more data
This commit is contained in:
@@ -49,8 +49,25 @@ def _read_csv(relative_path):
|
||||
return list(csv.DictReader(f))
|
||||
|
||||
|
||||
def _to_int(value, default=0):
|
||||
try:
|
||||
return int(float(value))
|
||||
except (TypeError, ValueError):
|
||||
return default
|
||||
|
||||
|
||||
def load_data():
|
||||
"""Loads all dummy CSVs into plain dict/list structures."""
|
||||
"""
|
||||
Loads the real UORM CSV data into plain dict/list structures.
|
||||
|
||||
courses/courses.csv is the full course catalog in the university's
|
||||
wide export format (Department_Name, Program_Name, Course_Code,
|
||||
T_Contact_Hours, P_Contact_Hours, ...). courses/course_offering.csv is
|
||||
a curated subset of that catalog — only rows with an instructor
|
||||
assigned there are schedulable. Course rows are looked up by
|
||||
Course_Code (e.g. "BBA-300"), which is what course_offering.csv's
|
||||
course_id column refers to.
|
||||
"""
|
||||
rooms = _read_csv(os.path.join("blocks", "rooms.csv"))
|
||||
blocks = _read_csv(os.path.join("blocks", "blocks.csv"))
|
||||
instructors = _read_csv(os.path.join("instructor", "instructors.csv"))
|
||||
@@ -65,22 +82,51 @@ def load_data():
|
||||
instructor_names_by_code = {i["instructor_code"]: i["instructor_name"] for i in instructors}
|
||||
batches_by_code = {b["batch_code"]: b for b in batches}
|
||||
|
||||
# normalize courses: numeric fields + a lookup by id
|
||||
# normalize courses: derive scheduler fields from the real catalog's
|
||||
# contact-hour columns, and key by Course_Code (what offerings reference).
|
||||
courses_by_id = {}
|
||||
for c in courses:
|
||||
c["credit_hours"] = int(c["credit_hours"])
|
||||
c["lectures_per_week"] = int(c["lectures_per_week"])
|
||||
c["labs_per_week"] = int(c["labs_per_week"])
|
||||
c["is_general"] = c["is_general"].lower() == "yes"
|
||||
c["lecture_duration_minutes"] = int(c.get("lecture_duration_minutes") or 90)
|
||||
courses_by_id[c["course_id"]] = c
|
||||
course_code = c["Course_Code"].strip()
|
||||
t_contact_hours = _to_int(c.get("T_Contact_Hours"))
|
||||
p_contact_hours = _to_int(c.get("P_Contact_Hours"))
|
||||
courses_by_id[course_code] = {
|
||||
"course_id": course_code,
|
||||
"course_name": c["Course_Title"].strip(),
|
||||
"department": c["Department_Name"],
|
||||
"program": c["Program_Name"],
|
||||
"credit_hours": _to_int(c.get("Total_Credit_Hours")),
|
||||
# One 90-minute lecture slot per weekly theory contact hour;
|
||||
# no reliable "general" flag exists in this data, so every
|
||||
# course is treated as non-general (see solver/phase1_general.py).
|
||||
"lectures_per_week": t_contact_hours,
|
||||
# A course with any practical/lab contact hours gets exactly
|
||||
# one 180-minute lab session/week (never more), per spec.
|
||||
"labs_per_week": 1 if p_contact_hours > 0 else 0,
|
||||
"is_general": False,
|
||||
"lecture_duration_minutes": 90,
|
||||
# No per-course room assignment in this data yet; leaving
|
||||
# these unset makes room_candidates() fall back to the full
|
||||
# room pool with no preference (handled by PlacementBoard).
|
||||
"lecture_room_id": None,
|
||||
"lab_room_id": None,
|
||||
}
|
||||
|
||||
# normalize offerings: expand "batch1|batch2" section batch groups into
|
||||
# a list of batch codes, and attach the course record for convenience
|
||||
# a list of batch codes (kept for forward compatibility even though
|
||||
# this real data has one batch per offering), and attach the course
|
||||
# record for convenience. Offerings whose course_id has no matching
|
||||
# catalog row are skipped (curated subset referencing a typo/removed
|
||||
# course) rather than crashing the whole load.
|
||||
normalized_offerings = []
|
||||
for off in offerings:
|
||||
course = courses_by_id.get(off["course_id"])
|
||||
if course is None:
|
||||
print(f"WARNING: course_offering.csv references unknown course_id '{off['course_id']}' — skipping")
|
||||
continue
|
||||
off["batch_codes"] = off["batch_codes"].split("|")
|
||||
off["course"] = courses_by_id[off["course_id"]]
|
||||
off["course"] = course
|
||||
off["instructor_name"] = instructor_names_by_code.get(off["instructor_code"], off["instructor_code"])
|
||||
normalized_offerings.append(off)
|
||||
|
||||
return {
|
||||
"rooms": rooms,
|
||||
@@ -90,7 +136,7 @@ def load_data():
|
||||
"instructor_names_by_code": instructor_names_by_code,
|
||||
"courses": courses,
|
||||
"courses_by_id": courses_by_id,
|
||||
"offerings": offerings,
|
||||
"offerings": normalized_offerings,
|
||||
"students": students,
|
||||
"batches": batches,
|
||||
"batches_by_code": batches_by_code,
|
||||
|
||||
Reference in New Issue
Block a user