more data

This commit is contained in:
Talhadeveloperr
2026-09-11 16:54:23 +05:00
parent 9079ba2739
commit 71042c2aae
352 changed files with 1715 additions and 3098 deletions

View File

@@ -49,8 +49,25 @@ def _read_csv(relative_path):
return list(csv.DictReader(f))
def _to_int(value, default=0):
try:
return int(float(value))
except (TypeError, ValueError):
return default
def load_data():
"""Loads all dummy CSVs into plain dict/list structures."""
"""
Loads the real UORM CSV data into plain dict/list structures.
courses/courses.csv is the full course catalog in the university's
wide export format (Department_Name, Program_Name, Course_Code,
T_Contact_Hours, P_Contact_Hours, ...). courses/course_offering.csv is
a curated subset of that catalog — only rows with an instructor
assigned there are schedulable. Course rows are looked up by
Course_Code (e.g. "BBA-300"), which is what course_offering.csv's
course_id column refers to.
"""
rooms = _read_csv(os.path.join("blocks", "rooms.csv"))
blocks = _read_csv(os.path.join("blocks", "blocks.csv"))
instructors = _read_csv(os.path.join("instructor", "instructors.csv"))
@@ -65,22 +82,51 @@ def load_data():
instructor_names_by_code = {i["instructor_code"]: i["instructor_name"] for i in instructors}
batches_by_code = {b["batch_code"]: b for b in batches}
# normalize courses: numeric fields + a lookup by id
# normalize courses: derive scheduler fields from the real catalog's
# contact-hour columns, and key by Course_Code (what offerings reference).
courses_by_id = {}
for c in courses:
c["credit_hours"] = int(c["credit_hours"])
c["lectures_per_week"] = int(c["lectures_per_week"])
c["labs_per_week"] = int(c["labs_per_week"])
c["is_general"] = c["is_general"].lower() == "yes"
c["lecture_duration_minutes"] = int(c.get("lecture_duration_minutes") or 90)
courses_by_id[c["course_id"]] = c
course_code = c["Course_Code"].strip()
t_contact_hours = _to_int(c.get("T_Contact_Hours"))
p_contact_hours = _to_int(c.get("P_Contact_Hours"))
courses_by_id[course_code] = {
"course_id": course_code,
"course_name": c["Course_Title"].strip(),
"department": c["Department_Name"],
"program": c["Program_Name"],
"credit_hours": _to_int(c.get("Total_Credit_Hours")),
# One 90-minute lecture slot per weekly theory contact hour;
# no reliable "general" flag exists in this data, so every
# course is treated as non-general (see solver/phase1_general.py).
"lectures_per_week": t_contact_hours,
# A course with any practical/lab contact hours gets exactly
# one 180-minute lab session/week (never more), per spec.
"labs_per_week": 1 if p_contact_hours > 0 else 0,
"is_general": False,
"lecture_duration_minutes": 90,
# No per-course room assignment in this data yet; leaving
# these unset makes room_candidates() fall back to the full
# room pool with no preference (handled by PlacementBoard).
"lecture_room_id": None,
"lab_room_id": None,
}
# normalize offerings: expand "batch1|batch2" section batch groups into
# a list of batch codes, and attach the course record for convenience
# a list of batch codes (kept for forward compatibility even though
# this real data has one batch per offering), and attach the course
# record for convenience. Offerings whose course_id has no matching
# catalog row are skipped (curated subset referencing a typo/removed
# course) rather than crashing the whole load.
normalized_offerings = []
for off in offerings:
course = courses_by_id.get(off["course_id"])
if course is None:
print(f"WARNING: course_offering.csv references unknown course_id '{off['course_id']}' — skipping")
continue
off["batch_codes"] = off["batch_codes"].split("|")
off["course"] = courses_by_id[off["course_id"]]
off["course"] = course
off["instructor_name"] = instructor_names_by_code.get(off["instructor_code"], off["instructor_code"])
normalized_offerings.append(off)
return {
"rooms": rooms,
@@ -90,7 +136,7 @@ def load_data():
"instructor_names_by_code": instructor_names_by_code,
"courses": courses,
"courses_by_id": courses_by_id,
"offerings": offerings,
"offerings": normalized_offerings,
"students": students,
"batches": batches,
"batches_by_code": batches_by_code,