more data

This commit is contained in:
Talhadeveloperr
2026-09-11 16:54:23 +05:00
parent 9079ba2739
commit 71042c2aae
352 changed files with 1715 additions and 3098 deletions

View File

@@ -1,7 +1,7 @@
"""
Phase 1: general courses (shared across every batch of a session).
Simple strategy — small offering count, Tue/Thu preferred first. After
Simple strategy — small offering count, Mon/Wed preferred first. After
placing everything, self-repair: any batch left with two general-course
sessions in the same slot has the losing session unmarked and re-placed
(scanning every day) until this phase's own sessions are clash-free.
@@ -10,7 +10,7 @@ sessions in the same slot has the losing session unmarked and re-placed
from . import grid
def run(board, data, preferred_days=("Tuesday", "Thursday"), max_repair_attempts=20):
def run(board, data, preferred_days=("Monday", "Wednesday"), max_repair_attempts=20):
"""Places every general-course offering's weekly lectures. Returns this phase's sessions."""
general = [o for o in data["offerings"] if o["course"]["is_general"]]
preferred_order = list(preferred_days) + [d for d in grid.DAYS if d not in preferred_days]

View File

@@ -10,11 +10,21 @@ phases' placements as fixed/locked.
from . import grid
def _is_lab_room_type(room_type):
"""
Real room data uses free-text room_type labels (e.g. "Computing
Laboratories", "electrical", "Classrooms", "Civil", "physics") rather
than a fixed classroom/lab enum. Any label mentioning "lab" is treated
as a lab room; everything else is a classroom.
"""
return "lab" in (room_type or "").lower()
class PlacementBoard:
def __init__(self, data):
self.data = data
self.classrooms = [r for r in data["rooms"] if r["room_type"] == "classroom"]
self.labs = [r for r in data["rooms"] if r["room_type"] == "lab"]
self.classrooms = [r for r in data["rooms"] if not _is_lab_room_type(r["room_type"])]
self.labs = [r for r in data["rooms"] if _is_lab_room_type(r["room_type"])]
# busy[(day, slot_index)] -> set of keys already occupied
self.room_busy = {}

View File

@@ -49,8 +49,25 @@ def _read_csv(relative_path):
return list(csv.DictReader(f))
def _to_int(value, default=0):
try:
return int(float(value))
except (TypeError, ValueError):
return default
def load_data():
"""Loads all dummy CSVs into plain dict/list structures."""
"""
Loads the real UORM CSV data into plain dict/list structures.
courses/courses.csv is the full course catalog in the university's
wide export format (Department_Name, Program_Name, Course_Code,
T_Contact_Hours, P_Contact_Hours, ...). courses/course_offering.csv is
a curated subset of that catalog — only rows with an instructor
assigned there are schedulable. Course rows are looked up by
Course_Code (e.g. "BBA-300"), which is what course_offering.csv's
course_id column refers to.
"""
rooms = _read_csv(os.path.join("blocks", "rooms.csv"))
blocks = _read_csv(os.path.join("blocks", "blocks.csv"))
instructors = _read_csv(os.path.join("instructor", "instructors.csv"))
@@ -65,22 +82,51 @@ def load_data():
instructor_names_by_code = {i["instructor_code"]: i["instructor_name"] for i in instructors}
batches_by_code = {b["batch_code"]: b for b in batches}
# normalize courses: numeric fields + a lookup by id
# normalize courses: derive scheduler fields from the real catalog's
# contact-hour columns, and key by Course_Code (what offerings reference).
courses_by_id = {}
for c in courses:
c["credit_hours"] = int(c["credit_hours"])
c["lectures_per_week"] = int(c["lectures_per_week"])
c["labs_per_week"] = int(c["labs_per_week"])
c["is_general"] = c["is_general"].lower() == "yes"
c["lecture_duration_minutes"] = int(c.get("lecture_duration_minutes") or 90)
courses_by_id[c["course_id"]] = c
course_code = c["Course_Code"].strip()
t_contact_hours = _to_int(c.get("T_Contact_Hours"))
p_contact_hours = _to_int(c.get("P_Contact_Hours"))
courses_by_id[course_code] = {
"course_id": course_code,
"course_name": c["Course_Title"].strip(),
"department": c["Department_Name"],
"program": c["Program_Name"],
"credit_hours": _to_int(c.get("Total_Credit_Hours")),
# One 90-minute lecture slot per weekly theory contact hour;
# no reliable "general" flag exists in this data, so every
# course is treated as non-general (see solver/phase1_general.py).
"lectures_per_week": t_contact_hours,
# A course with any practical/lab contact hours gets exactly
# one 180-minute lab session/week (never more), per spec.
"labs_per_week": 1 if p_contact_hours > 0 else 0,
"is_general": False,
"lecture_duration_minutes": 90,
# No per-course room assignment in this data yet; leaving
# these unset makes room_candidates() fall back to the full
# room pool with no preference (handled by PlacementBoard).
"lecture_room_id": None,
"lab_room_id": None,
}
# normalize offerings: expand "batch1|batch2" section batch groups into
# a list of batch codes, and attach the course record for convenience
# a list of batch codes (kept for forward compatibility even though
# this real data has one batch per offering), and attach the course
# record for convenience. Offerings whose course_id has no matching
# catalog row are skipped (curated subset referencing a typo/removed
# course) rather than crashing the whole load.
normalized_offerings = []
for off in offerings:
course = courses_by_id.get(off["course_id"])
if course is None:
print(f"WARNING: course_offering.csv references unknown course_id '{off['course_id']}' — skipping")
continue
off["batch_codes"] = off["batch_codes"].split("|")
off["course"] = courses_by_id[off["course_id"]]
off["course"] = course
off["instructor_name"] = instructor_names_by_code.get(off["instructor_code"], off["instructor_code"])
normalized_offerings.append(off)
return {
"rooms": rooms,
@@ -90,7 +136,7 @@ def load_data():
"instructor_names_by_code": instructor_names_by_code,
"courses": courses,
"courses_by_id": courses_by_id,
"offerings": offerings,
"offerings": normalized_offerings,
"students": students,
"batches": batches,
"batches_by_code": batches_by_code,