"""Label-preserving interchange of taught/required course assignments.
The two CSV tables follow the CDSS column order (learning object, skill).
JSON additionally retains requirement alternatives and unused domain labels.
"""
from __future__ import annotations
import csv
import json
from collections.abc import Collection
from pathlib import Path
from typing import Any, Literal
from knowledgespaces.curriculum import LearningObject, SkillAssignment
[docs]
def assignment_to_dict(assignment: SkillAssignment) -> dict[str, Any]:
"""Serialize all alternatives and the explicit skill domain."""
return {
"skills": sorted(assignment.skills),
"objects": [
{"id": row.identifier, "taught": sorted(row.taught), "required": sorted(row.required)}
for row in assignment.rows
],
}
def _label_list(data: Any, field: str) -> list[str]:
if (
not isinstance(data, list)
or any(not isinstance(s, str) or not s for s in data)
or len(set(data)) != len(data)
):
raise ValueError(f"{field} must be a list of unique nonempty string labels.")
return data
[docs]
def dict_to_assignment(data: dict[str, Any]) -> SkillAssignment:
"""Read assignment JSON, validating nested records before construction."""
if not isinstance(data, dict) or "skills" not in data or "objects" not in data:
raise ValueError("Assignment JSON needs skills and objects fields.")
skills = _label_list(data["skills"], "skills")
if not isinstance(data["objects"], list):
raise ValueError("objects must be a list of learning-object records.")
rows = []
for row in data["objects"]:
if not isinstance(row, dict) or not {"id", "taught", "required"} <= row.keys():
raise ValueError("Each object record needs id, taught and required fields.")
rows.append(
LearningObject(
row["id"],
frozenset(_label_list(row["taught"], "taught")),
frozenset(_label_list(row["required"], "required")),
)
)
return SkillAssignment(rows, skills=skills)
[docs]
def write_assignment_json(assignment: SkillAssignment, path: str | Path) -> None:
"""Write a UTF-8 JSON assignment, including alternative requirements."""
with open(path, "w", encoding="utf-8") as stream:
json.dump(assignment_to_dict(assignment), stream, indent=2, ensure_ascii=False)
stream.write("\n")
[docs]
def read_assignment_json(path: str | Path) -> SkillAssignment:
"""Read a UTF-8 JSON assignment."""
with open(path, encoding="utf-8") as stream:
return dict_to_assignment(json.load(stream))
[docs]
def read_assignment_csv(
taught_path: str | Path,
required_path: str | Path,
*,
delimiter: str = ",",
header: bool = True,
learning_objects: Collection[str] | None = None,
skills: Collection[str] | None = None,
) -> SkillAssignment:
"""Read two CDSS-style (object, skill) tables as a single assignment.
Column order is positional; an optional header must have two columns.
Empty records are ignored. Labels, including surrounding whitespace,
are preserved exactly. Explicit domains retain absent objects/skills.
Diagnostics are available on the result; semantic noncompliance does
not prevent importing a course that needs correction.
"""
if not isinstance(header, bool):
raise ValueError("header must be a bool.")
def read(path: str | Path) -> list[tuple[str, str]]:
pairs = []
with open(path, encoding="utf-8-sig", newline="") as stream:
records = (row for row in csv.reader(stream, delimiter=delimiter) if row)
if header:
headings = next(records, None)
if headings is None or len(headings) != 2:
raise ValueError("Assignment CSV requires a two-column header.")
for row in records:
if len(row) != 2 or not all(row):
raise ValueError("Assignment CSV needs two nonempty labels per row.")
pairs.append((row[0], row[1]))
return pairs
return SkillAssignment.from_pairs(
read(taught_path), read(required_path), learning_objects=learning_objects, skills=skills
)
[docs]
def write_assignment_csv(
assignment: SkillAssignment,
taught_path: str | Path,
required_path: str | Path,
*,
delimiter: str = ",",
header: bool = True,
) -> None:
"""Write two CDSS-style tables without silently losing information.
This representation cannot retain alternative requirements or domain
labels absent from both tables; such inputs are rejected. Use JSON for
those cases. A reader infers object order from the taught table first;
pass ``learning_objects`` when the precise order of a nonteaching object
matters. Column headers are ``LO,Skill`` when requested.
"""
if not isinstance(header, bool):
raise ValueError("header must be a bool.")
if Path(taught_path).resolve() == Path(required_path).resolve():
raise ValueError("Taught and required tables need distinct paths.")
_pair_exportable(assignment)
for path, attribute in ((taught_path, "taught"), (required_path, "required")):
with open(path, "w", encoding="utf-8", newline="") as stream:
writer = csv.writer(stream, delimiter=delimiter)
if header:
writer.writerow(["LO", "Skill"])
for row in assignment.rows:
writer.writerows((row.identifier, s) for s in sorted(getattr(row, attribute)))
[docs]
def read_assignment(
path: str | Path,
required_path: str | Path | None = None,
*,
taught_sheet: str | int = "Taught",
required_sheet: str | int = "Required",
header: bool = True,
delimiter: str = ",",
learning_objects: Collection[str] | None = None,
skills: Collection[str] | None = None,
formula_policy: Literal["reject", "cached"] = "reject",
max_cells: int = 1_000_000,
) -> SkillAssignment:
"""Read a course from JSON, two CSV files, or a two-sheet XLSX/ODS workbook.
Format is detected by extension. Workbook sheets use the CDSS pair-table
layout and default names Taught/Required; sheet indices are zero-based.
Noncompliant courses remain available for diagnostics. Call ``.derive()``
on the returned assignment to run completion and both structural workflows.
JSON carries its own domains; external domain/header/sheet options apply
only to tabular inputs. No workbook formula is evaluated.
"""
from knowledgespaces.io.tables import read_table
suffix = Path(path).suffix.lower()
if suffix == ".csv":
if required_path is None:
raise ValueError("CSV assignments require a second path for required skills.")
return read_assignment_csv(
path,
required_path,
delimiter=delimiter,
header=header,
learning_objects=learning_objects,
skills=skills,
)
if required_path is not None:
raise ValueError("A second path is only used for two-file CSV assignments.")
if suffix == ".json":
if learning_objects is not None or skills is not None:
raise ValueError("JSON already declares its domains; do not override them on import.")
return read_assignment_json(path)
if suffix not in (".xlsx", ".ods"):
raise ValueError("Assignment file must be CSV, JSON, XLSX or ODS.")
if not isinstance(header, bool):
raise ValueError("header must be a bool.")
if taught_sheet == required_sheet:
raise ValueError("Taught and required sheets must be distinct.")
if formula_policy not in ("reject", "cached"):
raise ValueError("formula_policy must be reject or cached.")
def pairs(sheet: str | int) -> list[tuple[str, str]]:
from knowledgespaces.io.interchange import _label
table = read_table(path, sheet=sheet, formula_policy=formula_policy, max_cells=max_cells)
if header:
if not table or len(table[0]) != 2:
raise ValueError("Assignment sheet needs a two-column header.")
table = table[1:]
result = []
for row in table:
if all(v in (None, "") for v in row):
continue
if len(row) != 2:
raise ValueError("Assignment sheets need two columns (object, skill).")
result.append((_label(row[0]), _label(row[1])))
return result
return SkillAssignment.from_pairs(
pairs(taught_sheet), pairs(required_sheet), learning_objects=learning_objects, skills=skills
)
[docs]
def write_assignment(
assignment: SkillAssignment,
path: str | Path,
required_path: str | Path | None = None,
*,
taught_sheet: str = "Taught",
required_sheet: str = "Required",
header: bool = True,
delimiter: str = ",",
max_cells: int = 1_000_000,
) -> None:
"""Export JSON, two CSV files, or a new CDSS-style XLSX/ODS workbook.
Pair-table formats reject alternatives and unrepresented domain labels.
Use JSON to preserve those cases. Workbook exports contain exactly the
requested taught/required sheets, with literal identifiers.
"""
from knowledgespaces.io.tables import Cell, write_tables
suffix = Path(path).suffix.lower()
if suffix == ".csv":
if required_path is None:
raise ValueError("CSV assignments require a second path for required skills.")
write_assignment_csv(assignment, path, required_path, header=header, delimiter=delimiter)
return
if required_path is not None:
raise ValueError("A second path is only used for two-file CSV assignments.")
if suffix == ".json":
write_assignment_json(assignment, path)
return
if suffix not in (".xlsx", ".ods"):
raise ValueError("Assignment file must be CSV, JSON, XLSX or ODS.")
if not isinstance(header, bool):
raise ValueError("header must be a bool.")
if taught_sheet == required_sheet:
raise ValueError("Taught and required sheets must be distinct.")
_pair_exportable(assignment)
tables: dict[str, list[list[Cell]]] = {}
for name, attribute in ((taught_sheet, "taught"), (required_sheet, "required")):
rows: list[list[Cell]] = [["LO", "Skill"]] if header else []
for row in assignment.rows:
rows.extend([[row.identifier, s] for s in sorted(getattr(row, attribute))])
tables[name] = rows
write_tables(tables, path, max_cells=max_cells)
def _pair_exportable(assignment: SkillAssignment) -> None:
if not assignment.is_single_assignment:
raise ValueError("Pair tables cannot preserve alternative requirements; use JSON.")
represented = frozenset().union(*(r.taught | r.required for r in assignment.rows))
if represented != assignment.skills or any(
not (r.taught | r.required) for r in assignment.rows
):
raise ValueError("Pair tables cannot preserve absent domain labels; use JSON.")