Source code for knowledgespaces.io.curriculum

"""Label-preserving interchange of taught/required course assignments.

The two CSV tables follow the CDSS column order (learning object, skill).
JSON additionally retains requirement alternatives and unused domain labels.
"""

from __future__ import annotations

import csv
import json
from collections.abc import Collection
from pathlib import Path
from typing import Any, Literal

from knowledgespaces.curriculum import LearningObject, SkillAssignment


[docs] def assignment_to_dict(assignment: SkillAssignment) -> dict[str, Any]: """Serialize all alternatives and the explicit skill domain.""" return { "skills": sorted(assignment.skills), "objects": [ {"id": row.identifier, "taught": sorted(row.taught), "required": sorted(row.required)} for row in assignment.rows ], }
def _label_list(data: Any, field: str) -> list[str]: if ( not isinstance(data, list) or any(not isinstance(s, str) or not s for s in data) or len(set(data)) != len(data) ): raise ValueError(f"{field} must be a list of unique nonempty string labels.") return data
[docs] def dict_to_assignment(data: dict[str, Any]) -> SkillAssignment: """Read assignment JSON, validating nested records before construction.""" if not isinstance(data, dict) or "skills" not in data or "objects" not in data: raise ValueError("Assignment JSON needs skills and objects fields.") skills = _label_list(data["skills"], "skills") if not isinstance(data["objects"], list): raise ValueError("objects must be a list of learning-object records.") rows = [] for row in data["objects"]: if not isinstance(row, dict) or not {"id", "taught", "required"} <= row.keys(): raise ValueError("Each object record needs id, taught and required fields.") rows.append( LearningObject( row["id"], frozenset(_label_list(row["taught"], "taught")), frozenset(_label_list(row["required"], "required")), ) ) return SkillAssignment(rows, skills=skills)
[docs] def write_assignment_json(assignment: SkillAssignment, path: str | Path) -> None: """Write a UTF-8 JSON assignment, including alternative requirements.""" with open(path, "w", encoding="utf-8") as stream: json.dump(assignment_to_dict(assignment), stream, indent=2, ensure_ascii=False) stream.write("\n")
[docs] def read_assignment_json(path: str | Path) -> SkillAssignment: """Read a UTF-8 JSON assignment.""" with open(path, encoding="utf-8") as stream: return dict_to_assignment(json.load(stream))
[docs] def read_assignment_csv( taught_path: str | Path, required_path: str | Path, *, delimiter: str = ",", header: bool = True, learning_objects: Collection[str] | None = None, skills: Collection[str] | None = None, ) -> SkillAssignment: """Read two CDSS-style (object, skill) tables as a single assignment. Column order is positional; an optional header must have two columns. Empty records are ignored. Labels, including surrounding whitespace, are preserved exactly. Explicit domains retain absent objects/skills. Diagnostics are available on the result; semantic noncompliance does not prevent importing a course that needs correction. """ if not isinstance(header, bool): raise ValueError("header must be a bool.") def read(path: str | Path) -> list[tuple[str, str]]: pairs = [] with open(path, encoding="utf-8-sig", newline="") as stream: records = (row for row in csv.reader(stream, delimiter=delimiter) if row) if header: headings = next(records, None) if headings is None or len(headings) != 2: raise ValueError("Assignment CSV requires a two-column header.") for row in records: if len(row) != 2 or not all(row): raise ValueError("Assignment CSV needs two nonempty labels per row.") pairs.append((row[0], row[1])) return pairs return SkillAssignment.from_pairs( read(taught_path), read(required_path), learning_objects=learning_objects, skills=skills )
[docs] def write_assignment_csv( assignment: SkillAssignment, taught_path: str | Path, required_path: str | Path, *, delimiter: str = ",", header: bool = True, ) -> None: """Write two CDSS-style tables without silently losing information. This representation cannot retain alternative requirements or domain labels absent from both tables; such inputs are rejected. Use JSON for those cases. A reader infers object order from the taught table first; pass ``learning_objects`` when the precise order of a nonteaching object matters. Column headers are ``LO,Skill`` when requested. """ if not isinstance(header, bool): raise ValueError("header must be a bool.") if Path(taught_path).resolve() == Path(required_path).resolve(): raise ValueError("Taught and required tables need distinct paths.") _pair_exportable(assignment) for path, attribute in ((taught_path, "taught"), (required_path, "required")): with open(path, "w", encoding="utf-8", newline="") as stream: writer = csv.writer(stream, delimiter=delimiter) if header: writer.writerow(["LO", "Skill"]) for row in assignment.rows: writer.writerows((row.identifier, s) for s in sorted(getattr(row, attribute)))
[docs] def read_assignment( path: str | Path, required_path: str | Path | None = None, *, taught_sheet: str | int = "Taught", required_sheet: str | int = "Required", header: bool = True, delimiter: str = ",", learning_objects: Collection[str] | None = None, skills: Collection[str] | None = None, formula_policy: Literal["reject", "cached"] = "reject", max_cells: int = 1_000_000, ) -> SkillAssignment: """Read a course from JSON, two CSV files, or a two-sheet XLSX/ODS workbook. Format is detected by extension. Workbook sheets use the CDSS pair-table layout and default names Taught/Required; sheet indices are zero-based. Noncompliant courses remain available for diagnostics. Call ``.derive()`` on the returned assignment to run completion and both structural workflows. JSON carries its own domains; external domain/header/sheet options apply only to tabular inputs. No workbook formula is evaluated. """ from knowledgespaces.io.tables import read_table suffix = Path(path).suffix.lower() if suffix == ".csv": if required_path is None: raise ValueError("CSV assignments require a second path for required skills.") return read_assignment_csv( path, required_path, delimiter=delimiter, header=header, learning_objects=learning_objects, skills=skills, ) if required_path is not None: raise ValueError("A second path is only used for two-file CSV assignments.") if suffix == ".json": if learning_objects is not None or skills is not None: raise ValueError("JSON already declares its domains; do not override them on import.") return read_assignment_json(path) if suffix not in (".xlsx", ".ods"): raise ValueError("Assignment file must be CSV, JSON, XLSX or ODS.") if not isinstance(header, bool): raise ValueError("header must be a bool.") if taught_sheet == required_sheet: raise ValueError("Taught and required sheets must be distinct.") if formula_policy not in ("reject", "cached"): raise ValueError("formula_policy must be reject or cached.") def pairs(sheet: str | int) -> list[tuple[str, str]]: from knowledgespaces.io.interchange import _label table = read_table(path, sheet=sheet, formula_policy=formula_policy, max_cells=max_cells) if header: if not table or len(table[0]) != 2: raise ValueError("Assignment sheet needs a two-column header.") table = table[1:] result = [] for row in table: if all(v in (None, "") for v in row): continue if len(row) != 2: raise ValueError("Assignment sheets need two columns (object, skill).") result.append((_label(row[0]), _label(row[1]))) return result return SkillAssignment.from_pairs( pairs(taught_sheet), pairs(required_sheet), learning_objects=learning_objects, skills=skills )
[docs] def write_assignment( assignment: SkillAssignment, path: str | Path, required_path: str | Path | None = None, *, taught_sheet: str = "Taught", required_sheet: str = "Required", header: bool = True, delimiter: str = ",", max_cells: int = 1_000_000, ) -> None: """Export JSON, two CSV files, or a new CDSS-style XLSX/ODS workbook. Pair-table formats reject alternatives and unrepresented domain labels. Use JSON to preserve those cases. Workbook exports contain exactly the requested taught/required sheets, with literal identifiers. """ from knowledgespaces.io.tables import Cell, write_tables suffix = Path(path).suffix.lower() if suffix == ".csv": if required_path is None: raise ValueError("CSV assignments require a second path for required skills.") write_assignment_csv(assignment, path, required_path, header=header, delimiter=delimiter) return if required_path is not None: raise ValueError("A second path is only used for two-file CSV assignments.") if suffix == ".json": write_assignment_json(assignment, path) return if suffix not in (".xlsx", ".ods"): raise ValueError("Assignment file must be CSV, JSON, XLSX or ODS.") if not isinstance(header, bool): raise ValueError("header must be a bool.") if taught_sheet == required_sheet: raise ValueError("Taught and required sheets must be distinct.") _pair_exportable(assignment) tables: dict[str, list[list[Cell]]] = {} for name, attribute in ((taught_sheet, "taught"), (required_sheet, "required")): rows: list[list[Cell]] = [["LO", "Skill"]] if header else [] for row in assignment.rows: rows.extend([[row.identifier, s] for s in sorted(getattr(row, attribute))]) tables[name] = rows write_tables(tables, path, max_cells=max_cells)
def _pair_exportable(assignment: SkillAssignment) -> None: if not assignment.is_single_assignment: raise ValueError("Pair tables cannot preserve alternative requirements; use JSON.") represented = frozenset().union(*(r.taught | r.required for r in assignment.rows)) if represented != assignment.skills or any( not (r.taught | r.required) for r in assignment.rows ): raise ValueError("Pair tables cannot preserve absent domain labels; use JSON.")