Source code for knowledgespaces.io.interchange

"""Strict KST conversions for the tabular layouts of kstIO and CbKST.

Explicit ``kind`` selects the mathematical interpretation; no closure,
endpoint insertion, base reduction, frequency aggregation or imputation is
performed by a reader. Use the raw table API to inspect nonconforming files.
"""

from __future__ import annotations

from collections.abc import Sequence
from pathlib import Path
from typing import Literal

import numpy as np

from knowledgespaces.derivation.skill_map import SkillMap, SkillMultiMap
from knowledgespaces.estimation.blim_em import ResponseMatrix
from knowledgespaces.estimation.incomplete import IncompleteResponseMatrix
from knowledgespaces.io.tables import Cell, FormulaPolicy, TableFormat, read_table, write_tables
from knowledgespaces.structures.attribution import Attribution
from knowledgespaces.structures.knowledge_base import KnowledgeBase
from knowledgespaces.structures.knowledge_structure import KnowledgeStructure
from knowledgespaces.structures.relations import SurmiseRelation
from knowledgespaces.structures.set_family import SetFamily
from knowledgespaces.structures.surmise_function import SurmiseFunction

KSTKind = Literal[
    "family",
    "structure",
    "space",
    "basis",
    "relation",
    "attribution",
    "surmise_function",
    "skill_map",
    "skill_multimap",
    "data",
    "incomplete_data",
]
KSTObject = (
    SetFamily
    | KnowledgeStructure
    | KnowledgeBase
    | SurmiseRelation
    | Attribution
    | SurmiseFunction
    | SkillMap
    | SkillMultiMap
    | ResponseMatrix
    | IncompleteResponseMatrix
)
_KINDS = {
    "family",
    "structure",
    "space",
    "basis",
    "relation",
    "attribution",
    "surmise_function",
    "skill_map",
    "skill_multimap",
    "data",
    "incomplete_data",
}
_CLAUSES = {"attribution", "surmise_function", "skill_map", "skill_multimap"}


def _label(value: Cell) -> str:
    if isinstance(value, str) and value:
        return value
    if isinstance(value, (int, float)) and not isinstance(value, bool) and np.isfinite(value):
        return str(int(value)) if int(value) == value else str(value)
    raise ValueError("Labels must be nonempty text or finite numeric identifiers.")


def _binary(value: Cell) -> int:
    if isinstance(value, str):
        if value.lower() in ("true", "false"):
            return int(value.lower() == "true")
        try:
            value = float(value)
        except ValueError as exc:
            raise ValueError("Expected binary 0/1 table cells.") from exc
    if value not in (0, 1):
        raise ValueError("Expected binary 0/1 table cells; blanks are not zero.")
    return int(value)


def _check_marker(marker: str) -> None:
    if not isinstance(marker, str) or not marker:
        raise ValueError("missing_marker must be nonempty text distinct from binary values.")
    try:
        _binary(marker)
    except ValueError:
        return
    raise ValueError("missing_marker must not represent a binary value.")


[docs] def table_to_kst( table: Sequence[Sequence[Cell]], *, kind: KSTKind, header: bool = True, items: Sequence[str] | None = None, count_column: str | None = None, missing_marker: str = "NA", ) -> KSTObject: """Validate and convert an already loaded table to a mathematical object. State/base/response tables have one column per item. Clause and multimap tables have an initial item-ID column and one column per item/skill. Relation rows and columns follow the same label order, with entries `(prerequisite, dependent)`. Repeated response rows remain repeated. ``count_column`` explicitly selects a frequency column for response data; it is never guessed. Incomplete data additionally recognize empty cells and ``missing_marker``; missingness assumptions remain the caller's concern. """ if kind not in _KINDS: raise ValueError("Unknown KST table kind.") if not isinstance(header, bool): raise ValueError("header must be a bool.") _check_marker(missing_marker) rows = [list(row) for row in table] if not rows or not rows[0] or any(len(r) != len(rows[0]) for r in rows): raise ValueError("Expected a nonempty rectangular table.") leading = int(kind in _CLAUSES) width = len(rows[0]) - leading labels = ( [_label(v) for v in rows.pop(0)[leading:]] if header else [str(i + 1) for i in range(width)] ) if items is not None: explicit = list(items) if any(not isinstance(q, str) or not q for q in explicit) or len(explicit) != width: raise ValueError("items must be nonempty string labels matching table columns.") if header and labels != explicit: raise ValueError("Explicit items disagree with the file header order.") labels = explicit if len(set(labels)) != len(labels): raise ValueError("Duplicate column labels.") if count_column is not None and kind not in ("data", "incomplete_data"): raise ValueError("count_column is only valid for response data.") counts = None if count_column is not None: if count_column not in labels: raise ValueError("The requested count_column is absent.") index = labels.index(count_column) labels.pop(index) try: counts = np.array([r.pop(index) for r in rows], dtype=float) except (TypeError, ValueError) as exc: raise ValueError("Invalid frequency column.") from exc if not labels and kind not in ("skill_map", "skill_multimap"): raise ValueError("This KST object requires a nonempty item domain.") if kind in ("data", "incomplete_data"): matrix = np.array( [ [ np.nan if kind == "incomplete_data" and v in (None, "", missing_marker) else _binary(v) for v in row ] for row in rows ], dtype=float, ).reshape(len(rows), len(labels)) if kind == "incomplete_data": return IncompleteResponseMatrix(labels, matrix, counts) return ResponseMatrix(labels, matrix, counts) matrix_rows = [[_binary(v) for v in r[leading:]] for r in rows] if leading: identifiers = [_label(r[0]) for r in rows] if kind in ("skill_map", "skill_multimap"): multi = SkillMultiMap.from_matrix(identifiers, labels, matrix_rows) if kind == "skill_multimap": return multi if any(len(multi.competencies_for(q)) != 1 for q in multi.items): raise ValueError("A conjunctive skill map cannot contain requirement alternatives.") return SkillMap( multi.items, multi.skills, {q: multi.competencies_for(q)[0] for q in multi.items} ) clauses: dict[str, list[frozenset[str]]] = {} for q, row in zip(identifiers, matrix_rows, strict=True): clauses.setdefault(q, []).append( frozenset(s for s, b in zip(labels, row, strict=True) if b) ) if kind == "attribution": return Attribution(labels, clauses) return SurmiseFunction(labels, clauses) if kind == "relation": if len(matrix_rows) != len(labels) or any( matrix_rows[i][i] != 1 for i in range(len(labels)) ): raise ValueError("Surmise relation must be square and reflexive.") relation = SurmiseRelation( labels, [ (a, b) for a, row in zip(labels, matrix_rows, strict=True) for b, bit in zip(labels, row, strict=True) if bit ], ) if relation.transitive_closure().relations != relation.relations: raise ValueError("Surmise relation must already be transitive.") return relation family = SetFamily( labels, [frozenset(q for q, bit in zip(labels, r, strict=True) if bit) for r in matrix_rows] ) if kind == "family": return family if kind == "basis": base = family.to_knowledge_base() if base.base != family.sets: raise ValueError("A base table must be irredundant and exclude the empty set.") return base structure = family.to_knowledge_structure() if kind == "space" and not structure.is_knowledge_space: raise ValueError("Knowledge-space table must already be union-closed.") return structure
[docs] def read_kst( path: str | Path, *, kind: KSTKind, sheet: str | int = 0, format: TableFormat = "auto", header: bool = True, items: Sequence[str] | None = None, count_column: str | None = None, missing_marker: str = "NA", delimiter: str = ",", formula_policy: FormulaPolicy = "reject", max_cells: int = 1_000_000, ) -> KSTObject: """Read a CSV/XLSX/ODS table with an explicit, strictly checked KST kind. For KST/SRBT/plain binary text use the legacy readers. Spreadsheet engines load only when needed. See :func:`table_to_kst` for layout conventions. """ return table_to_kst( read_table( path, sheet=sheet, format=format, delimiter=delimiter, formula_policy=formula_policy, max_cells=max_cells, ), kind=kind, header=header, items=items, count_column=count_column, missing_marker=missing_marker, )
[docs] def kst_to_table( obj: KSTObject, *, items: Sequence[str] | None = None, header: bool = True, count_column: str | None = None, missing_marker: str = "NA", ) -> list[list[Cell]]: """Export the exact represented family, clauses, relation, map or responses. A base exports its base sets, not its span. A function exports canonical clauses, not all states. Frequencies require an explicit ``count_column``; they are not discarded or expanded. Incomplete responses use an explicit text marker so entirely missing trailing respondents survive workbook I/O. """ if not isinstance(header, bool): raise ValueError("header must be a bool.") _check_marker(missing_marker) if isinstance(obj, (SkillMap, SkillMultiMap)): domain = obj.skills elif isinstance(obj, (ResponseMatrix, IncompleteResponseMatrix, SurmiseRelation)): domain = frozenset(obj.items) else: domain = obj.domain labels = ( ( list(obj.items) if isinstance(obj, (ResponseMatrix, IncompleteResponseMatrix)) else sorted(domain) ) if items is None else list(items) ) if ( len(set(labels)) != len(labels) or set(labels) != domain or any(not isinstance(q, str) or not q for q in labels) ): raise ValueError( "items must order the complete domain using unique nonempty string labels." ) if not labels and not isinstance(obj, (SkillMap, SkillMultiMap)): raise ValueError( "Zero-column families cannot be represented in this table layout; use JSON." ) response = isinstance(obj, (ResponseMatrix, IncompleteResponseMatrix)) if count_column is not None and ( not response or not isinstance(count_column, str) or not count_column or count_column in labels ): raise ValueError("count_column must be a new label for response frequencies only.") rows: list[list[Cell]] = [] headings = list(labels) if isinstance(obj, (Attribution, SurmiseFunction, SkillMap, SkillMultiMap)): prefix = "Item" while prefix in labels: prefix = "_" + prefix headings.insert(0, prefix) objects = obj.items if isinstance(obj, (SkillMap, SkillMultiMap)) else sorted(obj.domain) for q in objects: if isinstance(obj, SkillMap): clauses = [obj.skills_for(q)] elif isinstance(obj, SkillMultiMap): clauses = list(obj.competencies_for(q)) else: clauses = sorted(obj.clauses_for(q), key=lambda c: (len(c), sorted(c))) rows.extend([[q, *[int(s in c) for s in labels]] for c in clauses]) elif isinstance(obj, (ResponseMatrix, IncompleteResponseMatrix)): # ResponseMatrix is mutable; validate its current arrays before exporting. if isinstance(obj, ResponseMatrix): ResponseMatrix(list(obj.items), obj.patterns, obj.counts) if obj.counts is not None and count_column is None: raise ValueError("Specify count_column to preserve response frequencies.") order = [list(obj.items).index(q) for q in labels] for i, row in enumerate(obj.patterns[:, order]): values: list[Cell] = [missing_marker if np.isnan(v) else int(v) for v in row] if count_column is not None: values.append(float(obj.effective_counts[i])) rows.append(values) if count_column is not None: headings.append(count_column) elif isinstance(obj, SurmiseRelation): if obj.transitive_closure().relations != obj.relations: raise ValueError("Close the relation explicitly before exporting it.") rows = [[int((a, b) in obj) for b in labels] for a in labels] else: family = ( obj.base if isinstance(obj, KnowledgeBase) else obj.states if isinstance(obj, KnowledgeStructure) else obj.sets ) rows = [ [int(q in s) for q in labels] for s in sorted(family, key=lambda s: (len(s), sorted(s))) ] header_row: list[Cell] = list(headings) return [header_row, *rows] if header else rows
[docs] def write_kst( obj: KSTObject, path: str | Path, *, sheet: str = "Data", format: TableFormat = "auto", items: Sequence[str] | None = None, header: bool = True, count_column: str | None = None, missing_marker: str = "NA", delimiter: str = ",", max_cells: int = 1_000_000, ) -> None: """Write one KST object as a new CSV/XLSX/ODS export. Use ``write_tables`` with ``kst_to_table`` to put several objects into separate sheets of the same workbook. """ table = kst_to_table( obj, items=items, header=header, count_column=count_column, missing_marker=missing_marker ) write_tables({sheet: table}, path, format=format, delimiter=delimiter, max_cells=max_cells)