"""Strict KST conversions for the tabular layouts of kstIO and CbKST.
Explicit ``kind`` selects the mathematical interpretation; no closure,
endpoint insertion, base reduction, frequency aggregation or imputation is
performed by a reader. Use the raw table API to inspect nonconforming files.
"""
from __future__ import annotations
from collections.abc import Sequence
from pathlib import Path
from typing import Literal
import numpy as np
from knowledgespaces.derivation.skill_map import SkillMap, SkillMultiMap
from knowledgespaces.estimation.blim_em import ResponseMatrix
from knowledgespaces.estimation.incomplete import IncompleteResponseMatrix
from knowledgespaces.io.tables import Cell, FormulaPolicy, TableFormat, read_table, write_tables
from knowledgespaces.structures.attribution import Attribution
from knowledgespaces.structures.knowledge_base import KnowledgeBase
from knowledgespaces.structures.knowledge_structure import KnowledgeStructure
from knowledgespaces.structures.relations import SurmiseRelation
from knowledgespaces.structures.set_family import SetFamily
from knowledgespaces.structures.surmise_function import SurmiseFunction
KSTKind = Literal[
"family",
"structure",
"space",
"basis",
"relation",
"attribution",
"surmise_function",
"skill_map",
"skill_multimap",
"data",
"incomplete_data",
]
KSTObject = (
SetFamily
| KnowledgeStructure
| KnowledgeBase
| SurmiseRelation
| Attribution
| SurmiseFunction
| SkillMap
| SkillMultiMap
| ResponseMatrix
| IncompleteResponseMatrix
)
_KINDS = {
"family",
"structure",
"space",
"basis",
"relation",
"attribution",
"surmise_function",
"skill_map",
"skill_multimap",
"data",
"incomplete_data",
}
_CLAUSES = {"attribution", "surmise_function", "skill_map", "skill_multimap"}
def _label(value: Cell) -> str:
if isinstance(value, str) and value:
return value
if isinstance(value, (int, float)) and not isinstance(value, bool) and np.isfinite(value):
return str(int(value)) if int(value) == value else str(value)
raise ValueError("Labels must be nonempty text or finite numeric identifiers.")
def _binary(value: Cell) -> int:
if isinstance(value, str):
if value.lower() in ("true", "false"):
return int(value.lower() == "true")
try:
value = float(value)
except ValueError as exc:
raise ValueError("Expected binary 0/1 table cells.") from exc
if value not in (0, 1):
raise ValueError("Expected binary 0/1 table cells; blanks are not zero.")
return int(value)
def _check_marker(marker: str) -> None:
if not isinstance(marker, str) or not marker:
raise ValueError("missing_marker must be nonempty text distinct from binary values.")
try:
_binary(marker)
except ValueError:
return
raise ValueError("missing_marker must not represent a binary value.")
[docs]
def table_to_kst(
table: Sequence[Sequence[Cell]],
*,
kind: KSTKind,
header: bool = True,
items: Sequence[str] | None = None,
count_column: str | None = None,
missing_marker: str = "NA",
) -> KSTObject:
"""Validate and convert an already loaded table to a mathematical object.
State/base/response tables have one column per item. Clause and multimap
tables have an initial item-ID column and one column per item/skill.
Relation rows and columns follow the same label order, with entries
`(prerequisite, dependent)`. Repeated response rows remain repeated.
``count_column`` explicitly selects a frequency column for response data;
it is never guessed. Incomplete data additionally recognize empty cells
and ``missing_marker``; missingness assumptions remain the caller's concern.
"""
if kind not in _KINDS:
raise ValueError("Unknown KST table kind.")
if not isinstance(header, bool):
raise ValueError("header must be a bool.")
_check_marker(missing_marker)
rows = [list(row) for row in table]
if not rows or not rows[0] or any(len(r) != len(rows[0]) for r in rows):
raise ValueError("Expected a nonempty rectangular table.")
leading = int(kind in _CLAUSES)
width = len(rows[0]) - leading
labels = (
[_label(v) for v in rows.pop(0)[leading:]] if header else [str(i + 1) for i in range(width)]
)
if items is not None:
explicit = list(items)
if any(not isinstance(q, str) or not q for q in explicit) or len(explicit) != width:
raise ValueError("items must be nonempty string labels matching table columns.")
if header and labels != explicit:
raise ValueError("Explicit items disagree with the file header order.")
labels = explicit
if len(set(labels)) != len(labels):
raise ValueError("Duplicate column labels.")
if count_column is not None and kind not in ("data", "incomplete_data"):
raise ValueError("count_column is only valid for response data.")
counts = None
if count_column is not None:
if count_column not in labels:
raise ValueError("The requested count_column is absent.")
index = labels.index(count_column)
labels.pop(index)
try:
counts = np.array([r.pop(index) for r in rows], dtype=float)
except (TypeError, ValueError) as exc:
raise ValueError("Invalid frequency column.") from exc
if not labels and kind not in ("skill_map", "skill_multimap"):
raise ValueError("This KST object requires a nonempty item domain.")
if kind in ("data", "incomplete_data"):
matrix = np.array(
[
[
np.nan
if kind == "incomplete_data" and v in (None, "", missing_marker)
else _binary(v)
for v in row
]
for row in rows
],
dtype=float,
).reshape(len(rows), len(labels))
if kind == "incomplete_data":
return IncompleteResponseMatrix(labels, matrix, counts)
return ResponseMatrix(labels, matrix, counts)
matrix_rows = [[_binary(v) for v in r[leading:]] for r in rows]
if leading:
identifiers = [_label(r[0]) for r in rows]
if kind in ("skill_map", "skill_multimap"):
multi = SkillMultiMap.from_matrix(identifiers, labels, matrix_rows)
if kind == "skill_multimap":
return multi
if any(len(multi.competencies_for(q)) != 1 for q in multi.items):
raise ValueError("A conjunctive skill map cannot contain requirement alternatives.")
return SkillMap(
multi.items, multi.skills, {q: multi.competencies_for(q)[0] for q in multi.items}
)
clauses: dict[str, list[frozenset[str]]] = {}
for q, row in zip(identifiers, matrix_rows, strict=True):
clauses.setdefault(q, []).append(
frozenset(s for s, b in zip(labels, row, strict=True) if b)
)
if kind == "attribution":
return Attribution(labels, clauses)
return SurmiseFunction(labels, clauses)
if kind == "relation":
if len(matrix_rows) != len(labels) or any(
matrix_rows[i][i] != 1 for i in range(len(labels))
):
raise ValueError("Surmise relation must be square and reflexive.")
relation = SurmiseRelation(
labels,
[
(a, b)
for a, row in zip(labels, matrix_rows, strict=True)
for b, bit in zip(labels, row, strict=True)
if bit
],
)
if relation.transitive_closure().relations != relation.relations:
raise ValueError("Surmise relation must already be transitive.")
return relation
family = SetFamily(
labels, [frozenset(q for q, bit in zip(labels, r, strict=True) if bit) for r in matrix_rows]
)
if kind == "family":
return family
if kind == "basis":
base = family.to_knowledge_base()
if base.base != family.sets:
raise ValueError("A base table must be irredundant and exclude the empty set.")
return base
structure = family.to_knowledge_structure()
if kind == "space" and not structure.is_knowledge_space:
raise ValueError("Knowledge-space table must already be union-closed.")
return structure
[docs]
def read_kst(
path: str | Path,
*,
kind: KSTKind,
sheet: str | int = 0,
format: TableFormat = "auto",
header: bool = True,
items: Sequence[str] | None = None,
count_column: str | None = None,
missing_marker: str = "NA",
delimiter: str = ",",
formula_policy: FormulaPolicy = "reject",
max_cells: int = 1_000_000,
) -> KSTObject:
"""Read a CSV/XLSX/ODS table with an explicit, strictly checked KST kind.
For KST/SRBT/plain binary text use the legacy readers. Spreadsheet engines
load only when needed. See :func:`table_to_kst` for layout conventions.
"""
return table_to_kst(
read_table(
path,
sheet=sheet,
format=format,
delimiter=delimiter,
formula_policy=formula_policy,
max_cells=max_cells,
),
kind=kind,
header=header,
items=items,
count_column=count_column,
missing_marker=missing_marker,
)
[docs]
def kst_to_table(
obj: KSTObject,
*,
items: Sequence[str] | None = None,
header: bool = True,
count_column: str | None = None,
missing_marker: str = "NA",
) -> list[list[Cell]]:
"""Export the exact represented family, clauses, relation, map or responses.
A base exports its base sets, not its span. A function exports canonical
clauses, not all states. Frequencies require an explicit ``count_column``;
they are not discarded or expanded. Incomplete responses use an explicit
text marker so entirely missing trailing respondents survive workbook I/O.
"""
if not isinstance(header, bool):
raise ValueError("header must be a bool.")
_check_marker(missing_marker)
if isinstance(obj, (SkillMap, SkillMultiMap)):
domain = obj.skills
elif isinstance(obj, (ResponseMatrix, IncompleteResponseMatrix, SurmiseRelation)):
domain = frozenset(obj.items)
else:
domain = obj.domain
labels = (
(
list(obj.items)
if isinstance(obj, (ResponseMatrix, IncompleteResponseMatrix))
else sorted(domain)
)
if items is None
else list(items)
)
if (
len(set(labels)) != len(labels)
or set(labels) != domain
or any(not isinstance(q, str) or not q for q in labels)
):
raise ValueError(
"items must order the complete domain using unique nonempty string labels."
)
if not labels and not isinstance(obj, (SkillMap, SkillMultiMap)):
raise ValueError(
"Zero-column families cannot be represented in this table layout; use JSON."
)
response = isinstance(obj, (ResponseMatrix, IncompleteResponseMatrix))
if count_column is not None and (
not response
or not isinstance(count_column, str)
or not count_column
or count_column in labels
):
raise ValueError("count_column must be a new label for response frequencies only.")
rows: list[list[Cell]] = []
headings = list(labels)
if isinstance(obj, (Attribution, SurmiseFunction, SkillMap, SkillMultiMap)):
prefix = "Item"
while prefix in labels:
prefix = "_" + prefix
headings.insert(0, prefix)
objects = obj.items if isinstance(obj, (SkillMap, SkillMultiMap)) else sorted(obj.domain)
for q in objects:
if isinstance(obj, SkillMap):
clauses = [obj.skills_for(q)]
elif isinstance(obj, SkillMultiMap):
clauses = list(obj.competencies_for(q))
else:
clauses = sorted(obj.clauses_for(q), key=lambda c: (len(c), sorted(c)))
rows.extend([[q, *[int(s in c) for s in labels]] for c in clauses])
elif isinstance(obj, (ResponseMatrix, IncompleteResponseMatrix)):
# ResponseMatrix is mutable; validate its current arrays before exporting.
if isinstance(obj, ResponseMatrix):
ResponseMatrix(list(obj.items), obj.patterns, obj.counts)
if obj.counts is not None and count_column is None:
raise ValueError("Specify count_column to preserve response frequencies.")
order = [list(obj.items).index(q) for q in labels]
for i, row in enumerate(obj.patterns[:, order]):
values: list[Cell] = [missing_marker if np.isnan(v) else int(v) for v in row]
if count_column is not None:
values.append(float(obj.effective_counts[i]))
rows.append(values)
if count_column is not None:
headings.append(count_column)
elif isinstance(obj, SurmiseRelation):
if obj.transitive_closure().relations != obj.relations:
raise ValueError("Close the relation explicitly before exporting it.")
rows = [[int((a, b) in obj) for b in labels] for a in labels]
else:
family = (
obj.base
if isinstance(obj, KnowledgeBase)
else obj.states
if isinstance(obj, KnowledgeStructure)
else obj.sets
)
rows = [
[int(q in s) for q in labels] for s in sorted(family, key=lambda s: (len(s), sorted(s)))
]
header_row: list[Cell] = list(headings)
return [header_row, *rows] if header else rows
[docs]
def write_kst(
obj: KSTObject,
path: str | Path,
*,
sheet: str = "Data",
format: TableFormat = "auto",
items: Sequence[str] | None = None,
header: bool = True,
count_column: str | None = None,
missing_marker: str = "NA",
delimiter: str = ",",
max_cells: int = 1_000_000,
) -> None:
"""Write one KST object as a new CSV/XLSX/ODS export.
Use ``write_tables`` with ``kst_to_table`` to put several objects into
separate sheets of the same workbook.
"""
table = kst_to_table(
obj, items=items, header=header, count_column=count_column, missing_marker=missing_marker
)
write_tables({sheet: table}, path, format=format, delimiter=delimiter, max_cells=max_cells)