"""
Classic Knowledge Space Theory datasets.
Ships the standard examples used across the KST software literature so
that analyses are directly comparable with published results (e.g., the
R packages ``pks`` and ``DAKS``). Structure/frequency loaders return a
:class:`KSTDataset`; the individual probability loader retains source
case identifiers, paired responses, metadata and provenance.
Currently included:
- :func:`load_pisa` — 340 original DAKS PISA response rows, or aggregated
frequencies, without an assumed ground-truth knowledge structure.
- :func:`load_probability_individual` — all 504 retained source cases,
with paired waves, raw answers, original scores and codebook.
- :func:`load_doignon_falmagne_7` — the canonical five-item fictitious
example of Doignon & Falmagne (1999, chapter 7).
- :func:`load_probability` — aggregated pre/post responses of the 345
completers in Anselmi & Wickelmaier's probability study, with the
original GPL-2.0-or-later data terms preserved separately from the code.
"""
from __future__ import annotations
import csv
from dataclasses import dataclass
from importlib.resources import files
from typing import Literal
import numpy as np
from knowledgespaces.datasets.individual import (
ProbabilityIndividualDataset,
load_probability_individual,
)
from knowledgespaces.datasets.observed import ResponseDataset, load_pisa
from knowledgespaces.derivation.cbkst import derive_knowledge_structure
from knowledgespaces.derivation.skill_map import SkillMap, SkillMultiMap
from knowledgespaces.estimation.blim_em import ResponseMatrix
from knowledgespaces.structures.knowledge_structure import KnowledgeStructure
from knowledgespaces.structures.relations import SurmiseRelation
__all__ = [
"KSTDataset",
"ProbabilityDataset",
"ProbabilityIndividualDataset",
"ResponseDataset",
"load_doignon_falmagne_7",
"load_pisa",
"load_probability",
"load_probability_individual",
]
[docs]
@dataclass(frozen=True)
class KSTDataset:
"""A packaged KST dataset with provenance.
Attributes
----------
name : str
Short identifier (matches the loader name).
description : str
What the data are and how they were collected/constructed.
source : str
Full bibliographic source of the data.
structure : KnowledgeStructure
The knowledge structure associated with the data in the source.
data : ResponseMatrix
Observed response-pattern frequencies, items in sorted order.
"""
name: str
description: str
source: str
structure: KnowledgeStructure
data: ResponseMatrix
[docs]
@dataclass(frozen=True)
class ProbabilityDataset(KSTDataset):
"""Aggregated probability data with a delineating skill function.
``wave`` identifies pre/post instruction; ``model`` selects K1 (sf1,
conjunctive) or K2 (sf2, alternative competencies). ``skill_map``
exposes the original item-to-skill mapping. Both waves describe the
same 345 completers out of 504 source cases, but aggregate frequencies
do not retain respondent pairing, treatment groups, or missing cases.
They cannot support longitudinal individual-level or causal analyses.
Data and skill-function CSVs retain GPL-2.0-or-later terms from pks;
see the packaged data/probability/README.md and COPYING files.
"""
skill_map: SkillMap | SkillMultiMap
wave: str
model: str
source_license: str = "GPL-2.0-or-later"
[docs]
def load_probability(
*, wave: Literal["pre", "post"] = "pre", model: Literal["K1", "K2"] = "K2"
) -> ProbabilityDataset:
"""Load the probability study's completer sample, without R or downloads.
Data collected by Pasquale Anselmi and Florian Wickelmaier,
University of Tuebingen, February-March 2010; distributed in pks
0.7-0 as ``probability``. Selection follows the source example:
``!is.na(probability$b201)`` (N=345). Binary items b101..b112 (pre)
or b201..b212 (post) are aligned as i01..i12. Frequencies describe
110 (pre) or 84 (post) distinct response patterns.
K1 and K2 are derived from the source skill functions, with 16 and
13 states respectively; neither is a knowledge space. They are
candidate models, not observed ground-truth knowledge states.
"""
if wave not in ("pre", "post"):
raise ValueError("wave must be 'pre' or 'post'.")
if model not in ("K1", "K2"):
raise ValueError("model must be 'K1' or 'K2'.")
root = files(__package__).joinpath("data").joinpath("probability")
with root.joinpath(f"freq_{wave}.csv").open("r", encoding="utf-8") as handle:
rows = list(csv.DictReader(handle))
items = [f"i{i:02d}" for i in range(1, 13)]
data = ResponseMatrix(
items=items,
patterns=np.array([[int(c) for c in row["pattern"]] for row in rows], dtype=int),
counts=np.array([int(row["freq"]) for row in rows], dtype=np.float64),
)
sf = "sf1" if model == "K1" else "sf2"
with root.joinpath(f"skill_function_{sf}.csv").open("r", encoding="utf-8") as handle:
clauses = list(csv.DictReader(handle))
skills = ["cp", "id", "pb", "un"]
mapping: dict[str, list[frozenset[str]]] = {q: [] for q in items}
for row in clauses:
mapping[f"i{int(row['item']):02d}"].append(frozenset(s for s in skills if row[s] == "1"))
skill_map: SkillMap | SkillMultiMap
if model == "K1":
skill_map = SkillMap(items, skills, {q: comp[0] for q, comp in mapping.items()})
else:
skill_map = SkillMultiMap(items, skills, mapping)
structure = derive_knowledge_structure(
skill_map, SurmiseRelation(skills, [])
).knowledge_structure
return ProbabilityDataset(
name=f"probability_{wave}_{model}",
description=(
f"Elementary probability responses before/after instruction: {wave} wave, "
f"345 completers, 12 binary items, candidate structure {model}. "
"Aggregated frequencies; respondent pairing is unavailable."
),
source=(
"Anselmi, P., & Wickelmaier, F. (data collected 2010, Tuebingen). "
"Problems in Elementary Probability Theory, probability dataset in "
"pks 0.7-0, https://CRAN.R-project.org/package=pks. "
"K1/sf1 and K2/sf2 from the accompanying probability documentation."
),
structure=structure,
data=data,
skill_map=skill_map,
wave=wave,
model=model,
)
# Response-pattern frequencies of the fictitious 1000-respondent example in
# Doignon & Falmagne (1999, ch. 7). Pattern strings are in item order
# "abcde". Cross-checked against the `DoignonFalmagne7` dataset of the R
# package pks (v0.7-0); the numbers originate in the book, which is the
# canonical source to cite.
_DF7_FREQUENCIES: dict[str, int] = {
"00000": 80,
"10000": 92,
"01000": 89,
"00100": 3,
"00010": 2,
"00001": 1,
"11000": 89,
"10100": 16,
"10010": 18,
"10001": 10,
"01100": 18,
"01010": 20,
"01001": 4,
"00110": 2,
"00101": 2,
"00011": 3,
"11100": 89,
"11010": 89,
"11001": 19,
"10110": 16,
"10101": 16,
"10011": 3,
"01110": 18,
"01101": 16,
"01011": 2,
"00111": 2,
"11110": 73,
"11101": 82,
"11011": 19,
"10111": 15,
"01111": 15,
"11111": 77,
}
_DF7_STATES: list[frozenset[str]] = [
frozenset(),
frozenset({"a"}),
frozenset({"b"}),
frozenset({"a", "b"}),
frozenset({"a", "b", "c"}),
frozenset({"a", "b", "d"}),
frozenset({"a", "b", "c", "d"}),
frozenset({"a", "b", "c", "e"}),
frozenset({"a", "b", "c", "d", "e"}),
]
[docs]
def load_doignon_falmagne_7() -> KSTDataset:
"""Load the Doignon & Falmagne (1999, ch. 7) five-item example.
The canonical toy example of knowledge space theory: five items
(a-e), a knowledge structure with nine states that is a learning
space, and the response patterns of 1000 fictitious respondents.
Known as ``DoignonFalmagne7`` in the R package ``pks`` — the "7"
refers to the book chapter.
Returns
-------
KSTDataset
``structure`` has 9 states over items ``a..e``; ``data`` holds
all 32 response patterns with their frequencies (N = 1000).
Examples
--------
>>> ds = load_doignon_falmagne_7()
>>> ds.structure.n_states
9
>>> int(ds.data.effective_counts.sum())
1000
"""
structure = KnowledgeStructure.from_states(_DF7_STATES)
items = ["a", "b", "c", "d", "e"]
patterns = np.array([[int(c) for c in p] for p in _DF7_FREQUENCIES], dtype=int)
counts = np.array(list(_DF7_FREQUENCIES.values()), dtype=np.float64)
data = ResponseMatrix(items=items, patterns=patterns, counts=counts)
return KSTDataset(
name="doignon_falmagne_7",
description=(
"Fictitious responses of 1000 respondents to five items (a-e) "
"under a nine-state learning space; the canonical textbook "
"example of knowledge space theory."
),
source=(
"Doignon, J.-P., & Falmagne, J.-C. (1999). Knowledge Spaces, "
"chapter 7. Springer-Verlag. (Distributed as DoignonFalmagne7 "
"in the R package pks.)"
),
structure=structure,
data=data,
)