Source code for knowledgespaces.datasets.observed
"""Published response data without imposing a latent knowledge structure."""
from __future__ import annotations
import csv
from dataclasses import dataclass
from importlib.resources import files
import numpy as np
from knowledgespaces.estimation.blim_em import ResponseMatrix
[docs]
@dataclass(frozen=True)
class ResponseDataset:
"""Observed data with source and terms, without a ground-truth structure.
data contains responses or frequency rows, as specified by the loader.
Fitting a structure to these responses is an analysis step, not evidence
that the fitted states or implications were observed directly.
"""
name: str
description: str
source: str
source_license: str
data: ResponseMatrix
[docs]
def load_pisa(*, aggregate: bool = False) -> ResponseDataset:
"""Load all 340 German students and five items from DAKS::pisa.
Data are dichotomized PISA 2003 mathematical-literacy responses, in
original a..e column order. By default each row is a student in the R
data-frame order. aggregate=True returns lexicographically ordered
distinct patterns with frequencies. No rows are filtered or imputed.
The source does not provide item wording, original PISA item identifiers,
original polytomous scores or dichotomization rules. Source row numbers
are export positions, not recovered OECD participant identifiers.
No true latent-state structure or relation is supplied. GPL-2.0-or-later
terms and the original codebook are packaged in data/pisa, separately
from the MIT implementation. Loading requires neither R nor a network.
"""
if not isinstance(aggregate, bool):
raise ValueError("aggregate must be a boolean.")
root = files("knowledgespaces.datasets").joinpath("data").joinpath("pisa")
with root.joinpath("responses.csv").open("r", encoding="utf-8") as handle:
rows = list(csv.DictReader(handle))
items = list("abcde")
patterns = np.array([[int(row[item]) for item in items] for row in rows], dtype=np.int8)
counts = None
if aggregate:
patterns, frequency = np.unique(patterns, axis=0, return_counts=True)
counts = frequency.astype(float)
counts.flags.writeable = False
patterns.flags.writeable = False
return ResponseDataset(
"pisa",
"Dichotomized PISA 2003 mathematical-literacy responses: 340 German students, "
"five items a..e, exported unchanged from DAKS 2.1-3.",
"OECD PISA 2003; DAKS 2.1-3 pisa dataset. Ünlü & Sargin (2010), "
"Journal of Statistical Software, 37(2), https://doi.org/10.18637/jss.v037.i02.",
"GPL-2.0-or-later",
ResponseMatrix(items, patterns, counts),
)