Source code for knowledgespaces.datasets.observed

"""Published response data without imposing a latent knowledge structure."""

from __future__ import annotations

import csv
from dataclasses import dataclass
from importlib.resources import files

import numpy as np

from knowledgespaces.estimation.blim_em import ResponseMatrix


[docs] @dataclass(frozen=True) class ResponseDataset: """Observed data with source and terms, without a ground-truth structure. data contains responses or frequency rows, as specified by the loader. Fitting a structure to these responses is an analysis step, not evidence that the fitted states or implications were observed directly. """ name: str description: str source: str source_license: str data: ResponseMatrix
[docs] def load_pisa(*, aggregate: bool = False) -> ResponseDataset: """Load all 340 German students and five items from DAKS::pisa. Data are dichotomized PISA 2003 mathematical-literacy responses, in original a..e column order. By default each row is a student in the R data-frame order. aggregate=True returns lexicographically ordered distinct patterns with frequencies. No rows are filtered or imputed. The source does not provide item wording, original PISA item identifiers, original polytomous scores or dichotomization rules. Source row numbers are export positions, not recovered OECD participant identifiers. No true latent-state structure or relation is supplied. GPL-2.0-or-later terms and the original codebook are packaged in data/pisa, separately from the MIT implementation. Loading requires neither R nor a network. """ if not isinstance(aggregate, bool): raise ValueError("aggregate must be a boolean.") root = files("knowledgespaces.datasets").joinpath("data").joinpath("pisa") with root.joinpath("responses.csv").open("r", encoding="utf-8") as handle: rows = list(csv.DictReader(handle)) items = list("abcde") patterns = np.array([[int(row[item]) for item in items] for row in rows], dtype=np.int8) counts = None if aggregate: patterns, frequency = np.unique(patterns, axis=0, return_counts=True) counts = frequency.astype(float) counts.flags.writeable = False patterns.flags.writeable = False return ResponseDataset( "pisa", "Dichotomized PISA 2003 mathematical-literacy responses: 340 German students, " "five items a..e, exported unchanged from DAKS 2.1-3.", "OECD PISA 2003; DAKS 2.1-3 pisa dataset. Ünlü & Sargin (2010), " "Journal of Statistical Software, 37(2), https://doi.org/10.18637/jss.v037.i02.", "GPL-2.0-or-later", ResponseMatrix(items, patterns, counts), )