Source code for knowledgespaces.datasets.individual

"""Individual probability data with source identifiers, coding and pairing."""

from __future__ import annotations

import csv
import json
from collections.abc import Mapping
from dataclasses import dataclass
from importlib.resources import files
from types import MappingProxyType
from typing import Literal

import numpy as np

from knowledgespaces.estimation.incomplete import IncompleteResponseMatrix


[docs] @dataclass(frozen=True) class ProbabilityIndividualDataset: """Row-aligned pre/post source scores, metadata and raw numeric responses. case_ids retain the original pseudonymous case key; no pairing is inferred from frequency tables. records contains all 68 source columns as numeric values, strings/factor labels, ISO UTC timestamps or None. Source p values are numeric answers; source b values are the original correctness coding. schema() and source_documentation() expose types, levels and the original item wording/codebook. The source dataset and resources are GPL >= 2. """ case_ids: tuple[str, ...] pre: IncompleteResponseMatrix post: IncompleteResponseMatrix records: tuple[Mapping[str, str | float | None], ...] sample: str mode: str condition: str source: str = "Anselmi & Wickelmaier, Tuebingen 2010; probability in pks 0.7-0." source_license: str = "GPL-2.0-or-later"
[docs] def raw_answers(self, wave: Literal["pre", "post"]) -> np.ndarray: """Numeric p responses, including original NaN; not dichotomous scores.""" if wave not in ("pre", "post"): raise ValueError("wave must be 'pre' or 'post'.") part = 1 if wave == "pre" else 2 result = np.array( [[row[f"p{part}{i:02}"] for i in range(1, 13)] for row in self.records], dtype=float ) result.flags.writeable = False return result
[docs] @staticmethod def schema() -> dict: """Fresh JSON schema with source column types, factor levels and transformations.""" return json.loads(_root().joinpath("individual_schema.json").read_text(encoding="utf-8"))
[docs] @staticmethod def source_documentation() -> str: """Original pks probability Rd: item wordings, scoring and design notes.""" return _root().joinpath("PROBABILITY_SOURCE.Rd").read_text(encoding="utf-8")
def _root(): return files("knowledgespaces.datasets").joinpath("data").joinpath("probability")
[docs] def load_probability_individual( *, sample: Literal["all", "completers"] = "all", mode: Literal["all", "lab", "online"] = "all", condition: Literal["all", "basic", "enhan"] = "all", ) -> ProbabilityIndividualDataset: """Load all 504 source participants, with explicit optional selections. Completers follow the source !is.na(b201) selection (345 cases); masks in b201..b212 are identical. Pre b101..b112 is complete for all 504. Source p omissions already coded b=0 remain zeros: no rescoring/imputation. 26 lab cases all have 'enhan'; do not infer randomized lab treatment balance from the study's general description. Empty selections raise. The 504 retained source cases are not the whole recruited population. """ if ( sample not in ("all", "completers") or mode not in ("all", "lab", "online") or condition not in ("all", "basic", "enhan") ): raise ValueError("Unknown sample, mode or condition selection.") schema = ProbabilityIndividualDataset.schema() numeric = {field["name"] for field in schema["columns"] if field["type"] == "number"} with _root().joinpath("individual.csv").open("r", encoding="utf-8", newline="") as handle: original = list(csv.DictReader(handle)) rows: list[Mapping[str, str | float | None]] = [] for record in original: if sample == "completers" and record["b201"] == "": continue if (mode != "all" and record["mode"] != mode) or ( condition != "all" and record["learnobj"] != condition ): continue rows.append( MappingProxyType( { key: None if value == "" else float(value) if key in numeric else value for key, value in record.items() } ) ) if not rows: raise ValueError("The requested selection contains no participants.") ids = tuple(str(row["case"]) for row in rows) items = tuple(f"i{i:02}" for i in range(1, 13)) waves = [ IncompleteResponseMatrix( items, np.array([[row[f"b{part}{i:02}"] for i in range(1, 13)] for row in rows], dtype=float), ) for part in (1, 2) ] return ProbabilityIndividualDataset( ids, waves[0], waves[1], tuple(rows), sample, mode, condition )