[
  {
    "id": "corbett_anderson_1995_bkt",
    "tier": "Tier 1",
    "full_citation": "Corbett, A. T., & Anderson, J. R. (1995). Knowledge tracing: Modeling the acquisition of procedural knowledge. User Modeling and User-Adapted Interaction, 4(4), 253–278. https://doi.org/10.1007/BF01099821",
    "link": "https://doi.org/10.1007/BF01099821",
    "peer_reviewed": true,
    "population": "Students using an ACT programming tutor across four empirical studies; later reported study groups included 20, 20, and 25 students.",
    "design": "An expert ideal-student model supplied production rules. Model tracing matched actions to applicable rules, and knowledge tracing maintained a learner-specific two-state learned/unlearned probability for each credited rule, updated after successive opportunities.",
    "finding_verbatim": "“the tutor also maintains an estimate of the probability that the student has learned each of the rules in the ideal model”",
    "effect_size": "Experiment 1 prediction with previously estimated common parameters gave r=.47 and MAE=.16; same-data rule-specific fitting gave r=.85 and MAE=.07. In Experiment 4, 56% of the knowledge-tracing group versus 24% of a comparison group met a 90% test criterion, z=2.21, p<.05, but the tracing group completed about 76% more exercises.",
    "conditions_and_limits": "Canonical BKT uses P(L0), P(T), P(G), and P(S), assumes a binary state, one-way acquisition, no forgetting, constant parameters, and usually one credited production per action. Opportunity count, not elapsed time, is the clock. Some reported fits were in-sample.",
    "criticism": "The paper established Bayesian longitudinal inference of node-level rule mastery, but observed performance rather than durable learning. Post-mastery predictions depended heavily on slip, whose meaning could be transient error, weak knowledge, or model-content mismatch.",
    "which_workflow": "Q10: canonical prior art and the baseline comparison for every claimed Bayesian learner-model contribution.",
    "claimable_sentence": "Canonical BKT has represented learners as person-specific Bayesian probabilities of mastering expert-defined rules since 1995, updating those probabilities from sequential performance evidence."
  },
  {
    "id": "baker_corbett_aleven_2008_contextual_bkt",
    "tier": "Tier 2",
    "full_citation": "Baker, R. S. J. d., Corbett, A. T., & Aleven, V. (2008). More accurate student modeling through contextual estimation of slip and guess probabilities in Bayesian Knowledge Tracing. In B. P. Woolf, E. Aïmeur, R. Nkambou, & S. Lajoie (Eds.), Intelligent Tutoring Systems (LNCS 5091, pp. 406–415). Springer. https://doi.org/10.1007/978-3-540-69132-7_44",
    "link": "https://doi.org/10.1007/978-3-540-69132-7_44",
    "peer_reviewed": true,
    "population": "232 students aged approximately 12–14 using a middle-school mathematics tutor in 2002–2003; 581,785 transactions on 171,987 tagged problem steps covering 253 skills in 37 lessons or units.",
    "design": "Comparison of baseline, bounded, Dirichlet-prior, and contextual BKT. The contextual model used machine-learned action features to estimate whether each correct response was a guess and each incorrect response was a slip.",
    "finding_verbatim": "“our method leads to substantially higher accuracy than prior methods”",
    "effect_size": "Baseline A′=.66 and r=.29; contextual BKT A′=.75 and r=.43. The authors reported 27% of the possible A′ gain, Z=2.86, p<.01, and a 48% correlation increase, t(171984)=69.12, p<.0001; student-collapsed t(231)=7.95, p<.0001.",
    "conditions_and_limits": "The guess/slip classifiers were trained on 64 skills and tested across 253. Prediction used tagged tutor actions, and the principal outcomes were fit to action correctness. Thirteen of 758 ten-correct sequences still remained below the .95 mastery threshold.",
    "criticism": "The study shows that BKT's observation model can incorporate action context. It does not infer associations among knowledge elements, and improved next-action prediction is not independent evidence of retention, transfer, or learning.",
    "which_workflow": "Q10: established variants of BKT's evidence model and the distinction between behavioral prediction and learning measurement.",
    "claimable_sentence": "BKT variants already use contextual features of learner actions to update latent mastery more accurately, but their validation remains primarily performance-predictive."
  },
  {
    "id": "yudelson_koedinger_gordon_2013_individualized_bkt",
    "tier": "Tier 2",
    "full_citation": "Yudelson, M. V., Koedinger, K. R., & Gordon, G. J. (2013). Individualized Bayesian Knowledge Tracing models. In H. C. Lane, K. Yacef, J. Mostow, & P. Pavlik (Eds.), Artificial Intelligence in Education (LNAI 7926, pp. 171–180). Springer. https://doi.org/10.1007/978-3-642-39112-5_18",
    "link": "https://doi.org/10.1007/978-3-642-39112-5_18",
    "peer_reviewed": true,
    "population": "Learners in Cognitive Tutor Algebra I and Bridge to Algebra datasets, each analyzed under two alternative skill models.",
    "design": "User-stratified ten-fold cross-validation compared standard skill-parameter BKT with variants adding student-specific initial-mastery, learning-rate, guess, and slip parameters. Parameters were optimized without changing the underlying hidden Markov structure.",
    "finding_verbatim": "“student-specific parameters lead to a tangible improvement when predicting the data of unseen students”",
    "effect_size": "For one Algebra I skill model, student-specific P(T) changed RMSE from .36273 to .36116 and accuracy from .82755 to .82896. For one Bridge-to-Algebra model, RMSE changed from .36294 to .35851 and accuracy from .82261 to .82957.",
    "conditions_and_limits": "Student-specific learning-rate parameters consistently outperformed standard BKT and initial-mastery individualization, but gains were small and the result table did not report significance tests. Outcomes were next-response predictions for held-out learners.",
    "criticism": "Individualization strengthens the prior-art case for a person-specific Bayesian learner representation. The represented variables remain mastery probabilities for individual skills, not posterior strengths of learner-specific edges among heuristics.",
    "which_workflow": "Q10: prior art for learner-specific BKT parameters and out-of-sample prediction.",
    "claimable_sentence": "Individualized BKT predates the proposed mechanism and can fit person-specific learning-rate parameters, although its latent target remains mastery of individual skills."
  },
  {
    "id": "beck_chang_2007_identifiability",
    "tier": "Tier 2",
    "full_citation": "Beck, J. E., & Chang, K.-m. (2007). Identifiability: A fundamental problem of student modeling. In C. Conati, K. McCoy, & G. Paliouras (Eds.), User Modeling 2007 (LNAI 4511, pp. 137–146). Springer. https://doi.org/10.1007/978-3-540-73078-1_17",
    "link": "https://doi.org/10.1007/978-3-540-73078-1_17",
    "peer_reviewed": true,
    "population": "360 children, mostly aged six to eight, who used Project LISTEN's Reading Tutor in 2002–2003, reading approximately 1.95 million words; 3,532 distinct words were treated as skills.",
    "design": "The paper analytically compared BKT parameterizations with the same marginal performance curve, then compared expectation-maximization fits with and without Dirichlet parameter priors using an approximately equal student-level train/test split.",
    "finding_verbatim": "“observed student performance corresponds to an infinite family of possible model parameter estimates”",
    "effect_size": "Training AUC was .653±.002 for both models. On unseen test data, the Dirichlet-prior model achieved AUC .620±.002 versus .614±.002 for baseline BKT. Prior-knowledge parameter variance explained by log word frequency was 16% with priors versus 1% without.",
    "conditions_and_limits": "The empirical evaluation used one reading-tutor dataset. Its broad non-identifiability argument compared marginal learning curves rather than the full joint distribution of response histories and was later corrected by Doroudi and Brunskill.",
    "criticism": "The lasting warning is that predictive equivalence at the aggregate-curve level does not validate a latent knowledge interpretation. The stronger claim that full BKT is fundamentally unidentifiable is not supported after later formal analysis.",
    "which_workflow": "Q10: historical identifiability critique, parameter plausibility, and construct-validity warning.",
    "claimable_sentence": "Aggregate BKT learning curves can be compatible with substantively different mastery estimates, so performance fit alone cannot validate the meaning of a latent learning state."
  },
  {
    "id": "doroudi_brunskill_2017_identifiability",
    "tier": "Tier 2",
    "full_citation": "Doroudi, S., & Brunskill, E. (2017). The misidentified identifiability problem of Bayesian Knowledge Tracing. Proceedings of the 10th International Conference on Educational Data Mining, 143–149.",
    "link": "https://files.eric.ed.gov/fulltext/ED577166.pdf",
    "peer_reviewed": true,
    "population": "No human intervention sample; the paper uses formal hidden-Markov-model analysis and simulations, including data generated from alternative ten-state learning models.",
    "design": "The authors distinguish marginal response curves from full joint response-sequence distributions, apply existing hidden-Markov identifiability results to BKT, and simulate model fitting to examine semantic degeneracy and sequence-length effects.",
    "finding_verbatim": "“under mild conditions on the parameters, BKT is actually identifiable”",
    "effect_size": "No intervention effect size is applicable. The formal result requires P(L0) not equal to 0 or 1, P(T) not equal to 1, and P(G) not equal to 1−P(S); joint distributions through three observations can identify the model in principle.",
    "conditions_and_limits": "Structural identifiability assumes the data-generating process is accurately represented by standard BKT and does not remove finite-sample error or local optimization failures. Simulations show that misspecification can yield semantically degenerate parameters.",
    "criticism": "A unique statistical solution can still be inconsistent with BKT's conceptual mastery assumptions. Identifiability therefore does not establish that the hidden state is learning, mastery, or any other intended psychological construct.",
    "which_workflow": "Q10: corrected identifiability verdict and the distinction between statistical identification and construct validity.",
    "claimable_sentence": "Standard BKT is generically identifiable from full response histories under mild nondegeneracy conditions, but identifiable parameters can still lack a defensible mastery interpretation."
  },
  {
    "id": "van_de_sande_2013_bkt_properties",
    "tier": "Tier 1",
    "full_citation": "van de Sande, B. (2013). Properties of the Bayesian Knowledge Tracing model. Journal of Educational Data Mining, 5(2), 1–10. https://doi.org/10.5281/zenodo.3554629",
    "link": "https://doi.org/10.5281/zenodo.3554629",
    "peer_reviewed": true,
    "population": "No learner sample; this is a mathematical analysis of the population-average Markov-chain form and the individual response-conditioned hidden-Markov form of BKT.",
    "design": "Analytical solution of the population-average learning curve and fixed-point analysis of individual Bayesian filtering, with explicit characterization of parameter constraints needed for intended mastery behavior.",
    "finding_verbatim": "“The probability that the student has learned a skill is updated by student performance at each opportunity to apply that skill.”",
    "effect_size": "No empirical effect size is applicable. The population-average curve reduces to a three-parameter exponential, while the individual algorithm conditions on each learner's response history.",
    "conditions_and_limits": "The analysis concerns properties of the mathematical model. It assumes binary observations and states, supplied parameters, constant P(T), and the standard no-forgetting transition structure.",
    "criticism": "The distinction explains why an aggregate curve can appear underdetermined while an individual sequence model is informative. Neither mathematical behavior nor filter stability validates the hidden state's interpretation as durable learning.",
    "which_workflow": "Q10: formal mechanics of BKT and reconciliation of population-curve and individual-filter arguments.",
    "claimable_sentence": "BKT's individual hidden-state filter updates mastery after every observed opportunity, whereas its population-average learning curve discards response-history information."
  },
  {
    "id": "pardos_heffernan_2011_kt_idem",
    "tier": "Tier 2",
    "full_citation": "Pardos, Z. A., & Heffernan, N. T. (2011). KT-IDEM: Introducing item difficulty to the Knowledge Tracing model. In J. A. Konstan, R. Conejo, J. L. Marzo, & N. Oliver (Eds.), User Modeling, Adaption and Personalization (LNCS 6787, pp. 243–254). Springer. https://doi.org/10.1007/978-3-642-22362-4_21",
    "link": "https://doi.org/10.1007/978-3-642-22362-4_21",
    "peer_reviewed": true,
    "population": "Student-response data from ten ASSISTments datasets and twelve Cognitive Tutor datasets representing multiple real-world tutoring scenarios.",
    "design": "Student-level cross-validation compared standard BKT with KT-IDEM, a minimally altered Bayesian-network topology that assigns item-specific guess and slip parameters to represent item difficulty.",
    "finding_verbatim": "“substantial performance gains can be achieved with this minor modification that incorporates item difficulty”",
    "effect_size": "Across ten ASSISTments datasets, mean AUC was .669 for BKT and .699 for KT-IDEM, p<.05. Across twelve Cognitive Tutor datasets, .6457 versus .6441, p=.96; for five datasets with more than six observations per problem, .6124 versus .7108, p=.02.",
    "conditions_and_limits": "Benefits depended on platform and observations per item. Sparse item data caused overparameterization, and the principal criterion was prediction of held-out responses rather than an independent learning measure.",
    "criticism": "Item difficulty is established BKT prior art, showing that the observation model need not be uniform. It still updates skill mastery nodes rather than learner-specific relations among heuristics.",
    "which_workflow": "Q10: scope of BKT variants and evidence that additional contextual parameters have conditional, not universal, benefits.",
    "claimable_sentence": "BKT has been extended to model item-specific difficulty, with predictive gains in some sufficiently dense datasets and no overall gain in another tutor corpus."
  },
  {
    "id": "kaeser_klingler_schwing_gross_2017_dbn",
    "tier": "Tier 1",
    "full_citation": "Käser, T., Klingler, S., Schwing, A. G., & Gross, M. (2017). Dynamic Bayesian networks for student modeling. IEEE Transactions on Learning Technologies, 10(4), 450–462. https://doi.org/10.1109/TLT.2017.2689017",
    "link": "https://doi.org/10.1109/TLT.2017.2689017",
    "peer_reviewed": true,
    "population": "Five tutoring-system datasets spanning mathematics, spelling, algebra, and physics, with learners from elementary school through university and datasets containing up to approximately 7,000 students.",
    "design": "Predictive comparison of conventional independent-skill BKT with dynamic Bayesian networks that jointly represent multiple latent skills on domain-specific prerequisite topologies; parameters were fitted by constrained optimization and evaluated on unseen responses.",
    "finding_verbatim": "“Dynamic Bayesian networks (DBN) on the other hand are able to represent multiple skills jointly within one model.”",
    "effect_size": "Number representation: cross-entropy .3141 to .2783, RMSE .4550 to .4378, AUC .5975 to .7093. Subtraction: cross-entropy .2818 to .2580, RMSE .4368 to .4215, AUC .5996 to .6916. Physics: cross-entropy .2930 to .2616 and RMSE .4530 to .4244.",
    "conditions_and_limits": "Benefits were largest in domains with meaningful hierarchy and small in spelling. Domain topology was supplied or designed; the model learned conditional parameters and learner mastery states rather than discovering a separate graph for every learner.",
    "criticism": "This directly occupies Bayesian modeling of relations among skills but not the narrower mechanism of estimating person-specific association edges from a retrieved-versus-enacted contrast. Prediction gains do not validate graph movement as learning.",
    "which_workflow": "Q10/Q12 bridge: relationship-aware Bayesian student models and the boundary between skill mastery and edge estimation.",
    "claimable_sentence": "Dynamic Bayesian student models already represent prerequisite relations among multiple skills and can outperform independent-skill BKT on hierarchical domains."
  },
  {
    "id": "saric_grgic_grubisic_gaspar_2024_systematic_review",
    "tier": "Tier 1",
    "full_citation": "Šarić-Grgić, I., Grubišić, A., & Gašpar, A. (2024). Twenty-five years of Bayesian knowledge tracing: A systematic review. User Modeling and User-Adapted Interaction, 34, 1127–1173. https://doi.org/10.1007/s11257-023-09389-4",
    "link": "https://doi.org/10.1007/s11257-023-09389-4",
    "peer_reviewed": true,
    "population": "The BKT-enhancement literature identified through a PRISMA-guided systematic review and classified across 13 enhancement and computational-method dimensions.",
    "design": "Systematic review distinguishing evaluations based on student-answer prediction from evaluations of estimated knowledge mastery; it also catalogs student, tutor, domain, item-difficulty, and architectural extensions.",
    "finding_verbatim": "“only a few studies further investigated the systems’ estimations of knowledge mastery by correlating it to knowledge on post-tests”",
    "effect_size": "No new pooled effect size was calculated. The review reports that enhanced BKT models generally outperformed vanilla BKT on RMSE, AUC-ROC, or accuracy, while only a few studies checked mastery estimates against post-tests.",
    "conditions_and_limits": "This is secondary evidence across heterogeneous datasets, variants, and metrics. Its wording supports a field-level validation gap and should not be converted into an aggregate numerical effect or treated as validation of any one mechanism.",
    "criticism": "The review makes the use-versus-learning problem central: most evidence concerns response prediction, not durable acquisition, retention, transfer, or independent construct validation.",
    "which_workflow": "Q10 and the paper's construct-validity synthesis: distinguish modeled mastery and next-response prediction from demonstrated learning.",
    "claimable_sentence": "Across BKT research, next-answer prediction is evaluated far more often than latent mastery against post-tests, leaving the learning interpretation comparatively weakly tested."
  },
  {
    "id": "chen_gonzalez_brenes_tian_2016_command",
    "tier": "Tier 2",
    "full_citation": "Chen, Y., González-Brenes, J. P., & Tian, J. (2016). Joint discovery of skill prerequisite graphs and student models. Proceedings of the 9th International Conference on Educational Data Mining, 46–53.",
    "link": "https://www.educationaldatamining.org/EDM2016/proceedings/paper_89.pdf",
    "peer_reviewed": true,
    "population": "ECPE: 2,922 examinees, 28 items, three English skills. Mathematics Chapter 2: 1,720 students, 30 items, six skills. Chapter 3: 1,245 students, 33 items, seven skills.",
    "design": "COMMAND uses Structural EM to learn a Bayesian-network structure among latent skill variables while estimating conditional-probability parameters and student mastery from item correctness and a supplied Q-matrix.",
    "finding_verbatim": "“simultaneously discovering prerequisite structure of skills and a student model from student performance data”",
    "effect_size": "Ten-fold cross-validation with reported 95% confidence intervals: Chapter 2 AUC .803 ± .008 for COMMAND versus .791 ± .007 for a fully connected model, p=.0022; Chapter 3 .775 ± .007 versus .765 ± .008, p=.01.",
    "conditions_and_limits": "Mathematics data were tests administered after all relevant skills had been taught; filtered datasets had complete responses. Reversible edges required a substantive prerequisite assumption for orientation.",
    "criticism": "COMMAND learns one population/domain graph from a cohort. The graph is not updated online for each learner, and observed evidence is answer correctness rather than use of a knowledge element or strategy.",
    "which_workflow": "Q12: direct prior art for data-driven skill-to-skill Bayesian-network structure learning.",
    "claimable_sentence": "COMMAND shows that a domain-level graph of latent skill dependencies can be learned jointly with student mastery from cohort response data."
  },
  {
    "id": "han_yoon_yoo_2017_bayesian_prerequisites",
    "tier": "Tier 2",
    "full_citation": "Han, S.-Y., Yoon, J., & Yoo, Y. J. (2017). Discovering skill prerequisite structure through Bayesian estimation and nested model comparison. Proceedings of the 10th International Conference on Educational Data Mining, 398–399.",
    "link": "https://educationaldatamining.org/EDM2017/proc_files/papers/paper_149.pdf",
    "peer_reviewed": true,
    "population": "Simulation: 500 datasets of 1,000 students for each of five prerequisite structures. Real data: 936 eighth-grade students, 16 items, and four linear-equation/inequality skills.",
    "design": "Static Bayesian skill network with a DINA response model; MCMC estimation and pseudo-Bayes-factor nested-model tests determine whether strict prerequisite constraints fit the data.",
    "finding_verbatim": "“discover prerequisite structure from data using nested model comparisons in the context of Bayesian estimation”",
    "effect_size": "Simulation true-structure recovery ranged .816–.926 and true-edge recovery .937–.962 under low slip and guess probabilities. In real data, every expert-specified edge and one additional edge were recovered.",
    "conditions_and_limits": "The simulation used balanced Q-matrices, 1,000 students, and slip/guess probabilities drawn from Uniform(0,.05). The real application had only four skills; the authors call for evaluation under broader conditions.",
    "criticism": "This is strictly Bayesian edge evidence, but it estimates a static population prerequisite structure at one assessment occasion, not longitudinal learner-specific associations or learning from observed use.",
    "which_workflow": "Q12: strongest direct Bayesian precedent for estimating edges between educational knowledge elements.",
    "claimable_sentence": "Bayesian MCMC and model comparison have been used to estimate population-level prerequisite edges among latent skills from item responses."
  },
  {
    "id": "zemla_austerweil_2018_uinvite",
    "tier": "Tier 1",
    "full_citation": "Zemla, J. C., & Austerweil, J. L. (2018). Estimating semantic networks of groups and individuals from fluency data. Computational Brain & Behavior, 1(1), 36–58. https://doi.org/10.1007/s42113-018-0003-7",
    "link": "https://doi.org/10.1007/s42113-018-0003-7",
    "peer_reviewed": true,
    "population": "Simulation used a 160-node, 393-edge animal network and 50 simulated participants with three lists of 35 animals. Human study used 50 participants who produced three lists in each of animals, foods, and tools; 101 additional raters judged edge similarity.",
    "design": "U-INVITE infers an unweighted semantic network from fluency sequences under a censored-random-walk retrieval model. Hierarchical U-INVITE jointly estimates a group graph and separate individual graphs in a Bayesian model.",
    "finding_verbatim": "“estimating semantic networks from semantic fluency data ... based on a censored random walk model of memory retrieval”",
    "effect_size": "In the human validation, the hierarchical group network's estimated edges had mean similarity 66.4 versus 23.9 for randomly selected non-edges. The simulation demonstrated convergence under the assumed retrieval process; no standardized intervention effect was reported.",
    "conditions_and_limits": "Validity is conditional on fluency sequences being generated by the censored-random-walk process. Networks are binary, symmetric, and undirected. Multiple lists per individual are required, and estimation is computationally intensive.",
    "criticism": "This occupies Bayesian inference of person-specific concept-association edges from retrieval behavior, but it is cross-sectional semantic-memory estimation, not a longitudinal learning measure or educational strategy-use model.",
    "which_workflow": "Q12: closest generic prior art for Bayesian estimation of an individual's association graph from observed retrieval.",
    "claimable_sentence": "A Bayesian method already estimates individual concept-association networks from personal retrieval sequences, so individual Bayesian association graphs are not new in themselves."
  },
  {
    "id": "allegre_yessad_luengo_2023_eprism",
    "tier": "Tier 2",
    "full_citation": "Allègre, O., Yessad, A., & Luengo, V. (2023). Discovering prerequisite relationships between knowledge components from an interpretable learner model. Proceedings of the 16th International Conference on Educational Data Mining, 490–496. https://doi.org/10.5281/zenodo.8115738",
    "link": "https://doi.org/10.5281/zenodo.8115738",
    "peer_reviewed": true,
    "population": "Selected pairs of mathematics knowledge components from ASSISTments 2012, Eedi 2020, and Kartable; only learners who practiced both KCs were included, and seven transactions per learner were retained for tractability.",
    "design": "E-PRISM is a dynamic Bayesian knowledge-tracing model with interpretable parameters for learning, forgetting, and candidate prerequisite effects. Parameters are learned with Monte Carlo EM; derived metrics infer an edge's existence, direction, and strength.",
    "finding_verbatim": "“retrieve the prerequisite structure of a domain model from learner traces”",
    "effect_size": "Agreement among edge-existence metrics was weak: Cohen's kappa .133, -.071, and .053. Direction agreement was .325 or .55. Two metrics had kappa .778 when existence and direction were combined.",
    "conditions_and_limits": "Analyses were pairwise rather than full-graph because of tractability. There was no expert gold-standard evaluation; one result reversed the expected addition/multiplication order, and the authors explicitly call for expert validation.",
    "criticism": "The method learns domain-level candidate relations from aggregated learner traces rather than a changing edge graph for each learner. Weak metric agreement makes edge-existence claims preliminary.",
    "which_workflow": "Q12: near-direct temporal Bayesian-network prior art and a caution about identifiability and validation.",
    "claimable_sentence": "E-PRISM extracts candidate KC prerequisite relations from temporal learner traces, but its real-data edge evidence is preliminary and population-level."
  },
  {
    "id": "desmarais_meshkinfam_gagnon_2006_item_graphs",
    "tier": "Tier 1",
    "full_citation": "Desmarais, M. C., Meshkinfam, P., & Gagnon, M. (2006). Learned student models with item to item knowledge structures. User Modeling and User-Adapted Interaction, 16(5), 403–434. https://doi.org/10.1007/s11257-006-9016-3",
    "link": "https://doi.org/10.1007/s11257-006-9016-3",
    "peer_reviewed": true,
    "population": "Two item-response datasets: 149 grade 6–8 pupils answering 20 fraction-arithmetic items and 47 respondents answering 33 UNIX shell-command items.",
    "design": "Comparison of Bayesian-network structure learning and a partial-order knowledge-structure induction method for learning item-to-item probabilistic relations and predicting unobserved item outcomes from partial evidence.",
    "finding_verbatim": "“improve predictive power of a Bayesian network for item outcome, but that improvement does not transfer to the concept assessment”",
    "effect_size": "No standardized effect is reported for the claim at issue; the paper evaluates curves of item-outcome prediction as evidence accumulates and reports that the added data improved item prediction without improving concept assessment in that experiment.",
    "conditions_and_limits": "Nodes are observable test items rather than latent knowledge components. The structures are induced from cohort response matrices and evaluated mainly for item prediction; the true item graph is unavailable.",
    "criticism": "This is early Bayesian/data-driven edge learning in student models, but the edges join items, not a learner's heuristics, and the learned structure is not an individual longitudinal graph.",
    "which_workflow": "Q12: early prior art showing that Bayesian item-to-item knowledge structures were learned from response data by 2006.",
    "claimable_sentence": "Student-model research has long learned probabilistic item-to-item structures from response data, although item graphs are not the same construct as individual heuristic associations."
  },
  {
    "id": "shaffer_collier_ruis_2016_ena",
    "tier": "Tier 1",
    "full_citation": "Shaffer, D. W., Collier, W., & Ruis, A. R. (2016). A tutorial on epistemic network analysis: Analyzing the structure of connections in cognitive, social, and interaction data. Journal of Learning Analytics, 3(3), 9–45. https://doi.org/10.18608/jla.2016.33.3",
    "link": "https://doi.org/10.18608/jla.2016.33.3",
    "peer_reviewed": true,
    "population": "Methodological tutorial illustrated with coded cognitive, discourse, and interaction data; not a single intervention sample.",
    "design": "Epistemic Network Analysis accumulates co-occurrences among a fixed set of coded elements within defined conversational or temporal windows, producing weighted networks for units such as individuals or groups and supporting comparison over time.",
    "finding_verbatim": "“identifying and quantifying connections among elements in coded data and representing them in dynamic network models”",
    "effect_size": "No standardized effect; this is a methodological article.",
    "conditions_and_limits": "Edge meaning depends on the codebook, unit of analysis, conversation definition, and co-occurrence window. ENA is not a Bayesian posterior-update model.",
    "criticism": "ENA directly occupies graphs of associations among enacted ideas, skills, practices, or habits of mind, but its edges are constructed from coded co-occurrence and do not by themselves demonstrate latent learning.",
    "which_workflow": "Q12 and construct validity: non-Bayesian prior art for individual dynamic use-association networks.",
    "claimable_sentence": "Epistemic Network Analysis already represents patterns among coded knowledge, skills, and practices as individual or group networks that can change over time."
  },
  {
    "id": "bernholt_lossjew_gombert_2026_longitudinal_networks",
    "tier": "Tier 1",
    "full_citation": "Bernholt, S., Lossjew, J., & Gombert, S. (2026). Analyzing students’ conceptual understanding over the course of a teaching unit: Tracking changes in knowledge structures over time. Unterrichtswissenschaft. Advance online publication. https://doi.org/10.1007/s42010-026-00244-0",
    "link": "https://doi.org/10.1007/s42010-026-00244-0",
    "peer_reviewed": true,
    "population": "N=300 students in grades 11–13 across 15 regular chemistry classes; a 10–12-week chemical-kinetics unit with 86 tasks and 17 knowledge elements.",
    "design": "Proof-of-concept longitudinal observational study. Knowledge elements detected in each learner's answers and artifacts became nodes; co-occurrence in the same response created edges. Individual growing networks were summarized and related to an end-of-unit test.",
    "finding_verbatim": "“individual knowledge networks that reflect students’ ability to enact and connect specific knowledge elements across the unit”",
    "effect_size": "Combined network-metric model: R²=.51, adjusted R²=.47. Size summaries: R²=.44, adjusted .42. Density: R²=.40, adjusted .38. Connectedness: R²=.31, adjusted .28. Area under the network-density trajectory was negative in phase 1 (standardized beta=-.46) and positive in phase 2 (beta=.90).",
    "conditions_and_limits": "Human scoring of posttest knowledge elements had kappa=.91 and test WLE reliability=.85. Artifact scoring model overall F1=89.85. Although 300 students enrolled, the reported regression degrees of freedom imply complete-case analytic samples of roughly 202–204, depending on model; this is an inference from the degrees of freedom because attrition is not stated in the searchable report text. The design is observational and domain-specific; the pretest was collected but not included in reported models, and no control group supports causal attribution.",
    "criticism": "This is the closest educational construct precedent: an individual evolving co-enactment graph related to later performance. It is not Bayesian, lacks an available-versus-enacted contrast, and shows that denser graphs can relate to lower performance early in learning.",
    "which_workflow": "Q12 pivotal prior art and §5 construct validity: direct evidence for, and limits of, interpreting edge movement as learning.",
    "claimable_sentence": "Individual knowledge-element co-enactment networks can relate substantially to later achievement, but the sign of a connectivity–performance relation can reverse across phases, so edge growth is not learning by definition."
  },
  {
    "id": "ji_wang_wu_zhou_2026_hmckt",
    "tier": "Tier 1",
    "full_citation": "Ji, W., Wang, H., Wu, Q., & Zhou, G. (2026). Knowledge tracing model based on human-machine collaboration: An analysis of the impact of perceptual ambiguity, selective attention, and heuristic judgment on learning performance. Journal of Big Data, 13, Article 47. https://doi.org/10.1186/s40537-026-01385-w",
    "link": "https://doi.org/10.1186/s40537-026-01385-w",
    "peer_reviewed": true,
    "population": "Three public knowledge-tracing datasets: ASSISTments (4,151 learners; 325,678 interactions; 110 knowledge components), KDD Cup (574 learners; 607,026 interactions; 436 concepts), and STATICS2011 (333 learners; 189,927 interactions; 1,223 knowledge components).",
    "design": "Benchmark prediction study using 80/20 sequence splits and a spatiotemporal graph-convolution model with a learnable knowledge-component adjacency matrix; reported experiments were repeated three to five times per dataset.",
    "finding_verbatim": "“This study dynamically adjusts the weights of spatial dimensions using a learnable adjacency matrix.”",
    "effect_size": "The article reports AUC=.8593 for the full HMCKT model and .8194 after removing its active-learning component; these are predictive benchmark values, not human-learning effects.",
    "conditions_and_limits": "Edges optimize response prediction from correct-or-incorrect logs. The heuristic analyses scale selected graph weights and node representations to simulate availability and representativeness; they do not observe invoked heuristics or test independent acquisition, transfer, or durability.",
    "criticism": "This is close prior art for learning relations among knowledge components, but the prediction-optimized adjacency matrix is not a Bayesian posterior over a learner's changing heuristic-to-heuristic associations and should not be read as a validated cognitive structure.",
    "which_workflow": "Q12 pivotal prior art: learned knowledge-component adjacency and simulated heuristic modulation in graph-based knowledge tracing.",
    "claimable_sentence": "A peer-reviewed knowledge-tracing model already learns weighted relations among knowledge components, but not person-specific heuristic-use edges from available-versus-enacted observations."
  },
  {
    "id": "sung_et_al_2025_beyond_frequency",
    "tier": "Tier 1",
    "full_citation": "Sung, H., Bernacki, M. L., Greene, J. A., Yu, L., & Plumley, R. D. (2025). Beyond frequency: Using epistemic network analysis and multimodal traces to understand temporal dynamics of self-regulated learning. Journal of Science Education and Technology, 34, 1110–1127. https://doi.org/10.1007/s10956-024-10164-2",
    "link": "https://doi.org/10.1007/s10956-024-10164-2",
    "peer_reviewed": true,
    "population": "N=48 students classified into progressing and mastery groups during a science learning task; coded think-aloud processes were combined with digital learning traces.",
    "design": "Cross-sectional comparison of individual Epistemic Network Analysis models constructed from multimodal self-regulated-learning traces; network positions and mean edge patterns were compared between performance groups.",
    "finding_verbatim": "“statistically significant differences in the patterns of multimodal traces during SRL processing between the progressing and mastery groups”",
    "effect_size": "ENA network-position difference: Mprogressing=.06, Mmastery=-.08, t(45.99)=2.66, p<.05, Cohen's d=.75. The mastery group showed stronger monitoring-to-domain-specific-strategy connections; progressing learners showed stronger monitoring-to-domain-general-strategy connections.",
    "conditions_and_limits": "Group labels were based on task performance. Edges represent relative co-occurrence of coded verbalizations and trace events; the comparison does not test within-person change or retention/transfer.",
    "criticism": "The study supports criterion-related differences in strategy-association networks, not the claim that Bayesian movement of an individual's edges constitutes learning.",
    "which_workflow": "Q12 adjacent evidence: strategy-use association networks can discriminate performance groups.",
    "claimable_sentence": "Networks of co-occurring self-regulated-learning strategies differed between mastery and progressing groups, but this group association does not validate edge change as individual learning."
  },
  {
    "id": "Anderson1982AcquisitionCognitiveSkill",
    "tier": "Tier 3",
    "full_citation": "Anderson, J. R. (1982). Acquisition of cognitive skill. Psychological Review, 89(4), 369–406.",
    "link": "https://doi.org/10.1037/0033-295X.89.4.369",
    "peer_reviewed": true,
    "population": "No single participant sample; the article integrates cognitive experiments and ACT production-system models.",
    "design": "Theoretical synthesis and computational account of cognitive skill acquisition.",
    "finding_verbatim": "Knowledge compilation consists of proceduralization and composition.",
    "effect_size": "Not applicable; the article reports a theory and model demonstrations rather than one intervention effect.",
    "conditions_and_limits": "Declarative knowledge is modeled propositionally and procedural knowledge as productions; later changes include generalization, discrimination, and strengthening.",
    "criticism": "A representational theory does not establish that an observed behavioral graph is a literal mental structure or a valid measure of learning.",
    "which_workflow": "HS-WP-2026-09—production-rule prior art for structured heuristic knowledge.",
    "claimable_sentence": "Production-system theories have represented procedural expertise as structured rules for decades; the novelty question concerns observation and validation, not the existence of structured procedural models."
  },
  {
    "id": "ChiFeltovichGlaser1981PhysicsCategorization",
    "tier": "Tier 2",
    "full_citation": "Chi, M. T. H., Feltovich, P. J., & Glaser, R. (1981). Categorization and representation of physics problems by experts and novices. Cognitive Science, 5(2), 121–152.",
    "link": "https://doi.org/10.1207/s15516709cog0502_2",
    "peer_reviewed": true,
    "population": "Physics experts and novices across four studies of problem categorization and representation.",
    "design": "Experimental comparison of sorting, representation, and problem-solving organization.",
    "finding_verbatim": "Experts emphasized major physics principles; novices emphasized objects and surface features.",
    "effect_size": "No standardized aggregate effect is reported in the article's abstract; do not invent one.",
    "conditions_and_limits": "Classic mechanics problems and relatively small expert–novice samples.",
    "criticism": "The study supports structured expert–novice differences but does not validate a passively observed personal heuristic graph.",
    "which_workflow": "HS-WP-2026-09—expert-knowledge organization prior art.",
    "claimable_sentence": "Experts and novices can organize the same domain around different relations, but this finding does not identify a unique graph format or make use traces a learning measure."
  },
  {
    "id": "GoldsmithJohnsonActon1991StructuralKnowledge",
    "tier": "Tier 2",
    "full_citation": "Goldsmith, T. E., Johnson, P. J., & Acton, W. H. (1991). Assessing structural knowledge. Journal of Educational Psychology, 83(1), 88–96.",
    "link": "https://doi.org/10.1037/0022-0663.83.1.88",
    "peer_reviewed": true,
    "population": "40 students in a semester course; relatedness judgments involved 30 domain concepts.",
    "design": "Correlational structural assessment: pairwise concept-relatedness ratings were transformed into Pathfinder networks and compared with an instructor network.",
    "finding_verbatim": "Student–instructor network similarity was strongly related to classroom examination performance.",
    "effect_size": "Network similarity with the instructor correlated r = .74, p < .01, with semester examination performance.",
    "conditions_and_limits": "Explicit pairwise judgments, one course and referent, and a global similarity criterion.",
    "criticism": "The network is elicited through ratings rather than passively inferred from use, and the design is correlational.",
    "which_workflow": "HS-WP-2026-09—direct graph-based learner-knowledge prior art.",
    "claimable_sentence": "Pathfinder research already estimated learner-specific knowledge networks and related their similarity to an expert referent to course performance."
  },
  {
    "id": "TrumpowerShararaGoldsmith2010Specificity",
    "tier": "Tier 2",
    "full_citation": "Trumpower, D. L., Sharara, H., & Goldsmith, T. E. (2010). Specificity of structural assessment of knowledge. Journal of Technology, Learning, and Assessment, 8(5), 1–32.",
    "link": "https://ejournals.bc.edu/index.php/jtla/article/view/1624",
    "peer_reviewed": true,
    "population": "35 undergraduates who learned a computer programming language.",
    "design": "Correlational study comparing particular Pathfinder link subsets with performance on matching and nonmatching programming problems.",
    "finding_verbatim": "Two link subsets differentially predicted performance on two types of problems.",
    "effect_size": "Pointer links: matching U = 57.50, p = .032; nonmatching U = 84.00, p = .234. Go-To links: matching U = 85.00, p = .035; nonmatching U = 134.50, p = .892.",
    "conditions_and_limits": "Only 8 of 35 learners had the full Pointer subset and 12 had the Go-To subset; edges arose from pairwise ratings and expert-defined subsets.",
    "criticism": "The authors call the evidence correlational. It is edge-specific criterion evidence, not Bayesian estimation from observed co-use and not proof of learning.",
    "which_workflow": "HS-WP-2026-09—closest direct prior art for learner-specific knowledge edges.",
    "claimable_sentence": "Individual structural-knowledge edges have already shown content-specific relations to matching problem performance, although they were elicited explicitly and tested correlationally."
  },
  {
    "id": "WoutersVanDerSpekVanOostendorp2011PathfinderGame",
    "tier": "Tier 2",
    "full_citation": "Wouters, P. J. M., van der Spek, E. D., & van Oostendorp, H. (2011). Measuring learning in serious games: A case study with structural assessment. Educational Technology Research and Development, 59(6), 741–763.",
    "link": "https://doi.org/10.1007/s11423-010-9183-0",
    "peer_reviewed": true,
    "population": "Novice and advanced learners who completed a triage serious game and pre/post assessment.",
    "design": "Pre/post case study combining a traditional knowledge test with Pathfinder network similarity and coherence indices.",
    "finding_verbatim": "Structural assessment produced information beyond traditional verbal assessment.",
    "effect_size": "Traditional scores improved for advanced learners, Z = −2.121, p < .05, and novices, Z = −2.230, p < .05. Expert similarity rose for novices, Z = −1.96, p < .05, but not advanced learners, Z = −.78, p > .05.",
    "conditions_and_limits": "Immediate pre/post design in one serious game; structural coherence and similarity did not move together.",
    "criticism": "Results depend on expert agreement, choice of referent and concepts, and the burdensome pairwise-rating task; different structural indices can diverge.",
    "which_workflow": "HS-WP-2026-09—structural change versus learning-score boundary.",
    "claimable_sentence": "A Pathfinder graph can change after instruction, but graph indices and conventional learning scores may diverge, so graph movement is not a unitary learning quantity."
  },
  {
    "id": "RuizPrimoShavelson1996ConceptMapAssessment",
    "tier": "Tier 3",
    "full_citation": "Ruiz-Primo, M. A., & Shavelson, R. J. (1996). Problems and issues in the use of concept maps in science assessment. Journal of Research in Science Teaching, 33(6), 569–600.",
    "link": "https://doi.org/10.1002/(SICI)1098-2736(199608)33:6%3C569::AID-TEA1%3E3.0.CO;2-M",
    "peer_reviewed": true,
    "population": "Review of concept-mapping assessment studies; no new participant sample.",
    "design": "Methodological review of elicitation tasks, response formats, scoring, reliability, and validity.",
    "finding_verbatim": "Little attention had been paid to reliability and validity across mapping variations.",
    "effect_size": "Not applicable; the review did not synthesize a single pooled effect.",
    "conditions_and_limits": "Science concept maps varied extensively in task, response format, and scoring.",
    "criticism": "A graph label does not define one measure; technique-specific validity evidence is required.",
    "which_workflow": "HS-WP-2026-09—graph-assessment validity warning.",
    "claimable_sentence": "Concept maps are established knowledge-structure assessments, but their interpretation depends on the elicitation and scoring procedure rather than graph form alone."
  },
  {
    "id": "FanEtAl2022TraceValidity",
    "tier": "Tier 2",
    "full_citation": "Fan, Y., van der Graaf, J., Lim, L., Raković, M., Singh, S., Kilgour, J., Moore, J., Molenaar, I., Bannert, M., & Gašević, D. (2022). Towards investigating the validity of measurement of self-regulated learning based on trace data. Metacognition and Learning, 17, 949–987.",
    "link": "https://doi.org/10.1007/s11409-022-09291-1",
    "peer_reviewed": true,
    "population": "44 learners who studied artificial intelligence in education for 45 minutes in a technology-enhanced environment.",
    "design": "Laboratory validation study aligning trace-derived SRL processes with think-aloud data and reporting match rate, sensitivity, specificity, and coverage.",
    "finding_verbatim": "The validation approach improved alignment between trace and think-aloud process evidence.",
    "effect_size": "Match rate improved from 38.97% to 54.24% in training and from 34.54% to 55.09% in testing.",
    "conditions_and_limits": "One short laboratory task; think-aloud was treated as a reference rather than an infallible ground truth.",
    "criticism": "Even after improvement, only about half of reference events matched; trace meaning must be validated, not inferred from event labels.",
    "which_workflow": "HS-WP-2026-09—direct evidence standard for invocation coding.",
    "claimable_sentence": "Trace-derived strategy processes can be validated, but a carefully aligned protocol still left substantial mismatch."
  },
  {
    "id": "BernackiEtAl2024MultimodalTraceValidation",
    "tier": "Tier 2",
    "full_citation": "Bernacki, M. L., Yu, L., Kuhlmann, S. L., Plumley, R. D., Greene, J. A., Duke, R. F., Freed, R., Hollander-Blackmon, C., & Hogan, K. A. (2025). Using multimodal learning analytics to validate digital traces of self-regulated learning in a laboratory study and predict performance in undergraduate courses. Journal of Educational Psychology, 117(2), 176–205. (Published online October 3, 2024.)",
    "link": "https://doi.org/10.1037/edu0000890",
    "peer_reviewed": true,
    "population": "Study 1: 48 biology learners; Study 2: 307 course learners; next-semester replication: 432 learners.",
    "design": "Temporal alignment of digital events with verbal SRL evidence, followed by field prediction and replication.",
    "finding_verbatim": "Ten digital events co-occurred at least 70% with verbalized macroprocesses.",
    "effect_size": "Ten events met the ≥70% co-occurrence rule. Exact regression coefficients are not available in the abstract and are not entered here.",
    "conditions_and_limits": "Some events mapped to multiple SRL processes; low laboratory variance yielded nonsignificant lesson-quiz prediction, while field traces predicted several course outcomes.",
    "criticism": "Many-to-many trace mappings and setting-specific variance limit a simple one-event/one-construct interpretation.",
    "which_workflow": "HS-WP-2026-09—multistudy trace-validation precedent.",
    "claimable_sentence": "Digital events can support SRL interpretations after alignment and replication, but event meanings can be many-to-many and laboratory validity need not yield criterion prediction."
  },
  {
    "id": "FoxEricssonBest2011VerbalReports",
    "tier": "Tier 1",
    "full_citation": "Fox, M. C., Ericsson, K. A., & Best, R. (2011). Do procedures for verbal reporting of thinking have to be reactive? A meta-analysis and recommendations for best reporting methods. Psychological Bulletin, 137(2), 316–344.",
    "link": "https://doi.org/10.1037/a0021663",
    "peer_reviewed": true,
    "population": "94 studies with nearly 3,500 participants.",
    "design": "Meta-analysis comparing concurrent verbal-report procedures with matched silent controls.",
    "finding_verbatim": "Classic think-aloud was nonreactive for accuracy; directed explanation was reactive.",
    "effect_size": "Think-aloud accuracy effect r = −.03, statistically indistinguishable from zero; all verbal-report procedures increased completion time.",
    "conditions_and_limits": "Results distinguish nondirected concurrent verbalization from prompts to explain or describe particular information.",
    "criticism": "A nondirected protocol can preserve accuracy on average while remaining incomplete; directed elicitation changes the process it seeks to observe.",
    "which_workflow": "HS-WP-2026-09—reactivity and observability boundary.",
    "claimable_sentence": "Concurrent think-aloud can preserve accuracy under narrow conditions, whereas prompted explanation changes performance and all verbalization slows work."
  },
  {
    "id": "LemaireSiegler1995FourAspects",
    "tier": "Tier 2",
    "full_citation": "Lemaire, P., & Siegler, R. S. (1995). Four aspects of strategic change: Contributions to children's learning of multiplication. Journal of Experimental Psychology: General, 124(1), 83–97.",
    "link": "https://doi.org/10.1037/0096-3445.124.1.83",
    "peer_reviewed": true,
    "population": "French second graders assessed three times during the year in which they learned single-digit multiplication.",
    "design": "Longitudinal trial-level assessment of speed, accuracy, and strategy use.",
    "finding_verbatim": "Improvement reflected repertoire, frequency, execution, and adaptive-selection changes.",
    "effect_size": "No single standardized effect captures the four longitudinal processes; none is entered.",
    "conditions_and_limits": "Single-digit multiplication in children; multiple strategies persisted at all three measurement points.",
    "criticism": "A single frequency or association score collapses distinct learning processes.",
    "which_workflow": "HS-WP-2026-09—strategy repertoire versus use and selection prior art.",
    "claimable_sentence": "Strategy learning comprises at least changes in repertoire, frequency, execution efficiency, and adaptive selection; use frequency alone is incomplete."
  },
  {
    "id": "SieglerLemaire1997ChoiceNoChoice",
    "tier": "Tier 2",
    "full_citation": "Siegler, R. S., & Lemaire, P. (1997). Older and younger adults' strategy choices in multiplication: Testing predictions of ASCM using the choice/no-choice method. Journal of Experimental Psychology: General, 126(1), 71–92.",
    "link": "https://doi.org/10.1037/0096-3445.126.1.71",
    "peer_reviewed": true,
    "population": "Adults around ages 20 and 70 across three multidigit-multiplication experiments.",
    "design": "Choice/no-choice experiments comparing free selection with required use of mental calculation, calculator, and paper-and-pencil strategies.",
    "finding_verbatim": "Speed and accuracy were the strongest predictors of strategy frequency.",
    "effect_size": "The abstract reports directional model tests but no single standardized effect; none is entered.",
    "conditions_and_limits": "Strategies were discrete and observable in arithmetic; the method is harder where strategies blend or cannot be verified.",
    "criticism": "Observed choice is endogenous: without no-choice trials, performance estimates for unchosen strategies are selection-biased.",
    "which_workflow": "HS-WP-2026-09—direct analogue for available versus enacted options.",
    "claimable_sentence": "Use reveals selection, not the potential performance or cognitive availability of an unchosen strategy."
  },
  {
    "id": "SieglerStern1998UnconsciousDiscovery",
    "tier": "Tier 2",
    "full_citation": "Siegler, R. S., & Stern, E. (1998). Conscious and unconscious strategy discoveries: A microgenetic analysis. Journal of Experimental Psychology: General, 127(4), 377–397.",
    "link": "https://doi.org/10.1037/0096-3445.127.4.377",
    "peer_reviewed": true,
    "population": "Second-grade children solving arithmetic problems through computation or insight.",
    "design": "Microgenetic trial-by-trial study using implicit performance and explicit-report measures of strategy discovery.",
    "finding_verbatim": "Strategies can be discovered without conscious awareness.",
    "effect_size": "Almost 90% showed the insight implicitly before explicit report; 80% reported it within five relevant trials of first implicit use.",
    "conditions_and_limits": "One arithmetic insight task with dense trial-level observation.",
    "criticism": "Explicit non-invocation can understate available procedural knowledge.",
    "which_workflow": "HS-WP-2026-09—absence-of-trace boundary.",
    "claimable_sentence": "Failure to verbalize or explicitly invoke a strategy is not evidence that the strategy is absent."
  },
  {
    "id": "MillerSeierBarronProbert1994UtilizationDeficiency",
    "tier": "Tier 2",
    "full_citation": "Miller, P. H., Seier, W. L., Barron, K. L., & Probert, J. S. (1994). What causes a memory strategy utilization deficiency? Cognitive Development, 9(1), 77–101.",
    "link": "https://doi.org/10.1016/0885-2014(94)90020-5",
    "peer_reviewed": true,
    "population": "Three child studies; Study 1 included 83 kindergarten and first-grade children.",
    "design": "Experiments on spontaneous selective-attention strategy production, contextual knowledge, labeling, and recall benefit.",
    "finding_verbatim": "Children can produce an appropriate strategy yet receive little recall benefit.",
    "effect_size": "Study 1 n = 83; selectivity predicted recall with a familiar story context but not without it. A standardized effect is not reported in the abstract.",
    "conditions_and_limits": "Young children, selective attention, labeling, and location recall.",
    "criticism": "Production and benefit are separable; neither spontaneous use nor frequency proves effective learning.",
    "which_workflow": "HS-WP-2026-09—use-versus-benefit discriminant evidence.",
    "claimable_sentence": "A learner can enact an appropriate strategy without gaining the performance benefit that would justify calling the enactment learned competence."
  },
  {
    "id": "RollEtAl2011HelpTutor",
    "tier": "Tier 1",
    "full_citation": "Roll, I., Aleven, V., McLaren, B. M., & Koedinger, K. R. (2011). Improving students' help-seeking skills using metacognitive feedback in an intelligent tutoring system. Learning and Instruction, 21(2), 267–280.",
    "link": "https://doi.org/10.1016/j.learninstruc.2010.07.004",
    "peer_reviewed": true,
    "population": "Study 1: 58 students over six 45-minute sessions; Study 2: 67 students over four months.",
    "design": "Two classroom studies of a Help Tutor embedded in a Geometry Cognitive Tutor, with independent strategy measures, supported behavior, unsupported transfer, and domain outcomes.",
    "finding_verbatim": "Metacognitive feedback improved help-seeking behavior and some later unsupported use.",
    "effect_size": "Study 1 online error versus independent paper hint use: partial r = −.50, p < .01; dilemma scores 77% versus 59%, d = .83. Faulty-hint requests d = 1.51; bottom-out behavior d = 1.07. Study 2 supported effects reached d = 1.47.",
    "conditions_and_limits": "Geometry tutor context; some behavioral transfer appeared only after extended support. Domain learning did not differ in months when measured.",
    "criticism": "Large, validated trace changes coexisted with null domain-learning effects, directly separating use from learning.",
    "which_workflow": "HS-WP-2026-09—strongest behavioral-validity versus learning counterexample.",
    "claimable_sentence": "A trace can validly measure improved strategy use while providing no evidence of improved domain learning."
  },
  {
    "id": "Winne2020TraceConstructValidity",
    "tier": "Tier 3",
    "full_citation": "Winne, P. H. (2020). Construct and consequential validity for learning analytics based on trace data. Computers in Human Behavior, 112, 106457.",
    "link": "https://doi.org/10.1016/j.chb.2020.106457",
    "peer_reviewed": true,
    "population": "No participant sample.",
    "design": "Validity analysis focused on theory, reliability, generalizability, and consequences of trace-based learning analytics.",
    "finding_verbatim": "Raw trace data are biased by the theory that recommends observing them.",
    "effect_size": "Not applicable.",
    "conditions_and_limits": "Addresses trace analytics generally rather than heuristic graphs or human–AI work specifically.",
    "criticism": "Trace collection and interpretation are part of the measure; learner agency and changing tasks complicate reliability and generalization.",
    "which_workflow": "HS-WP-2026-09—central trace-validity framework.",
    "claimable_sentence": "Trace data are theory-selected observations whose reliability and meaning must generalize across relevant facets of learners, tasks, and environments."
  },
  {
    "id": "SoderstromBjork2015LearningPerformance",
    "tier": "Tier 3",
    "full_citation": "Soderstrom, N. C., & Bjork, R. A. (2015). Learning versus performance: An integrative review. Perspectives on Psychological Science, 10(2), 176–199.",
    "link": "https://doi.org/10.1177/1745691615569000",
    "peer_reviewed": true,
    "population": "Review of laboratory and applied learning research; no new sample.",
    "design": "Narrative integrative review separating acquisition-phase performance from durable learning.",
    "finding_verbatim": "Performance during acquisition can be a poor index of long-term learning.",
    "effect_size": "Not applicable; do not aggregate the reviewed studies into a new effect.",
    "conditions_and_limits": "Defines learning through relatively permanent changes supporting retention and transfer.",
    "criticism": "The review supplies a construct boundary, not an empirical validation of any heuristic graph.",
    "which_workflow": "HS-WP-2026-09—operative definition of learning.",
    "claimable_sentence": "Observed use during assisted work is performance; learning requires evidence that change persists and transfers."
  },
  {
    "id": "RoedigerKarpicke2006TestingMemory",
    "tier": "Tier 1",
    "full_citation": "Roediger, H. L., III, & Karpicke, J. D. (2006). Test-enhanced learning: Taking memory tests improves long-term retention. Psychological Science, 17(3), 249–255.",
    "link": "https://doi.org/10.1111/j.1467-9280.2006.01693.x",
    "peer_reviewed": true,
    "population": "Undergraduates learning prose passages in two experiments.",
    "design": "Randomized comparison of repeated study and repeated testing with immediate and delayed recall.",
    "finding_verbatim": "Testing produced better long-term retention despite an immediate study advantage.",
    "effect_size": "Experiment 2: repeated study led at 5 minutes, d = 1.22; repeated testing led at one week, 61% versus 40%, d = 1.26. Proportional forgetting was 10% versus 52%.",
    "conditions_and_limits": "Prose recall with a one-week delay; not human–AI work or heuristic graphs.",
    "criticism": "Immediate performance can reverse at delay, so contemporaneous edge movement cannot define durable learning.",
    "which_workflow": "HS-WP-2026-09—performance/retention dissociation.",
    "claimable_sentence": "The sign of an acquisition-phase advantage can reverse on a delayed test, making immediate use an unsafe proxy for learning."
  },
  {
    "id": "SalomonPerkinsGloberson1991WithOfTechnology",
    "tier": "Tier 3",
    "full_citation": "Salomon, G., Perkins, D. N., & Globerson, T. (1991). Partners in cognition: Extending human intelligence with intelligent technologies. Educational Researcher, 20(3), 2–9.",
    "link": "https://doi.org/10.3102/0013189X020003002",
    "peer_reviewed": true,
    "population": "No new participant sample.",
    "design": "Conceptual distinction between joint performance with technology and cognitive effects that remain after technology use.",
    "finding_verbatim": "Effects with technology differ from effects of technology.",
    "effect_size": "Not applicable.",
    "conditions_and_limits": "General intelligent technologies, predating contemporary generative AI.",
    "criticism": "The distinction is conceptual; empirical tests still require unassisted outcomes.",
    "which_workflow": "HS-WP-2026-09—human–technology learning boundary.",
    "claimable_sentence": "A network learned from assisted work initially represents effects with the system; effects of the experience require performance after assistance is removed."
  },
  {
    "id": "BastaniEtAl2025GenAIWithoutGuardrails",
    "tier": "Tier 1",
    "full_citation": "Bastani, H., Bastani, O., Sungu, A., Ge, H., Kabakcı, Ö., & Mariman, R. (2025). Generative AI without guardrails can harm learning: Evidence from high school mathematics. Proceedings of the National Academy of Sciences, 122(26), e2422633122.",
    "link": "https://doi.org/10.1073/pnas.2422633122",
    "peer_reviewed": true,
    "population": "Nearly 1,000 students in grades 9–11 at one private high school in Turkey.",
    "design": "Preregistered classroom-cluster randomized trial across four sessions comparing control, generic GPT-4, and a teacher-grounded GPT tutor; assisted practice was followed by immediate unassisted exams.",
    "finding_verbatim": "Generative AI without guardrails can harm learning.",
    "effect_size": "Assisted practice: generic GPT about +48% and guarded tutor about +127% relative to control. Unassisted exam: generic GPT about −17%; guarded tutor estimate −.004, nonsignificant.",
    "conditions_and_limits": "One private school, mathematics, four sessions, and an immediate rather than delayed unassisted test.",
    "criticism": "Large assisted-performance gains did not imply learning; the guarded tutor removed the penalty but did not yield a positive unassisted effect.",
    "which_workflow": "HS-WP-2026-09—direct human–AI performance/learning boundary.",
    "claimable_sentence": "Successful AI-assisted practice can coexist with worse or unchanged unassisted performance."
  },
  {
    "id": "BorsboomMellenberghVanHeerden2004Validity",
    "tier": "Tier 3",
    "full_citation": "Borsboom, D., Mellenbergh, G. J., & van Heerden, J. (2004). The concept of validity. Psychological Review, 111(4), 1061–1071.",
    "link": "https://doi.org/10.1037/0033-295X.111.4.1061",
    "peer_reviewed": true,
    "population": "No participant sample.",
    "design": "Conceptual analysis of measurement validity.",
    "finding_verbatim": "Validity requires the attribute to exist and causally affect measurement outcomes.",
    "effect_size": "Not applicable.",
    "conditions_and_limits": "General measurement theory rather than educational trace data specifically.",
    "criticism": "Correlation with outcomes alone does not establish that learning caused edge movement.",
    "which_workflow": "HS-WP-2026-09—construct-validity standard.",
    "claimable_sentence": "A learning interpretation requires evidence for the process by which durable knowledge change, rather than irrelevant trace factors, produces the graph change."
  },
  {
    "id": "BullKay2016SMILI",
    "tier": "Tier 3",
    "full_citation": "Bull, S., & Kay, J. (2016). SMILI☺: A framework for interfaces to learning data in open learner models, learning analytics and related fields. International Journal of Artificial Intelligence in Education, 26(1), 293–331.",
    "link": "https://doi.org/10.1007/s40593-015-0090-8",
    "peer_reviewed": true,
    "population": "No new participant sample.",
    "design": "Revised framework for describing, comparing, and critiquing open learner-model interfaces.",
    "finding_verbatim": "The framework defines diverse ways to invite learners into their models.",
    "effect_size": "Not applicable.",
    "conditions_and_limits": "Inspectable, editable, negotiable, and related learner-model interfaces; not a test of a heuristic-use graph.",
    "criticism": "Opening a learner model can promote reflection and contestability, but learner endorsement is not criterion validity.",
    "which_workflow": "HS-WP-2026-09—open-learner-model prior art.",
    "claimable_sentence": "Inspectable and negotiable learner models are established prior art for exposing and contesting a system's representation of a learner."
  },
  {
    "id": "HooshyarEtAl2020OpenLearnerModelsReview",
    "tier": "Tier 3",
    "full_citation": "Hooshyar, D., Pedaste, M., Saks, K., Leijen, Ä., Bardone, E., & Wang, M. (2020). Open learner models in supporting self-regulated learning in higher education: A systematic literature review. Computers & Education, 154, 103878.",
    "link": "https://doi.org/10.1016/j.compedu.2020.103878",
    "peer_reviewed": true,
    "population": "64 higher-education open-learner-model articles spanning approximately 30 years.",
    "design": "Systematic literature review organized by self-regulated-learning dimensions and phases.",
    "finding_verbatim": "OLMs mostly supported cognition, then metacognition and motivation; emotion was rarely supported.",
    "effect_size": "No pooled effect size was reported; do not aggregate the studies into a new number.",
    "conditions_and_limits": "Heterogeneous systems and outcomes; simple inspectable models were more common than advanced editable or negotiable forms.",
    "criticism": "The review demonstrates a mature design space, not criterion validity for any one model or evidence that visualization measures learning.",
    "which_workflow": "HS-WP-2026-09—open learner model landscape.",
    "claimable_sentence": "Open learner models are established, but evidence is heterogeneous and concentrated on appraisal and performance rather than the full learning process."
  },
  {
    "id": "VisserVanDerTogtVanRiel2016TheoryAction",
    "tier": "Tier 3",
    "full_citation": "Visser, M., & van der Togt, K. (2016). Learning in public sector organizations: A theory of action approach. Public Organization Review, 16, 235–249.",
    "link": "https://doi.org/10.1007/s11115-015-0303-5",
    "peer_reviewed": true,
    "population": "One Dutch municipal policy case used to illustrate an Argyris–Schön theory-of-action framework.",
    "design": "Conceptual synthesis and qualitative case application.",
    "finding_verbatim": "Espoused theory and theory-in-use may or may not coincide.",
    "effect_size": "Not applicable.",
    "conditions_and_limits": "Organizational and public-policy learning, not individual educational trace measurement.",
    "criticism": "The espoused/enacted distinction is strong conceptual prior art, but reconstruction from action remains interpretive and learning additionally involves error correction.",
    "which_workflow": "HS-WP-2026-09—espoused versus enacted analogue.",
    "claimable_sentence": "The difference between stated and enacted heuristics has a canonical precursor in espoused theory versus theory-in-use; action still does not equal learning."
  },
  {
    "id": "KizilcecLee2022AlgorithmicFairnessEducation",
    "tier": "Tier 3",
    "full_citation": "Kizilcec, R. F., & Lee, H. (2022). Algorithmic fairness in education. In W. Holmes & K. Porayska-Pomsta (Eds.), The ethics of artificial intelligence in education (pp. 174–202). Routledge.",
    "link": "https://doi.org/10.4324/9780429329067-10",
    "peer_reviewed": false,
    "population": "No new participant sample.",
    "design": "Conceptual review of statistical, similarity-based, and causal fairness in educational algorithms.",
    "finding_verbatim": "Bias can enter through measurement, model learning, and action.",
    "effect_size": "Not applicable.",
    "conditions_and_limits": "Broad educational predictions and decisions rather than heuristic invocation specifically.",
    "criticism": "Fairness definitions can conflict; group parity does not by itself establish construct equivalence.",
    "which_workflow": "HS-WP-2026-09—trace fairness and differential validity.",
    "claimable_sentence": "Fairness auditing must examine how traces are measured, modeled, and acted upon, not only aggregate predictive accuracy."
  },
  {
    "id": "ShaGasevicChen2023DebiasingEducation",
    "tier": "Tier 2",
    "full_citation": "Sha, L., Gašević, D., & Chen, G. (2023). Lessons from debiasing data for fair and accurate predictive modeling in education. Expert Systems with Applications, 228, 120323.",
    "link": "https://doi.org/10.1016/j.eswa.2023.120323",
    "peer_reviewed": true,
    "population": "Educational datasets used across five frequently performed prediction tasks, audited by sex and first-language background.",
    "design": "Empirical evaluation of distribution and hardness bias, balancing methods, prediction accuracy, and fairness.",
    "finding_verbatim": "Fairness improved substantially with a small accuracy sacrifice.",
    "effect_size": "Reported fairness improvement up to 66% with less than 1% loss of prediction accuracy.",
    "conditions_and_limits": "Five predictive tasks and the fairness metrics used by the authors; not a validation of trace constructs or learning claims.",
    "criticism": "Aggregate accuracy can conceal subgroup disparities, but debiasing a prediction does not establish measurement invariance.",
    "which_workflow": "HS-WP-2026-09—empirical fairness-audit precedent.",
    "claimable_sentence": "Educational models can display material subgroup unfairness despite high aggregate accuracy, making groupwise trace validation necessary."
  },
  {
    "id": "PelanekRihakPapousek2016DataCollectionBias",
    "tier": "Tier 2",
    "full_citation": "Pelánek, R., Řihák, J., & Papoušek, J. (2016). Impact of data collection on interpretation and evaluation of student models. In Proceedings of the Sixth International Conference on Learning Analytics & Knowledge (pp. 40–47). ACM.",
    "link": "https://doi.org/10.1145/2883851.2883868",
    "peer_reviewed": true,
    "population": "Simulated and real student-modeling datasets.",
    "design": "Methodological demonstrations of mastery attrition bias and adaptive item-selection effects.",
    "finding_verbatim": "The way data are collected can alter model evaluation and interpretation.",
    "effect_size": "No single effect size summarizes the simulations and demonstrations; none is entered.",
    "conditions_and_limits": "Historical adaptive-learning data and student-model evaluation.",
    "criticism": "An adaptive observation policy makes exposure endogenous, so edge movement may reflect what was shown or sampled rather than learner change.",
    "which_workflow": "HS-WP-2026-09—observation-policy and opportunity-denominator warning.",
    "claimable_sentence": "When data collection adapts to a learner, model estimates and retrospective validation can be biased by the observation process itself."
  },
  {
    "id": "BarnettCeci2002TransferTaxonomy",
    "tier": "Tier 3",
    "full_citation": "Barnett, S. M., & Ceci, S. J. (2002). When and where do we apply what we learn? A taxonomy for far transfer. Psychological Bulletin, 128(4), 612–637.",
    "link": "https://doi.org/10.1037/0033-2909.128.4.612",
    "peer_reviewed": true,
    "population": "Review of transfer research; no new participant sample.",
    "design": "Theoretical taxonomy separating content and context dimensions of transfer.",
    "finding_verbatim": "Transfer varies across multiple content and context dimensions.",
    "effect_size": "Not applicable; the article argues against collapsing heterogeneous transfer conditions into one label.",
    "conditions_and_limits": "Knowledge domain, task, temporal, physical, functional, social, and modality dimensions can differ.",
    "criticism": "A generic claim of transfer is underspecified; near tool-bound generalization and far unassisted transfer are different tests.",
    "which_workflow": "HS-WP-2026-09—transfer criterion specification.",
    "claimable_sentence": "A learning validation must state which knowledge and context dimensions changed rather than report undifferentiated transfer."
  },
  {
    "id": "HadwinEtAl2007ExaminingTraceData",
    "tier": "Tier 2",
    "full_citation": "Hadwin, A. F., Nesbit, J. C., Jamieson-Noel, D., Code, J., & Winne, P. H. (2007). Examining trace data to explore self-regulated learning. Metacognition and Learning, 2, 107–124.",
    "link": "https://doi.org/10.1007/s11409-007-9016-7",
    "peer_reviewed": true,
    "population": "Eight learners observed during two studying episodes.",
    "design": "Exploratory case-based construction of trace profiles for frequency, pattern, timing, and content of study tactics.",
    "finding_verbatim": "Trace profiles made the temporal and contextual character of studying visible.",
    "effect_size": "Not applicable; n = 8 exploratory profiles and no population effect estimate.",
    "conditions_and_limits": "Tool-mediated study tactics in two episodes; the authors did not intend statistical generalization.",
    "criticism": "Trace events are artifacts of both learner regulation and the environment's affordances; a tiny exploratory sample cannot validate a general construct.",
    "which_workflow": "HS-WP-2026-09—early trace-based SRL prior art.",
    "claimable_sentence": "Learning traces can characterize how tactics unfold within an environment, but early evidence was explicitly exploratory rather than a validation of latent learning."
  },
  {
    "id": "BannertReimannSonnenberg2014ProcessMining",
    "tier": "Tier 2",
    "full_citation": "Bannert, M., Reimann, P., & Sonnenberg, C. (2014). Process mining techniques for analysing patterns and strategies in students' self-regulated learning. Metacognition and Learning, 9(2), 161–185.",
    "link": "https://doi.org/10.1007/s11409-013-9107-6",
    "peer_reviewed": true,
    "population": "38 learners studying in a hypermedia environment.",
    "design": "Process-mining analysis of coded think-aloud events to identify temporal sequences of self-regulatory activity.",
    "finding_verbatim": "Process mining represented specific sequences of regulatory activities.",
    "effect_size": "No single standardized effect summarizes the methodological demonstration; none is entered.",
    "conditions_and_limits": "Small hypermedia sample, coded think-aloud reference, and task-specific process definitions.",
    "criticism": "A mined sequence is a pattern of observed regulation, not proof of retained knowledge or a learned association.",
    "which_workflow": "HS-WP-2026-09—temporal trace-pattern prior art.",
    "claimable_sentence": "Process mining already identifies temporal strategy patterns from learning-process traces, but pattern change requires separate learning validation."
  },
  {
    "id": "SaintEtAl2020TraceSRL",
    "tier": "Tier 2",
    "full_citation": "Saint, J., Whitelock-Wainwright, A., Gašević, D., & Pardo, A. (2020). Trace-SRL: A framework for analysis of microlevel processes of self-regulated learning from trace data. IEEE Transactions on Learning Technologies, 13(4), 861–877.",
    "link": "https://doi.org/10.1109/TLT.2020.3027496",
    "peer_reviewed": true,
    "population": "Nearly 300 computer-engineering undergraduates in a flipped course.",
    "design": "Theory-driven transformation of raw trace events into SRL processes, strategic clustering, and process mining.",
    "finding_verbatim": "Theory-driven process mining revealed detail unavailable from frequency measures alone.",
    "effect_size": "The abstract reports group-pattern differences rather than one standardized effect; none is entered.",
    "conditions_and_limits": "One flipped engineering course; success groups and trace patterns are associational.",
    "criticism": "The framework provides an interpretable representation of activity but does not establish that inferred process patterns are durable learning.",
    "which_workflow": "HS-WP-2026-09—trace transformation and process-mining prior art.",
    "claimable_sentence": "The trace-SRL literature already transforms platform events into theory-defined strategy processes and temporal patterns."
  },
  {
    "id": "Kane2013ValidationArgument",
    "tier": "Tier 3",
    "full_citation": "Kane, M. T. (2013). Validating the interpretations and uses of test scores. Journal of Educational Measurement, 50(1), 1–73.",
    "link": "https://doi.org/10.1111/jedm.12000",
    "peer_reviewed": true,
    "population": "No participant sample.",
    "design": "Argument-based framework for evaluating score interpretations and uses.",
    "finding_verbatim": "More ambitious claims require more support than less ambitious claims.",
    "effect_size": "Not applicable.",
    "conditions_and_limits": "General educational measurement rather than trace graphs specifically.",
    "criticism": "Validation attaches to a stated interpretation and use, not to a score or algorithm in the abstract.",
    "which_workflow": "HS-WP-2026-09—evidence-chain framework.",
    "claimable_sentence": "The validity argument must separately support scoring, generalization, extrapolation, and use; evidence for observed co-use does not automatically support a learning interpretation."
  },
  {
    "id": "MuthenEtAl1995OpportunityToLearn",
    "tier": "Tier 2",
    "full_citation": "Muthén, B., Huang, L.-C., Jo, B., Khoo, S.-T., Nelson Goff, G., Novak, J. R., & Shih, J. C. (1995). Opportunity-to-learn effects on achievement: Analytical aspects. Educational Evaluation and Policy Analysis, 17(3), 371–403.",
    "link": "https://doi.org/10.3102/01623737017003371",
    "peer_reviewed": true,
    "population": "Mathematics data from the National Assessment of Educational Progress and the National Education Longitudinal Study.",
    "design": "Methods article with empirical illustrations for estimating opportunity-to-learn effects while accounting for prior performance and background factors.",
    "finding_verbatim": "Opportunity-to-learn effects require analysis alongside prior performance and background factors.",
    "effect_size": "No single effect summarizes the multiple methods and illustrations; none is entered.",
    "conditions_and_limits": "Large-scale mathematics assessment and content exposure, not AI-presented heuristic candidates.",
    "criticism": "Opportunity is an input or exposure construct, not achieved learning; candidate supply therefore belongs in the observation model.",
    "which_workflow": "HS-WP-2026-09—opportunity-versus-achievement analogue.",
    "claimable_sentence": "The available candidate set is analogous to opportunity to learn and must be modeled separately from enactment and achievement."
  },
  {
    "id": "LarkinEtAl1980ExpertNoviceRepresentation",
    "tier": "Tier 2",
    "full_citation": "Larkin, J., McDermott, J., Simon, D. P., & Simon, H. A. (1980). Expert and novice performance in solving physics problems. Science, 208(4450), 1335–1342.",
    "link": "https://doi.org/10.1126/science.208.4450.1335",
    "peer_reviewed": true,
    "population": "Experts and novices solving physics problems, together with computational models of their problem-solving processes.",
    "design": "Comparative cognitive-process study and production-system modeling of expert and novice problem solving.",
    "finding_verbatim": "Experts and novices used differently organized knowledge and search procedures.",
    "effect_size": "No single standardized effect is reported; the evidence is process- and model-based.",
    "conditions_and_limits": "Classical physics problems and small expert–novice samples typical of cognitive-process research.",
    "criticism": "The study supports structured procedural differences but does not validate passive trace-derived edges or a learning interpretation.",
    "which_workflow": "HS-WP-2026-09—expert procedural-representation prior art.",
    "claimable_sentence": "Expertise has long been modeled as differently organized procedural knowledge, so structured heuristic representations are not themselves novel."
  },
  {
    "id": "SR-WP09-AFM-2006",
    "tier": "Tier 2",
    "full_citation": "Cen, H., Koedinger, K. R., & Junker, B. (2006). Learning Factors Analysis—A general method for cognitive model evaluation and improvement. In Intelligent Tutoring Systems (LNCS 4053, pp. 164–175). Springer. https://doi.org/10.1007/11774303_17",
    "link": "https://doi.org/10.1007/11774303_17",
    "peer_reviewed": true,
    "population": "Cognitive Tutor Geometry Area 1996–97 log data; later dataset documentation identifies 59 students and 5,388 step-level observations.",
    "design": "Method-development study. Learning Factors Analysis used the Additive Factor Model (AFM), expert-specified difficulty factors, and combinatorial search to compare alternative knowledge-component decompositions against observed step correctness.",
    "finding_verbatim": "For the best BIC model, its BIC is reduced by 37, and AIC by 62.",
    "effect_size": "For the best BIC-selected alternative, BIC decreased by 37, AIC by 62, and mean absolute deviation by .012 relative to the original cognitive model.",
    "conditions_and_limits": "Single geometry-tutor domain; fit to logged correctness. Standard AFM represents log-odds of a correct response using a learner intercept, KC easiness, practice opportunities, and KC-specific practice slopes. It does not separately model successes and failures, forgetting, transfer, or delayed retention.",
    "criticism": "The practice coefficient is named a learning rate, but the criterion is fit to performance observations. Better AIC, BIC, or MAD does not independently validate acquisition, transfer, or durability.",
    "which_workflow": "WP-09 — P4/Q11: AFM comparison and the performance-versus-learning boundary",
    "claimable_sentence": "AFM estimates practice-related changes in response probability at the level of tagged knowledge components; its original evidence was improved fit to tutor performance, not independent evidence of durable learning."
  },
  {
    "id": "SR-WP09-PFA-2009",
    "tier": "Tier 2",
    "full_citation": "Pavlik, P. I., Jr., Cen, H., & Koedinger, K. R. (2009). Performance Factors Analysis—A new alternative to Knowledge Tracing. In Artificial Intelligence in Education (pp. 531–538). IOS Press. https://doi.org/10.3233/978-1-60750-028-5-531",
    "link": "https://doi.org/10.3233/978-1-60750-028-5-531",
    "peer_reviewed": true,
    "population": "Four tutor-log datasets: physics (40,930 observations), geometry (44,780), algebra (46,570), and fractions (approximately 101,000).",
    "design": "Method comparison using seven-fold cross-validation. PFA predicted correctness from KC difficulty plus separate counts/slopes for each learner’s prior successes and failures on the KCs required by an item, and was compared with AFM variants and standard Knowledge Tracing.",
    "finding_verbatim": "These comparisons demonstrate that the PFA model is a new alternative that may be useful for detecting and reacting to student learning in a tutor.",
    "effect_size": "Cross-validated MAD for PFA versus AFM was .346 versus .340 (physics), .297 versus .295 (geometry), .204 versus .203 (algebra), and .207 versus .208 (fractions).",
    "conditions_and_limits": "PFA was approximately as predictive as AFM across these four datasets and slightly better only on the reported fractions MAD. Standard PFA is a logistic performance model, not a latent-state model, and contains no elapsed-time, transfer, or retention term.",
    "criticism": "Calling the success/failure slopes learning parameters does not change the observed target: future correctness. Counts can absorb prior proficiency, item selection, hints, task difficulty, and opportunity differences.",
    "which_workflow": "WP-09 — P4/Q11: PFA comparison and observable performance factors",
    "claimable_sentence": "PFA directly predicts performance from prior successes and failures on tagged KCs; it is prior art for longitudinal performance accumulation, not evidence that a changing parameter is durable learning."
  },
  {
    "id": "SR-WP09-DKT-2015",
    "tier": "Tier 2",
    "full_citation": "Piech, C., Bassen, J., Huang, J., Ganguli, S., Sahami, M., Guibas, L. J., & Sohl-Dickstein, J. (2015). Deep Knowledge Tracing. Advances in Neural Information Processing Systems, 28, 505–513.",
    "link": "https://proceedings.neurips.cc/paper_files/paper/2015/file/bac9162b47c56fc8a4d2a519803d51b3-Paper.pdf",
    "peer_reviewed": true,
    "population": "Simulated-5: 4,000 simulated students and 200,000 answers; Khan Math: 47,495 learners and 1.435 million answers; ASSISTments: 15,931 learners and 526,000 answers.",
    "design": "Recurrent-neural-network sequence model trained to predict correctness on a learner’s next interaction; five-fold cross-validation on the two real datasets and held-out simulated learners.",
    "finding_verbatim": "Knowledge tracing is the task of modelling student knowledge over time so that we can accurately predict how students will perform on future interactions.",
    "effect_size": "AUC: Khan Math DKT .85 versus standard BKT .68 and marginal .63; ASSISTments DKT .86 versus BKT .67, prior-best BKT variant .69, and marginal .62; Simulated-5 DKT .75 versus BKT .54.",
    "conditions_and_limits": "The real-data outcome was next-response correctness. The curriculum experiment simulated students using the trained DKT itself and was not a human learning trial. The hidden vector was not independently matched to delayed retention or transfer.",
    "criticism": "Higher next-answer AUC establishes predictive discrimination under those datasets, not that the hidden state is a valid representation of acquisition, transfer, or durability.",
    "which_workflow": "WP-09 — P4/Q11: DKT prior art and predictive-target distinction",
    "claimable_sentence": "DKT’s landmark gains were gains in next-response AUC; the study did not independently validate its hidden state as durable learning."
  },
  {
    "id": "SR-WP09-DKT-PROBLEMS-2018",
    "tier": "Tier 2",
    "full_citation": "Yeung, C.-K., & Yeung, D.-Y. (2018). Addressing two problems in deep knowledge tracing via prediction-consistent regularization. Proceedings of the Fifth Annual ACM Conference on Learning at Scale, Article 5. https://doi.org/10.1145/3231644.3231647",
    "link": "https://doi.org/10.1145/3231644.3231647",
    "peer_reviewed": true,
    "population": "Knowledge-tracing benchmark interaction data used to reproduce and diagnose DKT behavior; the paper reports extensive experiments rather than an intervention measuring learner outcomes.",
    "design": "Diagnostic replication and model modification. The authors tested input reconstruction and temporal consistency, then added reconstruction and waviness regularizers to DKT’s loss.",
    "finding_verbatim": "The first problem is that the model fails to reconstruct the observed input.",
    "effect_size": "The regularized model alleviated reconstruction and prediction-waviness failures without degrading the original next-response prediction task; the abstract reports no single aggregate effect size.",
    "conditions_and_limits": "The critique concerns internal prediction behavior. It does not itself validate a learning construct against post-tests, transfer tasks, or delayed measures.",
    "criticism": "A hidden state labeled mastery can move opposite to the learner’s observed performance and fluctuate implausibly while retaining predictive accuracy; predictive success and interpretable learning state are separable.",
    "which_workflow": "WP-09 — P4/Q11: DKT construct and interpretability failure modes",
    "claimable_sentence": "DKT can preserve next-response performance while exhibiting internally implausible state changes, so its latent vector cannot be treated as learning without separate validation."
  },
  {
    "id": "SR-WP09-IRT-BKT-2018",
    "tier": "Tier 1",
    "full_citation": "Deonovic, B., Yudelson, M., Bolsinova, M., Attali, M., & Maris, G. (2018). Learning meets assessment: On the relation between Item Response Theory and Bayesian Knowledge Tracing. Behaviormetrika, 45(2), 457–474. https://doi.org/10.1007/s41237-018-0070-z",
    "link": "https://link.springer.com/article/10.1007/s41237-018-0070-z",
    "peer_reviewed": true,
    "population": "Theoretical comparison; no human sample.",
    "design": "Mathematical analysis relating stationary distributions in BKT to an IRT model and comparing the study designs and educational interpretations for which the traditions were developed.",
    "finding_verbatim": "Bayesian knowledge tracing is designed to analyze longitudinal data while item response theory is built for cross-sectional data.",
    "effect_size": "Not applicable; theoretical derivation rather than an empirical effect estimate.",
    "conditions_and_limits": "The statement characterizes standard IRT and BKT; longitudinal and dynamic IRT extensions exist. The authors identify the role of education as a missing component in both traditions.",
    "criticism": "IRT ability parameters are not learning merely because data are educational. Standard cross-sectional IRT has no acquisition transition, and dynamic extensions still require external construct validation.",
    "which_workflow": "WP-09 — P4/Q11: IRT alternative tradition",
    "claimable_sentence": "Standard IRT is principally a cross-sectional measurement model; adding time variation can describe change but does not by itself identify that change as learning."
  },
  {
    "id": "SR-WP09-KT-COMPARISON-2020",
    "tier": "Tier 1",
    "full_citation": "Gervet, T., Koedinger, K., Schneider, J., & Mitchell, T. (2020). When is deep learning the best approach to knowledge tracing? Journal of Educational Data Mining, 12(3), 31–54. https://doi.org/10.5281/zenodo.4143614",
    "link": "https://theophilegervet.github.io/assets/pdf/gervet2020deep.pdf",
    "peer_reviewed": true,
    "population": "Nine real-world intelligent-tutoring datasets spanning different dataset sizes, sequence lengths, and temporal structures.",
    "design": "Student-level held-out comparison of Markov-process, logistic-regression, and deep-learning learner-performance models, with feature ablations and calibration analyses.",
    "finding_verbatim": "In this paper, we investigated which approach to knowledge tracing makes the most accurate predictions, in what conditions.",
    "effect_size": "No cross-study aggregate was created. Logistic regression led on moderate datasets or very long learner histories; DKT led on large datasets or where precise temporal order mattered; BKT lagged. On ASSISTments 2009, leading models were systematically miscalibrated at probability extremes.",
    "conditions_and_limits": "Cross-validation generalized to new learners, not unseen items. The authors note that this design disadvantaged IRT because learner ability was not refit. The criterion throughout was learner performance prediction.",
    "criticism": "The study is strong evidence about comparative prediction under specific data regimes, not about whether any model’s internal state corresponds to acquisition, transfer, or durability.",
    "which_workflow": "WP-09 — P4/Q11: current comparative evidence on KT families",
    "claimable_sentence": "No knowledge-tracing family dominates across data regimes, and calibration can fail even when ranking performance is acceptable; these are predictive-model results rather than learning validation."
  },
  {
    "id": "SR-WP09-SPACING-MODEL-2005",
    "tier": "Tier 2",
    "full_citation": "Pavlik, P. I., Jr., & Anderson, J. R. (2005). Practice and forgetting effects on vocabulary memory: An activation-based model of the spacing effect. Cognitive Science, 29(4), 559–586. https://doi.org/10.1207/s15516709cog0000_14",
    "link": "https://onlinelibrary.wiley.com/doi/10.1207/s15516709cog0000_14",
    "peer_reviewed": true,
    "population": "40 native English-speaking participants learning 104 Japanese–English paired associates; 20 were tested after one day and 20 after seven days.",
    "design": "Within-subject vocabulary experiment varying repetitions, intervening trials (2, 14, or 98), and retention interval, plus comparison of activation-based memory models.",
    "finding_verbatim": "The relative benefit of spacing increased with increased practice and with longer retention intervals.",
    "effect_size": "The paper reports significant repetition-by-spacing and repetition-by-retention-interval interactions; the abstract does not provide standardized effect sizes. Wider spacing reduced forgetting at both one- and seven-day tests.",
    "conditions_and_limits": "Paired-associate vocabulary under controlled retrieval practice. The model uses power-law decay whose rate depends on activation at practice; results do not imply one universal decay rate for complex judgment or associations.",
    "criticism": "Elapsed time cannot be interpreted alone: frequency, spacing, retrieval success, and retention horizon jointly determine later performance.",
    "which_workflow": "WP-09 — P4/Q13: forgetting curves and spacing",
    "claimable_sentence": "Forgetting evidence supports time-sensitive models, but also shows that recency is not a sufficient statistic: the effect of time depends on practice history and the retention horizon."
  },
  {
    "id": "SR-WP09-SPACING-META-2006",
    "tier": "Tier 1",
    "full_citation": "Cepeda, N. J., Pashler, H., Vul, E., Wixted, J. T., & Rohrer, D. (2006). Distributed practice in verbal recall tasks: A review and quantitative synthesis. Psychological Bulletin, 132(3), 354–380. https://doi.org/10.1037/0033-2909.132.3.354",
    "link": "https://pubmed.ncbi.nlm.nih.gov/16719566/",
    "peer_reviewed": true,
    "population": "839 assessments of distributed practice from 317 experiments reported in 184 articles.",
    "design": "Meta-analysis of massed versus spaced study, lag, expanding intervals, and the interaction between interstudy interval (ISI) and retention interval.",
    "finding_verbatim": "the ISI producing maximal retention increased as retention interval increased.",
    "effect_size": "The published synthesis included 839 assessments; its central quantitative result was an ISI-by-retention-interval interaction rather than one context-free optimal interval.",
    "conditions_and_limits": "Verbal-recall tasks dominate. The optimal spacing interval changes with the intended retention interval; direct transfer to complex heuristic associations requires testing.",
    "criticism": "A fixed time-decay rule risks contradicting the evidence that the value of an interval depends on when retention will be tested and how practice was distributed.",
    "which_workflow": "WP-09 — P4/Q13: defensible temporal assumptions",
    "claimable_sentence": "Spacing and forgetting are jointly time-dependent; there is no literature basis for treating the same elapsed interval as equally diagnostic in every retention context."
  },
  {
    "id": "SR-WP09-DAS3H-2019",
    "tier": "Tier 2",
    "full_citation": "Choffin, B., Popineau, F., Bourda, Y., & Vie, J.-J. (2019). DAS3H: Modeling student learning and forgetting for optimally scheduling distributed practice of skills. Proceedings of the 12th International Conference on Educational Data Mining, 29–38.",
    "link": "https://files.eric.ed.gov/fulltext/ED599174.pdf",
    "peer_reviewed": true,
    "population": "ASSISTments 2012–13: 24,750 users and 2,692,889 interactions; Bridge to Algebra 2006–07: 1,135 users and 1,817,427 interactions; Algebra 2005–06: 569 users and 607,000 interactions.",
    "design": "Five-fold predictive comparison of a logistic model containing temporal windows and item–skill relations against DASH, IRT, PFA, AFM, and multidimensional variants; ablations compared skill-specific and shared temporal parameters.",
    "finding_verbatim": "We observe that using time window features consistently boosts the AUC of the model.",
    "effect_size": "With no latent embedding, Algebra AUC was .826 ± .003 for DAS3H versus .775 ± .005 DASH, .771 ± .007 IRT, .744 ± .004 PFA, and .707 ± .005 AFM. Skill-specific versus shared temporal parameters improved AUC by .03–.04 across the three datasets.",
    "conditions_and_limits": "Standard deviations are across five folds. Outcomes were next-response correctness, not delayed independent retention tests. The temporal windows estimate predictive recency patterns and do not alone identify psychological forgetting.",
    "criticism": "The skill-specific gain is evidence against a uniform temporal effect, but predictive improvement can reflect curriculum timing, item sequencing, and changing opportunity as well as forgetting.",
    "which_workflow": "WP-09 — P4/Q13: time-aware learner models",
    "claimable_sentence": "Time-aware, skill-specific features can improve next-answer prediction, but those gains do not establish that a decaying trace is durable learning or forgetting."
  },
  {
    "id": "SR-WP09-FAIR-SLICING-2019",
    "tier": "Tier 2",
    "full_citation": "Gardner, J., Brooks, C., & Baker, R. (2019). Evaluating the fairness of predictive student models through slicing analysis. Proceedings of the 9th International Learning Analytics & Knowledge Conference, 225–234. https://doi.org/10.1145/3303772.3303791",
    "link": "https://doi.org/10.1145/3303772.3303791",
    "peer_reviewed": true,
    "population": "More than four million learners in 44 MOOCs; five replicated dropout-prediction models.",
    "design": "Large-scale replication and gender-based slicing analysis using Absolute Between-ROC Area (ABROCA) to compare subgroup discrimination across algorithms, feature sets, courses, and curricular areas.",
    "finding_verbatim": "there is not evidence of a strict tradeoff between performance and fairness.",
    "effect_size": "Mean ABROCA differed by model (Kruskal–Wallis p = 8.106×10^-5) and feature set (p = 1.99×10^-5). AUC and ABROCA correlated r = .029, p = .6692. The classification tree’s mean ABROCA was nearly 25% below LSTM/logistic regression while its average AUC was highest.",
    "conditions_and_limits": "Outcome was MOOC dropout, not learning. Gender was binary and inferred from names; other protected and intersectional groups were not tested. Associations with course context were exploratory, not causal.",
    "criticism": "Overall accuracy can hide subgroup error. Trace volume, opportunity, language style, course context, and protected-group slices should be examined separately, with uncertainty reported for each slice.",
    "which_workflow": "WP-09 — Q16: fairness audit standard",
    "claimable_sentence": "Fairness cannot be inferred from aggregate AUC: model and feature choices produced different subgroup performance even when overall predictive performance did not trade off with the fairness metric."
  },
  {
    "id": "SR-WP09-BIAS-REVIEW-2022",
    "tier": "Tier 1",
    "full_citation": "Baker, R. S., & Hawn, A. (2022). Algorithmic bias in education. International Journal of Artificial Intelligence in Education, 32(4), 1052–1092. https://doi.org/10.1007/s40593-021-00285-9",
    "link": "https://learninganalytics.upenn.edu/ryanbaker/AlgorithmicBiasInEducation_rsb3.7.pdf",
    "peer_reviewed": true,
    "population": "Review of empirical educational-algorithm bias research spanning race/ethnicity, gender, nationality, socioeconomic status, disability, and military-connected status.",
    "design": "Education-specific narrative review mapping bias to stages and actors in the machine-learning pipeline, summarizing evidence by affected group, and proposing a framework from unknown to known bias and from fairness to equity.",
    "finding_verbatim": "Acknowledging the gaps in what has been studied, we propose a framework for moving from unknown bias to known bias and from fairness to equity.",
    "effect_size": "No pooled effect size was reported. Evidence coverage was uneven, with race/ethnicity, gender, and nationality studied more than socioeconomic status, disability, or military-connected status.",
    "conditions_and_limits": "A review of a heterogeneous, incomplete literature; it cannot establish that every educational algorithm is biased in the same direction or magnitude.",
    "criticism": "A fairness audit limited to the most available demographic attribute can leave other or intersectional harms unknown. Bias can enter during measurement, labeling, modeling, deployment, and feedback—not only through protected attributes supplied to a model.",
    "which_workflow": "WP-09 — Q16: education-specific algorithmic bias evidence",
    "claimable_sentence": "Educational algorithms have documented group-specific biases, but coverage is uneven; a learning representation therefore needs pre-specified subgroup and intersectional audits across its full evidence pipeline."
  },
  {
    "id": "SR-WP09-FAIR-NLP-2019",
    "tier": "Tier 2",
    "full_citation": "Loukina, A., Madnani, N., & Zechner, K. (2019). The many dimensions of algorithmic fairness in educational applications. Proceedings of the Fourteenth Workshop on Innovative Use of NLP for Building Educational Applications, 1–10. https://doi.org/10.18653/v1/W19-4401",
    "link": "https://aclanthology.org/W19-4401/",
    "peer_reviewed": true,
    "population": "26,710 spoken responses from 4,452 English-proficiency test takers, balanced across six self-reported first-language groups; 6,768 responses from 1,128 test takers formed the test set.",
    "design": "Simulated and operational automated-scoring analyses compared overall accuracy, overall score differences, and conditional score differences across native-language groups.",
    "finding_verbatim": "We illustrate that total fairness may not be achievable and that different definitions of fairness may require different solutions.",
    "effect_size": "Even a near-perfect simulated model (human-score correlation r = .97) failed overall score-difference equality when score distributions differed. In a controlled equal-distribution subset, non-metadata models’ absolute standardized mean differences fell below .02.",
    "conditions_and_limits": "Automated English speech scoring, not learner modeling. Human scores were treated as gold standard despite possible rater bias; most responses had one human rating and test-set human–human agreement was r = .66.",
    "criticism": "Fairness is construct- and use-dependent. Equal aggregate accuracy, equal score distributions, and equal conditional errors answer different questions and can conflict.",
    "which_workflow": "WP-09 — Q16: fairness definitions for dialogue-derived evidence",
    "claimable_sentence": "Dialogue-derived learner evidence should be audited for conditional error across language groups, not only aggregate accuracy, because score distributions can make fairness metrics disagree."
  },
  {
    "id": "KleinEtAl1989",
    "tier": "Tier 3",
    "full_citation": "Klein, G. A., Calderwood, R., & MacGregor, D. (1989). Critical decision method for eliciting knowledge. IEEE Transactions on Systems, Man, and Cybernetics, 19(3), 462–472.",
    "link": "https://doi.org/10.1109/21.31053",
    "peer_reviewed": true,
    "population": "Experienced practitioners drawn from applied projects involving urban and wildland fire command, tank command, structural and design engineering, paramedicine, and computer programming; the methodological article does not report one pooled sample size.",
    "design": "Foundational methodological and case-based article describing the Critical Decision Method: multiple-pass retrospective interviews about personally experienced nonroutine incidents, beginning with chronology and decision points and then probing cues, goals, expectancies, options, discriminations, typicality, and counterfactuals.",
    "finding_verbatim": "The method is a variant of the critical incident technique extended to include probes",
    "effect_size": "No comparative effect size, confidence interval, or controlled validation statistic was reported.",
    "conditions_and_limits": "The method depends on a concrete consequential incident personally experienced by the interviewee, reconstruction of the event timeline before cognitive probes, and interviewers who avoid substituting generic doctrine for case-specific recall. It intentionally samples difficult or unusual cases rather than estimating the frequency of routine behavior.",
    "criticism": "The paper establishes a replicable elicitation procedure and illustrates useful outputs, but it does not test recall against contemporaneous process data, quantify completeness or inter-interviewer reliability, or show that knowledge from atypical incidents generalizes to ordinary cases.",
    "which_workflow": "Expert-judgment capture—foundational lineage for eliciting decision points, cues, expectancies, options, and counterfactuals from critical incidents.",
    "claimable_sentence": "The Critical Decision Method provides a structured way to elicit decisions and cues from a specific experienced incident, but its foundational paper did not establish that retrospective accounts are complete, independently veridical, or representative of routine work."
  },
  {
    "id": "MilitelloHutton1998",
    "tier": "Tier 2",
    "full_citation": "Militello, L. G., & Hutton, R. J. B. (1998). Applied cognitive task analysis (ACTA): A practitioner's toolkit for understanding cognitive task demands. Ergonomics, 41(11), 1618–1641.",
    "link": "https://doi.org/10.1080/001401398186108",
    "peer_reviewed": true,
    "population": "Twenty-three graduate psychology students without prior domain, CTA, or instructional-design experience: 12 worked in firefighting and 11 in electronic warfare. Domain experts had at least 10 years of fire-command experience or at least six years of electronic-warfare experience.",
    "design": "Matched comparative evaluation of ACTA versus unstructured interviewing after common introductory training; ACTA participants received six additional hours of method training, conducted expert interviews, and produced cognitive-demands tables and training materials that independent subject-matter experts and cognitive psychologists rated.",
    "finding_verbatim": "ACTA techniques were found to be easy to use, flexible, and to provide clear output.",
    "effect_size": "For ACTA outputs in firefighting and electronic warfare, respectively, 92% and 94% of items were rated cognitive, 95% and 90% contained expert-only content, and 73% and 87% were relevant. Proposed manual modifications were rated accurate for 89% and 65% of items, and learning objectives for 92% and 54%. No standardized between-group effect sizes or confidence intervals were reported.",
    "conditions_and_limits": "ACTA combined a Task Diagram, Knowledge Audit, Simulation Interview, and Cognitive Demands Table. Its evaluation concerned usability, cognitive content, relevance, and SME-rated training utility after substantial method training; it did not compare elicited accounts with contemporaneous cognitive process data.",
    "criticism": "The groups were small and variable; the comparison group received some cognitively oriented training, reducing separation. Item-level overlap could not be scored without unacceptable inference, and some electronic-warfare SME agreement was poor, forcing reliance on one expert's ratings. The evaluation supports practicability more strongly than reliability or construct validity.",
    "which_workflow": "Expert-judgment capture—streamlined CTA workflow producing difficult elements, cues, strategies, and common errors.",
    "claimable_sentence": "ACTA proved usable for trained novice analysts and produced SME-rated cognitive and relevant content, but its evaluation did not establish that the elicited model was complete, reproducible, or independently veridical."
  },
  {
    "id": "SminkEtAl2012",
    "tier": "Tier 3",
    "full_citation": "Smink, D. S., Peyre, S. E., Soybel, D. I., Tavakkolizadeh, A., Vernon, A. H., & Anastakis, D. J. (2012). Utilization of a cognitive task analysis for laparoscopic appendectomy to identify differentiated intraoperative teaching objectives. American Journal of Surgery, 203(4), 540–545.",
    "link": "https://doi.org/10.1016/j.amjsurg.2011.11.002",
    "peer_reviewed": true,
    "population": "Three local expert surgeons experienced in laparoscopic appendectomy.",
    "design": "Critical-Decision-Method-based CTA interviews; transcripts were converted into individual cognitive-demands tables, member-checked, merged, and reviewed in a consensus meeting. The authors calculated expert agreement on operative steps and decision points and compared teaching priorities for junior and senior residents.",
    "finding_verbatim": "Of the 27 decision points, only 5 (19%) were identified by all 3 surgeon experts.",
    "effect_size": "All three experts identified 18 of 24 operative steps (75%) but only 5 of 27 decision points (19%). Individual coverage was 96%, 79%, and 83% for operative steps and 78%, 59%, and 48% for decision points. Experts selected nine operative steps and six decision points for junior residents versus four operative steps and 13 decision points for senior residents, p<.01. No standardized effect sizes or confidence intervals were reported.",
    "conditions_and_limits": "The procedure was narrowly bounded, each table was returned to its expert for completeness review, and disagreements were handled through a consensus meeting. The resulting master table reflects aggregation and consensus rather than a demonstrated population-complete model.",
    "criticism": "The very low three-expert overlap for decision points shows that decision capture is more expert-sample-dependent than action-step capture. With three experts, one procedure, and no independent performance criterion, consensus can conceal legitimate strategy variation as well as repair omissions.",
    "which_workflow": "Expert-judgment capture—direct reliability evidence for action steps versus decision points.",
    "claimable_sentence": "Three surgeons converged on most operative steps but on only 19% of the combined decision points, indicating that decision-point capture is highly sensitive to which experts are interviewed."
  },
  {
    "id": "TverskyKahneman1974HeuristicsBiases",
    "tier": "Tier 1",
    "full_citation": "Tversky, A., & Kahneman, D. (1974). Judgment under uncertainty: Heuristics and biases. Science, 185(4157), 1124–1131. https://doi.org/10.1126/science.185.4157.1124",
    "link": "https://doi.org/10.1126/science.185.4157.1124",
    "peer_reviewed": true,
    "population": "Multiple laboratory judgment demonstrations using students, professionals, and other adult samples; the article does not present one pooled population.",
    "design": "Programmatic synthesis of experimental demonstrations of representativeness, availability, and anchoring in judgments under uncertainty.",
    "finding_verbatim": "people rely on a limited number of heuristic principles",
    "effect_size": "No single pooled effect size was reported; the paper presents separate demonstrations under different tasks and samples.",
    "conditions_and_limits": "The research emphasizes systematic errors in probabilistic judgment, often in laboratory or hypothetical tasks. It does not define expert instructional heuristics or a learner representation.",
    "criticism": "The biases tradition is one meaning of heuristic, not the only one. Treating every heuristic as an error would misstate ecological and naturalistic decision research.",
    "which_workflow": "P1 — foundational heuristics-and-biases lineage and a boundary on terminology.",
    "claimable_sentence": "The heuristics-and-biases tradition treats heuristics as economical judgment procedures that can generate systematic errors under specified conditions."
  },
  {
    "id": "GigerenzerGaissmaier2011HeuristicDecisionMaking",
    "tier": "Tier 1",
    "full_citation": "Gigerenzer, G., & Gaissmaier, W. (2011). Heuristic decision making. Annual Review of Psychology, 62, 451–482. https://doi.org/10.1146/annurev-psych-120709-145346",
    "link": "https://doi.org/10.1146/annurev-psych-120709-145346",
    "peer_reviewed": true,
    "population": "No single participant sample; integrative review of theoretical, experimental, and applied heuristic research.",
    "design": "Annual Review synthesis of fast-and-frugal heuristics, ecological rationality, strategy selection, and environment–strategy fit.",
    "finding_verbatim": "A heuristic is a strategy that ignores part of the information",
    "effect_size": "Not applicable; the article is an integrative review and does not pool studies into one effect.",
    "conditions_and_limits": "The review covers many domains and heuristic forms. It does not validate a decision-point/cue/error template, human–AI trace, or learner graph.",
    "criticism": "A heuristic's success depends on its fit to an environment, so frequency of use cannot be interpreted without task ecology and boundary conditions.",
    "which_workflow": "P1 and P3 — ecological-rationality lineage for contextual expert heuristics and principled non-use.",
    "claimable_sentence": "Heuristics can be adaptive strategies whose validity depends on environmental structure, not merely cognitive shortcuts or errors."
  },
  {
    "id": "Sfard1998TwoMetaphorsLearning",
    "tier": "Tier 1",
    "full_citation": "Sfard, A. (1998). On two metaphors for learning and the dangers of choosing just one. Educational Researcher, 27(2), 4–13. https://doi.org/10.3102/0013189X027002004",
    "link": "https://doi.org/10.3102/0013189X027002004",
    "peer_reviewed": true,
    "population": "No participant sample.",
    "design": "Conceptual analysis comparing acquisition and participation metaphors of learning and their entailments.",
    "finding_verbatim": "two metaphors of learning are identified",
    "effect_size": "Not applicable; this is a theoretical article.",
    "conditions_and_limits": "The article provides conceptual vocabulary rather than an empirical test or measurement rule. It does not concern AI or trace analytics.",
    "criticism": "Participation language cannot make an unreliable trace a learning measure; it changes the target construct and therefore the required validity evidence.",
    "which_workflow": "Central framing — distinguishes learning as acquired capacity from learning as changing participation in practice.",
    "claimable_sentence": "Learning has established acquisition and participation interpretations; a process record can address changing participation without automatically measuring acquired capacity."
  },
  {
    "id": "Greeno1998SituativityKnowingLearning",
    "tier": "Tier 1",
    "full_citation": "Greeno, J. G. (1998). The situativity of knowing, learning, and research. American Psychologist, 53(1), 5–26. https://doi.org/10.1037/0003-066X.53.1.5",
    "link": "https://doi.org/10.1037/0003-066X.53.1.5",
    "peer_reviewed": true,
    "population": "No participant sample.",
    "design": "Theoretical synthesis of situative, cognitive, behaviorist, ecological, and interactional perspectives on knowing and learning.",
    "finding_verbatim": "the focus of analysis is on activity systems",
    "effect_size": "Not applicable; this is a theoretical synthesis.",
    "conditions_and_limits": "The situative perspective supplies a unit of analysis, not an empirical validation of a graph or a claim about durable individual outcomes.",
    "criticism": "A system-level account can obscure person-level differences unless the record remains learner-indexed and later independent criteria test what carries forward.",
    "which_workflow": "Central framing — learner, task, AI, and representational resources as an interacting activity system.",
    "claimable_sentence": "A situative account locates knowing in coordinated activity among people and environmental resources, making mediated reasoning a legitimate object of study."
  },
  {
    "id": "WiseShaffer2015TheoryBigData",
    "tier": "Tier 1",
    "full_citation": "Wise, A. F., & Shaffer, D. W. (2015). Why theory matters more than ever in the age of big data. Journal of Learning Analytics, 2(2), 5–13. https://doi.org/10.18608/jla.2015.22.2",
    "link": "https://doi.org/10.18608/jla.2015.22.2",
    "peer_reviewed": true,
    "population": "No participant sample.",
    "design": "Conceptual analysis of theory's roles in selecting variables, interpreting patterns, identifying confounds and subgroups, and making analytics actionable.",
    "finding_verbatim": "theory plays an ever-more critical role in analysis",
    "effect_size": "Not applicable; this is a conceptual article.",
    "conditions_and_limits": "The article addresses learning analytics broadly rather than heuristic graphs or human–AI dialogue specifically.",
    "criticism": "Theory can guide an interpretation but cannot replace empirical reliability, validity, and fairness evidence.",
    "which_workflow": "Trace interpretation — why more interaction data do not make construct definitions self-evident.",
    "claimable_sentence": "Large trace datasets increase the need for theory about what to record, what alternatives to test, and how to interpret the resulting patterns."
  },
  {
    "id": "Molenaar2022HybridHumanAILearning",
    "tier": "Tier 1",
    "full_citation": "Molenaar, I. (2022). Towards hybrid human–AI learning technologies. European Journal of Education, 57(4), 632–645. https://doi.org/10.1111/ejed.12527",
    "link": "https://doi.org/10.1111/ejed.12527",
    "peer_reviewed": true,
    "population": "No participant sample.",
    "design": "Conceptual framework combining an augmentation perspective, hybrid intelligence, detect–diagnose–act functions, and levels of automation in education.",
    "finding_verbatim": "learning remains an essentially human activity",
    "effect_size": "Not applicable; this is a conceptual framework.",
    "conditions_and_limits": "The article proposes a common language and research agenda; it does not test learning effects or validate a human–AI reasoning trace.",
    "criticism": "Hybrid-system framing clarifies distributed control but does not determine which observed content belongs to the learner or persists beyond the episode.",
    "which_workflow": "P2 — positions educational AI as augmentation and human–AI work as a mediated learning arrangement.",
    "claimable_sentence": "Hybrid human–AI learning frameworks treat control and cognition as distributed during assisted activity while retaining learning as a human educational objective."
  },
  {
    "id": "GajosMamykina2022CognitiveEngagementAI",
    "tier": "Tier 2",
    "full_citation": "Gajos, K. Z., & Mamykina, L. (2022). Do people engage cognitively with AI? Impact of AI assistance on incidental learning. Proceedings of the 27th International Conference on Intelligent User Interfaces, 794–806. https://doi.org/10.1145/3490099.3511138",
    "link": "https://doi.org/10.1145/3490099.3511138",
    "peer_reviewed": true,
    "population": "Online adults making nutritional decisions: Experiment 2 included 268 participants; Experiment 3 included 221 participants and an independent replication included 270.",
    "design": "Three online experiments compared feedback, recommendations, explanations, delayed recommendations, and explanation-only assistance using pretest-to-posttest normalized changes.",
    "finding_verbatim": "it may not be sufficient to provide people with AI-generated recommendations and explanations",
    "effect_size": "In Experiment 3, explanation-only versus minimal feedback yielded r=.31 for immediate benefit and r=.23 for learning. The replication yielded r=.34 and r=.20, respectively. Explanation-only and explanation-feedback learning did not differ significantly in Experiment 3, r=.03.",
    "conditions_and_limits": "Simulated AI, online food-choice judgments, short-term incidental learning, and self-selected online samples; not a course, generative dialogue, delayed test, or heuristic graph.",
    "criticism": "The results show that assistance format changes cognitive engagement and learning, making observed uptake partly an interface effect rather than a stable learner property.",
    "which_workflow": "P2 and P3 — direct evidence that supplied AI reasoning, user decision format, immediate performance, and learning can diverge.",
    "claimable_sentence": "Requiring users to reason from AI-provided explanations without a recommended answer improved both immediate decisions and incidental learning in one experiment and replication."
  },
  {
    "id": "LuEtAl2025HumanAICoLearning",
    "tier": "Tier 1",
    "full_citation": "Lu, J., Yan, Y., Huang, K., Yin, M., & Zhang, F. (2025). Do we learn from each other: Understanding the human–AI co-learning process embedded in human–AI collaboration. Group Decision and Negotiation, 34(2), 235–271. https://doi.org/10.1007/s10726-024-09912-x",
    "link": "https://doi.org/10.1007/s10726-024-09912-x",
    "peer_reviewed": true,
    "population": "Participants in three online behavioral experiments on emotion classification; the article's public abstract does not state the combined sample size.",
    "design": "Two-stage, between-subject experiments separated collaborative training from later independent human performance and model updating. Human and AI baselines were designed to be moderate and comparable at about 50% accuracy.",
    "finding_verbatim": "this expected dual-pathway co-learning process ... does not occur spontaneously",
    "effect_size": "The public abstract reports directional effects but no standardized effect sizes; the study is used here chiefly as a design precedent for separating collaboration from later independent performance.",
    "conditions_and_limits": "Online emotion classification, deliberate moderate baseline accuracy, and subgroup differences by cognitive reflection; not an educational course, expert-heuristic model, or delayed-transfer study.",
    "criticism": "The study supports the need for a two-stage co-learning design but does not validate learning from human–AI dialogue generally or any longitudinal learner graph.",
    "which_workflow": "P2 and learning-layer distinction — collaboration performance versus what the human and AI separately carry forward.",
    "claimable_sentence": "Human–AI co-learning experiments already separate joint performance from subsequent independent performance and show that co-learning depends on feedback, workflow, and participant characteristics."
  },
  {
    "id": "XuEtAl2017PotentialPracticalFlexibility",
    "tier": "Tier 1",
    "full_citation": "Xu, L., Liu, R.-D., Star, J. R., Wang, J., Liu, Y., & Zhen, R. (2017). Measures of potential flexibility and practical flexibility in equation solving. Frontiers in Psychology, 8, 1368. https://doi.org/10.3389/fpsyg.2017.01368",
    "link": "https://doi.org/10.3389/fpsyg.2017.01368",
    "peer_reviewed": true,
    "population": "158 Chinese seventh-grade students from six classrooms; ages 11–14, mean 12.74 years. Students had been taught standard and innovative equation-solving strategies.",
    "design": "A three-phase assessment separated first-attempt strategy use from the ability to generate multiple strategies and identify the innovative strategy.",
    "finding_verbatim": "Potential flexibility and practical flexibility were found to be distinct but related.",
    "effect_size": "Potential-flexibility mean 5.85 versus practical-flexibility mean 1.44 on 12-point scales, t=12.97, p<.001; their correlation was r=.27, p<.01. In 38% of item cases, students were accurate and showed potential flexibility without practical flexibility.",
    "conditions_and_limits": "Cross-sectional, same-session assessment in linear equations. Practical flexibility meant innovative first-attempt use; the study did not test acquisition, delayed retention, or transfer to another domain.",
    "criticism": "The study directly undermines treating non-use as absence of strategy knowledge, but its elicited potential-flexibility measure is not a longitudinal learning outcome.",
    "which_workflow": "P3 — available-versus-enacted distinction and strategy-selection literature.",
    "claimable_sentence": "Knowledge of an applicable strategy and spontaneous use of that strategy are empirically distinct, so non-enactment cannot be scored as nonknowledge without further evidence."
  }
]
