[
{
  "id": "hhitl_evomas_2026_openreview",
  "tier": "Tier 3 \u2014 preprint or unreviewed manuscript",
  "full_citation": "Wei, Y., Huang, Z., Xu, R., Wang, H., & Xing, W. W. (2026 manuscript). EvoMAS: Heuristics in the Loop\u2014Evolving Smarter Agentic Workflows. OpenReview manuscript.",
  "link": "https://openreview.net/pdf?id=0rJUulYnow",
  "peer_reviewed": false,
  "population": "Computational multi-agent workflow benchmarks; no human learner sample.",
  "design": "A meta-controller formulates and updates evolutionary rules and strategies while optimizing multi-agent workflows.",
  "finding_verbatim": "heuristics-in-the-loop learning by formulating and reflectively updating evolutionary rules and strategies",
  "effect_size": "The manuscript reports benchmark and cost-efficiency improvements; no human outcome and no single standardized effect is used in this audit.",
  "conditions_and_limits": "The accessible version states that it was under double-blind review. Its object is machine-agent workflow optimization, not human learning.",
  "criticism": "It occupies the shorter phrase 'Heuristics in the Loop' but not the proposed human-learning configuration.",
  "which_workflow": "HHITL audit \u2014 exact-name collision.",
  "claimable_sentence": "The shorter phrase 'Heuristics in the Loop' was already used in a 2026 OpenReview manuscript for machine-agent workflow optimization."
},
{
  "id": "hhitl_bajestani_et_al_2025_hhitl",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Bajestani, M. S., Mahdi, M. M., Mun, D., & Kim, D. B. (2025). Human and Humanoid-in-the-Loop (HHitL) Ecosystem: An Industry 5.0 Perspective. Machines, 13(6), 510.",
  "link": "https://doi.org/10.3390/machines13060510",
  "peer_reviewed": true,
  "population": "No participant sample.",
  "design": "Conceptual Industry 5.0 framework integrating humans and humanoid robots in cyber-physical manufacturing.",
  "finding_verbatim": "integrates both humans and humanoid robots as collaborative agents",
  "effect_size": "Not applicable; conceptual communication.",
  "conditions_and_limits": "Manufacturing ecosystem proposal; no educational or human-learning evaluation.",
  "criticism": "The paper occupies the HHitL acronym even though its construct is unrelated to heuristics or learning.",
  "which_workflow": "HHITL audit \u2014 acronym collision.",
  "claimable_sentence": "HHitL already abbreviates 'Human and Humanoid-in-the-Loop' in a peer-reviewed 2025 publication."
},
{
  "id": "hhitl_ravichandran_et_al_2024_active_learning",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Ravichandran, S., Sudarsanam, N., Ravindran, B., & Katsikopoulos, K. V. (2024). Active learning with human heuristics: An algorithm robust to labeling bias. Frontiers in Artificial Intelligence, 7, 1491932.",
  "link": "https://doi.org/10.3389/frai.2024.1491932",
  "peer_reviewed": true,
  "population": "Fifteen benchmark datasets from domains in which people provide labels; the human oracle was computationally modeled rather than observed live.",
  "design": "Simulation study crossing two human-heuristic labelers, four active-learning algorithms, and three classifiers, including a proposed inverse-information-density method.",
  "finding_verbatim": "if a heuristic provides labels, the performance of active learning algorithms significantly drops",
  "effect_size": "The proposed algorithm achieved an author-reported overall improvement of 87% over the best other algorithm under the authors' aggregation; some heuristic-label conditions fell below random performance.",
  "conditions_and_limits": "The 87% figure is an algorithm benchmark aggregate, not a human or educational outcome. Human behavior was modeled rather than directly measured, and the study concerns biased labels rather than provenance-bearing expert heuristics offered during reasoning.",
  "criticism": "This is phrase-level and mechanism-adjacent prior art, but its learned object is a classifier and its aggregate improvement must not be generalized to HHITL.",
  "which_workflow": "HHITL naming, human-heuristic bias, and interactive-machine-learning boundary.",
  "claimable_sentence": "Human heuristics have already been modeled inside active-learning loops, principally as potentially biased label-generating processes rather than inspectable expert reasoning aids."
},
{
  "id": "hhitl_chen_cao_2024_hlg",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Chen, B., & Cao, Z. (2024). HLG: Bridging Human Heuristic Knowledge and Deep Reinforcement Learning for Optimal Agent Performance. Proceedings of the 23rd International Conference on Autonomous Agents and Multiagent Systems, 2189\u20132191.",
  "link": "https://ifaamas.csc.liv.ac.uk/Proceedings/aamas2024/pdfs/p2189.pdf",
  "peer_reviewed": true,
  "population": "MiniGrid reinforcement-learning environments; no human participants.",
  "design": "High-level human knowledge was represented as heuristic rules in differentiable decision trees and injected into trainable policy guides.",
  "finding_verbatim": "HLG deployed the heuristic rules from human knowledge in differential decision trees",
  "effect_size": "The authors report at least 25% improvement in training efficiency and exploration capability relative to PPO and PROLONET on MiniGrid tasks.",
  "conditions_and_limits": "Three-page extended abstract and computational evaluation; the machine agent is the learner.",
  "criticism": "It is close component prior art for explicit human heuristic knowledge guiding AI, not for representing human learning.",
  "which_workflow": "HHITL audit \u2014 heuristic knowledge infusion and HITL reinforcement learning.",
  "claimable_sentence": "Peer-reviewed work already injects explicit human heuristic rules into a machine-learning loop to improve agent training."
},
{
  "id": "hhitl_kumar_et_al_2023_design_heuristics",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Kumar, R. S., Srivatsa, S., Baker, E., Silberstein, M., & Selva, D. (2023). Identifying and Leveraging Promising Design Heuristics for Multi-Objective Combinatorial Design Optimization. Journal of Mechanical Design, 145(12), 121702.",
  "link": "https://doi.org/10.1115/1.4063238",
  "peer_reviewed": true,
  "population": "Four computational design-optimization problems; no human learner sample.",
  "design": "The study screened design heuristics and represented them as soft constraints, repair operators, or biased sampling functions.",
  "finding_verbatim": "enforcing only the promising heuristics as repair operators enables finding good designs faster",
  "effect_size": "Promising-heuristic enforcement was benchmarked on four design problems; the abstract reports consistent faster discovery but no single standardized effect.",
  "conditions_and_limits": "Engineering design optimization; heuristics improve automated search rather than model a person's learning.",
  "criticism": "It occupies formal selection and use of expert design heuristics in optimization workflows.",
  "which_workflow": "HHITL audit \u2014 expert heuristics and optimization.",
  "claimable_sentence": "Design-optimization research already formalizes, screens, and selectively enforces expert heuristics in several computational forms."
},
{
  "id": "hhitl_liu_2026_bounded_minds",
  "tier": "Tier 3 \u2014 preprint or unreviewed manuscript",
  "full_citation": "Liu, J. (2026). Bounded Minds, Generative Machines: Envisioning Conversational AI that Works with Human Heuristics and Reduces Bias Risk. arXiv:2601.13376.",
  "link": "https://arxiv.org/abs/2601.13376",
  "peer_reviewed": false,
  "population": "No participant sample.",
  "design": "Conceptual research agenda grounded in bounded rationality for conversational AI design.",
  "finding_verbatim": "conversational AI should be designed to work with human heuristics rather than against them",
  "effect_size": "Not applicable; conceptual article.",
  "conditions_and_limits": "The public version is a preprint. The author's site lists acceptance in Communications of the ACM, but a final publisher record was not located during the audit.",
  "criticism": "This is the closest framing neighbor, but it does not propose or test a longitudinal learner representation.",
  "which_workflow": "HHITL audit \u2014 conversational AI framing.",
  "claimable_sentence": "A 2026 conceptual article already argues that conversational AI should work with human heuristics and evaluate cognitive outcomes beyond factual accuracy."
},
{
  "id": "hhitl_kang_et_al_2026_conversation_heuristics",
  "tier": "Tier 3 \u2014 preprint or unreviewed manuscript",
  "full_citation": "Kang, S., Jeon, S., Eun, J., Lee, K., Song, C., Joo, M., & Lee, J. (2026). Analyzing Human Heuristics and Strategies in Everyday Decision-Making Conversations for Conversational AI Design. arXiv:2605.07789.",
  "link": "https://arxiv.org/abs/2605.07789",
  "peer_reviewed": false,
  "population": "955 Korean food and travel decision conversations containing 15,476 utterances; 3,416 coded heuristic uses in the reported analysis.",
  "design": "Conversation analysis with an LLM-assisted coding pipeline linked heuristic categories to decision resolution.",
  "finding_verbatim": "people prioritize satisficing over optimization",
  "effect_size": "Attribute elimination appeared in 1.6% of coded uses and had a 6.86:1 success ratio; affect appeared in 46.2% with a 2.59:1 ratio. These are author-reported descriptive ratios, not causal effects.",
  "conditions_and_limits": "Preprint; human-human Korean conversations; decision resolution rather than learning; LLM-assisted coding introduces measurement dependence.",
  "criticism": "It is close prior art for inferring heuristics-in-use from conversational traces but not for human-AI co-learning or longitudinal person graphs.",
  "which_workflow": "HHITL audit \u2014 observational heuristic traces.",
  "claimable_sentence": "Naturalistic conversation has already been coded for heuristic use and related descriptively to whether a decision is resolved."
},
{
  "id": "hhitl_ibs_et_al_2024_human_explanations",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Ibs, I., Ott, C., J\u00e4kel, F., & Rothkopf, C. A. (2024). From human explanations to explainable AI: Insights from constrained optimization. Cognitive Systems Research, 88, 101297.",
  "link": "https://doi.org/10.1016/j.cogsys.2024.101297",
  "peer_reviewed": true,
  "population": "Three studies using a constrained-optimization game; the third behavioral dataset included more than 150 participants.",
  "design": "Two qualitative studies elicited representations and heuristics; formal strategy combinations were then matched to decisions in a larger behavioral dataset.",
  "finding_verbatim": "formalize strategies that in combination can act as descriptors for participants' behavior",
  "effect_size": "The abstract reports a large behavioral dataset exceeding 150 participants but no standardized effect size.",
  "conditions_and_limits": "Constrained optimization microworld; the representation supports cognitively adequate AI explanations, not longitudinal learner assessment.",
  "criticism": "This is strong prior art for eliciting, composing, and behaviorally matching human heuristics.",
  "which_workflow": "HHITL audit \u2014 closest heuristic-composition mechanism precedent.",
  "claimable_sentence": "Peer-reviewed work already formalizes combinations of human heuristics and matches those combinations to observed decision behavior."
},
{
  "id": "hhitl_ibs_rothkopf_2025_rationales",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Ibs, I., & Rothkopf, C. A. (2025). Generating Rationales Based on Human Explanations for Constrained Optimization. In Explainable Artificial Intelligence: xAI 2025 (pp. 162\u2013184). Springer.",
  "link": "https://doi.org/10.1007/978-3-032-08317-3_8",
  "peer_reviewed": true,
  "population": "Human-generated solutions from a prior constrained-optimization study plus 100 randomly generated benchmark problems.",
  "design": "A probabilistic context-free grammar derived from human heuristic strategies composes logical programs and matches rationales to decision sequences through probabilistic program induction.",
  "finding_verbatim": "combinations of programs from a probabilistic context-free grammar",
  "effect_size": "The paper reports high dataset description and human-comparable complexity but does not provide a standardized human-learning effect in the accessible abstract.",
  "conditions_and_limits": "The target is rationale generation and behavioral alignment, not an evolving person-specific learning representation.",
  "criticism": "It occupies structured associations and compositions among heuristics inferred against human decision sequences.",
  "which_workflow": "HHITL audit \u2014 closest compositional inference precedent.",
  "claimable_sentence": "Probabilistic program induction has already been used to compose human heuristic strategies into rationales aligned with decision sequences."
},
{
  "id": "hhitl_callaway_et_al_2022_planning_strategies",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Callaway, F., Jain, Y. R., van Opheusden, B., Das, P., Iwama, G., Gul, S., Krueger, P. M., Becker, F., Griffiths, T. L., & Lieder, F. (2022). Leveraging artificial intelligence to improve people's planning strategies. Proceedings of the National Academy of Sciences, 119(12), e2117432119.",
  "link": "https://doi.org/10.1073/pnas.2117432119",
  "peer_reviewed": true,
  "population": "Online adult participants across seven experiments; Experiment 1 recruited 151 and Experiment 6 recruited 417.",
  "design": "An intelligent tutor generated metacognitive feedback on participants' planning operations and tested strategy acquisition, transfer, retention, and feedback components.",
  "finding_verbatim": "The basic idea is to give people feedback on how they reach their decisions.",
  "effect_size": "In Experiment 6, full feedback yielded 87.5 relative-score points versus 75.8 for no feedback, d=0.32, p=.024; either component alone was not significant, d=0.05 and d=-0.09.",
  "conditions_and_limits": "Designed planning tasks with observable information-gathering operations; some far-transfer effects were limited and unobserved cognition remained possible.",
  "criticism": "This is direct prior art for AI teaching heuristics, but it does not create a longitudinal heuristic-association graph.",
  "which_workflow": "HHITL audit \u2014 human heuristic learning outcome precedent.",
  "claimable_sentence": "AI-generated metacognitive feedback has taught planning strategies under controlled conditions and produced bounded transfer and retention effects."
},
{
  "id": "hhitl_kefalidou_2017_interactive_feedback",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Kefalidou, G. (2017). When immediate interactive feedback boosts optimization problem solving: A 'human-in-the-loop' approach for solving Capacitated Vehicle Routing Problems. Computers in Human Behavior, 73, 110\u2013124.",
  "link": "https://doi.org/10.1016/j.chb.2017.03.019",
  "peer_reviewed": true,
  "population": "Experiment 1 included 69 postgraduate students; Experiment 2 included 77 postgraduate students across paper, stripped-tool, and full-tool conditions.",
  "design": "Two laboratory experiments compared delayed feedback and interactive computerized support; the full tool included two heuristic planning supports plus live feedback.",
  "finding_verbatim": "the aim of this paper is to move a step back and look upon human performance per se and not learning",
  "effect_size": "Experiment 2 showed a group effect on percentage above optimal, F(2,73)=7.09, partial eta squared=.163, p=.002; the two software versions did not differ significantly in solution quality.",
  "conditions_and_limits": "Route-optimization tasks with postgraduate students; heuristic support was bundled with other features; no independent learning or skill-acquisition test.",
  "criticism": "The study puts heuristics and feedback in a human task loop but explicitly does not support a learning claim.",
  "which_workflow": "HHITL audit \u2014 closest HITL heuristic-support experiment.",
  "claimable_sentence": "An interactive tool with heuristic planning support improved in-task optimization performance, but heuristic support did not outperform explanatory feedback alone and learning was not tested."
},
{
  "id": "hhitl_bucinca_et_al_2021_cognitive_forcing",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Bu\u00e7inca, Z., Malaya, M. B., & Gajos, K. Z. (2021). To trust or to think: Cognitive forcing functions can reduce overreliance on AI in AI-assisted decision-making. Proceedings of the ACM on Human-Computer Interaction, 5(CSCW1), Article 188.",
  "link": "https://doi.org/10.1145/3449287",
  "peer_reviewed": true,
  "population": "199 participants completing nutrition-related decisions with AI recommendations and explanations.",
  "design": "Randomized comparison of a simple explainable-AI interface and cognitive-forcing interfaces that required an initial judgment or delayed the recommendation.",
  "finding_verbatim": "cognitive forcing significantly reduced overreliance compared to the simple explainable AI approaches",
  "effect_size": "On wrong-AI trials, overall correctness was .03 with simple XAI versus .09 with forcing (d = .37); for carbohydrate source, .08 versus .27 (d = .66).",
  "conditions_and_limits": "Forcing did not significantly improve total performance over simple XAI, created friction, was least preferred, and benefited higher-need-for-cognition participants more. The study used a simulated AI and one short decision task.",
  "criticism": "Cognitive forcing can change engagement and reliance under bounded conditions; it is not evidence of durable strategy learning or a universal performance gain.",
  "which_workflow": "HHITL cognitive forcing, automation bias, and evaluation heterogeneity.",
  "claimable_sentence": "Cognitive forcing reduced overreliance when the AI was wrong, but did not yield a general performance gain and imposed usability costs."
},
{
  "id": "hhitl_gajos_mamykina_2022_incidental_learning",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Gajos, K. Z., & Mamykina, L. (2022). Do People Engage Cognitively with AI? Impact of AI Assistance on Incidental Learning. Proceedings of the 27th International Conference on Intelligent User Interfaces, 794\u2013806.",
  "link": "https://doi.org/10.1145/3490099.3511138",
  "peer_reviewed": true,
  "population": "Online adults making nutritional decisions: Experiment 2 N=268; Experiment 3 N=221; independent replication N=270.",
  "design": "Three experiments compared feedback, recommendations, explanations, delayed recommendations, and explanation-only assistance using pretest-to-posttest normalized changes.",
  "finding_verbatim": "it may not be sufficient to provide people with AI-generated recommendations and explanations",
  "effect_size": "In Experiment 3, explanation-only versus minimal feedback yielded r=.31 for immediate benefit and r=.23 for learning; replication yielded r=.34 and r=.20. Explanation-only versus explanation-feedback learning was r=.03.",
  "conditions_and_limits": "Simulated AI, online food-choice judgments, short-term incidental learning, and no delayed transfer or heuristic graph.",
  "criticism": "The study demonstrates that assistance format changes learning, making in-loop uptake partly an interface effect.",
  "which_workflow": "HHITL audit \u2014 immediate performance versus learning.",
  "claimable_sentence": "Requiring users to reason from an AI explanation without a recommendation improved immediate decisions and incidental learning in one experiment and replication."
},
{
  "id": "hhitl_lu_et_al_2025_colearning",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Lu, J., Yan, Y., Huang, K., Yin, M., & Zhang, F. (2025). Do we learn from each other: Understanding the human\u2013AI co-learning process embedded in human\u2013AI collaboration. Group Decision and Negotiation, 34(2), 235\u2013271.",
  "link": "https://doi.org/10.1007/s10726-024-09912-x",
  "peer_reviewed": true,
  "population": "Participants in three online emotion-classification experiments; the public article record does not expose one combined sample size suitable for this ledger.",
  "design": "Two-stage between-subject experiments separated collaborative training from later independent human performance and model updating.",
  "finding_verbatim": "this expected dual-pathway co-learning process does not occur spontaneously",
  "effect_size": "The public article record reports directional and conditional effects but no single standardized effect suitable for this ledger.",
  "conditions_and_limits": "Online emotion classification used deliberately moderate human and AI baselines; effects varied with workflow, feedback, and cognitive reflection. The study does not validate heuristic graphs or educational writing transfer.",
  "criticism": "Interaction is not evidence of mutual learning unless later unsupported human and machine performance are tested separately.",
  "which_workflow": "HHITL co-learning construct, independent criteria, and staged evaluation.",
  "claimable_sentence": "Human\u2013AI co-learning experiments already separate collaboration from later independent performance and show that co-learning depends on workflow and feedback."
},
{
  "id": "hhitl_molenaar_2022_hybrid_learning",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Molenaar, I. (2022). Towards hybrid human\u2013AI learning technologies. European Journal of Education, 57(4), 632\u2013645.",
  "link": "https://doi.org/10.1111/ejed.12527",
  "peer_reviewed": true,
  "population": "No participant sample.",
  "design": "Conceptual framework combining augmentation, hybrid intelligence, detect-diagnose-act functions, and levels of automation in education.",
  "finding_verbatim": "learning remains an essentially human activity",
  "effect_size": "Not applicable; conceptual framework.",
  "conditions_and_limits": "The article proposes a common language and research agenda rather than testing a learning effect.",
  "criticism": "It clarifies distributed control but does not specify how a human reasoning trace becomes evidence of learning.",
  "which_workflow": "HHITL audit \u2014 hybrid educational AI.",
  "claimable_sentence": "Hybrid educational AI frameworks distribute control while retaining human learning as the educational objective."
},
{
  "id": "hhitl_dellermann_et_al_2019_hybrid_intelligence",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Dellermann, D., Ebel, P., S\u00f6llner, M., & Leimeister, J. M. (2019). Hybrid intelligence. Business & Information Systems Engineering, 61(5), 637\u2013643.",
  "link": "https://doi.org/10.1007/s12599-019-00595-2",
  "peer_reviewed": true,
  "population": "No participant sample; conceptual synthesis of human and machine capabilities in hybrid systems.",
  "design": "Definition and research agenda for systems intended to combine complementary human and machine strengths and improve through interaction.",
  "finding_verbatim": "combining human and artificial intelligence to collectively achieve superior results",
  "effect_size": "No empirical effect size was applicable.",
  "conditions_and_limits": "Broad conceptual framing rather than proof that hybrid systems achieve complementarity; no educational heuristic trace or validation study was reported.",
  "criticism": "HHITL is nested within, not a replacement for, hybrid-intelligence prior art; configuration claims must therefore be narrower than complementarity itself.",
  "which_workflow": "HHITL hybrid-intelligence boundary and complementarity.",
  "claimable_sentence": "Hybrid intelligence already frames human\u2013AI systems around complementary capabilities and mutual improvement rather than mere co-presence."
},
{
  "id": "hhitl_horvitz_1999_mixed_initiative",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Horvitz, E. (1999). Principles of mixed-initiative user interfaces. Proceedings of CHI '99, 159\u2013166.",
  "link": "https://doi.org/10.1145/302979.303030",
  "peer_reviewed": true,
  "population": "Interactive system designs and case examples, including the LookOut scheduling system, rather than a single controlled participant sample.",
  "design": "Principles for coupling direct manipulation with automated services under uncertainty, limited attention, and interruption costs.",
  "finding_verbatim": "an elegant coupling of automated services with direct manipulation",
  "effect_size": "No single empirical effect size was applicable.",
  "conditions_and_limits": "The design framework predates contemporary generative AI and explicit expert-heuristic corpora; it allocates control rather than modeling human learning.",
  "criticism": "Dynamic human\u2013machine agency is mature prior art, so HHITL requires a more specific judgment object and discriminating test.",
  "which_workflow": "HHITL mixed initiative, consequential human agency, and category boundary.",
  "claimable_sentence": "Mixed-initiative interface research already treats automation and direct human control as a dynamic coordination problem."
},
{
  "id": "hhitl_fails_olsen_2003_iml",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Fails, J. A., & Olsen, D. R. (2003). Interactive Machine Learning. Proceedings of IUI '03, 39\u201345.",
  "link": "https://doi.org/10.1145/604045.604056",
  "peer_reviewed": true,
  "population": "Algorithm evaluations and an interactive image-processing tool; no human-learning outcome in the abstract.",
  "design": "Users train a classifier, inspect classifications, and correct them in an iterative loop.",
  "finding_verbatim": "allows users to train, classify/view and correct the classifications",
  "effect_size": "Two algorithms were evaluated; no standardized human effect is reported in the abstract.",
  "conditions_and_limits": "Perceptual-interface classifier construction; the machine model is the primary learner.",
  "criticism": "The train-inspect-correct loop is foundational component prior art but differs in the object of learning.",
  "which_workflow": "HHITL audit \u2014 interactive machine learning.",
  "claimable_sentence": "Interactive machine learning has let users iteratively train, inspect, and correct models since at least 2003."
},
{
  "id": "hhitl_amershi_et_al_2014_iml_review",
  "tier": "Tier 1 \u2014 peer-reviewed evidence synthesis",
  "full_citation": "Amershi, S., Cakmak, M., Knox, W. B., & Kulesza, T. (2014). Power to the People: The Role of Humans in Interactive Machine Learning. AI Magazine, 35(4), 105\u2013120.",
  "link": "https://doi.org/10.1609/aimag.v35i4.2513",
  "peer_reviewed": true,
  "population": "Review and case studies; no single participant sample.",
  "design": "Synthesis of interactive-machine-learning systems and human-centered design challenges.",
  "finding_verbatim": "Intelligent systems that learn interactively from their end-users are quickly becoming widespread.",
  "effect_size": "Not applicable; review and case-study article.",
  "conditions_and_limits": "Focuses on effective machine learning and user experience, not human strategy acquisition.",
  "criticism": "It establishes the breadth of human contributions to machine learning and the need to study user behavior.",
  "which_workflow": "HHITL audit \u2014 interactive machine learning field boundary.",
  "claimable_sentence": "Interactive machine learning is a mature field in which systems update from deliberate end-user input."
},
{
  "id": "hhitl_holzinger_2016_iml_health",
  "tier": "Tier 1 \u2014 peer-reviewed evidence synthesis",
  "full_citation": "Holzinger, A. (2016). Interactive Machine Learning for Health Informatics: When do we need the human-in-the-loop? Brain Informatics, 3(2), 119\u2013131.",
  "link": "https://doi.org/10.1007/s40708-016-0042-6",
  "peer_reviewed": true,
  "population": "No single participant sample; methodological review in health informatics.",
  "design": "Review of interactive machine learning for problems requiring expert knowledge and human involvement.",
  "finding_verbatim": "human expertise can help to reduce an exponential search space through heuristic selection of samples",
  "effect_size": "Not applicable; review article.",
  "conditions_and_limits": "Health-informatics and computational-hardness framing; machine outcome rather than human learning.",
  "criticism": "Human heuristic selection inside an algorithmic loop is established but serves computational search.",
  "which_workflow": "HHITL audit \u2014 expert HITL machine learning.",
  "claimable_sentence": "HITL research already treats expert heuristic selection as a way to reduce computational search."
},
{
  "id": "hhitl_knox_stone_2009_tamer",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Knox, W. B., & Stone, P. (2009). Interactively Shaping Agents via Human Reinforcement: The TAMER Framework. Proceedings of K-CAP '09.",
  "link": "https://users.cs.utah.edu/~dsbrown/readings/tamer.pdf",
  "peer_reviewed": true,
  "population": "Human trainers providing evaluative feedback to learning agents, including a Tetris demonstration setting.",
  "design": "A learning agent models a human's reinforcement signal and uses that learned model to select behavior.",
  "finding_verbatim": "a tamer agent models the human's reinforcement and exploits its model",
  "effect_size": "No single standardized effect is used in this audit.",
  "conditions_and_limits": "Human feedback shapes an agent policy; the human trainer's learning is not represented.",
  "criticism": "It is foundational prior art for consequential human evaluation in an iterative learning loop.",
  "which_workflow": "HHITL audit \u2014 human feedback.",
  "claimable_sentence": "Human evaluative feedback has long been modeled and exploited to shape agent behavior interactively."
},
{
  "id": "hhitl_christiano_et_al_2017_preferences",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Christiano, P. F., Leike, J., Brown, T., Martic, M., Legg, S., & Amodei, D. (2017). Deep Reinforcement Learning from Human Preferences. Advances in Neural Information Processing Systems, 30.",
  "link": "https://papers.nips.cc/paper/7017-deep-reinforcement-learning",
  "peer_reviewed": true,
  "population": "Non-expert humans comparing pairs of agent trajectory segments in Atari and simulated locomotion tasks.",
  "design": "Pairwise human preferences trained a reward model used to optimize an agent policy.",
  "finding_verbatim": "providing feedback on about 0.1% of our agent's interactions with the environment",
  "effect_size": "Complex novel behaviors were learned with about one hour of human feedback and judgments on about 0.1% of interactions.",
  "conditions_and_limits": "Machine reward and policy learning; the paper does not model what the human learns.",
  "criticism": "Preference traces are established machine-learning inputs and cannot be reinterpreted as human learning without additional evidence.",
  "which_workflow": "HHITL audit \u2014 preference learning and human feedback.",
  "claimable_sentence": "Human choices between alternatives can train machine reward models, but those choices are not evidence of human learning."
},
{
  "id": "hhitl_edwards_cooley_1993_expertise",
  "tier": "Tier 1 \u2014 peer-reviewed evidence synthesis",
  "full_citation": "Edwards, M., & Cooley, R. E. (1993). Expertise in expert systems: Knowledge acquisition for biological expert systems. Computer Applications in the Biosciences, 9(6), 657\u2013665.",
  "link": "https://doi.org/10.1093/bioinformatics/9.6.657",
  "peer_reviewed": true,
  "population": "No participant sample; review of knowledge-acquisition techniques.",
  "design": "Review of techniques for eliciting heuristic knowledge from experts for biological expert systems.",
  "finding_verbatim": "The additional knowledge consists of the heuristics or 'rules of thumb' used by an expert",
  "effect_size": "Not applicable; review article.",
  "conditions_and_limits": "Expert-system knowledge acquisition; no human-AI learning or longitudinal learner model.",
  "criticism": "Capturing expert heuristics as computational knowledge is longstanding prior art.",
  "which_workflow": "HHITL audit \u2014 knowledge engineering.",
  "claimable_sentence": "Knowledge engineering has long treated expert heuristics, not facts alone, as a necessary part of expertise."
},
{
  "id": "hhitl_clancey_1983_rule_expert_system",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Clancey, W. J. (1983). The epistemology of a rule-based expert system\u2014a framework for explanation. Artificial Intelligence, 20(3), 215\u2013251.",
  "link": "https://doi.org/10.1016/0004-3702(83)90008-5",
  "peer_reviewed": true,
  "population": "Analysis of MYCIN's expert-system rule representation; no participant intervention sample.",
  "design": "Epistemological and representational analysis of production rules from the perspective of explanation and teaching.",
  "finding_verbatim": "Production rules are a popular representation for encoding heuristic knowledge",
  "effect_size": "Not applicable; analytical article.",
  "conditions_and_limits": "A historical expert system in scientific and medical reasoning.",
  "criticism": "The paper shows that weakly structured rules can hide the strategic and justificatory knowledge required for teaching and reuse.",
  "which_workflow": "HHITL audit \u2014 expert heuristic representation and critique.",
  "claimable_sentence": "Rule-based expert systems established heuristic representation decades ago while revealing why rule collections are not automatically pedagogical models."
},
{
  "id": "hhitl_gaur_et_al_2022_knowledge_infused",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Gaur, M., Gunaratna, K., Bhatt, S., & Sheth, A. (2022). Knowledge-Infused Learning: A Sweet Spot in Neuro-Symbolic AI. IEEE Internet Computing, 26(4), 5\u201311.",
  "link": "https://doi.org/10.1109/MIC.2022.3179759",
  "peer_reviewed": true,
  "population": "No participant sample.",
  "design": "Conceptual taxonomy of shallow, semi-deep, and deep infusion of explicit knowledge into data-driven learning.",
  "finding_verbatim": "external and expert-curated knowledge in data-driven learning methodologies",
  "effect_size": "Not applicable; conceptual special-issue article.",
  "conditions_and_limits": "Machine learning for consistency, robustness, explainability, and control; no human learner outcome.",
  "criticism": "Expert knowledge inside an AI learning workflow is established component prior art.",
  "which_workflow": "HHITL audit \u2014 knowledge infusion and neuro-symbolic AI.",
  "claimable_sentence": "Knowledge-infused learning already integrates explicit expert-curated knowledge into machine-learning methods."
},
{
  "id": "hhitl_annervaz_et_al_2018_kg_augmented",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Annervaz, K. M., Chowdhury, S. B. R., & Dukkipati, A. (2018). Learning beyond datasets: Knowledge Graph Augmented Neural Networks for Natural Language Processing. Proceedings of NAACL-HLT 2018, 313\u2013322.",
  "link": "https://aclanthology.org/N18-1029/",
  "peer_reviewed": true,
  "population": "News20, DBPedia, and Stanford Natural Language Inference benchmark datasets; no human learner sample.",
  "design": "A neural model attends to relevant knowledge-graph facts as external prior support for prediction.",
  "finding_verbatim": "enhance learning models with world knowledge in the form of Knowledge Graph fact triples",
  "effect_size": "The paper reports significant task improvements and reduced labeled-data needs; no single standardized effect is used here.",
  "conditions_and_limits": "NLP model performance; retrieved graph knowledge is machine context, not human enactment.",
  "criticism": "It occupies graph-structured external knowledge infusion but not the proposed human-learning construct.",
  "which_workflow": "HHITL audit \u2014 graph-based knowledge infusion.",
  "claimable_sentence": "Knowledge graphs have already been used as on-demand structured prior knowledge for machine learning."
},
{
  "id": "hhitl_lewis_et_al_2020_rag",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Lewis, P., Perez, E., Piktus, A., Petroni, F., Karpukhin, V., Goyal, N., K\u00fcttler, H., Lewis, M., Yih, W.-t., Rockt\u00e4schel, T., Riedel, S., & Kiela, D. (2020). Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks. Advances in Neural Information Processing Systems, 33.",
  "link": "https://papers.neurips.cc/paper/2020/hash/6b493230205f780e1bc26945df7481e5-Abstract.html",
  "peer_reviewed": true,
  "population": "Knowledge-intensive NLP benchmark datasets; no human learner sample.",
  "design": "Generation combines a parametric sequence model with retrieved non-parametric memory from a dense Wikipedia index.",
  "finding_verbatim": "models which combine pre-trained parametric and non-parametric memory for language generation",
  "effect_size": "State of the art was reported on three open-domain QA tasks; no standardized human effect applies.",
  "conditions_and_limits": "Model benchmark performance. Retrieval establishes availability to the model, not human use or learning.",
  "criticism": "RAG occupies retrieval of external knowledge into an AI turn but not an availability-versus-human-enactment measurement model.",
  "which_workflow": "HHITL audit \u2014 retrieval-augmented generation.",
  "claimable_sentence": "RAG established retrieval of explicit non-parametric knowledge into generation, but retrieval alone is not evidence of human enactment."
},
{
  "id": "hhitl_wei_et_al_2022_cot",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Wei, J., Wang, X., Schuurmans, D., Bosma, M., Ichter, B., Xia, F., Chi, E., Le, Q. V., & Zhou, D. (2022). Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. Advances in Neural Information Processing Systems, 35.",
  "link": "https://proceedings.neurips.cc/paper_files/paper/2022/hash/9d5609613524ecf4f15af0f7b31abca4-Abstract.html",
  "peer_reviewed": true,
  "population": "Three large language models evaluated on arithmetic, commonsense, and symbolic reasoning benchmarks; no human learner sample.",
  "design": "Few-shot prompts included worked intermediate reasoning steps as exemplars.",
  "finding_verbatim": "a series of intermediate reasoning steps",
  "effect_size": "Eight exemplars with a 540B-parameter model achieved then-state-of-the-art GSM8K accuracy; no human-learning effect applies.",
  "conditions_and_limits": "Large-model benchmark performance; generated reasoning text is not a faithful trace of a person's cognition.",
  "criticism": "Structured prompting occupies the provision of reasoning scaffolds but does not establish human strategy uptake or learning.",
  "which_workflow": "HHITL audit \u2014 structured prompting.",
  "claimable_sentence": "Intermediate reasoning exemplars can improve model performance, but model reasoning traces are not human learning traces."
},
{
  "id": "hhitl_bull_kay_2016_smili",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Bull, S., & Kay, J. (2016). SMILI\u263a: A framework for interfaces to learning data in open learner models, learning analytics and related fields. International Journal of Artificial Intelligence in Education, 26(1), 293\u2013331.",
  "link": "https://doi.org/10.1007/s40593-015-0090-8",
  "peer_reviewed": true,
  "population": "Prior open-learner-model systems and interface designs; no new intervention sample.",
  "design": "Framework synthesis covering purposes, model openness, presentation, learner control, access, and interaction.",
  "finding_verbatim": "a guide for designers of interfaces for open learner models",
  "effect_size": "No intervention effect size was applicable.",
  "conditions_and_limits": "This is a design framework rather than a causal evaluation, and it does not test an expert-heuristic-use graph.",
  "criticism": "Inspectable, editable, and negotiable learner models are established prior art; learner endorsement still does not establish criterion validity.",
  "which_workflow": "HHITL inspectability and contestability of personalized learner representations.",
  "claimable_sentence": "Open learner modeling already treats visibility, editability, negotiation, and access as distinct design choices."
},
{
  "id": "hhitl_corbett_anderson_1995_bkt",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Corbett, A. T., & Anderson, J. R. (1995). Knowledge tracing: Modeling the acquisition of procedural knowledge. User Modeling and User-Adapted Interaction, 4(4), 253\u2013278.",
  "link": "https://doi.org/10.1007/BF01099821",
  "peer_reviewed": true,
  "population": "Students using an ACT programming tutor across four empirical studies; later reported groups included 20, 20, and 25 students.",
  "design": "Model tracing credited actions to expert production rules; knowledge tracing maintained a learner-specific mastery probability for each rule and updated it over opportunities.",
  "finding_verbatim": "the tutor also maintains an estimate of the probability that the student has learned each of the rules",
  "effect_size": "Experiment 1 out-of-sample prediction gave r=.47 and MAE=.16; same-data rule-specific fitting gave r=.85 and MAE=.07. In Experiment 4, 56% versus 24% met a 90% criterion, z=2.21, p<.05, with unequal practice volume.",
  "conditions_and_limits": "Canonical BKT assumes binary mastery, one-way acquisition, constant parameters, and node-level rule states; performance is observed, learning is latent.",
  "criticism": "Bayesian longitudinal learner modeling is established, but canonical BKT does not estimate changing heuristic-to-heuristic edges.",
  "which_workflow": "HHITL audit \u2014 Bayesian learner-model baseline.",
  "claimable_sentence": "Bayesian Knowledge Tracing has maintained person-specific mastery probabilities for expert-defined rules from sequential performance since the 1990s."
},
{
  "id": "hhitl_zemla_austerweil_2018_uinvite",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Zemla, J. C., & Austerweil, J. L. (2018). Estimating semantic networks of groups and individuals from fluency data. Computational Brain & Behavior, 1(1), 36\u201358.",
  "link": "https://doi.org/10.1007/s42113-018-0003-7",
  "peer_reviewed": true,
  "population": "Simulation used 50 modeled participants; human validation used 50 fluency participants and 101 edge-similarity raters.",
  "design": "Hierarchical U-INVITE jointly infers group and individual semantic networks from retrieval sequences under a Bayesian censored-random-walk model.",
  "finding_verbatim": "estimating semantic networks from semantic fluency data ... based on a censored random walk model of memory retrieval",
  "effect_size": "In human validation, estimated group-network edges had mean similarity 66.4 versus 23.9 for randomly selected non-edges.",
  "conditions_and_limits": "Binary, symmetric, undirected semantic networks; validity depends on the assumed retrieval process; not longitudinal learning.",
  "criticism": "This occupies Bayesian inference of person-specific concept-association edges from use sequences.",
  "which_workflow": "HHITL audit \u2014 Bayesian relational prior art.",
  "claimable_sentence": "A Bayesian method already estimates individual association graphs from personal retrieval sequences."
},
{
  "id": "hhitl_shaffer_et_al_2016_ena",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Shaffer, D. W., Collier, W., & Ruis, A. R. (2016). A tutorial on epistemic network analysis: Analyzing the structure of connections in cognitive, social, and interaction data. Journal of Learning Analytics, 3(3), 9\u201345.",
  "link": "https://doi.org/10.18608/jla.2016.33.3",
  "peer_reviewed": true,
  "population": "Methodological tutorial illustrated with coded cognitive, discourse, and interaction data; no single intervention sample.",
  "design": "Epistemic Network Analysis accumulates co-occurrences among coded elements within defined windows and produces weighted networks for individuals or groups.",
  "finding_verbatim": "identifying and quantifying connections among elements in coded data and representing them in dynamic network models",
  "effect_size": "Not applicable; methodological article.",
  "conditions_and_limits": "Edge meaning depends on codebook, unit, and co-occurrence window; ENA is not a Bayesian posterior-update model.",
  "criticism": "ENA occupies dynamic use-association networks but its edges do not demonstrate latent learning by themselves.",
  "which_workflow": "HHITL audit \u2014 relational trace prior art.",
  "claimable_sentence": "Epistemic Network Analysis already represents connections among coded knowledge, skills, and practices as changing individual or group networks."
},
{
  "id": "hhitl_bernholt_et_al_2026_longitudinal_networks",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Bernholt, S., Lossjew, J., & Gombert, S. (2026). Analyzing students' conceptual understanding over the course of a teaching unit: Tracking changes in knowledge structures over time. Unterrichtswissenschaft. Advance online publication.",
  "link": "https://doi.org/10.1007/s42010-026-00244-0",
  "peer_reviewed": true,
  "population": "N=300 students in grades 11\u201313 across 15 chemistry classes during a 10\u201312-week unit with 86 tasks and 17 knowledge elements.",
  "design": "Knowledge elements detected in each learner's answers and artifacts became nodes; co-occurrence in a response created edges in individual growing networks related to an end-of-unit test.",
  "finding_verbatim": "individual knowledge networks that reflect students' ability to enact and connect specific knowledge elements across the unit",
  "effect_size": "Combined network-metric model R\u00b2=.51, adjusted R\u00b2=.47; density standardized beta=-.46 in phase 1 and beta=.90 in phase 2.",
  "conditions_and_limits": "Observational, domain-specific, and complete-case analysis; no control group and no available-versus-enacted contrast.",
  "criticism": "This is the closest educational relational precedent and shows that the sign of a connectivity-performance relation can reverse across phases.",
  "which_workflow": "HHITL audit \u2014 closest longitudinal educational network precedent.",
  "claimable_sentence": "Individual knowledge-element co-enactment networks can predict later achievement, but edge density is not learning by definition."
},
{
  "id": "hhitl_ait_chabane_et_al_2026_ktdusl",
  "tier": "Tier 2 \u2014 peer-reviewed conference paper",
  "full_citation": "Ait Chabane, R., Brun, A., & Roussanaly, A. (2026). A New Domain-Informed Learner Model with Uncertainty-Aware Knowledge Mastery Propagation. Proceedings of the 19th International Conference on Educational Data Mining.",
  "link": "https://doi.org/10.5281/zenodo.21040060",
  "peer_reviewed": true,
  "population": "Junyi: 72,758 learners and 16,217,311 interactions; EEDI: 118,971 learners and 15,867,850 interactions.",
  "design": "A person-specific knowledge graph stores knowledge-component mastery and epistemic uncertainty; subjective-logic evidence is propagated through a fixed domain hierarchy.",
  "finding_verbatim": "Learner models encode the learner's current and evolving knowledge with respect to a domain",
  "effect_size": "On mono-concept EEDI, AUC improved from .680 to .702 and LogLoss from .616 to .601; on multi-concept EEDI, AUC changed from .730 to .729 and LogLoss from .578 to .576.",
  "conditions_and_limits": "Prediction of next-response correctness on mathematics logs; fixed expert domain relations; no heuristic enactment, durable learning test, or causal intervention.",
  "criticism": "It is close prior art for uncertain person-specific graph learner models and shows that relational propagation adds value only in some data regimes.",
  "which_workflow": "HHITL audit \u2014 graph-structured probabilistic learner modeling.",
  "claimable_sentence": "Person-specific graph learner models already propagate mastery and uncertainty across knowledge-component relations, with context-dependent predictive gains."
},
{
  "id": "hhitl_ji_et_al_2026_hmckt",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Ji, W., Wang, H., Wu, Q., & Zhou, G. (2026). Knowledge tracing model based on human-machine collaboration: An analysis of the impact of perceptual ambiguity, selective attention, and heuristic judgment on learning performance. Journal of Big Data, 13, Article 47.",
  "link": "https://doi.org/10.1186/s40537-026-01385-w",
  "peer_reviewed": true,
  "population": "Three public knowledge-tracing datasets: ASSISTments (4,151 learners; 325,678 interactions; 110 knowledge components), KDD Cup (574 learners; 607,026 interactions; 436 concepts), and STATICS2011 (333 learners; 189,927 interactions; 1,223 knowledge components).",
  "design": "Benchmark prediction study using 80/20 sequence splits and a spatiotemporal graph-convolution model with a learnable knowledge-component adjacency matrix; reported experiments were repeated three to five times per dataset.",
  "finding_verbatim": "This study dynamically adjusts the weights of spatial dimensions using a learnable adjacency matrix.",
  "effect_size": "The paper reports AUC=.8593 for HMCKT and .8194 after removing its active-learning component; these are predictive benchmark values, not human-learning effects.",
  "conditions_and_limits": "Edges optimize response prediction from correct-or-incorrect logs. The heuristic analyses scale selected graph weights and node representations to simulate availability and representativeness; they do not observe invoked heuristics or test independent acquisition, transfer, or durability.",
  "criticism": "This is close prior art for learning relations among knowledge components, but the shared predictive adjacency matrix is not a Bayesian posterior over a learner's heuristic-to-heuristic associations and should not be read as a validated cognitive structure.",
  "which_workflow": "HHITL audit \u2014 learned relational knowledge-tracing prior art.",
  "claimable_sentence": "A peer-reviewed knowledge-tracing model already learns weighted relations among knowledge components, but not person-specific heuristic-use edges from availability-versus-enactment observations."
},
{
  "id": "hhitl_chi_feltovich_glaser1981",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Chi, M. T. H., Feltovich, P. J., & Glaser, R. (1981). Categorization and representation of physics problems by experts and novices. Cognitive Science, 5(2), 121\u2013152.",
  "link": "https://doi.org/10.1207/s15516709cog0502_2",
  "peer_reviewed": true,
  "population": "Advanced physics doctoral students and undergraduates with introductory mechanics experience across four exploratory studies.",
  "design": "Problem-sorting, categorization, representation, and solution studies comparing expertise levels.",
  "finding_verbatim": "novices and experts categorized and represented physics problems quite differently",
  "effect_size": "Experts and novices produced similar numbers of categories, 8.4 versus 8.6, but only 5 of 20 distinct category labels overlapped; no standardized effect was reported.",
  "conditions_and_limits": "Small foundational physics samples and exploratory tasks; the studies do not establish a universal representational format for expertise.",
  "criticism": "The result supports relationally organized expertise, not the literal truth or completeness of any single heuristic graph.",
  "which_workflow": "Theoretical foundation for structured expert knowledge.",
  "claimable_sentence": "Experts often organize problems around governing principles while novices rely more on surface features, supporting structured but domain-bound representations of expertise."
},
{
  "id": "hhitl_klein_calderwood_mac_gregor1989_cdm",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Klein, G. A., Calderwood, R., & MacGregor, D. (1989). Critical decision method for eliciting knowledge. IEEE Transactions on Systems, Man, and Cybernetics, 19(3), 462\u2013472.",
  "link": "https://doi.org/10.1109/21.31053",
  "peer_reviewed": true,
  "population": "Experienced practitioners recounting consequential incidents in naturalistic work domains.",
  "design": "Method paper describing repeated interview passes, incident timelines, decision-point identification, and probes for cues, goals, expectancies, options, and counterfactuals.",
  "finding_verbatim": "a retrospective interview strategy that applies a set of cognitive probes",
  "effect_size": "No causal intervention effect size was reported.",
  "conditions_and_limits": "Retrospective reconstruction is vulnerable to omission and rationalization; the method samples difficult incidents rather than exhaustively reading expert cognition.",
  "criticism": "Foundational prior art for decision-point and cue elicitation, but not evidence that the elicited representation improves learning.",
  "which_workflow": "Expert-heuristic elicitation and prior art.",
  "claimable_sentence": "Critical Decision Method long predates HHITL as a structured way to externalize cues, goals, expectancies, and decisions from expert incidents."
},
{
  "id": "hhitl_militello_hutton1998_acta",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Militello, L. G., & Hutton, R. J. B. (1998). Applied cognitive task analysis (ACTA): A practitioner's toolkit for understanding cognitive task demands. Ergonomics, 41(11), 1618\u20131641.",
  "link": "https://doi.org/10.1080/001401398186108",
  "peer_reviewed": true,
  "population": "CTA-naive interviewers and subject-matter experts in firefighting and electronic-warfare tasks.",
  "design": "Matched comparison after a shared two-hour orientation; the ACTA group received six further hours of method training and produced task diagrams, audits, simulations, and cognitive-demand tables.",
  "finding_verbatim": "ACTA is intended to provide a flexible toolkit",
  "effect_size": "Experts judged 92% and 94% of items cognitive, 95% and 90% expert-only, and 73% and 87% relevant in the two domains; no standardized between-group effect was reported.",
  "conditions_and_limits": "Developer-led, small evaluation; both groups received cognitive-task-analysis framing, and inter-rater agreement was weak for parts of one domain.",
  "criticism": "The paper supports a practical decomposition but calls for stronger reliability and validity evidence.",
  "which_workflow": "Externalization method and provenance requirements.",
  "claimable_sentence": "ACTA provides established tools for eliciting cognitive demands, cues, strategies, and common errors, but its outputs require independent reliability and validity checks."
},
{
  "id": "hhitl_smink_et_al2012_appendectomy_cta",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Smink, D. S., Peyre, S. E., Soybel, D. I., Tavakkolizadeh, A., Vernon, A. H., & Anastakis, D. J. (2012). Utilization of a cognitive task analysis for laparoscopic appendectomy to identify differentiated intraoperative teaching objectives. American Journal of Surgery, 203(4), 540\u2013545.",
  "link": "https://doi.org/10.1016/j.amjsurg.2011.11.002",
  "peer_reviewed": true,
  "population": "Three expert surgeons decomposing laparoscopic appendectomy.",
  "design": "Cognitive task analysis comparing convergence on operative steps, decision points, and trainee-level teaching objectives.",
  "finding_verbatim": "Significant differences emerged in the decision points identified",
  "effect_size": "All three experts identified 18 of 24 operative steps, 75%, but only 5 of 27 decision points, 19%; junior-versus-senior priority differences were significant at p<.01.",
  "conditions_and_limits": "Only three experts and one surgical procedure; the study measured agreement, not downstream learning or patient outcomes.",
  "criticism": "The cognitive structure was much less reproducible than the visible procedure, making source plurality and disagreement essential.",
  "which_workflow": "Expert disagreement and heuristic provenance.",
  "claimable_sentence": "Experts may converge on procedural steps while diverging sharply on decision points, so one canonical heuristic graph can conceal material disagreement."
},
{
  "id": "hhitl_hinds1999_curse_of_expertise",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Hinds, P. J. (1999). The curse of expertise: The effects of expertise and debiasing methods on predictions of novice performance. Journal of Experimental Psychology: Applied, 5(2), 205\u2013221.",
  "link": "https://doi.org/10.1037/1076-898X.5.2.205",
  "peer_reviewed": true,
  "population": "Voicemail users with different expertise and participants given brief LEGO-task familiarity.",
  "design": "Observational and experimental comparisons of predicted versus actual novice completion time, with debiasing prompts.",
  "finding_verbatim": "people with more task knowledge were less accurate at predicting novice task completion times",
  "effect_size": "Voicemail actual median was 33 minutes; expert estimate 12.86, intermediate 20.09, novice 15.71, F(2,92)=3.83, p<.05. LEGO actual was 12.27; high-expertise estimate 8.5 versus low 12.9, F(1,45)=8.96, p<.05.",
  "conditions_and_limits": "The tasks were bounded and prediction focused; tested debiasing did not generalize to all possible methods.",
  "criticism": "Expert accounts of novice errors should be treated as hypotheses and validated against novice behavior.",
  "which_workflow": "Expert selection and novice-error validation.",
  "claimable_sentence": "Greater task expertise can impair predictions of novice performance, so expert-authored novice-error heuristics need independent novice data."
},
{
  "id": "hhitl_fox_ericsson_best2011_verbal_reports",
  "tier": "Tier 1 \u2014 peer-reviewed evidence synthesis",
  "full_citation": "Fox, M. C., Ericsson, K. A., & Best, R. (2011). Do procedures for verbal reporting of thinking have to be reactive? A meta-analysis and recommendations for best reporting methods. Psychological Bulletin, 137(2), 316\u2013344.",
  "link": "https://doi.org/10.1037/a0021663",
  "peer_reviewed": true,
  "population": "Ninety-four studies and nearly 3,500 participants across verbal-report tasks.",
  "design": "Meta-analysis comparing strict concurrent think-aloud, directed explanation or description, and silent controls.",
  "finding_verbatim": "The think-aloud effect size is indistinguishable from zero",
  "effect_size": "Strict concurrent reporting r=-.03, 95% CI [-.10,.03], p=.35; directed explanation r=.23, 95% CI [.14,.31], p<.001; verbal reporting generally increased completion time.",
  "conditions_and_limits": "An average nonreactive accuracy effect does not imply completeness or validity of the reports. Directed explanation is a different, reactive procedure.",
  "criticism": "A heuristic-use interface that requests explanation is partly producing the process it records.",
  "which_workflow": "Measurement reactivity and multimethod process validation.",
  "claimable_sentence": "Neutral concurrent reporting was not detectably accuracy-reactive on average, while directed explanation altered performance and all verbal reporting added time."
},
{
  "id": "hhitl_van_gog_et_al2005_process_tracing",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "van Gog, T., Paas, F., van Merri\u00ebnboer, J. J. G., & Witte, P. (2005). Uncovering the problem-solving process: Cued retrospective reporting versus concurrent and retrospective reporting. Journal of Experimental Psychology: Applied, 11(4), 237\u2013244.",
  "link": "https://doi.org/10.1037/1076-898X.11.4.237",
  "peer_reviewed": true,
  "population": "Twenty-six participants troubleshooting electrical circuits.",
  "design": "Within-person comparison of concurrent reports, uncued retrospective reports, and retrospective reports cued by replay.",
  "finding_verbatim": "cued retrospective reporting was superior to concurrent and retrospective reporting",
  "effect_size": "No single standardized intervention effect was reported; methods differed in the quantity and type of process information recovered.",
  "conditions_and_limits": "One technical troubleshooting task and a small sample. Cued retrospection can invite reconstruction; concurrent reports can omit metacognition.",
  "criticism": "No one trace channel supplied a complete account, so interface logs should not be treated as a ground truth for reasoning.",
  "which_workflow": "Triangulation of heuristic enactment.",
  "claimable_sentence": "Concurrent, retrospective, and replay-cued reports reveal different parts of problem solving; none should be assumed complete."
},
{
  "id": "hhitl_dhami_ayton2001_bail_cue_use",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Dhami, M. K., & Ayton, P. (2001). Bailing and jailing the fast and frugal way. Journal of Behavioral Decision Making, 14(2), 141\u2013168.",
  "link": "https://doi.org/10.1002/bdm.371",
  "peer_reviewed": true,
  "population": "Eighty-one magistrates from 44 courts judging hypothetical bail cases.",
  "design": "Judgment-modeling study comparing stated cue use, behaviorally inferred cue use, confidence, and consistency.",
  "finding_verbatim": "magistrates' decisions were better described by a simple matching heuristic",
  "effect_size": "The study reported divergence between stated and inferred cue use plus intra- and inter-rater inconsistency; no single standardized effect summarizes the design.",
  "conditions_and_limits": "Hypothetical cases and one judicial context. Behavioral fit describes choice patterns but does not establish normativity or fairness.",
  "criticism": "Self-report and behavior answer different questions and need separate measures.",
  "which_workflow": "Espoused versus enacted heuristics.",
  "claimable_sentence": "Magistrates' stated cue use diverged from cue use inferred from their judgments, supporting separate measures of espoused and enacted strategy."
},
{
  "id": "hhitl_dhami2003_bail_heuristic",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Dhami, M. K. (2003). Psychological models of professional decision making. Psychological Science, 14(2), 175\u2013180.",
  "link": "https://doi.org/10.1111/1467-9280.01438",
  "peer_reviewed": true,
  "population": "Three hundred forty-two real bail decisions from two courts.",
  "design": "Field judgment modeling with cross-validation of a sparse matching heuristic against a 25-cue compensatory model.",
  "finding_verbatim": "The simple matching heuristic predicted judges' decisions more accurately",
  "effect_size": "Held-out accuracy was 91.8% versus 86.3% in one court and 85.4% versus 73.4% in the other.",
  "conditions_and_limits": "Predicting professional decisions is not the same as predicting legally or ethically valid outcomes. Dominant cues reflected prior institutional decisions.",
  "criticism": "Descriptive fidelity can fossilize institutional practice and invalid proxies.",
  "which_workflow": "Heuristic validity and bias fossilization.",
  "claimable_sentence": "A sparse heuristic can predict expert practice very well while relying on institutionally troubling cues; fidelity is not validity."
},
{
  "id": "hhitl_gick_holyoak1980_analogical_transfer",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Gick, M. L., & Holyoak, K. J. (1980). Analogical problem solving. Cognitive Psychology, 12(3), 306\u2013355.",
  "link": "https://doi.org/10.1016/0010-0285(80)90013-4",
  "peer_reviewed": true,
  "population": "Participants across five laboratory experiments solving analogically related problems.",
  "design": "Experiments varying exposure to a source analogy and whether participants received an explicit hint to use it.",
  "finding_verbatim": "many subjects failed to make spontaneous use of the analogy",
  "effect_size": "Spontaneous transfer was roughly 20\u201330%, whereas an explicit hint raised solution rates to roughly 75\u201392%, depending on experiment.",
  "conditions_and_limits": "Classic laboratory insight problems; exact rates vary by experiment and should not be pooled into one new estimate.",
  "criticism": "Prompted use measures supported retrieval, not necessarily spontaneous acquisition or far transfer.",
  "which_workflow": "Availability, retrieval, and transfer distinction.",
  "claimable_sentence": "Encountering a relevant strategy did not ensure spontaneous transfer; explicit retrieval cues sharply increased its use."
},
{
  "id": "hhitl_tofel_grehl_feldon2013_ctainstruction",
  "tier": "Tier 1 \u2014 peer-reviewed evidence synthesis",
  "full_citation": "Tofel-Grehl, C., & Feldon, D. F. (2013). Cognitive task analysis-based training: A meta-analysis of studies. Journal of Cognitive Engineering and Decision Making, 7(3), 293\u2013304.",
  "link": "https://doi.org/10.1177/1555343412474821",
  "peer_reviewed": true,
  "population": "Twenty studies contributing 56 comparisons of cognitive-task-analysis-informed instruction.",
  "design": "Meta-analysis of downstream training outcomes by elicitation and instructional method.",
  "finding_verbatim": "CTA-based instruction has a large positive effect on learning",
  "effect_size": "Overall Hedges' g=.871. Reported method subgroups included CDM g=.329 across 4 comparisons and PARI g=1.598 across 11; heterogeneity and non-independence were substantial.",
  "conditions_and_limits": "Many included studies had weak reporting or non-peer-reviewed origins; measures and methods varied, and multiple comparisons came from the same studies.",
  "criticism": "The pooled result supports potential, not a general effect for any heuristic-externalization method.",
  "which_workflow": "Evidence for CTA-derived instruction.",
  "claimable_sentence": "CTA-informed instruction improved assessed performance on average, but effects and methods were highly heterogeneous."
},
{
  "id": "hhitl_feldon_et_al2010_biology_cta",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Feldon, D. F., Timmerman, B. C., Stowe, K. A., & Showman, R. (2010). Translating expertise into effective instruction: The impacts of cognitive task analysis-based training. Journal of Research in Science Teaching, 47(6), 678\u2013701.",
  "link": "https://doi.org/10.1002/tea.20382",
  "peer_reviewed": true,
  "population": "Three hundred fourteen undergraduate biology students.",
  "design": "Double-blind course study comparing traditional instruction with supplements derived from cognitive task analysis.",
  "finding_verbatim": "students receiving CTA-based instruction were significantly less likely to withdraw",
  "effect_size": "Withdrawal was 8.1% under traditional instruction versus 1.4% with CTA-derived supplementation; completers improved on several scientific-discussion-writing dimensions.",
  "conditions_and_limits": "Course-level bundled instruction; it did not isolate heuristic invocation, AI assistance, or far transfer.",
  "criticism": "The result supports task-specific instructional externalization but not the full HHITL mechanism.",
  "which_workflow": "Externalized expertise as instruction.",
  "claimable_sentence": "CTA-derived instructional supplements improved persistence and selected writing outcomes in one undergraduate biology course."
},
{
  "id": "hhitl_kalyuga_et_al2003_expertise_reversal",
  "tier": "Tier 1 \u2014 peer-reviewed evidence synthesis",
  "full_citation": "Kalyuga, S., Ayres, P., Chandler, P., & Sweller, J. (2003). The expertise reversal effect. Educational Psychologist, 38(1), 23\u201331.",
  "link": "https://doi.org/10.1207/S15326985EP3801_4",
  "peer_reviewed": true,
  "population": "Prior instructional experiments across learner expertise levels.",
  "design": "Conceptual and empirical review of interactions between learner expertise and instructional guidance.",
  "finding_verbatim": "instructional techniques that are effective with inexperienced learners can lose their effectiveness",
  "effect_size": "No single pooled effect size was reported.",
  "conditions_and_limits": "A review of cognitive-load studies rather than HHITL or generative AI; reversal depends on task and prior knowledge.",
  "criticism": "A fixed heuristic layer may burden advanced learners, so adaptive fading and learner control require direct testing.",
  "which_workflow": "Scaffolding, adaptation, and equity by expertise.",
  "claimable_sentence": "Guidance that helps novices can become redundant or harmful for more knowledgeable learners."
},
{
  "id": "hhitl_hooshyar_et_al2020_olmreview",
  "tier": "Tier 1 \u2014 peer-reviewed evidence synthesis",
  "full_citation": "Hooshyar, D., Pedaste, M., Saks, K., Leijen, \u00c4., Bardone, E., & Wang, M. (2020). Open learner models in supporting self-regulated learning in higher education: A systematic literature review. Computers & Education, 154, 103878.",
  "link": "https://doi.org/10.1016/j.compedu.2020.103878",
  "peer_reviewed": true,
  "population": "Sixty-four higher-education articles spanning 30 years of open-learner-model research.",
  "design": "Systematic review of model types, self-regulated-learning phases, design features, and evidence gaps.",
  "finding_verbatim": "a simple inspectable OLM is more preferred",
  "effect_size": "No pooled intervention effect was reported.",
  "conditions_and_limits": "Heterogeneous studies and outcomes; preference does not establish learning efficacy. Preparation and emotion were less often supported.",
  "criticism": "The review shows a mature adjacency and unresolved questions about granularity, control, transparency, and theoretical mechanism.",
  "which_workflow": "Open learner model prior art and learner agency.",
  "claimable_sentence": "Open learner models have mainly supported cognition and monitoring; their learning effects and preferred control levels remain context dependent."
},
{
  "id": "hhitl_long_aleven2017_olmexperiments",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Long, Y., & Aleven, V. (2017). Enhancing learning outcomes through self-regulated learning support with an Open Learner Model. User Modeling and User-Adapted Interaction, 27, 55\u201388.",
  "link": "https://doi.org/10.1007/s11257-016-9186-6",
  "peer_reviewed": true,
  "population": "A total of 302 seventh- and eighth-grade students across two classroom experiments.",
  "design": "Factorial classroom experiments testing access to an open learner model and shared control of problem selection in an equation-solving tutor.",
  "finding_verbatim": "the hypothesized main effects were not confirmed",
  "effect_size": "The smaller experiment found an OLM main effect; the larger did not, but found a significant interaction when OLM access was paired with shared problem selection. A single standardized effect was not reported in the abstract.",
  "conditions_and_limits": "Equation solving in one tutoring system; findings differed across experiments and depended on combined supports.",
  "criticism": "Inspectability can help, but its effect is not separable from the activity and control structure around it.",
  "which_workflow": "Inspectable learner models and conditional learning effects.",
  "claimable_sentence": "Open learner models produced conditional classroom benefits rather than a stable main effect across two experiments."
},
{
  "id": "hhitl_salomon_perkins_globerson1991_with_of_technology",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Salomon, G., Perkins, D. N., & Globerson, T. (1991). Partners in cognition: Extending human intelligence with intelligent technologies. Educational Researcher, 20(3), 2\u20139.",
  "link": "https://doi.org/10.3102/0013189X020003002",
  "peer_reviewed": true,
  "population": "Conceptual synthesis of research on cognition and intelligent technologies.",
  "design": "Theoretical distinction between performance effects with technology and lasting cognitive effects of technology.",
  "finding_verbatim": "effects with a technology and effects of it",
  "effect_size": "No empirical effect size was applicable.",
  "conditions_and_limits": "Conceptual paper predating generative AI; it supplies a construct distinction rather than an HHITL test.",
  "criticism": "The distinction is essential but must be operationalized with unaided transfer and retention measures.",
  "which_workflow": "Learning versus supported performance.",
  "claimable_sentence": "Joint performance with a tool and durable learning from using it are different outcomes that require different tests."
},
{
  "id": "hhitl_hollan_hutchins_kirsh2000_distributed_cognition",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Hollan, J., Hutchins, E., & Kirsh, D. (2000). Distributed cognition: Toward a new foundation for human-computer interaction research. ACM Transactions on Computer-Human Interaction, 7(2), 174\u2013196.",
  "link": "https://doi.org/10.1145/353485.353487",
  "peer_reviewed": true,
  "population": "Conceptual HCI analysis drawing on cognitive ethnography and distributed-cognition research.",
  "design": "Theoretical account of cognition distributed across people, artifacts, representations, and time.",
  "finding_verbatim": "cognitive processes may be distributed across the members of a social group",
  "effect_size": "No empirical effect size was applicable.",
  "conditions_and_limits": "The framework does not identify when distributed systems improve learning or who should receive authorship credit.",
  "criticism": "Distributed competence during assisted work should not be mistaken for an internal learner state.",
  "which_workflow": "Theoretical foundation for joint cognition and scaffolding.",
  "claimable_sentence": "Cognition can be distributed across people and artifacts, making assisted system performance a property of the joint arrangement."
},
{
  "id": "hhitl_haynes_et_al2009_checklist",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Haynes, A. B., Weiser, T. G., Berry, W. R., et al. (2009). A surgical safety checklist to reduce morbidity and mortality in a global population. New England Journal of Medicine, 360, 491\u2013499.",
  "link": "https://doi.org/10.1056/NEJMsa0810119",
  "peer_reviewed": true,
  "population": "3,733 baseline and 3,955 post-implementation noncardiac surgery patients in eight hospitals across eight cities.",
  "design": "Prospective multicenter before\u2013after implementation of a 19-item surgical safety checklist and team-process program.",
  "finding_verbatim": "The rate of death was 1.5% before the checklist and declined to 0.8% afterward",
  "effect_size": "Deaths 1.5% to .8%, p=.003; inpatient complications 11.0% to 7.0%, p<.001.",
  "conditions_and_limits": "Nonrandomized before\u2013after bundle; secular trends, observation, culture change, and co-interventions remain alternative explanations.",
  "criticism": "A large implementation-associated effect cannot be attributed to a decontextualized checklist item alone.",
  "which_workflow": "Positive checklist evidence and implementation mechanism.",
  "claimable_sentence": "A supported checklist rollout was associated with large reductions in surgical complications and mortality across eight hospitals."
},
{
  "id": "hhitl_urbach_et_al2014_ontario_checklist",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Urbach, D. R., Govindarajan, A., Saskin, R., Wilton, A. S., & Baxter, N. N. (2014). Introduction of surgical safety checklists in Ontario, Canada. New England Journal of Medicine, 370, 1029\u20131038.",
  "link": "https://doi.org/10.1056/NEJMsa1308261",
  "peer_reviewed": true,
  "population": "101 Ontario hospitals; 109,341 procedures before and 106,370 after checklist adoption.",
  "design": "Population natural experiment using administrative outcomes in three-month windows before and after mandated rollout.",
  "finding_verbatim": "not associated with significant reductions in operative mortality or complications",
  "effect_size": "Adjusted mortality .71% to .65%, OR .91, 95% CI [.80,1.03], p=.13; complications 3.86% to 3.82%, OR .97, 95% CI [.90,1.03], p=.29.",
  "conditions_and_limits": "Administrative data and brief before\u2013after windows; implementation fidelity beyond reported adoption varied and rare outcomes limit power at individual hospitals.",
  "criticism": "Mandated presence and high reported compliance did not reproduce the earlier large effects.",
  "which_workflow": "Null checklist evidence and implementation limits.",
  "claimable_sentence": "Checklist adoption alone did not significantly reduce mortality or complications across Ontario hospitals."
},
{
  "id": "hhitl_arriaga_et_al2013_crisis_checklists",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Arriaga, A. F., Bader, A. M., Wong, J. M., et al. (2013). Simulation-based trial of surgical-crisis checklists. New England Journal of Medicine, 368, 246\u2013253.",
  "link": "https://doi.org/10.1056/NEJMsa1204720",
  "peer_reviewed": true,
  "population": "Seventeen operating-room teams participating in 106 high-fidelity simulated surgical crises.",
  "design": "Counterbalanced simulation trial comparing crisis management with and without event-specific checklists.",
  "finding_verbatim": "Failure to adhere to lifesaving processes of care was less common",
  "effect_size": "Critical steps missed: 6% with checklist versus 23% without; adjusted RR .28, 95% CI [.18,.42], p<.001.",
  "conditions_and_limits": "High-fidelity simulation, not patient outcomes; teams knew they were observed and repeatedly encountered scenarios.",
  "criticism": "Strong task-matched performance evidence should not be generalized to routine or educational uses without testing.",
  "which_workflow": "Point-of-work cognitive aid efficacy.",
  "claimable_sentence": "Event-specific checklists sharply reduced missed critical steps during simulated operating-room crises."
},
{
  "id": "hhitl_morewedge_et_al2015_debiasing",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Morewedge, C. K., Yoon, H., Scopelliti, I., Symborski, C. W., Korris, J. H., & Kassam, K. S. (2015). Debiasing decisions: Improved decision making with a single training intervention. Policy Insights from the Behavioral and Brain Sciences, 2(1), 129\u2013140.",
  "link": "https://doi.org/10.1177/2372732215600886",
  "peer_reviewed": true,
  "population": "Participants in two longitudinal experiments receiving a debiasing game, instructional video, or comparison intervention.",
  "design": "Training experiments covering six biases, with immediate and delayed assessments and taught and untaught formats.",
  "finding_verbatim": "Both kinds of interventions produced medium to large debiasing effects immediately",
  "effect_size": "Immediate reductions: games at least 31.94%, videos at least 18.60%; after at least two months: games at least 23.57%, videos at least 19.20%.",
  "conditions_and_limits": "Specific games and videos about selected biases; percentage reductions are study-specific and not a general effect for all heuristics.",
  "criticism": "The interventions combined concepts, practice, feedback, and salience, so the active component is not a mere rule display.",
  "which_workflow": "Positive debiasing and durability evidence.",
  "claimable_sentence": "Practice-rich bias training produced sizable immediate and two-month reductions on selected judgment biases."
},
{
  "id": "hhitl_osullivan_schofield2019_slow",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "O'Sullivan, E. D., & Schofield, S. J. (2019). A cognitive forcing tool to mitigate cognitive bias: A randomised control trial. BMC Medical Education, 19, 12.",
  "link": "https://doi.org/10.1186/s12909-018-1444-3",
  "peer_reviewed": true,
  "population": "Seventy-six retained medical participants from 300 recruits.",
  "design": "Randomized evaluation of the generic SLOW cognitive-forcing mnemonic versus control on diagnostic cases.",
  "finding_verbatim": "No difference was found between the intervention and control groups",
  "effect_size": "Mean correct 2.8 versus 3.1; between-group 95% CI [-.94,.45], p=.49.",
  "conditions_and_limits": "Severe attrition left the study underpowered and vulnerable to selection bias; the mnemonic was generic rather than error-specific.",
  "criticism": "Positive subjective reactions did not translate into measured diagnostic accuracy.",
  "which_workflow": "Null cognitive-forcing evidence.",
  "claimable_sentence": "A generic forcing mnemonic did not improve diagnostic accuracy in a small randomized trial with substantial attrition."
},
{
  "id": "hhitl_vaccaro_almaatouq_malone2024_meta",
  "tier": "Tier 1 \u2014 peer-reviewed evidence synthesis",
  "full_citation": "Vaccaro, M., Almaatouq, A., & Malone, T. W. (2024). When combinations of humans and AI are useful: A systematic review and meta-analysis. Nature Human Behaviour, 8, 2293\u20132303.",
  "link": "https://doi.org/10.1038/s41562-024-02024-1",
  "peer_reviewed": true,
  "population": "106 human-participant experiments reporting 370 effects, published from January 2020 through June 2023.",
  "design": "Preregistered three-level meta-analysis requiring human-alone, AI-alone, and human\u2013AI performance.",
  "finding_verbatim": "human\u2013AI combinations performed significantly worse than the best of humans or AI alone",
  "effect_size": "Strong synergy g=-.23, 95% CI [-.39,-.07]; augmentation over human alone g=.64, 95% CI [.53,.74]. Predetermined subtask allocation across four effects: g=.22, 95% CI [-.42,.87], p=.494.",
  "conditions_and_limits": "Extreme heterogeneity, possible publication bias, varied tasks and measures, and a rapidly changing technology period.",
  "criticism": "Improvement over humans alone is not evidence that a human\u2013AI system outperforms the better solo agent.",
  "which_workflow": "Comparative baseline taxonomy and complementarity standard.",
  "claimable_sentence": "Human\u2013AI systems often augment humans but, on average, underperform the better of human or AI alone."
},
{
  "id": "hhitl_poursabzi_sangdeh_et_al2021_transparency",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Poursabzi-Sangdeh, F., Goldstein, D. G., Hofman, J. M., Wortman Vaughan, J., & Wallach, H. (2021). Manipulating and measuring model interpretability. Proceedings of CHI 2021, Article 580, 1\u201352.",
  "link": "https://doi.org/10.1145/3411764.3445315",
  "peer_reviewed": true,
  "population": "3,800 participants completing prediction tasks with models whose complexity and transparency were experimentally varied.",
  "design": "Large preregistered randomized experiments comparing model simulatability, reliance, and the ability to detect and correct model mistakes.",
  "finding_verbatim": "transparency made people less willing to correct the model's mistakes",
  "effect_size": "N = 3,800; sparse transparent models improved simulation but did not improve appropriate reliance, and no single standardized effect summarizes the experiments.",
  "conditions_and_limits": "Structured prediction tasks with simplified models; transparency was operationalized as showing model details, not as eliciting or inspecting expert heuristics.",
  "criticism": "Inspectability can improve mental-model accuracy without improving error correction and may increase deference.",
  "which_workflow": "Inspectability, contestability, and reliance calibration.",
  "claimable_sentence": "Making an AI model easier to inspect does not by itself make people better at detecting or correcting its errors."
},
{
  "id": "hhitl_bansal_et_al2021_explanations_complementarity",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Bansal, G., Wu, T., Zhou, J., Fok, R., Nushi, B., Kamar, E., Ribeiro, M. T., & Weld, D. S. (2021). Does the whole exceed its parts? The effect of AI explanations on complementary team performance. Proceedings of CHI 2021, Article 81, 1\u201316.",
  "link": "https://doi.org/10.1145/3411764.3445717",
  "peer_reviewed": true,
  "population": "Participants making decisions on three datasets with AI systems calibrated to roughly human-level accuracy.",
  "design": "Controlled human\u2013AI decision experiments comparing predictions with and without explanations and testing whether teams exceeded the better solo member.",
  "finding_verbatim": "explanations increased the likelihood that participants accepted the AI's recommendation",
  "effect_size": "Across three datasets, explanations did not produce complementary team performance; the accessible report does not provide one cross-study standardized effect.",
  "conditions_and_limits": "Tasks used specific explanation forms and AI performance comparable to human performance; results do not show that all explanations are harmful.",
  "criticism": "Acceptance is not calibrated reliance when explanations raise agreement with correct and incorrect recommendations alike.",
  "which_workflow": "Complementarity, explanation, and acceptance-error asymmetry.",
  "claimable_sentence": "Explanations can increase acceptance of AI advice without making the human\u2013AI team better than its strongest member."
},
{
  "id": "hhitl_parasuraman_manzey2010_complacency_bias",
  "tier": "Tier 1 \u2014 peer-reviewed evidence synthesis",
  "full_citation": "Parasuraman, R., & Manzey, D. H. (2010). Complacency and bias in human use of automation: An attentional integration. Human Factors, 52(3), 381\u2013410.",
  "link": "https://doi.org/10.1177/0018720810376055",
  "peer_reviewed": true,
  "population": "Evidence from laboratory and applied studies of novice and expert automation users.",
  "design": "Integrative review and attentional account of automation complacency and automation bias.",
  "finding_verbatim": "automation bias and complacency can occur in both na\u00efve and expert participants",
  "effect_size": "Review article; no defensible single pooled effect size was reported.",
  "conditions_and_limits": "Much of the underlying evidence predates generative AI and concerns monitoring or decision aids, but the reliance mechanisms remain directly relevant.",
  "criticism": "Expertise, instructions, and training do not eliminate omission and commission errors induced by automation.",
  "which_workflow": "Automation bias, expertise, and governance safeguards.",
  "claimable_sentence": "Automation bias is not confined to novices, and training alone is not a sufficient control."
},
{
  "id": "hhitl_endsley_kiris1995_out_of_loop",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Endsley, M. R., & Kiris, E. O. (1995). The out-of-the-loop performance problem and level of control in automation. Human Factors, 37(2), 381\u2013394.",
  "link": "https://doi.org/10.1518/001872095779064555",
  "peer_reviewed": true,
  "population": "Participants operating a simulated navigation task under differing levels of automated control.",
  "design": "Controlled experiment measuring situation awareness and the ability to resume manual performance after automation failures.",
  "finding_verbatim": "a significant out-of-the-loop performance problem",
  "effect_size": "Automation reduced situation awareness and slowed recovery after failures; the accessible record does not supply a single standardized effect.",
  "conditions_and_limits": "Simulation involved navigation automation rather than generative reasoning, and active control moderated the observed costs.",
  "criticism": "A system that produces good in-loop output may still degrade the user's ability to perform when support is removed.",
  "which_workflow": "Deskilling, transfer, and human-agency outcomes.",
  "claimable_sentence": "Evaluation must measure unsupported recovery and transfer, because supported performance can coexist with out-of-the-loop degradation."
},
{
  "id": "hhitl_tschandl_et_al2020_human_computer_collaboration",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Tschandl, P., Rinner, C., Apalla, Z., Argenziano, G., Codella, N., Halpern, A., Janda, M., Lallas, A., Longo, C., Malvehy, J., Paoli, J., Puig, S., Rosendahl, C., Soyer, H. P., Zalaudek, I., & Kittler, H. (2020). Human\u2013computer collaboration for skin cancer recognition. Nature Medicine, 26, 1229\u20131234.",
  "link": "https://doi.org/10.1038/s41591-020-0942-0",
  "peer_reviewed": true,
  "population": "302 medical raters evaluating 1,412 dermoscopic and close-up images.",
  "design": "Large reader study comparing unaided decisions, valid AI support, and systematically corrupted AI support across expertise levels.",
  "finding_verbatim": "all raters were susceptible to being misled by the algorithms",
  "effect_size": "Valid AI increased accuracy from 63.6% to 77.0%, +13.3 points (95% CI 11.5\u201315.2); corrupted AI reduced median accuracy 6.3 points (p = 6\u00d710\u221213).",
  "conditions_and_limits": "Medical image classification with a specific support interface; benefits depended on AI validity and do not automatically generalize to open-ended academic work.",
  "criticism": "The same interface can substantially help or harm depending on model error, including for experts.",
  "which_workflow": "Complementarity, corrupted-advice stress testing, and expertise moderation.",
  "claimable_sentence": "Valid AI support improved diagnostic accuracy, while corrupted support substantially misled raters at every expertise level."
},
{
  "id": "hhitl_fan_et_al2025_gen_aiwriting_transfer",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Fan, Y., Tang, L., Le, H., Shen, K., Tan, S., Zhao, Y., Shen, Y., Li, X., & Ga\u0161evi\u0107, D. (2025). Beware of metacognitive laziness: Effects of generative artificial intelligence on learning motivation, processes, and performance. British Journal of Educational Technology, 56, 489\u2013530.",
  "link": "https://doi.org/10.1111/bjet.13544",
  "peer_reviewed": true,
  "population": "117 students assigned to generative-AI, human-expert, or no-assistance conditions for essay writing.",
  "design": "Randomized experiment measuring supported essay performance, learning process, motivation, and later transfer performance.",
  "finding_verbatim": "students relied on GenAI feedback without engaging in higher-order thinking",
  "effect_size": "AI versus control improved supported essays by 1.970 points (95% CI .083\u20133.858), but transfer showed F = .019, p = .996, \u03b7\u00b2 = .000.",
  "conditions_and_limits": "One essay context, one support design, and a modest sample; a null transfer effect is not proof of no possible learning benefit.",
  "criticism": "Immediate output gains cannot be described as learning without retention or transfer evidence.",
  "which_workflow": "Learning definition, metacognition, transfer, and deskilling.",
  "claimable_sentence": "Generative-AI support improved the assisted essay but produced no detectable advantage on the transfer task in this experiment."
},
{
  "id": "hhitl_bassner_et_al_2026_ai_supported_programming",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Bassner, P., Lenk-Ostendorf, B., Beinstingel, R., Wasner, T., & Krusche, S. (2026). Less stress, better scores, same learning: The dissociation of performance and learning in AI-supported programming education. Computers & Education: Artificial Intelligence, 10, 100537.",
  "link": "https://doi.org/10.1016/j.caeai.2025.100537",
  "peer_reviewed": true,
  "population": "275 participants in an introductory programming course assigned to a scaffolded AI tutor, unrestricted AI assistance, or a no-AI control using conventional web resources.",
  "design": "Three-arm randomized controlled trial separating exercise performance from pre\u2013post knowledge gains and code-comprehension performance.",
  "finding_verbatim": "neither AI condition produced greater pre\u2013post knowledge gains or code-comprehension advantages",
  "effect_size": "Supported exercise performance differed, F(2,272) = 29.693, p < .001, generalized \u03b7\u00b2 = .179; the time-by-group learning interaction was F = .258, p = .773, \u03b7\u00b2 = .0003.",
  "conditions_and_limits": "One 90-minute concurrency exercise in one CS1 course; learning was measured over the study's pre\u2013post window, not as long-term retention or broad transfer.",
  "criticism": "A substantial assistance effect and a near-zero differential learning interaction can coexist; neither generalizes automatically to every AI design.",
  "which_workflow": "HHITL performance-versus-learning distinction and comparative evaluation.",
  "claimable_sentence": "This experiment found a substantial supported-performance difference but essentially no differential learning over time."
},
{
  "id": "hhitl_friedman_nissenbaum1996_computer_bias",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Friedman, B., & Nissenbaum, H. (1996). Bias in computer systems. ACM Transactions on Information Systems, 14(3), 330\u2013347.",
  "link": "https://doi.org/10.1145/230538.230561",
  "peer_reviewed": true,
  "population": "Conceptual analysis illustrated with cases of computer systems and institutional practice.",
  "design": "Taxonomy distinguishing preexisting, technical, and emergent sources of bias.",
  "finding_verbatim": "three categories of bias can arise in computer systems",
  "effect_size": "Conceptual paper; no empirical effect size was applicable.",
  "conditions_and_limits": "The taxonomy predates foundation models but applies to encoded heuristics, interfaces, data, and deployment contexts.",
  "criticism": "A heuristic repository can inherit social bias, introduce technical bias, or become biased as its context changes.",
  "which_workflow": "Bias fossilization and governance taxonomy.",
  "claimable_sentence": "Bias controls must address what experts contribute, how heuristics are formalized, and how their meaning changes in use."
},
{
  "id": "hhitl_selbst_et_al2019_abstraction_traps",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Selbst, A. D., Boyd, D., Friedler, S. A., Venkatasubramanian, S., & Vertesi, J. (2019). Fairness and abstraction in sociotechnical systems. Proceedings of FAT* 2019, 59\u201368.",
  "link": "https://doi.org/10.1145/3287560.3287598",
  "peer_reviewed": true,
  "population": "Conceptual analysis of fairness interventions in sociotechnical systems.",
  "design": "Identification of framing, portability, formalism, ripple-effect, and solutionism traps.",
  "finding_verbatim": "technical interventions can fail to account for the full social context",
  "effect_size": "Conceptual paper; no empirical effect size was applicable.",
  "conditions_and_limits": "Fairness-focused analysis rather than an empirical study of heuristic systems.",
  "criticism": "Portable expert rules can lose meaning when detached from the institutions, incentives, and exceptions that made them valid.",
  "which_workflow": "Decontextualization, portability, and sociotechnical governance.",
  "claimable_sentence": "Heuristics cannot be treated as context-free objects merely because they have been made machine-readable."
},
{
  "id": "hhitl_obermeyer_et_al2019_racial_bias",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Obermeyer, Z., Powers, B., Vogeli, C., & Mullainathan, S. (2019). Dissecting racial bias in an algorithm used to manage the health of populations. Science, 366(6464), 447\u2013453.",
  "link": "https://doi.org/10.1126/science.aax2342",
  "peer_reviewed": true,
  "population": "Patients scored by a widely used population-health management algorithm.",
  "design": "Empirical audit comparing algorithmic risk scores, healthcare costs, and illness burden by race, followed by target-variable simulation.",
  "finding_verbatim": "Black patients are considerably sicker than White patients at a given risk score",
  "effect_size": "Replacing cost with illness burden as the target would increase Black patients selected for extra help from 17.7% to 46.5%.",
  "conditions_and_limits": "One healthcare allocation setting; it demonstrates proxy bias rather than bias in expert heuristics specifically.",
  "criticism": "A seemingly reasonable measurable proxy can preserve structural inequity even without an explicit protected attribute.",
  "which_workflow": "Bias fossilization, target validity, and subgroup audit.",
  "claimable_sentence": "Governance must audit what a recorded heuristic optimizes, because plausible proxies can reproduce large inequities."
},
{
  "id": "hhitl_glickman_sharot2025_bias_amplification",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Glickman, M., & Sharot, T. (2025). How human\u2013AI feedback loops alter human perceptual, emotional and social judgements. Nature Human Behaviour, 9, 345\u2013359.",
  "link": "https://doi.org/10.1038/s41562-024-02077-2",
  "peer_reviewed": true,
  "population": "1,401 participants across experiments involving judgments and interaction with AI trained on human-produced labels.",
  "design": "Experiments tracing bias from human labels into AI and back into later human judgments during repeated interaction.",
  "finding_verbatim": "AI can amplify biases and humans can internalize these amplified biases",
  "effect_size": "A human bias of d = .34 became d = 1.33 in AI; interacting humans shifted from 49.9% to 56.3% biased judgments (d = .84, p < .001).",
  "conditions_and_limits": "Perceptual, emotional, and social judgment tasks; not a direct study of educational expert heuristics.",
  "criticism": "Human-origin content does not remain normatively safe merely because people can inspect or revise it.",
  "which_workflow": "Feedback-loop bias and longitudinal governance.",
  "claimable_sentence": "Human\u2013AI feedback loops can amplify an initial human bias and then teach the amplified bias back to people."
},
{
  "id": "hhitl_burgman_et_al2011_expert_status_accuracy",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Burgman, M. A., McBride, M., Ashton, R., Speirs-Bridge, A., Flander, L., Wintle, B., Fidler, F., Rumpff, L., & Twardy, C. (2011). Expert status and performance. PLOS ONE, 6(7), e22998.",
  "link": "https://doi.org/10.1371/journal.pone.0022998",
  "peer_reviewed": true,
  "population": "Experts participating in six structured-judgment workshops, with approximately 13\u201325 participants per workshop.",
  "design": "Comparison of peer-rated expertise, self-assessment, and performance on seed questions with known outcomes.",
  "finding_verbatim": "peer ratings of expertise were not significantly related to performance",
  "effect_size": "Across the six workshops, expert rankings did not significantly predict estimation accuracy; no pooled standardized effect was reported.",
  "conditions_and_limits": "Small workshop samples and estimation tasks in selected domains; results do not deny the value of domain expertise.",
  "criticism": "Credentials and reputation cannot substitute for calibration against outcomes when deciding which heuristics deserve authority.",
  "which_workflow": "Expert selection, validation, and provenance.",
  "claimable_sentence": "Expert-contributed heuristics require outcome calibration because peer reputation did not reliably predict estimation accuracy in these workshops."
},
{
  "id": "hhitl_kosinski_et_al2013_digital_records_privacy",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Kosinski, M., Stillwell, D., & Graepel, T. (2013). Private traits and attributes are predictable from digital records of human behavior. Proceedings of the National Academy of Sciences, 110(15), 5802\u20135805.",
  "link": "https://doi.org/10.1073/pnas.1218772110",
  "peer_reviewed": true,
  "population": "More than 58,000 volunteers whose Facebook Likes were linked to psychometric and demographic records.",
  "design": "Predictive modeling of sensitive personal attributes from sparse digital behavior records.",
  "finding_verbatim": "a wide range of people's personal attributes can be automatically and accurately inferred",
  "effect_size": "Reported prediction accuracy included 88% for male sexual orientation, 95% for ethnicity, and 85% for political-party affiliation.",
  "conditions_and_limits": "Historic Facebook-Like data and specific prediction tasks; these values should not be transferred to heuristic-use traces.",
  "criticism": "A personalized record of reasoning choices can reveal sensitive traits beyond the purpose for which it was collected.",
  "which_workflow": "Privacy, inference risk, and purpose limitation.",
  "claimable_sentence": "Behavioral traces can support accurate inference of sensitive traits, so heuristic-use histories require more protection than ordinary content logs."
},
{
  "id": "hhitl_ifenthaler_schumacher2016_learning_analytics_privacy",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Ifenthaler, D., & Schumacher, C. (2016). Student perceptions of privacy principles for learning analytics. Educational Technology Research and Development, 64, 923\u2013938.",
  "link": "https://doi.org/10.1007/s11423-016-9477-y",
  "peer_reviewed": true,
  "population": "330 university students reporting attitudes toward data collection, personalization, and sharing in learning analytics.",
  "design": "Survey study of perceived acceptability and privacy principles for educational data practices.",
  "finding_verbatim": "students are rather conservative regarding the sharing of personal data",
  "effect_size": "N = 330; the study reports item-level perceptions rather than a causal effect or one standardized summary.",
  "conditions_and_limits": "Self-reported attitudes in one higher-education context, not observed consent behavior or reactions to heuristic graphs.",
  "criticism": "Educational usefulness does not erase expectations of transparency, control, and restricted sharing.",
  "which_workflow": "Learner privacy, consent, and dashboard governance.",
  "claimable_sentence": "Students may value personalized analytics while remaining cautious about sharing the underlying personal data."
},
{
  "id": "hhitl_carlini_et_al2021_training_data_extraction",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Carlini, N., Tramer, F., Wallace, E., Jagielski, M., Herbert-Voss, A., Lee, K., Roberts, A., Brown, T., Song, D., Erlingsson, U., Oprea, A., & Raffel, C. (2021). Extracting training data from large language models. 30th USENIX Security Symposium, 2633\u20132650.",
  "link": "https://www.usenix.org/conference/usenixsecurity21/presentation/carlini-extracting",
  "peer_reviewed": true,
  "population": "A large language model and its generated outputs; no human participant sample.",
  "design": "Adversarial generation and ranking pipeline used to identify verbatim memorized training examples, followed by validation against source data.",
  "finding_verbatim": "an adversary can recover individual training examples by querying the language model",
  "effect_size": "From 600,000 generated samples, researchers manually inspected 1,800 candidates and confirmed 604 unique memorized training examples.",
  "conditions_and_limits": "One model generation and attack setting; counts are not a universal leakage rate and do not directly measure retrieval from a heuristic store.",
  "criticism": "Proprietary expert language and personal traces can become extractable if incorporated into poorly governed models or outputs.",
  "which_workflow": "Intellectual property, memorization, and data-boundary controls.",
  "claimable_sentence": "Language models can reproduce memorized training examples, making provenance, access boundaries, and non-training commitments material governance requirements."
},
{
  "id": "hhitl_draxler_et_al2024_aighostwriter",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Draxler, F., Werner, A., Lehmann, F., Hoppe, M., Schmidt, A., Buschek, D., & Welsch, R. (2024). The AI ghostwriter effect: When users do not perceive ownership of AI-generated text but self-declare as authors. ACM Transactions on Computer-Human Interaction, 31(2), Article 25.",
  "link": "https://doi.org/10.1145/3637875",
  "peer_reviewed": true,
  "population": "30 participants in an exploratory study and 96 participants in a controlled follow-up writing study.",
  "design": "Mixed-method and controlled experiments varying AI influence and measuring psychological ownership, authorship attribution, and disclosure.",
  "finding_verbatim": "users do not perceive ownership of AI-generated text but self-declare as authors",
  "effect_size": "Study 1 n = 30 and Study 2 n = 96; increasing human influence increased ownership, while the article reports no single cross-study standardized effect.",
  "conditions_and_limits": "Specific text-production interfaces and self-report measures; legal authorship and disciplinary norms were outside the experiment.",
  "criticism": "Attribution, felt ownership, actual contribution, and disclosure are separable and can diverge.",
  "which_workflow": "Authorship, agency, provenance, and disclosure.",
  "claimable_sentence": "People may claim authorship of AI-assisted text even when they report weak ownership and fail to disclose the assistance."
},
{
  "id": "hhitl_gao_et_al2023_chat_gptabstracts",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Gao, C. A., Howard, F. M., Markov, N. S., Dyer, E. C., Ramesh, S., Luo, Y., & Pearson, A. T. (2023). Comparing scientific abstracts generated by ChatGPT to real abstracts with detectors and blinded human reviewers. npj Digital Medicine, 6, 75.",
  "link": "https://doi.org/10.1038/s41746-023-00819-6",
  "peer_reviewed": true,
  "population": "Blinded human reviewers evaluating 50 generated medical abstracts alongside genuine abstracts.",
  "design": "Generation study comparing plagiarism checking, an AI-output detector, journal-format compliance, and blinded human identification.",
  "finding_verbatim": "human reviewers correctly identified only 68% of the generated abstracts",
  "effect_size": "Reviewers identified 68% of generated abstracts and falsely labeled 14% of real abstracts; only 8 of 50 generated abstracts followed journal formatting.",
  "conditions_and_limits": "Early ChatGPT version, medical abstracts, and contemporaneous detector; detector AUROC .94 should not be generalized to current models.",
  "criticism": "Surface plausibility and human detection are inadequate provenance controls for scholarly work.",
  "which_workflow": "Scholarly writing, disclosure, and output verification.",
  "claimable_sentence": "Blinded reviewers often failed to identify generated abstracts and sometimes mislabeled genuine ones, so disclosure cannot depend on detection."
},
{
  "id": "hhitl_chen_jia_2026_ai_disclosure",
  "tier": "Tier 1 \u2014 peer-reviewed primary research",
  "full_citation": "Chen, C., & Jia, X. (2026). When researchers use AI: Public trust, ethical judgments, and the perceived value of academic research. AI and Ethics, 6, Article 223.",
  "link": "https://doi.org/10.1007/s43681-026-01039-w",
  "peer_reviewed": true,
  "population": "806 participants evaluating hypothetical research descriptions with varied statements about human and AI contributions.",
  "design": "Preregistered survey experiment randomizing six disclosure statements for theory, methods, and language assistance and measuring trust-related evaluations.",
  "finding_verbatim": "the public reported strong preference for human-only research and researcher",
  "effect_size": "Individual methodological-versus-theoretical disclosure contrasts were d = .03 to .22 and not statistically significant; an exploratory, non-preregistered aggregation gave d = .13 for eight directional outcomes and d = .15 for all ten.",
  "conditions_and_limits": "Participants evaluated disclosure statements rather than completed papers. The aggregated contrasts were post hoc, and the study measured perceptions rather than research validity.",
  "criticism": "Disclosure can alter trust perceptions, but the study cannot determine whether disclosed research was more or less valid.",
  "which_workflow": "HHITL disclosure wording, audience trust, and publication evaluation.",
  "claimable_sentence": "AI-use disclosures can modestly affect perceived trustworthiness, although the study tested labels rather than completed scholarship."
},
{
  "id": "hhitl_salloch_eriksen2024_co_reasoning",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Salloch, S., & Eriksen, A. (2024). What does it mean to co-reason with AI? The American Journal of Bioethics, 24(7), 24\u201326.",
  "link": "https://doi.org/10.1080/15265161.2024.2353800",
  "peer_reviewed": true,
  "population": "Conceptual analysis in the context of clinical and ethical judgment.",
  "design": "Normative argument about practical reasoning, responsibility, and the limits of describing AI as a co-reasoner.",
  "finding_verbatim": "AI systems do not co-reason in the full sense of the term",
  "effect_size": "Conceptual commentary; no empirical effect size was applicable.",
  "conditions_and_limits": "Clinical ethics framing; it does not test an HHITL workflow or settle all philosophical accounts of agency.",
  "criticism": "Language such as collaboration and co-reasoning can obscure asymmetries of judgment, understanding, and responsibility.",
  "which_workflow": "Human agency, responsibility, and category discipline.",
  "claimable_sentence": "Calling a system a co-reasoner should not blur the human responsibility for contextual judgment and consequences."
},
{
  "id": "hhitl_icmje2026_aiuse",
  "tier": "Authoritative guidance \u2014 non-peer-reviewed",
  "full_citation": "International Committee of Medical Journal Editors. (2026). Use of artificial intelligence in publishing. Recommendations for the Conduct, Reporting, Editing, and Publication of Scholarly Work in Medical Journals.",
  "link": "https://www.icmje.org/recommendations/browse/artificial-intelligence/ai-use-by-authors.html",
  "peer_reviewed": false,
  "population": "Authors, editors, reviewers, and publishers in biomedical scholarship.",
  "design": "Consensus publication policy addressing disclosure, authorship, confidentiality, and human accountability for AI use.",
  "finding_verbatim": "AI-assisted technologies should not be listed as an author or co-author",
  "effect_size": "Policy statement; no empirical effect size was applicable.",
  "conditions_and_limits": "Biomedical publishing guidance; individual journals and other disciplines may impose additional or different requirements.",
  "criticism": "A human-authored heuristic trail cannot transfer accountability to an AI system or erase the need to disclose material assistance.",
  "which_workflow": "Authorship, disclosure, and human accountability.",
  "claimable_sentence": "Current ICMJE guidance requires disclosure of AI assistance, excludes AI from authorship, and leaves humans responsible for accuracy and integrity."
},
{
  "id": "hhitl_wame2023_chatbots",
  "tier": "Authoritative guidance \u2014 non-peer-reviewed",
  "full_citation": "World Association of Medical Editors. (2023). Chatbots, generative AI, and scholarly manuscripts: WAME recommendations on chatbots and generative artificial intelligence in relation to scholarly publications.",
  "link": "https://wame.org/pdf/Chatbots-Generative-AI-and-Scholarly-Manuscripts.pdf",
  "peer_reviewed": false,
  "population": "Journal authors, reviewers, and editors.",
  "design": "Editorial-policy recommendations for attribution, disclosure, responsibility, and confidentiality when using generative AI.",
  "finding_verbatim": "Authors should be transparent when chatbots are used",
  "effect_size": "Policy statement; no empirical effect size was applicable.",
  "conditions_and_limits": "Editorial guidance rather than law or empirical evidence; implementation varies by venue.",
  "criticism": "A provenance-rich system can support disclosure, but the existence of a log does not ensure complete or intelligible reporting.",
  "which_workflow": "Scholarly disclosure and reproducible provenance.",
  "claimable_sentence": "WAME recommends transparent reporting of generative-AI use and sufficient detail about the tool and its role for accountability."
},
{
  "id": "hhitl_nisocredi_t2022",
  "tier": "Authoritative standard \u2014 non-peer-reviewed",
  "full_citation": "National Information Standards Organization. (2022). ANSI/NISO Z39.104-2022, CRediT: Contributor Roles Taxonomy.",
  "link": "https://www.niso.org/publications/z39104-2022-credit",
  "peer_reviewed": false,
  "population": "Research contributors, publishers, institutions, and research-information systems.",
  "design": "Consensus standard defining 14 contributor roles for transparent reporting of who performed which kinds of work.",
  "finding_verbatim": "a high-level taxonomy, including 14 roles, that can be used to represent the roles",
  "effect_size": "Standards document; no empirical effect size was applicable.",
  "conditions_and_limits": "Contributor-role reporting does not itself determine authorship, responsibility, contribution quality, or AI disclosure sufficiency.",
  "criticism": "Process provenance should map human contributions to recognizable roles without treating logged interaction as proof of intellectual ownership.",
  "which_workflow": "Contribution provenance and authorship reporting.",
  "claimable_sentence": "CRediT supplies a standardized vocabulary for human contributions, but it does not decide authorship or validate the contribution."
},
{
  "id": "hhitl_gigerenzer_gaissmaier2011_heuristic_decision",
  "tier": "Tier 2 \u2014 peer-reviewed method, framework, or conceptual analysis",
  "full_citation": "Gigerenzer, G., & Gaissmaier, W. (2011). Heuristic decision making. Annual Review of Psychology, 62, 451\u2013482.",
  "link": "https://doi.org/10.1146/annurev-psych-120709-145346",
  "peer_reviewed": true,
  "population": "Research on heuristic decision making across psychology, economics, law, medicine, organizational studies, and related fields.",
  "design": "Interdisciplinary review of formal heuristic models, ecological rationality, and evidence comparing simple strategies with more information-intensive approaches.",
  "finding_verbatim": "A heuristic is a strategy that ignores part of the information",
  "effect_size": "Review article spanning heterogeneous tasks and designs; no single pooled effect size was reported or applicable.",
  "conditions_and_limits": "The review concerns formal and psychologically plausible decision strategies whose performance depends on environmental structure; it does not validate any particular expert-authored heuristic or HHITL workflow.",
  "criticism": "Calling a rule a heuristic does not establish its accuracy, expertise, fairness, or fit to a new environment.",
  "which_workflow": "Construct definition, bounded rationality, and environment\u2013strategy fit.",
  "claimable_sentence": "Heuristics are selective decision strategies whose usefulness depends on their fit with the structure of the task environment."
}
]
