[
  {
    "id": "Alfieri2013",
    "tier": "Tier 2",
    "full_citation": "Alfieri, L., Nokes-Malach, T. J., & Schunn, C. D. (2013). Learning through case comparisons: A meta-analytic review. Educational Psychologist, 48(2), 87–113.",
    "link": "https://doi.org/10.1080/00461520.2013.775712",
    "peer_reviewed": "Yes",
    "population": "57 experiments and 336 outcome tests across varied ages, academic domains, and laboratory/classroom contexts; not restricted to higher education.",
    "design": "Random-effects meta-analysis of explicit case-comparison activities versus sequential cases, single cases, nonanalogous cases, traditional instruction, or control.",
    "finding_verbatim": "Case comparison activities commonly led to greater learning outcomes than other forms of case study.",
    "effect_size": "Test-level random-effects d=0.50, 95% CI [0.44, 0.56], k=335, I²=68.05%; experiment-level d=0.60, 95% CI [0.47, 0.72], k=56, I²=59.05%. Sequential-cases comparison d=0.37, 95% CI [0.31, 0.44]; nonanalogous-cases comparison d=0.82, 95% CI [0.55, 1.08].",
    "conditions_and_limits": "Benefits were larger when learners searched for similarities, a governing principle followed comparison, content was perceptual, and testing was immediate. Mere co-presentation is not guided relational comparison.",
    "criticism": "Populations, domains, controls, and outcomes were highly heterogeneous; many outcomes were immediate and were not decision-quality measures. The headline test-level estimate treats multiple tests within experiments as units, although the authors also report an experiment-level synthesis. No modern study-level risk-of-bias assessment was reported. This supports case comparison, not role/persona multiplicity or an interface.",
    "which_workflow": "WP-05—contrasting cases; direct mechanism evidence, indirect for higher-education decision-making, no AI-interface test.",
    "claimable_sentence": "Across a heterogeneous literature, explicitly comparing cases improved learning relative to studying cases without comparison, with the clearest support for guided attention to shared structure."
  },
  {
    "id": "AtkinsonRenklMerrill2003",
    "tier": "Tier 2",
    "full_citation": "Atkinson, R. K., Renkl, A., & Merrill, M. M. (2003). Transitioning from studying examples to solving problems: Effects of self-explanation prompts and fading worked-out steps. Journal of Educational Psychology, 95(4), 774–783.",
    "link": "https://doi.org/10.1037/0022-0663.95.4.774",
    "peer_reviewed": "Yes",
    "population": "Experiment 1: 78 education/psychology undergraduates at a large southeastern US university. Experiment 2: 40 advanced-algebra high-school students. Probability problems.",
    "design": "Experiment 1 randomized a 2×2 design: backward fading versus example–problem pairs, crossed with principle-choice prompts plus correctness feedback versus no prompts. Experiment 2 randomized backward fading with versus without prompts.",
    "finding_verbatim": "This combination produced medium to large effects on near and far transfer without requiring additional time on task.",
    "effect_size": "No confidence intervals reported. Experiment 1: fading near-transfer Cohen's f=0.23, F(1,73)=4.50, p<.05; far-transfer f=0.27, F=5.99, p<.05. Prompt near f=0.25, F=5.01, p<.05; far f=0.23, F=4.50, p<.05. Experiment 2 prompt effects: near f=0.42, F(1,37)=6.65, p<.05; far f=0.37, F=5.14, p<.05.",
    "conditions_and_limits": "Backward-fading sequences moved from complete examples toward missing steps and independent problem solving. Prompts required choosing a probability principle and supplied correctness feedback; this is not purely generative self-explanation.",
    "criticism": "Small samples, one narrow probability domain, immediate post-tests, and no delayed retention; Experiment 2 lacked a non-fading comparison. Prompt, choice, and feedback are bundled. The article appears twice to mislabel its far-transfer f as a near-transfer effect, so outcomes were taken from the tables/design.",
    "which_workflow": "WP-05—fading and guided practice; direct mechanism study, not evidence for an AI expert.",
    "claimable_sentence": "A small randomized probability study supports a transition from complete examples through faded steps to problem solving, with principle-choice prompts and feedback, on immediate transfer."
  },
  {
    "id": "Barbieri2023",
    "tier": "Tier 2",
    "full_citation": "Barbieri, C. A., Miller-Cotto, D., Clerjuste, S. N., & Chawla, K. (2023). A meta-analysis of the worked examples effect on mathematics performance. Educational Psychology Review, 35, Article 11.",
    "link": "https://doi.org/10.1007/s10648-023-09745-1",
    "peer_reviewed": "Yes",
    "population": "Elementary through postsecondary mathematics learners; 43 articles, 55 studies, and 181 effect sizes. Most samples were middle school or older; 53 studies randomized and two were quasi-experimental. Several samples were undergraduates/college students.",
    "design": "Random-effects robust-variance meta-analysis of worked examples versus problem solving/no-example comparison, for initial acquisition or practice.",
    "finding_verbatim": "The worked examples effect yields a medium effect on mathematics outcomes whether used for practice or initial skill acquisition.",
    "effect_size": "Overall Hedges' g=0.48, 95% CI [0.36, 0.60], p=.01; τ²=.5854; I²=93.72%. Egger intercept indicated asymmetry, B=-0.45, 95% CI [-0.78, -0.12], p<.0001; trim-and-fill g=0.44, 95% CI [0.32, 0.56].",
    "conditions_and_limits": "Mathematics, mostly well-structured tasks; complete correct examples were associated with larger effects. Practice versus initial-acquisition timing did not significantly moderate effects. Prior knowledge and dosage could not be synthesized because of inconsistent reporting.",
    "criticism": "Very high heterogeneity; effects ranged from -0.88 to 4.66; 93% of manuscripts used percent accuracy as the sole outcome; evidence of publication bias; about half the studies were US/English-only. A negative cross-study coefficient for self-explanation prompts is not a causal test. Does not test consequential decision-making or an AI persona.",
    "which_workflow": "WP-05—worked examples; direct for mathematics learning, indirect for higher-education decision-making.",
    "claimable_sentence": "Worked examples improve mathematics performance on average, but the effect is highly heterogeneous and the evidence is concentrated in structured tasks and immediate accuracy outcomes."
  },
  {
    "id": "BasuRoyMcMahon2012",
    "tier": "Tier 2",
    "full_citation": "Basu Roy, R., & McMahon, G. T. (2012). Video-based cases disrupt deep critical thinking in problem-based learning. Medical Education, 46(4), 426–435.",
    "link": "https://doi.org/10.1111/j.1365-2923.2011.04197.x",
    "peer_reviewed": "Yes",
    "population": "Four second-year medical tutorial groups in an endocrine and reproductive pathophysiology course; a prior cohort of 165 students and 18 tutors supplied preference data.",
    "design": "Randomized crossover comparison of one video-based and one text-based PBL case; 5,224 tutorial utterances were coded for depth of thinking by a blinded coder, with generalized estimating equations adjusting for groups, cases, and tutor participation.",
    "finding_verbatim": "Video-based material that refers to cases without dynamic physical signs is associated with a reduction in deep thinking.",
    "effect_size": "Deep versus superficial thinking, video versus text: OR=0.663, 95% CI [0.582, 0.754], p<.0001. Problem-exploration domain: OR=0.559, 95% CI [0.355, 0.882], p=.0125. Students preferring video: 59%; tutors: 78%.",
    "conditions_and_limits": "The cases involved psychosocial elements but no dynamic physical signs, so video had no necessary representational advantage. The outcome was coded tutorial discourse rather than later performance.",
    "criticism": "Only four groups, two cases, one course and one institution; a crossover at tutorial level; preference data came from a prior cohort. The study cannot establish that video is generally inferior where motion, affect, examination signs, or repeated practice matter.",
    "which_workflow": "WP-05—case presentation and adverse interface evidence; no AI test.",
    "claimable_sentence": "In one medical PBL crossover study, students preferred video cases, but video without dynamic physical signs elicited less deep discussion than text."
  },
  {
    "id": "Chernikova2020",
    "tier": "Tier 2",
    "full_citation": "Chernikova, O., Heitzmann, N., Stadler, M., Holzberger, D., Seidel, T., & Fischer, F. (2020). Simulation-based learning in higher education: A meta-analysis. Review of Educational Research, 90(4), 499–541.",
    "link": "https://doi.org/10.3102/0034654320933544",
    "peer_reviewed": "Yes",
    "population": "145 higher-education studies, 409 effects, N=10,532; overwhelmingly medical education (126 studies), with seven teacher-education and 12 other-domain studies.",
    "design": "Random-effects meta-analysis with robust-variance methods; eligible studies compared simulation with a control or baseline and used objective measures of complex skills.",
    "finding_verbatim": "The results of this meta-analysis show that simulation-based learning has large positive overall effects.",
    "effect_size": "Overall simulation versus control/baseline g=0.85, 95% CI [0.69, 1.02], k=145, I²=95.86%. Diagnostic skills g=0.82, 95% CI [0.41, 1.22], k=18, I²=92.02%; problem solving g=0.88 [0.68, 1.08], k=58; situation management g=0.72 [0.14, 1.31], k=21; role-play/standardized-patient format g=0.63 [0.38, 0.89], k=26, I²=87.02%.",
    "conditions_and_limits": "Simulations provided repeated practice of complex skills and almost always included feedback; scaffolding could include examples, prompts, and reflection. Support should be matched to learners' developing knowledge.",
    "criticism": "Extreme heterogeneity; mixed control types; medical education dominates. Simulation type, authenticity, technology, and expertise/scaffolding findings are mostly between-study moderators, not randomized head-to-head estimates. Prior knowledge was proxied by education level/context familiarity, and only about 4% of heterogeneity was explained. Presence-versus-absence comparisons did not show added value for particular scaffolds overall. It does not isolate simulated stakeholders or AI.",
    "which_workflow": "WP-05—guided simulation, role-play and diagnostic practice; direct mechanism, no AI-interface test.",
    "claimable_sentence": "Across a heterogeneous higher-education literature dominated by medicine, simulation improved objectively assessed complex skills; the evidence supports guided practice, not a particular stakeholder or agent interface."
  },
  {
    "id": "ChernikovaDiagnostic2020",
    "tier": "Tier 2",
    "full_citation": "Chernikova, O., Heitzmann, N., Fink, M. C., Timothy, V., Seidel, T., Fischer, F., & DFG Research Group COSIMA. (2020). Facilitating diagnostic competences in higher education—a meta-analysis in medical and teacher education. Educational Psychology Review, 32, 157–196.",
    "link": "https://doi.org/10.1007/s10648-019-09492-2",
    "peer_reviewed": "Yes",
    "population": "35 empirical studies, N=3,472, in medical and teacher higher education; 26 studies assigned an active professional role during all or part of learning.",
    "design": "Random-effects meta-analysis and moderator analysis of instructional supports for procedural diagnostic competence; 69% of included studies used random assignment.",
    "finding_verbatim": "Role-taking (taking an agent's role during the learning phase) had a significant positive effect on advancing diagnostic competences.",
    "effect_size": "Instructional support overall g=0.39, 95% CI [0.22, 0.56], I²=79.60%. Roles included g=0.49; the published table reports 95% CI [0.20, 0.58], n=2,472 across 26 studies, I²=74.65%, while the text reports SE=.11. Roles absent g=0.00, 95% CI [-0.18, 0.17]. Role moderator Q(1,33)=35.79 in the table, p<.001; the prose instead prints Q=19.09.",
    "conditions_and_limits": "Every role study used the active agent role—physician or teacher—and role-taking co-occurred with problem solving or other scaffolds. Primary studies measured procedural diagnostic competence; conceptual and strategic aspects were not reported separately.",
    "criticism": "Between-study moderator, not a randomized pooled role-versus-no-role contrast. Research on patient, student, and observer roles was lacking. The table's role confidence interval is difficult to reconcile with g=.49 and SE=.11, and the moderator Q differs between table and prose. Six of 38 p-curve results had insufficient evidential value and 10 had inadequate R-indices; substantial heterogeneity remained.",
    "which_workflow": "WP-05—role assignment and diagnostic practice; direct but bundled mechanism evidence, no AI-interface test.",
    "claimable_sentence": "Role assignment was associated with greater procedural diagnostic learning across medical and teacher-education interventions, but active roles co-occurred with other supports and the reported statistics contain inconsistencies."
  },
  {
    "id": "Colliver2008",
    "tier": "Tier 2",
    "full_citation": "Colliver, J. A., Kucera, K., & Verhulst, S. J. (2008). Meta-analysis of quasi-experimental research: Are systematic narrative reviews indicated? Medical Education, 42(9), 858–865.",
    "link": "https://doi.org/10.1111/j.1365-2923.2008.03144.x",
    "peer_reviewed": "Yes",
    "population": "The 11 studies contributing outcomes to the principles category of Gijbels et al. (2005).",
    "design": "Study-by-study methodological re-examination of design confounds, selection bias, intervention–outcome alignment, and effect-size calculations.",
    "finding_verbatim": "All 10 were subject to constant biases and confounds that favoured the intervention condition.",
    "effect_size": "No new pooled effect or confidence interval. Two extreme estimates reported as d=-8.291 and d=-7.91 were recalculated as d=-1.09 and d=-0.67 after correcting use of SEM instead of SD.",
    "conditions_and_limits": "The critique applies specifically to the studies behind Gijbels's strongest principles estimate and illustrates the consequences of pooling confounded quasi-experiments.",
    "criticism": "This is a validity audit, not an independent intervention trial, and plausible confounds do not demonstrate that the true PBL effect is zero. It focuses deliberately on a problematic subset.",
    "which_workflow": "WP-05—quality assessment for PBL evidence; no AI-interface evidence.",
    "claimable_sentence": "The strongest older PBL category relied predominantly on confounded quasi-experiments, and two extreme effect-size figures in the underlying synthesis were miscalculated."
  },
  {
    "id": "Duchatelet2019",
    "tier": "Tier 2",
    "full_citation": "Duchatelet, D., Gijbels, D., Bursens, P., Donche, V., & Spooren, P. (2019). Looking at role-play simulations of political decision-making in higher education through a contextual lens: A state-of-the-art. Educational Research Review, 27, 126–139.",
    "link": "https://doi.org/10.1016/j.edurev.2019.03.002",
    "peer_reviewed": "Yes",
    "population": "36 peer-reviewed higher-education studies published 1974–2016; 81% US, 72% undergraduate, one graduate-level study; mostly classes of 15–35 students.",
    "design": "Systematically searched state-of-the-art narrative review using SSCI and ERIC of face-to-face role-play simulations of political decision-making.",
    "finding_verbatim": "Learning outcomes have never been studied in relation to the simulation structure or the broader teaching-learning context.",
    "effect_size": "No pooled effect size or confidence interval was calculated. Findings were mapped narratively; the review characterized outcome findings as inconclusive.",
    "conditions_and_limits": "Included simulations assigned learners actual political or public-policy actor roles in a structured face-to-face decision process, generally with preparation, interaction, and teacher mediation.",
    "criticism": "The evidence base is narrow, US-heavy, and methodologically inconsistent. More than half of studies examined outcomes without connecting them to design/context; measures varied, and preparation may itself produce knowledge gains. Transfer and long-term outcomes were not studied. Screening was led by one author with peer debriefing; there was no pooled analysis or formal risk-of-bias synthesis. It cannot be generalized to online or AI role-play.",
    "which_workflow": "WP-05—political stakeholder role-play; direct descriptive evidence, causal decision improvement and AI cast not established.",
    "claimable_sentence": "Political role-play is widely used to rehearse complex decisions, but the reviewed literature could not establish which design features improve learning or whether effects transfer."
  },
  {
    "id": "Gijbels2005",
    "tier": "Tier 2",
    "full_citation": "Gijbels, D., Dochy, F., Van den Bossche, P., & Segers, M. (2005). Effects of problem-based learning: A meta-analysis from the angle of assessment. Review of Educational Research, 75(1), 27–61.",
    "link": "https://doi.org/10.3102/00346543075001027",
    "peer_reviewed": "Yes",
    "population": "Approximately 40 studies of problem-based learning, predominantly in higher and medical education.",
    "design": "Meta-analysis reclassifying assessments as understanding concepts, understanding principles linking concepts, or applying concepts and principles.",
    "finding_verbatim": "PBL had the most positive effects when the focal constructs being assessed were at the level of understanding principles that link concepts.",
    "effect_size": "Weighted ES: concepts=0.068, reported 95% CI half-width ±0.864; principles=0.795, ±0.782; application=0.339, ±0.662. Only the principles estimate was statistically significant.",
    "conditions_and_limits": "Results depended on what the assessment elicited; the positive result concerned organizing and relating concepts, not every form of applied performance.",
    "criticism": "Substantial overlap with Dochy et al. (2003), so this is not independent confirmation. Most evidence was quasi-experimental. Colliver et al. later found systematic validity threats in 10 of the 11 studies behind the principles category.",
    "which_workflow": "WP-05—case/problem-based learning; direct mechanism, AI expert cast indirect and untested.",
    "claimable_sentence": "An assessment-sensitive meta-analysis found PBL's clearest positive result for understanding principles linking concepts; concept and application estimates were not statistically secure."
  },
  {
    "id": "Hayashi2018",
    "tier": "Tier 3",
    "full_citation": "Hayashi, Y. (2018). The power of a 'maverick' in collaborative problem solving: An experimental investigation of individual perspective-taking within a group. Cognitive Science, 42(S1), 69–104.",
    "link": "https://doi.org/10.1111/cogs.12587",
    "peer_reviewed": "Yes",
    "population": "344 Japanese undergraduates recruited for two main laboratory experiments: 154 first-/second-year humanities students (Experiment 1; 127 analyzed after exclusions) and 190 second-/third-year psychology students (Experiment 2; 173 analyzed).",
    "design": "Controlled between-groups laboratory experiments using a simple visual rule-discovery task. Each student believed they were collaborating with five human partners; scripted conversational agents supplied zero, one, or three alternative perspectives. Experiment 2 crossed one-versus-three dissenting partners with positive-versus-negative group tone. Allocation method was not clearly reported.",
    "finding_verbatim": "Results show that when participants interacted with a maverick during the task, they were able to take others' perspectives.",
    "effect_size": "Experiment 1 final solution across 6:0, 5:1, and 3:3 groups: χ²(2)=8.342, p=.015, Cramér's V=.256; 5:1 versus 3:3: χ²(1)=0.019, p=.890, φ=.015. Perspective-taking dialogue: χ²(2)=11.763, p=.003, V=.304. In the 5:1 group, positive impression and successful integration were associated, χ²(1)=7.883, p=.005, φ=.433. Experiment 2: main effects of one versus three alternative perspectives, χ²(1)=10.650, p=.001, and positive versus negative tone, χ²(1)=6.275, p=.012; no interaction for final solution, p=.74. No confidence intervals reported.",
    "conditions_and_limits": "The alternative perspective had to be consistent and task-relevant; the student led each exchange; positive tone supported attention to the dissenter. The outcome was one short, artificial rule-discovery task, not retained learning or authentic professional decision quality.",
    "criticism": "No human-partner, no-agent, static-text, or alternative-interface control isolates the agents' contribution; agents were a delivery/control device and participants were deceived. Twenty-seven Experiment 1 recruits and 17 Experiment 2 recruits were excluded, including suspected nonhuman partners. One versus three dissenters did not differ in Experiment 1. An additional 62-student test found that being the lone maverick was frustrating. No independent direct replication was identified.",
    "which_workflow": "WP-05—multiple contrasting perspectives; direct mechanism test, AI expert-cast interface not tested.",
    "claimable_sentence": "In two tightly controlled undergraduate lab experiments, a consistent dissenting perspective could prompt perspective integration; the study did not test whether conversational agents or an AI cast improved learning."
  },
  {
    "id": "Maia2023",
    "tier": "Tier 2",
    "full_citation": "Maia, D., Andrade, R., Afonso, J., Costa, P., Valente, C., & Espregueira-Mendes, J. (2023). Academic performance and perceptions of undergraduate medical students in case-based learning compared to other teaching strategies: A systematic review with meta-analysis. Education Sciences, 13(3), 238.",
    "link": "https://doi.org/10.3390/educsci13030238",
    "peer_reviewed": "Yes",
    "population": "41 studies involving 7,667 undergraduate medical students.",
    "design": "Systematic review and random-effects meta-analysis; risk assessed with RoBANS and certainty with GRADE.",
    "finding_verbatim": "However, the certainty of evidence was very low.",
    "effect_size": "Exam scores SMD=2.37, 95% CI [1.25, 3.49], I²=94%. Interest/motivation SMD=0.79, 95% CI [0.13, 1.44]. After trim-and-fill added 11 studies, exam-score SMD=0.66, 95% CI [-0.77, 2.10], no longer significant.",
    "conditions_and_limits": "Findings concern undergraduate medical courses and mostly short-term academic outcomes; effects differed by discipline and comparator.",
    "criticism": "Extreme heterogeneity, very-low-certainty evidence, funnel-plot asymmetry, and significant Egger test, t=5.37, p<.001. The headline exam estimate did not survive the authors' publication-bias adjustment. Several disciplinary subgroups showed no significant advantage.",
    "which_workflow": "WP-05—case-based learning; direct, but decision-making and AI expert cast not established.",
    "claimable_sentence": "A recent medical-education meta-analysis reported higher case-based exam scores, but evidence was very uncertain and the estimate became nonsignificant after publication-bias adjustment."
  },
  {
    "id": "Nievelstein2013",
    "tier": "Tier 2",
    "full_citation": "Nievelstein, F., van Gog, T., van Dijck, G., & Boshuizen, H. P. A. (2013). The worked example and expertise reversal effect in less structured tasks: Learning to reason about legal cases. Contemporary Educational Psychology, 38(2), 118–125.",
    "link": "https://doi.org/10.1016/j.cedpsych.2012.12.004",
    "peer_reviewed": "Yes",
    "population": "75 first-year and 36 third-year law students at a Dutch university; third-years had completed several private-law courses but were not expert lawyers.",
    "design": "Randomized 2×2 within experience level: study worked legal-case examples versus solve cases, crossed with general Toulmin process steps versus no process steps; two learning cases and one immediate test case.",
    "finding_verbatim": "A worked example effect was found for both novice and advanced students, and no evidence for an expertise-reversal effect was found.",
    "effect_size": "No confidence intervals reported. First-years: worked M=58.63, SD=27.74 versus solve M=8.59, SD=4.06; F(1,68)=105.221, p<.001, Cohen's f=1.24. Third-years: M=82.28, SD=18.11 versus M=37.06, SD=26.71; F(1,32)=37.03, p<.001, f=1.02. Third-year nonparametric check Z=4.13, p<.001, r=.688.",
    "conditions_and_limits": "Less-structured but introductory civil-law cases; immediate reasoning test after two learning cases. Generic process steps had no significant performance effect or interaction.",
    "criticism": "Small single-university study, especially nine learners per third-year cell; paid volunteers; first-year outliers removed and heteroscedasticity required robustness checks; one immediate case; third-years were advanced students, not experts; material was set at first-year level.",
    "which_workflow": "WP-05—worked examples for legal reasoning; direct higher-education boundary study.",
    "claimable_sentence": "In one small law experiment, worked cases improved immediate legal-case reasoning for both first- and third-year students; expertise reversal did not appear."
  },
  {
    "id": "Tetzlaff2025",
    "tier": "Tier 2",
    "full_citation": "Tetzlaff, L., Simonsmeier, B., Peters, T., & Brod, G. (2025). A cornerstone of adaptivity – A meta-analysis of the expertise reversal effect. Learning and Instruction, 98, 102142.",
    "link": "https://doi.org/10.1016/j.learninstruc.2025.102142",
    "peer_reviewed": "Yes",
    "population": "5,924 learners in 60 experimental studies, 176 effect sizes, published 1992–2024 across primary, secondary, vocational, and higher education; higher education contributed 22 studies/28 effects in the educational-status moderator.",
    "design": "Cluster-robust random-effects meta-analysis of experiments comparing high- versus low-assistance instruction for learners grouped by domain-specific prior knowledge.",
    "finding_verbatim": "Low prior knowledge learners learn better from high-assistance instruction. High prior knowledge learners learn better from low-assistance instruction.",
    "effect_size": "Low-prior-knowledge learners: d=0.505, 95% CI [0.260, 0.750], I²=90.87%. High-prior-knowledge learners: d=-0.428, 95% CI [-0.647, -0.209], I²=87.55%. Difference-of-differences d=0.971, 95% CI [0.631, 1.312]. HE-versus-primary dummy coefficient: low prior knowledge 0.99 [0.34, 1.63], p=.003; high prior knowledge -0.09 [-0.63, 0.46], p=.749—these are not standalone HE subgroup effects.",
    "conditions_and_limits": "Domain-specific prior knowledge and degree of assistance must be considered together. The synthesis covers many assistance forms, not worked examples alone. Authors characterize the asymmetry as a stronger benefit from assisting novices than from withholding assistance from experts.",
    "criticism": "Very high heterogeneity; prior knowledge and assistance were categorized despite being continuous; educational status is partly confounded with age; several moderator cells were small; only PsycINFO and ERIC were searched and 32 eligible reports lacked usable data. Supports adaptation/fading, not a universal claim that examples harm every advanced learner or that a particular interface works.",
    "which_workflow": "WP-05—expertise reversal and adaptive guidance; direct for assistance-by-expertise interaction, indirect for a specific interface.",
    "claimable_sentence": "Instructional support should change with domain knowledge: pooled evidence favors more assistance for novices and less for advanced learners, with substantial between-study variation."
  },
  {
    "id": "Thistlethwaite2012",
    "tier": "Tier 2",
    "full_citation": "Thistlethwaite, J. E., Davies, D., Ekeocha, S., Kidd, J. M., MacDougall, C., Matthews, P., Purkis, J., & Clay, D. (2012). The effectiveness of case-based learning in health professional education: A BEME systematic review: BEME Guide No. 23. Medical Teacher, 34(6), e421–e444.",
    "link": "https://doi.org/10.3109/0142159X.2012.680939",
    "peer_reviewed": "Yes",
    "population": "104 reports in health-professional education; medicine was the most common discipline. Interventions ranged from a single case to year-long curricula.",
    "design": "Systematic review with narrative synthesis; 61% of reports used a single cohort and 75% collected post-intervention data only.",
    "finding_verbatim": "The short answer is there is little good and reliable evidence.",
    "effect_size": "No pooled effect size or confidence interval; heterogeneity and weak reporting prevented meta-analysis.",
    "conditions_and_limits": "Promising implementations generally used relevant and realistic cases, guided inquiry, small-group discussion, and explicit application of knowledge. Learners often preferred structured guidance to wholly open inquiry.",
    "criticism": "Only 23 reports were judged sufficiently strong and significant for detailed synthesis. Comparative studies generally found no significant difference from other methods. Case definitions, implementation, outcomes, sample sizes, and follow-up varied substantially; sustained behavioral evidence was scarce.",
    "which_workflow": "WP-05—case-based learning; direct, simulated stakeholders and AI expert cast not evaluated.",
    "claimable_sentence": "Case-based learning is feasible and often well received, but the principal systematic review found little reliable evidence that it outperformed alternative instruction."
  },
  {
    "id": "Thompson2000",
    "tier": "Tier 2",
    "full_citation": "Thompson, L., Gentner, D., & Loewenstein, J. (2000). Avoiding missed opportunities in managerial life: Analogical training more powerful than individual case training. Organizational Behavior and Human Decision Processes, 82(1), 60–75.",
    "link": "https://doi.org/10.1006/obhd.2000.2887",
    "peer_reviewed": "Yes",
    "population": "88 Masters of Management students in a 10-week negotiation course at Northwestern University's Kellogg School; analyzed as 44 negotiating dyads.",
    "design": "Students studied the same two structurally analogous but surface-dissimilar negotiation cases. The comparison condition derived a common principle; the advice condition analyzed each protagonist separately. One week later, dyads completed a face-to-face negotiation with an opportunity to use a contingency contract.",
    "finding_verbatim": "Management students in the Comparison condition were nearly three times more likely to transfer the principle.",
    "effect_size": "Contingency-contract use: 14/22 comparison dyads (64%) versus 5/22 advice dyads (23%), χ²(1,N=44)=7.503, p<.01; no confidence interval or standardized effect reported. Contract versus compromise joint profit: $1,049,500 versus $987,500, t(42)=2.484, p<.05. Comparison-versus-advice monetary gain was only 2% and nonsignificant.",
    "conditions_and_limits": "Both cases instantiated the same contingency-contract principle; students were explicitly asked to compare and extract it; the transfer negotiation occurred one week later.",
    "criticism": "Small, single-school, single-domain sample; dyadic outcome; no blinding; the manipulation bundles simultaneous comparison with an instruction to derive a principle. Behavioral transfer did not translate into a significant overall monetary advantage by instructional condition, and long-term workplace transfer was not tested.",
    "which_workflow": "WP-05—contrasting cases and decision transfer; direct mechanism, no AI-interface test.",
    "claimable_sentence": "In one management master's course, explicitly comparing two analogous cases increased use of the target strategy in a later face-to-face negotiation; separately analyzing both cases did not."
  },
  {
    "id": "VanGogRummel2010",
    "tier": "Tier 2",
    "full_citation": "van Gog, T., & Rummel, N. (2010). Example-based learning: Integrating cognitive and social-cognitive research perspectives. Educational Psychology Review, 22, 155–174.",
    "link": "https://doi.org/10.1007/s10648-010-9134-7",
    "peer_reviewed": "Yes",
    "population": "Narrative review of heterogeneous worked-example and modeling-example studies across ages, domains, and delivery modes; no pooled population count.",
    "design": "Non-exhaustive narrative review comparing cognitive worked-example research with social-cognitive modeling research.",
    "finding_verbatim": "Cognitive research has mainly focused on worked examples; social-cognitive research has mostly focused on modeling examples.",
    "effect_size": "No pooled effect size or confidence interval reported.",
    "conditions_and_limits": "Worked examples are typically written worked-out solutions; modeling examples involve observing an adult or peer performing a task, live or via media. The literatures overlap but test distinct instructional forms.",
    "criticism": "Explicitly non-exhaustive; heterogeneous tasks and learners prevent definitive quantitative conclusions; published before current generative-AI interfaces. It is definitional/integrative evidence, not an effect estimate.",
    "which_workflow": "WP-05—scope distinction between worked examples and expert modeling.",
    "claimable_sentence": "Written worked solutions and observed expert or peer demonstrations come from related but distinct traditions; efficacy claims should not be transferred silently between them."
  },
  {
    "id": "WalkerLeary2009",
    "tier": "Tier 2",
    "full_citation": "Walker, A., & Leary, H. (2009). A problem based learning meta analysis: Differences across problem types, implementation types, disciplines, and assessment levels. Interdisciplinary Journal of Problem-Based Learning, 3(1), 12–43.",
    "link": "https://doi.org/10.7771/1541-5015.1061",
    "peer_reviewed": "Yes",
    "population": "82 studies and 201 cognitive outcomes across disciplines and educational levels; medical and allied-health outcomes predominated.",
    "design": "Meta-analysis comparing PBL with lecture-based instruction and coding problem type, implementation, discipline, and assessment level.",
    "finding_verbatim": "Across 82 studies and 201 outcomes the findings favor PBL.",
    "effect_size": "Overall d=0.13 with reported 95% CI half-width ±0.025. Assessment-level estimates: concept d=-0.04, principles d=0.21, application d=0.33; reasoning-process outcomes d=0.49 ±0.23.",
    "conditions_and_limits": "More favorable estimates appeared for application-level assessment and fuller closed-loop implementations. Design and strategic-performance cells were very small.",
    "criticism": "Marked heterogeneity, Q=954.27. The analysis used 201 outcomes rather than statistically independent study-level effects; some studies contributed multiple outcomes. Implementation details were often missing and moderator cells could contain only a few studies. The review located no outcomes for decision-making problem types.",
    "which_workflow": "WP-05—case/problem-based learning; direct mechanism, decision-making indirect, AI interface untested.",
    "claimable_sentence": "Across varied settings, PBL had a small average advantage, with more favorable estimates on application than concept assessments, but heterogeneity and outcome non-independence limit precision."
  },
  {
    "id": "WittwerRenkl2010",
    "tier": "Tier 2",
    "full_citation": "Wittwer, J., & Renkl, A. (2010). How effective are instructional explanations in example-based learning? A meta-analytic review. Educational Psychology Review, 22(4), 393–409.",
    "link": "https://doi.org/10.1007/s10648-010-9136-5",
    "peer_reviewed": "Yes",
    "population": "Learners in 21 experimental publications yielding 28 comparisons across mathematics, science, instructional design, bookkeeping, and undergraduate medicine; aggregate N and education-stage breakdown were not reported.",
    "design": "Random-effects meta-analysis comparing worked examples plus instructional explanations with the same examples without explanations; searched through 2009. Authors used 90% confidence intervals because the literature was small.",
    "finding_verbatim": "The benefits of instructional explanations for example-based learning per se are minimal.",
    "effect_size": "Overall Cohen's d=0.16, 90% CI [0.03, 0.30], p=.04. Conceptual knowledge d=0.36 [0.17, 0.55]; near transfer d=0.10 [-0.04, 0.24]; far transfer d=0.23 [-0.02, 0.43], all 90% CIs. With prompted self-explanation in control, d=-0.01 [-0.28, 0.26].",
    "conditions_and_limits": "The intervention is explanatory text/information added to written worked examples, not live expert modeling. Benefit was clearer for conceptual knowledge than procedural knowledge or transfer; explanations added nothing when controls were prompted to self-explain.",
    "criticism": "Old, small synthesis; 90% rather than 95% CIs; heterogeneous tasks; many moderator cells small; studies missing statistics were excluded. It cannot establish effects of embodied experts, social personas, simulated stakeholders, or AI casts.",
    "which_workflow": "WP-05—expert explanation added to examples; direct for written explanations, not expert-interface claims.",
    "claimable_sentence": "Adding an instructional explanation to an example produced only a small average benefit, concentrated in conceptual knowledge rather than reliable transfer."
  },
  {
    "id": "XiaoFu2025",
    "tier": "Tier 3",
    "full_citation": "Xiao, J., & Fu, X. (2025). Is the use of standardized patients more effective than role-playing in medical education? A meta-analysis. Frontiers in Medicine, 12, 1601116.",
    "link": "https://doi.org/10.3389/fmed.2025.1601116",
    "peer_reviewed": "Yes",
    "population": "10 studies, 27 reported effects, N=721 medical/health learners from China, Germany, Indonesia, Korea, and the United States; five RCTs and five quasi-experiments.",
    "design": "Meta-analysis directly comparing trained human standardized patients with peer/student role-play; neither condition was a no-simulation control.",
    "finding_verbatim": "However, in other aspects, the two methods showed similar outcomes.",
    "effect_size": "Overall Hedges' g=0.073, 95% CI [-0.104, 0.250], p=.418, I²=74.248%, 27 effects. Self-confidence favored standardized patients: g=0.415 [0.18, 0.64], five effects, I²=0%. Communication g=0.200 [-0.52, 0.92]; general performance g=-0.089 [-0.67, 0.50]; knowledge g=0.221 [-0.21, 0.65]; professional competence g=-0.068 [-0.97, 0.84]; all nonsignificant.",
    "conditions_and_limits": "The contrast is trained human standardized patients versus learners enacting patient or clinician roles in health-professions scenarios. It tests relative delivery fidelity, not whether either method beats ordinary instruction or improves decision quality specifically.",
    "criticism": "Only 10 studies, mixed RCT/quasi-experimental designs, high overall heterogeneity, and very small outcome subgroups. No multilevel or robust adjustment for 27 effects nested in 10 studies is described. The article text incorrectly says p<.001 for the nonsignificant overall result although its table gives p=.418. Journal-article effects were null while thesis effects favored standardized patients.",
    "which_workflow": "WP-05—simulated stakeholders and role-play fidelity; direct comparison, AI expert cast not tested.",
    "claimable_sentence": "A small meta-analysis found no overall learning advantage for trained standardized patients over peer role-play; only self-confidence favored standardized patients, and objective outcomes were imprecise."
  },
  {
    "id": "Dochy2003",
    "tier": "Tier 2",
    "full_citation": "Dochy, F., Segers, M., Van den Bossche, P., & Gijbels, D. (2003). Effects of problem-based learning: A meta-analysis. Learning and Instruction, 13(5), 533–568.",
    "link": "https://doi.org/10.1016/S0959-4752(02)00025-7",
    "peer_reviewed": "Yes",
    "population": "43 empirical studies of problem-based learning in tertiary education in real classrooms; most were medical or health-professions curricula.",
    "design": "Meta-analysis of quasi-experimental comparisons of problem-based learning with conventional instruction.",
    "finding_verbatim": "There is a robust positive effect from PBL on the skills of students.",
    "effect_size": "Skills/application ES=+0.460 ±0.058. Knowledge ES=−0.223 ±0.058; the authors reported that two outliers drove the negative result, and removing them reduced the estimate to −0.107 ±0.058. These are the authors' reported uncertainty expressions, not confidence intervals reconstructed for this review.",
    "conditions_and_limits": "Effects varied with learner expertise, implementation scope, retention interval, assessment method and research design. The skills result concerned applying knowledge; the knowledge result was negative and explicitly judged non-robust.",
    "criticism": "No included study was randomized; heterogeneity and medical dominance limit generalization. A PBL curriculum bundles cases, facilitation, collaboration, study time and assessment and is not equivalent to a case interface or expert cast.",
    "which_workflow": "WP-05—case- and problem-organized learning; direct for tertiary PBL, not for a case label, interface or topology.",
    "claimable_sentence": "In a classic tertiary review, PBL favored skills or knowledge application, while its negative knowledge estimate depended on two outliers and was non-robust."
  },
  {
    "id": "Atkinson2000",
    "tier": "Tier 2",
    "full_citation": "Atkinson, R. K., Derry, S. J., Renkl, A., & Wortham, D. (2000). Learning from examples: Instructional principles from the worked examples research. Review of Educational Research, 70(2), 181–214.",
    "link": "https://doi.org/10.3102/00346543070002181",
    "peer_reviewed": "Yes",
    "population": "Experimental worked-example literature across mathematics and other problem-solving domains; learners were largely in early skill acquisition.",
    "design": "Narrative research review and instructional-design synthesis; no pooled meta-analytic estimate.",
    "finding_verbatim": "Worked examples are associated with early stages of skill development.",
    "effect_size": "No single pooled effect size or confidence interval reported.",
    "conditions_and_limits": "Benefits depend on integrating mutually referring information, using multiple examples, signalling deep structure, eliciting self-explanation and transitioning toward learner problem solving as knowledge develops.",
    "criticism": "Narrative rather than meta-analytic, with substantial laboratory and school evidence. It reviews worked solutions, not a demonstrated effect of an embodied expert, social persona, AI cast or interaction topology.",
    "which_workflow": "WP-05—worked-example mechanisms and expertise-sensitive transition design.",
    "claimable_sentence": "Worked examples can support early skill development when they expose conceptual structure, elicit processing and lead toward independent problem solving."
  },
  {
    "id": "Freeman2014",
    "tier": "Tier 2",
    "full_citation": "Freeman, S., Eddy, S. L., McDonough, M., Smith, M. K., Okoroafor, N., Jordt, H., & Wenderoth, M. P. (2014). Active learning increases student performance in science, engineering, and mathematics. Proceedings of the National Academy of Sciences, 111(23), 8410–8415.",
    "link": "https://doi.org/10.1073/pnas.1319030111",
    "peer_reviewed": "Yes",
    "population": "225 studies in undergraduate STEM; 158 contributed examination or concept-inventory effects and 67 contributed failure rates.",
    "design": "Meta-analysis comparing broadly defined active learning with traditional lecturing.",
    "finding_verbatim": "Student performance on examinations and concept inventories increased by 0.47 SDs under active learning.",
    "effect_size": "Mean examination/concept-inventory difference=0.47 SD. The odds of failure under traditional lecturing were 1.95 times the odds under active learning; average examination scores were about 6% higher under active learning.",
    "conditions_and_limits": "Undergraduate STEM; active learning included many intervention types and section-level instructional comparisons.",
    "criticism": "Interventions and study quality were heterogeneous, many studies were not randomized, and possible publication bias was assessed. The review does not isolate cases, PBL, role-play, expert modelling, AI delivery or topology.",
    "which_workflow": "WP-05—broad boundary evidence only; not an effect estimate for any focal case mechanism.",
    "claimable_sentence": "Broad undergraduate STEM active-learning evidence cannot be used as an effect estimate for cases, worked examples, role-play or expert casts."
  },
  {
    "id": "Bisra2018",
    "tier": "Tier 2",
    "full_citation": "Bisra, K., Liu, Q., Nesbit, J. C., Salimi, F., & Winne, P. H. (2018). Inducing self-explanation: A meta-analysis. Educational Psychology Review, 30(3), 703–725.",
    "link": "https://doi.org/10.1007/s10648-018-9434-x",
    "peer_reviewed": "Yes",
    "population": "64 research reports yielding 69 effects and approximately 5,917 learners across school, undergraduate and adult settings and multiple subjects.",
    "design": "Random-effects meta-analysis of prompted self-explanation during study or problem solving.",
    "finding_verbatim": "Self-explanation prompts are a potentially powerful intervention across a range of instructional conditions.",
    "effect_size": "Overall Hedges' g=0.55, 95% CI approximately [0.45, 0.65]. The three-effect visual pedagogical-agent subgroup was g=0.641, 95% CI [−0.018, 1.300], and therefore imprecise and compatible with no effect.",
    "conditions_and_limits": "Prompts induced explanations while studying or solving; effects varied by prompt format, knowledge type, task and educational level. Written and spoken explanations were combined.",
    "criticism": "Heterogeneous tasks and populations; agent subgroup contained only three effects and crossed zero. The overall estimate does not show that a visual agent, humanlike model, AI expert or case topology caused learning.",
    "which_workflow": "WP-05—active processing of examples and the boundary on pedagogical-agent inference.",
    "claimable_sentence": "Prompted self-explanation had a positive heterogeneous average, while the very small visual-agent subgroup was too imprecise to establish an interface effect."
  }
]
