[
  {
    "id": "AtkinsonRenklMerrill2003",
    "tier": "Tier 2",
    "full_citation": "Atkinson, R. K., Renkl, A., & Merrill, M. M. (2003). Transitioning from studying examples to solving problems: Effects of self-explanation prompts and fading worked-out steps. Journal of Educational Psychology, 95(4), 774–783.",
    "link": "https://doi.org/10.1037/0022-0663.95.4.774",
    "peer_reviewed": "Yes",
    "population": "Experiment 1: 78 psychology and educational-psychology undergraduates. Experiment 2: 40 high-school students enrolled in advanced algebra.",
    "design": "Two randomized computer-based experiments on probability problems; backward fading versus example-problem pairs and self-explanation principle prompts versus no prompts.",
    "finding_verbatim": "Across 2 experiments, this combination produced medium to large effects on near and far transfer without requiring additional time on task.",
    "effect_size": "Experiment 1 backward fading: near-transfer Cohen f=.23, far-transfer f=.27; prompting: near f=.25, far f=.23. Experiment 2 prompting within backward fading: near f=.42, far f=.37. No confidence intervals reported.",
    "conditions_and_limits": "Well-structured probability tasks; backward fading removed later solution steps first; prompts required selecting an underlying principle and supplied correctness feedback; outcomes were immediate.",
    "criticism": "Small, short studies; no confidence intervals; prompting was bundled with correctness feedback; the tasks predate LLMs and do not test open-ended Socratic dialogue or delayed retention.",
    "which_workflow": "AI teaching assistant—worked examples, fading, and self-explanation mechanisms; indirect technology evidence.",
    "claimable_sentence": "Two small randomized studies found that backward fading and principle-identification prompts with correctness feedback improved immediate near and far transfer in structured probability problems."
  },
  {
    "id": "Barbieri2023",
    "tier": "Tier 1",
    "full_citation": "Barbieri, C. A., Miller-Cotto, D., Clerjuste, S. N., & Chawla, K. (2023). A meta-analysis of the worked examples effect on mathematics performance. Educational Psychology Review, 35, 11.",
    "link": "https://doi.org/10.1007/s10648-023-09745-1",
    "peer_reviewed": "Yes",
    "population": "43 articles, 55 studies, and 181 effects spanning elementary through postsecondary mathematics; 53 randomized and two quasi-experimental studies.",
    "design": "Robust-variance meta-analysis of worked examples delivered by teachers, researchers, and computer systems.",
    "finding_verbatim": "Pairing examples with self-explanation prompts may not be a fruitful design modification.",
    "effect_size": "Overall Hedges g=.48, 95% CI [.36,.60], I²=93.72%; trim-and-fill g=.44, 95% CI [.32,.56]. Self-explanation-prompt moderator beta=-.24, SE=.11, p=.042; no CI reported for the moderator.",
    "conditions_and_limits": "Mathematics worked examples compared with problem solving or other instruction; examples were mostly used with middle-school students and older learners.",
    "criticism": "Extreme heterogeneity; effects ranged from negative to very large; Egger intercept indicated asymmetry. Most moderators were cross-study and noncausal. Only six articles examined faded examples, and prior-knowledge interactions varied.",
    "which_workflow": "AI teaching assistant—worked-example mechanism; direct pedagogy and indirect technology evidence.",
    "claimable_sentence": "Worked examples have a positive average effect in mathematics, but effects are highly heterogeneous and the synthesis does not establish that adding self-explanation prompts causes a larger benefit."
  },
  {
    "id": "Barcaui2025",
    "tier": "Tier 2",
    "full_citation": "Barcaui, A. (2025). ChatGPT as a cognitive crutch: Evidence from a randomized controlled trial on knowledge retention. Social Sciences & Humanities Open, 12, 102287.",
    "link": "https://doi.org/10.1016/j.ssaho.2025.102287",
    "peer_reviewed": "Yes",
    "population": "120 Brazilian undergraduate business students randomized 60/60; 85 completed the 45-day outcome (43 ChatGPT, 42 traditional study).",
    "design": "Randomized study of unrestricted ChatGPT-4 web access versus traditional non-AI study while preparing a short presentation; surprise proctored 20-item retention test after 45 days.",
    "finding_verbatim": "Students who used ChatGPT scored significantly lower on the retention test (57.5% correct) compared to those who studied traditionally (68.5% correct).",
    "effect_size": "57.5% versus 68.5%; t(83)=-3.19, p=.002, reported Cohen d=.68 in magnitude. No confidence interval reported for d or the mean difference.",
    "conditions_and_limits": "Unrestricted naturalistic ChatGPT use; a presentation-preparation task; one university; surprise multiple-choice outcome after 45 days.",
    "criticism": "29.2% attrition with complete-case analysis; AI and non-AI use were not fully logged; the AI group reported less study time; no guarded-tutor arm, preregistration, process mediation, or independent replication located.",
    "which_workflow": "AI teaching assistant—withdrawal and delayed-retention evidence.",
    "claimable_sentence": "In one randomized business-course study with substantial attrition, students assigned to unrestricted ChatGPT study scored lower on a surprise retention test 45 days later."
  },
  {
    "id": "Bassner2026",
    "tier": "Tier 1",
    "full_citation": "Bassner, P., Lenk-Ostendorf, B., Beinstingel, R., Wasner, T., & Krusche, S. (2026). Less stress, better scores, same learning: The dissociation of performance and learning in AI-supported programming education. Computers and Education: Artificial Intelligence, 10, 100537.",
    "link": "https://doi.org/10.1016/j.caeai.2025.100537",
    "peer_reviewed": "Yes",
    "population": "275 introductory-programming students at the Technical University of Munich.",
    "design": "Three-arm randomized controlled trial during one 90-minute concurrency exercise: Iris scaffolded hints withholding full solutions, unrestricted ChatGPT, or no-AI web resources; pre/post knowledge and post-support code-comprehension tests.",
    "finding_verbatim": "Despite these performance gains, neither AI condition produced greater pre–post knowledge gains or code-comprehension advantages.",
    "effect_size": "Knowledge time×group F(2,272)=0.258, p=.773, generalized eta squared=.0003; exercise performance F(2,272)=29.693, p<.001, generalized eta squared=.179. No numerical confidence intervals reported for these effects.",
    "conditions_and_limits": "One programming task; the scaffolded tutor supplied calibrated hints and withheld full solutions; ChatGPT could provide complete solutions; immediate outcomes only.",
    "criticism": "Single course and 90-minute task; no delayed retention; no teacher-authored solution or generic-versus-teacher-grounded contrast; scaffolding was a package and was not decomposed; no independent replication located.",
    "which_workflow": "AI teaching assistant—direct comparison of hint-first and unrestricted assistance; withdrawal evidence.",
    "claimable_sentence": "In one programming RCT, both AI conditions improved assisted exercise performance, but neither the hint-first tutor nor unrestricted ChatGPT improved immediate conceptual learning or code comprehension."
  },
  {
    "id": "Belland2017",
    "tier": "Tier 1",
    "full_citation": "Belland, B. R., Walker, A. E., Kim, N. J., & Lefler, M. (2017). Synthesizing results from empirical research on computer-based scaffolding in STEM education: A meta-analysis. Review of Educational Research, 87(2), 309–344.",
    "link": "https://doi.org/10.3102/0034654316670999",
    "peer_reviewed": "Yes",
    "population": "144 experimental studies and 333 cognitive outcomes across primary through adult STEM learners in ill-structured, problem-centered curricula.",
    "design": "Random-effects meta-analysis with composite outcomes, dependency sensitivity checks, outlier removal, and moderator analyses.",
    "finding_verbatim": "Computer-based scaffolding showed a consistently positive (g = 0.46) effect on cognitive outcomes across various contexts of use.",
    "effect_size": "Overall Hedges g=.46, z=18.19, p<.01, I²=69.7%. The pooled 95% CI is plotted but not stated numerically in the article text.",
    "conditions_and_limits": "Computer scaffolds supported ill-structured STEM problem solving; 82% of outcomes used context-specific scaffolds; only 16.5% included fading.",
    "criticism": "Predates LLMs; high heterogeneity; five outlier outcomes from three studies were removed; 64% of outcomes lacked reliability reporting. Fading, adjustment logic, and context specificity were not significant moderators, so the average cannot be attributed to those features.",
    "which_workflow": "AI teaching assistant—scaffolding and fading mechanisms; indirect technology evidence.",
    "claimable_sentence": "Computer-based STEM scaffolding has positive package-level evidence, but this synthesis did not find fading superior to adding, combining both, or leaving support unchanged."
  },
  {
    "id": "ContractorReyes2026",
    "tier": "Tier 2",
    "full_citation": "Contractor, Z., & Reyes, G. (2026). Experimental evidence on the learning impact of generative AI [Preprint]. arXiv.",
    "link": "https://doi.org/10.48550/arXiv.2607.08849",
    "peer_reviewed": "No—preprint",
    "population": "211 Middlebury College undergraduates attended the randomized first session; 204 returned approximately one week later.",
    "design": "Preregistered randomized laboratory experiment: unrestricted off-the-shelf GenAI access versus no AI with Google, Wikipedia, and library access during a fixed 35-minute unfamiliar-topic learning and essay task; unaided tests immediately and one week later.",
    "finding_verbatim": "AI access raises immediate test scores by 0.27 standard deviations. These gains persist one week later.",
    "effect_size": "Immediate ITT +.067 fraction correct, robust SE .032, or +.266 SD, SE .125. One-week ITT +.051, SE .023, or +.268 SD, SE .120. Numerical 95% CIs are shown graphically but not tabulated.",
    "conditions_and_limits": "Equal, proctored time on task; strong web-resource control; unfamiliar topics; 67.3% treatment take-up; short factual/conceptual tests and an analytical essay.",
    "criticism": "Selective single college, one 35-minute task, one-week horizon, small tests, preprint, and no replication. Test-rule violations rose 12.6 percentage points. Post-treatment augmentation/automation subgroups are nonrandom and not causal.",
    "which_workflow": "AI teaching assistant—generic assistance and withdrawal/transfer counterevidence.",
    "claimable_sentence": "A preregistered college preprint found a positive one-week unaided knowledge effect from unrestricted AI under tightly controlled equal-time conditions."
  },
  {
    "id": "Fan2025",
    "tier": "Tier 2",
    "full_citation": "Fan, Y., Tang, L., Le, H., Shen, K., Tan, S., Zhao, Y., Shen, Y., Li, X., & Gašević, D. (2025). Beware of metacognitive laziness: Effects of generative artificial intelligence on learning motivation, processes, and performance. British Journal of Educational Technology, 56(2), 489–530.",
    "link": "https://doi.org/10.1111/bjet.13544",
    "peer_reviewed": "Yes",
    "population": "117 Chinese university students from multiple disciplines; English was a second language; 70% female.",
    "design": "Randomized laboratory experiment comparing ChatGPT-4, a human writing expert, an analytics checklist, and no added support during a two-stage English reading-and-writing task; process mining plus knowledge and transfer tests.",
    "finding_verbatim": "ChatGPT can significantly improve short-term task performance, but it may not boost intrinsic motivation and knowledge gain and transfer.",
    "effect_size": "Essay improvement, AI minus control: 1.970 points, 95% CI [.083,3.858]; AI minus human expert: 2.120 [.191,4.049]. Transfer F=.019, p=.996, eta squared=.000; no numerical CI for the null omnibus effect.",
    "conditions_and_limits": "ChatGPT received reading material, requirements, rubric, and the learner's essay; one lab task; immediate knowledge and cross-domain transfer measures.",
    "criticism": "Small and gender-imbalanced sample; one task; variable missingness; no delayed follow-up. Metacognitive laziness was inferred from process patterns and not measured with a validated targeted instrument or identified as a causal mediator.",
    "which_workflow": "AI teaching assistant—performance/learning dissociation, metacognitive offloading, and transfer.",
    "claimable_sentence": "ChatGPT improved essay revision in one randomized study without improving knowledge or transfer, alongside process patterns consistent with—but not proof of—metacognitive offloading."
  },
  {
    "id": "Futterer2026",
    "tier": "Tier 1",
    "full_citation": "Fütterer, T., Bardach, L., Kuhn, J., Keller, S. D., & Gerjets, P. (2026). Enhancing school students' self-regulated learning through generative AI support: A randomized controlled trial. Educational Psychology Review, 38, 42.",
    "link": "https://doi.org/10.1007/s10648-026-10133-8",
    "peer_reviewed": "Yes",
    "population": "371 German students in grades 7–9 from secondary schools in Baden-Württemberg; physics and English lessons.",
    "design": "Individually randomized classroom RCT across six 45-minute sessions: GPT-4o utility-value reflection, GPT-4o cognitive-strategy/Socratic prompting, or light task conversation; all systems shared teacher- and expert-developed curriculum, tasks, solutions, and answer withholding.",
    "finding_verbatim": "No statistically significant advantages of either intervention over the control condition were found for effort, domain-specific knowledge, or elaboration-based strategy use.",
    "effect_size": "Domain-knowledge condition×time chi-square(1)=1.28, p=.257; strategy-use condition×time chi-square(2)=.08, p=.960. No numerical effect confidence intervals reported; figures display 95% intervals.",
    "conditions_and_limits": "Conditions differed mainly in theory-informed prompts; immediate posttest without the chatbot; four learning sessions in regular classes; all systems had the same grounding and answer policy.",
    "criticism": "35% attrition, convenience sample, immediate outcomes, a short domain test, no no-AI arm, and no independent replication. Exploratory engagement moderators are noncausal and sensitive to multiple testing.",
    "which_workflow": "AI teaching assistant—direct test of course-grounded Socratic/metacognitive prompting.",
    "claimable_sentence": "In a prompt-focused classroom RCT, course-grounded Socratic and metacognitive prompting did not improve immediate knowledge or strategy performance over a minimally pedagogical GPT condition."
  },
  {
    "id": "Kosmyna2025",
    "tier": "Contested",
    "full_citation": "Kosmyna, N., Hauptmann, E., Yuan, Y. T., Situ, J., Liao, X.-H., Beresnitzky, A. V., Braunstein, I., & Maes, P. (2025). Your brain on ChatGPT: Accumulation of cognitive debt when using an AI assistant for essay writing task [Preprint]. arXiv.",
    "link": "https://doi.org/10.48550/arXiv.2506.08872",
    "peer_reviewed": "No—preprint",
    "population": "54 adults, 18 per LLM, search-engine, and unaided group for three sessions; 18 total completed the fourth crossover session.",
    "design": "Multi-session randomized essay-writing study using EEG connectivity, NLP and human/AI essay scoring, recall of one's own text, and self-reported ownership; crossover of LLM and unaided groups in session four.",
    "finding_verbatim": "In session 4, LLM-to-Brain participants showed reduced alpha and beta connectivity, indicating under-engagement.",
    "effect_size": "No single standardized educational-learning effect or confidence interval reported; the paper reports many EEG connections, linguistic comparisons, and p values.",
    "conditions_and_limits": "Essay writing across four months; EEG during task performance; no curriculum knowledge test or validated learning-transfer outcome.",
    "criticism": "Small groups and crossover, tiny topic cells, extensive multiplicity, unclear reproducibility, inconsistent labels and scores, and no replication. A scholarly commentary argues that fewer significant EEG connections do not establish lower absolute neural activity or impaired learning ability.",
    "which_workflow": "AI teaching assistant—cognitive-debt hypothesis and offloading context only.",
    "claimable_sentence": "An unreviewed, small EEG essay study raises hypotheses about reduced engagement with LLM assistance but does not establish cognitive debt as a learning effect."
  },
  {
    "id": "LearnLMTeam2025",
    "tier": "Tier 3",
    "full_citation": "LearnLM Team Google & Eedi. (2025). AI tutoring can safely and effectively support students: An exploratory RCT in UK classrooms [Preprint]. arXiv.",
    "link": "https://doi.org/10.48550/arXiv.2512.23633",
    "peer_reviewed": "No—industry/platform-authored preprint",
    "population": "165 students aged 13–15 in Years 9–10 across five UK secondary schools; 17 qualified expert tutors.",
    "design": "Exploratory randomized trial on the Eedi mathematics platform: static hints versus interactive tutoring, then session-level human tutor versus LearnLM with every drafted AI message reviewed, edited, or replaced by a human tutor.",
    "finding_verbatim": "Students guided by LearnLM performed at least as well as students chatting with human tutors on each learning outcome we measured.",
    "effect_size": "Next-unit transfer: supervised LearnLM 66.2%, 95% credible interval [61.1,71.2], versus human tutoring 60.7% [55.8,65.4]; ATE +5.5 percentage points, 95% credible interval [-1.4,12.4].",
    "conditions_and_limits": "LearnLM was pedagogically fine-tuned, received question text, incorrect response and misconception explanations, and was prompted Socratically. Every student-facing AI message had expert human review; 74.4% were accepted unchanged.",
    "criticism": "Preprint authored by the model and platform teams; short, one domain, no generic autonomous-AI arm, next-unit rather than delayed transfer, AI-versus-human interval includes zero, and possible tutor/session contamination. No independent replication located.",
    "which_workflow": "AI teaching assistant—human-supervised, grounded Socratic tutoring; lower-certainty direct evidence.",
    "claimable_sentence": "In an industry-authored preprint, a pedagogically tuned and fully human-supervised AI tutor performed at least as well as human tutoring on measured outcomes; its estimated transfer advantage over humans remained uncertain."
  },
  {
    "id": "Roll2011",
    "tier": "Tier 2",
    "full_citation": "Roll, I., Aleven, V., McLaren, B. M., & Koedinger, K. R. (2011). Improving students' help-seeking skills using metacognitive feedback in an intelligent tutoring system. Learning and Instruction, 21(2), 267–280.",
    "link": "https://doi.org/10.1016/j.learninstruc.2010.07.004",
    "peer_reviewed": "Yes",
    "population": "Study 1: 58 students in grades 10–11. Study 2: 67 vocational students in grades 10–11.",
    "design": "Two studies of a Help Tutor layered onto the Geometry Cognitive Tutor; Study 1 balanced assignment, Study 2 nonrandom class assignment and a four-month support/withdrawal sequence.",
    "finding_verbatim": "Both studies did not show improvement to students' domain learning while receiving help-seeking support.",
    "effect_size": "Study 1 faulty hint requests d=1.51 and reaching the bottom-out hint d=1.07, with no posttest or geometry-learning-gain difference. Study 2 supported-month hint-error d=1.20 and 1.47; no confidence intervals reported.",
    "conditions_and_limits": "Hint sequence moved from orientation to instrumental question to governing rule to bottom-out answer; the intervention discouraged premature bottom-out use rather than prohibiting answers; longer exposure preceded behavior persistence.",
    "criticism": "Study 2 was nonrandom and bundled a video and self-assessment tutor; behavior persisted in the same tutoring environment, but domain learning was not measured during withdrawal; pre-LLM geometry setting; no confidence intervals.",
    "which_workflow": "AI teaching assistant—hint sequencing, bottom-out answers, metacognitive help seeking, and fading/withdrawal behavior.",
    "claimable_sentence": "Metacognitive tutoring changed how students requested hints and reduced premature bottom-out use, but it did not improve measured geometry learning."
  },
  {
    "id": "Stankovic2026",
    "tier": "Contested",
    "full_citation": "Stanković, M., Hirche, E., Kollatzsch, S., & Doetsch, J. N. (2026). Comment on: Your brain on ChatGPT: Accumulation of cognitive debt when using an AI assistant for essay writing tasks [Preprint]. arXiv.",
    "link": "https://doi.org/10.48550/arXiv.2601.00856",
    "peer_reviewed": "No—preprint commentary",
    "population": "Methodological commentary on Kosmyna et al.'s N=54 study and N=18 crossover session; no new participant sample.",
    "design": "Scholarly reappraisal of study design, statistical reporting, EEG interpretation, reproducibility, and transparency.",
    "finding_verbatim": "Some results by Kosmyna et al. (2025) could be interpreted more conservatively.",
    "effect_size": "No new effect size or confidence interval; this is a methodological commentary rather than a reanalysis with a pooled estimate.",
    "conditions_and_limits": "Addresses the cognitive-debt preprint's methods and claims; identifies small topic cells, multiplicity, omitted effect sizes, reporting inconsistencies, and EEG-interpretation problems.",
    "criticism": "The commentary is itself an unreviewed preprint and does not independently replicate or reanalyze the underlying raw data; it challenges inference rather than estimating a corrected effect.",
    "which_workflow": "AI teaching assistant—criticism record for the cognitive-debt claim.",
    "claimable_sentence": "A scholarly preprint commentary identifies substantial design, reproducibility, EEG-analysis, and reporting concerns that require conservative interpretation of the cognitive-debt study."
  },
  {
    "id": "Steindl2025",
    "tier": "Tier 2",
    "full_citation": "Steindl, S., Brunner, F., Sissouno, N., Schwagerl, D., Schöler-Niewiera, F., & Schäfer, U. (2025). On the effectiveness of prompt-moderated LLMs for math tutoring at the tertiary level. In Findings of the Association for Computational Linguistics: EMNLP 2025 (pp. 11310–11323).",
    "link": "https://doi.org/10.18653/v1/2025.findings-emnlp.605",
    "peer_reviewed": "Yes—conference findings",
    "population": "49 volunteer first-semester engineering and computer-science mathematics students: no AI n=17, generic GPT n=17, prompt-moderated tutor n=15.",
    "design": "Randomized same-day review, assisted practice, and unassisted examination using GPT-4o-mini; moderated prompt instructed incremental guidance and no direct answers but lacked teacher-authored solutions.",
    "finding_verbatim": "When the assistance was removed again, both LLM groups performed better than the control group, contradicting concerns about shallow learning.",
    "effect_size": "Unassisted exam means: no AI 24.3, generic GPT 38.2, moderated tutor 29.5. No exam-specific inferential estimate or CI. Pooled-phase Tukey no-AI minus GPT mean-difference 95% CI [.162,20.38], p=.046.",
    "conditions_and_limits": "One 40-minute tertiary-mathematics practice period followed by a same-day exam; volunteer sample; generic and moderated interfaces used the same small model.",
    "criticism": "Very small sample; near and immediate transfer; no exam-specific test, no delayed retention, no human or teacher-grounded arm, occasional solution leakage, differential frustration and task exposure, and no independent replication.",
    "which_workflow": "AI teaching assistant—generic versus prompt-moderated tutoring and withdrawal.",
    "claimable_sentence": "A very small tertiary-mathematics trial descriptively favored generic GPT on a same-day unassisted exam, but did not report an exam-specific inferential comparison."
  },
  {
    "id": "Tetzlaff2025",
    "tier": "Tier 1",
    "full_citation": "Tetzlaff, L., Simonsmeier, B. A., Peters, T., & Brod, G. (2025). A cornerstone of adaptivity—A meta-analysis of the expertise reversal effect. Learning and Instruction, 98, 102142.",
    "link": "https://doi.org/10.1016/j.learninstruc.2025.102142",
    "peer_reviewed": "Yes",
    "population": "60 experimental studies, 176 effect sizes, and 5,924 participants across primary, secondary, higher, vocational, and adult education.",
    "design": "PRISMA-guided meta-analysis of domain-prior-knowledge by instructional-assistance interactions, accounting for dependent effects.",
    "finding_verbatim": "Providing additional assistance to learners with low domain knowledge appears to have stronger effects than withholding assistance from experts.",
    "effect_size": "Low prior knowledge, high versus low assistance: d=.505, 95% CI [.260,.750], I²=90.87%. High prior knowledge: d=-.428, 95% CI [-.647,-.209], I²=87.55%. Difference-of-differences d=.971 [.631,1.312].",
    "conditions_and_limits": "Expertise was domain-specific prior knowledge; instructional assistance varied widely; effects were moderated by educational status, domain, and prior-knowledge assessment.",
    "criticism": "Very high heterogeneity; continuous knowledge and assistance were dichotomized; sparse moderator cells; educational status is partly confounded with age and domain; the synthesis is not GenAI-specific and cannot identify mechanisms.",
    "which_workflow": "AI teaching assistant—adaptivity, assistance level, fading, and expertise reversal.",
    "claimable_sentence": "Assistance should be contingent on domain knowledge: higher support helped low-knowledge learners on average, while the direction reversed among high-knowledge learners, with substantial heterogeneity."
  },
  {
    "id": "Vanzo2025",
    "tier": "Tier 2",
    "full_citation": "Vanzo, A., Pal Chowdhury, S., & Sachan, M. (2025). GPT-4 as a homework tutor can improve student engagement and learning outcomes. In Proceedings of the 63rd Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (pp. 31119–31136).",
    "link": "https://doi.org/10.18653/v1/2025.acl-long.1502",
    "peer_reviewed": "Yes—conference paper",
    "population": "76 enrolled Italian technical-high-school students in four English classes taught by one teacher; the descriptive table totals 75 students.",
    "design": "Eight-week within-class stratified randomization to teacher-specified GPT-4 homework tutoring or traditional homework; tutor used stepwise questioning and a no-answer instruction; teacher-made pre/post multiple-choice tests.",
    "finding_verbatim": "We also observe significant improvement in 3rd year students in learning as measured by tests, while 5th year students maintain their performances.",
    "effect_size": "Pooled learning gain d=.251, one-sided p=.156; third-year subgroup d=.603, one-sided p=.044; fifth-year subgroup d=-.004, p=.505. No confidence intervals reported.",
    "conditions_and_limits": "Teacher supplied the task and purpose; GPT generated an instructional plan; treatment students interacted at home; neither condition was prohibited from private ChatGPT use.",
    "criticism": "Tiny subgroups, one-sided tests, no multiplicity correction for three learning tests, one teacher/school, contamination, no delayed retention, test-format misalignment, reporting inconsistency, and answer leakage despite the prompt. Words typed—not assignment—predicted gains in the authors' regression.",
    "which_workflow": "AI teaching assistant—teacher-specified homework tutoring, answer withholding, and engagement.",
    "claimable_sentence": "A small teacher-specified homework trial found no pooled learning advantage, one positive grade subgroup, and no effect in the other grade."
  },
  {
    "id": "Xue2026",
    "tier": "Tier 2",
    "full_citation": "Xue, H., Lin, C., Xie, B., Fu, M., Jiang, L., Sui, Y., Wu, X., & Xu, N. (2026). More than scores: AI-assisted instruction in long-term knowledge retention and critical thinking skills for diagnostic education. Medical Science Educator. Advance online publication.",
    "link": "https://doi.org/10.1007/s40670-026-02830-4",
    "peer_reviewed": "Yes",
    "population": "84 fourth-year medical students, 42 per condition.",
    "design": "Cluster-assigned quasi-experiment comparing DeepSeek-supported case-based learning with textbooks and guidelines; equal 120-minute sessions, two feedback rounds, baseline, immediate, and six-month theory tests; mixed models with class random intercept and false-discovery control.",
    "finding_verbatim": "GenAI-assisted learning was associated with improvements in critical thinking, self-directed learning, and satisfaction, and showed promise for supporting knowledge retention.",
    "effect_size": "At six months: symptomatology adjusted difference 2.1, 95% CI [.5,3.7], q=.015; laboratory/instrumental 2.4 [.2,4.6], q=.042; physical-examination knowledge null, q=.335.",
    "conditions_and_limits": "Grounded case-based diagnostic instruction; equal scheduled time; two feedback rounds; six-month follow-up; the immediate knowledge outcomes did not differ.",
    "criticism": "Quasi-random cluster assignment, small single center and likely few classes, multi-component intervention, self-reported higher-order skills, no generic-versus-grounded contrast, and no independent replication.",
    "which_workflow": "AI teaching assistant—longer-term retention from grounded case-based learning.",
    "claimable_sentence": "A small medical-education quasi-experiment found six-month advantages in two of three knowledge domains after grounded AI case-based learning, with one domain null."
  },
  {
    "id": "Zhao2026",
    "tier": "Tier 3",
    "full_citation": "Zhao, C., Zhu, J., Liu, J., Zhao, W., & Pang, Y. (2026). Effectiveness of a generative AI-powered digital tutor integrated with a knowledge graph in anatomy education for nursing students: A randomized controlled trial. BMC Medical Education, 26, 1026.",
    "link": "https://doi.org/10.1186/s12909-026-09469-0",
    "peer_reviewed": "Yes",
    "population": "362 first-year nursing students in six intact classes at one Chinese medical college; 301 complete cases analyzed.",
    "design": "Six classes assigned to AI plus knowledge-graph adaptive tutoring, the same knowledge graph with instructor tutoring, or traditional lectures for 16 weeks; one-month follow-up retention ratio and case inference outcomes.",
    "finding_verbatim": "blended teaching model integrating a generative AI-powered digital tutor with a knowledge graph significantly improved nursing students' academic performance, knowledge retention, and clinical reasoning ability",
    "effect_size": "One-month retention: AI 82.3% versus instructor/knowledge graph 71.5%; mean difference 10.8 percentage points, 95% CI [7.4,14.2]; overall eta squared=.25. Reported intervals do not account for class clustering.",
    "conditions_and_limits": "Course-limited tutor with knowledge graph, personalized preview, real-time questions, adaptive feedback, and review plans; same knowledge graph in instructor-tutoring arm; one semester and one-month follow-up.",
    "criticism": "Only six randomized clusters, but ordinary individual-level ANOVA ignored clustering, creating pseudoreplication and overprecise intervals. Also 17% attrition, complete-case analysis, bundled AI/knowledge graph/adaptive practice, no generic arm, no AI-response audit, and no replication.",
    "which_workflow": "AI teaching assistant—course-grounded digital tutoring and one-month retention; very-low-certainty positive evidence.",
    "claimable_sentence": "A six-class anatomy trial reported higher one-month retention with an AI/knowledge-graph package, but its student-level analysis ignored class randomization and materially overstates precision."
  },
  {
    "id": "Kestin2025",
    "tier": "Tier 1",
    "full_citation": "Kestin, G., Miller, K., Klales, A., Milbourne, T., & Ponti, G. (2025). AI tutoring outperforms in-class active learning: an RCT introducing a novel research-based design in an authentic educational setting. Scientific Reports, 15, 17458.",
    "link": "https://doi.org/10.1038/s41598-025-97652-6",
    "peer_reviewed": "Yes",
    "population": "Introductory physics for life sciences; 194 eligible undergraduates; Harvard University, USA.",
    "design": "Cluster-randomized crossover RCT; two topics over two consecutive weeks; immediate pre/post-tests.",
    "finding_verbatim": "Students learn significantly more in less time when using the AI tutor.",
    "effect_size": "Linear-regression estimate 0.63 SD; quantile-regression estimate 0.73–1.3 SD; median post-test 4.5 vs 3.5; p<10^-8.",
    "conditions_and_limits": "GPT-4; expert-crafted question-specific prompts; pre-written solutions; instructional videos; identical worksheets; understanding/applying/analyzing outcomes; immediate tests only.",
    "criticism": "Single selective institution and course; two lessons; no delayed retention; attrition from 233 enrolled to 194 eligible; ceiling-effect adjustment produces a wide estimate; no generic-AI arm; authors designed/taught the intervention.",
    "which_workflow": "AI teaching assistant—direct; case/worked-example scaffolding—indirect.",
    "claimable_sentence": "In one randomized Harvard physics study, an instructor-authored, carefully scaffolded AI tutor produced higher immediate post-test performance than in-class active learning."
  },
  {
    "id": "Bastani2025",
    "tier": "Tier 1",
    "full_citation": "Bastani, H., Bastani, O., Sungu, A., Ge, H., Kabakcı, Ö., & Mariman, R. (2025). Generative AI without guardrails can harm learning: Evidence from high school mathematics. Proceedings of the National Academy of Sciences, 122(26), e2422633122.",
    "link": "https://doi.org/10.1073/pnas.2422633122",
    "peer_reviewed": "Yes",
    "population": "Nearly 1,000 students in grades 9–11; mathematics; one private high school in Turkey.",
    "design": "Preregistered cluster RCT across four sessions; control vs generic GPT-4 vs teacher-grounded guarded GPT-4; assisted practice followed by unassisted exams.",
    "finding_verbatim": "When access is subsequently taken away, students actually perform worse than those who never had access.",
    "effect_size": "Practice: GPT Base +0.137/1 (48%) and GPT Tutor +0.361/1 (127%) vs control. Unassisted exam: GPT Base −0.054/1 (−17%); GPT Tutor −0.004, not significant.",
    "conditions_and_limits": "GPT Tutor prompt contained teacher-designed hints and correct solutions and withheld direct answers; outcomes were session-level math practice/exams.",
    "criticism": "K–12, not higher education; one school; guarded tutor prevented harm but did not improve unassisted learning; short duration; classroom-level randomization.",
    "which_workflow": "AI teaching assistant—direct design contrast; Socratic withholding—direct.",
    "claimable_sentence": "A large classroom experiment found that generic AI reduced subsequent unassisted performance, while teacher-designed hints and guardrails eliminated that harm."
  },
  {
    "id": "KulikFletcher2016",
    "tier": "Tier 1",
    "full_citation": "Kulik, J. A., & Fletcher, J. D. (2016). Effectiveness of intelligent tutoring systems: A meta-analytic review. Review of Educational Research, 86(1), 42–78.",
    "link": "https://doi.org/10.3102/0034654315581420",
    "peer_reviewed": "Yes",
    "population": "50 controlled evaluations across educational levels, subjects, and intelligent tutoring systems.",
    "design": "Meta-analysis of controlled evaluations.",
    "finding_verbatim": "The median effect of intelligent tutoring in the 50 evaluations was to raise test scores 0.66 standard deviations.",
    "effect_size": "Median ES=0.66 SD; authors equate this with a move from the 50th to 75th percentile.",
    "conditions_and_limits": "Effects depended strongly on alignment: locally developed tests yielded larger effects than standardized tests; adequate implementation and conventional comparison groups mattered.",
    "criticism": "Predates generative AI; heterogeneous populations and systems; not specific to higher education or instructor-configured LLMs; median is not a pooled causal effect for a new product.",
    "which_workflow": "AI teaching assistant—direct mechanism, indirect technology.",
    "claimable_sentence": "Across 50 controlled evaluations, intelligent tutoring systems improved test performance relative to conventional instruction, with results strongly dependent on test alignment and implementation."
  },
  {
    "id": "VanLehn2011",
    "tier": "Tier 1",
    "full_citation": "VanLehn, K. (2011). The relative effectiveness of human tutoring, intelligent tutoring systems, and other tutoring systems. Educational Psychologist, 46(4), 197–221.",
    "link": "https://doi.org/10.1080/00461520.2011.611369",
    "peer_reviewed": "Yes",
    "population": "Experimental comparisons of human tutoring, computer tutoring, and matched no-tutoring instruction; mixed educational settings.",
    "design": "Quantitative research review.",
    "finding_verbatim": "This review did not confirm these beliefs.",
    "effect_size": "Human tutoring d=0.79; intelligent tutoring systems d=0.76; answer-based systems approximately d=0.31 vs no tutoring.",
    "conditions_and_limits": "Comparisons attempted to hold content and instructional time constant; effect varied with interaction granularity.",
    "criticism": "Not a formal modern meta-analysis; small sets for some comparisons; heterogeneous tutors and outcomes; predates LLMs; does not establish equivalence in every context.",
    "which_workflow": "AI teaching assistant—direct mechanism; Bloom critique.",
    "claimable_sentence": "A quantitative review found substantially smaller tutoring effects than Bloom’s 2σ benchmark and similar average effects for human and step-based computer tutoring."
  },
  {
    "id": "Wisniewski2020",
    "tier": "Tier 2",
    "full_citation": "Wisniewski, B., Zierer, K., & Hattie, J. (2020). The power of feedback revisited: A meta-analysis of educational feedback research. Frontiers in Psychology, 10, 3087.",
    "link": "https://doi.org/10.3389/fpsyg.2019.03087",
    "peer_reviewed": "Yes",
    "population": "435 studies; 994 effects; more than 61,000 learners across educational levels and domains.",
    "design": "Random-effects meta-analysis.",
    "finding_verbatim": "Feedback cannot be understood as a single consistent form of treatment.",
    "effect_size": "Overall d=0.48; significant heterogeneity.",
    "conditions_and_limits": "Effects were larger for cognitive and motor outcomes and depended substantially on the information content of feedback.",
    "criticism": "Broad mixture of populations, feedback types, designs, and outcomes; includes Hattie as coauthor but uses study-level meta-analysis; not specific to AI or timely feedback.",
    "which_workflow": "AI teaching assistant—timely formative feedback; case practice—expert feedback.",
    "claimable_sentence": "Information-rich feedback can improve learning, but its impact varies substantially with what the feedback communicates."
  },
  {
    "id": "Atkinson2000",
    "tier": "Tier 2",
    "full_citation": "Atkinson, R. K., Derry, S. J., Renkl, A., & Wortham, D. (2000). Learning from examples: Instructional principles from the worked examples research. Review of Educational Research, 70(2), 181–214.",
    "link": "https://doi.org/10.3102/00346543070002181",
    "peer_reviewed": "Yes",
    "population": "Experimental worked-example literature across mathematics and other problem-solving domains; learners largely in early skill acquisition.",
    "design": "Narrative research review and design-principle synthesis.",
    "finding_verbatim": "Worked examples are associated with early stages of skill development.",
    "effect_size": "No single pooled effect size reported.",
    "conditions_and_limits": "Benefits depend on integrated components, multiple examples, signaling deep structure, self-explanation, and learner expertise; support should fade as expertise grows.",
    "criticism": "Narrative rather than meta-analytic; much laboratory and school research; expertise-reversal limits; expert modeling by an AI persona was not tested.",
    "which_workflow": "Case study/AI expert cast—expert modeling and worked examples, indirect.",
    "claimable_sentence": "Worked examples can support early skill development when they expose conceptual structure and are adapted as learners gain expertise."
  },
  {
    "id": "Bisra2018",
    "tier": "Tier 2",
    "full_citation": "Bisra, K., Liu, Q., Nesbit, J. C., Salimi, F., & Winne, P. H. (2018). Inducing self-explanation: A meta-analysis. Educational Psychology Review, 30(3), 703–725.",
    "link": "https://doi.org/10.1007/s10648-018-9434-x",
    "peer_reviewed": "Yes",
    "population": "64 research reports; 69 effects; approximately 5,917 learners across school, undergraduate, and adult settings and multiple subjects.",
    "design": "Random-effects meta-analysis.",
    "finding_verbatim": "Self-explanation prompts are a potentially powerful intervention across a range of instructional conditions.",
    "effect_size": "Overall Hedges g=0.55, 95% CI approximately [0.45, 0.65].",
    "conditions_and_limits": "Prompts induced explanations while studying or problem solving; effects varied with prompt format, knowledge type, task, and educational level.",
    "criticism": "Mixed age groups and tasks; not specifically oral or AI-mediated; possible publication dependence and heterogeneity; explaining aloud was not isolated from written self-explanation.",
    "which_workflow": "Voice assessment—self-explanation; case practice—direct mechanism.",
    "claimable_sentence": "Prompting learners to explain relationships and reasoning improves learning across a range of study and problem-solving conditions."
  },
  {
    "id": "Bloom1984",
    "tier": "Contested",
    "full_citation": "Bloom, B. S. (1984). The 2 sigma problem: The search for methods of group instruction as effective as one-to-one tutoring. Educational Researcher, 13(6), 4–16.",
    "link": "https://doi.org/10.3102/0013189X013006004",
    "peer_reviewed": "Yes",
    "population": "Selected mastery-learning and tutoring studies, including small studies conducted by Bloom’s students.",
    "design": "Narrative synthesis and problem statement; not a modern meta-analysis.",
    "finding_verbatim": "The average tutored student was above 98% of the students in the control class.",
    "effect_size": "Claimed approximately 2 SD for mastery-based one-to-one tutoring vs conventional instruction in selected studies.",
    "conditions_and_limits": "Intensive one-to-one tutoring combined with mastery learning and corrective feedback; historical, selected comparisons.",
    "criticism": "The 2σ figure is not a replicable general tutoring estimate; selection, small samples, bundled treatment, and historical controls limit inference. VanLehn found human tutoring d=0.79; other syntheses are smaller.",
    "which_workflow": "AI teaching assistant—historical context only.",
    "claimable_sentence": "Bloom framed an influential challenge about scaling mastery tutoring, but later reviews report much smaller typical effects."
  }
]
