[
  {
    "id": "Bisra2018",
    "tier": "Tier 2",
    "full_citation": "Bisra, K., Liu, Q., Nesbit, J. C., Salimi, F., & Winne, P. H. (2018). Inducing self-explanation: A meta-analysis. Educational Psychology Review, 30(3), 703–725.",
    "link": "https://doi.org/10.1007/s10648-018-9434-x",
    "peer_reviewed": true,
    "population": "64 research reports yielding 69 effects and approximately 5,917 learners across school, undergraduate, and adult settings and multiple subjects.",
    "design": "Random-effects meta-analysis of prompted self-explanation during study or problem solving.",
    "finding_verbatim": "Self-explanation prompts are a potentially powerful intervention across a range of instructional conditions.",
    "effect_size": "Overall Hedges' g=.55, with a 95% confidence interval approximately [.45, .65].",
    "conditions_and_limits": "Effects varied by prompt format, knowledge type, task, and educational level. The evidence combined written and spoken explanation rather than isolating an oral mechanism.",
    "criticism": "Heterogeneous tasks and populations; speech was not isolated, and the synthesis addresses learning outcomes rather than the validity or reliability of an oral-assessment score.",
    "which_workflow": "Voice assessment—self-explanation learning evidence and the boundary between prompted activity and score interpretation.",
    "claimable_sentence": "Prompted self-explanation produced a positive average learning effect across heterogeneous conditions, but the synthesis does not isolate speech or establish oral-assessment validity."
  },
  {
    "id": "FoxEricssonBest2011",
    "tier": "Tier 1",
    "full_citation": "Fox, M. C., Ericsson, K. A., & Best, R. (2011). Do procedures for verbal reporting of thinking have to be reactive? A meta-analysis and recommendations for best reporting methods. Psychological Bulletin, 137(2), 316–344.",
    "link": "https://doi.org/10.1037/a0021663",
    "peer_reviewed": true,
    "population": "94 studies and nearly 3,500 participants completing diverse cognitive and performance tasks with concurrent verbalization or a matched silent condition.",
    "design": "Meta-analysis comparing concurrent neutral think-aloud, directed description or explanation, and matched silent performance.",
    "finding_verbatim": "procedures that entail describing or explaining thoughts and actions are significantly reactive, leading to higher performance than silent control conditions.",
    "effect_size": "Neutral think-aloud accuracy effect r=-.03, 95% CI [-.10, .03]. Directed description or explanation current-task performance effect r=.23, 95% CI [.14, .31]. All verbal-report procedures tended to increase completion time.",
    "conditions_and_limits": "The near-zero result applies to neutral concurrent reporting of information already in attention, not to why-questions, explanations, retrospective reports, or examiner-directed probing. Outcomes concerned performance while verbalizing, not subsequent retention or transfer.",
    "criticism": "Tasks and populations were heterogeneous and predominantly experimental rather than higher-education assessments. The review does not establish oral-assessment validity, durable learning, or the effects of interactive examiner follow-ups.",
    "which_workflow": "Voice assessment—direct evidence on measurement reactivity; self-explanation—distinguishes elicitation effects from learning.",
    "claimable_sentence": "No average accuracy effect was detected for neutral think-aloud, whereas directed requests to explain changed current performance and cannot be treated as passive measurement."
  },
  {
    "id": "GrevingRichter2022",
    "tier": "Tier 2",
    "full_citation": "Greving, S., & Richter, T. (2022). Practicing retrieval in university teaching: Short-answer questions are beneficial, whereas multiple-choice questions are not. Journal of Cognitive Psychology, 34(5), 657–674.",
    "link": "https://doi.org/10.1080/20445911.2022.2085281",
    "peer_reviewed": true,
    "population": "53 second-semester undergraduates in a German introductory psychology lecture; 74% women and 98% psychology majors, with modeled short-answer and multiple-choice comparisons based on 43 participants each.",
    "design": "Experiment embedded in five weekly lecture sessions. Information units were assigned to short-answer retrieval, difficult multiple-response multiple-choice retrieval, or restudy; review occupied the final ten minutes, no feedback was given, and retention was measured on an unannounced test in Session 10.",
    "finding_verbatim": "Answering short-answer questions only once and without feedback benefits retention of learning compared to restudying key points of the lecture.",
    "effect_size": "Short-answer retrieval increased predicted probability correct by 17 percentage points, 95% credible interval [5.63, 27.00], OR=2.06. Multiple-choice retrieval increased it by 1 point, 95% credible interval [-9.48, 11.48], OR=1.04; model coefficient beta=-0.54, 95% credible interval [-1.33, 0.18], BF01=5.06.",
    "conditions_and_limits": "The benefit occurred after one short-answer retrieval opportunity without feedback and was strongest for low-retrievability items. The criterion test used alternate but closely related questions.",
    "criticism": "Small, selective, high-achieving, predominantly psychology sample; voluntary participation and variable attendance; unannounced researcher test rather than a graded course examination; one course and near-transfer outcome. It demonstrates format dependence rather than a universal benefit from any quiz.",
    "which_workflow": "Voice assessment—retrieval-practice mechanism; question-format boundary.",
    "claimable_sentence": "In one authentic psychology lecture, a single short-answer retrieval opportunity improved later retention relative to restudy, while difficult multiple-choice retrieval did not."
  },
  {
    "id": "GurungBurns2019",
    "tier": "Tier 2",
    "full_citation": "Gurung, R. A. R., & Burns, K. (2019). Putting evidence-based claims to the test: A multi-site classroom study of retrieval practice and spaced practice. Applied Cognitive Psychology, 33(5), 732–743.",
    "link": "https://doi.org/10.1002/acp.3507",
    "peer_reviewed": true,
    "population": "351 introductory-psychology students at nine colleges and universities; the reported exam-score regression used approximately 206 linked cases.",
    "design": "Multi-site classroom comparison in which course implementations varied the amount of quiz repetition and the spacing between quizzes; outcomes included authentic class examinations and a common standardized psychology test.",
    "finding_verbatim": "In practice, closer attention needs to be paid to quiz and exam content overlap and the duration within which repetition is used.",
    "effect_size": "GPA accounted for R-squared=.05, F(1,204)=10.84, p=.001. Retrieval, spacing, and their interaction added R-squared=.28, F(3,201)=27.38, p<.001. No standardized simple-effect estimate or confidence interval was reported in the accessible primary record.",
    "conditions_and_limits": "Results depended jointly on spacing and repetition. Students with greater spacing, and those in the low-spacing/low-repetition condition, outscored the low-spacing/high-repetition condition; quiz–exam content overlap and timing mattered.",
    "criticism": "Students were not individually randomized across a common implementation; courses, sites, instructors, quiz schedules, and perceived difficulty differed. Linked outcome data were substantially smaller than the enrolled sample, and the design cannot isolate retrieval amount from spacing, instructor, site, or content alignment.",
    "which_workflow": "Voice assessment—retrieval-practice implementation complication; authentic multi-site course evidence.",
    "claimable_sentence": "A nine-campus field study did not support a simple more-quizzing-is-better rule: spacing and quiz–exam alignment materially changed the pattern."
  },
  {
    "id": "HardersEbersbach2026",
    "tier": "Tier 1",
    "full_citation": "Harders, B., & Ebersbach, M. (2026). No causal self-explanation effect for factual knowledge. Applied Cognitive Psychology, 40(3), e70174.",
    "link": "https://doi.org/10.1002/acp.70174",
    "peer_reviewed": true,
    "population": "Final N=208 adults, including 193 university students; mean age 23.1 years. Of 250 initial participants, 42 did not complete the two-week test. Participants were randomized to written self-explanation (n=111) or rereading (n=97).",
    "design": "Preregistered randomized active-control laboratory experiment with equal 15-minute study periods using facts about a fictional ecosystem; immediate and two-week delayed multiple-choice tests.",
    "finding_verbatim": "self-explanation itself had no effect on factual learning compared to re-reading, neither immediately nor after a delay.",
    "effect_size": "Learning-strategy main effect F(1,205)=2.29, p=.132; strategy-by-retention interaction F(1,205)=2.74, p=.099. The authors reported no standardized treatment effect or confidence interval for these null effects. The retention-interval effect was F(1,205)=236.17, p<.001, eta-squared=.27, but this is forgetting over time, not a self-explanation effect. Within the self-explanation group, correct text-based inferences predicted aggregate test performance, b=.26, SE=.07, p<.001; no CI was reported.",
    "conditions_and_limits": "The comparison equated study time and controlled prior knowledge. Explanations were written and outcomes assessed factual recall, not spoken reasoning, transfer, or oral-assessment performance.",
    "criticism": "Immediate scores showed a ceiling effect; 42 initial participants were lost before the delayed test. Materials were fictional and laboratory-based. Reliability for coding false inferences was poor, r=.47, so that category was excluded.",
    "which_workflow": "Self-explanation—strong counterevidence for factual learning; indirect to oral assessment.",
    "claimable_sentence": "A preregistered active-control study did not detect an advantage of written self-explanation for immediate or two-week factual recall."
  },
  {
    "id": "Huxham2012",
    "tier": "Tier 2",
    "full_citation": "Huxham, M., Campbell, F., & Westwood, J. (2012). Oral versus written assessments: A test of student performance and attitudes. Assessment & Evaluation in Higher Education, 37(1), 125–136.",
    "link": "https://doi.org/10.1080/02602938.2010.515012",
    "peer_reviewed": true,
    "population": "Two biology cohorts at Edinburgh Napier University: 99 students randomized between modes and 29 students completing both modes.",
    "design": "Randomized response-mode comparison plus a within-student comparison and an attitude study using comparable biology questions delivered orally or in writing.",
    "finding_verbatim": "Both cohorts showed highly significant differences in the mean marks achieved, with better performance in the oral assessment.",
    "effect_size": "Higher oral marks were reported in both cohorts; the accessible source record did not report a standardized effect or confidence interval.",
    "conditions_and_limits": "The study compared performance modes at one institution. It did not assign an oral learning intervention or measure later retention or transfer.",
    "criticism": "Higher marks do not identify the more valid mode or establish more learning. Clarification, response demands, examiner interaction, anxiety, and writing burden remain plausible explanations.",
    "which_workflow": "Voice assessment—direct evidence that response mode changes observed performance; boundary on learning and validity claims.",
    "claimable_sentence": "Students in two biology cohorts earned higher marks orally than in writing, a mode difference that does not by itself establish learning or validity."
  },
  {
    "id": "IpeirotisRizakos2026",
    "tier": "Tier 3",
    "full_citation": "Ipeirotis, P., & Rizakos, K. (2026). Scalable and personalized oral assessments using voice AI. Authors' version identifying a Communications of the ACM publication, arXiv:2603.18221v3; assigned DOI 10.1145/3831714.",
    "link": "https://arxiv.org/abs/2603.18221",
    "peer_reviewed": false,
    "population": "Two undergraduate AI/ML Product Management cohorts at NYU Stern: 36 students in Fall 2025 and 37 in Spring 2026.",
    "design": "End-to-end field deployment as a final examination. Voice AI conducted personalized project and case questioning; Claude, Gemini, and GPT-5 independently scored transcripts on five 0–4 rubric dimensions, deliberated after seeing peer ratings and evidence, and a chair model synthesized the grade. Instructor and teaching assistant also graded all 36 Fall examinations, but the comparison was informal.",
    "finding_verbatim": "This comparison was informal, not a controlled validation study.",
    "effect_size": "Dimension-pooled ordinal Krippendorff's alpha increased from .52 to .86 in Fall 2025 and from .65 to .90 in Spring 2026 after AI deliberation. Overall 0–20 score alpha after deliberation was .83 and .95, respectively. No confidence intervals were reported. More-stressful-than-written responses were 83% among 30 of 36 Fall students and 63% among 32 of 37 Spring students; these are respondent proportions and had no confidence intervals. No blinded same-rubric faculty agreement or decision-accuracy coefficient was reported.",
    "conditions_and_limits": "Results depend on the named model versions, a five-dimension course-specific rubric, automatic transcript evidence, a two-round multi-model council, chair synthesis, and human audit of high-disagreement cases. The audit trigger flagged about 3% of Spring exams; this is an operational guardrail, not validation. The 2026-07-26 author version says it is published in Communications of the ACM and supplies DOI 10.1145/3831714, but the publisher DOI/Crossref record remained inaccessible on 2026-08-05; the accessible version is therefore recorded as not independently verified as peer reviewed.",
    "criticism": "One course, institution, and instructor team; no control group; no blinded same-rubric faculty comparison; instructor grading was holistic and confounded by knowledge of teaching gaps; no subgroup fairness analysis; no end-to-end assessment of speech-recognition error; no confidence intervals. Models exchanged ratings before high post-deliberation agreement was calculated, so the coefficient does not test independence or stability across reruns, model updates, cases, or occasions. The system also showed question stacking, silence interruptions, non-random case selection, voice-related stress, and stable model-leniency differences.",
    "which_workflow": "Voice assessment—direct automated-administration and scoring evidence; boundary on validity.",
    "claimable_sentence": "A 2026 NYU field deployment reports high post-deliberation inter-model agreement on oral-exam transcripts, but no blinded same-rubric faculty comparison or full validity argument."
  },
  {
    "id": "Kane2013",
    "tier": "Tier 1",
    "full_citation": "Kane, M. T. (2013). Validating the interpretations and uses of test scores. Journal of Educational Measurement, 50(1), 1–73.",
    "link": "https://doi.org/10.1111/jedm.12000",
    "peer_reviewed": true,
    "population": "Conceptual and methodological analysis of score validation across educational and psychological measurement settings; no participant sample.",
    "design": "Argument-based validity framework specifying the inferences and assumptions that connect observed test responses to score interpretations and uses.",
    "finding_verbatim": "It is the proposed score interpretations and uses that are validated and not the test or the test scores.",
    "effect_size": "No effect size or confidence interval; this is a measurement framework rather than an intervention study.",
    "conditions_and_limits": "The framework requires evidence proportionate to each proposed interpretation and use. It does not prescribe one universal oral-assessment design or validate any particular rubric or technology.",
    "criticism": "Argument structure makes assumptions visible but does not itself supply empirical evidence for scoring consistency, domain representation, extrapolation, consequences, or fairness.",
    "which_workflow": "Voice assessment—foundational validity framework for separating elicitation, scoring, interpretation, and use.",
    "claimable_sentence": "An oral format is not valid in the abstract; evidence must support the specific inferences and uses attached to its scores."
  },
  {
    "id": "LarsenButlerRoediger2013",
    "tier": "Tier 2",
    "full_citation": "Larsen, D. P., Butler, A. C., & Roediger, H. L., III. (2013). Comparative effects of test-enhanced learning and self-explanation on long-term retention. Medical Education, 47(7), 674–682.",
    "link": "https://doi.org/10.1111/medu.12141",
    "peer_reviewed": true,
    "population": "47 first-year medical students completing weekly activities on four course-relevant topics.",
    "design": "Within-participant randomized assignment of topics to testing plus self-explanation, testing alone, review plus self-explanation, or study; six-month free-recall test including clinical application questions.",
    "finding_verbatim": "Testing had a more robust effect on long-term retention and application than self-explanation.",
    "effect_size": "Testing main effect eta-squared=.33; self-explanation main effect eta-squared=.08. Six-month comparisons collapsed over topics: testing plus explanation 40% versus explanation 29%, d=.70, p=.001; testing 36% versus explanation 29%, d=.48, p=.02; explanation 29% versus study 20%, d=.68, p=.001. No confidence intervals were reported.",
    "conditions_and_limits": "Course-relevant medical material; repeated weekly activity; six-month free recall and clinical application. Students completed activities outside normal class for payment, and review/explanation conditions had answer materials.",
    "criticism": "Small sample; topic-by-intervention interaction; outside-course paid participation; written rather than spoken self-explanation; answer-key and sample-explanation access complicate the mechanism; no confidence intervals.",
    "which_workflow": "Voice assessment—retrieval and self-explanation learning mechanisms, indirect to oral assessment.",
    "claimable_sentence": "In one small medical-course experiment, testing showed a more robust six-month benefit than self-explanation, while self-explanation still exceeded study."
  },
  {
    "id": "Nallaya2024",
    "tier": "Tier 2",
    "full_citation": "Nallaya, S., Gentili, S., Weeks, S., & Baldock, K. (2024). The validity, reliability, academic integrity and integration of oral assessments in higher education: A systematic review. Issues in Educational Research, 34(2), 629–646.",
    "link": "https://www.iier.org.au/iier34/nallaya.pdf",
    "peer_reviewed": true,
    "population": "17 peer-reviewed higher-education studies published from 2010 through 2021 across multiple disciplines and countries.",
    "design": "PRISMA systematic review with narrative synthesis of validity, reliability, academic integrity, and integration evidence for oral assessment.",
    "finding_verbatim": "Oral assessments can be both valid and reliable.",
    "effect_size": "No pooled effect size or confidence interval; the review narratively summarized study-level evidence and conditions.",
    "conditions_and_limits": "Recurring conditions included clear criteria, authentic alignment, assessor training, moderation, practice, and inclusive design.",
    "criticism": "Only 17 heterogeneous studies met criteria; the review did not pool effects or provide a formal study-level risk-of-bias synthesis, and many integrity claims were perceptual or local.",
    "which_workflow": "Voice assessment—direct synthesis of higher-education oral-assessment validity and reliability evidence.",
    "claimable_sentence": "Oral assessment can support valid and reliable interpretations under defined design conditions; the review does not establish validity as an inherent property of the format."
  },
  {
    "id": "NieminenMorinaBiagiotti2024",
    "tier": "Tier 2",
    "full_citation": "Nieminen, J. H., Moriña, A., & Biagiotti, G. (2024). Assessment as a matter of inclusion: A meta-ethnographic review of the assessment experiences of students with disabilities in higher education. Educational Research Review, 42, 100582.",
    "link": "https://doi.org/10.1016/j.edurev.2023.100582",
    "peer_reviewed": true,
    "population": "42 qualitative higher-education studies published 2010–2022, comprising 868 students with diverse disabilities across multiple countries.",
    "design": "Meta-ethnographic synthesis of disabled students' reported assessment experiences, organized around access, participation, inclusion, and exclusion.",
    "finding_verbatim": "Overall, our review shows that assessment is a primary barrier to the inclusion of students with disabilities in higher education.",
    "effect_size": "No quantitative effect size or confidence interval; 40 of 42 studies reported experiences of exclusion, 22 of 42 reported access-related inclusion, and 5 of 42 reported participation-related inclusion. These are counts of studies or themes, not prevalence estimates.",
    "conditions_and_limits": "Addresses assessment broadly—especially examinations and accommodations—not oral or viva assessment specifically. Evidence concerns lived experience, implementation, and access rather than causal effects on marks.",
    "criticism": "Qualitative theme counts cannot estimate prevalence or mitigation efficacy; English-language restriction, heterogeneous disability and assessment contexts, and variable primary-study quality. It cannot establish that oral assessment is inherently more or less accessible.",
    "which_workflow": "Voice assessment—accessibility and disability, indirect boundary evidence.",
    "claimable_sentence": "Assessment formats and accommodations can both enable and exclude disabled students; accessibility must be designed around the target construct and access needs, not assumed from format."
  },
  {
    "id": "Peixoto2017",
    "tier": "Tier 3",
    "full_citation": "Peixoto, J. M., Mamede, S., de Faria, R. M. D., de Moura, A. S., Santos, S. M. E., & Schmidt, H. G. (2017). The effect of self-explanation of pathophysiological mechanisms of diseases on medical students' diagnostic performance. Advances in Health Sciences Education, 22(5), 1183–1197.",
    "link": "https://doi.org/10.1007/s10459-017-9757-2",
    "peer_reviewed": true,
    "population": "39 fourth-year medical students: 20 assigned to pathophysiological self-explanation and 19 to solve cases without self-explanation.",
    "design": "Randomized experiment using six training cases and six novel same-syndrome diagnostic cases one week later.",
    "finding_verbatim": "Self-explanation of pathophysiology did not improve students' diagnostic performance for all diseases.",
    "effect_size": "No overall phase effect, p=.34; no overall condition effect, p=.10; no phase-by-condition interaction, p=.42. A post hoc syndrome interaction was p=.022; within the self-explanation group, jaundice improved, p=.035, while chest-pain cases did not. No standardized effect sizes or confidence intervals were reported.",
    "conditions_and_limits": "The proposed benefit was limited to diseases that may share a pathophysiological mechanism; diagnostic transfer was tested one week later. Explanations were generated during case solving, not used as an oral assessment score.",
    "criticism": "Very small single-program sample; the syndrome-specific finding was post hoc; no standardized effect or interval; clinical-domain task; not a spoken-versus-written comparison and not evidence of assessment validity.",
    "which_workflow": "Voice assessment—self-explanation learning counterevidence, indirect to oral scoring.",
    "claimable_sentence": "A small randomized medical study found no general diagnostic benefit from pathophysiological self-explanation and only a post hoc syndrome-specific pattern."
  },
  {
    "id": "Rasalkar2025",
    "tier": "Tier 3",
    "full_citation": "Rasalkar, K., Tripathy, S., Sinha, S., Mukherjee, B., Takkella, N., Dadel, E. V., Sundriyal, M., & Prasad, S. (2025). Enhancing medical assessment strategies: A comparative study between structured, traditional and hybrid viva-voce assessment. BMC Medical Education, 25, 835.",
    "link": "https://doi.org/10.1186/s12909-025-07428-9",
    "peer_reviewed": true,
    "population": "151 first-year medical students in biochemistry at one Indian medical college; four faculty examiners.",
    "design": "Within-student observational comparison: two sequential examiner rounds, each containing a five-minute structured and five-minute traditional viva; the hybrid score was their sum; paired score comparisons, correlations, alpha, and one-way intraclass correlation.",
    "finding_verbatim": "Low consistency in scoring between examiners for both structured and traditional viva-voce, with statistically significant differences in scores across different sets of examiners.",
    "effect_size": "Overall structured r=.424, alpha=.595, ICC=.511; traditional r=.458, alpha=.626, ICC=.551; hybrid r=.496, alpha=.663, ICC=.591. Set-2 hybrid alpha=.729 and ICC=.685. No confidence intervals were reported.",
    "conditions_and_limits": "Expert-reviewed cards and trained examiners; each student encountered different cards, raters, and occasions in fixed sequence; the hybrid score mechanically aggregates structured and traditional scores.",
    "criticism": "Rater, question, order, and occasion effects are confounded; hybrid reliability can improve through aggregation alone; one site and no external criterion. The text says only hybrid alpha exceeded .70 although overall hybrid alpha was .663 and only set 2 was .729. Structure alone did not produce high reliability.",
    "which_workflow": "Voice assessment—adverse reliability evidence, direct.",
    "claimable_sentence": "In one medical cohort, neither structured nor traditional viva scores showed high inter-examiner reliability, and the combined score remained only moderate overall."
  },
  {
    "id": "RenNguyenBernackiYuGreene2026",
    "tier": "Tier 3",
    "full_citation": "Ren, S., Nguyen, H., Bernacki, M. L., Yu, L., & Greene, J. A. (2026). Using large language models for automated coding of self-regulated learning think-aloud protocol data. Journal of Learning Analytics, Early Access Articles, 1–24.",
    "link": "https://doi.org/10.18608/jla.2026.9025",
    "peer_reviewed": true,
    "population": "98 undergraduates aged 17–26: 49 enrolled in introductory biology and 49 in precalculus. Each completed a 90-minute controlled laboratory session; the evaluation sample comprised 600 balanced, human-coded utterances selected from 10,391 transcribed verbalizations.",
    "design": "Factorial binary-classification benchmark. Professional transcripts of concurrent think-aloud were coded by three research assistants; for each of six self-regulated-learning codes and each of two tasks, researchers randomly sampled 25 code-positive and 25 code-negative utterances. GPT-4o, Claude 3.5 Sonnet, and Gemini 1.5 Pro classified each utterance under six zero- or few-shot prompt conditions, producing 10,800 observations analyzed with binomial mixed models.",
    "finding_verbatim": "These findings highlighted both the promise and limitations of LLMs for scaling SRL research.",
    "effect_size": "Accuracy across conditions ranged from .49 to .90. Mathematics accuracy was M=.78 (SD=.41), versus biology M=.56 (SD=.50): beta=2.39, SE=.30, OR=10.93, 95% CI [6.04, 19.78], p<.001. Relative to Claude 3.5 Sonnet, GPT-4o OR=.81, 95% CI [.68, .96], p=.016, and Gemini 1.5 Pro OR=.39, 95% CI [.33, .46], p<.001. The overall few-shot effect was not significant, OR=.90, 95% CI [.79, 1.02], p=.111; for sub-goal setting, few-shot underperformed zero-shot, OR=.56, 95% CI [.41, .75], Bonferroni-adjusted p<.001.",
    "conditions_and_limits": "The results apply to six binary self-regulated-learning process codes, deliberately balanced positive and negative examples, professional human transcription, 2024 model versions, and the tested prompts in controlled mathematics and biology tasks. Participants were instructed to verbalize without explaining their mental processes. The study did not test automated speech recognition, spoken self-explanation, authentic graded oral examinations, instructor-rubric scores, pass/fail decisions, or learning outcomes.",
    "criticism": "Human reference coding was itself imperfect: kappa=.65 in mathematics and .66 in biology, with 74.0% and 79.2% agreement; no confidence intervals were reported. Balanced 50% code prevalence makes raw accuracy unlike natural classroom prevalence. Performance varied substantially by task, code, model, and prompting, and no subgroup-fairness analysis was reported. The benchmark does not establish validity for automated assessment scoring, and no independent replication was identified by the 2026-08-05 cutoff.",
    "which_workflow": "Voice assessment—adjacent evidence on automated coding of transcribed verbalizations; a boundary case, not validation of AI scoring against instructor rubrics.",
    "claimable_sentence": "In one laboratory benchmark, LLM classification of professionally transcribed think-aloud utterances varied by task, code, model, and prompt; it did not validate grades against instructor rubrics."
  },
  {
    "id": "Ringeisen2019",
    "tier": "Tier 3",
    "full_citation": "Ringeisen, T., Lichtenfeld, S., Becker, S., & Minkley, N. (2019). Stress experience and performance during an oral exam: The role of self-efficacy, threat appraisals, anxiety, and cortisol. Anxiety, Stress, & Coping, 32(1), 50–66.",
    "link": "https://doi.org/10.1080/10615806.2018.1528528",
    "peer_reviewed": true,
    "population": "92 psychology students, ages 20–36, at one German university; 18 had non-German cultural backgrounds and all had C2 German proficiency; cortisol analysis had fewer complete cases.",
    "design": "Within-person observational study of a mandatory, authentic 30-minute oral exam, with anxiety and salivary cortisol measured on a matched control day and repeatedly before and after the exam; ANCOVA and latent-growth models.",
    "finding_verbatim": "Compared to a control day, the level of either stress response was higher on the exam day.",
    "effect_size": "Anxiety: F(2,83)=8.59, p<.01, partial eta-squared=.09; cortisol: F(1,54)=19.17, p<.01, partial eta-squared=.26; confidence intervals were not reported. Better grade was associated with a steeper anxiety decline, beta=-.52, SE=.07, p<.05; grade was unrelated to anxiety level, threat, self-efficacy, or cortisol in the final model.",
    "conditions_and_limits": "Authentic exam with the same examiner, common topics but adaptive questions, and intraindividual control-day measurement; findings identify stress around this exam, not the incremental effect of oral versus written format.",
    "criticism": "No written-format comparator; one course, university, and examiner; modest sample and subgroup sizes; adaptive questions; possible wake-time or diurnal cortisol confounding. Observational mediation cannot establish that changing self-efficacy or anxiety would improve performance.",
    "which_workflow": "Voice assessment—anxiety and physiological stress, direct.",
    "claimable_sentence": "One authentic oral exam produced higher anxiety and cortisol than a control day, but anxiety level itself was not associated with the grade."
  },
  {
    "id": "RoedigerKarpicke2006",
    "tier": "Tier 2",
    "full_citation": "Roediger, H. L., III, & Karpicke, J. D. (2006). Test-enhanced learning: Taking memory tests improves long-term retention. Psychological Science, 17(3), 249–255.",
    "link": "https://doi.org/10.1111/j.1467-9280.2006.01693.x",
    "peer_reviewed": true,
    "population": "Undergraduate psychology participant pools in two experiments using educational prose passages.",
    "design": "Randomized laboratory experiments comparing repeated study with one or repeated free-recall tests without feedback at delays up to one week.",
    "finding_verbatim": "On the delayed tests, prior testing produced substantially greater retention than studying.",
    "effect_size": "In Experiment 2 at one week, repeated testing produced approximately 61% recall versus approximately 40% after repeated study; no standardized effect was reported for this comparison.",
    "conditions_and_limits": "Free recall without feedback, prose learning, and delays of five minutes, two days, and one week; the result concerns delayed retention rather than course grades or oral reasoning.",
    "criticism": "Laboratory setting, narrow materials, short maximum delay, and written recall. The experiments do not establish oral-assessment validity, course-level transfer, or automated scoring accuracy.",
    "which_workflow": "Voice assessment—foundational retrieval-practice mechanism, indirect to oral assessment.",
    "claimable_sentence": "Retrieving prose produced better one-week retention than repeated study in a controlled experiment, without establishing effects on oral reasoning or course assessment."
  },
  {
    "id": "RyanKoppenhofer2024",
    "tier": "Tier 2",
    "full_citation": "Ryan, R. S., & Koppenhofer, J. A. (2024). Prompted self-explanations improve learning in statistics but not retention. Teaching of Psychology, 51(4), 402–413.",
    "link": "https://doi.org/10.1177/00986283221114196",
    "peer_reviewed": true,
    "population": "364 introductory-psychology participant-pool undergraduates completed acquisition testing; 199 returned the following semester for retention testing, comprising 101 self-explanation and 98 restatement participants.",
    "design": "Preregistered randomized between-participant laboratory experiment comparing written electronic self-explanation prompts with restating information; pretest, immediate posttest, and next-semester retention test.",
    "finding_verbatim": "The self-explanations that we elicited improved initial learning and were superior to students' usual study methods, but did not benefit retention.",
    "effect_size": "Condition-by-time interaction: F(2,394)=12.46, p<.001, partial eta-squared=.06. Exploratory immediate posttest contrast: 16 percentage points, t(591)=5.37, p<.001, d=.76; reported 95% CI [10.27, 22.07] is in percentage-point units, not a CI for d. Preregistered retention contrast: less than 2 percentage points favoring control, t(591)=0.50, p=.619, d=-.07; reported 95% CI [-7.40, 4.40] is likewise in percentage-point units; no CI for d was reported.",
    "conditions_and_limits": "Participants received brief instruction and studied eight statistics examples individually. The intervention was written, not spoken, and retention was assessed the next semester with varying elapsed intervals.",
    "criticism": "Loss to follow-up was substantial, from 364 to 199. The immediate between-condition contrast was exploratory. Self-explanation participants retained access to rationale material longer than controls, so extra exposure could explain the initial difference; participants also did not necessarily retrieve the rationale from memory.",
    "which_workflow": "Voice assessment—indirect learning-mechanism evidence; direct counterevidence on durable retention.",
    "claimable_sentence": "In this randomized laboratory study, the self-explanation condition scored higher immediately, but the preregistered next-semester comparison found no retention advantage."
  },
  {
    "id": "SabqatAinKhan2026",
    "tier": "Tier 3",
    "full_citation": "Sabqat, M., Ain, N., & Khan, R. A. (2026). Validity and reliability of SCOPE (Structured Comprehensive Oral Problem-based Examination) using generalizability and decision study. Pakistan Journal of Medical Sciences, 42(3), 697–703.",
    "link": "https://doi.org/10.12669/pjms.42.3.13939",
    "peer_reviewed": true,
    "population": "37 final-year medical students at one Pakistani medical college; ten medical-education experts contributed content validation and four trained examiners administered the pilot.",
    "design": "Expert content-validity study plus person-by-problem-by-cognitive-category generalizability study and decision study of a structured problem-based oral exam; each student was scored by one examiner.",
    "finding_verbatim": "The Phi-coefficient reflects moderate reliability for absolute decisions (pass/fail).",
    "effect_size": "S-CVI/Ave=.92 after revision. With two problems and two categories: G=.793, Phi=.696, relative SEM=3.55, absolute SEM=4.30; no confidence intervals. Decision-study projections: four problems and two categories G=.885, Phi=.766; six G=.920, Phi=.792; ten G=.950, Phi=.815. Student variance=48.13%; student-by-problem-by-category=19.8%.",
    "conditions_and_limits": "Structured recall and application scoring and two sampled problems; estimates support ranking more strongly than pass/fail use. Increasing problem samples was projected to improve reliability, with workload not tested.",
    "criticism": "Small single-site pilot; each student had one examiner, so examiner variance was omitted; examiner-generated problems and rater effects may be confounded. Validity was expert content relevance only. Decision-study gains are model projections, not observed replications; no external criterion or confidence intervals.",
    "which_workflow": "Voice assessment—generalizability and absolute-decision reliability, direct.",
    "claimable_sentence": "A small G-theory pilot found stronger reliability for ranking than for pass/fail decisions and projected that broader problem sampling would improve both."
  },
  {
    "id": "StephensonJohnsonGlauchCruchley2025",
    "tier": "Tier 2",
    "full_citation": "Stephenson, Z., Johnson-Glauch, N., & Cruchley, S. (2025). Interventions and facilitators of oral assessment performance in higher education: A systematic review. Assessment & Evaluation in Higher Education, 50(7), 1140–1153.",
    "link": "https://doi.org/10.1080/02602938.2025.2504621",
    "peer_reviewed": true,
    "population": "24 higher-education studies published 2015–2024 across disciplines; 16 undergraduate, six postgraduate, and two unspecified; samples ranged from 11 to 789, with two unreported.",
    "design": "Systematic review with MMAT appraisal and narrative synthesis of experimental, quasi-experimental, qualitative, survey, mixed-method, and correlational studies of interventions for oral-assessment performance.",
    "finding_verbatim": "The synthesis identified several facilitators of student performance, including structured opportunities for practice, timely and constructive feedback from instructors and peers.",
    "effect_size": "No pooled effect size or confidence interval. Selected study results included VR performance M=42.82 to 49.85, p<.001; another study found higher anxiety after VR than face-to-face practice, p=.05, but no final-performance difference, p=.39; no confidence intervals were reported in the review for these results.",
    "conditions_and_limits": "Practice, instructor or peer feedback, video or self-reflection, and some technology interventions were associated with better performance; most evidence concerned oral presentations or communication, not interactive viva reasoning or score validity.",
    "criticism": "Heterogeneous interventions, outcomes, and designs; only 15% of screening duplicated and quality appraisal by one author; all studies retained regardless of appraisal. No meta-analysis; mostly selective p values without effects or CIs. The printed method says four databases while the abstract lists six. Anxiety findings were mixed, so practice and feedback are not proven anxiety treatments for vivas.",
    "which_workflow": "Voice assessment—practice and anxiety-mitigation evidence, indirect.",
    "claimable_sentence": "Practice and constructive feedback are plausible supports for oral performance, but the review does not establish a pooled effect or a reliable reduction in viva anxiety."
  },
  {
    "id": "TurnerDavilaRoss2015",
    "tier": "Tier 3",
    "full_citation": "Turner, M., & Davila-Ross, M. (2015). Using oral exams to assess psychological literacy: The final year research project interview. Psychology Teaching Review, 21(2), 48–68.",
    "link": "https://doi.org/10.53841/bpsptr.2015.21.2.48",
    "peer_reviewed": true,
    "population": "454 final-year undergraduate psychology students on two accredited programs at one UK university; 443 across three cohorts completed the interview.",
    "design": "Three-cohort assessment study of a 15-minute, 10%-weighted project interview; two trained academic raters scored independently before agreeing a final mark; correlations, paired-rater comparisons, and team or day analyses.",
    "finding_verbatim": "A significant difference was found in marks awarded by the first interviewer and second interviewer, although the effect size was small.",
    "effect_size": "First versus second rater: M=66.5% (SD=8.3) versus 67.1% (SD=8.9), t(391)=3.78, p<.001, d=.06; within-cohort rater correlations r=.94–.95, p<.001. Interview–project-report r=.37–.40 and interview–final-year-average r=.49–.58, all p<.001. No confidence intervals were reported.",
    "conditions_and_limits": "Two raters; supervisor excluded; common five-theme prompt framework, fixed timing, rubric, rater training, audio recording, disclosed opening questions, examples, workshop, and mock interview; interview counted for 10% of the unit.",
    "criticism": "Pearson r measures rank consistency, not absolute agreement; consensus can conceal initial disagreement. Team assignment was not randomized, flexible follow-ups reduce standardization, and grade correlations do not establish the intended construct against an external criterion. One institution; no inter-case or test–retest reliability and no learning or retention outcome.",
    "which_workflow": "Voice assessment—validity and human-scoring reliability, direct.",
    "claimable_sentence": "In one highly structured psychology interview, two raters ranked students similarly, but a small systematic mark difference remained and the evidence does not generalize to all oral assessments."
  },
  {
    "id": "Yang2021",
    "tier": "Tier 1",
    "full_citation": "Yang, C., Luo, L., Vadillo, M. A., Yu, R., & Shanks, D. R. (2021). Testing (quizzing) boosts classroom learning: A systematic and meta-analytic review. Psychological Bulletin, 147(4), 399–435.",
    "link": "https://doi.org/10.1037/bul0000309",
    "peer_reviewed": true,
    "population": "48,478 students from 222 independent classroom studies yielding 573 effect sizes across elementary, secondary, university or college, and continuing-education settings; the university or college subgroup contained 335 effect sizes.",
    "design": "Systematic review and multilevel random-effects meta-analysis of classroom studies comparing testing or quizzing with no quizzing or fewer quiz questions.",
    "finding_verbatim": "The results show that overall testing (quizzing) raises student academic achievement to a medium extent.",
    "effect_size": "Overall g=.499, 95% CI [.442, .557], p<.001. University or college subgroup g=.486, 95% CI [.420, .552], p<.001; problem-solving outcomes g=.453, 95% CI [.309, .596]. Against no activity or filler g=.610 [.547, .673]; against restudy g=.330 [.256, .404]; against elaborative strategies g=.095 [-.005, .194]. Corrective feedback g=.537 [.469, .604], versus no feedback g=.374 [.278, .471]. Of point estimates, 82.9% were positive, 15.5% negative, and 1.6% zero; sign does not establish statistical significance.",
    "conditions_and_limits": "Initial instruction had to occur in a classroom, although quizzes could occur inside or outside class. Effects varied with control activity, quiz–criterion format and material alignment, corrective feedback, repetition, treatment duration, and study design. The category included many forms of quizzing and academic achievement.",
    "criticism": "Heterogeneity was substantial. Within-subject effects were larger than between-subject effects: g=.674, 95% CI [.592, .757], versus g=.415, 95% CI [.348, .481]. Within-subject correlations were usually unreported and therefore imputed. Many outcomes reused, rephrased, or closely related quiz material; the synthesis does not isolate far transfer, spoken reasoning, or a pure retrieval mechanism from exposure, feedback, and motivation.",
    "which_workflow": "Voice assessment—retrieval-practice mechanism; authentic-course evidence.",
    "claimable_sentence": "Across classroom studies, quizzing improved academic outcomes on average, including at university level, but effects depended on feedback, alignment, repetition, and research design."
  }
]
