[
    {
        "id":  "AERA2014Standards",
        "tier":  "Authoritative professional standard",
        "full_citation":  "American Educational Research Association, American Psychological Association, \u0026 National Council on Measurement in Education. (2014). Standards for educational and psychological testing.",
        "link":  "https://www.testingstandards.net/uploads/7/6/6/4/76643089/standards_2014edition.pdf",
        "peer_reviewed":  false,
        "population":  "Educational and psychological testing programs and score uses.",
        "design":  "Consensus standards covering validation, scoring, automated scoring, reliability, fairness, and reporting.",
        "finding_verbatim":  "Validity refers to the degree to which evidence and theory support score interpretations for proposed uses.",
        "effect_size":  "Not applicable.",
        "conditions_and_limits":  "The required evidence depends on the proposed interpretation, use, population, and consequences; the Standards do not prescribe one coefficient or cutoff.",
        "criticism":  "Normative guidance rather than treatment evidence.",
        "which_workflow":  "WP-07 validity framework and vendor checklist.",
        "claimable_sentence":  "Validation attaches to an intended interpretation and use; neither a rubric nor human-machine agreement validates a scoring system by itself."
    },
    {
        "id":  "JonssonSvingby2007",
        "tier":  "Tier 2",
        "full_citation":  "Jönsson, A., \u0026 Svingby, G. (2007). The use of scoring rubrics: Reliability, validity and educational consequences. Educational Research Review, 2(2), 130–144.",
        "link":  "https://doi.org/10.1016/j.edurev.2007.05.002",
        "peer_reviewed":  true,
        "population":  "Seventy-five empirical rubric studies across educational tasks and settings.",
        "design":  "Systematic literature review.",
        "finding_verbatim":  "Rubrics can improve scoring reliability, especially when complemented with exemplars and rater training.",
        "effect_size":  "No pooled effect. The review reports heterogeneous coefficients and notes that most exact-agreement estimates discussed were below 70%.",
        "conditions_and_limits":  "Findings vary with analytic versus holistic design, topic specificity, criteria, task, training, and exemplars.",
        "criticism":  "Heterogeneous tasks, designs, metrics, and eras preclude a general human IRR rate; validity evidence was comparatively sparse.",
        "which_workflow":  "WP-07 Half A, rubric reliability and validity.",
        "claimable_sentence":  "Rubrics and training can improve agreement, but rubric possession alone is not reliability or validity evidence."
    },
    {
        "id":  "JonssonBalan2018",
        "tier":  "Tier 3",
        "full_citation":  "Jönsson, A., \u0026 Balan, A. (2018). Analytic or holistic: A study of agreement between different grading models. Practical Assessment, Research \u0026 Evaluation, 23, Article 12.",
        "link":  "https://eric.ed.gov/?id=EJ1191403",
        "peer_reviewed":  true,
        "population":  "Twenty-four teachers grading the same performance.",
        "design":  "Randomized comparison of analytic and holistic grading.",
        "finding_verbatim":  "Analytic grading models improve agreement between graders.",
        "effect_size":  "Analytic grading: 66% exact agreement, Cohen\u0027s kappa .60, Spearman rho .97. Holistic: 46%, .41, and .94.",
        "conditions_and_limits":  "One task and a small teacher sample.",
        "criticism":  "The design does not show that analytic grading improves construct validity; high rank correlations in both conditions conceal different exact agreement.",
        "which_workflow":  "WP-07 metric interpretation.",
        "claimable_sentence":  "In one randomized teacher study, analytic grading improved exact and chance-corrected agreement while rank correlations remained near one in both groups."
    },
    {
        "id":  "RezaeiLovorn2010",
        "tier":  "Tier 3",
        "full_citation":  "Rezaei, A. R., \u0026 Lovorn, M. (2010). Reliability and validity of rubrics for assessment through writing. Assessing Writing, 15(1), 18–39.",
        "link":  "https://doi.org/10.1016/j.asw.2010.01.003",
        "peer_reviewed":  true,
        "population":  "326 college students, mostly novice raters, evaluating two engineered essays.",
        "design":  "Controlled between-group writing-assessment experiment.",
        "finding_verbatim":  "Rubrics may not improve reliability or validity without effective design and rater training.",
        "effect_size":  "The fluent off-prompt essay averaged 68.21/100 and received 8.65/15 for nonexistent citations; the content-correct essay with 20 mechanical errors averaged 58.77/100. Within-group SDs increased after the rubric.",
        "conditions_and_limits":  "Novice raters, two artificial essays, and score dispersion rather than a conventional IRR coefficient.",
        "criticism":  "Local experimental manipulation; results do not imply every analytic rubric harms scoring.",
        "which_workflow":  "WP-07 adverse rubric evidence.",
        "claimable_sentence":  "A rubric did not prevent fluency and mechanics from overwhelming task response and content for novice raters in this experiment."
    },
    {
        "id":  "Williamson2012",
        "tier":  "Tier 2",
        "full_citation":  "Williamson, D. M., Xi, X., \u0026 Breyer, F. J. (2012). A framework for evaluation and use of automated scoring. Educational Measurement: Issues and Practice, 31(1), 2–13.",
        "link":  "https://doi.org/10.1111/j.1745-3992.2011.00223.x",
        "peer_reviewed":  true,
        "population":  "Automated scoring systems and their proposed operational uses.",
        "design":  "Validity and operational evaluation framework.",
        "finding_verbatim":  "Automated scoring evaluation requires more than agreement with human scores.",
        "effect_size":  "Not applicable.",
        "conditions_and_limits":  "Evidence should cover intended purpose, system capability, human-machine agreement, external criteria, generalization, consequences, subgroups, and maintenance.",
        "criticism":  "Framework rather than a single empirical test.",
        "which_workflow":  "WP-07 automated scoring and procurement.",
        "claimable_sentence":  "Automated scoring evaluation is a validity program, not an accuracy contest against one human score."
    },
    {
        "id":  "Teckwani2024",
        "tier":  "Tier 3",
        "full_citation":  "Teckwani, S. H., Wong, A. H.-P., Luke, N. V., \u0026 Low, I. C. C. (2024). Accuracy and reliability of large language models in assessing learning outcomes achievement across cognitive domains. Advances in Physiology Education, 48(4), 904–914.",
        "link":  "https://doi.org/10.1152/advan.00137.2024",
        "peer_reviewed":  true,
        "population":  "117 deidentified assignments in one university scientific-inquiry course; two human graders and two fresh-thread runs of GPT-3.5, GPT-4o, and Gemini.",
        "design":  "Repeated rubric scoring comparison across four learning categories.",
        "finding_verbatim":  "Human graders surpassed the LLMs in interrater reliability across all rubric categories.",
        "effect_size":  "Human overall exact agreement was 80%; repeated overall agreement was 40% for GPT-3.5 and 49% for GPT-4o. No human-LLM criterion correlation was significant.",
        "conditions_and_limits":  "One course, one rubric, hosted model versions from the study period; percent agreement and Pearson correlation were primary reliability analyses.",
        "criticism":  "Correlation is not agreement, and the local design does not establish brand-level performance.",
        "which_workflow":  "WP-07 Half A adverse LLM evidence.",
        "claimable_sentence":  "Mean score similarity did not imply repeatable or human-aligned criterion scoring in this university assignment study."
    },
    {
        "id":  "JohnsonZhang2024",
        "tier":  "Tier 3",
        "full_citation":  "Johnson, R. L., \u0026 Zhang, S. (2024). Examining responsible use of zero-shot AI approaches to scoring essays. Scientific Reports, 14, 30064.",
        "link":  "https://doi.org/10.1038/s41598-024-79208-2",
        "peer_reviewed":  true,
        "population":  "13,121 PERSUADE 2.0 essays across eight prompts, grades 8–12.",
        "design":  "Large zero-shot GPT-4o scoring study with agreement, predictive, and subgroup analyses.",
        "finding_verbatim":  "GPT-4o scores were systematically lower than human scores.",
        "effect_size":  "Mean difference -0.90 points, 95% CI [-.922, -.875]; exact agreement about 30%; QWK .437, 95% CI [.387, .487]; within-one agreement 77%; disattenuated correlation .757.",
        "conditions_and_limits":  "One zero-shot prompt/model/batch on a public secondary-school corpus.",
        "criticism":  "Observed subgroup differences do not identify causal discrimination; human scores are not error-free. The study does not test fine-tuning or evidence gates.",
        "which_workflow":  "WP-07 correlation-versus-agreement and fairness.",
        "claimable_sentence":  "A moderately high correlation coexisted with low exact agreement, systematic under-scoring, and a subgroup discrepancy in this large zero-shot study."
    },
    {
        "id":  "Pack2024",
        "tier":  "Tier 3",
        "full_citation":  "Pack, A., Barrett, A., \u0026 Escalante, J. (2024). Large language models and automated essay scoring of English language learner writing. Computers and Education: Artificial Intelligence, 6, 100234.",
        "link":  "https://doi.org/10.1016/j.caeai.2024.100234",
        "peer_reviewed":  true,
        "population":  "119 ESL admissions essays; two experienced teacher ratings per essay; four LLM services at two dates separated by a mean 134.3 days.",
        "design":  "Longitudinal repeated automated scoring comparison.",
        "finding_verbatim":  "Model performance varied over time despite using the same prompts.",
        "effect_size":  "GPT-4 within-time ICCs .897 and .927; human-alignment ICC fell .843 to .779; mean score shifted 4.35 to 4.09, t=4.07, p\u003c.001, d=.69.",
        "conditions_and_limits":  "Small local ESL corpus, zero-shot prompting, and hosted services that changed between dates.",
        "criticism":  "The study measures changing operational services, not a frozen model; ICC variants were not always fully identified.",
        "which_workflow":  "WP-07 model drift and vendor change control.",
        "claimable_sentence":  "Saving the prompt did not freeze the measurement instrument across hosted-model updates."
    },
    {
        "id":  "Crossley2026",
        "tier":  "Tier 3",
        "full_citation":  "Crossley, S. A., Holmes, L., \u0026 Morris, W. (2026). Assessing the reliability and validity of large language models in automatic essay scoring. Assessing Writing, 69, 101082.",
        "link":  "https://doi.org/10.1016/j.asw.2026.101082",
        "peer_reviewed":  true,
        "population":  "11,826 ASAP2 source-based argumentative essays; five prompts; 23 trained raters.",
        "design":  "Supervised 70/15/15 split comparing fine-tuned and zero-shot models on held-out same-corpus essays.",
        "finding_verbatim":  "Fine-tuned language models demonstrated high agreement with human ratings.",
        "effect_size":  "Human ICC(2,1) .828 and QWK .706; held-out QWK .79 for fine-tuned ModernBERT, .78 for fine-tuned GPT-4o-mini, and .51 for zero-shot o3-mini.",
        "conditions_and_limits":  "Strong results depend on same-corpus supervised fine-tuning and a bounded benchmark.",
        "criticism":  "Selected convergent linguistic relations do not establish fairness, consequences, transportability, or rationale correctness.",
        "which_workflow":  "WP-07 positive bound on LLM scoring.",
        "claimable_sentence":  "A fine-tuned LLM can reach strong held-out agreement on the same essay corpus; that is not plug-and-play validity for a new course."
    },
    {
        "id":  "Naidu2026",
        "tier":  "Tier 3",
        "full_citation":  "Naidu, M., Montaquila, N. S., Roa, J. P., \u0026 Achilli, T.-M. (2026). Evaluating large language models for rubric-based essay grading in an undergraduate biology course. Journal of Microbiology \u0026 Biology Education, e00095-26.",
        "link":  "https://doi.org/10.1128/jmbe.00095-26",
        "peer_reviewed":  true,
        "population":  "200 deidentified authentic essays from one undergraduate biology course; one human reference scorer; three LLMs under zero- and five-example prompts.",
        "design":  "Rubric-guided human-LLM and inter-model scoring comparison.",
        "finding_verbatim":  "LLM grading behavior varied across models, prompting strategies, and rubric components.",
        "effect_size":  "Zero-shot human correlations: ChatGPT .49, Claude .65, Gemini .35; exact agreement 36%, 45%, 32%. Few-shot correlations .48, .51, .54; human-plus-LLM ICC .56, 95% CI [.49, .63].",
        "conditions_and_limits":  "Single course/institution and one human reference; five examples; checklist-style rubric.",
        "criticism":  "No human IRR gold standard. Examples did not improve all models or criteria, and one condition collapsed a writing criterion to a constant maximum.",
        "which_workflow":  "WP-07 authentic higher-education LLM scoring.",
        "claimable_sentence":  "Few-shot calibration did not uniformly improve scoring; effects differed by model and rubric component."
    },
    {
        "id":  "Schoepp2018",
        "tier":  "Tier 3",
        "full_citation":  "Schoepp, K., Danaher, M., \u0026 Ater Kranov, A. (2018). An effective rubric norming process. Practical Assessment, Research, and Evaluation, 23, Article 11.",
        "link":  "https://doi.org/10.7275/z3gm-fp34",
        "peer_reviewed":  true,
        "population":  "Higher-education faculty rating transcripts with the six-level Computing Professional Skills Assessment rubric.",
        "design":  "Consensus-driven rubric norming case and procedural report.",
        "finding_verbatim":  "Norming should be built around meaningful, evidence-driven consensus.",
        "effect_size":  "Authors report prior cumulative exact agreement of 75% and later 87–100% IRR, but describe the later figures as unpublished.",
        "conditions_and_limits":  "Raters identify/highlight transcript evidence before scoring and point to transcript/rubric during consensus.",
        "criticism":  "Training, evidence capture, discussion, and consensus are bundled; no randomized evidence-gate comparison.",
        "which_workflow":  "WP-07 evidence-before-score human precedent and rater norming.",
        "claimable_sentence":  "Higher-education raters were instructed to identify transcript evidence before scoring, directly contradicting a broad novelty claim."
    },
    {
        "id":  "TakanoIchikawa2022",
        "tier":  "Tier 3",
        "full_citation":  "Takano, S., \u0026 Ichikawa, O. (2022). Automatic scoring of short answers using justification cues estimated by BERT. Proceedings of the 17th Workshop on Innovative Use of NLP for Building Educational Applications, 8–13.",
        "link":  "https://doi.org/10.18653/v1/2022.bea-1.2",
        "peer_reviewed":  true,
        "population":  "Six Japanese high-school reading-comprehension prompts with 2,100 answers each.",
        "design":  "Criterion-specific cue extraction followed by cue-only LightGBM scoring; absent cue forces zero.",
        "finding_verbatim":  "Justification cues enable accurate scoring with substantially less training data.",
        "effect_size":  "At training n=400 per prompt, QWK .877 with predicted cues, .706 without cues, and .959 with gold cues.",
        "conditions_and_limits":  "Short Japanese answers with three or four additive criteria; representation and gate change together.",
        "criticism":  "Not a clean UI/procedural gate ablation and not authentic higher-education multilevel work.",
        "which_workflow":  "WP-07 Q11 architectural hard-gate precedent.",
        "claimable_sentence":  "A cue-only scorer that forces zero without evidence is direct prior art for an evidence bottleneck."
    },
    {
        "id":  "Lee2024",
        "tier":  "Tier 3",
        "full_citation":  "Lee, G.-G., Latif, E., Wu, X., Liu, N., \u0026 Zhai, X. (2024). Applying large language models and chain-of-thought for automatic scoring. Computers and Education: Artificial Intelligence, 6, 100213.",
        "link":  "https://doi.org/10.1016/j.caeai.2024.100213",
        "peer_reviewed":  true,
        "population":  "1,650 middle-school science constructed responses across six tasks.",
        "design":  "GPT-3.5/GPT-4 comparison of six prompt strategies. Few-shot examples quote response evidence, map it to rubric components, then emit a level.",
        "finding_verbatim":  "No single prompting strategy was optimal across every assessment task.",
        "effect_size":  "GPT-4 mean accuracy: .5487 zero-shot/no-CoT; .6831 zero-shot/CoT plus context/rubric; .6604 few-shot/no-CoT; .6975 few-shot/CoT plus context/rubric. CoT alone was .5532 zero-shot and .6515 few-shot.",
        "conditions_and_limits":  "K–12 science responses; evidence ordering, context, rubric, demonstrations, and reasoning are bundled; quotations were not independently validated.",
        "criticism":  "At least one item worsened under the fuller strategy; the prompt is not a mechanically enforced gate.",
        "which_workflow":  "WP-07 pivotal Q11 prior art.",
        "claimable_sentence":  "Evidence-before-level LLM prompting is peer-reviewed prior art, but its incremental gate effect was not isolated."
    },
    {
        "id":  "WangAutoSCORE2026",
        "tier":  "Tier 3",
        "full_citation":  "Wang, Y., Ding, Z., Wu, X., Sun, S., Liu, N., \u0026 Zhai, X. (2026). AutoSCORE: Enhancing automated scoring with multi-agent large language models via structured component recognition. Proceedings of the AAAI Conference on Artificial Intelligence, 40(48), 40898–40906.",
        "link":  "https://doi.org/10.1609/aaai.v40i48.42123",
        "peer_reviewed":  true,
        "population":  "6,656 ASAP responses: 4,871 short answers and 1,785 essays across four datasets.",
        "design":  "Extraction agent emits structured rubric components before a scoring agent; comparison with direct prompting across GPT-4o and Llama models.",
        "finding_verbatim":  "Structured component recognition improves scoring in several settings but not uniformly.",
        "effect_size":  "GPT-4o QWK .540 to .629 in English and .251 to .344 in essays, while essay accuracy fell .280 to .269 and biology accuracy .819 to .806; several Llama QWK values declined.",
        "conditions_and_limits":  "The scorer still sees the full response; extracted components need not be quotations; the multi-agent package is bundled.",
        "criticism":  "Counter-results preclude a monotonic benefit claim and the design does not isolate a hard evidence gate.",
        "which_workflow":  "WP-07 Q11 multi-agent precedent and adverse evidence.",
        "claimable_sentence":  "Extract-then-score improved some metrics and worsened others; structured extraction is not a universal gain."
    },
    {
        "id":  "AnghelGradeAgentOps2026",
        "tier":  "Tier 3",
        "full_citation":  "Anghel, C., Anghel, A. A., Craciun, M. V., Cocu, A., Vulpe, D.-E., Andrei, C. A., Maier, C., Scheau, C., Dragosloveanu, S., \u0026 Cergan, R. (2026). GradeAgentOps: A verification-first framework for evidence-anchored LLM exam grading. AI, 7(6), 198.",
        "link":  "https://doi.org/10.3390/ai7060198",
        "peer_reviewed":  true,
        "population":  "1,000 university short open-ended answers: 100 students by 10 questions; two expert graders.",
        "design":  "Evidence-bearing JSON contract with deterministic total and evidence-string recoverability verification and repair.",
        "finding_verbatim":  "Verification-first grading produced agreement near the human-expert baseline.",
        "effect_size":  "Human E1/E2 QWK .678, MAE 1.827; full system versus E2 QWK .652, MAE 1.935, within ±2 points .667.",
        "conditions_and_limits":  "Technical and argumentative short answers; verification checks string recoverability rather than semantic entailment.",
        "criticism":  "Every configuration retained the evidence contract/verifier, so the gate\u0027s effect was not isolated.",
        "which_workflow":  "WP-07 Q11 higher-education verification precedent.",
        "claimable_sentence":  "A university LLM grader already requires recoverable evidence, but its evidence gate was not separately ablated."
    },
    {
        "id":  "HongRulers2026",
        "tier":  "Tier 4",
        "full_citation":  "Hong, Y., Yao, H., Shen, B., Xu, W., Wei, H., \u0026 Dong, Y. (2026). From rubrics to reliable scores: Evidence-grounded text evaluation with LLM judges (arXiv:2601.08654, v2).",
        "link":  "https://arxiv.org/abs/2601.08654",
        "peer_reviewed":  false,
        "population":  "Held-out ASAP 2.0 n=7,421; SummHF n=1,038; DREsS n=1,979; WebNLG n=611; calibration n=200.",
        "design":  "Typed extractive evidence, mechanical grounding checks, and calibration, with removal of the evidence phase.",
        "finding_verbatim":  "Ablations confirm that rubric locking, evidence grounding, and distribution alignment each contribute to performance.",
        "effect_size":  "GPT-4o-mini full versus no Phase II evidence QWK, reported as bootstrap mean ± SE: ASAP .7077±.0058 versus .6833±.0408; SummHF .3984±.0305 versus .3737±.0260; DREsS .5292±.0203 versus .5088±.0218; WebNLG .6135±.0294 versus .5613±.0319.",
        "conditions_and_limits":  "arXiv v2 preprint as verified 5 August 2026; Phase II bundles evidence production and mechanical verification; the ± values are bootstrap SE, not confidence intervals; no paired-difference interval or significance test is reported; benchmarks are not authentic higher-education multilevel performance tasks.",
        "criticism":  "The closest controlled novelty challenge is not yet peer reviewed and does not isolate all evidence operations.",
        "which_workflow":  "WP-07 bounded novelty conclusion.",
        "claimable_sentence":  "A contemporaneous preprint already ablates an evidence phase, precluding any broad first-test claim."
    },
    {
        "id":  "Zeng2022",
        "tier":  "Tier 3",
        "full_citation":  "Zeng, Z., Li, S., Gašević, D., \u0026 Chen, G. (2022). Do deep neural nets display human-like attention in short answer scoring? Proceedings of NAACL-HLT 2022.",
        "link":  "https://doi.org/10.18653/v1/2022.naacl-main.14",
        "peer_reviewed":  true,
        "population":  "Study 1: 20 raters and 60 selected ASAP answers; Study 2: 10 participants and 36 answers.",
        "design":  "Human evidence-span annotation and a controlled comparison of scoring with or without model highlights.",
        "finding_verbatim":  "Model-provided highlights did not clearly improve human scoring efficiency.",
        "effect_size":  "Descriptive QWK .71 control versus .74 highlight; mean time 42.17 versus 54.83 seconds, with the time difference p\u003c.01; no inferential QWK test reported.",
        "conditions_and_limits":  "Small selected samples; model-generated highlights rather than a hard evidence gate.",
        "criticism":  "Agreement change was descriptive and evidence cues increased time; highlights require independent validation.",
        "which_workflow":  "WP-07 evidence-gate cost and counterevidence.",
        "claimable_sentence":  "Evidence cues can impose time cost even when agreement changes little."
    },
    {
        "id":  "Madhusudhan2025Abstention",
        "tier":  "Tier 2",
        "full_citation":  "Madhusudhan, N., Madhusudhan, S. T., Yadav, V., \u0026 Hashemi, M. (2025). Do LLMs know when to NOT answer? Investigating abstention abilities of large language models. Proceedings of the 31st International Conference on Computational Linguistics, 9329–9345.",
        "link":  "https://aclanthology.org/2025.coling-main.627/",
        "peer_reviewed":  true,
        "population":  "Black-box language models evaluated on Abstain-QA across answerable and unanswerable, represented and under-represented, factual and reasoning questions.",
        "design":  "Benchmark evaluation using an answerable-unanswerable confusion matrix and three prompting strategies.",
        "finding_verbatim":  "Even strong models had difficulty abstaining; strict prompting and chain-of-thought improved selected abstention results.",
        "effect_size":  "Model- and condition-specific values; no classroom scoring effect is transferred into WP-07.",
        "conditions_and_limits":  "Question answering rather than consequential higher-education scoring; abstention depends on the error taxonomy and prompt.",
        "criticism":  "General abstention competence does not establish calibrated insufficient-evidence decisions under a rubric.",
        "which_workflow":  "WP-07 Questions 12, 13, and 22; vendor checklist.",
        "claimable_sentence":  "An abstain option must be tested because instruction alone does not guarantee reliable abstention."
    },
    {
        "id":  "ZhouRubricBench2026",
        "tier":  "Tier 2",
        "full_citation":  "Zhou, J., Zhang, Q., Wang, Y., Lyu, F., Ming, Y., Xu, C., Sun, Q., Zheng, K., Kang, P., Liu, X., \u0026 Ma, C. (2026). RubricBench: Aligning model-generated rubrics with human standards. Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics, 31179–31200.",
        "link":  "https://doi.org/10.18653/v1/2026.acl-long.1439",
        "peer_reviewed":  true,
        "population":  "1,147 curated hard pairwise comparisons with expert-annotated atomic rubrics.",
        "design":  "Benchmark comparison of human-annotated and model-generated rubrics for rubric-guided evaluation.",
        "finding_verbatim":  "The authors report a substantial capability gap between human-annotated and model-generated rubrics.",
        "effect_size":  "1,147 comparisons; model- and condition-specific benchmark results are not converted into a pooled effect.",
        "conditions_and_limits":  "Reward-model evaluation benchmark, not authentic course rubric development or a higher-education IRR trial.",
        "criticism":  "The benchmark establishes a drafting-quality concern, not the effectiveness of any local rubric-review process.",
        "which_workflow":  "WP-07 Question 19; expert-review requirements.",
        "claimable_sentence":  "AI-drafted rubrics require expert construct review and local pilot testing rather than automatic adoption."
    },
    {
        "id":  "AACSB2026Standards",
        "tier":  "Official normative source",
        "full_citation":  "AACSB International. (2026). AACSB global standards for business education.",
        "link":  "https://www.aacsb.edu/educators/global-standards",
        "peer_reviewed":  false,
        "population":  "Business schools within AACSB accreditation scope.",
        "design":  "Normative accreditation standards, including Standard 5 Assurance of Learning.",
        "finding_verbatim":  "Schools use documented assessment tied to competencies and use results for curricular improvement.",
        "effect_size":  "Not applicable.",
        "conditions_and_limits":  "Institutional requirements vary by mission and jurisdiction; the standards do not preapprove AI scorers.",
        "criticism":  "Accreditation language cannot substitute for local validity, fairness, privacy, or human-review evidence.",
        "which_workflow":  "WP-07 Question 18 and checklist governance section.",
        "claimable_sentence":  "Accreditation keeps responsibility for aligned evidence and improvement with the institution."
    },
    {
        "id":  "ABET2026Criteria",
        "tier":  "Official normative source",
        "full_citation":  "ABET. (2026). Criteria for accrediting engineering programs, 2026–2027.",
        "link":  "https://www.abet.org/accreditation/accreditation-criteria/criteria-for-accrediting-engineering-programs-2026-2027/",
        "peer_reviewed":  false,
        "population":  "Engineering programs seeking ABET accreditation in the 2026–2027 cycle.",
        "design":  "Normative program accreditation criteria.",
        "finding_verbatim":  "Programs must have documented processes for assessing student outcomes and using results for continuous improvement.",
        "effect_size":  "Not applicable.",
        "conditions_and_limits":  "Program-specific requirements apply; no automated scorer is preapproved.",
        "criticism":  "Compliance evidence is use- and program-specific and does not validate a vendor system by association.",
        "which_workflow":  "WP-07 Question 18 and checklist governance section.",
        "claimable_sentence":  "An institution must show its assessment process and improvement use, regardless of scoring technology."
    },
    {
        "id":  "HLC2025Criteria",
        "tier":  "Official normative source",
        "full_citation":  "Higher Learning Commission. (2025). Criteria for accreditation (CRRT.B.10.010; revised June 2024, effective September 2025).",
        "link":  "https://www.hlcommission.org/accreditation/policies/criteria/",
        "peer_reviewed":  false,
        "population":  "HLC-accredited and candidate institutions.",
        "design":  "Normative institutional accreditation criteria.",
        "finding_verbatim":  "Institutions provide evidence for teaching, learning, student success, and systematic improvement.",
        "effect_size":  "Not applicable.",
        "conditions_and_limits":  "Evidence is interpreted in institutional context; an AI citation is not, by itself, validity evidence.",
        "criticism":  "The criteria define accountability rather than a psychometric acceptance threshold.",
        "which_workflow":  "WP-07 Question 18 and checklist governance section.",
        "claimable_sentence":  "Institutional accreditation requires evidence and improvement, not a vendor-branded scoring method."
    },
    {
        "id":  "MSCHE2026Standards",
        "tier":  "Official normative source",
        "full_citation":  "Middle States Commission on Higher Education. (2026). Standards for accreditation and requirements of affiliation (15th ed.).",
        "link":  "https://www.msche.org/standards/standards-15/",
        "peer_reviewed":  false,
        "population":  "Institutions entering the applicable MSCHE review cycle on or after the Fifteenth Edition effective date.",
        "design":  "Normative institutional accreditation standards, effective July 1, 2026 for specified cycles and actions.",
        "finding_verbatim":  "The standards require documented assessment and institutional responsibility for educational effectiveness.",
        "effect_size":  "Not applicable.",
        "conditions_and_limits":  "Fourteenth- versus Fifteenth-Edition applicability depends on the institution\u0027s review timing.",
        "criticism":  "The standards do not certify individual scoring products or erase local validation obligations.",
        "which_workflow":  "WP-07 Question 18 and checklist governance section.",
        "claimable_sentence":  "Accreditation review examines the institution\u0027s evidence system, not an AI scorer in isolation."
    },
    {
        "id":  "Weigle1998",
        "tier":  "Tier 2",
        "full_citation":  "Weigle, S. C. (1998). Using FACETS to model rater training effects. Language Testing, 15(2), 263–287.",
        "link":  "https://doi.org/10.1177/026553229801500205",
        "peer_reviewed":  true,
        "population":  "Sixteen ESL composition raters receiving a 90-minute training session.",
        "design":  "Pre/post rater-training study analyzed with FACETS.",
        "finding_verbatim":  "Training reduced some severity and consistency differences, while meaningful individual rater differences remained.",
        "effect_size":  "Condition-specific FACETS estimates; WP-07 does not create a pooled training effect.",
        "conditions_and_limits":  "Small ESL-writing rater sample and brief training; effects are not transferable as a universal calibration rate.",
        "criticism":  "Training helps conditionally but does not make raters interchangeable.",
        "which_workflow":  "WP-07 Question 3 and human-baseline design.",
        "claimable_sentence":  "Rater training may reduce disagreement, but persistent individual severity requires monitoring."
    },
    {
        "id":  "Bridgeman2012",
        "tier":  "Tier 2",
        "full_citation":  "Bridgeman, B., Trapani, C., \u0026 Attali, Y. (2012). Comparison of human and machine scoring of essays: Differences by gender, ethnicity, and country. Applied Measurement in Education, 25(1), 27–40.",
        "link":  "https://doi.org/10.1080/08957347.2012.635502",
        "peer_reviewed":  true,
        "population":  "Operational essay responses examined across gender, ethnicity, and country groupings.",
        "design":  "Comparative subgroup analysis of human and conventional automated essay scores.",
        "finding_verbatim":  "Human-machine score discrepancies varied across some groups and contexts.",
        "effect_size":  "Group- and prompt-specific results; no pooled fairness coefficient is used in WP-07.",
        "conditions_and_limits":  "Conventional e-rater setting, not an LLM scorer; discrepancy does not identify which scorer is biased.",
        "criticism":  "Aggregate agreement can conceal subgroup differences, but subgroup difference alone is not a causal bias diagnosis.",
        "which_workflow":  "WP-07 Question 9 and vendor subgroup audit.",
        "claimable_sentence":  "Fairness review needs disaggregated directional errors and uncertainty, not one overall agreement value."
    },
    {
        "id":  "ZhengLLMJudge2023",
        "tier":  "Tier 2",
        "full_citation":  "Zheng, L., Chiang, W.-L., Sheng, Y., Zhuang, S., Wu, Z., Zhuang, Y., Lin, Z., Li, Z., Li, D., Xing, E. P., Zhang, H., Gonzalez, J. E., \u0026 Stoica, I. (2023). Judging LLM-as-a-judge with MT-Bench and Chatbot Arena. Advances in Neural Information Processing Systems, 36, 46595–46623.",
        "link":  "https://proceedings.neurips.cc/paper_files/paper/2023/hash/91f18a1287b398d378ef22505bf41832-Abstract-Datasets_and_Benchmarks.html",
        "peer_reviewed":  true,
        "population":  "Pairwise and single-answer evaluations in MT-Bench and Chatbot Arena.",
        "design":  "Benchmark evaluation of LLM judge agreement and systematic biases.",
        "finding_verbatim":  "The study documents position, verbosity, and self-enhancement biases in LLM judging.",
        "effect_size":  "Benchmark- and model-specific; not transferred as an educational scoring rate.",
        "conditions_and_limits":  "General dialogue evaluation rather than rubric-based student assessment.",
        "criticism":  "Bias mechanisms motivate controls but need local replication in the assessment use.",
        "which_workflow":  "WP-07 Question 7 and scorer-protocol design.",
        "claimable_sentence":  "LLM judges require counterbalancing and repeated evaluation because presentation order and response style can change judgments."
    },
    {
        "id":  "Dawson2017",
        "tier":  "Tier 2",
        "full_citation":  "Dawson, P. (2017). Assessment rubrics: Towards clearer and more replicable design, research and practice. Assessment \u0026 Evaluation in Higher Education, 42(3), 347–360.",
        "link":  "https://doi.org/10.1080/02602938.2015.1111294",
        "peer_reviewed":  true,
        "population":  "Published higher-education rubric definitions and design practices.",
        "design":  "Conceptual and methodological synthesis producing a taxonomy of rubric-design features.",
        "finding_verbatim":  "A rubric is a polysemous term, with no widely accepted definition.",
        "effect_size":  "Not applicable; conceptual taxonomy rather than an effect study.",
        "conditions_and_limits":  "The article identifies 14 design elements hidden by the word rubric; it does not estimate the causal effect of any one element.",
        "criticism":  "A design taxonomy improves replicability but does not validate a local rubric or score use.",
        "which_workflow":  "WP-07 rubric specification and construct section.",
        "claimable_sentence":  "Rubric studies must report design choices because materially different scoring instruments share the same label."
    },
    {
        "id":  "Deane2013",
        "tier":  "Tier 2",
        "full_citation":  "Deane, P. (2013). On the relation between automated essay scoring and modern views of the writing construct. Assessing Writing, 18(1), 7–24.",
        "link":  "https://doi.org/10.1016/j.asw.2012.10.002",
        "peer_reviewed":  true,
        "population":  "Automated essay-scoring systems and contemporary theories of writing.",
        "design":  "Construct-focused conceptual analysis of automated essay scoring.",
        "finding_verbatim":  "The validity of automated scoring depends critically on how well it represents the writing construct.",
        "effect_size":  "Not applicable; conceptual validity analysis.",
        "conditions_and_limits":  "The paper addresses automated essay scoring, not current LLM evidence gates or a specific classroom implementation.",
        "criticism":  "It supplies a construct warning rather than a performance estimate.",
        "which_workflow":  "WP-07 automated-scoring construct representation.",
        "claimable_sentence":  "Agreement with historical ratings can coexist with construct underrepresentation."
    },
    {
        "id":  "Hallgren2012",
        "tier":  "Tier 2",
        "full_citation":  "Hallgren, K. A. (2012). Computing inter-rater reliability for observational data: An overview and tutorial. Tutorials in Quantitative Methods for Psychology, 8(1), 23–34.",
        "link":  "https://doi.org/10.20982/tqmp.08.1.p023",
        "peer_reviewed":  true,
        "population":  "Observational ratings involving categorical, ordinal, and continuous measurements.",
        "design":  "Methodological tutorial comparing common interrater reliability statistics.",
        "finding_verbatim":  "Inter-rater reliability is the relative consistency between two or more judges.",
        "effect_size":  "Not applicable; methodological guidance.",
        "conditions_and_limits":  "Statistic choice depends on scale, number of raters, rating design, and the intended reliability question.",
        "criticism":  "A general tutorial does not select the correct statistic without the actual assignment structure.",
        "which_workflow":  "WP-07 human-reference and metric specification.",
        "claimable_sentence":  "Reliability coefficients are interpretable only when the rater design and scale are named."
    },
    {
        "id":  "Kane2013",
        "tier":  "Tier 1",
        "full_citation":  "Kane, M. T. (2013). Validating the interpretations and uses of test scores. Journal of Educational Measurement, 50(1), 1–73.",
        "link":  "https://doi.org/10.1111/jedm.12000",
        "peer_reviewed":  true,
        "population":  "Educational and psychological score interpretations and uses.",
        "design":  "Argument-based validity framework.",
        "finding_verbatim":  "Validation involves an evaluation of the plausibility of the proposed interpretations and uses of test scores.",
        "effect_size":  "Not applicable; validity framework.",
        "conditions_and_limits":  "The specific inferences and evidence requirements depend on the proposed interpretation, use, population, and consequences.",
        "criticism":  "A framework does not supply empirical support for any particular scoring system.",
        "which_workflow":  "WP-07 validity definitions and institutional use.",
        "claimable_sentence":  "Validity is an argument about a proposed score interpretation and use, not a permanent property of a model."
    },
    {
        "id":  "KooLi2016",
        "tier":  "Tier 2",
        "full_citation":  "Koo, T. K., \u0026 Li, M. Y. (2016). A guideline of selecting and reporting intraclass correlation coefficients for reliability research. Journal of Chiropractic Medicine, 15(2), 155–163.",
        "link":  "https://doi.org/10.1016/j.jcm.2016.02.012",
        "peer_reviewed":  true,
        "population":  "Reliability studies using intraclass correlation coefficients.",
        "design":  "Methodological guideline and decision framework for ICC selection and reporting.",
        "finding_verbatim":  "Researchers should select the correct form of ICC and report both the ICC value and its 95% confidence interval.",
        "effect_size":  "Not applicable; methodological guideline.",
        "conditions_and_limits":  "ICC interpretation depends on one- versus two-way design, fixed versus random raters, single versus average measures, and consistency versus absolute agreement.",
        "criticism":  "Common verbal cutoffs are not substitutes for decision-specific consequences or a valid rater design.",
        "which_workflow":  "WP-07 human-reference and metric specification.",
        "claimable_sentence":  "An ICC is incomplete unless its model, unit, agreement target, and uncertainty are reported."
    },
    {
        "id":  "Mizumoto2019",
        "tier":  "Tier 3",
        "full_citation":  "Mizumoto, T., Ouchi, H., Isobe, Y., Reisert, P., Nagata, R., Sekine, S., \u0026 Inui, K. (2019). Analytic score prediction and justification identification in automated short answer scoring. Proceedings of the Fourteenth Workshop on Innovative Use of NLP for Building Educational Applications, 316–325.",
        "link":  "https://doi.org/10.18653/v1/W19-4433",
        "peer_reviewed":  true,
        "population":  "Six Japanese high-school reading-comprehension prompts with 2,100 answers each, 12,600 total.",
        "design":  "Joint neural prediction of analytic criterion scores and token-level justification cues with supervised-attention comparisons.",
        "finding_verbatim":  "We propose and formalize two novel analytical assessment tasks: analytic score prediction and justification identification.",
        "effect_size":  "At 200 training responses per prompt, average analytic QWK was .822 with supervised justification attention versus .794 without it; authors report paired-bootstrap p\u003c.01 across prompts. Human QWK was .873 in the reported comparison.",
        "conditions_and_limits":  "Scores and evidence cues are predicted jointly; the system does not enforce a verified evidence-before-level gate.",
        "criticism":  "Short, additive, Japanese-language responses do not establish transport to holistic university artifacts.",
        "which_workflow":  "WP-07 evidence-aware automated-scoring lineage.",
        "claimable_sentence":  "Joint score-and-cue modeling is direct evidence-aware prior art but not a hard operational gate."
    },
    {
        "id":  "Hellman2023",
        "tier":  "Tier 3",
        "full_citation":  "Hellman, S., Andrade, A., \u0026 Habermehl, K. (2023). Scalable and explainable automated scoring for open-ended constructed response math word problems. Proceedings of the 18th Workshop on Innovative Use of NLP for Building Educational Applications, 137–147.",
        "link":  "https://doi.org/10.18653/v1/2023.bea-1.12",
        "peer_reviewed":  true,
        "population":  "34,417 proprietary responses across 14 math-plus-text constructed-response items.",
        "design":  "Rule-based detection of exact evidence spans grouped into positive and negative scorable traits that determine rubric points.",
        "finding_verbatim":  "We develop a novel technique for the automated scoring of MPT items that leverages these rubrics to provide explainable scoring.",
        "effect_size":  "The paper evaluates 34,417 responses across 14 items and 30 stratified train/test splits; most comparative outcomes are reported graphically rather than as one portable numeric estimate.",
        "conditions_and_limits":  "Binary additive traits and proprietary data; automatic rule induction failed on one item; no evidence-gate ablation.",
        "criticism":  "Transparent exact-span rules can be brittle and do not generalize automatically to distributed qualities.",
        "which_workflow":  "WP-07 span-detection-then-score prior art.",
        "claimable_sentence":  "Detected evidence spans have already been used as the direct basis for rubric points."
    },
    {
        "id":  "Tang2024",
        "tier":  "Tier 3",
        "full_citation":  "Tang, X., Chen, H., Lin, D., \u0026 Li, K. (2024). Harnessing LLMs for multi-dimensional writing assessment: Reliability and alignment with human judgments. Heliyon, 10(14), e34262.",
        "link":  "https://doi.org/10.1016/j.heliyon.2024.e34262",
        "peer_reviewed":  true,
        "population":  "1,730 grade-7 essays from ASAP Set 7, scored on four analytic dimensions.",
        "design":  "Prompt-package and sampling-temperature comparison for GPT-3.5, GPT-4, and Claude 2 against two human raters.",
        "finding_verbatim":  "The reliability of LLMs is significantly influenced by the prompt design and hyperparameters.",
        "effect_size":  "GPT-4 QWK .1947 under an overall-score-only prompt and .5677 under the fuller criteria/reference/justification package; at temperatures above zero the latter fell to .2478, .3699, .2232, and .3519. Human QWK was .6573.",
        "conditions_and_limits":  "One public benchmark; prompt package changes several components; justification was not independently validated as evidence.",
        "criticism":  "The design supports configuration sensitivity, not a standalone effect of reasoning or evidence.",
        "which_workflow":  "WP-07 configuration sensitivity and drift controls.",
        "claimable_sentence":  "Prompt specification and sampling parameters materially changed agreement in one writing benchmark."
    },
    {
        "id":  "Ward2019",
        "tier":  "Tier 3",
        "full_citation":  "Ward, K., Kinney, K., Patania, R., Savage, L., Motley, J., \u0026 Smith, M. (2019). Development of a student grading rubric and testing for interrater agreement in a doctor of chiropractic competency program. Journal of Chiropractic Education, 33(2), 140–144.",
        "link":  "https://doi.org/10.7899/JCE-18-9",
        "peer_reviewed":  true,
        "population":  "Four instructors conducting paired ratings across 16 pilot, 16 first-version, and 14 revised senior-intern assessments.",
        "design":  "Iterative local rubric development with paired interrater agreement analysis.",
        "finding_verbatim":  "Interrater agreement improved with faculty experience and rubric refinement.",
        "effect_size":  "Pilot exact/adjacent/pass-fail agreement 46%/80%/63%; first version 49%/86%/70%; revised version 60%/93%/81%.",
        "conditions_and_limits":  "Very small local samples; rubric revision and rater learning are confounded; concurrent validity was not established.",
        "criticism":  "Adjacent agreement can look excellent while exact and decision agreement remain materially lower.",
        "which_workflow":  "WP-07 metric interpretation and human baseline.",
        "claimable_sentence":  "Within-one agreement cannot substitute for exact or consequential decision agreement."
    },
    {
        "id":  "ZhaoEduMARS2026",
        "tier":  "Tier 3",
        "full_citation":  "Zhao, X., Chen, J., Xu, W., Yan, H., Fang, C., \u0026 Wei, X. (2026). EduMARS: Can vision-language models grade like teachers? Benchmarking multimodal, rubric-based assessment on Chinese K–12 answers. Findings of the Association for Computational Linguistics: ACL 2026, 9561–9583.",
        "link":  "https://doi.org/10.18653/v1/2026.findings-acl.466",
        "peer_reviewed":  true,
        "population":  "4,501 authentic handwritten Chinese K–12 high-stakes exam responses across eight subjects; reported model subset n=1,200.",
        "design":  "Multimodal benchmark and retrieval-augmented adaptive-rubric grading with structured steps and evidence-referencing rationales.",
        "finding_verbatim":  "RARG effectively enhances the performance and interpretability of various MLLMs on EduMARS.",
        "effect_size":  "In the reported 1,200-response subset, GPT-5 raw-image zero-shot to multi-turn pipeline Spearman .638→.680 and normalized MAE .191→.170; Gemini .645→.705 and .204→.165.",
        "conditions_and_limits":  "The manipulation bundles transcription, step decomposition, evidence rationales, retrieval, and additional calls; K–12 Chinese handwritten exams are not higher-education artifacts.",
        "criticism":  "The study is multimodal evidence-first prior art but does not isolate a hard evidence gate.",
        "which_workflow":  "WP-07 multimodal evidence-first lineage.",
        "claimable_sentence":  "Evidence-referencing multimodal scoring has peer-reviewed prior art, with gains attributable only to a bundled pipeline."
    },
    {
        "id":  "Cai2026",
        "tier":  "Tier 3",
        "full_citation":  "Cai, Y. (2026). Prompt injection attacks on educational large language models for higher and vocational education. Scientific Reports, 16, 15594.",
        "link":  "https://doi.org/10.1038/s41598-026-46563-1",
        "peer_reviewed":  true,
        "population":  "Four educational benchmarks spanning essay grading, short-answer grading, multi-scenario educational prompts, and academic reasoning: ASAP, SciEntsBank, EduBench, and MMLU-Edu.",
        "design":  "Black-box prompt-injection experiments plus three proposed defenses for educational LLM pipelines, including a two-stage Evidence-First Scoring architecture.",
        "finding_verbatim":  "EFS decomposes grading into two explicit stages: evidence extraction followed by scoring.",
        "effect_size":  "The attack framework achieved author-reported attack success rates of .82, .79, .76, and .73 across the four benchmarks, exceeding comparison attacks by .19–.33 absolute on average; no isolated EFS scoring-effect estimate is reported.",
        "conditions_and_limits":  "The empirical outcomes concern attack success under the authors' benchmark, model, prompt, and black-box settings. Evidence-First Scoring is presented as a defense architecture rather than evaluated as an isolated criterion-level scoring treatment.",
        "criticism":  "The paper establishes terminology and conceptual architecture, not the validity or incremental scoring effect of an evidence gate.",
        "which_workflow":  "WP-07 named Evidence-First Scoring prior art and adversarial-risk analysis.",
        "claimable_sentence":  "Evidence-First Scoring is already named in peer-reviewed literature as a two-stage defense, but its isolated scoring effect is not established there."
    }
]
