[
  {
    "id": "Perkins2024Bypass",
    "tier": "Tier 3",
    "full_citation": "Perkins, M., Roe, J., Vu, B. H., Postma, D., Hickerson, D., McGaughran, J., & Khuat, H. Q. (2024). Simple techniques to bypass GenAI text detectors: Implications for inclusive education. International Journal of Educational Technology in Higher Education, 21, 53.",
    "link": "https://doi.org/10.1186/s41239-024-00487-w",
    "peer_reviewed": true,
    "population": "A constructed corpus of 15 original AI texts (five each from GPT-4, Claude 2, and Bard), 89 AI-manipulated variants, and 10 human controls written by academic staff and Vietnamese undergraduates; 114 texts were crossed with seven detectors, producing 797 valid tests.",
    "design": "Experimental comparative accuracy study. Six adversarial techniques were applied to AI outputs, and heterogeneous detector outputs were converted using binary, semi-binary, and logarithmic scoring approaches. Testing occurred in September–October 2023.",
    "finding_verbatim": "AI text detection technologies are not highly reliable and are vulnerable to multiple adversarial techniques.",
    "effect_size": "Authors' mean accuracy across three scoring approaches was 39.5% for unmanipulated AI text and 22.14% after manipulation; human-control accuracy was 67%. The reported mean reduction was 17.4 percentage points (17.5 elsewhere in the paper). Copyleaks reached 58.7% after manipulation; GPTKit 4.5%; Turnitin fell 42.1 points. No confidence intervals were reported.",
    "conditions_and_limits": "Texts were 350–625 words; generators and detectors were the versions available in late 2023. Manipulations were produced with the originating model or a paraphraser. Results depend on the authors' mappings of proprietary scores and a small, purpose-built corpus.",
    "criticism": "The 797 tests are repeated tool-by-text observations, not independent student submissions. There were only 15 AI seed texts and 10 human controls; some error-insertion outputs were implausibly poor student work. The abstract says six detectors and n=805, whereas the methods list seven detectors and 797 valid tests. Averaging tools and scoring approaches obscures operating differences, and the snapshot is rapidly obsolete. No independent replication of the exact protocol was located.",
    "which_workflow": "WP-06—detector reliability and adversarial failure modes.",
    "claimable_sentence": "In a small 2023 benchmark, simple transformations substantially reduced already uneven detector performance; the reported figures are a tool-and-corpus snapshot, not a campus error rate."
  },
  {
    "id": "Waltzer2024",
    "tier": "Tier 3",
    "full_citation": "Waltzer, T., Pilegard, C., & Heyman, G. D. (2024). Can you spot the bot? Identifying AI-generated writing in college essays. International Journal for Educational Integrity, 20, 11.",
    "link": "https://doi.org/10.1007/s40979-024-00158-3",
    "peer_reviewed": true,
    "population": "Study 1: a convenience sample of 140 college instructors and 145 undergraduates, primarily in the United States. Study 2: 40 separate ChatGPT survey runs. Stimuli came from 50 undergraduates' proctored psychology exam essays and ChatGPT responses to the same two prompts.",
    "design": "Preregistered online forced-choice experiment. Participants viewed six pairs, each containing one student essay and one ChatGPT essay, and identified the generated item. Binomial tests, group comparisons, regression, and mixed models evaluated accuracy and correlates.",
    "finding_verbatim": "Neither experience with ChatGPT nor content expertise improved performance.",
    "effect_size": "Instructors were correct on 70% of trials, 95% CI [66%, 73%]; students on 60%, 95% CI [57%, 64%]. Instructors outperformed students, t(283)=3.30, p=.001. ChatGPT was correct on 63%, 95% CI [57%, 69%], and did not differ significantly from the human groups in post-hoc comparisons.",
    "conditions_and_limits": "Exactly one essay in every pair was AI-generated. Essays addressed two psychology prompts; obvious spelling, punctuation, and capitalization errors in student work were corrected. ChatGPT texts were generated in January 2023 and were unedited, not hybrid.",
    "criticism": "Forced choice supplies a 50% base rate and removes the operational question of whether any AI is present. The instructor sample was nonprobability and mostly US-based; the stimulus domain was narrow; each participant saw only six pairs. Findings do not validate unaided suspicion for misconduct adjudication or edited student work.",
    "which_workflow": "WP-06—human detectability in assessment-like writing.",
    "claimable_sentence": "In a constrained paired-essay task, instructors detected ChatGPT writing above chance but with substantial error; the design does not estimate real-world authorship judgments."
  },
  {
    "id": "Jiang2024",
    "tier": "Tier 3",
    "full_citation": "Jiang, Y., Hao, J., Fauss, M., & Li, C. (2024). Detecting ChatGPT-generated essays in a large-scale writing assessment: Is there a bias against non-native English speakers? Computers & Education, 217, 105070.",
    "link": "https://doi.org/10.1016/j.compedu.2024.105070",
    "peer_reviewed": true,
    "population": "A carefully sampled large-scale GRE writing corpus containing authentic essays by native- and non-native-English test takers and ChatGPT-generated essays. The accessible primary report does not state the exact group counts.",
    "design": "Large-scale, matched-task detector-development and fairness study. The authors built custom classifiers from ETS e-rater linguistic features and GPT-based perplexity features and evaluated detection performance and differential results by English-language background.",
    "finding_verbatim": "showed no evidence of bias disadvantaging non-native English speakers",
    "effect_size": "The accessible primary abstract reports 'near-perfect detection accuracy' but provides no numerical estimate or confidence interval. This review therefore does not attach a numeric effect to the result.",
    "conditions_and_limits": "The finding concerns purpose-built, in-domain detectors using carefully sampled GRE essays and matched prompts. It is not a test of public commercial detectors on ordinary coursework and does not contradict the possibility of bias after domain or population shift.",
    "criticism": "The custom e-rater features are proprietary, limiting reproduction. The accessible publisher abstract does not report thresholds, subgroup error estimates, numerical accuracy, or confidence intervals; the closed full text was not independently inspected for this review. Generalization to other genres, generators, edited text, and campus policies is unestablished. No independent replication of the complete system was located.",
    "which_workflow": "WP-06—counterevidence on non-native-English detector bias.",
    "claimable_sentence": "A matched, large-scale GRE study reported near-perfect in-domain detection and no detected disadvantage to non-native-English writers, cutting against a universal-bias claim under those conditions."
  },
  {
    "id": "VanVlasselaer2026",
    "tier": "Tier 3",
    "full_citation": "Van Vlasselaer, M., Van Droogenbroeck, F., & Spruyt, B. (2026). Who wrote this? Evaluating the reliability of AI detection tools in higher education. International Journal for Educational Integrity, 22, 16.",
    "link": "https://doi.org/10.1007/s40979-026-00226-w",
    "peer_reviewed": true,
    "population": "A constructed ground-truth corpus of 160 long papers: 40 pre-2019 human papers by non-native-English graduate students, 40 GPT-4o Deep Research papers, 40 hybrid papers, and 40 prompt-humanised hybrid papers. A post-hoc authentic corpus contained 1,163 master's theses from one Belgian social-sciences and economics faculty.",
    "design": "Preregistered comparative evaluation of Turnitin, GPTZero, Copyleaks, and Pangram, with two disclosed deviations and a later exploratory authentic-corpus analysis. Categorical accuracy and mixed-effects error analyses compared tool scores with known AI proportions in the constructed corpus; the authentic corpus had no ground truth.",
    "finding_verbatim": "they should not be used as sole evidence in high-stakes decision-making",
    "effect_size": "Pangram classified 37/40 hybrid and 37/40 humanised papers in the expected range (strict accuracy 92.5%; inclusive accuracy 95.0%). For fully AI papers, Turnitin classified 100% as false negatives; GPTZero 70% false negative and 30% partially false negative; Copyleaks 75% and 25%, respectively. The categorical proportions had no reported confidence intervals; the paper reports model-based 95% CIs for continuous error estimates. Pangram flagged 529/1,163 authentic theses (45.5%), but ground truth was unavailable.",
    "conditions_and_limits": "Papers were long and drawn from human and social sciences; GPT-4o Deep Research and the detectors were accessed in 2025. One prompt generated each hybrid/humanisation condition. Study-defined ranges determined categorical accuracy.",
    "criticism": "The strong Pangram result is from one constructed corpus and needs independent prospective replication. Tools may trade low false-positive rates for false negatives. The authentic-thesis analysis was added post hoc and cannot estimate sensitivity, specificity, or prevalence: its 45.5% is only a flagging rate. The paper acknowledges this but later infers common AI-assisted writing, which its no-ground-truth design cannot confirm.",
    "which_workflow": "WP-06—recent detector counterevidence and unsupported prevalence inference.",
    "claimable_sentence": "One detector performed strongly on a 2026 constructed corpus, but its 45.5% flagging rate in real theses was not a prevalence estimate because those theses had no ground truth."
  },
  {
    "id": "TurnerDavilaRoss2015",
    "tier": "Tier 2",
    "full_citation": "Turner, M., & Davila Ross, M. (2015). Using oral exams to assess psychological literacy: The final year research project interview. Psychology Teaching Review, 21(2), 48–68.",
    "link": "https://files.eric.ed.gov/fulltext/EJ1146560.pdf",
    "peer_reviewed": true,
    "population": "454 final-year undergraduates on two British Psychological Society-accredited psychology programs at one UK university across 2013–2015; 443 attended the assessed interview.",
    "design": "Observational evaluation of a 15-minute structured interview seven weeks after an 8,000-word research report. Two trained academics used fixed themes, core questions, a shared follow-up pool, explicit criteria, independent initial ratings, and an agreed mark; interviews were recorded.",
    "finding_verbatim": "a reliable means of evaluating skills that contribute to psychological literacy",
    "effect_size": "Interview M=67.2% (SD=8.9) versus report M=66.6% (SD=7.1), t(437)=1.27, p=.21, d=.07; no CI reported. Interview–final-year r=.52 versus interview–report r=.38, Z=2.35, p=.018; no CIs. Rater r≥.94, but rater means differed by d=.06, t(391)=3.78, p<.001; no CI.",
    "conditions_and_limits": "The interview sampled explanation of purpose, rationale, methods, interpretation, reflection, and communication rather than detailed report recall. It depended on training, common prompts and criteria, two-rater panels, practice, recordings, and time control.",
    "criticism": "Nonrandomized and single-institution. Correlation with course performance is concurrent evidence, not causal learning evidence or proof of authorship. Follow-ups varied; a low-response survey identified stress and possible disadvantage for shy or nervous students. High rater correlations coexisted with a systematic mean difference. The study predates GenAI and did not test who authored the report.",
    "which_workflow": "WP-06—structured oral evidence of live explanation, not authorship.",
    "claimable_sentence": "A structured project interview yielded reproducible ratings and sampled live explanation, but it did not establish who authored the earlier report."
  },
  {
    "id": "RobertsShadbolt2014",
    "tier": "Tier 2",
    "full_citation": "Roberts, C., Shadbolt, N., Clark, T., & Simpson, P. (2014). The reliability and validity of a portfolio designed as a programmatic assessment of performance in an integrated clinical placement. BMC Medical Education, 14, 197.",
    "link": "https://doi.org/10.1186/1472-6920-14-197",
    "peer_reviewed": true,
    "population": "257 graduate-entry medical students completing an eight-week integrated primary and community clinical placement at the University of Sydney; 372 raters contributed across six portfolio tasks.",
    "design": "Secondary analysis of deidentified records. The portfolio combined two written projects, a peer-teaching case, two workplace supervisor assessments, and a 60-item MCQ. Generalizability theory decomposed score variance, and a decision study modeled measurement error around a 57% pass standard.",
    "finding_verbatim": "It has modest precision in assessing students' achievement of a competency standard.",
    "effect_size": "Mean portfolio mark 75.56 (SD=6.68). For one student on one task, 11% of score variance reflected student capability, 49% student-by-task context specificity, and 29% rater subjectivity. With six tasks, reliability=.43 and absolute SEM=4.74 points; the authors reported a 95% CI of ±9.30 points around the 57% cut (47.70%–66.30%).",
    "conditions_and_limits": "Interpretation depended on six heterogeneous tasks, multiple rater types, rubrics, orientation, standard setting, and supplementary assessment for low scores. The portfolio combined products, observed performance, and workplace judgments rather than relying on one artifact.",
    "criticism": "One medical school, one placement and year; secondary observational data; partially crossed raters; task-total rather than item-level data. The six-task portfolio did not meet conventional reliability, and modeled reliability of .70 required 20 tasks, judged infeasible. Written components could be outsourced, and neither identity nor authorship was tested.",
    "which_workflow": "WP-06—programmatic assessment and triangulation.",
    "claimable_sentence": "A six-task medical portfolio showed why one task is a noisy basis for high-stakes inference and why multiple aligned samples are needed."
  },
  {
    "id": "Ellis2020",
    "tier": "Tier 3",
    "full_citation": "Ellis, C., van Haeringen, K., Harper, R., Bretag, T., Zucker, I., McBride, S., Rozenberg, P., Newton, P., & Saddiqui, S. (2020). Does authentic assessment assure academic integrity? Evidence from contract cheating data. Higher Education Research & Development, 39(3), 454–469.",
    "link": "https://doi.org/10.1080/07294360.2019.1680956",
    "peer_reviewed": true,
    "population": "Two positive-case datasets: 221 assignment orders placed on academic custom-writing sites and 198 higher-education assessment tasks in which contract cheating had been detected.",
    "design": "Descriptive content analysis. Assessment authenticity was coded using five literature-derived factors: frequency, fidelity, complexity, real-world impact, and feed-forward.",
    "finding_verbatim": "assessment tasks with no, some, or all of the five authenticity factors are routinely outsourced by students",
    "effect_size": "No causal or comparative effect size and no confidence interval were reported. The paper reports dataset counts and authenticity characteristics among already outsourced or detected cases.",
    "conditions_and_limits": "The finding applies to contract-cheating cases available to the researchers. It establishes that authenticity factors do not make outsourcing impossible; it does not estimate prevalence or compare rates for authentic and non-authentic tasks.",
    "criticism": "Both datasets were conditioned on outsourcing or detection and lacked a denominator or non-cheated comparison, so they cannot estimate risk reduction or causality. The study concerns human third-party contract cheating, not GenAI. No authorship-detection performance was evaluated.",
    "which_workflow": "WP-06—counterevidence to integrity claims for authentic assessment.",
    "claimable_sentence": "Authentic task features may improve alignment, but positive-case contract-cheating data show that they do not by themselves assure who produced the artifact."
  },
  {
    "id": "AACSB2026",
    "tier": "Official normative source",
    "full_citation": "AACSB International. (2026). AACSB Global Standards for Business Education (effective July 1, 2026), Standard 5, pp. 74–79; glossary, pp. 140, 142.",
    "link": "https://www.aacsb.edu/educators/global-standards",
    "peer_reviewed": false,
    "population": "AACSB-accredited and applicant business schools; degree programs in scope for business accreditation.",
    "design": "Principles-based accreditation standard with binding language, bases for judgment, suggested documentation, and glossary; non-empirical.",
    "finding_verbatim": "Both direct and indirect measures are tied to clearly articulated learning competencies or objectives, as opposed to simple satisfaction measures.",
    "effect_size": "Not applicable—normative accreditation document; no empirical effect or confidence interval.",
    "conditions_and_limits": "Standard 5.1 requires direct and indirect evidence across the school's overall portfolio, tied to competencies or objectives and used for curricular improvement. Individual programs may use one type with a rationale. Direct measures involve individually observed learner performance or work. The 2026–27 cycle is phased, with many reviews continuing under 2020 standards.",
    "criticism": "The standard does not validate an instrument, establish authorship, prescribe oral assessment, or endorse automated scoring. Its effective date does not mean every 2026–27 review uses it.",
    "which_workflow": "WP-06—AACSB assurance of learning.",
    "claimable_sentence": "AACSB requires competency-aligned direct and indirect evidence across a school's portfolio and documented use of results for curricular improvement."
  },
  {
    "id": "ABET2026",
    "tier": "Official normative source",
    "full_citation": "ABET Engineering Accreditation Commission. (2025). Criteria for Accrediting Engineering Programs, 2026–2027, definitions and Criterion 4.",
    "link": "https://www.abet.org/2026-2027_eac_criteria/",
    "peer_reviewed": false,
    "population": "Engineering programs accredited by or seeking accreditation from ABET's Engineering Accreditation Commission.",
    "design": "Binding programmatic accreditation criteria and definitions; non-empirical.",
    "finding_verbatim": "Effective assessment uses relevant direct, indirect, quantitative and qualitative measures as appropriate to the outcome being measured.",
    "effect_size": "Not applicable—normative accreditation document; no empirical effect or confidence interval.",
    "conditions_and_limits": "Criterion 4 requires regular, appropriate, documented assessment and evaluation of student-outcome attainment and systematic use of results for improvement. Sampling is permitted; measures are selected as appropriate to the outcome.",
    "criticism": "The criteria do not require one measure from every category, prescribe a medium, identify an authorship test, or validate automated scoring.",
    "which_workflow": "WP-06—ABET evidence requirements.",
    "claimable_sentence": "ABET requires documented outcome assessment with measures appropriate to the outcome and systematic use of findings for improvement."
  },
  {
    "id": "RogersABET",
    "tier": "Official implementation guidance",
    "full_citation": "Rogers, G. (n.d.). Direct and indirect assessments. ABET Assessment Resources.",
    "link": "https://assessment.abet.org/planning_article/direct-and-indirect-assessment-methods/",
    "peer_reviewed": false,
    "population": "Faculty and programs designing program-level student-outcome assessment.",
    "design": "Explanatory guidance with a classification table; non-empirical and nonbinding.",
    "finding_verbatim": "The key is whether the results are a product of direct observation or the self-report or opinion of the respondent.",
    "effect_size": "Not applicable—guidance; no empirical effect or confidence interval.",
    "conditions_and_limits": "Direct examples include exams, demonstrations, oral and written reports, oral exams, observed performance, portfolios, simulations, and standardized exams. Classification depends on how a method is used; interviews, focus groups, and surveys usually supply indirect perceptions.",
    "criticism": "This is guidance, not accreditation-criterion text. It supplies no reliability, validity, fairness, authorship, or automated-scoring evidence for any listed method.",
    "which_workflow": "WP-06—ABET examples of evidence.",
    "claimable_sentence": "ABET guidance recognizes oral exams as one possible direct measure of observed knowledge while leaving validation to the program."
  },
  {
    "id": "HLC2025",
    "tier": "Official normative source",
    "full_citation": "Higher Learning Commission. (2024). Criteria for Accreditation (CRRT.B.10.010), revised June 2024, effective September 1, 2025.",
    "link": "https://www.hlcommission.org/accreditation/policies/criteria/",
    "peer_reviewed": false,
    "population": "HLC-accredited and candidate institutions and institutions seeking HLC accreditation.",
    "design": "Binding institutional accreditation criteria; non-empirical.",
    "finding_verbatim": "The institution improves the quality of educational programs based on its assessment of student learning.",
    "effect_size": "Not applicable—normative accreditation policy; no empirical effect or confidence interval.",
    "conditions_and_limits": "Core Component 3.E operates within Criterion 3's expectation of responsibility for educational quality across modalities and locations. Institutional mission informs how compliance is demonstrated.",
    "criticism": "The criterion does not prescribe direct or indirect measures, a particular artifact, an individual-level design, an authorship test, or automated scoring.",
    "which_workflow": "WP-06—HLC learning-assessment requirement.",
    "claimable_sentence": "HLC requires institutions to show that assessment of student learning results in improvement of educational programs."
  },
  {
    "id": "HLCEvidence2024",
    "tier": "Official implementation guidance",
    "full_citation": "Higher Learning Commission. (2024, September). Providing Evidence for the Criteria for Accreditation: Updated for Criteria effective September 1, 2025.",
    "link": "https://download.hlcommission.org/ProvidingEvidence-2025Criteria_INF.pdf",
    "peer_reviewed": false,
    "population": "Institutions preparing accreditation evidence and HLC peer reviewers.",
    "design": "Nonbinding examples of evidence aligned with the Criteria; explicitly not a checklist.",
    "finding_verbatim": "Institutions are not required to use these examples and peer reviewers should defer to institutional determinations.",
    "effect_size": "Not applicable—implementation resource; no empirical effect or confidence interval.",
    "conditions_and_limits": "For 3.E, examples include minutes showing use and action, curriculum maps, rubrics, benchmarking, student work, employer or graduate-school data, outcomes, reports, faculty involvement, assessment plans, and documents using direct measures.",
    "criticism": "The list does not establish that any example is sufficient or psychometrically sound. Relevance, persuasiveness, mission, student population, and institutional context remain controlling.",
    "which_workflow": "WP-06—HLC examples of evidence.",
    "claimable_sentence": "HLC offers nonexclusive evidence examples and leaves institutions responsible for showing their relevance and use."
  },
  {
    "id": "MSCHE2026",
    "tier": "Official normative source",
    "full_citation": "Middle States Commission on Higher Education. (2026). Standards for Accreditation and Requirements of Affiliation (15th ed.; effective July 1, 2026), Standard 3 and Examples of Evidence §§3.3, 3.5.",
    "link": "https://www.msche.org/standards/standards-15/",
    "peer_reviewed": false,
    "population": "MSCHE applicant, candidate, and accredited institutions.",
    "design": "Binding accreditation standards followed by official, nonexclusive examples of evidence; non-empirical.",
    "finding_verbatim": "All student learning experiences (credit bearing or otherwise) are intentionally designed, effectively delivered, and regularly assessed by qualified professionals.",
    "effect_size": "Not applicable—normative accreditation document; no empirical effect or confidence interval.",
    "conditions_and_limits": "Section 3.5 examples include policies, documented approaches, sample instruments and analysis, results and follow-up, curriculum mapping, and assessment at course, program, and institutional levels. The examples also include guidance on acceptable and ethical AI use and mechanisms protecting academic integrity. Section 3.3 treats direct assessment and CBE as alternative program-delivery formats requiring comparable quality, support, faculty interaction, and resources.",
    "criticism": "'Direct assessment' denotes a program format, not direct evidence in the AACSB or ABET sense. The standard does not prescribe an instrument, establish authorship, or validate automated scoring. Implementation is cohort-based.",
    "which_workflow": "WP-06—MSCHE evidence requirements and alternative formats.",
    "claimable_sentence": "MSCHE requires regular assessment by qualified professionals and names AI guidance and academic-integrity mechanisms among nonexclusive evidence examples; it does not validate a particular instrument."
  },
  {
    "id": "MSCHE2023",
    "tier": "Superseded/transitional official standard",
    "full_citation": "Middle States Commission on Higher Education. (2023). Standards for Accreditation and Requirements of Affiliation (14th ed.; effective July 1, 2023), Standard V.",
    "link": "https://www.msche.org/standards/fourteenth-edition/",
    "peer_reviewed": false,
    "population": "MSCHE institutions governed by Fourteenth Edition self-study and review arrangements.",
    "design": "Binding accreditation standards for applicable transitional cohorts; non-empirical.",
    "finding_verbatim": "defensible standards for assessing whether students are achieving those outcomes",
    "effect_size": "Not applicable—normative accreditation document; no empirical effect or confidence interval.",
    "conditions_and_limits": "Fourteenth Edition Standard V also required organized and systematic assessment, faculty or qualified-professional involvement, use of disaggregated results for improvement, oversight of third-party assessment services, and review of assessment processes.",
    "criticism": "The Fifteenth Edition became effective July 1, 2026 and does not retain the quoted phrase. It should not be described as an unqualified current Fifteenth Edition requirement, though it remains relevant to transitional cohorts.",
    "which_workflow": "WP-06—edition-specific interpretation of MSCHE terminology.",
    "claimable_sentence": "The phrase 'defensible standards' comes from MSCHE's Fourteenth Edition, not its Fifteenth Edition."
  },
  {
    "id": "Tufts2025",
    "tier": "Tier 3",
    "full_citation": "Tufts, B., Zhao, X., & Li, L. (2025). A practical examination of AI-generated text detectors for large language models. In Findings of the Association for Computational Linguistics: NAACL 2025 (pp. 4839–4856). Association for Computational Linguistics.",
    "link": "https://doi.org/10.18653/v1/2025.findings-naacl.271",
    "peer_reviewed": true,
    "population": "Seven tasks—question answering, summarization, dialogue, code, scientific abstracts, peer reviews, and translation—using four generators and English, Spanish, French, and Chinese where applicable. Each evaluated slice used 500 human texts and 100 generated texts per model.",
    "design": "Out-of-distribution benchmark of three trained and four zero-shot detectors on datasets and generators excluded from their prior evaluations, with ordinary prompts, 'sound human' prompts, and AI rewriting of human responses.",
    "finding_verbatim": "These detectors perform poorly in certain settings, with TPR@.01 as low as 0%.",
    "effect_size": "Table 4, entire dataset at FPR=.01: RADAR TPR=.05/AUROC=.6009; Fast-DetectGPT .49/.8405; Wild .11/.6841; PHD .08/.6790; LogRank .09/.7763; T5Sentinel .03/.5179; Binoculars .58/.8485. No confidence intervals were reported.",
    "conditions_and_limits": "Thresholds were evaluated at fixed false-positive rates of .01, .05, and .10; generators were GPT-4o, Llama-3-Instruct-8B, Mistral-Instruct-v0.3, and Phi-3-Mini-Instruct-4k.",
    "criticism": "This was a technical benchmark, not authentic student assessment; human data settings differ from proposed deployment, and internet corpora may contain AI text. Operational tool versions will change. The paper demonstrates that AUROC can conceal poor low-FPR sensitivity, not that every detector fails in every task. No exact independent replication was located.",
    "which_workflow": "WP-06—threshold choice, domain shift, and model drift.",
    "claimable_sentence": "Across unfamiliar tasks, languages, and models, several detectors missed most generated text when calibrated to keep false positives at 1%."
  },
  {
    "id": "AlAli2026",
    "tier": "Tier 3",
    "full_citation": "Al Ali, A., Helcl, J., & Libovický, J. (2026). Different time, different language: Revisiting the bias against non-native speakers in GPT detectors. In Proceedings of the 19th Conference of the European Chapter of the Association for Computational Linguistics, Volume 4: Student Research Workshop (pp. 277–291). Association for Computational Linguistics.",
    "link": "https://doi.org/10.18653/v1/2026.eacl-srw.20",
    "peer_reviewed": true,
    "population": "Czech essays from 450 moderately proficient non-native writers and 450 native school-age writers; 29 proficient non-native and 29 advanced native writers; additional Czech corpora. Liang's TOEFL-91 and Hewlett English corpora were re-tested with one contemporary commercial detector.",
    "design": "Explicit follow-up to Liang using entropy analysis and three detector families: TF-IDF naïve Bayes, fine-tuned RobeCzech, and closed-source Plagramme.",
    "finding_verbatim": "None of the detectors exhibited a systematic bias against non-native speakers of Czech.",
    "effect_size": "Czech entropy was M=3.48, SD=.57 for non-native versus M=3.19, SD=.49 for native-youth essays, p<10^-14. Naïve Bayes false-positive rates were 8.7% versus 6.2%, p>.23; Plagramme Czech rates were 2.0% versus 1.0%, p>.11. On Liang's English corpora, Plagramme rates were 23.1% for TOEFL-91 versus 0% for Hewlett. No confidence intervals were reported.",
    "conditions_and_limits": "Results depend on Czech morphology, proficiency and error patterns, detector versions, and task matching. Plagramme was limited to at most 100 randomly selected texts per corpus and 512 words.",
    "criticism": "This was not an exact replication because language and tool generation changed. The proficient Czech groups contained only 29 texts each; the native comparator was school-age; custom detectors were brittle; Plagramme was proprietary. The English re-test retained a substantial gap. Weak entropy-output correlations challenge a general claim that current detectors primarily depend on perplexity.",
    "which_workflow": "WP-06—bias, criticism, and temporal/language dependence.",
    "claimable_sentence": "A 2026 follow-up found no systematic Czech non-native-speaker bias, while one contemporary detector retained a nontrivial false-positive gap on Liang's English corpora."
  },
  {
    "id": "Hadra2026",
    "tier": "Tier 3",
    "full_citation": "Hadra, M., Cambridge, K., & Mesbah, M. (2026). Evaluating the accuracy and reliability of AI content detectors in academic contexts. International Journal for Educational Integrity, 22, 4.",
    "link": "https://doi.org/10.1007/s40979-026-00213-1",
    "peer_reviewed": true,
    "population": "192 texts, 48 each: authentic pre-August-2022 EFL foundation-program coursework, professional XSum texts, GPT-4.1/Claude 3 Opus texts, and constructed 50/50 hybrid texts.",
    "design": "Three-class evaluation of Turnitin and Originality using Human=0–20%, Hybrid=21–79%, and AI≥80%, with performance compared by text length, genre, and EFL/professional source.",
    "finding_verbatim": "Neither Turnitin nor Originality can currently be considered sufficiently reliable as AI detection solutions in academic settings.",
    "effect_size": "Macro accuracy was .69 for Originality and .61 for Turnitin; macro recall .60 versus .51; both macro F1<.55. Genre accuracy was .86 humanities versus .51 science for Turnitin, chi-square(1)=19.109, authors report p=.000; .96 versus .58 for Originality, chi-square(1)=26.429, p=.000. Turnitin EFL/professional Fisher p=.50; Originality one-sided p=.058. No confidence intervals were reported.",
    "conditions_and_limits": "Two proprietary tools; January–May 2025 AI outputs; predetermined three-class thresholds; 135 science and 57 humanities texts; 23 short, 144 medium, and 25 long texts.",
    "criticism": "EFL coursework versus professional BBC/XSum writing confounds proficiency with task, genre, provenance, and author population. Originality's bias result was nonsignificant and one-sided, so it is suggestive rather than confirmatory. Equal class construction does not resemble campus prevalence; hybrid texts were artificially fixed at 50/50; thresholds were not calibrated on this corpus.",
    "which_workflow": "WP-06—authentic EFL writing, hybrid failure, and contextual bias.",
    "claimable_sentence": "In one 192-text study, detector performance varied by genre and length; Turnitin showed no measurable EFL disparity, while Originality showed only a nonsignificant adverse trend."
  },
  {
    "id": "Scarfe2024",
    "tier": "Tier 3",
    "full_citation": "Scarfe, P., Watcham, K., Clarke, A., & Roesch, E. (2024). A real-world test of artificial intelligence infiltration of a university examinations system: A ‘Turing Test’ case study. PLOS ONE, 19(6), e0305354.",
    "link": "https://doi.org/10.1371/journal.pone.0305354",
    "peer_reviewed": true,
    "population": "Five undergraduate psychology modules across three years at one UK university; 63 inserted AI submissions and 1,134 authentic student submissions.",
    "design": "Blind real-world infiltration study in live online take-home examinations. Wholly GPT-4-generated answers were submitted under fictitious student identities and marked through the ordinary university system.",
    "finding_verbatim": "We found that 94% of our AI submissions were undetected.",
    "effect_size": "Markers did not flag 94% of the 63 inserted AI submissions. AI answers averaged half a grade boundary above authentic submissions; the authors reported an 83.4% probability that the AI set would outperform an equally sized random set of authentic submissions.",
    "conditions_and_limits": "Wholly GPT-4-written answers were inserted into 2022–23 assessments at approximately 5% experimental prevalence. Detection meant unsolicited marker concern during ordinary grading, not a prespecified classifier threshold.",
    "criticism": "One institution, discipline, academic year, and model family; only 63 AI submissions and three finalist-level submissions. The result is not a detector false-negative rate, a misconduct prevalence estimate, or evidence about edited, hybrid, or disclosed AI use.",
    "which_workflow": "WP-06—ecological evidence that generated work can pass ordinary marking without concern.",
    "claimable_sentence": "In one blind study of five UK psychology modules, ordinary markers did not flag 94% of wholly AI-written exam submissions."
  },
  {
    "id": "WeberWulff2023",
    "tier": "Tier 3",
    "full_citation": "Weber-Wulff, D., Anohina-Naumeca, A., Bjelobaba, S., Foltýnek, T., Guerrero-Dib, J., Popoola, O., Šigut, P., & Waddington, L. (2023). Testing of detection tools for AI-generated text. International Journal for Educational Integrity, 19, 26.",
    "link": "https://doi.org/10.1007/s40979-023-00146-z",
    "peer_reviewed": true,
    "population": "Fourteen AI-text detectors evaluated on human, machine-translated, AI-generated, edited, paraphrased, and obfuscated texts.",
    "design": "Comparative diagnostic-accuracy evaluation of 12 public tools plus Turnitin and PlagiarismCheck under several text-transformation conditions.",
    "finding_verbatim": "The available detection tools are neither accurate nor reliable.",
    "effect_size": "In the authors' inclusive binary analysis, the highest tool accuracies remained below 80%; performance varied substantially by tool and text condition. No universal campus operating characteristic was estimated.",
    "conditions_and_limits": "Tools and models available in 2023; small purpose-built document set; several transformations and author-defined mappings of detector outputs.",
    "criticism": "Tool versions age rapidly. Accuracy depends on threshold, prevalence, language, genre, text length, and scoring rule. The study was not a live disciplinary misconduct process and does not establish permanent performance for later detectors.",
    "which_workflow": "WP-06—comparative detector reliability and transformation sensitivity.",
    "claimable_sentence": "In a 2023 comparison, detector accuracy varied across tools and text transformations; the result is a time-bounded benchmark, not a universal detector verdict."
  },
  {
    "id": "Liang2023",
    "tier": "Tier 3",
    "full_citation": "Liang, W., Yuksekgonul, M., Mao, Y., Wu, E., & Zou, J. (2023). GPT detectors are biased against non-native English writers. Patterns, 4(7), 100779.",
    "link": "https://doi.org/10.1016/j.patter.2023.100779",
    "peer_reviewed": true,
    "population": "Ninety-one human-written TOEFL essays by non-native English writers and 88 essays by US eighth-grade students, evaluated with seven detectors.",
    "design": "Comparative diagnostic and perturbation study examining false-positive classifications and the relation between textual predictability and detector output.",
    "finding_verbatim": "They incorrectly labeled more than half of the TOEFL essays as ‘AI-generated.’",
    "effect_size": "The average false-positive rate across detectors was 61.3% on TOEFL essays; all seven flagged 19.8% and at least one flagged 97.8%. The paper did not supply confidence intervals for these headline proportions.",
    "conditions_and_limits": "Specific 2023 detectors and short essay corpora; the native and non-native corpora differed in task, age, and provenance as well as language background.",
    "criticism": "The convenience-group comparison is confounded and is not a clean causal estimate of language status. The large disparity is consequential for those texts and tools but cannot be treated as an invariant rate for every detector or deployment.",
    "which_workflow": "WP-06—demonstrated subgroup disparity and the limits of a universal bias claim.",
    "claimable_sentence": "Seven 2023 detectors falsely labeled the tested non-native-English TOEFL essays at a high average rate, under a comparison confounded by task and provenance."
  },
  {
    "id": "Huxham2012",
    "tier": "Tier 2",
    "full_citation": "Huxham, M., Campbell, F., & Westwood, J. (2012). Oral versus written assessments: A test of student performance and attitudes. Assessment & Evaluation in Higher Education, 37(1), 125–136.",
    "link": "https://doi.org/10.1080/02602938.2010.515012",
    "peer_reviewed": true,
    "population": "Two biology cohorts at Edinburgh Napier University: 99 students randomized between modes and 29 students completing both modes.",
    "design": "Randomized response-mode comparison plus a within-student comparison and an attitude study using comparable biology questions delivered orally or in writing.",
    "finding_verbatim": "Both cohorts showed highly significant differences in the mean marks achieved, with better performance in the oral assessment.",
    "effect_size": "Higher oral marks were reported in both cohorts; the accessible source record did not report a standardized effect or confidence interval.",
    "conditions_and_limits": "The study compared performance modes at one institution. It did not assign an oral learning intervention or measure later retention or transfer.",
    "criticism": "Higher marks do not identify the more valid mode or establish more learning. Clarification, response demands, examiner interaction, anxiety, and writing burden remain plausible explanations.",
    "which_workflow": "WP-06—response-mode evidence and boundary on learning and validity claims.",
    "claimable_sentence": "Students in two biology cohorts earned higher marks orally than in writing, a mode difference that does not by itself establish learning or validity."
  },
  {
    "id": "Nallaya2024",
    "tier": "Tier 2",
    "full_citation": "Nallaya, S., Gentili, S., Weeks, S., & Baldock, K. (2024). The validity, reliability, academic integrity and integration of oral assessments in higher education: A systematic review. Issues in Educational Research, 34(2), 629–646.",
    "link": "https://www.iier.org.au/iier34/nallaya.pdf",
    "peer_reviewed": true,
    "population": "Seventeen peer-reviewed higher-education studies published from 2010 through 2021 across multiple disciplines and countries.",
    "design": "PRISMA systematic review with narrative synthesis of validity, reliability, academic-integrity, and integration evidence for oral assessment.",
    "finding_verbatim": "Oral assessments can be both valid and reliable.",
    "effect_size": "No pooled effect size or confidence interval; the review narratively summarized study-level evidence and conditions.",
    "conditions_and_limits": "Recurring conditions included clear criteria, authentic alignment, assessor training, moderation, practice, and inclusive design.",
    "criticism": "Only 17 heterogeneous studies met the criteria. The review did not pool effects or provide a formal study-level risk-of-bias synthesis, and many integrity claims were perceptual or local. It did not validate oral assessment as an authorship test.",
    "which_workflow": "WP-06—higher-education oral-assessment validity, reliability, accessibility, and authorship boundary.",
    "claimable_sentence": "Oral assessment can support valid and reliable interpretations under defined design conditions; validity is not an inherent property of the format."
  },
  {
    "id": "Kane2013",
    "tier": "Tier 1 conceptual foundation",
    "full_citation": "Kane, M. T. (2013). Validating the interpretations and uses of test scores. Journal of Educational Measurement, 50(1), 1–73.",
    "link": "https://doi.org/10.1111/jedm.12000",
    "peer_reviewed": true,
    "population": "Conceptual paper in educational measurement; no participant sample.",
    "design": "Argument-based validity framework specifying the inferences and assumptions connecting observed responses, scores, interpretations, and uses.",
    "finding_verbatim": "It is the proposed score interpretations and uses that are validated and not the test or the test scores.",
    "effect_size": "Not applicable—conceptual validity paper; no empirical effect or confidence interval.",
    "conditions_and_limits": "The framework requires the intended interpretation and use to be specified and treats more ambitious claims as requiring more support.",
    "criticism": "The paper supplies a validation logic, not evidence that any detector, oral task, process record, or assessment architecture supports a particular AI-era inference.",
    "which_workflow": "WP-06—foundation for separating artifact quality, authorship, independent competence, and learning inferences.",
    "claimable_sentence": "Assessment evidence is evaluated against a specified interpretation and use; a stronger inference requires a stronger supporting argument."
  },
  {
    "id": "Ebrahimzadeh2026",
    "tier": "Tier 3—peer-reviewed conceptual and preliminary prototype evaluation",
    "full_citation": "Ebrahimzadeh, M., Shibani, A., & Buckingham Shum, S. (2026). Coauthorship integrity: Reconceptualising assessment validity for the age of generative artificial intelligence. Computers and Education: Artificial Intelligence, 10, 100609.",
    "link": "https://doi.org/10.1016/j.caeai.2026.100609",
    "peer_reviewed": true,
    "population": "A prototype AI-mediated viva evaluated by educators, some of whom were assessment experts. The accessible primary report did not state the exact participant count; no student-validation sample was reported there.",
    "design": "Conceptual validity proposal, technical prototype, and preliminary qualitative and quantitative expert evaluation of LLM-generated comprehension and dialogic questions.",
    "finding_verbatim": "We argue for a new category of evidence for validity, called Coauthorship Integrity.",
    "effect_size": "No student effect, authorship-diagnostic accuracy, subgroup estimate, or confidence interval was available in the accessible primary report.",
    "conditions_and_limits": "The paper addresses non-proctored writing and asks whether students understand AI-assisted content. The accessible primary material supports the stated concept and prototype scope, not a complete appraisal of the expert-evaluation statistics.",
    "criticism": "Preliminary expert evaluation does not establish validity with students, independent competence, authorship detection, learning, scalability, or fairness. Complete primary full text could not be retrieved during the update, so no numerical result carries a claim in WP-06.",
    "which_workflow": "WP-06—direct prior art on coauthorship integrity and AI-mediated viva design.",
    "claimable_sentence": "A peer-reviewed 2026 paper proposes coauthorship integrity and an AI-mediated viva, but its accessible evidence does not validate student authorship or learning inferences."
  },
  {
    "id": "GreenawayQuinceMunn2026",
    "tier": "Official regulator-hosted case study",
    "full_citation": "Greenaway, R., Quince, Z., & Munn, J. (2026). Adapting assessment in the age of generative AI: The Assessment Adaptation Model. TEQSA Academic Integrity Toolkit.",
    "link": "https://www.teqsa.gov.au/guides-resources/protecting-academic-integrity/academic-integrity-toolkit/risks-academic-integrity-ai/adapting-assessment-age-generative-ai-assessment-adaptation-model",
    "peer_reviewed": false,
    "population": "Southern Cross University assessment-design practice presented through TEQSA's revised Academic Integrity Toolkit; no outcome-evaluation sample.",
    "design": "Non-empirical case study describing seven connected components across the assessment lifecycle: design, analyse, act, inform, educate, check, and evaluate.",
    "finding_verbatim": "A security risk matrix is a conversation starting point to reconsider assessment design, it is not a definitive measure of assessment security.",
    "effect_size": "Not applicable—official practice case study; no comparative effect or confidence interval.",
    "conditions_and_limits": "The page was last updated 19 March 2026 and describes one university's model within Australian regulator-hosted guidance.",
    "criticism": "The source presents an assessment-adaptation framework rather than causal evidence that the model reduces misconduct, improves learning, or produces valid authorship decisions.",
    "which_workflow": "WP-06—current holistic redesign guidance and explicit boundary on security matrices.",
    "claimable_sentence": "A current regulator-hosted model treats assessment integrity as a lifecycle design problem and explicitly rejects its risk matrix as a definitive security measure."
  },
  {
    "id": "TEQSA2026RoleGuide",
    "tier": "Official regulator guidance",
    "full_citation": "Tertiary Education Quality and Standards Agency. (2026). Role-specific guide to promoting academic integrity, and managing and investigating academic misconduct.",
    "link": "https://www.teqsa.gov.au/sites/default/files/2026-05/role-specific-guide-to-promoting-academic-integrity.pdf",
    "peer_reviewed": false,
    "population": "Australian higher-education providers and staff roles responsible for academic-integrity education, evidence gathering, investigation, and decision-making.",
    "design": "Normative regulator guidance describing reasonable cause, balance of probabilities, evidentiary strength, natural justice, and procedural fairness; non-empirical.",
    "finding_verbatim": "The more serious the potential outcome, the stronger the evidence is required to support the finding/s.",
    "effect_size": "Not applicable—official procedural guidance; no diagnostic effect or confidence interval.",
    "conditions_and_limits": "Published in May 2026 for the Australian regulatory context; institutional policies govern the operational process.",
    "criticism": "The guidance treats inability to answer assignment questions as potentially consequential evidence, but it does not validate oral discrepancy as an authorship diagnostic or report false-positive and false-negative rates.",
    "which_workflow": "WP-06—current due-process, evidence-strength, and oral-questioning guidance.",
    "claimable_sentence": "Current Australian guidance links evidentiary strength to consequence and requires procedural fairness; it does not supply diagnostic accuracy for oral authorship inference."
  },
  {
    "id": "TEQSA2025InvestigationProcess",
    "tier": "Official regulator student guidance",
    "full_citation": "Tertiary Education Quality and Standards Agency. (2025). Student academic misconduct—the investigation process.",
    "link": "https://www.teqsa.gov.au/students/student-academic-misconduct-resources/investigation-process",
    "peer_reviewed": false,
    "population": "Students in Australian higher education who receive an academic-misconduct allegation; no empirical sample.",
    "design": "Student-facing description of a typical institutional process from preliminary communication through notice, response, investigation, decision, outcome, and appeal; non-empirical.",
    "finding_verbatim": "The student is notified of the allegations and given a chance to respond.",
    "effect_size": "Not applicable—official procedural guidance; no empirical effect or confidence interval.",
    "conditions_and_limits": "Page last updated 18 November 2025. It expressly directs students to the policy and process of their own institution.",
    "criticism": "The page describes a typical process and cannot establish that all providers implement it consistently, that any evidentiary instrument is valid, or that appeals correct error at a known rate.",
    "which_workflow": "WP-06—notice, response, outcome, and appeal as components of due process.",
    "claimable_sentence": "Current student-facing regulator guidance describes notice, an opportunity to respond, an evidence-gathering process, an outcome, and an avenue of appeal."
  }
]
