[
  {
    "id": "AcostaPradoEtAl2026LeadershipHAI",
    "tier": "Tier 3",
    "full_citation": "Acosta-Prado, J. C., Camargo, J. P., Zárate-Torres, R. A., & Rey-Sarmiento, C. F. (2026). Leadership and human–AI collaboration: A measurement scale. Behavioral Sciences, 16(7), 1208.",
    "link": "https://doi.org/10.3390/bs16071208",
    "peer_reviewed": true,
    "population": "170 self-administered online responses from Colombian companies spanning organization sizes and economic sectors.",
    "design": "Instrument-development study using content evidence, exploratory factor analysis, confirmatory factor analysis, and internal-consistency estimates for a 30-item five-point self-report scale.",
    "finding_verbatim": "The results show that the proposed measurement scale meets the psychometric properties required of a social-science instrument.",
    "effect_size": "For the six-factor human–AI collaboration CFA, CFI = .970, TLI = .965, RMSEA = .074, and SRMR = .057; reported alpha ranged .763–.881 and omega .784–.891 across dimensions.",
    "conditions_and_limits": "Items assess respondent perceptions, including balance between human judgment and AI output, assignment of final responsibility, and preservation of individual autonomy.",
    "criticism": "EFA and CFA used the same n = 170 sample; there was no independent cross-validation, behavioral trace, episode-level unit, external criterion, or predictive validation. Positive-worded self-reports are vulnerable to common-response bias.",
    "which_workflow": "WP-08A Phase A: construct comparison and field-level prior-art boundary.",
    "claimable_sentence": "A self-report scale already measures perceived human–AI balance, responsibility, and autonomy; behavioral episode attribution remains a separate measurement task."
  },
  {
    "id": "Ali2026",
    "tier": "Tier 3",
    "full_citation": "Ali, M. S. (2026). From assistants to agents: A relational framework for human–AI co-agency. AI and Ethics, 6(3), Article 280.",
    "link": "https://doi.org/10.1007/s43681-026-01111-5",
    "peer_reviewed": true,
    "population": "No empirical participant sample.",
    "design": "Conceptual and normative relational framework organized around initiative, decision scope, oversight, and responsibility attribution.",
    "finding_verbatim": "The article frames human–AI agency as “co-agency” across initiative, decision scope, oversight, and responsibility.",
    "effect_size": "Not applicable; no empirical test or numerical effect is reported.",
    "conditions_and_limits": "General human–AI relations rather than educational or entrepreneurial traces. The paper provides no episode boundaries, behavioral anchors, opportunity rules, reliability, or criterion evidence.",
    "criticism": "It occupies the conceptual space but cannot validate actor attribution or a score. Normative responsibility and visibly enacted commitment should not be collapsed.",
    "which_workflow": "Driver’s Seat—relational-agency precursor and construct-boundary source.",
    "claimable_sentence": "Initiative, decision scope, oversight, and responsibility are established co-agency dimensions; Driver’s Seat must contribute an operational test rather than rename them."
  },
  {
    "id": "AngelopoulosEtAl2023",
    "tier": "Tier 3",
    "full_citation": "Angelopoulos, S., Bendoly, E., Fransoo, J., Hoberg, K., Ou, C., & Tenhiälä, A. (2023). Digital transformation in operations management: Fundamental change through agency reversal. Journal of Operations Management, 69(6), 876–889.",
    "link": "https://doi.org/10.1002/joom.1271",
    "peer_reviewed": true,
    "population": "No empirical participant sample.",
    "design": "Conceptual account of agency reversal along a human-driven/technology-supported to technology-driven/human-supported spectrum.",
    "finding_verbatim": "The article names the shift “agency reversal.”",
    "effect_size": "Not applicable; no empirical effect, sample, or psychometric coefficient is reported.",
    "conditions_and_limits": "Operations-management framing and broad digital transformation; the article calls for future operationalization.",
    "criticism": "It is not an empirical study and cannot validate Driver’s Seat, but it does establish that technology may direct actions humans carry out.",
    "which_workflow": "Driver’s Seat—authority-reversal precursor and automation/augmentation boundary.",
    "claimable_sentence": "Agency reversal already describes technology directing moves performed by humans; Driver’s Seat must operationalize that possibility rather than rename it."
  },
  {
    "id": "BairdMaruping2021",
    "tier": "Tier 3",
    "full_citation": "Baird, A., & Maruping, L. M. (2021). The next generation of research on IS use: A theoretical framework of delegation to and from agentic IS artifacts. MIS Quarterly, 45(1), 315–341.",
    "link": "https://doi.org/10.25300/MISQ/2021/15882",
    "peer_reviewed": true,
    "population": "No empirical participant sample.",
    "design": "Conceptual IS-delegation framework with the human–agentic-IS-artifact dyad as the elemental unit. It organizes delegation around agent endowments, preferences, and roles and the mechanisms of appraisal, distribution, and coordination.",
    "finding_verbatim": "The authors “introduce delegation ... as a foundational and powerful lens” for human–agentic-artifact relationships.",
    "effect_size": "Not applicable; the article develops theory and testable-model guidance but reports no empirical effect.",
    "conditions_and_limits": "General information-systems theory rather than generative-AI dialogue, entrepreneurship, education, or a validated behavioral measure. Rights and responsibilities are theorized, not trace-coded.",
    "criticism": "The framework establishes delegation as prior conceptual territory but cannot validate episode boundaries, actor attribution, opportunity rules, score reliability, or links to quality and learning.",
    "which_workflow": "WP-08A/B: foundational delegation and rights/responsibilities precursor for actor allocation.",
    "claimable_sentence": "Baird and Maruping make appraisal, distribution, and coordination of delegation in human–agentic-artifact dyads established IS theory, not a new Driver’s Seat idea."
  },
  {
    "id": "BastaniEtAl2025",
    "tier": "Tier 1",
    "full_citation": "Bastani, H., Bastani, O., Sungu, A., Ge, H., Kabakcı, Ö., & Mariman, R. (2025). Generative AI without guardrails can harm learning: Evidence from high school mathematics. Proceedings of the National Academy of Sciences, 122(26), e2422633122.",
    "link": "https://doi.org/10.1073/pnas.2422633122",
    "peer_reviewed": true,
    "population": "Nearly 1,000 students in grades 9–11 at one private high school in Turkey.",
    "design": "Preregistered classroom-cluster randomized trial across four sessions comparing control, generic GPT-4, and a teacher-grounded GPT Tutor that supplied hints and withheld direct answers; assisted practice was followed by immediate unassisted exams.",
    "finding_verbatim": "The authors conclude that “generative AI without guardrails can harm learning.”",
    "effect_size": "Assisted practice: GPT Base +.137 on the normalized outcome, about 48% relative to the control mean; GPT Tutor +.361, about 127%. Unassisted exam: GPT Base −.054, about −17%; GPT Tutor −.004, nonsignificant.",
    "conditions_and_limits": "One private high school, mathematics, short-duration intervention, classroom-level randomization, and immediate unassisted exams. The guarded tutor contained teacher-designed answers and scaffolds.",
    "criticism": "The guarded tutor removed the detected exam penalty but did not improve the unassisted exam. The study does not measure Driver’s Seat, durable learning, or long-term transfer.",
    "which_workflow": "Driver’s Seat—learning-outcome boundary and guardrail mechanism, not construct validation.",
    "claimable_sentence": "Generic GPT improved assisted practice but reduced immediate unassisted performance, while a teacher-grounded tutor removed that penalty without producing a positive unassisted-exam effect."
  },
  {
    "id": "BilalEtAl2026FinancialDelegation",
    "tier": "Tier 3",
    "full_citation": "Bilal, I. M., Wang, Y. C., Raj, A., Giovagnini, F., Tewari, P., Zhang, Y., Liou, M.-C. Z., & Zaman, Q. (2026). From information to delegation: Mapping human-AI financial decision making [Preprint]. arXiv:2608.02100.",
    "link": "https://arxiv.org/abs/2608.02100",
    "peer_reviewed": false,
    "population": "1.53 million user prompts from 6,304 voluntary opt-in ChatGPT and Gemini users in the United States and India, covering August–October 2025.",
    "design": "Behavioral trace classification of finance conversations by financial domain, intent, and a three-level decision-authority mapping (Inform, Shape, Act), using manually annotated and synthetic training data.",
    "finding_verbatim": "Consumers overwhelmingly use AI to retrieve information and shape financial judgement, while delegation of financial execution remains rare.",
    "effect_size": "Finance classifier accuracy=96.5 and F1=97.3. Intent classifier F1-micro=70.6 and F1-macro=72.3. These are classifier-performance metrics, not effects of AI use.",
    "conditions_and_limits": "The sample skews young and differs from national populations; actual transactions and off-platform decisions are unobserved. High-authority intent classes used 82.03%–96.01% synthetic training examples because such organic cases were under 1%.",
    "criticism": "Preprint; all authors are affiliated with Stripe Partners; authority is assigned by fixed intent-to-level mapping rather than independently coded right by right; annotator reliability is not reported; sessions of ten or fewer prompts are not segmented; observed requests cannot prove who retained ultimate authority.",
    "which_workflow": "WP-08A/B: large-scale trace-derived delegated decision authority in a consequential domain.",
    "claimable_sentence": "Bilal and colleagues already operationalize delegated decision authority from real conversational traces, eliminating any defensible claim that Driver's Seat is the first behavioral authority measure; the remaining gap is a validated, multi-right episode measure with explicit abstention."
  },
  {
    "id": "Bousmah2026",
    "tier": "Tier 3",
    "full_citation": "Bousmah, M. (2026). LLMography: Transforming human–AI conversations into traceability, oversight, and auditability indicators [Preprint]. arXiv.",
    "link": "https://doi.org/10.48550/arXiv.2606.29437",
    "peer_reviewed": false,
    "population": "19 anonymized engineering-student audit reports containing 462 turns; the paper also applies the prototype to its own writing process.",
    "design": "Exploratory prototype in which an LLM analyzer generates Prompt Quality, Human Direction, AI Dependency, Auditability, Final Output Traceability, and Privacy Risk indicators plus categorical labels.",
    "finding_verbatim": "“Outputs are not enough; we need the history of interaction.”",
    "effect_size": "Reported sample means were Human Direction 86.8/100, Prompt Quality 81.9/100, Auditability 72.8/100, and Final Output Traceability 77.1/100. Labels were 14 co-produced, three human-directed, two minimal-AI, and none AI-dominant or insufficient.",
    "conditions_and_limits": "Small, selected engineering-report sample and whole-conversation scores. The analyzer prompt permits the model to compute or estimate values; some single-turn records receive extreme scores.",
    "criticism": "No published formula, behavioral anchors, independent human-coded ground truth, calibration, inter-rater reliability, repeatability study, or external criterion. Reported values are prototype outputs, not validated measurements.",
    "which_workflow": "WP-08A/B: direct trace-based human-direction and provenance comparator.",
    "claimable_sentence": "LLMography already generates Human Direction and AI Dependency indicators from traces, but its exploratory LLM-estimated scores have no reported independent validation."
  },
  {
    "id": "CampbellFiske1959MTMM",
    "tier": "Tier 1",
    "full_citation": "Campbell, D. T., & Fiske, D. W. (1959). Convergent and discriminant validation by the multitrait-multimethod matrix. Psychological Bulletin, 56(2), 81–105.",
    "link": "https://doi.org/10.1037/h0046016",
    "peer_reviewed": true,
    "population": "No new single participant sample. The methodological paper develops and illustrates a validity framework using matrices of correlations among multiple traits measured by multiple methods.",
    "design": "Foundational construct-validity framework requiring both convergence across different methods for the same trait and discrimination from different traits, while making method effects inspectable.",
    "finding_verbatim": "“Measures of the same trait should correlate higher with each other than they do with measures of different traits involving separate methods.”",
    "effect_size": "Not applicable. The paper presents a validation logic and illustrative correlation matrices rather than one participant-level intervention effect.",
    "conditions_and_limits": "The approach requires at least two traits and at least two methods, with measures designed so trait and method variance can be compared.",
    "criticism": "The original criteria rely substantially on matrix inspection and correlations. Correlation is not agreement, a favorable matrix does not prove causal validity, and later latent-variable MTMM methods address limitations not resolved in the 1959 paper.",
    "which_workflow": "WP-08b B1: prospective convergent and discriminant validation of Driver’s Seat against AI contribution, authorship, output quality, and simpler authority measures.",
    "claimable_sentence": "Driver’s Seat needs evidence that it converges with measures of the same governance construct while remaining distinguishable from production volume, authorship, output quality, and adjacent traits."
  },
  {
    "id": "Chen2026AlgorithmicBoundedRationality",
    "tier": "Tier 3",
    "full_citation": "Chen, Z. S. (2026). Rethinking managerial rationality in the age of AI: A human–machine collaboration perspective on organizational decision-making. Management Decision, 1–18. Advance online publication.",
    "link": "https://doi.org/10.1108/MD-07-2025-1890",
    "peer_reviewed": true,
    "population": "Managerial decision architectures involving AI; no participant sample.",
    "design": "Conceptual synthesis of bounded rationality and socio-technical systems theory, deriving an algorithmic-bounded-rationality framework and propositions.",
    "finding_verbatim": "Human–AI collaboration represents a distinct rationality configuration rather than a midpoint between automation and human judgment.",
    "effect_size": "Not applicable; no empirical study.",
    "conditions_and_limits": "The framework distinguishes AI-led, human-first, and collaborative modes as contingent on data intensity, contextual ambiguity, accountability demands, and governance maturity.",
    "criticism": "The modes and fit propositions are untested. The article provides no behavioral coding, episode score, psychometric validation, entrepreneurship-specific judgment model, or outcome comparison.",
    "which_workflow": "WP-08a Phase A: post-cutoff managerial-mode landscape and two-axis comparison.",
    "claimable_sentence": "New managerial theory treats AI-led, human-first, and collaborative modes as contingent configurations, reinforcing the need to separate configuration description from claims of superiority."
  },
  {
    "id": "Chow1970RejectTradeoff",
    "tier": "Tier 2",
    "full_citation": "Chow, C. K. (1970). On optimum recognition error and reject tradeoff. IEEE Transactions on Information Theory, 16(1), 41–46.",
    "link": "https://doi.org/10.1109/TIT.1970.1054406",
    "peer_reviewed": true,
    "population": "Formal pattern-recognition systems; worked examples use normal and uniform distributions rather than human participants.",
    "design": "Mathematical derivation of an optimum rejection rule and the relationship between recognition error and rejection probability.",
    "finding_verbatim": "The performance of a pattern recognition system is characterized by its error and reject tradeoff.",
    "effect_size": "Not applicable; this is a theoretical result, not an empirical effect-size study.",
    "conditions_and_limits": "The result concerns classification with specified posterior probabilities and rejection costs. Rejecting more cases changes both coverage and conditional error.",
    "criticism": "Driver's Seat episodes are not ordinary labeled pattern-recognition cases, and score eligibility may itself be informative. Chow supplies an analogy and reporting principle, not validity evidence for the measure.",
    "which_workflow": "WP-08a Phase A: abstention and risk–coverage rationale.",
    "claimable_sentence": "A reject option creates an explicit error–coverage trade-off; reporting only performance among retained cases is therefore incomplete."
  },
  {
    "id": "CoreMooreZinn2003",
    "tier": "Tier 2",
    "full_citation": "Core, M. G., Moore, J. D., & Zinn, C. (2003). The role of initiative in tutorial dialogue. In Proceedings of the 10th Conference of the European Chapter of the Association for Computational Linguistics (pp. 67–74). Association for Computational Linguistics.",
    "link": "https://doi.org/10.3115/1067807.1067818",
    "peer_reviewed": true,
    "population": "Twenty-three human–human typed tutoring sessions: three trial, ten Socratic, and ten didactic; the same tutor conducted the sessions.",
    "design": "Corpus annotation study comparing initiative management in the ten Socratic and ten didactic sessions, with learning correlations across all 23 sessions and reliability coding.",
    "finding_verbatim": "“There was no direct relationship between student initiative and learning.”",
    "effect_size": "Initiative coding κ=.92 on 757 examples. Initiative versus learning gain r=−.0689, n=23, nonsignificant. Student word share r=.60, p<.005; utterance share r=.56, p<.005; tutor-question share r=.46, p<.05.",
    "conditions_and_limits": "Small human–human tutoring corpus, one tutor, typed basic-electronics dialogue, and a crude initiative code that omitted task initiative.",
    "criticism": "High coding agreement did not make initiative a learning indicator. Turn control and word share are not equivalent to substantive evaluative or commitment governance.",
    "which_workflow": "Driver’s Seat—classic measurement warning separating reliability, initiative, interactivity, and learning.",
    "claimable_sentence": "A reliably coded dialogue-initiative measure was unrelated to learning gain in this small tutoring study, showing that reliability does not establish substantive validity or learning relevance."
  },
  {
    "id": "CristofaroGiardinoMuldoon2026",
    "tier": "Tier 2",
    "full_citation": "Cristofaro, M., Giardino, P. L., & Muldoon, J. (2026). Entrepreneurial decision-making in the age of AI: Sector knowledge at the balance of intuition and analysis. Technology in Society, 85, 103200.",
    "link": "https://doi.org/10.1016/j.techsoc.2025.103200",
    "peer_reviewed": true,
    "population": "135 entrepreneurs recruited and 124 retained, with 31 cases per cell; outputs were assessed by three blinded investor raters.",
    "design": "Controlled 90-minute GPT-4 entrepreneurial decision task with a factorial cell structure. Pre-existing sector knowledge was measured rather than experimentally assigned.",
    "finding_verbatim": "The authors report enhanced “contextual understanding” alongside lower novelty and innovation under AI-assisted conditions.",
    "effect_size": "No numerical effect is entered because the published main-effect and cell-mean tables conflict in direction and magnitude. The discrepancy requires author or publisher clarification.",
    "conditions_and_limits": "One controlled task, short duration, output/rater outcomes, and measured sector expertise. No trace-based actor allocation or independent learning/transfer test.",
    "criticism": "The published numerical tables conflict, so effects should not be quoted, and sector knowledge should not be described as randomized.",
    "which_workflow": "Driver’s Seat—entrepreneurship-adjacent criterion candidate and expertise contingency, currently unsuitable for exact-effect validation.",
    "claimable_sentence": "The experiment reports more opportunity generation and analytical depth with stronger contextual understanding but lower novelty; exact effects remain unusable because its tables conflict."
  },
  {
    "id": "Cukurova2026",
    "tier": "Tier 3",
    "full_citation": "Cukurova, M. (2026). Agency as a system property in human–AI interaction in education. British Journal of Educational Technology, 57(4), 1065–1070.",
    "link": "https://doi.org/10.1111/bjet.70060",
    "peer_reviewed": true,
    "population": "No new empirical participant sample.",
    "design": "Conceptual commentary treating agency as a property of sociotechnical configurations and focusing on accept, reject, and transform micro-decisions.",
    "finding_verbatim": "Agency is framed as “a system property” rather than a stable possession of either a learner or an AI system.",
    "effect_size": "Not applicable; no empirical effect is reported.",
    "conditions_and_limits": "Education-focused conceptual argument. Calls for process traces, reasoning quality, ownership/control, and transfer but does not validate an instrument.",
    "criticism": "It supports context-sensitive interpretation but supplies no scoring rules, reliability, criterion evidence, or entrepreneurship-specific rights.",
    "which_workflow": "Driver’s Seat—system-property theoretical anchor and warning against trait-like person scores.",
    "claimable_sentence": "A trace profile should be interpreted as a state of an observed sociotechnical configuration, not as proof that a student possesses a fixed amount of agency."
  },
  {
    "id": "DaiEtAl2026",
    "tier": "Tier 1",
    "full_citation": "Dai, Y., Liu, S., Zhou, S., Lai, S., Liu, A., & Lim, C. P. (2026). Redefining and measuring student agency in AI-assisted learning: Development and validation of the agentic engagement with AI (AE-AI) scale. Computers & Education, 253, 105687.",
    "link": "https://doi.org/10.1016/j.compedu.2026.105687",
    "peer_reviewed": true,
    "population": "Interviews with 26 students, exploratory factor analysis with 340 respondents, and confirmatory factor analysis with 256 respondents.",
    "design": "Mixed-method scale development and psychometric validation. An initial 28-item pool was reduced to a 16-item self-report scale with Adaptive Direction, Critical Integration, Cross-Source Inquiry, and Reflective Calibration factors.",
    "finding_verbatim": "The validated dimensions are “adaptive direction, critical integration, cross-source inquiry, and reflective calibration.”",
    "effect_size": "Factor-loadings and fit evidence are reported in the article; no single intervention effect applies. No trace-to-reference-coder agreement coefficient was tested.",
    "conditions_and_limits": "Self-reported engagement and agency, not native behavioral observation. The abstract’s use of “observing” does not indicate log or dialogue coding.",
    "criticism": "The scale cannot allocate rights to actors or establish temporal transitions, commitments, learning, or transfer. Convergence with Driver’s Seat would be informative but would not imply equivalence.",
    "which_workflow": "Driver’s Seat—principal validated self-report comparator for convergent and discriminant evidence.",
    "claimable_sentence": "AE-AI is a validated four-factor self-report measure of agentic engagement with AI, not an actor-by-right trace measure."
  },
  {
    "id": "DarvishiEtAl2024",
    "tier": "Tier 1",
    "full_citation": "Darvishi, A., Khosravi, H., Sadiq, S., Gašević, D., & Siemens, G. (2024). Impact of AI assistance on student agency. Computers & Education, 210, 104967.",
    "link": "https://doi.org/10.1016/j.compedu.2023.104967",
    "peer_reviewed": true,
    "population": "1,625 undergraduates across ten courses.",
    "design": "Randomized multi-course study after four common weeks of AI-assisted peer-feedback activity, comparing withdrawal of AI, continued AI, and continued AI plus a self-monitoring checklist.",
    "finding_verbatim": "The study examines the “impact of AI assistance on student agency.”",
    "effect_size": "Withdrawal worsened several automated peer-feedback indicators; AI plus checklist did not outperform AI alone. The review does not supply a standardized effect or confidence interval, so none is reconstructed here.",
    "conditions_and_limits": "Outcomes were automated feedback features rather than independent knowledge, retention, or transfer. Effects concern a specific peer-feedback system and withdrawal schedule.",
    "criticism": "The phrase “relied rather than learned” is an interpretation, not a tested learning contrast. The study does not validate Driver’s Seat or establish durable development.",
    "which_workflow": "Driver’s Seat—agency-withdrawal and outcome-boundary evidence; warning against equating assisted performance with learning.",
    "claimable_sentence": "Withdrawing AI worsened several peer-feedback proxies, while adding a checklist did not outperform continued AI; the study did not independently test learning or transfer."
  },
  {
    "id": "DebeerEtAl2017IRTrees",
    "tier": "Tier 2",
    "full_citation": "Debeer, D., Janssen, R., & De Boeck, P. (2017). Modeling skipped and not-reached items using IRTrees. Journal of Educational Measurement, 54(3), 333–363.",
    "link": "https://doi.org/10.1111/jedm.12147",
    "peer_reviewed": true,
    "population": "Simulated item-response data and an empirical illustration using the 2009 PISA reading assessment.",
    "design": "Tree-based item-response framework jointly modeling correct responses, skipped items, and not-reached items, with person and item contributions to omission processes.",
    "finding_verbatim": "When the occurrence of these omissions is related to the proficiency process the missingness is nonignorable.",
    "effect_size": "No single effect size; the paper reports simulation behavior and a PISA application rather than a treatment contrast.",
    "conditions_and_limits": "The framework distinguishes intermittent skips from terminal not-reached responses and assumes a specified IRTree structure for the response and omission processes.",
    "criticism": "Test-item omission is not the same phenomenon as an unscoreable human–AI episode. The transfer is a warning about informative missingness, not evidence that this IRTree is the correct Driver's Seat model.",
    "which_workflow": "WP-08a Phase A: informative abstention and reason-coded missingness.",
    "claimable_sentence": "When omission is related to proficiency, treating missing responses as ignorable can bias interpretation; skipped and not-reached cases should be distinguished."
  },
  {
    "id": "DelikouraPapadopoulosHui2026Agnoagentia",
    "tier": "Tier 2",
    "full_citation": "Delikoura, I., Papadopoulos, P. M., & Hui, P. (2026). Agnoagentia: The illusion of agency in AI-assisted learning. In Artificial Intelligence in Education: 27th International Conference, AIED 2026, Proceedings, Part III (Lecture Notes in Artificial Intelligence, Vol. 16583, pp. 1–9). Springer Nature Switzerland.",
    "link": "https://doi.org/10.1007/978-3-032-29760-0_1",
    "peer_reviewed": true,
    "population": "52 university students randomly grouped into 26 dyads.",
    "design": "Each dyad completed three collaborative-writing tasks in a fixed sequence: baseline, a structured pedagogical agent (Clair), and individual GPT-5 access; perceived agency was self-reported and enacted agency was dialogue-coded using Bandura's four properties.",
    "finding_verbatim": "ChatGPT-assisted students reported high perceived agency, while demonstrating low enacted agency.",
    "effect_size": "The authors report significantly reduced communication volume and increased offloading under ChatGPT, but the indexed abstract does not supply a standardized effect size.",
    "conditions_and_limits": "The contrast is between perceived agency and coded enacted agency in dyadic writing; all dyads experienced the conditions in the same order.",
    "criticism": "Fixed order confounds tool condition with task and time; small sample; no established learning-outcome criterion; Bandura-property coding is not the same as entrepreneurial judgment-right attribution.",
    "which_workflow": "Driver's Seat construct validity and abstention—observed versus perceived agency; education boundary.",
    "claimable_sentence": "Perceived agency can diverge from enacted behavior, strengthening the case for trace-based assessment while providing no evidence that Driver's Seat measures learning."
  },
  {
    "id": "ElYanivWiener2010RiskCoverage",
    "tier": "Tier 2",
    "full_citation": "El-Yaniv, R., & Wiener, Y. (2010). On the foundations of noise-free selective classification. Journal of Machine Learning Research, 11(53), 1605–1641.",
    "link": "https://jmlr.org/papers/v11/el-yaniv10a.html",
    "peer_reviewed": true,
    "population": "No human participant population. The paper studies theoretical noise-free classification models with a reject option.",
    "design": "Theoretical analysis of selective classification, defining coverage and selective risk and constructing controlled risk–coverage trade-offs.",
    "finding_verbatim": "“The essence in selective classification is to trade-off classifier coverage for higher accuracy.”",
    "effect_size": "Not applicable. The contribution consists of formal characterizations and algorithms rather than a conventional empirical effect size.",
    "conditions_and_limits": "The results concern selective classifiers under the paper’s noise-free theoretical settings and specified loss functions; the reference labels and acceptance function must be defined.",
    "criticism": "The framework does not validate the reference labels, construct, or uncertainty score and does not itself address clustered educational traces, multiclass imbalance, or class-conditional rejection.",
    "which_workflow": "WP-08b B1: formal definitions of system coverage, selective risk, abstention operating points, and risk–coverage curves.",
    "claimable_sentence": "Abstention must be evaluated jointly with the error among accepted cases because lower risk can be purchased by reducing coverage."
  },
  {
    "id": "EssienEtAl2026",
    "tier": "Tier 2",
    "full_citation": "Essien, A., Zhou, X., Kremantzis, M., & Teng, D. (2026). The agency gap: Perceived human AI agency, reflection and generative AI learning across UK and China based higher education contexts. Studies in Higher Education, 1–22. Advance online publication.",
    "link": "https://doi.org/10.1080/03075079.2026.2686986",
    "peer_reviewed": true,
    "population": "309 higher-education respondents: 145 UK-based and 164 China-based; data collected September–December 2025.",
    "design": "Cross-sectional online self-report survey analyzed with PLS-SEM. Eight agency items cover reported initiative, self-direction, monitoring, control, responsibility, and final decision.",
    "finding_verbatim": "The paper studies “perceived human AI agency,” rather than observed behavior or reciprocal trace-coded co-agency.",
    "effect_size": "Reported agency-to-reflection path coefficients were approximately β=.721 in the UK sample and β=.554 in the China sample. Outcomes were self-reported reflection, critical thinking, and self-concept.",
    "conditions_and_limits": "Cross-sectional associations with only partial measurement invariance; no behavioral traces, randomized exposure, objective learning performance, retention, or transfer criterion.",
    "criticism": "Common-method self-report cannot identify behavioral decision locus or causal direction. Country differences should not be treated as tested cultural mechanisms.",
    "which_workflow": "Driver’s Seat—convergent self-report comparator and evidence for separating perceived from enacted agency.",
    "claimable_sentence": "Perceived AI-assisted agency was associated with self-reported reflection in two national samples, but the study did not observe decision locus or learning performance."
  },
  {
    "id": "FeinsteinCicchetti1990Kappa",
    "tier": "Tier 2",
    "full_citation": "Feinstein, A. R., & Cicchetti, D. V. (1990). High agreement but low kappa: I. The problems of two paradoxes. Journal of Clinical Epidemiology, 43(6), 543–549.",
    "link": "https://doi.org/10.1016/0895-4356(90)90158-L",
    "peer_reviewed": true,
    "population": "No participant sample. The analysis uses constructed binary fourfold tables to examine observed agreement, marginal imbalance, and kappa.",
    "design": "Methodological demonstration of two kappa paradoxes produced by symmetric and asymmetric imbalance in marginal totals.",
    "finding_verbatim": "“A high value of P0 can be drastically lowered by a substantial imbalance in the table’s marginal totals.”",
    "effect_size": "Not applicable. The paper demonstrates algebraic and table-based discrepancies rather than estimating a population intervention effect.",
    "conditions_and_limits": "The demonstrations concern two observers and binary 2×2 tables; they show dependence on marginals under the specified chance correction.",
    "criticism": "The paper diagnoses kappa’s behavior but does not establish one universally superior replacement. Raw agreement, category-specific errors, and the substantive meaning of prevalence must still be reported.",
    "which_workflow": "WP-08b B1: severe-imbalance warning for levels, actor categories, abstention, and uncertainty labels.",
    "claimable_sentence": "With levels 4–5 dominating the distribution, a kappa coefficient can diverge from observed agreement, so no single chance-corrected scalar should carry the reliability claim."
  },
  {
    "id": "FossFossKlein2007",
    "tier": "Tier 3",
    "full_citation": "Foss, K., Foss, N. J., & Klein, P. G. (2007). Original and derived judgment: An entrepreneurial theory of economic organization. Organization Studies, 28(12), 1893–1912.",
    "link": "https://doi.org/10.1177/0170840606076179",
    "peer_reviewed": true,
    "population": "No empirical participant sample.",
    "design": "Entrepreneurial theory of economic organization distinguishing owner-held original judgment from decision rights delegated to subordinates as derived judgment.",
    "finding_verbatim": "The article distinguishes “original and derived judgment” within economic organization.",
    "effect_size": "Not applicable; no empirical effect is reported.",
    "conditions_and_limits": "Original judgment is tied to ownership, uncertainty bearing, and residual control; the delegated actor in the paper is organizational, not generative AI.",
    "criticism": "Extending derived judgment to AI is an analogy, not the authors’ claim. Visible chat activity cannot prove economic ownership or uncertainty bearing.",
    "which_workflow": "Driver’s Seat—entrepreneurial-judgment foundation for selective delegation and commitment boundaries.",
    "claimable_sentence": "Owners may delegate decision rights as derived judgment while retaining original judgment, but extending that distinction to AI requires a new argument and evidence."
  },
  {
    "id": "FournierInkpen2012Segmentation",
    "tier": "Tier 2",
    "full_citation": "Fournier, C., & Inkpen, D. (2012). Segmentation similarity and agreement. In Proceedings of the 2012 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (pp. 152–161). Association for Computational Linguistics.",
    "link": "https://aclanthology.org/N12-1016/",
    "peer_reviewed": true,
    "population": "No human participant population. The paper uses worked and benchmark segmentations to evaluate a proposed segmentation metric and adapted agreement coefficients.",
    "design": "Methodological paper defining Segmentation Similarity S through edit-distance transformations of boundaries and adapting chance-corrected agreement coefficients to segmentation.",
    "finding_verbatim": "“We propose a new segmentation evaluation metric, called segmentation similarity (S).”",
    "effect_size": "No single participant-level effect size is applicable. The paper compares metric behavior across segmentation examples and benchmark conditions.",
    "conditions_and_limits": "Segmentations must share a common ordered continuum, and results depend on the selected atomic unit and transformation costs for additions, deletions, substitutions, and boundary transpositions.",
    "criticism": "S is a similarity measure, not by itself a complete validity argument. Exact-boundary performance, splits, merges, unmatched episodes, and a chance-corrected agreement form should also be reported.",
    "which_workflow": "WP-08b B1: human–human and human–system episode-boundary reliability before downstream right attribution.",
    "claimable_sentence": "Episode segmentation can be evaluated with partial credit for near boundaries and explicit penalties for splits and merges instead of requiring only exact boundary matches."
  },
  {
    "id": "Fox2026EntrepreneurialBenchmarks",
    "tier": "Tier 2",
    "full_citation": "Fox, J. D. (2026). Developing artificial intelligence benchmarks for entrepreneurial tasks. Small Enterprise Research, 1–18. Advance online publication.",
    "link": "https://doi.org/10.1080/13215906.2026.2705499",
    "peer_reviewed": true,
    "population": "Published entrepreneurship literature used to identify tasks that could be assigned to artificial intelligence; no human participant sample.",
    "design": "Systematic review and prospective entrepreneurship-as-design-science framework for process benchmarking of AI on entrepreneurial tasks.",
    "finding_verbatim": "The review catalogues 131 entrepreneurial tasks as candidates for AI benchmarking.",
    "effect_size": "Not applicable; 131 is a task count from the review, not an effect size.",
    "conditions_and_limits": "The contribution concerns defining task benchmarks and evaluating AI capability on entrepreneurial work, not attributing governance between a human and AI within an interaction.",
    "criticism": "Task assignability and benchmarking do not establish that AI should govern a task, that a human delegated it, or that either party exercised entrepreneurial judgment well. The paper supplies no Driver's Seat-like measure or human outcome.",
    "which_workflow": "WP-08a Phase A: post-cutoff entrepreneurship-task landscape and novelty boundary.",
    "claimable_sentence": "Entrepreneurship-specific AI task benchmarks now exist, narrowing task-taxonomy novelty but not duplicating episode-level measurement of who governs."
  },
  {
    "id": "GluszakGluszak2026DAGM",
    "tier": "Tier 3",
    "full_citation": "Gluszak, L., & Gluszak, F. (2026). Delegated agentic governance: A delegation-centred framework for managing autonomous AI in organisations. Journal of Information & Knowledge Management, Article 2650048.",
    "link": "https://doi.org/10.1142/S0219649226500486",
    "peer_reviewed": true,
    "population": "A bibliometric corpus of 795 peer-reviewed publications from 2020–2026; no human participant sample.",
    "design": "Multi-corpus bibliometric analysis plus conceptual derivation of a three-tier Delegated Agentic Governance Model covering advisory, operational, and autonomous delegation.",
    "finding_verbatim": "Delegation, not architecture, is the primary variable that governance frameworks for agentic AI must address.",
    "effect_size": "Not applicable; 795 is a literature-corpus count, not an effect size.",
    "conditions_and_limits": "The framework assigns governance requirements by delegated-autonomy tier and proposes design principles and untested propositions for enterprise settings.",
    "criticism": "The governance-readiness instrument is conceptual rather than psychometrically validated. The paper does not observe or score enacted judgment rights, episode-level authority, or human outcomes.",
    "which_workflow": "WP-08a Phase A: current prior-art comparison and delegation theory.",
    "claimable_sentence": "Delegation-calibrated organizational governance frameworks now exist, but they do not duplicate a validated episode-level behavioral attribution measure."
  },
  {
    "id": "GneitingRaftery2007ProperScoring",
    "tier": "Tier 2",
    "full_citation": "Gneiting, T., & Raftery, A. E. (2007). Strictly proper scoring rules, prediction, and estimation. Journal of the American Statistical Association, 102(477), 359–378.",
    "link": "https://doi.org/10.1198/016214506000001437",
    "peer_reviewed": true,
    "population": "No human participant population. The paper develops a general mathematical framework for probabilistic forecasts and scoring rules.",
    "design": "Theoretical and methodological treatment of proper and strictly proper scoring rules, including representations and examples for probabilistic prediction and estimation.",
    "finding_verbatim": "“Scoring rules assess the quality of probabilistic forecasts.”",
    "effect_size": "Not applicable. The paper provides definitions, representation results, and worked forecasting applications rather than a participant-level effect.",
    "conditions_and_limits": "A proper score evaluates an issued predictive distribution against an observed outcome; strict propriety rewards truthful probabilistic reporting under the chosen outcome and score.",
    "criticism": "A proper score does not establish that reference labels are true, that a construct is valid, or that probabilities remain calibrated across rights, levels, contexts, or deployment shifts.",
    "which_workflow": "WP-08b B1: evaluation of system uncertainty using Brier/log scores and calibration diagnostics rather than unvalidated verbal confidence.",
    "claimable_sentence": "If Driver’s Seat emits probabilities, uncertainty should be evaluated with proper scoring rules and calibration diagnostics on held-out human-referenced cases."
  },
  {
    "id": "GordetzkiEtAl2026",
    "tier": "Tier 1",
    "full_citation": "Gordetzki, P., Blohm, I., Clegg, M., Schakols, F., & Hofstetter, R. (2026). Agency configurations in generative AI ideation: How textual and visual idea concretizations shape idea creativity and ideator effort. Information Systems Research. Advance online publication.",
    "link": "https://doi.org/10.1287/isre.2024.0952",
    "peer_reviewed": true,
    "population": "276 US Prolific participants producing 428 idea refinements; novice participants completed one ideation task.",
    "design": "Online experiment comparing textual versus visual AI concretization with imagine/control conditions and testing idea-maturity moderation.",
    "finding_verbatim": "The authors describe “agency configurations” in which representation changes creativity and ideator effort.",
    "effect_size": "Textual relative to visual support reportedly increased creativity by 18% and required 30% more effort. Effects were most pronounced for mature ideas; visual support improved very immature ideas without added effort.",
    "conditions_and_limits": "One short ideation task with novices and no longitudinal learning, expertise, consequential commitment, or transfer criterion.",
    "criticism": "Agency configuration is inferred from treatments and effort rather than directly coded actor-by-right allocation. More effort is not itself more governance.",
    "which_workflow": "Driver’s Seat—configuration-contingency evidence, principally relevant to option formation.",
    "claimable_sentence": "AI representation changed creativity and effort under specific idea-maturity conditions, but the experiment did not measure who governed framing, evaluation, or commitment."
  },
  {
    "id": "GuTopol2026DecisionAuthority",
    "tier": "Tier 3",
    "full_citation": "Gu, Y., & Topol, E. J. (2026). Decision authority in health AI. Nature Health. Advance online publication.",
    "link": "https://doi.org/10.1038/s44360-026-00185-z",
    "peer_reviewed": false,
    "population": "Health-AI systems used inside and outside clinical settings; no participant sample.",
    "design": "Policy and governance commentary organized around degrees of AI influence over individual health trajectories.",
    "finding_verbatim": "They need to be regulated and evaluated with respect to the degree to which they can influence and shape individual health trajectories.",
    "effect_size": "Not applicable; no empirical study.",
    "conditions_and_limits": "The argument addresses consequential healthcare authority and regulation, including use beyond formal clinical care.",
    "criticism": "The Comment provides no operationalization, reliability evidence, behavioral trace analysis, or entrepreneurship application. One author is affiliated with ByteDance, which should be disclosed when assessing perspective.",
    "which_workflow": "WP-08a Phase A: post-cutoff decision-authority landscape and novelty boundary.",
    "claimable_sentence": "A post-cutoff health-AI commentary independently centers degree of decision authority, so the general question of who governs cannot be claimed as novel."
  },
  {
    "id": "Gwet2008AC1",
    "tier": "Tier 2",
    "full_citation": "Gwet, K. L. (2008). Computing inter-rater reliability and its variance in the presence of high agreement. British Journal of Mathematical and Statistical Psychology, 61(1), 29–48.",
    "link": "https://doi.org/10.1348/000711006X126600",
    "peer_reviewed": true,
    "population": "No human participant population. The paper uses mathematical analysis and Monte Carlo simulation of multiple-rater nominal ratings.",
    "design": "Methodological study tracing limitations of pi and kappa, introducing AC1, and proposing variance estimators that do not require rater independence.",
    "finding_verbatim": "“This paper … introduces an alternative and more stable agreement coefficient referred to as the AC1 coefficient.”",
    "effect_size": "No conventional participant effect size is applicable. Monte Carlo results support the proposed variance estimators and AC1 behavior under simulated conditions.",
    "conditions_and_limits": "The 2008 paper addresses nominal ratings and high-agreement settings; AC1 uses a different chance-agreement model from kappa.",
    "criticism": "AC1’s stability under skew does not remove the need for raw, class-specific, and confusion-matrix reporting. Kappa interpretation bands should not be transferred mechanically to AC1.",
    "which_workflow": "WP-08b B1: primary chance-corrected reliability for nominal opportunity, observability, actor, configuration, and abstention labels under imbalance.",
    "claimable_sentence": "AC1 is a defensible chance-corrected companion to exact and class-specific agreement when dominant categories make kappa unstable."
  },
  {
    "id": "HayesKrippendorff2007Alpha",
    "tier": "Tier 1",
    "full_citation": "Hayes, A. F., & Krippendorff, K. (2007). Answering the call for a standard reliability measure for coding data. Communication Methods and Measures, 1(1), 77–89.",
    "link": "https://doi.org/10.1080/19312450709336664",
    "peer_reviewed": true,
    "population": "No human participant population. The methodological brief uses a worked coding example and software implementation.",
    "design": "Methodological evaluation of requirements for coding-reliability coefficients and proposal of Krippendorff’s alpha as a general standard across coders, scales, sample sizes, and missing data.",
    "finding_verbatim": "“We propose Krippendorff’s alpha as the standard reliability measure.”",
    "effect_size": "Not applicable. The paper presents methodological criteria, a worked example, and implementation rather than an intervention effect.",
    "conditions_and_limits": "Alpha requires independent codings of common units and a distance function appropriate to the measurement level; its interpretation depends on sampling and coding design.",
    "criticism": "A single alpha can conceal systematic class, right, or context failures, and ordinary alpha for predefined units does not solve disagreement about episode boundaries.",
    "which_workflow": "WP-08b B1: sensitivity reliability statistic for nominal and ordinal labels and explicit distinction between unitizing and coding reliability.",
    "claimable_sentence": "Krippendorff’s alpha provides a common sensitivity analysis across nominal and ordinal coding, but segmentation and class-specific performance still require separate evidence."
  },
  {
    "id": "HendrickxEtAl2024RejectOption",
    "tier": "Tier 2",
    "full_citation": "Hendrickx, K., Perini, L., Van der Plas, D., Meert, W., & Davis, J. (2024). Machine learning with a reject option: A survey. Machine Learning, 113(5), 3073–3110.",
    "link": "https://doi.org/10.1007/s10994-024-06534-x",
    "peer_reviewed": true,
    "population": "Published machine-learning methods for prediction with a reject option across application domains.",
    "design": "Field survey and taxonomy covering reasons for rejection, model architectures, learning methods, and evaluation of predictive and rejective quality.",
    "finding_verbatim": "We introduce the conditions leading to two types of rejection, ambiguity and novelty rejection, which we carefully formalize.",
    "effect_size": "Not applicable; the survey does not pool an effect size.",
    "conditions_and_limits": "Predictive and rejective performance must be considered together; ambiguity rejection and novelty rejection address different failure conditions.",
    "criticism": "The review concerns machine-learning prediction, not formative educational measurement or conversational observability. Its taxonomies do not establish a valid abstention threshold for Driver's Seat.",
    "which_workflow": "WP-08a Phase A: abstention taxonomy and evaluation requirements.",
    "claimable_sentence": "Modern reject-option research distinguishes ambiguity from novelty rejection and evaluates prediction quality jointly with rejection behavior."
  },
  {
    "id": "IssaPetaniGlavas2026AgenticLoafing",
    "tier": "Tier 2",
    "full_citation": "Issa, H., Petani, F. J., & Glavas, D. (2026). Agentic loafing: An AI decision delegation risk. Risk Analysis, 46(8), e70306.",
    "link": "https://doi.org/10.1111/risa.70306",
    "peer_reviewed": true,
    "population": "Professional podcasts and gray literature concerning clinicians' delegation to AI; no conventional participant sample is reported in the abstract.",
    "design": "Netnography of professional podcasts plus secondary analysis of gray literature; conceptual development of three institutional drivers and a preliminary risk diagnostic.",
    "finding_verbatim": "the systematic disappearance of accountability when AI-assisted decisions fail, leaving no single party responsible",
    "effect_size": "No quantitative effect size; qualitative/conceptual risk analysis.",
    "conditions_and_limits": "The proposed drivers are performance-culture conformity, structural isolation of responsibility, and legitimization through quantification in healthcare decision delegation.",
    "criticism": "Podcast and gray-literature material, no representative sampling, no behavioral validation, and a preliminary diagnostic that has not undergone psychometric or criterion validation.",
    "which_workflow": "Driver's Seat risk interpretation—responsibility erosion and delegation without monitoring.",
    "claimable_sentence": "Agentic loafing identifies responsibility erosion as a delegation risk, but its preliminary organizational diagnostic does not duplicate an episode-level behavioral measure."
  },
  {
    "id": "IssakRezwanaHarteveld2025MOSAAIC",
    "tier": "Tier 2",
    "full_citation": "Issak, A., Rezwana, J., & Harteveld, C. (2025). MOSAAIC: Managing optimization towards shared autonomy, authority, and initiative in co-creation. In Proceedings of the Sixteenth International Conference on Computational Creativity (pp. 97–107). Association for Computational Creativity.",
    "link": "https://computationalcreativity.net/iccc25/wp-content/uploads/papers/iccc25-issak2025mosaaic.pdf",
    "peer_reviewed": true,
    "population": "A systematic review of 172 full-length publications from ACM and Association for Computational Creativity venues, followed by six published co-creative systems used as cases.",
    "design": "Systematic literature review and framework synthesis; two authors independently coded six co-creative systems and resolved discrepancies through discussion.",
    "finding_verbatim": "We define control as the power to determine, initiate, and direct the process of co-creation.",
    "effect_size": "No quantitative effect size; this is a review-derived framework and six-case demonstration.",
    "conditions_and_limits": "The framework concerns human–AI co-creativity and allocates autonomy, initiative, and authority as full-human, shared, or full-AI; the authors state that an appropriate balance depends on the task.",
    "criticism": "It does not validate a behavioral score, use consequential entrepreneurial episodes, or report inter-rater reliability for the six-case coding. Its dimensions are broader than five specific judgment rights.",
    "which_workflow": "WP-08A/B: allocation of authority, initiative, and autonomy.",
    "claimable_sentence": "MOSAAIC frames human–AI control as separable allocations of autonomy, initiative, and authority; the general control-allocation question is established prior art."
  },
  {
    "id": "IssakRezwanaHarteveld2026ControlTrajectory",
    "tier": "Tier 2",
    "full_citation": "Issak, A., Rezwana, J., & Harteveld, C. (2026). “Control is a trajectory, not a point”: Conceptualizing control in human-AI co-creativity. In Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems (pp. 1–17). ACM.",
    "link": "https://doi.org/10.1145/3772318.3790861",
    "peer_reviewed": true,
    "population": "Nine experts in HCI, co-creativity, and AI.",
    "design": "Semi-structured expert focus group using MOSAAIC as a theoretical probe; qualitative thematic analysis.",
    "finding_verbatim": "control is widely viewed as a dynamic, context-dependent construct that should adapt across different phases of co-creation, domains, and levels of trust in AI",
    "effect_size": "No quantitative effect size; qualitative expert study.",
    "conditions_and_limits": "Findings concern expert conceptions and preferences for co-creative systems, not observed consequential decisions by end users.",
    "criticism": "One focus group with nine experts cannot validate a trajectory measure or establish how control actually shifts in use. The MOSAAIC probe may also have shaped the categories discussed.",
    "which_workflow": "Driver's Seat dynamics—within-episode movement and control trajectories.",
    "claimable_sentence": "Dynamic, phase-sensitive control is established conceptual territory; a particular operationalization must be evaluated on its measurement evidence."
  },
  {
    "id": "JiangEtAl2026",
    "tier": "Tier 2",
    "full_citation": "Jiang, Y., Wu, Q., Yang, Y., Jian, C., & Zhao, J. (2026). Learner agency in revising GenAI-generated statements of purpose. British Journal of Educational Technology, 57(4), 965–983.",
    "link": "https://doi.org/10.1111/bjet.70041",
    "peer_reviewed": true,
    "population": "121 English-major students at one southern-China university; nine completed retrospective interviews.",
    "design": "ChatGPT-4 generated a statement-of-purpose draft from each CV and a standardized prompt; students evaluated and edited it. The study combined text comparison, nonparametric ratings, and thematic interviews.",
    "finding_verbatim": "The authors identify “compliance-oriented acceptance, form-oriented modification, and content-oriented innovation.”",
    "effect_size": "The paper reports rating differences among the three patterns; no standardized causal effect from a randomized intervention applies and none is entered here.",
    "conditions_and_limits": "One revision task, one institution, fixed AI draft, nine interviews, and no conversational log, independent quality criterion, retention, or transfer measure.",
    "criticism": "The patterns describe revision behavior and voice, not a complete actor-by-right allocation. Suggested instructional enhancements are not experimentally tested effects.",
    "which_workflow": "Driver’s Seat—evaluation, modification, ownership, and voice comparator.",
    "claimable_sentence": "Students displayed compliance, form-level modification, and content-level innovation when revising AI drafts, but the study does not establish learning or full decision-right governance."
  },
  {
    "id": "Kane2013Validity",
    "tier": "Tier 1",
    "full_citation": "Kane, M. T. (2013). Validating the interpretations and uses of test scores. Journal of Educational Measurement, 50(1), 1–73.",
    "link": "https://doi.org/10.1111/jedm.12000",
    "peer_reviewed": true,
    "population": "Proposed interpretations and uses of assessment scores across testing contexts; no participant sample.",
    "design": "Argument-based validity framework specifying inferences, assumptions, evidence, and consequences for score interpretations and uses.",
    "finding_verbatim": "It is the proposed score interpretations and uses that are validated and not the test or the test scores.",
    "effect_size": "Not applicable; this is a validity framework rather than an empirical treatment study.",
    "conditions_and_limits": "The evidence burden depends on the ambition of the proposed interpretation or use; consequences matter for use claims, and validating an interpretation does not automatically validate a use.",
    "criticism": "Kane supplies a framework, not a reliability cutoff, abstention rule, or empirical validation result. Invoking the framework cannot substitute for testing Driver's Seat's scoring, generalization, extrapolation, and use assumptions.",
    "which_workflow": "WP-08a Phase A and Phase B: claim ladder and interpretation/use argument.",
    "claimable_sentence": "Driver's Seat validation must target each proposed interpretation and use; a defensible conditional trace description would not by itself validate ranking, development, or learning claims."
  },
  {
    "id": "KimSoPark2026",
    "tier": "Tier 1",
    "full_citation": "Kim, S., So, H.-J., & Park, K. (2026). Supporting learner agency in collaborative writing with generative AI. British Journal of Educational Technology, 57(4), 984–1008.",
    "link": "https://doi.org/10.1111/bjet.70015",
    "peer_reviewed": true,
    "population": "52 Korean students randomized to Similarity Viewer only (n=26) or Similarity Viewer plus Argument Outline (n=26); eight participants were interviewed.",
    "design": "Randomized two-condition collaborative-writing study with two 30-minute tasks, trace coding, text-similarity analysis, epistemic network analysis, and interviews.",
    "finding_verbatim": "The authors warn that optimal learner agency is not simply “higher human input.”",
    "effect_size": "No significant overall behavior-frequency differences. In session 2, the outline group had lower similarity, F=5.48, p=.02, η²=.11, with final analytic groups of 25 and 23.",
    "conditions_and_limits": "Short collaborative-writing tasks; similarity is not direct agency or learning evidence. Codes include Compose, Revise, Seek Suggestion, Dismiss, Accept, and low/high modification.",
    "criticism": "The study does not allocate five decision rights, test consequential commitment, or demonstrate learning and transfer. Modification amount cannot stand alone as governance.",
    "which_workflow": "Driver’s Seat—behavioral accept/reject/modify comparator and warning against activity-volume scoring.",
    "claimable_sentence": "A small randomized writing study found a session-specific similarity difference but no overall behavior-frequency difference, and explicitly rejected equating more human input with better agency."
  },
  {
    "id": "Krippendorff1995Unitizing",
    "tier": "Tier 2",
    "full_citation": "Krippendorff, K. (1995). On the reliability of unitizing continuous data. Sociological Methodology, 25, 47–76.",
    "link": "https://doi.org/10.2307/271061",
    "peer_reviewed": true,
    "population": "No human participant population. The paper develops a reliability coefficient for observers who identify units within textual, audiovisual, temporal, or spatial continua.",
    "design": "Foundational derivation of a chance-corrected alpha-family coefficient for unitizing continuous phenomena rather than only categorizing predefined units.",
    "finding_verbatim": "“Unitizing underlies all quantifications.”",
    "effect_size": "Not applicable. The contribution is a derived reliability coefficient and worked methodological analysis, not a participant-level effect.",
    "conditions_and_limits": "Coders must independently inspect the same continuum; applications must define the smallest indivisible unit, gaps, unit lengths, and category distance.",
    "criticism": "The coefficient is technically demanding and depends on continuum and gap definitions. Later unitizing work refined implementations and multi-annotator inference, so software and formula versions must be documented.",
    "which_workflow": "WP-08b B1: sensitivity analysis for reliability of locating Driver’s Seat episodes in continuous conversation traces.",
    "claimable_sentence": "Agreement about labels on predefined episodes is insufficient when coders may disagree about where an episode begins and ends."
  },
  {
    "id": "KrushinskaiaElenRaes2026",
    "tier": "Tier 1",
    "full_citation": "Krushinskaia, K., Elen, J., & Raes, A. (2026). Pre-service teachers’ agency during their interactions with generative AI while designing for learning—A process view on Intelligent-TPACK. Computers and Education Open, 10, 100325.",
    "link": "https://doi.org/10.1016/j.caeo.2025.100325",
    "peer_reviewed": true,
    "population": "78 pre-service teachers randomized to a custom expert bot or basic ChatGPT.",
    "design": "Randomized process study using interaction length, proportion of self-generated prompts/words, and coded collaborative problem-solving behavior as agency indicators.",
    "finding_verbatim": "The study takes “a process view” of agency during generative-AI-supported learning design.",
    "effect_size": "Omnibus MANOVA: Pillai’s trace=.262, F(4,50)=4.43, p=.004. Expert-bot users produced more prompts, F(1,41)=8.20, corrected p=.007, η²=.126; deeper coding indicated mixed or reduced substantive agency.",
    "conditions_and_limits": "Pre-service-teacher design task, two tool configurations, and multiple proxy indicators that did not converge on a single agency interpretation.",
    "criticism": "Length and self-generated prompting can suggest increased agency while collaboration coding suggests passivity or delegation. No learning, retention, transfer, or five-right reference coding was tested.",
    "which_workflow": "Driver’s Seat—strong adverse evidence on proxy disagreement and validation design.",
    "claimable_sentence": "In one randomized study, an expert bot increased prompt production while deeper process coding suggested mixed or reduced agency, showing that surface activity is not a valid stand-alone proxy."
  },
  {
    "id": "Lee2026ReadinessMetrics",
    "tier": "Tier 2",
    "full_citation": "Lee, M. H. (2026). From accuracy to readiness: Metrics and benchmarks for human-AI decision-making: An initial exploration. In Extended Abstracts of the 2026 CHI Conference on Human Factors in Computing Systems (pp. 1–10). ACM.",
    "link": "https://doi.org/10.1145/3772363.3798377",
    "peer_reviewed": true,
    "population": "No participant sample; conceptual metrics paper.",
    "design": "Conceptual synthesis proposing four metric families—outcomes, reliance and interaction, safety and harm, and learning and readiness—mapped to an Understand–Control–Improve lifecycle.",
    "finding_verbatim": "This paper proposes a measurement framework for evaluating human-AI decision-making centered on team readiness.",
    "effect_size": "Not applicable; no empirical test.",
    "conditions_and_limits": "Metrics such as accept-on-wrong, changed-to-wrong, override timing, rollback, escalation, and transfer presuppose suitable interaction logs and, for some metrics, ground truth.",
    "criticism": "The framework is unvalidated and does not attribute entrepreneurial judgment rights. It nevertheless establishes trace-based governance-in-use and longitudinal readiness as prior measurement ideas.",
    "which_workflow": "Driver's Seat measurement design—trace-based behavior, abstention prerequisites, and criterion validation.",
    "claimable_sentence": "Trace-based evaluation of reliance and governance is already proposed; Driver's Seat's narrower contribution must lie in its particular judgment-right rubric, abstention rule, and validation program."
  },
  {
    "id": "Leonardi2025",
    "tier": "Tier 3",
    "full_citation": "Leonardi, P. M. (2025). Homo agenticus in the age of agentic AI: Agency loops, power displacement, and the circulation of responsibility. Information and Organization, 35(3), 100582.",
    "link": "https://doi.org/10.1016/j.infoandorg.2025.100582",
    "peer_reviewed": true,
    "population": "No new participant sample; the article synthesizes and illustrates three prior empirical studies.",
    "design": "Conceptual five-phase agency loop: delegation, attribution, contingency, reassertion, and reconfiguration.",
    "finding_verbatim": "The framework follows “delegation, attribution, contingency, reassertion, and reconfiguration.”",
    "effect_size": "Not applicable; no new empirical effect is estimated.",
    "conditions_and_limits": "Organizational theory of attributed agency and responsibility, not a trace-scoring or psychometric study.",
    "criticism": "Perceived or attributed agency can shift while formal authority remains unchanged. Driver’s Seat must not conflate subjective attribution, formal authority, and visible enacted governance.",
    "which_workflow": "Driver’s Seat—temporal-transition and responsibility-circulation precursor.",
    "claimable_sentence": "Delegation and reassertion are established temporal agency ideas; a transition-coding proposal requires separate validation."
  },
  {
    "id": "Li2026AIEE",
    "tier": "Tier 3",
    "full_citation": "Li, Y. (2026). The associations of AI-integrated entrepreneurship education versus traditional entrepreneurship education on undergraduates' entrepreneurial intention and its antecedents. Humanities and Social Sciences Communications. Advance online publication.",
    "link": "https://doi.org/10.1057/s41599-026-08504-1",
    "peer_reviewed": true,
    "population": "133 undergraduates in two intact classes at one university in Suzhou, China: AI-integrated entrepreneurship education n = 68 and traditional entrepreneurship education n = 65.",
    "design": "One-semester, non-individually-randomized, post-test comparison analyzed with multigroup PLS-SEM using self-reported entrepreneurial intention and antecedents.",
    "finding_verbatim": "Paths from ATE and PBC to EI, alongside the group-specific indirect effects via PD and PF, are significantly stronger in the AIEE group.",
    "effect_size": "ATE→EI: .310 versus .200, difference .110, one-tailed p = .023; PBC→EI: .350 versus .190, difference .160, p = .012. Indirect ATE→PD→EI: .247 versus .037, difference .210, p = .009; PBC→PF→EI: .231 versus .036, difference .195, p = .002. EI R²: .650 versus .318, difference .332, p < .001.",
    "conditions_and_limits": "The comparison bundles AI tools with the full instructional package. The authors explicitly interpret results as correlational because assignment was nonrandom, data were cross-sectional, the site was singular, confounders were unmeasured, and pure AI effects could not be isolated.",
    "criticism": "Outcomes are self-reported intentions and perceived antecedents, not objective learning, transfer, durable development, or Driver's Seat. One-tailed path comparisons, small intact classes, common method, and PLS prediction can make differences look stronger than warranted.",
    "which_workflow": "WP-08a Phase A: post-cutoff entrepreneurship-education evidence and learning-claim boundary.",
    "claimable_sentence": "One small nonrandom class comparison associated an AI-integrated curriculum with stronger self-reported intention pathways, but it does not establish learning or any relation between Driver's Seat and outcomes."
  },
  {
    "id": "MadjdiWurth2026",
    "tier": "Tier 3",
    "full_citation": "Madjdi, F., & Wurth, B. (2026). AI-mediated plausibility regimes: Entrepreneurial judgment, epistemic risk, and the distribution of entrepreneurial futures. Journal of Business Venturing Insights, 26, e00644.",
    "link": "https://doi.org/10.1016/j.jbvi.2026.e00644",
    "peer_reviewed": true,
    "population": "No empirical participants or dataset.",
    "design": "Conceptual theory of generative plausibility inflation and screening-based plausibility suppression under Knightian uncertainty.",
    "finding_verbatim": "The authors argue that “AI systems are not epistemically neutral.”",
    "effect_size": "Not applicable; no empirical effect is estimated.",
    "conditions_and_limits": "Propositions depend on AI system class, inferential logic, judgment intensity, institutionalization, and the authority granted to AI-generated plausibility signals.",
    "criticism": "The article theorizes possible false-positive and false-negative mechanisms; it does not demonstrate that AI caused either effect in entrepreneurs or ecosystems.",
    "which_workflow": "Driver’s Seat—entrepreneurial context, grounding, evaluative governance, and epistemic-risk theory.",
    "claimable_sentence": "Madjdi and Wurth theorize that generative systems may inflate confidence and screening systems may suppress low-precedent futures under specified conditions."
  },
  {
    "id": "MargaridoEtAl2024",
    "tier": "Tier 3",
    "full_citation": "Margarido, S., Roque, L., Machado, P., & Martins, P. (2024). MI-CCy Quantifier: A framework for quantifying mixed-initiative co-creativity in human-AI collaborations. In M. F. Santos, J. Machado, P. Novais, P. Cortez, & P. M. Moreira (Eds.), Progress in artificial intelligence: 23rd EPIA Conference on Artificial Intelligence, EPIA 2024, proceedings, Part I (pp. 3–15). Springer.",
    "link": "https://doi.org/10.1007/978-3-031-73497-7_1",
    "peer_reviewed": true,
    "population": "No human participant sample. The paper demonstrates the framework through a subjective analysis of one co-creative system, 1001 Nights.",
    "design": "Conceptual/descriptive framework with human-to-computer spectra for Initial Setting, Initiative, Evaluation, and Final Decision, plus ordinal visual criteria for Task Assignment, Intervention Pace, and Explainability.",
    "finding_verbatim": "The authors call it “a directive framework for analyzing works based on their level of MI-CCy in a descriptive and visually gradable manner.”",
    "effect_size": "Not applicable. The paper reports no numerical score, participant effect, reliability coefficient, or criterion-validity estimate.",
    "conditions_and_limits": "The analysis is system-level and subjectively positioned on visual spectra. Final Decision chiefly concerns when to end a creative process, not consequential entrepreneurial commitment.",
    "criticism": "The framework lacks behavioral anchors, reproducible scoring rules, participants, reliability, and validation. The authors explicitly warn that a higher MI-CCy level is not necessarily better.",
    "which_workflow": "WP-08A/B: allocation-framework comparator for problem framing and evaluative governance.",
    "claimable_sentence": "MI-CCy already allocates human and computer influence over initial setting, initiative, evaluation, and process closure, but it is not a validated behavioral measure."
  },
  {
    "id": "MishraHenriksen2026AgentAgency",
    "tier": "Tier 3",
    "full_citation": "Mishra, P., & Henriksen, D. (2026). Agentic AI in education: Whose agent? Whose agency? TechTrends. Advance online publication.",
    "link": "https://doi.org/10.1007/s11528-026-01213-1",
    "peer_reviewed": true,
    "population": "Educational discourse and historical concepts of agents, agency, learner voice, and teacher voice; no participant sample.",
    "design": "Conceptual and historical analysis of the language of agentic AI in education.",
    "finding_verbatim": "Agent is a role-word that extends easily to software, while agency is a rights-word.",
    "effect_size": "Not applicable; no empirical study.",
    "conditions_and_limits": "The argument concerns moral and educational meanings of agency and the institutional principals served by educational AI platforms.",
    "criticism": "The paper offers a normative linguistic distinction, not an operational measure, behavioral coding study, or test of learning and governance outcomes.",
    "which_workflow": "WP-08a Phase A: post-cutoff terminology and novelty boundary.",
    "claimable_sentence": "Calling software an agent does not establish that it possesses, preserves, or redistributes human agency; role and rights should remain analytically distinct."
  },
  {
    "id": "MurrayRhymerSirmon2021",
    "tier": "Tier 3",
    "full_citation": "Murray, A., Rhymer, J., & Sirmon, D. G. (2021). Humans and technology: Forms of conjoined agency in organizations. Academy of Management Review, 46(3), 552–571.",
    "link": "https://doi.org/10.5465/amr.2019.0186",
    "peer_reviewed": true,
    "population": "No empirical participant sample.",
    "design": "Conceptual 2×2 framework allocating intentionality over protocol development and action selection to a human or technology, yielding assisting, arresting, augmenting, and automating forms.",
    "finding_verbatim": "The paper theorizes “forms of conjoined agency” from protocol development and action selection.",
    "effect_size": "Not applicable; no empirical effect is estimated.",
    "conditions_and_limits": "General organizational technology theory, not generative-AI dialogue, entrepreneurship, or a validated trace measure.",
    "criticism": "It is a direct conceptual precursor to Randazzo’s how/what allocation. Any claim that actor-by-decision-dimension allocation is new is untenable.",
    "which_workflow": "WP-08A/B: foundational precursor for actor-by-decision-dimension allocation.",
    "claimable_sentence": "Murray et al. already allocate protocol development and action selection between humans and technology, making the broad how/what authority question prior art."
  },
  {
    "id": "PackardBylund2025",
    "tier": "Tier 3",
    "full_citation": "Packard, M. D., & Bylund, P. L. (2025). Towards an entrepreneurial judgement theory: Building the cognitive microfoundations of entrepreneurial judgement. International Small Business Journal: Researching Entrepreneurship, 43(1), 53–75.",
    "link": "https://doi.org/10.1177/02662426241269772",
    "peer_reviewed": true,
    "population": "No empirical participant sample.",
    "design": "Conceptual entrepreneurial-judgment theory using nested distal and proximal intentions and a dynamic intentionality scaffold.",
    "finding_verbatim": "The authors define judgment as “the determination and instigation of intentions.”",
    "effect_size": "Not applicable; no empirical effect is reported.",
    "conditions_and_limits": "A cognitive and intentional theory. Text can contain evidence consistent with commitment but cannot reveal private intention or establish later action.",
    "criticism": "A visible commitment utterance is not identical to the formation, instigation, or persistence of an entrepreneurial intention.",
    "which_workflow": "Driver’s Seat—commitment-right and intention boundary theory.",
    "claimable_sentence": "Entrepreneurial judgment theory links judgment to determining and instigating nested intentions; dialogue evidence can only proxy, not prove, that private process."
  },
  {
    "id": "RafnerEtAl2025CreativeAgency",
    "tier": "Tier 2",
    "full_citation": "Rafner, J., Zana, B., Hansen, I. B., Ceh, S., Sherson, J., Benedek, M., & Lebuda, I. (2025). Agency in human-AI collaboration for image generation and creative writing: Preliminary insights from think-aloud protocols. Creativity Research Journal, advance online publication, 1–24.",
    "link": "https://doi.org/10.1080/10400419.2025.2587803",
    "peer_reviewed": true,
    "population": "Study 1: six participants completing AI-assisted image generation; Study 2: seven participants completing AI-assisted creative writing.",
    "design": "Two exploratory qualitative studies using think-aloud protocols, post-task semi-structured interviews, and within-subject comparisons across tools.",
    "finding_verbatim": "agency in human–AI co-creation fluctuates across the creative process",
    "effect_size": "No quantitative effect size; exploratory qualitative evidence.",
    "conditions_and_limits": "Agency was organized around creative self-efficacy, control over creative action, autonomy in process, and ownership of product in image-generation and writing tasks.",
    "criticism": "Extremely small samples and exploratory coding; no consequential decision, validated coefficient, independent outcome, or evidence of durable development.",
    "which_workflow": "Driver's Seat dynamics—fluctuation and reassertion during an episode.",
    "claimable_sentence": "Small think-aloud studies show that experienced creative agency can fluctuate during AI use, supporting a dynamic premise but not validating Driver's Seat."
  },
  {
    "id": "RandazzoEtAl2025",
    "tier": "Tier 3",
    "full_citation": "Randazzo, S., Lifshitz, H., Kellogg, K. C., Dell’Acqua, F., Mollick, E., Candelon, F., & Lakhani, K. R. (2025). Cyborgs, centaurs and self-automators: The three modes of human–GenAI knowledge work and their implications for skilling and the future of expertise (Harvard Business School Working Paper No. 26-036). Harvard Business School.",
    "link": "https://doi.org/10.2139/ssrn.4921696",
    "peer_reviewed": false,
    "population": "244 global junior management consultants completing a strategic investment task; the mode counts reported in the paper total 243. The authors describe approximately 4,975 logged interactions/prompts and 237 follow-up interviews.",
    "design": "Field study of human–GPT-4 work across seven subtasks. Session logs, final products, and interviews were used to derive Directed/Centaur, Fused/Cyborg, and Abdicated/Self-Automator modes and examine sequences across the workflow.",
    "finding_verbatim": "The paper asks: “Who selects what needs to be done?” and “Who identifies how it gets done?”",
    "effect_size": "No standardized effect size for actor allocation, skilling, or transfer was reported. Descriptive mode counts were 146 Cyborgs, 34 Centaurs, and 63 Self-Automators, totaling 243 rather than the stated N=244.",
    "conditions_and_limits": "One consulting task, one organizational partner, junior consultants, and GPT-4. The modes are session-level patterns, not right-level scores; sequence analysis is present, but not a formal decision-right transition measure.",
    "criticism": "No public coder-reliability estimate, independent learning or transfer criterion, or journal peer review was located. Interview claims about upskilling/newskilling should not be treated as demonstrated durable development.",
    "which_workflow": "WP-08A/B: closest full-workflow trace and what/how authority precursor.",
    "claimable_sentence": "A 2025 working paper uses the driver’s-seat metaphor and classifies who directs what and how across a logged consulting workflow; those broad ideas are established prior art."
  },
  {
    "id": "RappOlbrich2023",
    "tier": "Tier 3",
    "full_citation": "Rapp, D. J., & Olbrich, M. (2023). From Knightian uncertainty to real-structuredness: Further opening the judgment black box. Strategic Entrepreneurship Journal, 17(1), 186–209.",
    "link": "https://doi.org/10.1002/sej.1443",
    "peer_reviewed": true,
    "population": "No empirical participant sample.",
    "design": "Conceptual dimensionalization of entrepreneurial judgment through goal, causality, appraisal, and solution judgments under real-structured decision problems.",
    "finding_verbatim": "The article’s “four-part dimensionalization” covers effects, appraisal of alternatives, goals, and resolution of the decision problem.",
    "effect_size": "Not applicable; no empirical effect is estimated.",
    "conditions_and_limits": "The categories concern entrepreneurial decision problems and selective communicability/delegation; they are not dialogue codes or a validated scale.",
    "criticism": "Mapping goal, causality, appraisal, and solution judgments to PF/OH/EG is interpretive. Consequential commitment is not fully represented, and external input may be advice rather than derived judgment.",
    "which_workflow": "Driver’s Seat—principal content-domain precursor for differentiated entrepreneurial judgment rights.",
    "claimable_sentence": "Goal, causality, appraisal, and solution judgments are established entrepreneurship constructs that can inform, but do not validate, a trace operationalization."
  },
  {
    "id": "RetamalSaavedraEtAl2026",
    "tier": "Tier 1",
    "full_citation": "Retamal-Saavedra, C. D., Andrade-Valbuena, N. A., Contreras Navarro, J. E., Inostroza Caceres, F., & Vidal-Rebolledo, I. (2026). Artificial intelligence in entrepreneurship: Mapping a fragmented field and advancing a cognitive research agenda. Journal of Management & Organization, 32(2), 473–500.",
    "link": "https://doi.org/10.1017/jmo.2026.10082",
    "peer_reviewed": true,
    "population": "372 peer-reviewed Web of Science documents retained after screening 933 records. The publisher page inconsistently describes the covered period as both 2010–2025 and 2001–2025.",
    "design": "Hybrid systematic review, bibliometric co-word analysis, thematic mapping, and qualitative synthesis across selected AI-and-entrepreneurship categories.",
    "finding_verbatim": "The field is described as “fragmented, technology centered, and weakly connected to theories of entrepreneurial decision-making.”",
    "effect_size": "Not applicable; the review maps 372 documents and does not estimate a pooled intervention effect.",
    "conditions_and_limits": "Web of Science only, selected search categories and terms, bibliometric co-occurrence, and inconsistent stated search periods.",
    "criticism": "The review supports fragmentation within its retrieved corpus but cannot establish field-wide absence of cognitive or behavioral judgment research.",
    "which_workflow": "Driver’s Seat—field-positioning review and rationale for a cognitive/process research agenda.",
    "claimable_sentence": "A 2026 hybrid review maps a fragmented, technology-centered AI–entrepreneurship literature in which cognitive and behavioral judgment work remains marginal within the retrieved corpus."
  },
  {
    "id": "SaglamEtAl2026BiasAwareReject",
    "tier": "Tier 2",
    "full_citation": "Sağlam, F., Özgen, Ü., Uygun, A., Dinçer, O. S., & Albayrak, C. (2026). Selective classification under imbalance in multiclass settings: A novel metric for bias-aware risk–coverage evaluation. Journal of Biomedical Informatics, 181, 105084.",
    "link": "https://doi.org/10.1016/j.jbi.2026.105084",
    "peer_reviewed": true,
    "population": "A clinical complete-blood-count dataset with 3,316 patient records, 11 laboratory features, and 9 diagnostic classes, plus three public benchmark datasets.",
    "design": "Multidataset comparison of conventional and class-averaged risk–coverage metrics and a class-conditional coverage-matching selection strategy across five uncertainty measures.",
    "finding_verbatim": "Commonly used evaluation metrics obscure fairness issues under class imbalance.",
    "effect_size": "Rank-based comparisons favored the class-conditional strategy for CA-AUGRC (p = .009, r = .638) and AUIC (p < .001, r = .957).",
    "conditions_and_limits": "The reported advantage is for class-aware evaluation and rejection in four imbalanced multiclass datasets; results concern diagnostic/benchmark labels and specified uncertainty estimators.",
    "criticism": "Clinical classes are not educational contexts or governance opportunities. The study shows how aggregate risk–coverage can hide selective rejection, but it does not establish fairness or validity for Driver's Seat.",
    "which_workflow": "WP-08A/B: class- and context-conditional coverage and abstention comparison.",
    "claimable_sentence": "Aggregate risk–coverage summaries can conceal disproportionate rejection under imbalance, so Driver's Seat coverage should be reported by context and relevant groups."
  },
  {
    "id": "ShresthaBenMenahemVonKrogh2019",
    "tier": "Tier 3",
    "full_citation": "Shrestha, Y. R., Ben-Menahem, S. M., & von Krogh, G. (2019). Organizational decision-making structures in the age of artificial intelligence. California Management Review, 61(4), 66–83.",
    "link": "https://doi.org/10.1177/0008125619862257",
    "peer_reviewed": true,
    "population": "No empirical participant sample.",
    "design": "Conceptual comparison of human and AI decision making across search-space specificity, interpretability, alternative-set size, speed, and replicability, used to derive organizational decision structures.",
    "finding_verbatim": "The framework includes “full human to AI delegation,” sequential human–AI hybrids, and “aggregated human–AI decision making.”",
    "effect_size": "Not applicable; no empirical treatment effect or validation coefficient is reported.",
    "conditions_and_limits": "Organization-level decision-structure theory focused on combining human and AI decisions. It does not operationalize conversational episodes, entrepreneurship-specific rights, trace provenance, or learning outcomes.",
    "criticism": "The framework makes delegation direction and hybrid decision structures prior art, but it supplies no behavioral attribution instrument and does not show which configuration is superior across untested settings.",
    "which_workflow": "Driver’s Seat—foundational full-delegation, sequential-hybrid, and aggregation precursor.",
    "claimable_sentence": "Full delegation, human-to-AI and AI-to-human sequential hybrids, and aggregated human–AI decisions were already theorized in 2019."
  },
  {
    "id": "SrinivasChetan2026DelegationAugmentation",
    "tier": "Tier 2",
    "full_citation": "Srinivas, R., & Chetan, S. S. (2026). Integrating artificial intelligence in strategic decision-making: Contexts for delegation and augmentation. Group Decision and Negotiation, 35(3), Article 59.",
    "link": "https://doi.org/10.1007/s10726-026-10016-x",
    "peer_reviewed": true,
    "population": "33 C-suite and senior executives across three multinational organizations with customized LLM-based tools.",
    "design": "Gioia-method qualitative study using semi-structured interviews, structured demonstrations of in-house tools, and secondary documents.",
    "finding_verbatim": "a grounded framework distinguishing high-confidence AI delegation of analytical work from iterative human-AI co-deliberation",
    "effect_size": "No quantitative effect size; qualitative study.",
    "conditions_and_limits": "Delegation was described for structured, predictable analytical work; ill-structured and uncertain strategic decisions elicited human primacy and iterative co-deliberation.",
    "criticism": "Self-reported executive accounts and staged demonstrations, not observation of live strategic decisions; three AI-mature multinationals; no performance-optimality test or behavioral coefficient.",
    "which_workflow": "Driver's Seat contingency question—when delegation versus augmentation is appropriate.",
    "claimable_sentence": "Executives described delegation as contingent on problem structuredness and situational predictability, supporting a task-fit interpretation rather than a universal preference for more human control."
  },
  {
    "id": "Srivastava2026AIDelegationModel",
    "tier": "Tier 2",
    "full_citation": "Srivastava, A. (2026). EXPRESS: Governing AI-enabled decision making: Delegation, autonomy, and control at the operations–marketing interface. Production and Operations Management, advance online publication.",
    "link": "https://doi.org/10.1177/10591478261473004",
    "peer_reviewed": true,
    "population": "No human sample; modeled firm jointly chooses pricing, inventory, and governance under demand uncertainty.",
    "design": "Analytical operations model comparing human control, full AI autonomy, and human-in-the-loop governance, with extensions for endogenous investment, learning, lead times, oversight, and scale.",
    "finding_verbatim": "partial delegation can strictly dominate both full autonomy and full human control",
    "effect_size": "Not applicable; theoretical thresholds and equilibria rather than empirical effects.",
    "conditions_and_limits": "Results follow from stated assumptions about demand variance, AI imperfection, responsiveness, buffers, risk asymmetry, and organizational structure.",
    "criticism": "No observed humans, interactions, or measurement instrument; optimality is model-contingent and cannot validate a trace score or learning claim.",
    "which_workflow": "Driver's Seat theory boundary—economic contingency and hybrid delegation.",
    "claimable_sentence": "Analytical work shows that partial delegation may be optimal under some assumptions, reinforcing that the normative target is appropriate allocation rather than maximum human activity."
  },
  {
    "id": "TownsendHunt2019",
    "tier": "Tier 3",
    "full_citation": "Townsend, D. M., & Hunt, R. A. (2019). Entrepreneurial action, creativity, & judgment in the age of artificial intelligence. Journal of Business Venturing Insights, 11, e00126.",
    "link": "https://doi.org/10.1016/j.jbvi.2019.e00126",
    "peer_reviewed": true,
    "population": "No empirical participant sample.",
    "design": "Conceptual agenda connecting AI, entrepreneurial action, creativity, desirability, possibility, and judgment under modal uncertainty.",
    "finding_verbatim": "The article centers “entrepreneurial action, creativity, & judgment in the age of artificial intelligence.”",
    "effect_size": "Not applicable; no empirical effect is reported.",
    "conditions_and_limits": "Early conceptual agenda rather than an agency measure, trace study, or validation test.",
    "criticism": "Describe it as an early JBVI agenda, not the journal’s singular foundational agenda. It supplies no actor-allocation instrument.",
    "which_workflow": "Driver’s Seat—journal and domain positioning for AI-enabled entrepreneurial judgment.",
    "claimable_sentence": "JBVI had already placed AI, creativity, entrepreneurial action, and judgment under ambiguity on its agenda by 2019."
  },
  {
    "id": "VaccaroAlmaatouqMalone2024",
    "tier": "Tier 1",
    "full_citation": "Vaccaro, M., Almaatouq, A., & Malone, T. W. (2024). When combinations of humans and AI are useful: A systematic review and meta-analysis. Nature Human Behaviour, 8(12), 2293–2303.",
    "link": "https://doi.org/10.1038/s41562-024-02024-1",
    "peer_reviewed": true,
    "population": "74 peer-reviewed papers reporting 106 human-participant experiments and 370 effect sizes. Eligible experiments compared human-only, AI-only, and combined human–AI performance; searches covered 1 January 2020 through 30 June 2023.",
    "design": "Preregistered interdisciplinary systematic review with a three-level meta-analytic model. Primary outcomes were synergy relative to the better solo performer and augmentation relative to the human alone; task type and relative solo performance were tested as moderators.",
    "finding_verbatim": "On average, “human–AI combinations performed significantly worse than the best of humans or AI alone.”",
    "effect_size": "Synergy versus the better solo performer: Hedges’ g=−.23, 95% CI [−.39, −.07], p=.005. Augmentation versus humans alone: g=.64, 95% CI [.53, .74]. Decision-task synergy: g=−.27, 95% CI [−.44, −.10]; creation-task synergy: g=.19, 95% CI [−.09, .48], p=.180. When humans outperformed AI alone, synergy g=.46, 95% CI [.28, .66]; when AI outperformed humans, synergy g=−.54, 95% CI [−.71, −.37]. Heterogeneity was I²=97.7% for synergy and 93.8% for augmentation.",
    "conditions_and_limits": "The result applies to experiments reporting all three performance baselines and to the tasks, processes, and populations selected for study. About 85% of effect sizes involved finite-choice decision tasks and about 10% creation tasks. More than 95% of systems had humans make the final decision; only three experiments predetermined separate human/AI subtasks.",
    "criticism": "Very high heterogeneity, possible topic-selection and publication bias, varied outcome metrics and measurement quality, and predominantly laboratory configurations limit generalization. A positive creation-task point estimate was not statistically different from zero. The meta-analysis measures performance, not governance or learning.",
    "which_workflow": "Driver’s Seat—configuration-contingency proposition, independent performance criterion, and evidence against assuming that human–AI combination or greater human participation is inherently superior.",
    "claimable_sentence": "Across 106 experiments, human–AI systems outperformed humans alone on average but underperformed the better solo performer; losses were concentrated in decision tasks and varied sharply with relative human-versus-AI capability."
  },
  {
    "id": "WuYao2026AfterInterface",
    "tier": "Tier 2",
    "full_citation": "Wu, M., & Yao, M. (2026). After the interface: Relocating human agency in the age of conversational AI. In Proceedings of the 8th ACM Conference on Conversational User Interfaces (pp. 1–7). ACM.",
    "link": "https://doi.org/10.1145/3816046.3816301",
    "peer_reviewed": true,
    "population": "No participant sample; conceptual CUI paper.",
    "design": "Conceptual argument distinguishing process control from outcome control and mapping conversational, generative, and agentic systems across that space.",
    "finding_verbatim": "Agency has not diminished but has relocated.",
    "effect_size": "Not applicable; no empirical test.",
    "conditions_and_limits": "The authors locate contemporary agency in goal articulation, output evaluation, and outcome negotiation and explicitly acknowledge that outcome-based agency may be illusory for unverifiable outputs.",
    "criticism": "The central assertion is a conceptual provocation rather than an empirical finding; the paper offers no scoring rule, reliability, external validation, or consequential-task data.",
    "which_workflow": "Driver's Seat construct boundary—process versus outcome control and goal/evaluation/negotiation rights.",
    "claimable_sentence": "Goal articulation, evaluation, and outcome negotiation are already identified as relocated forms of agency, narrowing novelty around purpose framing, evaluation, and commitment."
  },
  {
    "id": "WuEtAl2026HumanAgencyCreativity",
    "tier": "Tier 2",
    "full_citation": "Wu, S. H., Yang, Y., Lee, A. Y., Liebscher, A., Rapuano, K., Niederhoffer, K., & Hancock, J. T. (2026). The role of human agency in human-AI co-creativity. In Proceedings of the 2026 Conference on Creativity and Cognition (pp. 1510–1515). ACM.",
    "link": "https://doi.org/10.1145/3803784.3816857",
    "peer_reviewed": true,
    "population": "150 idea submitters and 142 evaluators were recruited; inferential analyses used smaller samples after exclusions.",
    "design": "Preregistered experiment assigning participants to high-agency AI, low-agency AI, or human-only conditions for alternative-uses ideation, with separate evaluations and similarity analyses.",
    "finding_verbatim": "ideas did not differ in individual-level quality (overall creativity, originality, and usefulness)",
    "effect_size": "The paper reports a low-agency usefulness coefficient of β=.20, p=.04, and a homogenization test of F(2,121.67)=3.15, p<.05, partial η²=.07; interpret within the paper's final analytic samples.",
    "conditions_and_limits": "Agency was experimentally framed for creative ideation with LLM access; collective homogenization was assessed through similarity-to-centroid and exploratory clustering.",
    "criticism": "A brief conference paper, artificial ideation task, and framing manipulation do not show judgment-right governance in consequential decisions. No individual-level quality advantage emerged across conditions.",
    "which_workflow": "Driver's Seat value question—why more human agency is not automatically better; collective homogenization as a possible outcome.",
    "claimable_sentence": "Agency framing did not improve individual idea quality, although low-agency AI use showed evidence of collective homogenization; more human activity should not be treated as universally superior."
  },
  {
    "id": "XieEtAl2026",
    "tier": "Tier 2",
    "full_citation": "Xie, Y., Qi, T., Yi, J., Yang, X., Whalen, R., Huang, J., Ding, Q., Xie, Y., Xie, X., & Wu, F. (2026). Measuring human contribution in AI-assisted content generation. In Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (pp. 6168–6190). Association for Computational Linguistics.",
    "link": "https://doi.org/10.18653/v1/2026.acl-long.279",
    "peer_reviewed": true,
    "population": "Computational dataset with 2,000 entries per domain across abstracts, news, patents, and poems, spanning four input regimes, six LLMs, and five outputs per input. Human validation used 1,500 pairwise comparisons with three annotators.",
    "design": "Information-theoretic measure of human contribution, φ=I(x;y)/I(y), followed by corpus experiments and deliberately separated pairwise human comparisons.",
    "finding_verbatim": "The authors state: “we quantify the proportional information contribution of humans in content generation.”",
    "effect_size": "Human ranking aligned with the metric on 95.93% of the evaluated pairs, which were deliberately selected to differ by more than .1. This does not estimate adjacent-score precision.",
    "conditions_and_limits": "Text-generation domains and informational provenance, not governing authority, judgment quality, or consequential commitment. Human Contribution Ratio is convenient shorthand, not clearly the paper’s branded term.",
    "criticism": "A short decisive human intervention and verbose low-governance input can receive misleadingly different contribution values. The selected-pair validation should not be generalized to fine-grained score accuracy.",
    "which_workflow": "Driver’s Seat—nearest formal human-contribution coefficient and discriminator for the governance-versus-contribution distinction.",
    "claimable_sentence": "Xie et al. quantify informational human contribution to AI-assisted text, but the coefficient does not identify who held evaluative or commitment authority."
  },
  {
    "id": "XuEtAl2026",
    "tier": "Tier 2",
    "full_citation": "Xu, T., Chen, Y., Zhu, B., Fan, B., Wu, Y., & Jiang, Y. (2026). AI agency drives college students’ entrepreneurial thinking through human sense of agency in human and AI symbiosis. Scientific Reports. Advance online publication.",
    "link": "https://doi.org/10.1038/s41598-026-60406-z",
    "peer_reviewed": true,
    "population": "Two Chinese college-student samples: n=398 for scale development and n=574 for the structural model/fsQCA, rather than one 972-person SEM sample.",
    "design": "Cross-sectional self-report scale development followed by SEM and fsQCA linking perceived AI cognitive, interaction, and action support with human sense of agency, opportunity recognition, and creativity.",
    "finding_verbatim": "“AI agency is positively associated with entrepreneurial thinking both directly and indirectly through the human sense of agency.”",
    "effect_size": "The review does not quote path coefficients or standardized effects; none are reconstructed here. The two samples should not be pooled into a new effect or described as a single SEM sample.",
    "conditions_and_limits": "Self-reported constructs, cross-sectional common method, college entrepreneurship-learning context, and no observed interaction trace or independent transfer measure.",
    "criticism": "The title’s causal verb exceeds what the cross-sectional design establishes. Perceived support and perceived agency do not identify behavioral decision locus.",
    "which_workflow": "Driver’s Seat—entrepreneurship-adjacent convergent comparator and causal-language caution.",
    "claimable_sentence": "Across separate scale-development and modeling samples, perceived AI support related to self-reported agency and entrepreneurial thinking; the study did not observe who governed decisions."
  },
  {
    "id": "YunTaranovaWang2026ChatbotAgency",
    "tier": "Tier 2",
    "full_citation": "Yun, B., Taranova, E., & Wang, A. Y. (2026). Does my chatbot have an agenda? Understanding human and AI agency in human-human-like chatbot interaction. In Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems (pp. 1–32). ACM.",
    "link": "https://doi.org/10.1145/3772318.3791620",
    "peer_reviewed": true,
    "population": "22 adults who used an LLM companion during a month-long study.",
    "design": "Longitudinal use of a purpose-built companion chatbot, followed by semi-structured interviews, post-hoc elicitation, cross-participant chat review, and a strategy reveal.",
    "finding_verbatim": "control shifted and was co-constructed turn-by-turn",
    "effect_size": "No quantitative effect size; qualitative longitudinal study.",
    "conditions_and_limits": "The framework maps human, AI, or hybrid agency across intention, execution, adaptation, delimitation, and negotiation in companion chat.",
    "criticism": "Small, purpose-built companion-chat context; participant perceptions and selected moments do not validate a general-purpose behavioral coefficient or link agency to performance or learning.",
    "which_workflow": "Driver's Seat dynamics and construct boundary—actor-by-action attribution over time.",
    "claimable_sentence": "Yun and colleagues show qualitative, turn-by-turn redistribution of agency, closely occupying the dynamic actor-by-right space without offering a validated score."
  },
  {
    "id": "ZhangEtAl2026MNARMissing",
    "tier": "Tier 2",
    "full_citation": "Zhang, J., Lu, J., & Zhang, Z. (2026). Modeling missing response data in item response theory: Addressing missing not at random mechanism with monotone missing characteristics. Journal of Educational Measurement, 63(1), e12428. First published online February 24, 2025.",
    "link": "https://doi.org/10.1111/jedm.12428",
    "peer_reviewed": true,
    "population": "Four simulation studies and an application to PISA 2015 science data.",
    "design": "Bayesian item-response model for missing-not-at-random responses using monotone missing characteristics, model-comparison criteria, and slice-sampling estimation.",
    "finding_verbatim": "This study introduces a missing data model based on the missing not at random (MNAR) mechanism.",
    "effect_size": "No single effect size; the evidence consists of simulation performance and an empirical model illustration.",
    "conditions_and_limits": "The model uses cumulative prior missing indicators to represent monotone individual missingness and compares constrained MAR and MNAR specifications.",
    "criticism": "Monotone item nonresponse in an assessment is not equivalent to trace eligibility in conversation. Driver's Seat abstention can arise from several nonmonotone task, capture, opportunity, and behavior mechanisms.",
    "which_workflow": "WP-08a Phase A: sensitivity to missing-not-at-random mechanisms.",
    "claimable_sentence": "Educational nonresponse can require an MNAR model when missingness depends on latent or prior response processes; scored-only summaries should not assume ignorability."
  },
  {
    "id": "ZhangWangYi2025AgencyReview",
    "tier": "Tier 2",
    "full_citation": "Zhang, S., Wang, H., & Yi, X. (2025). Exploring collaboration patterns and strategies in human-AI co-creation through the lens of agency: A scoping review of the top-tier HCI literature. Proceedings of the ACM on Human-Computer Interaction, 9(7), Article CSCW413, 1–43.",
    "link": "https://doi.org/10.1145/3757594",
    "peer_reviewed": true,
    "population": "134 papers from top-tier HCI and CSCW venues over approximately 20 years.",
    "design": "Scoping review mapping agency configurations, control mechanisms, and interaction contexts.",
    "finding_verbatim": "an integrated theoretical framework structuring agency patterns, control mechanisms, and interaction contexts",
    "effect_size": "No quantitative effect size; scoping synthesis.",
    "conditions_and_limits": "The corpus is restricted to selected top-tier HCI/CSCW venues and co-creative settings. ACM records a 2026 corrigendum (doi:10.1145/3779003); use the corrected version of record.",
    "criticism": "The review demonstrates a mature agency literature but does not provide a validated episode-level coefficient or establish prediction of outcomes.",
    "which_workflow": "WP-08A/B: agency configurations and operational control mechanisms.",
    "claimable_sentence": "A 134-paper scoping review already systematizes agency patterns and control mechanisms, ruling out a claim that Driver's Seat is the first human–AI agency framework."
  },
  {
    "id": "ZhuEtAl2026",
    "tier": "Tier 3",
    "full_citation": "Zhu, L., Lu, Q., Ding, M., Lee, S. U., & Wang, C. (2026). Designing meaningful human oversight in AI. AI and Ethics, 6(3), Article 286.",
    "link": "https://doi.org/10.1007/s43681-026-01147-7",
    "peer_reviewed": true,
    "population": "No human participant sample. A structured known-use analysis screened 54 public cases and retained 12.",
    "design": "Conceptual/interpretive framework separating AI operative agency from human evaluative agency; two authors independently extracted and adjudicated evidence from retained cases.",
    "finding_verbatim": "The authors distinguish “AI operative agency” from “human evaluative agency” and reject a zero-sum relation between them.",
    "effect_size": "Not applicable. The case analysis reports no participant-level effect or psychometric coefficient.",
    "conditions_and_limits": "The 12 cases rely heavily on vendor technical documents and product whitepapers; the authors characterize the work as interpretive and call for empirical validation.",
    "criticism": "Under the library’s no-vendor-material rule, cite the peer-reviewed theoretical distinction, not the vendor case claims. Nominal human presence is not meaningful oversight.",
    "which_workflow": "Driver’s Seat—direct precursor to the separate governance and AI-operative-contribution axes.",
    "claimable_sentence": "Human evaluative agency and AI operative agency can both be high, so AI contribution must not be mechanically subtracted from human governance."
  }
]
