[
    {
        "id":  "AcostaPradoEtAl2026LeadershipHAI",
        "tier":  "Tier 3",
        "full_citation":  "Acosta-Prado, J. C., Camargo, J. P., Zárate-Torres, R. A., \u0026 Rey-Sarmiento, C. F. (2026). Leadership and human–AI collaboration: A measurement scale. Behavioral Sciences, 16(7), 1208.",
        "link":  "https://doi.org/10.3390/bs16071208",
        "peer_reviewed":  true,
        "population":  "170 self-administered online responses from Colombian companies spanning organization sizes and economic sectors.",
        "design":  "Instrument-development study using content evidence, exploratory factor analysis, confirmatory factor analysis, and internal-consistency estimates for a 30-item five-point self-report scale.",
        "finding_verbatim":  "The results show that the proposed measurement scale meets the psychometric properties required of a social-science instrument.",
        "effect_size":  "For the six-factor human–AI collaboration CFA, CFI = .970, TLI = .965, RMSEA = .074, and SRMR = .057; reported alpha ranged .763–.881 and omega .784–.891 across dimensions.",
        "conditions_and_limits":  "Items assess respondent perceptions, including balance between human judgment and AI output, assignment of final responsibility, and preservation of individual autonomy.",
        "criticism":  "EFA and CFA used the same n = 170 sample; there was no independent cross-validation, behavioral trace, episode-level unit, external criterion, or predictive validation. Positive-worded self-reports are vulnerable to common-response bias.",
        "which_workflow":  "WP-08A Phase A: construct comparison and field-level prior-art boundary.",
        "claimable_sentence":  "A self-report scale already measures perceived human–AI balance, responsibility, and autonomy; behavioral episode attribution remains a separate measurement task."
    },
    {
        "id":  "Ali2026",
        "tier":  "Tier 3",
        "full_citation":  "Ali, M. S. (2026). From assistants to agents: A relational framework for human–AI co-agency. AI and Ethics, 6(3), Article 280.",
        "link":  "https://doi.org/10.1007/s43681-026-01111-5",
        "peer_reviewed":  true,
        "population":  "No empirical participant sample.",
        "design":  "Conceptual and normative relational framework organized around initiative, decision scope, oversight, and responsibility attribution.",
        "finding_verbatim":  "The article frames human–AI agency as “co-agency” across initiative, decision scope, oversight, and responsibility.",
        "effect_size":  "Not applicable; no empirical test or numerical effect is reported.",
        "conditions_and_limits":  "General human–AI relations rather than educational or entrepreneurial traces. The paper provides no episode boundaries, behavioral anchors, opportunity rules, reliability, or criterion evidence.",
        "criticism":  "It occupies the conceptual space but cannot validate actor attribution or a score. Normative responsibility and visibly enacted commitment should not be collapsed.",
        "which_workflow":  "Driver’s Seat—relational-agency precursor and construct-boundary source.",
        "claimable_sentence":  "Initiative, decision scope, oversight, and responsibility are established co-agency dimensions; Driver’s Seat must contribute an operational test rather than rename them."
    },
    {
        "id":  "AngelopoulosEtAl2023",
        "tier":  "Tier 3",
        "full_citation":  "Angelopoulos, S., Bendoly, E., Fransoo, J., Hoberg, K., Ou, C., \u0026 Tenhiälä, A. (2023). Digital transformation in operations management: Fundamental change through agency reversal. Journal of Operations Management, 69(6), 876–889.",
        "link":  "https://doi.org/10.1002/joom.1271",
        "peer_reviewed":  true,
        "population":  "No empirical participant sample.",
        "design":  "Conceptual account of agency reversal along a human-driven/technology-supported to technology-driven/human-supported spectrum.",
        "finding_verbatim":  "The article names the shift “agency reversal.”",
        "effect_size":  "Not applicable; no empirical effect, sample, or psychometric coefficient is reported.",
        "conditions_and_limits":  "Operations-management framing and broad digital transformation; the article calls for future operationalization.",
        "criticism":  "It is not an empirical study and cannot validate Driver’s Seat, but it does establish that technology may direct actions humans carry out.",
        "which_workflow":  "Driver’s Seat—authority-reversal precursor and automation/augmentation boundary.",
        "claimable_sentence":  "Agency reversal already describes technology directing moves performed by humans; Driver’s Seat must operationalize that possibility rather than rename it."
    },
    {
        "id":  "BairdMaruping2021",
        "tier":  "Tier 3",
        "full_citation":  "Baird, A., \u0026 Maruping, L. M. (2021). The next generation of research on IS use: A theoretical framework of delegation to and from agentic IS artifacts. MIS Quarterly, 45(1), 315–341.",
        "link":  "https://doi.org/10.25300/MISQ/2021/15882",
        "peer_reviewed":  true,
        "population":  "No empirical participant sample.",
        "design":  "Conceptual IS-delegation framework with the human–agentic-IS-artifact dyad as the elemental unit. It organizes delegation around agent endowments, preferences, and roles and the mechanisms of appraisal, distribution, and coordination.",
        "finding_verbatim":  "The authors “introduce delegation ... as a foundational and powerful lens” for human–agentic-artifact relationships.",
        "effect_size":  "Not applicable; the article develops theory and testable-model guidance but reports no empirical effect.",
        "conditions_and_limits":  "General information-systems theory rather than generative-AI dialogue, entrepreneurship, education, or a validated behavioral measure. Rights and responsibilities are theorized, not trace-coded.",
        "criticism":  "The framework establishes delegation as prior conceptual territory but cannot validate episode boundaries, actor attribution, opportunity rules, score reliability, or links to quality and learning.",
        "which_workflow":  "WP-08A Phase A: foundational delegation and rights/responsibilities precursor for actor allocation.",
        "claimable_sentence":  "Baird and Maruping make appraisal, distribution, and coordination of delegation in human–agentic-artifact dyads established IS theory, not a new Driver’s Seat idea."
    },
    {
        "id":  "BastaniEtAl2025",
        "tier":  "Tier 1",
        "full_citation":  "Bastani, H., Bastani, O., Sungu, A., Ge, H., Kabakcı, Ö., \u0026 Mariman, R. (2025). Generative AI without guardrails can harm learning: Evidence from high school mathematics. Proceedings of the National Academy of Sciences, 122(26), e2422633122.",
        "link":  "https://doi.org/10.1073/pnas.2422633122",
        "peer_reviewed":  true,
        "population":  "Nearly 1,000 students in grades 9–11 at one private high school in Turkey.",
        "design":  "Preregistered classroom-cluster randomized trial across four sessions comparing control, generic GPT-4, and a teacher-grounded GPT Tutor that supplied hints and withheld direct answers; assisted practice was followed by immediate unassisted exams.",
        "finding_verbatim":  "The authors conclude that “generative AI without guardrails can harm learning.”",
        "effect_size":  "Assisted practice: GPT Base +.137 on the normalized outcome, about 48% relative to the control mean; GPT Tutor +.361, about 127%. Unassisted exam: GPT Base −.054, about −17%; GPT Tutor −.004, nonsignificant.",
        "conditions_and_limits":  "One private high school, mathematics, short-duration intervention, classroom-level randomization, and immediate unassisted exams. The guarded tutor contained teacher-designed answers and scaffolds.",
        "criticism":  "The guarded tutor removed the detected exam penalty but did not improve the unassisted exam. The study does not measure Driver’s Seat, durable learning, or long-term transfer.",
        "which_workflow":  "Driver’s Seat—learning-outcome boundary and guardrail mechanism, not construct validation.",
        "claimable_sentence":  "Generic GPT improved assisted practice but reduced immediate unassisted performance, while a teacher-grounded tutor removed that penalty without producing a positive unassisted-exam effect."
    },
    {
        "id":  "BilalEtAl2026FinancialDelegation",
        "tier":  "Tier 3",
        "full_citation":  "Bilal, I. M., Wang, Y. C., Raj, A., Giovagnini, F., Tewari, P., Zhang, Y., Liou, M.-C. Z., \u0026 Zaman, Q. (2026). From information to delegation: Mapping human-AI financial decision making [Preprint]. arXiv:2608.02100.",
        "link":  "https://arxiv.org/abs/2608.02100",
        "peer_reviewed":  false,
        "population":  "1.53 million user prompts from 6,304 voluntary opt-in ChatGPT and Gemini users in the United States and India, covering August–October 2025.",
        "design":  "Behavioral trace classification of finance conversations by financial domain, intent, and a three-level decision-authority mapping (Inform, Shape, Act), using manually annotated and synthetic training data.",
        "finding_verbatim":  "Consumers overwhelmingly use AI to retrieve information and shape financial judgement, while delegation of financial execution remains rare.",
        "effect_size":  "Finance classifier accuracy=96.5 and F1=97.3. Intent classifier F1-micro=70.6 and F1-macro=72.3. These are classifier-performance metrics, not effects of AI use.",
        "conditions_and_limits":  "The sample skews young and differs from national populations; actual transactions and off-platform decisions are unobserved. High-authority intent classes used 82.03%–96.01% synthetic training examples because such organic cases were under 1%.",
        "criticism":  "Preprint; all authors are affiliated with Stripe Partners; authority is assigned by fixed intent-to-level mapping rather than independently coded right by right; annotator reliability is not reported; sessions of ten or fewer prompts are not segmented; observed requests cannot prove who retained ultimate authority.",
        "which_workflow":  "WP-08A Phase A: large-scale trace-derived delegated decision authority in a consequential domain.",
        "claimable_sentence":  "Bilal and colleagues already operationalize delegated decision authority from real conversational traces, eliminating any defensible claim that Driver\u0027s Seat is the first behavioral authority measure; the remaining gap is a validated, multi-right episode measure with explicit abstention."
    },
    {
        "id":  "Bousmah2026",
        "tier":  "Tier 3",
        "full_citation":  "Bousmah, M. (2026). LLMography: Transforming human–AI conversations into traceability, oversight, and auditability indicators [Preprint]. arXiv.",
        "link":  "https://doi.org/10.48550/arXiv.2606.29437",
        "peer_reviewed":  false,
        "population":  "19 anonymized engineering-student audit reports containing 462 turns; the paper also applies the prototype to its own writing process.",
        "design":  "Exploratory prototype in which an LLM analyzer generates Prompt Quality, Human Direction, AI Dependency, Auditability, Final Output Traceability, and Privacy Risk indicators plus categorical labels.",
        "finding_verbatim":  "“Outputs are not enough; we need the history of interaction.”",
        "effect_size":  "Reported sample means were Human Direction 86.8/100, Prompt Quality 81.9/100, Auditability 72.8/100, and Final Output Traceability 77.1/100. Labels were 14 co-produced, three human-directed, two minimal-AI, and none AI-dominant or insufficient.",
        "conditions_and_limits":  "Small, selected engineering-report sample and whole-conversation scores. The analyzer prompt permits the model to compute or estimate values; some single-turn records receive extreme scores.",
        "criticism":  "No published formula, behavioral anchors, independent human-coded ground truth, calibration, inter-rater reliability, repeatability study, or external criterion. Reported values are prototype outputs, not validated measurements.",
        "which_workflow":  "WP-08A Phase A: direct trace-based human-direction and provenance comparator.",
        "claimable_sentence":  "LLMography already generates Human Direction and AI Dependency indicators from traces, but its exploratory LLM-estimated scores have no reported independent validation."
    }
,
    {
        "id":  "Chen2026AlgorithmicBoundedRationality",
        "tier":  "Tier 3",
        "full_citation":  "Chen, Z. S. (2026). Rethinking managerial rationality in the age of AI: A human–machine collaboration perspective on organizational decision-making. Management Decision, 1–18. Advance online publication.",
        "link":  "https://doi.org/10.1108/MD-07-2025-1890",
        "peer_reviewed":  true,
        "population":  "Managerial decision architectures involving AI; no participant sample.",
        "design":  "Conceptual synthesis of bounded rationality and socio-technical systems theory, deriving an algorithmic-bounded-rationality framework and propositions.",
        "finding_verbatim":  "Human–AI collaboration represents a distinct rationality configuration rather than a midpoint between automation and human judgment.",
        "effect_size":  "Not applicable; no empirical study.",
        "conditions_and_limits":  "The framework distinguishes AI-led, human-first, and collaborative modes as contingent on data intensity, contextual ambiguity, accountability demands, and governance maturity.",
        "criticism":  "The modes and fit propositions are untested. The article provides no behavioral coding, episode score, psychometric validation, entrepreneurship-specific judgment model, or outcome comparison.",
        "which_workflow":  "WP-08a Phase A: post-cutoff managerial-mode landscape and two-axis comparison.",
        "claimable_sentence":  "New managerial theory treats AI-led, human-first, and collaborative modes as contingent configurations, reinforcing the need to separate configuration description from claims of superiority."
    },
    {
        "id":  "Chow1970RejectTradeoff",
        "tier":  "Tier 2",
        "full_citation":  "Chow, C. K. (1970). On optimum recognition error and reject tradeoff. IEEE Transactions on Information Theory, 16(1), 41–46.",
        "link":  "https://doi.org/10.1109/TIT.1970.1054406",
        "peer_reviewed":  true,
        "population":  "Formal pattern-recognition systems; worked examples use normal and uniform distributions rather than human participants.",
        "design":  "Mathematical derivation of an optimum rejection rule and the relationship between recognition error and rejection probability.",
        "finding_verbatim":  "The performance of a pattern recognition system is characterized by its error and reject tradeoff.",
        "effect_size":  "Not applicable; this is a theoretical result, not an empirical effect-size study.",
        "conditions_and_limits":  "The result concerns classification with specified posterior probabilities and rejection costs. Rejecting more cases changes both coverage and conditional error.",
        "criticism":  "Driver\u0027s Seat episodes are not ordinary labeled pattern-recognition cases, and score eligibility may itself be informative. Chow supplies an analogy and reporting principle, not validity evidence for the measure.",
        "which_workflow":  "WP-08a Phase A: abstention and risk–coverage rationale.",
        "claimable_sentence":  "A reject option creates an explicit error–coverage trade-off; reporting only performance among retained cases is therefore incomplete."
    },
    {
        "id":  "CoreMooreZinn2003",
        "tier":  "Tier 2",
        "full_citation":  "Core, M. G., Moore, J. D., \u0026 Zinn, C. (2003). The role of initiative in tutorial dialogue. In Proceedings of the 10th Conference of the European Chapter of the Association for Computational Linguistics (pp. 67–74). Association for Computational Linguistics.",
        "link":  "https://doi.org/10.3115/1067807.1067818",
        "peer_reviewed":  true,
        "population":  "Twenty-three human–human typed tutoring sessions: three trial, ten Socratic, and ten didactic; the same tutor conducted the sessions.",
        "design":  "Corpus annotation study comparing initiative management in the ten Socratic and ten didactic sessions, with learning correlations across all 23 sessions and reliability coding.",
        "finding_verbatim":  "“There was no direct relationship between student initiative and learning.”",
        "effect_size":  "Initiative coding κ=.92 on 757 examples. Initiative versus learning gain r=−.0689, n=23, nonsignificant. Student word share r=.60, p\u003c.005; utterance share r=.56, p\u003c.005; tutor-question share r=.46, p\u003c.05.",
        "conditions_and_limits":  "Small human–human tutoring corpus, one tutor, typed basic-electronics dialogue, and a crude initiative code that omitted task initiative.",
        "criticism":  "High coding agreement did not make initiative a learning indicator. Turn control and word share are not equivalent to substantive evaluative or commitment governance.",
        "which_workflow":  "Driver’s Seat—classic measurement warning separating reliability, initiative, interactivity, and learning.",
        "claimable_sentence":  "A reliably coded dialogue-initiative measure was unrelated to learning gain in this small tutoring study, showing that reliability does not establish substantive validity or learning relevance."
    },
    {
        "id":  "CristofaroGiardinoMuldoon2026",
        "tier":  "Tier 2",
        "full_citation":  "Cristofaro, M., Giardino, P. L., \u0026 Muldoon, J. (2026). Entrepreneurial decision-making in the age of AI: Sector knowledge at the balance of intuition and analysis. Technology in Society, 85, 103200.",
        "link":  "https://doi.org/10.1016/j.techsoc.2025.103200",
        "peer_reviewed":  true,
        "population":  "135 entrepreneurs recruited and 124 retained, with 31 cases per cell; outputs were assessed by three blinded investor raters.",
        "design":  "Controlled 90-minute GPT-4 entrepreneurial decision task with a factorial cell structure. Pre-existing sector knowledge was measured rather than experimentally assigned.",
        "finding_verbatim":  "The authors report enhanced “contextual understanding” alongside lower novelty and innovation under AI-assisted conditions.",
        "effect_size":  "No numerical effect is entered because the published main-effect and cell-mean tables conflict in direction and magnitude. The discrepancy requires author or publisher clarification.",
        "conditions_and_limits":  "One controlled task, short duration, output/rater outcomes, and measured sector expertise. No trace-based actor allocation or independent learning/transfer test.",
        "criticism":  "The published numerical tables conflict, so effects should not be quoted, and sector knowledge should not be described as randomized.",
        "which_workflow":  "Driver’s Seat—entrepreneurship-adjacent criterion candidate and expertise contingency, currently unsuitable for exact-effect validation.",
        "claimable_sentence":  "The experiment reports more opportunity generation and analytical depth with stronger contextual understanding but lower novelty; exact effects remain unusable because its tables conflict."
    },
    {
        "id":  "Cukurova2026",
        "tier":  "Tier 3",
        "full_citation":  "Cukurova, M. (2026). Agency as a system property in human–AI interaction in education. British Journal of Educational Technology, 57(4), 1065–1070.",
        "link":  "https://doi.org/10.1111/bjet.70060",
        "peer_reviewed":  true,
        "population":  "No new empirical participant sample.",
        "design":  "Conceptual commentary treating agency as a property of sociotechnical configurations and focusing on accept, reject, and transform micro-decisions.",
        "finding_verbatim":  "Agency is framed as “a system property” rather than a stable possession of either a learner or an AI system.",
        "effect_size":  "Not applicable; no empirical effect is reported.",
        "conditions_and_limits":  "Education-focused conceptual argument. Calls for process traces, reasoning quality, ownership/control, and transfer but does not validate an instrument.",
        "criticism":  "It supports context-sensitive interpretation but supplies no scoring rules, reliability, criterion evidence, or entrepreneurship-specific rights.",
        "which_workflow":  "Driver’s Seat—system-property theoretical anchor and warning against trait-like person scores.",
        "claimable_sentence":  "A trace profile should be interpreted as a state of an observed sociotechnical configuration, not as proof that a student possesses a fixed amount of agency."
    },
    {
        "id":  "DaiEtAl2026",
        "tier":  "Tier 1",
        "full_citation":  "Dai, Y., Liu, S., Zhou, S., Lai, S., Liu, A., \u0026 Lim, C. P. (2026). Redefining and measuring student agency in AI-assisted learning: Development and validation of the agentic engagement with AI (AE-AI) scale. Computers \u0026 Education, 253, 105687.",
        "link":  "https://doi.org/10.1016/j.compedu.2026.105687",
        "peer_reviewed":  true,
        "population":  "Interviews with 26 students, exploratory factor analysis with 340 respondents, and confirmatory factor analysis with 256 respondents.",
        "design":  "Mixed-method scale development and psychometric validation. An initial 28-item pool was reduced to a 16-item self-report scale with Adaptive Direction, Critical Integration, Cross-Source Inquiry, and Reflective Calibration factors.",
        "finding_verbatim":  "The validated dimensions are “adaptive direction, critical integration, cross-source inquiry, and reflective calibration.”",
        "effect_size":  "Factor-loadings and fit evidence are reported in the article; no single intervention effect applies. No trace-to-reference-coder agreement coefficient was tested.",
        "conditions_and_limits":  "Self-reported engagement and agency, not native behavioral observation. The abstract’s use of “observing” does not indicate log or dialogue coding.",
        "criticism":  "The scale cannot allocate rights to actors or establish temporal transitions, commitments, learning, or transfer. Convergence with Driver’s Seat would be informative but would not imply equivalence.",
        "which_workflow":  "Driver’s Seat—principal validated self-report comparator for convergent and discriminant evidence.",
        "claimable_sentence":  "AE-AI is a validated four-factor self-report measure of agentic engagement with AI, not an actor-by-right trace measure."
    },
    {
        "id":  "DarvishiEtAl2024",
        "tier":  "Tier 1",
        "full_citation":  "Darvishi, A., Khosravi, H., Sadiq, S., Gašević, D., \u0026 Siemens, G. (2024). Impact of AI assistance on student agency. Computers \u0026 Education, 210, 104967.",
        "link":  "https://doi.org/10.1016/j.compedu.2023.104967",
        "peer_reviewed":  true,
        "population":  "1,625 undergraduates across ten courses.",
        "design":  "Randomized multi-course study after four common weeks of AI-assisted peer-feedback activity, comparing withdrawal of AI, continued AI, and continued AI plus a self-monitoring checklist.",
        "finding_verbatim":  "The study examines the “impact of AI assistance on student agency.”",
        "effect_size":  "Withdrawal worsened several automated peer-feedback indicators; AI plus checklist did not outperform AI alone. The review does not supply a standardized effect or confidence interval, so none is reconstructed here.",
        "conditions_and_limits":  "Outcomes were automated feedback features rather than independent knowledge, retention, or transfer. Effects concern a specific peer-feedback system and withdrawal schedule.",
        "criticism":  "The phrase “relied rather than learned” is an interpretation, not a tested learning contrast. The study does not validate Driver’s Seat or establish durable development.",
        "which_workflow":  "Driver’s Seat—agency-withdrawal and outcome-boundary evidence; warning against equating assisted performance with learning.",
        "claimable_sentence":  "Withdrawing AI worsened several peer-feedback proxies, while adding a checklist did not outperform continued AI; the study did not independently test learning or transfer."
    }
,
    {
        "id":  "DebeerEtAl2017IRTrees",
        "tier":  "Tier 2",
        "full_citation":  "Debeer, D., Janssen, R., \u0026 De Boeck, P. (2017). Modeling skipped and not-reached items using IRTrees. Journal of Educational Measurement, 54(3), 333–363.",
        "link":  "https://doi.org/10.1111/jedm.12147",
        "peer_reviewed":  true,
        "population":  "Simulated item-response data and an empirical illustration using the 2009 PISA reading assessment.",
        "design":  "Tree-based item-response framework jointly modeling correct responses, skipped items, and not-reached items, with person and item contributions to omission processes.",
        "finding_verbatim":  "When the occurrence of these omissions is related to the proficiency process the missingness is nonignorable.",
        "effect_size":  "No single effect size; the paper reports simulation behavior and a PISA application rather than a treatment contrast.",
        "conditions_and_limits":  "The framework distinguishes intermittent skips from terminal not-reached responses and assumes a specified IRTree structure for the response and omission processes.",
        "criticism":  "Test-item omission is not the same phenomenon as an unscoreable human–AI episode. The transfer is a warning about informative missingness, not evidence that this IRTree is the correct Driver\u0027s Seat model.",
        "which_workflow":  "WP-08a Phase A: informative abstention and reason-coded missingness.",
        "claimable_sentence":  "When omission is related to proficiency, treating missing responses as ignorable can bias interpretation; skipped and not-reached cases should be distinguished."
    },
    {
        "id":  "DelikouraPapadopoulosHui2026Agnoagentia",
        "tier":  "Tier 2",
        "full_citation":  "Delikoura, I., Papadopoulos, P. M., \u0026 Hui, P. (2026). Agnoagentia: The illusion of agency in AI-assisted learning. In Artificial Intelligence in Education: 27th International Conference, AIED 2026, Proceedings, Part III (Lecture Notes in Artificial Intelligence, Vol. 16583, pp. 1–9). Springer Nature Switzerland.",
        "link":  "https://doi.org/10.1007/978-3-032-29760-0_1",
        "peer_reviewed":  true,
        "population":  "52 university students randomly grouped into 26 dyads.",
        "design":  "Each dyad completed three collaborative-writing tasks in a fixed sequence: baseline, a structured pedagogical agent (Clair), and individual GPT-5 access; perceived agency was self-reported and enacted agency was dialogue-coded using Bandura\u0027s four properties.",
        "finding_verbatim":  "ChatGPT-assisted students reported high perceived agency, while demonstrating low enacted agency.",
        "effect_size":  "The authors report significantly reduced communication volume and increased offloading under ChatGPT, but the indexed abstract does not supply a standardized effect size.",
        "conditions_and_limits":  "The contrast is between perceived agency and coded enacted agency in dyadic writing; all dyads experienced the conditions in the same order.",
        "criticism":  "Fixed order confounds tool condition with task and time; small sample; no established learning-outcome criterion; Bandura-property coding is not the same as entrepreneurial judgment-right attribution.",
        "which_workflow":  "Driver\u0027s Seat construct validity and abstention—observed versus perceived agency; education boundary.",
        "claimable_sentence":  "Perceived agency can diverge from enacted behavior, strengthening the case for trace-based assessment while providing no evidence that Driver\u0027s Seat measures learning."
    },
    {
        "id":  "EssienEtAl2026",
        "tier":  "Tier 2",
        "full_citation":  "Essien, A., Zhou, X., Kremantzis, M., \u0026 Teng, D. (2026). The agency gap: Perceived human AI agency, reflection and generative AI learning across UK and China based higher education contexts. Studies in Higher Education, 1–22. Advance online publication.",
        "link":  "https://doi.org/10.1080/03075079.2026.2686986",
        "peer_reviewed":  true,
        "population":  "309 higher-education respondents: 145 UK-based and 164 China-based; data collected September–December 2025.",
        "design":  "Cross-sectional online self-report survey analyzed with PLS-SEM. Eight agency items cover reported initiative, self-direction, monitoring, control, responsibility, and final decision.",
        "finding_verbatim":  "The paper studies “perceived human AI agency,” rather than observed behavior or reciprocal trace-coded co-agency.",
        "effect_size":  "Reported agency-to-reflection path coefficients were approximately β=.721 in the UK sample and β=.554 in the China sample. Outcomes were self-reported reflection, critical thinking, and self-concept.",
        "conditions_and_limits":  "Cross-sectional associations with only partial measurement invariance; no behavioral traces, randomized exposure, objective learning performance, retention, or transfer criterion.",
        "criticism":  "Common-method self-report cannot identify behavioral decision locus or causal direction. Country differences should not be treated as tested cultural mechanisms.",
        "which_workflow":  "Driver’s Seat—convergent self-report comparator and evidence for separating perceived from enacted agency.",
        "claimable_sentence":  "Perceived AI-assisted agency was associated with self-reported reflection in two national samples, but the study did not observe decision locus or learning performance."
    },
    {
        "id":  "FossFossKlein2007",
        "tier":  "Tier 3",
        "full_citation":  "Foss, K., Foss, N. J., \u0026 Klein, P. G. (2007). Original and derived judgment: An entrepreneurial theory of economic organization. Organization Studies, 28(12), 1893–1912.",
        "link":  "https://doi.org/10.1177/0170840606076179",
        "peer_reviewed":  true,
        "population":  "No empirical participant sample.",
        "design":  "Entrepreneurial theory of economic organization distinguishing owner-held original judgment from decision rights delegated to subordinates as derived judgment.",
        "finding_verbatim":  "The article distinguishes “original and derived judgment” within economic organization.",
        "effect_size":  "Not applicable; no empirical effect is reported.",
        "conditions_and_limits":  "Original judgment is tied to ownership, uncertainty bearing, and residual control; the delegated actor in the paper is organizational, not generative AI.",
        "criticism":  "Extending derived judgment to AI is an analogy, not the authors’ claim. Visible chat activity cannot prove economic ownership or uncertainty bearing.",
        "which_workflow":  "Driver’s Seat—entrepreneurial-judgment foundation for selective delegation and commitment boundaries.",
        "claimable_sentence":  "Owners may delegate decision rights as derived judgment while retaining original judgment, but extending that distinction to AI requires a new argument and evidence."
    },
    {
        "id":  "Fox2026EntrepreneurialBenchmarks",
        "tier":  "Tier 2",
        "full_citation":  "Fox, J. D. (2026). Developing artificial intelligence benchmarks for entrepreneurial tasks. Small Enterprise Research, 1–18. Advance online publication.",
        "link":  "https://doi.org/10.1080/13215906.2026.2705499",
        "peer_reviewed":  true,
        "population":  "Published entrepreneurship literature used to identify tasks that could be assigned to artificial intelligence; no human participant sample.",
        "design":  "Systematic review and prospective entrepreneurship-as-design-science framework for process benchmarking of AI on entrepreneurial tasks.",
        "finding_verbatim":  "The review catalogues 131 entrepreneurial tasks as candidates for AI benchmarking.",
        "effect_size":  "Not applicable; 131 is a task count from the review, not an effect size.",
        "conditions_and_limits":  "The contribution concerns defining task benchmarks and evaluating AI capability on entrepreneurial work, not attributing governance between a human and AI within an interaction.",
        "criticism":  "Task assignability and benchmarking do not establish that AI should govern a task, that a human delegated it, or that either party exercised entrepreneurial judgment well. The paper supplies no Driver\u0027s Seat-like measure or human outcome.",
        "which_workflow":  "WP-08a Phase A: post-cutoff entrepreneurship-task landscape and novelty boundary.",
        "claimable_sentence":  "Entrepreneurship-specific AI task benchmarks now exist, narrowing task-taxonomy novelty but not duplicating episode-level measurement of who governs."
    },
    {
        "id":  "GluszakGluszak2026DAGM",
        "tier":  "Tier 3",
        "full_citation":  "Gluszak, L., \u0026 Gluszak, F. (2026). Delegated agentic governance: A delegation-centred framework for managing autonomous AI in organisations. Journal of Information \u0026 Knowledge Management, Article 2650048.",
        "link":  "https://doi.org/10.1142/S0219649226500486",
        "peer_reviewed":  true,
        "population":  "A bibliometric corpus of 795 peer-reviewed publications from 2020–2026; no human participant sample.",
        "design":  "Multi-corpus bibliometric analysis plus conceptual derivation of a three-tier Delegated Agentic Governance Model covering advisory, operational, and autonomous delegation.",
        "finding_verbatim":  "Delegation, not architecture, is the primary variable that governance frameworks for agentic AI must address.",
        "effect_size":  "Not applicable; 795 is a literature-corpus count, not an effect size.",
        "conditions_and_limits":  "The framework assigns governance requirements by delegated-autonomy tier and proposes design principles and untested propositions for enterprise settings.",
        "criticism":  "The governance-readiness instrument is conceptual rather than psychometrically validated. The paper does not observe or score enacted judgment rights, episode-level authority, or human outcomes.",
        "which_workflow":  "WP-08a Phase A: current prior-art comparison and delegation theory.",
        "claimable_sentence":  "Delegation-calibrated organizational governance frameworks now exist, but they do not duplicate a validated episode-level behavioral attribution measure."
    },
    {
        "id":  "GordetzkiEtAl2026",
        "tier":  "Tier 1",
        "full_citation":  "Gordetzki, P., Blohm, I., Clegg, M., Schakols, F., \u0026 Hofstetter, R. (2026). Agency configurations in generative AI ideation: How textual and visual idea concretizations shape idea creativity and ideator effort. Information Systems Research. Advance online publication.",
        "link":  "https://doi.org/10.1287/isre.2024.0952",
        "peer_reviewed":  true,
        "population":  "276 US Prolific participants producing 428 idea refinements; novice participants completed one ideation task.",
        "design":  "Online experiment comparing textual versus visual AI concretization with imagine/control conditions and testing idea-maturity moderation.",
        "finding_verbatim":  "The authors describe “agency configurations” in which representation changes creativity and ideator effort.",
        "effect_size":  "Textual relative to visual support reportedly increased creativity by 18% and required 30% more effort. Effects were most pronounced for mature ideas; visual support improved very immature ideas without added effort.",
        "conditions_and_limits":  "One short ideation task with novices and no longitudinal learning, expertise, consequential commitment, or transfer criterion.",
        "criticism":  "Agency configuration is inferred from treatments and effort rather than directly coded actor-by-right allocation. More effort is not itself more governance.",
        "which_workflow":  "Driver’s Seat—configuration-contingency evidence, principally relevant to option formation.",
        "claimable_sentence":  "AI representation changed creativity and effort under specific idea-maturity conditions, but the experiment did not measure who governed framing, evaluation, or commitment."
    }
,
    {
        "id":  "GuTopol2026DecisionAuthority",
        "tier":  "Tier 3",
        "full_citation":  "Gu, Y., \u0026 Topol, E. J. (2026). Decision authority in health AI. Nature Health. Advance online publication.",
        "link":  "https://doi.org/10.1038/s44360-026-00185-z",
        "peer_reviewed":  false,
        "population":  "Health-AI systems used inside and outside clinical settings; no participant sample.",
        "design":  "Policy and governance commentary organized around degrees of AI influence over individual health trajectories.",
        "finding_verbatim":  "They need to be regulated and evaluated with respect to the degree to which they can influence and shape individual health trajectories.",
        "effect_size":  "Not applicable; no empirical study.",
        "conditions_and_limits":  "The argument addresses consequential healthcare authority and regulation, including use beyond formal clinical care.",
        "criticism":  "The Comment provides no operationalization, reliability evidence, behavioral trace analysis, or entrepreneurship application. One author is affiliated with ByteDance, which should be disclosed when assessing perspective.",
        "which_workflow":  "WP-08a Phase A: post-cutoff decision-authority landscape and novelty boundary.",
        "claimable_sentence":  "A post-cutoff health-AI commentary independently centers degree of decision authority, so the general question of who governs cannot be claimed as novel."
    },
    {
        "id":  "HendrickxEtAl2024RejectOption",
        "tier":  "Tier 2",
        "full_citation":  "Hendrickx, K., Perini, L., Van der Plas, D., Meert, W., \u0026 Davis, J. (2024). Machine learning with a reject option: A survey. Machine Learning, 113(5), 3073–3110.",
        "link":  "https://doi.org/10.1007/s10994-024-06534-x",
        "peer_reviewed":  true,
        "population":  "Published machine-learning methods for prediction with a reject option across application domains.",
        "design":  "Field survey and taxonomy covering reasons for rejection, model architectures, learning methods, and evaluation of predictive and rejective quality.",
        "finding_verbatim":  "We introduce the conditions leading to two types of rejection, ambiguity and novelty rejection, which we carefully formalize.",
        "effect_size":  "Not applicable; the survey does not pool an effect size.",
        "conditions_and_limits":  "Predictive and rejective performance must be considered together; ambiguity rejection and novelty rejection address different failure conditions.",
        "criticism":  "The review concerns machine-learning prediction, not formative educational measurement or conversational observability. Its taxonomies do not establish a valid abstention threshold for Driver\u0027s Seat.",
        "which_workflow":  "WP-08a Phase A: abstention taxonomy and evaluation requirements.",
        "claimable_sentence":  "Modern reject-option research distinguishes ambiguity from novelty rejection and evaluates prediction quality jointly with rejection behavior."
    },
    {
        "id":  "IssaPetaniGlavas2026AgenticLoafing",
        "tier":  "Tier 2",
        "full_citation":  "Issa, H., Petani, F. J., \u0026 Glavas, D. (2026). Agentic loafing: An AI decision delegation risk. Risk Analysis, 46(8), e70306.",
        "link":  "https://doi.org/10.1111/risa.70306",
        "peer_reviewed":  true,
        "population":  "Professional podcasts and gray literature concerning clinicians\u0027 delegation to AI; no conventional participant sample is reported in the abstract.",
        "design":  "Netnography of professional podcasts plus secondary analysis of gray literature; conceptual development of three institutional drivers and a preliminary risk diagnostic.",
        "finding_verbatim":  "the systematic disappearance of accountability when AI-assisted decisions fail, leaving no single party responsible",
        "effect_size":  "No quantitative effect size; qualitative/conceptual risk analysis.",
        "conditions_and_limits":  "The proposed drivers are performance-culture conformity, structural isolation of responsibility, and legitimization through quantification in healthcare decision delegation.",
        "criticism":  "Podcast and gray-literature material, no representative sampling, no behavioral validation, and a preliminary diagnostic that has not undergone psychometric or criterion validation.",
        "which_workflow":  "Driver\u0027s Seat risk interpretation—responsibility erosion and delegation without monitoring.",
        "claimable_sentence":  "Agentic loafing identifies responsibility erosion as a delegation risk, but its preliminary organizational diagnostic does not duplicate an episode-level behavioral measure."
    },
    {
        "id":  "IssakRezwanaHarteveld2025MOSAAIC",
        "tier":  "Tier 2",
        "full_citation":  "Issak, A., Rezwana, J., \u0026 Harteveld, C. (2025). MOSAAIC: Managing optimization towards shared autonomy, authority, and initiative in co-creation. In Proceedings of the Sixteenth International Conference on Computational Creativity (pp. 97–107). Association for Computational Creativity.",
        "link":  "https://computationalcreativity.net/iccc25/wp-content/uploads/papers/iccc25-issak2025mosaaic.pdf",
        "peer_reviewed":  true,
        "population":  "A systematic review of 172 full-length publications from ACM and Association for Computational Creativity venues, followed by six published co-creative systems used as cases.",
        "design":  "Systematic literature review and framework synthesis; two authors independently coded six co-creative systems and resolved discrepancies through discussion.",
        "finding_verbatim":  "We define control as the power to determine, initiate, and direct the process of co-creation.",
        "effect_size":  "No quantitative effect size; this is a review-derived framework and six-case demonstration.",
        "conditions_and_limits":  "The framework concerns human–AI co-creativity and allocates autonomy, initiative, and authority as full-human, shared, or full-AI; the authors state that an appropriate balance depends on the task.",
        "criticism":  "It does not validate a behavioral score, use consequential entrepreneurial episodes, or report inter-rater reliability for the six-case coding. Its dimensions are broader than five specific judgment rights.",
        "which_workflow":  "WP-08A Phase A: allocation of authority, initiative, and autonomy.",
        "claimable_sentence":  "MOSAAIC frames human–AI control as separable allocations of autonomy, initiative, and authority; the general control-allocation question is established prior art."
    },
    {
        "id":  "IssakRezwanaHarteveld2026ControlTrajectory",
        "tier":  "Tier 2",
        "full_citation":  "Issak, A., Rezwana, J., \u0026 Harteveld, C. (2026). “Control is a trajectory, not a point”: Conceptualizing control in human-AI co-creativity. In Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems (pp. 1–17). ACM.",
        "link":  "https://doi.org/10.1145/3772318.3790861",
        "peer_reviewed":  true,
        "population":  "Nine experts in HCI, co-creativity, and AI.",
        "design":  "Semi-structured expert focus group using MOSAAIC as a theoretical probe; qualitative thematic analysis.",
        "finding_verbatim":  "control is widely viewed as a dynamic, context-dependent construct that should adapt across different phases of co-creation, domains, and levels of trust in AI",
        "effect_size":  "No quantitative effect size; qualitative expert study.",
        "conditions_and_limits":  "Findings concern expert conceptions and preferences for co-creative systems, not observed consequential decisions by end users.",
        "criticism":  "One focus group with nine experts cannot validate a trajectory measure or establish how control actually shifts in use. The MOSAAIC probe may also have shaped the categories discussed.",
        "which_workflow":  "Driver\u0027s Seat dynamics—within-episode movement and control trajectories.",
        "claimable_sentence":  "Dynamic, phase-sensitive control is established conceptual territory; a particular operationalization must be evaluated on its measurement evidence."
    },
    {
        "id":  "JiangEtAl2026",
        "tier":  "Tier 2",
        "full_citation":  "Jiang, Y., Wu, Q., Yang, Y., Jian, C., \u0026 Zhao, J. (2026). Learner agency in revising GenAI-generated statements of purpose. British Journal of Educational Technology, 57(4), 965–983.",
        "link":  "https://doi.org/10.1111/bjet.70041",
        "peer_reviewed":  true,
        "population":  "121 English-major students at one southern-China university; nine completed retrospective interviews.",
        "design":  "ChatGPT-4 generated a statement-of-purpose draft from each CV and a standardized prompt; students evaluated and edited it. The study combined text comparison, nonparametric ratings, and thematic interviews.",
        "finding_verbatim":  "The authors identify “compliance-oriented acceptance, form-oriented modification, and content-oriented innovation.”",
        "effect_size":  "The paper reports rating differences among the three patterns; no standardized causal effect from a randomized intervention applies and none is entered here.",
        "conditions_and_limits":  "One revision task, one institution, fixed AI draft, nine interviews, and no conversational log, independent quality criterion, retention, or transfer measure.",
        "criticism":  "The patterns describe revision behavior and voice, not a complete actor-by-right allocation. Suggested instructional enhancements are not experimentally tested effects.",
        "which_workflow":  "Driver’s Seat—evaluation, modification, ownership, and voice comparator.",
        "claimable_sentence":  "Students displayed compliance, form-level modification, and content-level innovation when revising AI drafts, but the study does not establish learning or full decision-right governance."
    },
    {
        "id":  "Kane2013Validity",
        "tier":  "Tier 1",
        "full_citation":  "Kane, M. T. (2013). Validating the interpretations and uses of test scores. Journal of Educational Measurement, 50(1), 1–73.",
        "link":  "https://doi.org/10.1111/jedm.12000",
        "peer_reviewed":  true,
        "population":  "Proposed interpretations and uses of assessment scores across testing contexts; no participant sample.",
        "design":  "Argument-based validity framework specifying inferences, assumptions, evidence, and consequences for score interpretations and uses.",
        "finding_verbatim":  "It is the proposed score interpretations and uses that are validated and not the test or the test scores.",
        "effect_size":  "Not applicable; this is a validity framework rather than an empirical treatment study.",
        "conditions_and_limits":  "The evidence burden depends on the ambition of the proposed interpretation or use; consequences matter for use claims, and validating an interpretation does not automatically validate a use.",
        "criticism":  "Kane supplies a framework, not a reliability cutoff, abstention rule, or empirical validation result. Invoking the framework cannot substitute for testing Driver\u0027s Seat\u0027s scoring, generalization, extrapolation, and use assumptions.",
        "which_workflow":  "WP-08a Phase A and Phase B: claim ladder and interpretation/use argument.",
        "claimable_sentence":  "Driver\u0027s Seat validation must target each proposed interpretation and use; a defensible conditional trace description would not by itself validate ranking, development, or learning claims."
    }
,
    {
        "id":  "KimSoPark2026",
        "tier":  "Tier 1",
        "full_citation":  "Kim, S., So, H.-J., \u0026 Park, K. (2026). Supporting learner agency in collaborative writing with generative AI. British Journal of Educational Technology, 57(4), 984–1008.",
        "link":  "https://doi.org/10.1111/bjet.70015",
        "peer_reviewed":  true,
        "population":  "52 Korean students randomized to Similarity Viewer only (n=26) or Similarity Viewer plus Argument Outline (n=26); eight participants were interviewed.",
        "design":  "Randomized two-condition collaborative-writing study with two 30-minute tasks, trace coding, text-similarity analysis, epistemic network analysis, and interviews.",
        "finding_verbatim":  "The authors warn that optimal learner agency is not simply “higher human input.”",
        "effect_size":  "No significant overall behavior-frequency differences. In session 2, the outline group had lower similarity, F=5.48, p=.02, η²=.11, with final analytic groups of 25 and 23.",
        "conditions_and_limits":  "Short collaborative-writing tasks; similarity is not direct agency or learning evidence. Codes include Compose, Revise, Seek Suggestion, Dismiss, Accept, and low/high modification.",
        "criticism":  "The study does not allocate five decision rights, test consequential commitment, or demonstrate learning and transfer. Modification amount cannot stand alone as governance.",
        "which_workflow":  "Driver’s Seat—behavioral accept/reject/modify comparator and warning against activity-volume scoring.",
        "claimable_sentence":  "A small randomized writing study found a session-specific similarity difference but no overall behavior-frequency difference, and explicitly rejected equating more human input with better agency."
    },
    {
        "id":  "KrushinskaiaElenRaes2026",
        "tier":  "Tier 1",
        "full_citation":  "Krushinskaia, K., Elen, J., \u0026 Raes, A. (2026). Pre-service teachers’ agency during their interactions with generative AI while designing for learning—A process view on Intelligent-TPACK. Computers and Education Open, 10, 100325.",
        "link":  "https://doi.org/10.1016/j.caeo.2025.100325",
        "peer_reviewed":  true,
        "population":  "78 pre-service teachers randomized to a custom expert bot or basic ChatGPT.",
        "design":  "Randomized process study using interaction length, proportion of self-generated prompts/words, and coded collaborative problem-solving behavior as agency indicators.",
        "finding_verbatim":  "The study takes “a process view” of agency during generative-AI-supported learning design.",
        "effect_size":  "Omnibus MANOVA: Pillai’s trace=.262, F(4,50)=4.43, p=.004. Expert-bot users produced more prompts, F(1,41)=8.20, corrected p=.007, η²=.126; deeper coding indicated mixed or reduced substantive agency.",
        "conditions_and_limits":  "Pre-service-teacher design task, two tool configurations, and multiple proxy indicators that did not converge on a single agency interpretation.",
        "criticism":  "Length and self-generated prompting can suggest increased agency while collaboration coding suggests passivity or delegation. No learning, retention, transfer, or five-right reference coding was tested.",
        "which_workflow":  "Driver’s Seat—strong adverse evidence on proxy disagreement and validation design.",
        "claimable_sentence":  "In one randomized study, an expert bot increased prompt production while deeper process coding suggested mixed or reduced agency, showing that surface activity is not a valid stand-alone proxy."
    },
    {
        "id":  "Lee2026ReadinessMetrics",
        "tier":  "Tier 2",
        "full_citation":  "Lee, M. H. (2026). From accuracy to readiness: Metrics and benchmarks for human-AI decision-making: An initial exploration. In Extended Abstracts of the 2026 CHI Conference on Human Factors in Computing Systems (pp. 1–10). ACM.",
        "link":  "https://doi.org/10.1145/3772363.3798377",
        "peer_reviewed":  true,
        "population":  "No participant sample; conceptual metrics paper.",
        "design":  "Conceptual synthesis proposing four metric families—outcomes, reliance and interaction, safety and harm, and learning and readiness—mapped to an Understand–Control–Improve lifecycle.",
        "finding_verbatim":  "This paper proposes a measurement framework for evaluating human-AI decision-making centered on team readiness.",
        "effect_size":  "Not applicable; no empirical test.",
        "conditions_and_limits":  "Metrics such as accept-on-wrong, changed-to-wrong, override timing, rollback, escalation, and transfer presuppose suitable interaction logs and, for some metrics, ground truth.",
        "criticism":  "The framework is unvalidated and does not attribute entrepreneurial judgment rights. It nevertheless establishes trace-based governance-in-use and longitudinal readiness as prior measurement ideas.",
        "which_workflow":  "Driver\u0027s Seat measurement design—trace-based behavior, abstention prerequisites, and criterion validation.",
        "claimable_sentence":  "Trace-based evaluation of reliance and governance is already proposed; Driver\u0027s Seat\u0027s narrower contribution must lie in its particular judgment-right rubric, abstention rule, and validation program."
    },
    {
        "id":  "Leonardi2025",
        "tier":  "Tier 3",
        "full_citation":  "Leonardi, P. M. (2025). Homo agenticus in the age of agentic AI: Agency loops, power displacement, and the circulation of responsibility. Information and Organization, 35(3), 100582.",
        "link":  "https://doi.org/10.1016/j.infoandorg.2025.100582",
        "peer_reviewed":  true,
        "population":  "No new participant sample; the article synthesizes and illustrates three prior empirical studies.",
        "design":  "Conceptual five-phase agency loop: delegation, attribution, contingency, reassertion, and reconfiguration.",
        "finding_verbatim":  "The framework follows “delegation, attribution, contingency, reassertion, and reconfiguration.”",
        "effect_size":  "Not applicable; no new empirical effect is estimated.",
        "conditions_and_limits":  "Organizational theory of attributed agency and responsibility, not a trace-scoring or psychometric study.",
        "criticism":  "Perceived or attributed agency can shift while formal authority remains unchanged. Driver’s Seat must not conflate subjective attribution, formal authority, and visible enacted governance.",
        "which_workflow":  "Driver’s Seat—temporal-transition and responsibility-circulation precursor.",
        "claimable_sentence":  "Delegation and reassertion are established temporal agency ideas; a transition-coding proposal requires separate validation."
    },
    {
        "id":  "Li2026AIEE",
        "tier":  "Tier 3",
        "full_citation":  "Li, Y. (2026). The associations of AI-integrated entrepreneurship education versus traditional entrepreneurship education on undergraduates\u0027 entrepreneurial intention and its antecedents. Humanities and Social Sciences Communications. Advance online publication.",
        "link":  "https://doi.org/10.1057/s41599-026-08504-1",
        "peer_reviewed":  true,
        "population":  "133 undergraduates in two intact classes at one university in Suzhou, China: AI-integrated entrepreneurship education n = 68 and traditional entrepreneurship education n = 65.",
        "design":  "One-semester, non-individually-randomized, post-test comparison analyzed with multigroup PLS-SEM using self-reported entrepreneurial intention and antecedents.",
        "finding_verbatim":  "Paths from ATE and PBC to EI, alongside the group-specific indirect effects via PD and PF, are significantly stronger in the AIEE group.",
        "effect_size":  "ATE→EI: .310 versus .200, difference .110, one-tailed p = .023; PBC→EI: .350 versus .190, difference .160, p = .012. Indirect ATE→PD→EI: .247 versus .037, difference .210, p = .009; PBC→PF→EI: .231 versus .036, difference .195, p = .002. EI R²: .650 versus .318, difference .332, p \u003c .001.",
        "conditions_and_limits":  "The comparison bundles AI tools with the full instructional package. The authors explicitly interpret results as correlational because assignment was nonrandom, data were cross-sectional, the site was singular, confounders were unmeasured, and pure AI effects could not be isolated.",
        "criticism":  "Outcomes are self-reported intentions and perceived antecedents, not objective learning, transfer, durable development, or Driver\u0027s Seat. One-tailed path comparisons, small intact classes, common method, and PLS prediction can make differences look stronger than warranted.",
        "which_workflow":  "WP-08a Phase A: post-cutoff entrepreneurship-education evidence and learning-claim boundary.",
        "claimable_sentence":  "One small nonrandom class comparison associated an AI-integrated curriculum with stronger self-reported intention pathways, but it does not establish learning or any relation between Driver\u0027s Seat and outcomes."
    },
    {
        "id":  "MadjdiWurth2026",
        "tier":  "Tier 3",
        "full_citation":  "Madjdi, F., \u0026 Wurth, B. (2026). AI-mediated plausibility regimes: Entrepreneurial judgment, epistemic risk, and the distribution of entrepreneurial futures. Journal of Business Venturing Insights, 26, e00644.",
        "link":  "https://doi.org/10.1016/j.jbvi.2026.e00644",
        "peer_reviewed":  true,
        "population":  "No empirical participants or dataset.",
        "design":  "Conceptual theory of generative plausibility inflation and screening-based plausibility suppression under Knightian uncertainty.",
        "finding_verbatim":  "The authors argue that “AI systems are not epistemically neutral.”",
        "effect_size":  "Not applicable; no empirical effect is estimated.",
        "conditions_and_limits":  "Propositions depend on AI system class, inferential logic, judgment intensity, institutionalization, and the authority granted to AI-generated plausibility signals.",
        "criticism":  "The article theorizes possible false-positive and false-negative mechanisms; it does not demonstrate that AI caused either effect in entrepreneurs or ecosystems.",
        "which_workflow":  "Driver’s Seat—entrepreneurial context, grounding, evaluative governance, and epistemic-risk theory.",
        "claimable_sentence":  "Madjdi and Wurth theorize that generative systems may inflate confidence and screening systems may suppress low-precedent futures under specified conditions."
    },
    {
        "id":  "MargaridoEtAl2024",
        "tier":  "Tier 3",
        "full_citation":  "Margarido, S., Roque, L., Machado, P., \u0026 Martins, P. (2024). MI-CCy Quantifier: A framework for quantifying mixed-initiative co-creativity in human-AI collaborations. In M. F. Santos, J. Machado, P. Novais, P. Cortez, \u0026 P. M. Moreira (Eds.), Progress in artificial intelligence: 23rd EPIA Conference on Artificial Intelligence, EPIA 2024, proceedings, Part I (pp. 3–15). Springer.",
        "link":  "https://doi.org/10.1007/978-3-031-73497-7_1",
        "peer_reviewed":  true,
        "population":  "No human participant sample. The paper demonstrates the framework through a subjective analysis of one co-creative system, 1001 Nights.",
        "design":  "Conceptual/descriptive framework with human-to-computer spectra for Initial Setting, Initiative, Evaluation, and Final Decision, plus ordinal visual criteria for Task Assignment, Intervention Pace, and Explainability.",
        "finding_verbatim":  "The authors call it “a directive framework for analyzing works based on their level of MI-CCy in a descriptive and visually gradable manner.”",
        "effect_size":  "Not applicable. The paper reports no numerical score, participant effect, reliability coefficient, or criterion-validity estimate.",
        "conditions_and_limits":  "The analysis is system-level and subjectively positioned on visual spectra. Final Decision chiefly concerns when to end a creative process, not consequential entrepreneurial commitment.",
        "criticism":  "The framework lacks behavioral anchors, reproducible scoring rules, participants, reliability, and validation. The authors explicitly warn that a higher MI-CCy level is not necessarily better.",
        "which_workflow":  "WP-08A Phase A: allocation-framework comparator for problem framing and evaluative governance.",
        "claimable_sentence":  "MI-CCy already allocates human and computer influence over initial setting, initiative, evaluation, and process closure, but it is not a validated behavioral measure."
    }
,
    {
        "id":  "MishraHenriksen2026AgentAgency",
        "tier":  "Tier 3",
        "full_citation":  "Mishra, P., \u0026 Henriksen, D. (2026). Agentic AI in education: Whose agent? Whose agency? TechTrends. Advance online publication.",
        "link":  "https://doi.org/10.1007/s11528-026-01213-1",
        "peer_reviewed":  true,
        "population":  "Educational discourse and historical concepts of agents, agency, learner voice, and teacher voice; no participant sample.",
        "design":  "Conceptual and historical analysis of the language of agentic AI in education.",
        "finding_verbatim":  "Agent is a role-word that extends easily to software, while agency is a rights-word.",
        "effect_size":  "Not applicable; no empirical study.",
        "conditions_and_limits":  "The argument concerns moral and educational meanings of agency and the institutional principals served by educational AI platforms.",
        "criticism":  "The paper offers a normative linguistic distinction, not an operational measure, behavioral coding study, or test of learning and governance outcomes.",
        "which_workflow":  "WP-08a Phase A: post-cutoff terminology and novelty boundary.",
        "claimable_sentence":  "Calling software an agent does not establish that it possesses, preserves, or redistributes human agency; role and rights should remain analytically distinct."
    },
    {
        "id":  "MurrayRhymerSirmon2021",
        "tier":  "Tier 3",
        "full_citation":  "Murray, A., Rhymer, J., \u0026 Sirmon, D. G. (2021). Humans and technology: Forms of conjoined agency in organizations. Academy of Management Review, 46(3), 552–571.",
        "link":  "https://doi.org/10.5465/amr.2019.0186",
        "peer_reviewed":  true,
        "population":  "No empirical participant sample.",
        "design":  "Conceptual 2×2 framework allocating intentionality over protocol development and action selection to a human or technology, yielding assisting, arresting, augmenting, and automating forms.",
        "finding_verbatim":  "The paper theorizes “forms of conjoined agency” from protocol development and action selection.",
        "effect_size":  "Not applicable; no empirical effect is estimated.",
        "conditions_and_limits":  "General organizational technology theory, not generative-AI dialogue, entrepreneurship, or a validated trace measure.",
        "criticism":  "It is a direct conceptual precursor to Randazzo’s how/what allocation. Any claim that actor-by-decision-dimension allocation is new is untenable.",
        "which_workflow":  "WP-08A Phase A: foundational precursor for actor-by-decision-dimension allocation.",
        "claimable_sentence":  "Murray et al. already allocate protocol development and action selection between humans and technology, making the broad how/what authority question prior art."
    },
    {
        "id":  "PackardBylund2025",
        "tier":  "Tier 3",
        "full_citation":  "Packard, M. D., \u0026 Bylund, P. L. (2025). Towards an entrepreneurial judgement theory: Building the cognitive microfoundations of entrepreneurial judgement. International Small Business Journal: Researching Entrepreneurship, 43(1), 53–75.",
        "link":  "https://doi.org/10.1177/02662426241269772",
        "peer_reviewed":  true,
        "population":  "No empirical participant sample.",
        "design":  "Conceptual entrepreneurial-judgment theory using nested distal and proximal intentions and a dynamic intentionality scaffold.",
        "finding_verbatim":  "The authors define judgment as “the determination and instigation of intentions.”",
        "effect_size":  "Not applicable; no empirical effect is reported.",
        "conditions_and_limits":  "A cognitive and intentional theory. Text can contain evidence consistent with commitment but cannot reveal private intention or establish later action.",
        "criticism":  "A visible commitment utterance is not identical to the formation, instigation, or persistence of an entrepreneurial intention.",
        "which_workflow":  "Driver’s Seat—commitment-right and intention boundary theory.",
        "claimable_sentence":  "Entrepreneurial judgment theory links judgment to determining and instigating nested intentions; dialogue evidence can only proxy, not prove, that private process."
    },
    {
        "id":  "RafnerEtAl2025CreativeAgency",
        "tier":  "Tier 2",
        "full_citation":  "Rafner, J., Zana, B., Hansen, I. B., Ceh, S., Sherson, J., Benedek, M., \u0026 Lebuda, I. (2025). Agency in human-AI collaboration for image generation and creative writing: Preliminary insights from think-aloud protocols. Creativity Research Journal, advance online publication, 1–24.",
        "link":  "https://doi.org/10.1080/10400419.2025.2587803",
        "peer_reviewed":  true,
        "population":  "Study 1: six participants completing AI-assisted image generation; Study 2: seven participants completing AI-assisted creative writing.",
        "design":  "Two exploratory qualitative studies using think-aloud protocols, post-task semi-structured interviews, and within-subject comparisons across tools.",
        "finding_verbatim":  "agency in human–AI co-creation fluctuates across the creative process",
        "effect_size":  "No quantitative effect size; exploratory qualitative evidence.",
        "conditions_and_limits":  "Agency was organized around creative self-efficacy, control over creative action, autonomy in process, and ownership of product in image-generation and writing tasks.",
        "criticism":  "Extremely small samples and exploratory coding; no consequential decision, validated coefficient, independent outcome, or evidence of durable development.",
        "which_workflow":  "Driver\u0027s Seat dynamics—fluctuation and reassertion during an episode.",
        "claimable_sentence":  "Small think-aloud studies show that experienced creative agency can fluctuate during AI use, supporting a dynamic premise but not validating Driver\u0027s Seat."
    },
    {
        "id":  "RandazzoEtAl2025",
        "tier":  "Tier 3",
        "full_citation":  "Randazzo, S., Lifshitz, H., Kellogg, K. C., Dell’Acqua, F., Mollick, E., Candelon, F., \u0026 Lakhani, K. R. (2025). Cyborgs, centaurs and self-automators: The three modes of human–GenAI knowledge work and their implications for skilling and the future of expertise (Harvard Business School Working Paper No. 26-036). Harvard Business School.",
        "link":  "https://doi.org/10.2139/ssrn.4921696",
        "peer_reviewed":  false,
        "population":  "244 global junior management consultants completing a strategic investment task; the mode counts reported in the paper total 243. The authors describe approximately 4,975 logged interactions/prompts and 237 follow-up interviews.",
        "design":  "Field study of human–GPT-4 work across seven subtasks. Session logs, final products, and interviews were used to derive Directed/Centaur, Fused/Cyborg, and Abdicated/Self-Automator modes and examine sequences across the workflow.",
        "finding_verbatim":  "The paper asks: “Who selects what needs to be done?” and “Who identifies how it gets done?”",
        "effect_size":  "No standardized effect size for actor allocation, skilling, or transfer was reported. Descriptive mode counts were 146 Cyborgs, 34 Centaurs, and 63 Self-Automators, totaling 243 rather than the stated N=244.",
        "conditions_and_limits":  "One consulting task, one organizational partner, junior consultants, and GPT-4. The modes are session-level patterns, not right-level scores; sequence analysis is present, but not a formal decision-right transition measure.",
        "criticism":  "No public coder-reliability estimate, independent learning or transfer criterion, or journal peer review was located. Interview claims about upskilling/newskilling should not be treated as demonstrated durable development.",
        "which_workflow":  "WP-08A Phase A: closest full-workflow trace and what/how authority precursor.",
        "claimable_sentence":  "A 2025 working paper uses the driver’s-seat metaphor and classifies who directs what and how across a logged consulting workflow; those broad ideas are established prior art."
    },
    {
        "id":  "RappOlbrich2023",
        "tier":  "Tier 3",
        "full_citation":  "Rapp, D. J., \u0026 Olbrich, M. (2023). From Knightian uncertainty to real-structuredness: Further opening the judgment black box. Strategic Entrepreneurship Journal, 17(1), 186–209.",
        "link":  "https://doi.org/10.1002/sej.1443",
        "peer_reviewed":  true,
        "population":  "No empirical participant sample.",
        "design":  "Conceptual dimensionalization of entrepreneurial judgment through goal, causality, appraisal, and solution judgments under real-structured decision problems.",
        "finding_verbatim":  "The article’s “four-part dimensionalization” covers effects, appraisal of alternatives, goals, and resolution of the decision problem.",
        "effect_size":  "Not applicable; no empirical effect is estimated.",
        "conditions_and_limits":  "The categories concern entrepreneurial decision problems and selective communicability/delegation; they are not dialogue codes or a validated scale.",
        "criticism":  "Mapping goal, causality, appraisal, and solution judgments to PF/OH/EG is interpretive. Consequential commitment is not fully represented, and external input may be advice rather than derived judgment.",
        "which_workflow":  "Driver’s Seat—principal content-domain precursor for differentiated entrepreneurial judgment rights.",
        "claimable_sentence":  "Goal, causality, appraisal, and solution judgments are established entrepreneurship constructs that can inform, but do not validate, a trace operationalization."
    },
    {
        "id":  "RetamalSaavedraEtAl2026",
        "tier":  "Tier 1",
        "full_citation":  "Retamal-Saavedra, C. D., Andrade-Valbuena, N. A., Contreras Navarro, J. E., Inostroza Caceres, F., \u0026 Vidal-Rebolledo, I. (2026). Artificial intelligence in entrepreneurship: Mapping a fragmented field and advancing a cognitive research agenda. Journal of Management \u0026 Organization, 32(2), 473–500.",
        "link":  "https://doi.org/10.1017/jmo.2026.10082",
        "peer_reviewed":  true,
        "population":  "372 peer-reviewed Web of Science documents retained after screening 933 records. The publisher page inconsistently describes the covered period as both 2010–2025 and 2001–2025.",
        "design":  "Hybrid systematic review, bibliometric co-word analysis, thematic mapping, and qualitative synthesis across selected AI-and-entrepreneurship categories.",
        "finding_verbatim":  "The field is described as “fragmented, technology centered, and weakly connected to theories of entrepreneurial decision-making.”",
        "effect_size":  "Not applicable; the review maps 372 documents and does not estimate a pooled intervention effect.",
        "conditions_and_limits":  "Web of Science only, selected search categories and terms, bibliometric co-occurrence, and inconsistent stated search periods.",
        "criticism":  "The review supports fragmentation within its retrieved corpus but cannot establish field-wide absence of cognitive or behavioral judgment research.",
        "which_workflow":  "Driver’s Seat—field-positioning review and rationale for a cognitive/process research agenda.",
        "claimable_sentence":  "A 2026 hybrid review maps a fragmented, technology-centered AI–entrepreneurship literature in which cognitive and behavioral judgment work remains marginal within the retrieved corpus."
    }
,
    {
        "id":  "SaglamEtAl2026BiasAwareReject",
        "tier":  "Tier 2",
        "full_citation":  "Sağlam, F., Özgen, Ü., Uygun, A., Dinçer, O. S., \u0026 Albayrak, C. (2026). Selective classification under imbalance in multiclass settings: A novel metric for bias-aware risk–coverage evaluation. Journal of Biomedical Informatics, 181, 105084.",
        "link":  "https://doi.org/10.1016/j.jbi.2026.105084",
        "peer_reviewed":  true,
        "population":  "A clinical complete-blood-count dataset with 3,316 patient records, 11 laboratory features, and 9 diagnostic classes, plus three public benchmark datasets.",
        "design":  "Multidataset comparison of conventional and class-averaged risk–coverage metrics and a class-conditional coverage-matching selection strategy across five uncertainty measures.",
        "finding_verbatim":  "Commonly used evaluation metrics obscure fairness issues under class imbalance.",
        "effect_size":  "Rank-based comparisons favored the class-conditional strategy for CA-AUGRC (p = .009, r = .638) and AUIC (p \u003c .001, r = .957).",
        "conditions_and_limits":  "The reported advantage is for class-aware evaluation and rejection in four imbalanced multiclass datasets; results concern diagnostic/benchmark labels and specified uncertainty estimators.",
        "criticism":  "Clinical classes are not educational contexts or governance opportunities. The study shows how aggregate risk–coverage can hide selective rejection, but it does not establish fairness or validity for Driver\u0027s Seat.",
        "which_workflow":  "WP-08A Phase A: class- and context-conditional coverage and abstention comparison.",
        "claimable_sentence":  "Aggregate risk–coverage summaries can conceal disproportionate rejection under imbalance, so Driver\u0027s Seat coverage should be reported by context and relevant groups."
    },
    {
        "id":  "ShresthaBenMenahemVonKrogh2019",
        "tier":  "Tier 3",
        "full_citation":  "Shrestha, Y. R., Ben-Menahem, S. M., \u0026 von Krogh, G. (2019). Organizational decision-making structures in the age of artificial intelligence. California Management Review, 61(4), 66–83.",
        "link":  "https://doi.org/10.1177/0008125619862257",
        "peer_reviewed":  true,
        "population":  "No empirical participant sample.",
        "design":  "Conceptual comparison of human and AI decision making across search-space specificity, interpretability, alternative-set size, speed, and replicability, used to derive organizational decision structures.",
        "finding_verbatim":  "The framework includes “full human to AI delegation,” sequential human–AI hybrids, and “aggregated human–AI decision making.”",
        "effect_size":  "Not applicable; no empirical treatment effect or validation coefficient is reported.",
        "conditions_and_limits":  "Organization-level decision-structure theory focused on combining human and AI decisions. It does not operationalize conversational episodes, entrepreneurship-specific rights, trace provenance, or learning outcomes.",
        "criticism":  "The framework makes delegation direction and hybrid decision structures prior art, but it supplies no behavioral attribution instrument and does not show which configuration is superior across untested settings.",
        "which_workflow":  "Driver’s Seat—foundational full-delegation, sequential-hybrid, and aggregation precursor.",
        "claimable_sentence":  "Full delegation, human-to-AI and AI-to-human sequential hybrids, and aggregated human–AI decisions were already theorized in 2019."
    },
    {
        "id":  "SrinivasChetan2026DelegationAugmentation",
        "tier":  "Tier 2",
        "full_citation":  "Srinivas, R., \u0026 Chetan, S. S. (2026). Integrating artificial intelligence in strategic decision-making: Contexts for delegation and augmentation. Group Decision and Negotiation, 35(3), Article 59.",
        "link":  "https://doi.org/10.1007/s10726-026-10016-x",
        "peer_reviewed":  true,
        "population":  "33 C-suite and senior executives across three multinational organizations with customized LLM-based tools.",
        "design":  "Gioia-method qualitative study using semi-structured interviews, structured demonstrations of in-house tools, and secondary documents.",
        "finding_verbatim":  "a grounded framework distinguishing high-confidence AI delegation of analytical work from iterative human-AI co-deliberation",
        "effect_size":  "No quantitative effect size; qualitative study.",
        "conditions_and_limits":  "Delegation was described for structured, predictable analytical work; ill-structured and uncertain strategic decisions elicited human primacy and iterative co-deliberation.",
        "criticism":  "Self-reported executive accounts and staged demonstrations, not observation of live strategic decisions; three AI-mature multinationals; no performance-optimality test or behavioral coefficient.",
        "which_workflow":  "Driver\u0027s Seat contingency question—when delegation versus augmentation is appropriate.",
        "claimable_sentence":  "Executives described delegation as contingent on problem structuredness and situational predictability, supporting a task-fit interpretation rather than a universal preference for more human control."
    },
    {
        "id":  "Srivastava2026AIDelegationModel",
        "tier":  "Tier 2",
        "full_citation":  "Srivastava, A. (2026). EXPRESS: Governing AI-enabled decision making: Delegation, autonomy, and control at the operations–marketing interface. Production and Operations Management, advance online publication.",
        "link":  "https://doi.org/10.1177/10591478261473004",
        "peer_reviewed":  true,
        "population":  "No human sample; modeled firm jointly chooses pricing, inventory, and governance under demand uncertainty.",
        "design":  "Analytical operations model comparing human control, full AI autonomy, and human-in-the-loop governance, with extensions for endogenous investment, learning, lead times, oversight, and scale.",
        "finding_verbatim":  "partial delegation can strictly dominate both full autonomy and full human control",
        "effect_size":  "Not applicable; theoretical thresholds and equilibria rather than empirical effects.",
        "conditions_and_limits":  "Results follow from stated assumptions about demand variance, AI imperfection, responsiveness, buffers, risk asymmetry, and organizational structure.",
        "criticism":  "No observed humans, interactions, or measurement instrument; optimality is model-contingent and cannot validate a trace score or learning claim.",
        "which_workflow":  "Driver\u0027s Seat theory boundary—economic contingency and hybrid delegation.",
        "claimable_sentence":  "Analytical work shows that partial delegation may be optimal under some assumptions, reinforcing that the normative target is appropriate allocation rather than maximum human activity."
    },
    {
        "id":  "TownsendHunt2019",
        "tier":  "Tier 3",
        "full_citation":  "Townsend, D. M., \u0026 Hunt, R. A. (2019). Entrepreneurial action, creativity, \u0026 judgment in the age of artificial intelligence. Journal of Business Venturing Insights, 11, e00126.",
        "link":  "https://doi.org/10.1016/j.jbvi.2019.e00126",
        "peer_reviewed":  true,
        "population":  "No empirical participant sample.",
        "design":  "Conceptual agenda connecting AI, entrepreneurial action, creativity, desirability, possibility, and judgment under modal uncertainty.",
        "finding_verbatim":  "The article centers “entrepreneurial action, creativity, \u0026 judgment in the age of artificial intelligence.”",
        "effect_size":  "Not applicable; no empirical effect is reported.",
        "conditions_and_limits":  "Early conceptual agenda rather than an agency measure, trace study, or validation test.",
        "criticism":  "Describe it as an early JBVI agenda, not the journal’s singular foundational agenda. It supplies no actor-allocation instrument.",
        "which_workflow":  "Driver’s Seat—journal and domain positioning for AI-enabled entrepreneurial judgment.",
        "claimable_sentence":  "JBVI had already placed AI, creativity, entrepreneurial action, and judgment under ambiguity on its agenda by 2019."
    },
    {
        "id":  "VaccaroAlmaatouqMalone2024",
        "tier":  "Tier 1",
        "full_citation":  "Vaccaro, M., Almaatouq, A., \u0026 Malone, T. W. (2024). When combinations of humans and AI are useful: A systematic review and meta-analysis. Nature Human Behaviour, 8(12), 2293–2303.",
        "link":  "https://doi.org/10.1038/s41562-024-02024-1",
        "peer_reviewed":  true,
        "population":  "74 peer-reviewed papers reporting 106 human-participant experiments and 370 effect sizes. Eligible experiments compared human-only, AI-only, and combined human–AI performance; searches covered 1 January 2020 through 30 June 2023.",
        "design":  "Preregistered interdisciplinary systematic review with a three-level meta-analytic model. Primary outcomes were synergy relative to the better solo performer and augmentation relative to the human alone; task type and relative solo performance were tested as moderators.",
        "finding_verbatim":  "On average, “human–AI combinations performed significantly worse than the best of humans or AI alone.”",
        "effect_size":  "Synergy versus the better solo performer: Hedges’ g=−.23, 95% CI [−.39, −.07], p=.005. Augmentation versus humans alone: g=.64, 95% CI [.53, .74]. Decision-task synergy: g=−.27, 95% CI [−.44, −.10]; creation-task synergy: g=.19, 95% CI [−.09, .48], p=.180. When humans outperformed AI alone, synergy g=.46, 95% CI [.28, .66]; when AI outperformed humans, synergy g=−.54, 95% CI [−.71, −.37]. Heterogeneity was I²=97.7% for synergy and 93.8% for augmentation.",
        "conditions_and_limits":  "The result applies to experiments reporting all three performance baselines and to the tasks, processes, and populations selected for study. About 85% of effect sizes involved finite-choice decision tasks and about 10% creation tasks. More than 95% of systems had humans make the final decision; only three experiments predetermined separate human/AI subtasks.",
        "criticism":  "Very high heterogeneity, possible topic-selection and publication bias, varied outcome metrics and measurement quality, and predominantly laboratory configurations limit generalization. A positive creation-task point estimate was not statistically different from zero. The meta-analysis measures performance, not governance or learning.",
        "which_workflow":  "Driver’s Seat—configuration-contingency proposition, independent performance criterion, and evidence against assuming that human–AI combination or greater human participation is inherently superior.",
        "claimable_sentence":  "Across 106 experiments, human–AI systems outperformed humans alone on average but underperformed the better solo performer; losses were concentrated in decision tasks and varied sharply with relative human-versus-AI capability."
    },
    {
        "id":  "WuYao2026AfterInterface",
        "tier":  "Tier 2",
        "full_citation":  "Wu, M., \u0026 Yao, M. (2026). After the interface: Relocating human agency in the age of conversational AI. In Proceedings of the 8th ACM Conference on Conversational User Interfaces (pp. 1–7). ACM.",
        "link":  "https://doi.org/10.1145/3816046.3816301",
        "peer_reviewed":  true,
        "population":  "No participant sample; conceptual CUI paper.",
        "design":  "Conceptual argument distinguishing process control from outcome control and mapping conversational, generative, and agentic systems across that space.",
        "finding_verbatim":  "Agency has not diminished but has relocated.",
        "effect_size":  "Not applicable; no empirical test.",
        "conditions_and_limits":  "The authors locate contemporary agency in goal articulation, output evaluation, and outcome negotiation and explicitly acknowledge that outcome-based agency may be illusory for unverifiable outputs.",
        "criticism":  "The central assertion is a conceptual provocation rather than an empirical finding; the paper offers no scoring rule, reliability, external validation, or consequential-task data.",
        "which_workflow":  "Driver\u0027s Seat construct boundary—process versus outcome control and goal/evaluation/negotiation rights.",
        "claimable_sentence":  "Goal articulation, evaluation, and outcome negotiation are already identified as relocated forms of agency, narrowing novelty around purpose framing, evaluation, and commitment."
    }
,
    {
        "id":  "WuEtAl2026HumanAgencyCreativity",
        "tier":  "Tier 2",
        "full_citation":  "Wu, S. H., Yang, Y., Lee, A. Y., Liebscher, A., Rapuano, K., Niederhoffer, K., \u0026 Hancock, J. T. (2026). The role of human agency in human-AI co-creativity. In Proceedings of the 2026 Conference on Creativity and Cognition (pp. 1510–1515). ACM.",
        "link":  "https://doi.org/10.1145/3803784.3816857",
        "peer_reviewed":  true,
        "population":  "150 idea submitters and 142 evaluators were recruited; inferential analyses used smaller samples after exclusions.",
        "design":  "Preregistered experiment assigning participants to high-agency AI, low-agency AI, or human-only conditions for alternative-uses ideation, with separate evaluations and similarity analyses.",
        "finding_verbatim":  "ideas did not differ in individual-level quality (overall creativity, originality, and usefulness)",
        "effect_size":  "The paper reports a low-agency usefulness coefficient of β=.20, p=.04, and a homogenization test of F(2,121.67)=3.15, p\u003c.05, partial η²=.07; interpret within the paper\u0027s final analytic samples.",
        "conditions_and_limits":  "Agency was experimentally framed for creative ideation with LLM access; collective homogenization was assessed through similarity-to-centroid and exploratory clustering.",
        "criticism":  "A brief conference paper, artificial ideation task, and framing manipulation do not show judgment-right governance in consequential decisions. No individual-level quality advantage emerged across conditions.",
        "which_workflow":  "Driver\u0027s Seat value question—why more human agency is not automatically better; collective homogenization as a possible outcome.",
        "claimable_sentence":  "Agency framing did not improve individual idea quality, although low-agency AI use showed evidence of collective homogenization; more human activity should not be treated as universally superior."
    },
    {
        "id":  "XieEtAl2026",
        "tier":  "Tier 2",
        "full_citation":  "Xie, Y., Qi, T., Yi, J., Yang, X., Whalen, R., Huang, J., Ding, Q., Xie, Y., Xie, X., \u0026 Wu, F. (2026). Measuring human contribution in AI-assisted content generation. In Proceedings of the 64th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers) (pp. 6168–6190). Association for Computational Linguistics.",
        "link":  "https://doi.org/10.18653/v1/2026.acl-long.279",
        "peer_reviewed":  true,
        "population":  "Computational dataset with 2,000 entries per domain across abstracts, news, patents, and poems, spanning four input regimes, six LLMs, and five outputs per input. Human validation used 1,500 pairwise comparisons with three annotators.",
        "design":  "Information-theoretic measure of human contribution, φ=I(x;y)/I(y), followed by corpus experiments and deliberately separated pairwise human comparisons.",
        "finding_verbatim":  "The authors state: “we quantify the proportional information contribution of humans in content generation.”",
        "effect_size":  "Human ranking aligned with the metric on 95.93% of the evaluated pairs, which were deliberately selected to differ by more than .1. This does not estimate adjacent-score precision.",
        "conditions_and_limits":  "Text-generation domains and informational provenance, not governing authority, judgment quality, or consequential commitment. Human Contribution Ratio is convenient shorthand, not clearly the paper’s branded term.",
        "criticism":  "A short decisive human intervention and verbose low-governance input can receive misleadingly different contribution values. The selected-pair validation should not be generalized to fine-grained score accuracy.",
        "which_workflow":  "Driver’s Seat—nearest formal human-contribution coefficient and discriminator for the governance-versus-contribution distinction.",
        "claimable_sentence":  "Xie et al. quantify informational human contribution to AI-assisted text, but the coefficient does not identify who held evaluative or commitment authority."
    },
    {
        "id":  "XuEtAl2026",
        "tier":  "Tier 2",
        "full_citation":  "Xu, T., Chen, Y., Zhu, B., Fan, B., Wu, Y., \u0026 Jiang, Y. (2026). AI agency drives college students’ entrepreneurial thinking through human sense of agency in human and AI symbiosis. Scientific Reports. Advance online publication.",
        "link":  "https://doi.org/10.1038/s41598-026-60406-z",
        "peer_reviewed":  true,
        "population":  "Two Chinese college-student samples: n=398 for scale development and n=574 for the structural model/fsQCA, rather than one 972-person SEM sample.",
        "design":  "Cross-sectional self-report scale development followed by SEM and fsQCA linking perceived AI cognitive, interaction, and action support with human sense of agency, opportunity recognition, and creativity.",
        "finding_verbatim":  "“AI agency is positively associated with entrepreneurial thinking both directly and indirectly through the human sense of agency.”",
        "effect_size":  "The review does not quote path coefficients or standardized effects; none are reconstructed here. The two samples should not be pooled into a new effect or described as a single SEM sample.",
        "conditions_and_limits":  "Self-reported constructs, cross-sectional common method, college entrepreneurship-learning context, and no observed interaction trace or independent transfer measure.",
        "criticism":  "The title’s causal verb exceeds what the cross-sectional design establishes. Perceived support and perceived agency do not identify behavioral decision locus.",
        "which_workflow":  "Driver’s Seat—entrepreneurship-adjacent convergent comparator and causal-language caution.",
        "claimable_sentence":  "Across separate scale-development and modeling samples, perceived AI support related to self-reported agency and entrepreneurial thinking; the study did not observe who governed decisions."
    },
    {
        "id":  "YunTaranovaWang2026ChatbotAgency",
        "tier":  "Tier 2",
        "full_citation":  "Yun, B., Taranova, E., \u0026 Wang, A. Y. (2026). Does my chatbot have an agenda? Understanding human and AI agency in human-human-like chatbot interaction. In Proceedings of the 2026 CHI Conference on Human Factors in Computing Systems (pp. 1–32). ACM.",
        "link":  "https://doi.org/10.1145/3772318.3791620",
        "peer_reviewed":  true,
        "population":  "22 adults who used an LLM companion during a month-long study.",
        "design":  "Longitudinal use of a purpose-built companion chatbot, followed by semi-structured interviews, post-hoc elicitation, cross-participant chat review, and a strategy reveal.",
        "finding_verbatim":  "control shifted and was co-constructed turn-by-turn",
        "effect_size":  "No quantitative effect size; qualitative longitudinal study.",
        "conditions_and_limits":  "The framework maps human, AI, or hybrid agency across intention, execution, adaptation, delimitation, and negotiation in companion chat.",
        "criticism":  "Small, purpose-built companion-chat context; participant perceptions and selected moments do not validate a general-purpose behavioral coefficient or link agency to performance or learning.",
        "which_workflow":  "Driver\u0027s Seat dynamics and construct boundary—actor-by-action attribution over time.",
        "claimable_sentence":  "Yun and colleagues show qualitative, turn-by-turn redistribution of agency, closely occupying the dynamic actor-by-right space without offering a validated score."
    },
    {
        "id":  "ZhangEtAl2026MNARMissing",
        "tier":  "Tier 2",
        "full_citation":  "Zhang, J., Lu, J., \u0026 Zhang, Z. (2026). Modeling missing response data in item response theory: Addressing missing not at random mechanism with monotone missing characteristics. Journal of Educational Measurement, 63(1), e12428. First published online February 24, 2025.",
        "link":  "https://doi.org/10.1111/jedm.12428",
        "peer_reviewed":  true,
        "population":  "Four simulation studies and an application to PISA 2015 science data.",
        "design":  "Bayesian item-response model for missing-not-at-random responses using monotone missing characteristics, model-comparison criteria, and slice-sampling estimation.",
        "finding_verbatim":  "This study introduces a missing data model based on the missing not at random (MNAR) mechanism.",
        "effect_size":  "No single effect size; the evidence consists of simulation performance and an empirical model illustration.",
        "conditions_and_limits":  "The model uses cumulative prior missing indicators to represent monotone individual missingness and compares constrained MAR and MNAR specifications.",
        "criticism":  "Monotone item nonresponse in an assessment is not equivalent to trace eligibility in conversation. Driver\u0027s Seat abstention can arise from several nonmonotone task, capture, opportunity, and behavior mechanisms.",
        "which_workflow":  "WP-08a Phase A: sensitivity to missing-not-at-random mechanisms.",
        "claimable_sentence":  "Educational nonresponse can require an MNAR model when missingness depends on latent or prior response processes; scored-only summaries should not assume ignorability."
    },
    {
        "id":  "ZhangWangYi2025AgencyReview",
        "tier":  "Tier 2",
        "full_citation":  "Zhang, S., Wang, H., \u0026 Yi, X. (2025). Exploring collaboration patterns and strategies in human-AI co-creation through the lens of agency: A scoping review of the top-tier HCI literature. Proceedings of the ACM on Human-Computer Interaction, 9(7), Article CSCW413, 1–43.",
        "link":  "https://doi.org/10.1145/3757594",
        "peer_reviewed":  true,
        "population":  "134 papers from top-tier HCI and CSCW venues over approximately 20 years.",
        "design":  "Scoping review mapping agency configurations, control mechanisms, and interaction contexts.",
        "finding_verbatim":  "an integrated theoretical framework structuring agency patterns, control mechanisms, and interaction contexts",
        "effect_size":  "No quantitative effect size; scoping synthesis.",
        "conditions_and_limits":  "The corpus is restricted to selected top-tier HCI/CSCW venues and co-creative settings. ACM records a 2026 corrigendum (doi:10.1145/3779003); use the corrected version of record.",
        "criticism":  "The review demonstrates a mature agency literature but does not provide a validated episode-level coefficient or establish prediction of outcomes.",
        "which_workflow":  "WP-08A Phase A: agency configurations and operational control mechanisms.",
        "claimable_sentence":  "A 134-paper scoping review already systematizes agency patterns and control mechanisms, ruling out a claim that Driver\u0027s Seat is the first human–AI agency framework."
    },
    {
        "id":  "ZhuEtAl2026",
        "tier":  "Tier 3",
        "full_citation":  "Zhu, L., Lu, Q., Ding, M., Lee, S. U., \u0026 Wang, C. (2026). Designing meaningful human oversight in AI. AI and Ethics, 6(3), Article 286.",
        "link":  "https://doi.org/10.1007/s43681-026-01147-7",
        "peer_reviewed":  true,
        "population":  "No human participant sample. A structured known-use analysis screened 54 public cases and retained 12.",
        "design":  "Conceptual/interpretive framework separating AI operative agency from human evaluative agency; two authors independently extracted and adjudicated evidence from retained cases.",
        "finding_verbatim":  "The authors distinguish “AI operative agency” from “human evaluative agency” and reject a zero-sum relation between them.",
        "effect_size":  "Not applicable. The case analysis reports no participant-level effect or psychometric coefficient.",
        "conditions_and_limits":  "The 12 cases rely heavily on vendor technical documents and product whitepapers; the authors characterize the work as interpretive and call for empirical validation.",
        "criticism":  "Under the library’s no-vendor-material rule, cite the peer-reviewed theoretical distinction, not the vendor case claims. Nominal human presence is not meaningful oversight.",
        "which_workflow":  "Driver’s Seat—direct precursor to the separate governance and AI-operative-contribution axes.",
        "claimable_sentence":  "Human evaluative agency and AI operative agency can both be high, so AI contribution must not be mechanically subtracted from human governance."
    },
    {
        "id":  "CampbellFiske1959MTMM",
        "tier":  "Tier 1",
        "full_citation":  "Campbell, D. T., & Fiske, D. W. (1959). Convergent and discriminant validation by the multitrait-multimethod matrix. Psychological Bulletin, 56(2), 81–105.",
        "link":  "https://doi.org/10.1037/h0046016",
        "peer_reviewed":  true,
        "population":  "No new single participant sample. The methodological paper develops and illustrates a validity framework using matrices of correlations among multiple traits measured by multiple methods.",
        "design":  "Foundational construct-validity framework requiring both convergence across different methods for the same trait and discrimination from different traits, while making method effects inspectable.",
        "finding_verbatim":  "“Measures of the same trait should correlate higher with each other than they do with measures of different traits involving separate methods.”",
        "effect_size":  "Not applicable. The paper presents a validation logic and illustrative correlation matrices rather than one participant-level intervention effect.",
        "conditions_and_limits":  "The approach requires at least two traits and at least two methods, with measures designed so trait and method variance can be compared.",
        "criticism":  "The original criteria rely substantially on matrix inspection and correlations. Correlation is not agreement, a favorable matrix does not prove causal validity, and later latent-variable MTMM methods address limitations not resolved in the 1959 paper.",
        "which_workflow":  "WP-08A Phase A: prospective convergent and discriminant validation of displayed judgment governance against contribution, authorship, output quality, and adjacent constructs.",
        "claimable_sentence":  "A proposed governance measure needs evidence that it converges with measures of the same construct while remaining distinguishable from production volume, authorship, output quality, and adjacent traits."
    }
]
