[
  {
    "id": "ChaseSimon1973",
    "tier": "Tier 3",
    "full_citation": "Chase, W. G., & Simon, H. A. (1973). Perception in chess. Cognitive Psychology, 4(1), 55\u201381.",
    "link": "https://doi.org/10.1016/0010-0285(73)90004-2",
    "peer_reviewed": "Yes",
    "population": "Three chess players: one master, one Class A player, and one beginner.",
    "design": "Foundational descriptive expert\u2013novice study using reconstruction of 20 game positions and 8 randomized positions, both during viewing and after 5-second exposure, plus placement-interval analysis and a small follow-up using 9 puzzle positions.",
    "finding_verbatim": "the amount of information extracted from a briefly exposed position varies with playing strength",
    "effect_size": "No standardized effect size or confidence interval reported. Approximate first-trial recall for actual middlegame positions was 16, 8, and 4 pieces for the master, Class A player, and beginner; follow-up puzzle-position recall was 81%, 49%, and 33%, respectively. Mean between-glance intervals were 2.8, 3.2, and 3.5 seconds, p<.05.",
    "conditions_and_limits": "The large skill difference appeared for familiar, meaningful configurations. The proposed chunks linked pieces through relations such as defense, proximity, attack, color, and type. The authors cautioned that rapid perceptual processing may not be available to conscious introspection.",
    "criticism": "N=3, descriptive, and chess-specific; no modern blinding, reliability analysis, standardized effects, or confidence intervals. Gobet and Simon (1996) later showed that the common claim that the skill advantage disappears completely for random positions is too strong.",
    "which_workflow": "Expert-judgment capture\u2014structural expert\u2013novice differences in pattern recognition and cue organization.",
    "claimable_sentence": "In a foundational but extremely small chess study, stronger players reconstructed meaningful positions by encoding larger relational chunks, not by displaying a general memory advantage for arbitrary material."
  },
  {
    "id": "GobetSimon1996",
    "tier": "Tier 2",
    "full_citation": "Gobet, F., & Simon, H. A. (1996). Recall of rapidly presented random chess positions is a function of skill. Psychonomic Bulletin & Review, 3(2), 159\u2013163.",
    "link": "https://doi.org/10.3758/BF03212414",
    "peer_reviewed": "Yes",
    "population": "Previously published samples of chess players at different skill levels who completed rapid-recall tasks using game and randomized positions.",
    "design": "Quantitative review and reanalysis of chess-position recall studies using presentation times of approximately 3\u201310 seconds; no new participant sample.",
    "finding_verbatim": "strong players generally maintain some superiority over weak players even with random positions",
    "effect_size": "No new pooled standardized effect size or confidence interval reported. The paper reanalyzed results across prior studies and concluded that the relative skill difference was substantially smaller for randomized than for game positions.",
    "conditions_and_limits": "Random boards can contain accidental familiar chunks, while normal game positions contain many more meaningful relations. The paper narrows, rather than rejects, the familiar-configuration account of expert recall.",
    "criticism": "This is a reanalysis of heterogeneous earlier studies, not a preregistered meta-analysis or new experiment, and it reports no pooled effect or confidence interval. Its principal value is correcting the widely repeated absolute claim that expert superiority vanishes on random boards.",
    "which_workflow": "Expert-judgment capture\u2014boundary condition and correction to the classic chunking result.",
    "claimable_sentence": "Randomizing chess positions greatly reduces the expert recall advantage but does not reliably eliminate it, so the familiar claim that the advantage vanishes is not supported as an absolute."
  },
  {
    "id": "GegenfurtnerEtAl2011",
    "tier": "Tier 1",
    "full_citation": "Gegenfurtner, A., Lehtinen, E., & S\u00e4lj\u00f6, R. (2011). Expertise differences in the comprehension of visualizations: A meta-analysis of eye-tracking research in professional domains. Educational Psychology Review, 23(4), 523\u2013552.",
    "link": "https://doi.org/10.1007/s10648-011-9174-7",
    "peer_reviewed": "Yes",
    "population": "819 experts, 187 intermediates, and 893 novices across sports, medicine, transportation, and other professional domains.",
    "design": "Random-effects meta-analysis of 65 articles, 73 independent data sources, and 296 effect sizes from eye-tracking studies of visualization comprehension.",
    "finding_verbatim": "experts, when compared with non-experts, had shorter fixation durations, more fixations on task-relevant areas, and fewer fixations on task-redundant areas",
    "effect_size": "Corrected expert\u2013novice correlations with 99% CIs: relevant-fixation count r_c=.53 [.49, .57] (k=8, N=185); redundant-fixation count r_c=\u2212.31 [\u2212.35, \u2212.27] (k=3, N=65); relevant-fixation duration r_c=.27 [.21, .33] (k=15, N=325); redundant-fixation duration r_c=\u2212.43 [\u2212.48, \u2212.39] (k=8, N=147); time to first relevant fixation r_c=\u2212.31 [\u2212.34, \u2212.28] (k=7, N=125); saccade length r_c=.30 [.25, .35] (k=8, N=196); response time r_c=\u2212.38 [\u2212.40, \u2212.36] (k=37, N=1,050); performance accuracy r_c=.45 [.43, .47] (k=46, N=1,175).",
    "conditions_and_limits": "Effects were heterogeneous and moderated by visualization dynamics, realism, dimensionality, modality, and annotation; task complexity, time, and control; and professional domain. Eye gaze indexes attention only when the visible environment is pertinent to the task.",
    "criticism": "The underlying evidence is overwhelmingly cross-sectional and cannot establish developmental causality. Constituent studies usually had small groups, some meta-analytic cells were very small, estimates corrected only sampling error, and the selected parameters and moderators were incomplete.",
    "which_workflow": "Expert-judgment capture\u2014cross-domain evidence on selective attention to relevant and redundant cues.",
    "claimable_sentence": "Across professional eye-tracking studies, experts tended to reach relevant information sooner, attend more selectively to it, and spend less attention on redundant information, with substantial variation by task and display."
  },
  {
    "id": "ChiEtAl1981",
    "tier": "Tier 3",
    "full_citation": "Chi, M. T. H., Feltovich, P. J., & Glaser, R. (1981). Categorization and representation of physics problems by experts and novices. Cognitive Science, 5(2), 121\u2013152.",
    "link": "https://doi.org/10.1207/s15516709cog0502_2",
    "peer_reviewed": "Yes",
    "population": "The central card-sort study compared 8 advanced physics PhDs with 8 undergraduates who had completed one semester of mechanics; follow-up protocol studies used still smaller subsets, often two experts and two novices.",
    "design": "Four exploratory studies using repeated sorting of 24 mechanics problems, explanations of categories, constructed problems, category elaboration, and think-aloud accounts of a basic solution approach.",
    "finding_verbatim": "the actual cues used by the experts are not the words themselves, but what they signify",
    "effect_size": "No standardized effect size, inferential confidence interval, or group-level confidence interval reported. Experts and novices formed similar numbers of categories (8.4 vs. 8.6), but only 5 of 20 distinct category labels overlapped; experts predominantly used governing principles or solution methods, while novices more often used literal objects or terms.",
    "conditions_and_limits": "Experts interpreted visible words and objects as signs of derived, relational features and tested whether the conditions of a principle schema held. The result supports representing cue meaning and relations, not merely listing literal cues.",
    "criticism": "Tiny convenience samples, investigator interpretation of protocols, no modern blind coding or reliability statistics, and textbook-mechanics tasks. Categorization was treated as a proxy for expertise rather than shown to cause superior solving; later studies found considerable overlap across nominal expertise groups.",
    "which_workflow": "Expert-judgment capture\u2014problem representation, relational cue meaning, and limits of literal cue lists.",
    "claimable_sentence": "In small foundational physics studies, experts organized problems around governing relations and solution principles, whereas novices more often organized them around literal features, but the result should not be treated as a universal binary."
  },
  {
    "id": "HardimanEtAl1989",
    "tier": "Tier 2",
    "full_citation": "Hardiman, P. T., Dufresne, R., & Mestre, J. P. (1989). The relation between problem categorization and problem solving among experts and novices. Memory & Cognition, 17(5), 627\u2013638.",
    "link": "https://doi.org/10.3758/BF03197085",
    "peer_reviewed": "Yes",
    "population": "Experiment 1 included 45 undergraduates who had earned at least B+ in mechanics and 10 physicists; Experiment 2 included 44 undergraduates and 7 PhD physicists.",
    "design": "Two cross-sectional experiments using similarity triads that crossed deep and surface structure, followed by categorization and solution of four single-principle mechanics problems. Experiment 2 classified novices as surface-oriented (n=17), principle-oriented (n=11), or mixed (n=16).",
    "finding_verbatim": "the conclusion that novices focus almost exclusively on surface feature similarity is unwarranted",
    "effect_size": "No standardized effect size or confidence interval reported for the main comparisons. Experts selected the deep match 78% of the time versus 59% for novices, F(1,53)=28.78, p<.0001; surface similarity also harmed experts, F(3,27)=10.56, p=.0001. In Experiment 2, solution means were 14%, 32%, and 57% for surface, mixed, and principle-oriented novices, F(2,41)=16.19, p<.0001. Principle-reason frequency correlated with solving, r=.63, p<.0001; partial r=.505, p<.0004.",
    "conditions_and_limits": "Novices were heterogeneous; stronger novices already used principles, and salient surface similarities affected experts as well. The categorization\u2013performance association was correlational and did not establish causal direction.",
    "criticism": "Cross-sectional design, a mixed expert group, only four retained single-principle solution items, and no standardized effects or confidence intervals. Categorization performance is evidence about one dimension of representation, not an exhaustive measure of expertise.",
    "which_workflow": "Expert-judgment capture\u2014novice heterogeneity, deep versus surface representation, and surface-cue error.",
    "claimable_sentence": "Physics novices were not uniformly surface-bound: those who reasoned more often from governing principles solved more problems, while surface similarity also misled experts under some conditions."
  },
  {
    "id": "MasonSingh2011",
    "tier": "Tier 2",
    "full_citation": "Mason, A., & Singh, C. (2011). Assessing expertise in introductory physics using categorization task. Physical Review Special Topics\u2013Physics Education Research, 7(2), 020110.",
    "link": "https://doi.org/10.1103/PhysRevSTPER.7.020110",
    "peer_reviewed": "Yes",
    "population": "403 introductory physics students in two algebra-based courses (n=109 and n=114) and one calculus-based course (n=180), plus 21 physics graduate students and 7 faculty members.",
    "design": "Large cross-sectional replication and extension using two versions of a 25-problem mechanics card sort, with think-aloud interviews in a small subset. Only seven problems from the original Chi et al. study were available.",
    "finding_verbatim": "there is a wide distribution of expertise in mechanics among introductory and graduate students",
    "effect_size": "No standardized effect size or confidence interval reported. Mean percentage assigned to categories judged good was 34.4% for calculus-based introductory students and 18.7% for algebra-based students; calculus-based and graduate-student distributions overlapped substantially, and faculty performed best. Performance on the 15 common problems did not differ between problem-set versions, p=.90.",
    "conditions_and_limits": "Categorization was curriculum-sensitive and distributed along a continuum rather than cleanly separating nominal novices and experts. It is one proxy predictor of expertise, not a complete definition.",
    "criticism": "One university, cross-sectional design, partly judgment-based category scoring, two problem-set versions, and only seven original Chi items, so this was not an exact replication. Course selection and preparation may explain part of the calculus\u2013algebra difference.",
    "which_workflow": "Expert-judgment capture\u2014partial replication and evidence against binary expert\u2013novice labeling.",
    "claimable_sentence": "A much larger physics card-sort study found wide within-group variation and substantial overlap between calculus-based introductory students and graduate students, complicating a simple expert\u2013novice dichotomy."
  },
  {
    "id": "KleinEtAl2010",
    "tier": "Tier 3",
    "full_citation": "Klein, G., Calderwood, R., & Clinton-Cirocco, A. (2010). Rapid decision making on the fire ground: The original study plus a postscript. Journal of Cognitive Engineering and Decision Making, 4(3), 186\u2013209.",
    "link": "https://doi.org/10.1518/155534310X12844000801203",
    "peer_reviewed": "Yes",
    "population": "26 experienced fireground commanders from seven U.S. municipalities; mean fire-service experience 23.2 years, range 12\u201337 years.",
    "design": "Peer-reviewed publication of an edited 1985 field study with a 2010 postscript: 29 retrospective, semistructured critical-incident interviews covering 32 nonroutine incidents and 156 probed decision points.",
    "finding_verbatim": "In less than 12% of them was there any evidence of simultaneous comparisons and relative evaluation of two or more options.",
    "effect_size": "No inferential effect size or confidence interval reported. Of 156 decision points, 28 involved more than one identified option and 16 involved relative evaluation; more than 80% were classified as recognitional, and 78% of decisions were estimated to have taken less than one minute.",
    "conditions_and_limits": "The findings concern highly experienced commanders making time-pressured decisions in a familiar, high-stakes domain. In an unfamiliar oil-pumping-station incident, commanders shifted toward explicit option analysis and external consultation.",
    "criticism": "Retrospective sampling of memorable incidents; decision points and alternatives were often inferred through interviewer probes; no formal intercoder reliability; recall, reconstruction, and selection bias; one domain. The authors explicitly declined to present the data as firm evidence for the synthesized model.",
    "which_workflow": "Expert-judgment capture\u2014direct lineage for decision points, situational cues, recognitional patterns, and missing-cue probes.",
    "claimable_sentence": "The original recognition-primed-decision field study shows that decision points and cue interpretations can structure expert elicitation, but its retrospective methods do not validate them as an exhaustive or reliably captured ontology of expertise."
  },
  {
    "id": "Hinds1999",
    "tier": "Tier 2",
    "full_citation": "Hinds, P. J. (1999). The curse of expertise: The effects of expertise and debiasing methods on predictions of novice performance. Journal of Experimental Psychology: Applied, 5(2), 205\u2013221.",
    "link": "https://doi.org/10.1037/1076-898X.5.2.205",
    "peer_reviewed": "Yes",
    "population": "Study 1 included 95 cellular-company sales staff, recent customers, and people with no cellular-telephone experience. Study 2 included 49 participants assigned to experimentally induced high- or low-expertise conditions on a LEGO task.",
    "design": "Study 1 compared expert, intermediate, and novice predictions of novice voicemail-task time and tested recall/list debiasing. Study 2 used a 2\u00d72 between-subjects design: five prior LEGO builds versus no practice, crossed with an unaided estimate versus a list of novice difficulties.",
    "finding_verbatim": "those with more expertise were worse predictors of novice performance times and were resistant to debiasing techniques intended to reduce underestimation",
    "effect_size": "No standardized effect size or numerical confidence interval reported; a figure displayed 95% CI bars without numerical bounds. Study 1 actual median was 33 minutes; unaided estimates were 12.86 (SD=4.23) for experts, 20.09 (SD=11.97) for intermediates, and 15.71 (SD=9.46) for novices, F(2,92)=3.83, p<.05. Study 2 actual median was 12.27 minutes; high-expertise estimate 8.5 (SD=3.36) versus low-expertise 12.9 (SD=6.73), F(1,45)=8.96, p<.05. The list main effect, F(1,45)=1.44, and interaction, F(1,45)=0.82, were nonsignificant.",
    "conditions_and_limits": "The outcome was accuracy in predicting novice completion time in two procedural tasks. The result directly concerns experts' ability to anticipate novice difficulty, not the accuracy of their own task performance.",
    "criticism": "Study 1 compared groups drawn from different populations with potentially different motivations. Study 2 created short-term task familiarity rather than professional expertise. Both samples were small and the article reported imprecise p<.05 thresholds and no standardized effects or numerical CIs.",
    "which_workflow": "Expert-judgment capture\u2014direct negative evidence on expert-authored predictions of novice errors and difficulty.",
    "claimable_sentence": "In two procedural tasks, more knowledgeable participants underestimated novice completion time, so proposed novice errors should be validated with actual novice performance rather than accepted from expert prediction alone."
  },
  {
    "id": "KahnemanKlein2009",
    "tier": "Tier 3",
    "full_citation": "Kahneman, D., & Klein, G. (2009). Conditions for intuitive expertise: A failure to disagree. American Psychologist, 64(6), 515\u2013526.",
    "link": "https://doi.org/10.1037/a0016755",
    "peer_reviewed": "Yes",
    "population": "No new participant sample; the article integrates evidence from heuristics-and-biases research and naturalistic decision-making research across multiple professional domains.",
    "design": "Integrative narrative review and theoretical reconciliation, not a systematic review, meta-analysis, or primary experiment.",
    "finding_verbatim": "we do not believe that subjective confidence reliably indicates whether intuitive judgments or decisions are valid",
    "effect_size": "Not applicable: no new quantitative synthesis, effect size, or confidence interval reported.",
    "conditions_and_limits": "The authors argue that skilled intuition requires an environment containing sufficiently valid cues and adequate opportunity to learn its regularities through prolonged practice and feedback that is rapid and reasonably unequivocal.",
    "criticism": "Narrative and theoretical rather than systematic or quantitative; claims synthesize heterogeneous literatures and do not estimate how often specified environments meet the proposed conditions. It is a boundary framework, not direct validation of any elicitation method.",
    "which_workflow": "Expert-judgment capture\u2014ecological boundary conditions for trusting intuition, cues, and confidence.",
    "claimable_sentence": "Intuitive expertise is most defensible where the environment contains learnable regularities and supplies rapid, clear feedback; an expert's confidence alone is not evidence of validity."
  },
  {
    "id": "SinhaKapur2021",
    "tier": "Tier 1",
    "full_citation": "Sinha, T., & Kapur, M. (2021). When problem solving followed by instruction works: Evidence for productive failure. Review of Educational Research, 91(5), 761\u2013798.",
    "link": "https://doi.org/10.3102/00346543211019105",
    "peer_reviewed": "Yes",
    "population": "Learners from Grade 2 through university across 53 studies and 166 comparisons; studies were concentrated in mathematics and other STEM domains and in Grades 6\u201310 and undergraduate education.",
    "design": "Meta-analysis comparing problem solving followed by instruction (PS-I) with instruction followed by problem solving (I-PS), with experimental and quasi-experimental studies.",
    "finding_verbatim": "Our results showed a significant, moderate effect in favor of PS-I.",
    "effect_size": "Combined conceptual-knowledge and transfer outcomes: Hedges's g=0.36, 95% CI [0.20, 0.51]. Procedural knowledge: g=-0.03, 95% CI [-0.20, 0.15]. Heterogeneity for the main model: I\u00b2=42.01%.",
    "conditions_and_limits": "Benefits were associated with higher-fidelity productive-failure designs and were concentrated in conceptual learning and transfer. Results did not support a procedural-fluency advantage; younger learners in Grades 2\u20135 and domain-general-skills tasks tended to favor instruction first.",
    "criticism": "The synthesis combines experiments and quasi-experiments, and implementation fidelity, duration, population, and domain covary. The average therefore does not establish that problem solving first is universally preferable, and the null procedural result is material negative evidence.",
    "which_workflow": "Expertise transfer\u2014sequencing practice and instruction; productive-failure boundary evidence.",
    "claimable_sentence": "Problem solving before instruction can improve later conceptual understanding and transfer under well-aligned conditions, but the meta-analysis found no procedural-knowledge advantage and identified important reversals by learner age and task type."
  },
  {
    "id": "KeithFrese2008",
    "tier": "Tier 1",
    "full_citation": "Keith, N., & Frese, M. (2008). Effectiveness of error management training: A meta-analysis. Journal of Applied Psychology, 93(1), 59\u201369.",
    "link": "https://doi.org/10.1037/0021-9010.93.1.59",
    "peer_reviewed": "Yes",
    "population": "Twenty-four studies with 2,183 participants; 21 studies trained software skills and three used computer-delivered decision tasks.",
    "design": "Meta-analysis of error-management training, which combines active exploration with explicit encouragement to make, examine, and learn from errors, compared with error-avoidant or other training.",
    "finding_verbatim": "Results suggest that EMT may be better suited than error-avoidant training methods for promotion of transfer to novel tasks.",
    "effect_size": "The article reports 90% confidence intervals: overall Cohen's d=0.44, 90% CI [0.27, 0.61]; within-training performance d=-0.15, 90% CI [-0.47, 0.17]; post-training transfer d=0.56, 90% CI [0.40, 0.73]; analogical transfer d=0.20, 90% CI [0.02, 0.39]; adaptive transfer d=0.80, 90% CI [0.56, 1.05]. These must not be relabeled as 95% intervals.",
    "conditions_and_limits": "The intervention requires a psychologically safe framing of errors plus opportunities to explore, receive information, and revise performance. The strongest reported effect concerned adaptive transfer rather than reproduction of the trained task.",
    "criticism": "The evidence base is overwhelmingly software training, with little direct evidence for interpersonal, managerial, or authentic professional judgment. Error framing and active exploration are bundled, feedback coding was uncertain, and study-author clustering may contribute to the results.",
    "which_workflow": "Expertise transfer\u2014error-management practice and adaptive performance.",
    "claimable_sentence": "Error-management training improved transfer on average in this meta-analysis, but nearly all included studies involved software skills, so transfer to complex professional judgment remains uncertain."
  },
  {
    "id": "DyreEtAl2017",
    "tier": "Tier 1",
    "full_citation": "Dyre, L., Tabor, A., Ringsted, C., & Tolsgaard, M. G. (2017). Imperfect practice makes perfect: Error management training improves transfer of learning. Medical Education, 51(2), 196\u2013206.",
    "link": "https://doi.org/10.1111/medu.13208",
    "peer_reviewed": "Yes",
    "population": "Sixty medical students who were novices in fetal ultrasonography were randomized; 56 completed the study.",
    "design": "Randomized trial of a three-hour fetal-ultrasound simulator curriculum using error-management training versus error-avoidance training, followed by blinded assessment on real patients 7\u201310 days later.",
    "finding_verbatim": "The provision of error management instructions during simulation-based training improves the transfer of learning to the clinical setting.",
    "effect_size": "Clinical transfer performance was 67.7%, 95% CI [62.4, 72.9], after error-management training versus 51.7%, 95% CI [45.8, 57.6], after error-avoidance training; Cohen's d=1.1, 95% CI [0.5, 1.7], p<.001. Diagnostic-accuracy difference: d=0.46, 95% CI [-0.06, 1.0], p=.082.",
    "conditions_and_limits": "Training involved simulator practice, deliberate exposure to errors and exploration, and transfer to the same procedural domain after a short delay. The significant result was for rated clinical performance; the diagnostic-accuracy interval included zero.",
    "criticism": "The trial was small, involved novices and one motor-diagnostic procedure, and had a short follow-up with no patient outcomes. The intervention bundles error framing and exploration, so their separate causal contributions cannot be identified.",
    "which_workflow": "Expertise transfer\u2014error-management simulation with short-delay clinical transfer.",
    "claimable_sentence": "In one small randomized ultrasound study, error-management simulation improved blinded ratings of clinical transfer after 7\u201310 days, while the diagnostic-accuracy estimate remained compatible with no effect."
  },
  {
    "id": "AliagaEtAl2024",
    "tier": "Tier 1",
    "full_citation": "Aliaga, L., Bavolek, R. A., Cooper, B., Mariorenzi, A., Ahn, J., Kraut, A., Duong, D., Burger, C., & Gisondi, M. A. (2024). Error management training and adaptive expertise in learning computed tomography interpretation: A randomized clinical trial. JAMA Network Open, 7(9), e2431600.",
    "link": "https://doi.org/10.1001/jamanetworkopen.2024.31600",
    "peer_reviewed": "Yes",
    "population": "Two hundred twelve postgraduate-year 1\u20134 emergency-medicine residents at seven geographically diverse US residency programs were randomized; 150 completed the post-test.",
    "design": "Three-arm randomized clinical trial comparing difficult error-management training, easy error-management training, and error-avoidance or instruction-first training in a one-hour online head-CT curriculum, followed by an immediate post-test of familiar and novel cases.",
    "finding_verbatim": "Residents who made more errors during training made fewer errors when they subsequently evaluated novel head CT cases on a posttest.",
    "effect_size": "Novel-case performance: difficult EMT 60.6%, 95% CI [56.1, 65.1]; easy EMT 45.2%, 95% CI [39.9, 50.6]; error-avoidance training 40.9%, 95% CI [36.0, 45.7]; \u03b7\u00b2=0.19, p<.001, with no confidence interval reported for \u03b7\u00b2. Familiar-case performance was 75.1%, 73.6%, and 74.4%, respectively, p=.84.",
    "conditions_and_limits": "The benefit required difficult questions that induced errors before instruction and appeared on structurally related but novel CT cases. It did not improve performance on familiar cases and was measured immediately in an unsupervised online setting.",
    "criticism": "There was no baseline pre-test, 29.2% of randomized residents did not complete the post-test, and completion differed across arms. Outcomes were immediate simulated interpretations, with no retention, workplace-transfer, diagnostic-management, or patient outcome.",
    "which_workflow": "Expertise transfer\u2014adaptive expertise through difficult error-management practice.",
    "claimable_sentence": "Difficult error-first training improved immediate performance on novel head-CT cases in a multicenter resident trial, without improving familiar-case performance; longer-term clinical transfer is unknown."
  },
  {
    "id": "BrushEtAl2019",
    "tier": "Tier 2",
    "full_citation": "Brush, J. E., Jr., Lee, M., Sherbino, J., Taylor-Fishwick, J. C., & Norman, G. (2019). Effect of teaching Bayesian methods using learning by concept vs learning by example on medical students' ability to estimate probability of a diagnosis: A randomized clinical trial. JAMA Network Open, 2(12), e1918023.",
    "link": "https://doi.org/10.1001/jamanetworkopen.2019.18023",
    "peer_reviewed": "Yes",
    "population": "Sixty-one medical students at McMaster University and Eastern Virginia Medical School: 22 assigned to concept instruction, 20 to repeated examples, and 19 to control materials.",
    "design": "Randomized trial comparing an 18-minute video on Bayesian concepts and an anchoring-and-adjustment heuristic, 27 practice cases with feedback, and a reading control; testing included a new but structurally similar diagnosis.",
    "finding_verbatim": "The study showed a modest advantage for students who received theoretical instruction on bayesian concepts.",
    "effect_size": "Overall signed discrepancy from Bayes' rule: concept 0.4% (SE 0.7), examples 3.5% (SE 0.7), control 4.3% (SE 0.7), F=9.07, p<.001. On the new diagnosis: concept -2.0% (SE 1.6), examples 4.6% (SE 1.7), control 7.9% (SE 1.8), F=8.74, p<.001. Standardized effects and confidence intervals were not reported.",
    "conditions_and_limits": "The instructed principle and heuristic were directly applicable to a near-transfer probability-estimation task. Testing was short term, and concept-instruction participants spent more time on the intervention.",
    "criticism": "The sample and assessment were small, all groups performed near the Bayesian benchmark, and the observed advantage was modest. Probability estimates are not equivalent to diagnostic decisions or patient outcomes, and long-term retention was not tested.",
    "which_workflow": "Heuristic transfer\u2014explicit conceptual instruction versus repeated examples.",
    "claimable_sentence": "A small medical-student trial found that explicit Bayesian concept and heuristic instruction transferred modestly to a new, structurally similar probability-estimation problem."
  },
  {
    "id": "MamedeEtAl2020",
    "tier": "Tier 1",
    "full_citation": "Mamede, S., de Carvalho-Filho, M. A., de Faria, R. M. D., Franci, D., Nunes, M. D. P. T., Ribeiro, L. M. C., Biegelmeyer, J., Zwaan, L., & Schmidt, H. G. (2020). \u2018Immunising\u2019 physicians against availability bias in diagnostic reasoning: A randomised controlled experiment. BMJ Quality & Safety, 29(7), 550\u2013559.",
    "link": "https://doi.org/10.1136/bmjqs-2019-010079",
    "peer_reviewed": "Yes",
    "population": "Ninety-one second-year internal-medicine residents at eight teaching hospitals in five Brazilian cities.",
    "design": "Three-phase multicenter randomized experiment: compare-and-contrast instruction with feedback for one of two disease clusters, availability-bias induction one week later, and diagnostic testing on vignettes.",
    "finding_verbatim": "An intervention directed at increasing knowledge of clinical findings that discriminate between similar-looking diseases decreased physicians' susceptibility to availability bias.",
    "effect_size": "For bias-exposed vignettes, diagnostic accuracy was 0.40 after relevant immunization versus 0.24 without it; difference=0.16, 95% CI [0.05, 0.27], p=.004. For vignettes not exposed to bias, difference=-0.05, 95% CI [-0.17, 0.08], p=.45. No standardized effect was reported.",
    "conditions_and_limits": "Instruction required comparison of look-alike diseases, feedback, and attention to discriminating clinical findings. The effect appeared only for the trained disease cluster under induced availability bias after one week.",
    "criticism": "The outcome was diagnosis of simulated vignettes involving only two disease clusters, not workplace behavior or patient outcomes. The specificity of the benefit argues against treating it as a general debiasing skill.",
    "which_workflow": "Heuristic transfer\u2014contrastive cue instruction and domain-specific resistance to novice error.",
    "claimable_sentence": "Compare-and-contrast instruction on discriminating clinical cues reduced availability-bias errors one week later, but only within the disease cluster for which residents had received the instruction."
  },
  {
    "id": "OSullivanSchofield2019",
    "tier": "Tier 2",
    "full_citation": "O'Sullivan, E. D., & Schofield, S. J. (2019). A cognitive forcing tool to mitigate cognitive bias\u2014a randomised control trial. BMC Medical Education, 19, 12.",
    "link": "https://doi.org/10.1186/s12909-018-1444-3",
    "peer_reviewed": "Yes",
    "population": "Seventy-six analyzable volunteer medical students and physicians from the UK, Republic of Ireland, and North America, spanning student to attending level; a separate convenience sample of 20 UK doctors completed think-aloud interviews.",
    "design": "Single-blind online randomized trial of the SLOW cognitive-forcing mnemonic versus no prompt on bias-inducing clinical vignettes, supplemented by a qualitative think-aloud study.",
    "finding_verbatim": "The quantitative data failed to show an improvement in accuracy despite a positive qualitative experience.",
    "effect_size": "Mean correct answers were 2.8 with the tool and 3.1 in control; between-group difference 95% CI [-0.94, 0.45], p=.49. A standardized effect was not reported.",
    "conditions_and_limits": "The tool was a brief generic mnemonic delivered online and used during written vignettes. Participants reported a positive experience, but the randomized accuracy outcome was null.",
    "criticism": "Of 300 recruits, 244 produced incomplete or insufficient responses, a 74.6% dropout rate. Engagement with the tool was uncertain, the sample was heterogeneous and self-selected, the study was underpowered relative to its target, and no clinical behavior was observed.",
    "which_workflow": "Heuristic transfer\u2014direct negative evidence for a generic debiasing checklist.",
    "claimable_sentence": "A generic cognitive-forcing mnemonic did not improve vignette accuracy in a randomized trial, despite favorable think-aloud impressions and substantial uncertainty created by very high attrition."
  },
  {
    "id": "TofelGrehlFeldon2013",
    "tier": "Tier 2",
    "full_citation": "Tofel-Grehl, C., & Feldon, D. F. (2013). Cognitive task analysis\u2013based training: A meta-analysis of studies. Journal of Cognitive Engineering and Decision Making, 7(3), 293\u2013304.",
    "link": "https://doi.org/10.1177/1555343412474821",
    "peer_reviewed": "Yes",
    "population": "Fifty-six coded outcome cases from a small controlled-study corpus spanning military, medical, industrial, university, and government training contexts.",
    "design": "Meta-analysis comparing instruction whose content was derived through cognitive task analysis with alternative ways of identifying and representing instructional content.",
    "finding_verbatim": "Overall, the effect of CTA-based instruction is large (Hedges's g = 0.871).",
    "effect_size": "Overall Hedges's g=0.871; no confidence interval was reported. Unadjusted mean effects by elicitation method were g=1.598 for PARI (11 cases), g=0.329 for the Critical Decision Method (4 cases), and g=0.729 for other or unreported methods (41 cases); subgroup confidence intervals were not reported.",
    "conditions_and_limits": "Effects varied substantially with the CTA method and training setting. The analysis concerns the downstream instructional value of CTA-derived content, not the independent completeness, reliability, or construct validity of the elicited expert model.",
    "criticism": "The underlying study corpus was small and heterogeneous; methods bearing the same name could use different techniques, many reports omitted the CTA method and outcome reliability, and missing details were sometimes obtained by author contact. The pooled estimate should not be treated as validation of CTA as one uniform intervention.",
    "which_workflow": "Expert capture-to-transfer\u2014CTA-informed instruction.",
    "claimable_sentence": "CTA-derived instruction outperformed comparison instruction on average in a small heterogeneous literature, but effects differed markedly by elicitation method and context."
  },
  {
    "id": "EdwardsEtAl2021",
    "tier": "Tier 1",
    "full_citation": "Edwards, T. C., Coombs, A. W., Szyszka, B., Logishetty, K., & Cobb, J. P. (2021). Cognitive task analysis-based training in surgery: A meta-analysis. BJS Open, 5(6), zrab122.",
    "link": "https://doi.org/10.1093/bjsopen/zrab122",
    "peer_reviewed": "Yes",
    "population": "Twelve surgical-training studies: seven with surgical trainees, four with medical students, and one with a mixed population.",
    "design": "Systematic review of randomized and observational studies, with random-effects meta-analyses restricted to randomized trials reporting comparable procedural-knowledge or technical-performance outcomes.",
    "finding_verbatim": "CTA-based training is an effective way to learn the cognitive skills of a surgical procedure.",
    "effect_size": "Procedural knowledge among surgical trainees: SMD=1.36, 95% CI [0.67, 2.05], p<.001. Technical performance among trainees: SMD=2.06, 95% CI [1.17, 2.96], p<.001, I\u00b2=61%. Including the medical-student study: SMD=1.58, 95% CI [0.31, 2.85], p=.010; initial heterogeneity I\u00b2=87%.",
    "conditions_and_limits": "CTA-derived instruction addressed discrete invasive procedures and was assessed mainly on simulators, synthetic models, or structured performance ratings; some studies included real-patient performance. Results were sensitive to learner population and whether the CTA method was documented.",
    "criticism": "Only 12 heterogeneous surgical studies met criteria, pooled effects were very large and imprecise, protocols were generally unavailable, and selective outcome reporting was possible. Long-term retention, independent practice, and patient outcomes were not established.",
    "which_workflow": "Expert capture-to-transfer\u2014CTA-informed procedural instruction in surgery.",
    "claimable_sentence": "CTA-informed surgical instruction improved procedural knowledge and rated technical performance in small randomized-study syntheses, although heterogeneity and outcome-reporting concerns limit generalization."
  },
  {
    "id": "EricssonEtAl1993",
    "tier": "Contested",
    "full_citation": "Ericsson, K. A., Krampe, R. T., & Tesch-R\u00f6mer, C. (1993). The role of deliberate practice in the acquisition of expert performance. Psychological Review, 100(3), 363\u2013406.",
    "link": "https://doi.org/10.1037/0033-295X.100.3.363",
    "peer_reviewed": "Yes",
    "population": "Study 1 included 30 young West Berlin music-academy violinists\u201410 faculty-nominated best, 10 good, and 10 music-teacher-track students\u2014plus 10 middle-aged orchestra violinists. Study 2 compared 12 expert and 12 amateur pianists.",
    "design": "Observational expertise studies using structured retrospective practice-history interviews, current-week estimates, seven-day diaries, faculty-defined skill groups, and music-performance tasks; not an instructional experiment.",
    "finding_verbatim": "Hence, there is complete correspondence between the skill level of the groups and their average accumulation of practice time alone with the violin.",
    "effect_size": "At age 18, retrospective accumulated practice averaged 7,410 hours for best versus 5,301 for good violinists, F(1,27)=4.59, p<.05; the combined best-and-good average exceeded the music-teacher-track mean of 3,420 hours, F(1,27)=11.86, p<.01. Expert pianists reported 7,606 versus amateurs' 1,606 hours, F(1,22)=26.29, p<.001. Standardized effects and confidence intervals were not reported.",
    "conditions_and_limits": "The account assumes prolonged, effortful, goal-directed practice with appropriate difficulty, instruction, feedback, resources, and recovery in mature domains. The data establish associations among selected musicians, not that hours alone cause expertise or transfer across domains.",
    "criticism": "Samples were tiny and selected; practice histories were retrospective; current weekly estimates exceeded diaries by 5.2 hours; and the design was correlational. The best-versus-good analysis excluded a group while retaining full-sample denominator degrees of freedom. A preregistered replication did not reproduce monotonic correspondence. The paper did not test a universal 10,000-hour threshold or guarantee.",
    "which_workflow": "Deliberate practice and expertise acquisition\u2014foundational but contested evidence.",
    "claimable_sentence": "The seminal musician studies linked accumulated practice with performance level, but their small observational design does not support a universal hours threshold, causal guarantee, or claim that practice alone fully determines expert rank."
  },
  {
    "id": "MacnamaraEtAl2014",
    "tier": "Tier 1",
    "full_citation": "Macnamara, B. N., Hambrick, D. Z., & Oswald, F. L. (2014). Deliberate practice and performance in music, games, sports, education, and professions: A meta-analysis. Psychological Science, 25(8), 1608\u20131618.",
    "link": "https://doi.org/10.1177/0956797614535810",
    "peer_reviewed": "Yes",
    "population": "Eighty-eight studies contributing 157 practice\u2013performance correlations across games, music, sports, education, and professions.",
    "design": "Meta-analysis of associations between accumulated structured practice and attained performance; central estimates below are the corrected 2018 results.",
    "finding_verbatim": "We conclude that deliberate practice is important, but not as important as has been argued.",
    "effect_size": "Corrected mean r=0.38, 95% CI [0.33, 0.42]; variance explained 14%, 95% CI [11%, 18%]; I\u00b2=88.54%. Corrected domain estimates: games r=0.49, music r=0.48, sports r=0.45, education r=0.22, and professions r=0.09, p=.377; domain confidence intervals were not reported.",
    "conditions_and_limits": "Associations were larger in more predictable activities\u2014high predictability r=0.48, moderate r=0.37, low r=0.25\u2014and varied with how practice was measured. These are between-person observational associations, not randomized estimates of practice effects.",
    "criticism": "The 2018 corrigendum changed the main correlation from 0.35 to 0.38 and variance explained from 12% to 14% after correcting the adjustment for dependent outcomes. Heterogeneity was extreme, practice definitions varied, self-report and range restriction remain concerns, and Ericsson and Harwell argue that many included activities were structured practice rather than deliberate practice under the original narrower definition.",
    "which_workflow": "Deliberate practice and expertise acquisition\u2014cross-domain quantitative critique.",
    "claimable_sentence": "In corrected estimates, accumulated structured practice correlated moderately with performance overall, but the association was heterogeneous, weak and nonsignificant in professions, and not a causal estimate."
  },
  {
    "id": "EricssonHarwell2019",
    "tier": "Contested",
    "full_citation": "Ericsson, K. A., & Harwell, K. W. (2019). Deliberate practice and proposed limits on the effects of practice on the acquisition of expert performance: Why the original definition matters and recommendations for future research. Frontiers in Psychology, 10, 2396.",
    "link": "https://doi.org/10.3389/fpsyg.2019.02396",
    "peer_reviewed": "Yes",
    "population": "Reanalysis of Macnamara et al.'s 157 effects after applying three narrower criteria; 14 effects remained: games k=5, music k=3, sports k=5, education k=1, and professions k=0.",
    "design": "Critical narrative review, definitional reclassification, and random-effects reanalysis restricted to reproducibly superior performance and individualized purposeful or deliberate practice.",
    "finding_verbatim": "A reanalysis of the remaining effects estimated that accumulated duration of practice explained considerably more variance in performance.",
    "effect_size": "Accumulated purposeful or deliberate practice and performance: r=0.54, 95% CI [0.44, 0.63], p<.001, approximately 29% variance. The authors report 61% after correction for attenuation; a confidence interval for that assumption-dependent correction was not reported.",
    "conditions_and_limits": "The estimate applies only to a narrow definition requiring reproducibly superior performance, practice designed for the target performance, and solitary or individually supervised practice. No professional effect and only one education effect met those criteria.",
    "criticism": "The reanalysis retains a very small, selected, and nonrepresentative subset; classification was conducted by the theory's originator; and the evidence remains correlational. Narrowing the construct can improve internal coherence while reducing transportability and risking circular exclusion of disconfirming domains. Its key methodological contribution is showing how strongly results depend on definitions.",
    "which_workflow": "Deliberate practice and expertise acquisition\u2014definitional counteranalysis.",
    "claimable_sentence": "A narrow reanalysis found a stronger practice\u2013performance correlation, but retained no professional studies and only one education study, so it cannot establish the effect of deliberate practice in professional judgment."
  },
  {
    "id": "MacnamaraMaitra2019",
    "tier": "Tier 2",
    "full_citation": "Macnamara, B. N., & Maitra, M. (2019). The role of deliberate practice in expert performance: Revisiting Ericsson, Krampe & Tesch-R\u00f6mer (1993). Royal Society Open Science, 6, 190327.",
    "link": "https://doi.org/10.1098/rsos.190327",
    "peer_reviewed": "Yes",
    "population": "Thirty-nine violinists: 13 faculty-nominated best and 13 good students at the Cleveland Institute of Music, plus 13 less-accomplished students at affiliated Case Western Reserve University.",
    "design": "Preregistered, results-blind accepted, double-blind close replication using retrospective practice interviews and seven-day diaries, with an added measure of teacher-designed practice.",
    "finding_verbatim": "We did not replicate the core finding that accumulated amounts of deliberate practice corresponded to each skill level.",
    "effect_size": "Accumulated practice-alone group effect \u03b7\u00b2=0.26, 95% CI [0.03, 0.44], p=.001. Best M=8,224 hours, 95% CI [6,400, 10,048], versus good M=9,844, 95% CI [6,937, 12,751], d=-0.38, p=.364; no CI for d was reported. Good versus less-accomplished M=4,558, 95% CI [3,264, 5,851], d=1.33, p=.005; no CI for d was reported.",
    "conditions_and_limits": "Practice differentiated the less-accomplished group from the two more accomplished groups but did not order the best and good groups. Both best and good groups exceeded 10,000 accumulated hours on average by age 20.",
    "criticism": "The replication remained small, retrospective, and observational, and its less-accomplished group came from a different institution or department. Faculty nominations and group membership are imperfect performance measures. It directly challenges monotonic correspondence among top tiers, not the relevance of practice to becoming highly skilled.",
    "which_workflow": "Deliberate practice and expertise acquisition\u2014preregistered replication and boundary evidence.",
    "claimable_sentence": "A preregistered replication found much more practice among more- than less-accomplished violinists, but no evidence that accumulated hours distinguished the best from the good group."
  },
  {
    "id": "KleinEtAl1989",
    "tier": "Tier 3",
    "full_citation": "Klein, G. A., Calderwood, R., & MacGregor, D. (1989). Critical decision method for eliciting knowledge. IEEE Transactions on Systems, Man, and Cybernetics, 19(3), 462\u2013472.",
    "link": "https://doi.org/10.1109/21.31053",
    "peer_reviewed": "Yes",
    "population": "Experienced practitioners drawn from applied projects involving urban and wildland fire command, tank command, structural and design engineering, paramedicine, and computer programming; the methodological article does not report one pooled sample size.",
    "design": "Foundational methodological and case-based article describing the Critical Decision Method: multiple-pass retrospective interviews about personally experienced nonroutine incidents, beginning with chronology and decision points and then probing cues, goals, expectancies, options, discriminations, typicality, and counterfactuals.",
    "finding_verbatim": "The method is a variant of the critical incident technique extended to include probes",
    "effect_size": "No comparative effect size, confidence interval, or controlled validation statistic was reported.",
    "conditions_and_limits": "The method depends on a concrete consequential incident personally experienced by the interviewee, reconstruction of the event timeline before cognitive probes, and interviewers who avoid substituting generic doctrine for case-specific recall. It intentionally samples difficult or unusual cases rather than estimating the frequency of routine behavior.",
    "criticism": "The paper establishes a replicable elicitation procedure and illustrates useful outputs, but it does not test recall against contemporaneous process data, quantify completeness or inter-interviewer reliability, or show that knowledge from atypical incidents generalizes to ordinary cases.",
    "which_workflow": "Expert-judgment capture\u2014foundational lineage for eliciting decision points, cues, expectancies, options, and counterfactuals from critical incidents.",
    "claimable_sentence": "The Critical Decision Method provides a structured way to elicit decisions and cues from a specific experienced incident, but its foundational paper did not establish that retrospective accounts are complete, independently veridical, or representative of routine work."
  },
  {
    "id": "BurtonEtAl1990",
    "tier": "Tier 3",
    "full_citation": "Burton, A. M., Shadbolt, N. R., Rugg, G., & Hedgecock, A. P. (1990). The efficacy of knowledge elicitation techniques: A comparison across domains and levels of expertise. Knowledge Acquisition, 2(2), 167\u2013178.",
    "link": "https://doi.org/10.1016/S1042-8143(05)80010-X",
    "peer_reviewed": "Yes",
    "population": "Sixteen professional archaeologists: eight specialists in flint and eight in pottery classification.",
    "design": "Within-person comparison in which each archaeologist completed one traditional elicitation session, using either a structured interview or think-aloud protocol, and one contrived session, using either a laddered grid or card sort; outputs were converted to pseudo-English production-rule clauses.",
    "finding_verbatim": "Subjects entirely ignorant of a domain are able to construct plausible knowledge bases from common sense alone.",
    "effect_size": "Clauses judged true were 63% for interview, 46% for think-aloud protocol, 43% for card sort, and 55% for laddered grid. Think-aloud produced 32% garbled clauses and card sort 44% trivial clauses. First-pass rule overlap with the experts' later direct transcription was 54%, 25%, 29%, and 45%, respectively. No standardized effect sizes or confidence intervals were reported.",
    "conditions_and_limits": "The results concern archaeological classification and a particular conversion of elicited material into production rules. Method cells were extremely small, and the same experts contributed to the later comparison representation.",
    "criticism": "The study shows that plausible and abundant elicited content can be wrong, trivial, or garbled. It lacks an independent performance criterion, and representation and coding choices may explain some method differences.",
    "which_workflow": "Expert-judgment capture\u2014comparative knowledge-elicitation evidence from the expert-systems era.",
    "claimable_sentence": "A small comparative study found substantial differences in the truth, triviality, and garbling of knowledge produced by interviews, protocols, card sorts, and grids, demonstrating that plausibility or volume alone is not evidence of valid capture."
  },
  {
    "id": "MilitelloHutton1998",
    "tier": "Tier 2",
    "full_citation": "Militello, L. G., & Hutton, R. J. B. (1998). Applied cognitive task analysis (ACTA): A practitioner's toolkit for understanding cognitive task demands. Ergonomics, 41(11), 1618\u20131641.",
    "link": "https://doi.org/10.1080/001401398186108",
    "peer_reviewed": "Yes",
    "population": "Twenty-three graduate psychology students without prior domain, CTA, or instructional-design experience: 12 worked in firefighting and 11 in electronic warfare. Domain experts had at least 10 years of fire-command experience or at least six years of electronic-warfare experience.",
    "design": "Matched comparative evaluation of ACTA versus unstructured interviewing after common introductory training; ACTA participants received six additional hours of method training, conducted expert interviews, and produced cognitive-demands tables and training materials that independent subject-matter experts and cognitive psychologists rated.",
    "finding_verbatim": "ACTA techniques were found to be easy to use, flexible, and to provide clear output.",
    "effect_size": "For ACTA outputs in firefighting and electronic warfare, respectively, 92% and 94% of items were rated cognitive, 95% and 90% contained expert-only content, and 73% and 87% were relevant. Proposed manual modifications were rated accurate for 89% and 65% of items, and learning objectives for 92% and 54%. No standardized between-group effect sizes or confidence intervals were reported.",
    "conditions_and_limits": "ACTA combined a Task Diagram, Knowledge Audit, Simulation Interview, and Cognitive Demands Table. Its evaluation concerned usability, cognitive content, relevance, and SME-rated training utility after substantial method training; it did not compare elicited accounts with contemporaneous cognitive process data.",
    "criticism": "The groups were small and variable; the comparison group received some cognitively oriented training, reducing separation. Item-level overlap could not be scored without unacceptable inference, and some electronic-warfare SME agreement was poor, forcing reliance on one expert's ratings. The evaluation supports practicability more strongly than reliability or construct validity.",
    "which_workflow": "Expert-judgment capture\u2014streamlined CTA workflow producing difficult elements, cues, strategies, and common errors.",
    "claimable_sentence": "ACTA proved usable for trained novice analysts and produced SME-rated cognitive and relevant content, but its evaluation did not establish that the elicited model was complete, reproducible, or independently veridical."
  },
  {
    "id": "PhippsEtAl2011",
    "tier": "Tier 3",
    "full_citation": "Phipps, D. L., Meakin, G. H., & Beatty, P. C. W. (2011). Extending hierarchical task analysis to identify cognitive demands and information design requirements. Applied Ergonomics, 42(5), 741\u2013748.",
    "link": "https://doi.org/10.1016/j.apergo.2010.11.009",
    "peer_reviewed": "Yes",
    "population": "Two analysts independently classified steps in an existing hierarchical task analysis of the planning and delivery of anaesthesia; a third author, an anaesthetist, supplied domain content.",
    "design": "Inter-rater reliability evaluation of two extensions to hierarchical task analysis: the sub-goal template and the skill\u2013rule\u2013knowledge framework, applied to preoperative and perioperative anaesthesia task steps.",
    "finding_verbatim": "In quantitative terms, the two methods were found to have relatively poor inter-rater reliability",
    "effect_size": "Sub-goal-template Cohen's kappa was \u22120.017 for preoperative steps, 0.228 for perioperative steps, and 0.211 combined (N=144). Skill\u2013rule\u2013knowledge kappa was 0.273, 0.377, and 0.385 combined (N=166). Confidence intervals were not reported.",
    "conditions_and_limits": "Both analysts knew the frameworks but had little practical experience applying them. The sub-goal template was developed in process-control settings, and category interpretation required analyst judgment when transferred to anaesthesia.",
    "criticism": "The authors found useful qualitative insights despite low agreement, but inconsistent classification can produce different inferred cognitive demands and different design recommendations. Two analysts and one domain application provide a weak basis for generalizing reliability.",
    "which_workflow": "Expert-judgment capture\u2014direct evidence on analyst reliability when coding cognitive demands from task analysis.",
    "claimable_sentence": "Two cognitive extensions to anaesthesia task analysis generated useful hypotheses but poor analyst agreement, showing that a plausible representation can be unstable at the coding stage."
  },
  {
    "id": "SminkEtAl2012",
    "tier": "Tier 3",
    "full_citation": "Smink, D. S., Peyre, S. E., Soybel, D. I., Tavakkolizadeh, A., Vernon, A. H., & Anastakis, D. J. (2012). Utilization of a cognitive task analysis for laparoscopic appendectomy to identify differentiated intraoperative teaching objectives. American Journal of Surgery, 203(4), 540\u2013545.",
    "link": "https://doi.org/10.1016/j.amjsurg.2011.11.002",
    "peer_reviewed": "Yes",
    "population": "Three local expert surgeons experienced in laparoscopic appendectomy.",
    "design": "Critical-Decision-Method-based CTA interviews; transcripts were converted into individual cognitive-demands tables, member-checked, merged, and reviewed in a consensus meeting. The authors calculated expert agreement on operative steps and decision points and compared teaching priorities for junior and senior residents.",
    "finding_verbatim": "Of the 27 decision points, only 5 (19%) were identified by all 3 surgeon experts.",
    "effect_size": "All three experts identified 18 of 24 operative steps (75%) but only 5 of 27 decision points (19%). Individual coverage was 96%, 79%, and 83% for operative steps and 78%, 59%, and 48% for decision points. Experts selected nine operative steps and six decision points for junior residents versus four operative steps and 13 decision points for senior residents, p<.01. No standardized effect sizes or confidence intervals were reported.",
    "conditions_and_limits": "The procedure was narrowly bounded, each table was returned to its expert for completeness review, and disagreements were handled through a consensus meeting. The resulting master table reflects aggregation and consensus rather than a demonstrated population-complete model.",
    "criticism": "The very low three-expert overlap for decision points shows that decision capture is more expert-sample-dependent than action-step capture. With three experts, one procedure, and no independent performance criterion, consensus can conceal legitimate strategy variation as well as repair omissions.",
    "which_workflow": "Expert-judgment capture\u2014direct reliability evidence for action steps versus decision points.",
    "claimable_sentence": "Three surgeons converged on most operative steps but on only 19% of the combined decision points, indicating that decision-point capture is highly sensitive to which experts are interviewed."
  },
  {
    "id": "ClarkEtAl2012",
    "tier": "Tier 3",
    "full_citation": "Clark, R. E., Pugh, C. M., Yates, K. A., Inaba, K., Green, D. J., & Sullivan, M. E. (2012). The use of cognitive task analysis to improve instructional descriptions of procedures. Journal of Surgical Research, 173(1), e37\u2013e42.",
    "link": "https://doi.org/10.1016/j.jss.2011.09.003",
    "peer_reviewed": "Yes",
    "population": "Ten expert trauma surgeons: four gave unaided free-recall descriptions, five gave free recall with simulation materials, and one was interviewed with CTA; an eleventh vascular surgeon reviewed the final procedure representation.",
    "design": "Nonrandomized, highly unbalanced method comparison for descriptions of emergency femoral-artery shunt insertion. Interview products were corrected by participants, coded independently with reported inter-rater reliability of 0.87, aggregated into a criterion protocol, and compared for completeness and accuracy.",
    "finding_verbatim": "Surgeons in the unaided group omitted nearly 70% of necessary decision steps.",
    "effect_size": "Unaided descriptions contained approximately 25% of criterion content initially and 31.25% after participant correction, while the single CTA interview contained 68.75%. The authors reported one-sample tests across three dependent measures at p<.001, but no standardized effect size or confidence interval.",
    "conditions_and_limits": "The finding concerns one emergency surgical procedure and a criterion representation assembled from the same interview corpus. Simulation appeared to improve some free-recall descriptions, but differences were not statistically significant.",
    "criticism": "Only one surgeon received CTA, and that surgeon's elicited content contributed to the criterion against which the methods were scored, creating circularity. Groups were tiny and nonparallel, and there was no external observation or performance criterion. The widely repeated roughly 70% omission figure is not a cross-domain constant.",
    "which_workflow": "Expert-judgment capture\u2014comparison of CTA with unaided procedural recall and a key warning about the unsupported universalization of a local omission figure.",
    "claimable_sentence": "In one small and circular surgical comparison, a CTA interview produced a more complete description than unaided recall; the study does not support a general claim that experts omit 70% of decision knowledge."
  },
  {
    "id": "FoxEtAl2011",
    "tier": "Tier 1",
    "full_citation": "Fox, M. C., Ericsson, K. A., & Best, R. (2011). Do procedures for verbal reporting of thinking have to be reactive? A meta-analysis and recommendations for best reporting methods. Psychological Bulletin, 137(2), 316\u2013344.",
    "link": "https://doi.org/10.1037/a0021663",
    "peer_reviewed": "Yes",
    "population": "Nearly 3,500 participants across 94 studies comparing concurrent verbal reporting with a matched silent condition; constituent tasks and populations were diverse but predominantly laboratory-based and not restricted to experts.",
    "design": "Meta-analysis separating strict concurrent think-aloud from directed verbalization that asks participants to describe or explain selected information or actions.",
    "finding_verbatim": "the \u2018think-aloud\u2019 effect size is indistinguishable from zero (r = \u2212.03)",
    "effect_size": "Strict concurrent think-aloud versus silence: r=\u2212.03, 95% CI [\u2212.10, .03], Z=\u22120.94, p=.35. Directed description or explanation versus silence: r=.23, 95% CI [.14, .31], Z=5.05, p<.001. All verbal-reporting procedures tended to increase completion time.",
    "conditions_and_limits": "The average performance-nonreactivity result applies to strict, nondirective reporting of thoughts already in focal attention with a matched silent control. Requests for explanation, reasons, or particular kinds of information are a different and reactive procedure.",
    "criticism": "Average nonreactivity does not establish that verbal reports are complete, veridical, or exhaustive of nonverbal expertise. The evidence is mainly laboratory work, and a pooled null can coexist with reactivity for particular tasks such as insight problems.",
    "which_workflow": "Expert-judgment capture\u2014best quantitative evidence on when concurrent verbalization changes task performance.",
    "claimable_sentence": "Strict nondirective think-aloud did not change accuracy on average across 94 studies, whereas directed explanation did; neither result establishes that spoken content is a complete record of cognition."
  },
  {
    "id": "RussoEtAl1989",
    "tier": "Tier 2",
    "full_citation": "Russo, J. E., Johnson, E. J., & Stephens, D. L. (1989). The validity of verbal protocols. Memory & Cognition, 17(6), 759\u2013769.",
    "link": "https://doi.org/10.3758/BF03202637",
    "peer_reviewed": "Yes",
    "population": "Twenty-four paid student volunteers who each completed five approximately two-hour sessions.",
    "design": "Repeated-measures comparison of concurrent and several retrospective verbal-protocol procedures across five-letter anagrams, simple gamble choices, Raven matrices, and mental addition of three three-digit numbers.",
    "finding_verbatim": "Retrospective protocols yielded substantial forgetting or fabrication in all tasks",
    "effect_size": "Concurrent verbalization altered accuracy in two of four tasks, mental addition and gamble choice, and generally prolonged response times; retrospective reports showed omission or fabrication in every task. The paper did not report standardized effect sizes or confidence intervals for these findings.",
    "conditions_and_limits": "Reactivity varied by task, and retrospective validity depended on memory demands and the type of retrieval cue. The authors concluded that protocol validity should be checked empirically for the task rather than presumed from theory.",
    "criticism": "The sample was small, student-based, and tested artificial tasks rather than professional expertise. Fox et al. (2011) later found average nonreactivity for strict think-aloud, so Russo et al. are best read as evidence of task-specific failures rather than universal invalidity.",
    "which_workflow": "Expert-judgment capture\u2014direct tests of reactivity and retrospective omission or fabrication.",
    "claimable_sentence": "A small multi-task experiment found that concurrent reporting changed performance on some tasks and that retrospective reports omitted or fabricated information, supporting task-specific validity checks."
  },
  {
    "id": "BrunyeEtAl2023",
    "tier": "Tier 2",
    "full_citation": "Bruny\u00e9, T. T., Balla, A., Drew, T., Elmore, J. G., Kerr, K. F., Shucard, H., & Weaver, D. L. (2023). From image to diagnosis: Characterizing sources of error in histopathologic interpretation. Modern Pathology, 36(7), 100162.",
    "link": "https://doi.org/10.1016/j.modpat.2023.100162",
    "peer_reviewed": "Yes",
    "population": "Ninety pathology residents and attending pathologists recruited at nine major United States medical centers; each reviewed 14 digitized whole-slide breast-biopsy images.",
    "design": "Cross-sectional mixed-methods study using unrestricted slide viewing, participant-drawn regions of interest, open-ended feature descriptions, diagnoses, software traces, and eye tracking. Performance was compared with consensus diagnoses and consensus-defined critical regions; generalized estimating equations accounted for repeated cases within participants.",
    "finding_verbatim": "Failure to accurately describe features was the only factor strongly associated with an incorrect diagnosis.",
    "effect_size": "Both experience groups detected the critical region about 94% of the time. Trainees used incorrect feature terminology in 41% of cases versus 21% for attendings. In the covariate-adjusted model, accurate feature description was associated with diagnostic accuracy, OR=10.37, 95% CI [7.8, 13.8], p<.001; recognizing relevance was OR=1.21, 95% CI [0.91, 1.63], p=.19. Attendings had higher odds of accurate feature description than trainees, OR=2.96, 95% CI [1.8, 4.9], p<.001.",
    "conditions_and_limits": "The four-phase model separated detecting a critical region, recognizing its relevance, describing its features, and deciding on a diagnosis. Results apply to 14 digitized breast-biopsy cases reviewed with zoom and pan available and scored against expert consensus.",
    "criticism": "Cross-sectional experience comparison cannot identify how expertise developed. The study used a small standardized case set in an experimental setting; feature-description scoring was subjective, 136 of 1,211 annotations were excluded for ambiguity or disagreement, and a perfect pathology gold standard does not exist. Expert-rater agreement improved through discussion, so final agreement should not be treated as independent reliability.",
    "which_workflow": "Expert-judgment capture\u2014direct evidence for separating detection, relevance, feature description, and diagnostic decision points.",
    "claimable_sentence": "In one multisite breast-pathology study, residents and attendings usually looked at the critical region, but the groups differed in how accurately they described its features; looking at a cue was not equivalent to interpreting it correctly."
  },
  {
    "id": "MacnamaraEtAl2018Corrigendum",
    "tier": "Tier 1",
    "full_citation": "Macnamara, B. N., Hambrick, D. Z., & Oswald, F. L. (2018). Corrigendum: Deliberate practice and performance in music, games, sports, education, and professions: A meta-analysis. Psychological Science, 29(7), 1202\u20131204.",
    "link": "https://doi.org/10.1177/0956797618769891",
    "peer_reviewed": "Yes\u2014published correction",
    "population": "The correction applies to the 2014 meta-analysis of 88 studies and 157 correlations across games, music, sports, education, and professions.",
    "design": "Published corrigendum recalculating dependency-adjusted meta-analytic estimates after identifying an error in the original adjustment method.",
    "finding_verbatim": "Results Being Corrected",
    "effect_size": "Corrected mean correlation r=.38, 95% CI [.33, .42], replacing r=.35, 95% CI [.30, .39]. Corrected explained variance was 14%, 95% CI [11%, 18%], replacing 12%, 95% CI [9%, 15%]. Corrected heterogeneity was I\u00b2=88.54.",
    "conditions_and_limits": "The corrected estimates retain the original study-selection and construct definitions. They describe observational associations between accumulated practice and performance, not an experimentally assigned practice effect.",
    "criticism": "The correction materially changes headline estimates and demonstrates why the original numbers should not be quoted. It does not resolve high heterogeneity, self-report and range-restriction concerns, or the dispute over which activities qualify as deliberate practice.",
    "which_workflow": "Expertise transfer\u2014required correction to the deliberate-practice evidence base.",
    "claimable_sentence": "The published correction reports an average practice\u2013performance correlation of .38, not .35, with high heterogeneity; the estimate is observational and does not establish a universal practice dose."
  }
]
