[
  {
    "id": "AndreoniPetrie2004",
    "tier": "Tier 3",
    "full_citation": "Andreoni, J., & Petrie, R. (2004). Public goods experiments without confidentiality: A glimpse into fund-raising. Journal of Public Economics, 88(7–8), 1605–1623.",
    "link": "https://doi.org/10.1016/S0047-2727(03)00040-9",
    "peer_reviewed": true,
    "population": "200 mostly undergraduate participants; mean age 20.22 years (SD=2.13), 41% women.",
    "design": "Laboratory experiment using repeated five-person linear public-goods games; factorial manipulation of photographs/identity and information about each person's contribution.",
    "finding_verbatim": "Adding just information on generosity has no significant effect … and neither does adding just the identity.",
    "effect_size": "Mean endowment contributed: baseline 30.3%, contribution-information only 26.9%, photographs only 39.5%, and both 48.1%. Both versus photographs was 21.7% higher, Wilcoxon p=.0423; both versus information was 78.8% higher, p=.0001. No standardized effect size or numeric confidence interval was reported.",
    "conditions_and_limits": "Identity had to be linkable to individual conduct, permitting pride, shame, status and social comparison. The task was incentivized, repeated and conducted in small groups.",
    "criticism": "A monetary laboratory game is not an academic group project; repeated observations and group membership create dependence; learning and contribution quality were not measured. The study tests socially attributable giving, not a neutral activity trace.",
    "which_workflow": "Group work/accountability—indirect cross-domain mechanism evidence.",
    "claimable_sentence": "In a public-goods experiment, neither contribution information nor identity alone significantly changed giving, whereas linking the two did."
  },
  {
    "id": "ApuglieseLewis2017",
    "tier": "Tier 2",
    "full_citation": "Apugliese, A., & Lewis, S. E. (2017). Impact of instructional decisions on the effectiveness of cooperative learning in chemistry through meta-analysis. Chemistry Education Research and Practice, 18(1), 271–278.",
    "link": "https://doi.org/10.1039/C6RP00195E",
    "peer_reviewed": true,
    "population": "25 independent samples from 24 chemistry studies; 16 college-level and eight high-school articles.",
    "design": "Random-effects meta-analysis of experimental and quasi-experimental cooperative-learning comparisons with traditional instruction.",
    "finding_verbatim": "The lack of impact of CL on cumulative exams does give pause.",
    "effect_size": "Overall Hedges' g=0.586, 95% CI [0.339, 0.834]. Cumulative assessments: g=-0.088, 95% CI [-0.479, 0.392], k=6; single-topic assessments: g=1.12, 95% CI [0.78, 1.45], k=13; between-category Q=16.3, p<.01. Sensitivity analysis with nine cumulative samples: g=-0.020, 95% CI [-0.414, 0.372].",
    "conditions_and_limits": "Effects were examined by assessment scope, frequency of use, group size and assessment openness. The review did not code whether individual accountability was actually implemented.",
    "criticism": "School and college populations were mixed; moderator cells were small and observational across studies; search terms omitted several named cooperative-learning variants. Positive single-topic outcomes did not extend to cumulative assessments.",
    "which_workflow": "Group work/shared team model—direct for chemistry learning, indirect for contribution.",
    "claimable_sentence": "A chemistry meta-analysis found a positive overall cooperative-learning effect but no detectable advantage on cumulative assessments."
  },
  {
    "id": "Biesma2019",
    "tier": "Tier 2",
    "full_citation": "Biesma, R., Kennedy, M.-C., Pawlikowska, T., Brugha, R., Conroy, R., & Doyle, F. (2019). Peer assessment to improve medical student’s contributions to team-based projects: Randomised controlled trial and qualitative follow-up. BMC Medical Education, 19, 371.",
    "link": "https://doi.org/10.1186/s12909-019-1783-8",
    "peer_reviewed": true,
    "population": "223 second-year medical students in 37 teams at the Royal College of Surgeons in Ireland.",
    "design": "Cluster randomized controlled trial with qualitative focus-group follow-up; 19 intervention teams and 18 control teams; CATME baseline at week 2 and follow-up at week 10.",
    "finding_verbatim": "There was no difference in team contribution, and other forms of team effectiveness.",
    "effect_size": "Primary contribution outcome β=0.76, 95% CI [−0.58, 2.09], p=.26; all CATME secondary outcomes were nonsignificant.",
    "conditions_and_limits": "Intervention teams allocated a fixed pool of peer marks openly and by consensus during a 10-week project; six focus groups examined implementation. Students had no prior peer-assessment experience in the medical curriculum.",
    "criticism": "Most intervention teams did not implement differentiation as intended because of relationship concerns, reciprocity fears, and objections to zero-sum marking. The per-protocol analysis and single culturally diverse medical school limit generalization, but non-use is part of the causal result for a socially demanding intervention.",
    "which_workflow": "Group work/accountability—direct negative classroom evidence.",
    "claimable_sentence": "A cluster trial found that transparent, consequential peer marking did not improve team contribution because students largely resisted differentiating their peers."
  },
  {
    "id": "Black2021",
    "tier": "Tier 2",
    "full_citation": "Black, E. W., Dickson, T., & Blue, A. V. (2021). Exploring item discrimination in an online self and peer assessment of interprofessional teamwork. Journal of Interprofessional Education & Practice, 22, 100396.",
    "link": "https://doi.org/10.1016/j.xjep.2020.100396",
    "peer_reviewed": true,
    "population": "2,731 students in a longitudinal interprofessional service-learning program, 2014–2017.",
    "design": "Secondary psychometric analysis of CATME self- and peer-rating data collected twice over nine months; generalized partial-credit item-response model.",
    "finding_verbatim": "Items could not discriminate between individuals at or above the estimated population mean.",
    "effect_size": "Classical reliability ranged from 0.84 to 0.95; DETECT indices exceeded 1.0, indicating high multidimensionality; all six modeled items showed misfit. No effect-size confidence interval was reported.",
    "conditions_and_limits": "Only three CATME competencies were represented as six self/peer items in one institutional interprofessional program.",
    "criticism": "This is not a test of the complete five-dimension CATME instrument and has no independent behavioral criterion. High reliability did not imply good item fit or discrimination.",
    "which_workflow": "Group work/accountability—direct instrument-validity evidence.",
    "claimable_sentence": "An IRT analysis found that the studied CATME items identified low teamwork ratings but did not discriminate at or above the sample mean."
  },
  {
    "id": "BrooksAmmons2003",
    "tier": "Tier 3",
    "full_citation": "Brooks, C. M., & Ammons, J. L. (2003). Free riding in group projects and the effects of timing, frequency, and specificity of criteria in peer assessments. Journal of Education for Business, 78(5), 268–272.",
    "link": "https://doi.org/10.1080/08832320309598613",
    "peer_reviewed": true,
    "population": "340 students in 60 teams in a first-semester interdisciplinary business course; 330 completed the end-course questionnaire.",
    "design": "Uncontrolled repeated peer-evaluation study across three four-week projects; consequential, fixed-total ratings multiplied the team project grade.",
    "finding_verbatim": "The variance in group evaluation scores should decrease over time.",
    "effect_size": "Peer-rating variance declined from 140.249 at administration one to 78.023 at administration two, Levene statistic=20.894, p<.001, and was 78.781 at administration three; administration two versus three p=.760. No standardized effect size or confidence interval was reported.",
    "conditions_and_limits": "Evaluations were early, repeated, behaviorally specific, written and grade-consequential; raters divided a fixed total averaging 100 points per member.",
    "criticism": "Contribution was not observed and there was no untreated comparison. Lower rating dispersion can reflect equalized work, leniency, collusion or learning not to differentiate; it is not itself evidence that free riding declined.",
    "which_workflow": "Group work/accountability—direct but weak peer-assessment evidence.",
    "claimable_sentence": "Repeated consequential peer evaluation compressed rating variance; the study did not establish a change in contribution behavior."
  },
  {
    "id": "Colliver2003",
    "tier": "Tier 2",
    "full_citation": "Colliver, J. A., Feltovich, P. J., & Verhulst, S. J. (2003). Small group learning in medical education: A second look at the Springer, Stanne, and Donovan meta-analysis. Teaching and Learning in Medicine, 15(1), 2–5.",
    "link": "https://doi.org/10.1207/S15328015TLM1501_01",
    "peer_reviewed": true,
    "population": "The subset of studies in Springer et al. (1999) considered relevant to medical education, including nine randomized studies.",
    "design": "Critical reanalysis and close reading; not a new intervention or new meta-analysis.",
    "finding_verbatim": "All in all, the evidence is not convincing.",
    "effect_size": "No new pooled effect size or confidence interval was reported. Of nine randomized studies, four used conventional small-group learning; one was judged uninterpretable and the remaining three yielded one null, one negative and one positive result.",
    "conditions_and_limits": "The critique is directed particularly at causal claims for conventional small-group learning in medical education.",
    "criticism": "A selective domain-specific reappraisal cannot invalidate the full Springer synthesis or later active-learning evidence, but it materially weakens strong causal readings of the randomized subset.",
    "which_workflow": "Group work/shared team model—direct critical evidence.",
    "claimable_sentence": "A medical-education reappraisal found the randomized subset behind the Springer synthesis too inconsistent for a strong causal conclusion."
  },
  {
    "id": "deJong2010",
    "tier": "Tier 2",
    "full_citation": "de Jong, Z., van Nies, J. A. B., Peters, S. W. M., Vink, S., Dekker, F. W., & Scherpbier, A. (2010). Interactive seminars or small group tutorials in preclinical medical education: Results of a randomized controlled trial. BMC Medical Education, 10, 79.",
    "link": "https://doi.org/10.1186/1472-6920-10-79",
    "peer_reviewed": true,
    "population": "107 third-year Leiden medical students randomized; 96 randomized participants completed the examination.",
    "design": "Randomized comparison of tutorials of approximately 15 students with interactive seminars of 50–60 students.",
    "finding_verbatim": "Small group tutorials leads to greater satisfaction but not to better learning results.",
    "effect_size": "Mean examination grade was 6.6 in both conditions; 42/48 tutorial students and 41/48 seminar students passed. Satisfaction was 86% versus 39%, p<.001. No standardized learning effect or confidence interval was reported.",
    "conditions_and_limits": "Both conditions were interactive and covered one preclinical medical course; the intervention primarily isolated group size rather than interaction versus lecture.",
    "criticism": "Volunteer randomized sample at one institution; learning measure may not capture all benefits. The study nevertheless separates satisfaction from achievement and shows that smaller groups alone are insufficient.",
    "which_workflow": "Group work/shared team model—direct boundary-condition evidence.",
    "claimable_sentence": "In one randomized medical course, smaller tutorials increased satisfaction but did not improve examination performance over large interactive seminars."
  },
  {
    "id": "Freeman2014",
    "tier": "Tier 2",
    "full_citation": "Freeman, S., Eddy, S. L., McDonough, M., Smith, M. K., Okoroafor, N., Jordt, H., & Wenderoth, M. P. (2014). Active learning increases student performance in science, engineering, and mathematics. Proceedings of the National Academy of Sciences, 111(23), 8410–8415.",
    "link": "https://doi.org/10.1073/pnas.1319030111",
    "peer_reviewed": true,
    "population": "225 undergraduate STEM studies; 158 contributed examination or concept-inventory effects and 67 contributed failure-rate comparisons.",
    "design": "Meta-analysis comparing broadly defined active learning with traditional lecturing in undergraduate STEM.",
    "finding_verbatim": "Student performance on examinations and concept inventories increased by 0.47 SDs under active learning.",
    "effect_size": "Overall examination/concept-inventory difference=0.47 SD; odds of failure under traditional lecture OR=1.95. The randomized/crossover subgroup estimate was g=0.514, 95% CI [0.322, 0.706]. Raw mean failure rates were 33.8% under lecture and 21.8% under active learning, but the failure subset contained no randomized designs.",
    "conditions_and_limits": "Active learning included heterogeneous practices such as peer instruction, group problem solving, worksheets, tutorials, workshops, and studio formats; comparisons were at the course-section level.",
    "criticism": "The synthesis supports broad active learning, not group work, visibility, accountability, or a specific collaborative design. Interventions and study quality varied, many studies were not randomized, and the failure analysis cannot support the same causal reading as the randomized/crossover performance subgroup.",
    "which_workflow": "Group work/shared team model—broad indirect learning evidence.",
    "claimable_sentence": "Across undergraduate STEM studies, broadly defined active-learning sections produced higher assessment performance and lower observed failure rates than lecture, without isolating group work or accountability."
  },
  {
    "id": "Hoenow2025",
    "tier": "Tier 3",
    "full_citation": "Hoenow, N. C. (2025). Disclosing group members’ identities reduces cooperation in an artefactual public goods field experiment. Human Nature, 36(3), 337–359.",
    "link": "https://doi.org/10.1007/s12110-025-09508-7",
    "peer_reviewed": true,
    "population": "144 rural Namibian villagers from 10 villages; 53% women, mean age 35.85 years.",
    "design": "Randomized artefactual field experiment; single-round four-person public-goods game with disclosed versus undisclosed group identities while individual contributions remained private.",
    "finding_verbatim": "Contributions to the public good are significantly higher when group members cannot identify one another.",
    "effect_size": "Full sample means were 4.44 coins (SD=3.67) without disclosure and 3.31 (SD=3.66) with disclosure; mean difference=-1.14, SE=0.61, p=.065. Comprehension-passing subset n=114: 4.70 versus 2.92; difference=-1.78, SE=0.68, p=.009. Adjusted ordered-logit identification OR=0.269, p=.03; no numeric confidence interval was reported.",
    "conditions_and_limits": "Only partner identities, not individual behavior, were disclosed; communication was prohibited and participants had pre-existing village relationships. Social closeness predicted more giving within the identified condition.",
    "criticism": "The primary full-sample mean test was underpowered and nonsignificant at .05; the comprehension subset and relational mechanisms are post hoc or exploratory. Strong village clustering and a field context limit transfer to higher education.",
    "which_workflow": "Group work/accountability—indirect counterevidence on identity visibility.",
    "claimable_sentence": "In one randomized field setting, revealing group identities reduced rather than increased cooperation, especially among socially distant members."
  },
  {
    "id": "Kalaian2018",
    "tier": "Tier 2",
    "full_citation": "Kalaian, S. A., Kasim, R. M., & Nims, J. K. (2018). Effectiveness of small-group learning pedagogies in engineering and technology education: A meta-analysis. Journal of Technology Education, 29(2), 20–35.",
    "link": "https://doi.org/10.21061/jte.v29i2.a.2",
    "peer_reviewed": true,
    "population": "Undergraduate engineering and technology classrooms; 18 primary studies published from 1995 to 2010, yielding 26 independent samples/effects.",
    "design": "Random-effects meta-analysis of cooperative, collaborative, problem-based and peer-led team learning versus lecture-based or individualized instruction.",
    "finding_verbatim": "Most of the primary studies supported the effectiveness of the small-group learning methods.",
    "effect_size": "Overall weighted estimate d=0.449, 95% CI [0.278, 0.620], p<.001. Effects were heterogeneous, Q=115.81, p<.001, and individual estimates ranged from -0.284 to 1.399.",
    "conditions_and_limits": "Domain was undergraduate engineering/technology; included published articles and dissertations, varied research designs, four pedagogical families and different grouping practices.",
    "criticism": "Only 18 studies; interventions and designs were heterogeneous; most diversity characteristics were unreported; study-level moderator comparisons were not randomized. The analysis does not isolate individual accountability or contribution behavior.",
    "which_workflow": "Group work/shared team model—direct outcome evidence.",
    "claimable_sentence": "A meta-analysis found a positive average achievement effect for small-group learning in undergraduate engineering and technology, with substantial heterogeneity."
  },
  {
    "id": "KarauWilliams1993",
    "tier": "Tier 2",
    "full_citation": "Karau, S. J., & Williams, K. D. (1993). Social loafing: A meta-analytic review and theoretical integration. Journal of Personality and Social Psychology, 65(4), 681–706.",
    "link": "https://doi.org/10.1037/0022-3514.65.4.681",
    "peer_reviewed": true,
    "population": "78 studies/documents represented in 163 objective comparison units for the main analysis; predominantly laboratory, Western and student samples.",
    "design": "Meta-analysis comparing effort in collective with coactive/individual conditions and testing motivational moderators.",
    "finding_verbatim": "Individuals tended to engage in social loafing, producing lower effort levels in the collective condition than in the coactive condition.",
    "effect_size": "Overall d=0.44, 95% CI [0.39, 0.48], 163 units, Q=964.70, p<.001. After excluding 64 outlier units: d=0.24, 95% CI [0.19, 0.29]. Evaluation only in the coactive condition: d=0.59, 95% CI [0.55, 0.64], n=115; evaluation in both conditions: d=0.08, 95% CI [-0.01, 0.17], n=27; no evaluation: d=-0.12, 95% CI [-0.33, 0.08], n=5.",
    "conditions_and_limits": "Evaluation potential, meaningful tasks, nonredundant inputs, expectations about co-workers, culture and established relationships moderated effort losses.",
    "criticism": "Extreme heterogeneity, many partitioned/nonindependent comparison units, heavy reliance on brief artificial tasks, substantial outlier exclusion and older samples limit transfer to semester-long undergraduate projects.",
    "which_workflow": "Group work/accountability—indirect mechanism synthesis.",
    "claimable_sentence": "Social loafing was conditional: the average loss was small and statistically indistinguishable from zero when collective and individual effort were equally evaluable."
  },
  {
    "id": "Linton2014",
    "tier": "Tier 2",
    "full_citation": "Linton, D. L., Pangle, W. M., Wyatt, K. H., Powell, K. N., & Sherwood, R. E. (2014). Identifying key features of effective active learning: The effects of writing and peer discussion. CBE—Life Sciences Education, 13(3), 469–477.",
    "link": "https://doi.org/10.1187/cbe.13-12-0242",
    "peer_reviewed": true,
    "population": "346 students in three sections of a majors' introductory biology course.",
    "design": "Within-course rotation of 10 concept activities across discussion-only, individual-writing-only, and discussion-plus-writing conditions and three instructors.",
    "finding_verbatim": "Enforcing individual writing requires each student to explain his or her thinking.",
    "effect_size": "Immediate clicker improvement did not differ, χ²(2)=2.8, p=.24. Written examination performance differed by treatment, χ²(2)=7.2, p=.027; both writing conditions exceeded discussion only. Instructor effect χ²(2)=19.3, p<.0001; treatment-by-instructor interaction χ²(4)=78.1, p<.0001. No standardized effect size or confidence interval was reported.",
    "conditions_and_limits": "Attributable individual writing occurred before or alongside peer discussion; outcomes were written answers on course assessments.",
    "criticism": "A large instructor/concept interaction limits generalization. Better written answers may reflect retention or written communication rather than deeper conceptual understanding. Contribution behavior was not measured.",
    "which_workflow": "Group work/accountability—direct evidence for individual cognitive processing, not contribution.",
    "claimable_sentence": "Requiring individual writing alongside active learning improved later written responses in one biology course, with strong instructor dependence."
  },
  {
    "id": "LountWilk2014",
    "tier": "Tier 3",
    "full_citation": "Lount, R. B., Jr., & Wilk, S. L. (2014). Working harder or hardly working? Posting performance eliminates social loafing and promotes social laboring in workgroups. Management Science, 60(5), 1098–1106.",
    "link": "https://doi.org/10.1287/mnsc.2013.1820",
    "peer_reviewed": true,
    "population": "21 employees in one call center; 90% women; 737 employee-day/project observations.",
    "design": "Longitudinal workplace field study with six weeks of weekly public named/ranked performance posting followed by six weeks without posting; temporary solo/team assignment was largely random but phase order was fixed.",
    "finding_verbatim": "When individual performance was publicly posted in the workplace, employees working in a group performed better than when working alone.",
    "effect_size": "Posting-by-group-work hierarchical-model coefficient b=0.042, t=4.20, p<.001. Under posting, group versus solo b=0.019, t=2.18, p=.03; without posting b=-0.023, t=-4.33, p<.001. No numeric confidence intervals were reported.",
    "conditions_and_limits": "Performance was named, ranked, repeatedly publicized and embedded in a management/evaluation context.",
    "criticism": "Very small employee sample, one workplace, fixed posting-first sequence, low-interdependence teams and an unmeasured mechanism. Posting combined visibility, comparison, ranking and reputational pressure.",
    "which_workflow": "Group work/accountability—indirect workplace visibility evidence.",
    "claimable_sentence": "Named public rankings were associated with greater individual effort during group work in one small call-center study."
  },
  {
    "id": "Magin2001",
    "tier": "Tier 3",
    "full_citation": "Magin, D. (2001). Reciprocity as a source of bias in multiple peer assessment of group work. Studies in Higher Education, 26(1), 53–63.",
    "link": "https://doi.org/10.1080/03075070020030715",
    "peer_reviewed": true,
    "population": "169 medical students in 16 groups of nine to eleven completing a behavioral/community-medicine report.",
    "design": "Observational analysis of reciprocal rater–ratee correlations in peer ratings of contribution to discussion and group development.",
    "finding_verbatim": "Accounting for only 1% of the variance.",
    "effect_size": "Group correlations ranged approximately from -0.02 to 0.31; Fisher-transformed mean r=0.11, 95% CI [0.07, 0.15], corresponding to about 1% explained variance.",
    "conditions_and_limits": "Students worked in unusually large groups and each member received multiple peer ratings.",
    "criticism": "The method excludes raters who give every teammate the same score and detects symmetric association, a narrower construct than fear of retaliation, friendship leniency or prearranged equality. It does not show that relational fears are absent.",
    "which_workflow": "Group work/accountability—direct evidence on one rating-bias mechanism.",
    "claimable_sentence": "Measured reciprocity explained little rating variance in one medical-student peer-assessment setting, despite the plausibility of reciprocity fears."
  },
  {
    "id": "Meijer2022",
    "tier": "Tier 2",
    "full_citation": "Meijer, H., Brouwer, J., Hoekstra, R., & Strijbos, J.-W. (2022). Exploring construct and consequential validity of collaborative learning assessment in higher education. Small Group Research, 53(6), 891–925.",
    "link": "https://doi.org/10.1177/10464964221095545",
    "peer_reviewed": true,
    "population": "Two Dutch final-year undergraduate teacher-training cohorts: n=28 in 11 groups and n=42 in 14 triads.",
    "design": "Observational comparison across an eight-week design course of a group assignment, a near-identical attributable individual component, an independent open-book essay examination and simulated combined grades.",
    "finding_verbatim": "Validity can vary widely within and across cohorts.",
    "effect_size": "Group grade versus individual examination: r=0.20, p=.301, and r=0.15, p=.343. Group grade versus near-identical individual part: r=0.33, p=.083, and r=0.74, p<.001. No numeric confidence intervals were reported. Lower individual performers gained 25.94% and 12.79% relative to their individual grade from a group grade; higher performers lost 12.41% and 8.47%.",
    "conditions_and_limits": "The design included a separately attributable individual response and an independent exam, allowing product and individual evidence to be compared.",
    "criticism": "Very small single-course cohorts, self-selected teams, COVID/online disruption, nonrandom design and noisy grades. Some combined-grade correlations are mechanically induced.",
    "which_workflow": "Group work/assessment—direct validity evidence.",
    "claimable_sentence": "In two small cohorts, a common group grade related weakly to an independent individual exam and redistributed marks across individual performance levels."
  },
  {
    "id": "Ohland2012",
    "tier": "Tier 2",
    "full_citation": "Ohland, M. W., Loughry, M. L., Woehr, D. J., Bullard, L. G., Felder, R. M., Finelli, C. J., Layton, R. A., Pomeranz, H. R., & Schmucker, D. G. (2012). The Comprehensive Assessment of Team Member Effectiveness: Development of a behaviorally anchored rating scale for self- and peer evaluation. Academy of Management Learning & Education, 11(4), 609–630.",
    "link": "https://doi.org/10.5465/amle.2010.0177",
    "peer_reviewed": true,
    "population": "Three student studies: n=86 in a crossover comparison; n=104, with 98 analyzed, in convergent/criterion analyses; and 570 students in 113 teams, with 358 retained for analyses requiring three peer raters.",
    "design": "Multistudy development and validation of CATME behaviorally anchored rating scales for five dimensions of team-member effectiveness.",
    "finding_verbatim": "Training will not motivate students to rate accurately.",
    "effect_size": "Generalizability coefficients were approximately 0.70–0.90 and absolute-agreement coefficients approximately 0.44–0.82 across dimensions. Composite α=0.94; convergence with another teamwork measure r=0.64; association with course points r=0.51; five dimensions explained R²=0.58 of desire to work together again. No numeric confidence intervals were reported.",
    "conditions_and_limits": "Behaviorally specific anchors, multiple peer raters and rater practice/training supported the instrument; the intended construct was perceived teamwork behavior.",
    "criticism": "Ratings clustered near 4.2/5 and did not use the full scale. Rater and halo effects remained plausible. External criteria were course points, liking and future-work preference, not independently observed contribution or individual learning.",
    "which_workflow": "Group work/accountability—direct instrument-validity evidence.",
    "claimable_sentence": "CATME BARS structures perceived teamwork judgments with useful but variable agreement; it does not validate honest consequential use or individual learning inference."
  },
  {
    "id": "ONeill2020",
    "tier": "Tier 2",
    "full_citation": "O’Neill, T. A., Boyce, M., & McLarnon, M. J. W. (2020). Team health and project quality are improved when peer evaluation scores affect grades on team projects. Frontiers in Education, 5, 49.",
    "link": "https://doi.org/10.3389/feduc.2020.00049",
    "peer_reviewed": true,
    "population": "873 introductory-psychology students in 162 self-selected teams across three successive fall cohorts.",
    "design": "ABA cohort comparison: completion-only peer evaluation, grade-consequential peer evaluation, then return to completion-only evaluation.",
    "finding_verbatim": "It is not possible to rule out changes … that may be extraneous.",
    "effect_size": "Graded cohort versus the two baseline cohorts: peer-rating b=0.69 and 0.35; team-health b=0.48 and 0.28; project-grade b=8.87 and 5.53; all p<.01. Graded-cohort means were 4.70 for peer ratings, 4.47 for team health and 86.01 for project grade. Figures show 95% confidence intervals, but numeric limits and standardized effects were not reported.",
    "conditions_and_limits": "Private behavioral ratings, criteria disclosed at the outset and a 4% course-grade consequence for the mean peer rating.",
    "criticism": "Cohorts were not randomized; timing, assignment clarity and teaching assistants differed. The rating used for grading was also the behavioral outcome, so strategic generosity or rating inflation can imitate behavior change. Accountability and social loafing were not directly measured.",
    "which_workflow": "Group work/accountability—direct but quasi-experimental evidence.",
    "claimable_sentence": "A grade-contingent peer-rating cohort had higher ratings and project grades, but the design cannot establish that contribution behavior changed."
  },
  {
    "id": "Panadero2013",
    "tier": "Tier 3",
    "full_citation": "Panadero, E., Romero, M., & Strijbos, J.-W. (2013). The impact of a rubric and friendship on peer assessment: Effects on construct validity, performance, and perceptions of fairness and comfort. Studies in Educational Evaluation, 39(4), 195–203.",
    "link": "https://doi.org/10.1016/j.stueduc.2013.10.005",
    "peer_reviewed": true,
    "population": "209 third-year Spanish preservice teachers; rubric n=104 and criteria-only n=105.",
    "design": "Quasi-experimental class-level comparison; each student rated one known peer's concept map against an expert score.",
    "finding_verbatim": "Amplify – or make more visible – potential friendship bias.",
    "effect_size": "Rubric reduced peer–expert deviation: M=0.84 versus 2.33 on a 20-point scale, F(1,203)=5.66, p=.018, η²=.03. Friendship main effect η²=.016 and rubric-by-friendship interaction η²=.018 were nonsignificant. Within-rubric exploratory high-friendship contrast d=.98, p=.022; no confidence interval was reported.",
    "conditions_and_limits": "Students rated a familiar peer's product in a low-stakes preservice-teacher task; a rubric or criteria list structured judgment.",
    "criticism": "All groups over-scored. Only 14 dyads were classified high-friendship, and the prespecified interaction was null; class assignment, teacher differences and one-item friendship measurement limit causal interpretation. The task was not teammate-contribution rating.",
    "which_workflow": "Group work/accountability—indirect evidence on friendship and rubric effects.",
    "claimable_sentence": "A rubric improved average peer–expert agreement but did not eliminate over-scoring; the friendship moderation signal was exploratory."
  },
  {
    "id": "RieglerGuest2026",
    "tier": "Tier 3",
    "full_citation": "Riegler, R., & Guest, J. (2026). Does widespread collusion undermine the case for using peer-assessment schemes with assessed group work? Studies in Higher Education, 51(2), 295–308.",
    "link": "https://doi.org/10.1080/03075079.2025.2465687",
    "peer_reviewed": true,
    "population": "First-year applied-economics module; 179 students completed the project, with 411 unique dyads across 31 randomly assigned teams retained after rating-data exclusions.",
    "design": "Observational pattern-detection analysis of formative and summative fixed-point peer ratings; other peers' ratings served as a proxy for actual contribution.",
    "finding_verbatim": "All the estimates are upper limits.",
    "effect_size": "Threshold-sensitive potential-collusion estimates were 3%–9% of dyads and 19%–55% of teams containing at least one flagged dyad. Fifty-five percent of students equal-scored every teammate; seven of 31 teams unanimously reported equal contribution. No standardized effect or confidence interval was reported.",
    "conditions_and_limits": "Students allocated 100 effort points among teammates after submission but before the group mark was known; thresholds for mutual high scoring determined the estimates.",
    "criticism": "No agreement to collude or actual contribution was observed. Proxy peer ratings can carry the same biases as the tested ratings, and results vary materially by threshold. One module limits prevalence claims.",
    "which_workflow": "Group work/accountability—direct but inferential evidence on potential collusion.",
    "claimable_sentence": "Potential collusion was uncommon by dyad but threshold-sensitive and distributed across a meaningful share of teams; the study did not observe collusion itself."
  },
  {
    "id": "Schurmann2024",
    "tier": "Tier 2",
    "full_citation": "Schürmann, V., Marquardt, N., & Bodemer, D. (2024). Conceptualization and measurement of peer collaboration in higher education: A systematic review. Small Group Research, 55(1), 89–138.",
    "link": "https://doi.org/10.1177/10464964231200191",
    "peer_reviewed": true,
    "population": "28 included best-practice publications on adult peer collaboration; 24 primarily involved students, usually in dyads or groups of three to six.",
    "design": "Systematic, selective and purposive review linking conceptual definitions of collaborative process with measurement methods; qualitative content analysis.",
    "finding_verbatim": "Collaboration is simply operationalized as interaction and participation rates.",
    "effect_size": "No intervention effect or confidence interval was estimated. Study-selection agreement κ=0.74 on an approximately 10% double-coded sample; coding agreement exceeded 0.85 for each of six double-coded included publications. Twenty-one of 28 studies used additional or relational data for triangulation.",
    "conditions_and_limits": "Measurement must specify the intended cognitive, metacognitive, affective or behavioral process, level of inference, task and context, and link recorded traces to the construct.",
    "criticism": "The review deliberately selected 28 'best practice' studies rather than exhaustively estimating prevalence; most primary samples were small and technology-supported. It provides a measurement framework, not validation of any contribution score.",
    "which_workflow": "Group work/assessment—direct team-process measurement evidence.",
    "claimable_sentence": "A higher-education review warns that participation traces require a theory-backed link to the collaborative process they are claimed to measure."
  },
  {
    "id": "Slavin1983",
    "tier": "Tier 3",
    "full_citation": "Slavin, R. E. (1983). When does cooperative learning increase student achievement? Psychological Bulletin, 94(3), 429–445.",
    "link": "https://doi.org/10.1037/0033-2909.94.3.429",
    "peer_reviewed": true,
    "population": "Elementary and secondary classroom experiments of cooperative learning lasting at least two weeks; not higher education.",
    "design": "Methodologically screened narrative review organized by cooperative-learning reward and task structures.",
    "finding_verbatim": "Group rewards and individual accountability are held to be essential.",
    "effect_size": "No pooled standardized effect size or confidence interval was reported in this review.",
    "conditions_and_limits": "The most consistently positive methods linked group rewards to the individual learning of members rather than to one shared product.",
    "criticism": "Old K–12 achievement literature, not undergraduate contribution behavior; methods were classified in bundled forms and the review cannot isolate accountability as a component effect.",
    "which_workflow": "Group work/accountability—indirect historical design-principle evidence.",
    "claimable_sentence": "The canonical individual-accountability principle comes largely from school achievement studies in which group success depended on each member's individual learning."
  },
  {
    "id": "Smith2009",
    "tier": "Tier 2",
    "full_citation": "Smith, M. K., Wood, W. B., Adams, W. K., Wieman, C., Knight, J. K., Guild, N., & Su, T. T. (2009). Why peer discussion improves student performance on in-class concept questions. Science, 323(5910), 122–124.",
    "link": "https://doi.org/10.1126/science.1165919",
    "peer_reviewed": true,
    "population": "350 undergraduate biology majors in one genetics course.",
    "design": "Sixteen isomorphic question pairs: individual question, peer discussion and revote, then an individual transfer question before feedback; item order randomized.",
    "finding_verbatim": "Peer discussion enhances understanding, even when none of the students in a discussion group originally knows the correct answer.",
    "effect_size": "Across items, first-question postdiscussion minus prediscussion performance was +16 percentage points (SEM=1); individual transfer question minus first prediscussion answer was +21 points (SEM=1); transfer minus postdiscussion answer was +5 points (SEM=1). No numeric confidence interval was reported.",
    "conditions_and_limits": "Students first committed to an individual answer, discussed, and then answered an isomorphic transfer item individually before receiving correctness feedback.",
    "criticism": "No no-discussion control, one course, immediate transfer and possible testing/isomorphic-item effects. The study demonstrates a useful assessment design but does not validate activity traces as learning evidence.",
    "which_workflow": "Group work/assessment—direct evidence for individual assessment after collaboration.",
    "claimable_sentence": "An individual transfer item after peer discussion provided attributable evidence of immediate conceptual improvement in one genetics course."
  },
  {
    "id": "Springer1999",
    "tier": "Tier 2",
    "full_citation": "Springer, L., Stanne, M. E., & Donovan, S. S. (1999). Effects of small-group learning on undergraduates in science, mathematics, engineering, and technology: A meta-analysis. Review of Educational Research, 69(1), 21–51.",
    "link": "https://doi.org/10.3102/00346543069001021",
    "peer_reviewed": true,
    "population": "Undergraduate STEM; 39 included studies; achievement N=3,472, persistence N=2,014, and attitudes N=1,293.",
    "design": "Meta-analysis of controlled and pre/post field studies of cooperative, collaborative, and mixed small-group learning.",
    "finding_verbatim": "Students who learned in small groups demonstrated greater achievement.",
    "effect_size": "Achievement d=0.51; persistence d=0.46; attitudes d=0.55. The paper states that the 95% confidence intervals excluded zero but does not print numeric limits. Achievement and attitude effects were heterogeneous. The frequently repeated 22% attrition figure is an illustrative binomial-effect conversion of d=0.46, not a pooled observed risk reduction.",
    "conditions_and_limits": "North American postsecondary STEM studies published from 1980 onward; effects varied by research design and instructional setting, and descriptions were too sparse to test many implementation features.",
    "criticism": "The literature is older, heterogeneous, and not specific to assessed projects, contribution, or AI-supported work. Instructor-investigator studies had larger effects, and a later medical-education reappraisal found the relevant randomized subset inconsistent.",
    "which_workflow": "Group work/shared team model—direct broad synthesis.",
    "claimable_sentence": "Across heterogeneous undergraduate STEM studies, structured small-group learning was associated with higher achievement, persistence, and more favorable attitudes than comparison instruction."
  },
  {
    "id": "Sridharan2019",
    "tier": "Tier 2",
    "full_citation": "Sridharan, B., Tai, J., & Boud, D. (2019). Does the use of summative peer assessment in collaborative group work inhibit good judgement? Higher Education, 77(5), 853–870.",
    "link": "https://doi.org/10.1007/s10734-018-0305-7",
    "peer_reviewed": true,
    "population": "98 capstone information-systems students—71 undergraduate and 27 postgraduate—in teams of three to five.",
    "design": "Anonymous SPARKPLUS formative and summative peer ratings; comparison of rating discrimination under no-grade and grade-adjusting conditions.",
    "finding_verbatim": "Reluctant to honestly assess their peers.",
    "effect_size": "Formative ratings differentiated contribution categories, F(2,92)=21.9, η²=.322; under- versus equal-contributor mean difference=-20.05, 95% CI [-26.91, -13.19]. Summative grade-adjusting ratings did not differentiate categories, F(2,92)=1.8, p=.20, reported η²=0.0.",
    "conditions_and_limits": "Formative ratings carried no grade consequence; summative ratings adjusted the individual share of a heavily weighted group product.",
    "criticism": "The 'actual contribution' category and outcome both derive from peer ratings, creating circularity. Formative and summative criteria and cohort policies differed; no independently observed behavior was available.",
    "which_workflow": "Group work/accountability—direct evidence on rating compression under stakes.",
    "claimable_sentence": "Peer ratings differentiated contribution categories formatively but compressed when they adjusted grades; actual behavior was not independently measured."
  },
  {
    "id": "Tan2011",
    "tier": "Tier 2",
    "full_citation": "Tan, N. C. K., Kandiah, N., Chan, Y. H., Umapathi, T., Lee, S. H., & Tan, K. (2011). A controlled study of team-based learning for undergraduate clinical neurology education. BMC Medical Education, 11, 91.",
    "link": "https://doi.org/10.1186/1472-6920-11-91",
    "peer_reviewed": true,
    "population": "49 third-year medical students at the National University of Singapore.",
    "design": "Modified crossover comparison of team-based learning with self-reading across two neurology topics.",
    "finding_verbatim": "Compared to PL, TBL showed greater improvement in knowledge scores.",
    "effect_size": "Immediate adjusted score change: 10.3% team-based learning versus 5.8% passive learning; difference=4.5 percentage points, 95% CI [0.7, 8.3], p=.021. Forty-eight-hour change: 13.0% versus 4.9%; difference=8.1 points, 95% CI [3.7, 12.5], p=.001.",
    "conditions_and_limits": "Advance reading, individual readiness test, team retest and consensus, simultaneous answers, facilitator correction, team scores and clinical application scenarios.",
    "criticism": "Small single-course sample, repeated true/false questions and only 48-hour follow-up. The intervention bundled accountability, testing, discussion and feedback, so it does not isolate any component or measure contribution.",
    "which_workflow": "Group work/accountability—direct bundled learning-outcome evidence.",
    "claimable_sentence": "A small medical-course crossover found greater immediate and 48-hour knowledge gains for a structured team-based-learning bundle than self-reading."
  },
  {
    "id": "Torka2021",
    "tier": "Tier 2",
    "full_citation": "Torka, A.-K., Mazei, J., & Hüffmeier, J. (2021). Together, everyone achieves more—or, less? An interdisciplinary meta-analysis on effort gains and losses in teams. Psychological Bulletin, 147(5), 504–534.",
    "link": "https://doi.org/10.1037/bul0000251",
    "peer_reviewed": true,
    "population": "158 studies, 622 effect sizes and 320,632 participants across disciplines; 573 objective effects in the main model.",
    "design": "Preregistered interdisciplinary three-level meta-analysis of individual effort when working in teams versus alone/coactively.",
    "finding_verbatim": "Teamwork did not have a motivating or demotivating effect on effort per se.",
    "effect_size": "Main teamwork effect Hedges' g=-0.04, t=-0.70, p=.485; numeric confidence interval not reported; I²=99.97%. Indispensability moderator coefficient=0.59, p=.007; indispensable g=0.59, p<.001, dispensable g=-0.35, p<.001. Social-comparison-potential coefficient=0.44, p=.040; evaluation-potential coefficient=0.40, p=.036. Numeric subgroup confidence intervals were not reported.",
    "conditions_and_limits": "Indispensability, social comparison and evaluation potential moderated effort; evaluation and comparison were confounded in parts of the evidence.",
    "criticism": "Extreme heterogeneity, dependent effects, mixed domains and tasks, and confounded moderators prevent a simple causal account. The synthesis is not specific to education or long student projects.",
    "which_workflow": "Group work/accountability—indirect updated mechanism synthesis.",
    "claimable_sentence": "A preregistered meta-analysis found no average teamwork effect on effort; indispensability, comparison and evaluation shaped gains and losses."
  },
  {
    "id": "Williams1981",
    "tier": "Tier 2",
    "full_citation": "Williams, K., Harkins, S., & Latané, B. (1981). Identifiability as a deterrent to social loafing: Two cheering experiments. Journal of Personality and Social Psychology, 40(2), 303–311.",
    "link": "https://doi.org/10.1037/0022-3514.40.2.303",
    "peer_reviewed": true,
    "population": "Experiment 1: 48 male undergraduates in eight six-person groups; Experiment 2: 108 male undergraduates in four-person sessions.",
    "design": "Two deceptive laboratory cheering experiments manipulating whether participants believed individual output could be measured; pseudo-groups separated motivational from coordination loss.",
    "finding_verbatim": "This manipulation eliminated social loafing.",
    "effect_size": "Experiment 1 unidentifiable pseudo-pairs and pseudo-groups of six produced 69% and 63% of solitary effort; identifiable conditions produced 98% and 92%. Identifiability change F(1,7)=37.8, p<.0005. Experiment 2 instruction-by-group-size F(4,48)=5.25, p<.01; never-identifiable below always-identifiable F(1,48)=11.5, p<.01. No standardized effect or confidence interval was reported.",
    "conditions_and_limits": "Output was objectively measurable, maximal effort was instructed and participants believed an experimenter could attribute individual performance; public peer exposure was unnecessary.",
    "criticism": "Brief artificial task, all-male samples, deception and small group-level degrees of freedom; no learning, contribution quality or durable relationship. No contemporary direct replication was located.",
    "which_workflow": "Group work/accountability—indirect causal mechanism evidence.",
    "claimable_sentence": "Believed individual identifiability eliminated effort loss in two short laboratory cheering experiments, not in an academic project."
  }
]
