From e369f0cac6452674a2e262ceac659e68406e83e6 Mon Sep 17 00:00:00 2001 From: Jacob Maynard Date: Sat, 19 Sep 2026 17:31:22 -0500 Subject: [PATCH 1/2] Report inter-rater reliability per tool with weighted kappa Replace the AMSTAR-only percent agreement and unweighted kappa with a shared reliability module that pools every dual-reviewed cell per tool. Domain judgements (RoB 2, ROBINS-I) and item answers (AMSTAR 2) get a linear weighted Cohen's kappa with a Fleiss-Cohen-Everitt confidence interval, percent agreement, a per-domain breakdown, and a signaling question agreement line. Only pairs where both reviewers gave a substantive answer are compared; one-sided not-applicable pairs and domains assessed under different aims are reported rather than counted. The overview shows one card per tool with a dialog that explains the calculation from the same stats object, so the explanation cannot drift from the numbers. Claude-Session: https://claude.ai/code/session_01TAEtViwHmBCJSDSVKkzTD6 --- packages/docs/glossary.md | 7 +- packages/shared/package.json | 4 + packages/shared/src/checklists/index.ts | 3 + .../reliability/__tests__/adapters.test.ts | 257 ++++++++++++++++++ .../reliability/__tests__/stats.test.ts | 171 ++++++++++++ .../src/checklists/reliability/amstar2.ts | 78 ++++++ .../src/checklists/reliability/index.ts | 117 ++++++++ .../shared/src/checklists/reliability/rob2.ts | 105 +++++++ .../src/checklists/reliability/robins-i.ts | 130 +++++++++ .../src/checklists/reliability/stats.ts | 222 +++++++++++++++ .../src/checklists/reliability/types.ts | 31 +++ .../web/src/components/FeatureShowcase.tsx | 2 +- .../project/overview-tab/OverviewTab.tsx | 11 +- .../overview-tab/ReliabilityAboutDialog.tsx | 215 +++++++++++++++ .../overview-tab/ReliabilitySection.tsx | 189 ++++++++++--- .../__tests__/ReliabilitySection.test.tsx | 102 +++++++ .../web/src/lib/inter-rater-reliability.ts | 228 ---------------- 17 files changed, 1597 insertions(+), 275 deletions(-) create mode 100644 packages/shared/src/checklists/reliability/__tests__/adapters.test.ts create mode 100644 packages/shared/src/checklists/reliability/__tests__/stats.test.ts create mode 100644 packages/shared/src/checklists/reliability/amstar2.ts create mode 100644 packages/shared/src/checklists/reliability/index.ts create mode 100644 packages/shared/src/checklists/reliability/rob2.ts create mode 100644 packages/shared/src/checklists/reliability/robins-i.ts create mode 100644 packages/shared/src/checklists/reliability/stats.ts create mode 100644 packages/shared/src/checklists/reliability/types.ts create mode 100644 packages/web/src/components/project/overview-tab/ReliabilityAboutDialog.tsx create mode 100644 packages/web/src/components/project/overview-tab/__tests__/ReliabilitySection.test.tsx delete mode 100644 packages/web/src/lib/inter-rater-reliability.ts diff --git a/packages/docs/glossary.md b/packages/docs/glossary.md index 1abb13ce4..5a4ae2a86 100644 --- a/packages/docs/glossary.md +++ b/packages/docs/glossary.md @@ -66,9 +66,10 @@ The process of resolving disagreements between multiple reviewers' checklist ass - Compares two completed checklists item-by-item - Shows agreements, disagreements, and missing responses -- Calculates inter-rater reliability (Cohen's kappa, percent agreement) - Creates a final reconciled checklist +Inter-rater reliability (percent agreement and weighted kappa per tool) is shown on the project overview and is computed from the reviewer checklists before reconciliation. + **Related:** `packages/web/src/components/project/reconcile-tab/ReconciliationWrapper.tsx` --- @@ -180,9 +181,9 @@ Full-stack React framework with file-based server routing, used for the main app ### Cohen's Kappa -Inter-rater reliability statistic measuring agreement between two reviewers beyond chance. Range: -1 to 1 (>0.8 = excellent agreement). +Inter-rater reliability statistic measuring agreement between two reviewers beyond chance. Range: -1 to 1, read on the Landis and Koch bands (0.8 and above = almost perfect). CoRATES reports a linear weighted kappa per tool over domain judgements (RoB 2, ROBINS-I) or item answers (AMSTAR 2), with percent agreement alongside it and a 95% confidence interval from the Fleiss, Cohen and Everitt standard error. -**Related:** `packages/web/src/lib/inter-rater-reliability.ts` +**Related:** `packages/shared/src/checklists/reliability/` ### Systematic Review diff --git a/packages/shared/package.json b/packages/shared/package.json index 4d8eca8de..dcde6bfb5 100644 --- a/packages/shared/package.json +++ b/packages/shared/package.json @@ -36,6 +36,10 @@ "types": "./dist/checklists/rob2/index.d.ts", "import": "./dist/checklists/rob2/index.js" }, + "./checklists/reliability": { + "types": "./dist/checklists/reliability/index.d.ts", + "import": "./dist/checklists/reliability/index.js" + }, "./sync": { "types": "./dist/sync/index.d.ts", "import": "./dist/sync/index.js" diff --git a/packages/shared/src/checklists/index.ts b/packages/shared/src/checklists/index.ts index 9cae8e0af..20d581380 100644 --- a/packages/shared/src/checklists/index.ts +++ b/packages/shared/src/checklists/index.ts @@ -58,3 +58,6 @@ export { scoreRob2Domain, scoreAllDomains as scoreAllROB2Domains, } from './rob2/index.js'; + +// Inter-rater reliability +export * as reliability from './reliability/index.js'; diff --git a/packages/shared/src/checklists/reliability/__tests__/adapters.test.ts b/packages/shared/src/checklists/reliability/__tests__/adapters.test.ts new file mode 100644 index 000000000..98867d954 --- /dev/null +++ b/packages/shared/src/checklists/reliability/__tests__/adapters.test.ts @@ -0,0 +1,257 @@ +/** + * Tests for the per-tool reliability adapters and the project roll-up + * + * INTENDED BEHAVIOR: + * - RoB 2: domain judgements are derived from the signaling answers, a + * domain assessed under different aims is not compared, and a question + * that branching skipped is treated as not applicable + * - ROBINS-I: a Section B Critical rating leaves no domain judgements and + * sets the overall judgement to Critical + * - AMSTAR 2: "No MA" is not applicable, the overall confidence rating is + * compared on its own scale + * - calculateProjectReliability: pools every cell with two completed reviewer + * checklists per tool, ignoring consensus and single-reviewer cells + */ + +import { describe, it, expect } from 'vitest'; +import { extractRob2Pairs } from '../rob2.js'; +import { extractRobinsIPairs } from '../robins-i.js'; +import { extractAmstar2Pairs } from '../amstar2.js'; +import { calculateProjectReliability } from '../index.js'; +import { NOT_APPLICABLE } from '../stats.js'; +import { createROB2Checklist } from '../../rob2/create.js'; +import { createROBINSIChecklist } from '../../robins-i/create.js'; +import { createAMSTAR2Checklist } from '../../amstar2/create.js'; +import { CHECKLIST_STATUS } from '../../status.js'; +import type { AMSTAR2Checklist, AMSTAR2Question, Study } from '../../types.js'; + +function rob2(answers: Record>, aim = 'ASSIGNMENT') { + const checklist = createROB2Checklist({ id: 'c', name: 'c' }); + checklist.preliminary.aim = aim as 'ASSIGNMENT' | 'ADHERING'; + for (const [domainKey, domainAnswers] of Object.entries(answers)) { + const domain = checklist[domainKey as 'domain1']; + for (const [qKey, answer] of Object.entries(domainAnswers)) { + domain.answers[qKey] = { answer: answer as 'Y', comment: '' }; + } + } + return checklist; +} + +function pairFor(pairs: { item: string; a: string | null; b: string | null }[], item: string) { + return pairs.find(p => p.item === item); +} + +describe('extractRob2Pairs', () => { + it('derives domain judgements from the signaling answers', () => { + const a = rob2({ domain1: { d1_1: 'Y', d1_2: 'Y', d1_3: 'N' } }); + const b = rob2({ domain1: { d1_1: 'Y', d1_2: 'N', d1_3: 'N' } }); + const { judgements } = extractRob2Pairs(a, b); + expect(pairFor(judgements, 'domain1')).toEqual({ item: 'domain1', a: 'Low', b: 'High' }); + expect(pairFor(judgements, 'domain3')).toEqual({ item: 'domain3', a: null, b: null }); + }); + + it('skips domain 2 when the reviewers chose different aims', () => { + const a = rob2({}, 'ASSIGNMENT'); + const b = rob2({}, 'ADHERING'); + const { judgements, questions } = extractRob2Pairs(a, b); + expect(judgements.map(p => p.item)).toEqual(['domain1', 'domain3', 'domain4', 'domain5']); + expect(questions.some(p => p.item.startsWith('d2'))).toBe(false); + }); + + it('marks a question skipped by branching as not applicable', () => { + // 3.1 = Y ends domain 3 at Low, so 3.2 to 3.4 are off the path. + const a = rob2({ domain3: { d3_1: 'Y' } }); + const b = rob2({ domain3: { d3_1: 'Y', d3_2: 'NA' } }); + const { questions } = extractRob2Pairs(a, b); + expect(pairFor(questions, 'd3_2')).toEqual({ + item: 'd3_2', + a: NOT_APPLICABLE, + b: NOT_APPLICABLE, + }); + expect(pairFor(questions, 'd3_1')).toEqual({ item: 'd3_1', a: 'Y', b: 'Y' }); + }); + + it('keeps a substantive answer against a skip so the pair reads as one-sided', () => { + const a = rob2({ domain3: { d3_1: 'N', d3_2: 'Y' } }); + const b = rob2({ domain3: { d3_1: 'Y' } }); + const { questions } = extractRob2Pairs(a, b); + expect(pairFor(questions, 'd3_2')).toEqual({ item: 'd3_2', a: 'Y', b: NOT_APPLICABLE }); + }); + + it('leaves an unanswered question as null', () => { + const a = rob2({ domain1: { d1_2: 'Y' } }); + const b = rob2({ domain1: { d1_2: 'Y', d1_1: 'Y' } }); + const { questions } = extractRob2Pairs(a, b); + expect(pairFor(questions, 'd1_1')).toEqual({ item: 'd1_1', a: null, b: 'Y' }); + }); +}); + +describe('extractRobinsIPairs', () => { + it('turns a Section B Critical rating into a Critical overall with no domain judgements', () => { + const a = createROBINSIChecklist({ id: 'a', name: 'a' }); + a.sectionB.b1.answer = 'N'; + a.sectionB.b2.answer = 'Y'; + const b = createROBINSIChecklist({ id: 'b', name: 'b' }); + b.sectionB.b1.answer = 'Y'; + b.sectionB.b2.answer = 'N'; + b.sectionB.b3.answer = 'N'; + b.domain2.answers.d2_1 = { answer: 'Y', comment: '' }; + b.domain2.answers.d2_4 = { answer: 'SY', comment: '' }; + b.domain2.answers.d2_5 = { answer: 'N', comment: '' }; + + const { judgements, overall, questions } = extractRobinsIPairs(a, b); + expect(pairFor(judgements, 'domain2')).toEqual({ item: 'domain2', a: null, b: 'Serious' }); + expect(overall).toEqual([{ item: 'overall', a: 'Critical', b: null }]); + expect(pairFor(questions, 'b2')).toEqual({ item: 'b2', a: 'Y', b: 'N' }); + // Every domain question is skipped for the reviewer who stopped at Section B. + expect(pairFor(questions, 'd2_1')).toEqual({ item: 'd2_1', a: NOT_APPLICABLE, b: 'Y' }); + }); + + it('merges both domain 1 variants onto one row', () => { + const a = createROBINSIChecklist({ id: 'a', name: 'a' }); + const b = createROBINSIChecklist({ id: 'b', name: 'b' }); + const { judgements } = extractRobinsIPairs(a, b); + expect(judgements.map(p => p.item)).toEqual([ + 'domain1', + 'domain2', + 'domain3', + 'domain4', + 'domain5', + 'domain6', + ]); + }); + + it('skips domain 1 when only one reviewer assessed the per-protocol effect', () => { + const a = createROBINSIChecklist({ id: 'a', name: 'a' }); + const b = createROBINSIChecklist({ id: 'b', name: 'b' }); + b.sectionC.isPerProtocol = true; + const { judgements } = extractRobinsIPairs(a, b); + expect(judgements.map(p => p.item)).not.toContain('domain1'); + }); +}); + +function amstar2(answers: Record): AMSTAR2Checklist { + const checklist = createAMSTAR2Checklist({ id: 'c', name: 'c' }); + for (const [key, index] of Object.entries(answers)) { + const question = checklist[key as keyof AMSTAR2Checklist] as AMSTAR2Question; + const last = question.answers[question.answers.length - 1]; + question.answers[question.answers.length - 1] = last.map((_, i) => i === index); + } + return checklist; +} + +describe('extractAmstar2Pairs', () => { + it('compares final answers per item and maps No MA to not applicable', () => { + const a = amstar2({ q1: 0, q2: 1, q9a: 3 }); + const b = amstar2({ q1: 0, q2: 2, q9a: 0 }); + const { judgements } = extractAmstar2Pairs(a, b); + expect(pairFor(judgements, 'q1')).toEqual({ item: 'q1', a: 'Yes', b: 'Yes' }); + expect(pairFor(judgements, 'q2')).toEqual({ item: 'q2', a: 'Partial Yes', b: 'No' }); + expect(pairFor(judgements, 'q9a')).toEqual({ item: 'q9a', a: NOT_APPLICABLE, b: 'Yes' }); + expect(pairFor(judgements, 'q3')).toEqual({ item: 'q3', a: null, b: null }); + }); + + it('compares the overall confidence rating only for complete checklists', () => { + const allYes = Object.fromEntries( + [ + 'q1', + 'q2', + 'q3', + 'q4', + 'q5', + 'q6', + 'q7', + 'q8', + 'q9a', + 'q9b', + 'q10', + 'q11a', + 'q11b', + 'q12', + 'q13', + 'q14', + 'q15', + 'q16', + ].map(key => [key, 0]), + ); + const a = amstar2(allYes); + const b = amstar2({ ...allYes, q1: 1 }); + const incomplete = amstar2({ q1: 0 }); + expect(extractAmstar2Pairs(a, b).overall).toEqual([{ item: 'overall', a: 'High', b: 'High' }]); + expect(extractAmstar2Pairs(a, incomplete).overall).toEqual([ + { item: 'overall', a: 'High', b: null }, + ]); + }); +}); + +describe('calculateProjectReliability', () => { + const data = new Map(); + const getChecklistData = (_studyId: string, checklistId: string) => { + const answers = data.get(checklistId); + return answers ? { answers } : null; + }; + + function reviewer(id: string, type: string, assignedTo: string, outcomeId: string | null) { + return { id, type, assignedTo, outcomeId, status: CHECKLIST_STATUS.REVIEWER_COMPLETED }; + } + + it('pools cells per tool and skips consensus, single-reviewer and same-reviewer cells', () => { + data.set('r1', rob2({ domain1: { d1_1: 'Y', d1_2: 'Y', d1_3: 'N' } })); + data.set('r2', rob2({ domain1: { d1_1: 'Y', d1_2: 'N', d1_3: 'N' } })); + data.set('r3', rob2({ domain1: { d1_1: 'Y', d1_2: 'Y', d1_3: 'N' } })); + data.set('r4', rob2({ domain1: { d1_1: 'Y', d1_2: 'Y', d1_3: 'N' } })); + data.set('a1', amstar2({ q1: 0 })); + data.set('a2', amstar2({ q1: 0 })); + + const studies: Study[] = [ + { + id: 's1', + checklists: [ + reviewer('r1', 'ROB2', 'u1', 'o1'), + reviewer('r2', 'ROB2', 'u2', 'o1'), + { id: 'x1', type: 'ROB2', kind: 'consensus', outcomeId: 'o1', status: 'finalized' }, + reviewer('r3', 'ROB2', 'u1', 'o2'), + reviewer('r4', 'ROB2', 'u2', 'o2'), + ], + }, + { + id: 's2', + checklists: [ + reviewer('a1', 'AMSTAR2', 'u1', null), + reviewer('a2', 'AMSTAR2', 'u2', null), + reviewer('r5', 'ROB2', 'u1', 'o3'), + ], + }, + { + id: 's3', + checklists: [reviewer('r6', 'ROB2', 'u1', 'o4'), reviewer('r7', 'ROB2', 'u1', 'o4')], + }, + ]; + + const result = calculateProjectReliability(studies, getChecklistData); + expect(result.map(t => t.definition.type)).toEqual(['ROB2', 'AMSTAR2']); + + const rob = result[0]; + expect(rob.cells).toBe(2); + expect(rob.studies).toBe(1); + expect(rob.judgements.compared).toBe(2); + expect(rob.judgements.agreed).toBe(1); + expect(rob.judgements.items.find(i => i.key === 'domain1')).toMatchObject({ + compared: 2, + agreed: 1, + }); + expect(rob.questions?.compared).toBe(6); + expect(rob.overall.compared).toBe(0); + + const amstar = result[1]; + expect(amstar.cells).toBe(1); + expect(amstar.judgements.compared).toBe(1); + expect(amstar.judgements.agreed).toBe(1); + expect(amstar.questions).toBeNull(); + }); + + it('returns nothing for an empty project', () => { + expect(calculateProjectReliability([], getChecklistData)).toEqual([]); + expect(calculateProjectReliability(null, getChecklistData)).toEqual([]); + }); +}); diff --git a/packages/shared/src/checklists/reliability/__tests__/stats.test.ts b/packages/shared/src/checklists/reliability/__tests__/stats.test.ts new file mode 100644 index 000000000..1dd454188 --- /dev/null +++ b/packages/shared/src/checklists/reliability/__tests__/stats.test.ts @@ -0,0 +1,171 @@ +/** + * Tests for the reliability statistics core + * + * INTENDED BEHAVIOR: + * - classifyPair: a pair is compared only when both sides are substantive + * and on the scale; one not-applicable side is reported separately, + * unanswered is excluded (the counts are checked through summarizeLevel) + * - weightedKappa: linear weighted Cohen's kappa with the Fleiss, Cohen and + * Everitt (1969) standard error; identity weights reduce to plain kappa + * - summarizeLevel: per-item agreement, confusion matrix, and a kappa only + * once enough pairs have been compared + */ + +import { describe, it, expect } from 'vitest'; +import { + classifyPair, + weightedKappa, + summarizeLevel, + getKappaInterpretation, + MIN_PAIRS_FOR_KAPPA, + NOT_APPLICABLE, +} from '../stats.js'; + +function repeat(value: T, times: number): T[] { + return Array.from({ length: times }, () => value); +} + +describe('classifyPair', () => { + it('excludes an answer that is off the scale', () => { + expect(classifyPair({ item: 'x', a: 'Maybe', b: 'Low' }, ['Low', 'High'])).toBe('excluded'); + }); +}); + +describe('weightedKappa', () => { + it('reduces to plain Cohen kappa on a two-category scale', () => { + // a\b table [[20, 5], [10, 15]]: Po = 0.7, Pe = 0.5, kappa = 0.4. + // SE hand-computed from the unweighted Fleiss formula: 0.1270. + const pairs: Array<[string, string]> = [ + ...repeat<[string, string]>(['Y', 'Y'], 20), + ...repeat<[string, string]>(['Y', 'N'], 5), + ...repeat<[string, string]>(['N', 'Y'], 10), + ...repeat<[string, string]>(['N', 'N'], 15), + ]; + const result = weightedKappa(pairs, ['Y', 'N'])!; + expect(result.observed).toBeCloseTo(0.7, 10); + expect(result.expected).toBeCloseTo(0.5, 10); + expect(result.kappa).toBeCloseTo(0.4, 10); + expect(result.se).toBeCloseTo(0.127, 3); + expect(result.ci[0]).toBeCloseTo(0.4 - 1.96 * 0.127, 2); + expect(result.ci[1]).toBeCloseTo(0.4 + 1.96 * 0.127, 2); + }); + + it('gives half credit to a one-step disagreement on a three-category scale', () => { + // a\b table [[10, 4, 2], [2, 5, 1], [1, 0, 5]], n = 30. + // Linear weights: Po = 23.5/30, Pe = 511/900, kappa = 0.4987. + const pairs: Array<[string, string]> = [ + ...repeat<[string, string]>(['L', 'L'], 10), + ...repeat<[string, string]>(['L', 'S'], 4), + ...repeat<[string, string]>(['L', 'H'], 2), + ...repeat<[string, string]>(['S', 'L'], 2), + ...repeat<[string, string]>(['S', 'S'], 5), + ...repeat<[string, string]>(['S', 'H'], 1), + ...repeat<[string, string]>(['H', 'L'], 1), + ...repeat<[string, string]>(['H', 'H'], 5), + ]; + const result = weightedKappa(pairs, ['L', 'S', 'H'])!; + expect(result.observed).toBeCloseTo(23.5 / 30, 10); + expect(result.expected).toBeCloseTo(511 / 900, 10); + expect(result.kappa).toBeCloseTo(0.4987, 4); + expect(result.se).toBeGreaterThan(0); + }); + + it('is 1 on perfect agreement with both categories used', () => { + const pairs: Array<[string, string]> = [ + ...repeat<[string, string]>(['L', 'L'], 5), + ...repeat<[string, string]>(['H', 'H'], 5), + ]; + const result = weightedKappa(pairs, ['L', 'H'])!; + expect(result.kappa).toBe(1); + expect(result.se).toBe(0); + }); + + it('is undefined when every rating is the same category', () => { + expect(weightedKappa(repeat<[string, string]>(['L', 'L'], 10), ['L', 'H'])).toBeNull(); + }); + + it('is undefined for a rating outside the scale or an empty input', () => { + expect(weightedKappa([['L', 'X']], ['L', 'H'])).toBeNull(); + expect(weightedKappa([], ['L', 'H'])).toBeNull(); + }); +}); + +describe('summarizeLevel', () => { + const scale = ['Low', 'Some concerns', 'High']; + const items = [ + { key: 'domain1', label: 'D1', title: 'Domain 1' }, + { key: 'domain2', label: 'D2', title: 'Domain 2' }, + ]; + + it('counts compared, agreed, one-sided and excluded pairs', () => { + const result = summarizeLevel( + [ + { item: 'domain1', a: 'Low', b: 'Low' }, + { item: 'domain1', a: 'Low', b: 'High' }, + { item: 'domain2', a: NOT_APPLICABLE, b: 'Low' }, + { item: 'domain2', a: null, b: 'Low' }, + { item: 'domain2', a: 'High', b: 'High' }, + ], + scale, + items, + ); + expect(result.compared).toBe(3); + expect(result.agreed).toBe(2); + expect(result.oneSided).toBe(1); + expect(result.excluded).toBe(1); + expect(result.percentAgreement).toBeCloseTo(66.67, 1); + expect(result.items).toEqual([ + { key: 'domain1', label: 'D1', title: 'Domain 1', compared: 2, agreed: 1 }, + { key: 'domain2', label: 'D2', title: 'Domain 2', compared: 1, agreed: 1 }, + ]); + expect(result.matrix).toEqual([ + [1, 0, 1], + [0, 0, 0], + [0, 0, 1], + ]); + }); + + it('withholds the kappa until enough pairs have been compared', () => { + const pair = { item: 'domain1', a: 'Low', b: 'High' }; + const agree = { item: 'domain1', a: 'Low', b: 'Low' }; + const few = summarizeLevel(repeat(pair, MIN_PAIRS_FOR_KAPPA - 1), scale, items); + expect(few.kappa).toBeNull(); + const enough = summarizeLevel( + [...repeat(pair, MIN_PAIRS_FOR_KAPPA / 2), ...repeat(agree, MIN_PAIRS_FOR_KAPPA / 2)], + scale, + items, + ); + expect(enough.kappa).not.toBeNull(); + }); + + it('gives percent agreement only when there is no scale', () => { + const result = summarizeLevel( + repeat({ item: 'd1_1', a: 'Y', b: 'Y' }, MIN_PAIRS_FOR_KAPPA + 5), + null, + [], + ); + expect(result.percentAgreement).toBe(100); + expect(result.kappa).toBeNull(); + expect(result.matrix).toBeNull(); + }); + + it('returns nulls for an empty level', () => { + const result = summarizeLevel([], scale, items); + expect(result.percentAgreement).toBeNull(); + expect(result.kappa).toBeNull(); + expect(result.matrix).toBeNull(); + }); +}); + +describe('getKappaInterpretation', () => { + it('places values on the Landis and Koch bands', () => { + expect(getKappaInterpretation(null)).toBe('N/A'); + expect(getKappaInterpretation(-0.1)).toBe('Poor'); + expect(getKappaInterpretation(0)).toBe('Slight'); + expect(getKappaInterpretation(0.2)).toBe('Fair'); + expect(getKappaInterpretation(0.4)).toBe('Moderate'); + expect(getKappaInterpretation(0.6)).toBe('Substantial'); + expect(getKappaInterpretation(0.8)).toBe('Almost perfect'); + expect(getKappaInterpretation(1)).toBe('Almost perfect'); + }); +}); diff --git a/packages/shared/src/checklists/reliability/amstar2.ts b/packages/shared/src/checklists/reliability/amstar2.ts new file mode 100644 index 000000000..af84671a6 --- /dev/null +++ b/packages/shared/src/checklists/reliability/amstar2.ts @@ -0,0 +1,78 @@ +/** + * AMSTAR 2 reliability adapter: turns two reviewer checklists into rating pairs. + * + * AMSTAR 2 has no domains, so its items are the judgements. The sub-criteria + * checkboxes are working notes and are not compared. + */ + +import { AMSTAR_CHECKLIST, AMSTAR2_DATA_KEYS } from '../amstar2/schema.js'; +import { getSelectedAnswer } from '../amstar2/answers.js'; +import { scoreAMSTAR2Checklist, isAMSTAR2Complete } from '../amstar2/score.js'; +import type { AMSTAR2Checklist, AMSTAR2Question } from '../types.js'; +import { NOT_APPLICABLE, type ItemDefinition, type RatingPair } from './stats.js'; +import type { ToolPairs, ToolReliabilityDefinition } from './types.js'; + +type Checklist = Partial; + +export const AMSTAR2_ITEM_SCALE = ['Yes', 'Partial Yes', 'No']; +export const AMSTAR2_OVERALL_SCALE = ['High', 'Moderate', 'Low', 'Critically Low']; + +function itemDefinition(dataKey: string): ItemDefinition { + const questionKey = dataKey.replace(/[ab]$/, ''); + const schema = AMSTAR_CHECKLIST[questionKey]; + const part = + dataKey.endsWith('a') ? schema?.subtitle + : dataKey.endsWith('b') ? schema?.subtitle2 + : null; + const number = questionKey.slice(1); + return { + key: dataKey, + label: part ? `Q${number} ${part}` : `Q${number}`, + title: schema?.text ?? dataKey, + }; +} + +const ITEMS: ItemDefinition[] = AMSTAR2_DATA_KEYS.map(itemDefinition); + +function itemAnswer(checklist: Checklist, dataKey: string): string | null { + const question = checklist[dataKey as keyof AMSTAR2Checklist] as AMSTAR2Question | undefined; + if (!question || !Array.isArray(question.answers)) return null; + const selected = getSelectedAnswer(question.answers, dataKey); + // "No MA" also stands in for "Includes only NRSI/RCTs" on Q9 and Q11. + if (selected === 'No MA') return NOT_APPLICABLE; + return selected; +} + +function overallRating(checklist: Checklist): string | null { + const full = checklist as AMSTAR2Checklist; + if (!isAMSTAR2Complete(full)) return null; + const score = scoreAMSTAR2Checklist(full); + return score === 'Error' ? null : score; +} + +export function extractAmstar2Pairs(a: Checklist, b: Checklist): ToolPairs { + const judgements: RatingPair[] = AMSTAR2_DATA_KEYS.map(dataKey => ({ + item: dataKey, + a: itemAnswer(a, dataKey), + b: itemAnswer(b, dataKey), + })); + const overall: RatingPair[] = [{ item: 'overall', a: overallRating(a), b: overallRating(b) }]; + return { judgements, overall, questions: [] }; +} + +export const AMSTAR2_RELIABILITY: ToolReliabilityDefinition = { + type: 'AMSTAR2', + label: 'AMSTAR 2', + unit: 'study', + judgementLabel: 'Item answers', + judgementScale: AMSTAR2_ITEM_SCALE, + overallScale: AMSTAR2_OVERALL_SCALE, + items: ITEMS, + hasQuestions: false, + notes: [ + '18 items compared on their final answer (Q9 and Q11 have RCT and NRSI parts). Sub-criteria checkboxes are not compared.', + 'Yes/No-only items sit on the same scale, so Yes against No is a full disagreement.', + 'The overall confidence rating follows from the items, so it is compared on its own scale.', + ], + extractPairs: (a, b) => extractAmstar2Pairs(a as Checklist, b as Checklist), +}; diff --git a/packages/shared/src/checklists/reliability/index.ts b/packages/shared/src/checklists/reliability/index.ts new file mode 100644 index 000000000..a0f80b1d8 --- /dev/null +++ b/packages/shared/src/checklists/reliability/index.ts @@ -0,0 +1,117 @@ +/** + * Inter-rater reliability across a project + * + * Every (study, tool, outcome) cell with two completed reviewer checklists is + * one reviewer pair. Pairs are pooled per tool, never across tools, because + * each tool has its own scale. The reviewer checklists keep their status after + * the consensus is finalized, so the numbers describe agreement before + * reconciliation and stay stable afterwards. + */ + +import { CHECKLIST_STATUS } from '../status.js'; +import { getAppraisalCells, isReconciledChecklist } from '../domain.js'; +import type { Study } from '../types.js'; +import { summarizeLevel, type LevelStats } from './stats.js'; +import type { ToolPairs, ToolReliabilityDefinition, ReliabilityToolType } from './types.js'; +import { ROB2_RELIABILITY } from './rob2.js'; +import { ROBINS_I_RELIABILITY } from './robins-i.js'; +import { AMSTAR2_RELIABILITY } from './amstar2.js'; + +export * from './stats.js'; +export * from './types.js'; +export { extractRob2Pairs, ROB2_JUDGEMENT_SCALE } from './rob2.js'; +export { extractRobinsIPairs, ROBINS_I_JUDGEMENT_SCALE } from './robins-i.js'; +export { extractAmstar2Pairs, AMSTAR2_ITEM_SCALE, AMSTAR2_OVERALL_SCALE } from './amstar2.js'; + +export const RELIABILITY_TOOLS: Record = { + ROB2: ROB2_RELIABILITY, + ROBINS_I: ROBINS_I_RELIABILITY, + AMSTAR2: AMSTAR2_RELIABILITY, +}; + +const TOOL_ORDER: ReliabilityToolType[] = ['ROB2', 'ROBINS_I', 'AMSTAR2']; + +export interface ToolReliability { + definition: ToolReliabilityDefinition; + /** Reviewer pairs compared. */ + cells: number; + studies: number; + judgements: LevelStats; + overall: LevelStats; + questions: LevelStats | null; +} + +export type ChecklistDataGetter = ( + studyId: string, + checklistId: string, +) => { answers?: unknown } | null | undefined; + +function isReliabilityTool(type: string): type is ReliabilityToolType { + return type in RELIABILITY_TOOLS; +} + +export function calculateProjectReliability( + studies: Study[] | null | undefined, + getChecklistData: ChecklistDataGetter, +): ToolReliability[] { + const pools = new Map< + ReliabilityToolType, + { pairs: ToolPairs; cells: number; studies: Set } + >(); + + for (const study of studies ?? []) { + for (const cell of getAppraisalCells(study)) { + if (!isReliabilityTool(cell.type)) continue; + const reviewers = cell.checklists.filter( + c => !isReconciledChecklist(c) && c.status === CHECKLIST_STATUS.REVIEWER_COMPLETED, + ); + if (reviewers.length !== 2) continue; + if (reviewers[0].assignedTo && reviewers[0].assignedTo === reviewers[1].assignedTo) continue; + + // Stable orientation for the matrix: rows are the lower user id. + reviewers.sort((x, y) => (x.assignedTo ?? '').localeCompare(y.assignedTo ?? '')); + const dataA = getChecklistData(study.id, reviewers[0].id)?.answers; + const dataB = getChecklistData(study.id, reviewers[1].id)?.answers; + if (!dataA || !dataB) continue; + + const definition = RELIABILITY_TOOLS[cell.type]; + const extracted = definition.extractPairs(dataA, dataB); + let pool = pools.get(cell.type); + if (!pool) { + pool = { + pairs: { judgements: [], overall: [], questions: [] }, + cells: 0, + studies: new Set(), + }; + pools.set(cell.type, pool); + } + pool.pairs.judgements.push(...extracted.judgements); + pool.pairs.overall.push(...extracted.overall); + pool.pairs.questions.push(...extracted.questions); + pool.cells += 1; + pool.studies.add(study.id); + } + } + + const results: ToolReliability[] = []; + for (const type of TOOL_ORDER) { + const pool = pools.get(type); + if (!pool) continue; + const definition = RELIABILITY_TOOLS[type]; + results.push({ + definition, + cells: pool.cells, + studies: pool.studies.size, + judgements: summarizeLevel( + pool.pairs.judgements, + definition.judgementScale, + definition.items, + ), + overall: summarizeLevel(pool.pairs.overall, definition.overallScale, [ + { key: 'overall', label: 'Overall', title: 'Overall judgement' }, + ]), + questions: definition.hasQuestions ? summarizeLevel(pool.pairs.questions, null, []) : null, + }); + } + return results; +} diff --git a/packages/shared/src/checklists/reliability/rob2.ts b/packages/shared/src/checklists/reliability/rob2.ts new file mode 100644 index 000000000..29cb735d6 --- /dev/null +++ b/packages/shared/src/checklists/reliability/rob2.ts @@ -0,0 +1,105 @@ +/** + * RoB 2 reliability adapter: turns two reviewer checklists into rating pairs. + */ + +import { + ROB2_CHECKLIST, + JUDGEMENTS, + getActiveDomainKeys, + getDomainQuestions, +} from '../rob2/schema.js'; +import { scoreRob2Domain, scoreAllDomains, type ChecklistState } from '../rob2/scoring.js'; +import { getSkippedDomainQuestions, isEffectivelyNotApplicable } from '../rob2/skipped.js'; +import type { ROB2Checklist, ROB2DomainState } from '../types.js'; +import { NOT_APPLICABLE, type ItemDefinition, type RatingPair } from './stats.js'; +import type { ToolPairs, ToolReliabilityDefinition } from './types.js'; + +type Checklist = Partial; + +export const ROB2_JUDGEMENT_SCALE = [JUDGEMENTS.LOW, JUDGEMENTS.SOME_CONCERNS, JUDGEMENTS.HIGH]; + +// Both domain 2 variants land on the same row so the breakdown is per domain, not per aim. +const ITEM_KEY: Record = { + domain1: 'domain1', + domain2a: 'domain2', + domain2b: 'domain2', + domain3: 'domain3', + domain4: 'domain4', + domain5: 'domain5', +}; + +const DOMAIN_ITEMS: ItemDefinition[] = [ + { key: 'domain1', label: 'D1', title: ROB2_CHECKLIST.domain1.name }, + { key: 'domain2', label: 'D2', title: ROB2_CHECKLIST.domain2a.name }, + { key: 'domain3', label: 'D3', title: ROB2_CHECKLIST.domain3.name }, + { key: 'domain4', label: 'D4', title: ROB2_CHECKLIST.domain4.name }, + { key: 'domain5', label: 'D5', title: ROB2_CHECKLIST.domain5.name }, +]; + +function normalizeAnswer( + questionKey: string, + answer: string | null | undefined, + skipped: Set, +): string | null { + if (isEffectivelyNotApplicable(questionKey, answer, skipped)) return NOT_APPLICABLE; + return answer ?? null; +} + +export function extractRob2Pairs(a: Checklist, b: Checklist): ToolPairs { + const activeA = getActiveDomainKeys(a.preliminary?.aim === 'ADHERING'); + const activeB = getActiveDomainKeys(b.preliminary?.aim === 'ADHERING'); + // A domain assessed under different aims is a different set of questions, so + // it is only comparable when both reviewers chose the same aim. + const shared = activeA.filter(key => activeB.includes(key)); + + const judgements: RatingPair[] = []; + const questions: RatingPair[] = []; + + for (const domainKey of shared) { + const domainA = a[domainKey] as ROB2DomainState | undefined; + const domainB = b[domainKey] as ROB2DomainState | undefined; + + judgements.push({ + item: ITEM_KEY[domainKey], + a: scoreRob2Domain(domainKey, domainA?.answers).judgement, + b: scoreRob2Domain(domainKey, domainB?.answers).judgement, + }); + + const skippedA = getSkippedDomainQuestions(domainKey, domainA?.answers); + const skippedB = getSkippedDomainQuestions(domainKey, domainB?.answers); + for (const qKey of Object.keys(getDomainQuestions(domainKey))) { + questions.push({ + item: qKey, + a: normalizeAnswer(qKey, domainA?.answers?.[qKey]?.answer, skippedA), + b: normalizeAnswer(qKey, domainB?.answers?.[qKey]?.answer, skippedB), + }); + } + } + + const overall: RatingPair[] = [ + { + item: 'overall', + a: scoreAllDomains(a as ChecklistState).overall, + b: scoreAllDomains(b as ChecklistState).overall, + }, + ]; + + return { judgements, overall, questions }; +} + +export const ROB2_RELIABILITY: ToolReliabilityDefinition = { + type: 'ROB2', + label: 'RoB 2', + unit: 'outcome', + judgementLabel: 'Domain judgements', + judgementScale: ROB2_JUDGEMENT_SCALE, + overallScale: ROB2_JUDGEMENT_SCALE, + items: DOMAIN_ITEMS, + hasQuestions: true, + notes: [ + 'Domain judgements come from the RoB 2 algorithm applied to the signaling answers, as the checklist shows them.', + 'Domain 2 is compared only when both reviewers assessed the same effect (assignment or adhering).', + 'The overall judgement follows from the domains, so it is compared separately.', + ], + extractPairs: (a, b) => extractRob2Pairs(a as Checklist, b as Checklist), +}; diff --git a/packages/shared/src/checklists/reliability/robins-i.ts b/packages/shared/src/checklists/reliability/robins-i.ts new file mode 100644 index 000000000..e515c532b --- /dev/null +++ b/packages/shared/src/checklists/reliability/robins-i.ts @@ -0,0 +1,130 @@ +/** + * ROBINS-I reliability adapter: turns two reviewer checklists into rating pairs. + */ + +import { ROBINS_I_CHECKLIST, getActiveDomainKeys, getDomainQuestions } from '../robins-i/schema.js'; +import { JUDGEMENTS, scoreRobinsDomain, scoreAllDomains } from '../robins-i/scoring.js'; +import { + getSkippedQuestions, + isEffectivelyNotApplicable, + isSectionBCritical, +} from '../robins-i/skipped.js'; +import type { ROBINSIChecklist, ROBINSIDomainState } from '../types.js'; +import { NOT_APPLICABLE, type ItemDefinition, type RatingPair } from './stats.js'; +import type { ToolPairs, ToolReliabilityDefinition } from './types.js'; + +type Checklist = Partial; + +export const ROBINS_I_JUDGEMENT_SCALE = [ + JUDGEMENTS.LOW, + JUDGEMENTS.LOW_EXCEPT_CONFOUNDING, + JUDGEMENTS.MODERATE, + JUDGEMENTS.SERIOUS, + JUDGEMENTS.CRITICAL, +]; + +// Both domain 1 variants land on the same row so the breakdown is per domain, not per effect. +const ITEM_KEY: Record = { + domain1a: 'domain1', + domain1b: 'domain1', + domain2: 'domain2', + domain3: 'domain3', + domain4: 'domain4', + domain5: 'domain5', + domain6: 'domain6', +}; + +const DOMAIN_ITEMS: ItemDefinition[] = [ + { key: 'domain1', label: 'D1', title: ROBINS_I_CHECKLIST.domain1a.name }, + { key: 'domain2', label: 'D2', title: ROBINS_I_CHECKLIST.domain2.name }, + { key: 'domain3', label: 'D3', title: ROBINS_I_CHECKLIST.domain3.name }, + { key: 'domain4', label: 'D4', title: ROBINS_I_CHECKLIST.domain4.name }, + { key: 'domain5', label: 'D5', title: ROBINS_I_CHECKLIST.domain5.name }, + { key: 'domain6', label: 'D6', title: ROBINS_I_CHECKLIST.domain6.name }, +]; + +type ScoringState = Parameters[0]; +type SectionBState = Parameters[0]; + +function overallJudgement(checklist: Checklist): string | null { + if (isSectionBCritical(checklist.sectionB as SectionBState)) return JUDGEMENTS.CRITICAL; + return scoreAllDomains(checklist as ScoringState).overall; +} + +function normalizeAnswer( + questionKey: string, + answer: string | null | undefined, + skipped: Set, +): string | null { + if (isEffectivelyNotApplicable(questionKey, answer, skipped)) return NOT_APPLICABLE; + return answer ?? null; +} + +export function extractRobinsIPairs(a: Checklist, b: Checklist): ToolPairs { + const activeA = getActiveDomainKeys(a.sectionC?.isPerProtocol || false); + const activeB = getActiveDomainKeys(b.sectionC?.isPerProtocol || false); + // Domain 1 differs between the assignment and per-protocol effects, so it is + // only comparable when both reviewers chose the same effect. + const shared = activeA.filter(key => activeB.includes(key)); + + // A Critical rating in Section B ends the assessment, so that reviewer has + // no domain judgements to compare; the overall pair carries the difference. + const criticalA = isSectionBCritical(a.sectionB as SectionBState); + const criticalB = isSectionBCritical(b.sectionB as SectionBState); + const skippedA = getSkippedQuestions(a as Parameters[0]); + const skippedB = getSkippedQuestions(b as Parameters[0]); + + const judgements: RatingPair[] = []; + const questions: RatingPair[] = []; + + for (const key of Object.keys(ROBINS_I_CHECKLIST.sectionB)) { + questions.push({ + item: key, + a: a.sectionB?.[key as 'b1' | 'b2' | 'b3']?.answer ?? null, + b: b.sectionB?.[key as 'b1' | 'b2' | 'b3']?.answer ?? null, + }); + } + + for (const domainKey of shared) { + const domainA = a[domainKey] as ROBINSIDomainState | undefined; + const domainB = b[domainKey] as ROBINSIDomainState | undefined; + + judgements.push({ + item: ITEM_KEY[domainKey], + a: criticalA ? null : scoreRobinsDomain(domainKey, domainA?.answers).judgement, + b: criticalB ? null : scoreRobinsDomain(domainKey, domainB?.answers).judgement, + }); + + for (const qKey of Object.keys(getDomainQuestions(domainKey))) { + questions.push({ + item: qKey, + a: normalizeAnswer(qKey, domainA?.answers?.[qKey]?.answer, skippedA), + b: normalizeAnswer(qKey, domainB?.answers?.[qKey]?.answer, skippedB), + }); + } + } + + const overall: RatingPair[] = [ + { item: 'overall', a: overallJudgement(a), b: overallJudgement(b) }, + ]; + + return { judgements, overall, questions }; +} + +export const ROBINS_I_RELIABILITY: ToolReliabilityDefinition = { + type: 'ROBINS_I', + label: 'ROBINS-I', + unit: 'outcome', + judgementLabel: 'Domain judgements', + judgementScale: ROBINS_I_JUDGEMENT_SCALE, + overallScale: ROBINS_I_JUDGEMENT_SCALE, + items: DOMAIN_ITEMS, + hasQuestions: true, + notes: [ + 'Domain judgements come from the ROBINS-I algorithm applied to the signaling answers, as the checklist shows them.', + 'Domain 1 is compared only when both reviewers assessed the same effect (assignment or per-protocol).', + 'A Critical rating in Section B leaves that reviewer with no domain judgements; the difference shows in the overall judgement.', + 'The overall judgement follows from the domains, so it is compared separately.', + ], + extractPairs: (a, b) => extractRobinsIPairs(a as Checklist, b as Checklist), +}; diff --git a/packages/shared/src/checklists/reliability/stats.ts b/packages/shared/src/checklists/reliability/stats.ts new file mode 100644 index 000000000..0e48f6cf7 --- /dev/null +++ b/packages/shared/src/checklists/reliability/stats.ts @@ -0,0 +1,222 @@ +/** + * Inter-rater reliability statistics + * + * Tool-agnostic: works on pairs of ratings and an ordered scale. Tool adapters + * turn checklists into pairs; the UI explains the numbers using the same + * constants exported here so the explanation cannot drift from the maths. + */ + +/** Sentinel for "not applicable", whether chosen explicitly or skipped by branching. */ +export const NOT_APPLICABLE = 'NA'; + +/** Below this many compared pairs a kappa swings too much to be worth showing. */ +export const MIN_PAIRS_FOR_KAPPA = 20; + +/** Two-sided 95% normal quantile used for the kappa confidence interval. */ +export const CI_Z = 1.96; + +/** Landis and Koch (1977) bands, lower bound inclusive. */ +export const KAPPA_BANDS = [ + { min: 0.8, label: 'Almost perfect' }, + { min: 0.6, label: 'Substantial' }, + { min: 0.4, label: 'Moderate' }, + { min: 0.2, label: 'Fair' }, + { min: 0, label: 'Slight' }, + { min: -Infinity, label: 'Poor' }, +] as const; + +export interface RatingPair { + /** Which domain, question or item the pair belongs to. */ + item: string; + a: string | null; + b: string | null; +} + +export type PairClass = 'compared' | 'one-sided' | 'excluded'; + +export interface KappaResult { + kappa: number; + se: number; + ci: [number, number]; + observed: number; + expected: number; +} + +export interface ItemStats { + key: string; + label: string; + title: string; + compared: number; + agreed: number; +} + +export interface LevelStats { + /** Ordered categories the kappa is weighted over; null for percent agreement only. */ + scale: string[] | null; + compared: number; + agreed: number; + percentAgreement: number | null; + /** Not applicable for exactly one reviewer, so no comparison was possible. */ + oneSided: number; + /** Unanswered by either reviewer, or not applicable for both. */ + excluded: number; + kappa: KappaResult | null; + /** Counts of compared pairs, rows are reviewer A and columns reviewer B, in scale order. */ + matrix: number[][] | null; + items: ItemStats[]; +} + +export interface ItemDefinition { + key: string; + label: string; + title: string; +} + +/** + * A pair is compared only when both reviewers gave a substantive answer. When + * exactly one side is not applicable the pair is reported separately rather + * than counted as a disagreement, because the branching answer that caused it + * is already compared on its own. + */ +export function classifyPair(pair: RatingPair, scale: string[] | null): PairClass { + const { a, b } = pair; + if (a == null || b == null) return 'excluded'; + const aNa = a === NOT_APPLICABLE; + const bNa = b === NOT_APPLICABLE; + if (aNa && bNa) return 'excluded'; + if (aNa || bNa) return 'one-sided'; + if (scale && (!scale.includes(a) || !scale.includes(b))) return 'excluded'; + return 'compared'; +} + +/** Linear disagreement weight: distance along the scale as a fraction of its length. */ +export function linearDisagreement(i: number, j: number, categories: number): number { + if (categories < 2) return 0; + return Math.abs(i - j) / (categories - 1); +} + +/** + * Linear weighted Cohen's kappa with the large-sample standard error of + * Fleiss, Cohen and Everitt (1969). With identity weights this is plain + * Cohen's kappa. Pairs must already be on the scale. + */ +export function weightedKappa(pairs: Array<[string, string]>, scale: string[]): KappaResult | null { + const k = scale.length; + const n = pairs.length; + if (k < 2 || n === 0) return null; + + const index = new Map(scale.map((category, i) => [category, i])); + const counts: number[][] = Array.from({ length: k }, () => Array(k).fill(0)); + for (const [a, b] of pairs) { + const i = index.get(a); + const j = index.get(b); + if (i == null || j == null) return null; + counts[i][j] += 1; + } + + const p = counts.map(row => row.map(c => c / n)); + const rowMarginal = p.map(row => row.reduce((s, v) => s + v, 0)); + const colMarginal = scale.map((_, j) => p.reduce((s, row) => s + row[j], 0)); + const w = (i: number, j: number) => 1 - linearDisagreement(i, j, k); + + let observed = 0; + let expected = 0; + for (let i = 0; i < k; i++) { + for (let j = 0; j < k; j++) { + observed += p[i][j] * w(i, j); + expected += rowMarginal[i] * colMarginal[j] * w(i, j); + } + } + + const denominator = 1 - expected; + if (Math.abs(denominator) < 1e-12) return null; + const kappa = (observed - expected) / denominator; + + const rowWeight = scale.map((_, i) => + scale.reduce((s, _c, j) => s + colMarginal[j] * w(i, j), 0), + ); + const colWeight = scale.map((_, j) => + scale.reduce((s, _c, i) => s + rowMarginal[i] * w(i, j), 0), + ); + let sum = 0; + for (let i = 0; i < k; i++) { + for (let j = 0; j < k; j++) { + const term = w(i, j) - (rowWeight[i] + colWeight[j]) * (1 - kappa); + sum += p[i][j] * term * term; + } + } + const centre = kappa - expected * (1 - kappa); + const variance = Math.max(0, (sum - centre * centre) / (n * denominator * denominator)); + const se = Math.sqrt(variance); + + return { + kappa, + se, + ci: [Math.max(-1, kappa - CI_Z * se), Math.min(1, kappa + CI_Z * se)], + observed, + expected, + }; +} + +export function getKappaInterpretation(kappa: number | null): string { + if (kappa == null) return 'N/A'; + return KAPPA_BANDS.find(band => kappa >= band.min)?.label ?? 'Poor'; +} + +/** + * Summarise one level of comparison (domain judgements, overall judgement or + * signaling questions) from its pairs. + */ +export function summarizeLevel( + pairs: RatingPair[], + scale: string[] | null, + items: ItemDefinition[], +): LevelStats { + const itemStats = new Map( + items.map(item => [item.key, { ...item, compared: 0, agreed: 0 }]), + ); + const compared: Array<[string, string]> = []; + let agreed = 0; + let oneSided = 0; + let excluded = 0; + + for (const pair of pairs) { + const cls = classifyPair(pair, scale); + if (cls === 'excluded') { + excluded += 1; + continue; + } + if (cls === 'one-sided') { + oneSided += 1; + continue; + } + const a = pair.a as string; + const b = pair.b as string; + compared.push([a, b]); + const item = itemStats.get(pair.item); + if (item) item.compared += 1; + if (a === b) { + agreed += 1; + if (item) item.agreed += 1; + } + } + + let matrix: number[][] | null = null; + if (scale && compared.length > 0) { + const index = new Map(scale.map((c, i) => [c, i])); + matrix = scale.map(() => Array(scale.length).fill(0)); + for (const [a, b] of compared) matrix[index.get(a)!][index.get(b)!] += 1; + } + + return { + scale, + compared: compared.length, + agreed, + percentAgreement: compared.length > 0 ? (agreed / compared.length) * 100 : null, + oneSided, + excluded, + kappa: scale && compared.length >= MIN_PAIRS_FOR_KAPPA ? weightedKappa(compared, scale) : null, + matrix, + items: Array.from(itemStats.values()), + }; +} diff --git a/packages/shared/src/checklists/reliability/types.ts b/packages/shared/src/checklists/reliability/types.ts new file mode 100644 index 000000000..5a63d5386 --- /dev/null +++ b/packages/shared/src/checklists/reliability/types.ts @@ -0,0 +1,31 @@ +/** + * Shared shapes for the per-tool reliability adapters. + */ + +import type { ItemDefinition, RatingPair } from './stats.js'; + +export type ReliabilityToolType = 'ROB2' | 'ROBINS_I' | 'AMSTAR2'; + +export interface ToolPairs { + /** Domain judgements (RoB 2, ROBINS-I) or item answers (AMSTAR 2). */ + judgements: RatingPair[]; + /** One pair per checklist pair: the overall judgement or confidence rating. */ + overall: RatingPair[]; + /** Signaling questions; empty for tools without them. */ + questions: RatingPair[]; +} + +export interface ToolReliabilityDefinition { + type: ReliabilityToolType; + label: string; + /** What one reviewer pair appraises: an outcome or a whole study. */ + unit: 'outcome' | 'study'; + judgementLabel: string; + judgementScale: string[]; + overallScale: string[]; + items: ItemDefinition[]; + hasQuestions: boolean; + /** Tool-specific rules the explanation must state, in plain language. */ + notes: string[]; + extractPairs: (a: unknown, b: unknown) => ToolPairs; +} diff --git a/packages/web/src/components/FeatureShowcase.tsx b/packages/web/src/components/FeatureShowcase.tsx index 638f3a126..e67d2ce17 100644 --- a/packages/web/src/components/FeatureShowcase.tsx +++ b/packages/web/src/components/FeatureShowcase.tsx @@ -549,7 +549,7 @@ export default function FeatureShowcase() { illustration: , bullets: [ 'Independent ratings with blinded mode', - 'Automatic inter-rater agreement for AMSTAR 2 appraisals', + 'Automatic inter-rater agreement for every appraisal tool', 'Live, real-time collaboration with instant updates', ], }, diff --git a/packages/web/src/components/project/overview-tab/OverviewTab.tsx b/packages/web/src/components/project/overview-tab/OverviewTab.tsx index 94c3a05c6..27431d71a 100644 --- a/packages/web/src/components/project/overview-tab/OverviewTab.tsx +++ b/packages/web/src/components/project/overview-tab/OverviewTab.tsx @@ -13,10 +13,7 @@ import { project } from '@/project'; import { useProjectContext, type ProjectMember } from '../ProjectContext'; import { Collapsible, CollapsibleTrigger, CollapsibleContent } from '@/components/ui/collapsible'; import { CHECKLIST_STATUS } from '@corates/shared/checklists'; -import { - calculateInterRaterReliability, - type InterRaterMetrics, -} from '@/lib/inter-rater-reliability.js'; +import { calculateProjectReliability } from '@corates/shared/checklists/reliability'; import { ChartSection } from './ChartSection'; import { ResultsTables } from './ResultsTables'; import { ProgressSection } from './ProgressSection'; @@ -93,7 +90,7 @@ export function OverviewTab() { return map; }, [studies]); - const interRaterMetrics: InterRaterMetrics = useMemo(() => { + const reliability = useMemo(() => { // getData throws while the pool has no active connection (a cold refresh // renders this tab from cached rows before the gate's effects run) -- // treat that window as "no data" rather than crashing into the section @@ -105,7 +102,7 @@ export function OverviewTab() { return null; } }; - return calculateInterRaterReliability(studies, getChecklistData); + return calculateProjectReliability(studies, getChecklistData); }, [studies]); return ( @@ -118,7 +115,7 @@ export function OverviewTab() { - {interRaterMetrics.studyCount > 0 && } + {reliability.length > 0 && } {!empty && (
diff --git a/packages/web/src/components/project/overview-tab/ReliabilityAboutDialog.tsx b/packages/web/src/components/project/overview-tab/ReliabilityAboutDialog.tsx new file mode 100644 index 000000000..a32e24e0b --- /dev/null +++ b/packages/web/src/components/project/overview-tab/ReliabilityAboutDialog.tsx @@ -0,0 +1,215 @@ +/** + * ReliabilityAboutDialog - spells out how one tool's reliability numbers were + * produced. Every figure and rule comes from the stats object and the tool + * definition, so the explanation cannot drift from the calculation. + */ + +import { + CI_Z, + KAPPA_BANDS, + MIN_PAIRS_FOR_KAPPA, + linearDisagreement, + type LevelStats, + type ToolReliability, +} from '@corates/shared/checklists/reliability'; +import { + Dialog, + DialogContent, + DialogDescription, + DialogHeader, + DialogTitle, +} from '@/components/ui/dialog'; +import { formatPercent } from './ReliabilitySection'; + +interface ReliabilityAboutDialogProps { + tool: ToolReliability; + open: boolean; + onOpenChange: (open: boolean) => void; +} + +function Section({ title, children }: { title: string; children: React.ReactNode }) { + return ( +
+

{title}

+
{children}
+
+ ); +} + +function plural(n: number, singular: string, pluralForm = `${singular}s`): string { + return `${n} ${n === 1 ? singular : pluralForm}`; +} + +function ConfusionMatrix({ level }: { level: LevelStats }) { + if (!level.scale || !level.matrix) return null; + const { scale, matrix } = level; + return ( + + + + + + ))} + + + + {scale.map((rowCategory, i) => ( + + + {matrix[i].map((count, j) => ( + + ))} + + ))} + +
+ Rows are one reviewer of each pair, columns the other. Matches sit on the diagonal. +
+ {scale.map(category => ( + + {category} +
+ {rowCategory} + + {count} +
+ ); +} + +function ScaleWeights({ scale }: { scale: string[] }) { + const steps = scale.length - 1; + const credit = (distance: number) => + `${Math.round((1 - linearDisagreement(0, distance, scale.length)) * 100)}%`; + return ( +
    +
  • Same category: 100% agreement
  • + {Array.from({ length: steps }, (_, k) => k + 1).map(distance => ( +
  • + {distance === 1 ? 'One step apart' : `${distance} steps apart`} (for example {scale[0]}{' '} + against {scale[distance]}): {credit(distance)} +
  • + ))} +
+ ); +} + +export function ReliabilityAboutDialog({ tool, open, onOpenChange }: ReliabilityAboutDialogProps) { + const { definition, judgements, overall, questions } = tool; + const unit = definition.unit === 'study' ? 'study' : 'outcome'; + const judgementNoun = definition.judgementLabel.toLowerCase(); + const kappa = judgements.kappa; + + return ( + + + + How reliability is calculated for {definition.label} + + Two reviewers per {unit}, using their own completed checklists, before reconciliation. + + + +
+
+

+ {plural(tool.cells, unit)} across {plural(tool.studies, 'study', 'studies')}. + Consensus checklists are not used, so the numbers do not change after reconciliation. +

+

+ {definition.judgementLabel} on the scale {definition.judgementScale.join(', ')}. +

+
    + {definition.notes.map(note => ( +
  • {note}
  • + ))} +
+
+ +
+

+ {judgements.agreed} of {judgements.compared} compared {judgementNoun} matched:{' '} + {formatPercent(judgements.percentAgreement)}. +

+
+ +
+

+ Agreement corrected for chance: kappa = (Po - Pe) / (1 - Pe). Po is the observed + agreement, Pe the agreement expected from how often each reviewer used each category. +

+

Near misses get partial credit:

+ + {kappa ? +

+ Po {kappa.observed.toFixed(3)}, Pe {kappa.expected.toFixed(3)}, kappa{' '} + {kappa.kappa.toFixed(3)}. 95% CI {kappa.ci[0].toFixed(2)} to{' '} + {kappa.ci[1].toFixed(2)}: kappa plus or minus {CI_Z} standard errors, SE{' '} + {kappa.se.toFixed(3)} (Fleiss, Cohen and Everitt, 1969). +

+ :

+ Shown after {MIN_PAIRS_FOR_KAPPA} compared {judgementNoun}. Undefined when every + rating is the same category. +

+ } +

+ Kappa drops when most judgements share one category, so read it with percent + agreement. Reviewer pairs vary between studies; the value is pooled across pairs. +

+

+ Bands (Landis and Koch, 1977):{' '} + {KAPPA_BANDS.filter(band => Number.isFinite(band.min)) + .map(band => `${band.label} from ${band.min}`) + .join(', ')} + , Poor below 0. +

+ {judgements.matrix && } +
+ +
+

Only pairs where both reviewers gave a substantive answer are compared.

+

+ {plural(judgements.excluded, 'pair')} left out: a reviewer had no answer, or the item + was not applicable to both. +

+

+ {plural(judgements.oneSided, 'pair')} not applicable to one reviewer only. Not counted + as disagreements; the earlier answer that caused the split is compared on its own. +

+
+ +
+

+ Compared on its own scale, {definition.overallScale.join(', ')}: {overall.agreed} of{' '} + {overall.compared} matched ({formatPercent(overall.percentAgreement)}). +

+
+ + {questions && ( +
+

+ {questions.agreed} of {questions.compared} matched ( + {formatPercent(questions.percentAgreement)}). {questions.oneSided} applicable to one + reviewer only, {questions.excluded} left out. +

+

+ An explicit not-applicable answer and a question skipped by branching count the + same. Options differ between questions, so only percent agreement is reported. +

+
+ )} +
+
+
+ ); +} diff --git a/packages/web/src/components/project/overview-tab/ReliabilitySection.tsx b/packages/web/src/components/project/overview-tab/ReliabilitySection.tsx index b4fa83695..1c199bec4 100644 --- a/packages/web/src/components/project/overview-tab/ReliabilitySection.tsx +++ b/packages/web/src/components/project/overview-tab/ReliabilitySection.tsx @@ -1,17 +1,38 @@ /** - * ReliabilitySection - reviewer agreement before reconciliation, as a strip - * of three tiles with the kappa value placed on the Landis and Koch scale + * ReliabilitySection - reviewer agreement before reconciliation, one card per + * appraisal tool. Each card has a strip of three tiles, a per-domain (or + * per-item) breakdown, and a dialog explaining the calculation from the same + * numbers. */ -import { getKappaInterpretation, type InterRaterMetrics } from '@/lib/inter-rater-reliability.js'; +import { useState } from 'react'; +import { CircleHelpIcon } from 'lucide-react'; +import { + getKappaInterpretation, + KAPPA_BANDS, + MIN_PAIRS_FOR_KAPPA, + type LevelStats, + type ToolReliability, +} from '@corates/shared/checklists/reliability'; +import { Button } from '@/components/ui/button'; import { COLORS } from '@/components/charts/chartConfigs'; +import { ReliabilityAboutDialog } from './ReliabilityAboutDialog'; interface ReliabilitySectionProps { - metrics: InterRaterMetrics; + tools: ToolReliability[]; } // Landis and Koch band boundaries, which getKappaInterpretation also uses -const KAPPA_TICKS = [0, 0.2, 0.4, 0.6, 0.8, 1]; +const KAPPA_TICKS = KAPPA_BANDS.map(band => band.min) + .filter(min => Number.isFinite(min)) + .concat(1) + .sort((a, b) => a - b); + +const TILE_CLASS = 'border-border flex flex-col gap-0.5 rounded-lg border px-3.5 py-3'; + +export function formatPercent(value: number | null): string { + return value == null ? 'N/A' : `${value.toFixed(1)}%`; +} function KappaScale({ kappa }: { kappa: number }) { const position = Math.min(Math.max(kappa, 0), 1) * 100; @@ -39,54 +60,150 @@ function KappaScale({ kappa }: { kappa: number }) { function Tile({ label, children }: { label: string; children: React.ReactNode }) { return ( -
+
{label} {children}
); } -export function ReliabilitySection({ metrics }: ReliabilitySectionProps) { - const { percentAgreement, cohensKappa, studyCount, totalComparisons, agreementCount } = metrics; +function KappaTile({ level }: { level: LevelStats }) { + const { kappa, compared } = level; + if (kappa) { + return ( + + + {kappa.kappa.toFixed(2)} + + {getKappaInterpretation(kappa.kappa)} + + + + 95% CI {kappa.ci[0].toFixed(2)} to {kappa.ci[1].toFixed(2)} + + + + ); + } + const short = MIN_PAIRS_FOR_KAPPA - compared; + return ( + + N/A + + {short > 0 ? + `Needs ${short} more ${short === 1 ? 'comparison' : 'comparisons'}` + : 'Undefined when every judgement is the same category'} + + + ); +} +function ItemBreakdown({ level }: { level: LevelStats }) { + const items = level.items.filter(item => item.compared > 0); + if (items.length === 0) return null; return ( -
-
-

- Inter-rater reliability -

- - Before reconciliation, across {studyCount} finalized{' '} - {studyCount === 1 ? 'study' : 'studies'} - +
    + {items.map(item => { + const percent = (item.agreed / item.compared) * 100; + return ( +
  • + + {item.label} + {Math.round(percent)}% + +
  • + ); + })} +
+ ); +} + +function scopeText(tool: ToolReliability): string { + const studies = `${tool.studies} ${tool.studies === 1 ? 'study' : 'studies'}`; + if (tool.definition.unit === 'study') return studies; + return `${tool.cells} ${tool.cells === 1 ? 'outcome' : 'outcomes'} across ${studies}`; +} + +function ToolCard({ tool }: { tool: ToolReliability }) { + const [aboutOpen, setAboutOpen] = useState(false); + const { definition, judgements, overall, questions } = tool; + const overallLabel = + definition.type === 'AMSTAR2' ? 'Overall confidence rating' : 'Overall judgement'; + + return ( +
+
+
+

{definition.label}

+ {scopeText(tool)} +
+
+
- + - {percentAgreement != null ? `${percentAgreement.toFixed(1)}%` : 'N/A'} + {formatPercent(judgements.percentAgreement)} - {agreementCount} of {totalComparisons} judgements matched + {judgements.agreed} of {judgements.compared} matched - - {cohensKappa != null ? - <> - - {cohensKappa.toFixed(2)} - - {getKappaInterpretation(cohensKappa)} - - - - - : N/A} - - - {totalComparisons} - Two reviewers, same item + + + + {formatPercent(overall.percentAgreement)} + + + {overall.agreed} of {overall.compared} matched +
+ + + + {questions && questions.compared > 0 && ( +

+ Signaling questions: {formatPercent(questions.percentAgreement)} agreement across{' '} + {questions.compared} compared + {questions.oneSided > 0 && `, ${questions.oneSided} applicable to one reviewer only`} +

+ )} + + +
+ ); +} + +export function ReliabilitySection({ tools }: ReliabilitySectionProps) { + return ( +
+
+

+ Inter-rater reliability +

+ + Before reconciliation, pooled across reviewer pairs + +
+
+ {tools.map(tool => ( + + ))} +
); } diff --git a/packages/web/src/components/project/overview-tab/__tests__/ReliabilitySection.test.tsx b/packages/web/src/components/project/overview-tab/__tests__/ReliabilitySection.test.tsx new file mode 100644 index 000000000..06a5645bd --- /dev/null +++ b/packages/web/src/components/project/overview-tab/__tests__/ReliabilitySection.test.tsx @@ -0,0 +1,102 @@ +import { describe, it, expect } from 'vitest'; +import '@testing-library/jest-dom/vitest'; +import { render, screen, fireEvent } from '@testing-library/react'; +import { + RELIABILITY_TOOLS, + summarizeLevel, + MIN_PAIRS_FOR_KAPPA, + type ToolReliability, +} from '@corates/shared/checklists/reliability'; +import { ReliabilitySection } from '../ReliabilitySection'; + +function repeat(value: T, times: number): T[] { + return Array.from({ length: times }, () => value); +} + +function rob2Tool(): ToolReliability { + const definition = RELIABILITY_TOOLS.ROB2; + const judgements = summarizeLevel( + [ + ...repeat({ item: 'domain1', a: 'Low', b: 'Low' }, MIN_PAIRS_FOR_KAPPA), + ...repeat({ item: 'domain3', a: 'High', b: 'High' }, 4), + ...repeat({ item: 'domain3', a: 'Low', b: 'High' }, 4), + { item: 'domain4', a: null, b: 'Low' }, + { item: 'domain4', a: 'NA', b: 'Low' }, + ], + definition.judgementScale, + definition.items, + ); + const overall = summarizeLevel( + [ + { item: 'overall', a: 'Low', b: 'Low' }, + { item: 'overall', a: 'Low', b: 'High' }, + ], + definition.overallScale, + [{ key: 'overall', label: 'Overall', title: 'Overall judgement' }], + ); + const questions = summarizeLevel( + [ + ...repeat({ item: 'd1_1', a: 'Y', b: 'Y' }, 9), + { item: 'd1_2', a: 'Y', b: 'N' }, + { item: 'd3_2', a: 'NA', b: 'Y' }, + ], + null, + [], + ); + return { definition, cells: 8, studies: 6, judgements, overall, questions }; +} + +describe('ReliabilitySection', () => { + it('renders one card per tool with agreement, kappa, overall and breakdown', () => { + render(); + + expect(screen.getByRole('heading', { name: 'RoB 2' })).toBeInTheDocument(); + expect(screen.getByText('8 outcomes across 6 studies')).toBeInTheDocument(); + expect(screen.getByText('85.7%')).toBeInTheDocument(); + expect(screen.getByText('24 of 28 matched')).toBeInTheDocument(); + expect(screen.getByText('50.0%')).toBeInTheDocument(); + expect(screen.getByText('1 of 2 matched')).toBeInTheDocument(); + + const kappaTile = screen.getByText('Weighted kappa').parentElement!; + expect(kappaTile).toHaveTextContent(/95% CI -?\d\.\d\d to \d\.\d\d/); + + const breakdown = screen.getByRole('list', { name: 'Agreement by item' }); + expect(breakdown).toHaveTextContent('D1100%'); + expect(breakdown).toHaveTextContent('D350%'); + expect(breakdown).not.toHaveTextContent('D4'); + + expect( + screen.getByText( + 'Signaling questions: 90.0% agreement across 10 compared, 1 applicable to one reviewer only', + ), + ).toBeInTheDocument(); + }); + + it('explains a missing kappa by how many comparisons are still needed', () => { + const tool = rob2Tool(); + tool.judgements = summarizeLevel( + [{ item: 'domain1', a: 'Low', b: 'Low' }], + tool.definition.judgementScale, + tool.definition.items, + ); + render(); + expect( + screen.getByText(`Needs ${MIN_PAIRS_FOR_KAPPA - 1} more comparisons`), + ).toBeInTheDocument(); + }); + + it('opens a dialog with the tool notes and the confusion matrix', () => { + render(); + fireEvent.click(screen.getByRole('button', { name: 'How this is calculated' })); + + const dialog = screen.getByRole('dialog', { + name: 'How reliability is calculated for RoB 2', + }); + for (const note of RELIABILITY_TOOLS.ROB2.notes) expect(dialog).toHaveTextContent(note); + + const matrix = screen.getByRole('table'); + const rows = matrix.querySelectorAll('tbody tr'); + expect(rows[0]).toHaveTextContent('Low2004'); + expect(rows[2]).toHaveTextContent('High004'); + }); +}); diff --git a/packages/web/src/lib/inter-rater-reliability.ts b/packages/web/src/lib/inter-rater-reliability.ts deleted file mode 100644 index c487ee88a..000000000 --- a/packages/web/src/lib/inter-rater-reliability.ts +++ /dev/null @@ -1,228 +0,0 @@ -/** - * Inter-rater Reliability Calculation Utilities - * - * Calculates simple percent agreement and Cohen's Kappa for AMSTAR2 checklists - * across dual-reviewer studies. - */ - -import { CHECKLIST_STATUS } from '@corates/shared/checklists'; -import type { Study } from '@corates/shared/checklists'; -import { getAnswers, getQuestionKeys } from '@corates/shared/checklists/amstar2'; -import type { AMSTAR2Checklist } from '@corates/shared/checklists'; - -interface ChecklistData { - answers?: Record; -} - -interface Comparison { - questionKey: string; - reviewer1: string; - reviewer2: string; - agree: boolean; -} - -export interface InterRaterMetrics { - percentAgreement: number | null; - cohensKappa: number | null; - studyCount: number; - totalComparisons: number; - agreementCount: number; -} - -/** - * Calculate inter-rater reliability metrics for all eligible studies - */ -export function calculateInterRaterReliability( - studies: Study[] | null | undefined, - getChecklistData: - ((studyId: string, checklistId: string) => ChecklistData | null | undefined) | null | undefined, -): InterRaterMetrics { - if (!studies || !Array.isArray(studies) || studies.length === 0) { - return { - percentAgreement: null, - cohensKappa: null, - studyCount: 0, - totalComparisons: 0, - agreementCount: 0, - }; - } - - // Filter studies with dual reviewers - const dualReviewerStudies = studies.filter(s => s.reviewer1 && s.reviewer2); - - if (dualReviewerStudies.length === 0) { - return { - percentAgreement: null, - cohensKappa: null, - studyCount: 0, - totalComparisons: 0, - agreementCount: 0, - }; - } - - // Collect all question comparisons across all studies - const allComparisons: Comparison[] = []; - let eligibleStudyCount = 0; - - for (const study of dualReviewerStudies) { - const checklists = study.checklists || []; - - // Find 2 reviewer-completed AMSTAR2 checklists (one per reviewer) - const completedChecklists = checklists.filter( - c => c.status === CHECKLIST_STATUS.REVIEWER_COMPLETED && c.type === 'AMSTAR2', - ); - - // Must have exactly 2 completed checklists - if (completedChecklists.length !== 2) continue; - - // Verify one is from reviewer1 and one is from reviewer2 - const reviewer1Checklist = completedChecklists.find(c => c.assignedTo === study.reviewer1); - const reviewer2Checklist = completedChecklists.find(c => c.assignedTo === study.reviewer2); - - if (!reviewer1Checklist || !reviewer2Checklist) continue; - - // Get checklist data - const checklist1Data = getChecklistData?.(study.id, reviewer1Checklist.id); - const checklist2Data = getChecklistData?.(study.id, reviewer2Checklist.id); - - if (!checklist1Data?.answers || !checklist2Data?.answers) continue; - - // Extract answers using getAnswers function - const answers1 = getAnswers(checklist1Data.answers as unknown as AMSTAR2Checklist); - const answers2 = getAnswers(checklist2Data.answers as unknown as AMSTAR2Checklist); - - if (!answers1 || !answers2) continue; - - // Get question keys (q1-q16, with q9 and q11 consolidated) - const questionKeys = getQuestionKeys(); - - // Compare each question - for (const questionKey of questionKeys) { - const answer1 = answers1[questionKey]; - const answer2 = answers2[questionKey]; - - // Skip if either answer is missing/null - if (answer1 == null || answer2 == null) continue; - - allComparisons.push({ - questionKey, - reviewer1: answer1, - reviewer2: answer2, - agree: answer1 === answer2, - }); - } - - eligibleStudyCount++; - } - - if (allComparisons.length === 0) { - return { - percentAgreement: null, - cohensKappa: null, - studyCount: 0, - totalComparisons: 0, - agreementCount: 0, - }; - } - - // Calculate percent agreement - const agreements = allComparisons.filter(c => c.agree).length; - const percentAgreement = (agreements / allComparisons.length) * 100; - - // Calculate Cohen's Kappa - const cohensKappa = calculateCohensKappa(allComparisons); - - return { - percentAgreement, - cohensKappa, - studyCount: eligibleStudyCount, - totalComparisons: allComparisons.length, - agreementCount: agreements, - }; -} - -/** - * Calculate Cohen's Kappa from comparison data - */ -function calculateCohensKappa(comparisons: Comparison[]): number | null { - if (!comparisons || comparisons.length === 0) return null; - - // Get all unique answer values - const allAnswers = new Set(); - comparisons.forEach(c => { - allAnswers.add(c.reviewer1); - allAnswers.add(c.reviewer2); - }); - - const answerCategories = Array.from(allAnswers).sort(); - - if (answerCategories.length === 0) return null; - - // Build confusion matrix: [reviewer1][reviewer2] = count - const matrix: Record> = {}; - answerCategories.forEach(a1 => { - matrix[a1] = {}; - answerCategories.forEach(a2 => { - matrix[a1][a2] = 0; - }); - }); - - comparisons.forEach(c => { - matrix[c.reviewer1][c.reviewer2]++; - }); - - // Calculate observed agreement (P_o) - let observedAgreement = 0; - answerCategories.forEach(category => { - observedAgreement += matrix[category][category] || 0; - }); - const P_o = observedAgreement / comparisons.length; - - // Calculate expected agreement (P_e) from marginal distributions - const n = comparisons.length; - const reviewer1Marginals: Record = {}; - const reviewer2Marginals: Record = {}; - - answerCategories.forEach(category => { - reviewer1Marginals[category] = 0; - reviewer2Marginals[category] = 0; - }); - - comparisons.forEach(c => { - reviewer1Marginals[c.reviewer1]++; - reviewer2Marginals[c.reviewer2]++; - }); - - let expectedAgreement = 0; - answerCategories.forEach(category => { - const p1 = reviewer1Marginals[category] / n; - const p2 = reviewer2Marginals[category] / n; - expectedAgreement += p1 * p2; - }); - - const P_e = expectedAgreement; - - // Calculate Cohen's Kappa: k = (P_o - P_e) / (1 - P_e) - // Handle edge case where P_e is very close to 1 (numerical precision) - const denominator = 1 - P_e; - if (Math.abs(denominator) < 1e-10) { - // Perfect expected agreement means no variance, return 1 if observed is also perfect - return Math.abs(P_o - 1) < 1e-10 ? 1 : null; - } - - const kappa = (P_o - P_e) / denominator; - return kappa; -} - -/** - * Get interpretation text for Cohen's Kappa value - */ -export function getKappaInterpretation(kappa: number | null): string { - if (kappa == null) return 'N/A'; - if (kappa < 0) return 'Poor'; - if (kappa <= 0.2) return 'Slight'; - if (kappa <= 0.4) return 'Fair'; - if (kappa <= 0.6) return 'Moderate'; - if (kappa <= 0.8) return 'Substantial'; - return 'Almost Perfect'; -} From 50b945969c144a3c3c7ffea65dcd44ac59c411ba Mon Sep 17 00:00:00 2001 From: Jacob Maynard Date: Sat, 19 Sep 2026 17:44:42 -0500 Subject: [PATCH 2/2] Require two identified reviewers before counting a reliability pair A reviewer checklist created without an assignee can still reach reviewer-completed, since that status follows the study's reviewer slots. Two such checklists in one cell cannot be attributed to two people, so they no longer count as a pair. Claude-Session: https://claude.ai/code/session_01TAEtViwHmBCJSDSVKkzTD6 --- .../src/checklists/reliability/__tests__/adapters.test.ts | 8 ++++++-- packages/shared/src/checklists/reliability/index.ts | 6 +++++- 2 files changed, 11 insertions(+), 3 deletions(-) diff --git a/packages/shared/src/checklists/reliability/__tests__/adapters.test.ts b/packages/shared/src/checklists/reliability/__tests__/adapters.test.ts index 98867d954..0f284f726 100644 --- a/packages/shared/src/checklists/reliability/__tests__/adapters.test.ts +++ b/packages/shared/src/checklists/reliability/__tests__/adapters.test.ts @@ -191,11 +191,11 @@ describe('calculateProjectReliability', () => { return answers ? { answers } : null; }; - function reviewer(id: string, type: string, assignedTo: string, outcomeId: string | null) { + function reviewer(id: string, type: string, assignedTo: string | null, outcomeId: string | null) { return { id, type, assignedTo, outcomeId, status: CHECKLIST_STATUS.REVIEWER_COMPLETED }; } - it('pools cells per tool and skips consensus, single-reviewer and same-reviewer cells', () => { + it('pools cells per tool and skips consensus, single-reviewer, same-reviewer and unassigned cells', () => { data.set('r1', rob2({ domain1: { d1_1: 'Y', d1_2: 'Y', d1_3: 'N' } })); data.set('r2', rob2({ domain1: { d1_1: 'Y', d1_2: 'N', d1_3: 'N' } })); data.set('r3', rob2({ domain1: { d1_1: 'Y', d1_2: 'Y', d1_3: 'N' } })); @@ -226,6 +226,10 @@ describe('calculateProjectReliability', () => { id: 's3', checklists: [reviewer('r6', 'ROB2', 'u1', 'o4'), reviewer('r7', 'ROB2', 'u1', 'o4')], }, + { + id: 's4', + checklists: [reviewer('r8', 'ROB2', null, 'o5'), reviewer('r9', 'ROB2', null, 'o5')], + }, ]; const result = calculateProjectReliability(studies, getChecklistData); diff --git a/packages/shared/src/checklists/reliability/index.ts b/packages/shared/src/checklists/reliability/index.ts index a0f80b1d8..f4f417b57 100644 --- a/packages/shared/src/checklists/reliability/index.ts +++ b/packages/shared/src/checklists/reliability/index.ts @@ -66,7 +66,11 @@ export function calculateProjectReliability( c => !isReconciledChecklist(c) && c.status === CHECKLIST_STATUS.REVIEWER_COMPLETED, ); if (reviewers.length !== 2) continue; - if (reviewers[0].assignedTo && reviewers[0].assignedTo === reviewers[1].assignedTo) continue; + // A pair only counts when it can be attributed to two different people. + const [first, second] = reviewers; + if (!first.assignedTo || !second.assignedTo || first.assignedTo === second.assignedTo) { + continue; + } // Stable orientation for the matrix: rows are the lower user id. reviewers.sort((x, y) => (x.assignedTo ?? '').localeCompare(y.assignedTo ?? ''));