diff --git a/docs/branch-review-records/2d2b3c78d86d6c1599ede00358ad6ffc66a5adba0563fd5dee8656f0ef55ea87.record.md b/docs/branch-review-records/2d2b3c78d86d6c1599ede00358ad6ffc66a5adba0563fd5dee8656f0ef55ea87.record.md new file mode 100644 index 0000000000..bf90a1de87 --- /dev/null +++ b/docs/branch-review-records/2d2b3c78d86d6c1599ede00358ad6ffc66a5adba0563fd5dee8656f0ef55ea87.record.md @@ -0,0 +1 @@ +| 2026-08-21 | claude/gate-e-blinded-eval-b6076d | af8afeba09916f44462ba53609560f0118ca0940 | Gate E blinded-eval capture tooling: eval-answer-quality --extra-cases + gate-outcome dump fields, new scripts/blind-answer-pairs.ts build/unblind, tests, docs (PR #2208) | PR #2208 opened; offline-only, no retrieval behaviour change (#E0N0QC); paid v18-vs-v19 capture pending owner approval | focused vitest 36/36; offline contract 26 suites/627; adversarial fixtures 24 recorded + harness 25/25; check:rag:fixtures 36/26; typecheck 0; lint 0; docs checks green; full suite: load-flake timeouts only, disjoint sets, unrelated files | diff --git a/docs/rag-improvement/HANDOVER.md b/docs/rag-improvement/HANDOVER.md index 3ac6860afa..b383e9e5da 100644 --- a/docs/rag-improvement/HANDOVER.md +++ b/docs/rag-improvement/HANDOVER.md @@ -56,6 +56,11 @@ generation-quality verdict on fallback`), merged 2026-08-13 — structured `answerSections.maxItems` 6, adversarial baseline re-captured for v19 (`baseline-record.md` §4). Its canary pair 32100681177 -> 32111839806 is green, and `eval:answer-quality` was neutral within nondeterminism (owner blinded read pending). **Track A is complete with S3 (A4).** +- **Gate E tooling landed (2026-08-21, offline-only; `#E0N0QC`):** `eval-answer-quality` + gained `--extra-cases` (owner capture-only questions) and gate-outcome dump fields, and + `scripts/blind-answer-pairs.ts` builds the blinded reading pack / verdict sheet / + assignment key and unblinds verdicts. The paid v18-vs-v19 capture and the owner's blinded + read are still pending — procedure in §2a below; every provider step needs owner approval. - **Owner decisions 2026-08-17:** (1) **R1 before S2** — A2/A3 add answer length, and length under the still-unbudgeted strong retry pushes more dosing queries into `provider_timeout`, not fewer; (2) **governance Option B** for the document-summary `similarity: 1` question @@ -94,6 +99,7 @@ generation-quality verdict on fallback`), merged 2026-08-13 — structured | S6 | B3: Docling lab benchmark | `claude/packet-s6-docling-lab-d6foa6` | #2057 | Merged 2026-08-17 (merge `5a6418636`) | Offline only: `check:docling-lab` 36 fixtures / 10 hostile / 6 canaries + Gate B template valid; `verify:pr-local` heavy plan failed:(none); contract test 20/20; legacy smoke 46 docs, 10/10 hostile contained, canary-clean report. Verdict is a separate owner dispatch of `docling-lab.yml` | | S6b | Gate B run: docling-lab dispatch, verdict, decision record | `claude/docling-gate-b-eval-5czgln` | (this PR) | Gate B **PASS** 2026-08-18 (evidence run 32176604314 at `8a92378`); four latent harness defects found+fixed en route (setuptools pin, libGL, torch.compile/no-toolchain, HTML-entity scoring) | All five gates pass at pre-agreed 0 pp margins: parse 36/36 both engines, exactness 162/162 both, table F1 parity at ceiling (agreed 0 pp target; fixtures.v2 hardness follow-up queued), hostile 10/10 contained / 0 crash / 0 canary echo, resources max 12.6 s P95 / 1.40 GiB vs 120 s / 6 GiB caps; record: `docs/rag-improvement/gate-b-decision-record-2026-08-18.{md,json}`, validated `--final` | | S7 | B4: Docling worker shadow mode | `claude/docling-worker-shadow-mode-b6fa17` | #2170 | Merged 2026-08-19 (squash `5437c309f`), landed by content; default legacy — shadow is an operator flag (Railway); ledger #9DGA6R closed | Offline only: `WORKER_DOCUMENT_EXTRACTOR_MODE=legacy\ | shadow`(default legacy) +`WORKER_SHADOW_EXTRACTION_COHORT_PERCENT`(1–5, default 2) +`WORKER_DOCLING_PYTHON_BIN`; shadow runs after `commitDocumentIndexGeneration`, aggregate record in `documents.metadata.shadow_extraction` via the existing final metadata merge, no chunk/embedding/index/table-fact/`document_index_quality`writes; bounded 120 s / 40 pages / 1 process; vitest`tests/worker-shadow-extraction.test.ts`19/19 + Python unittest 7/7; caveats: table-heavy leg passed at parity-on-ceiling (fixtures.v2 first), eager-mode latency budgeted by the three bounds; rollback`WORKER_DOCUMENT_EXTRACTOR_MODE=legacy` | +| Gate E | Blinded before/after read tooling (capture `--extra-cases` + gate-outcome dump fields; `blind-answer-pairs.ts` build/unblind) | `claude/gate-e-blinded-eval-b6076d` | (this PR) | Tooling merged, offline-only; paid v18 (4ea310e48) vs v19 capture and the owner's blinded read still pending (`#E0N0QC`; procedure in §2a) | Offline only: blind-answer-pairs + eval-answer-quality focused suites 36/36 (blinding swap-proof, key round-trip, byte-stable builds); eval:rag:offline + eval:rag:adversarial:offline unchanged; no canary — no runtime behaviour change | | S8+ | B5 Ragas / B6 reranker / B7 DSPy | — | — | Still gated — owner decision | — | | #212 T1–T3 | Runtime row contracts (rag.ts, rag-candidate-sources.ts, src/app/api) — sibling stream sharing `src/lib/rag/**` | — | #1946 / #1981 / #2023 | Merged (T3 squash `440a34f71` 2026-08-17) | see the #212 ledger row; RAG surface complete for the cast class | | #212 T4 | Runtime row contracts: `worker/main.ts` (11 casts) — sibling stream | `claude/ledger-212-tranche-4-worker-q3y6i4` | #2037 | Merged 2026-08-17 (squash `1726537b7`); #212 closed by reconcile PR #2045 | Governance Preflight complete; audit: 1 inbound cast (claim rows, per-row fail-soft) + 2 read-back param casts contracted, 9 outbound/interop left; closes #212 (inbox `done` queued in the PR) | @@ -102,6 +108,38 @@ Update rule: the session that opens a packet's PR edits its row (branch, PR numb state) in the same PR. A later session updating another packet may also correct stale rows it can verify from GitHub/git state. Keep rows one line. +## 2a. Gate E blinded read — owner procedure + +Compares prompt v18 (commit `4ea310e48`, the S2 baseline canary half) against v19 (current +`main`) on the 30 `answerQualityEvalCases` plus up to ~10 owner-chosen live questions. +`scripts/eval-answer-quality.ts` and `scripts/eval-utils.ts` are byte-identical between +`4ea310e48` and current `main` and every new dump field is defensive, so the updated script +runs unmodified in a v18 worktree. Steps 3 and 4 are **provider-backed and paid** (OpenAI + +Supabase, ~80 cache-bypassed answers, est ~$4–8); everything else is offline. + +1. _(offline)_ Author `.local/gate-e/extra-cases.json` in the main checkout: + `{"questions":[{"id":"live-01","question":"..."}, ...]}` — ids `live-*`; the tool rejects + collisions with the fixed `quality-*` ids. +2. _(offline; handles credentials)_ `git worktree add ..\gate-e-v18 4ea310e48`, then in it + `npm ci --include=dev`; copy in from main: `scripts\eval-answer-quality.ts`, + `.local\gate-e\extra-cases.json`, and `.env.local` (never commit; delete before removing + the worktree). +3. **PROVIDER (paid)** — in the v18 worktree: + `npm run eval:answer-quality -- --json --extra-cases .local/gate-e/extra-cases.json --dump-answers output/gate-e/dump-v18.json > output/gate-e/summary-v18.json` +4. **PROVIDER (paid)** — in the main checkout at current HEAD: same command with + `dump-v19` in both paths. +5. _(offline)_ Copy the v18 dump into the main checkout's `output/gate-e/`, then: + `node scripts/run-tsx.mjs scripts/blind-answer-pairs.ts build --before output/gate-e/dump-v18.json --after output/gate-e/dump-v19.json --before-label v18-4ea310e48 --after-label v19- --out-dir output/gate-e/blind` + — check the printed pair count (expect 30 + the extra questions) and any unpaired ids. +6. _(offline, human)_ Open **only** `output/gate-e/blind/reading-pack.md`; record every + verdict in `verdict-sheet.md` (`verdict=A|B|tie|neither`, optional `notes=`). Do not open + `assignment-key.json` until every verdict is recorded. +7. _(offline)_ + `node scripts/run-tsx.mjs scripts/blind-answer-pairs.ts unblind --key output/gate-e/blind/assignment-key.json --verdicts output/gate-e/blind/verdict-sheet.md --out output/gate-e/blind/unblinded-report.md` +8. _(offline)_ Record the Gate E verdict (tallies + dump digests from the key) in the S2 and + Gate E rows above via a docs PR + ledger append; delete `.env.local` from the v18 + worktree, then `git worktree remove ..\gate-e-v18`. + ## 3. Session packets Every packet inherits the standing rules in §6. "Done" for a packet always ends at an open diff --git a/docs/scripts-index.md b/docs/scripts-index.md index a299169969..f939741d25 100644 --- a/docs/scripts-index.md +++ b/docs/scripts-index.md @@ -1,6 +1,6 @@ # Scripts index -Curated map of `scripts/` (249 files) and the `package.json` script surface (252 entries), +Curated map of `scripts/` (250 files) and the `package.json` script surface (252 entries), grouped by purpose. This is orientation, not an exhaustive per-file listing — the authoritative command list is `package.json`, and `npm run docs:check-scripts` verifies every `npm run ` referenced in docs resolves to a real script. `npm run docs:update` refreshes the exact counts above. @@ -97,7 +97,12 @@ validation of the synthetic adversarial fixture dataset and its baseline record; `eval-rag-adversarial-offline.mjs` (packet B2: fixture validation then the offline Vitest adversarial harness `tests/rag-adversarial-harness.test.ts`; `npm run eval:rag:adversarial:offline`, routed by `ci-change-scope.mjs` to RAG-surface PRs only; fails closed on missing fixture, -network attempt, or round-trip budget breach). +network attempt, or round-trip budget breach), +`blind-answer-pairs.ts` (Gate E offline blinded A/B pairing over two +`eval-answer-quality --dump-answers` artefacts — `build` emits reading-pack/verdict-sheet/ +assignment-key under `output/` or `.local/` only, `unblind` resolves recorded verdicts back to +version labels; pure file transformation, no provider access, node-builtin imports only; +`/issues` `#E0N0QC`; run via `node scripts/run-tsx.mjs scripts/blind-answer-pairs.ts`). Golden fixtures: `scripts/fixtures/rag-retrieval-golden.json`, `scripts/fixtures/assertion-golden.json`. Adversarial fixtures: `scripts/fixtures/rag-adversarial-cases.v1.json` (+ its schema) and diff --git a/scripts/blind-answer-pairs.ts b/scripts/blind-answer-pairs.ts new file mode 100644 index 0000000000..aa237dbe93 --- /dev/null +++ b/scripts/blind-answer-pairs.ts @@ -0,0 +1,473 @@ +// Gate E blinded before/after pairing (docs/rag-improvement/README.md §"Gates A–F"; ledger +// #E0N0QC). Takes two `eval-answer-quality --dump-answers` artefacts — a "before" and an +// "after" capture of the same question set — pairs cases by id, and emits three files: +// +// reading-pack.md the ONLY file the blinded reader opens: question + Answer A / Answer B +// (text, sections, citations, gate outcome), with no version labels, +// commit SHAs, models, timestamps, latencies, costs, or input paths. +// verdict-sheet.md one `: verdict= notes=` line per pair for the reader to fill +// (verdict = A | B | tie | neither). +// assignment-key.json the per-pair A→before|after mapping. The reader never opens this. +// +// `unblind` maps a filled verdict sheet back through the key into a labelled report. +// +// Pure offline file transformation: this script deliberately imports nothing from `src/` or +// other scripts (no `@/lib/*`, no `@next/env`), so it can never touch provider env or clients. +// The A/B assignment derives from the two input files' content digests, which appear only in +// the key file — the reading pack is byte-identical whichever way the inputs are labelled, so +// the assignment cannot be recovered from the pack alone (proven in +// tests/blind-answer-pairs.test.ts). +import { createHash } from "node:crypto"; +import { mkdirSync, readFileSync, writeFileSync } from "node:fs"; +import { dirname, isAbsolute, join, relative, resolve } from "node:path"; +import { pathToFileURL } from "node:url"; + +export const BLIND_TOOL_VERSION = "1"; + +export function sha256Hex(text: string): string { + return createHash("sha256").update(text, "utf8").digest("hex"); +} + +// Deliberate duplicate of resolveLocalDiagnosticOutputPath in scripts/eval-answer-quality.ts: +// that script must stay a single-file drop-in for the prompt-v18 capture worktree (Gate E owner +// procedure, docs/rag-improvement/HANDOVER.md) and this one must not import it, so the small +// guard is copied rather than shared. +export function resolveLocalArtifactPath(requestedPath: string, cwd = process.cwd()) { + const candidate = requestedPath.trim(); + if (!candidate) throw new Error("Artefact output path must not be empty."); + const resolvedPath = resolve(cwd, candidate); + const allowedRoots = [resolve(cwd, ".local"), resolve(cwd, "output")]; + const allowed = allowedRoots.some((root) => { + const fromRoot = relative(root, resolvedPath); + return Boolean(fromRoot) && !fromRoot.startsWith("..") && !isAbsolute(fromRoot); + }); + if (!allowed) throw new Error("Blind artefacts must be repo-local files under .local/ or output/."); + return resolvedPath; +} + +export type BlindGateOutcome = { + grounded: boolean | null; + tier: string | null; + degraded: boolean; + degraded_reason: string | null; + fallback_reason: string | null; + gate_reasons: string[]; + // Whether the dump record carried each key at all. An older capture script omits the newer + // keys entirely; the pack must not render a field one side recorded and the other did not, + // or the asymmetry becomes a systematic side marker (PR #2208 review). + recorded: { tier: boolean; degraded: boolean; fallback: boolean; gate_reasons: boolean }; +}; + +export type BlindNormalizedCase = { + id: string; + question: string; + answer: string; + sections: Array<{ heading: string; body: string }>; + citation_count: number | null; + // The sources the answer actually cites (dump field cited_sources); null when the capture + // predates that field. The broader retrieved-source diagnostics are never rendered. + cited_sources: Array<{ title: string; filename: string; page: number | null }> | null; + gate: BlindGateOutcome; +}; + +function asString(value: unknown): string { + return typeof value === "string" ? value : ""; +} + +export function parseAnswerDump(jsonText: string, label: string): BlindNormalizedCase[] { + let parsed: unknown; + try { + parsed = JSON.parse(jsonText); + } catch { + throw new Error(`${label}: dump is not valid JSON.`); + } + const cases = (parsed as { cases?: unknown } | null)?.cases; + if (!Array.isArray(cases) || cases.length === 0) { + throw new Error(`${label}: dump must contain a non-empty "cases" array.`); + } + const seen = new Set(); + return cases.map((entry, index) => { + const record = (entry ?? {}) as Record; + const id = asString(record.id).trim(); + if (!id) throw new Error(`${label}: case ${index + 1} has no id.`); + if (seen.has(id)) throw new Error(`${label}: duplicate case id ${id}.`); + seen.add(id); + const sections = Array.isArray(record.answer_sections) ? record.answer_sections : []; + const citedSources = Array.isArray(record.cited_sources) ? record.cited_sources : null; + const degraded = (record.degraded_mode ?? null) as { active?: unknown; reason?: unknown } | null; + const gateReasons = Array.isArray(record.generation_quality_gate_reasons) + ? record.generation_quality_gate_reasons + : []; + return { + id, + question: asString(record.question), + answer: asString(record.answer), + sections: sections.map((section) => { + const raw = (section ?? {}) as Record; + return { heading: asString(raw.heading), body: asString(raw.body) }; + }), + citation_count: typeof record.citation_count === "number" ? record.citation_count : null, + cited_sources: citedSources + ? citedSources.map((source) => { + const raw = (source ?? {}) as Record; + return { + title: asString(raw.title), + filename: asString(raw.filename), + page: typeof raw.page === "number" ? raw.page : null, + }; + }) + : null, + gate: { + grounded: typeof record.grounded === "boolean" ? record.grounded : null, + tier: asString(record.answer_quality_tier) || null, + degraded: degraded ? Boolean(degraded.active) : false, + degraded_reason: degraded ? asString(degraded.reason) || null : null, + fallback_reason: asString(record.fallback_reason) || null, + gate_reasons: gateReasons.filter((reason): reason is string => typeof reason === "string"), + recorded: { + tier: "answer_quality_tier" in record, + degraded: "degraded_mode" in record, + fallback: "fallback_reason" in record, + gate_reasons: "generation_quality_gate_reasons" in record, + }, + }, + }; + }); +} + +export function deriveAssignment(beforeText: string, afterText: string, caseIds: string[]) { + const beforeSha256 = sha256Hex(beforeText); + const afterSha256 = sha256Hex(afterText); + if (beforeSha256 === afterSha256) { + throw new Error("Before and after dumps are byte-identical — a blinded comparison is meaningless."); + } + // Swap-symmetric core: sorting the digests makes every derived value identical whichever + // input is labelled "before", so the reading pack is byte-stable under relabelling and the + // assignment is recoverable only with the key (which records the labels and digests). + const core = sha256Hex([beforeSha256, afterSha256].sort().join("\0")); + const lowSide: "before" | "after" = beforeSha256 < afterSha256 ? "before" : "after"; + const highSide: "before" | "after" = lowSide === "before" ? "after" : "before"; + const assignment = new Map(); + for (const caseId of caseIds) { + const bit = parseInt(sha256Hex(`${core}\0${caseId}`).slice(0, 2), 16) & 1; + assignment.set(caseId, bit === 0 ? lowSide : highSide); + } + return { assignment, beforeSha256, afterSha256 }; +} + +export type AssignmentKey = { + tool: "blind-answer-pairs"; + tool_version: string; + before_label: string; + after_label: string; + before_sha256: string; + after_sha256: string; + pair_count: number; + unpaired_before_ids: string[]; + unpaired_after_ids: string[]; + pairs: Array<{ id: string; a_is: "before" | "after" }>; +}; + +function assertBlindSafe(artifactName: string, text: string, forbidden: string[]) { + for (const token of forbidden) { + if (token && text.includes(token)) { + throw new Error( + `${artifactName} would contain a blinding token ("${token.slice(0, 12)}…") — pick more distinctive labels.`, + ); + } + } +} + +type PairRenderPlan = { + citedSources: boolean; + tier: boolean; + degraded: boolean; + fallback: boolean; + gateReasons: boolean; +}; + +// A field appears in the pack only when BOTH sides of the pair recorded it. Rendering +// "unknown" on one side against a concrete value on the other would be a systematic side +// marker that defeats the blinding (PR #2208 review). +function pairRenderPlan(a: BlindNormalizedCase, b: BlindNormalizedCase): PairRenderPlan { + return { + citedSources: a.cited_sources !== null && b.cited_sources !== null, + tier: a.gate.recorded.tier && b.gate.recorded.tier, + degraded: a.gate.recorded.degraded && b.gate.recorded.degraded, + fallback: a.gate.recorded.fallback && b.gate.recorded.fallback, + gateReasons: a.gate.recorded.gate_reasons && b.gate.recorded.gate_reasons, + }; +} + +function renderAnswerSide(letter: "A" | "B", side: BlindNormalizedCase, plan: PairRenderPlan): string[] { + const lines: string[] = [`### Answer ${letter}`, ""]; + lines.push(side.answer.trim() ? side.answer.trim() : "_(no answer text)_"); + for (const section of side.sections) { + lines.push("", `#### ${section.heading.trim() || "(untitled section)"}`, "", section.body.trim()); + } + lines.push("", "Citations:"); + if (plan.citedSources) { + const cited = side.cited_sources ?? []; + if (cited.length === 0) lines.push("- (none)"); + for (const citation of cited) { + const page = citation.page === null ? "" : `, p. ${citation.page}`; + lines.push(`- ${citation.title || "(untitled)"} (${citation.filename || "unknown file"}${page})`); + } + } else { + lines.push( + `- ${side.citation_count ?? "an unknown number of"} cited source(s); details not recorded in this capture`, + ); + } + const gate = side.gate; + const gateParts = [`grounded=${gate.grounded === null ? "unknown" : gate.grounded}`]; + if (plan.tier) gateParts.push(`tier=${gate.tier ?? "unspecified"}`); + if (plan.degraded) { + gateParts.push(`degraded=${gate.degraded}${gate.degraded_reason ? ` (${gate.degraded_reason})` : ""}`); + } + if (plan.fallback) gateParts.push(`fallback=${gate.fallback_reason ?? "none"}`); + if (plan.gateReasons) gateParts.push(`gate_reasons=[${gate.gate_reasons.join(", ")}]`); + lines.push("", `Gate outcome: ${gateParts.join(", ")}`); + return lines; +} + +export type BlindBuildInput = { + beforeText: string; + afterText: string; + beforeLabel: string; + afterLabel: string; +}; + +export function buildBlindArtifacts(input: BlindBuildInput) { + const beforeLabel = input.beforeLabel.trim(); + const afterLabel = input.afterLabel.trim(); + if (!beforeLabel || !afterLabel) throw new Error("Both --before-label and --after-label must be non-empty."); + if (beforeLabel === afterLabel) throw new Error("--before-label and --after-label must differ."); + const beforeCases = parseAnswerDump(input.beforeText, "--before"); + const afterCases = parseAnswerDump(input.afterText, "--after"); + const beforeById = new Map(beforeCases.map((entry) => [entry.id, entry] as const)); + const afterById = new Map(afterCases.map((entry) => [entry.id, entry] as const)); + const pairedIds = [...beforeById.keys()].filter((id) => afterById.has(id)).sort(); + if (pairedIds.length === 0) throw new Error("No case ids are present in both dumps — nothing to pair."); + const unpairedBeforeIds = [...beforeById.keys()].filter((id) => !afterById.has(id)).sort(); + const unpairedAfterIds = [...afterById.keys()].filter((id) => !beforeById.has(id)).sort(); + const { assignment, beforeSha256, afterSha256 } = deriveAssignment(input.beforeText, input.afterText, pairedIds); + + const packLines: string[] = [ + "# Gate E blinded reading pack", + "", + "For each pair below, read the question and both answers, then record a verdict in", + "`verdict-sheet.md` (verdict = A | B | tie | neither). Do not open `assignment-key.json`", + "until every verdict is recorded.", + "", + ]; + const sheetLines: string[] = [ + "# Gate E verdict sheet", + "", + "", + "", + "", + ]; + const normalizeQuestion = (value: string) => value.trim().replace(/\s+/g, " ").toLowerCase(); + pairedIds.forEach((id, index) => { + const before = beforeById.get(id)!; + const after = afterById.get(id)!; + if (normalizeQuestion(before.question) !== normalizeQuestion(after.question)) { + throw new Error(`Paired case ${id} has different question text in the two dumps — the captures do not match.`); + } + const plan = pairRenderPlan(before, after); + const aSide = assignment.get(id) === "before" ? before : after; + const bSide = assignment.get(id) === "before" ? after : before; + const pairNumber = String(index + 1).padStart(2, "0"); + packLines.push(`## Pair ${pairNumber} — ${id}`, "", `**Question:** ${aSide.question || bSide.question}`, ""); + packLines.push(...renderAnswerSide("A", aSide, plan), ""); + packLines.push(...renderAnswerSide("B", bSide, plan), ""); + packLines.push("### Verdict", "", `Record it in verdict-sheet.md under \`${id}\`.`, ""); + sheetLines.push(`${id}: verdict= notes=`); + }); + const readingPack = `${packLines + .join("\n") + .replace(/\n{3,}/g, "\n\n") + .trimEnd()}\n`; + const verdictSheet = `${sheetLines.join("\n").trimEnd()}\n`; + + const key: AssignmentKey = { + tool: "blind-answer-pairs", + tool_version: BLIND_TOOL_VERSION, + before_label: beforeLabel, + after_label: afterLabel, + before_sha256: beforeSha256, + after_sha256: afterSha256, + pair_count: pairedIds.length, + unpaired_before_ids: unpairedBeforeIds, + unpaired_after_ids: unpairedAfterIds, + pairs: pairedIds.map((id) => ({ id, a_is: assignment.get(id)! })), + }; + const assignmentKeyJson = `${JSON.stringify(key, null, 2)}\n`; + + for (const [name, text] of [ + ["reading-pack.md", readingPack], + ["verdict-sheet.md", verdictSheet], + ] as const) { + assertBlindSafe(name, text, [beforeLabel, afterLabel, beforeSha256, afterSha256]); + } + return { readingPack, verdictSheet, assignmentKeyJson, pairedIds, unpairedBeforeIds, unpairedAfterIds }; +} + +export type BlindVerdict = { verdict: "A" | "B" | "tie" | "neither"; notes: string }; + +export function parseVerdictSheet(text: string): Map { + const verdicts = new Map(); + text.split(/\r?\n/).forEach((rawLine, index) => { + const line = rawLine.trim(); + if (!line || line.startsWith("#") || line.startsWith("