diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a7deb70cb..8b449ad52 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -57,7 +57,7 @@ jobs: codex_autofix_changed: ${{ steps.scope.outputs.codex_autofix_changed }} build_changed: ${{ steps.scope.outputs.build_changed }} lockfile_changed: ${{ steps.scope.outputs.lockfile_changed }} - pr_policy_body_present: ${{ steps.scope.outputs.pr_policy_body_present }} + pr_policy_body_changed: ${{ steps.scope.outputs.pr_policy_body_changed }} steps: - name: Checkout uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 @@ -88,7 +88,7 @@ jobs: sync-pr-policy-body: name: Sync PR policy body needs: changes - if: github.event_name == 'pull_request' && needs.changes.outputs.pr_policy_body_present == 'true' + if: github.event_name == 'pull_request' && needs.changes.outputs.pr_policy_body_changed == 'true' runs-on: ubuntu-24.04 timeout-minutes: 5 permissions: diff --git a/docs/branch-review-ledger.md b/docs/branch-review-ledger.md index 25e18f4c6..7fd691776 100644 --- a/docs/branch-review-ledger.md +++ b/docs/branch-review-ledger.md @@ -899,3 +899,4 @@ Records before 2026-07-28 were written by hand and had drifted: 146 lines carrie | 2026-08-11 | claude/spacing-icon-design-review-rxwh28 | 5b96281ee7da817d5ce7f1102004ebe6f861b920 | pr-1815 heavy review-and-fix | remote already merged main (shadow-tight Switch kept); cherry-picked privacy -mb-4 reclaim + calculators dock cancel; removed duplicate UniversalSearchAlsoMatches; rail-aware section-sheet focus restore; dispositioned CodeRabbit docs/ledger/gates nits and outdated Sentry skeleton gap | verify:cheap PASS prior tip; verify:pr-local PASS prior tip; vitest privacy+in-page-nav 28 passed on cherry-pick; merge-tree clean vs origin/main | | 2026-08-12 | PR #1815 / claude/spacing-icon-design-review-rxwh28 | 9f266210f02081be54d407c70a85f52fed436128 | babysit | no remaining actionable findings; one pre-existing thread resolved as no-change (Dockerfile.worker follow-up needed) | required checks: Gitleaks PR policy PR required (all pass); targeted vitest passed: tests/document-frame-contract.test.ts + tests/in-page-nav-header.dom.test.tsx | | 2026-08-12 | 1815 | 27ce96e1755055ceee2eeae02d6efdf11259fcde | babysit | fixed | Unit coverage: targeted vitest passed: tests/shared-home-empty-state.dom.test.tsx (17 passed). PR required still blocked on pre-existing check failure at old remote head before sync. | +| 2026-08-12 | codex/pr-workflow-safety-230-296 | bc0a491fdf4146775629f9b2b03e2a2cc61bd7cb | pr-1830 unblock | unblocked: merged origin/main (outstanding-issues conflict), PR body RAG impact + governance, resolved Copilot thread; merge-tree clean; required CI in progress | check:outstanding-issues pass; evaluatePullRequestPolicy ok; merge-tree clean; PR policy/mergeability/Change scope in progress | diff --git a/docs/outstanding-issues.md b/docs/outstanding-issues.md index 4cadf4ea4..0604ffcb2 100644 --- a/docs/outstanding-issues.md +++ b/docs/outstanding-issues.md @@ -56,114 +56,111 @@ removed after current-main verification; it is not missing recommended work. | 1 | `#059` | A1 | Operator security + independent reviewer | Immediate approved security window | 1–3 hours plus verification | Verify every reported exposed credential (GitHub, OpenAI, Supabase service role/database, E2E) is retired; rotate anything still valid and update only intended secret stores. Never record values; stop before provider action without approval. | | 2 | `#053` | A1 | Operator — legal/privacy | Start now; finish before real patient use/privacy-approved release | 4–8 hours internal; 1–6 weeks elapsed | Execute DPAs; decide ZDR/residency; obtain cache behavior in writing; review subprocessors; obtain APP 8 and APP 5/1 counsel sign-off. Do not change public copy before approval. | | 3 | `#231` | A1 | Specialist — answer path + Operator | Immediate approved live investigation | 2–4 hours plus provider | Live answers degrade to source-only when `answerRouteBudgetMs.fast` (25000) binds while retrieval is healthy. Measure and fix the fast-route budget / generation timeout; keep conservative source-only fallback. **Gate:** focused answer-route tests offline first; live probe only with explicit provider approval. **Stop:** do not weaken quality gates to hide timeouts; flag RAG surfaces before edit. | -| 4 | `#207` | A1 | Specialist — clinical answer UI | Before AnswerCard / PR-13 shell adoption | 2–4 hours | AnswerState has no ungrounded-answer channel — `grounded:false` / unsupported confidence currently maps to `ready`, which would silently retire the live "Review source match" gate when AnswerCard is adopted. Add the ungrounded channel and wire producers. **Gate:** focused AnswerState/DOM tests + clinical-governance review. **Stop:** do not adopt AnswerCard (`#216`) until this lands. | -| 5 | `#226` | A1 | High — phone chrome + clinical owner | After clinical owner decides notice-vs-runway | 2–4 hours | VerificationNotice pushes the phone short-answer runway past the in-flow activation band; CI pins were widened as a temporary contract. Decide notice vs runway with the clinical owner, then restore a deliberate budget. **Gate:** `verify:phone-chrome` / Production UI pins from CI only. **Stop:** never pin runway budgets from a developer machine. | -| 6 | `#024` | A2 | High — browser/Next diagnostics | Provider-free macOS Safari host available | 1–2 hours | Reproduce document-source fallbacks in Safari/STP without Playwright interception; capture `_rsc` response evidence. Treat as an app defect only if native Safari reproduces; otherwise return to the harness. Never suppress `pageerror` or change CORS without proof. | -| 7 | `#022` | A2 | Operator — clinical governance + Specialist | Policy implemented locally; hosted apply and human review pending | 1–2 hours apply; 0.5–1 day first ten | The auditable BMJ `third_party_reference_attested` policy, migration and top-ten evidence manifest are prepared without changing `clinical_validation_status=unverified`. A qualified operator must review evidence, apply the migration deliberately, attest eligible records, review the ten visible local documents, then remeasure warnings. | -| 8 | `#023` | A2 | Specialist — RAG/browser diagnostics | After next weekly/manual matrix green (audit no longer blocks it) | 1–2 hours | Capture one Firefox/WebKit scheduled/manual datapoint and disposition the human irrelevant-at-10 labels. Matrix is structurally unblocked from blocking audit; do not spend on another RAG run. | -| 9 | `#018` | A2 | Specialist — clinical RAG/retrieval | Lithium closed; ADHD/metabolic evidence debt remains | Corpus/operator follow-up | Lithium's bounded subject/row-aware fix passed its targeted answer plus the full 36-case retrieval and 44-case answer canaries. ADHD's expected CAMHS document remains absent and the surfaced chart has no accessible table; metabolic schedule evidence remains unavailable and its standalone classifier candidate was reverted. | -| 10 | `#001` | A2 | Specialist — retrieval/ranking | After rollout approval | 0.5–1 day plus canary | Keep semantic reranking off unless an approved ambiguity comparison preserves 36/36, recall 1.0, zero per-case regressions, and shows measured gain; otherwise record keep-off and stop. | -| 11 | `#025` | A2 | Operator — Railway/GitHub/chat/Supabase | Next approved observability window | 1–3 hours/channel | Choose owned deployment, CI, ingestion, and SLO alerts; mock first, then one approved controlled provider event/channel. The merged Supabase trigger remains inert until its verified inputs are configured. Stop without an accountable responder. | -| 12 | `#055` | A2 | Specialist release owner + Operator | Before next full-confidence release/handoff | 2–4 hours plus runtime | On one exact SHA, run local/provider gates, Firefox/WebKit, required hosted CI, and close actionable GitHub threads. Stop at first failure and rerun only the repaired smallest gate. | -| 13 | `#056` | A2 | Operator — Supabase/Railway + Specialist | Next approved staging schema window | 2–4 hours | Reconcile the existing healthy, empty staging tier's 24-migration history gap using the exact repository migration chain, then re-run indexing, health, identity and data-boundary proof. Never recreate it or copy production clinical documents. | -| 14 | `#057` | A2 | High — release/SRE + Operator | After `#056` | 2–4 hours plus soak | Run documented staging soak and rollback against an exact candidate. Retain latency/error/rollback evidence; stop on unsafe data, identity mismatch, or unowned rollback. | -| 15 | `#011` | A3 | Operator — Supabase capacity | Immediately before first compute scale-up | 30–60 min plus observation | Switch Auth to percentage allocation, record before/after, and run approved advisor/health checks. Stop if no scale-up is planned. | -| 16 | `#147` | A3 | High — frontend layout/performance | Attribution done; the fix is next and is phone-chrome governed | 2–4 hours | Stop `usePhoneOverlayChromeReserve` publishing a stale 200px reserve it revises to 72px 15-60ms later — that round trip moves all main content down 128px and back, and is 100% of `/documents/search`'s CLS and ~75% of `/dsm`'s. Defer the first publish until the stack settles, or let the ResizeObserver be the only writer and trust the CSS seed (which is already correct) until it fires. Phone-chrome surface: read `docs/search-chrome-behaviour.md`, run `npm run verify:phone-chrome`, and produce a before/after CLS pair from the offline harness. **Stop:** do not re-dispatch the live workflow, do not read LCP from local runs, and do not change the CSS seed — it is not the cause. | -| 17 | `#117` | A3 | High — frontend/perf + product | After per-field card-vs-search decision | 0.5–1 day once fields decided | Cut `/therapy-compass` mobile LCP by trimming or deferring the 690 KB `therapies-index.json` prose payload. Confirm each long-form field is card-rendered, search-matched, or neither before dropping it. Gate: `check:therapy-data-index`, therapy Playwright journeys, `verify:lighthouse`. **Stop:** do not strip fields without that confirmation. | -| 18 | `#118` | A3 | High — CI/visual/perf gates | After `#147` and `#117` (do not bake current breaches) | 2–4 hours | Commit the CI-uploaded visual baselines under the platform-scoped snapshot path, run `check:lighthouse-budget -- --update` on a known-good build, then flip `enforce` so `visual-baseline` and `lighthouse-budget` block. **Stop:** never commit baselines from a developer machine; never enforce while `#147`/`#117` still breach. | -| 19 | `#033` | A3 | Specialist — prompt/source governance | After `#022` and explicit evaluation approval | 1–2 days plus approved eval | Design unknown-vs-adverse metadata wording and prompt tests. Require no supported-grounding drop and zero citation failures; stop on broad over-caveating or degradation. | -| 20 | `#013`, `#016` | A3 | High — bundling/runtime performance | After `#147`/`#117` or equivalent evidence | 0.5–2 days/route | Optimize only a production route with measured payload/render/motion harm. Require material gain plus focused, `verify:cheap`, and browser evidence; stop on small gain. | -| 21 | `#035` | A3 | Specialist — evidence rules | After a demonstrated missed conflict | 0.5–1 day design; code separate | Define a clinically reviewed conflict class with positive and negative fixtures. Stop if no bounded class can be shown; behavior change requires protected review. | -| 22 | `#027` | Optional | Operator — SRE/provider | When an owned external alert path is wanted | 1–2 hours | Decide vendor/cost/privacy/owner; if accepted, prove one non-PHI outage and recovery alert. Stop when no responder owns it. | -| 23 | `#039` | Optional | High — frontend architecture | During a concrete catalogue-toolbar project | 0.5–1 day inventory; 1–3 days code | Converge only repeated toolbar behavior without flattening search semantics. Stop when there is no bounded implementation target. | -| 24 | `#079` | Optional | High — repository hygiene | In explicitly scheduled batches | 30–60 minutes per batch | Disposition at most ten retained worktrees per pass using owner, PR, review-ledger, ancestry, and patch evidence. Preserve every dirty, active, secret-bearing, post-freeze, or ambiguous worktree and stop rather than broad-cleaning. | -| 25 | `#149` | A2 | High — install/gate integrity | Next verify-tooling pass | 1–3 hours | Widen `check:installed-lock-parity` beyond the seven top-level packages so transitive drift (e.g. `brace-expansion` CVE patch) fails closed. Prove with a fixture where only a nested dependency mismatches. **Stop:** do not weaken SessionStart skip-install behaviour without the wider check. | -| 26 | `#086` | A3 | High — repository structure + Specialist | On explicit go-ahead for X3; later packages own their gates | 1 PR per work order | Ship remaining maturity backlog (X3 rag.ts; X7 src/lib reorg; X6 coverage floors; X5 ACL consolidation; L1 one-shot archive; M1 host hardening) as verified draft PRs from `docs/maturity-backlog-workorders.md`. L4 ledger rotation shipped in #1418. Start with X3 after go-ahead; stop before RAG edits without the flag or X5 without live-DB approval. | -| 27 | `#098` | A3 | High — test infrastructure | Before `#099` or `#101`; it is their enabler | 2–4 hours | Generalise the answer-route preamble guard into a counting-proxy round-trip budget harness over the existing offline fixtures. Must enforce admission-before-scope, never the reverse. No providers, no DB. Stop if it would require live credentials. | -| 28 | `#102` | A3 | Operator — Supabase + Specialist | Next approved index window, after the ordering question is settled | 1–2 hours plus apply | Author the migration (operator SQL alone never reaches staging/DR/local replay), then apply → mirror `schema.sql` → regenerate drift manifest → register `required_indexes`. **Stop:** the RAG-path index is canary-gated, and ordering `fetchDocumentTitleAliasRows`'s unordered `.limit(12)` does not lift that — an imposed order can select a different twelve, so it is a second canary-gated change, not a way out of the first. The byte-identical claim was retracted. | -| 29 | `#099` | A3 | Specialist — answer path | After `#098` | Half a day per sub-item | Remaining fixed per-request round trips: the 8 `setCachedSearch` deferrals (abort semantics + mutation window), the anonymous subject+global limiter pair (needs a new atomic RPC first), and proxy→route identity duplication. Stop before hand-authoring locking SQL. | -| 30 | `#162` | A3 | High — frontend/UI | When starting the mode search redesign package | 0.5–1.5 days | Redesign `/tools?q=` as Compact Results Instrument (direction A): query-as-H1, one composer, dense tool rows, demote cross-mode cards, remove success-green filter banner and home hero on results. Comps in `public/mockups/mode-page-redesign-2026-07/tools-search/`. Verify phone+desktop chrome ownership and `verify:phone-chrome` / focused UI. Stop if scope expands into Tools home redesign without an explicit ask. | -| 31 | `#163` | A3 | High — frontend/UI | After or with `#162` | 0.5–1.5 days | Redesign `/services?q=` as Progressive Referral Workflow (direction B): H1 = query (not match count), progressive shortlist/compare (no always-on decision panel or giant step rail). Comps in `public/mockups/mode-page-redesign-2026-07/services-search/`. Verify referral shortlist still works; stop before changing Services home ModeHome. | -| 32 | `#164` | A3 | High — frontend/UI | Product confirmed Favourites is hybrid dashboard+search (no ModeHome) | 1–2 days | Redesign Favourites as one dashboard + search page: recommended Search-Led Workspace (direction B) — persistent search, sets as chips, Continue + recent + table on empty query, in-place filter on typed query. Comps in `public/mockups/mode-page-redesign-2026-07/favourites-hybrid/`. Do not reintroduce ModeHome for Favourites. Verify desktop+phone; stop before splitting into separate ModeHome routes. | -| 33 | `#202` | A3 | High — agent process | Next issues-skill touch | 30–60 min | Require recommendation/`/issues` answers to revalidate against `origin/main`'s ledger (or state checkout lag). Prevents re-proposing closed work from stale worktrees. | -| 34 | `#186` | A3 | Specialist — RAG ledger accuracy | Before any `#101` canary | 30–60 min | Rewrite open `#101` to credit PR 1474 hydration parallelisation and list only remaining canary-gated candidates. | -| 35 | `#187` | Optional | High — ledger hygiene | When writing the durable notes | 30–60 min | After one-line notes land for `#151`/`#154`, archive those process-lesson rows so the open table stays actionable. | -| 36 | `#090` | A3 | High — eslint toolchain | When ESLint 10 plugin peers are compatible | blocked; revisit monthly | Upgrade the eslint ecosystem to clear remaining dev-scoped high advisories — full `npm audit` reports zero high advisories from the eslint toolchain. | -| 37 | `#100` | A3 | Specialist — answer streaming | After offline Phase 0/1 design proof | provider-gated rollout | Buffered answer generation has no incremental verified delivery — [`verified-answer-incremental-delivery-design.md`](verified-answer-incremental-delivery-design.md) records the clinical-governance decision and staged co… | -| 38 | `#150` | Optional | Operator — review tooling | Next CodeRabbit billing/policy decision | 30–60 min decision | CodeRabbit reviewed none of a full day's PRs; spending cap reached — the repo's second automated reviewer is either funded or acknowledged as absent, rather than appearing to review while skipping. | -| 39 | `#152` | A2 | High — worktree hygiene | Next cleanup batch with #079 | 1–2 hours | Uncommitted work sits in worktrees whose branches are already merged — work that exists in no branch and no PR is either committed or knowingly discarded, not lost to a disk reclaim. | -| 40 | `#155` | A2 | High — agent process | Standing rule; next multi-agent session | process change | Several agent sessions edit the same branch and ledger concurrently — concurrent sessions stop silently undoing each other on shared `claude/*` branches and on this file. | -| 41 | `#159` | A3 | High — test hygiene | Next test-infra pass | 1–2 hours | Lists naming test files are duplicated, and the stale copy fails by running nothing — no gate, plan or config names a set of test files in a second place without being derived from the filesystem or asserted against it. | -| 42 | `#165` | A2 | High — clinical UI | Next answer-home UX pass | 0.5–1 day | Adopt a consolidated answer-home notice block — the studies exist, nothing adopts them — the answer hero states its safety obligation, its scope, and its verification requirement as one block in one voice. | -| 43 | `#166` | A2 | High — clinical safety UI | With #165 or next clinical chrome pass | 2–4 hours | Answer mode ships no verify-before-use caveat; every other clinical mode does — the surface that actually generates prose from retrieved sources says so, and says it must be checked. | -| 44 | `#168` | A3 | High — ledger architecture | With #156 / id-scheme redesign | design first | Sequential issue ids force every concurrent append to conflict — two sessions can append to this ledger at the same time without conflicting. | -| 45 | `#169` | A3 | High — git hygiene | Next branch cleanup batch | 1–2 hours | Local branches carry work that exists on no remote — committed work is not lost when a machine or worktree is reclaimed. | -| 46 | `#170` | A2 | High — phone UI | Next documents/filter phone pass | 0.5–1 day | Documents and therapy already have page-owned phone filter sheets; remaining modes still use inline controls — shared-band Filter+Sheet adoption without regressing those two or sheetless Sort. | -| 47 | `#171` | A2 | High — documents UI | With #170 / filter consolidation | 0.5–1.5 days | Documents mode has four overlapping filtering surfaces, two of them the same job — one Filter control opening one panel, so a reader learns filtering once. | -| 48 | `#175` | A2 | Operator — clinical data + Standard | Next therapy catalogue curation window | 2–4 hours | Therapy modality is now null on all 205 records and needs curation or removal — the Therapy detail and recommend screens either show a curated modality or stop carrying the field at all. | -| 49 | `#178` | A3 | High — PR policy | Next pr-policy change | 1–2 hours | pr-policy does not flag operational risk bundled with clinical or UI risk — a PR that mixes operational-risk paths with clinical or UI risk is called out before it merges, because squash-merging that mix destroys per-it… | -| 50 | `#189` | A2 | Specialist — search/RAG budgets | After #098 route residual; before collapsing RPCs | 2–4 hours + canary if behaviour | Pin /api/search route-level round trips and disposition the x3 text RPC probes — a counting-proxy budget drives `POST` `/api/search` (auth/ratelimit/scope/enrichment/telemetry), and the retrieval-core finding that `matc… | -| 51 | `#036` | Optional | Specialist — privacy/schema | When visibility model is redesigned | design + migration | No explicit `is_public` visibility flag on documents — Public-corpus visibility is implicit: `owner_id IS NULL` on an `indexed` document (`resolveSearchScope`). The `metadata.public_corpus` marker is written by the prom… | -| 52 | `#101` | A3 | Specialist — RAG/retrieval | After #186 update + canary approval | canary-gated | Canary-gated retrieval parallelisation candidates — independent retrieval stages stop running serially, proven by a live canary pair. Candidates: metadata/memory/visual hydration triples repeated on four branches (`rag.… | -| 53 | `#142` | Optional | High — docs hygiene | Next docs filing pass | 1–2 hours | Four loose dated docs need source and migration edits before they can be filed — every dated point-in-time doc lives in `docs/audit/` or `docs/archive/` as `docs/README.md` requires, not loose at the `docs/` top level. | -| 54 | `#151` | Optional | Operator — GitHub PAT | When writing the durable note (#187) | 15–30 min | `gh pr checks` cannot read CI, but the Actions API can — nobody concludes CI is unverifiable when it is merely reached through a different endpoint. | -| 55 | `#154` | Optional | High — agent process | When writing the durable note (#187) | 15–30 min | Row ids are not stable identifiers for "did my change land" — an agent confirms work reached `main` by content, never by id, title or PR state. | -| 56 | `#156` | A3 | High — ledger architecture | With #168 id-scheme work | design first | Outstanding-issues ids are still allocated read-modify-write, and Update-branch corrupts the merge — two branches cannot silently claim the same outstanding-issues id, and no merge path can commit a file where they have. | -| 57 | `#172` | A3 | High — documents UI | With #171 filter consolidation | 1–2 hours | `Sources` sits in the results bar but is navigation, not a filter — the results bar holds only controls that act on the current results. | -| 58 | `#174` | A3 | High — search facets | When facet UX is redesigned | 0.5–1 day | Facets AND within a group, so two values from one group almost always return nothing — a decision on record, either way. | -| 59 | `#177` | A3 | High — therapy catalogue build | Next therapy-index build change | 1–2 hours | Therapy catalogue aliases duplicate 2.53 MB of bytes instead of pointing at the hashed file — the unversioned catalogue aliases stop costing a second copy of every payload in the repo and the image. | -| 60 | `#179` | A3 | High — therapy catalogue build | With #177/#180 | 1–2 hours | The full therapy catalogue silently switched from minified to pretty-printed — the full catalogue's on-disk format is a decision someone made, not a side effect. | -| 61 | `#180` | A3 | High — therapy catalogue build | With #177/#179 | 1–2 hours | build-therapies-index now overwrites its own source input — the therapy catalogue generator has a source it does not also destroy. | -| 62 | `#181` | Optional | High — documents UI clarification | When updating #171 | 15–30 min | Correction to `#171`: source-type does NOT duplicate the `Document type` facet group — `#171` states that the documents source-type control "duplicates the facet group already named `Document type`". That is wrong, and … | -| 63 | `#188` | A3 | Operator — DR/SRE | After any schema restore drill, or next DR review | checklist-owned | Document and track disaster-recovery re-creation checklist as ledger work — the five DR items that do not survive a schema restore are tracked with owners and verify steps, not only in `docs/operator-backlog.md`. | -| 64 | `#190` | A3 | Specialist — RAG structure | On explicit X3 go-ahead | 1 PR per extraction unit | X3: Finish rag.ts monolith decomposition — `src/lib/rag/rag.ts` is decomposed into focused modules per `docs/maturity-backlog-workorders.md` X3, with existing offline RAG contracts green. | -| 65 | `#191` | A3 | Operator — DB + Specialist | Approved live-DB window only | provider-gated | X5: ACL-migration consolidation (provider-gated) — ACL-related migrations are consolidated per maturity work-order X5 without weakening owner-scope/RLS. | -| 66 | `#192` | A3 | High — test coverage | Next coverage-floor pass | 0.5–1 day | X6: Raise clinical/retrieval/answer coverage floors — coverage floors for clinical, retrieval, and answer domains meet the maturity X6 targets with CI enforcing them. | -| 67 | `#193` | A3 | High — src/lib structure | After/with X3 non-protected clusters | 1 PR per cluster | X7: Complete the remaining src/lib domain-directory reorg — remaining `src/lib` clusters sit in their domain directories per X7 follow-on to X2. | -| 68 | `#194` | A3 | High — scripts/docs hygiene | Next scripts archive pass | 1–2 hours | L1: Archive retired backfill one-shots and dead ci-change-scope token — retired `backfill:*` one-shots and the dead `ci-change-scope` token are archived/removed with docs/script index updated. | -| 69 | `#195` | A3 | Operator — GitHub maintainer | Maintainer UI window | 30–60 min | M1: Repo-host hardening (branch protection and required checks) — GitHub branch-protection rulesets and required checks match audit §8 / maturity M1. | -| 70 | `#196` | A3 | Operator — DR/SRE | After schema restore drill | 1–2 hours | DR: Re-create pg_cron schedules after schema restore — ingestion/retention and related pg_cron schedules exist on the target DB after any schema restore. | -| 71 | `#197` | A3 | Operator — DR/SRE | After schema restore drill | 30–60 min | DR: Re-add Vault secrets including cron_ingestion_jwt — required Vault secrets (at least `cron_ingestion_jwt`) are present after schema restore. | -| 72 | `#198` | A3 | Operator — DR/SRE | After schema restore drill | 30–60 min | DR: Re-set custom database GUCs after schema restore — custom `app.*` GUCs required by the app/worker are set on the restored database. | -| 73 | `#199` | A3 | Operator — DR/SRE | After schema restore; Deno v2 available | 1–2 hours | DR: Redeploy Supabase edge functions (Deno v2.x) — required edge functions are deployed to the target project with Deno v2.x. | -| 74 | `#200` | A3 | Operator — DR/SRE | After schema restore drill | 1–2 hours | DR: Re-enter dashboard config after schema restore — auth providers/SSO redirect URLs, connection-pool caps, per-project keys, and `E2E_USER_*` are re-entered in the Supabase/Railway dashboards after restore. | -| 75 | `#183` | A2 | Operator — Sentry + Specialist | Next approved observability window with SENTRY_AUTH_TOKEN | 1–2 hours | Create Sentry metric alert for production DB span p95 > 500ms (`span.op:db`, environment production). **Stop:** no secret printing; blocked until token/env available. | -| 76 | `#206` | A2 | Specialist — answer UI contract | With AnswerState producer work (`#207`) | 2–4 hours | `partial_retrieval` has no app-facing producer — decide RAG contract vs UI-only mapping before AnswerCard. **Stop:** no retrieval behaviour change without RAG flag. | -| 77 | `#208` | A2 | Specialist — clinical copy | With answer clipboard / PR-13 work | 1–2 hours | `answerClipboardText` must not replace `formatAnswerRenderCopyText` — compose render-policy warnings. **Gate:** focused clipboard/copy tests. **Stop:** do not drop render-policy caveats. | -| 78 | `#216` | A2 | High — design-system answer shell | After `#207` and clinical surface decision | 0.5–1 day | Adopt AnswerCard on the answer surface (deferred from PR-J). Own PR, own `verify:ui`. **Stop:** not before `#207`; show both surface treatments before choosing. | -| 79 | `#230` | A2 | High — PR policy / CI | Next ci.yml / pr-policy change | 1–2 hours | PR-policy body sync must no-op unless `PR_POLICY_BODY.md` is new in that PR's own diff (or move body out of repo). **Gate:** `check:github-actions` / workflow self-test. **Stop:** do not re-commit scratch bodies to main. | -| 80 | `#232` | A2 | High — review ledger hygiene | Next ledger touch for PR-J | 30–60 min | Supersede the PR-J clinical-governance ledger row so it describes the merged head (`ledger:append --supersede`). **Stop:** append-only — never edit/delete the old row. | -| 81 | `#209` | A3 | High — design tokens / contrast | Next Gate 1 / warning-token pass | 1–2 hours | Add contrast pair for `--warning` used as body text (VerificationNotice / DoseLine). **Gate:** design-system contrast checks. **Stop:** do not invent a new status token without TOKENS.md. | -| 82 | `#211` | A3 | High — TypeScript strictness | Dedicated migration branch | multi-PR | Plan and start `noUncheckedIndexedAccess` migration (1266 errors); highest-risk files first. **Stop:** do not flip the flag on main without a staged plan. | -| 83 | `#212` | A3 | High — runtime validation | After highest-risk cast inventory | multi-PR | Replace `as unknown as` and unvalidated `JSON.parse` with Zod/guards at trust boundaries. **Stop:** RAG/provider boundaries need clinical/privacy care. | -| 84 | `#213` | A3 | High — error handling | Next fetch/stream hardening pass | 0.5–1 day | Stop swallowing fetch/stream errors with empty catches; check `response.ok`. **Stop:** do not change telemetry contracts silently. | -| 85 | `#215` | Optional | High — image perf | Next image/PWA pass | 2–4 hours | Image-optimization basics for lightbox, PWA lifecycle, demo PNGs. **Stop:** optional until measured need. | -| 86 | `#221` | A3 | High — design-system convergence | After `#218` cn() decision | 0.5–1 day | Converge remaining local EmptyState/LoadingState/Chip duplicates. **Stop:** not piecemeal before cn()/Chip decisions. | -| 87 | `#222` | A3 | High — headers / search chrome | During headers redesign decision | 2–4 hours | Decide whether mode-home-template / search-results-header-band are in PageHeader scope or permanently out. **Stop:** do not flatten phone composer ownership. | -| 88 | `#233` | A3 | High — design-system docs | Next COMPONENTS.md docs PR | 1–2 hours | Refresh section 0 maturity matrix and document FormField optionality-marker contract. **Gate:** docs checks. **Stop:** docs-only; no product behaviour change. | -| 89 | `#234` | A3 | High — design-system docs | With answer-surface docs | 30–60 min | Document `answer-copy-payload.ts` as the clipboard contract for three surfaces. **Stop:** do not add a second copy builder. | -| 90 | `#235` | A3 | High — design-system evidence | Next warmed local proof-shot pass | 1–2 hours | Capture missing ADOPTION.md §7 proof shots for adopted surfaces. **Stop:** not visual-baseline PNGs (`#118`); no Playwright snapshot commit. | -| 91 | `#236` | Optional | High — branch hygiene | Next cleanup batch with `#079` | 30–60 min | Dispose orphan DS V2 builder branches and leftover wave-5 dev servers. **Stop:** content-verify before delete; no force-clean. | -| 92 | `#237` | A3 | High — design-system a11y | Before freezing Linux visual baselines (#242) | 30–60 min | Eyeball low-confidence AccessibleTable densities at 320px; MissingValue phrases must remain readable. **Gate:** visual spot-check only. **Stop:** do not abbreviate MissingValue to a dash. | -| 93 | `#238` | A3 | High — overlays/UI | After Sheet portal default change (#1616) | 30–60 min | Visual pass for Sheet portal default on settings, sidebar, and answer overlays under OverlayRoot. **Stop:** do not revert portal default without evidence. | -| 94 | `#239` | Optional | High — phone chrome | When phone orientation QA is available | 15–30 min | Manual phone rotation check for ResizeObserver-only phone chrome reserve. **Gate:** `verify:phone-chrome` still owns automated coverage. **Stop:** do not widen reserve heuristics without reproduction. | -| 95 | `#240` | Optional | High — design tokens | Next design-owner review | 15–30 min | Confirm tooltip visual hard-clip asymmetry with design owner (sr-only keeps full text). **Stop:** no product change without that confirmation. | -| 96 | `#241` | A3 | High — therapy catalogue | Standing; with any therapy-home change | 15–30 min | Therapy home summary count/slugs remain build-time; keep `build-therapies-index --check` load-bearing. **Stop:** do not bypass the check. | -| 97 | `#242` | A2 | High — design-system baselines | After human review of Linux baselines | 1–2 hours | Commit approved Linux visual baselines and promote adoption not-committed → committed. **Stop:** never commit baselines from an unreviewed machine run. | -| 98 | `#244` | A3 | High — design tokens / forced-colors | With any ckb-v2 forced-colours edit | 15–30 min | Keep grouped dark selectors in the forced-colours media block so specificity matches dark rules. **Stop:** do not trim to a single `.ckb-v2.ckb-v2` selector. | -| 99 | `#245` | A3 | High — cross-mode links | Next CrossModeLinks / analytics pass | 30–60 min | responsive-compact CrossModeLinks keeps duplicate rails in the DOM; prefer one mount or accept test double-counts. **Stop:** do not break phone-only rail contract. | -| 100 | `#248` | A2 | Operator — Supabase + Specialist | After PR #1614 symptom repair; approved live/history window | 1–2 hours | Investigate why 20260705180000 search-health indexes were missing on live despite applied history; decide if drift checks should catch this class. **Stop:** no hosted mutation without approval. | -| 101 | `#249` | A3 | High — agent process | Next issues-skill / plan touch | 1–2 hours | Extend issues/plan with an agent-safe wins classifier (optional filter; no new skill unless reused thrice). **Stop:** do not outrank A1 operator work. | -| 102 | `#250` | A2 | High — multi-agent execution | After Wave 0 queue repair on main (done in this capture); run remaining Wave 0/#202 process gates next on the engineering track | multi-wave | Execute the fastest-wins multi-wave plan (Waves 0–4 + operator track) with parallel agents and per-PR gates. Waves do not outrank A1 acuity. **Stop:** provider/RAG approvals still required where flagged. | -| 103 | `#251` | Optional | High — agent process | Next handoff/gates doc touch | 15–30 min | Handoff checklist pairs gates skill with verification-router; paste decisive proof line. **Stop:** do not stack broad gates by default. | -| 104 | `#253` | A2 | High — phone results UI | Next open-PR sweep | 15–30 min | Decide #1606's fate: the `MobileResultFilterControl` it rewrites was deleted by #247, so there is nothing left to hand-merge. Verify keyboard parity of the replacement sheet on a real device, then close #1606 as superseded. **Stop:** the decision is a human's; do not close #1606 automatically. | -| 105 | `#254` | A2 | Operator — Codex Cloud | Before #1617 leaves draft | 1–2 hours | Re-run Codex Cloud acceptance at the exact current head or mark head-independent evidence. **Stop:** do not treat stale pins as coverage. | -| 106 | `#256` | A2 | High — mode section nav | Next information-page / mode-nav pass | 2–4 hours | Declared information-page section sets whose target ids nothing renders — verify each set against the rendered DOM per route; render anchors or delete the set. **Stop:** do not audit by grepping for `id=` alone (sectionId props exist). | -| 107 | `#257` | Optional | High — formulation/specifiers flake | Standing until second reproduction | 15–30 min | Single unreproduced ui-formulation flake when run with ui-specifiers — record a second sighting only; do not quarantine until three on the same SHA. **Stop:** do not weaken assertions. | -| 108 | `#286` | A3 | High — in-page nav + frontend | After owner go-ahead for the information-page series | 1–2 days | Convert the six pill-rail information pages onto `InPageNavHeader`, widen Server Component–safe actions, then delete `informationPageSectionDefinitions`. **Gate:** focused DOM/contract tests + `verify:phone-chrome` for touched owners. **Stop:** do not convert DocumentViewer here; do not verify anchors by grepping `id=` alone. | -| 109 | `#287` | A3 | High — in-page nav + clinical owner | After `#286`; medications needs an owner product call | 0.5–1 day design + convert | Decide medications tab model, presentations MobileTabs vs `InPageNavHeader`, and factsheets heading→id scheme; convert or record lasting exceptions. **Stop:** do not port medications mechanically. | -| 110 | `#288` | Optional | High — document chrome | After `#286`/`#287`, or when declaring the series complete | 30–60 min | Confirm DocumentViewer non-adoption (already noted in `docs/search-chrome-behaviour.md`) as the final end state, or schedule a separate convergence PR that leaves pinned `--document-*` CSS names untouched. | -| 111 | `#289` | A2 | High — auth/identity | Next auth module touch | 1–2 hours | Export a named helper (e.g. `authorizationIdentity(headers)`) from the auth module and use it at every property-access call site; consider a lint rule or branded type so `.Authorization` stops type-checking at all. **Stop:** do not change `authorizationHeadersForAccessToken` to emit uppercase — lowercase is the correct Fetch/Headers convention and callers that pass the object wholesale to `fetch` depend on it. | +| 4 | `#024` | A2 | High — browser/Next diagnostics | Provider-free macOS Safari host available | 1–2 hours | Reproduce document-source fallbacks in Safari/STP without Playwright interception; capture `_rsc` response evidence. Treat as an app defect only if native Safari reproduces; otherwise return to the harness. Never suppress `pageerror` or change CORS without proof. | +| 5 | `#022` | A2 | Operator — clinical governance + Specialist | Policy implemented locally; hosted apply and human review pending | 1–2 hours apply; 0.5–1 day first ten | The auditable BMJ `third_party_reference_attested` policy, migration and top-ten evidence manifest are prepared without changing `clinical_validation_status=unverified`. A qualified operator must review evidence, apply the migration deliberately, attest eligible records, review the ten visible local documents, then remeasure warnings. | +| 6 | `#023` | A2 | Specialist — RAG/browser diagnostics | After next weekly/manual matrix green (audit no longer blocks it) | 1–2 hours | Capture one Firefox/WebKit scheduled/manual datapoint and disposition the human irrelevant-at-10 labels. Matrix is structurally unblocked from blocking audit; do not spend on another RAG run. | +| 7 | `#018` | A2 | Specialist — clinical RAG/retrieval | Lithium closed; ADHD/metabolic evidence debt remains | Corpus/operator follow-up | Lithium's bounded subject/row-aware fix passed its targeted answer plus the full 36-case retrieval and 44-case answer canaries. ADHD's expected CAMHS document remains absent and the surfaced chart has no accessible table; metabolic schedule evidence remains unavailable and its standalone classifier candidate was reverted. | +| 8 | `#001` | A2 | Specialist — retrieval/ranking | After rollout approval | 0.5–1 day plus canary | Keep semantic reranking off unless an approved ambiguity comparison preserves 36/36, recall 1.0, zero per-case regressions, and shows measured gain; otherwise record keep-off and stop. | +| 9 | `#025` | A2 | Operator — Railway/GitHub/chat/Supabase | Next approved observability window | 1–3 hours/channel | Choose owned deployment, CI, ingestion, and SLO alerts; mock first, then one approved controlled provider event/channel. The merged Supabase trigger remains inert until its verified inputs are configured. Stop without an accountable responder. | +| 10 | `#055` | A2 | Specialist release owner + Operator | Before next full-confidence release/handoff | 2–4 hours plus runtime | On one exact SHA, run local/provider gates, Firefox/WebKit, required hosted CI, and close actionable GitHub threads. Stop at first failure and rerun only the repaired smallest gate. | +| 11 | `#056` | A2 | Operator — Supabase/Railway + Specialist | Next approved staging schema window | 2–4 hours | Reconcile the existing healthy, empty staging tier's 24-migration history gap using the exact repository migration chain, then re-run indexing, health, identity and data-boundary proof. Never recreate it or copy production clinical documents. | +| 12 | `#057` | A2 | High — release/SRE + Operator | After `#056` | 2–4 hours plus soak | Run documented staging soak and rollback against an exact candidate. Retain latency/error/rollback evidence; stop on unsafe data, identity mismatch, or unowned rollback. | +| 13 | `#011` | A3 | Operator — Supabase capacity | Immediately before first compute scale-up | 30–60 min plus observation | Switch Auth to percentage allocation, record before/after, and run approved advisor/health checks. Stop if no scale-up is planned. | +| 14 | `#147` | A3 | High — frontend layout/performance | Attribution done; the fix is next and is phone-chrome governed | 2–4 hours | Stop `usePhoneOverlayChromeReserve` publishing a stale 200px reserve it revises to 72px 15-60ms later — that round trip moves all main content down 128px and back, and is 100% of `/documents/search`'s CLS and ~75% of `/dsm`'s. Defer the first publish until the stack settles, or let the ResizeObserver be the only writer and trust the CSS seed (which is already correct) until it fires. Phone-chrome surface: read `docs/search-chrome-behaviour.md`, run `npm run verify:phone-chrome`, and produce a before/after CLS pair from the offline harness. **Stop:** do not re-dispatch the live workflow, do not read LCP from local runs, and do not change the CSS seed — it is not the cause. | +| 15 | `#117` | A3 | High — frontend/perf + product | After per-field card-vs-search decision | 0.5–1 day once fields decided | Cut `/therapy-compass` mobile LCP by trimming or deferring the 690 KB `therapies-index.json` prose payload. Confirm each long-form field is card-rendered, search-matched, or neither before dropping it. Gate: `check:therapy-data-index`, therapy Playwright journeys, `verify:lighthouse`. **Stop:** do not strip fields without that confirmation. | +| 16 | `#118` | A3 | High — CI/visual/perf gates | After `#147` and `#117` (do not bake current breaches) | 2–4 hours | Commit the CI-uploaded visual baselines under the platform-scoped snapshot path, run `check:lighthouse-budget -- --update` on a known-good build, then flip `enforce` so `visual-baseline` and `lighthouse-budget` block. **Stop:** never commit baselines from a developer machine; never enforce while `#147`/`#117` still breach. | +| 17 | `#033` | A3 | Specialist — prompt/source governance | After `#022` and explicit evaluation approval | 1–2 days plus approved eval | Design unknown-vs-adverse metadata wording and prompt tests. Require no supported-grounding drop and zero citation failures; stop on broad over-caveating or degradation. | +| 18 | `#013`, `#016` | A3 | High — bundling/runtime performance | After `#147`/`#117` or equivalent evidence | 0.5–2 days/route | Optimize only a production route with measured payload/render/motion harm. Require material gain plus focused, `verify:cheap`, and browser evidence; stop on small gain. | +| 19 | `#035` | A3 | Specialist — evidence rules | After a demonstrated missed conflict | 0.5–1 day design; code separate | Define a clinically reviewed conflict class with positive and negative fixtures. Stop if no bounded class can be shown; behavior change requires protected review. | +| 20 | `#027` | Optional | Operator — SRE/provider | When an owned external alert path is wanted | 1–2 hours | Decide vendor/cost/privacy/owner; if accepted, prove one non-PHI outage and recovery alert. Stop when no responder owns it. | +| 21 | `#039` | Optional | High — frontend architecture | During a concrete catalogue-toolbar project | 0.5–1 day inventory; 1–3 days code | Converge only repeated toolbar behavior without flattening search semantics. Stop when there is no bounded implementation target. | +| 22 | `#079` | Optional | High — repository hygiene | In explicitly scheduled batches | 30–60 minutes per batch | Disposition at most ten retained worktrees per pass using owner, PR, review-ledger, ancestry, and patch evidence. Preserve every dirty, active, secret-bearing, post-freeze, or ambiguous worktree and stop rather than broad-cleaning. | +| 23 | `#149` | A2 | High — install/gate integrity | Next verify-tooling pass | 1–3 hours | Widen `check:installed-lock-parity` beyond the seven top-level packages so transitive drift (e.g. `brace-expansion` CVE patch) fails closed. Prove with a fixture where only a nested dependency mismatches. **Stop:** do not weaken SessionStart skip-install behaviour without the wider check. | +| 24 | `#086` | A3 | High — repository structure + Specialist | On explicit go-ahead for X3; later packages own their gates | 1 PR per work order | Ship remaining maturity backlog (X3 rag.ts; X7 src/lib reorg; X6 coverage floors; X5 ACL consolidation; L1 one-shot archive; M1 host hardening) as verified draft PRs from `docs/maturity-backlog-workorders.md`. L4 ledger rotation shipped in #1418. Start with X3 after go-ahead; stop before RAG edits without the flag or X5 without live-DB approval. | +| 25 | `#098` | A3 | High — test infrastructure | Before `#099` or `#101`; it is their enabler | 2–4 hours | Generalise the answer-route preamble guard into a counting-proxy round-trip budget harness over the existing offline fixtures. Must enforce admission-before-scope, never the reverse. No providers, no DB. Stop if it would require live credentials. | +| 26 | `#102` | A3 | Operator — Supabase + Specialist | Next approved index window, after the ordering question is settled | 1–2 hours plus apply | Author the migration (operator SQL alone never reaches staging/DR/local replay), then apply → mirror `schema.sql` → regenerate drift manifest → register `required_indexes`. **Stop:** the RAG-path index is canary-gated, and ordering `fetchDocumentTitleAliasRows`'s unordered `.limit(12)` does not lift that — an imposed order can select a different twelve, so it is a second canary-gated change, not a way out of the first. The byte-identical claim was retracted. | +| 27 | `#099` | A3 | Specialist — answer path | After `#098` | Half a day per sub-item | Remaining fixed per-request round trips: the 8 `setCachedSearch` deferrals (abort semantics + mutation window), the anonymous subject+global limiter pair (needs a new atomic RPC first), and proxy→route identity duplication. Stop before hand-authoring locking SQL. | +| 28 | `#162` | A3 | High — frontend/UI | When starting the mode search redesign package | 0.5–1.5 days | Redesign `/tools?q=` as Compact Results Instrument (direction A): query-as-H1, one composer, dense tool rows, demote cross-mode cards, remove success-green filter banner and home hero on results. Comps in `public/mockups/mode-page-redesign-2026-07/tools-search/`. Verify phone+desktop chrome ownership and `verify:phone-chrome` / focused UI. Stop if scope expands into Tools home redesign without an explicit ask. | +| 29 | `#163` | A3 | High — frontend/UI | After or with `#162` | 0.5–1.5 days | Redesign `/services?q=` as Progressive Referral Workflow (direction B): H1 = query (not match count), progressive shortlist/compare (no always-on decision panel or giant step rail). Comps in `public/mockups/mode-page-redesign-2026-07/services-search/`. Verify referral shortlist still works; stop before changing Services home ModeHome. | +| 30 | `#164` | A3 | High — frontend/UI | Product confirmed Favourites is hybrid dashboard+search (no ModeHome) | 1–2 days | Redesign Favourites as one dashboard + search page: recommended Search-Led Workspace (direction B) — persistent search, sets as chips, Continue + recent + table on empty query, in-place filter on typed query. Comps in `public/mockups/mode-page-redesign-2026-07/favourites-hybrid/`. Do not reintroduce ModeHome for Favourites. Verify desktop+phone; stop before splitting into separate ModeHome routes. | +| 31 | `#202` | A3 | High — agent process | Next issues-skill touch | 30–60 min | Require recommendation/`/issues` answers to revalidate against `origin/main`'s ledger (or state checkout lag). Prevents re-proposing closed work from stale worktrees. | +| 32 | `#186` | A3 | Specialist — RAG ledger accuracy | Before any `#101` canary | 30–60 min | Rewrite open `#101` to credit PR 1474 hydration parallelisation and list only remaining canary-gated candidates. | +| 33 | `#187` | Optional | High — ledger hygiene | When writing the durable notes | 30–60 min | After one-line notes land for `#151`/`#154`, archive those process-lesson rows so the open table stays actionable. | +| 34 | `#090` | A3 | High — eslint toolchain | When ESLint 10 plugin peers are compatible | blocked; revisit monthly | Upgrade the eslint ecosystem to clear remaining dev-scoped high advisories — full `npm audit` reports zero high advisories from the eslint toolchain. | +| 35 | `#100` | A3 | Specialist — answer streaming | After offline Phase 0/1 design proof | provider-gated rollout | Buffered answer generation has no incremental verified delivery — [`verified-answer-incremental-delivery-design.md`](verified-answer-incremental-delivery-design.md) records the clinical-governance decision and staged co… | +| 36 | `#150` | Optional | Operator — review tooling | Next CodeRabbit billing/policy decision | 30–60 min decision | CodeRabbit reviewed none of a full day's PRs; spending cap reached — the repo's second automated reviewer is either funded or acknowledged as absent, rather than appearing to review while skipping. | +| 37 | `#152` | A2 | High — worktree hygiene | Next cleanup batch with #079 | 1–2 hours | Uncommitted work sits in worktrees whose branches are already merged — work that exists in no branch and no PR is either committed or knowingly discarded, not lost to a disk reclaim. | +| 38 | `#155` | A2 | High — agent process | Standing rule; next multi-agent session | process change | Several agent sessions edit the same branch and ledger concurrently — concurrent sessions stop silently undoing each other on shared `claude/*` branches and on this file. | +| 39 | `#159` | A3 | High — test hygiene | Next test-infra pass | 1–2 hours | Lists naming test files are duplicated, and the stale copy fails by running nothing — no gate, plan or config names a set of test files in a second place without being derived from the filesystem or asserted against it. | +| 40 | `#165` | A2 | High — clinical UI | Next answer-home UX pass | 0.5–1 day | Adopt a consolidated answer-home notice block — the studies exist, nothing adopts them — the answer hero states its safety obligation, its scope, and its verification requirement as one block in one voice. | +| 41 | `#166` | A2 | High — clinical safety UI | With #165 or next clinical chrome pass | 2–4 hours | Answer mode ships no verify-before-use caveat; every other clinical mode does — the surface that actually generates prose from retrieved sources says so, and says it must be checked. | +| 42 | `#168` | A3 | High — ledger architecture | With #156 / id-scheme redesign | design first | Sequential issue ids force every concurrent append to conflict — two sessions can append to this ledger at the same time without conflicting. | +| 43 | `#169` | A3 | High — git hygiene | Next branch cleanup batch | 1–2 hours | Local branches carry work that exists on no remote — committed work is not lost when a machine or worktree is reclaimed. | +| 44 | `#170` | A2 | High — phone UI | Next documents/filter phone pass | 0.5–1 day | Documents and therapy already have page-owned phone filter sheets; remaining modes still use inline controls — shared-band Filter+Sheet adoption without regressing those two or sheetless Sort. | +| 45 | `#171` | A2 | High — documents UI | With #170 / filter consolidation | 0.5–1.5 days | Documents mode has four overlapping filtering surfaces, two of them the same job — one Filter control opening one panel, so a reader learns filtering once. | +| 46 | `#175` | A2 | Operator — clinical data + Standard | Next therapy catalogue curation window | 2–4 hours | Therapy modality is now null on all 205 records and needs curation or removal — the Therapy detail and recommend screens either show a curated modality or stop carrying the field at all. | +| 47 | `#178` | A3 | High — PR policy | Next pr-policy change | 1–2 hours | pr-policy does not flag operational risk bundled with clinical or UI risk — a PR that mixes operational-risk paths with clinical or UI risk is called out before it merges, because squash-merging that mix destroys per-it… | +| 48 | `#189` | A2 | Specialist — search/RAG budgets | After #098 route residual; before collapsing RPCs | 2–4 hours + canary if behaviour | Pin /api/search route-level round trips and disposition the x3 text RPC probes — a counting-proxy budget drives `POST` `/api/search` (auth/ratelimit/scope/enrichment/telemetry), and the retrieval-core finding that `matc… | +| 49 | `#036` | Optional | Specialist — privacy/schema | When visibility model is redesigned | design + migration | No explicit `is_public` visibility flag on documents — Public-corpus visibility is implicit: `owner_id IS NULL` on an `indexed` document (`resolveSearchScope`). The `metadata.public_corpus` marker is written by the prom… | +| 50 | `#101` | A3 | Specialist — RAG/retrieval | After #186 update + canary approval | canary-gated | Canary-gated retrieval parallelisation candidates — independent retrieval stages stop running serially, proven by a live canary pair. Candidates: metadata/memory/visual hydration triples repeated on four branches (`rag.… | +| 51 | `#142` | Optional | High — docs hygiene | Next docs filing pass | 1–2 hours | Four loose dated docs need source and migration edits before they can be filed — every dated point-in-time doc lives in `docs/audit/` or `docs/archive/` as `docs/README.md` requires, not loose at the `docs/` top level. | +| 52 | `#151` | Optional | Operator — GitHub PAT | When writing the durable note (#187) | 15–30 min | `gh pr checks` cannot read CI, but the Actions API can — nobody concludes CI is unverifiable when it is merely reached through a different endpoint. | +| 53 | `#154` | Optional | High — agent process | When writing the durable note (#187) | 15–30 min | Row ids are not stable identifiers for "did my change land" — an agent confirms work reached `main` by content, never by id, title or PR state. | +| 54 | `#156` | A3 | High — ledger architecture | With #168 id-scheme work | design first | Outstanding-issues ids are still allocated read-modify-write, and Update-branch corrupts the merge — two branches cannot silently claim the same outstanding-issues id, and no merge path can commit a file where they have. | +| 55 | `#172` | A3 | High — documents UI | With #171 filter consolidation | 1–2 hours | `Sources` sits in the results bar but is navigation, not a filter — the results bar holds only controls that act on the current results. | +| 56 | `#174` | A3 | High — search facets | When facet UX is redesigned | 0.5–1 day | Facets AND within a group, so two values from one group almost always return nothing — a decision on record, either way. | +| 57 | `#177` | A3 | High — therapy catalogue build | Next therapy-index build change | 1–2 hours | Therapy catalogue aliases duplicate 2.53 MB of bytes instead of pointing at the hashed file — the unversioned catalogue aliases stop costing a second copy of every payload in the repo and the image. | +| 58 | `#179` | A3 | High — therapy catalogue build | With #177/#180 | 1–2 hours | The full therapy catalogue silently switched from minified to pretty-printed — the full catalogue's on-disk format is a decision someone made, not a side effect. | +| 59 | `#180` | A3 | High — therapy catalogue build | With #177/#179 | 1–2 hours | build-therapies-index now overwrites its own source input — the therapy catalogue generator has a source it does not also destroy. | +| 60 | `#181` | Optional | High — documents UI clarification | When updating #171 | 15–30 min | Correction to `#171`: source-type does NOT duplicate the `Document type` facet group — `#171` states that the documents source-type control "duplicates the facet group already named `Document type`". That is wrong, and … | +| 61 | `#188` | A3 | Operator — DR/SRE | After any schema restore drill, or next DR review | checklist-owned | Document and track disaster-recovery re-creation checklist as ledger work — the five DR items that do not survive a schema restore are tracked with owners and verify steps, not only in `docs/operator-backlog.md`. | +| 62 | `#190` | A3 | Specialist — RAG structure | On explicit X3 go-ahead | 1 PR per extraction unit | X3: Finish rag.ts monolith decomposition — `src/lib/rag/rag.ts` is decomposed into focused modules per `docs/maturity-backlog-workorders.md` X3, with existing offline RAG contracts green. | +| 63 | `#191` | A3 | Operator — DB + Specialist | Approved live-DB window only | provider-gated | X5: ACL-migration consolidation (provider-gated) — ACL-related migrations are consolidated per maturity work-order X5 without weakening owner-scope/RLS. | +| 64 | `#192` | A3 | High — test coverage | Next coverage-floor pass | 0.5–1 day | X6: Raise clinical/retrieval/answer coverage floors — coverage floors for clinical, retrieval, and answer domains meet the maturity X6 targets with CI enforcing them. | +| 65 | `#193` | A3 | High — src/lib structure | After/with X3 non-protected clusters | 1 PR per cluster | X7: Complete the remaining src/lib domain-directory reorg — remaining `src/lib` clusters sit in their domain directories per X7 follow-on to X2. | +| 66 | `#194` | A3 | High — scripts/docs hygiene | Next scripts archive pass | 1–2 hours | L1: Archive retired backfill one-shots and dead ci-change-scope token — retired `backfill:*` one-shots and the dead `ci-change-scope` token are archived/removed with docs/script index updated. | +| 67 | `#195` | A3 | Operator — GitHub maintainer | Maintainer UI window | 30–60 min | M1: Repo-host hardening (branch protection and required checks) — GitHub branch-protection rulesets and required checks match audit §8 / maturity M1. | +| 68 | `#196` | A3 | Operator — DR/SRE | After schema restore drill | 1–2 hours | DR: Re-create pg_cron schedules after schema restore — ingestion/retention and related pg_cron schedules exist on the target DB after any schema restore. | +| 69 | `#197` | A3 | Operator — DR/SRE | After schema restore drill | 30–60 min | DR: Re-add Vault secrets including cron_ingestion_jwt — required Vault secrets (at least `cron_ingestion_jwt`) are present after schema restore. | +| 70 | `#198` | A3 | Operator — DR/SRE | After schema restore drill | 30–60 min | DR: Re-set custom database GUCs after schema restore — custom `app.*` GUCs required by the app/worker are set on the restored database. | +| 71 | `#199` | A3 | Operator — DR/SRE | After schema restore; Deno v2 available | 1–2 hours | DR: Redeploy Supabase edge functions (Deno v2.x) — required edge functions are deployed to the target project with Deno v2.x. | +| 72 | `#200` | A3 | Operator — DR/SRE | After schema restore drill | 1–2 hours | DR: Re-enter dashboard config after schema restore — auth providers/SSO redirect URLs, connection-pool caps, per-project keys, and `E2E_USER_*` are re-entered in the Supabase/Railway dashboards after restore. | +| 73 | `#183` | A2 | Operator — Sentry + Specialist | Next approved observability window with SENTRY_AUTH_TOKEN | 1–2 hours | Create Sentry metric alert for production DB span p95 > 500ms (`span.op:db`, environment production). **Stop:** no secret printing; blocked until token/env available. | +| 74 | `#206` | A2 | Specialist — answer UI contract | With AnswerState producer work (`#207`) | 2–4 hours | `partial_retrieval` has no app-facing producer — decide RAG contract vs UI-only mapping before AnswerCard. **Stop:** no retrieval behaviour change without RAG flag. | +| 75 | `#208` | A2 | Specialist — clinical copy | With answer clipboard / PR-13 work | 1–2 hours | `answerClipboardText` must not replace `formatAnswerRenderCopyText` — compose render-policy warnings. **Gate:** focused clipboard/copy tests. **Stop:** do not drop render-policy caveats. | +| 76 | `#216` | A2 | High — design-system answer shell | After `#207` and clinical surface decision | 0.5–1 day | Adopt AnswerCard on the answer surface (deferred from PR-J). Own PR, own `verify:ui`. **Stop:** not before `#207`; show both surface treatments before choosing. | +| 77 | `#232` | A2 | High — review ledger hygiene | Next ledger touch for PR-J | 30–60 min | Supersede the PR-J clinical-governance ledger row so it describes the merged head (`ledger:append --supersede`). **Stop:** append-only — never edit/delete the old row. | +| 78 | `#209` | A3 | High — design tokens / contrast | Next Gate 1 / warning-token pass | 1–2 hours | Add contrast pair for `--warning` used as body text (VerificationNotice / DoseLine). **Gate:** design-system contrast checks. **Stop:** do not invent a new status token without TOKENS.md. | +| 79 | `#211` | A3 | High — TypeScript strictness | Dedicated migration branch | multi-PR | Plan and start `noUncheckedIndexedAccess` migration (1266 errors); highest-risk files first. **Stop:** do not flip the flag on main without a staged plan. | +| 80 | `#212` | A3 | High — runtime validation | After highest-risk cast inventory | multi-PR | Replace `as unknown as` and unvalidated `JSON.parse` with Zod/guards at trust boundaries. **Stop:** RAG/provider boundaries need clinical/privacy care. | +| 81 | `#213` | A3 | High — error handling | Next fetch/stream hardening pass | 0.5–1 day | Stop swallowing fetch/stream errors with empty catches; check `response.ok`. **Stop:** do not change telemetry contracts silently. | +| 82 | `#215` | Optional | High — image perf | Next image/PWA pass | 2–4 hours | Image-optimization basics for lightbox, PWA lifecycle, demo PNGs. **Stop:** optional until measured need. | +| 83 | `#221` | A3 | High — design-system convergence | After `#218` cn() decision | 0.5–1 day | Converge remaining local EmptyState/LoadingState/Chip duplicates. **Stop:** not piecemeal before cn()/Chip decisions. | +| 84 | `#222` | A3 | High — headers / search chrome | During headers redesign decision | 2–4 hours | Decide whether mode-home-template / search-results-header-band are in PageHeader scope or permanently out. **Stop:** do not flatten phone composer ownership. | +| 85 | `#233` | A3 | High — design-system docs | Next COMPONENTS.md docs PR | 1–2 hours | Refresh section 0 maturity matrix and document FormField optionality-marker contract. **Gate:** docs checks. **Stop:** docs-only; no product behaviour change. | +| 86 | `#234` | A3 | High — design-system docs | With answer-surface docs | 30–60 min | Document `answer-copy-payload.ts` as the clipboard contract for three surfaces. **Stop:** do not add a second copy builder. | +| 87 | `#235` | A3 | High — design-system evidence | Next warmed local proof-shot pass | 1–2 hours | Capture missing ADOPTION.md §7 proof shots for adopted surfaces. **Stop:** not visual-baseline PNGs (`#118`); no Playwright snapshot commit. | +| 88 | `#236` | Optional | High — branch hygiene | Next cleanup batch with `#079` | 30–60 min | Dispose orphan DS V2 builder branches and leftover wave-5 dev servers. **Stop:** content-verify before delete; no force-clean. | +| 89 | `#237` | A3 | High — design-system a11y | Before freezing Linux visual baselines (#242) | 30–60 min | Eyeball low-confidence AccessibleTable densities at 320px; MissingValue phrases must remain readable. **Gate:** visual spot-check only. **Stop:** do not abbreviate MissingValue to a dash. | +| 90 | `#238` | A3 | High — overlays/UI | After Sheet portal default change (#1616) | 30–60 min | Visual pass for Sheet portal default on settings, sidebar, and answer overlays under OverlayRoot. **Stop:** do not revert portal default without evidence. | +| 91 | `#239` | Optional | High — phone chrome | When phone orientation QA is available | 15–30 min | Manual phone rotation check for ResizeObserver-only phone chrome reserve. **Gate:** `verify:phone-chrome` still owns automated coverage. **Stop:** do not widen reserve heuristics without reproduction. | +| 92 | `#240` | Optional | High — design tokens | Next design-owner review | 15–30 min | Confirm tooltip visual hard-clip asymmetry with design owner (sr-only keeps full text). **Stop:** no product change without that confirmation. | +| 93 | `#241` | A3 | High — therapy catalogue | Standing; with any therapy-home change | 15–30 min | Therapy home summary count/slugs remain build-time; keep `build-therapies-index --check` load-bearing. **Stop:** do not bypass the check. | +| 94 | `#242` | A2 | High — design-system baselines | After human review of Linux baselines | 1–2 hours | Commit approved Linux visual baselines and promote adoption not-committed → committed. **Stop:** never commit baselines from an unreviewed machine run. | +| 95 | `#244` | A3 | High — design tokens / forced-colors | With any ckb-v2 forced-colours edit | 15–30 min | Keep grouped dark selectors in the forced-colours media block so specificity matches dark rules. **Stop:** do not trim to a single `.ckb-v2.ckb-v2` selector. | +| 96 | `#245` | A3 | High — cross-mode links | Next CrossModeLinks / analytics pass | 30–60 min | responsive-compact CrossModeLinks keeps duplicate rails in the DOM; prefer one mount or accept test double-counts. **Stop:** do not break phone-only rail contract. | +| 97 | `#248` | A2 | Operator — Supabase + Specialist | After PR #1614 symptom repair; approved live/history window | 1–2 hours | Investigate why 20260705180000 search-health indexes were missing on live despite applied history; decide if drift checks should catch this class. **Stop:** no hosted mutation without approval. | +| 98 | `#249` | A3 | High — agent process | Next issues-skill / plan touch | 1–2 hours | Extend issues/plan with an agent-safe wins classifier (optional filter; no new skill unless reused thrice). **Stop:** do not outrank A1 operator work. | +| 99 | `#250` | A2 | High — multi-agent execution | After Wave 0 queue repair on main (done in this capture); run remaining Wave 0/#202 process gates next on the engineering track | multi-wave | Execute the fastest-wins multi-wave plan (Waves 0–4 + operator track) with parallel agents and per-PR gates. Waves do not outrank A1 acuity. **Stop:** provider/RAG approvals still required where flagged. | +| 100 | `#251` | Optional | High — agent process | Next handoff/gates doc touch | 15–30 min | Handoff checklist pairs gates skill with verification-router; paste decisive proof line. **Stop:** do not stack broad gates by default. | +| 101 | `#253` | A2 | High — phone results UI | Next open-PR sweep | 15–30 min | Decide #1606's fate: the `MobileResultFilterControl` it rewrites was deleted by #247, so there is nothing left to hand-merge. Verify keyboard parity of the replacement sheet on a real device, then close #1606 as superseded. **Stop:** the decision is a human's; do not close #1606 automatically. | +| 102 | `#254` | A2 | Operator — Codex Cloud | Before #1617 leaves draft | 1–2 hours | Re-run Codex Cloud acceptance at the exact current head or mark head-independent evidence. **Stop:** do not treat stale pins as coverage. | +| 103 | `#256` | A2 | High — mode section nav | Next information-page / mode-nav pass | 2–4 hours | Declared information-page section sets whose target ids nothing renders — verify each set against the rendered DOM per route; render anchors or delete the set. **Stop:** do not audit by grepping for `id=` alone (sectionId props exist). | +| 104 | `#257` | Optional | High — formulation/specifiers flake | Standing until second reproduction | 15–30 min | Single unreproduced ui-formulation flake when run with ui-specifiers — record a second sighting only; do not quarantine until three on the same SHA. **Stop:** do not weaken assertions. | +| 105 | `#286` | A3 | High — in-page nav + frontend | After owner go-ahead for the information-page series | 1–2 days | Convert the six pill-rail information pages onto `InPageNavHeader`, widen Server Component–safe actions, then delete `informationPageSectionDefinitions`. **Gate:** focused DOM/contract tests + `verify:phone-chrome` for touched owners. **Stop:** do not convert DocumentViewer here; do not verify anchors by grepping `id=` alone. | +| 106 | `#287` | A3 | High — in-page nav + clinical owner | After `#286`; medications needs an owner product call | 0.5–1 day design + convert | Decide medications tab model, presentations MobileTabs vs `InPageNavHeader`, and factsheets heading→id scheme; convert or record lasting exceptions. **Stop:** do not port medications mechanically. | +| 107 | `#288` | Optional | High — document chrome | After `#286`/`#287`, or when declaring the series complete | 30–60 min | Confirm DocumentViewer non-adoption (already noted in `docs/search-chrome-behaviour.md`) as the final end state, or schedule a separate convergence PR that leaves pinned `--document-*` CSS names untouched. | +| 108 | `#289` | A2 | High — auth/identity | Next auth module touch | 1–2 hours | Export a named helper (e.g. `authorizationIdentity(headers)`) from the auth module and use it at every property-access call site; consider a lint rule or branded type so `.Authorization` stops type-checking at all. **Stop:** do not change `authorizationHeadersForAccessToken` to emit uppercase — lowercase is the correct Fetch/Headers convention and callers that pass the object wholesale to `fetch` depend on it. | @@ -258,7 +255,6 @@ removed after current-main verification; it is not missing recommended work. | #199 | P3 | task | DR: Redeploy Supabase edge functions (Deno v2.x) | **Outcome:** required edge functions are deployed to the target project with Deno v2.x. **Next:** operator deploy after restore; confirm function list/health. Parent `#188`. **Stop:** needs Deno toolchain and explicit approval for hosted deploy. | docs/operator-backlog.md; #188 | 2026-07-31 | | #200 | P3 | task | DR: Re-enter dashboard config after schema restore | **Outcome:** auth providers/SSO redirect URLs, connection-pool caps, per-project keys, and `E2E_USER_*` are re-entered in the Supabase/Railway dashboards after restore. **Next:** operator checklist in `docs/operator-backlog.md`. Parent `#188`. **Stop:** do not commit dashboard secrets. | docs/operator-backlog.md; #188 | 2026-07-31 | | #206 | P2 | task | AnswerState partial_retrieval has no app-facing producer | PR-E step 0 found nothing in the client payload names which expected sources were unavailable (retrievalDiagnostics = candidate counts; conflictsOrGaps = prose). RetrievalStateBanner supports the state but PR-J adoption can only emit ready/stale_evidence/source_only. Next action: decide whether a separate RAG contract PR should add a named missing-source signal (governance preflight + RAG impact line + offline eval); until then do not synthesise the state from counts. Pinned by tests/answer-state-contract.test.ts and SPEC 13 / COMPONENTS 2. | PR-E step 0, session 2026-08-02 | 2026-08-02 | -| #207 | P1 | task | DS V2 PR 13 blocker: AnswerState has no ungrounded-answer channel | answerStateFromRetrieval() maps a grounded:false / confidence:'unsupported' answer over current sources to 'ready'. The live product already gates on grounded/confidence/unverifiedNumericTokens (evidence-panels.tsx, answer-thread-turn.tsx) to show 'Review source match', so adopting AnswerCard as-is would silently retire a warning shipped today. Needs a fifth state or companion flag; wording is a clinical-owner decision. | clinical-governance-reviewer P1-2 on PR 6 (claude/ds-v2-answer-safety); recorded in docs/design-system/SPEC.md PR 6 clinical review note and COMPONENTS.md 9.13 | 2026-08-02 | | #208 | P2 | task | DS V2: answerClipboardText must not replace formatAnswerRenderCopyText in PR 13 | PR 6 strengthened answerClipboardText (unconditional attribution + verify line, enumerated sources, provenance suppressed where it would contradict the caveat). It is still narrower than formatAnswerRenderCopyText (src/lib/answer-render-policy.ts), which carries the render policy's own warnings. PR 13 must compose the two or extend answerClipboardText with clinical-owner review, never swap it in. | clinical-governance-reviewer P1-1 on PR 6; recorded in docs/design-system/SPEC.md and COMPONENTS.md 9.13 | 2026-08-02 | | #209 | P3 | task | DS V2 Gate 1: add contrast pair for --warning used as body text | VerificationNotice's caution variant and DoseLine's overdue label use --warning at text tier — the only place a status hue is used as body-text colour rather than a --text-* token. Gate 1's contrast checking must add that pair explicitly rather than assuming the text tiers cover it. Also note: the logged-once Sets in missing-value, date-display, verification-notice, answer-state and retrieval-state-banner are module-level, so on the server they are per-process and unbounded; a persistent data defect logs once at boot then is swallowed. Acceptable while unregistered. | clinical-governance-reviewer P3 findings on PR 6; recorded in docs/design-system/SPEC.md PR 6 clinical review note | 2026-08-02 | | #210 | P2 | task | Restore the npm run typecheck gate | Clear stale .next generated types or update validator references so npm run typecheck reflects source health. Evidence: 14 errors in .next/dev/types/validator.ts referencing removed mockup pages; source-only typecheck is clean. See docs/review-findings-2026-08-02.md sections 2.1 and 9. WIDER THAN TYPECHECK (session 2026-08-06, PR #1647): the trigger is running npm run ensure, which makes the dev server generate .next/dev/types/validator.ts; tsconfig.json includes .next/dev/types/**/*.ts, so that file breaks BOTH repo-wide npm run typecheck AND every Playwright production build, because scripts/run-playwright.mjs writes an isolated tsconfig that extends the root one. Observed as 'Type error: Cannot find name __IsExpected' aborting the isolated build; rm -rf .next/dev restored both. This bites anyone following the documented 'run npm run ensure before browser work' instruction. Next: give the generated Playwright tsconfig its own include rather than inheriting the root include, or stop the dev server writing types into a path the production typecheck reads. | session 2026-08-02 /ledger sweep; docs/review-findings-2026-08-02.md | 2026-08-02 | @@ -269,9 +265,7 @@ removed after current-main verification; it is not missing recommended work. | #216 | P2 | task | Adopt the AnswerCard container on the answer surface (deferred from PR-J) | PR-J adopted the answer safety components (VerificationNotice, RetrievalStateBanner, AnswerState projection, composed clipboard) but NOT AnswerCard itself, so it stays at zero product imports and PR 13's answer surface is adopted in substance, not in shell. The swap replaces the answerSurface wrapper with AnswerCard's article, its query echo with UserQuestionBubble, wraps NaturalLanguageAnswer in its --measure-clamped prose div, maps copy/feedback/follow-up onto its structured actions array, and adds AnswerFooter (needs publisher/version/reviewDate/generatedAt threaded). Deferred deliberately: two competing surface treatments must be resolved by the design owner (SPEC 2.4 border-or-ring), the measure clamp reflows every answer and moves the phone scroll-runway pins that cost PR-V two CI cycles (finding L), and bundling it would make a red verify:ui unattributable across PR-J's other five surfaces. Reasons recorded in docs/design-system/ADOPTION.md 2.6. Next action: own PR after PR-J lands and soaks, own verify:ui pass, own glance; show the user both surface treatments before choosing. | session 2026-08-03 (PR-J Wave 5 controller) | 2026-08-02 | | #221 | P3 | task | Local EmptyState, LoadingState and Chip duplicates still unconverged after PR-J | PR-J converged what it could inside its allowlists and left four known duplicates, each blocked for a stated reason rather than missed. therapy-compass/ui.tsx defines its own LoadingState AND its own EmptyState used across nine screens (whole-module job, not a one-call-site conversion). mode-home-template.tsx ModeHomeStatusNotice is an EmptyState duplicate that four catalogue homes delegate to, which is why those four files show no diff. differentials-home.tsx has a local two-density Chip blocked by the cn() tailwind-merge gap. favourites-command-library-page.tsx SmallChip is driven by an eight-entry type-token map that Chip's five-tone vocabulary cannot express. Next action: take these as one convergence PR after the cn() decision lands, not piecemeal. Found during PR-J adoption, 2026-08-03. | session 2026-08-03 (PR-J Wave 5, Builder B) | 2026-08-02 | | #222 | P3 | task | Headers surface only partially converged in PR-J: mode-home-template and search-results-header-band untouched | Builder A converged DsmPageHeader, InformationPageHeader and InformationPageBreadcrumbs onto PageHeader plus Breadcrumb, and declined two files with reasons. mode-home-template.tsx ModeHomeHero is a centred display hero on the fluid text-hero token and is the slot the in-flow phone composer sits in, so converging it onto a left-aligned PageHeader is a redesign of 13 mode homes that collides with the one-composer-per-page contract. search-results-header-band.tsx is a results spine carrying status, counts and filters, not a page-title stack, so its pin tests/search-results-header-band.dom.test.tsx remains unflipped. Both are defensible; both leave the headers surface partially adopted. Next action: decide whether either is in scope at all, or record them as permanently out of the PageHeader vocabulary. Found during PR-J adoption, 2026-08-03. | session 2026-08-03 (PR-J Wave 5, Builder A) | 2026-08-02 | -| #226 | P1 | task | PR-J: VerificationNotice pushes the phone short-answer runway past the in-flow activation band | CI run 30820496984 reported the real numbers: ui-smoke phantom-scroll 97 against an 8px budget, short-runway maxOffset 251 against a 200px ceiling. Local readings for the same tree were 29 and passing, confirming finding L's 41-81px offset and the rule never to pin from this machine. Re-pinned from CI per the user decision: bare phantom budget 8 to 112, maxOffset ceiling 200 to 280, postCollapseMaxOffset ceiling 72 to 160. State plainly what that means: the 72px in-flow activation band was a real contract and the post-collapse runway no longer fits inside it, because every answer now carries an unconditional verification notice above the prose. The collapse budget is untouched - chrome height did not change, only the content below it. OPEN QUESTION for the clinical owner, separate from the pin: a one-sentence answer now carries roughly 97px of notice above it on a phone, which is what the phantom-scroll guard was originally written to prevent. Widening the pin accepts that as intended product behaviour. | session 2026-08-03 (PR-J Wave 5, gate window, verify:phone-chrome) | 2026-08-03 | -| #230 | P2 | issue | CI's PR-policy body sync can overwrite PR descriptions from a committed scratch file | ci.yml's Sync PR policy body job runs only when the PR head contains PR_POLICY_BODY.md (pr_policy_body_present) and then replaces that PR's description with the file. #1546 committed the scratch file to main; any open PR whose head still contained the file (including those that merged or rebased onto that tip) inherited #1546's body — and pr-policy.mjs parses the body as merge-gating input, so governance checklists and verification claims appeared on PRs they did not belong to. Heads without the file are not rewritten. #1548 deleted the file, which removes today's symptom; nothing stops the next branch committing one. Next action: make the sync job no-op unless the file is new in that PR's own diff, or move the handoff body out of the repo entirely. | Observed during the search-results-bar work (#1555); recorded in that branch's handoff doc rather than the ledger at the time | 2026-08-04 | -| #231 | P1 | issue | Live answers degrade to source-only: fast-route answerRouteBudgetMs.fast (25000) binds while retrieval is healthy | Measured live 2026-08-04 on psychiatry.tools, container d6c2b9526. POST /api/answer for 'Lithium dosing?' took 24.7s and returned degradedMode {active:true, reason:'generation_fallback:provider_timeout'}, routingReason 'clinical_fast_grounded_synthesis; generation_fallback:provider_timeout; source_backed_extractive_fallback', modelUsed null, answerQualityTier 'source_only', openAIUsage 6349 input / 586 output tokens (billed then discarded). Retrieval is NOT the fault: retrievalDiagnostics gateStatus 'passed', candidateCount 12, distinctDocumentCount 5, topScore 0.7125, fallbackReason null; POST /api/search returned 8 correct results for both 'lithium dosing' and 'clozapine monitoring'; the top lithium document is document_status 'current', review_date 2029-11-30, clinical_validation_status 'locally_reviewed'; and the dose path sets embedding_skipped true (dose_evidence_text_match) so it never calls OpenAI for retrieval. This corrects the 2026-08-04 production-triage handover, whose four ranked hypotheses (query embedding, Supabase RPC, source governance, corpus) are all ruled out. THE BINDING BUDGET IS NOT OPENAI_ANSWER_TIMEOUT_MS - do not raise it, and note production does not even set it. The query routed fast (routeMode 'fast', smart_api_response_mode 'fast_grounded_answer'), so answerRouteBudgetMs.fast = 25000 is the ceiling and generationRequestTimeoutMs returns min(30000, remaining - generationRecoveryReserveMs 2000); after ~2s retrieval that leaves roughly 21s for generation, and the observed 24.7s is the route budget being spent, not the 30s env knob. Production sets OPENAI_MAX_OUTPUT_TOKENS=16000 and RAG_PROVIDER_MODE=auto with no timeout or reasoning-effort override. Second finding: answerRouteResultCanBeCached returns !deadlineExceeded, which correctly excludes a route-deadline-exceeded answer but NOT one degraded by the OpenAI request timing out - deadlineExceeded stays false there - so a generation_fallback answer is cacheable in rag_response_cache and keeps being served after the provider recovers, which is what user-observed ~2s 'No current source ...' responses were. Also live only since this window: 47c512e07 bumped openai 6.49.0 to 7.3.0 (major). Next action: own PR, because every candidate fix is inside behaviour-protected `src/lib/rag/**` - flag it first, carry a RAG impact line, and pair it with a live eval canary. Decide (a) whether medication_dose_risk belongs on the fast route when it consistently overruns 25s, and (b) whether answerRouteResultCanBeCached should also exclude generation_fallback results, which is the half with clinical consequence. Stop: do not edit `src/lib/rag/**` without flagging, and do not raise a timeout that cannot bind. Renumbered from this PR branch original #230 after main claimed #230 for the PR-policy body-sync issue (PR #1609). | session 2026-08-04 (production triage, live /api/search + /api/answer) | 2026-08-04 | +| #231 | P1 | issue | Generation fallbacks no longer stick in answer cache; lithium generation quality still falls back safely | PARTIAL 2026-08-12: This PR fixes the clinically consequential stale-fallback path: every answer whose routing or degraded reason contains generation_fallback is excluded from rag_response_cache. Offline evidence: 96 focused answer-route tests and 574 RAG fixture/contract tests passed. Approved live baseline/final canaries preserved 36/36 document and content recall at 1.0 with zero per-case reciprocal-rank regressions; the final 44-case answer gate had zero citation or numeric-grounding failures. A budget extension was tested and rejected: four cache-bypassed 'Lithium dosing?' probes remained grounded, cited safe extractive fallbacks at 35-40 second candidate budgets; the decisive 40-second probe completed generation in 25.272 seconds and 27.237 seconds total with route_deadline_exceeded=false, but failed generation quality. Therefore OPENAI_ANSWER_TIMEOUT_MS and the route budget are not the current residual binding cause. Next: instrument and reproduce the structured generation-quality failure using provider-safe metadata, then make a separate bounded output-quality fix with an offline fixture and live canary. Stop: do not increase route/provider timeouts or cache any generation fallback. | session 2026-08-04 (production triage, live /api/search + /api/answer) | 2026-08-04 | | #232 | P2 | issue | The PR-J clinical governance review record describes a head that was never merged | docs/branch-review-ledger.md records the clinical-governance review of PR #1595 (PR-J) at head f9f73c707. The head that actually merged was 590eb6cfb (squashed to f4448f8c1 on main). Three things landed after the reviewed head: the bot's answer-state projection changes, the two Codex review fixes, and the #228 attribution wording. The user passed the review knowing this, so the review decision itself is not in question - the record simply does not describe the merged tree, which defeats the ledger's purpose of proving what was reviewed. Next action: append a superseding record (never edit the existing row) with npm run ledger:append -- --supersede against the merged head, scoped to the delta between f9f73c707 and 590eb6cfb, or state explicitly that the delta was accepted unreviewed. Stop: the ledger is append-only - do not rewrite or delete the original row. | session 2026-08-04 (DS V2 Wave 5 close-out capture) | 2026-08-04 | | #233 | P3 | task | COMPONENTS.md section 0 describes the pre-adoption world, and the optionality-marker contract change is undocumented | Two documentation debts left by PR-J, both in docs/design-system/COMPONENTS.md, naturally one PR. First: section 0's maturity matrix is stale. FormField, TextField, SearchField, Select, Checkbox, RadioGroup, PageHeader and Breadcrumb now have real product mounts, so 0.1 and 0.2 misdescribe what is registered versus built-but-unregistered, and 0.4's field-shell defects are closed by the five-control fold. A reader deciding whether a component is safe to adopt is reading the wrong answer. Second: FormField now marks only the requirement and leaves optional fields unmarked - (optional) was removed app-wide by design decision and is pinned by tests/ui-v2-form-field.dom.test.tsx - which is a design-system contract change that appears in no document. It belongs in COMPONENTS.md section 4 and probably DECISIONS.md. Next action: one docs PR updating section 0 from the actual mount list and recording the optionality rule with its rationale. Stop: do not re-add (optional) markers to satisfy a generic form-accessibility rule - the removal was deliberate and is test-pinned. | session 2026-08-04 (DS V2 Wave 5 close-out capture) | 2026-08-04 | | #234 | P3 | task | answer-copy-payload.ts is the single clipboard payload builder for three surfaces and has no documentation | src/lib/answer-copy-payload.ts arrived in PR-J exporting answerStateForAnswer, buildAnswerClipboardText, resolveAnswerSources, citedSourcesOnly and singleDocumentClipboardMetadata. It is now the one place three product surfaces build a clipboard payload, which makes it a contract rather than a helper: a future caller that bypasses it can reintroduce the false-attribution defect the module exists to prevent (see #228). Nothing in docs/design-system mentions it. Next action: document the module and its five exports where the answer surface's copy contract is described, and state that new copy paths route through it rather than composing their own text. Found during PR-J close-out, 2026-08-04. | session 2026-08-04 (DS V2 Wave 5 close-out capture) | 2026-08-04 | @@ -295,8 +289,7 @@ removed after current-main verification; it is not missing recommended work. | #257 | P3 | issue | Single unreproduced ui-formulation flake: keeps specifier and formulation route families clinically separate | Observed once on 2026-08-06 at PR #1647 head f5833acc, running tests/ui-formulation.spec.ts + tests/ui-specifiers.spec.ts together against local Chromium (1 failed, 11 passed). Did NOT reproduce: passed in isolation with --grep, and passed again on a full-file re-run (7/7). Recorded only so a second sighting is recognisable as a second rather than looking like a first. Per docs/testing.md this is one reproduction of three — do NOT quarantine, and do not weaken the assertion. Next: no action unless it recurs; if a second reproduction lands on the same SHA, note it here, and only on a third open a tests/flake-ledger.json entry with @quarantine and a <=30-day expiry. | session 2026-08-06; PR #1647 | 2026-08-06 | | #258 | P2 | rec | The PR-handoff stop rule is enforced for Claude Code only; Codex and Cursor get prose with no gate | **Outcome:** a session that opens a PR stops following it in every agent this repo supports, not just Claude Code. **Detail:** PR #1649 added `.claude/hooks/pr-handoff-stop.sh` plus the AGENTS.md "Stop when the pull request is open" section. The hook is registered in `.claude/settings.json`, which only Claude Code reads, so the PostToolUse marker and the PreToolUse denials (shell `gh pr checks/status/view/run watch`, GitHub MCP tools named pull_request/workflow_run/workflow_job/check_run/check_suite/job_log/update_branch, and Monitor/ScheduleWakeup/CronCreate) simply do not exist for Codex or Cursor sessions. Those agents get the AGENTS.md prose and nothing else — and prose alone is exactly what was already in force, and already insufficient, before #1649. Cost is the same long tail of post-handoff CI polling the hook was built to cut, just relocated to whichever agent lacks the gate; a cloud Codex session is the worst case because nothing naturally ends it. **Next:** cheapest first — check whether Codex and Cursor expose any pre-tool interception this repo can register (Codex plugin hooks under `plugins/clinical-kb/`, Cursor rules under `.cursor/`); if neither offers a deny path, the fallback is a shared marker file plus a wrapper the agent is told to route `gh` through, which is weaker but still detectable. If no mechanism exists at all, record that explicitly here so the gap is a known limit rather than an open task. **Stop:** do not weaken the Claude Code hook to make the tools symmetric, and do not add a second copy of the deny list — one script, multiple registrations. | PR #1649; .claude/hooks/pr-handoff-stop.sh; .claude/settings.json; AGENTS.md "Stop when the pull request is open"; session 2026-08-07 | 2026-08-07 | | #260 | P2 | task | Two unpushed Sentry commits are stranded on a Windows-only branch and will be lost with that machine | **Outcome:** the Sentry setup/logging-hardening work is either shipped or consciously discarded, not left sitting in one machine's reflog. **Detail:** `claude/cloud-pr-loop-prevention-bc052b` carries two commits — `c3c9d6a31` and `abbcdc8e9`, ~389 lines across `src/sentry.*.config.ts`, `src/lib/env.ts`, `src/lib/supabase/client.tsx`, `src/components/ui-primitives.tsx` — that were never pushed and are not the authoring session's own work. The branch does not exist on the remote, so the commits are unreachable from any cloud or remote container; a 2026-08-07 remote session could not inspect, verify, or ship them and could only record their existence. The same worktree (`.claude/worktrees/pensive-borg-6be2f0`) still holds the same four files uncommitted. Two Sentry branches DO exist on origin — `claude/sentry-nextjs-sdk-setup-2v24q5` and `cursor/sentry-nextjs-sdk-7cee` — but whether either already carries this change is unconfirmed: a three-dot diff against `origin/main` from the remote container returned empty for both, which is not trustworthy as proof either way and was not pursued further. Note this touches `src/lib/env.ts` and `src/lib/supabase/client.tsx`, so it is not a docs-class change and needs a real gate whenever it does ship. **Next:** from the Windows machine, diff those two commits against the two remote Sentry branches to decide whether the work is already represented. If it is, delete the branch; if it is not, push it and open a PR rather than leaving it local. **Stop:** do not discard the commits blind, and do not assume the remote Sentry branches supersede them without a content diff — nothing has yet compared them. | session 2026-08-07 remote container; handoff notes from the PR #1649 session; origin branches claude/sentry-nextjs-sdk-setup-2v24q5 and cursor/sentry-nextjs-sdk-7cee | 2026-08-07 | -| #261 | P2 | task | DS Track A2: retire --shadow-focus from the search composer | Replace the composer's companion focus ring with the sanctioned outline / --focus treatment used everywhere else, then delete the token (both theme declarations). Live consumer is .chat-composer-shell-delta:focus-within in globals.css — a --include=*.tsx grep reports zero consumers and is wrong. This is a visible focus-state change on the search composer: read docs/search-chrome-behaviour.md first and get a Chromium look. Gate: npm run check:design-system-contract + npm run verify:phone-chrome. | session 2026-08-07 — design-system HANDOVER-2026-08-07 Track A1 handoff (PR #1678) | 2026-08-07 | -| #262 | P2 | task | DS Track A3: finish the design-token debt | Three parts. (1) Move --shadow-tight onto the --eN elevation ladder. SCOPE RE-MEASURED 2026-08-08 against origin/main 2675e6e1d, running analyzeClassContractsInSource + analyzeCssContractsInSource over the same walk check-design-system-contract.mjs uses (src/**, .ts/.tsx/.css, mockups excluded). The inherited figures were wrong in three ways. First, the legacyShadowAliases metric counts SEVEN tokens, not one: measured total 228 = tight 100, soft 72, elevated 17, hover 17, card 12, lux 8, lift 2. So the '229 --shadow-tight aliases' in HANDOVER-2026-08-07 is the all-token total mislabelled, and this row's earlier '155 consumers' was closer to a raw repo-wide grep (160 occurrences including mockups) than to the gated number. Second, the real scope is 100 production --shadow-tight sites across 55 files, so the inherited figure overstates the work by roughly 1.55x, and clearing all 100 will NOT zero the ratchet: 128 aliases across the six other tokens remain, so do not treat legacyShadowAliases=0 as the success criterion. Third, --shadow-focus is NOT in this metric at all: LEGACY_SHADOW_ALIAS has matched exactly tight\|card\|soft\|hover\|elevated\|lux\|lift since PR #1616 and has never included focus, so an earlier note claiming 'eight tokens, focus 2' and an overlap with #261 was wrong. #261 is a separate token with one consumer (src/app/globals.css:1476) and two theme declarations (lines 423, 664); the two tasks do not share this metric. Baseline pins legacyShadowAliases at 231 and the baseline is a ceiling, so today's 228 already passes. Re-measure before starting rather than trusting any of these numbers. (2) Add a step-SELECTION lint for the eight non-standard type steps (1318 sites) — check:type-scale already blocks arbitrary text-[12px], so do NOT write a lint duplicating the half that ships. (3) Extend the contract ratchet to raw padding / radius / line-height literals; it covers colour, shadow, tap and tracking today. Gate: npm run check:design-system-contract. | session 2026-08-07 — design-system HANDOVER-2026-08-07 Track A1 handoff (PR #1678) | 2026-08-07 | +| #262 | P2 | task | DS Track A3: finish the design-token debt | Three parts. (1) DONE 2026-08-10 - --shadow-tight is retired outright: 90 gated production sites across 48 files (plus 60 mockup occurrences, migrated in the same pass so no file names a dead token) now reach for var(--e1), and all three declarations - both themes and the forced-colors flattening - are deleted. The alias resolved to exactly var(--e1) in every scope and the forced-colors block already flattened --e1 alongside the roles, so the substitution was value-preserving in light, dark and forced-colors and needed no visual review. Do NOT take that from the declarations alone for the remaining tranches: ckb-v2-tokens.css redeclares --e1 (light 13 40 71 / 5% vs globals 11 42 56 / 7%) and never redeclares the roles, and a custom property containing var() substitutes on the element it is DECLARED on - an alias declared in an outer scope and overridden in a narrower one freezes at the outer value. This migration is safe only because .ckb-v2 is on (layout.tsx) and .ckb-v2.ckb-v2 outspecifies :root, so the alias substitutes against the winning v2 tier; measured in Chromium, both spellings compute to rgba(13, 40, 71, 0.05) 0px 1px 2px 0px. Re-run that check per alias, it is about where a declaration sits. legacyShadowAliases 220 -> 127 with per-path counts pinned to measured, which also closed 3 aliases of re-accumulated stale slack across the other six roles (measured 217 against a 220 ceiling - the same drift #264 found on 9 Aug). design-token-contract.test.ts now asserts the token is absent from the whole stylesheet, mutation-verified. Remaining 127: soft 71, elevated 17, hover 17, card 12, lux 8, lift 2 - and count a token by reading the var() call, not the declaration it sits in, because two of the soft hits are the VALUE of the --shadow-focus declarations. Parts (2) and (3) below are untouched; (3) landed separately in PR #1780 per #301. ORIGINAL SCOPE NOTE, kept for the remaining tranches: SCOPE RE-MEASURED 2026-08-08 against origin/main 2675e6e1d, running analyzeClassContractsInSource + analyzeCssContractsInSource over the same walk check-design-system-contract.mjs uses (src/**, .ts/.tsx/.css, mockups excluded). The inherited figures were wrong in three ways. First, the legacyShadowAliases metric counts SEVEN tokens, not one: measured total 228 = tight 100, soft 72, elevated 17, hover 17, card 12, lux 8, lift 2. So the '229 --shadow-tight aliases' in HANDOVER-2026-08-07 is the all-token total mislabelled, and this row's earlier '155 consumers' was closer to a raw repo-wide grep (160 occurrences including mockups) than to the gated number. Second, the real scope is 100 production --shadow-tight sites across 55 files, so the inherited figure overstates the work by roughly 1.55x, and clearing all 100 will NOT zero the ratchet: 128 aliases across the six other tokens remain, so do not treat legacyShadowAliases=0 as the success criterion. Third, --shadow-focus is NOT in this metric at all: LEGACY_SHADOW_ALIAS has matched exactly tight\|card\|soft\|hover\|elevated\|lux\|lift since PR #1616 and has never included focus, so an earlier note claiming 'eight tokens, focus 2' and an overlap with #261 was wrong. #261 is a separate token with one consumer (src/app/globals.css:1476) and two theme declarations (lines 423, 664); the two tasks do not share this metric. Baseline pins legacyShadowAliases at 231 and the baseline is a ceiling, so today's 228 already passes. Re-measure before starting rather than trusting any of these numbers. (2) Add a step-SELECTION lint for the eight non-standard type steps (1318 sites) — check:type-scale already blocks arbitrary text-[12px], so do NOT write a lint duplicating the half that ships. (3) Extend the contract ratchet to raw padding / radius / line-height literals; it covers colour, shadow, tap and tracking today. Gate: npm run check:design-system-contract. | session 2026-08-07 — design-system HANDOVER-2026-08-07 Track A1 handoff (PR #1678) | 2026-08-07 | | #265 | P2 | task | DS Track A6: move design-system gates 2, 4, 7 and 8 from partial to blocking | RE-MEASURED AND PART-CLOSED 2026-08-09 against origin/main 8db1e53937. GATE 4 CLOSED: colourOnlyStatusIndicators in check:design-system-contract is the repository-wide enumeration this row asked for - a status hue on a box with no children, no aria-label/aria-labelledby/title on it or any ancestor, no sibling text, and not a StatusMark. It also flags shared swatch recipes, because the analyzer is per-file and cannot follow an imported statusDotReady to its call sites. Ratcheted at 4 with per-path pins (the two bare statusDot recipes GATES.md named, a calculator risk band, a therapy meter fill); a new colour-only indicator anywhere in src now fails. Mutation-verified. GATE 2 NOT CLOSED, and this row's description of it was wrong in a way that cost a session. It is NOT true that test:e2e:style-contract needs wiring into verify:cheap: the npm script is only an alias for running that one spec, the spec matches productionSpecPattern in playwright.config.ts and is listed in scripts/playwright-pr-shards.mjs, so it ALREADY runs in the required Production UI job. It must NOT be added to verify:cheap:internal, because check:gate-manifest then demands a matching step in static-pr, which has no browser and no server. The real gap is the h-10 blind spot inside the audit itself, and an enumeration for it was written, shown to find genuine defects, and then reverted rather than landed because it is not deterministic on a live-search route - see #293 for the six-run evidence and the follow-up. REMAINING: gate 2's enumeration (needs a deterministic surface first, #293), gate 7 (elevation child/parent, needs a render-tree check, untouched), and gate 8's recorded debt only - its two checks already ship and ratchet per path, so that work is retiring 27 edge conflicts across 15 files and 2 globals.css spreads, then pinning both at zero. | session 2026-08-07 — design-system HANDOVER-2026-08-07 Track A1 handoff (PR #1678) | 2026-08-07 | | #266 | P2 | task | DS Track B1: adopt the 23 unadopted components demand-driven, never as a race to 53/53 | Pick a surface and let it pull, the way PR #1658 did for AnswerCard. COUNT RE-MEASURED 2026-08-08 from docs/design-system/adoption-manifest.json on origin/main: 53 registered, 30 with at least one productImportFiles entry, 23 UNADOPTED — not 24. Button moved into the adopted set when AccessibleTable's expand control stopped being a hand-rolled recipe (#263, PR #1712); its sole production importer is src/components/AccessibleTable.tsx, which is the demand-driven route this row describes, so it is the pattern to copy rather than an exception. The 23 measured today: AnswerFooter, Checkbox, Citation, CitationList, ConfirmDialog, Disclosure, DisclosureGroup, DoseLine, DownloadLink, ErrorSummary, ExternalTextLink, FieldError, FieldHint, LinkAction, Pagination, Progress, RadioGroup, SearchField, StageList, Tabs, TextLink, ToastRegion, Tooltip. Forms are still the largest single tranche: FieldError, FieldHint, ErrorSummary, SearchField, Checkbox and RadioGroup all land together on one form conversion. Do not stub a component to move the adoption count. Regenerate with npm run design-system:adoption:update after any import change; ALSO run npm run design-system:design-sync:update, because changing any *Props type or adopting a component fails check:design-sync-contract with 'dtsPropsFor must be generated from source public Props types' if only the first is run. Both manifests are generated, never hand-edited. | session 2026-08-07 — design-system HANDOVER-2026-08-07 Track A1 handoff (PR #1678) | 2026-08-07 | | #267 | P3 | task | DS Track B2: AnswerFooter and DoseLine need a provenance/dose payload the answer surface does not produce | Backend-shaped work, not a component swap: the two components cannot be adopted until the answer surface emits the provenance and dose data they render. Do not stub one to make the adoption count look better. Sequence after the payload exists, then adopt via the Track B1 demand-driven route. | session 2026-08-07 — design-system HANDOVER-2026-08-07 Track A1 handoff (PR #1678) | 2026-08-07 | @@ -312,7 +305,6 @@ removed after current-main verification; it is not missing recommended work. | #281 | P2 | rec | The phone document route renders two clinical-summary surfaces and neither is canonical | **Outcome:** one clinical summary on the document route, chosen deliberately. **Detail:** a phone reader gets the gradient 'High-yield clinical summary' card (DocumentClinicalSummary, built by buildDocumentClinicalSummaryModel) and, further down, the rail's '#source-summary' / 'high-yield-summary' disclosure (DocumentSectionSummary + FormattedHighYieldSummary + BadgeCluster). They render the same document.summary row two different ways. The rail is not hidden on phones — only its DocumentSectionIndexCard is lg:block — so both appear. Only the rail panel carries the section anchor, so the more prominent card is the unnavigable one. Note the two disagree about emptiness as well: the card now renders nothing when the model yields no usable text, while the rail panel still renders for its label badges, which is why 'hasStoredSummary' was deliberately left keyed to the stored row rather than to card content. **Next:** decide which rendering is canonical — this is a clinical-content judgement about how a summary should read, not a layout fix — then delete the other and give the survivor the 'source-summary' anchor. If the rail's badges are the part worth keeping, they can move without the second summary body. **Stop:** do not merge the two renderings mechanically; they format clinical text differently and the difference is the decision. | session 2026-08-08 document-viewer optimisation; document-rail-panels.tsx; document-clinical-summary.tsx | 2026-08-08 | | #282 | P3 | task | Probe the corpus for JBIG2/JPX before deciding whether pdf.js needs its decoder assets shipped | **Outcome:** a measured decision about pdf.js's cMap/standard-font/WASM assets rather than an assumption either way. **Detail:** getDocument is configured with url plus the on-demand fetch flags and nothing else, so 'wasmUrl', 'standardFontDataUrl', 'cMapUrl' and 'iccUrl' are all unset. pdfjs-dist ships those assets (wasm 1.5 MB, standard_fonts 804 KB, cmaps 1.7 MB) and nothing copies them into public/. With wasmUrl null, 'useWorkerFetch' resolves false and the WASM image decoders cannot load, so JBIG2 and JPEG2000 images fall back to the JS decoders or fail; those are exactly the encodings a scanned guideline uses, and this repo runs an OCR pipeline, which implies scanned sources exist. Non-embedded standard-14 fonts fall back to system fonts, which is a fidelity risk on a clinical document rather than a failure. **Next:** sample the real corpus for JBIG2/JPX-encoded images and for PDFs relying on the standard 14 before shipping ~2 MB of static assets; if the corpus does use them, copy into public/pdfjs, set the URLs, and add immutable cache headers in next.config.ts (public/ is not counted by check:bundle-budget, so there is no budget risk — the cost is bytes over the wire on first use). **Stop:** do not ship the assets on the assumption alone. | session 2026-08-08 document-viewer optimisation; node_modules/pdfjs-dist/types/src/display/api.d.ts | 2026-08-08 | | #283 | P3 | rec | The 100-id batch signed-URL route still has no caller | **Outcome:** either the batch minter is used or it is retired, rather than sitting as an untested, unreachable privileged surface. **Detail:** src/app/api/images/signed-urls/route.ts POSTs up to 100 image ids and returns their signed URLs, with its own rate limit, owner scoping and committed-generation filter. Nothing in src/ calls it — only tests/private-access-routes.test.ts imports it. **DEFERRED AGAIN, DELIBERATELY, 2026-08-09 (document viewer Phase 3, Task 3).** The user chose deferral over wiring when asked. Two reasons beyond cost: (a) wiring it puts a privileged owner-scoped API route into a diff that is otherwise confined to src/components/document-viewer/**, and it matches clinicalRiskPatterns (/^src\/app\/api\//) so pr-policy hard-blocks the merge without a complete Clinical Governance Preflight; (b) Phase 3 Task 2 windowed the rail to six rows and tightened its IntersectionObserver root margin from 640px to 240px, so the many-distinct-images case the batch route was meant to serve is now materially smaller — a page of N figures no longer mounts N rows at once. The batching win should be re-measured against the windowed rail before it is wired at all, rather than assumed from the pre-window numbers. **Next:** decide deliberately — measure concurrent distinct-image requests on a figure-heavy document with the windowed rail, then either wire the batch route in its own PR or delete it and its tests. **Stop:** if wiring it, keep the per-image endpoint for the lightbox's retry path; do not make the batch the only way to mint a URL. | session 2026-08-08 document-viewer optimisation; src/app/api/images/signed-urls/route.ts | 2026-08-08 | -| #284 | P3 | issue | tests/pr-handoff-stop.test.ts fails whenever the suite runs as root | **Outcome:** 'npm run test' is green in a root container, so a real failure is not hidden behind a known one. **Detail:** 'pr-handoff-stop hook > emits handoff context only when the marker file exists' expects markerExists('sess-readonly') to be false — it makes the marker directory read-only and asserts the hook could not write there. Root ignores the permission bits, so the write succeeds and the assertion fails. Reproduced on an unmodified bc33d41 checkout as well as on the viewer-optimisation branch, so it is environment-dependent, not a regression. Cost is that every full-suite run in a root container reports '1 failed', which trains readers to skim past the failure count. **Next:** skip the case when 'process.getuid?.() === 0' with an explicit reason, or drop privileges for that assertion. **Stop:** do not delete the coverage — the read-only case is the point of the test on a normal user account. | session 2026-08-08 full-suite runs; reproduced on bc33d41 | 2026-08-08 | | #286 | P2 | task | PR 2 of the in-page nav series - convert the six pill-rail information pages onto InPageNavHeader | IMPLEMENTED in PR #1766 (branch claude/inpage-nav-pr-2-6d32f9); close this row when that PR merges. All six routes (seven components: services, forms, specifiers record + catalogue reference, formulation, /dsm/diagnoses/ and its /differentials child) mount InPageNavHeader, drop their breadcrumb row, keep their in-body h1, and move record actions into the ellipsis sheet. Three things the conversion needed first, none of which were in the original sketch: the actions render prop had to widen to ReactNode \| ((close) => ReactNode) because four of the seven are Server Components, and onSelectSection plus PageSection.icon have the same RSC-boundary problem, so those four mount the header through a 'use client' sibling module; both sheets derive open state from the pathname so navigation closes them; and information-page sections had no scroll-mt at all, so a shared inPageAnchor token consuming --inpage-anchor-offset was added, published by useInPageChromeMetrics from the live chrome height. The pill rail is deleted with it: hasLocalInformationPageNavigation collapses to isInformationPage and the section kind leaves secondary-navigation.tsx. Guard shipped: tests/in-page-nav-route-sections.dom.test.tsx asserts every declared section id against rendered DOM for all seven components. | session 2026-08-08; PR #1740 (2806d5e) | 2026-08-08 | | #287 | P2 | task | PR 3 of the in-page nav series - the three locally-owned routes each need a decision, not just a conversion | **Outcome:** every information page uses the documented in-page navigation template, or has a recorded reason not to. **Detail:** the last three routes each own a different bespoke pattern, and none is a mechanical port. (1) /medications/[slug] - SectionTabs at medication-record-page.tsx:183 SWAPS CONTENT rather than scrolling: sectionsByTab[activeTab] at :391 filters record.sections by type, so a different set mounts per tab. The InPageNavHeader track is scroll-spy over anchors that all exist at once, so adopting it means either driving tab state from the track (the track stops meaning where am I on the page) or flattening to one scrolling page - a real behaviour change to a clinical record, and a product call. (2) /differentials/presentations/[slug] - MobileTabs at differential-presentation-workflow-page.tsx plus the xl review sidebar; the old `differentialPresentationSections` shell set is gone and the route is locally owned (`page-secondary-navigation.tsx`). Remaining work is the product decision to adopt `InPageNavHeader` (or keep the tab/sidebar model with a recorded reason), not resurrecting deleted section targetIds. (3) /factsheets/[slug] - the On this page list at factsheet-detail-page.tsx:365-371 is li text with no link, button or handler, and the sections themselves carry no ids at all (:214, :267, :313, :447, :453, :471); the tail is data-driven via factsheet.sections.map keyed on section.heading (:538), so anchor ids must be generated deterministically from headings and that generator becomes the contract the section list depends on. Therapy Compass is deliberately excluded from the whole series - ModeNav is a different multi-route pattern. **Next:** decide the medications tab model first (owner decision, blocks planning); decide whether presentations keep MobileTabs/sidebar or adopt InPageNavHeader; choose the factsheets heading-to-id scheme. Then convert. **Stop:** do not port medications mechanically - swapping the tablist for a scroll track silently changes what a clinician sees on a medication record. | session 2026-08-08; follows #286 | 2026-08-08 | | #288 | P3 | rec | Decide whether DocumentViewer adopts the template it was extracted from, or the partial adoption is recorded as final | **Outcome:** the in-page navigation template has one deliberate owner story rather than an unexplained gap. **Detail:** PR 1 (2806d5e, #1740) extracted the header from DocumentViewer.tsx and differential-detail-page.tsx, which held it near-verbatim twice, into src/components/in-page-nav/InPageNavHeader. differential-detail-page was converted onto it; DocumentViewer was deliberately NOT, because its own useDocumentSectionSpy and useDocumentChromeMetrics wiring and its CSS custom-property names (--document-anchor-offset, --document-sticky-header-height, [data-document-sticky-header]) are pinned verbatim by tests/header-scroll-hide-contract.test.ts:110-113. Once #286 and #287 land, the template is adopted on every information page EXCEPT DocumentViewer. `docs/search-chrome-behaviour.md` (Default in-page navigation template) already records that DocumentViewer keeps its own header copy because it owns the page h1, uses edge-glass-header, and is pinned by visual baselines — so the gap is documented, not overlooked. #286 generalises chrome metrics for information pages only; it does not close DocumentViewer convergence. **Next:** owner decision only — convert DocumentViewer later (leaving pinned `--document-*` property names untouched per tests/header-scroll-hide-contract.test.ts:110-113), or explicitly mark the documented non-adoption as the final end state in this ledger when the series closes. **Stop:** do not rename or repoint the pinned document CSS custom properties to unify them with the information-page ones - the contract test pins those exact strings and the document route is the highest-traffic surface in the app. | session 2026-08-08; PR #1740 | 2026-08-08 | @@ -322,13 +314,13 @@ removed after current-main verification; it is not missing recommended work. | #292 | P2 | rec | Two assistants built the same queued conversion twice because neither workflow checks the open-PR list before starting | **Outcome:** picking up a queued ledger item cannot silently duplicate work another session already has in flight. **Detail:** on 2026-08-09 two assistants took the same queued `/issues` item roughly four hours apart and independently built the same in-page-nav conversion — PR #1766 (merged) and PR #1767 (closed as duplicate). Neither had any way to see the other: the ledger row was the only shared state. Correcting an earlier version of this row after CodeRabbit's review on PR #1773: it is not true that the ledger "has no in-progress state" — some rows do carry a progress marker in their prose (`IN PROGRESS` appears on two, and `IMPLEMENTED in PR #1766` on another). The accurate gap is narrower and worse: there is no structured status field and no atomic claim, so a marker is written by whoever did the work, usually after the fact, and nothing requires or checks one — which means the ABSENCE of a marker carries no information at all. Both sessions read it, both correctly concluded it was open, both built it. The wasted effort is the smaller cost; the larger one is that the two implementations diverged in shape, which is what forced the separate `PageSection` ownership decision recorded in `docs/search-chrome-behaviour.md`. Distinct from `#156`/`#168`, which are about two branches colliding on an **id** while appending; this is two sessions colliding on the **work** a row describes, and a collision-free id scheme would leave it untouched. **Mitigation landed 2026-08-09 (same PR as this row):** the check is now written into the three places an assistant actually reads before starting queued work — `.claude/skills/newtask/SKILL.md` "Before you start" (which already performed an open-PR read for PR bundling, so this asks that same list a second question and costs no extra call), `.claude/skills/issues/SKILL.md` after the read-only flow, and the `/issues` section of `AGENTS.md` so Codex and Cursor get it too rather than Claude Code only. All three say to scan for the **route, component or surface**, not the ledger id, because a duplicate PR rarely quotes the id; all three degrade to a warning when GitHub is unreachable so an offline session can still start work. **Next:** leave open for one or two queued-item cycles to see whether prose is enough. If a second duplicate lands anyway, this becomes the same class as `#258` — a rule enforced for one tool by prose with no gate — and the answer is a check, not more wording. **Stop:** do not implement a claim marker written back into the row when a session starts an item; that reintroduces exactly the read-modify-write contention `#168` exists to remove. Do not make the open-PR read a hard blocker. | session 2026-08-09; PR #1766 (merged); PR #1767 (closed duplicate) | 2026-08-09 | | #293 | P2 | issue | Controls declare min-h-tap and compute min-height 0px; a rendered-interactive tap audit needs a deterministic surface first | Two findings, one robust and one that blocked the gate. FINDING 1 (robust, reproduced in ALL SIX runs): controls that carry min-h-tap compute min-height 0px and render far below the 48px floor. Six distinct shapes seen across runs - 'a.inline-flex min-h-tap items-center' 16px, 'a.inline-flex min-h-tap shrink-0' 16.5px, 'button.flex min-h-tap w-full' 26.6px, 'button.grid min-h-tap min-w-tap' 36px, 'button.inline-flex min-h-tap items-center' 16px, 'button.inline-flex min-h-tap min-w-[94px]' 36px. min-h-tap works in general (the existing declared-carrier audit still measures carriers at or above 48px), so these elements have the declaration overridden to 0 rather than the utility being absent; likely an unlayered component class in globals.css, which by design outranks Tailwind utilities here. This was invisible because the pre-existing audit in tests/ui-style-contract.spec.ts only measures elements whose COMPUTED min-height is already at or above the floor (declared < tapFloor - 0.5 continue), so a floor overridden downward is skipped rather than flagged - the same structural blind spot as the h-10 case #265 named. FINDING 2 (why gate 2 did NOT land 2026-08-09): a rendered-interactive enumeration on /services?q=CMHT&run=1 is NOT DETERMINISTIC. Six runs against one production build returned 6, 5, 4, 3, 3 and 9 distinct control shapes, largely disjoint - one run saw answer-suggestion chips and a sort band, another a settled services results list. waitForLoadState('networkidle') plus deduplication to distinct shapes (instance counts measure how many results the query returned, and gave 9 vs 39) did NOT fix it; two consecutive agreeing runs were coincidence, and the next run differed again. The enumeration was written, proven to find real defects, and then REVERTED rather than landed, because tests/ui-style-contract.spec.ts runs in the required Production UI job via productionSpecPattern and scripts/playwright-pr-shards.mjs, so an intermittent version of it would block every merge in the repo. Next, in order: (1) find a deterministic surface for the audit - a static route with no async search, or a fixed seeded state - before re-attempting the enumeration; (2) separately, find what zeroes min-height on the min-h-tap carriers and fix or write a stated exception. Stop: do not re-land the enumeration on a live-search route, do not quarantine a brand-new test to get it merged (quarantine is for keeping flaky tests we already trust, and repo policy needs three reproductions on one SHA via tests/flake-ledger.json), do not lower any production tap target, and never to min-h-11 (known ui-smoke sub-pixel flake; production uses min-h-12). | session 2026-08-09 — M2 gate 2 enumeration (#265) | 2026-08-09 | | #294 | P3 | rec | OffscreenCanvas for the PDF raster is unjustified until the page-flip cost is read from CI | **Outcome:** the worker-raster question is settled by a number rather than left as a standing 'optional' item in the redesign plan. **Detail:** docs/plans/document-viewer-redesign-plan.md conditions OffscreenCanvas on 'measured main-thread paint cost'. Phase 3 (Task 5) did not implement it, deliberately: virtualization now keeps the reader's page and one neighbour already rastered, so the cold-render-per-flip cost that motivated a worker raster is largely gone before any threading work starts, and moving pdf.js rendering off the main thread would put the canvas the clinical source is drawn into behind a transfer boundary — a real risk on the one surface where a blank page is a clinical failure. **No number exists yet and none could be produced locally:** pdfjs-dist@6 needs Map.prototype.getOrInsertComputed, which this container's Chromium 141 lacks and Node 24.13.0 also lacks, so neither a browser nor a headless harness here can raster a page (see #279). **Next:** read the measurement the gate already captures. tests/ui-document-canvas.spec.ts attaches page-flip-raster-cost.json (flipToPaintedMs, longTaskCount, longTaskTotalMs, longestTaskMs, canvasBackingPixels) and logs a '[viewer-canvas] page flip painted in Nms' line, on every Production UI run and on any host with the pinned Chromium 151 build: npm ci --include=dev && npx playwright install chromium && npm run ensure && npm run test:e2e -- tests/ui-document-canvas.spec.ts --project=chromium. Close this as not-worth-doing and strike the row from the plan's Phase 3 table only when a Production UI (or equivalent Chromium 151) run records decisive log lines for all three: longestTaskMs comfortably under ~50ms, plus explicit flipToPaintedMs and longTaskTotalMs budgets agreed for that host class and met on the same run. Do not close on longestTaskMs alone. **Stop:** do not implement OffscreenCanvas on principle because the plan lists it — the plan conditions it on the measurement, and the measurement is now cheap to obtain. | session 2026-08-09 document viewer Phase 3, Task 5; docs/plans/document-viewer-phase3-handover.md | 2026-08-09 | -| #296 | P3 | issue | tests/pr-handoff-stop.test.ts fails whenever the unit suite runs as root | The case 'emits handoff context only when the marker file exists' chmods the fixture git dir to 0o555 to force the marker write to fail, then asserts the hook failed open with no marker. Root ignores permission bits, so the write succeeds and the assertion flips: 'expected true to be false' at tests/pr-handoff-stop.test.ts:180. Confirmed environmental and pre-existing, not diff-induced — reproduced on a clean checkout of af85cbc with every working change stashed (1 failed \| 10 passed), and 'id -u' returns 0 in the remote container. Cost is that 'npm run test' and therefore 'npm run verify:pr-local' cannot reach a clean exit in any root container, so a real regression later in the run is masked by a known-red file and the gate has to be interpreted by hand every time. CI is unaffected because its runner is non-root, which is why this has not surfaced there. Next action: make the test skip or change technique when 'process.getuid?.() === 0' — either skip with an explicit reason, or force the write failure a way root cannot bypass (point the marker path at a directory that does not exist, or at a path whose parent is a file), which is portable and keeps the assertion meaningful for every user. Stop: do not delete the case or relax it to 'marker may or may not exist' — failing open without telling the model that tools are denied is the actual contract it guards. | session 2026-08-09; verify:pr-local run on af85cbc | 2026-08-09 | | #298 | P2 | task | ErrorState is built and registered but nothing enforces it - the '0 matches after a failed request' row is still planned | GATES.md still reads: Render "0 matches" after a failed request \| ErrorState adoption + check \| planned. The component now exists (src/components/ui/error-state.tsx) and is registered across all gate-11/12 surfaces, but grep over scripts/ and eslint-rules/ returns ZERO references to ErrorState, so nothing prevents a new surface rendering a count under a failed status. Building the component closed the 'documented gate with nothing behind it' half; the enforcement half is untouched, and the row was deliberately NOT flipped to implemented because that is exactly the drift GATES.md exists to stop. Next action: add a metric to check:design-system-contract that fails when a count-bearing node renders under an error/failed status - the analyzer already resolves class roots and JSX children (childrenAreNumeralOnly, used by statusColouredNumerals), so it needs no new npm script. Do this BEFORE adoption: a check with no adoption still stops the next regression, adoption with no check does not. Stop: do not flip the GATES.md row until a check actually runs in verify:cheap. | session 2026-08-09 M4 - ErrorState build | 2026-08-09 | | #299 | P3 | task | Adopt ErrorState at the three surfaces that genuinely hand-roll the failed-request guard | Three surfaces hand-roll the guard and their comments state the rule outright: src/components/clinical-dashboard/search-results-header-band.tsx:210 ('no number may reach the DOM'), src/components/services/services-navigator-page.tsx:634 ('a blocked registry must not reach the band as 0 matches'), src/components/clinical-dashboard/favourites-command-library-page.tsx:1182. They are CORRECT today, just not shared, so this is convergence rather than a bug fix. The band's fault panel is the richest existing implementation (role=alert, warning tokens, AsyncButton retry with busy state, faultAction slot) and ErrorState was modelled on it, so the shapes already line up. Live-look change: own PR, Chromium pass. Per the M4 brief it sits DOWNSTREAM of design decisions the owner has not made, so doing it before the site-wide redesign risks redoing it. Do NOT bundle with the enforcement check. Stop: only these three - see the sibling row for three sites that were miscarried as guards. | session 2026-08-09 M4 - ErrorState build | 2026-08-09 | | #300 | P2 | issue | Three sites carried into M4 as hand-rolled '0 matches' guards are not guards - do not convert them | Re-measured 2026-08-09. The M4 handover listed six surfaces hand-rolling the failed-request guard; three do not survive measurement and converting them to ErrorState would be WRONG. (1) src/components/clinical-dashboard/differentials-home.tsx:716,729 renders '0 matches' and 'No matches' when sourcesChecked is TRUE - sourcesChecked: boolean means the source search RAN, so that is a legitimate zero after a search that SUCCEEDED and must keep reporting its count. (2) specifiers-home-page.tsx is not under clinical-dashboard/ at all - the real path is src/components/specifiers/specifiers-home-page.tsx and its line 211 is a comment about not showing a stale zero ABOVE real catalogue results, a different problem. (3) document-search-results.tsx:1508 gates on recordStatus for LOADING (recordSearchStillRunning), not for a failed count; its genuine fault handling is recordBandOwnsFault, which delegates to the band. Only search-results-header-band, services-navigator-page and favourites-command-library-page are real. Next action: none - this row exists so the next reader does not convert the wrong three. Stop: do not 'fix' differentials-home to suppress its count. | session 2026-08-09 M4 - ErrorState build | 2026-08-09 | | #301 | P3 | issue | Two sessions built #262 part 3 in parallel because the GATES.md row understated what had shipped | On 2026-08-09 two branches implemented the same raw-value ratchet independently. PR #1780 landed rawPaddingLiterals/rawRadiusLiterals/rawLineHeightLiterals; a concurrent session built arbitraryPadding/arbitraryGap/arbitraryRadius/arbitraryLeading against the same four files and discovered the collision only when syncing before PR. The duplicate was dropped and only the uncovered gap family was rebuilt on #1780's predicate (rawGapLiterals, 34 sites). Root cause is the same failure this document keeps producing: the §3 row read 'Contract ratchet \| implemented-partial (colour/shadow/tap literals only)' and named none of the metrics #1780 had just shipped, so the row still advertised the work as unstarted. Identical to the 2026-08-09 finding that four of #264's six prohibitions were already gated while their rows read 'planned'. Both rows are corrected now. Next action: when a gate lands, update its §3 row IN THE SAME COMMIT - a row that understates shipped work is not a stale doc, it is a duplicate-work generator. Consider asserting in a test that every metric key in design-system-contract-baseline.json appears somewhere in GATES.md. Stop: do not rely on the ledger alone to prevent this - both sessions had ledger access. | session 2026-08-09 M4; PR #1780 collision | 2026-08-09 | -| #302 | P3 | issue | `tests/helpers/style-contracts.ts` contains escaped line-break artifacts in the exemption map | `smart-search-phone-ticker*` entries were merged with literal backtick-`r`n escapes, which makes the style-exemptions object invalid for the required parse and blocks local checks. Cleanly split each ticker exemption to one line and keep the same reason text so the exception intent is preserved. | PR #1815 unblock follow-up (`tests/helpers/style-contracts.ts`) | 2026-08-11 | -| #303 | P3 | task | `issues:next-id` is out of sync with declared rows | The outstanding-issues marker is `issues:next-id=302` with no `#302`/`#303` rows in either open or resolved tables, which `check:outstanding-issues` flags as missing-issue failures. Add both rows and bump marker to `304` to keep the ledger monotonic. | `docs/outstanding-issues.md` | 2026-08-11 | +| #302 | P3 | rec | Design-system contract ratchets re-accumulate slack because paying debt down does not re-pin the ceiling | On 2026-08-10 the legacyShadowAliases ceiling in scripts/design-system-contract-baseline.json read 220 against a measured 217, so three files could each have gained an alias without failing. Ledger #264 corrected exactly this on 2026-08-09 (edgeOwnershipConflicts 28 to 27, legacyShadowAliases 231 to 224) and it had already re-accumulated one day later. The mechanism is structural, not a one-off: a metric only moves when someone hand-edits the baseline, so every paydown that forgets to re-pin leaves headroom, and nothing in the gate output shows the gap - the check prints the measured value and passes silently while under the ceiling. Per-path pins limit the blast radius but do not close it, since a path whose measured count fell below its pin still carries per-file headroom. Next action: have check:design-system-contract print measured-vs-baseline and the resulting slack per metric, and consider failing when total slack crosses a small threshold, so a forgotten re-pin is visible in the gate rather than found by the next person who measures. Stop: do not auto-write the baseline from measured inside the check - that direction silently absorbs a real regression instead of reporting it. | session 2026-08-10 shadow-tight retirement (PR #1803) | 2026-08-10 | +| #303 | P3 | issue | ledger:append rejects any flag value that starts with a double-dash token, which is every design-token name | npm run ledger:append -- --scope "--shadow-tight migration onto the elevation ladder" fails with 'missing required flag(s): --scope'. The arg parser reads the token after --scope, sees it begin with a double dash, and treats it as the next flag rather than the value, so the required flag reads as absent. Hit on 2026-08-10 recording PR #1803; the workaround was to rewrite the prose so no value begins with a token name, which drops the exact CSS custom-property identifier from the permanent record - the one thing a token-retirement row most needs to name. Every future design-token ledger row hits this, and the error message points at the wrong cause (it reads as a forgotten flag, not a swallowed value). Next action: take the argv entry immediately after a required flag verbatim, or accept the --scope= spelling; check whether scripts/outstanding-issues.mjs shares the parser before fixing only one. Stop: do not settle for a docs note telling authors to avoid leading token names - that is what already costs the identifier. | session 2026-08-10 shadow-tight retirement (PR #1803) | 2026-08-10 | + ## Resolved / archive @@ -337,6 +329,7 @@ Move resolved rows here with the resolution date and a one-line outcome. Keep th | ID | Type | Summary | Outcome | Resolved | | ---- | ----- | --------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------- | +| #261 | task | DS Track A2: retire --shadow-focus from the search composer | RESOLVED 2026-08-11. Both theme declarations deleted; the one consumer, `.chat-composer-shell-delta:focus-within`, now uses the sanctioned `outline: 2px solid var(--focus)` at `outline-offset: 2px` and no longer overrides `box-shadow`, so the pill keeps its resting elevation while focused instead of re-seating (the retired token carried `--shadow-soft` as its second layer). Measured in Chromium both themes: light `solid 2px rgb(29, 111, 184)`, dark `solid 2px rgb(116, 189, 240)`, box-shadow identical resting vs focused in both. CORRECTION to this row's own premise, which was inherited from HANDOVER-2026-08-07 A2: it is NOT a visible production focus change. The consumer class reaches no production route - `chatComposerShell` is imported only by calculators/search-detail.tsx and its mockup twin, production /calculators renders `chatComposerShellBase` + `answer-footer-search-pill` instead, and `CalculatorSearchHome` is reached only from two unrouted mockup exports. Probed all 37 static production routes in Chromium: zero render the class; the single live render is /mockups/calculators-search, which is where the Chromium look was taken. The row was right that a `--include=*.tsx` grep misses the consumer - it is in CSS - but wrong about its reach. Guard: design-token-contract.test.ts rejects both a `--shadow-focus:` declaration and a `var(--shadow-focus)` consumer, in globals.css and the v2 layer, mutation-verified both ways; it is deliberately not the whole-file substring check used for --shadow-tight, because the composer rule names the retired token in a comment on purpose. legacyShadowAliases 127 -> 125 (soft 71 -> 69) with the globals.css per-path pin tightened 3 -> 1: the deleted declarations' VALUE ended in `var(--shadow-soft)`, so they scored as two `soft` aliases - the indirection GATES documents. NOT ratcheted, and still open as pre-existing slack unrelated to this diff: rawPaddingLiterals 67 -> 65 and rawGapLiterals 34 -> 32 (therapy-compass/therapy-card.tsx), layoutTransitionExceptions 12 -> 11 (secondary-navigation.tsx). Gate: npm run check:design-system-contract passed. verify:phone-chrome NOT run - this container's Chromium is rev 1194 against the repo's pinned 1234 (#255 drift), and no production phone chrome renders the class; browser proof delegated to CI. | 2026-08-11 | | #173 | issue | Facet counts and format counts are computed against different sets, so half the filter panel goes stale | RESOLVED 2026-07-31 in PR #1526. `projectSmartTagFacetGroups` recounts an already-built facet index against the live selection so each row answers how many documents remain if that facet is also ticked; zero-count facets stay visible (not removed) and the facet rail disables unselected zeros so they cannot advertise a dead end. Originally captured as open `#172` on this branch before `main` claimed `#169` for the local-branches finding; renumbered to `#173` on merge. | 2026-07-31 | | #160 | task | Reland closed PR #1515 (#093 + #138 fixes never reached main) | RESOLVED 2026-07-31: capture recorded while #1515 was closed unmerged; #1515 then landed on `main` as squash `ca2c4de51faae9a0502b0b0570b6866acbb943fe`, which also archived `#093`/`#138`. **Content-verified on `origin/main` (not SHA/PR state alone):** `tests/playwright-settlement.ts` exports `visibleByTestId` (`.filter({ visible: true })`) and it is used from `tests/ui-tools.spec.ts`, `tests/ui-smoke.spec.ts`, and `tests/ui-accessibility.spec.ts`; `.github/workflows/ci-triage.yml` enables by default with `vars.CI_TRIAGE_ENABLED != 'false'`. Reland no longer needed; chat archive unblocked. | 2026-07-31 | | #093 | issue | Next streaming `S:` clone causes Playwright strict-mode violations under CI load | RESOLVED 2026-07-31: shared `visibleByTestId` scopes page-root/shell testids to the visible DOM owner (not bare `.first()`), applied to the known hotspots in `ui-tools` / `ui-smoke` / `ui-accessibility`. `expectSingleSettledOwner` remains for full-convergence races. Product mount bisect remains optional if a new surface appears. | 2026-07-31 | @@ -499,3 +492,9 @@ Move resolved rows here with the resolution date and a one-line outcome. Keep th | #167 | issue | `verify:pr-local` exits 0 when its own build step refuses to run | Resolved 2026-08-09: guard-next-build now exits 76 when it refuses a build, verify:pr-local propagates the failed selected step, and its self-test plus focused contracts prove the aggregate cannot report green when the build never ran. | 2026-08-09 | | #204 | issue | npm 11.6.2 regenerates a lockfile its own `npm ci` rejects, reddening every CI job | Resolved 2026-08-09: verify:pr-local now selects npm ci --dry-run --ignore-scripts before broad checks whenever package.json or package-lock.json changes; plan self-tests and focused CLI contracts cover both paths. | 2026-08-09 | | #255 | issue | Remote/Cloud containers cannot run any browser gate: Playwright lock drift plus a missing Chromium build | Resolved 2026-08-09: the actual Playwright launch preflight checks the locked browser revision before an explicit executable override or production build in the download-disabled container, while docs/testing.md preserves CI delegation and the verified image-recovery recipe. | 2026-08-09 | +| #284 | issue | tests/pr-handoff-stop.test.ts fails whenever the suite runs as root | Closed 2026-08-10 as a duplicate, superseded by #296. Both rows describe the same environmental failure - tests/pr-handoff-stop.test.ts 'emits handoff context only when the marker file exists' chmods the fixture git dir read-only and asserts the marker write failed, which root ignores. #296 is the row to keep: it cites the exact assertion line (:180), records the reproduction on a clean af85cbc checkout, and offers a portable fix (point the marker path at a parent that is not a directory) alongside the getuid skip. Re-confirmed today on untouched base a16dd26 while verifying PR #1803, which is how the duplicate surfaced. No behaviour changed; the defect stays open under #296. | 2026-08-10 | +| #207 | task | DS V2 PR 13 blocker: AnswerState has no ungrounded-answer channel | Resolved on current main by PR #1658: AnswerState includes an ungrounded channel, the live result surface projects it, and focused answer-state tests protect the warning. | 2026-08-12 | +| #226 | task | PR-J: VerificationNotice pushes the phone short-answer runway past the in-flow activation band | Resolved on current main: the phone short-answer contract is back within the deliberate bounded runway, with ui-smoke asserting maxOffset above 100 and below 200, collapse budget 112 to 128, and post-collapse maxOffset below 72. | 2026-08-12 | +| #230 | issue | CI's PR-policy body sync can overwrite PR descriptions from a committed scratch file | Resolved by this PR: PR_POLICY_BODY.md sync is gated by the file appearing in the current PR diff; inherited files no longer overwrite descriptions, and deleted files safely no-op at runtime. | 2026-08-12 | +| #296 | issue | tests/pr-handoff-stop.test.ts fails whenever the unit suite runs as root | Resolved by this PR: the marker-write failure test uses a directory at the marker path, proving fail-open behavior for root and non-root Linux users without permission-bit assumptions. | 2026-08-12 | + diff --git a/scripts/ci-change-scope.mjs b/scripts/ci-change-scope.mjs index b6c548ccf..27d58c90b 100644 --- a/scripts/ci-change-scope.mjs +++ b/scripts/ci-change-scope.mjs @@ -1,6 +1,6 @@ #!/usr/bin/env node import { execFileSync } from "node:child_process"; -import { appendFileSync, existsSync, readFileSync } from "node:fs"; +import { appendFileSync, readFileSync } from "node:fs"; const zeroSha = /^0{40}$/; @@ -34,7 +34,7 @@ const outputs = [ "codex_autofix_changed", "build_changed", "lockfile_changed", - "pr_policy_body_present", + "pr_policy_body_changed", ]; function normalizePath(filePath) { @@ -362,7 +362,7 @@ function isRecognisedLightPath(filePath) { // drive both the empty and non-empty cases without touching the real file. const readFlakeLedger = () => readFileSync("tests/flake-ledger.json", "utf8"); -function classify(files, { readLedger = readFlakeLedger, prPolicyBodyPresent = existsSync("PR_POLICY_BODY.md") } = {}) { +function classify(files, { readLedger = readFlakeLedger } = {}) { const normalized = [...new Set(files.map(normalizePath).filter(Boolean))].sort(); const docsChanged = normalized.some((file) => pathMatches(file, docPatterns)); const sourceChanged = normalized.some((file) => pathMatches(file, [...sourcePatterns, ...staticConfigPatterns])); @@ -376,6 +376,7 @@ function classify(files, { readLedger = readFlakeLedger, prPolicyBodyPresent = e const workflowChanged = normalized.some((file) => pathMatches(file, workflowPatterns)); const codexAutofixChanged = normalized.some((file) => pathMatches(file, codexAutofixPatterns)); const lockfileChanged = normalized.some((file) => pathMatches(file, lockfilePatterns)); + const prPolicyBodyChanged = normalized.includes("PR_POLICY_BODY.md"); const buildChanged = normalized.some((file) => pathMatches(file, buildPatterns)) || containerChanged; // Only two categories are allowed to take the lightweight path: recognised // documentation and recognised non-executable workflow/policy surfaces. @@ -415,7 +416,7 @@ function classify(files, { readLedger = readFlakeLedger, prPolicyBodyPresent = e codex_autofix_changed: codexAutofixChanged, build_changed: buildChanged, lockfile_changed: lockfileChanged, - pr_policy_body_present: prPolicyBodyPresent, + pr_policy_body_changed: prPolicyBodyChanged, }; } @@ -1172,12 +1173,12 @@ function selfTest() { static_heavy_changed: true, coverage_changed: true, }); - assertScope( - "pr-policy-body-presence-is-routed-without-a-second-checkout", - ["PR_POLICY_BODY.md"], - { pr_policy_body_present: true }, - { prPolicyBodyPresent: true }, - ); + assertScope("pr-policy-body-change-is-routed-from-the-pr-diff", ["PR_POLICY_BODY.md"], { + pr_policy_body_changed: true, + }); + assertScope("inherited-pr-policy-body-does-not-sync", ["docs/testing.md"], { + pr_policy_body_changed: false, + }); console.log("CI change scope self-test passed."); } diff --git a/src/lib/rag/rag-cache.ts b/src/lib/rag/rag-cache.ts index ea4248066..8290b9452 100644 --- a/src/lib/rag/rag-cache.ts +++ b/src/lib/rag/rag-cache.ts @@ -447,6 +447,14 @@ function sharedCacheSelector( return query; } +const GENERATION_FALLBACK_MARKER = /(?:^|;\s*)generation_fallback(?::|$)/i; +function isGenerationFallbackAnswer(answer: Pick) { + return ( + GENERATION_FALLBACK_MARKER.test(answer.routingReason ?? "") || + GENERATION_FALLBACK_MARKER.test(answer.degradedMode?.reason ?? "") + ); +} + export async function cacheIndexingVersion( args: Pick, options?: { forceRefresh?: boolean }, @@ -613,6 +621,10 @@ export async function getSharedCachedAnswer( ).maybeSingle(); if (error || !data?.payload) return null; const answer = cloneAnswer((data.payload as { answer: RagAnswer }).answer); + if (isGenerationFallbackAnswer(answer)) { + await deleteSharedCachedAnswerRow({ ...args, accessScope: retrievalAccessScopeForArgs(args) }, indexingVersion); + return null; + } answer.routingReason = answer.routingReason ? `${answer.routingReason}; shared_answer_cache_hit` : "shared_answer_cache_hit"; diff --git a/src/lib/rag/rag-route-budget.ts b/src/lib/rag/rag-route-budget.ts index 9f412bd5a..f6e371a27 100644 --- a/src/lib/rag/rag-route-budget.ts +++ b/src/lib/rag/rag-route-budget.ts @@ -1,4 +1,5 @@ import type { AnswerRouteMode } from "@/lib/rag/rag-routing"; +import type { RagAnswer } from "@/lib/types"; export const answerRouteBudgetMs = { unsupported: 0, @@ -17,6 +18,12 @@ export const generationRecoveryReserveMs = 2_000; // spends MORE reasoning under a boosted cap, so anything shorter is a guaranteed-discard. export const minimumGenerationRetryMs = 5_000; +// Keep in lockstep with rag-cache / isProviderGenerationDegraded: bare +// `generation_fallback` (no `:reason`) and case variants must refuse the cache. +// Inline the marker here so this leaf module does not import rag-answer-support +// (that path pulls env through deep-memory and freezes the offline vitest snapshot). +const GENERATION_FALLBACK_MARKER = /(?:^|;\s*)generation_fallback(?::|$)/i; + export class AnswerRouteDeadlineExceededError extends Error { readonly routeMode: AnswerRouteMode; readonly budgetMs: number; @@ -49,8 +56,15 @@ export function deadlineAllowsGenerationRetry(deadline: Pick= generationRecoveryReserveMs + minimumGenerationRetryMs; } -export function answerRouteResultCanBeCached(deadline: Pick) { - return !deadline.deadlineExceeded; +export function answerRouteResultCanBeCached( + deadline: Pick, + answer: Pick, +) { + return ( + !deadline.deadlineExceeded && + !GENERATION_FALLBACK_MARKER.test(answer.routingReason ?? "") && + !GENERATION_FALLBACK_MARKER.test(answer.degradedMode?.reason ?? "") + ); } function abortReason(signal: AbortSignal) { diff --git a/src/lib/rag/rag.ts b/src/lib/rag/rag.ts index b32bc1a40..debe71a51 100644 --- a/src/lib/rag/rag.ts +++ b/src/lib/rag/rag.ts @@ -2413,23 +2413,6 @@ function buildContextDerivedArtifacts(query: string, results: SearchResult[]) { }; } -export function isCacheableGroundedGenerationFallback( - answer: Pick< - RagAnswer, - "routingMode" | "routingReason" | "grounded" | "confidence" | "citations" | "unverifiedNumericTokens" - >, -) { - return ( - answer.routingMode === "extractive" && - answer.grounded && - answer.confidence !== "unsupported" && - answer.citations.length > 0 && - (answer.unverifiedNumericTokens?.length ?? 0) === 0 && - /(?:source_backed_extractive_fallback|comparison_source_safe_fallback)/.test(answer.routingReason ?? "") && - !(answer.routingReason ?? "").includes(SOURCE_BACKED_REVIEW_FALLBACK_REASON) - ); -} - /** Answer question. */ export async function answerQuestion(query: string, documentId?: string) { return answerQuestionWithScope({ query, documentId, allowGlobalSearch: true }); @@ -2963,7 +2946,7 @@ async function answerQuestionWithScopeUncoalesced( // Soft-tail unsupported refusals must not stick in the 5-minute answer cache. if ( - answerRouteResultCanBeCached(routeDeadline) && + answerRouteResultCanBeCached(routeDeadline, finalizedAnswer) && !shouldSkipUnsupportedSoftTailAnswerCacheWrite({ resultCount: results.length, retrievalStrategy: search.telemetry.retrieval_strategy, @@ -3158,7 +3141,7 @@ async function answerQuestionWithScopeUncoalesced( }, }); - if (!routeDeadline.deadlineExceeded) + if (answerRouteResultCanBeCached(routeDeadline, finalizedAnswer)) await setCachedAnswer(args, finalizedAnswer, { indexingVersionAtRetrievalStart }); routeDeadline.dispose(); return finalizedAnswer; @@ -3924,7 +3907,7 @@ ${qualityRetryInstruction}` }, }); - if (answerRouteResultCanBeCached(routeDeadline)) + if (answerRouteResultCanBeCached(routeDeadline, answer)) await setCachedAnswer(args, answer, { indexingVersionAtRetrievalStart }); routeDeadline.dispose(); return answer; @@ -4279,7 +4262,7 @@ ${qualityRetryInstruction}` }, }); - if (isCacheableGroundedGenerationFallback(fallbackAnswer) && !routeDeadline.deadlineExceeded) { + if (answerRouteResultCanBeCached(routeDeadline, fallbackAnswer)) { await setCachedAnswer(args, fallbackAnswer, { indexingVersionAtRetrievalStart }); } routeDeadline.dispose(); diff --git a/tests/ci-cache-safety.test.ts b/tests/ci-cache-safety.test.ts index d321dfaf7..2073e357c 100644 --- a/tests/ci-cache-safety.test.ts +++ b/tests/ci-cache-safety.test.ts @@ -131,7 +131,7 @@ describe("CI cache safety", () => { expect(opsDigestWorkflow).toContain("eval-canary-liveness:"); expect(opsDigestWorkflow).toContain("github.rest.actions.listWorkflowRuns"); expect(workflow).toContain( - "if: github.event_name == 'pull_request' && needs.changes.outputs.pr_policy_body_present == 'true'", + "if: github.event_name == 'pull_request' && needs.changes.outputs.pr_policy_body_changed == 'true'", ); }); diff --git a/tests/pr-handoff-stop.test.ts b/tests/pr-handoff-stop.test.ts index 44a26aaf6..33e3a2d09 100644 --- a/tests/pr-handoff-stop.test.ts +++ b/tests/pr-handoff-stop.test.ts @@ -1,5 +1,5 @@ import { execFileSync, spawnSync } from "node:child_process"; -import { chmodSync, existsSync, mkdtempSync, rmSync, utimesSync, writeFileSync } from "node:fs"; +import { existsSync, lstatSync, mkdirSync, mkdtempSync, rmSync, utimesSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { afterEach, describe, expect, it } from "vitest"; @@ -162,26 +162,24 @@ describe("pr-handoff-stop hook", () => { it("emits handoff context only when the marker file exists", () => { const { root, gitDir } = freshRepo(); - // Make the git dir unwritable so the marker write fails; post must fail - // open with no additionalContext (model must not be told tools are denied). - chmodSync(gitDir, 0o555); + // A directory at the exact marker path makes shell redirection fail for + // root and non-root users. Post must fail open with no additionalContext + // (the model must not be told tools are denied when no marker file landed). + const marker = join(gitDir, "claude-pr-handoff-sess-readonly"); + mkdirSync(marker); - try { - const out = runHook( - "post", - { - tool_name: "create_pull_request", - session_id: "sess-readonly", - tool_response: "Opened https://github.com/BigSimmo/Database/pull/1649", - }, - root, - ); - expect(out.status).toBe(0); - expect(out.markerExists("sess-readonly")).toBe(false); - expect(out.stdout).toBe(""); - } finally { - chmodSync(gitDir, 0o755); - } + const out = runHook( + "post", + { + tool_name: "create_pull_request", + session_id: "sess-readonly", + tool_response: "Opened https://github.com/BigSimmo/Database/pull/1649", + }, + root, + ); + expect(out.status).toBe(0); + expect(lstatSync(marker).isDirectory()).toBe(true); + expect(out.stdout).toBe(""); }); it("denies quoted compound follow commands when jq is unavailable", () => { diff --git a/tests/rag-answer-fallback.test.ts b/tests/rag-answer-fallback.test.ts index 9a4124f8e..5d05302bb 100644 --- a/tests/rag-answer-fallback.test.ts +++ b/tests/rag-answer-fallback.test.ts @@ -1,6 +1,10 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; import { citationFromResult } from "../src/lib/citations"; -import { answerRouteBudgetMs, generationRecoveryReserveMs } from "../src/lib/rag/rag-route-budget"; +import { + answerRouteBudgetMs, + answerRouteResultCanBeCached, + generationRecoveryReserveMs, +} from "../src/lib/rag/rag-route-budget"; import type { RagAnswer, SearchResult } from "../src/lib/types"; function retrievalRpcBaseName(name: string) { @@ -90,8 +94,13 @@ async function answerFromTextSources( generatedAnswer?: GeneratedAnswerPayload | Error, options: { sourceOnly?: boolean } = {}, ) { + // `src/lib/env.ts` freezes process.env at module load. The offline vitest wrapper + // starts every worker as RAG_PROVIDER_MODE=offline with a blank OpenAI key, so we + // must re-parse env after stubbing — otherwise the first test in this file keeps + // the runner's offline snapshot and never exercises the mocked provider path. + vi.resetModules(); vi.stubEnv("OPENAI_API_KEY", options.sourceOnly ? "" : "test-key"); - if (options.sourceOnly) vi.stubEnv("RAG_PROVIDER_MODE", "offline"); + vi.stubEnv("RAG_PROVIDER_MODE", options.sourceOnly ? "offline" : "auto"); vi.stubEnv("RAG_SEARCH_CACHE_TTL_MS", "0"); vi.stubEnv("RAG_ANSWER_CACHE_TTL_MS", "0"); @@ -3387,7 +3396,7 @@ describe("RAG structured-output fallback", () => { generateStructuredTextResult, })); - const { answerQuestionWithScope, isCacheableGroundedGenerationFallback } = await import("../src/lib/rag/rag"); + const { answerQuestionWithScope } = await import("../src/lib/rag/rag"); const progressEvents: Array<{ stage: string; selectedContextCount?: number; @@ -3430,7 +3439,7 @@ describe("RAG structured-output fallback", () => { ]); expect(answer.openAIRequestIds).toEqual(["req_truncated_1", "req_truncated_2"]); expect(answer.openAIUsage).toMatchObject({ output_tokens: 1300, total_tokens: 1500 }); - expect(isCacheableGroundedGenerationFallback(answer)).toBe(true); + expect(answerRouteResultCanBeCached({ deadlineExceeded: false }, answer)).toBe(false); expect(progressEvents).toContainEqual( expect.objectContaining({ stage: "ranking", @@ -3471,14 +3480,12 @@ describe("RAG structured-output fallback", () => { ], new Error("OpenAI generation incomplete: max_output_tokens"), ); - const { isCacheableGroundedGenerationFallback } = await import("../src/lib/rag/rag"); - expect(answer.answer).not.toMatch(/fluoxetine|citalopram|60 mg|40 mg/i); expect(answer.answer).toMatch(/source|guidance|support|evidence/i); expect(answer.routingReason).toContain("generation_fallback:provider_incomplete_max_output_tokens"); expect(answer.routingReason).toContain("source_backed_review_fallback"); expect(answer.unverifiedNumericTokens ?? []).toEqual([]); - expect(isCacheableGroundedGenerationFallback(answer)).toBe(false); + expect(answerRouteResultCanBeCached({ deadlineExceeded: false }, answer)).toBe(false); }); it("prefers the safe single-chunk fallback candidate that carries the asked-for dose figure", async () => { @@ -3523,40 +3530,25 @@ describe("RAG structured-output fallback", () => { expect(new Set(answer.citations.map((citation) => citation.chunk_id))).toEqual(new Set(["quetiapine-maximum-1"])); }); - it("never marks the generic source-review fallback as cacheable", async () => { - const { isCacheableGroundedGenerationFallback } = await import("../src/lib/rag/rag"); - + it("never marks provider-generation fallbacks as cacheable", async () => { expect( - isCacheableGroundedGenerationFallback({ - routingMode: "unsupported", - routingReason: "strong_generation; generation_fallback:provider_timeout", - grounded: false, - confidence: "low", - citations: [], - unverifiedNumericTokens: [], - }), + answerRouteResultCanBeCached( + { deadlineExceeded: false }, + { + routingReason: "strong_generation; generation_fallback:provider_timeout", + degradedMode: { active: true, reason: "generation_fallback:provider_timeout" }, + }, + ), ).toBe(false); expect( - isCacheableGroundedGenerationFallback({ - routingMode: "extractive", - routingReason: - "strong_generation; generation_fallback:provider_timeout; source_backed_review_fallback; extractive_quality_gate:weak", - grounded: true, - confidence: "low", - citations: [ - { - chunk_id: "source-1", - document_id: "document-1", - title: "Source", - file_name: "source.pdf", - page_number: 1, - chunk_index: 0, - source_metadata: null, - provenance: "deterministic_support", - }, - ], - unverifiedNumericTokens: [], - }), + answerRouteResultCanBeCached( + { deadlineExceeded: false }, + { + routingReason: + "strong_generation; generation_fallback:provider_timeout; source_backed_review_fallback; extractive_quality_gate:weak", + degradedMode: { active: true, reason: "generation_fallback:provider_timeout" }, + }, + ), ).toBe(false); }); diff --git a/tests/rag-route-budget.test.ts b/tests/rag-route-budget.test.ts index b9e4f5701..b61bce01e 100644 --- a/tests/rag-route-budget.test.ts +++ b/tests/rag-route-budget.test.ts @@ -61,13 +61,50 @@ describe("RAG route deadlines", () => { it("does not allow a result to be cached after its route deadline", async () => { vi.useFakeTimers(); const deadline = createAnswerRouteDeadline({ routeMode: "extractive" }); + const answer = { routingReason: "high_confidence_extractive_retrieval", degradedMode: undefined }; - expect(answerRouteResultCanBeCached(deadline)).toBe(true); + expect(answerRouteResultCanBeCached(deadline, answer)).toBe(true); await vi.advanceTimersByTimeAsync(answerRouteBudgetMs.extractive); - expect(answerRouteResultCanBeCached(deadline)).toBe(false); + expect(answerRouteResultCanBeCached(deadline, answer)).toBe(false); deadline.dispose(); }); + + it("never caches provider-generation fallbacks before the route deadline", () => { + const deadline = { deadlineExceeded: false }; + + expect( + answerRouteResultCanBeCached(deadline, { + routingReason: "clinical_fast_grounded_synthesis; generation_fallback:provider_timeout", + degradedMode: { active: true, reason: "generation_fallback:provider_timeout" }, + }), + ).toBe(false); + expect( + answerRouteResultCanBeCached(deadline, { + routingReason: "clinical_fast_grounded_synthesis; source_backed_extractive_fallback", + degradedMode: { active: true, reason: "generation_fallback:provider_timeout" }, + }), + ).toBe(false); + expect( + answerRouteResultCanBeCached(deadline, { + routingReason: "strong_generation; generation_fallback", + degradedMode: { active: true, reason: "generation_fallback" }, + }), + ).toBe(false); + // Bare routing marker alone (no degradedMode) must refuse the cache too. + expect( + answerRouteResultCanBeCached(deadline, { + routingReason: "strong_generation; generation_fallback", + degradedMode: undefined, + }), + ).toBe(false); + expect( + answerRouteResultCanBeCached(deadline, { + routingReason: "clinical_fast; Generation_Fallback:provider_timeout", + degradedMode: undefined, + }), + ).toBe(false); + }); }); describe("budget-aware generation deadlines (E-3b)", () => { diff --git a/tests/rag-shared-cache.test.ts b/tests/rag-shared-cache.test.ts index a74f1eacf..3d3a7f58d 100644 --- a/tests/rag-shared-cache.test.ts +++ b/tests/rag-shared-cache.test.ts @@ -1,10 +1,64 @@ import { afterEach, describe, expect, it, vi } from "vitest"; +const ownerId = "aaaaaaaa-aaaa-4aaa-8aaa-aaaaaaaaaaaa"; + afterEach(() => { vi.restoreAllMocks(); vi.resetModules(); }); +function createSharedCacheBuilder(payload: { + answer: { routingReason: string; degradedMode?: { active: boolean; reason: string } }; +}) { + const sharedBuilder = { + select: () => sharedBuilder, + eq: () => sharedBuilder, + is: () => sharedBuilder, + in: () => sharedBuilder, + gt: () => sharedBuilder, + order: () => sharedBuilder, + limit: () => sharedBuilder, + maybeSingle: async () => ({ data: { payload }, error: null }), + then: (resolve: (value: { data: unknown; error: null }) => unknown) => + Promise.resolve({ data: null, error: null }).then(resolve), + }; + + let deletionCalls = 0; + const deleteBuilder = { + eq: () => deleteBuilder, + is: () => deleteBuilder, + in: () => deleteBuilder, + then: (resolve: (value: { data: unknown; error: null }) => unknown) => { + deletionCalls += 1; + return Promise.resolve({ data: null, error: null }).then(resolve); + }, + }; + + return { + builder: { + select: () => sharedBuilder, + delete: () => deleteBuilder, + eq: () => sharedBuilder, + is: () => sharedBuilder, + in: () => sharedBuilder, + gt: () => sharedBuilder, + order: () => sharedBuilder, + limit: () => sharedBuilder, + then: (resolve: (value: { data: unknown[]; error: null }) => unknown, reject?: (reason: unknown) => unknown) => { + return Promise.resolve({ + data: [{ id: "doc-1", updated_at: "2026-07-14T00:00:00.000Z", metadata: {} }], + error: null, + }).then(resolve, reject); + }, + }, + sharedBuilder, + deleteBuilder, + get deletionCalls() { + return deletionCalls; + }, + } as const; +} + describe("shared RAG search cache", () => { it("uses one shared-cache read for a cold filtered miss", async () => { vi.resetModules(); @@ -125,4 +179,71 @@ describe("shared RAG search cache", () => { expect(result?.kind).toBe("hit"); expect(result?.kind === "hit" && result.telemetry.corpus_grounding).toBe("in_corpus_topic"); }); + + it("does not serve generation-fallback answers from shared cache and removes them", async () => { + vi.resetModules(); + const fallback = { + answer: { + routingReason: "clinical_fast_grounded_synthesis; generation_fallback", + degradedMode: { active: true, reason: "generation_fallback" }, + }, + }; + const shared = createSharedCacheBuilder(fallback); + + vi.doMock("@/lib/env", () => ({ + env: { + RAG_SEARCH_CACHE_TTL_MS: 60_000, + RAG_SEARCH_CACHE_SIZE: 200, + RAG_ANSWER_CACHE_TTL_MS: 60_000, + RAG_ANSWER_CACHE_SIZE: 200, + RAG_PERSIST_RAW_QUERY_TEXT: false, + RAG_QUERY_HASH_SECRET: "test-query-hash-secret", + }, + isDemoMode: () => false, + isLocalNoAuthMode: () => false, + })); + vi.doMock("@/lib/deep-memory", () => ({ ragDeepMemoryVersion: "test-rag-version" })); + vi.doMock("@/lib/clinical-search", () => ({ + buildClinicalTextSearchQuery: (query: string) => query.trim(), + })); + vi.doMock("@/lib/supabase/admin", () => ({ + createAdminClient: () => ({ + from: (table: string) => { + if (table === "rag_response_cache") return shared.builder; + return { + select: () => shared.sharedBuilder, + eq: () => shared.sharedBuilder, + is: () => shared.sharedBuilder, + or: () => shared.sharedBuilder, + gt: () => shared.sharedBuilder, + order: () => shared.sharedBuilder, + limit: () => shared.sharedBuilder, + then: ( + resolve: (value: { data: unknown[]; error: null }) => unknown, + reject?: (reason: unknown) => unknown, + ) => + Promise.resolve({ + data: [{ id: "doc-1", updated_at: "2026-07-14T00:00:00.000Z", metadata: {} }], + error: null, + }).then(resolve, reject), + maybeSingle: async () => ({ data: null, error: null }), + }; + }, + }), + })); + + const { getSharedCachedAnswer } = await import("../src/lib/rag/rag-cache"); + const result = await getSharedCachedAnswer( + { + query: "clinical deterioration", + ownerId, + queryMode: "auto", + forceEmbedding: false, + }, + Date.now(), + ); + + expect(result).toBeNull(); + expect(shared.deletionCalls).toBe(1); + }); });