diff --git a/.claude/hooks/issues-surface.sh b/.claude/hooks/issues-surface.sh index f22141372..e0e25fbc6 100755 --- a/.claude/hooks/issues-surface.sh +++ b/.claude/hooks/issues-surface.sh @@ -1,11 +1,10 @@ #!/usr/bin/env bash # SessionStart hook — surface the outstanding-work memory into context. # -# Reads docs/outstanding-issues.md (the /issues ledger) and prints a compact, -# glanceable summary of the OPEN items so every session starts already aware of -# what is outstanding. When the trigger is a context reset (compact / resume / -# clear) it also emits a reminder to run `/issues capture` — that is the moment -# a session's in-flight follow-ups are most likely to be lost. +# Reads docs/outstanding-issues.md (the universal /issues ledger) and prints the +# ordered recommended tasks plus open-item counts so every session starts with +# the same repository-wide priorities. When the trigger is a context reset +# (compact / resume / clear) it also emits a reminder to run `/issues capture`. # # Contract: READ-ONLY. Never writes, never commits, never fails a session — it # always exits 0, and every step is guarded so a parse error just yields less @@ -43,25 +42,29 @@ rows="$(awk ' } ' "$ledger" 2>/dev/null || true)" -# --- parse the recommended execution queue ---------------------------------- -queue_rows="$(awk ' - /^## Recommended execution queue/ { inqueue=1; next } - /^## / { if (inqueue) inqueue=0 } - inqueue && /^\|[[:space:]]*[0-9]+[[:space:]]*\|/ { +# --- parse the ordered recommended execution queue -------------------------- +# Emit "ORDERIDSUMMARYACUITYWHENESTIMATE" per recommended row. +recommended="$(awk ' + /^## Recommended execution queue/ { inrecommended=1; next } + /^## / { if (inrecommended) inrecommended=0 } + inrecommended && /^\|[[:space:]]*[0-9]+[[:space:]]*\|/ { n=split($0, c, "|") - ord=c[2]; ids=c[3]; acuity=c[4]; capability=c[5]; timing=c[6] - gsub(/^[ \t]+|[ \t]+$/, "", ord) - gsub(/^[ \t]+|[ \t]+$/, "", ids) + order=c[2]; id=c[3]; summary=c[4]; acuity=c[5]; timing=c[6]; estimate=c[7] + gsub(/^[ \t]+|[ \t]+$/, "", order) + gsub(/^[ \t]+|[ \t]+$/, "", id) + gsub(/^[ \t]+|[ \t]+$/, "", summary) gsub(/^[ \t]+|[ \t]+$/, "", acuity) - gsub(/^[ \t]+|[ \t]+$/, "", capability) gsub(/^[ \t]+|[ \t]+$/, "", timing) - printf "%s\t%s\t%s\t%s\t%s\n", ord, ids, acuity, capability, timing + gsub(/^[ \t]+|[ \t]+$/, "", estimate) + if (length(summary) > 100) summary=substr(summary, 1, 97) "..." + printf "%s\t%s\t%s\t%s\t%s\t%s\n", order, id, summary, acuity, timing, estimate } ' "$ledger" 2>/dev/null || true)" total="$(printf '%s' "$rows" | grep -c . || true)" -if [ "${total:-0}" -eq 0 ]; then - echo "[issues] Outstanding-work memory (docs/outstanding-issues.md): no open items. Record one with /issues add …" +recommended_total="$(printf '%s' "$recommended" | grep -c . || true)" +if [ "${total:-0}" -eq 0 ] && [ "${recommended_total:-0}" -eq 0 ]; then + echo "[issues] Universal task ledger (docs/outstanding-issues.md): no recommended or open items. Record one with /issues add …" exit 0 fi @@ -70,25 +73,25 @@ count() { printf '%s' "$1" | grep -c . || true; } p1="$(group P1)"; p2="$(group P2)"; p3="$(group P3)" c1="$(count "$p1")"; c2="$(count "$p2")"; c3="$(count "$p3")" -echo "[issues] Outstanding-work memory — ${total} open (${c1}×P1, ${c2}×P2, ${c3}×P3). Source of truth: docs/outstanding-issues.md · read the full list back with /issues." - -queue_total="$(printf '%s' "$queue_rows" | grep -c . || true)" -if [ "${queue_total:-0}" -gt 0 ]; then - echo "[issues] Recommended execution queue — ${queue_total} retained tasks (first 8):" - printf '%s\n' "$queue_rows" | head -n 8 | while IFS=$'\t' read -r ord ids acuity capability timing; do - echo " ${ord}. ${ids} · ${acuity} · ${timing} · ${capability}" - done -fi +echo "[issues] Universal task ledger — ${recommended_total} recommended · ${total} open (${c1}×P1, ${c2}×P2, ${c3}×P3). Source of truth: docs/outstanding-issues.md · read the full ledger with /issues." -# Keep the priority summary complementary to the queue instead of repeating -# the same recommended IDs in both sections. -queued_ids=" $(printf '%s\n' "$queue_rows" | grep -oE '#[0-9]+' | tr '\n' ' ' || true)" -unqueued_rows="$(printf '%s\n' "$rows" | awk -F'\t' -v queued="$queued_ids" ' - index(queued, " " $2 " ") == 0 -' || true)" -ungroup() { printf '%s\n' "$unqueued_rows" | awk -F'\t' -v p="$1" '$1==p'; } -u1="$(ungroup P1)"; u2="$(ungroup P2)"; u3="$(ungroup P3)" -uc1="$(count "$u1")"; uc2="$(count "$u2")"; uc3="$(count "$u3")" +print_recommended() { # $1=max-to-list + local limit="$1" shown=0 more=0 order id summary acuity timing estimate + [ -z "$recommended" ] && return 0 + while IFS=$'\t' read -r order id summary acuity timing estimate; do + [ -z "$order" ] && continue + if [ "$shown" -lt "$limit" ]; then + echo " ${order} ${id} ${acuity} — ${summary} (${timing}; ${estimate})" + shown=$((shown + 1)) + else + more=$((more + 1)) + fi + done <`** — append a row to **Open items**. Infer `Pri`/`Type` from the text (ask only if genuinely ambiguous; default `P2`/`task`). Allocate the ID from the `` marker, then bump that marker. Fill `Source` with - `session ` unless the user names one; `Added` is today's date. + `session ` unless the user names one; `Added` is today's date. If the work is currently + recommended, also add it to the ordered execution ledger with acuity, intelligence, timing, + estimate, dependency, and completion signal; otherwise retain it only in Open items. - **`/issues done [outcome]`** — move that row from **Open items** to **Resolved / archive** - with today's date and a one-line outcome. Archive, never delete. + with today's date and a one-line outcome, remove it from the recommended execution queue, and + close the order gap. Archive, never delete. - **`/issues update `** — edit an open row's summary or next action in place. - **`/issues capture`** — scan the current session for recommendations, follow-ups, deferrals, and unfixed problems that surfaced but were not recorded. Propose them as a numbered list and add the @@ -60,6 +65,9 @@ paragraph; put the smallest next action in **Detail / next action**. - Remove a task from the recommended queue when it completes or is no longer recommended; retain its evidence in the open or resolved table as appropriate. - IDs are monotonic and never reused — always allocate from the `issues:next-id` marker and bump it. +- Keep the recommended execution queue dependency-ordered, gap-free, deduplicated, and synchronized + with its referenced open rows. Never add refuted, parked, superseded, resolved, or decision-only + records to the active recommendation view. - Escape `|` inside cell text (write `\|`) so the markdown table stays intact. - Respect the repo's RAG/clinical/privacy flagging rules if an item _itself_ touches a protected surface — recording it here is fine, but acting on it later still needs the usual gate. diff --git a/AGENTS.md b/AGENTS.md index 13deab25f..d8f0c3891 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -448,28 +448,29 @@ Run the matching planner command in `docs/productivity-workflows.md` without sid -## Outstanding-work memory (`/issues`) +## Universal repository task ledger (`/issues`) `docs/outstanding-issues.md` is the single universal, durable, cross-session ledger for every -outstanding **task**, **recommendation**, and **issue** in this repo. It owns evidence, resolution -history, recommended order, acuity, capability, timing, effort, approvals, verification, and stop -rules. Chat context resets; this file does not, so anything worth remembering belongs there. -Detailed runbooks such as `docs/operator-backlog.md` may support a task but must not become a second -status ledger. Update the universal ledger when work completes, is dropped, becomes stale, or is -materially re-scoped. Never restore completed, duplicate, speculative, superseded, or rejected work -to the recommended queue. - +outstanding **task**, **recommendation**, and **issue** in this repo. It owns the recommended +execution order, acuity, required capability, timing, effort, approvals, evidence, status, success +criteria, stop rules, and resolution history. Chat context resets between sessions; that file does +not, so anything worth remembering after a session ends belongs there. Do not create or maintain a +second task ledger. + +- Before starting or recommending repository work, read the ordered **Recommended execution + queue** and the referenced open item. Only rows in that ordered view are active recommendations; + refuted, parked, superseded, resolved, and decision-only records remain audit history, not tasks. - When the user types `/issues`, invoke the `issues` skill (`.claude/skills/issues/SKILL.md`): read - `docs/outstanding-issues.md`, state the recommended queue in order, then summarize other open - items by priority. A plain `/issues` is read-only — it mutates and commits nothing. + `docs/outstanding-issues.md` and state the recommended execution queue in order, then summarize + the wider open-item counts. A plain `/issues` is read-only — it mutates and commits nothing. - `/issues add|done|update|capture …` mutate the ledger; each mutation commits **only** `docs/outstanding-issues.md` (no push unless the user asks or you are already handing off). - Proactively offer to `capture` unresolved follow-ups, deferrals, and known risks into the ledger before a session's context is lost — that is what keeps it a memory rather than a stale list. - A `SessionStart` hook (`.claude/hooks/issues-surface.sh`, wired in `.claude/settings.json`) - auto-surfaces the open items into context at the start of every session and, on a context reset - (`compact`/`resume`/`clear`), nudges a `/issues capture`. It is read-only — it never writes the - ledger. `/issues` is still the way to read the full list or mutate it. + auto-surfaces the ordered recommended tasks plus open-item counts at the start of every session + and, on a context reset (`compact`/`resume`/`clear`), nudges a `/issues capture`. It is read-only — + it never writes the ledger. `/issues` is still the way to read the full list or mutate it. ## Codex GitHub review behavior diff --git a/docs/README.md b/docs/README.md index cb90326ff..5adb6a414 100644 --- a/docs/README.md +++ b/docs/README.md @@ -72,6 +72,7 @@ npm run docs:check-links ## Plans and workstreams (living) +- [outstanding-issues.md](outstanding-issues.md) — universal task ledger, recommended execution order, evidence, status, and resolution history - [maturity-backlog-workorders.md](maturity-backlog-workorders.md) — actionable work orders tracking the repository-maturity audit backlog - [framework-dependency-modernization-checklist.md](framework-dependency-modernization-checklist.md) — ordered Next.js 16, runtime, dependency, Turbopack, and verification migration program - [search-rag-master-plan.md](search-rag-master-plan.md) / [search-rag-master-context.md](search-rag-master-context.md) — search/RAG roadmap and shared context diff --git a/docs/branch-review-ledger.md b/docs/branch-review-ledger.md index 65d17fc47..96e57f5b7 100644 --- a/docs/branch-review-ledger.md +++ b/docs/branch-review-ledger.md @@ -20,8 +20,8 @@ Use this ledger to prevent repeated branch and PR reviews when the reviewed HEAD | Date | Branch or ref | Reviewed HEAD | Scope | Outcome | Checks | | ---------- | -------------------------------------------------------- | ---------------------------------------- | ---------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| 2026-07-24 | PR #1106 / `codex/task-ledger-final-11318f` | `5d128a2844c2298d0da36df64e5e2f7dda11e14b` + reviewed follow-up diff | Universal task-ledger workflow and protected-main merge readiness | APPROVE after follow-up. `docs/outstanding-issues.md` is the single durable task ledger, with retained work carrying order, acuity, timing, capability, effort, dependencies, success criteria, verification and stop rules. Four actionable review findings were fixed: filtered `/issues` reads now apply the filter to open items before rendering queued and non-queued results; the session hook excludes queued IDs from its priority summary; `#030` is consistently P2/A2 in the canonical open table and queue; and the sole A1/P1 blocker is first while `#052` is explicitly the first code task. No other actionable review thread remains in the reviewed scope. | Protected CI at the initial reviewed head passed policy, static, safety/config, unit coverage, Semgrep, Gitleaks, GitGuardian and the required aggregate; UI, build, migration replay and release browser matrix were correctly skipped for the docs/workflow scope. Follow-up proof: scoped Prettier; hook syntax/runtime plus exact ID-deduplication, P2-count and A1-first assertions; docs links (1,136 references); canonical skill catalog (32 skills, 8 aliases); `git diff --check`. Exact-head hosted CI remains required after the follow-up push. No OpenAI, Supabase, Railway, deployment or production-data operation ran. | -| 2026-07-24 | `codex/supabase-document-change-trigger` | `9c7d9edf509a51478f5bebbabcca64e3926dc877` + reviewed working diff | Document-change ingestion trigger migration, schema mirror, grants, privacy and fail-safe delivery | APPROVE. No P0-P2 finding. The trigger is update-only, acts solely on a strict JSON boolean false/absent-to-true transition, sends only the receiver's allowlisted owner-scoped fields, fails open for document writes when Vault/GUC/pg_net is unavailable, and revokes execution from public/anon/authenticated. No production URL fallback exists. Highest residual risk is deliberate pg_net at-most-once delivery; the clear-then-flip recovery and data-preserving rollback are documented, and the trigger remains inert until both the Vault secret and environment base-URL GUC are configured. | Disposable Supabase Postgres `17.6.1.127` schema replay and drift-manifest regeneration passed (16s; scratch container removed); focused schema/drift/receiver Vitest 89/89; migration-role, function-grant (30 SECURITY DEFINER functions) and owner-scope guards; production-readiness CI mode READY with expected secretless-worktree warnings; offline RAG 21 suites/307 tests; `verify:cheap` 365 files, 3,241 passed/1 skipped; static trace of receiver payload, authoritative owner-scoped reload and idempotent enqueue path. No live provider mutation or migration apply. | +| 2026-07-24 | PR #1109 / `codex/universal-ledger-security-followup-082afe` | `3996b866ca69fa8dfc171fe51f41832eb269006c` + reviewed working diff | Universal task ledger, `/issues` lifecycle and SessionStart surfacing | READY AFTER FIXES. Two P2 lifecycle defects were confirmed and corrected before merge: the SessionStart hook now preserves a non-empty recommended queue when the wider open table is empty, and every active queue package has a durable open-item ID so `/issues done` can archive it without leaving stale recommendations. The ledger provenance was also advanced to the actual merged `origin/main` base. Highest residual risk is human maintenance drift between the queue and open table; the documented mutation contract and invariant check are the current guardrails. | Bash syntax and live SessionStart execution passed (33 recommended, 45 open); queue order is contiguous and unique; every queue source contains only tracked open IDs; `docs:check-links` passed (1,142 references); `docs:check-index` passed (42 modules/routes plus all schema tables); `docs:check-scripts` passed (334 npm script references); `check:skills` passed (32 canonical skills, 8 aliases); Prettier and `git diff --check` passed. No live application, OpenAI, Supabase, Railway, deployment, or production-data action. | +| 2026-07-24 | `codex/supabase-document-change-trigger` | `9c7d9edf509a51478f5bebbabcca64e3926dc877` + reviewed working diff | Document-change ingestion trigger migration, schema mirror, grants, privacy and fail-safe delivery | APPROVE. No P0-P2 finding. The trigger is update-only, acts solely on a strict JSON boolean false/absent-to-true transition, sends only the receiver's allowlisted owner-scoped fields, fails open for document writes when Vault/GUC/pg_net is unavailable, and revokes execution from public/anon/authenticated. No production URL fallback exists. Highest residual risk is deliberate pg_net at-most-once delivery; the clear-then-flip recovery and data-preserving rollback are documented, and the trigger remains inert until both the Vault secret and environment base-URL GUC are configured. | Disposable Supabase Postgres 17.6.1.127 image replay and drift-manifest regeneration passed (16s; scratch container removed); focused schema/drift/receiver Vitest 89/89; migration-role, function-grant (30 SECURITY DEFINER functions) and owner-scope guards; production-readiness CI mode READY with expected secretless-worktree warnings; offline RAG 21 suites/307 tests; `verify:cheap` 365 files, 3,241 passed/1 skipped; static trace of receiver payload, authoritative owner-scoped reload and idempotent enqueue path. No live provider mutation or migration apply. | | 2026-07-23 | PR #1090 / `cursor/fix-phone-dock-edge-1b1d` | `761de7e9ad623b6bd8d634d849a9eb465d622e48` (merged as `09028ef217209fceb53f1122ac7738b509bce323`) | Phone safe-area and edge-to-edge search-dock UI review | MERGED. No P0-P2 finding. The branch was three commits behind, so current `origin/main` was merged before landing; the actual merge tree matched the reviewed synthetic tree. The dock remains flush to the viewport with safe-area padding inside the form, and the phone shell no longer retains the `dvh` clamp that created the Safari toolbar band. Zero actionable review threads. | `npm run ensure`; focused `ui-tools.spec.ts` phone-home and edge-to-edge scenarios: Chromium 2/2 and WebKit 2/2; refreshed hosted policy, security, unit, build, advisory UI, Production UI and required aggregate checks green; exact-head ancestry and local-main tree equality proved after merge. | | 2026-07-22 | PR #1087 / `codex/reconcile-product-truth` | `edbc2260fef59ca2fa7c6973dffb85e32354bce1` (merged as `05dc52fd8408a65117e22a6236e43252203bea92`) | Product-truth copy, account persistence and unavailable-SSO presentation | MERGED. Cross-device claims now match favourites/preferences persistence; recent searches are identified as browser-session data; the contradictory “never shared” statement is removed. All unavailable setup providers and Apple elsewhere use the connected accessible “coming soon” placeholder pattern. The single review finding was fixed, replied to and resolved. | Red DOM proof; focused 19/19; `verify:cheap` 3,220 passed / 1 skipped; `verify:ui` 265/265; PR-local build/secret scan/offline RAG; final hosted required, Production UI, policy and security checks green. No provider calls or RAG spend. | | 2026-07-22 | PR #1086 / `codex/reconcile-xlsx-budgets` | `5376880a40749b6526fd7e4603a7be9d04bc9624` (merged as `2963fba46eacd644618a588fa283f7597faa2644`) | XLSX resource-boundary review | MERGED. Enforces worksheet, non-empty-row, rendered-cell and UTF-8 output ceilings before result fragments are appended; sparse-column output is preserved. No actionable review threads. | Red 257-sheet reproducer; focused 4/4; `verify:cheap` 3,218 passed / 1 skipped; PR-local build/scan/offline RAG; hosted required/security/policy green. | diff --git a/docs/maturity-backlog-workorders.md b/docs/maturity-backlog-workorders.md index 2c3a945c9..3d207ed7f 100644 --- a/docs/maturity-backlog-workorders.md +++ b/docs/maturity-backlog-workorders.md @@ -7,6 +7,10 @@ files**, **risk**, **verification**, and **status**. High-risk items are deliber their own work order — the audit's rule is one dedicated PR + full-suite verification per structural change, not a single mixed PR. +This file is supporting work-order detail, not a second task ledger. Only work represented in the +recommended queue in [`outstanding-issues.md`](outstanding-issues.md) is an active repository +recommendation; that universal ledger owns current status, priority, and execution order. + **Status legend:** `DONE` (landed) · `IN PROGRESS` (partially landed; more PRs remain) · `READY` (scoped, safe to start) · `OPEN` (needs a decision or a dedicated PR) · `PROVIDER-GATED` (touches live DB/CI/provider — needs explicit confirmation) · `SATISFIED` diff --git a/docs/operator-backlog.md b/docs/operator-backlog.md index 31d9ec96a..01b641776 100644 --- a/docs/operator-backlog.md +++ b/docs/operator-backlog.md @@ -1,9 +1,9 @@ # Operator action detail Detailed runbook index for **human-only / provider-gated actions** that cannot be done from a coding -session. Canonical task status, order, acuity, and completion live only in -[`outstanding-issues.md`](outstanding-issues.md); this file supplies provider-specific steps and -presence claims that must be verified before acting. +session (they touch Supabase, Railway, OpenAI, or GitHub settings, per the AGENTS.md provider boundary). +Canonical task status, priority, and execution order live only in +[`outstanding-issues.md`](outstanding-issues.md); this file supplies operator procedures and evidence. **How to use:** work top to bottom; each row links to the detailed runbook. `Status` values are `⏳ pending`, `🔎 verify` (may already be done — confirm before repeating), `✅ done`, `—` (n/a). diff --git a/docs/outstanding-issues.md b/docs/outstanding-issues.md index 5ba66d462..4977d723b 100644 --- a/docs/outstanding-issues.md +++ b/docs/outstanding-issues.md @@ -7,13 +7,14 @@ resolution history. Chat context is ephemeral; this file is the single universal **Rule of thumb:** if it is worth remembering after this session ends, it belongs here. -Detailed runbooks may live elsewhere, including [`operator-backlog.md`](operator-backlog.md), but -task status, priority, order, dependencies, and completion state are canonical only here. +This is the repository's **single universal task ledger**. It owns recommended execution order, +acuity, required capability, timing, effort, dependencies, approvals, evidence, status, success +criteria, stop rules, and resolution history. Do not create or maintain a second task ledger. ## How this is used -- Say `/issues` in Claude Code → the skill reads this file and states the open items back, - grouped by priority with a one-line summary count. Nothing is mutated on a plain read. +- Say `/issues` in Claude Code → the skill reads this file and states the recommended queue in + order, then summarizes the wider open-item counts. Nothing is mutated on a plain read. - `/issues add …`, `/issues done `, `/issues capture`, and friends mutate the tables below. The full command surface lives in the skill file. - Every mutation keeps this file committed so the memory survives across sessions and worktrees. @@ -30,54 +31,72 @@ task status, priority, order, dependencies, and completion state are canonical o - Resolving an item moves its row to **Resolved / archive** with the date and a one-line outcome — rows are archived, not deleted, so the history stays auditable. +## Execution scales + +- **A1 urgent:** active safety/privacy/data-loss/release blocker; `#057` and `#053` are the current A1 items. +- **A2 important:** confirmed correctness, clinical, privacy, reliability, or evaluation-integrity + work that leads its available lane. +- **A3 planned:** worthwhile work deferred to its stated trigger. +- **Optional:** do only when measured need, ownership, and cost justify it. +- **Standard:** experienced generalist. **High:** senior cross-module reasoning. **Specialist:** + database/RAG/clinical/privacy/security/evaluation expertise. **Operator:** authorised human owner. +- Effort is active work, excluding approvals, hosted/provider waits, soak, and review time. + +The order below is planning guidance, not authority to call providers, spend money, change +production, commit, push, merge, or deploy. A waiting dependency does not block an independent +executable item below it. + ## Recommended execution queue -This queue contains only work that remains recommended. Revalidate a row against current `main` -before starting. Its position is not authority to call providers, spend money, change production, -commit, push, open a PR, or deploy. - -Order numbers remain stable within a reconciliation snapshot. A gap means an item was completed or -removed after current-main verification; it is not missing recommended work. - -- **A1 — urgent:** active safety, privacy, data-loss, or release/launch blocker. One operator/legal - item is retained; there is no current A1 code defect. -- **A2 — important:** confirmed correctness, clinical, privacy, reliability, or evaluation work. -- **A3 — planned:** worthwhile work that is safe to defer until its trigger. -- **Optional:** start only when measured need, ownership, and cost justify it. -- **Capability:** Standard = established pattern; High = cross-module senior work; Specialist = - database/RAG/clinical/privacy expertise; Operator = named provider/product/legal authority. -- **Estimate:** focused active time, excluding approval, hosted runtime, soak, and review waits. - -| Order | ID(s) | Acuity | Capability | When | Estimate | Outcome, gate, verification, and stopping condition | -| ----: | ---------------------- | -------- | ------------------------------------------- | ------------------------------------------------------------------ | ---------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| 1 | `#053` | A1 | Operator — legal/privacy | Start now; finish before real patient use/privacy-approved release | 4–8 hours internal; 1–6 weeks elapsed | Execute DPAs; decide ZDR/residency; obtain cache behavior in writing; review subprocessors; obtain APP 8 and APP 5/1 counsel sign-off. Do not change public copy before approval. | -| 2 | `#052` | A2 | High — ingestion concurrency | Ready; first code task | 0.5–1 day | Add red single/bulk tests, then block full/retry reindex during a fresh agent-enrichment lease while keeping stale leases and enrichment mode unchanged. Run focused safety tests and `verify:cheap`; stop if current `main` cannot reproduce it. | -| 3 | `#030` | A2 | High — evaluation semantics | After order 2 | 2–4 hours | Require distinct source identities for distinct comparison slots. Run focused matching tests, typecheck, and `verify:cheap`; stop without changing strict aliases, retrieval, or ranking. | -| 4 | `#019` | A2 | Specialist — RAG answer pipeline | Local reproducer now; behavior change after order 8 | 0.5–1 day reproducer | Reproduce admission-source loss in the fallback layer using PR #1096’s source shape. Any behavior change needs protected review and an approved baseline/post canary; stop if independently non-reproducible. | -| 6 | `#054` | A2 | Standard locally; Operator hosted | Local safety identifier now; hosted next approved window | 15–30 min local; 1–2 hours hosted | Presence-check and fill confirmed secret/config gaps with distinct per-environment values. Never record values. Require clean readiness/secret checks; stop on ambiguous environment or project identity. | -| 7 | `#022` | A2 | Operator — clinical governance + Specialist | Decision-ready | 1–2 hours policy; 0.5–1 day first ten | Decide BMJ attestation policy and review the ten highest-impact local documents. Record reviewer/evidence/time; stop after ten and remeasure warning debt. | -| 8 | `#051`, `#023` | A2 | Specialist — RAG diagnostics | After scheduled 2026-07-26 run | 2–4 hours | With GitHub-read approval, compare structured canary/browser/irrelevant-at-10 artifacts without dispatching a rerun. Record deterministic/provider/latency deltas and disposition residuals; stop without spending. | -| 9 | `#018` | A2 | Specialist — clinical RAG/retrieval | After order 8, one mechanism at a time | 1–2 days diagnosis | Give lithium, ADHD, and metabolic residuals separate current-main reproducers and candidates. Behavior canaries require approval; stop any item without a deterministic reproducer or on regression. | -| 10 | `#029` | A2 | Specialist — answer quality/clinical safety | After orders 8–9 | 0.5–1 day inventory; 1–3 days per fix | Re-enumerate current fallback stubs and fix one causal cluster at a time without weakening grounding/citation gates. Stop if a change merely makes the metric easier to pass. | -| 11 | `#001` | A2 | Specialist — retrieval/ranking | After order 8 and rollout approval | 0.5–1 day plus canary | Keep semantic reranking off unless an approved ambiguity comparison preserves 36/36, recall 1.0, zero per-case regressions, and shows measured gain; otherwise record keep-off and stop. | -| 12 | `#025` | A2 | Operator — Railway/GitHub/chat/Supabase | Next approved observability window | 1–3 hours/channel | Choose owned deployment, CI, ingestion, and SLO alerts; mock first, then one approved controlled provider event/channel. The merged Supabase trigger remains inert until its verified inputs are configured. Stop without an accountable responder. | -| 13 | `#055` | A2 | Specialist release owner + Operator | Before next full-confidence release/handoff | 2–4 hours plus runtime | On one exact SHA, run local/provider gates, Firefox/WebKit, required hosted CI, and close actionable GitHub threads. Stop at first failure and rerun only the repaired smallest gate. | -| 14 | `#056` | A2 | Operator — Supabase/Railway + Specialist | After cost/ownership approval | 0.5–1 day | Provision isolated `Clinical KB Staging` with synthetic data and distinct secrets. Verify identity, schema, indexing, health, and data boundary; never copy production clinical documents. | -| 15 | `#057` | A2 | High — release/SRE + Operator | After order 14 | 2–4 hours plus soak | Run documented staging soak and rollback against an exact candidate. Retain latency/error/rollback evidence; stop on unsafe data, identity mismatch, or unowned rollback. | -| 16 | `#058` | A2 | Operator — production data + Specialist | Next approved production verification window | 30–60 min read-only; 1–2 hours if needed | Verify registry/differentials/medications are non-empty before writing; seed only confirmed gaps idempotently. Stop when healthy or owner/project identity is ambiguous. | -| 17 | `#007` | A3 | Operator decision + Standard frontend | When product chooses canonical Tools route | 15–30 min decision; 0.5 day | Align navigation, redirect, sitemap, and reachability around one entry point. Stop while the standalone page has an unresolved requirement. | -| 18 | `#011` | A3 | Operator — Supabase capacity | Immediately before first compute scale-up | 30–60 min plus observation | Switch Auth to percentage allocation, record before/after, and run approved advisor/health checks. Stop if no scale-up is planned. | -| 19 | `#017` | A3 | High — performance/browser | Before order 23; approved live-site window | 1–2 hours | Capture reproducible mobile/desktop Lighthouse/Web-Vitals evidence and decide whether payload work is justified. Stop if metrics are acceptable or evidence is too noisy. | -| 20 | `#024` | A3 | High — Next.js/Playwright/WebKit | After order 8 or real Safari reproduction | 0.5–1 day | Distinguish test interception from a Safari defect. Apply test-only correction only with a discriminating repro; keep access-control assertions meaningful. | -| 21 | `#033` | A3 | Specialist — prompt/source governance | After orders 7–8 | 1–2 days plus approved eval | Design unknown-vs-adverse metadata wording and prompt tests. Require no supported-grounding drop and zero citation failures; stop on broad over-caveating or degradation. | -| 22 | `#037` | A3 | Operator — clinical/product + Standard | Next trust-policy review | 30–60 min; up to 0.5 day | Decide whether routine claims cap at medium trust. Record policy; if accepted, change only the flag/expectations and run focused tests. | -| 23 | `#012`, `#013`, `#016` | A3 | High — bundling/runtime performance | After order 19 or equivalent evidence | 0.5–2 days/route | Optimize only a production route with measured payload/render/motion harm. Require material gain plus focused, `verify:cheap`, and browser evidence; stop on small gain. | -| 24 | `#027` | Optional | Operator — SRE/provider | When an owned external alert path is wanted | 1–2 hours | Decide vendor/cost/privacy/owner; if accepted, prove one non-PHI outage and recovery alert. Stop when no responder owns it. | -| 25 | `#028` | Optional | Specialist privacy/observability + Operator | After privacy/ownership/cost approval | 1–3 days | Define vendor/region/retention/redaction/sampling/source-map envelope before SDK work. Prove no clinical text, identifiers, or secrets leave; stop if unacceptable. | -| 26 | `#038` | Optional | High — product/design architecture | When a new comparison surface is approved | 0.5–1 day | Define a shared interaction contract without flattening mode-specific content. Stop when no concrete new surface exists. | -| 27 | `#040` | Optional | High — visual QA/accessibility | When baseline owner/update workflow exist | 1–2 days | Establish a small stable desktop/mobile/accessibility baseline set. Do not make it blocking if flake or maintenance cost outweighs detection value. | - - +Last reconciled on **2026-07-24** against fetched `origin/main` +`527988c2ccabc98b4d0673d33360c971df65fa0e`. No live application or provider state was queried. +Revalidate the referenced evidence against current `main` immediately before starting a row. + +| Order | Source | Recommended outcome / next action | Acuity / capability | When | Active effort | Dependencies / approval | Done, verification, and stop rule | +| ----: | --------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -------------------------------------------------- | ----------------------------------------------------------- | -------------------------------------- | -------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| 1 | `#057` | Verify provider-side containment of credentials previously exposed outside authorised stores; revoke or rotate any still-valid OpenAI, Supabase service-role/database, and E2E credentials, then update only intended secret stores. | A1 / Operator security + independent reviewer | Immediate approved security window | 1–3 hours | Credentialed provider owners; correct target identity; explicit provider and secret-store approval | Every old credential is rejected or retired; replacements exist only in intended stores; presence/readiness checks pass; no value enters Git, logs, issues, or chat. Use approved provider evidence and secret scans. Stop before provider action without approval and do not rewrite history without separate evidence. | +| 2 | `#053` | Align the Safety Plan Generator with the repository's no-patient-data contract before real-patient use. | A1 / Specialist privacy + clinical | Ready; first | 3–5 hours | Privacy/clinical owner | Copy, behavior, print, and tests agree; no identifier is persisted/transmitted without an approved basis. Run focused UI/privacy/a11y checks, `verify:cheap`, production-readiness. Stop before adding storage/provider flow. | +| 3 | `#054` | Fail closed when answer relevance metadata is absent; add the red policy test, then make the smallest render-policy correction. | A2 / Specialist clinical safety | Ready; after `#053` | 2–4 hours | Local/offline | Missing and explicit-false relevance are conservative; explicit source-backed behavior is unchanged. Run focused policy/provenance tests, `verify:cheap`, production-readiness. Stop before retrieval/ranking/generation. | +| 4 | `#052` | Prevent full/retry reindex overlapping a fresh agent-enrichment lease; add red single/bulk tests, then reuse `hasActiveAgentEnrichmentJob` before mutation. | A2 / High concurrency | Ready; after `#054` | 0.5–1 day | Local/offline | Fresh processing leases conflict; stale leases and enrichment mode retain behavior. Run focused route/safety tests, `verify:cheap`, production-readiness. Stop if current `main` does not reproduce it. | +| 5 | `#055` | Recover aged `queued` documents that have no open ingestion job; begin with a stranded-row reproducer and the smallest idempotent owner-scoped recovery path. | A2 / Specialist queue reliability | Ready; after `#052` | 0.5–1.5 days | Local replay for schema/scheduling; hosted changes need approval | Exactly one recoverable job is created under crash/concurrency/repeat; fresh queues and open jobs are untouched. Run focused recovery/schema tests, disposable replay if needed, `verify:cheap`, production-readiness. Stop if safe age/ownership cannot be proven. | +| 6 | `#030` | Require distinct source identities for distinct expected comparison slots; begin with a red shared-title/two-expectation test. | A2 / High evaluation semantics | Decision-ready; code only after owner choice | 2–4 hours | Local/offline; protected-eval review | One source cannot satisfy both slots; existing cases stay green. Run focused matching, typecheck, `verify:cheap`. Stop before alias/retrieval/ranking changes. | +| 7 | `#019` | Prove admission evidence is lost at the comparison fallback boundary; scope the smallest correction without automatically shipping it. | A2 / Specialist RAG | Reproducer now; behavior after `#051`/`#023` | 0.5–1 day; fix separate | Protected review; approved baseline/post canary for behavior | Reproducer fails while packing retains both sources. Stop without behavior change if the boundary defect is not independently reproducible. | +| 8 | `#058` | Execute OpenAI/Railway DPAs; decide ZDR/residency; obtain cache/subprocessor answers and APP 8 plus APP 5/1 counsel sign-off. | A2 / Operator legal/privacy | Start now; before real-patient use/privacy-approved release | 4–8 hours internal; 1–6 weeks elapsed | Signers, counsel, providers | Record countersigned evidence, IDs, dates, decisions, and approved wording in PIA/cross-border docs. Stop before public-copy changes without counsel. | +| 9 | `#059` | Set a unique 32+ character `OPENAI_SAFETY_IDENTIFIER_SECRET` per environment with `OPENAI_API_KEY`, never in Git/output. | A2 / Standard local + Operator hosted | Local now; hosted approved window | 15–30 min local; 30–60 min/env | Secret manager; hosted approval | Presence-only readiness passes, HMAC pseudonymity remains, secret scans clean. Stop if governance rejects stable identifiers. | +| 10 | `#060` | Reconcile query-hash, deep-probe, Supabase/OpenAI, project identity, and schedules using read-only presence checks before setting anything. | A2 / Operator platform | Next approved readiness window | 1–2 hours | Provider approval, env owners | Record status, never values; set only confirmed gaps; identity/readiness pass. Stop on target ambiguity or cross-env key reuse. | +| 11 | `#022` | Decide auditable BMJ reference policy, then review the ten highest-impact local WA documents. | A2 / Operator governance + Specialist | Decision-ready | 1–2 hours policy; 0.5–1 day reviews | Clinical authority; live writes/canary approval | Preserve third-party/unverified provenance; record reviewer/evidence/time/rationale. Stop after ten and remeasure debt. | +| 12 | `#051` + `#023` | Compare run `30018289898` with the scheduled 2026-07-26 canary/browser/label artifacts; disposition residuals without rerun. | A2 / Specialist evaluation | About 2026-07-27 02:00 AWST plus runtime | 2–4 hours | Approval to read hosted artifacts; no dispatch | Record tree/run, content/provider/latency, browser, and labeling outcomes. Stop without spend or archived lithium changes. | +| 13 | `#018` | Diagnose lithium, ADHD, metabolic residuals independently: table fast path, extractive budget, schedule-free selection. | A2 / Specialist clinical RAG | After `#051`/`#023`; one mechanism at a time | 1–2 days diagnosis; fixes separate | Protected review; approved behavior canary | Each has a current-main fixture and scoped candidate; stop without deterministic repro or on individual canary regression. | +| 14 | `#001` | Reassess semantic reranking without changing default; enable only after accepted ambiguity comparison. | A2 / Specialist ranking | After `#051`/`#023` and rollout approval | 0.5–1 day plus canary review | Provider approval/budget | Keep off unless 36/36, recalls 1.0, zero case regressions, measured gain. Otherwise record keep-off decision. | +| 15 | `#033` | Decide whether governance metadata enters the LLM source block, distinguishing unknown from adverse status. | A3 / Specialist prompt governance | After `#022` and `#051`/`#023` | 1–2 days plus eval | Better metadata, stable canary, provider approval | Prompt/serialization tests; no grounded-supported drop; zero citation failures. Stop on over-caveating/degradation. | +| 16 | `#024` | Separate Playwright `_rsc` interception failure from a real Safari defect. | A3 / High Next.js/WebKit | After scheduled matrix data or real Safari repro | 0.5–1 day | Device/provider check only if local ambiguity | Test-only fix needs discriminating repro and meaningful access-control assertions; keep Chromium green. | +| 17 | `#007` | Choose canonical Tools experience; align nav, redirects, sitemap, reachability. | A3 / Operator product + Standard frontend | Product decision window | 15–30 min decision; 0.5 day code | Product owner | One canonical path and intentional redirect with green route/UI tests. Stop on unresolved standalone need. | +| 18 | `#025` + `#061` | Select owned deploy/CI/ingestion/SLO channels; configure only those, ingestion after `#055`. | A2 / Operator integrations | Approved observability window | 1–3 hours/channel | Provider approval, secrets, destination owner | Mocked tests then one controlled non-PHI event/channel. Stop on missing owner or unsafe delivery. | +| 19 | `#062` | Run release/clinical gates once against an exact candidate, not a moving branch. | A2 / High release + Operator | Before release/handoff needing full confidence | 0.5 day plus waits | Exact SHA, provider approval, heavy-command lock | All required gates recorded against one SHA; stop/classify first failure and do not repeat unchanged pass. | +| 20 | `#063` | Provision dedicated Clinical KB staging on Supabase/Railway with isolated keys and synthetic/non-clinical data. | A2 / Operator + Specialist DB | After cost/ownership approval | 0.5–1 day | Billable provider approval, owner | Identity, tenancy, secrets, migrations, app/worker health, data boundary pass. Stop on target ambiguity or production data/key reuse. | +| 21 | `#064` | Run documented soak and rollback rehearsal in dedicated staging. | A2 / Operator reliability | After `#063` | 0.5 day plus soak | Staging and provider approval | SLO, rollback, data integrity, recovery evidence pass. Stop before production if unproven. | +| 22 | `#065` | Verify registry/differentials/medications non-empty before writes; seed only confirmed governed gaps. | A2 / Operator data + Specialist | Approved production window | 1–3 hours plus seed time | Correct owner/project, provider approval | Read-only proof precedes idempotent writes; stop if populated or target ambiguous. | +| 23 | `#011` | Switch Auth DB cap to percentage immediately before first compute scale-up. | A3 / Operator capacity | Scale-up trigger only | 30–60 min plus soak | Dashboard, planned scale-up, approval | Target only `sjrfecxgysukkwxsowpy`; record before/after and advisor/health recheck. Do not create/use staging for this task. | +| 24 | `#017` | Capture reproducible mobile/desktop production LCP, INP, CLS before performance work. | A3 / High web performance | Before `#012`/`#013`; approved live window | 1–2 hours | Live-site approval; record route/build/throttle | Record metrics and accept/reject decision. Stop if acceptable or too noisy. | +| 25 | `#037` | Decide whether routine supported claims cap at medium trust. | A3 / Operator clinical-product + Standard frontend | Next trust-policy review | 30–60 min decision; up to 0.5 day code | Clinical/product authority | Record policy; if accepted, change flag/render expectations only and run focused tests. Stop without owner acceptance. | +| 26 | `#027` | Add uptime monitor outside GitHub/Railway only if service, budget, responder justified. | Optional / Operator SRE | Owned external-alert trigger | 1–2 hours | Vendor/privacy/owner/provider approval | Controlled non-PHI failure/recovery alerts. Stop without responder or if current monitoring accepted. | +| 27 | `#028` | Define vendor/region/retention/redaction/sampling/maps/owner before runtime error tracking. | Optional / Specialist privacy + Operator | After privacy/ownership/cost approval | 1–3 days | Privacy and provider approval | Redaction tests exclude clinical data/IDs/secrets; one non-sensitive event arrives. Stop if envelope unacceptable. | +| 28 | `#012` + `#013` | Re-measure route payloads and slim only one production route with a demonstrated budget problem; ignore mockup-only code unless it enters production. | A3 / High Next.js performance | After `#017` or equivalent evidence; one route | 0.5–2 days/route | Measured target, Next.js guide, UI verification | Demonstrate a material parsed/gzip or interaction gain with unchanged behavior; run analysis, focused tests, `verify:cheap`, browser smoke. Stop if gain is small. | +| 29 | `#040` | Add small stable visual-regression baselines with owner/update workflow. | Optional / High visual QA | Stable surfaces and owner | 1–2 days | Stable browser, owner | Low-flake repeat runs and documented updates. Stop before blocking if churn high. | +| 30 | `#038` | Define shared comparison interaction contract before another comparison surface; keep clinical content mode-specific. | Optional / High design-system | Approved new comparison surface | 0.5–1 day | Concrete surface, product/design owner | Inventory patterns and make shared behavior testable. Stop if no new surface. | +| 31 | `#035` | Expand conflict detection only for concrete clinically reviewed class with positive/negative fixtures. | A3 / Specialist evidence rules | Demonstrated missed conflict | 0.5–1 day design; code separate | Clinical review; provider approval only for live validation | Fixtures discriminate class without unrelated warnings. Stop if no bounded class. | +| 32 | `#039` | Converge repeated catalogue toolbar behavior only during a concrete toolbar project. | Optional / High frontend architecture | Concrete project | 0.5–1 day inventory; 1–3 days code | Product/design owner, UI verification | Prove shared behavior without flattening search semantics; stop after bounded contract. | +| 33 | `#056` | Write a product/privacy/persistence brief for “Current Clinical Work” before any storage or UI implementation. | A3 / High product architecture + privacy | Only when the product owner wants to evaluate the feature | 0.5–1 day | Product owner, privacy/retention decision, evidence of demand | Define users, data classes, lifecycle, cross-device expectations, deletion, failure states, and a smallest testable slice. Stop if demand or safe persistence cannot be established. | + +### Queue maintenance + +1. Revalidate against current `main` before starting and remove/rewrite contradicted work. +2. Do not combine protected RAG residuals or change scores, comparators, aliases, clamps, or semantic + reranking without separate reproducers and required validation. +3. Treat provider status as a claim: verify only after approval and never store secret values. +4. Close/reclassify when a success or stop condition is met; do not preserve work for its own sake. + + ## Open items @@ -87,15 +106,22 @@ removed after current-main verification; it is not missing recommended work. | ID | Pri | Type | Summary | Detail / next action | Source | Added | | ---- | --- | ----- | --------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------- | ---------- | +| #057 | P1 | task | Verify containment of previously exposed credentials | **Outcome:** credentials previously exposed outside authorised stores can no longer authenticate. **Next:** in approved provider dashboards, revoke or rotate any still-valid OpenAI, Supabase service-role/database, and E2E credentials; update only intended secret stores and record status without values. **Success:** provider evidence shows every old credential rejected or retired, replacements are scoped to intended environments, and readiness checks pass. **Verify:** approved provider audit/rotation evidence, presence-only checks, and secret scanning that never prints values. **Stop:** no provider or secret-store action without explicit approval; never paste values into Git, logs, issues, or chat, and do not rewrite history without separate evidence. | session 2026-07-24 security reconciliation | 2026-07-24 | | #051 | P2 | task | Stabilise the live answer-quality canary before more RAG tuning | Diagnostics landed in PR #1095: structured JSON/Markdown artifacts now record the actual checked-out SHA, run identity and latency context, and the offline trend tool separates content, provider-route and latency outcomes. First validating run `30018289898` recorded the expected tree and cost, with 36/36 retrieval green, but one report cannot establish variability; PR #1097 prevents a single failure being mislabeled as repeated. Next: compare the scheduled 2026-07-26 structured report with this run. Do not spend on an immediate retry or reapply the archived lithium guard before that comparison. | PR #1095; run `30018289898`; PR #1097; archive ref `refs/archive/rejected-rag/20260723/monitoring-subject-gate` | 2026-07-23 | | #001 | P2 | task | Semantic reranking still gated off | `RAG_SEMANTIC_RERANK_ENABLED=false` from PR #901. Do not enable until the provider-backed 36/36 retrieval-quality gate **and** an ambiguity-focused canary are explicitly approved and recorded. | `docs/process-hardening.md` (Semantic reranking rollout debt); PR #901 | 2026-07-21 | -| #052 | P2 | issue | Reindex can overlap a fresh agent-enrichment pass | Full/retry single and bulk routes consult `ingestion_jobs` but not the implemented `hasActiveAgentEnrichmentJob` predicate. Add red route tests, then block fresh `indexing_v3_agent_jobs.status='processing'` leases before mutation while keeping stale leases and enrichment mode unchanged. | `src/lib/ingestion-mutation-safety.ts`; single/bulk reindex routes; repository audit 2026-07-24 | 2026-07-24 | -| #053 | P1 | task | Execute cross-border privacy/legal package | Execute OpenAI and Railway DPAs; decide ZDR and Australian data residency; obtain prompt-cache behavior in writing; review subprocessors; obtain APP 8 and APP 5/1 counsel sign-off. Do not represent the release as privacy-approved or alter final public privacy wording before sign-off. | `docs/openai-cross-border-basis.md`; `docs/privacy-impact-assessment.md` | 2026-07-24 | -| #054 | P2 | task | Reconcile local and hosted secrets/config | Presence-check safety-identifier, query-hash, deep-probe, Supabase service-role, OpenAI, project-identity, and schedule settings; set only confirmed gaps with distinct per-environment values. Never record secret values. Provider reads/writes require approval. | `.env.example`; production-readiness warning; `docs/operator-backlog.md` | 2026-07-24 | -| #055 | P2 | task | Run one exact-SHA full release and PR gate | Before the next full-confidence release/handoff, record the candidate/PR SHA and run the local/provider release gates, Firefox/WebKit, required hosted CI, and actionable GitHub review-thread closure once. Stop at the first actionable failure and rerun only the repaired smallest gate. | `docs/launch-operator-runbook.md`; `docs/codex-review-protocol.md` | 2026-07-24 | -| #056 | P2 | task | Provision isolated staging environment | After explicit cost/ownership approval, provision `Clinical KB Staging` Supabase and Railway tiers with distinct secrets and synthetic/non-clinical data. Verify identity, schema, indexing, health, and the production-data boundary. | `docs/staging-setup.md`; `docs/operator-backlog.md` | 2026-07-24 | -| #057 | P2 | task | Complete staging soak and rollback rehearsal | After #056, run the documented soak and rollback against an exact candidate; retain latency/error/rollback evidence. Stop on unsafe data, identity mismatch, or an unowned rollback decision. | `docs/launch-operator-runbook.md`; `docs/capacity-review.md` | 2026-07-24 | -| #058 | P2 | task | Verify production content before any seed write | Against `Clinical KB Database`, verify registry, differentials, and medications surfaces are non-empty before writing. Seed only confirmed gaps idempotently with approved owner/project identity and confirmation flags. | `docs/launch-operator-runbook.md`; `docs/operator-backlog.md` | 2026-07-24 | +| #052 | P2 | issue | Reindex can overlap a fresh agent-enrichment pass | Full/retry reindex preflights check `ingestion_jobs` but do not call the existing `hasActiveAgentEnrichmentJob`. A fresh `indexing_v3_agent_jobs.status='processing'` lease can therefore overlap destructive artifact work. Extend single and bulk full/retry preflights before mutation; keep stale leases non-blocking and preserve enrichment-mode behavior. | `src/lib/ingestion-mutation-safety.ts:116-159`; single/bulk reindex routes; ingestion audit 2026-07-24 | 2026-07-24 | +| #053 | P1 | issue | Safety Plan Generator contradicts the privacy contract | **Outcome:** the tool, privacy notice, PIA, and tests agree on whether patient identifiers may be entered, copied, printed, or saved. `patient-safety-plan.tsx` asks for “Patient (name or initials)” and produces a patient copy, while `/privacy` and the PIA say the product does not ask for patient data. **Next:** obtain a privacy/clinical decision, then default to identifier-free behavior unless transient identifier processing is explicitly approved and documented. **Success:** no contradictory copy; no identifier is persisted or transmitted without an approved basis. **Verify:** focused component/privacy/copy/accessibility tests, browser print/copy smoke, `verify:cheap`, production-readiness. **Stop:** any new storage or provider transmission requires a separate review. | `src/components/patient-safety-plan.tsx:649`; `src/app/privacy/page.tsx:28`; `docs/privacy-impact-assessment.md:18` | 2026-07-24 | +| #054 | P2 | issue | Missing answer relevance metadata is treated as source-backed | **Outcome:** absent `relevance` metadata renders conservatively. `RagAnswer.relevance` is optional, but `relevance?.isSourceBacked !== false` treats `undefined` as source-backed. **Next:** add the red render-policy test, then make the smallest policy-only fix. **Success:** missing and explicit-false relevance fail closed; explicit source-backed relevance is unchanged. **Verify:** focused answer-render-policy, provenance, clinical-safety, `verify:cheap`, production-readiness. **Stop:** do not expand into retrieval, ranking, or generation. | `src/lib/types.ts:1029`; `src/lib/answer-render-policy.ts:145` | 2026-07-24 | +| #055 | P2 | issue | Upload crash can strand a queued document without a job | **Outcome:** a crash between document and job creation cannot strand an upload indefinitely. **Next:** add the stranded-row reproducer, then choose the smallest idempotent atomic-enqueue RPC or bounded scheduled sweep consistent with current ownership and rollback contracts. **Success:** exactly one recoverable job is created; existing open jobs do not duplicate; owner scope, retry, audit, and rollback remain intact. **Verify:** focused upload/recovery/schema tests, migration guards, disposable replay if needed, `verify:cheap`, production-readiness. **Stop:** hosted changes require approval, and no at-least-once claim is valid until the crash case passes. | `src/app/api/upload/route.ts`; `docs/webhooks.md:163-192` | 2026-07-24 | +| #058 | P2 | task | Complete privacy and cross-border legal package | Execute the OpenAI and Railway DPAs; decide ZDR and residency; obtain written prompt-cache/subprocessor answers; and obtain APP 8 plus APP 5/1 counsel sign-off. Record evidence IDs, dates, decisions, and only counsel-approved public wording. Stop before provider outreach, signatures, account changes, or public-copy edits without the applicable owner approval. | `docs/openai-cross-border-basis.md`; `docs/privacy-impact-assessment.md` | 2026-07-24 | +| #059 | P2 | task | Configure privacy-preserving OpenAI safety identifiers | Set a unique 32+ character `OPENAI_SAFETY_IDENTIFIER_SECRET` in each approved environment that has `OPENAI_API_KEY`, using only the intended secret manager. Verify presence-only production readiness, HMAC pseudonymity, and clean secret scans. Stop on cross-environment reuse, target ambiguity, or a governance decision against stable identifiers. | `.env.example`; `src/lib/env.ts`; `scripts/production-readiness.ts` | 2026-07-24 | +| #060 | P2 | task | Reconcile operator configuration backlog | In the next approved readiness window, use read-only presence checks to reconcile query-hash, deep-probe, Supabase/OpenAI, project-identity, and schedule configuration before setting anything. Record status, never values; set only confirmed gaps. Stop on target ambiguity or cross-environment key reuse. | `docs/operator-backlog.md`; `docs/launch-operator-runbook.md` | 2026-07-24 | +| #061 | P2 | task | Select and verify owned SLO alert channels | Choose owned deploy, CI, ingestion, and SLO alert destinations, then configure only those channels; defer ingestion alerts until `#055` is resolved. Run mocked checks first and one controlled non-PHI event per approved channel. Stop without a destination owner or if safe delivery cannot be demonstrated. | `docs/operator-backlog.md`; `docs/webhooks.md`; `docs/capacity-review.md` | 2026-07-24 | +| #062 | P2 | task | Run the full release gate against an exact candidate | Run the release and clinical gates once against one immutable candidate SHA before a release or handoff that requires full confidence. Record every gate against that SHA, classify the first failure, and do not repeat an unchanged pass. Provider-backed portions and the heavy-command lock require their normal approvals. | `docs/production-readiness-checklist.md`; `.agents/skills/release/SKILL.md` | 2026-07-24 | +| #063 | P2 | task | Provision an isolated Clinical KB staging environment | After billable provider cost and ownership approval, provision dedicated Supabase/Railway staging with isolated keys and synthetic or non-clinical data. Verify target identity, tenancy, secrets, migrations, app/worker health, and the data boundary. Stop on target ambiguity or any production data/key reuse. | `docs/staging-setup.md`; `docs/deployment-architecture.md` | 2026-07-24 | +| #064 | P2 | task | Rehearse staging soak and rollback | After `#063`, run the documented soak and rollback rehearsal in dedicated staging. Record SLO, rollback, data-integrity, and recovery evidence. Stop before production if any outcome is unproven. | `docs/launch-operator-runbook.md`; `docs/capacity-review.md` | 2026-07-24 | +| #065 | P2 | task | Verify governed production seed coverage | In an approved production window, first prove read-only that registry, differentials, and medications data are non-empty; seed only confirmed governed gaps with idempotent procedures. Stop if the corpus is already populated or the owner/project target is ambiguous. | `docs/launch-operator-runbook.md`; `docs/operator-backlog.md` | 2026-07-24 | +| #056 | P3 | rec | Define “Current Clinical Work” before implementation | **Outcome:** decide whether a workspace combining saved comparisons, partial formulation work, recent tools, and pinned source sets is worth building. **Next:** write a product/privacy/persistence brief only; do not build storage or UI. **Success:** the brief defines users, data classes, lifecycle, cross-device expectations, deletion, failure states, evidence of demand, and the smallest testable slice. **Stop:** close the idea if demand or safe persistence cannot be established. | session 2026-07-24; `src/lib/tools-catalog.ts:253-262` | 2026-07-24 | | #005 | P3 | rec | `finalScore` saturates at clamp ceiling | Base + ~40 stacked boosts routinely exceed 1.0, so strong matches tie at 1.0 and order by an arbitrary `document_id` tiebreak. If ranking is ever revisited, break ties by the **pre-clamp** score rather than raising the `[0,1]` ceiling (downstream gates assume `[0,1]`). Ordering already sorts by the unbounded pre-clamp `rankScore` (`clinical-search.ts:1735,1927,1950-1955`), so the clamp confines only the reported confidence value, not result order. Not a defect on the current golden set; any change here is a protected RAG surface (canary required). | `docs/rag-hybrid-findings-and-todo.md` P1 item 4; `src/lib/clinical-search.ts:1735` | 2026-07-21 | | #007 | P3 | rec | `/tools` vs `/?mode=tools` parallel Tools entry points | `/tools` (standalone `ApplicationsLauncherPage`) has no inbound in-app link; the sidebar Tools item uses `/?mode=tools`. Decide the canonical entry point and wire nav consistently, or drop the standalone `/tools` page + `/applications` redirect. Currently allowlisted in `tests/route-reachability.test.ts`. | `src/app/tools/page.tsx`; `src/app/applications/route.ts` | 2026-07-21 | | #009 | P3 | rec | Confirm `/api/jobs` is intentionally server/ops-only | No client `fetch()` reaches `/api/jobs` (only tests import it). Confirm it is a deliberate ops/manual surface; if abandoned, remove it. | `src/app/api/jobs/route.ts` | 2026-07-21 | @@ -103,7 +129,6 @@ removed after current-main verification; it is not missing recommended work. | #011 | P3 | task | Auth DB-connection allocation is operator-only | Supabase Auth (GoTrue) is capped at ~10 absolute DB connections (Supabase perf advisor). Switch to **percentage-based** allocation in the Supabase **dashboard** before the first compute scale-up — **not settable via SQL/MCP** (operator-owned). Verify via a staging soak + an approval-gated read-only advisor re-check. | `docs/auth-connection-cap-runbook.md`; `docs/process-hardening.md` (Known follow-up debts) | 2026-07-21 | | #012 | P3 | rec | Slim the lazy cross-mode differentials chunk | `cross-mode-differentials.ts` is dynamically imported (correctly code-split **out** of the initial/dashboard bundle — verified), but it pulls the full ~860 KB differentials snapshot (~125 KB gzip lazy chunk) just to build a tiny `{slug,title,clinicalHinge}` + presentations + aliases catalog. A precomputed lightweight index (generator + drift check, like the `specifiers-content` split / medications `fields=index`) would cut that lazy chunk ~5–10×. Not a bundle leak — an M-effort slim. | `src/lib/cross-mode-differentials.ts`; `src/components/clinical-dashboard/cross-mode-links.tsx:150`; session 2026-07-21 (build:analyze) | 2026-07-21 | | #013 | P3 | rec | Route-chunk + mockup catalogue JSON weight | `build:analyze`: `/specifiers` ships `specifiers-search-index.json` (~180 KB parsed), `/forms` ships `forms-catalog.json` (~132 KB), `/formulation` ships `formulation-content.json` (~52 KB, client-side local search — needs index/full split or a search endpoint, architectural). All route-scoped (not initial bundle). Also `*-mockups.tsx` (~100 KB across chunks) build though `/mockups` 404s in prod — exclude from the prod artifact. | session 2026-07-21 (build:analyze) | 2026-07-21 | -| #014 | P3 | rec | Realize the `next/image` win on signed previews | `next.config` `images` (AVIF + `*.supabase.co` `remotePatterns` pinned to the project host, from #1024) is currently inert — signed document/image previews still render as raw ``. Route them through `next/image` to actually get AVIF + lazy optimization. | #1024; `src/components/clinical-dashboard/signed-image.tsx`; session 2026-07-21 | 2026-07-21 | | #016 | P3 | rec | "Big but not easy" structural + motion perf | Deferred larger levers: (a) nonce-CSP forces every product route to `ƒ Dynamic` (zero static generation) — evaluate Partial Prerendering / static shells for the static clinical catalogues (DSM/differentials/therapy/specifiers/formulation); (b) sidebar expand/collapse animates `grid-template-columns` (biggest smoothness cost, motion-gated — needs a transform-overlay rethink); (c) Therapy Compass fetches 692 KB / 2.5 MB JSON client-side (defer until interaction + confirm brotli); (d) settings/setup/admin dialogs static-imported into the home chunk (`next/dynamic` them). | session 2026-07-21 (build route table + design audit) | 2026-07-21 | | #017 | P3 | task | Field Web-Vitals baseline via live Lighthouse | In-sandbox runtime vitals were blocked (prod server hard-requires Supabase secrets; dev-mode CLS measured excellent at 0.00–0.04, content-first pages 0.000). Run Lighthouse against `psychiatry.tools` for real LCP/INP/CLS to prioritize #012–#016 by measured impact rather than reasoning. | session 2026-07-21 (measurement pass) | 2026-07-21 | | #018 | P2 | task | Split the lithium, ADHD and metabolic residuals by mechanism | Revalidated on current main 2026-07-23: these are not one composer defect. Lithium reproduced an unrelated-table retrieval fast-path defect; ADHD retrieves a relevant chart-heavy CAMHS source but exhausts the extractive route budget; metabolic retrieves the correct AKG source but selects schedule-free prose. The narrow lithium subject-evidence guard improved targeting from 0 to 1 with golden recall 1.0 and no reciprocal-rank regressions, but it was reverted because the full canary failed. After #051 stabilises the canary, add independent current-main reproducers and assess each mechanism separately. Do not widen the matcher or combine these into a broad ranking/composer change. | runs `30007833352` and `30009207429`; PR #1093; session 2026-07-23 | 2026-07-21 | @@ -116,10 +141,9 @@ removed after current-main verification; it is not missing recommended work. | #027 | P3 | rec | External uptime monitor independent of GitHub/Railway | `live-domain-monitor.yml` runs on GitHub's cron, so it won't run in exactly the outage it should catch (Actions or the deploy itself down). Add an off-platform synthetic monitor (UptimeRobot / Better Stack / Checkly) hitting `/api/health` with a webhook alert. Provider setup, not code. | session 2026-07-22 webhook review | 2026-07-22 | | #028 | P3 | rec | Runtime error tracking (Sentry or similar) | No error tracking in the repo — production exceptions on `psychiatry.tools`, including how often `RAG_PROVIDER_MODE=auto` silently degrades to source-only, are invisible. Weigh adding `@sentry/nextjs` (dependency + DSN secret + instrumentation) vs cost; alert → chat/issue. Provider-backed; needs explicit sign-off before adding the dependency. | session 2026-07-22 webhook review | 2026-07-22 | | #029 | P2 | issue | 12 of 30 answer-quality cases return the fallback stub | run #61 --dump-answers: 12/30 quality cases emit the source_backed_review_fallback boilerplate with answer_sections: [], all grounded with 4-6 citations. Some still PASS targeting because the stub echoes query keywords (the contraindication/document_lookup matchers need only a keyword), so the targeting metric MASKS the problem for those intents. Superset of #018 — fix in the extractive composer, validate with the provider-backed answer eval. | run #61 dump artifact; session 2026-07-22 | 2026-07-22 | -| #030 | P2 | issue | Wide-tier alias lets one doc satisfy both comparison slots | In src/lib/eval-document-matching.ts, "Admission to Discharge for Mental Health Inpatients" appears in BOTH the AdmissionCommunityPts and Discharge alias lists, so a single document can satisfy both expectedFiles slots and make allHit true — a latent false-pass on admission-discharge cases. Not firing today (that doc is not in the failing top-5) but it would mask a real miss. Tighten the tables so one doc cannot fill both sides. | src/lib/eval-document-matching.ts:32-65; session 2026-07-22 | 2026-07-22 | +| #030 | P3 | issue | Wide-tier alias lets one doc satisfy both comparison slots | In src/lib/eval-document-matching.ts, "Admission to Discharge for Mental Health Inpatients" appears in BOTH the AdmissionCommunityPts and Discharge alias lists, so a single document can satisfy both expectedFiles slots and make allHit true — a latent false-pass on admission-discharge cases. Not firing today (that doc is not in the failing top-5) but it would mask a real miss. Tighten the tables so one doc cannot fill both sides. | src/lib/eval-document-matching.ts:32-65; session 2026-07-22 | 2026-07-22 | | #032 | P3 | rec | Governance ranking weighting: REFUTED, not debt | The source-governance audit (PR #1051) flagged three "gaps": `review_due` carries no ranking penalty, `unknownCurrentnessPenalty` ships at 0, and `selectBestSourceRecommendation` ignores governance metadata. **These are deliberate, measured decisions — do NOT implement them as written.** Blanket metadata boosts/penalties in selection ordering were measured on 2026-07-02 to regress the golden retrieval eval to 16/23 (doc-recall@5 1.0→0.76, mrr 0.75→0.64). Two corpus facts make it unsafe: scores saturate at the clamp so stacked boosts fully override lexical relevance, and the corpus is only partially metadata-enriched while `normalizeSourceMetadata` coerces unenriched docs to `unknown`/`unverified` — so "unknown" ≠ "bad" and blanket weighting swings ranking approx. 0.35 for reasons unrelated to relevance. Even governance-as-tiebreak buried correct unenriched docs (3 designs bisected). Next action: none — treat as a guardrail. If ever revisited, RC8 (source-strength as a _filter_) is the tracked path, gated on `eval:retrieval:quality` 36/36 plus a live canary pair. | PR #118; `docs/rag-behaviour/refuted-approaches.md`; PR #1051 items 4/5/6 | 2026-07-22 | | #033 | P3 | rec | Source governance metadata absent from the LLM prompt | `buildRagSourceBlock` omits `document_status`, `clinical_validation_status`, and `extraction_quality`, so the model cannot self-caveat during generation and governance is enforced only post-hoc. Generation-surface change: needs `eval:rag` plus `eval:quality --rag-only` (grounded-supported must not drop, citation-failure 0) and explicit approval. Carries the same "unknown ≠ bad" hazard as #032 — on a partially-enriched corpus the model would likely over-caveat correct sources, so design the prompt wording before spending an eval. | `src/lib/rag/rag-source-block.ts:126-198`; PR #1051 audit item 8 | 2026-07-22 | -| #034 | P3 | issue | Answer cache can serve stale governance metadata | `cacheIndexingVersion` derives the version from `updated_at` / `indexed_at` / `index_generation_id`, so a metadata-only `document_status` flip that bumps none of those is invisible to the passive guard. **Already mitigated**: every known status-write path calls `invalidateRagCachesForOwner` or `invalidateRagCachesForDocumentMutation`. Residual risk only — a future write path that omits the invalidator would serve stale governance until TTL. Next action: add a regression test pinning the invalidator call on status-mutating routes (cheaper and safer than touching the protected cache key). | `src/lib/rag/rag-cache.ts:382-438`; PR #1051 audit item 10 | 2026-07-22 | | #035 | P3 | rec | Threshold-conflict detection covers only 3 params | `detectThresholdDisagreements` checks only ANC, WBC, and platelets paired with withholding verbs, so cross-source conflicts on medication doses, lithium/thyroid levels, or vital signs go undetected. Deliberately narrow (see the comment at `:469-474`). Broadening changes when an answer is classified `conflicting` and adds warnings — real false-positive risk. Needs new fixtures plus a behaviour review before any change. | `src/lib/evidence.ts:469-574`; PR #1051 audit item 7 | 2026-07-22 | | #036 | P3 | rec | No explicit `is_public` visibility flag on documents | Public-corpus visibility is implicit: `owner_id IS NULL` on an `indexed` document (`resolveSearchScope`). The `metadata.public_corpus` marker is written by the promotion migrations but never used as a retrieval filter. Promotion is unconditional on `clinical_validation_status`, so unverified documents are publicly searchable — compensated by keeping `unverified_source` in the frontend-visible warning set. A hard schema flag touches RLS and the clinical-risk-gated retrieval RPCs; weigh against the existing compensating control before acting. | `supabase/schema.sql:61-108`; `src/lib/search-scope.ts:181-236`; PR #1051 audit item 3 | 2026-07-22 | | #037 | P3 | rec | D5 trust-cap-all-claims flag parked OFF | `NEXT_PUBLIC_RAG_TRUST_CAP_ALL_CLAIMS` extends authority gating from high-risk claims to **all** supported claims (`deriveTrust`). Ships OFF by design; flipping it caps trust to `medium` for routine claims across the board — a product/clinical-UX decision, not a defect. Both states are test-pinned. Next action: product decision, then flip and re-baseline the UI expectations. | `src/lib/answer-render-policy.ts:159-177`; PR #1051 audit item 11 | 2026-07-22 | @@ -135,6 +159,8 @@ Move resolved rows here with the resolution date and a one-line outcome. Keep th | ID | Type | Summary | Outcome | Resolved | | ---- | ----- | ---------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------- | | #026 | task | Wire the Supabase document-change trigger | PR #1100 merged after disposable PostgreSQL replay and hosted migration replay. Production migration history and read-only catalog proof confirm the enabled metadata trigger, security-definer function, pinned search path and denied anonymous/authenticated execution; `npm run check:drift` reports no unexpected live drift. Delivery remains intentionally inert until the operator inputs tracked in #025 are configured. | 2026-07-24 | +| #034 | issue | Answer cache can serve stale governance metadata | Current-source verification found direct route coverage already asserts RAG-cache invalidation on document PATCH, source review, label, bulk and reindex mutation paths. The residual test recommendation is therefore already met; changing the protected cache key is unnecessary. | 2026-07-24 | +| #014 | rec | Realize the `next/image` win on signed previews | Superseded: `SignedImage` uses `next/image` for layout/sizing but deliberately sets `unoptimized`, preventing bearer signed URLs from entering the unauthenticated optimizer cache where cached content could outlive the token. No optimization task remains unless private-image delivery changes. | 2026-07-24 | | #031 | issue | Populate canary Source Governance table | The answer-quality step now consumes the preceding `golden-retrieval.json` only for source-governance reporting. Offline replay of run `30018289898` populated 338 top results, including 202 review-required entries, while retaining zero retrieval cases and no additional threshold failures. Retrieval and ranking behavior are unchanged. | 2026-07-24 | | #020 | task | Validate eval:quality cost readout post-fix | Confirmed on merged-main canary run `30018289898`: Answer Metrics reported 9 nonzero-cost cases and an estimated answer cost of `$0.234736`; the structured report retained the same value. The PR #1050 estimator fix is operationally proven. | 2026-07-23 | | #003 | task | Staging tenancy release evidence outstanding | Ran GitHub Action and validated isolation | 2026-07-21 |