diff --git a/docs/branch-review-records/381b41511e145024c0d34d173ed186f1659871b294a41e390fa36fdf6f5bdfb9.record.md b/docs/branch-review-records/381b41511e145024c0d34d173ed186f1659871b294a41e390fa36fdf6f5bdfb9.record.md new file mode 100644 index 0000000000..6fcf9cf51a --- /dev/null +++ b/docs/branch-review-records/381b41511e145024c0d34d173ed186f1659871b294a41e390fa36fdf6f5bdfb9.record.md @@ -0,0 +1 @@ +| 2026-08-18 | claude/docling-gate-b-eval-5czgln | 397f40d96c4c08c5da61b8fcf07e166a93cf7aba | packet S6b: Gate B run + decision record (eval/docling harness fixes + docs/rag-improvement records) | Gate B PASS recorded from evidence run 32176604314; four latent harness defects fixed (setuptools pin, libGL, torch.compile toolchain, HTML-entity scoring) | verify:pr-local selected plan green (lint, typecheck, docs/ledger contracts); npm run test 673 files / 7281 passed / 4 skipped; check:rag:fixtures 36 golden / 26 suites; check:docling-lab contract passed; gate-b record valid (final mode) | diff --git a/docs/outstanding-issues-inbox/abc21f52-0d39-490d-9676-d106b9af4302.json b/docs/outstanding-issues-inbox/abc21f52-0d39-490d-9676-d106b9af4302.json new file mode 100644 index 0000000000..aa912ac1d5 --- /dev/null +++ b/docs/outstanding-issues-inbox/abc21f52-0d39-490d-9676-d106b9af4302.json @@ -0,0 +1,14 @@ +{ + "version": 2, + "id": "abc21f52-0d39-490d-9676-d106b9af4302", + "createdOn": "2026-08-18", + "action": "add", + "payload": { + "pri": "P2", + "type": "task", + "summary": "Build packet B4: Docling worker shadow mode (WORKER_DOCUMENT_EXTRACTOR_MODE=legacy|shadow) — authorised by the Gate B PASS of 2026-08-18", + "detail": "Gate B passed on 2026-08-18 (evidence run 32176604314 at 8a92378; record docs/rag-improvement/gate-b-decision-record-2026-08-18.md): design-only authorisation for README section B4 per HANDOVER S7+. Scope: typed WORKER_DOCUMENT_EXTRACTOR_MODE env defaulting to legacy, shadow runs after legacy success on a 1-5 percent cohort selected by src/lib/index-quality.ts signals, aggregate metadata only, no chunk/embedding/index writes, kill switch, one-step rollback to legacy; ingestion-worker-reviewer reviews the PR. Two caveats travel from the decision record: the table-heavy leg passed at parity-on-ceiling (fixtures.v2 hardness corpus precedes any table-quality promotion argument), and docling runs eager at roughly 9-19 s/doc on 2 CPUs vs legacy's 1 s — cohort sizing must budget for it.", + "source": "packet S6b Gate B decision record, 2026-08-18", + "issueUlid": "01M0B66FJ69DGA6R5W8QN9V7CY" + } +} diff --git a/docs/outstanding-issues-inbox/fb87f710-b46c-452a-b349-e3ce71c36959.json b/docs/outstanding-issues-inbox/fb87f710-b46c-452a-b349-e3ce71c36959.json new file mode 100644 index 0000000000..fa7a4ccca2 --- /dev/null +++ b/docs/outstanding-issues-inbox/fb87f710-b46c-452a-b349-e3ce71c36959.json @@ -0,0 +1,14 @@ +{ + "version": 2, + "id": "fb87f710-b46c-452a-b349-e3ce71c36959", + "createdOn": "2026-08-18", + "action": "add", + "payload": { + "pri": "P3", + "type": "task", + "summary": "Docling lab fixtures.v2 table-hardness corpus: add unruled, merged-cell, and rotated-header tables so the Gate B table-heavy improvement leg has measurable headroom", + "detail": "The v1 corpus's table strata are cleanly ruled grids on which the legacy extractor already scores cell F1 1.0 (S6 smoke run and the S6b Gate B run), so the pre-agreed table-heavy improvement target was set to 0 pp (parity at ceiling) by owner decision on 2026-08-18. Before any table-heavy delta is treated as decisive for a Docling promotion beyond B4 shadow design, add a docling-lab-fixtures.v2 stratum set where the legacy find_tables path is expected to degrade: unruled tables, merged cells, rotated headers. Fixture-hardness change only — eval/docling/ manifest + generator, no worker or extractor edit. See eval/docling/README.md 'Known limitation (v1 corpus)' and docs/rag-improvement/gate-b-decision-record-2026-08-18.md.", + "source": "packet S6b (Gate B run), owner threshold decision 2026-08-18", + "issueUlid": "01M0AXXCJMBSBE9BYR4QX6QBC8" + } +} diff --git a/docs/rag-improvement/HANDOVER.md b/docs/rag-improvement/HANDOVER.md index 260855b4de..ffc47e0227 100644 --- a/docs/rag-improvement/HANDOVER.md +++ b/docs/rag-improvement/HANDOVER.md @@ -76,25 +76,26 @@ generation-quality verdict on fallback`), merged 2026-08-13 — structured ## 2. Status table — update in every programme PR -| Packet | Scope | Branch | PR | State | Canary / evidence refs | -| ---------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------ | --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| Guide | Programme guide | `claude/rag-plan-review-guide-vhrls9` | #1895 | Merged 2026-08-13 | docs-only | -| Handover | Multi-session handover + coordination | `claude/rag-plan-review-guide-vhrls9` | #1908 / #2024 | Merged 2026-08-13; coordination layer PR #2024 | docs-only | -| S0 | A1 phase 1: structured fallback diagnostics | `claude/lithium-generation-quality-debug-ji1vce` | #1899 | Merged 2026-08-13 | offline 93/93 focused | -| S1 | A1 phase 2: rung-1 verification-faithfulness fixes | `claude/s1-rag-mitigation-231-86c182` | #2022 | Merged 2026-08-17 (squash `2bd146eed`, landed by content) | 8 pre-fix + 5 post-fix live probes 2026-08-17; offline 583/583; canary pair run 31964560921 (baseline `8f8d111ab`) -> run 32025082010 (`2bd146eed`): recall 1.0/1.0, zero per-case rr regressions, answer gate 44/44 (denominator reconciled by S5; see baseline-record §3); rung-2 measurement in `docs/audit/live-drift-forensics-2026-08.md` §5 | -| S1b | A1 rung 3 (R1): pre-deadline strong routing for dosing class | `claude/s1b-rag-dosing-routing-6u1mik` | #2035 | Merged 2026-08-17 (PR #2035, merge `92f7618`) | canary pair pending: baseline run 32025082010 (`2bd146eed`) -> post-merge dispatch (owner-approved); offline 586/586 + verify:pr-local heavy scope green | -| S1c | A1 residuals R2 + R3: claim-support strictness | `claude/s1c-residuals-r2-r3-4pb1at` | #2052 | Merged 2026-08-17 (merge `b8e774bcd`; follow-up #2063 kept; follow-up #2065 reverted by PR #2088 after canary regression) | canary pair: baseline run 32049952885 -> post run 32052479537 (`084f63799`): recall 1.0/1.0, zero per-case rr regressions, answer gate 44/44 | -| S1d | A1 final-gate gap recovery: hedged cited low-confidence fast answers must recover extractively, not collapse to a citation-free `provider_source_gap` | `claude/s1d-final-gate-gap-recovery-dxgrn2` | #2054 | Merged 2026-08-17 (merge `0bbd64fbc`); landed by content; canary pair green after the #2065 revert (PR #2088) | canary pair 32052479537 -> 32100681177 (`4ea310e48`) green: recall 1.0/1.0, zero per-case rr regressions, answer gate 44/44. Interim post run 32097916649 (`9904fbda8`) was RED on `agitation-im-po-route-short-terms` — bisected live to PR #2065 (S1c follow-up condition-first regex), not S1d; reverted by PR #2088; the confirmation run is 32100681177 | -| G1 | Governance: provenance tag for document-summary rows (Option B) | `claude/g1-rag-document-context-qn9ubx` | #2053 | Merged 2026-08-17 (merge `125e98526`); rows #J912J9 / #0MSNT8 closed at reconcile | no canary (no behaviour change); `document_context` tag at `types.ts` + `rag-row-contracts.ts`, deriveConfidence pinned | -| S2 | A2 + A3: composition menu + moderate length | `claude/s2-rag-composition-7330b0` | #2097 | Merged 2026-08-18 (squash dda4956ff), landed by content; A2 and A3 together; prompt clinical-rag-answer-v19, schema v4, answerSections.maxItems 6 | offline: composition 9/9 + prompt pins 5/5, eval:rag:offline 26 suites / 623 tests, eval:rag:adversarial:offline 25/25 (3 divergences still pinned), rag.ts 4362/4362; canary pair 32100681177 (4ea310e48) -> 32111839806 GREEN (document/content recall 1.0/1.0, zero per-case rr regressions, answer gate clean; one non-blocking 20 s latency advisory on neuroleptic-side-effect-escalation); eval:answer-quality v18 (4ea310e48) vs v19 run 2026-08-18 neutral within nondeterminism (relevance 0.633->0.567 on two nondeterministic timeout/gap cases, targeting 0.409->0.429, readability at ceiling), owner blinded read pending. Note: the 220-word total-length readability ceiling in scoreAnswerQualityEvalCase is a known metric confound for A3 (baseline-record §4) | -| S2b | A3: moderate length (if separate review needed) | — | — | Not needed — A3 shipped inside S2 (combined diff stayed reviewable) | — | -| S3 | A4: follow-up suggestion refinement | `claude/s3-follow-up-suggestions-95e160` | #2108 | Merged 2026-08-18 (squash 511d22f4d), landed by content; Track A complete | offline only: answer-follow-up 27/27 (menu alignment across all 48 class×intent cells, evidence gate positive + discriminating negatives, suppression), answer-follow-up-chips DOM 6/6 (desktop + phone composer surfaces), answer-composition 9/9 unchanged; no canary — deterministic composition only, generation prompt untouched | -| S4 | B0: adversarial fixtures + baseline + register | `claude/packet-s4-adversarial-fixtures-5ho5tp` | #2036 | Merged 2026-08-17 (squash `f5b093291`) | Offline only: `check:rag:adversarial-fixtures` 24 cases / 8 categories / 6 canaries; `eval:rag:offline` 24 suites, 597 tests. Baseline `scripts/fixtures/rag-adversarial-baseline.v1.json` marks the three provider-backed gates `pending_owner_run` | -| S5 | B1+B2: telemetry assessment + offline harness | `claude/s5-rag-telemetry-harness-2wvis7` | #2056 | Merged 2026-08-17 (merge `093f9340c`); post-merge canary run 32049952885 | Offline only: `eval:rag:adversarial:offline` 25/25 (24 cases + canary-free report; 3 divergences pinned in `KNOWN_DIVERGENCES`); B1 gap = `verification_latency_ms` behind `RAG_TELEMETRY_EXTENDED` (default false); canary-absence tests green; 44/44 denominator reconciled | -| S6 | B3: Docling lab benchmark | `claude/packet-s6-docling-lab-d6foa6` | #2057 | Merged 2026-08-17 (merge `5a6418636`) | Offline only: `check:docling-lab` 36 fixtures / 10 hostile / 6 canaries + Gate B template valid; `verify:pr-local` heavy plan failed:(none); contract test 20/20; legacy smoke 46 docs, 10/10 hostile contained, canary-clean report. Verdict is a separate owner dispatch of `docling-lab.yml` | -| S7+ | B4 shadow / B5 Ragas / B6 reranker / B7 DSPy | — | — | Gated — owner decision | — | -| #212 T1–T3 | Runtime row contracts (rag.ts, rag-candidate-sources.ts, src/app/api) — sibling stream sharing `src/lib/rag/**` | — | #1946 / #1981 / #2023 | Merged (T3 squash `440a34f71` 2026-08-17) | see the #212 ledger row; RAG surface complete for the cast class | -| #212 T4 | Runtime row contracts: `worker/main.ts` (11 casts) — sibling stream | `claude/ledger-212-tranche-4-worker-q3y6i4` | #2037 | Merged 2026-08-17 (squash `1726537b7`); #212 closed by reconcile PR #2045 | Governance Preflight complete; audit: 1 inbound cast (claim rows, per-row fail-soft) + 2 read-back param casts contracted, 9 outbound/interop left; closes #212 (inbox `done` queued in the PR) | +| Packet | Scope | Branch | PR | State | Canary / evidence refs | +| ---------- | ----------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------ | --------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Guide | Programme guide | `claude/rag-plan-review-guide-vhrls9` | #1895 | Merged 2026-08-13 | docs-only | +| Handover | Multi-session handover + coordination | `claude/rag-plan-review-guide-vhrls9` | #1908 / #2024 | Merged 2026-08-13; coordination layer PR #2024 | docs-only | +| S0 | A1 phase 1: structured fallback diagnostics | `claude/lithium-generation-quality-debug-ji1vce` | #1899 | Merged 2026-08-13 | offline 93/93 focused | +| S1 | A1 phase 2: rung-1 verification-faithfulness fixes | `claude/s1-rag-mitigation-231-86c182` | #2022 | Merged 2026-08-17 (squash `2bd146eed`, landed by content) | 8 pre-fix + 5 post-fix live probes 2026-08-17; offline 583/583; canary pair run 31964560921 (baseline `8f8d111ab`) -> run 32025082010 (`2bd146eed`): recall 1.0/1.0, zero per-case rr regressions, answer gate 44/44 (denominator reconciled by S5; see baseline-record §3); rung-2 measurement in `docs/audit/live-drift-forensics-2026-08.md` §5 | +| S1b | A1 rung 3 (R1): pre-deadline strong routing for dosing class | `claude/s1b-rag-dosing-routing-6u1mik` | #2035 | Merged 2026-08-17 (PR #2035, merge `92f7618`) | canary pair pending: baseline run 32025082010 (`2bd146eed`) -> post-merge dispatch (owner-approved); offline 586/586 + verify:pr-local heavy scope green | +| S1c | A1 residuals R2 + R3: claim-support strictness | `claude/s1c-residuals-r2-r3-4pb1at` | #2052 | Merged 2026-08-17 (merge `b8e774bcd`; follow-up #2063 kept; follow-up #2065 reverted by PR #2088 after canary regression) | canary pair: baseline run 32049952885 -> post run 32052479537 (`084f63799`): recall 1.0/1.0, zero per-case rr regressions, answer gate 44/44 | +| S1d | A1 final-gate gap recovery: hedged cited low-confidence fast answers must recover extractively, not collapse to a citation-free `provider_source_gap` | `claude/s1d-final-gate-gap-recovery-dxgrn2` | #2054 | Merged 2026-08-17 (merge `0bbd64fbc`); landed by content; canary pair green after the #2065 revert (PR #2088) | canary pair 32052479537 -> 32100681177 (`4ea310e48`) green: recall 1.0/1.0, zero per-case rr regressions, answer gate 44/44. Interim post run 32097916649 (`9904fbda8`) was RED on `agitation-im-po-route-short-terms` — bisected live to PR #2065 (S1c follow-up condition-first regex), not S1d; reverted by PR #2088; the confirmation run is 32100681177 | +| G1 | Governance: provenance tag for document-summary rows (Option B) | `claude/g1-rag-document-context-qn9ubx` | #2053 | Merged 2026-08-17 (merge `125e98526`); rows #J912J9 / #0MSNT8 closed at reconcile | no canary (no behaviour change); `document_context` tag at `types.ts` + `rag-row-contracts.ts`, deriveConfidence pinned | +| S2 | A2 + A3: composition menu + moderate length | `claude/s2-rag-composition-7330b0` | #2097 | Merged 2026-08-18 (squash dda4956ff), landed by content; A2 and A3 together; prompt clinical-rag-answer-v19, schema v4, answerSections.maxItems 6 | offline: composition 9/9 + prompt pins 5/5, eval:rag:offline 26 suites / 623 tests, eval:rag:adversarial:offline 25/25 (3 divergences still pinned), rag.ts 4362/4362; canary pair 32100681177 (4ea310e48) -> 32111839806 GREEN (document/content recall 1.0/1.0, zero per-case rr regressions, answer gate clean; one non-blocking 20 s latency advisory on neuroleptic-side-effect-escalation); eval:answer-quality v18 (4ea310e48) vs v19 run 2026-08-18 neutral within nondeterminism (relevance 0.633->0.567 on two nondeterministic timeout/gap cases, targeting 0.409->0.429, readability at ceiling), owner blinded read pending. Note: the 220-word total-length readability ceiling in scoreAnswerQualityEvalCase is a known metric confound for A3 (baseline-record §4) | +| S2b | A3: moderate length (if separate review needed) | — | — | Not needed — A3 shipped inside S2 (combined diff stayed reviewable) | — | +| S3 | A4: follow-up suggestion refinement | `claude/s3-follow-up-suggestions-95e160` | #2108 | Merged 2026-08-18 (squash 511d22f4d), landed by content; Track A complete | offline only: answer-follow-up 27/27 (menu alignment across all 48 class×intent cells, evidence gate positive + discriminating negatives, suppression), answer-follow-up-chips DOM 6/6 (desktop + phone composer surfaces), answer-composition 9/9 unchanged; no canary — deterministic composition only, generation prompt untouched | +| S4 | B0: adversarial fixtures + baseline + register | `claude/packet-s4-adversarial-fixtures-5ho5tp` | #2036 | Merged 2026-08-17 (squash `f5b093291`) | Offline only: `check:rag:adversarial-fixtures` 24 cases / 8 categories / 6 canaries; `eval:rag:offline` 24 suites, 597 tests. Baseline `scripts/fixtures/rag-adversarial-baseline.v1.json` marks the three provider-backed gates `pending_owner_run` | +| S5 | B1+B2: telemetry assessment + offline harness | `claude/s5-rag-telemetry-harness-2wvis7` | #2056 | Merged 2026-08-17 (merge `093f9340c`); post-merge canary run 32049952885 | Offline only: `eval:rag:adversarial:offline` 25/25 (24 cases + canary-free report; 3 divergences pinned in `KNOWN_DIVERGENCES`); B1 gap = `verification_latency_ms` behind `RAG_TELEMETRY_EXTENDED` (default false); canary-absence tests green; 44/44 denominator reconciled | +| S6 | B3: Docling lab benchmark | `claude/packet-s6-docling-lab-d6foa6` | #2057 | Merged 2026-08-17 (merge `5a6418636`) | Offline only: `check:docling-lab` 36 fixtures / 10 hostile / 6 canaries + Gate B template valid; `verify:pr-local` heavy plan failed:(none); contract test 20/20; legacy smoke 46 docs, 10/10 hostile contained, canary-clean report. Verdict is a separate owner dispatch of `docling-lab.yml` | +| S6b | Gate B run: docling-lab dispatch, verdict, decision record | `claude/docling-gate-b-eval-5czgln` | (this PR) | Gate B **PASS** 2026-08-18 (evidence run 32176604314 at `8a92378`); four latent harness defects found+fixed en route (setuptools pin, libGL, torch.compile/no-toolchain, HTML-entity scoring) | All five gates pass at pre-agreed 0 pp margins: parse 36/36 both engines, exactness 162/162 both, table F1 parity at ceiling (agreed 0 pp target; fixtures.v2 hardness follow-up queued), hostile 10/10 contained / 0 crash / 0 canary echo, resources max 12.6 s P95 / 1.40 GiB vs 120 s / 6 GiB caps; record: `docs/rag-improvement/gate-b-decision-record-2026-08-18.{md,json}`, validated `--final` | +| S7+ | B4 shadow / B5 Ragas / B6 reranker / B7 DSPy | — | — | B4: **unblocked — Gate B PASS 2026-08-18** (design-only authorisation, caveats in the decision record); B5/B6/B7 still gated — owner decision | — | +| #212 T1–T3 | Runtime row contracts (rag.ts, rag-candidate-sources.ts, src/app/api) — sibling stream sharing `src/lib/rag/**` | — | #1946 / #1981 / #2023 | Merged (T3 squash `440a34f71` 2026-08-17) | see the #212 ledger row; RAG surface complete for the cast class | +| #212 T4 | Runtime row contracts: `worker/main.ts` (11 casts) — sibling stream | `claude/ledger-212-tranche-4-worker-q3y6i4` | #2037 | Merged 2026-08-17 (squash `1726537b7`); #212 closed by reconcile PR #2045 | Governance Preflight complete; audit: 1 inbound cast (claim rows, per-row fail-soft) + 2 read-back param casts contracted, 9 outbound/interop left; closes #212 (inbox `done` queued in the PR) | Update rule: the session that opens a packet's PR edits its row (branch, PR number, state) in the same PR. A later session updating another packet may also correct stale rows diff --git a/docs/rag-improvement/gate-b-decision-record-2026-08-18.json b/docs/rag-improvement/gate-b-decision-record-2026-08-18.json new file mode 100644 index 0000000000..49ca341773 --- /dev/null +++ b/docs/rag-improvement/gate-b-decision-record-2026-08-18.json @@ -0,0 +1,106 @@ +{ + "recordVersion": "docling-lab-gate-b.v1", + "status": "pass", + "note": "Owner-run Gate B decision record for the Docling lab benchmark (packet S6b). Thresholds were agreed and committed BEFORE any benchmark dispatch (branch commit 6d05c07525e9bc30b43edd0eaeda279a57ae607b, 2026-08-18). The recorded evidence run is workflow run 32176604314 (artifact docling-lab-report-32176604314) at commit 8a923787c81e4256d4e22e45eed65ccb26f5fba8 — the first run in the dispatch chain whose docling pass is a valid measurement; see runHistory. Human twin: docs/rag-improvement/gate-b-decision-record-2026-08-18.md. Validate: node eval/docling/report/build-report.mjs --validate-record docs/rag-improvement/gate-b-decision-record-2026-08-18.json --final.", + "preAgreedThresholds": { + "agreedBeforeRun": true, + "parseSuccessNonInferiorityMarginPp": 0, + "numericExactnessNonInferiorityMarginPp": 0, + "tableHeavyCellF1ImprovementTargetPp": 0, + "thresholdRationale": "Zero-pp non-inferiority margins match the programme's zero-tolerance posture (36/36 golden fixture, zero per-case regressions): docling must lose no parse and no dose/unit/comparator assertion legacy preserves, in any stratum. The table-heavy improvement target is 0 pp by explicit owner decision (this session, 2026-08-18): the v1 corpus's cleanly ruled tables put the legacy extractor at cell F1 1.0 in the S6 smoke run, so parity at ceiling satisfies the improvement leg, the headroom caveat is recorded in the decision, and a docling-lab-fixtures.v2 table-hardness follow-up (unruled, merged-cell, rotated-header tables) is queued in the issues inbox.", + "resourceCeilings": "eval/docling/report/lab-config.json (sandbox + outputCaps) at the run's commit_sha" + }, + "reportKey": { + "commit_sha": "8a923787c81e4256d4e22e45eed65ccb26f5fba8", + "dataset_version": "docling-lab-fixtures.v1", + "eval_config_version": "docling-lab-config-v1", + "model_version": "answer=gpt-5.6-terra; fast=gpt-5.6-terra; strong=gpt-5.6-sol", + "embedding_version": "text-embedding-3-small@1536", + "index_version": "20260818113000_forward_codify_hybrid_owner_matches_bodies" + }, + "reportKeyNote": "Copied verbatim from the run artifact's stamped reportKey. model_version, embedding_version, and index_version are programme-wide comparability qualifiers not exercised by this extraction benchmark — no model, embedding, or database call occurs inside the sandbox.", + "qualifiers": { + "extractorVersions": { + "legacy": "src/lib/extractors/document.ts + worker/python/extract_pdf_assets.py (pymupdf==1.28.0, worker/python/requirements.txt)", + "docling": "docling==2.120.2 (eval/docling/requirements.txt)" + }, + "workflowRunUrl": "https://github.com/BigSimmo/Database/actions/runs/32176604314", + "artifactName": "docling-lab-report-32176604314", + "doclingPhaseEnvironment": "TORCHDYNAMO_DISABLE=1 (eager mode; the sandbox image ships no C++ toolchain for torch.compile — eval/docling/harness/entry.sh)" + }, + "runHistory": [ + { + "runId": "32164356999", + "commit": "0216f18e97c224ce9bf50799ba8c4837909e781f", + "outcome": "infrastructure failure — docker build: pip --require-hashes refused torch's setuptools>=77.0.3 runtime requirement (lock generated without --allow-unsafe); fixed by 779af80" + }, + { + "runId": "32165911181", + "commit": "779af8094827d9cd8ad32d3d164ca2b0c5fe6e4e", + "outcome": "infrastructure failure — docling model prefetch: rapidocr imports cv2, slim base lacked libGL.so.1/glib; fixed by 958a70d" + }, + { + "runId": "32166445937", + "commit": "958a70d90b373ee45d26645f4cc1cddf492209d0", + "outcome": "workflow green but not a valid docling measurement — all 46 docling documents errored uniformly; cause locked in never-uploaded raw output; diagnostic error surface added by c66faf8" + }, + { + "runId": "32171549648", + "commit": "c66faf89072011dc3307b86bcc2be8425e1cbaff", + "outcome": "diagnostic run — named the failure: InvalidCxxCompiler (torch.compile needs a C++ compiler the sandbox image deliberately lacks); fixed by 6c6e80e (TORCHDYNAMO_DISABLE=1)" + }, + { + "runId": "32174653778", + "commit": "6c6e80ea1cdc31865f01c8c8e8aac06b40cb55b6", + "outcome": "first valid docling pass — parse 36/36, table F1 parity, hostile clean, resources in bounds; numeric exactness scored -25 to -62.5pp purely because docling's markdown export HTML-escapes comparators ('>= 1.4 mmol/L' as '>= 1.4 mmol/L') and the scorer did not unescape; values verified intact locally; scorer made escape-neutral by 8a92378" + }, + { + "runId": "32176604314", + "commit": "8a923787c81e4256d4e22e45eed65ccb26f5fba8", + "outcome": "recorded evidence run — all five gates pass under the pre-agreed thresholds" + } + ], + "gates": [ + { + "id": "parse_success", + "caseCount": 36, + "status": "recorded", + "result": "PASS — docling 36/36 and legacy 36/36; comparison.perStratum[*].parseSuccessDeltaPp = 0 in all six strata (rule: >= 0, margin 0 pp)", + "evidence": "https://github.com/BigSimmo/Database/actions/runs/32176604314 artifact docling-lab-report-32176604314" + }, + { + "id": "resource_bounds", + "caseCount": 46, + "status": "recorded", + "result": "PASS — no timeout, signal kill, or output-cap breach on any of 46 documents for either engine; docling max wallClockMsP95 12,558 ms (cap 120,000 ms; hostile cap 60,000 ms), max peakRssBytesMax 1,504,313,344 bytes (~1.40 GiB, cgroup cap 6 GiB), run completed inside the 3,600 s cap, report 1,285 bytes (cap 1 MiB)", + "evidence": "https://github.com/BigSimmo/Database/actions/runs/32176604314 artifact docling-lab-report-32176604314" + }, + { + "id": "table_precision_recall", + "caseCount": 12, + "status": "recorded", + "result": "PASS — tableCellPrecision/Recall/F1 = 1.0 for both engines on table_simple and table_heavy; tableCellF1DeltaPp = 0 on both (non-inferiority rule: >= 0); table-heavy improvement leg met at the pre-agreed 0 pp parity-at-ceiling target — the v1 corpus gives legacy F1 1.0, so no improvement was demonstrable by construction (fixtures.v2 follow-up queued)", + "evidence": "https://github.com/BigSimmo/Database/actions/runs/32176604314 artifact docling-lab-report-32176604314" + }, + { + "id": "numeric_exactness", + "caseCount": 36, + "status": "recorded", + "result": "PASS — docling 162/162 and legacy 162/162 dose/unit/comparator assertions found; comparison.perStratum[*].assertionCoverageDeltaPp = 0 in all six strata (rule: >= 0, margin 0 pp); scored escape-neutrally (html.unescape in score.py normalise) so entity-encoded comparators compare by value", + "evidence": "https://github.com/BigSimmo/Database/actions/runs/32176604314 artifact docling-lab-report-32176604314" + }, + { + "id": "hostile_containment", + "caseCount": 10, + "status": "recorded", + "result": "PASS — docling containedCount 10/10 and legacy 10/10; containedDeltaCount 0 (rule: >= 0), crashArtifactTotal 0 (rule: = 0), canaryEchoTotal 0 (rule: = 0); every hostile failure was a clean bounded error within its per-document limit", + "evidence": "https://github.com/BigSimmo/Database/actions/runs/32176604314 artifact docling-lab-report-32176604314" + } + ], + "decision": { + "outcome": "pass", + "ownerSignoff": "Thresholds, every dispatch (six approvals), and the parity-at-ceiling table target were explicitly owner-approved in the S6b session on 2026-08-18; final countersign is the owner's review/merge of the packet S6b pull request.", + "date": "2026-08-18", + "consequence": "Gate B pass authorises DESIGNING packet B4 (WORKER_DOCUMENT_EXTRACTOR_MODE=legacy|shadow worker shadow mode) only, with two explicit caveats carried forward: (1) the table-heavy improvement leg was satisfied at parity-on-ceiling, not by a demonstrated gain — a docling-lab-fixtures.v2 hardness corpus is queued before any promotion beyond shadow design leans on a table-quality argument; (2) docling ran in eager mode (TORCHDYNAMO_DISABLE=1) at ~9-19 s per document on 2 CPUs versus legacy's ~1 s — shadow-cohort sizing must budget for that. The worker remains untouched; the ingestion-worker-reviewer subagent reviews the B4 PR." + } +} diff --git a/docs/rag-improvement/gate-b-decision-record-2026-08-18.md b/docs/rag-improvement/gate-b-decision-record-2026-08-18.md new file mode 100644 index 0000000000..ddbb79384c --- /dev/null +++ b/docs/rag-improvement/gate-b-decision-record-2026-08-18.md @@ -0,0 +1,94 @@ +# Gate B decision record — Docling extraction benchmark (owner run, 2026-08-18) + +**Status: PASS.** This is the owner's filled copy of +`docs/rag-improvement/gate-b-decision-record.md` for packet **S6b** (the Gate B run the +S6 harness deliberately shipped without). Per the template's §3 rule, the thresholds below +were agreed in the S6b session and **committed before any benchmark dispatch** (branch +commit `6d05c07`); the gate results are filled from the recorded evidence run's +`docling-lab-report-32176604314` artifact. Machine-readable twin: +`docs/rag-improvement/gate-b-decision-record-2026-08-18.json`, validated with +`node eval/docling/report/build-report.mjs --validate-record docs/rag-improvement/gate-b-decision-record-2026-08-18.json --final`. + +Gate B (README §Gates A–F): **non-inferior on all safety/exactness measures, improved on +the pre-agreed table-heavy metric, no budget breach.** This pass authorises _designing_ +packet B4 (worker shadow mode) only; the worker remains untouched. + +## 1. Provenance discipline + +As `baseline-record.md`: a gate result is either `recorded` with a result **and** the run +or artifact it came from, or `pending_owner_run` with a stated reason and no result. A +number cannot be entered without provenance. + +## 2. Report key + +Copied verbatim from the run artifact's stamped key. + +| Field | This run | +| --------------------- | -------------------------------------------------------------- | +| `commit_sha` | `8a923787c81e4256d4e22e45eed65ccb26f5fba8` | +| `dataset_version` | `docling-lab-fixtures.v1` | +| `eval_config_version` | `docling-lab-config-v1` | +| `model_version` | `answer=gpt-5.6-terra; fast=gpt-5.6-terra; strong=gpt-5.6-sol` | +| `embedding_version` | `text-embedding-3-small@1536` | +| `index_version` | `20260818113000_forward_codify_hybrid_owner_matches_bodies` | + +`model_version`, `embedding_version`, and `index_version` are programme-wide +comparability qualifiers not exercised by this extraction benchmark — no model, embedding, +or database call occurs inside the sandbox. Outside the key: `extractorVersions.legacy` +(pymupdf==1.28.0 path) and `extractorVersions.docling` (docling==2.120.2) from +`eval/docling/report/lab-config.json`; evidence run +, artifact +`docling-lab-report-32176604314`; docling phase environment `TORCHDYNAMO_DISABLE=1` +(eager mode — the sandbox image ships no C++ toolchain for torch.compile). + +## 3. Pre-agreed thresholds — agreed and committed BEFORE the dispatch + +| Threshold | Agreed value | Rationale (owner) | +| ------------------------------------------------------------- | -------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Parse-success non-inferiority margin (pp, per stratum) | **0** | Zero-tolerance posture: docling must parse no fewer clean fixtures than legacy in any stratum. | +| Numeric/unit/comparator exactness non-inferiority margin (pp) | **0** | Any dose, unit, or comparator string legacy preserves that docling loses is clinically meaningful; per stratum, `assertionCoverageDeltaPp >= 0`. | +| Table-heavy improvement target (pp cell F1 on `table_heavy`) | **0** | Owner decision 2026-08-18: the v1 corpus's cleanly ruled tables put legacy at cell F1 1.0 (S6 smoke), leaving no headroom; parity at ceiling satisfies the leg, and a fixtures.v2 table-hardness follow-up is queued. | +| Resource ceilings | `eval/docling/report/lab-config.json` (sandbox + outputCaps) at the run's `commit_sha` | Fixed by the lab. | + +Hostile-corpus treatment (hard safety gate, not a threshold to tune): +`canaryEchoTotal` must be 0, `crashArtifactTotal` must be 0, `containedDeltaCount >= 0`. + +## 4. Gate results + +Evidence for every row: run , +artifact `docling-lab-report-32176604314`. + +| Gate | Cases | Status | Result | +| ------------------------ | ----- | ---------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `parse_success` | 36 | `recorded` | **PASS** — docling 36/36, legacy 36/36; `parseSuccessDeltaPp` = 0 in all six strata (rule ≥ 0) | +| `resource_bounds` | 46 | `recorded` | **PASS** — no timeout/signal/output-cap breach either engine; docling max wall P95 12,558 ms (cap 120 s; hostile 60 s), max peak RSS 1,504,313,344 B (~1.40 GiB, cap 6 GiB), report 1,285 B | +| `table_precision_recall` | 12 | `recorded` | **PASS** — cell F1 1.0 both engines on `table_simple` and `table_heavy`; deltas 0 (rule ≥ 0); improvement leg met at the pre-agreed 0 pp parity-at-ceiling target | +| `numeric_exactness` | 36 | `recorded` | **PASS** — docling 162/162, legacy 162/162 assertions; `assertionCoverageDeltaPp` = 0 in all six strata (rule ≥ 0); scored escape-neutrally (`html.unescape`) so comparators compare by value | +| `hostile_containment` | 10 | `recorded` | **PASS** — contained 10/10 both engines; `containedDeltaCount` 0, `crashArtifactTotal` 0, `canaryEchoTotal` 0; every hostile failure a clean bounded error | + +### Run history (the dispatch chain behind the evidence run) + +| Run | Commit | Outcome | +| ---------------------------------------------------------------------------- | --------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| [32164356999](https://github.com/BigSimmo/Database/actions/runs/32164356999) | `0216f18` | Infra failure — hashed lock omitted setuptools (`pip-compile` without `--allow-unsafe`); torch requires it at runtime. Fixed by `779af80`. | +| [32165911181](https://github.com/BigSimmo/Database/actions/runs/32165911181) | `779af80` | Infra failure — model prefetch: rapidocr imports cv2, slim base lacked `libGL.so.1`/glib. Fixed by `958a70d`. | +| [32166445937](https://github.com/BigSimmo/Database/actions/runs/32166445937) | `958a70d` | Workflow green but **not a valid docling measurement** — all 46 docling docs errored uniformly; cause invisible in aggregate output. Diagnostic surface: `c66faf8`. | +| [32171549648](https://github.com/BigSimmo/Database/actions/runs/32171549648) | `c66faf8` | Diagnostic — named the failure: `InvalidCxxCompiler` (torch.compile needs a C++ compiler the sandbox deliberately lacks). Fixed by `6c6e80e` (`TORCHDYNAMO_DISABLE=1`). | +| [32174653778](https://github.com/BigSimmo/Database/actions/runs/32174653778) | `6c6e80e` | First valid docling pass — exactness scored −25…−62.5 pp purely on HTML-entity encoding (`>=` exported as `>=`); values verified intact locally. Fixed by `8a92378`. | +| [32176604314](https://github.com/BigSimmo/Database/actions/runs/32176604314) | `8a92378` | **Recorded evidence run — all five gates pass.** | + +## 5. Decision + +| Field | Value | +| -------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Outcome | **`pass`** | +| Owner sign-off | Thresholds, every dispatch (six approvals), and the parity-at-ceiling table target were explicitly owner-approved in the S6b session on 2026-08-18; final countersign is the owner's review/merge of the packet S6b pull request. | +| Date | 2026-08-18 | +| Consequence | Packet **B4 (shadow mode) may be designed**: `WORKER_DOCUMENT_EXTRACTOR_MODE=legacy\|shadow`, default `legacy`; `ingestion-worker-reviewer` reviews that PR. Two caveats travel with the pass: (1) the table-heavy leg passed at parity-on-ceiling, not by a demonstrated gain — the queued `docling-lab-fixtures.v2` hardness corpus precedes any promotion argument built on table quality; (2) docling ran eager at ~9–19 s/doc on 2 CPUs vs legacy's ~1 s — shadow-cohort sizing must budget for that. | + +## 6. Related + +- `docs/rag-improvement/gate-b-decision-record.md` (the committed template this copies) +- `docs/rag-improvement/README.md` §B3, §B4, §Gates A–F +- `docs/rag-improvement/HANDOVER.md` packets S6 / S6b +- `eval/docling/README.md` (harness, sandbox contract, known v1 table-ceiling limitation) diff --git a/eval/docling/Dockerfile b/eval/docling/Dockerfile index 2d259a0738..a977e0db4f 100644 --- a/eval/docling/Dockerfile +++ b/eval/docling/Dockerfile @@ -9,6 +9,10 @@ # python3 is 3.11, matching both hashed locks consumed below). FROM node:24-bookworm-slim@sha256:235600a8101ab264e117b1768e925532262668dc9b581ef1dd7d96ced463b8e7 +# libgl1 + libglib2.0-0: OpenCV's runtime shared libraries. docling's rapidocr +# stage imports cv2 during `docling-tools models download` (and again at run +# time), and the slim base ships neither — run 32165911181 failed the prefetch +# with `ImportError: libGL.so.1: cannot open shared object file`. RUN apt-get update \ && apt-get install -y --no-install-recommends \ python3 \ @@ -16,6 +20,8 @@ RUN apt-get update \ tesseract-ocr \ fonts-dejavu-core \ ca-certificates \ + libgl1 \ + libglib2.0-0 \ && rm -rf /var/lib/apt/lists/* # Legacy comparator venv: read-only consumption of the worker's production hashed diff --git a/eval/docling/generate-lock.mjs b/eval/docling/generate-lock.mjs index c495ed6067..aa78e324cb 100644 --- a/eval/docling/generate-lock.mjs +++ b/eval/docling/generate-lock.mjs @@ -68,9 +68,17 @@ function main() { const python = venvPython(venvDir); run(python, ["-m", "pip", "install", "setuptools", "wheel"]); run(python, ["-m", "pip", "install", `pip-tools==${PIP_TOOLS_VERSION}`]); - run(python, ["-m", "piptools", "compile", "--generate-hashes", "--output-file", OUT_FILE, IN_FILE], { - env: { ...process.env, CUSTOM_COMPILE_COMMAND: GENERATE_COMMAND }, - }); + // --allow-unsafe pins setuptools/wheel INTO the lock. Without it pip-tools omits + // them as "unsafe", and the image build's `pip install --require-hashes` then + // refuses torch's `setuptools>=77.0.3` runtime requirement as an unpinned + // dependency — the exact failure of docling-lab run 32164356999 (2026-08-18). + run( + python, + ["-m", "piptools", "compile", "--generate-hashes", "--allow-unsafe", "--output-file", OUT_FILE, IN_FILE], + { + env: { ...process.env, CUSTOM_COMPILE_COMMAND: GENERATE_COMMAND }, + }, + ); assertLockShape(); console.log(`Generated ${OUT_FILE} for Python ${REQUIRED_PYTHON}`); } finally { diff --git a/eval/docling/harness/entry.sh b/eval/docling/harness/entry.sh index 0d0aacb7c9..b86b0a64b3 100755 --- a/eval/docling/harness/entry.sh +++ b/eval/docling/harness/entry.sh @@ -33,7 +33,13 @@ PYTHON_BIN="$LEGACY_PY" "$LEGACY_PY" "$LAB/harness/run_corpus.py" \ --out "$OUT/raw/legacy.json" echo "== phase 3: docling engine ==" -DOCLING_PYTHON="$DOCLING_PY" DOCLING_ARTIFACTS_PATH=/opt/docling-models \ +# TORCHDYNAMO_DISABLE=1: docling's models call torch.compile, whose Inductor +# backend needs a C++ compiler at runtime — the sandbox image deliberately has +# none, which failed every clean conversion in run 32171549648 with +# InvalidCxxCompiler. Eager mode is deterministic and needs no toolchain; each +# document runs in a fresh process anyway, so per-doc compilation only added +# overhead. +TORCHDYNAMO_DISABLE=1 DOCLING_PYTHON="$DOCLING_PY" DOCLING_ARTIFACTS_PATH=/opt/docling-models \ "$DOCLING_PY" "$LAB/harness/run_corpus.py" \ --engine docling \ --corpus "$OUT/corpus" \ diff --git a/eval/docling/harness/run_corpus.py b/eval/docling/harness/run_corpus.py index 523377779f..4ddede249a 100755 --- a/eval/docling/harness/run_corpus.py +++ b/eval/docling/harness/run_corpus.py @@ -125,6 +125,40 @@ def run_one(cmd: list[str], timeout_seconds: int, cwd: Path) -> dict: } +def collect_canary_tokens(manifest: dict) -> list[str]: + tokens = [] + for entry in manifest.get("canaryRegistry", []): + token = entry.get("token") if isinstance(entry, dict) else entry + if isinstance(token, str) and token: + tokens.append(token) + return tokens + + +def sanitised_error_summary(result_path: Path, canary_tokens: list[str]) -> str | None: + """Bounded, canary-redacted error description for the per-doc progress line. + + The workflow log is a reportable sink, so this prints only the runner's + recorded exception name/message — never stream tails or extracted text — + with every registered canary token redacted and the whole line truncated. + Diagnosing run 32166445937 (all 46 docling docs uniformly `error` with the + cause locked inside never-uploaded raw output) required exactly this. + """ + if not result_path.exists(): + return None + try: + payload = json.loads(result_path.read_text(encoding="utf-8")) + except (OSError, ValueError): + return None + error = payload.get("error") + if not isinstance(error, dict): + return None + name = str(error.get("name", ""))[:80] + message = " ".join(str(error.get("message", "")).split()) + for token in canary_tokens: + message = message.replace(token, "[CANARY]") + return f"{name}: {message[:200]}" if name or message else None + + def main() -> None: parser = argparse.ArgumentParser() parser.add_argument("--engine", required=True, choices=["legacy", "docling"]) @@ -141,6 +175,7 @@ def main() -> None: config = json.loads(Path(args.config).read_text(encoding="utf-8")) sandbox = config["sandbox"] caps = config["outputCaps"] + canary_tokens = collect_canary_tokens(manifest) os.environ.setdefault("LAB_PER_DOC_TEXT_BYTES", str(caps["perDocumentTextBytes"])) documents = [ @@ -190,11 +225,15 @@ def main() -> None: record["resultBytes"] = result_bytes record["resultPath"] = str(result_path.relative_to(results_dir.parent)) if result_bytes > 0 else None per_doc.append(record) - print( + line = ( f"run_corpus[{args.engine}] {doc['id']}: {record['exitReason']} " - f"{record['wallClockMs']}ms rss={record['peakRssBytes'] // (1024 * 1024)}MB", - flush=True, + f"{record['wallClockMs']}ms rss={record['peakRssBytes'] // (1024 * 1024)}MB" ) + if record["exitReason"] not in ("completed",): + summary = sanitised_error_summary(result_path, canary_tokens) + if summary: + line += f" [{summary}]" + print(line, flush=True) Path(args.out).write_text( json.dumps( diff --git a/eval/docling/harness/score.py b/eval/docling/harness/score.py index cdc6edc9ab..9851fdf240 100755 --- a/eval/docling/harness/score.py +++ b/eval/docling/harness/score.py @@ -35,6 +35,7 @@ from __future__ import annotations import argparse +import html import json import re from collections import Counter @@ -44,7 +45,12 @@ def normalise(text: str) -> str: - return " ".join(text.replace("\\|", "|").split()).lower() + # Escape-neutral like the existing `\|` handling: docling's markdown export + # HTML-escapes angle brackets, so a comparator assertion like '>= 1.4 mmol/L' + # surfaces as '>= 1.4 mmol/L' — the value intact, only entity-encoded. + # Run 32174653778 scored those as missing (-25 to -62.5pp per stratum) purely + # on this encoding difference; exactness compares values, not escaping. + return " ".join(html.unescape(text.replace("\\|", "|")).split()).lower() def percentile(values: list[int], fraction: float) -> int: diff --git a/eval/docling/requirements.txt b/eval/docling/requirements.txt index 371dfde22b..8ef71561c0 100644 --- a/eval/docling/requirements.txt +++ b/eval/docling/requirements.txt @@ -2054,7 +2054,8 @@ xlsxwriter==3.2.9 \ --hash=sha256:9a5db42bc5dff014806c58a20b9eae7322a134abb6fce3c92c181bfb275ec5b3 # via python-pptx -# WARNING: The following packages were not pinned, but pip requires them to be -# pinned when the requirements file includes hashes and the requirement is not -# satisfied by a package already installed. Consider using the --allow-unsafe flag. -# setuptools +# The following packages are considered to be unsafe in a requirements file: +setuptools==84.0.0 \ + --hash=sha256:51a52592b3b99e102b609654876bd65f19f999935166d1352678931132b0c670 \ + --hash=sha256:f4695c21257f0d9b537ec2692c941d02ee143b7cc1276941349a546573b2ef73 + # via torch