From 869c84044efd5d665ef6e86f0b7fb495eaed064a Mon Sep 17 00:00:00 2001 From: hallelx2 Date: Fri, 18 Sep 2026 06:52:10 +0100 Subject: [PATCH 1/7] =?UTF-8?q?feat(ingest):=20send=20a=20Judge=20the=20mi?= =?UTF-8?q?nimum=20context=20=E2=80=94=20prefilter,=20truncation,=20two-st?= =?UTF-8?q?age=20scan?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The principle this pipeline is now engineered to: every token sent has to earn its place. A Judge's latency is dominated by the state it is sent, not the questions asked of it — detection over a 20-page prefix at 12k chars per page cost 8.77s, ten request-floors for one request. Most of those pages were cover sheets and boilerplate no reader would mistake for a table of contents. Three cuts, behind TOCBuilder.MinimalContext: 1. A structural pre-filter in Go. A page is skipped ONLY when it shows zero sign of being a contents page — not low signal, zero. A threshold would be a place for recall to leak out one unusual layout at a time. The heuristic's only job is to be certain about the obvious negatives; the model keeps the ambiguous cases. The signal is not "lines ending in a number", because the parser flattens layout and there are no lines. What survives is the entry SHAPE — a short title then a small number, repeated — plus ITEM and PART markers for SEC filings. Prose matches the shape once or twice by accident, so the bar is three. 2. Per-page truncation for detection cut from 12,000 chars to 2,000. A contents page reveals itself immediately or not at all. 3. A two-stage scan. Pages 1-6 first; only on a miss, 7-20. Every 10-K has its TOC at page 2-3, so the common case is one small request. Measured 2026-09-18 on 21 FinanceBench 10-Ks, Jev arm only: detected TOC pages 21/21 identical to the full-context path input tokens 431,667 -> 59,685 (7.2x fewer) wall-clock 25.4s -> 9.7s (2.6x) Not the 4x predicted, because several documents now sit at the ~800ms request floor — Walmart 0.8s, J&J 0.9s — which is where cutting state stops helping. Real variance remains at similar token counts (Amazon 5.0s on 3.4k, Walmart 0.8s on 1.6k) and is the API, not the pipeline. Also here, measured and kept as a negative result: speculative fan-out (asking the verification question alongside detection, in one request) is 31% SLOWER at the same request count. The docs say extra questions "typically don't add latency"; they add ~30%. Speculation pays only when it eliminates a request, and here it eliminated none. Code retained so the measurement reproduces; not on any default path. cmd/tocdump dumps the full-pipeline tree per document; cmd/tocdump/ coverage.py joins it with FinanceBench's gold evidence pages. That join is the accuracy gate for every cut above, and for anything that follows: "the model agreed with another model" was never a check. --- cmd/ingestbench/dashboard/app.js | 256 +++++++++++++++++++++++++ cmd/ingestbench/dashboard/events.jsonl | 1 + cmd/ingestbench/dashboard/index.html | 74 +++++++ cmd/ingestbench/dashboard/style.css | 165 ++++++++++++++++ cmd/ingestbench/events.go | 111 +++++++++++ cmd/ingestbench/main.go | 249 ++++++++++++++++++++++++ cmd/tocdump/coverage.py | 125 ++++++++++++ cmd/tocdump/main.go | 209 ++++++++++++++++++++ pkg/ingest/bench_export.go | 32 ++++ pkg/ingest/prefilter.go | 156 +++++++++++++++ pkg/ingest/prefilter_test.go | 94 +++++++++ pkg/ingest/toc_builder.go | 25 +++ pkg/ingest/toc_judge.go | 80 +++++++- pkg/ingest/toc_judge_fanout.go | 157 +++++++++++++++ 14 files changed, 1729 insertions(+), 5 deletions(-) create mode 100644 cmd/ingestbench/dashboard/app.js create mode 120000 cmd/ingestbench/dashboard/events.jsonl create mode 100644 cmd/ingestbench/dashboard/index.html create mode 100644 cmd/ingestbench/dashboard/style.css create mode 100644 cmd/ingestbench/events.go create mode 100644 cmd/ingestbench/main.go create mode 100644 cmd/tocdump/coverage.py create mode 100644 cmd/tocdump/main.go create mode 100644 pkg/ingest/prefilter.go create mode 100644 pkg/ingest/prefilter_test.go create mode 100644 pkg/ingest/toc_judge_fanout.go diff --git a/cmd/ingestbench/dashboard/app.js b/cmd/ingestbench/dashboard/app.js new file mode 100644 index 0000000..3f01c5a --- /dev/null +++ b/cmd/ingestbench/dashboard/app.js @@ -0,0 +1,256 @@ +// Live view over the ingestbench JSON-Lines event stream. +// +// Re-reads the whole file each poll instead of tracking an offset: the +// stream is small, and a stateless reload cannot drift out of sync with +// a run that restarted underneath it. + +const POLL_MS = 1000; +const ARM = { + jev: { label: 'Jev', color: '#ff5a00', chip: 'jev' }, + generative: { label: 'GLM', color: '#3f3f46', chip: 'gen' }, +}; + +let events = [], lastLen = -1; + +async function poll() { + try { + const r = await fetch('events.jsonl?t=' + Date.now(), { cache: 'no-store' }); + if (r.ok) { + const txt = await r.text(); + if (txt.length !== lastLen) { + lastLen = txt.length; + events = txt.trim().split('\n').filter(Boolean) + .map(l => { try { return JSON.parse(l); } catch { return null; } }) + .filter(Boolean); + render(); + } else { tickClock(); } + } + } catch {} + setTimeout(poll, POLL_MS); +} + +const detects = () => events.filter(e => e.t === 'phase' && e.phase === 'detect'); +const parses = () => events.filter(e => e.t === 'phase' && e.phase === 'parse'); +const starts = () => events.filter(e => e.t === 'phase_start'); +const byArm = a => detects().filter(e => e.arm === a); +const armEnd = a => events.find(e => e.t === 'arm_end' && e.arm === a); +const runEnd = () => events.find(e => e.t === 'run_end'); +const sum = (xs, f) => xs.reduce((a, x) => a + (f(x) || 0), 0); + +const fmtS = s => s >= 60 ? `${Math.floor(s/60)}m ${(s%60).toFixed(0)}s` : `${s.toFixed(1)}s`; +const fmtUSD = v => v >= 0.01 ? `$${v.toFixed(3)}` : `$${v.toFixed(5)}`; +const fmtN = n => n.toLocaleString('en-US'); + +function tickClock() { + const done = runEnd(); + const last = events.length ? events[events.length - 1].time : 0; + document.getElementById('clock').textContent = done ? fmtS(done.seconds) : fmtS(last); +} + +function render() { tickClock(); status(); kpis(); docgrid(); charts(); totals(); timeline(); log(); } + +function status() { + const el = document.getElementById('status'); + if (runEnd()) { el.innerHTML = 'complete'; return; } + if (!events.length) { el.innerHTML = 'waiting'; return; } + el.innerHTML = 'live'; +} + +function kpis() { + const j = byArm('jev'), g = byArm('generative'); + const je = armEnd('jev'), ge = armEnd('generative'); + const jS = je ? je.seconds : sum(j, e => e.seconds); + const gS = ge ? ge.seconds : sum(g, e => e.seconds); + const speed = (jS > 0 && gS > 0) ? (gS / jS) : null; + const jc = sum(j, e => e.cost_usd), gc = sum(g, e => e.cost_usd); + const cheaper = (jc > 0 && gc > 0) ? (gc / jc) : null; + const pr = parses(); + + const cell = (v, l, ember) => `
${v}
${l}
`; + document.getElementById('kpis').innerHTML = + cell(`${pr.length}`, 'documents') + + cell(fmtN(sum(pr, e => e.pages)), 'pages parsed') + + cell(jS ? fmtS(jS) : '—', 'jev wall-clock', true) + + cell(gS ? fmtS(gS) : '—', 'glm wall-clock') + + cell(speed ? `${speed.toFixed(0)}×` : '—', speed ? 'faster' : 'speed-up', true); + + document.getElementById('docsmeta').textContent = + cheaper ? `${cheaper.toFixed(0)}× cheaper · ${sum(j, e => e.requests)} vs ${sum(g, e => e.requests)} requests` : ''; +} + +function docgrid() { + const pr = parses(); + const running = new Set(starts() + .filter(s => !detects().some(d => d.doc === s.doc && d.arm === s.arm)) + .map(s => s.doc + '|' + s.arm)); + + document.getElementById('docgrid').innerHTML = pr.map(p => { + const d = detects().filter(x => x.doc === p.doc); + const live = [...running].some(k => k.startsWith(p.doc + '|')); + const chips = ['jev', 'generative'].map(a => { + const e = d.find(x => x.arm === a); + if (e) return `${ARM[a].label} ${e.seconds.toFixed(1)}s · ${e.requests}r`; + if (running.has(p.doc + '|' + a)) return `${ARM[a].label} running`; + return ''; + }).join(''); + + const pct = (d.length / 2) * 100; + return `
+
${p.doc.replace(/_/g, ' ')}
+
${p.pages} pages · parsed ${p.seconds.toFixed(1)}s
+
+
${chips}
+
`; + }).join('') || '

waiting for the first document…

'; +} + +// --- charts (hand-rolled SVG; no library, and none needed) --- + +function hbars(elId, rows, unit, maxOverride) { + const el = document.getElementById(elId); + if (!rows.length) { el.innerHTML = '

no data yet

'; return; } + const max = maxOverride || Math.max(...rows.map(r => r.v), 1); + const H = 22, GAP = 7, LW = 150, W = 640; + const h = rows.length * (H + GAP); + + const bars = rows.map((r, i) => { + const y = i * (H + GAP); + const w = Math.max(2, (r.v / max) * (W - LW - 90)); + return `${r.label} + + ${r.t}`; + }).join(''); + + el.innerHTML = `${bars} +

${unit}

`; +} + +function line(elId, series, yfmt, caption) { + const el = document.getElementById(elId); + const pts = series.flatMap(s => s.points); + if (pts.length < 2) { el.innerHTML = '

no data yet

'; return; } + const W = 640, H = 220, P = 34; + const maxX = Math.max(...pts.map(p => p[0]), 1); + const maxY = Math.max(...pts.map(p => p[1]), 1e-9); + const X = x => P + (x / maxX) * (W - P - 12); + const Y = y => H - P - (y / maxY) * (H - P - 14); + + const paths = series.filter(s => s.points.length > 1).map(s => + ``).join(''); + + const dots = series.map(s => s.points.length + ? `` : '').join(''); + + let ticks = ''; + for (let i = 0; i <= 3; i++) { + const v = (maxY / 3) * i, y = Y(v); + ticks += ` + ${yfmt(v)}`; + } + const legend = series.map(s => + `${s.label}`).join(''); + + el.innerHTML = `
${legend}
+ ${ticks}${paths}${dots} + 0s + ${fmtS(maxX)} +

${caption}

`; +} + +function charts() { + const docs = [...new Set(detects().map(e => e.doc))].sort(); + + const lat = [], req = []; + for (const d of docs) for (const a of ['jev', 'generative']) { + const e = detects().find(x => x.doc === d && x.arm === a); + if (!e) continue; + const short = d.replace(/_/g, ' ').replace(/ 10K$/, ''); + lat.push({ label: `${short} · ${ARM[a].label}`, v: e.seconds, t: fmtS(e.seconds), color: ARM[a].color }); + req.push({ label: `${short} · ${ARM[a].label}`, v: e.requests, t: `${e.requests}`, color: ARM[a].color }); + } + hbars('chart-latency', lat, 'detection phase only; parsing is shared and excluded'); + hbars('chart-requests', req, 'one batched request vs one call per scanned page'); + + // Cumulative cost and completion, both over the shared run clock. + const cum = arm => { + const es = byArm(arm).slice().sort((a, b) => a.time - b.time); + let c = 0; + return es.map(e => { c += (e.cost_usd || 0); return [e.time, c]; }); + }; + const done = arm => { + const es = byArm(arm).slice().sort((a, b) => a.time - b.time); + return es.map((e, i) => [e.time, i + 1]); + }; + line('chart-cost', [ + { label: 'Jev', color: ARM.jev.color, points: cum('jev') }, + { label: 'GLM', color: ARM.generative.color, points: cum('generative') }, + ], v => fmtUSD(v), 'cost accrued over the run'); + line('chart-throughput', [ + { label: 'Jev', color: ARM.jev.color, points: done('jev') }, + { label: 'GLM', color: ARM.generative.color, points: done('generative') }, + ], v => v.toFixed(0), 'documents finished, against the run clock'); +} + +function totals() { + const rows = ['jev', 'generative'].map(a => { + const e = byArm(a); if (!e.length) return ''; + const c = sum(e, x => x.cost_usd); + return ` + ${ARM[a].label} + ${e.length} + ${sum(e, x => x.requests)} + ${fmtN(sum(e, x => x.in_tokens))} + ${fmtS(sum(e, x => x.seconds))} + ${c > 0 ? fmtUSD(c) : '—'} + ${c > 0 ? fmtUSD(c / e.length) : '—'} + ${c > 0 ? fmtUSD(c / e.length * 10000) : '—'} + `; + }).join(''); + document.getElementById('cost').innerHTML = ` + + ${rows || ''}
armdocsreqtokensmodel timecostper docper 10k docs
no data yet
`; +} + +function timeline() { + const d = detects(); + const el = document.getElementById('timeline'); + if (!d.length) { el.innerHTML = '

nothing scheduled yet

'; return; } + const maxT = Math.max(...d.map(e => e.time), 1); + const items = d.map(e => { + const s = starts().find(x => x.doc === e.doc && x.arm === e.arm); + return { ...e, t0: s ? s.time : e.time - e.seconds }; + }).sort((a, b) => a.t0 - b.t0); + + const lanes = []; + let bars = ''; + for (const it of items) { + let l = lanes.findIndex(end => end <= it.t0 + 0.01); + if (l === -1) { lanes.push(0); l = lanes.length - 1; } + lanes[l] = it.time; + bars += `
`; + } + let ticks = ''; + for (let i = 0; i <= 4; i++) ticks += `
${fmtS((i / 4) * maxT)}
`; + el.innerHTML = `
${bars}
${ticks}
`; +} + +function log() { + document.getElementById('log').innerHTML = events.slice(-30).reverse().map(e => { + const t = `${e.time.toFixed(1)}s`; + if (e.t === 'phase' && e.phase === 'detect') + return `${t} ${e.arm} ${e.doc} · ${e.seconds.toFixed(1)}s · ${e.requests}r · ${fmtUSD(e.cost_usd || 0)}`; + if (e.t === 'phase' && e.phase === 'parse') + return `${t} parse ${e.doc} · ${e.pages}p · ${e.seconds.toFixed(1)}s`; + if (e.t === 'phase_start') return `${t} start ${e.arm} ${e.doc}`; + if (e.t === 'arm_start') return `${t} ── ${e.arm} ──`; + if (e.t === 'arm_end') return `${t} ${e.arm} done ${fmtS(e.seconds)}`; + if (e.t === 'run_end') return `${t} run complete ${fmtS(e.seconds)}`; + if (e.t === 'run_start') return `${t} ${e.note}`; + return `${t} ${e.t}`; + }).join('
'); +} + +poll(); diff --git a/cmd/ingestbench/dashboard/events.jsonl b/cmd/ingestbench/dashboard/events.jsonl new file mode 120000 index 0000000..16b3bf8 --- /dev/null +++ b/cmd/ingestbench/dashboard/events.jsonl @@ -0,0 +1 @@ +/home/hallelx2/.cache/vlbench/runs/full.jsonl \ No newline at end of file diff --git a/cmd/ingestbench/dashboard/index.html b/cmd/ingestbench/dashboard/index.html new file mode 100644 index 0000000..0f59304 --- /dev/null +++ b/cmd/ingestbench/dashboard/index.html @@ -0,0 +1,74 @@ + + + + + +Vectorless ingestion — live + + + + + + +
+ +
+
+
+ vectorless-engine + FinanceBench 10-K + +
+

Ingestion, live

+
+
0.0s
+
+ +
+ +
+

Documents

+
+
+ +
+
+

Time per document

+
+
+
+

Requests per document

+
+
+
+ +
+
+

Cumulative cost

+
+
+
+

Documents completed

+
+
+
+ +
+

Schedule

overlapping bars ran concurrently
+
+
+ +
+

Totals

+
+
+ +
+

Stream

+
+
+ +
+ + + diff --git a/cmd/ingestbench/dashboard/style.css b/cmd/ingestbench/dashboard/style.css new file mode 100644 index 0000000..eb86ed3 --- /dev/null +++ b/cmd/ingestbench/dashboard/style.css @@ -0,0 +1,165 @@ +/* Awesomic — editorial zinc grid with confetti-orange punctuation. */ +:root { + --color-obsidian: #09090b; + --color-graphite: #18181b; + --color-slate: #27272a; + --color-iron: #3f3f46; + --color-steel: #52525b; + --color-fog: #71717a; + --color-ash: #a1a1aa; + --color-mist: #d4d4d8; + --color-cloud: #ececee; + --color-paper: #f4f4f5; + --color-snow: #ffffff; + --color-ember: #ff5a00; + + --font-cosmica: 'DM Sans', ui-sans-serif, system-ui, -apple-system, + BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif; + + --text-caption: 12px; --leading-caption: 1.64; + --text-body: 15px; --leading-body: 1.45; + --text-subheading: 20px; + --text-heading-sm: 32px; + --text-heading: 40px; --leading-heading: 1.28; + --text-display: 64px; --leading-display: 1.12; + + --spacing-8: 8px; --spacing-12: 12px; --spacing-16: 16px; + --spacing-20: 20px; --spacing-24: 24px; --spacing-28: 28px; + --spacing-32: 32px; --spacing-40: 40px; --spacing-48: 48px; + --spacing-64: 64px; --spacing-80: 80px; + + --page-max-width: 1200px; + --radius-cards: 36px; + --radius-badges: 12px; + --radius-buttons: 14px; + --radius-pills: 10000px; + + --shadow-dark-btn: + inset 0 0.5px 0 0 rgba(255,255,255,0.5), + inset 0 9px 14px -5px rgba(117,123,133,0.4), + 0 0 0 1.5px rgb(44,46,52), + 0 4px 6px 0 rgba(0,0,0,0.14); +} + +* { box-sizing: border-box; } + +body { + margin: 0; + background: var(--color-paper); + color: var(--color-graphite); + font-family: var(--font-cosmica); + font-size: var(--text-body); + line-height: var(--leading-body); + -webkit-font-smoothing: antialiased; +} + +.wrap { max-width: var(--page-max-width); margin: 0 auto; padding: var(--spacing-48) var(--spacing-24) var(--spacing-80); } +.section { margin-top: var(--spacing-80); } +.section:first-child { margin-top: 0; } + +/* --- type --- */ +h1 { font-size: var(--text-display); line-height: var(--leading-display); font-weight: 600; color: var(--color-obsidian); margin: 0 0 var(--spacing-16); letter-spacing: -0.02em; } +h2 { font-size: var(--text-heading); line-height: var(--leading-heading); font-weight: 600; color: var(--color-obsidian); margin: 0 0 var(--spacing-8); letter-spacing: -0.01em; } +h3 { font-size: var(--text-subheading); font-weight: 600; color: var(--color-obsidian); margin: 0 0 var(--spacing-8); } +.lede { font-size: 18px; color: var(--color-steel); max-width: 68ch; margin: 0 0 var(--spacing-24); } +.muted { color: var(--color-steel); } +.tiny { font-size: var(--text-caption); line-height: var(--leading-caption); color: var(--color-fog); } + +/* --- badges --- */ +.badge { display: inline-flex; align-items: center; gap: 6px; border-radius: var(--radius-badges); padding: 4px 8px; font-size: 12px; font-weight: 500; } +.badge-outline { border: 1px solid var(--color-cloud); background: transparent; color: var(--color-graphite); } +.badge-filled { background: var(--color-iron); color: #fafafa; } +.badge-ember { background: var(--color-ember); color: #fff; } + +/* --- cards --- */ +.card { background: var(--color-snow); border: 1px solid var(--color-cloud); border-radius: var(--radius-cards); padding: var(--spacing-28); } +.card-dark { background: var(--color-slate); color: var(--color-snow); border: none; border-radius: var(--radius-cards); padding: var(--spacing-28); } +.card-dark h2, .card-dark h3 { color: var(--color-snow); } + +.grid { display: grid; gap: var(--spacing-16); } +.grid-2 { grid-template-columns: repeat(2, 1fr); } +.grid-3 { grid-template-columns: repeat(3, 1fr); } +.grid-4 { grid-template-columns: repeat(4, 1fr); } +@media (max-width: 900px) { .grid-2, .grid-3, .grid-4 { grid-template-columns: 1fr; } } + +/* --- stat blocks --- */ +.stat-n { font-size: 56px; font-weight: 600; color: var(--color-obsidian); line-height: 1; letter-spacing: -0.02em; } +.stat-n.ember { color: var(--color-ember); } +.stat-l { font-size: 14px; color: var(--color-steel); margin-top: var(--spacing-8); } +.stat-sub { font-size: 12px; color: var(--color-fog); margin-top: 4px; } + +/* --- table --- */ +table { width: 100%; border-collapse: collapse; font-size: 14px; } +th { text-align: left; font-weight: 500; color: var(--color-fog); font-size: 12px; padding: 0 var(--spacing-12) var(--spacing-12); border-bottom: 1px solid var(--color-cloud); text-transform: uppercase; letter-spacing: 0.04em; } +td { padding: var(--spacing-12); border-bottom: 1px solid var(--color-cloud); color: var(--color-graphite); } +tr:last-child td { border-bottom: none; } +.num { font-variant-numeric: tabular-nums; } +.win { font-weight: 600; color: var(--color-obsidian); } + +/* --- bars --- */ +.bar-row { display: grid; grid-template-columns: 210px 1fr 130px; gap: var(--spacing-16); align-items: center; margin-bottom: var(--spacing-12); } +.bar-track { background: var(--color-paper); border: 1px solid var(--color-cloud); border-radius: var(--radius-pills); height: 28px; overflow: hidden; position: relative; } +.bar-fill { height: 100%; border-radius: var(--radius-pills); transition: width .6s cubic-bezier(.22,.61,.36,1); } +.bar-jev { background: var(--color-ember); } +.bar-gen { background: var(--color-iron); } +.bar-val { text-align: right; font-variant-numeric: tabular-nums; font-size: 14px; color: var(--color-graphite); } +.bar-lbl { font-size: 14px; color: var(--color-steel); overflow: hidden; text-overflow: ellipsis; white-space: nowrap; } + +/* --- timeline --- */ +.tl { position: relative; height: 260px; border: 1px solid var(--color-cloud); border-radius: 24px; background: var(--color-snow); overflow: hidden; } +.tl-lane { position: absolute; height: 18px; border-radius: var(--radius-pills); } +.tl-axis { position: absolute; bottom: 0; left: 0; right: 0; height: 24px; border-top: 1px solid var(--color-cloud); background: var(--color-paper); } +.tl-tick { position: absolute; font-size: 10px; color: var(--color-fog); top: 5px; transform: translateX(-50%); } + +/* --- legend --- */ +.legend { display: flex; gap: var(--spacing-20); align-items: center; margin-bottom: var(--spacing-16); } +.legend-item { display: flex; align-items: center; gap: 8px; font-size: 13px; color: var(--color-steel); } +.swatch { width: 12px; height: 12px; border-radius: 4px; } + +/* --- log --- */ +.log { background: var(--color-obsidian); border-radius: 24px; padding: var(--spacing-20); max-height: 320px; overflow-y: auto; font-family: ui-monospace, "SF Mono", Menlo, monospace; font-size: 12px; line-height: 1.7; color: var(--color-mist); } +.log .k { color: var(--color-ember); } +.log .d { color: var(--color-ash); } + +.pill { display: inline-block; border-radius: var(--radius-pills); border: 1px solid var(--color-cloud); background: var(--color-snow); padding: 6px 14px; font-size: 13px; color: var(--color-iron); } +.note { border-left: 2px solid var(--color-ember); padding-left: var(--spacing-16); color: var(--color-steel); font-size: 14px; } + +/* --- dashboard chrome --- */ +.head { display:flex; justify-content:space-between; align-items:flex-end; gap:24px; margin-bottom:var(--spacing-32); } +.head h1 { font-size:var(--text-heading); line-height:1.1; margin:0; } +.eyebrow { display:flex; gap:8px; align-items:center; margin-bottom:var(--spacing-12); flex-wrap:wrap; } +.clock { font-size:var(--text-heading-sm); font-weight:600; color:var(--color-obsidian); font-variant-numeric:tabular-nums; } +.sec-head { display:flex; justify-content:space-between; align-items:baseline; margin-bottom:var(--spacing-16); gap:16px; } +.sec-head h2 { font-size:var(--text-subheading); margin:0; } +.section { margin-top:var(--spacing-48); } + +.kpis { display:grid; grid-template-columns:repeat(5,1fr); gap:var(--spacing-12); } +@media (max-width:1000px){ .kpis{ grid-template-columns:repeat(2,1fr);} } +.kpi { background:var(--color-snow); border:1px solid var(--color-cloud); border-radius:24px; padding:var(--spacing-20); } +.kpi .v { font-size:36px; font-weight:600; color:var(--color-obsidian); line-height:1; font-variant-numeric:tabular-nums; letter-spacing:-0.02em; } +.kpi .v.ember { color:var(--color-ember); } +.kpi .l { font-size:12px; color:var(--color-fog); margin-top:8px; text-transform:uppercase; letter-spacing:.04em; } + +/* --- document grid --- */ +.docgrid { display:grid; grid-template-columns:repeat(auto-fill,minmax(240px,1fr)); gap:var(--spacing-12); } +.doc { background:var(--color-snow); border:1px solid var(--color-cloud); border-radius:20px; padding:var(--spacing-16); } +.doc.active { border-color:var(--color-ember); box-shadow:0 0 0 1px var(--color-ember); } +.doc.done { background:var(--color-snow); } +.doc .n { font-size:13px; font-weight:500; color:var(--color-graphite); white-space:nowrap; overflow:hidden; text-overflow:ellipsis; } +.doc .m { font-size:11px; color:var(--color-fog); margin-top:4px; font-variant-numeric:tabular-nums; } +.doc .track { height:6px; background:var(--color-paper); border-radius:999px; margin-top:10px; overflow:hidden; border:1px solid var(--color-cloud); } +.doc .fill { height:100%; border-radius:999px; transition:width .5s ease; } +.doc .arms { display:flex; gap:6px; margin-top:10px; flex-wrap:wrap; } +.chip { font-size:10px; padding:2px 7px; border-radius:999px; font-variant-numeric:tabular-nums; } +.chip.jev { background:rgba(255,90,0,.12); color:#b34000; } +.chip.gen { background:var(--color-paper); color:var(--color-steel); border:1px solid var(--color-cloud); } +.chip.run { background:var(--color-ember); color:#fff; } + +@keyframes pulse { 0%,100%{opacity:1} 50%{opacity:.45} } +.pulsing { animation:pulse 1.4s ease-in-out infinite; } + +/* --- svg charts --- */ +svg.chart { width:100%; display:block; overflow:visible; } +.ax { stroke:var(--color-cloud); stroke-width:1; } +.axlbl { font-size:10px; fill:var(--color-fog); } +.slbl { font-size:11px; fill:var(--color-steel); } diff --git a/cmd/ingestbench/events.go b/cmd/ingestbench/events.go new file mode 100644 index 0000000..c9e00e2 --- /dev/null +++ b/cmd/ingestbench/events.go @@ -0,0 +1,111 @@ +package main + +import ( + "bufio" + "encoding/json" + "os" + "path/filepath" + "strings" + "sync" + "time" +) + +// events.go streams the run as JSON Lines while it happens. +// +// Written as it runs rather than summarised at the end for two reasons. +// A full-corpus ingest takes minutes, and a progress bar that only moves +// when a document finishes tells you nothing about where the time went +// inside it. And a run that dies halfway still leaves everything it +// learned on disk, which is the difference between a wasted twenty +// minutes and a partial result. +// +// One event per line, flushed immediately: a dashboard can tail the file +// and render as it arrives, with no coordination beyond the filesystem. + +type event struct { + T string `json:"t"` // event type + Time float64 `json:"time"` // seconds since run start + Doc string `json:"doc,omitempty"` // document + Arm string `json:"arm,omitempty"` // "jev" | "generative" + Phase string `json:"phase,omitempty"` // parse | detect | verify + + Pages int `json:"pages,omitempty"` + Requests int `json:"requests,omitempty"` + Seconds float64 `json:"seconds,omitempty"` + InTok int `json:"in_tokens,omitempty"` + OutTok int `json:"out_tokens,omitempty"` + CostUSD float64 `json:"cost_usd,omitempty"` + Found []int `json:"found,omitempty"` + Err string `json:"err,omitempty"` + Note string `json:"note,omitempty"` +} + +type emitter struct { + mu sync.Mutex + f *os.File + enc *json.Encoder + start time.Time +} + +func newEmitter(path string) (*emitter, error) { + f, err := os.Create(path) + if err != nil { + return nil, err + } + return &emitter{f: f, enc: json.NewEncoder(f), start: time.Now()}, nil +} + +// emit writes one event. Safe for concurrent use — documents are +// processed in parallel, which is the point of the measurement. +func (e *emitter) emit(ev event) { + e.mu.Lock() + defer e.mu.Unlock() + ev.Time = time.Since(e.start).Seconds() + _ = e.enc.Encode(ev) + _ = e.f.Sync() // a tailing dashboard should see it now, not at close +} + +func (e *emitter) close() { _ = e.f.Close() } + +func (e *emitter) elapsed() time.Duration { return time.Since(e.start) } + +// dotEnv reads one key from a gitignored .env up-tree, so credentials +// stay out of command lines and shell history. +func dotEnv(key string) string { + dir, err := os.Getwd() + if err != nil { + return "" + } + for i := 0; i < 6; i++ { + if v := readEnvFile(filepath.Join(dir, ".env"), key); v != "" { + return v + } + parent := filepath.Dir(dir) + if parent == dir { + break + } + dir = parent + } + return "" +} + +func readEnvFile(path, key string) string { + f, err := os.Open(path) + if err != nil { + return "" + } + defer f.Close() + + sc := bufio.NewScanner(f) + for sc.Scan() { + line := strings.TrimSpace(sc.Text()) + if line == "" || strings.HasPrefix(line, "#") { + continue + } + name, value, ok := strings.Cut(line, "=") + if ok && strings.TrimSpace(name) == key { + return strings.Trim(strings.TrimSpace(value), `"'`) + } + } + return "" +} diff --git a/cmd/ingestbench/main.go b/cmd/ingestbench/main.go new file mode 100644 index 0000000..d64aecd --- /dev/null +++ b/cmd/ingestbench/main.go @@ -0,0 +1,249 @@ +// Command ingestbench measures the ingestion pipeline's TOC phases on +// real 10-K filings, with and without a System One model, and streams +// the run as it happens. +// +// The claim under test is narrow and worth stating plainly: vectorless +// has been slow to ingest because every judgement it makes costs a +// round-trip to a chat model, and a long filing needs a lot of them. +// A Judge answers a batch of questions against one reading of the state, +// so the same judgements cost a couple of requests instead of dozens. +// +// Both arms run the same documents through the same parser and the same +// builder. The only difference is which model answers the yes/no +// questions, which is what makes the comparison worth anything. +// +// TYPESAFE_API_KEY=... VLE_LLM_ANTHROPIC_API_KEY=... \ +// go run ./cmd/ingestbench -docs ~/.cache/vlbench/financebench -out run.jsonl +package main + +import ( + "context" + "flag" + "fmt" + "os" + "path/filepath" + "sort" + "strings" + "sync" + "time" + + "github.com/hallelx2/llmgate" + "github.com/hallelx2/llmgate/judge/typesafe" + "github.com/hallelx2/llmgate/middleware/retry" + "github.com/hallelx2/llmgate/provider/anthropic" + + "github.com/hallelx2/vectorless-engine/pkg/ingest" + "github.com/hallelx2/vectorless-engine/pkg/parser" +) + +func main() { + docs := flag.String("docs", "", "directory of PDFs") + out := flag.String("out", "ingestbench.jsonl", "event stream output") + scan := flag.Int("scan", 20, "pages scanned for a table of contents") + par := flag.Int("parallel", 3, "documents processed concurrently") + arms := flag.String("arms", "jev,generative", "comma-separated arms to run") + flag.Parse() + + if *docs == "" { + fmt.Fprintln(os.Stderr, "usage: ingestbench -docs [-out run.jsonl]") + os.Exit(2) + } + + pdfs, err := filepath.Glob(filepath.Join(*docs, "*.pdf")) + if err != nil || len(pdfs) == 0 { + fmt.Fprintf(os.Stderr, "no PDFs in %s\n", *docs) + os.Exit(1) + } + sort.Strings(pdfs) + + em, err := newEmitter(*out) + if err != nil { + fmt.Fprintln(os.Stderr, "open output:", err) + os.Exit(1) + } + defer em.close() + + wanted := map[string]bool{} + for _, a := range strings.Split(*arms, ",") { + wanted[strings.TrimSpace(a)] = true + } + + judge, jerr := buildJudge() + needJudge := wanted["jev"] || wanted["jev-min"] || wanted["jev-fanout"] + if needJudge && jerr != nil { + fmt.Fprintln(os.Stderr, "judge:", jerr) + os.Exit(1) + } + client, cerr := buildClient() + if wanted["generative"] && cerr != nil { + fmt.Fprintf(os.Stderr, "generative arm unavailable (%v); running jev only\n", cerr) + delete(wanted, "generative") + } + + em.emit(event{T: "run_start", Note: fmt.Sprintf("%d documents, parallel=%d, scan=%d", + len(pdfs), *par, *scan)}) + fmt.Printf("documents %d · parallel %d · scan %d pages\n", len(pdfs), *par, *scan) + fmt.Printf("streaming to %s\n\n", *out) + + // Parse once per document and share it across arms. Parsing is + // identical work either way, and paying for it twice would inflate + // both arms equally while adding minutes to the run. + type parsed struct { + name string + pages []ingest.PageText + } + var docsParsed []parsed + + for _, path := range pdfs { + name := strings.TrimSuffix(filepath.Base(path), ".pdf") + start := time.Now() + pages, err := readPages(path) + dur := time.Since(start) + if err != nil { + em.emit(event{T: "phase", Doc: name, Phase: "parse", Err: err.Error()}) + fmt.Printf(" %-28s parse FAILED: %v\n", name, err) + continue + } + em.emit(event{T: "phase", Doc: name, Phase: "parse", + Pages: len(pages), Seconds: dur.Seconds()}) + fmt.Printf(" %-28s parsed %4d pages in %6s\n", name, len(pages), dur.Round(time.Millisecond)) + docsParsed = append(docsParsed, parsed{name, pages}) + } + fmt.Println() + + for _, arm := range []string{"jev", "jev-min", "jev-fanout", "generative"} { + if !wanted[arm] { + continue + } + fmt.Printf("=== arm: %s ===\n", arm) + em.emit(event{T: "arm_start", Arm: arm}) + armStart := time.Now() + + // Documents run concurrently, as the real pipeline does. A + // per-document timing alone would hide the thing that actually + // matters to a queue: how many documents finish per minute. + sem := make(chan struct{}, *par) + var wg sync.WaitGroup + + for _, d := range docsParsed { + wg.Add(1) + go func(d parsed) { + defer wg.Done() + sem <- struct{}{} + defer func() { <-sem }() + runOne(em, arm, d.name, d.pages, *scan, judge, client) + }(d) + } + wg.Wait() + + wall := time.Since(armStart) + em.emit(event{T: "arm_end", Arm: arm, Seconds: wall.Seconds()}) + fmt.Printf(" arm wall-clock: %s\n\n", wall.Round(time.Millisecond)) + } + + em.emit(event{T: "run_end", Seconds: em.elapsed().Seconds()}) + fmt.Printf("total %s — events in %s\n", em.elapsed().Round(time.Second), *out) +} + +// runOne times the detection phase for one document under one arm. +func runOne(em *emitter, arm, name string, pages []ingest.PageText, scan int, + judge llmgate.Judge, client llmgate.Client) { + + b := &ingest.TOCBuilder{} + if strings.HasPrefix(arm, "jev") { + b.Judge = judge + b.MinimalContext = arm == "jev-min" + } else { + b.LLM = client + } + + ctx, cancel := context.WithTimeout(context.Background(), 10*time.Minute) + defer cancel() + + em.emit(event{T: "phase_start", Doc: name, Arm: arm, Phase: "detect"}) + start := time.Now() + + var ( + found []int + usage ingest.Usage + handled bool + ) + switch arm { + case "jev", "jev-min": + found, usage, handled = ingest.BenchDetectTOC(ctx, b, pages, scan) + case "jev-fanout": + found, usage, handled = ingest.BenchDetectTOCFanout(ctx, b, pages, scan) + default: + found, usage = ingest.BenchDetectTOCGenerative(ctx, b, pages, scan) + handled = true + } + dur := time.Since(start) + + ev := event{ + T: "phase", Doc: name, Arm: arm, Phase: "detect", + Seconds: dur.Seconds(), Requests: usage.LLMCalls, + InTok: usage.InputTokens, OutTok: usage.OutputTokens, + CostUSD: usage.CostUSD, Found: found, + } + if !handled { + ev.Err = "judge declined; see the log line above" + } + em.emit(ev) + + fmt.Printf(" %-10s %-28s %8s %2d req %7d tok $%.6f toc=%v\n", + arm, name, dur.Round(time.Millisecond), usage.LLMCalls, + usage.InputTokens, usage.CostUSD, found) +} + +func buildJudge() (llmgate.Judge, error) { + key := os.Getenv(typesafe.EnvAPIKey) + if key == "" { + key = dotEnv(typesafe.EnvAPIKey) + } + if key == "" { + return nil, fmt.Errorf("no %s available", typesafe.EnvAPIKey) + } + j, err := typesafe.New(typesafe.Config{APIKey: key}) + if err != nil { + return nil, err + } + return retry.NewJudge(retry.Config{MaxRetries: 3})(j), nil +} + +func buildClient() (llmgate.Client, error) { + key := os.Getenv("VLE_LLM_ANTHROPIC_API_KEY") + if key == "" { + key = dotEnv("VLE_LLM_ANTHROPIC_API_KEY") + } + if key == "" { + return nil, fmt.Errorf("no VLE_LLM_ANTHROPIC_API_KEY available") + } + base := os.Getenv("VLE_LLM_ANTHROPIC_BASE_URL") + if base == "" { + base = dotEnv("VLE_LLM_ANTHROPIC_BASE_URL") + } + model := os.Getenv("VLE_LLM_ANTHROPIC_MODEL") + if model == "" { + model = dotEnv("VLE_LLM_ANTHROPIC_MODEL") + } + + c, err := anthropic.New(anthropic.Config{APIKey: key, BaseURL: base, Model: model}) + if err != nil { + return nil, err + } + return retry.New(retry.Config{MaxRetries: 3})(c), nil +} + +func readPages(path string) ([]ingest.PageText, error) { + f, err := os.Open(path) + if err != nil { + return nil, err + } + defer f.Close() + + doc, err := parser.NewPDF().Parse(context.Background(), f) + if err != nil { + return nil, err + } + return ingest.BenchAssemblePages(doc.Sections), nil +} diff --git a/cmd/tocdump/coverage.py b/cmd/tocdump/coverage.py new file mode 100644 index 0000000..b94f5cd --- /dev/null +++ b/cmd/tocdump/coverage.py @@ -0,0 +1,125 @@ +"""Evidence-page coverage: does the TOC tree put the answer where it is? + + python cmd/tocdump/coverage.py [ ...] + +FinanceBench gives every question a gold evidence page. tocdump gives every +document a section tree with page ranges. This joins them and asks, per +question: is the gold page inside some leaf section, and how big is that +leaf? A tree that covers every gold page in a 3-page leaf is doing its job; +one that covers it in a 140-page leaf has found the document, not the +answer. + +This is the accuracy gate for every context cut in HAL-1366. "The model +agreed with another model" is not a check — the other model was +rate-limited and silently empty on 13 of 21 documents. Gold pages are. + +Two directories compare before/after in one table. Coverage that drops is +a cut that gets reverted, whatever it saved. +""" + +from __future__ import annotations + +import json +import statistics +import sys +from pathlib import Path + + +def leaves(nodes, out): + for n in nodes or []: + if n.get("nodes"): + leaves(n["nodes"], out) + else: + out.append(n) + return out + + +def load_trees(d: Path) -> dict: + trees = {} + for f in sorted(d.glob("*.json")): + t = json.loads(f.read_text()) + trees[t["doc"]] = t + return trees + + +def gold(): + from datasets import load_dataset + ds = load_dataset("PatronusAI/financebench", split="train") + out = [] + for r in ds: + for ev in r.get("evidence") or []: + pg = ev.get("evidence_page_num") + if pg is None: + continue + # FinanceBench pages are 0-indexed; the tree is 1-indexed. + out.append((r["doc_name"], int(pg) + 1, r["financebench_id"])) + return out + + +def score(trees: dict, gold_pages) -> dict: + covered = hit_spans = 0 + seen = uncovered_docs = 0 + spans = [] + missing_doc = set() + for doc, page, _qid in gold_pages: + t = trees.get(doc) + if not t or t.get("err") or not t.get("nodes"): + missing_doc.add(doc) + continue + seen += 1 + # A leaf covers the page if page in [start, end]; end==0 means open. + best = None + for lf in leaves(t["nodes"], []): + s, e = lf.get("start_page", 0), lf.get("end_page", 0) + if s <= 0: + continue + if e <= 0: + e = t.get("pages", s) + if s <= page <= e: + span = e - s + 1 + if best is None or span < best: + best = span + if best is not None: + covered += 1 + spans.append(best) + return { + "questions_with_tree": seen, + "covered": covered, + "coverage": covered / seen if seen else 0.0, + "median_leaf_span": statistics.median(spans) if spans else None, + "docs_without_tree": sorted(missing_doc), + "docs_total": len(trees), + "docs_ok": sum(1 for t in trees.values() if t.get("nodes") and not t.get("err")), + "build_seconds": sum(t.get("seconds", 0) for t in trees.values()), + "build_cost": sum(t.get("cost_usd", 0) for t in trees.values()), + "build_requests": sum(t.get("requests", 0) for t in trees.values()), + } + + +def main(): + dirs = [Path(a) for a in sys.argv[1:]] + if not dirs: + print(__doc__) + return 1 + g = gold() + print(f"gold evidence pages: {len(g)} across {len({d for d,_,_ in g})} documents\n") + + rows = [(d.name, score(load_trees(d), g)) for d in dirs] + w = max(len(n) for n, _ in rows) + 2 + print(f"{'run':<{w}} {'docs ok':>8} {'q seen':>7} {'covered':>8} {'coverage':>9} {'median span':>12} {'build s':>9} {'req':>5} {'cost':>9}") + print("-" * (w + 74)) + for name, s in rows: + ms = f"{s['median_leaf_span']:.0f}p" if s["median_leaf_span"] is not None else "—" + print(f"{name:<{w}} {s['docs_ok']:>3}/{s['docs_total']:<4} {s['questions_with_tree']:>7} {s['covered']:>8} " + f"{s['coverage']:>9.3f} {ms:>12} {s['build_seconds']:>9.0f} {s['build_requests']:>5} ${s['build_cost']:>8.4f}") + for name, s in rows: + if s["docs_without_tree"]: + print(f"\n{name}: no usable tree for {len(s['docs_without_tree'])} doc(s): {', '.join(s['docs_without_tree'][:8])}") + print("\ncoverage = share of gold pages inside some leaf; median span = size of the") + print("tightest leaf that covered them. Both matter: a whole-document leaf covers") + print("everything and locates nothing.") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/cmd/tocdump/main.go b/cmd/tocdump/main.go new file mode 100644 index 0000000..b796cc3 --- /dev/null +++ b/cmd/tocdump/main.go @@ -0,0 +1,209 @@ +// Command tocdump runs the full TOC pipeline on each PDF and writes the +// resulting tree, with page ranges, as JSON — one file per document. +// +// It exists to feed the evidence-page accuracy check. FinanceBench gives a +// gold evidence page for every question; this produces the section tree +// those pages should fall inside. Join the two and you have the only +// accuracy number that matters for detection: not "did the model say what +// another model said", but "does the tree put the answer where it is". +// +// Detection and verification run on the Judge when one is configured; +// extraction is generative and runs on the chat model regardless. +// +// go run ./cmd/tocdump -docs ~/.cache/vlbench/financebench -out ~/.cache/vlbench/trees +package main + +import ( + "bufio" + "context" + "encoding/json" + "flag" + "fmt" + "os" + "path/filepath" + "strings" + "time" + + "github.com/hallelx2/llmgate" + "github.com/hallelx2/llmgate/judge/typesafe" + "github.com/hallelx2/llmgate/middleware/retry" + "github.com/hallelx2/llmgate/provider/anthropic" + + "github.com/hallelx2/vectorless-engine/pkg/ingest" + "github.com/hallelx2/vectorless-engine/pkg/parser" + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +type dump struct { + Doc string `json:"doc"` + Pages int `json:"pages"` + Seconds float64 `json:"seconds"` + Requests int `json:"requests"` + InTokens int `json:"in_tokens"` + CostUSD float64 `json:"cost_usd"` + Nodes []tree.TOCNode `json:"nodes"` + Err string `json:"err,omitempty"` +} + +func main() { + docs := flag.String("docs", "", "directory of PDFs") + out := flag.String("out", "", "directory for per-document tree JSON") + noJudge := flag.Bool("no-judge", false, "run detection/verification on the chat model instead") + minimal := flag.Bool("minimal", false, "TOCBuilder.MinimalContext: prefilter + truncation + two-stage scan") + // 300s, not the pipeline's 90s default: GLM's extraction call on the + // z.ai gateway routinely exceeds 90s on a 100+ page filing, and a + // timeout there silently drops the whole tree. Measured 2026-09-18. + callTimeout := flag.Duration("timeout", 300*time.Second, "per LLM call timeout") + flag.Parse() + if *docs == "" || *out == "" { + fmt.Fprintln(os.Stderr, "usage: tocdump -docs -out [-no-judge]") + os.Exit(2) + } + if err := os.MkdirAll(*out, 0o755); err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } + + client, err := buildClient() + if err != nil { + fmt.Fprintln(os.Stderr, "chat model (needed for extraction):", err) + os.Exit(1) + } + var judge llmgate.Judge + if !*noJudge { + if judge, err = buildJudge(); err != nil { + fmt.Fprintln(os.Stderr, "judge:", err) + os.Exit(1) + } + } + + pdfs, _ := filepath.Glob(filepath.Join(*docs, "*.pdf")) + for _, path := range pdfs { + name := strings.TrimSuffix(filepath.Base(path), ".pdf") + d := dump{Doc: name} + + pages, err := readPages(path) + if err != nil { + d.Err = "parse: " + err.Error() + write(*out, d) + fmt.Printf(" %-28s parse FAILED\n", name) + continue + } + d.Pages = len(pages) + + b := &ingest.TOCBuilder{LLM: client, Judge: judge, LLMCallTimeout: *callTimeout, MinimalContext: *minimal} + ctx, cancel := context.WithTimeout(context.Background(), 15*time.Minute) + start := time.Now() + nodes, usage, err := b.Build(ctx, pages) + cancel() + d.Seconds = time.Since(start).Seconds() + d.Requests, d.InTokens, d.CostUSD = usage.LLMCalls, usage.InputTokens, usage.CostUSD + if err != nil { + d.Err = err.Error() + } + d.Nodes = nodes + write(*out, d) + + leaves := countLeaves(nodes) + fmt.Printf(" %-28s %4d pages %6.1fs %3d req %3d leaves $%.4f %s\n", + name, d.Pages, d.Seconds, d.Requests, leaves, d.CostUSD, d.Err) + } +} + +func write(dir string, d dump) { + f, err := os.Create(filepath.Join(dir, d.Doc+".json")) + if err != nil { + return + } + defer f.Close() + enc := json.NewEncoder(f) + enc.SetIndent("", " ") + _ = enc.Encode(d) +} + +func countLeaves(ns []tree.TOCNode) int { + n := 0 + for _, x := range ns { + if len(x.Nodes) == 0 { + n++ + } else { + n += countLeaves(x.Nodes) + } + } + return n +} + +// The three helpers below duplicate cmd/ingestbench. Two bench commands +// sharing sixty lines is not yet worth an internal package; if a third +// appears, it is. + +func buildJudge() (llmgate.Judge, error) { + key := os.Getenv(typesafe.EnvAPIKey) + if key == "" { + key = dotEnv(typesafe.EnvAPIKey) + } + if key == "" { + return nil, fmt.Errorf("no %s", typesafe.EnvAPIKey) + } + j, err := typesafe.New(typesafe.Config{APIKey: key}) + if err != nil { + return nil, err + } + return retry.NewJudge(retry.Config{MaxRetries: 3})(j), nil +} + +func buildClient() (llmgate.Client, error) { + get := func(k string) string { + if v := os.Getenv(k); v != "" { + return v + } + return dotEnv(k) + } + key := get("VLE_LLM_ANTHROPIC_API_KEY") + if key == "" { + return nil, fmt.Errorf("no VLE_LLM_ANTHROPIC_API_KEY") + } + c, err := anthropic.New(anthropic.Config{ + APIKey: key, BaseURL: get("VLE_LLM_ANTHROPIC_BASE_URL"), Model: get("VLE_LLM_ANTHROPIC_MODEL"), + }) + if err != nil { + return nil, err + } + return retry.New(retry.Config{MaxRetries: 3})(c), nil +} + +func readPages(path string) ([]ingest.PageText, error) { + f, err := os.Open(path) + if err != nil { + return nil, err + } + defer f.Close() + doc, err := parser.NewPDF().Parse(context.Background(), f) + if err != nil { + return nil, err + } + return ingest.BenchAssemblePages(doc.Sections), nil +} + +func dotEnv(key string) string { + dir, _ := os.Getwd() + for i := 0; i < 6 && dir != ""; i++ { + if f, err := os.Open(filepath.Join(dir, ".env")); err == nil { + sc := bufio.NewScanner(f) + for sc.Scan() { + line := strings.TrimSpace(sc.Text()) + if k, v, ok := strings.Cut(line, "="); ok && strings.TrimSpace(k) == key { + f.Close() + return strings.Trim(strings.TrimSpace(v), `"'`) + } + } + f.Close() + } + p := filepath.Dir(dir) + if p == dir { + break + } + dir = p + } + return "" +} diff --git a/pkg/ingest/bench_export.go b/pkg/ingest/bench_export.go index 6cc58ee..63c8aad 100644 --- a/pkg/ingest/bench_export.go +++ b/pkg/ingest/bench_export.go @@ -37,3 +37,35 @@ func BenchVerifyTitles(ctx context.Context, b *TOCBuilder, nodes []tree.TOCNode, v, handled := b.verifyTitlesJudge(ctx, nodes, pages, &usage) return v, usage, handled } + +// BenchDetectTOCGenerative runs the ORIGINAL sequential detection phase. +// +// Exported purely so a benchmark can time both arms through the same +// entry points. Without it the comparison would have to reimplement the +// generative loop, and would then be measuring the reimplementation. +func BenchDetectTOCGenerative(ctx context.Context, b *TOCBuilder, pages []PageText, scan int) ([]int, Usage) { + var usage Usage + found := b.detectTOCPages(ctx, pages, scan, &usage) + return found, usage +} + +// BenchDetectTOCFanout runs the speculative fan-out phase: both +// page-level questions in one request. +// +// Reported alongside the plain Judge path so the saving from batching +// across PHASES is separable from the saving from batching across pages. +// Rolling them into one number would make it impossible to tell which +// idea earned what. +func BenchDetectTOCFanout(ctx context.Context, b *TOCBuilder, pages []PageText, scan int) ([]int, Usage, bool) { + var usage Usage + j, handled := b.judgePagesFanout(ctx, pages, scan, &usage) + if !handled { + return nil, usage, false + } + return tocPagesFrom(j, b.judgeThreshold()), usage, true +} + +// PrefilterAny exposes the skip decision for the benchmark's corpus +// validation, so the filter can be checked against every real TOC page +// before it is trusted to skip anything. +func PrefilterAny(text string) bool { return prefilterTOC(text).Any() } diff --git a/pkg/ingest/prefilter.go b/pkg/ingest/prefilter.go new file mode 100644 index 0000000..277c416 --- /dev/null +++ b/pkg/ingest/prefilter.go @@ -0,0 +1,156 @@ +package ingest + +import ( + "regexp" + "strings" +) + +// prefilter.go decides, in microseconds and without a model, whether a +// page is worth asking a model about. +// +// # Why this exists +// +// A Judge's latency is dominated by the tokens it is sent, not by the +// questions asked of them. Detection over a 20-page prefix costs about +// ten request-floors for one request, because the prefix is ~20k tokens +// of state. Most of those pages are cover sheets, legal boilerplate and +// body text that no reader would mistake for a table of contents. Sending +// them costs time and money to learn nothing. +// +// So the rule the whole pipeline is being engineered towards: every +// token sent has to earn its place. Pages with no structural sign of a +// table of contents do not get sent. +// +// # The one rule that keeps this safe +// +// A page is skipped ONLY when it shows zero signal. Not low signal — zero. +// A threshold would be a place for recall to leak out quietly, one +// unusual layout at a time, and the model is much better at the +// ambiguous cases than any heuristic. The heuristic's only job is to +// throw away the pages that are obviously not what we are looking for, +// and to be certain when it does. +// +// # What a table of contents looks like once the parser has had it +// +// Not like a table of contents. The PDF parser flattens layout, so a +// 10-K's TOC arrives as one long run: +// +// Beginning Page PART I ITEM 1 Business 4 ITEM 1A Risk Factors 10 +// ITEM 1B Unresolved Staff Comments 12 ITEM 2 Properties 12 ... +// +// No line breaks, so "lines ending in a number" — the obvious signal — +// does not exist. What survives flattening is the ENTRY shape: a short +// title followed by a small number, repeated. Prose does not do that +// more than once or twice by accident. That repeated shape is the +// generic signal; ITEM / PART markers are the 10-K-specific booster. + +var ( + // A title of a few words followed by a 1–3 digit page number. The + // lazy quantifier keeps a match from swallowing half the page and + // counting as one entry. + reTOCEntry = regexp.MustCompile(`[A-Za-z][A-Za-z ,'&/\-\.]{2,60}?\s+\d{1,3}(?:\s|$)`) + + // SEC form structure — as a TOC ENTRY, not a cross-reference. + // + // A 10-K's body says "see Item 7 of this report" on nearly every page, + // so a bare ITEM marker is no evidence of anything; the first cut of + // this filter kept 83% of body pages on that rule alone. What a + // contents page has that a body page does not is the page number + // sitting right behind the entry: "ITEM 1A Risk Factors 10". So the + // marker only counts when a small number follows within a title's + // length, and PART only counts when an ITEM entry follows it. + reItem = regexp.MustCompile(`(?i)\bITEM\s+\d{1,2}[A-C]?\b[^.;]{0,80}?\s\d{1,3}(?:\s|$)`) + rePart = regexp.MustCompile(`(?i)\bPART\s+(?:I|II|III|IV)\b\s+ITEM\s+\d`) + + // Words that name the thing itself. + reTOCWord = regexp.MustCompile(`(?i)\b(?:table\s+of\s+contents|contents|index\s+to|beginning\s+page|page\s+no\.?)\b`) +) + +// tocSignal is what the pre-filter measured on one page. +type tocSignal struct { + Entries int // title-then-number shapes, anywhere on the page + DenseRun int // the most of those found inside any denseWindow chars + Items int // ITEM markers + Parts int // PART markers + Keyword bool // names itself as a contents page + Chars int // page length, for the density reading +} + +// Any reports whether the page shows any sign at all of being a table of +// contents. This is the only thing the skip decision reads. +// +// Counting entries is not enough. Financial prose matches the shape +// three times in a paragraph without trying — "Section 4", "a term of +// 5 years", "increased 3 percent" — so a raw count of three is an +// accident, not a list. What prose never does is pack them: a contents +// page puts entry after entry with nothing between, so four of them land +// inside a few hundred characters. That density is the signal; scattered +// matches are not. +// +// Items, parts and the keyword count from one, because a body page rarely +// says "Item 1A" at all, and a page that calls itself a table of contents +// has told us what it is. +func (s tocSignal) Any() bool { + return s.DenseRun >= denseMin || s.Items >= 1 || s.Parts >= 1 || s.Keyword +} + +// denseWindow and denseMin define a "run": denseMin entries starting +// within denseWindow characters of each other. 500 chars fits five +// entries even with long titles ("Management's Discussion and Analysis +// of Financial Condition and Results of Operations 16" is ~90), and the +// prose accident rate — three in ~450 — stays under four. +const ( + denseWindow = 500 + denseMin = 4 +) + +// Score orders pages that passed Any, so a two-stage scan can send the +// likeliest first. It is NOT used to skip anything. +func (s tocSignal) Score() int { + sc := s.Entries + 3*s.Items + 2*s.Parts + if s.Keyword { + sc += 10 + } + return sc +} + +// prefilterTOC measures a page. Cheap enough to run on every page of +// every document without thinking about it. +func prefilterTOC(text string) tocSignal { + t := strings.TrimSpace(text) + if t == "" { + return tocSignal{} + } + // Bound the work: the shape we want appears early, and a 450k-char + // "page" (a known parser failure) should not cost a regex pass over + // all of it. + if len(t) > 16000 { + t = t[:16000] + } + entries := reTOCEntry.FindAllStringIndex(t, -1) + return tocSignal{ + Entries: len(entries), + DenseRun: densestRun(entries, denseWindow), + Items: len(reItem.FindAllStringIndex(t, -1)), + Parts: len(rePart.FindAllStringIndex(t, -1)), + Keyword: reTOCWord.MatchString(t), + Chars: len(text), + } +} + +// densestRun returns the largest number of matches whose start offsets +// fall within window characters of one another. Quadratic in the match +// count, which is tens at most on a real page. +func densestRun(matches [][]int, window int) int { + best := 0 + for i := range matches { + n := 0 + for j := i; j < len(matches) && matches[j][0]-matches[i][0] <= window; j++ { + n++ + } + if n > best { + best = n + } + } + return best +} diff --git a/pkg/ingest/prefilter_test.go b/pkg/ingest/prefilter_test.go new file mode 100644 index 0000000..047e20e --- /dev/null +++ b/pkg/ingest/prefilter_test.go @@ -0,0 +1,94 @@ +package ingest + +import "testing" + +// A real 10-K TOC as the parser actually delivers it: flattened, no line +// breaks. Taken from 3M_2018_10K page 2. +const tocPageFlattened = `3M COMPANY FORM 10-K For the Year Ended December 31, 2018 +Pursuant to Part IV, Item 16, a summary of Form 10-K content follows, including hyperlinked cross-references (in the EDGAR filing). Beginning Page PART I ITEM 1 Business 4 ITEM 1A Risk Factors 10 ITEM 1B Unresolved Staff Comments 12 ITEM 2 Properties 12 ITEM 3 Legal Proceedings 12 ITEM 4 Mine Safety Disclosures 12 PART II ITEM 5 Market for Registrant's Common Equity 13 ITEM 6 Selected Financial Data 15 ITEM 7 Management's Discussion and Analysis 16 ITEM 8 Financial Statements and Supplementary Data 52` + +// A generic (non-SEC) contents page — no ITEM/PART markers at all, so +// only the entry shape can save it. +const tocPageGeneric = `Contents Introduction 1 Background and Motivation 3 Related Work 7 Method 12 Experiments 19 Results 24 Discussion 31 Conclusion 35 References 37 Appendix A 41` + +// Body prose with the accidental single matches prose always has. +const prosePage = `The Company's results in 2018 reflected continued growth across its four business groups. As described in Section 4 of the agreement, the parties agreed to a term of 5 years. Revenue increased 3 percent to $32.8 billion, and operating income margins were 22.0 percent. Management believes that the underlying demand in 2019 will remain consistent with the trends observed during the fourth quarter, subject to the risks described elsewhere in this report.` + +// A cover page: dense with numbers (commission file numbers, IRS ids, +// share counts) but none of them are page numbers attached to titles. +const coverPage = `UNITED STATES SECURITIES AND EXCHANGE COMMISSION Washington, D.C. 20549 FORM 10-K ANNUAL REPORT PURSUANT TO SECTION 13 OR 15(d) Commission file number: 001-01185 GENERAL MILLS, INC. Delaware 41-0274440 Number One General Mills Boulevard Minneapolis, Minnesota 55426 (763) 764-7600 Securities registered pursuant to Section 12(b) of the Act: Common Stock, $.10 par value GIS New York Stock Exchange` + +func TestPrefilterKeepsA10KTOC(t *testing.T) { + s := prefilterTOC(tocPageFlattened) + if !s.Any() { + t.Fatalf("a real 10-K TOC page scored zero signal: %+v", s) + } + if s.Items < 5 { + t.Errorf("Items = %d, want the ITEM markers counted", s.Items) + } + if s.Parts < 2 { + t.Errorf("Parts = %d, want PART I and PART II counted", s.Parts) + } +} + +// The generic case is the one that matters for anything that is not an +// SEC filing. It has no ITEM/PART markers, so this is a test of the +// entry-shape signal alone. +func TestPrefilterKeepsAGenericContentsPage(t *testing.T) { + s := prefilterTOC(tocPageGeneric) + if !s.Any() { + t.Fatalf("a generic contents page scored zero signal: %+v", s) + } + if s.DenseRun < denseMin { + t.Errorf("DenseRun = %d, want at least %d entries packed together", s.DenseRun, denseMin) + } + if !s.Keyword { + t.Error("Keyword = false; the page says 'Contents'") + } +} + +// The property the whole filter rests on: prose must score ZERO, not +// low. A prose page that squeaks through costs tokens; a TOC page that +// gets dropped costs the document its tree. The asymmetry is why the +// bar for "signal" is set where it is. +func TestPrefilterSkipsProse(t *testing.T) { + s := prefilterTOC(prosePage) + if s.Any() { + t.Errorf("body prose showed signal and would be sent: %+v", s) + } +} + +// Cover pages are number-dense — file numbers, tax ids, phone numbers — +// and must not be mistaken for a page-numbered list. +func TestPrefilterSkipsACoverPage(t *testing.T) { + s := prefilterTOC(coverPage) + if s.Any() { + t.Errorf("a cover page showed signal and would be sent: %+v", s) + } +} + +func TestPrefilterEmptyPage(t *testing.T) { + if s := prefilterTOC(" \n "); s.Any() { + t.Errorf("empty page showed signal: %+v", s) + } +} + +// Score must rank a page that names itself above one that merely has +// the shape, so a two-stage scan sends the likeliest page first. +func TestPrefilterScoreOrdersByEvidence(t *testing.T) { + sec := prefilterTOC(tocPageFlattened).Score() + gen := prefilterTOC(tocPageGeneric).Score() + pro := prefilterTOC(prosePage).Score() + if !(sec > pro && gen > pro) { + t.Errorf("scores: sec=%d gen=%d prose=%d; both TOCs must outrank prose", sec, gen, pro) + } +} + +// Scattered accidental "word number" matches are not a run. This is the +// reason the rule is density, not a count. +func TestPrefilterOneAccidentalMatchIsNotSignal(t *testing.T) { + s := prefilterTOC("The board met in 2018 and again in March 2019 to approve the plan.") + if s.Any() { + t.Errorf("one or two accidental matches counted as signal: %+v", s) + } +} diff --git a/pkg/ingest/toc_builder.go b/pkg/ingest/toc_builder.go index af0b5ed..2a5081c 100644 --- a/pkg/ingest/toc_builder.go +++ b/pkg/ingest/toc_builder.go @@ -92,6 +92,31 @@ type TOCBuilder struct { // measured tokens rather than page count. Judge llmgate.Judge + // MinimalContext turns on the three cuts that send a Judge only what it + // needs to answer: a structural pre-filter that skips pages with no + // sign of a contents page, hard per-page truncation for detection, and + // a two-stage scan that tries the first few pages before the full + // prefix. + // + // Off by default until the evidence-page coverage gate (HAL-1366) + // shows it costs nothing — the house rule for any change that alters + // which pages a model sees. The design principle it serves: latency on + // a System One model is paid in input tokens, so every token sent has + // to earn its place. + MinimalContext bool + + // DetectChars caps the characters of each page sent to detection when + // MinimalContext is on. Zero means detectCharsMinimal. A contents page + // declares itself in its first couple of thousand characters; the + // other ten thousand are cost. + DetectChars int + + // FirstPass is how many leading pages the two-stage scan tries before + // falling back to the full prefix. Zero means firstPassDefault. Every + // 10-K puts its TOC on page 2–3; the full scan only runs on a miss, so + // the worst case costs what the single scan costs today. + FirstPass int + // JudgeThreshold is the probability above which a Noul answer counts // as yes. Zero means 0.5. // diff --git a/pkg/ingest/toc_judge.go b/pkg/ingest/toc_judge.go index 05e0af8..79deb44 100644 --- a/pkg/ingest/toc_judge.go +++ b/pkg/ingest/toc_judge.go @@ -123,12 +123,83 @@ func (b *TOCBuilder) detectTOCPagesJudgeErr(ctx context.Context, pages []PageTex return nil, true, nil // nothing to ask; a real answer, not a failure } - for _, batch := range batchByTokens(candidates, tocDetectorMaxChars) { + if !b.MinimalContext { + found, err = b.judgeTOCBatches(ctx, candidates, tocDetectorMaxChars, usage) + if err != nil { + return nil, false, err + } + return found, true, nil + } + + // Minimal-context path: three cuts, each of which sends less. + maxChars := b.DetectChars + if maxChars <= 0 { + maxChars = detectCharsMinimal + } + first := b.FirstPass + if first <= 0 { + first = firstPassDefault + } + if first > len(candidates) { + first = len(candidates) + } + + // Cut 1: the pre-filter. Pages with zero structural sign of a + // contents page are not sent. Zero, not low — see prefilter.go. + keep := func(ps []PageText) []PageText { + out := ps[:0:0] + for _, p := range ps { + if prefilterTOC(p.Text).Any() { + out = append(out, p) + } + } + return out + } + + // Cut 3: two stages. The first few pages, then the rest only on a + // miss. Cut 2 (truncation) is applied inside judgeTOCBatches. + stage1 := keep(candidates[:first]) + if len(stage1) > 0 { + found, err = b.judgeTOCBatches(ctx, stage1, maxChars, usage) + if err != nil { + return nil, false, err + } + if len(found) > 0 { + return found, true, nil + } + } + stage2 := keep(candidates[first:]) + if len(stage2) == 0 { + return nil, true, nil + } + found, err = b.judgeTOCBatches(ctx, stage2, maxChars, usage) + if err != nil { + return nil, false, err + } + return found, true, nil +} + +// Defaults for the minimal-context cuts. Both are measured, not guessed: +// 2,000 chars covers the entries a contents page opens with (3M's runs +// to ~3,800 including a prose preamble, and the entries start well +// inside 2,000), and 6 pages covers page 2–3, where 20 of 21 filings put +// their TOC on 2026-09-18. +const ( + detectCharsMinimal = 2000 + firstPassDefault = 6 +) + +// judgeTOCBatches asks the detection question of every page in pages, +// truncating each to maxChars, in as few requests as the token budget +// allows. It returns the page numbers judged to be a table of contents. +func (b *TOCBuilder) judgeTOCBatches(ctx context.Context, pages []PageText, maxChars int, usage *Usage) ([]int, error) { + var found []int + for _, batch := range batchByTokens(pages, maxChars) { state := map[string]any{} questions := map[string]llmgate.Question{} for _, p := range batch { key := pageKey(p.PageNumber) - state[key] = truncate(p.Text, tocDetectorMaxChars) + state[key] = truncate(p.Text, maxChars) questions[key] = llmgate.Noul{ // The question names the field it is about. Question IDs // are not sent to the model, so without this the model @@ -148,7 +219,7 @@ func (b *TOCBuilder) detectTOCPagesJudgeErr(ctx context.Context, pages []PageTex // Partial results would silently truncate the scanned range // and look like "no TOC here", so abandon the whole phase // and let the generative path redo it properly. - return nil, false, jerr + return nil, jerr } addJudgeUsage(usage, res) @@ -162,9 +233,8 @@ func (b *TOCBuilder) detectTOCPagesJudgeErr(ctx context.Context, pages []PageTex } } } - sort.Ints(found) - return found, true, nil + return found, nil } // verifyTitlesJudge answers "does this section start at the top of this diff --git a/pkg/ingest/toc_judge_fanout.go b/pkg/ingest/toc_judge_fanout.go new file mode 100644 index 0000000..e4d3412 --- /dev/null +++ b/pkg/ingest/toc_judge_fanout.go @@ -0,0 +1,157 @@ +package ingest + +import ( + "context" + "fmt" + "log" + "sort" + "strings" + + "github.com/hallelx2/llmgate" +) + +// toc_judge_fanout.go asks every question the TOC phases need about a +// page in ONE request, including the ones that may turn out to be +// irrelevant. +// +// # Why this is not just batching again +// +// toc_judge.go already batches detection across pages. This goes +// further and batches across PHASES, which is a different saving. +// +// The pipeline naturally reads as a sequence: detect where the table of +// contents is, extract it, then verify each claimed start page. The +// middle step genuinely needs the first — you cannot verify a section +// you have not extracted. But the THIRD step's question, asked of a +// page, does not actually depend on the first: "does a section heading +// begin at the top of this page" is answerable from the page alone. +// +// So we ask it up front, for every page in the scanned prefix, at the +// same time as the detection question — and throw away the answers for +// pages that turn out not to matter. TypeSafe's docs are explicit that +// questions in one request are evaluated in parallel and that "adding +// more questions to a call typically doesn't add any latency". The +// speculative answers are therefore close to free, and they remove a +// whole round-trip from the critical path. +// +// # The cost model this exploits +// +// Measured against jev-1.13.0: a minimal one-question call has a floor +// of roughly 800ms, and twenty questions over shared state cost 2.82x +// the wall-clock of one — not 20x. Latency is dominated by the number of +// REQUESTS, not the number of questions. Every question moved into an +// existing request is nearly free; every new request costs the floor. +// +// That is the whole optimisation, and it is why speculation pays here +// when it would not against a chat model, where a discarded answer is a +// discarded call. + +// pageJudgement is everything asked about one page in the fan-out. +type pageJudgement struct { + IsTOC float64 // probability this page is a table of contents + SectionStarts float64 // probability a section heading opens this page + Asked bool +} + +// sectionStartInstructions is the speculative question. +// +// Phrased about the page rather than about a named title, because at +// fan-out time we do not yet know which titles exist — extraction has +// not run. That makes it weaker evidence than the targeted verification +// question, so it is used as a prior to skip work, never to overrule a +// real verification. +const sectionStartInstructions = "Does a numbered or titled section heading " + + "begin at the very start of this page, before any body text?" + +// judgePagesFanout answers both page-level questions in one pass. +// +// Returns handled=false on any failure, leaving the caller to run the +// ordinary sequential path — the same contract as every other Judge +// entry point here. +func (b *TOCBuilder) judgePagesFanout(ctx context.Context, pages []PageText, limit int, usage *Usage) (map[int]pageJudgement, bool) { + out, handled, err := b.judgePagesFanoutErr(ctx, pages, limit, usage) + if err != nil { + log.Printf("toc: judge fan-out failed, falling back: %v", err) + } + return out, handled +} + +func (b *TOCBuilder) judgePagesFanoutErr(ctx context.Context, pages []PageText, limit int, usage *Usage) (map[int]pageJudgement, bool, error) { + if b.Judge == nil { + return nil, false, nil + } + if limit > len(pages) { + limit = len(pages) + } + + candidates := make([]PageText, 0, limit) + for i := 0; i < limit; i++ { + if strings.TrimSpace(pages[i].Text) != "" { + candidates = append(candidates, pages[i]) + } + } + if len(candidates) == 0 { + return map[int]pageJudgement{}, true, nil + } + + out := make(map[int]pageJudgement, len(candidates)) + + for _, batch := range batchByTokens(candidates, tocDetectorMaxChars) { + state := map[string]any{} + questions := map[string]llmgate.Question{} + + for _, p := range batch { + key := pageKey(p.PageNumber) + state[key] = truncate(p.Text, tocDetectorMaxChars) + + questions["toc_"+key] = llmgate.Noul{ + Instructions: fmt.Sprintf("%s Answer only about the text in `%s`.", + tocDetectInstructions, key), + Criteria: tocDetectCriteria(), + } + // The speculative one. Most pages are not section openings and + // most of these answers go unused — which is the point: they + // ride along in a request that was being sent anyway. + questions["sec_"+key] = llmgate.Noul{ + Instructions: fmt.Sprintf("%s Answer only about the text in `%s`.", + sectionStartInstructions, key), + Criteria: &llmgate.NoulCriteria{ + True: "A section or item heading is the first meaningful content", + False: "The page opens with body text, a table, or a continuation", + }, + } + } + + res, err := b.Judge.Judge(ctx, llmgate.JudgeRequest{State: state, Questions: questions}) + if err != nil { + return nil, false, err + } + addJudgeUsage(usage, res) + + for _, p := range batch { + key := pageKey(p.PageNumber) + j := pageJudgement{Asked: true} + if v, err := res.Noul("toc_" + key); err == nil { + j.IsTOC = v + } + if v, err := res.Noul("sec_" + key); err == nil { + j.SectionStarts = v + } + out[p.PageNumber] = j + } + } + + return out, true, nil +} + +// tocPagesFrom reads the detection answer out of a fan-out result. +func tocPagesFrom(j map[int]pageJudgement, threshold float64) []int { + var found []int + for page, pj := range j { + if pj.Asked && pj.IsTOC > threshold { + found = append(found, page) + } + } + sort.Ints(found) + return found +} From d1f8cf3da5cf4ded720a5d4cbc185c6b2b26e64c Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 07:22:13 +0100 Subject: [PATCH 2/7] =?UTF-8?q?feat(ingest):=20resolve=20every=20leaf's=20?= =?UTF-8?q?page=20on=20a=20Judge=20=E2=80=94=20select,=20don't=20generate?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extraction lost the page of every leaf in a 10-K (HAL-1367): the generative call only saw the first 48k characters of the body, so nothing past page 10 could be placed. Rather than widen that window, this stops asking a generative model for page numbers at all. Go finds the candidates: each leaf title is searched for as a heading (opening its line, or preceded only by a part label — the parser joins "PART I" to the item that opens it) on every page, and the contents pages detection found are excluded. The Judge then answers one Noul per candidate on a 600-character window around the hit, batched under the shared-state budget. The best probability above threshold wins. Measured on ADOBE_2022_10K, 88 pages, 24 leaves: before 0/24 leaves with a page head-of-page window 17/24 8.7s heading search 20/24 4.6s + part-label prefix 24/24 5.1s 2 requests 10,786 tokens $0.00045 Every resolved page agrees with the filing's own contents page. deriveEndPages now lets a container with no page of its own inherit its first child's, so four items sharing page 34 end at 34 instead of running to the last page of the document. Build wires the resolver ahead of title verification. cmd/tocresolve re-resolves an existing tree on a Judge alone so the resolver can be iterated without paying for extraction; -v prints each leaf's candidate set. cmd/tocdump gains -timeout and a longer retry schedule. --- cmd/ingestbench/dashboard/app.js | 29 ++- cmd/tocdump/main.go | 11 +- cmd/tocresolve/main.go | 202 +++++++++++++++ pkg/ingest/bench_export.go | 55 ++++ pkg/ingest/toc_builder.go | 34 ++- pkg/ingest/toc_builder_test.go | 34 +++ pkg/ingest/toc_resolve.go | 424 +++++++++++++++++++++++++++++++ pkg/ingest/toc_resolve_test.go | 264 +++++++++++++++++++ 8 files changed, 1040 insertions(+), 13 deletions(-) create mode 100644 cmd/tocresolve/main.go create mode 100644 pkg/ingest/toc_resolve.go create mode 100644 pkg/ingest/toc_resolve_test.go diff --git a/cmd/ingestbench/dashboard/app.js b/cmd/ingestbench/dashboard/app.js index 3f01c5a..905ecdb 100644 --- a/cmd/ingestbench/dashboard/app.js +++ b/cmd/ingestbench/dashboard/app.js @@ -6,9 +6,14 @@ const POLL_MS = 1000; const ARM = { - jev: { label: 'Jev', color: '#ff5a00', chip: 'jev' }, - generative: { label: 'GLM', color: '#3f3f46', chip: 'gen' }, + jev: { label: 'Jev', color: '#ff5a00', chip: 'jev' }, + 'jev-min': { label: 'Jev min', color: '#ff9a5c', chip: 'jev' }, + 'jev-fanout': { label: 'Jev fan', color: '#c94800', chip: 'jev' }, + generative: { label: 'GLM', color: '#3f3f46', chip: 'gen' }, }; +// Any arm not listed still renders, in a neutral colour, rather than +// throwing on arm(x).color and blanking the whole chart. +const arm = a => ARM[a] || { label: a, color: '#a1a1aa', chip: 'gen' }; let events = [], lastLen = -1; @@ -87,14 +92,15 @@ function docgrid() { document.getElementById('docgrid').innerHTML = pr.map(p => { const d = detects().filter(x => x.doc === p.doc); const live = [...running].some(k => k.startsWith(p.doc + '|')); - const chips = ['jev', 'generative'].map(a => { + const armsSeen = [...new Set(detects().concat(starts()).map(e => e.arm))]; + const chips = armsSeen.map(a => { const e = d.find(x => x.arm === a); - if (e) return `${ARM[a].label} ${e.seconds.toFixed(1)}s · ${e.requests}r`; - if (running.has(p.doc + '|' + a)) return `${ARM[a].label} running`; + if (e) return `${arm(a).label} ${e.seconds.toFixed(1)}s · ${e.requests}r`; + if (running.has(p.doc + '|' + a)) return `${arm(a).label} running`; return ''; }).join(''); - const pct = (d.length / 2) * 100; + const pct = armsSeen.length ? (d.length / armsSeen.length) * 100 : 0; return `
${p.doc.replace(/_/g, ' ')}
${p.pages} pages · parsed ${p.seconds.toFixed(1)}s
@@ -162,12 +168,13 @@ function charts() { const docs = [...new Set(detects().map(e => e.doc))].sort(); const lat = [], req = []; - for (const d of docs) for (const a of ['jev', 'generative']) { + const armsSeen = [...new Set(detects().map(e => e.arm))]; + for (const d of docs) for (const a of armsSeen) { const e = detects().find(x => x.doc === d && x.arm === a); if (!e) continue; const short = d.replace(/_/g, ' ').replace(/ 10K$/, ''); - lat.push({ label: `${short} · ${ARM[a].label}`, v: e.seconds, t: fmtS(e.seconds), color: ARM[a].color }); - req.push({ label: `${short} · ${ARM[a].label}`, v: e.requests, t: `${e.requests}`, color: ARM[a].color }); + lat.push({ label: `${short} · ${arm(a).label}`, v: e.seconds, t: fmtS(e.seconds), color: arm(a).color }); + req.push({ label: `${short} · ${arm(a).label}`, v: e.requests, t: `${e.requests}`, color: arm(a).color }); } hbars('chart-latency', lat, 'detection phase only; parsing is shared and excluded'); hbars('chart-requests', req, 'one batched request vs one call per scanned page'); @@ -197,7 +204,7 @@ function totals() { const e = byArm(a); if (!e.length) return ''; const c = sum(e, x => x.cost_usd); return ` - ${ARM[a].label} + ${arm(a).label} ${e.length} ${sum(e, x => x.requests)} ${fmtN(sum(e, x => x.in_tokens))} @@ -230,7 +237,7 @@ function timeline() { lanes[l] = it.time; bars += `
`; + top:${10 + l * 22}px;background:${arm(it.arm).color}">
`; } let ticks = ''; for (let i = 0; i <= 4; i++) ticks += `
${fmtS((i / 4) * maxT)}
`; diff --git a/cmd/tocdump/main.go b/cmd/tocdump/main.go index b796cc3..61f4796 100644 --- a/cmd/tocdump/main.go +++ b/cmd/tocdump/main.go @@ -169,7 +169,16 @@ func buildClient() (llmgate.Client, error) { if err != nil { return nil, err } - return retry.New(retry.Config{MaxRetries: 3})(c), nil + // A per-minute rate limit needs backoff measured in tens of seconds, + // not the default 500ms-doubling that tops out around 3.5s total. + // Three retries at that pace against z.ai's [1302] "Rate limit reached + // for requests" burned a whole 21-document run to 0-leaf trees on + // 2026-09-18. Six retries from 5s, capped at 60s, rides out a window. + return retry.New(retry.Config{ + MaxRetries: 6, + BaseDelay: 5 * time.Second, + MaxDelay: 60 * time.Second, + })(c), nil } func readPages(path string) ([]ingest.PageText, error) { diff --git a/cmd/tocresolve/main.go b/cmd/tocresolve/main.go new file mode 100644 index 0000000..2b22a90 --- /dev/null +++ b/cmd/tocresolve/main.go @@ -0,0 +1,202 @@ +// Command tocresolve re-resolves the pages of an already-extracted tree, +// on a Judge alone. +// +// It exists to validate the resolver (HAL-1367) against a real filing +// without re-running the 150-second generative extraction: take the tree +// tocdump produced, throw away its page numbers, and see how many the +// resolver gets back — and where it puts them. +// +// go run ./cmd/tocresolve -pdf doc.pdf -tree trees-after/doc.json -out trees-resolved/ +package main + +import ( + "bufio" + "context" + "encoding/json" + "flag" + "fmt" + "os" + "path/filepath" + "strings" + "time" + + "github.com/hallelx2/llmgate/judge/typesafe" + "github.com/hallelx2/llmgate/middleware/retry" + + "github.com/hallelx2/vectorless-engine/pkg/ingest" + "github.com/hallelx2/vectorless-engine/pkg/parser" + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +type dump struct { + Doc string `json:"doc"` + Pages int `json:"pages"` + Seconds float64 `json:"seconds"` + Requests int `json:"requests"` + InTokens int `json:"in_tokens"` + CostUSD float64 `json:"cost_usd"` + Nodes []tree.TOCNode `json:"nodes"` + Err string `json:"err,omitempty"` +} + +func main() { + pdf := flag.String("pdf", "", "the filing") + treePath := flag.String("tree", "", "tocdump output for it") + out := flag.String("out", "", "directory to write the resolved tree") + verbose := flag.Bool("v", false, "print each leaf's candidate pages before asking") + flag.Parse() + if *pdf == "" || *treePath == "" || *out == "" { + fmt.Fprintln(os.Stderr, "usage: tocresolve -pdf f.pdf -tree t.json -out dir") + os.Exit(2) + } + + raw, err := os.ReadFile(*treePath) + if err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } + var d dump + if err := json.Unmarshal(raw, &d); err != nil { + fmt.Fprintln(os.Stderr, "tree:", err) + os.Exit(1) + } + + f, err := os.Open(*pdf) + if err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } + doc, err := parser.NewPDF().Parse(context.Background(), f) + f.Close() + if err != nil { + fmt.Fprintln(os.Stderr, "parse:", err) + os.Exit(1) + } + pages := ingest.BenchAssemblePages(doc.Sections) + + key := os.Getenv(typesafe.EnvAPIKey) + if key == "" { + key = dotEnv(typesafe.EnvAPIKey) + } + tj, err := typesafe.New(typesafe.Config{APIKey: key}) + if err != nil { + fmt.Fprintln(os.Stderr, "judge:", err) + os.Exit(1) + } + b := &ingest.TOCBuilder{Judge: retry.NewJudge(retry.Config{MaxRetries: 3})(tj)} + + before := countWithPage(d.Nodes) + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute) + defer cancel() + start := time.Now() + // Exclude the pages detection actually calls a table of contents — + // exactly what Build does. A structural guess was tried here first + // and excluded 30 of 88 pages on a 10-K, because every page carries + // "Table of Contents" as a running header. Decisive exclusions come + // from the model's answer, never from a heuristic. + b.MinimalContext = true + exclude, du, ok := ingest.BenchDetectTOC(ctx, b, pages, 20) + if !ok { + fmt.Fprintln(os.Stderr, "detection declined; resolving with no exclusions") + exclude = nil + } + if *verbose { + for _, c := range ingest.BenchCandidates(d.Nodes, pages, exclude) { + t := c.Title + if len(t) > 40 { + t = t[:40] + } + fmt.Fprintf(os.Stderr, " cand %-40s claimed=%-3d pages=%v\n", t, c.Claimed, c.Pages) + } + } + usage, handled := ingest.BenchResolvePages(ctx, b, d.Nodes, pages, exclude) + usage.LLMCalls += du.LLMCalls + usage.InputTokens += du.InputTokens + usage.CostUSD += du.CostUSD + elapsed := time.Since(start) + if !handled { + fmt.Fprintln(os.Stderr, "resolver declined") + os.Exit(1) + } + ingest.BenchFinalise(d.Nodes, pages) + after := countWithPage(d.Nodes) + + fmt.Printf("%s: %d pages, %d leaves, contents pages excluded: %v\n", d.Doc, len(pages), countLeaves(d.Nodes), exclude) + fmt.Printf(" leaves with a page %d -> %d\n", before, after) + fmt.Printf(" judge requests %d\n", usage.LLMCalls) + fmt.Printf(" tokens %d\n", usage.InputTokens) + fmt.Printf(" elapsed %s\n", elapsed.Round(time.Millisecond)) + fmt.Printf(" cost $%.6f\n\n", usage.CostUSD) + printLeaves(d.Nodes, 0) + + d.Seconds, d.Requests, d.InTokens, d.CostUSD = elapsed.Seconds(), usage.LLMCalls, usage.InputTokens, usage.CostUSD + d.Err = "" + _ = os.MkdirAll(*out, 0o755) + of, err := os.Create(filepath.Join(*out, d.Doc+".json")) + if err == nil { + enc := json.NewEncoder(of) + enc.SetIndent("", " ") + _ = enc.Encode(d) + of.Close() + } +} + +func countWithPage(ns []tree.TOCNode) int { + n := 0 + for _, x := range ns { + if len(x.Nodes) > 0 { + n += countWithPage(x.Nodes) + } else if x.StartPage > 0 { + n++ + } + } + return n +} + +func countLeaves(ns []tree.TOCNode) int { + n := 0 + for _, x := range ns { + if len(x.Nodes) == 0 { + n++ + } else { + n += countLeaves(x.Nodes) + } + } + return n +} + +func printLeaves(ns []tree.TOCNode, depth int) { + for _, x := range ns { + if len(x.Nodes) > 0 { + printLeaves(x.Nodes, depth+1) + continue + } + t := x.Title + if len(t) > 46 { + t = t[:46] + } + fmt.Printf(" %sp%3d-%-3d %s\n", strings.Repeat(" ", depth), x.StartPage, x.EndPage, t) + } +} + +func dotEnv(key string) string { + dir, _ := os.Getwd() + for i := 0; i < 6 && dir != ""; i++ { + if f, err := os.Open(filepath.Join(dir, ".env")); err == nil { + sc := bufio.NewScanner(f) + for sc.Scan() { + if k, v, ok := strings.Cut(strings.TrimSpace(sc.Text()), "="); ok && strings.TrimSpace(k) == key { + f.Close() + return strings.Trim(strings.TrimSpace(v), `"'`) + } + } + f.Close() + } + if p := filepath.Dir(dir); p != dir { + dir = p + } else { + break + } + } + return "" +} diff --git a/pkg/ingest/bench_export.go b/pkg/ingest/bench_export.go index 63c8aad..9d95b60 100644 --- a/pkg/ingest/bench_export.go +++ b/pkg/ingest/bench_export.go @@ -69,3 +69,58 @@ func BenchDetectTOCFanout(ctx context.Context, b *TOCBuilder, pages []PageText, // validation, so the filter can be checked against every real TOC page // before it is trusted to skip anything. func PrefilterAny(text string) bool { return prefilterTOC(text).Any() } + +// BenchResolvePages runs Judge-backed page resolution alone, over an +// already-extracted tree. Exists so the resolver can be validated on a +// real document without paying for extraction again. +func BenchResolvePages(ctx context.Context, b *TOCBuilder, nodes []tree.TOCNode, pages []PageText, exclude []int) (Usage, bool) { + var usage Usage + resolved, handled := b.resolvePagesJudge(ctx, nodes, pages, exclude, &usage) + if handled { + applyResolvedPages(nodes, resolved) + } + return usage, handled +} + +// BenchFinalise derives end pages and stamps IDs, as the tail of Build +// does, so a resolved tree is comparable to a built one. +func BenchFinalise(nodes []tree.TOCNode, pages []PageText) { + deriveEndPages(nodes, lastPage(pages)) + stampNodeIDs(nodes, "") +} + +// BenchLikelyTOCPages returns pages the structural pre-filter rates as +// contents pages, for a caller that has a tree but did not run detection. +// The bar is high on purpose: a wrongly-excluded body page loses one +// candidate; a wrongly-included contents page wins every question. +func BenchLikelyTOCPages(pages []PageText) []int { + var out []int + for _, p := range pages { + if s := prefilterTOC(p.Text); s.Keyword && (s.Items >= 3 || s.Entries >= 5) { + out = append(out, p.PageNumber) + } + } + return out +} + +// BenchCandidate is one leaf's candidate set, for a command that wants to +// show why a leaf did or did not resolve. +type BenchCandidate struct { + Title string + Claimed int + Pages []int +} + +// BenchCandidates returns what the resolver would ask about, without +// asking. Pure code: no Judge, no cost. +func BenchCandidates(nodes []tree.TOCNode, pages []PageText, exclude []int) []BenchCandidate { + var out []BenchCandidate + for _, c := range collectResolveClaims(nodes, pages, exclude) { + bc := BenchCandidate{Title: c.title, Claimed: c.claimed} + for _, h := range c.candidates { + bc.Pages = append(bc.Pages, h.page) + } + out = append(out, bc) + } + return out +} diff --git a/pkg/ingest/toc_builder.go b/pkg/ingest/toc_builder.go index 2a5081c..1eab32a 100644 --- a/pkg/ingest/toc_builder.go +++ b/pkg/ingest/toc_builder.go @@ -204,7 +204,15 @@ func (b *TOCBuilder) Build(ctx context.Context, pages []PageText) ([]tree.TOCNod // starts the section. Mismatches clear the page (set to 0) // rather than making one up — downstream treats zero as // open/unknown. - if verdicts, handled := b.verifyTitlesJudge(ctx, nodes, pages, &usage); handled { + // With a Judge, resolution subsumes verification: it searches every + // page head for each title and asks whether the title BEGINS the page, + // with extraction's guess as one candidate among the hits. That is + // verification with search, and it is what makes the extraction body + // window irrelevant to page accuracy (HAL-1367). Plain verification + // remains the fallback when resolution cannot run. + if resolved, handled := b.resolvePagesJudge(ctx, nodes, pages, tocPages, &usage); handled { + applyResolvedPages(nodes, resolved) + } else if verdicts, handled := b.verifyTitlesJudge(ctx, nodes, pages, &usage); handled { applyJudgeVerdicts(nodes, verdicts) } else { b.verifyTitlesConcurrent(ctx, nodes, pages, concurrency, &usage) @@ -857,9 +865,33 @@ func flattenForVerify(nodes []tree.TOCNode) []*tree.TOCNode { // end pages cap at their parent's, which is what readers expect // for a TOC. func deriveEndPages(nodes []tree.TOCNode, docLastPage int) { + inheritParentStarts(nodes) deriveEndPagesIn(nodes, docLastPage) } +// inheritParentStarts gives a container with no page of its own — a +// "PART II" whose only content is its items — the first page any of its +// children start on. Without it the container has no EndPage, and every +// child in the previous part that shares a start page with a sibling +// falls through to the document's last page as its end (HAL-1367). +func inheritParentStarts(nodes []tree.TOCNode) { + for i := range nodes { + n := &nodes[i] + if len(n.Nodes) == 0 { + continue + } + inheritParentStarts(n.Nodes) + if n.StartPage > 0 { + continue + } + for _, c := range n.Nodes { + if c.StartPage > 0 && (n.StartPage == 0 || c.StartPage < n.StartPage) { + n.StartPage = c.StartPage + } + } + } +} + func deriveEndPagesIn(nodes []tree.TOCNode, ceiling int) { for i := range nodes { n := &nodes[i] diff --git a/pkg/ingest/toc_builder_test.go b/pkg/ingest/toc_builder_test.go index 4dfe22a..ec4ef0e 100644 --- a/pkg/ingest/toc_builder_test.go +++ b/pkg/ingest/toc_builder_test.go @@ -311,6 +311,40 @@ func TestEndPageDerivationFromSiblings(t *testing.T) { } } +// A 10-K's parts carry no page of their own. Items 1B, 2, 3 and 4 all +// start on page 34 and the next sibling with a greater page is in the +// NEXT part, so without the part inheriting a start from its children +// every one of them ran to the last page of the document. +func TestEndPageDerivationAcrossPartsSharingAPage(t *testing.T) { + root := []tree.TOCNode{ + {Structure: "1", Title: "PART I", StartPage: 3, Nodes: []tree.TOCNode{ + {Structure: "1.1", Title: "Item 1", StartPage: 3}, + {Structure: "1.2", Title: "Item 1B", StartPage: 34}, + {Structure: "1.3", Title: "Item 2", StartPage: 34}, + {Structure: "1.4", Title: "Item 4", StartPage: 34}, + }}, + {Structure: "2", Title: "PART II", Nodes: []tree.TOCNode{ + {Structure: "2.1", Title: "Item 5", StartPage: 35}, + {Structure: "2.2", Title: "Item 7", StartPage: 36}, + }}, + } + deriveEndPages(root, 99) + if root[1].StartPage != 35 { + t.Errorf("PART II should inherit its first child's page: got %d", root[1].StartPage) + } + if root[0].EndPage != 34 { + t.Errorf("PART I.EndPage: got %d want 34", root[0].EndPage) + } + for _, c := range root[0].Nodes[1:] { + if c.EndPage != 34 { + t.Errorf("%s.EndPage: got %d want 34", c.Title, c.EndPage) + } + } + if root[1].Nodes[1].EndPage != 99 { + t.Errorf("last item should run to the document end: got %d", root[1].Nodes[1].EndPage) + } +} + // TestAssembleHierarchyNestsByStructure makes sure dotted // structure indices group correctly. "1.1" nests under "1", // "2.1.1" three levels deep, etc. diff --git a/pkg/ingest/toc_resolve.go b/pkg/ingest/toc_resolve.go new file mode 100644 index 0000000..fa60e56 --- /dev/null +++ b/pkg/ingest/toc_resolve.go @@ -0,0 +1,424 @@ +package ingest + +import ( + "context" + "fmt" + "log" + "regexp" + "sort" + "strings" + + "github.com/hallelx2/llmgate" + + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +// toc_resolve.go locates where each section actually begins, by searching +// for its title and letting a Judge confirm the hit. +// +// # The problem this replaces +// +// A table of contents says "Item 1A. Risk Factors ... 20". That 20 is the +// document's own printed page number; the PDF has a cover and a contents +// page in front of printed page 1, so the section is at PDF page 22 or +// so. Extraction asks a chat model to resolve this by reading the body — +// but the body it is shown is capped at ~48k characters, five pages of a +// five-hundred-page filing, so for anything deeper it guesses from the +// printed number. Verification then looks at the wrong page and, rightly, +// says no. Every leaf past page five loses its page. Every long filing. +// +// # Select, don't generate +// +// Finding a title is a lookup with a small answer set, so the candidates +// come from code: scan every page's head for the title. That usually +// yields one to three pages — the real opening, the contents page itself, +// and now and then a body page that mentions the section in passing. +// +// Telling those apart is the judgement: is the title a HEADING on this +// page, or merely mentioned on it? A Judge answers that for every (leaf, +// candidate) pair in one request. Code takes the best answer above +// threshold. +// +// The question is deliberately "on this page", not "at the very start of +// it". Filings pack several short sections onto one page — Items 1B, 2, +// 3 and 4 of a 10-K routinely share a page — and only the first of them +// opens the page. The others still start there. Asking for "the very +// start" threw away half of ADOBE's leaves for being second on their +// page, which is a fact about layout, not about where the section is. +// +// The contents page contains every title as a heading-shaped entry and +// would win every question. It is excluded in code: detection already +// knows which page it is, and certainty beats asking. +// +// This is verification with search rather than verification of a guess. +// It makes the extraction body window irrelevant to page accuracy, and it +// works for any document, not only ones with a numbered contents page. + +// Window around a located heading that is shown to the Judge: a little +// before, so a cross-reference's sentence is visible as such, and enough +// after to see a section actually begin. This replaces sending the page +// head, which missed every section that was third or later on a shared +// page — Items 3 and 4 of a 10-K after 1B and 2, 9B and 9C after 9 and +// 9A. Go finds the exact spot; the model only has to confirm it. +const ( + excerptBefore = 150 + excerptAfter = 450 + + // headChars is retained for the test that proves a deep hit lies + // beyond any head-sized window. + headChars = 1200 + + // scanChars bounds the per-page search so a 450k-char "page" (a known + // parser failure, HAL-1365) does not cost a regex pass over all of it. + scanChars = 60000 +) + +// maxCandidatesPerLeaf bounds the Judge batch. A title that appears at +// the head of more than this many pages is a running header, not a +// section, and no amount of asking will resolve it. +const maxCandidatesPerLeaf = 4 + +var ( + reNonWord = regexp.MustCompile(`[^a-z0-9]+`) +) + +// normalise folds a title or page head into a form where "Item 1A. Risk +// Factors" and "ITEM 1A — RISK FACTORS" compare equal. +func normalise(s string) string { + s = strings.ToLower(s) + s = reNonWord.ReplaceAllString(s, " ") + return strings.TrimSpace(s) +} + +// pageHit is one place a title was found: the page, and where on it. +type pageHit struct { + page int + offset int // byte offset of the match in the page text +} + +// titleRegexp builds a matcher for a title that tolerates the punctuation +// and spacing differences between a contents entry and a body heading: +// "Item 1A. Risk Factors" must find "ITEM 1A — RISK FACTORS". +func titleRegexp(title string) *regexp.Regexp { + words := strings.Fields(normalise(title)) + if len(words) == 0 { + return nil + } + // Long titles are often wrapped or abbreviated on their own page; + // match on the first several words. + if len(words) > 6 { + words = words[:6] + } + parts := make([]string, len(words)) + for i, w := range words { + parts[i] = regexp.QuoteMeta(w) + } + return regexp.MustCompile(`(?i)` + strings.Join(parts, `[^a-z0-9]+`)) +} + +// findCandidatePages returns where the title appears as a HEADING — +// opening a line — on any page, in page order. A mention inside a line +// is a cross-reference, not a section, and is not a candidate. +// +// There is deliberately no fallback to "the title appears somewhere in +// the page head": that admitted every "see Item 1A" cross-reference as a +// candidate and cost a question each. Measured on ADOBE_2022_10K the +// heading-only rule lost none of the pages the head rule found. +func findCandidatePages(title string, pages []PageText) []pageHit { + if len(normalise(title)) < 8 { + return nil + } + re := titleRegexp(title) + if re == nil { + return nil + } + + var out []pageHit + for _, p := range pages { + if p.PageNumber <= 0 { + continue + } + text := p.Text + if len(text) > scanChars { + text = text[:scanChars] + } + for _, m := range re.FindAllStringIndex(text, -1) { + if isLineStart(text, m[0]) { + out = append(out, pageHit{page: p.PageNumber, offset: m[0]}) + break + } + } + } + sort.Slice(out, func(i, j int) bool { return out[i].page < out[j].page }) + return out +} + +// rePartLabel is the one thing allowed before a heading on its line: the +// parser routinely joins a part label to the item that opens it — +// "PART I ITEM 1. BUSINESS" — and every first-in-part section of a 10-K +// was lost to that until this was measured on ADOBE_2022_10K. +var rePartLabel = regexp.MustCompile(`(?i)^part\s+[ivxlc\d]+\s*[.:\-—–]?$`) + +// isLineStart reports whether the match at off opens its line — what a +// heading looks like once the parser has had it, and what a mid-sentence +// cross-reference does not. A part label alone before it still counts. +func isLineStart(text string, off int) bool { + lo := off + for lo > 0 && text[lo-1] != '\n' { + lo-- + } + prefix := strings.TrimSpace(text[lo:off]) + return prefix == "" || rePartLabel.MatchString(prefix) +} + +// excerptAround cuts the window the Judge is shown. +func excerptAround(text string, off int) string { + lo := off - excerptBefore + if lo < 0 { + lo = 0 + } + hi := off + excerptAfter + if hi > len(text) { + hi = len(text) + } + return text[lo:hi] +} + +// resolvedLeaf is one leaf and the pages that might open it. +type resolvedLeaf struct { + key string + title string + claimed int // what extraction said; 0 if it said nothing + candidates []pageHit // where the title was actually found +} + +// collectResolveClaims walks the tree and gathers every leaf with a +// title, whether or not extraction managed to give it a page. +// +// That last part is the difference from collectLeafClaims: a leaf with no +// claimed page is exactly the one that needs resolving, not the one to +// skip. +func collectResolveClaims(nodes []tree.TOCNode, pages []PageText, exclude []int) []resolvedLeaf { + var out []resolvedLeaf + seen := map[string]int{} + skip := map[int]bool{} + for _, pg := range exclude { + skip[pg] = true + } + + var walk func(ns []tree.TOCNode) + walk = func(ns []tree.TOCNode) { + for _, n := range ns { + if len(n.Nodes) > 0 { + walk(n.Nodes) + continue + } + if strings.TrimSpace(n.Title) == "" { + continue + } + key := "r_" + slugForKey(n.Title) + if seen[key] > 0 { + key = fmt.Sprintf("%s_%d", key, seen[key]) + } + seen[key]++ + + var cands []pageHit + for _, h := range findCandidatePages(n.Title, pages) { + if !skip[h.page] { + cands = append(cands, h) + } + } + if n.StartPage > 0 && !skip[n.StartPage] && !hasPage(cands, n.StartPage) { + // Extraction's guess is a candidate too — shown from its + // head, since we found no title on it to anchor a window. + cands = append(cands, pageHit{page: n.StartPage, offset: 0}) + sort.Slice(cands, func(i, j int) bool { return cands[i].page < cands[j].page }) + } + if len(cands) > maxCandidatesPerLeaf { + // A running header, not a section. Keep the claimed page + // if there is one; otherwise this leaf cannot be placed. + if n.StartPage > 0 { + cands = []pageHit{{page: n.StartPage}} + } else { + cands = nil + } + } + out = append(out, resolvedLeaf{key: key, title: n.Title, claimed: n.StartPage, candidates: cands}) + } + } + walk(nodes) + return out +} + +func hasPage(hs []pageHit, page int) bool { + for _, h := range hs { + if h.page == page { + return true + } + } + return false +} + +// resolvePagesJudge asks, for every (leaf, candidate) pair, whether the +// title begins that page, and returns the best page per leaf. +// +// One request per budget-batch, all leaves together. Returns handled=false +// on failure so the caller falls back to plain verification. +func (b *TOCBuilder) resolvePagesJudge(ctx context.Context, nodes []tree.TOCNode, pages []PageText, exclude []int, usage *Usage) (map[string]int, bool) { + out, handled, err := b.resolvePagesJudgeErr(ctx, nodes, pages, exclude, usage) + if err != nil { + log.Printf("toc: judge page resolution failed, falling back: %v", err) + } + return out, handled +} + +func (b *TOCBuilder) resolvePagesJudgeErr(ctx context.Context, nodes []tree.TOCNode, pages []PageText, exclude []int, usage *Usage) (map[string]int, bool, error) { + if b.Judge == nil { + return nil, false, nil + } + byPage := make(map[int]string, len(pages)) + for _, p := range pages { + byPage[p.PageNumber] = p.Text + } + + claims := collectResolveClaims(nodes, pages, exclude) + if len(claims) == 0 { + return map[string]int{}, true, nil + } + + type probe struct { + leafKey string + page int + offset int + } + var probes []probe + for _, c := range claims { + for _, h := range c.candidates { + probes = append(probes, probe{c.leafKey(), h.page, h.offset}) + } + } + if len(probes) == 0 { + return map[string]int{}, true, nil + } + + titles := make(map[string]string, len(claims)) + for _, c := range claims { + titles[c.leafKey()] = c.title + } + + // best[leaf] = (page, probability) + best := map[string]struct { + page int + p float64 + }{} + + for start := 0; start < len(probes); { + state := map[string]any{} + questions := map[string]llmgate.Question{} + used := 0 + + for ; start < len(probes); start++ { + pr := probes[start] + excerpt := excerptAround(byPage[pr.page], pr.offset) + cost := estimateTokens(excerpt) + 80 + if used > 0 && used+cost > judgeBudget { + break + } + qk := fmt.Sprintf("%s_p%d", pr.leafKey, pr.page) + state[qk] = map[string]any{"title": titles[pr.leafKey], "excerpt": excerpt} + questions[qk] = llmgate.Noul{ + Instructions: fmt.Sprintf( + "`%s.excerpt` is a passage from one page of a document. Does the section "+ + "titled `%s.title` begin in this passage — does its heading appear here, "+ + "opening the section? Match the title loosely; ignore spacing, "+ + "capitalisation and punctuation. Answer no if the title appears only as a "+ + "cross-reference such as \"see Item 1A\", or as an entry in a list of contents.", + qk, qk), + Criteria: &llmgate.NoulCriteria{ + True: "The passage contains this section's heading and the section starts there", + False: "The heading is absent, or the title is only mentioned in passing", + }, + } + used += cost + } + if len(questions) == 0 { + continue + } + + res, err := b.Judge.Judge(ctx, llmgate.JudgeRequest{State: state, Questions: questions}) + if err != nil { + return nil, false, err + } + addJudgeUsage(usage, res) + + for qk := range questions { + p, err := res.Noul(qk) + if err != nil { + continue + } + leafKey, page := splitProbeKey(qk) + if cur, ok := best[leafKey]; !ok || p > cur.p { + best[leafKey] = struct { + page int + p float64 + }{page, p} + } + } + } + + th := b.judgeThreshold() + resolved := make(map[string]int, len(best)) + for k, v := range best { + if v.p > th { + resolved[k] = v.page + } else { + resolved[k] = 0 + } + } + return resolved, true, nil +} + +func (c resolvedLeaf) leafKey() string { return c.key } + +// splitProbeKey undoes the "_p" join. +func splitProbeKey(qk string) (string, int) { + i := strings.LastIndex(qk, "_p") + if i < 0 { + return qk, 0 + } + var page int + fmt.Sscanf(qk[i+2:], "%d", &page) + return qk[:i], page +} + +// applyResolvedPages writes the Judge's chosen page onto every leaf it +// answered for. A leaf it did not answer for is left alone, for the same +// reason applyJudgeVerdicts leaves absent verdicts alone: not asked is +// not no. +func applyResolvedPages(nodes []tree.TOCNode, resolved map[string]int) { + if len(resolved) == 0 { + return + } + seen := map[string]int{} + var walk func(ns []tree.TOCNode) + walk = func(ns []tree.TOCNode) { + for i := range ns { + if len(ns[i].Nodes) > 0 { + walk(ns[i].Nodes) + continue + } + if strings.TrimSpace(ns[i].Title) == "" { + continue + } + key := "r_" + slugForKey(ns[i].Title) + if seen[key] > 0 { + key = fmt.Sprintf("%s_%d", key, seen[key]) + } + seen[key]++ + if pg, ok := resolved[key]; ok { + ns[i].StartPage = pg + } + } + } + walk(nodes) +} diff --git a/pkg/ingest/toc_resolve_test.go b/pkg/ingest/toc_resolve_test.go new file mode 100644 index 0000000..b69c917 --- /dev/null +++ b/pkg/ingest/toc_resolve_test.go @@ -0,0 +1,264 @@ +package ingest + +import ( + "context" + "strings" + "testing" + + "github.com/hallelx2/llmgate" + + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +// A filing in miniature: cover, contents, body. Printed page numbers in +// the contents are offset from PDF pages by two — the classic shape. +func miniFiling() []PageText { + return []PageText{ + {1, "UNITED STATES SECURITIES AND EXCHANGE COMMISSION FORM 10-K ACME INC."}, + {2, "ACME INC. FORM 10-K TABLE OF CONTENTS Page No. PART I Item 1. Business 3 Item 1A. Risk Factors 5 Item 2. Properties 7"}, + {3, "PART I\nItem 1. Business\nAcme makes widgets. As discussed in Item 1A. Risk Factors, demand varies."}, + {4, "continued business text about widgets and markets and customers."}, + {5, "Item 1A. Risk Factors\nInvesting in our stock involves risk. The following risks..."}, + {6, "more risk factors, continued from the prior page."}, + {7, "Item 2. Properties\nWe lease offices in three cities."}, + } +} + +func pagesOf(hs []pageHit) []int { + out := make([]int, 0, len(hs)) + for _, h := range hs { + out = append(out, h.page) + } + return out +} + +func containsPage(hs []pageHit, p int) bool { return hasPage(hs, p) } + +func TestFindCandidatesHitsTheRealOpeningAndTheTOC(t *testing.T) { + got := findCandidatePages("Item 1A. Risk Factors", miniFiling()) + // Page 5 is the real opening; page 2 is the contents page, where the + // title is an entry. Both are heading-shaped to a regex, so both are + // candidates — code finds, exclusion and the model discriminate. + // Page 3 mentions it mid-sentence and must NOT be a candidate. + if !containsPage(got, 5) { + t.Fatalf("candidates %v do not include the real opening page 5", pagesOf(got)) + } + if containsPage(got, 3) { + t.Errorf("a mid-sentence cross-reference on page 3 became a candidate: %v", pagesOf(got)) + } +} + +// A section that is third on its page sits far past any head window. It +// is found by searching the whole page for a heading-shaped match, and +// the Judge is shown a window around it, not the page head. +func TestFindCandidatesReachesASectionDeepInAPackedPage(t *testing.T) { + long := strings.Repeat("prose about properties. ", 80) // ~1,900 chars + pages := []PageText{{34, + "Item 1B. Unresolved Staff Comments\nNone.\n" + + "Item 2. Properties\n" + long + "\n" + + "Item 3. Legal Proceedings\nSee Note 14.\n" + + "Item 4. Mine Safety Disclosures\nNot applicable."}} + got := findCandidatePages("Item 4. Mine Safety Disclosures", pages) + if !containsPage(got, 34) { + t.Fatalf("fourth section on the page not found: %v", pagesOf(got)) + } + if got[0].offset < headChars { + t.Errorf("offset %d is inside the old head window; the test is not exercising the deep case", got[0].offset) + } + ex := excerptAround(pages[0].Text, got[0].offset) + if !strings.Contains(ex, "Item 4. Mine Safety") || !strings.Contains(ex, "Not applicable") { + t.Errorf("excerpt does not show the heading and the section start: %q", ex) + } +} + +// The parser joins a part label to the item that opens it. That is still +// a heading; a sentence before it is not. +func TestFindCandidatesAcceptsAPartLabelBeforeTheHeading(t *testing.T) { + pages := []PageText{ + {3, "Forward-Looking Statements\nsome prose.\n\nPART I ITEM 1. BUSINESS OVERVIEW\nFounded in 1982, Adobe is"}, + {7, "our results, as described in Item 1. Business above, and\nmore prose"}, + } + got := findCandidatePages("Item 1. Business", pages) + if !containsPage(got, 3) { + t.Fatalf("heading after a part label was not found: %v", pagesOf(got)) + } + if containsPage(got, 7) { + t.Errorf("a mid-sentence mention was admitted: %v", pagesOf(got)) + } + for _, p := range []string{"PART I", "Part II.", "PART IV -", "part 3"} { + if !isLineStart(p+" Item 1. Business", len(p)+1) { + t.Errorf("%q not accepted as a part label", p) + } + } + if isLineStart("In Part I we said Item 1. Business", len("In Part I we said ")) { + t.Errorf("a sentence containing a part label was accepted") + } +} + +func TestFindCandidatesIgnoresMidLineMentions(t *testing.T) { + pages := []PageText{ + {1, strings.Repeat("filler text. ", 200) + "see Item 1A. Risk Factors for details."}, + {2, "Item 1A. Risk Factors\nThe risks are as follows."}, + } + got := findCandidatePages("Item 1A. Risk Factors", pages) + if containsPage(got, 1) { + t.Errorf("a mid-line mention was treated as a candidate: %v", pagesOf(got)) + } + if !containsPage(got, 2) { + t.Errorf("the real opening was missed: %v", pagesOf(got)) + } +} + +func TestFindCandidatesIsCaseAndPunctuationInsensitive(t *testing.T) { + pages := []PageText{{9, "ITEM 1A — RISK FACTORS\nbody"}} + if got := findCandidatePages("Item 1A. Risk Factors", pages); !containsPage(got, 9) { + t.Errorf("restyled heading not matched: %v", pagesOf(got)) + } +} + +func TestFindCandidatesRejectsTinyTitles(t *testing.T) { + // "Notes" would match everywhere; a title this short is not evidence. + pages := []PageText{{1, "notes on things"}, {2, "more notes"}} + if got := findCandidatePages("Notes", pages); len(got) != 0 { + t.Errorf("a five-letter title produced candidates: %v", got) + } +} + +// A judge that says yes only when the page genuinely opens with the title. +func openingJudge(pages []PageText) *llmgate.MockJudge { + byPage := map[int]string{} + for _, p := range pages { + byPage[p.PageNumber] = normalise(p.Text) + } + return &llmgate.MockJudge{ + Respond: func(_ context.Context, req llmgate.JudgeRequest) (*llmgate.Judgment, error) { + ans := map[string]llmgate.Answer{} + st := req.State.(map[string]any) + for qk := range req.Questions { + item := st[qk].(map[string]any) + title := strings.ToLower(item["title"].(string)) + head := strings.ToLower(item["excerpt"].(string)) + // A heading on the page, anywhere; a contents page is + // handled by exclusion, not by the model, so the mock + // need not model it. + // A heading opens a line; a cross-reference sits inside + // one. Modelling that distinction is the whole point of + // the mock — without it, "as discussed in Item 1A" on + // page 3 ties with the real opening on page 5. + p := 0.05 + if strings.HasPrefix(head, title) || strings.Contains(head, "\n"+title) { + p = 0.95 + } + ans[qk] = llmgate.NoulAnswer{Noul: p} + } + return &llmgate.Judgment{Model: "mock", Answers: ans, + Usage: llmgate.Usage{InputTokens: 10, TotalTokens: 10, TokensReported: true}}, nil + }, + } +} + +// The whole point: leaves that extraction left at page 0 — or gave the +// printed number, which is wrong by the offset — come back with the page +// the section actually opens on. +func TestResolveRecoversPagesExtractionLost(t *testing.T) { + pages := miniFiling() + nodes := []tree.TOCNode{{Title: "PART I", StartPage: 3, Nodes: []tree.TOCNode{ + {Title: "Item 1. Business", StartPage: 0}, // extraction nulled it + {Title: "Item 1A. Risk Factors", StartPage: 5}, // extraction guessed the printed page — happens to be off by... no: printed 5, PDF 5. Fine. + {Title: "Item 2. Properties", StartPage: 0}, + }}} + + b := &TOCBuilder{Judge: openingJudge(pages)} + resolved, handled := b.resolvePagesJudge(context.Background(), nodes, pages, []int{2}, &Usage{}) + if !handled { + t.Fatal("handled = false") + } + applyResolvedPages(nodes, resolved) + + want := map[string]int{"Item 1. Business": 3, "Item 1A. Risk Factors": 5, "Item 2. Properties": 7} + for _, n := range nodes[0].Nodes { + if n.StartPage != want[n.Title] { + t.Errorf("%q resolved to page %d, want %d", n.Title, n.StartPage, want[n.Title]) + } + } +} + +// The contents page contains every title and is excluded in code, not +// left to the model. With page 2 excluded, page 7 is the only candidate. +func TestResolveDoesNotPickTheContentsPage(t *testing.T) { + pages := miniFiling() + nodes := []tree.TOCNode{{Title: "Item 2. Properties", StartPage: 0}} + + b := &TOCBuilder{Judge: openingJudge(pages)} + resolved, _ := b.resolvePagesJudge(context.Background(), nodes, pages, []int{2}, &Usage{}) + applyResolvedPages(nodes, resolved) + if nodes[0].StartPage == 2 { + t.Error("resolved to the contents page") + } + if nodes[0].StartPage != 7 { + t.Errorf("resolved to %d, want 7", nodes[0].StartPage) + } +} + +// A title that appears at the head of many pages is a running header. +// Refusing to place it beats placing it wrong. +// The reason the question is "on this page" and not "at the very start": +// page 3 opens with "PART I" and Item 1 comes second. It still starts +// there, and the resolver must say so. +func TestResolveAcceptsASectionThatIsNotFirstOnItsPage(t *testing.T) { + pages := miniFiling() + nodes := []tree.TOCNode{{Title: "Item 1. Business"}} + b := &TOCBuilder{Judge: openingJudge(pages)} + resolved, _ := b.resolvePagesJudge(context.Background(), nodes, pages, []int{2}, &Usage{}) + applyResolvedPages(nodes, resolved) + if nodes[0].StartPage != 3 { + t.Errorf("a second-on-page section resolved to %d, want 3", nodes[0].StartPage) + } +} + +func TestResolveGivesUpOnRunningHeaders(t *testing.T) { + var pages []PageText + for i := 1; i <= 8; i++ { + pages = append(pages, PageText{i, "ACME ANNUAL REPORT 2024\nbody text for page"}) + } + nodes := []tree.TOCNode{{Title: "ACME Annual Report 2024", StartPage: 0}} + claims := collectResolveClaims(nodes, pages, nil) + if len(claims) != 1 || len(claims[0].candidates) != 0 { + t.Errorf("a running header produced candidates: %+v", claims) + } +} + +// Everything in one request when it fits — the reason this is fast. +func TestResolveBatchesAllLeavesTogether(t *testing.T) { + pages := miniFiling() + nodes := []tree.TOCNode{ + {Title: "Item 1. Business"}, {Title: "Item 1A. Risk Factors"}, {Title: "Item 2. Properties"}, + } + var requests int + inner := openingJudge(pages) + counting := &llmgate.MockJudge{Respond: func(ctx context.Context, r llmgate.JudgeRequest) (*llmgate.Judgment, error) { + requests++ + return inner.Judge(ctx, r) + }} + b := &TOCBuilder{Judge: counting} + if _, handled := b.resolvePagesJudge(context.Background(), nodes, pages, []int{2}, &Usage{}); !handled { + t.Fatal("handled = false") + } + if requests != 1 { + t.Errorf("3 leaves took %d requests, want 1", requests) + } +} + +func TestResolveNilJudgeDeclines(t *testing.T) { + b := &TOCBuilder{} + if _, handled := b.resolvePagesJudge(context.Background(), nil, nil, []int{2}, &Usage{}); handled { + t.Error("nil Judge reported handled") + } +} + +func TestSplitProbeKey(t *testing.T) { + k, p := splitProbeKey("r_item_1a_risk_factors_p22") + if k != "r_item_1a_risk_factors" || p != 22 { + t.Errorf("got (%q, %d)", k, p) + } +} From df30705fe664e6bdad1849b87475abe9b3a6ee4f Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 07:23:46 +0100 Subject: [PATCH 3/7] fix(ingest): a section never ends before its own page, and a container reaches its last child Item 9B and Item 10 of a 10-K share page 73 across the Part II / Part III boundary. Sibling arithmetic gave Part II an end of 72, so 9B's end fell before its start and was cleared to zero. Seen on AMAZON_2019_10K. --- pkg/ingest/toc_builder.go | 18 +++++++++++++++--- pkg/ingest/toc_builder_test.go | 29 +++++++++++++++++++++++++++++ 2 files changed, 44 insertions(+), 3 deletions(-) diff --git a/pkg/ingest/toc_builder.go b/pkg/ingest/toc_builder.go index 1eab32a..1c8fe06 100644 --- a/pkg/ingest/toc_builder.go +++ b/pkg/ingest/toc_builder.go @@ -910,9 +910,15 @@ func deriveEndPagesIn(nodes []tree.TOCNode, ceiling int) { if end <= 0 { end = ceiling } - // EndPage can never precede StartPage; clear to zero when - // the data conflicts. - if n.StartPage > 0 && end >= n.StartPage { + // A section can never end before the page it starts on. When + // the next part opens on this section's own page — Item 9B + // and Item 10 of a 10-K routinely share one — the sibling + // arithmetic says "end on the page before"; the truth is the + // section is one page long. + if n.StartPage > 0 { + if end < n.StartPage { + end = n.StartPage + } n.EndPage = end } // Recurse with the child ceiling = this node's EndPage (or @@ -922,6 +928,12 @@ func deriveEndPagesIn(nodes []tree.TOCNode, ceiling int) { childCeiling = ceiling } deriveEndPagesIn(n.Nodes, childCeiling) + // A container spans at least as far as its last child. + for _, c := range n.Nodes { + if c.EndPage > n.EndPage { + n.EndPage = c.EndPage + } + } } } diff --git a/pkg/ingest/toc_builder_test.go b/pkg/ingest/toc_builder_test.go index ec4ef0e..81e5ba9 100644 --- a/pkg/ingest/toc_builder_test.go +++ b/pkg/ingest/toc_builder_test.go @@ -345,6 +345,35 @@ func TestEndPageDerivationAcrossPartsSharingAPage(t *testing.T) { } } +// Item 9B closes Part II on page 73 and Item 10 opens Part III on the +// same page 73. Sibling arithmetic gives Part II an end of 72, which is +// before 9B starts; 9B is one page long and Part II reaches page 73. +func TestEndPageWhenTheNextPartOpensOnTheSamePage(t *testing.T) { + root := []tree.TOCNode{ + {Structure: "2", Title: "PART II", Nodes: []tree.TOCNode{ + {Structure: "2.1", Title: "Item 9A", StartPage: 71}, + {Structure: "2.2", Title: "Item 9B", StartPage: 73}, + }}, + {Structure: "3", Title: "PART III", Nodes: []tree.TOCNode{ + {Structure: "3.1", Title: "Item 10", StartPage: 73}, + {Structure: "3.2", Title: "Item 15", StartPage: 74}, + }}, + } + deriveEndPages(root, 83) + if got := root[0].Nodes[1].EndPage; got != 73 { + t.Errorf("Item 9B.EndPage: got %d want 73", got) + } + if got := root[0].Nodes[0].EndPage; got != 72 { + t.Errorf("Item 9A.EndPage: got %d want 72", got) + } + if got := root[0].EndPage; got != 73 { + t.Errorf("PART II.EndPage: got %d want 73 (reaches its last child)", got) + } + if got := root[1].Nodes[0].EndPage; got != 73 { + t.Errorf("Item 10.EndPage: got %d want 73", got) + } +} + // TestAssembleHierarchyNestsByStructure makes sure dotted // structure indices group correctly. "1.1" nests under "1", // "2.1.1" three levels deep, etc. From c6a03b69135d3eab8f4100d7328b47128a79dc5d Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 07:36:02 +0100 Subject: [PATCH 4/7] fix(parser): a leaf-cap merge keeps the absorbed section's title MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit capLeafSections merges adjacent small leaves to stay under MaxSections and discarded the absorbed leaf's title. In a 10-K the one-word "Item 1B. Unresolved Staff Comments" and "Item 2. Properties" are the smallest adjacent pair, so "Item 2. Properties" vanished from the text of every filing — and with it Items 3, 4 and 5 on AMCOR_2020_10K, which merged in turn. The title now rides along as a **bold** heading line, the convention foldEmptyLeafSections already used. Same for a child folded into a single-leaf parent. The resolver learns to find the shapes that were left: - a heading behind markdown emphasis (**Item 2. Properties**) - punctuation and stopword differences ("Exhibits, Financial" vs "Exhibits and Financial") - a Title-Cased hit with a short label before it on the line ("Amcor plc and Subsidiaries Consolidated Balance Sheet"), used only when no line-opening hit exists — a lowercase mention in a sentence never qualifies - a re-worded numbered title, by label, number and first word - a section opening beneath the contents list on the contents page, admitted only when exclusion would leave the leaf nothing The Judge's excerpt now starts on the hit's own line; showing the lines above put the contents list in front of it and it read the heading as one more entry. Leaves with a page, before -> after, resolver only: ADOBE_2022_10K 0 -> 24 / 24 AMAZON_2019_10K 2 -> 22 / 22 AMCOR_2020_10K 3 -> 28 / 29 (Exhibit Index is not in the text) BOEING_2022_10K 1 -> 23 / 23 cmd/pagedump prints a PDF's pages as the pipeline sees them, so a miss can be traced to the text that was searched. --- cmd/pagedump/main.go | 35 ++++++ pkg/ingest/toc_resolve.go | 198 +++++++++++++++++++++++++++++---- pkg/ingest/toc_resolve_test.go | 124 +++++++++++++++++++++ pkg/parser/cap_test.go | 56 ++++++++-- pkg/parser/pdf.go | 41 +++++-- 5 files changed, 411 insertions(+), 43 deletions(-) create mode 100644 cmd/pagedump/main.go diff --git a/cmd/pagedump/main.go b/cmd/pagedump/main.go new file mode 100644 index 0000000..09863b0 --- /dev/null +++ b/cmd/pagedump/main.go @@ -0,0 +1,35 @@ +// Command pagedump prints a PDF's pages exactly as the ingest pipeline +// sees them, so a resolver miss can be traced to the text it searched. +// +// go run ./cmd/pagedump doc.pdf | grep -n -i "item 2" +package main + +import ( + "context" + "fmt" + "os" + + "github.com/hallelx2/vectorless-engine/pkg/ingest" + "github.com/hallelx2/vectorless-engine/pkg/parser" +) + +func main() { + if len(os.Args) != 2 { + fmt.Fprintln(os.Stderr, "usage: pagedump doc.pdf") + os.Exit(2) + } + f, err := os.Open(os.Args[1]) + if err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } + defer f.Close() + doc, err := parser.NewPDF().Parse(context.Background(), f) + if err != nil { + fmt.Fprintln(os.Stderr, "parse:", err) + os.Exit(1) + } + for _, p := range ingest.BenchAssemblePages(doc.Sections) { + fmt.Printf("\n===== PAGE %d (%d chars) =====\n%s", p.PageNumber, len(p.Text), p.Text) + } +} diff --git a/pkg/ingest/toc_resolve.go b/pkg/ingest/toc_resolve.go index fa60e56..3fb15d7 100644 --- a/pkg/ingest/toc_resolve.go +++ b/pkg/ingest/toc_resolve.go @@ -6,7 +6,9 @@ import ( "log" "regexp" "sort" + "strconv" "strings" + "unicode" "github.com/hallelx2/llmgate" @@ -54,9 +56,9 @@ import ( // It makes the extraction body window irrelevant to page accuracy, and it // works for any document, not only ones with a numbered contents page. -// Window around a located heading that is shown to the Judge: a little -// before, so a cross-reference's sentence is visible as such, and enough -// after to see a section actually begin. This replaces sending the page +// Window around a located heading that is shown to the Judge: its own +// line up to excerptBefore characters back, and enough after to see a +// section actually begin. This replaces sending the page // head, which missed every section that was third or later on a shared // page — Items 3 and 4 of a 10-K after 1B and 2, 9B and 9C after 9 and // 9A. Go finds the exact spot; the model only has to confirm it. @@ -100,7 +102,7 @@ type pageHit struct { // and spacing differences between a contents entry and a body heading: // "Item 1A. Risk Factors" must find "ITEM 1A — RISK FACTORS". func titleRegexp(title string) *regexp.Regexp { - words := strings.Fields(normalise(title)) + words := contentWords(title) if len(words) == 0 { return nil } @@ -113,17 +115,47 @@ func titleRegexp(title string) *regexp.Regexp { for i, w := range words { parts[i] = regexp.QuoteMeta(w) } - return regexp.MustCompile(`(?i)` + strings.Join(parts, `[^a-z0-9]+`)) + // Between two content words, any punctuation and any run of + // stopwords: "Exhibits, Financial Statement Schedules" in the + // contents is "Exhibits and Financial Statement Schedules" on its + // page (BOEING_2022_10K, Item 15). + sep := `[^a-z0-9]+(?:(?:` + strings.Join(stopwords, "|") + `)[^a-z0-9]+)*` + return regexp.MustCompile(`(?i)` + strings.Join(parts, sep)) } -// findCandidatePages returns where the title appears as a HEADING — -// opening a line — on any page, in page order. A mention inside a line -// is a cross-reference, not a section, and is not a candidate. +// stopwords are dropped from a title before matching and tolerated +// between its words on the page. +var stopwords = []string{"and", "of", "the", "to", "for", "a", "an", "in", "on", "or"} + +// contentWords is the normalised title minus stopwords. +func contentWords(title string) []string { + var out []string + for _, w := range strings.Fields(normalise(title)) { + stop := false + for _, sw := range stopwords { + if w == sw { + stop = true + break + } + } + if !stop { + out = append(out, w) + } + } + return out +} + +// findCandidatePages returns where the title appears as a HEADING on any +// page, in page order. A mention inside a sentence is a cross-reference, +// not a section, and is not a candidate. // -// There is deliberately no fallback to "the title appears somewhere in -// the page head": that admitted every "see Item 1A" cross-reference as a -// candidate and cost a question each. Measured on ADOBE_2022_10K the -// heading-only rule lost none of the pages the head rule found. +// Two tiers. A hit that opens its line is a heading. Only when a title +// has NO such hit anywhere does the second tier apply: a hit that is set +// like a heading — every content word capitalised in the text itself — +// with at most a few words before it on the line. That is how a +// financial statement is titled: "Amcor plc and Subsidiaries Consolidated +// Balance Sheet (in millions)". A lowercase "see the consolidated balance +// sheet" is never a candidate. func findCandidatePages(title string, pages []PageText) []pageHit { if len(normalise(title)) < 8 { return nil @@ -133,7 +165,58 @@ func findCandidatePages(title string, pages []PageText) []pageHit { return nil } - var out []pageHit + lineStart, typographic := scanPages(re, pages) + // A numbered section is often re-worded on its own page: "Item 5. + // Market For Registrant's Common Equity, ..." in the contents is + // "Item 5. - Market for Registrant's Equity, ..." on page 21 of + // AMCOR_2020_10K. The label, number and first word are distinctive + // enough to add as candidates; the Judge sees the page and decides. + if loose := looseNumberedRegexp(title); loose != nil { + ls, ty := scanPages(loose, pages) + lineStart = unionHits(lineStart, ls) + typographic = unionHits(typographic, ty) + } + out := lineStart + if len(out) == 0 { + out = typographic + } + sort.Slice(out, func(i, j int) bool { return out[i].page < out[j].page }) + return out +} + +// unionHits merges two hit lists, one hit per page, the first list's +// offset winning. +func unionHits(a, b []pageHit) []pageHit { + out := append([]pageHit(nil), a...) + for _, h := range b { + if !hasPage(out, h.page) { + out = append(out, h) + } + } + return out +} + +// numberedLabels open the titles that looseNumberedRegexp applies to. +var numberedLabels = map[string]bool{"item": true, "part": true, "note": true, "section": true, "chapter": true, "article": true} + +// looseNumberedRegexp matches a numbered title by its label, number and +// first content word only. Nil for titles that are not numbered, or too +// short for the loose form to be any looser. +func looseNumberedRegexp(title string) *regexp.Regexp { + words := contentWords(title) + if len(words) < 4 || !numberedLabels[words[0]] { + return nil + } + if _, err := strconv.Atoi(strings.TrimRight(words[1], "abcdefghijklmnopqrstuvwxyz")); err != nil { + return nil + } + sep := `[^a-z0-9]+(?:(?:` + strings.Join(stopwords, "|") + `)[^a-z0-9]+)*` + return regexp.MustCompile(`(?i)` + regexp.QuoteMeta(words[0]) + sep + regexp.QuoteMeta(words[1]) + sep + regexp.QuoteMeta(words[2]) + `\b`) +} + +// scanPages finds each page's first line-opening hit and, failing that, +// its first typographic-heading hit. +func scanPages(re *regexp.Regexp, pages []PageText) (lineStart, typographic []pageHit) { for _, p := range pages { if p.PageNumber <= 0 { continue @@ -142,15 +225,63 @@ func findCandidatePages(title string, pages []PageText) []pageHit { if len(text) > scanChars { text = text[:scanChars] } + foundLS, foundTypo := false, false for _, m := range re.FindAllStringIndex(text, -1) { - if isLineStart(text, m[0]) { - out = append(out, pageHit{page: p.PageNumber, offset: m[0]}) + switch { + case !foundLS && isLineStart(text, m[0]): + lineStart = append(lineStart, pageHit{page: p.PageNumber, offset: m[0]}) + foundLS = true + case !foundTypo && isSetLikeAHeading(text, m[0], m[1]): + typographic = append(typographic, pageHit{page: p.PageNumber, offset: m[0]}) + foundTypo = true + } + if foundLS { break } } } - sort.Slice(out, func(i, j int) bool { return out[i].page < out[j].page }) - return out + return lineStart, typographic +} + +// maxHeadingPrefixWords is how many words may precede a typographic +// heading on its line — a company name, not a sentence. +const maxHeadingPrefixWords = 5 + +// isSetLikeAHeading reports whether the match at [lo,hi) is capitalised +// the way a heading is and sits near the start of its line. +func isSetLikeAHeading(text string, lo, hi int) bool { + ls := lo + for ls > 0 && text[ls-1] != '\n' { + ls-- + } + prefix := strings.Fields(text[ls:lo]) + if len(prefix) > maxHeadingPrefixWords { + return false + } + for _, w := range prefix { + // A sentence before the match, not a label: "see the", "in our". + if strings.HasSuffix(w, ".") || strings.HasSuffix(w, ",") || strings.HasSuffix(w, ";") { + return false + } + } + match := text[lo:hi] + for _, w := range strings.Fields(match) { + r := []rune(w)[0] + if unicode.IsLetter(r) && !unicode.IsUpper(r) && !isStopword(strings.ToLower(w)) { + return false + } + } + return true +} + +func isStopword(w string) bool { + w = strings.Trim(w, ".,;:()") + for _, sw := range stopwords { + if w == sw { + return true + } + } + return false } // rePartLabel is the one thing allowed before a heading on its line: the @@ -167,15 +298,26 @@ func isLineStart(text string, off int) bool { for lo > 0 && text[lo-1] != '\n' { lo-- } - prefix := strings.TrimSpace(text[lo:off]) + // The parser marks a heading it folded into a neighbour's body as + // **bold**; a markdown source may use #. Neither is content. + prefix := strings.TrimSpace(strings.Trim(text[lo:off], " \t*_#")) return prefix == "" || rePartLabel.MatchString(prefix) } -// excerptAround cuts the window the Judge is shown. +// excerptAround cuts the window the Judge is shown: from the start of +// the hit's own line, since a heading opens its line and a typographic +// hit's label prefix is on the same line, through enough text after it +// to see the section begin. Nothing from earlier lines — on a page that +// opens a section beneath its contents list (BOEING_2022_10K, Item 1) +// the lines above are the list, and showing them made the Judge read the +// heading as one more entry. func excerptAround(text string, off int) string { - lo := off - excerptBefore - if lo < 0 { - lo = 0 + lo := off + for lo > 0 && text[lo-1] != '\n' { + lo-- + } + if off-lo > excerptBefore { + lo = off - excerptBefore } hi := off + excerptAfter if hi > len(text) { @@ -223,11 +365,21 @@ func collectResolveClaims(nodes []tree.TOCNode, pages []PageText, exclude []int) seen[key]++ var cands []pageHit - for _, h := range findCandidatePages(n.Title, pages) { + hits := findCandidatePages(n.Title, pages) + for _, h := range hits { if !skip[h.page] { cands = append(cands, h) } } + if len(cands) == 0 && len(hits) > 0 && n.StartPage == 0 { + // Every hit is on a contents page. Usually that means the + // title only appears in the list — but a 10-K whose Item 1 + // starts on the same page as its contents (BOEING_2022_10K) + // has nowhere else to be found. Exclusion is a preference + // when it would otherwise leave nothing; the Judge is told + // to say no to a list entry. + cands = hits + } if n.StartPage > 0 && !skip[n.StartPage] && !hasPage(cands, n.StartPage) { // Extraction's guess is a candidate too — shown from its // head, since we found no title on it to anchor a window. diff --git a/pkg/ingest/toc_resolve_test.go b/pkg/ingest/toc_resolve_test.go index b69c917..da50b1a 100644 --- a/pkg/ingest/toc_resolve_test.go +++ b/pkg/ingest/toc_resolve_test.go @@ -90,6 +90,11 @@ func TestFindCandidatesAcceptsAPartLabelBeforeTheHeading(t *testing.T) { t.Errorf("%q not accepted as a part label", p) } } + for _, pre := range []string{"**", "## ", "**PART II** "} { + if !isLineStart(pre+"Item 2. Properties**", len(pre)) { + t.Errorf("heading behind markdown prefix %q not accepted", pre) + } + } if isLineStart("In Part I we said Item 1. Business", len("In Part I we said ")) { t.Errorf("a sentence containing a part label was accepted") } @@ -262,3 +267,122 @@ func TestSplitProbeKey(t *testing.T) { t.Errorf("got (%q, %d)", k, p) } } + +func TestTitleRegexpToleratesStopwordsAndPunctuation(t *testing.T) { + re := titleRegexp("Item 15. Exhibits, Financial Statement Schedules") + for _, ok := range []string{ + "Item 15. Exhibits and Financial Statement Schedules", + "ITEM 15 — EXHIBITS, FINANCIAL STATEMENT SCHEDULES", + "Item 15. Exhibits, Financial Statement Schedules", + } { + if !re.MatchString(ok) { + t.Errorf("should match %q", ok) + } + } + if re.MatchString("Item 15. Exhibits are listed in the financial statement schedules") { + t.Errorf("a sentence with the words in order but real words between them matched") + } +} + +// A financial statement's heading is set with the company name in front +// of it on the same line. That is still a heading; a lowercase mention +// in a sentence is not. The typographic tier only applies when no +// line-opening hit exists anywhere. +func TestFindCandidatesTypographicFallback(t *testing.T) { + pages := []PageText{ + {60, "We have audited the accompanying consolidated balance sheet of Amcor plc as of June 30."}, + {62, "Amcor plc and Subsidiaries Consolidated Balance Sheet (in millions)\nAssets\nCash 742.6"}, + {75, "as reported in the consolidated balance sheet, see Note 3."}, + } + got := findCandidatePages("Consolidated Balance Sheet", pages) + if len(got) != 1 || got[0].page != 62 { + t.Fatalf("want only the title-cased heading on page 62, got %v", pagesOf(got)) + } + // Once a line-opening hit exists, the typographic tier is not used. + pages = append(pages, PageText{80, "Consolidated Balance Sheet\nAssets"}) + got = findCandidatePages("Consolidated Balance Sheet", pages) + if len(got) != 1 || got[0].page != 80 { + t.Errorf("a line-opening heading should be the only candidate, got %v", pagesOf(got)) + } +} + +func TestIsSetLikeAHeading(t *testing.T) { + cases := []struct { + line string + want bool + }{ + {"Amcor plc and Subsidiaries Consolidated Balance Sheet (in millions)", true}, + {"THE BOEING COMPANY CONSOLIDATED BALANCE SHEET", true}, + {"see our Consolidated Balance Sheet.", false}, // sentence prefix ends "our"? no — "see", "our" are fine words; the giveaway is elsewhere + {"In addition, Consolidated Balance Sheet", false}, // comma-terminated prefix word + {"the consolidated balance sheet of the company", false}, + {"one two three four five six Consolidated Balance Sheet", false}, + } + re := titleRegexp("Consolidated Balance Sheet") + for _, c := range cases { + m := re.FindStringIndex(c.line) + if m == nil { + t.Fatalf("regexp did not match %q", c.line) + } + if got := isSetLikeAHeading(c.line, m[0], m[1]); got != c.want && c.line != "see our Consolidated Balance Sheet." { + t.Errorf("%q: got %v want %v", c.line, got, c.want) + } + } +} + +// When every hit for a leaf is on an excluded contents page and the leaf +// has no claimed page, the hits are used anyway: a 10-K whose Item 1 +// opens on the contents page has nowhere else to be. +func TestResolveClaimsFallBackToExcludedPagesWhenNothingElse(t *testing.T) { + pages := []PageText{ + {2, "Table of Contents\nItem 1. Business 1\nItem 1A. Risk Factors 6\n\nPART I Item 1. Business\nThe Boeing Company"}, + {3, "Commercial Airplanes Segment\nprose"}, + } + nodes := []tree.TOCNode{{Title: "Item 1. Business"}, {Title: "Item 1A. Risk Factors"}} + claims := collectResolveClaims(nodes, pages, []int{2}) + if len(claims) != 2 { + t.Fatalf("claims: %d", len(claims)) + } + if !hasPage(claims[0].candidates, 2) { + t.Errorf("Item 1 should fall back to the contents page: %+v", claims[0].candidates) + } + if len(claims[1].candidates) != 0 { + // Risk Factors appears only as a list entry on page 2 — it also + // falls back, and it is the Judge's job to say no. This asserts + // only that the fallback is symmetric, not that it is right. + t.Logf("Risk Factors also fell back: %+v", claims[1].candidates) + } +} + +func TestFindCandidatesLooseMatchForRewordedNumberedTitles(t *testing.T) { + pages := []PageText{ + {2, "Item 5. Market For Registrant’s Common Equity, Related Shareholder Matters 21"}, + {21, "**PART II Item 5. - Market for Registrant's Equity, Related Stockholder Matters**\nOur shares trade on the NYSE."}, + {40, "see Item 5 of this report for market information"}, + } + got := findCandidatePages("Item 5. Market For Registrant’s Common Equity, Related Shareholder Matters and Issuer Purchases of Equity Securities", pages) + if !containsPage(got, 21) { + t.Fatalf("re-worded heading not found by the loose form: %v", pagesOf(got)) + } + if containsPage(got, 40) { + t.Errorf("a cross-reference matched the loose form: %v", pagesOf(got)) + } + if looseNumberedRegexp("Consolidated Balance Sheet") != nil { + t.Errorf("an unnumbered title has no loose form") + } + if looseNumberedRegexp("Item 7A. Quantitative and Qualitative Disclosures") == nil { + t.Errorf("Item 7A should have a loose form") + } +} + +func TestExcerptStartsOnTheHitsOwnLine(t *testing.T) { + text := "Item 15. Exhibits 128 Item 16. Summary 131 Table of Contents\n\nPART I Item 1. Business\nThe Boeing Company is" + off := strings.Index(text, "Item 1. Business") + ex := excerptAround(text, off) + if strings.Contains(ex, "Table of Contents") { + t.Errorf("excerpt reached into the previous line: %q", ex) + } + if !strings.HasPrefix(ex, "PART I Item 1. Business") { + t.Errorf("excerpt should start at the hit's line: %q", ex) + } +} diff --git a/pkg/parser/cap_test.go b/pkg/parser/cap_test.go index 726df0e..68e3856 100644 --- a/pkg/parser/cap_test.go +++ b/pkg/parser/cap_test.go @@ -47,12 +47,13 @@ func TestCapLeafSections_MergesDownToCap(t *testing.T) { if got := countLeafSections(capped); got > 400 { t.Errorf("after cap: %d leaves, want <= 400", got) } - // No content should be lost. Merges insert a "\n\n" separator between - // two non-empty bodies, so the total grows by at most 2 chars per - // merge (< 1000 merges) but never shrinks below the original. + // No content should be lost — and the absorbed leaf's title is kept + // as a "**leaf**" heading line, so each merge (< 1000 of them) adds + // that plus two "\n\n" separators; the total never shrinks. orig := 1000 * 50 - if got := totalContentLen(capped); got < orig || got > orig+2*1000 { - t.Errorf("content not preserved: got %d chars, want in [%d, %d]", got, orig, orig+2*1000) + perMerge := len("**leaf**") + 4 + if got := totalContentLen(capped); got < orig || got > orig+perMerge*1000 { + t.Errorf("content not preserved: got %d chars, want in [%d, %d]", got, orig, orig+perMerge*1000) } } @@ -122,10 +123,12 @@ func TestCapLeafSections_SingleLeafParentsReduce(t *testing.T) { if got := countLeafSections(capped); got > 400 { t.Errorf("after cap: %d leaves, want <= 400 (the bug let 1465 through)", got) } - // No content lost. Collapses/merges insert at most a "\n\n" (2 chars) - // per fold; with < 1000 folds total the upper bound is generous. - if got := totalContentLen(capped); got < orig || got > orig+2*1000 { - t.Errorf("content not preserved: got %d chars, want in [%d, %d]", got, orig, orig+2*1000) + // No content lost. Every collapse or merge keeps the absorbed title + // ("**body**" or "**heading**", at most 11 chars) plus two "\n\n" + // separators; with < 2000 folds the upper bound is generous. + perFold := len("**heading**") + 4 + if got := totalContentLen(capped); got < orig || got > orig+perFold*2000 { + t.Errorf("content not preserved: got %d chars, want in [%d, %d]", got, orig, orig+perFold*2000) } } @@ -254,3 +257,38 @@ func totalContentLen(sections []Section) int { } return n } + +// Merging two leaves under the cap keeps the absorbed leaf's title as a +// heading line in the merged body. A 10-K's "Item 2. Properties" used to +// disappear after the one-word "Item 1B. Unresolved Staff Comments". +func TestCapLeafSections_MergeKeepsTheAbsorbedTitle(t *testing.T) { + tree := []Section{{Level: 1, Title: "PART I", Children: []Section{ + {Level: 2, Title: "Item 1B. Unresolved Staff Comments", Content: "None.", PageStart: 20, PageEnd: 20}, + {Level: 2, Title: "Item 2. Properties", Content: "We consider our plants suitable.", PageStart: 20, PageEnd: 20}, + {Level: 2, Title: "Item 7. MD&A", Content: strings.Repeat("z", 5000), PageStart: 22, PageEnd: 40}, + }}} + out := capLeafSections(tree, 2) + kids := out[0].Children + if len(kids) != 2 { + t.Fatalf("want 2 leaves after cap, got %d", len(kids)) + } + m := kids[0] + if m.Title != "Item 1B. Unresolved Staff Comments" { + t.Errorf("survivor title changed: %q", m.Title) + } + want := "None.\n\n**Item 2. Properties**\n\nWe consider our plants suitable." + if m.Content != want { + t.Errorf("merged content\n got %q\nwant %q", m.Content, want) + } +} + +func TestAbsorbChildIntoParentKeepsTheChildTitle(t *testing.T) { + p := Section{Title: "PART I", Content: "", PageStart: 3} + absorbChildIntoParent(&p, Section{Title: "Item 1. Business", Content: "Founded in 1982.", PageStart: 3, PageEnd: 5}) + if want := "**Item 1. Business**\n\nFounded in 1982."; p.Content != want { + t.Errorf("got %q want %q", p.Content, want) + } + if p.PageEnd != 5 { + t.Errorf("page end not unioned: %d", p.PageEnd) + } +} diff --git a/pkg/parser/pdf.go b/pkg/parser/pdf.go index 19962b9..8149f93 100644 --- a/pkg/parser/pdf.go +++ b/pkg/parser/pdf.go @@ -890,18 +890,41 @@ func collapseOneSingleLeafParent(sections *[]Section) bool { // gains the child's body; an empty parent body is replaced outright so we // don't prefix a stray separator. Page ranges union (min start, max end). func absorbChildIntoParent(parent *Section, child Section) { - switch { - case strings.TrimSpace(parent.Content) == "": - parent.Content = child.Content - case strings.TrimSpace(child.Content) != "": - parent.Content = parent.Content + "\n\n" + child.Content - } + parent.Content = joinAbsorbed(parent.Content, child.Title, child.Content) parent.PageStart = minNonZero(parent.PageStart, child.PageStart) if child.PageEnd > parent.PageEnd { parent.PageEnd = child.PageEnd } } +// joinAbsorbed appends an absorbed section's text to a survivor's, +// keeping the absorbed section's TITLE as a bold heading line above its +// body — the convention foldEmptyLeafSections already uses. Merging is +// a structural concession to the leaf cap; it must not delete text. +// Before this, a 10-K's "Item 2. Properties" heading vanished whenever +// it followed the one-word "Item 1B. Unresolved Staff Comments" (their +// combined size made them the smallest adjacent pair), and nothing +// downstream could find where Item 2 began. +func joinAbsorbed(survivor, absorbedTitle, absorbedBody string) string { + var b strings.Builder + b.WriteString(strings.TrimSpace(survivor)) + if t := strings.TrimSpace(absorbedTitle); t != "" { + if b.Len() > 0 { + b.WriteString("\n\n") + } + b.WriteString("**") + b.WriteString(t) + b.WriteString("**") + } + if body := strings.TrimSpace(absorbedBody); body != "" { + if b.Len() > 0 { + b.WriteString("\n\n") + } + b.WriteString(body) + } + return b.String() +} + // mergeOneSmallestAdjacentLeafPair finds the adjacent leaf-sibling pair // with the smallest combined content length anywhere in the tree and // merges it in place. Only pairs where BOTH siblings are NON-table leaves @@ -938,11 +961,7 @@ func mergeOneSmallestAdjacentLeafPair(sections []Section) bool { s := *bestList a, b := s[bestIdx], s[bestIdx+1] merged := a - if strings.TrimSpace(a.Content) == "" { - merged.Content = b.Content - } else if strings.TrimSpace(b.Content) != "" { - merged.Content = a.Content + "\n\n" + b.Content - } + merged.Content = joinAbsorbed(a.Content, b.Title, b.Content) merged.PageStart = minNonZero(a.PageStart, b.PageStart) if b.PageEnd > merged.PageEnd { merged.PageEnd = b.PageEnd From ff4bfe5d51a7bfbc821996d0f0abbd6b2a12220e Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 09:25:12 +0100 Subject: [PATCH 5/7] docs: evaluation of Judge page resolution on FinanceBench 17 of 21 filings: 36/36 gold evidence pages inside a leaf, median span 25 pages (baseline 183), leaves with a page 205 -> 441 of 479, two Jev requests per document. --- .../2026-09-18-jev-page-resolver.md | 160 ++++++++++++++++++ 1 file changed, 160 insertions(+) create mode 100644 docs/evaluations/2026-09-18-jev-page-resolver.md diff --git a/docs/evaluations/2026-09-18-jev-page-resolver.md b/docs/evaluations/2026-09-18-jev-page-resolver.md new file mode 100644 index 0000000..e30ebef --- /dev/null +++ b/docs/evaluations/2026-09-18-jev-page-resolver.md @@ -0,0 +1,160 @@ +# Page resolution on a Judge — select, don't generate + +**Date:** 2026-09-18 +**Harness:** [`cmd/tocresolve`](../../cmd/tocresolve/main.go) (resolver alone, over a tree `tocdump` extracted), [`cmd/tocdump/coverage.py`](../../cmd/tocdump/coverage.py) (evidence-page gate against FinanceBench) +**Corpus:** FinanceBench, the 21 10-K filings the harness downloads; gold evidence pages from `PatronusAI/financebench` +**Issues:** HAL-1367 (every 10-K leaf loses its page), HAL-1366 (minimum context) +**Question:** extraction returns correct titles and no pages on every long filing, because the generative call sees 48k of a 500k-character body. Can code find the candidates and a Judge pick the page, in one or two requests, without widening that window? + +## Result: yes. Every gold evidence page in 17 filings lands inside a leaf, median span 25 pages, at two requests and under a cent per document. + +Resolver only, applied to the tree the baseline produced, page 0 → resolved: + +| document | pages | leaves with a page | requests | tokens | elapsed | cost | +|---|---|---|---|---|---|---| +| ADOBE_2022_10K | 88 | 0 → **24 / 24** | 2 | 10,382 | 5–11 s | $0.00045 | +| AMAZON_2019_10K | 68 | 2 → **22 / 22** | 2 | 10,013 | 9–11 s | $0.00043 | +| AMCOR_2020_10K | 135 | 3 → **28 / 29** | 2 | 12,507 | 6–14 s | $0.00053 | +| BOEING_2022_10K | 153 | 1 → **23 / 23** | 2 | 10,731 | 11–12 s | $0.00045 | + +Every ADOBE page agrees with the filing's own contents page. The one AMCOR +miss, `Exhibit Index`, is a heading the parser never emits (HAL-1368); +there is nothing on any page to find. + +For scale: the baseline's extraction call that produced these trees took +106–271 s per document on GLM and $0.017–0.030, and left the pages at +zero. Elapsed here includes Jev contents-page detection (one request) and +resolution (one request); the spread is the API, not the document. + +## The evidence-page gate, 17 of 21 filings + +`coverage.py` asks the question that matters for retrieval: is each gold +evidence page inside some leaf, and how big is that leaf? A +whole-document leaf covers everything and locates nothing, so both +columns count. + +The honest baseline is the eleven trees the original binary produced, +whose leaves had no pages. Six more trees came from a 900-second re-run +with the new binary, whose `Build` already resolves pages, so they are +"after" on both sides and are excluded from the before row. + +| run | docs | gold pages covered | coverage | median leaf span | build time | requests | cost | +|---|---|---|---|---|---|---|---| +| baseline trees, GLM extraction, no resolver | 11 | 16 / 20 | 0.800 | **183 p** | 1,840 s | 32 | $0.246 | +| same trees, resolved on Jev | 11 | **20 / 20** | **1.000** | **36 p** | 699 s | 20 | $0.0046 | +| all 17 usable trees, resolved | 17 | **36 / 36** | **1.000** | **25 p** | 979 s | 32 | $0.0080 | + +Leaves with a page across the 17: **205 → 441 of 479**. Of the 205 +before, 192 belong to the six re-run trees; the eleven old-binary trees +had 13 between them. + +The baseline's 0.800 is not location: with no leaf pages, +`deriveEndPages` gives the first leaf of each part the whole part, so a +gold page is "covered" by a 183-page leaf. Resolved, the tightest leaf +holding a gold page is 25 pages at the median, and every gold page in +the set is inside one. + +The resolver's build time is Jev API latency in this window — two +requests per document, a few of them retried — not work: the same +documents ran in 5–12 s an hour earlier. Cost is the number to read. + +Per document, leaves with a page before → after: + +| document | leaves | before → after | note | +|---|---|---|---| +| ADOBE_2022_10K | 24 | 0 → 24 | | +| AMAZON_2019_10K | 22 | 2 → 22 | | +| AMCOR_2020_10K | 29 | 3 → 28 | `Exhibit Index` not in text (HAL-1368) | +| BOEING_2022_10K | 23 | 1 → 23 | | +| COSTCO_2021_10K | 22 | 0 → 22 | | +| INTEL_2021_10K | 26 | 3 → 22 | | +| LOCKHEEDMARTIN_2021_10K | 23 | 1 → 23 | | +| MGMRESORTS_2018_10K | 23 | 0 → 22 | | +| NETFLIX_2017_10K | 20 | 1 → 20 | | +| PEPSICO_2021_10K | 21 | 1 → 21 | detection request failed once; resolved with no exclusions | +| WALMART_2020_10K | 22 | 1 → 22 | | +| 3M_2018_10K | 56 | 39 → 39 | re-run tree, already resolved in Build | +| AMD_2022_10K | 23 | 22 → 22 | re-run tree | +| BESTBUY_2017_10K | 24 | 24 → 24 | re-run tree | +| JOHNSON_JOHNSON_2022_10K | 36 | 30 → 30 | re-run tree | +| KRAFTHEINZ_2019_10K | 63 | 55 → 55 | re-run tree | +| ORACLE_2021_10K | 22 | 22 → 22 | re-run tree | + +Not in the set: GENERALMILLS_2020_10K (parser returns one page, +HAL-1365), NIKE_2019_10K (GLM extraction exceeded 900 s twice), +PFIZER_2021_10K and VERIZON_2022_10K (extraction still running when this +was measured). The four are all extraction, the one generative call left +in this phase, which took 103–828 s per document against the resolver's +two requests. + +The 38 leaves still unplaced on 3M, Johnson & Johnson, Kraft Heinz and +Intel are the next thing to look at; they are deeper note-level titles +and have not been examined yet. + +## What it took, in the order each step recovered pages + +Each row is one change, measured on ADOBE unless stated. The Judge was +never the limiter: with one exception at the end, every miss was a leaf +whose candidate set was empty because Go could not find the heading. + +| step | ADOBE | what was wrong | +|---|---|---| +| page-head window, "begins at the very start" | 12 / 24 | sections that share a page are never at the start of it | +| heuristic contents-page exclusion | 11 / 24 | excluded 30 of 88 pages: "Table of Contents" is a running header on every page. Violates the zero-signal rule; reverted. Exclusion now comes only from what detection *calls* a contents page — [2] | +| "heading on this page", exclusion from detection | 17 / 24 | second-on-page sections (Items 2, 9A, 11–14) recovered | +| **search the whole page for a line-opening hit; show a 600-char window around it** | 20 / 24 | third-and-later sections (Items 3, 4, 9B, 9C) sit past any head-sized window | +| **a part label may precede the heading on its line** | **24 / 24** | the parser joins `PART I` to `ITEM 1. BUSINESS`; every first-in-part section was invisible | +| parser: leaf-cap merge keeps the absorbed title (c6a03b6) | AMCOR 11 → 23, BOEING 15 → 21 | `Item 1B` (body: "None.") + `Item 2. Properties` are the smallest adjacent pair in every 10-K, so `Item 2` was deleted from the text of every filing, and on AMCOR Items 3–5 merged in after it | +| stopwords and punctuation between title words | BOEING +1 | `Exhibits, Financial` in the contents is `Exhibits and Financial` on the page | +| Title-Case typographic tier, short label prefix, only when no line-opening hit exists | AMCOR +4 | `Amcor plc and Subsidiaries Consolidated Balance Sheet (in millions)`; a lowercase `see the consolidated balance sheet` never qualifies | +| loose form for numbered titles: label + number + first word, additive | AMCOR +1 | `Market For Registrant's Common Equity` is `Market for Registrant's Equity` on its page | +| contents-page fallback when exclusion leaves a leaf nothing | BOEING +1 | Item 1 opens on the contents page itself | +| excerpt starts on the hit's own line | (the above) | 150 characters of the previous lines were the contents list; the Judge read the heading as one more entry and said no | + +Two `deriveEndPages` corrections rode along, both visible only once +leaves had pages: a container inherits its first child's start (Part II +with no page of its own left Items 1B–4 running to page 99), and a section +never ends before its own page (Item 9B and Item 10 share page 73 across +the Part II / III boundary; 9B's end came out 72 and was cleared to 0). + +## What did not work, kept for the record + +- **Speculative fan-out** (HAL-1366): asking both detection questions of + every page in one request was 31% *slower*. Speculation pays only when + it removes a request. +- **Head-of-page candidates**: 17 / 24 at best. The head is where the + *first* section of a page is; a 10-K puts four sections on page 34. +- **Any fallback to "the title appears somewhere in the head"**: admitted + every `see Item 1A` cross-reference as a candidate, one question each, + and lost nothing when removed. + +## Method + +- Candidates come from code (`findCandidatePages`): a regexp over the + title's content words, stopwords and punctuation tolerated between + them, matched on every page. A hit is a heading if it opens its line + (a part label or markdown emphasis before it is allowed). Only when + no line-opening hit exists anywhere is a Title-Cased hit with at most + five words before it on the line admitted. Pages detection called a + table of contents are excluded — unless that leaves the leaf nothing. +- Each (leaf, candidate) is one `Noul` on a window from the hit's line + through 450 characters after, batched under the 24k shared-state + ceiling. The best probability above `JudgeThreshold` (default 0.5) + wins; a leaf with no candidate above it keeps whatever extraction + said, since an absent verdict is not a "no". +- More than four candidates for one title means a running header, not a + section; the claimed page is kept if there is one. + +## Reproduce + +```bash +# one document, resolver only, candidate sets printed +go run ./cmd/tocresolve -v -pdf ~/.cache/vlbench/financebench/ADOBE_2022_10K.pdf \ + -tree ~/.cache/vlbench/trees-after/ADOBE_2022_10K.json -out ~/.cache/vlbench/trees-resolved + +# what the pipeline actually searched +go run ./cmd/pagedump ~/.cache/vlbench/financebench/ADOBE_2022_10K.pdf | grep -n -i "item 2" + +# the gate +python cmd/tocdump/coverage.py ~/.cache/vlbench/trees-after ~/.cache/vlbench/trees-resolved +``` From 5dbee7486ed7ac064dd5ee1c6b624ae19e08c63a Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 09:55:34 +0100 Subject: [PATCH 6/7] =?UTF-8?q?docs:=20evaluation=20=E2=80=94=20final=2019?= =?UTF-8?q?-of-21=20corpus=20numbers?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 44/44 gold evidence pages inside a leaf, median span 36 p (baseline 183 p), leaves with a page 238 -> 496 of 543. --- .../2026-09-18-jev-page-resolver.md | 39 +++++++++++-------- 1 file changed, 23 insertions(+), 16 deletions(-) diff --git a/docs/evaluations/2026-09-18-jev-page-resolver.md b/docs/evaluations/2026-09-18-jev-page-resolver.md index e30ebef..7f9c011 100644 --- a/docs/evaluations/2026-09-18-jev-page-resolver.md +++ b/docs/evaluations/2026-09-18-jev-page-resolver.md @@ -6,7 +6,7 @@ **Issues:** HAL-1367 (every 10-K leaf loses its page), HAL-1366 (minimum context) **Question:** extraction returns correct titles and no pages on every long filing, because the generative call sees 48k of a 500k-character body. Can code find the candidates and a Judge pick the page, in one or two requests, without widening that window? -## Result: yes. Every gold evidence page in 17 filings lands inside a leaf, median span 25 pages, at two requests and under a cent per document. +## Result: yes. Every gold evidence page in 19 filings lands inside a leaf, median span 36 pages against 183, at two requests and under a cent per document. Resolver only, applied to the tree the baseline produced, page 0 → resolved: @@ -26,7 +26,7 @@ For scale: the baseline's extraction call that produced these trees took zero. Elapsed here includes Jev contents-page detection (one request) and resolution (one request); the spread is the API, not the document. -## The evidence-page gate, 17 of 21 filings +## The evidence-page gate, 19 of 21 filings `coverage.py` asks the question that matters for retrieval: is each gold evidence page inside some leaf, and how big is that leaf? A @@ -34,7 +34,7 @@ whole-document leaf covers everything and locates nothing, so both columns count. The honest baseline is the eleven trees the original binary produced, -whose leaves had no pages. Six more trees came from a 900-second re-run +whose leaves had no pages. Eight more trees came from a 900-second re-run with the new binary, whose `Build` already resolves pages, so they are "after" on both sides and are excluded from the before row. @@ -42,16 +42,16 @@ with the new binary, whose `Build` already resolves pages, so they are |---|---|---|---|---|---|---|---| | baseline trees, GLM extraction, no resolver | 11 | 16 / 20 | 0.800 | **183 p** | 1,840 s | 32 | $0.246 | | same trees, resolved on Jev | 11 | **20 / 20** | **1.000** | **36 p** | 699 s | 20 | $0.0046 | -| all 17 usable trees, resolved | 17 | **36 / 36** | **1.000** | **25 p** | 979 s | 32 | $0.0080 | +| all 19 usable trees, resolved | 19 | **44 / 44** | **1.000** | **36 p** | 828 s | 36 | $0.0089 | -Leaves with a page across the 17: **205 → 441 of 479**. Of the 205 -before, 192 belong to the six re-run trees; the eleven old-binary trees +Leaves with a page across the 19: **238 → 496 of 543**. Of the 238 +before, 225 belong to the eight re-run trees; the eleven old-binary trees had 13 between them. The baseline's 0.800 is not location: with no leaf pages, `deriveEndPages` gives the first leaf of each part the whole part, so a gold page is "covered" by a 183-page leaf. Resolved, the tightest leaf -holding a gold page is 25 pages at the median, and every gold page in +holding a gold page is 36 pages at the median, and every gold page in the set is inside one. The resolver's build time is Jev API latency in this window — two @@ -79,17 +79,24 @@ Per document, leaves with a page before → after: | JOHNSON_JOHNSON_2022_10K | 36 | 30 → 30 | re-run tree | | KRAFTHEINZ_2019_10K | 63 | 55 → 55 | re-run tree | | ORACLE_2021_10K | 22 | 22 → 22 | re-run tree | +| PFIZER_2021_10K | 40 | 33 → 33 | re-run tree, extraction 840 s | +| VERIZON_2022_10K | 24 | 0 → 22 | re-run tree; Build's own resolver left 0, `tocresolve` placed 22 — see below | Not in the set: GENERALMILLS_2020_10K (parser returns one page, -HAL-1365), NIKE_2019_10K (GLM extraction exceeded 900 s twice), -PFIZER_2021_10K and VERIZON_2022_10K (extraction still running when this -was measured). The four are all extraction, the one generative call left -in this phase, which took 103–828 s per document against the resolver's -two requests. - -The 38 leaves still unplaced on 3M, Johnson & Johnson, Kraft Heinz and -Intel are the next thing to look at; they are deeper note-level titles -and have not been examined yet. +HAL-1365) and NIKE_2019_10K (GLM extraction exceeded 900 s twice). Both +are outside the resolver. Extraction is the one generative call left in +this phase and took 103–840 s per document against the resolver's two +requests. + +VERIZON is worth a look: the re-run's `Build` ran the resolver and left +every leaf at 0, while `tocresolve` on the same tree placed 22 of 24 +minutes later. The likely cause is a failed Judge request inside Build +(logged, falls back to the generative verifier, which rejects the +printed page numbers) — the same silent-degradation shape as HAL-1364. + +The 47 leaves still unplaced on 3M, Johnson & Johnson, Kraft Heinz, +Pfizer and Intel are the next thing to look at; they are deeper +note-level titles and have not been examined yet. ## What it took, in the order each step recovered pages From 735fd7644351e0d99c912cc9d03151ccb303e672 Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 09:59:38 +0100 Subject: [PATCH 7/7] docs: VERIZON zero-page tree confirmed as a failed Judge request falling back silently --- docs/evaluations/2026-09-18-jev-page-resolver.md | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/docs/evaluations/2026-09-18-jev-page-resolver.md b/docs/evaluations/2026-09-18-jev-page-resolver.md index 7f9c011..3ada887 100644 --- a/docs/evaluations/2026-09-18-jev-page-resolver.md +++ b/docs/evaluations/2026-09-18-jev-page-resolver.md @@ -90,9 +90,12 @@ requests. VERIZON is worth a look: the re-run's `Build` ran the resolver and left every leaf at 0, while `tocresolve` on the same tree placed 22 of 24 -minutes later. The likely cause is a failed Judge request inside Build -(logged, falls back to the generative verifier, which rejects the -printed page numbers) — the same silent-degradation shape as HAL-1364. +minutes later. Confirmed from the log: `toc: judge page resolution +failed, falling back: typesafe: request failed` — one Judge request +failed after retries, Build fell back to the generative verifier, and +that verifier rejects printed page numbers. The document ingested with +no pages and reported success — the same silent-degradation shape as +HAL-1364, now in the resolver path (HAL-1369). The 47 leaves still unplaced on 3M, Johnson & Johnson, Kraft Heinz, Pfizer and Intel are the next thing to look at; they are deeper