From c3f36dd55a19747db4f4b4cca4dc95a698ff867a Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 16:21:34 +0100 Subject: [PATCH 1/2] =?UTF-8?q?feat(ingest):=20extract=20the=20table=20of?= =?UTF-8?q?=20contents=20on=20a=20Judge=20=E2=80=94=20code=20lists=20the?= =?UTF-8?q?=20entries,=20Jev=20confirms=20them?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit HAL-1370. The generative extractor was the last minutes-long call in the TOC stage: 103–840 s per FinanceBench filing, the whole wall clock once detection and resolution were on a Judge. It asked a chat model to do a lookup the contents page already answers. Go parses the detected contents pages into candidate entries — the parser runs them together on one line, so an entry closes at a printed page number followed by something that opens an entry — and derives structure from the numbering: Items under their Part, dotted numbers by depth. The Judge answers one Noul per entry, is this a section or the page's furniture, in one request with the contents text shared once. Pages come from the resolver; the printed numbers only calibrate an offset (median of physical − printed over resolved leaves) for whatever the resolver could not place. Fewer than three confirmed entries declines to the generative extractor unchanged. A failed Judge request also falls back, but is recorded on Usage.Degraded. Usage.GenerativeCalls counts chat calls so a Judge-only build is provable; tocdump -judge-only makes any generative call fail loudly. ADOBE_2022_10K, no chat model at all: 3 requests, 0 generative calls, 24/24 leaves, every page agreeing with the contents page, $0.0006. --- cmd/tocdump/main.go | 41 ++- cmd/tocdump/titles.py | 61 ++++ pkg/ingest/toc_builder.go | 43 ++- pkg/ingest/toc_extract_judge.go | 398 +++++++++++++++++++++++++++ pkg/ingest/toc_extract_judge_test.go | 335 ++++++++++++++++++++++ 5 files changed, 861 insertions(+), 17 deletions(-) create mode 100644 cmd/tocdump/titles.py create mode 100644 pkg/ingest/toc_extract_judge.go create mode 100644 pkg/ingest/toc_extract_judge_test.go diff --git a/cmd/tocdump/main.go b/cmd/tocdump/main.go index 61f4796..4344ae0 100644 --- a/cmd/tocdump/main.go +++ b/cmd/tocdump/main.go @@ -17,6 +17,7 @@ import ( "bufio" "context" "encoding/json" + "errors" "flag" "fmt" "os" @@ -35,20 +36,23 @@ import ( ) type dump struct { - Doc string `json:"doc"` - Pages int `json:"pages"` - Seconds float64 `json:"seconds"` - Requests int `json:"requests"` - InTokens int `json:"in_tokens"` - CostUSD float64 `json:"cost_usd"` - Nodes []tree.TOCNode `json:"nodes"` - Err string `json:"err,omitempty"` + Doc string `json:"doc"` + Pages int `json:"pages"` + Seconds float64 `json:"seconds"` + Requests int `json:"requests"` + InTokens int `json:"in_tokens"` + CostUSD float64 `json:"cost_usd"` + Generative int `json:"generative_calls"` + Degraded []string `json:"degraded,omitempty"` + Nodes []tree.TOCNode `json:"nodes"` + Err string `json:"err,omitempty"` } func main() { docs := flag.String("docs", "", "directory of PDFs") out := flag.String("out", "", "directory for per-document tree JSON") noJudge := flag.Bool("no-judge", false, "run detection/verification on the chat model instead") + judgeOnly := flag.Bool("judge-only", false, "no chat model at all: any generative call fails loudly, so the tree is provably Judge-built") minimal := flag.Bool("minimal", false, "TOCBuilder.MinimalContext: prefilter + truncation + two-stage scan") // 300s, not the pipeline's 90s default: GLM's extraction call on the // z.ai gateway routinely exceeds 90s on a 100+ page filing, and a @@ -91,13 +95,18 @@ func main() { } d.Pages = len(pages) - b := &ingest.TOCBuilder{LLM: client, Judge: judge, LLMCallTimeout: *callTimeout, MinimalContext: *minimal} + llm := client + if *judgeOnly { + llm = refusingClient{} + } + b := &ingest.TOCBuilder{LLM: llm, Judge: judge, LLMCallTimeout: *callTimeout, MinimalContext: *minimal} ctx, cancel := context.WithTimeout(context.Background(), 15*time.Minute) start := time.Now() nodes, usage, err := b.Build(ctx, pages) cancel() d.Seconds = time.Since(start).Seconds() d.Requests, d.InTokens, d.CostUSD = usage.LLMCalls, usage.InputTokens, usage.CostUSD + d.Generative, d.Degraded = usage.GenerativeCalls, usage.Degraded if err != nil { d.Err = err.Error() } @@ -105,8 +114,8 @@ func main() { write(*out, d) leaves := countLeaves(nodes) - fmt.Printf(" %-28s %4d pages %6.1fs %3d req %3d leaves $%.4f %s\n", - name, d.Pages, d.Seconds, d.Requests, leaves, d.CostUSD, d.Err) + fmt.Printf(" %-28s %4d pages %6.1fs %3d req %d gen %3d leaves $%.4f %s %s\n", + name, d.Pages, d.Seconds, d.Requests, d.Generative, leaves, d.CostUSD, d.Err, strings.Join(d.Degraded, "; ")) } } @@ -216,3 +225,13 @@ func dotEnv(key string) string { } return "" } + +// refusingClient is the chat model for -judge-only: every call fails, +// so a tree can only have come from the Judge. +type refusingClient struct{} + +func (refusingClient) Complete(context.Context, llmgate.Request) (*llmgate.Response, error) { + return nil, errors.New("tocdump -judge-only: a generative call was attempted") +} + +func (refusingClient) CountTokens(_ context.Context, s string) (int, error) { return len(s) / 4, nil } diff --git a/cmd/tocdump/titles.py b/cmd/tocdump/titles.py new file mode 100644 index 0000000..5acf92f --- /dev/null +++ b/cmd/tocdump/titles.py @@ -0,0 +1,61 @@ +#!/usr/bin/env python3 +"""Compare the leaf titles of two tocdump runs. + + python cmd/tocdump/titles.py trees-after trees-jev + +Reports, per document and overall, what share of the reference run's leaf +titles the candidate run also has (recall), and the reverse (precision), +after normalising case, punctuation and whitespace. Titles are compared +as sets; a title is matched if the normalised forms are equal or one is +a prefix of the other (contents entries are often wrapped or truncated). +""" +import json, re, sys +from pathlib import Path + +def norm(t): + t = t.lower() + t = re.sub(r"[^a-z0-9 ]+", " ", t) + return " ".join(t.split()) + +def leaves(ns): + for n in ns: + if n.get("nodes"): + yield from leaves(n["nodes"]) + else: + yield n + +def titles(path): + d = json.loads(Path(path).read_text()) + return [norm(n["title"]) for n in leaves(d.get("nodes") or [])], d + +def matched(a, bs): + for b in bs: + if a == b or (len(a) >= 12 and (a.startswith(b) or b.startswith(a))): + return True + return False + +def main(ref_dir, cand_dir): + ref_dir, cand_dir = Path(ref_dir), Path(cand_dir) + rows = [] + R = C = RM = CM = 0 + for rp in sorted(ref_dir.glob("*.json")): + cp = cand_dir / rp.name + if not cp.exists(): + continue + rt, rd = titles(rp); ct, cd = titles(cp) + if not rt or not ct: + rows.append((rp.stem, len(rt), len(ct), None, None, cd.get("seconds", 0), cd.get("generative_calls", "?"))) + continue + rm = sum(1 for t in rt if matched(t, ct)); cm = sum(1 for t in ct if matched(t, rt)) + R += len(rt); C += len(ct); RM += rm; CM += cm + rows.append((rp.stem, len(rt), len(ct), rm / len(rt), cm / len(ct), cd.get("seconds", 0), cd.get("generative_calls", "?"))) + print(f"{'document':26} {'ref':>4} {'cand':>4} {'recall':>7} {'prec':>6} {'cand s':>7} gen") + for name, r, c, rec, prec, sec, gen in rows: + rec_s = f"{rec:.3f}" if rec is not None else " n/a" + prec_s = f"{prec:.3f}" if prec is not None else " n/a" + print(f"{name:26} {r:4d} {c:4d} {rec_s:>7} {prec_s:>6} {sec:7.1f} {gen}") + if R and C: + print(f"\noverall: recall {RM/R:.3f} ({RM}/{R}) precision {CM/C:.3f} ({CM}/{C})") + +if __name__ == "__main__": + main(*sys.argv[1:3]) diff --git a/pkg/ingest/toc_builder.go b/pkg/ingest/toc_builder.go index 574ebd4..83f8f34 100644 --- a/pkg/ingest/toc_builder.go +++ b/pkg/ingest/toc_builder.go @@ -137,6 +137,11 @@ type Usage struct { CostUSD float64 LLMCalls int + // GenerativeCalls counts the chat-completion calls among LLMCalls. + // Zero on a Judge-only build; the number to watch when the goal is + // to have no generative call in the TOC stage at all. + GenerativeCalls int + // Degraded lists the Judge-path steps that could not complete and // what Build did instead. Empty means every step ran as configured. // A document built with a non-empty Degraded is not wrong, but it is @@ -162,6 +167,7 @@ func (u *Usage) add(r *llmgate.Response) { u.TotalTokens += r.Usage.TotalTokens u.CostUSD += r.Usage.CostUSD u.LLMCalls++ + u.GenerativeCalls++ } // Build runs the three-phase pipeline on pages and returns a @@ -199,15 +205,34 @@ func (b *TOCBuilder) Build(ctx context.Context, pages []PageText) ([]tree.TOCNod } // Phase 2: extract. + // + // With a Judge and a contents page, code lists the entries and the + // Judge confirms them — one request, seconds. The generative + // extractor runs only when the contents did not parse into enough + // entries to trust, or when the Judge failed (recorded as a + // degradation, since it is minutes and a body window instead). var nodes []tree.TOCNode + var printed map[string]int var err error - if len(tocPages) > 0 { - nodes, err = b.extractFromTOCPages(ctx, pages, tocPages, &usage) - } else { - nodes, err = b.generateNoTOC(ctx, pages, &usage) + extracted := false + if len(tocPages) > 0 && b.Judge != nil { + var handled bool + nodes, printed, handled, err = b.extractFromTOCPagesJudge(ctx, pages, tocPages, &usage) + if err != nil { + log.Printf("toc: judge extraction failed, falling back to the generative extractor: %v", err) + usage.degrade("extraction", "generative extractor after a failed Judge request: "+err.Error()) + } + extracted = handled && err == nil } - if err != nil { - return nil, usage, err + if !extracted { + if len(tocPages) > 0 { + nodes, err = b.extractFromTOCPages(ctx, pages, tocPages, &usage) + } else { + nodes, err = b.generateNoTOC(ctx, pages, &usage) + } + if err != nil { + return nil, usage, err + } } if len(nodes) == 0 { return nil, usage, nil @@ -237,6 +262,12 @@ func (b *TOCBuilder) Build(ctx context.Context, pages []PageText) ([]tree.TOCNod b.verifyTitlesConcurrent(ctx, nodes, pages, concurrency, &usage) } + // Leaves the resolver could not place get their printed page plus + // the offset the resolved leaves agree on. + if n := calibrateFromPrinted(nodes, printed, lastPage(pages)); n > 0 { + log.Printf("toc: %d leaves placed from printed page numbers", n) + } + // Derive end pages from sibling order. Done last so verified // start pages drive the derivation. deriveEndPages(nodes, lastPage(pages)) diff --git a/pkg/ingest/toc_extract_judge.go b/pkg/ingest/toc_extract_judge.go new file mode 100644 index 0000000..14e8534 --- /dev/null +++ b/pkg/ingest/toc_extract_judge.go @@ -0,0 +1,398 @@ +package ingest + +import ( + "context" + "fmt" + "regexp" + "sort" + "strconv" + "strings" + "unicode" + + "github.com/hallelx2/llmgate" + + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +// TOC extraction on a Judge — select, don't generate (HAL-1370). +// +// The generative extractor sends the contents pages plus 48k characters +// of body to a chat model and asks it to write the tree back as JSON. +// That call took 103–840 seconds per FinanceBench filing and was the +// entire wall clock of the TOC stage once detection and page resolution +// were on a Judge. It is asking a generative model to do a lookup: the +// contents page already lists every title, in order, with its printed +// page number. +// +// So: Go parses the contents pages into candidate entries, structure +// comes from the numbering, and the Judge answers one Noul per entry — +// is this a section a reader could turn to, or noise? — in one request. +// Pages are not taken from the printed numbers; the resolver finds +// them, and the printed numbers only calibrate an offset for whatever +// the resolver could not place. + +// contentsEntry is one candidate line of a table of contents. +type contentsEntry struct { + Title string + Printed int // page number as printed in the contents; 0 if none + Depth int // 1 = top level + Container bool // a PART header: kept without asking, carries no page +} + +// minContentsEntries is how many confirmed entries make a contents page +// worth trusting over the generative extractor. +const minContentsEntries = 3 + +// contentsStateMaxChars bounds the contents text shared with every +// question. Two 10-K contents pages are ~3k characters; this is a +// ceiling for the pathological case, well under the shared-state limit. +const contentsStateMaxChars = 12_000 + +var ( + // rePartHeader: "PART I", "Part II.", "**Part III**", alone on its + // line or closing a title line ("... Table of Contents Part I"). + rePartHeader = regexp.MustCompile(`(?i)(?:^|\s)(part\s+(?:[ivxlc]+|\d+))\s*[.:\-—–]?\s*$`) + // reEntryLabel opens a numbered entry: Item 1A., Note 12, Section 3. + reEntryLabel = regexp.MustCompile(`(?i)^(?:item|note|section|chapter|article|appendix|schedule|exhibit)$`) + // reDotted is a dotted heading number: 1, 1.2, 1.2.3, optionally with + // a trailing dot. + reDotted = regexp.MustCompile(`^\d+(?:\.\d+)*\.?$`) + // rePageNum is a printed page number token. + rePageNum = regexp.MustCompile(`^\d{1,3}$`) + // reLeaders strips dot leaders and trailing punctuation from a title. + reLeaders = regexp.MustCompile(`[.\s·…_\-]+$`) + // reNoise: lines that are the table's own furniture, not entries. + reNoise = regexp.MustCompile(`(?i)^(?:page(?:\s*no\.?)?|table\s+of\s+contents|contents|index)$`) +) + +// parseContentsEntries turns the text of the contents pages into +// candidate entries, in reading order. +// +// The parser runs entries together on one line — "Item 1. Business 3 +// Item 1A. Risk Factors 20 …" — so an entry closes at a 1–3 digit page +// number that is followed by the end of the line or by something that +// opens an entry: a label like Item, a dotted number, or a capitalised +// word. "Item 6 [Reserved] 35" survives because "[Reserved]" opens +// nothing; "Form 10-K Summary 96" because "10-K" is not a page number. +// A trailing "2 Table of Contents" footer has no page after it and is +// dropped. +func parseContentsEntries(text string) []contentsEntry { + var out []contentsEntry + inPart := false + for _, raw := range strings.Split(text, "\n") { + line := strings.TrimSpace(strings.ReplaceAll(raw, "**", " ")) + if line == "" { + continue + } + if m := rePartHeader.FindStringSubmatch(line); m != nil { + // A part header, possibly at the end of a title line. Only + // the part becomes an entry; the rest of the line is the + // table's own title. + out = append(out, contentsEntry{Title: strings.ToUpper(strings.Join(strings.Fields(m[1]), " ")), Depth: 1, Container: true}) + inPart = true + continue + } + toks := strings.Fields(line) + var cur []string + flush := func(page int) { + title := cleanContentsTitle(strings.Join(cur, " ")) + cur = nil + if title == "" || reNoise.MatchString(title) || !hasLetter(title) || len(title) < 3 { + return + } + out = append(out, contentsEntry{Title: title, Printed: page, Depth: entryDepth(title, inPart)}) + } + for i := 0; i < len(toks); i++ { + t := toks[i] + if rePageNum.MatchString(t) && len(cur) > 0 && (i+1 == len(toks) || opensEntry(toks[i+1])) { + n, _ := strconv.Atoi(t) + flush(n) + continue + } + cur = append(cur, t) + } + // Whatever is left had no page number: a wrapped title fragment + // or a footer. Neither is an entry on its own. + } + return out +} + +// opensEntry reports whether a token can begin a new contents entry. +func opensEntry(tok string) bool { + if reEntryLabel.MatchString(tok) || reDotted.MatchString(tok) { + return true + } + r := []rune(tok)[0] + return unicode.IsUpper(r) +} + +// entryDepth derives the outline depth from the title's own numbering. +// "Item N" sits under a part when one has been seen; "1.2.3" is depth 3 +// (+1 under a part); anything else is a sibling at the current level. +func entryDepth(title string, inPart bool) int { + base := 1 + if inPart { + base = 2 + } + f := strings.Fields(title) + if len(f) == 0 { + return base + } + if reEntryLabel.MatchString(f[0]) { + return base + } + if reDotted.MatchString(f[0]) { + return base + strings.Count(strings.TrimSuffix(f[0], "."), ".") + } + return base +} + +func cleanContentsTitle(s string) string { + s = reLeaders.ReplaceAllString(strings.TrimSpace(s), "") + return strings.Join(strings.Fields(s), " ") +} + +func hasLetter(s string) bool { + for _, r := range s { + if unicode.IsLetter(r) { + return true + } + } + return false +} + +// confirmEntriesJudge asks one Noul per non-container entry: is this a +// section of the document, or furniture? Returns which entries to keep. +func (b *TOCBuilder) confirmEntriesJudge(ctx context.Context, entries []contentsEntry, contents string, usage *Usage) ([]bool, error) { + keep := make([]bool, len(entries)) + if len(contents) > contentsStateMaxChars { + contents = contents[:contentsStateMaxChars] + } + th := b.judgeThreshold() + + // Batch so that shared contents + per-entry state stays under the + // shared-state ceiling. Entries are ~30 tokens each. + const perEntry = 40 + perBatch := (judgeBudget - estimateTokens(contents) - 200) / perEntry + if perBatch < 20 { + perBatch = 20 + } + for start := 0; start < len(entries); start += perBatch { + end := start + perBatch + if end > len(entries) { + end = len(entries) + } + state := map[string]any{"contents": contents} + questions := map[string]llmgate.Question{} + for i := start; i < end; i++ { + e := entries[i] + if e.Container { + keep[i] = true + continue + } + qk := fmt.Sprintf("e_%d", i) + state[qk] = map[string]any{"title": e.Title, "printed_page": e.Printed} + questions[qk] = llmgate.Noul{ + Instructions: fmt.Sprintf( + "`contents` is the text of a document's table of contents. Is `%s.title` "+ + "(listed at page `%s.printed_page`) one of the document's sections — a heading "+ + "a reader could turn to — rather than the table's own title, a running header "+ + "or footer, a column label such as \"Page\", or a fragment cut from a longer "+ + "title?", qk, qk), + Criteria: &llmgate.NoulCriteria{ + True: "A real section or chapter of the document, as listed", + False: "Furniture of the page, or a broken fragment of a title", + }, + } + } + if len(questions) == 0 { + continue + } + res, err := b.Judge.Judge(ctx, llmgate.JudgeRequest{State: state, Questions: questions}) + if err != nil { + return nil, err + } + addJudgeUsage(usage, res) + for qk := range questions { + p, err := res.Noul(qk) + if err != nil { + continue + } + i, _ := strconv.Atoi(strings.TrimPrefix(qk, "e_")) + keep[i] = p > th + } + } + return keep, nil +} + +// extractFromTOCPagesJudge builds the tree from the contents pages on a +// Judge alone. handled=false with a nil error means the contents did not +// parse into enough entries to trust — the generative extractor should +// run. An error means the Judge failed; the caller decides. +// +// The returned printed map carries each leaf's printed page number, +// keyed the way the resolver keys leaves, for calibrateFromPrinted. +func (b *TOCBuilder) extractFromTOCPagesJudge(ctx context.Context, pages []PageText, tocPages []int, usage *Usage) (nodes []tree.TOCNode, printed map[string]int, handled bool, err error) { + if b.Judge == nil || len(tocPages) == 0 { + return nil, nil, false, nil + } + contents := joinTOCPagesText(pages, tocPages) + entries := parseContentsEntries(contents) + if countRealEntries(entries) < minContentsEntries { + return nil, nil, false, nil + } + keep, err := b.confirmEntriesJudge(ctx, entries, contents, usage) + if err != nil { + return nil, nil, false, err + } + var kept []contentsEntry + for i, e := range entries { + if keep[i] { + kept = append(kept, e) + } + } + if countRealEntries(kept) < minContentsEntries { + return nil, nil, false, nil + } + + // Drop a container that ended up with nothing under it (a "Part V" + // the parser saw in a footer, say). + kept = dropEmptyContainers(kept) + + flat := make([]tocNodePayload, 0, len(kept)) + counters := map[int]int{} + printed = map[string]int{} + seen := map[string]int{} + for _, e := range kept { + d := e.Depth + if d < 1 { + d = 1 + } + counters[d]++ + for k := range counters { + if k > d { + delete(counters, k) + } + } + parts := make([]string, d) + for i := 1; i <= d; i++ { + c := counters[i] + if c == 0 { + c = 1 + counters[i] = 1 + } + parts[i-1] = strconv.Itoa(c) + } + flat = append(flat, tocNodePayload{Structure: strings.Join(parts, "."), Title: e.Title}) + if !e.Container && e.Printed > 0 { + key := "r_" + slugForKey(e.Title) + if seen[key] > 0 { + key = fmt.Sprintf("%s_%d", key, seen[key]) + } + seen[key]++ + printed[key] = e.Printed + } + } + return assembleHierarchy(flat), printed, true, nil +} + +func countRealEntries(es []contentsEntry) int { + n := 0 + for _, e := range es { + if !e.Container { + n++ + } + } + return n +} + +func dropEmptyContainers(es []contentsEntry) []contentsEntry { + var out []contentsEntry + for i, e := range es { + if e.Container { + if i+1 >= len(es) || es[i+1].Container { + continue + } + } + out = append(out, e) + } + return out +} + +// calibrateFromPrinted gives a page to every leaf the resolver could +// not place, from its printed page number and the offset the resolved +// leaves agree on. A 10-K's printed page N is physical page N+k for a +// constant k (the cover and contents come before printed page 1); the +// median of (physical − printed) over the resolved leaves is k. Leaves +// keyed as the resolver keys them. Returns how many were placed. +func calibrateFromPrinted(nodes []tree.TOCNode, printed map[string]int, lastPage int) int { + if len(printed) == 0 { + return 0 + } + var offsets []int + var unplaced []*tree.TOCNode + seen := map[string]int{} + var walk func(ns []tree.TOCNode) + walk = func(ns []tree.TOCNode) { + for i := range ns { + n := &ns[i] + if len(n.Nodes) > 0 { + walk(n.Nodes) + continue + } + key := "r_" + slugForKey(n.Title) + if seen[key] > 0 { + key = fmt.Sprintf("%s_%d", key, seen[key]) + } + seen[key]++ + pp, ok := printed[key] + if !ok || pp <= 0 { + continue + } + if n.StartPage > 0 { + offsets = append(offsets, n.StartPage-pp) + } else { + unplaced = append(unplaced, n) + } + } + } + walk(nodes) + if len(offsets) < 2 || len(unplaced) == 0 { + return 0 + } + sort.Ints(offsets) + k := offsets[len(offsets)/2] + placed := 0 + seen = map[string]int{} + // Second pass to read the printed page of each unplaced leaf again + // (keys were consumed in order above; recompute the same way). + var walk2 func(ns []tree.TOCNode) + walk2 = func(ns []tree.TOCNode) { + for i := range ns { + n := &ns[i] + if len(n.Nodes) > 0 { + walk2(n.Nodes) + continue + } + key := "r_" + slugForKey(n.Title) + if seen[key] > 0 { + key = fmt.Sprintf("%s_%d", key, seen[key]) + } + seen[key]++ + if n.StartPage != 0 { + continue + } + pp, ok := printed[key] + if !ok || pp <= 0 { + continue + } + p := pp + k + if p >= 1 && (lastPage == 0 || p <= lastPage) { + n.StartPage = p + placed++ + } + } + } + walk2(nodes) + return placed +} diff --git a/pkg/ingest/toc_extract_judge_test.go b/pkg/ingest/toc_extract_judge_test.go new file mode 100644 index 0000000..2ce315c --- /dev/null +++ b/pkg/ingest/toc_extract_judge_test.go @@ -0,0 +1,335 @@ +package ingest + +import ( + "context" + "errors" + "strings" + "sync/atomic" + "testing" + + "github.com/hallelx2/llmgate" + + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +// Contents pages exactly as the parser hands them to the builder. +const adobeContents = `ADOBE INC. FORM 10-K TABLE OF CONTENTS +Page No. + +PART I +Item 1. Business 3 Item 1A. Risk Factors 20 Item 1B. Unresolved Staff Comments 34 Item 2. Properties 34 Item 3. Legal Proceedings 34 Item 4. Mine Safety Disclosures 34 + +PART II +Item 5. Market for Registrant's Common Equity, Related Stockholder Matters and Issuer Purchases of Equity Securities 35 Item 6 [Reserved] 35 Item 7. Management's Discussion and Analysis of Financial Condition and Results of Operations 36 Item 7A. Quantitative and Qualitative Disclosures About Market Risk 50 Item 8. Financial Statements and Supplementary Data 52 Item 9. Changes in and Disagreements with Accountants on Accounting and Financial Disclosure 92 Item 9A. Controls and Procedures 92 Item 9B. Other Information 92 Item 9C. Disclosure Regarding Foreign Jurisdictions That Prevent Inspections 92 + +PART III +Item 10. Directors, Executive Officers and Corporate Governance 93 Item 11. Executive Compensation 93 Item 12. Security Ownership of Certain Beneficial Owners and Management and Related Stockholder Matters 93 Item 13. Certain Relationships and Related Transactions, and Director Independence 93 Item 14. Principal Accounting Fees and Services 93 + +PART IV +Item 15. Exhibits, Financial Statement Schedules 94 Item 16. Form 10-K Summary 96 Signatures 97 Summary of Trademarks 99 2 Table of Contents +` + +const amcorContents = `DOCUMENTS INCORPORATED BY REFERENCE +Certain information required for Part III of this Annual Report on Form 10-K is incorporated by reference to the Amcor plc definitive Proxy Statement. + +**Amcor plc Annual Report on Form 10-K Table of Contents Part I** + +Item 1. Business 6 Item 1A. Risk Factors 10 Item 1B. Unresolved Staff Comments 20 Item 2. Properties 20 Item 3. Legal Proceedings 20 Item 4. Mine Safety Disclosures 20 + +**Part II** + +Item 5. Market For Registrant’s Common Equity, Related Shareholder Matters and Issuer Purchases of Equity Securities 21 Item 6. Selected Financial Data 23 Item 7. Management’s Discussion and Analysis of Financial Condition and Results of Operations 24 + +**Part IV** + +Item 15. Exhibits and Financial Statement Schedules 116 Exhibit Index 116 Item 16. Form 10-K Summary (optional) 120 Signatures 121 3 +` + +const boeingContents = `THE BOEING COMPANY Index to the Form 10-K For the Fiscal Year Ended December 31, 2022 PART I Page +Item 1. Business 1 Item 1A. Risk Factors 6 Item 1B. Unresolved Staff Comments 17 Item 2. Properties 18 Item 3. Legal Proceedings 18 Item 4. Mine Safety Disclosures 18 + +**PART II** + +Item 5. Market for Registrant’s Common Equity, Related Stockholder Matters and Issuer Purchases of Equity Securities 19 Item 6. [Reserved] 19 Item 7. Management’s Discussion and Analysis of Financial Condition and Results of Operations 20 + +**PART IV** + +Item 15. Exhibits, Financial Statement Schedules 128 Item 16. Form 10-K Summary 131 Signatures 132 Table of Contents + +**PART I Item 1. Business** + +The Boeing Company, together with its subsidiaries, is one of the world’s major aerospace firms. +` + +func titlesOf(es []contentsEntry) []string { + out := make([]string, 0, len(es)) + for _, e := range es { + out = append(out, e.Title) + } + return out +} + +func hasTitle(es []contentsEntry, t string) *contentsEntry { + for i := range es { + if es[i].Title == t { + return &es[i] + } + } + return nil +} + +func TestParseContentsAdobe(t *testing.T) { + es := parseContentsEntries(adobeContents) + want := map[string]int{ + "Item 1. Business": 3, "Item 1A. Risk Factors": 20, "Item 1B. Unresolved Staff Comments": 34, + "Item 2. Properties": 34, "Item 4. Mine Safety Disclosures": 34, + "Item 5. Market for Registrant's Common Equity, Related Stockholder Matters and Issuer Purchases of Equity Securities": 35, + "Item 6 [Reserved]": 35, "Item 7A. Quantitative and Qualitative Disclosures About Market Risk": 50, + "Item 9C. Disclosure Regarding Foreign Jurisdictions That Prevent Inspections": 92, + "Item 14. Principal Accounting Fees and Services": 93, + "Item 15. Exhibits, Financial Statement Schedules": 94, "Item 16. Form 10-K Summary": 96, + "Signatures": 97, "Summary of Trademarks": 99, + } + for title, page := range want { + e := hasTitle(es, title) + if e == nil { + t.Errorf("missing %q in %v", title, titlesOf(es)) + continue + } + if e.Printed != page { + t.Errorf("%q printed page: got %d want %d", title, e.Printed, page) + } + if e.Depth != 2 { + t.Errorf("%q depth: got %d want 2 (under a part)", title, e.Depth) + } + } + if countRealEntries(es) != 24 { + t.Errorf("real entries: got %d want 24: %v", countRealEntries(es), titlesOf(es)) + } + parts := 0 + for _, e := range es { + if e.Container { + parts++ + } + } + if parts != 4 { + t.Errorf("parts: got %d want 4", parts) + } + for _, junk := range []string{"Page No.", "Table of Contents", "2 Table of Contents", "ADOBE INC. FORM 10-K TABLE OF CONTENTS"} { + if hasTitle(es, junk) != nil { + t.Errorf("junk entry admitted: %q", junk) + } + } +} + +func TestParseContentsAmcorPartAtEndOfTitleLine(t *testing.T) { + es := parseContentsEntries(amcorContents) + if e := hasTitle(es, "PART I"); e == nil || !e.Container { + t.Fatalf("PART I closing the table's title line should be a container: %v", titlesOf(es)) + } + if e := hasTitle(es, "Item 6. Selected Financial Data"); e == nil || e.Printed != 23 { + t.Errorf("Item 6: %+v", e) + } + if e := hasTitle(es, "Item 16. Form 10-K Summary (optional)"); e == nil || e.Printed != 120 { + t.Errorf("Item 16 with a parenthetical: %+v", e) + } + if e := hasTitle(es, "Exhibit Index"); e == nil || e.Printed != 116 { + t.Errorf("Exhibit Index: %+v", e) + } + // The prose paragraph has no page numbers and yields nothing. + for _, e := range es { + if strings.HasPrefix(e.Title, "Certain information") { + t.Errorf("prose admitted as an entry: %q", e.Title) + } + } +} + +func TestParseContentsBoeingBodyBelowTheList(t *testing.T) { + es := parseContentsEntries(boeingContents) + if e := hasTitle(es, "Item 6. [Reserved]"); e == nil || e.Printed != 19 { + t.Errorf("Item 6. [Reserved]: %+v", e) + } + if e := hasTitle(es, "Signatures"); e == nil || e.Printed != 132 { + t.Errorf("Signatures: %+v", e) + } + // "PART I Item 1. Business" is the body heading on the same page; + // it has no page number and is not an entry. The Judge would reject + // a fragment; the parser must not even produce one. + for _, e := range es { + if strings.HasPrefix(e.Title, "The Boeing Company, together") { + t.Errorf("body prose admitted: %q", e.Title) + } + } +} + +func TestParseContentsDottedNumbering(t *testing.T) { + es := parseContentsEntries("1 Introduction 3\n1.1 Background 4\n1.2 Scope 6\n2 Methods 9\n2.1.1 Sampling 10") + depths := map[string]int{"1 Introduction": 1, "1.1 Background": 2, "1.2 Scope": 2, "2 Methods": 1, "2.1.1 Sampling": 3} + for title, d := range depths { + e := hasTitle(es, title) + if e == nil { + t.Errorf("missing %q: %v", title, titlesOf(es)) + continue + } + if e.Depth != d { + t.Errorf("%q depth got %d want %d", title, e.Depth, d) + } + } +} + +func TestCalibrateFromPrinted(t *testing.T) { + nodes := []tree.TOCNode{{Title: "PART I", Nodes: []tree.TOCNode{ + {Title: "Item 1. Business", StartPage: 3}, + {Title: "Item 1A. Risk Factors", StartPage: 20}, + {Title: "Item 1B. Unresolved Staff Comments"}, // unplaced + {Title: "Item 2. Properties"}, // unplaced + }}} + printed := map[string]int{ + "r_" + slugForKey("Item 1. Business"): 1, "r_" + slugForKey("Item 1A. Risk Factors"): 18, + "r_" + slugForKey("Item 1B. Unresolved Staff Comments"): 32, "r_" + slugForKey("Item 2. Properties"): 32, + } + n := calibrateFromPrinted(nodes, printed, 99) + if n != 2 { + t.Fatalf("placed %d want 2", n) + } + if got := nodes[0].Nodes[2].StartPage; got != 34 { + t.Errorf("1B: got %d want 34 (printed 32 + offset 2)", got) + } + if got := nodes[0].Nodes[3].StartPage; got != 34 { + t.Errorf("Item 2: got %d want 34", got) + } + // One resolved leaf is not enough to trust an offset. + single := []tree.TOCNode{{Title: "A", StartPage: 5}, {Title: "B"}} + if calibrateFromPrinted(single, map[string]int{"r_" + slugForKey("A"): 3, "r_" + slugForKey("B"): 7}, 99) != 0 { + t.Errorf("a single offset should not be trusted") + } +} + +// End to end on a Judge: detection says the contents page is one, the +// entries are parsed and confirmed, the resolver places them — and the +// generative extractor is never called. +func TestBuildExtractsOnTheJudgeWithNoGenerativeCall(t *testing.T) { + llm := &scriptedLLM{} + var generativeExtract atomic.Int32 + llm.route = func(prompt string) string { + if strings.Contains(prompt, "hierarchical tree structure") { + generativeExtract.Add(1) + } + return `{"toc_detected":"no"}` + } + judge := &llmgate.MockJudge{Respond: func(_ context.Context, req llmgate.JudgeRequest) (*llmgate.Judgment, error) { + st, _ := req.State.(map[string]any) + ans := map[string]llmgate.Answer{} + for id := range req.Questions { + p := 0.0 + switch { + case strings.HasPrefix(id, "e_"): + // Confirm every entry except one planted piece of furniture. + item := st[id].(map[string]any) + if item["title"] != "Page No." { + p = 0.9 + } + case strings.HasPrefix(id, "r_"): + item := st[id].(map[string]any) + if strings.HasPrefix(item["excerpt"].(string), item["title"].(string)) { + p = 0.9 + } + default: + // Detection: the page that says Table of Contents. + if item, ok := st[id].(map[string]any); ok { + for _, v := range item { + if s, ok := v.(string); ok && strings.Contains(s, "TABLE OF CONTENTS") { + p = 0.9 + } + } + } else if s, ok := st[id].(string); ok && strings.Contains(s, "TABLE OF CONTENTS") { + p = 0.9 + } + } + ans[id] = llmgate.NoulAnswer{Noul: p} + } + return &llmgate.Judgment{Model: "mock", Answers: ans, Usage: llmgate.Usage{InputTokens: 10, TotalTokens: 10, TokensReported: true}}, nil + }} + pages := []PageText{ + {PageNumber: 1, Text: "Cover"}, + {PageNumber: 2, Text: "ACME FORM 10-K TABLE OF CONTENTS\nPage No.\nPART I\nItem 1. Business 1 Item 1A. Risk Factors 4 Item 2. Properties 9\nPART II\nItem 5. Market 11 Signatures 12"}, + {PageNumber: 3, Text: "PART I ITEM 1. BUSINESS\nWe make things."}, + {PageNumber: 6, Text: "Item 1A. Risk Factors\nThings go wrong."}, + {PageNumber: 11, Text: "Item 2. Properties\nWe own a shed."}, + {PageNumber: 13, Text: "PART II Item 5. Market\nShares trade."}, + {PageNumber: 14, Text: "Signatures\nSigned."}, + } + b := &TOCBuilder{LLM: llm, Judge: judge, TOCCheckPages: 3, Concurrency: 2, MinimalContext: true} + nodes, usage, err := b.Build(context.Background(), pages) + if err != nil { + t.Fatalf("Build: %v", err) + } + if generativeExtract.Load() != 0 || usage.GenerativeCalls != 0 { + t.Errorf("generative extractor ran: route=%d usage=%d", generativeExtract.Load(), usage.GenerativeCalls) + } + if len(usage.Degraded) != 0 { + t.Errorf("unexpected degradation: %v", usage.Degraded) + } + if len(nodes) != 2 || nodes[0].Title != "PART I" || nodes[1].Title != "PART II" { + t.Fatalf("top level: %+v", nodes) + } + got := map[string]int{} + for _, part := range nodes { + for _, n := range part.Nodes { + got[n.Title] = n.StartPage + } + } + want := map[string]int{"Item 1. Business": 3, "Item 1A. Risk Factors": 6, "Item 2. Properties": 11, "Item 5. Market": 13, "Signatures": 14} + for title, page := range want { + if got[title] != page { + t.Errorf("%q: got page %d want %d (all: %v)", title, got[title], page, got) + } + } + if _, ok := got["Page No."]; ok { + t.Errorf("furniture the Judge rejected was kept") + } + if nodes[0].Nodes[0].Structure != "1.1" || nodes[1].Nodes[0].Structure != "2.1" { + t.Errorf("structure: %q %q", nodes[0].Nodes[0].Structure, nodes[1].Nodes[0].Structure) + } +} + +// A failed Judge extraction is a degradation, not a silent fallback. +func TestBuildRecordsAFailedJudgeExtraction(t *testing.T) { + llm := &scriptedLLM{} + llm.route = func(prompt string) string { + if strings.Contains(prompt, "hierarchical tree structure") { + return `{"nodes":[{"structure":"1","title":"Item 1. Business","physical_index":""}]}` + } + return `{"toc_detected":"no"}` + } + judge := &llmgate.MockJudge{Respond: func(_ context.Context, req llmgate.JudgeRequest) (*llmgate.Judgment, error) { + for id := range req.Questions { + if strings.HasPrefix(id, "e_") { + return nil, errors.New("typesafe: request failed") + } + } + ans := map[string]llmgate.Answer{} + for id := range req.Questions { + ans[id] = llmgate.NoulAnswer{Noul: 0.9} + } + return &llmgate.Judgment{Model: "mock", Answers: ans, Usage: llmgate.Usage{TokensReported: true}}, nil + }} + pages := []PageText{ + {PageNumber: 2, Text: "TABLE OF CONTENTS\nItem 1. Business 1 Item 1A. Risk Factors 4 Item 2. Properties 9"}, + {PageNumber: 3, Text: "Item 1. Business\nWe make things."}, + } + b := &TOCBuilder{LLM: llm, Judge: judge, TOCCheckPages: 2, MinimalContext: true} + nodes, usage, err := b.Build(context.Background(), pages) + if err != nil { + t.Fatalf("Build: %v", err) + } + if usage.GenerativeCalls == 0 { + t.Errorf("the generative extractor should have run as the fallback") + } + if len(usage.Degraded) != 1 || !strings.Contains(usage.Degraded[0], "extraction") { + t.Errorf("degradation not recorded: %v", usage.Degraded) + } + if len(nodes) != 1 || nodes[0].Title != "Item 1. Business" { + t.Errorf("fallback tree: %+v", nodes) + } +} From 0c9b9b78ac7b6650b62a44a2298811d4c6a99256 Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 17:02:06 +0100 Subject: [PATCH 2/2] fix(ingest): contents parser learns the corpus's shapes; evaluation of the Judge-only TOC stage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Glued item numbers (Item 1Business), inline part labels (PART II Item 5.), a title wrapped around its own page number, an item with no page, parenthesised sub-entries, and the parser's bold-fold duplicates — each found by comparing the Jev tree with the GLM tree on one filing. FinanceBench, 21 filings, no chat model: 20 trees (NIKE, which GLM never extracted, now has one), 45/45 gold evidence pages inside a leaf, leaf title recall 0.961 / precision 0.965 against the GLM trees, 0 generative calls, $0.0003–0.0015 per document. --- .../2026-09-18-toc-extraction-on-a-judge.md | 111 ++++++++++++++ pkg/ingest/toc_extract_judge.go | 145 ++++++++++++++++-- pkg/ingest/toc_extract_judge_test.go | 72 +++++++++ 3 files changed, 315 insertions(+), 13 deletions(-) create mode 100644 docs/evaluations/2026-09-18-toc-extraction-on-a-judge.md diff --git a/docs/evaluations/2026-09-18-toc-extraction-on-a-judge.md b/docs/evaluations/2026-09-18-toc-extraction-on-a-judge.md new file mode 100644 index 0000000..adc1ec6 --- /dev/null +++ b/docs/evaluations/2026-09-18-toc-extraction-on-a-judge.md @@ -0,0 +1,111 @@ +# TOC extraction on a Judge — the TOC stage with no generative model + +**Date:** 2026-09-18 +**Harness:** [`cmd/tocdump -judge-only`](../../cmd/tocdump/main.go) (any chat-completion call fails loudly), [`cmd/tocdump/titles.py`](../../cmd/tocdump/titles.py) (leaf-title recall against the GLM trees), [`cmd/tocdump/coverage.py`](../../cmd/tocdump/coverage.py) (evidence-page gate) +**Corpus:** FinanceBench, 21 10-K filings; gold evidence pages from `PatronusAI/financebench` +**Issue:** HAL-1370 +**Question:** with detection and page resolution on Jev, extraction was the only generative call left in the TOC stage and all of its wall clock — 103–840 s per filing. Can code list the contents entries and Jev confirm them, with the same tree at the end? + +## Result: yes. The whole TOC stage runs with no chat model, in one to two minutes of API time per filing, for under a tenth of a cent, and lands every gold evidence page. + +| | GLM extraction + Jev resolver ([previous evaluation](2026-09-18-jev-page-resolver.md)) | **Jev only** | +|---|---|---| +| documents with a tree | 19 / 21 | **20 / 21** | +| generative calls per document | 1 | **0** | +| gold evidence pages inside a leaf | 44 / 44 | **45 / 45** | +| median leaf span | 36 p | 37 p | +| TOC stage per document | 103–840 s (extraction) + resolver | 12–143 s, mean 44 s, all of it Jev latency | +| cost per document | $0.017–0.095 | **$0.0003–0.0015** | +| requests per document | 3 | 3 | + +NIKE_2019_10K, which GLM never extracted (past 900 s twice), has a +21-leaf tree. GENERALMILLS_2020_10K is still the one-page parse +(HAL-1365); it is the only filing without a tree. + +## Are the trees the same trees? + +Leaf titles of the Jev-built tree against the GLM-built tree, per +document, after normalising case and punctuation and allowing a prefix +match for wrapped titles: + +| document | GLM leaves | Jev leaves | recall | precision | +|---|---|---|---|---| +| ADOBE, AMAZON, AMD, BESTBUY, COSTCO, NETFLIX, ORACLE, PEPSICO, WALMART | 20–24 | same | 1.000 | 1.000 | +| BOEING | 23 | 24 | 1.000 | 1.000 | +| LOCKHEEDMARTIN | 23 | 24 | 1.000 | 0.958 | +| JOHNSON_JOHNSON | 36 | 36 | 0.972 | 0.972 | +| KRAFTHEINZ | 63 | 65 | 0.968 | 0.954 | +| AMCOR | 29 | 30 | 0.966 | 0.933 | +| VERIZON | 24 | 23 | 0.958 | 1.000 | +| MGMRESORTS | 23 | 24 | 0.957 | 0.917 | +| 3M | 56 | 55 | 0.893 | 0.909 | +| PFIZER | 40 | 37 | 0.875 | 0.946 | +| INTEL | 26 | 25 | 0.846 | 0.880 | +| **overall** | 543 | 543 | **0.961** | **0.965** | + +The GLM tree is the reference only because it existed first; where the +two differ it is not always GLM that is right. The misses on PFIZER are +unnumbered sub-headings GLM read out of the body (`Available +Information`, `Forward-Looking Information…`) that the contents page +does not list; the extra on LOCKHEEDMARTIN is a real entry. INTEL's +contents page lists `Fundamentals of Our Business` and similar headings +under a different layout that the parser splits into fewer entries; it +is the one document under 0.9 and worth a look on its own. + +## What the parser had to learn + +The first corpus pass scored 0.866 recall with WALMART at 0.045. Every +miss was a shape of contents text, not a Judge verdict: + +| shape | filing | handling | +|---|---|---| +| no space between the item number and its title: `Item 1Business 7 Item 1ARisk Factors 14` | WALMART | re-insert the space; a number directly after a label is never a page | +| part label inline before its first item: `PART II Item 5.`, `PART I. Item 1.` | VERIZON, ORACLE, WALMART | the label becomes a container, the run restarts | +| a title wrapped so its page lands mid-title: `… Issuer Purchases of 20 Equity Securities Item 6.` | VERIZON | the words before the next label are the previous entry's tail | +| an item with no page at all: `Item 1. Business` alone on its line | VERIZON | an item-labelled run is an entry; the resolver finds its page | +| parenthesised sub-entries: `… SCHEDULES 111 15(a)(1) Financial Statements 111` | PFIZER | a digit opens an entry | +| the same line twice, once truncated in bold, from the parser's empty-leaf fold | 3M | exact duplicates dropped; a page-less prefix of a fuller entry dropped | +| a column label fused to the first entry: `Page Part I Item 1…` | WALMART | leading furniture skipped | + +Second pass: recall 0.961, precision 0.965, WALMART 1.000. + +## Method + +- **Candidates from code.** `parseContentsEntries` tokenises each line + of the detected contents pages; an entry closes at a 1–3 digit token + followed by the end of the line or something that opens an entry (a + label such as Item or Note, a number, a capitalised word). Structure + comes from the numbering: Items under the enclosing Part, dotted + numbers by depth. +- **One Judge request.** A `Noul` per entry — *is this a section a + reader could turn to, rather than the table's own title, a running + header, a column label, or a fragment?* — with the contents text + shared once (~3k characters) and the entry as per-question state. +- **Pages from the resolver** (HAL-1367), which searches the body for + each title. The printed page numbers are not trusted for placement; + after resolution they calibrate an offset — the median of physical + minus printed over the resolved leaves — that places whatever the + resolver could not. +- **Fallbacks.** Fewer than three confirmed entries declines to the + generative extractor unchanged. A failed Judge request also falls + back, recorded on `Usage.Degraded`. `Usage.GenerativeCalls` makes a + Judge-only build provable; `tocdump -judge-only` makes any generative + call an error. + +## What this means for the pipeline + +Detection, extraction, resolution: three requests, no generative call, +roughly a cent per hundred filings. The wall clock is now the Judge +API's latency in the hour of the run (the same documents took 5–12 s an +hour earlier and 40–140 s in this pass), not the work. The remaining +generative steps in ingest are the per-section summaries, HyDE +questions and multi-axis summaries, none of which the page-based +retrieval strategy needs. + +## Reproduce + +```bash +go run ./cmd/tocdump -docs ~/.cache/vlbench/financebench -out /tmp/trees-jev -judge-only -minimal +python cmd/tocdump/titles.py ~/.cache/vlbench/trees-after /tmp/trees-jev +python cmd/tocdump/coverage.py ~/.cache/vlbench/trees-resolved /tmp/trees-jev +``` diff --git a/pkg/ingest/toc_extract_judge.go b/pkg/ingest/toc_extract_judge.go index 14e8534..a430efb 100644 --- a/pkg/ingest/toc_extract_judge.go +++ b/pkg/ingest/toc_extract_judge.go @@ -65,18 +65,41 @@ var ( reNoise = regexp.MustCompile(`(?i)^(?:page(?:\s*no\.?)?|table\s+of\s+contents|contents|index)$`) ) +var ( + // reGluedLabel: the parser sometimes drops the space between an + // item number and its title — "Item 1Business", "Item 1ARisk + // Factors" (WALMART_2020_10K). + reGluedLabel = regexp.MustCompile(`\b((?i:item|note|part))\s*(\d+[A-C]?)([A-Z][a-z])`) + // reInlinePart is a part label appearing mid-line, before the item + // that opens it: "PART II Item 5.", "Part III. Item 10". + reRoman = regexp.MustCompile(`(?i)^(?:[ivxlc]+|\d+)\.?$`) + rePartTok = regexp.MustCompile(`(?i)^part$`) + // reLabelNum: the number after an entry label — 1, 1A, 12., 7A. + reLabelNum = regexp.MustCompile(`(?i)^\d+[a-c]?\.?$`) +) + // parseContentsEntries turns the text of the contents pages into // candidate entries, in reading order. // // The parser runs entries together on one line — "Item 1. Business 3 // Item 1A. Risk Factors 20 …" — so an entry closes at a 1–3 digit page // number that is followed by the end of the line or by something that -// opens an entry: a label like Item, a dotted number, or a capitalised -// word. "Item 6 [Reserved] 35" survives because "[Reserved]" opens -// nothing; "Form 10-K Summary 96" because "10-K" is not a page number. -// A trailing "2 Table of Contents" footer has no page after it and is +// opens an entry: a label like Item, a number, or a capitalised word. +// "Item 6 [Reserved] 35" survives because "[Reserved]" opens nothing; +// "Form 10-K Summary 96" because "10-K" is not a page number. A +// trailing "2 Table of Contents" footer has no page after it and is // dropped. +// +// Shapes met on FinanceBench and handled here: a part label inline +// before its first item; a title wrapped so that its page number lands +// mid-title ("… Issuer Purchases of 20 Equity Securities Item 6."), +// where the words before the next label are the previous entry's tail; +// an item with no page number at all ("Item 1. Business" alone on its +// line); the space missing between an item number and its title; and +// the same line emitted twice, once truncated in bold, by the parser's +// empty-leaf fold. func parseContentsEntries(text string) []contentsEntry { + text = reGluedLabel.ReplaceAllString(text, "$1 $2 $3") var out []contentsEntry inPart := false for _, raw := range strings.Split(text, "\n") { @@ -85,16 +108,25 @@ func parseContentsEntries(text string) []contentsEntry { continue } if m := rePartHeader.FindStringSubmatch(line); m != nil { - // A part header, possibly at the end of a title line. Only - // the part becomes an entry; the rest of the line is the - // table's own title. - out = append(out, contentsEntry{Title: strings.ToUpper(strings.Join(strings.Fields(m[1]), " ")), Depth: 1, Container: true}) + out = append(out, partEntry(m[1])) inPart = true continue } toks := strings.Fields(line) var cur []string flush := func(page int) { + if len(cur) == 0 { + return + } + // A label later in the run means what precedes it is the + // previous entry's wrapped tail, not this entry's title. + if k := firstLabelAt(cur); k > 0 { + tail := cleanContentsTitle(strings.Join(cur[:k], " ")) + if j := lastRealEntry(out); j >= 0 && tail != "" && !reNoise.MatchString(tail) { + out[j].Title = out[j].Title + " " + tail + } + cur = cur[k:] + } title := cleanContentsTitle(strings.Join(cur, " ")) cur = nil if title == "" || reNoise.MatchString(title) || !hasLetter(title) || len(title) < 3 { @@ -104,17 +136,103 @@ func parseContentsEntries(text string) []contentsEntry { } for i := 0; i < len(toks); i++ { t := toks[i] - if rePageNum.MatchString(t) && len(cur) > 0 && (i+1 == len(toks) || opensEntry(toks[i+1])) { + // A column label at the start of the run ("Page"). + if len(cur) == 0 && reNoise.MatchString(strings.Trim(t, ".:")) { + continue + } + // A part label inline, before the item that opens it. + if rePartTok.MatchString(t) && i+1 < len(toks) && reRoman.MatchString(toks[i+1]) { + if len(cur) > 0 && firstLabelAt(cur) == 0 { + flush(0) // an item with no page number, closed by the next part + } + cur = nil + out = append(out, partEntry(t+" "+toks[i+1])) + inPart = true + i++ + continue + } + // A number right after a label is the label's number ("Item 1 + // Business"), never a page. + afterLabel := len(cur) > 0 && reEntryLabel.MatchString(cur[len(cur)-1]) + if rePageNum.MatchString(t) && len(cur) > 0 && !afterLabel && (i+1 == len(toks) || opensEntry(toks[i+1])) { n, _ := strconv.Atoi(t) flush(n) continue } cur = append(cur, t) } - // Whatever is left had no page number: a wrapped title fragment - // or a footer. Neither is an entry on its own. + // What is left had no page number. An item-labelled run is still + // an entry — "Item 1. Business" alone on its line; the resolver + // finds its page. Anything else is a wrapped fragment or footer. + if len(cur) > 0 && firstLabelAt(cur) == 0 { + flush(0) + } } - return out + return dedupEntries(out) +} + +func partEntry(label string) contentsEntry { + label = strings.TrimRight(strings.Join(strings.Fields(label), " "), ".:") + return contentsEntry{Title: strings.ToUpper(label), Depth: 1, Container: true} +} + +// firstLabelAt returns the index of the first "Item 1A"-shaped label in +// the run, or -1. +func firstLabelAt(toks []string) int { + for i := 0; i+1 < len(toks); i++ { + if reEntryLabel.MatchString(toks[i]) && reLabelNum.MatchString(toks[i+1]) { + return i + } + } + return -1 +} + +func lastRealEntry(es []contentsEntry) int { + for j := len(es) - 1; j >= 0; j-- { + if !es[j].Container { + return j + } + } + return -1 +} + +// dedupEntries drops an exact repeat (the parser's bold-fold emits a +// truncated copy of a line before the line itself) and a page-less +// entry whose title is a prefix of a fuller one. +func dedupEntries(es []contentsEntry) []contentsEntry { + var out []contentsEntry + seen := map[string]bool{} + for _, e := range es { + if e.Container { + out = append(out, e) + continue + } + k := normalise(e.Title) + "|" + strconv.Itoa(e.Printed) + if seen[k] { + continue + } + seen[k] = true + out = append(out, e) + } + // Prefix fragments without a page. + kept := out[:0] + for i, e := range out { + if !e.Container && e.Printed == 0 { + n := normalise(e.Title) + frag := false + for j, o := range out { + if j != i && !o.Container && len(normalise(o.Title)) > len(n) && strings.HasPrefix(normalise(o.Title), n) { + frag = true + break + } + } + if frag { + continue + } + } + kept = append(kept, e) + } + return kept } // opensEntry reports whether a token can begin a new contents entry. @@ -123,7 +241,8 @@ func opensEntry(tok string) bool { return true } r := []rune(tok)[0] - return unicode.IsUpper(r) + // A digit opens "15(a)(1) Financial Statements" and "2 Methods". + return unicode.IsUpper(r) || unicode.IsDigit(r) } // entryDepth derives the outline depth from the title's own numbering. diff --git a/pkg/ingest/toc_extract_judge_test.go b/pkg/ingest/toc_extract_judge_test.go index 2ce315c..3ee7e89 100644 --- a/pkg/ingest/toc_extract_judge_test.go +++ b/pkg/ingest/toc_extract_judge_test.go @@ -333,3 +333,75 @@ func TestBuildRecordsAFailedJudgeExtraction(t *testing.T) { t.Errorf("fallback tree: %+v", nodes) } } + +func TestParseContentsShapesFromTheCorpus(t *testing.T) { + t.Run("walmart: glued item numbers and an inline part after a column label", func(t *testing.T) { + es := parseContentsEntries("Page Part I Item 1Business 7 Item 1ARisk Factors 14 Item 2Properties 24 Part II Item 5Market for Registrant's Common Equity 29") + if e := hasTitle(es, "Item 1 Business"); e == nil || e.Printed != 7 { + t.Errorf("Item 1: %+v in %v", e, titlesOf(es)) + } + if e := hasTitle(es, "Item 1A Risk Factors"); e == nil || e.Printed != 14 { + t.Errorf("Item 1A: %+v", e) + } + if e := hasTitle(es, "PART II"); e == nil || !e.Container { + t.Errorf("inline PART II not a container: %v", titlesOf(es)) + } + if e := hasTitle(es, "Item 5 Market for Registrant's Common Equity"); e == nil || e.Depth != 2 { + t.Errorf("Item 5 under Part II: %+v", e) + } + if hasTitle(es, "Page") != nil || hasTitle(es, "Page Part I Item 1 Business") != nil { + t.Errorf("column label leaked: %v", titlesOf(es)) + } + }) + t.Run("verizon: item without a page, page number inside a wrapped title", func(t *testing.T) { + es := parseContentsEntries("PART I\nItem 1. Business\nItem 1A. Risk Factors 14 PART II Item 5. Market for Registrant’s Common Equity, Related Stockholder Matters and Issuer Purchases of 20 Equity Securities Item 6. [Reserved] 21") + if e := hasTitle(es, "Item 1. Business"); e == nil || e.Printed != 0 { + t.Errorf("Item 1 without a page should still be an entry: %+v %v", e, titlesOf(es)) + } + if e := hasTitle(es, "Item 5. Market for Registrant’s Common Equity, Related Stockholder Matters and Issuer Purchases of Equity Securities"); e == nil || e.Printed != 20 { + t.Errorf("wrapped tail not re-attached: %v", titlesOf(es)) + } + if e := hasTitle(es, "Item 6. [Reserved]"); e == nil || e.Printed != 21 { + t.Errorf("Item 6: %+v", e) + } + if hasTitle(es, "Equity Securities Item 6. [Reserved]") != nil { + t.Errorf("tail glued to the next entry: %v", titlesOf(es)) + } + }) + t.Run("oracle: part with a trailing dot inline", func(t *testing.T) { + es := parseContentsEntries("Page PART I. Item 1. Business 3 PART II. Item 5. Market 21") + if e := hasTitle(es, "PART I"); e == nil || !e.Container { + t.Errorf("PART I.: %v", titlesOf(es)) + } + if e := hasTitle(es, "Item 1. Business"); e == nil || e.Printed != 3 || e.Depth != 2 { + t.Errorf("Item 1: %+v", e) + } + }) + t.Run("pfizer: parenthesised sub-entries", func(t *testing.T) { + es := parseContentsEntries("ITEM 15. EXHIBITS, FINANCIAL STATEMENT SCHEDULES 111 15(a)(1) Financial Statements 111 15(a)(2) Financial Statement Schedules 111 15(a)(3) Exhibits 111 ITEM 16. FORM 10-K SUMMARY 116") + if e := hasTitle(es, "ITEM 15. EXHIBITS, FINANCIAL STATEMENT SCHEDULES"); e == nil || e.Printed != 111 { + t.Errorf("Item 15 did not close at its page: %v", titlesOf(es)) + } + if e := hasTitle(es, "15(a)(2) Financial Statement Schedules"); e == nil || e.Printed != 111 { + t.Errorf("sub-entry: %v", titlesOf(es)) + } + }) + t.Run("3m: bold-fold duplicates and a truncated copy", func(t *testing.T) { + es := parseContentsEntries("**Geographic Area 38 Critical Accounting Estimates 39 New**\nGeographic Area 38 Critical Accounting Estimates 39 New Accounting Pronouncements 42 Financial Condition and Liquidity 43") + n := 0 + for _, e := range es { + if e.Title == "Geographic Area" { + n++ + } + } + if n != 1 { + t.Errorf("duplicate entry kept %d times: %v", n, titlesOf(es)) + } + if hasTitle(es, "New") != nil { + t.Errorf("truncated fragment 'New' kept: %v", titlesOf(es)) + } + if e := hasTitle(es, "New Accounting Pronouncements"); e == nil || e.Printed != 42 { + t.Errorf("full entry missing: %v", titlesOf(es)) + } + }) +}