diff --git a/cmd/engine/main.go b/cmd/engine/main.go index 9eed587..498f07d 100644 --- a/cmd/engine/main.go +++ b/cmd/engine/main.go @@ -11,6 +11,7 @@ import ( "flag" "fmt" "io" + "log" "log/slog" "net/http" "os" @@ -126,7 +127,17 @@ func run() error { return fmt.Errorf("init llm: %w", err) } } - strategy := buildStrategy(cfg.Retrieval, llmClient, store) + judge, err := buildJudge(cfg.LLM.Judge) + if err != nil { + logger.Error("judge: config invalid", "err", err) + os.Exit(1) + } + if judge != nil { + logger.Info("judge: typesafe enabled — TOC stage and judgewalk retrieval run on the Judge") + } else { + logger.Warn("judge: none configured — TOC judgements run on the generative driver, one call per page (set TYPESAFE_API_KEY)") + } + strategy := buildStrategy(cfg.Retrieval, llmClient, judge, store) // Wrap with caching if enabled. if cfg.Retrieval.Cache.Enabled { @@ -205,16 +216,6 @@ func run() error { ) } - judge, err := buildJudge(cfg.LLM.Judge) - if err != nil { - logger.Error("judge: config invalid", "err", err) - os.Exit(1) - } - if judge != nil { - logger.Info("judge: typesafe enabled — contents-page detection and page resolution run on the Judge") - } else { - logger.Warn("judge: none configured — TOC judgements run on the generative driver, one call per page (set TYPESAFE_API_KEY)") - } pipeline := ingest.NewPipeline(ingest.Pipeline{ DB: pool, Storage: store, @@ -517,8 +518,16 @@ func buildLLMFrom(c config.LLMConfig, provider, apiKey, baseURL, model string) ( } } -func buildStrategy(c config.RetrievalConfig, client llmgate.Client, store storage.Storage) retrieval.Strategy { +func buildStrategy(c config.RetrievalConfig, client llmgate.Client, judge llmgate.Judge, store storage.Storage) retrieval.Strategy { switch c.Strategy { + case "judgewalk": + if judge == nil { + log.Printf("retrieval: strategy judgewalk needs llm.judge configured; using treewalk") + return buildTreeWalkStrategy(c, client, store) + } + s := retrieval.NewJudgeWalkStrategy(judge) + s.PageLoader = storagePageLoader{s: store} + return s case "single-pass": return retrieval.NewSinglePass(client) case "chunked-tree": diff --git a/cmd/ingestbench/main.go b/cmd/ingestbench/main.go index d64aecd..42feac5 100644 --- a/cmd/ingestbench/main.go +++ b/cmd/ingestbench/main.go @@ -245,5 +245,5 @@ func readPages(path string) ([]ingest.PageText, error) { if err != nil { return nil, err } - return ingest.BenchAssemblePages(doc.Sections), nil + return ingest.BenchAssemblePages(doc), nil } diff --git a/cmd/jevbench/main.go b/cmd/jevbench/main.go index 5028620..9c920a5 100644 --- a/cmd/jevbench/main.go +++ b/cmd/jevbench/main.go @@ -136,7 +136,7 @@ func readPages(path string) ([]ingest.PageText, error) { if err != nil { return nil, err } - return ingest.BenchAssemblePages(doc.Sections), nil + return ingest.BenchAssemblePages(doc), nil } // dotEnv reads one key from a gitignored .env up-tree, so a credential diff --git a/cmd/navbench/main.go b/cmd/navbench/main.go new file mode 100644 index 0000000..c6cee7c --- /dev/null +++ b/cmd/navbench/main.go @@ -0,0 +1,314 @@ +// Command navbench measures retrieval navigation on a Judge against +// FinanceBench's gold evidence pages, with no generative model. +// +// For each question on a filing with a tree: rank the tree's leaves, +// read the pages of the best few, rank those pages, and ask whether the +// gold evidence pages are in the evidence set. +// +// go run ./cmd/navbench -questions ~/.cache/vlbench/financebench-questions.jsonl \ +// -trees ~/.cache/vlbench/trees-jev2 -pdfs ~/.cache/vlbench/financebench -out nav.jsonl +package main + +import ( + "bufio" + "context" + "encoding/json" + "flag" + "fmt" + "os" + "path/filepath" + "sort" + "strings" + "time" + + "github.com/hallelx2/llmgate/judge/typesafe" + "github.com/hallelx2/llmgate/middleware/retry" + + "github.com/hallelx2/vectorless-engine/pkg/ingest" + "github.com/hallelx2/vectorless-engine/pkg/parser" + "github.com/hallelx2/vectorless-engine/pkg/retrieval" + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +type question struct { + ID string `json:"id"` + Doc string `json:"doc"` + Question string `json:"question"` + Answer string `json:"answer"` + Evidence []int `json:"evidence_pages"` + Type string `json:"type"` +} + +type dump struct { + Doc string `json:"doc"` + Nodes []tree.TOCNode `json:"nodes"` +} + +type outcome struct { + ID string `json:"id"` + Doc string `json:"doc"` + Question string `json:"question"` + Gold []int `json:"gold_pages"` + Selected []string `json:"selected_leaves"` + SelectedPages [][2]int `json:"selected_ranges"` + Evidence []int `json:"evidence_pages"` + EvidenceP []float64 `json:"evidence_p"` + PagesRead int `json:"pages_read"` + LeafHit bool `json:"leaf_hit"` // every gold page inside a selected leaf + Recall float64 `json:"recall"` // share of gold pages in the evidence set + Hit bool `json:"hit"` // recall == 1 + Requests int `json:"requests"` + InTokens int `json:"in_tokens"` + CostUSD float64 `json:"cost_usd"` + Seconds float64 `json:"seconds"` + Err string `json:"err,omitempty"` +} + +func main() { + qPath := flag.String("questions", "", "FinanceBench questions JSONL") + trees := flag.String("trees", "", "directory of tocdump trees") + pdfs := flag.String("pdfs", "", "directory of filings") + out := flag.String("out", "", "JSONL of per-question outcomes") + maxLeaves := flag.Int("leaves", 3, "sections read per question") + maxPages := flag.Int("pages", 40, "pages judged per question") + limit := flag.Int("limit", 0, "stop after this many questions (0 = all)") + flag.Parse() + if *qPath == "" || *trees == "" || *pdfs == "" { + fmt.Fprintln(os.Stderr, "usage: navbench -questions q.jsonl -trees dir -pdfs dir [-out o.jsonl]") + os.Exit(2) + } + + key := os.Getenv(typesafe.EnvAPIKey) + tj, err := typesafe.New(typesafe.Config{APIKey: key}) + if err != nil { + fmt.Fprintln(os.Stderr, "judge:", err) + os.Exit(1) + } + nav := &retrieval.JudgeNavigator{Judge: retry.NewJudge(retry.Config{MaxRetries: 3})(tj), MaxLeaves: *maxLeaves, MaxPages: *maxPages} + + qs := readQuestions(*qPath) + if *limit > 0 && len(qs) > *limit { + qs = qs[:*limit] + } + var of *os.File + if *out != "" { + of, _ = os.Create(*out) + defer of.Close() + } + + pageCache := map[string][]ingest.PageText{} + leafCache := map[string][]retrieval.NavLeaf{} + var results []outcome + for _, q := range qs { + o := outcome{ID: q.ID, Doc: q.Doc, Question: q.Question, Gold: q.Evidence} + leaves, pages, err := load(q.Doc, *trees, *pdfs, leafCache, pageCache) + if err != nil { + o.Err = err.Error() + results = append(results, o) + report(o) + continue + } + byNum := map[int]string{} + for _, p := range pages { + byNum[p.PageNumber] = p.Text + } + loadPages := func(_ context.Context, l retrieval.NavLeaf) ([]retrieval.NavPage, error) { + var ps []retrieval.NavPage + for n := l.Start; n <= l.End; n++ { + if t, ok := byNum[n]; ok { + ps = append(ps, retrieval.NavPage{Number: n, Text: t}) + } + } + return ps, nil + } + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute) + start := time.Now() + res, err := nav.Navigate(ctx, q.Question, leaves, loadPages) + cancel() + o.Seconds = time.Since(start).Seconds() + if err != nil { + o.Err = err.Error() + results = append(results, o) + report(o) + continue + } + for _, l := range res.Selected { + o.Selected = append(o.Selected, l.Title) + o.SelectedPages = append(o.SelectedPages, [2]int{l.Start, l.End}) + } + for _, e := range res.Evidence { + o.Evidence = append(o.Evidence, e.Page.Number) + o.EvidenceP = append(o.EvidenceP, e.P) + } + o.PagesRead = len(res.Pages) + o.Requests, o.InTokens, o.CostUSD = res.Requests, res.Usage.InputTokens, res.Usage.CostUSD + o.LeafHit = allInRanges(q.Evidence, o.SelectedPages) + o.Recall = recall(q.Evidence, o.Evidence) + o.Hit = o.Recall == 1 + results = append(results, o) + report(o) + if of != nil { + b, _ := json.Marshal(o) + of.Write(append(b, '\n')) + } + } + summarise(results) +} + +func load(doc, trees, pdfs string, leafCache map[string][]retrieval.NavLeaf, pageCache map[string][]ingest.PageText) ([]retrieval.NavLeaf, []ingest.PageText, error) { + if l, ok := leafCache[doc]; ok { + return l, pageCache[doc], nil + } + raw, err := os.ReadFile(filepath.Join(trees, doc+".json")) + if err != nil { + return nil, nil, err + } + var d dump + if err := json.Unmarshal(raw, &d); err != nil { + return nil, nil, err + } + if len(d.Nodes) == 0 { + return nil, nil, fmt.Errorf("no tree") + } + f, err := os.Open(filepath.Join(pdfs, doc+".pdf")) + if err != nil { + return nil, nil, err + } + pd, err := parser.NewPDF().Parse(context.Background(), f) + f.Close() + if err != nil { + return nil, nil, err + } + pages := ingest.BenchAssemblePages(pd) + var leaves []retrieval.NavLeaf + var walk func(ns []tree.TOCNode, path string) + walk = func(ns []tree.TOCNode, path string) { + for _, n := range ns { + p := n.Title + if path != "" { + p = path + " > " + n.Title + } + if len(n.Nodes) > 0 { + walk(n.Nodes, p) + continue + } + if n.StartPage > 0 && n.EndPage >= n.StartPage { + leaves = append(leaves, retrieval.NavLeaf{ID: n.NodeID, Title: n.Title, Path: p, Start: n.StartPage, End: n.EndPage}) + } + } + } + walk(d.Nodes, "") + leafCache[doc], pageCache[doc] = leaves, pages + return leaves, pages, nil +} + +func allInRanges(gold []int, ranges [][2]int) bool { + for _, g := range gold { + in := false + for _, r := range ranges { + if g >= r[0] && g <= r[1] { + in = true + break + } + } + if !in { + return false + } + } + return len(gold) > 0 +} + +func recall(gold, got []int) float64 { + if len(gold) == 0 { + return 0 + } + set := map[int]bool{} + for _, g := range got { + set[g] = true + } + n := 0 + for _, g := range gold { + if set[g] { + n++ + } + } + return float64(n) / float64(len(gold)) +} + +func report(o outcome) { + mark := "MISS" + if o.Hit { + mark = "hit " + } else if o.LeafHit { + mark = "leaf" + } + if o.Err != "" { + mark = "err " + } + fmt.Printf("%s %-24s gold %-10v evidence %-16v read %2d %d req %5.1fs $%.5f %s\n", + mark, o.Doc, o.Gold, o.Evidence, o.PagesRead, o.Requests, o.Seconds, o.CostUSD, strings.TrimSpace(o.Err)) +} + +func summarise(rs []outcome) { + var n, hit, leaf, errs int + var rec, secs, cost float64 + var pages, reqs, toks int + for _, o := range rs { + if o.Err != "" { + errs++ + continue + } + n++ + if o.Hit { + hit++ + } + if o.LeafHit { + leaf++ + } + rec += o.Recall + secs += o.Seconds + cost += o.CostUSD + pages += o.PagesRead + reqs += o.Requests + toks += o.InTokens + } + if n == 0 { + fmt.Println("no questions answered") + return + } + var secList []float64 + for _, o := range rs { + if o.Err == "" { + secList = append(secList, o.Seconds) + } + } + sort.Float64s(secList) + fmt.Printf("\nquestions %d (errors %d)\n", n, errs) + fmt.Printf(" gold pages inside a selected section %d / %d (%.3f)\n", leaf, n, float64(leaf)/float64(n)) + fmt.Printf(" every gold page in the evidence set %d / %d (%.3f)\n", hit, n, float64(hit)/float64(n)) + fmt.Printf(" mean page recall %.3f\n", rec/float64(n)) + fmt.Printf(" pages read / question %.1f\n", float64(pages)/float64(n)) + fmt.Printf(" requests / question %.1f\n", float64(reqs)/float64(n)) + fmt.Printf(" input tokens / question %d\n", toks/n) + fmt.Printf(" seconds / question mean %.1f median %.1f\n", secs/float64(n), secList[len(secList)/2]) + fmt.Printf(" cost / question $%.5f total $%.4f\n", cost/float64(n), cost) +} + +func readQuestions(path string) []question { + f, err := os.Open(path) + if err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } + defer f.Close() + var qs []question + sc := bufio.NewScanner(f) + sc.Buffer(make([]byte, 1<<20), 1<<20) + for sc.Scan() { + var q question + if json.Unmarshal(sc.Bytes(), &q) == nil { + qs = append(qs, q) + } + } + return qs +} diff --git a/cmd/pagedump/main.go b/cmd/pagedump/main.go index 09863b0..c8570b5 100644 --- a/cmd/pagedump/main.go +++ b/cmd/pagedump/main.go @@ -29,7 +29,7 @@ func main() { fmt.Fprintln(os.Stderr, "parse:", err) os.Exit(1) } - for _, p := range ingest.BenchAssemblePages(doc.Sections) { + for _, p := range ingest.BenchAssemblePages(doc) { fmt.Printf("\n===== PAGE %d (%d chars) =====\n%s", p.PageNumber, len(p.Text), p.Text) } } diff --git a/cmd/server/main.go b/cmd/server/main.go index 13c1089..7f104b8 100644 --- a/cmd/server/main.go +++ b/cmd/server/main.go @@ -19,6 +19,7 @@ import ( "flag" "fmt" "io" + "log" "log/slog" "net/http" "os" @@ -143,7 +144,17 @@ func run() error { if err != nil { return fmt.Errorf("init llm: %w", err) } - strategy := buildStrategy(cfg.Engine.Retrieval, llmClient, store, pool) + judge, err := buildJudge(cfg.Engine.LLM.Judge) + if err != nil { + logger.Error("judge: config invalid", "err", err) + os.Exit(1) + } + if judge != nil { + logger.Info("judge: typesafe enabled — TOC stage and judgewalk retrieval run on the Judge") + } else { + logger.Warn("judge: none configured — TOC judgements run on the generative driver, one call per page (set TYPESAFE_API_KEY)") + } + strategy := buildStrategy(cfg.Engine.Retrieval, llmClient, judge, store, pool) // Wrap with caching if enabled in engine config. if cfg.Engine.Retrieval.Cache.Enabled { @@ -170,7 +181,7 @@ func run() error { // running engine without a redeploy. Built from the raw client so // each override behaves identically to booting with that strategy // as the default (no shared cache wrapper across overrides). - strategies := buildStrategySet(cfg.Engine.Retrieval, llmClient, store, pool) + strategies := buildStrategySet(cfg.Engine.Retrieval, llmClient, judge, store, pool) // Replay store: every /v1/answer and /v1/answer/treewalk response // is stamped with a deterministic trace_token and its body bytes @@ -203,16 +214,6 @@ func run() error { } // ── Ingest pipeline ─────────────────────────────────────────── - judge, err := buildJudge(cfg.Engine.LLM.Judge) - if err != nil { - logger.Error("judge: config invalid", "err", err) - os.Exit(1) - } - if judge != nil { - logger.Info("judge: typesafe enabled — contents-page detection and page resolution run on the Judge") - } else { - logger.Warn("judge: none configured — TOC judgements run on the generative driver, one call per page (set TYPESAFE_API_KEY)") - } pipeline := ingest.NewPipeline(ingest.Pipeline{ DB: pool, Storage: store, @@ -460,8 +461,14 @@ func buildLLM(c enginecfg.LLMConfig) (llmgate.Client, error) { // retrieval.strategy. The DB pool is threaded through so the // treewalk strategy can wire a TOC provider that reads // documents.toc_tree (the other strategies ignore it). -func buildStrategy(c enginecfg.RetrievalConfig, client llmgate.Client, store storage.Storage, pool *db.Pool) retrieval.Strategy { +func buildStrategy(c enginecfg.RetrievalConfig, client llmgate.Client, judge llmgate.Judge, store storage.Storage, pool *db.Pool) retrieval.Strategy { switch c.Strategy { + case "judgewalk": + if judge == nil { + log.Printf("retrieval: strategy judgewalk needs llm.judge configured; using treewalk") + return buildTreeWalkStrategy(c, client, store, pool) + } + return buildJudgeWalkStrategy(judge, store) case "single-pass": return retrieval.NewSinglePass(client) case "chunked-tree": @@ -493,20 +500,34 @@ func buildStrategy(c enginecfg.RetrievalConfig, client llmgate.Client, store sto // from the same config blocks the default builder reads, so an // override behaves identically to booting with that strategy as the // default. -func buildStrategySet(c enginecfg.RetrievalConfig, client llmgate.Client, store storage.Storage, pool *db.Pool) map[string]retrieval.Strategy { +func buildStrategySet(c enginecfg.RetrievalConfig, client llmgate.Client, judge llmgate.Judge, store storage.Storage, pool *db.Pool) map[string]retrieval.Strategy { agentic := retrieval.NewAgentic(client, storageFetcher{s: store}) if c.Agentic.MaxHops > 0 { agentic.MaxHops = c.Agentic.MaxHops } agentic.ModelOverride = c.Agentic.Model - return map[string]retrieval.Strategy{ + set := map[string]retrieval.Strategy{ "single-pass": retrieval.NewSinglePass(client), "chunked-tree": retrieval.NewChunkedTree(client), "agentic": agentic, "treewalk": buildTreeWalkStrategy(c, client, store, pool), "auto": retrieval.NewAuto(retrieval.NewSinglePass(client), buildTreeWalkStrategy(c, client, store, pool)), } + if judge != nil { + set["judgewalk"] = buildJudgeWalkStrategy(judge, store) + } + return set +} + +// buildJudgeWalkStrategy constructs navigation on the Judge: the tree's +// leaves ranked in one request, the best sections' bodies ranked in one +// or two more, no generative call (HAL-1371). Selectable per request as +// strategy=judgewalk whenever llm.judge is configured. +func buildJudgeWalkStrategy(judge llmgate.Judge, store storage.Storage) *retrieval.JudgeWalkStrategy { + s := retrieval.NewJudgeWalkStrategy(judge) + s.PageLoader = storagePageLoader{s: store} + return s } // buildTreeWalkStrategy constructs the page-based agentic strategy diff --git a/cmd/tocdump/main.go b/cmd/tocdump/main.go index 4344ae0..43e0d8e 100644 --- a/cmd/tocdump/main.go +++ b/cmd/tocdump/main.go @@ -200,7 +200,7 @@ func readPages(path string) ([]ingest.PageText, error) { if err != nil { return nil, err } - return ingest.BenchAssemblePages(doc.Sections), nil + return ingest.BenchAssemblePages(doc), nil } func dotEnv(key string) string { diff --git a/cmd/tocresolve/main.go b/cmd/tocresolve/main.go index 2b22a90..367d068 100644 --- a/cmd/tocresolve/main.go +++ b/cmd/tocresolve/main.go @@ -72,7 +72,7 @@ func main() { fmt.Fprintln(os.Stderr, "parse:", err) os.Exit(1) } - pages := ingest.BenchAssemblePages(doc.Sections) + pages := ingest.BenchAssemblePages(doc) key := os.Getenv(typesafe.EnvAPIKey) if key == "" { diff --git a/docs/evaluations/2026-09-18-retrieval-navigation-on-a-judge.md b/docs/evaluations/2026-09-18-retrieval-navigation-on-a-judge.md new file mode 100644 index 0000000..d3d9be1 --- /dev/null +++ b/docs/evaluations/2026-09-18-retrieval-navigation-on-a-judge.md @@ -0,0 +1,105 @@ +# Retrieval navigation on a Judge — and the page bug it found + +**Date:** 2026-09-18 +**Harness:** [`cmd/navbench`](../../cmd/navbench/main.go) (evidence-page recall per question, no generative call), [`cmd/tocdump -judge-only`](../../cmd/tocdump/main.go), [`cmd/tocdump/coverage.py`](../../cmd/tocdump/coverage.py), [`cmd/tocdump/titles.py`](../../cmd/tocdump/titles.py) +**Corpus:** FinanceBench, 40 questions on the 21 downloaded 10-K filings, gold evidence pages from `PatronusAI/financebench` +**Issues:** HAL-1371 (navigation), HAL-1375 (per-page text), HAL-1374 (leaf granularity) +**Question:** `TreeWalkStrategy` navigates with up to eight chat-completion hops. Can a Judge navigate instead — rank the tree's leaves, then the pages — and land the gold evidence page, with no generative call? + +## Result: 34 of 40 questions have every gold page in the evidence set, 37 of 40 have it inside the chosen section, at about four Jev requests, 40 pages read and $0.003 per question. + +| run | change | questions | right section | every gold page found | pages read | requests | s / question | $ / question | +|---|---|---|---|---|---|---|---|---| +| 1 | leaves → full pages, 3 leaves, 40 pages | 38 | 32 (0.842) | 30 (0.789) | 35.9 | 2.9 | 8.9 | 0.0021 | +| 2 | + coarse pass over page heads, page dedup | 35 † | 29 (0.829) | 25 (0.714) | 36.0 | 3.7 | 22.5 | 0.0028 | +| 3 | + leaf prompt no longer rules out risk sections | 38 | 34 (0.895) | 29 (0.763) | 36.2 | 3.7 | 16.4 | 0.0028 | +| **5** | **+ real per-page text (HAL-1375), + follow cross-references** | **40** | **37 (0.925)** | **34 (0.850)** | 40.7 | 4.1 | 20.6 | 0.0029 | + +† three Jev requests failed after four minutes of retries in run 2; those questions are excluded from its counts and its seconds are inflated by the retries. Run 4 (follow only, old pages) was stopped when the page bug was found. + +Seconds per question are Jev API latency in the hour of the run — the +same 21 filings' whole TOC stage took 44 s each in one pass and 9.5 s in +the next — and are reported, not engineered around. + +## The bug the near-misses found + +Run 3's misses were one page off: gold 12 with evidence 11, gold 59 +with 58, gold 57 with 56. Page text was being assembled from +*sections* — each section's whole content keyed under the page it +started on — so a section running from 58 into 59 put page 59's opening +under "page 58", and a page that started no section had no entry at +all. AMD_2022_10K had 104 page entries for a 121-page PDF; PEPSICO's +549 pages showed as 170. + +Every page-level stage had been working on that: detection, resolution, +the coverage gate, and navigation. `ParsedDoc.Pages` is now built from +the parser's own rows, grouped by physical page. Re-running the whole +TOC stage on real pages: + +| | section-start pages (jev2) | real pages (jev3) | +|---|---|---| +| filings with a tree | 20 / 21 | **21 / 21** (GENERALMILLS, the "one-page parse" of HAL-1365, was this bug) | +| gold evidence pages inside a leaf | 45 / 45 | **47 / 47** | +| median leaf span | 37 p | 37 p | +| leaf title recall / precision against jev2 | — | 0.975 / 0.966 | +| TOC stage per filing | 44 s | 9.5 s (Jev latency in the hour) | + +## What each navigation change did + +- **Leaf ranking, then full pages** (run 1): the shape works. One Noul + per leaf, one per page of the best three leaves. 30 of 38. +- **Coarse pass over page heads** (run 2): meant to let a 70-page Item + 8 into the page budget. It did not move hits on its own and adds a + request; kept because run 5 needs it to gather 120 pages from five + leaves. +- **The leaf prompt** (run 3): the first version said "a risk-factors + section holds neither" the figures nor the discussion. Three Boeing + questions are answered in Risk Factors; the Judge ranked it last + because it was told to. Right-section rate 0.842 → 0.895 once the + instruction described what each kind of section holds without ruling + any out. A prompt is a measured artifact; see the prompt-versioning + proposal. +- **Real pages + follow** (run 5): the page fix turned the off-by-one + misses into hits (0.763 → 0.850) and brought the two GENERALMILLS + questions into the set. Following "see Note 21" fires only when the + tree has a leaf to follow to; on 10-K trees with 20-odd leaves it + rarely does, which is HAL-1374's point. + +## The six that are still missed + +| filing | gold | evidence | why | +|---|---|---|---| +| BOEING | 113 | 26, 27 | "see Note 21" — the tree has no note-level leaf to follow to (HAL-1374) | +| BOEING | 8, 10, 14 | 12, 14 | right section; two of three pages ranked below threshold | +| BOEING | 55 | 24, 77, 76, 79 | right section; page missed | +| KRAFTHEINZ | 50, 52 | 50, 124 | one of two gold pages | +| PFIZER | 59 | 57, 41 | right section; neighbouring page chosen | +| PFIZER | 70, 71 | 51, 9 | wrong section | + +Four of six are page-level misses inside the right section — the +full-pass question, or how much of a page it sees, is the next thing +to measure. The other two want finer leaves. + +## Method + +- `JudgeNavigator.RankLeaves`: one request per 120 leaves; state is + the question plus each leaf's title, path, page range and summary. +- Gather the pages of the best five leaves, up to 120, each page once. +- Over 40 pages: `rankPages` on 700-character heads, keep the best 40. +- `RankPages` on full pages (6,000 chars), batched so the request's + whole state stays under 24k tokens — the provider treats the state + object as shared context for every question. Evidence is every page + at or above 0.5; never fewer than the best two. +- `referencedLeaves`: `Note N` / `Item N` mentions on evidence pages + that name an unread leaf → its pages ranked in one more request. +- `JudgeWalkStrategy` is the same over a section tree, selectable as + `strategy=judgewalk` whenever `llm.judge` is configured. It writes + no answer: that is the caller's one generative call, over the + evidence only. + +## Reproduce + +```bash +go run ./cmd/tocdump -docs ~/.cache/vlbench/financebench -out /tmp/trees -judge-only -minimal +go run ./cmd/navbench -questions ~/.cache/vlbench/financebench-questions.jsonl -trees /tmp/trees -pdfs ~/.cache/vlbench/financebench -out /tmp/nav.jsonl +``` diff --git a/pkg/ingest/bench_export.go b/pkg/ingest/bench_export.go index 9d95b60..2ec6e2a 100644 --- a/pkg/ingest/bench_export.go +++ b/pkg/ingest/bench_export.go @@ -21,8 +21,8 @@ import ( // BenchAssemblePages turns parsed sections into the per-page text the // TOC builder consumes. -func BenchAssemblePages(secs []parser.Section) []PageText { - return assemblePagesFromSections(secs) +func BenchAssemblePages(doc *parser.ParsedDoc) []PageText { + return assemblePages(doc) } // BenchDetectTOC runs the Judge-backed detection phase alone. diff --git a/pkg/ingest/ingest.go b/pkg/ingest/ingest.go index 9559fb5..4d2b6e0 100644 --- a/pkg/ingest/ingest.go +++ b/pkg/ingest/ingest.go @@ -457,7 +457,7 @@ func (p *Pipeline) Run(ctx context.Context, pl Payload) error { // SQL NULL to documents.toc_tree (which is the column's default, // so this is also the no-op). func (p *Pipeline) runTOCBuilder(ctx context.Context, docID tree.DocumentID, parsed *parser.ParsedDoc, log *slog.Logger) error { - pages := assemblePagesFromSections(parsed.Sections) + pages := assemblePages(parsed) if len(pages) == 0 { log.Info("ingest: toc-builder skipped; no per-page text available") return nil @@ -519,6 +519,26 @@ func (p *Pipeline) runTOCBuilder(ctx context.Context, docID tree.DocumentID, par // // Sections with PageStart == 0 are skipped (the parser couldn't // place them) so the builder never sees ambiguous page numbers. +// assemblePages returns the document's text per page. The parser's own +// pages are the truth when it has them; the section-based assembly is +// the fallback for formats with no page notion — and it is only ever an +// approximation, since a section's content spans pages (HAL-1375). +func assemblePages(parsed *parser.ParsedDoc) []PageText { + if parsed == nil { + return nil + } + if len(parsed.Pages) > 0 { + out := make([]PageText, 0, len(parsed.Pages)) + for _, p := range parsed.Pages { + if p.Number > 0 && strings.TrimSpace(p.Text) != "" { + out = append(out, PageText{PageNumber: p.Number, Text: p.Text}) + } + } + return out + } + return assemblePagesFromSections(parsed.Sections) +} + func assemblePagesFromSections(secs []parser.Section) []PageText { pageText := map[int]*strings.Builder{} pages := []int{} diff --git a/pkg/ingest/toc_builder_test.go b/pkg/ingest/toc_builder_test.go index 68733b8..5a6212d 100644 --- a/pkg/ingest/toc_builder_test.go +++ b/pkg/ingest/toc_builder_test.go @@ -640,3 +640,19 @@ func TestBuildKeepsClaimedPagesWhenTheResolverIsDown(t *testing.T) { t.Errorf("degradation not recorded: %v", usage.Degraded) } } + +func TestAssemblePagesPrefersTheParsersPages(t *testing.T) { + doc := &parser.ParsedDoc{ + Sections: []parser.Section{{Title: "Item 1", Content: "page 2 text page 3 text", PageStart: 2, PageEnd: 3}}, + Pages: []parser.Page{{Number: 2, Text: "Item 1\npage 2 text"}, {Number: 3, Text: "page 3 text"}}, + } + got := assemblePages(doc) + if len(got) != 2 || got[1].PageNumber != 3 || got[1].Text != "page 3 text" { + t.Fatalf("pages should come from Pages, per page: %+v", got) + } + doc.Pages = nil + got = assemblePages(doc) + if len(got) != 1 || got[0].PageNumber != 2 { + t.Errorf("without Pages, the section fallback: %+v", got) + } +} diff --git a/pkg/parser/pages_test.go b/pkg/parser/pages_test.go new file mode 100644 index 0000000..9e32757 --- /dev/null +++ b/pkg/parser/pages_test.go @@ -0,0 +1,54 @@ +package parser + +import ( + "context" + "os" + "testing" +) + +// Pages come from rows grouped by physical page, never from sections. +func TestPagesFromRowsGroupsByPhysicalPage(t *testing.T) { + rows := []pdfRow{ + {page: 1, text: "Cover"}, + {page: 2, text: "Item 1. Business"}, + {page: 2, text: "We make things."}, + {page: 3, text: "and keep making them."}, // same section, next page + {page: 5, text: "Item 1A. Risk Factors"}, // page 4 is blank + } + got := pagesFromRows(rows) + if len(got) != 4 { + t.Fatalf("pages: %d want 4 (1,2,3,5)", len(got)) + } + if got[2].Number != 3 || got[2].Text != "and keep making them." { + t.Errorf("page 3 should hold only its own text: %+v", got[2]) + } + if got[3].Number != 5 { + t.Errorf("a blank page keeps its number for the next: %+v", got[3]) + } + if got[1].Text != "Item 1. Business\nWe make things." { + t.Errorf("page 2: %q", got[1].Text) + } +} + +func TestParsedPDFExposesPages(t *testing.T) { + f, err := os.Open("testdata/tables-example.pdf") + if err != nil { + t.Skip("fixture missing") + } + defer f.Close() + doc, err := NewPDF().Parse(context.Background(), f) + if err != nil { + t.Fatal(err) + } + if len(doc.Pages) == 0 { + t.Fatal("no pages on a parsed PDF") + } + for i := 1; i < len(doc.Pages); i++ { + if doc.Pages[i].Number <= doc.Pages[i-1].Number { + t.Errorf("pages out of order: %d after %d", doc.Pages[i].Number, doc.Pages[i-1].Number) + } + } + if doc.Pages[0].Number != 1 { + t.Errorf("first page is %d", doc.Pages[0].Number) + } +} diff --git a/pkg/parser/parser.go b/pkg/parser/parser.go index b84f8e1..92c42f1 100644 --- a/pkg/parser/parser.go +++ b/pkg/parser/parser.go @@ -34,6 +34,22 @@ type ParsedDoc struct { // Metadata holds whatever extra structural hints the parser recovered // (author, created date, page count, etc.). Metadata map[string]string + + // Pages is the document's text page by page, for parsers that have + // pages. It is the ground truth for anything that reasons per page — + // contents-page detection, page resolution, page ranking — and must + // never be reconstructed from Sections: a section's content spans + // pages, and keying it under the section's first page puts the next + // page's opening under the wrong number (HAL-1375). Empty for + // formats without a page notion. + Pages []Page +} + +// Page is one page's text, in reading order, with the parser's +// running-header and boilerplate filtering already applied. +type Page struct { + Number int + Text string } // Section is one node in the parsed outline. diff --git a/pkg/parser/pdf.go b/pkg/parser/pdf.go index 8149f93..2abab77 100644 --- a/pkg/parser/pdf.go +++ b/pkg/parser/pdf.go @@ -345,6 +345,7 @@ func (p *PDF) parseDoc(_ context.Context, buf []byte) (*ParsedDoc, error) { if outline := reader.Outline(); len(outline.Child) > 0 { if doc, ok := parsePDFWithOutline(outline, rows); ok { doc.Sections = capLeafSections(doc.Sections, p.resolvedMaxSections()) + doc.Pages = pagesFromRows(rows) attachTableSections(doc, tableSections) return doc, nil } @@ -543,11 +544,36 @@ func (p *PDF) parseDoc(_ context.Context, buf []byte) (*ParsedDoc, error) { out := &ParsedDoc{ Title: title, Sections: capLeafSections(chunkOversizedLeaves(rootSec.Children), p.resolvedMaxSections()), + Pages: pagesFromRows(rows), } attachTableSections(out, tableSections) return out, nil } +// pagesFromRows groups the filtered rows by page, in order. Rows carry +// the physical page index they were read from, so a page with no rows +// (blank, or image-only) simply has no entry — its number is still +// its own, never reassigned. +func pagesFromRows(rows []pdfRow) []Page { + var out []Page + var cur *Page + for _, r := range rows { + text := strings.TrimSpace(r.text) + if text == "" || r.page <= 0 { + continue + } + if cur == nil || cur.Number != r.page { + out = append(out, Page{Number: r.page}) + cur = &out[len(out)-1] + } + if cur.Text != "" { + cur.Text += "\n" + } + cur.Text += text + } + return out +} + // resolvedMaxSections turns the configured MaxSections into the value // the cap actually uses: 0 selects defaultMaxLeafSections; a negative // value disables the cap (returns a non-positive number capLeafSections diff --git a/pkg/retrieval/judgewalk.go b/pkg/retrieval/judgewalk.go new file mode 100644 index 0000000..56acb8f --- /dev/null +++ b/pkg/retrieval/judgewalk.go @@ -0,0 +1,669 @@ +package retrieval + +import ( + "context" + "fmt" + "sort" + "strings" + + "regexp" + + "github.com/hallelx2/llmgate" + "github.com/hallelx2/llmgate/judge/typesafe" + + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +// Retrieval navigation on a Judge — select, don't generate (HAL-1371). +// +// TreeWalkStrategy navigates by running a chat model for up to eight +// hops, each carrying the structure and every page read so far. But +// navigation is two judgements over candidate sets the tree already +// provides: which sections could hold the answer, then which pages in +// them do. A Judge answers each set in one batched request. The +// generative model, if any, is then asked one question over the +// evidence pages only — and never asked to navigate. + +// NavLeaf is one navigable section: a leaf of the TOC tree with pages. +type NavLeaf struct { + ID string + Title string + Path string // "PART II > Item 7. Management's Discussion…" + Start int + End int + Summary string // optional; shown to the Judge when present +} + +// NavPage is one unit of text the page ranking judges: a real page, or +// a chunk of a section body when pages are not individually addressable. +type NavPage struct { + Number int + Text string + LeafID string +} + +// LeafScore is the Judge's probability that a leaf holds what the +// question needs. +type LeafScore struct { + Leaf NavLeaf + P float64 +} + +// PageScore is the Judge's probability that a page contains it. +type PageScore struct { + Page NavPage + P float64 +} + +// NavResult is what navigation found. +type NavResult struct { + Leaves []LeafScore // every leaf, best first + Selected []NavLeaf // the leaves whose pages were read + Coarse []PageScore // the coarse pass over page heads, when it ran; best first + Pages []PageScore // every page read in full, best first + Evidence []PageScore // pages above threshold, best first; never empty when any page was read + Followed []NavLeaf // leaves read because an evidence page referred to them + Usage Usage + Requests int +} + +// JudgeNavigator ranks leaves, then pages, on a Judge. +type JudgeNavigator struct { + Judge llmgate.Judge + + // Threshold is the probability at or above which a page counts as + // evidence. Zero selects 0.5. + Threshold float64 + + // MaxLeaves bounds how many sections' pages are gathered. Zero + // selects 5. + MaxLeaves int + + // MaxPages bounds how many pages are judged in full. Zero selects 40. + MaxPages int + + // CoarsePages bounds how many gathered pages the coarse pass may + // rank by their heads before the best MaxPages are read in full. + // Zero selects 120. A 10-K's Item 8 alone is 70 pages; without the + // coarse pass the page budget cut it off at 40 and the statement on + // page 113 was never read. + CoarsePages int + + // HeadChars is how much of a page the coarse pass sees. Zero selects + // 700 — the heading and the first rows of a table. + HeadChars int + + // FollowReferences, when true (the default), takes one more hop when + // the evidence pages point elsewhere — "see Note 21" — and the leaf + // they point to has not been read. A 10-K's Item 3 is one sentence + // that refers to the note where the legal proceedings actually are. + NoFollowReferences bool + + // PageChars truncates each page's text before the Judge sees it. + // Zero selects 6000 — a dense filing page is ~4–5k characters. + PageChars int + + // RequestBudgetTokens bounds the STATE of one page-ranking request. + // The provider treats the whole state object as shared context for + // every question — there is no per-question state — so state plus + // the longest question must stay under 32k tokens. Zero selects + // 24k, the same ceiling the TOC stage uses; that is six to eight + // dense filing pages per request. + RequestBudgetTokens int +} + +const ( + defaultNavThreshold = 0.5 + defaultNavMaxLeaves = 5 + defaultNavMaxPages = 40 + defaultNavCoarse = 120 + defaultNavHeadChars = 700 + defaultNavPageChars = 6000 + defaultNavReqTokens = 24_000 + navLeafBatch = 120 + navMinEvidencePages = 2 + navLeafStateMaxChars = 300 +) + +func (n *JudgeNavigator) threshold() float64 { + if n.Threshold > 0 { + return n.Threshold + } + return defaultNavThreshold +} + +func (n *JudgeNavigator) maxLeaves() int { + if n.MaxLeaves > 0 { + return n.MaxLeaves + } + return defaultNavMaxLeaves +} + +func (n *JudgeNavigator) maxPages() int { + if n.MaxPages > 0 { + return n.MaxPages + } + return defaultNavMaxPages +} + +func (n *JudgeNavigator) coarsePages() int { + if n.CoarsePages > 0 { + return n.CoarsePages + } + return defaultNavCoarse +} + +func (n *JudgeNavigator) headChars() int { + if n.HeadChars > 0 { + return n.HeadChars + } + return defaultNavHeadChars +} + +func (n *JudgeNavigator) pageChars() int { + if n.PageChars > 0 { + return n.PageChars + } + return defaultNavPageChars +} + +func (n *JudgeNavigator) reqTokens() int { + if n.RequestBudgetTokens > 0 { + return n.RequestBudgetTokens + } + return defaultNavReqTokens +} + +// RankLeaves asks, for every leaf, whether the section is where the +// question's answer would be found. One request per 120 leaves; the +// query is shared state, each leaf is ~40 tokens of question state. +func (n *JudgeNavigator) RankLeaves(ctx context.Context, query string, leaves []NavLeaf) ([]LeafScore, Usage, int, error) { + var usage Usage + if n.Judge == nil { + return nil, usage, 0, fmt.Errorf("judgewalk: no Judge configured") + } + scores := make([]LeafScore, len(leaves)) + for i, l := range leaves { + scores[i] = LeafScore{Leaf: l} + } + requests := 0 + for start := 0; start < len(leaves); start += navLeafBatch { + end := start + navLeafBatch + if end > len(leaves) { + end = len(leaves) + } + state := map[string]any{"question": query} + questions := map[string]llmgate.Question{} + for i := start; i < end; i++ { + l := leaves[i] + qk := fmt.Sprintf("l_%d", i) + item := map[string]any{"title": l.Title, "pages": fmt.Sprintf("%d-%d", l.Start, l.End)} + if l.Path != "" && l.Path != l.Title { + item["path"] = l.Path + } + if s := strings.TrimSpace(l.Summary); s != "" { + if len(s) > navLeafStateMaxChars { + s = s[:navLeafStateMaxChars] + } + item["summary"] = s + } + state[qk] = item + questions[qk] = llmgate.Noul{ + Instructions: fmt.Sprintf( + "`question` is a question about a document. `%s` is one section of that "+ + "document's table of contents: its title, where it sits, and its page range. "+ + "Would the information needed to answer the question be found in this "+ + "section? Judge from what such a section of such a document contains: "+ + "figures live in the statements and their notes, discussion of results in "+ + "the MD&A, customers, competition and outlook in the business and risk "+ + "sections.", qk), + Criteria: &llmgate.NoulCriteria{ + True: "This section is where a reader would look for the answer", + False: "The answer would not be in this section", + }, + } + } + res, err := n.Judge.Judge(ctx, llmgate.JudgeRequest{State: state, Questions: questions}) + if err != nil { + return nil, usage, requests, err + } + requests++ + usage.Add(judgeUsage(res)) + for qk := range questions { + p, err := res.Noul(qk) + if err != nil { + continue + } + var i int + fmt.Sscanf(qk, "l_%d", &i) + scores[i].P = p + } + } + sort.SliceStable(scores, func(i, j int) bool { return scores[i].P > scores[j].P }) + return scores, usage, requests, nil +} + +// RankPages asks, for every page, whether it contains the facts or +// figures the question asks for. Pages are batched so the request's +// whole state — query plus every page in the batch — stays under the +// budget, counted with the provider's tokenizer. +func (n *JudgeNavigator) RankPages(ctx context.Context, query string, pages []NavPage) ([]PageScore, Usage, int, error) { + return n.rankPages(ctx, query, pages, n.pageChars(), false) +} + +// rankPages is RankPages with the text window chosen by the caller: +// the full page, or just its head for the coarse pass. +func (n *JudgeNavigator) rankPages(ctx context.Context, query string, pages []NavPage, limit int, coarse bool) ([]PageScore, Usage, int, error) { + var usage Usage + if n.Judge == nil { + return nil, usage, 0, fmt.Errorf("judgewalk: no Judge configured") + } + scores := make([]PageScore, len(pages)) + for i, p := range pages { + scores[i] = PageScore{Page: p} + } + requests := 0 + budget := n.reqTokens() + for start := 0; start < len(pages); { + state := map[string]any{"question": query} + questions := map[string]llmgate.Question{} + used := countTokens(query) + 100 + end := start + for end < len(pages) { + text := pages[end].Text + if len(text) > limit { + text = text[:limit] + } + cost := countTokens(text) + 20 + if end > start && used+cost > budget { + break + } + qk := fmt.Sprintf("p_%d", end) + state[qk] = map[string]any{"page": pages[end].Number, "text": text} + if coarse { + questions[qk] = llmgate.Noul{ + Instructions: fmt.Sprintf( + "`question` is a question about a document. `%s.text` is the START of page "+ + "`%s.page` — its heading and first lines. Could the facts or figures needed "+ + "to answer the question be on this page, judging from what it opens with?", qk, qk), + Criteria: &llmgate.NoulCriteria{ + True: "This page's opening says it is the kind of page that would hold the answer", + False: "This page opens on something unrelated", + }, + } + } else { + questions[qk] = llmgate.Noul{ + Instructions: fmt.Sprintf( + "`question` is a question about a document. `%s.text` is the text of page "+ + "`%s.page`. Does this page contain the specific facts or figures needed to "+ + "answer the question — the number, the statement, the table row? A page that "+ + "only mentions the topic, or refers the reader elsewhere, does not.", qk, qk), + Criteria: &llmgate.NoulCriteria{ + True: "The answer, or a figure it is computed from, is on this page", + False: "The page is about something else, or only mentions the topic", + }, + } + } + used += cost + end++ + } + res, err := n.Judge.Judge(ctx, llmgate.JudgeRequest{State: state, Questions: questions}) + if err != nil { + return nil, usage, requests, err + } + requests++ + usage.Add(judgeUsage(res)) + for qk := range questions { + p, err := res.Noul(qk) + if err != nil { + continue + } + var i int + fmt.Sscanf(qk, "p_%d", &i) + scores[i].P = p + } + start = end + } + sort.SliceStable(scores, func(i, j int) bool { return scores[i].P > scores[j].P }) + return scores, usage, requests, nil +} + +// Navigate ranks the leaves, reads the pages of the best few, ranks +// those pages, and returns the evidence set. loadPages returns the +// pages (or body chunks) of one leaf. +func (n *JudgeNavigator) Navigate(ctx context.Context, query string, leaves []NavLeaf, loadPages func(ctx context.Context, leaf NavLeaf) ([]NavPage, error)) (*NavResult, error) { + out := &NavResult{} + if len(leaves) == 0 { + return out, nil + } + ranked, u, r, err := n.RankLeaves(ctx, query, leaves) + if err != nil { + return nil, fmt.Errorf("judgewalk: rank leaves: %w", err) + } + out.Usage.Add(u) + out.Requests += r + out.Leaves = ranked + + // Gather the pages of the best leaves, in rank order, up to the + // coarse budget. A leaf below threshold is still read when nothing + // better exists: a low-confidence best guess beats reading nothing. + // A page two overlapping leaves both cover is gathered once. + var pages []NavPage + seen := map[int]bool{} + coarseCap := n.coarsePages() + for _, ls := range ranked { + if len(out.Selected) >= n.maxLeaves() || len(pages) >= coarseCap { + break + } + ps, err := loadPages(ctx, ls.Leaf) + if err != nil { + return nil, fmt.Errorf("judgewalk: load %q: %w", ls.Leaf.Title, err) + } + added := 0 + for _, p := range ps { + if seen[p.Number] || len(pages) >= coarseCap { + continue + } + seen[p.Number] = true + p.LeafID = ls.Leaf.ID + pages = append(pages, p) + added++ + } + if added > 0 { + out.Selected = append(out.Selected, ls.Leaf) + } + } + if len(pages) == 0 { + return out, nil + } + + // More pages than the full-text budget: a coarse pass over page + // heads picks which ones deserve their whole text. Heads are + // ~150 tokens, so 120 of them is one request. + maxP := n.maxPages() + if len(pages) > maxP { + heads, u, r, err := n.rankPages(ctx, query, pages, n.headChars(), true) + if err != nil { + return nil, fmt.Errorf("judgewalk: rank page heads: %w", err) + } + out.Usage.Add(u) + out.Requests += r + out.Coarse = heads + pages = pages[:0] + for _, h := range heads[:maxP] { + pages = append(pages, h.Page) + } + sort.Slice(pages, func(i, j int) bool { return pages[i].Number < pages[j].Number }) + } + scored, u, r, err := n.RankPages(ctx, query, pages) + if err != nil { + return nil, fmt.Errorf("judgewalk: rank pages: %w", err) + } + out.Usage.Add(u) + out.Requests += r + out.Pages = scored + th := n.threshold() + fill := func(scored []PageScore) { + out.Evidence = out.Evidence[:0] + for _, ps := range scored { + if ps.P >= th { + out.Evidence = append(out.Evidence, ps) + } + } + // Never come back empty-handed: the best pages are the evidence, + // flagged by their probability. + if len(out.Evidence) < navMinEvidencePages { + out.Evidence = append(out.Evidence[:0], scored[:min(navMinEvidencePages, len(scored))]...) + } + } + fill(scored) + + // One more hop when the evidence points elsewhere. Code reads the + // reference, the tree names the leaf, the Judge reads its pages. + if !n.NoFollowReferences { + var refPages []NavPage + readLeaf := map[string]bool{} + for _, l := range out.Selected { + readLeaf[l.ID] = true + } + for _, ev := range out.Evidence { + for _, target := range referencedLeaves(ev.Page.Text, leaves) { + if readLeaf[target.ID] { + continue + } + readLeaf[target.ID] = true + ps, err := loadPages(ctx, target) + if err != nil { + return nil, fmt.Errorf("judgewalk: follow %q: %w", target.Title, err) + } + for _, p := range ps { + if !seen[p.Number] && len(refPages) < maxP { + seen[p.Number] = true + p.LeafID = target.ID + refPages = append(refPages, p) + } + } + out.Followed = append(out.Followed, target) + } + } + if len(refPages) > 0 { + more, u, r, err := n.RankPages(ctx, query, refPages) + if err != nil { + return nil, fmt.Errorf("judgewalk: rank referenced pages: %w", err) + } + out.Usage.Add(u) + out.Requests += r + out.Pages = append(out.Pages, more...) + sort.SliceStable(out.Pages, func(i, j int) bool { return out.Pages[i].P > out.Pages[j].P }) + fill(out.Pages) + } + } + return out, nil +} + +// countTokens uses the provider's tokenizer, so the batch budget is +// measured the way the request will be. Dense financial tables run +// near one token per two characters; a bytes/4 guess overflowed. +func countTokens(text string) int { + if n, err := typesafe.EstimateTokens(text); err == nil { + return n + } + return len(text)/3 + 1 +} + +// reReference finds the cross-references a page makes: "see Note 21", +// "Item 1A", "Note 14 to the Consolidated Financial Statements". +var reReference = regexp.MustCompile(`(?i)\b(note|item|part|section|schedule)\s+(\d+[a-c]?)\b`) + +// referencedLeaves returns the leaves a page's cross-references name, +// matched by label and number at the start of the leaf's title. +func referencedLeaves(text string, leaves []NavLeaf) []NavLeaf { + var out []NavLeaf + seen := map[string]bool{} + for _, m := range reReference.FindAllStringSubmatch(text, -1) { + label, num := strings.ToLower(m[1]), strings.ToLower(m[2]) + key := label + " " + num + if seen[key] { + continue + } + seen[key] = true + for _, l := range leaves { + t := strings.ToLower(strings.TrimSpace(l.Title)) + // "Note 21. Legal Proceedings", "NOTE 21 — …", "Item 1A." + if strings.HasPrefix(t, key) { + rest := t[len(key):] + if rest == "" || !(rest[0] >= '0' && rest[0] <= '9') && !(rest[0] >= 'a' && rest[0] <= 'z') { + out = append(out, l) + break + } + } + } + } + return out +} + +func judgeUsage(res *llmgate.Judgment) Usage { + if res == nil { + return Usage{} + } + return Usage{ + InputTokens: res.Usage.InputTokens, + OutputTokens: res.Usage.OutputTokens, + TotalTokens: res.Usage.TotalTokens, + CostUSD: res.Usage.CostUSD, + LLMCalls: 1, + } +} + +// JudgeWalkStrategy is JudgeNavigator as a Strategy over a section +// tree. Leaves are the sections that carry a page range; a section's +// body, loaded through PageLoader, is chunked into page-sized units for +// the page ranking. The result names the sections the evidence came +// from and their page ranges. It does not write an answer: answering +// is the one generative step, and it belongs to the caller, over the +// evidence only. +type JudgeWalkStrategy struct { + Navigator JudgeNavigator + PageLoader PageContentLoader +} + +const strategyNameJudgeWalk = "judgewalk" + +// NewJudgeWalkStrategy builds the strategy on a Judge. +func NewJudgeWalkStrategy(j llmgate.Judge) *JudgeWalkStrategy { + return &JudgeWalkStrategy{Navigator: JudgeNavigator{Judge: j}} +} + +func (s *JudgeWalkStrategy) Name() string { return strategyNameJudgeWalk } + +// Select returns the section IDs the evidence pages came from. +func (s *JudgeWalkStrategy) Select(ctx context.Context, t *tree.Tree, query string, budget ContextBudget) ([]tree.SectionID, error) { + res, err := s.SelectWithCost(ctx, t, query, budget) + if err != nil { + return nil, err + } + return res.SelectedIDs, nil +} + +// SelectWithCost runs navigation and reports usage. +func (s *JudgeWalkStrategy) SelectWithCost(ctx context.Context, t *tree.Tree, query string, _ ContextBudget) (*Result, error) { + if t == nil || t.Root == nil { + return &Result{}, nil + } + sections := flattenSectionsByPage(t) + byID := map[string]sectionPageEntry{} + paths := sectionPaths(t) + var leaves []NavLeaf + for _, sec := range sections { + byID[string(sec.id)] = sec + leaves = append(leaves, NavLeaf{ + ID: string(sec.id), Title: sec.title, Path: paths[sec.id], + Start: sec.start, End: sec.end, Summary: sec.summary, + }) + } + load := func(ctx context.Context, leaf NavLeaf) ([]NavPage, error) { + sec := byID[leaf.ID] + if s.PageLoader == nil || sec.contentRef == "" { + return nil, nil + } + b, err := s.PageLoader.Load(ctx, sec.contentRef) + if err != nil { + return nil, err + } + return chunkBody(string(b), sec.start, sec.end, s.Navigator.pageChars()), nil + } + nav, err := s.Navigator.Navigate(ctx, query, leaves, load) + if err != nil { + return nil, err + } + var ids []tree.SectionID + conf := map[tree.SectionID]float64{} + var ranges []pageRange + seen := map[string]bool{} + for _, ev := range nav.Evidence { + id := tree.SectionID(ev.Page.LeafID) + if !seen[ev.Page.LeafID] { + seen[ev.Page.LeafID] = true + ids = append(ids, id) + sec := byID[ev.Page.LeafID] + ranges = append(ranges, pageRange{Start: sec.start, End: sec.end}) + } + if ev.P > conf[id] { + conf[id] = ev.P + } + } + best := 0.0 + if len(nav.Evidence) > 0 { + best = nav.Evidence[0].P + } + return &Result{ + SelectedIDs: ids, + Confidences: conf, + Confidence: best, + CitedPages: rangesToPairs(ranges), + ModelUsed: "judge", + Usage: nav.Usage, + HopsTaken: nav.Requests, + }, nil +} + +// sectionPaths maps each section to its "parent > child" title path. +func sectionPaths(t *tree.Tree) map[tree.SectionID]string { + out := map[tree.SectionID]string{} + var walk func(s *tree.Section, prefix string) + walk = func(s *tree.Section, prefix string) { + p := s.Title + if prefix != "" && s.Title != "" { + p = prefix + " > " + s.Title + } else if prefix != "" { + p = prefix + } + out[s.ID] = p + for i := range s.Children { + walk(s.Children[i], p) + } + } + if t != nil && t.Root != nil { + for i := range t.Root.Children { + walk(t.Root.Children[i], "") + } + } + return out +} + +// chunkBody splits a section body into page-sized units, numbered +// through the section's page range so a chunk can be cited by page. +func chunkBody(body string, start, end, size int) []NavPage { + body = strings.TrimSpace(body) + if body == "" { + return nil + } + if size <= 0 { + size = defaultNavPageChars + } + var chunks []string + for len(body) > size { + cut := strings.LastIndex(body[:size], "\n") + if cut < size/2 { + cut = size + } + chunks = append(chunks, body[:cut]) + body = strings.TrimSpace(body[cut:]) + } + if body != "" { + chunks = append(chunks, body) + } + span := end - start + 1 + if span < 1 { + span = 1 + } + out := make([]NavPage, len(chunks)) + for i, c := range chunks { + page := start + i*span/len(chunks) + if page > end { + page = end + } + out[i] = NavPage{Number: page, Text: c} + } + return out +} diff --git a/pkg/retrieval/judgewalk_test.go b/pkg/retrieval/judgewalk_test.go new file mode 100644 index 0000000..8999cbd --- /dev/null +++ b/pkg/retrieval/judgewalk_test.go @@ -0,0 +1,285 @@ +package retrieval + +import ( + "context" + "errors" + "strings" + "testing" + + "github.com/hallelx2/llmgate" + + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +// navJudge answers leaf questions by title keyword and page questions +// by text keyword, and counts requests. +func navJudge(leafHit, pageHit string) (*llmgate.MockJudge, *int) { + calls := 0 + j := &llmgate.MockJudge{Respond: func(_ context.Context, req llmgate.JudgeRequest) (*llmgate.Judgment, error) { + calls++ + st := req.State.(map[string]any) + ans := map[string]llmgate.Answer{} + for id := range req.Questions { + item := st[id].(map[string]any) + p := 0.1 + if strings.HasPrefix(id, "l_") && strings.Contains(strings.ToLower(item["title"].(string)), leafHit) { + p = 0.9 + } + if strings.HasPrefix(id, "p_") && strings.Contains(item["text"].(string), pageHit) { + p = 0.95 + } + ans[id] = llmgate.NoulAnswer{Noul: p} + } + return &llmgate.Judgment{Model: "mock", Answers: ans, Usage: llmgate.Usage{InputTokens: 10, TotalTokens: 10, TokensReported: true}}, nil + }} + return j, &calls +} + +func tenKLeaves() []NavLeaf { + return []NavLeaf{ + {ID: "1", Title: "Item 1. Business", Path: "PART I > Item 1. Business", Start: 3, End: 19}, + {ID: "2", Title: "Item 1A. Risk Factors", Start: 20, End: 33}, + {ID: "7", Title: "Item 7. Management's Discussion and Analysis", Start: 36, End: 49}, + {ID: "8", Title: "Item 8. Financial Statements and Supplementary Data", Start: 52, End: 91}, + {ID: "9", Title: "Item 9. Changes in and Disagreements with Accountants", Start: 92, End: 92}, + } +} + +func TestNavigateReadsTheBestLeafAndFindsTheEvidencePage(t *testing.T) { + j, calls := navJudge("financial statements", "Total revenue 17,606") + n := &JudgeNavigator{Judge: j, MaxLeaves: 1} + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + var ps []NavPage + for p := l.Start; p <= l.End && p < l.Start+6; p++ { + text := "Notes to the statements, page " + strings.Repeat("x", 20) + if p == l.Start+2 { + text = "CONSOLIDATED STATEMENTS OF INCOME\nTotal revenue 17,606 15,785" + } + ps = append(ps, NavPage{Number: p, Text: text}) + } + return ps, nil + } + res, err := n.Navigate(context.Background(), "What was Adobe's total revenue in FY2022?", tenKLeaves(), load) + if err != nil { + t.Fatal(err) + } + if len(res.Selected) != 1 || res.Selected[0].ID != "8" { + t.Fatalf("selected %+v, want Item 8", res.Selected) + } + if len(res.Evidence) == 0 || res.Evidence[0].Page.Number != 54 { + t.Fatalf("evidence %+v, want page 54 first", res.Evidence) + } + if res.Evidence[0].Page.LeafID != "8" { + t.Errorf("evidence page should carry its leaf: %+v", res.Evidence[0].Page) + } + if *calls != 2 || res.Requests != 2 { + t.Errorf("requests: mock saw %d, result says %d; want 2 (leaves, pages)", *calls, res.Requests) + } + if res.Coarse != nil { + t.Errorf("six pages under a 40-page budget need no coarse pass") + } +} + +// A 70-page section: the coarse pass over page heads picks the pages +// worth reading in full, and the page deep in the section is found. +func TestNavigateCoarsePassReachesDeepIntoABigLeaf(t *testing.T) { + j, calls := navJudge("financial statements", "Total revenue 17,606") + n := &JudgeNavigator{Judge: j, MaxLeaves: 1, MaxPages: 10, CoarsePages: 120} + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + var ps []NavPage + for p := l.Start; p <= l.End; p++ { + text := "Notes to the statements " + strings.Repeat("x", 3000) + if p == 85 { + text = "CONSOLIDATED STATEMENTS OF INCOME\nTotal revenue 17,606 15,785" + strings.Repeat("y", 3000) + } + ps = append(ps, NavPage{Number: p, Text: text}) + } + return ps, nil + } + leaves := []NavLeaf{{ID: "8", Title: "Item 8. Financial Statements", Start: 52, End: 125}} + res, err := n.Navigate(context.Background(), "total revenue?", leaves, load) + if err != nil { + t.Fatal(err) + } + if len(res.Coarse) != 74 { + t.Errorf("coarse pass should have seen all 74 page heads, saw %d", len(res.Coarse)) + } + if len(res.Pages) != 10 { + t.Errorf("full pass should read MaxPages=10, read %d", len(res.Pages)) + } + if len(res.Evidence) == 0 || res.Evidence[0].Page.Number != 85 { + t.Fatalf("page 85 not found: %+v", res.Evidence) + } + if *calls < 3 { + t.Errorf("want leaves + coarse + full requests, got %d", *calls) + } +} + +func TestNavigateGathersASharedPageOnce(t *testing.T) { + j, _ := navJudge("item", "needle") + n := &JudgeNavigator{Judge: j, MaxLeaves: 2} + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + var ps []NavPage + for p := l.Start; p <= l.End; p++ { + ps = append(ps, NavPage{Number: p, Text: "needle on page"}) + } + return ps, nil + } + leaves := []NavLeaf{{ID: "a", Title: "Item 9A", Start: 60, End: 61}, {ID: "b", Title: "Item 9B", Start: 61, End: 62}} + res, err := n.Navigate(context.Background(), "q", leaves, load) + if err != nil { + t.Fatal(err) + } + if len(res.Pages) != 3 { + t.Errorf("pages 60,61,62 should be read once each, read %d", len(res.Pages)) + } +} + +func TestNavigateNeverReturnsEmptyEvidence(t *testing.T) { + j, _ := navJudge("nothing matches", "nothing matches") + n := &JudgeNavigator{Judge: j, MaxLeaves: 2} + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + return []NavPage{{Number: l.Start, Text: "prose"}, {Number: l.Start + 1, Text: "more prose"}}, nil + } + res, err := n.Navigate(context.Background(), "q", tenKLeaves(), load) + if err != nil { + t.Fatal(err) + } + if len(res.Selected) != 2 { + t.Errorf("below-threshold leaves should still be read up to MaxLeaves: %d", len(res.Selected)) + } + if len(res.Evidence) != navMinEvidencePages { + t.Errorf("want the best %d pages as low-confidence evidence, got %d", navMinEvidencePages, len(res.Evidence)) + } + if res.Evidence[0].P >= n.threshold() { + t.Errorf("fallback evidence should carry its real (low) probability, got %v", res.Evidence[0].P) + } +} + +func TestRankPagesBatchesUnderTheRequestBudget(t *testing.T) { + j, calls := navJudge("", "needle") + n := &JudgeNavigator{Judge: j, RequestBudgetTokens: 1200, PageChars: 2000} + var pages []NavPage + for i := 0; i < 10; i++ { + text := strings.Repeat("y", 1900) + if i == 7 { + text = "needle " + text + } + pages = append(pages, NavPage{Number: i + 1, Text: text}) + } + scored, _, reqs, err := n.RankPages(context.Background(), "q", pages) + if err != nil { + t.Fatal(err) + } + // 1900 chars of "y" is ~475 tokens; two fit under 1200 with the + // query, a third does not. + if reqs < 4 || *calls != reqs { + t.Errorf("10 pages of ~475 tokens under a 1200-token budget should take ≥4 requests, took %d (mock saw %d)", reqs, *calls) + } + if scored[0].Page.Number != 8 { + t.Errorf("best page should be the one with the needle, got %d", scored[0].Page.Number) + } +} + +func TestNavigateSurfacesAJudgeFailure(t *testing.T) { + j := &llmgate.MockJudge{Err: errors.New("typesafe: request failed")} + n := &JudgeNavigator{Judge: j} + _, err := n.Navigate(context.Background(), "q", tenKLeaves(), func(context.Context, NavLeaf) ([]NavPage, error) { return nil, nil }) + if err == nil || !strings.Contains(err.Error(), "request failed") { + t.Errorf("a failed Judge must be an error, not an empty result: %v", err) + } +} + +type mapLoader map[string]string + +func (m mapLoader) Load(_ context.Context, ref string) ([]byte, error) { return []byte(m[ref]), nil } + +func TestJudgeWalkStrategyOverATree(t *testing.T) { + j, _ := navJudge("financial", "Total revenue") + s := NewJudgeWalkStrategy(j) + s.Navigator.MaxLeaves = 1 + s.PageLoader = mapLoader{"ref8": "CONSOLIDATED STATEMENTS OF INCOME\nTotal revenue 17,606", "ref1": "We are a company."} + tr := &tree.Tree{DocumentID: "d", Root: &tree.Section{ID: "root", Children: []*tree.Section{ + {ID: "s1", Title: "Item 1. Business", PageStart: 3, PageEnd: 19, ContentRef: "ref1"}, + {ID: "s8", Title: "Item 8. Financial Statements", PageStart: 52, PageEnd: 91, ContentRef: "ref8"}, + }}} + res, err := s.SelectWithCost(context.Background(), tr, "total revenue?", ContextBudget{}) + if err != nil { + t.Fatal(err) + } + if len(res.SelectedIDs) != 1 || res.SelectedIDs[0] != "s8" { + t.Fatalf("selected %v want [s8]", res.SelectedIDs) + } + if len(res.CitedPages) != 1 || res.CitedPages[0] != [2]int{52, 91} { + t.Errorf("cited pages %v", res.CitedPages) + } + if res.Usage.LLMCalls != 2 || res.HopsTaken != 2 { + t.Errorf("usage %+v hops %d", res.Usage, res.HopsTaken) + } + if s.Name() != "judgewalk" { + t.Errorf("name %q", s.Name()) + } +} + +func TestChunkBodyNumbersChunksThroughThePageRange(t *testing.T) { + body := strings.Repeat("line of text\n", 1000) // 13k chars + chunks := chunkBody(body, 10, 19, 6000) + if len(chunks) != 3 { + t.Fatalf("chunks: %d", len(chunks)) + } + if chunks[0].Number != 10 || chunks[2].Number > 19 || chunks[1].Number < 10 { + t.Errorf("chunk pages: %d %d %d", chunks[0].Number, chunks[1].Number, chunks[2].Number) + } +} + +// Item 3 says "see Note 21"; the answer is in Note 21, which the leaf +// ranking did not pick. The navigator follows the reference. +func TestNavigateFollowsACrossReference(t *testing.T) { + j, calls := navJudge("legal proceedings", "class action filed") + n := &JudgeNavigator{Judge: j, MaxLeaves: 1} + leaves := []NavLeaf{ + {ID: "3", Title: "Item 3. Legal Proceedings", Start: 20, End: 20}, + {ID: "n21", Title: "Note 21 – Legal Proceedings", Start: 113, End: 114}, + {ID: "n20", Title: "Note 20 – Leases", Start: 110, End: 112}, + } + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + switch l.ID { + case "3": + return []NavPage{{Number: 20, Text: "Item 3. Legal Proceedings\nCurrently, we are involved in a number of legal proceedings. For a discussion of contingencies see Note 21 to our Consolidated Financial Statements."}}, nil + case "n21": + return []NavPage{{Number: 113, Text: "Note 21 – Legal Proceedings\nA class action filed in 2019 remains pending."}, {Number: 114, Text: "Other matters."}}, nil + } + return []NavPage{{Number: l.Start, Text: "leases"}}, nil + } + res, err := n.Navigate(context.Background(), "Has Boeing reported any materially important ongoing legal battles?", leaves, load) + if err != nil { + t.Fatal(err) + } + if len(res.Followed) != 1 || res.Followed[0].ID != "n21" { + t.Fatalf("followed %+v, want Note 21", res.Followed) + } + if len(res.Evidence) == 0 || res.Evidence[0].Page.Number != 113 { + t.Fatalf("evidence %+v, want page 113 first", res.Evidence) + } + if *calls != 3 { + t.Errorf("requests: %d, want 3 (leaves, Item 3 page, Note 21 pages)", *calls) + } + // Turned off, it stays on Item 3. + n.NoFollowReferences = true + res, _ = n.Navigate(context.Background(), "q", leaves, load) + if len(res.Followed) != 0 { + t.Errorf("NoFollowReferences ignored") + } +} + +func TestReferencedLeaves(t *testing.T) { + leaves := []NavLeaf{{ID: "a", Title: "Note 2. Revenue"}, {ID: "b", Title: "Note 21 – Legal Proceedings"}, {ID: "c", Title: "Item 1A. Risk Factors"}, {ID: "d", Title: "Item 1. Business"}} + got := referencedLeaves("see Note 21 and Item 1A; also note 2 above", leaves) + ids := []string{} + for _, l := range got { + ids = append(ids, l.ID) + } + if strings.Join(ids, ",") != "b,c,a" { + t.Errorf("got %v want [b c a] — and Note 2 must not match Note 21, Item 1 must not match Item 1A", ids) + } +}