From 831025b67a3f36ace80e6e26eba03a40a0591b54 Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 18:02:31 +0100 Subject: [PATCH 1/4] =?UTF-8?q?feat(retrieval):=20navigation=20on=20a=20Ju?= =?UTF-8?q?dge=20=E2=80=94=20rank=20the=20leaves,=20then=20the=20pages,=20?= =?UTF-8?q?no=20generative=20call?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit HAL-1371. TreeWalkStrategy navigates with up to eight chat-completion hops, each carrying the structure and every page read so far. But navigation is two judgements over candidate sets the tree already provides. JudgeNavigator asks one Noul per leaf — would the answer be found in this section — in one request, reads the pages of the best few, and asks one Noul per page — does this page contain the specific facts or figures — batched under the request budget. The pages above threshold are the evidence; when none are, the best two are returned with their real probability rather than nothing. JudgeWalkStrategy is the same over a section tree: sections with page ranges are the leaves, a section's body is chunked into page-sized units for the page ranking, and the result names the sections the evidence came from. It writes no answer: answering is the one generative step and belongs to the caller, over the evidence only. Selectable as strategy=judgewalk in cmd/server and cmd/engine whenever llm.judge is configured; the judge is now built before the strategies. cmd/navbench measures it on FinanceBench: for each question, are the gold evidence pages in the evidence set, how many pages were read, at what cost and time. --- cmd/engine/main.go | 33 ++- cmd/navbench/main.go | 314 ++++++++++++++++++++ cmd/server/main.go | 51 +++- pkg/retrieval/judgewalk.go | 502 ++++++++++++++++++++++++++++++++ pkg/retrieval/judgewalk_test.go | 173 +++++++++++ 5 files changed, 1046 insertions(+), 27 deletions(-) create mode 100644 cmd/navbench/main.go create mode 100644 pkg/retrieval/judgewalk.go create mode 100644 pkg/retrieval/judgewalk_test.go diff --git a/cmd/engine/main.go b/cmd/engine/main.go index 9eed587..498f07d 100644 --- a/cmd/engine/main.go +++ b/cmd/engine/main.go @@ -11,6 +11,7 @@ import ( "flag" "fmt" "io" + "log" "log/slog" "net/http" "os" @@ -126,7 +127,17 @@ func run() error { return fmt.Errorf("init llm: %w", err) } } - strategy := buildStrategy(cfg.Retrieval, llmClient, store) + judge, err := buildJudge(cfg.LLM.Judge) + if err != nil { + logger.Error("judge: config invalid", "err", err) + os.Exit(1) + } + if judge != nil { + logger.Info("judge: typesafe enabled — TOC stage and judgewalk retrieval run on the Judge") + } else { + logger.Warn("judge: none configured — TOC judgements run on the generative driver, one call per page (set TYPESAFE_API_KEY)") + } + strategy := buildStrategy(cfg.Retrieval, llmClient, judge, store) // Wrap with caching if enabled. if cfg.Retrieval.Cache.Enabled { @@ -205,16 +216,6 @@ func run() error { ) } - judge, err := buildJudge(cfg.LLM.Judge) - if err != nil { - logger.Error("judge: config invalid", "err", err) - os.Exit(1) - } - if judge != nil { - logger.Info("judge: typesafe enabled — contents-page detection and page resolution run on the Judge") - } else { - logger.Warn("judge: none configured — TOC judgements run on the generative driver, one call per page (set TYPESAFE_API_KEY)") - } pipeline := ingest.NewPipeline(ingest.Pipeline{ DB: pool, Storage: store, @@ -517,8 +518,16 @@ func buildLLMFrom(c config.LLMConfig, provider, apiKey, baseURL, model string) ( } } -func buildStrategy(c config.RetrievalConfig, client llmgate.Client, store storage.Storage) retrieval.Strategy { +func buildStrategy(c config.RetrievalConfig, client llmgate.Client, judge llmgate.Judge, store storage.Storage) retrieval.Strategy { switch c.Strategy { + case "judgewalk": + if judge == nil { + log.Printf("retrieval: strategy judgewalk needs llm.judge configured; using treewalk") + return buildTreeWalkStrategy(c, client, store) + } + s := retrieval.NewJudgeWalkStrategy(judge) + s.PageLoader = storagePageLoader{s: store} + return s case "single-pass": return retrieval.NewSinglePass(client) case "chunked-tree": diff --git a/cmd/navbench/main.go b/cmd/navbench/main.go new file mode 100644 index 0000000..51ba135 --- /dev/null +++ b/cmd/navbench/main.go @@ -0,0 +1,314 @@ +// Command navbench measures retrieval navigation on a Judge against +// FinanceBench's gold evidence pages, with no generative model. +// +// For each question on a filing with a tree: rank the tree's leaves, +// read the pages of the best few, rank those pages, and ask whether the +// gold evidence pages are in the evidence set. +// +// go run ./cmd/navbench -questions ~/.cache/vlbench/financebench-questions.jsonl \ +// -trees ~/.cache/vlbench/trees-jev2 -pdfs ~/.cache/vlbench/financebench -out nav.jsonl +package main + +import ( + "bufio" + "context" + "encoding/json" + "flag" + "fmt" + "os" + "path/filepath" + "sort" + "strings" + "time" + + "github.com/hallelx2/llmgate/judge/typesafe" + "github.com/hallelx2/llmgate/middleware/retry" + + "github.com/hallelx2/vectorless-engine/pkg/ingest" + "github.com/hallelx2/vectorless-engine/pkg/parser" + "github.com/hallelx2/vectorless-engine/pkg/retrieval" + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +type question struct { + ID string `json:"id"` + Doc string `json:"doc"` + Question string `json:"question"` + Answer string `json:"answer"` + Evidence []int `json:"evidence_pages"` + Type string `json:"type"` +} + +type dump struct { + Doc string `json:"doc"` + Nodes []tree.TOCNode `json:"nodes"` +} + +type outcome struct { + ID string `json:"id"` + Doc string `json:"doc"` + Question string `json:"question"` + Gold []int `json:"gold_pages"` + Selected []string `json:"selected_leaves"` + SelectedPages [][2]int `json:"selected_ranges"` + Evidence []int `json:"evidence_pages"` + EvidenceP []float64 `json:"evidence_p"` + PagesRead int `json:"pages_read"` + LeafHit bool `json:"leaf_hit"` // every gold page inside a selected leaf + Recall float64 `json:"recall"` // share of gold pages in the evidence set + Hit bool `json:"hit"` // recall == 1 + Requests int `json:"requests"` + InTokens int `json:"in_tokens"` + CostUSD float64 `json:"cost_usd"` + Seconds float64 `json:"seconds"` + Err string `json:"err,omitempty"` +} + +func main() { + qPath := flag.String("questions", "", "FinanceBench questions JSONL") + trees := flag.String("trees", "", "directory of tocdump trees") + pdfs := flag.String("pdfs", "", "directory of filings") + out := flag.String("out", "", "JSONL of per-question outcomes") + maxLeaves := flag.Int("leaves", 3, "sections read per question") + maxPages := flag.Int("pages", 40, "pages judged per question") + limit := flag.Int("limit", 0, "stop after this many questions (0 = all)") + flag.Parse() + if *qPath == "" || *trees == "" || *pdfs == "" { + fmt.Fprintln(os.Stderr, "usage: navbench -questions q.jsonl -trees dir -pdfs dir [-out o.jsonl]") + os.Exit(2) + } + + key := os.Getenv(typesafe.EnvAPIKey) + tj, err := typesafe.New(typesafe.Config{APIKey: key}) + if err != nil { + fmt.Fprintln(os.Stderr, "judge:", err) + os.Exit(1) + } + nav := &retrieval.JudgeNavigator{Judge: retry.NewJudge(retry.Config{MaxRetries: 3})(tj), MaxLeaves: *maxLeaves, MaxPages: *maxPages} + + qs := readQuestions(*qPath) + if *limit > 0 && len(qs) > *limit { + qs = qs[:*limit] + } + var of *os.File + if *out != "" { + of, _ = os.Create(*out) + defer of.Close() + } + + pageCache := map[string][]ingest.PageText{} + leafCache := map[string][]retrieval.NavLeaf{} + var results []outcome + for _, q := range qs { + o := outcome{ID: q.ID, Doc: q.Doc, Question: q.Question, Gold: q.Evidence} + leaves, pages, err := load(q.Doc, *trees, *pdfs, leafCache, pageCache) + if err != nil { + o.Err = err.Error() + results = append(results, o) + report(o) + continue + } + byNum := map[int]string{} + for _, p := range pages { + byNum[p.PageNumber] = p.Text + } + loadPages := func(_ context.Context, l retrieval.NavLeaf) ([]retrieval.NavPage, error) { + var ps []retrieval.NavPage + for n := l.Start; n <= l.End; n++ { + if t, ok := byNum[n]; ok { + ps = append(ps, retrieval.NavPage{Number: n, Text: t}) + } + } + return ps, nil + } + ctx, cancel := context.WithTimeout(context.Background(), 5*time.Minute) + start := time.Now() + res, err := nav.Navigate(ctx, q.Question, leaves, loadPages) + cancel() + o.Seconds = time.Since(start).Seconds() + if err != nil { + o.Err = err.Error() + results = append(results, o) + report(o) + continue + } + for _, l := range res.Selected { + o.Selected = append(o.Selected, l.Title) + o.SelectedPages = append(o.SelectedPages, [2]int{l.Start, l.End}) + } + for _, e := range res.Evidence { + o.Evidence = append(o.Evidence, e.Page.Number) + o.EvidenceP = append(o.EvidenceP, e.P) + } + o.PagesRead = len(res.Pages) + o.Requests, o.InTokens, o.CostUSD = res.Requests, res.Usage.InputTokens, res.Usage.CostUSD + o.LeafHit = allInRanges(q.Evidence, o.SelectedPages) + o.Recall = recall(q.Evidence, o.Evidence) + o.Hit = o.Recall == 1 + results = append(results, o) + report(o) + if of != nil { + b, _ := json.Marshal(o) + of.Write(append(b, '\n')) + } + } + summarise(results) +} + +func load(doc, trees, pdfs string, leafCache map[string][]retrieval.NavLeaf, pageCache map[string][]ingest.PageText) ([]retrieval.NavLeaf, []ingest.PageText, error) { + if l, ok := leafCache[doc]; ok { + return l, pageCache[doc], nil + } + raw, err := os.ReadFile(filepath.Join(trees, doc+".json")) + if err != nil { + return nil, nil, err + } + var d dump + if err := json.Unmarshal(raw, &d); err != nil { + return nil, nil, err + } + if len(d.Nodes) == 0 { + return nil, nil, fmt.Errorf("no tree") + } + f, err := os.Open(filepath.Join(pdfs, doc+".pdf")) + if err != nil { + return nil, nil, err + } + pd, err := parser.NewPDF().Parse(context.Background(), f) + f.Close() + if err != nil { + return nil, nil, err + } + pages := ingest.BenchAssemblePages(pd.Sections) + var leaves []retrieval.NavLeaf + var walk func(ns []tree.TOCNode, path string) + walk = func(ns []tree.TOCNode, path string) { + for _, n := range ns { + p := n.Title + if path != "" { + p = path + " > " + n.Title + } + if len(n.Nodes) > 0 { + walk(n.Nodes, p) + continue + } + if n.StartPage > 0 && n.EndPage >= n.StartPage { + leaves = append(leaves, retrieval.NavLeaf{ID: n.NodeID, Title: n.Title, Path: p, Start: n.StartPage, End: n.EndPage}) + } + } + } + walk(d.Nodes, "") + leafCache[doc], pageCache[doc] = leaves, pages + return leaves, pages, nil +} + +func allInRanges(gold []int, ranges [][2]int) bool { + for _, g := range gold { + in := false + for _, r := range ranges { + if g >= r[0] && g <= r[1] { + in = true + break + } + } + if !in { + return false + } + } + return len(gold) > 0 +} + +func recall(gold, got []int) float64 { + if len(gold) == 0 { + return 0 + } + set := map[int]bool{} + for _, g := range got { + set[g] = true + } + n := 0 + for _, g := range gold { + if set[g] { + n++ + } + } + return float64(n) / float64(len(gold)) +} + +func report(o outcome) { + mark := "MISS" + if o.Hit { + mark = "hit " + } else if o.LeafHit { + mark = "leaf" + } + if o.Err != "" { + mark = "err " + } + fmt.Printf("%s %-24s gold %-10v evidence %-16v read %2d %d req %5.1fs $%.5f %s\n", + mark, o.Doc, o.Gold, o.Evidence, o.PagesRead, o.Requests, o.Seconds, o.CostUSD, strings.TrimSpace(o.Err)) +} + +func summarise(rs []outcome) { + var n, hit, leaf, errs int + var rec, secs, cost float64 + var pages, reqs, toks int + for _, o := range rs { + if o.Err != "" { + errs++ + continue + } + n++ + if o.Hit { + hit++ + } + if o.LeafHit { + leaf++ + } + rec += o.Recall + secs += o.Seconds + cost += o.CostUSD + pages += o.PagesRead + reqs += o.Requests + toks += o.InTokens + } + if n == 0 { + fmt.Println("no questions answered") + return + } + var secList []float64 + for _, o := range rs { + if o.Err == "" { + secList = append(secList, o.Seconds) + } + } + sort.Float64s(secList) + fmt.Printf("\nquestions %d (errors %d)\n", n, errs) + fmt.Printf(" gold pages inside a selected section %d / %d (%.3f)\n", leaf, n, float64(leaf)/float64(n)) + fmt.Printf(" every gold page in the evidence set %d / %d (%.3f)\n", hit, n, float64(hit)/float64(n)) + fmt.Printf(" mean page recall %.3f\n", rec/float64(n)) + fmt.Printf(" pages read / question %.1f\n", float64(pages)/float64(n)) + fmt.Printf(" requests / question %.1f\n", float64(reqs)/float64(n)) + fmt.Printf(" input tokens / question %d\n", toks/n) + fmt.Printf(" seconds / question mean %.1f median %.1f\n", secs/float64(n), secList[len(secList)/2]) + fmt.Printf(" cost / question $%.5f total $%.4f\n", cost/float64(n), cost) +} + +func readQuestions(path string) []question { + f, err := os.Open(path) + if err != nil { + fmt.Fprintln(os.Stderr, err) + os.Exit(1) + } + defer f.Close() + var qs []question + sc := bufio.NewScanner(f) + sc.Buffer(make([]byte, 1<<20), 1<<20) + for sc.Scan() { + var q question + if json.Unmarshal(sc.Bytes(), &q) == nil { + qs = append(qs, q) + } + } + return qs +} diff --git a/cmd/server/main.go b/cmd/server/main.go index 13c1089..7f104b8 100644 --- a/cmd/server/main.go +++ b/cmd/server/main.go @@ -19,6 +19,7 @@ import ( "flag" "fmt" "io" + "log" "log/slog" "net/http" "os" @@ -143,7 +144,17 @@ func run() error { if err != nil { return fmt.Errorf("init llm: %w", err) } - strategy := buildStrategy(cfg.Engine.Retrieval, llmClient, store, pool) + judge, err := buildJudge(cfg.Engine.LLM.Judge) + if err != nil { + logger.Error("judge: config invalid", "err", err) + os.Exit(1) + } + if judge != nil { + logger.Info("judge: typesafe enabled — TOC stage and judgewalk retrieval run on the Judge") + } else { + logger.Warn("judge: none configured — TOC judgements run on the generative driver, one call per page (set TYPESAFE_API_KEY)") + } + strategy := buildStrategy(cfg.Engine.Retrieval, llmClient, judge, store, pool) // Wrap with caching if enabled in engine config. if cfg.Engine.Retrieval.Cache.Enabled { @@ -170,7 +181,7 @@ func run() error { // running engine without a redeploy. Built from the raw client so // each override behaves identically to booting with that strategy // as the default (no shared cache wrapper across overrides). - strategies := buildStrategySet(cfg.Engine.Retrieval, llmClient, store, pool) + strategies := buildStrategySet(cfg.Engine.Retrieval, llmClient, judge, store, pool) // Replay store: every /v1/answer and /v1/answer/treewalk response // is stamped with a deterministic trace_token and its body bytes @@ -203,16 +214,6 @@ func run() error { } // ── Ingest pipeline ─────────────────────────────────────────── - judge, err := buildJudge(cfg.Engine.LLM.Judge) - if err != nil { - logger.Error("judge: config invalid", "err", err) - os.Exit(1) - } - if judge != nil { - logger.Info("judge: typesafe enabled — contents-page detection and page resolution run on the Judge") - } else { - logger.Warn("judge: none configured — TOC judgements run on the generative driver, one call per page (set TYPESAFE_API_KEY)") - } pipeline := ingest.NewPipeline(ingest.Pipeline{ DB: pool, Storage: store, @@ -460,8 +461,14 @@ func buildLLM(c enginecfg.LLMConfig) (llmgate.Client, error) { // retrieval.strategy. The DB pool is threaded through so the // treewalk strategy can wire a TOC provider that reads // documents.toc_tree (the other strategies ignore it). -func buildStrategy(c enginecfg.RetrievalConfig, client llmgate.Client, store storage.Storage, pool *db.Pool) retrieval.Strategy { +func buildStrategy(c enginecfg.RetrievalConfig, client llmgate.Client, judge llmgate.Judge, store storage.Storage, pool *db.Pool) retrieval.Strategy { switch c.Strategy { + case "judgewalk": + if judge == nil { + log.Printf("retrieval: strategy judgewalk needs llm.judge configured; using treewalk") + return buildTreeWalkStrategy(c, client, store, pool) + } + return buildJudgeWalkStrategy(judge, store) case "single-pass": return retrieval.NewSinglePass(client) case "chunked-tree": @@ -493,20 +500,34 @@ func buildStrategy(c enginecfg.RetrievalConfig, client llmgate.Client, store sto // from the same config blocks the default builder reads, so an // override behaves identically to booting with that strategy as the // default. -func buildStrategySet(c enginecfg.RetrievalConfig, client llmgate.Client, store storage.Storage, pool *db.Pool) map[string]retrieval.Strategy { +func buildStrategySet(c enginecfg.RetrievalConfig, client llmgate.Client, judge llmgate.Judge, store storage.Storage, pool *db.Pool) map[string]retrieval.Strategy { agentic := retrieval.NewAgentic(client, storageFetcher{s: store}) if c.Agentic.MaxHops > 0 { agentic.MaxHops = c.Agentic.MaxHops } agentic.ModelOverride = c.Agentic.Model - return map[string]retrieval.Strategy{ + set := map[string]retrieval.Strategy{ "single-pass": retrieval.NewSinglePass(client), "chunked-tree": retrieval.NewChunkedTree(client), "agentic": agentic, "treewalk": buildTreeWalkStrategy(c, client, store, pool), "auto": retrieval.NewAuto(retrieval.NewSinglePass(client), buildTreeWalkStrategy(c, client, store, pool)), } + if judge != nil { + set["judgewalk"] = buildJudgeWalkStrategy(judge, store) + } + return set +} + +// buildJudgeWalkStrategy constructs navigation on the Judge: the tree's +// leaves ranked in one request, the best sections' bodies ranked in one +// or two more, no generative call (HAL-1371). Selectable per request as +// strategy=judgewalk whenever llm.judge is configured. +func buildJudgeWalkStrategy(judge llmgate.Judge, store storage.Storage) *retrieval.JudgeWalkStrategy { + s := retrieval.NewJudgeWalkStrategy(judge) + s.PageLoader = storagePageLoader{s: store} + return s } // buildTreeWalkStrategy constructs the page-based agentic strategy diff --git a/pkg/retrieval/judgewalk.go b/pkg/retrieval/judgewalk.go new file mode 100644 index 0000000..c25bd88 --- /dev/null +++ b/pkg/retrieval/judgewalk.go @@ -0,0 +1,502 @@ +package retrieval + +import ( + "context" + "fmt" + "sort" + "strings" + + "github.com/hallelx2/llmgate" + + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +// Retrieval navigation on a Judge — select, don't generate (HAL-1371). +// +// TreeWalkStrategy navigates by running a chat model for up to eight +// hops, each carrying the structure and every page read so far. But +// navigation is two judgements over candidate sets the tree already +// provides: which sections could hold the answer, then which pages in +// them do. A Judge answers each set in one batched request. The +// generative model, if any, is then asked one question over the +// evidence pages only — and never asked to navigate. + +// NavLeaf is one navigable section: a leaf of the TOC tree with pages. +type NavLeaf struct { + ID string + Title string + Path string // "PART II > Item 7. Management's Discussion…" + Start int + End int + Summary string // optional; shown to the Judge when present +} + +// NavPage is one unit of text the page ranking judges: a real page, or +// a chunk of a section body when pages are not individually addressable. +type NavPage struct { + Number int + Text string + LeafID string +} + +// LeafScore is the Judge's probability that a leaf holds what the +// question needs. +type LeafScore struct { + Leaf NavLeaf + P float64 +} + +// PageScore is the Judge's probability that a page contains it. +type PageScore struct { + Page NavPage + P float64 +} + +// NavResult is what navigation found. +type NavResult struct { + Leaves []LeafScore // every leaf, best first + Selected []NavLeaf // the leaves whose pages were read + Pages []PageScore // every page read, best first + Evidence []PageScore // pages above threshold, best first; never empty when any page was read + Usage Usage + Requests int +} + +// JudgeNavigator ranks leaves, then pages, on a Judge. +type JudgeNavigator struct { + Judge llmgate.Judge + + // Threshold is the probability at or above which a page counts as + // evidence. Zero selects 0.5. + Threshold float64 + + // MaxLeaves bounds how many sections' pages are read. Zero selects 3. + MaxLeaves int + + // MaxPages bounds how many pages are judged in total. Zero selects 40. + MaxPages int + + // PageChars truncates each page's text before the Judge sees it. + // Zero selects 6000 — a dense filing page is ~4–5k characters. + PageChars int + + // RequestBudgetTokens bounds one page-ranking request. Zero selects + // 40k, under the provider's 64k request ceiling with room for the + // shared query. + RequestBudgetTokens int +} + +const ( + defaultNavThreshold = 0.5 + defaultNavMaxLeaves = 3 + defaultNavMaxPages = 40 + defaultNavPageChars = 6000 + defaultNavReqTokens = 40_000 + navLeafBatch = 120 + navMinEvidencePages = 2 + navLeafStateMaxChars = 300 +) + +func (n *JudgeNavigator) threshold() float64 { + if n.Threshold > 0 { + return n.Threshold + } + return defaultNavThreshold +} + +func (n *JudgeNavigator) maxLeaves() int { + if n.MaxLeaves > 0 { + return n.MaxLeaves + } + return defaultNavMaxLeaves +} + +func (n *JudgeNavigator) maxPages() int { + if n.MaxPages > 0 { + return n.MaxPages + } + return defaultNavMaxPages +} + +func (n *JudgeNavigator) pageChars() int { + if n.PageChars > 0 { + return n.PageChars + } + return defaultNavPageChars +} + +func (n *JudgeNavigator) reqTokens() int { + if n.RequestBudgetTokens > 0 { + return n.RequestBudgetTokens + } + return defaultNavReqTokens +} + +// RankLeaves asks, for every leaf, whether the section is where the +// question's answer would be found. One request per 120 leaves; the +// query is shared state, each leaf is ~40 tokens of question state. +func (n *JudgeNavigator) RankLeaves(ctx context.Context, query string, leaves []NavLeaf) ([]LeafScore, Usage, int, error) { + var usage Usage + if n.Judge == nil { + return nil, usage, 0, fmt.Errorf("judgewalk: no Judge configured") + } + scores := make([]LeafScore, len(leaves)) + for i, l := range leaves { + scores[i] = LeafScore{Leaf: l} + } + requests := 0 + for start := 0; start < len(leaves); start += navLeafBatch { + end := start + navLeafBatch + if end > len(leaves) { + end = len(leaves) + } + state := map[string]any{"question": query} + questions := map[string]llmgate.Question{} + for i := start; i < end; i++ { + l := leaves[i] + qk := fmt.Sprintf("l_%d", i) + item := map[string]any{"title": l.Title, "pages": fmt.Sprintf("%d-%d", l.Start, l.End)} + if l.Path != "" && l.Path != l.Title { + item["path"] = l.Path + } + if s := strings.TrimSpace(l.Summary); s != "" { + if len(s) > navLeafStateMaxChars { + s = s[:navLeafStateMaxChars] + } + item["summary"] = s + } + state[qk] = item + questions[qk] = llmgate.Noul{ + Instructions: fmt.Sprintf( + "`question` is a question about a document. `%s` is one section of that "+ + "document's table of contents: its title, where it sits, and its page range. "+ + "Would the information needed to answer the question be found in this "+ + "section? Judge from what such a section of such a document contains — a "+ + "financial statement holds the figures, an MD&A holds the discussion, a "+ + "risk-factors section holds neither.", qk), + Criteria: &llmgate.NoulCriteria{ + True: "This section is where a reader would look for the answer", + False: "The answer would not be in this section", + }, + } + } + res, err := n.Judge.Judge(ctx, llmgate.JudgeRequest{State: state, Questions: questions}) + if err != nil { + return nil, usage, requests, err + } + requests++ + usage.Add(judgeUsage(res)) + for qk := range questions { + p, err := res.Noul(qk) + if err != nil { + continue + } + var i int + fmt.Sscanf(qk, "l_%d", &i) + scores[i].P = p + } + } + sort.SliceStable(scores, func(i, j int) bool { return scores[i].P > scores[j].P }) + return scores, usage, requests, nil +} + +// RankPages asks, for every page, whether it contains the facts or +// figures the question asks for. Pages are batched so one request stays +// under the budget; each page's text is its own question state, so the +// per-question ceiling is a page, not the sum. +func (n *JudgeNavigator) RankPages(ctx context.Context, query string, pages []NavPage) ([]PageScore, Usage, int, error) { + var usage Usage + if n.Judge == nil { + return nil, usage, 0, fmt.Errorf("judgewalk: no Judge configured") + } + scores := make([]PageScore, len(pages)) + for i, p := range pages { + scores[i] = PageScore{Page: p} + } + requests := 0 + budget := n.reqTokens() + limit := n.pageChars() + for start := 0; start < len(pages); { + state := map[string]any{"question": query} + questions := map[string]llmgate.Question{} + used := len(query) / 4 + end := start + for end < len(pages) { + text := pages[end].Text + if len(text) > limit { + text = text[:limit] + } + cost := len(text)/4 + 60 + if end > start && used+cost > budget { + break + } + qk := fmt.Sprintf("p_%d", end) + state[qk] = map[string]any{"page": pages[end].Number, "text": text} + questions[qk] = llmgate.Noul{ + Instructions: fmt.Sprintf( + "`question` is a question about a document. `%s.text` is the text of page "+ + "`%s.page`. Does this page contain the specific facts or figures needed to "+ + "answer the question — the number, the statement, the table row? A page that "+ + "only mentions the topic, or refers the reader elsewhere, does not.", qk, qk), + Criteria: &llmgate.NoulCriteria{ + True: "The answer, or a figure it is computed from, is on this page", + False: "The page is about something else, or only mentions the topic", + }, + } + used += cost + end++ + } + res, err := n.Judge.Judge(ctx, llmgate.JudgeRequest{State: state, Questions: questions}) + if err != nil { + return nil, usage, requests, err + } + requests++ + usage.Add(judgeUsage(res)) + for qk := range questions { + p, err := res.Noul(qk) + if err != nil { + continue + } + var i int + fmt.Sscanf(qk, "p_%d", &i) + scores[i].P = p + } + start = end + } + sort.SliceStable(scores, func(i, j int) bool { return scores[i].P > scores[j].P }) + return scores, usage, requests, nil +} + +// Navigate ranks the leaves, reads the pages of the best few, ranks +// those pages, and returns the evidence set. loadPages returns the +// pages (or body chunks) of one leaf. +func (n *JudgeNavigator) Navigate(ctx context.Context, query string, leaves []NavLeaf, loadPages func(ctx context.Context, leaf NavLeaf) ([]NavPage, error)) (*NavResult, error) { + out := &NavResult{} + if len(leaves) == 0 { + return out, nil + } + ranked, u, r, err := n.RankLeaves(ctx, query, leaves) + if err != nil { + return nil, fmt.Errorf("judgewalk: rank leaves: %w", err) + } + out.Usage.Add(u) + out.Requests += r + out.Leaves = ranked + + // The best leaves, in rank order, until the page budget is spent. + // A leaf below threshold is still read when nothing better exists: + // a low-confidence best guess beats reading nothing. + var pages []NavPage + maxP := n.maxPages() + for _, ls := range ranked { + if len(out.Selected) >= n.maxLeaves() { + break + } + ps, err := loadPages(ctx, ls.Leaf) + if err != nil { + return nil, fmt.Errorf("judgewalk: load %q: %w", ls.Leaf.Title, err) + } + if len(ps) == 0 { + continue + } + for i := range ps { + ps[i].LeafID = ls.Leaf.ID + } + if len(pages)+len(ps) > maxP { + ps = ps[:maxP-len(pages)] + } + pages = append(pages, ps...) + out.Selected = append(out.Selected, ls.Leaf) + if len(pages) >= maxP { + break + } + } + if len(pages) == 0 { + return out, nil + } + scored, u, r, err := n.RankPages(ctx, query, pages) + if err != nil { + return nil, fmt.Errorf("judgewalk: rank pages: %w", err) + } + out.Usage.Add(u) + out.Requests += r + out.Pages = scored + th := n.threshold() + for _, ps := range scored { + if ps.P >= th { + out.Evidence = append(out.Evidence, ps) + } + } + // Never come back empty-handed: the best pages are the evidence, + // flagged by their probability. + if len(out.Evidence) < navMinEvidencePages { + out.Evidence = scored[:min(navMinEvidencePages, len(scored))] + } + return out, nil +} + +func judgeUsage(res *llmgate.Judgment) Usage { + if res == nil { + return Usage{} + } + return Usage{ + InputTokens: res.Usage.InputTokens, + OutputTokens: res.Usage.OutputTokens, + TotalTokens: res.Usage.TotalTokens, + CostUSD: res.Usage.CostUSD, + LLMCalls: 1, + } +} + +// JudgeWalkStrategy is JudgeNavigator as a Strategy over a section +// tree. Leaves are the sections that carry a page range; a section's +// body, loaded through PageLoader, is chunked into page-sized units for +// the page ranking. The result names the sections the evidence came +// from and their page ranges. It does not write an answer: answering +// is the one generative step, and it belongs to the caller, over the +// evidence only. +type JudgeWalkStrategy struct { + Navigator JudgeNavigator + PageLoader PageContentLoader +} + +const strategyNameJudgeWalk = "judgewalk" + +// NewJudgeWalkStrategy builds the strategy on a Judge. +func NewJudgeWalkStrategy(j llmgate.Judge) *JudgeWalkStrategy { + return &JudgeWalkStrategy{Navigator: JudgeNavigator{Judge: j}} +} + +func (s *JudgeWalkStrategy) Name() string { return strategyNameJudgeWalk } + +// Select returns the section IDs the evidence pages came from. +func (s *JudgeWalkStrategy) Select(ctx context.Context, t *tree.Tree, query string, budget ContextBudget) ([]tree.SectionID, error) { + res, err := s.SelectWithCost(ctx, t, query, budget) + if err != nil { + return nil, err + } + return res.SelectedIDs, nil +} + +// SelectWithCost runs navigation and reports usage. +func (s *JudgeWalkStrategy) SelectWithCost(ctx context.Context, t *tree.Tree, query string, _ ContextBudget) (*Result, error) { + if t == nil || t.Root == nil { + return &Result{}, nil + } + sections := flattenSectionsByPage(t) + byID := map[string]sectionPageEntry{} + paths := sectionPaths(t) + var leaves []NavLeaf + for _, sec := range sections { + byID[string(sec.id)] = sec + leaves = append(leaves, NavLeaf{ + ID: string(sec.id), Title: sec.title, Path: paths[sec.id], + Start: sec.start, End: sec.end, Summary: sec.summary, + }) + } + load := func(ctx context.Context, leaf NavLeaf) ([]NavPage, error) { + sec := byID[leaf.ID] + if s.PageLoader == nil || sec.contentRef == "" { + return nil, nil + } + b, err := s.PageLoader.Load(ctx, sec.contentRef) + if err != nil { + return nil, err + } + return chunkBody(string(b), sec.start, sec.end, s.Navigator.pageChars()), nil + } + nav, err := s.Navigator.Navigate(ctx, query, leaves, load) + if err != nil { + return nil, err + } + var ids []tree.SectionID + conf := map[tree.SectionID]float64{} + var ranges []pageRange + seen := map[string]bool{} + for _, ev := range nav.Evidence { + id := tree.SectionID(ev.Page.LeafID) + if !seen[ev.Page.LeafID] { + seen[ev.Page.LeafID] = true + ids = append(ids, id) + sec := byID[ev.Page.LeafID] + ranges = append(ranges, pageRange{Start: sec.start, End: sec.end}) + } + if ev.P > conf[id] { + conf[id] = ev.P + } + } + best := 0.0 + if len(nav.Evidence) > 0 { + best = nav.Evidence[0].P + } + return &Result{ + SelectedIDs: ids, + Confidences: conf, + Confidence: best, + CitedPages: rangesToPairs(ranges), + ModelUsed: "judge", + Usage: nav.Usage, + HopsTaken: nav.Requests, + }, nil +} + +// sectionPaths maps each section to its "parent > child" title path. +func sectionPaths(t *tree.Tree) map[tree.SectionID]string { + out := map[tree.SectionID]string{} + var walk func(s *tree.Section, prefix string) + walk = func(s *tree.Section, prefix string) { + p := s.Title + if prefix != "" && s.Title != "" { + p = prefix + " > " + s.Title + } else if prefix != "" { + p = prefix + } + out[s.ID] = p + for i := range s.Children { + walk(s.Children[i], p) + } + } + if t != nil && t.Root != nil { + for i := range t.Root.Children { + walk(t.Root.Children[i], "") + } + } + return out +} + +// chunkBody splits a section body into page-sized units, numbered +// through the section's page range so a chunk can be cited by page. +func chunkBody(body string, start, end, size int) []NavPage { + body = strings.TrimSpace(body) + if body == "" { + return nil + } + if size <= 0 { + size = defaultNavPageChars + } + var chunks []string + for len(body) > size { + cut := strings.LastIndex(body[:size], "\n") + if cut < size/2 { + cut = size + } + chunks = append(chunks, body[:cut]) + body = strings.TrimSpace(body[cut:]) + } + if body != "" { + chunks = append(chunks, body) + } + span := end - start + 1 + if span < 1 { + span = 1 + } + out := make([]NavPage, len(chunks)) + for i, c := range chunks { + page := start + i*span/len(chunks) + if page > end { + page = end + } + out[i] = NavPage{Number: page, Text: c} + } + return out +} diff --git a/pkg/retrieval/judgewalk_test.go b/pkg/retrieval/judgewalk_test.go new file mode 100644 index 0000000..7dae00d --- /dev/null +++ b/pkg/retrieval/judgewalk_test.go @@ -0,0 +1,173 @@ +package retrieval + +import ( + "context" + "errors" + "strings" + "testing" + + "github.com/hallelx2/llmgate" + + "github.com/hallelx2/vectorless-engine/pkg/tree" +) + +// navJudge answers leaf questions by title keyword and page questions +// by text keyword, and counts requests. +func navJudge(leafHit, pageHit string) (*llmgate.MockJudge, *int) { + calls := 0 + j := &llmgate.MockJudge{Respond: func(_ context.Context, req llmgate.JudgeRequest) (*llmgate.Judgment, error) { + calls++ + st := req.State.(map[string]any) + ans := map[string]llmgate.Answer{} + for id := range req.Questions { + item := st[id].(map[string]any) + p := 0.1 + if strings.HasPrefix(id, "l_") && strings.Contains(strings.ToLower(item["title"].(string)), leafHit) { + p = 0.9 + } + if strings.HasPrefix(id, "p_") && strings.Contains(item["text"].(string), pageHit) { + p = 0.95 + } + ans[id] = llmgate.NoulAnswer{Noul: p} + } + return &llmgate.Judgment{Model: "mock", Answers: ans, Usage: llmgate.Usage{InputTokens: 10, TotalTokens: 10, TokensReported: true}}, nil + }} + return j, &calls +} + +func tenKLeaves() []NavLeaf { + return []NavLeaf{ + {ID: "1", Title: "Item 1. Business", Path: "PART I > Item 1. Business", Start: 3, End: 19}, + {ID: "2", Title: "Item 1A. Risk Factors", Start: 20, End: 33}, + {ID: "7", Title: "Item 7. Management's Discussion and Analysis", Start: 36, End: 49}, + {ID: "8", Title: "Item 8. Financial Statements and Supplementary Data", Start: 52, End: 91}, + {ID: "9", Title: "Item 9. Changes in and Disagreements with Accountants", Start: 92, End: 92}, + } +} + +func TestNavigateReadsTheBestLeafAndFindsTheEvidencePage(t *testing.T) { + j, calls := navJudge("financial statements", "Total revenue 17,606") + n := &JudgeNavigator{Judge: j, MaxLeaves: 1} + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + var ps []NavPage + for p := l.Start; p <= l.End && p < l.Start+6; p++ { + text := "Notes to the statements, page " + strings.Repeat("x", 20) + if p == l.Start+2 { + text = "CONSOLIDATED STATEMENTS OF INCOME\nTotal revenue 17,606 15,785" + } + ps = append(ps, NavPage{Number: p, Text: text}) + } + return ps, nil + } + res, err := n.Navigate(context.Background(), "What was Adobe's total revenue in FY2022?", tenKLeaves(), load) + if err != nil { + t.Fatal(err) + } + if len(res.Selected) != 1 || res.Selected[0].ID != "8" { + t.Fatalf("selected %+v, want Item 8", res.Selected) + } + if len(res.Evidence) == 0 || res.Evidence[0].Page.Number != 54 { + t.Fatalf("evidence %+v, want page 54 first", res.Evidence) + } + if res.Evidence[0].Page.LeafID != "8" { + t.Errorf("evidence page should carry its leaf: %+v", res.Evidence[0].Page) + } + if *calls != 2 || res.Requests != 2 { + t.Errorf("requests: mock saw %d, result says %d; want 2 (leaves, pages)", *calls, res.Requests) + } +} + +func TestNavigateNeverReturnsEmptyEvidence(t *testing.T) { + j, _ := navJudge("nothing matches", "nothing matches") + n := &JudgeNavigator{Judge: j, MaxLeaves: 2} + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + return []NavPage{{Number: l.Start, Text: "prose"}, {Number: l.Start + 1, Text: "more prose"}}, nil + } + res, err := n.Navigate(context.Background(), "q", tenKLeaves(), load) + if err != nil { + t.Fatal(err) + } + if len(res.Selected) != 2 { + t.Errorf("below-threshold leaves should still be read up to MaxLeaves: %d", len(res.Selected)) + } + if len(res.Evidence) != navMinEvidencePages { + t.Errorf("want the best %d pages as low-confidence evidence, got %d", navMinEvidencePages, len(res.Evidence)) + } + if res.Evidence[0].P >= n.threshold() { + t.Errorf("fallback evidence should carry its real (low) probability, got %v", res.Evidence[0].P) + } +} + +func TestRankPagesBatchesUnderTheRequestBudget(t *testing.T) { + j, calls := navJudge("", "needle") + n := &JudgeNavigator{Judge: j, RequestBudgetTokens: 1000, PageChars: 2000} + var pages []NavPage + for i := 0; i < 10; i++ { + text := strings.Repeat("y", 1900) + if i == 7 { + text = "needle " + text + } + pages = append(pages, NavPage{Number: i + 1, Text: text}) + } + scored, _, reqs, err := n.RankPages(context.Background(), "q", pages) + if err != nil { + t.Fatal(err) + } + if reqs < 5 || *calls != reqs { + t.Errorf("10 pages of ~500 tokens under a 1000-token budget should take ≥5 requests, took %d (mock saw %d)", reqs, *calls) + } + if scored[0].Page.Number != 8 { + t.Errorf("best page should be the one with the needle, got %d", scored[0].Page.Number) + } +} + +func TestNavigateSurfacesAJudgeFailure(t *testing.T) { + j := &llmgate.MockJudge{Err: errors.New("typesafe: request failed")} + n := &JudgeNavigator{Judge: j} + _, err := n.Navigate(context.Background(), "q", tenKLeaves(), func(context.Context, NavLeaf) ([]NavPage, error) { return nil, nil }) + if err == nil || !strings.Contains(err.Error(), "request failed") { + t.Errorf("a failed Judge must be an error, not an empty result: %v", err) + } +} + +type mapLoader map[string]string + +func (m mapLoader) Load(_ context.Context, ref string) ([]byte, error) { return []byte(m[ref]), nil } + +func TestJudgeWalkStrategyOverATree(t *testing.T) { + j, _ := navJudge("financial", "Total revenue") + s := NewJudgeWalkStrategy(j) + s.Navigator.MaxLeaves = 1 + s.PageLoader = mapLoader{"ref8": "CONSOLIDATED STATEMENTS OF INCOME\nTotal revenue 17,606", "ref1": "We are a company."} + tr := &tree.Tree{DocumentID: "d", Root: &tree.Section{ID: "root", Children: []*tree.Section{ + {ID: "s1", Title: "Item 1. Business", PageStart: 3, PageEnd: 19, ContentRef: "ref1"}, + {ID: "s8", Title: "Item 8. Financial Statements", PageStart: 52, PageEnd: 91, ContentRef: "ref8"}, + }}} + res, err := s.SelectWithCost(context.Background(), tr, "total revenue?", ContextBudget{}) + if err != nil { + t.Fatal(err) + } + if len(res.SelectedIDs) != 1 || res.SelectedIDs[0] != "s8" { + t.Fatalf("selected %v want [s8]", res.SelectedIDs) + } + if len(res.CitedPages) != 1 || res.CitedPages[0] != [2]int{52, 91} { + t.Errorf("cited pages %v", res.CitedPages) + } + if res.Usage.LLMCalls != 2 || res.HopsTaken != 2 { + t.Errorf("usage %+v hops %d", res.Usage, res.HopsTaken) + } + if s.Name() != "judgewalk" { + t.Errorf("name %q", s.Name()) + } +} + +func TestChunkBodyNumbersChunksThroughThePageRange(t *testing.T) { + body := strings.Repeat("line of text\n", 1000) // 13k chars + chunks := chunkBody(body, 10, 19, 6000) + if len(chunks) != 3 { + t.Fatalf("chunks: %d", len(chunks)) + } + if chunks[0].Number != 10 || chunks[2].Number > 19 || chunks[1].Number < 10 { + t.Errorf("chunk pages: %d %d %d", chunks[0].Number, chunks[1].Number, chunks[2].Number) + } +} From 337375c6153150a27235bbe1a9b273cc40429fa1 Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 18:40:45 +0100 Subject: [PATCH 2/4] feat(retrieval): judgewalk reads page heads before full pages, follows cross-references, and stops prejudging risk sections MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A 10-K's Item 8 is 70 pages; the 40-page budget cut it off. A coarse pass over page heads (700 chars, ~150 tokens each) now picks which of up to 120 gathered pages deserve their full text. A page two overlapping leaves both cover is gathered once. Item 3 of a 10-K is one sentence that says "see Note 21". When an evidence page refers to a leaf that was not read, the navigator reads it — code finds the reference, the tree names the leaf, the Judge judges its pages — for one more request. The leaf instruction said a risk-factors section "holds neither" the figures nor the discussion; three Boeing questions answered in Risk Factors ranked it last. The instruction now says what each kind of section holds without ruling any out. --- pkg/retrieval/judgewalk.go | 269 ++++++++++++++++++++++++++------ pkg/retrieval/judgewalk_test.go | 118 +++++++++++++- 2 files changed, 333 insertions(+), 54 deletions(-) diff --git a/pkg/retrieval/judgewalk.go b/pkg/retrieval/judgewalk.go index c25bd88..56acb8f 100644 --- a/pkg/retrieval/judgewalk.go +++ b/pkg/retrieval/judgewalk.go @@ -6,7 +6,10 @@ import ( "sort" "strings" + "regexp" + "github.com/hallelx2/llmgate" + "github.com/hallelx2/llmgate/judge/typesafe" "github.com/hallelx2/vectorless-engine/pkg/tree" ) @@ -56,8 +59,10 @@ type PageScore struct { type NavResult struct { Leaves []LeafScore // every leaf, best first Selected []NavLeaf // the leaves whose pages were read - Pages []PageScore // every page read, best first + Coarse []PageScore // the coarse pass over page heads, when it ran; best first + Pages []PageScore // every page read in full, best first Evidence []PageScore // pages above threshold, best first; never empty when any page was read + Followed []NavLeaf // leaves read because an evidence page referred to them Usage Usage Requests int } @@ -70,28 +75,51 @@ type JudgeNavigator struct { // evidence. Zero selects 0.5. Threshold float64 - // MaxLeaves bounds how many sections' pages are read. Zero selects 3. + // MaxLeaves bounds how many sections' pages are gathered. Zero + // selects 5. MaxLeaves int - // MaxPages bounds how many pages are judged in total. Zero selects 40. + // MaxPages bounds how many pages are judged in full. Zero selects 40. MaxPages int + // CoarsePages bounds how many gathered pages the coarse pass may + // rank by their heads before the best MaxPages are read in full. + // Zero selects 120. A 10-K's Item 8 alone is 70 pages; without the + // coarse pass the page budget cut it off at 40 and the statement on + // page 113 was never read. + CoarsePages int + + // HeadChars is how much of a page the coarse pass sees. Zero selects + // 700 — the heading and the first rows of a table. + HeadChars int + + // FollowReferences, when true (the default), takes one more hop when + // the evidence pages point elsewhere — "see Note 21" — and the leaf + // they point to has not been read. A 10-K's Item 3 is one sentence + // that refers to the note where the legal proceedings actually are. + NoFollowReferences bool + // PageChars truncates each page's text before the Judge sees it. // Zero selects 6000 — a dense filing page is ~4–5k characters. PageChars int - // RequestBudgetTokens bounds one page-ranking request. Zero selects - // 40k, under the provider's 64k request ceiling with room for the - // shared query. + // RequestBudgetTokens bounds the STATE of one page-ranking request. + // The provider treats the whole state object as shared context for + // every question — there is no per-question state — so state plus + // the longest question must stay under 32k tokens. Zero selects + // 24k, the same ceiling the TOC stage uses; that is six to eight + // dense filing pages per request. RequestBudgetTokens int } const ( defaultNavThreshold = 0.5 - defaultNavMaxLeaves = 3 + defaultNavMaxLeaves = 5 defaultNavMaxPages = 40 + defaultNavCoarse = 120 + defaultNavHeadChars = 700 defaultNavPageChars = 6000 - defaultNavReqTokens = 40_000 + defaultNavReqTokens = 24_000 navLeafBatch = 120 navMinEvidencePages = 2 navLeafStateMaxChars = 300 @@ -118,6 +146,20 @@ func (n *JudgeNavigator) maxPages() int { return defaultNavMaxPages } +func (n *JudgeNavigator) coarsePages() int { + if n.CoarsePages > 0 { + return n.CoarsePages + } + return defaultNavCoarse +} + +func (n *JudgeNavigator) headChars() int { + if n.HeadChars > 0 { + return n.HeadChars + } + return defaultNavHeadChars +} + func (n *JudgeNavigator) pageChars() int { if n.PageChars > 0 { return n.PageChars @@ -171,9 +213,10 @@ func (n *JudgeNavigator) RankLeaves(ctx context.Context, query string, leaves [] "`question` is a question about a document. `%s` is one section of that "+ "document's table of contents: its title, where it sits, and its page range. "+ "Would the information needed to answer the question be found in this "+ - "section? Judge from what such a section of such a document contains — a "+ - "financial statement holds the figures, an MD&A holds the discussion, a "+ - "risk-factors section holds neither.", qk), + "section? Judge from what such a section of such a document contains: "+ + "figures live in the statements and their notes, discussion of results in "+ + "the MD&A, customers, competition and outlook in the business and risk "+ + "sections.", qk), Criteria: &llmgate.NoulCriteria{ True: "This section is where a reader would look for the answer", False: "The answer would not be in this section", @@ -201,10 +244,16 @@ func (n *JudgeNavigator) RankLeaves(ctx context.Context, query string, leaves [] } // RankPages asks, for every page, whether it contains the facts or -// figures the question asks for. Pages are batched so one request stays -// under the budget; each page's text is its own question state, so the -// per-question ceiling is a page, not the sum. +// figures the question asks for. Pages are batched so the request's +// whole state — query plus every page in the batch — stays under the +// budget, counted with the provider's tokenizer. func (n *JudgeNavigator) RankPages(ctx context.Context, query string, pages []NavPage) ([]PageScore, Usage, int, error) { + return n.rankPages(ctx, query, pages, n.pageChars(), false) +} + +// rankPages is RankPages with the text window chosen by the caller: +// the full page, or just its head for the coarse pass. +func (n *JudgeNavigator) rankPages(ctx context.Context, query string, pages []NavPage, limit int, coarse bool) ([]PageScore, Usage, int, error) { var usage Usage if n.Judge == nil { return nil, usage, 0, fmt.Errorf("judgewalk: no Judge configured") @@ -215,33 +264,45 @@ func (n *JudgeNavigator) RankPages(ctx context.Context, query string, pages []Na } requests := 0 budget := n.reqTokens() - limit := n.pageChars() for start := 0; start < len(pages); { state := map[string]any{"question": query} questions := map[string]llmgate.Question{} - used := len(query) / 4 + used := countTokens(query) + 100 end := start for end < len(pages) { text := pages[end].Text if len(text) > limit { text = text[:limit] } - cost := len(text)/4 + 60 + cost := countTokens(text) + 20 if end > start && used+cost > budget { break } qk := fmt.Sprintf("p_%d", end) state[qk] = map[string]any{"page": pages[end].Number, "text": text} - questions[qk] = llmgate.Noul{ - Instructions: fmt.Sprintf( - "`question` is a question about a document. `%s.text` is the text of page "+ - "`%s.page`. Does this page contain the specific facts or figures needed to "+ - "answer the question — the number, the statement, the table row? A page that "+ - "only mentions the topic, or refers the reader elsewhere, does not.", qk, qk), - Criteria: &llmgate.NoulCriteria{ - True: "The answer, or a figure it is computed from, is on this page", - False: "The page is about something else, or only mentions the topic", - }, + if coarse { + questions[qk] = llmgate.Noul{ + Instructions: fmt.Sprintf( + "`question` is a question about a document. `%s.text` is the START of page "+ + "`%s.page` — its heading and first lines. Could the facts or figures needed "+ + "to answer the question be on this page, judging from what it opens with?", qk, qk), + Criteria: &llmgate.NoulCriteria{ + True: "This page's opening says it is the kind of page that would hold the answer", + False: "This page opens on something unrelated", + }, + } + } else { + questions[qk] = llmgate.Noul{ + Instructions: fmt.Sprintf( + "`question` is a question about a document. `%s.text` is the text of page "+ + "`%s.page`. Does this page contain the specific facts or figures needed to "+ + "answer the question — the number, the statement, the table row? A page that "+ + "only mentions the topic, or refers the reader elsewhere, does not.", qk, qk), + Criteria: &llmgate.NoulCriteria{ + True: "The answer, or a figure it is computed from, is on this page", + False: "The page is about something else, or only mentions the topic", + }, + } } used += cost end++ @@ -283,37 +344,57 @@ func (n *JudgeNavigator) Navigate(ctx context.Context, query string, leaves []Na out.Requests += r out.Leaves = ranked - // The best leaves, in rank order, until the page budget is spent. - // A leaf below threshold is still read when nothing better exists: - // a low-confidence best guess beats reading nothing. + // Gather the pages of the best leaves, in rank order, up to the + // coarse budget. A leaf below threshold is still read when nothing + // better exists: a low-confidence best guess beats reading nothing. + // A page two overlapping leaves both cover is gathered once. var pages []NavPage - maxP := n.maxPages() + seen := map[int]bool{} + coarseCap := n.coarsePages() for _, ls := range ranked { - if len(out.Selected) >= n.maxLeaves() { + if len(out.Selected) >= n.maxLeaves() || len(pages) >= coarseCap { break } ps, err := loadPages(ctx, ls.Leaf) if err != nil { return nil, fmt.Errorf("judgewalk: load %q: %w", ls.Leaf.Title, err) } - if len(ps) == 0 { - continue - } - for i := range ps { - ps[i].LeafID = ls.Leaf.ID - } - if len(pages)+len(ps) > maxP { - ps = ps[:maxP-len(pages)] + added := 0 + for _, p := range ps { + if seen[p.Number] || len(pages) >= coarseCap { + continue + } + seen[p.Number] = true + p.LeafID = ls.Leaf.ID + pages = append(pages, p) + added++ } - pages = append(pages, ps...) - out.Selected = append(out.Selected, ls.Leaf) - if len(pages) >= maxP { - break + if added > 0 { + out.Selected = append(out.Selected, ls.Leaf) } } if len(pages) == 0 { return out, nil } + + // More pages than the full-text budget: a coarse pass over page + // heads picks which ones deserve their whole text. Heads are + // ~150 tokens, so 120 of them is one request. + maxP := n.maxPages() + if len(pages) > maxP { + heads, u, r, err := n.rankPages(ctx, query, pages, n.headChars(), true) + if err != nil { + return nil, fmt.Errorf("judgewalk: rank page heads: %w", err) + } + out.Usage.Add(u) + out.Requests += r + out.Coarse = heads + pages = pages[:0] + for _, h := range heads[:maxP] { + pages = append(pages, h.Page) + } + sort.Slice(pages, func(i, j int) bool { return pages[i].Number < pages[j].Number }) + } scored, u, r, err := n.RankPages(ctx, query, pages) if err != nil { return nil, fmt.Errorf("judgewalk: rank pages: %w", err) @@ -322,19 +403,105 @@ func (n *JudgeNavigator) Navigate(ctx context.Context, query string, leaves []Na out.Requests += r out.Pages = scored th := n.threshold() - for _, ps := range scored { - if ps.P >= th { - out.Evidence = append(out.Evidence, ps) + fill := func(scored []PageScore) { + out.Evidence = out.Evidence[:0] + for _, ps := range scored { + if ps.P >= th { + out.Evidence = append(out.Evidence, ps) + } + } + // Never come back empty-handed: the best pages are the evidence, + // flagged by their probability. + if len(out.Evidence) < navMinEvidencePages { + out.Evidence = append(out.Evidence[:0], scored[:min(navMinEvidencePages, len(scored))]...) } } - // Never come back empty-handed: the best pages are the evidence, - // flagged by their probability. - if len(out.Evidence) < navMinEvidencePages { - out.Evidence = scored[:min(navMinEvidencePages, len(scored))] + fill(scored) + + // One more hop when the evidence points elsewhere. Code reads the + // reference, the tree names the leaf, the Judge reads its pages. + if !n.NoFollowReferences { + var refPages []NavPage + readLeaf := map[string]bool{} + for _, l := range out.Selected { + readLeaf[l.ID] = true + } + for _, ev := range out.Evidence { + for _, target := range referencedLeaves(ev.Page.Text, leaves) { + if readLeaf[target.ID] { + continue + } + readLeaf[target.ID] = true + ps, err := loadPages(ctx, target) + if err != nil { + return nil, fmt.Errorf("judgewalk: follow %q: %w", target.Title, err) + } + for _, p := range ps { + if !seen[p.Number] && len(refPages) < maxP { + seen[p.Number] = true + p.LeafID = target.ID + refPages = append(refPages, p) + } + } + out.Followed = append(out.Followed, target) + } + } + if len(refPages) > 0 { + more, u, r, err := n.RankPages(ctx, query, refPages) + if err != nil { + return nil, fmt.Errorf("judgewalk: rank referenced pages: %w", err) + } + out.Usage.Add(u) + out.Requests += r + out.Pages = append(out.Pages, more...) + sort.SliceStable(out.Pages, func(i, j int) bool { return out.Pages[i].P > out.Pages[j].P }) + fill(out.Pages) + } } return out, nil } +// countTokens uses the provider's tokenizer, so the batch budget is +// measured the way the request will be. Dense financial tables run +// near one token per two characters; a bytes/4 guess overflowed. +func countTokens(text string) int { + if n, err := typesafe.EstimateTokens(text); err == nil { + return n + } + return len(text)/3 + 1 +} + +// reReference finds the cross-references a page makes: "see Note 21", +// "Item 1A", "Note 14 to the Consolidated Financial Statements". +var reReference = regexp.MustCompile(`(?i)\b(note|item|part|section|schedule)\s+(\d+[a-c]?)\b`) + +// referencedLeaves returns the leaves a page's cross-references name, +// matched by label and number at the start of the leaf's title. +func referencedLeaves(text string, leaves []NavLeaf) []NavLeaf { + var out []NavLeaf + seen := map[string]bool{} + for _, m := range reReference.FindAllStringSubmatch(text, -1) { + label, num := strings.ToLower(m[1]), strings.ToLower(m[2]) + key := label + " " + num + if seen[key] { + continue + } + seen[key] = true + for _, l := range leaves { + t := strings.ToLower(strings.TrimSpace(l.Title)) + // "Note 21. Legal Proceedings", "NOTE 21 — …", "Item 1A." + if strings.HasPrefix(t, key) { + rest := t[len(key):] + if rest == "" || !(rest[0] >= '0' && rest[0] <= '9') && !(rest[0] >= 'a' && rest[0] <= 'z') { + out = append(out, l) + break + } + } + } + } + return out +} + func judgeUsage(res *llmgate.Judgment) Usage { if res == nil { return Usage{} diff --git a/pkg/retrieval/judgewalk_test.go b/pkg/retrieval/judgewalk_test.go index 7dae00d..8999cbd 100644 --- a/pkg/retrieval/judgewalk_test.go +++ b/pkg/retrieval/judgewalk_test.go @@ -75,6 +75,64 @@ func TestNavigateReadsTheBestLeafAndFindsTheEvidencePage(t *testing.T) { if *calls != 2 || res.Requests != 2 { t.Errorf("requests: mock saw %d, result says %d; want 2 (leaves, pages)", *calls, res.Requests) } + if res.Coarse != nil { + t.Errorf("six pages under a 40-page budget need no coarse pass") + } +} + +// A 70-page section: the coarse pass over page heads picks the pages +// worth reading in full, and the page deep in the section is found. +func TestNavigateCoarsePassReachesDeepIntoABigLeaf(t *testing.T) { + j, calls := navJudge("financial statements", "Total revenue 17,606") + n := &JudgeNavigator{Judge: j, MaxLeaves: 1, MaxPages: 10, CoarsePages: 120} + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + var ps []NavPage + for p := l.Start; p <= l.End; p++ { + text := "Notes to the statements " + strings.Repeat("x", 3000) + if p == 85 { + text = "CONSOLIDATED STATEMENTS OF INCOME\nTotal revenue 17,606 15,785" + strings.Repeat("y", 3000) + } + ps = append(ps, NavPage{Number: p, Text: text}) + } + return ps, nil + } + leaves := []NavLeaf{{ID: "8", Title: "Item 8. Financial Statements", Start: 52, End: 125}} + res, err := n.Navigate(context.Background(), "total revenue?", leaves, load) + if err != nil { + t.Fatal(err) + } + if len(res.Coarse) != 74 { + t.Errorf("coarse pass should have seen all 74 page heads, saw %d", len(res.Coarse)) + } + if len(res.Pages) != 10 { + t.Errorf("full pass should read MaxPages=10, read %d", len(res.Pages)) + } + if len(res.Evidence) == 0 || res.Evidence[0].Page.Number != 85 { + t.Fatalf("page 85 not found: %+v", res.Evidence) + } + if *calls < 3 { + t.Errorf("want leaves + coarse + full requests, got %d", *calls) + } +} + +func TestNavigateGathersASharedPageOnce(t *testing.T) { + j, _ := navJudge("item", "needle") + n := &JudgeNavigator{Judge: j, MaxLeaves: 2} + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + var ps []NavPage + for p := l.Start; p <= l.End; p++ { + ps = append(ps, NavPage{Number: p, Text: "needle on page"}) + } + return ps, nil + } + leaves := []NavLeaf{{ID: "a", Title: "Item 9A", Start: 60, End: 61}, {ID: "b", Title: "Item 9B", Start: 61, End: 62}} + res, err := n.Navigate(context.Background(), "q", leaves, load) + if err != nil { + t.Fatal(err) + } + if len(res.Pages) != 3 { + t.Errorf("pages 60,61,62 should be read once each, read %d", len(res.Pages)) + } } func TestNavigateNeverReturnsEmptyEvidence(t *testing.T) { @@ -100,7 +158,7 @@ func TestNavigateNeverReturnsEmptyEvidence(t *testing.T) { func TestRankPagesBatchesUnderTheRequestBudget(t *testing.T) { j, calls := navJudge("", "needle") - n := &JudgeNavigator{Judge: j, RequestBudgetTokens: 1000, PageChars: 2000} + n := &JudgeNavigator{Judge: j, RequestBudgetTokens: 1200, PageChars: 2000} var pages []NavPage for i := 0; i < 10; i++ { text := strings.Repeat("y", 1900) @@ -113,8 +171,10 @@ func TestRankPagesBatchesUnderTheRequestBudget(t *testing.T) { if err != nil { t.Fatal(err) } - if reqs < 5 || *calls != reqs { - t.Errorf("10 pages of ~500 tokens under a 1000-token budget should take ≥5 requests, took %d (mock saw %d)", reqs, *calls) + // 1900 chars of "y" is ~475 tokens; two fit under 1200 with the + // query, a third does not. + if reqs < 4 || *calls != reqs { + t.Errorf("10 pages of ~475 tokens under a 1200-token budget should take ≥4 requests, took %d (mock saw %d)", reqs, *calls) } if scored[0].Page.Number != 8 { t.Errorf("best page should be the one with the needle, got %d", scored[0].Page.Number) @@ -171,3 +231,55 @@ func TestChunkBodyNumbersChunksThroughThePageRange(t *testing.T) { t.Errorf("chunk pages: %d %d %d", chunks[0].Number, chunks[1].Number, chunks[2].Number) } } + +// Item 3 says "see Note 21"; the answer is in Note 21, which the leaf +// ranking did not pick. The navigator follows the reference. +func TestNavigateFollowsACrossReference(t *testing.T) { + j, calls := navJudge("legal proceedings", "class action filed") + n := &JudgeNavigator{Judge: j, MaxLeaves: 1} + leaves := []NavLeaf{ + {ID: "3", Title: "Item 3. Legal Proceedings", Start: 20, End: 20}, + {ID: "n21", Title: "Note 21 – Legal Proceedings", Start: 113, End: 114}, + {ID: "n20", Title: "Note 20 – Leases", Start: 110, End: 112}, + } + load := func(_ context.Context, l NavLeaf) ([]NavPage, error) { + switch l.ID { + case "3": + return []NavPage{{Number: 20, Text: "Item 3. Legal Proceedings\nCurrently, we are involved in a number of legal proceedings. For a discussion of contingencies see Note 21 to our Consolidated Financial Statements."}}, nil + case "n21": + return []NavPage{{Number: 113, Text: "Note 21 – Legal Proceedings\nA class action filed in 2019 remains pending."}, {Number: 114, Text: "Other matters."}}, nil + } + return []NavPage{{Number: l.Start, Text: "leases"}}, nil + } + res, err := n.Navigate(context.Background(), "Has Boeing reported any materially important ongoing legal battles?", leaves, load) + if err != nil { + t.Fatal(err) + } + if len(res.Followed) != 1 || res.Followed[0].ID != "n21" { + t.Fatalf("followed %+v, want Note 21", res.Followed) + } + if len(res.Evidence) == 0 || res.Evidence[0].Page.Number != 113 { + t.Fatalf("evidence %+v, want page 113 first", res.Evidence) + } + if *calls != 3 { + t.Errorf("requests: %d, want 3 (leaves, Item 3 page, Note 21 pages)", *calls) + } + // Turned off, it stays on Item 3. + n.NoFollowReferences = true + res, _ = n.Navigate(context.Background(), "q", leaves, load) + if len(res.Followed) != 0 { + t.Errorf("NoFollowReferences ignored") + } +} + +func TestReferencedLeaves(t *testing.T) { + leaves := []NavLeaf{{ID: "a", Title: "Note 2. Revenue"}, {ID: "b", Title: "Note 21 – Legal Proceedings"}, {ID: "c", Title: "Item 1A. Risk Factors"}, {ID: "d", Title: "Item 1. Business"}} + got := referencedLeaves("see Note 21 and Item 1A; also note 2 above", leaves) + ids := []string{} + for _, l := range got { + ids = append(ids, l.ID) + } + if strings.Join(ids, ",") != "b,c,a" { + t.Errorf("got %v want [b c a] — and Note 2 must not match Note 21, Item 1 must not match Item 1A", ids) + } +} From 6e734d4aa8356e9387b377661897701dcb8a3517 Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 18:54:46 +0100 Subject: [PATCH 3/4] fix(parser): expose real per-page text; every page-level stage reads it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit HAL-1375. Page text was reconstructed from sections by keying each section's whole content under its first page. A section that starts on page 58 and runs into 59 put page 59's opening under "page 58"; a page that started no section had no entry at all. AMD_2022_10K had 104 page entries for a 121-page PDF. Detection, resolution, the coverage gate and judgewalk's page ranking all worked on that — the navigation bench showed it as evidence one page before the gold page. ParsedDoc.Pages is now built from the parser's own rows, grouped by the physical page they were read from, with the running-header and boilerplate filtering already applied. Ingest and every bench command take pages from it; the section assembly stays only as the fallback for formats without pages. AMD: 121 entries, contiguous. --- cmd/ingestbench/main.go | 2 +- cmd/jevbench/main.go | 2 +- cmd/navbench/main.go | 2 +- cmd/pagedump/main.go | 2 +- cmd/tocdump/main.go | 2 +- cmd/tocresolve/main.go | 2 +- pkg/ingest/bench_export.go | 4 +-- pkg/ingest/ingest.go | 22 +++++++++++++- pkg/ingest/toc_builder_test.go | 16 ++++++++++ pkg/parser/pages_test.go | 54 ++++++++++++++++++++++++++++++++++ pkg/parser/parser.go | 16 ++++++++++ pkg/parser/pdf.go | 26 ++++++++++++++++ 12 files changed, 141 insertions(+), 9 deletions(-) create mode 100644 pkg/parser/pages_test.go diff --git a/cmd/ingestbench/main.go b/cmd/ingestbench/main.go index d64aecd..42feac5 100644 --- a/cmd/ingestbench/main.go +++ b/cmd/ingestbench/main.go @@ -245,5 +245,5 @@ func readPages(path string) ([]ingest.PageText, error) { if err != nil { return nil, err } - return ingest.BenchAssemblePages(doc.Sections), nil + return ingest.BenchAssemblePages(doc), nil } diff --git a/cmd/jevbench/main.go b/cmd/jevbench/main.go index 5028620..9c920a5 100644 --- a/cmd/jevbench/main.go +++ b/cmd/jevbench/main.go @@ -136,7 +136,7 @@ func readPages(path string) ([]ingest.PageText, error) { if err != nil { return nil, err } - return ingest.BenchAssemblePages(doc.Sections), nil + return ingest.BenchAssemblePages(doc), nil } // dotEnv reads one key from a gitignored .env up-tree, so a credential diff --git a/cmd/navbench/main.go b/cmd/navbench/main.go index 51ba135..c6cee7c 100644 --- a/cmd/navbench/main.go +++ b/cmd/navbench/main.go @@ -179,7 +179,7 @@ func load(doc, trees, pdfs string, leafCache map[string][]retrieval.NavLeaf, pag if err != nil { return nil, nil, err } - pages := ingest.BenchAssemblePages(pd.Sections) + pages := ingest.BenchAssemblePages(pd) var leaves []retrieval.NavLeaf var walk func(ns []tree.TOCNode, path string) walk = func(ns []tree.TOCNode, path string) { diff --git a/cmd/pagedump/main.go b/cmd/pagedump/main.go index 09863b0..c8570b5 100644 --- a/cmd/pagedump/main.go +++ b/cmd/pagedump/main.go @@ -29,7 +29,7 @@ func main() { fmt.Fprintln(os.Stderr, "parse:", err) os.Exit(1) } - for _, p := range ingest.BenchAssemblePages(doc.Sections) { + for _, p := range ingest.BenchAssemblePages(doc) { fmt.Printf("\n===== PAGE %d (%d chars) =====\n%s", p.PageNumber, len(p.Text), p.Text) } } diff --git a/cmd/tocdump/main.go b/cmd/tocdump/main.go index 4344ae0..43e0d8e 100644 --- a/cmd/tocdump/main.go +++ b/cmd/tocdump/main.go @@ -200,7 +200,7 @@ func readPages(path string) ([]ingest.PageText, error) { if err != nil { return nil, err } - return ingest.BenchAssemblePages(doc.Sections), nil + return ingest.BenchAssemblePages(doc), nil } func dotEnv(key string) string { diff --git a/cmd/tocresolve/main.go b/cmd/tocresolve/main.go index 2b22a90..367d068 100644 --- a/cmd/tocresolve/main.go +++ b/cmd/tocresolve/main.go @@ -72,7 +72,7 @@ func main() { fmt.Fprintln(os.Stderr, "parse:", err) os.Exit(1) } - pages := ingest.BenchAssemblePages(doc.Sections) + pages := ingest.BenchAssemblePages(doc) key := os.Getenv(typesafe.EnvAPIKey) if key == "" { diff --git a/pkg/ingest/bench_export.go b/pkg/ingest/bench_export.go index 9d95b60..2ec6e2a 100644 --- a/pkg/ingest/bench_export.go +++ b/pkg/ingest/bench_export.go @@ -21,8 +21,8 @@ import ( // BenchAssemblePages turns parsed sections into the per-page text the // TOC builder consumes. -func BenchAssemblePages(secs []parser.Section) []PageText { - return assemblePagesFromSections(secs) +func BenchAssemblePages(doc *parser.ParsedDoc) []PageText { + return assemblePages(doc) } // BenchDetectTOC runs the Judge-backed detection phase alone. diff --git a/pkg/ingest/ingest.go b/pkg/ingest/ingest.go index 9559fb5..4d2b6e0 100644 --- a/pkg/ingest/ingest.go +++ b/pkg/ingest/ingest.go @@ -457,7 +457,7 @@ func (p *Pipeline) Run(ctx context.Context, pl Payload) error { // SQL NULL to documents.toc_tree (which is the column's default, // so this is also the no-op). func (p *Pipeline) runTOCBuilder(ctx context.Context, docID tree.DocumentID, parsed *parser.ParsedDoc, log *slog.Logger) error { - pages := assemblePagesFromSections(parsed.Sections) + pages := assemblePages(parsed) if len(pages) == 0 { log.Info("ingest: toc-builder skipped; no per-page text available") return nil @@ -519,6 +519,26 @@ func (p *Pipeline) runTOCBuilder(ctx context.Context, docID tree.DocumentID, par // // Sections with PageStart == 0 are skipped (the parser couldn't // place them) so the builder never sees ambiguous page numbers. +// assemblePages returns the document's text per page. The parser's own +// pages are the truth when it has them; the section-based assembly is +// the fallback for formats with no page notion — and it is only ever an +// approximation, since a section's content spans pages (HAL-1375). +func assemblePages(parsed *parser.ParsedDoc) []PageText { + if parsed == nil { + return nil + } + if len(parsed.Pages) > 0 { + out := make([]PageText, 0, len(parsed.Pages)) + for _, p := range parsed.Pages { + if p.Number > 0 && strings.TrimSpace(p.Text) != "" { + out = append(out, PageText{PageNumber: p.Number, Text: p.Text}) + } + } + return out + } + return assemblePagesFromSections(parsed.Sections) +} + func assemblePagesFromSections(secs []parser.Section) []PageText { pageText := map[int]*strings.Builder{} pages := []int{} diff --git a/pkg/ingest/toc_builder_test.go b/pkg/ingest/toc_builder_test.go index 68733b8..5a6212d 100644 --- a/pkg/ingest/toc_builder_test.go +++ b/pkg/ingest/toc_builder_test.go @@ -640,3 +640,19 @@ func TestBuildKeepsClaimedPagesWhenTheResolverIsDown(t *testing.T) { t.Errorf("degradation not recorded: %v", usage.Degraded) } } + +func TestAssemblePagesPrefersTheParsersPages(t *testing.T) { + doc := &parser.ParsedDoc{ + Sections: []parser.Section{{Title: "Item 1", Content: "page 2 text page 3 text", PageStart: 2, PageEnd: 3}}, + Pages: []parser.Page{{Number: 2, Text: "Item 1\npage 2 text"}, {Number: 3, Text: "page 3 text"}}, + } + got := assemblePages(doc) + if len(got) != 2 || got[1].PageNumber != 3 || got[1].Text != "page 3 text" { + t.Fatalf("pages should come from Pages, per page: %+v", got) + } + doc.Pages = nil + got = assemblePages(doc) + if len(got) != 1 || got[0].PageNumber != 2 { + t.Errorf("without Pages, the section fallback: %+v", got) + } +} diff --git a/pkg/parser/pages_test.go b/pkg/parser/pages_test.go new file mode 100644 index 0000000..9e32757 --- /dev/null +++ b/pkg/parser/pages_test.go @@ -0,0 +1,54 @@ +package parser + +import ( + "context" + "os" + "testing" +) + +// Pages come from rows grouped by physical page, never from sections. +func TestPagesFromRowsGroupsByPhysicalPage(t *testing.T) { + rows := []pdfRow{ + {page: 1, text: "Cover"}, + {page: 2, text: "Item 1. Business"}, + {page: 2, text: "We make things."}, + {page: 3, text: "and keep making them."}, // same section, next page + {page: 5, text: "Item 1A. Risk Factors"}, // page 4 is blank + } + got := pagesFromRows(rows) + if len(got) != 4 { + t.Fatalf("pages: %d want 4 (1,2,3,5)", len(got)) + } + if got[2].Number != 3 || got[2].Text != "and keep making them." { + t.Errorf("page 3 should hold only its own text: %+v", got[2]) + } + if got[3].Number != 5 { + t.Errorf("a blank page keeps its number for the next: %+v", got[3]) + } + if got[1].Text != "Item 1. Business\nWe make things." { + t.Errorf("page 2: %q", got[1].Text) + } +} + +func TestParsedPDFExposesPages(t *testing.T) { + f, err := os.Open("testdata/tables-example.pdf") + if err != nil { + t.Skip("fixture missing") + } + defer f.Close() + doc, err := NewPDF().Parse(context.Background(), f) + if err != nil { + t.Fatal(err) + } + if len(doc.Pages) == 0 { + t.Fatal("no pages on a parsed PDF") + } + for i := 1; i < len(doc.Pages); i++ { + if doc.Pages[i].Number <= doc.Pages[i-1].Number { + t.Errorf("pages out of order: %d after %d", doc.Pages[i].Number, doc.Pages[i-1].Number) + } + } + if doc.Pages[0].Number != 1 { + t.Errorf("first page is %d", doc.Pages[0].Number) + } +} diff --git a/pkg/parser/parser.go b/pkg/parser/parser.go index b84f8e1..92c42f1 100644 --- a/pkg/parser/parser.go +++ b/pkg/parser/parser.go @@ -34,6 +34,22 @@ type ParsedDoc struct { // Metadata holds whatever extra structural hints the parser recovered // (author, created date, page count, etc.). Metadata map[string]string + + // Pages is the document's text page by page, for parsers that have + // pages. It is the ground truth for anything that reasons per page — + // contents-page detection, page resolution, page ranking — and must + // never be reconstructed from Sections: a section's content spans + // pages, and keying it under the section's first page puts the next + // page's opening under the wrong number (HAL-1375). Empty for + // formats without a page notion. + Pages []Page +} + +// Page is one page's text, in reading order, with the parser's +// running-header and boilerplate filtering already applied. +type Page struct { + Number int + Text string } // Section is one node in the parsed outline. diff --git a/pkg/parser/pdf.go b/pkg/parser/pdf.go index 8149f93..2abab77 100644 --- a/pkg/parser/pdf.go +++ b/pkg/parser/pdf.go @@ -345,6 +345,7 @@ func (p *PDF) parseDoc(_ context.Context, buf []byte) (*ParsedDoc, error) { if outline := reader.Outline(); len(outline.Child) > 0 { if doc, ok := parsePDFWithOutline(outline, rows); ok { doc.Sections = capLeafSections(doc.Sections, p.resolvedMaxSections()) + doc.Pages = pagesFromRows(rows) attachTableSections(doc, tableSections) return doc, nil } @@ -543,11 +544,36 @@ func (p *PDF) parseDoc(_ context.Context, buf []byte) (*ParsedDoc, error) { out := &ParsedDoc{ Title: title, Sections: capLeafSections(chunkOversizedLeaves(rootSec.Children), p.resolvedMaxSections()), + Pages: pagesFromRows(rows), } attachTableSections(out, tableSections) return out, nil } +// pagesFromRows groups the filtered rows by page, in order. Rows carry +// the physical page index they were read from, so a page with no rows +// (blank, or image-only) simply has no entry — its number is still +// its own, never reassigned. +func pagesFromRows(rows []pdfRow) []Page { + var out []Page + var cur *Page + for _, r := range rows { + text := strings.TrimSpace(r.text) + if text == "" || r.page <= 0 { + continue + } + if cur == nil || cur.Number != r.page { + out = append(out, Page{Number: r.page}) + cur = &out[len(out)-1] + } + if cur.Text != "" { + cur.Text += "\n" + } + cur.Text += text + } + return out +} + // resolvedMaxSections turns the configured MaxSections into the value // the cap actually uses: 0 selects defaultMaxLeafSections; a negative // value disables the cap (returns a non-positive number capLeafSections From b5d94fea743a0a3e3416d1ddef48251a152b80c6 Mon Sep 17 00:00:00 2001 From: Halleluyah Oludele Date: Fri, 18 Sep 2026 19:15:07 +0100 Subject: [PATCH 4/4] docs: evaluation of retrieval navigation on a Judge and the per-page text fix 40 FinanceBench questions: 34/40 with every gold page in the evidence set, 37/40 in the right section, ~4 Jev requests and $0.003 each. The per-page fix also took the corpus to 21/21 trees and 47/47 gold pages. --- ...6-09-18-retrieval-navigation-on-a-judge.md | 105 ++++++++++++++++++ 1 file changed, 105 insertions(+) create mode 100644 docs/evaluations/2026-09-18-retrieval-navigation-on-a-judge.md diff --git a/docs/evaluations/2026-09-18-retrieval-navigation-on-a-judge.md b/docs/evaluations/2026-09-18-retrieval-navigation-on-a-judge.md new file mode 100644 index 0000000..d3d9be1 --- /dev/null +++ b/docs/evaluations/2026-09-18-retrieval-navigation-on-a-judge.md @@ -0,0 +1,105 @@ +# Retrieval navigation on a Judge — and the page bug it found + +**Date:** 2026-09-18 +**Harness:** [`cmd/navbench`](../../cmd/navbench/main.go) (evidence-page recall per question, no generative call), [`cmd/tocdump -judge-only`](../../cmd/tocdump/main.go), [`cmd/tocdump/coverage.py`](../../cmd/tocdump/coverage.py), [`cmd/tocdump/titles.py`](../../cmd/tocdump/titles.py) +**Corpus:** FinanceBench, 40 questions on the 21 downloaded 10-K filings, gold evidence pages from `PatronusAI/financebench` +**Issues:** HAL-1371 (navigation), HAL-1375 (per-page text), HAL-1374 (leaf granularity) +**Question:** `TreeWalkStrategy` navigates with up to eight chat-completion hops. Can a Judge navigate instead — rank the tree's leaves, then the pages — and land the gold evidence page, with no generative call? + +## Result: 34 of 40 questions have every gold page in the evidence set, 37 of 40 have it inside the chosen section, at about four Jev requests, 40 pages read and $0.003 per question. + +| run | change | questions | right section | every gold page found | pages read | requests | s / question | $ / question | +|---|---|---|---|---|---|---|---|---| +| 1 | leaves → full pages, 3 leaves, 40 pages | 38 | 32 (0.842) | 30 (0.789) | 35.9 | 2.9 | 8.9 | 0.0021 | +| 2 | + coarse pass over page heads, page dedup | 35 † | 29 (0.829) | 25 (0.714) | 36.0 | 3.7 | 22.5 | 0.0028 | +| 3 | + leaf prompt no longer rules out risk sections | 38 | 34 (0.895) | 29 (0.763) | 36.2 | 3.7 | 16.4 | 0.0028 | +| **5** | **+ real per-page text (HAL-1375), + follow cross-references** | **40** | **37 (0.925)** | **34 (0.850)** | 40.7 | 4.1 | 20.6 | 0.0029 | + +† three Jev requests failed after four minutes of retries in run 2; those questions are excluded from its counts and its seconds are inflated by the retries. Run 4 (follow only, old pages) was stopped when the page bug was found. + +Seconds per question are Jev API latency in the hour of the run — the +same 21 filings' whole TOC stage took 44 s each in one pass and 9.5 s in +the next — and are reported, not engineered around. + +## The bug the near-misses found + +Run 3's misses were one page off: gold 12 with evidence 11, gold 59 +with 58, gold 57 with 56. Page text was being assembled from +*sections* — each section's whole content keyed under the page it +started on — so a section running from 58 into 59 put page 59's opening +under "page 58", and a page that started no section had no entry at +all. AMD_2022_10K had 104 page entries for a 121-page PDF; PEPSICO's +549 pages showed as 170. + +Every page-level stage had been working on that: detection, resolution, +the coverage gate, and navigation. `ParsedDoc.Pages` is now built from +the parser's own rows, grouped by physical page. Re-running the whole +TOC stage on real pages: + +| | section-start pages (jev2) | real pages (jev3) | +|---|---|---| +| filings with a tree | 20 / 21 | **21 / 21** (GENERALMILLS, the "one-page parse" of HAL-1365, was this bug) | +| gold evidence pages inside a leaf | 45 / 45 | **47 / 47** | +| median leaf span | 37 p | 37 p | +| leaf title recall / precision against jev2 | — | 0.975 / 0.966 | +| TOC stage per filing | 44 s | 9.5 s (Jev latency in the hour) | + +## What each navigation change did + +- **Leaf ranking, then full pages** (run 1): the shape works. One Noul + per leaf, one per page of the best three leaves. 30 of 38. +- **Coarse pass over page heads** (run 2): meant to let a 70-page Item + 8 into the page budget. It did not move hits on its own and adds a + request; kept because run 5 needs it to gather 120 pages from five + leaves. +- **The leaf prompt** (run 3): the first version said "a risk-factors + section holds neither" the figures nor the discussion. Three Boeing + questions are answered in Risk Factors; the Judge ranked it last + because it was told to. Right-section rate 0.842 → 0.895 once the + instruction described what each kind of section holds without ruling + any out. A prompt is a measured artifact; see the prompt-versioning + proposal. +- **Real pages + follow** (run 5): the page fix turned the off-by-one + misses into hits (0.763 → 0.850) and brought the two GENERALMILLS + questions into the set. Following "see Note 21" fires only when the + tree has a leaf to follow to; on 10-K trees with 20-odd leaves it + rarely does, which is HAL-1374's point. + +## The six that are still missed + +| filing | gold | evidence | why | +|---|---|---|---| +| BOEING | 113 | 26, 27 | "see Note 21" — the tree has no note-level leaf to follow to (HAL-1374) | +| BOEING | 8, 10, 14 | 12, 14 | right section; two of three pages ranked below threshold | +| BOEING | 55 | 24, 77, 76, 79 | right section; page missed | +| KRAFTHEINZ | 50, 52 | 50, 124 | one of two gold pages | +| PFIZER | 59 | 57, 41 | right section; neighbouring page chosen | +| PFIZER | 70, 71 | 51, 9 | wrong section | + +Four of six are page-level misses inside the right section — the +full-pass question, or how much of a page it sees, is the next thing +to measure. The other two want finer leaves. + +## Method + +- `JudgeNavigator.RankLeaves`: one request per 120 leaves; state is + the question plus each leaf's title, path, page range and summary. +- Gather the pages of the best five leaves, up to 120, each page once. +- Over 40 pages: `rankPages` on 700-character heads, keep the best 40. +- `RankPages` on full pages (6,000 chars), batched so the request's + whole state stays under 24k tokens — the provider treats the state + object as shared context for every question. Evidence is every page + at or above 0.5; never fewer than the best two. +- `referencedLeaves`: `Note N` / `Item N` mentions on evidence pages + that name an unread leaf → its pages ranked in one more request. +- `JudgeWalkStrategy` is the same over a section tree, selectable as + `strategy=judgewalk` whenever `llm.judge` is configured. It writes + no answer: that is the caller's one generative call, over the + evidence only. + +## Reproduce + +```bash +go run ./cmd/tocdump -docs ~/.cache/vlbench/financebench -out /tmp/trees -judge-only -minimal +go run ./cmd/navbench -questions ~/.cache/vlbench/financebench-questions.jsonl -trees /tmp/trees -pdfs ~/.cache/vlbench/financebench -out /tmp/nav.jsonl +```