diff --git a/codewiki/src/be/agent_tools/generate_sub_module_documentations.py b/codewiki/src/be/agent_tools/generate_sub_module_documentations.py index 86693700..938a34cd 100644 --- a/codewiki/src/be/agent_tools/generate_sub_module_documentations.py +++ b/codewiki/src/be/agent_tools/generate_sub_module_documentations.py @@ -3,7 +3,7 @@ from pydantic_ai import RunContext, Tool, Agent from codewiki.src.be.agent_tools.deps import CodeWikiDeps -from codewiki.src.be.module_naming import normalize_sub_module_specs +from codewiki.src.be.module_naming import plan_sub_module_specs from codewiki.src.be.agent_tools.read_code_components import read_code_components_tool from codewiki.src.be.agent_tools.str_replace_editor import str_replace_editor_tool from codewiki.src.be.llm_services import create_fallback_models @@ -25,9 +25,10 @@ async def generate_sub_module_documentation( sub_module_specs: A dictionary mapping sub-module names to their core component IDs. Example: {"authentication": ["auth_handler.py::AuthHandler", "auth_middleware.py::verify_token"], "database": ["db_client.py::DBClient", "models.py::UserModel"]} Each key is a descriptive sub-module name, and the value is a list of component IDs from the current module's core components that belong to that sub-module. - Sub-module names must be unique across the whole wiki; a name that is already used by another - module is automatically prefixed with the current module name, and the tool result reports the - final file names actually saved. + Sub-module names must be unique across the whole wiki. A name already used by another module is + prefixed with the current module name; a name that is already documented (plain or prefixed) + is SKIPPED, not regenerated. The tool result reports the final file names actually saved and + every skipped request. """ deps = ctx.deps @@ -38,15 +39,21 @@ async def generate_sub_module_documentation( # Resolve name collisions against the module tree and files already on disk # before touching the tree (issue #76): docs live in one flat directory. - name_map = normalize_sub_module_specs( + # A request that is already documented (plain or parent-prefixed name) is + # skipped rather than renamed x_2, x_3, ... (issue #113). + plan = plan_sub_module_specs( sub_module_specs, previous_module_name, deps.module_tree, deps.absolute_docs_path, ) + name_map = plan.name_map + if not name_map: + return _skipped_report(plan.skipped, deps.current_module_name) final_specs = { name_map[requested_name]: core_component_ids for requested_name, core_component_ids in sub_module_specs.items() + if requested_name in name_map } # add the sub-module to the module tree @@ -135,9 +142,22 @@ async def generate_sub_module_documentation( if missing: report += f" MISSING (generation did not produce these files): {', '.join(missing)}." logger.warning("Sub-module documentation missing after generation: %s", ", ".join(missing)) + if plan.skipped: + report += " " + _skipped_report(plan.skipped, deps.current_module_name) return report +def _skipped_report(skipped: dict[str, str], current_module_name: str) -> str: + """Tell the parent agent, unambiguously, not to retry skipped sub-modules.""" + if not skipped: + return "No sub-modules were generated." + items = ", ".join(f"'{name}' ({reason})" for name, reason in skipped.items()) + return ( + f"Skipped sub-modules: {items}. Do NOT call generate_sub_module_documentation again " + f"for these; link the existing pages from `{current_module_name}.md` instead." + ) + + generate_sub_module_documentation_tool = Tool( function=generate_sub_module_documentation, name="generate_sub_module_documentation", diff --git a/codewiki/src/be/caw_toolkit.py b/codewiki/src/be/caw_toolkit.py index ba296f06..76f887c1 100644 --- a/codewiki/src/be/caw_toolkit.py +++ b/codewiki/src/be/caw_toolkit.py @@ -28,7 +28,7 @@ from mcp.server.fastmcp import Context from codewiki.src.be.agent_tools.deps import CodeWikiDeps -from codewiki.src.be.module_naming import normalize_sub_module_specs +from codewiki.src.be.module_naming import plan_sub_module_specs if TYPE_CHECKING: from codewiki.src.be.caw_backend import CawBackend @@ -242,9 +242,10 @@ async def str_replace_editor( "sub_module_specs: a dictionary mapping sub-module names to their core component IDs. " "Example: {'authentication': ['auth_handler.py::AuthHandler'], " "'database': ['db_client.py::DBClient']}\n" - "Sub-module names must be unique across the whole wiki; a name already used by another " - "module is automatically prefixed with the current module name, and the tool result " - "reports the final file names actually saved." + "Sub-module names must be unique across the whole wiki. A name already used by another " + "module is prefixed with the current module name; a name that is already documented " + "(plain or prefixed) is SKIPPED, not regenerated. The result reports the files actually " + "saved and every skipped request." ) ) async def generate_sub_module_documentation( @@ -284,15 +285,21 @@ def _run_sub_modules(self, sub_module_specs: dict[str, list[str]]) -> str: # Resolve name collisions against the module tree and files already on # disk before touching the tree (issue #76): docs live in one flat directory. - name_map = normalize_sub_module_specs( + # A request that is already documented (plain or parent-prefixed name) is + # skipped rather than renamed x_2, x_3, ... (issue #113). + plan = plan_sub_module_specs( sub_module_specs, previous_module_name, deps.module_tree, deps.absolute_docs_path, ) + name_map = plan.name_map + if not name_map: + return _skipped_report(plan.skipped, deps.current_module_name) final_specs = { name_map[requested_name]: core_ids for requested_name, core_ids in sub_module_specs.items() + if requested_name in name_map } # Add sub-modules to the in-memory module tree. @@ -352,4 +359,17 @@ def _run_sub_modules(self, sub_module_specs: dict[str, list[str]]) -> str: logger.warning( "Sub-module documentation missing after generation: %s", ", ".join(missing) ) + if plan.skipped: + report += " " + _skipped_report(plan.skipped, deps.current_module_name) return report + + +def _skipped_report(skipped: dict[str, str], current_module_name: str) -> str: + """Tell the parent agent, unambiguously, not to retry skipped sub-modules.""" + if not skipped: + return "No sub-modules were generated." + items = ", ".join(f"'{name}' ({reason})" for name, reason in skipped.items()) + return ( + f"Skipped sub-modules: {items}. Do NOT call generate_sub_module_documentation again " + f"for these; link the existing pages from `{current_module_name}.md` instead." + ) diff --git a/codewiki/src/be/module_naming.py b/codewiki/src/be/module_naming.py index 367a3df3..48e23176 100644 --- a/codewiki/src/be/module_naming.py +++ b/codewiki/src/be/module_naming.py @@ -8,6 +8,7 @@ """ import os +from dataclasses import dataclass, field from typing import Any, Dict, List, Optional, Set import logging @@ -97,6 +98,66 @@ def normalize_sub_module_specs( return name_map +@dataclass +class SubModulePlan: + """What ``generate_sub_module_documentation`` will actually create. + + ``name_map``: requested name -> final, unique file stem to generate. + ``skipped``: requested name -> reason it is not generated (already + documented under that name or its parent-prefixed variant). + """ + + name_map: Dict[str, str] = field(default_factory=dict) + skipped: Dict[str, str] = field(default_factory=dict) + + +def plan_sub_module_specs( + sub_module_specs: Dict[str, Any], + parent_name: Optional[str], + module_tree: Dict[str, Any], + working_dir: str, +) -> SubModulePlan: + """Decide which requested sub-modules to generate (issue #113). + + Like :func:`normalize_sub_module_specs`, a name that collides with the + tree, a ``.md`` on disk or a reserved stem gets the parent prefix. Unlike + it, a request whose plain *and* prefixed names are both taken is treated + as a repeat of something already documented and skipped, never renamed + with a numeric suffix: that suffixing is what let one agent regenerate the + same modules as ``x_2``, ``x_3``, ... without ever converging. + """ + taken = collect_module_tree_names(module_tree) + taken |= _existing_doc_stems(working_dir) + taken |= RESERVED_STEMS + + plan = SubModulePlan() + for requested_name in sub_module_specs: + name = sanitize_module_name(requested_name) + if name not in taken: + final_name = name + else: + prefixed = f"{sanitize_module_name(parent_name)}_{name}" if parent_name else name + if prefixed in taken: + existing = name if name in taken else prefixed + reason = f"already documented as {existing}.md; do not request it again" + plan.skipped[requested_name] = reason + logger.info( + "Sub-module '%s' already documented as '%s'; skipping duplicate request.", + requested_name, + existing, + ) + continue + final_name = prefixed + logger.info( + "Sub-module name '%s' collides with an existing module or file; renamed to '%s'.", + requested_name, + final_name, + ) + taken.add(final_name) + plan.name_map[requested_name] = final_name + return plan + + def dedupe_module_tree_names(module_tree: Dict[str, Any]) -> Dict[str, Any]: """Sanitize and uniquify all module names in a freshly clustered tree. diff --git a/codewiki/src/be/updater/orchestrator.py b/codewiki/src/be/updater/orchestrator.py index 83884b6c..3f7ebba2 100644 --- a/codewiki/src/be/updater/orchestrator.py +++ b/codewiki/src/be/updater/orchestrator.py @@ -9,7 +9,7 @@ from typing import Any from codewiki.src.be.backend import LLMBackend -from codewiki.src.be.cluster_modules import cluster_modules +from codewiki.src.be.cluster_modules import cluster_modules, get_clustering_input_token_count from codewiki.src.be.dependency_analyzer.models.core import Node from codewiki.src.be.updater import pages as P from codewiki.src.be.updater import tree as T @@ -41,7 +41,12 @@ ) from codewiki.src.be.updater.routing import RoutingAgent from codewiki.src.be.updater.stale_scan import StaleScanner -from codewiki.src.be.updater.tree_repair import RepairResult, repair_tree +from codewiki.src.be.updater.tree_repair import ( + RULE_AGENT, + RepairResult, + RoutingDecision, + repair_tree, +) from codewiki.src.config import MODULE_TREE_FILENAME, Config from codewiki.src.utils import file_manager @@ -85,6 +90,17 @@ def _load_state(self, old_graph_path: str) -> tuple[dict[str, Node], dict[str, A self.record.detector_notes.append("whole-repository mode: one virtual leaf (overview)") return old_graph, tree + def _route_to_overview( + self, orphans: list[str], context: dict[str, Any] + ) -> list[RoutingDecision]: + """Whole-repository mode: every orphan belongs to the single page.""" + return [ + RoutingDecision( + cid, RULE_AGENT, (P.OVERVIEW_STEM,), detail="whole-repository mode: single page" + ) + for cid in orphans + ] + def _module_path(self, path: tuple[str, ...]) -> list[str]: if self.whole_repo and path == (P.OVERVIEW_STEM,): return [] @@ -253,6 +269,12 @@ async def _run( if self.opts.use_routing_agent else None ) + if self.whole_repo: + # One page documents everything: orphans go to the virtual leaf + # without an LLM call and without creating leaves. Each created + # leaf would otherwise map back onto the overview page and rewrite + # it once more in Step 5 (issue #113). + router = self._route_to_overview repair: RepairResult = repair_tree( old_tree, diff, new_graph, tracked_new, self.opts, route_orphans=router ) @@ -288,11 +310,43 @@ async def _run( ratios["fired"] = ( ratios["r_leaf"] >= self.opts.tau_full or ratios["r_tree"] >= self.opts.tau_tree ) - if self.whole_repo and ratios["fired"]: - # One virtual leaf: any change is 100% active by construction, and - # patching that single page is exactly the incremental step. - ratios["fired"] = False - ratios["note"] = "whole-repository mode: fallback rule not applied" + if self.whole_repo: + # One virtual leaf: any change is 100% active by construction, so + # r_leaf says nothing. What matters is whether the *current* scope + # still fits one page under the same threshold a fresh build uses + # to skip clustering. A whole-repo baseline built from a narrow + # --include and then updated against the full repo does not, and + # feeding that to the single-page agent runs unbounded (issue #113). + scope = [c for c in leaf_nodes if c in new_graph] + tokens = get_clustering_input_token_count(scope, new_graph) + threshold = self.config.max_token_per_module + ratios["clustering_tokens"] = tokens + ratios["tau_cluster"] = threshold + if tokens <= threshold: + ratios["fired"] = False + note = "whole-repository mode: scope still fits one page; fallback rule not applied" + ratios["note"] = note + logger.info( + "Whole-repository mode: current scope is %d clustering tokens " + "(threshold %d, %d leaf nodes); updating the single page in place", + tokens, + threshold, + len(scope), + ) + else: + ratios["fired"] = True + ratios["note"] = ( + "whole-repository baseline but current scope exceeds the clustering " + "threshold; a fresh build would cluster, so fall back to a full build" + ) + rec.detector_notes.append(ratios["note"]) + logger.warning( + "Whole-repository baseline cannot be updated in place: current scope is " + "%d clustering tokens (threshold %d, %d leaf nodes); falling back to a full build", + tokens, + threshold, + len(scope), + ) rec.fallback = ratios if ratios["fired"]: rec.outcome = OUTCOME_FULL_FALLBACK @@ -311,6 +365,13 @@ async def _run( # ---- Step 5: sequential leaf agents order = order_active(active, new_tree, new_graph) + if self.whole_repo and len(order) > 1: + # Every active unit is the same overview page; regenerate it once. + overview_path = (P.OVERVIEW_STEM,) + order = [overview_path if overview_path in reports else order[0]] + rec.detector_notes.append( + f"whole-repository mode: {len(active)} active units collapsed into one overview run" + ) dep = T.leaf_dependents(new_tree, new_graph) inv = inverse(ref_index) rec.active = [ diff --git a/tests/test_module_naming.py b/tests/test_module_naming.py new file mode 100644 index 00000000..e87806cc --- /dev/null +++ b/tests/test_module_naming.py @@ -0,0 +1,168 @@ +"""Unit tests for module-name collision handling (issue #76).""" + +from codewiki.src.be.module_naming import ( + collect_module_tree_names, + dedupe_module_tree_names, + normalize_sub_module_specs, + plan_sub_module_specs, + resolve_unique_name, + sanitize_module_name, +) + + +class TestSanitizeModuleName: + def test_spaces_become_underscores(self): + assert sanitize_module_name("text ui") == "text_ui" + + def test_path_separators_removed(self): + assert "/" not in sanitize_module_name("engine/search") + assert "\\" not in sanitize_module_name("engine\\search") + + def test_case_preserved(self): + assert sanitize_module_name("TextUI") == "TextUI" + + def test_empty_falls_back(self): + assert sanitize_module_name(" ") == "module" + + +class TestResolveUniqueName: + def test_unique_name_unchanged(self): + assert resolve_unique_name("search", "engine", {"evaluation"}) == "search" + + def test_conflict_gets_parent_prefix(self): + assert resolve_unique_name("search", "engine", {"search"}) == "engine_search" + + def test_prefixed_name_also_taken_gets_numeric_suffix(self): + taken = {"search", "engine_search"} + assert resolve_unique_name("search", "engine", taken) == "engine_search_2" + + def test_conflict_without_parent_gets_numeric_suffix(self): + assert resolve_unique_name("search", None, {"search"}) == "search_2" + + +class TestNormalizeSubModuleSpecs: + def test_no_conflicts_names_unchanged(self, tmp_path): + module_tree = {"engine": {"components": [], "children": {}}} + name_map = normalize_sub_module_specs( + {"search": [], "evaluation": []}, "engine", module_tree, str(tmp_path) + ) + assert name_map == {"search": "search", "evaluation": "evaluation"} + + def test_conflict_with_tree_name_gets_parent_prefix(self, tmp_path): + # Issue #76 variant 1: engine's sub-modules named like top-level modules + module_tree = { + "engine": {"components": [], "children": {}}, + "search": {"components": [], "children": {}}, + "evaluation": {"components": [], "children": {}}, + } + name_map = normalize_sub_module_specs( + {"search": [], "evaluation": []}, "engine", module_tree, str(tmp_path) + ) + assert name_map == { + "search": "engine_search", + "evaluation": "engine_evaluation", + } + + def test_conflict_with_existing_md_file_gets_parent_prefix(self, tmp_path): + (tmp_path / "search.md").write_text("existing docs") + name_map = normalize_sub_module_specs({"search": []}, "engine", {}, str(tmp_path)) + assert name_map == {"search": "engine_search"} + + def test_reserved_stems_are_taken(self, tmp_path): + name_map = normalize_sub_module_specs({"overview": []}, "engine", {}, str(tmp_path)) + assert name_map == {"overview": "engine_overview"} + + def test_batch_internal_collision_after_sanitization(self, tmp_path): + # Two requested names that sanitize to the same stem + name_map = normalize_sub_module_specs( + {"text ui": [], "text_ui": []}, "textui", {}, str(tmp_path) + ) + assert name_map["text ui"] == "text_ui" + assert name_map["text_ui"] == "textui_text_ui" + assert len(set(name_map.values())) == 2 + + +class TestDedupeModuleTreeNames: + def test_nested_vs_toplevel_collision(self): + tree = { + "engine": { + "components": [], + "children": {"search": {"components": [], "children": {}}}, + }, + "search": {"components": [], "children": {}}, + } + deduped = dedupe_module_tree_names(tree) + names = [] + stack = [deduped] + while stack: + level = stack.pop() + for name, info in level.items(): + names.append(name) + if isinstance(info.get("children"), dict): + stack.append(info["children"]) + assert len(names) == len(set(names)) + assert "engine" in names + + def test_no_collision_tree_unchanged(self): + tree = { + "engine": { + "components": ["a"], + "children": {"engine_search": {"components": ["b"], "children": {}}}, + }, + } + assert dedupe_module_tree_names(tree) == tree + + +class TestCollectModuleTreeNames: + def test_collects_all_depths(self): + tree = { + "a": {"children": {"b": {"children": {"c": {"children": {}}}}}}, + } + assert collect_module_tree_names(tree) == {"a", "b", "c"} + + +class TestPlanSubModuleSpecs: + """Issue #113: agent-inserted sub-modules are never renamed x_2, x_3, ...; + a request that is already documented is skipped instead.""" + + def test_no_conflicts_matches_normalize(self, tmp_path): + module_tree = {"engine": {"components": [], "children": {}}} + specs = {"search": [], "evaluation": []} + plan = plan_sub_module_specs(specs, "engine", module_tree, str(tmp_path)) + assert plan.name_map == normalize_sub_module_specs( + specs, "engine", module_tree, str(tmp_path) + ) + assert plan.skipped == {} + + def test_conflict_gets_parent_prefix(self, tmp_path): + (tmp_path / "search.md").write_text("existing docs") + plan = plan_sub_module_specs({"search": []}, "engine", {}, str(tmp_path)) + assert plan.name_map == {"search": "engine_search"} + assert plan.skipped == {} + + def test_plain_and_prefixed_taken_is_skipped_not_suffixed(self, tmp_path): + (tmp_path / "approval.md").write_text("round 1") + (tmp_path / "overview_approval.md").write_text("round 2") + plan = plan_sub_module_specs({"approval": []}, "overview", {}, str(tmp_path)) + assert plan.name_map == {} + assert "already documented as approval.md" in plan.skipped["approval"] + + def test_tree_names_count_as_taken(self, tmp_path): + module_tree = { + "search": {"components": [], "children": {}}, + "engine_search": {"components": [], "children": {}}, + } + plan = plan_sub_module_specs( + {"search": [], "ranking": []}, "engine", module_tree, str(tmp_path) + ) + assert plan.name_map == {"ranking": "ranking"} + assert set(plan.skipped) == {"search"} + + def test_batch_internal_duplicate_is_prefixed_then_skipped(self, tmp_path): + # Same stem requested three times in one batch: plain, prefixed, then skipped. + plan = plan_sub_module_specs( + {"text ui": [], "text_ui": [], " text ui": []}, "textui", {}, str(tmp_path) + ) + assert plan.name_map == {"text ui": "text_ui", "text_ui": "textui_text_ui"} + assert " text ui" in plan.skipped + assert not any(v.endswith("_2") for v in plan.name_map.values()) diff --git a/tests/test_sub_module_dedupe.py b/tests/test_sub_module_dedupe.py new file mode 100644 index 00000000..83a78a3c --- /dev/null +++ b/tests/test_sub_module_dedupe.py @@ -0,0 +1,106 @@ +"""The pydantic-ai sub-module tool never regenerates an already documented +module under a suffixed name (issue #113).""" + +from __future__ import annotations + +import asyncio +from types import SimpleNamespace + +import pytest + +from codewiki.src.be.agent_tools import generate_sub_module_documentations as mod +from codewiki.src.be.agent_tools.deps import CodeWikiDeps +from codewiki.src.be.dependency_analyzer.models.core import Node + + +class FakeAgent: + """Stands in for pydantic_ai.Agent: writes the sub-module page and returns.""" + + runs: list[str] = [] + + def __init__(self, *args, **kwargs): + self.name = kwargs.get("name") + + async def run(self, prompt, deps): + FakeAgent.runs.append(deps.current_module_name) + page = f"{deps.absolute_docs_path}/{deps.current_module_name}.md" + with open(page, "w", encoding="utf-8") as f: + f.write(f"# {deps.current_module_name}\n") + return SimpleNamespace(output="ok") + + +def _node(cid: str) -> Node: + rel, name = cid.split("::", 1) + return Node( + id=cid, + name=name, + component_type="function", + file_path=f"/repo/{rel}", + relative_path=rel, + source_code=f"def {name}():\n pass\n", + ) + + +def _deps(tmp_path) -> CodeWikiDeps: + components = {c: _node(c) for c in ("a.py::fa", "b.py::fb")} + return CodeWikiDeps( + absolute_docs_path=str(tmp_path), + absolute_repo_path="/repo", + registry={}, + components=components, + path_to_current_module=[], + current_module_name="overview", + module_tree={}, + max_depth=2, + current_depth=1, + config=SimpleNamespace(max_token_per_leaf_module=4000), + custom_instructions="", + ) + + +@pytest.fixture(autouse=True) +def _fake_agent(monkeypatch): + FakeAgent.runs = [] + monkeypatch.setattr(mod, "Agent", FakeAgent) + monkeypatch.setattr(mod, "create_fallback_models", lambda config: None) + + +def _call(deps, specs): + return asyncio.run(mod.generate_sub_module_documentation(SimpleNamespace(deps=deps), specs)) + + +def test_repeat_request_is_skipped_not_suffixed(tmp_path): + deps = _deps(tmp_path) + specs = {"approval": ["a.py::fa"], "billing": ["b.py::fb"]} + + first = _call(deps, specs) + assert "approval.md" in first and "billing.md" in first + assert (tmp_path / "approval.md").exists() and (tmp_path / "billing.md").exists() + assert set(deps.module_tree) == {"approval", "billing"} + + # Round 2: same names again -> parent prefix, once (issue #76 behaviour). + _call(deps, specs) + assert (tmp_path / "overview_approval.md").exists() + assert (tmp_path / "overview_billing.md").exists() + + # Round 3: plain and prefixed both exist -> skipped, nothing new anywhere. + files_before = sorted(p.name for p in tmp_path.iterdir()) + tree_before = dict(deps.module_tree) + runs_before = len(FakeAgent.runs) + third = _call(deps, specs) + assert "Skipped sub-modules" in third and "Do NOT call" in third + assert sorted(p.name for p in tmp_path.iterdir()) == files_before + assert deps.module_tree == tree_before + assert len(FakeAgent.runs) == runs_before + assert not any(p.name.endswith("_2.md") for p in tmp_path.iterdir()) + + +def test_mixed_batch_generates_only_new_names(tmp_path): + deps = _deps(tmp_path) + (tmp_path / "approval.md").write_text("old") + (tmp_path / "overview_approval.md").write_text("old") + report = _call(deps, {"approval": ["a.py::fa"], "billing": ["b.py::fb"]}) + assert (tmp_path / "billing.md").exists() + assert FakeAgent.runs == ["billing"] + assert "billing.md" in report and "'approval'" in report and "Skipped" in report + assert set(deps.module_tree) == {"billing"} diff --git a/tests/test_updater_orchestrator.py b/tests/test_updater_orchestrator.py index 6bfb2a3f..be1892d4 100644 --- a/tests/test_updater_orchestrator.py +++ b/tests/test_updater_orchestrator.py @@ -8,7 +8,7 @@ from pathlib import Path from types import SimpleNamespace -from updater_toy import OAUTH, USER, graph_r1, graph_r2, tracked_r2, tree_r1, write_pages +from updater_toy import OAUTH, USER, graph_r1, graph_r2, node, tracked_r2, tree_r1, write_pages from codewiki.src.be.agent_tools.str_replace_editor import str_replace_editor from codewiki.src.be.backend import AgentReply @@ -228,3 +228,53 @@ def test_whole_repo_mode(tmp_path): assert [a["page"] for a in rec.active] == ["overview"] assert json.load(open(docs / "module_tree.json")) == {} assert backend.update_calls and backend.update_calls[0][0] == "overview" + assert rec.fallback["fired"] is False + assert "fits" in rec.fallback["note"] + assert rec.fallback["clustering_tokens"] <= rec.fallback["tau_cluster"] + + +def test_whole_repo_orphans_go_to_overview_without_new_leaves(tmp_path): + # Issue #113: an added component with no neighbour in the tree used to be + # routed by the LLM agent into a *new* leaf; each such leaf then rewrote + # the overview page once more. In whole-repo mode orphans belong to the + # single page, with no routing call and no created leaves. + docs, config, prev = _setup(tmp_path, tree={}) + for stem in ("auth", "api", "core", "storage", "pipeline"): + (docs / f"{stem}.md").unlink() + backend = FakeBackend() + g2 = graph_r2() + for i in range(3): + cid = f"src/new/mod{i}.py::Standalone{i}" + g2[cid] = node(cid, "class", f"class Standalone{i}:\n pass\n") + tracked = sorted(tracked_r2() | {c for c in g2 if "Standalone" in c}) + gen = _generator(config, backend) + upd = IncrementalUpdater(config, backend, gen, UpdateOptions()) + rec = asyncio.run(upd.run(prev, g2, tracked, {"old_commit": "old", "new_commit": "new"})) + assert rec.outcome == "incremental" + assert rec.repair["created_leaves"] == [] + assert rec.repair["orphans"] == [] + assert [a["page"] for a in rec.active] == ["overview"] + assert backend.complete_calls == [] # no LLM routing + # The single page was written exactly once. + assert backend.update_calls.count(("overview", backend.update_calls[0][1])) == 1 + assert len(backend.update_calls) + len(backend.module_calls) <= 2 + + +def test_whole_repo_baseline_falls_back_when_scope_needs_clustering(tmp_path): + # Issue #113: a whole-repo baseline (empty tree) updated against a scope + # that a fresh build would cluster must not be patched as one page. + docs, config, prev = _setup(tmp_path, tree={}) + for stem in ("auth", "api", "core", "storage", "pipeline"): + (docs / f"{stem}.md").unlink() + config.max_token_per_module = 1 + backend = FakeBackend() + rec = _run(docs, config, prev, backend, UpdateOptions()) + assert rec.outcome == "full_fallback" + assert rec.fallback["fired"] is True + assert rec.fallback["clustering_tokens"] > rec.fallback["tau_cluster"] == 1 + assert any("exceeds the clustering threshold" in n for n in rec.detector_notes) + # No agent ran and nothing on disk changed. + assert backend.module_calls == [] + assert backend.update_calls == [] + assert json.load(open(docs / "module_tree.json")) == {} + assert (docs / "overview.md").exists()