Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 25 additions & 5 deletions codewiki/src/be/agent_tools/generate_sub_module_documentations.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
from pydantic_ai import RunContext, Tool, Agent

from codewiki.src.be.agent_tools.deps import CodeWikiDeps
from codewiki.src.be.module_naming import normalize_sub_module_specs
from codewiki.src.be.module_naming import plan_sub_module_specs
from codewiki.src.be.agent_tools.read_code_components import read_code_components_tool
from codewiki.src.be.agent_tools.str_replace_editor import str_replace_editor_tool
from codewiki.src.be.llm_services import create_fallback_models
Expand All @@ -25,9 +25,10 @@ async def generate_sub_module_documentation(
sub_module_specs: A dictionary mapping sub-module names to their core component IDs.
Example: {"authentication": ["auth_handler.py::AuthHandler", "auth_middleware.py::verify_token"], "database": ["db_client.py::DBClient", "models.py::UserModel"]}
Each key is a descriptive sub-module name, and the value is a list of component IDs from the current module's core components that belong to that sub-module.
Sub-module names must be unique across the whole wiki; a name that is already used by another
module is automatically prefixed with the current module name, and the tool result reports the
final file names actually saved.
Sub-module names must be unique across the whole wiki. A name already used by another module is
prefixed with the current module name; a name that is already documented (plain or prefixed)
is SKIPPED, not regenerated. The tool result reports the final file names actually saved and
every skipped request.
"""

deps = ctx.deps
Expand All @@ -38,15 +39,21 @@ async def generate_sub_module_documentation(

# Resolve name collisions against the module tree and files already on disk
# before touching the tree (issue #76): docs live in one flat directory.
name_map = normalize_sub_module_specs(
# A request that is already documented (plain or parent-prefixed name) is
# skipped rather than renamed x_2, x_3, ... (issue #113).
plan = plan_sub_module_specs(
sub_module_specs,
previous_module_name,
deps.module_tree,
deps.absolute_docs_path,
)
name_map = plan.name_map
if not name_map:
return _skipped_report(plan.skipped, deps.current_module_name)
final_specs = {
name_map[requested_name]: core_component_ids
for requested_name, core_component_ids in sub_module_specs.items()
if requested_name in name_map
}

# add the sub-module to the module tree
Expand Down Expand Up @@ -135,9 +142,22 @@ async def generate_sub_module_documentation(
if missing:
report += f" MISSING (generation did not produce these files): {', '.join(missing)}."
logger.warning("Sub-module documentation missing after generation: %s", ", ".join(missing))
if plan.skipped:
report += " " + _skipped_report(plan.skipped, deps.current_module_name)
return report


def _skipped_report(skipped: dict[str, str], current_module_name: str) -> str:
"""Tell the parent agent, unambiguously, not to retry skipped sub-modules."""
if not skipped:
return "No sub-modules were generated."
items = ", ".join(f"'{name}' ({reason})" for name, reason in skipped.items())
return (
f"Skipped sub-modules: {items}. Do NOT call generate_sub_module_documentation again "
f"for these; link the existing pages from `{current_module_name}.md` instead."
)


generate_sub_module_documentation_tool = Tool(
function=generate_sub_module_documentation,
name="generate_sub_module_documentation",
Expand Down
30 changes: 25 additions & 5 deletions codewiki/src/be/caw_toolkit.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,7 @@
from mcp.server.fastmcp import Context

from codewiki.src.be.agent_tools.deps import CodeWikiDeps
from codewiki.src.be.module_naming import normalize_sub_module_specs
from codewiki.src.be.module_naming import plan_sub_module_specs

if TYPE_CHECKING:
from codewiki.src.be.caw_backend import CawBackend
Expand Down Expand Up @@ -242,9 +242,10 @@ async def str_replace_editor(
"sub_module_specs: a dictionary mapping sub-module names to their core component IDs. "
"Example: {'authentication': ['auth_handler.py::AuthHandler'], "
"'database': ['db_client.py::DBClient']}\n"
"Sub-module names must be unique across the whole wiki; a name already used by another "
"module is automatically prefixed with the current module name, and the tool result "
"reports the final file names actually saved."
"Sub-module names must be unique across the whole wiki. A name already used by another "
"module is prefixed with the current module name; a name that is already documented "
"(plain or prefixed) is SKIPPED, not regenerated. The result reports the files actually "
"saved and every skipped request."
)
)
async def generate_sub_module_documentation(
Expand Down Expand Up @@ -284,15 +285,21 @@ def _run_sub_modules(self, sub_module_specs: dict[str, list[str]]) -> str:

# Resolve name collisions against the module tree and files already on
# disk before touching the tree (issue #76): docs live in one flat directory.
name_map = normalize_sub_module_specs(
# A request that is already documented (plain or parent-prefixed name) is
# skipped rather than renamed x_2, x_3, ... (issue #113).
plan = plan_sub_module_specs(
sub_module_specs,
previous_module_name,
deps.module_tree,
deps.absolute_docs_path,
)
name_map = plan.name_map
if not name_map:
return _skipped_report(plan.skipped, deps.current_module_name)
final_specs = {
name_map[requested_name]: core_ids
for requested_name, core_ids in sub_module_specs.items()
if requested_name in name_map
}

# Add sub-modules to the in-memory module tree.
Expand Down Expand Up @@ -352,4 +359,17 @@ def _run_sub_modules(self, sub_module_specs: dict[str, list[str]]) -> str:
logger.warning(
"Sub-module documentation missing after generation: %s", ", ".join(missing)
)
if plan.skipped:
report += " " + _skipped_report(plan.skipped, deps.current_module_name)
return report


def _skipped_report(skipped: dict[str, str], current_module_name: str) -> str:
"""Tell the parent agent, unambiguously, not to retry skipped sub-modules."""
if not skipped:
return "No sub-modules were generated."
items = ", ".join(f"'{name}' ({reason})" for name, reason in skipped.items())
return (
f"Skipped sub-modules: {items}. Do NOT call generate_sub_module_documentation again "
f"for these; link the existing pages from `{current_module_name}.md` instead."
)
61 changes: 61 additions & 0 deletions codewiki/src/be/module_naming.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
"""

import os
from dataclasses import dataclass, field
from typing import Any, Dict, List, Optional, Set

import logging
Expand Down Expand Up @@ -97,6 +98,66 @@ def normalize_sub_module_specs(
return name_map


@dataclass
class SubModulePlan:
"""What ``generate_sub_module_documentation`` will actually create.

``name_map``: requested name -> final, unique file stem to generate.
``skipped``: requested name -> reason it is not generated (already
documented under that name or its parent-prefixed variant).
"""

name_map: Dict[str, str] = field(default_factory=dict)
skipped: Dict[str, str] = field(default_factory=dict)


def plan_sub_module_specs(
sub_module_specs: Dict[str, Any],
parent_name: Optional[str],
module_tree: Dict[str, Any],
working_dir: str,
) -> SubModulePlan:
"""Decide which requested sub-modules to generate (issue #113).

Like :func:`normalize_sub_module_specs`, a name that collides with the
tree, a ``.md`` on disk or a reserved stem gets the parent prefix. Unlike
it, a request whose plain *and* prefixed names are both taken is treated
as a repeat of something already documented and skipped, never renamed
with a numeric suffix: that suffixing is what let one agent regenerate the
same modules as ``x_2``, ``x_3``, ... without ever converging.
"""
taken = collect_module_tree_names(module_tree)
taken |= _existing_doc_stems(working_dir)
taken |= RESERVED_STEMS

plan = SubModulePlan()
for requested_name in sub_module_specs:
name = sanitize_module_name(requested_name)
if name not in taken:
final_name = name
else:
prefixed = f"{sanitize_module_name(parent_name)}_{name}" if parent_name else name
if prefixed in taken:
existing = name if name in taken else prefixed
reason = f"already documented as {existing}.md; do not request it again"
plan.skipped[requested_name] = reason
logger.info(
"Sub-module '%s' already documented as '%s'; skipping duplicate request.",
requested_name,
existing,
)
continue
final_name = prefixed
logger.info(
"Sub-module name '%s' collides with an existing module or file; renamed to '%s'.",
requested_name,
final_name,
)
taken.add(final_name)
plan.name_map[requested_name] = final_name
return plan


def dedupe_module_tree_names(module_tree: Dict[str, Any]) -> Dict[str, Any]:
"""Sanitize and uniquify all module names in a freshly clustered tree.

Expand Down
75 changes: 68 additions & 7 deletions codewiki/src/be/updater/orchestrator.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@
from typing import Any

from codewiki.src.be.backend import LLMBackend
from codewiki.src.be.cluster_modules import cluster_modules
from codewiki.src.be.cluster_modules import cluster_modules, get_clustering_input_token_count
from codewiki.src.be.dependency_analyzer.models.core import Node
from codewiki.src.be.updater import pages as P
from codewiki.src.be.updater import tree as T
Expand Down Expand Up @@ -41,7 +41,12 @@
)
from codewiki.src.be.updater.routing import RoutingAgent
from codewiki.src.be.updater.stale_scan import StaleScanner
from codewiki.src.be.updater.tree_repair import RepairResult, repair_tree
from codewiki.src.be.updater.tree_repair import (
RULE_AGENT,
RepairResult,
RoutingDecision,
repair_tree,
)
from codewiki.src.config import MODULE_TREE_FILENAME, Config
from codewiki.src.utils import file_manager

Expand Down Expand Up @@ -85,6 +90,17 @@ def _load_state(self, old_graph_path: str) -> tuple[dict[str, Node], dict[str, A
self.record.detector_notes.append("whole-repository mode: one virtual leaf (overview)")
return old_graph, tree

def _route_to_overview(
self, orphans: list[str], context: dict[str, Any]
) -> list[RoutingDecision]:
"""Whole-repository mode: every orphan belongs to the single page."""
return [
RoutingDecision(
cid, RULE_AGENT, (P.OVERVIEW_STEM,), detail="whole-repository mode: single page"
)
for cid in orphans
]

def _module_path(self, path: tuple[str, ...]) -> list[str]:
if self.whole_repo and path == (P.OVERVIEW_STEM,):
return []
Expand Down Expand Up @@ -253,6 +269,12 @@ async def _run(
if self.opts.use_routing_agent
else None
)
if self.whole_repo:
# One page documents everything: orphans go to the virtual leaf
# without an LLM call and without creating leaves. Each created
# leaf would otherwise map back onto the overview page and rewrite
# it once more in Step 5 (issue #113).
router = self._route_to_overview
repair: RepairResult = repair_tree(
old_tree, diff, new_graph, tracked_new, self.opts, route_orphans=router
)
Expand Down Expand Up @@ -288,11 +310,43 @@ async def _run(
ratios["fired"] = (
ratios["r_leaf"] >= self.opts.tau_full or ratios["r_tree"] >= self.opts.tau_tree
)
if self.whole_repo and ratios["fired"]:
# One virtual leaf: any change is 100% active by construction, and
# patching that single page is exactly the incremental step.
ratios["fired"] = False
ratios["note"] = "whole-repository mode: fallback rule not applied"
if self.whole_repo:
# One virtual leaf: any change is 100% active by construction, so
# r_leaf says nothing. What matters is whether the *current* scope
# still fits one page under the same threshold a fresh build uses
# to skip clustering. A whole-repo baseline built from a narrow
# --include and then updated against the full repo does not, and
# feeding that to the single-page agent runs unbounded (issue #113).
scope = [c for c in leaf_nodes if c in new_graph]
tokens = get_clustering_input_token_count(scope, new_graph)
threshold = self.config.max_token_per_module
ratios["clustering_tokens"] = tokens
ratios["tau_cluster"] = threshold
if tokens <= threshold:
ratios["fired"] = False
note = "whole-repository mode: scope still fits one page; fallback rule not applied"
ratios["note"] = note
logger.info(
"Whole-repository mode: current scope is %d clustering tokens "
"(threshold %d, %d leaf nodes); updating the single page in place",
tokens,
threshold,
len(scope),
)
else:
ratios["fired"] = True
ratios["note"] = (
"whole-repository baseline but current scope exceeds the clustering "
"threshold; a fresh build would cluster, so fall back to a full build"
)
rec.detector_notes.append(ratios["note"])
logger.warning(
"Whole-repository baseline cannot be updated in place: current scope is "
"%d clustering tokens (threshold %d, %d leaf nodes); falling back to a full build",
tokens,
threshold,
len(scope),
)
rec.fallback = ratios
if ratios["fired"]:
rec.outcome = OUTCOME_FULL_FALLBACK
Expand All @@ -311,6 +365,13 @@ async def _run(

# ---- Step 5: sequential leaf agents
order = order_active(active, new_tree, new_graph)
if self.whole_repo and len(order) > 1:
# Every active unit is the same overview page; regenerate it once.
overview_path = (P.OVERVIEW_STEM,)
order = [overview_path if overview_path in reports else order[0]]
rec.detector_notes.append(
f"whole-repository mode: {len(active)} active units collapsed into one overview run"
)
dep = T.leaf_dependents(new_tree, new_graph)
inv = inverse(ref_index)
rec.active = [
Expand Down
Loading
Loading