From 1697ec8946e95bb54bb6b7d6b85d0b0b4f5b6183 Mon Sep 17 00:00:00 2001 From: anhnh2002 Date: Tue, 29 Sep 2026 14:45:00 +0700 Subject: [PATCH 1/2] agent run limits, fallback on malformed responses, overview instructions, ship updater - Package: add codewiki.src.be.updater to [tool.setuptools] packages; a non-editable install had no updater, so every --update failed on import. New test checks every package directory is listed. (#119) - Agent limits: pass UsageLimits(request_limit=...) to every agent run and retries=... to every Agent. pydantic-ai defaults are 50 requests per run and 1 retry per failing tool call, which complex modules and generate_sub_module_documentation hit. New settings request_limit (default 100) and agent_retries (default 3), in `codewiki config set`, `config show` and as per-run overrides on `codewiki generate`. (#115, #118) - Fallback: FallbackModel now also falls back on UnexpectedModelBehavior (a 200 response whose body does not parse), keeping ModelAPIError. (#117) - Overview pages: MODULE_OVERVIEW_PROMPT / REPO_OVERVIEW_PROMPT had no slot for the user's instructions. complete() takes an optional system_prompt (OpenAI- compatible, litellm, Azure, and caw via CawAgent(system_prompt=...)), and parent/repo overviews send the instructions as a system message. (#116) - Sub-module agents built their system prompt with a raw .format(), so with no instructions the prompt ended in the literal text "None"; they now use format_system_prompt / format_leaf_system_prompt like the top-level agents. Reported in #115, #116, #117, #118, #119. --- .gitignore | 1 + codewiki/cli/adapters/doc_generator.py | 2 + codewiki/cli/commands/config.py | 32 +++++ codewiki/cli/commands/generate.py | 21 +++ codewiki/cli/config_manager.py | 8 ++ codewiki/cli/models/config.py | 10 ++ .../generate_sub_module_documentations.py | 18 ++- codewiki/src/be/backend.py | 3 +- codewiki/src/be/caw_backend.py | 2 + codewiki/src/be/documentation_generator.py | 6 +- codewiki/src/be/llm_services.py | 36 ++++-- codewiki/src/be/prompt_template.py | 17 +++ codewiki/src/be/pydantic_ai_backend.py | 14 +- codewiki/src/config.py | 14 ++ pyproject.toml | 1 + tests/test_agent_limits.py | 121 ++++++++++++++++++ tests/test_processing_order_update.py | 6 +- tests/test_sub_module_dedupe.py | 4 +- tests/test_updater_orchestrator.py | 2 +- 19 files changed, 293 insertions(+), 25 deletions(-) create mode 100644 tests/test_agent_limits.py diff --git a/.gitignore b/.gitignore index d36f2b71..ecd28ba0 100644 --- a/.gitignore +++ b/.gitignore @@ -47,6 +47,7 @@ tests/* !tests/test_branding.py !tests/test_artifact_analyzer.py !tests/test_scala_analyzer.py +!tests/test_agent_limits.py # Jupyter *.ipynb diff --git a/codewiki/cli/adapters/doc_generator.py b/codewiki/cli/adapters/doc_generator.py index 92c066dd..07f425fc 100644 --- a/codewiki/cli/adapters/doc_generator.py +++ b/codewiki/cli/adapters/doc_generator.py @@ -145,6 +145,8 @@ def generate(self) -> DocumentationJob: max_token_per_leaf_module=self.config.get("max_token_per_leaf_module", 16000), max_leaf_nodes_per_cluster=self.config.get("max_leaf_nodes_per_cluster", 600), max_depth=self.config.get("max_depth", 2), + request_limit=self.config.get("request_limit", 100), + agent_retries=self.config.get("agent_retries", 3), agent_instructions=self.config.get("agent_instructions"), use_gitignore=self.config.get("use_gitignore", True), prompt_caching=self.config.get("prompt_caching", True), diff --git a/codewiki/cli/commands/config.py b/codewiki/cli/commands/config.py index 17bed6cd..f0e6d18b 100644 --- a/codewiki/cli/commands/config.py +++ b/codewiki/cli/commands/config.py @@ -51,6 +51,12 @@ def config_group(): @click.option( "--max-depth", type=int, help="Maximum depth for hierarchical decomposition (default: 2)" ) +@click.option( + "--request-limit", type=int, help="Maximum model requests per agent run (default: 100)" +) +@click.option( + "--agent-retries", type=int, help="Retries for a failing agent tool call (default: 3)" +) @click.option( "--provider", type=click.Choice( @@ -97,6 +103,8 @@ def config_set( max_token_per_module: Optional[int], max_token_per_leaf_module: Optional[int], max_depth: Optional[int], + request_limit: Optional[int] = None, + agent_retries: Optional[int] = None, provider: Optional[str] = None, aws_region: Optional[str] = None, api_version: Optional[str] = None, @@ -166,6 +174,8 @@ def config_set( max_token_per_module, max_token_per_leaf_module, max_depth, + request_limit is not None, + agent_retries is not None, provider, aws_region, api_version, @@ -226,6 +236,16 @@ def config_set( raise ConfigurationError("max_depth must be a positive integer") validated_data["max_depth"] = max_depth + if request_limit is not None: + if request_limit < 1: + raise ConfigurationError("request_limit must be a positive integer") + validated_data["request_limit"] = request_limit + + if agent_retries is not None: + if agent_retries < 0: + raise ConfigurationError("agent_retries must be zero or a positive integer") + validated_data["agent_retries"] = agent_retries + if provider is not None: validated_data["provider"] = provider @@ -258,6 +278,8 @@ def config_set( max_token_per_module=validated_data.get("max_token_per_module"), max_token_per_leaf_module=validated_data.get("max_token_per_leaf_module"), max_depth=validated_data.get("max_depth"), + request_limit=validated_data.get("request_limit"), + agent_retries=validated_data.get("agent_retries"), provider=validated_data.get("provider"), aws_region=validated_data.get("aws_region"), api_version=validated_data.get("api_version"), @@ -311,6 +333,12 @@ def config_set( if max_depth: click.secho(f"✓ Max depth: {max_depth}", fg="green") + if request_limit: + click.secho(f"✓ Request limit: {request_limit}", fg="green") + + if agent_retries is not None: + click.secho(f"✓ Agent retries: {agent_retries}", fg="green") + if provider: click.secho(f"✓ Provider: {provider}", fg="green") @@ -384,6 +412,8 @@ def config_show(output_json: bool): "max_token_per_module": config.max_token_per_module if config else 36369, "max_token_per_leaf_module": config.max_token_per_leaf_module if config else 16000, "max_depth": config.max_depth if config else 2, + "request_limit": config.request_limit if config else 100, + "agent_retries": config.agent_retries if config else 3, "use_gitignore": config.use_gitignore if config else True, "prompt_caching": config.prompt_caching if config else True, "agent_instructions": config.agent_instructions.to_dict() @@ -450,6 +480,8 @@ def config_show(output_json: bool): click.secho("Decomposition Settings", fg="cyan", bold=True) if config: click.echo(f" Max Depth: {config.max_depth}") + click.echo(f" Request Limit: {config.request_limit}") + click.echo(f" Agent Retries: {config.agent_retries}") click.echo(f" Use Gitignore: {config.use_gitignore}") click.echo() diff --git a/codewiki/cli/commands/generate.py b/codewiki/cli/commands/generate.py index 2e476c86..b0df3a84 100644 --- a/codewiki/cli/commands/generate.py +++ b/codewiki/cli/commands/generate.py @@ -296,6 +296,18 @@ def _find_affected(tree, parent_names=None): default=None, help="Maximum depth for hierarchical decomposition (overrides config)", ) +@click.option( + "--request-limit", + type=click.IntRange(min=1), + default=None, + help="Maximum model requests per agent run (overrides config)", +) +@click.option( + "--agent-retries", + type=click.IntRange(min=0), + default=None, + help="Retries for a failing agent tool call (overrides config)", +) @click.option( "--prompt-caching/--no-prompt-caching", default=None, @@ -398,6 +410,8 @@ def generate_command( max_token_per_module: int | None, max_token_per_leaf_module: int | None, max_depth: int | None, + request_limit: int | None, + agent_retries: int | None, prompt_caching: bool | None, artifacts: bool = True, artifact_token_budget: int = 200_000, @@ -717,6 +731,13 @@ def generate_command( else config.max_token_per_leaf_module, # Max depth setting (runtime override takes precedence) "max_depth": max_depth if max_depth is not None else config.max_depth, + # Agent run limits (runtime override takes precedence) + "request_limit": request_limit + if request_limit is not None + else config.request_limit, + "agent_retries": agent_retries + if agent_retries is not None + else config.agent_retries, # Gitignore setting (runtime override takes precedence) "use_gitignore": use_gitignore if use_gitignore is not None diff --git a/codewiki/cli/config_manager.py b/codewiki/cli/config_manager.py index 3cbadcff..15f9f820 100644 --- a/codewiki/cli/config_manager.py +++ b/codewiki/cli/config_manager.py @@ -136,6 +136,8 @@ def save( max_token_per_module: Optional[int] = None, max_token_per_leaf_module: Optional[int] = None, max_depth: Optional[int] = None, + request_limit: Optional[int] = None, + agent_retries: Optional[int] = None, provider: Optional[str] = None, aws_region: Optional[str] = None, api_version: Optional[str] = None, @@ -157,6 +159,8 @@ def save( max_token_per_module: Maximum tokens per module for clustering max_token_per_leaf_module: Maximum tokens per leaf module max_depth: Maximum depth for hierarchical decomposition + request_limit: Maximum model requests per agent run + agent_retries: Retries for a failing tool call provider: LLM provider type (openai-compatible, anthropic, bedrock, azure-openai) aws_region: AWS region for Bedrock provider api_version: Azure OpenAI API version @@ -205,6 +209,10 @@ def save( self._config.max_token_per_leaf_module = max_token_per_leaf_module if max_depth is not None: self._config.max_depth = max_depth + if request_limit is not None: + self._config.request_limit = request_limit + if agent_retries is not None: + self._config.agent_retries = agent_retries if provider is not None: self._config.provider = provider if aws_region is not None: diff --git a/codewiki/cli/models/config.py b/codewiki/cli/models/config.py index c1cabe07..f0f99cff 100644 --- a/codewiki/cli/models/config.py +++ b/codewiki/cli/models/config.py @@ -130,6 +130,8 @@ class Configuration: max_token_per_module: Maximum tokens per module for clustering (default: 36369) max_token_per_leaf_module: Maximum tokens per leaf module (default: 16000) max_depth: Maximum depth for hierarchical decomposition (default: 2) + request_limit: Maximum model requests per agent run (default: 100) + agent_retries: Retries for a failing tool call (default: 3) use_gitignore: Apply Git ignore rules during repository analysis prompt_caching: Add prompt-cache breakpoints to agentic LLM calls (default: True) agent_instructions: Custom agent instructions for documentation generation @@ -148,6 +150,8 @@ class Configuration: max_token_per_module: int = 36369 max_token_per_leaf_module: int = 16000 max_depth: int = 2 + request_limit: int = 100 + agent_retries: int = 3 use_gitignore: bool = True prompt_caching: bool = True agent_instructions: AgentInstructions = field(default_factory=AgentInstructions) @@ -187,6 +191,8 @@ def to_dict(self) -> dict: "max_token_per_module": self.max_token_per_module, "max_token_per_leaf_module": self.max_token_per_leaf_module, "max_depth": self.max_depth, + "request_limit": self.request_limit, + "agent_retries": self.agent_retries, "use_gitignore": self.use_gitignore, "prompt_caching": self.prompt_caching, "fallback_model": self.fallback_model, @@ -224,6 +230,8 @@ def from_dict(cls, data: dict) -> "Configuration": max_token_per_module=data.get("max_token_per_module", 36369), max_token_per_leaf_module=data.get("max_token_per_leaf_module", 16000), max_depth=data.get("max_depth", 2), + request_limit=data.get("request_limit", 100), + agent_retries=data.get("agent_retries", 3), use_gitignore=data.get("use_gitignore", True), prompt_caching=data.get("prompt_caching", True), agent_instructions=agent_instructions, @@ -300,6 +308,8 @@ def to_backend_config( max_token_per_module=self.max_token_per_module, max_token_per_leaf_module=self.max_token_per_leaf_module, max_depth=self.max_depth, + request_limit=self.request_limit, + agent_retries=self.agent_retries, agent_instructions=final_instructions.to_dict() if final_instructions else None, use_gitignore=self.use_gitignore, prompt_caching=self.prompt_caching, diff --git a/codewiki/src/be/agent_tools/generate_sub_module_documentations.py b/codewiki/src/be/agent_tools/generate_sub_module_documentations.py index 938a34cd..9583bd61 100644 --- a/codewiki/src/be/agent_tools/generate_sub_module_documentations.py +++ b/codewiki/src/be/agent_tools/generate_sub_module_documentations.py @@ -1,13 +1,18 @@ import os from pydantic_ai import RunContext, Tool, Agent +from pydantic_ai.usage import UsageLimits from codewiki.src.be.agent_tools.deps import CodeWikiDeps from codewiki.src.be.module_naming import plan_sub_module_specs from codewiki.src.be.agent_tools.read_code_components import read_code_components_tool from codewiki.src.be.agent_tools.str_replace_editor import str_replace_editor_tool from codewiki.src.be.llm_services import create_fallback_models -from codewiki.src.be.prompt_template import SYSTEM_PROMPT, LEAF_SYSTEM_PROMPT, format_user_prompt +from codewiki.src.be.prompt_template import ( + format_leaf_system_prompt, + format_system_prompt, + format_user_prompt, +) from codewiki.src.be.utils import is_complex_module, count_tokens from codewiki.src.be.cluster_modules import format_potential_core_components @@ -83,9 +88,8 @@ async def generate_sub_module_documentation( model=fallback_models, name=sub_module_name, deps_type=CodeWikiDeps, - system_prompt=SYSTEM_PROMPT.format( - module_name=sub_module_name, custom_instructions=ctx.deps.custom_instructions - ), + system_prompt=format_system_prompt(sub_module_name, ctx.deps.custom_instructions), + retries=ctx.deps.config.agent_retries, tools=[ read_code_components_tool, str_replace_editor_tool, @@ -97,9 +101,10 @@ async def generate_sub_module_documentation( model=fallback_models, name=sub_module_name, deps_type=CodeWikiDeps, - system_prompt=LEAF_SYSTEM_PROMPT.format( - module_name=sub_module_name, custom_instructions=ctx.deps.custom_instructions + system_prompt=format_leaf_system_prompt( + sub_module_name, ctx.deps.custom_instructions ), + retries=ctx.deps.config.agent_retries, tools=[read_code_components_tool, str_replace_editor_tool], ) @@ -117,6 +122,7 @@ async def generate_sub_module_documentation( module_tree=ctx.deps.module_tree, ), deps=ctx.deps, + usage_limits=UsageLimits(request_limit=ctx.deps.config.request_limit), ) # remove the sub-module name from the path to current module and the module tree diff --git a/codewiki/src/be/backend.py b/codewiki/src/be/backend.py index 294a3fc8..d31733f9 100644 --- a/codewiki/src/be/backend.py +++ b/codewiki/src/be/backend.py @@ -89,8 +89,9 @@ def complete( prompt: str, *, model: str | None = None, + system_prompt: str | None = None, ) -> str: - """Single-shot text completion.""" + """Single-shot text completion, with an optional system message.""" @abc.abstractmethod async def run_module_agent( diff --git a/codewiki/src/be/caw_backend.py b/codewiki/src/be/caw_backend.py index 6a2c6970..f10c46a7 100644 --- a/codewiki/src/be/caw_backend.py +++ b/codewiki/src/be/caw_backend.py @@ -246,6 +246,7 @@ def complete( prompt: str, *, model: str | None = None, + system_prompt: str | None = None, ) -> str: # Blocks the calling thread for the lifetime of the claude/codex # subprocess. Callers running this from an async context (e.g. the @@ -256,6 +257,7 @@ def complete( provider=self._caw_provider, model=effective_model, tools=ToolGroup.READER, + system_prompt=system_prompt, ) traj = agent.completion(prompt) self.last_usage = usage_to_dict(getattr(traj, "total_usage", None)) diff --git a/codewiki/src/be/documentation_generator.py b/codewiki/src/be/documentation_generator.py index 3f1a5b9d..fad342a2 100644 --- a/codewiki/src/be/documentation_generator.py +++ b/codewiki/src/be/documentation_generator.py @@ -24,6 +24,7 @@ MODULE_OVERVIEW_PROMPT, REPO_OVERVIEW_ARTIFACT_ADDENDUM, REPO_OVERVIEW_PROMPT, + format_overview_system_prompt, ) from codewiki.src.config import ( FIRST_MODULE_TREE_FILENAME, @@ -354,7 +355,10 @@ async def generate_parent_module_docs( logger.debug(f"Overview prompt for {module_name}: {len(prompt)} chars") try: - parent_docs = self.backend.complete(prompt) + parent_docs = self.backend.complete( + prompt, + system_prompt=format_overview_system_prompt(self.config.get_prompt_addition()), + ) if not parent_docs: raise RuntimeError( f"LLM returned empty content for {module_name} overview " diff --git a/codewiki/src/be/llm_services.py b/codewiki/src/be/llm_services.py index 38fca92b..e677061e 100644 --- a/codewiki/src/be/llm_services.py +++ b/codewiki/src/be/llm_services.py @@ -13,7 +13,7 @@ from openai.types import chat -from pydantic_ai.exceptions import ModelHTTPError +from pydantic_ai.exceptions import ModelAPIError, ModelHTTPError, UnexpectedModelBehavior from pydantic_ai.models.openai import OpenAIChatModel, OpenAIChatModelSettings from pydantic_ai.models.fallback import FallbackModel from pydantic_ai.providers.openai import OpenAIProvider @@ -254,7 +254,10 @@ def create_fallback_models(config: Config) -> FallbackModel: """Create fallback models chain from configuration.""" main = create_main_model(config) fallback = create_fallback_model(config) - return FallbackModel(main, fallback) + # The default fallback_on=(ModelAPIError,) misses UnexpectedModelBehavior, which + # pydantic-ai raises for a 200 response whose body does not parse (e.g. choices=None + # from an OpenAI-compatible gateway); without it such a response skips the fallback. + return FallbackModel(main, fallback, fallback_on=(ModelAPIError, UnexpectedModelBehavior)) def create_openai_client(config: Config) -> OpenAI: @@ -308,7 +311,15 @@ def _extract_content(response, model: str) -> Optional[str]: return content -def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]: +def _messages(prompt: str, system_prompt: str | None) -> list[dict[str, str]]: + messages = [{"role": "system", "content": system_prompt}] if system_prompt else [] + messages.append({"role": "user", "content": prompt}) + return messages + + +def call_llm( + prompt: str, config: Config, model: str = None, system_prompt: str | None = None +) -> Optional[str]: """ Call LLM with the given prompt. @@ -322,6 +333,7 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]: prompt: The prompt to send config: Configuration containing LLM settings model: Model name (defaults to config.main_model) + system_prompt: Optional system message sent before the prompt Returns: LLM response text, or None when the provider returned no content @@ -333,10 +345,10 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]: provider = getattr(config, "provider", "openai-compatible") if provider in ("bedrock", "anthropic"): - return _call_llm_via_litellm(prompt, config, model) + return _call_llm_via_litellm(prompt, config, model, system_prompt) if provider == "azure-openai": - return _call_llm_via_azure(prompt, config, model) + return _call_llm_via_azure(prompt, config, model, system_prompt) # Default: OpenAI-compatible client = create_openai_client(config) @@ -349,7 +361,7 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]: base_kwargs = { "model": model, - "messages": [{"role": "user", "content": prompt}], + "messages": _messages(prompt, system_prompt), } try: @@ -387,7 +399,9 @@ def _is_unsupported_token_param_error(err: BadRequestError, param: str) -> bool: return "unsupported parameter" in msg and param in msg -def _call_llm_via_litellm(prompt: str, config: Config, model: str) -> Optional[str]: +def _call_llm_via_litellm( + prompt: str, config: Config, model: str, system_prompt: str | None = None +) -> Optional[str]: """ Call LLM via litellm for Bedrock/Anthropic providers. @@ -407,14 +421,16 @@ def _call_llm_via_litellm(prompt: str, config: Config, model: str) -> Optional[s response = litellm.completion( model=litellm_model, - messages=[{"role": "user", "content": prompt}], + messages=_messages(prompt, system_prompt), max_tokens=config.max_tokens, api_key=config.llm_api_key if config.provider != "bedrock" else None, ) return _extract_content(response, litellm_model) -def _call_llm_via_azure(prompt: str, config: Config, model: str) -> Optional[str]: +def _call_llm_via_azure( + prompt: str, config: Config, model: str, system_prompt: str | None = None +) -> Optional[str]: """ Call LLM via Azure OpenAI. @@ -436,7 +452,7 @@ def _call_llm_via_azure(prompt: str, config: Config, model: str) -> Optional[str response = client.chat.completions.create( model=deployment, - messages=[{"role": "user", "content": prompt}], + messages=_messages(prompt, system_prompt), max_tokens=config.max_tokens, ) return _extract_content(response, deployment) diff --git a/codewiki/src/be/prompt_template.py b/codewiki/src/be/prompt_template.py index ed9f736b..20429318 100644 --- a/codewiki/src/be/prompt_template.py +++ b/codewiki/src/be/prompt_template.py @@ -645,3 +645,20 @@ def format_leaf_system_prompt(module_name: str, custom_instructions: str | None return LEAF_SYSTEM_PROMPT.format( module_name=module_name, custom_instructions=custom_section ).strip() + + +def format_overview_system_prompt(custom_instructions: str | None = None) -> str | None: + """ + System message for the module / repository overview completions. + + MODULE_OVERVIEW_PROMPT and REPO_OVERVIEW_PROMPT have no custom-instructions + slot, so without this the user's instructions (language, audience, style) + never reach overview pages. Returns None when there is nothing to add. + """ + if not custom_instructions: + return None + return ( + f"\n{custom_instructions}\n\n\n" + "Follow these instructions when writing the overview; they take precedence over " + "conflicting defaults in the request (language, audience, structure, style)." + ) diff --git a/codewiki/src/be/pydantic_ai_backend.py b/codewiki/src/be/pydantic_ai_backend.py index 23ad08e9..3be7dada 100644 --- a/codewiki/src/be/pydantic_ai_backend.py +++ b/codewiki/src/be/pydantic_ai_backend.py @@ -15,6 +15,7 @@ from typing import Any from pydantic_ai import Agent +from pydantic_ai.usage import UsageLimits from codewiki.src.be.agent_tools.deps import CodeWikiDeps from codewiki.src.be.agent_tools.generate_sub_module_documentations import ( @@ -63,9 +64,10 @@ def complete( prompt: str, *, model: str | None = None, + system_prompt: str | None = None, ) -> str: pop_last_usage() - result = call_llm(prompt, self._config, model=model) + result = call_llm(prompt, self._config, model=model, system_prompt=system_prompt) self.last_usage = pop_last_usage() return result @@ -81,9 +83,14 @@ async def run_update_agent( deps_type=CodeWikiDeps, tools=[read_code_components_tool, str_replace_editor_tool], system_prompt=system_prompt, + retries=self._config.agent_retries, ) started = time.time() - result = await agent.run(user_prompt, deps=deps) + result = await agent.run( + user_prompt, + deps=deps, + usage_limits=UsageLimits(request_limit=self._config.request_limit), + ) seconds = time.time() - started usage = _run_usage(result) self.last_usage = usage @@ -127,6 +134,7 @@ async def run_module_agent( generate_sub_module_documentation_tool, ], system_prompt=format_system_prompt(module_name, self._custom_instructions), + retries=config.agent_retries, ) else: agent = Agent( @@ -135,6 +143,7 @@ async def run_module_agent( deps_type=CodeWikiDeps, tools=[read_code_components_tool, str_replace_editor_tool], system_prompt=format_leaf_system_prompt(module_name, self._custom_instructions), + retries=config.agent_retries, ) deps = CodeWikiDeps( @@ -160,6 +169,7 @@ async def run_module_agent( module_tree=deps.module_tree, ), deps=deps, + usage_limits=UsageLimits(request_limit=config.request_limit), ) self.last_usage = _run_usage(result) file_manager.save_json(deps.module_tree, module_tree_path) diff --git a/codewiki/src/config.py b/codewiki/src/config.py index b98208ed..63b98f98 100644 --- a/codewiki/src/config.py +++ b/codewiki/src/config.py @@ -19,6 +19,11 @@ DEFAULT_MAX_TOKENS = 32_768 DEFAULT_MAX_TOKEN_PER_MODULE = 36_369 DEFAULT_MAX_TOKEN_PER_LEAF_MODULE = 4_000 +# pydantic-ai caps each agent run at 50 model requests by default; complex +# modules that read many components and delegate sub-modules can need more. +DEFAULT_REQUEST_LIMIT = 100 +# pydantic-ai retries a failing tool call (and output validation) once by default. +DEFAULT_AGENT_RETRIES = 3 # Super-group the flat top level into architectural subsystems only when it # has more than this many modules; 0 or negative disables the pass. DEFAULT_MIN_MODULES_FOR_SUPER_GROUPING = 3 @@ -91,6 +96,9 @@ class Config: max_token_per_leaf_module: int = DEFAULT_MAX_TOKEN_PER_LEAF_MODULE min_modules_for_super_grouping: int = DEFAULT_MIN_MODULES_FOR_SUPER_GROUPING max_leaf_nodes_per_cluster: int = DEFAULT_MAX_LEAF_NODES_PER_CLUSTER + # Agent run limits (model requests per run, retries per failing tool call) + request_limit: int = DEFAULT_REQUEST_LIMIT + agent_retries: int = DEFAULT_AGENT_RETRIES # Prompt caching for agentic/multi-turn calls (auto-disables per model if # the provider rejects cache_control markers) prompt_caching: bool = True @@ -217,6 +225,8 @@ def from_cli( min_modules_for_super_grouping: int = DEFAULT_MIN_MODULES_FOR_SUPER_GROUPING, max_leaf_nodes_per_cluster: int = DEFAULT_MAX_LEAF_NODES_PER_CLUSTER, max_depth: int = MAX_DEPTH, + request_limit: int = DEFAULT_REQUEST_LIMIT, + agent_retries: int = DEFAULT_AGENT_RETRIES, agent_instructions: dict[str, Any] | None = None, use_gitignore: bool = True, prompt_caching: bool = True, @@ -248,6 +258,8 @@ def from_cli( max_leaf_nodes_per_cluster: Partition clustering inputs into structure-based batches of at most this many leaf nodes max_depth: Maximum depth for hierarchical decomposition + request_limit: Maximum model requests per agent run + agent_retries: Retries for a failing tool call or output validation agent_instructions: Custom agent instructions dict use_gitignore: Whether to apply Git ignore rules prompt_caching: Whether to add prompt-cache breakpoints to agentic calls @@ -281,6 +293,8 @@ def from_cli( max_token_per_leaf_module=max_token_per_leaf_module, min_modules_for_super_grouping=min_modules_for_super_grouping, max_leaf_nodes_per_cluster=max_leaf_nodes_per_cluster, + request_limit=request_limit, + agent_retries=agent_retries, agent_instructions=agent_instructions, use_gitignore=use_gitignore, prompt_caching=prompt_caching, diff --git a/pyproject.toml b/pyproject.toml index 78d72c69..5a3c0716 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -108,6 +108,7 @@ packages = [ "codewiki.src.be.dependency_analyzer.analyzers", "codewiki.src.be.dependency_analyzer.models", "codewiki.src.be.dependency_analyzer.utils", + "codewiki.src.be.updater", "codewiki.src.fe", "codewiki.mcp", "codewiki.mcp.tools" diff --git a/tests/test_agent_limits.py b/tests/test_agent_limits.py new file mode 100644 index 00000000..c99f173d --- /dev/null +++ b/tests/test_agent_limits.py @@ -0,0 +1,121 @@ +"""Agent run limits, fallback triggers, overview instructions and packaging.""" + +import asyncio +import tomllib +from pathlib import Path +from types import SimpleNamespace + +from pydantic_ai.exceptions import ModelAPIError, UnexpectedModelBehavior + +import codewiki.src.be.agent_tools.generate_sub_module_documentations as sub_mod +import codewiki.src.be.llm_services as llm_services +from codewiki.cli.models.config import Configuration +from codewiki.src.be.agent_tools.deps import CodeWikiDeps +from codewiki.src.be.prompt_template import format_overview_system_prompt +from codewiki.src.config import DEFAULT_AGENT_RETRIES, DEFAULT_REQUEST_LIMIT, Config + +ROOT = Path(__file__).resolve().parents[1] + + +def test_configuration_round_trips_agent_limits(): + cfg = Configuration.from_dict( + { + "base_url": "http://x", + "main_model": "m", + "cluster_model": "m", + "request_limit": 250, + "agent_retries": 5, + } + ) + assert (cfg.request_limit, cfg.agent_retries) == (250, 5) + assert Configuration.from_dict(cfg.to_dict()).request_limit == 250 + backend = cfg.to_backend_config(repo_path=".", output_dir="/tmp/out", api_key="k") + assert (backend.request_limit, backend.agent_retries) == (250, 5) + + +def test_backend_config_defaults(): + cfg = Config.from_cli( + repo_path=".", + output_dir="/tmp/out", + llm_base_url="http://x", + llm_api_key="k", + main_model="m", + cluster_model="m", + ) + assert cfg.request_limit == DEFAULT_REQUEST_LIMIT == 100 + assert cfg.agent_retries == DEFAULT_AGENT_RETRIES == 3 + + +def test_fallback_covers_malformed_responses(monkeypatch): + captured = {} + monkeypatch.setattr(llm_services, "create_main_model", lambda config: "main") + monkeypatch.setattr(llm_services, "create_fallback_model", lambda config: "fallback") + monkeypatch.setattr( + llm_services, "FallbackModel", lambda *models, **kwargs: captured.update(kwargs) + ) + llm_services.create_fallback_models(SimpleNamespace()) + # a 200 response with an unparseable body must also switch to the fallback model + assert set(captured["fallback_on"]) == {ModelAPIError, UnexpectedModelBehavior} + + +def test_overview_system_prompt(): + assert format_overview_system_prompt("") is None + assert format_overview_system_prompt(None) is None + text = format_overview_system_prompt("Write in Portuguese.") + assert "\nWrite in Portuguese.\n" in text + + +def test_messages_put_system_first(): + assert llm_services._messages("hi", None) == [{"role": "user", "content": "hi"}] + assert llm_services._messages("hi", "sys") == [ + {"role": "system", "content": "sys"}, + {"role": "user", "content": "hi"}, + ] + + +def test_sub_module_agents_get_limits_and_no_literal_none(tmp_path, monkeypatch): + seen = {} + + class FakeAgent: + def __init__(self, *args, **kwargs): + seen["init"] = kwargs + + async def run(self, prompt, deps, **kwargs): + seen["run"] = kwargs + Path(deps.absolute_docs_path, f"{deps.current_module_name}.md").write_text("# x\n") + return SimpleNamespace(output="ok") + + monkeypatch.setattr(sub_mod, "Agent", FakeAgent) + monkeypatch.setattr(sub_mod, "create_fallback_models", lambda config: None) + deps = CodeWikiDeps( + absolute_docs_path=str(tmp_path), + absolute_repo_path=str(tmp_path), + registry={}, + components={}, + path_to_current_module=[], + current_module_name="root", + module_tree={}, + max_depth=2, + current_depth=1, + config=SimpleNamespace(max_token_per_leaf_module=4000, agent_retries=7, request_limit=321), + custom_instructions=None, + ) + asyncio.run( + sub_mod.generate_sub_module_documentation( + SimpleNamespace(deps=deps), {"billing": ["b.py::fb"]} + ) + ) + assert seen["init"]["retries"] == 7 + assert seen["run"]["usage_limits"].request_limit == 321 + # custom_instructions=None used to be formatted into the prompt as the text "None" + assert not seen["init"]["system_prompt"].endswith("None") + + +def test_every_package_is_listed_in_pyproject(): + listed = set( + tomllib.loads((ROOT / "pyproject.toml").read_text())["tool"]["setuptools"]["packages"] + ) + on_disk = { + ".".join(p.parent.relative_to(ROOT).parts) for p in (ROOT / "codewiki").rglob("__init__.py") + } + assert on_disk - listed == set() diff --git a/tests/test_processing_order_update.py b/tests/test_processing_order_update.py index 8a3a08ce..d3449bf1 100644 --- a/tests/test_processing_order_update.py +++ b/tests/test_processing_order_update.py @@ -78,7 +78,7 @@ async def run_module_agent( Path(working_dir, f"{module_name}.md").write_text(f"# {module_name}\n") return module_tree - def complete(self, prompt, model=None): + def complete(self, prompt, model=None, system_prompt=None): self.complete_calls += 1 return "overview" @@ -86,7 +86,9 @@ def complete(self, prompt, model=None): def _generator(docs_dir: Path) -> tuple[DocumentationGenerator, FakeBackend]: # Bypass __init__: it wires a real LLM backend and dependency analyzer. gen = object.__new__(DocumentationGenerator) - gen.config = SimpleNamespace(docs_dir=str(docs_dir), repo_path=str(docs_dir)) + gen.config = SimpleNamespace( + docs_dir=str(docs_dir), repo_path=str(docs_dir), get_prompt_addition=lambda: "" + ) gen.backend = FakeBackend(docs_dir) return gen, gen.backend diff --git a/tests/test_sub_module_dedupe.py b/tests/test_sub_module_dedupe.py index 83a78a3c..0c0e0728 100644 --- a/tests/test_sub_module_dedupe.py +++ b/tests/test_sub_module_dedupe.py @@ -21,7 +21,7 @@ class FakeAgent: def __init__(self, *args, **kwargs): self.name = kwargs.get("name") - async def run(self, prompt, deps): + async def run(self, prompt, deps, **kwargs): FakeAgent.runs.append(deps.current_module_name) page = f"{deps.absolute_docs_path}/{deps.current_module_name}.md" with open(page, "w", encoding="utf-8") as f: @@ -53,7 +53,7 @@ def _deps(tmp_path) -> CodeWikiDeps: module_tree={}, max_depth=2, current_depth=1, - config=SimpleNamespace(max_token_per_leaf_module=4000), + config=SimpleNamespace(max_token_per_leaf_module=4000, agent_retries=3, request_limit=100), custom_instructions="", ) diff --git a/tests/test_updater_orchestrator.py b/tests/test_updater_orchestrator.py index be1892d4..292e4c05 100644 --- a/tests/test_updater_orchestrator.py +++ b/tests/test_updater_orchestrator.py @@ -29,7 +29,7 @@ def __init__(self, own_verdict="patch"): self.complete_calls = [] self.last_usage = None - def complete(self, prompt, *, model=None): + def complete(self, prompt, *, model=None, system_prompt=None): self.complete_calls.append(prompt[:80]) self.last_usage = {"prompt_tokens": 10, "completion_tokens": 5} return "regenerated overview" From af8d1676e650d48973543cf87abed35b89b30982 Mon Sep 17 00:00:00 2001 From: anhnh2002 Date: Tue, 29 Sep 2026 14:53:01 +0700 Subject: [PATCH 2/2] llm_services: fall back to ModelHTTPError when pydantic-ai has no ModelAPIError requirements.txt pins pydantic-ai 1.0.6, which predates ModelAPIError, so the import failed and CI could not collect any test. --- codewiki/src/be/llm_services.py | 14 ++++++++++---- tests/test_agent_limits.py | 4 ++-- 2 files changed, 12 insertions(+), 6 deletions(-) diff --git a/codewiki/src/be/llm_services.py b/codewiki/src/be/llm_services.py index e677061e..f3b4acb2 100644 --- a/codewiki/src/be/llm_services.py +++ b/codewiki/src/be/llm_services.py @@ -13,7 +13,12 @@ from openai.types import chat -from pydantic_ai.exceptions import ModelAPIError, ModelHTTPError, UnexpectedModelBehavior +from pydantic_ai.exceptions import ModelHTTPError, UnexpectedModelBehavior + +try: # pydantic-ai >= 1.x adds ModelAPIError as the parent of ModelHTTPError + from pydantic_ai.exceptions import ModelAPIError +except ImportError: # older releases (e.g. the pinned 1.0.6) only have ModelHTTPError + ModelAPIError = ModelHTTPError from pydantic_ai.models.openai import OpenAIChatModel, OpenAIChatModelSettings from pydantic_ai.models.fallback import FallbackModel from pydantic_ai.providers.openai import OpenAIProvider @@ -254,9 +259,10 @@ def create_fallback_models(config: Config) -> FallbackModel: """Create fallback models chain from configuration.""" main = create_main_model(config) fallback = create_fallback_model(config) - # The default fallback_on=(ModelAPIError,) misses UnexpectedModelBehavior, which - # pydantic-ai raises for a 200 response whose body does not parse (e.g. choices=None - # from an OpenAI-compatible gateway); without it such a response skips the fallback. + # The default fallback_on (ModelAPIError, or ModelHTTPError on older pydantic-ai) + # misses UnexpectedModelBehavior, which pydantic-ai raises for a 200 response whose + # body does not parse (e.g. choices=None from an OpenAI-compatible gateway); without + # it such a response skips the fallback model. return FallbackModel(main, fallback, fallback_on=(ModelAPIError, UnexpectedModelBehavior)) diff --git a/tests/test_agent_limits.py b/tests/test_agent_limits.py index c99f173d..fffa1fd8 100644 --- a/tests/test_agent_limits.py +++ b/tests/test_agent_limits.py @@ -5,7 +5,7 @@ from pathlib import Path from types import SimpleNamespace -from pydantic_ai.exceptions import ModelAPIError, UnexpectedModelBehavior +from pydantic_ai.exceptions import UnexpectedModelBehavior import codewiki.src.be.agent_tools.generate_sub_module_documentations as sub_mod import codewiki.src.be.llm_services as llm_services @@ -55,7 +55,7 @@ def test_fallback_covers_malformed_responses(monkeypatch): ) llm_services.create_fallback_models(SimpleNamespace()) # a 200 response with an unparseable body must also switch to the fallback model - assert set(captured["fallback_on"]) == {ModelAPIError, UnexpectedModelBehavior} + assert set(captured["fallback_on"]) == {llm_services.ModelAPIError, UnexpectedModelBehavior} def test_overview_system_prompt():