diff --git a/.gitignore b/.gitignore
index d36f2b71..ecd28ba0 100644
--- a/.gitignore
+++ b/.gitignore
@@ -47,6 +47,7 @@ tests/*
!tests/test_branding.py
!tests/test_artifact_analyzer.py
!tests/test_scala_analyzer.py
+!tests/test_agent_limits.py
# Jupyter
*.ipynb
diff --git a/codewiki/cli/adapters/doc_generator.py b/codewiki/cli/adapters/doc_generator.py
index 92c066dd..07f425fc 100644
--- a/codewiki/cli/adapters/doc_generator.py
+++ b/codewiki/cli/adapters/doc_generator.py
@@ -145,6 +145,8 @@ def generate(self) -> DocumentationJob:
max_token_per_leaf_module=self.config.get("max_token_per_leaf_module", 16000),
max_leaf_nodes_per_cluster=self.config.get("max_leaf_nodes_per_cluster", 600),
max_depth=self.config.get("max_depth", 2),
+ request_limit=self.config.get("request_limit", 100),
+ agent_retries=self.config.get("agent_retries", 3),
agent_instructions=self.config.get("agent_instructions"),
use_gitignore=self.config.get("use_gitignore", True),
prompt_caching=self.config.get("prompt_caching", True),
diff --git a/codewiki/cli/commands/config.py b/codewiki/cli/commands/config.py
index 17bed6cd..f0e6d18b 100644
--- a/codewiki/cli/commands/config.py
+++ b/codewiki/cli/commands/config.py
@@ -51,6 +51,12 @@ def config_group():
@click.option(
"--max-depth", type=int, help="Maximum depth for hierarchical decomposition (default: 2)"
)
+@click.option(
+ "--request-limit", type=int, help="Maximum model requests per agent run (default: 100)"
+)
+@click.option(
+ "--agent-retries", type=int, help="Retries for a failing agent tool call (default: 3)"
+)
@click.option(
"--provider",
type=click.Choice(
@@ -97,6 +103,8 @@ def config_set(
max_token_per_module: Optional[int],
max_token_per_leaf_module: Optional[int],
max_depth: Optional[int],
+ request_limit: Optional[int] = None,
+ agent_retries: Optional[int] = None,
provider: Optional[str] = None,
aws_region: Optional[str] = None,
api_version: Optional[str] = None,
@@ -166,6 +174,8 @@ def config_set(
max_token_per_module,
max_token_per_leaf_module,
max_depth,
+ request_limit is not None,
+ agent_retries is not None,
provider,
aws_region,
api_version,
@@ -226,6 +236,16 @@ def config_set(
raise ConfigurationError("max_depth must be a positive integer")
validated_data["max_depth"] = max_depth
+ if request_limit is not None:
+ if request_limit < 1:
+ raise ConfigurationError("request_limit must be a positive integer")
+ validated_data["request_limit"] = request_limit
+
+ if agent_retries is not None:
+ if agent_retries < 0:
+ raise ConfigurationError("agent_retries must be zero or a positive integer")
+ validated_data["agent_retries"] = agent_retries
+
if provider is not None:
validated_data["provider"] = provider
@@ -258,6 +278,8 @@ def config_set(
max_token_per_module=validated_data.get("max_token_per_module"),
max_token_per_leaf_module=validated_data.get("max_token_per_leaf_module"),
max_depth=validated_data.get("max_depth"),
+ request_limit=validated_data.get("request_limit"),
+ agent_retries=validated_data.get("agent_retries"),
provider=validated_data.get("provider"),
aws_region=validated_data.get("aws_region"),
api_version=validated_data.get("api_version"),
@@ -311,6 +333,12 @@ def config_set(
if max_depth:
click.secho(f"✓ Max depth: {max_depth}", fg="green")
+ if request_limit:
+ click.secho(f"✓ Request limit: {request_limit}", fg="green")
+
+ if agent_retries is not None:
+ click.secho(f"✓ Agent retries: {agent_retries}", fg="green")
+
if provider:
click.secho(f"✓ Provider: {provider}", fg="green")
@@ -384,6 +412,8 @@ def config_show(output_json: bool):
"max_token_per_module": config.max_token_per_module if config else 36369,
"max_token_per_leaf_module": config.max_token_per_leaf_module if config else 16000,
"max_depth": config.max_depth if config else 2,
+ "request_limit": config.request_limit if config else 100,
+ "agent_retries": config.agent_retries if config else 3,
"use_gitignore": config.use_gitignore if config else True,
"prompt_caching": config.prompt_caching if config else True,
"agent_instructions": config.agent_instructions.to_dict()
@@ -450,6 +480,8 @@ def config_show(output_json: bool):
click.secho("Decomposition Settings", fg="cyan", bold=True)
if config:
click.echo(f" Max Depth: {config.max_depth}")
+ click.echo(f" Request Limit: {config.request_limit}")
+ click.echo(f" Agent Retries: {config.agent_retries}")
click.echo(f" Use Gitignore: {config.use_gitignore}")
click.echo()
diff --git a/codewiki/cli/commands/generate.py b/codewiki/cli/commands/generate.py
index 2e476c86..b0df3a84 100644
--- a/codewiki/cli/commands/generate.py
+++ b/codewiki/cli/commands/generate.py
@@ -296,6 +296,18 @@ def _find_affected(tree, parent_names=None):
default=None,
help="Maximum depth for hierarchical decomposition (overrides config)",
)
+@click.option(
+ "--request-limit",
+ type=click.IntRange(min=1),
+ default=None,
+ help="Maximum model requests per agent run (overrides config)",
+)
+@click.option(
+ "--agent-retries",
+ type=click.IntRange(min=0),
+ default=None,
+ help="Retries for a failing agent tool call (overrides config)",
+)
@click.option(
"--prompt-caching/--no-prompt-caching",
default=None,
@@ -398,6 +410,8 @@ def generate_command(
max_token_per_module: int | None,
max_token_per_leaf_module: int | None,
max_depth: int | None,
+ request_limit: int | None,
+ agent_retries: int | None,
prompt_caching: bool | None,
artifacts: bool = True,
artifact_token_budget: int = 200_000,
@@ -717,6 +731,13 @@ def generate_command(
else config.max_token_per_leaf_module,
# Max depth setting (runtime override takes precedence)
"max_depth": max_depth if max_depth is not None else config.max_depth,
+ # Agent run limits (runtime override takes precedence)
+ "request_limit": request_limit
+ if request_limit is not None
+ else config.request_limit,
+ "agent_retries": agent_retries
+ if agent_retries is not None
+ else config.agent_retries,
# Gitignore setting (runtime override takes precedence)
"use_gitignore": use_gitignore
if use_gitignore is not None
diff --git a/codewiki/cli/config_manager.py b/codewiki/cli/config_manager.py
index 3cbadcff..15f9f820 100644
--- a/codewiki/cli/config_manager.py
+++ b/codewiki/cli/config_manager.py
@@ -136,6 +136,8 @@ def save(
max_token_per_module: Optional[int] = None,
max_token_per_leaf_module: Optional[int] = None,
max_depth: Optional[int] = None,
+ request_limit: Optional[int] = None,
+ agent_retries: Optional[int] = None,
provider: Optional[str] = None,
aws_region: Optional[str] = None,
api_version: Optional[str] = None,
@@ -157,6 +159,8 @@ def save(
max_token_per_module: Maximum tokens per module for clustering
max_token_per_leaf_module: Maximum tokens per leaf module
max_depth: Maximum depth for hierarchical decomposition
+ request_limit: Maximum model requests per agent run
+ agent_retries: Retries for a failing tool call
provider: LLM provider type (openai-compatible, anthropic, bedrock, azure-openai)
aws_region: AWS region for Bedrock provider
api_version: Azure OpenAI API version
@@ -205,6 +209,10 @@ def save(
self._config.max_token_per_leaf_module = max_token_per_leaf_module
if max_depth is not None:
self._config.max_depth = max_depth
+ if request_limit is not None:
+ self._config.request_limit = request_limit
+ if agent_retries is not None:
+ self._config.agent_retries = agent_retries
if provider is not None:
self._config.provider = provider
if aws_region is not None:
diff --git a/codewiki/cli/models/config.py b/codewiki/cli/models/config.py
index c1cabe07..f0f99cff 100644
--- a/codewiki/cli/models/config.py
+++ b/codewiki/cli/models/config.py
@@ -130,6 +130,8 @@ class Configuration:
max_token_per_module: Maximum tokens per module for clustering (default: 36369)
max_token_per_leaf_module: Maximum tokens per leaf module (default: 16000)
max_depth: Maximum depth for hierarchical decomposition (default: 2)
+ request_limit: Maximum model requests per agent run (default: 100)
+ agent_retries: Retries for a failing tool call (default: 3)
use_gitignore: Apply Git ignore rules during repository analysis
prompt_caching: Add prompt-cache breakpoints to agentic LLM calls (default: True)
agent_instructions: Custom agent instructions for documentation generation
@@ -148,6 +150,8 @@ class Configuration:
max_token_per_module: int = 36369
max_token_per_leaf_module: int = 16000
max_depth: int = 2
+ request_limit: int = 100
+ agent_retries: int = 3
use_gitignore: bool = True
prompt_caching: bool = True
agent_instructions: AgentInstructions = field(default_factory=AgentInstructions)
@@ -187,6 +191,8 @@ def to_dict(self) -> dict:
"max_token_per_module": self.max_token_per_module,
"max_token_per_leaf_module": self.max_token_per_leaf_module,
"max_depth": self.max_depth,
+ "request_limit": self.request_limit,
+ "agent_retries": self.agent_retries,
"use_gitignore": self.use_gitignore,
"prompt_caching": self.prompt_caching,
"fallback_model": self.fallback_model,
@@ -224,6 +230,8 @@ def from_dict(cls, data: dict) -> "Configuration":
max_token_per_module=data.get("max_token_per_module", 36369),
max_token_per_leaf_module=data.get("max_token_per_leaf_module", 16000),
max_depth=data.get("max_depth", 2),
+ request_limit=data.get("request_limit", 100),
+ agent_retries=data.get("agent_retries", 3),
use_gitignore=data.get("use_gitignore", True),
prompt_caching=data.get("prompt_caching", True),
agent_instructions=agent_instructions,
@@ -300,6 +308,8 @@ def to_backend_config(
max_token_per_module=self.max_token_per_module,
max_token_per_leaf_module=self.max_token_per_leaf_module,
max_depth=self.max_depth,
+ request_limit=self.request_limit,
+ agent_retries=self.agent_retries,
agent_instructions=final_instructions.to_dict() if final_instructions else None,
use_gitignore=self.use_gitignore,
prompt_caching=self.prompt_caching,
diff --git a/codewiki/src/be/agent_tools/generate_sub_module_documentations.py b/codewiki/src/be/agent_tools/generate_sub_module_documentations.py
index 938a34cd..9583bd61 100644
--- a/codewiki/src/be/agent_tools/generate_sub_module_documentations.py
+++ b/codewiki/src/be/agent_tools/generate_sub_module_documentations.py
@@ -1,13 +1,18 @@
import os
from pydantic_ai import RunContext, Tool, Agent
+from pydantic_ai.usage import UsageLimits
from codewiki.src.be.agent_tools.deps import CodeWikiDeps
from codewiki.src.be.module_naming import plan_sub_module_specs
from codewiki.src.be.agent_tools.read_code_components import read_code_components_tool
from codewiki.src.be.agent_tools.str_replace_editor import str_replace_editor_tool
from codewiki.src.be.llm_services import create_fallback_models
-from codewiki.src.be.prompt_template import SYSTEM_PROMPT, LEAF_SYSTEM_PROMPT, format_user_prompt
+from codewiki.src.be.prompt_template import (
+ format_leaf_system_prompt,
+ format_system_prompt,
+ format_user_prompt,
+)
from codewiki.src.be.utils import is_complex_module, count_tokens
from codewiki.src.be.cluster_modules import format_potential_core_components
@@ -83,9 +88,8 @@ async def generate_sub_module_documentation(
model=fallback_models,
name=sub_module_name,
deps_type=CodeWikiDeps,
- system_prompt=SYSTEM_PROMPT.format(
- module_name=sub_module_name, custom_instructions=ctx.deps.custom_instructions
- ),
+ system_prompt=format_system_prompt(sub_module_name, ctx.deps.custom_instructions),
+ retries=ctx.deps.config.agent_retries,
tools=[
read_code_components_tool,
str_replace_editor_tool,
@@ -97,9 +101,10 @@ async def generate_sub_module_documentation(
model=fallback_models,
name=sub_module_name,
deps_type=CodeWikiDeps,
- system_prompt=LEAF_SYSTEM_PROMPT.format(
- module_name=sub_module_name, custom_instructions=ctx.deps.custom_instructions
+ system_prompt=format_leaf_system_prompt(
+ sub_module_name, ctx.deps.custom_instructions
),
+ retries=ctx.deps.config.agent_retries,
tools=[read_code_components_tool, str_replace_editor_tool],
)
@@ -117,6 +122,7 @@ async def generate_sub_module_documentation(
module_tree=ctx.deps.module_tree,
),
deps=ctx.deps,
+ usage_limits=UsageLimits(request_limit=ctx.deps.config.request_limit),
)
# remove the sub-module name from the path to current module and the module tree
diff --git a/codewiki/src/be/backend.py b/codewiki/src/be/backend.py
index 294a3fc8..d31733f9 100644
--- a/codewiki/src/be/backend.py
+++ b/codewiki/src/be/backend.py
@@ -89,8 +89,9 @@ def complete(
prompt: str,
*,
model: str | None = None,
+ system_prompt: str | None = None,
) -> str:
- """Single-shot text completion."""
+ """Single-shot text completion, with an optional system message."""
@abc.abstractmethod
async def run_module_agent(
diff --git a/codewiki/src/be/caw_backend.py b/codewiki/src/be/caw_backend.py
index 6a2c6970..f10c46a7 100644
--- a/codewiki/src/be/caw_backend.py
+++ b/codewiki/src/be/caw_backend.py
@@ -246,6 +246,7 @@ def complete(
prompt: str,
*,
model: str | None = None,
+ system_prompt: str | None = None,
) -> str:
# Blocks the calling thread for the lifetime of the claude/codex
# subprocess. Callers running this from an async context (e.g. the
@@ -256,6 +257,7 @@ def complete(
provider=self._caw_provider,
model=effective_model,
tools=ToolGroup.READER,
+ system_prompt=system_prompt,
)
traj = agent.completion(prompt)
self.last_usage = usage_to_dict(getattr(traj, "total_usage", None))
diff --git a/codewiki/src/be/documentation_generator.py b/codewiki/src/be/documentation_generator.py
index 3f1a5b9d..fad342a2 100644
--- a/codewiki/src/be/documentation_generator.py
+++ b/codewiki/src/be/documentation_generator.py
@@ -24,6 +24,7 @@
MODULE_OVERVIEW_PROMPT,
REPO_OVERVIEW_ARTIFACT_ADDENDUM,
REPO_OVERVIEW_PROMPT,
+ format_overview_system_prompt,
)
from codewiki.src.config import (
FIRST_MODULE_TREE_FILENAME,
@@ -354,7 +355,10 @@ async def generate_parent_module_docs(
logger.debug(f"Overview prompt for {module_name}: {len(prompt)} chars")
try:
- parent_docs = self.backend.complete(prompt)
+ parent_docs = self.backend.complete(
+ prompt,
+ system_prompt=format_overview_system_prompt(self.config.get_prompt_addition()),
+ )
if not parent_docs:
raise RuntimeError(
f"LLM returned empty content for {module_name} overview "
diff --git a/codewiki/src/be/llm_services.py b/codewiki/src/be/llm_services.py
index 38fca92b..f3b4acb2 100644
--- a/codewiki/src/be/llm_services.py
+++ b/codewiki/src/be/llm_services.py
@@ -13,7 +13,12 @@
from openai.types import chat
-from pydantic_ai.exceptions import ModelHTTPError
+from pydantic_ai.exceptions import ModelHTTPError, UnexpectedModelBehavior
+
+try: # pydantic-ai >= 1.x adds ModelAPIError as the parent of ModelHTTPError
+ from pydantic_ai.exceptions import ModelAPIError
+except ImportError: # older releases (e.g. the pinned 1.0.6) only have ModelHTTPError
+ ModelAPIError = ModelHTTPError
from pydantic_ai.models.openai import OpenAIChatModel, OpenAIChatModelSettings
from pydantic_ai.models.fallback import FallbackModel
from pydantic_ai.providers.openai import OpenAIProvider
@@ -254,7 +259,11 @@ def create_fallback_models(config: Config) -> FallbackModel:
"""Create fallback models chain from configuration."""
main = create_main_model(config)
fallback = create_fallback_model(config)
- return FallbackModel(main, fallback)
+ # The default fallback_on (ModelAPIError, or ModelHTTPError on older pydantic-ai)
+ # misses UnexpectedModelBehavior, which pydantic-ai raises for a 200 response whose
+ # body does not parse (e.g. choices=None from an OpenAI-compatible gateway); without
+ # it such a response skips the fallback model.
+ return FallbackModel(main, fallback, fallback_on=(ModelAPIError, UnexpectedModelBehavior))
def create_openai_client(config: Config) -> OpenAI:
@@ -308,7 +317,15 @@ def _extract_content(response, model: str) -> Optional[str]:
return content
-def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]:
+def _messages(prompt: str, system_prompt: str | None) -> list[dict[str, str]]:
+ messages = [{"role": "system", "content": system_prompt}] if system_prompt else []
+ messages.append({"role": "user", "content": prompt})
+ return messages
+
+
+def call_llm(
+ prompt: str, config: Config, model: str = None, system_prompt: str | None = None
+) -> Optional[str]:
"""
Call LLM with the given prompt.
@@ -322,6 +339,7 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]:
prompt: The prompt to send
config: Configuration containing LLM settings
model: Model name (defaults to config.main_model)
+ system_prompt: Optional system message sent before the prompt
Returns:
LLM response text, or None when the provider returned no content
@@ -333,10 +351,10 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]:
provider = getattr(config, "provider", "openai-compatible")
if provider in ("bedrock", "anthropic"):
- return _call_llm_via_litellm(prompt, config, model)
+ return _call_llm_via_litellm(prompt, config, model, system_prompt)
if provider == "azure-openai":
- return _call_llm_via_azure(prompt, config, model)
+ return _call_llm_via_azure(prompt, config, model, system_prompt)
# Default: OpenAI-compatible
client = create_openai_client(config)
@@ -349,7 +367,7 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]:
base_kwargs = {
"model": model,
- "messages": [{"role": "user", "content": prompt}],
+ "messages": _messages(prompt, system_prompt),
}
try:
@@ -387,7 +405,9 @@ def _is_unsupported_token_param_error(err: BadRequestError, param: str) -> bool:
return "unsupported parameter" in msg and param in msg
-def _call_llm_via_litellm(prompt: str, config: Config, model: str) -> Optional[str]:
+def _call_llm_via_litellm(
+ prompt: str, config: Config, model: str, system_prompt: str | None = None
+) -> Optional[str]:
"""
Call LLM via litellm for Bedrock/Anthropic providers.
@@ -407,14 +427,16 @@ def _call_llm_via_litellm(prompt: str, config: Config, model: str) -> Optional[s
response = litellm.completion(
model=litellm_model,
- messages=[{"role": "user", "content": prompt}],
+ messages=_messages(prompt, system_prompt),
max_tokens=config.max_tokens,
api_key=config.llm_api_key if config.provider != "bedrock" else None,
)
return _extract_content(response, litellm_model)
-def _call_llm_via_azure(prompt: str, config: Config, model: str) -> Optional[str]:
+def _call_llm_via_azure(
+ prompt: str, config: Config, model: str, system_prompt: str | None = None
+) -> Optional[str]:
"""
Call LLM via Azure OpenAI.
@@ -436,7 +458,7 @@ def _call_llm_via_azure(prompt: str, config: Config, model: str) -> Optional[str
response = client.chat.completions.create(
model=deployment,
- messages=[{"role": "user", "content": prompt}],
+ messages=_messages(prompt, system_prompt),
max_tokens=config.max_tokens,
)
return _extract_content(response, deployment)
diff --git a/codewiki/src/be/prompt_template.py b/codewiki/src/be/prompt_template.py
index ed9f736b..20429318 100644
--- a/codewiki/src/be/prompt_template.py
+++ b/codewiki/src/be/prompt_template.py
@@ -645,3 +645,20 @@ def format_leaf_system_prompt(module_name: str, custom_instructions: str | None
return LEAF_SYSTEM_PROMPT.format(
module_name=module_name, custom_instructions=custom_section
).strip()
+
+
+def format_overview_system_prompt(custom_instructions: str | None = None) -> str | None:
+ """
+ System message for the module / repository overview completions.
+
+ MODULE_OVERVIEW_PROMPT and REPO_OVERVIEW_PROMPT have no custom-instructions
+ slot, so without this the user's instructions (language, audience, style)
+ never reach overview pages. Returns None when there is nothing to add.
+ """
+ if not custom_instructions:
+ return None
+ return (
+ f"\n{custom_instructions}\n\n\n"
+ "Follow these instructions when writing the overview; they take precedence over "
+ "conflicting defaults in the request (language, audience, structure, style)."
+ )
diff --git a/codewiki/src/be/pydantic_ai_backend.py b/codewiki/src/be/pydantic_ai_backend.py
index 23ad08e9..3be7dada 100644
--- a/codewiki/src/be/pydantic_ai_backend.py
+++ b/codewiki/src/be/pydantic_ai_backend.py
@@ -15,6 +15,7 @@
from typing import Any
from pydantic_ai import Agent
+from pydantic_ai.usage import UsageLimits
from codewiki.src.be.agent_tools.deps import CodeWikiDeps
from codewiki.src.be.agent_tools.generate_sub_module_documentations import (
@@ -63,9 +64,10 @@ def complete(
prompt: str,
*,
model: str | None = None,
+ system_prompt: str | None = None,
) -> str:
pop_last_usage()
- result = call_llm(prompt, self._config, model=model)
+ result = call_llm(prompt, self._config, model=model, system_prompt=system_prompt)
self.last_usage = pop_last_usage()
return result
@@ -81,9 +83,14 @@ async def run_update_agent(
deps_type=CodeWikiDeps,
tools=[read_code_components_tool, str_replace_editor_tool],
system_prompt=system_prompt,
+ retries=self._config.agent_retries,
)
started = time.time()
- result = await agent.run(user_prompt, deps=deps)
+ result = await agent.run(
+ user_prompt,
+ deps=deps,
+ usage_limits=UsageLimits(request_limit=self._config.request_limit),
+ )
seconds = time.time() - started
usage = _run_usage(result)
self.last_usage = usage
@@ -127,6 +134,7 @@ async def run_module_agent(
generate_sub_module_documentation_tool,
],
system_prompt=format_system_prompt(module_name, self._custom_instructions),
+ retries=config.agent_retries,
)
else:
agent = Agent(
@@ -135,6 +143,7 @@ async def run_module_agent(
deps_type=CodeWikiDeps,
tools=[read_code_components_tool, str_replace_editor_tool],
system_prompt=format_leaf_system_prompt(module_name, self._custom_instructions),
+ retries=config.agent_retries,
)
deps = CodeWikiDeps(
@@ -160,6 +169,7 @@ async def run_module_agent(
module_tree=deps.module_tree,
),
deps=deps,
+ usage_limits=UsageLimits(request_limit=config.request_limit),
)
self.last_usage = _run_usage(result)
file_manager.save_json(deps.module_tree, module_tree_path)
diff --git a/codewiki/src/config.py b/codewiki/src/config.py
index b98208ed..63b98f98 100644
--- a/codewiki/src/config.py
+++ b/codewiki/src/config.py
@@ -19,6 +19,11 @@
DEFAULT_MAX_TOKENS = 32_768
DEFAULT_MAX_TOKEN_PER_MODULE = 36_369
DEFAULT_MAX_TOKEN_PER_LEAF_MODULE = 4_000
+# pydantic-ai caps each agent run at 50 model requests by default; complex
+# modules that read many components and delegate sub-modules can need more.
+DEFAULT_REQUEST_LIMIT = 100
+# pydantic-ai retries a failing tool call (and output validation) once by default.
+DEFAULT_AGENT_RETRIES = 3
# Super-group the flat top level into architectural subsystems only when it
# has more than this many modules; 0 or negative disables the pass.
DEFAULT_MIN_MODULES_FOR_SUPER_GROUPING = 3
@@ -91,6 +96,9 @@ class Config:
max_token_per_leaf_module: int = DEFAULT_MAX_TOKEN_PER_LEAF_MODULE
min_modules_for_super_grouping: int = DEFAULT_MIN_MODULES_FOR_SUPER_GROUPING
max_leaf_nodes_per_cluster: int = DEFAULT_MAX_LEAF_NODES_PER_CLUSTER
+ # Agent run limits (model requests per run, retries per failing tool call)
+ request_limit: int = DEFAULT_REQUEST_LIMIT
+ agent_retries: int = DEFAULT_AGENT_RETRIES
# Prompt caching for agentic/multi-turn calls (auto-disables per model if
# the provider rejects cache_control markers)
prompt_caching: bool = True
@@ -217,6 +225,8 @@ def from_cli(
min_modules_for_super_grouping: int = DEFAULT_MIN_MODULES_FOR_SUPER_GROUPING,
max_leaf_nodes_per_cluster: int = DEFAULT_MAX_LEAF_NODES_PER_CLUSTER,
max_depth: int = MAX_DEPTH,
+ request_limit: int = DEFAULT_REQUEST_LIMIT,
+ agent_retries: int = DEFAULT_AGENT_RETRIES,
agent_instructions: dict[str, Any] | None = None,
use_gitignore: bool = True,
prompt_caching: bool = True,
@@ -248,6 +258,8 @@ def from_cli(
max_leaf_nodes_per_cluster: Partition clustering inputs into
structure-based batches of at most this many leaf nodes
max_depth: Maximum depth for hierarchical decomposition
+ request_limit: Maximum model requests per agent run
+ agent_retries: Retries for a failing tool call or output validation
agent_instructions: Custom agent instructions dict
use_gitignore: Whether to apply Git ignore rules
prompt_caching: Whether to add prompt-cache breakpoints to agentic calls
@@ -281,6 +293,8 @@ def from_cli(
max_token_per_leaf_module=max_token_per_leaf_module,
min_modules_for_super_grouping=min_modules_for_super_grouping,
max_leaf_nodes_per_cluster=max_leaf_nodes_per_cluster,
+ request_limit=request_limit,
+ agent_retries=agent_retries,
agent_instructions=agent_instructions,
use_gitignore=use_gitignore,
prompt_caching=prompt_caching,
diff --git a/pyproject.toml b/pyproject.toml
index 78d72c69..5a3c0716 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -108,6 +108,7 @@ packages = [
"codewiki.src.be.dependency_analyzer.analyzers",
"codewiki.src.be.dependency_analyzer.models",
"codewiki.src.be.dependency_analyzer.utils",
+ "codewiki.src.be.updater",
"codewiki.src.fe",
"codewiki.mcp",
"codewiki.mcp.tools"
diff --git a/tests/test_agent_limits.py b/tests/test_agent_limits.py
new file mode 100644
index 00000000..fffa1fd8
--- /dev/null
+++ b/tests/test_agent_limits.py
@@ -0,0 +1,121 @@
+"""Agent run limits, fallback triggers, overview instructions and packaging."""
+
+import asyncio
+import tomllib
+from pathlib import Path
+from types import SimpleNamespace
+
+from pydantic_ai.exceptions import UnexpectedModelBehavior
+
+import codewiki.src.be.agent_tools.generate_sub_module_documentations as sub_mod
+import codewiki.src.be.llm_services as llm_services
+from codewiki.cli.models.config import Configuration
+from codewiki.src.be.agent_tools.deps import CodeWikiDeps
+from codewiki.src.be.prompt_template import format_overview_system_prompt
+from codewiki.src.config import DEFAULT_AGENT_RETRIES, DEFAULT_REQUEST_LIMIT, Config
+
+ROOT = Path(__file__).resolve().parents[1]
+
+
+def test_configuration_round_trips_agent_limits():
+ cfg = Configuration.from_dict(
+ {
+ "base_url": "http://x",
+ "main_model": "m",
+ "cluster_model": "m",
+ "request_limit": 250,
+ "agent_retries": 5,
+ }
+ )
+ assert (cfg.request_limit, cfg.agent_retries) == (250, 5)
+ assert Configuration.from_dict(cfg.to_dict()).request_limit == 250
+ backend = cfg.to_backend_config(repo_path=".", output_dir="/tmp/out", api_key="k")
+ assert (backend.request_limit, backend.agent_retries) == (250, 5)
+
+
+def test_backend_config_defaults():
+ cfg = Config.from_cli(
+ repo_path=".",
+ output_dir="/tmp/out",
+ llm_base_url="http://x",
+ llm_api_key="k",
+ main_model="m",
+ cluster_model="m",
+ )
+ assert cfg.request_limit == DEFAULT_REQUEST_LIMIT == 100
+ assert cfg.agent_retries == DEFAULT_AGENT_RETRIES == 3
+
+
+def test_fallback_covers_malformed_responses(monkeypatch):
+ captured = {}
+ monkeypatch.setattr(llm_services, "create_main_model", lambda config: "main")
+ monkeypatch.setattr(llm_services, "create_fallback_model", lambda config: "fallback")
+ monkeypatch.setattr(
+ llm_services, "FallbackModel", lambda *models, **kwargs: captured.update(kwargs)
+ )
+ llm_services.create_fallback_models(SimpleNamespace())
+ # a 200 response with an unparseable body must also switch to the fallback model
+ assert set(captured["fallback_on"]) == {llm_services.ModelAPIError, UnexpectedModelBehavior}
+
+
+def test_overview_system_prompt():
+ assert format_overview_system_prompt("") is None
+ assert format_overview_system_prompt(None) is None
+ text = format_overview_system_prompt("Write in Portuguese.")
+ assert "\nWrite in Portuguese.\n" in text
+
+
+def test_messages_put_system_first():
+ assert llm_services._messages("hi", None) == [{"role": "user", "content": "hi"}]
+ assert llm_services._messages("hi", "sys") == [
+ {"role": "system", "content": "sys"},
+ {"role": "user", "content": "hi"},
+ ]
+
+
+def test_sub_module_agents_get_limits_and_no_literal_none(tmp_path, monkeypatch):
+ seen = {}
+
+ class FakeAgent:
+ def __init__(self, *args, **kwargs):
+ seen["init"] = kwargs
+
+ async def run(self, prompt, deps, **kwargs):
+ seen["run"] = kwargs
+ Path(deps.absolute_docs_path, f"{deps.current_module_name}.md").write_text("# x\n")
+ return SimpleNamespace(output="ok")
+
+ monkeypatch.setattr(sub_mod, "Agent", FakeAgent)
+ monkeypatch.setattr(sub_mod, "create_fallback_models", lambda config: None)
+ deps = CodeWikiDeps(
+ absolute_docs_path=str(tmp_path),
+ absolute_repo_path=str(tmp_path),
+ registry={},
+ components={},
+ path_to_current_module=[],
+ current_module_name="root",
+ module_tree={},
+ max_depth=2,
+ current_depth=1,
+ config=SimpleNamespace(max_token_per_leaf_module=4000, agent_retries=7, request_limit=321),
+ custom_instructions=None,
+ )
+ asyncio.run(
+ sub_mod.generate_sub_module_documentation(
+ SimpleNamespace(deps=deps), {"billing": ["b.py::fb"]}
+ )
+ )
+ assert seen["init"]["retries"] == 7
+ assert seen["run"]["usage_limits"].request_limit == 321
+ # custom_instructions=None used to be formatted into the prompt as the text "None"
+ assert not seen["init"]["system_prompt"].endswith("None")
+
+
+def test_every_package_is_listed_in_pyproject():
+ listed = set(
+ tomllib.loads((ROOT / "pyproject.toml").read_text())["tool"]["setuptools"]["packages"]
+ )
+ on_disk = {
+ ".".join(p.parent.relative_to(ROOT).parts) for p in (ROOT / "codewiki").rglob("__init__.py")
+ }
+ assert on_disk - listed == set()
diff --git a/tests/test_processing_order_update.py b/tests/test_processing_order_update.py
index 8a3a08ce..d3449bf1 100644
--- a/tests/test_processing_order_update.py
+++ b/tests/test_processing_order_update.py
@@ -78,7 +78,7 @@ async def run_module_agent(
Path(working_dir, f"{module_name}.md").write_text(f"# {module_name}\n")
return module_tree
- def complete(self, prompt, model=None):
+ def complete(self, prompt, model=None, system_prompt=None):
self.complete_calls += 1
return "overview"
@@ -86,7 +86,9 @@ def complete(self, prompt, model=None):
def _generator(docs_dir: Path) -> tuple[DocumentationGenerator, FakeBackend]:
# Bypass __init__: it wires a real LLM backend and dependency analyzer.
gen = object.__new__(DocumentationGenerator)
- gen.config = SimpleNamespace(docs_dir=str(docs_dir), repo_path=str(docs_dir))
+ gen.config = SimpleNamespace(
+ docs_dir=str(docs_dir), repo_path=str(docs_dir), get_prompt_addition=lambda: ""
+ )
gen.backend = FakeBackend(docs_dir)
return gen, gen.backend
diff --git a/tests/test_sub_module_dedupe.py b/tests/test_sub_module_dedupe.py
index 83a78a3c..0c0e0728 100644
--- a/tests/test_sub_module_dedupe.py
+++ b/tests/test_sub_module_dedupe.py
@@ -21,7 +21,7 @@ class FakeAgent:
def __init__(self, *args, **kwargs):
self.name = kwargs.get("name")
- async def run(self, prompt, deps):
+ async def run(self, prompt, deps, **kwargs):
FakeAgent.runs.append(deps.current_module_name)
page = f"{deps.absolute_docs_path}/{deps.current_module_name}.md"
with open(page, "w", encoding="utf-8") as f:
@@ -53,7 +53,7 @@ def _deps(tmp_path) -> CodeWikiDeps:
module_tree={},
max_depth=2,
current_depth=1,
- config=SimpleNamespace(max_token_per_leaf_module=4000),
+ config=SimpleNamespace(max_token_per_leaf_module=4000, agent_retries=3, request_limit=100),
custom_instructions="",
)
diff --git a/tests/test_updater_orchestrator.py b/tests/test_updater_orchestrator.py
index be1892d4..292e4c05 100644
--- a/tests/test_updater_orchestrator.py
+++ b/tests/test_updater_orchestrator.py
@@ -29,7 +29,7 @@ def __init__(self, own_verdict="patch"):
self.complete_calls = []
self.last_usage = None
- def complete(self, prompt, *, model=None):
+ def complete(self, prompt, *, model=None, system_prompt=None):
self.complete_calls.append(prompt[:80])
self.last_usage = {"prompt_tokens": 10, "completion_tokens": 5}
return "regenerated overview"