Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,7 @@ tests/*
!tests/test_branding.py
!tests/test_artifact_analyzer.py
!tests/test_scala_analyzer.py
!tests/test_agent_limits.py

# Jupyter
*.ipynb
Expand Down
2 changes: 2 additions & 0 deletions codewiki/cli/adapters/doc_generator.py
Original file line number Diff line number Diff line change
Expand Up @@ -145,6 +145,8 @@ def generate(self) -> DocumentationJob:
max_token_per_leaf_module=self.config.get("max_token_per_leaf_module", 16000),
max_leaf_nodes_per_cluster=self.config.get("max_leaf_nodes_per_cluster", 600),
max_depth=self.config.get("max_depth", 2),
request_limit=self.config.get("request_limit", 100),
agent_retries=self.config.get("agent_retries", 3),
agent_instructions=self.config.get("agent_instructions"),
use_gitignore=self.config.get("use_gitignore", True),
prompt_caching=self.config.get("prompt_caching", True),
Expand Down
32 changes: 32 additions & 0 deletions codewiki/cli/commands/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,12 @@ def config_group():
@click.option(
"--max-depth", type=int, help="Maximum depth for hierarchical decomposition (default: 2)"
)
@click.option(
"--request-limit", type=int, help="Maximum model requests per agent run (default: 100)"
)
@click.option(
"--agent-retries", type=int, help="Retries for a failing agent tool call (default: 3)"
)
@click.option(
"--provider",
type=click.Choice(
Expand Down Expand Up @@ -97,6 +103,8 @@ def config_set(
max_token_per_module: Optional[int],
max_token_per_leaf_module: Optional[int],
max_depth: Optional[int],
request_limit: Optional[int] = None,
agent_retries: Optional[int] = None,
provider: Optional[str] = None,
aws_region: Optional[str] = None,
api_version: Optional[str] = None,
Expand Down Expand Up @@ -166,6 +174,8 @@ def config_set(
max_token_per_module,
max_token_per_leaf_module,
max_depth,
request_limit is not None,
agent_retries is not None,
provider,
aws_region,
api_version,
Expand Down Expand Up @@ -226,6 +236,16 @@ def config_set(
raise ConfigurationError("max_depth must be a positive integer")
validated_data["max_depth"] = max_depth

if request_limit is not None:
if request_limit < 1:
raise ConfigurationError("request_limit must be a positive integer")
validated_data["request_limit"] = request_limit

if agent_retries is not None:
if agent_retries < 0:
raise ConfigurationError("agent_retries must be zero or a positive integer")
validated_data["agent_retries"] = agent_retries

if provider is not None:
validated_data["provider"] = provider

Expand Down Expand Up @@ -258,6 +278,8 @@ def config_set(
max_token_per_module=validated_data.get("max_token_per_module"),
max_token_per_leaf_module=validated_data.get("max_token_per_leaf_module"),
max_depth=validated_data.get("max_depth"),
request_limit=validated_data.get("request_limit"),
agent_retries=validated_data.get("agent_retries"),
provider=validated_data.get("provider"),
aws_region=validated_data.get("aws_region"),
api_version=validated_data.get("api_version"),
Expand Down Expand Up @@ -311,6 +333,12 @@ def config_set(
if max_depth:
click.secho(f"✓ Max depth: {max_depth}", fg="green")

if request_limit:
click.secho(f"✓ Request limit: {request_limit}", fg="green")

if agent_retries is not None:
click.secho(f"✓ Agent retries: {agent_retries}", fg="green")

if provider:
click.secho(f"✓ Provider: {provider}", fg="green")

Expand Down Expand Up @@ -384,6 +412,8 @@ def config_show(output_json: bool):
"max_token_per_module": config.max_token_per_module if config else 36369,
"max_token_per_leaf_module": config.max_token_per_leaf_module if config else 16000,
"max_depth": config.max_depth if config else 2,
"request_limit": config.request_limit if config else 100,
"agent_retries": config.agent_retries if config else 3,
"use_gitignore": config.use_gitignore if config else True,
"prompt_caching": config.prompt_caching if config else True,
"agent_instructions": config.agent_instructions.to_dict()
Expand Down Expand Up @@ -450,6 +480,8 @@ def config_show(output_json: bool):
click.secho("Decomposition Settings", fg="cyan", bold=True)
if config:
click.echo(f" Max Depth: {config.max_depth}")
click.echo(f" Request Limit: {config.request_limit}")
click.echo(f" Agent Retries: {config.agent_retries}")
click.echo(f" Use Gitignore: {config.use_gitignore}")

click.echo()
Expand Down
21 changes: 21 additions & 0 deletions codewiki/cli/commands/generate.py
Original file line number Diff line number Diff line change
Expand Up @@ -296,6 +296,18 @@ def _find_affected(tree, parent_names=None):
default=None,
help="Maximum depth for hierarchical decomposition (overrides config)",
)
@click.option(
"--request-limit",
type=click.IntRange(min=1),
default=None,
help="Maximum model requests per agent run (overrides config)",
)
@click.option(
"--agent-retries",
type=click.IntRange(min=0),
default=None,
help="Retries for a failing agent tool call (overrides config)",
)
@click.option(
"--prompt-caching/--no-prompt-caching",
default=None,
Expand Down Expand Up @@ -398,6 +410,8 @@ def generate_command(
max_token_per_module: int | None,
max_token_per_leaf_module: int | None,
max_depth: int | None,
request_limit: int | None,
agent_retries: int | None,
prompt_caching: bool | None,
artifacts: bool = True,
artifact_token_budget: int = 200_000,
Expand Down Expand Up @@ -717,6 +731,13 @@ def generate_command(
else config.max_token_per_leaf_module,
# Max depth setting (runtime override takes precedence)
"max_depth": max_depth if max_depth is not None else config.max_depth,
# Agent run limits (runtime override takes precedence)
"request_limit": request_limit
if request_limit is not None
else config.request_limit,
"agent_retries": agent_retries
if agent_retries is not None
else config.agent_retries,
# Gitignore setting (runtime override takes precedence)
"use_gitignore": use_gitignore
if use_gitignore is not None
Expand Down
8 changes: 8 additions & 0 deletions codewiki/cli/config_manager.py
Original file line number Diff line number Diff line change
Expand Up @@ -136,6 +136,8 @@ def save(
max_token_per_module: Optional[int] = None,
max_token_per_leaf_module: Optional[int] = None,
max_depth: Optional[int] = None,
request_limit: Optional[int] = None,
agent_retries: Optional[int] = None,
provider: Optional[str] = None,
aws_region: Optional[str] = None,
api_version: Optional[str] = None,
Expand All @@ -157,6 +159,8 @@ def save(
max_token_per_module: Maximum tokens per module for clustering
max_token_per_leaf_module: Maximum tokens per leaf module
max_depth: Maximum depth for hierarchical decomposition
request_limit: Maximum model requests per agent run
agent_retries: Retries for a failing tool call
provider: LLM provider type (openai-compatible, anthropic, bedrock, azure-openai)
aws_region: AWS region for Bedrock provider
api_version: Azure OpenAI API version
Expand Down Expand Up @@ -205,6 +209,10 @@ def save(
self._config.max_token_per_leaf_module = max_token_per_leaf_module
if max_depth is not None:
self._config.max_depth = max_depth
if request_limit is not None:
self._config.request_limit = request_limit
if agent_retries is not None:
self._config.agent_retries = agent_retries
if provider is not None:
self._config.provider = provider
if aws_region is not None:
Expand Down
10 changes: 10 additions & 0 deletions codewiki/cli/models/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -130,6 +130,8 @@ class Configuration:
max_token_per_module: Maximum tokens per module for clustering (default: 36369)
max_token_per_leaf_module: Maximum tokens per leaf module (default: 16000)
max_depth: Maximum depth for hierarchical decomposition (default: 2)
request_limit: Maximum model requests per agent run (default: 100)
agent_retries: Retries for a failing tool call (default: 3)
use_gitignore: Apply Git ignore rules during repository analysis
prompt_caching: Add prompt-cache breakpoints to agentic LLM calls (default: True)
agent_instructions: Custom agent instructions for documentation generation
Expand All @@ -148,6 +150,8 @@ class Configuration:
max_token_per_module: int = 36369
max_token_per_leaf_module: int = 16000
max_depth: int = 2
request_limit: int = 100
agent_retries: int = 3
use_gitignore: bool = True
prompt_caching: bool = True
agent_instructions: AgentInstructions = field(default_factory=AgentInstructions)
Expand Down Expand Up @@ -187,6 +191,8 @@ def to_dict(self) -> dict:
"max_token_per_module": self.max_token_per_module,
"max_token_per_leaf_module": self.max_token_per_leaf_module,
"max_depth": self.max_depth,
"request_limit": self.request_limit,
"agent_retries": self.agent_retries,
"use_gitignore": self.use_gitignore,
"prompt_caching": self.prompt_caching,
"fallback_model": self.fallback_model,
Expand Down Expand Up @@ -224,6 +230,8 @@ def from_dict(cls, data: dict) -> "Configuration":
max_token_per_module=data.get("max_token_per_module", 36369),
max_token_per_leaf_module=data.get("max_token_per_leaf_module", 16000),
max_depth=data.get("max_depth", 2),
request_limit=data.get("request_limit", 100),
agent_retries=data.get("agent_retries", 3),
use_gitignore=data.get("use_gitignore", True),
prompt_caching=data.get("prompt_caching", True),
agent_instructions=agent_instructions,
Expand Down Expand Up @@ -300,6 +308,8 @@ def to_backend_config(
max_token_per_module=self.max_token_per_module,
max_token_per_leaf_module=self.max_token_per_leaf_module,
max_depth=self.max_depth,
request_limit=self.request_limit,
agent_retries=self.agent_retries,
agent_instructions=final_instructions.to_dict() if final_instructions else None,
use_gitignore=self.use_gitignore,
prompt_caching=self.prompt_caching,
Expand Down
18 changes: 12 additions & 6 deletions codewiki/src/be/agent_tools/generate_sub_module_documentations.py
Original file line number Diff line number Diff line change
@@ -1,13 +1,18 @@
import os

from pydantic_ai import RunContext, Tool, Agent
from pydantic_ai.usage import UsageLimits

from codewiki.src.be.agent_tools.deps import CodeWikiDeps
from codewiki.src.be.module_naming import plan_sub_module_specs
from codewiki.src.be.agent_tools.read_code_components import read_code_components_tool
from codewiki.src.be.agent_tools.str_replace_editor import str_replace_editor_tool
from codewiki.src.be.llm_services import create_fallback_models
from codewiki.src.be.prompt_template import SYSTEM_PROMPT, LEAF_SYSTEM_PROMPT, format_user_prompt
from codewiki.src.be.prompt_template import (
format_leaf_system_prompt,
format_system_prompt,
format_user_prompt,
)
from codewiki.src.be.utils import is_complex_module, count_tokens
from codewiki.src.be.cluster_modules import format_potential_core_components

Expand Down Expand Up @@ -83,9 +88,8 @@ async def generate_sub_module_documentation(
model=fallback_models,
name=sub_module_name,
deps_type=CodeWikiDeps,
system_prompt=SYSTEM_PROMPT.format(
module_name=sub_module_name, custom_instructions=ctx.deps.custom_instructions
),
system_prompt=format_system_prompt(sub_module_name, ctx.deps.custom_instructions),
retries=ctx.deps.config.agent_retries,
tools=[
read_code_components_tool,
str_replace_editor_tool,
Expand All @@ -97,9 +101,10 @@ async def generate_sub_module_documentation(
model=fallback_models,
name=sub_module_name,
deps_type=CodeWikiDeps,
system_prompt=LEAF_SYSTEM_PROMPT.format(
module_name=sub_module_name, custom_instructions=ctx.deps.custom_instructions
system_prompt=format_leaf_system_prompt(
sub_module_name, ctx.deps.custom_instructions
),
retries=ctx.deps.config.agent_retries,
tools=[read_code_components_tool, str_replace_editor_tool],
)

Expand All @@ -117,6 +122,7 @@ async def generate_sub_module_documentation(
module_tree=ctx.deps.module_tree,
),
deps=ctx.deps,
usage_limits=UsageLimits(request_limit=ctx.deps.config.request_limit),
)

# remove the sub-module name from the path to current module and the module tree
Expand Down
3 changes: 2 additions & 1 deletion codewiki/src/be/backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -89,8 +89,9 @@ def complete(
prompt: str,
*,
model: str | None = None,
system_prompt: str | None = None,
) -> str:
"""Single-shot text completion."""
"""Single-shot text completion, with an optional system message."""

@abc.abstractmethod
async def run_module_agent(
Expand Down
2 changes: 2 additions & 0 deletions codewiki/src/be/caw_backend.py
Original file line number Diff line number Diff line change
Expand Up @@ -246,6 +246,7 @@ def complete(
prompt: str,
*,
model: str | None = None,
system_prompt: str | None = None,
) -> str:
# Blocks the calling thread for the lifetime of the claude/codex
# subprocess. Callers running this from an async context (e.g. the
Expand All @@ -256,6 +257,7 @@ def complete(
provider=self._caw_provider,
model=effective_model,
tools=ToolGroup.READER,
system_prompt=system_prompt,
)
traj = agent.completion(prompt)
self.last_usage = usage_to_dict(getattr(traj, "total_usage", None))
Expand Down
6 changes: 5 additions & 1 deletion codewiki/src/be/documentation_generator.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,7 @@
MODULE_OVERVIEW_PROMPT,
REPO_OVERVIEW_ARTIFACT_ADDENDUM,
REPO_OVERVIEW_PROMPT,
format_overview_system_prompt,
)
from codewiki.src.config import (
FIRST_MODULE_TREE_FILENAME,
Expand Down Expand Up @@ -354,7 +355,10 @@ async def generate_parent_module_docs(
logger.debug(f"Overview prompt for {module_name}: {len(prompt)} chars")

try:
parent_docs = self.backend.complete(prompt)
parent_docs = self.backend.complete(
prompt,
system_prompt=format_overview_system_prompt(self.config.get_prompt_addition()),
)
if not parent_docs:
raise RuntimeError(
f"LLM returned empty content for {module_name} overview "
Expand Down
Loading
Loading