diff --git a/codewiki/src/be/backend.py b/codewiki/src/be/backend.py
index 294a3fc8..6fc6693d 100644
--- a/codewiki/src/be/backend.py
+++ b/codewiki/src/be/backend.py
@@ -89,6 +89,7 @@ def complete(
prompt: str,
*,
model: str | None = None,
+ system_prompt: str | None = None,
) -> str:
"""Single-shot text completion."""
diff --git a/codewiki/src/be/caw_backend.py b/codewiki/src/be/caw_backend.py
index 6a2c6970..5555e38e 100644
--- a/codewiki/src/be/caw_backend.py
+++ b/codewiki/src/be/caw_backend.py
@@ -246,18 +246,23 @@ def complete(
prompt: str,
*,
model: str | None = None,
+ system_prompt: str | None = None,
) -> str:
# Blocks the calling thread for the lifetime of the claude/codex
# subprocess. Callers running this from an async context (e.g. the
# documentation_generator) accept this — there is no concurrent work
# to do while clustering is in flight anyway.
effective_model = model or self._model
+ # CawAgent has no separate system-role slot exposed here; best-effort
+ # fold it into the prompt rather than silently drop it (see
+ # llm_services.call_llm for the primary, tested path).
+ effective_prompt = f"{system_prompt}\n\n{prompt}" if system_prompt else prompt
agent = CawAgent(
provider=self._caw_provider,
model=effective_model,
tools=ToolGroup.READER,
)
- traj = agent.completion(prompt)
+ traj = agent.completion(effective_prompt)
self.last_usage = usage_to_dict(getattr(traj, "total_usage", None))
return traj.result
diff --git a/codewiki/src/be/documentation_generator.py b/codewiki/src/be/documentation_generator.py
index 3f1a5b9d..e8442e0d 100644
--- a/codewiki/src/be/documentation_generator.py
+++ b/codewiki/src/be/documentation_generator.py
@@ -351,10 +351,26 @@ async def generate_parent_module_docs(
prompt += "\n\n" + REPO_OVERVIEW_ARTIFACT_ADDENDUM.format(
artifact_index=artifact_index
)
+ # Unlike SYSTEM_PROMPT/LEAF_SYSTEM_PROMPT, neither MODULE_OVERVIEW_PROMPT nor
+ # REPO_OVERVIEW_PROMPT has a {custom_instructions} slot — without this, `--instructions`
+ # (language, audience, forbidden diagram styles, link rules) silently never reaches
+ # module-parent or repo-root overview generation. Passing it as a trailing addition to
+ # the single user-role prompt measurably failed to change output (confirmed empirically:
+ # still English, still the default structure). Passing it as its own system-role message
+ # is the fix that actually works for SYSTEM_PROMPT/LEAF_SYSTEM_PROMPT (both put
+ # `--instructions` in the Agent's system_prompt, not the user turn), so mirror that here.
+ prompt_addition = self.config.get_prompt_addition()
+ overview_system_prompt = (
+ f"\n{prompt_addition}\n\n\n"
+ "These instructions override any conflicting default in the user message below "
+ "(audience, language, structure, diagram style, and link format all included)."
+ if prompt_addition
+ else None
+ )
logger.debug(f"Overview prompt for {module_name}: {len(prompt)} chars")
try:
- parent_docs = self.backend.complete(prompt)
+ parent_docs = self.backend.complete(prompt, system_prompt=overview_system_prompt)
if not parent_docs:
raise RuntimeError(
f"LLM returned empty content for {module_name} overview "
diff --git a/codewiki/src/be/llm_services.py b/codewiki/src/be/llm_services.py
index 38fca92b..c6749a64 100644
--- a/codewiki/src/be/llm_services.py
+++ b/codewiki/src/be/llm_services.py
@@ -308,7 +308,9 @@ def _extract_content(response, model: str) -> Optional[str]:
return content
-def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]:
+def call_llm(
+ prompt: str, config: Config, model: str = None, system_prompt: str | None = None
+) -> Optional[str]:
"""
Call LLM with the given prompt.
@@ -322,6 +324,13 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]:
prompt: The prompt to send
config: Configuration containing LLM settings
model: Model name (defaults to config.main_model)
+ system_prompt: Optional system-role message. `--instructions` reaching a model
+ only as trailing text appended to a single user-role prompt (as opposed to a
+ dedicated system message, or a `` block inside an actual
+ agent system_prompt) gets materially weaker adherence — confirmed empirically
+ on `MODULE_OVERVIEW_PROMPT`/`REPO_OVERVIEW_PROMPT` output (still English,
+ still the un-instructed default structure, after the instructions were
+ appended to the prompt string).
Returns:
LLM response text, or None when the provider returned no content
@@ -333,10 +342,10 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]:
provider = getattr(config, "provider", "openai-compatible")
if provider in ("bedrock", "anthropic"):
- return _call_llm_via_litellm(prompt, config, model)
+ return _call_llm_via_litellm(prompt, config, model, system_prompt=system_prompt)
if provider == "azure-openai":
- return _call_llm_via_azure(prompt, config, model)
+ return _call_llm_via_azure(prompt, config, model, system_prompt=system_prompt)
# Default: OpenAI-compatible
client = create_openai_client(config)
@@ -347,9 +356,14 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]:
primary_key = "max_completion_tokens" if use_completion_tokens else "max_tokens"
fallback_key = "max_tokens" if use_completion_tokens else "max_completion_tokens"
+ messages = []
+ if system_prompt:
+ messages.append({"role": "system", "content": system_prompt})
+ messages.append({"role": "user", "content": prompt})
+
base_kwargs = {
"model": model,
- "messages": [{"role": "user", "content": prompt}],
+ "messages": messages,
}
try:
@@ -387,7 +401,9 @@ def _is_unsupported_token_param_error(err: BadRequestError, param: str) -> bool:
return "unsupported parameter" in msg and param in msg
-def _call_llm_via_litellm(prompt: str, config: Config, model: str) -> Optional[str]:
+def _call_llm_via_litellm(
+ prompt: str, config: Config, model: str, system_prompt: str | None = None
+) -> Optional[str]:
"""
Call LLM via litellm for Bedrock/Anthropic providers.
@@ -405,16 +421,23 @@ def _call_llm_via_litellm(prompt: str, config: Config, model: str) -> Optional[s
elif config.provider == "anthropic":
logger.debug("Calling Anthropic model %s via litellm", litellm_model)
+ messages = []
+ if system_prompt:
+ messages.append({"role": "system", "content": system_prompt})
+ messages.append({"role": "user", "content": prompt})
+
response = litellm.completion(
model=litellm_model,
- messages=[{"role": "user", "content": prompt}],
+ messages=messages,
max_tokens=config.max_tokens,
api_key=config.llm_api_key if config.provider != "bedrock" else None,
)
return _extract_content(response, litellm_model)
-def _call_llm_via_azure(prompt: str, config: Config, model: str) -> Optional[str]:
+def _call_llm_via_azure(
+ prompt: str, config: Config, model: str, system_prompt: str | None = None
+) -> Optional[str]:
"""
Call LLM via Azure OpenAI.
@@ -434,9 +457,14 @@ def _call_llm_via_azure(prompt: str, config: Config, model: str) -> Optional[str
"Calling Azure OpenAI deployment %s (api_version=%s)", deployment, config.api_version
)
+ messages = []
+ if system_prompt:
+ messages.append({"role": "system", "content": system_prompt})
+ messages.append({"role": "user", "content": prompt})
+
response = client.chat.completions.create(
model=deployment,
- messages=[{"role": "user", "content": prompt}],
+ messages=messages,
max_tokens=config.max_tokens,
)
return _extract_content(response, deployment)
diff --git a/codewiki/src/be/pydantic_ai_backend.py b/codewiki/src/be/pydantic_ai_backend.py
index 23ad08e9..f5065686 100644
--- a/codewiki/src/be/pydantic_ai_backend.py
+++ b/codewiki/src/be/pydantic_ai_backend.py
@@ -63,9 +63,10 @@ def complete(
prompt: str,
*,
model: str | None = None,
+ system_prompt: str | None = None,
) -> str:
pop_last_usage()
- result = call_llm(prompt, self._config, model=model)
+ result = call_llm(prompt, self._config, model=model, system_prompt=system_prompt)
self.last_usage = pop_last_usage()
return result
diff --git a/tests/test_processing_order_update.py b/tests/test_processing_order_update.py
index 8a3a08ce..d3449bf1 100644
--- a/tests/test_processing_order_update.py
+++ b/tests/test_processing_order_update.py
@@ -78,7 +78,7 @@ async def run_module_agent(
Path(working_dir, f"{module_name}.md").write_text(f"# {module_name}\n")
return module_tree
- def complete(self, prompt, model=None):
+ def complete(self, prompt, model=None, system_prompt=None):
self.complete_calls += 1
return "overview"
@@ -86,7 +86,9 @@ def complete(self, prompt, model=None):
def _generator(docs_dir: Path) -> tuple[DocumentationGenerator, FakeBackend]:
# Bypass __init__: it wires a real LLM backend and dependency analyzer.
gen = object.__new__(DocumentationGenerator)
- gen.config = SimpleNamespace(docs_dir=str(docs_dir), repo_path=str(docs_dir))
+ gen.config = SimpleNamespace(
+ docs_dir=str(docs_dir), repo_path=str(docs_dir), get_prompt_addition=lambda: ""
+ )
gen.backend = FakeBackend(docs_dir)
return gen, gen.backend
diff --git a/tests/test_updater_orchestrator.py b/tests/test_updater_orchestrator.py
index 6bfb2a3f..65a8fd61 100644
--- a/tests/test_updater_orchestrator.py
+++ b/tests/test_updater_orchestrator.py
@@ -29,7 +29,7 @@ def __init__(self, own_verdict="patch"):
self.complete_calls = []
self.last_usage = None
- def complete(self, prompt, *, model=None):
+ def complete(self, prompt, *, model=None, system_prompt=None):
self.complete_calls.append(prompt[:80])
self.last_usage = {"prompt_tokens": 10, "completion_tokens": 5}
return "regenerated overview"