diff --git a/codewiki/src/be/backend.py b/codewiki/src/be/backend.py index 294a3fc8..6fc6693d 100644 --- a/codewiki/src/be/backend.py +++ b/codewiki/src/be/backend.py @@ -89,6 +89,7 @@ def complete( prompt: str, *, model: str | None = None, + system_prompt: str | None = None, ) -> str: """Single-shot text completion.""" diff --git a/codewiki/src/be/caw_backend.py b/codewiki/src/be/caw_backend.py index 6a2c6970..5555e38e 100644 --- a/codewiki/src/be/caw_backend.py +++ b/codewiki/src/be/caw_backend.py @@ -246,18 +246,23 @@ def complete( prompt: str, *, model: str | None = None, + system_prompt: str | None = None, ) -> str: # Blocks the calling thread for the lifetime of the claude/codex # subprocess. Callers running this from an async context (e.g. the # documentation_generator) accept this — there is no concurrent work # to do while clustering is in flight anyway. effective_model = model or self._model + # CawAgent has no separate system-role slot exposed here; best-effort + # fold it into the prompt rather than silently drop it (see + # llm_services.call_llm for the primary, tested path). + effective_prompt = f"{system_prompt}\n\n{prompt}" if system_prompt else prompt agent = CawAgent( provider=self._caw_provider, model=effective_model, tools=ToolGroup.READER, ) - traj = agent.completion(prompt) + traj = agent.completion(effective_prompt) self.last_usage = usage_to_dict(getattr(traj, "total_usage", None)) return traj.result diff --git a/codewiki/src/be/documentation_generator.py b/codewiki/src/be/documentation_generator.py index 3f1a5b9d..e8442e0d 100644 --- a/codewiki/src/be/documentation_generator.py +++ b/codewiki/src/be/documentation_generator.py @@ -351,10 +351,26 @@ async def generate_parent_module_docs( prompt += "\n\n" + REPO_OVERVIEW_ARTIFACT_ADDENDUM.format( artifact_index=artifact_index ) + # Unlike SYSTEM_PROMPT/LEAF_SYSTEM_PROMPT, neither MODULE_OVERVIEW_PROMPT nor + # REPO_OVERVIEW_PROMPT has a {custom_instructions} slot — without this, `--instructions` + # (language, audience, forbidden diagram styles, link rules) silently never reaches + # module-parent or repo-root overview generation. Passing it as a trailing addition to + # the single user-role prompt measurably failed to change output (confirmed empirically: + # still English, still the default structure). Passing it as its own system-role message + # is the fix that actually works for SYSTEM_PROMPT/LEAF_SYSTEM_PROMPT (both put + # `--instructions` in the Agent's system_prompt, not the user turn), so mirror that here. + prompt_addition = self.config.get_prompt_addition() + overview_system_prompt = ( + f"\n{prompt_addition}\n\n\n" + "These instructions override any conflicting default in the user message below " + "(audience, language, structure, diagram style, and link format all included)." + if prompt_addition + else None + ) logger.debug(f"Overview prompt for {module_name}: {len(prompt)} chars") try: - parent_docs = self.backend.complete(prompt) + parent_docs = self.backend.complete(prompt, system_prompt=overview_system_prompt) if not parent_docs: raise RuntimeError( f"LLM returned empty content for {module_name} overview " diff --git a/codewiki/src/be/llm_services.py b/codewiki/src/be/llm_services.py index 38fca92b..c6749a64 100644 --- a/codewiki/src/be/llm_services.py +++ b/codewiki/src/be/llm_services.py @@ -308,7 +308,9 @@ def _extract_content(response, model: str) -> Optional[str]: return content -def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]: +def call_llm( + prompt: str, config: Config, model: str = None, system_prompt: str | None = None +) -> Optional[str]: """ Call LLM with the given prompt. @@ -322,6 +324,13 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]: prompt: The prompt to send config: Configuration containing LLM settings model: Model name (defaults to config.main_model) + system_prompt: Optional system-role message. `--instructions` reaching a model + only as trailing text appended to a single user-role prompt (as opposed to a + dedicated system message, or a `` block inside an actual + agent system_prompt) gets materially weaker adherence — confirmed empirically + on `MODULE_OVERVIEW_PROMPT`/`REPO_OVERVIEW_PROMPT` output (still English, + still the un-instructed default structure, after the instructions were + appended to the prompt string). Returns: LLM response text, or None when the provider returned no content @@ -333,10 +342,10 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]: provider = getattr(config, "provider", "openai-compatible") if provider in ("bedrock", "anthropic"): - return _call_llm_via_litellm(prompt, config, model) + return _call_llm_via_litellm(prompt, config, model, system_prompt=system_prompt) if provider == "azure-openai": - return _call_llm_via_azure(prompt, config, model) + return _call_llm_via_azure(prompt, config, model, system_prompt=system_prompt) # Default: OpenAI-compatible client = create_openai_client(config) @@ -347,9 +356,14 @@ def call_llm(prompt: str, config: Config, model: str = None) -> Optional[str]: primary_key = "max_completion_tokens" if use_completion_tokens else "max_tokens" fallback_key = "max_tokens" if use_completion_tokens else "max_completion_tokens" + messages = [] + if system_prompt: + messages.append({"role": "system", "content": system_prompt}) + messages.append({"role": "user", "content": prompt}) + base_kwargs = { "model": model, - "messages": [{"role": "user", "content": prompt}], + "messages": messages, } try: @@ -387,7 +401,9 @@ def _is_unsupported_token_param_error(err: BadRequestError, param: str) -> bool: return "unsupported parameter" in msg and param in msg -def _call_llm_via_litellm(prompt: str, config: Config, model: str) -> Optional[str]: +def _call_llm_via_litellm( + prompt: str, config: Config, model: str, system_prompt: str | None = None +) -> Optional[str]: """ Call LLM via litellm for Bedrock/Anthropic providers. @@ -405,16 +421,23 @@ def _call_llm_via_litellm(prompt: str, config: Config, model: str) -> Optional[s elif config.provider == "anthropic": logger.debug("Calling Anthropic model %s via litellm", litellm_model) + messages = [] + if system_prompt: + messages.append({"role": "system", "content": system_prompt}) + messages.append({"role": "user", "content": prompt}) + response = litellm.completion( model=litellm_model, - messages=[{"role": "user", "content": prompt}], + messages=messages, max_tokens=config.max_tokens, api_key=config.llm_api_key if config.provider != "bedrock" else None, ) return _extract_content(response, litellm_model) -def _call_llm_via_azure(prompt: str, config: Config, model: str) -> Optional[str]: +def _call_llm_via_azure( + prompt: str, config: Config, model: str, system_prompt: str | None = None +) -> Optional[str]: """ Call LLM via Azure OpenAI. @@ -434,9 +457,14 @@ def _call_llm_via_azure(prompt: str, config: Config, model: str) -> Optional[str "Calling Azure OpenAI deployment %s (api_version=%s)", deployment, config.api_version ) + messages = [] + if system_prompt: + messages.append({"role": "system", "content": system_prompt}) + messages.append({"role": "user", "content": prompt}) + response = client.chat.completions.create( model=deployment, - messages=[{"role": "user", "content": prompt}], + messages=messages, max_tokens=config.max_tokens, ) return _extract_content(response, deployment) diff --git a/codewiki/src/be/pydantic_ai_backend.py b/codewiki/src/be/pydantic_ai_backend.py index 23ad08e9..f5065686 100644 --- a/codewiki/src/be/pydantic_ai_backend.py +++ b/codewiki/src/be/pydantic_ai_backend.py @@ -63,9 +63,10 @@ def complete( prompt: str, *, model: str | None = None, + system_prompt: str | None = None, ) -> str: pop_last_usage() - result = call_llm(prompt, self._config, model=model) + result = call_llm(prompt, self._config, model=model, system_prompt=system_prompt) self.last_usage = pop_last_usage() return result diff --git a/tests/test_processing_order_update.py b/tests/test_processing_order_update.py index 8a3a08ce..d3449bf1 100644 --- a/tests/test_processing_order_update.py +++ b/tests/test_processing_order_update.py @@ -78,7 +78,7 @@ async def run_module_agent( Path(working_dir, f"{module_name}.md").write_text(f"# {module_name}\n") return module_tree - def complete(self, prompt, model=None): + def complete(self, prompt, model=None, system_prompt=None): self.complete_calls += 1 return "overview" @@ -86,7 +86,9 @@ def complete(self, prompt, model=None): def _generator(docs_dir: Path) -> tuple[DocumentationGenerator, FakeBackend]: # Bypass __init__: it wires a real LLM backend and dependency analyzer. gen = object.__new__(DocumentationGenerator) - gen.config = SimpleNamespace(docs_dir=str(docs_dir), repo_path=str(docs_dir)) + gen.config = SimpleNamespace( + docs_dir=str(docs_dir), repo_path=str(docs_dir), get_prompt_addition=lambda: "" + ) gen.backend = FakeBackend(docs_dir) return gen, gen.backend diff --git a/tests/test_updater_orchestrator.py b/tests/test_updater_orchestrator.py index 6bfb2a3f..65a8fd61 100644 --- a/tests/test_updater_orchestrator.py +++ b/tests/test_updater_orchestrator.py @@ -29,7 +29,7 @@ def __init__(self, own_verdict="patch"): self.complete_calls = [] self.last_usage = None - def complete(self, prompt, *, model=None): + def complete(self, prompt, *, model=None, system_prompt=None): self.complete_calls.append(prompt[:80]) self.last_usage = {"prompt_tokens": 10, "completion_tokens": 5} return "regenerated overview"