From 3b520c062a3f18adf67c4dfc8e4e460da471d0b3 Mon Sep 17 00:00:00 2001 From: Eric Law <39393654+acn-ericlaw@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:46:30 -0700 Subject: [PATCH] feat: the LLM helper app - llm.chat, llm.stream and llm.health on the Anthropic SDK The dedicated function host that gives a Mercury engine an LLM: examples/llm-helper/ (llm_helper.py, README, resources) serves llm.chat (one answer, with optional JSON-schema structured output), llm.stream (the model's token batches, each forwarded the moment it arrives and never gathered) and llm.health (a credential check with no network traffic) on the official Anthropic SDK. The default model is claude-opus-5-5, refusal fallbacks are on by default, and a Backend seam is documented for AWS Bedrock through IAM (planned, not built). One shared contract file, tests/vectors/llm-helper-vectors.json (byte-identical in mercury-nodejs), runs 63 cases against a fake of the SDK. tests/test_llm_helper.py adds the tests a vector cannot express, 86 in all: token batches are never held back, no prompt text reaches a log, the deadline cancels the call, trace annotations, the backend seam. The demo is a plain demo again: examples/demo_app.py moves to examples/demo-app/ with its resources and a README, and loses its LLM code. The Gemini provider (google-genai) is gone, because the helper serves Claude only. READ: the contract is stricter than the demo nodes were. A params key outside provider, model, max_tokens, timeout_ms, effort and stop_sequences is a 400, a schema on llm.stream is a 400, and a reply with nothing usable is a 422, never an empty success. Each example now lives in a folder of its own: examples/demo_app.py and examples/resources/ moved. Certified live through the Java and Rust engines (docs/test-reports/llm-helper-certification.md). Co-Authored-By: Claude Sonnet 5.5 --- CHANGELOG.md | 51 + README.md | 2 +- docs/guides/config-logging-actuators.md | 2 +- docs/test-reports/llm-helper-certification.md | 75 + examples/demo-app/README.md | 109 + examples/demo-app/demo_app.py | 125 + .../{ => demo-app}/resources/application.yml | 2 +- examples/demo_app.py | 459 ---- examples/llm-helper/README.md | 178 ++ examples/llm-helper/llm_helper.py | 626 +++++ examples/llm-helper/resources/application.yml | 49 + mkdocs.yml | 1 + pyproject.toml | 6 +- tests/test_llm_chat.py | 252 -- tests/test_llm_helper.py | 531 ++++ tests/test_llm_stream.py | 234 -- tests/vectors/llm-helper-vectors.json | 2186 +++++++++++++++++ 17 files changed, 3937 insertions(+), 951 deletions(-) create mode 100644 docs/test-reports/llm-helper-certification.md create mode 100644 examples/demo-app/README.md create mode 100644 examples/demo-app/demo_app.py rename examples/{ => demo-app}/resources/application.yml (94%) delete mode 100644 examples/demo_app.py create mode 100644 examples/llm-helper/README.md create mode 100644 examples/llm-helper/llm_helper.py create mode 100644 examples/llm-helper/resources/application.yml delete mode 100644 tests/test_llm_chat.py create mode 100644 tests/test_llm_helper.py delete mode 100644 tests/test_llm_stream.py create mode 100644 tests/vectors/llm-helper-vectors.json diff --git a/CHANGELOG.md b/CHANGELOG.md index 0f2ae07..b8fbfc6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,56 @@ # Changelog +## Unreleased + +### Added + +- The **LLM helper app** (`examples/llm-helper/`): a dedicated function host for the AI nodes of the + agent-orchestration experiment, on the official Anthropic SDK (default model `claude-opus-5-5`). + `llm.chat` answers once, with optional JSON-schema structured output returned parsed as `data`; + `llm.stream` relays the model's token batches over the multi-shot reply contract, each batch + forwarded the moment it is produced and never gathered; `llm.health` reports whether a call could + be sent, with no network traffic. The contract, the `llm.*` configuration keys, the error contract + and the backend seam (AWS Bedrock through IAM is the planned second route) are in its README. +- One shared contract file, `tests/vectors/llm-helper-vectors.json`, byte-identical in mercury-nodejs: + 63 cases (request validation, the exact SDK call, replies, errors, streaming) run against a fake + of the SDK in each pack, so the two helpers cannot drift apart. No token is spent, no credential + needed. A separate test fails if a token batch is held back. +- A reply that carries nothing usable is an error, never an empty success: a refusal with no text, an + empty reply, or a structured reply cut off before it is valid JSON is a 422 that names the + `stop_reason`, the tokens spent and what to raise. A stream that ends before its first token fails + with a 422 instead of an empty 200. +- Server-side refusal fallbacks (`fallbacks: "default"`) on the models that support them, switchable + with `llm.fallbacks=off`; `request_id` in every reply and terminal event; usage on the trace + record (`llm_model`, `llm_stop_reason`, `llm_input_tokens`, `llm_output_tokens`, `llm_request_id`). +- `llm.log.batches` (off by default): log each streamed batch's number, size and arrival time, never its + text, to tell the hop that holds batches back from the one that forwards them. The certification drive + used it to show that every batch the helper forwarded reached the engine edge as its own frame. It + also showed that the cadence of progressive rendering is the API's and differs by model: Haiku 4.5 + streams continuously, Opus 5.5 in bursts about every 600 ms (documented in the README). + +### Changed + +- The AI nodes moved out of the demo. `examples/demo-app/demo_app.py` is the minimal polyglot demo + again, with no LLM code; `llm.chat` and `llm.stream` live in `examples/llm-helper/llm_helper.py` on + the same default port (8086), with the same request surface, so a graph that names those routes (the + `support-triage` graph) works unchanged. +- **READ - each example app now lives in its own folder**, with its own README and its own `resources/`: + `examples/demo-app/` and `examples/llm-helper/`. The demo moved from `examples/demo_app.py`, and its + sample configuration from `examples/resources/application.yml` to + `examples/demo-app/resources/application.yml`. Start it with + `mercury-serve examples/demo-app/demo_app.py`; the routes and the default port are unchanged. +- **READ - the contract is stricter than the demo nodes were.** A `params` key outside `provider`, + `model`, `max_tokens`, `timeout_ms`, `effort` and `stop_sequences` is a 400 (the demo forwarded any + key to the provider SDK); a `schema` on `llm.stream` is a 400 (it was ignored); on `llm.chat`, + `params.timeout_ms` bounds the whole call, retries included; provider errors read + `LLM provider error - {status} {type}: {message} (request_id …)`. +- The `llm` extra is `anthropic>=1,<2` only. + +### Removed + +- The Gemini provider (`google-genai`) and its selection: the helper serves Claude only, so + `-Dllm.provider=gemini` is now a 400 that names what is served. + ## Version 4.12.15, 9/22/2026 The lock-step round with the engines: the pack moves from 4.12.1 to 4.12.15, the number the Java engine and the diff --git a/README.md b/README.md index 1f92970..b5facb2 100644 --- a/README.md +++ b/README.md @@ -138,7 +138,7 @@ The same conventions as the engines, so a polyglot installation stays uniform: Configuration lives in the `resources` folder, mirroring the engines: `resources/application.yml` (or `.yaml` / `.properties`) in the working directory or next to the application file, or an explicit `--config` path — see -[`examples/resources/application.yml`](examples/resources/application.yml) for a worked +[`examples/demo-app/resources/application.yml`](examples/demo-app/resources/application.yml) for a worked sample. Values support `${ENV_VAR:default}` substitution. Runtime parameter overrides use the same `-D` syntax as the Java engine and the Rust port — checked first on every read (`AppConfig.set(key, value)` does the same programmatically, the `f:setConfig` analog): diff --git a/docs/guides/config-logging-actuators.md b/docs/guides/config-logging-actuators.md index 08aea75..dc54a8c 100644 --- a/docs/guides/config-logging-actuators.md +++ b/docs/guides/config-logging-actuators.md @@ -29,7 +29,7 @@ mercury-serve app.py -Drest.server.port=8090 -Dlog.format=compact ``` See the worked sample -[`examples/resources/application.yml`](https://github.com/Accenture/mercury-python/blob/main/examples/resources/application.yml) +[`examples/demo-app/resources/application.yml`](https://github.com/Accenture/mercury-python/blob/main/examples/demo-app/resources/application.yml) and the full key table in the [Configuration Reference](configuration-reference.md). ## Logging — one aggregation, three presentations diff --git a/docs/test-reports/llm-helper-certification.md b/docs/test-reports/llm-helper-certification.md new file mode 100644 index 0000000..e0c4fde --- /dev/null +++ b/docs/test-reports/llm-helper-certification.md @@ -0,0 +1,75 @@ +--- +title: Test Report — The LLM helper in the Python host +summary: This pack's view of the live LLM helper certification of 2026-10-01 - the Python helper + (llm.chat, llm.stream, llm.health on the Anthropic SDK) behind the Java and Rust engines with real + Claude calls - the token-free proof, the live results, and what the Python SDK needed. +layer: reference +audience: [developer, architect] +keywords: [llm helper, llm.chat, llm.stream, claude, anthropic, streaming, certification, test report] +--- + +# Test Report — The LLM helper in the Python host + +*This pack's view of the live certification of 2026-10-01 (UTC). The full report — all four engine and host +pairs, the three layers, the progressive-rendering evidence and the findings — is the +[engine report](https://accenture.github.io/mercury-composable/test-reports/llm-helper-certification/), which the +Rust engine's docs carry too. The helper is +[`examples/llm-helper`](https://github.com/Accenture/mercury-python/tree/main/examples/llm-helper); its README holds the contract.* + +## Verified without a credential + +- **86 helper tests, no token spent** (the whole suite: 187). `tests/test_llm_helper.py` runs a fake of the Python SDK + that builds the SDK's own error objects, so the status codes and messages are the real ones. +- **One contract, shared with the Node.js twin.** `tests/vectors/llm-helper-vectors.json` is byte-identical in both packs (SHA-256 `1f4823d9…8fa0`, pinned by + a test in each) and holds 63 cases: 22 request validations, 29 chat outcomes and 12 stream outcomes, each fixing the exact SDK + call, the reply or the error, and the segments. A change that drifts one helper away from the other fails a case. +- **The tests can fail.** A helper mutated to gather the token batches and send them at the end fails the sentinel test (the + fake model refuses to produce batch *k* until the caller holds batches 0 to *k*-1) and the vector for tokens delivered before a + mid-stream error; a helper whose default model is changed fails 13 cases. The unmutated helper passes all of them. +- **Static checks:** `ruff check .` clean, `basedpyright` 0 errors on the whole repository, `pytest -q` 187 passed. + +## Verified live + +Behind the Java engine and behind the Rust engine, the Python helper answered every scenario: 40 results per pair and no +failed check, across a streaming service (Layer 1), an Event Script flow (Layer 2) and two graphs (Layer 3), on +`claude-opus-5-5` and, where a request named it, `claude-haiku-4-5`. + +| Progressive stream | Helper batches | Edge frames | Offset, median / spread | Longest gap | +|---|---|---|---|---| +| Java → Python, Opus 5.5 | 67 | 67 | 12 / 10 ms | 1201 / 1206 ms | +| Java → Python, Haiku 4.5 | 101 | 101 | 7 / 4 ms | 191 / 191 ms | +| Rust → Python, Opus 5.5 | 50 | 50 | 4 / 9 ms | 922 / 922 ms | +| Rust → Python, Haiku 4.5 | 82 | 82 | 3 / 5 ms | 228 / 227 ms | + +Every batch the helper forwarded reached the engine's HTTP edge as its own frame, within a few milliseconds and without drift: +nothing is gathered or sent once. The long gaps are the API's own pacing, the same at both ends; Haiku 4.5 streams continuously +(a batch about every 25 ms) while Opus 5.5 arrives in bursts about every 600 ms. Credential states, measured on the real SDK: +no credential gave 503 through Java on all three layers and through Rust on the graph, and an invalid key gave Anthropic's 401 with its request id when this pack's client called the helper directly. The same message came from both helpers, through every layer. All traces that touched this helper rebuild +as one connected tree. + +## What the Python SDK needed + +- **No credential is not an API error.** `anthropic` 1.x raises a bare `TypeError` ("Could not resolve authentication method…") + before anything is sent, so the helper maps it by its text to a 503 `LLM provider credential missing - set ANTHROPIC_API_KEY in the + environment`. A missing key is also what `llm.health` reports, from the client's own attributes and with no network traffic. +- **The SDK takes no sampling parameters.** `temperature`, `top_p` and `top_k` are gone from its signatures, and the current models + reject them, so the contract refuses them (`params` outside `provider`, `model`, `max_tokens`, `timeout_ms`, `effort` and + `stop_sequences` is a 400). +- **A stream's request id is in the response headers.** The final message of a stream carries none (`_request_id` is empty), so the helper + reads `stream.response.headers["request-id"]`; every reply and terminal event carries it. +- **`fallbacks="default"` is a typed value** on the beta namespace, and the API accepted it on Opus 5.5. The helper sends it only for + the models that support it and never on a route that cannot. +- **The deadline is `asyncio.wait_for`** around the whole call, SDK retries included; the abandoned call is cancelled, which a test + asserts. A stream has no total deadline: `timeout_ms` is its idle allowance. + +## Reproduce + +```bash +pip install -e '.[dev]' +pytest tests/test_llm_helper.py # token-free +export ANTHROPIC_API_KEY=... # live: the credential reaches the helper only +mercury-serve examples/llm-helper/llm_helper.py -Dlog.format=compact -Dllm.log.batches=true +``` + +The commands for the engines, the deploy folder and the `curl` calls for each layer are in the +[engine report](https://accenture.github.io/mercury-composable/test-reports/llm-helper-certification/#reproduce). diff --git a/examples/demo-app/README.md b/examples/demo-app/README.md new file mode 100644 index 0000000..0a01ae0 --- /dev/null +++ b/examples/demo-app/README.md @@ -0,0 +1,109 @@ +# Demo app + +The minimal polyglot function host: seven small functions that show what a Mercury engine can reach +over Event-over-HTTP and what a function host does with a call. Run it to watch a Python function +answer an engine, and copy it as the starting point for your own app. It never calls a model and needs +no credential; the LLM routes are a separate app, [`llm-helper`](../llm-helper/README.md). + +```text + caller ── REST ──> Java or Rust engine ── Event-over-HTTP ──> demo app + flow, graph or service POST /api/event this app +``` + +The Node.js twin ([mercury-nodejs `examples/demo-app`](https://github.com/Accenture/mercury-nodejs)) +offers the same routes, `hello.node` in place of `hello.python` and without `hello.sync.chain` (a +JavaScript handler has no sync flavour), with the same behaviour. + +## Run it + +```bash +pip install mercury-composable +mercury-serve examples/demo-app/demo_app.py +``` + +The app listens on port 8086 (`rest.server.port` in `resources/application.yml`; `-Drest.server.port=8090` +overrides it). The [LLM helper](../llm-helper/README.md) defaults to 8086 as well, so give one of the two +another port when you run both on one machine. Map a route from the engine's `event-over-http.yaml`: + +```yaml +event: + http: + - route: 'hello.python' + target: 'http://127.0.0.1:8086/api/event' +``` + +Then call it from a flow or a graph task like any local function, or ad hoc from the package itself, as +the [getting-started guide](../../docs/guides/getting-started.md) shows: + +```python +import asyncio +from mercury_composable import PostOffice + +async def main(): + async with PostOffice(endpoint="http://127.0.0.1:8086/api/event") as po: + reply = await po.request("hello.python", body={"text": "polyglot"}, timeout_ms=5000) + print(reply.get_status(), reply.body) + +asyncio.run(main()) +``` + +```text +200 {'text': 'POLYGLOT', 'language': 'python'} +``` + +## The functions + +| Route | Visibility | What it shows | +|---|---|---| +| `hello.python` | public | The hello world: uppercases `text` and answers `{text, language}`. A body without `text` is a 400 `missing 'text'`, the portable error contract. It annotates the call's trace with `language`. | +| `hello.declarative` | public | Echoes the body and headers it received. The target of the engines' declarative Event-over-HTTP demos. | +| `hello.chain` | public | Local composition: calls the private `demo.suffix.helper` through the in-process bus and returns its reply. | +| `hello.sync.chain` | public | The same composition from a plain `def` handler (the `requests` and NumPy world), through the sync bridge. It blocks its own worker thread, never the event loop. | +| `hello.tokens` | public | Streaming: paced messages over the multi-shot reply contract. See below. | +| `demo.suffix.helper` | private | Appends `!` to `text`. In-app only: the HTTP host answers 403 for a private route. | +| `demo.health` | private | A health check speaking the engines' interface contract (`type=info` and `type=health`). `/health` includes it through `mandatory.health.dependencies`. | + +### Streaming + +`hello.tokens` sends an introductory message at once, then `count` messages, each after `delay` +milliseconds, then the terminal event. Two optional headers set the pace: `delay` (default 500, clamped +to 50 to 5000) and `count` (default 5, clamped to 1 to 100). The terminal's trailing metadata echoes +`count`, `language`, `trace_id` and `my_correlation_id`, so a calling engine's edge shows both +continuity dimensions, the distributed trace and the business correlation id, end to end. + +An engine consumes it progressively (`accept: text/event-stream` on the outbound event) and can render +it out its own HTTP edge, so this is the smallest engine-to-wrapper streaming demonstration. The Java +`lambda-example` and the Rust `hello-world` example drive it that way. + +## Configuration + +`resources/application.yml`, or `-Dkey=value` at run time. + +| Key | Default | Meaning | +|---|---|---| +| `application.name` | `demo-app` | The identity in logs and on the actuator endpoints. | +| `info.app.description` | `Mercury Composable polyglot demo` | The description `/info` reports. | +| `rest.server.port` | `8086` | The Event API and actuator port. | +| `log.format` | `text` | `text`, `json` (pretty-printed) or `compact` (single-line JSONL). | +| `log.level` | `INFO` | The log level; the `LOG_LEVEL` environment variable wins. | +| `mandatory.health.dependencies` | `demo.health` | The health-check routes `/health` requires. | +| `show.env.variables`, `show.application.properties` | `LOG_LEVEL`; `application.name, rest.server.port` | The opt-in lists behind `/env`. Nothing is ever dumped wholesale. | +| `otel.forwarding` | `false` | Opt-in OpenTelemetry export. See below. | + +To export the app's trace spans, run with `-Dotel.forwarding=true`. The endpoint and credential come from +the environment (`OTLP_API_ENDPOINT`, `OTLP_AUTH_HEADER`, `OTLP_TOKEN`, and optionally +`OTLP_SERVICE_NAME`), never from the file, and header names are logged, never values. The +[OpenTelemetry certification report](../../docs/test-reports/otel-dynatrace-certification.md) records a +run of this host against a live backend. + +## What it does not do + +It calls no model, holds no credential and keeps no state between calls. It is a demonstration of the +host's contracts, not a template for business logic. + +## Tests + +The demo has no unit tests of its own. The mechanisms it demonstrates are covered in `tests/`: the local +bus, private routes and the sync bridge in `tests/test_bus.py`, streaming in `tests/test_event_stream.py`. +`hello.tokens` was driven through both engines in the +[progressive-rendering interop report](../../docs/test-reports/progressive-rendering-interop.md). diff --git a/examples/demo-app/demo_app.py b/examples/demo-app/demo_app.py new file mode 100644 index 0000000..ed1f2c8 --- /dev/null +++ b/examples/demo-app/demo_app.py @@ -0,0 +1,125 @@ +""" +Demo polyglot functions. + +Run: mercury-serve examples/demo-app/demo_app.py + +Configuration comes from examples/demo-app/resources/application.yml (the +engines' "resources" convention - port 8086, the demo.health dependency, log +format); override any key with -Dkey=value, e.g. -Drest.server.port=8090. + +Then map a route from a Mercury engine application (event-over-http.yaml): + + event.http: + - route: 'hello.python' + target: 'http://127.0.0.1:8086/api/event' +""" + +import asyncio + +from mercury_composable import ( + AppException, + Body, + EventEnvelope, + EventStreamWriter, + PostOffice, + annotate_trace, + get_logger, + get_trace, + platform, + preload, +) + +log = get_logger(__name__) + +# the streaming head's content type (every streaming demo renders as SSE) +TEXT_EVENT_STREAM = "text/event-stream" + + +@preload(route="hello.python", instances=10) +def handle_event(_headers: dict[str, str], body: Body): + """Uppercase transform - the polyglot hello world. + + The (headers, body) two-part signature is the function contract (the + TypedLambdaFunction mirror) - a handler that does not need headers keeps + the parameter, underscore-prefixed per Python convention. + """ + if not isinstance(body, dict) or "text" not in body: + raise AppException(400, "missing 'text'") + annotate_trace("language", "python") + log.info("Transforming text of length %d", len(str(body["text"]))) + return {"text": str(body["text"]).upper(), "language": "python"} + + +@preload(route="hello.declarative", instances=10) +async def declarative_echo(headers: dict[str, str], body: Body): + """Echo for the composable-example declarative Event-over-HTTP demo.""" + return {"body": body, "headers": headers, "language": "python"} + + +@preload(route="demo.suffix.helper", instances=10, private=True) +async def suffix_helper(_headers: dict[str, str], body: Body): + """Private helper - callable in-app only (the HTTP host answers 403 for it).""" + assert isinstance(body, dict) + return {"text": f"{body.get('text', '')}!", "language": "python"} + + +@preload(route="hello.chain", instances=10) +async def chain(_headers: dict[str, str], body: Body): + """Local composition: a public function calls a private sibling through the bus.""" + reply = await PostOffice().request("demo.suffix.helper", body=body, timeout_ms=5000) + return reply.body + + +@preload(route="hello.sync.chain", instances=10) +def sync_chain(_headers: dict[str, str], body: Body): + """Sync composition: a plain-def handler (the requests/NumPy world) calls a + sibling through the sync bridge - blocking its own worker thread only, + never the event loop.""" + reply = PostOffice().request_sync("demo.suffix.helper", body=body, timeout_ms=5000) + return reply.body + + +@preload(route="hello.tokens", instances=10, interceptor=True) +async def stream_tokens(headers: dict[str, str], event: EventEnvelope): + """Streaming demo: paced test messages over the multi-shot reply contract. + + A calling engine consumes this progressively through Event-over-HTTP + (accept: text/event-stream on the outbound event) and can render it out + its own HTTP edge - engine-to-wrapper token streaming. Optional headers: + "delay" ms between messages (default 500, clamped 50-5000) and "count" + messages (default 5, clamped 1-100). + """ + delay = min(5000, max(50, int(headers.get("delay", "500") or 500))) / 1000 + count = min(100, max(1, int(headers.get("count", "5") or 5))) + # with log.format=json/compact, this line carries the application log + # "context" block (trace ids, business cid) - see the streaming guide + log.info("Streaming %d messages", count) + out = EventStreamWriter.from_request(event) + out.first(200, TEXT_EVENT_STREAM) + out.write("The following messages are rendered slowly to demonstrate streaming:") + for n in range(1, count + 1): + await asyncio.sleep(delay) + out.write(f"test message {n} (python)") + # the trailing metadata echoes the distributed trace id and the business + # correlation-id, so a calling engine's edge shows both continuity + # dimensions end to end + info = get_trace() + out.close({"count": count, "language": "python", + "trace_id": info.trace_id if info else None, + "my_correlation_id": headers.get("my_correlation_id")}) + + +@preload(route="demo.health", instances=5, private=True) +async def health_check(headers: dict[str, str], _body: Body): + """Health check speaking the engines' interface contract (type=info / type=health). + + Activated for the /health actuator endpoint by mandatory.health.dependencies + in examples/demo-app/resources/application.yml (or a -D override). + """ + if headers.get("type") == "info": + return {"service": "demo.service", "href": "http://127.0.0.1"} + return "demo.service is running fine" + + +if __name__ == "__main__": + platform.run() diff --git a/examples/resources/application.yml b/examples/demo-app/resources/application.yml similarity index 94% rename from examples/resources/application.yml rename to examples/demo-app/resources/application.yml index 72c2c71..6101d0e 100644 --- a/examples/resources/application.yml +++ b/examples/demo-app/resources/application.yml @@ -2,7 +2,7 @@ # mercury-serve loads resources/application.yml from the working directory, or from # the resources folder next to the application file (this one). Any key can be # overridden at run time with the engines' -D syntax, e.g.: -# mercury-serve examples/demo_app.py -Drest.server.port=8090 -Dlog.format=compact +# mercury-serve examples/demo-app/demo_app.py -Drest.server.port=8090 -Dlog.format=compact application.name: 'demo-app' info.app.description: 'Mercury Composable polyglot demo' diff --git a/examples/demo_app.py b/examples/demo_app.py deleted file mode 100644 index 7cad69d..0000000 --- a/examples/demo_app.py +++ /dev/null @@ -1,459 +0,0 @@ -""" -Demo polyglot functions. - -Run: mercury-serve examples/demo_app.py - -Configuration comes from examples/resources/application.yml (the engines' -"resources" convention - port 8086, the demo.health dependency, log format); -override any key with -Dkey=value, e.g. -Drest.server.port=8090. - -Then map a route from a Mercury engine application (event-over-http.yaml): - - event.http: - - route: 'hello.python' - target: 'http://127.0.0.1:8086/api/event' -""" - -import asyncio -import json -from typing import Any - -from mercury_composable import ( - AppException, - Body, - EventEnvelope, - EventStreamWriter, - PostOffice, - annotate_trace, - app_config, - get_logger, - get_trace, - platform, - preload, -) - -log = get_logger(__name__) - -# the AI node's provider surface: llm.provider / llm.model in the app config -# (or -D overrides), params.provider / params.model per call -LLM_DEFAULT_PROVIDER = "anthropic" -# gemini-flash-latest is the stable alias: a dated flash id stops being served when the -# provider retires it (gemini-3.6-flash 404s on a current key), the alias moves with it -LLM_DEFAULT_MODELS = {"anthropic": "claude-opus-5", "gemini": "gemini-flash-latest"} -LLM_DEFAULT_MAX_TOKENS = 16000 -LLM_DEFAULT_TIMEOUT_MS = 60000 - -# the streaming head's content type (every streaming demo renders as SSE) -TEXT_EVENT_STREAM = "text/event-stream" - -# lazily built provider clients - module-level so tests can inject fakes -_llm_client: Any = None -_gemini_client: Any = None - - -@preload(route="hello.python", instances=10) -def handle_event(_headers: dict[str, str], body: Body): - """Uppercase transform - the polyglot hello world. - - The (headers, body) two-part signature is the function contract (the - TypedLambdaFunction mirror) - a handler that does not need headers keeps - the parameter, underscore-prefixed per Python convention. - """ - if not isinstance(body, dict) or "text" not in body: - raise AppException(400, "missing 'text'") - annotate_trace("language", "python") - log.info("Transforming text of length %d", len(str(body["text"]))) - return {"text": str(body["text"]).upper(), "language": "python"} - - -@preload(route="hello.declarative", instances=10) -async def declarative_echo(headers: dict[str, str], body: Body): - """Echo for the composable-example declarative Event-over-HTTP demo.""" - return {"body": body, "headers": headers, "language": "python"} - - -@preload(route="demo.suffix.helper", instances=10, private=True) -async def suffix_helper(_headers: dict[str, str], body: Body): - """Private helper - callable in-app only (the HTTP host answers 403 for it).""" - assert isinstance(body, dict) - return {"text": f"{body.get('text', '')}!", "language": "python"} - - -@preload(route="hello.chain", instances=10) -async def chain(_headers: dict[str, str], body: Body): - """Local composition: a public function calls a private sibling through the bus.""" - reply = await PostOffice().request("demo.suffix.helper", body=body, timeout_ms=5000) - return reply.body - - -@preload(route="hello.sync.chain", instances=10) -def sync_chain(_headers: dict[str, str], body: Body): - """Sync composition: a plain-def handler (the requests/NumPy world) calls a - sibling through the sync bridge - blocking its own worker thread only, - never the event loop.""" - reply = PostOffice().request_sync("demo.suffix.helper", body=body, timeout_ms=5000) - return reply.body - - -@preload(route="hello.tokens", instances=10, interceptor=True) -async def stream_tokens(headers: dict[str, str], event: EventEnvelope): - """Streaming demo: paced test messages over the multi-shot reply contract. - - A calling engine consumes this progressively through Event-over-HTTP - (accept: text/event-stream on the outbound event) and can render it out - its own HTTP edge - engine-to-wrapper token streaming. Optional headers: - "delay" ms between messages (default 500, clamped 50-5000) and "count" - messages (default 5, clamped 1-100). - """ - delay = min(5000, max(50, int(headers.get("delay", "500") or 500))) / 1000 - count = min(100, max(1, int(headers.get("count", "5") or 5))) - # with log.format=json/compact, this line carries the application log - # "context" block (trace ids, business cid) - see the streaming guide - log.info("Streaming %d messages", count) - out = EventStreamWriter.from_request(event) - out.first(200, TEXT_EVENT_STREAM) - out.write("The following messages are rendered slowly to demonstrate streaming:") - for n in range(1, count + 1): - await asyncio.sleep(delay) - out.write(f"test message {n} (python)") - # the trailing metadata echoes the distributed trace id and the business - # correlation-id, so a calling engine's edge shows both continuity - # dimensions end to end - info = get_trace() - out.close({"count": count, "language": "python", - "trace_id": info.trace_id if info else None, - "my_correlation_id": headers.get("my_correlation_id")}) - - -@preload(route="llm.chat", instances=50) -async def llm_chat(_headers: dict[str, str], body: Body): - """The AI node (agent-orchestration experiment E0): a provider-neutral LLM - adapter as a plain wrapper-side function. The engine and this host stay - LLM-free - a graph or flow reaches this route like any other function, so - the certified graph decides control flow while the model advises within it. - - Input (map): - prompt | messages single-turn text, or conversation turns [{role, content}] - system optional system prompt - schema optional JSON schema -> structured output (the graph - needs parseable verdicts for decision routing; - additionalProperties defaults to false) - params provider, model, max_tokens, timeout_ms + provider - pass-through - - Provider selection: params.provider, else the llm.provider config key - (e.g. mercury-serve ... -Dllm.provider=gemini), else anthropic. The default - model per provider comes from params.model, the llm.model config key, or - LLM_DEFAULT_MODELS. - - Output (map): text | data, model, stop_reason (check for "refusal"), - usage {input_tokens, output_tokens}. - - Provider errors ride the envelope status - portable to a graph's error - context (error.code / error.message). The remaining time budget maps onto - the SDK timeout (params.timeout_ms), the x-ttl pattern. - """ - provider, model, max_tokens, timeout_ms, messages, system, params = _llm_request_prep(body) - assert isinstance(body, dict) # narrowed by _llm_request_prep - schema = body.get("schema") - if isinstance(schema, dict): - # structured output: a closed schema is what a bounded verdict wants, so - # default additionalProperties to false when the caller omits it - schema = {"additionalProperties": False, **schema} - else: - schema = None - if provider == "gemini": - result = await _call_gemini(model, messages, system, schema, max_tokens, timeout_ms, params) - else: - result = await _call_anthropic(model, messages, system, schema, max_tokens, timeout_ms, params) - annotate_trace("llm_model", str(result.get("model", ""))) - text = str(result.pop("text", "")) - if schema is not None and text: - # both providers guarantee schema-constrained output as one JSON text - result["data"] = json.loads(text) - else: - result["text"] = text - return result - - -def _llm_request_prep( - body: Body, -) -> tuple[str, str, int, int, Any, Any, dict[str, Any]]: - """The shared request surface of the AI nodes (llm.chat and llm.stream): - provider and model resolution (params -> llm.provider/llm.model config -> - defaults), token/time budgets and message shaping.""" - if not isinstance(body, dict) or not (body.get("prompt") or body.get("messages")): - raise AppException(400, "missing 'prompt' or 'messages'") - raw_params = body.get("params") - params: dict[str, Any] = dict(raw_params) if isinstance(raw_params, dict) else {} - provider = str( - params.pop("provider", None) - or app_config().get_property("llm.provider", LLM_DEFAULT_PROVIDER) - or LLM_DEFAULT_PROVIDER - ).lower() - if provider not in LLM_DEFAULT_MODELS: - raise AppException(400, f"unknown LLM provider '{provider}' - use one of " - f"{sorted(LLM_DEFAULT_MODELS)}") - model = str( - params.pop("model", None) - or app_config().get_property("llm.model", None) - or LLM_DEFAULT_MODELS[provider] - ) - max_tokens = int(params.pop("max_tokens", LLM_DEFAULT_MAX_TOKENS)) - timeout_ms = int(params.pop("timeout_ms", LLM_DEFAULT_TIMEOUT_MS)) - messages = body.get("messages") or [{"role": "user", "content": str(body["prompt"])}] - return provider, model, max_tokens, timeout_ms, messages, body.get("system"), params - - -async def _call_anthropic(model: str, messages: Any, system: Any, schema: dict[str, Any] | None, - max_tokens: int, timeout_ms: int, extra: dict[str, Any]) -> dict[str, Any]: - """Anthropic SDK edition of the llm.chat contract (lazy optional import).""" - try: - import anthropic - except ImportError as exc: # the SDK is an optional extra - teach, don't crash the app - raise AppException(501, "llm.chat requires the Anthropic SDK - pip install anthropic") from exc - request: dict[str, Any] = {"model": model, "max_tokens": max_tokens, "messages": messages} - if system: - request["system"] = system - if schema is not None: - request["output_config"] = {"format": {"type": "json_schema", "schema": schema}} - request.update(extra) # provider pass-through (e.g. output_config.effort) wins verbatim - global _llm_client - if _llm_client is None: - _llm_client = anthropic.AsyncAnthropic() - try: - response = await _llm_client.with_options(timeout=max(1.0, timeout_ms / 1000)) \ - .messages.create(**request) - except anthropic.RateLimitError as exc: - raise AppException(429, f"LLM provider rate limit - {exc}") from exc - except anthropic.APIStatusError as exc: - raise AppException(int(exc.status_code), f"LLM provider error - {exc}") from exc - except anthropic.APIConnectionError as exc: - raise AppException(503, f"LLM provider unreachable - {exc}") from exc - text = "" - for block in response.content: - if getattr(block, "type", "") == "text" and block.text: - text = block.text - break - return { - "text": text, - "model": response.model, - "stop_reason": str(response.stop_reason), - "usage": { - "input_tokens": response.usage.input_tokens, - "output_tokens": response.usage.output_tokens, - }, - } - - -def _gemini_request(system: Any, messages: Any, max_tokens: int, timeout_ms: int, - extra: dict[str, Any]) -> tuple[Any, list[Any]]: - """Config and role-mapped contents shared by the single-shot and streaming - Gemini editions. HttpOptions.timeout is in milliseconds - params.timeout_ms - passes through. Conversation turns map onto Gemini roles (assistant -> model). - """ - from google.genai import types as genai_types - config = genai_types.GenerateContentConfig( - max_output_tokens=max_tokens, - http_options=genai_types.HttpOptions(timeout=timeout_ms), - **extra, # provider pass-through (e.g. temperature) wins verbatim - ) - if config.automatic_function_calling is None: - # the AI nodes expose no tool surface (the graph decides, the model - # advises), so the SDK's automatic-function-calling loop is opted out - - # which also silences its AFC advisory warning on direct - # generate_content(_stream) calls - config.automatic_function_calling = genai_types.AutomaticFunctionCallingConfig(disable=True) - if system: - config.system_instruction = str(system) - contents = [ - genai_types.Content( - role="model" if turn.get("role") == "assistant" else "user", - parts=[genai_types.Part(text=str(turn.get("content", "")))], - ) - for turn in messages - if isinstance(turn, dict) - ] - return config, contents - - -def _gemini_finish(candidates: Any) -> str: - """Finish-reason name from a response/chunk's candidates, or empty.""" - if candidates: - reason = candidates[0].finish_reason - if reason is not None: - return getattr(reason, "name", str(reason)) - return "" - - -def _gemini_usage(usage: Any) -> dict[str, int]: - """Usage metadata (absent until the final stream chunk) in the contract shape.""" - return { - "input_tokens": (usage.prompt_token_count or 0) if usage else 0, - "output_tokens": (usage.candidates_token_count or 0) if usage else 0, - } - - -async def _call_gemini(model: str, messages: Any, system: Any, schema: dict[str, Any] | None, - max_tokens: int, timeout_ms: int, extra: dict[str, Any]) -> dict[str, Any]: - """Gemini SDK edition of the llm.chat contract (lazy optional import). - - The client reads GEMINI_API_KEY (or GOOGLE_API_KEY) from the environment. - """ - try: - from google import genai - from google.genai import errors as genai_errors - except ImportError as exc: - raise AppException(501, "llm.chat requires the Gemini SDK - pip install google-genai") from exc - config, contents = _gemini_request(system, messages, max_tokens, timeout_ms, extra) - if schema is not None: - config.response_mime_type = "application/json" - config.response_json_schema = schema - global _gemini_client - if _gemini_client is None: - _gemini_client = genai.Client() - try: - response = await _gemini_client.aio.models.generate_content( - model=model, contents=contents, config=config) - except genai_errors.APIError as exc: - raise AppException(int(exc.code) if exc.code else 500, f"LLM provider error - {exc}") from exc - except OSError as exc: # connection-level failures (DNS, refused, timeout) - raise AppException(503, f"LLM provider unreachable - {exc}") from exc - return { - "text": response.text or "", - "model": getattr(response, "model_version", None) or model, - "stop_reason": _gemini_finish(response.candidates), - "usage": _gemini_usage(response.usage_metadata), - } - - -@preload(route="llm.stream", instances=50, interceptor=True) -async def llm_stream(headers: dict[str, str], event: EventEnvelope): - """The streaming AI node (agent-orchestration follow-up to E0): pulls the - provider's REAL token stream and relays each token batch over the multi-shot - reply contract - a calling engine renders it progressively out its own HTTP - edge (SSE), with the same provider neutrality as llm.chat. - - Body: prompt | messages, optional system, params (provider, model, - max_tokens, timeout_ms + provider pass-through). Structured output (schema) - is deliberately not part of the streaming contract - a schema verdict is a - single-shot reply (use llm.chat). - - The terminal event's trailing metadata carries model, stop_reason, usage - and the trace/business correlation ids. - """ - out = EventStreamWriter.from_request(event) - try: - provider, model, max_tokens, timeout_ms, messages, system, params = \ - _llm_request_prep(event.body) - except AppException as exc: - out.fail(exc) - return - info = get_trace() - meta: dict[str, Any] = { - "language": "python", - "trace_id": info.trace_id if info else None, - "my_correlation_id": headers.get("my_correlation_id"), - } - log.info("Streaming tokens from %s via %s", model, provider) - if provider == "gemini": - await _stream_gemini(out, model, messages, system, max_tokens, timeout_ms, params, meta) - else: - await _stream_anthropic(out, model, messages, system, max_tokens, timeout_ms, params, meta) - - -async def _stream_anthropic(out: EventStreamWriter, model: str, messages: Any, system: Any, - max_tokens: int, timeout_ms: int, extra: dict[str, Any], - meta: dict[str, Any]) -> None: - """Anthropic SDK edition of the streaming contract (lazy optional import).""" - try: - import anthropic - except ImportError: - out.fail(AppException(501, "llm.stream requires the Anthropic SDK - pip install anthropic")) - return - request: dict[str, Any] = {"model": model, "max_tokens": max_tokens, "messages": messages} - if system: - request["system"] = system - request.update(extra) # provider pass-through wins verbatim - global _llm_client - if _llm_client is None: - _llm_client = anthropic.AsyncAnthropic() - try: - async with _llm_client.with_options(timeout=max(1.0, timeout_ms / 1000)) \ - .messages.stream(**request) as stream: - out.first(200, TEXT_EVENT_STREAM) - async for text in stream.text_stream: - if text: - out.write(text) - message = await stream.get_final_message() - except anthropic.RateLimitError as exc: - out.fail(AppException(429, f"LLM provider rate limit - {exc}")) - return - except anthropic.APIStatusError as exc: - out.fail(AppException(int(exc.status_code), f"LLM provider error - {exc}")) - return - except anthropic.APIConnectionError as exc: - out.fail(AppException(503, f"LLM provider unreachable - {exc}")) - return - out.close({ - "model": message.model, - "stop_reason": str(message.stop_reason), - "usage": { - "input_tokens": message.usage.input_tokens, - "output_tokens": message.usage.output_tokens, - }, - **meta, - }) - - -async def _stream_gemini(out: EventStreamWriter, model: str, messages: Any, system: Any, - max_tokens: int, timeout_ms: int, extra: dict[str, Any], - meta: dict[str, Any]) -> None: - """Gemini SDK edition of the streaming contract: each chunk is a token batch.""" - try: - from google import genai - from google.genai import errors as genai_errors - except ImportError: - out.fail(AppException(501, "llm.stream requires the Gemini SDK - pip install google-genai")) - return - config, contents = _gemini_request(system, messages, max_tokens, timeout_ms, extra) - global _gemini_client - if _gemini_client is None: - _gemini_client = genai.Client() - usage = None - finish = "" - version = model - try: - stream = await _gemini_client.aio.models.generate_content_stream( - model=model, contents=contents, config=config) - out.first(200, TEXT_EVENT_STREAM) - async for chunk in stream: - if chunk.text: - out.write(chunk.text) - # usage/finish arrive on the final chunk; the model version on any - usage = chunk.usage_metadata or usage - finish = _gemini_finish(chunk.candidates) or finish - version = getattr(chunk, "model_version", None) or version - except genai_errors.APIError as exc: - out.fail(AppException(int(exc.code) if exc.code else 500, f"LLM provider error - {exc}")) - return - except OSError as exc: - out.fail(AppException(503, f"LLM provider unreachable - {exc}")) - return - out.close({"model": version, "stop_reason": finish, "usage": _gemini_usage(usage), **meta}) - - -@preload(route="demo.health", instances=5, private=True) -async def health_check(headers: dict[str, str], _body: Body): - """Health check speaking the engines' interface contract (type=info / type=health). - - Activated for the /health actuator endpoint by mandatory.health.dependencies - in examples/resources/application.yml (or a -D override). - """ - if headers.get("type") == "info": - return {"service": "demo.service", "href": "http://127.0.0.1"} - return "demo.service is running fine" - - -if __name__ == "__main__": - platform.run() diff --git a/examples/llm-helper/README.md b/examples/llm-helper/README.md new file mode 100644 index 0000000..b5c32de --- /dev/null +++ b/examples/llm-helper/README.md @@ -0,0 +1,178 @@ +# LLM helper + +A dedicated function host that gives a Mercury engine an LLM: `llm.chat`, `llm.stream` and +`llm.health`, on the official Anthropic SDK. The engines stay LLM-free. A graph's `graph.task` +node or a flow's task names the route like any other function, the engine reaches this app +through declarative Event-over-HTTP, and the certified graph decides control flow while the +model advises within it. + +```text + caller ── REST (SSE / JSON) ──> Java or Rust engine ── Event-over-HTTP ──> llm-helper ──> Claude + Layer 1 service, flow or graph task this app + flow, or graph +``` + +The Node.js twin ([mercury-nodejs `examples/llm-helper`](https://github.com/Accenture/mercury-nodejs)) +speaks the same contract. Both run one shared vector file, so they cannot drift apart. + +## Run it + +```bash +pip install 'mercury-composable[llm]' # the SDK is this app's one dependency +export ANTHROPIC_API_KEY=... # from the environment, never from a config file +mercury-serve examples/llm-helper/llm_helper.py +``` + +The app listens on port 8086 (`rest.server.port` in `resources/application.yml`). Map its routes +from the engine's `event-over-http.yaml`, as the Java and Rust MiniGraph Playgrounds do: + +```yaml +event: + http: + - route: 'llm.chat' + target: 'http://127.0.0.1:8086/api/event' + - route: 'llm.stream' + target: 'http://127.0.0.1:8086/api/event' +``` + +## The contract + +Both routes take the same request. `llm.chat` answers once; `llm.stream` answers with the model's +token batches as they are produced. + +| Field | Meaning | +|---|---| +| `prompt` or `messages` | A single user turn, or conversation turns `[{role: user\|assistant, content}]`. `messages` wins when both are given. Content is plain text. | +| `system` | Optional system prompt (text). | +| `schema` | `llm.chat` only. A JSON schema; the reply is then structured output, returned parsed as `data`. `additionalProperties` defaults to `false`. A schema on `llm.stream` is a 400. | +| `params.model` | The model. Default `claude-opus-5-5` (`llm.model`). | +| `params.max_tokens` | Output cap. Default 16000 (`llm.max.tokens`). | +| `params.timeout_ms` | `llm.chat`: a deadline for the whole call, SDK retries included. `llm.stream`: the idle allowance between events. Default 60000 (`llm.timeout.ms`). | +| `params.effort` | `low`, `medium`, `high`, `xhigh` or `max`. Sent only when set; otherwise the model's own default applies. A model that takes no effort parameter answers 400. | +| `params.stop_sequences` | A list of strings. | +| `params.provider` | Accepted for compatibility: only `anthropic`. | + +Any other `params` key is a 400 that names the supported ones. The current Claude models take no +sampling parameters (`temperature`, `top_p`, `top_k`), so none is forwarded. + +**`llm.chat` reply** (a map): `text`, or `data` for a `schema` request; `model` (the model that +answered); `stop_reason`; `usage {input_tokens, output_tokens}`; `request_id` when the API sent one; +`stop_details {category, explanation}` when the model refused. + +**`llm.stream` reply**: one segment per token batch, in order, then a terminal event whose +trailing metadata carries `model`, `stop_reason`, `usage`, `request_id`, `language`, `trace_id` +and `my_correlation_id`. + +### Progressive rendering + +The point of `llm.stream` is to deliver batches continuously, so each batch the model produces +leaves this app as its own segment at that moment. Nothing is gathered and sent once. The stream's +head (status 200, `text/event-stream`) rides the first batch. A test pins this: the fake model +refuses to produce batch *k* until the caller already holds batches 0 to *k*-1. + +**What a viewer sees is bounded by how the API delivers.** The helper forwards every batch the moment it +arrives, and the engine edge adds nothing: in the certification drive, every batch the helper forwarded +reached the edge as its own frame, within a few milliseconds and without drift. The cadence itself is the +API's, and it differs by model. Measured on the raw API with no SDK in the path, Haiku 4.5 sends about 2 +tokens per batch every 25 ms and renders continuously; Sonnet 5.5 sends about 4 tokens in bursts every +350 ms or so; Opus 5.5 sends about 5 tokens, a dozen batches at a time, every 600 ms or so. A burst is +several batches arriving together, so the edge shows the same cadence. Choose `llm.model` (or +`params.model`) for the rendering you want. To attribute a slow or bursty stream, switch on +`llm.log.batches`: each batch then logs its number, size and arrival time (never its text), so the hop +that holds batches back can be told apart from the one that forwards them. + +### Errors + +Every failure is an `AppException(status, message)`, the portable error contract: the status rides +the envelope into a graph's error context or an HTTP edge. + +| Status | When | Message starts | +|---|---|---| +| 400 | A malformed request, or a 400 from the API | `missing 'prompt' or 'messages'`, `unsupported params: …`, `LLM provider error - 400 …` | +| 401 403 404 413 5xx 529 | The API's status, passed through | `LLM provider error - {status} {type}: {message} (request_id …)` | +| 408 | The deadline or the SDK's timeout | `LLM request timed out after {n} ms` | +| 422 | The reply carries nothing usable | `LLM refused the request …`, `LLM reply is empty …`, `LLM reply is not valid JSON for the requested schema …` | +| 429 | Rate limited | `LLM provider rate limit - 429 rate_limit_error: …` | +| 503 | No credential, or the API is unreachable | `LLM provider credential missing - set ANTHROPIC_API_KEY in the environment`, `LLM provider unreachable - …` | + +**A reply that carries nothing usable is an error, never an empty success.** A refusal with no +text, an empty reply, or a structured reply cut off before it is valid JSON is a 422 that says why +(`stop_reason`, the tokens spent, and what to raise). A reply that stopped early after real text +(`max_tokens`, or a refusal part-way) is returned with its `stop_reason`, and `stop_details` when +the model refused. A stream that ends before its first token fails with a 422 instead of an empty +200. Opus 5.5 thinks before it answers, and thinking tokens count against `max_tokens`, so a budget that +is too small for the question can end with no text at all (seen live with a 120-token budget): that is +the 422 above, not an empty stream. Raise `params.max_tokens` or lower `params.effort`. + +## Configuration + +`resources/application.yml`, or `-Dkey=value` at run time. A call's `params` win over these keys. + +| Key | Default | Meaning | +|---|---|---| +| `llm.backend` | `anthropic` | The route to the models; see Backends. | +| `llm.model` | `claude-opus-5-5` | The default model. | +| `llm.max.tokens` | `16000` | The default output cap. | +| `llm.timeout.ms` | `60000` | The default deadline (chat) or idle allowance (stream). | +| `llm.max.retries` | `2` | SDK retries for connection errors, 408, 409, 429 and 5xx. | +| `llm.fallbacks` | `default` | `default` or `off`. Server-side refusal fallbacks (`fallbacks: "default"`) on the models that support them (the Opus 5, Fable 5 and Sonnet 5.5 families). The reply's `model` names the model that answered. | +| `llm.effort` | unset | The default effort. | +| `llm.log.batches` | `false` | Log each streamed batch's number, size and arrival time (never its text). | +| `llm.provider` | `anthropic` | Only `anthropic`; any other value is a 400. | + +The credential is never configured here. `ANTHROPIC_API_KEY` (or whatever the SDK resolves: an +auth token, or a profile) comes from the environment. `llm.health` reports whether a call could +be sent (a credential is present) with no network traffic, so a health probe never spends a token. +`/health` includes it through `mandatory.health.dependencies`. + +### The default model and the token budget + +The default stays `claude-opus-5-5`, on purpose: it is the most capable model, and a graph's AI node is where +capability counts. It thinks before it answers, and its thinking tokens count against `max_tokens`, so give a +call a generous budget. The engines' demos ask for 2000 and the helper's own default is 16000; a budget of a few +hundred can end with no text at all, which the helper reports as `422 LLM reply is empty`, never as an empty +success. + +For the lowest latency and a continuous token trickle, choose Haiku 4.5 (`llm.model: claude-haiku-4-5`, or +`params.model` on a call). The helper sends no thinking parameter, so Haiku does not think, and it takes no +`effort` parameter (a call that sets one answers 400). Sonnet 5.5 sits between the two. The progressive rendering +section above gives each model's cadence. + +## Logs and traces + +A call logs its model, `stop_reason`, token usage, `request_id` and elapsed time. A prompt, a +system prompt or a completion never reaches a log. Usage also rides the call's trace record as +annotations (`llm_model`, `llm_stop_reason`, `llm_input_tokens`, `llm_output_tokens`, +`llm_request_id`), so telemetry shows what a call cost. + +## Backends + +`get_backend()` builds the one client the app uses, and a `Backend` says what that route to the +models supports. Today there is one, `anthropic` (the Claude API). + +**Planned: AWS Bedrock through IAM.** It is additive. Build the SDK's Bedrock client in +`get_backend()` (it authenticates with the AWS credentials of the default chain instead of an API +key, and takes a region), give the `Backend` a `provider_model` that prefixes the model id (`anthropic.claude-opus-5-5`), +report `supports_fallbacks=False` (server-side fallbacks are not offered on Bedrock), and describe a +missing AWS credential in `credential_problem()`. Select it with `llm.backend: 'bedrock'`. The routes, +the contract and the vector file stay as they are, because nothing outside `Backend` knows which +route answered. + +## What it does not do + +No tools or function calling, no images or documents, no `system` turns inside `messages`, no +sampling parameters, no prompt caching controls, and no provider other than Claude. Those are +deliberate: the helper is the bounded AI node of a graph, not an agent runtime. + +## Tests + +`tests/test_llm_helper.py` runs the shared `tests/vectors/llm-helper-vectors.json` against a fake of +the SDK (63 cases: request validation, the exact SDK call, replies, the error contract, streaming), +plus the tests a vector cannot express (token batches are never held back, no prompt text in a +log, the deadline cancels the call, trace annotations, health, the backend seam). No token is +spent and no credential is needed. + +```bash +pip install -e '.[dev]' +pytest tests/test_llm_helper.py +``` diff --git a/examples/llm-helper/llm_helper.py b/examples/llm-helper/llm_helper.py new file mode 100644 index 0000000..de33708 --- /dev/null +++ b/examples/llm-helper/llm_helper.py @@ -0,0 +1,626 @@ +""" +The LLM helper - a dedicated polyglot function host for the AI nodes of the +agent-orchestration experiment: ``llm.chat``, ``llm.stream`` and ``llm.health``, on the +official Anthropic SDK. The engines stay LLM-free: a graph or a flow reaches these routes +through declarative Event-over-HTTP like any other function, so the certified model decides +control flow while the LLM advises within it. + +Run: pip install 'mercury-composable[llm]' + mercury-serve examples/llm-helper/llm_helper.py + +The credential comes from the environment (ANTHROPIC_API_KEY), never from a config file. +Settings live in resources/application.yml (or -Dkey=value): llm.backend, llm.model, +llm.max.tokens, llm.timeout.ms, llm.max.retries, llm.fallbacks, llm.effort. The contract, +the keys and the backend seam are documented in README.md next to this file. The Node.js +twin (mercury-nodejs examples/llm-helper) speaks the same contract, and both are pinned by +one shared vector file. + +Then map the routes from a Mercury engine application (event-over-http.yaml): + + event: + http: + - route: 'llm.chat' + target: 'http://127.0.0.1:8086/api/event' + - route: 'llm.stream' + target: 'http://127.0.0.1:8086/api/event' +""" + +from __future__ import annotations + +import asyncio +import json +import time +from collections.abc import Callable +from dataclasses import dataclass +from typing import Any + +try: + import anthropic +except ImportError as missing_sdk: # the SDK is this app's one dependency - teach the fix + raise SystemExit( + "The LLM helper needs the Anthropic SDK - pip install 'mercury-composable[llm]'" + ) from missing_sdk + +from mercury_composable import ( + AppException, + Body, + EventEnvelope, + EventStreamWriter, + annotate_trace, + app_config, + get_logger, + get_trace, + platform, + preload, +) + +log = get_logger(__name__) + +# --- the contract's defaults and limits ---------------------------------------------- + +DEFAULT_PROVIDER = "anthropic" +DEFAULT_BACKEND = "anthropic" +DEFAULT_MODEL = "claude-opus-5-5" +DEFAULT_MAX_TOKENS = 16000 +DEFAULT_TIMEOUT_MS = 60000 +DEFAULT_MAX_RETRIES = 2 +PROVIDERS = (DEFAULT_PROVIDER,) +BACKENDS = (DEFAULT_BACKEND,) +EFFORTS = ("low", "medium", "high", "xhigh", "max") +ROLES = ("user", "assistant") +# per-call params. Anything else is rejected: the current models take no sampling +# parameters, and a mistyped key must not vanish silently +PARAMS = ("provider", "model", "max_tokens", "timeout_ms", "effort", "stop_sequences") +TEXT_EVENT_STREAM = "text/event-stream" +# server-side refusal fallbacks, "default" form: the Claude API only, current models only +FALLBACKS_BETA = "server-side-fallback-2026-07-01" +FALLBACK_MODEL_PREFIXES = ("claude-opus-5", "claude-fable-5", "claude-sonnet-5-5") +MISSING_CREDENTIAL = "LLM provider credential missing - set ANTHROPIC_API_KEY in the environment" +# the SDK raises a bare TypeError, before sending anything, when no credential resolves +CREDENTIAL_TYPE_ERROR = "Could not resolve authentication method" + + +def _prop(key: str, default: str | None = None) -> str | None: + return app_config().get_property(key, default) + + +# --- the request --------------------------------------------------------------------- + + +@dataclass(frozen=True) +class LlmRequest: + """A validated request. The contract's one request surface, shared by both routes.""" + + model: str + max_tokens: int + timeout_ms: int + max_retries: int + messages: list[dict[str, str]] + system: str | None + effort: str | None + stop_sequences: list[str] | None + schema: dict[str, Any] | None + fallbacks: bool + + +def _whole_number(name: str, value: Any, minimum: int) -> int: + """A whole number from a JSON number or a config string, or a 400 naming the field.""" + valid = (isinstance(value, int) and not isinstance(value, bool)) or ( + isinstance(value, str) and value.strip().isdigit() + ) + if not valid or int(value) < minimum: + raise AppException(400, f"{name} must be a whole number >= {minimum}") + return int(value) + + +def _setting(params: dict[str, Any], key: str, config_key: str, default: str) -> Any: + """Precedence: the call's param, then the config key, then the built-in default.""" + value = params.get(key) + if value is None: + value = _prop(config_key, default) + return value + + +def _turns(body: dict[str, Any]) -> list[dict[str, str]]: + messages = body.get("messages") + if not isinstance(messages, list) or not messages: + return [{"role": "user", "content": str(body["prompt"])}] + turns: list[dict[str, str]] = [] + for index, turn in enumerate(messages): + if not isinstance(turn, dict) or turn.get("role") not in ROLES: + raise AppException(400, f"messages[{index}].role must be one of: {', '.join(ROLES)}") + content = turn.get("content") + if not isinstance(content, str) or not content: + raise AppException(400, f"messages[{index}].content must be a non-empty string") + turns.append({"role": str(turn["role"]), "content": content}) + return turns + + +def _params(body: dict[str, Any]) -> dict[str, Any]: + """The call's params: a map holding only keys this contract supports.""" + raw = body.get("params") + if raw is not None and not isinstance(raw, dict): + raise AppException(400, "params must be a map") + params: dict[str, Any] = dict(raw or {}) + unknown = sorted(key for key in params if key not in PARAMS) + if unknown: + raise AppException( + 400, f"unsupported params: {', '.join(unknown)} - supported: {', '.join(PARAMS)}" + ) + return params + + +def _check_provider(params: dict[str, Any]) -> None: + provider = str(_setting(params, "provider", "llm.provider", DEFAULT_PROVIDER)).lower() + if provider not in PROVIDERS: + raise AppException( + 400, f"unknown LLM provider '{provider}' - this helper serves: {', '.join(PROVIDERS)}" + ) + + +def _system(body: dict[str, Any]) -> str | None: + system = body.get("system") + if system is not None and not isinstance(system, str): + raise AppException(400, "system must be a string") + return system or None + + +def _effort(params: dict[str, Any]) -> str | None: + effort = params.get("effort") or _prop("llm.effort") + if effort is None: + return None + effort = str(effort).lower() + if effort not in EFFORTS: + raise AppException(400, f"params.effort must be one of: {', '.join(EFFORTS)}") + return effort + + +def _stop_sequences(params: dict[str, Any]) -> list[str] | None: + stops = params.get("stop_sequences") + if stops is None: + return None + if not (isinstance(stops, list) and all(isinstance(s, str) for s in stops)): + raise AppException(400, "params.stop_sequences must be a list of strings") + return stops or None + + +def _schema(body: dict[str, Any], streaming: bool) -> dict[str, Any] | None: + schema = body.get("schema") + if schema is None: + return None + if streaming: + raise AppException(400, "schema is not part of the streaming contract - use llm.chat") + if not isinstance(schema, dict): + raise AppException(400, "schema must be a JSON schema map") + # a closed schema is what a bounded verdict wants + return {"additionalProperties": False, **schema} + + +def _fallbacks_enabled() -> bool: + mode = str(_prop("llm.fallbacks", "default")).lower() + if mode not in ("default", "off"): + raise AppException(500, "llm.fallbacks must be 'default' or 'off'") + return mode == "default" + + +def prepare(body: Body, *, streaming: bool) -> LlmRequest: + """Validate a request body into an LlmRequest, or raise AppException(400). + + The checks run in a fixed order, the same in the Node.js twin: body, params and provider, + the numbers, the turns, system, effort, stop sequences, schema, the fallbacks switch. + """ + turns = body.get("messages") if isinstance(body, dict) else None + has_turns = isinstance(turns, list) and bool(turns) + if not isinstance(body, dict) or not (body.get("prompt") or has_turns): + raise AppException(400, "missing 'prompt' or 'messages'") + params = _params(body) + _check_provider(params) + model = str(_setting(params, "model", "llm.model", DEFAULT_MODEL)) + max_tokens = _whole_number( + "params.max_tokens", + _setting(params, "max_tokens", "llm.max.tokens", str(DEFAULT_MAX_TOKENS)), + 1, + ) + timeout_ms = _whole_number( + "params.timeout_ms", + _setting(params, "timeout_ms", "llm.timeout.ms", str(DEFAULT_TIMEOUT_MS)), + 1, + ) + max_retries = _whole_number( + "llm.max.retries", _prop("llm.max.retries", str(DEFAULT_MAX_RETRIES)), 0 + ) + return LlmRequest( + model=model, + max_tokens=max_tokens, + timeout_ms=timeout_ms, + max_retries=max_retries, + messages=_turns(body), + system=_system(body), + effort=_effort(params), + stop_sequences=_stop_sequences(params), + schema=_schema(body, streaming), + fallbacks=_fallbacks_enabled(), + ) + + +# --- the backend seam ------------------------------------------------------------------ + + +def _same_model(model: str) -> str: + """The model id as the Claude API takes it; another route may spell it differently.""" + return model + + +@dataclass(frozen=True) +class Backend: + """One way to reach Claude: the client, and what this route to the model supports. + + This is the seam for a second route to the same models - AWS Bedrock through IAM is the + planned one (see README.md, "Backends"): build its client in get_backend(), give it a + provider_model that prefixes the model id, report no server-side fallbacks, and describe + a missing credential. Nothing else in this module knows which route answered. + """ + + name: str + client: Any + supports_fallbacks: bool + provider_model: Callable[[str], str] = _same_model + + def credential_problem(self) -> str | None: + """What is missing for a call to be sent, or None. No network traffic.""" + names = ("api_key", "auth_token", "credentials", "custom_auth") + if any(getattr(self.client, name, None) for name in names): + return None + return MISSING_CREDENTIAL + + +_backend: Backend | None = None # built on first use - module-level so tests can inject + + +def get_backend() -> Backend: + global _backend + backend = _backend + if backend is None: + name = str(_prop("llm.backend", DEFAULT_BACKEND) or DEFAULT_BACKEND).lower() + if name not in BACKENDS: + raise AppException( + 501, f"unknown LLM backend '{name}' - this helper serves: {', '.join(BACKENDS)}" + ) + # credentials resolve per call, so the app starts (and reports its health) without one + backend = Backend(name, anthropic.AsyncAnthropic(), supports_fallbacks=True) + _backend = backend + return backend + + +def _plan(request: LlmRequest, backend: Backend) -> tuple[Any, dict[str, Any]]: + """The SDK call for a request: the messages namespace to use and its keyword arguments.""" + client = backend.client.with_options( + timeout=request.timeout_ms / 1000, max_retries=request.max_retries + ) + kwargs: dict[str, Any] = { + "model": backend.provider_model(request.model), + "max_tokens": request.max_tokens, + "messages": request.messages, + } + if request.system: + kwargs["system"] = request.system + output: dict[str, Any] = {} + if request.schema is not None: + output["format"] = {"type": "json_schema", "schema": request.schema} + if request.effort: + output["effort"] = request.effort + if output: + kwargs["output_config"] = output + if request.stop_sequences: + kwargs["stop_sequences"] = request.stop_sequences + if ( + request.fallbacks + and backend.supports_fallbacks + and request.model.startswith(FALLBACK_MODEL_PREFIXES) + ): + kwargs["betas"] = [FALLBACKS_BETA] + kwargs["fallbacks"] = "default" + return client.beta.messages, kwargs + return client.messages, kwargs + + +# --- one provider call, shared by both routes ------------------------------------------ + + +@dataclass(frozen=True) +class Completion: + text: str + model: str + stop_reason: str + input_tokens: int + output_tokens: int + request_id: str | None + refusal: dict[str, Any] | None + + +def _refusal(details: Any) -> dict[str, Any] | None: + if details is None: + return None + return { + "category": getattr(details, "category", None), + "explanation": getattr(details, "explanation", None), + } + + +def _request_id_of(stream: Any) -> str | None: + """The API's request id, read from the response headers of an open stream.""" + headers: Any = getattr(getattr(stream, "response", None), "headers", None) + if headers is None: + return None + return headers.get("request-id") + + +async def _complete( + request: LlmRequest, + backend: Backend, + on_text: Callable[[str], None] | None = None, +) -> Completion: + """Run the call on the SDK's streaming transport. The transport never trips the SDK's + long-request guard, however large max_tokens is; llm.chat just takes the final message + while llm.stream forwards the token batches as they arrive.""" + api, kwargs = _plan(request, backend) + async with api.stream(**kwargs) as stream: + request_id = _request_id_of(stream) + if on_text is not None: + async for text in stream.text_stream: + if text: + on_text(text) + message = await stream.get_final_message() + return Completion( + text="".join(b.text for b in message.content if getattr(b, "type", "") == "text"), + model=message.model, + stop_reason=message.stop_reason or "", + input_tokens=message.usage.input_tokens, + output_tokens=message.usage.output_tokens, + request_id=request_id or getattr(message, "_request_id", None), + refusal=_refusal(getattr(message, "stop_details", None)), + ) + + +# --- outcomes: a reply with nothing usable is an error, never an empty success --------- + + +def _judge(request: LlmRequest, done: Completion, has_content: bool) -> Any: + """The parsed JSON for a schema request (None otherwise), or a 422 when the reply is + refused, empty, or cut off before it could be used.""" + if done.stop_reason == "refusal" and (request.schema is not None or not has_content): + category = (done.refusal or {}).get("category") + raise AppException( + 422, + "LLM refused the request - stop_reason=refusal" + + (f", category={category}" if category else ""), + ) + if not has_content: + hint = ( + " (raise params.max_tokens or lower params.effort)" + if done.stop_reason == "max_tokens" + else "" + ) + raise AppException( + 422, + f"LLM reply is empty - stop_reason={done.stop_reason}, " + f"output_tokens={done.output_tokens}{hint}", + ) + if request.schema is None: + return None + try: + return json.loads(done.text) + except ValueError as exc: + hint = ( + " (the reply was cut off - raise params.max_tokens)" + if done.stop_reason == "max_tokens" + else "" + ) + raise AppException( + 422, + "LLM reply is not valid JSON for the requested schema - " + f"stop_reason={done.stop_reason}{hint}", + ) from exc + + +# --- provider failures as the portable error contract ------------------------------------ + + +def _detail(exc: anthropic.APIStatusError) -> str: + """status, error type, message and request id - the same text on every runtime.""" + body = exc.body if isinstance(exc.body, dict) else {} + raw = body.get("error") + error: dict[str, Any] = raw if isinstance(raw, dict) else {} + kind = error.get("type") or "error" + message = error.get("message") or exc.message + request_id = f" (request_id {exc.request_id})" if exc.request_id else "" + return f"{exc.status_code} {kind}: {message}{request_id}" + + +def _status_failure(exc: anthropic.APIStatusError) -> AppException: + """An HTTP error from the API, its status passed through (429 named for what it is).""" + if isinstance(exc, anthropic.RateLimitError): + return AppException(429, f"LLM provider rate limit - {_detail(exc)}") + status = exc.status_code if 400 <= exc.status_code <= 599 else 502 + return AppException(status, f"LLM provider error - {_detail(exc)}") + + +def _failure(exc: BaseException, timeout_ms: int) -> AppException | None: + """Map a provider failure to AppException(status, message); None for anything else.""" + if isinstance(exc, AppException): + return exc + if isinstance(exc, (asyncio.TimeoutError, anthropic.APITimeoutError)): + return AppException(408, f"LLM request timed out after {timeout_ms} ms") + if isinstance(exc, anthropic.APIStatusError): + return _status_failure(exc) + if isinstance(exc, anthropic.APIConnectionError): + return AppException(503, f"LLM provider unreachable - {exc}") + if isinstance(exc, TypeError) and CREDENTIAL_TYPE_ERROR in str(exc): + return AppException(503, MISSING_CREDENTIAL) + return None + + +def _annotate(done: Completion) -> None: + """Usage rides the trace record, so telemetry shows what a call cost (never its text).""" + annotate_trace("llm_model", done.model) + annotate_trace("llm_stop_reason", done.stop_reason) + annotate_trace("llm_input_tokens", str(done.input_tokens)) + annotate_trace("llm_output_tokens", str(done.output_tokens)) + if done.request_id: + annotate_trace("llm_request_id", done.request_id) + + +def _summary(route: str, done: Completion, started: float) -> None: + # model and usage only - a prompt or a completion never reaches a log + log.info( + "%s model=%s stop_reason=%s input_tokens=%d output_tokens=%d request_id=%s elapsed_ms=%d", + route, + done.model, + done.stop_reason, + done.input_tokens, + done.output_tokens, + done.request_id or "none", + round((time.perf_counter() - started) * 1000), + ) + + +# --- the functions ----------------------------------------------------------------------- + + +@preload(route="llm.chat", instances=50) +async def llm_chat(_headers: dict[str, str], body: Body) -> dict[str, Any]: + """Single-shot completion - the AI node a graph's ``graph.task`` or a flow's task calls. + + Input (map): + prompt | messages single-turn text, or conversation turns [{role, content}] + system optional system prompt + schema optional JSON schema -> structured output (the graph needs parseable + verdicts for decision routing; additionalProperties defaults to false) + params model, max_tokens, timeout_ms, effort, stop_sequences, provider + + Output (map): text | data, model, stop_reason, usage {input_tokens, output_tokens}, + request_id; stop_details when the model refused. + + ``params.timeout_ms`` bounds the whole call, SDK retries included - the x-ttl pattern. + A reply that carries nothing usable (refused, empty, or a schema reply cut off) is a 422. + """ + request = prepare(body, streaming=False) + started = time.perf_counter() + try: + done = await asyncio.wait_for( + _complete(request, get_backend()), timeout=request.timeout_ms / 1000 + ) + except Exception as exc: # anything unmapped is re-raised below + failure = _failure(exc, request.timeout_ms) + if failure is None: + raise + log.warning("llm.chat failed - status=%d", failure.status) + raise failure from exc + data = _judge(request, done, has_content=bool(done.text)) + _annotate(done) + _summary("llm.chat", done, started) + result: dict[str, Any] = { + "model": done.model, + "stop_reason": done.stop_reason, + "usage": {"input_tokens": done.input_tokens, "output_tokens": done.output_tokens}, + } + if request.schema is not None: + result["data"] = data + else: + result["text"] = done.text + if done.request_id: + result["request_id"] = done.request_id + if done.stop_reason == "refusal" and done.refusal: + result["stop_details"] = done.refusal + return result + + +@preload(route="llm.stream", instances=50, interceptor=True) +async def llm_stream(headers: dict[str, str], event: EventEnvelope) -> None: + """Streaming completion: the model's real token batches over the multi-shot reply + contract - a calling engine renders them progressively out its own HTTP edge (SSE). + + Same request surface as llm.chat minus ``schema`` (a verdict is a single-shot reply). + ``params.timeout_ms`` is the idle allowance between events, not a total deadline - a + stream runs as long as tokens keep flowing. The terminal event's trailing metadata + carries model, stop_reason, usage, request_id and the trace and business correlation + ids. A stream that ends with no token at all fails in-band with a 422. + """ + out = EventStreamWriter.from_request(event) + started = time.perf_counter() + try: + request = prepare(event.body, streaming=True) + backend = get_backend() + except AppException as exc: + out.fail(exc) + return + info = get_trace() + meta: dict[str, Any] = { + "language": "python", + "trace_id": info.trace_id if info else None, + "my_correlation_id": headers.get("my_correlation_id"), + } + frames = 0 + batch_log = str(_prop("llm.log.batches", "false")).lower() == "true" + + def forward(text: str) -> None: + nonlocal frames + if frames == 0: + # the head rides the first token; a stream that never gets one fails cleanly + out.first(200, TEXT_EVENT_STREAM) + out.write(text) + frames += 1 + if batch_log: + # the diagnostics switch: a batch's number, size and arrival time - never its text + elapsed = round((time.perf_counter() - started) * 1000) + log.info("llm.stream batch=%d chars=%d t_ms=%d", frames, len(text), elapsed) + + try: + done = await _complete(request, backend, on_text=forward) + _judge(request, done, has_content=frames > 0) + except Exception as exc: # anything unmapped is re-raised below + failure = _failure(exc, request.timeout_ms) + if failure is None: + raise + log.warning("llm.stream failed - status=%d frames=%d", failure.status, frames) + out.fail(failure) + return + _annotate(done) + _summary("llm.stream", done, started) + trailing: dict[str, Any] = { + "model": done.model, + "stop_reason": done.stop_reason, + "usage": {"input_tokens": done.input_tokens, "output_tokens": done.output_tokens}, + **meta, + } + if done.request_id: + trailing["request_id"] = done.request_id + if done.stop_reason == "refusal" and done.refusal: + trailing["stop_details"] = done.refusal + out.close(trailing) + + +@preload(route="llm.health", instances=5, private=True) +async def llm_health(headers: dict[str, str], _body: Body) -> Any: + """Health check speaking the engines' interface contract (type=info / type=health). + + Activated for the /health actuator endpoint by mandatory.health.dependencies in + resources/application.yml. It reports whether a call could be sent (a credential is + present) without any network traffic, so a probe never spends a token. + """ + backend = get_backend() + if headers.get("type") == "info": + return { + "service": "llm.helper", + "href": "http://127.0.0.1", + "backend": backend.name, + "model": _prop("llm.model", DEFAULT_MODEL), + } + problem = backend.credential_problem() + if problem: + raise AppException(503, problem) + return "llm.helper is running fine" + + +if __name__ == "__main__": + platform.run() diff --git a/examples/llm-helper/resources/application.yml b/examples/llm-helper/resources/application.yml new file mode 100644 index 0000000..94cfe03 --- /dev/null +++ b/examples/llm-helper/resources/application.yml @@ -0,0 +1,49 @@ +# The LLM helper's configuration - the engines' "resources" convention. +# mercury-serve loads resources/application.yml from the working directory, or from the +# resources folder next to the application file (this one). Any key can be overridden at +# run time with the engines' -D syntax, e.g.: +# mercury-serve examples/llm-helper/llm_helper.py -Dllm.model=claude-haiku-4-5 + +application.name: 'llm-helper' +info.app.description: 'Mercury Composable LLM helper (llm.chat / llm.stream on the Anthropic SDK)' + +rest.server.port: 8086 + +# text (default) | json (pretty-printed) | compact (single-line JSONL) +log.format: text +log.level: INFO + +# Actuator /health dependencies - routes of health check functions speaking the +# engines' type=info / type=health contract (llm.health in llm_helper.py). It reports +# whether a call could be sent (a credential is present) with no network traffic. +mandatory.health.dependencies: 'llm.health' + +# Opt-in lists for the /env endpoint - nothing is ever dumped wholesale. +show.env.variables: 'LOG_LEVEL' +show.application.properties: 'application.name, rest.server.port, llm.backend, llm.model' + +# --- the helper's settings (a call's params.* win over these) ----------------------------- +# The credential is never configured here: ANTHROPIC_API_KEY comes from the environment. +llm.backend: 'anthropic' +llm.model: 'claude-opus-5-5' +llm.max.tokens: 16000 +# llm.chat: the whole call, SDK retries included. llm.stream: the idle allowance between events. +llm.timeout.ms: 60000 +llm.max.retries: 2 +# server-side refusal fallbacks on the models that support them: default | off +llm.fallbacks: 'default' +# low | medium | high | xhigh | max - left unset, the model's own default applies (a model +# that takes no effort parameter rejects one, so set this only for a model that does) +# llm.effort: 'medium' +# diagnostics for a slow or bursty stream: log each batch's number, size and arrival time (never +# its text), so the hop that holds batches back can be told apart from the one that forwards them +# llm.log.batches: true + +# OpenTelemetry forwarder (opt-in) - the engines' opentelemetry-forwarder twin. Switch it on +# at run time with -Dotel.forwarding=true; the endpoint and credential come from the +# environment (a Dynatrace endpoint ends in /api/v2/otlp/v1/traces; the header is the whole +# 'Authorization=Api-Token' prefix, the token its value). Header names are logged, never values. +otel.forwarding: false +otel.exporter.otlp.endpoint: '${OTLP_API_ENDPOINT:http://localhost:4318/v1/traces}' +otel.exporter.otlp.headers: '${OTLP_AUTH_HEADER} ${OTLP_TOKEN}' +otel.service.name: '${OTLP_SERVICE_NAME:llm-helper}' diff --git a/mkdocs.yml b/mkdocs.yml index 0fa8072..883b710 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -126,5 +126,6 @@ nav: - HTTP Surface Reference: guides/http-surface-reference.md - Interop Test Report — Progressive Rendering: test-reports/progressive-rendering-interop.md - Test Report — OpenTelemetry Certification: test-reports/otel-dynatrace-certification.md + - Test Report — The LLM helper: test-reports/llm-helper-certification.md - Release Notes: https://github.com/Accenture/mercury-python/blob/main/CHANGELOG.md - Contributing: https://github.com/Accenture/mercury-python/blob/main/CONTRIBUTING.md diff --git a/pyproject.toml b/pyproject.toml index 1578cc4..797695d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -38,9 +38,9 @@ Repository = "https://github.com/Accenture/mercury-python" Changelog = "https://github.com/Accenture/mercury-python/blob/main/CHANGELOG.md" [project.optional-dependencies] -# the llm.chat demo function (examples/demo_app.py) - the package itself stays SDK-free -llm = ["anthropic>=1,<2", "google-genai>=2,<3"] -dev = ["pytest>=8", "pytest-asyncio>=0.23", "ruff>=0.9", "anthropic>=1,<2", "google-genai>=2,<3"] +# the LLM helper app (examples/llm-helper) - the package itself stays SDK-free +llm = ["anthropic>=1,<2"] +dev = ["pytest>=8", "pytest-asyncio>=0.23", "ruff>=0.9", "anthropic>=1,<2"] [project.scripts] mercury-serve = "mercury_composable.cli:main" diff --git a/tests/test_llm_chat.py b/tests/test_llm_chat.py deleted file mode 100644 index d2e3142..0000000 --- a/tests/test_llm_chat.py +++ /dev/null @@ -1,252 +0,0 @@ -"""The llm.chat demo function (agent-orchestration experiment E0) - token-free: -the provider client is a fake, so these tests pin the adapter contract (request -shaping, structured output, usage surfacing, provider-error mapping) without -spending tokens or needing credentials. -""" - -import importlib.util -import json -import sys -from pathlib import Path -from types import SimpleNamespace -from typing import Any - -import anthropic -import pytest - -from mercury_composable import AppException - -_DEMO = Path(__file__).resolve().parent.parent / "examples" / "demo_app.py" -_spec = importlib.util.spec_from_file_location("demo_app_under_test", _DEMO) -assert _spec is not None -assert _spec.loader is not None -# typed Any: the module is loaded dynamically from a file path, so its attributes -# are unknowable statically - Any tells every analyzer to trust the runtime -demo_app: Any = importlib.util.module_from_spec(_spec) -sys.modules.setdefault("demo_app_under_test", demo_app) -_spec.loader.exec_module(demo_app) - - -def _provider_response(text: str) -> SimpleNamespace: - return SimpleNamespace( - content=[SimpleNamespace(type="text", text=text)], - usage=SimpleNamespace(input_tokens=42, output_tokens=7), - stop_reason="end_turn", - model="claude-opus-5", - ) - - -class _FakeMessages: - def __init__(self, outcome: Any): - self.outcome = outcome - self.requests: list[dict[str, Any]] = [] - - async def create(self, **kwargs: Any) -> Any: - self.requests.append(kwargs) - if isinstance(self.outcome, Exception): - raise self.outcome - return self.outcome - - -class _FakeClient: - def __init__(self, outcome: Any): - self.messages = _FakeMessages(outcome) - self.timeouts: list[float] = [] - - def with_options(self, timeout: float) -> "_FakeClient": - self.timeouts.append(timeout) - return self - - -def _patch_demo(monkeypatch: pytest.MonkeyPatch, attr: str, value: Any) -> None: - """monkeypatch.setattr on the dynamically loaded demo module. The attribute - name rides through a parameter because no static analyzer can verify names - on a module loaded from a file path; monkeypatch still validates the name - at run time and restores the original on teardown.""" - monkeypatch.setattr(demo_app, attr, value) - - -def _install(monkeypatch: pytest.MonkeyPatch, outcome: Any) -> _FakeClient: - fake = _FakeClient(outcome) - _patch_demo(monkeypatch, "_llm_client", fake) - return fake - - -async def test_prompt_mode_returns_text_usage_and_stop_reason( - monkeypatch: pytest.MonkeyPatch, -) -> None: - fake = _install(monkeypatch, _provider_response("Paris")) - result = await demo_app.llm_chat({}, {"prompt": "Capital of France?"}) - assert result == { - "model": "claude-opus-5", - "stop_reason": "end_turn", - "usage": {"input_tokens": 42, "output_tokens": 7}, - "text": "Paris", - } - request = fake.messages.requests[0] - # provider defaults: the documented model and max_tokens, single-turn message shaping - assert request["model"] == demo_app.LLM_DEFAULT_MODELS["anthropic"] - assert request["max_tokens"] == demo_app.LLM_DEFAULT_MAX_TOKENS - assert request["messages"] == [{"role": "user", "content": "Capital of France?"}] - assert "output_config" not in request - # the default time budget maps onto the SDK timeout (seconds) - assert fake.timeouts == [demo_app.LLM_DEFAULT_TIMEOUT_MS / 1000] - - -async def test_schema_mode_requests_structured_output_and_parses_data( - monkeypatch: pytest.MonkeyPatch, -) -> None: - schema = { - "type": "object", - "properties": {"label": {"type": "string", "enum": ["question", "bug", "feature"]}}, - "required": ["label"], - "additionalProperties": False, - } - fake = _install(monkeypatch, _provider_response(json.dumps({"label": "bug"}))) - result = await demo_app.llm_chat( - {}, - { - "prompt": "Classify: the app crashes on save", - "system": "You are a support triage assistant.", - "schema": schema, - "params": {"model": "claude-opus-5", "max_tokens": 512, "timeout_ms": 5000}, - }, - ) - assert result["data"] == {"label": "bug"} - assert "text" not in result - request = fake.messages.requests[0] - assert request["output_config"] == {"format": {"type": "json_schema", "schema": schema}} - assert request["system"] == "You are a support triage assistant." - assert request["max_tokens"] == 512 - assert fake.timeouts == [5.0] - - -async def test_schema_defaults_to_a_closed_object(monkeypatch: pytest.MonkeyPatch) -> None: - # a bounded verdict wants a closed schema - additionalProperties defaults to - # false when the caller omits it (callers that set it keep their value) - fake = _install(monkeypatch, _provider_response(json.dumps({"label": "bug"}))) - await demo_app.llm_chat({}, {"prompt": "x", "schema": {"type": "object"}}) - sent = fake.messages.requests[0]["output_config"]["format"]["schema"] - assert sent == {"additionalProperties": False, "type": "object"} - - -async def test_conversation_turns_pass_through(monkeypatch: pytest.MonkeyPatch) -> None: - fake = _install(monkeypatch, _provider_response("hi")) - turns = [{"role": "user", "content": "hello"}] - await demo_app.llm_chat({}, {"messages": turns}) - assert fake.messages.requests[0]["messages"] == turns - - -async def test_missing_prompt_and_messages_is_a_400(monkeypatch: pytest.MonkeyPatch) -> None: - _install(monkeypatch, _provider_response("unused")) - with pytest.raises(AppException) as error: - await demo_app.llm_chat({}, {"schema": {}}) - assert error.value.status == 400 - - -def _gemini_response(text: str) -> SimpleNamespace: - return SimpleNamespace( - text=text, - model_version="gemini-3.6-flash", - usage_metadata=SimpleNamespace(prompt_token_count=11, candidates_token_count=3), - candidates=[SimpleNamespace(finish_reason=SimpleNamespace(name="STOP"))], - ) - - -class _FakeGeminiModels: - def __init__(self, outcome: Any): - self.outcome = outcome - self.requests: list[dict[str, Any]] = [] - - async def generate_content(self, **kwargs: Any) -> Any: - self.requests.append(kwargs) - if isinstance(self.outcome, Exception): - raise self.outcome - return self.outcome - - -class _FakeGemini: - def __init__(self, outcome: Any): - self.aio = SimpleNamespace(models=_FakeGeminiModels(outcome)) - - -def _install_gemini(monkeypatch: pytest.MonkeyPatch, outcome: Any) -> _FakeGemini: - fake = _FakeGemini(outcome) - _patch_demo(monkeypatch, "_gemini_client", fake) - return fake - - -async def test_gemini_provider_speaks_the_same_contract( - monkeypatch: pytest.MonkeyPatch, -) -> None: - # provider swapped per call (or by the llm.provider config key) - the caller's - # contract and the graph above it do not change - schema = {"type": "object", "properties": {"label": {"type": "string"}}} - fake = _install_gemini(monkeypatch, _gemini_response(json.dumps({"label": "bug"}))) - result = await demo_app.llm_chat( - {}, - { - "prompt": "Classify: the app crashes on save", - "system": "You are a support triage assistant.", - "schema": schema, - "params": {"provider": "gemini", "timeout_ms": 5000}, - }, - ) - assert result == { - "model": "gemini-3.6-flash", - "stop_reason": "STOP", - "usage": {"input_tokens": 11, "output_tokens": 3}, - "data": {"label": "bug"}, - } - request = fake.aio.models.requests[0] - assert request["model"] == demo_app.LLM_DEFAULT_MODELS["gemini"] - config = request["config"] - assert config.system_instruction == "You are a support triage assistant." - assert config.response_mime_type == "application/json" - assert config.response_json_schema == {"additionalProperties": False, **schema} - # HttpOptions.timeout is milliseconds - timeout_ms passes through unchanged - assert config.http_options is not None - assert config.http_options.timeout == 5000 - contents = request["contents"] - assert len(contents) == 1 - assert contents[0].role == "user" - - -async def test_gemini_errors_map_to_envelope_status(monkeypatch: pytest.MonkeyPatch) -> None: - from google.genai import errors as genai_errors - - _install_gemini(monkeypatch, genai_errors.APIError(429, {"error": {"message": "quota"}})) - with pytest.raises(AppException) as error: - await demo_app.llm_chat({}, {"prompt": "x", "params": {"provider": "gemini"}}) - assert error.value.status == 429 - - -async def test_unknown_provider_is_a_400(monkeypatch: pytest.MonkeyPatch) -> None: - _install(monkeypatch, _provider_response("unused")) - with pytest.raises(AppException) as error: - await demo_app.llm_chat({}, {"prompt": "x", "params": {"provider": "openai"}}) - assert error.value.status == 400 - - -async def test_provider_errors_map_to_envelope_status(monkeypatch: pytest.MonkeyPatch) -> None: - # subclass the SDK exceptions so no httpx plumbing is needed - isinstance is - # what the mapping chain dispatches on - class _RateLimited(anthropic.RateLimitError): - def __init__(self) -> None: - Exception.__init__(self, "rate limited") - self.status_code = 429 - - class _Invalid(anthropic.APIStatusError): - def __init__(self) -> None: - Exception.__init__(self, "bad request") - self.status_code = 400 - - class _Unreachable(anthropic.APIConnectionError): - def __init__(self) -> None: - Exception.__init__(self, "connect timeout") - - for boom, expected in ((_RateLimited(), 429), (_Invalid(), 400), (_Unreachable(), 503)): - _install(monkeypatch, boom) - with pytest.raises(AppException) as error: - await demo_app.llm_chat({}, {"prompt": "x"}) - assert error.value.status == expected diff --git a/tests/test_llm_helper.py b/tests/test_llm_helper.py new file mode 100644 index 0000000..4b69982 --- /dev/null +++ b/tests/test_llm_helper.py @@ -0,0 +1,531 @@ +"""The LLM helper (examples/llm-helper/llm_helper.py) - token-free. + +One shared contract file, tests/vectors/llm-helper-vectors.json (byte-identical in +mercury-nodejs), drives every case below against a fake of the Anthropic SDK: the exact SDK +call, the reply map, the error contract and the streaming segments are pinned without +spending a token or needing a credential, and the Node.js twin runs the same file, so the two +helpers cannot drift apart. The tests after the vector runs pin what a vector cannot: token +batches are never held back, no prompt text reaches a log, the trace annotations, the health +route and the backend seam. +""" + +from __future__ import annotations + +import asyncio +import hashlib +import importlib.util +import json +import logging +import sys +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import anthropic +import httpx2 +import pytest + +from mercury_composable import AppException + +_TESTS = Path(__file__).resolve().parent +_HELPER = _TESTS.parent / "examples" / "llm-helper" / "llm_helper.py" +_VECTORS = _TESTS / "vectors" / "llm-helper-vectors.json" +# the Node.js twin pins the same digest: change the file in both packs or in neither +_VECTORS_SHA256 = "1f4823d9259ed6ff26e96d044b5f6b3c7bc9b342cc8ad76047434c2d3a2b8fa0" + +_spec = importlib.util.spec_from_file_location("llm_helper_under_test", _HELPER) +assert _spec is not None +assert _spec.loader is not None +# typed Any: the module is loaded dynamically from a file path, so its attributes are +# unknowable statically - Any tells every analyzer to trust the runtime +helper: Any = importlib.util.module_from_spec(_spec) +sys.modules.setdefault("llm_helper_under_test", helper) +_spec.loader.exec_module(helper) + +VECTORS: dict[str, Any] = json.loads(_VECTORS.read_text(encoding="utf-8")) +CHAT_CASES: list[dict[str, Any]] = [c for c in VECTORS["cases"] if c["route"] == "llm.chat"] +STREAM_CASES: list[dict[str, Any]] = [c for c in VECTORS["cases"] if c["route"] == "llm.stream"] + +_REQUEST = httpx2.Request("POST", "https://api.anthropic.com/v1/messages") +_STATUS_CLASSES: dict[int, type[anthropic.APIStatusError]] = { + 400: anthropic.BadRequestError, + 401: anthropic.AuthenticationError, + 403: anthropic.PermissionDeniedError, + 404: anthropic.NotFoundError, + 422: anthropic.UnprocessableEntityError, + 429: anthropic.RateLimitError, +} +_NO_CREDENTIAL = ( + '"Could not resolve authentication method. Expected one of api_key, auth_token, or ' + "credentials to be set. Or for one of the `X-Api-Key` or `Authorization` headers to be " + 'explicitly omitted"' +) + + +def _status_error(spec: dict[str, Any]) -> anthropic.APIStatusError: + """A real SDK error object, built the way the SDK builds one from a response.""" + status = int(spec["status"]) + headers = {"request-id": spec["request_id"]} if spec.get("request_id") else {} + body = {"type": "error", "error": {"type": spec["type"], "message": spec["message"]}} + response = httpx2.Response(status, request=_REQUEST, headers=headers, json=body) + error_class = _STATUS_CLASSES.get(status) + if error_class is None: + error_class = anthropic.InternalServerError if status >= 500 else anthropic.APIStatusError + return error_class(spec["message"], response=response, body=body) + + +class _Ledger: + def __init__(self) -> None: + self.calls: list[tuple[str, dict[str, Any]]] = [] + self.options: list[dict[str, int]] = [] + self.cancelled = False + + +class _FakeStream: + """What ``client.messages.stream(...)`` returns: an async context manager.""" + + def __init__(self, provider: dict[str, Any], ledger: _Ledger) -> None: + self._provider = provider + self._ledger = ledger + self.response: Any = None + + async def __aenter__(self) -> Any: + kind = self._provider["kind"] + if kind == "hang": + try: + await asyncio.Event().wait() + except asyncio.CancelledError: + self._ledger.cancelled = True + raise + if kind == "credential": + raise TypeError(_NO_CREDENTIAL) + if kind == "timeout": + raise anthropic.APITimeoutError(request=_REQUEST) + if kind == "connection": + raise anthropic.APIConnectionError(message="Connection error.", request=_REQUEST) + if kind == "status" and not self._provider.get("after"): + raise _status_error(self._provider) + request_id = self._provider.get("request_id") + self.response = SimpleNamespace(headers={"request-id": request_id} if request_id else {}) + return self + + async def __aexit__(self, *_exc: object) -> bool: + return False + + @property + def text_stream(self) -> Any: + provider = self._provider + + async def batches() -> Any: + deltas: list[str] = provider.get("deltas", []) + after = int(provider.get("after", 0)) + for index, text in enumerate(deltas): + if provider["kind"] == "status" and index == after: + raise _status_error(provider) + yield text + if provider["kind"] == "status" and after >= len(deltas): + raise _status_error(provider) + + return batches() + + async def get_final_message(self) -> Any: + provider = self._provider + details: Any = provider.get("stop_details") + return SimpleNamespace( + content=[ + SimpleNamespace(type="text", text=provider.get("text", "".join(provider["deltas"]))) + ], + model=provider["model"], + stop_reason=provider["stop_reason"], + usage=SimpleNamespace(**provider["usage"]), + stop_details=SimpleNamespace(**details) if details else None, + ) + + +class _Namespace: + def __init__(self, name: str, client: _FakeClient) -> None: + self._name = name + self._client = client + + def stream(self, **kwargs: Any) -> _FakeStream: + self._client.ledger.calls.append((self._name, kwargs)) + return self._client.make_stream() + + +class _FakeClient: + """The SDK client as the helper uses it: with_options(), then a message's namespace.""" + + def __init__(self, provider: dict[str, Any] | None) -> None: + self.provider = provider or {} + self.ledger = _Ledger() + self.api_key = "test-credential" + self.messages = _Namespace("messages", self) + self.beta = SimpleNamespace(messages=_Namespace("beta.messages", self)) + + def with_options(self, *, timeout: float, max_retries: int) -> _FakeClient: + self.ledger.options.append( + {"timeout_ms": round(timeout * 1000), "max_retries": max_retries} + ) + return self + + def make_stream(self) -> _FakeStream: + return _FakeStream(self.provider, self.ledger) + + +class _FakeWriter: + def __init__(self) -> None: + self.head: tuple[int, str] | None = None + self.segments: list[Any] = [] + self.trailing: Any = None + self.error: AppException | None = None + + def first(self, status: int, content_type: str, _ttl_seconds: int | None = None) -> None: + self.head = (status, content_type) + + def write(self, segment: Any) -> None: + self.segments.append(segment) + + def close(self, trailing_metadata: Any = None) -> None: + self.trailing = trailing_metadata + + def fail(self, error: Exception) -> None: + assert isinstance(error, AppException) + self.error = error + + +def _install_backend( + monkeypatch: pytest.MonkeyPatch, client: Any, config: dict[str, str] | None = None +) -> None: + settings = dict(config or {}) + monkeypatch.setattr(helper, "_prop", lambda key, default=None: settings.get(key, default)) + monkeypatch.setattr( + helper, "_backend", helper.Backend("anthropic", client, supports_fallbacks=True) + ) + + +def _install_writer(monkeypatch: pytest.MonkeyPatch) -> _FakeWriter: + out = _FakeWriter() + monkeypatch.setattr(helper, "EventStreamWriter", SimpleNamespace(from_request=lambda _e: out)) + return out + + +def _event(body: Any) -> SimpleNamespace: + return SimpleNamespace(body=body) + + +def _assert_sdk(client: _FakeClient, case: dict[str, Any]) -> None: + wanted = case["expect"].get("sdk") + if wanted is not None: + assert client.ledger.calls == [(wanted["namespace"], wanted["params"])] + assert client.ledger.options == [wanted["options"]] + elif case.get("provider") is None: + assert client.ledger.calls == [], "a rejected request must not reach the provider" + + +def _assert_error(error: AppException | None, wanted: dict[str, Any]) -> None: + assert error is not None, "an error was expected" + assert error.status == wanted["status"], error.message + for fragment in wanted["message_contains"]: + assert fragment in error.message, f"{fragment!r} not in {error.message!r}" + + +# --- the shared contract: llm.chat ------------------------------------------------------- + + +def test_the_vector_file_is_the_one_the_node_twin_runs() -> None: + digest = hashlib.sha256(_VECTORS.read_bytes()).hexdigest() + assert digest == _VECTORS_SHA256, "the shared vector file changed - update both packs" + assert VECTORS["contract"] == "llm-helper" + assert VECTORS["version"] == 1 + assert len(CHAT_CASES) + len(STREAM_CASES) == len(VECTORS["cases"]) + + +@pytest.mark.parametrize("case", CHAT_CASES, ids=[c["name"] for c in CHAT_CASES]) +async def test_chat_contract(case: dict[str, Any], monkeypatch: pytest.MonkeyPatch) -> None: + client = _FakeClient(case.get("provider")) + _install_backend(monkeypatch, client, case.get("config")) + result: Any = None + error: AppException | None = None + try: + result = await helper.llm_chat({}, case["body"]) + except AppException as exc: + error = exc + _assert_sdk(client, case) + if "error" in case["expect"]: + _assert_error(error, case["expect"]["error"]) + else: + assert error is None, error + assert result == case["expect"]["result"] + + +# --- the shared contract: llm.stream ----------------------------------------------------- + + +@pytest.mark.parametrize("case", STREAM_CASES, ids=[c["name"] for c in STREAM_CASES]) +async def test_stream_contract(case: dict[str, Any], monkeypatch: pytest.MonkeyPatch) -> None: + client = _FakeClient(case.get("provider")) + _install_backend(monkeypatch, client, case.get("config")) + out = _install_writer(monkeypatch) + await helper.llm_stream(case.get("headers", {}), _event(case["body"])) + expect = case["expect"] + _assert_sdk(client, case) + wanted_head = expect.get("head") + assert out.head == (tuple(wanted_head) if wanted_head else None) + assert out.segments == expect.get("frames", []) + terminal = expect.get("terminal") + if terminal is None: + assert out.trailing is None + else: + assert out.trailing is not None + for key, value in terminal.items(): + assert out.trailing[key] == value, key + assert out.trailing["language"] == "python" + if "error" in expect: + _assert_error(out.error, expect["error"]) + else: + assert out.error is None, out.error + + +# --- progressive rendering: a token batch is never held back ----------------------------- + + +async def test_each_token_batch_reaches_the_caller_before_the_next_is_produced( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """The point of the streaming route is to deliver batches continuously. The fake model + produces batch k only after asserting that the caller already holds batches 0 to k-1 - + a helper that gathered the batches and sent them once would fail this test.""" + out = _install_writer(monkeypatch) + batches = ["Event-", "driven ", "architecture ", "decouples ", "producers."] + + class _Paced(_FakeStream): + @property + def text_stream(self) -> Any: + async def paced() -> Any: + for index, text in enumerate(batches): + assert out.segments == batches[:index], f"batch {index} was produced early" + assert (out.head is not None) == (index > 0), "the head rides the first batch" + await asyncio.sleep(0) # a network read yields to the loop between batches + yield text + + return paced() + + class _PacedClient(_FakeClient): + def make_stream(self) -> _FakeStream: + return _Paced(self.provider, self.ledger) + + provider = { + "kind": "message", "model": "claude-opus-5-5", "stop_reason": "end_turn", + "deltas": batches, "usage": {"input_tokens": 9, "output_tokens": 11}, + } # fmt: skip + _install_backend(monkeypatch, _PacedClient(provider)) + await helper.llm_stream({}, _event({"prompt": "describe it"})) + assert out.error is None + assert out.segments == batches, "every batch is its own segment" + assert out.head == (200, "text/event-stream") + assert out.trailing is not None + assert out.trailing["usage"] == {"input_tokens": 9, "output_tokens": 11} + + +async def test_the_chat_route_sends_nothing_to_a_stream_writer( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """llm.chat is the single-shot route: one reply map, never segments.""" + out = _install_writer(monkeypatch) + provider = { + "kind": "message", "model": "claude-opus-5-5", "stop_reason": "end_turn", + "deltas": ["a", "b"], "usage": {"input_tokens": 1, "output_tokens": 2}, + } # fmt: skip + _install_backend(monkeypatch, _FakeClient(provider)) + result = await helper.llm_chat({}, {"prompt": "x"}) + assert result["text"] == "ab" + assert out.segments == [] + assert out.head is None + + +# --- what a vector cannot say ------------------------------------------------------------- + + +async def test_the_deadline_cancels_the_provider_call(monkeypatch: pytest.MonkeyPatch) -> None: + client = _FakeClient({"kind": "hang"}) + _install_backend(monkeypatch, client) + with pytest.raises(AppException) as raised: + await helper.llm_chat({}, {"prompt": "x", "params": {"timeout_ms": 30}}) + assert raised.value.status == 408 + assert client.ledger.cancelled, "the abandoned call must be cancelled, not left running" + + +async def test_neither_the_prompt_nor_the_reply_reaches_a_log( + monkeypatch: pytest.MonkeyPatch, caplog: pytest.LogCaptureFixture +) -> None: + secret_prompt, secret_reply = "PROMPT-MARKER-4471", "REPLY-MARKER-9023" + provider = { + "kind": "message", "model": "claude-opus-5-5", "stop_reason": "end_turn", + "deltas": [secret_reply], "usage": {"input_tokens": 7, "output_tokens": 5}, + "request_id": "req_log_001", + } # fmt: skip + _install_backend(monkeypatch, _FakeClient(provider)) + _install_writer(monkeypatch) + caplog.set_level(logging.DEBUG) + await helper.llm_chat({}, {"prompt": secret_prompt, "system": "SYSTEM-MARKER-1188"}) + await helper.llm_stream({}, _event({"prompt": secret_prompt})) + assert "llm.chat model=claude-opus-5-5" in caplog.text + assert "llm.stream model=claude-opus-5-5" in caplog.text + assert "request_id=req_log_001" in caplog.text + for marker in (secret_prompt, secret_reply, "SYSTEM-MARKER-1188"): + assert marker not in caplog.text + + +async def test_batch_timing_is_logged_when_asked_and_never_the_text( + monkeypatch: pytest.MonkeyPatch, caplog: pytest.LogCaptureFixture +) -> None: + provider = { + "kind": "message", "model": "claude-opus-5-5", "stop_reason": "end_turn", + "deltas": ["alpha-MARKER-1", "beta-MARKER-2", "gamma"], + "usage": {"input_tokens": 3, "output_tokens": 9}, + } # fmt: skip + _install_backend(monkeypatch, _FakeClient(provider), {"llm.log.batches": "true"}) + _install_writer(monkeypatch) + caplog.set_level(logging.INFO) + await helper.llm_stream({}, _event({"prompt": "x"})) + lines = [r.getMessage() for r in caplog.records if "llm.stream batch=" in r.getMessage()] + assert len(lines) == 3, lines + assert lines[0].startswith("llm.stream batch=1 chars=14 t_ms=") + assert lines[2].startswith("llm.stream batch=3 chars=5 t_ms=") + assert "MARKER" not in caplog.text + + +async def test_batch_timing_is_off_by_default( + monkeypatch: pytest.MonkeyPatch, caplog: pytest.LogCaptureFixture +) -> None: + provider = { + "kind": "message", "model": "claude-opus-5-5", "stop_reason": "end_turn", + "deltas": ["a", "b"], "usage": {"input_tokens": 1, "output_tokens": 2}, + } # fmt: skip + _install_backend(monkeypatch, _FakeClient(provider)) + _install_writer(monkeypatch) + caplog.set_level(logging.DEBUG) + await helper.llm_stream({}, _event({"prompt": "x"})) + assert "llm.stream batch=" not in caplog.text + + +async def test_usage_rides_the_trace_annotations(monkeypatch: pytest.MonkeyPatch) -> None: + seen: dict[str, Any] = {} + monkeypatch.setattr(helper, "annotate_trace", lambda key, value: seen.__setitem__(key, value)) + provider = { + "kind": "message", "model": "claude-opus-5-5", "stop_reason": "end_turn", + "deltas": ["ok"], "usage": {"input_tokens": 20, "output_tokens": 4}, + "request_id": "req_trace_001", + } # fmt: skip + _install_backend(monkeypatch, _FakeClient(provider)) + await helper.llm_chat({}, {"prompt": "x"}) + assert seen == { + "llm_model": "claude-opus-5-5", + "llm_stop_reason": "end_turn", + "llm_input_tokens": "20", + "llm_output_tokens": "4", + "llm_request_id": "req_trace_001", + } + + +@pytest.mark.parametrize( + ("model", "gets_fallbacks"), + [ + ("claude-opus-5-5", True), + ("claude-opus-5", True), + ("claude-fable-5-1", True), + ("claude-sonnet-5-5", True), + ("claude-sonnet-5", False), + ("claude-opus-4-8", False), + ("claude-haiku-4-5", False), + ], +) +def test_refusal_fallbacks_only_for_the_models_that_support_them( + model: str, gets_fallbacks: bool, monkeypatch: pytest.MonkeyPatch +) -> None: + client = _FakeClient(None) + _install_backend(monkeypatch, client) + request = helper.prepare({"prompt": "x", "params": {"model": model}}, streaming=False) + api, kwargs = helper._plan(request, helper.get_backend()) + assert (api is client.beta.messages) == gets_fallbacks + assert ("fallbacks" in kwargs) == gets_fallbacks + assert ("betas" in kwargs) == gets_fallbacks + + +def test_a_backend_can_spell_the_model_id_its_own_way(monkeypatch: pytest.MonkeyPatch) -> None: + """The seam for a second route to the models: its model ids may carry a prefix.""" + client = _FakeClient(None) + monkeypatch.setattr(helper, "_prop", lambda key, default=None: default) + backend = helper.Backend( + "anthropic", client, True, provider_model=lambda model: f"anthropic.{model}" + ) + request = helper.prepare({"prompt": "x"}, streaming=False) + _api, kwargs = helper._plan(request, backend) + assert kwargs["model"] == "anthropic.claude-opus-5-5" + + +def test_a_route_without_server_side_fallbacks_never_sends_them( + monkeypatch: pytest.MonkeyPatch, +) -> None: + client = _FakeClient(None) + monkeypatch.setattr(helper, "_prop", lambda key, default=None: default) + backend = helper.Backend("anthropic", client, supports_fallbacks=False) + request = helper.prepare({"prompt": "x"}, streaming=False) + api, kwargs = helper._plan(request, backend) + assert api is client.messages + assert "fallbacks" not in kwargs + + +# --- the health route and the backend seam ----------------------------------------------- + + +async def test_health_reports_the_backend_and_the_model(monkeypatch: pytest.MonkeyPatch) -> None: + _install_backend(monkeypatch, _FakeClient(None)) + info = await helper.llm_health({"type": "info"}, None) + assert info["service"] == "llm.helper" + assert info["backend"] == "anthropic" + assert info["model"] == "claude-opus-5-5" + + +async def test_health_is_up_when_a_credential_is_present(monkeypatch: pytest.MonkeyPatch) -> None: + _install_backend(monkeypatch, _FakeClient(None)) + assert await helper.llm_health({"type": "health"}, None) == "llm.helper is running fine" + + +async def test_health_is_down_without_a_credential(monkeypatch: pytest.MonkeyPatch) -> None: + client = _FakeClient(None) + client.api_key = None # type: ignore[assignment] + _install_backend(monkeypatch, client) + with pytest.raises(AppException) as raised: + await helper.llm_health({"type": "health"}, None) + assert raised.value.status == 503 + assert "credential missing" in raised.value.message + + +def test_the_real_client_reports_a_credential_it_was_given() -> None: + backend = helper.Backend("anthropic", anthropic.AsyncAnthropic(api_key="test-credential"), True) + assert backend.credential_problem() is None + + +def test_an_unknown_backend_is_a_501_that_names_what_is_served( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(helper, "_backend", None) + monkeypatch.setattr( + helper, "_prop", lambda key, default=None: "bedrock" if key == "llm.backend" else default + ) + with pytest.raises(AppException) as raised: + helper.get_backend() + assert raised.value.status == 501 + assert "unknown LLM backend 'bedrock'" in raised.value.message + assert "this helper serves: anthropic" in raised.value.message + + +def test_the_default_backend_is_the_anthropic_client(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(helper, "_backend", None) + monkeypatch.setattr(helper, "_prop", lambda key, default=None: default) + backend = helper.get_backend() + assert backend.name == "anthropic" + assert backend.supports_fallbacks is True + assert isinstance(backend.client, anthropic.AsyncAnthropic) + assert helper.get_backend() is backend, "the client is built once" diff --git a/tests/test_llm_stream.py b/tests/test_llm_stream.py deleted file mode 100644 index 0941594..0000000 --- a/tests/test_llm_stream.py +++ /dev/null @@ -1,234 +0,0 @@ -"""The llm.stream demo function (progressive token rendering, E0 follow-up) - -token-free: fake provider streams and a recording writer pin the relay contract -(head, ordered token batches, terminal metadata, provider-error fail) without -spending tokens or needing credentials. -""" - -import importlib.util -import sys -from pathlib import Path -from types import SimpleNamespace -from typing import Any - -import anthropic -import pytest - -from mercury_composable import AppException - -_DEMO = Path(__file__).resolve().parent.parent / "examples" / "demo_app.py" -_spec = importlib.util.spec_from_file_location("demo_app_stream_under_test", _DEMO) -assert _spec is not None -assert _spec.loader is not None -# typed Any: the module is loaded dynamically from a file path, so its attributes -# are unknowable statically - Any tells every analyzer to trust the runtime -demo_app: Any = importlib.util.module_from_spec(_spec) -sys.modules.setdefault("demo_app_stream_under_test", demo_app) -_spec.loader.exec_module(demo_app) - - -class _FakeWriter: - def __init__(self) -> None: - self.head: tuple[int, str] | None = None - self.segments: list[Any] = [] - self.trailing: Any = None - self.error: Exception | None = None - - def first(self, status: int, content_type: str) -> None: - self.head = (status, content_type) - - def write(self, segment: Any) -> None: - self.segments.append(segment) - - def close(self, trailing_metadata: Any = None) -> None: - self.trailing = trailing_metadata - - def fail(self, error: Exception) -> None: - self.error = error - - -class _FakeWriterFactory: - instance = _FakeWriter() - - @classmethod - def from_request(cls, _event: Any) -> _FakeWriter: - return cls.instance - - -def _patch_demo(monkeypatch: pytest.MonkeyPatch, attr: str, value: Any) -> None: - """monkeypatch.setattr on the dynamically loaded demo module. The attribute - name rides through a parameter because no static analyzer can verify names - on a module loaded from a file path; monkeypatch still validates the name - at run time and restores the original on teardown.""" - monkeypatch.setattr(demo_app, attr, value) - - -def _install_writer(monkeypatch: pytest.MonkeyPatch) -> _FakeWriter: - fake = _FakeWriter() - _FakeWriterFactory.instance = fake - _patch_demo(monkeypatch, "EventStreamWriter", _FakeWriterFactory) - return fake - - -def _event(body: Any) -> SimpleNamespace: - return SimpleNamespace(body=body) - - -# --- gemini streaming path --------------------------------------------------- - - -def _gemini_chunk(text: str | None, usage: Any = None, finish: str | None = None) -> SimpleNamespace: - return SimpleNamespace( - text=text, - usage_metadata=usage, - candidates=[SimpleNamespace(finish_reason=SimpleNamespace(name=finish))] if finish else [], - model_version="gemini-3.6-flash", - ) - - -class _FakeGeminiStreamModels: - def __init__(self, outcome: Any): - self.outcome = outcome - self.requests: list[dict[str, Any]] = [] - - async def generate_content_stream(self, **kwargs: Any) -> Any: - self.requests.append(kwargs) - if isinstance(self.outcome, Exception): - raise self.outcome - chunks = list(self.outcome) - - async def gen(): - for chunk in chunks: - yield chunk - - return gen() - - -def _install_gemini_stream(monkeypatch: pytest.MonkeyPatch, outcome: Any) -> _FakeGeminiStreamModels: - models = _FakeGeminiStreamModels(outcome) - fake = SimpleNamespace(aio=SimpleNamespace(models=models)) - _patch_demo(monkeypatch, "_gemini_client", fake) - return models - - -async def test_gemini_token_batches_relay_in_order(monkeypatch: pytest.MonkeyPatch) -> None: - out = _install_writer(monkeypatch) - final_usage = SimpleNamespace(prompt_token_count=9, candidates_token_count=17) - models = _install_gemini_stream(monkeypatch, [ - _gemini_chunk("Event-"), - _gemini_chunk("driven "), - _gemini_chunk("haiku", usage=final_usage, finish="STOP"), - ]) - await demo_app.llm_stream( - {"my_correlation_id": "biz-777"}, - _event({"prompt": "haiku please", "params": {"provider": "gemini", "timeout_ms": 5000}}), - ) - assert out.error is None - assert out.head == (200, "text/event-stream") - assert out.segments == ["Event-", "driven ", "haiku"] - assert out.trailing["usage"] == {"input_tokens": 9, "output_tokens": 17} - assert out.trailing["stop_reason"] == "STOP" - assert out.trailing["model"] == "gemini-3.6-flash" - assert out.trailing["my_correlation_id"] == "biz-777" - # HttpOptions.timeout is milliseconds - timeout_ms passes through unchanged - config = models.requests[0]["config"] - assert config.http_options is not None - assert config.http_options.timeout == 5000 - - -async def test_gemini_provider_error_fails_the_stream(monkeypatch: pytest.MonkeyPatch) -> None: - from google.genai import errors as genai_errors - - out = _install_writer(monkeypatch) - _install_gemini_stream(monkeypatch, genai_errors.APIError(429, {"error": {"message": "quota"}})) - await demo_app.llm_stream({}, _event({"prompt": "x", "params": {"provider": "gemini"}})) - assert isinstance(out.error, AppException) - assert out.error.status == 429 - assert out.trailing is None - - -async def test_missing_prompt_fails_before_any_provider_call( - monkeypatch: pytest.MonkeyPatch, -) -> None: - out = _install_writer(monkeypatch) - await demo_app.llm_stream({}, _event({"params": {"provider": "gemini"}})) - assert isinstance(out.error, AppException) - assert out.error.status == 400 - assert out.head is None - assert out.segments == [] - - -# --- anthropic streaming path ------------------------------------------------ - - -class _FakeAnthropicStream: - def __init__(self, texts: list[str], final: Any): - self._texts = texts - self._final = final - - async def __aenter__(self): - return self - - async def __aexit__(self, *_exc: object) -> bool: - return False - - @property - def text_stream(self) -> Any: - async def gen(): - for text in self._texts: - yield text - - return gen() - - async def get_final_message(self) -> Any: - return self._final - - -class _FakeAnthropicStreamClient: - def __init__(self, outcome: Any, final: Any): - self.outcome = outcome - self.final = final - self.requests: list[dict[str, Any]] = [] - self.messages = self - self.timeout: float | None = None - - def with_options(self, timeout: float) -> "_FakeAnthropicStreamClient": - self.timeout = timeout - return self - - def stream(self, **kwargs: Any) -> Any: - self.requests.append(kwargs) - if isinstance(self.outcome, Exception): - raise self.outcome - return _FakeAnthropicStream(self.outcome, self.final) - - -async def test_anthropic_token_batches_relay_in_order(monkeypatch: pytest.MonkeyPatch) -> None: - out = _install_writer(monkeypatch) - final = SimpleNamespace( - model="claude-opus-5", - stop_reason="end_turn", - usage=SimpleNamespace(input_tokens=12, output_tokens=34), - ) - client = _FakeAnthropicStreamClient(["Hello ", "world"], final) - _patch_demo(monkeypatch, "_llm_client", client) - await demo_app.llm_stream({}, _event({"prompt": "greet me"})) - assert out.error is None - assert out.head == (200, "text/event-stream") - assert out.segments == ["Hello ", "world"] - assert out.trailing["usage"] == {"input_tokens": 12, "output_tokens": 34} - assert out.trailing["model"] == "claude-opus-5" - assert client.requests[0]["model"] == demo_app.LLM_DEFAULT_MODELS["anthropic"] - - -async def test_anthropic_rate_limit_fails_the_stream(monkeypatch: pytest.MonkeyPatch) -> None: - class _RateLimited(anthropic.RateLimitError): - def __init__(self) -> None: - Exception.__init__(self, "rate limited") - self.status_code = 429 - - out = _install_writer(monkeypatch) - client = _FakeAnthropicStreamClient(_RateLimited(), None) - _patch_demo(monkeypatch, "_llm_client", client) - await demo_app.llm_stream({}, _event({"prompt": "x"})) - assert isinstance(out.error, AppException) - assert out.error.status == 429 diff --git a/tests/vectors/llm-helper-vectors.json b/tests/vectors/llm-helper-vectors.json new file mode 100644 index 0000000..e409679 --- /dev/null +++ b/tests/vectors/llm-helper-vectors.json @@ -0,0 +1,2186 @@ +{ + "contract": "llm-helper", + "version": 1, + "description": "One contract, two function hosts. mercury-python and mercury-nodejs ship this file byte-identical and run every case against their own fake of the Anthropic SDK, so the two helpers cannot drift apart. A case names the route, the config keys in force, the request body and what the fake provider does; the expectation is the exact SDK call (namespace, parameters, options), then either the reply map or the error (status and message fragments) for llm.chat, or the head, the segments in order, the terminal metadata (a subset) or the in-band error for llm.stream.", + "provider_kinds": { + "message": "succeeds: deltas are the token batches; text is the final message text (default: the deltas joined); usage, model, stop_reason, request_id (the request-id response header), stop_details", + "status": "an HTTP error from the API: status, type, message, request_id; after = the number of deltas delivered before it is raised (0: at open), deltas = those batches", + "connection": "a connection failure before any answer", + "timeout": "the SDK's own timeout error", + "credential": "the SDK found no credential (it raises a bare TypeError before sending anything)", + "hang": "never answers - the helper's own deadline must fire" + }, + "cases": [ + { + "name": "prep/missing-prompt-and-messages", + "route": "llm.chat", + "body": {}, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "missing 'prompt' or 'messages'" + ] + } + } + }, + { + "name": "prep/body-not-a-map", + "route": "llm.chat", + "body": "just some text", + "expect": { + "error": { + "status": 400, + "message_contains": [ + "missing 'prompt' or 'messages'" + ] + } + } + }, + { + "name": "prep/messages-not-a-list", + "route": "llm.chat", + "body": { + "messages": "hello" + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "missing 'prompt' or 'messages'" + ] + } + } + }, + { + "name": "prep/empty-messages-and-no-prompt", + "route": "llm.chat", + "body": { + "messages": [] + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "missing 'prompt' or 'messages'" + ] + } + } + }, + { + "name": "prep/params-not-a-map", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": "fast" + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "params must be a map" + ] + } + } + }, + { + "name": "prep/unsupported-param", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "temperature": 0.2 + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "unsupported params: temperature", + "supported: provider, model, max_tokens, timeout_ms, effort, stop_sequences" + ] + } + } + }, + { + "name": "prep/unknown-provider-param", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "provider": "gemini" + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "unknown LLM provider 'gemini'", + "this helper serves: anthropic" + ] + } + } + }, + { + "name": "prep/unknown-provider-config", + "route": "llm.chat", + "config": { + "llm.provider": "gemini" + }, + "body": { + "prompt": "x" + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "unknown LLM provider 'gemini'", + "this helper serves: anthropic" + ] + } + } + }, + { + "name": "prep/max-tokens-zero", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "max_tokens": 0 + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "params.max_tokens must be a whole number >= 1" + ] + } + } + }, + { + "name": "prep/max-tokens-text", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "max_tokens": "lots" + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "params.max_tokens must be a whole number >= 1" + ] + } + } + }, + { + "name": "prep/max-tokens-boolean", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "max_tokens": true + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "params.max_tokens must be a whole number >= 1" + ] + } + } + }, + { + "name": "prep/max-tokens-fraction", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "max_tokens": 2.5 + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "params.max_tokens must be a whole number >= 1" + ] + } + } + }, + { + "name": "prep/timeout-negative", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "timeout_ms": -5 + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "params.timeout_ms must be a whole number >= 1" + ] + } + } + }, + { + "name": "prep/bad-effort", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "effort": "extreme" + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "params.effort must be one of: low, medium, high, xhigh, max" + ] + } + } + }, + { + "name": "prep/bad-role", + "route": "llm.chat", + "body": { + "messages": [ + { + "role": "user", + "content": "hi" + }, + { + "role": "system", + "content": "be terse" + } + ] + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "messages[1].role must be one of: user, assistant" + ] + } + } + }, + { + "name": "prep/empty-content", + "route": "llm.chat", + "body": { + "messages": [ + { + "role": "user", + "content": "" + } + ] + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "messages[0].content must be a non-empty string" + ] + } + } + }, + { + "name": "prep/content-blocks-not-supported", + "route": "llm.chat", + "body": { + "messages": [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": "x" + } + ] + } + ] + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "messages[0].content must be a non-empty string" + ] + } + } + }, + { + "name": "prep/system-not-a-string", + "route": "llm.chat", + "body": { + "prompt": "x", + "system": 5 + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "system must be a string" + ] + } + } + }, + { + "name": "prep/schema-not-a-map", + "route": "llm.chat", + "body": { + "prompt": "x", + "schema": "object" + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "schema must be a JSON schema map" + ] + } + } + }, + { + "name": "prep/schema-on-the-streaming-route", + "route": "llm.stream", + "body": { + "prompt": "x", + "schema": { + "type": "object", + "properties": { + "label": { + "type": "string", + "enum": [ + "question", + "bug", + "feature" + ] + }, + "reason": { + "type": "string" + } + }, + "required": [ + "label", + "reason" + ] + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "schema is not part of the streaming contract - use llm.chat" + ] + } + } + }, + { + "name": "prep/stop-sequences-not-strings", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "stop_sequences": [ + 1 + ] + } + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "params.stop_sequences must be a list of strings" + ] + } + } + }, + { + "name": "prep/fallbacks-config-invalid", + "route": "llm.chat", + "config": { + "llm.fallbacks": "maybe" + }, + "body": { + "prompt": "x" + }, + "expect": { + "error": { + "status": 500, + "message_contains": [ + "llm.fallbacks must be 'default' or 'off'" + ] + } + } + }, + { + "name": "chat/prompt-with-defaults", + "route": "llm.chat", + "body": { + "prompt": "Capital of France?" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "Par", + "is" + ], + "usage": { + "input_tokens": 12, + "output_tokens": 3 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "Capital of France?" + } + ], + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "text": "Paris", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 12, + "output_tokens": 3 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/conversation-with-system", + "route": "llm.chat", + "body": { + "messages": [ + { + "role": "user", + "content": "My name is Alice." + }, + { + "role": "assistant", + "content": "Hello Alice!" + }, + { + "role": "user", + "content": "What is my name?" + } + ], + "system": "You are terse." + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "Alice." + ], + "usage": { + "input_tokens": 30, + "output_tokens": 4 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 16000, + "system": "You are terse.", + "messages": [ + { + "role": "user", + "content": "My name is Alice." + }, + { + "role": "assistant", + "content": "Hello Alice!" + }, + { + "role": "user", + "content": "What is my name?" + } + ], + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "text": "Alice.", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 30, + "output_tokens": 4 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/messages-win-over-prompt", + "route": "llm.chat", + "body": { + "prompt": "ignored", + "messages": [ + { + "role": "user", + "content": "kept" + } + ] + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "ok" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "kept" + } + ], + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "text": "ok", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/schema-verdict", + "route": "llm.chat", + "body": { + "prompt": "The app crashes when I upload a file", + "system": "You are a support triage assistant.", + "schema": { + "type": "object", + "properties": { + "label": { + "type": "string", + "enum": [ + "question", + "bug", + "feature" + ] + }, + "reason": { + "type": "string" + } + }, + "required": [ + "label", + "reason" + ] + }, + "params": { + "max_tokens": 512, + "timeout_ms": 25000 + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "{\"label\":\"bug\",\"reas", + "on\":\"The upload crashes the app\"}" + ], + "usage": { + "input_tokens": 302, + "output_tokens": 24 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 512, + "system": "You are a support triage assistant.", + "messages": [ + { + "role": "user", + "content": "The app crashes when I upload a file" + } + ], + "output_config": { + "format": { + "type": "json_schema", + "schema": { + "additionalProperties": false, + "type": "object", + "properties": { + "label": { + "type": "string", + "enum": [ + "question", + "bug", + "feature" + ] + }, + "reason": { + "type": "string" + } + }, + "required": [ + "label", + "reason" + ] + } + } + }, + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 25000, + "max_retries": 2 + } + }, + "result": { + "data": { + "label": "bug", + "reason": "The upload crashes the app" + }, + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 302, + "output_tokens": 24 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/schema-keeps-the-callers-additional-properties", + "route": "llm.chat", + "body": { + "prompt": "x", + "schema": { + "type": "object", + "additionalProperties": true, + "properties": { + "a": { + "type": "string" + } + } + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "{\"a\":\"b\"}" + ], + "usage": { + "input_tokens": 8, + "output_tokens": 6 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "x" + } + ], + "output_config": { + "format": { + "type": "json_schema", + "schema": { + "type": "object", + "additionalProperties": true, + "properties": { + "a": { + "type": "string" + } + } + } + } + }, + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "data": { + "a": "b" + }, + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 8, + "output_tokens": 6 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/effort-and-stop-sequences", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "effort": "high", + "stop_sequences": [ + "END" + ] + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "stop_sequence", + "deltas": [ + "done" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 2 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "x" + } + ], + "output_config": { + "effort": "high" + }, + "stop_sequences": [ + "END" + ], + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "text": "done", + "model": "claude-opus-5-5", + "stop_reason": "stop_sequence", + "usage": { + "input_tokens": 5, + "output_tokens": 2 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/numeric-strings-are-accepted", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "max_tokens": "512", + "timeout_ms": "2500" + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "ok" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 512, + "messages": [ + { + "role": "user", + "content": "x" + } + ], + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 2500, + "max_retries": 2 + } + }, + "result": { + "text": "ok", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/other-model-gets-no-fallbacks", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "model": "claude-haiku-4-5" + } + }, + "provider": { + "kind": "message", + "model": "claude-haiku-4-5-20251001", + "stop_reason": "end_turn", + "deltas": [ + "OK" + ], + "usage": { + "input_tokens": 14, + "output_tokens": 4 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "messages", + "params": { + "model": "claude-haiku-4-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "x" + } + ] + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "text": "OK", + "model": "claude-haiku-4-5-20251001", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 14, + "output_tokens": 4 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/fallbacks-off-by-config", + "route": "llm.chat", + "config": { + "llm.fallbacks": "off" + }, + "body": { + "prompt": "x" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "ok" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "x" + } + ] + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "text": "ok", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/config-supplies-defaults-and-params-win", + "route": "llm.chat", + "config": { + "llm.model": "claude-haiku-4-5", + "llm.max.tokens": "2048", + "llm.timeout.ms": "9000", + "llm.max.retries": "0", + "llm.effort": "low" + }, + "body": { + "prompt": "x", + "params": { + "model": "claude-opus-5-5" + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "ok" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 2048, + "messages": [ + { + "role": "user", + "content": "x" + } + ], + "output_config": { + "effort": "low" + }, + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 9000, + "max_retries": 0 + } + }, + "result": { + "text": "ok", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/no-request-id-leaves-the-key-out", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "ok" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 1 + } + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "x" + } + ], + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "text": "ok", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 5, + "output_tokens": 1 + } + } + } + }, + { + "name": "chat/refusal-after-partial-text-is-reported", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "refusal", + "deltas": [ + "I can" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 2 + }, + "request_id": "req_ok_001", + "stop_details": { + "category": "cyber", + "explanation": "declined" + } + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "x" + } + ], + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "text": "I can", + "model": "claude-opus-5-5", + "stop_reason": "refusal", + "usage": { + "input_tokens": 5, + "output_tokens": 2 + }, + "request_id": "req_ok_001", + "stop_details": { + "category": "cyber", + "explanation": "declined" + } + } + } + }, + { + "name": "chat/truncated-text-is-returned-with-its-stop-reason", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "max_tokens": 1 + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "max_tokens", + "deltas": [ + "O" + ], + "usage": { + "input_tokens": 20, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 1, + "messages": [ + { + "role": "user", + "content": "x" + } + ], + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "result": { + "text": "O", + "model": "claude-opus-5-5", + "stop_reason": "max_tokens", + "usage": { + "input_tokens": 20, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + } + } + }, + { + "name": "chat/unusable/refusal-with-no-text", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "refusal", + "deltas": [], + "usage": { + "input_tokens": 5, + "output_tokens": 0 + }, + "text": "", + "request_id": "req_ok_001", + "stop_details": { + "category": "cyber", + "explanation": "declined" + } + }, + "expect": { + "error": { + "status": 422, + "message_contains": [ + "LLM refused the request - stop_reason=refusal", + "category=cyber" + ] + } + } + }, + { + "name": "chat/unusable/refusal-of-a-schema-request", + "route": "llm.chat", + "body": { + "prompt": "x", + "schema": { + "type": "object", + "properties": { + "label": { + "type": "string", + "enum": [ + "question", + "bug", + "feature" + ] + }, + "reason": { + "type": "string" + } + }, + "required": [ + "label", + "reason" + ] + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "refusal", + "deltas": [ + "{" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001", + "stop_details": { + "category": null, + "explanation": null + } + }, + "expect": { + "error": { + "status": 422, + "message_contains": [ + "LLM refused the request - stop_reason=refusal" + ] + } + } + }, + { + "name": "chat/unusable/empty-reply", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [], + "usage": { + "input_tokens": 5, + "output_tokens": 0 + }, + "text": "", + "request_id": "req_ok_001" + }, + "expect": { + "error": { + "status": 422, + "message_contains": [ + "LLM reply is empty - stop_reason=end_turn", + "output_tokens=0" + ] + } + } + }, + { + "name": "chat/unusable/empty-reply-cut-off-by-max-tokens", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "max_tokens": 8 + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "max_tokens", + "deltas": [], + "usage": { + "input_tokens": 5, + "output_tokens": 8 + }, + "text": "", + "request_id": "req_ok_001" + }, + "expect": { + "error": { + "status": 422, + "message_contains": [ + "LLM reply is empty - stop_reason=max_tokens", + "output_tokens=8", + "raise params.max_tokens or lower params.effort" + ] + } + } + }, + { + "name": "chat/unusable/schema-reply-cut-off", + "route": "llm.chat", + "body": { + "prompt": "x", + "schema": { + "type": "object", + "properties": { + "label": { + "type": "string", + "enum": [ + "question", + "bug", + "feature" + ] + }, + "reason": { + "type": "string" + } + }, + "required": [ + "label", + "reason" + ] + }, + "params": { + "max_tokens": 1 + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "max_tokens", + "deltas": [ + "{" + ], + "usage": { + "input_tokens": 300, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + }, + "expect": { + "error": { + "status": 422, + "message_contains": [ + "LLM reply is not valid JSON for the requested schema - stop_reason=max_tokens", + "the reply was cut off - raise params.max_tokens" + ] + } + } + }, + { + "name": "chat/unusable/schema-reply-not-json", + "route": "llm.chat", + "body": { + "prompt": "x", + "schema": { + "type": "object", + "properties": { + "label": { + "type": "string", + "enum": [ + "question", + "bug", + "feature" + ] + }, + "reason": { + "type": "string" + } + }, + "required": [ + "label", + "reason" + ] + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "Sorry, I cannot do that." + ], + "usage": { + "input_tokens": 300, + "output_tokens": 6 + }, + "request_id": "req_ok_001" + }, + "expect": { + "error": { + "status": 422, + "message_contains": [ + "LLM reply is not valid JSON for the requested schema - stop_reason=end_turn" + ] + } + } + }, + { + "name": "chat/provider-error/http-404", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "status", + "status": 404, + "type": "not_found_error", + "message": "model: claude-no-such-model-9", + "request_id": "req_err_404" + }, + "expect": { + "error": { + "status": 404, + "message_contains": [ + "LLM provider error - 404 not_found_error: model: claude-no-such-model-9", + "request_id req_err_404" + ] + } + } + }, + { + "name": "chat/provider-error/http-401", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "status", + "status": 401, + "type": "authentication_error", + "message": "API key is invalid." + }, + "expect": { + "error": { + "status": 401, + "message_contains": [ + "LLM provider error - 401 authentication_error: API key is invalid." + ] + } + } + }, + { + "name": "chat/provider-error/http-400", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "status", + "status": 400, + "type": "invalid_request_error", + "message": "This model does not support the effort parameter.", + "request_id": "req_err_400" + }, + "expect": { + "error": { + "status": 400, + "message_contains": [ + "LLM provider error - 400 invalid_request_error: This model does not support the effort parameter.", + "request_id req_err_400" + ] + } + } + }, + { + "name": "chat/provider-error/http-429", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "status", + "status": 429, + "type": "rate_limit_error", + "message": "Number of request tokens has exceeded your per-minute rate limit.", + "request_id": "req_err_429" + }, + "expect": { + "error": { + "status": 429, + "message_contains": [ + "LLM provider rate limit - 429 rate_limit_error: Number of request tokens has exceeded your per-minute rate limit.", + "request_id req_err_429" + ] + } + } + }, + { + "name": "chat/provider-error/http-500", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "status", + "status": 500, + "type": "api_error", + "message": "Internal server error", + "request_id": "req_err_500" + }, + "expect": { + "error": { + "status": 500, + "message_contains": [ + "LLM provider error - 500 api_error: Internal server error", + "request_id req_err_500" + ] + } + } + }, + { + "name": "chat/provider-error/http-529", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "status", + "status": 529, + "type": "overloaded_error", + "message": "Overloaded", + "request_id": "req_err_529" + }, + "expect": { + "error": { + "status": 529, + "message_contains": [ + "LLM provider error - 529 overloaded_error: Overloaded", + "request_id req_err_529" + ] + } + } + }, + { + "name": "chat/provider-error/connection-failure", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "connection", + "message": "Connection error." + }, + "expect": { + "error": { + "status": 503, + "message_contains": [ + "LLM provider unreachable - " + ] + } + } + }, + { + "name": "chat/provider-error/sdk-timeout", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "timeout_ms": 1500 + } + }, + "provider": { + "kind": "timeout" + }, + "expect": { + "error": { + "status": 408, + "message_contains": [ + "LLM request timed out after 1500 ms" + ] + } + } + }, + { + "name": "chat/provider-error/no-credential", + "route": "llm.chat", + "body": { + "prompt": "x" + }, + "provider": { + "kind": "credential" + }, + "expect": { + "error": { + "status": 503, + "message_contains": [ + "LLM provider credential missing - set ANTHROPIC_API_KEY in the environment" + ] + } + } + }, + { + "name": "chat/provider-error/deadline-bounds-the-whole-call", + "route": "llm.chat", + "body": { + "prompt": "x", + "params": { + "timeout_ms": 50 + } + }, + "provider": { + "kind": "hang" + }, + "expect": { + "error": { + "status": 408, + "message_contains": [ + "LLM request timed out after 50 ms" + ] + } + } + }, + { + "name": "stream/batches-forwarded-in-order", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "greet me" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "Hello ", + "wor", + "ld" + ], + "usage": { + "input_tokens": 12, + "output_tokens": 34 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "beta.messages", + "params": { + "model": "claude-opus-5-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "greet me" + } + ], + "betas": [ + "server-side-fallback-2026-07-01" + ], + "fallbacks": "default" + }, + "options": { + "timeout_ms": 60000, + "max_retries": 2 + } + }, + "head": [ + 200, + "text/event-stream" + ], + "frames": [ + "Hello ", + "wor", + "ld" + ], + "terminal": { + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "usage": { + "input_tokens": 12, + "output_tokens": 34 + }, + "request_id": "req_ok_001", + "my_correlation_id": "biz-777", + "trace_id": null + } + } + }, + { + "name": "stream/empty-batches-are-skipped", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "a", + "", + "b", + "" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 2 + }, + "text": "ab", + "request_id": "req_ok_001" + }, + "expect": { + "head": [ + 200, + "text/event-stream" + ], + "frames": [ + "a", + "b" + ], + "terminal": { + "stop_reason": "end_turn" + } + } + }, + { + "name": "stream/idle-allowance-is-timeout-ms", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x", + "params": { + "timeout_ms": 7000, + "model": "claude-haiku-4-5" + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "end_turn", + "deltas": [ + "ok" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 1 + }, + "request_id": "req_ok_001" + }, + "expect": { + "sdk": { + "namespace": "messages", + "params": { + "model": "claude-haiku-4-5", + "max_tokens": 16000, + "messages": [ + { + "role": "user", + "content": "x" + } + ] + }, + "options": { + "timeout_ms": 7000, + "max_retries": 2 + } + }, + "head": [ + 200, + "text/event-stream" + ], + "frames": [ + "ok" + ], + "terminal": { + "stop_reason": "end_turn" + } + } + }, + { + "name": "stream/error-before-the-first-token", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x" + }, + "provider": { + "kind": "status", + "status": 404, + "type": "not_found_error", + "message": "model: nope", + "request_id": "req_err_404" + }, + "expect": { + "head": null, + "frames": [], + "terminal": null, + "error": { + "status": 404, + "message_contains": [ + "LLM provider error - 404 not_found_error: model: nope", + "request_id req_err_404" + ] + } + } + }, + { + "name": "stream/error-after-some-tokens", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x" + }, + "provider": { + "kind": "status", + "status": 529, + "type": "overloaded_error", + "message": "Overloaded", + "request_id": "req_err_529", + "after": 1, + "deltas": [ + "Once ", + "upon" + ] + }, + "expect": { + "head": [ + 200, + "text/event-stream" + ], + "frames": [ + "Once " + ], + "terminal": null, + "error": { + "status": 529, + "message_contains": [ + "LLM provider error - 529 overloaded_error: Overloaded" + ] + } + } + }, + { + "name": "stream/rate-limit", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x" + }, + "provider": { + "kind": "status", + "status": 429, + "type": "rate_limit_error", + "message": "slow down" + }, + "expect": { + "head": null, + "frames": [], + "terminal": null, + "error": { + "status": 429, + "message_contains": [ + "LLM provider rate limit - 429 rate_limit_error" + ] + } + } + }, + { + "name": "stream/no-token-at-all-is-a-422", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x", + "params": { + "max_tokens": 8 + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "max_tokens", + "deltas": [], + "usage": { + "input_tokens": 5, + "output_tokens": 8 + }, + "text": "", + "request_id": "req_ok_001" + }, + "expect": { + "head": null, + "frames": [], + "terminal": null, + "error": { + "status": 422, + "message_contains": [ + "LLM reply is empty - stop_reason=max_tokens", + "output_tokens=8" + ] + } + } + }, + { + "name": "stream/refusal-before-any-token-is-a-422", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "refusal", + "deltas": [], + "usage": { + "input_tokens": 5, + "output_tokens": 0 + }, + "text": "", + "request_id": "req_ok_001", + "stop_details": { + "category": "cyber", + "explanation": "declined" + } + }, + "expect": { + "head": null, + "frames": [], + "terminal": null, + "error": { + "status": 422, + "message_contains": [ + "LLM refused the request - stop_reason=refusal", + "category=cyber" + ] + } + } + }, + { + "name": "stream/refusal-after-tokens-closes-with-details", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x" + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "refusal", + "deltas": [ + "I can" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 2 + }, + "request_id": "req_ok_001", + "stop_details": { + "category": "cyber", + "explanation": "declined" + } + }, + "expect": { + "head": [ + 200, + "text/event-stream" + ], + "frames": [ + "I can" + ], + "terminal": { + "stop_reason": "refusal", + "stop_details": { + "category": "cyber", + "explanation": "declined" + } + } + } + }, + { + "name": "stream/truncated-stream-closes-with-its-stop-reason", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x", + "params": { + "max_tokens": 2 + } + }, + "provider": { + "kind": "message", + "model": "claude-opus-5-5", + "stop_reason": "max_tokens", + "deltas": [ + "Hel" + ], + "usage": { + "input_tokens": 5, + "output_tokens": 2 + }, + "request_id": "req_ok_001" + }, + "expect": { + "head": [ + 200, + "text/event-stream" + ], + "frames": [ + "Hel" + ], + "terminal": { + "stop_reason": "max_tokens" + } + } + }, + { + "name": "stream/sdk-timeout", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x", + "params": { + "timeout_ms": 900 + } + }, + "provider": { + "kind": "timeout" + }, + "expect": { + "head": null, + "frames": [], + "terminal": null, + "error": { + "status": 408, + "message_contains": [ + "LLM request timed out after 900 ms" + ] + } + } + }, + { + "name": "stream/no-credential", + "route": "llm.stream", + "headers": { + "my_correlation_id": "biz-777" + }, + "body": { + "prompt": "x" + }, + "provider": { + "kind": "credential" + }, + "expect": { + "head": null, + "frames": [], + "terminal": null, + "error": { + "status": 503, + "message_contains": [ + "LLM provider credential missing - set ANTHROPIC_API_KEY in the environment" + ] + } + } + } + ] +}