diff --git a/packages/ai/README.md b/packages/ai/README.md
index b1b2230a..5147a215 100644
--- a/packages/ai/README.md
+++ b/packages/ai/README.md
@@ -54,14 +54,13 @@ Never raises. Returns `{"enabled": bool, "config": dict | None, "meta": dict | N
## Evaluations from code
-`init_evaluations`, the criterion types, and the evaluations result types are all re-exported:
+`init_evaluations`, the criterion types, `EvalTool`, and the evaluations result types are all re-exported:
```python
from launchdarkly_ai_python import Judge, Scorer, init_evaluations
-evals = init_evaluations()
+evals = init_evaluations(project_key="my-project")
result = await evals.run(
- project_key="my-project",
key="unique-evaluation-key",
dataset="golden-dataset",
handler=my_handler,
@@ -73,7 +72,7 @@ result = await evals.run(
)
```
-`LD_API_TOKEN` is required. Configure `LD_SDK_KEY` — or initialize your own client with `init_client(client=...)` — to emit one `$ld:ai:offline-evals:generation` event per generated row, plus one `$ld:ai:offline-evals:criterion` event per `(row, criterion)` when `criteria` are supplied, through the standard SDK event transport. The SDK reports scores; LaunchDarkly rules on them at ingest. A judge served by a different provider than `generation` needs a handler for it in `judge_handlers`. Each row's tool calls are recorded during generation and rendered into the judge's `{{message_history}}`, between the row input and the generated output, so a rubric can grade the tool trajectory as well as the final answer. Use `LD_API_BASE_URI` for staging or local management API traffic; it is separate from the SDK delivery setting `LD_BASE_URI`. Evaluation-run links use the explicit `ui_base_uri` option or `LD_UI_BASE_URI`, defaulting to `https://app.launchdarkly.com`; set it when the project is not in production, or a run created elsewhere still links to the production app. See the [core evaluations guide](https://github.com/launchdarkly/python-ai-sdk/blob/main/packages/client/README.md#run-an-evaluation-from-code).
+`LD_API_TOKEN` is required. Configure `LD_SDK_KEY` — or initialize your own client with `init_client(client=...)` — to emit one `$ld:ai:offline-evals:generation` event per generated row, plus one `$ld:ai:offline-evals:criterion` event per `(row, criterion)` when `criteria` are supplied, through the standard SDK event transport. The SDK reports scores; LaunchDarkly rules on them at ingest. A judge served by a different provider than `generation` needs a handler for it in `judge_handlers`. Each row's tool calls are recorded during generation and rendered into the judge's `{{message_history}}`, between the row input and the generated output, so a rubric can grade the tool trajectory as well as the final answer. Use `LD_API_BASE_URI` for staging or local management API traffic; it is separate from the SDK delivery setting `LD_BASE_URI`. Evaluation-run links use the explicit `ui_base_uri` option or `LD_UI_BASE_URI`, defaulting to `https://app.launchdarkly.com`; set it when the project is not in production, or a run created elsewhere still links to the production app. `tools` is a list of `EvalTool`. Construct one to define a tool in code, or await `evals.tools.get()` for a tool that already exists in LaunchDarkly. See the [core evaluations guide](https://github.com/launchdarkly/python-ai-sdk/blob/main/packages/client/README.md#run-an-evaluation-from-code).
---
diff --git a/packages/client/README.md b/packages/client/README.md
index b752af4f..5b0aae02 100644
--- a/packages/client/README.md
+++ b/packages/client/README.md
@@ -39,7 +39,7 @@ No code changes are required — `init_client()` detects the packages at runtime
| `LD_SERVICE_NAME` | No | OTel `service.name` resource attribute (default: `python-sdk`) |
| `LD_ENVIRONMENT` | No | `deployment.environment` resource attribute attached to telemetry |
| `OTEL_EXPORTER_OTLP_ENDPOINT` | No | OTLP endpoint override (default: LaunchDarkly Observability backend) |
-| `LD_API_TOKEN` | For evaluations | API access token used by the evaluations management API |
+| `LD_API_TOKEN` | For evaluations | API key used by the evaluations management API |
| `LD_SDK_KEY` | For evaluations | SDK key whose event transport carries generation results to LaunchDarkly |
| `LD_API_BASE_URI` | No | Evaluations management API host override; intentionally separate from `LD_BASE_URI` |
| `LD_UI_BASE_URI` | No | LaunchDarkly application host for evaluation-run links (default: `https://app.launchdarkly.com`). Set it for a non-production project, or its runs still link to the production app |
@@ -59,9 +59,8 @@ from launchdarkly_ai_server import init_evaluations
async def main() -> int:
- evals = init_evaluations() # LD_API_TOKEN required; LD_SDK_KEY unless a client is already initialized
+ evals = init_evaluations(project_key="my-project") # LD_API_TOKEN required; LD_SDK_KEY unless a client is already initialized
result = await evals.run(
- project_key="my-project",
key="support-qa-2026-08-20",
dataset="support-golden",
handler=create_openai_messages_handler(),
@@ -78,7 +77,7 @@ async def main() -> int:
sys.exit(asyncio.run(main()))
```
-`project_key` is supplied per run rather than during initialization. `generation.instructions` is shorthand for one system message; use `generation.messages` instead for a full message list, but do not supply both. The harness never retries a handler invocation because doing so could repeat tool side effects. Its retries apply only to LaunchDarkly management API requests.
+`project_key` is supplied during initialization, either as an argument to `init_evaluations()` or through `LD_PROJECT_KEY`. `generation.instructions` is shorthand for one system message; use `generation.messages` instead for a full message list, but do not supply both. The harness never retries a handler invocation because doing so could repeat tool side effects. Its retries apply only to LaunchDarkly management API requests.
Generation and criterion events are the only path by which row results reach LaunchDarkly, so `init_evaluations()` raises rather than creating a run that can never complete unless it can resolve an event transport: either an SDK key (`sdk_key` or `LD_SDK_KEY`) or a client already initialized through `init_client(client=...)`. Bringing your own client lets a process emit evaluation events without an SDK key in scope. Every generated row is emitted and flushed unconditionally; no feature flag gates event publishing. The harness then polls the summary endpoint until row accounting shows processing is complete.
@@ -96,8 +95,7 @@ def mentions_policy(row: DatasetRow, output: str | None) -> bool:
return "refund policy" in (output or "").lower()
-result = await init_evaluations().run(
- project_key="my-project",
+result = await init_evaluations(project_key="my-project").run(
key="support-qa-2026-08-20",
dataset="support-golden",
handler=create_openai_messages_handler(),
@@ -200,6 +198,48 @@ asyncio.run(main())
| `shutdown()` | Flush all events and telemetry, then close the client. Await before process exit. |
| `inspect_config(key, context)` | Read an AI Config variation without invoking the model. Never raises. Returns `{"enabled", "config", "meta"}`. |
+### Give the evaluation tools
+
+Pass `tools` to `run()` as a list of `EvalTool`. Construct one to define a tool in code. Await `evals.tools.get()` to use a tool that already exists in your project's AI library, which reads the tool and pins its version at that point. One list can hold both kinds.
+
+```python
+from launchdarkly_ai_server import EvalTool, init_evaluations
+
+
+def lookup_order(order_id: str) -> str:
+ return f"order {order_id} shipped"
+
+
+evals = init_evaluations(project_key="my-project")
+
+# Read from the AI library now. Pins the version. Raises now if the tool is absent.
+search_docs_tool = await evals.tools.get("search_docs", implementation=search_docs)
+
+# Defined here. Needs no tool in LaunchDarkly.
+lookup_order_tool = EvalTool(
+ key="lookup_order",
+ implementation=lookup_order,
+ schema={
+ "type": "object",
+ "properties": {"order_id": {"type": "string"}},
+ "required": ["order_id"],
+ },
+ description="Look up an order by id",
+)
+
+result = await evals.run(
+ key="support-qa-2026-08-20",
+ dataset="support-golden",
+ handler=create_openai_messages_handler(),
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ tools=[search_docs_tool, lookup_order_tool],
+)
+```
+
+`run()` reads no tool from the API. `tools.get()` is a coroutine, so await it. It reads in a worker thread and does not block the event loop. A constructed `EvalTool` is always inline, because `source` and `version` are not constructor arguments. Only `tools.get()` produces a library tool. Handlers receive the same `{key: callable}` map whichever kind a tool is, so handler code needs no change.
+
+**The list is checked before any network I/O.** A blank key, an uppercase key, a `schema` that is not a JSON object, a schema that is not JSON-serializable (including a `NaN` or `Infinity` value), and a non-callable implementation each fail with zero requests issued. A repeated key fails too, and keys are compared without case. A `NativeTool` is valid only for a library tool, because the provider supplies its schema.
+
### `config(**args)`
The primary entry point for AI config invocations. Accepts either a single handler or a list of handlers and routes to the correct one at invoke-time based on the flag variation's provider and mode.
diff --git a/packages/client/agents.md b/packages/client/agents.md
index cb117045..a134b278 100644
--- a/packages/client/agents.md
+++ b/packages/client/agents.md
@@ -131,7 +131,7 @@ Handlers may return any of these — the client normalizes them before emitting
`init_evaluations()` creates an evaluations harness using `LD_API_TOKEN` and the management API host `LD_API_BASE_URI`. Do not reuse `LD_BASE_URI`: that variable configures SDK delivery and may point at a relay proxy. Evaluation-run links use the separate `ui_base_uri` option, then `LD_UI_BASE_URI`, then `https://app.launchdarkly.com`; do not derive their host from `LD_API_BASE_URI`. An event transport is resolved in `init_evaluations()`, which raises before any network I/O when it finds neither an SDK key (`sdk_key` or `LD_SDK_KEY`) nor an already-initialized event-capable client: generation events are the only ingest path for row results, so a run without a transport could never complete. The lifecycle module's bring-your-own-client path (`init_client(client=...)`) therefore satisfies the check on its own, and `run()` reuses that singleton through `_resolve_client`; `run()` raises if the client disappears before it emits. Both polling arguments reject NaN, which would otherwise never compare past a deadline and hang the run. The harness always queues one `$ld:ai:offline-evals:generation` custom event per row through the standard SDK event transport and flushes before returning. No feature flag gates event emission. The harness polls the run summary endpoint until a nonzero `total_rows` has `pending_rows == 0` and `passed + failed + error` rows accounting for the total, polling every `poll_interval_seconds` (default 2s) until `poll_timeout_seconds` (default 180s); both are `run()` arguments so large datasets can widen them. The summary endpoint does not return run state, so `RunSummary` exposes row counts only.
-`await EvaluationsModule.run(...)` takes `project_key` per call. Dataset lookup/row pagination, evaluation creation, and run creation are private helpers; only `run()` is public. Each call creates a new evaluation with `POST` and a run with `source="api"`, so its key must be unique. The harness directly invokes the supplied handler once per row and never retries it — event delivery is never a reason to rerun a handler because that would repeat tool side effects; retries apply only to management API requests. A 429 is replayed for any method, but 5xx responses and transport failures are replayed only for `GET`/`HEAD`, so an evaluation or run `POST` that the server may already have applied is never duplicated. Management API calls run in a worker thread (`asyncio.to_thread`) because the client is synchronous; the caller's event loop stays free. Generation events go through the already-initialized SDK client when the application has one — `init_client` is idempotent, so an existing singleton wins and the evaluations SDK key is ignored with a warning. Dataset-owned `input`, `expected_output`, `metadata`, and `variables` are deliberately excluded from the event payload. The harness flushes events, polls the run summary endpoint until row accounting is complete (`total_rows > 0`, `pending_rows == 0`, and `passed + failed + error == total_rows`), and raises a timeout once `poll_timeout_seconds` elapses if the backend never reaches one. `RunSummary` includes row counts only, and `EvalRunResult.passed` is true only when error and pending row counts are both zero.
+`init_evaluations()` takes `project_key`; `run()` does not. Dataset lookup/row pagination, evaluation creation, and run creation are private helpers; `run()` and `tools.get()` are the public surface. `tools` is a `list[EvalTool]`. `await evals.tools.get(key, implementation=...)` issues `GET projects/
/ai-tools/`, pins the returned version, and builds a library tool through `EvalTool._library`. `get` is a coroutine that reads in a worker thread. `EvalTool` is frozen with `schema` required, so a constructed tool is always inline. `_validate_tools` also rejects a library tool that has no `project_key` or one from another project. `run()` issues no tool request at all: a library tool was already read by `tools.get()`. Library tools are sent as `{key, version, source: "library"}` and inline tools as `{key, schema, description, source: "inline"}`. `_validate_tools` runs in `run()` before any I/O, so a bad list — a non-`EvalTool` entry, a blank or uppercase key, a non-object or non-serializable `schema` (`allow_nan=False`), a non-callable implementation, a repeated key (compared without case), or a `NativeTool` on an inline tool — fails with zero requests recorded. `_validate_tool_key` is shared by `tools.get()` and `_validate_tools`, so the key rules are identical on both paths. Handler config synthesis is unforked: `config["tools"][tool.key] = {"description", "parameters"}` is fed from whichever kind the tool is, so handlers cannot tell them apart, and `_tool_handlers` maps each tool to its executable. Each call creates a new evaluation with `POST` and a run with `source="api"`, so its key must be unique. The harness directly invokes the supplied handler once per row and never retries it — event delivery is never a reason to rerun a handler because that would repeat tool side effects; retries apply only to management API requests. A 429 is replayed for any method, but 5xx responses and transport failures are replayed only for `GET`/`HEAD`, so an evaluation or run `POST` that the server may already have applied is never duplicated. Management API calls run in a worker thread (`asyncio.to_thread`) because the client is synchronous; the caller's event loop stays free. Generation events go through the already-initialized SDK client when the application has one — `init_client` is idempotent, so an existing singleton wins and the evaluations SDK key is ignored with a warning. Dataset-owned `input`, `expected_output`, `metadata`, and `variables` are deliberately excluded from the event payload. The harness flushes events, polls the run summary endpoint until row accounting is complete (`total_rows > 0`, `pending_rows == 0`, and `passed + failed + error == total_rows`), and raises a timeout once `poll_timeout_seconds` elapses if the backend never reaches one. `RunSummary` includes row counts only, and `EvalRunResult.passed` is true only when error and pending row counts are both zero.
---
diff --git a/packages/client/src/launchdarkly_ai_server/__init__.py b/packages/client/src/launchdarkly_ai_server/__init__.py
index 4001239e..6e77b0bf 100644
--- a/packages/client/src/launchdarkly_ai_server/__init__.py
+++ b/packages/client/src/launchdarkly_ai_server/__init__.py
@@ -30,6 +30,7 @@
Criterion,
DatasetRow,
EvalRunResult,
+ EvalTool,
EvaluationsError,
EvaluationsModule,
GenerationConfig,
@@ -194,6 +195,7 @@
"EvaluationsError",
"EvaluationsModule",
"GenerationConfig",
+ "EvalTool",
"Judge",
"RunSummary",
"Scorer",
diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py b/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py
index 9cd942d9..5ba1535f 100644
--- a/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py
+++ b/packages/client/src/launchdarkly_ai_server/evaluations/__init__.py
@@ -11,6 +11,7 @@
)
from .criteria import Criterion, Judge, Scorer, SuccessDirection
from .module import EvaluationsModule, init_evaluations
+from .tools import EvalTool, ToolsClient
from .types import (
AIConfig,
DatasetRow,
@@ -26,6 +27,7 @@
"Criterion",
"DatasetRow",
"EvalRunResult",
+ "EvalTool",
"EvaluationsError",
"EvaluationsModule",
"GenerationConfig",
@@ -36,6 +38,7 @@
"RunSummary",
"Scorer",
"SuccessDirection",
+ "ToolsClient",
"Transport",
"Usage",
"init_evaluations",
diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/api.py b/packages/client/src/launchdarkly_ai_server/evaluations/api.py
index 3a46d23d..96328b65 100644
--- a/packages/client/src/launchdarkly_ai_server/evaluations/api.py
+++ b/packages/client/src/launchdarkly_ai_server/evaluations/api.py
@@ -6,7 +6,7 @@
import urllib.error
import urllib.parse
import urllib.request
-from collections.abc import Callable
+from collections.abc import Callable, Mapping
from dataclasses import dataclass, field
from datetime import UTC, datetime
from email.utils import parsedate_to_datetime
@@ -84,7 +84,7 @@ class LDApiClient:
def __init__(
self,
- api_token: str,
+ api_key: str,
base_uri: str = DEFAULT_BASE_URI,
transport: Transport = urllib_transport,
timeout: float = 30.0,
@@ -92,7 +92,7 @@ def __init__(
sleep: Callable[[float], None] = time.sleep,
random_value: Callable[[], float] = random.random,
) -> None:
- self.api_token = api_token
+ self.api_key = api_key
self.base_uri = base_uri.rstrip("/")
self._transport = transport
self._timeout = timeout
@@ -135,7 +135,7 @@ def request(
params: dict[str, Any] | None = None,
) -> Any:
headers = {
- "Authorization": self.api_token,
+ "Authorization": self.api_key,
"Accept": "application/json",
"LD-API-Version": "20240415",
"User-Agent": "launchdarkly-ai-evaluations-python",
@@ -192,3 +192,27 @@ def get(self, path: str, params: dict[str, Any] | None = None) -> Any:
def post(self, path: str, body: Any = None) -> Any:
return self.request("POST", path, body=body)
+
+
+def segment(value: str) -> str:
+ """Percent-encode one path segment."""
+ return urllib.parse.quote(value, safe="")
+
+
+def require_mapping(value: Any, *, description: str) -> Mapping[str, Any]:
+ """Return ``value`` as a mapping. Raises ``EvaluationsError``."""
+ if not isinstance(value, Mapping):
+ raise EvaluationsError(
+ f"LaunchDarkly returned an invalid {description} response"
+ )
+ return value
+
+
+def require_string(data: Mapping[str, Any], key: str, description: str) -> str:
+ """Return the non-empty string at ``key``. Raises ``EvaluationsError``."""
+ value = data.get(key)
+ if not isinstance(value, str) or not value:
+ raise EvaluationsError(
+ f"LaunchDarkly {description} response is missing string field {key!r}"
+ )
+ return value
diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/module.py b/packages/client/src/launchdarkly_ai_server/evaluations/module.py
index 4cc684aa..60a3bcc3 100644
--- a/packages/client/src/launchdarkly_ai_server/evaluations/module.py
+++ b/packages/client/src/launchdarkly_ai_server/evaluations/module.py
@@ -6,7 +6,7 @@
import math
import os
import time
-from collections.abc import Mapping
+from collections.abc import Mapping, Sequence
from typing import Any, cast
from ..lifecycle import get_client, init_client
@@ -15,16 +15,12 @@
EvaluationsError,
LDApiClient,
Transport,
+ segment,
urllib_transport,
)
from .criteria import Criterion, Judge
-from .runner import (
- EvalHandler,
- EvaluationsRunner,
- ToolImplementation,
- _provides_for,
- _segment,
-)
+from .runner import EvalHandler, EvaluationsRunner, _provides_for
+from .tools import EvalTool, ToolsClient, tool_handlers, validate_tools
from .types import AIConfig, EvalRunResult, GenerationConfig, RunSummary
logger = logging.getLogger(__name__)
@@ -98,18 +94,31 @@ class EvaluationsModule:
def __init__(
self,
api_client: LDApiClient,
+ project_key: str,
sdk_key: str | None,
ui_base_uri: str = DEFAULT_UI_BASE_URI,
) -> None:
self._api = api_client
+ self._project_key = project_key
self._sdk_key = sdk_key
self._ui_base_uri = ui_base_uri.rstrip("/")
self._runner = EvaluationsRunner(api_client)
+ self._tools = ToolsClient(api_client, project_key)
@property
def api(self) -> LDApiClient:
return self._api
+ @property
+ def project_key(self) -> str:
+ """Project that holds this module's evaluations, tools, and datasets."""
+ return self._project_key
+
+ @property
+ def tools(self) -> ToolsClient:
+ """Reader for tools in the LaunchDarkly tool library."""
+ return self._tools
+
@property
def sdk_key(self) -> str | None:
"""SDK key whose event transport carries generation results to LaunchDarkly."""
@@ -123,13 +132,12 @@ def ui_base_uri(self) -> str:
async def run(
self,
*,
- project_key: str,
key: str,
dataset: str,
handler: EvalHandler,
generation: GenerationConfig | None = None,
ai_config: AIConfig | None = None,
- tools: Mapping[str, ToolImplementation] | None = None,
+ tools: Sequence[EvalTool] | None = None,
criteria: list[Criterion] | None = None,
judge_handlers: list[EvalHandler] | None = None,
concurrency: int = 10,
@@ -145,6 +153,13 @@ async def run(
generated row, and one evaluation event is emitted per
``(row, criterion)`` result.
+ ``tools`` is a list of :class:`EvalTool`. Construct one to define a tool
+ in code. Call ``evals.tools.get(key, implementation=...)`` to use a
+ tool from the LaunchDarkly tool library, which reads the tool and pins
+ its version at that point. One list may hold both kinds. ``run`` reads
+ no tool from the API, and it checks the list before any network I/O.
+ Handlers receive a ``{key: executable}`` map either way.
+
A :class:`Judge` is an independent AI Config and may be served by a
different provider or mode than ``generation``. ``handler`` runs a judge
only when it provides for that judge's provider; pass handlers for any
@@ -171,7 +186,6 @@ async def run(
if poll_timeout_seconds is None:
poll_timeout_seconds = SUMMARY_POLL_TIMEOUT_SECONDS
self._validate_run_args(
- project_key=project_key,
key=key,
dataset=dataset,
handler=handler,
@@ -180,13 +194,16 @@ async def run(
poll_timeout_seconds=poll_timeout_seconds,
)
self._validate_config_source(generation=generation, ai_config=ai_config)
+ run_tools = list(tools or [])
+ validate_tools(run_tools, self._project_key)
+ run_tool_handlers = tool_handlers(run_tools)
pinned_tool_versions: dict[str, int] = {}
config_label = ""
if ai_config is not None:
config_label = f"{ai_config.key!r}/{ai_config.variation!r}"
ai_config_variation = await asyncio.to_thread(
self._runner._fetch_config_variation,
- project_key,
+ self._project_key,
ai_config.key,
ai_config.variation,
)
@@ -198,15 +215,25 @@ async def run(
+ ", ".join(
repr(name) for name in ai_config_variation.tool_versions
)
- + ". Pass tools= with an implementation for each."
+ + ". Pass tools= with an EvalTool for each."
)
+ # A caller who passes tools= replaces the variation's list, so an
+ # empty list runs the variation with no tools.
+ supplied = {tool.key for tool in run_tools}
+ for name in ai_config_variation.tool_versions:
+ if name not in supplied:
+ logger.warning(
+ "AI Config variation %s attaches tool %r, which this run "
+ "does not use.",
+ config_label,
+ name,
+ )
pinned_tool_versions = ai_config_variation.tool_versions
if criteria is None:
criteria = [
Judge(key=judge_key) for judge_key in ai_config_variation.judge_keys
]
generation = self._validate_generation(generation)
- run_tools = dict(tools or {})
run_criteria = list(criteria or [])
run_judge_handlers = list(judge_handlers or [])
self._validate_criteria(run_criteria)
@@ -218,58 +245,57 @@ async def run(
# The management API client is synchronous; running it in a worker thread
# keeps the caller's event loop free.
- # Tool/judge verification is deliberately first: a typo must not create records.
- resolved_tools = await asyncio.to_thread(
- self._runner._resolve_tools, project_key, run_tools
- )
- # The tool API serves only the latest version, so a variation pinned to
- # an older one is evaluated against the current schema.
+ # Judge verification is first: a typo must not create records.
+ # A variation pins a tool version. Compare it with the version the run
+ # uses, which tools.get() already read.
+ tools_by_key = {tool.key: tool for tool in run_tools}
for tool_key, pinned_version in pinned_tool_versions.items():
- resolved_tool = resolved_tools.get(tool_key)
- if resolved_tool is not None and resolved_tool.version != pinned_version:
- logger.warning(
- "AI Config variation %s pins tool %r at version %d; "
- "evaluating against the latest version %d.",
- config_label,
- tool_key,
- pinned_version,
- resolved_tool.version,
- )
+ tool = tools_by_key.get(tool_key)
+ if tool is None or tool.version is None or tool.version == pinned_version:
+ continue
+ logger.warning(
+ "AI Config variation %s pins tool %r at version %d; "
+ "the run uses version %d.",
+ config_label,
+ tool_key,
+ pinned_version,
+ tool.version,
+ )
resolved_judges = await self._runner._resolve_judges(
- project_key, ld_judges, handler, run_judge_handlers
+ self._project_key, ld_judges, handler, run_judge_handlers
)
dataset_ref = await asyncio.to_thread(
- self._runner._fetch_dataset, project_key, dataset
+ self._runner._fetch_dataset, self._project_key, dataset
)
rows = await asyncio.to_thread(
- self._runner._get_dataset_rows, project_key, dataset
+ self._runner._get_dataset_rows, self._project_key, dataset
)
evaluation = await asyncio.to_thread(
self._runner._create_evaluation,
- project_key,
+ self._project_key,
key,
generation,
- resolved_tools,
+ run_tools,
run_criteria,
)
evaluation_run = await asyncio.to_thread(
self._runner._create_evaluation_run,
- project_key,
+ self._project_key,
evaluation.id,
dataset_ref.id,
)
- config = self._runner._build_handler_config(generation, resolved_tools)
+ config = self._runner._build_handler_config(generation, run_tools)
results = await self._runner._run_rows(
rows,
handler,
config,
- run_tools,
+ run_tool_handlers,
concurrency,
)
try:
self._runner._emit_generation_events(
client,
- project_key=project_key,
+ project_key=self._project_key,
evaluation=evaluation,
evaluation_run=evaluation_run,
dataset=dataset_ref,
@@ -278,14 +304,14 @@ async def run(
if run_criteria:
criterion_results = await self._runner._run_criteria_for_results(
results,
- run_tools,
+ run_tool_handlers,
run_criteria,
resolved_judges,
concurrency,
)
self._runner._emit_evaluation_events(
client,
- project_key=project_key,
+ project_key=self._project_key,
evaluation=evaluation,
evaluation_run=evaluation_run,
dataset=dataset_ref,
@@ -298,15 +324,14 @@ async def run(
if inspect.isawaitable(flush_result):
await flush_result
summary = await self._poll_summary_until_terminal(
- project_key,
evaluation.id,
evaluation_run.id,
poll_interval_seconds,
poll_timeout_seconds,
)
url = (
- f"{self._ui_base_uri}/projects/{_segment(project_key)}/ai/evaluations/"
- f"{_segment(evaluation.id)}/runs/{_segment(evaluation_run.id)}"
+ f"{self._ui_base_uri}/projects/{segment(self._project_key)}/ai/evaluations/"
+ f"{segment(evaluation.id)}/runs/{segment(evaluation_run.id)}"
)
return EvalRunResult(
# failed_rows counts rows whose criteria were scored and did not
@@ -326,7 +351,6 @@ async def run(
async def _poll_summary_until_terminal(
self,
- project_key: str,
evaluation_id: str,
run_id: str,
poll_interval_seconds: float,
@@ -336,7 +360,7 @@ async def _poll_summary_until_terminal(
last_summary = None
while True:
last_summary = await asyncio.to_thread(
- self._runner._get_summary, project_key, evaluation_id, run_id
+ self._runner._get_summary, self._project_key, evaluation_id, run_id
)
if _is_terminal_summary(last_summary):
return last_summary
@@ -429,7 +453,6 @@ def _validate_judge_handlers(judge_handlers: list[EvalHandler]) -> None:
@staticmethod
def _validate_run_args(
*,
- project_key: str,
key: str,
dataset: str,
handler: EvalHandler,
@@ -438,7 +461,6 @@ def _validate_run_args(
poll_timeout_seconds: float,
) -> None:
for name, value in (
- ("project_key", project_key),
("key", key),
("dataset", dataset),
):
@@ -501,18 +523,30 @@ def _validate_generation(generation: GenerationConfig | None) -> GenerationConfi
def init_evaluations(
- api_token: str | None = None,
+ project_key: str | None = None,
+ api_key: str | None = None,
sdk_key: str | None = None,
base_uri: str | None = None,
ui_base_uri: str | None = None,
transport: Transport = urllib_transport,
) -> EvaluationsModule:
- """Resolve credentials and construct the evaluations module."""
- token = api_token or _env("LD_API_TOKEN")
+ """Resolve credentials and construct the evaluations module.
+
+ ``project_key`` names the project that holds the evaluations, the tools,
+ and the datasets this module uses.
+ """
+ resolved_project_key = (project_key or "").strip() or _env("LD_PROJECT_KEY")
+ if not resolved_project_key:
+ raise EvaluationsError(
+ "No LaunchDarkly project key provided. Set the LD_PROJECT_KEY "
+ "environment variable or pass project_key to init_evaluations()."
+ )
+
+ token = api_key or _env("LD_API_TOKEN")
if not token:
raise EvaluationsError(
- "No LaunchDarkly API access token provided. Set the LD_API_TOKEN "
- "environment variable or pass api_token to init_evaluations()."
+ "No LaunchDarkly API key provided. Set the LD_API_TOKEN "
+ "environment variable or pass api_key to init_evaluations()."
)
resolved_sdk_key = sdk_key or _env("LD_SDK_KEY")
@@ -529,12 +563,13 @@ def init_evaluations(
)
api_client = LDApiClient(
- api_token=token,
+ api_key=token,
base_uri=base_uri or _env("LD_API_BASE_URI") or DEFAULT_BASE_URI,
transport=transport,
)
return EvaluationsModule(
api_client=api_client,
+ project_key=resolved_project_key,
sdk_key=resolved_sdk_key,
ui_base_uri=ui_base_uri or _env("LD_UI_BASE_URI") or DEFAULT_UI_BASE_URI,
)
diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py
index 6240e980..4776e13f 100644
--- a/packages/client/src/launchdarkly_ai_server/evaluations/runner.py
+++ b/packages/client/src/launchdarkly_ai_server/evaluations/runner.py
@@ -6,8 +6,7 @@
import json
import logging
import time
-import urllib.parse
-from collections.abc import Awaitable, Callable, Mapping
+from collections.abc import Awaitable, Callable, Mapping, Sequence
from dataclasses import dataclass
from datetime import UTC, datetime
from typing import Any, Literal
@@ -24,7 +23,6 @@
render_row_trajectory,
row_fields,
)
-from ..types import NativeTool
from ..utils import (
collapse_messages_to_instructions,
normalize_mode,
@@ -32,7 +30,14 @@
parse_usage,
to_ld_context,
)
-from .api import EvaluationsError, LDApiClient, LDApiError
+from .api import (
+ EvaluationsError,
+ LDApiClient,
+ LDApiError,
+ require_mapping,
+ require_string,
+ segment,
+)
from .criteria import Criterion, Judge, Scorer
from .events import (
CriterionEventPayload,
@@ -41,6 +46,12 @@
LDJudgeCriterionEventPayload,
TokenUsage,
)
+from .tools import (
+ EvalTool,
+ ToolImplementation,
+ create_wire_tools,
+ handler_config_tools,
+)
from .types import (
AIConfigVariation,
DatasetRef,
@@ -49,7 +60,6 @@
EvaluationRunRef,
GenerationConfig,
ResolvedJudge,
- ResolvedTool,
RunSummary,
)
@@ -60,7 +70,6 @@
CRITERION_EVENT_NAME = "$ld:ai:offline-evals:criterion"
EvalHandler = Callable[..., Awaitable[dict[str, Any]]]
-ToolImplementation = Callable[..., Any] | NativeTool
@dataclass(frozen=True)
@@ -164,27 +173,6 @@ def _select_judge_handler(
return None
-def _segment(value: str) -> str:
- return urllib.parse.quote(value, safe="")
-
-
-def _mapping(value: Any, *, description: str) -> Mapping[str, Any]:
- if not isinstance(value, Mapping):
- raise EvaluationsError(
- f"LaunchDarkly returned an invalid {description} response"
- )
- return value
-
-
-def _required_string(data: Mapping[str, Any], key: str, description: str) -> str:
- value = data.get(key)
- if not isinstance(value, str) or not value:
- raise EvaluationsError(
- f"LaunchDarkly {description} response is missing string field {key!r}"
- )
- return value
-
-
class ConcurrencyController:
"""Owns row-worker permits."""
@@ -237,11 +225,11 @@ def _fetch_config_variation(
"""
description = f"AI Config variation {config_key!r}/{variation_key!r}"
path = (
- f"projects/{_segment(project_key)}/ai-configs/{_segment(config_key)}"
- f"/variations/{_segment(variation_key)}"
+ f"projects/{segment(project_key)}/ai-configs/{segment(config_key)}"
+ f"/variations/{segment(variation_key)}"
)
try:
- raw = _mapping(self._api.get(path), description=description)
+ raw = require_mapping(self._api.get(path), description=description)
except LDApiError as error:
if error.status == 404:
raise EvaluationsError(
@@ -282,13 +270,13 @@ def _fetch_model_config(
version: Any,
) -> Mapping[str, Any]:
path = (
- f"projects/{_segment(project_key)}/ai-configs/model-configs/"
- f"{_segment(model_config_key)}"
+ f"projects/{segment(project_key)}/ai-configs/model-configs/"
+ f"{segment(model_config_key)}"
)
# A pinned variation names the model-config version it was built against.
params = {"version": version} if isinstance(version, int) else None
try:
- return _mapping(
+ return require_mapping(
self._api.get(path, params=params),
description=f"model config {model_config_key!r}",
)
@@ -300,44 +288,6 @@ def _fetch_model_config(
) from error
raise
- def _resolve_tools(
- self,
- project_key: str,
- tools: Mapping[str, ToolImplementation],
- ) -> dict[str, ResolvedTool]:
- resolved: dict[str, ResolvedTool] = {}
- for key, implementation in tools.items():
- if not callable(implementation) and not isinstance(
- implementation, NativeTool
- ):
- raise EvaluationsError(
- f"Tool {key!r} must be callable or a NativeTool instance"
- )
- path = f"projects/{_segment(project_key)}/ai-tools/{_segment(key)}"
- try:
- raw = _mapping(self._api.get(path), description=f"tool {key!r}")
- except LDApiError as error:
- if error.status == 404:
- raise EvaluationsError(
- f"LaunchDarkly AI tool {key!r} was not found in project {project_key!r}"
- ) from error
- raise
- version = raw.get("version")
- if not isinstance(version, int):
- raise EvaluationsError(
- f"LaunchDarkly AI tool {key!r} has no integer version"
- )
- schema = raw.get("schema")
- if not isinstance(schema, Mapping):
- schema = {}
- resolved[key] = ResolvedTool(
- key=key,
- version=version,
- description=str(raw.get("description") or ""),
- schema=dict(schema),
- )
- return resolved
-
async def _resolve_judges(
self,
project_key: str,
@@ -407,16 +357,18 @@ async def _resolve_judges(
return resolved
def _fetch_dataset(self, project_key: str, dataset_key: str) -> DatasetRef:
- path = f"projects/{_segment(project_key)}/datasets/{_segment(dataset_key)}"
+ path = f"projects/{segment(project_key)}/datasets/{segment(dataset_key)}"
try:
- raw = _mapping(self._api.get(path), description=f"dataset {dataset_key!r}")
+ raw = require_mapping(
+ self._api.get(path), description=f"dataset {dataset_key!r}"
+ )
except LDApiError as error:
if error.status == 404:
raise EvaluationsError(
f"LaunchDarkly dataset {dataset_key!r} was not found in project {project_key!r}"
) from error
raise
- dataset_id = _required_string(raw, "id", "dataset")
+ dataset_id = require_string(raw, "id", "dataset")
response_key = raw.get("key", raw.get("name", dataset_key))
return DatasetRef(id=dataset_id, key=str(response_key))
@@ -427,8 +379,8 @@ def _fetch_dataset_rows_page(
*,
offset: int,
) -> Mapping[str, Any]:
- path = f"projects/{_segment(project_key)}/datasets/{_segment(dataset_key)}/rows"
- return _mapping(
+ path = f"projects/{segment(project_key)}/datasets/{segment(dataset_key)}/rows"
+ return require_mapping(
self._api.get(
path,
params={
@@ -458,7 +410,7 @@ def _get_dataset_rows(self, project_key: str, dataset_key: str) -> list[DatasetR
if not items:
break
for item_value in items:
- item = _mapping(item_value, description="dataset row")
+ item = require_mapping(item_value, description="dataset row")
row_index = item.get("rowIndex")
if not isinstance(row_index, int):
raise EvaluationsError(
@@ -512,7 +464,7 @@ def _create_evaluation(
project_key: str,
key: str,
generation: GenerationConfig,
- tools: Mapping[str, ResolvedTool],
+ tools: Sequence[EvalTool],
criteria: list[Criterion] | None = None,
) -> EvaluationRef:
body: dict[str, Any] = {
@@ -533,15 +485,13 @@ def _create_evaluation(
if "prompt_snippets" in generation:
body["promptSnippets"] = generation["prompt_snippets"]
if tools:
- body["tools"] = [
- {"key": tool.key, "version": tool.version} for tool in tools.values()
- ]
+ body["tools"] = create_wire_tools(tools)
if criteria:
body["criteria"] = [criterion.to_criteria_wire() for criterion in criteria]
- path = f"projects/{_segment(project_key)}/evaluations"
- raw = _mapping(self._api.post(path, body=body), description="evaluation")
- evaluation_id = _required_string(raw, "id", "evaluation")
+ path = f"projects/{segment(project_key)}/evaluations"
+ raw = require_mapping(self._api.post(path, body=body), description="evaluation")
+ evaluation_id = require_string(raw, "id", "evaluation")
response_key = raw.get("name", raw.get("label", key))
version = raw.get("version")
return EvaluationRef(
@@ -557,14 +507,13 @@ def _create_evaluation_run(
dataset_id: str,
) -> EvaluationRunRef:
path = (
- f"projects/{_segment(project_key)}/evaluations/"
- f"{_segment(evaluation_id)}/runs"
+ f"projects/{segment(project_key)}/evaluations/{segment(evaluation_id)}/runs"
)
body: dict[str, Any] = {
"source": "api",
"datasetId": dataset_id,
}
- raw = _mapping(
+ raw = require_mapping(
self._api.post(path, body=body),
description="evaluation run",
)
@@ -572,9 +521,9 @@ def _create_evaluation_run(
def _run_ref(self, raw: Mapping[str, Any]) -> EvaluationRunRef:
return EvaluationRunRef(
- id=_required_string(raw, "id", "evaluation run"),
- evaluation_id=_required_string(raw, "evaluationId", "evaluation run"),
- state=_required_string(raw, "state", "evaluation run"),
+ id=require_string(raw, "id", "evaluation run"),
+ evaluation_id=require_string(raw, "evaluationId", "evaluation run"),
+ state=require_string(raw, "state", "evaluation run"),
status_reason=(
str(raw["statusReason"])
if raw.get("statusReason") is not None
@@ -585,19 +534,13 @@ def _run_ref(self, raw: Mapping[str, Any]) -> EvaluationRunRef:
def _build_handler_config(
self,
generation: GenerationConfig,
- tools: Mapping[str, ResolvedTool],
+ tools: Sequence[EvalTool],
) -> dict[str, Any]:
parameters = generation.get("parameters")
config: dict[str, Any] = {
"provider": {"name": generation["provider"]},
"model": {"name": generation["model"], "parameters": parameters},
- "tools": {
- key: {
- "description": tool.description,
- "parameters": tool.schema,
- }
- for key, tool in tools.items()
- },
+ "tools": handler_config_tools(tools),
}
snippet_variables = {"snippet": generation.get("prompt_snippets", {})}
if "instructions" in generation:
@@ -1150,9 +1093,9 @@ def _get_summary(
self, project_key: str, evaluation_id: str, run_id: str
) -> RunSummary:
path = (
- f"projects/{_segment(project_key)}/evaluations/{_segment(evaluation_id)}"
- f"/runs/{_segment(run_id)}/summary"
+ f"projects/{segment(project_key)}/evaluations/{segment(evaluation_id)}"
+ f"/runs/{segment(run_id)}/summary"
)
return RunSummary.from_wire(
- _mapping(self._api.get(path), description="evaluation run summary")
+ require_mapping(self._api.get(path), description="evaluation run summary")
)
diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/tools.py b/packages/client/src/launchdarkly_ai_server/evaluations/tools.py
new file mode 100644
index 00000000..94845924
--- /dev/null
+++ b/packages/client/src/launchdarkly_ai_server/evaluations/tools.py
@@ -0,0 +1,238 @@
+"""Tools an evaluation run gives to its handler."""
+
+from __future__ import annotations
+
+import asyncio
+import json
+from collections.abc import Callable, Mapping, Sequence
+from dataclasses import dataclass, field
+from typing import Any, Literal
+
+from ..types import NativeTool
+from .api import (
+ EvaluationsError,
+ LDApiClient,
+ LDApiError,
+ require_mapping,
+ segment,
+)
+
+ToolImplementation = Callable[..., Any] | NativeTool
+
+
+@dataclass(frozen=True)
+class EvalTool:
+ """A tool a run gives to its handler. Pass a list of these to ``run``.
+
+ Construct one to define a tool in code. ``schema`` is required, and ``{}``
+ is valid. ``source`` is then ``"inline"`` and ``version`` is ``None``.
+ Neither is a constructor argument, and the object is frozen, so a
+ constructed tool is always inline.
+
+ Call ``await evals.tools.get(key, implementation=...)`` instead to use a
+ tool from the LaunchDarkly tool library. That returns a tool with
+ ``source`` ``"library"`` and the version it pinned.
+
+ ``implementation`` is the function the handler calls. A key must be
+ lowercase. A :class:`~launchdarkly_ai_server.NativeTool` is valid only for
+ a library tool, because the provider supplies its schema.
+ ``dataclasses.replace`` on a library tool returns an inline tool that
+ keeps the library schema.
+ """
+
+ key: str
+ implementation: ToolImplementation
+ schema: dict[str, Any]
+ description: str = ""
+ source: Literal["library", "inline"] = field(default="inline", init=False)
+ version: int | None = field(default=None, init=False)
+ project_key: str | None = field(default=None, init=False)
+
+ @classmethod
+ def _library(
+ cls,
+ key: str,
+ implementation: ToolImplementation,
+ *,
+ version: int,
+ schema: dict[str, Any],
+ description: str,
+ project_key: str,
+ ) -> EvalTool:
+ """Build a library tool. Used by ``ToolsClient.get``."""
+ tool = cls(
+ key=key,
+ implementation=implementation,
+ schema=schema,
+ description=description,
+ )
+ object.__setattr__(tool, "source", "library")
+ object.__setattr__(tool, "version", version)
+ object.__setattr__(tool, "project_key", project_key)
+ return tool
+
+ def to_create_wire(self) -> dict[str, Any]:
+ """The entry this tool contributes to the evaluation-create body."""
+ if self.source == "inline":
+ return {
+ "key": self.key,
+ "schema": self.schema,
+ "description": self.description,
+ "source": "inline",
+ }
+ return {"key": self.key, "version": self.version, "source": "library"}
+
+
+def validate_tool_key(key: str) -> None:
+ """Validate a tool key. Raises ``EvaluationsError``."""
+ if not isinstance(key, str) or not key.strip():
+ raise EvaluationsError("tool keys must not be blank")
+ # Keys are lowercase.
+ if key != key.lower():
+ raise EvaluationsError(
+ f"Tool key {key!r} must not use uppercase letters. Use "
+ f"{key.lower()!r} instead."
+ )
+
+
+def validate_tool_implementation(key: str, implementation: Any) -> None:
+ """Validate a tool implementation. Raises ``EvaluationsError``."""
+ if not callable(implementation) and not isinstance(implementation, NativeTool):
+ raise EvaluationsError(
+ f"Tool {key!r} implementation must be callable or a NativeTool, got "
+ f"{type(implementation).__name__}"
+ )
+
+
+def _validate_inline_tool(tool: EvalTool) -> None:
+ """Validate one inline tool. Raises ``EvaluationsError``."""
+ key = tool.key
+ if isinstance(tool.implementation, NativeTool):
+ raise EvaluationsError(
+ f"Inline tool {key!r} must not use a NativeTool. The provider "
+ "supplies the schema of a native tool. Call evals.tools.get() for "
+ "a library tool, or give this tool a function."
+ )
+ if not isinstance(tool.schema, Mapping):
+ raise EvaluationsError(
+ f"Inline tool {key!r} schema must be a JSON object, got "
+ f"{type(tool.schema).__name__}"
+ )
+ try:
+ # allow_nan=False rejects NaN and Infinity, which are not valid JSON.
+ json.dumps(dict(tool.schema), allow_nan=False)
+ except (TypeError, ValueError) as error:
+ raise EvaluationsError(
+ f"Inline tool {key!r} schema must be JSON-serializable: {error}"
+ ) from error
+
+
+def validate_tools(tools: Sequence[EvalTool], project_key: str) -> None:
+ """Validate the tools list. Raises ``EvaluationsError``.
+
+ Checks that each library tool has a project and that it matches
+ ``project_key``. Issues no requests.
+ """
+ keys_by_identity: dict[str, str] = {}
+ for tool in tools:
+ if not isinstance(tool, EvalTool):
+ raise EvaluationsError(
+ "each entry in tools must be an EvalTool. Construct one for an "
+ "inline tool, or call evals.tools.get() for a library tool, "
+ f"got {type(tool).__name__}"
+ )
+ key = tool.key
+ validate_tool_key(key)
+ validate_tool_implementation(key, tool.implementation)
+ if not isinstance(tool.description, str):
+ raise EvaluationsError(
+ f"Tool {key!r} description must be a string, got "
+ f"{type(tool.description).__name__}"
+ )
+ if tool.source == "inline":
+ _validate_inline_tool(tool)
+ elif tool.project_key is None:
+ raise EvaluationsError(
+ f"Library tool {key!r} has no project. Read it with evals.tools.get()."
+ )
+ elif tool.project_key != project_key:
+ raise EvaluationsError(
+ f"Tool {key!r} was read from project {tool.project_key!r} and "
+ f"cannot run in project {project_key!r}. Read it from "
+ f"{project_key!r} instead."
+ )
+ # One key names one tool. Keys are compared case-insensitively.
+ identity = key.strip().lower()
+ collision = keys_by_identity.get(identity)
+ if collision is not None:
+ if collision == key:
+ raise EvaluationsError(f"Tool {key!r} appears more than once in tools.")
+ raise EvaluationsError(
+ f"Tool {key!r} collides with {collision!r}. Two tools in one "
+ "run must not have keys that differ only by case."
+ )
+ keys_by_identity[identity] = key
+
+
+def tool_handlers(tools: Sequence[EvalTool]) -> dict[str, ToolImplementation]:
+ """Return the tools list as ``{key: executable}``."""
+ return {tool.key: tool.implementation for tool in tools}
+
+
+def handler_config_tools(tools: Sequence[EvalTool]) -> dict[str, dict[str, Any]]:
+ """Return the ``tools`` entry of a handler config."""
+ return {
+ tool.key: {"description": tool.description, "parameters": tool.schema}
+ for tool in tools
+ }
+
+
+def create_wire_tools(tools: Sequence[EvalTool]) -> list[dict[str, Any]]:
+ """Return the ``tools`` array of an evaluation-create body."""
+ return [tool.to_create_wire() for tool in tools]
+
+
+class ToolsClient:
+ """Reads tools from the LaunchDarkly tool library."""
+
+ def __init__(self, api_client: LDApiClient, project_key: str) -> None:
+ self._api = api_client
+ self._project_key = project_key
+
+ async def get(self, key: str, *, implementation: ToolImplementation) -> EvalTool:
+ """Return the library tool ``key``, paired with ``implementation``.
+
+ Reads the tool now and pins the version it returns. Raises
+ ``EvaluationsError`` when the tool does not exist in the project.
+
+ The read runs in a worker thread, so it does not block the event loop.
+ """
+ validate_tool_key(key)
+ validate_tool_implementation(key, implementation)
+ path = f"projects/{segment(self._project_key)}/ai-tools/{segment(key)}"
+ try:
+ response = await asyncio.to_thread(self._api.get, path)
+ raw = require_mapping(response, description=f"tool {key!r}")
+ except LDApiError as error:
+ if error.status == 404:
+ raise EvaluationsError(
+ f"LaunchDarkly AI tool {key!r} was not found in project "
+ f"{self._project_key!r}"
+ ) from error
+ raise
+ version = raw.get("version")
+ if not isinstance(version, int):
+ raise EvaluationsError(
+ f"LaunchDarkly AI tool {key!r} has no integer version"
+ )
+ schema = raw.get("schema")
+ if not isinstance(schema, Mapping):
+ schema = {}
+ return EvalTool._library(
+ key,
+ implementation,
+ version=version,
+ schema=dict(schema),
+ description=str(raw.get("description") or ""),
+ project_key=self._project_key,
+ )
diff --git a/packages/client/src/launchdarkly_ai_server/evaluations/types.py b/packages/client/src/launchdarkly_ai_server/evaluations/types.py
index 75af1aaa..fa1d4655 100644
--- a/packages/client/src/launchdarkly_ai_server/evaluations/types.py
+++ b/packages/client/src/launchdarkly_ai_server/evaluations/types.py
@@ -57,16 +57,6 @@ class DatasetRow:
metadata: dict[str, Any] | None = None
-@dataclass
-class ResolvedTool:
- """The schema and pinned version returned by the LaunchDarkly tool API."""
-
- key: str
- version: int
- description: str = ""
- schema: dict[str, Any] = field(default_factory=dict)
-
-
@dataclass(frozen=True)
class AIConfig:
"""Identifies an existing AI Config variation to seed an evaluation run.
diff --git a/packages/client/tests/test_evaluations.py b/packages/client/tests/test_evaluations.py
index e4c27226..282b07ab 100644
--- a/packages/client/tests/test_evaluations.py
+++ b/packages/client/tests/test_evaluations.py
@@ -67,34 +67,37 @@ def failing_transport(
def test_init_resolves_credentials_from_env(monkeypatch: pytest.MonkeyPatch) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.setenv("LD_API_TOKEN", "api-token-from-env")
monkeypatch.setenv("LD_SDK_KEY", "sdk-key-from-env")
evals = init_evaluations(transport=RecordingTransport())
- assert evals.api.api_token == "api-token-from-env"
+ assert evals.api.api_key == "api-token-from-env"
assert evals.sdk_key == "sdk-key-from-env"
assert evals.api.base_uri == DEFAULT_BASE_URI
assert evals.ui_base_uri == "https://app.launchdarkly.com"
def test_init_prefers_explicit_credentials(monkeypatch: pytest.MonkeyPatch) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.setenv("LD_API_TOKEN", "api-token-from-env")
monkeypatch.setenv("LD_SDK_KEY", "sdk-key-from-env")
evals = init_evaluations(
- api_token="explicit-token",
+ api_key="explicit-token",
sdk_key="explicit-sdk-key",
transport=RecordingTransport(),
)
- assert evals.api.api_token == "explicit-token"
+ assert evals.api.api_key == "explicit-token"
assert evals.sdk_key == "explicit-sdk-key"
def test_missing_api_token_raises_before_network_io(
monkeypatch: pytest.MonkeyPatch,
) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.delenv("LD_API_TOKEN", raising=False)
monkeypatch.setenv("LD_SDK_KEY", "sdk-key")
@@ -105,6 +108,7 @@ def test_missing_api_token_raises_before_network_io(
def test_blank_api_token_env_is_treated_as_unset(
monkeypatch: pytest.MonkeyPatch,
) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.setenv("LD_API_TOKEN", " ")
with pytest.raises(EvaluationsError):
@@ -114,6 +118,7 @@ def test_blank_api_token_env_is_treated_as_unset(
def test_missing_sdk_key_raises_before_network_io(
monkeypatch: pytest.MonkeyPatch,
) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.setenv("LD_API_TOKEN", "api-token")
monkeypatch.delenv("LD_SDK_KEY", raising=False)
@@ -124,6 +129,7 @@ def test_missing_sdk_key_raises_before_network_io(
def test_blank_sdk_key_env_is_treated_as_unset(
monkeypatch: pytest.MonkeyPatch,
) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.setenv("LD_API_TOKEN", "api-token")
monkeypatch.setenv("LD_SDK_KEY", " ")
@@ -134,6 +140,7 @@ def test_blank_sdk_key_env_is_treated_as_unset(
def test_missing_sdk_key_is_allowed_with_a_byoc_client(
monkeypatch: pytest.MonkeyPatch,
) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.setenv("LD_API_TOKEN", "api-token")
monkeypatch.delenv("LD_SDK_KEY", raising=False)
byoc_client = MagicMock()
@@ -149,6 +156,7 @@ def test_missing_sdk_key_is_allowed_with_a_byoc_client(
def test_missing_sdk_key_raises_when_the_byoc_client_cannot_emit_events(
monkeypatch: pytest.MonkeyPatch,
) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.setenv("LD_API_TOKEN", "api-token")
monkeypatch.delenv("LD_SDK_KEY", raising=False)
lifecycle_module._set_client_for_testing(object())
@@ -160,6 +168,7 @@ def test_missing_sdk_key_raises_when_the_byoc_client_cannot_emit_events(
def test_base_uri_override_isolated_from_sdk_delivery_uri(
monkeypatch: pytest.MonkeyPatch,
) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.setenv("LD_API_TOKEN", "api-token")
monkeypatch.setenv("LD_SDK_KEY", "sdk-key")
monkeypatch.setenv("LD_API_BASE_URI", "https://api.staging.example.com/")
@@ -177,6 +186,7 @@ def test_base_uri_override_isolated_from_sdk_delivery_uri(
def test_ui_base_uri_precedence_and_api_base_isolation(
monkeypatch: pytest.MonkeyPatch,
) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj")
monkeypatch.setenv("LD_API_TOKEN", "api-token")
monkeypatch.setenv("LD_SDK_KEY", "sdk-key")
monkeypatch.setenv("LD_API_BASE_URI", "https://api.staging.example.com")
@@ -194,7 +204,7 @@ def test_ui_base_uri_precedence_and_api_base_isolation(
def test_requests_carry_token_auth_and_json_body() -> None:
transport = RecordingTransport([HttpResponse(status=201, body='{"key": "run-1"}')])
- client = LDApiClient(api_token="api-token", transport=transport)
+ client = LDApiClient(api_key="api-token", transport=transport)
result = client.post("projects/proj/evaluations", body={"key": "support-qa"})
@@ -211,7 +221,7 @@ def test_requests_carry_token_auth_and_json_body() -> None:
def test_get_encodes_query_params_and_omits_none() -> None:
transport = RecordingTransport([HttpResponse(status=200, body='{"items": []}')])
client = LDApiClient(
- api_token="api-token", base_uri="https://ld.example.com", transport=transport
+ api_key="api-token", base_uri="https://ld.example.com", transport=transport
)
client.get("projects/proj/datasets/golden", params={"limit": 50, "offset": None})
@@ -238,7 +248,7 @@ def test_rate_limit_retries_and_honors_retry_after() -> None:
)
sleeps: list[float] = []
client = LDApiClient(
- api_token="api-token",
+ api_key="api-token",
transport=transport,
max_retries=1,
sleep=sleeps.append,
@@ -256,7 +266,7 @@ def test_server_error_retries_get_but_not_post() -> None:
[server_error, HttpResponse(200, '{"ok": true}')]
)
client = LDApiClient(
- api_token="api-token",
+ api_key="api-token",
transport=get_transport,
max_retries=2,
sleep=lambda _: None,
@@ -268,7 +278,7 @@ def test_server_error_retries_get_but_not_post() -> None:
post_transport = RecordingTransport([server_error])
client = LDApiClient(
- api_token="api-token",
+ api_key="api-token",
transport=post_transport,
max_retries=2,
sleep=lambda _: None,
@@ -296,7 +306,7 @@ def timing_out_transport(
raise TimeoutError("timed out")
client = LDApiClient(
- api_token="api-token",
+ api_key="api-token",
transport=timing_out_transport,
max_retries=2,
sleep=lambda _: None,
@@ -317,7 +327,7 @@ def test_rate_limited_post_is_retried() -> None:
]
)
client = LDApiClient(
- api_token="api-token",
+ api_key="api-token",
transport=transport,
max_retries=1,
sleep=lambda _: None,
@@ -334,7 +344,7 @@ def test_forbidden_response_is_not_retried() -> None:
transport = RecordingTransport(
[HttpResponse(status=403, body='{"message": "forbidden"}')]
)
- client = LDApiClient(api_token="api-token", transport=transport, max_retries=3)
+ client = LDApiClient(api_key="api-token", transport=transport, max_retries=3)
with pytest.raises(LDApiError) as excinfo:
client.get("projects/proj/evaluations")
@@ -347,7 +357,7 @@ def test_error_response_raises_ld_api_error() -> None:
transport = RecordingTransport(
[HttpResponse(status=404, body='{"message": "nope"}')]
)
- client = LDApiClient(api_token="api-token", transport=transport)
+ client = LDApiClient(api_key="api-token", transport=transport)
with pytest.raises(LDApiError) as excinfo:
client.get("projects/proj/ai-tools/missing")
@@ -358,7 +368,7 @@ def test_error_response_raises_ld_api_error() -> None:
def test_empty_response_body_is_none() -> None:
transport = RecordingTransport([HttpResponse(status=204, body="")])
- client = LDApiClient(api_token="api-token", transport=transport)
+ client = LDApiClient(api_key="api-token", transport=transport)
assert client.post("projects/proj/evaluations/support-qa/runs") is None
@@ -395,3 +405,32 @@ def test_run_summary_and_result() -> None:
assert RunSummary.from_wire(None) == RunSummary()
assert result.passed is False
assert result.run_id == "run-1"
+
+
+def test_project_key_resolves_from_the_environment(monkeypatch) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj-from-env")
+ monkeypatch.setenv("LD_API_TOKEN", "api-token")
+ monkeypatch.setenv("LD_SDK_KEY", "sdk-key")
+
+ evals = init_evaluations(transport=RecordingTransport())
+
+ assert evals.project_key == "proj-from-env"
+
+
+def test_explicit_project_key_wins_over_the_environment(monkeypatch) -> None:
+ monkeypatch.setenv("LD_PROJECT_KEY", "proj-from-env")
+ monkeypatch.setenv("LD_API_TOKEN", "api-token")
+ monkeypatch.setenv("LD_SDK_KEY", "sdk-key")
+
+ evals = init_evaluations(project_key="explicit", transport=RecordingTransport())
+
+ assert evals.project_key == "explicit"
+
+
+def test_missing_project_key_raises_before_network_io(monkeypatch) -> None:
+ monkeypatch.delenv("LD_PROJECT_KEY", raising=False)
+ monkeypatch.setenv("LD_API_TOKEN", "api-token")
+ monkeypatch.setenv("LD_SDK_KEY", "sdk-key")
+
+ with pytest.raises(EvaluationsError, match="LD_PROJECT_KEY"):
+ init_evaluations(transport=failing_transport)
diff --git a/packages/client/tests/test_evaluations_run.py b/packages/client/tests/test_evaluations_run.py
index 4a4b8571..5a8c296e 100644
--- a/packages/client/tests/test_evaluations_run.py
+++ b/packages/client/tests/test_evaluations_run.py
@@ -1,6 +1,8 @@
from __future__ import annotations
+import dataclasses
import json
+import threading
from collections.abc import Callable
from datetime import datetime
from typing import Any
@@ -8,10 +10,11 @@
import pytest
-from launchdarkly_ai_server import create_handler
+from launchdarkly_ai_server import NativeTool, create_handler
from launchdarkly_ai_server.evaluations import (
AIConfig,
DatasetRow,
+ EvalTool,
EvaluationsError,
HttpResponse,
Judge,
@@ -119,6 +122,10 @@ def lookup_order(order_id: str) -> str:
return order_id
+def refund_order(order_id: str) -> str:
+ return f"refunded {order_id}"
+
+
@pytest.mark.asyncio
async def test_complete_run_with_zero_failed_and_error_rows_passes(
monkeypatch: pytest.MonkeyPatch,
@@ -222,7 +229,8 @@ async def test_complete_run_with_zero_failed_and_error_rows_passes(
client.flush = AsyncMock()
init_client.return_value = client
evals = init_evaluations(
- api_token="token",
+ project_key="proj",
+ api_key="token",
sdk_key="sdk-key",
base_uri="https://api.example.com",
ui_base_uri="https://ui.example.com/",
@@ -231,11 +239,10 @@ async def test_complete_run_with_zero_failed_and_error_rows_passes(
assert evals.sdk_key == "sdk-key"
result = await evals.run(
- project_key="proj",
key="support-qa-unique",
dataset="golden",
handler=successful_handler,
- tools={"lookup_order": lookup_order},
+ tools=[await evals.tools.get("lookup_order", implementation=lookup_order)],
generation={
"provider": "OpenAI",
"model": "gpt-4o",
@@ -277,7 +284,7 @@ async def test_complete_run_with_zero_failed_and_error_rows_passes(
"generationModel": "gpt-4o",
"parameters": {"temperature": 0.2},
"messages": [{"role": "system", "content": "Help the user."}],
- "tools": [{"key": "lookup_order", "version": 7}],
+ "tools": [{"key": "lookup_order", "version": 7, "source": "library"}],
}
assert transport.requests[5]["url"].endswith(
"/api/v2/projects/proj/evaluations/11111111-1111-1111-1111-111111111111/runs"
@@ -401,13 +408,14 @@ async def fake_init_client(options: dict[str, Any]) -> MagicMock:
monkeypatch.setattr(
"launchdarkly_ai_server.evaluations.module.init_client", fake_init_client
)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -464,13 +472,12 @@ async def test_summary_is_polled_until_rows_are_accounted(
),
]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -526,13 +533,12 @@ async def test_summary_polling_completes_for_real_backend_summary_without_state(
),
]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -585,13 +591,12 @@ async def test_summary_polling_ignores_missing_state_even_when_pending_is_zero(
),
]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -635,7 +640,7 @@ async def test_summary_polling_times_out_waiting_for_rows_to_be_accounted(
),
]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
@@ -645,7 +650,6 @@ async def handler(*args: object) -> dict[str, Any]:
match=r"Timed out after 0 seconds.*rows to be fully accounted.*pending_rows=1",
):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -684,13 +688,12 @@ async def test_poll_timeout_and_interval_are_configurable_per_run() -> None:
),
]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -707,7 +710,6 @@ async def handler(*args: object) -> dict[str, Any]:
with pytest.raises(EvaluationsError, match="poll_timeout_seconds"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -725,14 +727,15 @@ async def handler(*args: object) -> dict[str, Any]:
async def test_nan_poll_values_are_rejected(
poll_interval_seconds: float, poll_timeout_seconds: float
) -> None:
- evals = init_evaluations(api_token="token", transport=failing_transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", transport=failing_transport
+ )
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
with pytest.raises(EvaluationsError, match="must be a number"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -777,14 +780,13 @@ async def test_run_uses_a_byoc_client_when_no_sdk_key_is_configured(
),
]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
assert evals.sdk_key is None
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -808,7 +810,9 @@ async def test_run_raises_when_no_sdk_key_and_no_initialized_client(
monkeypatch.setattr(
"launchdarkly_ai_server.evaluations.module.get_client", get_client
)
- evals = init_evaluations(api_token="token", transport=failing_transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", transport=failing_transport
+ )
get_client.side_effect = RuntimeError("client not initialized")
async def handler(*args: object) -> dict[str, Any]:
@@ -816,7 +820,6 @@ async def handler(*args: object) -> dict[str, Any]:
with pytest.raises(EvaluationsError, match="no initialized LaunchDarkly client"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -863,13 +866,12 @@ async def test_failed_rows_fail_the_result() -> None:
),
]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -883,11 +885,10 @@ async def handler(*args: object) -> dict[str, Any]:
@pytest.mark.asyncio
async def test_run_rejects_instructions_and_messages_before_network_io() -> None:
transport = SequencedTransport([])
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
with pytest.raises(EvaluationsError, match=r"instructions.*messages"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=successful_handler,
@@ -907,19 +908,457 @@ async def test_missing_tool_aborts_before_any_mutating_request() -> None:
transport = SequencedTransport(
[response(404, {"code": "not_found", "message": "not found"})]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
with pytest.raises(EvaluationsError, match="missing_tool"):
+ await evals.tools.get("missing_tool", implementation=lookup_order)
+
+ assert [request["method"] for request in transport.requests] == ["GET"]
+
+
+ORDER_SCHEMA: dict[str, Any] = {
+ "type": "object",
+ "properties": {"order_id": {"type": "string"}},
+ "required": ["order_id"],
+}
+
+
+def recorded_paths(transport: SequencedTransport) -> list[tuple[str, str]]:
+ """The recorded requests as ``(method, path)`` pairs, query strings dropped."""
+ return [
+ (
+ request["method"],
+ request["url"].split("/api/v2/", 1)[-1].split("?", 1)[0],
+ )
+ for request in transport.requests
+ ]
+
+
+def hosted_dataset_responses() -> list[HttpResponse]:
+ """Canned responses for a one-row hosted dataset run that passes."""
+ return [
+ response(200, {"id": "dataset-id", "name": "golden"}),
+ response(
+ 200,
+ dataset_page(
+ [{"rowIndex": 0, "input": "Order A19", "variables": {}}], total=1
+ ),
+ ),
+ response(201, {"id": "evaluation-id", "name": "eval-key", "version": 3}),
+ response(
+ 201,
+ {"id": "run-id", "evaluationId": "evaluation-id", "state": "PENDING"},
+ ),
+ response(
+ 200,
+ {
+ "statusCounts": {
+ "total": 1,
+ "passed": 1,
+ "failed": 0,
+ "error": 0,
+ "pending": 0,
+ }
+ },
+ ),
+ ]
+
+
+@pytest.mark.asyncio
+async def test_inline_tool_runs_without_reading_the_tool_api() -> None:
+ transport = SequencedTransport(hosted_dataset_responses())
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+ seen: dict[str, Any] = {}
+
+ async def handler(
+ config: dict[str, Any],
+ user_input: str | None,
+ tool_handlers: dict[str, Callable[..., Any]],
+ variables: dict[str, Any],
+ ) -> dict[str, Any]:
+ seen["config_tools"] = config["tools"]
+ seen["tool_handlers"] = tool_handlers
+ return {"output": "ok"}
+
+ result = await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=handler,
+ tools=[
+ EvalTool(
+ key="lookup_order",
+ implementation=lookup_order,
+ schema=ORDER_SCHEMA,
+ description="Look up an order",
+ )
+ ],
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ )
+
+ assert result.passed is True
+ # The full sequence, so a stray request anywhere in the run is visible.
+ assert recorded_paths(transport) == [
+ ("GET", "projects/proj/datasets/golden"),
+ ("GET", "projects/proj/datasets/golden/rows"),
+ ("POST", "projects/proj/evaluations"),
+ ("POST", "projects/proj/evaluations/evaluation-id/runs"),
+ ("GET", "projects/proj/evaluations/evaluation-id/runs/run-id/summary"),
+ ]
+ # Asserted explicitly rather than left to the sequence above: a transport
+ # that tolerated a surplus request would not fail on a stray tool GET, and
+ # skipping that request is the whole point of an inline definition.
+ assert not any("/ai-tools" in request["url"] for request in transport.requests), (
+ "an inline tool must not be looked up in the AI library"
+ )
+
+ assert transport.requests[2]["body"]["tools"] == [
+ {
+ "key": "lookup_order",
+ "schema": ORDER_SCHEMA,
+ "description": "Look up an order",
+ "source": "inline",
+ }
+ ]
+ # Same config shape a library tool produces, fed from the inline body, so a
+ # handler cannot tell the two sources apart.
+ assert seen["config_tools"] == {
+ "lookup_order": {
+ "description": "Look up an order",
+ "parameters": ORDER_SCHEMA,
+ }
+ }
+ # The handler is passed the executable, not the definition wrapping it.
+ assert list(seen["tool_handlers"]) == ["lookup_order"]
+ assert seen["tool_handlers"]["lookup_order"]("A1") == "A1"
+
+
+@pytest.mark.asyncio
+async def test_inline_tool_description_defaults_to_empty_string() -> None:
+ transport = SequencedTransport(hosted_dataset_responses())
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ async def handler(*args: object) -> dict[str, Any]:
+ return {"output": "ok"}
+
+ await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=handler,
+ tools=[
+ EvalTool(
+ key="lookup_order", implementation=lookup_order, schema=ORDER_SCHEMA
+ )
+ ],
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ )
+
+ assert transport.requests[2]["body"]["tools"] == [
+ {
+ "key": "lookup_order",
+ "schema": ORDER_SCHEMA,
+ "description": "",
+ "source": "inline",
+ }
+ ]
+
+
+@pytest.mark.asyncio
+async def test_mixed_library_and_inline_tools_each_keep_their_own_source() -> None:
+ transport = SequencedTransport(
+ [
+ response(
+ 200,
+ {
+ "key": "lookup_order",
+ "version": 7,
+ "description": "Look up an order",
+ "schema": {"type": "object"},
+ },
+ ),
+ *hosted_dataset_responses(),
+ ]
+ )
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+ seen: dict[str, Any] = {}
+
+ async def handler(
+ config: dict[str, Any],
+ user_input: str | None,
+ tool_handlers: dict[str, Callable[..., Any]],
+ variables: dict[str, Any],
+ ) -> dict[str, Any]:
+ seen["config_tools"] = config["tools"]
+ seen["tool_handlers"] = tool_handlers
+ return {"output": "ok"}
+
+ await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=handler,
+ tools=[
+ await evals.tools.get("lookup_order", implementation=lookup_order),
+ EvalTool(
+ key="refund_order",
+ implementation=refund_order,
+ schema=ORDER_SCHEMA,
+ description="Refund an order",
+ ),
+ ],
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ )
+
+ # Exactly one tool GET, for the library key only.
+ assert [
+ path for method, path in recorded_paths(transport) if "ai-tools" in path
+ ] == ["projects/proj/ai-tools/lookup_order"]
+ assert transport.requests[3]["body"]["tools"] == [
+ {"key": "lookup_order", "version": 7, "source": "library"},
+ {
+ "key": "refund_order",
+ "schema": ORDER_SCHEMA,
+ "description": "Refund an order",
+ "source": "inline",
+ },
+ ]
+ assert seen["config_tools"] == {
+ "lookup_order": {
+ "description": "Look up an order",
+ "parameters": {"type": "object"},
+ },
+ "refund_order": {
+ "description": "Refund an order",
+ "parameters": ORDER_SCHEMA,
+ },
+ }
+ assert sorted(seen["tool_handlers"]) == ["lookup_order", "refund_order"]
+ assert seen["tool_handlers"]["lookup_order"]("A1") == "A1"
+ assert seen["tool_handlers"]["refund_order"]("A1") == "refunded A1"
+
+
+@pytest.mark.asyncio
+@pytest.mark.parametrize(
+ ("tools", "match"),
+ [
+ pytest.param(
+ [EvalTool(key=" ", implementation=lookup_order, schema=ORDER_SCHEMA)],
+ "must not be blank",
+ id="blank_key",
+ ),
+ pytest.param(
+ [
+ EvalTool(
+ key="Lookup_Order", implementation=lookup_order, schema=ORDER_SCHEMA
+ )
+ ],
+ "must not use uppercase letters",
+ id="uppercase_key",
+ ),
+ pytest.param(
+ [EvalTool(key="lookup_order", implementation=lookup_order, schema=None)], # type: ignore[arg-type]
+ "schema must be a JSON object",
+ id="schema_is_none",
+ ),
+ pytest.param(
+ [
+ EvalTool(
+ key="lookup_order",
+ implementation=lookup_order,
+ schema=[{"type": "object"}],
+ )
+ ], # type: ignore[arg-type]
+ "schema must be a JSON object",
+ id="schema_is_a_list",
+ ),
+ pytest.param(
+ [
+ EvalTool(
+ key="lookup_order",
+ implementation=lookup_order,
+ schema={"default": object()},
+ )
+ ],
+ "schema must be JSON-serializable",
+ id="schema_is_not_serializable",
+ ),
+ pytest.param(
+ [
+ EvalTool(
+ key="lookup_order",
+ implementation=lookup_order,
+ schema={"default": float("nan")},
+ )
+ ],
+ "schema must be JSON-serializable",
+ id="schema_holds_nan",
+ ),
+ pytest.param(
+ [
+ EvalTool(
+ key="lookup_order",
+ implementation=lookup_order,
+ schema={"default": float("inf")},
+ )
+ ],
+ "schema must be JSON-serializable",
+ id="schema_holds_infinity",
+ ),
+ pytest.param(
+ [
+ EvalTool(
+ key="lookup_order",
+ implementation="not a function",
+ schema=ORDER_SCHEMA,
+ )
+ ], # type: ignore[arg-type]
+ "implementation must be callable",
+ id="implementation_is_not_callable",
+ ),
+ pytest.param(
+ [
+ EvalTool(
+ key="lookup_order",
+ implementation=lookup_order,
+ schema=ORDER_SCHEMA,
+ description=None,
+ )
+ ], # type: ignore[arg-type]
+ "description must be a string",
+ id="description_is_not_a_string",
+ ),
+ pytest.param(
+ ["not a tool"], # type: ignore[dict-item]
+ "each entry in tools must be an EvalTool",
+ id="entry_is_not_a_tool",
+ ),
+ ],
+)
+async def test_bad_tool_entry_is_rejected_with_zero_requests(
+ tools: list[Any], match: str
+) -> None:
+ transport = SequencedTransport([])
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ with pytest.raises(EvaluationsError, match=match):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=successful_handler,
- tools={"missing_tool": lookup_order},
+ tools=tools,
generation={"provider": "OpenAI", "model": "gpt-4o"},
)
- assert [request["method"] for request in transport.requests] == ["GET"]
+ assert transport.requests == []
+
+
+@pytest.mark.asyncio
+async def test_native_tool_paired_with_an_inline_definition_is_rejected() -> None:
+ transport = SequencedTransport([])
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ with pytest.raises(EvaluationsError, match=r"lookup_order.*NativeTool"):
+ await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=successful_handler,
+ tools=[
+ EvalTool(
+ key="lookup_order",
+ implementation=NativeTool("WebSearch"),
+ schema=ORDER_SCHEMA,
+ )
+ ],
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ )
+
+ assert transport.requests == []
+
+
+@pytest.mark.asyncio
+async def test_native_tool_on_its_own_still_resolves_from_the_library() -> None:
+ transport = SequencedTransport(
+ [
+ response(200, {"key": "web_search", "version": 2}),
+ *hosted_dataset_responses(),
+ ]
+ )
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ async def handler(*args: object) -> dict[str, Any]:
+ return {"output": "ok"}
+
+ await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=handler,
+ tools=[
+ await evals.tools.get("web_search", implementation=NativeTool("WebSearch"))
+ ],
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ )
+
+ assert recorded_paths(transport)[0] == ("GET", "projects/proj/ai-tools/web_search")
+ assert transport.requests[3]["body"]["tools"] == [
+ {"key": "web_search", "version": 2, "source": "library"}
+ ]
+
+
+@pytest.mark.asyncio
+async def test_a_repeated_tool_key_is_rejected_with_zero_requests() -> None:
+ """One key names one tool, whichever source each entry came from."""
+ transport = SequencedTransport([])
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ with pytest.raises(
+ EvaluationsError, match=r"'lookup_order' appears more than once"
+ ):
+ await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=successful_handler,
+ tools=[
+ EvalTool(
+ key="lookup_order",
+ implementation=lookup_order,
+ schema=ORDER_SCHEMA,
+ ),
+ EvalTool(
+ key="lookup_order",
+ implementation=refund_order,
+ schema=ORDER_SCHEMA,
+ ),
+ ],
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ )
+
+ assert transport.requests == []
+
+
+@pytest.mark.asyncio
+async def test_tools_get_refuses_an_uppercase_key_before_any_request() -> None:
+ """A tool key is lowercase, whether it is constructed or read from the library."""
+ transport = SequencedTransport([])
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ with pytest.raises(EvaluationsError, match="must not use uppercase letters"):
+ await evals.tools.get("Lookup_Order", implementation=lookup_order)
+
+ assert transport.requests == []
@pytest.mark.asyncio
@@ -936,11 +1375,10 @@ async def test_empty_dataset_fails_before_evaluation_or_run_creation() -> None:
response(200, dataset_page([], total=0)),
]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
with pytest.raises(EvaluationsError, match="empty"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=successful_handler,
@@ -1034,10 +1472,11 @@ async def fake_init_client(options: dict[str, Any]) -> MagicMock:
monkeypatch.setattr(
"launchdarkly_ai_server.evaluations.module.init_client", fake_init_client
)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
result = await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -1117,7 +1556,9 @@ async def fake_extract_variation(
"launchdarkly_ai_server.evaluations.runner.extract_variation",
fake_extract_variation,
)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(
config: dict[str, Any],
@@ -1156,7 +1597,6 @@ async def handler(
}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1268,7 +1708,9 @@ async def fake_extract_variation(
"launchdarkly_ai_server.evaluations.runner.extract_variation",
fake_extract_variation,
)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(
config: dict[str, Any],
@@ -1287,7 +1729,6 @@ async def handler(
}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1363,7 +1804,9 @@ async def fake_extract_variation(
"launchdarkly_ai_server.evaluations.runner.extract_variation",
fake_extract_variation,
)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(
config: dict[str, Any],
@@ -1376,7 +1819,6 @@ async def handler(
return {"output": "generated"}
await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1437,7 +1879,9 @@ async def fake_extract_variation(
"launchdarkly_ai_server.evaluations.runner.extract_variation",
fake_extract_variation,
)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
@@ -1447,7 +1891,6 @@ async def handler(*args: object) -> dict[str, Any]:
match=r"Failed to resolve LaunchDarkly judge 'security-judge': not found",
):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -1490,7 +1933,9 @@ async def test_run_with_deterministic_scorer_emits_scorer_evaluation_event(
),
]
)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(*args: object) -> dict[str, Any]:
return {
@@ -1505,7 +1950,6 @@ def check_refund(row: DatasetRow, output: Any) -> bool:
return "refund" in str(output)
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="support-golden-v3",
handler=handler,
@@ -1621,7 +2065,9 @@ async def test_bad_judge_output_emits_error_event_instead_of_crashing(
) -> None:
transport = judge_run_transport()
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(
config: dict[str, Any],
@@ -1634,7 +2080,6 @@ async def handler(
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1661,7 +2106,9 @@ async def test_generated_placeholders_are_not_expanded_into_judge_prompt(
transport = judge_run_transport()
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(
config: dict[str, Any],
@@ -1678,7 +2125,6 @@ async def handler(
return {"output": "{{expected_output}} leaked?"}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1717,7 +2163,9 @@ async def test_missing_expected_output_renders_empty_judge_variables(
]
)
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(
config: dict[str, Any],
@@ -1732,7 +2180,6 @@ async def handler(
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1745,14 +2192,15 @@ async def handler(
@pytest.mark.asyncio
async def test_duplicate_criteria_rejected_before_any_request() -> None:
transport = SequencedTransport([])
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
with pytest.raises(EvaluationsError, match="Duplicate evaluation criteria"):
await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1773,14 +2221,15 @@ async def test_duplicate_criteria_rejected_case_insensitively() -> None:
only by case would still collide there even though they'd look distinct
to a case-sensitive check."""
transport = SequencedTransport([])
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
with pytest.raises(EvaluationsError, match="Duplicate evaluation criteria"):
await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1803,7 +2252,9 @@ async def test_errored_generation_row_emits_generation_incomplete_criterion_even
summary={"statusCounts": {"total": 1, "passed": 0, "error": 1, "pending": 0}}
)
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(
config: dict[str, Any],
@@ -1816,7 +2267,6 @@ async def handler(
raise RuntimeError("provider unavailable")
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1854,7 +2304,9 @@ def track(event_name: str, *args: Any) -> None:
raise RuntimeError("event pipeline unavailable")
stub_sdk_client.track = MagicMock(side_effect=track)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(
config: dict[str, Any],
@@ -1868,7 +2320,6 @@ async def handler(
with pytest.raises(EvaluationsError) as error:
await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -1955,11 +2406,12 @@ async def test_judge_on_another_provider_fails_before_any_records_are_created(
"""
transport = judge_run_transport()
judge_variation(monkeypatch, provider="Anthropic")
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
with pytest.raises(EvaluationsError) as error:
await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=create_handler(("OpenAI", "messages"), _generation_only),
@@ -1979,7 +2431,9 @@ async def test_judge_handlers_route_a_judge_to_its_own_provider(
) -> None:
transport = judge_run_transport()
judge_variation(monkeypatch, provider="Anthropic")
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
judged: list[dict[str, Any]] = []
async def anthropic_judge(
@@ -1993,7 +2447,6 @@ async def anthropic_judge(
return {"output": '{"score": 0.75, "reasoning": "ok"}'}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=create_handler(("OpenAI", "messages"), _generation_only),
@@ -2024,7 +2477,9 @@ async def test_exact_provider_judge_handler_beats_a_wildcard_adapter(
"""
transport = judge_run_transport()
judge_variation(monkeypatch, provider="Anthropic")
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
chosen: list[str] = []
def judge_handler(name: str) -> Any:
@@ -2044,7 +2499,6 @@ async def run(
exact = create_handler(("Anthropic", "messages"), judge_handler("exact"))
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=create_handler(("OpenAI", "messages"), _generation_only),
@@ -2064,7 +2518,9 @@ async def test_wildcard_judge_handler_runs_a_judge_no_handler_names(
) -> None:
transport = judge_run_transport()
judge_variation(monkeypatch, provider="Anthropic")
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
judged: list[dict[str, Any]] = []
async def wildcard_judge(
@@ -2078,7 +2534,6 @@ async def wildcard_judge(
return {"output": '{"score": 1, "reasoning": "ok"}'}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=create_handler(("OpenAI", "messages"), _generation_only),
@@ -2109,7 +2564,9 @@ async def test_agent_handler_runs_a_messages_mode_judge_with_collapsed_messages(
]
},
)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
judged: list[dict[str, Any]] = []
async def anthropic_agent_judge(
@@ -2123,7 +2580,6 @@ async def anthropic_agent_judge(
return {"output": '{"score": 1, "reasoning": "ok"}'}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=create_handler(("OpenAI", "messages"), _generation_only),
@@ -2146,7 +2602,9 @@ async def test_generation_handler_runs_a_judge_on_the_same_provider(
) -> None:
transport = judge_run_transport()
judge_variation(monkeypatch, provider="OpenAI")
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
calls: list[str | None] = []
async def openai_handler(
@@ -2162,7 +2620,6 @@ async def openai_handler(
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=create_handler(("OpenAI", "messages"), openai_handler),
@@ -2181,11 +2638,12 @@ async def test_judge_handlers_must_declare_the_provider_they_serve(
"""An unrouted judge handler would silently never be selected."""
transport = judge_run_transport()
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
with pytest.raises(EvaluationsError, match="does not declare provides_for"):
await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=_generation_only,
@@ -2229,7 +2687,9 @@ async def test_criteria_run_concurrently_within_the_concurrency_bound(
]
)
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
in_flight = 0
max_in_flight = 0
@@ -2250,7 +2710,6 @@ async def handler(
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -2316,7 +2775,9 @@ async def test_tool_trajectory_reaches_the_judge_via_message_history(
"""
transport = tool_run_transport()
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
seen: dict[str, str] = {}
def lookup_order(args: dict[str, Any]) -> str:
@@ -2341,11 +2802,10 @@ async def handler(
return {"output": "Your order shipped."}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
- tools={"lookup_order": lookup_order},
+ tools=[await evals.tools.get("lookup_order", implementation=lookup_order)],
generation={"provider": "OpenAI", "model": "gpt-4o"},
criteria=[Judge(key="$ld:ai:judge:accuracy")],
)
@@ -2379,7 +2839,9 @@ async def test_each_row_gets_only_its_own_tool_trajectory(
transport = tool_run_transport(rows=2)
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
histories: dict[str, str] = {}
both_started = asyncio.Barrier(2)
@@ -2403,11 +2865,10 @@ async def handler(
return {"output": f"answered {row}"}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
- tools={"lookup_order": lookup_order},
+ tools=[await evals.tools.get("lookup_order", implementation=lookup_order)],
generation={"provider": "OpenAI", "model": "gpt-4o"},
criteria=[Judge(key="$ld:ai:judge:accuracy")],
concurrency=2,
@@ -2428,7 +2889,9 @@ async def test_a_row_that_called_no_tools_says_so_to_the_judge(
"""A judge grading tool selection needs to see the tool that went unused."""
transport = tool_run_transport()
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
seen: dict[str, str] = {}
async def handler(
@@ -2443,11 +2906,12 @@ async def handler(
return {"output": "I do not know."}
await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
- tools={"lookup_order": lambda args: "unused"},
+ tools=[
+ await evals.tools.get("lookup_order", implementation=lambda args: "unused")
+ ],
generation={"provider": "OpenAI", "model": "gpt-4o"},
criteria=[Judge(key="$ld:ai:judge:accuracy")],
)
@@ -2470,7 +2934,9 @@ async def test_a_run_without_tools_leaves_message_history_unchanged(
"""
transport = judge_run_transport()
accuracy_judge_variation(monkeypatch)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
seen: dict[str, str] = {}
async def handler(
@@ -2485,7 +2951,6 @@ async def handler(
return {"output": "generated"}
await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
@@ -2528,7 +2993,9 @@ async def fake_extract_variation(
"launchdarkly_ai_server.evaluations.runner.extract_variation",
fake_extract_variation,
)
- evals = init_evaluations(api_token="token", sdk_key="sdk-key", transport=transport)
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
async def handler(
config: dict[str, Any],
@@ -2545,11 +3012,15 @@ async def handler(
return {"output": "done"}
result = await evals.run(
- project_key="proj",
key="support-qa",
dataset="golden",
handler=handler,
- tools={"lookup_order": lambda args: "{{expected_output}} leaked?"},
+ tools=[
+ await evals.tools.get(
+ "lookup_order",
+ implementation=lambda args: "{{expected_output}} leaked?",
+ )
+ ],
generation={"provider": "OpenAI", "model": "gpt-4o"},
criteria=[Judge(key="$ld:ai:judge:accuracy")],
)
@@ -2564,7 +3035,7 @@ async def test_a_failed_row_keeps_the_calls_made_before_the_handler_raised() ->
from launchdarkly_ai_server.evaluations.runner import EvaluationsRunner
runner = EvaluationsRunner(
- LDApiClient(api_token="token", transport=failing_transport)
+ LDApiClient(api_key="token", transport=failing_transport)
)
async def handler(
@@ -2668,7 +3139,7 @@ def evaluation_post(transport: SequencedTransport) -> dict[str, Any]:
@pytest.mark.asyncio
async def test_run_seeds_generation_from_the_latest_ai_config_variation() -> None:
transport = SequencedTransport(fetched_run_responses(config_variation_page()))
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
seen_configs: list[dict[str, Any]] = []
async def handler(config: dict[str, Any], *args: object) -> dict[str, Any]:
@@ -2676,7 +3147,6 @@ async def handler(config: dict[str, Any], *args: object) -> dict[str, Any]:
return {"output": "generated"}
result = await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -2707,13 +3177,12 @@ async def handler(config: dict[str, Any], *args: object) -> dict[str, Any]:
@pytest.mark.asyncio
async def test_explicit_generation_overrides_the_fetched_variation() -> None:
transport = SequencedTransport(fetched_run_responses(config_variation_page()))
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -2747,11 +3216,10 @@ async def test_variation_tools_without_implementations_fail_before_mutating_requ
response(200, MODEL_CONFIG),
]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
with pytest.raises(EvaluationsError, match="'lookup_order'"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=successful_handler,
@@ -2790,7 +3258,7 @@ async def fake_extract_variation(
"launchdarkly_ai_server.evaluations.runner.extract_variation",
fake_extract_variation,
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
@@ -2798,7 +3266,6 @@ async def handler(*args: object) -> dict[str, Any]:
# Resolving the attached judge is what fails, so it was picked up as a criterion.
with pytest.raises(EvaluationsError, match="'security-judge'"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
@@ -2813,11 +3280,10 @@ async def test_unknown_variation_fails_before_any_records_are_created() -> None:
transport = SequencedTransport(
[response(404, {"code": "not_found", "message": "not found"})]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
with pytest.raises(EvaluationsError, match="was not found"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=successful_handler,
@@ -2846,11 +3312,10 @@ async def test_config_source_is_validated_before_network_io(
source: dict[str, AIConfig], message: str
) -> None:
transport = SequencedTransport([])
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
with pytest.raises(EvaluationsError, match=message):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=successful_handler,
@@ -2869,31 +3334,30 @@ async def test_tool_version_drift_from_the_variation_is_logged(
config_variation_page(tools=[{"key": "lookup_order", "version": 4}])
)
responses.insert(
- 2,
+ 0,
response(
200,
{"key": "lookup_order", "version": 7, "schema": {"type": "object"}},
),
)
transport = SequencedTransport(responses)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
async def handler(*args: object) -> dict[str, Any]:
return {"output": "generated"}
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=handler,
ai_config=AIConfig(key="support-agent", variation="control"),
- tools={"lookup_order": lookup_order},
+ tools=[await evals.tools.get("lookup_order", implementation=lookup_order)],
)
assert "pins tool 'lookup_order' at version 4" in caplog.text
- assert "latest version 7" in caplog.text
+ assert "the run uses version 7" in caplog.text
assert evaluation_post(transport)["tools"] == [
- {"key": "lookup_order", "version": 7}
+ {"key": "lookup_order", "version": 7, "source": "library"}
]
@@ -2905,11 +3369,10 @@ async def test_non_string_model_config_key_fails_loudly(
transport = SequencedTransport(
[response(200, config_variation_page(modelConfigKey=model_config_key))]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
with pytest.raises(EvaluationsError, match="non-string modelConfigKey"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=successful_handler,
@@ -2927,12 +3390,11 @@ async def test_variation_without_a_model_config_needs_an_explicit_provider(
transport = SequencedTransport(
[response(200, config_variation_page(modelConfigKey=model_config_key))]
)
- evals = init_evaluations(api_token="token", transport=transport)
+ evals = init_evaluations(project_key="proj", api_key="token", transport=transport)
# No model config is linked, so none is fetched and no provider is known.
with pytest.raises(EvaluationsError, match=r"generation\.provider is required"):
await evals.run(
- project_key="proj",
key="eval-key",
dataset="golden",
handler=successful_handler,
@@ -2965,3 +3427,208 @@ def test_ai_config_variation_from_api_layers_the_model_config() -> None:
unlinked = AIConfigVariation.from_api(latest)
assert "provider" not in unlinked.generation
assert unlinked.generation["parameters"] == {"temperature": 0.7}
+
+
+@pytest.mark.asyncio
+async def test_tools_get_pins_the_version_it_reads() -> None:
+ transport = SequencedTransport(
+ [
+ response(
+ 200,
+ {
+ "key": "lookup_order",
+ "version": 7,
+ "description": "Look up an order",
+ "schema": ORDER_SCHEMA,
+ },
+ )
+ ]
+ )
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ tool = await evals.tools.get("lookup_order", implementation=lookup_order)
+
+ assert tool.source == "library"
+ assert tool.version == 7
+ assert tool.description == "Look up an order"
+ assert tool.schema == ORDER_SCHEMA
+ assert tool.implementation is lookup_order
+ assert recorded_paths(transport) == [("GET", "projects/proj/ai-tools/lookup_order")]
+
+
+def test_a_constructed_tool_is_always_inline() -> None:
+ """``source`` and ``version`` are not constructor arguments."""
+ tool = EvalTool(
+ key="lookup_order", implementation=lookup_order, schema=ORDER_SCHEMA
+ )
+
+ assert tool.source == "inline"
+ assert tool.version is None
+ with pytest.raises(TypeError):
+ EvalTool( # type: ignore[call-arg]
+ key="lookup_order",
+ implementation=lookup_order,
+ schema=ORDER_SCHEMA,
+ source="library",
+ )
+
+
+def test_an_inline_tool_requires_a_schema() -> None:
+ with pytest.raises(TypeError, match="schema"):
+ EvalTool(key="lookup_order", implementation=lookup_order) # type: ignore[call-arg]
+
+ tool = EvalTool(key="lookup_order", implementation=lookup_order, schema={})
+ assert tool.schema == {}
+
+
+def test_a_tool_cannot_be_reassigned_to_a_library_tool() -> None:
+ tool = EvalTool(
+ key="lookup_order", implementation=lookup_order, schema=ORDER_SCHEMA
+ )
+
+ with pytest.raises(dataclasses.FrozenInstanceError):
+ tool.source = "library" # type: ignore[misc]
+ with pytest.raises(dataclasses.FrozenInstanceError):
+ tool.version = 99 # type: ignore[misc]
+ with pytest.raises(dataclasses.FrozenInstanceError):
+ tool.schema = {} # type: ignore[misc]
+
+
+@pytest.mark.asyncio
+async def test_a_library_tool_without_a_project_is_rejected() -> None:
+ forged = EvalTool._library(
+ "lookup_order",
+ lookup_order,
+ version=1,
+ schema={},
+ description="",
+ project_key="proj",
+ )
+ object.__setattr__(forged, "project_key", None)
+ transport = SequencedTransport([])
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ with pytest.raises(EvaluationsError, match="has no project"):
+ await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=successful_handler,
+ tools=[forged],
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ )
+
+ assert transport.requests == []
+
+
+@pytest.mark.asyncio
+async def test_tools_get_reads_in_a_worker_thread() -> None:
+ main_thread = threading.get_ident()
+ threads: list[int] = []
+
+ class RecordingTransport(SequencedTransport):
+ def __call__(self, *args: Any, **kwargs: Any) -> Any:
+ threads.append(threading.get_ident())
+ return super().__call__(*args, **kwargs)
+
+ transport = RecordingTransport(
+ [response(200, {"key": "lookup_order", "version": 7, "schema": {}})]
+ )
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ await evals.tools.get("lookup_order", implementation=lookup_order)
+
+ assert threads
+ assert main_thread not in threads
+
+
+@pytest.mark.asyncio
+async def test_run_reads_no_tool_from_the_api() -> None:
+ """``run`` resolves nothing: a library tool was already read by ``tools.get``."""
+ transport = SequencedTransport(
+ [
+ response(200, {"key": "lookup_order", "version": 7}),
+ *hosted_dataset_responses(),
+ ]
+ )
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+ library_tool = await evals.tools.get("lookup_order", implementation=lookup_order)
+ requests_before_run = len(transport.requests)
+
+ await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=successful_handler,
+ tools=[
+ library_tool,
+ EvalTool(
+ key="refund_order", implementation=refund_order, schema=ORDER_SCHEMA
+ ),
+ ],
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ )
+
+ run_paths = [path for _, path in recorded_paths(transport)[requests_before_run:]]
+ assert not any("ai-tools" in path for path in run_paths), run_paths
+
+
+@pytest.mark.asyncio
+async def test_an_empty_tools_list_runs_a_variation_with_no_tools() -> None:
+ """A caller who passes tools= replaces the variation's list."""
+ transport = SequencedTransport(
+ fetched_run_responses(
+ config_variation_page(tools=[{"key": "lookup_order", "version": 4}])
+ )
+ )
+ evals = init_evaluations(
+ project_key="proj", api_key="token", sdk_key="sdk-key", transport=transport
+ )
+
+ async def handler(*args: object) -> dict[str, Any]:
+ return {"output": "generated"}
+
+ await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=handler,
+ ai_config=AIConfig(key="support-agent", variation="control"),
+ tools=[],
+ )
+
+ assert "tools" not in evaluation_post(transport)
+
+
+@pytest.mark.asyncio
+async def test_a_tool_from_another_project_is_rejected() -> None:
+ transport = SequencedTransport(
+ [response(200, {"key": "lookup_order", "version": 4, "schema": {}})]
+ )
+ other = init_evaluations(
+ project_key="other-proj",
+ api_key="token",
+ sdk_key="sdk-key",
+ transport=transport,
+ )
+ foreign_tool = await other.tools.get("lookup_order", implementation=lookup_order)
+ evals = init_evaluations(
+ project_key="proj",
+ api_key="token",
+ sdk_key="sdk-key",
+ transport=SequencedTransport([]),
+ )
+
+ with pytest.raises(EvaluationsError, match="cannot run in project 'proj'"):
+ await evals.run(
+ key="eval-key",
+ dataset="golden",
+ handler=successful_handler,
+ tools=[foreign_tool],
+ generation={"provider": "OpenAI", "model": "gpt-4o"},
+ )