Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
48 changes: 48 additions & 0 deletions sdks/python/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -84,6 +84,54 @@ async with agent_control.AgentControlClient() as client:
The existing `evaluate_controls` helper remains available for callers that
prefer its field-based convenience arguments.

## Building trace/session steps

Trace- and session-level controls evaluate the aggregate of several
already-executed child steps, so `Step.children` has to be populated by the
caller. Building that tree by hand means hand-rolling a side-channel record
for every leaf call and converting it into `Step` objects afterward:

```python
# Before: a hand-built dict record next to the real return value
def execute_policy_search(query):
docs = search_policy_documents(query)
record = {"type": "retriever", "query": query, "docs": docs}
return StepExecution(value=docs, record=record)

spans = []
result = execute_policy_search(query)
spans.append(result.record)
...
trace_step = build_agent_control_step({"type": "trace", "spans": spans, ...})
```

`agent_control.record_step()` builds the same tree incrementally, in `Step`
vocabulary, with no intermediate dict and no converter:

```python
with agent_control.record_step("trace", "banking_trace", input={"request": req}) as trace:
with trace.child("retriever", "policy_lookup", input=query) as span:
span.output = search_policy_documents(query)

account = trace.call(lookup_account, account_id="acct-1001", step_type="tool")
plan = trace.call(run_banking_model, req, step_type="llm", tools=TOOL_DEFINITIONS)
trace.output = {"status": "planned", "message": plan["content"]}

result = await trace.evaluate(stage="post")
```

`trace.call(...)`/`await trace.acall(...)` run the function and record it as
a child using the same capture logic as `@control()` (input from bound
arguments, output from the return value), then return the real result -
removing the need for a separate `StepExecution`-style wrapper. A failed
call is still recorded (with the error in `context`) before the exception is
re-raised. `trace.child(...)` nests another recorder the same way, so
sessions nest traces with `session.child("trace", ...)`.

`trace.build()` produces the frozen `Step`; `trace.evaluate(...)` builds it
and evaluates it in one call via `evaluate_step()` - the same function
`evaluate_controls()` uses internally once its `Step` is built.

## Sharing an OpenTelemetry provider with Google ADK

When Google ADK and Agent Control should export through the same OpenTelemetry
Expand Down
7 changes: 6 additions & 1 deletion sdks/python/src/agent_control/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -92,7 +92,7 @@ async def handle_input(user_message: str) -> str:
)
from .client import AgentControlClient
from .control_decorators import ControlSteerError, ControlViolationError, control
from .evaluation import check_evaluation_with_local, evaluate_controls
from .evaluation import check_evaluation_with_local, evaluate_controls, evaluate_step
from .observability import (
LogConfig,
add_event,
Expand All @@ -116,6 +116,7 @@ async def handle_input(user_message: str) -> str:
)
from .otel_sink import control_event_to_otel_span
from .runtime_auth import validate_http_field_name
from .step_recorder import StepRecorder, record_step
from .tracing import (
get_current_span_id,
get_current_trace_id,
Expand Down Expand Up @@ -1619,6 +1620,10 @@ async def main():
# Local evaluation
"check_evaluation_with_local",
"evaluate_controls",
"evaluate_step",
# Step recorder (incremental trace/session Step tree builder)
"record_step",
"StepRecorder",
# Tracing
"get_trace_and_span_ids",
"get_current_trace_id",
Expand Down
90 changes: 63 additions & 27 deletions sdks/python/src/agent_control/evaluation.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
"""Evaluation check operations for Agent Control SDK."""

from collections.abc import Awaitable, Callable
from collections.abc import Awaitable, Callable, Mapping, Sequence
from dataclasses import dataclass
from inspect import iscoroutinefunction
from typing import Any, Literal, cast
Expand Down Expand Up @@ -515,6 +515,58 @@ def _with_parse_errors(result: EvaluationResult) -> EvaluationResult:
return _with_parse_errors(EvaluationResult(is_safe=True, confidence=1.0))


async def evaluate_step(
step: Step,
*,
agent_name: str,
stage: Literal["pre", "post"] = "pre",
target_type: str | None = None,
target_id: str | None = None,
trace_id: str | None = None,
span_id: str | None = None,
) -> EvaluationResult:
"""Evaluate controls for an already-built ``Step``.

This is the shared tail of :func:`evaluate_controls`: resolve the
session target, open a client, and run local/server evaluation. Use
it directly when the ``Step`` - including any ``children`` - was
already assembled, e.g. via :class:`~agent_control.step_recorder.StepRecorder`.

When ``target_type`` and ``target_id`` are both supplied, the request
is target-bearing: the server merges target bindings into the
effective control set. If they are omitted, the SDK falls back to the
target context fixed at ``init()`` time when present. A per-call
override that disagrees with the session target is rejected because
the cached controls were fetched for the session target and would
otherwise drive stale local-first evaluation.
"""
if state.server_url is None:
raise RuntimeError("Server URL not configured. Call agent_control.init() first.")

target_type, target_id = _resolve_session_target(target_type, target_id)
resolved_controls = state.server_controls or []

async with AgentControlClient(
base_url=state.server_url,
api_key=state.api_key,
api_key_header=state.api_key_header,
runtime_token_header=state.runtime_token_header,
runtime_token_cache=state.runtime_token_cache,
) as client:
return await check_evaluation_with_local(
client=client,
agent_name=agent_name,
step=step,
stage=stage,
controls=resolved_controls,
target_type=target_type,
target_id=target_id,
trace_id=trace_id,
span_id=span_id,
event_agent_name=agent_name,
)


async def evaluate_controls(
step_name: str,
*,
Expand All @@ -523,7 +575,7 @@ async def evaluate_controls(
context: dict[str, Any] | None = None,
tools: list[dict[str, JSONValue]] | None = None,
ground_truth: JSONValue | None = None,
children: list[Step] | None = None,
children: Sequence[Step | Mapping[str, Any]] | None = None,
step_type: str = "llm",
stage: Literal["pre", "post"] = "pre",
agent_name: str,
Expand All @@ -544,11 +596,6 @@ async def evaluate_controls(
"""
step_type = ensure_step_type(step_type)

if state.server_url is None:
raise RuntimeError("Server URL not configured. Call agent_control.init() first.")

target_type, target_id = _resolve_session_target(target_type, target_id)

default_value = {} if step_type == "tool" else ""
step_dict: dict[str, Any] = {
"type": step_type,
Expand All @@ -566,24 +613,13 @@ async def evaluate_controls(
step_dict["children"] = children

step_obj = Step(**step_dict) # type: ignore[arg-type]
resolved_controls = state.server_controls or []

async with AgentControlClient(
base_url=state.server_url,
api_key=state.api_key,
api_key_header=state.api_key_header,
runtime_token_header=state.runtime_token_header,
runtime_token_cache=state.runtime_token_cache,
) as client:
return await check_evaluation_with_local(
client=client,
agent_name=agent_name,
step=step_obj,
stage=stage,
controls=resolved_controls,
target_type=target_type,
target_id=target_id,
trace_id=trace_id,
span_id=span_id,
event_agent_name=agent_name,
)
return await evaluate_step(
step_obj,
agent_name=agent_name,
stage=stage,
target_type=target_type,
target_id=target_id,
trace_id=trace_id,
span_id=span_id,
)
Loading
Loading