From 50640d63658f443a3944ab0d1c410730a2be4943 Mon Sep 17 00:00:00 2001 From: Wrisa Date: Fri, 2 Oct 2026 12:15:11 -0700 Subject: [PATCH 01/10] feat(evaluators): Add evaluator for LLM --- evaluators/contrib/README.md | 2 +- evaluators/contrib/galileo/pyproject.toml | 3 +- .../__init__.py | 16 + .../llm/__init__.py | 30 + .../llm/client.py | 613 ++++++++++++++++++ .../llm/config.py | 124 ++++ .../llm/evaluator.py | 387 +++++++++++ sdks/python/ARCHITECTURE.md | 6 +- .../src/agent_control/control_decorators.py | 2 +- .../src/agent_control/evaluators/__init__.py | 25 +- .../tests/test_evaluators_optional_imports.py | 36 +- 11 files changed, 1235 insertions(+), 9 deletions(-) create mode 100644 evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py create mode 100644 evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py create mode 100644 evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py create mode 100644 evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py diff --git a/evaluators/contrib/README.md b/evaluators/contrib/README.md index 91beb9b9..4b852070 100644 --- a/evaluators/contrib/README.md +++ b/evaluators/contrib/README.md @@ -2,7 +2,7 @@ Contributed evaluators and templates for extending Agent Control. -- `galileo/` — Luna evaluator integration +- `galileo/` — Galileo evaluator integrations (Luna, LLM) - `template/` — Starter template for adding new evaluators Full guide: https://docs.agentcontrol.dev/concepts/evaluators/custom-evaluators diff --git a/evaluators/contrib/galileo/pyproject.toml b/evaluators/contrib/galileo/pyproject.toml index 7b212b96..7909b197 100644 --- a/evaluators/contrib/galileo/pyproject.toml +++ b/evaluators/contrib/galileo/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "agent-control-evaluator-galileo" version = "8.11.0" -description = "Galileo Luna evaluator for agent-control" +description = "Galileo evaluators (Luna, LLM-as-judge) for agent-control" readme = "README.md" requires-python = ">=3.12" license = { text = "Apache-2.0" } @@ -25,6 +25,7 @@ dev = [ [project.entry-points."agent_control.evaluators"] "galileo.luna" = "agent_control_evaluator_galileo.luna:LunaEvaluator" +"galileo.llm" = "agent_control_evaluator_galileo.llm:LlmEvaluator" [build-system] requires = ["hatchling"] diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py index 0183ad97..71e8e730 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py @@ -4,6 +4,7 @@ Available evaluators: - galileo.luna: Galileo Luna direct scorer evaluation + - galileo.llm: Galileo LLM-as-judge direct scorer evaluation Installation: pip install agent-control-evaluator-galileo @@ -30,6 +31,13 @@ ScorerInvokeRequest, ScorerInvokeResponse, ) +from agent_control_evaluator_galileo.llm import ( + LLM_AVAILABLE, + GalileoLLMClient, + LlmEvaluator, + LlmEvaluatorConfig, + LlmOperator, +) from agent_control_evaluator_galileo.records import ( GalileoRecord, GalileoRecordNormalizer, @@ -41,6 +49,7 @@ ) __all__ = [ + # luna "GalileoLunaClient", "ScorerInvokeRequest", "ScorerInvokeConfig", @@ -50,6 +59,13 @@ "LunaEvaluatorConfig", "LunaOperator", "LUNA_AVAILABLE", + # llm + "GalileoLLMClient", + "LlmEvaluator", + "LlmEvaluatorConfig", + "LlmOperator", + "LLM_AVAILABLE", + # records "GalileoRecord", "GalileoRecordNormalizer", "RecordFactoryError", diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py new file mode 100644 index 00000000..518febbe --- /dev/null +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py @@ -0,0 +1,30 @@ +"""Galileo LLM-as-judge direct scorer evaluator.""" + +from agent_control_evaluator_galileo.llm.client import ( + GalileoLLMClient, + GalileoExecutionContext, + ScorerInvokeInputs, + ScorerInvokeRecord, + ScorerInvokeRequest, + ScorerInvokeResponse, +) +from agent_control_evaluator_galileo.llm.config import ( + LlmEvaluatorConfig, + LlmOperator, + ScorerInvokeConfig, +) +from agent_control_evaluator_galileo.llm.evaluator import LLM_AVAILABLE, LlmEvaluator + +__all__ = [ + "GalileoLLMClient", + "GalileoExecutionContext", + "ScorerInvokeInputs", + "ScorerInvokeConfig", + "ScorerInvokeRecord", + "ScorerInvokeRequest", + "ScorerInvokeResponse", + "LlmEvaluatorConfig", + "LlmOperator", + "LlmEvaluator", + "LLM_AVAILABLE", +] diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py new file mode 100644 index 00000000..47b21e6b --- /dev/null +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py @@ -0,0 +1,613 @@ +"""Direct HTTP client for Galileo LLM-as-judge scorer invocation.""" + +from __future__ import annotations + +import logging +import os +import ssl +from asyncio import Lock +from base64 import urlsafe_b64encode +from hashlib import sha256 +from hmac import new as hmac_new +from json import dumps +from time import time +from typing import Literal, cast, get_args +from urllib.parse import urlsplit + +import httpx +from agent_control_models import JSONObject, JSONValue, Step +from pydantic import BaseModel, Field, PrivateAttr, model_validator + +from .config import ScorerInvokeConfig + +logger = logging.getLogger(__name__) + +DEFAULT_TIMEOUT_SECS = 10.0 +SERVER_TIMEOUT_RATIO = 0.8 +DEFAULT_INTERNAL_TOKEN_TTL_SECS = 3600 +DEFAULT_LLM_SCORER_INVOKE_PATH = "/api/v1/scorers/invoke" +LLM_INVOKE_URL_ENV = "GALILEO_LUNA_INVOKE_URL" +LLM_INVOKE_CA_FILE_ENV = "GALILEO_LUNA_INVOKE_CA_FILE" +AUTH_UPSTREAM_CA_FILE_ENV = "AGENT_CONTROL_AUTH_UPSTREAM_CA_FILE" + +# Headers that must never be forwarded to the scorer invoke endpoint (checked case-insensitively). +_BLOCKED_REQUEST_HEADERS = frozenset({"galileo-api-key"}) + +# Keep pooled-connection reuse shorter than typical server keepalive/worker +# recycle windows so requests do not pick up sockets the server already closed. +DEFAULT_KEEPALIVE_EXPIRY_SECS = 1.0 +DEFAULT_MAX_CONNECTIONS = 100 +DEFAULT_MAX_KEEPALIVE_CONNECTIONS = 20 +DEFAULT_CLIENT_POOL_SIZE = 1 +LLM_KEEPALIVE_EXPIRY_ENV = "GALILEO_LUNA_KEEPALIVE_EXPIRY_SECONDS" +LLM_MAX_CONNECTIONS_ENV = "GALILEO_LUNA_MAX_CONNECTIONS" +LLM_MAX_KEEPALIVE_CONNECTIONS_ENV = "GALILEO_LUNA_MAX_KEEPALIVE_CONNECTIONS" +LLM_CLIENT_POOL_SIZE_ENV = "GALILEO_LUNA_CLIENT_POOL_SIZE" + +# These values mirror Orbit's StepType discriminator. Agent Control's generic +# Step remains extensible; only the Galileo transport boundary is constrained. +ScorerInvokeRecordType = Literal[ + "llm", + "retriever", + "tool", + "workflow", + "agent", + "control", + "trace", + "session", +] +SUPPORTED_SCORER_INVOKE_RECORD_TYPES = frozenset(get_args(ScorerInvokeRecordType)) + + +def _b64url(data: bytes) -> str: + return urlsafe_b64encode(data).rstrip(b"=").decode("ascii") + + +def _internal_auth_token( + api_secret: str, + ttl_seconds: int = DEFAULT_INTERNAL_TOKEN_TTL_SECS, +) -> str: + """Create the internal JWT expected by LLM scorer invoke routes.""" + now = int(time()) + header = {"alg": "HS256", "typ": "JWT"} + payload = { + "internal": True, + "scope": "scorers.invoke", + "iat": now, + "exp": now + ttl_seconds, + } + signing_input = ".".join( + [ + _b64url(dumps(header, separators=(",", ":")).encode("utf-8")), + _b64url(dumps(payload, separators=(",", ":")).encode("utf-8")), + ] + ) + signature = hmac_new(api_secret.encode("utf-8"), signing_input.encode("ascii"), sha256).digest() + return f"{signing_input}.{_b64url(signature)}" + + +def _normalize_llm_invoke_url(raw_url: str) -> str: + """Use full invoke URLs as-is and append the default path to bare service roots.""" + url = raw_url.strip().rstrip("/") + parsed = urlsplit(url) + if parsed.path not in ("", "/") or parsed.query or parsed.fragment: + return url + return f"{url}{DEFAULT_LLM_SCORER_INVOKE_PATH}" + + +def _load_float_env(env_name: str, default: float) -> float: + raw = os.getenv(env_name) + if raw is None or raw.strip() == "": + return default + try: + return float(raw) + except ValueError as exc: + raise ValueError(f"{env_name}={raw!r} is not a number.") from exc + + +def _load_int_env(env_name: str, default: int) -> int: + raw = os.getenv(env_name) + if raw is None or raw.strip() == "": + return default + try: + return int(raw) + except ValueError as exc: + raise ValueError(f"{env_name}={raw!r} is not an integer.") from exc + + +def _validate_connection_config( + *, + keepalive_expiry_seconds: float, + max_connections: int, + max_keepalive_connections: int, + client_pool_size: int, +) -> None: + if keepalive_expiry_seconds < 0: + raise ValueError( + f"{LLM_KEEPALIVE_EXPIRY_ENV}={keepalive_expiry_seconds} " + "must be greater than or equal to 0." + ) + if max_connections <= 0: + raise ValueError(f"{LLM_MAX_CONNECTIONS_ENV}={max_connections} must be greater than 0.") + if max_keepalive_connections < 0: + raise ValueError( + f"{LLM_MAX_KEEPALIVE_CONNECTIONS_ENV}={max_keepalive_connections} " + "must be greater than or equal to 0." + ) + if max_keepalive_connections > max_connections: + raise ValueError( + f"{LLM_MAX_KEEPALIVE_CONNECTIONS_ENV}={max_keepalive_connections} " + f"must be less than or equal to {LLM_MAX_CONNECTIONS_ENV}={max_connections}." + ) + if client_pool_size <= 0: + raise ValueError(f"{LLM_CLIENT_POOL_SIZE_ENV}={client_pool_size} must be greater than 0.") + + +def _as_float_or_none(value: JSONValue) -> float | None: + if isinstance(value, bool) or value is None: + return None + if isinstance(value, (int, float)): + return float(value) + if isinstance(value, str): + try: + return float(value) + except ValueError: + return None + return None + + +def _has_value(value: JSONValue) -> bool: + if value is None: + return False + if isinstance(value, str): + return value.strip() != "" + if isinstance(value, (list, dict)): + return len(value) > 0 + return True + + +def _effective_scorer_timeout( + config: ScorerInvokeConfig, + *, + http_timeout_seconds: float, +) -> ScorerInvokeConfig: + """Resolve an Orbit execution timeout that expires before the HTTP request. + + The server execution budget defaults to 80% of the caller's HTTP deadline, + leaving time for Orbit to serialize and return the result. Explicit caller + overrides are preserved only when they maintain the same ordering. + + Args: + config: Caller-provided scorer-invoke configuration. + http_timeout_seconds: Agent Control's HTTP request deadline in seconds. + + Returns: + A scorer-invoke configuration with an effective execution timeout. + + Raises: + ValueError: If either deadline is invalid or the explicit server timeout + is not shorter than the HTTP deadline. + """ + if http_timeout_seconds <= 0: + raise ValueError("HTTP timeout must be greater than 0 seconds.") + + server_timeout = config.request_timeout_seconds + if server_timeout is None: + server_timeout = http_timeout_seconds * SERVER_TIMEOUT_RATIO + elif server_timeout >= http_timeout_seconds: + raise ValueError( + "config.request_timeout_seconds must be shorter than the HTTP " + f"timeout ({http_timeout_seconds:g} seconds)." + ) + + return config.model_copy(update={"request_timeout_seconds": server_timeout}) + + +class ScorerInvokeInputs(BaseModel): + """Input values sent to the LLM scorer invoke endpoint.""" + + query: JSONValue = "" + response: JSONValue = "" + ground_truth: JSONValue = None + tools: list[JSONObject] | None = None + + +class ScorerInvokeRecord(BaseModel): + """Caller-controlled subset of Orbit's partial runtime-record contract. + + Identity, ownership, persistence, and execution IDs are intentionally not + represented here. Orbit hydrates those fields from trusted server context. + """ + + type: ScorerInvokeRecordType + name: str | None = None + input: JSONValue = None + output: JSONValue = None + context: JSONObject | None = None + tools: list[JSONObject] | None = None + dataset_output: JSONValue = None + + +class GalileoExecutionContext(BaseModel): + """Authenticated identity fields required by Galileo scorer invocation.""" + + organization_id: str = Field(min_length=1) + user_id: str = Field(min_length=1) + project_id: str = Field(min_length=1) + run_id: str = Field(min_length=1) + + +class ScorerInvokeRequest(BaseModel): + """Request payload for LLM scorer invocation. + + Attributes: + scorer_id: Required scorer identifier. + scorer_version_id: Deprecated optional compatibility identifier. Orbit + currently invokes the scorer's current default version. + scorer_label: Optional display/metadata label. + inputs: Selected scorer input values. + record: Optional Orbit-compatible structured runtime record. + execution_context: Optional authenticated execution context. + config: Scorer-specific configuration, always emitted. + """ + + scorer_id: str = Field(min_length=1) + scorer_version_id: str | None = Field(default=None, min_length=1) + scorer_label: str | None = Field(default=None, min_length=1) + inputs: ScorerInvokeInputs + record: ScorerInvokeRecord | None = None + execution_context: GalileoExecutionContext | None = None + config: ScorerInvokeConfig = Field(default_factory=ScorerInvokeConfig) + + @model_validator(mode="after") + def ensure_required_values(self) -> ScorerInvokeRequest: + if not (_has_value(self.inputs.query) or _has_value(self.inputs.response)): + raise ValueError("Either inputs.query or inputs.response must be set.") + return self + + def to_dict(self) -> JSONObject: + """Convert to the LLM scorer invoke request shape.""" + return self.model_dump(mode="json", exclude_none=True) + + +def _orbit_record_from_step( + step: Step | None, + *, + selected_input: JSONValue, + selected_output: JSONValue, +) -> ScorerInvokeRecord | None: + """Translate a generic Agent Control step into Orbit's record contract. + + Selector-selected values remain the primary scorer input. When a selector + supplies one side, that value is written to both the legacy and structured + representations so Orbit's conflict validation cannot observe two meanings. + The complete step supplies the unselected side and additional record context. + + Unknown Agent Control step types intentionally fall back to the legacy + ``inputs`` contract. This keeps the open-source Step model extensible without + sending an invalid discriminator to Orbit. + + Args: + step: Complete Agent Control step, when contextual evaluation is used. + selected_input: Selector-selected value sent as ``inputs.query``. + selected_output: Selector-selected value sent as ``inputs.response``. + + Returns: + An Orbit-compatible record, or ``None`` for absent/unsupported steps. + """ + if step is None or step.type not in SUPPORTED_SCORER_INVOKE_RECORD_TYPES: + return None + + # The membership check above narrows the runtime value to Orbit's known + # discriminator set, but static type checkers cannot infer that relationship. + record_type = cast(ScorerInvokeRecordType, step.type) + return ScorerInvokeRecord( + type=record_type, + name=step.name, + input=selected_input if selected_input is not None else step.input, + output=selected_output if selected_output is not None else step.output, + context=step.context, + tools=step.tools, + dataset_output=step.ground_truth, + ) + + +class ScorerInvokeResponse(BaseModel): + """Response from LLM scorer invocation. + + Attributes: + scorer_label: Echoed scorer label, when returned. + score: Raw scorer value. + status: Invocation status. + execution_time: Execution time in seconds, when returned. + error_message: Error detail for non-success statuses. + """ + + scorer_label: str | None = None + score: JSONValue + status: str = "unknown" + execution_time: float | None = None + error_message: str | None = None + _raw_response: JSONObject = PrivateAttr(default_factory=dict) + + @property + def raw_response(self) -> JSONObject: + return self._raw_response + + @classmethod + def from_dict(cls, data: JSONObject) -> ScorerInvokeResponse: + """Create a response model from the LLM scorer invoke JSON object.""" + response = cls.model_validate( + data | {"execution_time": _as_float_or_none(data.get("execution_time"))} + ) + response._raw_response = data + return response + + +class GalileoLLMClient: + """Thin HTTP client for Galileo LLM-as-judge scorer invocation. + + Environment Variables: + GALILEO_API_SECRET_KEY or GALILEO_API_SECRET: JWT signing secret for internal auth. + GALILEO_LUNA_INVOKE_URL: LLM scorer invoke URL or service root (required). + GALILEO_LUNA_INVOKE_CA_FILE: CA bundle used to verify invoke TLS. + AGENT_CONTROL_AUTH_UPSTREAM_CA_FILE: Shared internal CA fallback. + GALILEO_LUNA_KEEPALIVE_EXPIRY_SECONDS: HTTP pooled connection expiry. + GALILEO_LUNA_MAX_CONNECTIONS: Maximum outbound HTTP connections. + GALILEO_LUNA_MAX_KEEPALIVE_CONNECTIONS: Maximum idle pooled HTTP connections. + GALILEO_LUNA_CLIENT_POOL_SIZE: Number of outbound HTTP clients to rotate across. + """ + + def __init__( + self, + api_secret: str | None = None, + llm_invoke_url: str | None = None, + llm_invoke_ca_file: str | None = None, + ) -> None: + """Initialize the Galileo LLM client. + + Args: + api_secret: Internal JWT signing secret. If not provided, reads from + GALILEO_API_SECRET_KEY or GALILEO_API_SECRET. + llm_invoke_url: LLM scorer invoke URL or service root. If not provided, + reads from GALILEO_LUNA_INVOKE_URL. + llm_invoke_ca_file: Optional CA bundle used to verify invoke TLS. If not + provided, reads from GALILEO_LUNA_INVOKE_CA_FILE, then + AGENT_CONTROL_AUTH_UPSTREAM_CA_FILE. + + Raises: + ValueError: If the API secret, invoke URL, CA bundle, or connection + tuning configuration is invalid. + """ + resolved_api_secret = ( + api_secret or os.getenv("GALILEO_API_SECRET_KEY") or os.getenv("GALILEO_API_SECRET") + ) + if not resolved_api_secret: + raise ValueError( + "GALILEO_API_SECRET_KEY or GALILEO_API_SECRET is required for LLM " + "scorer invocation. Set one as an environment variable or pass it " + "to the constructor." + ) + + resolved_llm_invoke_url = llm_invoke_url or os.getenv(LLM_INVOKE_URL_ENV) + if resolved_llm_invoke_url is None or resolved_llm_invoke_url.strip() == "": + raise ValueError( + "GALILEO_LUNA_INVOKE_URL is required for LLM scorer invocation. " + "Set it as an environment variable or pass it to the constructor." + ) + + self.api_secret = resolved_api_secret + self.llm_invoke_url = _normalize_llm_invoke_url(resolved_llm_invoke_url) + self.llm_invoke_ca_file = ( + llm_invoke_ca_file + or os.getenv(LLM_INVOKE_CA_FILE_ENV) + or os.getenv(AUTH_UPSTREAM_CA_FILE_ENV) + or "" + ).strip() or None + self._ssl_context = self._load_ssl_context(self.llm_invoke_ca_file) + self.keepalive_expiry_seconds = _load_float_env( + LLM_KEEPALIVE_EXPIRY_ENV, DEFAULT_KEEPALIVE_EXPIRY_SECS + ) + self.max_connections = _load_int_env(LLM_MAX_CONNECTIONS_ENV, DEFAULT_MAX_CONNECTIONS) + self.max_keepalive_connections = _load_int_env( + LLM_MAX_KEEPALIVE_CONNECTIONS_ENV, DEFAULT_MAX_KEEPALIVE_CONNECTIONS + ) + self.client_pool_size = _load_int_env(LLM_CLIENT_POOL_SIZE_ENV, DEFAULT_CLIENT_POOL_SIZE) + _validate_connection_config( + keepalive_expiry_seconds=self.keepalive_expiry_seconds, + max_connections=self.max_connections, + max_keepalive_connections=self.max_keepalive_connections, + client_pool_size=self.client_pool_size, + ) + self._client: httpx.AsyncClient | None = None + self._clients: list[httpx.AsyncClient] = [] + self._next_client_index = 0 + self._client_lock = Lock() + + @staticmethod + def _load_ssl_context(ca_file: str | None) -> ssl.SSLContext | None: + """Build a TLS verification context from a CA bundle path, if configured.""" + if ca_file is None: + return None + try: + return ssl.create_default_context(cafile=ca_file) + except (OSError, ssl.SSLError) as exc: + raise ValueError(f"Failed to load CA bundle from {ca_file!r}: {exc}") from exc + + def _create_client(self) -> httpx.AsyncClient: + """Create an HTTP client with the configured TLS and connection limits.""" + verify: ssl.SSLContext | bool = self._ssl_context if self._ssl_context is not None else True + return httpx.AsyncClient( + headers={"Content-Type": "application/json"}, + timeout=httpx.Timeout(DEFAULT_TIMEOUT_SECS), + limits=httpx.Limits( + max_connections=self.max_connections, + max_keepalive_connections=self.max_keepalive_connections, + keepalive_expiry=self.keepalive_expiry_seconds, + ), + verify=verify, + ) + + def _select_pooled_client(self) -> httpx.AsyncClient: + """Select the next pooled client while holding the client state lock.""" + client = self._clients[self._next_client_index % len(self._clients)] + self._next_client_index = (self._next_client_index + 1) % len(self._clients) + return client + + async def _get_client(self) -> httpx.AsyncClient: + """Get or create the next HTTP client.""" + async with self._client_lock: + self._clients = [client for client in self._clients if not client.is_closed] + + if self.client_pool_size == 1: + if self._client is not None and not self._client.is_closed: + return self._client + self._client = self._clients[0] if self._clients else self._create_client() + self._clients = [self._client] + return self._client + + self._client = None + while len(self._clients) < self.client_pool_size: + self._clients.append(self._create_client()) + + return self._select_pooled_client() + + def _endpoint_and_auth_header(self) -> tuple[str, str]: + token = _internal_auth_token(self.api_secret) + return self.llm_invoke_url, f"Bearer {token}" + + async def invoke( + self, + *, + scorer_id: str, + scorer_version_id: str | None = None, + scorer_label: str | None = None, + input: JSONValue = None, + output: JSONValue = None, + step: Step | None = None, + execution_context: GalileoExecutionContext | None = None, + config: ScorerInvokeConfig | JSONObject | None = None, + timeout: float = DEFAULT_TIMEOUT_SECS, + headers: dict[str, str] | None = None, + ) -> ScorerInvokeResponse: + """Invoke a Galileo LLM-as-judge scorer. + + Args: + scorer_id: Required scorer identifier. + scorer_version_id: Deprecated optional compatibility identifier. Orbit + currently invokes the scorer's current default version. + scorer_label: Optional display/metadata label. + input: Optional user/system prompt text. + output: Optional model response text. + step: Optional complete runtime step used for structured dual-write. + execution_context: Optional authenticated Galileo scorer context. + config: Optional Orbit-supported scorer invocation configuration. + timeout: Request timeout in seconds. + headers: Additional request headers. + + Returns: + Parsed scorer invocation response. + + Raises: + ValueError: If neither input nor output is provided, config contains + a field Orbit does not support, or the timeout ordering is invalid. + RuntimeError: If the API response is not a JSON object. + httpx.HTTPStatusError: If the invoke endpoint returns an error status code. + httpx.RequestError: If the request fails before a response is received. + """ + if not (_has_value(input) or _has_value(output)): + raise ValueError("At least one of input or output must be provided.") + + # Accept dictionaries for source compatibility with the original client, + # but validate them against Orbit's authoritative allowlist locally. + invoke_config = ( + ScorerInvokeConfig.model_validate(config) + if config is not None + else ScorerInvokeConfig() + ) + invoke_config = _effective_scorer_timeout( + invoke_config, + http_timeout_seconds=timeout, + ) + request_body = ScorerInvokeRequest( + scorer_id=scorer_id, + scorer_version_id=scorer_version_id, + scorer_label=scorer_label, + inputs=ScorerInvokeInputs( + query="" if input is None else input, + response="" if output is None else output, + ground_truth=step.ground_truth if step is not None else None, + tools=step.tools if step is not None else None, + ), + record=_orbit_record_from_step( + step, + selected_input=input, + selected_output=output, + ), + execution_context=execution_context, + config=invoke_config, + ).to_dict() + + endpoint, auth_header = self._endpoint_and_auth_header() + request_headers = { + k: v for k, v in (headers or {}).items() if k.lower() not in _BLOCKED_REQUEST_HEADERS + } + request_headers["Authorization"] = auth_header + + logger.debug("[GalileoLLMClient] POST %s", endpoint) + logger.debug("[GalileoLLMClient] Request body: %s", request_body) + + try: + client = await self._get_client() + response = await client.post( + endpoint, + json=request_body, + headers=request_headers, + timeout=timeout, + ) + response.raise_for_status() + response_data = response.json() + if not isinstance(response_data, dict): + raise RuntimeError("Invalid response payload: not a JSON object") + + parsed = ScorerInvokeResponse.from_dict(response_data) + logger.debug("[GalileoLLMClient] Response: %s", parsed.raw_response) + return parsed + except httpx.HTTPStatusError as exc: + logger.error( + "[GalileoLLMClient] API error: %s - %s", + exc.response.status_code, + exc.response.text, + ) + raise + except httpx.RequestError as exc: + logger.error("[GalileoLLMClient] Request failed: %s", exc) + raise + + async def close(self) -> None: + """Close HTTP clients and release resources.""" + async with self._client_lock: + clients: list[httpx.AsyncClient] = [] + seen_client_ids: set[int] = set() + if self._client is not None: + clients.append(self._client) + seen_client_ids.add(id(self._client)) + self._client = None + for client in self._clients: + if id(client) not in seen_client_ids: + clients.append(client) + seen_client_ids.add(id(client)) + self._clients = [] + self._next_client_index = 0 + + for client in clients: + if not client.is_closed: + await client.aclose() + + async def __aenter__(self) -> GalileoLLMClient: + """Async context manager entry.""" + return self + + async def __aexit__(self, exc_type: object, exc_val: object, exc_tb: object) -> None: + """Async context manager exit.""" + await self.close() diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py new file mode 100644 index 00000000..1b207f3a --- /dev/null +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py @@ -0,0 +1,124 @@ +"""Configuration model for direct Galileo LLM-as-judge scorer evaluation.""" + +from __future__ import annotations + +from typing import Literal + +from agent_control_evaluators import EvaluatorConfig +from agent_control_models import JSONValue +from pydantic import BaseModel, ConfigDict, Field, model_validator + +LlmOperator = Literal["gt", "gte", "lt", "lte", "eq", "ne", "contains", "any"] +LlmPayloadField = Literal["input", "output"] + +_NUMERIC_OPERATORS = frozenset({"gt", "gte", "lt", "lte"}) + + +class ScorerInvokeConfig(BaseModel): + """Orbit-supported overrides for a synchronous scorer invocation. + + Orbit owns the Galileo scorer-invoke wire contract. Keeping this model + strict makes an unsupported option fail locally instead of producing a + less actionable HTTP 422 response from Runners. + + Attributes: + threshold: Legacy threshold accepted by Orbit. Agent Control still + applies its evaluator threshold locally. + score_threshold: Legacy score threshold accepted by Orbit. + request_timeout_seconds: Optional upper bound for scorer execution in + Orbit. The Agent Control HTTP and evaluator deadlines must remain + longer than this value. + """ + + model_config = ConfigDict(extra="forbid") + + threshold: float | None = None + score_threshold: float | None = None + request_timeout_seconds: float | None = Field(default=None, gt=0) + + +def coerce_number(value: JSONValue) -> float | None: + """Return a numeric value for JSON scalars that can be compared numerically.""" + if isinstance(value, bool) or value is None: + return None + if isinstance(value, (int, float)): + return float(value) + if isinstance(value, str): + try: + return float(value) + except ValueError: + return None + return None + + +class LlmEvaluatorConfig(EvaluatorConfig): + """Configuration for direct LLM-as-judge scorer evaluation. + + Attributes: + scorer_id: Required scorer identifier for LLM scorer invocation. + scorer_version_id: Deprecated optional compatibility identifier. Orbit + currently invokes the scorer's current default version. + scorer_label: Optional display/metadata label. + threshold: Local threshold used by the evaluator for comparison. + operator: Local comparison operator. Numeric operators use threshold as a number. + scorer_config: Optional Orbit-supported scorer invocation config sent + as ``config``. + payload_field: Explicit scorer input side for scalar selected data. + timeout_ms: Request timeout in milliseconds. + """ + + scorer_id: str = Field( + min_length=1, + description="Required scorer identifier for LLM scorer invocation.", + ) + scorer_version_id: str | None = Field( + default=None, + min_length=1, + description=( + "Deprecated optional compatibility identifier. Orbit currently invokes " + "the scorer's current default version." + ), + ) + scorer_label: str | None = Field( + default=None, + min_length=1, + description="Optional display/metadata label.", + ) + threshold: JSONValue = Field( + default=0.5, + description="Local threshold used to decide whether the control matches.", + ) + operator: LlmOperator = Field( + default="gte", + description="Local comparison operator applied to the raw LLM scorer score.", + ) + scorer_config: ScorerInvokeConfig | None = Field( + default=None, + alias="config", + serialization_alias="config", + description=( + "Optional Orbit-supported configuration sent to the LLM scorer invoke endpoint." + ), + ) + payload_field: LlmPayloadField = Field( + default="input", + description=( + "Which scorer input side to use when selector output is a scalar value. " + "Structured selected data with input/output keys overrides this setting." + ), + ) + timeout_ms: int = Field( + default=10000, + ge=1000, + le=60000, + description="Request timeout in milliseconds (1-60 seconds)", + ) + + @model_validator(mode="after") + def validate_threshold(self) -> LlmEvaluatorConfig: + """Validate threshold compatibility with the configured operator.""" + if self.operator in _NUMERIC_OPERATORS and coerce_number(self.threshold) is None: + raise ValueError(f"operator '{self.operator}' requires a numeric threshold") + if self.operator != "any" and self.threshold is None: + raise ValueError("threshold is required unless operator is 'any'") + return self diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py new file mode 100644 index 00000000..7e7c8720 --- /dev/null +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py @@ -0,0 +1,387 @@ +"""Direct Galileo LLM-as-judge evaluator implementation.""" + +from __future__ import annotations + +import json +import logging +import os +from importlib.metadata import PackageNotFoundError, version +from typing import Any + +import httpx +from agent_control_evaluators import Evaluator, EvaluatorMetadata, register_evaluator +from agent_control_models import EvaluatorResult, JSONObject, JSONValue, Step + +from .client import GalileoExecutionContext, GalileoLLMClient, ScorerInvokeResponse +from .config import LlmEvaluatorConfig, coerce_number + +logger = logging.getLogger(__name__) + + +def _resolve_package_version() -> str: + """Return the installed package version, or a dev fallback during local imports.""" + try: + return version("agent-control-evaluator-galileo") + except PackageNotFoundError: + return "0.0.0.dev" + + +_PACKAGE_VERSION = _resolve_package_version() +LLM_AVAILABLE = True +_HTTP_ERROR_BODY_LIMIT = 500 + + +def _coerce_payload_text(value: Any) -> str | None: + """Coerce selected data into scorer text without losing structured values.""" + if value is None: + return None + if isinstance(value, str): + return value + if isinstance(value, (int, float, bool)): + return str(value) + try: + return json.dumps(value, ensure_ascii=False, sort_keys=True, default=str) + except TypeError: + return str(value) + + +def _has_text(value: str | None) -> bool: + return value is not None and value.strip() != "" + + +def _extract_dict_text(data: dict[str, Any], key: str) -> str | None: + if key not in data: + return None + return _coerce_payload_text(data.get(key)) + + +def _contains(score: JSONValue, threshold: JSONValue) -> bool: + if threshold is None: + return False + if isinstance(score, str): + return str(threshold) in score + if isinstance(score, list): + return threshold in score + if isinstance(score, dict): + return threshold in score.values() + return False + + +def _confidence_from_score(score: JSONValue) -> float: + if isinstance(score, bool): + return 1.0 if score else 0.0 + number = coerce_number(score) + if number is not None and 0.0 <= number <= 1.0: + return number + return 1.0 + + +def _truncated_http_response_body(body: str) -> tuple[str, bool]: + if len(body) <= _HTTP_ERROR_BODY_LIMIT: + return body, False + return body[:_HTTP_ERROR_BODY_LIMIT], True + + +def _http_status_error_metadata(error: httpx.HTTPStatusError) -> dict[str, Any]: + metadata: dict[str, Any] = {} + + request = error.request + metadata["http_method"] = request.method + metadata["http_endpoint_path"] = request.url.path + + response = error.response + metadata["http_status_code"] = response.status_code + metadata["http_response_content_type"] = response.headers.get("content-type") + + body = response.text + if body: + metadata["http_response_body"], metadata["http_response_body_truncated"] = ( + _truncated_http_response_body(body) + ) + + return {key: value for key, value in metadata.items() if value is not None} + + +@register_evaluator +class LlmEvaluator(Evaluator[LlmEvaluatorConfig]): + """Galileo LLM-as-judge evaluator using the direct scorer invocation API.""" + + metadata = EvaluatorMetadata( + name="galileo.llm", + version=_PACKAGE_VERSION, + description="Galileo LLM-as-judge direct scorer evaluation", + requires_api_key=True, + timeout_ms=10000, + ) + config_model = LlmEvaluatorConfig + + @classmethod + def is_available(cls) -> bool: + """Check whether required runtime dependencies are available.""" + return LLM_AVAILABLE + + def __init__(self, config: LlmEvaluatorConfig) -> None: + """Initialize the direct LLM-as-judge evaluator. + + Args: + config: Validated LlmEvaluatorConfig instance. + + Raises: + ValueError: If neither GALILEO_API_SECRET_KEY nor GALILEO_API_SECRET is set. + """ + has_secret = os.getenv("GALILEO_API_SECRET_KEY") or os.getenv("GALILEO_API_SECRET") + if not has_secret: + raise ValueError( + "GALILEO_API_SECRET_KEY or GALILEO_API_SECRET is required for LLM " + "scorer invocation. Set one as an environment variable before using " + "galileo.llm." + ) + + super().__init__(config) + self._client = GalileoLLMClient() + + def _get_client(self) -> GalileoLLMClient: + """Get the Galileo LLM client.""" + return self._client + + def _prepare_payload(self, data: Any) -> tuple[str | None, str | None]: + """Prepare scorer input/output fields from selected data.""" + if isinstance(data, dict): + input_text = _extract_dict_text(data, "input") + output_text = _extract_dict_text(data, "output") + if _has_text(input_text) or _has_text(output_text): + return input_text, output_text + + text = _coerce_payload_text(data) + if self.config.payload_field == "output": + return None, text + return text, None + + def _score_matches(self, score: JSONValue) -> bool: + """Apply the configured local threshold comparison to a raw LLM scorer score.""" + operator = self.config.operator + threshold = self.config.threshold + + if operator == "any": + return bool(score) + if operator == "eq": + return score == threshold + if operator == "ne": + return score != threshold + if operator == "contains": + return _contains(score, threshold) + + score_number = coerce_number(score) + threshold_number = coerce_number(threshold) + if score_number is None: + raise ValueError(f"LLM scorer score {score!r} is not numeric") + if threshold_number is None: + raise ValueError(f"LLM scorer threshold {threshold!r} is not numeric") + + if operator == "gt": + return score_number > threshold_number + if operator == "gte": + return score_number >= threshold_number + if operator == "lt": + return score_number < threshold_number + if operator == "lte": + return score_number <= threshold_number + + raise ValueError(f"Unsupported LLM scorer operator: {operator}") + + async def evaluate(self, data: Any) -> EvaluatorResult: + """Evaluate selected data with Galileo LLM-as-judge direct scorer invocation. + + Args: + data: The data selected from the runtime step. + + Returns: + EvaluatorResult with local threshold decision and scorer metadata. + """ + return await self._evaluate(data, step=None) + + async def evaluate_with_context(self, data: Any, step: Step) -> EvaluatorResult: + """Evaluate selected data while dual-writing the complete runtime step. + + Args: + data: Data selected by the configured control selector. + step: Complete runtime step for structured scorer context. + + Returns: + EvaluatorResult with local threshold decision and scorer metadata. + """ + return await self._evaluate(data, step=step) + + async def evaluate_with_extensions( + self, + data: Any, + step: Step, + extensions: JSONObject | None, + ) -> EvaluatorResult: + """Evaluate using opaque authenticated metadata supplied by Agent Control.""" + return await self._evaluate( + data, + step=step, + extensions=extensions, + ) + + @staticmethod + def _execution_context_from_extensions( + extensions: JSONObject | None, + ) -> GalileoExecutionContext | None: + """Translate trusted opaque auth metadata into Galileo's request context.""" + if extensions is None: + return None + + metadata = extensions.get("metadata") + metadata_obj = metadata if isinstance(metadata, dict) else {} + organization_id = extensions.get("namespace_key") + # Orbit's runtime-auth contract defines caller_id as the authenticated + # user ID for this flow. Keep that contract mapping inside Galileo. + user_id = extensions.get("caller_id") + project_id = metadata_obj.get("project_id") + target_type = extensions.get("target_type") + target_id = extensions.get("target_id") + run_id = target_id if target_type == "log_stream" else None + + missing = [ + field + for field, value in ( + ("organization_id", organization_id), + ("user_id", user_id), + ("project_id", project_id), + ("run_id", run_id), + ) + if not isinstance(value, str) or not value + ] + if missing: + raise ValueError( + "Authenticated execution metadata is missing required fields: " + + ", ".join(missing) + ) + + assert isinstance(organization_id, str) + assert isinstance(user_id, str) + assert isinstance(project_id, str) + assert isinstance(run_id, str) + return GalileoExecutionContext( + organization_id=organization_id, + user_id=user_id, + project_id=project_id, + run_id=run_id, + ) + + async def _evaluate( + self, + data: Any, + *, + step: Step | None, + extensions: JSONObject | None = None, + ) -> EvaluatorResult: + """Run an LLM scorer evaluation with optional structured runtime context.""" + input_text, output_text = self._prepare_payload(data) + if not (_has_text(input_text) or _has_text(output_text)): + return EvaluatorResult( + matched=False, + confidence=1.0, + message="No data to score with LLM scorer", + metadata=self._base_metadata(), + ) + + try: + execution_context = self._execution_context_from_extensions(extensions) + scorer_kwargs = self._scorer_kwargs() + if step is not None: + scorer_kwargs["step"] = step + if execution_context is not None: + scorer_kwargs["execution_context"] = execution_context + response = await self._get_client().invoke( + **scorer_kwargs, + input=input_text if _has_text(input_text) else None, + output=output_text if _has_text(output_text) else None, + config=self.config.scorer_config, + timeout=self.get_timeout_seconds(), + ) + + if response.status.lower() != "success": + message = response.error_message or f"LLM scorer status: {response.status}" + raise RuntimeError(message) + + matched = self._score_matches(response.score) + metadata = self._metadata(response) + operator = self.config.operator + threshold = self.config.threshold + state = "triggered" if matched else "not triggered" + return EvaluatorResult( + matched=matched, + confidence=_confidence_from_score(response.score), + message=( + f"LLM scorer score {response.score!r} {operator} threshold " + f"{threshold!r}: control {state}." + ), + metadata=metadata, + ) + except Exception as exc: + logger.error("LLM scorer evaluation error: %s", exc, exc_info=True) + return self._handle_error(exc) + + def _base_metadata(self) -> dict[str, Any]: + """Build result metadata without implying a requested version executed.""" + metadata: dict[str, Any] = {"scorer_id": self.config.scorer_id} + if self.config.scorer_version_id is not None: + metadata["requested_scorer_version_id"] = self.config.scorer_version_id + if self.config.scorer_label is not None: + metadata["scorer_label"] = self.config.scorer_label + return metadata + + def _scorer_kwargs(self) -> dict[str, Any]: + kwargs: dict[str, Any] = {"scorer_id": self.config.scorer_id} + if self.config.scorer_version_id is not None: + kwargs["scorer_version_id"] = self.config.scorer_version_id + if self.config.scorer_label is not None: + kwargs["scorer_label"] = self.config.scorer_label + return kwargs + + def _metadata( + self, + response: ScorerInvokeResponse, + ) -> dict[str, Any]: + metadata: dict[str, Any] = self._base_metadata() + echoed_label = response.scorer_label or self.config.scorer_label + if echoed_label is not None: + metadata["scorer_label"] = echoed_label + metadata.update( + { + "score": response.score, + "threshold": self.config.threshold, + "operator": self.config.operator, + "status": response.status, + "execution_time_seconds": response.execution_time, + "error_message": response.error_message, + } + ) + return metadata + + def _handle_error( + self, + error: Exception, + ) -> EvaluatorResult: + error_detail = str(error) + metadata: dict[str, Any] = { + **self._base_metadata(), + "error_type": type(error).__name__, + } + if isinstance(error, httpx.HTTPStatusError): + metadata.update(_http_status_error_metadata(error)) + + return EvaluatorResult( + matched=False, + confidence=0.0, + message=f"LLM scorer evaluation error: {error_detail}", + metadata=metadata, + error=error_detail, + ) + + async def aclose(self) -> None: + """Close the underlying Galileo LLM client.""" + await self._client.close() diff --git a/sdks/python/ARCHITECTURE.md b/sdks/python/ARCHITECTURE.md index 4a48c09a..a58fb4c7 100644 --- a/sdks/python/ARCHITECTURE.md +++ b/sdks/python/ARCHITECTURE.md @@ -20,7 +20,7 @@ sdks/python/src/agent_control/ ├── tracing.py # Distributed tracing support ├── py.typed # PEP 561 type marker └── evaluators/ # Evaluator base classes and discovery system - ├── __init__.py # Evaluator discovery, registration, and Luna integration + ├── __init__.py # Evaluator discovery, registration, and Luna/LLM integration └── base.py # Base Evaluator and EvaluatorMetadata classes ``` @@ -213,11 +213,11 @@ async def chat(message: str) -> str: **Key Components**: - Base evaluator classes (`Evaluator`, `EvaluatorMetadata`) - Evaluator discovery via entry points -- Third-party evaluator integration (e.g., Luna, Guardrails AI) +- Third-party evaluator integration (e.g., Luna, LLM, Guardrails AI) - Registration functions for custom evaluators **Structure**: -- `__init__.py` - Evaluator discovery (`discover_evaluators()`, `list_evaluators()`), registration (`register_evaluator()`), and optional Luna integration +- `__init__.py` - Evaluator discovery (`discover_evaluators()`, `list_evaluators()`), registration (`register_evaluator()`), and optional Luna/LLM integration - `base.py` - Base `Evaluator` and `EvaluatorMetadata` classes (re-exported from `agent_control_models`) **Usage**: diff --git a/sdks/python/src/agent_control/control_decorators.py b/sdks/python/src/agent_control/control_decorators.py index 8c12ae4c..6879ee0e 100644 --- a/sdks/python/src/agent_control/control_decorators.py +++ b/sdks/python/src/agent_control/control_decorators.py @@ -23,7 +23,7 @@ async def chat(message: str) -> str: # Server-side controls define: # - stage: "pre" or "post" # - selector.path: "input" or "output" - # - evaluator: regex, list, Luna evaluator, etc. + # - evaluator: regex, list, Luna evaluator, LLM evaluator, etc. # - action: deny, steer, or observe """ diff --git a/sdks/python/src/agent_control/evaluators/__init__.py b/sdks/python/src/agent_control/evaluators/__init__.py index 73714717..d239821e 100644 --- a/sdks/python/src/agent_control/evaluators/__init__.py +++ b/sdks/python/src/agent_control/evaluators/__init__.py @@ -1,7 +1,7 @@ """Evaluator system for agent_control. This module provides an evaluator architecture for extending agent_control -with external evaluation systems like Galileo Luna, Guardrails AI, etc. +with external evaluation systems like Galileo Luna, Galileo LLM, Guardrails AI, etc. Evaluator Discovery: Call `discover_evaluators()` at startup to load evaluators. This loads: @@ -14,6 +14,7 @@ When installed with galileo extras, the Galileo evaluator types are available: ```python from agent_control.evaluators import LunaEvaluator, LunaEvaluatorConfig # if galileo installed + from agent_control.evaluators import LlmEvaluator, LlmEvaluatorConfig # if galileo installed ``` """ @@ -62,3 +63,25 @@ ) except ImportError: pass + +# Optionally export LLM evaluator types when available +try: + from agent_control_evaluator_galileo.llm import ( # type: ignore[import-not-found] # noqa: F401 + LLM_AVAILABLE, + GalileoLLMClient, + LlmEvaluator, + LlmEvaluatorConfig, + LlmOperator, + ) + + __all__.extend( + [ + "GalileoLLMClient", + "LlmEvaluator", + "LlmEvaluatorConfig", + "LlmOperator", + "LLM_AVAILABLE", + ] + ) +except ImportError: + pass diff --git a/sdks/python/tests/test_evaluators_optional_imports.py b/sdks/python/tests/test_evaluators_optional_imports.py index b4560fc9..370e7a6d 100644 --- a/sdks/python/tests/test_evaluators_optional_imports.py +++ b/sdks/python/tests/test_evaluators_optional_imports.py @@ -28,6 +28,7 @@ def _module_available(name: str) -> bool: _GALILEO_INSTALLED = _module_available("agent_control_evaluator_galileo.luna") +_GALILEO_LLM_INSTALLED = _module_available("agent_control_evaluator_galileo.llm") def _reload_evaluators_with_blocked(prefix: str) -> object: @@ -70,21 +71,35 @@ def test_module_loads_when_galileo_luna_is_unavailable(): # Core names are always present. assert "Evaluator" in reloaded.__all__ - # Luna1 names are NOT present because the import failed. + # Luna names are NOT present because the import failed. assert "LunaEvaluator" not in reloaded.__all__ assert "GalileoLunaClient" not in reloaded.__all__ +def test_module_loads_when_galileo_llm_is_unavailable(): + """Hiding ``agent_control_evaluator_galileo.llm`` exercises its except branch.""" + reloaded = _reload_evaluators_with_blocked("agent_control_evaluator_galileo.llm") + + # Core names are always present. + assert "Evaluator" in reloaded.__all__ + # LLM names are NOT present because the import failed. + assert "LlmEvaluator" not in reloaded.__all__ + assert "GalileoLLMClient" not in reloaded.__all__ + + def test_module_loads_when_galileo_package_is_unavailable(): """Hiding the whole package exercises the ImportError fallback.""" reloaded = _reload_evaluators_with_blocked("agent_control_evaluator_galileo") assert "Evaluator" in reloaded.__all__ - # The optional luna names are absent. + # The optional luna and llm names are absent. for absent in ( "LunaEvaluator", "GalileoLunaClient", "LUNA_AVAILABLE", + "LlmEvaluator", + "GalileoLLMClient", + "LLM_AVAILABLE", ): assert absent not in reloaded.__all__ @@ -108,3 +123,20 @@ def test_module_loads_galileo_optional_imports_when_available(): finally: if saved is not None: sys.modules["agent_control.evaluators"] = saved + + +@pytest.mark.skipif( + not _GALILEO_LLM_INSTALLED, + reason="agent-control-evaluator-galileo extras not installed in this environment", +) +def test_module_loads_galileo_llm_optional_imports_when_available(): + """Sanity check: with galileo installed, the LLM optional names ARE exposed.""" + saved = sys.modules.pop("agent_control.evaluators", None) + try: + import agent_control.evaluators as reloaded + + reloaded = importlib.reload(reloaded) + assert "LlmEvaluator" in reloaded.__all__ + finally: + if saved is not None: + sys.modules["agent_control.evaluators"] = saved From 9e8df0586a94ac18afbf9d46ff968e1e9d578297 Mon Sep 17 00:00:00 2001 From: Wrisa Date: Fri, 2 Oct 2026 12:17:56 -0700 Subject: [PATCH 02/10] fix(evaluators): fix import sort order in galileo evaluator __init__ files --- .../agent_control_evaluator_galileo/__init__.py | 14 +++++++------- .../llm/__init__.py | 2 +- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py index 71e8e730..74f09225 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py @@ -20,6 +20,13 @@ except PackageNotFoundError: __version__ = "0.0.0.dev" +from agent_control_evaluator_galileo.llm import ( + LLM_AVAILABLE, + GalileoLLMClient, + LlmEvaluator, + LlmEvaluatorConfig, + LlmOperator, +) from agent_control_evaluator_galileo.luna import ( LUNA_AVAILABLE, GalileoLunaClient, @@ -31,13 +38,6 @@ ScorerInvokeRequest, ScorerInvokeResponse, ) -from agent_control_evaluator_galileo.llm import ( - LLM_AVAILABLE, - GalileoLLMClient, - LlmEvaluator, - LlmEvaluatorConfig, - LlmOperator, -) from agent_control_evaluator_galileo.records import ( GalileoRecord, GalileoRecordNormalizer, diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py index 518febbe..7d78b162 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py @@ -1,8 +1,8 @@ """Galileo LLM-as-judge direct scorer evaluator.""" from agent_control_evaluator_galileo.llm.client import ( - GalileoLLMClient, GalileoExecutionContext, + GalileoLLMClient, ScorerInvokeInputs, ScorerInvokeRecord, ScorerInvokeRequest, From 14c41abe529b02a5952f832296700992a9caca9c Mon Sep 17 00:00:00 2001 From: Wrisa Date: Sat, 3 Oct 2026 13:28:13 -0700 Subject: [PATCH 03/10] added execution context --- .../llm/client.py | 13 +- .../llm/evaluator.py | 38 ++-- .../galileo/tests/test_llm_evaluator.py | 191 ++++++++++++++++++ 3 files changed, 212 insertions(+), 30 deletions(-) create mode 100644 evaluators/contrib/galileo/tests/test_llm_evaluator.py diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py index 47b21e6b..396b0216 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py @@ -229,12 +229,12 @@ class ScorerInvokeRecord(BaseModel): class GalileoExecutionContext(BaseModel): - """Authenticated identity fields required by Galileo scorer invocation.""" + """Authenticated organization and optional identity fields for Galileo scorer invocation.""" organization_id: str = Field(min_length=1) - user_id: str = Field(min_length=1) - project_id: str = Field(min_length=1) - run_id: str = Field(min_length=1) + user_id: str | None = Field(default=None, min_length=1) + project_id: str | None = Field(default=None, min_length=1) + run_id: str | None = Field(default=None, min_length=1) class ScorerInvokeRequest(BaseModel): @@ -267,7 +267,10 @@ def ensure_required_values(self) -> ScorerInvokeRequest: def to_dict(self) -> JSONObject: """Convert to the LLM scorer invoke request shape.""" - return self.model_dump(mode="json", exclude_none=True) + request = self.model_dump(mode="json", exclude_none=True) + if self.execution_context is not None: + request["execution_context"] = self.execution_context.model_dump(mode="json") + return request def _orbit_record_from_step( diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py index 7e7c8720..6c12ddd1 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py @@ -236,38 +236,26 @@ def _execution_context_from_extensions( metadata = extensions.get("metadata") metadata_obj = metadata if isinstance(metadata, dict) else {} organization_id = extensions.get("namespace_key") - # Orbit's runtime-auth contract defines caller_id as the authenticated - # user ID for this flow. Keep that contract mapping inside Galileo. + # The runtime envelope's caller_id is supplied by the authenticated + # principal and is the caller identity Galileo associates with this run. user_id = extensions.get("caller_id") project_id = metadata_obj.get("project_id") target_type = extensions.get("target_type") target_id = extensions.get("target_id") - run_id = target_id if target_type == "log_stream" else None - - missing = [ - field - for field, value in ( - ("organization_id", organization_id), - ("user_id", user_id), - ("project_id", project_id), - ("run_id", run_id), - ) - if not isinstance(value, str) or not value - ] - if missing: - raise ValueError( - "Authenticated execution metadata is missing required fields: " - + ", ".join(missing) - ) + run_id = ( + target_id + if target_type in ("log_stream", "agent_stream") + and isinstance(target_id, str) + and target_id + else None + ) + if not isinstance(organization_id, str) or not organization_id: + raise ValueError("Authenticated execution metadata is missing organization_id") - assert isinstance(organization_id, str) - assert isinstance(user_id, str) - assert isinstance(project_id, str) - assert isinstance(run_id, str) return GalileoExecutionContext( organization_id=organization_id, - user_id=user_id, - project_id=project_id, + user_id=user_id if isinstance(user_id, str) and user_id else None, + project_id=project_id if isinstance(project_id, str) and project_id else None, run_id=run_id, ) diff --git a/evaluators/contrib/galileo/tests/test_llm_evaluator.py b/evaluators/contrib/galileo/tests/test_llm_evaluator.py new file mode 100644 index 00000000..814d9de3 --- /dev/null +++ b/evaluators/contrib/galileo/tests/test_llm_evaluator.py @@ -0,0 +1,191 @@ +"""Tests for the direct Galileo LLM-as-judge evaluator and client.""" + +from __future__ import annotations + +import os +from unittest.mock import AsyncMock, patch + +import pytest +from agent_control_models import Step + +LLM_ENV = { + "GALILEO_API_SECRET_KEY": "test-secret", + "GALILEO_LUNA_INVOKE_URL": "http://luna-invoke:8090", +} + + +class TestGalileoLLMClient: + """Tests for ``GalileoLLMClient`` and ``ScorerInvokeRequest``.""" + + def test_execution_context_serializes_missing_optional_claims_as_null(self) -> None: + from agent_control_evaluator_galileo.llm.client import ( + GalileoExecutionContext, + ScorerInvokeInputs, + ScorerInvokeRequest, + ) + + request = ScorerInvokeRequest( + scorer_id="scorer-123", + inputs=ScorerInvokeInputs(query="hello"), + execution_context=GalileoExecutionContext(organization_id="org-1"), + ) + + assert request.to_dict()["execution_context"] == { + "organization_id": "org-1", + "user_id": None, + "project_id": None, + "run_id": None, + } + + +class TestLlmEvaluator: + """Tests for ``LlmEvaluator`` execution context handling.""" + + @patch.dict(os.environ, LLM_ENV) + @pytest.mark.asyncio + @pytest.mark.parametrize("target_type", ["log_stream", "agent_stream"]) + async def test_evaluator_consumes_verified_opaque_extensions(self, target_type: str) -> None: + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import ( + GalileoExecutionContext, + GalileoLLMClient, + ScorerInvokeResponse, + ) + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "operator": "gte"} + ) + extensions = { + "namespace_key": "org-1", + "caller_id": "verified-user-2", + "target_type": target_type, + "target_id": "run-4", + "metadata": {"project_id": "project-3"}, + } + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate_with_extensions( + "selected input", Step(type="llm", name="answer", input="prompt"), extensions + ) + + assert result.matched is True + mock_invoke.assert_awaited_once_with( + scorer_id="scorer-123", + step=Step(type="llm", name="answer", input="prompt"), + execution_context=GalileoExecutionContext( + organization_id="org-1", + user_id="verified-user-2", + project_id="project-3", + run_id="run-4", + ), + input="selected input", + output=None, + config=None, + timeout=10.0, + ) + + @patch.dict(os.environ, LLM_ENV) + @pytest.mark.asyncio + async def test_evaluator_allows_missing_caller_id(self) -> None: + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import ( + GalileoExecutionContext, + GalileoLLMClient, + ScorerInvokeResponse, + ) + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123"}) + extensions = { + "namespace_key": "org-1", + "target_type": "log_stream", + "target_id": "run-4", + "metadata": {}, + } + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate_with_extensions( + "selected input", Step(type="llm", name="answer", input="prompt"), extensions + ) + + assert result.error is None + mock_invoke.assert_awaited_once_with( + scorer_id="scorer-123", + step=Step(type="llm", name="answer", input="prompt"), + execution_context=GalileoExecutionContext( + organization_id="org-1", + user_id=None, + project_id=None, + run_id="run-4", + ), + input="selected input", + output=None, + config=None, + timeout=10.0, + ) + + @patch.dict(os.environ, LLM_ENV) + @pytest.mark.asyncio + async def test_evaluator_omits_run_id_for_unsupported_target_type(self) -> None: + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import ( + GalileoExecutionContext, + GalileoLLMClient, + ScorerInvokeResponse, + ) + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123"}) + extensions = { + "namespace_key": "org-1", + "caller_id": "verified-user-2", + "target_type": "environment", + "target_id": "prod", + "metadata": {"project_id": "project-3"}, + } + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate_with_extensions( + "selected input", Step(type="llm", name="answer", input="prompt"), extensions + ) + + assert result.error is None + mock_invoke.assert_awaited_once_with( + scorer_id="scorer-123", + step=Step(type="llm", name="answer", input="prompt"), + execution_context=GalileoExecutionContext( + organization_id="org-1", + user_id="verified-user-2", + project_id="project-3", + run_id=None, + ), + input="selected input", + output=None, + config=None, + timeout=10.0, + ) + + @patch.dict(os.environ, LLM_ENV) + @pytest.mark.asyncio + async def test_evaluator_rejects_missing_organization_id(self) -> None: + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123"}) + extensions = { + "target_type": "log_stream", + "target_id": "run-4", + "metadata": {"project_id": "project-3"}, + } + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + result = await evaluator.evaluate_with_extensions( + "selected input", + Step(type="llm", name="answer", input="prompt"), + extensions, + ) + + assert result.error is not None + assert "organization_id" in result.error + mock_invoke.assert_not_called() From 103aebf0db57d0b67844b7b845d28c243360ea9a Mon Sep 17 00:00:00 2001 From: wrisa Date: Mon, 5 Oct 2026 14:55:06 -0700 Subject: [PATCH 04/10] added logs and identity log to llm and luna evaluators Port the structured logging and execution-context identity log from feature/SAO-16858-add-evaluator-for-LLM-thin (_shared/evaluator.py) to the separate llm/evaluator.py and luna/evaluator.py files on this branch. Adds dispatcher, skip, pre-invoke, result, and error logs to _evaluate, and an identity log to _execution_context_from_extensions. Co-Authored-By: Claude Opus 4.6 --- .../llm/evaluator.py | 44 +++++++++++++++++-- .../luna/evaluator.py | 44 +++++++++++++++++-- 2 files changed, 82 insertions(+), 6 deletions(-) diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py index 6c12ddd1..7fa9c3b7 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py @@ -252,10 +252,23 @@ def _execution_context_from_extensions( if not isinstance(organization_id, str) or not organization_id: raise ValueError("Authenticated execution metadata is missing organization_id") + resolved_user_id = user_id if isinstance(user_id, str) and user_id else None + resolved_project_id = project_id if isinstance(project_id, str) and project_id else None + raw_api_key = os.getenv("GALILEO_API_SECRET_KEY") or os.getenv("GALILEO_API_SECRET") or "" + api_key_hint = f"...{raw_api_key[-4:]}" if len(raw_api_key) >= 4 else "***" + logger.info( + "[execution_context] caller_id=%r → user_id=%r org=%r project=%r run=%r api_key=%s", + user_id, + resolved_user_id, + organization_id, + resolved_project_id, + run_id, + api_key_hint, + ) return GalileoExecutionContext( organization_id=organization_id, - user_id=user_id if isinstance(user_id, str) and user_id else None, - project_id=project_id if isinstance(project_id, str) and project_id else None, + user_id=resolved_user_id, + project_id=resolved_project_id, run_id=run_id, ) @@ -267,8 +280,12 @@ async def _evaluate( extensions: JSONObject | None = None, ) -> EvaluatorResult: """Run an LLM scorer evaluation with optional structured runtime context.""" + display = self.__class__.__name__ + scorer_id = self.config.scorer_id + logger.info("[%s] Dispatching scorer evaluation: scorer_id=%s", display, scorer_id) input_text, output_text = self._prepare_payload(data) if not (_has_text(input_text) or _has_text(output_text)): + logger.info("[%s] Skipping scorer invocation: no data to score", display) return EvaluatorResult( matched=False, confidence=1.0, @@ -283,6 +300,12 @@ async def _evaluate( scorer_kwargs["step"] = step if execution_context is not None: scorer_kwargs["execution_context"] = execution_context + logger.info( + "[%s] Invoking scorer: scorer_id=%s timeout=%.1fs", + display, + scorer_id, + self.get_timeout_seconds(), + ) response = await self._get_client().invoke( **scorer_kwargs, input=input_text if _has_text(input_text) else None, @@ -300,6 +323,15 @@ async def _evaluate( operator = self.config.operator threshold = self.config.threshold state = "triggered" if matched else "not triggered" + logger.info( + "[%s] Scorer result: scorer_id=%s score=%r %s threshold=%r → %s", + display, + scorer_id, + response.score, + operator, + threshold, + state, + ) return EvaluatorResult( matched=matched, confidence=_confidence_from_score(response.score), @@ -310,7 +342,13 @@ async def _evaluate( metadata=metadata, ) except Exception as exc: - logger.error("LLM scorer evaluation error: %s", exc, exc_info=True) + logger.error( + "[%s] Scorer evaluation error: scorer_id=%s error=%s", + display, + scorer_id, + exc, + exc_info=True, + ) return self._handle_error(exc) def _base_metadata(self) -> dict[str, Any]: diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/luna/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/luna/evaluator.py index a09b4ccf..23932bc0 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/luna/evaluator.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/luna/evaluator.py @@ -252,10 +252,23 @@ def _execution_context_from_extensions( if not isinstance(organization_id, str) or not organization_id: raise ValueError("Authenticated execution metadata is missing organization_id") + resolved_user_id = user_id if isinstance(user_id, str) and user_id else None + resolved_project_id = project_id if isinstance(project_id, str) and project_id else None + raw_api_key = os.getenv("GALILEO_API_SECRET_KEY") or os.getenv("GALILEO_API_SECRET") or "" + api_key_hint = f"...{raw_api_key[-4:]}" if len(raw_api_key) >= 4 else "***" + logger.info( + "[execution_context] caller_id=%r → user_id=%r org=%r project=%r run=%r api_key=%s", + user_id, + resolved_user_id, + organization_id, + resolved_project_id, + run_id, + api_key_hint, + ) return GalileoExecutionContext( organization_id=organization_id, - user_id=user_id if isinstance(user_id, str) and user_id else None, - project_id=project_id if isinstance(project_id, str) and project_id else None, + user_id=resolved_user_id, + project_id=resolved_project_id, run_id=run_id, ) @@ -267,8 +280,12 @@ async def _evaluate( extensions: JSONObject | None = None, ) -> EvaluatorResult: """Run a Luna evaluation with optional structured runtime context.""" + display = self.__class__.__name__ + scorer_id = self.config.scorer_id + logger.info("[%s] Dispatching scorer evaluation: scorer_id=%s", display, scorer_id) input_text, output_text = self._prepare_payload(data) if not (_has_text(input_text) or _has_text(output_text)): + logger.info("[%s] Skipping scorer invocation: no data to score", display) return EvaluatorResult( matched=False, confidence=1.0, @@ -285,6 +302,12 @@ async def _evaluate( scorer_kwargs["selected_data_payload_field"] = self.config.payload_field if execution_context is not None: scorer_kwargs["execution_context"] = execution_context + logger.info( + "[%s] Invoking scorer: scorer_id=%s timeout=%.1fs", + display, + scorer_id, + self.get_timeout_seconds(), + ) response = await self._get_client().invoke( **scorer_kwargs, input=input_text if _has_text(input_text) else None, @@ -302,6 +325,15 @@ async def _evaluate( operator = self.config.operator threshold = self.config.threshold state = "triggered" if matched else "not triggered" + logger.info( + "[%s] Scorer result: scorer_id=%s score=%r %s threshold=%r → %s", + display, + scorer_id, + response.score, + operator, + threshold, + state, + ) return EvaluatorResult( matched=matched, confidence=_confidence_from_score(response.score), @@ -312,7 +344,13 @@ async def _evaluate( metadata=metadata, ) except Exception as exc: - logger.error("Luna evaluation error: %s", exc, exc_info=True) + logger.error( + "[%s] Scorer evaluation error: scorer_id=%s error=%s", + display, + scorer_id, + exc, + exc_info=True, + ) return self._handle_error(exc) def _base_metadata(self) -> dict[str, Any]: From 8a4a718710f4cdea93d33d1653b76832ba3293a3 Mon Sep 17 00:00:00 2001 From: wrisa Date: Mon, 5 Oct 2026 15:26:26 -0700 Subject: [PATCH 05/10] Revert "added logs and identity log to llm and luna evaluators" This reverts commit ff00a4f4a32a3ec10c9f804e386c29649b107350. --- .../llm/evaluator.py | 44 ++----------------- .../luna/evaluator.py | 44 ++----------------- 2 files changed, 6 insertions(+), 82 deletions(-) diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py index 7fa9c3b7..6c12ddd1 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py @@ -252,23 +252,10 @@ def _execution_context_from_extensions( if not isinstance(organization_id, str) or not organization_id: raise ValueError("Authenticated execution metadata is missing organization_id") - resolved_user_id = user_id if isinstance(user_id, str) and user_id else None - resolved_project_id = project_id if isinstance(project_id, str) and project_id else None - raw_api_key = os.getenv("GALILEO_API_SECRET_KEY") or os.getenv("GALILEO_API_SECRET") or "" - api_key_hint = f"...{raw_api_key[-4:]}" if len(raw_api_key) >= 4 else "***" - logger.info( - "[execution_context] caller_id=%r → user_id=%r org=%r project=%r run=%r api_key=%s", - user_id, - resolved_user_id, - organization_id, - resolved_project_id, - run_id, - api_key_hint, - ) return GalileoExecutionContext( organization_id=organization_id, - user_id=resolved_user_id, - project_id=resolved_project_id, + user_id=user_id if isinstance(user_id, str) and user_id else None, + project_id=project_id if isinstance(project_id, str) and project_id else None, run_id=run_id, ) @@ -280,12 +267,8 @@ async def _evaluate( extensions: JSONObject | None = None, ) -> EvaluatorResult: """Run an LLM scorer evaluation with optional structured runtime context.""" - display = self.__class__.__name__ - scorer_id = self.config.scorer_id - logger.info("[%s] Dispatching scorer evaluation: scorer_id=%s", display, scorer_id) input_text, output_text = self._prepare_payload(data) if not (_has_text(input_text) or _has_text(output_text)): - logger.info("[%s] Skipping scorer invocation: no data to score", display) return EvaluatorResult( matched=False, confidence=1.0, @@ -300,12 +283,6 @@ async def _evaluate( scorer_kwargs["step"] = step if execution_context is not None: scorer_kwargs["execution_context"] = execution_context - logger.info( - "[%s] Invoking scorer: scorer_id=%s timeout=%.1fs", - display, - scorer_id, - self.get_timeout_seconds(), - ) response = await self._get_client().invoke( **scorer_kwargs, input=input_text if _has_text(input_text) else None, @@ -323,15 +300,6 @@ async def _evaluate( operator = self.config.operator threshold = self.config.threshold state = "triggered" if matched else "not triggered" - logger.info( - "[%s] Scorer result: scorer_id=%s score=%r %s threshold=%r → %s", - display, - scorer_id, - response.score, - operator, - threshold, - state, - ) return EvaluatorResult( matched=matched, confidence=_confidence_from_score(response.score), @@ -342,13 +310,7 @@ async def _evaluate( metadata=metadata, ) except Exception as exc: - logger.error( - "[%s] Scorer evaluation error: scorer_id=%s error=%s", - display, - scorer_id, - exc, - exc_info=True, - ) + logger.error("LLM scorer evaluation error: %s", exc, exc_info=True) return self._handle_error(exc) def _base_metadata(self) -> dict[str, Any]: diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/luna/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/luna/evaluator.py index 23932bc0..a09b4ccf 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/luna/evaluator.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/luna/evaluator.py @@ -252,23 +252,10 @@ def _execution_context_from_extensions( if not isinstance(organization_id, str) or not organization_id: raise ValueError("Authenticated execution metadata is missing organization_id") - resolved_user_id = user_id if isinstance(user_id, str) and user_id else None - resolved_project_id = project_id if isinstance(project_id, str) and project_id else None - raw_api_key = os.getenv("GALILEO_API_SECRET_KEY") or os.getenv("GALILEO_API_SECRET") or "" - api_key_hint = f"...{raw_api_key[-4:]}" if len(raw_api_key) >= 4 else "***" - logger.info( - "[execution_context] caller_id=%r → user_id=%r org=%r project=%r run=%r api_key=%s", - user_id, - resolved_user_id, - organization_id, - resolved_project_id, - run_id, - api_key_hint, - ) return GalileoExecutionContext( organization_id=organization_id, - user_id=resolved_user_id, - project_id=resolved_project_id, + user_id=user_id if isinstance(user_id, str) and user_id else None, + project_id=project_id if isinstance(project_id, str) and project_id else None, run_id=run_id, ) @@ -280,12 +267,8 @@ async def _evaluate( extensions: JSONObject | None = None, ) -> EvaluatorResult: """Run a Luna evaluation with optional structured runtime context.""" - display = self.__class__.__name__ - scorer_id = self.config.scorer_id - logger.info("[%s] Dispatching scorer evaluation: scorer_id=%s", display, scorer_id) input_text, output_text = self._prepare_payload(data) if not (_has_text(input_text) or _has_text(output_text)): - logger.info("[%s] Skipping scorer invocation: no data to score", display) return EvaluatorResult( matched=False, confidence=1.0, @@ -302,12 +285,6 @@ async def _evaluate( scorer_kwargs["selected_data_payload_field"] = self.config.payload_field if execution_context is not None: scorer_kwargs["execution_context"] = execution_context - logger.info( - "[%s] Invoking scorer: scorer_id=%s timeout=%.1fs", - display, - scorer_id, - self.get_timeout_seconds(), - ) response = await self._get_client().invoke( **scorer_kwargs, input=input_text if _has_text(input_text) else None, @@ -325,15 +302,6 @@ async def _evaluate( operator = self.config.operator threshold = self.config.threshold state = "triggered" if matched else "not triggered" - logger.info( - "[%s] Scorer result: scorer_id=%s score=%r %s threshold=%r → %s", - display, - scorer_id, - response.score, - operator, - threshold, - state, - ) return EvaluatorResult( matched=matched, confidence=_confidence_from_score(response.score), @@ -344,13 +312,7 @@ async def _evaluate( metadata=metadata, ) except Exception as exc: - logger.error( - "[%s] Scorer evaluation error: scorer_id=%s error=%s", - display, - scorer_id, - exc, - exc_info=True, - ) + logger.error("Luna evaluation error: %s", exc, exc_info=True) return self._handle_error(exc) def _base_metadata(self) -> dict[str, Any]: From 27d8ea0257d679d3a7014c98d0cb1ab835d6520e Mon Sep 17 00:00:00 2001 From: Wrisa Date: Mon, 5 Oct 2026 15:39:15 -0700 Subject: [PATCH 06/10] added coverage --- .../galileo/tests/test_llm_coverage_gaps.py | 1015 +++++++++++++++++ 1 file changed, 1015 insertions(+) create mode 100644 evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py diff --git a/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py b/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py new file mode 100644 index 00000000..e24b67de --- /dev/null +++ b/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py @@ -0,0 +1,1015 @@ +"""Targeted tests filling coverage gaps in llm/evaluator.py and llm/client.py. + +These tests cover the small utility functions and rare branches that the +integration-style tests in ``test_llm_evaluator.py`` skip past. +""" + +from __future__ import annotations + +import json +import os +from base64 import urlsafe_b64decode +from unittest.mock import AsyncMock, MagicMock, patch + +import httpx +import pytest + +LLM_ENV = { + "GALILEO_API_SECRET_KEY": "test-secret", + "GALILEO_LUNA_INVOKE_URL": "http://luna-invoke:8090", +} + + +def _decode_jwt_payload(token: str) -> dict[str, object]: + payload_segment = token.split(".")[1] + padded = payload_segment + ("=" * (-len(payload_segment) % 4)) + return json.loads(urlsafe_b64decode(padded.encode()).decode()) + + +# ============================================================================= +# llm/evaluator.py: utility helpers +# ============================================================================= + + +class TestCoercePayloadText: + """``_coerce_payload_text`` normalises arbitrary values to strings.""" + + def test_none_returns_none(self): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + assert _coerce_payload_text(None) is None + + def test_string_passed_through(self): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + assert _coerce_payload_text("hello") == "hello" + + @pytest.mark.parametrize("value", [42, 3.14, True]) + def test_scalars_stringified(self, value): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + assert _coerce_payload_text(value) == str(value) + + def test_dict_is_json_serialized(self): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + result = _coerce_payload_text({"a": 1, "b": 2}) + + assert json.loads(result) == {"a": 1, "b": 2} + + def test_unserialisable_falls_back_to_str(self): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + class CannotJson: + def __repr__(self): + return "" + + cannot = CannotJson() + result = _coerce_payload_text({"obj": cannot}) + + # default=str converts the inner object, so we still get a JSON string. + assert isinstance(result, str) + + +class TestExtractDictText: + """``_extract_dict_text`` returns ``None`` for missing keys.""" + + def test_missing_key_returns_none(self): + from agent_control_evaluator_galileo.llm.evaluator import _extract_dict_text + + assert _extract_dict_text({}, "absent") is None + + def test_present_key_coerced(self): + from agent_control_evaluator_galileo.llm.evaluator import _extract_dict_text + + assert _extract_dict_text({"x": 7}, "x") == "7" + + +class TestContains: + """``_contains`` supports str/list and dict values against a threshold.""" + + def test_none_threshold_is_no_match(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains("anything", None) is False + + def test_string_contains_substring(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains("hello world", "world") is True + assert _contains("hello world", "absent") is False + + def test_list_contains_value(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains(["a", "b", "c"], "b") is True + assert _contains(["a", "b", "c"], "z") is False + + def test_dict_threshold_does_not_match_key(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains({"toxicity": 0.9}, "toxicity") is False + + def test_dict_threshold_matches_value(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains({"label": "flagged"}, "flagged") is True + + def test_other_types_return_false(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains(42, 42) is False + + +class TestConfidenceFromScore: + """``_confidence_from_score`` maps a raw score to [0, 1].""" + + def test_true_bool_maps_to_one(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score(True) == 1.0 + + def test_false_bool_maps_to_zero(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score(False) == 0.0 + + def test_in_range_number_returned_as_is(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score(0.42) == 0.42 + + def test_out_of_range_falls_back_to_one(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score(7.2) == 1.0 + + def test_non_numeric_falls_back_to_one(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score("not-a-number") == 1.0 + + +# ============================================================================= +# llm/evaluator.py: _score_matches operator branches +# ============================================================================= + + +@pytest.fixture +def llm_evaluator(monkeypatch): + """A ready-to-use LlmEvaluator instance with auth env wired up.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + return LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5, "operator": "gte"}) + + +class TestScoreMatchesOperators: + """Every operator branch in ``_score_matches`` should evaluate.""" + + def _make(self, operator, threshold, monkeypatch): + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + return LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": threshold, "operator": operator} + ) + + def test_any_truthy_score_matches(self, monkeypatch): + evaluator = self._make("any", 0.5, monkeypatch) + assert evaluator._score_matches(1) is True + assert evaluator._score_matches(0) is False + + def test_eq_matches_threshold(self, monkeypatch): + evaluator = self._make("eq", "flagged", monkeypatch) + assert evaluator._score_matches("flagged") is True + assert evaluator._score_matches("safe") is False + + def test_ne_matches_when_different(self, monkeypatch): + evaluator = self._make("ne", "flagged", monkeypatch) + assert evaluator._score_matches("safe") is True + assert evaluator._score_matches("flagged") is False + + def test_contains_matches_substring(self, monkeypatch): + evaluator = self._make("contains", "flag", monkeypatch) + assert evaluator._score_matches("flagged") is True + assert evaluator._score_matches("clean") is False + + def test_numeric_operators_all_branches(self, monkeypatch): + for op, expectations in [ + ("gt", [(0.9, True), (0.5, False)]), + ("gte", [(0.5, True), (0.4, False)]), + ("lt", [(0.4, True), (0.5, False)]), + ("lte", [(0.5, True), (0.6, False)]), + ]: + evaluator = self._make(op, 0.5, monkeypatch) + for score, expected in expectations: + assert evaluator._score_matches(score) is expected, (op, score) + + def test_numeric_operator_rejects_non_numeric_score(self, monkeypatch): + evaluator = self._make("gte", 0.5, monkeypatch) + with pytest.raises(ValueError, match="not numeric"): + evaluator._score_matches("not-a-number") + + +# ============================================================================= +# llm/evaluator.py: payload preparation + aclose +# ============================================================================= + + +class TestPreparePayload: + """``_prepare_payload`` routes scalar data using explicit config.""" + + def test_scalar_routed_to_input_by_default(self, monkeypatch): + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + + input_text, output_text = evaluator._prepare_payload("hello") + + assert input_text == "hello" + assert output_text is None + + def test_scalar_routed_to_output_when_payload_field_is_output(self, monkeypatch): + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "payload_field": "output"} + ) + + input_text, output_text = evaluator._prepare_payload("hello") + + assert input_text is None + assert output_text == "hello" + + def test_structured_payload_uses_input_output_keys_over_payload_field(self, monkeypatch): + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "payload_field": "output"} + ) + + input_text, output_text = evaluator._prepare_payload( + {"input": "prompt", "output": "answer"} + ) + + assert input_text == "prompt" + assert output_text == "answer" + + +@pytest.mark.asyncio +async def test_evaluator_aclose_closes_underlying_client(monkeypatch): + """``aclose`` must release the eagerly-created client without clearing it.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + + fake = MagicMock() + fake.close = AsyncMock() + evaluator._client = fake + + await evaluator.aclose() + + fake.close.assert_awaited_once() + assert evaluator._client is fake + + +@pytest.mark.asyncio +async def test_evaluator_handles_non_success_status(monkeypatch): + """A non-success status from the scorer must surface as an error result.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator, ScorerInvokeResponse + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "operator": "gte"} + ) + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse( + scorer_label="toxicity", + score=None, + status="failed", + error_message="upstream timeout", + ) + + result = await evaluator.evaluate("hello") + + assert result.matched is False + assert result.error is not None + assert "upstream timeout" in result.error + + +@pytest.mark.asyncio +async def test_evaluator_skips_empty_data(monkeypatch): + """Evaluator skips invocation when there is no text to score.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + result = await evaluator.evaluate(None) + + assert result.matched is False + assert result.error is None + assert "No data to score" in result.message + mock_invoke.assert_not_called() + + +@pytest.mark.asyncio +async def test_evaluator_returns_error_result_on_http_error(monkeypatch): + """HTTP errors must surface as a non-matched EvaluatorResult with metadata.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + + fake_request = httpx.Request("POST", "http://luna-invoke:8090/api/v1/scorers/invoke") + fake_response = httpx.Response(500, text="internal server error", request=fake_request) + + with patch.object( + GalileoLLMClient, + "invoke", + new_callable=AsyncMock, + side_effect=httpx.HTTPStatusError("500", request=fake_request, response=fake_response), + ): + result = await evaluator.evaluate("hello") + + assert result.matched is False + assert result.error is not None + assert result.metadata.get("http_status_code") == 500 + + +# ============================================================================= +# llm/evaluator.py: package version fallback +# ============================================================================= + + +def test_resolve_package_version_falls_back_when_metadata_missing(): + """The dev fallback must trigger when the package isn't installed by metadata.""" + from importlib.metadata import PackageNotFoundError + + from agent_control_evaluator_galileo.llm import evaluator as evaluator_module + + with patch.object(evaluator_module, "version", side_effect=PackageNotFoundError): + result = evaluator_module._resolve_package_version() + + assert result == "0.0.0.dev" + + +# ============================================================================= +# llm/client.py: small helpers + branches +# ============================================================================= + + +class TestAsFloatOrNone: + """``_as_float_or_none`` parses scalar values; strings may fail.""" + + def test_returns_none_for_bool(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none(True) is None + + def test_returns_none_for_none(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none(None) is None + + def test_returns_float_for_int(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none(7) == 7.0 + + def test_returns_float_for_string_number(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none("0.42") == 0.42 + + def test_returns_none_for_unparseable_string(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none("not-a-number") is None + + def test_returns_none_for_other_types(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none([1, 2]) is None + + +class TestHasValue: + """``_has_value`` is the "is this scorable" predicate.""" + + def test_none_is_empty(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value(None) is False + + def test_empty_string_is_empty(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value("") is False + assert _has_value(" ") is False + + def test_non_empty_string_has_value(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value("hi") is True + + def test_empty_list_or_dict_is_empty(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value([]) is False + assert _has_value({}) is False + + def test_non_empty_list_or_dict_has_value(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value([1]) is True + assert _has_value({"k": "v"}) is True + + def test_scalar_other_types_have_value(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value(42) is True + assert _has_value(0) is True + assert _has_value(True) is True + + +class TestEffectiveScorerTimeout: + """``_effective_scorer_timeout`` applies 80% server timeout by default.""" + + def test_defaults_to_80_percent_of_http_timeout(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeConfig, + _effective_scorer_timeout, + ) + + config = _effective_scorer_timeout(ScorerInvokeConfig(), http_timeout_seconds=10.0) + assert config.request_timeout_seconds == pytest.approx(8.0) + + def test_explicit_server_timeout_preserved_when_shorter(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeConfig, + _effective_scorer_timeout, + ) + + config = _effective_scorer_timeout( + ScorerInvokeConfig(request_timeout_seconds=5.0), http_timeout_seconds=10.0 + ) + assert config.request_timeout_seconds == 5.0 + + def test_explicit_server_timeout_equal_to_http_raises(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeConfig, + _effective_scorer_timeout, + ) + + with pytest.raises(ValueError, match="shorter than the HTTP timeout"): + _effective_scorer_timeout( + ScorerInvokeConfig(request_timeout_seconds=10.0), http_timeout_seconds=10.0 + ) + + def test_zero_http_timeout_raises(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeConfig, + _effective_scorer_timeout, + ) + + with pytest.raises(ValueError, match="greater than 0"): + _effective_scorer_timeout(ScorerInvokeConfig(), http_timeout_seconds=0.0) + + +class TestLoadFloatEnv: + """``_load_float_env`` reads float env vars with a default.""" + + def test_returns_default_when_unset(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_float_env + + monkeypatch.delenv("TEST_FLOAT_ENV", raising=False) + assert _load_float_env("TEST_FLOAT_ENV", 3.14) == 3.14 + + def test_returns_default_when_blank(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_float_env + + monkeypatch.setenv("TEST_FLOAT_ENV", " ") + assert _load_float_env("TEST_FLOAT_ENV", 3.14) == 3.14 + + def test_parses_valid_float(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_float_env + + monkeypatch.setenv("TEST_FLOAT_ENV", "2.5") + assert _load_float_env("TEST_FLOAT_ENV", 1.0) == 2.5 + + def test_raises_on_invalid(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_float_env + + monkeypatch.setenv("TEST_FLOAT_ENV", "not-a-float") + with pytest.raises(ValueError, match="not a number"): + _load_float_env("TEST_FLOAT_ENV", 1.0) + + +class TestLoadIntEnv: + """``_load_int_env`` reads integer env vars with a default.""" + + def test_returns_default_when_unset(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_int_env + + monkeypatch.delenv("TEST_INT_ENV", raising=False) + assert _load_int_env("TEST_INT_ENV", 42) == 42 + + def test_returns_default_when_blank(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_int_env + + monkeypatch.setenv("TEST_INT_ENV", " ") + assert _load_int_env("TEST_INT_ENV", 42) == 42 + + def test_parses_valid_int(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_int_env + + monkeypatch.setenv("TEST_INT_ENV", "10") + assert _load_int_env("TEST_INT_ENV", 1) == 10 + + def test_raises_on_invalid(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_int_env + + monkeypatch.setenv("TEST_INT_ENV", "not-an-int") + with pytest.raises(ValueError, match="not an integer"): + _load_int_env("TEST_INT_ENV", 1) + + +class TestValidateConnectionConfig: + """``_validate_connection_config`` rejects invalid tuning parameters.""" + + def _valid(self): + return { + "keepalive_expiry_seconds": 1.0, + "max_connections": 100, + "max_keepalive_connections": 20, + "client_pool_size": 1, + } + + def test_negative_keepalive_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "keepalive_expiry_seconds": -1.0} + with pytest.raises(ValueError, match="greater than or equal to 0"): + _validate_connection_config(**cfg) + + def test_zero_max_connections_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "max_connections": 0} + with pytest.raises(ValueError, match="greater than 0"): + _validate_connection_config(**cfg) + + def test_negative_max_keepalive_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "max_keepalive_connections": -1} + with pytest.raises(ValueError, match="greater than or equal to 0"): + _validate_connection_config(**cfg) + + def test_keepalive_exceeds_connections_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "max_connections": 5, "max_keepalive_connections": 10} + with pytest.raises(ValueError, match="less than or equal to"): + _validate_connection_config(**cfg) + + def test_zero_pool_size_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "client_pool_size": 0} + with pytest.raises(ValueError, match="greater than 0"): + _validate_connection_config(**cfg) + + +class TestScorerInvokeRequestValidation: + """``ScorerInvokeRequest`` rejects malformed input combos.""" + + def test_missing_scorer_id_raises(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeInputs, + ScorerInvokeRequest, + ) + from pydantic import ValidationError + + with pytest.raises(ValidationError, match="scorer_id"): + ScorerInvokeRequest(inputs=ScorerInvokeInputs(query="hello")) + + +def test_client_raises_when_no_api_secret(monkeypatch): + """The client requires GALILEO_API_SECRET_KEY or GALILEO_API_SECRET.""" + for name in ("GALILEO_API_SECRET_KEY", "GALILEO_API_SECRET"): + monkeypatch.delenv(name, raising=False) + monkeypatch.setenv("GALILEO_LUNA_INVOKE_URL", "http://luna-invoke:8090") + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + with pytest.raises(ValueError, match="GALILEO_API_SECRET_KEY or GALILEO_API_SECRET"): + GalileoLLMClient() + + +def test_client_raises_when_no_llm_invoke_url(monkeypatch): + """The client requires GALILEO_LUNA_INVOKE_URL.""" + monkeypatch.setenv("GALILEO_API_SECRET_KEY", "test-secret") + monkeypatch.delenv("GALILEO_LUNA_INVOKE_URL", raising=False) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + with pytest.raises(ValueError, match="GALILEO_LUNA_INVOKE_URL"): + GalileoLLMClient() + + +def test_client_jwt_has_internal_scope(monkeypatch): + """JWT produced by the client must carry internal=True and scope=scorers.invoke.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + _, auth_header = client._endpoint_and_auth_header() + + assert auth_header.startswith("Bearer ") + payload = _decode_jwt_payload(auth_header.removeprefix("Bearer ")) + assert payload["internal"] is True + assert payload["scope"] == "scorers.invoke" + + +def test_client_posts_to_correct_llm_invoke_endpoint(monkeypatch): + """_endpoint_and_auth_header must return the LLM invoke endpoint path.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + endpoint, _ = client._endpoint_and_auth_header() + + assert endpoint == "http://luna-invoke:8090/api/v1/scorers/invoke" + + +def test_client_does_not_use_old_api_paths(monkeypatch): + """The client must not reference /internal/scorers/invoke.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + endpoint, _ = client._endpoint_and_auth_header() + + assert "/scorers/invoke" in endpoint + assert endpoint.startswith("http://luna-invoke:8090/api/v1/") + assert "/internal/scorers/invoke" not in endpoint + + +@pytest.mark.asyncio +async def test_get_client_does_not_set_galileo_api_key_header(monkeypatch): + """The HTTP client must never include a Galileo-API-Key header.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + http_client = await client._get_client() + try: + assert "Galileo-API-Key" not in http_client.headers + assert "galileo-api-key" not in http_client.headers + finally: + await client.close() + + +@pytest.mark.asyncio +async def test_get_client_uses_configured_llm_invoke_ca_file(monkeypatch): + """The HTTP client should verify TLS with the configured CA.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("GALILEO_LUNA_INVOKE_CA_FILE", "/etc/galileo/llm-invoke-ca.crt") + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + ssl_context = object() + with ( + patch.object(GalileoLLMClient, "_load_ssl_context", return_value=ssl_context), + patch("httpx.AsyncClient") as async_client, + ): + client = GalileoLLMClient() + await client._get_client() + + assert client.llm_invoke_ca_file == "/etc/galileo/llm-invoke-ca.crt" + assert async_client.call_args.kwargs["verify"] is ssl_context + + +@pytest.mark.asyncio +async def test_get_client_falls_back_to_agent_control_auth_upstream_ca_file(monkeypatch): + """Galileo in-cluster pods already mount the internal CA for auth upstream.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + monkeypatch.delenv("GALILEO_LUNA_INVOKE_CA_FILE", raising=False) + monkeypatch.setenv( + "AGENT_CONTROL_AUTH_UPSTREAM_CA_FILE", "/etc/agent-control/auth-upstream-ca/ca.crt" + ) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + ssl_context = object() + with ( + patch.object(GalileoLLMClient, "_load_ssl_context", return_value=ssl_context), + patch("httpx.AsyncClient") as async_client, + ): + client = GalileoLLMClient() + await client._get_client() + + assert client.llm_invoke_ca_file == "/etc/agent-control/auth-upstream-ca/ca.crt" + assert async_client.call_args.kwargs["verify"] is ssl_context + + +@pytest.mark.asyncio +async def test_invoke_raises_when_response_is_not_a_json_object(monkeypatch): + """A non-object JSON body must surface as a clear RuntimeError.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + + fake_response = MagicMock() + fake_response.raise_for_status = MagicMock() + fake_response.json = MagicMock(return_value=["not", "an", "object"]) + + fake_http = AsyncMock() + fake_http.post = AsyncMock(return_value=fake_response) + fake_http.is_closed = False + client._client = fake_http + + try: + with pytest.raises(RuntimeError, match="not a JSON object"): + await client.invoke(scorer_id="scorer-123", input="hello") + finally: + await client.close() + + +@pytest.mark.asyncio +async def test_invoke_propagates_http_status_error(monkeypatch): + """The client logs and re-raises HTTP status errors.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + + fake_response = MagicMock(spec=httpx.Response) + fake_response.status_code = 500 + fake_response.text = "internal error" + fake_response.raise_for_status = MagicMock( + side_effect=httpx.HTTPStatusError( + "boom", request=MagicMock(spec=httpx.Request), response=fake_response + ) + ) + + fake_http = AsyncMock() + fake_http.post = AsyncMock(return_value=fake_response) + fake_http.is_closed = False + client._client = fake_http + + try: + with pytest.raises(httpx.HTTPStatusError): + await client.invoke(scorer_id="scorer-123", input="hello") + finally: + await client.close() + + +@pytest.mark.asyncio +async def test_invoke_propagates_request_error(monkeypatch): + """RequestError is logged and re-raised so callers can decide policy.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + + fake_http = AsyncMock() + fake_http.post = AsyncMock(side_effect=httpx.RequestError("network down")) + fake_http.is_closed = False + client._client = fake_http + + try: + with pytest.raises(httpx.RequestError): + await client.invoke(scorer_id="scorer-123", input="hello") + finally: + await client.close() + + +@pytest.mark.asyncio +async def test_client_async_context_manager_closes_on_exit(monkeypatch): + """Entering/exiting the async context manager must close the client.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + async with GalileoLLMClient() as client: + await client._get_client() + assert client._client is not None + + assert client._client is None + + +@pytest.mark.asyncio +async def test_invoke_strips_caller_supplied_galileo_api_key_header(monkeypatch): + """Regression: a Galileo-API-Key passed via the headers kwarg must be stripped.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + captured: dict[str, object] = {} + + def handler(request: httpx.Request) -> httpx.Response: + captured["headers"] = dict(request.headers) + return httpx.Response(200, json={"score": 0.9, "status": "success"}) + + client = GalileoLLMClient() + client._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + + try: + await client.invoke( + scorer_id="scorer-123", + input="hello", + headers={"Galileo-API-Key": "should-be-stripped", "X-Custom": "keep-me"}, + ) + finally: + await client.close() + + headers = captured["headers"] + assert isinstance(headers, dict) + assert "galileo-api-key" not in headers + assert headers.get("x-custom") == "keep-me" + + +@pytest.mark.asyncio +async def test_invoke_always_emits_config_field(monkeypatch): + """The request always carries a server timeout below its HTTP deadline.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + captured: dict[str, object] = {} + + def handler(request: httpx.Request) -> httpx.Response: + captured["body"] = json.loads(request.content.decode()) + return httpx.Response(200, json={"score": 0.5, "status": "success"}) + + client = GalileoLLMClient() + client._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + + try: + await client.invoke(scorer_id="scorer-123", input="hello") + finally: + await client.close() + + assert "config" in captured["body"] + assert captured["body"]["config"] == {"request_timeout_seconds": 8.0} + + +@pytest.mark.asyncio +async def test_client_pool_round_robin(monkeypatch): + """With pool_size > 1, the client rotates through multiple connections.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("GALILEO_LUNA_CLIENT_POOL_SIZE", "2") + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + assert client.client_pool_size == 2 + + first = await client._get_client() + second = await client._get_client() + third = await client._get_client() + + # With pool_size=2, two distinct clients exist and we cycle back + assert len(client._clients) == 2 + assert first is not second + assert third is first # wraps around + + await client.close() + assert client._clients == [] + + +# ============================================================================= +# llm/config.py: threshold validator branches +# ============================================================================= + + +class TestLlmEvaluatorConfigValidation: + """``LlmEvaluatorConfig.validate_threshold`` exercises all branches.""" + + def test_numeric_operator_with_non_numeric_threshold_raises(self): + from agent_control_evaluator_galileo.llm.config import LlmEvaluatorConfig + from pydantic import ValidationError + + with pytest.raises(ValidationError, match="numeric threshold"): + LlmEvaluatorConfig(scorer_id="scorer-123", operator="gt", threshold="not-a-number") + + def test_none_threshold_with_non_any_operator_raises(self): + from agent_control_evaluator_galileo.llm.config import LlmEvaluatorConfig + from pydantic import ValidationError + + with pytest.raises(ValidationError, match="threshold is required"): + LlmEvaluatorConfig(scorer_id="scorer-123", operator="eq", threshold=None) + + def test_any_operator_allows_none_threshold(self): + from agent_control_evaluator_galileo.llm.config import LlmEvaluatorConfig + + config = LlmEvaluatorConfig(scorer_id="scorer-123", operator="any", threshold=None) + assert config.operator == "any" + + def test_coerce_number_returns_none_for_list(self): + from agent_control_evaluator_galileo.llm.config import coerce_number + + assert coerce_number([1, 2, 3]) is None + + def test_coerce_number_returns_none_for_non_numeric_string(self): + from agent_control_evaluator_galileo.llm.config import coerce_number + + assert coerce_number("abc") is None + + def test_coerce_number_parses_numeric_string(self): + from agent_control_evaluator_galileo.llm.config import coerce_number + + assert coerce_number("0.75") == pytest.approx(0.75) + + +@pytest.mark.asyncio +async def test_evaluator_with_context_uses_step(monkeypatch): + """evaluate_with_context passes the step to _evaluate.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_models import Step + + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient, ScorerInvokeResponse + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + step = Step(type="llm", name="test", input="prompt", output="answer") + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate_with_context("selected input", step) + + assert result.matched is True + assert mock_invoke.call_args.kwargs["step"] == step + + +@pytest.mark.asyncio +async def test_evaluator_scorer_label_echoed_in_metadata(monkeypatch): + """scorer_label echoed from the response is included in result metadata.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient, ScorerInvokeResponse + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "scorer_label": "my-label"} + ) + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse( + scorer_label="echoed-label", score=0.8, status="success" + ) + result = await evaluator.evaluate("hello") + + assert result.metadata.get("scorer_label") == "echoed-label" + + +@pytest.mark.asyncio +async def test_evaluator_scorer_version_id_in_metadata(monkeypatch): + """scorer_version_id is included in metadata when configured.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient, ScorerInvokeResponse + + evaluator = LlmEvaluator.from_dict( + { + "scorer_id": "scorer-123", + "scorer_version_id": "ver-456", + "threshold": 0.5, + } + ) + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate("hello") + + assert result.metadata.get("requested_scorer_version_id") == "ver-456" + assert mock_invoke.call_args.kwargs["scorer_version_id"] == "ver-456" From b5bcd8772314bc97a733c60969f71f28730ec8ad Mon Sep 17 00:00:00 2001 From: Wrisa Date: Wed, 7 Oct 2026 14:29:42 -0700 Subject: [PATCH 07/10] removed legacy fixes which were similar to luna --- .../llm/client.py | 98 ++++++++++++------- .../llm/config.py | 13 +-- .../llm/evaluator.py | 2 + .../galileo/tests/test_llm_evaluator.py | 6 ++ 4 files changed, 73 insertions(+), 46 deletions(-) diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py index 396b0216..2b7d7126 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py @@ -11,13 +11,14 @@ from hmac import new as hmac_new from json import dumps from time import time -from typing import Literal, cast, get_args +from typing import Any, Literal, get_args from urllib.parse import urlsplit import httpx from agent_control_models import JSONObject, JSONValue, Step -from pydantic import BaseModel, Field, PrivateAttr, model_validator +from pydantic import BaseModel, ConfigDict, Field, PrivateAttr, model_validator +from ..records import UnsupportedStepTypeError, record_from_step from .config import ScorerInvokeConfig logger = logging.getLogger(__name__) @@ -57,6 +58,7 @@ "session", ] SUPPORTED_SCORER_INVOKE_RECORD_TYPES = frozenset(get_args(ScorerInvokeRecordType)) +_MISSING_SELECTED_DATA = object() def _b64url(data: bytes) -> str: @@ -219,6 +221,8 @@ class ScorerInvokeRecord(BaseModel): represented here. Orbit hydrates those fields from trusted server context. """ + model_config = ConfigDict(extra="allow") + type: ScorerInvokeRecordType name: str | None = None input: JSONValue = None @@ -242,8 +246,8 @@ class ScorerInvokeRequest(BaseModel): Attributes: scorer_id: Required scorer identifier. - scorer_version_id: Deprecated optional compatibility identifier. Orbit - currently invokes the scorer's current default version. + scorer_version_id: Optional. When absent, contextual evaluation sends + legacy inputs only and skips the structured record. scorer_label: Optional display/metadata label. inputs: Selected scorer input values. record: Optional Orbit-compatible structured runtime record. @@ -273,11 +277,13 @@ def to_dict(self) -> JSONObject: return request -def _orbit_record_from_step( +def _scorer_invoke_record_from_step( step: Step | None, *, selected_input: JSONValue, selected_output: JSONValue, + selected_data: Any = _MISSING_SELECTED_DATA, + payload_field: str = "input", ) -> ScorerInvokeRecord | None: """Translate a generic Agent Control step into Orbit's record contract. @@ -294,25 +300,37 @@ def _orbit_record_from_step( step: Complete Agent Control step, when contextual evaluation is used. selected_input: Selector-selected value sent as ``inputs.query``. selected_output: Selector-selected value sent as ``inputs.response``. + selected_data: Raw selector-selected value used to construct the record, + preserving its structure independently from serialized legacy inputs. + payload_field: Record input/output side for scalar selector values. Returns: An Orbit-compatible record, or ``None`` for absent/unsupported steps. """ - if step is None or step.type not in SUPPORTED_SCORER_INVOKE_RECORD_TYPES: + if step is None: return None - - # The membership check above narrows the runtime value to Orbit's known - # discriminator set, but static type checkers cannot infer that relationship. - record_type = cast(ScorerInvokeRecordType, step.type) - return ScorerInvokeRecord( - type=record_type, - name=step.name, - input=selected_input if selected_input is not None else step.input, - output=selected_output if selected_output is not None else step.output, - context=step.context, - tools=step.tools, - dataset_output=step.ground_truth, - ) + try: + if selected_data is _MISSING_SELECTED_DATA: + galileo_core_record = record_from_step( + step, + selected_input=selected_input, + selected_output=selected_output, + ) + else: + galileo_core_record = record_from_step( + step, + selected_data=selected_data, + payload_field=payload_field, + ) + except UnsupportedStepTypeError: + return None + # record_from_step() returns Galileo Core models (LlmSpan, Trace, etc.). Dumping + # and revalidating coerces the model into the LLM request shape while preserving + # extra canonical fields that Orbit accepts but ScorerInvokeRecord does not declare. + canonical_record = galileo_core_record.model_dump(mode="json", exclude_none=True) + if canonical_record.get("type") not in SUPPORTED_SCORER_INVOKE_RECORD_TYPES: + return None + return ScorerInvokeRecord.model_validate(canonical_record) class ScorerInvokeResponse(BaseModel): @@ -488,8 +506,10 @@ async def invoke( input: JSONValue = None, output: JSONValue = None, step: Step | None = None, + selected_data: Any = _MISSING_SELECTED_DATA, + selected_data_payload_field: str = "input", execution_context: GalileoExecutionContext | None = None, - config: ScorerInvokeConfig | JSONObject | None = None, + config: ScorerInvokeConfig | None = None, timeout: float = DEFAULT_TIMEOUT_SECS, headers: dict[str, str] | None = None, ) -> ScorerInvokeResponse: @@ -497,14 +517,17 @@ async def invoke( Args: scorer_id: Required scorer identifier. - scorer_version_id: Deprecated optional compatibility identifier. Orbit - currently invokes the scorer's current default version. + scorer_version_id: Optional. When absent, contextual evaluation sends + legacy inputs only and skips the structured record. scorer_label: Optional display/metadata label. input: Optional user/system prompt text. output: Optional model response text. step: Optional complete runtime step used for structured dual-write. + selected_data: Raw selector-selected value used to construct the record, + preserving its structure independently from serialized legacy inputs. + selected_data_payload_field: Record input/output side for scalar selector values. execution_context: Optional authenticated Galileo scorer context. - config: Optional Orbit-supported scorer invocation configuration. + config: Optional scorer invocation configuration. timeout: Request timeout in seconds. headers: Additional request headers. @@ -512,8 +535,8 @@ async def invoke( Parsed scorer invocation response. Raises: - ValueError: If neither input nor output is provided, config contains - a field Orbit does not support, or the timeout ordering is invalid. + ValueError: If neither input nor output is provided, or the timeout + ordering is invalid. RuntimeError: If the API response is not a JSON object. httpx.HTTPStatusError: If the invoke endpoint returns an error status code. httpx.RequestError: If the request fails before a response is received. @@ -521,17 +544,22 @@ async def invoke( if not (_has_value(input) or _has_value(output)): raise ValueError("At least one of input or output must be provided.") - # Accept dictionaries for source compatibility with the original client, - # but validate them against Orbit's authoritative allowlist locally. - invoke_config = ( - ScorerInvokeConfig.model_validate(config) - if config is not None - else ScorerInvokeConfig() - ) + invoke_config = config if config is not None else ScorerInvokeConfig() invoke_config = _effective_scorer_timeout( invoke_config, http_timeout_seconds=timeout, ) + record = ( + _scorer_invoke_record_from_step( + step, + selected_input=input, + selected_output=output, + selected_data=selected_data, + payload_field=selected_data_payload_field, + ) + if step is not None and scorer_version_id is not None + else None + ) request_body = ScorerInvokeRequest( scorer_id=scorer_id, scorer_version_id=scorer_version_id, @@ -542,11 +570,7 @@ async def invoke( ground_truth=step.ground_truth if step is not None else None, tools=step.tools if step is not None else None, ), - record=_orbit_record_from_step( - step, - selected_input=input, - selected_output=output, - ), + record=record, execution_context=execution_context, config=invoke_config, ).to_dict() diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py index 1b207f3a..ed1a8a26 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py @@ -22,9 +22,6 @@ class ScorerInvokeConfig(BaseModel): less actionable HTTP 422 response from Runners. Attributes: - threshold: Legacy threshold accepted by Orbit. Agent Control still - applies its evaluator threshold locally. - score_threshold: Legacy score threshold accepted by Orbit. request_timeout_seconds: Optional upper bound for scorer execution in Orbit. The Agent Control HTTP and evaluator deadlines must remain longer than this value. @@ -32,8 +29,6 @@ class ScorerInvokeConfig(BaseModel): model_config = ConfigDict(extra="forbid") - threshold: float | None = None - score_threshold: float | None = None request_timeout_seconds: float | None = Field(default=None, gt=0) @@ -56,8 +51,8 @@ class LlmEvaluatorConfig(EvaluatorConfig): Attributes: scorer_id: Required scorer identifier for LLM scorer invocation. - scorer_version_id: Deprecated optional compatibility identifier. Orbit - currently invokes the scorer's current default version. + scorer_version_id: Optional. When absent, contextual evaluation sends + legacy inputs only and skips the structured record. scorer_label: Optional display/metadata label. threshold: Local threshold used by the evaluator for comparison. operator: Local comparison operator. Numeric operators use threshold as a number. @@ -75,8 +70,8 @@ class LlmEvaluatorConfig(EvaluatorConfig): default=None, min_length=1, description=( - "Deprecated optional compatibility identifier. Orbit currently invokes " - "the scorer's current default version." + "Optional. When absent, contextual evaluation sends legacy inputs only " + "and skips the structured record." ), ) scorer_label: str | None = Field( diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py index 6c12ddd1..4a94969c 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py @@ -281,6 +281,8 @@ async def _evaluate( scorer_kwargs = self._scorer_kwargs() if step is not None: scorer_kwargs["step"] = step + scorer_kwargs["selected_data"] = data + scorer_kwargs["selected_data_payload_field"] = self.config.payload_field if execution_context is not None: scorer_kwargs["execution_context"] = execution_context response = await self._get_client().invoke( diff --git a/evaluators/contrib/galileo/tests/test_llm_evaluator.py b/evaluators/contrib/galileo/tests/test_llm_evaluator.py index 814d9de3..907d1292 100644 --- a/evaluators/contrib/galileo/tests/test_llm_evaluator.py +++ b/evaluators/contrib/galileo/tests/test_llm_evaluator.py @@ -73,6 +73,8 @@ async def test_evaluator_consumes_verified_opaque_extensions(self, target_type: mock_invoke.assert_awaited_once_with( scorer_id="scorer-123", step=Step(type="llm", name="answer", input="prompt"), + selected_data="selected input", + selected_data_payload_field="input", execution_context=GalileoExecutionContext( organization_id="org-1", user_id="verified-user-2", @@ -113,6 +115,8 @@ async def test_evaluator_allows_missing_caller_id(self) -> None: mock_invoke.assert_awaited_once_with( scorer_id="scorer-123", step=Step(type="llm", name="answer", input="prompt"), + selected_data="selected input", + selected_data_payload_field="input", execution_context=GalileoExecutionContext( organization_id="org-1", user_id=None, @@ -154,6 +158,8 @@ async def test_evaluator_omits_run_id_for_unsupported_target_type(self) -> None: mock_invoke.assert_awaited_once_with( scorer_id="scorer-123", step=Step(type="llm", name="answer", input="prompt"), + selected_data="selected input", + selected_data_payload_field="input", execution_context=GalileoExecutionContext( organization_id="org-1", user_id="verified-user-2", From 5300544509a3ae1e8b6104b11eecb46a51e6cb22 Mon Sep 17 00:00:00 2001 From: Wrisa Date: Wed, 7 Oct 2026 17:47:41 -0700 Subject: [PATCH 08/10] feature flag and request runner's only when step and execution_context present --- .../llm/client.py | 19 +++--- .../llm/config.py | 12 ++++ .../llm/evaluator.py | 44 ++++++++++---- .../galileo/tests/test_llm_coverage_gaps.py | 60 +++++++++++++++---- .../galileo/tests/test_llm_evaluator.py | 1 + 5 files changed, 108 insertions(+), 28 deletions(-) diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py index 2b7d7126..859d1c61 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py @@ -522,11 +522,12 @@ async def invoke( scorer_label: Optional display/metadata label. input: Optional user/system prompt text. output: Optional model response text. - step: Optional complete runtime step used for structured dual-write. + step: Required runtime step; invocation is rejected without it. selected_data: Raw selector-selected value used to construct the record, preserving its structure independently from serialized legacy inputs. selected_data_payload_field: Record input/output side for scalar selector values. - execution_context: Optional authenticated Galileo scorer context. + execution_context: Required authenticated Galileo scorer context; Orbit uses + it to fetch LLM credentials. config: Optional scorer invocation configuration. timeout: Request timeout in seconds. headers: Additional request headers. @@ -535,12 +536,16 @@ async def invoke( Parsed scorer invocation response. Raises: - ValueError: If neither input nor output is provided, or the timeout - ordering is invalid. + ValueError: If step or execution_context are absent, neither input nor + output is provided, or the timeout ordering is invalid. RuntimeError: If the API response is not a JSON object. httpx.HTTPStatusError: If the invoke endpoint returns an error status code. httpx.RequestError: If the request fails before a response is received. """ + if step is None: + raise ValueError("LLM scorer invocation requires a runtime step.") + if execution_context is None: + raise ValueError("LLM scorer invocation requires an authenticated execution context.") if not (_has_value(input) or _has_value(output)): raise ValueError("At least one of input or output must be provided.") @@ -557,7 +562,7 @@ async def invoke( selected_data=selected_data, payload_field=selected_data_payload_field, ) - if step is not None and scorer_version_id is not None + if scorer_version_id is not None else None ) request_body = ScorerInvokeRequest( @@ -567,8 +572,8 @@ async def invoke( inputs=ScorerInvokeInputs( query="" if input is None else input, response="" if output is None else output, - ground_truth=step.ground_truth if step is not None else None, - tools=step.tools if step is not None else None, + ground_truth=step.ground_truth, + tools=step.tools, ), record=record, execution_context=execution_context, diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py index ed1a8a26..f42ceefe 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py @@ -2,6 +2,7 @@ from __future__ import annotations +import os from typing import Literal from agent_control_evaluators import EvaluatorConfig @@ -12,6 +13,17 @@ LlmPayloadField = Literal["input", "output"] _NUMERIC_OPERATORS = frozenset({"gt", "gte", "lt", "lte"}) +LLM_INVOKE_RUNTIME_FLAG_ENV = "GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME" + + +def llm_invoke_runtime_enabled() -> bool: + """Return whether the LLM scorer invoke runtime is enabled. + + The LLM evaluator requires Orbit support for ``execution_context`` to fetch + LLM credentials. Set ``GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME=enabled`` once + the Orbit-side support is confirmed ready. + """ + return os.getenv(LLM_INVOKE_RUNTIME_FLAG_ENV) == "enabled" class ScorerInvokeConfig(BaseModel): diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py index 4a94969c..332d1da2 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py @@ -13,7 +13,7 @@ from agent_control_models import EvaluatorResult, JSONObject, JSONValue, Step from .client import GalileoExecutionContext, GalileoLLMClient, ScorerInvokeResponse -from .config import LlmEvaluatorConfig, coerce_number +from .config import LlmEvaluatorConfig, coerce_number, llm_invoke_runtime_enabled logger = logging.getLogger(__name__) @@ -117,8 +117,13 @@ class LlmEvaluator(Evaluator[LlmEvaluatorConfig]): @classmethod def is_available(cls) -> bool: - """Check whether required runtime dependencies are available.""" - return LLM_AVAILABLE + """Return True only when the LLM invoke runtime feature flag is enabled. + + The LLM evaluator requires Orbit-side support for ``execution_context`` to + fetch LLM credentials. It is unavailable until + ``GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME=enabled`` is set. + """ + return LLM_AVAILABLE and llm_invoke_runtime_enabled() def __init__(self, config: LlmEvaluatorConfig) -> None: """Initialize the direct LLM-as-judge evaluator. @@ -267,6 +272,28 @@ async def _evaluate( extensions: JSONObject | None = None, ) -> EvaluatorResult: """Run an LLM scorer evaluation with optional structured runtime context.""" + if step is None: + return EvaluatorResult( + matched=False, + confidence=0.0, + message="LLM scorer requires a runtime step; evaluation skipped.", + metadata=self._base_metadata(), + error="LLM scorer requires a runtime step; evaluation skipped.", + ) + + try: + execution_context = self._execution_context_from_extensions(extensions) + except ValueError as exc: + return self._handle_error(exc) + if execution_context is None: + return EvaluatorResult( + matched=False, + confidence=0.0, + message="LLM scorer requires authenticated execution context; evaluation skipped.", + metadata=self._base_metadata(), + error="LLM scorer requires authenticated execution context; evaluation skipped.", + ) + input_text, output_text = self._prepare_payload(data) if not (_has_text(input_text) or _has_text(output_text)): return EvaluatorResult( @@ -277,14 +304,11 @@ async def _evaluate( ) try: - execution_context = self._execution_context_from_extensions(extensions) scorer_kwargs = self._scorer_kwargs() - if step is not None: - scorer_kwargs["step"] = step - scorer_kwargs["selected_data"] = data - scorer_kwargs["selected_data_payload_field"] = self.config.payload_field - if execution_context is not None: - scorer_kwargs["execution_context"] = execution_context + scorer_kwargs["step"] = step + scorer_kwargs["selected_data"] = data + scorer_kwargs["selected_data_payload_field"] = self.config.payload_field + scorer_kwargs["execution_context"] = execution_context response = await self._get_client().invoke( **scorer_kwargs, input=input_text if _has_text(input_text) else None, diff --git a/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py b/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py index e24b67de..d611784c 100644 --- a/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py +++ b/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py @@ -17,8 +17,29 @@ LLM_ENV = { "GALILEO_API_SECRET_KEY": "test-secret", "GALILEO_LUNA_INVOKE_URL": "http://luna-invoke:8090", + "GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME": "enabled", } +_EXTENSIONS = { + "namespace_key": "org-1", + "caller_id": "user-1", + "target_type": "log_stream", + "target_id": "run-1", + "metadata": {}, +} + + +def _make_invoke_kwargs() -> dict[str, object]: + """Minimal required kwargs for GalileoLLMClient.invoke().""" + from agent_control_models import Step + + from agent_control_evaluator_galileo.llm.client import GalileoExecutionContext + + return { + "step": Step(type="llm", name="test", input="prompt"), + "execution_context": GalileoExecutionContext(organization_id="org-1"), + } + def _decode_jwt_payload(token: str) -> dict[str, object]: payload_segment = token.split(".")[1] @@ -289,12 +310,15 @@ async def test_evaluator_handles_non_success_status(monkeypatch): """A non-success status from the scorer must surface as an error result.""" for key, value in LLM_ENV.items(): monkeypatch.setenv(key, value) + from agent_control_models import Step + from agent_control_evaluator_galileo.llm import LlmEvaluator, ScorerInvokeResponse from agent_control_evaluator_galileo.llm.client import GalileoLLMClient evaluator = LlmEvaluator.from_dict( {"scorer_id": "scorer-123", "threshold": 0.5, "operator": "gte"} ) + step = Step(type="llm", name="test", input="prompt") with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: mock_invoke.return_value = ScorerInvokeResponse( @@ -304,7 +328,7 @@ async def test_evaluator_handles_non_success_status(monkeypatch): error_message="upstream timeout", ) - result = await evaluator.evaluate("hello") + result = await evaluator.evaluate_with_extensions("hello", step, _EXTENSIONS) assert result.matched is False assert result.error is not None @@ -316,13 +340,16 @@ async def test_evaluator_skips_empty_data(monkeypatch): """Evaluator skips invocation when there is no text to score.""" for key, value in LLM_ENV.items(): monkeypatch.setenv(key, value) + from agent_control_models import Step + from agent_control_evaluator_galileo.llm import LlmEvaluator from agent_control_evaluator_galileo.llm.client import GalileoLLMClient evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + step = Step(type="llm", name="test", input="prompt") with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: - result = await evaluator.evaluate(None) + result = await evaluator.evaluate_with_extensions(None, step, _EXTENSIONS) assert result.matched is False assert result.error is None @@ -335,10 +362,13 @@ async def test_evaluator_returns_error_result_on_http_error(monkeypatch): """HTTP errors must surface as a non-matched EvaluatorResult with metadata.""" for key, value in LLM_ENV.items(): monkeypatch.setenv(key, value) + from agent_control_models import Step + from agent_control_evaluator_galileo.llm import LlmEvaluator from agent_control_evaluator_galileo.llm.client import GalileoLLMClient evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + step = Step(type="llm", name="test", input="prompt") fake_request = httpx.Request("POST", "http://luna-invoke:8090/api/v1/scorers/invoke") fake_response = httpx.Response(500, text="internal server error", request=fake_request) @@ -349,7 +379,7 @@ async def test_evaluator_returns_error_result_on_http_error(monkeypatch): new_callable=AsyncMock, side_effect=httpx.HTTPStatusError("500", request=fake_request, response=fake_response), ): - result = await evaluator.evaluate("hello") + result = await evaluator.evaluate_with_extensions("hello", step, _EXTENSIONS) assert result.matched is False assert result.error is not None @@ -755,7 +785,7 @@ async def test_invoke_raises_when_response_is_not_a_json_object(monkeypatch): try: with pytest.raises(RuntimeError, match="not a JSON object"): - await client.invoke(scorer_id="scorer-123", input="hello") + await client.invoke(scorer_id="scorer-123", input="hello", **_make_invoke_kwargs()) finally: await client.close() @@ -785,7 +815,7 @@ async def test_invoke_propagates_http_status_error(monkeypatch): try: with pytest.raises(httpx.HTTPStatusError): - await client.invoke(scorer_id="scorer-123", input="hello") + await client.invoke(scorer_id="scorer-123", input="hello", **_make_invoke_kwargs()) finally: await client.close() @@ -806,7 +836,7 @@ async def test_invoke_propagates_request_error(monkeypatch): try: with pytest.raises(httpx.RequestError): - await client.invoke(scorer_id="scorer-123", input="hello") + await client.invoke(scorer_id="scorer-123", input="hello", **_make_invoke_kwargs()) finally: await client.close() @@ -846,6 +876,7 @@ def handler(request: httpx.Request) -> httpx.Response: scorer_id="scorer-123", input="hello", headers={"Galileo-API-Key": "should-be-stripped", "X-Custom": "keep-me"}, + **_make_invoke_kwargs(), ) finally: await client.close() @@ -873,7 +904,7 @@ def handler(request: httpx.Request) -> httpx.Response: client._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) try: - await client.invoke(scorer_id="scorer-123", input="hello") + await client.invoke(scorer_id="scorer-123", input="hello", **_make_invoke_kwargs()) finally: await client.close() @@ -951,7 +982,7 @@ def test_coerce_number_parses_numeric_string(self): @pytest.mark.asyncio async def test_evaluator_with_context_uses_step(monkeypatch): - """evaluate_with_context passes the step to _evaluate.""" + """evaluate_with_extensions passes the step and execution_context to invoke.""" for key, value in LLM_ENV.items(): monkeypatch.setenv(key, value) from agent_control_models import Step @@ -964,10 +995,11 @@ async def test_evaluator_with_context_uses_step(monkeypatch): with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") - result = await evaluator.evaluate_with_context("selected input", step) + result = await evaluator.evaluate_with_extensions("selected input", step, _EXTENSIONS) assert result.matched is True assert mock_invoke.call_args.kwargs["step"] == step + assert mock_invoke.call_args.kwargs["execution_context"] is not None @pytest.mark.asyncio @@ -975,18 +1007,21 @@ async def test_evaluator_scorer_label_echoed_in_metadata(monkeypatch): """scorer_label echoed from the response is included in result metadata.""" for key, value in LLM_ENV.items(): monkeypatch.setenv(key, value) + from agent_control_models import Step + from agent_control_evaluator_galileo.llm import LlmEvaluator from agent_control_evaluator_galileo.llm.client import GalileoLLMClient, ScorerInvokeResponse evaluator = LlmEvaluator.from_dict( {"scorer_id": "scorer-123", "threshold": 0.5, "scorer_label": "my-label"} ) + step = Step(type="llm", name="test", input="prompt") with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: mock_invoke.return_value = ScorerInvokeResponse( scorer_label="echoed-label", score=0.8, status="success" ) - result = await evaluator.evaluate("hello") + result = await evaluator.evaluate_with_extensions("hello", step, _EXTENSIONS) assert result.metadata.get("scorer_label") == "echoed-label" @@ -996,6 +1031,8 @@ async def test_evaluator_scorer_version_id_in_metadata(monkeypatch): """scorer_version_id is included in metadata when configured.""" for key, value in LLM_ENV.items(): monkeypatch.setenv(key, value) + from agent_control_models import Step + from agent_control_evaluator_galileo.llm import LlmEvaluator from agent_control_evaluator_galileo.llm.client import GalileoLLMClient, ScorerInvokeResponse @@ -1006,10 +1043,11 @@ async def test_evaluator_scorer_version_id_in_metadata(monkeypatch): "threshold": 0.5, } ) + step = Step(type="llm", name="test", input="prompt") with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") - result = await evaluator.evaluate("hello") + result = await evaluator.evaluate_with_extensions("hello", step, _EXTENSIONS) assert result.metadata.get("requested_scorer_version_id") == "ver-456" assert mock_invoke.call_args.kwargs["scorer_version_id"] == "ver-456" diff --git a/evaluators/contrib/galileo/tests/test_llm_evaluator.py b/evaluators/contrib/galileo/tests/test_llm_evaluator.py index 907d1292..5136b1fc 100644 --- a/evaluators/contrib/galileo/tests/test_llm_evaluator.py +++ b/evaluators/contrib/galileo/tests/test_llm_evaluator.py @@ -11,6 +11,7 @@ LLM_ENV = { "GALILEO_API_SECRET_KEY": "test-secret", "GALILEO_LUNA_INVOKE_URL": "http://luna-invoke:8090", + "GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME": "enabled", } From 4ef26638f22367ce4d9877cdbf4b7d5ec22207a7 Mon Sep 17 00:00:00 2001 From: Wrisa Date: Wed, 7 Oct 2026 18:15:38 -0700 Subject: [PATCH 09/10] added documentation --- .../galileo/src/agent_control_evaluator_galileo/llm/client.py | 2 ++ sdks/python/src/agent_control/evaluators/__init__.py | 4 ++++ 2 files changed, 6 insertions(+) diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py index 859d1c61..0e8bea2b 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py @@ -369,6 +369,8 @@ class GalileoLLMClient: """Thin HTTP client for Galileo LLM-as-judge scorer invocation. Environment Variables: + GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME: Set to ``enabled`` to activate the evaluator. + Required; the evaluator is unavailable until this flag is set. GALILEO_API_SECRET_KEY or GALILEO_API_SECRET: JWT signing secret for internal auth. GALILEO_LUNA_INVOKE_URL: LLM scorer invoke URL or service root (required). GALILEO_LUNA_INVOKE_CA_FILE: CA bundle used to verify invoke TLS. diff --git a/sdks/python/src/agent_control/evaluators/__init__.py b/sdks/python/src/agent_control/evaluators/__init__.py index d239821e..5edaaebe 100644 --- a/sdks/python/src/agent_control/evaluators/__init__.py +++ b/sdks/python/src/agent_control/evaluators/__init__.py @@ -16,6 +16,10 @@ from agent_control.evaluators import LunaEvaluator, LunaEvaluatorConfig # if galileo installed from agent_control.evaluators import LlmEvaluator, LlmEvaluatorConfig # if galileo installed ``` + + Note: ``galileo.llm`` requires ``GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME=enabled`` to be set. + The evaluator is unavailable (``is_available()`` returns ``False``) until this flag is present, + as it depends on Orbit-side support for ``execution_context`` to fetch LLM credentials. """ from agent_control_engine import ( From 52024b0bc2a64efbfa571e6f45a69e0a351b7f55 Mon Sep 17 00:00:00 2001 From: Wrisa Date: Wed, 7 Oct 2026 19:14:42 -0700 Subject: [PATCH 10/10] Create override for testing in devstacks --- .../src/agent_control_evaluator_galileo/llm/client.py | 2 +- .../src/agent_control_evaluator_galileo/llm/config.py | 8 ++++---- .../src/agent_control_evaluator_galileo/llm/evaluator.py | 6 +++--- .../contrib/galileo/tests/test_llm_coverage_gaps.py | 2 +- evaluators/contrib/galileo/tests/test_llm_evaluator.py | 2 +- sdks/python/src/agent_control/evaluators/__init__.py | 2 +- 6 files changed, 11 insertions(+), 11 deletions(-) diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py index 0e8bea2b..f31e6a84 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py @@ -369,7 +369,7 @@ class GalileoLLMClient: """Thin HTTP client for Galileo LLM-as-judge scorer invocation. Environment Variables: - GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME: Set to ``enabled`` to activate the evaluator. + GALILEO_FEATURE_FLAG_LLM_EVALUATOR: Set to ``enabled`` to activate the evaluator. Required; the evaluator is unavailable until this flag is set. GALILEO_API_SECRET_KEY or GALILEO_API_SECRET: JWT signing secret for internal auth. GALILEO_LUNA_INVOKE_URL: LLM scorer invoke URL or service root (required). diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py index f42ceefe..ff9d7093 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py @@ -13,17 +13,17 @@ LlmPayloadField = Literal["input", "output"] _NUMERIC_OPERATORS = frozenset({"gt", "gte", "lt", "lte"}) -LLM_INVOKE_RUNTIME_FLAG_ENV = "GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME" +LLM_EVALUATOR_FLAG_ENV = "GALILEO_FEATURE_FLAG_LLM_EVALUATOR" -def llm_invoke_runtime_enabled() -> bool: +def llm_evaluator_enabled() -> bool: """Return whether the LLM scorer invoke runtime is enabled. The LLM evaluator requires Orbit support for ``execution_context`` to fetch - LLM credentials. Set ``GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME=enabled`` once + LLM credentials. Set ``GALILEO_FEATURE_FLAG_LLM_EVALUATOR=enabled`` once the Orbit-side support is confirmed ready. """ - return os.getenv(LLM_INVOKE_RUNTIME_FLAG_ENV) == "enabled" + return os.getenv(LLM_EVALUATOR_FLAG_ENV) == "enabled" class ScorerInvokeConfig(BaseModel): diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py index 332d1da2..c66de908 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py @@ -13,7 +13,7 @@ from agent_control_models import EvaluatorResult, JSONObject, JSONValue, Step from .client import GalileoExecutionContext, GalileoLLMClient, ScorerInvokeResponse -from .config import LlmEvaluatorConfig, coerce_number, llm_invoke_runtime_enabled +from .config import LlmEvaluatorConfig, coerce_number, llm_evaluator_enabled logger = logging.getLogger(__name__) @@ -121,9 +121,9 @@ def is_available(cls) -> bool: The LLM evaluator requires Orbit-side support for ``execution_context`` to fetch LLM credentials. It is unavailable until - ``GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME=enabled`` is set. + ``GALILEO_FEATURE_FLAG_LLM_EVALUATOR=enabled`` is set. """ - return LLM_AVAILABLE and llm_invoke_runtime_enabled() + return LLM_AVAILABLE and llm_evaluator_enabled() def __init__(self, config: LlmEvaluatorConfig) -> None: """Initialize the direct LLM-as-judge evaluator. diff --git a/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py b/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py index d611784c..c44c119b 100644 --- a/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py +++ b/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py @@ -17,7 +17,7 @@ LLM_ENV = { "GALILEO_API_SECRET_KEY": "test-secret", "GALILEO_LUNA_INVOKE_URL": "http://luna-invoke:8090", - "GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME": "enabled", + "GALILEO_FEATURE_FLAG_LLM_EVALUATOR": "enabled", } _EXTENSIONS = { diff --git a/evaluators/contrib/galileo/tests/test_llm_evaluator.py b/evaluators/contrib/galileo/tests/test_llm_evaluator.py index 5136b1fc..e7d093b3 100644 --- a/evaluators/contrib/galileo/tests/test_llm_evaluator.py +++ b/evaluators/contrib/galileo/tests/test_llm_evaluator.py @@ -11,7 +11,7 @@ LLM_ENV = { "GALILEO_API_SECRET_KEY": "test-secret", "GALILEO_LUNA_INVOKE_URL": "http://luna-invoke:8090", - "GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME": "enabled", + "GALILEO_FEATURE_FLAG_LLM_EVALUATOR": "enabled", } diff --git a/sdks/python/src/agent_control/evaluators/__init__.py b/sdks/python/src/agent_control/evaluators/__init__.py index 5edaaebe..33e648b7 100644 --- a/sdks/python/src/agent_control/evaluators/__init__.py +++ b/sdks/python/src/agent_control/evaluators/__init__.py @@ -17,7 +17,7 @@ from agent_control.evaluators import LlmEvaluator, LlmEvaluatorConfig # if galileo installed ``` - Note: ``galileo.llm`` requires ``GALILEO_FEATURE_FLAG_LLM_INVOKE_RUNTIME=enabled`` to be set. + Note: ``galileo.llm`` requires ``GALILEO_FEATURE_FLAG_LLM_EVALUATOR=enabled`` to be set. The evaluator is unavailable (``is_available()`` returns ``False``) until this flag is present, as it depends on Orbit-side support for ``execution_context`` to fetch LLM credentials. """