diff --git a/evaluators/contrib/README.md b/evaluators/contrib/README.md index 91beb9b9..4b852070 100644 --- a/evaluators/contrib/README.md +++ b/evaluators/contrib/README.md @@ -2,7 +2,7 @@ Contributed evaluators and templates for extending Agent Control. -- `galileo/` — Luna evaluator integration +- `galileo/` — Galileo evaluator integrations (Luna, LLM) - `template/` — Starter template for adding new evaluators Full guide: https://docs.agentcontrol.dev/concepts/evaluators/custom-evaluators diff --git a/evaluators/contrib/galileo/pyproject.toml b/evaluators/contrib/galileo/pyproject.toml index 7b212b96..7909b197 100644 --- a/evaluators/contrib/galileo/pyproject.toml +++ b/evaluators/contrib/galileo/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "agent-control-evaluator-galileo" version = "8.11.0" -description = "Galileo Luna evaluator for agent-control" +description = "Galileo evaluators (Luna, LLM-as-judge) for agent-control" readme = "README.md" requires-python = ">=3.12" license = { text = "Apache-2.0" } @@ -25,6 +25,7 @@ dev = [ [project.entry-points."agent_control.evaluators"] "galileo.luna" = "agent_control_evaluator_galileo.luna:LunaEvaluator" +"galileo.llm" = "agent_control_evaluator_galileo.llm:LlmEvaluator" [build-system] requires = ["hatchling"] diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py index 0183ad97..74f09225 100644 --- a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/__init__.py @@ -4,6 +4,7 @@ Available evaluators: - galileo.luna: Galileo Luna direct scorer evaluation + - galileo.llm: Galileo LLM-as-judge direct scorer evaluation Installation: pip install agent-control-evaluator-galileo @@ -19,6 +20,13 @@ except PackageNotFoundError: __version__ = "0.0.0.dev" +from agent_control_evaluator_galileo.llm import ( + LLM_AVAILABLE, + GalileoLLMClient, + LlmEvaluator, + LlmEvaluatorConfig, + LlmOperator, +) from agent_control_evaluator_galileo.luna import ( LUNA_AVAILABLE, GalileoLunaClient, @@ -41,6 +49,7 @@ ) __all__ = [ + # luna "GalileoLunaClient", "ScorerInvokeRequest", "ScorerInvokeConfig", @@ -50,6 +59,13 @@ "LunaEvaluatorConfig", "LunaOperator", "LUNA_AVAILABLE", + # llm + "GalileoLLMClient", + "LlmEvaluator", + "LlmEvaluatorConfig", + "LlmOperator", + "LLM_AVAILABLE", + # records "GalileoRecord", "GalileoRecordNormalizer", "RecordFactoryError", diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py new file mode 100644 index 00000000..7d78b162 --- /dev/null +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/__init__.py @@ -0,0 +1,30 @@ +"""Galileo LLM-as-judge direct scorer evaluator.""" + +from agent_control_evaluator_galileo.llm.client import ( + GalileoExecutionContext, + GalileoLLMClient, + ScorerInvokeInputs, + ScorerInvokeRecord, + ScorerInvokeRequest, + ScorerInvokeResponse, +) +from agent_control_evaluator_galileo.llm.config import ( + LlmEvaluatorConfig, + LlmOperator, + ScorerInvokeConfig, +) +from agent_control_evaluator_galileo.llm.evaluator import LLM_AVAILABLE, LlmEvaluator + +__all__ = [ + "GalileoLLMClient", + "GalileoExecutionContext", + "ScorerInvokeInputs", + "ScorerInvokeConfig", + "ScorerInvokeRecord", + "ScorerInvokeRequest", + "ScorerInvokeResponse", + "LlmEvaluatorConfig", + "LlmOperator", + "LlmEvaluator", + "LLM_AVAILABLE", +] diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py new file mode 100644 index 00000000..f31e6a84 --- /dev/null +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/client.py @@ -0,0 +1,647 @@ +"""Direct HTTP client for Galileo LLM-as-judge scorer invocation.""" + +from __future__ import annotations + +import logging +import os +import ssl +from asyncio import Lock +from base64 import urlsafe_b64encode +from hashlib import sha256 +from hmac import new as hmac_new +from json import dumps +from time import time +from typing import Any, Literal, get_args +from urllib.parse import urlsplit + +import httpx +from agent_control_models import JSONObject, JSONValue, Step +from pydantic import BaseModel, ConfigDict, Field, PrivateAttr, model_validator + +from ..records import UnsupportedStepTypeError, record_from_step +from .config import ScorerInvokeConfig + +logger = logging.getLogger(__name__) + +DEFAULT_TIMEOUT_SECS = 10.0 +SERVER_TIMEOUT_RATIO = 0.8 +DEFAULT_INTERNAL_TOKEN_TTL_SECS = 3600 +DEFAULT_LLM_SCORER_INVOKE_PATH = "/api/v1/scorers/invoke" +LLM_INVOKE_URL_ENV = "GALILEO_LUNA_INVOKE_URL" +LLM_INVOKE_CA_FILE_ENV = "GALILEO_LUNA_INVOKE_CA_FILE" +AUTH_UPSTREAM_CA_FILE_ENV = "AGENT_CONTROL_AUTH_UPSTREAM_CA_FILE" + +# Headers that must never be forwarded to the scorer invoke endpoint (checked case-insensitively). +_BLOCKED_REQUEST_HEADERS = frozenset({"galileo-api-key"}) + +# Keep pooled-connection reuse shorter than typical server keepalive/worker +# recycle windows so requests do not pick up sockets the server already closed. +DEFAULT_KEEPALIVE_EXPIRY_SECS = 1.0 +DEFAULT_MAX_CONNECTIONS = 100 +DEFAULT_MAX_KEEPALIVE_CONNECTIONS = 20 +DEFAULT_CLIENT_POOL_SIZE = 1 +LLM_KEEPALIVE_EXPIRY_ENV = "GALILEO_LUNA_KEEPALIVE_EXPIRY_SECONDS" +LLM_MAX_CONNECTIONS_ENV = "GALILEO_LUNA_MAX_CONNECTIONS" +LLM_MAX_KEEPALIVE_CONNECTIONS_ENV = "GALILEO_LUNA_MAX_KEEPALIVE_CONNECTIONS" +LLM_CLIENT_POOL_SIZE_ENV = "GALILEO_LUNA_CLIENT_POOL_SIZE" + +# These values mirror Orbit's StepType discriminator. Agent Control's generic +# Step remains extensible; only the Galileo transport boundary is constrained. +ScorerInvokeRecordType = Literal[ + "llm", + "retriever", + "tool", + "workflow", + "agent", + "control", + "trace", + "session", +] +SUPPORTED_SCORER_INVOKE_RECORD_TYPES = frozenset(get_args(ScorerInvokeRecordType)) +_MISSING_SELECTED_DATA = object() + + +def _b64url(data: bytes) -> str: + return urlsafe_b64encode(data).rstrip(b"=").decode("ascii") + + +def _internal_auth_token( + api_secret: str, + ttl_seconds: int = DEFAULT_INTERNAL_TOKEN_TTL_SECS, +) -> str: + """Create the internal JWT expected by LLM scorer invoke routes.""" + now = int(time()) + header = {"alg": "HS256", "typ": "JWT"} + payload = { + "internal": True, + "scope": "scorers.invoke", + "iat": now, + "exp": now + ttl_seconds, + } + signing_input = ".".join( + [ + _b64url(dumps(header, separators=(",", ":")).encode("utf-8")), + _b64url(dumps(payload, separators=(",", ":")).encode("utf-8")), + ] + ) + signature = hmac_new(api_secret.encode("utf-8"), signing_input.encode("ascii"), sha256).digest() + return f"{signing_input}.{_b64url(signature)}" + + +def _normalize_llm_invoke_url(raw_url: str) -> str: + """Use full invoke URLs as-is and append the default path to bare service roots.""" + url = raw_url.strip().rstrip("/") + parsed = urlsplit(url) + if parsed.path not in ("", "/") or parsed.query or parsed.fragment: + return url + return f"{url}{DEFAULT_LLM_SCORER_INVOKE_PATH}" + + +def _load_float_env(env_name: str, default: float) -> float: + raw = os.getenv(env_name) + if raw is None or raw.strip() == "": + return default + try: + return float(raw) + except ValueError as exc: + raise ValueError(f"{env_name}={raw!r} is not a number.") from exc + + +def _load_int_env(env_name: str, default: int) -> int: + raw = os.getenv(env_name) + if raw is None or raw.strip() == "": + return default + try: + return int(raw) + except ValueError as exc: + raise ValueError(f"{env_name}={raw!r} is not an integer.") from exc + + +def _validate_connection_config( + *, + keepalive_expiry_seconds: float, + max_connections: int, + max_keepalive_connections: int, + client_pool_size: int, +) -> None: + if keepalive_expiry_seconds < 0: + raise ValueError( + f"{LLM_KEEPALIVE_EXPIRY_ENV}={keepalive_expiry_seconds} " + "must be greater than or equal to 0." + ) + if max_connections <= 0: + raise ValueError(f"{LLM_MAX_CONNECTIONS_ENV}={max_connections} must be greater than 0.") + if max_keepalive_connections < 0: + raise ValueError( + f"{LLM_MAX_KEEPALIVE_CONNECTIONS_ENV}={max_keepalive_connections} " + "must be greater than or equal to 0." + ) + if max_keepalive_connections > max_connections: + raise ValueError( + f"{LLM_MAX_KEEPALIVE_CONNECTIONS_ENV}={max_keepalive_connections} " + f"must be less than or equal to {LLM_MAX_CONNECTIONS_ENV}={max_connections}." + ) + if client_pool_size <= 0: + raise ValueError(f"{LLM_CLIENT_POOL_SIZE_ENV}={client_pool_size} must be greater than 0.") + + +def _as_float_or_none(value: JSONValue) -> float | None: + if isinstance(value, bool) or value is None: + return None + if isinstance(value, (int, float)): + return float(value) + if isinstance(value, str): + try: + return float(value) + except ValueError: + return None + return None + + +def _has_value(value: JSONValue) -> bool: + if value is None: + return False + if isinstance(value, str): + return value.strip() != "" + if isinstance(value, (list, dict)): + return len(value) > 0 + return True + + +def _effective_scorer_timeout( + config: ScorerInvokeConfig, + *, + http_timeout_seconds: float, +) -> ScorerInvokeConfig: + """Resolve an Orbit execution timeout that expires before the HTTP request. + + The server execution budget defaults to 80% of the caller's HTTP deadline, + leaving time for Orbit to serialize and return the result. Explicit caller + overrides are preserved only when they maintain the same ordering. + + Args: + config: Caller-provided scorer-invoke configuration. + http_timeout_seconds: Agent Control's HTTP request deadline in seconds. + + Returns: + A scorer-invoke configuration with an effective execution timeout. + + Raises: + ValueError: If either deadline is invalid or the explicit server timeout + is not shorter than the HTTP deadline. + """ + if http_timeout_seconds <= 0: + raise ValueError("HTTP timeout must be greater than 0 seconds.") + + server_timeout = config.request_timeout_seconds + if server_timeout is None: + server_timeout = http_timeout_seconds * SERVER_TIMEOUT_RATIO + elif server_timeout >= http_timeout_seconds: + raise ValueError( + "config.request_timeout_seconds must be shorter than the HTTP " + f"timeout ({http_timeout_seconds:g} seconds)." + ) + + return config.model_copy(update={"request_timeout_seconds": server_timeout}) + + +class ScorerInvokeInputs(BaseModel): + """Input values sent to the LLM scorer invoke endpoint.""" + + query: JSONValue = "" + response: JSONValue = "" + ground_truth: JSONValue = None + tools: list[JSONObject] | None = None + + +class ScorerInvokeRecord(BaseModel): + """Caller-controlled subset of Orbit's partial runtime-record contract. + + Identity, ownership, persistence, and execution IDs are intentionally not + represented here. Orbit hydrates those fields from trusted server context. + """ + + model_config = ConfigDict(extra="allow") + + type: ScorerInvokeRecordType + name: str | None = None + input: JSONValue = None + output: JSONValue = None + context: JSONObject | None = None + tools: list[JSONObject] | None = None + dataset_output: JSONValue = None + + +class GalileoExecutionContext(BaseModel): + """Authenticated organization and optional identity fields for Galileo scorer invocation.""" + + organization_id: str = Field(min_length=1) + user_id: str | None = Field(default=None, min_length=1) + project_id: str | None = Field(default=None, min_length=1) + run_id: str | None = Field(default=None, min_length=1) + + +class ScorerInvokeRequest(BaseModel): + """Request payload for LLM scorer invocation. + + Attributes: + scorer_id: Required scorer identifier. + scorer_version_id: Optional. When absent, contextual evaluation sends + legacy inputs only and skips the structured record. + scorer_label: Optional display/metadata label. + inputs: Selected scorer input values. + record: Optional Orbit-compatible structured runtime record. + execution_context: Optional authenticated execution context. + config: Scorer-specific configuration, always emitted. + """ + + scorer_id: str = Field(min_length=1) + scorer_version_id: str | None = Field(default=None, min_length=1) + scorer_label: str | None = Field(default=None, min_length=1) + inputs: ScorerInvokeInputs + record: ScorerInvokeRecord | None = None + execution_context: GalileoExecutionContext | None = None + config: ScorerInvokeConfig = Field(default_factory=ScorerInvokeConfig) + + @model_validator(mode="after") + def ensure_required_values(self) -> ScorerInvokeRequest: + if not (_has_value(self.inputs.query) or _has_value(self.inputs.response)): + raise ValueError("Either inputs.query or inputs.response must be set.") + return self + + def to_dict(self) -> JSONObject: + """Convert to the LLM scorer invoke request shape.""" + request = self.model_dump(mode="json", exclude_none=True) + if self.execution_context is not None: + request["execution_context"] = self.execution_context.model_dump(mode="json") + return request + + +def _scorer_invoke_record_from_step( + step: Step | None, + *, + selected_input: JSONValue, + selected_output: JSONValue, + selected_data: Any = _MISSING_SELECTED_DATA, + payload_field: str = "input", +) -> ScorerInvokeRecord | None: + """Translate a generic Agent Control step into Orbit's record contract. + + Selector-selected values remain the primary scorer input. When a selector + supplies one side, that value is written to both the legacy and structured + representations so Orbit's conflict validation cannot observe two meanings. + The complete step supplies the unselected side and additional record context. + + Unknown Agent Control step types intentionally fall back to the legacy + ``inputs`` contract. This keeps the open-source Step model extensible without + sending an invalid discriminator to Orbit. + + Args: + step: Complete Agent Control step, when contextual evaluation is used. + selected_input: Selector-selected value sent as ``inputs.query``. + selected_output: Selector-selected value sent as ``inputs.response``. + selected_data: Raw selector-selected value used to construct the record, + preserving its structure independently from serialized legacy inputs. + payload_field: Record input/output side for scalar selector values. + + Returns: + An Orbit-compatible record, or ``None`` for absent/unsupported steps. + """ + if step is None: + return None + try: + if selected_data is _MISSING_SELECTED_DATA: + galileo_core_record = record_from_step( + step, + selected_input=selected_input, + selected_output=selected_output, + ) + else: + galileo_core_record = record_from_step( + step, + selected_data=selected_data, + payload_field=payload_field, + ) + except UnsupportedStepTypeError: + return None + # record_from_step() returns Galileo Core models (LlmSpan, Trace, etc.). Dumping + # and revalidating coerces the model into the LLM request shape while preserving + # extra canonical fields that Orbit accepts but ScorerInvokeRecord does not declare. + canonical_record = galileo_core_record.model_dump(mode="json", exclude_none=True) + if canonical_record.get("type") not in SUPPORTED_SCORER_INVOKE_RECORD_TYPES: + return None + return ScorerInvokeRecord.model_validate(canonical_record) + + +class ScorerInvokeResponse(BaseModel): + """Response from LLM scorer invocation. + + Attributes: + scorer_label: Echoed scorer label, when returned. + score: Raw scorer value. + status: Invocation status. + execution_time: Execution time in seconds, when returned. + error_message: Error detail for non-success statuses. + """ + + scorer_label: str | None = None + score: JSONValue + status: str = "unknown" + execution_time: float | None = None + error_message: str | None = None + _raw_response: JSONObject = PrivateAttr(default_factory=dict) + + @property + def raw_response(self) -> JSONObject: + return self._raw_response + + @classmethod + def from_dict(cls, data: JSONObject) -> ScorerInvokeResponse: + """Create a response model from the LLM scorer invoke JSON object.""" + response = cls.model_validate( + data | {"execution_time": _as_float_or_none(data.get("execution_time"))} + ) + response._raw_response = data + return response + + +class GalileoLLMClient: + """Thin HTTP client for Galileo LLM-as-judge scorer invocation. + + Environment Variables: + GALILEO_FEATURE_FLAG_LLM_EVALUATOR: Set to ``enabled`` to activate the evaluator. + Required; the evaluator is unavailable until this flag is set. + GALILEO_API_SECRET_KEY or GALILEO_API_SECRET: JWT signing secret for internal auth. + GALILEO_LUNA_INVOKE_URL: LLM scorer invoke URL or service root (required). + GALILEO_LUNA_INVOKE_CA_FILE: CA bundle used to verify invoke TLS. + AGENT_CONTROL_AUTH_UPSTREAM_CA_FILE: Shared internal CA fallback. + GALILEO_LUNA_KEEPALIVE_EXPIRY_SECONDS: HTTP pooled connection expiry. + GALILEO_LUNA_MAX_CONNECTIONS: Maximum outbound HTTP connections. + GALILEO_LUNA_MAX_KEEPALIVE_CONNECTIONS: Maximum idle pooled HTTP connections. + GALILEO_LUNA_CLIENT_POOL_SIZE: Number of outbound HTTP clients to rotate across. + """ + + def __init__( + self, + api_secret: str | None = None, + llm_invoke_url: str | None = None, + llm_invoke_ca_file: str | None = None, + ) -> None: + """Initialize the Galileo LLM client. + + Args: + api_secret: Internal JWT signing secret. If not provided, reads from + GALILEO_API_SECRET_KEY or GALILEO_API_SECRET. + llm_invoke_url: LLM scorer invoke URL or service root. If not provided, + reads from GALILEO_LUNA_INVOKE_URL. + llm_invoke_ca_file: Optional CA bundle used to verify invoke TLS. If not + provided, reads from GALILEO_LUNA_INVOKE_CA_FILE, then + AGENT_CONTROL_AUTH_UPSTREAM_CA_FILE. + + Raises: + ValueError: If the API secret, invoke URL, CA bundle, or connection + tuning configuration is invalid. + """ + resolved_api_secret = ( + api_secret or os.getenv("GALILEO_API_SECRET_KEY") or os.getenv("GALILEO_API_SECRET") + ) + if not resolved_api_secret: + raise ValueError( + "GALILEO_API_SECRET_KEY or GALILEO_API_SECRET is required for LLM " + "scorer invocation. Set one as an environment variable or pass it " + "to the constructor." + ) + + resolved_llm_invoke_url = llm_invoke_url or os.getenv(LLM_INVOKE_URL_ENV) + if resolved_llm_invoke_url is None or resolved_llm_invoke_url.strip() == "": + raise ValueError( + "GALILEO_LUNA_INVOKE_URL is required for LLM scorer invocation. " + "Set it as an environment variable or pass it to the constructor." + ) + + self.api_secret = resolved_api_secret + self.llm_invoke_url = _normalize_llm_invoke_url(resolved_llm_invoke_url) + self.llm_invoke_ca_file = ( + llm_invoke_ca_file + or os.getenv(LLM_INVOKE_CA_FILE_ENV) + or os.getenv(AUTH_UPSTREAM_CA_FILE_ENV) + or "" + ).strip() or None + self._ssl_context = self._load_ssl_context(self.llm_invoke_ca_file) + self.keepalive_expiry_seconds = _load_float_env( + LLM_KEEPALIVE_EXPIRY_ENV, DEFAULT_KEEPALIVE_EXPIRY_SECS + ) + self.max_connections = _load_int_env(LLM_MAX_CONNECTIONS_ENV, DEFAULT_MAX_CONNECTIONS) + self.max_keepalive_connections = _load_int_env( + LLM_MAX_KEEPALIVE_CONNECTIONS_ENV, DEFAULT_MAX_KEEPALIVE_CONNECTIONS + ) + self.client_pool_size = _load_int_env(LLM_CLIENT_POOL_SIZE_ENV, DEFAULT_CLIENT_POOL_SIZE) + _validate_connection_config( + keepalive_expiry_seconds=self.keepalive_expiry_seconds, + max_connections=self.max_connections, + max_keepalive_connections=self.max_keepalive_connections, + client_pool_size=self.client_pool_size, + ) + self._client: httpx.AsyncClient | None = None + self._clients: list[httpx.AsyncClient] = [] + self._next_client_index = 0 + self._client_lock = Lock() + + @staticmethod + def _load_ssl_context(ca_file: str | None) -> ssl.SSLContext | None: + """Build a TLS verification context from a CA bundle path, if configured.""" + if ca_file is None: + return None + try: + return ssl.create_default_context(cafile=ca_file) + except (OSError, ssl.SSLError) as exc: + raise ValueError(f"Failed to load CA bundle from {ca_file!r}: {exc}") from exc + + def _create_client(self) -> httpx.AsyncClient: + """Create an HTTP client with the configured TLS and connection limits.""" + verify: ssl.SSLContext | bool = self._ssl_context if self._ssl_context is not None else True + return httpx.AsyncClient( + headers={"Content-Type": "application/json"}, + timeout=httpx.Timeout(DEFAULT_TIMEOUT_SECS), + limits=httpx.Limits( + max_connections=self.max_connections, + max_keepalive_connections=self.max_keepalive_connections, + keepalive_expiry=self.keepalive_expiry_seconds, + ), + verify=verify, + ) + + def _select_pooled_client(self) -> httpx.AsyncClient: + """Select the next pooled client while holding the client state lock.""" + client = self._clients[self._next_client_index % len(self._clients)] + self._next_client_index = (self._next_client_index + 1) % len(self._clients) + return client + + async def _get_client(self) -> httpx.AsyncClient: + """Get or create the next HTTP client.""" + async with self._client_lock: + self._clients = [client for client in self._clients if not client.is_closed] + + if self.client_pool_size == 1: + if self._client is not None and not self._client.is_closed: + return self._client + self._client = self._clients[0] if self._clients else self._create_client() + self._clients = [self._client] + return self._client + + self._client = None + while len(self._clients) < self.client_pool_size: + self._clients.append(self._create_client()) + + return self._select_pooled_client() + + def _endpoint_and_auth_header(self) -> tuple[str, str]: + token = _internal_auth_token(self.api_secret) + return self.llm_invoke_url, f"Bearer {token}" + + async def invoke( + self, + *, + scorer_id: str, + scorer_version_id: str | None = None, + scorer_label: str | None = None, + input: JSONValue = None, + output: JSONValue = None, + step: Step | None = None, + selected_data: Any = _MISSING_SELECTED_DATA, + selected_data_payload_field: str = "input", + execution_context: GalileoExecutionContext | None = None, + config: ScorerInvokeConfig | None = None, + timeout: float = DEFAULT_TIMEOUT_SECS, + headers: dict[str, str] | None = None, + ) -> ScorerInvokeResponse: + """Invoke a Galileo LLM-as-judge scorer. + + Args: + scorer_id: Required scorer identifier. + scorer_version_id: Optional. When absent, contextual evaluation sends + legacy inputs only and skips the structured record. + scorer_label: Optional display/metadata label. + input: Optional user/system prompt text. + output: Optional model response text. + step: Required runtime step; invocation is rejected without it. + selected_data: Raw selector-selected value used to construct the record, + preserving its structure independently from serialized legacy inputs. + selected_data_payload_field: Record input/output side for scalar selector values. + execution_context: Required authenticated Galileo scorer context; Orbit uses + it to fetch LLM credentials. + config: Optional scorer invocation configuration. + timeout: Request timeout in seconds. + headers: Additional request headers. + + Returns: + Parsed scorer invocation response. + + Raises: + ValueError: If step or execution_context are absent, neither input nor + output is provided, or the timeout ordering is invalid. + RuntimeError: If the API response is not a JSON object. + httpx.HTTPStatusError: If the invoke endpoint returns an error status code. + httpx.RequestError: If the request fails before a response is received. + """ + if step is None: + raise ValueError("LLM scorer invocation requires a runtime step.") + if execution_context is None: + raise ValueError("LLM scorer invocation requires an authenticated execution context.") + if not (_has_value(input) or _has_value(output)): + raise ValueError("At least one of input or output must be provided.") + + invoke_config = config if config is not None else ScorerInvokeConfig() + invoke_config = _effective_scorer_timeout( + invoke_config, + http_timeout_seconds=timeout, + ) + record = ( + _scorer_invoke_record_from_step( + step, + selected_input=input, + selected_output=output, + selected_data=selected_data, + payload_field=selected_data_payload_field, + ) + if scorer_version_id is not None + else None + ) + request_body = ScorerInvokeRequest( + scorer_id=scorer_id, + scorer_version_id=scorer_version_id, + scorer_label=scorer_label, + inputs=ScorerInvokeInputs( + query="" if input is None else input, + response="" if output is None else output, + ground_truth=step.ground_truth, + tools=step.tools, + ), + record=record, + execution_context=execution_context, + config=invoke_config, + ).to_dict() + + endpoint, auth_header = self._endpoint_and_auth_header() + request_headers = { + k: v for k, v in (headers or {}).items() if k.lower() not in _BLOCKED_REQUEST_HEADERS + } + request_headers["Authorization"] = auth_header + + logger.debug("[GalileoLLMClient] POST %s", endpoint) + logger.debug("[GalileoLLMClient] Request body: %s", request_body) + + try: + client = await self._get_client() + response = await client.post( + endpoint, + json=request_body, + headers=request_headers, + timeout=timeout, + ) + response.raise_for_status() + response_data = response.json() + if not isinstance(response_data, dict): + raise RuntimeError("Invalid response payload: not a JSON object") + + parsed = ScorerInvokeResponse.from_dict(response_data) + logger.debug("[GalileoLLMClient] Response: %s", parsed.raw_response) + return parsed + except httpx.HTTPStatusError as exc: + logger.error( + "[GalileoLLMClient] API error: %s - %s", + exc.response.status_code, + exc.response.text, + ) + raise + except httpx.RequestError as exc: + logger.error("[GalileoLLMClient] Request failed: %s", exc) + raise + + async def close(self) -> None: + """Close HTTP clients and release resources.""" + async with self._client_lock: + clients: list[httpx.AsyncClient] = [] + seen_client_ids: set[int] = set() + if self._client is not None: + clients.append(self._client) + seen_client_ids.add(id(self._client)) + self._client = None + for client in self._clients: + if id(client) not in seen_client_ids: + clients.append(client) + seen_client_ids.add(id(client)) + self._clients = [] + self._next_client_index = 0 + + for client in clients: + if not client.is_closed: + await client.aclose() + + async def __aenter__(self) -> GalileoLLMClient: + """Async context manager entry.""" + return self + + async def __aexit__(self, exc_type: object, exc_val: object, exc_tb: object) -> None: + """Async context manager exit.""" + await self.close() diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py new file mode 100644 index 00000000..ff9d7093 --- /dev/null +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/config.py @@ -0,0 +1,131 @@ +"""Configuration model for direct Galileo LLM-as-judge scorer evaluation.""" + +from __future__ import annotations + +import os +from typing import Literal + +from agent_control_evaluators import EvaluatorConfig +from agent_control_models import JSONValue +from pydantic import BaseModel, ConfigDict, Field, model_validator + +LlmOperator = Literal["gt", "gte", "lt", "lte", "eq", "ne", "contains", "any"] +LlmPayloadField = Literal["input", "output"] + +_NUMERIC_OPERATORS = frozenset({"gt", "gte", "lt", "lte"}) +LLM_EVALUATOR_FLAG_ENV = "GALILEO_FEATURE_FLAG_LLM_EVALUATOR" + + +def llm_evaluator_enabled() -> bool: + """Return whether the LLM scorer invoke runtime is enabled. + + The LLM evaluator requires Orbit support for ``execution_context`` to fetch + LLM credentials. Set ``GALILEO_FEATURE_FLAG_LLM_EVALUATOR=enabled`` once + the Orbit-side support is confirmed ready. + """ + return os.getenv(LLM_EVALUATOR_FLAG_ENV) == "enabled" + + +class ScorerInvokeConfig(BaseModel): + """Orbit-supported overrides for a synchronous scorer invocation. + + Orbit owns the Galileo scorer-invoke wire contract. Keeping this model + strict makes an unsupported option fail locally instead of producing a + less actionable HTTP 422 response from Runners. + + Attributes: + request_timeout_seconds: Optional upper bound for scorer execution in + Orbit. The Agent Control HTTP and evaluator deadlines must remain + longer than this value. + """ + + model_config = ConfigDict(extra="forbid") + + request_timeout_seconds: float | None = Field(default=None, gt=0) + + +def coerce_number(value: JSONValue) -> float | None: + """Return a numeric value for JSON scalars that can be compared numerically.""" + if isinstance(value, bool) or value is None: + return None + if isinstance(value, (int, float)): + return float(value) + if isinstance(value, str): + try: + return float(value) + except ValueError: + return None + return None + + +class LlmEvaluatorConfig(EvaluatorConfig): + """Configuration for direct LLM-as-judge scorer evaluation. + + Attributes: + scorer_id: Required scorer identifier for LLM scorer invocation. + scorer_version_id: Optional. When absent, contextual evaluation sends + legacy inputs only and skips the structured record. + scorer_label: Optional display/metadata label. + threshold: Local threshold used by the evaluator for comparison. + operator: Local comparison operator. Numeric operators use threshold as a number. + scorer_config: Optional Orbit-supported scorer invocation config sent + as ``config``. + payload_field: Explicit scorer input side for scalar selected data. + timeout_ms: Request timeout in milliseconds. + """ + + scorer_id: str = Field( + min_length=1, + description="Required scorer identifier for LLM scorer invocation.", + ) + scorer_version_id: str | None = Field( + default=None, + min_length=1, + description=( + "Optional. When absent, contextual evaluation sends legacy inputs only " + "and skips the structured record." + ), + ) + scorer_label: str | None = Field( + default=None, + min_length=1, + description="Optional display/metadata label.", + ) + threshold: JSONValue = Field( + default=0.5, + description="Local threshold used to decide whether the control matches.", + ) + operator: LlmOperator = Field( + default="gte", + description="Local comparison operator applied to the raw LLM scorer score.", + ) + scorer_config: ScorerInvokeConfig | None = Field( + default=None, + alias="config", + serialization_alias="config", + description=( + "Optional Orbit-supported configuration sent to the LLM scorer invoke endpoint." + ), + ) + payload_field: LlmPayloadField = Field( + default="input", + description=( + "Which scorer input side to use when selector output is a scalar value. " + "Structured selected data with input/output keys overrides this setting." + ), + ) + timeout_ms: int = Field( + default=10000, + ge=1000, + le=60000, + description="Request timeout in milliseconds (1-60 seconds)", + ) + + @model_validator(mode="after") + def validate_threshold(self) -> LlmEvaluatorConfig: + """Validate threshold compatibility with the configured operator.""" + if self.operator in _NUMERIC_OPERATORS and coerce_number(self.threshold) is None: + raise ValueError(f"operator '{self.operator}' requires a numeric threshold") + if self.operator != "any" and self.threshold is None: + raise ValueError("threshold is required unless operator is 'any'") + return self diff --git a/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py new file mode 100644 index 00000000..c66de908 --- /dev/null +++ b/evaluators/contrib/galileo/src/agent_control_evaluator_galileo/llm/evaluator.py @@ -0,0 +1,401 @@ +"""Direct Galileo LLM-as-judge evaluator implementation.""" + +from __future__ import annotations + +import json +import logging +import os +from importlib.metadata import PackageNotFoundError, version +from typing import Any + +import httpx +from agent_control_evaluators import Evaluator, EvaluatorMetadata, register_evaluator +from agent_control_models import EvaluatorResult, JSONObject, JSONValue, Step + +from .client import GalileoExecutionContext, GalileoLLMClient, ScorerInvokeResponse +from .config import LlmEvaluatorConfig, coerce_number, llm_evaluator_enabled + +logger = logging.getLogger(__name__) + + +def _resolve_package_version() -> str: + """Return the installed package version, or a dev fallback during local imports.""" + try: + return version("agent-control-evaluator-galileo") + except PackageNotFoundError: + return "0.0.0.dev" + + +_PACKAGE_VERSION = _resolve_package_version() +LLM_AVAILABLE = True +_HTTP_ERROR_BODY_LIMIT = 500 + + +def _coerce_payload_text(value: Any) -> str | None: + """Coerce selected data into scorer text without losing structured values.""" + if value is None: + return None + if isinstance(value, str): + return value + if isinstance(value, (int, float, bool)): + return str(value) + try: + return json.dumps(value, ensure_ascii=False, sort_keys=True, default=str) + except TypeError: + return str(value) + + +def _has_text(value: str | None) -> bool: + return value is not None and value.strip() != "" + + +def _extract_dict_text(data: dict[str, Any], key: str) -> str | None: + if key not in data: + return None + return _coerce_payload_text(data.get(key)) + + +def _contains(score: JSONValue, threshold: JSONValue) -> bool: + if threshold is None: + return False + if isinstance(score, str): + return str(threshold) in score + if isinstance(score, list): + return threshold in score + if isinstance(score, dict): + return threshold in score.values() + return False + + +def _confidence_from_score(score: JSONValue) -> float: + if isinstance(score, bool): + return 1.0 if score else 0.0 + number = coerce_number(score) + if number is not None and 0.0 <= number <= 1.0: + return number + return 1.0 + + +def _truncated_http_response_body(body: str) -> tuple[str, bool]: + if len(body) <= _HTTP_ERROR_BODY_LIMIT: + return body, False + return body[:_HTTP_ERROR_BODY_LIMIT], True + + +def _http_status_error_metadata(error: httpx.HTTPStatusError) -> dict[str, Any]: + metadata: dict[str, Any] = {} + + request = error.request + metadata["http_method"] = request.method + metadata["http_endpoint_path"] = request.url.path + + response = error.response + metadata["http_status_code"] = response.status_code + metadata["http_response_content_type"] = response.headers.get("content-type") + + body = response.text + if body: + metadata["http_response_body"], metadata["http_response_body_truncated"] = ( + _truncated_http_response_body(body) + ) + + return {key: value for key, value in metadata.items() if value is not None} + + +@register_evaluator +class LlmEvaluator(Evaluator[LlmEvaluatorConfig]): + """Galileo LLM-as-judge evaluator using the direct scorer invocation API.""" + + metadata = EvaluatorMetadata( + name="galileo.llm", + version=_PACKAGE_VERSION, + description="Galileo LLM-as-judge direct scorer evaluation", + requires_api_key=True, + timeout_ms=10000, + ) + config_model = LlmEvaluatorConfig + + @classmethod + def is_available(cls) -> bool: + """Return True only when the LLM invoke runtime feature flag is enabled. + + The LLM evaluator requires Orbit-side support for ``execution_context`` to + fetch LLM credentials. It is unavailable until + ``GALILEO_FEATURE_FLAG_LLM_EVALUATOR=enabled`` is set. + """ + return LLM_AVAILABLE and llm_evaluator_enabled() + + def __init__(self, config: LlmEvaluatorConfig) -> None: + """Initialize the direct LLM-as-judge evaluator. + + Args: + config: Validated LlmEvaluatorConfig instance. + + Raises: + ValueError: If neither GALILEO_API_SECRET_KEY nor GALILEO_API_SECRET is set. + """ + has_secret = os.getenv("GALILEO_API_SECRET_KEY") or os.getenv("GALILEO_API_SECRET") + if not has_secret: + raise ValueError( + "GALILEO_API_SECRET_KEY or GALILEO_API_SECRET is required for LLM " + "scorer invocation. Set one as an environment variable before using " + "galileo.llm." + ) + + super().__init__(config) + self._client = GalileoLLMClient() + + def _get_client(self) -> GalileoLLMClient: + """Get the Galileo LLM client.""" + return self._client + + def _prepare_payload(self, data: Any) -> tuple[str | None, str | None]: + """Prepare scorer input/output fields from selected data.""" + if isinstance(data, dict): + input_text = _extract_dict_text(data, "input") + output_text = _extract_dict_text(data, "output") + if _has_text(input_text) or _has_text(output_text): + return input_text, output_text + + text = _coerce_payload_text(data) + if self.config.payload_field == "output": + return None, text + return text, None + + def _score_matches(self, score: JSONValue) -> bool: + """Apply the configured local threshold comparison to a raw LLM scorer score.""" + operator = self.config.operator + threshold = self.config.threshold + + if operator == "any": + return bool(score) + if operator == "eq": + return score == threshold + if operator == "ne": + return score != threshold + if operator == "contains": + return _contains(score, threshold) + + score_number = coerce_number(score) + threshold_number = coerce_number(threshold) + if score_number is None: + raise ValueError(f"LLM scorer score {score!r} is not numeric") + if threshold_number is None: + raise ValueError(f"LLM scorer threshold {threshold!r} is not numeric") + + if operator == "gt": + return score_number > threshold_number + if operator == "gte": + return score_number >= threshold_number + if operator == "lt": + return score_number < threshold_number + if operator == "lte": + return score_number <= threshold_number + + raise ValueError(f"Unsupported LLM scorer operator: {operator}") + + async def evaluate(self, data: Any) -> EvaluatorResult: + """Evaluate selected data with Galileo LLM-as-judge direct scorer invocation. + + Args: + data: The data selected from the runtime step. + + Returns: + EvaluatorResult with local threshold decision and scorer metadata. + """ + return await self._evaluate(data, step=None) + + async def evaluate_with_context(self, data: Any, step: Step) -> EvaluatorResult: + """Evaluate selected data while dual-writing the complete runtime step. + + Args: + data: Data selected by the configured control selector. + step: Complete runtime step for structured scorer context. + + Returns: + EvaluatorResult with local threshold decision and scorer metadata. + """ + return await self._evaluate(data, step=step) + + async def evaluate_with_extensions( + self, + data: Any, + step: Step, + extensions: JSONObject | None, + ) -> EvaluatorResult: + """Evaluate using opaque authenticated metadata supplied by Agent Control.""" + return await self._evaluate( + data, + step=step, + extensions=extensions, + ) + + @staticmethod + def _execution_context_from_extensions( + extensions: JSONObject | None, + ) -> GalileoExecutionContext | None: + """Translate trusted opaque auth metadata into Galileo's request context.""" + if extensions is None: + return None + + metadata = extensions.get("metadata") + metadata_obj = metadata if isinstance(metadata, dict) else {} + organization_id = extensions.get("namespace_key") + # The runtime envelope's caller_id is supplied by the authenticated + # principal and is the caller identity Galileo associates with this run. + user_id = extensions.get("caller_id") + project_id = metadata_obj.get("project_id") + target_type = extensions.get("target_type") + target_id = extensions.get("target_id") + run_id = ( + target_id + if target_type in ("log_stream", "agent_stream") + and isinstance(target_id, str) + and target_id + else None + ) + if not isinstance(organization_id, str) or not organization_id: + raise ValueError("Authenticated execution metadata is missing organization_id") + + return GalileoExecutionContext( + organization_id=organization_id, + user_id=user_id if isinstance(user_id, str) and user_id else None, + project_id=project_id if isinstance(project_id, str) and project_id else None, + run_id=run_id, + ) + + async def _evaluate( + self, + data: Any, + *, + step: Step | None, + extensions: JSONObject | None = None, + ) -> EvaluatorResult: + """Run an LLM scorer evaluation with optional structured runtime context.""" + if step is None: + return EvaluatorResult( + matched=False, + confidence=0.0, + message="LLM scorer requires a runtime step; evaluation skipped.", + metadata=self._base_metadata(), + error="LLM scorer requires a runtime step; evaluation skipped.", + ) + + try: + execution_context = self._execution_context_from_extensions(extensions) + except ValueError as exc: + return self._handle_error(exc) + if execution_context is None: + return EvaluatorResult( + matched=False, + confidence=0.0, + message="LLM scorer requires authenticated execution context; evaluation skipped.", + metadata=self._base_metadata(), + error="LLM scorer requires authenticated execution context; evaluation skipped.", + ) + + input_text, output_text = self._prepare_payload(data) + if not (_has_text(input_text) or _has_text(output_text)): + return EvaluatorResult( + matched=False, + confidence=1.0, + message="No data to score with LLM scorer", + metadata=self._base_metadata(), + ) + + try: + scorer_kwargs = self._scorer_kwargs() + scorer_kwargs["step"] = step + scorer_kwargs["selected_data"] = data + scorer_kwargs["selected_data_payload_field"] = self.config.payload_field + scorer_kwargs["execution_context"] = execution_context + response = await self._get_client().invoke( + **scorer_kwargs, + input=input_text if _has_text(input_text) else None, + output=output_text if _has_text(output_text) else None, + config=self.config.scorer_config, + timeout=self.get_timeout_seconds(), + ) + + if response.status.lower() != "success": + message = response.error_message or f"LLM scorer status: {response.status}" + raise RuntimeError(message) + + matched = self._score_matches(response.score) + metadata = self._metadata(response) + operator = self.config.operator + threshold = self.config.threshold + state = "triggered" if matched else "not triggered" + return EvaluatorResult( + matched=matched, + confidence=_confidence_from_score(response.score), + message=( + f"LLM scorer score {response.score!r} {operator} threshold " + f"{threshold!r}: control {state}." + ), + metadata=metadata, + ) + except Exception as exc: + logger.error("LLM scorer evaluation error: %s", exc, exc_info=True) + return self._handle_error(exc) + + def _base_metadata(self) -> dict[str, Any]: + """Build result metadata without implying a requested version executed.""" + metadata: dict[str, Any] = {"scorer_id": self.config.scorer_id} + if self.config.scorer_version_id is not None: + metadata["requested_scorer_version_id"] = self.config.scorer_version_id + if self.config.scorer_label is not None: + metadata["scorer_label"] = self.config.scorer_label + return metadata + + def _scorer_kwargs(self) -> dict[str, Any]: + kwargs: dict[str, Any] = {"scorer_id": self.config.scorer_id} + if self.config.scorer_version_id is not None: + kwargs["scorer_version_id"] = self.config.scorer_version_id + if self.config.scorer_label is not None: + kwargs["scorer_label"] = self.config.scorer_label + return kwargs + + def _metadata( + self, + response: ScorerInvokeResponse, + ) -> dict[str, Any]: + metadata: dict[str, Any] = self._base_metadata() + echoed_label = response.scorer_label or self.config.scorer_label + if echoed_label is not None: + metadata["scorer_label"] = echoed_label + metadata.update( + { + "score": response.score, + "threshold": self.config.threshold, + "operator": self.config.operator, + "status": response.status, + "execution_time_seconds": response.execution_time, + "error_message": response.error_message, + } + ) + return metadata + + def _handle_error( + self, + error: Exception, + ) -> EvaluatorResult: + error_detail = str(error) + metadata: dict[str, Any] = { + **self._base_metadata(), + "error_type": type(error).__name__, + } + if isinstance(error, httpx.HTTPStatusError): + metadata.update(_http_status_error_metadata(error)) + + return EvaluatorResult( + matched=False, + confidence=0.0, + message=f"LLM scorer evaluation error: {error_detail}", + metadata=metadata, + error=error_detail, + ) + + async def aclose(self) -> None: + """Close the underlying Galileo LLM client.""" + await self._client.close() diff --git a/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py b/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py new file mode 100644 index 00000000..c44c119b --- /dev/null +++ b/evaluators/contrib/galileo/tests/test_llm_coverage_gaps.py @@ -0,0 +1,1053 @@ +"""Targeted tests filling coverage gaps in llm/evaluator.py and llm/client.py. + +These tests cover the small utility functions and rare branches that the +integration-style tests in ``test_llm_evaluator.py`` skip past. +""" + +from __future__ import annotations + +import json +import os +from base64 import urlsafe_b64decode +from unittest.mock import AsyncMock, MagicMock, patch + +import httpx +import pytest + +LLM_ENV = { + "GALILEO_API_SECRET_KEY": "test-secret", + "GALILEO_LUNA_INVOKE_URL": "http://luna-invoke:8090", + "GALILEO_FEATURE_FLAG_LLM_EVALUATOR": "enabled", +} + +_EXTENSIONS = { + "namespace_key": "org-1", + "caller_id": "user-1", + "target_type": "log_stream", + "target_id": "run-1", + "metadata": {}, +} + + +def _make_invoke_kwargs() -> dict[str, object]: + """Minimal required kwargs for GalileoLLMClient.invoke().""" + from agent_control_models import Step + + from agent_control_evaluator_galileo.llm.client import GalileoExecutionContext + + return { + "step": Step(type="llm", name="test", input="prompt"), + "execution_context": GalileoExecutionContext(organization_id="org-1"), + } + + +def _decode_jwt_payload(token: str) -> dict[str, object]: + payload_segment = token.split(".")[1] + padded = payload_segment + ("=" * (-len(payload_segment) % 4)) + return json.loads(urlsafe_b64decode(padded.encode()).decode()) + + +# ============================================================================= +# llm/evaluator.py: utility helpers +# ============================================================================= + + +class TestCoercePayloadText: + """``_coerce_payload_text`` normalises arbitrary values to strings.""" + + def test_none_returns_none(self): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + assert _coerce_payload_text(None) is None + + def test_string_passed_through(self): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + assert _coerce_payload_text("hello") == "hello" + + @pytest.mark.parametrize("value", [42, 3.14, True]) + def test_scalars_stringified(self, value): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + assert _coerce_payload_text(value) == str(value) + + def test_dict_is_json_serialized(self): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + result = _coerce_payload_text({"a": 1, "b": 2}) + + assert json.loads(result) == {"a": 1, "b": 2} + + def test_unserialisable_falls_back_to_str(self): + from agent_control_evaluator_galileo.llm.evaluator import _coerce_payload_text + + class CannotJson: + def __repr__(self): + return "" + + cannot = CannotJson() + result = _coerce_payload_text({"obj": cannot}) + + # default=str converts the inner object, so we still get a JSON string. + assert isinstance(result, str) + + +class TestExtractDictText: + """``_extract_dict_text`` returns ``None`` for missing keys.""" + + def test_missing_key_returns_none(self): + from agent_control_evaluator_galileo.llm.evaluator import _extract_dict_text + + assert _extract_dict_text({}, "absent") is None + + def test_present_key_coerced(self): + from agent_control_evaluator_galileo.llm.evaluator import _extract_dict_text + + assert _extract_dict_text({"x": 7}, "x") == "7" + + +class TestContains: + """``_contains`` supports str/list and dict values against a threshold.""" + + def test_none_threshold_is_no_match(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains("anything", None) is False + + def test_string_contains_substring(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains("hello world", "world") is True + assert _contains("hello world", "absent") is False + + def test_list_contains_value(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains(["a", "b", "c"], "b") is True + assert _contains(["a", "b", "c"], "z") is False + + def test_dict_threshold_does_not_match_key(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains({"toxicity": 0.9}, "toxicity") is False + + def test_dict_threshold_matches_value(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains({"label": "flagged"}, "flagged") is True + + def test_other_types_return_false(self): + from agent_control_evaluator_galileo.llm.evaluator import _contains + + assert _contains(42, 42) is False + + +class TestConfidenceFromScore: + """``_confidence_from_score`` maps a raw score to [0, 1].""" + + def test_true_bool_maps_to_one(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score(True) == 1.0 + + def test_false_bool_maps_to_zero(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score(False) == 0.0 + + def test_in_range_number_returned_as_is(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score(0.42) == 0.42 + + def test_out_of_range_falls_back_to_one(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score(7.2) == 1.0 + + def test_non_numeric_falls_back_to_one(self): + from agent_control_evaluator_galileo.llm.evaluator import _confidence_from_score + + assert _confidence_from_score("not-a-number") == 1.0 + + +# ============================================================================= +# llm/evaluator.py: _score_matches operator branches +# ============================================================================= + + +@pytest.fixture +def llm_evaluator(monkeypatch): + """A ready-to-use LlmEvaluator instance with auth env wired up.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + return LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5, "operator": "gte"}) + + +class TestScoreMatchesOperators: + """Every operator branch in ``_score_matches`` should evaluate.""" + + def _make(self, operator, threshold, monkeypatch): + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + return LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": threshold, "operator": operator} + ) + + def test_any_truthy_score_matches(self, monkeypatch): + evaluator = self._make("any", 0.5, monkeypatch) + assert evaluator._score_matches(1) is True + assert evaluator._score_matches(0) is False + + def test_eq_matches_threshold(self, monkeypatch): + evaluator = self._make("eq", "flagged", monkeypatch) + assert evaluator._score_matches("flagged") is True + assert evaluator._score_matches("safe") is False + + def test_ne_matches_when_different(self, monkeypatch): + evaluator = self._make("ne", "flagged", monkeypatch) + assert evaluator._score_matches("safe") is True + assert evaluator._score_matches("flagged") is False + + def test_contains_matches_substring(self, monkeypatch): + evaluator = self._make("contains", "flag", monkeypatch) + assert evaluator._score_matches("flagged") is True + assert evaluator._score_matches("clean") is False + + def test_numeric_operators_all_branches(self, monkeypatch): + for op, expectations in [ + ("gt", [(0.9, True), (0.5, False)]), + ("gte", [(0.5, True), (0.4, False)]), + ("lt", [(0.4, True), (0.5, False)]), + ("lte", [(0.5, True), (0.6, False)]), + ]: + evaluator = self._make(op, 0.5, monkeypatch) + for score, expected in expectations: + assert evaluator._score_matches(score) is expected, (op, score) + + def test_numeric_operator_rejects_non_numeric_score(self, monkeypatch): + evaluator = self._make("gte", 0.5, monkeypatch) + with pytest.raises(ValueError, match="not numeric"): + evaluator._score_matches("not-a-number") + + +# ============================================================================= +# llm/evaluator.py: payload preparation + aclose +# ============================================================================= + + +class TestPreparePayload: + """``_prepare_payload`` routes scalar data using explicit config.""" + + def test_scalar_routed_to_input_by_default(self, monkeypatch): + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + + input_text, output_text = evaluator._prepare_payload("hello") + + assert input_text == "hello" + assert output_text is None + + def test_scalar_routed_to_output_when_payload_field_is_output(self, monkeypatch): + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "payload_field": "output"} + ) + + input_text, output_text = evaluator._prepare_payload("hello") + + assert input_text is None + assert output_text == "hello" + + def test_structured_payload_uses_input_output_keys_over_payload_field(self, monkeypatch): + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "payload_field": "output"} + ) + + input_text, output_text = evaluator._prepare_payload( + {"input": "prompt", "output": "answer"} + ) + + assert input_text == "prompt" + assert output_text == "answer" + + +@pytest.mark.asyncio +async def test_evaluator_aclose_closes_underlying_client(monkeypatch): + """``aclose`` must release the eagerly-created client without clearing it.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm import LlmEvaluator + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + + fake = MagicMock() + fake.close = AsyncMock() + evaluator._client = fake + + await evaluator.aclose() + + fake.close.assert_awaited_once() + assert evaluator._client is fake + + +@pytest.mark.asyncio +async def test_evaluator_handles_non_success_status(monkeypatch): + """A non-success status from the scorer must surface as an error result.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_models import Step + + from agent_control_evaluator_galileo.llm import LlmEvaluator, ScorerInvokeResponse + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "operator": "gte"} + ) + step = Step(type="llm", name="test", input="prompt") + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse( + scorer_label="toxicity", + score=None, + status="failed", + error_message="upstream timeout", + ) + + result = await evaluator.evaluate_with_extensions("hello", step, _EXTENSIONS) + + assert result.matched is False + assert result.error is not None + assert "upstream timeout" in result.error + + +@pytest.mark.asyncio +async def test_evaluator_skips_empty_data(monkeypatch): + """Evaluator skips invocation when there is no text to score.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_models import Step + + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + step = Step(type="llm", name="test", input="prompt") + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + result = await evaluator.evaluate_with_extensions(None, step, _EXTENSIONS) + + assert result.matched is False + assert result.error is None + assert "No data to score" in result.message + mock_invoke.assert_not_called() + + +@pytest.mark.asyncio +async def test_evaluator_returns_error_result_on_http_error(monkeypatch): + """HTTP errors must surface as a non-matched EvaluatorResult with metadata.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_models import Step + + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + step = Step(type="llm", name="test", input="prompt") + + fake_request = httpx.Request("POST", "http://luna-invoke:8090/api/v1/scorers/invoke") + fake_response = httpx.Response(500, text="internal server error", request=fake_request) + + with patch.object( + GalileoLLMClient, + "invoke", + new_callable=AsyncMock, + side_effect=httpx.HTTPStatusError("500", request=fake_request, response=fake_response), + ): + result = await evaluator.evaluate_with_extensions("hello", step, _EXTENSIONS) + + assert result.matched is False + assert result.error is not None + assert result.metadata.get("http_status_code") == 500 + + +# ============================================================================= +# llm/evaluator.py: package version fallback +# ============================================================================= + + +def test_resolve_package_version_falls_back_when_metadata_missing(): + """The dev fallback must trigger when the package isn't installed by metadata.""" + from importlib.metadata import PackageNotFoundError + + from agent_control_evaluator_galileo.llm import evaluator as evaluator_module + + with patch.object(evaluator_module, "version", side_effect=PackageNotFoundError): + result = evaluator_module._resolve_package_version() + + assert result == "0.0.0.dev" + + +# ============================================================================= +# llm/client.py: small helpers + branches +# ============================================================================= + + +class TestAsFloatOrNone: + """``_as_float_or_none`` parses scalar values; strings may fail.""" + + def test_returns_none_for_bool(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none(True) is None + + def test_returns_none_for_none(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none(None) is None + + def test_returns_float_for_int(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none(7) == 7.0 + + def test_returns_float_for_string_number(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none("0.42") == 0.42 + + def test_returns_none_for_unparseable_string(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none("not-a-number") is None + + def test_returns_none_for_other_types(self): + from agent_control_evaluator_galileo.llm.client import _as_float_or_none + + assert _as_float_or_none([1, 2]) is None + + +class TestHasValue: + """``_has_value`` is the "is this scorable" predicate.""" + + def test_none_is_empty(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value(None) is False + + def test_empty_string_is_empty(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value("") is False + assert _has_value(" ") is False + + def test_non_empty_string_has_value(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value("hi") is True + + def test_empty_list_or_dict_is_empty(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value([]) is False + assert _has_value({}) is False + + def test_non_empty_list_or_dict_has_value(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value([1]) is True + assert _has_value({"k": "v"}) is True + + def test_scalar_other_types_have_value(self): + from agent_control_evaluator_galileo.llm.client import _has_value + + assert _has_value(42) is True + assert _has_value(0) is True + assert _has_value(True) is True + + +class TestEffectiveScorerTimeout: + """``_effective_scorer_timeout`` applies 80% server timeout by default.""" + + def test_defaults_to_80_percent_of_http_timeout(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeConfig, + _effective_scorer_timeout, + ) + + config = _effective_scorer_timeout(ScorerInvokeConfig(), http_timeout_seconds=10.0) + assert config.request_timeout_seconds == pytest.approx(8.0) + + def test_explicit_server_timeout_preserved_when_shorter(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeConfig, + _effective_scorer_timeout, + ) + + config = _effective_scorer_timeout( + ScorerInvokeConfig(request_timeout_seconds=5.0), http_timeout_seconds=10.0 + ) + assert config.request_timeout_seconds == 5.0 + + def test_explicit_server_timeout_equal_to_http_raises(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeConfig, + _effective_scorer_timeout, + ) + + with pytest.raises(ValueError, match="shorter than the HTTP timeout"): + _effective_scorer_timeout( + ScorerInvokeConfig(request_timeout_seconds=10.0), http_timeout_seconds=10.0 + ) + + def test_zero_http_timeout_raises(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeConfig, + _effective_scorer_timeout, + ) + + with pytest.raises(ValueError, match="greater than 0"): + _effective_scorer_timeout(ScorerInvokeConfig(), http_timeout_seconds=0.0) + + +class TestLoadFloatEnv: + """``_load_float_env`` reads float env vars with a default.""" + + def test_returns_default_when_unset(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_float_env + + monkeypatch.delenv("TEST_FLOAT_ENV", raising=False) + assert _load_float_env("TEST_FLOAT_ENV", 3.14) == 3.14 + + def test_returns_default_when_blank(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_float_env + + monkeypatch.setenv("TEST_FLOAT_ENV", " ") + assert _load_float_env("TEST_FLOAT_ENV", 3.14) == 3.14 + + def test_parses_valid_float(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_float_env + + monkeypatch.setenv("TEST_FLOAT_ENV", "2.5") + assert _load_float_env("TEST_FLOAT_ENV", 1.0) == 2.5 + + def test_raises_on_invalid(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_float_env + + monkeypatch.setenv("TEST_FLOAT_ENV", "not-a-float") + with pytest.raises(ValueError, match="not a number"): + _load_float_env("TEST_FLOAT_ENV", 1.0) + + +class TestLoadIntEnv: + """``_load_int_env`` reads integer env vars with a default.""" + + def test_returns_default_when_unset(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_int_env + + monkeypatch.delenv("TEST_INT_ENV", raising=False) + assert _load_int_env("TEST_INT_ENV", 42) == 42 + + def test_returns_default_when_blank(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_int_env + + monkeypatch.setenv("TEST_INT_ENV", " ") + assert _load_int_env("TEST_INT_ENV", 42) == 42 + + def test_parses_valid_int(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_int_env + + monkeypatch.setenv("TEST_INT_ENV", "10") + assert _load_int_env("TEST_INT_ENV", 1) == 10 + + def test_raises_on_invalid(self, monkeypatch): + from agent_control_evaluator_galileo.llm.client import _load_int_env + + monkeypatch.setenv("TEST_INT_ENV", "not-an-int") + with pytest.raises(ValueError, match="not an integer"): + _load_int_env("TEST_INT_ENV", 1) + + +class TestValidateConnectionConfig: + """``_validate_connection_config`` rejects invalid tuning parameters.""" + + def _valid(self): + return { + "keepalive_expiry_seconds": 1.0, + "max_connections": 100, + "max_keepalive_connections": 20, + "client_pool_size": 1, + } + + def test_negative_keepalive_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "keepalive_expiry_seconds": -1.0} + with pytest.raises(ValueError, match="greater than or equal to 0"): + _validate_connection_config(**cfg) + + def test_zero_max_connections_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "max_connections": 0} + with pytest.raises(ValueError, match="greater than 0"): + _validate_connection_config(**cfg) + + def test_negative_max_keepalive_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "max_keepalive_connections": -1} + with pytest.raises(ValueError, match="greater than or equal to 0"): + _validate_connection_config(**cfg) + + def test_keepalive_exceeds_connections_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "max_connections": 5, "max_keepalive_connections": 10} + with pytest.raises(ValueError, match="less than or equal to"): + _validate_connection_config(**cfg) + + def test_zero_pool_size_raises(self): + from agent_control_evaluator_galileo.llm.client import _validate_connection_config + + cfg = {**self._valid(), "client_pool_size": 0} + with pytest.raises(ValueError, match="greater than 0"): + _validate_connection_config(**cfg) + + +class TestScorerInvokeRequestValidation: + """``ScorerInvokeRequest`` rejects malformed input combos.""" + + def test_missing_scorer_id_raises(self): + from agent_control_evaluator_galileo.llm.client import ( + ScorerInvokeInputs, + ScorerInvokeRequest, + ) + from pydantic import ValidationError + + with pytest.raises(ValidationError, match="scorer_id"): + ScorerInvokeRequest(inputs=ScorerInvokeInputs(query="hello")) + + +def test_client_raises_when_no_api_secret(monkeypatch): + """The client requires GALILEO_API_SECRET_KEY or GALILEO_API_SECRET.""" + for name in ("GALILEO_API_SECRET_KEY", "GALILEO_API_SECRET"): + monkeypatch.delenv(name, raising=False) + monkeypatch.setenv("GALILEO_LUNA_INVOKE_URL", "http://luna-invoke:8090") + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + with pytest.raises(ValueError, match="GALILEO_API_SECRET_KEY or GALILEO_API_SECRET"): + GalileoLLMClient() + + +def test_client_raises_when_no_llm_invoke_url(monkeypatch): + """The client requires GALILEO_LUNA_INVOKE_URL.""" + monkeypatch.setenv("GALILEO_API_SECRET_KEY", "test-secret") + monkeypatch.delenv("GALILEO_LUNA_INVOKE_URL", raising=False) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + with pytest.raises(ValueError, match="GALILEO_LUNA_INVOKE_URL"): + GalileoLLMClient() + + +def test_client_jwt_has_internal_scope(monkeypatch): + """JWT produced by the client must carry internal=True and scope=scorers.invoke.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + _, auth_header = client._endpoint_and_auth_header() + + assert auth_header.startswith("Bearer ") + payload = _decode_jwt_payload(auth_header.removeprefix("Bearer ")) + assert payload["internal"] is True + assert payload["scope"] == "scorers.invoke" + + +def test_client_posts_to_correct_llm_invoke_endpoint(monkeypatch): + """_endpoint_and_auth_header must return the LLM invoke endpoint path.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + endpoint, _ = client._endpoint_and_auth_header() + + assert endpoint == "http://luna-invoke:8090/api/v1/scorers/invoke" + + +def test_client_does_not_use_old_api_paths(monkeypatch): + """The client must not reference /internal/scorers/invoke.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + endpoint, _ = client._endpoint_and_auth_header() + + assert "/scorers/invoke" in endpoint + assert endpoint.startswith("http://luna-invoke:8090/api/v1/") + assert "/internal/scorers/invoke" not in endpoint + + +@pytest.mark.asyncio +async def test_get_client_does_not_set_galileo_api_key_header(monkeypatch): + """The HTTP client must never include a Galileo-API-Key header.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + http_client = await client._get_client() + try: + assert "Galileo-API-Key" not in http_client.headers + assert "galileo-api-key" not in http_client.headers + finally: + await client.close() + + +@pytest.mark.asyncio +async def test_get_client_uses_configured_llm_invoke_ca_file(monkeypatch): + """The HTTP client should verify TLS with the configured CA.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("GALILEO_LUNA_INVOKE_CA_FILE", "/etc/galileo/llm-invoke-ca.crt") + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + ssl_context = object() + with ( + patch.object(GalileoLLMClient, "_load_ssl_context", return_value=ssl_context), + patch("httpx.AsyncClient") as async_client, + ): + client = GalileoLLMClient() + await client._get_client() + + assert client.llm_invoke_ca_file == "/etc/galileo/llm-invoke-ca.crt" + assert async_client.call_args.kwargs["verify"] is ssl_context + + +@pytest.mark.asyncio +async def test_get_client_falls_back_to_agent_control_auth_upstream_ca_file(monkeypatch): + """Galileo in-cluster pods already mount the internal CA for auth upstream.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + monkeypatch.delenv("GALILEO_LUNA_INVOKE_CA_FILE", raising=False) + monkeypatch.setenv( + "AGENT_CONTROL_AUTH_UPSTREAM_CA_FILE", "/etc/agent-control/auth-upstream-ca/ca.crt" + ) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + ssl_context = object() + with ( + patch.object(GalileoLLMClient, "_load_ssl_context", return_value=ssl_context), + patch("httpx.AsyncClient") as async_client, + ): + client = GalileoLLMClient() + await client._get_client() + + assert client.llm_invoke_ca_file == "/etc/agent-control/auth-upstream-ca/ca.crt" + assert async_client.call_args.kwargs["verify"] is ssl_context + + +@pytest.mark.asyncio +async def test_invoke_raises_when_response_is_not_a_json_object(monkeypatch): + """A non-object JSON body must surface as a clear RuntimeError.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + + fake_response = MagicMock() + fake_response.raise_for_status = MagicMock() + fake_response.json = MagicMock(return_value=["not", "an", "object"]) + + fake_http = AsyncMock() + fake_http.post = AsyncMock(return_value=fake_response) + fake_http.is_closed = False + client._client = fake_http + + try: + with pytest.raises(RuntimeError, match="not a JSON object"): + await client.invoke(scorer_id="scorer-123", input="hello", **_make_invoke_kwargs()) + finally: + await client.close() + + +@pytest.mark.asyncio +async def test_invoke_propagates_http_status_error(monkeypatch): + """The client logs and re-raises HTTP status errors.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + + fake_response = MagicMock(spec=httpx.Response) + fake_response.status_code = 500 + fake_response.text = "internal error" + fake_response.raise_for_status = MagicMock( + side_effect=httpx.HTTPStatusError( + "boom", request=MagicMock(spec=httpx.Request), response=fake_response + ) + ) + + fake_http = AsyncMock() + fake_http.post = AsyncMock(return_value=fake_response) + fake_http.is_closed = False + client._client = fake_http + + try: + with pytest.raises(httpx.HTTPStatusError): + await client.invoke(scorer_id="scorer-123", input="hello", **_make_invoke_kwargs()) + finally: + await client.close() + + +@pytest.mark.asyncio +async def test_invoke_propagates_request_error(monkeypatch): + """RequestError is logged and re-raised so callers can decide policy.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + + fake_http = AsyncMock() + fake_http.post = AsyncMock(side_effect=httpx.RequestError("network down")) + fake_http.is_closed = False + client._client = fake_http + + try: + with pytest.raises(httpx.RequestError): + await client.invoke(scorer_id="scorer-123", input="hello", **_make_invoke_kwargs()) + finally: + await client.close() + + +@pytest.mark.asyncio +async def test_client_async_context_manager_closes_on_exit(monkeypatch): + """Entering/exiting the async context manager must close the client.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + async with GalileoLLMClient() as client: + await client._get_client() + assert client._client is not None + + assert client._client is None + + +@pytest.mark.asyncio +async def test_invoke_strips_caller_supplied_galileo_api_key_header(monkeypatch): + """Regression: a Galileo-API-Key passed via the headers kwarg must be stripped.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + captured: dict[str, object] = {} + + def handler(request: httpx.Request) -> httpx.Response: + captured["headers"] = dict(request.headers) + return httpx.Response(200, json={"score": 0.9, "status": "success"}) + + client = GalileoLLMClient() + client._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + + try: + await client.invoke( + scorer_id="scorer-123", + input="hello", + headers={"Galileo-API-Key": "should-be-stripped", "X-Custom": "keep-me"}, + **_make_invoke_kwargs(), + ) + finally: + await client.close() + + headers = captured["headers"] + assert isinstance(headers, dict) + assert "galileo-api-key" not in headers + assert headers.get("x-custom") == "keep-me" + + +@pytest.mark.asyncio +async def test_invoke_always_emits_config_field(monkeypatch): + """The request always carries a server timeout below its HTTP deadline.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + captured: dict[str, object] = {} + + def handler(request: httpx.Request) -> httpx.Response: + captured["body"] = json.loads(request.content.decode()) + return httpx.Response(200, json={"score": 0.5, "status": "success"}) + + client = GalileoLLMClient() + client._client = httpx.AsyncClient(transport=httpx.MockTransport(handler)) + + try: + await client.invoke(scorer_id="scorer-123", input="hello", **_make_invoke_kwargs()) + finally: + await client.close() + + assert "config" in captured["body"] + assert captured["body"]["config"] == {"request_timeout_seconds": 8.0} + + +@pytest.mark.asyncio +async def test_client_pool_round_robin(monkeypatch): + """With pool_size > 1, the client rotates through multiple connections.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + monkeypatch.setenv("GALILEO_LUNA_CLIENT_POOL_SIZE", "2") + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + client = GalileoLLMClient() + assert client.client_pool_size == 2 + + first = await client._get_client() + second = await client._get_client() + third = await client._get_client() + + # With pool_size=2, two distinct clients exist and we cycle back + assert len(client._clients) == 2 + assert first is not second + assert third is first # wraps around + + await client.close() + assert client._clients == [] + + +# ============================================================================= +# llm/config.py: threshold validator branches +# ============================================================================= + + +class TestLlmEvaluatorConfigValidation: + """``LlmEvaluatorConfig.validate_threshold`` exercises all branches.""" + + def test_numeric_operator_with_non_numeric_threshold_raises(self): + from agent_control_evaluator_galileo.llm.config import LlmEvaluatorConfig + from pydantic import ValidationError + + with pytest.raises(ValidationError, match="numeric threshold"): + LlmEvaluatorConfig(scorer_id="scorer-123", operator="gt", threshold="not-a-number") + + def test_none_threshold_with_non_any_operator_raises(self): + from agent_control_evaluator_galileo.llm.config import LlmEvaluatorConfig + from pydantic import ValidationError + + with pytest.raises(ValidationError, match="threshold is required"): + LlmEvaluatorConfig(scorer_id="scorer-123", operator="eq", threshold=None) + + def test_any_operator_allows_none_threshold(self): + from agent_control_evaluator_galileo.llm.config import LlmEvaluatorConfig + + config = LlmEvaluatorConfig(scorer_id="scorer-123", operator="any", threshold=None) + assert config.operator == "any" + + def test_coerce_number_returns_none_for_list(self): + from agent_control_evaluator_galileo.llm.config import coerce_number + + assert coerce_number([1, 2, 3]) is None + + def test_coerce_number_returns_none_for_non_numeric_string(self): + from agent_control_evaluator_galileo.llm.config import coerce_number + + assert coerce_number("abc") is None + + def test_coerce_number_parses_numeric_string(self): + from agent_control_evaluator_galileo.llm.config import coerce_number + + assert coerce_number("0.75") == pytest.approx(0.75) + + +@pytest.mark.asyncio +async def test_evaluator_with_context_uses_step(monkeypatch): + """evaluate_with_extensions passes the step and execution_context to invoke.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_models import Step + + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient, ScorerInvokeResponse + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123", "threshold": 0.5}) + step = Step(type="llm", name="test", input="prompt", output="answer") + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate_with_extensions("selected input", step, _EXTENSIONS) + + assert result.matched is True + assert mock_invoke.call_args.kwargs["step"] == step + assert mock_invoke.call_args.kwargs["execution_context"] is not None + + +@pytest.mark.asyncio +async def test_evaluator_scorer_label_echoed_in_metadata(monkeypatch): + """scorer_label echoed from the response is included in result metadata.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_models import Step + + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient, ScorerInvokeResponse + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "scorer_label": "my-label"} + ) + step = Step(type="llm", name="test", input="prompt") + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse( + scorer_label="echoed-label", score=0.8, status="success" + ) + result = await evaluator.evaluate_with_extensions("hello", step, _EXTENSIONS) + + assert result.metadata.get("scorer_label") == "echoed-label" + + +@pytest.mark.asyncio +async def test_evaluator_scorer_version_id_in_metadata(monkeypatch): + """scorer_version_id is included in metadata when configured.""" + for key, value in LLM_ENV.items(): + monkeypatch.setenv(key, value) + from agent_control_models import Step + + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient, ScorerInvokeResponse + + evaluator = LlmEvaluator.from_dict( + { + "scorer_id": "scorer-123", + "scorer_version_id": "ver-456", + "threshold": 0.5, + } + ) + step = Step(type="llm", name="test", input="prompt") + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate_with_extensions("hello", step, _EXTENSIONS) + + assert result.metadata.get("requested_scorer_version_id") == "ver-456" + assert mock_invoke.call_args.kwargs["scorer_version_id"] == "ver-456" diff --git a/evaluators/contrib/galileo/tests/test_llm_evaluator.py b/evaluators/contrib/galileo/tests/test_llm_evaluator.py new file mode 100644 index 00000000..e7d093b3 --- /dev/null +++ b/evaluators/contrib/galileo/tests/test_llm_evaluator.py @@ -0,0 +1,198 @@ +"""Tests for the direct Galileo LLM-as-judge evaluator and client.""" + +from __future__ import annotations + +import os +from unittest.mock import AsyncMock, patch + +import pytest +from agent_control_models import Step + +LLM_ENV = { + "GALILEO_API_SECRET_KEY": "test-secret", + "GALILEO_LUNA_INVOKE_URL": "http://luna-invoke:8090", + "GALILEO_FEATURE_FLAG_LLM_EVALUATOR": "enabled", +} + + +class TestGalileoLLMClient: + """Tests for ``GalileoLLMClient`` and ``ScorerInvokeRequest``.""" + + def test_execution_context_serializes_missing_optional_claims_as_null(self) -> None: + from agent_control_evaluator_galileo.llm.client import ( + GalileoExecutionContext, + ScorerInvokeInputs, + ScorerInvokeRequest, + ) + + request = ScorerInvokeRequest( + scorer_id="scorer-123", + inputs=ScorerInvokeInputs(query="hello"), + execution_context=GalileoExecutionContext(organization_id="org-1"), + ) + + assert request.to_dict()["execution_context"] == { + "organization_id": "org-1", + "user_id": None, + "project_id": None, + "run_id": None, + } + + +class TestLlmEvaluator: + """Tests for ``LlmEvaluator`` execution context handling.""" + + @patch.dict(os.environ, LLM_ENV) + @pytest.mark.asyncio + @pytest.mark.parametrize("target_type", ["log_stream", "agent_stream"]) + async def test_evaluator_consumes_verified_opaque_extensions(self, target_type: str) -> None: + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import ( + GalileoExecutionContext, + GalileoLLMClient, + ScorerInvokeResponse, + ) + + evaluator = LlmEvaluator.from_dict( + {"scorer_id": "scorer-123", "threshold": 0.5, "operator": "gte"} + ) + extensions = { + "namespace_key": "org-1", + "caller_id": "verified-user-2", + "target_type": target_type, + "target_id": "run-4", + "metadata": {"project_id": "project-3"}, + } + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate_with_extensions( + "selected input", Step(type="llm", name="answer", input="prompt"), extensions + ) + + assert result.matched is True + mock_invoke.assert_awaited_once_with( + scorer_id="scorer-123", + step=Step(type="llm", name="answer", input="prompt"), + selected_data="selected input", + selected_data_payload_field="input", + execution_context=GalileoExecutionContext( + organization_id="org-1", + user_id="verified-user-2", + project_id="project-3", + run_id="run-4", + ), + input="selected input", + output=None, + config=None, + timeout=10.0, + ) + + @patch.dict(os.environ, LLM_ENV) + @pytest.mark.asyncio + async def test_evaluator_allows_missing_caller_id(self) -> None: + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import ( + GalileoExecutionContext, + GalileoLLMClient, + ScorerInvokeResponse, + ) + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123"}) + extensions = { + "namespace_key": "org-1", + "target_type": "log_stream", + "target_id": "run-4", + "metadata": {}, + } + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate_with_extensions( + "selected input", Step(type="llm", name="answer", input="prompt"), extensions + ) + + assert result.error is None + mock_invoke.assert_awaited_once_with( + scorer_id="scorer-123", + step=Step(type="llm", name="answer", input="prompt"), + selected_data="selected input", + selected_data_payload_field="input", + execution_context=GalileoExecutionContext( + organization_id="org-1", + user_id=None, + project_id=None, + run_id="run-4", + ), + input="selected input", + output=None, + config=None, + timeout=10.0, + ) + + @patch.dict(os.environ, LLM_ENV) + @pytest.mark.asyncio + async def test_evaluator_omits_run_id_for_unsupported_target_type(self) -> None: + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import ( + GalileoExecutionContext, + GalileoLLMClient, + ScorerInvokeResponse, + ) + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123"}) + extensions = { + "namespace_key": "org-1", + "caller_id": "verified-user-2", + "target_type": "environment", + "target_id": "prod", + "metadata": {"project_id": "project-3"}, + } + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + mock_invoke.return_value = ScorerInvokeResponse(score=0.8, status="success") + result = await evaluator.evaluate_with_extensions( + "selected input", Step(type="llm", name="answer", input="prompt"), extensions + ) + + assert result.error is None + mock_invoke.assert_awaited_once_with( + scorer_id="scorer-123", + step=Step(type="llm", name="answer", input="prompt"), + selected_data="selected input", + selected_data_payload_field="input", + execution_context=GalileoExecutionContext( + organization_id="org-1", + user_id="verified-user-2", + project_id="project-3", + run_id=None, + ), + input="selected input", + output=None, + config=None, + timeout=10.0, + ) + + @patch.dict(os.environ, LLM_ENV) + @pytest.mark.asyncio + async def test_evaluator_rejects_missing_organization_id(self) -> None: + from agent_control_evaluator_galileo.llm import LlmEvaluator + from agent_control_evaluator_galileo.llm.client import GalileoLLMClient + + evaluator = LlmEvaluator.from_dict({"scorer_id": "scorer-123"}) + extensions = { + "target_type": "log_stream", + "target_id": "run-4", + "metadata": {"project_id": "project-3"}, + } + + with patch.object(GalileoLLMClient, "invoke", new_callable=AsyncMock) as mock_invoke: + result = await evaluator.evaluate_with_extensions( + "selected input", + Step(type="llm", name="answer", input="prompt"), + extensions, + ) + + assert result.error is not None + assert "organization_id" in result.error + mock_invoke.assert_not_called() diff --git a/sdks/python/ARCHITECTURE.md b/sdks/python/ARCHITECTURE.md index 4a48c09a..a58fb4c7 100644 --- a/sdks/python/ARCHITECTURE.md +++ b/sdks/python/ARCHITECTURE.md @@ -20,7 +20,7 @@ sdks/python/src/agent_control/ ├── tracing.py # Distributed tracing support ├── py.typed # PEP 561 type marker └── evaluators/ # Evaluator base classes and discovery system - ├── __init__.py # Evaluator discovery, registration, and Luna integration + ├── __init__.py # Evaluator discovery, registration, and Luna/LLM integration └── base.py # Base Evaluator and EvaluatorMetadata classes ``` @@ -213,11 +213,11 @@ async def chat(message: str) -> str: **Key Components**: - Base evaluator classes (`Evaluator`, `EvaluatorMetadata`) - Evaluator discovery via entry points -- Third-party evaluator integration (e.g., Luna, Guardrails AI) +- Third-party evaluator integration (e.g., Luna, LLM, Guardrails AI) - Registration functions for custom evaluators **Structure**: -- `__init__.py` - Evaluator discovery (`discover_evaluators()`, `list_evaluators()`), registration (`register_evaluator()`), and optional Luna integration +- `__init__.py` - Evaluator discovery (`discover_evaluators()`, `list_evaluators()`), registration (`register_evaluator()`), and optional Luna/LLM integration - `base.py` - Base `Evaluator` and `EvaluatorMetadata` classes (re-exported from `agent_control_models`) **Usage**: diff --git a/sdks/python/src/agent_control/control_decorators.py b/sdks/python/src/agent_control/control_decorators.py index 8c12ae4c..6879ee0e 100644 --- a/sdks/python/src/agent_control/control_decorators.py +++ b/sdks/python/src/agent_control/control_decorators.py @@ -23,7 +23,7 @@ async def chat(message: str) -> str: # Server-side controls define: # - stage: "pre" or "post" # - selector.path: "input" or "output" - # - evaluator: regex, list, Luna evaluator, etc. + # - evaluator: regex, list, Luna evaluator, LLM evaluator, etc. # - action: deny, steer, or observe """ diff --git a/sdks/python/src/agent_control/evaluators/__init__.py b/sdks/python/src/agent_control/evaluators/__init__.py index 73714717..33e648b7 100644 --- a/sdks/python/src/agent_control/evaluators/__init__.py +++ b/sdks/python/src/agent_control/evaluators/__init__.py @@ -1,7 +1,7 @@ """Evaluator system for agent_control. This module provides an evaluator architecture for extending agent_control -with external evaluation systems like Galileo Luna, Guardrails AI, etc. +with external evaluation systems like Galileo Luna, Galileo LLM, Guardrails AI, etc. Evaluator Discovery: Call `discover_evaluators()` at startup to load evaluators. This loads: @@ -14,7 +14,12 @@ When installed with galileo extras, the Galileo evaluator types are available: ```python from agent_control.evaluators import LunaEvaluator, LunaEvaluatorConfig # if galileo installed + from agent_control.evaluators import LlmEvaluator, LlmEvaluatorConfig # if galileo installed ``` + + Note: ``galileo.llm`` requires ``GALILEO_FEATURE_FLAG_LLM_EVALUATOR=enabled`` to be set. + The evaluator is unavailable (``is_available()`` returns ``False``) until this flag is present, + as it depends on Orbit-side support for ``execution_context`` to fetch LLM credentials. """ from agent_control_engine import ( @@ -62,3 +67,25 @@ ) except ImportError: pass + +# Optionally export LLM evaluator types when available +try: + from agent_control_evaluator_galileo.llm import ( # type: ignore[import-not-found] # noqa: F401 + LLM_AVAILABLE, + GalileoLLMClient, + LlmEvaluator, + LlmEvaluatorConfig, + LlmOperator, + ) + + __all__.extend( + [ + "GalileoLLMClient", + "LlmEvaluator", + "LlmEvaluatorConfig", + "LlmOperator", + "LLM_AVAILABLE", + ] + ) +except ImportError: + pass diff --git a/sdks/python/tests/test_evaluators_optional_imports.py b/sdks/python/tests/test_evaluators_optional_imports.py index b4560fc9..370e7a6d 100644 --- a/sdks/python/tests/test_evaluators_optional_imports.py +++ b/sdks/python/tests/test_evaluators_optional_imports.py @@ -28,6 +28,7 @@ def _module_available(name: str) -> bool: _GALILEO_INSTALLED = _module_available("agent_control_evaluator_galileo.luna") +_GALILEO_LLM_INSTALLED = _module_available("agent_control_evaluator_galileo.llm") def _reload_evaluators_with_blocked(prefix: str) -> object: @@ -70,21 +71,35 @@ def test_module_loads_when_galileo_luna_is_unavailable(): # Core names are always present. assert "Evaluator" in reloaded.__all__ - # Luna1 names are NOT present because the import failed. + # Luna names are NOT present because the import failed. assert "LunaEvaluator" not in reloaded.__all__ assert "GalileoLunaClient" not in reloaded.__all__ +def test_module_loads_when_galileo_llm_is_unavailable(): + """Hiding ``agent_control_evaluator_galileo.llm`` exercises its except branch.""" + reloaded = _reload_evaluators_with_blocked("agent_control_evaluator_galileo.llm") + + # Core names are always present. + assert "Evaluator" in reloaded.__all__ + # LLM names are NOT present because the import failed. + assert "LlmEvaluator" not in reloaded.__all__ + assert "GalileoLLMClient" not in reloaded.__all__ + + def test_module_loads_when_galileo_package_is_unavailable(): """Hiding the whole package exercises the ImportError fallback.""" reloaded = _reload_evaluators_with_blocked("agent_control_evaluator_galileo") assert "Evaluator" in reloaded.__all__ - # The optional luna names are absent. + # The optional luna and llm names are absent. for absent in ( "LunaEvaluator", "GalileoLunaClient", "LUNA_AVAILABLE", + "LlmEvaluator", + "GalileoLLMClient", + "LLM_AVAILABLE", ): assert absent not in reloaded.__all__ @@ -108,3 +123,20 @@ def test_module_loads_galileo_optional_imports_when_available(): finally: if saved is not None: sys.modules["agent_control.evaluators"] = saved + + +@pytest.mark.skipif( + not _GALILEO_LLM_INSTALLED, + reason="agent-control-evaluator-galileo extras not installed in this environment", +) +def test_module_loads_galileo_llm_optional_imports_when_available(): + """Sanity check: with galileo installed, the LLM optional names ARE exposed.""" + saved = sys.modules.pop("agent_control.evaluators", None) + try: + import agent_control.evaluators as reloaded + + reloaded = importlib.reload(reloaded) + assert "LlmEvaluator" in reloaded.__all__ + finally: + if saved is not None: + sys.modules["agent_control.evaluators"] = saved