From e2c88e6280b177ca50375a8cbb5e56815a112bf5 Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Fri, 25 Sep 2026 15:41:04 +0200 Subject: [PATCH 01/10] feat(python): separate Foundry Responses agent history and background --- python/packages/foundry_hosting/README.md | 105 ++- .../__init__.py | 3 + .../_request.py | 134 +++ .../_responses.py | 558 +++++++++++-- .../foundry_hosting/tests/test_request.py | 120 +++ .../foundry_hosting/tests/test_responses.py | 787 +++++++++++++++++- .../responses/basic/.env.example | 3 +- .../responses/basic/README.md | 91 +- .../responses/basic/agent_history.py | 31 + .../responses/basic/client.py | 50 ++ .../responses/basic/main.py | 14 +- .../responses/basic/options.py | 43 + .../responses/basic/provider_background.py | 39 + .../responses/basic/service_history.py | 30 + .../steerable_long_running_agent/README.md | 113 +-- .../steerable_long_running_agent/main.py | 3 +- .../verify_steering.py | 58 +- 17 files changed, 1863 insertions(+), 319 deletions(-) create mode 100644 python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py create mode 100644 python/packages/foundry_hosting/tests/test_request.py create mode 100644 python/samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py create mode 100644 python/samples/04-hosting/foundry-hosted-agents/responses/basic/client.py create mode 100644 python/samples/04-hosting/foundry-hosted-agents/responses/basic/options.py create mode 100644 python/samples/04-hosting/foundry-hosted-agents/responses/basic/provider_background.py create mode 100644 python/samples/04-hosting/foundry-hosted-agents/responses/basic/service_history.py diff --git a/python/packages/foundry_hosting/README.md b/python/packages/foundry_hosting/README.md index f1094969032..733a8471d03 100644 --- a/python/packages/foundry_hosting/README.md +++ b/python/packages/foundry_hosting/README.md @@ -29,52 +29,63 @@ The Responses host continues regular agents through its existing session store a existing checkpoint store. A callable does not make arbitrary instance fields persistent; state needed by later requests must remain in the supported stores. -## Conversation history +## Responses agent history and storage -`ResponsesHostServer` uses AgentServer response history as the model's conversation history by default: +The caller's `POST /responses` **`store` flag** controls whether the *outer* response is retrievable and whether +MAF session and approval state is saved. It does not choose the *inner* history source: -```python -server = ResponsesHostServer(agent) -``` - -In this mode, the configured AgentServer response provider supplies the prior transcript. Hosting rejects -`HistoryProvider` instances with `load_messages=True` and agents configured with a default `conversation_id`, -`previous_response_id`, or `conversation`, adds a transient in-memory provider for function-call loops, and clears -restored downstream service IDs. For clients that advertise `STORES_BY_DEFAULT=True`, hosting forces downstream -`store=False`; for other clients it removes an explicit agent-level `store` option and does not forward one. These -safeguards ensure the model receives the transcript once without sending unsupported storage options. - -AgentServer history requires a framework `RawAgent` whose client declares the boolean `STORES_BY_DEFAULT` capability; -the agent's runtime options then let hosting enforce downstream storage behavior. Custom `SupportsAgentRun` -implementations must use `history_source="agent"` because that protocol does not accept runtime chat options. - -`ResponsesHostServer` owns a supplied agent instance and may add hosting-specific context providers. Do not reuse that -instance with another host or invoke it directly after constructing the server. An agent returned by a callable belongs -to that request. - -To preserve the agent's regular history and service-storage behavior, select the agent as the history source: - -```python -server = ResponsesHostServer(agent, history_source="agent") -``` - -Hosting then passes only current request input, allows load-enabled history providers, and does not override the -agent's downstream `store` option. For example, `InMemoryHistoryProvider` stores messages in `AgentSession.state`, which -the default `FoundryAgentSessionStore` persists in Foundry: +| `inner_history` | What the model receives | Inner storage on `store=True` | +| --- | --- | --- | +| `"host"` (default) | Prior outer Responses transcript plus new input | Disabled. Storing clients run with `store=False`; non-storing clients receive no storage option. | +| `"service"` | New input only | Enabled. The service-issued `AgentSession.service_session_id` is saved privately under the outer response ID or conversation. | +| `"agent"` | New input plus the agent's `HistoryProvider` | Disabled. Agent history in `AgentSession.state` is saved by the host, without duplicating service history. | ```python -agent = Agent( - client=client, - context_providers=[InMemoryHistoryProvider()], - default_options={"store": False}, -) -server = ResponsesHostServer(agent, history_source="agent") +server = ResponsesHostServer(agent=agent, inner_history="service") ``` -The `store` argument remains independent: it selects the AgentServer response provider used for Responses API -persistence and retrieval. Omitting it or passing `None` selects the environment default. With -`history_source="agent_server"`, that response provider also supplies model history; with `history_source="agent"`, it -does not. +`"host"` and `"service"` reject a load-enabled `HistoryProvider` alongside their own history source; `"host"` +also rejects default downstream continuation IDs. Explicit modes require a `RawAgent` with a client declaring +`STORES_BY_DEFAULT`; `"service"` requires a storing client that returns a private continuation ID. Hosting may +add a transient in-memory provider to support function-call loops, but **never edits `agent.default_options`**. +The host owns the provided agent instance and any providers it adds; do not reuse it with another host. A factory +creates an independent agent for each request. + +The deprecated `history_source="agent_server"` still selects `"host"`. **Deprecated `history_source="agent"` is +not an alias for `inner_history="agent"`**: on stored requests, it preserves the old behavior of sending only new +input while the developer's defaults choose *either* a HistoryProvider *or* downstream service storage (including +`default_options={"store": True}`). Existing custom `SupportsAgentRun` implementations can continue using this +stored-request compatibility path. Each use of `history_source=` emits one deprecation warning per host; new code +should choose its explicit history mode. The outer storage-backend constructor argument is now `response_store=`. +The old `store=` backend argument remains an alias with its own once-per-host deprecation warning; supplying both +is an error. Neither constructor argument sets the caller's per-request `store` flag. + +`store=False` returns a one-shot response without **writing** host-managed session, conversation, or approval state; +it also disables inner service storage regardless of the developer's defaults. Unsafe custom agents, external +history providers that store messages, and fixed downstream continuation defaults fail with an actionable error +instead of silently persisting. An unstored service-mode request cannot resume a private service thread. The legacy +agent mode also rejects an unstored continuation if its restored session uses downstream storage. Application-owned +tools and external services may still have their own side effects. `background=True` requires outer `store=True`. + +Outer background work always uses the caller-visible `response.id` for polling; it does not enable provider-native +background automatically. `inner_background="provider"` is a separate opt-in for `"service"` with a storing +Responses client. Its private continuation token is saved under the outer ID and never returned to the caller. +Use `ResponsesServerOptions(resilient_background=True)` to permit recovery from a **saved** token; a crash before +the token is saved cannot safely restart the inner job. Provider background and steering cannot be combined. +Regular agent runs without this opt-in are not crash-replayable. `steerable_conversations=True` enables AgentServer's +process-wide multi-turn TaskManager; a superseded turn keeps its own response snapshot but cannot replace a later +CAS-protected conversation head. Start an in-progress background turn with `stream=True` before steering it: the +current AgentServer release can leave a superseded **non-streamed** background response in progress on retrieval. +Legacy `WorkflowAgent` dispatch is unchanged. + +Native CreateResponse generation fields become MAF runtime options (notably `max_output_tokens` -> `max_tokens` and +`parallel_tool_calls` -> `allow_multiple_tool_calls`). Flattened OpenAI `extra_body` fields overlay translated keys +**last**. A sync or async `prepare_options(request: HostedResponseRequest, options: dict)` hook can remove or replace +*caller* options before `Agent.run`; removed values fall back to the developer's unchanged agent defaults. Hosting +filters caller platform IDs and private continuation/storage controls from model options and rejects attempts to +reintroduce them through the hook. A custom agent cannot accept MAF runtime options: choose +`unsupported_options` as `"ignore"`, `"warn"` (default), or `"error"` for that case. See the +[agent history and options samples](../../samples/04-hosting/foundry-hosted-agents/responses/basic/). ## State store @@ -152,12 +163,14 @@ durably. By default they use `FoundryAgentSessionStore`, backed by Foundry stora and file-based storage locally. Responses sessions use the `agent_sessions` logical store; Invocations sessions use the separate `invocation_sessions` store. -Loaded MAF sessions are saved with an ETag condition. A competing turn that has -already advanced the same conversation causes a visible persistence failure instead -of silently overwriting its state. New hosted session keys are created only if absent; -turns using `previous_response_id` write their own new response ID, without applying -the predecessor's ETag to a different key. Local callers can still upsert directly -without first loading a session. +Each stored agent turn saves a snapshot under its **own** outer `response.id`. For a named `conversation`, a +separate mutable conversation-head key is also updated. Loaded MAF sessions use PR1's ETag condition for that key: +a competing turn that advanced the head causes a visible conflict rather than a stale overwrite. A superseded +steered turn saves its response snapshot but skips the head update. Service-backed history is linear: when +continuing by `previous_response_id`, the prior response is claimed with a conditional write so a second branch +cannot reuse the same downstream service thread; attempting to fork a named service conversation is also rejected. +New hosted keys are created only if absent. Custom store providers must provide equivalent scoped conditional +writes for concurrent turns. Local callers can still upsert directly without first loading a session. See the [custom storage provider sample](../../samples/04-hosting/foundry-hosted-agents/responses/custom_storage/) for an example that uses an in-memory session store locally and Azure Cosmos DB when hosted. diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/__init__.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/__init__.py index de96da16c4b..e5b59394e8a 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/__init__.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/__init__.py @@ -5,6 +5,7 @@ if TYPE_CHECKING: from ._invocations import InvocationsHostServer + from ._request import HostedResponseRequest from ._responses import ResponsesHostServer from ._scope import FoundryRequestScope from ._state_store import ( @@ -34,6 +35,7 @@ "FoundryFunctionApprovalStore": "._state_store", "FoundryRequestScope": "._scope", "FoundryToolbox": "._toolbox", + "HostedResponseRequest": "._request", "FunctionApprovalStore": "._state_store", "FunctionApprovalStoreProvider": "._state_store", "InvocationsHostServer": "._invocations", @@ -52,6 +54,7 @@ "FoundryToolbox", "FunctionApprovalStore", "FunctionApprovalStoreProvider", + "HostedResponseRequest", "InvocationsHostServer", "ResponsesHostServer", "StoreProvider", diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py new file mode 100644 index 00000000000..4754c9fda7f --- /dev/null +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py @@ -0,0 +1,134 @@ +# Copyright (c) Microsoft. All rights reserved. + +"""Request-scoped Responses options and a view for developer hooks.""" + +from __future__ import annotations + +import inspect +from collections.abc import Awaitable, Callable, Mapping +from copy import deepcopy +from types import MappingProxyType +from typing import Any, Literal, TypeAlias, cast + +from azure.ai.agentserver.responses import ResponseContext +from azure.ai.agentserver.responses.models import CreateResponse, Item + +from ._scope import FoundryRequestScope + +UnsupportedOptions: TypeAlias = Literal["ignore", "warn", "error"] + +_HOST_CONTROLLED_FIELDS = frozenset({ + "agent", + "agent_reference", + "agent_session_id", + "background", + "call_id", + "continuation_token", + "conversation", + "conversation_id", + "input", + "previous_response_id", + "response_id", + "service_session_id", + "session_id", + "store", + "stream", + "user", + "user_id", +}) +_OPTION_NAMES = {"max_output_tokens": "max_tokens", "parallel_tool_calls": "allow_multiple_tool_calls"} +_NATIVE_FIELDS = frozenset(CreateResponse.__annotations__) + + +def response_run_options(request: CreateResponse) -> dict[str, Any]: + """Translate native generation fields, with flattened extra-body values winning collisions.""" + native: dict[str, Any] = {} + extra: dict[str, Any] = {} + nested_extra: dict[str, Any] = {} + for name, value in request.items(): + if value is None or (name == "model" and value == ""): + continue + if name == "extra_body": + if not isinstance(value, Mapping) or any( + not isinstance(key, str) for key in cast(Mapping[object, object], value) + ): + raise TypeError("extra_body must be a mapping of model options.") + nested_extra.update(cast(Mapping[str, Any], value)) + elif name in _HOST_CONTROLLED_FIELDS: + continue + elif name in _NATIVE_FIELDS: + native[_OPTION_NAMES.get(name, name)] = value + else: + extra[name] = value + return { + **native, + **{name: value for name, value in extra.items() if name not in _HOST_CONTROLLED_FIELDS}, + **{name: value for name, value in nested_extra.items() if name not in _HOST_CONTROLLED_FIELDS}, + } + + +class HostedResponseRequest: + """Expose a copied request, trusted scope, and only this turn's input to an options hook.""" + + def __init__( + self, + request: CreateResponse, + context: ResponseContext, + scope: FoundryRequestScope, + options: Mapping[str, Any], + ) -> None: + self.request: Mapping[str, Any] = MappingProxyType(deepcopy(dict(request))) + self.scope = scope + self.response_id = context.response_id + self.conversation_id = context.conversation_id + self._context = context + self._options: Mapping[str, Any] = MappingProxyType(dict(options)) + + @property + def options(self) -> Mapping[str, Any]: + """Return this turn's effective caller options after the hook.""" + return self._options + + def set_options(self, options: Mapping[str, Any]) -> None: + """Replace the caller options without changing the agent's defaults.""" + self._options = MappingProxyType(dict(options)) + + async def get_input_items(self) -> list[Item]: + """Read only this turn's input items, not earlier conversation history.""" + return list(await self._context.get_input_items()) + + async def get_input_text(self) -> str | None: + """Read this turn's text input, if present.""" + return await self._context.get_input_text() + + +OptionsHook: TypeAlias = Callable[ + [HostedResponseRequest, dict[str, Any]], + Mapping[str, Any] | Awaitable[Mapping[str, Any]], +] + + +async def prepare_response_options(request: HostedResponseRequest, hook: OptionsHook | None) -> None: + """Apply a synchronous or asynchronous developer hook to a copy of caller options.""" + if hook is None: + return + result = hook(request, dict(request.options)) + if inspect.isawaitable(result): + result = await result + if not isinstance(result, Mapping): + raise TypeError("prepare_options must return a mapping of MAF run options.") + request.set_options(result) + + +def validate_request_options(options: Mapping[str, Any]) -> None: + """Keep hosting identity, storage decisions, and private continuation out of model options.""" + reserved = _HOST_CONTROLLED_FIELDS.intersection(options) + if reserved: + raise ValueError(f"prepare_options cannot set host-controlled fields: {', '.join(sorted(reserved))}.") + + +def validate_unsupported_options(mode: str) -> UnsupportedOptions: + """Reject misspelled unsupported-options policies at host construction.""" + if mode not in ("ignore", "warn", "error"): + raise ValueError("unsupported_options must be 'ignore', 'warn', or 'error'.") + return mode diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py index 11adf65ee1f..dce18a6b956 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py @@ -9,6 +9,8 @@ import logging import os import re +import uuid +import warnings from collections.abc import ( AsyncGenerator, AsyncIterable, @@ -25,7 +27,9 @@ from urllib.parse import urlparse from agent_framework import ( + AgentResponse, AgentResponseUpdate, + AgentSession, ChatOptions, CheckpointStorage, Content, @@ -86,6 +90,15 @@ from ._agent_source import is_agent, resolve_agent, validate_agent_source from ._feature_usage import FeatureIndex +from ._request import ( + HostedResponseRequest, + OptionsHook, + UnsupportedOptions, + prepare_response_options, + response_run_options, + validate_request_options, + validate_unsupported_options, +) from ._scope import FoundryRequestScope from ._state_store import ( AgentSessionStoreProvider, @@ -101,6 +114,11 @@ _MODEL_OUTPUT_KIND_KEY = "model_output_kind" _MODEL_OUTPUT_REFUSAL = "refusal" _HOSTED_RESPONSES_HISTORY_SOURCE_ID = "_foundry_responses_history" +_HOSTED_PROVIDER_STATE_KEY = "_foundry_provider_background" +_HOSTED_SOURCE_CONVERSATION_KEY = "_foundry_source_conversation" +_HOSTED_SERVICE_CHILD_KEY = "_foundry_service_child" + +_HistoryMode = Literal["host", "service", "agent", "legacy"] def _is_refusal_text_content(content: Content) -> bool: @@ -142,6 +160,27 @@ def _create_response_event_stream(context: ResponseContext) -> ResponseEventStre return ResponseEventStream(response_id=context.response_id) +def _agent_response_updates(response: AgentResponse[Any], response_id: str) -> list[AgentResponseUpdate]: + """Project a completed inner response without exposing its private provider token or ID.""" + updates = [ + AgentResponseUpdate( + contents=list(message.contents), + role=message.role, + author_name=message.author_name, + response_id=response_id, + message_id=message.message_id or uuid.uuid4().hex, + ) + for message in response.messages + ] + if response.usage_details is not None: + updates.append(AgentResponseUpdate(contents=[Content.from_usage(response.usage_details)])) + if response.finish_reason == "length": + updates.append(AgentResponseUpdate(finish_reason="length")) + elif response.finish_reason == "content_filter": + updates.append(AgentResponseUpdate(finish_reason="content_filter")) + return updates + + _T = TypeVar("_T") # Sentinel put on the internal queue by _SignalledIterator's driver task to signal that the @@ -425,8 +464,10 @@ class _AgentConfiguration: def _validate_agent_configuration( agent: SupportsAgentRun, - history_source: Literal["agent_server", "agent"], + inner_history: _HistoryMode, options: ResponsesServerOptions | None, + *, + inner_background: Literal["host", "provider"] = "host", ) -> _AgentConfiguration: is_workflow_agent = isinstance(agent, WorkflowAgent) if is_workflow_agent and agent.workflow._runner_context.has_checkpointing(): # pyright: ignore[reportPrivateUsage] @@ -436,53 +477,77 @@ def _validate_agent_configuration( ) resilient_background = bool(options and options.resilient_background) - if resilient_background and not is_workflow_agent: + if resilient_background and not is_workflow_agent and inner_background != "provider": raise RuntimeError( "resilient_background=True is only supported for workflow agents. " "Crash recovery cannot be provided for non-workflow agents." ) + if inner_background == "provider" and ( + not isinstance(agent, RawAgent) or getattr(cast(Any, agent).client, "STORES_BY_DEFAULT", None) is not True + ): + raise RuntimeError("Provider background requires a RawAgent with a storing, resumable Responses client.") if options and options.steerable_conversations and is_workflow_agent: raise RuntimeError( "steerable_conversations=True is only supported for non-workflow agents. " "Steering cannot be provided reliably for workflow agents." ) - uses_agent_server_history = history_source == "agent_server" + if is_workflow_agent and inner_history not in ("host", "legacy"): + raise ValueError("inner_history='service' and 'agent' are only supported for regular agents.") + + if not is_workflow_agent and isinstance(agent, RawAgent): + identity_defaults = ("session_id", "agent_session_id", "user_id", "call_id", "service_session_id") + if conflicting := [name for name in identity_defaults if agent.default_options.get(name) is not None]: + raise RuntimeError(f"Model defaults cannot supply Foundry platform identity: {', '.join(conflicting)}.") + + uses_agent_server_history = inner_history == "host" client_stores_by_default = False - if uses_agent_server_history and not is_workflow_agent: + if not is_workflow_agent and inner_history != "legacy": if not isinstance(agent, RawAgent): raise RuntimeError( - "history_source='agent_server' requires a RawAgent so hosting can enforce downstream " - "storage options. Construct ResponsesHostServer with history_source='agent' for a custom " - "SupportsAgentRun implementation." + "Explicit inner_history requires a RawAgent so hosting can enforce downstream storage. " + "For a custom SupportsAgentRun implementation, use the deprecated history_source='agent' " + "only for stored requests." ) - for provider in agent.context_providers: - if isinstance(provider, HistoryProvider) and provider.load_messages: - if _is_hosted_responses_history_sentinel(provider): - continue - raise RuntimeError( - "AgentServer response history is enabled, but the agent has a HistoryProvider " - "with load_messages=True. Remove that provider or construct ResponsesHostServer " - "with history_source='agent' to use the agent's regular history setup." - ) + + stores_by_default = getattr(cast(Any, agent).client, "STORES_BY_DEFAULT", None) + if not isinstance(stores_by_default, bool): + raise RuntimeError( + "Explicit inner_history requires the chat client to declare STORES_BY_DEFAULT " + "so hosting can enforce downstream storage behavior." + ) + client_stores_by_default = stores_by_default + + if inner_history == "service" and not stores_by_default: + raise RuntimeError( + "inner_history='service' requires a storing client that declares STORES_BY_DEFAULT=True." + ) + service_continuation_options = [ name - for name in ("conversation_id", "previous_response_id", "conversation") + for name in ("conversation_id", "previous_response_id", "conversation", "continuation_token", "response_id") if agent.default_options.get(name) is not None ] if service_continuation_options: raise RuntimeError( - "AgentServer response history is enabled, but the agent has downstream service continuation " - f"option(s): {', '.join(service_continuation_options)}. Remove them or construct " - "ResponsesHostServer with history_source='agent' to resume the downstream service conversation." + f"Explicit inner_history cannot use developer defaults for downstream continuation: " + f"{', '.join(service_continuation_options)}. Use history_source='agent' for legacy continuation." ) - stores_by_default = getattr(cast(Any, agent).client, "STORES_BY_DEFAULT", None) - if not isinstance(stores_by_default, bool): + + if inner_history in ("host", "service"): + for provider in agent.context_providers: + if isinstance(provider, HistoryProvider) and provider.load_messages: + if inner_history == "host" and _is_hosted_responses_history_sentinel(provider): + continue + raise RuntimeError( + "The selected inner_history conflicts with a load-enabled HistoryProvider. " + "Remove that provider or select inner_history='agent'." + ) + if inner_history != "service" and not stores_by_default and agent.default_options.get("store") is True: raise RuntimeError( - "history_source='agent_server' requires the agent's chat client to declare " - "STORES_BY_DEFAULT so hosting can enforce downstream storage behavior." + "The chat client does not store by default, but the agent sets a downstream store option. " + "Remove that developer-owned default rather than letting hosting change it." ) - client_stores_by_default = stores_by_default return _AgentConfiguration( workflow=is_workflow_agent, @@ -495,8 +560,6 @@ def _validate_agent_configuration( def _initialize_agent_history(agent: SupportsAgentRun, configuration: _AgentConfiguration) -> None: if not configuration.hosted_history or not isinstance(agent, RawAgent): return - if not configuration.client_stores_by_default: - agent.default_options.pop("store", None) if not any( _is_hosted_responses_history_sentinel(provider) for provider in cast(Sequence[ContextProvider], agent.context_providers) @@ -515,10 +578,15 @@ def __init__( prefix: str = "", options: ResponsesServerOptions | None = None, store: ResponseProviderProtocol | None = None, + response_store: ResponseProviderProtocol | None = None, agent_session_store_provider: StoreProvider[SessionStore] | None = None, checkpoint_store_provider: ContextScopedStoreProvider[CheckpointStorage] | None = None, function_approval_store_provider: StoreProvider[FunctionApprovalStore] | None = None, - history_source: Literal["agent_server", "agent"] = "agent_server", + inner_history: Literal["host", "service", "agent"] | None = None, + history_source: Literal["agent_server", "agent"] | None = None, + inner_background: Literal["host", "provider"] = "host", + prepare_options: OptionsHook | None = None, + unsupported_options: UnsupportedOptions = "warn", **kwargs: Any, ) -> None: """Initialize a ResponsesHostServer. @@ -528,20 +596,24 @@ def __init__( each request. Use a callable for agents that keep mutable state outside `AgentSession`. prefix: The URL prefix for the server. options: Optional server options. - store: Optional response store for input and history look up. + store: Deprecated alias for `response_store`. + response_store: Optional response store for caller-facing persistence and history. agent_session_store_provider: Optional provider for MAF agent session storage. If not provided, a default `AgentSessionStoreProvider` will be used. checkpoint_store_provider: Optional provider for workflow checkpoint storage. If not provided, a default `CheckpointStoreProvider` will be used. function_approval_store_provider: Optional provider for function approval storage. If not provided, a default `FunctionApprovalStoreProvider` will be used. - history_source: Source of conversation history supplied to the model for regular agents. - `"agent_server"` (default) uses the transcript from the configured response store, - requires a `RawAgent` whose client declares `STORES_BY_DEFAULT`, rejects load-enabled - agent history providers, and disables downstream service storage. - `"agent"` passes only the current request input and preserves the agent's normal - history-provider or service-storage behavior. AgentServer still manages Responses - API persistence through `store` in both modes. + inner_history: `"host"` (default) replays outer history with inner storage off; + `"service"` keeps the inner service continuation private; `"agent"` uses MAF history. + history_source: Deprecated alias. `"agent_server"` selects host history; `"agent"` + passes only new input and retains the developer's own history or service-storage + choice for stored requests. + inner_background: `"provider"` explicitly opts a storing, resumable client into its + background API; `"host"` (default) leaves provider background disabled. + prepare_options: Developer hook to remove or replace caller model options for an agent. + unsupported_options: `"warn"` (default), `"error"`, or `"ignore"` when a custom agent + cannot accept runtime model options. **kwargs: Additional keyword arguments. Note: @@ -569,37 +641,78 @@ def __init__( collected, and that isn't guaranteed to have happened in time. Raises: - ValueError: If `history_source` is not supported. + ValueError: If the history, background, or unsupported-options policy is invalid. RuntimeError: If the agent configuration conflicts with the selected history source, `resilient_background=True` is requested for a non-workflow agent, or `steerable_conversations=True` is requested for a workflow agent. """ - if history_source not in ("agent_server", "agent"): + if history_source is not None and history_source not in ("agent_server", "agent"): raise ValueError("history_source must be either 'agent_server' or 'agent'.") + if inner_history is not None and inner_history not in ("host", "service", "agent"): + raise ValueError("inner_history must be 'host', 'service', or 'agent'.") + if inner_history is not None and history_source is not None: + raise ValueError("inner_history and history_source cannot be combined.") + if inner_background not in ("host", "provider"): + raise ValueError("inner_background must be 'host' or 'provider'.") + if store is not None and response_store is not None: + raise ValueError("Pass response_store instead of store; they cannot be combined.") + + if inner_history is not None: + resolved_history: _HistoryMode = inner_history + elif history_source == "agent": + resolved_history = "legacy" + else: + resolved_history = "host" + if inner_background == "provider" and resolved_history != "service": + raise ValueError("Provider background requires inner_history='service'.") + if inner_background == "provider" and options and options.steerable_conversations: + raise ValueError("Provider background and steerable_conversations cannot be combined.") validate_agent_source(agent) resolved_agent = agent if is_agent(agent) else None configuration = ( - _validate_agent_configuration(resolved_agent, history_source, options) + _validate_agent_configuration(resolved_agent, resolved_history, options, inner_background=inner_background) if resolved_agent is not None else None ) # No caller-owned agent state is mutated until all validation and base-host construction succeed. - super().__init__(prefix=prefix, options=options, store=store, **kwargs) + super().__init__( + prefix=prefix, + options=options, + store=response_store if response_store is not None else store, + **kwargs, + ) + if options and options.steerable_conversations: + from azure.ai.agentserver.core.tasks import set_resilient_tasks_enabled + + set_resilient_tasks_enabled(True) self._agent_source = agent self._agent = resolved_agent self._configuration = configuration - self._history_source: Literal["agent_server", "agent"] = history_source + self._inner_history: _HistoryMode = resolved_history + self._inner_background: Literal["host", "provider"] = inner_background + self._prepare_options = prepare_options + self._unsupported_options = validate_unsupported_options(unsupported_options) self._host_options = options self._uses_agent_server_history = ( - configuration.agent_server_history if configuration is not None else history_source == "agent_server" + configuration.agent_server_history if configuration is not None else resolved_history == "host" ) self._resilient_background = bool(options and options.resilient_background) if resolved_agent is not None and configuration is not None: _initialize_agent_history(resolved_agent, configuration) + if history_source is not None: + warnings.warn( + "history_source is deprecated; use inner_history='host', 'service', or 'agent'. " + "history_source='agent' retains developer-controlled downstream storage for stored requests.", + DeprecationWarning, + stacklevel=2, + ) + if store is not None: + warnings.warn("store= is deprecated; use response_store=.", DeprecationWarning, stacklevel=2) + # Storage providers self._checkpoint_storage_provider = ( CheckpointStoreProvider() if checkpoint_store_provider is None else checkpoint_store_provider @@ -669,22 +782,30 @@ async def _handle_response( yield response_event_stream.emit_created() yield response_event_stream.emit_in_progress() - if self.config.is_hosted: - try: - FoundryRequestScope.from_context(self.config, get_request_context()) - except RuntimeError as exc: - logger.error("Invalid Foundry hosted request context: %s", exc) - for event in self._emit_failure(response_event_stream, None, exc): - yield event - return - terminal_event: ResponseStreamEvent | None = None - agent = await resolve_agent(self._agent_source) - configuration = self._configuration or _validate_agent_configuration( - agent, self._history_source, self._host_options - ) - if self._configuration is None: - _initialize_agent_history(agent, configuration) + try: + scope = FoundryRequestScope.from_context( + self.config, get_request_context(), local_session_id=context.response_id + ) + agent = await resolve_agent(self._agent_source) + configuration = self._configuration or _validate_agent_configuration( + agent, self._inner_history, self._host_options, inner_background=self._inner_background + ) + if self._configuration is None: + _initialize_agent_history(agent, configuration) + hosted_request: HostedResponseRequest | None = None + if configuration.workflow: + if self._prepare_options is not None: + raise ValueError("prepare_options is only supported for regular agents.") + else: + hosted_request = HostedResponseRequest(request, context, scope, response_run_options(request)) + await prepare_response_options(hosted_request, self._prepare_options) + validate_request_options(hosted_request.options) + except Exception as exc: + logger.error("Failed to prepare hosted Responses request", exc_info=(type(exc), exc, exc.__traceback__)) + for event in self._emit_failure(response_event_stream, None, exc): + yield event + return async with AsyncExitStack() as resources: inner = self._handle_prepared_response( @@ -695,6 +816,7 @@ async def _handle_response( agent, configuration, resources, + hosted_request, ) try: async for event in inner: @@ -720,6 +842,7 @@ async def _handle_prepared_response( agent: SupportsAgentRun, configuration: _AgentConfiguration, resources: AsyncExitStack, + hosted_request: HostedResponseRequest | None, ) -> AsyncGenerator[ResponseStreamEvent | ResponseCheckpointEvent]: # Lazy-enter the agent (and any MCP tools it owns). The MCP client wraps gateway # consent failures (and other connection-time errors) in AgentFrameworkException; if @@ -756,6 +879,13 @@ async def _handle_prepared_response( yield event return + if request.get("store") is False: + for event in self._emit_failure( + response_event_stream, None, ValueError("OAuth consent continuation requires store=true.") + ): + yield event + return + if not configuration.workflow: try: request_context = get_request_context() @@ -772,7 +902,17 @@ async def _handle_prepared_response( f"previous_response_id={previous_response_id}." ) session = agent.create_session() - await session_storage.set(context.conversation_id or context.response_id, session) + if previous_response_id is not None and context.conversation_id is None: + if session.service_session_id is not None and session.state.get( + _HOSTED_SOURCE_CONVERSATION_KEY + ): + raise ValueError("A service-managed downstream conversation cannot be forked.") + session.state.pop(_HOSTED_SOURCE_CONVERSATION_KEY, None) + if context.conversation_id is not None: + session.state[_HOSTED_SOURCE_CONVERSATION_KEY] = context.conversation_id + await session_storage.set(context.response_id, session) + if context.conversation_id is not None: + await session_storage.set(context.conversation_id, session) except Exception as save_error: logger.error( "Failed to persist the Agent Framework session for OAuth consent", @@ -810,6 +950,8 @@ async def _handle_prepared_response( cast(WorkflowAgent, agent), ) else: + if hosted_request is None: + raise RuntimeError("A regular agent requires a prepared Responses request.") inner = self._handle_inner_agent( request, context, @@ -818,6 +960,7 @@ async def _handle_prepared_response( cancellation_signal, agent, configuration, + hosted_request, ) try: @@ -921,6 +1064,7 @@ async def _handle_inner_agent( cancellation_signal: asyncio.Event, agent: SupportsAgentRun, configuration: _AgentConfiguration, + hosted_request: HostedResponseRequest, ) -> AsyncGenerator[ResponseStreamEvent]: """Handle a regular (non-workflow) agent. @@ -929,20 +1073,80 @@ async def _handle_inner_agent( into a terminal ``response.failed`` event (draining the tracker so the SSE stream stays well-formed). """ - if context.is_recovery: - logger.warning( - "Recovery mode is not supported for non-workflow agents. " - "The agent will restart from the original input." - ) + provider_background = self._inner_background == "provider" and request.get("background") is True + if context.is_recovery and not provider_background: + raise RuntimeError("A non-resumable agent cannot be replayed after a process crash.") + + stored = request.get("store") is not False request_messages_task: asyncio.Task[list[Message]] | None = None try: + if not stored: + if not isinstance(agent, RawAgent): + raise RuntimeError( + "store=false cannot guarantee that a custom agent will disable inner storage; " + "use a RawAgent or store=true." + ) + continuation_defaults = [ + name + for name in ( + "conversation_id", + "previous_response_id", + "conversation", + "continuation_token", + "response_id", + ) + if agent.default_options.get(name) is not None + ] + if continuation_defaults: + raise RuntimeError( + "store=false cannot use developer defaults for downstream service continuation: " + + ", ".join(continuation_defaults) + ) + if any( + isinstance(provider, HistoryProvider) + and not isinstance(provider, InMemoryHistoryProvider) + and (provider.store_inputs or provider.store_context_messages or provider.store_outputs) + for provider in agent.context_providers + ): + raise RuntimeError( + "store=false cannot prevent an external HistoryProvider from persisting responses. " + "Use an InMemoryHistoryProvider or store=true." + ) + if self._inner_history == "legacy": + stores_by_default = getattr(cast(Any, agent).client, "STORES_BY_DEFAULT", None) + if not isinstance(stores_by_default, bool): + raise RuntimeError( + "store=false with history_source='agent' requires a client declaring STORES_BY_DEFAULT; " + "select an explicit inner_history mode or use store=true." + ) + if not stores_by_default and agent.default_options.get("store") is True: + raise RuntimeError( + "store=false with history_source='agent' cannot safely override a storing " + "default on a client that does not declare storage support." + ) + request_context = get_request_context() - approval_storage = self._function_approval_storage_provider.get_store( - config=self.config, platform_context=request_context + approval_storage = ( + self._function_approval_storage_provider.get_store(config=self.config, platform_context=request_context) + if stored + else None ) - session_storage = self._session_storage_provider.get_store( - config=self.config, platform_context=request_context + previous_response_id = request.get("previous_response_id") + session_load_id = ( + context.response_id + if context.is_recovery and provider_background + else context.conversation_id or previous_response_id + ) + if not stored and session_load_id is not None and self._inner_history == "service": + raise ValueError( + "store=false cannot resume service-managed history; start a new one-shot request " + "or use inner_history='host' or 'agent'." + ) + session_storage = ( + self._session_storage_provider.get_store(config=self.config, platform_context=request_context) + if stored or (session_load_id is not None and self._inner_history in ("agent", "legacy")) + else None ) # Load the caller's input items and prior conversation history concurrently with the @@ -956,16 +1160,36 @@ async def _handle_inner_agent( ) ) - previous_response_id = request.get("previous_response_id") - session_load_id = context.conversation_id or previous_response_id - session = await session_storage.get(session_load_id) if session_load_id is not None else None + session = ( + await session_storage.get(session_load_id) + if session_storage is not None and session_load_id is not None + else None + ) if session is None: - if previous_response_id is not None and context.conversation_id is None: + if context.is_recovery and provider_background: + raise RuntimeError("Provider background was interrupted before its continuation token was stored.") + if stored and previous_response_id is not None and context.conversation_id is None: raise RuntimeError( f"Cannot find an existing agent session for previous_response_id={previous_response_id}." ) session = agent.create_session() - session_save_id = context.conversation_id or context.response_id + if not stored and self._inner_history == "legacy" and session.service_session_id is not None: + raise ValueError( + "store=false cannot continue legacy downstream service history; start a new one-shot request " + "or select an explicit inner_history mode." + ) + if previous_response_id is not None and context.conversation_id is None and not context.is_recovery: + if session.service_session_id is not None and session.state.get(_HOSTED_SOURCE_CONVERSATION_KEY): + raise ValueError("A service-managed downstream conversation cannot be forked.") + if stored and self._inner_history in ("service", "legacy") and session.service_session_id is not None: + if session.state.get(_HOSTED_SERVICE_CHILD_KEY): + raise ValueError("A service-managed downstream response cannot be forked.") + if session_storage is None: + raise RuntimeError("Service history requires agent session storage.") + session.state[_HOSTED_SERVICE_CHILD_KEY] = context.response_id + await session_storage.set(previous_response_id, session) + session.state.pop(_HOSTED_SERVICE_CHILD_KEY) + session.state.pop(_HOSTED_SOURCE_CONVERSATION_KEY, None) except BaseException as ex: # Session preparation failed (or the request was cancelled / the stream closed — # neither of which is an Exception). Cancel and drain the in-flight message-loading @@ -995,7 +1219,8 @@ async def _handle_inner_agent( "messages": messages, "session": session, } - chat_options, are_options_set = _to_chat_options(request) + chat_options = cast(ChatOptions[Any], dict(hosted_request.options)) + are_options_set = bool(chat_options) if configuration.agent_server_history: if configuration.client_stores_by_default: # The response provider already owns the transcript used for this run. Keep a @@ -1004,23 +1229,65 @@ async def _handle_inner_agent( else: # Do not pass a storage option to clients that do not advertise support for it. chat_options.pop("store", None) + elif self._inner_history == "service": + chat_options["store"] = stored + if not stored: + session.service_session_id = None + elif self._inner_history == "agent": + if configuration.client_stores_by_default: + chat_options["store"] = False + session.service_session_id = None + elif ( + not stored + and isinstance(agent, RawAgent) + and ( + getattr(cast(Any, agent).client, "STORES_BY_DEFAULT", False) + or agent.default_options.get("store") is not None + ) + ): + chat_options["store"] = False if isinstance(agent, RawAgent): run_kwargs["options"] = chat_options elif are_options_set: - logger.warning("Agent doesn't support runtime options. They will be ignored.") - - # Non-workflow agents can't be resilient, so there is no exit_for_recovery path here: - # both shutdown and steering/cancel just wind the turn down once observed. - agent_stream = _SignalledIterator( - agent.run(stream=True, **run_kwargs), # type: ignore[reportUnknownMemberType] - context.shutdown, - cancellation_signal, - ) - async with aclosing(agent_stream): - async for update in agent_stream: + if self._unsupported_options == "error": + raise TypeError("The hosted agent does not accept caller runtime options.") + if self._unsupported_options == "warn": + logger.warning("Agent doesn't support runtime options. They will be ignored.") + + inner_stream: ResponseStream[AgentResponseUpdate, AgentResponse[Any]] | None = None + if provider_background: + if session_storage is None or not isinstance(agent, RawAgent): + raise RuntimeError("Provider background requires a stored MAF agent session.") + updates = self._provider_background_updates( + agent=cast(RawAgent[ChatOptions[Any]], agent), + messages=messages, + session=session, + session_storage=session_storage, + options=chat_options, + context=context, + cancellation_signal=cancellation_signal, + ) + else: + inner_stream = agent.run(stream=True, **run_kwargs) # type: ignore[reportUnknownMemberType] + updates = _SignalledIterator(inner_stream, context.shutdown, cancellation_signal) + async with aclosing(updates): + async for update in updates: + if not stored and any( + content.type in ("function_approval_request", "oauth_consent_request") + or content.user_input_request + for content in update.contents + ): + raise ValueError("Approval and user-input continuation requires store=true.") async for event in tracker.handle_update(update, approval_storage=approval_storage): yield event + if inner_stream is not None and isinstance(updates, _SignalledIterator) and not updates.signalled: + final = await inner_stream.get_final_response() + if final.continuation_token is not None and final.finish_reason is None: + raise RuntimeError( + "The inner agent returned an unfinished provider response; " + "configure inner_background='provider' to resume it." + ) except (asyncio.CancelledError, GeneratorExit): request_interrupted = True raise @@ -1034,21 +1301,50 @@ async def _handle_inner_agent( if configuration.hosted_history: session.state.pop(_HOSTED_RESPONSES_HISTORY_SOURCE_ID, None) - # A service ID here means the client stored the turn despite the forced `store=False`. - # Do not persist a session that could resume that unreconciled history on a later turn. - stored_output_violation = configuration.agent_server_history and session.service_session_id is not None + # Never persist a session that could resume an inner response contrary to the + # history mode or the caller's explicit storage decision. + stored_output_violation = ( + self._inner_history in ("host", "agent") or not stored + ) and session.service_session_id is not None if stored_output_violation: misconfigured = RuntimeError( - "The agent's chat client stored this turn server-side while AgentServer response history " - "is supplying the conversation. Configure the client to honor store=False, or construct " - "ResponsesHostServer with history_source='agent' to use the agent's regular history setup." + "The agent's chat client stored this turn server-side despite store=False for the inner model. " + "Configure the client to honor store=False, or use inner_history='service' with store=true." ) logger.error("%s", misconfigured) if request_failure is None and not request_interrupted: request_failure = misconfigured + if ( + self._inner_history == "service" + and stored + and _HOSTED_PROVIDER_STATE_KEY not in session.state + and session.service_session_id is None + and request_failure is None + and not request_interrupted + ): + request_failure = RuntimeError( + "inner_history='service' requires the chat client to return a service continuation ID." + ) try: - if not stored_output_violation: - await session_storage.set(session_save_id, session) + final_provider_state = not provider_background or ( + request_failure is None + and not request_interrupted + and _HOSTED_PROVIDER_STATE_KEY not in session.state + ) + superseded_by_steering = bool(self._host_options and self._host_options.steerable_conversations) and ( + cancellation_signal.is_set() and not context.client_cancelled and not context.shutdown.is_set() + ) + if stored and session_storage is not None and not stored_output_violation and final_provider_state: + if context.conversation_id is not None: + session.state[_HOSTED_SOURCE_CONVERSATION_KEY] = context.conversation_id + await session_storage.set(context.response_id, session) + if ( + context.conversation_id is not None + and not superseded_by_steering + and not request_interrupted + and request_failure is None + ): + await session_storage.set(context.conversation_id, session) except Exception as save_error: save_failure = save_error if request_interrupted: @@ -1069,6 +1365,84 @@ async def _handle_inner_agent( elif save_failure is not None: raise save_failure + async def _provider_background_updates( + self, + *, + agent: RawAgent[ChatOptions[Any]], + messages: list[Message], + session: AgentSession, + session_storage: SessionStore, + options: ChatOptions[Any], + context: ResponseContext, + cancellation_signal: asyncio.Event, + ) -> AsyncGenerator[AgentResponseUpdate]: + """Keep the inner provider token private while the outer response ID is polled.""" + + async def run_provider( + options: ChatOptions[Any], *, input_messages: list[Message] | None + ) -> AgentResponse[Any]: + try: + return await agent.run(input_messages, session=session, options=options) + except Exception as exc: + raise RuntimeError("Inner provider background execution failed; inspect the host logs.") from exc + + async def save_private_state() -> None: + try: + await session_storage.set(context.response_id, session) + except Exception as exc: + raise RuntimeError("Could not save private provider background state; inspect the host logs.") from exc + + if context.is_recovery: + saved = session.state.get(_HOSTED_PROVIDER_STATE_KEY) + if not isinstance(saved, Mapping): + raise RuntimeError("Cannot recover a provider background job without its stored continuation token.") + saved_payload = cast(Mapping[str, Any], saved) + if saved_payload.get("outer_response_id") != context.response_id: + raise RuntimeError("Cannot recover a provider background job without its stored continuation token.") + token = saved_payload.get("continuation_token") + if not isinstance(token, Mapping): + raise RuntimeError("The stored provider continuation token is invalid.") + continuation_token: Mapping[str, Any] = cast(Mapping[str, Any], token) + else: + first = await run_provider(cast(ChatOptions[Any], {**options, "background": True}), input_messages=messages) + if first.continuation_token is None: + for update in _agent_response_updates(first, context.response_id): + yield update + return + if not isinstance(first.continuation_token, Mapping): + raise RuntimeError("The provider returned a continuation token that cannot be persisted.") + continuation_token = cast(Mapping[str, Any], first.continuation_token) + session.state[_HOSTED_PROVIDER_STATE_KEY] = { + "outer_response_id": context.response_id, + "continuation_token": dict(continuation_token), + } + await save_private_state() + + while True: + if context.shutdown.is_set() and self._resilient_background: + await context.exit_for_recovery() + if cancellation_signal.is_set() and context.client_cancelled: + return + await asyncio.sleep(2) + current = await run_provider( + cast(ChatOptions[Any], {"continuation_token": continuation_token, "store": True}), + input_messages=None, + ) + if current.continuation_token is None: + session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) + await save_private_state() + for update in _agent_response_updates(current, context.response_id): + yield update + return + if not isinstance(current.continuation_token, Mapping): + raise RuntimeError("The provider returned a continuation token that cannot be persisted.") + continuation_token = cast(Mapping[str, Any], current.continuation_token) + session.state[_HOSTED_PROVIDER_STATE_KEY] = { + "outer_response_id": context.response_id, + "continuation_token": dict(continuation_token), + } + await save_private_state() + async def _handle_inner_workflow( self, request: CreateResponse, diff --git a/python/packages/foundry_hosting/tests/test_request.py b/python/packages/foundry_hosting/tests/test_request.py new file mode 100644 index 00000000000..8bf62efbd15 --- /dev/null +++ b/python/packages/foundry_hosting/tests/test_request.py @@ -0,0 +1,120 @@ +# Copyright (c) Microsoft. All rights reserved. + +"""Only the Responses request view and option policy are part of this slice.""" + +from __future__ import annotations + +from typing import Any, cast +from unittest.mock import AsyncMock, MagicMock + +import pytest +from azure.ai.agentserver.responses import ResponseContext +from azure.ai.agentserver.responses.models import CreateResponse + +from agent_framework_foundry_hosting import HostedResponseRequest +from agent_framework_foundry_hosting._request import ( + prepare_response_options, + response_run_options, + validate_request_options, + validate_unsupported_options, +) +from agent_framework_foundry_hosting._scope import FoundryRequestScope + + +def test_native_options_are_translated_before_flattened_extra_body() -> None: + request = cast( + CreateResponse, + { + "input": "hello", + "model": "test-model", + "store": True, + "background": False, + "conversation": "outer-conversation", + "agent_session_id": "forged-sandbox", + "session_id": "caller-session", + "user": "forged-user", + "user_id": "forged-user", + "call_id": "forged-call", + "conversation_id": "inner-conversation", + "service_session_id": "private-session", + "continuation_token": {"response_id": "private"}, + "max_output_tokens": 300, + "max_tokens": 150, + "parallel_tool_calls": False, + "reasoning": {"effort": "high"}, + "slogan_style": "retro", + }, + ) + + assert response_run_options(request) == { + "model": "test-model", + "max_tokens": 150, + "allow_multiple_tool_calls": False, + "reasoning": {"effort": "high"}, + "slogan_style": "retro", + } + + +def test_nested_extra_body_is_also_overlaid_last_without_reserved_fields() -> None: + request = cast( + CreateResponse, + { + "max_output_tokens": 300, + "extra_body": { + "max_tokens": 150, + "session_id": "forged", + "continuation_token": {"response_id": "private"}, + }, + }, + ) + assert response_run_options(request) == {"max_tokens": 150} + with pytest.raises(TypeError, match="extra_body must be a mapping"): + response_run_options(cast(CreateResponse, {"extra_body": ["not a mapping"]})) + + +async def test_hook_uses_a_request_copy_and_does_not_mutate_defaults() -> None: + payload = cast(CreateResponse, {"input": "hello", "metadata": {"source": "caller"}}) + context = MagicMock(spec=ResponseContext) + context.response_id = "outer-id" + context.conversation_id = None + context.get_input_text = AsyncMock(return_value="hello") + scope = FoundryRequestScope(session_id="sandbox", user_id="user", call_id="call", is_hosted=True) + request = HostedResponseRequest(payload, context, scope, {"temperature": 0.8}) + original = {"temperature": 0.8} + + async def remove_temperature(view: HostedResponseRequest, options: dict[str, Any]) -> dict[str, Any]: + assert view.scope is scope + assert view.response_id == "outer-id" + assert await view.get_input_text() == "hello" + options.pop("temperature") + view.request["metadata"]["source"] = "copied" + return options + + await prepare_response_options(request, remove_temperature) + assert dict(request.options) == {} + assert original == {"temperature": 0.8} + assert payload["metadata"] == {"source": "caller"} + + +async def test_hook_rejects_reserved_fields_and_non_mapping_result() -> None: + context = MagicMock(spec=ResponseContext) + context.response_id = "outer-id" + context.conversation_id = None + scope = FoundryRequestScope(session_id="sandbox", user_id=None, call_id=None, is_hosted=False) + request = HostedResponseRequest(cast(CreateResponse, {"input": "hello"}), context, scope, {}) + + await prepare_response_options(request, lambda _view, _options: {"store": True, "session_id": "forged"}) + with pytest.raises(ValueError, match="session_id, store"): + validate_request_options(request.options) + with pytest.raises(TypeError, match="must return a mapping"): + await prepare_response_options(request, lambda _view, _options: cast(Any, None)) + + +@pytest.mark.parametrize("mode", ["ignore", "warn", "error"]) +def test_supported_unsupported_option_modes(mode: str) -> None: + assert validate_unsupported_options(mode) == mode + + +def test_unknown_unsupported_option_mode_fails_at_construction() -> None: + with pytest.raises(ValueError, match="unsupported_options"): + validate_unsupported_options("silent") diff --git a/python/packages/foundry_hosting/tests/test_responses.py b/python/packages/foundry_hosting/tests/test_responses.py index c80e8dc24db..e51c7199674 100644 --- a/python/packages/foundry_hosting/tests/test_responses.py +++ b/python/packages/foundry_hosting/tests/test_responses.py @@ -15,8 +15,11 @@ import json import logging import os +import threading +import time import uuid from collections.abc import AsyncGenerator, AsyncIterator, Awaitable, Callable, Generator, Mapping, Sequence +from concurrent.futures import ThreadPoolExecutor from contextlib import aclosing, asynccontextmanager from dataclasses import dataclass from importlib import import_module @@ -55,7 +58,7 @@ tool, ) from agent_framework.ag_ui import AgentFrameworkAgent, InMemoryAGUIThreadSnapshotStore -from agent_framework.openai import OpenAIChatClient +from agent_framework.openai import OpenAIChatClient, OpenAIChatOptions, OpenAIContinuationToken from azure.ai.agentserver.core import FoundryAgentRequestContext, get_request_context from azure.ai.agentserver.responses import ( FileResponseStore, @@ -70,6 +73,7 @@ from mcp import McpError from mcp.types import ErrorData from openai import AsyncOpenAI, DefaultAsyncHttpxClient +from starlette.testclient import TestClient from typing_extensions import Any from agent_framework_foundry_hosting import ResponsesHostServer @@ -147,6 +151,11 @@ async def _raising_updates( raise RuntimeError(message) +async def _single_state_update(session: AgentSession) -> AsyncIterator[AgentResponseUpdate]: + session.state["turn"] = 1 + yield AgentResponseUpdate(contents=[Content.from_text("recorded")], role="assistant") + + class _AgentProtocolMock(MagicMock): id = "test-agent" name: str | None = "Test Agent" @@ -424,6 +433,17 @@ async def set(self, session_id: str, session: AgentSession) -> None: raise OSError("session storage is full") +class _ConflictingConversationStore(SessionStore): + def __init__(self) -> None: + super().__init__() + self.fail_conversation = False + + async def set(self, session_id: str, session: AgentSession) -> None: + if self.fail_conversation and session_id == "conversation-head": + raise RuntimeError("Another request advanced this agent session; reload before writing.") + await super().set(session_id, session) + + _SESSION_STORE_UNSET = object() @@ -431,7 +451,7 @@ def _make_server(agent: Any, **kwargs: Any) -> ResponsesHostServer: """Create a ResponsesHostServer, optionally replacing its private store for tests.""" session_store = kwargs.pop("session_store", _SESSION_STORE_UNSET) response_store = kwargs.pop("response_store", InMemoryResponseProvider()) - server = ResponsesHostServer(agent, store=response_store, **kwargs) + server = ResponsesHostServer(agent, response_store=response_store, **kwargs) if session_store is not _SESSION_STORE_UNSET: provider = MagicMock(spec=AgentSessionStoreProvider) provider.get_store.return_value = cast(SessionStore | None, session_store) @@ -595,6 +615,14 @@ def _parse_sse_events(body: str) -> list[dict[str, Any]]: return events +def _failure_message(events: Sequence[Any]) -> str: + failed_events = [event for event in events if isinstance(event, Mapping) and event.get("type") == "response.failed"] + assert len(failed_events) == 1 + response = cast(Mapping[str, Any], failed_events[0]["response"]) + error = cast(Mapping[str, Any], response["error"]) + return cast(str, error["message"]) + + async def test_agui_service_storage_conversation_mode_sends_only_incremental_provider_input() -> None: """A provider conversation stays authoritative while AG-UI snapshots retain full history.""" hosted_agent_backend = _make_agent( @@ -933,7 +961,7 @@ def test_init_uses_default_store_providers(self) -> None: agent = _make_agent( response=AgentResponse(messages=[Message(role="assistant", contents=[Content.from_text("hi")])]) ) - server = ResponsesHostServer(agent, store=InMemoryResponseProvider()) + server = ResponsesHostServer(agent, response_store=InMemoryResponseProvider()) assert isinstance( server._session_storage_provider, # pyright: ignore[reportPrivateUsage] @@ -1058,6 +1086,57 @@ def test_init_rejects_steerable_conversations_for_workflow_agent(self) -> None: options=ResponsesServerOptions(steerable_conversations=True), ) + def test_steerable_agent_enables_multi_turn_task_manager(self) -> None: + with patch("azure.ai.agentserver.core.tasks.set_resilient_tasks_enabled") as enable: + _make_server(_make_agent(), options=ResponsesServerOptions(steerable_conversations=True)) + enable.assert_called_once_with(True) + + def test_provider_background_rejects_steering_and_wrong_history_mode(self) -> None: + agent = Agent(client=_ServiceStorageRecordingClient()) + with pytest.raises(ValueError, match="inner_history='service'"): + _make_server(agent, inner_background="provider") + with pytest.raises(ValueError, match="cannot be combined"): + _make_server( + agent, + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(steerable_conversations=True), + ) + + def test_legacy_aliases_warn_once_per_host(self) -> None: + agent = Agent(client=_ServiceStorageRecordingClient(), default_options=OpenAIChatOptions(store=True)) + with pytest.warns(DeprecationWarning) as recorded: + server = ResponsesHostServer( + agent, + history_source="agent", + store=InMemoryResponseProvider(), + ) + assert server is not None + assert len(recorded) == 2 + assert any("history_source" in str(warning.message) for warning in recorded) + assert any("response_store" in str(warning.message) for warning in recorded) + + def test_explicit_history_rejects_legacy_alias_and_invalid_policy(self) -> None: + agent = _make_agent() + with pytest.raises(ValueError, match="cannot be combined"): + _make_server(agent, inner_history="agent", history_source="agent") + with pytest.raises(ValueError, match="unsupported_options"): + _make_server(agent, unsupported_options="silent") + + @pytest.mark.parametrize( + "identity_field", ["session_id", "agent_session_id", "user_id", "call_id", "service_session_id"] + ) + @pytest.mark.parametrize("legacy", [False, True]) + def test_agent_defaults_cannot_supply_platform_identity(self, identity_field: str, legacy: bool) -> None: + agent = Agent( + client=_ServiceStorageRecordingClient(), + default_options=cast(Any, {identity_field: "forged"}), + ) + selection = {"history_source": "agent"} if legacy else {"inner_history": "host"} + + with pytest.raises(RuntimeError, match="Model defaults cannot supply Foundry platform identity"): + _make_server(agent, **selection) + async def test_previous_response_requires_existing_agent_session(self) -> None: agent = _make_agent() server = _make_server(agent, session_store=SessionStore()) @@ -1223,20 +1302,30 @@ async def test_agent_server_history_disables_service_storage(self) -> None: assert stored is not None assert stored.service_session_id is None - async def test_agent_server_history_removes_store_for_non_storing_client(self) -> None: + async def test_host_history_rejects_storing_default_without_mutating_it(self) -> None: client = _RecordingHistoryClient() agent = Agent( client=client, name="Non-Storing Agent", default_options={"store": True}, # pyrefly: ignore[bad-argument-type] ) - server = _make_server(agent, session_store=SessionStore()) + with pytest.raises(RuntimeError, match="Remove that developer-owned default"): + _make_server(agent, session_store=SessionStore()) + assert agent.default_options["store"] is True + assert agent.context_providers == [] - response = await _post(server, input_text="first") + async def test_host_history_preserves_non_storing_client_default(self) -> None: + client = _RecordingHistoryClient() + agent = Agent( + client=client, + default_options={"store": False}, # pyrefly: ignore[bad-argument-type] + ) + server = _make_server(agent) + response = await _post(server) assert response.json()["status"] == "completed" - assert "store" not in agent.default_options - assert "store" not in client.options[0] + assert agent.default_options["store"] is False + assert client.options[0]["store"] is False async def test_agent_history_does_not_forward_runtime_options_to_custom_agent(self) -> None: agent = _StrictCustomAgent() @@ -1291,6 +1380,688 @@ async def test_agent_history_preserves_service_storage(self) -> None: assert stored is not None assert stored.service_session_id == "service-thread-1" + async def test_service_history_preserves_private_continuation(self) -> None: + client = _ServiceStorageRecordingClient() + agent = Agent(client=client, default_options=OpenAIChatOptions(store=False)) + store = SessionStore() + server = _make_server(agent, session_store=store, inner_history="service") + + first = await _post(server, input_text="first") + second = await _post(server, input_text="second", previous_response_id=first.json()["id"]) + + assert second.json()["status"] == "completed" + assert [[message.text for message in call] for call in client.calls] == [["first"], ["second"]] + assert client.store_options == [True, True] + assert client.conversation_ids == [None, "service-thread-1"] + assert "service-thread-1" not in str(second.json()) + assert agent.default_options["store"] is False + stored = await store.get(second.json()["id"]) + assert stored is not None and stored.service_session_id == "service-thread-1" + + async def test_async_agent_factory_reuses_private_service_session_across_requests(self) -> None: + clients: list[_ServiceStorageRecordingClient] = [] + + async def create_agent() -> SupportsAgentRun: + client = _ServiceStorageRecordingClient() + clients.append(client) + return Agent(client=client, default_options=OpenAIChatOptions(store=True)) + + store = SessionStore() + server = _make_server(create_agent, session_store=store, inner_history="service") + first = await _post(server, input_text="first") + second = await _post(server, input_text="second", previous_response_id=first.json()["id"]) + + assert second.json()["status"] == "completed" + assert len(clients) == 2 + assert [client.store_options for client in clients] == [[True], [True]] + assert [client.conversation_ids for client in clients] == [[None], ["service-thread-1"]] + saved = await store.get(second.json()["id"]) + assert saved is not None and saved.service_session_id == "service-thread-1" + + async def test_service_history_rejects_second_child_of_provider_response(self) -> None: + client = _ServiceStorageRecordingClient() + server = _make_server(Agent(client=client), session_store=SessionStore(), inner_history="service") + first = await _post(server, input_text="first") + second = await _post(server, input_text="second", previous_response_id=first.json()["id"]) + branch = await _post(server, input_text="fork", previous_response_id=first.json()["id"]) + + assert second.json()["status"] == "completed" + assert branch.json()["status"] == "failed" + assert "cannot be forked" in branch.json()["error"]["message"] + assert len(client.calls) == 2 + + async def test_explicit_agent_history_uses_provider_without_inner_storage(self) -> None: + client = _ServiceStorageRecordingClient() + history = InMemoryHistoryProvider() + agent = Agent(client=client, context_providers=[history], default_options=OpenAIChatOptions(store=True)) + store = SessionStore() + server = _make_server(agent, session_store=store, inner_history="agent") + + first = await _post(server, input_text="first") + second = await _post(server, input_text="second", previous_response_id=first.json()["id"]) + + assert second.json()["status"] == "completed" + assert [[message.text for message in call] for call in client.calls] == [ + ["first"], + ["first", "recorded", "second"], + ] + assert client.store_options == [False, False] + assert agent.default_options["store"] is True + saved = await store.get(second.json()["id"]) + assert saved is not None and history.source_id in saved.state + assert saved.service_session_id is None + + @pytest.mark.parametrize("mode", ["host", "service", "agent", "legacy"]) + async def test_store_false_neither_saves_session_nor_stores_inner_response(self, mode: str) -> None: + client = _ServiceStorageRecordingClient() + agent = Agent(client=client, default_options=OpenAIChatOptions(store=True)) + store = SessionStore() + selection = {"history_source": "agent"} if mode == "legacy" else {"inner_history": mode} + server = _make_server(agent, session_store=store, **selection) + approvals = MagicMock(spec=FunctionApprovalStoreProvider) + server._function_approval_storage_provider = approvals # pyright: ignore[reportPrivateUsage] + + response = await _post_json(server, {"input": "one shot", "store": False, "model": "test-model"}) + + assert response.json()["status"] == "completed", response.json() + assert await store.get(response.json()["id"]) is None + assert client.store_options == [False] + assert client.conversation_ids == [None] + assert agent.default_options["store"] is True + provider = cast(MagicMock, server._session_storage_provider) # pyright: ignore[reportPrivateUsage] + provider.get_store.assert_not_called() + approvals.get_store.assert_not_called() + + async def test_store_false_rejects_external_history_and_custom_agent(self) -> None: + history = _PerServiceCallHistoryProvider() + agent = Agent(client=_RecordingHistoryClient(), context_providers=[history]) + server = _make_server(agent, history_source="agent") + response = await _post_json(server, {"input": "one shot", "store": False}) + assert response.json()["status"] == "failed" + assert "external HistoryProvider" in response.json()["error"]["message"] + + custom = _StrictCustomAgent() + server = _make_server(custom, history_source="agent") + response = await _post_json(server, {"input": "one shot", "store": False}) + assert response.json()["status"] == "failed" + assert "custom agent" in response.json()["error"]["message"] + assert custom.calls == [] + + async def test_store_false_legacy_continuation_fails_instead_of_using_service_history(self) -> None: + client = _ServiceStorageRecordingClient() + server = _make_server( + Agent(client=client, default_options=OpenAIChatOptions(store=True)), + session_store=SessionStore(), + history_source="agent", + ) + first = await _post(server) + request = CreateResponse(input="cannot resume", store=False, previous_response_id=first.json()["id"]) + context = ResponseContext(response_id="one-shot", mode_flags=MagicMock()) + events = [event async for event in server._handle_response(request, context, asyncio.Event())] + + assert "store=false cannot continue legacy" in _failure_message(events) + assert len(client.calls) == 1 + + async def test_extra_options_overlay_then_developer_hook_preserves_defaults(self) -> None: + client = _RecordingHistoryClient() + agent = Agent(client=client, default_options=OpenAIChatOptions(temperature=0.2, max_tokens=256)) + defaults = dict(agent.default_options) + + def prepare_options(_request: Any, options: dict[str, Any]) -> dict[str, Any]: + options.pop("temperature") + return options + + server = _make_server(agent, prepare_options=prepare_options) + response = await _post_json( + server, + { + "input": "hello", + "model": "test-model", + "temperature": 0.8, + "max_output_tokens": 300, + "max_tokens": 150, + "session_id": "forged-sandbox", + "call_id": "forged-call", + }, + ) + + assert response.json()["status"] == "completed" + assert client.options[0]["temperature"] == 0.2 + assert client.options[0]["max_tokens"] == 150 + assert "session_id" not in client.options[0] + assert "call_id" not in client.options[0] + assert agent.default_options == defaults + + async def test_hook_cannot_reintroduce_host_owned_storage_or_identity(self) -> None: + agent = _make_agent() + server = _make_server( + agent, + prepare_options=lambda _request, options: {**options, "store": True, "agent_session_id": "forged"}, + ) + response = await _post_json(server, {"input": "one shot", "store": False}) + + assert response.json()["status"] == "failed" + assert "host-controlled" in response.json()["error"]["message"] + agent.run.assert_not_called() + + async def test_unsupported_options_policy_rejects_custom_agent(self) -> None: + agent = _StrictCustomAgent() + server = _make_server(agent, history_source="agent", unsupported_options="error") + response = await _post_json(server, {"input": "hello", "temperature": 0.8}) + + assert response.json()["status"] == "failed" + assert "does not accept caller runtime options" in response.json()["error"]["message"] + assert agent.calls == [] + + @pytest.mark.parametrize(("policy", "warn"), [("warn", True), ("ignore", False)]) + async def test_custom_agent_unsupported_options_warning_policy( + self, policy: str, warn: bool, caplog: pytest.LogCaptureFixture + ) -> None: + agent = _StrictCustomAgent() + server = _make_server(agent, history_source="agent", unsupported_options=policy) + + with caplog.at_level(logging.WARNING): + response = await _post(server, temperature=0.5) + + assert response.json()["status"] == "completed" + assert ("Agent doesn't support runtime options" in caplog.text) is warn + + @pytest.mark.parametrize( + ("finish_reason", "expected_status"), + [(None, "failed"), ("stop", "completed"), ("length", "incomplete")], + ) + async def test_streaming_continuation_requires_an_unfinished_final_response( + self, finish_reason: str | None, expected_status: str, monkeypatch: pytest.MonkeyPatch + ) -> None: + agent = Agent(client=_ServiceStorageRecordingClient()) + + async def stream_updates() -> AsyncIterator[AgentResponseUpdate]: + yield AgentResponseUpdate(continuation_token=OpenAIContinuationToken(response_id="private-provider-token")) + if finish_reason is not None: + yield AgentResponseUpdate( + contents=[Content.from_text("finished")], + role="assistant", + finish_reason=cast(FinishReasonLiteral, finish_reason), + continuation_token=( + OpenAIContinuationToken(response_id="private-provider-token") + if finish_reason == "length" + else None + ), + ) + + monkeypatch.setattr( + agent, + "run", + MagicMock( + side_effect=lambda **_kwargs: ResponseStream(stream_updates(), finalizer=AgentResponse.from_updates) + ), + ) + server = _make_server(agent) + response = await _post_json(server, {"input": "hello", "store": True}) + + assert response.json()["status"] == expected_status, response.json() + assert "private-provider-token" not in str(response.json()) + + async def test_provider_background_keeps_private_token_under_outer_response( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + agent = Agent(client=_ServiceStorageRecordingClient(), default_options=OpenAIChatOptions(store=True)) + calls: list[dict[str, Any]] = [] + + async def run( + messages: Any = None, + *, + session: AgentSession, + options: Mapping[str, Any], + **kwargs: Any, + ) -> AgentResponse: + del messages, kwargs + calls.append(dict(options)) + if "continuation_token" not in options: + return AgentResponse( + messages=[], continuation_token=OpenAIContinuationToken(response_id="private-provider-token") + ) + session.service_session_id = "private-service-conversation" + return AgentResponse(messages=[Message(role="assistant", contents=[Content.from_text("finished")])]) + + monkeypatch.setattr(agent, "run", MagicMock(side_effect=run)) + store = SessionStore() + server = _make_server( + agent, + session_store=store, + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + request = CreateResponse(input="hello", store=True, background=True) + context = ResponseContext(response_id="outer-response", mode_flags=MagicMock()) + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + ): + events = [event async for event in server._handle_response(request, context, asyncio.Event())] + + assert [event.get("type") for event in events if isinstance(event, Mapping)][-1] == "response.completed" + assert [call.get("background") for call in calls] == [True, None] + assert calls[1]["continuation_token"] == {"response_id": "private-provider-token"} + assert "private-provider-token" not in str(events) + saved = await store.get("outer-response") + assert saved is not None + assert saved.service_session_id == "private-service-conversation" + assert "_foundry_provider_background" not in saved.state + + async def test_provider_recovery_without_saved_token_does_not_restart_job( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + agent = Agent(client=_ServiceStorageRecordingClient()) + run = MagicMock() + monkeypatch.setattr(agent, "run", run) + server = _make_server( + agent, + session_store=SessionStore(), + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + request = CreateResponse(input="hello", store=True, background=True) + context = ResponseContext(response_id="outer-response", mode_flags=MagicMock()) + context.is_recovery = True + + with patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])): + events = [event async for event in server._handle_response(request, context, asyncio.Event())] + + assert "before its continuation token was stored" in _failure_message(events) + run.assert_not_called() + + async def test_provider_recovery_polls_saved_token_without_reclaiming_parent( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + store = SessionStore() + await store.set("previous-outer-response", AgentSession(service_session_id="previous-service-session")) + calls: list[dict[str, Any]] = [] + + async def run( + messages: Any = None, + *, + session: AgentSession, + options: Mapping[str, Any], + **kwargs: Any, + ) -> AgentResponse: + del messages, kwargs + calls.append(dict(options)) + if "continuation_token" not in options: + return AgentResponse( + messages=[], continuation_token=OpenAIContinuationToken(response_id="private-provider-token") + ) + session.service_session_id = "next-service-session" + return AgentResponse(messages=[Message(role="assistant", contents=[Content.from_text("resumed")])]) + + first_agent = Agent(client=_ServiceStorageRecordingClient()) + monkeypatch.setattr(first_agent, "run", MagicMock(side_effect=run)) + first_server = _make_server( + first_agent, + session_store=store, + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + request = CreateResponse( + input="long job", store=True, background=True, previous_response_id="previous-outer-response" + ) + first_context = ResponseContext(response_id="outer-recover", mode_flags=MagicMock()) + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch( + "agent_framework_foundry_hosting._responses.asyncio.sleep", + new=AsyncMock(side_effect=ResponseExitForRecovery()), + ), + pytest.raises(ResponseExitForRecovery), + ): + _ = [event async for event in first_server._handle_response(request, first_context, asyncio.Event())] + + parent = await store.get("previous-outer-response") + assert parent is not None and parent.state["_foundry_service_child"] == "outer-recover" + token_session = await store.get("outer-recover") + assert token_session is not None + assert token_session.state["_foundry_provider_background"]["continuation_token"] == { + "response_id": "private-provider-token" + } + + resumed_agent = Agent(client=_ServiceStorageRecordingClient()) + monkeypatch.setattr(resumed_agent, "run", MagicMock(side_effect=run)) + resumed_server = _make_server( + resumed_agent, + session_store=store, + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + recovered_context = ResponseContext(response_id="outer-recover", mode_flags=MagicMock()) + recovered_context.is_recovery = True + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + ): + events = [ + event async for event in resumed_server._handle_response(request, recovered_context, asyncio.Event()) + ] + + assert [event.get("type") for event in events if isinstance(event, Mapping)][-1] == "response.completed" + assert [call.get("background") for call in calls] == [True, None] + saved = await store.get("outer-recover") + assert saved is not None and saved.service_session_id == "next-service-session" + assert "_foundry_provider_background" not in saved.state + assert "private-provider-token" not in str(events) + + async def test_provider_poll_errors_cannot_reveal_private_token( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + agent = Agent(client=_ServiceStorageRecordingClient()) + + async def run( + messages: Any = None, + *, + session: AgentSession, + options: Mapping[str, Any], + **kwargs: Any, + ) -> AgentResponse: + del messages, session, kwargs + if "continuation_token" not in options: + return AgentResponse( + messages=[], continuation_token=OpenAIContinuationToken(response_id="private-provider-token") + ) + raise RuntimeError("provider failed for private-provider-token") + + monkeypatch.setattr(agent, "run", MagicMock(side_effect=run)) + server = _make_server( + agent, + session_store=SessionStore(), + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + request = CreateResponse(input="hello", store=True, background=True) + context = ResponseContext(response_id="outer-response", mode_flags=MagicMock()) + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + ): + events = [event async for event in server._handle_response(request, context, asyncio.Event())] + + assert "Inner provider background execution failed" in _failure_message(events) + assert "private-provider-token" not in str(events) + + async def test_steering_keeps_superseded_snapshot_without_replacing_conversation_head(self) -> None: + store = SessionStore() + await store.set("conversation-head", AgentSession()) + first_continues = asyncio.Event() + calls = 0 + agent = _make_agent() + + def run_with_state(*args: Any, **kwargs: Any) -> ResponseStream[AgentResponseUpdate, AgentResponse]: + nonlocal calls + del args + calls += 1 + turn = calls + kwargs["session"].state["turn"] = turn + + async def updates() -> AsyncIterator[AgentResponseUpdate]: + yield AgentResponseUpdate(contents=[Content.from_text(f"turn {turn}")], role="assistant") + if turn == 1: + await first_continues.wait() + yield AgentResponseUpdate(contents=[Content.from_text("stale")], role="assistant") + + return ResponseStream(updates(), finalizer=AgentResponse.from_updates) + + agent.run = MagicMock(side_effect=run_with_state) + server = _make_server( + agent, + session_store=store, + options=ResponsesServerOptions(steerable_conversations=True), + ) + first = ResponseContext( + response_id="first-response", conversation_id="conversation-head", mode_flags=MagicMock() + ) + second = ResponseContext( + response_id="second-response", conversation_id="conversation-head", mode_flags=MagicMock() + ) + first_signal = asyncio.Event() + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch.object(ResponseContext, "get_history", new=AsyncMock(return_value=[])), + ): + first_handler = server._handle_response(CreateResponse(input="first"), first, first_signal) + async for event in first_handler: + if isinstance(event, Mapping) and event.get("type") == "response.output_text.delta": + break + newer = [ + event + async for event in server._handle_response(CreateResponse(input="second"), second, asyncio.Event()) + ] + first_signal.set() + older = [event async for event in first_handler] + + assert isinstance(newer[-1], Mapping) and newer[-1]["type"] == "response.completed" + assert isinstance(older[-1], Mapping) and older[-1]["type"] == "response.completed" + first_snapshot = await store.get("first-response") + second_snapshot = await store.get("second-response") + conversation_head = await store.get("conversation-head") + assert first_snapshot is not None and first_snapshot.state["turn"] == 1 + assert second_snapshot is not None and second_snapshot.state["turn"] == 2 + assert conversation_head is not None and conversation_head.state["turn"] == 2 + + async def test_conversation_write_conflict_keeps_response_snapshot(self) -> None: + store = _ConflictingConversationStore() + await store.set("conversation-head", AgentSession()) + store.fail_conversation = True + agent = _make_agent() + agent.run = MagicMock( + side_effect=lambda **kwargs: ResponseStream( + _single_state_update(kwargs["session"]), finalizer=AgentResponse.from_updates + ) + ) + server = _make_server(agent, session_store=store) + context = ResponseContext( + response_id="conflicting-response", conversation_id="conversation-head", mode_flags=MagicMock() + ) + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch.object(ResponseContext, "get_history", new=AsyncMock(return_value=[])), + ): + events = [ + event + async for event in server._handle_response(CreateResponse(input="hello"), context, asyncio.Event()) + ] + + assert "Another request advanced" in _failure_message(events) + snapshot = await store.get("conflicting-response") + conversation_head = await store.get("conversation-head") + assert snapshot is not None and snapshot.state["turn"] == 1 + assert conversation_head is not None and "turn" not in conversation_head.state + + async def test_http_outer_storage_is_independent_of_inner_service_history(self) -> None: + client = _ServiceStorageRecordingClient() + store = SessionStore() + server = _make_server(Agent(client=client), inner_history="service", session_store=store) + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=server), base_url="http://test") as http: + stored = await http.post("/responses", json={"input": "remember me", "store": True}) + stored_id = stored.json()["id"] + retrieved = await http.get(f"/responses/{stored_id}") + unstored = await http.post("/responses", json={"input": "one shot", "store": False}) + missing = await http.get(f"/responses/{unstored.json()['id']}") + + assert stored.json()["status"] == "completed" + assert retrieved.status_code == 200 and retrieved.json()["id"] == stored_id + assert unstored.json()["status"] == "completed" + assert missing.status_code == 404 + assert client.store_options == [True, False] + saved = await store.get(stored_id) + assert saved is not None and saved.service_session_id == "service-thread-1" + assert await store.get(unstored.json()["id"]) is None + + @pytest.mark.parametrize("mode", ["service", "agent"]) + async def test_http_outer_background_polling_is_independent_of_inner_history(self, mode: str) -> None: + client = _ServiceStorageRecordingClient() + history = [InMemoryHistoryProvider()] if mode == "agent" else [] + server = _make_server( + Agent(client=client, context_providers=history), session_store=SessionStore(), inner_history=mode + ) + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=server), base_url="http://test") as http: + pending = await http.post("/responses", json={"input": "background", "store": True, "background": True}) + assert pending.status_code == 200 + response_id = pending.json()["id"] + for _ in range(100): + final = await http.get(f"/responses/{response_id}") + if final.json()["status"] == "completed": + break + await asyncio.sleep(0.02) + else: + pytest.fail(f"Background response {response_id} did not finish.") + rejected = await http.post("/responses", json={"input": "invalid", "store": False, "background": True}) + + assert final.json()["id"] == response_id + assert "service-thread-1" not in str(final.json()) + assert client.store_options == ([True] if mode == "service" else [False]) + assert rejected.status_code == 400 + + async def test_http_background_polling_and_unstored_stream_use_outer_response_id(self) -> None: + started = asyncio.Event() + release = asyncio.Event() + agent = _make_agent() + + async def updates() -> AsyncIterator[AgentResponseUpdate]: + started.set() + await release.wait() + yield AgentResponseUpdate(contents=[Content.from_text("finished")], role="assistant") + + agent.run = MagicMock( + side_effect=lambda **_kwargs: ResponseStream(updates(), finalizer=AgentResponse.from_updates) + ) + server = _make_server(agent) + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=server), base_url="http://test", timeout=10 + ) as http: + pending = await http.post("/responses", json={"input": "long", "store": True, "background": True}) + assert pending.status_code == 200 + outer_id = pending.json()["id"] + assert pending.json()["status"] in ("queued", "in_progress") + await asyncio.wait_for(started.wait(), timeout=5) + release.set() + for _ in range(100): + result = await http.get(f"/responses/{outer_id}") + if result.json()["status"] == "completed": + break + await asyncio.sleep(0.02) + else: + pytest.fail("Outer background response did not complete.") + stream = await http.post("/responses", json={"input": "one shot", "store": False, "stream": True}) + completed = [ + event["data"]["response"] + for event in _parse_sse_events(stream.text) + if event["event"] == "response.completed" + ] + assert len(completed) == 1 + missing = await http.get(f"/responses/{completed[0]['id']}") + + assert result.json()["id"] == outer_id + assert "finished" in str(result.json()["output"]) + assert missing.status_code == 404 + assert agent.run.call_args_list[0].kwargs["options"].get("background") is None + + def test_http_steering_uses_task_manager_and_keeps_latest_conversation_state( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + monkeypatch.setenv("AGENTSERVER_STATE_ROOT", str(tmp_path / "state")) + first_output = threading.Event() + release = threading.Event() + turn_number = 0 + store = SessionStore() + agent = _make_agent() + + def run(*args: Any, **kwargs: Any) -> ResponseStream[AgentResponseUpdate, AgentResponse]: + nonlocal turn_number + del args + turn_number += 1 + turn = turn_number + kwargs["session"].state["turn"] = turn + + async def updates() -> AsyncIterator[AgentResponseUpdate]: + yield AgentResponseUpdate(contents=[Content.from_text(f"turn {turn}")], role="assistant") + if turn == 1: + await asyncio.to_thread(release.wait) + yield AgentResponseUpdate(contents=[Content.from_text("stale")], role="assistant") + + return ResponseStream(updates(), finalizer=AgentResponse.from_updates) + + agent.run = MagicMock(side_effect=run) + server = _make_server( + agent, + session_store=store, + response_store=FileResponseStore(storage_dir=tmp_path / "responses"), + options=ResponsesServerOptions(steerable_conversations=True), + ) + + async def observe_output(request: Any, context: Any, cancellation_signal: Any) -> AsyncIterator[Any]: + async for event in server._handle_response(request, context, cancellation_signal): + if isinstance(event, Mapping) and event.get("type") == "response.output_text.delta": + first_output.set() + yield event + + server.response_handler(observe_output) + with TestClient(server) as http, ThreadPoolExecutor(max_workers=1) as executor: + try: + first_future = executor.submit( + http.post, + "/responses", + json={ + "input": "first", + "store": True, + "stream": True, + "background": True, + "conversation": "http-steering", + }, + ) + assert first_output.wait(timeout=5), "The first turn did not produce streamed output." + second = http.post( + "/responses", + json={"input": "second", "store": True, "background": True, "conversation": "http-steering"}, + ) + assert second.status_code == 200 + assert second.json()["status"] in ("queued", "in_progress") + first_stream = first_future.result(timeout=8) + finally: + release.set() + + assert first_stream.status_code == 200 + first_events = _parse_sse_events(first_stream.text) + first_completions = [ + event["data"]["response"] for event in first_events if event["event"] == "response.completed" + ] + assert len(first_completions) == 1 + first_id = first_completions[0]["id"] + deadline = time.monotonic() + 10 + while time.monotonic() < deadline: + latest_response = http.get(f"/responses/{second.json()['id']}") + if latest_response.json()["status"] in ("completed", "incomplete", "failed"): + break + time.sleep(0.05) + else: + pytest.fail(f"Steered response {second.json()['id']} did not finish.") + first_retrieved = http.get(f"/responses/{first_id}") + + assert latest_response.json()["status"] == "completed" + assert first_retrieved.json()["status"] == "completed" + assert turn_number == 2 + snapshot = asyncio.run(store.get(first_id)) + latest = asyncio.run(store.get(second.json()["id"])) + assert snapshot is not None and snapshot.state["turn"] == 1 + assert latest is not None and latest.state["turn"] == 2 + async def test_agent_history_uses_in_memory_history_from_session_store(self) -> None: client = _RecordingHistoryClient() history = InMemoryHistoryProvider() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/.env.example b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/.env.example index 4d268b931b4..6bc4afcd8c3 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/.env.example +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/.env.example @@ -1,2 +1,3 @@ FOUNDRY_PROJECT_ENDPOINT="..." -AZURE_AI_MODEL_DEPLOYMENT_NAME="..." \ No newline at end of file +AZURE_AI_MODEL_DEPLOYMENT_NAME="..." +FOUNDRY_AGENT_NAME="agent-framework-agent-basic-responses" \ No newline at end of file diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md index 27d2e33cc69..0f4b5417ffe 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md @@ -1,52 +1,61 @@ -# What this sample demonstrates +# Responses agents: history, storage, and options -An [Agent Framework](https://github.com/microsoft/agent-framework) agent hosted using the **Responses protocol**. +The **outer** Responses request decides whether the hosted response is stored. The developer separately selects +where the **inner** agent gets its conversation history: -## How It Works +| Entry point | `inner_history` | Model history | +| --- | --- | --- | +| [main.py](main.py) | `"host"` | The outer Responses transcript; the inner client runs with `store=False`. | +| [service_history.py](service_history.py) | `"service"` | Only new input goes to the model; its private `service_session_id` persists for later turns. | +| [agent_history.py](agent_history.py) | `"agent"` | `InMemoryHistoryProvider` loads from `AgentSession.state`; inner client storage is off. | +| [options.py](options.py) | `"host"` | The hook removes the caller's token limit so the agent default is used. | +| [provider_background.py](provider_background.py) | `"service"` | Opt-in provider background with a private recovery token. | -### Model Integration +Run one entry point at a time. The deployment manifest targets `main.py`; select another script to deploy a different +mode. Set `FOUNDRY_PROJECT_ENDPOINT` and `AZURE_AI_MODEL_DEPLOYMENT_NAME` in `.env`, then run `python main.py`. +Follow the [parent hosting guide](../../README.md) for local and deployed setup. -The agent uses `FoundryChatClient` from the Agent Framework to create a Responses client from the project endpoint and model deployment. The agent supports both streaming (SSE events) and non-streaming (JSON) response modes. +## Outer response storage and background -See [main.py](main.py) for the full implementation. +`store=True` makes a response retrievable at `GET /responses/{response.id}` and available for continuation using +`previous_response_id` or a `conversation`. It does **not** choose inner history. With `store=False`, the response +is one-shot: the host writes no MAF session, approval, or conversation state, and it does not request inner service +storage. Application-owned tools and other external services can still have their own side effects. If a custom +agent or external history provider cannot guarantee that boundary, the host rejects the unstored request. -### Agent Hosting +`background=True` requires `store=True` and returns the **outer** `response.id` as the polling handle. Normal agent +background runs inside AgentServer even if the chat client cannot run in the background. Without durable inner +continuation, a process crash can leave such a response unfinished. The optional +[provider_background.py](provider_background.py) opts a storing Responses client into its *separate* background +mode: only this mode persists the provider's private token and polls it until completion/recovery. It is incompatible +with steering. The deployed identity needs Foundry User permission on the project for private provider polling. -The agent is hosted using the [Agent Framework](https://github.com/microsoft/agent-framework) with the `ResponsesHostServer`, which provisions a REST API endpoint compatible with the OpenAI Responses protocol. - -## Running the Agent Host - -Follow the instructions in the [Running the Agent Host Locally](../../README.md#running-the-agent-host-locally) section of the README in the parent directory to run the agent host. - -## Interacting with the agent - -> Depending on how you run the agent host, you can invoke the agent using `curl` (`Invoke-WebRequest` in PowerShell) or `azd`. Please refer to the [parent README](../../README.md) for more details. Use this README for sample queries you can send to the agent. - -Send a POST request to the server with a JSON body containing an `"input"` field to interact with the agent. For example: +[client.py](client.py) shows stored conversation turns, an unstored request, and background polling. It also needs +`FOUNDRY_AGENT_NAME` and Azure CLI authentication. For a local host, a simple multi-turn request is: ```bash -curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" -d '{"input": "Hi"}' +curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" \ + -d '{"input": "Hello", "store": true, "conversation": "my-conversation"}' ``` -The server will respond with a JSON object containing the response text and a response ID. You can use this response ID to continue the conversation in subsequent requests. - -### Multi-turn conversation - -To have a multi-turn conversation with the agent, include the previous response id in the request body. For example: - -```bash -curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" -d '{"input": "How are you?", "previous_response_id": "REPLACE_WITH_PREVIOUS_RESPONSE_ID"}' -``` - -When deployed to Foundry, reuse the same platform-provided agent session to continue -the agent's private state. The request's `previous_response_id` selects a Responses -continuation; it does not identify the Foundry sandbox. Hosted state is isolated by -the platform's user and sandbox IDs, so putting those IDs in the JSON body cannot -recover a different sandbox's state. After upgrading from unscoped hosted state, -start a fresh conversation: older MAF session data is not read by the new default -stores. See [State store](../../../../../packages/foundry_hosting/README.md#state-store) -for the ID mapping and migration implications. - -## Deploying the Agent to Foundry - -To host the agent on Foundry, follow the instructions in the [Deploying the Agent to Foundry](../../README.md#deploying-the-agent-to-foundry) section of the README in the parent directory. +Send another request with the same `conversation` to continue. When deployed, keep the same Foundry sandbox: +`agent_session_id` is a **platform** session, not the outer `response.id` or the private +`AgentSession.service_session_id`. A `conversation` binds that sandbox; with a bare `previous_response_id`, also +forward the earlier response's `agent_session_id`. Older unscoped hosted MAF state is not migrated; start a new +conversation after upgrading (see the [package state guide](../../../../../packages/foundry_hosting/README.md#state-store)). + +## Options and compatibility + +Native CreateResponse generation fields become MAF run options (`max_output_tokens` becomes `max_tokens` and +`parallel_tool_calls` becomes `allow_multiple_tool_calls`). Flattened OpenAI `extra_body` values overlay translated +keys **last**. The developer's `prepare_options(request, options)` hook can then remove or replace *caller* options; +removing one exposes the agent's own unchanged `default_options`. In [options.py](options.py), +`max_output_tokens=300` plus `extra_body={"max_tokens": 150}` becomes `max_tokens=150` before the hook removes it; +the model receives the agent's `max_tokens=256` default instead. Platform IDs, storage flags, and private +continuation tokens are never caller model options; the hook cannot add them back. + +`history_source="agent_server"` and `history_source="agent"` remain available with a per-host deprecation warning. +For **stored** requests, the latter preserves the former behavior: only new input goes to the agent, and its +developer-owned defaults may choose either a HistoryProvider **or downstream service storage**. It does **not** +silently become `inner_history="agent"`. Prefer the explicit mode in new code. `store=` as a constructor parameter +is a deprecated alias for `response_store=` (the outer storage *backend*, not the caller's `store` flag). diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py new file mode 100644 index 00000000000..9b7df61ec8d --- /dev/null +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py @@ -0,0 +1,31 @@ +# Copyright (c) Microsoft. All rights reserved. + +"""Use MAF history in AgentSession instead of replaying Responses history.""" + +import os + +from agent_framework import Agent, InMemoryHistoryProvider +from agent_framework.foundry import FoundryChatClient +from agent_framework_foundry_hosting import ResponsesHostServer +from azure.identity import DefaultAzureCredential +from dotenv import load_dotenv + +load_dotenv() + + +def main() -> None: + agent = Agent( + client=FoundryChatClient( + project_endpoint=os.environ["FOUNDRY_PROJECT_ENDPOINT"], + model=os.environ["AZURE_AI_MODEL_DEPLOYMENT_NAME"], + credential=DefaultAzureCredential(), + ), + instructions="Be concise.", + context_providers=[InMemoryHistoryProvider()], + default_options={"store": False}, + ) + ResponsesHostServer(agent=agent, inner_history="agent").run() + + +if __name__ == "__main__": + main() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/client.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/client.py new file mode 100644 index 00000000000..20f34d8f134 --- /dev/null +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/client.py @@ -0,0 +1,50 @@ +# Copyright (c) Microsoft. All rights reserved. + +"""Call a deployed Responses agent with stored, one-shot, and background requests.""" + +import os +from time import sleep + +from azure.ai.projects import AIProjectClient +from azure.identity import AzureCliCredential +from dotenv import load_dotenv + +load_dotenv() + + +def main() -> None: + with ( + AzureCliCredential() as credential, + AIProjectClient( + endpoint=os.environ["FOUNDRY_PROJECT_ENDPOINT"], + credential=credential, + allow_preview=True, + ) as project, + ): + openai = project.get_openai_client(agent_name=os.environ["FOUNDRY_AGENT_NAME"]) + conversation = openai.conversations.create() + first = openai.responses.create(input="Introduce yourself.", conversation=conversation.id, store=True) + second = openai.responses.create( + input="Answer briefly.", + conversation=conversation.id, + store=True, + max_output_tokens=300, + extra_body={"max_tokens": 150}, + ) + print(f"Stored response {first.id}; continued with {second.id}: {second.output_text}") + + one_shot = openai.responses.create(input="Say hello.", store=False) + print(f"Unstored response (id cannot be retrieved): {one_shot.output_text}") + + background = openai.responses.create(input="Write a report.", store=True, background=True) + response_id = background.id + while background.status in ("queued", "in_progress"): + sleep(2) + background = openai.responses.retrieve(response_id) + if background.status != "completed": + raise RuntimeError(f"Background response {response_id} ended with status {background.status}.") + print(f"Background response {response_id}: {background.output_text}") + + +if __name__ == "__main__": + main() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/main.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/main.py index ff1d5a3437e..0f607754d33 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/main.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/main.py @@ -1,17 +1,19 @@ # Copyright (c) Microsoft. All rights reserved. +"""Host a Responses agent with outer history and stateless model calls.""" + import os from agent_framework import Agent -from agent_framework.foundry import FoundryChatClient, ResponsesHostServer +from agent_framework.foundry import FoundryChatClient +from agent_framework_foundry_hosting import ResponsesHostServer from azure.identity import DefaultAzureCredential from dotenv import load_dotenv -# Load environment variables from .env file load_dotenv() -def main(): +def main() -> None: client = FoundryChatClient( project_endpoint=os.environ["FOUNDRY_PROJECT_ENDPOINT"], model=os.environ["AZURE_AI_MODEL_DEPLOYMENT_NAME"], @@ -21,13 +23,9 @@ def main(): agent = Agent( client=client, instructions="You are a friendly assistant. Keep your answers brief.", - # History will be managed by the hosting infrastructure, thus there - # is no need to store history by the service. Learn more at: - # https://developers.openai.com/api/reference/resources/responses/methods/create - default_options={"store": False}, ) - server = ResponsesHostServer(agent) + server = ResponsesHostServer(agent=agent, inner_history="host") server.run() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/options.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/options.py new file mode 100644 index 00000000000..8b94e03b61a --- /dev/null +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/options.py @@ -0,0 +1,43 @@ +# Copyright (c) Microsoft. All rights reserved. + +"""Let a developer option hook override a caller's model-option selection.""" + +import os +from typing import Any + +from agent_framework import Agent +from agent_framework.foundry import FoundryChatClient +from agent_framework.openai import OpenAIChatOptions +from agent_framework_foundry_hosting import HostedResponseRequest, ResponsesHostServer +from azure.identity import DefaultAzureCredential +from dotenv import load_dotenv + +load_dotenv() + + +def prepare_options(_request: HostedResponseRequest, options: dict[str, Any]) -> dict[str, Any]: + """Use the agent's output-token default instead of the caller's override.""" + options.pop("max_tokens", None) + return options + + +def main() -> None: + agent = Agent( + client=FoundryChatClient( + project_endpoint=os.environ["FOUNDRY_PROJECT_ENDPOINT"], + model=os.environ["AZURE_AI_MODEL_DEPLOYMENT_NAME"], + credential=DefaultAzureCredential(), + ), + instructions="Be concise.", + default_options=OpenAIChatOptions(max_tokens=256), + ) + ResponsesHostServer( + agent=agent, + inner_history="host", + prepare_options=prepare_options, + unsupported_options="warn", + ).run() + + +if __name__ == "__main__": + main() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/provider_background.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/provider_background.py new file mode 100644 index 00000000000..08db7670cd0 --- /dev/null +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/provider_background.py @@ -0,0 +1,39 @@ +# Copyright (c) Microsoft. All rights reserved. + +"""Opt into provider background processing without exposing its continuation token.""" + +import os + +from agent_framework import Agent +from agent_framework.foundry import FoundryChatClient +from agent_framework_foundry_hosting import ResponsesHostServer +from azure.ai.agentserver.responses import ResponsesServerOptions +from azure.identity import DefaultAzureCredential +from dotenv import load_dotenv + +load_dotenv() + + +def create_agent() -> Agent: + return Agent( + client=FoundryChatClient( + project_endpoint=os.environ["FOUNDRY_PROJECT_ENDPOINT"], + model=os.environ["AZURE_AI_MODEL_DEPLOYMENT_NAME"], + credential=DefaultAzureCredential(), + ), + instructions="Write thorough research reports.", + default_options={"store": True}, + ) + + +def main() -> None: + ResponsesHostServer( + agent=create_agent, + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(resilient_background=True), + ).run() + + +if __name__ == "__main__": + main() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/service_history.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/service_history.py new file mode 100644 index 00000000000..9fc52edb6da --- /dev/null +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/service_history.py @@ -0,0 +1,30 @@ +# Copyright (c) Microsoft. All rights reserved. + +"""Use downstream service history while retaining outer Responses results.""" + +import os + +from agent_framework import Agent +from agent_framework.foundry import FoundryChatClient +from agent_framework_foundry_hosting import ResponsesHostServer +from azure.identity import DefaultAzureCredential +from dotenv import load_dotenv + +load_dotenv() + + +def main() -> None: + agent = Agent( + client=FoundryChatClient( + project_endpoint=os.environ["FOUNDRY_PROJECT_ENDPOINT"], + model=os.environ["AZURE_AI_MODEL_DEPLOYMENT_NAME"], + credential=DefaultAzureCredential(), + ), + instructions="Be concise.", + default_options={"store": True}, + ) + ResponsesHostServer(agent=agent, inner_history="service").run() + + +if __name__ == "__main__": + main() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md index 32720cee1cf..0c516a9cf38 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md @@ -1,106 +1,45 @@ -# What this sample demonstrates +# Steerable Responses agent -A steerable multi-turn [Agent Framework](https://github.com/microsoft/agent-framework) agent hosted using the -**Responses protocol**. The agent is asked to count down from a target number, pacing its own output with a short -remark before each number so a real response takes a while to fully generate. With `steerable_conversations=True`, -sending a new turn on the same conversation while the countdown is still streaming **cancels the in-progress turn** -and drains the new turn next. +This sample hosts a single, slowly streaming [Agent Framework](https://github.com/microsoft/agent-framework) +agent using the Responses protocol. The agent counts down with a remark before each number. A new turn sent to +the same `conversation` while the first is running can steer it: the host stops the older turn, saves its partial +response under its own `response.id`, then lets the newer turn advance the CAS-protected conversation state. +Steering is supported for regular agents, not for workflow agents or provider-native background mode. -Steering is only supported for non-workflow agents. Steering a workflow is conceptually undefined: a workflow's -graph may have loops or parallel branches with no single well-defined "current point" to cancel and resume from, -unlike an agent's strictly linear execution. `ResponsesHostServer` rejects `steerable_conversations=True` for a -workflow agent with `RuntimeError`. +[main.py](main.py) sets `inner_history="host"` and +`ResponsesServerOptions(steerable_conversations=True)`. The host enables AgentServer's multi-turn TaskManager +before startup; no agent-side steering code is needed. Both requests use `store=True` so the outer Responses API +can retrieve their result; the inner Foundry client still runs with `store=False`. The outer `response.id` remains +the background polling handle. -## How It Works - -### Agent - -The agent (see [main.py](main.py)) is a single `Agent` backed by `FoundryChatClient`, with no workflow and no custom -agent class. Its instructions ask it to count down one integer per line, prefacing each with a brief remark, so a -real streamed generation takes long enough for a second turn to arrive mid-stream. Compared to the -[Basic](../basic/) sample, the only differences are the instructions and passing -`ResponsesServerOptions(steerable_conversations=True)` -- steering needs no special agent-side code. - -### Agent Hosting - -The agent is hosted using the [Agent Framework](https://github.com/microsoft/agent-framework) `ResponsesHostServer`. -Setting `steerable_conversations=True` in `ResponsesServerOptions` lets a new turn on the same conversation chain -preempt a still-running one: - -- The framework signals the in-progress turn's handler via the 3rd positional `cancellation_signal` argument. -- The handler (here, the Agent Framework agent-hosting layer) checks that signal between streamed model updates and - winds the turn down promptly, letting it complete with whatever partial output it had already produced. -- The new turn is then drained with `context.is_steered_turn == True` and `context.pending_input_count` reflecting - how many further turns are still queued behind it. -- **A single, linear chain**: every turn after the first must reference the immediately preceding turn's `id` via - `previous_response_id`, and a `previous_response_id` that doesn't point at the latest turn is rejected with HTTP - 409 (`conversation_fork_not_supported`). Chain identity in turn also depends on session continuity: without an - explicit `conversation`, the server derives a session id per request, and it's only deterministic across turns - when a client forwards the `x-agent-session-id` header from a prior response back as `agent_session_id` on the - next one (see the `curl` walkthrough below) -- otherwise a later turn resolves to a different session and starts - a brand new response instead of steering the earlier one. Sending the same explicit `conversation` value on every - turn sidesteps this entirely: it makes the derived session id (and the chain itself) a deterministic function of - that id, so no header needs to be echoed back, and `previous_response_id` becomes unnecessary for continuity. - -## Running the Agent Host - -Follow the instructions in the [Running the Agent Host Locally](../../README.md#running-the-agent-host-locally) section -of the README in the parent directory to run the agent host. - -## Interacting with the agent - -> Depending on how you run the agent host, you can invoke the agent using `curl` (`Invoke-WebRequest` in PowerShell) or -> `azd`. Please refer to the [parent README](../../README.md) for more details. Use this README for sample queries you -> can send to the agent. - -Start a long background countdown and note the response `id` from the JSON body and the `x-agent-session-id` -response header: - -```bash -curl -i -X POST http://localhost:8088/responses -H "Content-Type: application/json" \ - -d '{"input": "Count down from 30, slowly and with commentary.", "store": true, "background": true}' -``` - -While it is still generating, send a second turn on the same conversation with `previous_response_id` set to steer -it to a new target. Without an explicit `conversation_id`, also forward the `x-agent-session-id` value from the -first response as `agent_session_id` -- otherwise this turn resolves to a different session and starts a brand new -response instead of steering the first one: +Run the agent as described in the [parent guide](../../README.md), then start a long **streamed** +background turn. Keep the SSE connection open while you send the next request; `response.created` supplies +the first polling ID: ```bash curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" \ - -d '{"input": "Actually, count down from 3 instead.", "store": true, "background": true, "previous_response_id": "REPLACE_WITH_FIRST_RESPONSE_ID", "agent_session_id": "REPLACE_WITH_X-AGENT-SESSION-ID_HEADER"}' + -d '{"input": "Count down from 30, slowly and with commentary.", "store": true, "stream": true, "background": true, "conversation": "my-conversation"}' ``` -This second request returns immediately with `"status": "queued"`. Polling the *first* response's id will show it -completed early, with fewer tokens than a full 30-count run. Polling the *second* response's id will show a fresh -countdown from 3. - -Alternatively, send an explicit `conversation` id on every turn instead of forwarding `x-agent-session-id`. This is -simpler and also works without `previous_response_id` at all, since the `conversation` id alone identifies the chain: +While it runs, submit the second turn against that **same** conversation: ```bash curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" \ - -d '{"input": "Count down from 30, slowly and with commentary.", "store": true, "background": true, "conversation": "my-conversation-id"}' - -curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" \ - -d '{"input": "Actually, count down from 3 instead.", "store": true, "background": true, "conversation": "my-conversation-id"}' + -d '{"input": "Actually, count down from 3 instead.", "store": true, "background": true, "conversation": "my-conversation"}' ``` -## Testing steering +The second request is accepted as queued or in progress. Poll `GET /responses/{response.id}` for each outer +response ID; the older response should complete with partial output, and the newer one should contain its own +countdown. An explicit `conversation` binds the Foundry sandbox and avoids the need to forward the +`x-agent-session-id` response header. If continuing with only `previous_response_id`, forward the prior response's +`agent_session_id` as well to stay in the same sandbox. -[verify_steering.py](verify_steering.py) runs the whole scenario end to end: it starts the server, kicks off a -background streaming countdown, waits for it to stream a minimum number of tokens, sends a second turn with a new -target via `previous_response_id`, and asserts that the second turn is accepted immediately as `"queued"`, that the -first turn completes early, and that the second (steered) turn's output contains the new target's countdown in -order. Because this sample calls a real model, the assertions here are intentionally loose rather than an exact -output match. +[verify_steering.py](verify_steering.py) runs this check against a real model, using a fresh conversation ID and +isolated temporary AgentServer state. It never deletes your existing `~/.agentserver` data. Real model timing +varies; its assertions are intentionally looser than the package's deterministic local steering tests: ```bash python verify_steering.py --first-target 30 --second-target 3 ``` -## Deploying the Agent to Foundry - -To host the agent on Foundry, follow the instructions in the -[Deploying the Agent to Foundry](../../README.md#deploying-the-agent-to-foundry) section of the README in the parent -directory. +For a Foundry deployment, follow the [parent deployment guide](../../README.md). diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py index 8ed3a5a6acc..0fd671d4f4d 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py @@ -43,7 +43,8 @@ def main() -> None: ) server = ResponsesHostServer( - agent, + agent=agent, + inner_history="host", options=ResponsesServerOptions(steerable_conversations=True), log_level="DEBUG", ) diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/verify_steering.py b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/verify_steering.py index 7deaababc8f..4d72373b84d 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/verify_steering.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/verify_steering.py @@ -17,21 +17,23 @@ import argparse import json import os -import shutil import subprocess import sys import threading import time import urllib.error import urllib.request +import uuid from pathlib import Path +from tempfile import TemporaryDirectory from typing import IO, Any HOST = "127.0.0.1" PORT = 8088 BASE_URL = f"http://{HOST}:{PORT}" SAMPLE_DIR = Path(__file__).parent -LOG_PATH = SAMPLE_DIR / "verify_steering.log" +STATE_DIR = TemporaryDirectory(prefix="af-steering-") +LOG_PATH = Path(STATE_DIR.name) / "verify_steering.log" def _http_get(path: str, timeout: float = 5.0) -> tuple[int, dict[str, Any]]: @@ -72,7 +74,7 @@ def _start_server(log_file: IO[str]) -> subprocess.Popen: # type: ignore return subprocess.Popen( [sys.executable, "main.py"], cwd=SAMPLE_DIR, - env={**os.environ, "PYTHONIOENCODING": "utf-8"}, + env={**os.environ, "PYTHONIOENCODING": "utf-8", "AGENTSERVER_STATE_ROOT": STATE_DIR.name}, stdout=log_file, stderr=subprocess.STDOUT, ) @@ -100,18 +102,10 @@ def _kill(server: subprocess.Popen) -> None: # type: ignore def _watch_sse(request: "urllib.request.Request | str", progress: dict[str, Any]) -> None: """Read an SSE stream from a streaming create POST and track its progress. - Tracks the response id (on ``response.created``), a running count of text delta events (a - single-agent response streams as one message, not discrete output items), and signals - ``progress["done"]`` on any terminal event. + Tracks the response id (on ``response.created``), text deltas, and terminal events. """ try: with urllib.request.urlopen(request) as resp: - # Without an explicit conversation_id, the session id (which scopes the conversation - # chain id used to attach a steered turn to the same task) must be forwarded by the - # caller on later turns -- otherwise each turn derives a different session id locally. - session_id = resp.headers.get("x-agent-session-id") - if session_id: - progress["session_id"] = session_id current_event: str | None = None for raw_line in resp: line = raw_line.decode("utf-8").rstrip("\n") @@ -149,14 +143,6 @@ def _extract_output_text(output_items: list[dict[str, Any]]) -> str: return "".join(parts) -def _clear_stale_state() -> None: - """Wipe ~/.agentserver so a prior run's task/queue state never leaks into this run.""" - state_root = Path.home() / ".agentserver" - if state_root.exists(): - shutil.rmtree(state_root, ignore_errors=True) - print(f" cleared stale state: {state_root}") - - def main() -> None: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--first-target", type=int, default=30, help="First turn's countdown starting value.") @@ -164,17 +150,16 @@ def main() -> None: parser.add_argument( "--min-deltas-before-steering", type=int, - default=15, + default=5, help="Minimum text delta events to observe on turn 1 before sending the steering turn.", ) args = parser.parse_args() - _clear_stale_state() - log_file = LOG_PATH.open("w", encoding="utf-8") print(f"Server logs (DEBUG level) are redirected to {LOG_PATH}.") print(f"[1/5] Starting server (first target={args.first_target}, second target={args.second_target})...") + conversation_id = f"steering-{uuid.uuid4().hex}" server = _start_server(log_file) # type: ignore print(f" PID: {server.pid}") try: @@ -191,6 +176,7 @@ def main() -> None: "store": True, "background": True, "stream": True, + "conversation": conversation_id, } first_data = json.dumps(first_payload).encode("utf-8") first_request = urllib.request.Request( @@ -215,6 +201,10 @@ def main() -> None: ) time.sleep(0.1) count_at_steer_time = first_progress["delta_count"] + if first_progress["done"].is_set(): + raise SystemExit( + "FAIL: first turn finished before steering; increase --first-target or reduce --min-deltas." + ) print(f" turn 1 text deltas observed before steering: {count_at_steer_time}") print(f"[4/5] Sending the steering turn (new target={args.second_target})...") @@ -223,17 +213,13 @@ def main() -> None: "store": True, "background": True, "stream": False, - "previous_response_id": first_id, + "conversation": conversation_id, } - # Forward the session id turn 1 was assigned so this turn resolves to the same - # conversation chain and is queued as a steer instead of starting a fresh task. - if "session_id" in first_progress: - second_payload["agent_session_id"] = first_progress["session_id"] status, body = _http_post("/responses", second_payload) - if status != 200 or body.get("status") != "queued": - raise SystemExit(f"FAIL: expected an immediate queued response for the steering turn, got: {body}") + if status != 200 or body.get("status") not in ("queued", "in_progress"): + raise SystemExit(f"FAIL: expected an immediately accepted steering turn, got: {body}") second_id = body["id"] - print(f" steering turn accepted immediately as queued; response id: {second_id}") + print(f" steering turn accepted as {body['status']}; response id: {second_id}") print("[5/5] Watching turn 1 end early and the steered turn complete...") first_progress["done"].wait(timeout=120) @@ -246,11 +232,12 @@ def main() -> None: first_text = _extract_output_text(first_final.get("output", [])) second_text = _extract_output_text(second_final.get("output", [])) - print(f" turn 1 final status: {first_final['status']}, {len(first_text)} character(s): {first_text}") - print(f" turn 2 final status: {second_final['status']}, {len(second_text)} character(s): {second_text}") + print(f" turn 1 final status: {first_final['status']}, {len(first_text)} character(s)") + print(f" turn 2 final status: {second_final['status']}, {len(second_text)} character(s)") - if "Serving steered turn" in LOG_PATH.read_text(encoding="utf-8"): - print(" confirmed 'Serving steered turn' in the server log.") + if "Serving steered turn" not in LOG_PATH.read_text(encoding="utf-8"): + raise SystemExit("FAIL: the follow-up was not processed as a steered turn.") + print(" confirmed 'Serving steered turn' in the server log.") if first_final["status"] != "completed": raise SystemExit( @@ -280,6 +267,7 @@ def main() -> None: search_from = idx + 1 print("PASS: the steering turn cancelled the in-progress countdown early and completed its own countdown.") + STATE_DIR.cleanup() if __name__ == "__main__": From be003ffc94144cb25c97d0101e5d30b27f9f57eb Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Mon, 28 Sep 2026 12:30:16 +0200 Subject: [PATCH 02/10] docs(python): clarify Foundry provider background recovery --- .../_responses.py | 17 ++++++++--------- 1 file changed, 8 insertions(+), 9 deletions(-) diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py index dce18a6b956..1fc8f757ac9 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py @@ -617,8 +617,8 @@ def __init__( **kwargs: Additional keyword arguments. Note: - 1. When `history_source="agent_server"`, the agent must not have a history provider - with `load_messages=True`, because history is managed by the hosting infrastructure. + 1. With `inner_history="host"` (or the deprecated `history_source="agent_server"`), + the agent must not have a load-enabled history provider: the host supplies the transcript. 2. Context providers must not keep required state only on their Python instances, because the hosting environment may get deactivated between requests. Provider state carried by `AgentSession`, including `InMemoryHistoryProvider` messages in @@ -626,12 +626,11 @@ def __init__( 3. The server owns the supplied agent instance and may add hosting-specific providers. Do not reuse the same agent with another host or invoke it directly after construction. An agent returned by a callable belongs to that request. - 4. Resiliency (resilient_background=True) is ONLY supported for workflows; constructing this - server with a non-workflow agent and `resilient_background=True` raises `RuntimeError`. - When resiliency is enabled, and the server crashes mid-response: - - Background responses are automatically re-invoked on server restart (client won't see the crash). - - Stream events are preserved for client reconnection. - - State is maintained across crashes. + 4. `resilient_background=True` supports legacy workflows and regular agents configured with + `inner_history="service", inner_background="provider"`. For provider background, only + a saved private continuation token can be polled after a crash; a crash before the token + is saved cannot be replayed safely. Other regular-agent runs are not crash-recoverable. + Legacy workflow background responses retain their checkpoint-based recovery behavior. 5. Steering (steerable_conversations=True) is ONLY supported for non-workflow agents; constructing this server with a workflow agent and `steerable_conversations=True` raises `RuntimeError`. Steering a workflow is conceptually undefined -- a workflow's graph may have loops or parallel @@ -643,7 +642,7 @@ def __init__( Raises: ValueError: If the history, background, or unsupported-options policy is invalid. RuntimeError: If the agent configuration conflicts with the selected history source, - `resilient_background=True` is requested for a non-workflow agent, or + `resilient_background=True` is requested for a regular agent without provider background, or `steerable_conversations=True` is requested for a workflow agent. """ if history_source is not None and history_source not in ("agent_server", "agent"): From 57e2628306a91a00e09813a0be0e52e4208e80ce Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Mon, 28 Sep 2026 13:47:24 +0200 Subject: [PATCH 03/10] fix(python): stabilize Responses hosting CI checks --- python/packages/foundry_hosting/tests/test_responses.py | 8 +++++--- .../foundry-hosted-agents/responses/basic/client.py | 4 +++- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/python/packages/foundry_hosting/tests/test_responses.py b/python/packages/foundry_hosting/tests/test_responses.py index e51c7199674..cb5accd8782 100644 --- a/python/packages/foundry_hosting/tests/test_responses.py +++ b/python/packages/foundry_hosting/tests/test_responses.py @@ -18,6 +18,7 @@ import threading import time import uuid +import warnings from collections.abc import AsyncGenerator, AsyncIterator, Awaitable, Callable, Generator, Mapping, Sequence from concurrent.futures import ThreadPoolExecutor from contextlib import aclosing, asynccontextmanager @@ -1106,15 +1107,16 @@ def test_provider_background_rejects_steering_and_wrong_history_mode(self) -> No def test_legacy_aliases_warn_once_per_host(self) -> None: agent = Agent(client=_ServiceStorageRecordingClient(), default_options=OpenAIChatOptions(store=True)) with pytest.warns(DeprecationWarning) as recorded: + warnings.warn("unrelated SDK deprecation", DeprecationWarning, stacklevel=2) server = ResponsesHostServer( agent, history_source="agent", store=InMemoryResponseProvider(), ) assert server is not None - assert len(recorded) == 2 - assert any("history_source" in str(warning.message) for warning in recorded) - assert any("response_store" in str(warning.message) for warning in recorded) + messages = [str(warning.message) for warning in recorded] + assert sum(message.startswith("history_source is deprecated;") for message in messages) == 1 + assert sum(message.startswith("store= is deprecated;") for message in messages) == 1 def test_explicit_history_rejects_legacy_alias_and_invalid_policy(self) -> None: agent = _make_agent() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/client.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/client.py index 20f34d8f134..e2d46723e90 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/client.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/client.py @@ -21,7 +21,9 @@ def main() -> None: allow_preview=True, ) as project, ): - openai = project.get_openai_client(agent_name=os.environ["FOUNDRY_AGENT_NAME"]) + openai = project.get_openai_client( # ty: ignore[unresolved-attribute] # pyrefly: ignore + agent_name=os.environ["FOUNDRY_AGENT_NAME"] + ) conversation = openai.conversations.create() first = openai.responses.create(input="Introduce yourself.", conversation=conversation.id, store=True) second = openai.responses.create( From 4664826ad46468e535c78929551e3c3a9e56d4cd Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Mon, 28 Sep 2026 14:24:33 +0200 Subject: [PATCH 04/10] fix(python): protect hosted Responses provider state and options --- python/packages/foundry_hosting/README.md | 11 +- .../_request.py | 22 ++ .../_responses.py | 144 ++++++- .../foundry_hosting/tests/test_request.py | 15 + .../foundry_hosting/tests/test_responses.py | 371 +++++++++++++++++- 5 files changed, 541 insertions(+), 22 deletions(-) diff --git a/python/packages/foundry_hosting/README.md b/python/packages/foundry_hosting/README.md index 733a8471d03..4f098a26c17 100644 --- a/python/packages/foundry_hosting/README.md +++ b/python/packages/foundry_hosting/README.md @@ -71,7 +71,11 @@ Outer background work always uses the caller-visible `response.id` for polling; background automatically. `inner_background="provider"` is a separate opt-in for `"service"` with a storing Responses client. Its private continuation token is saved under the outer ID and never returned to the caller. Use `ResponsesServerOptions(resilient_background=True)` to permit recovery from a **saved** token; a crash before -the token is saved cannot safely restart the inner job. Provider background and steering cannot be combined. +the token is saved cannot safely restart the inner job. A final provider poll retains that token in the private +response-ID snapshot so recovery can re-poll it if the outer response was not yet committed; a later turn drops +it from its working session. Shutdown during initial submission fails rather than replaying a job whose +acceptance is unknown. Cancelling an in-flight submission does not prove the remote provider stopped it. +Provider background and steering cannot be combined. Regular agent runs without this opt-in are not crash-replayable. `steerable_conversations=True` enables AgentServer's process-wide multi-turn TaskManager; a superseded turn keeps its own response snapshot but cannot replace a later CAS-protected conversation head. Start an in-progress background turn with `stream=True` before steering it: the @@ -83,7 +87,10 @@ Native CreateResponse generation fields become MAF runtime options (notably `max **last**. A sync or async `prepare_options(request: HostedResponseRequest, options: dict)` hook can remove or replace *caller* options before `Agent.run`; removed values fall back to the developer's unchanged agent defaults. Hosting filters caller platform IDs and private continuation/storage controls from model options and rejects attempts to -reintroduce them through the hook. A custom agent cannot accept MAF runtime options: choose +reintroduce them through the hook. Nested `extra_body` transport overrides are rejected for both caller +input and developer hooks, because they could override the host's `store=False` after the OpenAI SDK merges +the body. Developer defaults also cannot use this transport channel for host-controlled fields on +explicit history modes or unstored requests. A custom agent cannot accept MAF runtime options: choose `unsupported_options` as `"ignore"`, `"warn"` (default), or `"error"` for that case. See the [agent history and options samples](../../samples/04-hosting/foundry-hosted-agents/responses/basic/). diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py index 4754c9fda7f..1be8e232698 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py @@ -26,6 +26,7 @@ "continuation_token", "conversation", "conversation_id", + "extra_body", "input", "previous_response_id", "response_id", @@ -53,6 +54,8 @@ def response_run_options(request: CreateResponse) -> dict[str, Any]: not isinstance(key, str) for key in cast(Mapping[object, object], value) ): raise TypeError("extra_body must be a mapping of model options.") + if "extra_body" in value: + raise ValueError("Nested extra_body is not supported; flatten provider options instead.") nested_extra.update(cast(Mapping[str, Any], value)) elif name in _HOST_CONTROLLED_FIELDS: continue @@ -127,6 +130,25 @@ def validate_request_options(options: Mapping[str, Any]) -> None: raise ValueError(f"prepare_options cannot set host-controlled fields: {', '.join(sorted(reserved))}.") +def validate_default_transport_options(defaults: Mapping[str, Any], *, allow_legacy_store: bool) -> None: + """Reject transport overrides that would bypass the host's inner storage and identity decisions.""" + extra_body = defaults.get("extra_body") + if extra_body is None: + return + if not isinstance(extra_body, Mapping) or any( + not isinstance(key, str) for key in cast(Mapping[object, object], extra_body) + ): + raise TypeError("Agent default extra_body must be a mapping of model options.") + reserved = _HOST_CONTROLLED_FIELDS.intersection(cast(Mapping[str, Any], extra_body)) + if allow_legacy_store: + reserved -= {"store"} + if reserved: + raise ValueError( + "Agent default extra_body cannot set host-controlled fields: " + f"{', '.join(sorted(reserved))}. Use explicit agent defaults or inner_history='service'." + ) + + def validate_unsupported_options(mode: str) -> UnsupportedOptions: """Reject misspelled unsupported-options policies at host construction.""" if mode not in ("ignore", "warn", "error"): diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py index 1fc8f757ac9..8617cff93a3 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py @@ -96,6 +96,7 @@ UnsupportedOptions, prepare_response_options, response_run_options, + validate_default_transport_options, validate_request_options, validate_unsupported_options, ) @@ -183,6 +184,30 @@ def _agent_response_updates(response: AgentResponse[Any], response_id: str) -> l _T = TypeVar("_T") + +async def _await_before_signal( + operation: Callable[[], Awaitable[_T]], *signals: asyncio.Event +) -> tuple[bool, _T | None]: + """Race a provider await against lifecycle signals without discarding a completed continuation token.""" + if any(signal.is_set() for signal in signals): + return False, None + task = asyncio.ensure_future(operation()) + waiters = [asyncio.ensure_future(signal.wait()) for signal in signals] + try: + finished, _ = await asyncio.wait([task, *waiters], return_when=asyncio.FIRST_COMPLETED) + if task in finished: + return True, await task + return False, None + finally: + if not task.done(): + task.cancel() + with suppress(asyncio.CancelledError): + await task + for waiter in waiters: + waiter.cancel() + await asyncio.gather(*waiters, return_exceptions=True) + + # Sentinel put on the internal queue by _SignalledIterator's driver task to signal that the # wrapped iterator is exhausted (distinct from `None`, which is a valid item value). _STOP_SENTINEL: Any = object() @@ -1080,6 +1105,11 @@ async def _handle_inner_agent( request_messages_task: asyncio.Task[list[Message]] | None = None try: + if isinstance(agent, RawAgent): + validate_default_transport_options( + agent.default_options, + allow_legacy_store=self._inner_history == "legacy" and stored, + ) if not stored: if not isinstance(agent, RawAgent): raise RuntimeError( @@ -1177,6 +1207,16 @@ async def _handle_inner_agent( "store=false cannot continue legacy downstream service history; start a new one-shot request " "or select an explicit inner_history mode." ) + provider_state = session.state.get(_HOSTED_PROVIDER_STATE_KEY) + if ( + not context.is_recovery + and provider_state is not None + and ( + not isinstance(provider_state, Mapping) + or cast(Mapping[str, Any], provider_state).get("completed") is not True + ) + ): + raise ValueError("A provider background response must complete before the next turn.") if previous_response_id is not None and context.conversation_id is None and not context.is_recovery: if session.service_session_id is not None and session.state.get(_HOSTED_SOURCE_CONVERSATION_KEY): raise ValueError("A service-managed downstream conversation cannot be forked.") @@ -1189,6 +1229,8 @@ async def _handle_inner_agent( await session_storage.set(previous_response_id, session) session.state.pop(_HOSTED_SERVICE_CHILD_KEY) session.state.pop(_HOSTED_SOURCE_CONVERSATION_KEY, None) + if not context.is_recovery: + session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) except BaseException as ex: # Session preparation failed (or the request was cancelled / the stream closed — # neither of which is an Exception). Cancel and drain the in-flight message-loading @@ -1320,15 +1362,24 @@ async def _handle_inner_agent( and session.service_session_id is None and request_failure is None and not request_interrupted + and not (cancellation_signal.is_set() and context.client_cancelled) ): request_failure = RuntimeError( "inner_history='service' requires the chat client to return a service continuation ID." ) try: + provider_state = session.state.get(_HOSTED_PROVIDER_STATE_KEY) final_provider_state = not provider_background or ( request_failure is None and not request_interrupted - and _HOSTED_PROVIDER_STATE_KEY not in session.state + and not (cancellation_signal.is_set() and context.client_cancelled) + and ( + provider_state is None + or ( + isinstance(provider_state, Mapping) + and cast(Mapping[str, Any], provider_state).get("completed") is True + ) + ) ) superseded_by_steering = bool(self._host_options and self._host_options.steerable_conversations) and ( cancellation_signal.is_set() and not context.client_cancelled and not context.shutdown.is_set() @@ -1343,6 +1394,8 @@ async def _handle_inner_agent( and not request_interrupted and request_failure is None ): + if provider_background: + session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) await session_storage.set(context.conversation_id, session) except Exception as save_error: save_failure = save_error @@ -1377,19 +1430,43 @@ async def _provider_background_updates( ) -> AsyncGenerator[AgentResponseUpdate]: """Keep the inner provider token private while the outer response ID is polled.""" + async def proceed(*, has_token: bool) -> bool: + if cancellation_signal.is_set() and context.client_cancelled: + if not has_token: + logger.warning( + "Provider background submission was cancelled before a token was saved; " + "remote work may continue." + ) + return False + if context.shutdown.is_set(): + if has_token and self._resilient_background: + await context.exit_for_recovery() + if has_token: + raise RuntimeError("Provider background recovery requires resilient_background=True.") + raise RuntimeError( + "Provider background submission stopped before its token was saved; cannot safely retry it." + ) + if cancellation_signal.is_set(): + raise RuntimeError("Provider background was interrupted without a client cancellation.") + return True + async def run_provider( - options: ChatOptions[Any], *, input_messages: list[Message] | None + options: ChatOptions[Any], *, input_messages: list[Message] | None, phase: Literal["submit", "poll"] ) -> AgentResponse[Any]: try: return await agent.run(input_messages, session=session, options=options) except Exception as exc: - raise RuntimeError("Inner provider background execution failed; inspect the host logs.") from exc + logger.warning("Inner provider background %s failed (%s).", phase, type(exc).__name__) + raise RuntimeError(f"Inner provider background {phase} failed; inspect the host logs.") async def save_private_state() -> None: try: await session_storage.set(context.response_id, session) except Exception as exc: - raise RuntimeError("Could not save private provider background state; inspect the host logs.") from exc + logger.warning("Private provider background state save failed (%s).", type(exc).__name__) + else: + return + raise RuntimeError("Could not save private provider background state; inspect the host logs.") if context.is_recovery: saved = session.state.get(_HOSTED_PROVIDER_STATE_KEY) @@ -1403,8 +1480,26 @@ async def save_private_state() -> None: raise RuntimeError("The stored provider continuation token is invalid.") continuation_token: Mapping[str, Any] = cast(Mapping[str, Any], token) else: - first = await run_provider(cast(ChatOptions[Any], {**options, "background": True}), input_messages=messages) + if not await proceed(has_token=False): + return + completed, first = await _await_before_signal( + lambda: run_provider( + cast(ChatOptions[Any], {**options, "background": True}), + input_messages=messages, + phase="submit", + ), + context.shutdown, + cancellation_signal, + ) + if not completed: + if not await proceed(has_token=False): + return + raise RuntimeError("Provider background submission was interrupted before its token was saved.") + if first is None: + raise RuntimeError("The provider did not return a background response.") if first.continuation_token is None: + if not await proceed(has_token=False): + return for update in _agent_response_updates(first, context.response_id): yield update return @@ -1418,20 +1513,43 @@ async def save_private_state() -> None: await save_private_state() while True: - if context.shutdown.is_set() and self._resilient_background: - await context.exit_for_recovery() - if cancellation_signal.is_set() and context.client_cancelled: + if not await proceed(has_token=True): + return + slept, _ = await _await_before_signal(lambda: asyncio.sleep(2), context.shutdown, cancellation_signal) + if not slept: + if not await proceed(has_token=True): + return + raise RuntimeError("Provider background polling was interrupted.") + if not await proceed(has_token=True): return - await asyncio.sleep(2) - current = await run_provider( - cast(ChatOptions[Any], {"continuation_token": continuation_token, "store": True}), - input_messages=None, + completed, current = await _await_before_signal( + lambda token=continuation_token: run_provider( + cast(ChatOptions[Any], {"continuation_token": token, "store": True}), + input_messages=None, + phase="poll", + ), + context.shutdown, + cancellation_signal, ) + if not completed: + if not await proceed(has_token=True): + return + raise RuntimeError("Provider background polling was interrupted.") + if current is None: + raise RuntimeError("The provider did not return a background response.") if current.continuation_token is None: - session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) + # Until AgentServer commits the outer terminal event, recovery may need to re-poll this ID. + session.state[_HOSTED_PROVIDER_STATE_KEY] = { + "outer_response_id": context.response_id, + "continuation_token": dict(continuation_token), + "completed": True, + } await save_private_state() for update in _agent_response_updates(current, context.response_id): + if not await proceed(has_token=True): + return yield update + await proceed(has_token=True) return if not isinstance(current.continuation_token, Mapping): raise RuntimeError("The provider returned a continuation token that cannot be persisted.") diff --git a/python/packages/foundry_hosting/tests/test_request.py b/python/packages/foundry_hosting/tests/test_request.py index 8bf62efbd15..8c0074242b4 100644 --- a/python/packages/foundry_hosting/tests/test_request.py +++ b/python/packages/foundry_hosting/tests/test_request.py @@ -15,6 +15,7 @@ from agent_framework_foundry_hosting._request import ( prepare_response_options, response_run_options, + validate_default_transport_options, validate_request_options, validate_unsupported_options, ) @@ -70,6 +71,8 @@ def test_nested_extra_body_is_also_overlaid_last_without_reserved_fields() -> No assert response_run_options(request) == {"max_tokens": 150} with pytest.raises(TypeError, match="extra_body must be a mapping"): response_run_options(cast(CreateResponse, {"extra_body": ["not a mapping"]})) + with pytest.raises(ValueError, match="Nested extra_body is not supported"): + response_run_options(cast(CreateResponse, {"extra_body": {"extra_body": {"store": True}}})) async def test_hook_uses_a_request_copy_and_does_not_mutate_defaults() -> None: @@ -106,10 +109,22 @@ async def test_hook_rejects_reserved_fields_and_non_mapping_result() -> None: await prepare_response_options(request, lambda _view, _options: {"store": True, "session_id": "forged"}) with pytest.raises(ValueError, match="session_id, store"): validate_request_options(request.options) + await prepare_response_options(request, lambda _view, _options: {"extra_body": {"store": True}}) + with pytest.raises(ValueError, match="extra_body"): + validate_request_options(request.options) with pytest.raises(TypeError, match="must return a mapping"): await prepare_response_options(request, lambda _view, _options: cast(Any, None)) +def test_default_transport_cannot_override_host_storage_or_identity() -> None: + with pytest.raises(ValueError, match="store"): + validate_default_transport_options({"extra_body": {"store": True}}, allow_legacy_store=False) + with pytest.raises(ValueError, match="extra_body"): + validate_default_transport_options({"extra_body": {"extra_body": {"store": True}}}, allow_legacy_store=True) + validate_default_transport_options({"extra_body": {"store": True}}, allow_legacy_store=True) + validate_default_transport_options({"extra_body": {"temperature": 0.5}}, allow_legacy_store=False) + + @pytest.mark.parametrize("mode", ["ignore", "warn", "error"]) def test_supported_unsupported_option_modes(mode: str) -> None: assert validate_unsupported_options(mode) == mode diff --git a/python/packages/foundry_hosting/tests/test_responses.py b/python/packages/foundry_hosting/tests/test_responses.py index cb5accd8782..2a5f1d754b1 100644 --- a/python/packages/foundry_hosting/tests/test_responses.py +++ b/python/packages/foundry_hosting/tests/test_responses.py @@ -21,7 +21,7 @@ import warnings from collections.abc import AsyncGenerator, AsyncIterator, Awaitable, Callable, Generator, Mapping, Sequence from concurrent.futures import ThreadPoolExecutor -from contextlib import aclosing, asynccontextmanager +from contextlib import aclosing, asynccontextmanager, suppress from dataclasses import dataclass from importlib import import_module from pathlib import Path @@ -83,6 +83,8 @@ _LATEST_CHECKPOINT_ID_KEY, # pyright: ignore[reportPrivateUsage] CONSENT_ERROR_CODE, ConsentError, + _agent_response_updates, # pyright: ignore[reportPrivateUsage] + _await_before_signal, # pyright: ignore[reportPrivateUsage] _item_to_message, # pyright: ignore[reportPrivateUsage] _json_safe_to_str, # pyright: ignore[reportPrivateUsage] _output_item_to_message, # pyright: ignore[reportPrivateUsage] @@ -1546,6 +1548,63 @@ async def test_hook_cannot_reintroduce_host_owned_storage_or_identity(self) -> N assert "host-controlled" in response.json()["error"]["message"] agent.run.assert_not_called() + async def test_nested_extra_body_cannot_override_unstored_openai_request(self) -> None: + outbound: list[dict[str, Any]] = [] + + def handle_request(request: Any) -> Any: + outbound.append(json.loads(request.content)) + response = { + "id": "resp_inner", + "object": "response", + "created_at": 0, + "model": "test-model", + "status": "completed", + "output": [], + } + event = {"type": "response.completed", "sequence_number": 1, "response": response} + return _OPENAI_HTTPX.Response( + 200, + headers={"content-type": "text/event-stream"}, + text=f"event: response.completed\ndata: {json.dumps(event)}\n\n", + ) + + async with AsyncOpenAI( + api_key="test-key", + base_url="https://example.test/v1", + http_client=DefaultAsyncHttpxClient(transport=_OPENAI_HTTPX.MockTransport(handle_request)), + max_retries=0, + ) as openai: + agent = Agent(client=OpenAIChatClient(model="test-model", async_client=openai)) + server = _make_server(agent) + rejected = await _post_json( + server, + {"input": "unsafe", "store": False, "extra_body": {"extra_body": {"store": True}}}, + ) + assert rejected.json()["status"] == "failed" + assert "Nested extra_body" in rejected.json()["error"]["message"] + assert outbound == [] + + legacy_default = Agent( + client=OpenAIChatClient(model="test-model", async_client=openai), + default_options=cast(Any, {"extra_body": {"store": True}}), + ) + legacy_server = _make_server(legacy_default, history_source="agent") + rejected_default = await _post_json(legacy_server, {"input": "unsafe", "store": False}) + assert rejected_default.json()["status"] == "failed" + assert "default extra_body" in rejected_default.json()["error"]["message"] + assert outbound == [] + + unstored = await _post_json( + server, + {"input": "safe", "store": False, "extra_body": {"temperature": 0.1}}, + ) + + assert unstored.json()["status"] == "completed", unstored.json() + assert len(outbound) == 1 + assert outbound[0]["store"] is False + assert outbound[0]["temperature"] == 0.1 + assert "extra_body" not in outbound[0] + async def test_unsupported_options_policy_rejects_custom_agent(self) -> None: agent = _StrictCustomAgent() server = _make_server(agent, history_source="agent", unsupported_options="error") @@ -1651,7 +1710,85 @@ async def run( saved = await store.get("outer-response") assert saved is not None assert saved.service_session_id == "private-service-conversation" - assert "_foundry_provider_background" not in saved.state + assert saved.state["_foundry_provider_background"]["completed"] is True + assert saved.state["_foundry_provider_background"]["continuation_token"] == { + "response_id": "private-provider-token" + } + + next_agent = _make_agent( + stream_updates=[AgentResponseUpdate(contents=[Content.from_text("next")], role="assistant")] + ) + next_agent.client.STORES_BY_DEFAULT = True + next_server = _make_server(next_agent, session_store=store, inner_history="service") + next_context = ResponseContext(response_id="outer-next", mode_flags=MagicMock()) + with patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])): + next_events = [ + event + async for event in next_server._handle_response( + CreateResponse(input="next", store=True, previous_response_id="outer-response"), + next_context, + asyncio.Event(), + ) + ] + + assert [event.get("type") for event in next_events if isinstance(event, Mapping)][-1] == "response.completed" + next_session = await store.get("outer-next") + assert next_session is not None and "_foundry_provider_background" not in next_session.state + original_session = await store.get("outer-response") + assert original_session is not None + assert original_session.state["_foundry_provider_background"]["completed"] is True + + async def test_provider_background_conversation_head_omits_private_recovery_token( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + agent = Agent(client=_ServiceStorageRecordingClient()) + + async def run( + messages: Any = None, + *, + session: AgentSession, + options: Mapping[str, Any], + **kwargs: Any, + ) -> AgentResponse: + del messages, kwargs + if "continuation_token" not in options: + return AgentResponse( + messages=[], continuation_token=OpenAIContinuationToken(response_id="private-provider-token") + ) + session.service_session_id = "private-service-conversation" + return AgentResponse(messages=[Message(role="assistant", contents=[Content.from_text("done")])]) + + monkeypatch.setattr(agent, "run", MagicMock(side_effect=run)) + store = SessionStore() + server = _make_server( + agent, + session_store=store, + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + context = ResponseContext( + response_id="outer-response", conversation_id="outer-conversation", mode_flags=MagicMock() + ) + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + ): + events = [ + event + async for event in server._handle_response( + CreateResponse(input="hello", store=True, background=True), + context, + asyncio.Event(), + ) + ] + + assert [event.get("type") for event in events if isinstance(event, Mapping)][-1] == "response.completed" + snapshot = await store.get("outer-response") + head = await store.get("outer-conversation") + assert snapshot is not None and snapshot.state["_foundry_provider_background"]["completed"] is True + assert head is not None and "_foundry_provider_background" not in head.state async def test_provider_recovery_without_saved_token_does_not_restart_job( self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch @@ -1756,14 +1893,223 @@ async def run( assert [call.get("background") for call in calls] == [True, None] saved = await store.get("outer-recover") assert saved is not None and saved.service_session_id == "next-service-session" - assert "_foundry_provider_background" not in saved.state + assert saved.state["_foundry_provider_background"]["completed"] is True assert "private-provider-token" not in str(events) - async def test_provider_poll_errors_cannot_reveal_private_token( + @pytest.mark.parametrize("stage", ["submit", "sleep", "poll"]) + @pytest.mark.parametrize("interruption", ["cancel", "shutdown"]) + async def test_provider_background_observes_lifecycle_during_blocked_work( + self, + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + caplog: pytest.LogCaptureFixture, + stage: str, + interruption: str, + ) -> None: + entered = asyncio.Event() + stopped = asyncio.Event() + blocked = asyncio.Event() + agent = Agent(client=_ServiceStorageRecordingClient()) + token = OpenAIContinuationToken(response_id="private-provider-token") + submissions = 0 + + async def run( + messages: Any = None, + *, + session: AgentSession, + options: Mapping[str, Any], + **kwargs: Any, + ) -> AgentResponse: + nonlocal submissions + del messages, session, kwargs + if "continuation_token" not in options: + submissions += 1 + if stage == "submit": + entered.set() + try: + await blocked.wait() + finally: + stopped.set() + return AgentResponse(messages=[], continuation_token=token) + if stage == "poll": + entered.set() + try: + await blocked.wait() + finally: + stopped.set() + return AgentResponse(messages=[Message(role="assistant", contents=[Content.from_text("done")])]) + + async def sleep(_seconds: float) -> None: + if stage == "sleep": + entered.set() + try: + await blocked.wait() + finally: + stopped.set() + + monkeypatch.setattr(agent, "run", MagicMock(side_effect=run)) + store = SessionStore() + server = _make_server( + agent, + session_store=store, + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + response_id = f"outer-{stage}-{interruption}" + context = ResponseContext(response_id=response_id, mode_flags=MagicMock()) + request = CreateResponse(input="hello", store=True, background=True) + cancellation_signal = asyncio.Event() + exit_for_recovery = AsyncMock(side_effect=ResponseExitForRecovery()) + + async def collect_events() -> list[Any]: + return [event async for event in server._handle_response(request, context, cancellation_signal)] + + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch.object(ResponseContext, "exit_for_recovery", new=exit_for_recovery), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=sleep), + caplog.at_level(logging.WARNING), + ): + task = asyncio.create_task(collect_events()) + try: + await asyncio.wait_for(entered.wait(), timeout=3) + if interruption == "cancel": + context.client_cancelled = True + cancellation_signal.set() + events = await asyncio.wait_for(task, timeout=3) + assert all( + event.get("type") not in ("response.completed", "response.failed") + for event in events + if isinstance(event, Mapping) + ), _failure_message(events) + exit_for_recovery.assert_not_awaited() + else: + context.shutdown.set() + if stage == "submit": + events = await asyncio.wait_for(task, timeout=3) + assert "cannot safely retry" in _failure_message(events) + exit_for_recovery.assert_not_awaited() + else: + with pytest.raises(ResponseExitForRecovery): + await asyncio.wait_for(task, timeout=3) + exit_for_recovery.assert_awaited_once() + finally: + if not task.done(): + task.cancel() + with suppress(asyncio.CancelledError): + await task + + assert stopped.is_set() + assert submissions == 1 + saved = await store.get(response_id) + if stage == "submit": + assert saved is None + if interruption == "cancel": + assert "remote work may continue" in caplog.text + else: + assert saved is not None + assert saved.state["_foundry_provider_background"]["continuation_token"] == token + assert saved.state["_foundry_provider_background"].get("completed") is None + + async def test_completed_provider_result_wins_simultaneous_shutdown(self) -> None: + shutdown = asyncio.Event() + + async def complete_with_token() -> str: + shutdown.set() + return "private-token" + + completed, result = await _await_before_signal(complete_with_token, shutdown) + assert completed is True and result == "private-token" + + operation = MagicMock() + completed, result = await _await_before_signal(operation, shutdown) + assert completed is False and result is None + operation.assert_not_called() + + async def test_final_provider_poll_keeps_token_for_crash_recovery( self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + store = SessionStore() + token = OpenAIContinuationToken(response_id="private-provider-token") + calls: list[dict[str, Any]] = [] + + async def run( + messages: Any = None, + *, + session: AgentSession, + options: Mapping[str, Any], + **kwargs: Any, + ) -> AgentResponse: + del messages, kwargs + calls.append(dict(options)) + if "continuation_token" not in options: + return AgentResponse(messages=[], continuation_token=token) + session.service_session_id = "private-conversation" + return AgentResponse(messages=[Message(role="assistant", contents=[Content.from_text("finished")])]) + + def make_host() -> ResponsesHostServer: + agent = Agent(client=_ServiceStorageRecordingClient()) + monkeypatch.setattr(agent, "run", MagicMock(side_effect=run)) + return _make_server( + agent, + session_store=store, + inner_history="service", + inner_background="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + + server = make_host() + request = CreateResponse(input="hello", store=True, background=True) + context = ResponseContext(response_id="outer-recovered", mode_flags=MagicMock()) + + def shutdown_before_output(response: AgentResponse, response_id: str) -> list[AgentResponseUpdate]: + context.shutdown.set() + return _agent_response_updates(response, response_id) + + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch.object(ResponseContext, "exit_for_recovery", new=AsyncMock(side_effect=ResponseExitForRecovery())), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + patch("agent_framework_foundry_hosting._responses._agent_response_updates", new=shutdown_before_output), + pytest.raises(ResponseExitForRecovery), + ): + _ = [event async for event in server._handle_response(request, context, asyncio.Event())] + + saved = await store.get("outer-recovered") + assert saved is not None + assert saved.state["_foundry_provider_background"] == { + "outer_response_id": "outer-recovered", + "continuation_token": token, + "completed": True, + } + + recovered = ResponseContext(response_id="outer-recovered", mode_flags=MagicMock()) + recovered.is_recovery = True + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + ): + events = [event async for event in make_host()._handle_response(request, recovered, asyncio.Event())] + + assert [event.get("type") for event in events if isinstance(event, Mapping)][-1] == "response.completed" + assert "private-provider-token" not in str(events) + assert [option.get("background") for option in calls] == [True, None, None] + assert all(option["continuation_token"] == token for option in calls[1:]) + + @pytest.mark.parametrize("phase", ["submit", "poll", "save"]) + async def test_provider_errors_cannot_reveal_private_token( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, caplog: pytest.LogCaptureFixture, phase: str ) -> None: agent = Agent(client=_ServiceStorageRecordingClient()) + class FailingTokenStore(SessionStore): + async def set(self, session_id: str, session: AgentSession) -> None: + del session_id, session + raise RuntimeError("storage failed for private-provider-token") + async def run( messages: Any = None, *, @@ -1773,15 +2119,19 @@ async def run( ) -> AgentResponse: del messages, session, kwargs if "continuation_token" not in options: + if phase == "submit": + raise RuntimeError("provider submit failed for private-provider-token") return AgentResponse( messages=[], continuation_token=OpenAIContinuationToken(response_id="private-provider-token") ) - raise RuntimeError("provider failed for private-provider-token") + if phase == "poll": + raise RuntimeError("provider poll failed for private-provider-token") + return AgentResponse(messages=[Message(role="assistant", contents=[Content.from_text("done")])]) monkeypatch.setattr(agent, "run", MagicMock(side_effect=run)) server = _make_server( agent, - session_store=SessionStore(), + session_store=FailingTokenStore() if phase == "save" else SessionStore(), inner_history="service", inner_background="provider", options=ResponsesServerOptions(resilient_background=True), @@ -1792,11 +2142,18 @@ async def run( with ( patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + caplog.at_level(logging.WARNING), ): events = [event async for event in server._handle_response(request, context, asyncio.Event())] - assert "Inner provider background execution failed" in _failure_message(events) + error = _failure_message(events) + assert "private-provider-token" not in error assert "private-provider-token" not in str(events) + assert "private-provider-token" not in caplog.text + for record in caplog.records: + if record.name == "agent_framework_foundry_hosting._responses" and record.exc_info is not None: + exception = record.exc_info[1] + assert exception is not None and exception.__cause__ is None and exception.__context__ is None async def test_steering_keeps_superseded_snapshot_without_replacing_conversation_head(self) -> None: store = SessionStore() From e81632c28640c2e44176dcb7945b0a4dd27e6ffb Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Mon, 28 Sep 2026 14:50:35 +0200 Subject: [PATCH 05/10] fix(python): gate unsafe Foundry Responses steering until SDK fix --- python/packages/foundry_hosting/README.md | 13 +- .../_responses.py | 21 +- .../foundry_hosting/tests/test_responses.py | 152 ++++------ .../foundry-hosted-agents/README.md | 2 +- .../steerable_long_running_agent/README.md | 60 ++-- .../agent.manifest.yaml | 7 +- .../steerable_long_running_agent/main.py | 11 +- .../verify_steering.py | 275 ++---------------- 8 files changed, 120 insertions(+), 421 deletions(-) diff --git a/python/packages/foundry_hosting/README.md b/python/packages/foundry_hosting/README.md index 4f098a26c17..ba1170bfdc8 100644 --- a/python/packages/foundry_hosting/README.md +++ b/python/packages/foundry_hosting/README.md @@ -76,11 +76,14 @@ response-ID snapshot so recovery can re-poll it if the outer response was not ye it from its working session. Shutdown during initial submission fails rather than replaying a job whose acceptance is unknown. Cancelling an in-flight submission does not prove the remote provider stopped it. Provider background and steering cannot be combined. -Regular agent runs without this opt-in are not crash-replayable. `steerable_conversations=True` enables AgentServer's -process-wide multi-turn TaskManager; a superseded turn keeps its own response snapshot but cannot replace a later -CAS-protected conversation head. Start an in-progress background turn with `stream=True` before steering it: the -current AgentServer release can leave a superseded **non-streamed** background response in progress on retrieval. -Legacy `WorkflowAgent` dispatch is unchanged. +Regular agent runs without this opt-in are not crash-replayable. **Steering is temporarily unavailable:** +`steerable_conversations=True` fails during host construction, before enabling the process-wide TaskManager. +The current AgentServer SDK retains unbounded futures for rejected turns when its steering queue fills. Do not +use a queue-length precheck: another worker can append before it. The guard can be removed only after +[Azure/azure-sdk-for-python#49233](https://github.com/Azure/azure-sdk-for-python/pull/49233) ships in an +official `azure-ai-agentserver-core` wheel, the minimum dependency and `uv.lock` are updated, and a +concurrent queue-overflow regression proves rejected turns leave no pending futures. No future SDK +version is assumed. Non-steerable background polling and legacy `WorkflowAgent` dispatch are unchanged. Native CreateResponse generation fields become MAF runtime options (notably `max_output_tokens` -> `max_tokens` and `parallel_tool_calls` -> `allow_multiple_tool_calls`). Flattened OpenAI `extra_body` fields overlay translated keys diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py index 8617cff93a3..1bedcbb16ca 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py @@ -656,20 +656,25 @@ def __init__( a saved private continuation token can be polled after a crash; a crash before the token is saved cannot be replayed safely. Other regular-agent runs are not crash-recoverable. Legacy workflow background responses retain their checkpoint-based recovery behavior. - 5. Steering (steerable_conversations=True) is ONLY supported for non-workflow agents; constructing - this server with a workflow agent and `steerable_conversations=True` raises `RuntimeError`. - Steering a workflow is conceptually undefined -- a workflow's graph may have loops or parallel - branches with no single well-defined "current point" to cancel and resume from, unlike an - agent's strictly linear execution. It's also not currently practical to implement: a workflow - instance cannot start a new run until its previous (steered-past) run has been garbage - collected, and that isn't guaranteed to have happened in time. + 5. Steering is temporarily unavailable for all agents. `steerable_conversations=True` fails + at construction, before starting a host or enabling the process-wide TaskManager. The current + AgentServer SDK retains futures for rejected turns after the steering queue fills. Raises: ValueError: If the history, background, or unsupported-options policy is invalid. RuntimeError: If the agent configuration conflicts with the selected history source, `resilient_background=True` is requested for a regular agent without provider background, or - `steerable_conversations=True` is requested for a workflow agent. + `steerable_conversations=True` is requested while steering is unavailable. """ + if options and options.steerable_conversations: + # TODO(foundry-hosting): Remove this guard after Azure/azure-sdk-for-python#49233 ships in an official + # azure-ai-agentserver-core wheel, the minimum and uv.lock are updated, and a concurrent + # queue-overflow regression proves rejected turns leave no pending futures. + raise RuntimeError( + "steerable_conversations=True is temporarily unavailable: the current AgentServer SDK " + "retains futures for rejected steering turns. Wait for the official fix in " + "Azure/azure-sdk-for-python#49233, then update the dependency and verify queue overflow." + ) if history_source is not None and history_source not in ("agent_server", "agent"): raise ValueError("history_source must be either 'agent_server' or 'agent'.") if inner_history is not None and inner_history not in ("host", "service", "agent"): diff --git a/python/packages/foundry_hosting/tests/test_responses.py b/python/packages/foundry_hosting/tests/test_responses.py index 2a5f1d754b1..dcbf238009d 100644 --- a/python/packages/foundry_hosting/tests/test_responses.py +++ b/python/packages/foundry_hosting/tests/test_responses.py @@ -15,8 +15,6 @@ import json import logging import os -import threading -import time import uuid import warnings from collections.abc import AsyncGenerator, AsyncIterator, Awaitable, Callable, Generator, Mapping, Sequence @@ -74,7 +72,6 @@ from mcp import McpError from mcp.types import ErrorData from openai import AsyncOpenAI, DefaultAsyncHttpxClient -from starlette.testclient import TestClient from typing_extensions import Any from agent_framework_foundry_hosting import ResponsesHostServer @@ -1082,23 +1079,63 @@ def test_init_rejects_resilient_background_for_non_workflow_agent(self, tmp_path def test_init_rejects_steerable_conversations_for_workflow_agent(self) -> None: workflow_agent = _build_text_workflow_agent("hello from workflow") - with pytest.raises(RuntimeError, match="steerable_conversations"): + with pytest.raises(RuntimeError, match="steerable_conversations=True is temporarily unavailable"): ResponsesHostServer( cast(SupportsAgentRun, workflow_agent), store=InMemoryResponseProvider(), options=ResponsesServerOptions(steerable_conversations=True), ) - def test_steerable_agent_enables_multi_turn_task_manager(self) -> None: - with patch("azure.ai.agentserver.core.tasks.set_resilient_tasks_enabled") as enable: - _make_server(_make_agent(), options=ResponsesServerOptions(steerable_conversations=True)) - enable.assert_called_once_with(True) + def test_steering_rejected_before_enabling_task_manager_or_starting_host(self) -> None: + agent_factory = MagicMock() + + def construct(_turn: int) -> str: + with pytest.raises(RuntimeError, match="steerable_conversations=True is temporarily unavailable") as error: + ResponsesHostServer( + agent=agent_factory, + options=ResponsesServerOptions(steerable_conversations=True), + ) + return str(error.value) + + with ( + patch("azure.ai.agentserver.core.tasks.set_resilient_tasks_enabled") as enable, + patch("agent_framework_foundry_hosting._responses.ResponsesAgentServerHost.__init__") as base_init, + ThreadPoolExecutor(max_workers=12) as executor, + ): + failures = list(executor.map(construct, range(24))) + + assert len(failures) == 24 + assert all("Azure/azure-sdk-for-python#49233" in failure for failure in failures) + enable.assert_not_called() + base_init.assert_not_called() + agent_factory.assert_not_called() + + async def test_non_steerable_background_still_uses_outer_response_id(self) -> None: + agent = _make_agent( + stream_updates=[AgentResponseUpdate(contents=[Content.from_text("finished")], role="assistant")] + ) + server = _make_server(agent, options=ResponsesServerOptions(steerable_conversations=False)) + + async with httpx.AsyncClient(transport=httpx.ASGITransport(app=server), base_url="http://test") as http: + pending = await http.post("/responses", json={"input": "hello", "store": True, "background": True}) + assert pending.status_code == 200 + response_id = pending.json()["id"] + for _ in range(100): + final = await http.get(f"/responses/{response_id}") + if final.json()["status"] == "completed": + break + await asyncio.sleep(0.02) + else: + pytest.fail(f"Non-steerable background response {response_id} did not finish.") + + assert final.json()["id"] == response_id + assert "finished" in str(final.json()["output"]) def test_provider_background_rejects_steering_and_wrong_history_mode(self) -> None: agent = Agent(client=_ServiceStorageRecordingClient()) with pytest.raises(ValueError, match="inner_history='service'"): _make_server(agent, inner_background="provider") - with pytest.raises(ValueError, match="cannot be combined"): + with pytest.raises(RuntimeError, match="temporarily unavailable"): _make_server( agent, inner_history="service", @@ -2178,11 +2215,9 @@ async def updates() -> AsyncIterator[AgentResponseUpdate]: return ResponseStream(updates(), finalizer=AgentResponse.from_updates) agent.run = MagicMock(side_effect=run_with_state) - server = _make_server( - agent, - session_store=store, - options=ResponsesServerOptions(steerable_conversations=True), - ) + server = _make_server(agent, session_store=store) + # Exercise the handler's superseded-turn snapshot invariant without starting SDK steering. + server._host_options = ResponsesServerOptions(steerable_conversations=True) # pyright: ignore[reportPrivateUsage] first = ResponseContext( response_id="first-response", conversation_id="conversation-head", mode_flags=MagicMock() ) @@ -2332,95 +2367,6 @@ async def updates() -> AsyncIterator[AgentResponseUpdate]: assert missing.status_code == 404 assert agent.run.call_args_list[0].kwargs["options"].get("background") is None - def test_http_steering_uses_task_manager_and_keeps_latest_conversation_state( - self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch - ) -> None: - monkeypatch.setenv("AGENTSERVER_STATE_ROOT", str(tmp_path / "state")) - first_output = threading.Event() - release = threading.Event() - turn_number = 0 - store = SessionStore() - agent = _make_agent() - - def run(*args: Any, **kwargs: Any) -> ResponseStream[AgentResponseUpdate, AgentResponse]: - nonlocal turn_number - del args - turn_number += 1 - turn = turn_number - kwargs["session"].state["turn"] = turn - - async def updates() -> AsyncIterator[AgentResponseUpdate]: - yield AgentResponseUpdate(contents=[Content.from_text(f"turn {turn}")], role="assistant") - if turn == 1: - await asyncio.to_thread(release.wait) - yield AgentResponseUpdate(contents=[Content.from_text("stale")], role="assistant") - - return ResponseStream(updates(), finalizer=AgentResponse.from_updates) - - agent.run = MagicMock(side_effect=run) - server = _make_server( - agent, - session_store=store, - response_store=FileResponseStore(storage_dir=tmp_path / "responses"), - options=ResponsesServerOptions(steerable_conversations=True), - ) - - async def observe_output(request: Any, context: Any, cancellation_signal: Any) -> AsyncIterator[Any]: - async for event in server._handle_response(request, context, cancellation_signal): - if isinstance(event, Mapping) and event.get("type") == "response.output_text.delta": - first_output.set() - yield event - - server.response_handler(observe_output) - with TestClient(server) as http, ThreadPoolExecutor(max_workers=1) as executor: - try: - first_future = executor.submit( - http.post, - "/responses", - json={ - "input": "first", - "store": True, - "stream": True, - "background": True, - "conversation": "http-steering", - }, - ) - assert first_output.wait(timeout=5), "The first turn did not produce streamed output." - second = http.post( - "/responses", - json={"input": "second", "store": True, "background": True, "conversation": "http-steering"}, - ) - assert second.status_code == 200 - assert second.json()["status"] in ("queued", "in_progress") - first_stream = first_future.result(timeout=8) - finally: - release.set() - - assert first_stream.status_code == 200 - first_events = _parse_sse_events(first_stream.text) - first_completions = [ - event["data"]["response"] for event in first_events if event["event"] == "response.completed" - ] - assert len(first_completions) == 1 - first_id = first_completions[0]["id"] - deadline = time.monotonic() + 10 - while time.monotonic() < deadline: - latest_response = http.get(f"/responses/{second.json()['id']}") - if latest_response.json()["status"] in ("completed", "incomplete", "failed"): - break - time.sleep(0.05) - else: - pytest.fail(f"Steered response {second.json()['id']} did not finish.") - first_retrieved = http.get(f"/responses/{first_id}") - - assert latest_response.json()["status"] == "completed" - assert first_retrieved.json()["status"] == "completed" - assert turn_number == 2 - snapshot = asyncio.run(store.get(first_id)) - latest = asyncio.run(store.get(second.json()["id"])) - assert snapshot is not None and snapshot.state["turn"] == 1 - assert latest is not None and latest.state["turn"] == 2 - async def test_agent_history_uses_in_memory_history_from_session_store(self) -> None: client = _RecordingHistoryClient() history = InMemoryHistoryProvider() diff --git a/python/samples/04-hosting/foundry-hosted-agents/README.md b/python/samples/04-hosting/foundry-hosted-agents/README.md index 2ec5b667a66..f9c9ef13e44 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/README.md +++ b/python/samples/04-hosting/foundry-hosted-agents/README.md @@ -24,7 +24,7 @@ This directory contains samples that demonstrate how to use hosted [Agent Framew | 11 | [Foundry Toolbox MCP Skills](responses/foundry_toolbox_mcp_skills/) | An agent that discovers MCP-based skills attached to a Foundry Toolbox and serves them via `SkillsProvider(MCPSkillsSource(...))`, fetching `SKILL.md` bodies and supplementary resources on demand. | | 13 | [Custom Storage](responses/custom_storage/) | An agent demonstrating how to implement a custom storage provider for agent sessions (in-memory and Cosmos DB). | | 14 | [Resilient Long-Running Workflow](responses/resilient_long_running_workflow/) | A long-running, crash-resilient workflow demonstrating how `resilient_background=True` lets a background response survive a hard crash of the server process and resume from its last checkpoint instead of restarting from scratch. | -| 15 | [Steerable Long-Running Agent](responses/steerable_long_running_agent/) | A long-running, non-workflow agent demonstrating how `steerable_conversations=True` lets a new turn on the same conversation cancel and replace a still-running turn instead of waiting for it to finish. Steering is only supported for non-workflow agents. | +| 15 | [Long-Running Agent (steering gated)](responses/steerable_long_running_agent/) | A working long-running Responses agent with ordinary background polling; steering currently fails at host construction until a patched AgentServer SDK is published and verified. | | 16 | [Using deployed agent](responses/using_deployed_agent.py) | Invoke an agent already deployed to Foundry using either a service-created or user-created hosted session, then delete the session after use. | ## Session Identifiers diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md index 0c516a9cf38..5ffb2b9c0ea 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md @@ -1,45 +1,35 @@ -# Steerable Responses agent - -This sample hosts a single, slowly streaming [Agent Framework](https://github.com/microsoft/agent-framework) -agent using the Responses protocol. The agent counts down with a remark before each number. A new turn sent to -the same `conversation` while the first is running can steer it: the host stops the older turn, saves its partial -response under its own `response.id`, then lets the newer turn advance the CAS-protected conversation state. -Steering is supported for regular agents, not for workflow agents or provider-native background mode. - -[main.py](main.py) sets `inner_history="host"` and -`ResponsesServerOptions(steerable_conversations=True)`. The host enables AgentServer's multi-turn TaskManager -before startup; no agent-side steering code is needed. Both requests use `store=True` so the outer Responses API -can retrieve their result; the inner Foundry client still runs with `store=False`. The outer `response.id` remains -the background polling handle. - -Run the agent as described in the [parent guide](../../README.md), then start a long **streamed** -background turn. Keep the SSE connection open while you send the next request; `response.created` supplies -the first polling ID: - -```bash -curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" \ - -d '{"input": "Count down from 30, slowly and with commentary.", "store": true, "stream": true, "background": true, "conversation": "my-conversation"}' -``` - -While it runs, submit the second turn against that **same** conversation: +# Long-running Responses agent (steering temporarily unavailable) + +**Steering is currently gated.** `ResponsesHostServer` rejects +`ResponsesServerOptions(steerable_conversations=True)` during construction rather than starting a +TaskManager that can retain unbounded futures after rejected steering requests. The upstream fix is +[Azure/azure-sdk-for-python#49233](https://github.com/Azure/azure-sdk-for-python/pull/49233); +steering will be re-enabled only after a patched official `azure-ai-agentserver-core` wheel is released, +the package's minimum dependency and lockfile are updated, and a concurrent overflow test verifies +that rejected turns leave no pending futures. Do not enable steering using a private SDK patch or a +queue-length precheck. + +[main.py](main.py) remains **runnable** as a regular non-steerable countdown agent. It uses +`inner_history="host"` with a `FoundryChatClient` and returns the caller's **outer** `response.id` for +background polling. Running a second turn concurrently does not steer the first. Start it using the +[parent hosting guide](../../README.md), then send one stored background request. The deployment +manifests keep their existing agent name for identity compatibility, not because steering is enabled: ```bash curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" \ - -d '{"input": "Actually, count down from 3 instead.", "store": true, "background": true, "conversation": "my-conversation"}' + -d '{"input": "Count down from 30, slowly and with commentary.", "store": true, "background": true}' ``` -The second request is accepted as queued or in progress. Poll `GET /responses/{response.id}` for each outer -response ID; the older response should complete with partial output, and the newer one should contain its own -countdown. An explicit `conversation` binds the Foundry sandbox and avoids the need to forward the -`x-agent-session-id` response header. If continuing with only `previous_response_id`, forward the prior response's -`agent_session_id` as well to stay in the same sandbox. +Poll `GET /responses/{response.id}` until the response is completed. The agent can also stream text +with `"stream": true`; normal background polling does not require provider-native background mode. -[verify_steering.py](verify_steering.py) runs this check against a real model, using a fresh conversation ID and -isolated temporary AgentServer state. It never deletes your existing `~/.agentserver` data. Real model timing -varies; its assertions are intentionally looser than the package's deterministic local steering tests: +[verify_steering.py](verify_steering.py) currently checks the **fail-fast gate** without Azure +credentials or a deployed host. It does not send steering turns or claim they work: ```bash -python verify_steering.py --first-target 30 --second-target 3 +python verify_steering.py ``` -For a Foundry deployment, follow the [parent deployment guide](../../README.md). +Restore the end-to-end steering verifier only after the SDK fix is consumed and the queue-overflow +regression passes. Do not claim crash replay for an ordinary background agent: a process crash can +leave a response unfinished without an opted-in private provider continuation. diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/agent.manifest.yaml b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/agent.manifest.yaml index 90c39b67d4b..d9a59b8b978 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/agent.manifest.yaml +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/agent.manifest.yaml @@ -1,7 +1,7 @@ name: agent-framework-steerable-long-running-agent description: > - A hosted agent that demonstrates steerable multi-turn conversations for a plain (non-workflow) - agent using the Responses protocol. + A long-running Responses agent with background polling. Steering is temporarily gated until + a patched AgentServer SDK is published and verified. metadata: tags: - Agent Framework @@ -9,8 +9,7 @@ metadata: - Azure AI AgentServer - Responses Protocol - Streaming - - Steering - - Multi-turn + - Background template: name: agent-framework-steerable-long-running-agent kind: hosted diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py index 0fd671d4f4d..51895babf76 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py @@ -1,12 +1,11 @@ # Copyright (c) Microsoft. All rights reserved. -"""Host a single (non-workflow) agent that counts down slowly, steerably. +"""Host a long-running countdown agent while steering is temporarily unavailable. The agent is asked to count down from a target number, pacing its own output with a short remark -before each number so a real response takes a while to fully generate. With -`steerable_conversations=True`, sending a new turn on the same conversation while the countdown is -still streaming cancels the in-progress turn and drains the new turn next. Steering is only -supported for non-workflow agents like this one. +before each number so a response takes a while to fully generate. This sample currently runs +without steering; the hosting layer rejects ``steerable_conversations=True`` until a patched +AgentServer SDK is released and verified. Environment variables: FOUNDRY_PROJECT_ENDPOINT: Microsoft Foundry project endpoint. @@ -18,7 +17,6 @@ from agent_framework import Agent from agent_framework.foundry import FoundryChatClient from agent_framework_foundry_hosting import ResponsesHostServer -from azure.ai.agentserver.responses import ResponsesServerOptions from azure.identity import DefaultAzureCredential from dotenv import load_dotenv @@ -45,7 +43,6 @@ def main() -> None: server = ResponsesHostServer( agent=agent, inner_history="host", - options=ResponsesServerOptions(steerable_conversations=True), log_level="DEBUG", ) server.run() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/verify_steering.py b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/verify_steering.py index 4d72373b84d..cfe7b1d47e0 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/verify_steering.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/verify_steering.py @@ -1,273 +1,32 @@ # Copyright (c) Microsoft. All rights reserved. -"""End-to-end steering test for the steerable single-agent countdown sample. +"""Verify that steering fails before starting a host while the AgentServer SDK is affected. -Starts the server, kicks off a background streaming countdown, then -- while the model is still -generating -- sends a second turn on the same conversation with a new target. Verifies the second -turn is accepted immediately as "queued", that the first turn is cancelled and completes early -(fewer tokens than a full run), and that the second (steered) turn completes with a countdown for -its own target. Because this sample uses a real model (no deterministic per-tick pacing), -assertions here are necessarily looser than an exact output match. Requires the -same environment (.env) as running main.py directly. +This script does not require Azure credentials or make model requests. The full end-to-end +steering verifier must be restored when a patched core wheel and overflow regression are in place. -Usage: - python verify_steering.py [--first-target N] [--second-target N] [--min-deltas-before-steering N] +Expected output: Steering is temporarily unavailable; no agent or host was started. """ -import argparse -import json -import os -import subprocess -import sys -import threading -import time -import urllib.error -import urllib.request -import uuid -from pathlib import Path -from tempfile import TemporaryDirectory -from typing import IO, Any +from agent_framework import SupportsAgentRun +from agent_framework_foundry_hosting import ResponsesHostServer +from azure.ai.agentserver.responses import ResponsesServerOptions -HOST = "127.0.0.1" -PORT = 8088 -BASE_URL = f"http://{HOST}:{PORT}" -SAMPLE_DIR = Path(__file__).parent -STATE_DIR = TemporaryDirectory(prefix="af-steering-") -LOG_PATH = Path(STATE_DIR.name) / "verify_steering.log" - -def _http_get(path: str, timeout: float = 5.0) -> tuple[int, dict[str, Any]]: - with urllib.request.urlopen(urllib.request.Request(f"{BASE_URL}{path}"), timeout=timeout) as resp: - return resp.status, json.loads(resp.read()) - - -def _http_post(path: str, payload: dict[str, Any], timeout: float = 30.0) -> tuple[int, dict[str, Any]]: - data = json.dumps(payload).encode("utf-8") - request = urllib.request.Request( - f"{BASE_URL}{path}", data=data, headers={"Content-Type": "application/json"}, method="POST" - ) - try: - with urllib.request.urlopen(request, timeout=timeout) as resp: - return resp.status, json.loads(resp.read()) - except urllib.error.HTTPError as exc: - return exc.code, json.loads(exc.read()) - - -def _poll_until_terminal(response_id: str, timeout: float) -> dict[str, Any]: - """Poll ``GET /responses/{id}`` until the response reaches a terminal status. - - A ``stream=false`` create only guarantees a fast initial ack (e.g. ``"queued"``); this polls - for the actual outcome instead of trying to replay it live (a ``?stream=true`` GET replay is - only valid for a response that was itself created with ``stream=true``). - """ - deadline = time.monotonic() + timeout - snapshot: dict[str, Any] = {} - while time.monotonic() < deadline: - snapshot = _http_get(f"/responses/{response_id}")[1] - if snapshot.get("status") in ("completed", "failed", "incomplete", "cancelled"): - return snapshot - time.sleep(0.5) - return snapshot - - -def _start_server(log_file: IO[str]) -> subprocess.Popen: # type: ignore - return subprocess.Popen( - [sys.executable, "main.py"], - cwd=SAMPLE_DIR, - env={**os.environ, "PYTHONIOENCODING": "utf-8", "AGENTSERVER_STATE_ROOT": STATE_DIR.name}, - stdout=log_file, - stderr=subprocess.STDOUT, - ) - - -def _wait_for_ready(timeout: float = 30.0) -> None: - deadline = time.monotonic() + timeout - while time.monotonic() < deadline: - try: - status, _ = _http_get("/readiness", timeout=2.0) - if status == 200: - return - except (urllib.error.URLError, ConnectionError, TimeoutError): - pass - time.sleep(0.5) - raise RuntimeError("Server did not become ready in time.") - - -def _kill(server: subprocess.Popen) -> None: # type: ignore - if server.poll() is None: - server.kill() - server.wait(timeout=10) - - -def _watch_sse(request: "urllib.request.Request | str", progress: dict[str, Any]) -> None: - """Read an SSE stream from a streaming create POST and track its progress. - - Tracks the response id (on ``response.created``), text deltas, and terminal events. - """ - try: - with urllib.request.urlopen(request) as resp: - current_event: str | None = None - for raw_line in resp: - line = raw_line.decode("utf-8").rstrip("\n") - if line.startswith("event:"): - current_event = line[len("event:") :].strip() - continue - if not line.startswith("data:"): - continue - data_obj = json.loads(line[len("data:") :].strip()) - if current_event == "response.created" and "id" not in progress: - progress["id"] = data_obj["response"]["id"] - progress["status"] = data_obj["response"]["status"] - progress["ready"].set() - elif current_event == "response.output_text.delta": - progress["delta_count"] += 1 - elif current_event in ("response.completed", "response.failed", "response.incomplete"): - progress["done"].set() - except urllib.error.HTTPError as exc: - progress["error"] = f"HTTP {exc.code}: {exc.read().decode('utf-8', errors='replace')}" - except (urllib.error.URLError, ConnectionError, TimeoutError, OSError) as exc: - progress["error"] = f"{type(exc).__name__}: {exc}" - finally: - progress["ready"].set() - progress["done"].set() - - -def _extract_output_text(output_items: list[dict[str, Any]]) -> str: - parts: list[str] = [] - for item in output_items: - if item.get("type") != "message": - continue - for part in item.get("content", []): - if part.get("type") == "output_text": - parts.append(part["text"]) - return "".join(parts) +def _agent_factory() -> SupportsAgentRun: + raise AssertionError("A guarded steering host must not create an agent.") def main() -> None: - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--first-target", type=int, default=30, help="First turn's countdown starting value.") - parser.add_argument("--second-target", type=int, default=3, help="Steered turn's countdown starting value.") - parser.add_argument( - "--min-deltas-before-steering", - type=int, - default=5, - help="Minimum text delta events to observe on turn 1 before sending the steering turn.", - ) - args = parser.parse_args() - - log_file = LOG_PATH.open("w", encoding="utf-8") - print(f"Server logs (DEBUG level) are redirected to {LOG_PATH}.") - - print(f"[1/5] Starting server (first target={args.first_target}, second target={args.second_target})...") - conversation_id = f"steering-{uuid.uuid4().hex}" - server = _start_server(log_file) # type: ignore - print(f" PID: {server.pid}") + """Confirm that the constructor rejects steering before creating an agent or host.""" try: - _wait_for_ready() - - print("[2/5] Starting the first turn's background streaming countdown...") - first_progress: dict[str, Any] = { - "delta_count": 0, - "ready": threading.Event(), - "done": threading.Event(), - } - first_payload = { - "input": f"Count down from {args.first_target}, slowly and with commentary.", - "store": True, - "background": True, - "stream": True, - "conversation": conversation_id, - } - first_data = json.dumps(first_payload).encode("utf-8") - first_request = urllib.request.Request( - f"{BASE_URL}/responses", data=first_data, headers={"Content-Type": "application/json"}, method="POST" - ) - first_watcher = threading.Thread(target=_watch_sse, args=(first_request, first_progress), daemon=True) - first_watcher.start() - if not first_progress["ready"].wait(timeout=60): - raise SystemExit("FAIL: did not receive response.created for turn 1 in time.") - if "id" not in first_progress: - raise SystemExit(f"FAIL: turn 1 create request failed: {first_progress.get('error', 'unknown error')}") - first_id = first_progress["id"] - print(f" turn 1 response id: {first_id}, status: {first_progress['status']}") - - print(f"[3/5] Waiting for turn 1 to stream at least {args.min_deltas_before_steering} tokens...") - deadline = time.monotonic() + 60 - while first_progress["delta_count"] < args.min_deltas_before_steering: - if first_progress["done"].is_set() or time.monotonic() > deadline: - raise SystemExit( - "FAIL: turn 1 finished or timed out before enough tokens streamed to steer reliably; " - f"observed {first_progress['delta_count']} delta(s). See {LOG_PATH} for server logs." - ) - time.sleep(0.1) - count_at_steer_time = first_progress["delta_count"] - if first_progress["done"].is_set(): - raise SystemExit( - "FAIL: first turn finished before steering; increase --first-target or reduce --min-deltas." - ) - print(f" turn 1 text deltas observed before steering: {count_at_steer_time}") - - print(f"[4/5] Sending the steering turn (new target={args.second_target})...") - second_payload = { - "input": f"Actually, count down from {args.second_target} instead.", - "store": True, - "background": True, - "stream": False, - "conversation": conversation_id, - } - status, body = _http_post("/responses", second_payload) - if status != 200 or body.get("status") not in ("queued", "in_progress"): - raise SystemExit(f"FAIL: expected an immediately accepted steering turn, got: {body}") - second_id = body["id"] - print(f" steering turn accepted as {body['status']}; response id: {second_id}") - - print("[5/5] Watching turn 1 end early and the steered turn complete...") - first_progress["done"].wait(timeout=120) - first_final = _http_get(f"/responses/{first_id}")[1] - second_final = _poll_until_terminal(second_id, timeout=60) - finally: - _kill(server) - log_file.close() - - first_text = _extract_output_text(first_final.get("output", [])) - second_text = _extract_output_text(second_final.get("output", [])) - - print(f" turn 1 final status: {first_final['status']}, {len(first_text)} character(s)") - print(f" turn 2 final status: {second_final['status']}, {len(second_text)} character(s)") - - if "Serving steered turn" not in LOG_PATH.read_text(encoding="utf-8"): - raise SystemExit("FAIL: the follow-up was not processed as a steered turn.") - print(" confirmed 'Serving steered turn' in the server log.") - - if first_final["status"] != "completed": - raise SystemExit( - f"FAIL: turn 1 did not complete; last status: {first_final['status']}. See {LOG_PATH} for server logs." - ) - # Loose bound: a steered turn 1 should have generated only a bit more than what we observed - # right before steering, not a whole additional full run's worth of tokens. - if first_progress["delta_count"] > count_at_steer_time * 3 + 20: - raise SystemExit( - "FAIL: turn 1 kept streaming long after the steering turn was sent -- steering did not " - f"cancel it in time. See {LOG_PATH} for server logs." - ) - - if second_final["status"] != "completed": - raise SystemExit( - f"FAIL: turn 2 did not complete; last status: {second_final['status']}. See {LOG_PATH} for server logs." - ) - # Weak ordering check: each number from the new target down to 1 must appear, in order. - search_from = 0 - for n in range(args.second_target, 0, -1): - idx = second_text.find(str(n), search_from) - if idx == -1: - raise SystemExit( - f"FAIL: steered turn output is missing '{n}' in order.\n got: {second_text!r}\n" - f"See {LOG_PATH} for server logs." - ) - search_from = idx + 1 - - print("PASS: the steering turn cancelled the in-progress countdown early and completed its own countdown.") - STATE_DIR.cleanup() + ResponsesHostServer(agent=_agent_factory, options=ResponsesServerOptions(steerable_conversations=True)) + except RuntimeError as exc: + if "steerable_conversations=True is temporarily unavailable" not in str(exc): + raise + print("Steering is temporarily unavailable; no agent or host was started.") + return + raise RuntimeError("Steering guard was removed; restore end-to-end and queue-overflow verification.") if __name__ == "__main__": From f79162b009d932f82d89b0c416e1b84ce79a24ed Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Mon, 28 Sep 2026 15:45:02 +0200 Subject: [PATCH 06/10] fix(python): preserve provider finish reasons in Responses updates --- .../_responses.py | 9 +++-- .../foundry_hosting/tests/test_responses.py | 37 +++++++++++++++++++ 2 files changed, 42 insertions(+), 4 deletions(-) diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py index 1bedcbb16ca..d430c923952 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py @@ -34,6 +34,7 @@ CheckpointStorage, Content, ContextProvider, + FinishReason, HistoryProvider, InMemoryHistoryProvider, Message, @@ -175,10 +176,10 @@ def _agent_response_updates(response: AgentResponse[Any], response_id: str) -> l ] if response.usage_details is not None: updates.append(AgentResponseUpdate(contents=[Content.from_usage(response.usage_details)])) - if response.finish_reason == "length": - updates.append(AgentResponseUpdate(finish_reason="length")) - elif response.finish_reason == "content_filter": - updates.append(AgentResponseUpdate(finish_reason="content_filter")) + if response.finish_reason is not None: + if not updates: + updates.append(AgentResponseUpdate()) + updates[-1].finish_reason = FinishReason(response.finish_reason) return updates diff --git a/python/packages/foundry_hosting/tests/test_responses.py b/python/packages/foundry_hosting/tests/test_responses.py index dcbf238009d..94143a30cba 100644 --- a/python/packages/foundry_hosting/tests/test_responses.py +++ b/python/packages/foundry_hosting/tests/test_responses.py @@ -39,6 +39,7 @@ ChatResponse, ChatResponseUpdate, Content, + FinishReason, FinishReasonLiteral, FunctionInvocationLayer, HistoryProvider, @@ -1700,6 +1701,42 @@ async def stream_updates() -> AsyncIterator[AgentResponseUpdate]: assert response.json()["status"] == expected_status, response.json() assert "private-provider-token" not in str(response.json()) + @pytest.mark.parametrize( + ("finish_reason", "incomplete_reason"), + [ + ("stop", None), + ("tool_calls", None), + ("length", ResponseIncompleteReason.MAX_OUTPUT_TOKENS), + ("content_filter", ResponseIncompleteReason.CONTENT_FILTER), + ("provider_specific", None), + ], + ) + def test_provider_background_forwards_finish_reason_without_extra_update( + self, finish_reason: str, incomplete_reason: ResponseIncompleteReason | None + ) -> None: + response = AgentResponse( + messages=[Message(role="assistant", contents=[Content.from_text("finished")])], + finish_reason=FinishReason(finish_reason), + continuation_token=OpenAIContinuationToken(response_id="private-provider-token"), + ) + updates = _agent_response_updates(response, "outer-response") + + assert len(updates) == 1 + assert updates[0].finish_reason == finish_reason + assert updates[0].response_id == "outer-response" + assert updates[0].continuation_token is None + tracker = _OutputItemTracker(ResponseEventStream(response_id="outer-response")) + tracker.record_finish_reason(updates[0].finish_reason) + assert tracker.incomplete_reason == incomplete_reason + + def test_provider_background_forwards_finish_reason_without_output(self) -> None: + response = AgentResponse(messages=[], finish_reason="content_filter") + updates = _agent_response_updates(response, "outer-response") + + assert len(updates) == 1 + assert updates[0].contents == [] + assert updates[0].finish_reason == "content_filter" + async def test_provider_background_keeps_private_token_under_outer_response( self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: From b2af0b8a1ea85885197fb76cea62559eefed1cf9 Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Mon, 28 Sep 2026 16:01:00 +0200 Subject: [PATCH 07/10] docs(python): explain Foundry Responses history ownership by mode --- python/packages/foundry_hosting/README.md | 24 +++++++++++++ .../_responses.py | 35 ++++++++++++++++--- .../responses/basic/README.md | 21 ++++++++++- 3 files changed, 74 insertions(+), 6 deletions(-) diff --git a/python/packages/foundry_hosting/README.md b/python/packages/foundry_hosting/README.md index ba1170bfdc8..b76fd18a299 100644 --- a/python/packages/foundry_hosting/README.md +++ b/python/packages/foundry_hosting/README.md @@ -40,10 +40,34 @@ MAF session and approval state is saved. It does not choose the *inner* history | `"service"` | New input only | Enabled. The service-issued `AgentSession.service_session_id` is saved privately under the outer response ID or conversation. | | `"agent"` | New input plus the agent's `HistoryProvider` | Disabled. Agent history in `AgentSession.state` is saved by the host, without duplicating service history. | +For example, suppose the first stored response answers **"My name is Ada"**, then the caller sends +**"What is my name?"** with `previous_response_id` set to that response's **outer** `response.id`: + +- **`"host"`:** The model receives the first user input, the first assistant output, and the new + question. The host reconstructs that transcript from the Responses store; the inner service + does not retain it. +- **`"service"`:** The model receives only the new question as *request input*, along with the + private `AgentSession.service_session_id` from the first turn. The downstream service retrieves + its own transcript. A second branch from the first response cannot safely reuse that service + thread and is rejected. +- **`"agent"`:** The model receives the new question plus earlier messages loaded by the agent's + `HistoryProvider` from its stored MAF session. The downstream service does not store either turn. + +All three still return **outer** Responses IDs for retrieval and background polling. `store=False` +requests are one-shot: they do not write host-managed state or ask the inner client to store, so +they cannot establish a persistent provider thread. Neither an outer `response.id` nor +`agent_session_id` should be used as an inner `service_session_id`. + +Choose **one** mode when constructing each host; do not reuse the same `Agent` instance across hosts. +For example, to use downstream service history: + ```python server = ResponsesHostServer(agent=agent, inner_history="service") ``` +Omitting `inner_history` instead selects `"host"`. For `"agent"`, construct the agent with a +`HistoryProvider`, as shown in [agent_history.py](../../samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py). + `"host"` and `"service"` reject a load-enabled `HistoryProvider` alongside their own history source; `"host"` also rejects default downstream continuation IDs. Explicit modes require a `RawAgent` with a client declaring `STORES_BY_DEFAULT`; `"service"` requires a storing client that returns a private continuation ID. Hosting may diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py index d430c923952..0a45843e6a1 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py @@ -630,11 +630,29 @@ def __init__( If not provided, a default `CheckpointStoreProvider` will be used. function_approval_store_provider: Optional provider for function approval storage. If not provided, a default `FunctionApprovalStoreProvider` will be used. - inner_history: `"host"` (default) replays outer history with inner storage off; - `"service"` keeps the inner service continuation private; `"agent"` uses MAF history. - history_source: Deprecated alias. `"agent_server"` selects host history; `"agent"` - passes only new input and retains the developer's own history or service-storage - choice for stored requests. + inner_history: Who supplies prior messages to the *model*, independently of the caller's + Responses `store` flag. Omitted means `"host"`: + + - `"host"`: Send the prior outer Responses transcript followed by this request's input. + For a storing chat client, force its per-run `store=False` and clear any restored + `service_session_id` to avoid replaying that transcript twice. A load-enabled + `HistoryProvider` or fixed downstream continuation default conflicts with this mode. + - `"service"`: Send only this request's input. On caller `store=True`, run the inner + client with `store=True` and save its issued `service_session_id` in the host's + private MAF session store. The next stored request resumes that provider thread; + it must not branch an already-used provider conversation. + - `"agent"`: Send only this request's input; run a storing client with `store=False`. + The agent's `HistoryProvider` loads prior messages from its MAF `AgentSession`, + which the host saves for stored requests. Do not also use downstream service history. + + For example, after a stored response to "My name is Ada", a stored follow-up + "What is my name?" sends both turns to the model in `"host"` mode, but sends + only the follow-up in `"service"` mode (with a private service continuation). + In `"agent"` mode the provider supplies the earlier messages instead. + history_source: Deprecated compatibility setting. `"agent_server"` selects `"host"`; + `"agent"` sends only current input but preserves developer-controlled downstream + `store=True` *or* a `HistoryProvider` on stored requests. It is not equivalent to + `inner_history="agent"`, which disables downstream storage. inner_background: `"provider"` explicitly opts a storing, resumable client into its background API; `"host"` (default) leaves provider background disabled. prepare_options: Developer hook to remove or replace caller model options for an agent. @@ -643,6 +661,13 @@ def __init__( **kwargs: Additional keyword arguments. Note: + The *caller* controls outer persistence with `POST /responses` `store=True/False`. + `store=False` writes no host-managed session or approval state and forces supported + inner clients not to store, regardless of `inner_history`. The constructor's + deprecated `store=` argument instead selects the outer response-store backend + (use `response_store=`). Outer `response.id` is the polling/continuation handle; + any inner `service_session_id` is private and never replaces it. + 1. With `inner_history="host"` (or the deprecated `history_source="agent_server"`), the agent must not have a load-enabled history provider: the host supplies the transcript. 2. Context providers must not keep required state only on their Python instances, diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md index 0f4b5417ffe..17808f71044 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md @@ -11,6 +11,14 @@ where the **inner** agent gets its conversation history: | [options.py](options.py) | `"host"` | The hook removes the caller's token limit so the agent default is used. | | [provider_background.py](provider_background.py) | `"service"` | Opt-in provider background with a private recovery token. | +For the same two stored requests—first **"My name is Ada"**, then **"What is my name?"** with the first +response's `previous_response_id`—the modes differ at the *model* boundary. `main.py` replays both +the first user message and the first assistant output before the follow-up. `service_history.py` +sends only the follow-up and privately resumes the service thread returned by the first call. +`agent_history.py` sends the follow-up plus earlier messages loaded from `InMemoryHistoryProvider` +inside the persisted MAF session. In none of these cases is the caller's `response.id` the +downstream service ID. + Run one entry point at a time. The deployment manifest targets `main.py`; select another script to deploy a different mode. Set `FOUNDRY_PROJECT_ENDPOINT` and `AZURE_AI_MODEL_DEPLOYMENT_NAME` in `.env`, then run `python main.py`. Follow the [parent hosting guide](../../README.md) for local and deployed setup. @@ -18,7 +26,18 @@ Follow the [parent hosting guide](../../README.md) for local and deployed setup. ## Outer response storage and background `store=True` makes a response retrievable at `GET /responses/{response.id}` and available for continuation using -`previous_response_id` or a `conversation`. It does **not** choose inner history. With `store=False`, the response +`previous_response_id` or a `conversation`. For a local two-turn example, capture the `"id"` returned by the +first request and use it in the second; keep the same Foundry sandbox when deployed: + +```bash +curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" \ + -d '{"input": "My name is Ada", "store": true}' +curl -X POST http://localhost:8088/responses -H "Content-Type: application/json" \ + -d '{"input": "What is my name?", "store": true, "previous_response_id": "REPLACE_WITH_FIRST_RESPONSE_ID"}' +``` + +The *same* outer request works with any of the three host entry points above; only the source of +inner model history changes. `store=True` does **not** choose inner history. With `store=False`, the response is one-shot: the host writes no MAF session, approval, or conversation state, and it does not request inner service storage. Application-owned tools and other external services can still have their own side effects. If a custom agent or external history provider cannot guarantee that boundary, the host rejects the unstored request. From f45d3b00455e718a8a434a9dd92a783e8ed5990e Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Mon, 28 Sep 2026 16:12:36 +0200 Subject: [PATCH 08/10] refactor(python): name Foundry Responses history and background sources --- python/packages/foundry_hosting/README.md | 49 +++--- .../_request.py | 6 +- .../_responses.py | 163 ++++++++---------- .../foundry_hosting/tests/test_request.py | 8 +- .../foundry_hosting/tests/test_responses.py | 105 +++++------ .../tests/test_responses_int.py | 4 +- .../responses/basic/README.md | 27 +-- .../responses/basic/agent_history.py | 2 +- .../responses/basic/main.py | 2 +- .../responses/basic/options.py | 2 +- .../responses/basic/provider_background.py | 4 +- .../responses/basic/service_history.py | 2 +- .../steerable_long_running_agent/README.md | 2 +- .../steerable_long_running_agent/main.py | 2 +- 14 files changed, 183 insertions(+), 195 deletions(-) diff --git a/python/packages/foundry_hosting/README.md b/python/packages/foundry_hosting/README.md index b76fd18a299..56155740d49 100644 --- a/python/packages/foundry_hosting/README.md +++ b/python/packages/foundry_hosting/README.md @@ -32,26 +32,27 @@ requests must remain in the supported stores. ## Responses agent history and storage The caller's `POST /responses` **`store` flag** controls whether the *outer* response is retrievable and whether -MAF session and approval state is saved. It does not choose the *inner* history source: +MAF session and approval state is saved. The host's `history_source` independently selects who supplies model history: -| `inner_history` | What the model receives | Inner storage on `store=True` | +| `history_source` | What the model receives | Inner storage on caller `store=True` | | --- | --- | --- | -| `"host"` (default) | Prior outer Responses transcript plus new input | Disabled. Storing clients run with `store=False`; non-storing clients receive no storage option. | +| `"agent_server"` (default) | Prior outer Responses transcript plus new input | Disabled. Storing clients run with `store=False`; non-storing clients receive no storage option. | | `"service"` | New input only | Enabled. The service-issued `AgentSession.service_session_id` is saved privately under the outer response ID or conversation. | -| `"agent"` | New input plus the agent's `HistoryProvider` | Disabled. Agent history in `AgentSession.state` is saved by the host, without duplicating service history. | +| `"agent"` | New input only; the agent chooses how to load history | Developer-owned: `HistoryProvider` with `default_options={"store": False}` loads from host-persisted `AgentSession.state`, **or** a storing client with `default_options={"store": True}` uses downstream service history. The existing behavior is unchanged. | For example, suppose the first stored response answers **"My name is Ada"**, then the caller sends **"What is my name?"** with `previous_response_id` set to that response's **outer** `response.id`: -- **`"host"`:** The model receives the first user input, the first assistant output, and the new +- **`"agent_server"`:** The model receives the first user input, the first assistant output, and the new question. The host reconstructs that transcript from the Responses store; the inner service does not retain it. - **`"service"`:** The model receives only the new question as *request input*, along with the private `AgentSession.service_session_id` from the first turn. The downstream service retrieves its own transcript. A second branch from the first response cannot safely reuse that service thread and is rejected. -- **`"agent"`:** The model receives the new question plus earlier messages loaded by the agent's - `HistoryProvider` from its stored MAF session. The downstream service does not store either turn. +- **`"agent"`:** The model receives the new question. With `InMemoryHistoryProvider` and agent + default `store=False`, the provider adds earlier messages from the saved MAF session; with agent + default `store=True`, the downstream service owns the prior transcript instead. All three still return **outer** Responses IDs for retrieval and background polling. `store=False` requests are one-shot: they do not write host-managed state or ask the inner client to store, so @@ -62,37 +63,37 @@ Choose **one** mode when constructing each host; do not reuse the same `Agent` i For example, to use downstream service history: ```python -server = ResponsesHostServer(agent=agent, inner_history="service") +server = ResponsesHostServer(agent=agent, history_source="service") ``` -Omitting `inner_history` instead selects `"host"`. For `"agent"`, construct the agent with a -`HistoryProvider`, as shown in [agent_history.py](../../samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py). +Omitting `history_source` selects `"agent_server"`. To use a provider in `"agent"` mode, +configure that agent with `store=False` as shown in +[agent_history.py](../../samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py). -`"host"` and `"service"` reject a load-enabled `HistoryProvider` alongside their own history source; `"host"` -also rejects default downstream continuation IDs. Explicit modes require a `RawAgent` with a client declaring +`"agent_server"` and `"service"` reject a load-enabled `HistoryProvider` alongside their own history source; +`"agent_server"` also rejects default downstream continuation IDs. Those modes require a `RawAgent` with a client declaring `STORES_BY_DEFAULT`; `"service"` requires a storing client that returns a private continuation ID. Hosting may add a transient in-memory provider to support function-call loops, but **never edits `agent.default_options`**. The host owns the provided agent instance and any providers it adds; do not reuse it with another host. A factory creates an independent agent for each request. -The deprecated `history_source="agent_server"` still selects `"host"`. **Deprecated `history_source="agent"` is -not an alias for `inner_history="agent"`**: on stored requests, it preserves the old behavior of sending only new -input while the developer's defaults choose *either* a HistoryProvider *or* downstream service storage (including -`default_options={"store": True}`). Existing custom `SupportsAgentRun` implementations can continue using this -stored-request compatibility path. Each use of `history_source=` emits one deprecation warning per host; new code -should choose its explicit history mode. The outer storage-backend constructor argument is now `response_store=`. -The old `store=` backend argument remains an alias with its own once-per-host deprecation warning; supplying both -is an error. Neither constructor argument sets the caller's per-request `store` flag. +The existing `history_source="agent"` still preserves the agent's own provider **or** service storage defaults +on stored requests, including `default_options={"store": True}`. Custom `SupportsAgentRun` +implementations can continue using that mode for stored requests; it is not a forced-provider mode. +The outer storage-backend constructor argument is now `response_store=`. The old `store=` backend +argument remains an alias with its own once-per-host deprecation warning; supplying both is an error. +Neither constructor argument sets the caller's per-request `store` flag. `store=False` returns a one-shot response without **writing** host-managed session, conversation, or approval state; -it also disables inner service storage regardless of the developer's defaults. Unsafe custom agents, external +it also disables downstream service storage regardless of the developer's defaults. Unsafe custom agents, external history providers that store messages, and fixed downstream continuation defaults fail with an actionable error -instead of silently persisting. An unstored service-mode request cannot resume a private service thread. The legacy -agent mode also rejects an unstored continuation if its restored session uses downstream storage. Application-owned +instead of silently persisting. An unstored service-mode request cannot resume a private service thread. +`history_source="agent"` also rejects an unstored continuation if its restored session uses downstream storage. Application-owned tools and external services may still have their own side effects. `background=True` requires outer `store=True`. Outer background work always uses the caller-visible `response.id` for polling; it does not enable provider-native -background automatically. `inner_background="provider"` is a separate opt-in for `"service"` with a storing +background automatically. `background_source="agent_server"` (default) uses only the outer background worker. +`background_source="provider"` is a separate opt-in for `history_source="service"` with a storing Responses client. Its private continuation token is saved under the outer ID and never returned to the caller. Use `ResponsesServerOptions(resilient_background=True)` to permit recovery from a **saved** token; a crash before the token is saved cannot safely restart the inner job. A final provider poll retains that token in the private diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py index 1be8e232698..1d645031436 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_request.py @@ -130,7 +130,7 @@ def validate_request_options(options: Mapping[str, Any]) -> None: raise ValueError(f"prepare_options cannot set host-controlled fields: {', '.join(sorted(reserved))}.") -def validate_default_transport_options(defaults: Mapping[str, Any], *, allow_legacy_store: bool) -> None: +def validate_default_transport_options(defaults: Mapping[str, Any], *, allow_agent_store: bool) -> None: """Reject transport overrides that would bypass the host's inner storage and identity decisions.""" extra_body = defaults.get("extra_body") if extra_body is None: @@ -140,12 +140,12 @@ def validate_default_transport_options(defaults: Mapping[str, Any], *, allow_leg ): raise TypeError("Agent default extra_body must be a mapping of model options.") reserved = _HOST_CONTROLLED_FIELDS.intersection(cast(Mapping[str, Any], extra_body)) - if allow_legacy_store: + if allow_agent_store: reserved -= {"store"} if reserved: raise ValueError( "Agent default extra_body cannot set host-controlled fields: " - f"{', '.join(sorted(reserved))}. Use explicit agent defaults or inner_history='service'." + f"{', '.join(sorted(reserved))}. Use explicit agent defaults or history_source='service'." ) diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py index 0a45843e6a1..5be28758b7f 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py @@ -120,7 +120,7 @@ _HOSTED_SOURCE_CONVERSATION_KEY = "_foundry_source_conversation" _HOSTED_SERVICE_CHILD_KEY = "_foundry_service_child" -_HistoryMode = Literal["host", "service", "agent", "legacy"] +_HistorySource = Literal["agent_server", "agent", "service"] def _is_refusal_text_content(content: Content) -> bool: @@ -490,10 +490,10 @@ class _AgentConfiguration: def _validate_agent_configuration( agent: SupportsAgentRun, - inner_history: _HistoryMode, + history_source: _HistorySource, options: ResponsesServerOptions | None, *, - inner_background: Literal["host", "provider"] = "host", + background_source: Literal["agent_server", "provider"] = "agent_server", ) -> _AgentConfiguration: is_workflow_agent = isinstance(agent, WorkflowAgent) if is_workflow_agent and agent.workflow._runner_context.has_checkpointing(): # pyright: ignore[reportPrivateUsage] @@ -503,12 +503,12 @@ def _validate_agent_configuration( ) resilient_background = bool(options and options.resilient_background) - if resilient_background and not is_workflow_agent and inner_background != "provider": + if resilient_background and not is_workflow_agent and background_source != "provider": raise RuntimeError( "resilient_background=True is only supported for workflow agents. " "Crash recovery cannot be provided for non-workflow agents." ) - if inner_background == "provider" and ( + if background_source == "provider" and ( not isinstance(agent, RawAgent) or getattr(cast(Any, agent).client, "STORES_BY_DEFAULT", None) is not True ): raise RuntimeError("Provider background requires a RawAgent with a storing, resumable Responses client.") @@ -518,35 +518,34 @@ def _validate_agent_configuration( "Steering cannot be provided reliably for workflow agents." ) - if is_workflow_agent and inner_history not in ("host", "legacy"): - raise ValueError("inner_history='service' and 'agent' are only supported for regular agents.") + if is_workflow_agent and history_source == "service": + raise ValueError("history_source='service' is only supported for regular agents.") if not is_workflow_agent and isinstance(agent, RawAgent): identity_defaults = ("session_id", "agent_session_id", "user_id", "call_id", "service_session_id") if conflicting := [name for name in identity_defaults if agent.default_options.get(name) is not None]: raise RuntimeError(f"Model defaults cannot supply Foundry platform identity: {', '.join(conflicting)}.") - uses_agent_server_history = inner_history == "host" + uses_agent_server_history = history_source == "agent_server" client_stores_by_default = False - if not is_workflow_agent and inner_history != "legacy": + if not is_workflow_agent and history_source != "agent": if not isinstance(agent, RawAgent): raise RuntimeError( - "Explicit inner_history requires a RawAgent so hosting can enforce downstream storage. " - "For a custom SupportsAgentRun implementation, use the deprecated history_source='agent' " - "only for stored requests." + "history_source='agent_server' and 'service' require a RawAgent so hosting can enforce downstream " + "storage. Use history_source='agent' for a custom SupportsAgentRun implementation on stored requests." ) stores_by_default = getattr(cast(Any, agent).client, "STORES_BY_DEFAULT", None) if not isinstance(stores_by_default, bool): raise RuntimeError( - "Explicit inner_history requires the chat client to declare STORES_BY_DEFAULT " + "history_source='agent_server' or 'service' requires the chat client to declare STORES_BY_DEFAULT " "so hosting can enforce downstream storage behavior." ) client_stores_by_default = stores_by_default - if inner_history == "service" and not stores_by_default: + if history_source == "service" and not stores_by_default: raise RuntimeError( - "inner_history='service' requires a storing client that declares STORES_BY_DEFAULT=True." + "history_source='service' requires a storing client that declares STORES_BY_DEFAULT=True." ) service_continuation_options = [ @@ -556,20 +555,21 @@ def _validate_agent_configuration( ] if service_continuation_options: raise RuntimeError( - f"Explicit inner_history cannot use developer defaults for downstream continuation: " - f"{', '.join(service_continuation_options)}. Use history_source='agent' for legacy continuation." + f"history_source='agent_server' or 'service' cannot use developer defaults for downstream " + f"continuation: {', '.join(service_continuation_options)}. " + "Use history_source='agent' to retain developer-controlled continuation." ) - if inner_history in ("host", "service"): + if history_source in ("agent_server", "service"): for provider in agent.context_providers: if isinstance(provider, HistoryProvider) and provider.load_messages: - if inner_history == "host" and _is_hosted_responses_history_sentinel(provider): + if history_source == "agent_server" and _is_hosted_responses_history_sentinel(provider): continue raise RuntimeError( - "The selected inner_history conflicts with a load-enabled HistoryProvider. " - "Remove that provider or select inner_history='agent'." + "The selected history_source conflicts with a load-enabled HistoryProvider. " + "Remove that provider or select history_source='agent'." ) - if inner_history != "service" and not stores_by_default and agent.default_options.get("store") is True: + if history_source == "agent_server" and not stores_by_default and agent.default_options.get("store") is True: raise RuntimeError( "The chat client does not store by default, but the agent sets a downstream store option. " "Remove that developer-owned default rather than letting hosting change it." @@ -608,9 +608,8 @@ def __init__( agent_session_store_provider: StoreProvider[SessionStore] | None = None, checkpoint_store_provider: ContextScopedStoreProvider[CheckpointStorage] | None = None, function_approval_store_provider: StoreProvider[FunctionApprovalStore] | None = None, - inner_history: Literal["host", "service", "agent"] | None = None, - history_source: Literal["agent_server", "agent"] | None = None, - inner_background: Literal["host", "provider"] = "host", + history_source: Literal["agent_server", "agent", "service"] = "agent_server", + background_source: Literal["agent_server", "provider"] = "agent_server", prepare_options: OptionsHook | None = None, unsupported_options: UnsupportedOptions = "warn", **kwargs: Any, @@ -630,10 +629,10 @@ def __init__( If not provided, a default `CheckpointStoreProvider` will be used. function_approval_store_provider: Optional provider for function approval storage. If not provided, a default `FunctionApprovalStoreProvider` will be used. - inner_history: Who supplies prior messages to the *model*, independently of the caller's - Responses `store` flag. Omitted means `"host"`: + history_source: Who supplies prior messages to the *model*, independently of the caller's + Responses `store` flag. Defaults to `"agent_server"`: - - `"host"`: Send the prior outer Responses transcript followed by this request's input. + - `"agent_server"`: Send the prior outer Responses transcript followed by this request's input. For a storing chat client, force its per-run `store=False` and clear any restored `service_session_id` to avoid replaying that transcript twice. A load-enabled `HistoryProvider` or fixed downstream continuation default conflicts with this mode. @@ -641,20 +640,19 @@ def __init__( client with `store=True` and save its issued `service_session_id` in the host's private MAF session store. The next stored request resumes that provider thread; it must not branch an already-used provider conversation. - - `"agent"`: Send only this request's input; run a storing client with `store=False`. - The agent's `HistoryProvider` loads prior messages from its MAF `AgentSession`, - which the host saves for stored requests. Do not also use downstream service history. + - `"agent"`: Send only this request's input and preserve the agent's developer-owned + storage choice on stored requests. With an `InMemoryHistoryProvider` and + `default_options={"store": False}`, the provider loads prior messages from + the host-persisted MAF `AgentSession`. With `default_options={"store": True}`, + the downstream service retains history instead (the previous meaning of `"agent"`). For example, after a stored response to "My name is Ada", a stored follow-up - "What is my name?" sends both turns to the model in `"host"` mode, but sends - only the follow-up in `"service"` mode (with a private service continuation). - In `"agent"` mode the provider supplies the earlier messages instead. - history_source: Deprecated compatibility setting. `"agent_server"` selects `"host"`; - `"agent"` sends only current input but preserves developer-controlled downstream - `store=True` *or* a `HistoryProvider` on stored requests. It is not equivalent to - `inner_history="agent"`, which disables downstream storage. - inner_background: `"provider"` explicitly opts a storing, resumable client into its - background API; `"host"` (default) leaves provider background disabled. + "What is my name?" sends both turns to the model with `"agent_server"`, but + sends only the follow-up with `"service"` (plus a private service continuation). + With `"agent"`, the developer chooses either of those agent-owned history sources. + background_source: `"agent_server"` (default) runs background work in the Responses host + without invoking provider-native background APIs; `"provider"` opts a storing, resumable + client into provider background polling and requires `history_source="service"`. prepare_options: Developer hook to remove or replace caller model options for an agent. unsupported_options: `"warn"` (default), `"error"`, or `"ignore"` when a custom agent cannot accept runtime model options. @@ -663,12 +661,12 @@ def __init__( Note: The *caller* controls outer persistence with `POST /responses` `store=True/False`. `store=False` writes no host-managed session or approval state and forces supported - inner clients not to store, regardless of `inner_history`. The constructor's + inner clients not to store, regardless of `history_source`. The constructor's deprecated `store=` argument instead selects the outer response-store backend (use `response_store=`). Outer `response.id` is the polling/continuation handle; any inner `service_session_id` is private and never replaces it. - 1. With `inner_history="host"` (or the deprecated `history_source="agent_server"`), + 1. With `history_source="agent_server"`, the agent must not have a load-enabled history provider: the host supplies the transcript. 2. Context providers must not keep required state only on their Python instances, because the hosting environment may get deactivated between requests. Provider @@ -678,7 +676,7 @@ def __init__( Do not reuse the same agent with another host or invoke it directly after construction. An agent returned by a callable belongs to that request. 4. `resilient_background=True` supports legacy workflows and regular agents configured with - `inner_history="service", inner_background="provider"`. For provider background, only + `history_source="service", background_source="provider"`. For provider background, only a saved private continuation token can be polled after a crash; a crash before the token is saved cannot be replayed safely. Other regular-agent runs are not crash-recoverable. Legacy workflow background responses retain their checkpoint-based recovery behavior. @@ -701,32 +699,22 @@ def __init__( "retains futures for rejected steering turns. Wait for the official fix in " "Azure/azure-sdk-for-python#49233, then update the dependency and verify queue overflow." ) - if history_source is not None and history_source not in ("agent_server", "agent"): - raise ValueError("history_source must be either 'agent_server' or 'agent'.") - if inner_history is not None and inner_history not in ("host", "service", "agent"): - raise ValueError("inner_history must be 'host', 'service', or 'agent'.") - if inner_history is not None and history_source is not None: - raise ValueError("inner_history and history_source cannot be combined.") - if inner_background not in ("host", "provider"): - raise ValueError("inner_background must be 'host' or 'provider'.") + if history_source not in ("agent_server", "agent", "service"): + raise ValueError("history_source must be 'agent_server', 'agent', or 'service'.") + if background_source not in ("agent_server", "provider"): + raise ValueError("background_source must be 'agent_server' or 'provider'.") if store is not None and response_store is not None: raise ValueError("Pass response_store instead of store; they cannot be combined.") - if inner_history is not None: - resolved_history: _HistoryMode = inner_history - elif history_source == "agent": - resolved_history = "legacy" - else: - resolved_history = "host" - if inner_background == "provider" and resolved_history != "service": - raise ValueError("Provider background requires inner_history='service'.") - if inner_background == "provider" and options and options.steerable_conversations: + if background_source == "provider" and history_source != "service": + raise ValueError("Provider background requires history_source='service'.") + if background_source == "provider" and options and options.steerable_conversations: raise ValueError("Provider background and steerable_conversations cannot be combined.") validate_agent_source(agent) resolved_agent = agent if is_agent(agent) else None configuration = ( - _validate_agent_configuration(resolved_agent, resolved_history, options, inner_background=inner_background) + _validate_agent_configuration(resolved_agent, history_source, options, background_source=background_source) if resolved_agent is not None else None ) @@ -746,25 +734,18 @@ def __init__( self._agent_source = agent self._agent = resolved_agent self._configuration = configuration - self._inner_history: _HistoryMode = resolved_history - self._inner_background: Literal["host", "provider"] = inner_background + self._history_source: _HistorySource = history_source + self._background_source: Literal["agent_server", "provider"] = background_source self._prepare_options = prepare_options self._unsupported_options = validate_unsupported_options(unsupported_options) self._host_options = options self._uses_agent_server_history = ( - configuration.agent_server_history if configuration is not None else resolved_history == "host" + configuration.agent_server_history if configuration is not None else history_source == "agent_server" ) self._resilient_background = bool(options and options.resilient_background) if resolved_agent is not None and configuration is not None: _initialize_agent_history(resolved_agent, configuration) - if history_source is not None: - warnings.warn( - "history_source is deprecated; use inner_history='host', 'service', or 'agent'. " - "history_source='agent' retains developer-controlled downstream storage for stored requests.", - DeprecationWarning, - stacklevel=2, - ) if store is not None: warnings.warn("store= is deprecated; use response_store=.", DeprecationWarning, stacklevel=2) @@ -844,7 +825,7 @@ async def _handle_response( ) agent = await resolve_agent(self._agent_source) configuration = self._configuration or _validate_agent_configuration( - agent, self._inner_history, self._host_options, inner_background=self._inner_background + agent, self._history_source, self._host_options, background_source=self._background_source ) if self._configuration is None: _initialize_agent_history(agent, configuration) @@ -1128,7 +1109,7 @@ async def _handle_inner_agent( into a terminal ``response.failed`` event (draining the tracker so the SSE stream stays well-formed). """ - provider_background = self._inner_background == "provider" and request.get("background") is True + provider_background = self._background_source == "provider" and request.get("background") is True if context.is_recovery and not provider_background: raise RuntimeError("A non-resumable agent cannot be replayed after a process crash.") @@ -1139,7 +1120,7 @@ async def _handle_inner_agent( if isinstance(agent, RawAgent): validate_default_transport_options( agent.default_options, - allow_legacy_store=self._inner_history == "legacy" and stored, + allow_agent_store=self._history_source == "agent" and stored, ) if not stored: if not isinstance(agent, RawAgent): @@ -1173,12 +1154,12 @@ async def _handle_inner_agent( "store=false cannot prevent an external HistoryProvider from persisting responses. " "Use an InMemoryHistoryProvider or store=true." ) - if self._inner_history == "legacy": + if self._history_source == "agent": stores_by_default = getattr(cast(Any, agent).client, "STORES_BY_DEFAULT", None) if not isinstance(stores_by_default, bool): raise RuntimeError( "store=false with history_source='agent' requires a client declaring STORES_BY_DEFAULT; " - "select an explicit inner_history mode or use store=true." + "select history_source='agent_server' or 'service', or use store=true." ) if not stores_by_default and agent.default_options.get("store") is True: raise RuntimeError( @@ -1198,14 +1179,14 @@ async def _handle_inner_agent( if context.is_recovery and provider_background else context.conversation_id or previous_response_id ) - if not stored and session_load_id is not None and self._inner_history == "service": + if not stored and session_load_id is not None and self._history_source == "service": raise ValueError( "store=false cannot resume service-managed history; start a new one-shot request " - "or use inner_history='host' or 'agent'." + "or use history_source='agent_server' or 'agent'." ) session_storage = ( self._session_storage_provider.get_store(config=self.config, platform_context=request_context) - if stored or (session_load_id is not None and self._inner_history in ("agent", "legacy")) + if stored or (session_load_id is not None and self._history_source == "agent") else None ) @@ -1233,10 +1214,10 @@ async def _handle_inner_agent( f"Cannot find an existing agent session for previous_response_id={previous_response_id}." ) session = agent.create_session() - if not stored and self._inner_history == "legacy" and session.service_session_id is not None: + if not stored and self._history_source == "agent" and session.service_session_id is not None: raise ValueError( - "store=false cannot continue legacy downstream service history; start a new one-shot request " - "or select an explicit inner_history mode." + "store=false cannot continue agent-managed downstream service history; " + "start a new one-shot request or use store=true." ) provider_state = session.state.get(_HOSTED_PROVIDER_STATE_KEY) if ( @@ -1251,7 +1232,7 @@ async def _handle_inner_agent( if previous_response_id is not None and context.conversation_id is None and not context.is_recovery: if session.service_session_id is not None and session.state.get(_HOSTED_SOURCE_CONVERSATION_KEY): raise ValueError("A service-managed downstream conversation cannot be forked.") - if stored and self._inner_history in ("service", "legacy") and session.service_session_id is not None: + if stored and self._history_source in ("service", "agent") and session.service_session_id is not None: if session.state.get(_HOSTED_SERVICE_CHILD_KEY): raise ValueError("A service-managed downstream response cannot be forked.") if session_storage is None: @@ -1301,14 +1282,10 @@ async def _handle_inner_agent( else: # Do not pass a storage option to clients that do not advertise support for it. chat_options.pop("store", None) - elif self._inner_history == "service": + elif self._history_source == "service": chat_options["store"] = stored if not stored: session.service_session_id = None - elif self._inner_history == "agent": - if configuration.client_stores_by_default: - chat_options["store"] = False - session.service_session_id = None elif ( not stored and isinstance(agent, RawAgent) @@ -1358,7 +1335,7 @@ async def _handle_inner_agent( if final.continuation_token is not None and final.finish_reason is None: raise RuntimeError( "The inner agent returned an unfinished provider response; " - "configure inner_background='provider' to resume it." + "configure background_source='provider' with history_source='service' to resume it." ) except (asyncio.CancelledError, GeneratorExit): request_interrupted = True @@ -1376,18 +1353,18 @@ async def _handle_inner_agent( # Never persist a session that could resume an inner response contrary to the # history mode or the caller's explicit storage decision. stored_output_violation = ( - self._inner_history in ("host", "agent") or not stored + self._history_source == "agent_server" or not stored ) and session.service_session_id is not None if stored_output_violation: misconfigured = RuntimeError( "The agent's chat client stored this turn server-side despite store=False for the inner model. " - "Configure the client to honor store=False, or use inner_history='service' with store=true." + "Configure the client to honor store=False, or use history_source='service' with store=true." ) logger.error("%s", misconfigured) if request_failure is None and not request_interrupted: request_failure = misconfigured if ( - self._inner_history == "service" + self._history_source == "service" and stored and _HOSTED_PROVIDER_STATE_KEY not in session.state and session.service_session_id is None @@ -1396,7 +1373,7 @@ async def _handle_inner_agent( and not (cancellation_signal.is_set() and context.client_cancelled) ): request_failure = RuntimeError( - "inner_history='service' requires the chat client to return a service continuation ID." + "history_source='service' requires the chat client to return a service continuation ID." ) try: provider_state = session.state.get(_HOSTED_PROVIDER_STATE_KEY) diff --git a/python/packages/foundry_hosting/tests/test_request.py b/python/packages/foundry_hosting/tests/test_request.py index 8c0074242b4..289b927c865 100644 --- a/python/packages/foundry_hosting/tests/test_request.py +++ b/python/packages/foundry_hosting/tests/test_request.py @@ -118,11 +118,11 @@ async def test_hook_rejects_reserved_fields_and_non_mapping_result() -> None: def test_default_transport_cannot_override_host_storage_or_identity() -> None: with pytest.raises(ValueError, match="store"): - validate_default_transport_options({"extra_body": {"store": True}}, allow_legacy_store=False) + validate_default_transport_options({"extra_body": {"store": True}}, allow_agent_store=False) with pytest.raises(ValueError, match="extra_body"): - validate_default_transport_options({"extra_body": {"extra_body": {"store": True}}}, allow_legacy_store=True) - validate_default_transport_options({"extra_body": {"store": True}}, allow_legacy_store=True) - validate_default_transport_options({"extra_body": {"temperature": 0.5}}, allow_legacy_store=False) + validate_default_transport_options({"extra_body": {"extra_body": {"store": True}}}, allow_agent_store=True) + validate_default_transport_options({"extra_body": {"store": True}}, allow_agent_store=True) + validate_default_transport_options({"extra_body": {"temperature": 0.5}}, allow_agent_store=False) @pytest.mark.parametrize("mode", ["ignore", "warn", "error"]) diff --git a/python/packages/foundry_hosting/tests/test_responses.py b/python/packages/foundry_hosting/tests/test_responses.py index 94143a30cba..e45d0c13ad9 100644 --- a/python/packages/foundry_hosting/tests/test_responses.py +++ b/python/packages/foundry_hosting/tests/test_responses.py @@ -1132,19 +1132,19 @@ async def test_non_steerable_background_still_uses_outer_response_id(self) -> No assert final.json()["id"] == response_id assert "finished" in str(final.json()["output"]) - def test_provider_background_rejects_steering_and_wrong_history_mode(self) -> None: + def test_provider_background_rejects_steering_and_wrong_history_source(self) -> None: agent = Agent(client=_ServiceStorageRecordingClient()) - with pytest.raises(ValueError, match="inner_history='service'"): - _make_server(agent, inner_background="provider") + with pytest.raises(ValueError, match="history_source='service'"): + _make_server(agent, background_source="provider") with pytest.raises(RuntimeError, match="temporarily unavailable"): _make_server( agent, - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(steerable_conversations=True), ) - def test_legacy_aliases_warn_once_per_host(self) -> None: + def test_store_alias_warns_once_without_deprecating_history_source(self) -> None: agent = Agent(client=_ServiceStorageRecordingClient(), default_options=OpenAIChatOptions(store=True)) with pytest.warns(DeprecationWarning) as recorded: warnings.warn("unrelated SDK deprecation", DeprecationWarning, stacklevel=2) @@ -1155,29 +1155,30 @@ def test_legacy_aliases_warn_once_per_host(self) -> None: ) assert server is not None messages = [str(warning.message) for warning in recorded] - assert sum(message.startswith("history_source is deprecated;") for message in messages) == 1 + assert not any(message.startswith("history_source is deprecated;") for message in messages) assert sum(message.startswith("store= is deprecated;") for message in messages) == 1 - def test_explicit_history_rejects_legacy_alias_and_invalid_policy(self) -> None: + def test_history_and_background_sources_reject_invalid_values(self) -> None: agent = _make_agent() - with pytest.raises(ValueError, match="cannot be combined"): - _make_server(agent, inner_history="agent", history_source="agent") + with pytest.raises(ValueError, match="history_source"): + _make_server(agent, history_source="host") + with pytest.raises(ValueError, match="background_source"): + _make_server(agent, background_source="host") with pytest.raises(ValueError, match="unsupported_options"): _make_server(agent, unsupported_options="silent") @pytest.mark.parametrize( "identity_field", ["session_id", "agent_session_id", "user_id", "call_id", "service_session_id"] ) - @pytest.mark.parametrize("legacy", [False, True]) - def test_agent_defaults_cannot_supply_platform_identity(self, identity_field: str, legacy: bool) -> None: + @pytest.mark.parametrize("history_source", ["agent_server", "agent", "service"]) + def test_agent_defaults_cannot_supply_platform_identity(self, identity_field: str, history_source: str) -> None: agent = Agent( client=_ServiceStorageRecordingClient(), default_options=cast(Any, {identity_field: "forged"}), ) - selection = {"history_source": "agent"} if legacy else {"inner_history": "host"} with pytest.raises(RuntimeError, match="Model defaults cannot supply Foundry platform identity"): - _make_server(agent, **selection) + _make_server(agent, history_source=history_source) async def test_previous_response_requires_existing_agent_session(self) -> None: agent = _make_agent() @@ -1401,12 +1402,13 @@ async def test_agent_server_history_clears_restored_service_session_id(self) -> assert stored is not None assert stored.service_session_id is None - async def test_agent_history_preserves_service_storage(self) -> None: + @pytest.mark.parametrize("explicit_store", [True, False], ids=["agent-default", "client-default"]) + async def test_agent_history_preserves_service_storage(self, explicit_store: bool) -> None: client = _ServiceStorageRecordingClient() agent = Agent( client=client, name="Agent Managed Service Storage", - default_options={"store": True}, # pyrefly: ignore[bad-argument-type] + default_options=OpenAIChatOptions(store=True) if explicit_store else None, ) store = SessionStore() server = _make_server(agent, session_store=store, history_source="agent") @@ -1416,7 +1418,10 @@ async def test_agent_history_preserves_service_storage(self) -> None: assert second.json()["status"] == "completed" assert [[message.text for message in call] for call in client.calls] == [["first"], ["second"]] - assert client.store_options == [True, True] + if explicit_store: + assert client.store_options == [True, True] + else: + assert client.store_options == [None, None] assert client.conversation_ids == [None, "service-thread-1"] stored = await store.get(second.json()["id"]) assert stored is not None @@ -1426,7 +1431,7 @@ async def test_service_history_preserves_private_continuation(self) -> None: client = _ServiceStorageRecordingClient() agent = Agent(client=client, default_options=OpenAIChatOptions(store=False)) store = SessionStore() - server = _make_server(agent, session_store=store, inner_history="service") + server = _make_server(agent, session_store=store, history_source="service") first = await _post(server, input_text="first") second = await _post(server, input_text="second", previous_response_id=first.json()["id"]) @@ -1449,7 +1454,7 @@ async def create_agent() -> SupportsAgentRun: return Agent(client=client, default_options=OpenAIChatOptions(store=True)) store = SessionStore() - server = _make_server(create_agent, session_store=store, inner_history="service") + server = _make_server(create_agent, session_store=store, history_source="service") first = await _post(server, input_text="first") second = await _post(server, input_text="second", previous_response_id=first.json()["id"]) @@ -1462,7 +1467,7 @@ async def create_agent() -> SupportsAgentRun: async def test_service_history_rejects_second_child_of_provider_response(self) -> None: client = _ServiceStorageRecordingClient() - server = _make_server(Agent(client=client), session_store=SessionStore(), inner_history="service") + server = _make_server(Agent(client=client), session_store=SessionStore(), history_source="service") first = await _post(server, input_text="first") second = await _post(server, input_text="second", previous_response_id=first.json()["id"]) branch = await _post(server, input_text="fork", previous_response_id=first.json()["id"]) @@ -1472,12 +1477,12 @@ async def test_service_history_rejects_second_child_of_provider_response(self) - assert "cannot be forked" in branch.json()["error"]["message"] assert len(client.calls) == 2 - async def test_explicit_agent_history_uses_provider_without_inner_storage(self) -> None: + async def test_agent_history_uses_provider_with_nonstoring_defaults(self) -> None: client = _ServiceStorageRecordingClient() history = InMemoryHistoryProvider() - agent = Agent(client=client, context_providers=[history], default_options=OpenAIChatOptions(store=True)) + agent = Agent(client=client, context_providers=[history], default_options=OpenAIChatOptions(store=False)) store = SessionStore() - server = _make_server(agent, session_store=store, inner_history="agent") + server = _make_server(agent, session_store=store, history_source="agent") first = await _post(server, input_text="first") second = await _post(server, input_text="second", previous_response_id=first.json()["id"]) @@ -1488,18 +1493,17 @@ async def test_explicit_agent_history_uses_provider_without_inner_storage(self) ["first", "recorded", "second"], ] assert client.store_options == [False, False] - assert agent.default_options["store"] is True + assert agent.default_options["store"] is False saved = await store.get(second.json()["id"]) assert saved is not None and history.source_id in saved.state assert saved.service_session_id is None - @pytest.mark.parametrize("mode", ["host", "service", "agent", "legacy"]) + @pytest.mark.parametrize("mode", ["agent_server", "service", "agent"]) async def test_store_false_neither_saves_session_nor_stores_inner_response(self, mode: str) -> None: client = _ServiceStorageRecordingClient() agent = Agent(client=client, default_options=OpenAIChatOptions(store=True)) store = SessionStore() - selection = {"history_source": "agent"} if mode == "legacy" else {"inner_history": mode} - server = _make_server(agent, session_store=store, **selection) + server = _make_server(agent, session_store=store, history_source=mode) approvals = MagicMock(spec=FunctionApprovalStoreProvider) server._function_approval_storage_provider = approvals # pyright: ignore[reportPrivateUsage] @@ -1529,7 +1533,7 @@ async def test_store_false_rejects_external_history_and_custom_agent(self) -> No assert "custom agent" in response.json()["error"]["message"] assert custom.calls == [] - async def test_store_false_legacy_continuation_fails_instead_of_using_service_history(self) -> None: + async def test_store_false_agent_continuation_fails_instead_of_using_service_history(self) -> None: client = _ServiceStorageRecordingClient() server = _make_server( Agent(client=client, default_options=OpenAIChatOptions(store=True)), @@ -1541,7 +1545,7 @@ async def test_store_false_legacy_continuation_fails_instead_of_using_service_hi context = ResponseContext(response_id="one-shot", mode_flags=MagicMock()) events = [event async for event in server._handle_response(request, context, asyncio.Event())] - assert "store=false cannot continue legacy" in _failure_message(events) + assert "store=false cannot continue agent-managed downstream service history" in _failure_message(events) assert len(client.calls) == 1 async def test_extra_options_overlay_then_developer_hook_preserves_defaults(self) -> None: @@ -1764,8 +1768,8 @@ async def run( server = _make_server( agent, session_store=store, - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(resilient_background=True), response_store=FileResponseStore(storage_dir=tmp_path), ) @@ -1793,7 +1797,7 @@ async def run( stream_updates=[AgentResponseUpdate(contents=[Content.from_text("next")], role="assistant")] ) next_agent.client.STORES_BY_DEFAULT = True - next_server = _make_server(next_agent, session_store=store, inner_history="service") + next_server = _make_server(next_agent, session_store=store, history_source="service") next_context = ResponseContext(response_id="outer-next", mode_flags=MagicMock()) with patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])): next_events = [ @@ -1837,8 +1841,8 @@ async def run( server = _make_server( agent, session_store=store, - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(resilient_background=True), response_store=FileResponseStore(storage_dir=tmp_path), ) @@ -1873,8 +1877,8 @@ async def test_provider_recovery_without_saved_token_does_not_restart_job( server = _make_server( agent, session_store=SessionStore(), - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(resilient_background=True), response_store=FileResponseStore(storage_dir=tmp_path), ) @@ -1916,8 +1920,8 @@ async def run( first_server = _make_server( first_agent, session_store=store, - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(resilient_background=True), response_store=FileResponseStore(storage_dir=tmp_path), ) @@ -1948,8 +1952,8 @@ async def run( resumed_server = _make_server( resumed_agent, session_store=store, - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(resilient_background=True), response_store=FileResponseStore(storage_dir=tmp_path), ) @@ -2026,8 +2030,8 @@ async def sleep(_seconds: float) -> None: server = _make_server( agent, session_store=store, - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(resilient_background=True), response_store=FileResponseStore(storage_dir=tmp_path), ) @@ -2129,8 +2133,8 @@ def make_host() -> ResponsesHostServer: return _make_server( agent, session_store=store, - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(resilient_background=True), response_store=FileResponseStore(storage_dir=tmp_path), ) @@ -2206,8 +2210,8 @@ async def run( server = _make_server( agent, session_store=FailingTokenStore() if phase == "save" else SessionStore(), - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(resilient_background=True), response_store=FileResponseStore(storage_dir=tmp_path), ) @@ -2318,7 +2322,7 @@ async def test_conversation_write_conflict_keeps_response_snapshot(self) -> None async def test_http_outer_storage_is_independent_of_inner_service_history(self) -> None: client = _ServiceStorageRecordingClient() store = SessionStore() - server = _make_server(Agent(client=client), inner_history="service", session_store=store) + server = _make_server(Agent(client=client), history_source="service", session_store=store) async with httpx.AsyncClient(transport=httpx.ASGITransport(app=server), base_url="http://test") as http: stored = await http.post("/responses", json={"input": "remember me", "store": True}) stored_id = stored.json()["id"] @@ -2336,11 +2340,14 @@ async def test_http_outer_storage_is_independent_of_inner_service_history(self) assert await store.get(unstored.json()["id"]) is None @pytest.mark.parametrize("mode", ["service", "agent"]) - async def test_http_outer_background_polling_is_independent_of_inner_history(self, mode: str) -> None: + async def test_http_outer_background_polling_is_independent_of_history_source(self, mode: str) -> None: client = _ServiceStorageRecordingClient() history = [InMemoryHistoryProvider()] if mode == "agent" else [] + default_options = OpenAIChatOptions(store=False) if mode == "agent" else None server = _make_server( - Agent(client=client, context_providers=history), session_store=SessionStore(), inner_history=mode + Agent(client=client, context_providers=history, default_options=default_options), + session_store=SessionStore(), + history_source=mode, ) async with httpx.AsyncClient(transport=httpx.ASGITransport(app=server), base_url="http://test") as http: pending = await http.post("/responses", json={"input": "background", "store": True, "background": True}) diff --git a/python/packages/foundry_hosting/tests/test_responses_int.py b/python/packages/foundry_hosting/tests/test_responses_int.py index 2648e220147..704e186817c 100644 --- a/python/packages/foundry_hosting/tests/test_responses_int.py +++ b/python/packages/foundry_hosting/tests/test_responses_int.py @@ -88,7 +88,7 @@ async def get_weather(location: Annotated[str, "The city name"]) -> str: return f"The weather in {location} is 72°F and sunny." -@pytest.fixture(params=["agent_server", "agent"], ids=["agent-server-history", "agent-history"]) +@pytest.fixture(params=["agent_server", "agent"], ids=["agent-server-history", "agent-managed-service-history"]) def history_server(request: pytest.FixtureRequest) -> ResponsesHostServer: """Create a real Foundry server for each model-history source.""" client = FoundryChatClient(credential=AzureCliCredential()) # pyrefly: ignore[bad-argument-type] @@ -104,7 +104,7 @@ def history_server(request: pytest.FixtureRequest) -> ResponsesHostServer: ) -@pytest.fixture(params=["agent_server", "agent"], ids=["agent-server-history", "agent-history"]) +@pytest.fixture(params=["agent_server", "agent"], ids=["agent-server-history", "agent-managed-service-history"]) def history_server_with_tools(request: pytest.FixtureRequest) -> ResponsesHostServer: """Create a real Foundry tool-calling server for each model-history source.""" client = FoundryChatClient(credential=AzureCliCredential()) # pyrefly: ignore[bad-argument-type] diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md index 17808f71044..b2987a8e464 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md @@ -1,22 +1,24 @@ # Responses agents: history, storage, and options The **outer** Responses request decides whether the hosted response is stored. The developer separately selects -where the **inner** agent gets its conversation history: +who supplies the agent's model history: -| Entry point | `inner_history` | Model history | +| Entry point | `history_source` | Model history | | --- | --- | --- | -| [main.py](main.py) | `"host"` | The outer Responses transcript; the inner client runs with `store=False`. | +| [main.py](main.py) | `"agent_server"` | The outer Responses transcript; the inner client runs with `store=False`. | | [service_history.py](service_history.py) | `"service"` | Only new input goes to the model; its private `service_session_id` persists for later turns. | -| [agent_history.py](agent_history.py) | `"agent"` | `InMemoryHistoryProvider` loads from `AgentSession.state`; inner client storage is off. | -| [options.py](options.py) | `"host"` | The hook removes the caller's token limit so the agent default is used. | -| [provider_background.py](provider_background.py) | `"service"` | Opt-in provider background with a private recovery token. | +| [agent_history.py](agent_history.py) | `"agent"` | This agent opts into `InMemoryHistoryProvider` with `default_options={"store": False}`; another `"agent"` configuration with `store=True` can use service history instead. | +| [options.py](options.py) | `"agent_server"` | The hook removes the caller's token limit so the agent default is used. | +| [provider_background.py](provider_background.py) | `"service"` | `background_source="provider"` opts into provider background with a private recovery token. | For the same two stored requests—first **"My name is Ada"**, then **"What is my name?"** with the first response's `previous_response_id`—the modes differ at the *model* boundary. `main.py` replays both the first user message and the first assistant output before the follow-up. `service_history.py` sends only the follow-up and privately resumes the service thread returned by the first call. `agent_history.py` sends the follow-up plus earlier messages loaded from `InMemoryHistoryProvider` -inside the persisted MAF session. In none of these cases is the caller's `response.id` the +inside the persisted MAF session. With `history_source="agent"`, the agent's storage default determines +whether it uses that provider or downstream service storage; the host does not force either one on +stored requests. In none of these cases is the caller's `response.id` the downstream service ID. Run one entry point at a time. The deployment manifest targets `main.py`; select another script to deploy a different @@ -73,8 +75,9 @@ removing one exposes the agent's own unchanged `default_options`. In [options.py the model receives the agent's `max_tokens=256` default instead. Platform IDs, storage flags, and private continuation tokens are never caller model options; the hook cannot add them back. -`history_source="agent_server"` and `history_source="agent"` remain available with a per-host deprecation warning. -For **stored** requests, the latter preserves the former behavior: only new input goes to the agent, and its -developer-owned defaults may choose either a HistoryProvider **or downstream service storage**. It does **not** -silently become `inner_history="agent"`. Prefer the explicit mode in new code. `store=` as a constructor parameter -is a deprecated alias for `response_store=` (the outer storage *backend*, not the caller's `store` flag). +`history_source="agent_server"` and `history_source="agent"` retain their existing meaning without a +deprecation warning. For **stored** requests, `"agent"` passes only new input to the agent, whose +developer-owned defaults may choose either a HistoryProvider **or downstream service storage**. +`history_source="service"` explicitly requires service-managed history and overrides the agent's +`store` default on stored requests. `store=` as a constructor parameter remains a deprecated alias +for `response_store=` (the outer storage *backend*, not the caller's `store` flag). diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py index 9b7df61ec8d..ddd03a02218 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/agent_history.py @@ -24,7 +24,7 @@ def main() -> None: context_providers=[InMemoryHistoryProvider()], default_options={"store": False}, ) - ResponsesHostServer(agent=agent, inner_history="agent").run() + ResponsesHostServer(agent=agent, history_source="agent").run() if __name__ == "__main__": diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/main.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/main.py index 0f607754d33..6fc1e5a766e 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/main.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/main.py @@ -25,7 +25,7 @@ def main() -> None: instructions="You are a friendly assistant. Keep your answers brief.", ) - server = ResponsesHostServer(agent=agent, inner_history="host") + server = ResponsesHostServer(agent=agent, history_source="agent_server") server.run() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/options.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/options.py index 8b94e03b61a..8b98afc6f22 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/options.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/options.py @@ -33,7 +33,7 @@ def main() -> None: ) ResponsesHostServer( agent=agent, - inner_history="host", + history_source="agent_server", prepare_options=prepare_options, unsupported_options="warn", ).run() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/provider_background.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/provider_background.py index 08db7670cd0..46909f6e1c1 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/provider_background.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/provider_background.py @@ -29,8 +29,8 @@ def create_agent() -> Agent: def main() -> None: ResponsesHostServer( agent=create_agent, - inner_history="service", - inner_background="provider", + history_source="service", + background_source="provider", options=ResponsesServerOptions(resilient_background=True), ).run() diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/service_history.py b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/service_history.py index 9fc52edb6da..ded7e8ddd06 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/service_history.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/service_history.py @@ -23,7 +23,7 @@ def main() -> None: instructions="Be concise.", default_options={"store": True}, ) - ResponsesHostServer(agent=agent, inner_history="service").run() + ResponsesHostServer(agent=agent, history_source="service").run() if __name__ == "__main__": diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md index 5ffb2b9c0ea..18b2388fb71 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/README.md @@ -10,7 +10,7 @@ that rejected turns leave no pending futures. Do not enable steering using a pri queue-length precheck. [main.py](main.py) remains **runnable** as a regular non-steerable countdown agent. It uses -`inner_history="host"` with a `FoundryChatClient` and returns the caller's **outer** `response.id` for +`history_source="agent_server"` with a `FoundryChatClient` and returns the caller's **outer** `response.id` for background polling. Running a second turn concurrently does not steer the first. Start it using the [parent hosting guide](../../README.md), then send one stored background request. The deployment manifests keep their existing agent name for identity compatibility, not because steering is enabled: diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py index 51895babf76..1f37842d505 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/steerable_long_running_agent/main.py @@ -42,7 +42,7 @@ def main() -> None: server = ResponsesHostServer( agent=agent, - inner_history="host", + history_source="agent_server", log_level="DEBUG", ) server.run() From bf94a545e764bf03086d352e5eb0527a832a10f9 Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Tue, 29 Sep 2026 08:23:35 +0200 Subject: [PATCH 09/10] Python: Protect service conversations and provider background polls --- python/packages/foundry_hosting/README.md | 17 +- .../_responses.py | 84 +++- .../foundry_hosting/tests/test_responses.py | 404 +++++++++++++++++- .../responses/basic/README.md | 9 +- 4 files changed, 494 insertions(+), 20 deletions(-) diff --git a/python/packages/foundry_hosting/README.md b/python/packages/foundry_hosting/README.md index 9420827025b..0d2e5677a73 100644 --- a/python/packages/foundry_hosting/README.md +++ b/python/packages/foundry_hosting/README.md @@ -100,6 +100,10 @@ the token is saved cannot safely restart the inner job. A final provider poll re response-ID snapshot so recovery can re-poll it if the outer response was not yet committed; a later turn drops it from its working session. Shutdown during initial submission fails rather than replaying a job whose acceptance is unknown. Cancelling an in-flight submission does not prove the remote provider stopped it. +Each poll retains the caller's generation options and `background=True`, so a tool-loop follow-up requests +another background response and saves its next token. A crash after a local tool side effect but before that +next token is saved can still repeat the tool on recovery; use idempotent tools or avoid provider background +for side-effecting local tools. This mechanism does not provide exactly-once tool execution. Provider background and steering cannot be combined. Regular agent runs without this opt-in are not crash-replayable. **Steering is temporarily unavailable:** `steerable_conversations=True` fails during host construction, before enabling the process-wide TaskManager. @@ -213,11 +217,18 @@ Invocations sessions use the separate `invocation_sessions` store. Each stored agent turn saves a snapshot under its **own** outer `response.id`. For a named `conversation`, a separate mutable conversation-head key is also updated. Loaded MAF sessions use PR1's ETag condition for that key: a competing turn that advanced the head causes a visible conflict rather than a stale overwrite. A superseded -steered turn saves its response snapshot but skips the head update. Service-backed history is linear: when -continuing by `previous_response_id`, the prior response is claimed with a conditional write so a second branch +steered turn saves its response snapshot but skips the head update. In `"service"` or `"agent"` history mode, a +stored named-conversation turn first claims that head with a conditional write *before* calling the inner agent. +Another request that read the old head loses the CAS; one that reads the claim fails before touching the provider. +The claim is cleared when the winning turn successfully commits the new head. If a dispatched turn fails or is +cancelled, the claim remains: the provider may already have changed its thread, so start a new conversation +instead of retrying this one blindly. A recovered provider-background turn must still own the same claim. +When continuing by `previous_response_id`, the prior response is claimed with a conditional write **after** +the input is validated, so an invalid approval response does not consume a usable parent. A second branch cannot reuse the same downstream service thread; attempting to fork a named service conversation is also rejected. New hosted keys are created only if absent. Custom store providers must provide equivalent scoped conditional -writes for concurrent turns. Local callers can still upsert directly without first loading a session. +writes for concurrent turns, including the pre-dispatch claim. Local callers can still upsert directly without +first loading a session. See the [custom storage provider sample](../../samples/04-hosting/foundry-hosted-agents/responses/custom_storage/) for an example that uses an in-memory session store locally and Azure Cosmos DB when hosted. diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py index 1ee0d8e841d..8f67416bced 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py @@ -125,6 +125,7 @@ _HOSTED_PROVIDER_STATE_KEY = "_foundry_provider_background" _HOSTED_SOURCE_CONVERSATION_KEY = "_foundry_source_conversation" _HOSTED_SERVICE_CHILD_KEY = "_foundry_service_child" +_HOSTED_CONVERSATION_CLAIM_KEY = "_foundry_conversation_claim" _HistorySource = Literal["agent_server", "agent", "service"] @@ -159,6 +160,15 @@ def _is_hosted_responses_history_sentinel(provider: ContextProvider) -> bool: ) +def _reject_busy_conversation(session: AgentSession) -> None: + if session.state.get(_HOSTED_CONVERSATION_CLAIM_KEY) is not None: + raise RuntimeError( + "A service-backed conversation already has an in-flight turn. Wait for it to finish; " + "if it failed or was cancelled after dispatch, start a new conversation rather than " + "reusing a possibly changed provider thread." + ) + + def _create_response_event_stream(context: ResponseContext) -> ResponseEventStream: """Create a response stream seeded from recovery state when available.""" if context.is_recovery: @@ -659,6 +669,8 @@ def __init__( background_source: `"agent_server"` (default) runs background work in the Responses host without invoking provider-native background APIs; `"provider"` opts a storing, resumable client into provider background polling and requires `history_source="service"`. + Each poll retains the caller's run options and `background=True`; local tools must be + idempotent because a crash before the next private token is saved can replay them. prepare_options: Developer hook to remove or replace caller model options for an agent. unsupported_options: `"warn"` (default), `"error"`, or `"ignore"` when a custom agent cannot accept runtime model options. @@ -689,6 +701,10 @@ def __init__( 5. Steering is temporarily unavailable for all agents. `steerable_conversations=True` fails at construction, before starting a host or enabling the process-wide TaskManager. The current AgentServer SDK retains futures for rejected turns after the steering queue fills. + 6. A stored, named `"service"` or `"agent"` conversation is claimed in the scoped session store + before the agent runs. Conditional writes prevent a concurrent turn from mutating the same + provider thread. If the turn fails or is cancelled, start a new conversation instead of + reusing a potentially changed downstream thread. Raises: ValueError: If the history, background, or unsupported-options policy is invalid. @@ -944,6 +960,8 @@ async def _handle_prepared_response( f"previous_response_id={previous_response_id}." ) session = agent.create_session() + if context.conversation_id is not None: + _reject_busy_conversation(session) if previous_response_id is not None and context.conversation_id is None: if session.service_session_id is not None and session.state.get( _HOSTED_SOURCE_CONVERSATION_KEY @@ -1220,6 +1238,18 @@ async def _handle_inner_agent( f"Cannot find an existing agent session for previous_response_id={previous_response_id}." ) session = agent.create_session() + if context.conversation_id is not None: + if context.is_recovery and provider_background: + if session_storage is None: + raise RuntimeError("Provider background recovery requires agent session storage.") + head = await session_storage.get(context.conversation_id) + if head is None or head.state.get(_HOSTED_CONVERSATION_CLAIM_KEY) != context.response_id: + raise RuntimeError( + "Cannot recover provider background: the service-backed conversation claim " + "is no longer held by this response." + ) + else: + _reject_busy_conversation(session) if not stored and self._history_source == "agent" and session.service_session_id is not None: raise ValueError( "store=false cannot continue agent-managed downstream service history; " @@ -1238,17 +1268,13 @@ async def _handle_inner_agent( if previous_response_id is not None and context.conversation_id is None and not context.is_recovery: if session.service_session_id is not None and session.state.get(_HOSTED_SOURCE_CONVERSATION_KEY): raise ValueError("A service-managed downstream conversation cannot be forked.") - if stored and self._history_source in ("service", "agent") and session.service_session_id is not None: - if session.state.get(_HOSTED_SERVICE_CHILD_KEY): - raise ValueError("A service-managed downstream response cannot be forked.") - if session_storage is None: - raise RuntimeError("Service history requires agent session storage.") - session.state[_HOSTED_SERVICE_CHILD_KEY] = context.response_id - await session_storage.set(previous_response_id, session) - session.state.pop(_HOSTED_SERVICE_CHILD_KEY) - session.state.pop(_HOSTED_SOURCE_CONVERSATION_KEY, None) - if not context.is_recovery: - session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) + if ( + stored + and self._history_source in ("service", "agent") + and session.service_session_id is not None + and session.state.get(_HOSTED_SERVICE_CHILD_KEY) + ): + raise ValueError("A service-managed downstream response cannot be forked.") except BaseException as ex: # Session preparation failed (or the request was cancelled / the stream closed — # neither of which is an Exception). Cancel and drain the in-flight message-loading @@ -1310,6 +1336,31 @@ async def _handle_inner_agent( if self._unsupported_options == "warn": logger.warning("Agent doesn't support runtime options. They will be ignored.") + if previous_response_id is not None and context.conversation_id is None and not context.is_recovery: + if stored and self._history_source in ("service", "agent") and session.service_session_id is not None: + if session_storage is None: + raise RuntimeError("Service history requires agent session storage.") + session.state[_HOSTED_SERVICE_CHILD_KEY] = context.response_id + try: + await session_storage.set(previous_response_id, session) + finally: + session.state.pop(_HOSTED_SERVICE_CHILD_KEY, None) + session.state.pop(_HOSTED_SOURCE_CONVERSATION_KEY, None) + + if not context.is_recovery: + session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) + + if stored and context.conversation_id is not None and not configuration.agent_server_history: + if session_storage is None: + raise RuntimeError("Service history requires agent session storage.") + # The scoped store's conditional write must succeed before the provider can + # mutate its linear thread; a losing turn never reaches agent.run(). + session.state[_HOSTED_CONVERSATION_CLAIM_KEY] = context.response_id + try: + await session_storage.set(context.conversation_id, session) + finally: + session.state.pop(_HOSTED_CONVERSATION_CLAIM_KEY, None) + inner_stream: ResponseStream[AgentResponseUpdate, AgentResponse[Any]] | None = None if provider_background: if session_storage is None or not isinstance(agent, RawAgent): @@ -1355,6 +1406,8 @@ async def _handle_inner_agent( finally: if configuration.hosted_history: session.state.pop(_HOSTED_RESPONSES_HISTORY_SOURCE_ID, None) + if not provider_background and not context.is_recovery: + session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) # Never persist a session that could resume an inner response contrary to the # history mode or the caller's explicit storage decision. @@ -1407,6 +1460,10 @@ async def _handle_inner_agent( and not superseded_by_steering and not request_interrupted and request_failure is None + and ( + configuration.agent_server_history + or (not cancellation_signal.is_set() and not context.shutdown.is_set()) + ) ): if provider_background: session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) @@ -1538,7 +1595,10 @@ async def save_private_state() -> None: return completed, current = await _await_before_signal( lambda token=continuation_token: run_provider( - cast(ChatOptions[Any], {"continuation_token": token, "store": True}), + cast( + ChatOptions[Any], + {**options, "background": True, "store": True, "continuation_token": token}, + ), input_messages=None, phase="poll", ), diff --git a/python/packages/foundry_hosting/tests/test_responses.py b/python/packages/foundry_hosting/tests/test_responses.py index 1a5393745d4..10cb97758c1 100644 --- a/python/packages/foundry_hosting/tests/test_responses.py +++ b/python/packages/foundry_hosting/tests/test_responses.py @@ -99,6 +99,7 @@ from agent_framework_foundry_hosting._state_store import ( AgentSessionStoreProvider, CheckpointStoreProvider, + FoundryAgentSessionStore, FunctionApprovalStoreProvider, ) @@ -1482,6 +1483,39 @@ async def test_service_history_rejects_second_child_of_provider_response(self) - assert "cannot be forked" in branch.json()["error"]["message"] assert len(client.calls) == 2 + async def test_invalid_approval_does_not_consume_service_parent(self) -> None: + client = _ServiceStorageRecordingClient() + store = SessionStore() + server = _make_server(Agent(client=client), session_store=store, history_source="service") + first = await _post(server, input_text="first") + parent_id = first.json()["id"] + + invalid = await _post_json( + server, + { + "model": "test-model", + "input": [ + {"type": "mcp_approval_response", "approval_request_id": "unknown-approval", "approve": True} + ], + "store": True, + "previous_response_id": parent_id, + }, + ) + + assert invalid.json()["status"] == "failed" + assert "unknown-approval" in invalid.json()["error"]["message"] + parent = await store.get(parent_id) + assert parent is not None and "_foundry_service_child" not in parent.state + assert len(client.calls) == 1 + + retry = await _post(server, input_text="corrected", previous_response_id=parent_id) + assert retry.json()["status"] == "completed" + assert [[message.text for message in call] for call in client.calls] == [["first"], ["corrected"]] + fork = await _post(server, input_text="another child", previous_response_id=parent_id) + assert fork.json()["status"] == "failed" + assert "cannot be forked" in fork.json()["error"]["message"] + assert len(client.calls) == 2 + async def test_agent_history_uses_provider_with_nonstoring_defaults(self) -> None: client = _ServiceStorageRecordingClient() history = InMemoryHistoryProvider() @@ -1778,7 +1812,7 @@ async def run( options=ResponsesServerOptions(resilient_background=True), response_store=FileResponseStore(storage_dir=tmp_path), ) - request = CreateResponse(input="hello", store=True, background=True) + request = CreateResponse(input="hello", store=True, background=True, temperature=0.35) context = ResponseContext(response_id="outer-response", mode_flags=MagicMock()) with ( patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), @@ -1787,7 +1821,8 @@ async def run( events = [event async for event in server._handle_response(request, context, asyncio.Event())] assert [event.get("type") for event in events if isinstance(event, Mapping)][-1] == "response.completed" - assert [call.get("background") for call in calls] == [True, None] + assert [call.get("background") for call in calls] == [True, True] + assert [call.get("temperature") for call in calls] == [0.35, 0.35] assert calls[1]["continuation_token"] == {"response_id": "private-provider-token"} assert "private-provider-token" not in str(events) saved = await store.get("outer-response") @@ -1973,12 +2008,194 @@ async def run( ] assert [event.get("type") for event in events if isinstance(event, Mapping)][-1] == "response.completed" - assert [call.get("background") for call in calls] == [True, None] + assert [call.get("background") for call in calls] == [True, True] saved = await store.get("outer-recover") assert saved is not None and saved.service_session_id == "next-service-session" assert saved.state["_foundry_provider_background"]["completed"] is True assert "private-provider-token" not in str(events) + @pytest.mark.parametrize("claim_stolen", [False, True]) + async def test_provider_recovery_retains_named_conversation_claim( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, claim_stolen: bool + ) -> None: + response_id = f"outer-{uuid.uuid4().hex}" + conversation_id = f"conversation-{uuid.uuid4().hex}" + token = OpenAIContinuationToken(response_id="private-provider-token") + calls: list[dict[str, Any]] = [] + + async def run( + messages: Any = None, + *, + session: AgentSession, + options: Mapping[str, Any], + **kwargs: Any, + ) -> AgentResponse: + del messages, kwargs + calls.append(dict(options)) + if "continuation_token" not in options: + return AgentResponse(messages=[], continuation_token=token) + session.service_session_id = "private-service-thread" + return AgentResponse(messages=[Message(role="assistant", contents=[Content.from_text("done")])]) + + def make_host() -> ResponsesHostServer: + agent = Agent(client=_ServiceStorageRecordingClient()) + monkeypatch.setattr(agent, "run", MagicMock(side_effect=run)) + return _make_server( + agent, + history_source="service", + background_source="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + + request = CreateResponse(input="hello", store=True, background=True, temperature=0.25) + initial_context = ResponseContext( + response_id=response_id, conversation_id=conversation_id, mode_flags=MagicMock() + ) + initial_host = make_host() + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch( + "agent_framework_foundry_hosting._responses.asyncio.sleep", + new=AsyncMock(side_effect=ResponseExitForRecovery()), + ), + pytest.raises(ResponseExitForRecovery), + ): + _ = [event async for event in initial_host._handle_response(request, initial_context, asyncio.Event())] + + store = AgentSessionStoreProvider().get_store( + config=initial_host.config, platform_context=get_request_context() + ) + claimed = await store.get(conversation_id) + assert claimed is not None and claimed.state["_foundry_conversation_claim"] == response_id + if claim_stolen: + claimed.state["_foundry_conversation_claim"] = "another-response" + await store.set(conversation_id, claimed) + + recovered_context = ResponseContext( + response_id=response_id, conversation_id=conversation_id, mode_flags=MagicMock() + ) + recovered_context.is_recovery = True + recovered_host = make_host() + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + ): + events = [ + event async for event in recovered_host._handle_response(request, recovered_context, asyncio.Event()) + ] + + head = await store.get(conversation_id) + assert head is not None + if claim_stolen: + assert "claim is no longer held" in _failure_message(events) + assert head.state["_foundry_conversation_claim"] == "another-response" + assert len(calls) == 1 + else: + terminal = events[-1] + assert isinstance(terminal, Mapping) and terminal["type"] == "response.completed" + assert [call.get("background") for call in calls] == [True, True] + assert [call.get("temperature") for call in calls] == [0.25, 0.25] + assert "_foundry_conversation_claim" not in head.state + assert head.service_session_id == "private-service-thread" + assert "private-provider-token" not in str(events) + + async def test_provider_background_polls_keep_options_and_new_token_after_tool(self, tmp_path: Path) -> None: + executions: list[str] = [] + + @tool(approval_mode="never_require") + def send_email(to: str) -> str: + executions.append(to) + return "sent" + + def openai_response(response_id: str, status: str, output: Any | None = None) -> MagicMock: + response = MagicMock() + response.id = response_id + response.status = status + response.conversation = None + response.model = "test-model" + response.created_at = 1_700_000_000 + response.usage = None + response.metadata = {} + response.incomplete_details = None + response.output = [] if output is None else [output] + response.parse = MagicMock(return_value=response) + response.headers = {} + return response + + call = MagicMock( + type="function_call", + call_id="call_1", + arguments='{"to": "bob"}', + id="fc_1", + status="completed", + ) + call.name = "send_email" + message = MagicMock( + type="message", + content=[MagicMock(type="output_text", text="Email sent.", annotations=[], logprobs=None)], + ) + create = AsyncMock( + side_effect=[ + openai_response("private-first-token", "in_progress"), + openai_response("private-second-token", "in_progress"), + ] + ) + retrieve = AsyncMock( + side_effect=[ + openai_response("private-first-token", "completed", call), + openai_response("private-second-token", "completed", message), + ] + ) + client = OpenAIChatClient(model="test-model", api_key="test-key") + client.function_invocation_configuration["max_iterations"] = 4 + agent = Agent(client=client, tools=[send_email]) + store = SessionStore() + server = _make_server( + agent, + session_store=store, + history_source="service", + background_source="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + context = ResponseContext(response_id="outer-tool-response", mode_flags=MagicMock()) + request = CreateResponse(input="email bob", store=True, background=True, temperature=0.42) + + with ( + patch.object( + ResponseContext, + "get_input_items", + new=AsyncMock(return_value=[cast(Item, {"type": "message", "role": "user", "content": "email bob"})]), + ), + patch.object(client.client.responses.with_raw_response, "create", new=create), + patch.object(client.client.responses.with_raw_response, "retrieve", new=retrieve), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + ): + events = [event async for event in server._handle_response(request, context, asyncio.Event())] + + terminal = events[-1] + assert isinstance(terminal, Mapping) + if terminal["type"] == "response.failed": + pytest.fail(_failure_message(events)) + assert terminal["type"] == "response.completed" + assert executions == ["bob"] + assert create.await_count == retrieve.await_count == 2 + assert [entry.kwargs["background"] for entry in create.await_args_list] == [True, True] + assert [entry.kwargs["temperature"] for entry in create.await_args_list] == [0.42, 0.42] + assert [entry.args[0] for entry in retrieve.await_args_list] == [ + "private-first-token", + "private-second-token", + ] + saved = await store.get(context.response_id) + assert saved is not None + assert saved.state["_foundry_provider_background"]["continuation_token"] == { + "response_id": "private-second-token" + } + assert saved.state["_foundry_provider_background"]["completed"] is True + assert "private-first-token" not in str(events) + assert "private-second-token" not in str(events) + @pytest.mark.parametrize("stage", ["submit", "sleep", "poll"]) @pytest.mark.parametrize("interruption", ["cancel", "shutdown"]) async def test_provider_background_observes_lifecycle_during_blocked_work( @@ -2179,7 +2396,7 @@ def shutdown_before_output(response: AgentResponse, response_id: str) -> list[Ag assert [event.get("type") for event in events if isinstance(event, Mapping)][-1] == "response.completed" assert "private-provider-token" not in str(events) - assert [option.get("background") for option in calls] == [True, None, None] + assert [option.get("background") for option in calls] == [True, True, True] assert all(option["continuation_token"] == token for option in calls[1:]) @pytest.mark.parametrize("phase", ["submit", "poll", "save"]) @@ -2295,6 +2512,185 @@ async def updates() -> AsyncIterator[AgentResponseUpdate]: assert second_snapshot is not None and second_snapshot.state["turn"] == 2 assert conversation_head is not None and conversation_head.state["turn"] == 2 + @pytest.mark.parametrize("existing_head", [False, True]) + async def test_service_conversation_claim_prevents_parallel_provider_dispatch(self, existing_head: bool) -> None: + conversation_id = f"conversation-{uuid.uuid4().hex}" + both_loaded = asyncio.Event() + release_provider = asyncio.Event() + provider_started = asyncio.Event() + reads = 0 + original_get = FoundryAgentSessionStore.get + + async def concurrent_get(store: FoundryAgentSessionStore, key: str) -> AgentSession | None: + nonlocal reads + session = await original_get(store, key) + if key == conversation_id and reads < 2: + reads += 1 + if reads == 2: + both_loaded.set() + await asyncio.wait_for(both_loaded.wait(), timeout=10) + return session + + calls: list[int] = [] + agent = _make_agent() + agent.client.STORES_BY_DEFAULT = True + + def run(*args: Any, **kwargs: Any) -> ResponseStream[AgentResponseUpdate, AgentResponse]: + del args + turn = len(calls) + 1 + calls.append(turn) + kwargs["session"].service_session_id = "private-service-thread" + + async def updates() -> AsyncIterator[AgentResponseUpdate]: + if turn == 1: + provider_started.set() + await release_provider.wait() + yield AgentResponseUpdate(contents=[Content.from_text(f"turn {turn}")], role="assistant") + + return ResponseStream(updates(), finalizer=AgentResponse.from_updates) + + agent.run = MagicMock(side_effect=run) + server = _make_server(agent, history_source="service") + if existing_head: + seed_store = server._session_storage_provider.get_store( # pyright: ignore[reportPrivateUsage] + config=server.config, platform_context=get_request_context() + ) + await seed_store.set(conversation_id, AgentSession(service_session_id="private-service-thread")) + + async def collect(response_id: str) -> list[Any]: + context = ResponseContext( + response_id=response_id, + conversation_id=conversation_id, + mode_flags=MagicMock(), + ) + return [ + event + async for event in server._handle_response( + CreateResponse(input="hello", store=True), context, asyncio.Event() + ) + ] + + with ( + patch.object(FoundryAgentSessionStore, "get", new=concurrent_get), + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + ): + pending = {asyncio.create_task(collect(f"response-{uuid.uuid4().hex}")) for _ in range(2)} + try: + await asyncio.wait_for(provider_started.wait(), timeout=10) + done, pending = await asyncio.wait(pending, timeout=10, return_when=asyncio.FIRST_COMPLETED) + assert len(done) == len(pending) == 1 + assert "Another request advanced this agent session" in _failure_message(next(iter(done)).result()) + assert calls == [1] + + blocked = await collect("blocked-while-provider-running") + assert "in-flight turn" in _failure_message(blocked) + assert calls == [1] + + release_provider.set() + winner = await asyncio.wait_for(next(iter(pending)), timeout=10) + assert winner[-1]["type"] == "response.completed" + head_store = server._session_storage_provider.get_store( # pyright: ignore[reportPrivateUsage] + config=server.config, platform_context=get_request_context() + ) + head = await head_store.get(conversation_id) + assert head is not None and "_foundry_conversation_claim" not in head.state + + following = await collect("after-claim-released") + assert following[-1]["type"] == "response.completed" + assert calls == [1, 2] + finally: + release_provider.set() + for task in pending: + task.cancel() + await asyncio.gather(*pending, return_exceptions=True) + + async def test_failed_service_conversation_dispatch_keeps_claim(self) -> None: + store = SessionStore() + agent = _make_agent() + agent.client.STORES_BY_DEFAULT = True + agent.run = MagicMock( + side_effect=lambda **kwargs: ResponseStream( + _raising_updates("provider timed out"), finalizer=AgentResponse.from_updates + ) + ) + server = _make_server(agent, session_store=store, history_source="service") + conversation_id = "failed-service-conversation" + + async def collect(response_id: str) -> list[Any]: + context = ResponseContext(response_id=response_id, conversation_id=conversation_id, mode_flags=MagicMock()) + return [ + event + async for event in server._handle_response( + CreateResponse(input="hello", store=True), context, asyncio.Event() + ) + ] + + with patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])): + failed = await collect("failed-dispatch") + blocked = await collect("retry-after-failure") + + assert "provider timed out" in _failure_message(failed) + assert "in-flight turn" in _failure_message(blocked) + agent.run.assert_called_once() + head = await store.get(conversation_id) + assert head is not None and head.state["_foundry_conversation_claim"] == "failed-dispatch" + + async def test_cancelled_service_conversation_does_not_release_claim(self) -> None: + store = SessionStore() + started = asyncio.Event() + stopped = asyncio.Event() + agent = _make_agent() + agent.client.STORES_BY_DEFAULT = True + + async def blocked() -> AsyncIterator[AgentResponseUpdate]: + started.set() + try: + await asyncio.Event().wait() + finally: + stopped.set() + yield AgentResponseUpdate(contents=[Content.from_text("too late")], role="assistant") + + agent.run = MagicMock(side_effect=lambda **kwargs: ResponseStream(blocked())) + server = _make_server(agent, session_store=store, history_source="service") + conversation_id = "cancelled-service-conversation" + context = ResponseContext( + response_id="cancelled-dispatch", conversation_id=conversation_id, mode_flags=MagicMock() + ) + signal = asyncio.Event() + + async def collect(response_context: ResponseContext, cancellation: asyncio.Event) -> list[Any]: + return [ + event + async for event in server._handle_response( + CreateResponse(input="hello", store=True), response_context, cancellation + ) + ] + + with patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])): + pending = asyncio.create_task(collect(context, signal)) + try: + await asyncio.wait_for(started.wait(), timeout=5) + context.client_cancelled = True + signal.set() + cancelled = await asyncio.wait_for(pending, timeout=5) + assert stopped.is_set() + assert not any( + isinstance(event, Mapping) and event.get("type") == "response.completed" for event in cancelled + ) + head = await store.get(conversation_id) + assert head is not None and head.state["_foundry_conversation_claim"] == context.response_id + + retry_context = ResponseContext( + response_id="retry-cancelled", conversation_id=conversation_id, mode_flags=MagicMock() + ) + retry = await collect(retry_context, asyncio.Event()) + assert "in-flight turn" in _failure_message(retry) + agent.run.assert_called_once() + finally: + pending.cancel() + with suppress(asyncio.CancelledError): + await pending + async def test_conversation_write_conflict_keeps_response_snapshot(self) -> None: store = _ConflictingConversationStore() await store.set("conversation-head", AgentSession()) diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md index b2987a8e464..9f6bd30feed 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md @@ -49,7 +49,10 @@ background runs inside AgentServer even if the chat client cannot run in the bac continuation, a process crash can leave such a response unfinished. The optional [provider_background.py](provider_background.py) opts a storing Responses client into its *separate* background mode: only this mode persists the provider's private token and polls it until completion/recovery. It is incompatible -with steering. The deployed identity needs Foundry User permission on the project for private provider polling. +with steering. Polls keep the original model options and `background=True`, including when a tool loop submits +its next leg. A crash between a local tool side effect and saving the next private token can still replay +that tool; use idempotent tools or avoid this mode for side-effecting local tools. The deployed identity +needs Foundry User permission on the project for private provider polling. [client.py](client.py) shows stored conversation turns, an unstored request, and background polling. It also needs `FOUNDRY_AGENT_NAME` and Azure CLI authentication. For a local host, a simple multi-turn request is: @@ -64,6 +67,10 @@ Send another request with the same `conversation` to continue. When deployed, ke `AgentSession.service_session_id`. A `conversation` binds that sandbox; with a bare `previous_response_id`, also forward the earlier response's `agent_session_id`. Older unscoped hosted MAF state is not migrated; start a new conversation after upgrading (see the [package state guide](../../../../../packages/foundry_hosting/README.md#state-store)). +For `"service"` and `"agent"` history, a stored named turn claims the conversation before inner dispatch: concurrent +turns fail rather than mutating the same service thread. An invalid input does not consume a +`previous_response_id` parent. If a named turn fails or is cancelled after dispatch, start a new conversation; +the existing thread may have changed even if its outer response did not complete. ## Options and compatibility From 8ffdbb35235a0f88b3f96b4a12ceb71f6582eed5 Mon Sep 17 00:00:00 2001 From: eavanvalkenburg Date: Wed, 30 Sep 2026 09:46:07 +0200 Subject: [PATCH 10/10] Python: Preserve provider background output through recovery --- python/packages/foundry_hosting/README.md | 14 +- .../_responses.py | 173 ++++++++++-- .../foundry_hosting/tests/test_responses.py | 250 ++++++++++++++++-- .../responses/basic/README.md | 7 +- 4 files changed, 395 insertions(+), 49 deletions(-) diff --git a/python/packages/foundry_hosting/README.md b/python/packages/foundry_hosting/README.md index f03146adff5..ab8fb4b5cc1 100644 --- a/python/packages/foundry_hosting/README.md +++ b/python/packages/foundry_hosting/README.md @@ -96,10 +96,13 @@ background automatically. `background_source="agent_server"` (default) uses only `background_source="provider"` is a separate opt-in for `history_source="service"` with a storing Responses client. Its private continuation token is saved under the outer ID and never returned to the caller. Use `ResponsesServerOptions(resilient_background=True)` to permit recovery from a **saved** token; a crash before -the token is saved cannot safely restart the inner job. A final provider poll retains that token in the private -response-ID snapshot so recovery can re-poll it if the outer response was not yet committed; a later turn drops -it from its working session. Shutdown during initial submission fails rather than replaying a job whose -acceptance is unknown. Cancelling an in-flight submission does not prove the remote provider stopped it. +the token is saved cannot safely restart the inner job. Completed polling output, including local function calls, +results, and usage, is saved together with the next token in the private response-ID snapshot before emission. +Outer output checkpoints record which saved batches have been emitted, so recovery restores their usage and +replays only uncheckpointed output. Once the final output is saved, recovery can finish from that snapshot without +calling the provider again; a later turn drops it from its working session. Shutdown during initial submission +fails rather than replaying a job whose acceptance is unknown. Cancelling an in-flight submission does not prove +the remote provider stopped it. Each poll retains the caller's generation options and `background=True`, so a tool-loop follow-up requests another background response and saves its next token. A crash after a local tool side effect but before that next token is saved can still repeat the tool on recovery; use idempotent tools or avoid provider background @@ -242,6 +245,9 @@ Another request that read the old head loses the CAS; one that reads the claim f The claim is cleared when the winning turn successfully commits the new head. If a dispatched turn fails or is cancelled, the claim remains: the provider may already have changed its thread, so start a new conversation instead of retrying this one blindly. A recovered provider-background turn must still own the same claim. +The committed head also records its completing outer response ID. If a crash occurs after the head write but before +outer completion, that response can recover its saved final output without reclaiming or rewriting the head. +A different in-flight claim or completing response is not accepted as ownership. When continuing by `previous_response_id`, the prior response is claimed with a conditional write **after** the input is validated, so an invalid approval response does not consume a usable parent. A second branch cannot reuse the same downstream service thread; attempting to fork a named service conversation is also rejected. diff --git a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py index 5fd49a5bca4..a8a1d037916 100644 --- a/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py +++ b/python/packages/foundry_hosting/agent_framework_foundry_hosting/_responses.py @@ -22,6 +22,7 @@ Sequence, ) from contextlib import AbstractAsyncContextManager, AsyncExitStack, aclosing, suppress +from copy import copy from dataclasses import asdict, dataclass, is_dataclass from typing import Generic, Literal, TypeGuard, TypeVar, cast from urllib.parse import urlparse @@ -126,6 +127,9 @@ _HOSTED_SOURCE_CONVERSATION_KEY = "_foundry_source_conversation" _HOSTED_SERVICE_CHILD_KEY = "_foundry_service_child" _HOSTED_CONVERSATION_CLAIM_KEY = "_foundry_conversation_claim" +_HOSTED_CONVERSATION_COMMITTED_KEY = "_foundry_conversation_committed" +_HOSTED_PROVIDER_OUTPUT_COUNT_KEY = "_foundry_provider_output_count" +_HOSTED_PROVIDER_USAGE_KEY = "_foundry_provider_usage" _HistorySource = Literal["agent_server", "agent", "service"] @@ -1172,7 +1176,7 @@ async def _handle_inner_agent( agent: SupportsAgentRun, configuration: _AgentConfiguration, hosted_request: HostedResponseRequest, - ) -> AsyncGenerator[ResponseStreamEvent]: + ) -> AsyncGenerator[ResponseStreamEvent | ResponseCheckpointEvent]: """Handle a regular (non-workflow) agent. The response stream, tracker, and opening lifecycle events are produced @@ -1187,6 +1191,7 @@ async def _handle_inner_agent( stored = request.get("store") is not False request_messages_task: asyncio.Task[list[Message]] | None = None + conversation_already_committed = False try: if isinstance(agent, RawAgent): validate_default_transport_options( @@ -1285,12 +1290,22 @@ async def _handle_inner_agent( f"Cannot find an existing agent session for previous_response_id={previous_response_id}." ) session = agent.create_session() + provider_state = session.state.get(_HOSTED_PROVIDER_STATE_KEY) if context.conversation_id is not None: if context.is_recovery and provider_background: if session_storage is None: raise RuntimeError("Provider background recovery requires agent session storage.") head = await session_storage.get(context.conversation_id) - if head is None or head.state.get(_HOSTED_CONVERSATION_CLAIM_KEY) != context.response_id: + conversation_already_committed = ( + head is not None + and head.state.get(_HOSTED_CONVERSATION_CLAIM_KEY) is None + and head.state.get(_HOSTED_CONVERSATION_COMMITTED_KEY) == context.response_id + and isinstance(provider_state, Mapping) + and cast(Mapping[str, Any], provider_state).get("completed") is True + ) + if not conversation_already_committed and ( + head is None or head.state.get(_HOSTED_CONVERSATION_CLAIM_KEY) != context.response_id + ): raise RuntimeError( "Cannot recover provider background: the service-backed conversation claim " "is no longer held by this response." @@ -1302,7 +1317,6 @@ async def _handle_inner_agent( "store=false cannot continue agent-managed downstream service history; " "start a new one-shot request or use store=true." ) - provider_state = session.state.get(_HOSTED_PROVIDER_STATE_KEY) if ( not context.is_recovery and provider_state is not None @@ -1396,8 +1410,14 @@ async def _handle_inner_agent( if not context.is_recovery: session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) + session.state.pop(_HOSTED_CONVERSATION_COMMITTED_KEY, None) - if stored and context.conversation_id is not None and not configuration.agent_server_history: + if ( + stored + and context.conversation_id is not None + and not configuration.agent_server_history + and not conversation_already_committed + ): if session_storage is None: raise RuntimeError("Service history requires agent session storage.") # The scoped store's conditional write must succeed before the provider can @@ -1409,10 +1429,14 @@ async def _handle_inner_agent( session.state.pop(_HOSTED_CONVERSATION_CLAIM_KEY, None) inner_stream: ResponseStream[AgentResponseUpdate, AgentResponse[Any]] | None = None + updates: _SignalledIterator[AgentResponseUpdate] | None = None if provider_background: if session_storage is None or not isinstance(agent, RawAgent): raise RuntimeError("Provider background requires a stored MAF agent session.") - updates = self._provider_background_updates( + output_count = response_event_stream.internal_metadata.get(_HOSTED_PROVIDER_OUTPUT_COUNT_KEY, 0) + if type(output_count) is not int or output_count < 0: + raise RuntimeError("The persisted provider output cursor is invalid.") + responses = self._provider_background_responses( agent=cast(RawAgent[ChatOptions[Any]], agent), messages=messages, session=session, @@ -1420,20 +1444,36 @@ async def _handle_inner_agent( options=chat_options, context=context, cancellation_signal=cancellation_signal, + emitted_output_count=output_count, ) + async with aclosing(responses): + async for response, output_count in responses: + for update in _agent_response_updates(response, context.response_id): + if context.shutdown.is_set() or cancellation_signal.is_set(): + break + async for event in tracker.handle_update(update, approval_storage=approval_storage): + yield event + if context.shutdown.is_set() or cancellation_signal.is_set(): + continue + for event in tracker.close(): + yield event + response_event_stream.internal_metadata[_HOSTED_PROVIDER_OUTPUT_COUNT_KEY] = output_count + response_event_stream.internal_metadata[_HOSTED_PROVIDER_USAGE_KEY] = tracker.usage_details + if self._resilient_background: + yield response_event_stream.checkpoint() else: inner_stream = agent.run(stream=True, **run_kwargs) # type: ignore[reportUnknownMemberType] updates = _SignalledIterator(inner_stream, context.shutdown, cancellation_signal) - async with aclosing(updates): - async for update in updates: - if not stored and any( - content.type in ("function_approval_request", "oauth_consent_request") - or content.user_input_request - for content in update.contents - ): - raise ValueError("Approval and user-input continuation requires store=true.") - async for event in tracker.handle_update(update, approval_storage=approval_storage): - yield event + async with aclosing(updates): + async for update in updates: + if not stored and any( + content.type in ("function_approval_request", "oauth_consent_request") + or content.user_input_request + for content in update.contents + ): + raise ValueError("Approval and user-input continuation requires store=true.") + async for event in tracker.handle_update(update, approval_storage=approval_storage): + yield event if inner_stream is not None and isinstance(updates, _SignalledIterator) and not updates.signalled: final = await inner_stream.get_final_response() if final.continuation_token is not None and final.finish_reason is None: @@ -1504,6 +1544,7 @@ async def _handle_inner_agent( await session_storage.set(context.response_id, session) if ( context.conversation_id is not None + and not conversation_already_committed and not superseded_by_steering and not request_interrupted and request_failure is None @@ -1514,6 +1555,7 @@ async def _handle_inner_agent( ): if provider_background: session.state.pop(_HOSTED_PROVIDER_STATE_KEY, None) + session.state[_HOSTED_CONVERSATION_COMMITTED_KEY] = context.response_id await session_storage.set(context.conversation_id, session) except Exception as save_error: save_failure = save_error @@ -1535,7 +1577,7 @@ async def _handle_inner_agent( elif save_failure is not None: raise save_failure - async def _provider_background_updates( + async def _provider_background_responses( self, *, agent: RawAgent[ChatOptions[Any]], @@ -1545,8 +1587,10 @@ async def _provider_background_updates( options: ChatOptions[Any], context: ResponseContext, cancellation_signal: asyncio.Event, - ) -> AsyncGenerator[AgentResponseUpdate]: - """Keep the inner provider token private while the outer response ID is polled.""" + emitted_output_count: int, + ) -> AsyncGenerator[tuple[AgentResponse[Any], int]]: + """Persist polling output with its private token and replay uncheckpointed output on recovery.""" + outputs: list[dict[str, Any]] = [] async def proceed(*, has_token: bool) -> bool: if cancellation_signal.is_set() and context.client_cancelled: @@ -1586,6 +1630,46 @@ async def save_private_state() -> None: return raise RuntimeError("Could not save private provider background state; inspect the host logs.") + def record_output(response: AgentResponse[Any]) -> None: + messages = response.messages + if response.continuation_token is not None: + # A tool loop prefixes completed calls/results to the unfinished model + # response. Retain that prefix, not partial output that polling will repeat. + for index in range(len(messages) - 1, -1, -1): + message = messages[index] + last_result = next( + ( + offset + for offset in range(len(message.contents) - 1, -1, -1) + if message.contents[offset].type == "function_result" + ), + None, + ) + if last_result is not None: + completed_message = copy(message) + completed_message.contents = message.contents[: last_result + 1] + messages = [*messages[:index], completed_message] + break + else: + return + outputs.append( + AgentResponse( + messages=messages, + usage_details=response.usage_details, + finish_reason=FinishReason(response.finish_reason) if response.finish_reason is not None else None, + ).to_dict() + ) + + async def pending_outputs() -> AsyncGenerator[tuple[AgentResponse[Any], int]]: + nonlocal emitted_output_count + while emitted_output_count < len(outputs): + if not await proceed(has_token=True): + return + yield AgentResponse.from_dict(outputs[emitted_output_count]), emitted_output_count + 1 + if not await proceed(has_token=True): + return + emitted_output_count += 1 + if context.is_recovery: saved = session.state.get(_HOSTED_PROVIDER_STATE_KEY) if not isinstance(saved, Mapping): @@ -1597,6 +1681,19 @@ async def save_private_state() -> None: if not isinstance(token, Mapping): raise RuntimeError("The stored provider continuation token is invalid.") continuation_token: Mapping[str, Any] = cast(Mapping[str, Any], token) + saved_outputs = saved_payload.get("outputs", []) + if not isinstance(saved_outputs, list) or any( + not isinstance(output, dict) for output in cast(list[object], saved_outputs) + ): + raise RuntimeError("The stored provider output is invalid.") + outputs = cast(list[dict[str, Any]], saved_outputs) + if emitted_output_count > len(outputs): + raise RuntimeError("The persisted provider output cursor exceeds its private snapshot.") + async for output in pending_outputs(): + yield output + if saved_payload.get("completed") is True and outputs: + await proceed(has_token=True) + return else: if not await proceed(has_token=False): return @@ -1618,17 +1715,21 @@ async def save_private_state() -> None: if first.continuation_token is None: if not await proceed(has_token=False): return - for update in _agent_response_updates(first, context.response_id): - yield update + yield first, 1 return if not isinstance(first.continuation_token, Mapping): raise RuntimeError("The provider returned a continuation token that cannot be persisted.") continuation_token = cast(Mapping[str, Any], first.continuation_token) + if any(message.contents for message in first.messages) or first.usage_details: + record_output(first) session.state[_HOSTED_PROVIDER_STATE_KEY] = { "outer_response_id": context.response_id, "continuation_token": dict(continuation_token), + "outputs": outputs, } await save_private_state() + async for output in pending_outputs(): + yield output while True: if not await proceed(has_token=True): @@ -1659,27 +1760,34 @@ async def save_private_state() -> None: if current is None: raise RuntimeError("The provider did not return a background response.") if current.continuation_token is None: - # Until AgentServer commits the outer terminal event, recovery may need to re-poll this ID. + # Save the completed output with its token before publishing it; recovery + # replays the uncheckpointed output without invoking the agent again. + record_output(current) session.state[_HOSTED_PROVIDER_STATE_KEY] = { "outer_response_id": context.response_id, "continuation_token": dict(continuation_token), "completed": True, + "outputs": outputs, } await save_private_state() - for update in _agent_response_updates(current, context.response_id): - if not await proceed(has_token=True): - return - yield update - await proceed(has_token=True) + async for output in pending_outputs(): + yield output return if not isinstance(current.continuation_token, Mapping): raise RuntimeError("The provider returned a continuation token that cannot be persisted.") + if current.continuation_token == continuation_token: + continue + if any(message.contents for message in current.messages) or current.usage_details: + record_output(current) continuation_token = cast(Mapping[str, Any], current.continuation_token) session.state[_HOSTED_PROVIDER_STATE_KEY] = { "outer_response_id": context.response_id, "continuation_token": dict(continuation_token), + "outputs": outputs, } await save_private_state() + async for output in pending_outputs(): + yield output async def _handle_inner_workflow( self, @@ -1957,6 +2065,14 @@ def __init__( self._stream = stream self._allowed_oauth_consent_origins = allowed_oauth_consent_origins self._usage_details: UsageDetails | None = None + persisted_usage = stream.internal_metadata.get(_HOSTED_PROVIDER_USAGE_KEY) + if persisted_usage is not None: + if not isinstance(persisted_usage, Mapping) or any( + not isinstance(key, str) or (value is not None and type(value) is not int) + for key, value in cast(Mapping[object, object], persisted_usage).items() + ): + raise RuntimeError("The persisted provider usage is invalid.") + self._usage_details = cast(UsageDetails, dict(cast(Mapping[str, int | None], persisted_usage))) self._active_type: str | None = None self._active_id: str | None = None # message_id of the update that opened the active text item, used to detect a new @@ -2005,6 +2121,11 @@ def __init__( if isinstance(consent_link, str) and isinstance(server_label, str): self._oauth_consent_requests.add((consent_link, server_label)) + @property + def usage_details(self) -> UsageDetails | None: + """Return usage retained with the provider output checkpoint.""" + return self._usage_details + @property def usage(self) -> ResponseUsage | None: """Return accumulated usage in the Responses API schema.""" diff --git a/python/packages/foundry_hosting/tests/test_responses.py b/python/packages/foundry_hosting/tests/test_responses.py index 08da6d70cc2..95c8feb7dfe 100644 --- a/python/packages/foundry_hosting/tests/test_responses.py +++ b/python/packages/foundry_hosting/tests/test_responses.py @@ -70,12 +70,19 @@ ) from azure.ai.agentserver.responses._id_generator import IdGenerator from azure.ai.agentserver.responses.aio import ResponseEventStream -from azure.ai.agentserver.responses.models import CreateResponse, Item, OutputItem, ResponseIncompleteReason +from azure.ai.agentserver.responses.models import ( + CreateResponse, + Item, + OutputItem, + ResponseIncompleteReason, + ResponseObject, +) from azure.ai.agentserver.responses.streaming._checkpoint import ResponseCheckpointEvent from mcp import McpError from mcp.types import ErrorData from openai import AsyncOpenAI, DefaultAsyncHttpxClient from openai.types.responses.response_input_item_param import ResponseInputItemParam +from openai.types.responses.response_usage import ResponseUsage as OpenAIResponseUsage from pydantic import TypeAdapter from typing_extensions import Any @@ -2102,7 +2109,127 @@ def make_host() -> ResponsesHostServer: assert head.service_session_id == "private-service-thread" assert "private-provider-token" not in str(events) - async def test_provider_background_polls_keep_options_and_new_token_after_tool(self, tmp_path: Path) -> None: + @pytest.mark.parametrize("ownership", ["same-response", "other-claim", "other-completion", "claim-during-recovery"]) + async def test_provider_recovery_after_conversation_head_commit( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ownership: str + ) -> None: + response_id = f"outer-{uuid.uuid4().hex}" + conversation_id = f"conversation-{uuid.uuid4().hex}" + token = OpenAIContinuationToken(response_id="private-provider-token") + run = AsyncMock() + + async def provider_run( + messages: Any = None, + *, + session: AgentSession, + options: Mapping[str, Any], + **kwargs: Any, + ) -> AgentResponse: + del messages, kwargs + await run(options) + if "continuation_token" not in options: + return AgentResponse(messages=[], continuation_token=token) + session.service_session_id = "private-service-thread" + return AgentResponse(messages=[Message(role="assistant", contents=[Content.from_text("done")])]) + + def make_host() -> ResponsesHostServer: + agent = Agent(client=_ServiceStorageRecordingClient()) + monkeypatch.setattr(agent, "run", MagicMock(side_effect=provider_run)) + return _make_server( + agent, + history_source="service", + background_source="provider", + options=ResponsesServerOptions(resilient_background=True), + response_store=FileResponseStore(storage_dir=tmp_path), + ) + + original_set = FoundryAgentSessionStore.set + + async def crash_after_head_write(store: FoundryAgentSessionStore, key: str, session: AgentSession) -> None: + await original_set(store, key, session) + if key == conversation_id and session.state.get("_foundry_conversation_committed") == response_id: + raise ResponseExitForRecovery + + server = make_host() + context = ResponseContext(response_id=response_id, conversation_id=conversation_id, mode_flags=MagicMock()) + request = CreateResponse(input="hello", store=True, background=True) + snapshots: list[dict[str, Any]] = [] + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch.object(FoundryAgentSessionStore, "set", new=crash_after_head_write), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + pytest.raises(ResponseExitForRecovery), + ): + async for event in server._handle_response(request, context, asyncio.Event()): + if isinstance(event, ResponseCheckpointEvent): + snapshots.append(copy.deepcopy(dict(event.response))) + + store = AgentSessionStoreProvider().get_store(config=server.config, platform_context=get_request_context()) + head = await store.get(conversation_id) + saved = await store.get(response_id) + assert head is not None and "_foundry_conversation_claim" not in head.state + assert head.state["_foundry_conversation_committed"] == response_id + assert saved is not None and saved.state["_foundry_provider_background"]["completed"] is True + assert snapshots and snapshots[-1]["status"] == "in_progress" + assert run.await_count == 2 + if ownership == "other-claim": + head.state["_foundry_conversation_claim"] = "newer-response" + await store.set(conversation_id, head) + elif ownership == "other-completion": + head.state["_foundry_conversation_committed"] = "newer-response" + await store.set(conversation_id, head) + + original_get = FoundryAgentSessionStore.get + raced = False + + async def claim_after_recovery_read(reader: FoundryAgentSessionStore, key: str) -> AgentSession | None: + nonlocal raced + loaded = await original_get(reader, key) + if key == conversation_id and ownership == "claim-during-recovery" and not raced: + raced = True + newer = await original_get(cast(FoundryAgentSessionStore, store), key) + assert newer is not None + newer.state["_foundry_conversation_claim"] = "newer-response" + await store.set(key, newer) + return loaded + + recovered = ResponseContext(response_id=response_id, conversation_id=conversation_id, mode_flags=MagicMock()) + recovered.is_recovery = True + recovered.persisted_response = cast(ResponseObject, snapshots[-1]) + recovered_host = make_host() + with ( + patch.object(ResponseContext, "get_input_items", new=AsyncMock(return_value=[])), + patch.object(FoundryAgentSessionStore, "get", new=claim_after_recovery_read), + patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), + ): + events = [event async for event in recovered_host._handle_response(request, recovered, asyncio.Event())] + + if ownership in ("other-claim", "other-completion"): + assert "claim is no longer held" in _failure_message(events) + else: + terminal = events[-1] + assert isinstance(terminal, Mapping) and terminal["type"] == "response.completed" + assert [item["type"] for item in terminal["response"]["output"]] == ["message"] + message = terminal["response"]["output"][0] + assert message["type"] == "message" + content = message["content"][0] + assert content["type"] == "output_text" + assert content["text"] == "done" + assert run.await_count == 2, "Recovery must not re-poll or submit provider work after this head commit." + final_head = await store.get(conversation_id) + assert final_head is not None + if ownership in ("other-claim", "claim-during-recovery"): + assert final_head.state["_foundry_conversation_claim"] == "newer-response" + elif ownership == "other-completion": + assert final_head.state["_foundry_conversation_committed"] == "newer-response" + else: + assert final_head.state["_foundry_conversation_committed"] == response_id + assert "private-provider-token" not in str(events) + + @pytest.mark.parametrize("recovery_stage", [None, "before-output", "during-output", "after-checkpoint"]) + async def test_provider_background_polls_keep_options_and_new_token_after_tool( + self, tmp_path: Path, recovery_stage: str | None + ) -> None: executions: list[str] = [] @tool(approval_mode="never_require") @@ -2110,14 +2237,19 @@ def send_email(to: str) -> str: executions.append(to) return "sent" - def openai_response(response_id: str, status: str, output: Any | None = None) -> MagicMock: + def openai_response( + response_id: str, + status: str, + output: Any | None = None, + usage: OpenAIResponseUsage | None = None, + ) -> MagicMock: response = MagicMock() response.id = response_id response.status = status response.conversation = None response.model = "test-model" response.created_at = 1_700_000_000 - response.usage = None + response.usage = usage response.metadata = {} response.incomplete_details = None response.output = [] if output is None else [output] @@ -2137,22 +2269,61 @@ def openai_response(response_id: str, status: str, output: Any | None = None) -> type="message", content=[MagicMock(type="output_text", text="Email sent.", annotations=[], logprobs=None)], ) + partial_message = MagicMock( + type="message", + content=[MagicMock(type="output_text", text="Email", annotations=[], logprobs=None)], + ) create = AsyncMock( side_effect=[ - openai_response("private-first-token", "in_progress"), - openai_response("private-second-token", "in_progress"), + openai_response("private-first-token", "in_progress", partial_message), + openai_response("private-second-token", "in_progress", partial_message), ] ) retrieve = AsyncMock( side_effect=[ - openai_response("private-first-token", "completed", call), - openai_response("private-second-token", "completed", message), + openai_response("private-first-token", "in_progress"), + openai_response( + "private-first-token", + "completed", + call, + OpenAIResponseUsage.model_validate({ + "input_tokens": 5, + "output_tokens": 2, + "total_tokens": 7, + "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, + "output_tokens_details": {"reasoning_tokens": 0}, + }), + ), + openai_response( + "private-second-token", + "completed", + message, + OpenAIResponseUsage.model_validate({ + "input_tokens": 3, + "output_tokens": 4, + "total_tokens": 7, + "input_tokens_details": {"cached_tokens": 0, "cache_write_tokens": 0}, + "output_tokens_details": {"reasoning_tokens": 0}, + }), + ), ] ) client = OpenAIChatClient(model="test-model", api_key="test-key") client.function_invocation_configuration["max_iterations"] = 4 agent = Agent(client=client, tools=[send_email]) - store = SessionStore() + + class CrashBeforeOutputStore(SessionStore): + crashed = False + + async def set(self, session_id: str, session: AgentSession) -> None: + # Exercise the same serialization boundary as durable storage. + await super().set(session_id, AgentSession.from_dict(session.to_dict())) + state = session.state.get("_foundry_provider_background", {}) + if recovery_stage == "before-output" and state.get("outputs") and not self.crashed: + self.crashed = True + raise ResponseExitForRecovery + + store = CrashBeforeOutputStore() server = _make_server( agent, session_store=store, @@ -2163,6 +2334,7 @@ def openai_response(response_id: str, status: str, output: Any | None = None) -> ) context = ResponseContext(response_id="outer-tool-response", mode_flags=MagicMock()) request = CreateResponse(input="email bob", store=True, background=True, temperature=0.42) + snapshots: list[dict[str, Any]] = [] with ( patch.object( @@ -2174,7 +2346,34 @@ def openai_response(response_id: str, status: str, output: Any | None = None) -> patch.object(client.client.responses.with_raw_response, "retrieve", new=retrieve), patch("agent_framework_foundry_hosting._responses.asyncio.sleep", new=AsyncMock()), ): - events = [event async for event in server._handle_response(request, context, asyncio.Event())] + handler = cast( + AsyncGenerator[Any, None], + server._handle_response(request, context, asyncio.Event()), + ) + events: list[Any] = [] + try: + async for event in handler: + if isinstance(event, ResponseCheckpointEvent): + snapshots.append(copy.deepcopy(dict(event.response))) + if recovery_stage == "after-checkpoint": + raise ResponseExitForRecovery + elif ( + recovery_stage == "during-output" + and isinstance(event, Mapping) + and event.get("type") == "response.function_call_arguments.delta" + ): + raise ResponseExitForRecovery + events.append(event) + except ResponseExitForRecovery: + assert recovery_stage is not None + finally: + await handler.aclose() + if recovery_stage is not None: + recovered = ResponseContext(response_id=context.response_id, mode_flags=MagicMock()) + recovered.is_recovery = True + if snapshots: + recovered.persisted_response = cast(ResponseObject, snapshots[-1]) + events = [event async for event in server._handle_response(request, recovered, asyncio.Event())] terminal = events[-1] assert isinstance(terminal, Mapping) @@ -2182,10 +2381,27 @@ def openai_response(response_id: str, status: str, output: Any | None = None) -> pytest.fail(_failure_message(events)) assert terminal["type"] == "response.completed" assert executions == ["bob"] - assert create.await_count == retrieve.await_count == 2 + output = terminal["response"]["output"] + assert [item["type"] for item in output] == ["function_call", "function_call_output", "message"] + assert output[0]["call_id"] == output[1]["call_id"] == "call_1" + assert output[0]["name"] == "send_email" + assert json.loads(output[0]["arguments"]) == {"to": "bob"} + assert output[1]["output"] == "sent" + assert output[2]["content"][0]["text"] == "Email sent." + assert terminal["response"]["usage"]["input_tokens"] == 8 + assert terminal["response"]["usage"]["output_tokens"] == 6 + assert terminal["response"]["usage"]["total_tokens"] == 14 + assert create.await_count == 2 + assert retrieve.await_count == 3 assert [entry.kwargs["background"] for entry in create.await_args_list] == [True, True] assert [entry.kwargs["temperature"] for entry in create.await_args_list] == [0.42, 0.42] + follow_up = create.await_args_list[1].kwargs + assert follow_up["previous_response_id"] == "private-first-token" + assert [ + (item["call_id"], item["output"]) for item in follow_up["input"] if item["type"] == "function_call_output" + ] == [("call_1", "sent")] assert [entry.args[0] for entry in retrieve.await_args_list] == [ + "private-first-token", "private-first-token", "private-second-token", ] @@ -2382,11 +2598,11 @@ def shutdown_before_output(response: AgentResponse, response_id: str) -> list[Ag saved = await store.get("outer-recovered") assert saved is not None - assert saved.state["_foundry_provider_background"] == { - "outer_response_id": "outer-recovered", - "continuation_token": token, - "completed": True, - } + state = saved.state["_foundry_provider_background"] + assert state["outer_response_id"] == "outer-recovered" + assert state["continuation_token"] == token + assert state["completed"] is True + assert len(state["outputs"]) == 1 recovered = ResponseContext(response_id="outer-recovered", mode_flags=MagicMock()) recovered.is_recovery = True @@ -2398,7 +2614,7 @@ def shutdown_before_output(response: AgentResponse, response_id: str) -> list[Ag assert [event.get("type") for event in events if isinstance(event, Mapping)][-1] == "response.completed" assert "private-provider-token" not in str(events) - assert [option.get("background") for option in calls] == [True, True, True] + assert [option.get("background") for option in calls] == [True, True] assert all(option["continuation_token"] == token for option in calls[1:]) @pytest.mark.parametrize("phase", ["submit", "poll", "save"]) diff --git a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md index 9f6bd30feed..6510e1a3e3a 100644 --- a/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md +++ b/python/samples/04-hosting/foundry-hosted-agents/responses/basic/README.md @@ -50,8 +50,11 @@ continuation, a process crash can leave such a response unfinished. The optional [provider_background.py](provider_background.py) opts a storing Responses client into its *separate* background mode: only this mode persists the provider's private token and polls it until completion/recovery. It is incompatible with steering. Polls keep the original model options and `background=True`, including when a tool loop submits -its next leg. A crash between a local tool side effect and saving the next private token can still replay -that tool; use idempotent tools or avoid this mode for side-effecting local tools. The deployed identity +its next leg. Each new private token is saved with the completed tool transcript and usage; recovery replays only +output that was not checkpointed. Saved final output can finish an interrupted outer response without another +provider call, even if that response already committed the conversation head. A crash between a local tool side +effect and saving the next private token can still replay that tool; use idempotent tools or avoid this mode for +side-effecting local tools. The deployed identity needs Foundry User permission on the project for private provider polling. [client.py](client.py) shows stored conversation turns, an unstored request, and background polling. It also needs