From 26c7c81c1bb313f19b04a7d2dda78f6371e3e45c Mon Sep 17 00:00:00 2001 From: Maxim Svistunov Date: Tue, 6 Oct 2026 14:47:02 +0200 Subject: [PATCH 1/4] LCORE-3789: carry the current turn's images into the compacted input In compacted mode the request input is overridden through extra_body with the explicit item list (summaries, recent turns, new query). That list is text only. Image attachments reach the model as ImageUrl parts of the pydantic-ai prompt, and the override replaces the prompt-derived input, so the images never reached the request body. OgxResponsesModel gets a step, run in request() and request_stream(). When the user prompt carries media, it replaces the trailing user message of the override with pydantic-ai's own mapping (the private _map_user_prompt, 2.27.1) of that message's text followed by the media, so the new query has the form it has outside compacted mode. The text comes from the override, not from the prompt: for an empty query agent_prompt_text returns the text of an earlier message. A text-only prompt leaves the override untouched. The 422 guard for such turns is still in place, so nothing changes for users yet. The next commit removes it. Tests: unit tests for the step, and two tests over httpx.MockTransport. One sends the same image prompt with and without the override through agent.run and compares the serialised bodies: no conversation parameter, text-only history unchanged, last input item equal to the non-compacted one, base64 payload present once. The other checks the body of a streamed request. --- src/pydantic_ai_lightspeed/ogx/_model.py | 83 ++++++- .../ogx/test_compacted_input_media.py | 224 ++++++++++++++++++ 2 files changed, 304 insertions(+), 3 deletions(-) create mode 100644 tests/unit/pydantic_ai_lightspeed/ogx/test_compacted_input_media.py diff --git a/src/pydantic_ai_lightspeed/ogx/_model.py b/src/pydantic_ai_lightspeed/ogx/_model.py index d5d21bc24..8cd0dbb97 100644 --- a/src/pydantic_ai_lightspeed/ogx/_model.py +++ b/src/pydantic_ai_lightspeed/ogx/_model.py @@ -20,7 +20,7 @@ from __future__ import annotations as _annotations from collections import defaultdict -from collections.abc import AsyncIterator, Callable +from collections.abc import AsyncIterator, Callable, Sequence from contextlib import asynccontextmanager from typing import Any, Final, Optional, cast @@ -31,7 +31,14 @@ from pydantic_ai import UnexpectedModelBehavior from pydantic_ai._run_context import RunContext from pydantic_ai._utils import PeekableAsyncStream, Unset, number_to_datetime -from pydantic_ai.messages import ModelMessage, ModelResponse +from pydantic_ai.messages import ( + ModelMessage, + ModelRequest, + ModelResponse, + UserContent, + UserPromptPart, + is_multi_modal_content, +) from pydantic_ai.models import ( ModelRequestParameters, StreamedResponse, @@ -82,7 +89,8 @@ def _model_settings_from_responses_params( # the wire ``input`` from the prompt alone. Overriding via extra_body # replaces it with the explicit list, exactly as the non-agent # /v1/responses path sends it. Dropped again on tool-loop - # continuations — see ``_prepare_compacted_input``. + # continuations — see ``_prepare_compacted_input``. The media of the + # current prompt is merged back in by ``_carry_prompt_media``. extra_body["input"] = payload["input"] settings_dict: dict[str, Any] = {} if extra_body: @@ -105,6 +113,28 @@ def _model_settings_from_responses_params( return cast("OpenAIResponsesModelSettings", settings_dict) +def _prompt_media(messages: Sequence[ModelMessage]) -> list[UserContent]: + """Return the media items, such as images, of the user prompt being sent. + + Parameters: + messages: Model messages for the request. + + Returns: + The media items of the last user prompt part, in prompt order. Empty + when there is no user prompt or it is text only. + """ + prompts = [ + part + for message in messages + if isinstance(message, ModelRequest) + for part in message.parts + if isinstance(part, UserPromptPart) + ] + if not prompts or isinstance(prompts[-1].content, str): + return [] + return [item for item in prompts[-1].content if is_multi_modal_content(item)] + + class _FilteredResponseStream: """Wraps an OpenAI AsyncStream to reorder spurious events from OGX. @@ -348,6 +378,7 @@ async def request( # pylint: disable=unused-argument messages, model_settings ) model_settings = self._prepare_compacted_input(messages, model_settings) + model_settings = await self._carry_prompt_media(messages, model_settings) return await super().request(messages, model_settings, model_request_parameters) def _prepare_conversation_continuation( @@ -419,6 +450,51 @@ def _prepare_compacted_input( new_settings["extra_body"] = new_extra_body return cast("ModelSettings", new_settings) + async def _carry_prompt_media( + self, + messages: list[ModelMessage], + model_settings: Optional[ModelSettings], + ) -> Optional[ModelSettings]: + """Give the compacted ``input`` override the media of the current prompt. + + The explicit item list is text only, while image attachments reach the + model as ``ImageUrl`` parts of the user prompt, so they would be lost + when the override replaces the prompt-derived ``input`` (LCORE-3789). + When the prompt carries media, the trailing user message of the + override (the new query) is replaced by pydantic-ai's own mapping of + its text followed by that media, which is the form the query has + outside compacted mode. The text stays the one the override carries: + for an empty query the prompt holds the text of an earlier message + (see ``agent_prompt_text``). A text-only prompt leaves the override as + it was built. + + Parameters: + messages: Model messages for the request. + model_settings: Model settings, possibly carrying the override. + + Returns: + The settings unchanged, or a copy whose override ends with the + new query and its media. + """ + if not model_settings or not isinstance(model_settings, dict): + return model_settings + extra_body = model_settings.get("extra_body") + if not isinstance(extra_body, dict) or "input" not in extra_body: + return model_settings + content = _prompt_media(messages) + if not content: + return model_settings + + history = list(extra_body["input"]) + query = history[-1] if history else {} + if query.get("role") == "user" and isinstance(query.get("content"), str): + history.pop() + content = [query["content"], *content] + new_query = await self._map_user_prompt(UserPromptPart(content=content)) + new_settings = dict(model_settings) + new_settings["extra_body"] = {**extra_body, "input": [*history, new_query]} + return cast("ModelSettings", new_settings) + @asynccontextmanager async def request_stream( # pylint: disable=unused-argument self, @@ -446,6 +522,7 @@ async def request_stream( # pylint: disable=unused-argument messages, model_settings ) model_settings = self._prepare_compacted_input(messages, model_settings) + model_settings = await self._carry_prompt_media(messages, model_settings) model_settings_cast = cast("OpenAIResponsesModelSettings", model_settings or {}) response = await self._responses_create( diff --git a/tests/unit/pydantic_ai_lightspeed/ogx/test_compacted_input_media.py b/tests/unit/pydantic_ai_lightspeed/ogx/test_compacted_input_media.py new file mode 100644 index 000000000..9cfcd9113 --- /dev/null +++ b/tests/unit/pydantic_ai_lightspeed/ogx/test_compacted_input_media.py @@ -0,0 +1,224 @@ +"""Unit tests for carrying the current prompt's media into the compacted input.""" + +# pylint: disable=protected-access + +import json + +import httpx +import pytest +from ogx_api.openai_responses import OpenAIResponseMessage +from pydantic_ai import Agent, ModelMessage +from pydantic_ai.messages import ImageUrl, ModelRequest, UserPromptPart +from pydantic_ai.models import ModelRequestParameters +from pydantic_ai.models.openai import OpenAIResponsesModelSettings +from pydantic_ai.settings import ModelSettings + +from models.common.responses.responses_api_params import ResponsesApiParams +from pydantic_ai_lightspeed.ogx._model import ( + OgxResponsesModel, + _model_settings_from_responses_params, +) +from pydantic_ai_lightspeed.ogx._provider import OgxProvider + +IMAGE_B64 = "aW1hZ2UtYnl0ZXM=" +IMAGE_URL = f"data:image/png;base64,{IMAGE_B64}" +IMAGE = ImageUrl(url=IMAGE_URL, media_type="image/png") +IMAGE_PART = {"image_url": IMAGE_URL, "type": "input_image", "detail": "auto"} +IMAGE_PROMPT = ["new question", IMAGE] +IMAGE_MESSAGES: list[ModelMessage] = [ + ModelRequest(parts=[UserPromptPart(content=IMAGE_PROMPT)]) +] +# The explicit input before the new query: a summary and one recent turn. +HISTORY = [ + { + "role": "user", + "content": "Summary of earlier conversation:\nS1", + "type": "message", + }, + {"role": "user", "content": "recent q", "type": "message"}, + {"role": "assistant", "content": "recent a", "type": "message"}, +] +ANSWER = { + "id": "resp-1", + "object": "response", + "created_at": 0, + "model": "test-model", + "status": "completed", + "output": [ + { + "id": "msg-1", + "type": "message", + "role": "assistant", + "status": "completed", + "content": [{"type": "output_text", "text": "ok", "annotations": []}], + } + ], + "parallel_tool_calls": False, + "tool_choice": "auto", + "tools": [], +} + + +def _make_settings(compacted: bool) -> OpenAIResponsesModelSettings: + """Build model settings for the new query, in compacted mode or outside it.""" + items = [ + OpenAIResponseMessage(**item) + for item in [*HISTORY, {"role": "user", "content": "new question"}] + ] + params = ResponsesApiParams( + input=items if compacted else "new question", + omit_conversation=compacted, + model="provider/model", + conversation="conv-1", + max_infer_iters=3, + store=True, + stream=False, + ) + return _model_settings_from_responses_params(params) + + +@pytest.fixture(name="bodies") +def bodies_fixture() -> list[bytes]: + """Collect the request bodies that reach the transport.""" + return [] + + +@pytest.fixture(name="provider") +def provider_fixture(bodies: list[bytes]) -> OgxProvider: + """Create a provider whose transport records requests and answers them.""" + + def handler(request: httpx.Request) -> httpx.Response: + bodies.append(request.content) + if not json.loads(request.content)["stream"]: + return httpx.Response(200, json=ANSWER) + event = {"type": "response.created", "sequence_number": 0, "response": ANSWER} + return httpx.Response( + 200, + headers={"content-type": "text/event-stream"}, + content=f"event: response.created\ndata: {json.dumps(event)}\n\n", + ) + + return OgxProvider( + base_url="http://ogx.test/v1", + http_client=httpx.AsyncClient(transport=httpx.MockTransport(handler)), + ) + + +@pytest.mark.asyncio +async def test_settings_without_input_override_are_unchanged( + provider: OgxProvider, +) -> None: + """Test that outside compacted mode an image prompt changes nothing.""" + model = OgxResponsesModel("test-model", provider=provider) + settings = _make_settings(compacted=False) + assert await model._carry_prompt_media(IMAGE_MESSAGES, settings) is settings + + +@pytest.mark.asyncio +async def test_text_prompt_keeps_the_override(provider: OgxProvider) -> None: + """Test that a text-only prompt leaves the explicit input as it was built.""" + model = OgxResponsesModel("test-model", provider=provider) + messages: list[ModelMessage] = [ + ModelRequest(parts=[UserPromptPart(content="new question")]) + ] + settings = _make_settings(compacted=True) + assert await model._carry_prompt_media(messages, settings) is settings + + +@pytest.mark.asyncio +async def test_image_prompt_adds_the_image_to_the_new_query( + provider: OgxProvider, +) -> None: + """Test that the trailing user message is sent with the prompt's image.""" + model = OgxResponsesModel("test-model", provider=provider) + settings = _make_settings(compacted=True) + + result = await model._carry_prompt_media(IMAGE_MESSAGES, settings) + + new_query = { + "role": "user", + "content": [{"text": "new question", "type": "input_text"}, IMAGE_PART], + } + assert result == { + **settings, + "extra_body": {**settings["extra_body"], "input": [*HISTORY, new_query]}, + } + # the settings passed in are not modified + assert settings["extra_body"]["input"][-1]["content"] == "new question" + + +@pytest.mark.asyncio +async def test_query_text_comes_from_the_explicit_input(provider: OgxProvider) -> None: + """Test that an empty query stays empty next to its image. + + For an empty query the agent prompt falls back to the text of an earlier + message. That text must not be sent again as the user's words. + """ + model = OgxResponsesModel("test-model", provider=provider) + messages: list[ModelMessage] = [ + ModelRequest(parts=[UserPromptPart(content=["recent a", IMAGE])]) + ] + empty_query = {"role": "user", "content": "", "type": "message"} + settings: ModelSettings = {"extra_body": {"input": [*HISTORY, empty_query]}} + + result = await model._carry_prompt_media(messages, settings) + + assert result is not None + assert result["extra_body"]["input"] == [ + *HISTORY, + {"role": "user", "content": [{"text": "", "type": "input_text"}, IMAGE_PART]}, + ] + + +@pytest.mark.asyncio +async def test_trailing_item_of_another_role_is_kept(provider: OgxProvider) -> None: + """Test that the image is appended when no user message ends the input.""" + model = OgxResponsesModel("test-model", provider=provider) + settings: ModelSettings = {"extra_body": {"input": HISTORY}} + + result = await model._carry_prompt_media(IMAGE_MESSAGES, settings) + + assert result is not None + assert result["extra_body"]["input"] == [ + *HISTORY, + {"role": "user", "content": [IMAGE_PART]}, + ] + + +@pytest.mark.asyncio +async def test_request_body_carries_the_image_as_outside_compacted_mode( + provider: OgxProvider, bodies: list[bytes] +) -> None: + """Test the serialised compacted request against the non-compacted one. + + The same prompt is sent once without and once with the compacted input + override, through the real OpenAI client. The new query must go out + identically both times, in compacted mode after a text-only history. + """ + for compacted in (False, True): + model = OgxResponsesModel( + "test-model", provider=provider, settings=_make_settings(compacted) + ) + await Agent(model, defer_model_check=True).run(IMAGE_PROMPT) + + plain_body, compacted_body = (json.loads(body) for body in bodies) + assert plain_body["conversation"] == "conv-1" + assert "conversation" not in compacted_body + assert compacted_body["input"][:-1] == HISTORY + assert compacted_body["input"][-1] == plain_body["input"][-1] + assert bodies[1].count(IMAGE_B64.encode()) == 1 + + +@pytest.mark.asyncio +async def test_streamed_request_body_carries_the_image( + provider: OgxProvider, bodies: list[bytes] +) -> None: + """Test that a streamed compacted request is sent with the image too.""" + settings = _make_settings(compacted=True) + model = OgxResponsesModel("test-model", provider=provider) + + async with model.request_stream(IMAGE_MESSAGES, settings, ModelRequestParameters()): + pass + + new_query = json.loads(bodies[0])["input"][-1] + assert new_query["content"][1]["image_url"] == IMAGE_URL From 694e1f0aa7a585df793c750548dc6aace99f04a5 Mon Sep 17 00:00:00 2001 From: Maxim Svistunov Date: Tue, 6 Oct 2026 15:18:34 +0200 Subject: [PATCH 2/4] LCORE-3789: stop rejecting compacted turns that carry images LCORE-3582 added reject_image_attachments_in_compacted_mode, which rejected a compacted turn with image attachments with a 422 error (the HTTP status on /v1/query; on /v1/streaming_query it was raised after the stream had started), because the explicit input could not carry the images and the model would have answered without seeing them. The previous commit sends the images of the current turn in compacted mode, so the guard has nothing left to protect against. This removes the guard, its two call sites in the /v1/query and /v1/streaming_query agent helpers, the imports that only it used, and its tests. The agent_prompt_text docstring now says where the images join the request. Tests: the two agent-helper tests that check the multimodal prompt for image attachments are parametrized over compacted mode. Before the removal the compacted cases failed with the 422. --- src/utils/agents/query.py | 4 -- src/utils/agents/streaming.py | 2 - src/utils/conversation_compaction.py | 48 ++----------------- tests/unit/utils/agents/test_query.py | 18 ++++++- tests/unit/utils/agents/test_streaming.py | 18 ++++++- .../utils/test_conversation_compaction.py | 33 ------------- 6 files changed, 37 insertions(+), 86 deletions(-) diff --git a/src/utils/agents/query.py b/src/utils/agents/query.py index 8b6db71c3..6cebd7d8b 100644 --- a/src/utils/agents/query.py +++ b/src/utils/agents/query.py @@ -38,7 +38,6 @@ ) from utils.conversation_compaction import ( agent_prompt_text, - reject_image_attachments_in_compacted_mode, store_compacted_turn, ) from utils.otel_tracing import ( @@ -282,9 +281,6 @@ async def retrieve_agent_response( no_tools=no_tools, ) logger.debug("Starting agent non-streaming response processing") - reject_image_attachments_in_compacted_mode( - responses_params, image_attachments - ) if image_attachments: prompt = build_multimodal_input( agent_prompt_text(responses_params), diff --git a/src/utils/agents/streaming.py b/src/utils/agents/streaming.py index 49637c92d..c937775f8 100644 --- a/src/utils/agents/streaming.py +++ b/src/utils/agents/streaming.py @@ -63,7 +63,6 @@ ) from utils.conversation_compaction import ( agent_prompt_text, - reject_image_attachments_in_compacted_mode, store_compacted_turn, ) from utils.otel_tracing import ( @@ -416,7 +415,6 @@ async def agent_response_generator( rag_id_mapping=context.rag_id_mapping, turn_summary=turn_summary, ) - reject_image_attachments_in_compacted_mode(responses_params, image_attachments) if image_attachments: prompt = build_multimodal_input( agent_prompt_text(responses_params), diff --git a/src/utils/conversation_compaction.py b/src/utils/conversation_compaction.py index 1b9f7c480..7ec21ab33 100644 --- a/src/utils/conversation_compaction.py +++ b/src/utils/conversation_compaction.py @@ -49,7 +49,6 @@ from dataclasses import dataclass from typing import Any, Optional, cast -from fastapi import HTTPException from ogx_api.openai_responses import OpenAIResponseMessage from ogx_client import AsyncOgxClient @@ -57,7 +56,6 @@ from cache.cache_error import CacheError from configuration import configuration from log import get_logger -from models.api.responses.error import UnprocessableEntityResponse from models.common.responses.responses_api_params import ResponsesApiParams from models.common.responses.types import ResponseInput from models.common.turn_summary import ContextStatus @@ -351,7 +349,9 @@ def agent_prompt_text(params: ResponsesApiParams) -> str: the new user query is the trailing message item. The agent pipeline still needs a plain string prompt (capabilities and multimodal input operate on it); the full explicit list reaches the request body separately via the - ``extra_body`` input override (LCORE-3582). + ``extra_body`` input override (LCORE-3582). Image attachments added to + this prompt are merged into that override by ``OgxResponsesModel`` + (LCORE-3789). Args: params: Prepared (possibly compaction-rewritten) request parameters. @@ -380,48 +380,6 @@ def agent_prompt_text(params: ResponsesApiParams) -> str: return "" -def reject_image_attachments_in_compacted_mode( - params: ResponsesApiParams, - image_attachments: Optional[Sequence[Any]], -) -> None: - """Reject a compacted turn that carries image attachments (LCORE-3582). - - In compacted mode the wire ``input`` is overridden with the explicit item - list built by :func:`_build_explicit_input`, which is text-only: image - attachments are converted to pydantic-ai ``ImageUrl`` parts on the prompt, - and the override replaces the prompt-derived input wholesale, so those - parts never reach the request body. - - Answering anyway would return a confident response that never saw the - image, with nothing to tell the caller their attachment was ignored. Fail - explicitly instead until the explicit input can carry ``input_image`` - content parts of its own (LCORE-3789). - - Args: - params: Prepared (possibly compaction-rewritten) request parameters. - image_attachments: Image attachments for this turn, if any. - - Raises: - HTTPException: 422 when the turn is compacted and carries images. - """ - if not image_attachments or not params.omit_conversation: - return - logger.warning( - "Rejecting compacted turn with %d image attachment(s): the explicit " - "input override cannot carry image content parts (LCORE-3789)", - len(image_attachments), - ) - response = UnprocessableEntityResponse( - response="Image attachments are not supported on this conversation", - cause=( - "This conversation has been compacted to fit the model's context " - "window, and compacted turns cannot carry image attachments yet. " - "Send the image in a new conversation, or retry without it." - ), - ) - raise HTTPException(**response.model_dump()) - - def _query_input_message(original_input: ResponseInput) -> list[Any]: """Render the new user query as explicit input items. diff --git a/tests/unit/utils/agents/test_query.py b/tests/unit/utils/agents/test_query.py index 038609ec5..b5a6eb406 100644 --- a/tests/unit/utils/agents/test_query.py +++ b/tests/unit/utils/agents/test_query.py @@ -486,14 +486,20 @@ async def test_unclassified_inference_error_is_logged_with_traceback( assert matching[0].exc_info is not None, "traceback was not attached" @pytest.mark.asyncio + @pytest.mark.parametrize("compacted", [False, True]) async def test_success_with_image_attachments_sends_multimodal_prompt( self, mocker: MockerFixture, make_agent_run_result: Callable[..., Any], make_responses_params: Callable[..., ResponsesApiParams], patch_recording_metrics: None, + compacted: bool, ) -> None: - """Test that image attachments produce a multimodal prompt.""" + """Test that image attachments produce a multimodal prompt. + + A compacted turn is no exception: it runs with the new query and the + image, rather than being rejected (LCORE-3789). + """ run_result = make_agent_run_result( content="I see a screenshot.", response_id="resp-image", @@ -512,6 +518,16 @@ async def test_success_with_image_attachments_sends_multimodal_prompt( content_type="image/jpeg", ) params = make_responses_params(input_text="describe this") + if compacted: + explicit = [ + OpenAIResponseMessage( + role="user", content="Summary of earlier conversation:\nS1" + ), + OpenAIResponseMessage(role="user", content="describe this"), + ] + params = params.model_copy( + update={"input": explicit, "omit_conversation": True} + ) summary = await retrieve_agent_response( client=mocker.AsyncMock(), diff --git a/tests/unit/utils/agents/test_streaming.py b/tests/unit/utils/agents/test_streaming.py index f614301df..bedde8394 100644 --- a/tests/unit/utils/agents/test_streaming.py +++ b/tests/unit/utils/agents/test_streaming.py @@ -1248,6 +1248,7 @@ async def test_compacted_input_streams_with_prompt_text( assert mock_agent.run_stream_events.call_args[0][0] == "new question" @pytest.mark.asyncio + @pytest.mark.parametrize("compacted", [False, True]) async def test_streams_with_image_attachments_passes_multimodal_prompt( self, mocker: MockerFixture, @@ -1255,8 +1256,13 @@ async def test_streams_with_image_attachments_passes_multimodal_prompt( make_responses_params: Callable[..., ResponsesApiParams], make_agent_run_result: Callable[..., Any], patch_recording_metrics: None, + compacted: bool, ) -> None: - """Test that image attachments produce a multimodal prompt for streaming.""" + """Test that image attachments produce a multimodal prompt for streaming. + + A compacted turn is no exception: it streams with the new query and + the image, rather than being rejected (LCORE-3789). + """ context = make_generator_context() turn_summary = TurnSummary() run_result = make_agent_run_result( @@ -1290,6 +1296,16 @@ async def test_streams_with_image_attachments_passes_multimodal_prompt( content_type="image/jpeg", ) params = make_responses_params(input_text="describe this") + if compacted: + explicit = [ + OpenAIResponseMessage( + role="user", content="Summary of earlier conversation:\nS1" + ), + OpenAIResponseMessage(role="user", content="describe this"), + ] + params = params.model_copy( + update={"input": explicit, "omit_conversation": True} + ) _ = [ event diff --git a/tests/unit/utils/test_conversation_compaction.py b/tests/unit/utils/test_conversation_compaction.py index b5bea40f7..b6c1d720a 100644 --- a/tests/unit/utils/test_conversation_compaction.py +++ b/tests/unit/utils/test_conversation_compaction.py @@ -8,7 +8,6 @@ from typing import Any, Optional, cast import pytest -from fastapi import HTTPException from ogx_api.openai_responses import ( OpenAIResponseInputMessageContentText, OpenAIResponseMessage, @@ -227,38 +226,6 @@ def test_agent_prompt_text_explicit_list_returns_last_message_text() -> None: assert cc.agent_prompt_text(compacted) == "new question" -@pytest.mark.parametrize( - ("omit_conversation", "has_images"), - [(True, False), (False, True), (False, False)], -) -def test_reject_image_attachments_allows_supported_combinations( - omit_conversation: bool, has_images: bool -) -> None: - """Only a compacted turn carrying images is rejected; the rest pass through.""" - params = _params().model_copy(update={"omit_conversation": omit_conversation}) - attachments = [object()] if has_images else None - cc.reject_image_attachments_in_compacted_mode(params, attachments) - - -def test_reject_image_attachments_in_compacted_mode_raises_422() -> None: - """A compacted turn with images fails explicitly instead of ignoring them. - - The explicit-input override replaces the prompt-derived request input, and - the explicit list is text-only, so the images would never reach the model. - Answering anyway would return a confident response that never saw the - image, with nothing telling the caller it was dropped (LCORE-3582). - """ - params = _params().model_copy(update={"omit_conversation": True}) - - with pytest.raises(HTTPException) as exc_info: - cc.reject_image_attachments_in_compacted_mode(params, [object()]) - - assert exc_info.value.status_code == 422 - detail = exc_info.value.detail - assert "Image attachments are not supported" in detail["response"] - assert "compacted" in detail["cause"] - - def test_agent_prompt_text_empty_list_returns_empty() -> None: """An empty explicit list yields an empty prompt rather than crashing.""" params = _params().model_copy(update={"input": [], "omit_conversation": True}) From 26ae711c30e6cc40184d7ce521b2ed62e24a7258 Mon Sep 17 00:00:00 2001 From: Maxim Svistunov Date: Tue, 6 Oct 2026 15:36:51 +0200 Subject: [PATCH 3/4] LCORE-3789: pin that earlier images are not replayed, document images Two properties of compacted mode held without a test, and the ticket asks that they stay true now that the current turn's images are sent: - An image from an earlier turn is not sent again. Recent turns are rendered from their text, so every explicit input item is plain text. - An image part adds nothing to the token estimate that triggers compaction. Only message text is counted, so a base64 payload never reaches the tokenizer. This adds one unit test for each. Each test fails when the property is broken on purpose: passing the stored content through in _verbatim_input_message, and counting image_url in extract_message_text. The design document and the user guide now describe image attachments in a compacted conversation: the current turn's images are sent, earlier images are not sent again, images are not counted in the estimate, and the stored turn holds text only. The design document also gets a changelog entry for the removed 422. --- .../conversation-compaction.md | 9 ++++ docs/user_doc/conversation_compaction.md | 9 ++++ .../utils/test_conversation_compaction.py | 49 +++++++++++++++++++ 3 files changed, 67 insertions(+) diff --git a/docs/design/conversation-compaction/conversation-compaction.md b/docs/design/conversation-compaction/conversation-compaction.md index 9abe1dbc0..cafcc7ff3 100644 --- a/docs/design/conversation-compaction/conversation-compaction.md +++ b/docs/design/conversation-compaction/conversation-compaction.md @@ -233,6 +233,8 @@ After compaction, lightspeed-stack writes the summary as a marked conversation i When building context for a compacted conversation, lightspeed-stack fetches the conversation items, reads the active summaries from the summary cache (LCORE-1571) — falling back to the marker texts when no persisting cache is configured — takes the items after the last marker as the recent verbatim buffer, and sends `[summaries] + [recent items] + [new query]` as **explicit input**, **without** the `conversation` parameter. This is necessary because OGX reloads the *full* stored message history whenever the `conversation` parameter is set — there is no marker-based selection hook (verified empirically; see the Changelog). Each completed turn is then appended back to the conversation items by lightspeed-stack, since OGX no longer auto-stores it. +Image attachments (LCORE-3789): on `/v1/query` and `/v1/streaming_query` an image attachment is not part of the text input. It reaches the model as a part of the pydantic-ai user prompt, and the explicit input is text only. Whenever the prompt carries an image, `OgxResponsesModel` therefore replaces the trailing message of the explicit input, the new query, with pydantic-ai's mapping of the same text followed by the images of the prompt, so the new query is sent in the form it has outside compacted mode. Only the images of the current turn are sent. Recent turns are rendered from their text and summaries are text, so an image from an earlier turn is not sent again once the conversation is compacted. Images are not counted in the token estimate that triggers compaction, which reads message text only. The turn that lightspeed-stack appends to the conversation holds the text input, without the image. + This preserves a single continuous conversation identity. The `conversation_id` never changes and the user sees one conversation in the UI. OGX stores the full history, including the summary marker items, and returns it through its Conversations API. lightspeed-stack's own `GET /v1/conversations/{conversation_id}` leaves the marker items out of the chat history it returns (LCORE-3909): they are stored as user messages, but the user never sent them. Stored conversations are not migrated: the markers must stay in storage as the fallback source of truth, so they are filtered when the conversation is read. ## API response changes @@ -446,6 +448,13 @@ never holds a marker, and is unchanged. Earlier revisions of this document said that the Conversations API returns the marker items; that holds for the OGX Conversations API only. +**2026-10-06 — Image attachments in compacted mode (LCORE-3789).** +A compacted turn on `/v1/query` or `/v1/streaming_query` that carried an image +attachment was rejected by a guard (LCORE-3582; HTTP 422 on `/v1/query`), +because the explicit input is text only. `OgxResponsesModel` now merges the +images of the current turn into the explicit input, and the guard is removed. +Details are in "Changed request flow after compaction". + # Appendix A: PoC Evidence A proof-of-concept was built and tested. diff --git a/docs/user_doc/conversation_compaction.md b/docs/user_doc/conversation_compaction.md index e7f5221fa..ef59fc514 100644 --- a/docs/user_doc/conversation_compaction.md +++ b/docs/user_doc/conversation_compaction.md @@ -128,6 +128,15 @@ Compaction triggers when **all** of the following are true: 3. The estimated token count of the conversation exceeds `threshold_ratio × context_window` 4. The estimated token count is at least `token_floor` +### Image attachments + +A query with image attachments on `/v1/query` or `/v1/streaming_query` is answered in a compacted conversation as well: + +- The images attached to the current query are sent to the LLM, together with the summary and the recent turns. +- Images from earlier turns are not sent again once the conversation is compacted. The LLM gets the text of those turns, or their summary. +- Images are not counted in the token estimate that triggers compaction. Only text is counted. +- The turn that the service stores in the conversation holds the text of the query, without the image. + ### Degrading guard The `buffer_turns` setting specifies a target number of recent turns to preserve. If the selected buffer turns exceed the available budget (`buffer_max_ratio × context_window`), the system reduces the buffer by one turn pair at a time until the budget fits. In extreme cases, the buffer can shrink to zero turns, meaning only the summary and the current query are sent to the LLM. diff --git a/tests/unit/utils/test_conversation_compaction.py b/tests/unit/utils/test_conversation_compaction.py index b6c1d720a..9fedcd8bb 100644 --- a/tests/unit/utils/test_conversation_compaction.py +++ b/tests/unit/utils/test_conversation_compaction.py @@ -9,6 +9,7 @@ import pytest from ogx_api.openai_responses import ( + OpenAIResponseInputMessageContentImage, OpenAIResponseInputMessageContentText, OpenAIResponseMessage, ) @@ -28,6 +29,19 @@ def _msg(role: str, text: str) -> OpenAIResponseMessage: return OpenAIResponseMessage(role=cast("Any", role), content=text) +def _msg_with_image(text: str) -> OpenAIResponseMessage: + """Build a user message item that carries an image next to its text.""" + return OpenAIResponseMessage( + role="user", + content=[ + OpenAIResponseInputMessageContentText(text=text), + OpenAIResponseInputMessageContentImage( + image_url="data:image/png;base64,AAAA" + ), + ], + ) + + def _marker(text: str) -> OpenAIResponseMessage: """Build a typed summary marker message item for tests (no covers count).""" return OpenAIResponseMessage(role="user", content=f"{cc.MARKER_SENTINEL} {text}") @@ -192,6 +206,27 @@ def test_build_explicit_input_shape() -> None: assert built[2].role == "assistant" +def test_build_explicit_input_does_not_replay_earlier_images() -> None: + """An image sent in an earlier turn is not sent again in compacted mode. + + Recent turns are rendered from their text, so the image part of a stored + message stays behind and every explicit item is plain text (LCORE-3789). + """ + built = cc._build_explicit_input( + summaries=[], + recent_items=[ + _msg_with_image("what is in this picture?"), + _msg("assistant", "a cat"), + ], + original_input="and what colour is it?", + ) + assert [message.content for message in built] == [ + "what is in this picture?", + "a cat", + "and what colour is it?", + ] + + def test_should_compact() -> None: """Trigger requires exceeding both the ratio threshold and the token floor.""" cfg = _compaction(threshold_ratio=0.7, token_floor=100) @@ -676,6 +711,20 @@ def test_estimate_response_input_tokens_counts_list_form() -> None: assert list_tokens > 10 +def test_estimate_does_not_count_image_parts() -> None: + """An image part adds nothing to the estimate that triggers compaction. + + Only the text of a message is counted, so a base64 payload is never run + through the tokenizer (LCORE-3789). + """ + text = "what is in this picture?" + with_image = cc._estimate_response_input_tokens( + cast("Any", [_msg_with_image(text)]), cc.DEFAULT_ENCODING_NAME + ) + text_only = cc._estimate_response_input_tokens(text, cc.DEFAULT_ENCODING_NAME) + assert with_image == text_only > 0 + + # --- per-conversation lock (R11): ref-counted cleanup --- From 8aafcb4bc0db8a20fac057e9775e088a19a05e19 Mon Sep 17 00:00:00 2001 From: Maxim Svistunov Date: Tue, 6 Oct 2026 19:11:06 +0200 Subject: [PATCH 4/4] LCORE-3789: keep an empty query empty in the agent prompt A request may carry an image and an empty query. In a compacted conversation agent_prompt_text skipped the empty trailing message of the explicit input and returned the text of the message before it, which is the last assistant answer or the summary. The first model request of the run was not affected, because its query text comes from the explicit input. Everything else that reads the prompt was: a shield got the earlier text to judge, and the second model request of a client-side tool loop sent the earlier text as the user's words next to the image. agent_prompt_text now returns the text of the trailing message item even when it is empty, which is what the prompt holds outside compacted mode. The warning stays for an explicit list without a message item. The _carry_prompt_media docstring no longer names the old fallback as the reason for taking the text from the explicit input. Tests: one unit test builds the explicit input for an empty query and expects an empty prompt. Before the change it got the text of the earlier assistant message. --- src/pydantic_ai_lightspeed/ogx/_model.py | 7 +++---- src/utils/conversation_compaction.py | 7 ++++--- .../ogx/test_compacted_input_media.py | 6 +----- tests/unit/utils/test_conversation_compaction.py | 11 +++++++++++ 4 files changed, 19 insertions(+), 12 deletions(-) diff --git a/src/pydantic_ai_lightspeed/ogx/_model.py b/src/pydantic_ai_lightspeed/ogx/_model.py index 8cd0dbb97..dcfbc002c 100644 --- a/src/pydantic_ai_lightspeed/ogx/_model.py +++ b/src/pydantic_ai_lightspeed/ogx/_model.py @@ -463,10 +463,9 @@ async def _carry_prompt_media( When the prompt carries media, the trailing user message of the override (the new query) is replaced by pydantic-ai's own mapping of its text followed by that media, which is the form the query has - outside compacted mode. The text stays the one the override carries: - for an empty query the prompt holds the text of an earlier message - (see ``agent_prompt_text``). A text-only prompt leaves the override as - it was built. + outside compacted mode. The text stays the one the override carries, + so the explicit list is the only source of the query text. A text-only + prompt leaves the override as it was built. Parameters: messages: Model messages for the request. diff --git a/src/utils/conversation_compaction.py b/src/utils/conversation_compaction.py index 7ec21ab33..0200b58a3 100644 --- a/src/utils/conversation_compaction.py +++ b/src/utils/conversation_compaction.py @@ -365,9 +365,10 @@ def agent_prompt_text(params: ResponsesApiParams) -> str: return params.input for item in reversed(list(params.input)): if is_message_item(item): - text = extract_message_text(item) - if text: - return text + # The trailing message is the new query. An empty one (an image + # sent without text) stays empty rather than borrowing the text + # of an earlier message. + return extract_message_text(item) # The wire input still carries the explicit list via the extra_body # override, so the request itself is well-formed — but capabilities and # multimodal construction operate on this prompt, and an explicit input diff --git a/tests/unit/pydantic_ai_lightspeed/ogx/test_compacted_input_media.py b/tests/unit/pydantic_ai_lightspeed/ogx/test_compacted_input_media.py index 9cfcd9113..daa2380cd 100644 --- a/tests/unit/pydantic_ai_lightspeed/ogx/test_compacted_input_media.py +++ b/tests/unit/pydantic_ai_lightspeed/ogx/test_compacted_input_media.py @@ -149,11 +149,7 @@ async def test_image_prompt_adds_the_image_to_the_new_query( @pytest.mark.asyncio async def test_query_text_comes_from_the_explicit_input(provider: OgxProvider) -> None: - """Test that an empty query stays empty next to its image. - - For an empty query the agent prompt falls back to the text of an earlier - message. That text must not be sent again as the user's words. - """ + """Test that the query text comes from the explicit input, not from the prompt.""" model = OgxResponsesModel("test-model", provider=provider) messages: list[ModelMessage] = [ ModelRequest(parts=[UserPromptPart(content=["recent a", IMAGE])]) diff --git a/tests/unit/utils/test_conversation_compaction.py b/tests/unit/utils/test_conversation_compaction.py index 9fedcd8bb..9291fddee 100644 --- a/tests/unit/utils/test_conversation_compaction.py +++ b/tests/unit/utils/test_conversation_compaction.py @@ -261,6 +261,17 @@ def test_agent_prompt_text_explicit_list_returns_last_message_text() -> None: assert cc.agent_prompt_text(compacted) == "new question" +def test_agent_prompt_text_empty_query_stays_empty() -> None: + """An empty query (an image sent without text) does not borrow earlier text.""" + explicit = cc._build_explicit_input( + ["earlier summary"], [_msg("assistant", "prior answer")], "" + ) + compacted = _params().model_copy( + update={"input": explicit, "omit_conversation": True} + ) + assert cc.agent_prompt_text(compacted) == "" + + def test_agent_prompt_text_empty_list_returns_empty() -> None: """An empty explicit list yields an empty prompt rather than crashing.""" params = _params().model_copy(update={"input": [], "omit_conversation": True})