From 0feed633bd49fd79d5648e0d8040e7d3c0b3d8af Mon Sep 17 00:00:00 2001 From: Zexin Liang Date: Sun, 4 Oct 2026 10:38:32 +0800 Subject: [PATCH] fix: route pasted images through put_task instead of patching backend.ask Patching backend.ask raised AttributeError on MixinSession (read-only .ask). Images now travel as task data: put_task(images=...) -> build_image_content -> agent_runner_loop(initial_user_content=...), with size/count caps so oversized files fall back to path references. Also keep non-text blocks (images) when filtering whitespace-only content in NativeToolClient.chat. Closes #824 --- agentmain.py | 32 +++++++++++++- frontends/desktop_bridge.py | 41 ------------------ frontends/tests/test_bridge_submit.py | 62 +++++++++++++++++++++++---- llmcore.py | 4 +- 4 files changed, 87 insertions(+), 52 deletions(-) diff --git a/agentmain.py b/agentmain.py index d8cab8cf0..34c87d744 100644 --- a/agentmain.py +++ b/agentmain.py @@ -38,6 +38,32 @@ def get_system_prompt(): prompt += get_global_memory() return prompt +_VISION_MIMES = {'image/png', 'image/jpeg', 'image/gif', 'image/webp'} +_MAX_IMAGE_BYTES = 10 * 1024 * 1024 +_MAX_IMAGE_TOTAL_BYTES = 20 * 1024 * 1024 +_MAX_IMAGE_COUNT = 8 + +def build_image_content(query, image_paths): + """Build first-turn multimodal content without allowing oversized requests.""" + import base64, mimetypes + blocks = [{"type": "text", "text": query}] + total_bytes = image_count = 0 + for p in image_paths: + try: + mime = mimetypes.guess_type(p)[0] or 'image/png' + size = os.path.getsize(p) + oversized = size > _MAX_IMAGE_BYTES or total_bytes + size > _MAX_IMAGE_TOTAL_BYTES + if mime not in _VISION_MIMES or oversized or image_count >= _MAX_IMAGE_COUNT: + blocks.append({"type": "text", "text": f"[attached file: {p}]"}) + continue + with open(p, 'rb') as f: + data = base64.b64encode(f.read()).decode() + total_bytes += size; image_count += 1 + blocks.append({"type": "image", "source": {"type": "base64", "media_type": mime, "data": data}}) + except OSError: + pass + return blocks + # SDK: # agent = GenericAgent(); threading.Thread(target=agent.run, daemon=True).start() # output1_queue = agent.put_task(prompt1) @@ -154,6 +180,7 @@ def run(self): task = self.task_queue.get() if isinstance(task, str): break raw_query, source, display_queue = task["query"], task["source"], task["output"] + images = task.get("images") or [] raw_query = self._handle_slash_cmd(raw_query, display_queue) if raw_query is None: self.task_queue.task_done(); continue @@ -180,8 +207,11 @@ def run(self): if self.force_non_stream: self.llmclient.backend.stream = False self.llmclient.backend.read_timeout = max(self.llmclient.backend.read_timeout, 1200) + native_images = images and isinstance(self.llmclient, NativeToolClient) + init_content = build_image_content(raw_query, images) if native_images else None gen = agent_runner_loop(self.llmclient, sys_prompt, raw_query, handler, TOOLS_SCHEMA, - max_turns=180, verbose=self.verbose, yield_info=True) + max_turns=180, verbose=self.verbose, yield_info=True, + initial_user_content=init_content) try: full_resp = ""; last_pos = 0; curr_turn = 0; turn_resps = self.all_outputs[-1]["outputs"] for chunk in gen: diff --git a/frontends/desktop_bridge.py b/frontends/desktop_bridge.py index f44ad287b..3b5e35ea0 100644 --- a/frontends/desktop_bridge.py +++ b/frontends/desktop_bridge.py @@ -1039,45 +1039,6 @@ def submit_prompt(self, sid: str, prompt: Any, images: Optional[list] = None, di emit_session_state(sess, "running") return {"ok": True, "sessionId": sid, "accepted": True, "userMessageId": user_msg["id"], "seq": seq} - @staticmethod - def _patch_chat_for_images(client, image_paths): - """Monkey-patch backend.ask to inject base64 image blocks on the first LLM call.""" - import base64 as b64, mimetypes - try: - from llmcore import NativeToolClient - except ImportError: - return - if not isinstance(client, NativeToolClient): - return - backend = client.backend - original_ask = backend.ask - - _VISION_MIMES = {'image/png', 'image/jpeg', 'image/gif', 'image/webp'} - - def patched_ask(msg): - try: - del backend.ask - except AttributeError: - backend.ask = original_ask - if isinstance(msg, dict) and isinstance(msg.get("content"), list): - for p in image_paths: - try: - mime = mimetypes.guess_type(p)[0] or 'image/png' - if mime not in _VISION_MIMES: - # Unsupported image format (e.g. SVG) — inject as text path reference - msg["content"].append({"type": "text", "text": f"[attached file: {p}]"}) - continue - with open(p, 'rb') as f: - raw = f.read() - data = b64.b64encode(raw).decode() - msg["content"].append({"type": "image", "source": {"type": "base64", "media_type": mime, "data": data}}) - except Exception: - pass - resp = yield from original_ask(msg) - return resp - - backend.ask = patched_ask - def run_agent_turn( self, sess: Session, @@ -1131,8 +1092,6 @@ def turn_state() -> tuple[bool, bool]: ) or None except Exception: sess.running_model = None - if images: - self._patch_chat_for_images(agent.llmclient, images) full = "" done_outputs = None # done时agent给的全量轮文本(turn_resps.copy()) if hasattr(agent, "put_task"): diff --git a/frontends/tests/test_bridge_submit.py b/frontends/tests/test_bridge_submit.py index 630c06b10..7ff5e9947 100644 --- a/frontends/tests/test_bridge_submit.py +++ b/frontends/tests/test_bridge_submit.py @@ -3,7 +3,24 @@ Tests the core file-attachment contract: files_meta → agent_prompt path prepend. Run: pytest frontends/tests/test_bridge_submit.py -v """ +import ast +import base64 import json +import os +from pathlib import Path + + +ROOT = Path(__file__).resolve().parent.parent.parent + + +def _load_image_builder(): + tree = ast.parse((ROOT / "agentmain.py").read_text(encoding="utf-8")) + names = {"_VISION_MIMES", "_MAX_IMAGE_BYTES", "_MAX_IMAGE_TOTAL_BYTES", "_MAX_IMAGE_COUNT"} + nodes = [n for n in tree.body if isinstance(n, ast.FunctionDef) and n.name == "build_image_content" + or isinstance(n, ast.Assign) and any(isinstance(t, ast.Name) and t.id in names for t in n.targets)] + ns = {"os": os} + exec(compile(ast.Module(nodes, []), "agentmain.py", "exec"), ns) + return ns["build_image_content"] class TestPathPrepend: @@ -116,7 +133,7 @@ def test_combined_files_and_images(self): class TestImagePathExtraction: - """Test image_paths extraction from image_metas (used for _patch_chat_for_images).""" + """Test image_paths extraction from image_metas.""" def test_extracts_paths(self): image_metas = [ @@ -140,6 +157,38 @@ def test_skips_entries_without_path(self): assert image_paths == ["/tmp/a.png"] +class TestImageContent: + def test_builds_image_and_limits_oversized_files(self, tmp_path): + small, large = tmp_path / "small.png", tmp_path / "large.png" + small.write_bytes(b"png-data") + large.write_bytes(b"x" * (10 * 1024 * 1024 + 1)) + blocks = _load_image_builder()("describe", [str(small)] + [str(large)] * 9) + assert blocks[0] == {"type": "text", "text": "describe"} + assert blocks[1]["source"] == {"type": "base64", "media_type": "image/png", + "data": base64.b64encode(b"png-data").decode()} + assert len(blocks) == 11 + assert all(b == {"type": "text", "text": f"[attached file: {large}]"} for b in blocks[2:]) + + def test_native_chat_keeps_image_blocks(self): + import importlib.util + spec = importlib.util.spec_from_file_location("image_test_llmcore", ROOT / "llmcore.py") + llmcore = importlib.util.module_from_spec(spec); spec.loader.exec_module(llmcore) + captured = {} + + class Backend: + history, tools, model = [], None, "test" + def set_system(self, _): pass + def ask(self, message): + captured["message"] = message + if False: yield + + client = llmcore.NativeToolClient.__new__(llmcore.NativeToolClient) + client.backend, client.log_path = Backend(), False + image = {"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": "x"}} + list(client.chat([{"role": "user", "content": [{"type": "text", "text": " "}, image]}])) + assert captured["message"]["content"] == [image] + + class TestSessionFiltering: """Test the tui_ prefix filter for conductor sessions.""" @@ -223,13 +272,10 @@ def test_run_agent_turn_does_not_reference_bare_sid(self): f"Use 'sess.id' instead — 'sid' is only defined in submit_prompt's scope." ) - def test_patch_chat_for_images_exists(self): - """_patch_chat_for_images must exist in bridge — it's the image injection path.""" + def test_images_flow_through_task(self): source = self._get_bridge_source() - assert "_patch_chat_for_images" in source, ( - "_patch_chat_for_images method missing from desktop_bridge.py — " - "image uploads will silently fail (agent won't see images)" - ) + assert "_patch_chat_for_images" not in source + assert "put_task(prompt, images=" in self._extract_method_body(source, "run_agent_turn") def test_submit_prompt_separates_agent_prompt_from_stored_message(self): """submit_prompt must store clean prompt in message but pass paths to agent.""" @@ -257,5 +303,5 @@ def test_image_paths_passed_to_run_agent_turn(self): body = self._extract_method_body(source, "submit_prompt") assert "image_paths" in body, ( "submit_prompt must extract image_paths from image_metas and pass to " - "run_agent_turn. Without this, _patch_chat_for_images receives None." + "run_agent_turn. Without this, put_task receives no images." ) diff --git a/llmcore.py b/llmcore.py index b9077f5df..b1a950250 100644 --- a/llmcore.py +++ b/llmcore.py @@ -1207,8 +1207,8 @@ def chat(self, messages, tools=None): for tid in self._pending_tool_ids: if tid not in tr_id_set: tool_result_blocks.append({"type": "tool_result", "tool_use_id": tid, "content": ""}) self._pending_tool_ids = [] - # Filter whitespace-only text blocks that cause 400 on strict API proxies - filtered_content = [c for c in combined_content if c.get("text", "").strip()] + # Filter whitespace-only text blocks that cause 400 on strict API proxies; keep non-text blocks (e.g. images) + filtered_content = [c for c in combined_content if c.get("type") != "text" or c.get("text", "").strip()] final_content = tool_result_blocks + filtered_content if not final_content: final_content = [{"type": "text", "text": "."}] merged = {"role": "user", "content": final_content}