Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
38 changes: 37 additions & 1 deletion agentmain.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,41 @@

script_dir = os.path.dirname(os.path.abspath(__file__))
BANNED_TOOLS = (['ask_user', 'start_long_term_update'] if '--no-user-tools' in sys.argv else [])

_VISION_MIMES = {'image/png', 'image/jpeg', 'image/gif', 'image/webp'}

def _multimodal_initial_content(raw_query, images, llmclient):
"""Build the first user turn's content blocks when the task carries images.

put_task() has always accepted images and every frontend passes them, but
run() used to drop them on the floor. Only NativeToolClient backends
understand Claude-style image blocks (the native Claude API takes them
as-is; the OAI paths convert them), so other clients keep plain text.
Returns None when there is nothing to add — callers pass the result
straight to agent_runner_loop(initial_user_content=...).
"""
import base64, mimetypes
if not images or not isinstance(llmclient, NativeToolClient):
return None
blocks = [{"type": "text", "text": raw_query}]
for img in images:
path = img if isinstance(img, str) else (img.get("path") if isinstance(img, dict) else None)
if not path:
continue
mime = mimetypes.guess_type(path)[0] or 'image/png'
if mime not in _VISION_MIMES:
# Unsupported format (e.g. SVG) — reference the path as text instead
blocks.append({"type": "text", "text": f"[attached file: {path}]"})
continue
try:
with open(path, 'rb') as f:
data = base64.b64encode(f.read()).decode('ascii')
except OSError as e:
blocks.append({"type": "text", "text": f"[image read failed: {path}: {e}]"})
continue
blocks.append({"type": "image", "source": {"type": "base64", "media_type": mime, "data": data}})
return blocks

def load_tool_schema(suffix=''):
global TOOLS_SCHEMA
TS = open(os.path.join(script_dir, f'assets/tools_schema{suffix}.json'), 'r', encoding='utf-8').read()
Expand Down Expand Up @@ -181,7 +216,8 @@ def run(self):
self.llmclient.backend.stream = False
self.llmclient.backend.read_timeout = max(self.llmclient.backend.read_timeout, 1200)
gen = agent_runner_loop(self.llmclient, sys_prompt, raw_query, handler, TOOLS_SCHEMA,
max_turns=180, verbose=self.verbose, yield_info=True)
max_turns=180, verbose=self.verbose, yield_info=True,
initial_user_content=_multimodal_initial_content(raw_query, task.get("images"), self.llmclient))
try:
full_resp = ""; last_pos = 0; curr_turn = 0; turn_resps = self.all_outputs[-1]["outputs"]
for chunk in gen:
Expand Down
41 changes: 0 additions & 41 deletions frontends/desktop_bridge.py
Original file line number Diff line number Diff line change
Expand Up @@ -1039,45 +1039,6 @@ def submit_prompt(self, sid: str, prompt: Any, images: Optional[list] = None, di
emit_session_state(sess, "running")
return {"ok": True, "sessionId": sid, "accepted": True, "userMessageId": user_msg["id"], "seq": seq}

@staticmethod
def _patch_chat_for_images(client, image_paths):
"""Monkey-patch backend.ask to inject base64 image blocks on the first LLM call."""
import base64 as b64, mimetypes
try:
from llmcore import NativeToolClient
except ImportError:
return
if not isinstance(client, NativeToolClient):
return
backend = client.backend
original_ask = backend.ask

_VISION_MIMES = {'image/png', 'image/jpeg', 'image/gif', 'image/webp'}

def patched_ask(msg):
try:
del backend.ask
except AttributeError:
backend.ask = original_ask
if isinstance(msg, dict) and isinstance(msg.get("content"), list):
for p in image_paths:
try:
mime = mimetypes.guess_type(p)[0] or 'image/png'
if mime not in _VISION_MIMES:
# Unsupported image format (e.g. SVG) — inject as text path reference
msg["content"].append({"type": "text", "text": f"[attached file: {p}]"})
continue
with open(p, 'rb') as f:
raw = f.read()
data = b64.b64encode(raw).decode()
msg["content"].append({"type": "image", "source": {"type": "base64", "media_type": mime, "data": data}})
except Exception:
pass
resp = yield from original_ask(msg)
return resp

backend.ask = patched_ask

def run_agent_turn(
self,
sess: Session,
Expand Down Expand Up @@ -1131,8 +1092,6 @@ def turn_state() -> tuple[bool, bool]:
) or None
except Exception:
sess.running_model = None
if images:
self._patch_chat_for_images(agent.llmclient, images)
full = ""
done_outputs = None # done时agent给的全量轮文本(turn_resps.copy())
if hasattr(agent, "put_task"):
Expand Down
41 changes: 35 additions & 6 deletions frontends/tests/test_bridge_submit.py
Original file line number Diff line number Diff line change
Expand Up @@ -223,12 +223,41 @@ def test_run_agent_turn_does_not_reference_bare_sid(self):
f"Use 'sess.id' instead — 'sid' is only defined in submit_prompt's scope."
)

def test_patch_chat_for_images_exists(self):
"""_patch_chat_for_images must exist in bridge — it's the image injection path."""
def test_core_forwards_images_to_agent_loop(self):
"""agentmain.run() must forward task images to agent_runner_loop.

Replaces the old desktop_bridge._patch_chat_for_images monkey-patch:
the core now injects image blocks via initial_user_content, so image
uploads cannot silently fail on any frontend (fsapp/tui/bridge alike),
and there is exactly one injection point.
"""
from pathlib import Path
candidates = [
Path(__file__).parent.parent.parent / "agentmain.py",
Path(__file__).parent.parent.parent.parent / "agentmain.py",
]
source = next((p.read_text(encoding="utf-8") for p in candidates if p.exists()), "")
assert source, "agentmain.py not found"
assert "_multimodal_initial_content" in source, (
"_multimodal_initial_content missing from agentmain.py — "
"images passed to put_task() would be silently dropped again"
)
assert "initial_user_content" in source, (
"agentmain.run() must pass initial_user_content to agent_runner_loop"
)

def test_run_agent_turn_does_not_monkeypatch_images(self):
"""run_agent_turn must not re-inject images at the backend level.

The core (put_task -> initial_user_content) is the single injection
point; a leftover backend.ask patch would duplicate every image.
"""
source = self._get_bridge_source()
assert "_patch_chat_for_images" in source, (
"_patch_chat_for_images method missing from desktop_bridge.py — "
"image uploads will silently fail (agent won't see images)"
body = self._extract_method_body(source, "run_agent_turn")
assert body, "Could not extract run_agent_turn body"
assert "_patch_chat_for_images" not in body, (
"run_agent_turn still monkey-patches backend.ask for images — "
"the core now injects them, so this would duplicate every image"
)

def test_submit_prompt_separates_agent_prompt_from_stored_message(self):
Expand Down Expand Up @@ -257,5 +286,5 @@ def test_image_paths_passed_to_run_agent_turn(self):
body = self._extract_method_body(source, "submit_prompt")
assert "image_paths" in body, (
"submit_prompt must extract image_paths from image_metas and pass to "
"run_agent_turn. Without this, _patch_chat_for_images receives None."
"run_agent_turn. Without this, the agent never receives the images."
)
Loading