Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
32 changes: 31 additions & 1 deletion agentmain.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,32 @@ def get_system_prompt():
prompt += get_global_memory()
return prompt

_VISION_MIMES = {'image/png', 'image/jpeg', 'image/gif', 'image/webp'}
_MAX_IMAGE_BYTES = 10 * 1024 * 1024
_MAX_IMAGE_TOTAL_BYTES = 20 * 1024 * 1024
_MAX_IMAGE_COUNT = 8

def build_image_content(query, image_paths):
"""Build first-turn multimodal content without allowing oversized requests."""
import base64, mimetypes
blocks = [{"type": "text", "text": query}]
total_bytes = image_count = 0
for p in image_paths:
try:
mime = mimetypes.guess_type(p)[0] or 'image/png'
size = os.path.getsize(p)
oversized = size > _MAX_IMAGE_BYTES or total_bytes + size > _MAX_IMAGE_TOTAL_BYTES
if mime not in _VISION_MIMES or oversized or image_count >= _MAX_IMAGE_COUNT:
blocks.append({"type": "text", "text": f"[attached file: {p}]"})
continue
with open(p, 'rb') as f:
data = base64.b64encode(f.read()).decode()
total_bytes += size; image_count += 1
blocks.append({"type": "image", "source": {"type": "base64", "media_type": mime, "data": data}})
except OSError:
pass
return blocks

# SDK:
# agent = GenericAgent(); threading.Thread(target=agent.run, daemon=True).start()
# output1_queue = agent.put_task(prompt1)
Expand Down Expand Up @@ -154,6 +180,7 @@ def run(self):
task = self.task_queue.get()
if isinstance(task, str): break
raw_query, source, display_queue = task["query"], task["source"], task["output"]
images = task.get("images") or []
raw_query = self._handle_slash_cmd(raw_query, display_queue)
if raw_query is None:
self.task_queue.task_done(); continue
Expand All @@ -180,8 +207,11 @@ def run(self):
if self.force_non_stream:
self.llmclient.backend.stream = False
self.llmclient.backend.read_timeout = max(self.llmclient.backend.read_timeout, 1200)
native_images = images and isinstance(self.llmclient, NativeToolClient)
init_content = build_image_content(raw_query, images) if native_images else None
gen = agent_runner_loop(self.llmclient, sys_prompt, raw_query, handler, TOOLS_SCHEMA,
max_turns=180, verbose=self.verbose, yield_info=True)
max_turns=180, verbose=self.verbose, yield_info=True,
initial_user_content=init_content)
try:
full_resp = ""; last_pos = 0; curr_turn = 0; turn_resps = self.all_outputs[-1]["outputs"]
for chunk in gen:
Expand Down
41 changes: 0 additions & 41 deletions frontends/desktop_bridge.py
Original file line number Diff line number Diff line change
Expand Up @@ -1039,45 +1039,6 @@ def submit_prompt(self, sid: str, prompt: Any, images: Optional[list] = None, di
emit_session_state(sess, "running")
return {"ok": True, "sessionId": sid, "accepted": True, "userMessageId": user_msg["id"], "seq": seq}

@staticmethod
def _patch_chat_for_images(client, image_paths):
"""Monkey-patch backend.ask to inject base64 image blocks on the first LLM call."""
import base64 as b64, mimetypes
try:
from llmcore import NativeToolClient
except ImportError:
return
if not isinstance(client, NativeToolClient):
return
backend = client.backend
original_ask = backend.ask

_VISION_MIMES = {'image/png', 'image/jpeg', 'image/gif', 'image/webp'}

def patched_ask(msg):
try:
del backend.ask
except AttributeError:
backend.ask = original_ask
if isinstance(msg, dict) and isinstance(msg.get("content"), list):
for p in image_paths:
try:
mime = mimetypes.guess_type(p)[0] or 'image/png'
if mime not in _VISION_MIMES:
# Unsupported image format (e.g. SVG) — inject as text path reference
msg["content"].append({"type": "text", "text": f"[attached file: {p}]"})
continue
with open(p, 'rb') as f:
raw = f.read()
data = b64.b64encode(raw).decode()
msg["content"].append({"type": "image", "source": {"type": "base64", "media_type": mime, "data": data}})
except Exception:
pass
resp = yield from original_ask(msg)
return resp

backend.ask = patched_ask

def run_agent_turn(
self,
sess: Session,
Expand Down Expand Up @@ -1131,8 +1092,6 @@ def turn_state() -> tuple[bool, bool]:
) or None
except Exception:
sess.running_model = None
if images:
self._patch_chat_for_images(agent.llmclient, images)
full = ""
done_outputs = None # done时agent给的全量轮文本(turn_resps.copy())
if hasattr(agent, "put_task"):
Expand Down
62 changes: 54 additions & 8 deletions frontends/tests/test_bridge_submit.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,24 @@
Tests the core file-attachment contract: files_meta → agent_prompt path prepend.
Run: pytest frontends/tests/test_bridge_submit.py -v
"""
import ast
import base64
import json
import os
from pathlib import Path


ROOT = Path(__file__).resolve().parent.parent.parent


def _load_image_builder():
tree = ast.parse((ROOT / "agentmain.py").read_text(encoding="utf-8"))
names = {"_VISION_MIMES", "_MAX_IMAGE_BYTES", "_MAX_IMAGE_TOTAL_BYTES", "_MAX_IMAGE_COUNT"}
nodes = [n for n in tree.body if isinstance(n, ast.FunctionDef) and n.name == "build_image_content"
or isinstance(n, ast.Assign) and any(isinstance(t, ast.Name) and t.id in names for t in n.targets)]
ns = {"os": os}
exec(compile(ast.Module(nodes, []), "agentmain.py", "exec"), ns)
return ns["build_image_content"]


class TestPathPrepend:
Expand Down Expand Up @@ -116,7 +133,7 @@ def test_combined_files_and_images(self):


class TestImagePathExtraction:
"""Test image_paths extraction from image_metas (used for _patch_chat_for_images)."""
"""Test image_paths extraction from image_metas."""

def test_extracts_paths(self):
image_metas = [
Expand All @@ -140,6 +157,38 @@ def test_skips_entries_without_path(self):
assert image_paths == ["/tmp/a.png"]


class TestImageContent:
def test_builds_image_and_limits_oversized_files(self, tmp_path):
small, large = tmp_path / "small.png", tmp_path / "large.png"
small.write_bytes(b"png-data")
large.write_bytes(b"x" * (10 * 1024 * 1024 + 1))
blocks = _load_image_builder()("describe", [str(small)] + [str(large)] * 9)
assert blocks[0] == {"type": "text", "text": "describe"}
assert blocks[1]["source"] == {"type": "base64", "media_type": "image/png",
"data": base64.b64encode(b"png-data").decode()}
assert len(blocks) == 11
assert all(b == {"type": "text", "text": f"[attached file: {large}]"} for b in blocks[2:])

def test_native_chat_keeps_image_blocks(self):
import importlib.util
spec = importlib.util.spec_from_file_location("image_test_llmcore", ROOT / "llmcore.py")
llmcore = importlib.util.module_from_spec(spec); spec.loader.exec_module(llmcore)
captured = {}

class Backend:
history, tools, model = [], None, "test"
def set_system(self, _): pass
def ask(self, message):
captured["message"] = message
if False: yield

client = llmcore.NativeToolClient.__new__(llmcore.NativeToolClient)
client.backend, client.log_path = Backend(), False
image = {"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": "x"}}
list(client.chat([{"role": "user", "content": [{"type": "text", "text": " "}, image]}]))
assert captured["message"]["content"] == [image]


class TestSessionFiltering:
"""Test the tui_ prefix filter for conductor sessions."""

Expand Down Expand Up @@ -223,13 +272,10 @@ def test_run_agent_turn_does_not_reference_bare_sid(self):
f"Use 'sess.id' instead — 'sid' is only defined in submit_prompt's scope."
)

def test_patch_chat_for_images_exists(self):
"""_patch_chat_for_images must exist in bridge — it's the image injection path."""
def test_images_flow_through_task(self):
source = self._get_bridge_source()
assert "_patch_chat_for_images" in source, (
"_patch_chat_for_images method missing from desktop_bridge.py — "
"image uploads will silently fail (agent won't see images)"
)
assert "_patch_chat_for_images" not in source
assert "put_task(prompt, images=" in self._extract_method_body(source, "run_agent_turn")

def test_submit_prompt_separates_agent_prompt_from_stored_message(self):
"""submit_prompt must store clean prompt in message but pass paths to agent."""
Expand Down Expand Up @@ -257,5 +303,5 @@ def test_image_paths_passed_to_run_agent_turn(self):
body = self._extract_method_body(source, "submit_prompt")
assert "image_paths" in body, (
"submit_prompt must extract image_paths from image_metas and pass to "
"run_agent_turn. Without this, _patch_chat_for_images receives None."
"run_agent_turn. Without this, put_task receives no images."
)
4 changes: 2 additions & 2 deletions llmcore.py
Original file line number Diff line number Diff line change
Expand Up @@ -1207,8 +1207,8 @@ def chat(self, messages, tools=None):
for tid in self._pending_tool_ids:
if tid not in tr_id_set: tool_result_blocks.append({"type": "tool_result", "tool_use_id": tid, "content": ""})
self._pending_tool_ids = []
# Filter whitespace-only text blocks that cause 400 on strict API proxies
filtered_content = [c for c in combined_content if c.get("text", "").strip()]
# Filter whitespace-only text blocks that cause 400 on strict API proxies; keep non-text blocks (e.g. images)
filtered_content = [c for c in combined_content if c.get("type") != "text" or c.get("text", "").strip()]
final_content = tool_result_blocks + filtered_content
if not final_content: final_content = [{"type": "text", "text": "."}]
merged = {"role": "user", "content": final_content}
Expand Down