diff --git a/assets/tools_schema_cn.json b/assets/tools_schema_cn.json index 98fbc211b..657865cfd 100644 --- a/assets/tools_schema_cn.json +++ b/assets/tools_schema_cn.json @@ -67,7 +67,7 @@ }}, {"type": "function", "function": { "name": "start_long_term_update", - "description": "准备开始提炼记忆。发现值得长期记忆的信息(环境事实/用户偏好/避坑经验)时调用此工具。已记忆更新或在自主流程内时无需调用。超15轮完成的任务必须调用以沉淀经验", + "description": "准备开始提炼记忆。发现值得长期记忆的信息(环境事实/用户偏好/避坑经验)时调用此工具。已记忆更新或在自主流程内时无需调用。达到15轮(含)完成的任务必须调用以沉淀经验", "parameters": {"type": "object", "properties": {}}} } ] \ No newline at end of file diff --git a/frontends/tests/test_long_term_settlement_gate.py b/frontends/tests/test_long_term_settlement_gate.py new file mode 100644 index 000000000..dfcd3c00a --- /dev/null +++ b/frontends/tests/test_long_term_settlement_gate.py @@ -0,0 +1,117 @@ +"""Tests for the completion-time long-term settlement gate (#789). + +ga.py imports with stdlib only (agent_loop falls back when plugins.hooks is +absent), so these tests exercise the real GenericAgentHandler. +Run: pytest frontends/tests/test_long_term_settlement_gate.py -v +""" +import sys +from pathlib import Path +from types import SimpleNamespace + +ROOT = Path(__file__).resolve().parent.parent.parent + +# test_bridge_sessions installs an attribute-less `agent_loop` stub into +# sys.modules; evict it so the real module can load (ga imports agent_loop). +_stub = sys.modules.get("agent_loop") +if _stub is not None and not hasattr(_stub, "BaseHandler"): + del sys.modules["agent_loop"] +sys.path.insert(0, str(ROOT)) + +from ga import GenericAgentHandler + + +def _handler(turn, history=None): + parent = SimpleNamespace(task_dir=None, verbose=False, extrakeyinfo=None, intervene=None) + h = GenericAgentHandler(parent, last_history=history or []) + h.current_turn = turn + return h + + +def _run(gen): + """Exhaust a do_* generator -> (yielded strings, returned StepOutcome).""" + yields, outcome = [], None + try: + while True: + yields.append(next(gen)) + except StopIteration as e: + outcome = e.value + return yields, outcome + + +def _response(content="任务已完成"): + return SimpleNamespace(content=content, thinking="", tool_calls=None) + + +_SETTLE_MARK = "记忆提纯" + + +class TestGateThreshold: + def test_turn_14_completes_normally(self): + h = _handler(14) + _, outcome = _run(h.do_no_tool({}, _response())) + assert outcome.next_prompt is None + assert not h._lt_started + + def test_turn_15_enters_settlement_once(self): + h = _handler(15) + yields, outcome = _run(h.do_no_tool({}, _response())) + assert h._lt_started + assert _SETTLE_MARK in outcome.next_prompt + assert any("distilling" in y for y in yields) + + def test_turn_40_enters_settlement(self): + h = _handler(40) + _, outcome = _run(h.do_no_tool({}, _response())) + assert _SETTLE_MARK in outcome.next_prompt + + def test_settlement_completion_exits_without_recursion(self): + h = _handler(20) + _, first = _run(h.do_no_tool({}, _response())) + assert first.next_prompt is not None # settlement phase started + _, second = _run(h.do_no_tool({}, _response())) + assert second.next_prompt is None # settlement done -> normal exit + + def test_explicit_call_blocks_gate(self): + h = _handler(20) + _, settled = _run(h.do_start_long_term_update({}, _response())) + assert settled.next_prompt is not None + _, outcome = _run(h.do_no_tool({}, _response())) + assert outcome.next_prompt is None # gate already satisfied + + def test_refused_early_call_does_not_block_gate(self): + h = _handler(5) + _, refused = _run(h.do_start_long_term_update({}, _response())) + assert "only used after completing" in refused.data + assert not h._lt_started + h.current_turn = 20 + _, outcome = _run(h.do_no_tool({}, _response())) + assert _SETTLE_MARK in outcome.next_prompt # gate still fires later + + +class TestExemptions: + def test_no_user_tools_exempt(self, monkeypatch): + monkeypatch.setattr(sys, "argv", ["ga", "--no-user-tools"]) + h = _handler(30) + _, outcome = _run(h.do_no_tool({}, _response())) + assert outcome.next_prompt is None + assert not h._lt_started + + def test_autonomous_flow_exempt(self): + h = _handler(30, history=["[USER]: [AUTO]🤖 用户已经离开超过30分钟,执行自动任务。"]) + _, outcome = _run(h.do_no_tool({}, _response())) + assert outcome.next_prompt is None + assert not h._lt_started + + def test_normal_user_message_not_exempt(self): + h = _handler(30, history=["[USER]: 帮我规划旅行"]) + _, outcome = _run(h.do_no_tool({}, _response())) + assert _SETTLE_MARK in outcome.next_prompt + + def test_autonomous_detection_uses_latest_user_message(self): + h = _handler(30, history=[ + "[USER]: [AUTO] auto task", + "[Agent] did stuff", + "[USER]: 普通问题", + ]) + _, outcome = _run(h.do_no_tool({}, _response())) + assert _SETTLE_MARK in outcome.next_prompt diff --git a/ga.py b/ga.py index b2a21df24..a0e786329 100644 --- a/ga.py +++ b/ga.py @@ -287,6 +287,7 @@ def __init__(self, parent, last_history=None, cwd='./temp'): self.code_stop_signal = [] self._done_hooks = [] self.print = safe_print + self._lt_started = False def _get_tool_maxlen(self, l, args, growth_rate=1.0): multiplier = 1 + (self.parent.get_ctx_multiplier() - 1) * growth_rate @@ -474,6 +475,21 @@ def _retry_or_exit(self, prompt): if self._empty_ct >= 3: return StepOutcome({}, should_exit=True) return StepOutcome({}, next_prompt=prompt) + _LONG_TASK_TURNS = 15 # schema:15+ turns / 达到15轮(含),turn >= 15 才门控 + + def _in_autonomous_flow(self): + for line in reversed(self.history_info): + if line.startswith('[USER]: '): + return line[8:].lstrip().startswith('[AUTO]') + return False + + def _needs_long_term_settlement(self): + '''完成时一次性结算门控(#789):schema 措辞只是建议,15+ 轮任务可能不带结算就退出。 + 已结算 / --no-user-tools / 自主流程 / 未到阈值 → 放行正常退出。''' + if self._lt_started or self.current_turn < self._LONG_TASK_TURNS: return False + if '--no-user-tools' in sys.argv: return False + return not self._in_autonomous_flow() + def do_no_tool(self, args, response): '''这是一个特殊工具,由引擎自主调用,不要包含在TOOLS_SCHEMA里。 当模型在一轮中未显式调用任何工具时,由引擎自动触发。 @@ -522,6 +538,8 @@ def do_no_tool(self, args, response): self._exit_plan_mode(); yield "[Info] Plan完成:plan.md中0个[ ]残留,退出plan模式。\n" #yield "[Info] Final response to user.\n" + if self._needs_long_term_settlement(): + return (yield from self.do_start_long_term_update(args, response)) return StepOutcome(response, next_prompt=None) def do_start_long_term_update(self, args, response): @@ -534,7 +552,10 @@ def do_start_long_term_update(self, args, response): path = './memory/memory_management_sop.md' if os.path.exists(path): result = 'This is L0:\n' + file_read(path, show_linenos=False) else: result = "Memory Management SOP not found. Do not update memory." - if self.current_turn < 10: result, prompt = 'start_long_term_update is only used after completing a long turn task!', '\n' + if self.current_turn < 10: + result, prompt = 'start_long_term_update is only used after completing a long turn task!', '\n' + else: + self._lt_started = True # 结算已启动:完成门控放行;settlement 自身完成也不会递归再触发 return StepOutcome(result, next_prompt=prompt) def _fold_earlier(self, lines):