From 7aa973caf30bb23875f45f33fc405afe10bfc0f9 Mon Sep 17 00:00:00 2001 From: lihujun Date: Fri, 18 Sep 2026 14:50:17 +0800 Subject: [PATCH 1/6] Fix Unicode input fallback when clipboard writes are denied --- artemis/clients/screen_client_factory.py | 14 ++- artemis/clients/ui_automator_client.py | 3 +- artemis/drivers/android/adb_driver.py | 23 ++-- .../clients/test_screen_client_factory.py | 46 ++++++++ tests/unit/test_android_text_input.py | 104 ++++++++++++++++++ 5 files changed, 175 insertions(+), 15 deletions(-) create mode 100644 tests/unit/test_android_text_input.py diff --git a/artemis/clients/screen_client_factory.py b/artemis/clients/screen_client_factory.py index 311f2629..93149d12 100644 --- a/artemis/clients/screen_client_factory.py +++ b/artemis/clients/screen_client_factory.py @@ -300,8 +300,18 @@ def get_screenshot_base64(self) -> str | None: def set_clipboard(self, text: str) -> bool: return self._call("set_clipboard", text) - def send_text(self, text: str) -> Any: - return self._call("send_text", text) + def send_text(self, text: str) -> bool: + result = self._call("send_text", text) + if result is False and self._active_backend == "helper": + # A healthy helper may observe a field but be unable to set its text. + # Try IME input without marking the whole helper service unavailable. + self._ensure_device_online() + try: + return self.uiautomator.send_text(text) + finally: + # UiAutomation can unbind the helper even when input fails. + self._set_active("uiautomator", "Helper rejected direct text input") + return result def clear_text(self) -> bool: return self._call("clear_text") diff --git a/artemis/clients/ui_automator_client.py b/artemis/clients/ui_automator_client.py index 12a73fb0..d3eb3da6 100644 --- a/artemis/clients/ui_automator_client.py +++ b/artemis/clients/ui_automator_client.py @@ -324,7 +324,7 @@ def press_key(self, key: str): device = self._ensure_connected() return device.press(key=key) - def send_text(self, text: str) -> None: + def send_text(self, text: str) -> bool: """Send text input to the device using FastInputIME. This method supports special characters (e.g., 'ö') that ADB shell @@ -343,6 +343,7 @@ def send_text(self, text: str) -> None: # Give FastInputIME time to process the broadcast and commit text # before switching it off and killing it. time.sleep(0.5) + return True finally: device.set_fastinput_ime(False) diff --git a/artemis/drivers/android/adb_driver.py b/artemis/drivers/android/adb_driver.py index 4a543810..639e54f8 100644 --- a/artemis/drivers/android/adb_driver.py +++ b/artemis/drivers/android/adb_driver.py @@ -325,21 +325,15 @@ async def input_text(self, text: str, clear_existing: bool = True) -> bool: # Normalize literal escaped newlines from LLM / tool call serialization norm_text = text.replace(r"\r\n", "\n").replace(r"\n", "\n").replace(r"\r", "\n") - # 1. Tier 1: Try clipboard injection + KEYCODE_PASTE (Zero IME interference, preserves multiline, works for all charsets) + # Prefer direct input. A background clipboard write can be silently + # denied by Android even when the helper reports success. if self._ui_adb_client: try: - set_clip_ok = False - if hasattr(self._ui_adb_client, "set_clipboard"): - set_clip_ok = self._ui_adb_client.set_clipboard(norm_text) - elif hasattr(self._ui_adb_client, "_device") and self._ui_adb_client._device: - self._ui_adb_client._device.set_clipboard(norm_text) - set_clip_ok = True - - if set_clip_ok: - await asyncio.to_thread(self.device.shell, "input keyevent 279") + result = self._ui_adb_client.send_text(norm_text) + if result is True: return True except Exception as e: - logger.debug(f"Clipboard paste fallback to ADB input: {e}") + logger.debug(f"Direct text input failed, trying ADBKeyboard: {e}") # 2. Tier 2: Check if ADBKeyboard is currently active try: @@ -355,7 +349,12 @@ async def input_text(self, text: str, clear_existing: bool = True) -> bool: # ADBKeyboard probe/broadcast failed; fall through to native input. logger.debug(f"ADBKeyboard IME path failed, falling back to ADB input: {e}") - # 3. Tier 3: Universal Native ADB input text fallback + # Native adb input text cannot reliably enter Unicode characters. + if not norm_text.isascii(): + logger.warning("Unicode input failed: no supported input channel succeeded") + return False + + # 3. Tier 3: Native ADB input text fallback for ASCII lines = norm_text.split("\n") for i, line in enumerate(lines): if i > 0: diff --git a/tests/unit/clients/test_screen_client_factory.py b/tests/unit/clients/test_screen_client_factory.py index 836af3b9..474d291a 100644 --- a/tests/unit/clients/test_screen_client_factory.py +++ b/tests/unit/clients/test_screen_client_factory.py @@ -225,3 +225,49 @@ def test_disconnect_stops_the_uiautomator_server_it_started(composite): client.connect() client.disconnect() u2.disconnect.assert_called_once_with(stop_server=True) + + +def test_rejected_helper_text_uses_uiautomator(composite): + client, helper, u2, _, _ = composite + helper.send_text.return_value = False + u2.send_text.return_value = True + assert client.send_text("你好") is True + u2.send_text.assert_called_once_with("你好") + assert client.active_backend == "uiautomator" + + +def test_successful_helper_text_is_not_duplicated(composite): + client, helper, u2, _, _ = composite + helper.send_text.return_value = True + assert client.send_text("你好") is True + u2.send_text.assert_not_called() + + +def test_rejected_uiautomator_text_is_not_retried(composite): + client, helper, u2, _, _ = composite + helper.send_text.side_effect = RuntimeError("unavailable") + u2.send_text.return_value = False + assert client.send_text("你好") is False + u2.send_text.assert_called_once_with("你好") + + +def test_failed_ime_fallback_releases_uiautomation_before_helper_retry(composite): + client, helper, u2, _, _ = composite + helper.send_text.return_value = False + u2.send_text.side_effect = RuntimeError("IME unavailable") + with pytest.raises(RuntimeError, match="IME unavailable"): + client.send_text("你好") + assert client.active_backend == "uiautomator" + helper.get_hierarchy.return_value = "" + assert client.get_hierarchy() == "" + u2.stop_server.assert_called_once() + assert client.active_backend == "helper" + + +def test_rejected_text_does_not_start_fallback_on_offline_device(composite): + client, helper, _, factory, clock = composite + helper.send_text.return_value = False + clock["state"] = "offline" + with pytest.raises(DeviceOfflineError): + client.send_text("你好") + factory.assert_not_called() diff --git a/tests/unit/test_android_text_input.py b/tests/unit/test_android_text_input.py new file mode 100644 index 00000000..02eab8ea --- /dev/null +++ b/tests/unit/test_android_text_input.py @@ -0,0 +1,104 @@ +"""Regression coverage for rejected clipboard writes and Unicode input fallback.""" + +import base64 +from unittest.mock import MagicMock + +import pytest + +from artemis.clients.ui_automator_client import UIAutomatorClient +from artemis.drivers.android.adb_driver import AndroidAdbDriver + + +@pytest.fixture +def input_path(): + device = MagicMock() + device.shell.return_value = "com.example/.Keyboard" + client = MagicMock() + client.send_text.return_value = True + # The real helper can claim clipboard success despite a denied write. + client.set_clipboard.return_value = True + adb = MagicMock() + adb.device.return_value = device + action = AndroidAdbDriver("test", adb, client).input_text + return action, device, client + + +@pytest.mark.asyncio +async def test_direct_unicode_input_does_not_paste_or_duplicate(input_path): + action, device, client = input_path + assert await action("héllo\n你好", clear_existing=False) is True + client.send_text.assert_called_once_with("héllo\n你好") + client.set_clipboard.assert_not_called() + assert [c.args[0] for c in device.shell.call_args_list] == ["input keyevent 123"] + + +@pytest.mark.asyncio +@pytest.mark.parametrize("failure", [False, RuntimeError("unsupported field")]) +async def test_failed_direct_input_falls_back_to_adbkeyboard(input_path, failure): + action, device, client = input_path + if isinstance(failure, Exception): + client.send_text.side_effect = failure + else: + client.send_text.return_value = failure + device.shell.return_value = "com.android.adbkeyboard/.AdbIME" + assert await action("你好", clear_existing=False) is True + payload = base64.b64encode("你好".encode()).decode() + device.shell.assert_any_call(f"am broadcast -a ADB_INPUT_B64 --es msg '{payload}'") + client.set_clipboard.assert_not_called() + + +@pytest.mark.asyncio +async def test_unicode_without_supported_channel_reports_failure(input_path): + action, device, client = input_path + client.send_text.return_value = False + assert await action("你好", clear_existing=False) is False + assert not any(c.args[0].startswith("input text ") for c in device.shell.call_args_list) + client.set_clipboard.assert_not_called() + + +def test_uiautomator_send_text_returns_success_and_restores_ime(monkeypatch): + device = MagicMock() + monkeypatch.setattr(UIAutomatorClient, "_ensure_connected", lambda self: device) + monkeypatch.setattr("artemis.clients.ui_automator_client.time.sleep", lambda _: None) + client = object.__new__(UIAutomatorClient) + assert client.send_text("你好") is True + device.send_keys.assert_called_once_with("你好") + assert [c.args for c in device.set_fastinput_ime.call_args_list] == [(True,), (False,)] + + +@pytest.mark.asyncio +async def test_ascii_fallback_still_inputs_text(input_path): + action, device, client = input_path + client.send_text.return_value = False + assert await action("hello world", clear_existing=False) is True + device.shell.assert_any_call("input text hello%sworld") + + +@pytest.mark.asyncio +async def test_driver_normalizes_escaped_newlines_before_direct_input(): + client = MagicMock() + client.send_text.return_value = True + driver = AndroidAdbDriver("test", MagicMock(), client) + assert await driver.input_text(r"你好\n世界") is True + client.send_text.assert_called_once_with("你好\n世界") + + +def test_uiautomator_send_text_restores_ime_on_error(monkeypatch): + device = MagicMock() + device.send_keys.side_effect = RuntimeError("input failed") + monkeypatch.setattr(UIAutomatorClient, "_ensure_connected", lambda self: device) + monkeypatch.setattr("artemis.clients.ui_automator_client.time.sleep", lambda _: None) + with pytest.raises(RuntimeError, match="input failed"): + object.__new__(UIAutomatorClient).send_text("你好") + device.set_fastinput_ime.assert_called_with(False) + + +@pytest.mark.asyncio +async def test_unicode_without_client_or_adbkeyboard_fails(): + device = MagicMock() + device.shell.return_value = "com.example/.Keyboard" + adb = MagicMock() + adb.device.return_value = device + action = AndroidAdbDriver("test", adb).input_text + assert await action("你好", clear_existing=False) is False + assert not any(c.args[0].startswith("input text ") for c in device.shell.call_args_list) From eb52e2f625cf2658bb595f610565955ba0def343 Mon Sep 17 00:00:00 2001 From: lihujun Date: Fri, 18 Sep 2026 15:01:02 +0800 Subject: [PATCH 2/6] Preserve literal Unicode goals in Flash prompts --- artemis/agents/flash/flash_runner.md | 6 ++++++ artemis/agents/flash/runner.py | 6 +++++- tests/unit/agents/test_flash_runner.py | 16 ++++++++++++++++ 3 files changed, 27 insertions(+), 1 deletion(-) diff --git a/artemis/agents/flash/flash_runner.md b/artemis/agents/flash/flash_runner.md index 3c21cf8d..9e259924 100644 --- a/artemis/agents/flash/flash_runner.md +++ b/artemis/agents/flash/flash_runner.md @@ -3,6 +3,12 @@ You are an autonomous and highly efficient Android Device Execution Agent. Your **Objective: {{ goal }}** +The same original objective as an ASCII-escaped JSON string (an exact character reference): +```json +{{ goal_json }} +``` +When entering user-provided text, copy the requested text exactly from the original objective. Preserve emoji, variation selectors, skin tones, joiners, punctuation and whitespace; do not substitute a visually similar character or reinterpret a symbol by its name. JSON Unicode escapes may be used in tool arguments; they decode to the actual characters, not literal backslash text. Before reporting completion, compare the field's contents with the original requested text, not just with your previous tool arguments. + --- # 1. COGNITIVE PROTOCOL diff --git a/artemis/agents/flash/runner.py b/artemis/agents/flash/runner.py index 936f6065..ec9a3160 100644 --- a/artemis/agents/flash/runner.py +++ b/artemis/agents/flash/runner.py @@ -296,7 +296,11 @@ def _render_system_prompt(self, tools_declaration: list) -> str: prompt_path = Path(__file__).parent / "flash_runner.md" prompt_template = prompt_path.read_text(encoding="utf-8") available_tools = frozenset(t.name for t in tools_declaration) - return Template(prompt_template).render(goal=self.goal, available_tools=available_tools) + return Template(prompt_template).render( + goal=self.goal, + goal_json=json.dumps(self.goal, ensure_ascii=True), + available_tools=available_tools, + ) # ------------------------------------------------------------------ # Per-turn helpers (observe / think) diff --git a/tests/unit/agents/test_flash_runner.py b/tests/unit/agents/test_flash_runner.py index f188b5e1..d1b153e2 100644 --- a/tests/unit/agents/test_flash_runner.py +++ b/tests/unit/agents/test_flash_runner.py @@ -556,3 +556,19 @@ async def test_final_report_persists_native_thinking(mock_context): assert kwargs["operator_raw_thinking"] == "final text" assert kwargs["operator_native_thinking"] == "native summary" assert isinstance(messages[-1], ToolMessage) + + +@pytest.mark.parametrize('text', ['你好中文英😅', '👍🏽👩‍💻🇨🇳❤️', r'路径\n"原文"']) +def test_prompt_preserves_literal_unicode_goal_as_json(mock_context, text): + import json + + goal = '在当前输入框输入:' + text + with patch('artemis.controllers.unified_controller.get_driver'): + runner = FlashRunner(mock_context, goal=goal) + prompt = runner._render_system_prompt(runner._get_tools()) + # A second, ASCII-only representation makes the exact code points available + # even when the model misreads an emoji glyph (😅 was changed to ㊅ in a trace). + encoded = json.dumps(goal, ensure_ascii=True) + assert encoded in prompt + assert json.loads(encoded) == goal + assert 'original objective' in prompt From 3446fd117359c4684b66a759618c84bf2d4ad61a Mon Sep 17 00:00:00 2001 From: lihujun101 <34243386+lihujun101@users.noreply.github.com> Date: Mon, 21 Sep 2026 17:15:06 +0800 Subject: [PATCH 3/6] fix: resolve visual summarizer from agent model configuration --- artemis/agents/flash/summarizer.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/artemis/agents/flash/summarizer.py b/artemis/agents/flash/summarizer.py index 56c5e098..e844d9c3 100644 --- a/artemis/agents/flash/summarizer.py +++ b/artemis/agents/flash/summarizer.py @@ -168,7 +168,7 @@ def __init__( if model_name: self._llm = get_google_llm(model_name=target_model, temperature=0.0) else: - self._llm = get_llm(ctx, name="summarizer", is_utils=True) + self._llm = get_llm(ctx, name="summarizer") except Exception: self._llm = get_google_llm(model_name=target_model, temperature=0.0) try: From 744c8b47a972e73d6bf9130108dd47fe09d75ad1 Mon Sep 17 00:00:00 2001 From: lihujun101 <34243386+lihujun101@users.noreply.github.com> Date: Tue, 22 Sep 2026 10:44:49 +0800 Subject: [PATCH 4/6] fix: make checker reports compatible with thinking models --- artemis/agents/checker/checker.py | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/artemis/agents/checker/checker.py b/artemis/agents/checker/checker.py index 40a90e29..77a145ed 100644 --- a/artemis/agents/checker/checker.py +++ b/artemis/agents/checker/checker.py @@ -408,7 +408,9 @@ def _format_check_items(check_items: list) -> str: async def _structured_report(llm, messages) -> CheckReport: - structured_llm = llm.with_structured_output(CheckReport) + structured_llm = llm.with_structured_output( + CheckReport, method="function_calling", tool_choice="auto" + ) # A conversation must end with a user turn: Gemini rejects requests whose # last message is the model's own answer ("Requests ending with a model # turn are not supported"), which is exactly the state after the loop's @@ -416,7 +418,7 @@ async def _structured_report(llm, messages) -> CheckReport: if messages and isinstance(messages[-1], AIMessage): messages = [ *messages, - HumanMessage(content="Now provide your structured verdict report."), + HumanMessage(content="Now call the CheckReport tool with your structured verdict report."), ] result = await invoke_llm_with_timeout_message(structured_llm.ainvoke(messages)) if isinstance(result, CheckReport): @@ -482,7 +484,7 @@ async def _run_check_loop( messages.append( HumanMessage( content=( - "This is your final iteration; provide your structured verdict report now." + "This is your final iteration; call CheckReport with your verdict report now." ) ) ) @@ -497,6 +499,7 @@ async def _run_check_loop( break messages.append(response) + deferred_images: list[HumanMessage] = [] for tc in response.tool_calls: tool_name = tc["name"].split(":")[-1] if ":" in tc["name"] else tc["name"] args = dict(tc["args"]) @@ -522,9 +525,15 @@ async def _run_check_loop( # Screenshots (get_step_screenshot) enter the conversation in the # carrier the model's provider accepts; text results stay a plain # ToolMessage. - messages.extend( - tool_result_messages(tc["id"], result_obj, name=tool_name, status=status, llm=llm) - ) + for message in tool_result_messages( + tc["id"], result_obj, name=tool_name, status=status, llm=llm + ): + if isinstance(message, HumanMessage): + deferred_images.append(message) + else: + messages.append(message) + # Complete every tool_call before introducing user-carried screenshots. + messages.extend(deferred_images) if report is None: report = CheckReport(verdicts=[]) From d19a9f431538091b515be8ab585b39b5a47e086e Mon Sep 17 00:00:00 2001 From: lihujun101 <34243386+lihujun101@users.noreply.github.com> Date: Tue, 22 Sep 2026 16:07:32 +0800 Subject: [PATCH 5/6] fix: preserve Pro outputter settings and persist task reports Preserve configured Outputter values when CLI overrides are omitted, so WebUI enablement does not reset force_synthesis. Save generated text or structured output to notes/output.md for the Task Report card. Add regression coverage for configuration inheritance, explicit overrides, report persistence, and disabled output. Validation: 57 related pytest tests passed; Ruff and git diff checks passed. --- artemis/interfaces/cli/commands/run.py | 18 +++- artemis/sdk/agent.py | 14 +++ tests/unit/sdk/test_output_report.py | 130 +++++++++++++++++++++++++ 3 files changed, 159 insertions(+), 3 deletions(-) create mode 100644 tests/unit/sdk/test_output_report.py diff --git a/artemis/interfaces/cli/commands/run.py b/artemis/interfaces/cli/commands/run.py index 91a5d044..a298dc1c 100644 --- a/artemis/interfaces/cli/commands/run.py +++ b/artemis/interfaces/cli/commands/run.py @@ -22,7 +22,12 @@ from adbutils import AdbClient from langchain_core.callbacks.base import Callbacks -from artemis.config import checker_overrides_for_level, initialize_llm_config, settings +from artemis.config import ( + checker_overrides_for_level, + initialize_llm_config, + load_agent_config, + settings, +) from artemis.utils.startup_progress import publish_startup_progress from artemis import Agent, Builders from artemis.sdk.types.task import AgentProfile @@ -116,9 +121,16 @@ async def execute_task( config.with_flash_step_summarizer(enabled=enable_step_summarizer) if enable_outputter is not None or force_output_synthesis is not None: + outputter_defaults = load_agent_config().outputter config.with_outputter( - enabled=enable_outputter if enable_outputter is not None else True, - force_synthesis=bool(force_output_synthesis), + enabled=enable_outputter + if enable_outputter is not None + else outputter_defaults.enabled, + force_synthesis=( + force_output_synthesis + if force_output_synthesis is not None + else outputter_defaults.force_synthesis + ), ) if ( diff --git a/artemis/sdk/agent.py b/artemis/sdk/agent.py index a0028eb1..72823a63 100644 --- a/artemis/sdk/agent.py +++ b/artemis/sdk/agent.py @@ -17,6 +17,7 @@ import asyncio import inspect +import json import os import re @@ -101,6 +102,7 @@ remove_steps_json_from_trace_folder, ) from artemis.utils.startup_progress import publish_startup_progress +from artemis.utils.notes import save_note_content logger = get_logger(__name__) @@ -1267,6 +1269,18 @@ async def _extract_output( output_config=output_config or OutputConfig(), graph_output=state, ) + if ctx.data_engine and structured_output is not None: + report = ( + structured_output + if isinstance(structured_output, str) + else "```json\n" + + json.dumps(structured_output, ensure_ascii=False, indent=2) + + "\n```" + ) + try: + save_note_content(ctx.data_engine.base_dir, "output", report) + except OSError as exc: + logger.warning(f"[{task_name}] Failed to save Task Report: {exc}") logger.info(f"[{task_name}] Structured output: {structured_output}") record_events( output_path=request.llm_output_path, diff --git a/tests/unit/sdk/test_output_report.py b/tests/unit/sdk/test_output_report.py new file mode 100644 index 00000000..b07f413e --- /dev/null +++ b/tests/unit/sdk/test_output_report.py @@ -0,0 +1,130 @@ +"""Regression coverage for Pro report configuration and UI report persistence.""" + +from types import SimpleNamespace +from unittest.mock import AsyncMock, MagicMock + +import pytest + +from artemis.config import OutputConfig +from artemis.sdk.agent import Agent + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "configured_enabled,configured_force,enabled,force,expected_enabled,expected_force", + [ + (True, True, True, None, True, True), + (True, True, True, False, True, False), + (True, True, False, None, False, True), + (False, False, None, True, False, True), + ], +) +async def test_cli_preserves_unspecified_outputter_settings( + monkeypatch, + configured_enabled, + configured_force, + enabled, + force, + expected_enabled, + expected_force, +): + import artemis.interfaces.cli.commands.run as cli + + builder = MagicMock() + agent = MagicMock() + agent.init = AsyncMock() + agent.run_task = AsyncMock() + agent.clean = AsyncMock() + monkeypatch.setattr(cli, "initialize_llm_config", MagicMock()) + monkeypatch.setattr(cli, "AgentProfile", MagicMock()) + builders = MagicMock() + builders.AgentConfig.with_default_profile.return_value = builder + monkeypatch.setattr(cli, "Builders", builders) + monkeypatch.setattr(cli, "Agent", MagicMock(return_value=agent)) + monkeypatch.setattr(cli, "publish_startup_progress", MagicMock()) + monkeypatch.setattr( + cli, + "load_agent_config", + lambda: SimpleNamespace( + outputter=SimpleNamespace(enabled=configured_enabled, force_synthesis=configured_force) + ), + ) + monkeypatch.setenv("ARTEMIS_TASK_INGRESS", "test") + + await cli.execute_task( + "Test goal", + device_serial="mock-device", + profile="pro", + enable_outputter=enabled, + force_output_synthesis=force, + ) + + builder.with_outputter.assert_called_once_with( + enabled=expected_enabled, + force_synthesis=expected_force, + ) + + +@pytest.mark.asyncio +@pytest.mark.parametrize("result", ["# 测试报告\n通过", {"结论": "通过"}]) +async def test_pro_synthesis_saves_ui_report_without_output_description( + monkeypatch, tmp_path, result +): + import artemis.sdk.agent as sdk + + synthesize = AsyncMock(return_value=result) + monkeypatch.setattr(sdk, "outputter", synthesize) + ctx = SimpleNamespace( + execution_setup=SimpleNamespace( + outputter=SimpleNamespace(enabled=True, force_synthesis=True), + disable_outputter=False, + ), + data_engine=SimpleNamespace(base_dir=tmp_path), + ) + request = SimpleNamespace(llm_output_path=None, output_format=None) + output = await Agent._extract_output( + object.__new__(Agent), + "test", + ctx, + request, + OutputConfig(), + SimpleNamespace(), + ) + + assert output == result + synthesize.assert_awaited_once() + report = (tmp_path / "notes" / "output.md").read_text(encoding="utf-8") + assert "通过" in report + if isinstance(result, str): + assert report == result + else: + import json + + assert json.loads(report.removeprefix("```json\n").removesuffix("\n```")) == result + + +@pytest.mark.asyncio +async def test_explicit_disable_does_not_generate_report(monkeypatch, tmp_path): + import artemis.sdk.agent as sdk + + synthesize = AsyncMock() + monkeypatch.setattr(sdk, "outputter", synthesize) + ctx = SimpleNamespace( + execution_setup=SimpleNamespace( + outputter=SimpleNamespace(enabled=True, force_synthesis=True), + disable_outputter=False, + ), + data_engine=SimpleNamespace(base_dir=tmp_path), + ) + output = await Agent._extract_output( + object.__new__(Agent), + "test", + ctx, + SimpleNamespace(llm_output_path=None, output_format=None), + OutputConfig(enable_outputter=False), + SimpleNamespace(), + ) + + assert output is None + synthesize.assert_not_awaited() + assert not (tmp_path / "notes" / "output.md").exists() From 755a1274edc3156f7d3a9c4b6974d687f9a80b9b Mon Sep 17 00:00:00 2001 From: lihujun101 <34243386+lihujun101@users.noreply.github.com> Date: Wed, 23 Sep 2026 15:03:23 +0800 Subject: [PATCH 6/6] fix: keep outputter tool replies before screenshot images --- artemis/agents/outputter/outputter.py | 22 +++++++--- tests/unit/agents/test_outputter.py | 60 +++++++++++++++++++++++++++ 2 files changed, 76 insertions(+), 6 deletions(-) diff --git a/artemis/agents/outputter/outputter.py b/artemis/agents/outputter/outputter.py index aed600d8..758fd844 100644 --- a/artemis/agents/outputter/outputter.py +++ b/artemis/agents/outputter/outputter.py @@ -19,7 +19,7 @@ from pathlib import Path from jinja2 import Template -from langchain_core.messages import BaseMessage, HumanMessage, SystemMessage +from langchain_core.messages import BaseMessage, HumanMessage, SystemMessage, ToolMessage from artemis.core.tool_failure import ToolFailure, is_tool_failure from artemis.config import OutputConfig from artemis.context import ArtemisContext @@ -268,6 +268,8 @@ async def outputter( raw_answer = response.content break + tool_replies: list[ToolMessage] = [] + image_messages: list[BaseMessage] = [] for tc in response.tool_calls: tool_name = tc["name"].split(":")[-1] if ":" in tc["name"] else tc["name"] args = tc["args"] @@ -304,11 +306,19 @@ async def outputter( result = f"Error running tool {tool_name}: {e}" status = "error" - # Step screenshots enter the conversation in the carrier the - # model's provider accepts; text results stay a plain ToolMessage. - messages.extend( - tool_result_messages(tc["id"], result, name=tool_name, status=status, llm=llm) - ) + # OpenAI-compatible providers carry screenshot images in a user + # message. All tool calls must receive their replies before that + # message is inserted into the conversation. + for message in tool_result_messages( + tc["id"], result, name=tool_name, status=status, llm=llm + ): + if isinstance(message, ToolMessage): + tool_replies.append(message) + else: + image_messages.append(message) + + messages.extend(tool_replies) + messages.extend(image_messages) if raw_answer is None: raw_answer = "Error: Outputter failed to resolve the query within maximum turns." diff --git a/tests/unit/agents/test_outputter.py b/tests/unit/agents/test_outputter.py index 668e5b3f..ad384ca0 100644 --- a/tests/unit/agents/test_outputter.py +++ b/tests/unit/agents/test_outputter.py @@ -453,6 +453,66 @@ def get_llm_side_effect(ctx, name, is_utils=False, use_fallback=False): assert result == "Video played successfully." +@patch("artemis.agents.outputter.outputter.get_read_note_tool_pure") +@patch("artemis.agents.outputter.outputter.get_history_tools") +@patch("artemis.agents.outputter.outputter.get_llm") +@pytest.mark.asyncio +async def test_outputter_replies_to_all_tools_before_screenshot_images( + mock_get_llm, mock_get_history_tools, mock_get_read_note, mock_context, mock_state +): + from langchain_core.messages import AIMessage, HumanMessage, ToolMessage + from langchain_core.tools import StructuredTool + + async def screenshot(step_number: int): + return [ + {"type": "text", "text": f"Screenshot of step {step_number}"}, + {"type": "image_url", "image_url": {"url": "data:image/jpeg;base64,AA=="}}, + ] + + async def read_note(key: str): + return f"Note {key}" + + screenshot_tool = StructuredTool.from_function( + coroutine=screenshot, name="get_step_screenshot", description="Get screenshot" + ) + read_note_tool = StructuredTool.from_function( + coroutine=read_note, name="read_note", description="Read note" + ) + mock_get_history_tools.return_value = (screenshot_tool, screenshot_tool, screenshot_tool) + mock_get_read_note.return_value = read_note_tool + + model = Mock() + model.endpoint.provider = "deepseek" + bound_model = Mock() + bound_model.ainvoke = AsyncMock( + side_effect=[ + AIMessage( + content="", + tool_calls=[ + {"name": "get_step_screenshot", "args": {"step_number": 10}, "id": "shot-10"}, + {"name": "read_note", "args": {"key": "result"}, "id": "note"}, + {"name": "get_step_screenshot", "args": {"step_number": 11}, "id": "shot-11"}, + ], + ), + AIMessage(content="Verified.", tool_calls=[]), + ] + ) + model.bind_tools.return_value = bound_model + mock_get_llm.return_value = model + + answer = await outputter( + ctx=mock_context, + output_config=OutputConfig(structured_output=None, output_description=None), + graph_output=mock_state, + ) + + sent = bound_model.ainvoke.call_args_list[1].args[0] + assert [message.tool_call_id for message in sent[3:6]] == ["shot-10", "note", "shot-11"] + assert all(isinstance(message, ToolMessage) for message in sent[3:6]) + assert all(isinstance(message, HumanMessage) for message in sent[6:8]) + assert answer == "Verified." + + @patch("artemis.agents.outputter.outputter.get_llm") @pytest.mark.asyncio async def test_outputter_executes_save_note_tool(mock_get_llm, mock_context, mock_state):