Files
agentic-mobile-control/tests/test_ai_planner.py
T
q792602257 a5aeb8889c
Tests / Test failed: 2, passed: 849
feat(runtime): add planner reflection history with rationale and thinking
- ToolCallDecision captures thinking blocks and pre-tool text output
- AnthropicToolCallingClient supports optional extended thinking (budget_tokens + beta header)
- PlannedStep carries rationale and thinking from each LLM decision
- WorldEvent replaces scene_summary with rationale/thinking/page fields (backward-compatible)
- AI planner system prompt instructs reflection before each tool call
- _history_summary() emits compact {page, rationale, action, success} dicts
- Cloud DB migration 0011 adds nullable rationale/thinking columns to planner_decision_log
- OpenAI client extracts reasoning_content into thinking field
2026-07-15 12:43:22 +08:00

257 lines
7.7 KiB
Python

from __future__ import annotations
from typing import Any
import pytest
from core.errors import TaskFailedError
from core.models import Bounds, Scene, SceneElement
from runtime.ai_planner import AIPlanner
from runtime.context import TaskContext
from runtime.planner_config import PlannerConfig
from runtime.tool_calling_client import ToolCallDecision
from runtime.tool_specs import ALL_TOOL_SPECS
class FakeToolCallingClient:
def __init__(self, decision: ToolCallDecision) -> None:
self.decision = decision
self.calls: list[dict[str, Any]] = []
def decide(
self,
*,
system_prompt: str,
user_prompt: str,
screenshot: bytes | None,
tools: list[Any],
timeout: float,
) -> ToolCallDecision:
self.calls.append(
{
"system_prompt": system_prompt,
"user_prompt": user_prompt,
"screenshot": screenshot,
"tools": tools,
"timeout": timeout,
}
)
return self.decision
def _scene() -> Scene:
return Scene(
width=10,
height=20,
elements=[
SceneElement(
id="send", type="button", text="Send", bounds=Bounds(1, 2, 3, 4)
)
],
)
def _context() -> TaskContext:
return TaskContext(task_id="task-1", goal="send a message")
def test_ai_planner_returns_single_planned_step_for_action_decision() -> None:
client = FakeToolCallingClient(
ToolCallDecision(tool_name="tap", arguments={"x": 1, "y": 2})
)
planner = AIPlanner(client=client)
steps = planner.plan(goal="send a message", scene=_scene(), context=_context())
assert len(steps) == 1
step = steps[0]
assert step.action == "tap"
assert step.description == "AI planner: tap({'x': 1, 'y': 2})"
assert step.args == {"x": 1, "y": 2}
def test_ai_planner_finish_task_success_returns_empty_plan() -> None:
client = FakeToolCallingClient(
ToolCallDecision(
tool_name="finish_task", arguments={"success": True, "reason": "done"}
)
)
planner = AIPlanner(client=client)
steps = planner.plan(goal="send a message", scene=_scene(), context=_context())
assert steps == []
def test_ai_planner_finish_task_failure_raises_task_failed_error_with_reason() -> None:
client = FakeToolCallingClient(
ToolCallDecision(
tool_name="finish_task",
arguments={"success": False, "reason": "stuck on login"},
)
)
planner = AIPlanner(client=client)
with pytest.raises(TaskFailedError, match="stuck on login"):
planner.plan(goal="send a message", scene=_scene(), context=_context())
def test_ai_planner_finish_task_failure_without_reason_uses_default_message() -> None:
client = FakeToolCallingClient(
ToolCallDecision(tool_name="finish_task", arguments={"success": False})
)
planner = AIPlanner(client=client)
with pytest.raises(TaskFailedError, match="task failed"):
planner.plan(goal="send a message", scene=_scene(), context=_context())
def test_ai_planner_goal_reached_is_always_false() -> None:
client = FakeToolCallingClient(
ToolCallDecision(tool_name="tap", arguments={"x": 1, "y": 2})
)
planner = AIPlanner(client=client)
assert (
planner.goal_reached(goal="anything", scene=_scene(), context=_context())
is False
)
def test_ai_planner_forwards_tools_screenshot_and_timeout_to_client() -> None:
client = FakeToolCallingClient(
ToolCallDecision(tool_name="tap", arguments={"x": 1, "y": 2})
)
planner = AIPlanner(client=client, config=PlannerConfig(timeout=12.5))
planner.plan(
goal="send a message",
scene=_scene(),
context=_context(),
screenshot=b"fake-bytes",
)
call = client.calls[0]
assert call["tools"] == ALL_TOOL_SPECS
assert call["screenshot"] == b"fake-bytes"
assert call["timeout"] == 12.5
assert "send a message" in call["user_prompt"]
def test_ai_planner_populates_step_prompt_from_user_prompt() -> None:
"""PlannedStep.prompt should carry the actual user prompt sent to the LLM,
not the bare task goal."""
client = FakeToolCallingClient(
ToolCallDecision(tool_name="tap", arguments={"x": 1, "y": 2})
)
planner = AIPlanner(client=client)
steps = planner.plan(goal="send a message", scene=_scene(), context=_context())
assert len(steps) == 1
assert steps[0].prompt is not None
# The per-step prompt contains the goal but also scene JSON and instruction text
assert "send a message" in steps[0].prompt
assert "Current Scene (JSON)" in steps[0].prompt
assert "Call exactly one tool" in steps[0].prompt
def test_ai_planner_step_prompt_reflects_scene_changes() -> None:
"""Per-step prompts differ when the scene changes, proving they are not
just the repeated task goal."""
from runtime.context import TaskContext
client = FakeToolCallingClient(
ToolCallDecision(tool_name="tap", arguments={"x": 1, "y": 2})
)
planner = AIPlanner(client=client)
scene_a = Scene(
width=10,
height=20,
elements=[
SceneElement(
id="btn_a", type="button", text="Alpha", bounds=Bounds(1, 2, 3, 4)
)
],
)
scene_b = Scene(
width=10,
height=20,
elements=[
SceneElement(
id="btn_b", type="button", text="Beta", bounds=Bounds(5, 6, 7, 8)
)
],
)
steps_a = planner.plan(
goal="test", scene=scene_a, context=TaskContext(task_id="t", goal="test")
)
steps_b = planner.plan(
goal="test", scene=scene_b, context=TaskContext(task_id="t", goal="test")
)
assert steps_a[0].prompt != steps_b[0].prompt
assert "Alpha" in steps_a[0].prompt
assert "Beta" in steps_b[0].prompt
def test_ai_planner_propagates_rationale_and_thinking_to_planned_step() -> None:
client = FakeToolCallingClient(
ToolCallDecision(
tool_name="tap",
arguments={"x": 1, "y": 2},
text_output="Previous step opened settings. Now tapping account.",
thinking="I need to navigate deeper.",
)
)
planner = AIPlanner(client=client)
steps = planner.plan(goal="open account", scene=_scene(), context=_context())
assert steps[0].rationale == "Previous step opened settings. Now tapping account."
assert steps[0].thinking == "I need to navigate deeper."
def test_history_summary_returns_compact_format() -> None:
from collections import deque
from runtime.ai_planner import _history_summary
from world.models import WorldEvent, WorldState
state = WorldState(
history=deque(
[
WorldEvent(
action="tap",
success=True,
rationale="Opened settings.",
page="Home",
),
WorldEvent(
action="swipe",
success=False,
rationale=None,
page="Settings",
),
]
)
)
summary = _history_summary(state)
assert summary == [
{"page": "Home", "action": "tap", "rationale": "Opened settings.", "success": True},
{"page": "Settings", "action": "swipe", "rationale": None, "success": False},
]
# Must not contain scene element data
for entry in summary:
assert "scene_summary" not in entry
assert "elements" not in entry
def test_history_summary_returns_empty_for_none_world() -> None:
from runtime.ai_planner import _history_summary
assert _history_summary(None) == []