Files
agentic-mobile-control/tests/test_task_loop.py
T
q792602257 8162509158
Tests / Test passed: 863
feat(host-agent): persist UI-tree evidence and add overlay/action visualization
Fixes issue 3: the host-agent console showed OCR results but never real
UI-tree data, because _ui_tree_nodes() checked for a get_ui_tree/ui_tree
tool action that has never existed anywhere in the codebase.

- storage/timeline.py: add a ui_tree_results field to TimelineRecord and
  Timeline.append(), mirroring the existing ocr_results field.
- runtime/task.py: _append_timeline() now extracts scene.elements with
  source == "ui" into ui_tree_results (scene_builder.build_scene() already
  preserved these; they were just never persisted).
- host_agent/web/app.py: _ui_tree_nodes() reads the new field directly
  instead of the dead tool-action check. New _overlay_payload() exposes
  each step's scene dimensions and fused element list for client-side
  rendering.
- task_detail.html: adds a toggle to overlay OCR (orange) and UI-tree
  (blue) bounding boxes on the before-action screenshot, plus a visual
  marker for the actually executed action (tap circle, or an animated
  swipe path) using an SVG viewBox so no manual coordinate-scaling JS is
  needed. Legacy/incomplete records degrade to no overlay, never an error.

Also corrects openspec/specs/runtime-task-evidence and
host-agent-console-task-pages, which had encoded the same nonexistent-tool
assumption, via the new host-agent-console-visual-evidence change.

600 tests passing; ruff/compileall/openspec validate all clean.
2026-07-15 14:39:28 +08:00

231 lines
7.3 KiB
Python

from __future__ import annotations
from pathlib import Path
from core.models import Bounds, Scene, SceneElement, Task
from runtime.executor import Executor, ExecutorConfig
from runtime.planner import PlannedStep, Planner
from runtime.task import TaskRunner, TaskRunnerConfig
from storage.artifact_store import ArtifactStore
from storage.task_metadata import TaskMetadataStore
from storage.timeline import Timeline
from tests.fakes import PNG_10X20
class ScriptedPlanner(Planner):
def __init__(self, steps: list[PlannedStep]) -> None:
self.steps = steps
def plan(self, *, goal, scene, context):
if len(context.step_results) >= len(self.steps):
return []
return [self.steps[len(context.step_results)]]
def goal_reached(self, *, goal, scene, context):
return len(context.step_results) >= len(self.steps) and all(
result.success for result in context.step_results
)
def test_task_runner_executes_loop_and_writes_timeline(tmp_path) -> None:
scene = Scene(
width=10,
height=20,
elements=[
SceneElement(
id="search",
type="input",
text="Search",
bounds=Bounds(1, 2, 4, 4),
)
],
)
planner = ScriptedPlanner(
[
PlannedStep(action="tap", description="tap search", args={"x": 3, "y": 4}),
PlannedStep(
action="input_text",
description="type query",
args={"text": "Mac mini"},
),
]
)
executor = Executor(
tools={
"tap": lambda **kwargs: {"ok": True, **kwargs},
"input_text": lambda **kwargs: {"ok": True, **kwargs},
},
config=ExecutorConfig(max_retries=1, backoff_seconds=0),
)
metadata = TaskMetadataStore(tmp_path / "tasks.sqlite3")
timeline = Timeline(ArtifactStore(tmp_path / "history"))
task = Task(goal="open app and search", device_id="iphone-1")
runner = TaskRunner(
planner=planner,
executor=executor,
metadata_store=metadata,
timeline=timeline,
config=TaskRunnerConfig(max_steps=5),
observer=lambda device_id: scene,
screenshot_provider=lambda device_id: PNG_10X20,
)
result = runner.run(task)
assert result.status == "completed"
assert len(timeline.read(task.id)) == 2
assert metadata.get_task(task.id)["status"] == "completed"
def test_task_runner_stops_before_the_next_planned_action() -> None:
scene = Scene(width=10, height=20, elements=[])
stop_requested = False
actions: list[str] = []
def record_action(**kwargs):
nonlocal stop_requested
actions.append("tap")
stop_requested = True
return {"ok": True}
runner = TaskRunner(
planner=ScriptedPlanner(
[
PlannedStep(action="tap", description="first", args={}),
PlannedStep(action="tap", description="second", args={}),
]
),
executor=Executor(
tools={"tap": record_action},
config=ExecutorConfig(max_retries=1, backoff_seconds=0),
),
config=TaskRunnerConfig(max_steps=5),
observer=lambda device_id: scene,
screenshot_provider=lambda device_id: PNG_10X20,
)
result = runner.run(
Task(goal="perform two actions", device_id="phone"),
should_stop=lambda: stop_requested,
)
assert result.status == "failed"
assert result.failure_reason == "execution interrupted"
assert actions == ["tap"]
def test_task_runner_persists_action_evidence_and_raw_ocr(tmp_path) -> None:
scene = Scene(
width=10,
height=20,
elements=[],
ocr_elements=[
SceneElement(
id="ocr-001",
type="text",
text="Search",
bounds=Bounds(1, 2, 3, 4),
confidence=0.98,
source="ocr",
)
],
)
screenshots = iter(
[
b"planning",
b"before-action",
b"after-action",
b"completion-check",
]
)
timeline = Timeline(ArtifactStore(tmp_path / "history"))
runner = TaskRunner(
planner=ScriptedPlanner(
[PlannedStep(action="tap", description="tap search", args={})]
),
executor=Executor(
tools={"tap": lambda **kwargs: {"ok": True}},
config=ExecutorConfig(max_retries=1, backoff_seconds=0),
),
timeline=timeline,
config=TaskRunnerConfig(max_steps=2),
observer=lambda device_id: scene,
screenshot_provider=lambda device_id: next(screenshots),
)
result = runner.run(Task(goal="tap search", device_id="iphone-1"))
assert result.status == "completed"
record = timeline.read(result.id)[0]
assert Path(record["before_screenshot_path"]).read_bytes() == b"before-action"
assert Path(record["after_screenshot_path"]).read_bytes() == b"after-action"
assert record["tool_call"]["description"] == "tap search"
assert record["ocr_results"][0]["text"] == "Search"
def test_task_runner_persists_ui_tree_elements_from_fused_scene(tmp_path) -> None:
scene = Scene(
width=10,
height=20,
elements=[
SceneElement(
id="ui-000",
type="button",
text="Search",
bounds=Bounds(1, 2, 3, 4),
source="ui",
),
SceneElement(
id="ocr-000",
type="text",
text="Unrelated label",
bounds=Bounds(5, 6, 3, 4),
source="ocr",
),
],
)
timeline = Timeline(ArtifactStore(tmp_path / "history"))
runner = TaskRunner(
planner=ScriptedPlanner(
[PlannedStep(action="tap", description="tap search", args={})]
),
executor=Executor(
tools={"tap": lambda **kwargs: {"ok": True}},
config=ExecutorConfig(max_retries=1, backoff_seconds=0),
),
timeline=timeline,
config=TaskRunnerConfig(max_steps=2),
observer=lambda device_id: scene,
screenshot_provider=lambda device_id: PNG_10X20,
)
result = runner.run(Task(goal="tap search", device_id="iphone-1"))
record = timeline.read(result.id)[0]
assert [element["text"] for element in record["ui_tree_results"]] == ["Search"]
assert [element["text"] for element in record["ocr_results"]] == ["Unrelated label"]
def test_task_runner_persists_empty_ui_tree_results_when_scene_has_no_ui_elements(
tmp_path,
) -> None:
scene = Scene(width=10, height=20, elements=[])
timeline = Timeline(ArtifactStore(tmp_path / "history"))
runner = TaskRunner(
planner=ScriptedPlanner(
[PlannedStep(action="tap", description="tap search", args={})]
),
executor=Executor(
tools={"tap": lambda **kwargs: {"ok": True}},
config=ExecutorConfig(max_retries=1, backoff_seconds=0),
),
timeline=timeline,
config=TaskRunnerConfig(max_steps=2),
observer=lambda device_id: scene,
screenshot_provider=lambda device_id: PNG_10X20,
)
result = runner.run(Task(goal="tap search", device_id="iphone-1"))
record = timeline.read(result.id)[0]
assert record["ui_tree_results"] == []