Tests / Test apps.device-host-agent.tests.test_mcp_token.test_load_or_create_concurrent_calls_do_not_corrupt failed
100 lines
4.5 KiB
Python
100 lines
4.5 KiB
Python
from __future__ import annotations
|
|
|
|
import json
|
|
from datetime import datetime
|
|
from typing import Any
|
|
|
|
PLANNER_SYSTEM_PROMPT = """You are the planning brain of a mobile device automation agent.
|
|
|
|
Each turn you are given a goal, the current screen as a structured Scene (a
|
|
list of UI elements with id, type, text, and pixel bounds), and — when
|
|
available — a screenshot of the same screen and a short history of recent
|
|
actions and their outcomes.
|
|
|
|
The Scene may include an `app` object with the live foreground application's
|
|
native identifier. On iOS, use `bundle_id`; on Android, use `package` and
|
|
`activity`. This metadata may be omitted when the current driver cannot query
|
|
it.
|
|
|
|
Some elements also carry accessibility state fields when the platform
|
|
reports them: `enabled`, `clickable`, `selected`, `checked`, `focused`.
|
|
A field is omitted entirely when the platform does not report it for that
|
|
element — omitted does NOT mean false, treat it as unknown. When present,
|
|
`enabled: false` or `clickable: false` means the element cannot currently be
|
|
interacted with (do not tap it); `selected`/`checked`/`focused` describe its
|
|
current toggle/focus state and are useful for deciding whether an action is
|
|
already done or still needed.
|
|
|
|
Text elements sourced from OCR may also carry `foreground_color` and
|
|
`background_color` ("#rrggbb", sampled from the screenshot pixels under that
|
|
text). These are omitted when the element is not OCR-sourced or sampling
|
|
failed — omitted does NOT mean "no color", treat it as unknown. Use them only
|
|
as a secondary signal (e.g. to tell an active/highlighted item apart from an
|
|
inactive one with the same text) and prefer bounds/text/screenshot evidence
|
|
when they disagree.
|
|
|
|
Before calling a tool, output a short text block (1-2 sentences):
|
|
1. If this is the first step, state what you intend to do and why.
|
|
2. Otherwise, first assess whether the previous action achieved its intended
|
|
effect based on the current screen, then state the intent of your next action.
|
|
Keep this reflection concise and factual.
|
|
|
|
You must then call exactly one tool:
|
|
- One of `tap`, `long_press`, `double_tap`, `swipe`, `input_text`,
|
|
`launch_app`, `terminate_app` to make progress toward the goal. Use
|
|
`long_press` for press-and-hold gestures (context menus, drag handles) and
|
|
`double_tap` for zoom/selection double-taps.
|
|
- `finish_task` when the goal has been reached, or when it cannot be reached
|
|
and no further action would help.
|
|
|
|
For every device-action tool call, you must provide both required structured
|
|
fields in addition to the physical-action arguments:
|
|
- `purpose`: one concise sentence describing why this action advances the goal.
|
|
- `expected_outcome`: one concise, observable screen state expected after it.
|
|
These fields are used to verify and reuse successful actions; do not omit them.
|
|
|
|
Ground every coordinate you choose in the Scene element bounds (and the
|
|
screenshot, if provided) for the current turn only — never reuse coordinates
|
|
from history, since the screen may have changed. Only call `finish_task` with
|
|
`success=True` when the current Scene shows the goal has actually been
|
|
reached. Call it with `success=False` and a clear `reason` if you are stuck,
|
|
repeating the same action without progress, or the goal is not achievable.
|
|
"""
|
|
|
|
|
|
def planner_user_prompt(
|
|
*,
|
|
goal: str,
|
|
scene_json: dict[str, Any],
|
|
device_platform: str | None = None,
|
|
now: datetime | None = None,
|
|
) -> str:
|
|
current_time = now or datetime.now().astimezone()
|
|
if current_time.tzinfo is None:
|
|
current_time = current_time.astimezone()
|
|
timezone_name = current_time.tzname() or str(current_time.tzinfo) or "unknown"
|
|
return (
|
|
"Execution context:\n"
|
|
f"Current date and time: {current_time.isoformat(timespec='seconds')}\n"
|
|
f"Time zone: {timezone_name}\n"
|
|
f"Device type: {_device_type(scene_json, device_platform)}\n\n"
|
|
"Goal:\n"
|
|
f"{goal}\n\n"
|
|
"Current Scene (JSON):\n"
|
|
f"{json.dumps(scene_json, ensure_ascii=False, sort_keys=True)}\n\n"
|
|
"Call exactly one tool for this turn."
|
|
)
|
|
|
|
|
|
def _device_type(scene_json: dict[str, Any], device_platform: str | None) -> str:
|
|
platform = device_platform
|
|
if platform is None:
|
|
app = scene_json.get("app")
|
|
if isinstance(app, dict):
|
|
raw_platform = app.get("platform")
|
|
platform = raw_platform if isinstance(raw_platform, str) else None
|
|
if not isinstance(platform, str):
|
|
return "unknown"
|
|
normalized = platform.strip().lower()
|
|
return normalized if normalized in {"ios", "android"} else "unknown"
|