Tests / Test tests.test_device_config.test_device_config_store_settings_get_set_and_defaults failed
Host-agent console showed OCR/UI-tree overlay boxes misaligned with the displayed screenshot. Two independent causes, both confirmed with real task data and pixel-level measurement of a user-provided screenshot: 1. perception/ui_parser.py parses XCUITest UI-tree bounds as iOS logical points, while scene_builder.py's Scene.width/height (via infer_png_size) and OCR bounds are in screenshot pixels, never reconciled (2.0x on Retina devices). build_scene() now detects the scale from the first x==0,y==0 UI element and rescales OCR bounds down to points-space, reporting Scene.width/height in points too. No-op for Android, where UiAutomator2 bounds already match pixels 1:1. This also fixes tap() landing at the wrong location for OCR-matched text, and lets the IOU fusion between UI-tree and OCR elements actually fire on iOS. 2. runtime/task.py captured `scene` (OCR/UI-tree data) before the LLM planning call, but re-captured `before_screenshot` for each step afterward - a real time gap during which on-screen content (e.g. a keyboard) could shift, producing a directional drift between the overlay and the displayed image. The first step of each plan batch now reuses the screenshot already taken for planning instead of capturing a new one; later steps in a multi-step batch still take a fresh capture (left unresolved, scoped out by request). Regression tests added for both the scale reconciliation (using real 828x1792 vs 414x896 numbers) and the screenshot reuse behavior.
139 lines
4.6 KiB
Python
139 lines
4.6 KiB
Python
from __future__ import annotations
|
|
|
|
from dataclasses import replace
|
|
from pathlib import Path
|
|
from struct import unpack
|
|
|
|
from core.models import Bounds, Scene, SceneElement
|
|
|
|
|
|
def build_scene(
|
|
*,
|
|
screen_width: int,
|
|
screen_height: int,
|
|
ui_elements: list[SceneElement] | None = None,
|
|
ocr_elements: list[SceneElement] | None = None,
|
|
iou_threshold: float = 0.5,
|
|
) -> Scene:
|
|
ui_elements = ui_elements or []
|
|
ocr_elements = ocr_elements or []
|
|
|
|
scale = _detect_pixel_scale(ui_elements, screen_width)
|
|
report_width, report_height = screen_width, screen_height
|
|
if abs(scale - 1.0) > 1e-6:
|
|
ocr_elements = [_scale_element(element, 1.0 / scale) for element in ocr_elements]
|
|
report_width = round(screen_width / scale)
|
|
report_height = round(screen_height / scale)
|
|
|
|
merged: list[SceneElement] = []
|
|
used_ocr: set[int] = set()
|
|
|
|
for ui_index, ui_element in enumerate(ui_elements):
|
|
best_index: int | None = None
|
|
best_iou = 0.0
|
|
for ocr_index, ocr_element in enumerate(ocr_elements):
|
|
if ocr_index in used_ocr:
|
|
continue
|
|
score = bbox_iou(ui_element.bounds, ocr_element.bounds)
|
|
if score > best_iou:
|
|
best_iou = score
|
|
best_index = ocr_index
|
|
|
|
element = _with_id(ui_element, f"ui-{ui_index:03d}")
|
|
if best_index is not None and best_iou >= iou_threshold:
|
|
used_ocr.add(best_index)
|
|
ocr_element = ocr_elements[best_index]
|
|
element = replace(
|
|
element,
|
|
text=element.text or ocr_element.text,
|
|
confidence=_best_confidence(element.confidence, ocr_element.confidence),
|
|
)
|
|
merged.append(element)
|
|
|
|
for ocr_index, ocr_element in enumerate(ocr_elements):
|
|
if ocr_index in used_ocr:
|
|
continue
|
|
merged.append(_with_id(ocr_element, f"ocr-{ocr_index:03d}"))
|
|
|
|
return Scene(
|
|
width=report_width,
|
|
height=report_height,
|
|
elements=merged,
|
|
ocr_elements=list(ocr_elements),
|
|
)
|
|
|
|
|
|
def bbox_iou(first: Bounds, second: Bounds) -> float:
|
|
x_left = max(first.x, second.x)
|
|
y_top = max(first.y, second.y)
|
|
x_right = min(first.right, second.right)
|
|
y_bottom = min(first.bottom, second.bottom)
|
|
if x_right <= x_left or y_bottom <= y_top:
|
|
return 0.0
|
|
|
|
intersection = (x_right - x_left) * (y_bottom - y_top)
|
|
first_area = first.width * first.height
|
|
second_area = second.width * second.height
|
|
union = first_area + second_area - intersection
|
|
if union <= 0:
|
|
return 0.0
|
|
return intersection / union
|
|
|
|
|
|
def infer_png_size(image: bytes | str | Path | None) -> tuple[int, int]:
|
|
if image is None:
|
|
return (0, 0)
|
|
data = Path(image).read_bytes() if not isinstance(image, bytes) else image
|
|
if len(data) >= 24 and data[:8] == b"\x89PNG\r\n\x1a\n":
|
|
width, height = unpack(">II", data[16:24])
|
|
return (int(width), int(height))
|
|
return (0, 0)
|
|
|
|
|
|
def _detect_pixel_scale(ui_elements: list[SceneElement], screen_width: int) -> float:
|
|
"""Detect the ratio between screenshot pixels and the UI tree's own unit (e.g. iOS points).
|
|
|
|
XCUITest reports UI tree bounds in logical points, which are half (or a
|
|
third, on some devices) of the screenshot's pixel dimensions on Retina
|
|
displays. Android's UiAutomator2 bounds already match screenshot pixels
|
|
1:1, so this returns 1.0 (no-op) for it. The first element rooted at
|
|
(0, 0) is used as the reference frame rather than the largest element,
|
|
since some system overlay elements report bounds that extend past the
|
|
visible screen.
|
|
"""
|
|
if screen_width <= 0:
|
|
return 1.0
|
|
for element in ui_elements:
|
|
bounds = element.bounds
|
|
if bounds.x == 0 and bounds.y == 0 and bounds.width > 0:
|
|
scale = screen_width / bounds.width
|
|
if scale > 0:
|
|
return scale
|
|
return 1.0
|
|
|
|
|
|
def _scale_element(element: SceneElement, factor: float) -> SceneElement:
|
|
bounds = element.bounds
|
|
return replace(
|
|
element,
|
|
bounds=Bounds(
|
|
x=bounds.x * factor,
|
|
y=bounds.y * factor,
|
|
width=bounds.width * factor,
|
|
height=bounds.height * factor,
|
|
),
|
|
)
|
|
|
|
|
|
def _with_id(element: SceneElement, fallback_id: str) -> SceneElement:
|
|
if element.id:
|
|
return element
|
|
return replace(element, id=fallback_id)
|
|
|
|
|
|
def _best_confidence(first: float | None, second: float | None) -> float | None:
|
|
values = [value for value in (first, second) if value is not None]
|
|
if not values:
|
|
return None
|
|
return max(values)
|