diff --git a/tests/tools/test_computer_use.py b/tests/tools/test_computer_use.py index 83ebd4581e9ab..23aaad83a42cb 100644 --- a/tests/tools/test_computer_use.py +++ b/tests/tools/test_computer_use.py @@ -1109,6 +1109,49 @@ def test_mixed_formats_in_single_tree(self): assert labels[15] == "Search" +class TestStructuredElementParsing: + """cua-driver 0.6.x emits a structuredContent.elements array with real + ``frame`` bounds. _parse_elements_from_structured must surface those bounds + (the markdown tree only ever yields zeroed bounds) so coordinate clicks and + spatial reasoning work. + """ + + def test_structured_elements_carry_real_bounds(self): + from tools.computer_use.cua_backend import _parse_elements_from_structured + raw = [ + {"element_index": 0, "role": "AXWindow", "label": "Calendar", + "frame": {"x": 218.0, "y": 56.0, "w": 1510.0, "h": 995.0}}, + {"element_index": 5, "role": "AXCheckBox", "label": "calendar-checkbox", + "frame": {"x": 244.4, "y": 148.6, "w": 16.0, "h": 16.0}}, + ] + els = _parse_elements_from_structured(raw) + assert len(els) == 2 + assert els[0].index == 0 + assert els[0].role == "AXWindow" + assert els[0].label == "Calendar" + assert els[0].bounds == (218, 56, 1510, 995) + # rounds floats to int logical px + assert els[1].bounds == (244, 149, 16, 16) + + def test_missing_frame_defaults_to_zero_bounds(self): + from tools.computer_use.cua_backend import _parse_elements_from_structured + els = _parse_elements_from_structured([ + {"element_index": 3, "role": "AXRow", "label": ""}, + ]) + assert len(els) == 1 + assert els[0].bounds == (0, 0, 0, 0) + + def test_entries_without_index_are_skipped(self): + from tools.computer_use.cua_backend import _parse_elements_from_structured + els = _parse_elements_from_structured([ + {"role": "AXGroup", "label": "no index"}, + {"element_index": 7, "role": "AXButton", "label": "ok", + "frame": {"x": 1, "y": 2, "w": 3, "h": 4}}, + ]) + assert len(els) == 1 + assert els[0].index == 7 + + class TestCaptureAfterAppContext: """Bug 2: capture_after=True loses app context after actions. diff --git a/tools/computer_use/cua_backend.py b/tools/computer_use/cua_backend.py index 4bacefa994bff..c39aeacbd2630 100644 --- a/tools/computer_use/cua_backend.py +++ b/tools/computer_use/cua_backend.py @@ -42,7 +42,7 @@ # Version pinning # --------------------------------------------------------------------------- -PINNED_CUA_DRIVER_VERSION = os.environ.get("HERMES_CUA_DRIVER_VERSION", "0.5.0") +PINNED_CUA_DRIVER_VERSION = os.environ.get("HERMES_CUA_DRIVER_VERSION", "0.6.5") _CUA_DRIVER_CMD = os.environ.get("HERMES_CUA_DRIVER_CMD", "cua-driver") _CUA_DRIVER_ARGS = ["mcp"] # stdio MCP transport @@ -126,6 +126,37 @@ def _parse_elements_from_tree(markdown: str) -> List[UIElement]: return elements +def _parse_elements_from_structured(raw_elements: List[Dict[str, Any]]) -> List[UIElement]: + """Parse UIElement list from cua-driver 0.6.x structuredContent.elements. + + Each entry carries ``element_index``, ``role``, ``label`` and a + ``frame: {x, y, w, h}`` rect (logical px). Unlike the markdown tree, this + gives us real bounds for coordinate-based clicks and spatial reasoning. + """ + elements: List[UIElement] = [] + for e in raw_elements: + frame = e.get("frame") or {} + try: + bounds = ( + int(round(frame.get("x", 0))), + int(round(frame.get("y", 0))), + int(round(frame.get("w", 0))), + int(round(frame.get("h", 0))), + ) + except (TypeError, ValueError): + bounds = (0, 0, 0, 0) + idx = e.get("element_index") + if idx is None: + continue + elements.append(UIElement( + index=int(idx), + role=e.get("role", "") or "", + label=e.get("label", "") or "", + bounds=bounds, + )) + return elements + + def _image_dimensions_from_bytes(raw: bytes) -> Tuple[int, int]: """Best-effort PNG/JPEG dimension sniffing without extra dependencies.""" if raw.startswith(b"\x89PNG\r\n\x1a\n") and len(raw) >= 24: @@ -492,51 +523,85 @@ def capture(self, mode: str = "som", app: Optional[str] = None) -> CaptureResult self._last_app = app_name # Step 2: capture. + # + # cua-driver 0.6.x unified capture into a single `get_window_state` + # tool that takes a `capture_mode` (som | vision | ax) and returns: + # * the screenshot as an MCP image content-part (som/vision) + # * an `elements` array in structuredContent with real `frame` bounds + # * `tree_markdown` in structuredContent / text for back-compat + # The standalone `screenshot` tool was removed in 0.6.x, so vision mode + # must also route through get_window_state. We prefer structuredContent + # (real bounds) and fall back to markdown parsing (zeroed bounds) for + # older 0.5.x daemons. png_b64: Optional[str] = None elements: List[UIElement] = [] width = height = 0 window_title = "" - if mode == "vision": - # screenshot tool: just the PNG, no AX walk. - sc_out = self._session.call_tool( - "screenshot", - {"window_id": self._active_window_id, "format": "jpeg", "quality": 85}, - ) - if sc_out["images"]: - png_b64 = sc_out["images"][0] - else: - # get_window_state: AX tree + optional screenshot. - gws_out = self._session.call_tool( - "get_window_state", - {"pid": self._active_pid, "window_id": self._active_window_id}, - ) - text = gws_out["data"] if isinstance(gws_out["data"], str) else "" - summary, tree = _split_tree_text(text) - - # Parse element count from summary e.g. "✅ AppName — 42 elements, turn 3..." - m = re.search(r'(\d+)\s+elements?', summary) - if tree and not gws_out["images"]: - # ax mode — no screenshot - elements = _parse_elements_from_tree(tree) - elif gws_out["images"]: - png_b64 = gws_out["images"][0] - elements = _parse_elements_from_tree(tree) - - # Extract window title from the AX tree first AXWindow line. - wt = re.search(r'AXWindow\s+"([^"]+)"', tree) + # cua-driver capture_mode values line up 1:1 with hermes modes. + cua_capture_mode = mode if mode in {"som", "vision", "ax"} else "som" + gws_out = self._session.call_tool( + "get_window_state", + { + "pid": self._active_pid, + "window_id": self._active_window_id, + "capture_mode": cua_capture_mode, + }, + ) + + structured = gws_out.get("structuredContent") or {} + + # Screenshot: 0.6.x delivers it as an MCP image-part; some builds also + # echo it as structuredContent.screenshot_png_b64. Prefer the image-part. + if gws_out.get("images"): + png_b64 = gws_out["images"][0] + elif structured.get("screenshot_png_b64"): + png_b64 = structured["screenshot_png_b64"] + + # Elements: prefer structuredContent.elements (real bounds). Fall back + # to markdown tree parsing (zeroed bounds) for older daemons. + text = gws_out["data"] if isinstance(gws_out["data"], str) else "" + tree_markdown = structured.get("tree_markdown") + if tree_markdown is None: + _summary, tree_markdown = _split_tree_text(text) + + if mode != "vision": + raw_elements = structured.get("elements") + if raw_elements: + elements = _parse_elements_from_structured(raw_elements) + elif tree_markdown: + elements = _parse_elements_from_tree(tree_markdown) + + # Window title: structured first AXWindow label, else markdown scan. + if structured.get("elements"): + for e in structured["elements"]: + if e.get("role") == "AXWindow" and e.get("label"): + window_title = e["label"] + break + if not window_title and tree_markdown: + wt = re.search(r'AXWindow\s+"([^"]+)"', tree_markdown) if wt: window_title = wt.group(1) + # Screenshot dimensions: structured reports them directly; otherwise + # sniff from the PNG/JPEG bytes below. + if structured.get("screenshot_width") and structured.get("screenshot_height"): + try: + width = int(structured["screenshot_width"]) + height = int(structured["screenshot_height"]) + except (TypeError, ValueError): + pass + png_bytes_len = 0 if png_b64: try: raw = base64.b64decode(png_b64, validate=False) png_bytes_len = len(raw) - detected_width, detected_height = _image_dimensions_from_bytes(raw) - if detected_width and detected_height: - width = detected_width - height = detected_height + if not (width and height): + detected_width, detected_height = _image_dimensions_from_bytes(raw) + if detected_width and detected_height: + width = detected_width + height = detected_height except Exception: png_bytes_len = len(png_b64) * 3 // 4