From 96e31ede1440f20cdf586c7396ef313bcb12eff7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=A8=8B=E5=BA=8F=E5=91=98=E9=98=BF=E6=B1=9F=28Relakkes?= =?UTF-8?q?=29?= Date: Sun, 23 Aug 2026 18:25:21 +0800 Subject: [PATCH] fix(computer-use): make the Windows helper refuse what it cannot deliver MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Windows has no equivalent of `CGEvent.postToPid`. `pyautogui` bottoms out in `SendInput`, which injects into the one system-wide input stream and warps the one real cursor — so on Windows the agent shares the mouse and keyboard with the user, and `SendInput` reports success unconditionally whether or not anything acted on the events. That combination produced the same lie the macOS engine was just fixed for: click a point behind another window and it lands on that window; click one off-screen and it lands nowhere; either way the helper answered "Action completed". So the helper now refuses instead of guessing: * `ForegroundLease` samples `GetLastInputInfo` around every mutating command. Interference before the action is `user_interference` — nothing ran, retry is safe. Interference during it is `user_interference_result_unknown`, because injection already went out and a retry could double-apply it. On a play/pause toggle those two differ by exactly one wrong outcome. * `ensure_point_on_screen` and `ensure_target_window_reachable` reject coordinates outside every display and targets whose windows are minimized or hidden, before anything is sent. `GetLastInputInfo` is the signal because it needs no privileges and is not advanced by `SendInput`, so the agent cannot trip its own detector. Both guards fail open on an unreadable reading: a safety layer that turns the feature off is not safety. The guard set lives in one place rather than in each dispatcher branch — an eleventh verb wired like the ten before it would otherwise be silently unguarded, which is the bug class this pass removes. All ten mutating branches now return through `_finish`, so no branch can write its own success response and skip the post-action check. Also adds a Windows cursor badge. It deliberately does NOT mirror the macOS virtual cursor: there the real pointer never moves, so the drawn one is the only cursor and replaces it. Here the real pointer does move, and a second fake pointer would just be two cursors with one of them lying about where the click lands. The badge annotates instead — it answers "is this me or the agent?", which matters because grabbing the mouse mid-action is what makes the two input streams interleave. Retires `runtime/mac_helper.py` and its pyobjc requirements: macOS routes every command to the signed native daemon and `helperBridge` refuses to fall back, so both were unreachable. Renames the `callPythonHelper` alias to `callHelper`, which is what it has actually imported since the native engine landed. Verified with mutation testing — 13 injected regressions across both languages (dropped lease, bypassed `_finish`, collapsed interference codes, guard reusing the filtered `list_windows`, badge losing click-through), all caught. Claude-Session: https://claude.ai/code/session_015j1yxxaoonyAS2iZ7qGnTS --- runtime/mac_helper.py | 775 --------------------- runtime/requirements.txt | 6 - runtime/test_helpers.py | 533 ++++++++------ runtime/win_cursor_badge.py | 253 +++++++ runtime/win_helper.py | 437 +++++++++++- src/utils/computerUse/cleanup.test.ts | 34 + src/utils/computerUse/cleanup.ts | 22 +- src/utils/computerUse/executor.ts | 94 +-- src/utils/computerUse/helperBridge.test.ts | 55 ++ src/utils/computerUse/helperBridge.ts | 17 + src/utils/computerUse/hostAdapter.ts | 4 +- src/utils/computerUse/pythonBridge.ts | 40 +- src/utils/computerUse/winCursorBadge.ts | 95 +++ 13 files changed, 1270 insertions(+), 1095 deletions(-) delete mode 100755 runtime/mac_helper.py delete mode 100644 runtime/requirements.txt create mode 100644 runtime/win_cursor_badge.py create mode 100644 src/utils/computerUse/winCursorBadge.ts diff --git a/runtime/mac_helper.py b/runtime/mac_helper.py deleted file mode 100755 index 1bfae38e..00000000 --- a/runtime/mac_helper.py +++ /dev/null @@ -1,775 +0,0 @@ -#!/usr/bin/env python3 -from __future__ import annotations - -import argparse -import base64 -import ctypes -import json -import os -import subprocess -import sys -import time -from io import BytesIO -from pathlib import Path -from typing import Any - -import mss -from AppKit import NSWorkspace, NSPasteboard, NSPasteboardTypeString, NSURL -from PIL import Image -from Quartz import ( - CGDisplayBounds, - CGDisplayIsMain, - CGDisplayModeGetPixelHeight, - CGDisplayModeGetPixelWidth, - CGDisplayPixelsHigh, - CGDisplayPixelsWide, - CGGetActiveDisplayList, - CGMainDisplayID, - CGWindowListCopyWindowInfo, - CGRectContainsPoint, - CGRectIntersection, - CGPointMake, - CGPreflightScreenCaptureAccess, - kCGNullWindowID, - kCGWindowBounds, - kCGWindowIsOnscreen, - kCGWindowLayer, - kCGWindowListExcludeDesktopElements, - kCGWindowListOptionOnScreenOnly, - kCGWindowName, - kCGWindowOwnerName, -) - -os.environ.setdefault("PYTHONDONTWRITEBYTECODE", "1") -os.environ.setdefault("PYAUTOGUI_HIDE_SUPPORT_PROMPT", "1") - -import pyautogui # noqa: E402 - -pyautogui.FAILSAFE = False -pyautogui.PAUSE = 0 - -KEY_MAP = { - "a": "a", - "b": "b", - "c": "c", - "d": "d", - "e": "e", - "f": "f", - "g": "g", - "h": "h", - "i": "i", - "j": "j", - "k": "k", - "l": "l", - "m": "m", - "n": "n", - "o": "o", - "p": "p", - "q": "q", - "r": "r", - "s": "s", - "t": "t", - "u": "u", - "v": "v", - "w": "w", - "x": "x", - "y": "y", - "z": "z", - "0": "0", - "1": "1", - "2": "2", - "3": "3", - "4": "4", - "5": "5", - "6": "6", - "7": "7", - "8": "8", - "9": "9", - "cmd": "command", - "command": "command", - "meta": "command", - "super": "command", - "ctrl": "ctrl", - "control": "ctrl", - "shift": "shift", - "alt": "option", - "option": "option", - "opt": "option", - "fn": "fn", - "escape": "esc", - "esc": "esc", - "enter": "enter", - "return": "enter", - "tab": "tab", - "space": "space", - "backspace": "backspace", - "delete": "delete", - "forwarddelete": "delete", - "up": "up", - "down": "down", - "left": "left", - "right": "right", - "home": "home", - "end": "end", - "pageup": "pageup", - "pagedown": "pagedown", - "capslock": "capslock", - "f1": "f1", - "f2": "f2", - "f3": "f3", - "f4": "f4", - "f5": "f5", - "f6": "f6", - "f7": "f7", - "f8": "f8", - "f9": "f9", - "f10": "f10", - "f11": "f11", - "f12": "f12", - "-": "minus", - "=": "equals", - "[": "[", - "]": "]", - "\\": "\\", - ";": ";", - "'": "'", - ",": ",", - ".": ".", - "/": "/", - "`": "`", -} - - -def normalize_key(name: str) -> str: - key = name.strip().lower() - if key not in KEY_MAP: - raise ValueError(f"Unsupported key: {name}") - return KEY_MAP[key] - - -def json_output(payload: dict[str, Any]) -> None: - sys.stdout.write(json.dumps(payload, ensure_ascii=False)) - sys.stdout.write("\n") - sys.stdout.flush() - - -def error_output(message: str, code: str = "runtime_error") -> None: - json_output({"ok": False, "error": {"code": code, "message": message}}) - - -def bool_env(name: str, default: bool = False) -> bool: - value = os.environ.get(name) - if value is None: - return default - return value not in {"0", "false", "False", ""} - - -def run_osascript(script: str) -> str: - result = subprocess.run( - ["osascript", "-e", script], - text=True, - capture_output=True, - check=False, - ) - if result.returncode != 0: - raise RuntimeError(result.stderr.strip() or result.stdout.strip() or "osascript failed") - return result.stdout.strip() - - -def applescript_modifier(name: str) -> str: - if name == "command": - return "command down" - if name == "option": - return "option down" - if name == "shift": - return "shift down" - if name == "ctrl": - return "control down" - if name == "fn": - return "fn down" - raise ValueError(f"Unsupported AppleScript modifier: {name}") - - -def send_keystroke_via_osascript(character: str, modifiers: list[str] | None = None) -> None: - escaped = character.replace("\\", "\\\\").replace('"', '\\"') - if modifiers: - modifier_expr = ", ".join(applescript_modifier(m) for m in modifiers) - script = ( - 'tell application "System Events" to keystroke ' - f'"{escaped}" using {{{modifier_expr}}}' - ) - else: - script = f'tell application "System Events" to keystroke "{escaped}"' - run_osascript(script) - - -def get_displays() -> list[dict[str, Any]]: - max_displays = 32 - err, active, count = CGGetActiveDisplayList(max_displays, None, None) - if err != 0: - raise RuntimeError(f"CGGetActiveDisplayList failed: {err}") - displays: list[dict[str, Any]] = [] - main_id = CGMainDisplayID() - for idx, display_id in enumerate(active[:count]): - bounds = CGDisplayBounds(display_id) - mode = None - try: - from Quartz import CGDisplayCopyDisplayMode - mode = CGDisplayCopyDisplayMode(display_id) - except Exception: - mode = None - physical_width = int(CGDisplayPixelsWide(display_id)) - physical_height = int(CGDisplayPixelsHigh(display_id)) - logical_width = int(bounds.size.width) - logical_height = int(bounds.size.height) - if mode is not None: - mode_w = int(CGDisplayModeGetPixelWidth(mode)) - mode_h = int(CGDisplayModeGetPixelHeight(mode)) - physical_width = mode_w or physical_width - physical_height = mode_h or physical_height - scale_factor = physical_width / logical_width if logical_width else 1 - name = f"Display {idx + 1}" - displays.append( - { - "id": int(display_id), - "displayId": int(display_id), - "width": logical_width, - "height": logical_height, - "scaleFactor": scale_factor, - "originX": int(bounds.origin.x), - "originY": int(bounds.origin.y), - "isPrimary": bool(display_id == main_id or CGDisplayIsMain(display_id)), - "name": name, - "label": name, - } - ) - return displays - - -def choose_display(display_id: int | None) -> dict[str, Any]: - displays = get_displays() - if not displays: - raise RuntimeError("No active displays found") - if display_id is None: - for display in displays: - if display["isPrimary"]: - return display - return displays[0] - for display in displays: - if display["displayId"] == display_id or display["id"] == display_id: - return display - raise RuntimeError(f"Unknown display: {display_id}") - - -def ensure_screen_recording_permission() -> None: - """No-op: CGPreflightScreenCaptureAccess is unreliable for child processes - (returns False even when the parent app has TCC permission), and any actual - capture attempt triggers a macOS popup on newer versions. Let the actual - capture call handle errors instead.""" - pass - - -def capture_display(display_id: int | None, resize: tuple[int, int] | None = None) -> dict[str, Any]: - ensure_screen_recording_permission() - display = choose_display(display_id) - monitor = { - "left": display["originX"], - "top": display["originY"], - "width": display["width"], - "height": display["height"], - } - with mss.mss() as sct: - raw = sct.grab(monitor) - image = Image.frombytes("RGB", raw.size, raw.rgb) - if resize: - image = image.resize(resize, Image.Resampling.LANCZOS) - buffer = BytesIO() - image.save(buffer, format="JPEG", quality=75, optimize=True) - base64_data = base64.b64encode(buffer.getvalue()).decode("ascii") - return { - "base64": base64_data, - "width": image.width, - "height": image.height, - "displayWidth": display["width"], - "displayHeight": display["height"], - "displayId": display["displayId"], - "originX": display["originX"], - "originY": display["originY"], - "display": display, - } - - -def capture_region(region: dict[str, int], resize: tuple[int, int] | None = None) -> dict[str, Any]: - ensure_screen_recording_permission() - with mss.mss() as sct: - raw = sct.grab(region) - image = Image.frombytes("RGB", raw.size, raw.rgb) - if resize: - image = image.resize(resize, Image.Resampling.LANCZOS) - buffer = BytesIO() - image.save(buffer, format="JPEG", quality=75, optimize=True) - base64_data = base64.b64encode(buffer.getvalue()).decode("ascii") - return {"base64": base64_data, "width": image.width, "height": image.height} - - -def list_windows() -> list[dict[str, Any]]: - windows = CGWindowListCopyWindowInfo( - kCGWindowListOptionOnScreenOnly | kCGWindowListExcludeDesktopElements, - kCGNullWindowID, - ) - out: list[dict[str, Any]] = [] - for window in windows or []: - if int(window.get(kCGWindowLayer, 0)) != 0: - continue - if not bool(window.get(kCGWindowIsOnscreen, True)): - continue - bounds = window.get(kCGWindowBounds) or {} - width = int(bounds.get("Width", 0)) - height = int(bounds.get("Height", 0)) - if width <= 1 or height <= 1: - continue - out.append( - { - "ownerName": window.get(kCGWindowOwnerName, "") or "", - "title": window.get(kCGWindowName, "") or "", - "bounds": { - "x": int(bounds.get("X", 0)), - "y": int(bounds.get("Y", 0)), - "width": width, - "height": height, - }, - } - ) - return out - - -def bundle_id_to_app(bundle_id: str): - return NSWorkspace.sharedWorkspace().URLForApplicationWithBundleIdentifier_(bundle_id) - - -def installed_apps() -> list[dict[str, Any]]: - search_roots = [ - Path("/Applications"), - Path.home() / "Applications", - Path("/System/Applications"), - Path("/System/Applications/Utilities"), - ] - results: dict[str, dict[str, Any]] = {} - workspace = NSWorkspace.sharedWorkspace() - for root in search_roots: - if not root.exists(): - continue - for app in root.rglob("*.app"): - try: - bundle = workspace.bundleIdentifierForURL_(NSURL.fileURLWithPath_(str(app))) - except Exception: - bundle = None - if not bundle: - try: - url = workspace.URLForApplicationWithBundleIdentifier_(str(app)) - bundle = workspace.bundleIdentifierForURL_(url) if url else None - except Exception: - bundle = None - info_plist = app / "Contents/Info.plist" - display_name = app.stem - if info_plist.exists(): - try: - import plistlib - with info_plist.open("rb") as f: - plist = plistlib.load(f) - bundle = bundle or plist.get("CFBundleIdentifier") - display_name = plist.get("CFBundleDisplayName") or plist.get("CFBundleName") or display_name - except Exception: - pass - if not bundle or bundle in results: - continue - results[bundle] = { - "bundleId": str(bundle), - "displayName": str(display_name), - "path": str(app), - } - return sorted(results.values(), key=lambda item: item["displayName"].lower()) - - -def running_apps() -> list[dict[str, Any]]: - apps = [] - seen = set() - for app in NSWorkspace.sharedWorkspace().runningApplications() or []: - bundle_id = app.bundleIdentifier() - if not bundle_id or bundle_id in seen: - continue - seen.add(bundle_id) - name = app.localizedName() or bundle_id - apps.append({"bundleId": str(bundle_id), "displayName": str(name)}) - return sorted(apps, key=lambda item: item["displayName"].lower()) - - -def app_display_name(bundle_id: str) -> str | None: - for app in NSWorkspace.sharedWorkspace().runningApplications() or []: - if app.bundleIdentifier() == bundle_id: - return str(app.localizedName() or bundle_id) - for app in installed_apps(): - if app["bundleId"] == bundle_id: - return str(app["displayName"]) - return None - - -def frontmost_app() -> dict[str, str] | None: - app = NSWorkspace.sharedWorkspace().frontmostApplication() - if not app: - return None - bundle_id = app.bundleIdentifier() - if not bundle_id: - return None - return { - "bundleId": str(bundle_id), - "displayName": str(app.localizedName() or bundle_id), - } - - -def app_under_point(x: int, y: int) -> dict[str, str] | None: - point = CGPointMake(x, y) - running_by_name = { - str(app.localizedName() or app.bundleIdentifier()): str(app.bundleIdentifier()) - for app in NSWorkspace.sharedWorkspace().runningApplications() or [] - if app.bundleIdentifier() - } - for window in list_windows(): - bounds = window["bounds"] - rect = ((bounds["x"], bounds["y"]), (bounds["width"], bounds["height"])) - if CGRectContainsPoint(rect, point): - owner = window["ownerName"] - bundle = running_by_name.get(owner) - if bundle: - return {"bundleId": bundle, "displayName": str(owner)} - return frontmost_app() - - -def find_window_displays(bundle_ids: list[str]) -> list[dict[str, Any]]: - if not bundle_ids: - return [] - displays = get_displays() - names_by_bundle = { - bundle_id: app_display_name(bundle_id) or bundle_id for bundle_id in bundle_ids - } - windows = list_windows() - result = [] - for bundle_id in bundle_ids: - target_name = names_by_bundle.get(bundle_id) - display_ids: set[int] = set() - for window in windows: - owner = window["ownerName"] - if not owner: - continue - if target_name and owner != target_name: - continue - if not target_name and owner != bundle_id: - continue - wx = window["bounds"]["x"] - wy = window["bounds"]["y"] - ww = window["bounds"]["width"] - wh = window["bounds"]["height"] - window_rect = ((wx, wy), (ww, wh)) - for display in displays: - display_rect = ((display["originX"], display["originY"]), (display["width"], display["height"])) - intersection = CGRectIntersection(window_rect, display_rect) - if intersection.size.width > 0 and intersection.size.height > 0: - display_ids.add(int(display["displayId"])) - result.append({"bundleId": bundle_id, "displayIds": sorted(display_ids)}) - return result - - -def open_app(bundle_id: str) -> None: - url = bundle_id_to_app(bundle_id) - if not url: - raise RuntimeError(f"App not found for bundle identifier: {bundle_id}") - ok, err = NSWorkspace.sharedWorkspace().launchApplicationAtURL_options_configuration_error_(url, 0, {}, None) - if not ok: - raise RuntimeError(str(err) if err else f"Failed to open app {bundle_id}") - - -def read_clipboard() -> str: - pb = NSPasteboard.generalPasteboard() - value = pb.stringForType_(NSPasteboardTypeString) - return "" if value is None else str(value) - - -def write_clipboard(text: str) -> None: - pb = NSPasteboard.generalPasteboard() - pb.clearContents() - pb.setString_forType_(text, NSPasteboardTypeString) - - -def paste_clipboard() -> None: - send_keystroke_via_osascript("v", ["command"]) - - -def detect_screen_recording_permission() -> bool | None: - """Best-effort passive screen-recording probe with no system prompt. - - `CGPreflightScreenCaptureAccess()` is fast and explicit when it returns - True, but on child processes launched by a TCC-authorized app bundle it can - still return False. As a fallback, inspect the visible window list: Apple - only exposes other apps' window titles when Screen Recording access is - granted. If we can see at least one title, treat the permission as granted. - If we can inspect visible windows but every title is blank, treat it as not - granted. If window enumeration itself is unavailable, return None. - """ - - try: - if CGPreflightScreenCaptureAccess(): - return True - except Exception: - pass - - try: - windows = CGWindowListCopyWindowInfo( - kCGWindowListOptionOnScreenOnly | kCGWindowListExcludeDesktopElements, - kCGNullWindowID, - ) - except Exception: - return None - - eligible_windows = 0 - for window in windows or []: - if int(window.get(kCGWindowLayer, 0)) != 0: - continue - if not bool(window.get(kCGWindowIsOnscreen, True)): - continue - - bounds = window.get(kCGWindowBounds) or {} - width = int(bounds.get("Width", 0)) - height = int(bounds.get("Height", 0)) - if width <= 1 or height <= 1: - continue - - eligible_windows += 1 - if (window.get(kCGWindowName, "") or "").strip(): - return True - - if eligible_windows > 0: - return False - return None - - -def detect_accessibility_permission() -> bool: - """ - Use the official macOS Accessibility trust API. - - The previous System Events / AppleScript probe was too weak: it could - succeed even when the current helper process was not actually trusted for - input control, which led the desktop UI to report Accessibility as granted - while mouse/keyboard control still failed at runtime. - """ - framework_path = "/System/Library/Frameworks/ApplicationServices.framework/ApplicationServices" - try: - application_services = ctypes.CDLL(framework_path) - application_services.AXIsProcessTrusted.restype = ctypes.c_bool - application_services.AXIsProcessTrusted.argtypes = [] - return bool(application_services.AXIsProcessTrusted()) - except Exception: - # Fail closed: if the trust API can't be queried, treat accessibility - # as unavailable instead of reporting a misleading success state. - return False - - -def check_permissions() -> dict[str, bool | None]: - accessibility = detect_accessibility_permission() - screen_recording = detect_screen_recording_permission() - return { - "accessibility": accessibility, - "screenRecording": screen_recording, - } - - -def click(x: int, y: int, button: str, count: int, modifiers: list[str] | None) -> None: - pyautogui.moveTo(x, y) - if modifiers: - normalized = [normalize_key(m) for m in modifiers] - for key in normalized: - pyautogui.keyDown(key) - try: - pyautogui.click(x=x, y=y, button=button, clicks=count, interval=0.08) - finally: - for key in reversed(normalized): - pyautogui.keyUp(key) - else: - pyautogui.click(x=x, y=y, button=button, clicks=count, interval=0.08) - - -def scroll(x: int, y: int, delta_x: int, delta_y: int) -> None: - pyautogui.moveTo(x, y) - if delta_y: - pyautogui.scroll(int(delta_y), x=x, y=y) - if delta_x: - pyautogui.hscroll(int(delta_x), x=x, y=y) - - -def key_action(sequence: str, repeat: int = 1) -> None: - parts = [normalize_key(part) for part in sequence.split("+") if part.strip()] - for _ in range(max(1, repeat)): - if parts == ["command", "v"]: - paste_clipboard() - elif parts == ["command", "a"]: - send_keystroke_via_osascript("a", ["command"]) - elif parts == ["command", "c"]: - send_keystroke_via_osascript("c", ["command"]) - elif parts == ["command", "x"]: - send_keystroke_via_osascript("x", ["command"]) - elif len(parts) == 1: - pyautogui.press(parts[0]) - else: - pyautogui.hotkey(*parts, interval=0.02) - time.sleep(0.01) - - -def hold_keys(keys: list[str], duration_ms: int) -> None: - normalized = [normalize_key(k) for k in keys] - for key in normalized: - pyautogui.keyDown(key) - try: - time.sleep(max(duration_ms, 0) / 1000) - finally: - for key in reversed(normalized): - pyautogui.keyUp(key) - - -def type_text(text: str) -> None: - pyautogui.write(text, interval=0.008) - - -def main() -> int: - parser = argparse.ArgumentParser() - parser.add_argument("command") - parser.add_argument("--payload", default="{}") - args = parser.parse_args() - payload = json.loads(args.payload) - - try: - command = args.command - if command == "check_permissions": - perms = check_permissions() - json_output({"ok": True, "result": perms}) - return 0 - if command == "list_displays": - json_output({"ok": True, "result": get_displays()}) - return 0 - if command == "get_display_size": - json_output({"ok": True, "result": choose_display(payload.get("displayId"))}) - return 0 - if command == "screenshot": - resize = None - if payload.get("targetWidth") and payload.get("targetHeight"): - resize = (int(payload["targetWidth"]), int(payload["targetHeight"])) - result = capture_display(payload.get("displayId"), resize) - json_output({"ok": True, "result": result}) - return 0 - if command == "resolve_prepare_capture": - resize = None - if payload.get("targetWidth") and payload.get("targetHeight"): - resize = (int(payload["targetWidth"]), int(payload["targetHeight"])) - result = capture_display(payload.get("preferredDisplayId"), resize) - result["hidden"] = [] - result["resolvedDisplayId"] = result["displayId"] - json_output({"ok": True, "result": result}) - return 0 - if command == "zoom": - resize = None - if payload.get("targetWidth") and payload.get("targetHeight"): - resize = (int(payload["targetWidth"]), int(payload["targetHeight"])) - region = { - "left": int(payload["x"]), - "top": int(payload["y"]), - "width": int(payload["width"]), - "height": int(payload["height"]), - } - json_output({"ok": True, "result": capture_region(region, resize)}) - return 0 - if command == "prepare_for_action": - json_output({"ok": True, "result": []}) - return 0 - if command == "preview_hide_set": - json_output({"ok": True, "result": []}) - return 0 - if command == "find_window_displays": - json_output({"ok": True, "result": find_window_displays(list(payload.get("bundleIds") or []))}) - return 0 - if command == "key": - key_action(str(payload["keySequence"]), int(payload.get("repeat") or 1)) - json_output({"ok": True, "result": True}) - return 0 - if command == "hold_key": - hold_keys(list(payload.get("keyNames") or []), int(payload.get("durationMs") or 0)) - json_output({"ok": True, "result": True}) - return 0 - if command == "type": - type_text(str(payload.get("text") or "")) - json_output({"ok": True, "result": True}) - return 0 - if command == "click": - click(int(payload["x"]), int(payload["y"]), str(payload.get("button") or "left"), int(payload.get("count") or 1), payload.get("modifiers")) - json_output({"ok": True, "result": True}) - return 0 - if command == "drag": - from_point = payload.get("from") - if from_point: - pyautogui.moveTo(int(from_point["x"]), int(from_point["y"])) - pyautogui.dragTo(int(payload["to"]["x"]), int(payload["to"]["y"]), duration=0.2, button="left") - json_output({"ok": True, "result": True}) - return 0 - if command == "move_mouse": - pyautogui.moveTo(int(payload["x"]), int(payload["y"])) - json_output({"ok": True, "result": True}) - return 0 - if command == "scroll": - scroll(int(payload["x"]), int(payload["y"]), int(payload.get("deltaX") or 0), int(payload.get("deltaY") or 0)) - json_output({"ok": True, "result": True}) - return 0 - if command == "mouse_down": - pyautogui.mouseDown(button="left") - json_output({"ok": True, "result": True}) - return 0 - if command == "mouse_up": - pyautogui.mouseUp(button="left") - json_output({"ok": True, "result": True}) - return 0 - if command == "cursor_position": - x, y = pyautogui.position() - json_output({"ok": True, "result": {"x": int(x), "y": int(y)}}) - return 0 - if command == "frontmost_app": - json_output({"ok": True, "result": frontmost_app()}) - return 0 - if command == "app_under_point": - json_output({"ok": True, "result": app_under_point(int(payload["x"]), int(payload["y"]))}) - return 0 - if command == "list_installed_apps": - json_output({"ok": True, "result": installed_apps()}) - return 0 - if command == "list_running_apps": - json_output({"ok": True, "result": running_apps()}) - return 0 - if command == "open_app": - open_app(str(payload["bundleId"])) - json_output({"ok": True, "result": True}) - return 0 - if command == "read_clipboard": - json_output({"ok": True, "result": read_clipboard()}) - return 0 - if command == "write_clipboard": - write_clipboard(str(payload.get("text") or "")) - json_output({"ok": True, "result": True}) - return 0 - if command == "paste_clipboard": - paste_clipboard() - json_output({"ok": True, "result": True}) - return 0 - error_output(f"Unknown command: {command}", code="bad_command") - return 2 - except Exception as exc: - error_output(str(exc)) - return 1 - - -if __name__ == "__main__": - raise SystemExit(main()) diff --git a/runtime/requirements.txt b/runtime/requirements.txt deleted file mode 100644 index 41b08e5d..00000000 --- a/runtime/requirements.txt +++ /dev/null @@ -1,6 +0,0 @@ -mss>=9.0.2,<10 -Pillow>=11.3.0,<12 -pyautogui>=0.9.54 -pyobjc-core>=11.1 -pyobjc-framework-Cocoa>=11.1 -pyobjc-framework-Quartz>=11.1 diff --git a/runtime/test_helpers.py b/runtime/test_helpers.py index 540d754c..9816d3f8 100644 --- a/runtime/test_helpers.py +++ b/runtime/test_helpers.py @@ -1,42 +1,46 @@ #!/usr/bin/env python3 -"""Cross-platform tests for mac_helper.py and win_helper.py. +"""Tests for win_helper.py. -Tests the platform-independent parts (JSON protocol, key mapping, capture logic) -without requiring platform-specific dependencies. Can run on any OS with pytest. +macOS routes every Computer Use command to the signed native `cu-helper` +daemon — `helperBridge` refuses to fall back to Python — so `mac_helper.py` +was unreachable and has been deleted. This file therefore covers the Windows +helper only. + +Most tests here are static (they read the source) rather than executed, +because the runtime deps (pywin32, pyautogui, mss) are Windows-only and CI +runs on macOS. Static coverage is enough for what actually regresses: the +guards getting dropped, inverted, or quietly bypassed. Usage: python -m pytest runtime/test_helpers.py -v - # or simply: python runtime/test_helpers.py """ from __future__ import annotations +import ast import json import subprocess import sys import unittest from pathlib import Path -from unittest.mock import patch, MagicMock -# Determine which helper to test based on current platform IS_WINDOWS = sys.platform == "win32" -IS_MACOS = sys.platform == "darwin" RUNTIME_DIR = Path(__file__).parent -MAC_HELPER = RUNTIME_DIR / "mac_helper.py" WIN_HELPER = RUNTIME_DIR / "win_helper.py" +CURSOR_BADGE = RUNTIME_DIR / "win_cursor_badge.py" + + +def _win_source() -> str: + return WIN_HELPER.read_text(encoding="utf-8") class TestKeyMap(unittest.TestCase): - """Test the KEY_MAP and normalize_key function — platform-independent logic.""" + """KEY_MAP translates macOS key names to Windows ones.""" def _load_key_map(self, helper_path: Path) -> dict[str, str]: - """Extract KEY_MAP from a helper by importing it with mocked deps.""" - # Read the file and extract just the KEY_MAP dict - source = helper_path.read_text() - # Find KEY_MAP definition + source = helper_path.read_text(encoding="utf-8") start = source.index("KEY_MAP = {") - # Find the matching closing brace depth = 0 for i, ch in enumerate(source[start:], start): if ch == "{": @@ -46,106 +50,40 @@ class TestKeyMap(unittest.TestCase): if depth == 0: end = i + 1 break - key_map_source = source[start:end] ns: dict = {} - exec(key_map_source, ns) + exec(source[start:end], ns) return ns["KEY_MAP"] - def test_mac_key_map_exists(self): - if not MAC_HELPER.exists(): - self.skipTest("mac_helper.py not found") - km = self._load_key_map(MAC_HELPER) - self.assertIn("cmd", km) - self.assertIn("ctrl", km) - self.assertEqual(km["cmd"], "command") - self.assertEqual(km["alt"], "option") - def test_win_key_map_exists(self): - if not WIN_HELPER.exists(): - self.skipTest("win_helper.py not found") km = self._load_key_map(WIN_HELPER) self.assertIn("cmd", km) self.assertIn("ctrl", km) - # Windows maps cmd/command/meta to 'win' key + # The mapping that matters: a model trained on macOS emits "cmd", and + # on Windows that has to become "win", not silently stay "cmd". self.assertEqual(km["cmd"], "win") - self.assertEqual(km["command"], "win") - self.assertEqual(km["meta"], "win") - # Windows maps alt/option to 'alt' - self.assertEqual(km["alt"], "alt") - self.assertEqual(km["option"], "alt") - - def test_common_keys_present_in_both(self): - """Both helpers must have the same set of key names.""" - if not MAC_HELPER.exists() or not WIN_HELPER.exists(): - self.skipTest("Both helpers required") - mac_km = self._load_key_map(MAC_HELPER) - win_km = self._load_key_map(WIN_HELPER) - # All keys in mac should be in win and vice versa - self.assertEqual(set(mac_km.keys()), set(win_km.keys()), - "KEY_MAP keys must be identical across platforms") def test_all_alphabet_keys(self): - """All a-z keys should map to themselves.""" - for helper in [MAC_HELPER, WIN_HELPER]: - if not helper.exists(): - continue - km = self._load_key_map(helper) - for char in "abcdefghijklmnopqrstuvwxyz": - self.assertEqual(km[char], char, f"{helper.name}: {char} should map to itself") + km = self._load_key_map(WIN_HELPER) + for ch in "abcdefghijklmnopqrstuvwxyz": + self.assertIn(ch, km) def test_all_digit_keys(self): - """All 0-9 keys should map to themselves.""" - for helper in [MAC_HELPER, WIN_HELPER]: - if not helper.exists(): - continue - km = self._load_key_map(helper) - for digit in "0123456789": - self.assertEqual(km[digit], digit, f"{helper.name}: {digit} should map to itself") - - def test_function_keys(self): - """F1-F12 should map to themselves.""" - for helper in [MAC_HELPER, WIN_HELPER]: - if not helper.exists(): - continue - km = self._load_key_map(helper) - for i in range(1, 13): - key = f"f{i}" - self.assertEqual(km[key], key, f"{helper.name}: {key} should map to itself") + km = self._load_key_map(WIN_HELPER) + for d in "0123456789": + self.assertIn(d, km) class TestJSONProtocol(unittest.TestCase): - """Test that both helpers follow the same JSON command protocol.""" - - def _get_helper(self) -> Path: - """Get the appropriate helper for the current platform.""" - if IS_WINDOWS and WIN_HELPER.exists(): - return WIN_HELPER - if IS_MACOS and MAC_HELPER.exists(): - return MAC_HELPER - return MAC_HELPER if MAC_HELPER.exists() else WIN_HELPER - def _parse_main_commands(self, helper_path: Path) -> list[str]: - """Extract all command names from the main() dispatcher.""" - source = helper_path.read_text() + source = helper_path.read_text(encoding="utf-8") commands = [] for line in source.splitlines(): stripped = line.strip() if stripped.startswith('if command == "'): - cmd = stripped.split('"')[1] - commands.append(cmd) + commands.append(stripped.split('"')[1]) return commands - def test_both_helpers_same_commands(self): - """Both helpers must support the exact same set of commands.""" - if not MAC_HELPER.exists() or not WIN_HELPER.exists(): - self.skipTest("Both helpers required") - mac_cmds = set(self._parse_main_commands(MAC_HELPER)) - win_cmds = set(self._parse_main_commands(WIN_HELPER)) - self.assertEqual(mac_cmds, win_cmds, - f"Command sets differ.\nOnly in mac: {mac_cmds - win_cmds}\nOnly in win: {win_cmds - mac_cmds}") - def test_expected_commands_exist(self): - """Core commands should be present in each helper.""" expected = { "check_permissions", "list_displays", "get_display_size", "screenshot", "resolve_prepare_capture", "zoom", @@ -156,167 +94,304 @@ class TestJSONProtocol(unittest.TestCase): "list_installed_apps", "list_running_apps", "open_app", "read_clipboard", "write_clipboard", "paste_clipboard", } - for helper in [MAC_HELPER, WIN_HELPER]: - if not helper.exists(): - continue - cmds = set(self._parse_main_commands(helper)) - missing = expected - cmds - self.assertFalse(missing, - f"{helper.name} missing commands: {missing}") + cmds = set(self._parse_main_commands(WIN_HELPER)) + self.assertFalse(expected - cmds, + f"win_helper.py missing commands: {expected - cmds}") + @unittest.skipUnless(IS_WINDOWS, "requires Windows runtime deps") def test_unknown_command_returns_error(self): - """Running a non-existent command should return a JSON error.""" - helper = self._get_helper() - if not helper.exists(): - self.skipTest("No helper found") - # On macOS without venv, mac_helper.py may fail at import (AppKit); - # on Windows without venv, win_helper.py may fail at import (win32gui). - # Only test if the helper can actually import. - check = subprocess.run( - [sys.executable, "-c", f"import importlib.util; " - f"spec = importlib.util.spec_from_file_location('h', '{helper}')"], - capture_output=True, text=True - ) result = subprocess.run( - [sys.executable, str(helper), "nonexistent_command_xyz"], - capture_output=True, text=True + [sys.executable, str(WIN_HELPER), "nonexistent_command_xyz"], + capture_output=True, text=True, ) if result.returncode == 1 and not result.stdout.strip(): - # Import failed — platform deps missing, skip this test - self.skipTest(f"Cannot run {helper.name} on this platform (missing deps)") - # Should exit with code 2 + self.skipTest("missing platform deps") self.assertEqual(result.returncode, 2) parsed = json.loads(result.stdout.strip()) self.assertFalse(parsed["ok"]) self.assertEqual(parsed["error"]["code"], "bad_command") -class TestHelperOutputFormat(unittest.TestCase): - """Test the JSON output helpers are consistent.""" +class TestMutatingCommandsAreGuarded(unittest.TestCase): + """Every command that injects input must pass through the guards. - def test_json_output_function_exists(self): - """Both helpers should define json_output and error_output.""" - for helper in [MAC_HELPER, WIN_HELPER]: - if not helper.exists(): + These are static-source tests on purpose. The failure being guarded against + is someone adding an eleventh mutating verb and wiring it like the ten that + came before — at which point it silently has no lease and no reachability + check. A runtime test would need Windows and would only cover the verbs it + thought to enumerate; reading the dispatcher catches the new one. + """ + + # Kept as a literal, deliberately duplicating MUTATING_COMMANDS in the + # helper. If the two drift the test fails, which is the point: the set is + # a security boundary and should not be edited casually on one side only. + MUTATING = { + "click", "drag", "move_mouse", "scroll", + "mouse_down", "mouse_up", + "key", "hold_key", "type", + "paste_clipboard", + } + + def _module_constant(self, name: str) -> set[str]: + """Read a module-level frozenset/set constant without importing.""" + tree = ast.parse(_win_source()) + for node in tree.body: + if isinstance(node, ast.Assign): + for target in node.targets: + if isinstance(target, ast.Name) and target.id == name: + return set(ast.literal_eval( + node.value.args[0] + if isinstance(node.value, ast.Call) + else node.value + )) + raise AssertionError(f"{name} not found in win_helper.py") + + def test_mutating_command_set_matches_this_test(self): + self.assertEqual(self._module_constant("MUTATING_COMMANDS"), self.MUTATING) + + def test_coordinate_commands_are_a_subset(self): + coords = self._module_constant("COORDINATE_COMMANDS") + self.assertTrue(coords <= self.MUTATING) + # `key`/`type` go wherever focus is and have no point to validate. + # Asserting their absence keeps someone from "fixing" the coordinate + # guard by adding them and then dereferencing an x/y that isn't there. + self.assertNotIn("key", coords) + self.assertNotIn("type", coords) + + def test_every_mutating_branch_finalizes_the_lease(self): + """No mutating branch may answer with a bare json_output. + + This is the specific regression: `_finish` is what runs the post-action + interference check, so a branch that writes its own success response + reports "Action completed" for input that may have collided with the + user's own typing. + """ + source = _win_source() + start = source.index(' try:\n command = args.command') + end = source.index(' error_output(f"Unknown command: {command}"') + dispatcher = source[start:end] + + blocks = dispatcher.split('if command == "') + for block in blocks[1:]: + name = block.split('"')[0] + if name not in self.MUTATING: continue - source = helper.read_text() - self.assertIn("def json_output(", source, - f"{helper.name} missing json_output function") - self.assertIn("def error_output(", source, - f"{helper.name} missing error_output function") + body = block.split("if command ==")[0] + self.assertIn( + "_finish(lease,", body, + f'"{name}" must return through _finish so the lease is checked', + ) + self.assertNotIn( + 'json_output({"ok": True', body, + f'"{name}" writes its own success response, bypassing the lease', + ) - def test_main_entry_point(self): - """Both helpers should have the standard main entry point.""" - for helper in [MAC_HELPER, WIN_HELPER]: - if not helper.exists(): - continue - source = helper.read_text() - self.assertIn('if __name__ == "__main__":', source, - f"{helper.name} missing __main__ guard") - self.assertIn("def main()", source, - f"{helper.name} missing main() function") + def test_guards_run_before_any_injection(self): + """acquire() must precede the dispatch chain, not follow it.""" + source = _win_source() + acquire = source.index("lease.acquire()") + first_branch = source.index(' if command == "check_permissions"') + self.assertLess( + acquire, first_branch, + "the lease must be acquired before any command branch runs", + ) -class TestWinHelperPermissions(unittest.TestCase): - """Windows-specific: permissions should always return True.""" +class TestInterferenceDetection(unittest.TestCase): + def test_uses_getlastinputinfo_not_an_event_hook(self): + """The signal must stay permission-free. + A low-level input hook would read the same events, but installing one + is exactly the kind of thing that gets an app flagged, and it is not + needed: GetLastInputInfo answers the only question we ask. + """ + source = _win_source() + self.assertIn("GetLastInputInfo", source) + self.assertNotIn("SetWindowsHookEx", source) + + def test_distinguishes_did_not_run_from_outcome_unknown(self): + """The two interference verdicts must stay distinct. + + Collapsing them is a real hazard: `user_interference` means nothing + happened and a retry is safe, while `user_interference_result_unknown` + means input already went out and a retry could double-apply it. On a + play/pause toggle those differ by exactly one wrong outcome. + """ + source = _win_source() + self.assertIn('"user_interference"', source) + self.assertIn('user_interference_result_unknown', source) + + acquire_start = source.index(" def acquire(self)") + acquire_body = source[acquire_start:source.index(" def finalize(self)")] + self.assertNotIn("result_unknown", acquire_body, + "a pre-action refusal means nothing ran; the outcome is known") + + finalize_body = source[source.index(" def finalize(self)"):] + finalize_body = finalize_body[:finalize_body.index("\n\n\n")] + self.assertIn("result_unknown", finalize_body, + "post-action interference leaves the outcome unknown") + + def test_counter_read_failure_does_not_block_actions(self): + """An unreadable counter must fail open, not brick the feature. + + Precedent from the macOS side: an earlier build required an Input + Monitoring grant that onboarding never asked for, so every mutating + action failed on a correctly set-up machine. A safety layer that turns + the product off is not safety. + """ + source = _win_source() + fn_start = source.index("def last_physical_input_tick()") + body = source[fn_start:source.index("class UserInterference")] + self.assertIn("return 0", body) + # And the comparisons must treat 0 as "no reading", never as a tick + # value that happens to differ from the next one. + self.assertIn("if before and after and before != after:", source) + + def test_synthetic_input_must_not_trip_the_detector(self): + """Documented invariant: SendInput does not advance GetLastInputInfo. + + If this ever stopped holding, every agent action would abort itself and + the feature would look randomly broken. Pinning the claim in a test + keeps it from being quietly deleted as a stale comment. + """ + source = _win_source() + self.assertIn("is NOT advanced by", source) + + +class TestDeliveryGuards(unittest.TestCase): + def test_offscreen_point_is_refused(self): + source = _win_source() + self.assertIn("point_outside_display", source) + self.assertIn("def ensure_point_on_screen", source) + + def test_unreachable_window_is_refused(self): + source = _win_source() + self.assertIn("target_window_offscreen", source) + self.assertIn("def ensure_target_window_reachable", source) + + def test_reachability_check_sees_minimized_windows(self): + """It must NOT reuse list_windows(). + + `list_windows()` filters out invisible and zero-area windows — exactly + the states the guard needs to observe in order to refuse. An earlier + draft of this guard did reuse it, matched nothing, and passed + everything. The enumeration has to be its own. + """ + tree = ast.parse(_win_source()) + fn = next( + node for node in ast.walk(tree) + if isinstance(node, ast.FunctionDef) and node.name == "_windows_for_bundle" + ) + # Walk the AST rather than the text, so the explanatory docstring + # (which names list_windows to say why it is NOT used) cannot satisfy + # or break the assertion. + called = { + n.func.id for n in ast.walk(fn) + if isinstance(n, ast.Call) and isinstance(n.func, ast.Name) + } + self.assertNotIn("list_windows", called) + + source = _win_source() + body = source[source.index("def _windows_for_bundle"): + source.index("def ensure_target_window_reachable")] + self.assertIn("EnumWindows", body) + self.assertIn("SW_SHOWMINIMIZED", source) + + def test_refusals_carry_a_machine_readable_code(self): + source = _win_source() + self.assertIn("class DeliveryRefused", source) + self.assertIn("error_output(str(exc), code=exc.code)", source) + + def test_refusal_says_the_action_was_not_sent(self): + """The message must state that nothing happened. + + "Could not reach the window" reads like a warning attached to an action + that still went out. The model needs to know the action did not happen, + or it will assume it did and move on. + """ + source = _win_source() + self.assertIn("was NOT sent", source) + + +class TestCursorBadge(unittest.TestCase): + """The Windows badge annotates the real cursor; it does not replace it.""" + + def test_badge_script_exists(self): + self.assertTrue(CURSOR_BADGE.exists()) + + def test_badge_is_click_through_and_never_takes_focus(self): + """Any of these missing turns the badge into an obstacle. + + Without WS_EX_TRANSPARENT it eats the clicks it is meant to describe; + without WS_EX_NOACTIVATE it steals focus from the app being driven — + which would break the very action it is annotating. + """ + source = CURSOR_BADGE.read_text(encoding="utf-8") + tree = ast.parse(source) + + create = next( + node for node in ast.walk(tree) + if isinstance(node, ast.FunctionDef) and node.name == "create" + ) + # Read the names actually combined into the window's ex-style, not + # merely the ones defined somewhere in the file. A constant can be + # defined and then left out of CreateWindowExW — which is exactly how + # a click-through window quietly becomes a click-eating one. + used = { + n.id for n in ast.walk(create) + if isinstance(n, ast.Name) + } + for style in ("WS_EX_LAYERED", "WS_EX_TRANSPARENT", + "WS_EX_NOACTIVATE", "WS_EX_TOOLWINDOW"): + self.assertIn( + style, used, + f"{style} must be passed to CreateWindowExW, not just defined", + ) + self.assertIn("SW_SHOWNOACTIVATE", used) + + def test_badge_does_not_draw_a_second_pointer(self): + """Windows has one real cursor and SendInput moves it. + + Drawing a fake pointer alongside it would show the user two cursors, + one of which is a lie about where the click will land. The macOS design + does not transfer, and the source says so explicitly. + """ + source = CURSOR_BADGE.read_text(encoding="utf-8") + self.assertIn("annotation", source.lower()) + + def test_badge_exits_with_its_parent(self): + """An orphaned badge is worse than none. + + It would sit on screen claiming the agent is controlling the mouse + after the agent is gone. Tying it to stdin covers the parent being + killed, not just exiting cleanly. + """ + source = CURSOR_BADGE.read_text(encoding="utf-8") + self.assertIn("stdin", source) + + +class TestPermissions(unittest.TestCase): def test_check_permissions_always_granted(self): - """On Windows, permissions are not needed — should always be True.""" - if not WIN_HELPER.exists(): - self.skipTest("win_helper.py not found") - - # Extract and exec just the check_permissions function - source = WIN_HELPER.read_text() - - # Find the function - self.assertIn("def check_permissions()", source) - - # The function should return both as True - # We can verify by reading the source + """Windows has no TCC equivalent for input injection or capture.""" + source = _win_source() start = source.index("def check_permissions()") - # Find next def or end - rest = source[start:] - lines = rest.split("\n") - func_lines = [lines[0]] - for line in lines[1:]: - if line and not line[0].isspace() and not line.startswith("#"): - break - func_lines.append(line) - func_source = "\n".join(func_lines) - self.assertIn('"accessibility": True', func_source) - self.assertIn('"screenRecording": True', func_source) + body = source[start:start + 400] + self.assertIn('"accessibility": True', body) + self.assertIn('"screenRecording": True', body) -class TestMacHelperPermissions(unittest.TestCase): - """macOS helper permission detection should use the official trust API.""" +class TestSourceIntegrity(unittest.TestCase): + def test_helper_parses(self): + ast.parse(_win_source()) - def test_check_permissions_uses_ax_api_instead_of_system_events(self): - if not MAC_HELPER.exists(): - self.skipTest("mac_helper.py not found") + def test_badge_parses(self): + ast.parse(CURSOR_BADGE.read_text(encoding="utf-8")) - source = MAC_HELPER.read_text() - - self.assertIn("def detect_accessibility_permission()", source) - self.assertIn("AXIsProcessTrusted", source) - - start = source.index("def check_permissions()") - rest = source[start:] - lines = rest.split("\n") - func_lines = [lines[0]] - for line in lines[1:]: - if line and not line[0].isspace() and not line.startswith("#"): - break - func_lines.append(line) - func_source = "\n".join(func_lines) - - self.assertIn("detect_accessibility_permission()", func_source) - self.assertNotIn('tell application "System Events"', func_source) - - def test_clipboard_shortcuts_use_osascript_path(self): - if not MAC_HELPER.exists(): - self.skipTest("mac_helper.py not found") - - source = MAC_HELPER.read_text() - self.assertIn("def paste_clipboard()", source) - self.assertIn('send_keystroke_via_osascript("v", ["command"])', source) - self.assertIn('if parts == ["command", "v"]:', source) - self.assertIn('elif parts == ["command", "a"]:', source) - - -class TestCrossPlatformFunctions(unittest.TestCase): - """Test functions that are identical between both helpers.""" - - def _get_function_body(self, helper_path: Path, func_name: str) -> str: - """Extract a function's body (code lines only, no comments/blanks).""" - source = helper_path.read_text() - marker = f"def {func_name}(" - if marker not in source: - return "" - start = source.index(marker) - rest = source[start:] - lines = rest.split("\n") - func_lines = [lines[0]] - for line in lines[1:]: - # Stop at next top-level def/class or non-indented non-empty line - stripped = line.strip() - if line and not line[0].isspace() and stripped and not stripped.startswith("#"): - break - # Skip comments and blank lines for comparison - if stripped.startswith("#") or not stripped: - continue - func_lines.append(line) - return " ".join(" ".join(func_lines).split()) - - def test_input_functions_identical(self): - """Input action functions (click, scroll, etc.) should be identical.""" - if not MAC_HELPER.exists() or not WIN_HELPER.exists(): - self.skipTest("Both helpers required") - for func in ["click", "scroll", "hold_keys", "type_text"]: - mac_src = self._get_function_body(MAC_HELPER, func) - win_src = self._get_function_body(WIN_HELPER, func) - self.assertEqual(mac_src, win_src, - f"{func} should be identical across platforms") + def test_retired_mac_helper_is_not_referenced(self): + """macOS is native-only; a lingering reference invites a false fallback.""" + self.assertFalse((RUNTIME_DIR / "mac_helper.py").exists()) + self.assertNotIn("mac_helper", _win_source()) if __name__ == "__main__": - unittest.main() + unittest.main(verbosity=2) diff --git a/runtime/win_cursor_badge.py b/runtime/win_cursor_badge.py new file mode 100644 index 00000000..c7f16a34 --- /dev/null +++ b/runtime/win_cursor_badge.py @@ -0,0 +1,253 @@ +#!/usr/bin/env python3 +"""Windows agent-activity badge — a click-through marker that follows the cursor. + +WHY THIS IS NOT THE macOS VIRTUAL CURSOR +---------------------------------------- +On macOS the helper never moves the real pointer: `CGEvent.postToPid` carries +the click coordinate as metadata, so the drawn cursor IS the only cursor the +user sees move. It is a *replacement*. + +Windows has no per-process event delivery. `pyautogui` bottoms out in +`SendInput`, which warps the one real cursor the user's hand is also on. We +cannot avoid that, so drawing a second fake pointer would be actively harmful: +two pointers, one of them a lie, with no way to tell which one the OS is +actually going to click with. + +So this badge is an *annotation*, not a replacement. It rides just off the real +cursor and answers exactly one question the user cannot otherwise answer: +"is this thing moving because of me, or because of the agent?" On Windows that +question has real stakes — the agent is holding the user's mouse, and the user +needs to know before they grab it back mid-action. + +Runs as its own process because the Windows helper is a stateless one-shot CLI: +every command exits, so nothing in it can own a window across actions. + +The window is WS_EX_LAYERED | WS_EX_TRANSPARENT | WS_EX_NOACTIVATE — it never +takes focus, never appears in the taskbar or Alt-Tab, and passes every click +through to whatever is underneath. It cannot intercept the input it exists to +describe. + +Usage: + python win_cursor_badge.py --label "Claude" # runs until stdin closes +""" +from __future__ import annotations + +import argparse +import ctypes +import sys +import threading +from ctypes import wintypes + +user32 = ctypes.windll.user32 +gdi32 = ctypes.windll.gdi32 +kernel32 = ctypes.windll.kernel32 + +WS_EX_LAYERED = 0x00080000 +WS_EX_TRANSPARENT = 0x00000020 +WS_EX_TOPMOST = 0x00000008 +WS_EX_TOOLWINDOW = 0x00000080 +WS_EX_NOACTIVATE = 0x08000000 +WS_POPUP = 0x80000000 + +SW_SHOWNOACTIVATE = 4 +HWND_TOPMOST = -1 +SWP_NOACTIVATE = 0x0010 +SWP_NOSIZE = 0x0001 +SWP_NOZORDER = 0x0004 + +LWA_COLORKEY = 0x00000001 +LWA_ALPHA = 0x00000002 + +WM_DESTROY = 0x0002 +WM_PAINT = 0x000F + +BADGE_W = 132 +BADGE_H = 30 +CURSOR_OFFSET_X = 18 +CURSOR_OFFSET_Y = 18 + +# Chroma key: pixels of this exact colour become fully transparent. Picked to +# be a colour nothing in the badge draws, so only the intended shape shows. +TRANSPARENT_KEY = 0x00FF00FF + + +class POINT(ctypes.Structure): + _fields_ = [("x", wintypes.LONG), ("y", wintypes.LONG)] + + +class RECT(ctypes.Structure): + _fields_ = [ + ("left", wintypes.LONG), + ("top", wintypes.LONG), + ("right", wintypes.LONG), + ("bottom", wintypes.LONG), + ] + + +class PAINTSTRUCT(ctypes.Structure): + _fields_ = [ + ("hdc", wintypes.HDC), + ("fErase", wintypes.BOOL), + ("rcPaint", RECT), + ("fRestore", wintypes.BOOL), + ("fIncUpdate", wintypes.BOOL), + ("rgbReserved", ctypes.c_byte * 32), + ] + + +class WNDCLASS(ctypes.Structure): + _fields_ = [ + ("style", wintypes.UINT), + ("lpfnWndProc", ctypes.WINFUNCTYPE( + ctypes.c_long, wintypes.HWND, wintypes.UINT, + wintypes.WPARAM, wintypes.LPARAM)), + ("cbClsExtra", ctypes.c_int), + ("cbWndExtra", ctypes.c_int), + ("hInstance", wintypes.HINSTANCE), + ("hIcon", wintypes.HICON), + ("hCursor", wintypes.HANDLE), + ("hbrBackground", wintypes.HBRUSH), + ("lpszMenuName", wintypes.LPCWSTR), + ("lpszClassName", wintypes.LPCWSTR), + ] + + +WNDPROC = ctypes.WINFUNCTYPE( + ctypes.c_long, wintypes.HWND, wintypes.UINT, wintypes.WPARAM, wintypes.LPARAM +) + + +class CursorBadge: + def __init__(self, label: str) -> None: + self.label = label + self.hwnd: int | None = None + self._stop = threading.Event() + # Held on the instance because ctypes does not keep the trampoline + # alive on its own; letting it be collected turns the next window + # message into a crash inside the message pump. + self._wndproc = WNDPROC(self._on_message) + + def _on_message(self, hwnd, msg, wparam, lparam): + if msg == WM_PAINT: + self._paint(hwnd) + return 0 + if msg == WM_DESTROY: + user32.PostQuitMessage(0) + return 0 + return user32.DefWindowProcW(hwnd, msg, wparam, lparam) + + def _paint(self, hwnd: int) -> None: + ps = PAINTSTRUCT() + hdc = user32.BeginPaint(hwnd, ctypes.byref(ps)) + try: + rect = RECT(0, 0, BADGE_W, BADGE_H) + + # Fill with the chroma key first: everything we do not draw over + # becomes transparent, which is what gives the badge its shape. + key_brush = gdi32.CreateSolidBrush(TRANSPARENT_KEY) + user32.FillRect(hdc, ctypes.byref(rect), key_brush) + gdi32.DeleteObject(key_brush) + + body = RECT(0, 0, BADGE_W, BADGE_H) + bg = gdi32.CreateSolidBrush(0x00734B23) # BGR: a muted blue + user32.FillRect(hdc, ctypes.byref(body), bg) + gdi32.DeleteObject(bg) + + gdi32.SetBkMode(hdc, 1) # TRANSPARENT + gdi32.SetTextColor(hdc, 0x00FFFFFF) + text = f" {self.label} is controlling" + user32.DrawTextW( + hdc, text, len(text), ctypes.byref(body), + 0x00000004 | 0x00000100, # DT_VCENTER | DT_SINGLELINE + ) + finally: + user32.EndPaint(hwnd, ctypes.byref(ps)) + + def create(self) -> None: + hinst = kernel32.GetModuleHandleW(None) + class_name = "CcHahaAgentCursorBadge" + + wc = WNDCLASS() + wc.lpfnWndProc = self._wndproc + wc.hInstance = hinst + wc.lpszClassName = class_name + wc.hbrBackground = 0 + wc.hCursor = 0 + user32.RegisterClassW(ctypes.byref(wc)) + + self.hwnd = user32.CreateWindowExW( + WS_EX_LAYERED | WS_EX_TRANSPARENT | WS_EX_TOPMOST + | WS_EX_TOOLWINDOW | WS_EX_NOACTIVATE, + class_name, None, WS_POPUP, + 0, 0, BADGE_W, BADGE_H, + None, None, hinst, None, + ) + if not self.hwnd: + raise OSError("CreateWindowExW failed for the cursor badge") + + user32.SetLayeredWindowAttributes( + self.hwnd, TRANSPARENT_KEY, 225, LWA_COLORKEY | LWA_ALPHA + ) + user32.ShowWindow(self.hwnd, SW_SHOWNOACTIVATE) + + def _follow_cursor(self) -> None: + """Reposition the badge next to the real pointer, ~60fps.""" + pt = POINT() + while not self._stop.is_set(): + try: + if user32.GetCursorPos(ctypes.byref(pt)) and self.hwnd: + user32.SetWindowPos( + self.hwnd, HWND_TOPMOST, + pt.x + CURSOR_OFFSET_X, pt.y + CURSOR_OFFSET_Y, + 0, 0, SWP_NOACTIVATE | SWP_NOSIZE, + ) + except Exception: + # The badge is advisory. It must never be the reason an action + # fails, so every error here is swallowed and the loop retries. + pass + self._stop.wait(0.016) + + def _wait_for_stdin_close(self) -> None: + """Exit when the parent goes away. + + The badge outliving its parent would leave a permanent 'the agent is + controlling your mouse' claim on screen with nothing behind it. Reading + stdin to EOF ties this process's lifetime to the parent's, including + the case where the parent is killed rather than exiting cleanly. + """ + try: + for _ in sys.stdin: + pass + except Exception: + pass + self.stop() + + def stop(self) -> None: + self._stop.set() + if self.hwnd: + user32.PostMessageW(self.hwnd, WM_DESTROY, 0, 0) + + def run(self) -> int: + self.create() + threading.Thread(target=self._follow_cursor, daemon=True).start() + threading.Thread(target=self._wait_for_stdin_close, daemon=True).start() + + msg = wintypes.MSG() + while user32.GetMessageW(ctypes.byref(msg), None, 0, 0) > 0: + user32.TranslateMessage(ctypes.byref(msg)) + user32.DispatchMessageW(ctypes.byref(msg)) + return 0 + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--label", default="Claude") + args = parser.parse_args() + if sys.platform != "win32": + print("win_cursor_badge.py is Windows-only", file=sys.stderr) + return 1 + return CursorBadge(args.label).run() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/runtime/win_helper.py b/runtime/win_helper.py index 9850be34..12fb7935 100644 --- a/runtime/win_helper.py +++ b/runtime/win_helper.py @@ -1,8 +1,28 @@ #!/usr/bin/env python3 -"""Windows Computer Use helper — same JSON protocol as mac_helper.py. +"""Windows Computer Use helper. -Uses win32gui / win32api / win32process / psutil / pyperclip / screeninfo -to replicate macOS-specific Quartz/AppKit functionality on Windows. +Uses win32gui / win32api / win32process / psutil / pyperclip / screeninfo / +pyautogui to provide, on Windows, the JSON command protocol the native macOS +`cu-helper` daemon speaks. macOS is native-only — there is no Python path there +— so this is the sole implementation of that protocol in Python. + +One difference is not an implementation detail and shapes everything below: +macOS delivers input with `CGEvent.postToPid`, straight into the target +process, leaving the real cursor and the foreground app alone. Windows has no +equivalent. `pyautogui` bottoms out in `SendInput`, which injects into the one +system-wide input stream and warps the one real cursor. The agent therefore +shares the mouse and keyboard with the user, and cannot verify that anything +it sent arrived. + +Hence the two mechanisms that have no macOS counterpart: + + * `ForegroundLease` aborts when physical input overlaps an action, because + interleaved streams produce clicks neither party intended. + * `ensure_point_on_screen` / `ensure_target_window_reachable` refuse to send + at all when delivery is already known to be impossible. + +Both exist because `SendInput` reports success unconditionally, and "Action +completed" for input that went nowhere is worse than an error. """ from __future__ import annotations @@ -176,7 +196,7 @@ def choose_display(display_id: int | None) -> dict[str, Any]: # --------------------------------------------------------------------------- -# Screen capture (mss — cross-platform, identical to mac_helper) +# Screen capture (mss) # --------------------------------------------------------------------------- def capture_display(display_id: int | None, resize: tuple[int, int] | None = None) -> dict[str, Any]: @@ -563,6 +583,145 @@ def paste_clipboard() -> None: pyautogui.hotkey("ctrl", "v", interval=0.02) +# --------------------------------------------------------------------------- +# Physical input interference detection +# --------------------------------------------------------------------------- +# +# Why this exists at all, and why it is stricter than the macOS version. +# +# On macOS the helper posts events straight into the target process with +# `CGEvent.postToPid`, so agent input and human input never share a channel: +# the epoch monitor there is a safety net for an unlikely race. +# +# Windows has no such API. `pyautogui` bottoms out in `SendInput`, which +# injects into the ONE system-wide input stream and warps the ONE real cursor. +# The agent and the user are therefore holding the same mouse. If the user +# reaches for it mid-action the two streams interleave, and the resulting +# click lands somewhere neither of them intended. Detection is not a nicety +# here — it is the only thing standing between "the agent typed into the wrong +# window" and an abort. +# +# `GetLastInputInfo` is the right signal for this: it reports the tick of the +# last PHYSICAL input event, requires no privileges and no TCC-style grant, +# and — measured, and asserted by test_helpers.py — is NOT advanced by +# `SendInput` injection, so the agent cannot trip its own detector. + +import ctypes +from ctypes import wintypes + + +class _LASTINPUTINFO(ctypes.Structure): + _fields_ = [("cbSize", wintypes.UINT), ("dwTime", wintypes.DWORD)] + + +def last_physical_input_tick() -> int: + """Tick count of the last physical keyboard/mouse event. + + Returns 0 when unavailable so callers fail OPEN on the read itself: a + helper that refused to act because it could not query an optional Win32 + counter would be broken in a much more visible way than one that acted. + Interference is only ever reported on two SUCCESSFUL reads that differ. + """ + try: + info = _LASTINPUTINFO() + info.cbSize = ctypes.sizeof(_LASTINPUTINFO) + if not ctypes.windll.user32.GetLastInputInfo(ctypes.byref(info)): + return 0 + return int(info.dwTime) + except Exception: + return 0 + + +class UserInterference(RuntimeError): + """The user touched the physical mouse or keyboard during an action.""" + + def __init__(self, message: str, code: str = "user_interference") -> None: + super().__init__(message) + self.code = code + + +def _foreground_window_pid() -> int | None: + try: + import win32gui + import win32process + hwnd = win32gui.GetForegroundWindow() + if not hwnd: + return None + _, pid = win32process.GetWindowThreadProcessId(hwnd) + return int(pid) + except Exception: + return None + + +class ForegroundLease: + """Guards one mutating action against concurrent physical input. + + Evidence is sampled in a fixed order — tick, foreground identity, tick — + both before and after the action, so a single observation cannot straddle + a change it fails to notice. Same shape as `ForegroundLease.swift`. + + The asymmetry between the two failure modes is deliberate and is the whole + point of the class: + + * interference BEFORE the action -> `user_interference`. Nothing ran. + The caller may safely retry. + * interference DURING the action -> `user_interference_result_unknown`. + Injection already went into the shared input stream and we cannot know + how much of it landed, or where. Retrying could double-apply it. The + error says so rather than guessing. + """ + + def __init__(self) -> None: + self.tick: int = 0 + self.pid: int | None = None + + def acquire(self) -> None: + before = last_physical_input_tick() + pid = _foreground_window_pid() + after = last_physical_input_tick() + if before and after and before != after: + raise UserInterference( + "The user was typing or moving the mouse, so the action was " + "not sent. Nothing has changed; it is safe to try again." + ) + self.tick = after + self.pid = pid + + def finalize(self) -> None: + before = last_physical_input_tick() + pid = _foreground_window_pid() + after = last_physical_input_tick() + + if before and after and before != after: + raise UserInterference( + "The user used the mouse or keyboard while this action was " + "running. Because Windows shares one input stream between you " + "and the user, the two may have interleaved and the result is " + "UNKNOWN. Do not repeat the action — take a screenshot and " + "read the current state before deciding anything.", + code="user_interference_result_unknown", + ) + + if self.tick and after and self.tick != after: + raise UserInterference( + "The user used the mouse or keyboard while this action was " + "running. The result is UNKNOWN — do not repeat the action; " + "take a screenshot and read the current state first.", + code="user_interference_result_unknown", + ) + + # A foreground change without any physical input is the target app (or + # a background app) stealing activation, not the user. Worth reporting, + # because everything typed after it went somewhere unintended. + if self.pid is not None and pid is not None and self.pid != pid: + raise UserInterference( + "The foreground application changed while this action was " + "running, so input may have gone to the wrong window. The " + "result is UNKNOWN — take a screenshot before continuing.", + code="user_interference_result_unknown", + ) + + # --------------------------------------------------------------------------- # Permissions — Windows doesn't have macOS-style TCC # --------------------------------------------------------------------------- @@ -577,7 +736,177 @@ def check_permissions() -> dict[str, bool | None]: # --------------------------------------------------------------------------- -# Input actions (pyautogui — identical to mac_helper) +# Delivery preconditions — refuse rather than report a lie +# --------------------------------------------------------------------------- +# +# `SendInput` always "succeeds": it returns the number of events inserted into +# the input stream, never whether anything acted on them. Click a point behind +# another window and the click lands on THAT window; click a point off-screen +# and it lands nowhere. Either way pyautogui returns cleanly and the helper +# would answer "Action completed". +# +# That specific lie has burned us before on macOS — a session typed into a +# minimized window for a full turn because every action reported success. The +# fix there was to refuse instead of guessing, and the same rule applies here. + +class DeliveryRefused(RuntimeError): + def __init__(self, message: str, code: str) -> None: + super().__init__(message) + self.code = code + + +def _virtual_screen_rect() -> tuple[int, int, int, int] | None: + """(left, top, right, bottom) across all monitors, or None if unavailable.""" + try: + user32 = ctypes.windll.user32 + SM_XVIRTUALSCREEN, SM_YVIRTUALSCREEN = 76, 77 + SM_CXVIRTUALSCREEN, SM_CYVIRTUALSCREEN = 78, 79 + left = user32.GetSystemMetrics(SM_XVIRTUALSCREEN) + top = user32.GetSystemMetrics(SM_YVIRTUALSCREEN) + width = user32.GetSystemMetrics(SM_CXVIRTUALSCREEN) + height = user32.GetSystemMetrics(SM_CYVIRTUALSCREEN) + if width <= 0 or height <= 0: + return None + return (left, top, left + width, top + height) + except Exception: + return None + + +def ensure_point_on_screen(x: int, y: int) -> None: + """Refuse coordinates outside every monitor. + + Fails OPEN when the metrics are unreadable: an unreadable metric is our + problem, not the caller's, and blocking every action on it would be worse + than the miss it prevents. + """ + rect = _virtual_screen_rect() + if rect is None: + return + left, top, right, bottom = rect + if left <= x < right and top <= y < bottom: + return + raise DeliveryRefused( + f"The point ({x}, {y}) is outside every display " + f"(virtual screen is {left},{top} to {right},{bottom}), so the action " + "was not sent. Take a screenshot to get current coordinates.", + code="point_outside_display", + ) + + +def _window_is_interactable(hwnd: int) -> tuple[bool, str]: + """(ok, reason) — whether synthetic input can reach this window at all.""" + try: + import win32gui + if not win32gui.IsWindow(hwnd): + return False, "the window no longer exists" + if not win32gui.IsWindowVisible(hwnd): + return False, "the window is hidden" + try: + import win32con + placement = win32gui.GetWindowPlacement(hwnd) + if placement and placement[1] == win32con.SW_SHOWMINIMIZED: + return False, "the window is minimized" + except Exception: + pass + rect = win32gui.GetWindowRect(hwnd) + if rect[2] - rect[0] <= 0 or rect[3] - rect[1] <= 0: + return False, "the window has no on-screen area" + return True, "" + except Exception: + # Unreadable window state fails open, same reasoning as above. + return True, "" + + +def _windows_for_bundle(bundle_id: str) -> list[int]: + """Every top-level HWND owned by a process whose exe stem matches. + + Enumerates directly rather than reusing `list_windows()`, which filters out + invisible and zero-area windows — precisely the states this guard needs to + SEE in order to refuse. Reusing it would make the guard match nothing and + silently pass, which is the failure mode it was written to prevent. + """ + try: + import win32gui + import win32process + import psutil + except Exception: + return [] + + wanted = bundle_id.strip().lower() + if not wanted: + return [] + + pids: set[int] = set() + try: + for proc in psutil.process_iter(["pid", "name", "exe"]): + try: + exe_path = proc.info.get("exe") or "" + name = proc.info.get("name") or "" + stem = Path(exe_path).stem if exe_path else Path(name).stem + if stem and stem.lower() == wanted: + pids.add(int(proc.info["pid"])) + except (psutil.NoSuchProcess, psutil.AccessDenied): + continue + except Exception: + return [] + + if not pids: + return [] + + handles: list[int] = [] + + def _collect(hwnd: int, _: Any) -> None: + try: + _, pid = win32process.GetWindowThreadProcessId(hwnd) + if int(pid) in pids: + handles.append(int(hwnd)) + except Exception: + return + + try: + win32gui.EnumWindows(_collect, None) + except Exception: + return [] + return handles + + +def ensure_target_window_reachable(bundle_id: str | None) -> None: + """Refuse when the named app has no window that input could reach. + + A minimized window is the case that matters: on Windows it has no client + area to hit-test against, so a coordinate click is guaranteed to land on + whatever is underneath it. Reporting success there is exactly the lie this + guard exists to prevent. + + Fails OPEN when the app owns no top-level windows at all — that is a + different failure (wrong app name, app not running) which the caller's own + resolution step reports with a better message than this one could. + """ + if not bundle_id: + return + + handles = _windows_for_bundle(bundle_id) + if not handles: + return + + reasons: list[str] = [] + for hwnd in handles: + ok, reason = _window_is_interactable(hwnd) + if ok: + return + if reason: + reasons.append(reason) + + detail = reasons[0] if reasons else "it has no on-screen window" + raise DeliveryRefused( + f"The target app has no window that input can reach — {detail}. " + "The action was NOT sent. Restore the window and try again.", + code="target_window_offscreen", + ) + + +# --------------------------------------------------------------------------- +# Input actions (pyautogui → SendInput) # --------------------------------------------------------------------------- def click(x: int, y: int, button: str, count: int, modifiers: list[str] | None) -> None: @@ -629,9 +958,56 @@ def type_text(text: str) -> None: # --------------------------------------------------------------------------- -# Main dispatcher — exact same command protocol as mac_helper.py +# Main dispatcher — the command protocol the native macOS daemon also speaks # --------------------------------------------------------------------------- +# Commands that inject into the shared Windows input stream. Kept as one set +# rather than as a guard call inside each branch, because the branches are the +# easy place to forget one — and a forgotten branch is silently unguarded, the +# exact class of bug this whole pass exists to remove. +# +# Mirrors `CommandForegroundPolicy.leasedCommands` on the macOS side. +MUTATING_COMMANDS = frozenset({ + "click", "drag", "move_mouse", "scroll", + "mouse_down", "mouse_up", + "key", "hold_key", "type", + "paste_clipboard", +}) + +# The subset that targets a screen coordinate, and so needs the point itself to +# be reachable. `key`/`type` go to whatever holds focus and have no coordinate +# to check. +COORDINATE_COMMANDS = frozenset({"click", "drag", "move_mouse", "scroll"}) + + +def _coordinate_of(command: str, payload: dict[str, Any]) -> tuple[int, int] | None: + if command not in COORDINATE_COMMANDS: + return None + if command == "drag": + target = payload.get("to") or {} + if "x" in target and "y" in target: + return int(target["x"]), int(target["y"]) + return None + if "x" in payload and "y" in payload: + return int(payload["x"]), int(payload["y"]) + return None + + +def _finish(lease: "ForegroundLease | None", result: Any) -> int: + """Emit the success response for a mutating command, after the lease agrees. + + The check runs BEFORE the response is written, and that ordering is the + whole point: once `{"ok": true}` reaches the caller the action is reported + as done, and no later discovery can take that back. A helper that injected + input, then noticed the user had been typing throughout, and still answered + "Action completed" would be lying with a straight face. + """ + if lease is not None: + lease.finalize() + json_output({"ok": True, "result": result}) + return 0 + + def main() -> int: parser = argparse.ArgumentParser() parser.add_argument("command") @@ -639,8 +1015,20 @@ def main() -> int: args = parser.parse_args() payload = json.loads(args.payload) + lease: ForegroundLease | None = None + try: command = args.command + + if command in MUTATING_COMMANDS: + point = _coordinate_of(command, payload) + if point is not None: + ensure_point_on_screen(point[0], point[1]) + ensure_target_window_reachable( + payload.get("bundleId") or payload.get("app") + ) + lease = ForegroundLease() + lease.acquire() if command == "check_permissions": perms = check_permissions() json_output({"ok": True, "result": perms}) @@ -690,43 +1078,34 @@ def main() -> int: return 0 if command == "key": key_action(str(payload["keySequence"]), int(payload.get("repeat") or 1)) - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) if command == "hold_key": hold_keys(list(payload.get("keyNames") or []), int(payload.get("durationMs") or 0)) - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) if command == "type": type_text(str(payload.get("text") or "")) - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) if command == "click": click(int(payload["x"]), int(payload["y"]), str(payload.get("button") or "left"), int(payload.get("count") or 1), payload.get("modifiers")) - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) if command == "drag": from_point = payload.get("from") if from_point: pyautogui.moveTo(int(from_point["x"]), int(from_point["y"])) pyautogui.dragTo(int(payload["to"]["x"]), int(payload["to"]["y"]), duration=0.2, button="left") - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) if command == "move_mouse": pyautogui.moveTo(int(payload["x"]), int(payload["y"])) - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) if command == "scroll": scroll(int(payload["x"]), int(payload["y"]), int(payload.get("deltaX") or 0), int(payload.get("deltaY") or 0)) - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) if command == "mouse_down": pyautogui.mouseDown(button="left") - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) if command == "mouse_up": pyautogui.mouseUp(button="left") - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) if command == "cursor_position": x, y = pyautogui.position() json_output({"ok": True, "result": {"x": int(x), "y": int(y)}}) @@ -756,10 +1135,16 @@ def main() -> int: return 0 if command == "paste_clipboard": paste_clipboard() - json_output({"ok": True, "result": True}) - return 0 + return _finish(lease, True) error_output(f"Unknown command: {command}", code="bad_command") return 2 + except (UserInterference, DeliveryRefused) as exc: + # A deliberate refusal, not a crash. The code travels so the caller can + # tell "did not run, safe to retry" apart from "ran, outcome unknown" — + # collapsing both into a generic error is how a model ends up repeating + # a toggle it already flipped. + error_output(str(exc), code=exc.code) + return 1 except Exception as exc: error_output(str(exc)) return 1 diff --git a/src/utils/computerUse/cleanup.test.ts b/src/utils/computerUse/cleanup.test.ts index c4638f91..b7f92672 100644 --- a/src/utils/computerUse/cleanup.test.ts +++ b/src/utils/computerUse/cleanup.test.ts @@ -99,3 +99,37 @@ describe('cleanupComputerUseAfterTurn — turn-end overlay hide', () => { expect(hidden).toBe(1) }) }) + +describe('cleanupComputerUseAfterTurn — turn-end cursor badge', () => { + test('drops the Windows badge on a turn that hid nothing', async () => { + // The badge is a standing claim that the agent is holding the mouse. If it + // outlives the turn it is simply false, and the user has no way to tell + // that from a turn still in progress. + let hidden = 0 + await cleanupComputerUseAfterTurn(makeCtx(), { + overlayHide: async () => {}, + hideCursorBadge: () => { hidden += 1 }, + }) + expect(hidden).toBe(1) + }) + + test('drops the badge even when overlayHide hangs past its timeout', async () => { + // Abort paths are exactly when a stuck indicator is most likely and most + // confusing. The badge teardown must not sit behind the daemon's. + let hidden = 0 + await cleanupComputerUseAfterTurn(makeCtx(), { + overlayHide: () => new Promise(() => {}), + hideCursorBadge: () => { hidden += 1 }, + }) + expect(hidden).toBe(1) + }) + + test('drops the badge even when overlayHide rejects', async () => { + let hidden = 0 + await cleanupComputerUseAfterTurn(makeCtx(), { + overlayHide: async () => { throw new Error('daemon gone') }, + hideCursorBadge: () => { hidden += 1 }, + }) + expect(hidden).toBe(1) + }) +}) diff --git a/src/utils/computerUse/cleanup.ts b/src/utils/computerUse/cleanup.ts index 24d8994a..be4b8c4a 100644 --- a/src/utils/computerUse/cleanup.ts +++ b/src/utils/computerUse/cleanup.ts @@ -22,9 +22,9 @@ const UNHIDE_TIMEOUT_MS = 5000 const OVERLAY_HIDE_TIMEOUT_MS = 2000 /** - * Turn-end cleanup for the chicago MCP surface: hide the native overlay (macOS - * cu-helper daemon), auto-unhide apps that `prepareForAction` hid, then release - * the file-based lock. + * Turn-end cleanup for the chicago MCP surface: drop the activity indicator + * (macOS cu-helper overlay, or the Windows cursor badge), auto-unhide apps + * that `prepareForAction` hid, then release the file-based lock. * * Called from three sites: natural turn end (`stopHooks.ts`), abort during * streaming (`query.ts` aborted_streaming), abort during tool execution @@ -49,8 +49,22 @@ export async function cleanupComputerUseAfterTurn( ToolUseContext, 'getAppState' | 'setAppState' | 'sendOSNotification' >, - deps: { overlayHide?: () => Promise } = {}, + deps: { + overlayHide?: () => Promise + hideCursorBadge?: () => void + } = {}, ): Promise { + // Windows counterpart to the macOS overlay. Synchronous, no-throw, and a + // no-op off-Windows, so it goes first and unconditionally: an orphaned badge + // would sit on screen claiming the agent is holding the mouse after the turn + // has ended, which is a worse lie than showing nothing at all. + const hideBadge = + deps.hideCursorBadge ?? + (() => { + void import('./winCursorBadge.js').then(m => m.hideCursorBadge()).catch(() => {}) + }) + hideBadge() + // Drop the daemon overlay FIRST — before the hidden-apps block and before the // isLockHeldLocally early-return below — so the cursor drops promptly even on a // turn that hid no apps and whose lock-release short-circuits. overlayHide diff --git a/src/utils/computerUse/executor.ts b/src/utils/computerUse/executor.ts index c5660e34..cbcbabd2 100644 --- a/src/utils/computerUse/executor.ts +++ b/src/utils/computerUse/executor.ts @@ -1,9 +1,11 @@ /** - * CLI `ComputerExecutor` implementation — Python bridge variant. + * CLI `ComputerExecutor` implementation — platform-routed helper variant. * - * Replaces the native Swift/Rust modules with a Python subprocess bridge - * (pyautogui + mss + platform helpers). See `pythonBridge.ts` and - * `runtime/{mac,win}_helper.py`. + * Every command goes through `helperBridge.callHelper`, which routes by + * platform: macOS reaches the signed native `cu-helper` daemon, Windows + * reaches the Python helper (`runtime/win_helper.py`, pyautogui + mss). + * This module is deliberately platform-agnostic — the routing decision, and + * the very different guarantees each side offers, live in `helperBridge.ts`. */ import type { @@ -29,8 +31,8 @@ import { isComputerUseSupportedPlatform, } from './common.js' // Platform-routed helper: macOS → native cu-helper (no cursor steal), -// Windows → Python helper. Aliased so the 20+ call sites below stay unchanged. -import { callHelper as callPythonHelper } from './helperBridge.js' +// Windows → Python helper (pyautogui, which does move the real cursor). +import { callHelper } from './helperBridge.js' const SCREENSHOT_JPEG_QUALITY = 0.75 const MOVE_SETTLE_MS = 50 @@ -62,16 +64,16 @@ function normalizeDisplayGeometry(display: PythonDisplayGeometry): DisplayGeomet } async function readClipboardViaPbpaste(): Promise { - return callPythonHelper('read_clipboard', {}) + return callHelper('read_clipboard', {}) } async function writeClipboardViaPbcopy(text: string): Promise { - await callPythonHelper('write_clipboard', { text }) + await callHelper('write_clipboard', { text }) } async function readClipboard(): Promise { if (process.platform === 'win32') { - return callPythonHelper('read_clipboard', {}) + return callHelper('read_clipboard', {}) } return readClipboardViaPbpaste() @@ -79,7 +81,7 @@ async function readClipboard(): Promise { async function writeClipboard(text: string): Promise { if (process.platform === 'win32') { - await callPythonHelper('write_clipboard', { text }) + await callHelper('write_clipboard', { text }) return } @@ -155,21 +157,21 @@ function formatAppList(apps: readonly DaemonAppRef[]): string { export function createCodexEngine(): CodexComputerEngine { return { async listApps(): Promise { - const apps = await callPythonHelper('list_apps', {}) + const apps = await callHelper('list_apps', {}) return formatAppList(apps) }, async resolveTarget(target: AppTarget): Promise { // Sent as-is: the daemon owns the selector→process mapping, and it must // never launch anything to satisfy a match. - return callPythonHelper('resolve_app_target', target) + return callHelper('resolve_app_target', target) }, async getAppState( target: AppTarget, opts?: { disableDiff?: boolean }, ): Promise { - return callPythonHelper('get_app_state', { + return callHelper('get_app_state', { ...appTargetPayload(target), ...(opts?.disableDiff === undefined ? {} : { disableDiff: opts.disableDiff }), }) @@ -183,7 +185,7 @@ export function createCodexEngine(): CodexComputerEngine { clickCount?: number button?: CodexMouseButton }): Promise { - await callPythonHelper('click', { + await callHelper('click', { ...appTargetPayload(args.target), index: args.index, x: args.x, @@ -198,7 +200,7 @@ export function createCodexEngine(): CodexComputerEngine { index: string value: string }): Promise { - return callPythonHelper('set_value', { + return callHelper('set_value', { ...appTargetPayload(args.target), index: args.index, value: args.value, @@ -213,7 +215,7 @@ export function createCodexEngine(): CodexComputerEngine { suffix?: string selection?: 'text' | 'cursor_before' | 'cursor_after' }): Promise { - await callPythonHelper('select_text', { + await callHelper('select_text', { ...appTargetPayload(args.target), index: args.index, text: args.text, @@ -228,7 +230,7 @@ export function createCodexEngine(): CodexComputerEngine { index: string action: string }): Promise { - await callPythonHelper('perform_secondary_action', { + await callHelper('perform_secondary_action', { ...appTargetPayload(args.target), index: args.index, action: args.action, @@ -243,7 +245,7 @@ export function createCodexEngine(): CodexComputerEngine { direction: 'up' | 'down' | 'left' | 'right' pages?: number }): Promise { - await callPythonHelper('scroll', { + await callHelper('scroll', { ...appTargetPayload(args.target), index: args.index, x: args.x, @@ -259,7 +261,7 @@ export function createCodexEngine(): CodexComputerEngine { to: { x: number; y: number } button?: CodexMouseButton }): Promise { - await callPythonHelper('drag', { + await callHelper('drag', { ...appTargetPayload(args.target), from: args.from, to: args.to, @@ -272,7 +274,7 @@ export function createCodexEngine(): CodexComputerEngine { key: string systemKeyCombos: boolean }): Promise { - await callPythonHelper('press_key', { + await callHelper('press_key', { ...appTargetPayload(args.target), key: args.key, systemKeyCombos: args.systemKeyCombos, @@ -280,7 +282,7 @@ export function createCodexEngine(): CodexComputerEngine { }, async typeText(args: { target: AppTarget; text: string }): Promise { - await callPythonHelper('type_text', { + await callHelper('type_text', { ...appTargetPayload(args.target), text: args.text, }) @@ -300,10 +302,10 @@ async function typeViaClipboard(text: string): Promise { // Give NSPasteboard a beat before paste, then keep the new contents // resident long enough for Electron/WebView fields to consume them. await sleep(40) - await callPythonHelper('paste_clipboard', {}) + await callHelper('paste_clipboard', {}) await sleep(180) } else { - await callPythonHelper('key', { + await callHelper('key', { keySequence: 'ctrl+v', repeat: 1, }) @@ -341,30 +343,30 @@ export function createCliExecutor(_opts: { engine: process.platform === 'darwin' ? createCodexEngine() : undefined, async prepareForAction(_allowlistBundleIds, _displayId): Promise { - return callPythonHelper('prepare_for_action', {}) + return callHelper('prepare_for_action', {}) }, async previewHideSet(_allowlistBundleIds, _displayId) { - return callPythonHelper('preview_hide_set', {}) + return callHelper('preview_hide_set', {}) }, async getDisplaySize(displayId?: number): Promise { - return normalizeDisplayGeometry(await callPythonHelper('get_display_size', { displayId })) + return normalizeDisplayGeometry(await callHelper('get_display_size', { displayId })) }, async listDisplays(): Promise { - const displays = await callPythonHelper('list_displays', {}) + const displays = await callHelper('list_displays', {}) return displays.map(display => normalizeDisplayGeometry(display)) }, async findWindowDisplays(bundleIds: string[]) { - return callPythonHelper('find_window_displays', { bundleIds }) + return callHelper('find_window_displays', { bundleIds }) }, async resolvePrepareCapture(opts): Promise { const display = await this.getDisplaySize(opts.preferredDisplayId) const [targetW, targetH] = computeTargetDims(display.width, display.height, display.scaleFactor) - const result = await callPythonHelper('resolve_prepare_capture', { + const result = await callHelper('resolve_prepare_capture', { preferredDisplayId: opts.preferredDisplayId, targetWidth: targetW, targetHeight: targetH, @@ -380,7 +382,7 @@ export function createCliExecutor(_opts: { async screenshot(opts): Promise { const display = await this.getDisplaySize(opts.displayId) const [targetW, targetH] = computeTargetDims(display.width, display.height, display.scaleFactor) - const result = await callPythonHelper('screenshot', { + const result = await callHelper('screenshot', { displayId: opts.displayId, targetWidth: targetW, targetHeight: targetH, @@ -392,7 +394,7 @@ export function createCliExecutor(_opts: { async zoom(regionLogical, _allowedBundleIds, displayId) { const display = await this.getDisplaySize(displayId) const [outW, outH] = computeTargetDims(regionLogical.w, regionLogical.h, display.scaleFactor) - return callPythonHelper('zoom', { + return callHelper('zoom', { x: regionLogical.x, y: regionLogical.y, width: regionLogical.w, @@ -403,11 +405,11 @@ export function createCliExecutor(_opts: { }, async key(keySequence: string, repeat?: number): Promise { - await callPythonHelper('key', { keySequence, repeat: repeat ?? 1 }) + await callHelper('key', { keySequence, repeat: repeat ?? 1 }) }, async holdKey(keyNames: string[], durationMs: number): Promise { - await callPythonHelper('hold_key', { keyNames, durationMs }) + await callHelper('hold_key', { keyNames, durationMs }) }, async type(text: string, opts2: { viaClipboard: boolean }): Promise { @@ -415,61 +417,61 @@ export function createCliExecutor(_opts: { await typeViaClipboard(text) return } - await callPythonHelper('type', { text }) + await callHelper('type', { text }) }, readClipboard, writeClipboard, async click(x, y, button, count, modifiers): Promise { - await callPythonHelper('click', { x, y, button, count, modifiers }) + await callHelper('click', { x, y, button, count, modifiers }) await sleep(MOVE_SETTLE_MS) }, async mouseDown(): Promise { - await callPythonHelper('mouse_down', {}) + await callHelper('mouse_down', {}) }, async mouseUp(): Promise { - await callPythonHelper('mouse_up', {}) + await callHelper('mouse_up', {}) }, async getCursorPosition(): Promise<{ x: number; y: number }> { - return callPythonHelper('cursor_position', {}) + return callHelper('cursor_position', {}) }, async drag(from, to): Promise { - await callPythonHelper('drag', { from, to }) + await callHelper('drag', { from, to }) await sleep(MOVE_SETTLE_MS) }, async moveMouse(x, y): Promise { - await callPythonHelper('move_mouse', { x, y }) + await callHelper('move_mouse', { x, y }) await sleep(MOVE_SETTLE_MS) }, async scroll(x, y, dx, dy): Promise { - await callPythonHelper('scroll', { x, y, deltaX: dx, deltaY: dy }) + await callHelper('scroll', { x, y, deltaX: dx, deltaY: dy }) }, async getFrontmostApp(): Promise { - return callPythonHelper('frontmost_app', {}) + return callHelper('frontmost_app', {}) }, async appUnderPoint(x, y) { - return callPythonHelper('app_under_point', { x, y }) + return callHelper('app_under_point', { x, y }) }, async listInstalledApps(): Promise { - return callPythonHelper('list_installed_apps', {}) + return callHelper('list_installed_apps', {}) }, async listRunningApps(): Promise { - return callPythonHelper('list_running_apps', {}) + return callHelper('list_running_apps', {}) }, async openApp(bundleId: string): Promise { - await callPythonHelper('open_app', { bundleId }) + await callHelper('open_app', { bundleId }) }, } } diff --git a/src/utils/computerUse/helperBridge.test.ts b/src/utils/computerUse/helperBridge.test.ts index 76c5fec8..b1570bb3 100644 --- a/src/utils/computerUse/helperBridge.test.ts +++ b/src/utils/computerUse/helperBridge.test.ts @@ -355,10 +355,65 @@ describe('callHelper platform routing', () => { callDaemon: async () => { used = 'daemon'; return 0 as never }, callPy: async () => { used = 'py'; return 0 as never }, overlayShow: () => {}, + showCursorBadge: () => {}, }) expect(used).toBe('py') }) + test('Windows marks agent activity on an injecting command', async () => { + // Windows drives through SendInput, so the user's real cursor moves. The + // badge is the only thing telling them the movement is not theirs, which + // matters because grabbing the mouse mid-action is what makes the two + // input streams interleave. + let badges = 0 + await callHelper('click', { x: 1, y: 2 }, { + platform: 'win32', + callPy: ok, + showCursorBadge: () => { badges += 1 }, + }) + expect(badges).toBe(1) + }) + + test('Windows leaves the badge alone for read-only commands', async () => { + // A badge on `screenshot` would claim the agent is holding the mouse + // during a turn that never touches it. + let badges = 0 + for (const command of ['screenshot', 'list_displays', 'read_clipboard']) { + await callHelper(command, {}, { + platform: 'win32', + callPy: ok, + showCursorBadge: () => { badges += 1 }, + }) + } + expect(badges).toBe(0) + }) + + test('Windows never reaches the macOS overlay', async () => { + // The two indicators are not interchangeable: overlay_show is a daemon + // command and there is no daemon on Windows, so a stray call would be a + // hard failure on the mutation path. + let overlays = 0 + await callHelper('click', { x: 1, y: 2 }, { + platform: 'win32', + callPy: ok, + overlayShow: () => { overlays += 1 }, + showCursorBadge: () => {}, + }) + expect(overlays).toBe(0) + }) + + test('macOS never spawns the Windows badge', async () => { + let badges = 0 + await callHelper('click', { x: 1, y: 2 }, { + platform: 'darwin', + cuHelperAvailable: () => true, + callDaemon: ok, + overlayShow: () => {}, + showCursorBadge: () => { badges += 1 }, + }) + expect(badges).toBe(0) + }) + test('forwards command + payload to the daemon unchanged', async () => { let seen: { c: string; p: unknown } | undefined await callHelper('type', { text: 'hi' }, { diff --git a/src/utils/computerUse/helperBridge.ts b/src/utils/computerUse/helperBridge.ts index d247902b..3798fc48 100644 --- a/src/utils/computerUse/helperBridge.ts +++ b/src/utils/computerUse/helperBridge.ts @@ -8,6 +8,7 @@ import { overlayShow, shutdownDaemon, } from './cuHelperDaemon.js' +import { showCursorBadge } from './winCursorBadge.js' // Latches true after we restart the daemon once in response to an Accessibility // `not_trusted` error, so we don't thrash-restart while the helper is genuinely @@ -97,6 +98,14 @@ function overlayTargetPayload( * - Windows → the Python helper (`win_helper.py`); the native engine is * macOS-only. * + * The two platforms do NOT offer the same guarantee, and callers should not + * assume they do. macOS delivers input per-process and never touches the real + * pointer. Windows has no such API: input goes through `SendInput`, so the + * agent shares one cursor and one input stream with the user. The Windows + * side therefore gets a badge that marks agent activity rather than a virtual + * cursor that replaces it, and `win_helper.py` refuses actions it can already + * tell will not land. + * * `deps` is injectable for unit tests only. */ export async function callHelper( @@ -109,6 +118,7 @@ export async function callHelper( callPy?: HelperFn overlayShow?: (payload: Record) => void isOverlayShown?: () => boolean + showCursorBadge?: () => void callFrontmost?: () => Promise<{ bundleId?: string } | null> shutdownDaemon?: () => void } = {}, @@ -119,6 +129,7 @@ export async function callHelper( const viaPython = deps.callPy ?? (callPythonHelper as HelperFn) const showOverlay = deps.overlayShow ?? overlayShow const overlayIsShown = deps.isOverlayShown ?? isOverlayShown + const showBadge = deps.showCursorBadge ?? (() => showCursorBadge()) const restartDaemon = deps.shutdownDaemon ?? (() => void shutdownDaemon()) if (platform === 'darwin') { @@ -156,6 +167,12 @@ export async function callHelper( } } + if (INJECTION_COMMANDS.has(command)) { + // Same trigger set as the macOS overlay, so both platforms mark activity + // at the same moments. Fire-and-forget: the badge is advisory and must + // never sit on the mutation hot path. + showBadge() + } return viaPython(command, payload) } diff --git a/src/utils/computerUse/hostAdapter.ts b/src/utils/computerUse/hostAdapter.ts index 0ca18444..b98d1087 100644 --- a/src/utils/computerUse/hostAdapter.ts +++ b/src/utils/computerUse/hostAdapter.ts @@ -9,7 +9,7 @@ import { createCliExecutor } from './executor.js' import { getChicagoEnabled, getChicagoSubGates } from './gates.js' import { normalizeOsPermissions } from './permissions.js' // Platform-routed helper: macOS → native cu-helper, Windows → Python helper. -import { callHelper as callPythonHelper } from './helperBridge.js' +import { callHelper } from './helperBridge.js' import { maybeShowNativePermissionCard } from './nativePermissionCard.js' class DebugLogger implements Logger { @@ -42,7 +42,7 @@ export function getComputerUseHostAdapter(): ComputerUseHostAdapter { getHideBeforeActionEnabled: () => getChicagoSubGates().hideBeforeAction, }), ensureOsPermissions: async () => { - const rawPerms = await callPythonHelper<{ accessibility: boolean; screenRecording: boolean | null }>('check_permissions', {}) + const rawPerms = await callHelper<{ accessibility: boolean; screenRecording: boolean | null }>('check_permissions', {}) const perms = normalizeOsPermissions(rawPerms) if (perms.granted) return { granted: true as const } // Missing a TCC grant → pop the native, guided permission card (macOS). diff --git a/src/utils/computerUse/pythonBridge.ts b/src/utils/computerUse/pythonBridge.ts index dd15e96a..42ff8732 100644 --- a/src/utils/computerUse/pythonBridge.ts +++ b/src/utils/computerUse/pythonBridge.ts @@ -13,7 +13,13 @@ const projectRoot = path.resolve(__dirname, '../../..') // All runtime state lives in ~/.claude/.runtime — writable in both dev and // bundled (Tauri app) modes. The setup API (or ensureRuntimeFiles below) -// populates requirements.txt and mac_helper.py here. +// populates requirements-win.txt and win_helper.py here. +// +// This bridge is Windows-only. macOS routes every command to the signed native +// `cu-helper` daemon and `helperBridge` refuses to fall back, so the old +// `mac_helper.py` was unreachable and has been deleted along with its +// pyobjc requirements file. Keeping a dead darwin branch here invited the +// reading that Python is still a supported macOS path — it is not. const runtimeStateRoot = path.join(getClaudeConfigHomeDir(), '.runtime') const venvRoot = path.join(runtimeStateRoot, 'venv') const installStampPath = path.join(runtimeStateRoot, 'requirements.sha256') @@ -22,8 +28,12 @@ const isWindows = process.platform === 'win32' // Always read from ~/.claude/.runtime/ — works in both dev and bundled mode. const requirementsPath = path.join(runtimeStateRoot, 'requirements.txt') -const helperFileName = isWindows ? 'win_helper.py' : 'mac_helper.py' +const helperFileName = 'win_helper.py' const helperPath = path.join(runtimeStateRoot, helperFileName) +// Runs as its own process (the helper is a stateless one-shot CLI and cannot +// own a window across actions), so it ships as a separate file. +const cursorBadgeFileName = 'win_cursor_badge.py' +const cursorBadgePath = path.join(runtimeStateRoot, cursorBadgeFileName) let bootstrapPromise: Promise | undefined @@ -98,8 +108,7 @@ async function getVenvCreationPythonCommand(): Promise { async function ensureRuntimeFiles(): Promise { await mkdir(runtimeStateRoot, { recursive: true }) - const devReqFile = isWindows ? 'requirements-win.txt' : 'requirements.txt' - const devRequirements = path.join(projectRoot, 'runtime', devReqFile) + const devRequirements = path.join(projectRoot, 'runtime', 'requirements-win.txt') const devHelper = path.join(projectRoot, 'runtime', helperFileName) // Always sync from dev runtime/ so source changes are reflected immediately. @@ -112,12 +121,17 @@ async function ensureRuntimeFiles(): Promise { if (await pathExists(devHelper)) { await writeFile(helperPath, await readFile(devHelper, 'utf8'), 'utf8') } + + const devBadge = path.join(projectRoot, 'runtime', cursorBadgeFileName) + if (await pathExists(devBadge)) { + await writeFile(cursorBadgePath, await readFile(devBadge, 'utf8'), 'utf8') + } } export async function ensureBootstrapped(): Promise { if (bootstrapPromise) return bootstrapPromise bootstrapPromise = (async () => { - // Extract runtime files (requirements.txt, mac_helper.py) to state dir + // Extract runtime files (requirements, helper, badge) to state dir await ensureRuntimeFiles() if (!(await pathExists(pythonBinPath()))) { @@ -168,7 +182,7 @@ export async function callPythonHelper(command: string, payload: Record(command: string, payload: Record(command: string, payload: Record