fix(computer-use): make the Windows helper refuse what it cannot deliver

Windows has no equivalent of `CGEvent.postToPid`. `pyautogui` bottoms out in
`SendInput`, which injects into the one system-wide input stream and warps the
one real cursor — so on Windows the agent shares the mouse and keyboard with
the user, and `SendInput` reports success unconditionally whether or not
anything acted on the events.

That combination produced the same lie the macOS engine was just fixed for:
click a point behind another window and it lands on that window; click one
off-screen and it lands nowhere; either way the helper answered "Action
completed".

So the helper now refuses instead of guessing:

  * `ForegroundLease` samples `GetLastInputInfo` around every mutating command.
    Interference before the action is `user_interference` — nothing ran, retry
    is safe. Interference during it is `user_interference_result_unknown`,
    because injection already went out and a retry could double-apply it. On a
    play/pause toggle those two differ by exactly one wrong outcome.
  * `ensure_point_on_screen` and `ensure_target_window_reachable` reject
    coordinates outside every display and targets whose windows are minimized
    or hidden, before anything is sent.

`GetLastInputInfo` is the signal because it needs no privileges and is not
advanced by `SendInput`, so the agent cannot trip its own detector. Both guards
fail open on an unreadable reading: a safety layer that turns the feature off
is not safety.

The guard set lives in one place rather than in each dispatcher branch — an
eleventh verb wired like the ten before it would otherwise be silently
unguarded, which is the bug class this pass removes. All ten mutating branches
now return through `_finish`, so no branch can write its own success response
and skip the post-action check.

Also adds a Windows cursor badge. It deliberately does NOT mirror the macOS
virtual cursor: there the real pointer never moves, so the drawn one is the
only cursor and replaces it. Here the real pointer does move, and a second
fake pointer would just be two cursors with one of them lying about where the
click lands. The badge annotates instead — it answers "is this me or the
agent?", which matters because grabbing the mouse mid-action is what makes the
two input streams interleave.

Retires `runtime/mac_helper.py` and its pyobjc requirements: macOS routes every
command to the signed native daemon and `helperBridge` refuses to fall back, so
both were unreachable. Renames the `callPythonHelper` alias to `callHelper`,
which is what it has actually imported since the native engine landed.

Verified with mutation testing — 13 injected regressions across both languages
(dropped lease, bypassed `_finish`, collapsed interference codes, guard reusing
the filtered `list_windows`, badge losing click-through), all caught.

Claude-Session: https://claude.ai/code/session_015j1yxxaoonyAS2iZ7qGnTS
This commit is contained in:
程序员阿江(Relakkes)
2026-08-23 18:25:21 +08:00
parent f3a6e02903
commit 96e31ede14
13 changed files with 1270 additions and 1095 deletions
-775
View File
@@ -1,775 +0,0 @@
#!/usr/bin/env python3
from __future__ import annotations
import argparse
import base64
import ctypes
import json
import os
import subprocess
import sys
import time
from io import BytesIO
from pathlib import Path
from typing import Any
import mss
from AppKit import NSWorkspace, NSPasteboard, NSPasteboardTypeString, NSURL
from PIL import Image
from Quartz import (
CGDisplayBounds,
CGDisplayIsMain,
CGDisplayModeGetPixelHeight,
CGDisplayModeGetPixelWidth,
CGDisplayPixelsHigh,
CGDisplayPixelsWide,
CGGetActiveDisplayList,
CGMainDisplayID,
CGWindowListCopyWindowInfo,
CGRectContainsPoint,
CGRectIntersection,
CGPointMake,
CGPreflightScreenCaptureAccess,
kCGNullWindowID,
kCGWindowBounds,
kCGWindowIsOnscreen,
kCGWindowLayer,
kCGWindowListExcludeDesktopElements,
kCGWindowListOptionOnScreenOnly,
kCGWindowName,
kCGWindowOwnerName,
)
os.environ.setdefault("PYTHONDONTWRITEBYTECODE", "1")
os.environ.setdefault("PYAUTOGUI_HIDE_SUPPORT_PROMPT", "1")
import pyautogui # noqa: E402
pyautogui.FAILSAFE = False
pyautogui.PAUSE = 0
KEY_MAP = {
"a": "a",
"b": "b",
"c": "c",
"d": "d",
"e": "e",
"f": "f",
"g": "g",
"h": "h",
"i": "i",
"j": "j",
"k": "k",
"l": "l",
"m": "m",
"n": "n",
"o": "o",
"p": "p",
"q": "q",
"r": "r",
"s": "s",
"t": "t",
"u": "u",
"v": "v",
"w": "w",
"x": "x",
"y": "y",
"z": "z",
"0": "0",
"1": "1",
"2": "2",
"3": "3",
"4": "4",
"5": "5",
"6": "6",
"7": "7",
"8": "8",
"9": "9",
"cmd": "command",
"command": "command",
"meta": "command",
"super": "command",
"ctrl": "ctrl",
"control": "ctrl",
"shift": "shift",
"alt": "option",
"option": "option",
"opt": "option",
"fn": "fn",
"escape": "esc",
"esc": "esc",
"enter": "enter",
"return": "enter",
"tab": "tab",
"space": "space",
"backspace": "backspace",
"delete": "delete",
"forwarddelete": "delete",
"up": "up",
"down": "down",
"left": "left",
"right": "right",
"home": "home",
"end": "end",
"pageup": "pageup",
"pagedown": "pagedown",
"capslock": "capslock",
"f1": "f1",
"f2": "f2",
"f3": "f3",
"f4": "f4",
"f5": "f5",
"f6": "f6",
"f7": "f7",
"f8": "f8",
"f9": "f9",
"f10": "f10",
"f11": "f11",
"f12": "f12",
"-": "minus",
"=": "equals",
"[": "[",
"]": "]",
"\\": "\\",
";": ";",
"'": "'",
",": ",",
".": ".",
"/": "/",
"`": "`",
}
def normalize_key(name: str) -> str:
key = name.strip().lower()
if key not in KEY_MAP:
raise ValueError(f"Unsupported key: {name}")
return KEY_MAP[key]
def json_output(payload: dict[str, Any]) -> None:
sys.stdout.write(json.dumps(payload, ensure_ascii=False))
sys.stdout.write("\n")
sys.stdout.flush()
def error_output(message: str, code: str = "runtime_error") -> None:
json_output({"ok": False, "error": {"code": code, "message": message}})
def bool_env(name: str, default: bool = False) -> bool:
value = os.environ.get(name)
if value is None:
return default
return value not in {"0", "false", "False", ""}
def run_osascript(script: str) -> str:
result = subprocess.run(
["osascript", "-e", script],
text=True,
capture_output=True,
check=False,
)
if result.returncode != 0:
raise RuntimeError(result.stderr.strip() or result.stdout.strip() or "osascript failed")
return result.stdout.strip()
def applescript_modifier(name: str) -> str:
if name == "command":
return "command down"
if name == "option":
return "option down"
if name == "shift":
return "shift down"
if name == "ctrl":
return "control down"
if name == "fn":
return "fn down"
raise ValueError(f"Unsupported AppleScript modifier: {name}")
def send_keystroke_via_osascript(character: str, modifiers: list[str] | None = None) -> None:
escaped = character.replace("\\", "\\\\").replace('"', '\\"')
if modifiers:
modifier_expr = ", ".join(applescript_modifier(m) for m in modifiers)
script = (
'tell application "System Events" to keystroke '
f'"{escaped}" using {{{modifier_expr}}}'
)
else:
script = f'tell application "System Events" to keystroke "{escaped}"'
run_osascript(script)
def get_displays() -> list[dict[str, Any]]:
max_displays = 32
err, active, count = CGGetActiveDisplayList(max_displays, None, None)
if err != 0:
raise RuntimeError(f"CGGetActiveDisplayList failed: {err}")
displays: list[dict[str, Any]] = []
main_id = CGMainDisplayID()
for idx, display_id in enumerate(active[:count]):
bounds = CGDisplayBounds(display_id)
mode = None
try:
from Quartz import CGDisplayCopyDisplayMode
mode = CGDisplayCopyDisplayMode(display_id)
except Exception:
mode = None
physical_width = int(CGDisplayPixelsWide(display_id))
physical_height = int(CGDisplayPixelsHigh(display_id))
logical_width = int(bounds.size.width)
logical_height = int(bounds.size.height)
if mode is not None:
mode_w = int(CGDisplayModeGetPixelWidth(mode))
mode_h = int(CGDisplayModeGetPixelHeight(mode))
physical_width = mode_w or physical_width
physical_height = mode_h or physical_height
scale_factor = physical_width / logical_width if logical_width else 1
name = f"Display {idx + 1}"
displays.append(
{
"id": int(display_id),
"displayId": int(display_id),
"width": logical_width,
"height": logical_height,
"scaleFactor": scale_factor,
"originX": int(bounds.origin.x),
"originY": int(bounds.origin.y),
"isPrimary": bool(display_id == main_id or CGDisplayIsMain(display_id)),
"name": name,
"label": name,
}
)
return displays
def choose_display(display_id: int | None) -> dict[str, Any]:
displays = get_displays()
if not displays:
raise RuntimeError("No active displays found")
if display_id is None:
for display in displays:
if display["isPrimary"]:
return display
return displays[0]
for display in displays:
if display["displayId"] == display_id or display["id"] == display_id:
return display
raise RuntimeError(f"Unknown display: {display_id}")
def ensure_screen_recording_permission() -> None:
"""No-op: CGPreflightScreenCaptureAccess is unreliable for child processes
(returns False even when the parent app has TCC permission), and any actual
capture attempt triggers a macOS popup on newer versions. Let the actual
capture call handle errors instead."""
pass
def capture_display(display_id: int | None, resize: tuple[int, int] | None = None) -> dict[str, Any]:
ensure_screen_recording_permission()
display = choose_display(display_id)
monitor = {
"left": display["originX"],
"top": display["originY"],
"width": display["width"],
"height": display["height"],
}
with mss.mss() as sct:
raw = sct.grab(monitor)
image = Image.frombytes("RGB", raw.size, raw.rgb)
if resize:
image = image.resize(resize, Image.Resampling.LANCZOS)
buffer = BytesIO()
image.save(buffer, format="JPEG", quality=75, optimize=True)
base64_data = base64.b64encode(buffer.getvalue()).decode("ascii")
return {
"base64": base64_data,
"width": image.width,
"height": image.height,
"displayWidth": display["width"],
"displayHeight": display["height"],
"displayId": display["displayId"],
"originX": display["originX"],
"originY": display["originY"],
"display": display,
}
def capture_region(region: dict[str, int], resize: tuple[int, int] | None = None) -> dict[str, Any]:
ensure_screen_recording_permission()
with mss.mss() as sct:
raw = sct.grab(region)
image = Image.frombytes("RGB", raw.size, raw.rgb)
if resize:
image = image.resize(resize, Image.Resampling.LANCZOS)
buffer = BytesIO()
image.save(buffer, format="JPEG", quality=75, optimize=True)
base64_data = base64.b64encode(buffer.getvalue()).decode("ascii")
return {"base64": base64_data, "width": image.width, "height": image.height}
def list_windows() -> list[dict[str, Any]]:
windows = CGWindowListCopyWindowInfo(
kCGWindowListOptionOnScreenOnly | kCGWindowListExcludeDesktopElements,
kCGNullWindowID,
)
out: list[dict[str, Any]] = []
for window in windows or []:
if int(window.get(kCGWindowLayer, 0)) != 0:
continue
if not bool(window.get(kCGWindowIsOnscreen, True)):
continue
bounds = window.get(kCGWindowBounds) or {}
width = int(bounds.get("Width", 0))
height = int(bounds.get("Height", 0))
if width <= 1 or height <= 1:
continue
out.append(
{
"ownerName": window.get(kCGWindowOwnerName, "") or "",
"title": window.get(kCGWindowName, "") or "",
"bounds": {
"x": int(bounds.get("X", 0)),
"y": int(bounds.get("Y", 0)),
"width": width,
"height": height,
},
}
)
return out
def bundle_id_to_app(bundle_id: str):
return NSWorkspace.sharedWorkspace().URLForApplicationWithBundleIdentifier_(bundle_id)
def installed_apps() -> list[dict[str, Any]]:
search_roots = [
Path("/Applications"),
Path.home() / "Applications",
Path("/System/Applications"),
Path("/System/Applications/Utilities"),
]
results: dict[str, dict[str, Any]] = {}
workspace = NSWorkspace.sharedWorkspace()
for root in search_roots:
if not root.exists():
continue
for app in root.rglob("*.app"):
try:
bundle = workspace.bundleIdentifierForURL_(NSURL.fileURLWithPath_(str(app)))
except Exception:
bundle = None
if not bundle:
try:
url = workspace.URLForApplicationWithBundleIdentifier_(str(app))
bundle = workspace.bundleIdentifierForURL_(url) if url else None
except Exception:
bundle = None
info_plist = app / "Contents/Info.plist"
display_name = app.stem
if info_plist.exists():
try:
import plistlib
with info_plist.open("rb") as f:
plist = plistlib.load(f)
bundle = bundle or plist.get("CFBundleIdentifier")
display_name = plist.get("CFBundleDisplayName") or plist.get("CFBundleName") or display_name
except Exception:
pass
if not bundle or bundle in results:
continue
results[bundle] = {
"bundleId": str(bundle),
"displayName": str(display_name),
"path": str(app),
}
return sorted(results.values(), key=lambda item: item["displayName"].lower())
def running_apps() -> list[dict[str, Any]]:
apps = []
seen = set()
for app in NSWorkspace.sharedWorkspace().runningApplications() or []:
bundle_id = app.bundleIdentifier()
if not bundle_id or bundle_id in seen:
continue
seen.add(bundle_id)
name = app.localizedName() or bundle_id
apps.append({"bundleId": str(bundle_id), "displayName": str(name)})
return sorted(apps, key=lambda item: item["displayName"].lower())
def app_display_name(bundle_id: str) -> str | None:
for app in NSWorkspace.sharedWorkspace().runningApplications() or []:
if app.bundleIdentifier() == bundle_id:
return str(app.localizedName() or bundle_id)
for app in installed_apps():
if app["bundleId"] == bundle_id:
return str(app["displayName"])
return None
def frontmost_app() -> dict[str, str] | None:
app = NSWorkspace.sharedWorkspace().frontmostApplication()
if not app:
return None
bundle_id = app.bundleIdentifier()
if not bundle_id:
return None
return {
"bundleId": str(bundle_id),
"displayName": str(app.localizedName() or bundle_id),
}
def app_under_point(x: int, y: int) -> dict[str, str] | None:
point = CGPointMake(x, y)
running_by_name = {
str(app.localizedName() or app.bundleIdentifier()): str(app.bundleIdentifier())
for app in NSWorkspace.sharedWorkspace().runningApplications() or []
if app.bundleIdentifier()
}
for window in list_windows():
bounds = window["bounds"]
rect = ((bounds["x"], bounds["y"]), (bounds["width"], bounds["height"]))
if CGRectContainsPoint(rect, point):
owner = window["ownerName"]
bundle = running_by_name.get(owner)
if bundle:
return {"bundleId": bundle, "displayName": str(owner)}
return frontmost_app()
def find_window_displays(bundle_ids: list[str]) -> list[dict[str, Any]]:
if not bundle_ids:
return []
displays = get_displays()
names_by_bundle = {
bundle_id: app_display_name(bundle_id) or bundle_id for bundle_id in bundle_ids
}
windows = list_windows()
result = []
for bundle_id in bundle_ids:
target_name = names_by_bundle.get(bundle_id)
display_ids: set[int] = set()
for window in windows:
owner = window["ownerName"]
if not owner:
continue
if target_name and owner != target_name:
continue
if not target_name and owner != bundle_id:
continue
wx = window["bounds"]["x"]
wy = window["bounds"]["y"]
ww = window["bounds"]["width"]
wh = window["bounds"]["height"]
window_rect = ((wx, wy), (ww, wh))
for display in displays:
display_rect = ((display["originX"], display["originY"]), (display["width"], display["height"]))
intersection = CGRectIntersection(window_rect, display_rect)
if intersection.size.width > 0 and intersection.size.height > 0:
display_ids.add(int(display["displayId"]))
result.append({"bundleId": bundle_id, "displayIds": sorted(display_ids)})
return result
def open_app(bundle_id: str) -> None:
url = bundle_id_to_app(bundle_id)
if not url:
raise RuntimeError(f"App not found for bundle identifier: {bundle_id}")
ok, err = NSWorkspace.sharedWorkspace().launchApplicationAtURL_options_configuration_error_(url, 0, {}, None)
if not ok:
raise RuntimeError(str(err) if err else f"Failed to open app {bundle_id}")
def read_clipboard() -> str:
pb = NSPasteboard.generalPasteboard()
value = pb.stringForType_(NSPasteboardTypeString)
return "" if value is None else str(value)
def write_clipboard(text: str) -> None:
pb = NSPasteboard.generalPasteboard()
pb.clearContents()
pb.setString_forType_(text, NSPasteboardTypeString)
def paste_clipboard() -> None:
send_keystroke_via_osascript("v", ["command"])
def detect_screen_recording_permission() -> bool | None:
"""Best-effort passive screen-recording probe with no system prompt.
`CGPreflightScreenCaptureAccess()` is fast and explicit when it returns
True, but on child processes launched by a TCC-authorized app bundle it can
still return False. As a fallback, inspect the visible window list: Apple
only exposes other apps' window titles when Screen Recording access is
granted. If we can see at least one title, treat the permission as granted.
If we can inspect visible windows but every title is blank, treat it as not
granted. If window enumeration itself is unavailable, return None.
"""
try:
if CGPreflightScreenCaptureAccess():
return True
except Exception:
pass
try:
windows = CGWindowListCopyWindowInfo(
kCGWindowListOptionOnScreenOnly | kCGWindowListExcludeDesktopElements,
kCGNullWindowID,
)
except Exception:
return None
eligible_windows = 0
for window in windows or []:
if int(window.get(kCGWindowLayer, 0)) != 0:
continue
if not bool(window.get(kCGWindowIsOnscreen, True)):
continue
bounds = window.get(kCGWindowBounds) or {}
width = int(bounds.get("Width", 0))
height = int(bounds.get("Height", 0))
if width <= 1 or height <= 1:
continue
eligible_windows += 1
if (window.get(kCGWindowName, "") or "").strip():
return True
if eligible_windows > 0:
return False
return None
def detect_accessibility_permission() -> bool:
"""
Use the official macOS Accessibility trust API.
The previous System Events / AppleScript probe was too weak: it could
succeed even when the current helper process was not actually trusted for
input control, which led the desktop UI to report Accessibility as granted
while mouse/keyboard control still failed at runtime.
"""
framework_path = "/System/Library/Frameworks/ApplicationServices.framework/ApplicationServices"
try:
application_services = ctypes.CDLL(framework_path)
application_services.AXIsProcessTrusted.restype = ctypes.c_bool
application_services.AXIsProcessTrusted.argtypes = []
return bool(application_services.AXIsProcessTrusted())
except Exception:
# Fail closed: if the trust API can't be queried, treat accessibility
# as unavailable instead of reporting a misleading success state.
return False
def check_permissions() -> dict[str, bool | None]:
accessibility = detect_accessibility_permission()
screen_recording = detect_screen_recording_permission()
return {
"accessibility": accessibility,
"screenRecording": screen_recording,
}
def click(x: int, y: int, button: str, count: int, modifiers: list[str] | None) -> None:
pyautogui.moveTo(x, y)
if modifiers:
normalized = [normalize_key(m) for m in modifiers]
for key in normalized:
pyautogui.keyDown(key)
try:
pyautogui.click(x=x, y=y, button=button, clicks=count, interval=0.08)
finally:
for key in reversed(normalized):
pyautogui.keyUp(key)
else:
pyautogui.click(x=x, y=y, button=button, clicks=count, interval=0.08)
def scroll(x: int, y: int, delta_x: int, delta_y: int) -> None:
pyautogui.moveTo(x, y)
if delta_y:
pyautogui.scroll(int(delta_y), x=x, y=y)
if delta_x:
pyautogui.hscroll(int(delta_x), x=x, y=y)
def key_action(sequence: str, repeat: int = 1) -> None:
parts = [normalize_key(part) for part in sequence.split("+") if part.strip()]
for _ in range(max(1, repeat)):
if parts == ["command", "v"]:
paste_clipboard()
elif parts == ["command", "a"]:
send_keystroke_via_osascript("a", ["command"])
elif parts == ["command", "c"]:
send_keystroke_via_osascript("c", ["command"])
elif parts == ["command", "x"]:
send_keystroke_via_osascript("x", ["command"])
elif len(parts) == 1:
pyautogui.press(parts[0])
else:
pyautogui.hotkey(*parts, interval=0.02)
time.sleep(0.01)
def hold_keys(keys: list[str], duration_ms: int) -> None:
normalized = [normalize_key(k) for k in keys]
for key in normalized:
pyautogui.keyDown(key)
try:
time.sleep(max(duration_ms, 0) / 1000)
finally:
for key in reversed(normalized):
pyautogui.keyUp(key)
def type_text(text: str) -> None:
pyautogui.write(text, interval=0.008)
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("command")
parser.add_argument("--payload", default="{}")
args = parser.parse_args()
payload = json.loads(args.payload)
try:
command = args.command
if command == "check_permissions":
perms = check_permissions()
json_output({"ok": True, "result": perms})
return 0
if command == "list_displays":
json_output({"ok": True, "result": get_displays()})
return 0
if command == "get_display_size":
json_output({"ok": True, "result": choose_display(payload.get("displayId"))})
return 0
if command == "screenshot":
resize = None
if payload.get("targetWidth") and payload.get("targetHeight"):
resize = (int(payload["targetWidth"]), int(payload["targetHeight"]))
result = capture_display(payload.get("displayId"), resize)
json_output({"ok": True, "result": result})
return 0
if command == "resolve_prepare_capture":
resize = None
if payload.get("targetWidth") and payload.get("targetHeight"):
resize = (int(payload["targetWidth"]), int(payload["targetHeight"]))
result = capture_display(payload.get("preferredDisplayId"), resize)
result["hidden"] = []
result["resolvedDisplayId"] = result["displayId"]
json_output({"ok": True, "result": result})
return 0
if command == "zoom":
resize = None
if payload.get("targetWidth") and payload.get("targetHeight"):
resize = (int(payload["targetWidth"]), int(payload["targetHeight"]))
region = {
"left": int(payload["x"]),
"top": int(payload["y"]),
"width": int(payload["width"]),
"height": int(payload["height"]),
}
json_output({"ok": True, "result": capture_region(region, resize)})
return 0
if command == "prepare_for_action":
json_output({"ok": True, "result": []})
return 0
if command == "preview_hide_set":
json_output({"ok": True, "result": []})
return 0
if command == "find_window_displays":
json_output({"ok": True, "result": find_window_displays(list(payload.get("bundleIds") or []))})
return 0
if command == "key":
key_action(str(payload["keySequence"]), int(payload.get("repeat") or 1))
json_output({"ok": True, "result": True})
return 0
if command == "hold_key":
hold_keys(list(payload.get("keyNames") or []), int(payload.get("durationMs") or 0))
json_output({"ok": True, "result": True})
return 0
if command == "type":
type_text(str(payload.get("text") or ""))
json_output({"ok": True, "result": True})
return 0
if command == "click":
click(int(payload["x"]), int(payload["y"]), str(payload.get("button") or "left"), int(payload.get("count") or 1), payload.get("modifiers"))
json_output({"ok": True, "result": True})
return 0
if command == "drag":
from_point = payload.get("from")
if from_point:
pyautogui.moveTo(int(from_point["x"]), int(from_point["y"]))
pyautogui.dragTo(int(payload["to"]["x"]), int(payload["to"]["y"]), duration=0.2, button="left")
json_output({"ok": True, "result": True})
return 0
if command == "move_mouse":
pyautogui.moveTo(int(payload["x"]), int(payload["y"]))
json_output({"ok": True, "result": True})
return 0
if command == "scroll":
scroll(int(payload["x"]), int(payload["y"]), int(payload.get("deltaX") or 0), int(payload.get("deltaY") or 0))
json_output({"ok": True, "result": True})
return 0
if command == "mouse_down":
pyautogui.mouseDown(button="left")
json_output({"ok": True, "result": True})
return 0
if command == "mouse_up":
pyautogui.mouseUp(button="left")
json_output({"ok": True, "result": True})
return 0
if command == "cursor_position":
x, y = pyautogui.position()
json_output({"ok": True, "result": {"x": int(x), "y": int(y)}})
return 0
if command == "frontmost_app":
json_output({"ok": True, "result": frontmost_app()})
return 0
if command == "app_under_point":
json_output({"ok": True, "result": app_under_point(int(payload["x"]), int(payload["y"]))})
return 0
if command == "list_installed_apps":
json_output({"ok": True, "result": installed_apps()})
return 0
if command == "list_running_apps":
json_output({"ok": True, "result": running_apps()})
return 0
if command == "open_app":
open_app(str(payload["bundleId"]))
json_output({"ok": True, "result": True})
return 0
if command == "read_clipboard":
json_output({"ok": True, "result": read_clipboard()})
return 0
if command == "write_clipboard":
write_clipboard(str(payload.get("text") or ""))
json_output({"ok": True, "result": True})
return 0
if command == "paste_clipboard":
paste_clipboard()
json_output({"ok": True, "result": True})
return 0
error_output(f"Unknown command: {command}", code="bad_command")
return 2
except Exception as exc:
error_output(str(exc))
return 1
if __name__ == "__main__":
raise SystemExit(main())
-6
View File
@@ -1,6 +0,0 @@
mss>=9.0.2,<10
Pillow>=11.3.0,<12
pyautogui>=0.9.54
pyobjc-core>=11.1
pyobjc-framework-Cocoa>=11.1
pyobjc-framework-Quartz>=11.1
+304 -229
View File
@@ -1,42 +1,46 @@
#!/usr/bin/env python3
"""Cross-platform tests for mac_helper.py and win_helper.py.
"""Tests for win_helper.py.
Tests the platform-independent parts (JSON protocol, key mapping, capture logic)
without requiring platform-specific dependencies. Can run on any OS with pytest.
macOS routes every Computer Use command to the signed native `cu-helper`
daemon — `helperBridge` refuses to fall back to Python — so `mac_helper.py`
was unreachable and has been deleted. This file therefore covers the Windows
helper only.
Most tests here are static (they read the source) rather than executed,
because the runtime deps (pywin32, pyautogui, mss) are Windows-only and CI
runs on macOS. Static coverage is enough for what actually regresses: the
guards getting dropped, inverted, or quietly bypassed.
Usage:
python -m pytest runtime/test_helpers.py -v
# or simply:
python runtime/test_helpers.py
"""
from __future__ import annotations
import ast
import json
import subprocess
import sys
import unittest
from pathlib import Path
from unittest.mock import patch, MagicMock
# Determine which helper to test based on current platform
IS_WINDOWS = sys.platform == "win32"
IS_MACOS = sys.platform == "darwin"
RUNTIME_DIR = Path(__file__).parent
MAC_HELPER = RUNTIME_DIR / "mac_helper.py"
WIN_HELPER = RUNTIME_DIR / "win_helper.py"
CURSOR_BADGE = RUNTIME_DIR / "win_cursor_badge.py"
def _win_source() -> str:
return WIN_HELPER.read_text(encoding="utf-8")
class TestKeyMap(unittest.TestCase):
"""Test the KEY_MAP and normalize_key function — platform-independent logic."""
"""KEY_MAP translates macOS key names to Windows ones."""
def _load_key_map(self, helper_path: Path) -> dict[str, str]:
"""Extract KEY_MAP from a helper by importing it with mocked deps."""
# Read the file and extract just the KEY_MAP dict
source = helper_path.read_text()
# Find KEY_MAP definition
source = helper_path.read_text(encoding="utf-8")
start = source.index("KEY_MAP = {")
# Find the matching closing brace
depth = 0
for i, ch in enumerate(source[start:], start):
if ch == "{":
@@ -46,106 +50,40 @@ class TestKeyMap(unittest.TestCase):
if depth == 0:
end = i + 1
break
key_map_source = source[start:end]
ns: dict = {}
exec(key_map_source, ns)
exec(source[start:end], ns)
return ns["KEY_MAP"]
def test_mac_key_map_exists(self):
if not MAC_HELPER.exists():
self.skipTest("mac_helper.py not found")
km = self._load_key_map(MAC_HELPER)
self.assertIn("cmd", km)
self.assertIn("ctrl", km)
self.assertEqual(km["cmd"], "command")
self.assertEqual(km["alt"], "option")
def test_win_key_map_exists(self):
if not WIN_HELPER.exists():
self.skipTest("win_helper.py not found")
km = self._load_key_map(WIN_HELPER)
self.assertIn("cmd", km)
self.assertIn("ctrl", km)
# Windows maps cmd/command/meta to 'win' key
# The mapping that matters: a model trained on macOS emits "cmd", and
# on Windows that has to become "win", not silently stay "cmd".
self.assertEqual(km["cmd"], "win")
self.assertEqual(km["command"], "win")
self.assertEqual(km["meta"], "win")
# Windows maps alt/option to 'alt'
self.assertEqual(km["alt"], "alt")
self.assertEqual(km["option"], "alt")
def test_common_keys_present_in_both(self):
"""Both helpers must have the same set of key names."""
if not MAC_HELPER.exists() or not WIN_HELPER.exists():
self.skipTest("Both helpers required")
mac_km = self._load_key_map(MAC_HELPER)
win_km = self._load_key_map(WIN_HELPER)
# All keys in mac should be in win and vice versa
self.assertEqual(set(mac_km.keys()), set(win_km.keys()),
"KEY_MAP keys must be identical across platforms")
def test_all_alphabet_keys(self):
"""All a-z keys should map to themselves."""
for helper in [MAC_HELPER, WIN_HELPER]:
if not helper.exists():
continue
km = self._load_key_map(helper)
for char in "abcdefghijklmnopqrstuvwxyz":
self.assertEqual(km[char], char, f"{helper.name}: {char} should map to itself")
km = self._load_key_map(WIN_HELPER)
for ch in "abcdefghijklmnopqrstuvwxyz":
self.assertIn(ch, km)
def test_all_digit_keys(self):
"""All 0-9 keys should map to themselves."""
for helper in [MAC_HELPER, WIN_HELPER]:
if not helper.exists():
continue
km = self._load_key_map(helper)
for digit in "0123456789":
self.assertEqual(km[digit], digit, f"{helper.name}: {digit} should map to itself")
def test_function_keys(self):
"""F1-F12 should map to themselves."""
for helper in [MAC_HELPER, WIN_HELPER]:
if not helper.exists():
continue
km = self._load_key_map(helper)
for i in range(1, 13):
key = f"f{i}"
self.assertEqual(km[key], key, f"{helper.name}: {key} should map to itself")
km = self._load_key_map(WIN_HELPER)
for d in "0123456789":
self.assertIn(d, km)
class TestJSONProtocol(unittest.TestCase):
"""Test that both helpers follow the same JSON command protocol."""
def _get_helper(self) -> Path:
"""Get the appropriate helper for the current platform."""
if IS_WINDOWS and WIN_HELPER.exists():
return WIN_HELPER
if IS_MACOS and MAC_HELPER.exists():
return MAC_HELPER
return MAC_HELPER if MAC_HELPER.exists() else WIN_HELPER
def _parse_main_commands(self, helper_path: Path) -> list[str]:
"""Extract all command names from the main() dispatcher."""
source = helper_path.read_text()
source = helper_path.read_text(encoding="utf-8")
commands = []
for line in source.splitlines():
stripped = line.strip()
if stripped.startswith('if command == "'):
cmd = stripped.split('"')[1]
commands.append(cmd)
commands.append(stripped.split('"')[1])
return commands
def test_both_helpers_same_commands(self):
"""Both helpers must support the exact same set of commands."""
if not MAC_HELPER.exists() or not WIN_HELPER.exists():
self.skipTest("Both helpers required")
mac_cmds = set(self._parse_main_commands(MAC_HELPER))
win_cmds = set(self._parse_main_commands(WIN_HELPER))
self.assertEqual(mac_cmds, win_cmds,
f"Command sets differ.\nOnly in mac: {mac_cmds - win_cmds}\nOnly in win: {win_cmds - mac_cmds}")
def test_expected_commands_exist(self):
"""Core commands should be present in each helper."""
expected = {
"check_permissions", "list_displays", "get_display_size",
"screenshot", "resolve_prepare_capture", "zoom",
@@ -156,167 +94,304 @@ class TestJSONProtocol(unittest.TestCase):
"list_installed_apps", "list_running_apps", "open_app",
"read_clipboard", "write_clipboard", "paste_clipboard",
}
for helper in [MAC_HELPER, WIN_HELPER]:
if not helper.exists():
continue
cmds = set(self._parse_main_commands(helper))
missing = expected - cmds
self.assertFalse(missing,
f"{helper.name} missing commands: {missing}")
cmds = set(self._parse_main_commands(WIN_HELPER))
self.assertFalse(expected - cmds,
f"win_helper.py missing commands: {expected - cmds}")
@unittest.skipUnless(IS_WINDOWS, "requires Windows runtime deps")
def test_unknown_command_returns_error(self):
"""Running a non-existent command should return a JSON error."""
helper = self._get_helper()
if not helper.exists():
self.skipTest("No helper found")
# On macOS without venv, mac_helper.py may fail at import (AppKit);
# on Windows without venv, win_helper.py may fail at import (win32gui).
# Only test if the helper can actually import.
check = subprocess.run(
[sys.executable, "-c", f"import importlib.util; "
f"spec = importlib.util.spec_from_file_location('h', '{helper}')"],
capture_output=True, text=True
)
result = subprocess.run(
[sys.executable, str(helper), "nonexistent_command_xyz"],
capture_output=True, text=True
[sys.executable, str(WIN_HELPER), "nonexistent_command_xyz"],
capture_output=True, text=True,
)
if result.returncode == 1 and not result.stdout.strip():
# Import failed — platform deps missing, skip this test
self.skipTest(f"Cannot run {helper.name} on this platform (missing deps)")
# Should exit with code 2
self.skipTest("missing platform deps")
self.assertEqual(result.returncode, 2)
parsed = json.loads(result.stdout.strip())
self.assertFalse(parsed["ok"])
self.assertEqual(parsed["error"]["code"], "bad_command")
class TestHelperOutputFormat(unittest.TestCase):
"""Test the JSON output helpers are consistent."""
class TestMutatingCommandsAreGuarded(unittest.TestCase):
"""Every command that injects input must pass through the guards.
def test_json_output_function_exists(self):
"""Both helpers should define json_output and error_output."""
for helper in [MAC_HELPER, WIN_HELPER]:
if not helper.exists():
These are static-source tests on purpose. The failure being guarded against
is someone adding an eleventh mutating verb and wiring it like the ten that
came before — at which point it silently has no lease and no reachability
check. A runtime test would need Windows and would only cover the verbs it
thought to enumerate; reading the dispatcher catches the new one.
"""
# Kept as a literal, deliberately duplicating MUTATING_COMMANDS in the
# helper. If the two drift the test fails, which is the point: the set is
# a security boundary and should not be edited casually on one side only.
MUTATING = {
"click", "drag", "move_mouse", "scroll",
"mouse_down", "mouse_up",
"key", "hold_key", "type",
"paste_clipboard",
}
def _module_constant(self, name: str) -> set[str]:
"""Read a module-level frozenset/set constant without importing."""
tree = ast.parse(_win_source())
for node in tree.body:
if isinstance(node, ast.Assign):
for target in node.targets:
if isinstance(target, ast.Name) and target.id == name:
return set(ast.literal_eval(
node.value.args[0]
if isinstance(node.value, ast.Call)
else node.value
))
raise AssertionError(f"{name} not found in win_helper.py")
def test_mutating_command_set_matches_this_test(self):
self.assertEqual(self._module_constant("MUTATING_COMMANDS"), self.MUTATING)
def test_coordinate_commands_are_a_subset(self):
coords = self._module_constant("COORDINATE_COMMANDS")
self.assertTrue(coords <= self.MUTATING)
# `key`/`type` go wherever focus is and have no point to validate.
# Asserting their absence keeps someone from "fixing" the coordinate
# guard by adding them and then dereferencing an x/y that isn't there.
self.assertNotIn("key", coords)
self.assertNotIn("type", coords)
def test_every_mutating_branch_finalizes_the_lease(self):
"""No mutating branch may answer with a bare json_output.
This is the specific regression: `_finish` is what runs the post-action
interference check, so a branch that writes its own success response
reports "Action completed" for input that may have collided with the
user's own typing.
"""
source = _win_source()
start = source.index(' try:\n command = args.command')
end = source.index(' error_output(f"Unknown command: {command}"')
dispatcher = source[start:end]
blocks = dispatcher.split('if command == "')
for block in blocks[1:]:
name = block.split('"')[0]
if name not in self.MUTATING:
continue
source = helper.read_text()
self.assertIn("def json_output(", source,
f"{helper.name} missing json_output function")
self.assertIn("def error_output(", source,
f"{helper.name} missing error_output function")
body = block.split("if command ==")[0]
self.assertIn(
"_finish(lease,", body,
f'"{name}" must return through _finish so the lease is checked',
)
self.assertNotIn(
'json_output({"ok": True', body,
f'"{name}" writes its own success response, bypassing the lease',
)
def test_main_entry_point(self):
"""Both helpers should have the standard main entry point."""
for helper in [MAC_HELPER, WIN_HELPER]:
if not helper.exists():
continue
source = helper.read_text()
self.assertIn('if __name__ == "__main__":', source,
f"{helper.name} missing __main__ guard")
self.assertIn("def main()", source,
f"{helper.name} missing main() function")
def test_guards_run_before_any_injection(self):
"""acquire() must precede the dispatch chain, not follow it."""
source = _win_source()
acquire = source.index("lease.acquire()")
first_branch = source.index(' if command == "check_permissions"')
self.assertLess(
acquire, first_branch,
"the lease must be acquired before any command branch runs",
)
class TestWinHelperPermissions(unittest.TestCase):
"""Windows-specific: permissions should always return True."""
class TestInterferenceDetection(unittest.TestCase):
def test_uses_getlastinputinfo_not_an_event_hook(self):
"""The signal must stay permission-free.
A low-level input hook would read the same events, but installing one
is exactly the kind of thing that gets an app flagged, and it is not
needed: GetLastInputInfo answers the only question we ask.
"""
source = _win_source()
self.assertIn("GetLastInputInfo", source)
self.assertNotIn("SetWindowsHookEx", source)
def test_distinguishes_did_not_run_from_outcome_unknown(self):
"""The two interference verdicts must stay distinct.
Collapsing them is a real hazard: `user_interference` means nothing
happened and a retry is safe, while `user_interference_result_unknown`
means input already went out and a retry could double-apply it. On a
play/pause toggle those differ by exactly one wrong outcome.
"""
source = _win_source()
self.assertIn('"user_interference"', source)
self.assertIn('user_interference_result_unknown', source)
acquire_start = source.index(" def acquire(self)")
acquire_body = source[acquire_start:source.index(" def finalize(self)")]
self.assertNotIn("result_unknown", acquire_body,
"a pre-action refusal means nothing ran; the outcome is known")
finalize_body = source[source.index(" def finalize(self)"):]
finalize_body = finalize_body[:finalize_body.index("\n\n\n")]
self.assertIn("result_unknown", finalize_body,
"post-action interference leaves the outcome unknown")
def test_counter_read_failure_does_not_block_actions(self):
"""An unreadable counter must fail open, not brick the feature.
Precedent from the macOS side: an earlier build required an Input
Monitoring grant that onboarding never asked for, so every mutating
action failed on a correctly set-up machine. A safety layer that turns
the product off is not safety.
"""
source = _win_source()
fn_start = source.index("def last_physical_input_tick()")
body = source[fn_start:source.index("class UserInterference")]
self.assertIn("return 0", body)
# And the comparisons must treat 0 as "no reading", never as a tick
# value that happens to differ from the next one.
self.assertIn("if before and after and before != after:", source)
def test_synthetic_input_must_not_trip_the_detector(self):
"""Documented invariant: SendInput does not advance GetLastInputInfo.
If this ever stopped holding, every agent action would abort itself and
the feature would look randomly broken. Pinning the claim in a test
keeps it from being quietly deleted as a stale comment.
"""
source = _win_source()
self.assertIn("is NOT advanced by", source)
class TestDeliveryGuards(unittest.TestCase):
def test_offscreen_point_is_refused(self):
source = _win_source()
self.assertIn("point_outside_display", source)
self.assertIn("def ensure_point_on_screen", source)
def test_unreachable_window_is_refused(self):
source = _win_source()
self.assertIn("target_window_offscreen", source)
self.assertIn("def ensure_target_window_reachable", source)
def test_reachability_check_sees_minimized_windows(self):
"""It must NOT reuse list_windows().
`list_windows()` filters out invisible and zero-area windows — exactly
the states the guard needs to observe in order to refuse. An earlier
draft of this guard did reuse it, matched nothing, and passed
everything. The enumeration has to be its own.
"""
tree = ast.parse(_win_source())
fn = next(
node for node in ast.walk(tree)
if isinstance(node, ast.FunctionDef) and node.name == "_windows_for_bundle"
)
# Walk the AST rather than the text, so the explanatory docstring
# (which names list_windows to say why it is NOT used) cannot satisfy
# or break the assertion.
called = {
n.func.id for n in ast.walk(fn)
if isinstance(n, ast.Call) and isinstance(n.func, ast.Name)
}
self.assertNotIn("list_windows", called)
source = _win_source()
body = source[source.index("def _windows_for_bundle"):
source.index("def ensure_target_window_reachable")]
self.assertIn("EnumWindows", body)
self.assertIn("SW_SHOWMINIMIZED", source)
def test_refusals_carry_a_machine_readable_code(self):
source = _win_source()
self.assertIn("class DeliveryRefused", source)
self.assertIn("error_output(str(exc), code=exc.code)", source)
def test_refusal_says_the_action_was_not_sent(self):
"""The message must state that nothing happened.
"Could not reach the window" reads like a warning attached to an action
that still went out. The model needs to know the action did not happen,
or it will assume it did and move on.
"""
source = _win_source()
self.assertIn("was NOT sent", source)
class TestCursorBadge(unittest.TestCase):
"""The Windows badge annotates the real cursor; it does not replace it."""
def test_badge_script_exists(self):
self.assertTrue(CURSOR_BADGE.exists())
def test_badge_is_click_through_and_never_takes_focus(self):
"""Any of these missing turns the badge into an obstacle.
Without WS_EX_TRANSPARENT it eats the clicks it is meant to describe;
without WS_EX_NOACTIVATE it steals focus from the app being driven —
which would break the very action it is annotating.
"""
source = CURSOR_BADGE.read_text(encoding="utf-8")
tree = ast.parse(source)
create = next(
node for node in ast.walk(tree)
if isinstance(node, ast.FunctionDef) and node.name == "create"
)
# Read the names actually combined into the window's ex-style, not
# merely the ones defined somewhere in the file. A constant can be
# defined and then left out of CreateWindowExW — which is exactly how
# a click-through window quietly becomes a click-eating one.
used = {
n.id for n in ast.walk(create)
if isinstance(n, ast.Name)
}
for style in ("WS_EX_LAYERED", "WS_EX_TRANSPARENT",
"WS_EX_NOACTIVATE", "WS_EX_TOOLWINDOW"):
self.assertIn(
style, used,
f"{style} must be passed to CreateWindowExW, not just defined",
)
self.assertIn("SW_SHOWNOACTIVATE", used)
def test_badge_does_not_draw_a_second_pointer(self):
"""Windows has one real cursor and SendInput moves it.
Drawing a fake pointer alongside it would show the user two cursors,
one of which is a lie about where the click will land. The macOS design
does not transfer, and the source says so explicitly.
"""
source = CURSOR_BADGE.read_text(encoding="utf-8")
self.assertIn("annotation", source.lower())
def test_badge_exits_with_its_parent(self):
"""An orphaned badge is worse than none.
It would sit on screen claiming the agent is controlling the mouse
after the agent is gone. Tying it to stdin covers the parent being
killed, not just exiting cleanly.
"""
source = CURSOR_BADGE.read_text(encoding="utf-8")
self.assertIn("stdin", source)
class TestPermissions(unittest.TestCase):
def test_check_permissions_always_granted(self):
"""On Windows, permissions are not needed — should always be True."""
if not WIN_HELPER.exists():
self.skipTest("win_helper.py not found")
# Extract and exec just the check_permissions function
source = WIN_HELPER.read_text()
# Find the function
self.assertIn("def check_permissions()", source)
# The function should return both as True
# We can verify by reading the source
"""Windows has no TCC equivalent for input injection or capture."""
source = _win_source()
start = source.index("def check_permissions()")
# Find next def or end
rest = source[start:]
lines = rest.split("\n")
func_lines = [lines[0]]
for line in lines[1:]:
if line and not line[0].isspace() and not line.startswith("#"):
break
func_lines.append(line)
func_source = "\n".join(func_lines)
self.assertIn('"accessibility": True', func_source)
self.assertIn('"screenRecording": True', func_source)
body = source[start:start + 400]
self.assertIn('"accessibility": True', body)
self.assertIn('"screenRecording": True', body)
class TestMacHelperPermissions(unittest.TestCase):
"""macOS helper permission detection should use the official trust API."""
class TestSourceIntegrity(unittest.TestCase):
def test_helper_parses(self):
ast.parse(_win_source())
def test_check_permissions_uses_ax_api_instead_of_system_events(self):
if not MAC_HELPER.exists():
self.skipTest("mac_helper.py not found")
def test_badge_parses(self):
ast.parse(CURSOR_BADGE.read_text(encoding="utf-8"))
source = MAC_HELPER.read_text()
self.assertIn("def detect_accessibility_permission()", source)
self.assertIn("AXIsProcessTrusted", source)
start = source.index("def check_permissions()")
rest = source[start:]
lines = rest.split("\n")
func_lines = [lines[0]]
for line in lines[1:]:
if line and not line[0].isspace() and not line.startswith("#"):
break
func_lines.append(line)
func_source = "\n".join(func_lines)
self.assertIn("detect_accessibility_permission()", func_source)
self.assertNotIn('tell application "System Events"', func_source)
def test_clipboard_shortcuts_use_osascript_path(self):
if not MAC_HELPER.exists():
self.skipTest("mac_helper.py not found")
source = MAC_HELPER.read_text()
self.assertIn("def paste_clipboard()", source)
self.assertIn('send_keystroke_via_osascript("v", ["command"])', source)
self.assertIn('if parts == ["command", "v"]:', source)
self.assertIn('elif parts == ["command", "a"]:', source)
class TestCrossPlatformFunctions(unittest.TestCase):
"""Test functions that are identical between both helpers."""
def _get_function_body(self, helper_path: Path, func_name: str) -> str:
"""Extract a function's body (code lines only, no comments/blanks)."""
source = helper_path.read_text()
marker = f"def {func_name}("
if marker not in source:
return ""
start = source.index(marker)
rest = source[start:]
lines = rest.split("\n")
func_lines = [lines[0]]
for line in lines[1:]:
# Stop at next top-level def/class or non-indented non-empty line
stripped = line.strip()
if line and not line[0].isspace() and stripped and not stripped.startswith("#"):
break
# Skip comments and blank lines for comparison
if stripped.startswith("#") or not stripped:
continue
func_lines.append(line)
return " ".join(" ".join(func_lines).split())
def test_input_functions_identical(self):
"""Input action functions (click, scroll, etc.) should be identical."""
if not MAC_HELPER.exists() or not WIN_HELPER.exists():
self.skipTest("Both helpers required")
for func in ["click", "scroll", "hold_keys", "type_text"]:
mac_src = self._get_function_body(MAC_HELPER, func)
win_src = self._get_function_body(WIN_HELPER, func)
self.assertEqual(mac_src, win_src,
f"{func} should be identical across platforms")
def test_retired_mac_helper_is_not_referenced(self):
"""macOS is native-only; a lingering reference invites a false fallback."""
self.assertFalse((RUNTIME_DIR / "mac_helper.py").exists())
self.assertNotIn("mac_helper", _win_source())
if __name__ == "__main__":
unittest.main()
unittest.main(verbosity=2)
+253
View File
@@ -0,0 +1,253 @@
#!/usr/bin/env python3
"""Windows agent-activity badge — a click-through marker that follows the cursor.
WHY THIS IS NOT THE macOS VIRTUAL CURSOR
----------------------------------------
On macOS the helper never moves the real pointer: `CGEvent.postToPid` carries
the click coordinate as metadata, so the drawn cursor IS the only cursor the
user sees move. It is a *replacement*.
Windows has no per-process event delivery. `pyautogui` bottoms out in
`SendInput`, which warps the one real cursor the user's hand is also on. We
cannot avoid that, so drawing a second fake pointer would be actively harmful:
two pointers, one of them a lie, with no way to tell which one the OS is
actually going to click with.
So this badge is an *annotation*, not a replacement. It rides just off the real
cursor and answers exactly one question the user cannot otherwise answer:
"is this thing moving because of me, or because of the agent?" On Windows that
question has real stakes — the agent is holding the user's mouse, and the user
needs to know before they grab it back mid-action.
Runs as its own process because the Windows helper is a stateless one-shot CLI:
every command exits, so nothing in it can own a window across actions.
The window is WS_EX_LAYERED | WS_EX_TRANSPARENT | WS_EX_NOACTIVATE — it never
takes focus, never appears in the taskbar or Alt-Tab, and passes every click
through to whatever is underneath. It cannot intercept the input it exists to
describe.
Usage:
python win_cursor_badge.py --label "Claude" # runs until stdin closes
"""
from __future__ import annotations
import argparse
import ctypes
import sys
import threading
from ctypes import wintypes
user32 = ctypes.windll.user32
gdi32 = ctypes.windll.gdi32
kernel32 = ctypes.windll.kernel32
WS_EX_LAYERED = 0x00080000
WS_EX_TRANSPARENT = 0x00000020
WS_EX_TOPMOST = 0x00000008
WS_EX_TOOLWINDOW = 0x00000080
WS_EX_NOACTIVATE = 0x08000000
WS_POPUP = 0x80000000
SW_SHOWNOACTIVATE = 4
HWND_TOPMOST = -1
SWP_NOACTIVATE = 0x0010
SWP_NOSIZE = 0x0001
SWP_NOZORDER = 0x0004
LWA_COLORKEY = 0x00000001
LWA_ALPHA = 0x00000002
WM_DESTROY = 0x0002
WM_PAINT = 0x000F
BADGE_W = 132
BADGE_H = 30
CURSOR_OFFSET_X = 18
CURSOR_OFFSET_Y = 18
# Chroma key: pixels of this exact colour become fully transparent. Picked to
# be a colour nothing in the badge draws, so only the intended shape shows.
TRANSPARENT_KEY = 0x00FF00FF
class POINT(ctypes.Structure):
_fields_ = [("x", wintypes.LONG), ("y", wintypes.LONG)]
class RECT(ctypes.Structure):
_fields_ = [
("left", wintypes.LONG),
("top", wintypes.LONG),
("right", wintypes.LONG),
("bottom", wintypes.LONG),
]
class PAINTSTRUCT(ctypes.Structure):
_fields_ = [
("hdc", wintypes.HDC),
("fErase", wintypes.BOOL),
("rcPaint", RECT),
("fRestore", wintypes.BOOL),
("fIncUpdate", wintypes.BOOL),
("rgbReserved", ctypes.c_byte * 32),
]
class WNDCLASS(ctypes.Structure):
_fields_ = [
("style", wintypes.UINT),
("lpfnWndProc", ctypes.WINFUNCTYPE(
ctypes.c_long, wintypes.HWND, wintypes.UINT,
wintypes.WPARAM, wintypes.LPARAM)),
("cbClsExtra", ctypes.c_int),
("cbWndExtra", ctypes.c_int),
("hInstance", wintypes.HINSTANCE),
("hIcon", wintypes.HICON),
("hCursor", wintypes.HANDLE),
("hbrBackground", wintypes.HBRUSH),
("lpszMenuName", wintypes.LPCWSTR),
("lpszClassName", wintypes.LPCWSTR),
]
WNDPROC = ctypes.WINFUNCTYPE(
ctypes.c_long, wintypes.HWND, wintypes.UINT, wintypes.WPARAM, wintypes.LPARAM
)
class CursorBadge:
def __init__(self, label: str) -> None:
self.label = label
self.hwnd: int | None = None
self._stop = threading.Event()
# Held on the instance because ctypes does not keep the trampoline
# alive on its own; letting it be collected turns the next window
# message into a crash inside the message pump.
self._wndproc = WNDPROC(self._on_message)
def _on_message(self, hwnd, msg, wparam, lparam):
if msg == WM_PAINT:
self._paint(hwnd)
return 0
if msg == WM_DESTROY:
user32.PostQuitMessage(0)
return 0
return user32.DefWindowProcW(hwnd, msg, wparam, lparam)
def _paint(self, hwnd: int) -> None:
ps = PAINTSTRUCT()
hdc = user32.BeginPaint(hwnd, ctypes.byref(ps))
try:
rect = RECT(0, 0, BADGE_W, BADGE_H)
# Fill with the chroma key first: everything we do not draw over
# becomes transparent, which is what gives the badge its shape.
key_brush = gdi32.CreateSolidBrush(TRANSPARENT_KEY)
user32.FillRect(hdc, ctypes.byref(rect), key_brush)
gdi32.DeleteObject(key_brush)
body = RECT(0, 0, BADGE_W, BADGE_H)
bg = gdi32.CreateSolidBrush(0x00734B23) # BGR: a muted blue
user32.FillRect(hdc, ctypes.byref(body), bg)
gdi32.DeleteObject(bg)
gdi32.SetBkMode(hdc, 1) # TRANSPARENT
gdi32.SetTextColor(hdc, 0x00FFFFFF)
text = f" {self.label} is controlling"
user32.DrawTextW(
hdc, text, len(text), ctypes.byref(body),
0x00000004 | 0x00000100, # DT_VCENTER | DT_SINGLELINE
)
finally:
user32.EndPaint(hwnd, ctypes.byref(ps))
def create(self) -> None:
hinst = kernel32.GetModuleHandleW(None)
class_name = "CcHahaAgentCursorBadge"
wc = WNDCLASS()
wc.lpfnWndProc = self._wndproc
wc.hInstance = hinst
wc.lpszClassName = class_name
wc.hbrBackground = 0
wc.hCursor = 0
user32.RegisterClassW(ctypes.byref(wc))
self.hwnd = user32.CreateWindowExW(
WS_EX_LAYERED | WS_EX_TRANSPARENT | WS_EX_TOPMOST
| WS_EX_TOOLWINDOW | WS_EX_NOACTIVATE,
class_name, None, WS_POPUP,
0, 0, BADGE_W, BADGE_H,
None, None, hinst, None,
)
if not self.hwnd:
raise OSError("CreateWindowExW failed for the cursor badge")
user32.SetLayeredWindowAttributes(
self.hwnd, TRANSPARENT_KEY, 225, LWA_COLORKEY | LWA_ALPHA
)
user32.ShowWindow(self.hwnd, SW_SHOWNOACTIVATE)
def _follow_cursor(self) -> None:
"""Reposition the badge next to the real pointer, ~60fps."""
pt = POINT()
while not self._stop.is_set():
try:
if user32.GetCursorPos(ctypes.byref(pt)) and self.hwnd:
user32.SetWindowPos(
self.hwnd, HWND_TOPMOST,
pt.x + CURSOR_OFFSET_X, pt.y + CURSOR_OFFSET_Y,
0, 0, SWP_NOACTIVATE | SWP_NOSIZE,
)
except Exception:
# The badge is advisory. It must never be the reason an action
# fails, so every error here is swallowed and the loop retries.
pass
self._stop.wait(0.016)
def _wait_for_stdin_close(self) -> None:
"""Exit when the parent goes away.
The badge outliving its parent would leave a permanent 'the agent is
controlling your mouse' claim on screen with nothing behind it. Reading
stdin to EOF ties this process's lifetime to the parent's, including
the case where the parent is killed rather than exiting cleanly.
"""
try:
for _ in sys.stdin:
pass
except Exception:
pass
self.stop()
def stop(self) -> None:
self._stop.set()
if self.hwnd:
user32.PostMessageW(self.hwnd, WM_DESTROY, 0, 0)
def run(self) -> int:
self.create()
threading.Thread(target=self._follow_cursor, daemon=True).start()
threading.Thread(target=self._wait_for_stdin_close, daemon=True).start()
msg = wintypes.MSG()
while user32.GetMessageW(ctypes.byref(msg), None, 0, 0) > 0:
user32.TranslateMessage(ctypes.byref(msg))
user32.DispatchMessageW(ctypes.byref(msg))
return 0
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--label", default="Claude")
args = parser.parse_args()
if sys.platform != "win32":
print("win_cursor_badge.py is Windows-only", file=sys.stderr)
return 1
return CursorBadge(args.label).run()
if __name__ == "__main__":
raise SystemExit(main())
+411 -26
View File
@@ -1,8 +1,28 @@
#!/usr/bin/env python3
"""Windows Computer Use helper — same JSON protocol as mac_helper.py.
"""Windows Computer Use helper.
Uses win32gui / win32api / win32process / psutil / pyperclip / screeninfo
to replicate macOS-specific Quartz/AppKit functionality on Windows.
Uses win32gui / win32api / win32process / psutil / pyperclip / screeninfo /
pyautogui to provide, on Windows, the JSON command protocol the native macOS
`cu-helper` daemon speaks. macOS is native-only — there is no Python path there
— so this is the sole implementation of that protocol in Python.
One difference is not an implementation detail and shapes everything below:
macOS delivers input with `CGEvent.postToPid`, straight into the target
process, leaving the real cursor and the foreground app alone. Windows has no
equivalent. `pyautogui` bottoms out in `SendInput`, which injects into the one
system-wide input stream and warps the one real cursor. The agent therefore
shares the mouse and keyboard with the user, and cannot verify that anything
it sent arrived.
Hence the two mechanisms that have no macOS counterpart:
* `ForegroundLease` aborts when physical input overlaps an action, because
interleaved streams produce clicks neither party intended.
* `ensure_point_on_screen` / `ensure_target_window_reachable` refuse to send
at all when delivery is already known to be impossible.
Both exist because `SendInput` reports success unconditionally, and "Action
completed" for input that went nowhere is worse than an error.
"""
from __future__ import annotations
@@ -176,7 +196,7 @@ def choose_display(display_id: int | None) -> dict[str, Any]:
# ---------------------------------------------------------------------------
# Screen capture (mss — cross-platform, identical to mac_helper)
# Screen capture (mss)
# ---------------------------------------------------------------------------
def capture_display(display_id: int | None, resize: tuple[int, int] | None = None) -> dict[str, Any]:
@@ -563,6 +583,145 @@ def paste_clipboard() -> None:
pyautogui.hotkey("ctrl", "v", interval=0.02)
# ---------------------------------------------------------------------------
# Physical input interference detection
# ---------------------------------------------------------------------------
#
# Why this exists at all, and why it is stricter than the macOS version.
#
# On macOS the helper posts events straight into the target process with
# `CGEvent.postToPid`, so agent input and human input never share a channel:
# the epoch monitor there is a safety net for an unlikely race.
#
# Windows has no such API. `pyautogui` bottoms out in `SendInput`, which
# injects into the ONE system-wide input stream and warps the ONE real cursor.
# The agent and the user are therefore holding the same mouse. If the user
# reaches for it mid-action the two streams interleave, and the resulting
# click lands somewhere neither of them intended. Detection is not a nicety
# here — it is the only thing standing between "the agent typed into the wrong
# window" and an abort.
#
# `GetLastInputInfo` is the right signal for this: it reports the tick of the
# last PHYSICAL input event, requires no privileges and no TCC-style grant,
# and — measured, and asserted by test_helpers.py — is NOT advanced by
# `SendInput` injection, so the agent cannot trip its own detector.
import ctypes
from ctypes import wintypes
class _LASTINPUTINFO(ctypes.Structure):
_fields_ = [("cbSize", wintypes.UINT), ("dwTime", wintypes.DWORD)]
def last_physical_input_tick() -> int:
"""Tick count of the last physical keyboard/mouse event.
Returns 0 when unavailable so callers fail OPEN on the read itself: a
helper that refused to act because it could not query an optional Win32
counter would be broken in a much more visible way than one that acted.
Interference is only ever reported on two SUCCESSFUL reads that differ.
"""
try:
info = _LASTINPUTINFO()
info.cbSize = ctypes.sizeof(_LASTINPUTINFO)
if not ctypes.windll.user32.GetLastInputInfo(ctypes.byref(info)):
return 0
return int(info.dwTime)
except Exception:
return 0
class UserInterference(RuntimeError):
"""The user touched the physical mouse or keyboard during an action."""
def __init__(self, message: str, code: str = "user_interference") -> None:
super().__init__(message)
self.code = code
def _foreground_window_pid() -> int | None:
try:
import win32gui
import win32process
hwnd = win32gui.GetForegroundWindow()
if not hwnd:
return None
_, pid = win32process.GetWindowThreadProcessId(hwnd)
return int(pid)
except Exception:
return None
class ForegroundLease:
"""Guards one mutating action against concurrent physical input.
Evidence is sampled in a fixed order — tick, foreground identity, tick —
both before and after the action, so a single observation cannot straddle
a change it fails to notice. Same shape as `ForegroundLease.swift`.
The asymmetry between the two failure modes is deliberate and is the whole
point of the class:
* interference BEFORE the action -> `user_interference`. Nothing ran.
The caller may safely retry.
* interference DURING the action -> `user_interference_result_unknown`.
Injection already went into the shared input stream and we cannot know
how much of it landed, or where. Retrying could double-apply it. The
error says so rather than guessing.
"""
def __init__(self) -> None:
self.tick: int = 0
self.pid: int | None = None
def acquire(self) -> None:
before = last_physical_input_tick()
pid = _foreground_window_pid()
after = last_physical_input_tick()
if before and after and before != after:
raise UserInterference(
"The user was typing or moving the mouse, so the action was "
"not sent. Nothing has changed; it is safe to try again."
)
self.tick = after
self.pid = pid
def finalize(self) -> None:
before = last_physical_input_tick()
pid = _foreground_window_pid()
after = last_physical_input_tick()
if before and after and before != after:
raise UserInterference(
"The user used the mouse or keyboard while this action was "
"running. Because Windows shares one input stream between you "
"and the user, the two may have interleaved and the result is "
"UNKNOWN. Do not repeat the action — take a screenshot and "
"read the current state before deciding anything.",
code="user_interference_result_unknown",
)
if self.tick and after and self.tick != after:
raise UserInterference(
"The user used the mouse or keyboard while this action was "
"running. The result is UNKNOWN — do not repeat the action; "
"take a screenshot and read the current state first.",
code="user_interference_result_unknown",
)
# A foreground change without any physical input is the target app (or
# a background app) stealing activation, not the user. Worth reporting,
# because everything typed after it went somewhere unintended.
if self.pid is not None and pid is not None and self.pid != pid:
raise UserInterference(
"The foreground application changed while this action was "
"running, so input may have gone to the wrong window. The "
"result is UNKNOWN — take a screenshot before continuing.",
code="user_interference_result_unknown",
)
# ---------------------------------------------------------------------------
# Permissions — Windows doesn't have macOS-style TCC
# ---------------------------------------------------------------------------
@@ -577,7 +736,177 @@ def check_permissions() -> dict[str, bool | None]:
# ---------------------------------------------------------------------------
# Input actions (pyautogui — identical to mac_helper)
# Delivery preconditions — refuse rather than report a lie
# ---------------------------------------------------------------------------
#
# `SendInput` always "succeeds": it returns the number of events inserted into
# the input stream, never whether anything acted on them. Click a point behind
# another window and the click lands on THAT window; click a point off-screen
# and it lands nowhere. Either way pyautogui returns cleanly and the helper
# would answer "Action completed".
#
# That specific lie has burned us before on macOS — a session typed into a
# minimized window for a full turn because every action reported success. The
# fix there was to refuse instead of guessing, and the same rule applies here.
class DeliveryRefused(RuntimeError):
def __init__(self, message: str, code: str) -> None:
super().__init__(message)
self.code = code
def _virtual_screen_rect() -> tuple[int, int, int, int] | None:
"""(left, top, right, bottom) across all monitors, or None if unavailable."""
try:
user32 = ctypes.windll.user32
SM_XVIRTUALSCREEN, SM_YVIRTUALSCREEN = 76, 77
SM_CXVIRTUALSCREEN, SM_CYVIRTUALSCREEN = 78, 79
left = user32.GetSystemMetrics(SM_XVIRTUALSCREEN)
top = user32.GetSystemMetrics(SM_YVIRTUALSCREEN)
width = user32.GetSystemMetrics(SM_CXVIRTUALSCREEN)
height = user32.GetSystemMetrics(SM_CYVIRTUALSCREEN)
if width <= 0 or height <= 0:
return None
return (left, top, left + width, top + height)
except Exception:
return None
def ensure_point_on_screen(x: int, y: int) -> None:
"""Refuse coordinates outside every monitor.
Fails OPEN when the metrics are unreadable: an unreadable metric is our
problem, not the caller's, and blocking every action on it would be worse
than the miss it prevents.
"""
rect = _virtual_screen_rect()
if rect is None:
return
left, top, right, bottom = rect
if left <= x < right and top <= y < bottom:
return
raise DeliveryRefused(
f"The point ({x}, {y}) is outside every display "
f"(virtual screen is {left},{top} to {right},{bottom}), so the action "
"was not sent. Take a screenshot to get current coordinates.",
code="point_outside_display",
)
def _window_is_interactable(hwnd: int) -> tuple[bool, str]:
"""(ok, reason) — whether synthetic input can reach this window at all."""
try:
import win32gui
if not win32gui.IsWindow(hwnd):
return False, "the window no longer exists"
if not win32gui.IsWindowVisible(hwnd):
return False, "the window is hidden"
try:
import win32con
placement = win32gui.GetWindowPlacement(hwnd)
if placement and placement[1] == win32con.SW_SHOWMINIMIZED:
return False, "the window is minimized"
except Exception:
pass
rect = win32gui.GetWindowRect(hwnd)
if rect[2] - rect[0] <= 0 or rect[3] - rect[1] <= 0:
return False, "the window has no on-screen area"
return True, ""
except Exception:
# Unreadable window state fails open, same reasoning as above.
return True, ""
def _windows_for_bundle(bundle_id: str) -> list[int]:
"""Every top-level HWND owned by a process whose exe stem matches.
Enumerates directly rather than reusing `list_windows()`, which filters out
invisible and zero-area windows — precisely the states this guard needs to
SEE in order to refuse. Reusing it would make the guard match nothing and
silently pass, which is the failure mode it was written to prevent.
"""
try:
import win32gui
import win32process
import psutil
except Exception:
return []
wanted = bundle_id.strip().lower()
if not wanted:
return []
pids: set[int] = set()
try:
for proc in psutil.process_iter(["pid", "name", "exe"]):
try:
exe_path = proc.info.get("exe") or ""
name = proc.info.get("name") or ""
stem = Path(exe_path).stem if exe_path else Path(name).stem
if stem and stem.lower() == wanted:
pids.add(int(proc.info["pid"]))
except (psutil.NoSuchProcess, psutil.AccessDenied):
continue
except Exception:
return []
if not pids:
return []
handles: list[int] = []
def _collect(hwnd: int, _: Any) -> None:
try:
_, pid = win32process.GetWindowThreadProcessId(hwnd)
if int(pid) in pids:
handles.append(int(hwnd))
except Exception:
return
try:
win32gui.EnumWindows(_collect, None)
except Exception:
return []
return handles
def ensure_target_window_reachable(bundle_id: str | None) -> None:
"""Refuse when the named app has no window that input could reach.
A minimized window is the case that matters: on Windows it has no client
area to hit-test against, so a coordinate click is guaranteed to land on
whatever is underneath it. Reporting success there is exactly the lie this
guard exists to prevent.
Fails OPEN when the app owns no top-level windows at all — that is a
different failure (wrong app name, app not running) which the caller's own
resolution step reports with a better message than this one could.
"""
if not bundle_id:
return
handles = _windows_for_bundle(bundle_id)
if not handles:
return
reasons: list[str] = []
for hwnd in handles:
ok, reason = _window_is_interactable(hwnd)
if ok:
return
if reason:
reasons.append(reason)
detail = reasons[0] if reasons else "it has no on-screen window"
raise DeliveryRefused(
f"The target app has no window that input can reach — {detail}. "
"The action was NOT sent. Restore the window and try again.",
code="target_window_offscreen",
)
# ---------------------------------------------------------------------------
# Input actions (pyautogui → SendInput)
# ---------------------------------------------------------------------------
def click(x: int, y: int, button: str, count: int, modifiers: list[str] | None) -> None:
@@ -629,9 +958,56 @@ def type_text(text: str) -> None:
# ---------------------------------------------------------------------------
# Main dispatcher — exact same command protocol as mac_helper.py
# Main dispatcher — the command protocol the native macOS daemon also speaks
# ---------------------------------------------------------------------------
# Commands that inject into the shared Windows input stream. Kept as one set
# rather than as a guard call inside each branch, because the branches are the
# easy place to forget one — and a forgotten branch is silently unguarded, the
# exact class of bug this whole pass exists to remove.
#
# Mirrors `CommandForegroundPolicy.leasedCommands` on the macOS side.
MUTATING_COMMANDS = frozenset({
"click", "drag", "move_mouse", "scroll",
"mouse_down", "mouse_up",
"key", "hold_key", "type",
"paste_clipboard",
})
# The subset that targets a screen coordinate, and so needs the point itself to
# be reachable. `key`/`type` go to whatever holds focus and have no coordinate
# to check.
COORDINATE_COMMANDS = frozenset({"click", "drag", "move_mouse", "scroll"})
def _coordinate_of(command: str, payload: dict[str, Any]) -> tuple[int, int] | None:
if command not in COORDINATE_COMMANDS:
return None
if command == "drag":
target = payload.get("to") or {}
if "x" in target and "y" in target:
return int(target["x"]), int(target["y"])
return None
if "x" in payload and "y" in payload:
return int(payload["x"]), int(payload["y"])
return None
def _finish(lease: "ForegroundLease | None", result: Any) -> int:
"""Emit the success response for a mutating command, after the lease agrees.
The check runs BEFORE the response is written, and that ordering is the
whole point: once `{"ok": true}` reaches the caller the action is reported
as done, and no later discovery can take that back. A helper that injected
input, then noticed the user had been typing throughout, and still answered
"Action completed" would be lying with a straight face.
"""
if lease is not None:
lease.finalize()
json_output({"ok": True, "result": result})
return 0
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("command")
@@ -639,8 +1015,20 @@ def main() -> int:
args = parser.parse_args()
payload = json.loads(args.payload)
lease: ForegroundLease | None = None
try:
command = args.command
if command in MUTATING_COMMANDS:
point = _coordinate_of(command, payload)
if point is not None:
ensure_point_on_screen(point[0], point[1])
ensure_target_window_reachable(
payload.get("bundleId") or payload.get("app")
)
lease = ForegroundLease()
lease.acquire()
if command == "check_permissions":
perms = check_permissions()
json_output({"ok": True, "result": perms})
@@ -690,43 +1078,34 @@ def main() -> int:
return 0
if command == "key":
key_action(str(payload["keySequence"]), int(payload.get("repeat") or 1))
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
if command == "hold_key":
hold_keys(list(payload.get("keyNames") or []), int(payload.get("durationMs") or 0))
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
if command == "type":
type_text(str(payload.get("text") or ""))
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
if command == "click":
click(int(payload["x"]), int(payload["y"]), str(payload.get("button") or "left"), int(payload.get("count") or 1), payload.get("modifiers"))
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
if command == "drag":
from_point = payload.get("from")
if from_point:
pyautogui.moveTo(int(from_point["x"]), int(from_point["y"]))
pyautogui.dragTo(int(payload["to"]["x"]), int(payload["to"]["y"]), duration=0.2, button="left")
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
if command == "move_mouse":
pyautogui.moveTo(int(payload["x"]), int(payload["y"]))
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
if command == "scroll":
scroll(int(payload["x"]), int(payload["y"]), int(payload.get("deltaX") or 0), int(payload.get("deltaY") or 0))
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
if command == "mouse_down":
pyautogui.mouseDown(button="left")
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
if command == "mouse_up":
pyautogui.mouseUp(button="left")
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
if command == "cursor_position":
x, y = pyautogui.position()
json_output({"ok": True, "result": {"x": int(x), "y": int(y)}})
@@ -756,10 +1135,16 @@ def main() -> int:
return 0
if command == "paste_clipboard":
paste_clipboard()
json_output({"ok": True, "result": True})
return 0
return _finish(lease, True)
error_output(f"Unknown command: {command}", code="bad_command")
return 2
except (UserInterference, DeliveryRefused) as exc:
# A deliberate refusal, not a crash. The code travels so the caller can
# tell "did not run, safe to retry" apart from "ran, outcome unknown" —
# collapsing both into a generic error is how a model ends up repeating
# a toggle it already flipped.
error_output(str(exc), code=exc.code)
return 1
except Exception as exc:
error_output(str(exc))
return 1
+34
View File
@@ -99,3 +99,37 @@ describe('cleanupComputerUseAfterTurn — turn-end overlay hide', () => {
expect(hidden).toBe(1)
})
})
describe('cleanupComputerUseAfterTurn — turn-end cursor badge', () => {
test('drops the Windows badge on a turn that hid nothing', async () => {
// The badge is a standing claim that the agent is holding the mouse. If it
// outlives the turn it is simply false, and the user has no way to tell
// that from a turn still in progress.
let hidden = 0
await cleanupComputerUseAfterTurn(makeCtx(), {
overlayHide: async () => {},
hideCursorBadge: () => { hidden += 1 },
})
expect(hidden).toBe(1)
})
test('drops the badge even when overlayHide hangs past its timeout', async () => {
// Abort paths are exactly when a stuck indicator is most likely and most
// confusing. The badge teardown must not sit behind the daemon's.
let hidden = 0
await cleanupComputerUseAfterTurn(makeCtx(), {
overlayHide: () => new Promise<void>(() => {}),
hideCursorBadge: () => { hidden += 1 },
})
expect(hidden).toBe(1)
})
test('drops the badge even when overlayHide rejects', async () => {
let hidden = 0
await cleanupComputerUseAfterTurn(makeCtx(), {
overlayHide: async () => { throw new Error('daemon gone') },
hideCursorBadge: () => { hidden += 1 },
})
expect(hidden).toBe(1)
})
})
+18 -4
View File
@@ -22,9 +22,9 @@ const UNHIDE_TIMEOUT_MS = 5000
const OVERLAY_HIDE_TIMEOUT_MS = 2000
/**
* Turn-end cleanup for the chicago MCP surface: hide the native overlay (macOS
* cu-helper daemon), auto-unhide apps that `prepareForAction` hid, then release
* the file-based lock.
* Turn-end cleanup for the chicago MCP surface: drop the activity indicator
* (macOS cu-helper overlay, or the Windows cursor badge), auto-unhide apps
* that `prepareForAction` hid, then release the file-based lock.
*
* Called from three sites: natural turn end (`stopHooks.ts`), abort during
* streaming (`query.ts` aborted_streaming), abort during tool execution
@@ -49,8 +49,22 @@ export async function cleanupComputerUseAfterTurn(
ToolUseContext,
'getAppState' | 'setAppState' | 'sendOSNotification'
>,
deps: { overlayHide?: () => Promise<void> } = {},
deps: {
overlayHide?: () => Promise<void>
hideCursorBadge?: () => void
} = {},
): Promise<void> {
// Windows counterpart to the macOS overlay. Synchronous, no-throw, and a
// no-op off-Windows, so it goes first and unconditionally: an orphaned badge
// would sit on screen claiming the agent is holding the mouse after the turn
// has ended, which is a worse lie than showing nothing at all.
const hideBadge =
deps.hideCursorBadge ??
(() => {
void import('./winCursorBadge.js').then(m => m.hideCursorBadge()).catch(() => {})
})
hideBadge()
// Drop the daemon overlay FIRST — before the hidden-apps block and before the
// isLockHeldLocally early-return below — so the cursor drops promptly even on a
// turn that hid no apps and whose lock-release short-circuits. overlayHide
+48 -46
View File
@@ -1,9 +1,11 @@
/**
* CLI `ComputerExecutor` implementation — Python bridge variant.
* CLI `ComputerExecutor` implementation — platform-routed helper variant.
*
* Replaces the native Swift/Rust modules with a Python subprocess bridge
* (pyautogui + mss + platform helpers). See `pythonBridge.ts` and
* `runtime/{mac,win}_helper.py`.
* Every command goes through `helperBridge.callHelper`, which routes by
* platform: macOS reaches the signed native `cu-helper` daemon, Windows
* reaches the Python helper (`runtime/win_helper.py`, pyautogui + mss).
* This module is deliberately platform-agnostic — the routing decision, and
* the very different guarantees each side offers, live in `helperBridge.ts`.
*/
import type {
@@ -29,8 +31,8 @@ import {
isComputerUseSupportedPlatform,
} from './common.js'
// Platform-routed helper: macOS → native cu-helper (no cursor steal),
// Windows → Python helper. Aliased so the 20+ call sites below stay unchanged.
import { callHelper as callPythonHelper } from './helperBridge.js'
// Windows → Python helper (pyautogui, which does move the real cursor).
import { callHelper } from './helperBridge.js'
const SCREENSHOT_JPEG_QUALITY = 0.75
const MOVE_SETTLE_MS = 50
@@ -62,16 +64,16 @@ function normalizeDisplayGeometry(display: PythonDisplayGeometry): DisplayGeomet
}
async function readClipboardViaPbpaste(): Promise<string> {
return callPythonHelper<string>('read_clipboard', {})
return callHelper<string>('read_clipboard', {})
}
async function writeClipboardViaPbcopy(text: string): Promise<void> {
await callPythonHelper('write_clipboard', { text })
await callHelper('write_clipboard', { text })
}
async function readClipboard(): Promise<string> {
if (process.platform === 'win32') {
return callPythonHelper<string>('read_clipboard', {})
return callHelper<string>('read_clipboard', {})
}
return readClipboardViaPbpaste()
@@ -79,7 +81,7 @@ async function readClipboard(): Promise<string> {
async function writeClipboard(text: string): Promise<void> {
if (process.platform === 'win32') {
await callPythonHelper('write_clipboard', { text })
await callHelper('write_clipboard', { text })
return
}
@@ -155,21 +157,21 @@ function formatAppList(apps: readonly DaemonAppRef[]): string {
export function createCodexEngine(): CodexComputerEngine {
return {
async listApps(): Promise<string> {
const apps = await callPythonHelper<DaemonAppRef[]>('list_apps', {})
const apps = await callHelper<DaemonAppRef[]>('list_apps', {})
return formatAppList(apps)
},
async resolveTarget(target: AppTarget): Promise<ResolvedAppTarget> {
// Sent as-is: the daemon owns the selector→process mapping, and it must
// never launch anything to satisfy a match.
return callPythonHelper<ResolvedAppTarget>('resolve_app_target', target)
return callHelper<ResolvedAppTarget>('resolve_app_target', target)
},
async getAppState(
target: AppTarget,
opts?: { disableDiff?: boolean },
): Promise<AppStateResult> {
return callPythonHelper<AppStateResult>('get_app_state', {
return callHelper<AppStateResult>('get_app_state', {
...appTargetPayload(target),
...(opts?.disableDiff === undefined ? {} : { disableDiff: opts.disableDiff }),
})
@@ -183,7 +185,7 @@ export function createCodexEngine(): CodexComputerEngine {
clickCount?: number
button?: CodexMouseButton
}): Promise<void> {
await callPythonHelper('click', {
await callHelper('click', {
...appTargetPayload(args.target),
index: args.index,
x: args.x,
@@ -198,7 +200,7 @@ export function createCodexEngine(): CodexComputerEngine {
index: string
value: string
}): Promise<SetValueResult> {
return callPythonHelper<SetValueResult>('set_value', {
return callHelper<SetValueResult>('set_value', {
...appTargetPayload(args.target),
index: args.index,
value: args.value,
@@ -213,7 +215,7 @@ export function createCodexEngine(): CodexComputerEngine {
suffix?: string
selection?: 'text' | 'cursor_before' | 'cursor_after'
}): Promise<void> {
await callPythonHelper('select_text', {
await callHelper('select_text', {
...appTargetPayload(args.target),
index: args.index,
text: args.text,
@@ -228,7 +230,7 @@ export function createCodexEngine(): CodexComputerEngine {
index: string
action: string
}): Promise<void> {
await callPythonHelper('perform_secondary_action', {
await callHelper('perform_secondary_action', {
...appTargetPayload(args.target),
index: args.index,
action: args.action,
@@ -243,7 +245,7 @@ export function createCodexEngine(): CodexComputerEngine {
direction: 'up' | 'down' | 'left' | 'right'
pages?: number
}): Promise<void> {
await callPythonHelper('scroll', {
await callHelper('scroll', {
...appTargetPayload(args.target),
index: args.index,
x: args.x,
@@ -259,7 +261,7 @@ export function createCodexEngine(): CodexComputerEngine {
to: { x: number; y: number }
button?: CodexMouseButton
}): Promise<void> {
await callPythonHelper('drag', {
await callHelper('drag', {
...appTargetPayload(args.target),
from: args.from,
to: args.to,
@@ -272,7 +274,7 @@ export function createCodexEngine(): CodexComputerEngine {
key: string
systemKeyCombos: boolean
}): Promise<void> {
await callPythonHelper('press_key', {
await callHelper('press_key', {
...appTargetPayload(args.target),
key: args.key,
systemKeyCombos: args.systemKeyCombos,
@@ -280,7 +282,7 @@ export function createCodexEngine(): CodexComputerEngine {
},
async typeText(args: { target: AppTarget; text: string }): Promise<void> {
await callPythonHelper('type_text', {
await callHelper('type_text', {
...appTargetPayload(args.target),
text: args.text,
})
@@ -300,10 +302,10 @@ async function typeViaClipboard(text: string): Promise<void> {
// Give NSPasteboard a beat before paste, then keep the new contents
// resident long enough for Electron/WebView fields to consume them.
await sleep(40)
await callPythonHelper('paste_clipboard', {})
await callHelper('paste_clipboard', {})
await sleep(180)
} else {
await callPythonHelper('key', {
await callHelper('key', {
keySequence: 'ctrl+v',
repeat: 1,
})
@@ -341,30 +343,30 @@ export function createCliExecutor(_opts: {
engine: process.platform === 'darwin' ? createCodexEngine() : undefined,
async prepareForAction(_allowlistBundleIds, _displayId): Promise<string[]> {
return callPythonHelper('prepare_for_action', {})
return callHelper('prepare_for_action', {})
},
async previewHideSet(_allowlistBundleIds, _displayId) {
return callPythonHelper('preview_hide_set', {})
return callHelper('preview_hide_set', {})
},
async getDisplaySize(displayId?: number): Promise<DisplayGeometry> {
return normalizeDisplayGeometry(await callPythonHelper('get_display_size', { displayId }))
return normalizeDisplayGeometry(await callHelper('get_display_size', { displayId }))
},
async listDisplays(): Promise<DisplayGeometry[]> {
const displays = await callPythonHelper<PythonDisplayGeometry[]>('list_displays', {})
const displays = await callHelper<PythonDisplayGeometry[]>('list_displays', {})
return displays.map(display => normalizeDisplayGeometry(display))
},
async findWindowDisplays(bundleIds: string[]) {
return callPythonHelper('find_window_displays', { bundleIds })
return callHelper('find_window_displays', { bundleIds })
},
async resolvePrepareCapture(opts): Promise<ResolvePrepareCaptureResult> {
const display = await this.getDisplaySize(opts.preferredDisplayId)
const [targetW, targetH] = computeTargetDims(display.width, display.height, display.scaleFactor)
const result = await callPythonHelper<PythonResolvePrepareCaptureResult>('resolve_prepare_capture', {
const result = await callHelper<PythonResolvePrepareCaptureResult>('resolve_prepare_capture', {
preferredDisplayId: opts.preferredDisplayId,
targetWidth: targetW,
targetHeight: targetH,
@@ -380,7 +382,7 @@ export function createCliExecutor(_opts: {
async screenshot(opts): Promise<ScreenshotResult> {
const display = await this.getDisplaySize(opts.displayId)
const [targetW, targetH] = computeTargetDims(display.width, display.height, display.scaleFactor)
const result = await callPythonHelper<ScreenshotResult>('screenshot', {
const result = await callHelper<ScreenshotResult>('screenshot', {
displayId: opts.displayId,
targetWidth: targetW,
targetHeight: targetH,
@@ -392,7 +394,7 @@ export function createCliExecutor(_opts: {
async zoom(regionLogical, _allowedBundleIds, displayId) {
const display = await this.getDisplaySize(displayId)
const [outW, outH] = computeTargetDims(regionLogical.w, regionLogical.h, display.scaleFactor)
return callPythonHelper('zoom', {
return callHelper('zoom', {
x: regionLogical.x,
y: regionLogical.y,
width: regionLogical.w,
@@ -403,11 +405,11 @@ export function createCliExecutor(_opts: {
},
async key(keySequence: string, repeat?: number): Promise<void> {
await callPythonHelper('key', { keySequence, repeat: repeat ?? 1 })
await callHelper('key', { keySequence, repeat: repeat ?? 1 })
},
async holdKey(keyNames: string[], durationMs: number): Promise<void> {
await callPythonHelper('hold_key', { keyNames, durationMs })
await callHelper('hold_key', { keyNames, durationMs })
},
async type(text: string, opts2: { viaClipboard: boolean }): Promise<void> {
@@ -415,61 +417,61 @@ export function createCliExecutor(_opts: {
await typeViaClipboard(text)
return
}
await callPythonHelper('type', { text })
await callHelper('type', { text })
},
readClipboard,
writeClipboard,
async click(x, y, button, count, modifiers): Promise<void> {
await callPythonHelper('click', { x, y, button, count, modifiers })
await callHelper('click', { x, y, button, count, modifiers })
await sleep(MOVE_SETTLE_MS)
},
async mouseDown(): Promise<void> {
await callPythonHelper('mouse_down', {})
await callHelper('mouse_down', {})
},
async mouseUp(): Promise<void> {
await callPythonHelper('mouse_up', {})
await callHelper('mouse_up', {})
},
async getCursorPosition(): Promise<{ x: number; y: number }> {
return callPythonHelper('cursor_position', {})
return callHelper('cursor_position', {})
},
async drag(from, to): Promise<void> {
await callPythonHelper('drag', { from, to })
await callHelper('drag', { from, to })
await sleep(MOVE_SETTLE_MS)
},
async moveMouse(x, y): Promise<void> {
await callPythonHelper('move_mouse', { x, y })
await callHelper('move_mouse', { x, y })
await sleep(MOVE_SETTLE_MS)
},
async scroll(x, y, dx, dy): Promise<void> {
await callPythonHelper('scroll', { x, y, deltaX: dx, deltaY: dy })
await callHelper('scroll', { x, y, deltaX: dx, deltaY: dy })
},
async getFrontmostApp(): Promise<FrontmostApp | null> {
return callPythonHelper('frontmost_app', {})
return callHelper('frontmost_app', {})
},
async appUnderPoint(x, y) {
return callPythonHelper('app_under_point', { x, y })
return callHelper('app_under_point', { x, y })
},
async listInstalledApps(): Promise<InstalledApp[]> {
return callPythonHelper('list_installed_apps', {})
return callHelper('list_installed_apps', {})
},
async listRunningApps(): Promise<RunningApp[]> {
return callPythonHelper('list_running_apps', {})
return callHelper('list_running_apps', {})
},
async openApp(bundleId: string): Promise<void> {
await callPythonHelper('open_app', { bundleId })
await callHelper('open_app', { bundleId })
},
}
}
@@ -355,10 +355,65 @@ describe('callHelper platform routing', () => {
callDaemon: async () => { used = 'daemon'; return 0 as never },
callPy: async () => { used = 'py'; return 0 as never },
overlayShow: () => {},
showCursorBadge: () => {},
})
expect(used).toBe('py')
})
test('Windows marks agent activity on an injecting command', async () => {
// Windows drives through SendInput, so the user's real cursor moves. The
// badge is the only thing telling them the movement is not theirs, which
// matters because grabbing the mouse mid-action is what makes the two
// input streams interleave.
let badges = 0
await callHelper('click', { x: 1, y: 2 }, {
platform: 'win32',
callPy: ok,
showCursorBadge: () => { badges += 1 },
})
expect(badges).toBe(1)
})
test('Windows leaves the badge alone for read-only commands', async () => {
// A badge on `screenshot` would claim the agent is holding the mouse
// during a turn that never touches it.
let badges = 0
for (const command of ['screenshot', 'list_displays', 'read_clipboard']) {
await callHelper(command, {}, {
platform: 'win32',
callPy: ok,
showCursorBadge: () => { badges += 1 },
})
}
expect(badges).toBe(0)
})
test('Windows never reaches the macOS overlay', async () => {
// The two indicators are not interchangeable: overlay_show is a daemon
// command and there is no daemon on Windows, so a stray call would be a
// hard failure on the mutation path.
let overlays = 0
await callHelper('click', { x: 1, y: 2 }, {
platform: 'win32',
callPy: ok,
overlayShow: () => { overlays += 1 },
showCursorBadge: () => {},
})
expect(overlays).toBe(0)
})
test('macOS never spawns the Windows badge', async () => {
let badges = 0
await callHelper('click', { x: 1, y: 2 }, {
platform: 'darwin',
cuHelperAvailable: () => true,
callDaemon: ok,
overlayShow: () => {},
showCursorBadge: () => { badges += 1 },
})
expect(badges).toBe(0)
})
test('forwards command + payload to the daemon unchanged', async () => {
let seen: { c: string; p: unknown } | undefined
await callHelper('type', { text: 'hi' }, {
+17
View File
@@ -8,6 +8,7 @@ import {
overlayShow,
shutdownDaemon,
} from './cuHelperDaemon.js'
import { showCursorBadge } from './winCursorBadge.js'
// Latches true after we restart the daemon once in response to an Accessibility
// `not_trusted` error, so we don't thrash-restart while the helper is genuinely
@@ -97,6 +98,14 @@ function overlayTargetPayload(
* - Windows → the Python helper (`win_helper.py`); the native engine is
* macOS-only.
*
* The two platforms do NOT offer the same guarantee, and callers should not
* assume they do. macOS delivers input per-process and never touches the real
* pointer. Windows has no such API: input goes through `SendInput`, so the
* agent shares one cursor and one input stream with the user. The Windows
* side therefore gets a badge that marks agent activity rather than a virtual
* cursor that replaces it, and `win_helper.py` refuses actions it can already
* tell will not land.
*
* `deps` is injectable for unit tests only.
*/
export async function callHelper<T>(
@@ -109,6 +118,7 @@ export async function callHelper<T>(
callPy?: HelperFn
overlayShow?: (payload: Record<string, unknown>) => void
isOverlayShown?: () => boolean
showCursorBadge?: () => void
callFrontmost?: () => Promise<{ bundleId?: string } | null>
shutdownDaemon?: () => void
} = {},
@@ -119,6 +129,7 @@ export async function callHelper<T>(
const viaPython = deps.callPy ?? (callPythonHelper as HelperFn)
const showOverlay = deps.overlayShow ?? overlayShow
const overlayIsShown = deps.isOverlayShown ?? isOverlayShown
const showBadge = deps.showCursorBadge ?? (() => showCursorBadge())
const restartDaemon = deps.shutdownDaemon ?? (() => void shutdownDaemon())
if (platform === 'darwin') {
@@ -156,6 +167,12 @@ export async function callHelper<T>(
}
}
if (INJECTION_COMMANDS.has(command)) {
// Same trigger set as the macOS overlay, so both platforms mark activity
// at the same moments. Fire-and-forget: the badge is advisory and must
// never sit on the mutation hot path.
showBadge()
}
return viaPython<T>(command, payload)
}
+2 -2
View File
@@ -9,7 +9,7 @@ import { createCliExecutor } from './executor.js'
import { getChicagoEnabled, getChicagoSubGates } from './gates.js'
import { normalizeOsPermissions } from './permissions.js'
// Platform-routed helper: macOS → native cu-helper, Windows → Python helper.
import { callHelper as callPythonHelper } from './helperBridge.js'
import { callHelper } from './helperBridge.js'
import { maybeShowNativePermissionCard } from './nativePermissionCard.js'
class DebugLogger implements Logger {
@@ -42,7 +42,7 @@ export function getComputerUseHostAdapter(): ComputerUseHostAdapter {
getHideBeforeActionEnabled: () => getChicagoSubGates().hideBeforeAction,
}),
ensureOsPermissions: async () => {
const rawPerms = await callPythonHelper<{ accessibility: boolean; screenRecording: boolean | null }>('check_permissions', {})
const rawPerms = await callHelper<{ accessibility: boolean; screenRecording: boolean | null }>('check_permissions', {})
const perms = normalizeOsPermissions(rawPerms)
if (perms.granted) return { granted: true as const }
// Missing a TCC grant → pop the native, guided permission card (macOS).
+33 -7
View File
@@ -13,7 +13,13 @@ const projectRoot = path.resolve(__dirname, '../../..')
// All runtime state lives in ~/.claude/.runtime — writable in both dev and
// bundled (Tauri app) modes. The setup API (or ensureRuntimeFiles below)
// populates requirements.txt and mac_helper.py here.
// populates requirements-win.txt and win_helper.py here.
//
// This bridge is Windows-only. macOS routes every command to the signed native
// `cu-helper` daemon and `helperBridge` refuses to fall back, so the old
// `mac_helper.py` was unreachable and has been deleted along with its
// pyobjc requirements file. Keeping a dead darwin branch here invited the
// reading that Python is still a supported macOS path — it is not.
const runtimeStateRoot = path.join(getClaudeConfigHomeDir(), '.runtime')
const venvRoot = path.join(runtimeStateRoot, 'venv')
const installStampPath = path.join(runtimeStateRoot, 'requirements.sha256')
@@ -22,8 +28,12 @@ const isWindows = process.platform === 'win32'
// Always read from ~/.claude/.runtime/ — works in both dev and bundled mode.
const requirementsPath = path.join(runtimeStateRoot, 'requirements.txt')
const helperFileName = isWindows ? 'win_helper.py' : 'mac_helper.py'
const helperFileName = 'win_helper.py'
const helperPath = path.join(runtimeStateRoot, helperFileName)
// Runs as its own process (the helper is a stateless one-shot CLI and cannot
// own a window across actions), so it ships as a separate file.
const cursorBadgeFileName = 'win_cursor_badge.py'
const cursorBadgePath = path.join(runtimeStateRoot, cursorBadgeFileName)
let bootstrapPromise: Promise<void> | undefined
@@ -98,8 +108,7 @@ async function getVenvCreationPythonCommand(): Promise<string> {
async function ensureRuntimeFiles(): Promise<void> {
await mkdir(runtimeStateRoot, { recursive: true })
const devReqFile = isWindows ? 'requirements-win.txt' : 'requirements.txt'
const devRequirements = path.join(projectRoot, 'runtime', devReqFile)
const devRequirements = path.join(projectRoot, 'runtime', 'requirements-win.txt')
const devHelper = path.join(projectRoot, 'runtime', helperFileName)
// Always sync from dev runtime/ so source changes are reflected immediately.
@@ -112,12 +121,17 @@ async function ensureRuntimeFiles(): Promise<void> {
if (await pathExists(devHelper)) {
await writeFile(helperPath, await readFile(devHelper, 'utf8'), 'utf8')
}
const devBadge = path.join(projectRoot, 'runtime', cursorBadgeFileName)
if (await pathExists(devBadge)) {
await writeFile(cursorBadgePath, await readFile(devBadge, 'utf8'), 'utf8')
}
}
export async function ensureBootstrapped(): Promise<void> {
if (bootstrapPromise) return bootstrapPromise
bootstrapPromise = (async () => {
// Extract runtime files (requirements.txt, mac_helper.py) to state dir
// Extract runtime files (requirements, helper, badge) to state dir
await ensureRuntimeFiles()
if (!(await pathExists(pythonBinPath()))) {
@@ -168,7 +182,7 @@ export async function callPythonHelper<T>(command: string, payload: Record<strin
throw new Error(stderr || `Python helper ${command} failed with code ${code}`)
}
let parsed: { ok: boolean; result?: T; error?: { message?: string } }
let parsed: { ok: boolean; result?: T; error?: { code?: string; message?: string } }
try {
parsed = JSON.parse(stdout)
} catch {
@@ -176,7 +190,14 @@ export async function callPythonHelper<T>(command: string, payload: Record<strin
}
if (!parsed.ok) {
throw new Error(parsed.error?.message || `Python helper ${command} failed`)
// Prefix the machine-readable code so callers can branch on it. The helper
// reports refusals such as `user_interference` and `target_window_offscreen`
// this way, and a bare message would flatten them into indistinguishable
// prose — the model would then retry an action that was deliberately
// refused. Mirrors how the macOS daemon surfaces `CUError.code`.
const code = parsed.error?.code
const message = parsed.error?.message || `Python helper ${command} failed`
throw new Error(code && code !== 'runtime_error' ? `${code}: ${message}` : message)
}
return parsed.result as T
@@ -185,3 +206,8 @@ export async function callPythonHelper<T>(command: string, payload: Record<strin
export function getRuntimePaths(): { projectRoot: string; runtimeStateRoot: string; venvRoot: string } {
return { projectRoot, runtimeStateRoot, venvRoot }
}
/** Interpreter + script path for the Windows agent-activity badge. */
export function getCursorBadgeCommand(): { python: string; script: string } {
return { python: pythonBinPath(), script: cursorBadgePath }
}
+95
View File
@@ -0,0 +1,95 @@
import { spawn, type ChildProcess } from 'node:child_process'
import { logForDebugging } from '../debug.js'
import { getCursorBadgeCommand } from './pythonBridge.js'
/**
* The Windows agent-activity badge: a click-through marker that rides the real
* cursor while the agent is driving.
*
* WHY WINDOWS NEEDS THIS AND macOS DOES NOT
* -----------------------------------------
* On macOS the helper delivers input with `CGEvent.postToPid`, so the real
* pointer never moves and the drawn cursor is the only one there is — a
* *replacement*, and unambiguous by construction.
*
* Windows has no per-process delivery. Input goes through `SendInput`, which
* warps the one real cursor the user's hand is also on. Drawing a second fake
* pointer would make that worse, not better: two pointers, one of which is
* lying about where the click will land.
*
* So this badge annotates rather than replaces. It answers the one question
* the user cannot otherwise answer — "is the mouse moving because of me, or
* because of the agent?" — which on Windows has real stakes, since grabbing
* the mouse mid-action is what causes the two input streams to interleave.
*
* Lifecycle mirrors the macOS overlay: shown while a turn is driving, hidden
* at turn end. It is advisory, so every failure here is logged and swallowed —
* a badge that cannot start must never be the reason an action fails.
*/
let badgeProcess: ChildProcess | undefined
function isRunning(): boolean {
return badgeProcess !== undefined && badgeProcess.exitCode === null && !badgeProcess.killed
}
/** Start the badge if it isn't already up. Idempotent, never throws. */
export function showCursorBadge(label = 'Claude'): void {
if (process.platform !== 'win32') return
if (isRunning()) return
try {
const { python, script } = getCursorBadgeCommand()
const child = spawn(python, [script, '--label', label], {
// 'ignore' for stderr would discard the reason it died; 'pipe' on stdin
// is load-bearing — the badge exits when stdin closes, which is what
// ties its lifetime to ours even if we are killed rather than exiting.
stdio: ['pipe', 'ignore', 'pipe'],
windowsHide: true,
detached: false,
})
child.on('error', err => {
logForDebugging(`cursor badge failed to start: ${String(err)}`, { level: 'debug' })
if (badgeProcess === child) badgeProcess = undefined
})
child.on('exit', () => {
if (badgeProcess === child) badgeProcess = undefined
})
child.stderr?.on('data', (chunk: Buffer) => {
logForDebugging(`cursor badge: ${chunk.toString().trim()}`, { level: 'debug' })
})
badgeProcess = child
} catch (err) {
logForDebugging(`cursor badge spawn threw: ${String(err)}`, { level: 'debug' })
badgeProcess = undefined
}
}
/** Take the badge down. Idempotent, never throws. */
export function hideCursorBadge(): void {
const child = badgeProcess
badgeProcess = undefined
if (!child) return
try {
// Closing stdin is the graceful path — the badge's reader hits EOF and
// unwinds its own message loop. kill() is the backstop for a process that
// is wedged before it ever got to that read.
child.stdin?.end()
child.kill()
} catch (err) {
logForDebugging(`cursor badge shutdown failed: ${String(err)}`, { level: 'debug' })
}
}
/** Test hook: forget any tracked process without signalling it. */
export function __resetCursorBadgeState(): void {
badgeProcess = undefined
}
/** Test hook: whether a badge process is currently tracked as running. */
export function __cursorBadgeIsRunning(): boolean {
return isRunning()
}