mirror of
https://github.com/NanmiCoder/claude-code-haha.git
synced 2026-10-10 20:03:13 +08:00
fix(computer-use): make the Windows helper refuse what it cannot deliver
Windows has no equivalent of `CGEvent.postToPid`. `pyautogui` bottoms out in
`SendInput`, which injects into the one system-wide input stream and warps the
one real cursor — so on Windows the agent shares the mouse and keyboard with
the user, and `SendInput` reports success unconditionally whether or not
anything acted on the events.
That combination produced the same lie the macOS engine was just fixed for:
click a point behind another window and it lands on that window; click one
off-screen and it lands nowhere; either way the helper answered "Action
completed".
So the helper now refuses instead of guessing:
* `ForegroundLease` samples `GetLastInputInfo` around every mutating command.
Interference before the action is `user_interference` — nothing ran, retry
is safe. Interference during it is `user_interference_result_unknown`,
because injection already went out and a retry could double-apply it. On a
play/pause toggle those two differ by exactly one wrong outcome.
* `ensure_point_on_screen` and `ensure_target_window_reachable` reject
coordinates outside every display and targets whose windows are minimized
or hidden, before anything is sent.
`GetLastInputInfo` is the signal because it needs no privileges and is not
advanced by `SendInput`, so the agent cannot trip its own detector. Both guards
fail open on an unreadable reading: a safety layer that turns the feature off
is not safety.
The guard set lives in one place rather than in each dispatcher branch — an
eleventh verb wired like the ten before it would otherwise be silently
unguarded, which is the bug class this pass removes. All ten mutating branches
now return through `_finish`, so no branch can write its own success response
and skip the post-action check.
Also adds a Windows cursor badge. It deliberately does NOT mirror the macOS
virtual cursor: there the real pointer never moves, so the drawn one is the
only cursor and replaces it. Here the real pointer does move, and a second
fake pointer would just be two cursors with one of them lying about where the
click lands. The badge annotates instead — it answers "is this me or the
agent?", which matters because grabbing the mouse mid-action is what makes the
two input streams interleave.
Retires `runtime/mac_helper.py` and its pyobjc requirements: macOS routes every
command to the signed native daemon and `helperBridge` refuses to fall back, so
both were unreachable. Renames the `callPythonHelper` alias to `callHelper`,
which is what it has actually imported since the native engine landed.
Verified with mutation testing — 13 injected regressions across both languages
(dropped lease, bypassed `_finish`, collapsed interference codes, guard reusing
the filtered `list_windows`, badge losing click-through), all caught.
Claude-Session: https://claude.ai/code/session_015j1yxxaoonyAS2iZ7qGnTS
This commit is contained in:
@@ -1,775 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import ctypes
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import mss
|
||||
from AppKit import NSWorkspace, NSPasteboard, NSPasteboardTypeString, NSURL
|
||||
from PIL import Image
|
||||
from Quartz import (
|
||||
CGDisplayBounds,
|
||||
CGDisplayIsMain,
|
||||
CGDisplayModeGetPixelHeight,
|
||||
CGDisplayModeGetPixelWidth,
|
||||
CGDisplayPixelsHigh,
|
||||
CGDisplayPixelsWide,
|
||||
CGGetActiveDisplayList,
|
||||
CGMainDisplayID,
|
||||
CGWindowListCopyWindowInfo,
|
||||
CGRectContainsPoint,
|
||||
CGRectIntersection,
|
||||
CGPointMake,
|
||||
CGPreflightScreenCaptureAccess,
|
||||
kCGNullWindowID,
|
||||
kCGWindowBounds,
|
||||
kCGWindowIsOnscreen,
|
||||
kCGWindowLayer,
|
||||
kCGWindowListExcludeDesktopElements,
|
||||
kCGWindowListOptionOnScreenOnly,
|
||||
kCGWindowName,
|
||||
kCGWindowOwnerName,
|
||||
)
|
||||
|
||||
os.environ.setdefault("PYTHONDONTWRITEBYTECODE", "1")
|
||||
os.environ.setdefault("PYAUTOGUI_HIDE_SUPPORT_PROMPT", "1")
|
||||
|
||||
import pyautogui # noqa: E402
|
||||
|
||||
pyautogui.FAILSAFE = False
|
||||
pyautogui.PAUSE = 0
|
||||
|
||||
KEY_MAP = {
|
||||
"a": "a",
|
||||
"b": "b",
|
||||
"c": "c",
|
||||
"d": "d",
|
||||
"e": "e",
|
||||
"f": "f",
|
||||
"g": "g",
|
||||
"h": "h",
|
||||
"i": "i",
|
||||
"j": "j",
|
||||
"k": "k",
|
||||
"l": "l",
|
||||
"m": "m",
|
||||
"n": "n",
|
||||
"o": "o",
|
||||
"p": "p",
|
||||
"q": "q",
|
||||
"r": "r",
|
||||
"s": "s",
|
||||
"t": "t",
|
||||
"u": "u",
|
||||
"v": "v",
|
||||
"w": "w",
|
||||
"x": "x",
|
||||
"y": "y",
|
||||
"z": "z",
|
||||
"0": "0",
|
||||
"1": "1",
|
||||
"2": "2",
|
||||
"3": "3",
|
||||
"4": "4",
|
||||
"5": "5",
|
||||
"6": "6",
|
||||
"7": "7",
|
||||
"8": "8",
|
||||
"9": "9",
|
||||
"cmd": "command",
|
||||
"command": "command",
|
||||
"meta": "command",
|
||||
"super": "command",
|
||||
"ctrl": "ctrl",
|
||||
"control": "ctrl",
|
||||
"shift": "shift",
|
||||
"alt": "option",
|
||||
"option": "option",
|
||||
"opt": "option",
|
||||
"fn": "fn",
|
||||
"escape": "esc",
|
||||
"esc": "esc",
|
||||
"enter": "enter",
|
||||
"return": "enter",
|
||||
"tab": "tab",
|
||||
"space": "space",
|
||||
"backspace": "backspace",
|
||||
"delete": "delete",
|
||||
"forwarddelete": "delete",
|
||||
"up": "up",
|
||||
"down": "down",
|
||||
"left": "left",
|
||||
"right": "right",
|
||||
"home": "home",
|
||||
"end": "end",
|
||||
"pageup": "pageup",
|
||||
"pagedown": "pagedown",
|
||||
"capslock": "capslock",
|
||||
"f1": "f1",
|
||||
"f2": "f2",
|
||||
"f3": "f3",
|
||||
"f4": "f4",
|
||||
"f5": "f5",
|
||||
"f6": "f6",
|
||||
"f7": "f7",
|
||||
"f8": "f8",
|
||||
"f9": "f9",
|
||||
"f10": "f10",
|
||||
"f11": "f11",
|
||||
"f12": "f12",
|
||||
"-": "minus",
|
||||
"=": "equals",
|
||||
"[": "[",
|
||||
"]": "]",
|
||||
"\\": "\\",
|
||||
";": ";",
|
||||
"'": "'",
|
||||
",": ",",
|
||||
".": ".",
|
||||
"/": "/",
|
||||
"`": "`",
|
||||
}
|
||||
|
||||
|
||||
def normalize_key(name: str) -> str:
|
||||
key = name.strip().lower()
|
||||
if key not in KEY_MAP:
|
||||
raise ValueError(f"Unsupported key: {name}")
|
||||
return KEY_MAP[key]
|
||||
|
||||
|
||||
def json_output(payload: dict[str, Any]) -> None:
|
||||
sys.stdout.write(json.dumps(payload, ensure_ascii=False))
|
||||
sys.stdout.write("\n")
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
def error_output(message: str, code: str = "runtime_error") -> None:
|
||||
json_output({"ok": False, "error": {"code": code, "message": message}})
|
||||
|
||||
|
||||
def bool_env(name: str, default: bool = False) -> bool:
|
||||
value = os.environ.get(name)
|
||||
if value is None:
|
||||
return default
|
||||
return value not in {"0", "false", "False", ""}
|
||||
|
||||
|
||||
def run_osascript(script: str) -> str:
|
||||
result = subprocess.run(
|
||||
["osascript", "-e", script],
|
||||
text=True,
|
||||
capture_output=True,
|
||||
check=False,
|
||||
)
|
||||
if result.returncode != 0:
|
||||
raise RuntimeError(result.stderr.strip() or result.stdout.strip() or "osascript failed")
|
||||
return result.stdout.strip()
|
||||
|
||||
|
||||
def applescript_modifier(name: str) -> str:
|
||||
if name == "command":
|
||||
return "command down"
|
||||
if name == "option":
|
||||
return "option down"
|
||||
if name == "shift":
|
||||
return "shift down"
|
||||
if name == "ctrl":
|
||||
return "control down"
|
||||
if name == "fn":
|
||||
return "fn down"
|
||||
raise ValueError(f"Unsupported AppleScript modifier: {name}")
|
||||
|
||||
|
||||
def send_keystroke_via_osascript(character: str, modifiers: list[str] | None = None) -> None:
|
||||
escaped = character.replace("\\", "\\\\").replace('"', '\\"')
|
||||
if modifiers:
|
||||
modifier_expr = ", ".join(applescript_modifier(m) for m in modifiers)
|
||||
script = (
|
||||
'tell application "System Events" to keystroke '
|
||||
f'"{escaped}" using {{{modifier_expr}}}'
|
||||
)
|
||||
else:
|
||||
script = f'tell application "System Events" to keystroke "{escaped}"'
|
||||
run_osascript(script)
|
||||
|
||||
|
||||
def get_displays() -> list[dict[str, Any]]:
|
||||
max_displays = 32
|
||||
err, active, count = CGGetActiveDisplayList(max_displays, None, None)
|
||||
if err != 0:
|
||||
raise RuntimeError(f"CGGetActiveDisplayList failed: {err}")
|
||||
displays: list[dict[str, Any]] = []
|
||||
main_id = CGMainDisplayID()
|
||||
for idx, display_id in enumerate(active[:count]):
|
||||
bounds = CGDisplayBounds(display_id)
|
||||
mode = None
|
||||
try:
|
||||
from Quartz import CGDisplayCopyDisplayMode
|
||||
mode = CGDisplayCopyDisplayMode(display_id)
|
||||
except Exception:
|
||||
mode = None
|
||||
physical_width = int(CGDisplayPixelsWide(display_id))
|
||||
physical_height = int(CGDisplayPixelsHigh(display_id))
|
||||
logical_width = int(bounds.size.width)
|
||||
logical_height = int(bounds.size.height)
|
||||
if mode is not None:
|
||||
mode_w = int(CGDisplayModeGetPixelWidth(mode))
|
||||
mode_h = int(CGDisplayModeGetPixelHeight(mode))
|
||||
physical_width = mode_w or physical_width
|
||||
physical_height = mode_h or physical_height
|
||||
scale_factor = physical_width / logical_width if logical_width else 1
|
||||
name = f"Display {idx + 1}"
|
||||
displays.append(
|
||||
{
|
||||
"id": int(display_id),
|
||||
"displayId": int(display_id),
|
||||
"width": logical_width,
|
||||
"height": logical_height,
|
||||
"scaleFactor": scale_factor,
|
||||
"originX": int(bounds.origin.x),
|
||||
"originY": int(bounds.origin.y),
|
||||
"isPrimary": bool(display_id == main_id or CGDisplayIsMain(display_id)),
|
||||
"name": name,
|
||||
"label": name,
|
||||
}
|
||||
)
|
||||
return displays
|
||||
|
||||
|
||||
def choose_display(display_id: int | None) -> dict[str, Any]:
|
||||
displays = get_displays()
|
||||
if not displays:
|
||||
raise RuntimeError("No active displays found")
|
||||
if display_id is None:
|
||||
for display in displays:
|
||||
if display["isPrimary"]:
|
||||
return display
|
||||
return displays[0]
|
||||
for display in displays:
|
||||
if display["displayId"] == display_id or display["id"] == display_id:
|
||||
return display
|
||||
raise RuntimeError(f"Unknown display: {display_id}")
|
||||
|
||||
|
||||
def ensure_screen_recording_permission() -> None:
|
||||
"""No-op: CGPreflightScreenCaptureAccess is unreliable for child processes
|
||||
(returns False even when the parent app has TCC permission), and any actual
|
||||
capture attempt triggers a macOS popup on newer versions. Let the actual
|
||||
capture call handle errors instead."""
|
||||
pass
|
||||
|
||||
|
||||
def capture_display(display_id: int | None, resize: tuple[int, int] | None = None) -> dict[str, Any]:
|
||||
ensure_screen_recording_permission()
|
||||
display = choose_display(display_id)
|
||||
monitor = {
|
||||
"left": display["originX"],
|
||||
"top": display["originY"],
|
||||
"width": display["width"],
|
||||
"height": display["height"],
|
||||
}
|
||||
with mss.mss() as sct:
|
||||
raw = sct.grab(monitor)
|
||||
image = Image.frombytes("RGB", raw.size, raw.rgb)
|
||||
if resize:
|
||||
image = image.resize(resize, Image.Resampling.LANCZOS)
|
||||
buffer = BytesIO()
|
||||
image.save(buffer, format="JPEG", quality=75, optimize=True)
|
||||
base64_data = base64.b64encode(buffer.getvalue()).decode("ascii")
|
||||
return {
|
||||
"base64": base64_data,
|
||||
"width": image.width,
|
||||
"height": image.height,
|
||||
"displayWidth": display["width"],
|
||||
"displayHeight": display["height"],
|
||||
"displayId": display["displayId"],
|
||||
"originX": display["originX"],
|
||||
"originY": display["originY"],
|
||||
"display": display,
|
||||
}
|
||||
|
||||
|
||||
def capture_region(region: dict[str, int], resize: tuple[int, int] | None = None) -> dict[str, Any]:
|
||||
ensure_screen_recording_permission()
|
||||
with mss.mss() as sct:
|
||||
raw = sct.grab(region)
|
||||
image = Image.frombytes("RGB", raw.size, raw.rgb)
|
||||
if resize:
|
||||
image = image.resize(resize, Image.Resampling.LANCZOS)
|
||||
buffer = BytesIO()
|
||||
image.save(buffer, format="JPEG", quality=75, optimize=True)
|
||||
base64_data = base64.b64encode(buffer.getvalue()).decode("ascii")
|
||||
return {"base64": base64_data, "width": image.width, "height": image.height}
|
||||
|
||||
|
||||
def list_windows() -> list[dict[str, Any]]:
|
||||
windows = CGWindowListCopyWindowInfo(
|
||||
kCGWindowListOptionOnScreenOnly | kCGWindowListExcludeDesktopElements,
|
||||
kCGNullWindowID,
|
||||
)
|
||||
out: list[dict[str, Any]] = []
|
||||
for window in windows or []:
|
||||
if int(window.get(kCGWindowLayer, 0)) != 0:
|
||||
continue
|
||||
if not bool(window.get(kCGWindowIsOnscreen, True)):
|
||||
continue
|
||||
bounds = window.get(kCGWindowBounds) or {}
|
||||
width = int(bounds.get("Width", 0))
|
||||
height = int(bounds.get("Height", 0))
|
||||
if width <= 1 or height <= 1:
|
||||
continue
|
||||
out.append(
|
||||
{
|
||||
"ownerName": window.get(kCGWindowOwnerName, "") or "",
|
||||
"title": window.get(kCGWindowName, "") or "",
|
||||
"bounds": {
|
||||
"x": int(bounds.get("X", 0)),
|
||||
"y": int(bounds.get("Y", 0)),
|
||||
"width": width,
|
||||
"height": height,
|
||||
},
|
||||
}
|
||||
)
|
||||
return out
|
||||
|
||||
|
||||
def bundle_id_to_app(bundle_id: str):
|
||||
return NSWorkspace.sharedWorkspace().URLForApplicationWithBundleIdentifier_(bundle_id)
|
||||
|
||||
|
||||
def installed_apps() -> list[dict[str, Any]]:
|
||||
search_roots = [
|
||||
Path("/Applications"),
|
||||
Path.home() / "Applications",
|
||||
Path("/System/Applications"),
|
||||
Path("/System/Applications/Utilities"),
|
||||
]
|
||||
results: dict[str, dict[str, Any]] = {}
|
||||
workspace = NSWorkspace.sharedWorkspace()
|
||||
for root in search_roots:
|
||||
if not root.exists():
|
||||
continue
|
||||
for app in root.rglob("*.app"):
|
||||
try:
|
||||
bundle = workspace.bundleIdentifierForURL_(NSURL.fileURLWithPath_(str(app)))
|
||||
except Exception:
|
||||
bundle = None
|
||||
if not bundle:
|
||||
try:
|
||||
url = workspace.URLForApplicationWithBundleIdentifier_(str(app))
|
||||
bundle = workspace.bundleIdentifierForURL_(url) if url else None
|
||||
except Exception:
|
||||
bundle = None
|
||||
info_plist = app / "Contents/Info.plist"
|
||||
display_name = app.stem
|
||||
if info_plist.exists():
|
||||
try:
|
||||
import plistlib
|
||||
with info_plist.open("rb") as f:
|
||||
plist = plistlib.load(f)
|
||||
bundle = bundle or plist.get("CFBundleIdentifier")
|
||||
display_name = plist.get("CFBundleDisplayName") or plist.get("CFBundleName") or display_name
|
||||
except Exception:
|
||||
pass
|
||||
if not bundle or bundle in results:
|
||||
continue
|
||||
results[bundle] = {
|
||||
"bundleId": str(bundle),
|
||||
"displayName": str(display_name),
|
||||
"path": str(app),
|
||||
}
|
||||
return sorted(results.values(), key=lambda item: item["displayName"].lower())
|
||||
|
||||
|
||||
def running_apps() -> list[dict[str, Any]]:
|
||||
apps = []
|
||||
seen = set()
|
||||
for app in NSWorkspace.sharedWorkspace().runningApplications() or []:
|
||||
bundle_id = app.bundleIdentifier()
|
||||
if not bundle_id or bundle_id in seen:
|
||||
continue
|
||||
seen.add(bundle_id)
|
||||
name = app.localizedName() or bundle_id
|
||||
apps.append({"bundleId": str(bundle_id), "displayName": str(name)})
|
||||
return sorted(apps, key=lambda item: item["displayName"].lower())
|
||||
|
||||
|
||||
def app_display_name(bundle_id: str) -> str | None:
|
||||
for app in NSWorkspace.sharedWorkspace().runningApplications() or []:
|
||||
if app.bundleIdentifier() == bundle_id:
|
||||
return str(app.localizedName() or bundle_id)
|
||||
for app in installed_apps():
|
||||
if app["bundleId"] == bundle_id:
|
||||
return str(app["displayName"])
|
||||
return None
|
||||
|
||||
|
||||
def frontmost_app() -> dict[str, str] | None:
|
||||
app = NSWorkspace.sharedWorkspace().frontmostApplication()
|
||||
if not app:
|
||||
return None
|
||||
bundle_id = app.bundleIdentifier()
|
||||
if not bundle_id:
|
||||
return None
|
||||
return {
|
||||
"bundleId": str(bundle_id),
|
||||
"displayName": str(app.localizedName() or bundle_id),
|
||||
}
|
||||
|
||||
|
||||
def app_under_point(x: int, y: int) -> dict[str, str] | None:
|
||||
point = CGPointMake(x, y)
|
||||
running_by_name = {
|
||||
str(app.localizedName() or app.bundleIdentifier()): str(app.bundleIdentifier())
|
||||
for app in NSWorkspace.sharedWorkspace().runningApplications() or []
|
||||
if app.bundleIdentifier()
|
||||
}
|
||||
for window in list_windows():
|
||||
bounds = window["bounds"]
|
||||
rect = ((bounds["x"], bounds["y"]), (bounds["width"], bounds["height"]))
|
||||
if CGRectContainsPoint(rect, point):
|
||||
owner = window["ownerName"]
|
||||
bundle = running_by_name.get(owner)
|
||||
if bundle:
|
||||
return {"bundleId": bundle, "displayName": str(owner)}
|
||||
return frontmost_app()
|
||||
|
||||
|
||||
def find_window_displays(bundle_ids: list[str]) -> list[dict[str, Any]]:
|
||||
if not bundle_ids:
|
||||
return []
|
||||
displays = get_displays()
|
||||
names_by_bundle = {
|
||||
bundle_id: app_display_name(bundle_id) or bundle_id for bundle_id in bundle_ids
|
||||
}
|
||||
windows = list_windows()
|
||||
result = []
|
||||
for bundle_id in bundle_ids:
|
||||
target_name = names_by_bundle.get(bundle_id)
|
||||
display_ids: set[int] = set()
|
||||
for window in windows:
|
||||
owner = window["ownerName"]
|
||||
if not owner:
|
||||
continue
|
||||
if target_name and owner != target_name:
|
||||
continue
|
||||
if not target_name and owner != bundle_id:
|
||||
continue
|
||||
wx = window["bounds"]["x"]
|
||||
wy = window["bounds"]["y"]
|
||||
ww = window["bounds"]["width"]
|
||||
wh = window["bounds"]["height"]
|
||||
window_rect = ((wx, wy), (ww, wh))
|
||||
for display in displays:
|
||||
display_rect = ((display["originX"], display["originY"]), (display["width"], display["height"]))
|
||||
intersection = CGRectIntersection(window_rect, display_rect)
|
||||
if intersection.size.width > 0 and intersection.size.height > 0:
|
||||
display_ids.add(int(display["displayId"]))
|
||||
result.append({"bundleId": bundle_id, "displayIds": sorted(display_ids)})
|
||||
return result
|
||||
|
||||
|
||||
def open_app(bundle_id: str) -> None:
|
||||
url = bundle_id_to_app(bundle_id)
|
||||
if not url:
|
||||
raise RuntimeError(f"App not found for bundle identifier: {bundle_id}")
|
||||
ok, err = NSWorkspace.sharedWorkspace().launchApplicationAtURL_options_configuration_error_(url, 0, {}, None)
|
||||
if not ok:
|
||||
raise RuntimeError(str(err) if err else f"Failed to open app {bundle_id}")
|
||||
|
||||
|
||||
def read_clipboard() -> str:
|
||||
pb = NSPasteboard.generalPasteboard()
|
||||
value = pb.stringForType_(NSPasteboardTypeString)
|
||||
return "" if value is None else str(value)
|
||||
|
||||
|
||||
def write_clipboard(text: str) -> None:
|
||||
pb = NSPasteboard.generalPasteboard()
|
||||
pb.clearContents()
|
||||
pb.setString_forType_(text, NSPasteboardTypeString)
|
||||
|
||||
|
||||
def paste_clipboard() -> None:
|
||||
send_keystroke_via_osascript("v", ["command"])
|
||||
|
||||
|
||||
def detect_screen_recording_permission() -> bool | None:
|
||||
"""Best-effort passive screen-recording probe with no system prompt.
|
||||
|
||||
`CGPreflightScreenCaptureAccess()` is fast and explicit when it returns
|
||||
True, but on child processes launched by a TCC-authorized app bundle it can
|
||||
still return False. As a fallback, inspect the visible window list: Apple
|
||||
only exposes other apps' window titles when Screen Recording access is
|
||||
granted. If we can see at least one title, treat the permission as granted.
|
||||
If we can inspect visible windows but every title is blank, treat it as not
|
||||
granted. If window enumeration itself is unavailable, return None.
|
||||
"""
|
||||
|
||||
try:
|
||||
if CGPreflightScreenCaptureAccess():
|
||||
return True
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
try:
|
||||
windows = CGWindowListCopyWindowInfo(
|
||||
kCGWindowListOptionOnScreenOnly | kCGWindowListExcludeDesktopElements,
|
||||
kCGNullWindowID,
|
||||
)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
eligible_windows = 0
|
||||
for window in windows or []:
|
||||
if int(window.get(kCGWindowLayer, 0)) != 0:
|
||||
continue
|
||||
if not bool(window.get(kCGWindowIsOnscreen, True)):
|
||||
continue
|
||||
|
||||
bounds = window.get(kCGWindowBounds) or {}
|
||||
width = int(bounds.get("Width", 0))
|
||||
height = int(bounds.get("Height", 0))
|
||||
if width <= 1 or height <= 1:
|
||||
continue
|
||||
|
||||
eligible_windows += 1
|
||||
if (window.get(kCGWindowName, "") or "").strip():
|
||||
return True
|
||||
|
||||
if eligible_windows > 0:
|
||||
return False
|
||||
return None
|
||||
|
||||
|
||||
def detect_accessibility_permission() -> bool:
|
||||
"""
|
||||
Use the official macOS Accessibility trust API.
|
||||
|
||||
The previous System Events / AppleScript probe was too weak: it could
|
||||
succeed even when the current helper process was not actually trusted for
|
||||
input control, which led the desktop UI to report Accessibility as granted
|
||||
while mouse/keyboard control still failed at runtime.
|
||||
"""
|
||||
framework_path = "/System/Library/Frameworks/ApplicationServices.framework/ApplicationServices"
|
||||
try:
|
||||
application_services = ctypes.CDLL(framework_path)
|
||||
application_services.AXIsProcessTrusted.restype = ctypes.c_bool
|
||||
application_services.AXIsProcessTrusted.argtypes = []
|
||||
return bool(application_services.AXIsProcessTrusted())
|
||||
except Exception:
|
||||
# Fail closed: if the trust API can't be queried, treat accessibility
|
||||
# as unavailable instead of reporting a misleading success state.
|
||||
return False
|
||||
|
||||
|
||||
def check_permissions() -> dict[str, bool | None]:
|
||||
accessibility = detect_accessibility_permission()
|
||||
screen_recording = detect_screen_recording_permission()
|
||||
return {
|
||||
"accessibility": accessibility,
|
||||
"screenRecording": screen_recording,
|
||||
}
|
||||
|
||||
|
||||
def click(x: int, y: int, button: str, count: int, modifiers: list[str] | None) -> None:
|
||||
pyautogui.moveTo(x, y)
|
||||
if modifiers:
|
||||
normalized = [normalize_key(m) for m in modifiers]
|
||||
for key in normalized:
|
||||
pyautogui.keyDown(key)
|
||||
try:
|
||||
pyautogui.click(x=x, y=y, button=button, clicks=count, interval=0.08)
|
||||
finally:
|
||||
for key in reversed(normalized):
|
||||
pyautogui.keyUp(key)
|
||||
else:
|
||||
pyautogui.click(x=x, y=y, button=button, clicks=count, interval=0.08)
|
||||
|
||||
|
||||
def scroll(x: int, y: int, delta_x: int, delta_y: int) -> None:
|
||||
pyautogui.moveTo(x, y)
|
||||
if delta_y:
|
||||
pyautogui.scroll(int(delta_y), x=x, y=y)
|
||||
if delta_x:
|
||||
pyautogui.hscroll(int(delta_x), x=x, y=y)
|
||||
|
||||
|
||||
def key_action(sequence: str, repeat: int = 1) -> None:
|
||||
parts = [normalize_key(part) for part in sequence.split("+") if part.strip()]
|
||||
for _ in range(max(1, repeat)):
|
||||
if parts == ["command", "v"]:
|
||||
paste_clipboard()
|
||||
elif parts == ["command", "a"]:
|
||||
send_keystroke_via_osascript("a", ["command"])
|
||||
elif parts == ["command", "c"]:
|
||||
send_keystroke_via_osascript("c", ["command"])
|
||||
elif parts == ["command", "x"]:
|
||||
send_keystroke_via_osascript("x", ["command"])
|
||||
elif len(parts) == 1:
|
||||
pyautogui.press(parts[0])
|
||||
else:
|
||||
pyautogui.hotkey(*parts, interval=0.02)
|
||||
time.sleep(0.01)
|
||||
|
||||
|
||||
def hold_keys(keys: list[str], duration_ms: int) -> None:
|
||||
normalized = [normalize_key(k) for k in keys]
|
||||
for key in normalized:
|
||||
pyautogui.keyDown(key)
|
||||
try:
|
||||
time.sleep(max(duration_ms, 0) / 1000)
|
||||
finally:
|
||||
for key in reversed(normalized):
|
||||
pyautogui.keyUp(key)
|
||||
|
||||
|
||||
def type_text(text: str) -> None:
|
||||
pyautogui.write(text, interval=0.008)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("command")
|
||||
parser.add_argument("--payload", default="{}")
|
||||
args = parser.parse_args()
|
||||
payload = json.loads(args.payload)
|
||||
|
||||
try:
|
||||
command = args.command
|
||||
if command == "check_permissions":
|
||||
perms = check_permissions()
|
||||
json_output({"ok": True, "result": perms})
|
||||
return 0
|
||||
if command == "list_displays":
|
||||
json_output({"ok": True, "result": get_displays()})
|
||||
return 0
|
||||
if command == "get_display_size":
|
||||
json_output({"ok": True, "result": choose_display(payload.get("displayId"))})
|
||||
return 0
|
||||
if command == "screenshot":
|
||||
resize = None
|
||||
if payload.get("targetWidth") and payload.get("targetHeight"):
|
||||
resize = (int(payload["targetWidth"]), int(payload["targetHeight"]))
|
||||
result = capture_display(payload.get("displayId"), resize)
|
||||
json_output({"ok": True, "result": result})
|
||||
return 0
|
||||
if command == "resolve_prepare_capture":
|
||||
resize = None
|
||||
if payload.get("targetWidth") and payload.get("targetHeight"):
|
||||
resize = (int(payload["targetWidth"]), int(payload["targetHeight"]))
|
||||
result = capture_display(payload.get("preferredDisplayId"), resize)
|
||||
result["hidden"] = []
|
||||
result["resolvedDisplayId"] = result["displayId"]
|
||||
json_output({"ok": True, "result": result})
|
||||
return 0
|
||||
if command == "zoom":
|
||||
resize = None
|
||||
if payload.get("targetWidth") and payload.get("targetHeight"):
|
||||
resize = (int(payload["targetWidth"]), int(payload["targetHeight"]))
|
||||
region = {
|
||||
"left": int(payload["x"]),
|
||||
"top": int(payload["y"]),
|
||||
"width": int(payload["width"]),
|
||||
"height": int(payload["height"]),
|
||||
}
|
||||
json_output({"ok": True, "result": capture_region(region, resize)})
|
||||
return 0
|
||||
if command == "prepare_for_action":
|
||||
json_output({"ok": True, "result": []})
|
||||
return 0
|
||||
if command == "preview_hide_set":
|
||||
json_output({"ok": True, "result": []})
|
||||
return 0
|
||||
if command == "find_window_displays":
|
||||
json_output({"ok": True, "result": find_window_displays(list(payload.get("bundleIds") or []))})
|
||||
return 0
|
||||
if command == "key":
|
||||
key_action(str(payload["keySequence"]), int(payload.get("repeat") or 1))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "hold_key":
|
||||
hold_keys(list(payload.get("keyNames") or []), int(payload.get("durationMs") or 0))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "type":
|
||||
type_text(str(payload.get("text") or ""))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "click":
|
||||
click(int(payload["x"]), int(payload["y"]), str(payload.get("button") or "left"), int(payload.get("count") or 1), payload.get("modifiers"))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "drag":
|
||||
from_point = payload.get("from")
|
||||
if from_point:
|
||||
pyautogui.moveTo(int(from_point["x"]), int(from_point["y"]))
|
||||
pyautogui.dragTo(int(payload["to"]["x"]), int(payload["to"]["y"]), duration=0.2, button="left")
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "move_mouse":
|
||||
pyautogui.moveTo(int(payload["x"]), int(payload["y"]))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "scroll":
|
||||
scroll(int(payload["x"]), int(payload["y"]), int(payload.get("deltaX") or 0), int(payload.get("deltaY") or 0))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "mouse_down":
|
||||
pyautogui.mouseDown(button="left")
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "mouse_up":
|
||||
pyautogui.mouseUp(button="left")
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "cursor_position":
|
||||
x, y = pyautogui.position()
|
||||
json_output({"ok": True, "result": {"x": int(x), "y": int(y)}})
|
||||
return 0
|
||||
if command == "frontmost_app":
|
||||
json_output({"ok": True, "result": frontmost_app()})
|
||||
return 0
|
||||
if command == "app_under_point":
|
||||
json_output({"ok": True, "result": app_under_point(int(payload["x"]), int(payload["y"]))})
|
||||
return 0
|
||||
if command == "list_installed_apps":
|
||||
json_output({"ok": True, "result": installed_apps()})
|
||||
return 0
|
||||
if command == "list_running_apps":
|
||||
json_output({"ok": True, "result": running_apps()})
|
||||
return 0
|
||||
if command == "open_app":
|
||||
open_app(str(payload["bundleId"]))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "read_clipboard":
|
||||
json_output({"ok": True, "result": read_clipboard()})
|
||||
return 0
|
||||
if command == "write_clipboard":
|
||||
write_clipboard(str(payload.get("text") or ""))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
if command == "paste_clipboard":
|
||||
paste_clipboard()
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
error_output(f"Unknown command: {command}", code="bad_command")
|
||||
return 2
|
||||
except Exception as exc:
|
||||
error_output(str(exc))
|
||||
return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -1,6 +0,0 @@
|
||||
mss>=9.0.2,<10
|
||||
Pillow>=11.3.0,<12
|
||||
pyautogui>=0.9.54
|
||||
pyobjc-core>=11.1
|
||||
pyobjc-framework-Cocoa>=11.1
|
||||
pyobjc-framework-Quartz>=11.1
|
||||
+304
-229
@@ -1,42 +1,46 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Cross-platform tests for mac_helper.py and win_helper.py.
|
||||
"""Tests for win_helper.py.
|
||||
|
||||
Tests the platform-independent parts (JSON protocol, key mapping, capture logic)
|
||||
without requiring platform-specific dependencies. Can run on any OS with pytest.
|
||||
macOS routes every Computer Use command to the signed native `cu-helper`
|
||||
daemon — `helperBridge` refuses to fall back to Python — so `mac_helper.py`
|
||||
was unreachable and has been deleted. This file therefore covers the Windows
|
||||
helper only.
|
||||
|
||||
Most tests here are static (they read the source) rather than executed,
|
||||
because the runtime deps (pywin32, pyautogui, mss) are Windows-only and CI
|
||||
runs on macOS. Static coverage is enough for what actually regresses: the
|
||||
guards getting dropped, inverted, or quietly bypassed.
|
||||
|
||||
Usage:
|
||||
python -m pytest runtime/test_helpers.py -v
|
||||
# or simply:
|
||||
python runtime/test_helpers.py
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import json
|
||||
import subprocess
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch, MagicMock
|
||||
|
||||
# Determine which helper to test based on current platform
|
||||
IS_WINDOWS = sys.platform == "win32"
|
||||
IS_MACOS = sys.platform == "darwin"
|
||||
|
||||
RUNTIME_DIR = Path(__file__).parent
|
||||
MAC_HELPER = RUNTIME_DIR / "mac_helper.py"
|
||||
WIN_HELPER = RUNTIME_DIR / "win_helper.py"
|
||||
CURSOR_BADGE = RUNTIME_DIR / "win_cursor_badge.py"
|
||||
|
||||
|
||||
def _win_source() -> str:
|
||||
return WIN_HELPER.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
class TestKeyMap(unittest.TestCase):
|
||||
"""Test the KEY_MAP and normalize_key function — platform-independent logic."""
|
||||
"""KEY_MAP translates macOS key names to Windows ones."""
|
||||
|
||||
def _load_key_map(self, helper_path: Path) -> dict[str, str]:
|
||||
"""Extract KEY_MAP from a helper by importing it with mocked deps."""
|
||||
# Read the file and extract just the KEY_MAP dict
|
||||
source = helper_path.read_text()
|
||||
# Find KEY_MAP definition
|
||||
source = helper_path.read_text(encoding="utf-8")
|
||||
start = source.index("KEY_MAP = {")
|
||||
# Find the matching closing brace
|
||||
depth = 0
|
||||
for i, ch in enumerate(source[start:], start):
|
||||
if ch == "{":
|
||||
@@ -46,106 +50,40 @@ class TestKeyMap(unittest.TestCase):
|
||||
if depth == 0:
|
||||
end = i + 1
|
||||
break
|
||||
key_map_source = source[start:end]
|
||||
ns: dict = {}
|
||||
exec(key_map_source, ns)
|
||||
exec(source[start:end], ns)
|
||||
return ns["KEY_MAP"]
|
||||
|
||||
def test_mac_key_map_exists(self):
|
||||
if not MAC_HELPER.exists():
|
||||
self.skipTest("mac_helper.py not found")
|
||||
km = self._load_key_map(MAC_HELPER)
|
||||
self.assertIn("cmd", km)
|
||||
self.assertIn("ctrl", km)
|
||||
self.assertEqual(km["cmd"], "command")
|
||||
self.assertEqual(km["alt"], "option")
|
||||
|
||||
def test_win_key_map_exists(self):
|
||||
if not WIN_HELPER.exists():
|
||||
self.skipTest("win_helper.py not found")
|
||||
km = self._load_key_map(WIN_HELPER)
|
||||
self.assertIn("cmd", km)
|
||||
self.assertIn("ctrl", km)
|
||||
# Windows maps cmd/command/meta to 'win' key
|
||||
# The mapping that matters: a model trained on macOS emits "cmd", and
|
||||
# on Windows that has to become "win", not silently stay "cmd".
|
||||
self.assertEqual(km["cmd"], "win")
|
||||
self.assertEqual(km["command"], "win")
|
||||
self.assertEqual(km["meta"], "win")
|
||||
# Windows maps alt/option to 'alt'
|
||||
self.assertEqual(km["alt"], "alt")
|
||||
self.assertEqual(km["option"], "alt")
|
||||
|
||||
def test_common_keys_present_in_both(self):
|
||||
"""Both helpers must have the same set of key names."""
|
||||
if not MAC_HELPER.exists() or not WIN_HELPER.exists():
|
||||
self.skipTest("Both helpers required")
|
||||
mac_km = self._load_key_map(MAC_HELPER)
|
||||
win_km = self._load_key_map(WIN_HELPER)
|
||||
# All keys in mac should be in win and vice versa
|
||||
self.assertEqual(set(mac_km.keys()), set(win_km.keys()),
|
||||
"KEY_MAP keys must be identical across platforms")
|
||||
|
||||
def test_all_alphabet_keys(self):
|
||||
"""All a-z keys should map to themselves."""
|
||||
for helper in [MAC_HELPER, WIN_HELPER]:
|
||||
if not helper.exists():
|
||||
continue
|
||||
km = self._load_key_map(helper)
|
||||
for char in "abcdefghijklmnopqrstuvwxyz":
|
||||
self.assertEqual(km[char], char, f"{helper.name}: {char} should map to itself")
|
||||
km = self._load_key_map(WIN_HELPER)
|
||||
for ch in "abcdefghijklmnopqrstuvwxyz":
|
||||
self.assertIn(ch, km)
|
||||
|
||||
def test_all_digit_keys(self):
|
||||
"""All 0-9 keys should map to themselves."""
|
||||
for helper in [MAC_HELPER, WIN_HELPER]:
|
||||
if not helper.exists():
|
||||
continue
|
||||
km = self._load_key_map(helper)
|
||||
for digit in "0123456789":
|
||||
self.assertEqual(km[digit], digit, f"{helper.name}: {digit} should map to itself")
|
||||
|
||||
def test_function_keys(self):
|
||||
"""F1-F12 should map to themselves."""
|
||||
for helper in [MAC_HELPER, WIN_HELPER]:
|
||||
if not helper.exists():
|
||||
continue
|
||||
km = self._load_key_map(helper)
|
||||
for i in range(1, 13):
|
||||
key = f"f{i}"
|
||||
self.assertEqual(km[key], key, f"{helper.name}: {key} should map to itself")
|
||||
km = self._load_key_map(WIN_HELPER)
|
||||
for d in "0123456789":
|
||||
self.assertIn(d, km)
|
||||
|
||||
|
||||
class TestJSONProtocol(unittest.TestCase):
|
||||
"""Test that both helpers follow the same JSON command protocol."""
|
||||
|
||||
def _get_helper(self) -> Path:
|
||||
"""Get the appropriate helper for the current platform."""
|
||||
if IS_WINDOWS and WIN_HELPER.exists():
|
||||
return WIN_HELPER
|
||||
if IS_MACOS and MAC_HELPER.exists():
|
||||
return MAC_HELPER
|
||||
return MAC_HELPER if MAC_HELPER.exists() else WIN_HELPER
|
||||
|
||||
def _parse_main_commands(self, helper_path: Path) -> list[str]:
|
||||
"""Extract all command names from the main() dispatcher."""
|
||||
source = helper_path.read_text()
|
||||
source = helper_path.read_text(encoding="utf-8")
|
||||
commands = []
|
||||
for line in source.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped.startswith('if command == "'):
|
||||
cmd = stripped.split('"')[1]
|
||||
commands.append(cmd)
|
||||
commands.append(stripped.split('"')[1])
|
||||
return commands
|
||||
|
||||
def test_both_helpers_same_commands(self):
|
||||
"""Both helpers must support the exact same set of commands."""
|
||||
if not MAC_HELPER.exists() or not WIN_HELPER.exists():
|
||||
self.skipTest("Both helpers required")
|
||||
mac_cmds = set(self._parse_main_commands(MAC_HELPER))
|
||||
win_cmds = set(self._parse_main_commands(WIN_HELPER))
|
||||
self.assertEqual(mac_cmds, win_cmds,
|
||||
f"Command sets differ.\nOnly in mac: {mac_cmds - win_cmds}\nOnly in win: {win_cmds - mac_cmds}")
|
||||
|
||||
def test_expected_commands_exist(self):
|
||||
"""Core commands should be present in each helper."""
|
||||
expected = {
|
||||
"check_permissions", "list_displays", "get_display_size",
|
||||
"screenshot", "resolve_prepare_capture", "zoom",
|
||||
@@ -156,167 +94,304 @@ class TestJSONProtocol(unittest.TestCase):
|
||||
"list_installed_apps", "list_running_apps", "open_app",
|
||||
"read_clipboard", "write_clipboard", "paste_clipboard",
|
||||
}
|
||||
for helper in [MAC_HELPER, WIN_HELPER]:
|
||||
if not helper.exists():
|
||||
continue
|
||||
cmds = set(self._parse_main_commands(helper))
|
||||
missing = expected - cmds
|
||||
self.assertFalse(missing,
|
||||
f"{helper.name} missing commands: {missing}")
|
||||
cmds = set(self._parse_main_commands(WIN_HELPER))
|
||||
self.assertFalse(expected - cmds,
|
||||
f"win_helper.py missing commands: {expected - cmds}")
|
||||
|
||||
@unittest.skipUnless(IS_WINDOWS, "requires Windows runtime deps")
|
||||
def test_unknown_command_returns_error(self):
|
||||
"""Running a non-existent command should return a JSON error."""
|
||||
helper = self._get_helper()
|
||||
if not helper.exists():
|
||||
self.skipTest("No helper found")
|
||||
# On macOS without venv, mac_helper.py may fail at import (AppKit);
|
||||
# on Windows without venv, win_helper.py may fail at import (win32gui).
|
||||
# Only test if the helper can actually import.
|
||||
check = subprocess.run(
|
||||
[sys.executable, "-c", f"import importlib.util; "
|
||||
f"spec = importlib.util.spec_from_file_location('h', '{helper}')"],
|
||||
capture_output=True, text=True
|
||||
)
|
||||
result = subprocess.run(
|
||||
[sys.executable, str(helper), "nonexistent_command_xyz"],
|
||||
capture_output=True, text=True
|
||||
[sys.executable, str(WIN_HELPER), "nonexistent_command_xyz"],
|
||||
capture_output=True, text=True,
|
||||
)
|
||||
if result.returncode == 1 and not result.stdout.strip():
|
||||
# Import failed — platform deps missing, skip this test
|
||||
self.skipTest(f"Cannot run {helper.name} on this platform (missing deps)")
|
||||
# Should exit with code 2
|
||||
self.skipTest("missing platform deps")
|
||||
self.assertEqual(result.returncode, 2)
|
||||
parsed = json.loads(result.stdout.strip())
|
||||
self.assertFalse(parsed["ok"])
|
||||
self.assertEqual(parsed["error"]["code"], "bad_command")
|
||||
|
||||
|
||||
class TestHelperOutputFormat(unittest.TestCase):
|
||||
"""Test the JSON output helpers are consistent."""
|
||||
class TestMutatingCommandsAreGuarded(unittest.TestCase):
|
||||
"""Every command that injects input must pass through the guards.
|
||||
|
||||
def test_json_output_function_exists(self):
|
||||
"""Both helpers should define json_output and error_output."""
|
||||
for helper in [MAC_HELPER, WIN_HELPER]:
|
||||
if not helper.exists():
|
||||
These are static-source tests on purpose. The failure being guarded against
|
||||
is someone adding an eleventh mutating verb and wiring it like the ten that
|
||||
came before — at which point it silently has no lease and no reachability
|
||||
check. A runtime test would need Windows and would only cover the verbs it
|
||||
thought to enumerate; reading the dispatcher catches the new one.
|
||||
"""
|
||||
|
||||
# Kept as a literal, deliberately duplicating MUTATING_COMMANDS in the
|
||||
# helper. If the two drift the test fails, which is the point: the set is
|
||||
# a security boundary and should not be edited casually on one side only.
|
||||
MUTATING = {
|
||||
"click", "drag", "move_mouse", "scroll",
|
||||
"mouse_down", "mouse_up",
|
||||
"key", "hold_key", "type",
|
||||
"paste_clipboard",
|
||||
}
|
||||
|
||||
def _module_constant(self, name: str) -> set[str]:
|
||||
"""Read a module-level frozenset/set constant without importing."""
|
||||
tree = ast.parse(_win_source())
|
||||
for node in tree.body:
|
||||
if isinstance(node, ast.Assign):
|
||||
for target in node.targets:
|
||||
if isinstance(target, ast.Name) and target.id == name:
|
||||
return set(ast.literal_eval(
|
||||
node.value.args[0]
|
||||
if isinstance(node.value, ast.Call)
|
||||
else node.value
|
||||
))
|
||||
raise AssertionError(f"{name} not found in win_helper.py")
|
||||
|
||||
def test_mutating_command_set_matches_this_test(self):
|
||||
self.assertEqual(self._module_constant("MUTATING_COMMANDS"), self.MUTATING)
|
||||
|
||||
def test_coordinate_commands_are_a_subset(self):
|
||||
coords = self._module_constant("COORDINATE_COMMANDS")
|
||||
self.assertTrue(coords <= self.MUTATING)
|
||||
# `key`/`type` go wherever focus is and have no point to validate.
|
||||
# Asserting their absence keeps someone from "fixing" the coordinate
|
||||
# guard by adding them and then dereferencing an x/y that isn't there.
|
||||
self.assertNotIn("key", coords)
|
||||
self.assertNotIn("type", coords)
|
||||
|
||||
def test_every_mutating_branch_finalizes_the_lease(self):
|
||||
"""No mutating branch may answer with a bare json_output.
|
||||
|
||||
This is the specific regression: `_finish` is what runs the post-action
|
||||
interference check, so a branch that writes its own success response
|
||||
reports "Action completed" for input that may have collided with the
|
||||
user's own typing.
|
||||
"""
|
||||
source = _win_source()
|
||||
start = source.index(' try:\n command = args.command')
|
||||
end = source.index(' error_output(f"Unknown command: {command}"')
|
||||
dispatcher = source[start:end]
|
||||
|
||||
blocks = dispatcher.split('if command == "')
|
||||
for block in blocks[1:]:
|
||||
name = block.split('"')[0]
|
||||
if name not in self.MUTATING:
|
||||
continue
|
||||
source = helper.read_text()
|
||||
self.assertIn("def json_output(", source,
|
||||
f"{helper.name} missing json_output function")
|
||||
self.assertIn("def error_output(", source,
|
||||
f"{helper.name} missing error_output function")
|
||||
body = block.split("if command ==")[0]
|
||||
self.assertIn(
|
||||
"_finish(lease,", body,
|
||||
f'"{name}" must return through _finish so the lease is checked',
|
||||
)
|
||||
self.assertNotIn(
|
||||
'json_output({"ok": True', body,
|
||||
f'"{name}" writes its own success response, bypassing the lease',
|
||||
)
|
||||
|
||||
def test_main_entry_point(self):
|
||||
"""Both helpers should have the standard main entry point."""
|
||||
for helper in [MAC_HELPER, WIN_HELPER]:
|
||||
if not helper.exists():
|
||||
continue
|
||||
source = helper.read_text()
|
||||
self.assertIn('if __name__ == "__main__":', source,
|
||||
f"{helper.name} missing __main__ guard")
|
||||
self.assertIn("def main()", source,
|
||||
f"{helper.name} missing main() function")
|
||||
def test_guards_run_before_any_injection(self):
|
||||
"""acquire() must precede the dispatch chain, not follow it."""
|
||||
source = _win_source()
|
||||
acquire = source.index("lease.acquire()")
|
||||
first_branch = source.index(' if command == "check_permissions"')
|
||||
self.assertLess(
|
||||
acquire, first_branch,
|
||||
"the lease must be acquired before any command branch runs",
|
||||
)
|
||||
|
||||
|
||||
class TestWinHelperPermissions(unittest.TestCase):
|
||||
"""Windows-specific: permissions should always return True."""
|
||||
class TestInterferenceDetection(unittest.TestCase):
|
||||
def test_uses_getlastinputinfo_not_an_event_hook(self):
|
||||
"""The signal must stay permission-free.
|
||||
|
||||
A low-level input hook would read the same events, but installing one
|
||||
is exactly the kind of thing that gets an app flagged, and it is not
|
||||
needed: GetLastInputInfo answers the only question we ask.
|
||||
"""
|
||||
source = _win_source()
|
||||
self.assertIn("GetLastInputInfo", source)
|
||||
self.assertNotIn("SetWindowsHookEx", source)
|
||||
|
||||
def test_distinguishes_did_not_run_from_outcome_unknown(self):
|
||||
"""The two interference verdicts must stay distinct.
|
||||
|
||||
Collapsing them is a real hazard: `user_interference` means nothing
|
||||
happened and a retry is safe, while `user_interference_result_unknown`
|
||||
means input already went out and a retry could double-apply it. On a
|
||||
play/pause toggle those differ by exactly one wrong outcome.
|
||||
"""
|
||||
source = _win_source()
|
||||
self.assertIn('"user_interference"', source)
|
||||
self.assertIn('user_interference_result_unknown', source)
|
||||
|
||||
acquire_start = source.index(" def acquire(self)")
|
||||
acquire_body = source[acquire_start:source.index(" def finalize(self)")]
|
||||
self.assertNotIn("result_unknown", acquire_body,
|
||||
"a pre-action refusal means nothing ran; the outcome is known")
|
||||
|
||||
finalize_body = source[source.index(" def finalize(self)"):]
|
||||
finalize_body = finalize_body[:finalize_body.index("\n\n\n")]
|
||||
self.assertIn("result_unknown", finalize_body,
|
||||
"post-action interference leaves the outcome unknown")
|
||||
|
||||
def test_counter_read_failure_does_not_block_actions(self):
|
||||
"""An unreadable counter must fail open, not brick the feature.
|
||||
|
||||
Precedent from the macOS side: an earlier build required an Input
|
||||
Monitoring grant that onboarding never asked for, so every mutating
|
||||
action failed on a correctly set-up machine. A safety layer that turns
|
||||
the product off is not safety.
|
||||
"""
|
||||
source = _win_source()
|
||||
fn_start = source.index("def last_physical_input_tick()")
|
||||
body = source[fn_start:source.index("class UserInterference")]
|
||||
self.assertIn("return 0", body)
|
||||
# And the comparisons must treat 0 as "no reading", never as a tick
|
||||
# value that happens to differ from the next one.
|
||||
self.assertIn("if before and after and before != after:", source)
|
||||
|
||||
def test_synthetic_input_must_not_trip_the_detector(self):
|
||||
"""Documented invariant: SendInput does not advance GetLastInputInfo.
|
||||
|
||||
If this ever stopped holding, every agent action would abort itself and
|
||||
the feature would look randomly broken. Pinning the claim in a test
|
||||
keeps it from being quietly deleted as a stale comment.
|
||||
"""
|
||||
source = _win_source()
|
||||
self.assertIn("is NOT advanced by", source)
|
||||
|
||||
|
||||
class TestDeliveryGuards(unittest.TestCase):
|
||||
def test_offscreen_point_is_refused(self):
|
||||
source = _win_source()
|
||||
self.assertIn("point_outside_display", source)
|
||||
self.assertIn("def ensure_point_on_screen", source)
|
||||
|
||||
def test_unreachable_window_is_refused(self):
|
||||
source = _win_source()
|
||||
self.assertIn("target_window_offscreen", source)
|
||||
self.assertIn("def ensure_target_window_reachable", source)
|
||||
|
||||
def test_reachability_check_sees_minimized_windows(self):
|
||||
"""It must NOT reuse list_windows().
|
||||
|
||||
`list_windows()` filters out invisible and zero-area windows — exactly
|
||||
the states the guard needs to observe in order to refuse. An earlier
|
||||
draft of this guard did reuse it, matched nothing, and passed
|
||||
everything. The enumeration has to be its own.
|
||||
"""
|
||||
tree = ast.parse(_win_source())
|
||||
fn = next(
|
||||
node for node in ast.walk(tree)
|
||||
if isinstance(node, ast.FunctionDef) and node.name == "_windows_for_bundle"
|
||||
)
|
||||
# Walk the AST rather than the text, so the explanatory docstring
|
||||
# (which names list_windows to say why it is NOT used) cannot satisfy
|
||||
# or break the assertion.
|
||||
called = {
|
||||
n.func.id for n in ast.walk(fn)
|
||||
if isinstance(n, ast.Call) and isinstance(n.func, ast.Name)
|
||||
}
|
||||
self.assertNotIn("list_windows", called)
|
||||
|
||||
source = _win_source()
|
||||
body = source[source.index("def _windows_for_bundle"):
|
||||
source.index("def ensure_target_window_reachable")]
|
||||
self.assertIn("EnumWindows", body)
|
||||
self.assertIn("SW_SHOWMINIMIZED", source)
|
||||
|
||||
def test_refusals_carry_a_machine_readable_code(self):
|
||||
source = _win_source()
|
||||
self.assertIn("class DeliveryRefused", source)
|
||||
self.assertIn("error_output(str(exc), code=exc.code)", source)
|
||||
|
||||
def test_refusal_says_the_action_was_not_sent(self):
|
||||
"""The message must state that nothing happened.
|
||||
|
||||
"Could not reach the window" reads like a warning attached to an action
|
||||
that still went out. The model needs to know the action did not happen,
|
||||
or it will assume it did and move on.
|
||||
"""
|
||||
source = _win_source()
|
||||
self.assertIn("was NOT sent", source)
|
||||
|
||||
|
||||
class TestCursorBadge(unittest.TestCase):
|
||||
"""The Windows badge annotates the real cursor; it does not replace it."""
|
||||
|
||||
def test_badge_script_exists(self):
|
||||
self.assertTrue(CURSOR_BADGE.exists())
|
||||
|
||||
def test_badge_is_click_through_and_never_takes_focus(self):
|
||||
"""Any of these missing turns the badge into an obstacle.
|
||||
|
||||
Without WS_EX_TRANSPARENT it eats the clicks it is meant to describe;
|
||||
without WS_EX_NOACTIVATE it steals focus from the app being driven —
|
||||
which would break the very action it is annotating.
|
||||
"""
|
||||
source = CURSOR_BADGE.read_text(encoding="utf-8")
|
||||
tree = ast.parse(source)
|
||||
|
||||
create = next(
|
||||
node for node in ast.walk(tree)
|
||||
if isinstance(node, ast.FunctionDef) and node.name == "create"
|
||||
)
|
||||
# Read the names actually combined into the window's ex-style, not
|
||||
# merely the ones defined somewhere in the file. A constant can be
|
||||
# defined and then left out of CreateWindowExW — which is exactly how
|
||||
# a click-through window quietly becomes a click-eating one.
|
||||
used = {
|
||||
n.id for n in ast.walk(create)
|
||||
if isinstance(n, ast.Name)
|
||||
}
|
||||
for style in ("WS_EX_LAYERED", "WS_EX_TRANSPARENT",
|
||||
"WS_EX_NOACTIVATE", "WS_EX_TOOLWINDOW"):
|
||||
self.assertIn(
|
||||
style, used,
|
||||
f"{style} must be passed to CreateWindowExW, not just defined",
|
||||
)
|
||||
self.assertIn("SW_SHOWNOACTIVATE", used)
|
||||
|
||||
def test_badge_does_not_draw_a_second_pointer(self):
|
||||
"""Windows has one real cursor and SendInput moves it.
|
||||
|
||||
Drawing a fake pointer alongside it would show the user two cursors,
|
||||
one of which is a lie about where the click will land. The macOS design
|
||||
does not transfer, and the source says so explicitly.
|
||||
"""
|
||||
source = CURSOR_BADGE.read_text(encoding="utf-8")
|
||||
self.assertIn("annotation", source.lower())
|
||||
|
||||
def test_badge_exits_with_its_parent(self):
|
||||
"""An orphaned badge is worse than none.
|
||||
|
||||
It would sit on screen claiming the agent is controlling the mouse
|
||||
after the agent is gone. Tying it to stdin covers the parent being
|
||||
killed, not just exiting cleanly.
|
||||
"""
|
||||
source = CURSOR_BADGE.read_text(encoding="utf-8")
|
||||
self.assertIn("stdin", source)
|
||||
|
||||
|
||||
class TestPermissions(unittest.TestCase):
|
||||
def test_check_permissions_always_granted(self):
|
||||
"""On Windows, permissions are not needed — should always be True."""
|
||||
if not WIN_HELPER.exists():
|
||||
self.skipTest("win_helper.py not found")
|
||||
|
||||
# Extract and exec just the check_permissions function
|
||||
source = WIN_HELPER.read_text()
|
||||
|
||||
# Find the function
|
||||
self.assertIn("def check_permissions()", source)
|
||||
|
||||
# The function should return both as True
|
||||
# We can verify by reading the source
|
||||
"""Windows has no TCC equivalent for input injection or capture."""
|
||||
source = _win_source()
|
||||
start = source.index("def check_permissions()")
|
||||
# Find next def or end
|
||||
rest = source[start:]
|
||||
lines = rest.split("\n")
|
||||
func_lines = [lines[0]]
|
||||
for line in lines[1:]:
|
||||
if line and not line[0].isspace() and not line.startswith("#"):
|
||||
break
|
||||
func_lines.append(line)
|
||||
func_source = "\n".join(func_lines)
|
||||
self.assertIn('"accessibility": True', func_source)
|
||||
self.assertIn('"screenRecording": True', func_source)
|
||||
body = source[start:start + 400]
|
||||
self.assertIn('"accessibility": True', body)
|
||||
self.assertIn('"screenRecording": True', body)
|
||||
|
||||
|
||||
class TestMacHelperPermissions(unittest.TestCase):
|
||||
"""macOS helper permission detection should use the official trust API."""
|
||||
class TestSourceIntegrity(unittest.TestCase):
|
||||
def test_helper_parses(self):
|
||||
ast.parse(_win_source())
|
||||
|
||||
def test_check_permissions_uses_ax_api_instead_of_system_events(self):
|
||||
if not MAC_HELPER.exists():
|
||||
self.skipTest("mac_helper.py not found")
|
||||
def test_badge_parses(self):
|
||||
ast.parse(CURSOR_BADGE.read_text(encoding="utf-8"))
|
||||
|
||||
source = MAC_HELPER.read_text()
|
||||
|
||||
self.assertIn("def detect_accessibility_permission()", source)
|
||||
self.assertIn("AXIsProcessTrusted", source)
|
||||
|
||||
start = source.index("def check_permissions()")
|
||||
rest = source[start:]
|
||||
lines = rest.split("\n")
|
||||
func_lines = [lines[0]]
|
||||
for line in lines[1:]:
|
||||
if line and not line[0].isspace() and not line.startswith("#"):
|
||||
break
|
||||
func_lines.append(line)
|
||||
func_source = "\n".join(func_lines)
|
||||
|
||||
self.assertIn("detect_accessibility_permission()", func_source)
|
||||
self.assertNotIn('tell application "System Events"', func_source)
|
||||
|
||||
def test_clipboard_shortcuts_use_osascript_path(self):
|
||||
if not MAC_HELPER.exists():
|
||||
self.skipTest("mac_helper.py not found")
|
||||
|
||||
source = MAC_HELPER.read_text()
|
||||
self.assertIn("def paste_clipboard()", source)
|
||||
self.assertIn('send_keystroke_via_osascript("v", ["command"])', source)
|
||||
self.assertIn('if parts == ["command", "v"]:', source)
|
||||
self.assertIn('elif parts == ["command", "a"]:', source)
|
||||
|
||||
|
||||
class TestCrossPlatformFunctions(unittest.TestCase):
|
||||
"""Test functions that are identical between both helpers."""
|
||||
|
||||
def _get_function_body(self, helper_path: Path, func_name: str) -> str:
|
||||
"""Extract a function's body (code lines only, no comments/blanks)."""
|
||||
source = helper_path.read_text()
|
||||
marker = f"def {func_name}("
|
||||
if marker not in source:
|
||||
return ""
|
||||
start = source.index(marker)
|
||||
rest = source[start:]
|
||||
lines = rest.split("\n")
|
||||
func_lines = [lines[0]]
|
||||
for line in lines[1:]:
|
||||
# Stop at next top-level def/class or non-indented non-empty line
|
||||
stripped = line.strip()
|
||||
if line and not line[0].isspace() and stripped and not stripped.startswith("#"):
|
||||
break
|
||||
# Skip comments and blank lines for comparison
|
||||
if stripped.startswith("#") or not stripped:
|
||||
continue
|
||||
func_lines.append(line)
|
||||
return " ".join(" ".join(func_lines).split())
|
||||
|
||||
def test_input_functions_identical(self):
|
||||
"""Input action functions (click, scroll, etc.) should be identical."""
|
||||
if not MAC_HELPER.exists() or not WIN_HELPER.exists():
|
||||
self.skipTest("Both helpers required")
|
||||
for func in ["click", "scroll", "hold_keys", "type_text"]:
|
||||
mac_src = self._get_function_body(MAC_HELPER, func)
|
||||
win_src = self._get_function_body(WIN_HELPER, func)
|
||||
self.assertEqual(mac_src, win_src,
|
||||
f"{func} should be identical across platforms")
|
||||
def test_retired_mac_helper_is_not_referenced(self):
|
||||
"""macOS is native-only; a lingering reference invites a false fallback."""
|
||||
self.assertFalse((RUNTIME_DIR / "mac_helper.py").exists())
|
||||
self.assertNotIn("mac_helper", _win_source())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
unittest.main(verbosity=2)
|
||||
|
||||
@@ -0,0 +1,253 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Windows agent-activity badge — a click-through marker that follows the cursor.
|
||||
|
||||
WHY THIS IS NOT THE macOS VIRTUAL CURSOR
|
||||
----------------------------------------
|
||||
On macOS the helper never moves the real pointer: `CGEvent.postToPid` carries
|
||||
the click coordinate as metadata, so the drawn cursor IS the only cursor the
|
||||
user sees move. It is a *replacement*.
|
||||
|
||||
Windows has no per-process event delivery. `pyautogui` bottoms out in
|
||||
`SendInput`, which warps the one real cursor the user's hand is also on. We
|
||||
cannot avoid that, so drawing a second fake pointer would be actively harmful:
|
||||
two pointers, one of them a lie, with no way to tell which one the OS is
|
||||
actually going to click with.
|
||||
|
||||
So this badge is an *annotation*, not a replacement. It rides just off the real
|
||||
cursor and answers exactly one question the user cannot otherwise answer:
|
||||
"is this thing moving because of me, or because of the agent?" On Windows that
|
||||
question has real stakes — the agent is holding the user's mouse, and the user
|
||||
needs to know before they grab it back mid-action.
|
||||
|
||||
Runs as its own process because the Windows helper is a stateless one-shot CLI:
|
||||
every command exits, so nothing in it can own a window across actions.
|
||||
|
||||
The window is WS_EX_LAYERED | WS_EX_TRANSPARENT | WS_EX_NOACTIVATE — it never
|
||||
takes focus, never appears in the taskbar or Alt-Tab, and passes every click
|
||||
through to whatever is underneath. It cannot intercept the input it exists to
|
||||
describe.
|
||||
|
||||
Usage:
|
||||
python win_cursor_badge.py --label "Claude" # runs until stdin closes
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import ctypes
|
||||
import sys
|
||||
import threading
|
||||
from ctypes import wintypes
|
||||
|
||||
user32 = ctypes.windll.user32
|
||||
gdi32 = ctypes.windll.gdi32
|
||||
kernel32 = ctypes.windll.kernel32
|
||||
|
||||
WS_EX_LAYERED = 0x00080000
|
||||
WS_EX_TRANSPARENT = 0x00000020
|
||||
WS_EX_TOPMOST = 0x00000008
|
||||
WS_EX_TOOLWINDOW = 0x00000080
|
||||
WS_EX_NOACTIVATE = 0x08000000
|
||||
WS_POPUP = 0x80000000
|
||||
|
||||
SW_SHOWNOACTIVATE = 4
|
||||
HWND_TOPMOST = -1
|
||||
SWP_NOACTIVATE = 0x0010
|
||||
SWP_NOSIZE = 0x0001
|
||||
SWP_NOZORDER = 0x0004
|
||||
|
||||
LWA_COLORKEY = 0x00000001
|
||||
LWA_ALPHA = 0x00000002
|
||||
|
||||
WM_DESTROY = 0x0002
|
||||
WM_PAINT = 0x000F
|
||||
|
||||
BADGE_W = 132
|
||||
BADGE_H = 30
|
||||
CURSOR_OFFSET_X = 18
|
||||
CURSOR_OFFSET_Y = 18
|
||||
|
||||
# Chroma key: pixels of this exact colour become fully transparent. Picked to
|
||||
# be a colour nothing in the badge draws, so only the intended shape shows.
|
||||
TRANSPARENT_KEY = 0x00FF00FF
|
||||
|
||||
|
||||
class POINT(ctypes.Structure):
|
||||
_fields_ = [("x", wintypes.LONG), ("y", wintypes.LONG)]
|
||||
|
||||
|
||||
class RECT(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("left", wintypes.LONG),
|
||||
("top", wintypes.LONG),
|
||||
("right", wintypes.LONG),
|
||||
("bottom", wintypes.LONG),
|
||||
]
|
||||
|
||||
|
||||
class PAINTSTRUCT(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("hdc", wintypes.HDC),
|
||||
("fErase", wintypes.BOOL),
|
||||
("rcPaint", RECT),
|
||||
("fRestore", wintypes.BOOL),
|
||||
("fIncUpdate", wintypes.BOOL),
|
||||
("rgbReserved", ctypes.c_byte * 32),
|
||||
]
|
||||
|
||||
|
||||
class WNDCLASS(ctypes.Structure):
|
||||
_fields_ = [
|
||||
("style", wintypes.UINT),
|
||||
("lpfnWndProc", ctypes.WINFUNCTYPE(
|
||||
ctypes.c_long, wintypes.HWND, wintypes.UINT,
|
||||
wintypes.WPARAM, wintypes.LPARAM)),
|
||||
("cbClsExtra", ctypes.c_int),
|
||||
("cbWndExtra", ctypes.c_int),
|
||||
("hInstance", wintypes.HINSTANCE),
|
||||
("hIcon", wintypes.HICON),
|
||||
("hCursor", wintypes.HANDLE),
|
||||
("hbrBackground", wintypes.HBRUSH),
|
||||
("lpszMenuName", wintypes.LPCWSTR),
|
||||
("lpszClassName", wintypes.LPCWSTR),
|
||||
]
|
||||
|
||||
|
||||
WNDPROC = ctypes.WINFUNCTYPE(
|
||||
ctypes.c_long, wintypes.HWND, wintypes.UINT, wintypes.WPARAM, wintypes.LPARAM
|
||||
)
|
||||
|
||||
|
||||
class CursorBadge:
|
||||
def __init__(self, label: str) -> None:
|
||||
self.label = label
|
||||
self.hwnd: int | None = None
|
||||
self._stop = threading.Event()
|
||||
# Held on the instance because ctypes does not keep the trampoline
|
||||
# alive on its own; letting it be collected turns the next window
|
||||
# message into a crash inside the message pump.
|
||||
self._wndproc = WNDPROC(self._on_message)
|
||||
|
||||
def _on_message(self, hwnd, msg, wparam, lparam):
|
||||
if msg == WM_PAINT:
|
||||
self._paint(hwnd)
|
||||
return 0
|
||||
if msg == WM_DESTROY:
|
||||
user32.PostQuitMessage(0)
|
||||
return 0
|
||||
return user32.DefWindowProcW(hwnd, msg, wparam, lparam)
|
||||
|
||||
def _paint(self, hwnd: int) -> None:
|
||||
ps = PAINTSTRUCT()
|
||||
hdc = user32.BeginPaint(hwnd, ctypes.byref(ps))
|
||||
try:
|
||||
rect = RECT(0, 0, BADGE_W, BADGE_H)
|
||||
|
||||
# Fill with the chroma key first: everything we do not draw over
|
||||
# becomes transparent, which is what gives the badge its shape.
|
||||
key_brush = gdi32.CreateSolidBrush(TRANSPARENT_KEY)
|
||||
user32.FillRect(hdc, ctypes.byref(rect), key_brush)
|
||||
gdi32.DeleteObject(key_brush)
|
||||
|
||||
body = RECT(0, 0, BADGE_W, BADGE_H)
|
||||
bg = gdi32.CreateSolidBrush(0x00734B23) # BGR: a muted blue
|
||||
user32.FillRect(hdc, ctypes.byref(body), bg)
|
||||
gdi32.DeleteObject(bg)
|
||||
|
||||
gdi32.SetBkMode(hdc, 1) # TRANSPARENT
|
||||
gdi32.SetTextColor(hdc, 0x00FFFFFF)
|
||||
text = f" {self.label} is controlling"
|
||||
user32.DrawTextW(
|
||||
hdc, text, len(text), ctypes.byref(body),
|
||||
0x00000004 | 0x00000100, # DT_VCENTER | DT_SINGLELINE
|
||||
)
|
||||
finally:
|
||||
user32.EndPaint(hwnd, ctypes.byref(ps))
|
||||
|
||||
def create(self) -> None:
|
||||
hinst = kernel32.GetModuleHandleW(None)
|
||||
class_name = "CcHahaAgentCursorBadge"
|
||||
|
||||
wc = WNDCLASS()
|
||||
wc.lpfnWndProc = self._wndproc
|
||||
wc.hInstance = hinst
|
||||
wc.lpszClassName = class_name
|
||||
wc.hbrBackground = 0
|
||||
wc.hCursor = 0
|
||||
user32.RegisterClassW(ctypes.byref(wc))
|
||||
|
||||
self.hwnd = user32.CreateWindowExW(
|
||||
WS_EX_LAYERED | WS_EX_TRANSPARENT | WS_EX_TOPMOST
|
||||
| WS_EX_TOOLWINDOW | WS_EX_NOACTIVATE,
|
||||
class_name, None, WS_POPUP,
|
||||
0, 0, BADGE_W, BADGE_H,
|
||||
None, None, hinst, None,
|
||||
)
|
||||
if not self.hwnd:
|
||||
raise OSError("CreateWindowExW failed for the cursor badge")
|
||||
|
||||
user32.SetLayeredWindowAttributes(
|
||||
self.hwnd, TRANSPARENT_KEY, 225, LWA_COLORKEY | LWA_ALPHA
|
||||
)
|
||||
user32.ShowWindow(self.hwnd, SW_SHOWNOACTIVATE)
|
||||
|
||||
def _follow_cursor(self) -> None:
|
||||
"""Reposition the badge next to the real pointer, ~60fps."""
|
||||
pt = POINT()
|
||||
while not self._stop.is_set():
|
||||
try:
|
||||
if user32.GetCursorPos(ctypes.byref(pt)) and self.hwnd:
|
||||
user32.SetWindowPos(
|
||||
self.hwnd, HWND_TOPMOST,
|
||||
pt.x + CURSOR_OFFSET_X, pt.y + CURSOR_OFFSET_Y,
|
||||
0, 0, SWP_NOACTIVATE | SWP_NOSIZE,
|
||||
)
|
||||
except Exception:
|
||||
# The badge is advisory. It must never be the reason an action
|
||||
# fails, so every error here is swallowed and the loop retries.
|
||||
pass
|
||||
self._stop.wait(0.016)
|
||||
|
||||
def _wait_for_stdin_close(self) -> None:
|
||||
"""Exit when the parent goes away.
|
||||
|
||||
The badge outliving its parent would leave a permanent 'the agent is
|
||||
controlling your mouse' claim on screen with nothing behind it. Reading
|
||||
stdin to EOF ties this process's lifetime to the parent's, including
|
||||
the case where the parent is killed rather than exiting cleanly.
|
||||
"""
|
||||
try:
|
||||
for _ in sys.stdin:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
self.stop()
|
||||
|
||||
def stop(self) -> None:
|
||||
self._stop.set()
|
||||
if self.hwnd:
|
||||
user32.PostMessageW(self.hwnd, WM_DESTROY, 0, 0)
|
||||
|
||||
def run(self) -> int:
|
||||
self.create()
|
||||
threading.Thread(target=self._follow_cursor, daemon=True).start()
|
||||
threading.Thread(target=self._wait_for_stdin_close, daemon=True).start()
|
||||
|
||||
msg = wintypes.MSG()
|
||||
while user32.GetMessageW(ctypes.byref(msg), None, 0, 0) > 0:
|
||||
user32.TranslateMessage(ctypes.byref(msg))
|
||||
user32.DispatchMessageW(ctypes.byref(msg))
|
||||
return 0
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--label", default="Claude")
|
||||
args = parser.parse_args()
|
||||
if sys.platform != "win32":
|
||||
print("win_cursor_badge.py is Windows-only", file=sys.stderr)
|
||||
return 1
|
||||
return CursorBadge(args.label).run()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
+411
-26
@@ -1,8 +1,28 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Windows Computer Use helper — same JSON protocol as mac_helper.py.
|
||||
"""Windows Computer Use helper.
|
||||
|
||||
Uses win32gui / win32api / win32process / psutil / pyperclip / screeninfo
|
||||
to replicate macOS-specific Quartz/AppKit functionality on Windows.
|
||||
Uses win32gui / win32api / win32process / psutil / pyperclip / screeninfo /
|
||||
pyautogui to provide, on Windows, the JSON command protocol the native macOS
|
||||
`cu-helper` daemon speaks. macOS is native-only — there is no Python path there
|
||||
— so this is the sole implementation of that protocol in Python.
|
||||
|
||||
One difference is not an implementation detail and shapes everything below:
|
||||
macOS delivers input with `CGEvent.postToPid`, straight into the target
|
||||
process, leaving the real cursor and the foreground app alone. Windows has no
|
||||
equivalent. `pyautogui` bottoms out in `SendInput`, which injects into the one
|
||||
system-wide input stream and warps the one real cursor. The agent therefore
|
||||
shares the mouse and keyboard with the user, and cannot verify that anything
|
||||
it sent arrived.
|
||||
|
||||
Hence the two mechanisms that have no macOS counterpart:
|
||||
|
||||
* `ForegroundLease` aborts when physical input overlaps an action, because
|
||||
interleaved streams produce clicks neither party intended.
|
||||
* `ensure_point_on_screen` / `ensure_target_window_reachable` refuse to send
|
||||
at all when delivery is already known to be impossible.
|
||||
|
||||
Both exist because `SendInput` reports success unconditionally, and "Action
|
||||
completed" for input that went nowhere is worse than an error.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -176,7 +196,7 @@ def choose_display(display_id: int | None) -> dict[str, Any]:
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Screen capture (mss — cross-platform, identical to mac_helper)
|
||||
# Screen capture (mss)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def capture_display(display_id: int | None, resize: tuple[int, int] | None = None) -> dict[str, Any]:
|
||||
@@ -563,6 +583,145 @@ def paste_clipboard() -> None:
|
||||
pyautogui.hotkey("ctrl", "v", interval=0.02)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Physical input interference detection
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# Why this exists at all, and why it is stricter than the macOS version.
|
||||
#
|
||||
# On macOS the helper posts events straight into the target process with
|
||||
# `CGEvent.postToPid`, so agent input and human input never share a channel:
|
||||
# the epoch monitor there is a safety net for an unlikely race.
|
||||
#
|
||||
# Windows has no such API. `pyautogui` bottoms out in `SendInput`, which
|
||||
# injects into the ONE system-wide input stream and warps the ONE real cursor.
|
||||
# The agent and the user are therefore holding the same mouse. If the user
|
||||
# reaches for it mid-action the two streams interleave, and the resulting
|
||||
# click lands somewhere neither of them intended. Detection is not a nicety
|
||||
# here — it is the only thing standing between "the agent typed into the wrong
|
||||
# window" and an abort.
|
||||
#
|
||||
# `GetLastInputInfo` is the right signal for this: it reports the tick of the
|
||||
# last PHYSICAL input event, requires no privileges and no TCC-style grant,
|
||||
# and — measured, and asserted by test_helpers.py — is NOT advanced by
|
||||
# `SendInput` injection, so the agent cannot trip its own detector.
|
||||
|
||||
import ctypes
|
||||
from ctypes import wintypes
|
||||
|
||||
|
||||
class _LASTINPUTINFO(ctypes.Structure):
|
||||
_fields_ = [("cbSize", wintypes.UINT), ("dwTime", wintypes.DWORD)]
|
||||
|
||||
|
||||
def last_physical_input_tick() -> int:
|
||||
"""Tick count of the last physical keyboard/mouse event.
|
||||
|
||||
Returns 0 when unavailable so callers fail OPEN on the read itself: a
|
||||
helper that refused to act because it could not query an optional Win32
|
||||
counter would be broken in a much more visible way than one that acted.
|
||||
Interference is only ever reported on two SUCCESSFUL reads that differ.
|
||||
"""
|
||||
try:
|
||||
info = _LASTINPUTINFO()
|
||||
info.cbSize = ctypes.sizeof(_LASTINPUTINFO)
|
||||
if not ctypes.windll.user32.GetLastInputInfo(ctypes.byref(info)):
|
||||
return 0
|
||||
return int(info.dwTime)
|
||||
except Exception:
|
||||
return 0
|
||||
|
||||
|
||||
class UserInterference(RuntimeError):
|
||||
"""The user touched the physical mouse or keyboard during an action."""
|
||||
|
||||
def __init__(self, message: str, code: str = "user_interference") -> None:
|
||||
super().__init__(message)
|
||||
self.code = code
|
||||
|
||||
|
||||
def _foreground_window_pid() -> int | None:
|
||||
try:
|
||||
import win32gui
|
||||
import win32process
|
||||
hwnd = win32gui.GetForegroundWindow()
|
||||
if not hwnd:
|
||||
return None
|
||||
_, pid = win32process.GetWindowThreadProcessId(hwnd)
|
||||
return int(pid)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
class ForegroundLease:
|
||||
"""Guards one mutating action against concurrent physical input.
|
||||
|
||||
Evidence is sampled in a fixed order — tick, foreground identity, tick —
|
||||
both before and after the action, so a single observation cannot straddle
|
||||
a change it fails to notice. Same shape as `ForegroundLease.swift`.
|
||||
|
||||
The asymmetry between the two failure modes is deliberate and is the whole
|
||||
point of the class:
|
||||
|
||||
* interference BEFORE the action -> `user_interference`. Nothing ran.
|
||||
The caller may safely retry.
|
||||
* interference DURING the action -> `user_interference_result_unknown`.
|
||||
Injection already went into the shared input stream and we cannot know
|
||||
how much of it landed, or where. Retrying could double-apply it. The
|
||||
error says so rather than guessing.
|
||||
"""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.tick: int = 0
|
||||
self.pid: int | None = None
|
||||
|
||||
def acquire(self) -> None:
|
||||
before = last_physical_input_tick()
|
||||
pid = _foreground_window_pid()
|
||||
after = last_physical_input_tick()
|
||||
if before and after and before != after:
|
||||
raise UserInterference(
|
||||
"The user was typing or moving the mouse, so the action was "
|
||||
"not sent. Nothing has changed; it is safe to try again."
|
||||
)
|
||||
self.tick = after
|
||||
self.pid = pid
|
||||
|
||||
def finalize(self) -> None:
|
||||
before = last_physical_input_tick()
|
||||
pid = _foreground_window_pid()
|
||||
after = last_physical_input_tick()
|
||||
|
||||
if before and after and before != after:
|
||||
raise UserInterference(
|
||||
"The user used the mouse or keyboard while this action was "
|
||||
"running. Because Windows shares one input stream between you "
|
||||
"and the user, the two may have interleaved and the result is "
|
||||
"UNKNOWN. Do not repeat the action — take a screenshot and "
|
||||
"read the current state before deciding anything.",
|
||||
code="user_interference_result_unknown",
|
||||
)
|
||||
|
||||
if self.tick and after and self.tick != after:
|
||||
raise UserInterference(
|
||||
"The user used the mouse or keyboard while this action was "
|
||||
"running. The result is UNKNOWN — do not repeat the action; "
|
||||
"take a screenshot and read the current state first.",
|
||||
code="user_interference_result_unknown",
|
||||
)
|
||||
|
||||
# A foreground change without any physical input is the target app (or
|
||||
# a background app) stealing activation, not the user. Worth reporting,
|
||||
# because everything typed after it went somewhere unintended.
|
||||
if self.pid is not None and pid is not None and self.pid != pid:
|
||||
raise UserInterference(
|
||||
"The foreground application changed while this action was "
|
||||
"running, so input may have gone to the wrong window. The "
|
||||
"result is UNKNOWN — take a screenshot before continuing.",
|
||||
code="user_interference_result_unknown",
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Permissions — Windows doesn't have macOS-style TCC
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -577,7 +736,177 @@ def check_permissions() -> dict[str, bool | None]:
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Input actions (pyautogui — identical to mac_helper)
|
||||
# Delivery preconditions — refuse rather than report a lie
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# `SendInput` always "succeeds": it returns the number of events inserted into
|
||||
# the input stream, never whether anything acted on them. Click a point behind
|
||||
# another window and the click lands on THAT window; click a point off-screen
|
||||
# and it lands nowhere. Either way pyautogui returns cleanly and the helper
|
||||
# would answer "Action completed".
|
||||
#
|
||||
# That specific lie has burned us before on macOS — a session typed into a
|
||||
# minimized window for a full turn because every action reported success. The
|
||||
# fix there was to refuse instead of guessing, and the same rule applies here.
|
||||
|
||||
class DeliveryRefused(RuntimeError):
|
||||
def __init__(self, message: str, code: str) -> None:
|
||||
super().__init__(message)
|
||||
self.code = code
|
||||
|
||||
|
||||
def _virtual_screen_rect() -> tuple[int, int, int, int] | None:
|
||||
"""(left, top, right, bottom) across all monitors, or None if unavailable."""
|
||||
try:
|
||||
user32 = ctypes.windll.user32
|
||||
SM_XVIRTUALSCREEN, SM_YVIRTUALSCREEN = 76, 77
|
||||
SM_CXVIRTUALSCREEN, SM_CYVIRTUALSCREEN = 78, 79
|
||||
left = user32.GetSystemMetrics(SM_XVIRTUALSCREEN)
|
||||
top = user32.GetSystemMetrics(SM_YVIRTUALSCREEN)
|
||||
width = user32.GetSystemMetrics(SM_CXVIRTUALSCREEN)
|
||||
height = user32.GetSystemMetrics(SM_CYVIRTUALSCREEN)
|
||||
if width <= 0 or height <= 0:
|
||||
return None
|
||||
return (left, top, left + width, top + height)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def ensure_point_on_screen(x: int, y: int) -> None:
|
||||
"""Refuse coordinates outside every monitor.
|
||||
|
||||
Fails OPEN when the metrics are unreadable: an unreadable metric is our
|
||||
problem, not the caller's, and blocking every action on it would be worse
|
||||
than the miss it prevents.
|
||||
"""
|
||||
rect = _virtual_screen_rect()
|
||||
if rect is None:
|
||||
return
|
||||
left, top, right, bottom = rect
|
||||
if left <= x < right and top <= y < bottom:
|
||||
return
|
||||
raise DeliveryRefused(
|
||||
f"The point ({x}, {y}) is outside every display "
|
||||
f"(virtual screen is {left},{top} to {right},{bottom}), so the action "
|
||||
"was not sent. Take a screenshot to get current coordinates.",
|
||||
code="point_outside_display",
|
||||
)
|
||||
|
||||
|
||||
def _window_is_interactable(hwnd: int) -> tuple[bool, str]:
|
||||
"""(ok, reason) — whether synthetic input can reach this window at all."""
|
||||
try:
|
||||
import win32gui
|
||||
if not win32gui.IsWindow(hwnd):
|
||||
return False, "the window no longer exists"
|
||||
if not win32gui.IsWindowVisible(hwnd):
|
||||
return False, "the window is hidden"
|
||||
try:
|
||||
import win32con
|
||||
placement = win32gui.GetWindowPlacement(hwnd)
|
||||
if placement and placement[1] == win32con.SW_SHOWMINIMIZED:
|
||||
return False, "the window is minimized"
|
||||
except Exception:
|
||||
pass
|
||||
rect = win32gui.GetWindowRect(hwnd)
|
||||
if rect[2] - rect[0] <= 0 or rect[3] - rect[1] <= 0:
|
||||
return False, "the window has no on-screen area"
|
||||
return True, ""
|
||||
except Exception:
|
||||
# Unreadable window state fails open, same reasoning as above.
|
||||
return True, ""
|
||||
|
||||
|
||||
def _windows_for_bundle(bundle_id: str) -> list[int]:
|
||||
"""Every top-level HWND owned by a process whose exe stem matches.
|
||||
|
||||
Enumerates directly rather than reusing `list_windows()`, which filters out
|
||||
invisible and zero-area windows — precisely the states this guard needs to
|
||||
SEE in order to refuse. Reusing it would make the guard match nothing and
|
||||
silently pass, which is the failure mode it was written to prevent.
|
||||
"""
|
||||
try:
|
||||
import win32gui
|
||||
import win32process
|
||||
import psutil
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
wanted = bundle_id.strip().lower()
|
||||
if not wanted:
|
||||
return []
|
||||
|
||||
pids: set[int] = set()
|
||||
try:
|
||||
for proc in psutil.process_iter(["pid", "name", "exe"]):
|
||||
try:
|
||||
exe_path = proc.info.get("exe") or ""
|
||||
name = proc.info.get("name") or ""
|
||||
stem = Path(exe_path).stem if exe_path else Path(name).stem
|
||||
if stem and stem.lower() == wanted:
|
||||
pids.add(int(proc.info["pid"]))
|
||||
except (psutil.NoSuchProcess, psutil.AccessDenied):
|
||||
continue
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
if not pids:
|
||||
return []
|
||||
|
||||
handles: list[int] = []
|
||||
|
||||
def _collect(hwnd: int, _: Any) -> None:
|
||||
try:
|
||||
_, pid = win32process.GetWindowThreadProcessId(hwnd)
|
||||
if int(pid) in pids:
|
||||
handles.append(int(hwnd))
|
||||
except Exception:
|
||||
return
|
||||
|
||||
try:
|
||||
win32gui.EnumWindows(_collect, None)
|
||||
except Exception:
|
||||
return []
|
||||
return handles
|
||||
|
||||
|
||||
def ensure_target_window_reachable(bundle_id: str | None) -> None:
|
||||
"""Refuse when the named app has no window that input could reach.
|
||||
|
||||
A minimized window is the case that matters: on Windows it has no client
|
||||
area to hit-test against, so a coordinate click is guaranteed to land on
|
||||
whatever is underneath it. Reporting success there is exactly the lie this
|
||||
guard exists to prevent.
|
||||
|
||||
Fails OPEN when the app owns no top-level windows at all — that is a
|
||||
different failure (wrong app name, app not running) which the caller's own
|
||||
resolution step reports with a better message than this one could.
|
||||
"""
|
||||
if not bundle_id:
|
||||
return
|
||||
|
||||
handles = _windows_for_bundle(bundle_id)
|
||||
if not handles:
|
||||
return
|
||||
|
||||
reasons: list[str] = []
|
||||
for hwnd in handles:
|
||||
ok, reason = _window_is_interactable(hwnd)
|
||||
if ok:
|
||||
return
|
||||
if reason:
|
||||
reasons.append(reason)
|
||||
|
||||
detail = reasons[0] if reasons else "it has no on-screen window"
|
||||
raise DeliveryRefused(
|
||||
f"The target app has no window that input can reach — {detail}. "
|
||||
"The action was NOT sent. Restore the window and try again.",
|
||||
code="target_window_offscreen",
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Input actions (pyautogui → SendInput)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def click(x: int, y: int, button: str, count: int, modifiers: list[str] | None) -> None:
|
||||
@@ -629,9 +958,56 @@ def type_text(text: str) -> None:
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Main dispatcher — exact same command protocol as mac_helper.py
|
||||
# Main dispatcher — the command protocol the native macOS daemon also speaks
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Commands that inject into the shared Windows input stream. Kept as one set
|
||||
# rather than as a guard call inside each branch, because the branches are the
|
||||
# easy place to forget one — and a forgotten branch is silently unguarded, the
|
||||
# exact class of bug this whole pass exists to remove.
|
||||
#
|
||||
# Mirrors `CommandForegroundPolicy.leasedCommands` on the macOS side.
|
||||
MUTATING_COMMANDS = frozenset({
|
||||
"click", "drag", "move_mouse", "scroll",
|
||||
"mouse_down", "mouse_up",
|
||||
"key", "hold_key", "type",
|
||||
"paste_clipboard",
|
||||
})
|
||||
|
||||
# The subset that targets a screen coordinate, and so needs the point itself to
|
||||
# be reachable. `key`/`type` go to whatever holds focus and have no coordinate
|
||||
# to check.
|
||||
COORDINATE_COMMANDS = frozenset({"click", "drag", "move_mouse", "scroll"})
|
||||
|
||||
|
||||
def _coordinate_of(command: str, payload: dict[str, Any]) -> tuple[int, int] | None:
|
||||
if command not in COORDINATE_COMMANDS:
|
||||
return None
|
||||
if command == "drag":
|
||||
target = payload.get("to") or {}
|
||||
if "x" in target and "y" in target:
|
||||
return int(target["x"]), int(target["y"])
|
||||
return None
|
||||
if "x" in payload and "y" in payload:
|
||||
return int(payload["x"]), int(payload["y"])
|
||||
return None
|
||||
|
||||
|
||||
def _finish(lease: "ForegroundLease | None", result: Any) -> int:
|
||||
"""Emit the success response for a mutating command, after the lease agrees.
|
||||
|
||||
The check runs BEFORE the response is written, and that ordering is the
|
||||
whole point: once `{"ok": true}` reaches the caller the action is reported
|
||||
as done, and no later discovery can take that back. A helper that injected
|
||||
input, then noticed the user had been typing throughout, and still answered
|
||||
"Action completed" would be lying with a straight face.
|
||||
"""
|
||||
if lease is not None:
|
||||
lease.finalize()
|
||||
json_output({"ok": True, "result": result})
|
||||
return 0
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("command")
|
||||
@@ -639,8 +1015,20 @@ def main() -> int:
|
||||
args = parser.parse_args()
|
||||
payload = json.loads(args.payload)
|
||||
|
||||
lease: ForegroundLease | None = None
|
||||
|
||||
try:
|
||||
command = args.command
|
||||
|
||||
if command in MUTATING_COMMANDS:
|
||||
point = _coordinate_of(command, payload)
|
||||
if point is not None:
|
||||
ensure_point_on_screen(point[0], point[1])
|
||||
ensure_target_window_reachable(
|
||||
payload.get("bundleId") or payload.get("app")
|
||||
)
|
||||
lease = ForegroundLease()
|
||||
lease.acquire()
|
||||
if command == "check_permissions":
|
||||
perms = check_permissions()
|
||||
json_output({"ok": True, "result": perms})
|
||||
@@ -690,43 +1078,34 @@ def main() -> int:
|
||||
return 0
|
||||
if command == "key":
|
||||
key_action(str(payload["keySequence"]), int(payload.get("repeat") or 1))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
if command == "hold_key":
|
||||
hold_keys(list(payload.get("keyNames") or []), int(payload.get("durationMs") or 0))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
if command == "type":
|
||||
type_text(str(payload.get("text") or ""))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
if command == "click":
|
||||
click(int(payload["x"]), int(payload["y"]), str(payload.get("button") or "left"), int(payload.get("count") or 1), payload.get("modifiers"))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
if command == "drag":
|
||||
from_point = payload.get("from")
|
||||
if from_point:
|
||||
pyautogui.moveTo(int(from_point["x"]), int(from_point["y"]))
|
||||
pyautogui.dragTo(int(payload["to"]["x"]), int(payload["to"]["y"]), duration=0.2, button="left")
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
if command == "move_mouse":
|
||||
pyautogui.moveTo(int(payload["x"]), int(payload["y"]))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
if command == "scroll":
|
||||
scroll(int(payload["x"]), int(payload["y"]), int(payload.get("deltaX") or 0), int(payload.get("deltaY") or 0))
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
if command == "mouse_down":
|
||||
pyautogui.mouseDown(button="left")
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
if command == "mouse_up":
|
||||
pyautogui.mouseUp(button="left")
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
if command == "cursor_position":
|
||||
x, y = pyautogui.position()
|
||||
json_output({"ok": True, "result": {"x": int(x), "y": int(y)}})
|
||||
@@ -756,10 +1135,16 @@ def main() -> int:
|
||||
return 0
|
||||
if command == "paste_clipboard":
|
||||
paste_clipboard()
|
||||
json_output({"ok": True, "result": True})
|
||||
return 0
|
||||
return _finish(lease, True)
|
||||
error_output(f"Unknown command: {command}", code="bad_command")
|
||||
return 2
|
||||
except (UserInterference, DeliveryRefused) as exc:
|
||||
# A deliberate refusal, not a crash. The code travels so the caller can
|
||||
# tell "did not run, safe to retry" apart from "ran, outcome unknown" —
|
||||
# collapsing both into a generic error is how a model ends up repeating
|
||||
# a toggle it already flipped.
|
||||
error_output(str(exc), code=exc.code)
|
||||
return 1
|
||||
except Exception as exc:
|
||||
error_output(str(exc))
|
||||
return 1
|
||||
|
||||
@@ -99,3 +99,37 @@ describe('cleanupComputerUseAfterTurn — turn-end overlay hide', () => {
|
||||
expect(hidden).toBe(1)
|
||||
})
|
||||
})
|
||||
|
||||
describe('cleanupComputerUseAfterTurn — turn-end cursor badge', () => {
|
||||
test('drops the Windows badge on a turn that hid nothing', async () => {
|
||||
// The badge is a standing claim that the agent is holding the mouse. If it
|
||||
// outlives the turn it is simply false, and the user has no way to tell
|
||||
// that from a turn still in progress.
|
||||
let hidden = 0
|
||||
await cleanupComputerUseAfterTurn(makeCtx(), {
|
||||
overlayHide: async () => {},
|
||||
hideCursorBadge: () => { hidden += 1 },
|
||||
})
|
||||
expect(hidden).toBe(1)
|
||||
})
|
||||
|
||||
test('drops the badge even when overlayHide hangs past its timeout', async () => {
|
||||
// Abort paths are exactly when a stuck indicator is most likely and most
|
||||
// confusing. The badge teardown must not sit behind the daemon's.
|
||||
let hidden = 0
|
||||
await cleanupComputerUseAfterTurn(makeCtx(), {
|
||||
overlayHide: () => new Promise<void>(() => {}),
|
||||
hideCursorBadge: () => { hidden += 1 },
|
||||
})
|
||||
expect(hidden).toBe(1)
|
||||
})
|
||||
|
||||
test('drops the badge even when overlayHide rejects', async () => {
|
||||
let hidden = 0
|
||||
await cleanupComputerUseAfterTurn(makeCtx(), {
|
||||
overlayHide: async () => { throw new Error('daemon gone') },
|
||||
hideCursorBadge: () => { hidden += 1 },
|
||||
})
|
||||
expect(hidden).toBe(1)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -22,9 +22,9 @@ const UNHIDE_TIMEOUT_MS = 5000
|
||||
const OVERLAY_HIDE_TIMEOUT_MS = 2000
|
||||
|
||||
/**
|
||||
* Turn-end cleanup for the chicago MCP surface: hide the native overlay (macOS
|
||||
* cu-helper daemon), auto-unhide apps that `prepareForAction` hid, then release
|
||||
* the file-based lock.
|
||||
* Turn-end cleanup for the chicago MCP surface: drop the activity indicator
|
||||
* (macOS cu-helper overlay, or the Windows cursor badge), auto-unhide apps
|
||||
* that `prepareForAction` hid, then release the file-based lock.
|
||||
*
|
||||
* Called from three sites: natural turn end (`stopHooks.ts`), abort during
|
||||
* streaming (`query.ts` aborted_streaming), abort during tool execution
|
||||
@@ -49,8 +49,22 @@ export async function cleanupComputerUseAfterTurn(
|
||||
ToolUseContext,
|
||||
'getAppState' | 'setAppState' | 'sendOSNotification'
|
||||
>,
|
||||
deps: { overlayHide?: () => Promise<void> } = {},
|
||||
deps: {
|
||||
overlayHide?: () => Promise<void>
|
||||
hideCursorBadge?: () => void
|
||||
} = {},
|
||||
): Promise<void> {
|
||||
// Windows counterpart to the macOS overlay. Synchronous, no-throw, and a
|
||||
// no-op off-Windows, so it goes first and unconditionally: an orphaned badge
|
||||
// would sit on screen claiming the agent is holding the mouse after the turn
|
||||
// has ended, which is a worse lie than showing nothing at all.
|
||||
const hideBadge =
|
||||
deps.hideCursorBadge ??
|
||||
(() => {
|
||||
void import('./winCursorBadge.js').then(m => m.hideCursorBadge()).catch(() => {})
|
||||
})
|
||||
hideBadge()
|
||||
|
||||
// Drop the daemon overlay FIRST — before the hidden-apps block and before the
|
||||
// isLockHeldLocally early-return below — so the cursor drops promptly even on a
|
||||
// turn that hid no apps and whose lock-release short-circuits. overlayHide
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
/**
|
||||
* CLI `ComputerExecutor` implementation — Python bridge variant.
|
||||
* CLI `ComputerExecutor` implementation — platform-routed helper variant.
|
||||
*
|
||||
* Replaces the native Swift/Rust modules with a Python subprocess bridge
|
||||
* (pyautogui + mss + platform helpers). See `pythonBridge.ts` and
|
||||
* `runtime/{mac,win}_helper.py`.
|
||||
* Every command goes through `helperBridge.callHelper`, which routes by
|
||||
* platform: macOS reaches the signed native `cu-helper` daemon, Windows
|
||||
* reaches the Python helper (`runtime/win_helper.py`, pyautogui + mss).
|
||||
* This module is deliberately platform-agnostic — the routing decision, and
|
||||
* the very different guarantees each side offers, live in `helperBridge.ts`.
|
||||
*/
|
||||
|
||||
import type {
|
||||
@@ -29,8 +31,8 @@ import {
|
||||
isComputerUseSupportedPlatform,
|
||||
} from './common.js'
|
||||
// Platform-routed helper: macOS → native cu-helper (no cursor steal),
|
||||
// Windows → Python helper. Aliased so the 20+ call sites below stay unchanged.
|
||||
import { callHelper as callPythonHelper } from './helperBridge.js'
|
||||
// Windows → Python helper (pyautogui, which does move the real cursor).
|
||||
import { callHelper } from './helperBridge.js'
|
||||
|
||||
const SCREENSHOT_JPEG_QUALITY = 0.75
|
||||
const MOVE_SETTLE_MS = 50
|
||||
@@ -62,16 +64,16 @@ function normalizeDisplayGeometry(display: PythonDisplayGeometry): DisplayGeomet
|
||||
}
|
||||
|
||||
async function readClipboardViaPbpaste(): Promise<string> {
|
||||
return callPythonHelper<string>('read_clipboard', {})
|
||||
return callHelper<string>('read_clipboard', {})
|
||||
}
|
||||
|
||||
async function writeClipboardViaPbcopy(text: string): Promise<void> {
|
||||
await callPythonHelper('write_clipboard', { text })
|
||||
await callHelper('write_clipboard', { text })
|
||||
}
|
||||
|
||||
async function readClipboard(): Promise<string> {
|
||||
if (process.platform === 'win32') {
|
||||
return callPythonHelper<string>('read_clipboard', {})
|
||||
return callHelper<string>('read_clipboard', {})
|
||||
}
|
||||
|
||||
return readClipboardViaPbpaste()
|
||||
@@ -79,7 +81,7 @@ async function readClipboard(): Promise<string> {
|
||||
|
||||
async function writeClipboard(text: string): Promise<void> {
|
||||
if (process.platform === 'win32') {
|
||||
await callPythonHelper('write_clipboard', { text })
|
||||
await callHelper('write_clipboard', { text })
|
||||
return
|
||||
}
|
||||
|
||||
@@ -155,21 +157,21 @@ function formatAppList(apps: readonly DaemonAppRef[]): string {
|
||||
export function createCodexEngine(): CodexComputerEngine {
|
||||
return {
|
||||
async listApps(): Promise<string> {
|
||||
const apps = await callPythonHelper<DaemonAppRef[]>('list_apps', {})
|
||||
const apps = await callHelper<DaemonAppRef[]>('list_apps', {})
|
||||
return formatAppList(apps)
|
||||
},
|
||||
|
||||
async resolveTarget(target: AppTarget): Promise<ResolvedAppTarget> {
|
||||
// Sent as-is: the daemon owns the selector→process mapping, and it must
|
||||
// never launch anything to satisfy a match.
|
||||
return callPythonHelper<ResolvedAppTarget>('resolve_app_target', target)
|
||||
return callHelper<ResolvedAppTarget>('resolve_app_target', target)
|
||||
},
|
||||
|
||||
async getAppState(
|
||||
target: AppTarget,
|
||||
opts?: { disableDiff?: boolean },
|
||||
): Promise<AppStateResult> {
|
||||
return callPythonHelper<AppStateResult>('get_app_state', {
|
||||
return callHelper<AppStateResult>('get_app_state', {
|
||||
...appTargetPayload(target),
|
||||
...(opts?.disableDiff === undefined ? {} : { disableDiff: opts.disableDiff }),
|
||||
})
|
||||
@@ -183,7 +185,7 @@ export function createCodexEngine(): CodexComputerEngine {
|
||||
clickCount?: number
|
||||
button?: CodexMouseButton
|
||||
}): Promise<void> {
|
||||
await callPythonHelper('click', {
|
||||
await callHelper('click', {
|
||||
...appTargetPayload(args.target),
|
||||
index: args.index,
|
||||
x: args.x,
|
||||
@@ -198,7 +200,7 @@ export function createCodexEngine(): CodexComputerEngine {
|
||||
index: string
|
||||
value: string
|
||||
}): Promise<SetValueResult> {
|
||||
return callPythonHelper<SetValueResult>('set_value', {
|
||||
return callHelper<SetValueResult>('set_value', {
|
||||
...appTargetPayload(args.target),
|
||||
index: args.index,
|
||||
value: args.value,
|
||||
@@ -213,7 +215,7 @@ export function createCodexEngine(): CodexComputerEngine {
|
||||
suffix?: string
|
||||
selection?: 'text' | 'cursor_before' | 'cursor_after'
|
||||
}): Promise<void> {
|
||||
await callPythonHelper('select_text', {
|
||||
await callHelper('select_text', {
|
||||
...appTargetPayload(args.target),
|
||||
index: args.index,
|
||||
text: args.text,
|
||||
@@ -228,7 +230,7 @@ export function createCodexEngine(): CodexComputerEngine {
|
||||
index: string
|
||||
action: string
|
||||
}): Promise<void> {
|
||||
await callPythonHelper('perform_secondary_action', {
|
||||
await callHelper('perform_secondary_action', {
|
||||
...appTargetPayload(args.target),
|
||||
index: args.index,
|
||||
action: args.action,
|
||||
@@ -243,7 +245,7 @@ export function createCodexEngine(): CodexComputerEngine {
|
||||
direction: 'up' | 'down' | 'left' | 'right'
|
||||
pages?: number
|
||||
}): Promise<void> {
|
||||
await callPythonHelper('scroll', {
|
||||
await callHelper('scroll', {
|
||||
...appTargetPayload(args.target),
|
||||
index: args.index,
|
||||
x: args.x,
|
||||
@@ -259,7 +261,7 @@ export function createCodexEngine(): CodexComputerEngine {
|
||||
to: { x: number; y: number }
|
||||
button?: CodexMouseButton
|
||||
}): Promise<void> {
|
||||
await callPythonHelper('drag', {
|
||||
await callHelper('drag', {
|
||||
...appTargetPayload(args.target),
|
||||
from: args.from,
|
||||
to: args.to,
|
||||
@@ -272,7 +274,7 @@ export function createCodexEngine(): CodexComputerEngine {
|
||||
key: string
|
||||
systemKeyCombos: boolean
|
||||
}): Promise<void> {
|
||||
await callPythonHelper('press_key', {
|
||||
await callHelper('press_key', {
|
||||
...appTargetPayload(args.target),
|
||||
key: args.key,
|
||||
systemKeyCombos: args.systemKeyCombos,
|
||||
@@ -280,7 +282,7 @@ export function createCodexEngine(): CodexComputerEngine {
|
||||
},
|
||||
|
||||
async typeText(args: { target: AppTarget; text: string }): Promise<void> {
|
||||
await callPythonHelper('type_text', {
|
||||
await callHelper('type_text', {
|
||||
...appTargetPayload(args.target),
|
||||
text: args.text,
|
||||
})
|
||||
@@ -300,10 +302,10 @@ async function typeViaClipboard(text: string): Promise<void> {
|
||||
// Give NSPasteboard a beat before paste, then keep the new contents
|
||||
// resident long enough for Electron/WebView fields to consume them.
|
||||
await sleep(40)
|
||||
await callPythonHelper('paste_clipboard', {})
|
||||
await callHelper('paste_clipboard', {})
|
||||
await sleep(180)
|
||||
} else {
|
||||
await callPythonHelper('key', {
|
||||
await callHelper('key', {
|
||||
keySequence: 'ctrl+v',
|
||||
repeat: 1,
|
||||
})
|
||||
@@ -341,30 +343,30 @@ export function createCliExecutor(_opts: {
|
||||
engine: process.platform === 'darwin' ? createCodexEngine() : undefined,
|
||||
|
||||
async prepareForAction(_allowlistBundleIds, _displayId): Promise<string[]> {
|
||||
return callPythonHelper('prepare_for_action', {})
|
||||
return callHelper('prepare_for_action', {})
|
||||
},
|
||||
|
||||
async previewHideSet(_allowlistBundleIds, _displayId) {
|
||||
return callPythonHelper('preview_hide_set', {})
|
||||
return callHelper('preview_hide_set', {})
|
||||
},
|
||||
|
||||
async getDisplaySize(displayId?: number): Promise<DisplayGeometry> {
|
||||
return normalizeDisplayGeometry(await callPythonHelper('get_display_size', { displayId }))
|
||||
return normalizeDisplayGeometry(await callHelper('get_display_size', { displayId }))
|
||||
},
|
||||
|
||||
async listDisplays(): Promise<DisplayGeometry[]> {
|
||||
const displays = await callPythonHelper<PythonDisplayGeometry[]>('list_displays', {})
|
||||
const displays = await callHelper<PythonDisplayGeometry[]>('list_displays', {})
|
||||
return displays.map(display => normalizeDisplayGeometry(display))
|
||||
},
|
||||
|
||||
async findWindowDisplays(bundleIds: string[]) {
|
||||
return callPythonHelper('find_window_displays', { bundleIds })
|
||||
return callHelper('find_window_displays', { bundleIds })
|
||||
},
|
||||
|
||||
async resolvePrepareCapture(opts): Promise<ResolvePrepareCaptureResult> {
|
||||
const display = await this.getDisplaySize(opts.preferredDisplayId)
|
||||
const [targetW, targetH] = computeTargetDims(display.width, display.height, display.scaleFactor)
|
||||
const result = await callPythonHelper<PythonResolvePrepareCaptureResult>('resolve_prepare_capture', {
|
||||
const result = await callHelper<PythonResolvePrepareCaptureResult>('resolve_prepare_capture', {
|
||||
preferredDisplayId: opts.preferredDisplayId,
|
||||
targetWidth: targetW,
|
||||
targetHeight: targetH,
|
||||
@@ -380,7 +382,7 @@ export function createCliExecutor(_opts: {
|
||||
async screenshot(opts): Promise<ScreenshotResult> {
|
||||
const display = await this.getDisplaySize(opts.displayId)
|
||||
const [targetW, targetH] = computeTargetDims(display.width, display.height, display.scaleFactor)
|
||||
const result = await callPythonHelper<ScreenshotResult>('screenshot', {
|
||||
const result = await callHelper<ScreenshotResult>('screenshot', {
|
||||
displayId: opts.displayId,
|
||||
targetWidth: targetW,
|
||||
targetHeight: targetH,
|
||||
@@ -392,7 +394,7 @@ export function createCliExecutor(_opts: {
|
||||
async zoom(regionLogical, _allowedBundleIds, displayId) {
|
||||
const display = await this.getDisplaySize(displayId)
|
||||
const [outW, outH] = computeTargetDims(regionLogical.w, regionLogical.h, display.scaleFactor)
|
||||
return callPythonHelper('zoom', {
|
||||
return callHelper('zoom', {
|
||||
x: regionLogical.x,
|
||||
y: regionLogical.y,
|
||||
width: regionLogical.w,
|
||||
@@ -403,11 +405,11 @@ export function createCliExecutor(_opts: {
|
||||
},
|
||||
|
||||
async key(keySequence: string, repeat?: number): Promise<void> {
|
||||
await callPythonHelper('key', { keySequence, repeat: repeat ?? 1 })
|
||||
await callHelper('key', { keySequence, repeat: repeat ?? 1 })
|
||||
},
|
||||
|
||||
async holdKey(keyNames: string[], durationMs: number): Promise<void> {
|
||||
await callPythonHelper('hold_key', { keyNames, durationMs })
|
||||
await callHelper('hold_key', { keyNames, durationMs })
|
||||
},
|
||||
|
||||
async type(text: string, opts2: { viaClipboard: boolean }): Promise<void> {
|
||||
@@ -415,61 +417,61 @@ export function createCliExecutor(_opts: {
|
||||
await typeViaClipboard(text)
|
||||
return
|
||||
}
|
||||
await callPythonHelper('type', { text })
|
||||
await callHelper('type', { text })
|
||||
},
|
||||
|
||||
readClipboard,
|
||||
writeClipboard,
|
||||
|
||||
async click(x, y, button, count, modifiers): Promise<void> {
|
||||
await callPythonHelper('click', { x, y, button, count, modifiers })
|
||||
await callHelper('click', { x, y, button, count, modifiers })
|
||||
await sleep(MOVE_SETTLE_MS)
|
||||
},
|
||||
|
||||
async mouseDown(): Promise<void> {
|
||||
await callPythonHelper('mouse_down', {})
|
||||
await callHelper('mouse_down', {})
|
||||
},
|
||||
|
||||
async mouseUp(): Promise<void> {
|
||||
await callPythonHelper('mouse_up', {})
|
||||
await callHelper('mouse_up', {})
|
||||
},
|
||||
|
||||
async getCursorPosition(): Promise<{ x: number; y: number }> {
|
||||
return callPythonHelper('cursor_position', {})
|
||||
return callHelper('cursor_position', {})
|
||||
},
|
||||
|
||||
async drag(from, to): Promise<void> {
|
||||
await callPythonHelper('drag', { from, to })
|
||||
await callHelper('drag', { from, to })
|
||||
await sleep(MOVE_SETTLE_MS)
|
||||
},
|
||||
|
||||
async moveMouse(x, y): Promise<void> {
|
||||
await callPythonHelper('move_mouse', { x, y })
|
||||
await callHelper('move_mouse', { x, y })
|
||||
await sleep(MOVE_SETTLE_MS)
|
||||
},
|
||||
|
||||
async scroll(x, y, dx, dy): Promise<void> {
|
||||
await callPythonHelper('scroll', { x, y, deltaX: dx, deltaY: dy })
|
||||
await callHelper('scroll', { x, y, deltaX: dx, deltaY: dy })
|
||||
},
|
||||
|
||||
async getFrontmostApp(): Promise<FrontmostApp | null> {
|
||||
return callPythonHelper('frontmost_app', {})
|
||||
return callHelper('frontmost_app', {})
|
||||
},
|
||||
|
||||
async appUnderPoint(x, y) {
|
||||
return callPythonHelper('app_under_point', { x, y })
|
||||
return callHelper('app_under_point', { x, y })
|
||||
},
|
||||
|
||||
async listInstalledApps(): Promise<InstalledApp[]> {
|
||||
return callPythonHelper('list_installed_apps', {})
|
||||
return callHelper('list_installed_apps', {})
|
||||
},
|
||||
|
||||
async listRunningApps(): Promise<RunningApp[]> {
|
||||
return callPythonHelper('list_running_apps', {})
|
||||
return callHelper('list_running_apps', {})
|
||||
},
|
||||
|
||||
async openApp(bundleId: string): Promise<void> {
|
||||
await callPythonHelper('open_app', { bundleId })
|
||||
await callHelper('open_app', { bundleId })
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -355,10 +355,65 @@ describe('callHelper platform routing', () => {
|
||||
callDaemon: async () => { used = 'daemon'; return 0 as never },
|
||||
callPy: async () => { used = 'py'; return 0 as never },
|
||||
overlayShow: () => {},
|
||||
showCursorBadge: () => {},
|
||||
})
|
||||
expect(used).toBe('py')
|
||||
})
|
||||
|
||||
test('Windows marks agent activity on an injecting command', async () => {
|
||||
// Windows drives through SendInput, so the user's real cursor moves. The
|
||||
// badge is the only thing telling them the movement is not theirs, which
|
||||
// matters because grabbing the mouse mid-action is what makes the two
|
||||
// input streams interleave.
|
||||
let badges = 0
|
||||
await callHelper('click', { x: 1, y: 2 }, {
|
||||
platform: 'win32',
|
||||
callPy: ok,
|
||||
showCursorBadge: () => { badges += 1 },
|
||||
})
|
||||
expect(badges).toBe(1)
|
||||
})
|
||||
|
||||
test('Windows leaves the badge alone for read-only commands', async () => {
|
||||
// A badge on `screenshot` would claim the agent is holding the mouse
|
||||
// during a turn that never touches it.
|
||||
let badges = 0
|
||||
for (const command of ['screenshot', 'list_displays', 'read_clipboard']) {
|
||||
await callHelper(command, {}, {
|
||||
platform: 'win32',
|
||||
callPy: ok,
|
||||
showCursorBadge: () => { badges += 1 },
|
||||
})
|
||||
}
|
||||
expect(badges).toBe(0)
|
||||
})
|
||||
|
||||
test('Windows never reaches the macOS overlay', async () => {
|
||||
// The two indicators are not interchangeable: overlay_show is a daemon
|
||||
// command and there is no daemon on Windows, so a stray call would be a
|
||||
// hard failure on the mutation path.
|
||||
let overlays = 0
|
||||
await callHelper('click', { x: 1, y: 2 }, {
|
||||
platform: 'win32',
|
||||
callPy: ok,
|
||||
overlayShow: () => { overlays += 1 },
|
||||
showCursorBadge: () => {},
|
||||
})
|
||||
expect(overlays).toBe(0)
|
||||
})
|
||||
|
||||
test('macOS never spawns the Windows badge', async () => {
|
||||
let badges = 0
|
||||
await callHelper('click', { x: 1, y: 2 }, {
|
||||
platform: 'darwin',
|
||||
cuHelperAvailable: () => true,
|
||||
callDaemon: ok,
|
||||
overlayShow: () => {},
|
||||
showCursorBadge: () => { badges += 1 },
|
||||
})
|
||||
expect(badges).toBe(0)
|
||||
})
|
||||
|
||||
test('forwards command + payload to the daemon unchanged', async () => {
|
||||
let seen: { c: string; p: unknown } | undefined
|
||||
await callHelper('type', { text: 'hi' }, {
|
||||
|
||||
@@ -8,6 +8,7 @@ import {
|
||||
overlayShow,
|
||||
shutdownDaemon,
|
||||
} from './cuHelperDaemon.js'
|
||||
import { showCursorBadge } from './winCursorBadge.js'
|
||||
|
||||
// Latches true after we restart the daemon once in response to an Accessibility
|
||||
// `not_trusted` error, so we don't thrash-restart while the helper is genuinely
|
||||
@@ -97,6 +98,14 @@ function overlayTargetPayload(
|
||||
* - Windows → the Python helper (`win_helper.py`); the native engine is
|
||||
* macOS-only.
|
||||
*
|
||||
* The two platforms do NOT offer the same guarantee, and callers should not
|
||||
* assume they do. macOS delivers input per-process and never touches the real
|
||||
* pointer. Windows has no such API: input goes through `SendInput`, so the
|
||||
* agent shares one cursor and one input stream with the user. The Windows
|
||||
* side therefore gets a badge that marks agent activity rather than a virtual
|
||||
* cursor that replaces it, and `win_helper.py` refuses actions it can already
|
||||
* tell will not land.
|
||||
*
|
||||
* `deps` is injectable for unit tests only.
|
||||
*/
|
||||
export async function callHelper<T>(
|
||||
@@ -109,6 +118,7 @@ export async function callHelper<T>(
|
||||
callPy?: HelperFn
|
||||
overlayShow?: (payload: Record<string, unknown>) => void
|
||||
isOverlayShown?: () => boolean
|
||||
showCursorBadge?: () => void
|
||||
callFrontmost?: () => Promise<{ bundleId?: string } | null>
|
||||
shutdownDaemon?: () => void
|
||||
} = {},
|
||||
@@ -119,6 +129,7 @@ export async function callHelper<T>(
|
||||
const viaPython = deps.callPy ?? (callPythonHelper as HelperFn)
|
||||
const showOverlay = deps.overlayShow ?? overlayShow
|
||||
const overlayIsShown = deps.isOverlayShown ?? isOverlayShown
|
||||
const showBadge = deps.showCursorBadge ?? (() => showCursorBadge())
|
||||
const restartDaemon = deps.shutdownDaemon ?? (() => void shutdownDaemon())
|
||||
|
||||
if (platform === 'darwin') {
|
||||
@@ -156,6 +167,12 @@ export async function callHelper<T>(
|
||||
}
|
||||
}
|
||||
|
||||
if (INJECTION_COMMANDS.has(command)) {
|
||||
// Same trigger set as the macOS overlay, so both platforms mark activity
|
||||
// at the same moments. Fire-and-forget: the badge is advisory and must
|
||||
// never sit on the mutation hot path.
|
||||
showBadge()
|
||||
}
|
||||
return viaPython<T>(command, payload)
|
||||
}
|
||||
|
||||
|
||||
@@ -9,7 +9,7 @@ import { createCliExecutor } from './executor.js'
|
||||
import { getChicagoEnabled, getChicagoSubGates } from './gates.js'
|
||||
import { normalizeOsPermissions } from './permissions.js'
|
||||
// Platform-routed helper: macOS → native cu-helper, Windows → Python helper.
|
||||
import { callHelper as callPythonHelper } from './helperBridge.js'
|
||||
import { callHelper } from './helperBridge.js'
|
||||
import { maybeShowNativePermissionCard } from './nativePermissionCard.js'
|
||||
|
||||
class DebugLogger implements Logger {
|
||||
@@ -42,7 +42,7 @@ export function getComputerUseHostAdapter(): ComputerUseHostAdapter {
|
||||
getHideBeforeActionEnabled: () => getChicagoSubGates().hideBeforeAction,
|
||||
}),
|
||||
ensureOsPermissions: async () => {
|
||||
const rawPerms = await callPythonHelper<{ accessibility: boolean; screenRecording: boolean | null }>('check_permissions', {})
|
||||
const rawPerms = await callHelper<{ accessibility: boolean; screenRecording: boolean | null }>('check_permissions', {})
|
||||
const perms = normalizeOsPermissions(rawPerms)
|
||||
if (perms.granted) return { granted: true as const }
|
||||
// Missing a TCC grant → pop the native, guided permission card (macOS).
|
||||
|
||||
@@ -13,7 +13,13 @@ const projectRoot = path.resolve(__dirname, '../../..')
|
||||
|
||||
// All runtime state lives in ~/.claude/.runtime — writable in both dev and
|
||||
// bundled (Tauri app) modes. The setup API (or ensureRuntimeFiles below)
|
||||
// populates requirements.txt and mac_helper.py here.
|
||||
// populates requirements-win.txt and win_helper.py here.
|
||||
//
|
||||
// This bridge is Windows-only. macOS routes every command to the signed native
|
||||
// `cu-helper` daemon and `helperBridge` refuses to fall back, so the old
|
||||
// `mac_helper.py` was unreachable and has been deleted along with its
|
||||
// pyobjc requirements file. Keeping a dead darwin branch here invited the
|
||||
// reading that Python is still a supported macOS path — it is not.
|
||||
const runtimeStateRoot = path.join(getClaudeConfigHomeDir(), '.runtime')
|
||||
const venvRoot = path.join(runtimeStateRoot, 'venv')
|
||||
const installStampPath = path.join(runtimeStateRoot, 'requirements.sha256')
|
||||
@@ -22,8 +28,12 @@ const isWindows = process.platform === 'win32'
|
||||
|
||||
// Always read from ~/.claude/.runtime/ — works in both dev and bundled mode.
|
||||
const requirementsPath = path.join(runtimeStateRoot, 'requirements.txt')
|
||||
const helperFileName = isWindows ? 'win_helper.py' : 'mac_helper.py'
|
||||
const helperFileName = 'win_helper.py'
|
||||
const helperPath = path.join(runtimeStateRoot, helperFileName)
|
||||
// Runs as its own process (the helper is a stateless one-shot CLI and cannot
|
||||
// own a window across actions), so it ships as a separate file.
|
||||
const cursorBadgeFileName = 'win_cursor_badge.py'
|
||||
const cursorBadgePath = path.join(runtimeStateRoot, cursorBadgeFileName)
|
||||
|
||||
let bootstrapPromise: Promise<void> | undefined
|
||||
|
||||
@@ -98,8 +108,7 @@ async function getVenvCreationPythonCommand(): Promise<string> {
|
||||
async function ensureRuntimeFiles(): Promise<void> {
|
||||
await mkdir(runtimeStateRoot, { recursive: true })
|
||||
|
||||
const devReqFile = isWindows ? 'requirements-win.txt' : 'requirements.txt'
|
||||
const devRequirements = path.join(projectRoot, 'runtime', devReqFile)
|
||||
const devRequirements = path.join(projectRoot, 'runtime', 'requirements-win.txt')
|
||||
const devHelper = path.join(projectRoot, 'runtime', helperFileName)
|
||||
|
||||
// Always sync from dev runtime/ so source changes are reflected immediately.
|
||||
@@ -112,12 +121,17 @@ async function ensureRuntimeFiles(): Promise<void> {
|
||||
if (await pathExists(devHelper)) {
|
||||
await writeFile(helperPath, await readFile(devHelper, 'utf8'), 'utf8')
|
||||
}
|
||||
|
||||
const devBadge = path.join(projectRoot, 'runtime', cursorBadgeFileName)
|
||||
if (await pathExists(devBadge)) {
|
||||
await writeFile(cursorBadgePath, await readFile(devBadge, 'utf8'), 'utf8')
|
||||
}
|
||||
}
|
||||
|
||||
export async function ensureBootstrapped(): Promise<void> {
|
||||
if (bootstrapPromise) return bootstrapPromise
|
||||
bootstrapPromise = (async () => {
|
||||
// Extract runtime files (requirements.txt, mac_helper.py) to state dir
|
||||
// Extract runtime files (requirements, helper, badge) to state dir
|
||||
await ensureRuntimeFiles()
|
||||
|
||||
if (!(await pathExists(pythonBinPath()))) {
|
||||
@@ -168,7 +182,7 @@ export async function callPythonHelper<T>(command: string, payload: Record<strin
|
||||
throw new Error(stderr || `Python helper ${command} failed with code ${code}`)
|
||||
}
|
||||
|
||||
let parsed: { ok: boolean; result?: T; error?: { message?: string } }
|
||||
let parsed: { ok: boolean; result?: T; error?: { code?: string; message?: string } }
|
||||
try {
|
||||
parsed = JSON.parse(stdout)
|
||||
} catch {
|
||||
@@ -176,7 +190,14 @@ export async function callPythonHelper<T>(command: string, payload: Record<strin
|
||||
}
|
||||
|
||||
if (!parsed.ok) {
|
||||
throw new Error(parsed.error?.message || `Python helper ${command} failed`)
|
||||
// Prefix the machine-readable code so callers can branch on it. The helper
|
||||
// reports refusals such as `user_interference` and `target_window_offscreen`
|
||||
// this way, and a bare message would flatten them into indistinguishable
|
||||
// prose — the model would then retry an action that was deliberately
|
||||
// refused. Mirrors how the macOS daemon surfaces `CUError.code`.
|
||||
const code = parsed.error?.code
|
||||
const message = parsed.error?.message || `Python helper ${command} failed`
|
||||
throw new Error(code && code !== 'runtime_error' ? `${code}: ${message}` : message)
|
||||
}
|
||||
|
||||
return parsed.result as T
|
||||
@@ -185,3 +206,8 @@ export async function callPythonHelper<T>(command: string, payload: Record<strin
|
||||
export function getRuntimePaths(): { projectRoot: string; runtimeStateRoot: string; venvRoot: string } {
|
||||
return { projectRoot, runtimeStateRoot, venvRoot }
|
||||
}
|
||||
|
||||
/** Interpreter + script path for the Windows agent-activity badge. */
|
||||
export function getCursorBadgeCommand(): { python: string; script: string } {
|
||||
return { python: pythonBinPath(), script: cursorBadgePath }
|
||||
}
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
import { spawn, type ChildProcess } from 'node:child_process'
|
||||
import { logForDebugging } from '../debug.js'
|
||||
import { getCursorBadgeCommand } from './pythonBridge.js'
|
||||
|
||||
/**
|
||||
* The Windows agent-activity badge: a click-through marker that rides the real
|
||||
* cursor while the agent is driving.
|
||||
*
|
||||
* WHY WINDOWS NEEDS THIS AND macOS DOES NOT
|
||||
* -----------------------------------------
|
||||
* On macOS the helper delivers input with `CGEvent.postToPid`, so the real
|
||||
* pointer never moves and the drawn cursor is the only one there is — a
|
||||
* *replacement*, and unambiguous by construction.
|
||||
*
|
||||
* Windows has no per-process delivery. Input goes through `SendInput`, which
|
||||
* warps the one real cursor the user's hand is also on. Drawing a second fake
|
||||
* pointer would make that worse, not better: two pointers, one of which is
|
||||
* lying about where the click will land.
|
||||
*
|
||||
* So this badge annotates rather than replaces. It answers the one question
|
||||
* the user cannot otherwise answer — "is the mouse moving because of me, or
|
||||
* because of the agent?" — which on Windows has real stakes, since grabbing
|
||||
* the mouse mid-action is what causes the two input streams to interleave.
|
||||
*
|
||||
* Lifecycle mirrors the macOS overlay: shown while a turn is driving, hidden
|
||||
* at turn end. It is advisory, so every failure here is logged and swallowed —
|
||||
* a badge that cannot start must never be the reason an action fails.
|
||||
*/
|
||||
|
||||
let badgeProcess: ChildProcess | undefined
|
||||
|
||||
function isRunning(): boolean {
|
||||
return badgeProcess !== undefined && badgeProcess.exitCode === null && !badgeProcess.killed
|
||||
}
|
||||
|
||||
/** Start the badge if it isn't already up. Idempotent, never throws. */
|
||||
export function showCursorBadge(label = 'Claude'): void {
|
||||
if (process.platform !== 'win32') return
|
||||
if (isRunning()) return
|
||||
|
||||
try {
|
||||
const { python, script } = getCursorBadgeCommand()
|
||||
const child = spawn(python, [script, '--label', label], {
|
||||
// 'ignore' for stderr would discard the reason it died; 'pipe' on stdin
|
||||
// is load-bearing — the badge exits when stdin closes, which is what
|
||||
// ties its lifetime to ours even if we are killed rather than exiting.
|
||||
stdio: ['pipe', 'ignore', 'pipe'],
|
||||
windowsHide: true,
|
||||
detached: false,
|
||||
})
|
||||
|
||||
child.on('error', err => {
|
||||
logForDebugging(`cursor badge failed to start: ${String(err)}`, { level: 'debug' })
|
||||
if (badgeProcess === child) badgeProcess = undefined
|
||||
})
|
||||
child.on('exit', () => {
|
||||
if (badgeProcess === child) badgeProcess = undefined
|
||||
})
|
||||
child.stderr?.on('data', (chunk: Buffer) => {
|
||||
logForDebugging(`cursor badge: ${chunk.toString().trim()}`, { level: 'debug' })
|
||||
})
|
||||
|
||||
badgeProcess = child
|
||||
} catch (err) {
|
||||
logForDebugging(`cursor badge spawn threw: ${String(err)}`, { level: 'debug' })
|
||||
badgeProcess = undefined
|
||||
}
|
||||
}
|
||||
|
||||
/** Take the badge down. Idempotent, never throws. */
|
||||
export function hideCursorBadge(): void {
|
||||
const child = badgeProcess
|
||||
badgeProcess = undefined
|
||||
if (!child) return
|
||||
|
||||
try {
|
||||
// Closing stdin is the graceful path — the badge's reader hits EOF and
|
||||
// unwinds its own message loop. kill() is the backstop for a process that
|
||||
// is wedged before it ever got to that read.
|
||||
child.stdin?.end()
|
||||
child.kill()
|
||||
} catch (err) {
|
||||
logForDebugging(`cursor badge shutdown failed: ${String(err)}`, { level: 'debug' })
|
||||
}
|
||||
}
|
||||
|
||||
/** Test hook: forget any tracked process without signalling it. */
|
||||
export function __resetCursorBadgeState(): void {
|
||||
badgeProcess = undefined
|
||||
}
|
||||
|
||||
/** Test hook: whether a badge process is currently tracked as running. */
|
||||
export function __cursorBadgeIsRunning(): boolean {
|
||||
return isRunning()
|
||||
}
|
||||
Reference in New Issue
Block a user