feat: merge AT-SPI targeting and delay post-action screenshots

This commit is contained in:
Zoe
2026-09-19 18:39:33 -05:00
9 changed files with 294 additions and 7 deletions
+116
View File
@@ -0,0 +1,116 @@
"""Bounded AT-SPI collection runs in a disposable process to contain D-Bus stalls."""
import json
import subprocess
import sys
from pathlib import Path
def collect(width, height):
try:
result = subprocess.run(
[sys.executable, str(Path(__file__).resolve()), str(width), str(height)],
capture_output=True, text=True, timeout=3, check=True,
)
return json.loads(result.stdout)
except (subprocess.SubprocessError, ValueError) as error:
return {"targets": [], "warning": f"Accessibility unavailable ({type(error).__name__}); use pixel clicks."}
def resolve_target(saved, current, observation_id, target_id, width, height):
if saved.get("observationId") != observation_id or saved.get("size") != [width, height]:
raise ValueError("Stale target observation; capture again")
target = next((item for item in saved["targets"] if item["id"] == target_id), None)
if target is None:
raise ValueError("Unknown target; capture again")
matching = next((item for item in current["targets"] if item["path"] == target["path"]), None)
if matching is None or any(matching[key] != target[key] for key in ("name", "role", "bounds", "app")):
raise ValueError("Target changed or disappeared; capture again")
x, y, w, h = target["bounds"]
return x + w // 2, y + h // 2
def badge_position(bounds, badge_width, badge_height, image_width, image_height):
x, y, w, h = bounds
# Keep row labels within their own row, never above it in the previous item.
left = min(x + 2, max(0, image_width - badge_width))
top = min(y + max(0, (h - badge_height) // 2), max(0, image_height - badge_height))
return left, top
def annotate(image, targets):
from PIL import ImageDraw, ImageFont
marked = image.copy()
draw = ImageDraw.Draw(marked)
font = ImageFont.load_default(size=14)
for target in targets:
x, y, w, h = target["bounds"]
draw.rectangle((x, y, x + w - 1, y + h - 1), outline="#ff00cc", width=2)
# Draw badges last so another element's outline cannot cross out a number.
for target in targets:
label = str(target["id"])
box = draw.textbbox((0, 0), label, font=font)
label_width, label_height = box[2] - box[0] + 6, box[3] - box[1] + 6
left, top = badge_position(target["bounds"], label_width, label_height, image.width, image.height)
draw.rectangle((left, top, left + label_width - 1, top + label_height - 1), fill="#ffff00", outline="black")
draw.text((left + 3 - box[0], top + 3 - box[1]), label, fill="black", font=font)
return marked
def scan(width, height):
import pyatspi
desktop = pyatspi.Registry.getDesktop(0)
targets = []
visited = 0
truncated = False
def walk(node, path, app, depth):
nonlocal visited, truncated
visited += 1
if visited > 1500 or len(targets) >= 100 or depth > 30:
truncated = True
return
try:
state = node.getState()
if not state.contains(pyatspi.STATE_SHOWING):
return
role = node.getRoleName()
actionable = state.contains(pyatspi.STATE_FOCUSABLE)
try:
actionable = actionable or node.queryAction().nActions > 0
except NotImplementedError:
pass
if actionable and state.contains(pyatspi.STATE_ENABLED):
rect = node.queryComponent().getExtents(pyatspi.DESKTOP_COORDS)
x, y = max(0, rect.x), max(0, rect.y)
right, bottom = min(width, rect.x + rect.width), min(height, rect.y + rect.height)
if right > x and bottom > y:
targets.append({"id": str(len(targets) + 1), "path": path, "app": app,
"name": (node.name or "")[:200], "role": role,
"bounds": [x, y, right - x, bottom - y]})
for index in range(min(node.childCount, 1500)):
if visited >= 1500 or len(targets) >= 100:
truncated = True
break
walk(node[index], path + [index], app, depth + 1)
except Exception:
# Individual applications can disappear or expose incomplete interfaces.
return
# Restrict marks to active windows, avoiding targets in covered background windows.
for app_index in range(min(desktop.childCount, 100)):
try:
app = desktop[app_index]
for window_index in range(min(app.childCount, 100)):
window = app[window_index]
if window.getState().contains(pyatspi.STATE_ACTIVE):
walk(window, [app_index, window_index], app.name or "", 0)
except Exception:
continue
warning = "Accessibility target list truncated." if truncated else None
if not targets:
warning = "No actionable targets in an active accessible window; use pixel clicks."
return {"targets": targets, "warning": warning}
if __name__ == "__main__":
print(json.dumps(scan(int(sys.argv[1]), int(sys.argv[2]))))
+28 -2
View File
@@ -9,6 +9,7 @@ import sys
import uuid
import fcntl
import tempfile
import accessibility
STATE = Path.home() / ".local/state/desktop-harness"
@@ -49,6 +50,13 @@ def validate(request, width, height):
click_pixels(action, width, height, space)
if action.get("button", "left") not in ("left", "middle", "right"):
raise ValueError("Unknown mouse button")
elif kind == "click_target":
if not isinstance(action.get("target"), str) or not action["target"].isdigit():
raise ValueError("Target must be a numbered label string")
if not isinstance(action.get("observationId"), str):
raise ValueError("click_target requires observationId")
if action is not actions[0] or sum(a.get("type") == "click_target" for a in actions) > 1:
raise ValueError("click_target must be the first and only target click; capture again before another")
elif kind == "scroll":
if action.get("direction") not in ("up", "down"):
raise ValueError("Unknown scroll direction")
@@ -143,6 +151,11 @@ def main():
x, y = click_pixels(action, width, height, space)
run("xdotool", "mousemove", "--sync", str(x), str(y), "click", button)
clicks.append({"supplied": [action["x"], action["y"]], "pixels": [x, y]})
elif kind == "click_target":
saved = json.loads((STATE / "targets.json").read_text())
current = accessibility.collect(width, height)
x, y = accessibility.resolve_target(saved, current, action["observationId"], action["target"], width, height)
run("xdotool", "mousemove", "--sync", str(x), str(y), "click", "1")
elif kind == "scroll":
run("xdotool", "click", "--repeat", str(action["steps"]), "--delay", "50", "4" if action["direction"] == "up" else "5")
elif kind == "keys":
@@ -156,11 +169,24 @@ def main():
completed += 1
except Exception as error:
raise RuntimeError(f"Action {completed} failed after {completed} completed actions; partial effects possible: {error}") from error
if actions:
# Input delivery can finish before the application has repainted.
time.sleep(0.5)
screenshot = ImageGrab.grab(xdisplay=os.environ["DISPLAY"])
accessible = accessibility.collect(screenshot.width, screenshot.height)
observation_id = str(uuid.uuid4())
(STATE / "targets.json").write_text(json.dumps({
"observationId": observation_id, "size": list(screenshot.size), "targets": accessible["targets"],
}))
raw_buffer = io.BytesIO()
screenshot.save(raw_buffer, format="PNG")
buffer = io.BytesIO()
screenshot.save(buffer, format="PNG")
accessibility.annotate(screenshot, accessible["targets"]).save(buffer, format="PNG")
print(json.dumps({
"observationId": str(uuid.uuid4()),
"observationId": observation_id,
"targets": accessible["targets"],
"accessibilityWarning": accessible["warning"],
"rawImage": base64.b64encode(raw_buffer.getvalue()).decode(),
"capturedAt": datetime.datetime.now(datetime.timezone.utc).isoformat(),
"width": screenshot.width,
"height": screenshot.height,