feat: AT-SPI interactions

This commit is contained in:
Zoe
2026-09-19 17:59:06 -05:00
parent c49f228847
commit e73af2bcf9
9 changed files with 237 additions and 6 deletions
+106
View File
@@ -0,0 +1,106 @@
"""Bounded AT-SPI collection runs in a disposable process to contain D-Bus stalls."""
import json
import subprocess
import sys
from pathlib import Path
def collect(width, height):
try:
result = subprocess.run(
[sys.executable, str(Path(__file__).resolve()), str(width), str(height)],
capture_output=True, text=True, timeout=3, check=True,
)
return json.loads(result.stdout)
except (subprocess.SubprocessError, ValueError) as error:
return {"targets": [], "warning": f"Accessibility unavailable ({type(error).__name__}); use pixel clicks."}
def resolve_target(saved, current, observation_id, target_id, width, height):
if saved.get("observationId") != observation_id or saved.get("size") != [width, height]:
raise ValueError("Stale target observation; capture again")
target = next((item for item in saved["targets"] if item["id"] == target_id), None)
if target is None:
raise ValueError("Unknown target; capture again")
matching = next((item for item in current["targets"] if item["path"] == target["path"]), None)
if matching is None or any(matching[key] != target[key] for key in ("name", "role", "bounds", "app")):
raise ValueError("Target changed or disappeared; capture again")
x, y, w, h = target["bounds"]
return x + w // 2, y + h // 2
def annotate(image, targets):
from PIL import ImageDraw
marked = image.copy()
draw = ImageDraw.Draw(marked)
for target in targets:
x, y, w, h = target["bounds"]
draw.rectangle((x, y, x + w - 1, y + h - 1), outline="#ff00cc", width=2)
label = str(target["id"])
box = draw.textbbox((0, 0), label)
label_width, label_height = box[2] + 6, box[3] - box[1] + 6
left = min(x, max(0, image.width - label_width))
top = max(0, y - label_height)
draw.rectangle((left, top, left + label_width, top + label_height), fill="#ffff00")
draw.text((left + 3, top + 3 - box[1]), label, fill="black")
return marked
def scan(width, height):
import pyatspi
desktop = pyatspi.Registry.getDesktop(0)
targets = []
visited = 0
truncated = False
def walk(node, path, app, depth):
nonlocal visited, truncated
visited += 1
if visited > 1500 or len(targets) >= 100 or depth > 30:
truncated = True
return
try:
state = node.getState()
if not state.contains(pyatspi.STATE_SHOWING):
return
role = node.getRoleName()
actionable = state.contains(pyatspi.STATE_FOCUSABLE)
try:
actionable = actionable or node.queryAction().nActions > 0
except NotImplementedError:
pass
if actionable and state.contains(pyatspi.STATE_ENABLED):
rect = node.queryComponent().getExtents(pyatspi.DESKTOP_COORDS)
x, y = max(0, rect.x), max(0, rect.y)
right, bottom = min(width, rect.x + rect.width), min(height, rect.y + rect.height)
if right > x and bottom > y:
targets.append({"id": str(len(targets) + 1), "path": path, "app": app,
"name": (node.name or "")[:200], "role": role,
"bounds": [x, y, right - x, bottom - y]})
for index in range(min(node.childCount, 1500)):
if visited >= 1500 or len(targets) >= 100:
truncated = True
break
walk(node[index], path + [index], app, depth + 1)
except Exception:
# Individual applications can disappear or expose incomplete interfaces.
return
# Restrict marks to active windows, avoiding targets in covered background windows.
for app_index in range(min(desktop.childCount, 100)):
try:
app = desktop[app_index]
for window_index in range(min(app.childCount, 100)):
window = app[window_index]
if window.getState().contains(pyatspi.STATE_ACTIVE):
walk(window, [app_index, window_index], app.name or "", 0)
except Exception:
continue
warning = "Accessibility target list truncated." if truncated else None
if not targets:
warning = "No actionable targets in an active accessible window; use pixel clicks."
return {"targets": targets, "warning": warning}
if __name__ == "__main__":
print(json.dumps(scan(int(sys.argv[1]), int(sys.argv[2]))))
+25 -2
View File
@@ -9,6 +9,7 @@ import sys
import uuid
import fcntl
import tempfile
import accessibility
STATE = Path.home() / ".local/state/desktop-harness"
@@ -39,6 +40,13 @@ def validate(request, width, height):
integer(action.get("y"), 0, height - 1)
if action.get("button", "left") not in ("left", "middle", "right"):
raise ValueError("Unknown mouse button")
elif kind == "click_target":
if not isinstance(action.get("target"), str) or not action["target"].isdigit():
raise ValueError("Target must be a numbered label string")
if not isinstance(action.get("observationId"), str):
raise ValueError("click_target requires observationId")
if action is not actions[0] or sum(a.get("type") == "click_target" for a in actions) > 1:
raise ValueError("click_target must be the first and only target click; capture again before another")
elif kind == "scroll":
if action.get("direction") not in ("up", "down"):
raise ValueError("Unknown scroll direction")
@@ -129,6 +137,11 @@ def main():
if kind == "click":
button = {"left": "1", "middle": "2", "right": "3"}[action.get("button", "left")]
run("xdotool", "mousemove", "--sync", str(action["x"]), str(action["y"]), "click", button)
elif kind == "click_target":
saved = json.loads((STATE / "targets.json").read_text())
current = accessibility.collect(width, height)
x, y = accessibility.resolve_target(saved, current, action["observationId"], action["target"], width, height)
run("xdotool", "mousemove", "--sync", str(x), str(y), "click", "1")
elif kind == "scroll":
run("xdotool", "click", "--repeat", str(action["steps"]), "--delay", "50", "4" if action["direction"] == "up" else "5")
elif kind == "keys":
@@ -143,10 +156,20 @@ def main():
except Exception as error:
raise RuntimeError(f"Action {completed} failed after {completed} completed actions; partial effects possible: {error}") from error
screenshot = ImageGrab.grab(xdisplay=os.environ["DISPLAY"])
accessible = accessibility.collect(screenshot.width, screenshot.height)
observation_id = str(uuid.uuid4())
(STATE / "targets.json").write_text(json.dumps({
"observationId": observation_id, "size": list(screenshot.size), "targets": accessible["targets"],
}))
raw_buffer = io.BytesIO()
screenshot.save(raw_buffer, format="PNG")
buffer = io.BytesIO()
screenshot.save(buffer, format="PNG")
accessibility.annotate(screenshot, accessible["targets"]).save(buffer, format="PNG")
print(json.dumps({
"observationId": str(uuid.uuid4()),
"observationId": observation_id,
"targets": accessible["targets"],
"accessibilityWarning": accessible["warning"],
"rawImage": base64.b64encode(raw_buffer.getvalue()).decode(),
"capturedAt": datetime.datetime.now(datetime.timezone.utc).isoformat(),
"width": screenshot.width,
"height": screenshot.height,