Files
DarkRoom/tools/manual/drive.py
T
dtourolle a6ea6ba83f Let the manual's scripts find a control by its name
Every scene in tools/manual aimed at window pixels written in by hand, so
a panel that gained a row moved every slider under it and the recording
went on dragging where the slider used to be. The develop column has
already moved that way (Compose now sits above Adjust), and nothing said.

A build with the `automation` feature listens on the Unix socket named
by DR_AUTOMATION and answers where an element is: by its accessible
label, the name a screen reader reads, or by its markup id for the few
things that are not controls (the canvas, the crop rectangle). It uses
Slint's element queries, which need the compiler's debug tables, so the
feature also turns those on in build.rs. It only answers questions; the
input is still xdotool's real pointer. No default build has the feature,
and one that has it listens only when the variable is set.

drive.py gains click-on, drag-on, hold-on, wait-for, wait-gone, labels
and ids. The grid's cells are now named by their file, each rating star
by its value, the sidebar's + as "New collection", and the Adjust
heading's reset as "Reset all adjustments" - controls a screen reader
could not reach before either.
2026-09-25 07:26:36 -04:00

423 lines
13 KiB
Python
Executable File
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""Drive the desktop build from outside, for the manual's screenshots.
drive.py launch [args...] start the app on the private X server
drive.py stop
drive.py shot OUT.png
drive.py click X Y [button]
drive.py move X Y
drive.py drag X1 Y1 X2 Y2 [steps]
drive.py type TEXT
drive.py key KEYSYM...
drive.py rec OUT.mp4 / cut start and stop a recording of the window
drive.py where the pointer, in window coordinates
By name — the accessible label a screen reader reads, answered by the app
itself when it is built with `--features automation` (see record.sh and
ui/dr-ui/src/automation.rs):
drive.py labels [TEXT] every labelled element on screen (containing TEXT)
drive.py ids [TEXT] every element with a markup id (containing TEXT)
drive.py locate NAME what NAME finds, as JSON
drive.py click-on NAME [button]
drive.py drag-on NAME DX DY [steps] from the element's centre
drive.py hold-on NAME SECONDS
drive.py wait-for NAME [timeout] until it is on screen
drive.py wait-gone NAME [timeout] until it is not
A NAME is an accessible label, exactly: `Exposure`, `Select`, `_MG_8393`.
Suffixes narrow it — `@ROLE` takes only elements of that role (`Button`,
`Slider`, `ListItem`, `Text`...), `#N` the N-th match in tree order from 0,
`#-1` the last (a sheet is drawn after what it covers). A leading `id:`
names an element by its id in the markup or its component type instead
(`id:canvas-image`, `id:move-area`, `id:Timeline`), for the few things a
scene aims at that are not controls. Only what Slint draws is answered:
nothing hidden, nothing scrolled out of its list.
The input itself is still xdotool — a real pointer, pressed and held — so a
scene exercises exactly what a hand would. The hook only says where to aim.
Everything is in *window* pixels, at scale 1, with the window at the origin
of a 1920×1200 Xvfb on `DR_DISPLAY` (`:7`). The app gets its own XDG
profile under `DR_HOME` (`/var/tmp/dr-manual`), so nothing here touches the
library or settings of whoever is logged in; `DR_BIN` names the binary.
Two things that cost an afternoon, kept here so they cost nobody else one:
a Slint `TouchArea` wants a *held* press (`mousedown`, a beat, `mouseup`) —
xdotool's `click` is sometimes dropped; and under XWayland at scale 2 a
pointer warp lands at twice the coordinate asked for, which is why this runs
on Xvfb rather than the desktop.
"""
import json
import os
import re
import socket
import subprocess
import sys
import time
HOME = os.environ.get('DR_HOME', '/var/tmp/dr-manual')
DISPLAY = os.environ.get('DR_DISPLAY', ':7')
BIN = os.environ.get('DR_BIN', 'target/release/darkroom-desktop')
SOCKET = os.environ.get('DR_AUTOMATION', f'{HOME}/automation.sock')
def app_env(xdg=None):
"""The app's environment: its own XDG tree (`DR_HOME/xdg` unless told
otherwise), X11 at scale 1, and the automation socket."""
xdg = xdg or f'{HOME}/xdg'
env = dict(
os.environ,
XDG_CONFIG_HOME=f'{xdg}/config',
XDG_DATA_HOME=f'{xdg}/data',
XDG_STATE_HOME=f'{xdg}/state',
RUST_LOG='info',
WINIT_X11_SCALE_FACTOR='1',
DISPLAY=DISPLAY,
DR_AUTOMATION=SOCKET,
)
env.pop('WAYLAND_DISPLAY', None)
return env
os.environ['DISPLAY'] = DISPLAY
def x(*args, check=True):
return subprocess.run(
['xdotool', *map(str, args)], capture_output=True, text=True, check=check
).stdout.strip()
def win():
return open(f'{HOME}/app.win').read().strip()
def geometry():
out = x('getwindowgeometry', win())
pos = out.split('Position: ')[1].split(' ')[0]
px, py = map(int, pos.split(','))
return px, py
def launch(args, xdg=None):
os.makedirs(HOME, exist_ok=True)
log = open(f'{HOME}/app.log', 'w')
p = subprocess.Popen([BIN, *args], env=app_env(xdg), stdout=log, stderr=subprocess.STDOUT)
open(f'{HOME}/app.pid', 'w').write(str(p.pid))
w = ''
for _ in range(120):
w = x('search', '--pid', p.pid, '--name', 'DarkRoom', check=False).split('\n')[-1]
if w:
break
time.sleep(0.5)
time.sleep(1.5)
W, H = os.environ.get('WIDTH', '1600'), os.environ.get('HEIGHT', '1100')
x('windowsize', '--sync', w, W, H)
time.sleep(0.5)
x('windowmove', '--sync', w, 0, 0)
x('windowfocus', '--sync', w, check=False)
open(f'{HOME}/app.win', 'w').write(w)
print(f'pid {p.pid} win {w}')
def stop():
"""SIGTERM to the app this profile launched, and wait for it to go.
Not `windowclose`: winit panics on that under Xvfb."""
try:
pid = int(open(f'{HOME}/app.pid').read())
os.kill(pid, 15)
except (OSError, ValueError) as e:
print(e)
return
for _ in range(100):
try:
os.kill(pid, 0)
time.sleep(0.1)
except OSError:
break
time.sleep(0.5)
def shot(out):
subprocess.run(['import', '-window', win(), out], check=True)
def move(px, py):
ox, oy = geometry()
x('mousemove', '--sync', ox + int(px), oy + int(py))
def click(px, py, button=1):
x('windowfocus', '--sync', win(), check=False)
move(px, py)
time.sleep(0.15)
x('mousedown', button)
time.sleep(0.12)
x('mouseup', button)
def drag(x1, y1, x2, y2, steps=20):
x('windowfocus', '--sync', win(), check=False)
move(x1, y1)
time.sleep(0.15)
x('mousedown', 1)
time.sleep(0.15)
for i in range(1, int(steps) + 1):
t = i / int(steps)
move(int(x1) + (int(x2) - int(x1)) * t, int(y1) + (int(y2) - int(y1)) * t)
time.sleep(0.03)
time.sleep(0.15)
x('mouseup', 1)
def drag_path(points, steps=20):
"""A drag through several points: out of the grid sideways first, then
to the row. A diagonal with much vertical in it is taken by the grid's
Flickable as a scroll before the DragArea can claim it."""
x('windowfocus', '--sync', win(), check=False)
(x1, y1), rest = points[0], points[1:]
move(x1, y1)
time.sleep(0.15)
x('mousedown', 1)
time.sleep(0.15)
for x2, y2 in rest:
for i in range(1, int(steps) + 1):
t = i / int(steps)
move(int(x1 + (x2 - x1) * t), int(y1 + (y2 - y1) * t))
time.sleep(0.03)
x1, y1 = x2, y2
time.sleep(0.4)
x('mouseup', 1)
# --- by name ----------------------------------------------------------------
class NotFound(Exception):
pass
def ask(request):
"""One request to the app's automation hook, and its JSON answer."""
with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as s:
s.settimeout(15)
s.connect(SOCKET)
s.sendall(request.encode() + b'\n')
buf = b''
while not buf.endswith(b'\n'):
chunk = s.recv(65536)
if not chunk:
break
buf += chunk
answer = json.loads(buf)
if isinstance(answer, dict) and 'error' in answer:
raise RuntimeError(f'{request}: {answer["error"]}')
return answer
def wait_ready(timeout=120):
"""Until the hook answers: the app is up and its event loop running."""
t0 = time.time()
while True:
try:
if ask('ping') == 'ok':
return
except OSError:
pass
if time.time() - t0 > timeout:
raise TimeoutError('the automation hook never answered; was the app built '
'with --features automation?')
time.sleep(0.5)
def parse_name(name):
"""`LABEL`, `id:ID`, each with optional `@ROLE` and `#N`."""
m = re.match(r'^(.*?)(?:@(\w+))?(?:#(-?\d+))?$', name, re.S)
what, role, n = m.group(1), m.group(2), m.group(3)
return what, role, int(n) if n is not None else 0
def matches(name, within=None):
"""Every on-screen element `name` names, in tree order. `within` is an
optional (x0, y0, x1, y1) the element's centre must fall inside."""
what, role, _ = parse_name(name)
if what.startswith('id:'):
hits = ask(f'locate-id {what[3:]}')
else:
hits = ask(f'locate {what}')
hits = [e for e in hits if e['w'] > 0 and e['h'] > 0 and e['opacity'] > 0]
if role:
hits = [e for e in hits if (e['role'] or '').lower() == role.lower()]
if within:
x0, y0, x1, y1 = within
hits = [e for e in hits if x0 <= e['x'] + e['w'] / 2 < x1 and y0 <= e['y'] + e['h'] / 2 < y1]
return hits
def find(name, within=None):
"""The element `name` names, or NotFound."""
index = parse_name(name)[2]
hits = matches(name, within)
if not -len(hits) <= index < len(hits):
raise NotFound(f'{name!r}: {len(hits)} on screen')
return hits[index]
def rect(name, within=None):
"""(x0, y0, x1, y1) of `name`."""
e = find(name, within)
return e['x'], e['y'], e['x'] + e['w'], e['y'] + e['h']
def point(name, fx=0.5, fy=0.5, within=None):
"""A point inside `name`, as fractions of its width and height."""
e = find(name, within)
return int(e['x'] + e['w'] * float(fx)), int(e['y'] + e['h'] * float(fy))
def centre(name, within=None):
return point(name, within=within)
def present(name, within=None):
try:
find(name, within)
return True
except NotFound:
return False
def wait_for(name, timeout=30, within=None):
t0 = time.time()
while True:
try:
return find(name, within)
except (NotFound, OSError):
if time.time() - t0 > float(timeout):
raise
time.sleep(0.25)
def wait_gone(name, timeout=30):
t0 = time.time()
while time.time() - t0 < float(timeout):
if not present(name):
return
time.sleep(0.25)
raise TimeoutError(f'{name!r} still on screen after {timeout}s')
def click_on(name, button=1, within=None):
click(*centre(name, within), button)
def drag_on(name, dx, dy, steps=20, within=None, fx=0.5, fy=0.5):
"""Drag from a point in `name` (its centre by default) by (dx, dy)."""
cx, cy = point(name, fx, fy, within)
drag(cx, cy, cx + int(dx), cy + int(dy), steps)
def hold_on(name, seconds, within=None):
x('windowfocus', '--sync', win(), check=False)
move(*centre(name, within))
time.sleep(0.15)
x('mousedown', 1)
time.sleep(float(seconds))
x('mouseup', 1)
def photo(fx=0.5, fy=0.5):
"""A point on the photograph in develop, as fractions of the picture as
drawn — `contain`-fitted into the canvas, so not of the canvas itself."""
cx0, cy0, cx1, cy1 = rect('id:canvas-image')
img = ask('canvas')
cw, ch = cx1 - cx0, cy1 - cy0
if not img['w'] or not img['h']:
return int(cx0 + cw * fx), int(cy0 + ch * fy)
s = min(cw / img['w'], ch / img['h'])
w, h = img['w'] * s, img['h'] * s
return int(cx0 + (cw - w) / 2 + w * fx), int(cy0 + (ch - h) / 2 + h * fy)
def rec_start(out):
size = x('getwindowgeometry', win()).split('Geometry: ')[1].strip()
ox, oy = geometry()
p = subprocess.Popen([
'ffmpeg', '-hide_banner', '-loglevel', 'error', '-f', 'x11grab',
'-framerate', '15', '-video_size', size, '-i', f'{DISPLAY}+{ox},{oy}',
'-c:v', 'libx264', '-preset', 'ultrafast', '-qp', '0', '-y', out,
])
open(f'{HOME}/rec.pid', 'w').write(str(p.pid))
time.sleep(0.5)
def rec_stop():
pid = int(open(f'{HOME}/rec.pid').read())
os.kill(pid, 2)
for _ in range(50):
try:
os.kill(pid, 0)
time.sleep(0.1)
except OSError:
break
def where():
out = x('getmouselocation')
mx, my = [int(v.split(':')[1]) for v in out.split()[:2]]
ox, oy = geometry()
print(mx - ox, my - oy)
def print_elements(elements):
for e in elements:
name = e['label'] if e['label'] else f"id:{e['id']}" if e['id'] else f"({e['type']})"
print(f"{e['x']:5.0f} {e['y']:5.0f} {e['w']:5.0f}x{e['h']:<4.0f} "
f"{(e['role'] or ''):14} {name!r}"
+ (f" = {e['value']!r}" if e['value'] else ''))
if __name__ == '__main__':
cmd, *a = sys.argv[1:]
if cmd == 'launch':
launch(a)
elif cmd == 'stop':
stop()
elif cmd == 'shot':
shot(a[0])
elif cmd == 'click':
click(*a)
elif cmd == 'move':
move(*a)
elif cmd == 'drag':
drag(*a)
elif cmd == 'type':
x('windowfocus', '--sync', win(), check=False)
x('type', '--delay', '120', a[0])
elif cmd == 'key':
x('windowfocus', '--sync', win(), check=False)
x('key', '--delay', '80', *a)
elif cmd == 'rec':
rec_start(a[0])
elif cmd == 'cut':
rec_stop()
elif cmd == 'where':
where()
elif cmd == 'labels':
print_elements(ask('labels ' + (a[0] if a else '')))
elif cmd == 'ids':
print_elements(ask('ids ' + (a[0] if a else '')))
elif cmd == 'locate':
print(json.dumps(matches(a[0]), indent=1))
elif cmd == 'click-on':
click_on(*a)
elif cmd == 'drag-on':
drag_on(*a)
elif cmd == 'hold-on':
hold_on(*a)
elif cmd == 'wait-for':
print(wait_for(*a))
elif cmd == 'wait-gone':
wait_gone(*a)
else:
sys.exit(__doc__)