summaryrefslogtreecommitdiff
path: root/test/hil/helper
diff options
context:
space:
mode:
authorHiFiPHile <[email protected]>2026-08-25 09:27:48 +0200
committerHiFiPHile <[email protected]>2026-08-25 09:27:48 +0200
commitdfac26a272fa7bbbca2050fbe9f1ca09008e548e (patch)
treeefbc53f8f2c1e5d9c7f38e5fef6d774a20053cec /test/hil/helper
parente590b45fcf51f9ddace73178074e4fe6d691e319 (diff)
parent5c0e31cdabaf37f14e1f5e988a020abfc1000495 (diff)
Merge master updates into the UAC1 host branch
Bring the audio work onto the current host core and build files before applying the remaining review fixes. Signed-off-by: HiFiPHile <[email protected]>
Diffstat (limited to 'test/hil/helper')
-rw-r--r--test/hil/helper/__init__.py4
-rw-r--r--test/hil/helper/hil_health.py380
-rwxr-xr-xtest/hil/helper/hil_lock.py525
-rw-r--r--test/hil/helper/hil_pool_check.py1019
-rwxr-xr-xtest/hil/helper/hil_select.py524
-rw-r--r--test/hil/helper/hil_util.py585
6 files changed, 3037 insertions, 0 deletions
diff --git a/test/hil/helper/__init__.py b/test/hil/helper/__init__.py
new file mode 100644
index 000000000..a080a2f55
--- /dev/null
+++ b/test/hil/helper/__init__.py
@@ -0,0 +1,4 @@
+# Marks helper/ as a REGULAR package. Without this it is only a PEP 420 namespace portion,
+# and a regular package named `helper` anywhere on sys.path wins over it even though
+# test/hil is sys.path[0] -- one transitive pip install would break every HIL entry point
+# at import. `helper` is a real distribution name on PyPI.
diff --git a/test/hil/helper/hil_health.py b/test/hil/helper/hil_health.py
new file mode 100644
index 000000000..92f0accc8
--- /dev/null
+++ b/test/hil/helper/hil_health.py
@@ -0,0 +1,380 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: MIT
+"""Shutting a wedged HIL run down: kill what the workers spawned, then report.
+
+A device whose usbfs node is held by a D-state process cannot be freed -- SIGKILL is not
+delivered in uninterruptible sleep -- so the goal is never to fix the rig from here. It is
+to free the runner's single job slot and leave a report naming what survived, instead of
+letting the job sit until GitHub cancels it with nothing to show.
+
+Deliberately shallow. We SIGKILL the process groups the workers spawned, wait a grace,
+and report whoever is still alive; we do not re-scan groups, prove pid ownership or
+escalate through sudo. A root-owned survivor is named in the report for hil_pool_check
+and the usb-kernel-recover skill to deal with -- signalling a pid we cannot prove is ours
+is the worse failure, and the job ceiling backstops whatever this misses.
+
+Everything here is stdlib-only and reads /proc unprivileged (dmesg is restricted on the
+rig), which keeps it importable -- and testable -- on a bare runner.
+"""
+import os
+import signal
+import threading
+import time
+from pathlib import Path
+
+PROC = Path('/proc')
+
+
+# How long to let a SIGKILL land before calling a process a survivor. Generous enough to
+# cover scheduling delay on a loaded rig, short enough that a fleet-wide sweep stays quick.
+CONFIRM_KILL_GRACE = 2.0
+
+
+def _p(*args, **kwargs) -> None:
+ # These run on the free-the-runner path, where stdout can already be a dead pipe (a
+ # dropped ssh session). An unguarded print would raise BrokenPipeError out of
+ # hil_test's inner finally, skipping shutdown_pool AND the report writing.
+ try:
+ print(*args, **kwargs)
+ except (OSError, ValueError):
+ # ValueError, not just OSError: printing to a CLOSED stream raises
+ # "ValueError: I/O operation on closed file", and both hil_pool_check and
+ # hil_test redirect stdout into a StringIO that can be closed under us. Escaping
+ # here skips shutdown_pool/kill_pool_children/os._exit -- stranding the runner,
+ # the exact failure this wrapper exists to prevent.
+ pass
+
+
+def _state(pid_dir: Path) -> str:
+ """The state letter from /proc/<pid>/stat. comm can contain ')', so the field is
+ located from the right rather than by splitting."""
+ # bytes, not read_text(): read_text decodes with the LOCALE encoding, so under LANG=C
+ # (systemd services, self-hosted runners) a non-ASCII comm raises UnicodeDecodeError
+ # and the entry silently vanishes from the scan.
+ stat = (pid_dir / 'stat').read_bytes()
+ return chr(stat[stat.rindex(b')') + 2])
+
+
+def _pids():
+ """/proc pid entries. Yields nothing rather than raising if /proc is unreadable."""
+ try:
+ entries = list(PROC.iterdir())
+ except OSError:
+ return
+ for entry in entries:
+ if entry.name.isdigit():
+ yield entry
+
+
+def d_state_note() -> str:
+ """Pids in uninterruptible sleep, for the report. Never aborts, never blocks.
+
+ A D-state process at start-up is NOT a fault on its own -- a healthy in-flight testusb
+ looks exactly like this, and the rig supports a dev run alongside CI. It is a hint for
+ whoever reads a red cell below. Diagnosis proper is hil_pool_check and the
+ usb-kernel-recover skill; this is one line, not a probe."""
+ stuck = []
+ for d in PROC.glob('[0-9]*'):
+ try:
+ if _state(d) == 'D':
+ stuck.append(d.name)
+ except (OSError, ValueError, IndexError):
+ pass # raced with exit, or /proc is restricted: not our problem here
+ if not stuck:
+ return ''
+ return (f'{len(stuck)} process(es) in D state when this run started: '
+ f'{sorted(stuck)[:10]}')
+
+
+def shutdown_pool(pool, grace: float = 30) -> bool:
+ """terminate() a worker Pool without ever blocking forever.
+
+ multiprocessing joins its workers unbounded (util.py _exit_function terminate()s the
+ daemonic ones, then calls p.join() -- no timeout -- on every remaining active child,
+ CPython 3.13.5), and a worker in uninterruptible sleep never
+ reaps -- so terminate() itself hangs, taking the runner's only job slot with it. False
+ when the pool refuses to die within `grace` (the caller must then abandon it); a
+ terminate() that *raises* counts as failure too, the pool being just as alive."""
+ outcome = {}
+
+ def _term():
+ try:
+ pool.terminate()
+ outcome['ok'] = True
+ except BaseException as e: # noqa: BLE001 - any failure means the pool is still up
+ # Say what happened: Pool._terminate_pool really can raise (CPython:
+ # AssertionError 'Cannot have cache with result_handler not alive'), and a
+ # swallowed one is indistinguishable from an unkillable D-state worker.
+ outcome['err'] = e
+ _p(f'warning: Pool.terminate() raised {type(e).__name__}: {e}', flush=True)
+
+ t = threading.Thread(target=_term, daemon=True)
+ t.start()
+ t.join(grace)
+ # Decide on the thread, not the dict: _term may set outcome['ok'] after join(grace)
+ # expired, reporting a merely-slow terminate as success on one read and abandoned on
+ # another. Still inside terminate() == not shut down.
+ if t.is_alive():
+ return False
+ return outcome.get('ok', False)
+
+
+def child_procs(pids) -> dict:
+ """{ancestor pid in `pids`: [(descendant pid, its pgid), ...]}, from ONE walk of /proc.
+
+ DESCENDANTS, not direct children: a worker's usbtest.py spawns its recovery flasher
+ through run_cmd (own session), so it is a GRANDCHILD that a direct-child sweep misses
+ and a kill mid-recovery would orphan on the probe. pgid comes back too because the two
+ kinds of child need different signals (see kill_pool_children)."""
+ wanted = set(pids)
+ by_parent: dict = {} # ppid -> [(pid, pgid), ...] for EVERY process
+ for entry in _pids():
+ try:
+ stat = (entry / 'stat').read_bytes()
+ except OSError:
+ continue # exited between the scan and the read, or not readable
+ # comm (field 2) is parenthesised and may contain spaces and ')' -- so split only
+ # what follows the LAST ')': state, ppid, pgrp, ...
+ try:
+ fields = stat[stat.rindex(b')') + 2:].split()
+ ppid, pgid = int(fields[1]), int(fields[2])
+ except (ValueError, IndexError):
+ continue # truncated or unparsable stat line
+ by_parent.setdefault(ppid, []).append((int(entry.name), pgid))
+ out: dict = {}
+ for root in wanted:
+ todo = list(by_parent.get(root, []))
+ while todo:
+ pid, pgid = todo.pop()
+ out.setdefault(root, []).append((pid, pgid))
+ todo += by_parent.get(pid, [])
+ return out
+
+
+def _pool_procs(pool, extra) -> list:
+ """The pool's worker Process objects, plus each extra's own process.
+
+ Manager() runs in its own child process and inherits the same descriptors as the
+ workers, so leaving it behind defeats the point: os._exit skips its finalizer."""
+ procs = list(getattr(pool, '_pool', []) or [])
+ for e in extra:
+ procs.append(getattr(e, '_process', e))
+ return procs
+
+
+
+
+def kill_worker_children(pool, *extra) -> int:
+ """SIGKILL what the pool's workers spawned; returns how many SURVIVED.
+
+ For the TIMEOUT path only. On the normal path each worker has already run
+ kill_own_children() and retired (maxtasksperchild=1), so this walks fresh idle workers
+ and finds nothing -- measured: 4 tasks, zero overlap with the pool at sweep time.
+
+ Call it BEFORE shutdown_pool(): terminate() reaps the (interruptible) worker and its
+ flasher is reparented to init, erasing the ppid link this matches on. Signalling the
+ worker's own group instead cannot work -- a forked pool worker inherits OUR group
+ (CPython 3.13.5 multiprocessing never setsid/setpgid) and run_cmd gives every flasher
+ a session of its own.
+
+ TWO passes because our SIGKILL can fail an in-flight flash and the worker then retries
+ in a fresh session, which one /proc snapshot misses. `seen` stops a pid signalled in
+ pass 1 being confirmed twice.
+ """
+ seen: set = set()
+ total = 0
+ for i in range(2):
+ if i:
+ time.sleep(0.5)
+ procs = _pool_procs(pool, extra)
+ total += _kill_kids(
+ child_procs(getattr(p, 'pid', None) for p in procs if p is not None), seen)
+ return total
+
+
+def kill_own_children() -> int:
+ """SIGKILL what THIS process spawned. Returns how many survived.
+
+ For the worker to call before it returns. maxtasksperchild=1 retires it the moment the
+ task ends, reparenting its children to init, so main()'s sweep walks fresh idle workers
+ and finds nothing (measured over 4 tasks: zero overlap, sweep 0, 4 strays alive).
+ Inside the worker the ppid link is still live.
+ """
+ return _kill_kids(child_procs([os.getpid()]), set())
+
+
+def _kill_kids(kids: dict, seen: set) -> int:
+ """SIGKILL every pid in a ppid-tree snapshot; return how many survived.
+
+ Every pid here is a DESCENDANT of a process we own, so it is ours by construction -- no
+ argv identity check, because we never signal anything we did not discover through our
+ own ppid tree.
+ """
+ try:
+ own = os.getpgid(0)
+ except OSError:
+ own = None # cannot tell our own group apart: never killpg, signal pids only
+ # One list: every pid here is a DESCENDANT of one of our own workers, so it is ours by
+ # construction -- no argv identity check needed, because we never signal anything we
+ # did not discover through our own ppid tree.
+ touched: list = []
+ for children in kids.values():
+ for cpid, cpgid in children:
+ if cpid in seen:
+ continue # a previous pass already signalled it
+ seen.add(cpid)
+ try:
+ if own is not None and cpgid != own:
+ # A run_cmd child: its own session, so one killpg also reaps what it
+ # spawned. Recorded because killpg cannot report a partial kill.
+ os.killpg(cpgid, signal.SIGKILL)
+ else:
+ # Shares our group (a plain subprocess.run), so killpg would take
+ # down the whole run -- it is signalled by pid in _kill_and_confirm.
+ pass
+ touched.append(cpid)
+ except PermissionError:
+ # NOT "already gone": the signal did not land, so this pid MUST still be
+ # confirmed, or the one case this handler exists for (an all-root session:
+ # the sudo wrapper died, its root members did not) is the one case that
+ # never reaches the report.
+ touched.append(cpid)
+ except ProcessLookupError:
+ pass # already gone
+ except OSError:
+ pass
+ # Both paths need confirming: a killpg'd flasher and a same-group mtype blocked on a
+ # wedged device are both in D state, and os.kill reported success on either.
+ denied = _kill_and_confirm(touched)
+ if denied:
+ _p(f'warning: could not kill {sorted(denied)}; they still hold whatever they '
+ f'had open (probe, usbfs node) into the next job', flush=True)
+ # SURVIVORS, not the signalled-child count: the caller needs to know the rig is dirty
+ # for the next job, and a count of what we successfully signalled cannot tell it that.
+ # (They are different units anyway -- a killpg is counted once per child sharing the
+ # group -- so the old return was never comparable to anything.)
+ return len(denied)
+
+
+def _kill_and_confirm(pids) -> list:
+ """SIGKILL every pid, then return those STILL alive after ONE grace window.
+
+ SIGKILL is QUEUED, not delivered, for a task in uninterruptible sleep -- and testusb
+ waits in a plain wait_for_completion() with no timeout (v6.12.96 usbtest.c:1404;
+ usb_sg_wait, message.c:765), so that is the normal state of a healthy in-flight case
+ too. os.kill returning success proves nothing; only the recheck does. It is also
+ asynchronous, so probing immediately reports a process we just killed as a survivor
+ (measured: 11 of 20 plain `sleep`s with no grace).
+
+ Signal all, then poll the set against ONE shared deadline: per-pid windows made this
+ scale with stray count, minutes on a convoy. A pid we cannot signal is reported, never
+ sudo-killed.
+ """
+ pending = []
+ for pid in pids:
+ try:
+ os.kill(pid, signal.SIGKILL)
+ except ProcessLookupError:
+ continue # already gone
+ except OSError:
+ pass # EPERM (root-owned): it stays, and the poll below reports it
+ pending.append(pid)
+
+ deadline = time.monotonic() + CONFIRM_KILL_GRACE
+ while True:
+ alive = []
+ for pid in pending:
+ try:
+ os.kill(pid, 0)
+ except ProcessLookupError:
+ continue # ESRCH: genuinely gone
+ except OSError:
+ pass # EPERM: it exists; the state check decides
+ try:
+ # a ZOMBIE answers kill(pid, 0) too: dead, merely unreaped. Not a survivor.
+ if _state(PROC / str(pid)) == 'Z':
+ continue
+ except (OSError, ValueError, IndexError):
+ continue # unreadable: assume gone rather than cry wolf
+ alive.append(pid)
+ pending = alive
+ if not pending or time.monotonic() >= deadline:
+ return pending # outlasted SIGKILL: D state, or not ours to kill
+ time.sleep(0.02)
+
+
+def kill_pool_children(pool, *extra) -> int:
+ """SIGKILL the pool's worker processes themselves. Returns how many are STILL ALIVE
+ after the grace -- not how many were signalled.
+
+ Survivors, not signals: the caller turns this number into "power-cycle the host", so
+ counting signals would send someone to a hypervisor over workers that all died.
+
+ Call after a shutdown_pool() that returned False, and after kill_worker_children().
+ A D-state worker ignores SIGKILL, but every other worker dies and drops the inherited
+ descriptors -- a survivor holds the runner's stdout pipe open and the runner waits for
+ EOF even after we exit, so the early exit would not free the job slot."""
+ killed_procs: list = []
+ for proc in _pool_procs(pool, extra):
+ try:
+ # Process.kill(), never a raw pid: multiprocessing's _send_signal re-checks
+ # `self.returncode is None` first, so once shutdown_pool's thread has reaped a
+ # worker this is a no-op instead of signalling a pid the OS may have recycled.
+ # os.pidfd_open(proc.pid) is worse: it skips that guard entirely.
+ if proc is None or not proc.is_alive():
+ continue
+ proc.kill()
+ killed_procs.append(proc)
+ except (OSError, AttributeError, ValueError):
+ continue # already reaped, never started, or not a real process
+ # Re-check the Process objects, never the pids collected a moment ago: shutdown_pool's
+ # thread is STILL join()ing workers, so a pid killed here can be reaped and RECYCLED
+ # before _kill_and_confirm signals it -- and on EPERM that escalates to `sudo -n kill
+ # -9 <stale pid>`, killing an unrelated ROOT process as the last act before os._exit.
+ killed_pids = []
+ for proc in killed_procs:
+ try:
+ if proc.is_alive() and proc.pid is not None:
+ killed_pids.append(proc.pid)
+ except (OSError, AttributeError, ValueError):
+ continue
+ # SIGKILL is asynchronous and a D-state task ignores it: only a confirmed survivor
+ # justifies the caller's power-cycle wording
+ return len(_kill_and_confirm(killed_pids)) if killed_pids else 0
+
+
+def write_timeout_report(report_dir: Path, boards, secs: int, md_name: str,
+ banner: str = '', prefix: str = '') -> None:
+ """Leave a report behind when the worker pool has to be abandoned.
+
+ map_async is all-or-nothing, so a timeout loses every per-board result and the report
+ dir would stay empty with no reason for the failure. Any prior attempt's markdown is
+ kept below the banner."""
+ # `prefix` carries the preflight rig-health verdict: the timeout aborts before
+ # accumulate_report, so without it the report loses the one line saying WHY the pool
+ # never finished. The '\n' stops Markdown lazy continuation pulling the banner into
+ # the blockquote.
+ try:
+ # Built INSIDE the try: a roster entry without a 'name' key raises KeyError while
+ # assembling the board list, and outside the try that escaped and stranded the
+ # runner -- which is exactly what the broad handler below exists to prevent.
+ head = (prefix + '\n' if prefix else '') + (banner or (
+ f'**HIL run abandoned: worker pool timed out after {secs}s.**\n\n'
+ f'No per-board results could be collected for this attempt, so the '
+ f'table below (if any) is from an earlier one. Boards dispatched:\n\n'
+ + '\n'.join(f'- {b.get("name", "?")}' for b in boards) + '\n'))
+ report_dir.mkdir(parents=True, exist_ok=True)
+ md_path = report_dir / md_name
+ # Its own handler so it cannot take the write down with it: a report torn by an
+ # attempt killed mid-write raises UnicodeDecodeError (a ValueError, and prior
+ # reports always contain status emoji), which under a shared try skipped the write
+ # entirely. Losing the old table is a nicety; losing the banner is the failure.
+ try:
+ prior = md_path.read_text(encoding='utf-8') if md_path.is_file() else ''
+ except (OSError, ValueError):
+ prior = ''
+ md_path.write_text(head + (f'\n{prior}' if prior else ''), encoding='utf-8')
+ except Exception as e: # noqa: BLE001
+ # Deliberately broad: this is the first statement of the pool-abandon path, so ANY
+ # escape skips kill_pool_children and os._exit and strands the runner.
+ _p(f'warning: cannot write {md_name} to {report_dir}: {e}', flush=True)
diff --git a/test/hil/helper/hil_lock.py b/test/hil/helper/hil_lock.py
new file mode 100755
index 000000000..7757ef17d
--- /dev/null
+++ b/test/hil/helper/hil_lock.py
@@ -0,0 +1,525 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: MIT
+"""Board locks + controller permits for the TinyUSB HIL rig.
+
+Board locks are kernel flocks in BOARD_LOCK_DIR arbitrating hardware access
+between dev sessions and CI's hil_test.py (never stop the actions-runner).
+Controller permits are in-process semaphores budgeting flashes and usbtest
+batteries per host controller; they have no CLI meaning. The CLI below
+(hold/release/status) manages board locks only.
+"""
+import argparse
+import fcntl
+import json
+import os
+import re
+import select
+import signal
+import sys
+import time
+
+sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # helper/ scripts import via the test/hil root
+from helper import hil_util
+
+BOARD_LOCK_DIR = '/tmp/tinyusb-hil-locks'
+CI_REASON = 'hil_test.py' # release-protected holder tag (release refuses to kill it)
+PROTECTED_REASONS = {CI_REASON, 'pool_check'} # cmd_release refuses to SIGTERM these holders
+PROFILE = os.environ.get('HIL_PROFILE') == '1'
+
+
+def lock_path(board: str) -> str:
+ return os.path.join(BOARD_LOCK_DIR, f'{board}.lock')
+
+
+def flock_nb(board: str):
+ """Open-or-create the lock file WITHOUT truncating (a losing racer must not
+ wipe the winner's record) and take LOCK_EX|LOCK_NB. Returns the open handle;
+ raises OSError when the flock is held elsewhere (handle already closed)."""
+ fd = os.open(lock_path(board), os.O_RDWR | os.O_CREAT, 0o666)
+ fh = os.fdopen(fd, 'r+')
+ try:
+ fcntl.flock(fh, fcntl.LOCK_EX | fcntl.LOCK_NB)
+ except OSError:
+ fh.close()
+ raise
+ return fh
+
+
+def write_record(fh, reason: str) -> bool:
+ """Holder record; the flock itself is already held. Returns False on a write failure:
+ acquire_board_lock stays best-effort (the flock is the authority), but cmd_hold aborts
+ -- a hold whose record is missing is invisible to status/release."""
+ try:
+ fh.truncate(0)
+ fh.seek(0)
+ json.dump({'pid': os.getpid(), 'reason': reason,
+ 'since': time.strftime('%Y-%m-%dT%H:%M:%S%z')}, fh)
+ fh.flush()
+ return True
+ except OSError:
+ return False
+
+
+def clear_record(fh) -> None:
+ """Clear our record before dropping the flock so records stay truthful."""
+ try:
+ fh.truncate(0)
+ except OSError:
+ pass
+
+
+def read_record(board: str):
+ try:
+ with open(lock_path(board)) as f:
+ return json.load(f)
+ except (OSError, ValueError):
+ return None
+
+
+# --- per-board dev-session locks ------------------------------------------
+def acquire_board_lock(board_name, reason=CI_REASON):
+ """Take this board's flock for the duration of its flash+test.
+ Returns an open file handle (keep it referenced; closing releases it),
+ or None when HIL_NO_BOARD_LOCK=1 or the lock dir is unusable (fail-open:
+ locking must never break a test run by itself).
+ Raises RuntimeError only when another session holds the board."""
+ import fcntl
+ if os.environ.get('HIL_NO_BOARD_LOCK') == '1':
+ return None # user-authorized bypass — see hil skill
+ try:
+ os.makedirs(BOARD_LOCK_DIR, exist_ok=True)
+ fd = os.open(os.path.join(BOARD_LOCK_DIR, f'{board_name}.lock'),
+ os.O_RDWR | os.O_CREAT, 0o666)
+ fh = os.fdopen(fd, 'r+')
+ except OSError as e:
+ # odd lock dir (perms, path collision): proceed unlocked, but say so —
+ # a silent fail-open is indistinguishable from the intentional bypass
+ print(f'warning: board lock unavailable for {board_name} ({e}); proceeding unlocked',
+ flush=True)
+ return None
+ try:
+ fcntl.flock(fh, fcntl.LOCK_EX | fcntl.LOCK_NB)
+ except OSError:
+ try:
+ info = fh.read(500).strip()
+ except (OSError, UnicodeDecodeError):
+ info = ''
+ fh.close()
+ raise RuntimeError(f'board locked: {info or "unknown holder"}')
+ # announce ourselves so the other side's conflict message is truthful;
+ # best-effort — the flock itself is already held
+ try:
+ fh.truncate(0)
+ fh.seek(0)
+ json.dump({'pid': os.getpid(), 'reason': reason,
+ 'since': time.strftime('%Y-%m-%dT%H:%M:%S%z')}, fh)
+ fh.flush()
+ except OSError:
+ pass
+ return fh
+
+
+# Per-host-controller concurrency (see controller_of/controller_slot below): a usbtest
+# battery saturates its DUT's host controller, so batteries and flashes are budgeted per
+# controller. The 4/2 defaults trade ~3.5 min on the usbtest leg for bandwidth margin on
+# the shared leaf-hub uplinks, where battery case failures were observed from 12/8
+# (profiled 2026-07-13/14: 22.2/14.3/12.5/10.8 min at usbtest width 1/2/3/4, plateau
+# after). Raise per run via HIL_FLASH_PARALLEL/HIL_USBTEST_PARALLEL.
+# - uPD720201 cards need firmware >= 2.0.2.6 (RAM-uploaded, reloads every power cycle):
+# the ROM firmware dies under battery + re-enumeration churn.
+# - a marginal DUT port bouncing during concurrent batteries can kill a uPD720201 ("xHCI
+# host not responding to stop endpoint command"): fix the port/cable or pull the board
+# -- lowering the widths does not fix a bad port (2026-07-16, every death).
+FLASH_PARALLEL = hil_util.pos_int_env('HIL_FLASH_PARALLEL', 4)
+USBTEST_PARALLEL = hil_util.pos_int_env('HIL_USBTEST_PARALLEL', 2)
+CONTROLLER_SLOTS = 12 # lock slots; controllers are assigned to slots on first sight
+# Bound on ONE permit wait. Generous: a real queue behind a slow board is normal,
+# and this only has to beat the pool guard so a leaked permit cannot consume it.
+PERMIT_TIMEOUT = hil_util.pos_int_env('HIL_PERMIT_TIMEOUT', 900)
+# CONTROLLER_SLOTS + 1 entries each, built by make_permit_sems: UNKNOWN_SLOT indexes the
+# extra one. Sized to CONTROLLER_SLOTS instead, the first unresolved board IndexErrors
+# inside a pool worker -- which now surfaces through drain_pool as a worker-raise (the
+# finished boards survive), but still loses this board and aborts the run.
+usbtest_sems = None # per-slot usbtest-battery permits
+flash_sems = None # per-slot flash permits
+controller_map = None # shared dict: 'pci:<addr>' -> slot, 'uid:<uid>' -> pci addr cache
+controller_meta = None # guards slot assignment in controller_map
+controller_hints = {} # static uid -> pci from the last run's cache (read-only per worker)
+
+
+log = print # hil_test.init_worker points this at log_line via init_scheduling
+
+
+def init_scheduling(b_sems, f_sems, cmap, cmeta, hints, log_fn=None):
+ """Install per-worker scheduling state (called from hil_test.init_worker)."""
+ global usbtest_sems, flash_sems, controller_map, controller_meta, controller_hints, log
+ usbtest_sems, flash_sems = b_sems, f_sems
+ controller_map, controller_meta, controller_hints = cmap, cmeta, hints
+ if log_fn is not None:
+ log = log_fn
+
+
+# -------------------------------------------------------------
+# Per-controller scheduling
+# -------------------------------------------------------------
+def controller_of(uid: str):
+ """Resolve a DUT uid to its root host controller's PCI address, or None when it cannot
+ be resolved — the device is not enumerated (e.g. parked in board_test firmware with USB
+ off), or sysfs would not answer. Successful resolutions are cached — cabling does not
+ change mid-run. Dual-port parts (e.g. CH32V307 usbhs/usbfs variants) share one uid and
+ one cache entry: budgeting is only exact when both ports sit on the same controller
+ (true on this rig)."""
+ if controller_map is None:
+ return None
+ cached = controller_map.get(f'uid:{uid}')
+ if cached:
+ return cached
+ # vid='cafe' first: the target is always a TinyUSB DUT, and the VID is a lock-free
+ # descriptor field. Without it this read every probe's and hub's `serial` -- the
+ # attribute served under device_lock -- so a HEALTHY peer mid-usbtest would strand a
+ # reader here and spend one of this worker's four blindness credits.
+ devs, _ = hil_util.usb_scan(vid='cafe', serial=uid)
+ for dev in devs:
+ busnum = hil_util.read_sysfs(os.path.join(dev['dir'], 'busnum'))
+ if busnum is None or busnum is hil_util.SYSFS_UNKNOWN:
+ continue
+ try:
+ root = os.path.realpath(f'/sys/bus/usb/devices/usb{int(busnum)}')
+ except ValueError:
+ continue
+ m = re.findall(r'[0-9a-f]{4}:[0-9a-f]{2}:[0-9a-f]{2}\.[0-9a-f]', root)
+ if m:
+ controller_map[f'uid:{uid}'] = m[-1]
+ return m[-1]
+ return None
+
+
+def controller_slot(pci: str) -> int:
+ """Map a controller PCI address to a lock slot (assigned on first sight)."""
+ key = f'pci:{pci}'
+ with controller_meta:
+ slot = controller_map.get(key)
+ if slot is None:
+ slot = controller_map.get('nslots', 0)
+ if slot >= CONTROLLER_SLOTS:
+ slot = 0 # more controllers than slots: overflow shares slot 0 (safe, over-serialized)
+ else:
+ controller_map['nslots'] = slot + 1
+ controller_map[key] = slot
+ return slot
+
+
+# Unresolved boards budget in a slot of their OWN, one past the real ones, and that slot
+# holds exactly ONE permit whatever the per-controller width is. Neither neighbour works:
+# a permit on every slot (the old fail-closed rule) serialized the whole fleet the moment
+# a worker went blind, while a full private budget let unknown boards run a second
+# controller's worth of batteries on top of the resolved ones -- doubling the load on
+# whichever physical controller they actually sit on, which is the saturation the
+# uPD720201 deaths above are attributed to. Width 1 caps the over-subscription at +1.
+UNKNOWN_SLOT = CONTROLLER_SLOTS
+
+
+def make_permit_sems(semaphore, width: int) -> list:
+ """One semaphore per controller slot at `width`, plus the unknown bucket at 1."""
+ return [semaphore(width) for _ in range(CONTROLLER_SLOTS)] + [semaphore(1)]
+
+
+class controller_permit:
+ """Context manager: one permit from `sems` on the board's controller slot. An
+ unresolved controller budgets in UNKNOWN_SLOT, which admits one at a time: unresolved
+ boards serialize against each other, never against the whole rig, and never add a
+ second full budget to a controller. `warn_unknown` logs that fallback (used by
+ usbtest, where the device is expected to be enumerated by the caller)."""
+ def __init__(self, sems, uid: str, warn_unknown: bool = False):
+ self.sems = sems
+ self.slots = None
+ self.uid = uid
+ # what __enter__ actually ACQUIRED. Not the same as self.slots: a bounded acquire
+ # that times out is skipped on purpose, and releasing it anyway would add a permit
+ # that was never taken -- multiprocessing semaphores are unbounded, so the width
+ # grows for the rest of the run, on the controller throttle that exists to keep
+ # concurrent batteries from killing the uPD720201 xHCI.
+ self.taken: list = []
+ if sems is None:
+ return
+ # Hint FIRST for flash budgeting: a mis-budgeted flash is harmless, and the board
+ # is usually parked in board_test with USB off at this point, so controller_of
+ # cannot resolve it anyway -- it just walks the whole bus to say so, once per
+ # flash permit (~14 examples x ~21 boards a leg), each walk spawning a bounded
+ # reader per device. usbtest still resolves for real (warn_unknown), and by then
+ # the DUT is enumerated, so that walk succeeds and caches.
+ pci = None if warn_unknown else controller_hints.get(uid)
+ if pci is None:
+ pci = controller_of(uid)
+ if pci is None and warn_unknown:
+ log(f'warning: cannot resolve {uid} to a host controller'
+ f'{hil_util.sysfs_blind_note()}; budgeting it in the unknown bucket')
+ self.slots = [controller_slot(pci) if pci else UNKNOWN_SLOT]
+
+ def __enter__(self):
+ if self.slots:
+ t0 = time.monotonic()
+ taken = self.taken = []
+ try:
+ for s in self.slots:
+ # BOUNDED. multiprocessing semaphores are NOT released when a holder
+ # dies, and the pool sweep SIGKILLs workers -- so a permit lost that
+ # way would block every later worker on this controller forever, and
+ # boards unrelated to the wedge would burn the whole pool guard. On
+ # expiry proceed over-subscribed and say so: a slower controller is a
+ # far better failure than a hung run.
+ if not self.sems[s].acquire(timeout=PERMIT_TIMEOUT):
+ log(f'warning: waited {PERMIT_TIMEOUT}s for a permit on slot {s} '
+ f'(uid {self.uid}); a holder probably died without releasing '
+ f'it -- proceeding over-subscribed')
+ continue
+ taken.append(s)
+ # inside the try: a failed __enter__ never gets its __exit__, so a raise
+ # here (e.g. broken stdout) must still release the permits
+ if PROFILE and time.monotonic() - t0 > 1.0:
+ log(f'[prof] permit wait {time.monotonic() - t0:.1f}s '
+ f'(uid {self.uid}, slots {self.slots})')
+ except BaseException:
+ for s in reversed(taken):
+ self.sems[s].release()
+ raise
+ return self
+
+ def __exit__(self, *exc):
+ if self.slots:
+ for s in reversed(self.taken):
+ self.sems[s].release()
+ self.taken = []
+ return False
+
+
+def flash_permit(uid: str) -> controller_permit:
+ return controller_permit(flash_sems, uid)
+
+
+def usbtest_permit(uid: str) -> controller_permit:
+ return controller_permit(usbtest_sems, uid, warn_unknown=True)
+
+
+# --- operator CLI (hold/release/status) ------------------------------------
+def boards_from_config(config: str) -> list:
+ """All board names, INCLUDING boards-skip: `hold --all` guards rig-wide
+ operations, and parked boards can still be touched (pool_check -b names them
+ explicitly), so a rig-wide hold that skipped them would leave a gap."""
+ try:
+ with open(config) as f:
+ cfg = json.load(f)
+ return [b['name'] for b in cfg['boards'] + cfg.get('boards-skip', [])]
+ except (OSError, ValueError, KeyError) as e:
+ print(f'ERROR: cannot read board roster {config}: {e}', file=sys.stderr)
+ sys.exit(1)
+
+
+def is_locked(board: str) -> bool:
+ """True if the recorded holder process is still alive.
+
+ Deliberately never touches the flock: even a momentary probe lock would
+ make a concurrent acquirer's LOCK_NB attempt fail spuriously. The flock
+ taken by acquirers themselves stays the only authority."""
+ info = read_record(board)
+ pid = info.get('pid') if isinstance(info, dict) else None
+ if not isinstance(pid, int) or pid <= 0:
+ return False
+ try:
+ os.kill(pid, 0)
+ except ProcessLookupError:
+ return False
+ except PermissionError:
+ return True # alive but owned by another user (e.g. the CI runner)
+ return True
+
+
+def cmd_hold(boards, reason):
+ os.makedirs(BOARD_LOCK_DIR, exist_ok=True)
+ # No pre-check: the holder's own LOCK_NB flock is the only authority, since a recorded
+ # pid may be stale or recycled. The holder signals success through this pipe because a
+ # generic is_locked() poll would be fooled by a RIVAL invocation's flock — only the
+ # holder knows whether it won every board.
+ r_fd, w_fd = os.pipe()
+ pid = os.fork()
+ if pid > 0:
+ os.close(w_fd)
+ os.waitpid(pid, 0) # reap intermediate child
+ ready, _, _ = select.select([r_fd], [], [], 10)
+ ok = bool(ready) and os.read(r_fd, 1) == b'1'
+ os.close(r_fd)
+ if ok:
+ print(f'held: {", ".join(boards)}')
+ return 0
+ for b in boards:
+ info = read_record(b)
+ if info:
+ print(f'ERROR: {b} locked: {info}', file=sys.stderr)
+ print('ERROR: holder failed to acquire locks', file=sys.stderr)
+ return 1
+ # intermediate child: detach, then spawn the actual holder
+ os.setsid()
+ if os.fork() > 0:
+ os._exit(0)
+ # holder (grandchild): acquire all flocks, signal the parent, sleep until killed
+ os.close(r_fd)
+ # Keep the success pipe clear of fds 0-2: invoked with stdio closed, os.pipe() can
+ # land there and the dup2 loop below would clobber it.
+ if w_fd <= 2:
+ w_fd = fcntl.fcntl(w_fd, fcntl.F_DUPFD, 3)
+ # Detach stdio: a `hold` whose output is captured must see EOF when the front-end
+ # exits — the immortal holder must not keep that pipe open.
+ devnull = os.open(os.devnull, os.O_RDWR)
+ for std_fd in (0, 1, 2):
+ os.dup2(devnull, std_fd)
+ if devnull > 2:
+ os.close(devnull)
+ try:
+ handles = []
+ for b in boards:
+ fh = flock_nb(b)
+ if not write_record(fh, reason):
+ raise OSError(f'cannot write holder record for {b}')
+ handles.append(fh)
+ except OSError:
+ try:
+ os.write(w_fd, b'0')
+ except OSError:
+ pass
+ os._exit(1) # lost a race; parent reports the failure
+ os.write(w_fd, b'1')
+ os.close(w_fd)
+
+ def _bow_out(*_):
+ # clear the records before dying so read_record/status stay truthful (the kernel
+ # drops the flocks themselves on exit either way)
+ for h in handles:
+ clear_record(h)
+ os._exit(0)
+
+ signal.signal(signal.SIGTERM, _bow_out)
+ while True:
+ signal.pause()
+
+
+def cmd_release(boards):
+ rc = 0
+ victims = set()
+ for b in boards:
+ try:
+ fd = os.open(lock_path(b), os.O_RDWR)
+ except OSError:
+ continue # no lock file (or another user's): nothing we can release
+ fh = os.fdopen(fd, 'r+')
+ try:
+ fcntl.flock(fh, fcntl.LOCK_EX | fcntl.LOCK_NB)
+ except OSError:
+ # flock genuinely held — never SIGTERM on a mere pid record: the pid may be
+ # recycled, or a live worker that already moved on.
+ fh.close()
+ info = read_record(b) or {}
+ pid = info.get('pid')
+ reason = info.get('reason')
+ if reason in PROTECTED_REASONS:
+ print(f'ERROR: {b} is mid-test by {reason} (pid {pid}) — not killing it; '
+ 'wait for it to finish', file=sys.stderr)
+ rc = 1
+ elif isinstance(pid, int) and pid > 0:
+ victims.add(pid)
+ else:
+ print(f'ERROR: {b} is held but its record is unreadable', file=sys.stderr)
+ rc = 1
+ continue
+ # flock was free: only a stale record remained — clear it
+ clear_record(fh)
+ fh.close()
+ for holder in sorted(victims):
+ try:
+ os.kill(holder, signal.SIGTERM)
+ print(f'released holder pid {holder}')
+ except ProcessLookupError:
+ pass
+ except PermissionError:
+ print(f'ERROR: holder pid {holder} belongs to another user — cannot signal it',
+ file=sys.stderr)
+ rc = 1
+ time.sleep(0.3)
+ still = [b for b in boards if is_locked(b)]
+ if still:
+ print(f'ERROR: still locked: {", ".join(still)}', file=sys.stderr)
+ return 1
+ return rc
+
+
+def cmd_status():
+ if not os.path.isdir(BOARD_LOCK_DIR):
+ print('no locks')
+ return 0
+ any_locked = False
+ for fn in sorted(os.listdir(BOARD_LOCK_DIR)):
+ if not fn.endswith('.lock'):
+ continue
+ b = fn[:-5]
+ if is_locked(b):
+ any_locked = True
+ print(f'{b}: {read_record(b)}')
+ if not any_locked:
+ print('no locks')
+ return 0
+
+
+_CLI_USAGE = """Per-board advisory locks for the HIL rig.
+
+Arbitrates board access between dev sessions and CI's hil_test.py without
+stopping the actions-runner. Locks are kernel flocks: the kernel releases
+them automatically when the holder process dies, and holders clear their
+lock-file record on release so records stay truthful (/tmp also clears on
+reboot).
+
+Usage:
+ hil_lock.py hold BOARD [BOARD...] --reason TEXT
+ hil_lock.py hold --all [--config CONFIG.json] --reason TEXT
+ hil_lock.py release BOARD [BOARD...] | release --all
+ hil_lock.py status
+
+A holder process holds ALL boards given in one `hold` call; releasing any of
+them kills that holder and releases all of its boards.
+"""
+
+
+def main():
+ ap = argparse.ArgumentParser(description=_CLI_USAGE,
+ formatter_class=argparse.RawDescriptionHelpFormatter)
+ sub = ap.add_subparsers(dest='cmd', required=True)
+ p_hold = sub.add_parser('hold')
+ p_hold.add_argument('boards', nargs='*')
+ p_hold.add_argument('--all', action='store_true')
+ p_hold.add_argument('--config',
+ default=os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
+ 'tinyusb.json'),
+ help='board roster JSON (default: tinyusb.json in test/hil, one level above this script)')
+ p_hold.add_argument('--reason', required=True)
+ p_rel = sub.add_parser('release')
+ p_rel.add_argument('boards', nargs='*')
+ p_rel.add_argument('--all', action='store_true')
+ sub.add_parser('status')
+ a = ap.parse_args()
+ if a.cmd == 'hold':
+ boards = boards_from_config(a.config) if a.all else a.boards
+ if not boards:
+ ap.error('no boards given (name boards or use --all)')
+ sys.exit(cmd_hold(boards, a.reason))
+ if a.cmd == 'release':
+ if a.all:
+ boards = ([fn[:-5] for fn in os.listdir(BOARD_LOCK_DIR) if fn.endswith('.lock')]
+ if os.path.isdir(BOARD_LOCK_DIR) else [])
+ else:
+ boards = a.boards
+ if not boards:
+ ap.error('no boards given (name boards or use --all)')
+ sys.exit(cmd_release(boards))
+ sys.exit(cmd_status())
+
+
+if __name__ == '__main__':
+ main()
diff --git a/test/hil/helper/hil_pool_check.py b/test/hil/helper/hil_pool_check.py
new file mode 100644
index 000000000..371aff1e1
--- /dev/null
+++ b/test/hil/helper/hil_pool_check.py
@@ -0,0 +1,1019 @@
+#!/usr/bin/env python3
+"""Quick HIL pool health check.
+
+For every board in the rig's HIL config: is the flash probe on the USB bus, does a
+light example flash, and does the board's USB device (uid) come back up? Missing
+firmware is BUILT on the spot (tools/build.py, idf.py for espressif; one get_deps
+retry) — never skipped; --no-build opts out. Applies only per-device-safe recovery
+(probe authorized-toggle, board reset/re-flash) and prints a markdown summary
+table. Row statuses: ok (flashed and verified; under --scan-only: probe present —
+the scan checks presence only), flash-failed (firmware delivery failed: probe
+missing, build failed, flasher error, silent flash no-op, park not verified),
+failed (the check ran but did not verify: flashed with no enumeration/serial, or
+the check itself errored), locked (board flock held by another process —
+reported, never waited on or bypassed).
+
+Config is picked by hostname unless given: ci -> tinyusb.json, tusb (hifiphile
+rig) -> hfp.json, anything else is a dev PC -> local.json.
+
+Lives in test/hil/helper/ beside hil_lock.py; imports it and hil_flash; board
+recovery uses the repo's .claude/skills/usb-kernel-recover/scripts/usb_recover.sh.
+"""
+
+import argparse
+import io
+import json
+import glob
+import os
+import re
+import shlex
+import shutil
+import socket
+import sys
+import threading
+import time
+from concurrent.futures import ThreadPoolExecutor
+from pathlib import Path
+
+sys.path.insert(0, str(Path(__file__).resolve().parents[1])) # hil_flash + the helper package
+import hil_flash
+from helper import hil_lock, hil_util
+
+REPO_ROOT = hil_util.TINYUSB_ROOT
+USB_RECOVER = REPO_ROOT / '.claude' / 'skills' / 'usb-kernel-recover' / 'scripts' / 'usb_recover.sh'
+SEEN_CACHE = Path.home() / '.cache' / 'tinyusb-hil' / 'pool_seen.json'
+CONFIG_BY_HOST = {'ci': 'tinyusb.json', 'tusb': 'hfp.json'} # anything else: dev PC -> local.json
+
+# light-example preference; first built wins
+DEVICE_CANDIDATES = ['device/dfu_runtime', 'device/cdc_msc', 'device/cdc_msc_freertos',
+ 'device/hid_composite_freertos', 'device/cdc_dual_ports']
+HOST_CANDIDATES = ['host/device_info', 'host/cdc_msc_hid', 'host/msc_file_explorer_freertos']
+
+ENUM_WAIT = 12 # s, uid wait after flash
+ENUM_WAIT_RETRY = 8 # s, uid wait after a recovery reset/re-flash
+SERIAL_WAIT = 6 # s, host-board serial-output wait
+
+print_mutex = threading.Lock()
+_UNKNOWN_WARNED = False # scan_usb's caveat: once per process, not once per poll
+t0 = time.monotonic()
+
+
+def say(msg: str) -> None:
+ with print_mutex:
+ print(f'[{time.monotonic() - t0:6.1f}s] {msg}', file=sys.__stdout__, flush=True)
+
+
+def scan_usb() -> dict:
+ """busport -> {'serial', 'vidpid', 'ino'} for every enumerated USB device. Only
+ <bus>-<port>[.<port>...] dirs match; root hubs ('usbN', no dash) are excluded because
+ their 'serial' is a fabricated PCI address, and including them measured 6-7s/scan slower
+ (an observation; NOT an autosuspend wake -- that read is cached and does no I/O).
+ Keyed by busport, not serial: two devices can share a serial (an Espressif
+ USB-Serial-JTAG bridge and the cafe device it flashes both derive it from the same
+ MAC), and one dict slot would silently drop whichever lost the race."""
+ found = {}
+ # `unknown` matters BEFORE the blindness latch trips: one wedged device is the normal
+ # reason this tool is run, and its serial read stranding makes it absent from `devs`.
+ # Reported as fact, that is "probe MISSING" for hardware that is physically present.
+ devs, unknown = hil_util.usb_scan()
+ # ONCE per process: this is called from 0.5s poll loops across 4 worker threads and
+ # ~26 boards, so warning per call buried the table it exists to qualify under 600+
+ # identical lines. The memo in read_sysfs makes the condition sticky, so one line is
+ # as true as six hundred.
+ global _UNKNOWN_WARNED
+ if unknown and not _UNKNOWN_WARNED:
+ _UNKNOWN_WARNED = True
+ say('WARNING: at least one device did not answer a bounded read; rows below that '
+ 'say a probe or board is missing may be this scan losing sight of healthy '
+ 'hardware. Find the wedged device (usb-kernel-recover) and re-run.')
+ for dev in devs:
+ try:
+ found[dev['busport']] = {
+ 'serial': dev['serial'].lower(),
+ 'vidpid': f"{dev['vid']}:{dev['pid']}",
+ 'ino': os.stat(dev['dir'] + '/').st_ino}
+ except OSError:
+ continue
+ return found
+
+
+def find_usb(uid: str, devs: dict | None = None):
+ """Locate a flasher probe by uid, excluding VID cafe (TinyUSB DUT firmware): a
+ probe's uid can coincidentally equal its DUT's (Espressif USB-Serial-JTAG
+ bridges derive both from the same MAC), and the DUT is never the probe.
+
+ J-Link zero-pads numeric serials (681295394 -> 000681295394): an all-digit uid
+ matches an all-digit serial only when that serial equals the uid zero-padded to
+ the serial's own length (leading zeros only) — never when the zero-stripped uid
+ is empty, so a placeholder serial (metro_m4_express's probe legitimately reports
+ '123456') can't be mistaken for an unrelated device."""
+ devs = devs if devs is not None else scan_usb()
+ u = uid.lower()
+ candidates = [(bp, dev) for bp, dev in devs.items() if not dev['vidpid'].startswith('cafe:')]
+ for bp, dev in candidates:
+ if dev['serial'] == u:
+ return bp, dev['vidpid'], dev['ino']
+ stripped = u.lstrip('0')
+ if u.isdigit() and stripped:
+ for bp, dev in candidates:
+ s = dev['serial']
+ if s.isdigit() and s == stripped.zfill(len(s)):
+ return bp, dev['vidpid'], dev['ino']
+ return None
+
+
+def find_device(uid: str, pid: str | None):
+ """Board-online check: TinyUSB device (idVendor cafe) with this uid, optionally
+ PID-pinned. VID cafe keeps an Espressif USB-Serial-JTAG (303a) that shares the MAC
+ serial from false-passing."""
+ for busport, dev in scan_usb().items():
+ if (dev['serial'] == uid.lower() and dev['vidpid'].startswith('cafe:')
+ and (pid is None or dev['vidpid'].endswith(pid))):
+ return busport, dev['vidpid'], dev['ino']
+ return None
+
+
+def wait_device(uid: str, pid: str | None, old_ino, budget: float):
+ """Wait for the board's device with a NEW sysfs inode (flash resets the MCU, so a
+ genuine flash must re-enumerate; the inode is the re-enumeration marker)."""
+ deadline = time.monotonic() + budget
+ while time.monotonic() < deadline:
+ hit = find_device(uid, pid)
+ if hit and hit[2] != old_ino:
+ return hit
+ time.sleep(0.5)
+ return None
+
+
+def lock_board(name: str):
+ """Nonblocking flock per hil_lock.py protocol. Returns the handle, or a str with the
+ holder's info when the board is locked elsewhere. Board locks are ALWAYS respected: a
+ held board is reported and skipped, never waited on, and there is no bypass here."""
+ os.makedirs(hil_lock.BOARD_LOCK_DIR, exist_ok=True)
+ try:
+ fh = hil_lock.flock_nb(name)
+ except OSError:
+ # NB: conflates a held flock with open() failures (EACCES/EROFS/ENOSPC) — benign
+ # while everything on the rig runs as one uid
+ info = hil_lock.read_record(name)
+ return json.dumps(info) if info else 'unknown holder'
+ if not hil_lock.write_record(fh, 'pool_check'):
+ # an invisible lock (flock held, no record) is worse than no lock: status cannot
+ # show us and release cannot recognize the protected holder
+ hil_lock.clear_record(fh)
+ fh.close()
+ return 'ERROR: holder record write failed (lock dir unwritable?)'
+ return fh
+
+
+def unlock_board(fh) -> None:
+ hil_lock.clear_record(fh)
+ fh.close()
+
+
+def can_recover() -> bool:
+ if not USB_RECOVER.is_file():
+ return False
+ try:
+ # run_cmd, not subprocess.run: run's post-timeout reap is an UNBOUNDED wait(), and
+ # our kill bounces off a setuid-root sudo with EPERM, leaving communicate() on a
+ # pipe that never closes. run_cmd killpgs, escalates through sudo, reaps bounded.
+ r = hil_util.run_cmd('sudo -n true', timeout=10, quiet=True)
+ except OSError: # sudo not installed
+ return False
+ return r.returncode == 0
+
+
+def recover_probe(uid: str, busport: str) -> bool:
+ """Soft-replug an enumerated-but-wedged probe: deauthorize+reauthorize (no VBUS cut,
+ touches only this device). Success = the probe re-enumerated (new sysfs inode), not the
+ helper's exit code, which flakes while the toggle works. J-Links respond with a full
+ disconnect and can stay off the bus for >8 s."""
+ pre = find_usb(uid)
+ # Bounded through run_cmd (same reason as can_recover): the sysfs authorized store can
+ # block in D state on a wedged device, and this runs while the board's release-
+ # PROTECTED flock is held -- a hang here would lock the board until the host reboots.
+ cmd = ' '.join(shlex.quote(a) for a in
+ ['sudo', '-n', str(USB_RECOVER), 'authorized', busport])
+ if hil_util.run_cmd(cmd, timeout=30, quiet=True).returncode == 124:
+ return False
+ deadline = time.monotonic() + 20
+ while time.monotonic() < deadline:
+ post = find_usb(uid)
+ if post and (pre is None or post[2] != pre[2]):
+ return True
+ time.sleep(0.5)
+ return False
+
+
+def resolve_variant(board: dict, example: str, note: list | None = None) -> str:
+ """Build-dir variant name for `example`: the first of the board's variants with
+ already-built firmware, falling back to the board name. Notes the pick when it
+ differs from the board name (e.g. nanoch32v203's build dir is variant
+ 'nanoch32v203-fsdev', not the board name)."""
+ name = board['name']
+ for v in board.get('variant') or [{'name': name}]:
+ vn = v['name']
+ if hil_flash.find_firmware(vn, example, flasher=board['flasher']['name']):
+ if vn != name and note is not None and f'variant: {vn}' not in note:
+ note.append(f'variant: {vn}')
+ return vn
+ return name
+
+
+def pick_example(board: dict, note: list, build_missing: bool = True):
+ """(example, kind, variant, fw) with built firmware for this board; kind is
+ 'device' (uid check) or 'host' (serial-output check); variant is the resolved
+ build-dir variant that has it (see resolve_variant); fw is the firmware path to
+ flash, extension included. When nothing is built and build_missing is set (the default —
+ never skip a board for lack of a build), the preferred candidate is built on
+ the spot via ensure_fw."""
+ tests = board.get('tests', {})
+ only = tests.get('only', [])
+ skip = set(tests.get('skip', [])) # config's known-broken examples: never pick one
+ is_device = tests.get('device') or any(t.startswith('device/') for t in only)
+ if is_device:
+ cand = DEVICE_CANDIDATES + [t for t in only if t.startswith('device/') and t != 'device/usbtest']
+ kind = 'device'
+ else:
+ cand = HOST_CANDIDATES + [t for t in only if t.startswith('host/')]
+ kind = 'host'
+ for ex in dict.fromkeys(cand):
+ if ex in skip:
+ continue
+ variant = resolve_variant(board, ex, note)
+ fw = hil_flash.find_firmware(variant, ex, flasher=board['flasher']['name'])
+ if fw:
+ return ex, kind, variant, fw
+ if not build_missing:
+ return None, kind, None, None
+ # nothing built anywhere: build the preferred candidate (an only-list board
+ # must get one of its own examples — dfu_runtime etc. may not even configure)
+ pref = [c for c in dict.fromkeys(cand) if c not in skip and (not only or c in only)]
+ if not pref:
+ return None, kind, None, None
+ variant = (board.get('variant') or [{'name': board['name']}])[0]['name']
+ for ex in pref[:2]: # the second candidate covers a preferred example that fails to build
+ fw = ensure_fw(board, variant, ex, note)
+ if fw:
+ return ex, kind, variant, fw
+ return None, kind, None, None
+
+
+_pid_cache: dict[str, str | None] = {}
+
+
+def get_expected_pid(example: str) -> str | None:
+ """USB_PID for `example`'s device descriptor (examples/<example>/src/
+ usb_descriptors.c, '#define USB_PID 0x....'), lowercased and without the 0x
+ prefix to match sysfs idProduct. Cached per example; None (also cached) when
+ the file or define isn't there — host examples have no usb_descriptors.c, and
+ the caller must stay quiet rather than false-warn."""
+ if example not in _pid_cache:
+ pid = None
+ try:
+ text = (REPO_ROOT / 'examples' / example / 'src' / 'usb_descriptors.c').read_text()
+ # optional parens as in tools/check_example_pids.py's parser
+ m = re.search(r'#define\s+USB_PID\s+\(?\s*(0x[0-9a-fA-F]+)', text)
+ if m:
+ pid = m.group(1)[2:].lower()
+ except OSError:
+ pass
+ _pid_cache[example] = pid
+ return _pid_cache[example]
+
+
+def call_flasher(fn, *fn_args) -> tuple[int, str]:
+ """Run a hil_flash flash_*/reset_* backend, normalizing raises to a failure: several
+ backends raise instead of returning nonzero (get_serial_dev when a bridge's
+ /dev/serial/by-id node vanishes, a missing config.env, a .jlink script OSError), and an
+ exception must not skip the caller's retry/recovery ladder. Returns (rc, error line)."""
+ try:
+ ret = fn(*fn_args)
+ if ret.returncode == 0:
+ return 0, ''
+ err = flash_error_line(hil_util.cmd_stdout_text(ret.stdout))
+ return ret.returncode, err or f'rc={ret.returncode}'
+ except Exception as e:
+ return -1, repr(e)[:90]
+
+
+def flash(board: dict, fw, allow_recovery: bool, probe_port: str, note: list) -> bool:
+ """Flash the resolved firmware with one retry; on repeated failure soft-replug the
+ probe and always make one final attempt afterward, confirmed replug or not — some
+ probes (WCH-Link, ST-Link, CP210x, picoprobe) keep their sysfs kobject across an
+ authorized toggle instead of dropping off the bus. Returns True on success.
+
+ `fw` comes from pick_example: a re-resolve here would use the global search policy and
+ miss a firmware ensure_fw just built into cmake-build/ under an exclusive -B."""
+ fn = getattr(hil_flash, f'flash_{board["flasher"]["name"].lower()}')
+ for attempt in range(3):
+ if attempt == 2:
+ if not (allow_recovery and probe_port):
+ return False
+ cur = find_usb(board['flasher']['uid'])
+ if cur is None:
+ # probe gone from the bus: its old busport may now hold an UNRELATED device
+ # (bus renumbering) and the helper only checks occupancy, so toggling would
+ # deauthorize an innocent fixture
+ note.append('probe vanished before toggle')
+ else:
+ say(f'{board["name"]:26} recovery: replugging probe {cur[0]} (authorized toggle)')
+ if recover_probe(board['flasher']['uid'], cur[0]):
+ note.append('probe replugged')
+ time.sleep(2) # udev recreates /dev/serial/by-id symlinks after re-enumeration
+ else:
+ note.append('probe toggle unconfirmed')
+ rc, err = call_flasher(fn, board, str(fw))
+ if rc == 0:
+ return True
+ if rc == 127: # flasher binary missing: retries/probe recovery can't fix env
+ note.append(f'flasher tool missing ({err}) — esptool needs the ESP-IDF env (get-idf)'
+ if board['flasher']['name'].lower() == 'esptool' else
+ f'flasher tool missing: {err}')
+ return False
+ if attempt == 0:
+ say(f'{board["name"]:26} flash retry: {err}')
+ else:
+ note.append(f'flash: {err}')
+ return False
+
+
+def flash_error_line(out: str) -> str:
+ """Most informative line of a failed flash's output: last error-looking line,
+ else the last non-empty one."""
+ lines = [l.strip() for l in out.splitlines() if l.strip()]
+ for l in reversed(lines):
+ if any(k in l.lower() for k in ('error', 'fail', 'unknown', 'cannot', 'timeout',
+ 'no valid', 'not found', 'unable')):
+ return l[:90]
+ return lines[-1][:90] if lines else ''
+
+
+def check_host_serial(board: dict, do_reset: bool = True, want_hello: bool = False) -> bytes | None:
+ """Host-only boards never enumerate their uid (their USB port is the host side);
+ aliveness = output on the flasher's UART bridge after a reset. A probe byte is
+ written each poll so an echo-only firmware (board_test) also answers. Returns
+ the first output chunk (b'' when silent, None when the port is absent/drops) so
+ the caller can also judge WHAT answered — see boardtest_output().
+
+ do_reset=False listens to the firmware as-is: used right after a flash whose
+ own reset already started it — a second openocd/JLink session back-to-back on
+ the same probe can fail transiently and leave the target halted."""
+ import serial
+ try:
+ port = hil_util.get_serial_dev(board['flasher']['uid'], None, None, 0)
+ ser = serial.Serial(port, baudrate=115200, timeout=0.3, write_timeout=1)
+ except Exception as e:
+ say(f'{board["name"]:26} no flasher serial port: {e}')
+ return None
+ try:
+ # flush BEFORE the reset: this drops the pre-reset CDC backlog (which must not
+ # count as life) while keeping the post-reset boot banner, which prints while the
+ # reset tool is still tearing down and a post-reset flush would eat
+ ser.reset_input_buffer()
+ if do_reset:
+ getattr(hil_flash, f'reset_{board["flasher"]["name"].lower()}')(board)
+ # judge the WHOLE window, not the first chunk: the probe's CDC bridge has its own
+ # FIFO, so stale pre-flash output (e.g. board_test hellos) can arrive after our
+ # host-side flush and must not decide the verdict alone.
+ data = b''
+ deadline = time.monotonic() + SERIAL_WAIT
+ while time.monotonic() < deadline:
+ try:
+ ser.write(b'U')
+ data += ser.read(256)
+ except serial.SerialTimeoutException:
+ pass
+ except serial.SerialException:
+ return None # port dropped mid-poll (bridge re-enumerating)
+ # early-exit on the caller's positive signal (board_test hello for park
+ # verification, any non-board_test output for example liveness): stale
+ # bridge-FIFO backlog of the OTHER kind must not end the window
+ if want_hello:
+ if b'Hello from TinyUSB' in data:
+ return data
+ elif data and not boardtest_output(data):
+ return data
+ return data
+ finally:
+ ser.close()
+
+
+def boardtest_output(data: bytes) -> bool:
+ """True when (non-empty) serial output is recognizably ONLY board_test's: its
+ periodic HELLO_STR and echoes of our b'U' pokes, nothing else. Any residue
+ beyond that (an example banner, log lines) proves other firmware is talking,
+ however much stale board_test backlog surrounds it. Used as a negative
+ identity marker — after flashing a host example, board_test-only chatter
+ means the flash silently didn't take (the host analog of the PID check)."""
+ residue = data.replace(b'Hello from TinyUSB', b'')
+ for junk in (b'U', b'\r', b'\n'):
+ residue = residue.replace(junk, b'')
+ return len(residue) == 0
+
+
+def build_example(board: dict, variant: str, example: str) -> int:
+ """Build one example for this board: tools/build.py (same invocation shape as
+ hil_test.build_board), or idf.py directly for espressif (tools/build.py's esp branch
+ ignores -T and builds everything; variant flags travel as -DCFLAGS_CLI, the channel
+ tools/build.py uses). Bounded and process-group-killed via run_cmd; 600 s covers a
+ first configure+build of an SDK-heavy family (pico, nrf, esp). Builds normally run
+ pre-lock, so a board flock is not held here except on rare recovery paths. Per-build
+ compile parallelism is capped at cpu/-j so -j concurrent builds cannot swamp sibling
+ workers' verification windows. Returns the returncode (127 = ESP-IDF env missing)."""
+ name = board['name']
+ variants = board.get('variant') or [{'name': name}]
+ vcfg = next((v for v in variants if v['name'] == variant), variants[0])
+ if board['flasher']['name'].lower() == 'esptool':
+ if not shutil.which('idf.py'):
+ return 127 # ESP-IDF env not sourced in this shell
+ # -B keyed off the VARIANT so ensure_fw's post-build lookup finds it
+ cmd = ['idf.py', '-C', f'examples/{example}',
+ '-B', f'cmake-build/cmake-build-{vcfg["name"]}/{example}',
+ '-G', 'Ninja', f'-DBOARD={name}', 'build']
+ for d in board.get('build', {}).get('args', []) + vcfg.get('defines', []):
+ cmd.insert(-1, f'-D{d}')
+ if vcfg.get('flags'):
+ cmd.insert(-1, f'-DCFLAGS_CLI={vcfg["flags"]}')
+ # the IDF component manager writes examples/<ex>/dependencies.lock in the
+ # SOURCE tree (idf.py -B relocates only the build dir), so concurrent esp
+ # builds of one example for different targets corrupt each other's solve
+ with _esp_lock, _build_sem:
+ return hil_util.run_cmd(shlex.join(cmd), cwd=str(hil_util.TINYUSB_ROOT),
+ timeout=600).returncode
+ cmd = [sys.executable, str(hil_util.TINYUSB_ROOT / 'tools' / 'build.py'),
+ '-b', name, '-T', Path(example).name,
+ '-j', str(max(1, (os.cpu_count() or _jobs) // _jobs))]
+ for d in board.get('build', {}).get('args', []):
+ cmd += ['-D', d]
+ if vcfg['name'] != name:
+ cmd += ['--build-name', vcfg['name']]
+ for d in vcfg.get('defines', []):
+ cmd += ['-D', d]
+ for tok in vcfg.get('flags', '').split():
+ cmd += [f'--cflag={tok}']
+ with _build_sem:
+ return hil_util.run_cmd(shlex.join(cmd), cwd=str(hil_util.TINYUSB_ROOT),
+ timeout=600).returncode
+
+
+_deps_lock = threading.Lock() # one get_deps at a time (it also drains _build_sem)
+_esp_lock = threading.Lock() # idf.py mutates source-tree dependencies.lock per example
+_no_build = False # --no-build: ensure_fw never invokes a build
+_jobs = 4 # mirrors -j; set in main before the pool starts
+_build_sem = threading.BoundedSemaphore(4) # build slots; get_deps drains ALL (exclusive)
+_builds: dict = {} # (variant, example) -> (fw|None, reason): one attempt per run
+
+
+def ensure_fw(board: dict, variant: str, example: str, note: list):
+ """Firmware for `example`, building it when absent — never skip a board for lack of a
+ build (--no-build opts out). One retry with deps fetched and the CMake caches dropped
+ when the first build fails (fresh checkouts lack the family deps; a cache configured
+ in a broken env poisons every later attempt). Returns the firmware path, or None with
+ the failure noted. Call BEFORE taking the board lock: builds are long. One attempt per
+ (variant, example) per run, memoized in _builds, so a repeat call (park, under the
+ held flock) resolves instantly even when an exclusive -B hides the fresh artifact."""
+ fw = hil_flash.find_firmware(variant, example, flasher=board['flasher']['name'])
+ if fw:
+ return fw
+ key, base = (variant, example), Path(example).name
+ if key in _builds:
+ return _builds[key][0]
+ if _no_build:
+ _builds[key] = (None, 'disabled')
+ note.append(f'build skipped (--no-build): {base}')
+ return None
+ rc = build_example(board, variant, example)
+ if rc == 127 and board['flasher']['name'].lower() == 'esptool':
+ _builds[key] = (None, 'no-env')
+ note.append(f'cannot build {base}: ESP-IDF env missing (get-idf)')
+ return None
+ if rc == 124: # hung build: a deps/cache retry cannot cure it, don't double the stall
+ _builds[key] = (None, 'timeout')
+ note.append(f'build timeout: {base}')
+ return None
+ if rc != 0:
+ # retry once with deps fetched and the CMake caches dropped (cache only — a tree
+ # wipe would destroy every other example's firmware). get_deps git-resets shared
+ # deps that are already present, so it drains ALL build slots first.
+ with _deps_lock:
+ for _ in range(_jobs):
+ _build_sem.acquire()
+ try:
+ r = hil_util.run_cmd(shlex.join([sys.executable, str(hil_util.TINYUSB_ROOT / 'tools' / 'get_deps.py'),
+ '-b', board['name']]),
+ cwd=str(hil_util.TINYUSB_ROOT), timeout=600)
+ finally:
+ for _ in range(_jobs):
+ _build_sem.release()
+ if r.returncode != 0:
+ note.append('get_deps failed')
+ bd = hil_util.TINYUSB_ROOT / 'cmake-build' / f'cmake-build-{variant}'
+ # esp configures one level deeper (<variant>/<example>/): wipe both layouts
+ for d in (bd, bd / example):
+ shutil.rmtree(d / 'CMakeFiles', ignore_errors=True)
+ (d / 'CMakeCache.txt').unlink(missing_ok=True)
+ rc = build_example(board, variant, example)
+ if rc != 0:
+ _builds[key] = (None, 'fail')
+ note.append(f'build failed: {base}')
+ return None
+ # both build paths write to cmake-build/, so look there even when an explicit -B
+ # narrowed the global search — this is OUR fresh build, not a stale fallback
+ fw = hil_flash.find_firmware(variant, example,
+ roots=[hil_flash.build_dir, 'cmake-build'],
+ flasher=board['flasher']['name'])
+ _builds[key] = (fw, 'ok' if fw else 'no-fw')
+ note.append(f'built {base}' if fw else f'build produced no firmware: {base}')
+ return fw
+
+
+def ensure_board_test(board: dict, variant: str, note: list):
+ """board_test firmware for parking, building it if absent (via ensure_fw).
+ Espressif included — tools/build.py builds board_test for that family too;
+ the build just needs the ESP-IDF env (127 → noted, park is then skipped)."""
+ fw = hil_flash.find_firmware(variant, 'device/board_test', flasher=board['flasher']['name'])
+ if fw:
+ return fw
+ variants = board.get('variant') or [{'name': board['name']}]
+ if not any(v['name'] == variant for v in variants):
+ variant = variants[0]['name']
+ return ensure_fw(board, variant, 'device/board_test', note)
+
+
+def verdict(row: dict, ok: bool) -> str:
+ """Row status for a verification result, preserving a 'flash-failed' a deeper
+ layer already recorded (silent flash no-op, board_test delivery failure)."""
+ return 'ok' if ok else ('flash-failed' if row['status'] == 'flash-failed' else 'failed')
+
+
+def host_alive(board: dict, note: list, row: dict, flashed_example: bool = False) -> bool:
+ """Serial aliveness with recovery: silent -> (build and) flash board_test (it
+ hellos every second and echoes) -> recheck. Also cures a silent flash no-op
+ that left the board crashed.
+
+ With flashed_example=True (a host example was just flashed), board_test-shaped
+ output FAILS the check: the parked image still talking means the example flash
+ silently didn't take — the host analog of the device path's PID check.
+
+ Side effect: delivery-class failures (silent no-op, board_test build/flash
+ failure) set row['status'] = 'flash-failed' so verdict() preserves the cause;
+ the caller derives the final status from the return value via verdict()."""
+ data = check_host_serial(board)
+ if data:
+ if flashed_example and boardtest_output(data):
+ note.append('board_test output after example flash: silent flash no-op')
+ row['status'] = 'flash-failed'
+ return False
+ return True
+ variant = resolve_variant(board, 'device/board_test', note)
+ fw = ensure_board_test(board, variant, note)
+ if fw is None:
+ note.append('serial silent; board_test unavailable')
+ row['status'] = 'flash-failed'
+ return False
+ say(f'{board["name"]:26} recovery: serial silent, flashing board_test')
+ rc, err = call_flasher(getattr(hil_flash, f'flash_{board["flasher"]["name"].lower()}'), board, str(fw))
+ if rc != 0:
+ note.append(f'serial silent; board_test flash failed: {err}')
+ row['status'] = 'flash-failed'
+ return False
+ if not check_host_serial(board):
+ return False
+ if flashed_example:
+ # board_test talking proves the BOARD is alive, but the just-flashed
+ # example never produced serial — that verification still fails
+ note.append('example silent; board alive via board_test reflash')
+ return False
+ note.append('recovered via board_test reflash')
+ return True
+
+
+def device_recover_and_check(board: dict, example: str, variant: str, old_ino, note: list, row: dict, seen: dict) -> bool:
+ """Wait for the flashed board's uid to re-enumerate; on timeout, try one board
+ reset (skipped for flashers with no hardware reset — see hil_flash.RESET_NOOP,
+ it would just burn the wait) and wait again.
+
+ The PID policy is deliberately asymmetric. Pre-reset, the re-enumeration was
+ caused by the flash itself, so a PID mismatch most likely means the build dir
+ is stale (the flash DID write what find_firmware found) — warn, don't fail —
+ UNLESS the firmware was built this very run: then 'stale build' is impossible
+ and the mismatch can only be a silent flash no-op, which fails. Post-reset,
+ the re-enumeration proves nothing about the flash (the reset alone explains
+ it), so a mismatch is treated as a silent flash no-op and fails; an unknown
+ expected PID scores ok with a 'pid unverified' note in both paths."""
+ name = board['name']
+ expected_pid = get_expected_pid(example)
+ built_this_run = _builds.get((variant, example), (None, ''))[1] == 'ok'
+
+ def seen_hit(hit):
+ seen[board['uid']] = {'name': name, 'busport': hit[0], 'when': time.strftime('%Y-%m-%d %H:%M')}
+
+ hit = wait_device(board['uid'], None, old_ino, ENUM_WAIT)
+ if hit:
+ if expected_pid is not None and not hit[1].endswith(expected_pid):
+ if built_this_run:
+ row['device'] = f'❌ {hit[1]}'
+ note.append(f'pid {hit[1]}, this run built {expected_pid}: silent flash no-op')
+ row['status'] = 'flash-failed'
+ return False
+ note.append(f'⚠ pid {hit[1]}, source says {expected_pid}: stale build or silent flash no-op')
+ elif expected_pid is None:
+ note.append('pid unverified')
+ row['device'] = f'✅ {hit[1]}'
+ seen_hit(hit)
+ return True
+
+ flasher_name = board['flasher']['name'].lower()
+ if flasher_name in hil_flash.RESET_NOOP:
+ note.append(f'no hardware reset available for {flasher_name}')
+ row['device'] = '❌ not enumerated'
+ return False
+
+ say(f'{name:26} recovery: uid not up, resetting board')
+ rc, err = call_flasher(getattr(hil_flash, f'reset_{flasher_name}'), board)
+ if rc != 0:
+ note.append(f'reset failed: {err}')
+ hit = wait_device(board['uid'], None, old_ino, ENUM_WAIT_RETRY)
+ if not hit:
+ row['device'] = '❌ not enumerated'
+ note.append('reset did not help')
+ return False
+ if expected_pid is None:
+ row['device'] = f'✅ {hit[1]}'
+ note.append('reset recovered (pid unverified)')
+ seen_hit(hit)
+ return True
+ if hit[1].endswith(expected_pid):
+ row['device'] = f'✅ {hit[1]}'
+ note.append('reset recovered')
+ seen_hit(hit)
+ return True
+ row['device'] = f'❌ {hit[1]}'
+ note.append(f'reset recovered wrong pid, expected {expected_pid}: silent flash no-op')
+ row['status'] = 'flash-failed'
+ return False
+
+
+def check_board(board: dict, args, allow_recovery: bool, seen: dict) -> dict:
+ name = board['name']
+ row = {'name': name, 'probe': '❌ missing', 'flash': '–', 'device': '–', 'note': [], 'status': 'failed'}
+ note = row['note']
+
+ probe = find_usb(board['flasher']['uid'])
+ if probe:
+ row['probe'] = f'✅ {probe[0]}'
+ seen[board['flasher']['uid']] = {'name': f'{name} probe', 'busport': probe[0],
+ 'when': time.strftime('%Y-%m-%d %H:%M')}
+ else:
+ last = seen.get(board['flasher']['uid'])
+ note.append(f'probe last seen {last["busport"]} {last["when"]}' if last
+ else 'probe never seen by pool_check')
+ say(f'{name:26} probe MISSING ({board["flasher"]["name"]} {board["flasher"]["uid"]})')
+
+ # existing firmware only; a missing build is built further down (after a lock peek),
+ # except in scan/no-build modes and never for a missing probe
+ example, kind, variant, fw = pick_example(board, note, build_missing=False)
+ if kind == 'host':
+ note.append('host-only board')
+
+ if args.scan_only:
+ hit = find_device(board['uid'], None)
+ # report the BOARD's usb state too: enumerated (with busport), off-bus (normal
+ # when parked in board_test), or n/a for host-only boards
+ if hit:
+ row['device'] = f'✅ {hit[1]} @{hit[0]}'
+ elif kind == 'host':
+ row['device'] = '– n/a (host-only)'
+ else:
+ row['device'] = '⚫ off bus (parked?)'
+ # scan verifies probe presence only, so probe present is ok; a missing probe means
+ # no firmware could be delivered → flash-failed
+ row['status'] = 'ok' if probe else 'flash-failed'
+ if probe:
+ say(f'{name:26} probe ✅ {probe[0]}' + (f' device {hit[1]}' if hit else ''))
+ return row
+ if not probe:
+ row['status'] = 'flash-failed'
+ return row
+
+ bt_variant = resolve_variant(board, 'device/board_test', note)
+ need_example = example is None and not args.no_build
+ # board_test is also host_alive's recovery image, so host boards pre-build it
+ # even under --no-park; --no-build gates EVERY build, board_test included
+ need_bt = (not args.no_build
+ and (not args.no_park or kind == 'host')
+ and hil_flash.find_firmware(bt_variant, 'device/board_test',
+ flasher=board['flasher']['name']) is None)
+ if need_example or need_bt:
+ # builds are long and run BEFORE locking (park must never hold the flock through
+ # one); peek first so minutes of building are not wasted on — or a rebuilt tree
+ # swapped under — a board CI holds right now
+ peek = lock_board(name)
+ if isinstance(peek, str):
+ if peek.startswith('ERROR:'): # environment failure, not a held lock
+ row['flash'] = '❌ lock'
+ row['status'] = 'failed'
+ else:
+ row['flash'] = '🔒 locked'
+ row['status'] = 'locked'
+ note.append(peek)
+ say(f'{name:26} locked: {peek}')
+ return row
+ unlock_board(peek)
+ if need_example:
+ example, kind, variant, fw = pick_example(board, note, build_missing=True)
+ if need_bt and (example is not None or kind == 'host'):
+ # skip the park build when the example build already failed on a device board:
+ # the row returns before any flash/park could use it
+ ensure_board_test(board, bt_variant, note)
+
+ if example is None:
+ if not any(n.startswith(('build failed', 'build timeout', 'build produced',
+ 'build skipped', 'cannot build')) for n in note):
+ note.append('no firmware built')
+ if kind != 'host':
+ row['status'] = 'flash-failed'
+ say(f'{name:26} probe ✅ {probe[0]} (no firmware to flash)')
+ return row
+ # host-only board: aliveness is still checkable without flashing — reset and listen
+ # to whatever is on it (parked board_test echoes and hellos on the flasher UART)
+
+ lk = lock_board(name)
+ if isinstance(lk, str):
+ if lk.startswith('ERROR:'): # environment failure, not a held lock
+ row['flash'] = '❌ lock'
+ row['status'] = 'failed'
+ else:
+ row['flash'] = '🔒 locked'
+ row['status'] = 'locked'
+ note.append(lk)
+ say(f'{name:26} locked: {lk}')
+ return row
+ try:
+ if example is None: # host-only without firmware: UART-only aliveness check
+ ok = host_alive(board, note, row)
+ row['device'] = '✅ serial out' if ok else '❌ no serial out'
+ row['status'] = verdict(row, ok)
+ say(f'{name:26} – {row["device"]} (existing firmware)')
+ return row
+
+ pre = find_device(board['uid'], None)
+ old_ino = pre[2] if pre else None
+
+ try:
+ if not flash(board, fw, allow_recovery, probe[0], note):
+ row['flash'] = f'❌ {Path(example).name}'
+ row['status'] = 'flash-failed'
+ say(f'{name:26} flash FAILED ({example})')
+ return row
+ row['flash'] = f'✅ {Path(example).name}'
+
+ if kind == 'host':
+ ok = host_alive(board, note, row, flashed_example=True)
+ row['device'] = '✅ serial out' if ok else '❌ no serial out'
+ else:
+ ok = device_recover_and_check(board, example, variant, old_ino, note, row, seen)
+ row['status'] = verdict(row, ok)
+ say(f'{name:26} {row["flash"]} {row["device"]}')
+ return row
+ finally:
+ # teardown for EVERY path that attempted a flash (a failed programmer op can
+ # still have erased/half-written the target), while the lock is still held
+ if not args.no_park:
+ park_board(board, kind, row, note)
+ finally:
+ unlock_board(lk)
+
+
+def park_board(board: dict, kind: str, row: dict, note: list) -> None:
+ """Re-park with board_test, building it if absent (ensure_board_test), and
+ VERIFY it took: board_test never enumerates USB, so a device board's cafe
+ device must drop off the bus, and a host board must answer with board_test's
+ own output — a rc=0 park that changed nothing (silent no-op) must not pass.
+ A board left unparked marks an ok row flash-failed (never downgrading a
+ 'failed' verify verdict — that is the more diagnostic signal), with one
+ exception: an espressif board without the ESP-IDF env cannot build
+ board_test — noted, not a board fault."""
+ # capture BEFORE the park flash: uid-disappearance only verifies the park if the
+ # device was on the bus to begin with
+ on_bus_before = kind != 'host' and find_device(board['uid'], None) is not None
+ variant = resolve_variant(board, 'device/board_test', note)
+ fw = ensure_board_test(board, variant, note)
+ if fw is None:
+ if any(n.startswith('cannot build board_test') for n in note):
+ note.append('park skipped (no ESP-IDF env)')
+ else:
+ # --no-build disables builds, not parking (--no-park is that opt-out):
+ # a board left running a USB-active image is unparked either way
+ note.append('unparked: board_test not built (--no-build)'
+ if any(n.startswith('build skipped (--no-build): board_test') for n in note)
+ else 'unparked: board_test unavailable (build failed/timed out)')
+ if row['status'] == 'ok':
+ row['status'] = 'flash-failed'
+ return
+ rc, err = call_flasher(getattr(hil_flash, f'flash_{board["flasher"]["name"].lower()}'),
+ board, str(fw))
+ if rc != 0:
+ note.append(f'park flash failed: {err}')
+ if row['status'] == 'ok':
+ row['status'] = 'flash-failed'
+ return
+ if kind == 'host':
+ # no second reset (the park flash's own reset started board_test); POSITIVE
+ # marker: its hello must appear, and stale bridge-FIFO output alongside it is not
+ # disqualifying
+ data = check_host_serial(board, do_reset=False, want_hello=True)
+ if not (data and b'Hello from TinyUSB' in data):
+ note.append('park unverified: no board_test output')
+ if row['status'] == 'ok':
+ row['status'] = 'flash-failed'
+ return
+ if not on_bus_before:
+ # never enumerated this run: uid-disappearance cannot tell a verified park from a
+ # silent no-op — say so instead of passing vacuously
+ note.append('park unverified (device already off bus)')
+ return
+ deadline = time.monotonic() + 6
+ while time.monotonic() < deadline:
+ if find_device(board['uid'], None) is None:
+ return
+ time.sleep(0.5)
+ note.append('park unverified: device still enumerated')
+ if row['status'] == 'ok':
+ row['status'] = 'flash-failed'
+
+
+def check_board_safe(board: dict, args, allow_recovery: bool, seen: dict) -> dict:
+ """Isolate one board's exceptions: a crashing worker must not discard every
+ other board's row, the table, the topology, and the seen-cache write."""
+ try:
+ return check_board(board, args, allow_recovery, seen)
+ except Exception as e:
+ name = board.get('name', '?')
+ say(f'{name:26} INTERNAL ERROR: {e!r}')
+ return {'name': name, 'probe': '–', 'flash': '–', 'device': '❌ error',
+ 'note': [repr(e)[:120]], 'status': 'failed'}
+
+
+def controller_summary() -> list[str]:
+ """USB topology: controller (PCI addr, vendor) -> bus -> root-port subtree device
+ counts (hubs included, interfaces/root hubs not). Bus numbers renumber every boot;
+ PCI addresses and root-port numbers are stable."""
+ vendor_names = {'0x1022': 'AMD', '0x1912': 'Renesas', '0x8086': 'Intel', '0x1b21': 'ASMedia'}
+ ctrl = {}
+ for root in glob.glob('/sys/bus/usb/devices/usb*'):
+ bus = int(os.path.basename(root)[3:])
+ m = re.findall(r'[0-9a-f]{4}:[0-9a-f]{2}:[0-9a-f]{2}\.[0-9a-f]', os.path.realpath(root))
+ pci = m[-1] if m else '?'
+ c = ctrl.setdefault(pci, {'vendor': '?', 'buses': {}})
+ subtrees = {}
+ for d in glob.glob(f'/sys/bus/usb/devices/{bus}-*'):
+ b = os.path.basename(d)
+ if ':' in b:
+ continue
+ subtrees[b.split('.')[0]] = subtrees.get(b.split('.')[0], 0) + 1
+ c['buses'][bus] = subtrees
+ try:
+ vid = open(f'/sys/bus/pci/devices/{pci}/vendor').read().strip()
+ c['vendor'] = vendor_names.get(vid, vid)
+ except OSError:
+ pass
+
+ lines = []
+ for pci, c in sorted(ctrl.items()):
+ lines.append(f'{pci} ({c["vendor"]})')
+ for bus, subtrees in sorted(c['buses'].items()):
+ detail = ' '.join(f'{k}: {n} dev' for k, n in
+ sorted(subtrees.items(), key=lambda i: int(i[0].split('-')[1])))
+ lines.append(f' bus {bus}: {sum(subtrees.values())} devices'
+ + (f' {detail}' if detail else ''))
+ return lines
+
+
+def main() -> None:
+ # toolchain/flasher CLIs live in the user bin dirs, which non-login shells may lack --
+ # the same PATH shim hil_ci.sh applies on the remote side
+ for d in (Path.home() / 'bin', Path.home() / '.local' / 'bin'):
+ if d.is_dir() and str(d) not in os.environ.get('PATH', '').split(os.pathsep):
+ os.environ['PATH'] = f'{d}{os.pathsep}{os.environ.get("PATH", "")}'
+
+ parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
+ parser.add_argument('config', nargs='?', help='HIL config json (default: by hostname)')
+ parser.add_argument('-b', '--board', action='append', default=[], help='only these boards')
+ parser.add_argument('-B', '--build-dir', default=None,
+ help='firmware parent dir, searched EXCLUSIVELY when given '
+ '(default: examples, plus cmake-build as fallback)')
+ parser.add_argument('--scan-only', action='store_true',
+ help='USB presence scan only: no locks, no flashing')
+ parser.add_argument('--no-build', action='store_true',
+ help='do not build missing firmware (default: build the light example on the spot)')
+ parser.add_argument('--no-park', action='store_true',
+ help='leave the light example running (default: park with board_test)')
+ # no cross-process flash budget against a concurrent hil_test.py run (its semaphores
+ # are in-process), so keep this modest
+ parser.add_argument('-j', '--jobs', type=int, default=4)
+ parser.add_argument('-v', '--verbose', action='store_true')
+ args = parser.parse_args()
+ global _no_build, _jobs, _build_sem
+ _no_build = args.no_build
+ _jobs = max(1, args.jobs)
+ _build_sem = threading.BoundedSemaphore(_jobs)
+
+ host = socket.gethostname()
+ cfg_name = args.config or CONFIG_BY_HOST.get(host, 'local.json')
+ cfg_path = Path(cfg_name)
+ if not cfg_path.exists():
+ cfg_path = REPO_ROOT / 'test' / 'hil' / cfg_name
+ if not cfg_path.exists():
+ sys.exit(f'config not found: {cfg_name} (host {host}; dev PCs need test/hil/local.json)')
+ with cfg_path.open() as f:
+ config = json.load(f)
+
+ boards = list(config['boards']) # boards-skip (parked hardware) is not scanned by default
+ if args.board:
+ boards += config.get('boards-skip', []) # explicitly named parked boards are fair game
+ unknown = set(args.board) - {b['name'] for b in boards}
+ if unknown:
+ sys.exit(f'board(s) not in {cfg_path.name}: {", ".join(sorted(unknown))}')
+ boards = [b for b in boards if b['name'] in args.board]
+
+ hil_flash.build_dir = args.build_dir or 'examples'
+ hil_util.verbose = args.verbose
+ if args.build_dir is None:
+ # default mode: search both standard layouts (cmake-build/ from tools/build.py and
+ # ESP-IDF, examples/ from manual builds). An EXPLICIT -B stays exclusive: the caller
+ # named an artifact tree, so a miss must report rather than flash an older build.
+ hil_flash.EXTRA_BUILD_DIRS = ['cmake-build', 'examples']
+ allow_recovery = not args.scan_only and can_recover()
+ seen = {}
+ try:
+ loaded = json.loads(SEEN_CACHE.read_text())
+ if isinstance(loaded, dict): # tolerate a torn/hand-edited cache
+ seen = {k: v for k, v in loaded.items() if isinstance(v, dict)}
+ except (OSError, ValueError):
+ pass
+
+ roots = ' + '.join(dict.fromkeys([hil_flash.build_dir, *hil_flash.EXTRA_BUILD_DIRS]))
+ say(f'pool check: host {host}, config {cfg_path.name}, {len(boards)} boards, '
+ f'{"scan-only" if args.scan_only else f"flash via {{{roots}}}/cmake-build-<board>"}'
+ f'{"" if allow_recovery or args.scan_only else ", recovery unavailable (no sudo -n / usb_recover.sh)"}')
+
+ if args.verbose:
+ rows = [check_board_safe(b, args, allow_recovery, seen) for b in boards]
+ else:
+ with io.StringIO() as spool, ThreadPoolExecutor(max_workers=args.jobs) as pool:
+ sys.stdout = spool # silence hil_util.run_cmd's COMMAND FAILED dumps; say() uses __stdout__
+ try:
+ rows = list(pool.map(lambda b: check_board_safe(b, args, allow_recovery, seen), boards))
+ finally:
+ sys.stdout = sys.__stdout__
+
+ try:
+ SEEN_CACHE.parent.mkdir(parents=True, exist_ok=True)
+ tmp = SEEN_CACHE.with_suffix('.json.tmp')
+ tmp.write_text(json.dumps(seen, indent=1, sort_keys=True) + '\n')
+ tmp.replace(SEEN_CACHE) # atomic: a killed run can't tear the cache
+ except OSError:
+ pass
+
+ status_mark = {'ok': '✅ ok', 'flash-failed': '❌ flash-failed', 'failed': '❌ failed',
+ 'locked': '🔒 locked'}
+ headers = ['Board', 'Probe', 'Flash', 'Device', 'Status', 'Note']
+ cells = [[r['name'], r['probe'], r['flash'], r['device'],
+ status_mark.get(r['status'], r['status']), '; '.join(r['note'])] for r in rows]
+ widths = [max(len(h), *(len(c[i]) for c in cells)) if cells else len(h)
+ for i, h in enumerate(headers)]
+ line = lambda vals: '| ' + ' | '.join(v.ljust(w) for v, w in zip(vals, widths)) + ' |'
+ print()
+ print(line(headers))
+ print('|' + '|'.join('-' * (w + 2) for w in widths) + '|')
+ for c in cells:
+ print(line(c))
+
+ print('\nUSB topology (controller → root-port subtree):')
+ for line in controller_summary():
+ print(f' {line}')
+
+ counts = {'ok': 0, 'flash-failed': 0, 'failed': 0, 'locked': 0}
+ for r in rows:
+ counts[r.get('status', 'failed')] += 1
+ print(f'\n{counts["ok"]} ok · {counts["flash-failed"]} flash-failed · {counts["failed"]} failed '
+ f'· {counts["locked"]} locked · in {time.monotonic() - t0:.0f}s')
+ if hil_util.sysfs_blind():
+ # Without this the table is the worst kind of wrong: once the process latches
+ # blind, every read answers SYSFS_UNKNOWN, scan_usb() returns {}, and EVERY board
+ # prints "probe MISSING"/"off bus" -- a clean-looking report declaring the whole
+ # fleet dead, produced during exactly the incident this tool is run to diagnose,
+ # and it sends the operator to power-cycle a rig where one device is wedged.
+ print('WARNING: this scan lost sight of the bus'
+ f'{hil_util.sysfs_blind_note()}. Rows above that say a probe or board is '
+ f'missing may be this tool losing sight of healthy hardware, not absent '
+ f'hardware. Find the wedged device (see the usb-kernel-recover skill) and '
+ f're-run before acting on the table.')
+ sys.exit(min(counts['flash-failed'] + counts['failed'], 125))
+
+
+if __name__ == '__main__':
+ main()
diff --git a/test/hil/helper/hil_select.py b/test/hil/helper/hil_select.py
new file mode 100755
index 000000000..f0d4f0b9f
--- /dev/null
+++ b/test/hil/helper/hil_select.py
@@ -0,0 +1,524 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: MIT
+"""PR-diff -> HIL selection: which rig boards and which tests a change can affect.
+
+Stdlib-only (runs on bare CI runners; imports hil_util for the example rosters,
+never hil_test/pyserial — test_hil_util.BottomLayer enforces the stdlib closure).
+Fail-open: any file no rule classifies forces the full matrix. See
+docs/superpowers/specs/2026-07-29-hil-pr-scoped-selection-design.md.
+
+JSON: full, boards (name -> 'all' | [tests]), families (bsp families the diff
+touches, including ones with no rig board - build-only consumers such as /pre-pr
+sample from these), args (hil_test.py args per config) and args_flasher (the same
+args split by each board's flasher, for CI legs that split one rig by flasher).
+"""
+import argparse
+import functools
+import glob
+import json
+import os
+import re
+import subprocess
+import sys
+
+sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # helper/ scripts import via the test/hil root
+from helper.hil_util import device_tests, dual_tests, host_test
+
+ALL_TESTS = {'device': device_tests, 'dual': dual_tests, 'host': host_test}
+
+# class dir -> config macro suffix exceptions (rule 3); dfu is per-file, handled inline
+NET_MACROS = ('ECM_RNDIS', 'NCM')
+
+_NONCODE_RE = re.compile(
+ r'^(docs/|\.claude/|.*\.(md|rst)$|LICENSE)')
+_FULL_RE = re.compile(
+ r'^(src/common/|src/osal/|src/tusb\.c$|src/tusb\.h$|src/tusb_option\.h$|'
+ r'test/hil/|\.github/workflows/build.*\.yml$|\.github/actions/|\.github/scripts/|'
+ r'tools/build\.py$|tools/get_deps\.py$|tools/cmake/|hw/mcu/|lib/|'
+ r'hw/bsp/(family_support\.cmake|board_api\.h|board\.c|ansi_escape\.h)$|'
+ r'examples/build_system/|examples/CMakeLists\.txt$|'
+ # board_test is HIL infrastructure, not a test: hil_test.py flashes it to park
+ # every board (variant boundary + end-of-board teardown), so every board depends on it
+ r'examples/device/board_test/)')
+
+# --no-renames: with rename detection git reports only a rename's destination, so code
+# moved out of an HIL-relevant path would be classified by its new path alone
+GIT_DIFF_ARGV = ['git', 'diff', '--no-renames', '--name-only']
+
+
+def test_role(test: str) -> str:
+ return test.split('/', 1)[0] # 'device' | 'dual' | 'host'
+
+
+def board_roles(board: dict) -> set:
+ t = board.get('tests', {})
+ roles = set()
+ if t.get('device'):
+ roles.add('device')
+ if t.get('host'):
+ roles.add('host')
+ if t.get('dual'):
+ roles.update(('device', 'host'))
+ for only in t.get('only', []):
+ r = test_role(only)
+ roles.update(('device', 'host') if r == 'dual' else (r,))
+ return roles
+
+
+def board_tests(board: dict) -> list:
+ """Every test this board would run today (mirrors hil_test.test_board's default)."""
+ t = board.get('tests', {})
+ if 'only' in t:
+ run = list(t['only'])
+ else:
+ run = []
+ if t.get('device'):
+ run += device_tests
+ if t.get('dual'):
+ run += dual_tests
+ if t.get('host'):
+ run += host_test
+ return [x for x in run if x not in t.get('skip', [])]
+
+
+# cached: called per changed file x roster board, and the tree doesn't change mid-run
[email protected]_cache(maxsize=None)
+def board_family(board_name: str, repo_root: str):
+ hits = glob.glob(os.path.join(repo_root, 'hw/bsp/*/boards', board_name))
+ return os.path.basename(os.path.dirname(os.path.dirname(hits[0]))) if hits else None
+
+
+# `if (OPTION STREQUAL "1")` guards in family_support.cmake, and the option tokens
+# a roster entry passes to the build (NAME=VALUE / -DNAME=VALUE)
+_CM_IF_RE = re.compile(r'if\s*\(')
+_CM_ELSE_RE = re.compile(r'else(if)?\s*\(')
+_CM_ENDIF_RE = re.compile(r'endif\s*\(')
+_CM_OPT_RE = re.compile(r'if\s*\(\s*\$?\{?([A-Za-z_]\w*)\}?\s+STREQUAL\s+"?1"?\s*\)')
+_CM_PORT_RE = re.compile(r'src/portable/((?:[^/\s]+/)?[^/\s]+)/')
+_FALSY = ('', '0', 'off', 'false', 'no')
+
+
[email protected]_cache(maxsize=None)
+def port_option_gates(repo_root: str) -> dict:
+ """port dir -> build options that compile it regardless of the board's family
+ file, e.g. {'analog/max3421': {'MAX3421_HOST'}} from family_support.cmake."""
+ gates = {}
+ try:
+ text = open(os.path.join(repo_root, 'hw/bsp/family_support.cmake')).read()
+ except OSError:
+ return gates
+ stack = [] # one entry per open if(): its option, or None
+ for line in text.splitlines():
+ line = line.strip()
+ if _CM_IF_RE.match(line):
+ m = _CM_OPT_RE.match(line)
+ stack.append(m.group(1) if m else None)
+ elif _CM_ELSE_RE.match(line):
+ if stack:
+ stack[-1] = None # the guard doesn't hold in this branch
+ elif _CM_ENDIF_RE.match(line):
+ if stack:
+ stack.pop()
+ opts = {o for o in stack if o}
+ m = _CM_PORT_RE.search(line)
+ if opts and m:
+ gates.setdefault(m.group(1), set()).update(opts)
+ return gates
+
+
+_CM_SET_RE = re.compile(r'set\s*\(\s*([A-Za-z_]\w*)\s+([^)\s]+)\s*\)')
+
+
+# cached: called per changed portable file x roster board
[email protected]_cache(maxsize=None)
+def bsp_board_options(board_name: str, repo_root: str) -> frozenset:
+ """Build options a board turns on in its own BSP: `set(<OPT> <value>)` in
+ hw/bsp/<family>/boards/<board>/board.cmake, e.g. MAX3421_HOST on the espressif
+ and rp2040 max3421 boards. CMake only - HIL CI builds nothing with Make, so a
+ board.mk-only option (e.g. nrf5340dk's MAX3421_HOST) compiles no port here."""
+ fam = board_family(board_name, repo_root)
+ if not fam:
+ return frozenset()
+ path = os.path.join(repo_root, 'hw/bsp', fam, 'boards', board_name, 'board.cmake')
+ try:
+ text = open(path).read()
+ except OSError:
+ return frozenset()
+ out = set()
+ for line in text.splitlines():
+ line = line.strip()
+ if line.startswith('#'):
+ continue
+ m = _CM_SET_RE.match(line)
+ if m and m.group(2).strip('"').lower() not in _FALSY:
+ out.add(m.group(1))
+ return frozenset(out)
+
+
+def board_options(board: dict, repo_root: str) -> set:
+ """Build options a board has truthy: the roster entry's build.args plus each
+ variant's defines (NAME=VALUE) and raw CFLAGS (-DNAME=VALUE), plus whatever its
+ own board.cmake sets (a board can enable a gated port without the roster saying so)."""
+ toks = list(board.get('build', {}).get('args', []))
+ for v in board.get('variant', []):
+ toks += list(v.get('defines', []))
+ toks += v.get('flags', '').split()
+ out = set(bsp_board_options(board['name'], repo_root))
+ for t in toks:
+ name, _, val = (t[2:] if t.startswith('-D') else t).partition('=')
+ if name and val.strip().strip('"').lower() not in _FALSY:
+ out.add(name.strip())
+ return out
+
+
[email protected]_cache(maxsize=None)
+def port_families(port_dir: str, repo_root: str) -> set:
+ """Board families that compile this src/portable dir. CMake only: HIL CI builds
+ every board with CMake, so a port wired up in family.mk alone is compiled for no
+ HIL board and must not select one. family.cmake lists portable sources directly
+ for most families; espressif instead references them from a nested component
+ CMakeLists.txt (hw/bsp/espressif/components/tinyusb_src/CMakeLists.txt)."""
+ fams = set()
+ bsp_root = os.path.join(repo_root, 'hw/bsp')
+ # trailing '/' so a port dir is not a prefix of a sibling: bare 'microchip/pic'
+ # would otherwise match '.../microchip/pic32mz/...' and inherit its families
+ needle = port_dir + '/'
+ for f in glob.glob(os.path.join(bsp_root, '*/family.cmake')) + \
+ glob.glob(os.path.join(bsp_root, '*/components/*/CMakeLists.txt')):
+ try:
+ if needle in open(f).read():
+ fam = os.path.relpath(f, bsp_root).split(os.sep, 1)[0]
+ fams.add(fam)
+ except OSError:
+ pass
+ return fams
+
+
+_CLS_INC_RE = re.compile(r'#\s*include\s*[<"]class/([^/"<>]+)/([^"<>]+)[">]')
+
+
[email protected]_cache(maxsize=None)
+def class_include_edges(repo_root: str) -> dict:
+ """'<class>/<header>' -> the other class dirs that include it. A class header
+ pulled in by a second class ships in every firmware enabling that second class:
+ src/class/midi/midi{,2}_{device,host}.h include class/audio/audio.h, and
+ net_device.h includes class/cdc/cdc.h. The class rule derives macros from the
+ directory name alone, so without this edge a change to the included header
+ selects only its own class's examples - and on a board that skips those (e.g.
+ metro_m4_express skips audio_test_freertos), nothing at all.
+
+ Derived from the actual #include lines rather than a hand-written table so it
+ cannot rot when a class picks up or drops a cross-class include."""
+ edges = {}
+ for f in sorted(glob.glob(os.path.join(repo_root, 'src/class/*/*.[ch]'))):
+ cls = os.path.basename(os.path.dirname(f))
+ try:
+ text = open(f).read()
+ except OSError:
+ continue
+ for inc_cls, inc_hdr in _CLS_INC_RE.findall(text):
+ if inc_cls != cls:
+ edges.setdefault(f'{inc_cls}/{inc_hdr}', set()).add(cls)
+ return edges
+
+
+def class_macros(cls: str, base: str, prefix: str) -> list:
+ """Config macros that compile a class dir's code, for role prefix TUD/TUH.
+ `base` refines dfu only (it splits DFU from DFU_RUNTIME per file); pass '' for
+ a class reached through an include edge, where the widest set is correct."""
+ if cls == 'net':
+ return [f'CFG_{prefix}_{m}' for m in NET_MACROS]
+ if cls == 'dfu':
+ if base.startswith('dfu_rt'):
+ return [f'CFG_{prefix}_DFU_RUNTIME']
+ if base.startswith('dfu_device') or base.startswith('dfu_host'):
+ return [f'CFG_{prefix}_DFU']
+ return [f'CFG_{prefix}_DFU', f'CFG_{prefix}_DFU_RUNTIME']
+ return [f'CFG_{prefix}_{cls.upper()}']
+
+
+def _config_enables(cfg_path: str, macros) -> bool:
+ try:
+ text = open(cfg_path).read()
+ except OSError:
+ return False
+ return any(re.search(rf'#define\s+{m}\s+\(?\s*0*[1-9]', text) for m in macros)
+
+
+def roster_only_tests(all_boards) -> set:
+ """Test paths that only appear in a roster board's tests.only list (e.g.
+ espressif boards), not in the shared device/dual/host_test lists."""
+ out = set()
+ for b in all_boards:
+ out.update(b.get('tests', {}).get('only', []))
+ return out
+
+
+def class_examples(macros, role: str, repo_root: str, extra_tests: set) -> set:
+ """Tests (from role's + dual lists, plus roster-only-list tests of that role)
+ whose example config enables any macro."""
+ pool = role_tests({role}, extra_tests)
+ out = set()
+ for test in pool:
+ cfg = os.path.join(repo_root, 'examples', test, 'src', 'tusb_config.h')
+ if _config_enables(cfg, macros):
+ out.add(test)
+ return out
+
+
+def role_tests(roles: set, extras: set) -> set:
+ """Every test for the given role(s): each role's own list + dual tests,
+ plus roster-only-list tests (extras) matching those roles or 'dual'."""
+ pool = set(dual_tests)
+ for r in roles:
+ pool |= set(ALL_TESTS[r])
+ pool |= {t for t in extras if test_role(t) in roles or test_role(t) == 'dual'}
+ return pool
+
+
+class _Sel:
+ """Accumulates contributions. board->set(tests) plus 'all-board' markers."""
+ def __init__(self):
+ self.full = False
+ self.by_board = {} # name -> set of tests, or 'all'
+ self.roles = set() # roles touched by any contribution
+ self.families = set() # bsp families touched (incl. off-rig ones: build-only consumers)
+ self.reasons = []
+
+ def add(self, boards, tests, reason):
+ """tests: 'all' or iterable of test paths."""
+ self.reasons.append(reason)
+ for b in boards:
+ cur = self.by_board.get(b)
+ if tests == 'all' or cur == 'all':
+ self.by_board[b] = 'all'
+ else:
+ self.by_board[b] = (cur or set()) | set(tests)
+
+ def force_full(self, reason):
+ self.full = True
+ self.reasons.append(reason)
+
+
+def _classify_one(path, repo_root, roster_boards, extras: set, s: _Sel):
+ base = os.path.basename(path)
+ if _NONCODE_RE.match(path):
+ s.reasons.append(f'{path}: non-code, no contribution')
+ return
+ if _FULL_RE.match(path):
+ s.force_full(f'{path}: core/infra -> full matrix')
+ return
+
+ m = re.match(r'src/portable/((?:[^/]+/)?[^/]+)/', path)
+ if m:
+ port = m.group(1)
+ if re.match(r'(dcd_|.*_device)', base):
+ roles = {'device'}
+ elif re.match(r'(hcd_|.*_host)', base):
+ roles = {'host'}
+ else:
+ roles = {'device', 'host'}
+ fams = port_families(port, repo_root)
+ if not fams:
+ # no family references this port: either a new/renamed port dir or a
+ # family.cmake layout the scan misses - widen instead of contributing nothing
+ s.force_full(f'{path}: port {port} maps to no board family -> full matrix')
+ return
+ s.families.update(fams)
+ # a board can also pull the port in through a build option (e.g. MAX3421_HOST=1
+ # from the roster on metro_m4_express, or from its own board.cmake), which its
+ # family file never names
+ gates = port_option_gates(repo_root).get(port, set())
+ boards = [b['name'] for b in roster_boards
+ if (board_family(b['name'], repo_root) in fams or
+ (gates and board_options(b, repo_root) & gates)) and (board_roles(b) & roles)]
+ tests = role_tests(roles, extras)
+ s.roles.update(roles)
+ why = f'{path}: port {port} -> families {sorted(fams)}'
+ if gates:
+ why += f' + option {sorted(gates)}'
+ s.add(boards, tests, f'{why} -> boards {boards} ({"/".join(sorted(roles))})')
+ return
+
+ m = re.match(r'src/class/([^/]+)/', path)
+ if m:
+ cls = m.group(1)
+ if re.search(r'_device\.[ch]$', base):
+ roles = {'device'}
+ elif re.search(r'_host\.[ch]$', base):
+ roles = {'host'}
+ else:
+ roles = {'device', 'host'}
+ # this file's own class, plus any class whose headers include it
+ via = sorted(class_include_edges(repo_root).get(f'{cls}/{base}', ()))
+
+ def macros(prefix):
+ return (class_macros(cls, base, prefix) +
+ [m2 for c in via for m2 in class_macros(c, '', prefix)])
+ tests = set()
+ if 'device' in roles:
+ tests |= class_examples(macros('TUD'), 'device', repo_root, extras)
+ if 'host' in roles:
+ tests |= class_examples(macros('TUH'), 'host', repo_root, extras)
+ boards = [b['name'] for b in roster_boards if board_roles(b) & roles]
+ s.roles.update(roles)
+ why = f'{path}: class {cls}' + (f' (+ included by {via})' if via else '')
+ s.add(boards, tests, f'{why} -> {sorted(tests)} ({"/".join(sorted(roles))})')
+ return
+
+ m = re.match(r'src/(device|host)/', path)
+ if m:
+ role = m.group(1)
+ boards = [b['name'] for b in roster_boards if role in board_roles(b)]
+ s.roles.add(role)
+ s.add(boards, role_tests({role}, extras), f'{path}: core {role} stack -> all {role} tests')
+ return
+
+ m = re.match(r'hw/bsp/([^/]+)/(?:boards/([^/]+)/)?', path)
+ if m:
+ fam, brd = m.group(1), m.group(2)
+ s.families.add(fam)
+ if brd:
+ boards = [b['name'] for b in roster_boards if b['name'] == brd]
+ why = f'{path}: bsp board {brd}'
+ else:
+ boards = [b['name'] for b in roster_boards
+ if board_family(b['name'], repo_root) == fam]
+ why = f'{path}: bsp family {fam}'
+ s.roles.update(('device', 'host'))
+ s.add(boards, 'all', f'{why} -> boards {boards}')
+ return
+
+ m = re.match(r'examples/(device|host|dual)/([^/]+)/', path)
+ if m:
+ test = f'{m.group(1)}/{m.group(2)}'
+ known = any(test in pool for pool in ALL_TESTS.values()) or test in extras
+ if known:
+ boards = [b['name'] for b in roster_boards]
+ role = test_role(test)
+ s.roles.update(('device', 'host') if role == 'dual' else (role,))
+ s.add(boards, [test], f'{path}: example -> {test} on all boards')
+ else:
+ s.reasons.append(f'{path}: example not in HIL lists, no contribution')
+ return
+
+ s.force_full(f'{path}: unclassified -> full matrix')
+
+
+def classify(changed_files, repo_root, rosters):
+ all_boards = []
+ seen = set()
+ for _, boards in rosters:
+ for b in boards:
+ if b['name'] not in seen:
+ seen.add(b['name'])
+ all_boards.append(b)
+
+ extras = roster_only_tests(all_boards)
+ s = _Sel()
+ # no early exit once full: keep classifying so `families` still reports every
+ # family the diff touches (build-only consumers need it). Nothing after the first
+ # force_full can change full/boards/args - the full branch below ignores by_board.
+ for path in changed_files:
+ _classify_one(path, repo_root, all_boards, extras, s)
+
+ if s.full:
+ return {'full': True, 'boards': {b['name']: 'all' for b in all_boards},
+ 'families': sorted(s.families), 'reasons': s.reasons}
+
+ # role pruning: single-role selections drop the other role's tests and boards
+ by_name = {b['name']: b for b in all_boards}
+ out = {}
+ for name, tests in s.by_board.items():
+ allowed = board_tests(by_name[name])
+ if tests == 'all':
+ kept = list(allowed)
+ else:
+ kept = [t for t in allowed if t in tests]
+ if s.roles and s.roles != {'device', 'host'}:
+ role = next(iter(s.roles))
+ kept = [t for t in kept if test_role(t) in (role, 'dual')]
+ if kept:
+ out[name] = 'all' if set(kept) == set(allowed) else sorted(kept)
+ return {'full': False, 'boards': out, 'families': sorted(s.families),
+ 'reasons': s.reasons}
+
+
+def _board_args(name, chosen) -> list:
+ parts = [f'-b {name}']
+ if chosen != 'all':
+ parts.append(f'-bt {name}:{",".join(chosen)}')
+ return parts
+
+
+def selection_args(sel, rosters):
+ """hil_test.py args per config. Empty means either 'full matrix' or 'nothing
+ selected' - callers must read sel['full'] to tell them apart."""
+ args = {}
+ for cfg_path, boards in rosters:
+ parts = []
+ if not sel['full']:
+ for b in boards:
+ chosen = sel['boards'].get(b['name'])
+ if chosen is not None:
+ parts += _board_args(b['name'], chosen)
+ args[os.path.basename(cfg_path)] = ' '.join(parts)
+ return args
+
+
+def selection_args_by_flasher(sel, rosters):
+ """{config: {flasher name: args}}. CI runs one rig as several jobs split by
+ flasher (esptool vs the rest); each must gate on its own subset, otherwise the
+ other leg runs a filter matching zero boards and reports a vacuous green."""
+ out = {}
+ for cfg_path, boards in rosters:
+ per = {}
+ if not sel['full']:
+ for b in boards:
+ chosen = sel['boards'].get(b['name'])
+ if chosen is None:
+ continue
+ per.setdefault(b.get('flasher', {}).get('name', ''), []).extend(
+ _board_args(b['name'], chosen))
+ out[os.path.basename(cfg_path)] = {f: ' '.join(p) for f, p in per.items()}
+ return out
+
+
+def changed_files_from_git(base, repo_root):
+ mb = subprocess.run(['git', 'merge-base', 'HEAD', base], cwd=repo_root,
+ capture_output=True, text=True, check=True).stdout.strip()
+ diff = subprocess.run(GIT_DIFF_ARGV + [f'{mb}..HEAD'], cwd=repo_root,
+ capture_output=True, text=True, check=True).stdout
+ return [l for l in diff.splitlines() if l.strip()]
+
+
+def main():
+ ap = argparse.ArgumentParser(description=__doc__)
+ g = ap.add_mutually_exclusive_group(required=True)
+ g.add_argument('--base', help='git ref to diff against (merge-base..HEAD)')
+ g.add_argument('--diff-file', help='newline-separated changed-file list')
+ ap.add_argument('configs', nargs='+', help='rig roster JSON file(s)')
+ a = ap.parse_args()
+
+ # test/hil/helper/ -> repo root is FOUR levels up; three left this at <repo>/test
+ # after the helper/ move and every repo-relative glob silently matched nothing
+ repo_root = os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
+ rosters = []
+ for c in a.configs:
+ with open(c) as f:
+ rosters.append((c, json.load(f)['boards']))
+
+ files = (open(a.diff_file).read().splitlines() if a.diff_file
+ else changed_files_from_git(a.base, repo_root))
+ files = [f for f in files if f.strip()]
+
+ s = classify(files, repo_root, rosters)
+ s['args'] = selection_args(s, rosters)
+ s['args_flasher'] = selection_args_by_flasher(s, rosters)
+ for r in s['reasons']:
+ print(f'hil_select: {r}', file=sys.stderr)
+ print(json.dumps(s))
+
+
+if __name__ == '__main__':
+ main()
diff --git a/test/hil/helper/hil_util.py b/test/hil/helper/hil_util.py
new file mode 100644
index 000000000..54984d20f
--- /dev/null
+++ b/test/hil/helper/hil_util.py
@@ -0,0 +1,585 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: MIT
+# Bottom layer of the HIL harness: the bounded command runner plus the shared helpers and
+# data every other module needs. Stays stdlib-only and imports nothing local -- everything
+# else imports this, including the unit tests on GitHub's bare runner; never import them
+# from here. Callers set the module global `verbose`.
+
+from __future__ import annotations
+
+import glob
+import os
+import signal
+import subprocess
+import threading
+import sys
+from pathlib import Path
+from typing import Any
+
+
+# -------------------------------------------------------------
+# HIL example test lists, shared by hil_test.py (runner) and hil_select.py (PR-diff
+# selector). Run order is shuffled per board (see test_board); every example carries a
+# unique hardcoded idProduct (see its usb_descriptors.c).
+# -------------------------------------------------------------
+
+# device tests
+device_tests = [
+ 'device/cdc_dual_ports',
+ 'device/cdc_msc',
+ 'device/dfu',
+ 'device/cdc_msc_throughput',
+ 'device/audio_test_freertos',
+ 'device/dfu_runtime',
+ 'device/cdc_msc_freertos',
+ 'device/hid_boot_interface',
+ 'device/msc_dual_lun',
+ 'device/hid_generic_inout',
+ 'device/printer_to_cdc',
+ 'device/midi_test',
+ 'device/mtp',
+ 'device/usbtest', # cafe:4010, unique PID; runs the Linux testusb tier-4 battery via usbtest.py
+ # 'device/net_lwip_webserver', # disabled for PR #3605: USB net iface enum is flaky on the CI HIL host
+]
+
+dual_tests = [
+ 'dual/host_info_to_device_cdc',
+]
+
+host_test = [
+ 'host/cdc_msc_hid',
+ 'host/msc_file_explorer',
+ 'host/msc_file_explorer_freertos',
+ 'host/device_info',
+]
+
+verbose = False
+
+def pos_int_env(name: str, default: int) -> int:
+ # One parsing policy for every HIL_* knob: a bare int() crashes every run at import
+ # on a malformed value, and 0/negative silently removes the bound the knob enforces.
+ try:
+ v = int(os.getenv(name, str(default)))
+ except ValueError:
+ print(f'warning: {name} is not an integer; using {default}',
+ file=sys.stderr, flush=True)
+ return default
+ if v <= 0:
+ print(f'warning: {name}={v} is not usable; using {default}',
+ file=sys.stderr, flush=True)
+ return default
+ return v
+
+
+def pos_float_env(name: str, default: float) -> float:
+ try:
+ v = float(os.getenv(name, str(default)))
+ except ValueError:
+ print(f'warning: {name} is not a number; using {default}',
+ file=sys.stderr, flush=True)
+ return default
+ # float() accepts 'inf'/'nan': an infinite serial timeout is an unbounded read, the
+ # very thing these knobs exist to prevent, and nan fails every comparison silently
+ if not (v > 0 and v < float('inf')):
+ print(f'warning: {name}={v} is not usable; using {default}',
+ file=sys.stderr, flush=True)
+ return default
+ return v
+
+
+CMD_TIMEOUT = pos_int_env('HIL_CMD_TIMEOUT', 180)
+
+TINYUSB_ROOT = Path(__file__).resolve().parents[3] # test/hil/helper/ -> repo root
+
+
+def cmd_stdout_text(out: Any) -> str:
+ if out is None:
+ return ''
+ if isinstance(out, bytes):
+ return out.decode('utf-8', errors='ignore')
+ return str(out)
+
+
+def _banner_body(out: Any, err: Any) -> str:
+ # split_stderr callers keep the diagnostic in stderr — a banner of stdout alone
+ # would be blank exactly when something went wrong
+ body = cmd_stdout_text(out)
+ err_text = cmd_stdout_text(err)
+ if err_text:
+ body = f'{body}\n{err_text}' if body else err_text
+ return body
+
+
+# Shared with compact_output's stripper in hil_test: duplicated literals let the two
+# layers drift and reintroduce literal marker noise mid-row in the GitHub log.
+GROUP_MARK, ENDGROUP_MARK = '::group::', '::endgroup::'
+
+
+def strip_workflow_markers(line: str) -> str:
+ # run_cmd only ever emits markers at line start; mid-line is not a real case.
+ return line.removeprefix(GROUP_MARK).removeprefix(ENDGROUP_MARK)
+
+
+def _ci_log_groups() -> bool:
+ # GitHub folds ::group::/::endgroup:: only at line start of the JOB's real stdout; a
+ # pool worker's capture is compacted into one row line, where they render literally.
+ return bool(os.getenv('CI')) and sys.stdout is sys.__stdout__
+
+
+def _print_banner(title: str, out: Any, err: Any) -> None:
+ print()
+ if _ci_log_groups():
+ print(f'{GROUP_MARK}{title}')
+ print(_banner_body(out, err))
+ print(ENDGROUP_MARK)
+ else:
+ print(title)
+ print(_banner_body(out, err))
+
+
+SYSFS_READ_GRACE = 2.0 # bound on one attribute read of a possibly-wedged device
+SYSFS_STUCK_MAX = 4 # stranded readers tolerated before read_sysfs goes blind
+_sysfs_stuck = 0 # each costs a thread + an fd for the life of the process
+_sysfs_stuck_lock = threading.Lock()
+_sysfs_blind_logged = False
+
+
+class _SysfsUnknown:
+ """Sentinel: the read did not answer. NOT "the attribute is absent" -- reading it as
+ absence turns a healthy board into a firmware regression in the report."""
+ __slots__ = ()
+
+ def __bool__(self) -> bool:
+ return False
+
+ def __repr__(self) -> str:
+ return 'SYSFS_UNKNOWN'
+
+
+SYSFS_UNKNOWN = _SysfsUnknown()
+
+
+def sysfs_blind() -> bool:
+ """True once this process has stranded SYSFS_STUCK_MAX readers: every later read
+ answers SYSFS_UNKNOWN, so nothing it reports about a device is a fact any more."""
+ return _sysfs_stuck >= SYSFS_STUCK_MAX
+
+
+def sysfs_blind_note() -> str:
+ """Suffix for a failure message, so a blind worker's verdict never reads as hardware."""
+ return (f' (this worker is blind: {SYSFS_STUCK_MAX} sysfs reads stranded on a wedged '
+ f'device, so the check could not see the bus)') if sysfs_blind() else ''
+
+
+def read_sysfs(path: str, grace: float = SYSFS_READ_GRACE) -> str | None | _SysfsUnknown:
+ """Read a sysfs attribute with a WALL-CLOCK bound.
+
+ The value, None when the attribute is genuinely unreadable (OSError), or SYSFS_UNKNOWN
+ when the read did not answer -- it timed out, or this process is already blind. Callers
+ MUST keep those apart: absence is a fact, unknown is not.
+
+ usb_string_attr (serial/product/manufacturer) is served under the device lock a wedged
+ usbfs ioctl holds, so a plain open().read() blocks for as long as the wedge lasts, on
+ exactly the board an incident is about. The reader sleeps INTERRUPTIBLY (every read
+ takes usb_lock_device_interruptible, v6.12.96 sysfs.c:124-139 -- uninterruptible is the
+ ioctl holder, not us), so it dies with a SIGKILLed worker; what it costs meanwhile is a
+ thread and an fd for this process's life, because on sysfs the open() SUCCEEDS and only
+ the read blocks. Measured: 20 blocking reads leave 20 live threads.
+
+ Hence the cap: callers rescan (hil_lock's controller_of re-reads every unresolved
+ device on EVERY permit), and hitting RLIMIT_NOFILE or the thread ceiling raises inside
+ the worker and loses every board's result -- worse than the hang this prevents.
+ """
+ if sysfs_blind():
+ return SYSFS_UNKNOWN
+ # Known-stranded? Re-reading costs another permanent thread+fd and a blindness credit
+ # to learn what we already know. Lives HERE, not at the call sites: a call-site memo
+ # has to be remembered by every new scanner, and twice it was not.
+ was = _sysfs_stranded.get(path, _STRAND_MISS)
+ if was is not _STRAND_MISS:
+ if was is None:
+ return SYSFS_UNKNOWN # stranded, inode unknown: never re-read it
+ try:
+ if os.stat(path).st_ino == was:
+ return SYSFS_UNKNOWN # same node, still wedged
+ except OSError:
+ pass # gone: fall through, the read reports it
+ _sysfs_stranded.pop(path, None) # replaced or gone -> re-read it
+ out: dict = {}
+
+ def _read():
+ try:
+ with open(path) as f:
+ out['v'] = f.read().strip()
+ except (OSError, ValueError):
+ pass # no such attribute, or not text: unreadable, and that IS a fact
+
+ t = threading.Thread(target=_read, daemon=True)
+ t.start()
+ t.join(grace)
+ # `out` FIRST, not is_alive() alone: a reader can deposit its value and still be alive
+ # for a moment afterwards, and counting that as a strand memoises a healthy attribute as
+ # unreadable and spends one of four blindness credits. bounded_open has always checked
+ # its box for the same reason.
+ if t.is_alive() and 'v' not in out:
+ # Count the PATH once, not once per reader. hil_pool_check runs -j4 by default,
+ # which equals SYSFS_STUCK_MAX, so four threads hitting ONE wedged device used to
+ # spend the entire blindness budget between them -- latching blind on the single
+ # wedge the tool was run to find. The strand is real for each thread, but the
+ # DEVICE is what the cap is about.
+ # Under the SAME lock as the counter: check-then-act here is a race, and
+ # hil_pool_check runs a ThreadPoolExecutor of exactly SYSFS_STUCK_MAX workers in
+ # ONE process, so four threads on one wedged path could each see `first` before any
+ # of them recorded it -- spending the whole blindness budget on a single device,
+ # which is what this memo exists to prevent. note_sysfs_strand takes the lock
+ # itself, so call it after releasing.
+ with _sysfs_stuck_lock:
+ first = path not in _sysfs_stranded
+ if first:
+ try:
+ # stat, never the thread's own open(): stat does not call ->show(), so
+ # it cannot block on the device lock the reader is stuck behind
+ _sysfs_stranded[path] = os.stat(path).st_ino
+ except OSError:
+ _sysfs_stranded[path] = None # unstattable, but still known-stranded
+ if first:
+ note_sysfs_strand()
+ return SYSFS_UNKNOWN
+ return out.get('v')
+
+
+def note_sysfs_strand() -> None:
+ """Record ONE stranded sysfs reader. Shared by read_sysfs and bounded_open so both
+ account against a single counter -- the report caveat keys off it."""
+ global _sysfs_stuck, _sysfs_blind_logged
+ with _sysfs_stuck_lock:
+ _sysfs_stuck += 1
+ announce = sysfs_blind() and not _sysfs_blind_logged
+ _sysfs_blind_logged = _sysfs_blind_logged or announce
+ if announce:
+ # once per process, on stderr: a worker's stdout is compacted into one report
+ # row, where this would be lost among the test output
+ print(f'warning: {SYSFS_STUCK_MAX} sysfs reads stranded on a wedged device; '
+ f'this process is now blind and answers SYSFS_UNKNOWN for every '
+ f'attribute -- its verdicts about device presence are not evidence',
+ file=sys.stderr, flush=True)
+
+
+# path -> the inode it had when its read stranded. A stranded attribute stays
+# stranded until the DEVICE is replaced, and a re-enumeration destroys the kernfs
+# node and makes a new one -- so a changed inode is the all-clear. Keyed by path
+# alone it would outlive the wedge: a busport does not change when a board comes
+# back on the same port, so the HUNG reflash this branch performs would recover a
+# board the harness could then never see again.
+_sysfs_stranded: dict = {}
+# A stranded path whose inode could not be read is stored as None, so a plain .get() cannot
+# tell 'known stranded, inode unknown' from 'never seen' -- and treating the first as the
+# second re-reads it, stranding another permanent thread and fd every call. Distinct miss
+# sentinel, so None keeps its own meaning.
+_STRAND_MISS = object()
+
+
+def usb_scan(vid_pid=None, serial=None, vid=None) -> tuple[list, bool]:
+ """Enumerated USB devices matching the filters, and whether anything is unknown.
+
+ Returns ([{busport, dir, vid, pid, serial}], unknown). `unknown` True means a bounded
+ read did not answer, so absence is NOT proven -- the same contract as read_sysfs.
+
+ Three rules, one implementation for every caller:
+
+ * Root hubs excluded (glob `*-*`): no DUT is one, and scans including them measured
+ seconds slower (observation, no mechanism -- the "autosuspend wake" explanation was
+ wrong; usb_string_attr reads a cached string, sysfs.c:124-139).
+ * idVendor/idProduct first: lock-free `sysfs_emit` from udev->descriptor
+ (sysfs.c:688-705), so they rule out nearly every device for free.
+ * `serial` last and bounded: it is served under the lock a wedged ioctl holds, and a
+ path that already stranded is never re-read (each strand costs a thread and an fd
+ for this process's life).
+ """
+ out = []
+ unknown = False
+ for d in glob.glob('/sys/bus/usb/devices/*-*'):
+ # Interfaces are '<busport>:<cfg>.<ifnum>' (e.g. 2-4:1.0) -- they CONTAIN the
+ # colon, they do not end with it, so the original endswith() never fired and every
+ # scan opened idVendor/idProduct on all of them (measured: 31 of 44 matches).
+ if ':' in os.path.basename(d):
+ continue
+ try:
+ with open(os.path.join(d, 'idVendor')) as f:
+ dev_vid = f.read().strip()
+ with open(os.path.join(d, 'idProduct')) as f:
+ dev_pid = f.read().strip()
+ except OSError:
+ continue # vanished mid-walk, or not a device dir: a fact, not unknown
+ if vid_pid is not None and (dev_vid, dev_pid) != tuple(vid_pid):
+ continue # ruled out for free, without touching the locked attribute
+ if vid is not None and dev_vid != vid:
+ continue # same, for callers that know the VID but not the PID
+ sn = read_sysfs(os.path.join(d, 'serial'))
+ if sn is SYSFS_UNKNOWN:
+ unknown = True # read_sysfs memoises it; a repeat scan costs nothing
+ continue
+ if sn is None:
+ continue # no serial attribute: a fact
+ if serial is not None and sn.lower() != serial.lower():
+ continue
+ out.append({'busport': os.path.basename(d), 'dir': d,
+ 'vid': dev_vid, 'pid': dev_pid, 'serial': sn})
+ return out, unknown
+
+
+def bounded_open(path: str, flags: int, timeout: float = SYSFS_READ_GRACE):
+ """os.open() with a wall-clock bound.
+
+ The fd, None when the open genuinely FAILED (OSError: EBUSY, ENOENT, EACCES), or
+ SYSFS_UNKNOWN when it did not answer -- the same three-valued contract as read_sysfs,
+ and for the same reason: folding a fact into an unknown made an ordinary EBUSY read as
+ a wedged device and sent the operator hunting hardware that is healthy.
+
+ An open CAN block on a wedged device -- not on O_NONBLOCK, which usblp_open never
+ consults, but on usb_autopm_get_interface(), a runtime-PM resume that does I/O
+ (v6.12.96 drivers/usb/class/usblp.c). It holds usblp_mutex while it waits, and that
+ mutex is driver-GLOBAL, so one wedged printer blocks opens of every usblp node.
+
+ Unlike read_sysfs the stranded thread cleans up after itself: if we have given up it
+ closes the fd it eventually got, so only the thread leaks. Both sides take `handoff`
+ -- "store or close" and "abandon and drain" are a check-then-act pair that can
+ interleave into an fd stored after the box was drained, which would leak it into a
+ node that allows a SINGLE opener (usblp_open returns -EBUSY when usblp->used).
+ """
+ # Same short-circuit as read_sysfs: once blind, another stranded thread buys nothing
+ # and the cap exists precisely to stop them accumulating.
+ if sysfs_blind():
+ return SYSFS_UNKNOWN
+ # Known-stranded? Re-opening costs another thread, another fd and another blindness
+ # credit to learn what we already know -- and the printer test re-opens ONE lp node on
+ # every retry. Same memo and same inode check as read_sysfs.
+ was = _sysfs_stranded.get(path, _STRAND_MISS)
+ if was is not _STRAND_MISS:
+ if was is None:
+ return SYSFS_UNKNOWN # stranded, inode unknown: never re-read it
+ try:
+ if os.stat(path).st_ino == was:
+ return SYSFS_UNKNOWN
+ except OSError:
+ pass
+ _sysfs_stranded.pop(path, None)
+ box: dict = {}
+ done, abandoned = threading.Event(), threading.Event()
+ handoff = threading.Lock()
+
+ def _open():
+ try:
+ fd = os.open(path, flags)
+ except OSError:
+ done.set()
+ return
+ with handoff:
+ stored = not abandoned.is_set()
+ if stored:
+ box['fd'] = fd
+ if not stored:
+ try:
+ os.close(fd)
+ except OSError:
+ pass
+ done.set()
+
+ threading.Thread(target=_open, daemon=True).start()
+ if not done.wait(timeout):
+ with handoff:
+ abandoned.set()
+ fd = box.pop('fd', None) # completed in the gap between timeout and flag
+ if fd is not None:
+ # It DID open, just after our deadline -- the thread finished, so nothing is
+ # stranded. Report unknown (we already gave up on it) but do not spend a
+ # blindness credit, and do not call a merely-slow node wedged.
+ try:
+ os.close(fd)
+ except OSError:
+ pass
+ return SYSFS_UNKNOWN
+ # counted like a stranded read_sysfs: the thread and (eventually) its fd are gone
+ # for the life of the process, and the cap exists to stop that reaching the
+ # thread/fd ceiling -- an exception there escapes the worker and loses every board.
+ # Memoised by inode so a retry of the same node does not pay again.
+ # same lock as read_sysfs, same reason
+ with _sysfs_stuck_lock:
+ first = path not in _sysfs_stranded
+ if first:
+ try:
+ _sysfs_stranded[path] = os.stat(path).st_ino
+ except OSError:
+ _sysfs_stranded[path] = None
+ if first:
+ note_sysfs_strand()
+ return SYSFS_UNKNOWN
+ return box.get('fd')
+
+
+def _close_pipes(p: subprocess.Popen) -> None:
+ """Close OUR ends of an abandoned child's pipes. Never raises."""
+ for pipe in (p.stdout, p.stderr, p.stdin):
+ try:
+ if pipe is not None:
+ pipe.close()
+ except OSError:
+ pass
+
+
+def run_alongside(argv: list, work, timeout: int) -> subprocess.CompletedProcess:
+ """Run `argv` alongside `work()`, which runs in THIS thread, then reap it -- bounded.
+
+ The read-while-we-write shape run_cmd cannot express: the caller needs the child
+ RUNNING while it does something else. Everything else about the contract is run_cmd's
+ -- own session, killpg, bounded reap, our pipe ends closed, rc 124 on the kill.
+
+ A PROCESS, not a thread: an abandoned thread keeps the fd, and usblp_open returns
+ -EBUSY while usblp->used (v6.12.96 usblp.c), so every later open in this long-lived
+ worker would read as a wedged device. A killed process takes its fd with it.
+
+ stdout is captured as BYTES and kept CLEAN -- a caller byte-compares it against the
+ payload it sent, so a single stderr byte (a PYTHONWARNINGS chirp, a sitecustomize
+ print, a .pth deprecation from a venv) would read as USB data corruption. stderr gets
+ its own pipe; communicate() drains both, so the split cannot deadlock.
+ `work` runs even if the child dies immediately -- the caller's own asserts decide.
+ """
+ p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
+ start_new_session=True)
+
+ def _reap() -> subprocess.CompletedProcess:
+ try:
+ out, err = p.communicate(timeout=timeout)
+ return subprocess.CompletedProcess(argv, p.returncode, out, err)
+ except subprocess.TimeoutExpired:
+ try:
+ os.killpg(p.pid, signal.SIGKILL)
+ except OSError:
+ p.kill()
+ try:
+ out, err = p.communicate(timeout=5)
+ except subprocess.TimeoutExpired:
+ # Outlasted SIGKILL: uninterruptible, still holding whatever it opened.
+ # Abandoned like any other stray -- but as a real child in its own
+ # session, so the containment sweep FINDS it (child_procs walks the ppid
+ # tree) and the report names it. That is the whole difference from a
+ # blocked thread, which no sweep can see and no signal can reach.
+ out, err = b'', b''
+ _close_pipes(p) # our own fds must not leak either
+ return subprocess.CompletedProcess(argv, 124, out, err)
+
+ try:
+ work()
+ except BaseException:
+ # Reap first so the child never outlives us, then let the caller's error through.
+ # A `return` inside a `finally` would SWALLOW it -- an assert in `work` would
+ # vanish and the caller would compare data it never finished sending.
+ _reap()
+ raise
+ return _reap()
+
+
+def run_cmd(cmd: str, cwd: str | None = None, timeout: int | None = None,
+ binary: bool = False, split_stderr: bool = False,
+ quiet: bool = False) -> subprocess.CompletedProcess:
+ if timeout is None:
+ timeout = CMD_TIMEOUT
+ # binary: raw bytes (text mode's errors='replace' mangles non-UTF-8 file content).
+ # split_stderr: keep stderr out of stdout, for callers that parse stdout. quiet: no
+ # COMMAND FAILED banner, for retry loops that report failures themselves (timeouts
+ # still print: a killed child is always noteworthy).
+ popen_kwargs = {
+ 'cwd': cwd,
+ 'shell': True,
+ 'stdout': subprocess.PIPE,
+ 'stderr': subprocess.PIPE if split_stderr else subprocess.STDOUT,
+ }
+ if not binary:
+ popen_kwargs.update({'text': True, 'encoding': 'utf-8', 'errors': 'replace'})
+ if os.name != 'nt':
+ # C-level setsid, same process-group semantics as preexec_fn=os.setsid but
+ # safe when called from threads (pool_check runs flashes from a thread pool)
+ popen_kwargs['start_new_session'] = True
+
+ p = subprocess.Popen(cmd, **popen_kwargs)
+ try:
+ out, err = p.communicate(timeout=timeout)
+ r = subprocess.CompletedProcess(args=cmd, returncode=p.returncode, stdout=out, stderr=err)
+ except subprocess.TimeoutExpired as ex:
+ if os.name != 'nt':
+ try:
+ os.killpg(p.pid, signal.SIGKILL)
+ except OSError:
+ # ProcessLookupError: already gone. PermissionError: an all-root group
+ # refuses the group kill -- letting either escape would skip the bounded
+ # reap, the pipe close and the rc-124 return this handler exists for.
+ pass
+ else:
+ p.kill()
+ try:
+ out, err = p.communicate(timeout=10)
+ except subprocess.TimeoutExpired:
+ # Something in the group outlived SIGKILL: D state (truly unkillable), or
+ # root-owned because sudo FORKS rather than execs, so the wrapper dies and its
+ # root child does not. Abandon it and let the report name it; the harness never
+ # sudo-kills its way out. Our ends of its pipes must not leak, though: a pool
+ # worker lives for the whole run, so every wedged command would cost it two fds.
+ out, err = None, None
+ _close_pipes(p)
+ # prefer the post-kill buffers (supersets of the exception's), falling back to ex.*
+ # when the child was unkillable. TimeoutExpired carries BYTES even for a text-mode
+ # Popen, so the fallbacks must be decoded or a text-mode caller gets bytes exactly
+ # when the child wedged in D state.
+ def _typed(v):
+ if not binary and isinstance(v, bytes):
+ return v.decode('utf-8', errors='replace')
+ return v
+
+ timeout_out = _typed(out or ex.stdout) or (b'' if binary else '')
+ # ...and never None: with split_stderr the SUCCESS path always yields a str/bytes,
+ # so a caller that does `r.stderr.strip()` works everywhere except the timeout --
+ # the one path it was written for. Without split_stderr stderr stays None, as on
+ # the success path (it was merged into stdout).
+ timeout_err = _typed(err if err is not None else ex.stderr)
+ if split_stderr and timeout_err is None:
+ timeout_err = b'' if binary else ''
+ _print_banner(f'COMMAND TIMEOUT ({timeout}s): {cmd}', timeout_out, timeout_err)
+ return subprocess.CompletedProcess(args=cmd, returncode=124, stdout=timeout_out, stderr=timeout_err)
+ except BaseException:
+ # BaseException, not Exception (as in CPython's own subprocess.run):
+ # KeyboardInterrupt is the case that matters, and start_new_session put the child in
+ # its OWN group, so it never got the terminal's SIGINT -- without this, Ctrl-C
+ # leaves the flasher or testusb holding the probe and its usbfs node. Kill and
+ # close, never wait: this path must not add a hang of its own.
+ if os.name != 'nt':
+ try:
+ os.killpg(p.pid, signal.SIGKILL)
+ except OSError:
+ pass
+ else:
+ p.kill()
+ _close_pipes(p)
+ raise
+
+ if r.returncode != 0 and not quiet:
+ _print_banner(f'COMMAND FAILED: {cmd}', r.stdout, r.stderr)
+ elif verbose:
+ print(cmd)
+ print(cmd_stdout_text(r.stdout))
+ return r
+
+
+# get usb serial by id
+def get_serial_dev(id, vendor_str, product_str, ifnum):
+ if vendor_str and product_str:
+ # known vendor and product
+ vendor_str = vendor_str.replace(' ', '_')
+ product_str = product_str.replace(' ', '_')
+ return f'/dev/serial/by-id/usb-{vendor_str}_{product_str}_{id}-if{ifnum:02d}'
+ else:
+ # just use id: mostly for cp210x/ftdi flasher
+ pattern = f'/dev/serial/by-id/usb-*_{id}-if*'
+ port_list = glob.glob(pattern)
+ if len(port_list) == 0:
+ raise RuntimeError(f'No serial device found for {pattern}')
+ return port_list[0]