summaryrefslogtreecommitdiff
path: root/test/hil/helper
diff options
context:
space:
mode:
Diffstat (limited to 'test/hil/helper')
-rw-r--r--test/hil/helper/hil_health.py48
-rwxr-xr-xtest/hil/helper/hil_lock.py15
-rw-r--r--test/hil/helper/hil_pool_check.py105
-rw-r--r--test/hil/helper/hil_report.py578
-rwxr-xr-xtest/hil/helper/hil_select.py524
-rw-r--r--test/hil/helper/hil_summary.py115
-rw-r--r--test/hil/helper/hil_util.py478
7 files changed, 893 insertions, 970 deletions
diff --git a/test/hil/helper/hil_health.py b/test/hil/helper/hil_health.py
index 92f0accc8..d78d0f220 100644
--- a/test/hil/helper/hil_health.py
+++ b/test/hil/helper/hil_health.py
@@ -1,6 +1,6 @@
#!/usr/bin/env python3
# SPDX-License-Identifier: MIT
-"""Shutting a wedged HIL run down: kill what the workers spawned, then report.
+"""Shutting a wedged HIL run down: kill what the workers spawned.
A device whose usbfs node is held by a D-state process cannot be freed -- SIGKILL is not
delivered in uninterruptible sleep -- so the goal is never to fix the rig from here. It is
@@ -214,9 +214,6 @@ def _kill_kids(kids: dict, seen: set) -> int:
own = os.getpgid(0)
except OSError:
own = None # cannot tell our own group apart: never killpg, signal pids only
- # One list: every pid here is a DESCENDANT of one of our own workers, so it is ours by
- # construction -- no argv identity check needed, because we never signal anything we
- # did not discover through our own ppid tree.
touched: list = []
for children in kids.values():
for cpid, cpgid in children:
@@ -249,10 +246,8 @@ def _kill_kids(kids: dict, seen: set) -> int:
if denied:
_p(f'warning: could not kill {sorted(denied)}; they still hold whatever they '
f'had open (probe, usbfs node) into the next job', flush=True)
- # SURVIVORS, not the signalled-child count: the caller needs to know the rig is dirty
- # for the next job, and a count of what we successfully signalled cannot tell it that.
- # (They are different units anyway -- a killpg is counted once per child sharing the
- # group -- so the old return was never comparable to anything.)
+ # SURVIVORS, not the count we signalled: the caller needs to know the rig is dirty for
+ # the next job, and a killpg is counted once per child sharing the group anyway.
return len(denied)
@@ -341,40 +336,3 @@ def kill_pool_children(pool, *extra) -> int:
# SIGKILL is asynchronous and a D-state task ignores it: only a confirmed survivor
# justifies the caller's power-cycle wording
return len(_kill_and_confirm(killed_pids)) if killed_pids else 0
-
-
-def write_timeout_report(report_dir: Path, boards, secs: int, md_name: str,
- banner: str = '', prefix: str = '') -> None:
- """Leave a report behind when the worker pool has to be abandoned.
-
- map_async is all-or-nothing, so a timeout loses every per-board result and the report
- dir would stay empty with no reason for the failure. Any prior attempt's markdown is
- kept below the banner."""
- # `prefix` carries the preflight rig-health verdict: the timeout aborts before
- # accumulate_report, so without it the report loses the one line saying WHY the pool
- # never finished. The '\n' stops Markdown lazy continuation pulling the banner into
- # the blockquote.
- try:
- # Built INSIDE the try: a roster entry without a 'name' key raises KeyError while
- # assembling the board list, and outside the try that escaped and stranded the
- # runner -- which is exactly what the broad handler below exists to prevent.
- head = (prefix + '\n' if prefix else '') + (banner or (
- f'**HIL run abandoned: worker pool timed out after {secs}s.**\n\n'
- f'No per-board results could be collected for this attempt, so the '
- f'table below (if any) is from an earlier one. Boards dispatched:\n\n'
- + '\n'.join(f'- {b.get("name", "?")}' for b in boards) + '\n'))
- report_dir.mkdir(parents=True, exist_ok=True)
- md_path = report_dir / md_name
- # Its own handler so it cannot take the write down with it: a report torn by an
- # attempt killed mid-write raises UnicodeDecodeError (a ValueError, and prior
- # reports always contain status emoji), which under a shared try skipped the write
- # entirely. Losing the old table is a nicety; losing the banner is the failure.
- try:
- prior = md_path.read_text(encoding='utf-8') if md_path.is_file() else ''
- except (OSError, ValueError):
- prior = ''
- md_path.write_text(head + (f'\n{prior}' if prior else ''), encoding='utf-8')
- except Exception as e: # noqa: BLE001
- # Deliberately broad: this is the first statement of the pool-abandon path, so ANY
- # escape skips kill_pool_children and os._exit and strands the runner.
- _p(f'warning: cannot write {md_name} to {report_dir}: {e}', flush=True)
diff --git a/test/hil/helper/hil_lock.py b/test/hil/helper/hil_lock.py
index 7757ef17d..91f05ca86 100755
--- a/test/hil/helper/hil_lock.py
+++ b/test/hil/helper/hil_lock.py
@@ -175,13 +175,12 @@ def controller_of(uid: str):
if cached:
return cached
# vid='cafe' first: the target is always a TinyUSB DUT, and the VID is a lock-free
- # descriptor field. Without it this read every probe's and hub's `serial` -- the
- # attribute served under device_lock -- so a HEALTHY peer mid-usbtest would strand a
- # reader here and spend one of this worker's four blindness credits.
- devs, _ = hil_util.usb_scan(vid='cafe', serial=uid)
+ # descriptor field. Without it this reads every probe's and hub's `serial` -- the one
+ # attribute served under device_lock -- so a wedged peer would block us here.
+ devs = hil_util.usb_scan(vid='cafe', serial=uid)
for dev in devs:
busnum = hil_util.read_sysfs(os.path.join(dev['dir'], 'busnum'))
- if busnum is None or busnum is hil_util.SYSFS_UNKNOWN:
+ if busnum is None:
continue
try:
root = os.path.realpath(f'/sys/bus/usb/devices/usb{int(busnum)}')
@@ -212,7 +211,7 @@ def controller_slot(pci: str) -> int:
# Unresolved boards budget in a slot of their OWN, one past the real ones, and that slot
# holds exactly ONE permit whatever the per-controller width is. Neither neighbour works:
# a permit on every slot (the old fail-closed rule) serialized the whole fleet the moment
-# a worker went blind, while a full private budget let unknown boards run a second
+# one board could not be resolved, while a full private budget let unknown boards run a second
# controller's worth of batteries on top of the resolved ones -- doubling the load on
# whichever physical controller they actually sit on, which is the saturation the
# uPD720201 deaths above are attributed to. Width 1 caps the over-subscription at +1.
@@ -252,8 +251,8 @@ class controller_permit:
if pci is None:
pci = controller_of(uid)
if pci is None and warn_unknown:
- log(f'warning: cannot resolve {uid} to a host controller'
- f'{hil_util.sysfs_blind_note()}; budgeting it in the unknown bucket')
+ log(f'warning: cannot resolve {uid} to a host controller; '
+ f'budgeting it in the unknown bucket')
self.slots = [controller_slot(pci) if pci else UNKNOWN_SLOT]
def __enter__(self):
diff --git a/test/hil/helper/hil_pool_check.py b/test/hil/helper/hil_pool_check.py
index d926bbe3d..d98b92bd4 100644
--- a/test/hil/helper/hil_pool_check.py
+++ b/test/hil/helper/hil_pool_check.py
@@ -54,7 +54,7 @@ ENUM_WAIT_RETRY = 8 # s, uid wait after a recovery reset/re-flash
SERIAL_WAIT = 6 # s, host-board serial-output wait
print_mutex = threading.Lock()
-_UNKNOWN_WARNED = False # scan_usb's caveat: once per process, not once per poll
+_STRANDED_WARNED = False # scan_usb's caveat: once per process, not once per poll
t0 = time.monotonic()
@@ -72,20 +72,20 @@ def scan_usb() -> dict:
USB-Serial-JTAG bridge and the cafe device it flashes both derive it from the same
MAC), and one dict slot would silently drop whichever lost the race."""
found = {}
- # `unknown` matters BEFORE the blindness latch trips: one wedged device is the normal
- # reason this tool is run, and its serial read stranding makes it absent from `devs`.
- # Reported as fact, that is "probe MISSING" for hardware that is physically present.
- devs, unknown = hil_util.usb_scan()
- # ONCE per process: this is called from 0.5s poll loops across 4 worker threads and
- # ~26 boards, so warning per call buried the table it exists to qualify under 600+
- # identical lines. The memo in read_sysfs makes the condition sticky, so one line is
- # as true as six hundred.
- global _UNKNOWN_WARNED
- if unknown and not _UNKNOWN_WARNED:
- _UNKNOWN_WARNED = True
- say('WARNING: at least one device did not answer a bounded read; rows below that '
- 'say a probe or board is missing may be this scan losing sight of healthy '
- 'hardware. Find the wedged device (usb-kernel-recover) and re-run.')
+ # usb_scan's `serial` read is bounded by default (see hil_util.read_sysfs) -- this tool
+ # has no pool guard behind it and is run exactly when a device is suspected wedged. A
+ # device that will not answer is simply absent from the table; the footer says so.
+ devs = hil_util.usb_scan()
+ # ONCE per process, at SCAN time, not only in the footer: this tool prints rows as it
+ # goes over minutes, so a board dropped from the scan says "probe MISSING" within
+ # seconds while the only qualification would arrive after the final counts -- and an
+ # operator acting on the streaming output, or a run cut short by ^C, never sees it.
+ global _STRANDED_WARNED
+ if hil_util.sysfs_stranded() and not _STRANDED_WARNED:
+ _STRANDED_WARNED = True
+ say('WARNING: a bounded sysfs read gave up; rows below that say a probe or board '
+ 'is missing may be this scan losing sight of healthy hardware. Find the '
+ 'wedged device (usb-kernel-recover) and re-run.')
for dev in devs:
try:
found[dev['busport']] = {
@@ -360,7 +360,47 @@ def check_host_serial(board: dict, do_reset: bool = True, want_hello: bool = Fal
do_reset=False listens to the firmware as-is: used right after a flash whose
own reset already started it โ€” a second openocd/JLink session back-to-back on
- the same probe can fail transiently and leave the target halted."""
+ the same probe can fail transiently and leave the target halted.
+
+ "logger": "rtt" boards have no VCOM: the same check runs over the probe's RTT
+ console instead. The reset happens BEFORE the console opens (it owns the probe),
+ which also zeroes the .bss ring โ€” so pre-reset backlog cannot count as life, and
+ without a reset Commander delivers the boot burst the preceding flash left."""
+ if board.get('logger') == 'rtt':
+ if do_reset:
+ # a failed reset leaves the previous run's ring intact: attaching anyway would
+ # score stale output as life, so bail to host_alive's board_test reflash ladder
+ rc, err = call_flasher(getattr(hil_flash, f'reset_{board["flasher"]["name"].lower()}'), board)
+ if rc:
+ say(f'{board["name"]:26} reset failed: {err}')
+ return None
+ try:
+ ser = hil_util.JlinkRtt(board, timeout=0.3)
+ except hil_util.RttError as e:
+ say(f'{board["name"]:26} no RTT console: {e}')
+ return None
+ try:
+ data = b''
+ deadline = time.monotonic() + SERIAL_WAIT
+ while time.monotonic() < deadline:
+ ser.write(b'U')
+ data += ser.read(256)
+ # JLinkExe's banner arrives whether or not the target is alive --
+ # judged unfiltered it scores a dead board 'alive'. Same shared filter
+ # as test_host_device_info; complete_only drops a trailing partial
+ # line, so a banner FRAGMENT split by this read boundary cannot count
+ # as target output either.
+ td = hil_util.strip_banner(data, complete_only=True)
+ if want_hello:
+ if b'Hello from TinyUSB' in td:
+ return td
+ elif td and not boardtest_output(td):
+ return td
+ return hil_util.strip_banner(data)
+ except hil_util.RttError:
+ return None # console died mid-poll (server exited, probe dropped)
+ finally:
+ ser.close()
import serial
try:
port = hil_util.get_serial_dev(board['flasher']['uid'], None, None, 0)
@@ -433,7 +473,7 @@ def build_example(board: dict, variant: str, example: str) -> int:
cmd = ['idf.py', '-C', f'examples/{example}',
'-B', f'cmake-build/cmake-build-{vcfg["name"]}/{example}',
'-G', 'Ninja', f'-DBOARD={name}', 'build']
- for d in board.get('build', {}).get('args', []) + vcfg.get('defines', []):
+ for d in vcfg.get('defines', []):
cmd.insert(-1, f'-D{d}')
if vcfg.get('flags'):
cmd.insert(-1, f'-DCFLAGS_CLI={vcfg["flags"]}')
@@ -446,8 +486,6 @@ def build_example(board: dict, variant: str, example: str) -> int:
cmd = [sys.executable, str(hil_util.TINYUSB_ROOT / 'tools' / 'build.py'),
'-b', name, '-T', Path(example).name,
'-j', str(max(1, (os.cpu_count() or _jobs) // _jobs))]
- for d in board.get('build', {}).get('args', []):
- cmd += ['-D', d]
if vcfg['name'] != name:
cmd += ['--build-name', vcfg['name']]
for d in vcfg.get('defines', []):
@@ -985,9 +1023,13 @@ def main() -> None:
headers = ['Board', 'Probe', 'Flash', 'Device', 'Status', 'Note']
cells = [[r['name'], r['probe'], r['flash'], r['device'],
status_mark.get(r['status'], r['status']), '; '.join(r['note'])] for r in rows]
- widths = [max(len(h), *(len(c[i]) for c in cells)) if cells else len(h)
+ # display_width, not len(): โœ… / โŒ / ๐Ÿ”’ / โš  are one character and two columns, so
+ # len() pads every row holding one a column short of the header rule
+ _w = hil_util.display_width
+ widths = [max(_w(h), *(_w(c[i]) for c in cells)) if cells else _w(h)
for i, h in enumerate(headers)]
- line = lambda vals: '| ' + ' | '.join(v.ljust(w) for v, w in zip(vals, widths)) + ' |'
+ line = lambda vals: ('| ' + ' | '.join(hil_util.pad(v, w)
+ for v, w in zip(vals, widths)) + ' |')
print()
print(line(headers))
print('|' + '|'.join('-' * (w + 2) for w in widths) + '|')
@@ -1003,17 +1045,16 @@ def main() -> None:
counts[r.get('status', 'failed')] += 1
print(f'\n{counts["ok"]} ok ยท {counts["flash-failed"]} flash-failed ยท {counts["failed"]} failed '
f'ยท {counts["locked"]} locked ยท in {time.monotonic() - t0:.0f}s')
- if hil_util.sysfs_blind():
- # Without this the table is the worst kind of wrong: once the process latches
- # blind, every read answers SYSFS_UNKNOWN, scan_usb() returns {}, and EVERY board
- # prints "probe MISSING"/"off bus" -- a clean-looking report declaring the whole
- # fleet dead, produced during exactly the incident this tool is run to diagnose,
- # and it sends the operator to power-cycle a rig where one device is wedged.
- print('WARNING: this scan lost sight of the bus'
- f'{hil_util.sysfs_blind_note()}. Rows above that say a probe or board is '
- f'missing may be this tool losing sight of healthy hardware, not absent '
- f'hardware. Find the wedged device (see the usb-kernel-recover skill) and '
- f're-run before acting on the table.')
+ if hil_util.sysfs_stranded():
+ # Without this the table is the worst kind of wrong: a device whose `serial` never
+ # answered is absent from the scan, which prints as "probe MISSING"/"off bus" for
+ # hardware that is physically present -- during exactly the incident this tool is
+ # run to diagnose, and it sends the operator to power-cycle a healthy rig.
+ print('WARNING: at least one sysfs read did not answer within '
+ f'{hil_util.SYSFS_READ_GRACE:.0f}s, so rows above that say a probe or board '
+ f'is missing may be this tool losing sight of healthy hardware rather than '
+ f'absent hardware. Find the wedged device (see the usb-kernel-recover '
+ f'skill) and re-run before acting on the table.')
sys.exit(min(counts['flash-failed'] + counts['failed'], 125))
diff --git a/test/hil/helper/hil_report.py b/test/hil/helper/hil_report.py
new file mode 100644
index 000000000..c93c8e6a1
--- /dev/null
+++ b/test/hil/helper/hil_report.py
@@ -0,0 +1,578 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: MIT
+"""The HIL report document: one owner for hil_report.json and hil_report.md.
+
+The markdown IS a rendering of the sidecar -- every writer goes through render_report(), so
+a table can never contain something the JSON does not. This module owns the whole life of
+that document: the cell vocabulary, the one classifier both artifacts share, rendering, the
+writers, and the fold to one machine-readable verdict per board.
+
+Dual-mode by design: imported as `helper.hil_report` by hil_test.py, and run as a script by
+the operator (see .claude/agents/hil-operator.md). A script run puts test/hil/helper on
+sys.path rather than test/hil, so this module imports no sibling helper at all --
+_p and the width helpers below are defined locally for that reason.
+"""
+import argparse
+import json
+import sys
+import unicodedata
+from pathlib import Path
+
+
+def _w(s: str) -> int:
+ """Terminal COLUMNS, not characters. Every status mark in REPORT_CELL is one Python
+ character and TWO columns wide, so len() pads a cell holding one a column short and
+ the pipes drift out of line with the header rule for the whole table.
+
+ Local, like _p above and for the same reason: this module is also run as a script, and
+ under PYTHONSAFEPATH=1 a sibling import dies before argparse runs. hil_util carries the
+ same pair for callers that can import it.
+ """
+ return sum(2 if unicodedata.east_asian_width(c) in 'WF' else 1 for c in s)
+
+
+def _pad(s: str, width: int, center: bool = False) -> str:
+ """str.ljust/center, measured in display columns. See _w."""
+ room = max(0, width - _w(s))
+ if not center:
+ return s + ' ' * room
+ left = room // 2
+ return ' ' * left + s + ' ' * (room - left)
+
+
+def _p(*args, **kwargs) -> None:
+ """Print that cannot raise. Defined here rather than imported from hil_health: this
+ module is ALSO run as a script (hil-operator.md invokes it by path), and under
+ PYTHONSAFEPATH=1 -- which the suite's own MTP fixtures set -- sys.path[0] is not the
+ script dir, so any sibling import dies before argparse runs. Five lines beat that."""
+ try:
+ print(*args, **kwargs)
+ except (OSError, ValueError):
+ # ValueError too: printing to a CLOSED stream raises "I/O operation on closed
+ # file", and escaping here skips the containment path's os._exit.
+ pass
+
+REPORT_MD = 'hil_report.md'
+REPORT_JSON = 'hil_report.json'
+# The status vocabulary, shared by the code that WRITES a cell (hil_test's test runners) and
+# the code that reads one back (cell_state). One dict, so the human's table and the agent's
+# verdict cannot drift apart.
+REPORT_CELL = {'pass': 'โœ…', 'fail': 'โŒ', 'skip': 'โšช'}
+BOUNDARY_CELL = 'same-PID boundary'
+LOCKED_CELL = 'board-locked'
+# A pseudo-test column, not a real one: write_timeout_report marks the boards that were
+# still dispatched when the pool guard fired. accumulate_report clears it on a retry.
+POOL_TIMEOUT_CELL = 'pool-timeout'
+# The other way a board can fail to report: the pool did not expire, a worker RAISED. Same
+# shape, different cause, and naming the cause is the whole point of the column -- a board
+# marked pool-timeout by an abort that never timed out sends the reader after the guard.
+RUN_ABORTED_CELL = 'run-aborted'
+
+
+def _load(report_dir: Path) -> tuple:
+ """(doc, readable) for the sidecar, coerced to the canonical shape.
+
+ hil_ci.sh uploads a sidecar as the --accumulate merge base, so a non-conforming one is
+ reachable from OUTSIDE the harness -- and every writer here runs on a path where a
+ TypeError costs the whole report. Coerce once, at the boundary, instead of guarding
+ each use: `banner: null` used to kill a fully successful run with a traceback and no
+ artifact at all, and `cells: null` sent write_timeout_report down its fallback so a
+ board that ate the whole pool guard was published as a pass.
+
+ `readable` is False only when a sidecar EXISTS but could not be parsed, or is absent --
+ both mean its rows are unrecoverable, which callers use to avoid destroying a markdown
+ that may still hold them."""
+ jpath = report_dir / REPORT_JSON
+ if not jpath.is_file():
+ return {'rows': [], 'banner': '', 'scope': '', 'caveat': ''}, False
+ try:
+ raw = json.loads(jpath.read_text())
+ if not isinstance(raw, dict):
+ raise ValueError('sidecar is not an object')
+ except (OSError, ValueError, TypeError):
+ return {'rows': [], 'banner': '', 'scope': '', 'caveat': ''}, False
+ rows = []
+ # isinstance, not `or []`: a sidecar with `rows: 1` iterates an int and raises outside
+ # the parse handler above.
+ for r in (raw.get('rows') if isinstance(raw.get('rows'), list) else []):
+ if not isinstance(r, dict) or 'board' not in r:
+ continue
+ cells = r.get('cells')
+ dur = r.get('duration')
+ # VALUES as well as keys: render_matrix does REPORT_CELL.get(v, v), which raises
+ # TypeError on an unhashable value, and cell_state does v.startswith. A non-str
+ # cell is corrupt, and dropping it renders blank -- "not run" -- which is the
+ # honest reading. Coercing it to str would make it classify as a PASS.
+ rows.append({'board': str(r['board']),
+ 'cells': {str(k): v for k, v in cells.items() if isinstance(v, str)}
+ if isinstance(cells, dict) else {},
+ 'duration': dur if isinstance(dur, str) else None})
+ text = lambda k: raw[k] if isinstance(raw.get(k), str) else ''
+ return {'rows': rows, 'banner': text('banner'), 'scope': text('scope'),
+ 'caveat': text('caveat')}, True
+
+
+def cell_state(v) -> str:
+ """'pass' | 'fail' | 'skip' for one report cell.
+
+ THE classifier -- the markdown tally and the per-board verdict both call this, so they
+ cannot disagree. 'fail' or a fail-icon prefix is a failure, 'skip' or a skip-icon prefix
+ is a skip, and EVERYTHING ELSE is a pass. That last arm is load-bearing: a passing test
+ may return a plain metric string ('480.0 MBps') that lands in the cell unprefixed, while
+ failures are guaranteed marked -- TestFail's docstring pins that its metric is
+ icon-prefixed precisely so render and tally treat it as a failure. Classifying unknown
+ shapes as fail here would publish a green table as a red verdict.
+
+ isinstance-guarded: cells are usually str but a caller may hand over None or a number,
+ and .startswith on those raises inside a report writer that must not raise."""
+ if v == 'fail' or (isinstance(v, str) and v.startswith(REPORT_CELL['fail'])):
+ return 'fail'
+ if v == 'skip' or (isinstance(v, str) and v.startswith(REPORT_CELL['skip'])):
+ return 'skip'
+ return 'pass'
+
+
+def render_matrix(rows_all: list) -> str:
+ """Render rows (list of (row_label, {example: status}, duration)) as an aligned
+ markdown matrix: columns = tests (bare names) centered, boards left-aligned,
+ per-row duration as the trailing column."""
+ seen = set()
+ for _, cells, _ in rows_all:
+ seen.update(cells)
+ if not seen:
+ return 'No tests were run.'
+
+ # metric-bearing columns pinned first, the rest alphabetical: stable regardless of the
+ # shuffled execution order
+ pinned = ['usbtest', 'cdc_msc_throughput', 'msc_file_explorer', 'msc_file_explorer_freertos']
+
+ def col_key(t):
+ name = t.rsplit('/', 1)[-1]
+ return (pinned.index(name) if name in pinned else len(pinned), name, t)
+
+ columns = sorted(seen, key=col_key)
+ headers = [c.rsplit('/', 1)[-1] for c in columns] + ['duration'] # bare example names
+
+ def cell(cells, col):
+ v = cells.get(col)
+ if v is None:
+ return ''
+ return REPORT_CELL.get(v, v) # status symbol, or a metric string (e.g. speed) verbatim
+
+ rows_vals = [(lbl, [cell(cells, c) for c in columns] + [dur or ''])
+ for lbl, cells, dur in rows_all]
+ board_hdr = 'Board'
+ # display_width, not len(): the โœ…/โŒ/โšช marks are one character and two columns
+ board_w = max([_w(board_hdr)] + [_w(lbl) for lbl, _ in rows_vals])
+ col_w = [max([_w(h)] + [_w(vals[i]) for _, vals in rows_vals])
+ for i, h in enumerate(headers)]
+
+ def line(label, values):
+ padded = [_pad(label, board_w)] + [_pad(v, w, center=True)
+ for v, w in zip(values, col_w)]
+ return '| ' + ' | '.join(padded) + ' |'
+
+ header = line(board_hdr, headers)
+ sep = '| ' + '-' * board_w + ' | ' + ' | '.join(':' + '-' * (w - 2) + ':' for w in col_w) + ' |'
+ body = [line(lbl, vals) for lbl, vals in rows_vals]
+
+ # tally run cells (not-run cells are absent from the dicts). A cell is a bare status or
+ # a metric string carrying its own icon ("โŒ 29/30"), so classify by the leading icon --
+ # through cell_state, the same call the per-board verdict makes.
+ kinds = [cell_state(v) for _, cells, _ in rows_all for v in cells.values()]
+ failed = kinds.count('fail')
+ skipped = kinds.count('skip')
+ passed = kinds.count('pass')
+ summary = (f'**{REPORT_CELL["pass"]} {passed} passed ยท {REPORT_CELL["fail"]} {failed} failed ยท '
+ f'{REPORT_CELL["skip"]} {skipped} skipped ยท blank not run**')
+
+ return summary + '\n\n' + '\n'.join([header, sep] + body)
+
+
+def render_report(doc: dict) -> str:
+ """The markdown IS a rendering of the sidecar. Every writer goes through here, so a
+ table can never contain something the JSON does not."""
+ # .get throughout, not subscripts: mark_report_abandoned renders a sidecar it did NOT
+ # write (hil_ci.sh reuses a persistent REMOTE_DIR, so it may be an older version's or
+ # a torn one) on the way to os._exit, and a KeyError there is not in its handler --
+ # it would unwind into multiprocessing's unbounded join and hang the runner it is
+ # trying to free. Same reason summarize() below reads cells as `r.get('cells') or {}`.
+ md = render_matrix([(r.get('board', '?'), r.get('cells') or {}, r.get('duration'))
+ for r in doc.get('rows') or [] if isinstance(r, dict)])
+ if doc.get('scope'):
+ # a scoped run's small table is otherwise indistinguishable from a full one, and
+ # it replaces the previous full table in the sticky PR comment
+ md = f'_Scoped run: {doc["scope"]}. Boards/tests not listed were not run._\n\n' + md
+ # banner, then caveat: a rig-health caveat outranks the table AND the scope note, and an
+ # abandon notice outranks even that -- the top of the report is where hil/SKILL.md tells
+ # the agent to look
+ if doc.get('banner'):
+ md = doc['banner'] + '\n' + md
+ if doc.get('caveat'):
+ md = doc['caveat'] + '\n' + md
+ return md
+
+
+def write_report(report_dir: Path, doc: dict) -> None:
+ """Write both artifacts from one document.
+
+ RAISES on failure, deliberately: every caller is on a path whose own handler exists to
+ report exactly this (write_timeout_report's _p warning, hil_test's fallback-of-the-
+ fallback). Swallowing OSError here made both of those dead code, so an unwritable or
+ root-owned report dir produced no artifact AND no message.
+
+ Renders BEFORE writing anything: committing the JSON first and then raising in
+ render_report left a sidecar saying "abandoned" beside a markdown still reading as a
+ clean green table -- the one invariant this module exists to hold."""
+ md = render_report(doc) + '\n'
+ report_dir.mkdir(parents=True, exist_ok=True)
+ (report_dir / REPORT_JSON).write_text(json.dumps(doc, indent=2) + '\n')
+ (report_dir / REPORT_MD).write_text(md, encoding='utf-8')
+
+
+def _abandon_notice(why: str) -> str:
+ # Wording is a CONTRACT: .claude/skills/hil/SKILL.md pins this banner as the case where
+ # "the table below IS this run's ... Report the results AND the abandonment". Calling
+ # the table partial would send the reading agent to re-run boards that already passed.
+ return (f'**HIL run abandoned: {why}** The table below was collected before the '
+ f'abandon; treat board results as unverified.\n')
+
+
+def _already_abandoned(doc: dict) -> bool:
+ """Whether THIS attempt already recorded how it ended.
+
+ `caveat` only. It used to check `banner` too, because hil_test.py folded its abandon
+ notices in there -- but banner is carried across an --accumulate retry by design, so a
+ stale notice from an earlier attempt silenced a genuinely new abandon and the run's own
+ failure went unrecorded. banner now carries rig HEALTH (which describes the conditions
+ the cells were collected under, and so must persist); caveat carries the run's OUTCOME
+ (which must not)."""
+ return '**HIL run ab' in doc.get('caveat', '')
+
+
+def _stamp_markdown(report_dir: Path, notice: str) -> None:
+ """Last line of defence: prepend the notice to the markdown itself.
+
+ pr_comment.yml cats only hil_report.md, so a path that gives up here publishes a clean
+ green table under an abandoned, non-zero job. Master did this unconditionally."""
+ mpath = report_dir / REPORT_MD
+ if not mpath.is_file():
+ return
+ # errors='replace' and catch ValueError: a torn report or a LANG=C locale raises
+ # UnicodeDecodeError -- NOT an OSError -- straight past os._exit.
+ body = mpath.read_text(encoding='utf-8', errors='replace')
+ if '**HIL run ab' not in body[:2000]:
+ mpath.write_text(notice + '\n' + body, encoding='utf-8')
+
+
+def mark_report_abandoned(report_dir: Path, why: str) -> None:
+ """Stamp an existing report as abandoned, in BOTH artifacts.
+
+ Best-effort and silent: this runs while the interpreter is being torn down, and an
+ exception here hangs the process in multiprocessing's unbounded join()."""
+ notice = _abandon_notice(why)
+ try:
+ doc, readable = _load(report_dir)
+ if readable:
+ if _already_abandoned(doc):
+ return # whoever got there first wins, WRITE included
+ doc['caveat'] = notice
+ write_report(report_dir, doc)
+ return
+ except (OSError, ValueError, TypeError, AttributeError):
+ pass # fall through -- a failure here must not cost the stamp entirely
+ # Unreadable sidecar, or the document write failed. Either way the markdown is what
+ # the PR comment reads, so stamp it directly rather than giving up.
+ try:
+ _stamp_markdown(report_dir, notice)
+ except (OSError, ValueError, TypeError, AttributeError):
+ pass
+
+
+def mark_report_no_boards(report_dir: Path, msg: str, fresh: bool = True) -> None:
+ """Record that the board filters intersected to nothing.
+
+ `fresh` mirrors hil_test's own flag, because this runs BEFORE the fresh wipe: without
+ it a fresh run whose filter emptied re-published the PREVIOUS run's green rows under
+ this run's red job -- the stale-table failure it exists to prevent. An --accumulate run
+ keeps them, since nothing this attempt did invalidates them."""
+ try:
+ doc, _ = _load(report_dir)
+ if not fresh and _already_abandoned(doc):
+ # SKILL.md gives the two notices OPPOSITE rules, and an abandon outranks a
+ # filter that matched nothing -- do not overwrite the record of a failed run.
+ # Only while ACCUMULATING, though: this runs before the fresh wipe, so guarding
+ # a fresh run would leave the previous attempt's rows AND its abandon notice
+ # published as this run's.
+ return
+ # A fresh run carries NOTHING from the prior sidecar -- rows, banner and scope
+ # alike, matching accumulate_report, which builds from an empty prior when fresh.
+ # Resetting only rows republished a stale rig-health note and a stale scope line
+ # under this run's notice, from a leftover or uploaded sidecar.
+ prior = {'rows': [], 'banner': '', 'scope': ''} if fresh else doc
+ write_report(report_dir, {'rows': prior['rows'], 'banner': prior['banner'],
+ 'scope': prior['scope'],
+ 'caveat': f'**HIL run selected no boards.** {msg}\n'})
+ except (OSError, ValueError, TypeError, AttributeError):
+ pass # loud on stdout already; the exit code is what the job reads
+
+
+def accumulate_report(mret: list, report_dir: Path, fresh: bool, scope: str = '',
+ banner: str = '', caveat: str = '') -> str:
+ """Merge this run's results into json in report_dir, then (re)write
+ the markdown matrix to md. `fresh` (a first run, no --accumulate)
+ starts a new report; otherwise a re-run accumulates so boards/tests that
+ already passed are preserved while re-run cells are updated. `scope` names the
+ board filter, if any, so a scoped table is not mistaken for a full one.
+ Returns the md.
+
+ `mret` is hil_test.py's worker-result shape (name, err, fts, rows, ...), so this one
+ function knows something about its caller that the rest of the module does not. Folding
+ mret into rows could live in hil_test and only the merge here, but that would rewrite
+ the subtle parts -- stale board-locked clearing, BOUNDARY_CELL dropping, duration=None
+ preservation -- for a tidier seam. Data-shape coupling, not an import cycle."""
+ # ONE canonical load: a sidecar reaching here may have been uploaded by hil_ci.sh as
+ # the merge base, so it is untrusted input. `banner` carries forward -- it describes
+ # the conditions the earlier cells were collected under, and the .failed spec re-runs
+ # only FAILURES so those passes are never re-earned. `caveat` does NOT: it records how
+ # a RUN ENDED, and this attempt has not ended yet. Carrying it made a clean retry
+ # publish "HIL run abandoned" over a run where nothing was abandoned.
+ prior = {'rows': [], 'banner': ''}
+ if not fresh:
+ prior, _ = _load(report_dir)
+ acc = {r['board']: [dict(r['cells']), r['duration']] for r in prior['rows']}
+ prior_banner = prior['banner']
+
+ # current cells override prior for boards/tests that ran; a filtered run reports
+ # duration None, keeping the previous full-run value
+ for name, _, _, rows, *_ in mret:
+ if rows and not any(LOCKED_CELL in cells for _, cells, _ in rows):
+ # board ran for real: clear a stale lock-failure cell (its row is keyed by
+ # board name; test rows may be variant names)
+ stale = acc.get(name)
+ if stale is not None:
+ stale[0].pop(LOCKED_CELL, None)
+ # and the pool-timeout mark: write_timeout_report stamps it on a board that
+ # never reported, and update() below MERGES, so without this a board that
+ # passed clean on the retry kept a red cell for ever.
+ stale[0].pop(POOL_TIMEOUT_CELL, None)
+ stale[0].pop(RUN_ABORTED_CELL, None)
+ if not stale[0]:
+ # variant-keyed boards never repopulate the board-name row, so drop it
+ # or it renders as a blank ghost row
+ del acc[name]
+ for row_label, cells, dur in rows:
+ row = acc.setdefault(row_label, [{}, None])
+ # a row that ran is no longer pool-timed-out, whatever it is keyed by
+ row[0].pop(POOL_TIMEOUT_CELL, None)
+ row[0].pop(RUN_ABORTED_CELL, None)
+ # the boundary cell is only ever written on failure, so a re-run of this
+ # variant that cleared the boundary must drop the previous attempt's โŒ
+ if BOUNDARY_CELL not in cells:
+ row[0].pop(BOUNDARY_CELL, None)
+ row[0].update(cells)
+ if dur is not None:
+ row[1] = dur
+
+ report_dir.mkdir(parents=True, exist_ok=True)
+ # by LINE, deduped: attempts repeat the same caveat far more often than they add a new
+ # one, and three copies of the D-state note reads as three incidents
+ seen, merged = set(), []
+ for line in (prior_banner + banner).splitlines():
+ if line.strip() and line not in seen:
+ seen.add(line)
+ merged.append(line)
+ banner = '\n'.join(merged) + '\n' if merged else ''
+ doc = {'rows': [{'board': k, 'cells': c, 'duration': d} for k, (c, d) in acc.items()],
+ 'banner': banner, 'scope': scope, 'caveat': caveat}
+ # through write_report, not hand-rolled: writing the JSON and only then rendering is
+ # the ordering write_report exists to forbid -- a render failure left the sidecar ahead
+ # of the markdown, which is the one invariant this module holds.
+ write_report(report_dir, doc)
+ return render_report(doc)
+
+
+def _write_stuck_over_prior_md(report_dir: Path, doc: dict) -> None:
+ """Sidecar unrecoverable: rebuild it from the stuck rows alone, but leave the
+ markdown's existing table beneath the caveat rather than throwing real results away.
+
+ The one place the md-is-a-rendering-of-the-json invariant is deliberately suspended,
+ because there is no readable json left for it to be a rendering of."""
+ try:
+ prior = (report_dir / REPORT_MD).read_text(encoding='utf-8')
+ except (OSError, ValueError):
+ prior = ''
+ # Say so explicitly: those rows exist only as rendered text, so no later --accumulate
+ # can merge them back. Claiming the sidecar represents them would be false.
+ note = ('_The table below is a previous attempt\'s rendered output. The sidecar could '
+ 'not be read, so those rows are NOT in it and will not survive another run._\n')
+ head = (doc['banner'] + '\n' if doc['banner'] else '') + doc['caveat'] + '\n' + note
+ body = prior if prior.strip() else render_matrix(
+ [(r['board'], r['cells'], r['duration']) for r in doc['rows']])
+ report_dir.mkdir(parents=True, exist_ok=True)
+ (report_dir / REPORT_JSON).write_text(json.dumps(doc, indent=2) + '\n')
+ (report_dir / REPORT_MD).write_text(head + '\n' + body, encoding='utf-8')
+
+
+def write_timeout_report(report_dir: Path, boards, secs: int,
+ banner: str = '', prefix: str = '',
+ cell: str = POOL_TIMEOUT_CELL) -> None:
+ """Leave a report behind when the worker pool has to be abandoned.
+
+ map_async is all-or-nothing, so a timeout loses every per-board result and the report
+ dir would stay empty with no reason for the failure. Any prior attempt's rows are kept
+ and each stuck board is marked with a POOL_TIMEOUT_CELL beside them.
+
+ `prefix` is the preflight rig-health verdict and goes to the BANNER, where rig health
+ lives and where an --accumulate retry carries it forward; the abandon notice goes to
+ the caveat, which does not carry. Folding both into the caveat is what made a clean
+ retry report an abandonment that had not happened."""
+ try:
+ # names INSIDE the try: a roster entry that is not a dict raises here, and outside
+ # it that escaped and stranded the runner.
+ names = [b.get('name', '?') if isinstance(b, dict) else '?' for b in boards]
+ caveat = banner or (
+ f'**HIL run abandoned: worker pool timed out after {secs}s.**\n\n'
+ f'No per-board results could be collected for this attempt. Rows other than '
+ f'the {cell} cells below are from an earlier attempt. Boards '
+ f'dispatched:\n\n' + '\n'.join(f'- {n}' for n in names) + '\n')
+ doc, readable = _load(report_dir)
+ rows = doc['rows']
+ by_board = {r['board']: r for r in rows}
+ for name in names:
+ row = by_board.get(name)
+ if row is None:
+ rows.append({'board': name, 'cells': {cell: 'fail'},
+ 'duration': None})
+ else:
+ # _load guarantees `cells` is a dict, so a null-cells row from an uploaded
+ # sidecar can no longer send this down the fallback and publish a board
+ # that ate the whole pool guard as a pass.
+ row['cells'][cell] = 'fail'
+ out = {'rows': rows, 'scope': doc['scope'], 'caveat': caveat,
+ 'banner': ((doc['banner'] + prefix) if prefix not in doc['banner']
+ else doc['banner'])}
+ if not readable and (report_dir / REPORT_MD).is_file():
+ # `readable` covers ABSENT as well as torn: an absent sidecar beside an intact
+ # markdown used to re-render from the stuck row alone and destroy real results.
+ _write_stuck_over_prior_md(report_dir, out)
+ return
+ write_report(report_dir, out)
+ except Exception as e: # noqa: BLE001
+ # Deliberately broad: this is the first statement of the pool-abandon path, so ANY
+ # escape skips kill_pool_children and os._exit and strands the runner.
+ _p(f'warning: cannot write {REPORT_MD} to {report_dir}: {e}', flush=True)
+ try:
+ # Same wording as above and the same guarded name extraction -- the fallback
+ # used to re-derive b.get("name") outside any try and raise identically, so a
+ # malformed roster left NO artifact at all.
+ names = [b.get('name', '?') if isinstance(b, dict) else '?' for b in boards]
+ head = (prefix + '\n' if prefix else '') + (banner or (
+ f'**HIL run abandoned: worker pool timed out after {secs}s.**\n\n'
+ f'No per-board results could be collected for this attempt, so the table '
+ f'below (if any) is from an earlier one. Boards dispatched:\n\n'
+ + '\n'.join(f'- {n}' for n in names) + '\n'))
+ try:
+ prior = (report_dir / REPORT_MD).read_text(encoding='utf-8')
+ except (OSError, ValueError):
+ prior = ''
+ report_dir.mkdir(parents=True, exist_ok=True)
+ (report_dir / REPORT_MD).write_text(
+ head + (f'\n{prior}' if prior else ''), encoding='utf-8')
+ except Exception as e2: # noqa: BLE001
+ _p(f'warning: fallback {REPORT_MD} write failed too: {e2}', flush=True)
+
+
+def variants_of(cfg: dict, board: str) -> list:
+ for b in cfg.get('boards', []):
+ if b['name'] == board:
+ return [v['name'] for v in (b.get('variant') or [])] or [board]
+ return [board]
+
+
+def summarize(cfg: dict, boards: list, report: dict) -> dict:
+ # .get, not a subscript: this is the one reader an agent's verdict depends on, and a
+ # row without 'board' used to kill the CLI with a traceback and no results at all --
+ # hil-validate.js then reports every board as "hil-operator returned no entry".
+ rows = {r['board']: r.get('cells') or {}
+ for r in (report.get('rows') or [])
+ if isinstance(r, dict) and 'board' in r}
+ owner = {v['name']: b['name'] for b in cfg.get('boards', [])
+ for v in (b.get('variant') or [])}
+ results = []
+ for board in boards:
+ names = variants_of(cfg, board)
+ mine = {n: rows[n] for n in names if n in rows}
+ # a variant name that is neither declared nor prefixed cannot be attributed; the
+ # `<board>-` fallback only helps ad-hoc builds, it is not the primary path. It must
+ # also never steal a row DECLARED by another board: a declared variant need not start
+ # with its own board's name, so it may happen to start with this board's name plus '-'.
+ mine.update({n: c for n, c in rows.items()
+ if n.startswith(f'{board}-') and n not in mine
+ and owner.get(n, board) == board})
+ # the BOARD-name row too: hil_test writes lock contention and pool timeouts keyed
+ # by board name, but variants_of returns only DECLARED variant names -- and
+ # nanoch32v203 / ch32v307v_r1_1v0 declare none equal to their board name. Without
+ # this those rows are invisible, so a lock held by concurrent CI is published as a
+ # hardware FAIL and hil-validate.js never retries it.
+ if board in rows and board not in mine:
+ mine[board] = rows[board]
+ if not mine:
+ results.append({'board': board, 'ran': False, 'pass': False, 'locked': False,
+ 'detail': 'no report row for this board'})
+ continue
+ # a wedge outranks lock contention: `locked` short-circuits `detail` below, so a
+ # stale board-locked cell from an earlier attempt used to mask the pool-timeout
+ # cell the retry added -- publishing a board that hung the rig as LOCKED, which
+ # hil-validate.js then RE-RUNS, paying another pool guard on it. RUN_ABORTED_CELL
+ # is written by the same _abort_report path for a board the guard never reached,
+ # and must outrank it for the same reason.
+ wedged = any(POOL_TIMEOUT_CELL in cells or RUN_ABORTED_CELL in cells
+ for cells in mine.values())
+ locked = not wedged and any(LOCKED_CELL in cells for cells in mine.values())
+ bad = []
+ for vname, cells in sorted(mine.items()):
+ for test, val in sorted(cells.items()):
+ if test == LOCKED_CELL:
+ continue
+ if cell_state(val) == 'fail':
+ bad.append(f'{vname} {test}: {val}')
+ ok = not bad and not locked
+ if locked:
+ detail = 'held by another holder; not flashed'
+ elif bad:
+ detail = '; '.join(bad)
+ else:
+ detail = f'{len(mine)} variant(s), {sum(len(c) for c in mine.values())} cell(s) ok'
+ results.append({'board': board, 'ran': True, 'pass': ok, 'locked': locked,
+ 'detail': detail})
+ # `caveat` too: an abandoned or no-boards run says so THERE, and this JSON is all
+ # an agent gets -- leaving it in the sidecar puts it back where only a human looks.
+ return {'results': results, 'banner': report.get('banner', ''),
+ 'caveat': report.get('caveat', '')}
+
+
+def main() -> int:
+ ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
+ ap.add_argument('config_file')
+ ap.add_argument('-b', '--board', action='append', default=[],
+ help='boards to report on; default: every board in the config')
+ ap.add_argument('--report-dir', default='.', help=f'where {REPORT_JSON} lives (default: cwd)')
+ a = ap.parse_args()
+
+ cfg = json.loads(Path(a.config_file).read_text())
+ boards = a.board or [b['name'] for b in cfg.get('boards', [])]
+ jpath = Path(a.report_dir) / REPORT_JSON
+ if not jpath.is_file():
+ print(f'error: {jpath} not found -- did hil_test.py run in this directory?',
+ file=sys.stderr)
+ return 1
+ # through _load, like every writer: feeding raw JSON to summarize left the one reader an
+ # agent's verdict depends on crashing on the malformed sidecars the writers tolerate.
+ doc, _ = _load(Path(a.report_dir))
+ json.dump(summarize(cfg, boards, doc), sys.stdout, indent=2)
+ print()
+ return 0
+
+if __name__ == '__main__':
+ sys.exit(main())
diff --git a/test/hil/helper/hil_select.py b/test/hil/helper/hil_select.py
deleted file mode 100755
index f0d4f0b9f..000000000
--- a/test/hil/helper/hil_select.py
+++ /dev/null
@@ -1,524 +0,0 @@
-#!/usr/bin/env python3
-# SPDX-License-Identifier: MIT
-"""PR-diff -> HIL selection: which rig boards and which tests a change can affect.
-
-Stdlib-only (runs on bare CI runners; imports hil_util for the example rosters,
-never hil_test/pyserial โ€” test_hil_util.BottomLayer enforces the stdlib closure).
-Fail-open: any file no rule classifies forces the full matrix. See
-docs/superpowers/specs/2026-07-29-hil-pr-scoped-selection-design.md.
-
-JSON: full, boards (name -> 'all' | [tests]), families (bsp families the diff
-touches, including ones with no rig board - build-only consumers such as /pre-pr
-sample from these), args (hil_test.py args per config) and args_flasher (the same
-args split by each board's flasher, for CI legs that split one rig by flasher).
-"""
-import argparse
-import functools
-import glob
-import json
-import os
-import re
-import subprocess
-import sys
-
-sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # helper/ scripts import via the test/hil root
-from helper.hil_util import device_tests, dual_tests, host_test
-
-ALL_TESTS = {'device': device_tests, 'dual': dual_tests, 'host': host_test}
-
-# class dir -> config macro suffix exceptions (rule 3); dfu is per-file, handled inline
-NET_MACROS = ('ECM_RNDIS', 'NCM')
-
-_NONCODE_RE = re.compile(
- r'^(docs/|\.claude/|.*\.(md|rst)$|LICENSE)')
-_FULL_RE = re.compile(
- r'^(src/common/|src/osal/|src/tusb\.c$|src/tusb\.h$|src/tusb_option\.h$|'
- r'test/hil/|\.github/workflows/build.*\.yml$|\.github/actions/|\.github/scripts/|'
- r'tools/build\.py$|tools/get_deps\.py$|tools/cmake/|hw/mcu/|lib/|'
- r'hw/bsp/(family_support\.cmake|board_api\.h|board\.c|ansi_escape\.h)$|'
- r'examples/build_system/|examples/CMakeLists\.txt$|'
- # board_test is HIL infrastructure, not a test: hil_test.py flashes it to park
- # every board (variant boundary + end-of-board teardown), so every board depends on it
- r'examples/device/board_test/)')
-
-# --no-renames: with rename detection git reports only a rename's destination, so code
-# moved out of an HIL-relevant path would be classified by its new path alone
-GIT_DIFF_ARGV = ['git', 'diff', '--no-renames', '--name-only']
-
-
-def test_role(test: str) -> str:
- return test.split('/', 1)[0] # 'device' | 'dual' | 'host'
-
-
-def board_roles(board: dict) -> set:
- t = board.get('tests', {})
- roles = set()
- if t.get('device'):
- roles.add('device')
- if t.get('host'):
- roles.add('host')
- if t.get('dual'):
- roles.update(('device', 'host'))
- for only in t.get('only', []):
- r = test_role(only)
- roles.update(('device', 'host') if r == 'dual' else (r,))
- return roles
-
-
-def board_tests(board: dict) -> list:
- """Every test this board would run today (mirrors hil_test.test_board's default)."""
- t = board.get('tests', {})
- if 'only' in t:
- run = list(t['only'])
- else:
- run = []
- if t.get('device'):
- run += device_tests
- if t.get('dual'):
- run += dual_tests
- if t.get('host'):
- run += host_test
- return [x for x in run if x not in t.get('skip', [])]
-
-
-# cached: called per changed file x roster board, and the tree doesn't change mid-run
[email protected]_cache(maxsize=None)
-def board_family(board_name: str, repo_root: str):
- hits = glob.glob(os.path.join(repo_root, 'hw/bsp/*/boards', board_name))
- return os.path.basename(os.path.dirname(os.path.dirname(hits[0]))) if hits else None
-
-
-# `if (OPTION STREQUAL "1")` guards in family_support.cmake, and the option tokens
-# a roster entry passes to the build (NAME=VALUE / -DNAME=VALUE)
-_CM_IF_RE = re.compile(r'if\s*\(')
-_CM_ELSE_RE = re.compile(r'else(if)?\s*\(')
-_CM_ENDIF_RE = re.compile(r'endif\s*\(')
-_CM_OPT_RE = re.compile(r'if\s*\(\s*\$?\{?([A-Za-z_]\w*)\}?\s+STREQUAL\s+"?1"?\s*\)')
-_CM_PORT_RE = re.compile(r'src/portable/((?:[^/\s]+/)?[^/\s]+)/')
-_FALSY = ('', '0', 'off', 'false', 'no')
-
-
[email protected]_cache(maxsize=None)
-def port_option_gates(repo_root: str) -> dict:
- """port dir -> build options that compile it regardless of the board's family
- file, e.g. {'analog/max3421': {'MAX3421_HOST'}} from family_support.cmake."""
- gates = {}
- try:
- text = open(os.path.join(repo_root, 'hw/bsp/family_support.cmake')).read()
- except OSError:
- return gates
- stack = [] # one entry per open if(): its option, or None
- for line in text.splitlines():
- line = line.strip()
- if _CM_IF_RE.match(line):
- m = _CM_OPT_RE.match(line)
- stack.append(m.group(1) if m else None)
- elif _CM_ELSE_RE.match(line):
- if stack:
- stack[-1] = None # the guard doesn't hold in this branch
- elif _CM_ENDIF_RE.match(line):
- if stack:
- stack.pop()
- opts = {o for o in stack if o}
- m = _CM_PORT_RE.search(line)
- if opts and m:
- gates.setdefault(m.group(1), set()).update(opts)
- return gates
-
-
-_CM_SET_RE = re.compile(r'set\s*\(\s*([A-Za-z_]\w*)\s+([^)\s]+)\s*\)')
-
-
-# cached: called per changed portable file x roster board
[email protected]_cache(maxsize=None)
-def bsp_board_options(board_name: str, repo_root: str) -> frozenset:
- """Build options a board turns on in its own BSP: `set(<OPT> <value>)` in
- hw/bsp/<family>/boards/<board>/board.cmake, e.g. MAX3421_HOST on the espressif
- and rp2040 max3421 boards. CMake only - HIL CI builds nothing with Make, so a
- board.mk-only option (e.g. nrf5340dk's MAX3421_HOST) compiles no port here."""
- fam = board_family(board_name, repo_root)
- if not fam:
- return frozenset()
- path = os.path.join(repo_root, 'hw/bsp', fam, 'boards', board_name, 'board.cmake')
- try:
- text = open(path).read()
- except OSError:
- return frozenset()
- out = set()
- for line in text.splitlines():
- line = line.strip()
- if line.startswith('#'):
- continue
- m = _CM_SET_RE.match(line)
- if m and m.group(2).strip('"').lower() not in _FALSY:
- out.add(m.group(1))
- return frozenset(out)
-
-
-def board_options(board: dict, repo_root: str) -> set:
- """Build options a board has truthy: the roster entry's build.args plus each
- variant's defines (NAME=VALUE) and raw CFLAGS (-DNAME=VALUE), plus whatever its
- own board.cmake sets (a board can enable a gated port without the roster saying so)."""
- toks = list(board.get('build', {}).get('args', []))
- for v in board.get('variant', []):
- toks += list(v.get('defines', []))
- toks += v.get('flags', '').split()
- out = set(bsp_board_options(board['name'], repo_root))
- for t in toks:
- name, _, val = (t[2:] if t.startswith('-D') else t).partition('=')
- if name and val.strip().strip('"').lower() not in _FALSY:
- out.add(name.strip())
- return out
-
-
[email protected]_cache(maxsize=None)
-def port_families(port_dir: str, repo_root: str) -> set:
- """Board families that compile this src/portable dir. CMake only: HIL CI builds
- every board with CMake, so a port wired up in family.mk alone is compiled for no
- HIL board and must not select one. family.cmake lists portable sources directly
- for most families; espressif instead references them from a nested component
- CMakeLists.txt (hw/bsp/espressif/components/tinyusb_src/CMakeLists.txt)."""
- fams = set()
- bsp_root = os.path.join(repo_root, 'hw/bsp')
- # trailing '/' so a port dir is not a prefix of a sibling: bare 'microchip/pic'
- # would otherwise match '.../microchip/pic32mz/...' and inherit its families
- needle = port_dir + '/'
- for f in glob.glob(os.path.join(bsp_root, '*/family.cmake')) + \
- glob.glob(os.path.join(bsp_root, '*/components/*/CMakeLists.txt')):
- try:
- if needle in open(f).read():
- fam = os.path.relpath(f, bsp_root).split(os.sep, 1)[0]
- fams.add(fam)
- except OSError:
- pass
- return fams
-
-
-_CLS_INC_RE = re.compile(r'#\s*include\s*[<"]class/([^/"<>]+)/([^"<>]+)[">]')
-
-
[email protected]_cache(maxsize=None)
-def class_include_edges(repo_root: str) -> dict:
- """'<class>/<header>' -> the other class dirs that include it. A class header
- pulled in by a second class ships in every firmware enabling that second class:
- src/class/midi/midi{,2}_{device,host}.h include class/audio/audio.h, and
- net_device.h includes class/cdc/cdc.h. The class rule derives macros from the
- directory name alone, so without this edge a change to the included header
- selects only its own class's examples - and on a board that skips those (e.g.
- metro_m4_express skips audio_test_freertos), nothing at all.
-
- Derived from the actual #include lines rather than a hand-written table so it
- cannot rot when a class picks up or drops a cross-class include."""
- edges = {}
- for f in sorted(glob.glob(os.path.join(repo_root, 'src/class/*/*.[ch]'))):
- cls = os.path.basename(os.path.dirname(f))
- try:
- text = open(f).read()
- except OSError:
- continue
- for inc_cls, inc_hdr in _CLS_INC_RE.findall(text):
- if inc_cls != cls:
- edges.setdefault(f'{inc_cls}/{inc_hdr}', set()).add(cls)
- return edges
-
-
-def class_macros(cls: str, base: str, prefix: str) -> list:
- """Config macros that compile a class dir's code, for role prefix TUD/TUH.
- `base` refines dfu only (it splits DFU from DFU_RUNTIME per file); pass '' for
- a class reached through an include edge, where the widest set is correct."""
- if cls == 'net':
- return [f'CFG_{prefix}_{m}' for m in NET_MACROS]
- if cls == 'dfu':
- if base.startswith('dfu_rt'):
- return [f'CFG_{prefix}_DFU_RUNTIME']
- if base.startswith('dfu_device') or base.startswith('dfu_host'):
- return [f'CFG_{prefix}_DFU']
- return [f'CFG_{prefix}_DFU', f'CFG_{prefix}_DFU_RUNTIME']
- return [f'CFG_{prefix}_{cls.upper()}']
-
-
-def _config_enables(cfg_path: str, macros) -> bool:
- try:
- text = open(cfg_path).read()
- except OSError:
- return False
- return any(re.search(rf'#define\s+{m}\s+\(?\s*0*[1-9]', text) for m in macros)
-
-
-def roster_only_tests(all_boards) -> set:
- """Test paths that only appear in a roster board's tests.only list (e.g.
- espressif boards), not in the shared device/dual/host_test lists."""
- out = set()
- for b in all_boards:
- out.update(b.get('tests', {}).get('only', []))
- return out
-
-
-def class_examples(macros, role: str, repo_root: str, extra_tests: set) -> set:
- """Tests (from role's + dual lists, plus roster-only-list tests of that role)
- whose example config enables any macro."""
- pool = role_tests({role}, extra_tests)
- out = set()
- for test in pool:
- cfg = os.path.join(repo_root, 'examples', test, 'src', 'tusb_config.h')
- if _config_enables(cfg, macros):
- out.add(test)
- return out
-
-
-def role_tests(roles: set, extras: set) -> set:
- """Every test for the given role(s): each role's own list + dual tests,
- plus roster-only-list tests (extras) matching those roles or 'dual'."""
- pool = set(dual_tests)
- for r in roles:
- pool |= set(ALL_TESTS[r])
- pool |= {t for t in extras if test_role(t) in roles or test_role(t) == 'dual'}
- return pool
-
-
-class _Sel:
- """Accumulates contributions. board->set(tests) plus 'all-board' markers."""
- def __init__(self):
- self.full = False
- self.by_board = {} # name -> set of tests, or 'all'
- self.roles = set() # roles touched by any contribution
- self.families = set() # bsp families touched (incl. off-rig ones: build-only consumers)
- self.reasons = []
-
- def add(self, boards, tests, reason):
- """tests: 'all' or iterable of test paths."""
- self.reasons.append(reason)
- for b in boards:
- cur = self.by_board.get(b)
- if tests == 'all' or cur == 'all':
- self.by_board[b] = 'all'
- else:
- self.by_board[b] = (cur or set()) | set(tests)
-
- def force_full(self, reason):
- self.full = True
- self.reasons.append(reason)
-
-
-def _classify_one(path, repo_root, roster_boards, extras: set, s: _Sel):
- base = os.path.basename(path)
- if _NONCODE_RE.match(path):
- s.reasons.append(f'{path}: non-code, no contribution')
- return
- if _FULL_RE.match(path):
- s.force_full(f'{path}: core/infra -> full matrix')
- return
-
- m = re.match(r'src/portable/((?:[^/]+/)?[^/]+)/', path)
- if m:
- port = m.group(1)
- if re.match(r'(dcd_|.*_device)', base):
- roles = {'device'}
- elif re.match(r'(hcd_|.*_host)', base):
- roles = {'host'}
- else:
- roles = {'device', 'host'}
- fams = port_families(port, repo_root)
- if not fams:
- # no family references this port: either a new/renamed port dir or a
- # family.cmake layout the scan misses - widen instead of contributing nothing
- s.force_full(f'{path}: port {port} maps to no board family -> full matrix')
- return
- s.families.update(fams)
- # a board can also pull the port in through a build option (e.g. MAX3421_HOST=1
- # from the roster on metro_m4_express, or from its own board.cmake), which its
- # family file never names
- gates = port_option_gates(repo_root).get(port, set())
- boards = [b['name'] for b in roster_boards
- if (board_family(b['name'], repo_root) in fams or
- (gates and board_options(b, repo_root) & gates)) and (board_roles(b) & roles)]
- tests = role_tests(roles, extras)
- s.roles.update(roles)
- why = f'{path}: port {port} -> families {sorted(fams)}'
- if gates:
- why += f' + option {sorted(gates)}'
- s.add(boards, tests, f'{why} -> boards {boards} ({"/".join(sorted(roles))})')
- return
-
- m = re.match(r'src/class/([^/]+)/', path)
- if m:
- cls = m.group(1)
- if re.search(r'_device\.[ch]$', base):
- roles = {'device'}
- elif re.search(r'_host\.[ch]$', base):
- roles = {'host'}
- else:
- roles = {'device', 'host'}
- # this file's own class, plus any class whose headers include it
- via = sorted(class_include_edges(repo_root).get(f'{cls}/{base}', ()))
-
- def macros(prefix):
- return (class_macros(cls, base, prefix) +
- [m2 for c in via for m2 in class_macros(c, '', prefix)])
- tests = set()
- if 'device' in roles:
- tests |= class_examples(macros('TUD'), 'device', repo_root, extras)
- if 'host' in roles:
- tests |= class_examples(macros('TUH'), 'host', repo_root, extras)
- boards = [b['name'] for b in roster_boards if board_roles(b) & roles]
- s.roles.update(roles)
- why = f'{path}: class {cls}' + (f' (+ included by {via})' if via else '')
- s.add(boards, tests, f'{why} -> {sorted(tests)} ({"/".join(sorted(roles))})')
- return
-
- m = re.match(r'src/(device|host)/', path)
- if m:
- role = m.group(1)
- boards = [b['name'] for b in roster_boards if role in board_roles(b)]
- s.roles.add(role)
- s.add(boards, role_tests({role}, extras), f'{path}: core {role} stack -> all {role} tests')
- return
-
- m = re.match(r'hw/bsp/([^/]+)/(?:boards/([^/]+)/)?', path)
- if m:
- fam, brd = m.group(1), m.group(2)
- s.families.add(fam)
- if brd:
- boards = [b['name'] for b in roster_boards if b['name'] == brd]
- why = f'{path}: bsp board {brd}'
- else:
- boards = [b['name'] for b in roster_boards
- if board_family(b['name'], repo_root) == fam]
- why = f'{path}: bsp family {fam}'
- s.roles.update(('device', 'host'))
- s.add(boards, 'all', f'{why} -> boards {boards}')
- return
-
- m = re.match(r'examples/(device|host|dual)/([^/]+)/', path)
- if m:
- test = f'{m.group(1)}/{m.group(2)}'
- known = any(test in pool for pool in ALL_TESTS.values()) or test in extras
- if known:
- boards = [b['name'] for b in roster_boards]
- role = test_role(test)
- s.roles.update(('device', 'host') if role == 'dual' else (role,))
- s.add(boards, [test], f'{path}: example -> {test} on all boards')
- else:
- s.reasons.append(f'{path}: example not in HIL lists, no contribution')
- return
-
- s.force_full(f'{path}: unclassified -> full matrix')
-
-
-def classify(changed_files, repo_root, rosters):
- all_boards = []
- seen = set()
- for _, boards in rosters:
- for b in boards:
- if b['name'] not in seen:
- seen.add(b['name'])
- all_boards.append(b)
-
- extras = roster_only_tests(all_boards)
- s = _Sel()
- # no early exit once full: keep classifying so `families` still reports every
- # family the diff touches (build-only consumers need it). Nothing after the first
- # force_full can change full/boards/args - the full branch below ignores by_board.
- for path in changed_files:
- _classify_one(path, repo_root, all_boards, extras, s)
-
- if s.full:
- return {'full': True, 'boards': {b['name']: 'all' for b in all_boards},
- 'families': sorted(s.families), 'reasons': s.reasons}
-
- # role pruning: single-role selections drop the other role's tests and boards
- by_name = {b['name']: b for b in all_boards}
- out = {}
- for name, tests in s.by_board.items():
- allowed = board_tests(by_name[name])
- if tests == 'all':
- kept = list(allowed)
- else:
- kept = [t for t in allowed if t in tests]
- if s.roles and s.roles != {'device', 'host'}:
- role = next(iter(s.roles))
- kept = [t for t in kept if test_role(t) in (role, 'dual')]
- if kept:
- out[name] = 'all' if set(kept) == set(allowed) else sorted(kept)
- return {'full': False, 'boards': out, 'families': sorted(s.families),
- 'reasons': s.reasons}
-
-
-def _board_args(name, chosen) -> list:
- parts = [f'-b {name}']
- if chosen != 'all':
- parts.append(f'-bt {name}:{",".join(chosen)}')
- return parts
-
-
-def selection_args(sel, rosters):
- """hil_test.py args per config. Empty means either 'full matrix' or 'nothing
- selected' - callers must read sel['full'] to tell them apart."""
- args = {}
- for cfg_path, boards in rosters:
- parts = []
- if not sel['full']:
- for b in boards:
- chosen = sel['boards'].get(b['name'])
- if chosen is not None:
- parts += _board_args(b['name'], chosen)
- args[os.path.basename(cfg_path)] = ' '.join(parts)
- return args
-
-
-def selection_args_by_flasher(sel, rosters):
- """{config: {flasher name: args}}. CI runs one rig as several jobs split by
- flasher (esptool vs the rest); each must gate on its own subset, otherwise the
- other leg runs a filter matching zero boards and reports a vacuous green."""
- out = {}
- for cfg_path, boards in rosters:
- per = {}
- if not sel['full']:
- for b in boards:
- chosen = sel['boards'].get(b['name'])
- if chosen is None:
- continue
- per.setdefault(b.get('flasher', {}).get('name', ''), []).extend(
- _board_args(b['name'], chosen))
- out[os.path.basename(cfg_path)] = {f: ' '.join(p) for f, p in per.items()}
- return out
-
-
-def changed_files_from_git(base, repo_root):
- mb = subprocess.run(['git', 'merge-base', 'HEAD', base], cwd=repo_root,
- capture_output=True, text=True, check=True).stdout.strip()
- diff = subprocess.run(GIT_DIFF_ARGV + [f'{mb}..HEAD'], cwd=repo_root,
- capture_output=True, text=True, check=True).stdout
- return [l for l in diff.splitlines() if l.strip()]
-
-
-def main():
- ap = argparse.ArgumentParser(description=__doc__)
- g = ap.add_mutually_exclusive_group(required=True)
- g.add_argument('--base', help='git ref to diff against (merge-base..HEAD)')
- g.add_argument('--diff-file', help='newline-separated changed-file list')
- ap.add_argument('configs', nargs='+', help='rig roster JSON file(s)')
- a = ap.parse_args()
-
- # test/hil/helper/ -> repo root is FOUR levels up; three left this at <repo>/test
- # after the helper/ move and every repo-relative glob silently matched nothing
- repo_root = os.path.dirname(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))))
- rosters = []
- for c in a.configs:
- with open(c) as f:
- rosters.append((c, json.load(f)['boards']))
-
- files = (open(a.diff_file).read().splitlines() if a.diff_file
- else changed_files_from_git(a.base, repo_root))
- files = [f for f in files if f.strip()]
-
- s = classify(files, repo_root, rosters)
- s['args'] = selection_args(s, rosters)
- s['args_flasher'] = selection_args_by_flasher(s, rosters)
- for r in s['reasons']:
- print(f'hil_select: {r}', file=sys.stderr)
- print(json.dumps(s))
-
-
-if __name__ == '__main__':
- main()
diff --git a/test/hil/helper/hil_summary.py b/test/hil/helper/hil_summary.py
deleted file mode 100644
index e566bead0..000000000
--- a/test/hil/helper/hil_summary.py
+++ /dev/null
@@ -1,115 +0,0 @@
-#!/usr/bin/env python3
-# SPDX-License-Identifier: MIT
-"""Fold hil_report.json into one machine-readable verdict per BOARD.
-
-A workflow driving hil_test.py through an operator agent has no filesystem access, so the
-agent has to carry the results across. It must carry them, not retype them: the previous
-design asked the agent to transcribe the markdown table, and every defect found in four
-review rounds came from re-parsing that prose -- variant row names vs board names,
-`board locked` vs `board-locked`, folding several variant rows into one verdict, rows that
-matched no board. All of it is a join, and the join belongs here, where the roster is.
-
-Report rows are named per VARIANT (hil_test.py builds them from `vname`), and a variant name
-is not required to start with the board name -- nanoch32v203 produces only `-fsdev`/`-usbfs`,
-ch32v307v_r1_1v0 only `-usbhs`/`-usbfs`. The config is what maps them back.
-
-Emits, on stdout:
- {"results": [{"board", "ran", "pass", "locked", "detail"}...], "banner": str}
-
-`locked` is a field, not a prefix to grep for. `ran` false means the board produced no row at
-all, which is not the same as failing.
-
-Usage: hil_summary.py <config.json> [-b BOARD]... [--report-dir DIR]
-"""
-import argparse
-import json
-import sys
-from pathlib import Path
-
-FAIL_ICON, SKIP_ICON = 'โŒ', 'โšช' # a pass needs no icon: unmarked = pass
-LOCKED_CELL = 'board-locked'
-
-
-def cell_state(v: str) -> str:
- """'pass' | 'fail' | 'skip' -- the EXACT classifier hil_test.py's own tally uses
- (cell_kind in render_matrix): 'fail' or a โŒ prefix is a failure, 'skip' or a โšช
- prefix is a skip, and EVERYTHING ELSE is a pass. That last arm is load-bearing: a
- passing test may return a plain metric string ('480.0 MBps') that lands in the cell
- unprefixed, while failures are guaranteed marked -- TestFail's docstring pins that its
- metric is icon-prefixed precisely so render/tally treat it as a failure. Classifying
- unknown shapes as fail here would publish a green table as a red verdict."""
- if v == 'fail' or v.startswith(FAIL_ICON):
- return 'fail'
- if v == 'skip' or v.startswith(SKIP_ICON):
- return 'skip'
- return 'pass'
-
-
-def variants_of(cfg: dict, board: str) -> list:
- for b in cfg.get('boards', []):
- if b['name'] == board:
- return [v['name'] for v in (b.get('variant') or [])] or [board]
- return [board]
-
-
-def summarize(cfg: dict, boards: list, report: dict) -> dict:
- rows = {r['board']: r.get('cells') or {} for r in report.get('rows', [])}
- owner = {v['name']: b['name'] for b in cfg.get('boards', [])
- for v in (b.get('variant') or [])}
- results = []
- for board in boards:
- names = variants_of(cfg, board)
- mine = {n: rows[n] for n in names if n in rows}
- # a variant name that is neither declared nor prefixed cannot be attributed; the
- # `<board>-` fallback only helps ad-hoc builds, it is not the primary path. It must
- # also never steal a row DECLARED by another board: a declared variant need not start
- # with its own board's name, so it may happen to start with this board's name plus '-'.
- mine.update({n: c for n, c in rows.items()
- if n.startswith(f'{board}-') and n not in mine
- and owner.get(n, board) == board})
- if not mine:
- results.append({'board': board, 'ran': False, 'pass': False, 'locked': False,
- 'detail': 'no report row for this board'})
- continue
- locked = any(LOCKED_CELL in cells for cells in mine.values())
- bad = []
- for vname, cells in sorted(mine.items()):
- for test, val in sorted(cells.items()):
- if test == LOCKED_CELL:
- continue
- if cell_state(str(val)) == 'fail':
- bad.append(f'{vname} {test}: {val}')
- ok = not bad and not locked
- if locked:
- detail = 'held by another holder; not flashed'
- elif bad:
- detail = '; '.join(bad)
- else:
- detail = f'{len(mine)} variant(s), {sum(len(c) for c in mine.values())} cell(s) ok'
- results.append({'board': board, 'ran': True, 'pass': ok, 'locked': locked,
- 'detail': detail})
- return {'results': results, 'banner': report.get('banner', '')}
-
-
-def main() -> int:
- ap = argparse.ArgumentParser()
- ap.add_argument('config_file')
- ap.add_argument('-b', '--board', action='append', default=[],
- help='boards to report on; default: every board in the config')
- ap.add_argument('--report-dir', default='.', help='where hil_report.json lives (default: cwd)')
- a = ap.parse_args()
-
- cfg = json.loads(Path(a.config_file).read_text())
- boards = a.board or [b['name'] for b in cfg.get('boards', [])]
- jpath = Path(a.report_dir) / 'hil_report.json'
- if not jpath.is_file():
- print(f'error: {jpath} not found -- did hil_test.py run in this directory?',
- file=sys.stderr)
- return 1
- json.dump(summarize(cfg, boards, json.loads(jpath.read_text())), sys.stdout, indent=2)
- print()
- return 0
-
-
-if __name__ == '__main__':
- sys.exit(main())
diff --git a/test/hil/helper/hil_util.py b/test/hil/helper/hil_util.py
index 54984d20f..6f84c143d 100644
--- a/test/hil/helper/hil_util.py
+++ b/test/hil/helper/hil_util.py
@@ -1,7 +1,8 @@
#!/usr/bin/env python3
# SPDX-License-Identifier: MIT
# Bottom layer of the HIL harness: the bounded command runner plus the shared helpers and
-# data every other module needs. Stays stdlib-only and imports nothing local -- everything
+# data every other module needs. Stays stdlib-only; its one local dependency is
+# tools/rtt.py (the RTT console, loaded by path below) -- everything
# else imports this, including the unit tests on GitHub's bare runner; never import them
# from here. Callers set the module global `verbose`.
@@ -11,6 +12,7 @@ import glob
import os
import signal
import subprocess
+import unicodedata
import threading
import sys
from pathlib import Path
@@ -18,7 +20,7 @@ from typing import Any
# -------------------------------------------------------------
-# HIL example test lists, shared by hil_test.py (runner) and hil_select.py (PR-diff
+# HIL example test lists, shared by hil_test.py (runner) and ci_select.py (PR-diff
# selector). Run order is shuffled per board (see test_board); every example carries a
# unique hardcoded idProduct (see its usb_descriptors.c).
# -------------------------------------------------------------
@@ -88,10 +90,33 @@ def pos_float_env(name: str, default: float) -> float:
CMD_TIMEOUT = pos_int_env('HIL_CMD_TIMEOUT', 180)
+# Post-SIGKILL reap, spent ON TOP of a run_cmd timeout whenever the child has to be killed.
+# A caller budgeting several bounded steps must add one of these PER STEP, or its own outer
+# bound fires mid-step -- for a flasher, orphaning it on the probe.
+REAP_GRACE = 10
TINYUSB_ROOT = Path(__file__).resolve().parents[3] # test/hil/helper/ -> repo root
+def display_width(s: str) -> int:
+ """Terminal COLUMNS, not characters.
+
+ The status marks the reports use -- โœ… โŒ โšช โš  ๐Ÿ”’ -- are one Python character and TWO
+ columns wide. Measuring with len() pads every cell containing one a column short, so
+ the pipes drift out of line against the header rule for the whole table.
+ """
+ return sum(2 if unicodedata.east_asian_width(c) in 'WF' else 1 for c in s)
+
+
+def pad(s: str, width: int, center: bool = False) -> str:
+ """str.ljust/center, measured in display columns. See display_width."""
+ room = max(0, width - display_width(s))
+ if not center:
+ return s + ' ' * room
+ left = room // 2
+ return ' ' * left + s + ' ' * (room - left)
+
+
def cmd_stdout_text(out: Any) -> str:
if out is None:
return ''
@@ -137,74 +162,119 @@ def _print_banner(title: str, out: Any, err: Any) -> None:
print(_banner_body(out, err))
-SYSFS_READ_GRACE = 2.0 # bound on one attribute read of a possibly-wedged device
-SYSFS_STUCK_MAX = 4 # stranded readers tolerated before read_sysfs goes blind
-_sysfs_stuck = 0 # each costs a thread + an fd for the life of the process
-_sysfs_stuck_lock = threading.Lock()
-_sysfs_blind_logged = False
+SYSFS_READ_GRACE = 2.0 # default bound on one attribute read; see read_sysfs
+# path -> the kernfs inode the node had when its bounded read gave up. Keyed by INODE, not
+# by path alone: a busport does not change when a board returns to the same physical port,
+# so a path-only blacklist outlives the wedge -- hil_pool_check resets or reflashes the
+# board, wait_device polls that busport for the new inode, and the scan it polls through
+# would never look at the device again. A re-enumeration destroys the kernfs node and makes
+# a new one, so a CHANGED inode is the all-clear. os.stat is safe on a wedged device: it
+# does not call ->show(), so it cannot block on the lock the reader is stuck behind.
+_stranded: dict = {}
+_strand_hits: dict = {} # path -> how many times it has stranded, ever
+_refused: set = set() # paths answered None WITHOUT reading, once past _STRAND_MAX
+_strand_lock = threading.Lock()
+_ever_stranded = False
-class _SysfsUnknown:
- """Sentinel: the read did not answer. NOT "the attribute is absent" -- reading it as
- absence turns a healthy board into a firmware regression in the report."""
- __slots__ = ()
+# Each strand costs a thread AND an fd for the life of the process -- on sysfs the open()
+# SUCCEEDS and only the read blocks. Two ceilings, because they bound different things:
+#
+# _PATH_STRAND_MAX -- a device that FLAPS while still wedged re-enumerates, clears the
+# inode memo, and strands again. Per path, so one sick board cannot leak without bound.
+# After this many it stays memoised whatever its inode says.
+# _STRAND_MAX -- a whole-process backstop against RLIMIT_NOFILE or the thread ceiling,
+# which would raise inside a worker and lose every board's result. Counted PER PATH, not
+# per reader: hil_pool_check runs four poll threads over one bus, and counting each
+# reader let four threads on ONE wedged device spend four credits between them. With
+# per-path counting a 27-board rig cannot approach this.
+_PATH_STRAND_MAX = 4
+_STRAND_MAX = 64
- def __bool__(self) -> bool:
- return False
- def __repr__(self) -> str:
- return 'SYSFS_UNKNOWN'
+def sysfs_stranded() -> bool:
+ """True once any bounded read has given up, and it STAYS true.
+
+ A sticky, process-wide fact, so it answers exactly one question: "could anything in
+ this process's output be the tool losing sight of healthy hardware?" -- which is what
+ hil_pool_check's footer needs. It canNOT answer "is THIS device unreadable" for a
+ caller deciding what a single missing device means; use path_stranded() for that.
+ """
+ return _ever_stranded
-SYSFS_UNKNOWN = _SysfsUnknown()
+def strand_note() -> str:
+ """Suffix for an absence claim, so "not found" never reads as proven absence.
+ Lives here because every caller that can say "not found" needs the same sentence, and
+ the one that had to re-invent it got missed: a wedged-but-enumerated printer was
+ reported as an enumeration failure, sending a maintainer after firmware.
+ """
+ return (' (a bounded sysfs read gave up, so "not found" here means "could not tell"'
+ ' -- see the usb-kernel-recover skill)') if sysfs_stranded() else ''
-def sysfs_blind() -> bool:
- """True once this process has stranded SYSFS_STUCK_MAX readers: every later read
- answers SYSFS_UNKNOWN, so nothing it reports about a device is a fact any more."""
- return _sysfs_stuck >= SYSFS_STUCK_MAX
+def path_stranded(path: str) -> bool:
+ """Whether THIS attribute is currently memoised as unreadable.
-def sysfs_blind_note() -> str:
- """Suffix for a failure message, so a blind worker's verdict never reads as hardware."""
- return (f' (this worker is blind: {SYSFS_STUCK_MAX} sysfs reads stranded on a wedged '
- f'device, so the check could not see the bus)') if sysfs_blind() else ''
+ The per-device question sysfs_stranded() cannot answer. usbtest uses it to tell a DUT
+ whose `serial` is held under device_lock from one that genuinely left the bus, because
+ the difference decides whether it performs driver-registry writes that take the
+ UNINTERRUPTIBLE device_lock.
+ """
+ with _strand_lock:
+ return path in _stranded or path in _refused
-def read_sysfs(path: str, grace: float = SYSFS_READ_GRACE) -> str | None | _SysfsUnknown:
- """Read a sysfs attribute with a WALL-CLOCK bound.
+def read_sysfs(path: str, timeout: float = SYSFS_READ_GRACE) -> str | None:
+ """A sysfs attribute's value, or None when it did not answer.
- The value, None when the attribute is genuinely unreadable (OSError), or SYSFS_UNKNOWN
- when the read did not answer -- it timed out, or this process is already blind. Callers
- MUST keep those apart: absence is a fact, unknown is not.
+ BOUNDED BY DEFAULT, and it has to be. `serial` is served by usb_string_attr, which
+ takes usb_lock_device_interruptible (v6.12.96 sysfs.c:141-143) -- the same lock a
+ wedged usbfs ioctl holds. Every OTHER attribute the harness reads (idVendor, idProduct,
+ bcdDevice, busnum, devnum, speed) is a lock-free sysfs_emit from a cached field and
+ cannot block.
- usb_string_attr (serial/product/manufacturer) is served under the device lock a wedged
- usbfs ioctl holds, so a plain open().read() blocks for as long as the wedge lasts, on
- exactly the board an incident is about. The reader sleeps INTERRUPTIBLY (every read
- takes usb_lock_device_interruptible, v6.12.96 sysfs.c:124-139 -- uninterruptible is the
- ioctl holder, not us), so it dies with a SIGKILLed worker; what it costs meanwhile is a
- thread and an fd for this process's life, because on sysfs the open() SUCCEEDS and only
- the read blocks. Measured: 20 blocking reads leave 20 live threads.
+ "Only the wedged board's own worker pays" is FALSE, which is why the bound is not
+ opt-in: usb_scan reads `serial` on every device matching the VID to find the one it
+ wants, so resolving MY board touches every peer's locked attribute. hil_lock's
+ controller_of does that from controller_permit, on essentially every board -- one
+ wedged DUT would stall every worker, not one. hil_pool_check has no guard at all.
- Hence the cap: callers rescan (hil_lock's controller_of re-reads every unresolved
- device on EVERY permit), and hitting RLIMIT_NOFILE or the thread ceiling raises inside
- the worker and loses every board's result -- worse than the hang this prevents.
+ A give-up reads as None, the same as unreadable: there is no third value and no
+ per-attribute blindness. The memo is keyed by inode so the cost stays on the device
+ that is actually wedged; path_stranded() tells a caller which device that was.
"""
- if sysfs_blind():
- return SYSFS_UNKNOWN
- # Known-stranded? Re-reading costs another permanent thread+fd and a blindness credit
- # to learn what we already know. Lives HERE, not at the call sites: a call-site memo
- # has to be remembered by every new scanner, and twice it was not.
- was = _sysfs_stranded.get(path, _STRAND_MISS)
- if was is not _STRAND_MISS:
- if was is None:
- return SYSFS_UNKNOWN # stranded, inode unknown: never re-read it
+ with _strand_lock:
+ was = _stranded.get(path)
+ stuck_for_good = _strand_hits.get(path, 0) >= _PATH_STRAND_MAX
+ budget_spent = len(_stranded) >= _STRAND_MAX
+ if was is not None:
try:
if os.stat(path).st_ino == was:
- return SYSFS_UNKNOWN # same node, still wedged
+ return None # same kernfs node, still wedged
except OSError:
- pass # gone: fall through, the read reports it
- _sysfs_stranded.pop(path, None) # replaced or gone -> re-read it
+ pass # gone: let the read below report it
+ if stuck_for_good:
+ return None # flapped too many times; see _PATH_STRAND_MAX
+ with _strand_lock:
+ _stranded.pop(path, None) # a different inode is the all-clear
+ elif budget_spent:
+ # see _STRAND_MAX. Recorded, not just returned: usbtest fails CLOSED on
+ # path_stranded() before the lock-taking cleanup, and a path we declined to read
+ # is exactly the case it must not be told is readable-and-absent.
+ with _strand_lock:
+ _refused.add(path)
+ return None
+
+ # BEFORE the read, not after: a node that re-enumerates DURING the grace would
+ # otherwise have its brand-new HEALTHY inode recorded as the wedged one, and only a
+ # second re-enumeration could ever clear it. If it cannot be stat'd there is no key to
+ # memoise against, so the path is simply re-read next time -- the open fails fast.
+ try:
+ ino = os.stat(path).st_ino
+ except OSError:
+ ino = None
out: dict = {}
def _read():
@@ -212,96 +282,70 @@ def read_sysfs(path: str, grace: float = SYSFS_READ_GRACE) -> str | None | _Sysf
with open(path) as f:
out['v'] = f.read().strip()
except (OSError, ValueError):
- pass # no such attribute, or not text: unreadable, and that IS a fact
+ pass
t = threading.Thread(target=_read, daemon=True)
t.start()
- t.join(grace)
- # `out` FIRST, not is_alive() alone: a reader can deposit its value and still be alive
- # for a moment afterwards, and counting that as a strand memoises a healthy attribute as
- # unreadable and spends one of four blindness credits. bounded_open has always checked
- # its box for the same reason.
+ t.join(timeout)
+ # `out` FIRST: a reader can deposit its value and still be alive for a moment
+ # afterwards, and counting that as a strand blacklists a healthy attribute forever
+ if 'v' in out:
+ # a path that answered is not refused any more: _refused feeds path_stranded(),
+ # and a stale entry makes usbtest read a LATER genuine disconnect as "cannot tell"
+ with _strand_lock:
+ _refused.discard(path)
if t.is_alive() and 'v' not in out:
- # Count the PATH once, not once per reader. hil_pool_check runs -j4 by default,
- # which equals SYSFS_STUCK_MAX, so four threads hitting ONE wedged device used to
- # spend the entire blindness budget between them -- latching blind on the single
- # wedge the tool was run to find. The strand is real for each thread, but the
- # DEVICE is what the cap is about.
- # Under the SAME lock as the counter: check-then-act here is a race, and
- # hil_pool_check runs a ThreadPoolExecutor of exactly SYSFS_STUCK_MAX workers in
- # ONE process, so four threads on one wedged path could each see `first` before any
- # of them recorded it -- spending the whole blindness budget on a single device,
- # which is what this memo exists to prevent. note_sysfs_strand takes the lock
- # itself, so call it after releasing.
- with _sysfs_stuck_lock:
- first = path not in _sysfs_stranded
- if first:
- try:
- # stat, never the thread's own open(): stat does not call ->show(), so
- # it cannot block on the device lock the reader is stuck behind
- _sysfs_stranded[path] = os.stat(path).st_ino
- except OSError:
- _sysfs_stranded[path] = None # unstattable, but still known-stranded
- if first:
- note_sysfs_strand()
- return SYSFS_UNKNOWN
+ global _ever_stranded
+ announce = False
+ if ino is None:
+ # the pre-read stat lost a race the open then won -- the node was replaced
+ # between them. Re-stat now: the reader is blocked on whatever node exists,
+ # so this is the key it is stuck on. Without a key nothing is memoised and
+ # every later poll starts another permanent thread and fd for this path.
+ try:
+ ino = os.stat(path).st_ino
+ except OSError:
+ pass
+ with _strand_lock:
+ _ever_stranded = True
+ if ino is not None:
+ first = path not in _stranded # count the PATH once, not each reader
+ _stranded[path] = ino
+ if first:
+ _strand_hits[path] = _strand_hits.get(path, 0) + 1
+ announce = len(_stranded) == _STRAND_MAX
+ else:
+ _refused.add(path) # unkeyable: at least do not vouch for it
+ if announce:
+ print(f'warning: {_STRAND_MAX} devices have unreadable sysfs attributes; '
+ f'refusing to start more bounded readers, so later reads answer None '
+ f'without looking. Find the wedged device (usb-kernel-recover skill).',
+ file=sys.stderr, flush=True)
+ return None
return out.get('v')
-def note_sysfs_strand() -> None:
- """Record ONE stranded sysfs reader. Shared by read_sysfs and bounded_open so both
- account against a single counter -- the report caveat keys off it."""
- global _sysfs_stuck, _sysfs_blind_logged
- with _sysfs_stuck_lock:
- _sysfs_stuck += 1
- announce = sysfs_blind() and not _sysfs_blind_logged
- _sysfs_blind_logged = _sysfs_blind_logged or announce
- if announce:
- # once per process, on stderr: a worker's stdout is compacted into one report
- # row, where this would be lost among the test output
- print(f'warning: {SYSFS_STUCK_MAX} sysfs reads stranded on a wedged device; '
- f'this process is now blind and answers SYSFS_UNKNOWN for every '
- f'attribute -- its verdicts about device presence are not evidence',
- file=sys.stderr, flush=True)
-
-
-# path -> the inode it had when its read stranded. A stranded attribute stays
-# stranded until the DEVICE is replaced, and a re-enumeration destroys the kernfs
-# node and makes a new one -- so a changed inode is the all-clear. Keyed by path
-# alone it would outlive the wedge: a busport does not change when a board comes
-# back on the same port, so the HUNG reflash this branch performs would recover a
-# board the harness could then never see again.
-_sysfs_stranded: dict = {}
-# A stranded path whose inode could not be read is stored as None, so a plain .get() cannot
-# tell 'known stranded, inode unknown' from 'never seen' -- and treating the first as the
-# second re-reads it, stranding another permanent thread and fd every call. Distinct miss
-# sentinel, so None keeps its own meaning.
-_STRAND_MISS = object()
-
-
-def usb_scan(vid_pid=None, serial=None, vid=None) -> tuple[list, bool]:
- """Enumerated USB devices matching the filters, and whether anything is unknown.
-
- Returns ([{busport, dir, vid, pid, serial}], unknown). `unknown` True means a bounded
- read did not answer, so absence is NOT proven -- the same contract as read_sysfs.
+def usb_scan(vid_pid=None, serial=None, vid=None, timeout=SYSFS_READ_GRACE) -> list:
+ """Enumerated USB devices matching the filters: [{busport, dir, vid, pid, serial}].
Three rules, one implementation for every caller:
* Root hubs excluded (glob `*-*`): no DUT is one, and scans including them measured
seconds slower (observation, no mechanism -- the "autosuspend wake" explanation was
- wrong; usb_string_attr reads a cached string, sysfs.c:124-139).
+ wrong; usb_string_attr reads a cached string, sysfs.c:141-143).
* idVendor/idProduct first: lock-free `sysfs_emit` from udev->descriptor
(sysfs.c:688-705), so they rule out nearly every device for free.
- * `serial` last and bounded: it is served under the lock a wedged ioctl holds, and a
- path that already stranded is never re-read (each strand costs a thread and an fd
- for this process's life).
+ * `serial` LAST and BOUNDED: it is the only attribute here served under the device
+ lock, so it is the only one that can block. Filtering on the lock-free pair first
+ keeps most devices out of it, but a scan for ONE board still reads the serial of
+ every peer that shares its VID -- so the bound is what stops one wedged DUT from
+ stalling every caller (see read_sysfs).
"""
out = []
- unknown = False
for d in glob.glob('/sys/bus/usb/devices/*-*'):
- # Interfaces are '<busport>:<cfg>.<ifnum>' (e.g. 2-4:1.0) -- they CONTAIN the
- # colon, they do not end with it, so the original endswith() never fired and every
- # scan opened idVendor/idProduct on all of them (measured: 31 of 44 matches).
+ # `in`, not endswith: an interface is '<busport>:<cfg>.<ifnum>' (2-4:1.0), which
+ # CONTAINS the colon rather than ending with it. Screening them out here is worth
+ # real time -- they were 31 of 44 matches on this rig.
if ':' in os.path.basename(d):
continue
try:
@@ -315,106 +359,14 @@ def usb_scan(vid_pid=None, serial=None, vid=None) -> tuple[list, bool]:
continue # ruled out for free, without touching the locked attribute
if vid is not None and dev_vid != vid:
continue # same, for callers that know the VID but not the PID
- sn = read_sysfs(os.path.join(d, 'serial'))
- if sn is SYSFS_UNKNOWN:
- unknown = True # read_sysfs memoises it; a repeat scan costs nothing
- continue
+ sn = read_sysfs(os.path.join(d, 'serial'), timeout)
if sn is None:
- continue # no serial attribute: a fact
+ continue # no serial attribute
if serial is not None and sn.lower() != serial.lower():
continue
out.append({'busport': os.path.basename(d), 'dir': d,
'vid': dev_vid, 'pid': dev_pid, 'serial': sn})
- return out, unknown
-
-
-def bounded_open(path: str, flags: int, timeout: float = SYSFS_READ_GRACE):
- """os.open() with a wall-clock bound.
-
- The fd, None when the open genuinely FAILED (OSError: EBUSY, ENOENT, EACCES), or
- SYSFS_UNKNOWN when it did not answer -- the same three-valued contract as read_sysfs,
- and for the same reason: folding a fact into an unknown made an ordinary EBUSY read as
- a wedged device and sent the operator hunting hardware that is healthy.
-
- An open CAN block on a wedged device -- not on O_NONBLOCK, which usblp_open never
- consults, but on usb_autopm_get_interface(), a runtime-PM resume that does I/O
- (v6.12.96 drivers/usb/class/usblp.c). It holds usblp_mutex while it waits, and that
- mutex is driver-GLOBAL, so one wedged printer blocks opens of every usblp node.
-
- Unlike read_sysfs the stranded thread cleans up after itself: if we have given up it
- closes the fd it eventually got, so only the thread leaks. Both sides take `handoff`
- -- "store or close" and "abandon and drain" are a check-then-act pair that can
- interleave into an fd stored after the box was drained, which would leak it into a
- node that allows a SINGLE opener (usblp_open returns -EBUSY when usblp->used).
- """
- # Same short-circuit as read_sysfs: once blind, another stranded thread buys nothing
- # and the cap exists precisely to stop them accumulating.
- if sysfs_blind():
- return SYSFS_UNKNOWN
- # Known-stranded? Re-opening costs another thread, another fd and another blindness
- # credit to learn what we already know -- and the printer test re-opens ONE lp node on
- # every retry. Same memo and same inode check as read_sysfs.
- was = _sysfs_stranded.get(path, _STRAND_MISS)
- if was is not _STRAND_MISS:
- if was is None:
- return SYSFS_UNKNOWN # stranded, inode unknown: never re-read it
- try:
- if os.stat(path).st_ino == was:
- return SYSFS_UNKNOWN
- except OSError:
- pass
- _sysfs_stranded.pop(path, None)
- box: dict = {}
- done, abandoned = threading.Event(), threading.Event()
- handoff = threading.Lock()
-
- def _open():
- try:
- fd = os.open(path, flags)
- except OSError:
- done.set()
- return
- with handoff:
- stored = not abandoned.is_set()
- if stored:
- box['fd'] = fd
- if not stored:
- try:
- os.close(fd)
- except OSError:
- pass
- done.set()
-
- threading.Thread(target=_open, daemon=True).start()
- if not done.wait(timeout):
- with handoff:
- abandoned.set()
- fd = box.pop('fd', None) # completed in the gap between timeout and flag
- if fd is not None:
- # It DID open, just after our deadline -- the thread finished, so nothing is
- # stranded. Report unknown (we already gave up on it) but do not spend a
- # blindness credit, and do not call a merely-slow node wedged.
- try:
- os.close(fd)
- except OSError:
- pass
- return SYSFS_UNKNOWN
- # counted like a stranded read_sysfs: the thread and (eventually) its fd are gone
- # for the life of the process, and the cap exists to stop that reaching the
- # thread/fd ceiling -- an exception there escapes the worker and loses every board.
- # Memoised by inode so a retry of the same node does not pay again.
- # same lock as read_sysfs, same reason
- with _sysfs_stuck_lock:
- first = path not in _sysfs_stranded
- if first:
- try:
- _sysfs_stranded[path] = os.stat(path).st_ino
- except OSError:
- _sysfs_stranded[path] = None
- if first:
- note_sysfs_strand()
- return SYSFS_UNKNOWN
- return box.get('fd')
+ return out
def _close_pipes(p: subprocess.Popen) -> None:
@@ -457,7 +409,7 @@ def run_alongside(argv: list, work, timeout: int) -> subprocess.CompletedProcess
except OSError:
p.kill()
try:
- out, err = p.communicate(timeout=5)
+ out, err = p.communicate(timeout=REAP_GRACE)
except subprocess.TimeoutExpired:
# Outlasted SIGKILL: uninterruptible, still holding whatever it opened.
# Abandoned like any other stray -- but as a real child in its own
@@ -479,9 +431,49 @@ def run_alongside(argv: list, work, timeout: int) -> subprocess.CompletedProcess
return _reap()
-def run_cmd(cmd: str, cwd: str | None = None, timeout: int | None = None,
+# The RTT console implementation lives in tools/rtt.py (importable classes + CLI,
+# stdlib-only, harness-critical โ€” see its module docstring). Loaded by file path so
+# no sys.path entry for tools/ can shadow other imports; re-exported here so the
+# harness keeps addressing hil_util.JlinkRtt.
+import importlib.util as _ilu
+
+_rtt_path = TINYUSB_ROOT / 'tools' / 'rtt.py'
+if not _rtt_path.exists():
+ # name the real cause: a bare FileNotFoundError out of an exec_module here reads
+ # as a harness bug, when the actual problem is an incompletely staged tree
+ raise ImportError(f'{_rtt_path} is missing โ€” the RTT console lives there and the '
+ f'harness depends on it; stage it alongside test/hil (hil_ci.sh does)')
+_rtt_spec = _ilu.spec_from_file_location('tinyusb_tools_rtt', _rtt_path)
+_rtt = _ilu.module_from_spec(_rtt_spec)
+sys.modules[_rtt_spec.name] = _rtt # registered: RttError must be picklable across the fork Pool
+_rtt_spec.loader.exec_module(_rtt)
+JlinkRtt = _rtt.JlinkRtt
+OpenocdRtt = _rtt.OpenocdRtt
+RttError = _rtt.RttError
+RTT_BANNER_RE = _rtt.RTT_BANNER_RE
+strip_banner = _rtt.strip_banner
+
+
+def _cmd_label(cmd) -> str:
+ """A one-line name for a banner. An argv whose payload is a `python3 -c` program would
+ otherwise dump the whole body into the CI log, where run_cmd's banners are already the
+ noisiest thing in a failing row."""
+ if isinstance(cmd, str):
+ return cmd
+ parts = [a if len(a) <= 60 else f'<{len(a)}-char program>' for a in cmd]
+ return ' '.join(parts)
+
+
+def run_cmd(cmd: str | list, cwd: str | None = None, timeout: int | None = None,
binary: bool = False, split_stderr: bool = False,
quiet: bool = False) -> subprocess.CompletedProcess:
+ """Bounded subprocess: own session, killpg on expiry, rc 124 when it had to be killed.
+
+ `cmd` is a shell STRING or an argv LIST. argv exists for a program that cannot survive
+ a trip through the shell -- a multi-line `python3 -c` body -- which is how the harness
+ runs a library call that no in-process bound can contain. A daemon thread cannot bound
+ a C call that holds the GIL, so for those the child process IS the bound.
+ """
if timeout is None:
timeout = CMD_TIMEOUT
# binary: raw bytes (text mode's errors='replace' mangles non-UTF-8 file content).
@@ -490,34 +482,31 @@ def run_cmd(cmd: str, cwd: str | None = None, timeout: int | None = None,
# still print: a killed child is always noteworthy).
popen_kwargs = {
'cwd': cwd,
- 'shell': True,
+ # a list goes straight to execve; only a string needs a shell to parse it
+ 'shell': isinstance(cmd, str),
'stdout': subprocess.PIPE,
'stderr': subprocess.PIPE if split_stderr else subprocess.STDOUT,
}
if not binary:
popen_kwargs.update({'text': True, 'encoding': 'utf-8', 'errors': 'replace'})
- if os.name != 'nt':
- # C-level setsid, same process-group semantics as preexec_fn=os.setsid but
- # safe when called from threads (pool_check runs flashes from a thread pool)
- popen_kwargs['start_new_session'] = True
+ # C-level setsid, same process-group semantics as preexec_fn=os.setsid but safe when
+ # called from threads (pool_check runs flashes from a thread pool)
+ popen_kwargs['start_new_session'] = True
p = subprocess.Popen(cmd, **popen_kwargs)
try:
out, err = p.communicate(timeout=timeout)
r = subprocess.CompletedProcess(args=cmd, returncode=p.returncode, stdout=out, stderr=err)
except subprocess.TimeoutExpired as ex:
- if os.name != 'nt':
- try:
- os.killpg(p.pid, signal.SIGKILL)
- except OSError:
- # ProcessLookupError: already gone. PermissionError: an all-root group
- # refuses the group kill -- letting either escape would skip the bounded
- # reap, the pipe close and the rc-124 return this handler exists for.
- pass
- else:
- p.kill()
try:
- out, err = p.communicate(timeout=10)
+ os.killpg(p.pid, signal.SIGKILL)
+ except OSError:
+ # ProcessLookupError: already gone. PermissionError: an all-root group refuses
+ # the group kill -- letting either escape would skip the bounded reap, the pipe
+ # close and the rc-124 return this handler exists for.
+ pass
+ try:
+ out, err = p.communicate(timeout=REAP_GRACE)
except subprocess.TimeoutExpired:
# Something in the group outlived SIGKILL: D state (truly unkillable), or
# root-owned because sudo FORKS rather than execs, so the wrapper dies and its
@@ -543,7 +532,7 @@ def run_cmd(cmd: str, cwd: str | None = None, timeout: int | None = None,
timeout_err = _typed(err if err is not None else ex.stderr)
if split_stderr and timeout_err is None:
timeout_err = b'' if binary else ''
- _print_banner(f'COMMAND TIMEOUT ({timeout}s): {cmd}', timeout_out, timeout_err)
+ _print_banner(f'COMMAND TIMEOUT ({timeout}s): {_cmd_label(cmd)}', timeout_out, timeout_err)
return subprocess.CompletedProcess(args=cmd, returncode=124, stdout=timeout_out, stderr=timeout_err)
except BaseException:
# BaseException, not Exception (as in CPython's own subprocess.run):
@@ -551,18 +540,15 @@ def run_cmd(cmd: str, cwd: str | None = None, timeout: int | None = None,
# its OWN group, so it never got the terminal's SIGINT -- without this, Ctrl-C
# leaves the flasher or testusb holding the probe and its usbfs node. Kill and
# close, never wait: this path must not add a hang of its own.
- if os.name != 'nt':
- try:
- os.killpg(p.pid, signal.SIGKILL)
- except OSError:
- pass
- else:
- p.kill()
+ try:
+ os.killpg(p.pid, signal.SIGKILL)
+ except OSError:
+ pass
_close_pipes(p)
raise
if r.returncode != 0 and not quiet:
- _print_banner(f'COMMAND FAILED: {cmd}', r.stdout, r.stderr)
+ _print_banner(f'COMMAND FAILED: {_cmd_label(cmd)}', r.stdout, r.stderr)
elif verbose:
print(cmd)
print(cmd_stdout_text(r.stdout))