summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--.claude/agents/builder.md1
-rw-r--r--.claude/agents/code-verifier.md (renamed from .claude/agents/driver-reviewer.md)3
-rw-r--r--.claude/agents/code-writer.md (renamed from .claude/agents/port-dev.md)3
-rw-r--r--.claude/agents/hil-operator.md7
-rw-r--r--.claude/agents/pr-ci-watcher.md26
-rw-r--r--.claude/agents/pr-monitor.md38
-rw-r--r--.claude/agents/pr-review-validator.md26
-rw-r--r--.claude/agents/static-analyzer.md1
-rw-r--r--.claude/agents/target-debugger.md1
-rw-r--r--.claude/skills/update-sponsor/SKILL.md28
-rw-r--r--.claude/skills/update-sponsor/config.json15
-rw-r--r--.claude/skills/update-sponsor/update_sponsor.py505
-rw-r--r--.claude/workflows/driver-review.js9
-rw-r--r--.claude/workflows/fanout-dev.js10
-rw-r--r--.claude/workflows/hil-validate.js37
-rw-r--r--.claude/workflows/pr-babysit.js362
-rw-r--r--.claude/workflows/test-hil-validate.mjs29
-rw-r--r--.claude/workflows/validate.js60
-rwxr-xr-x.github/scripts/ci_set_matrix.py8
-rw-r--r--.github/workflows/build.yml35
-rw-r--r--.github/workflows/build_util.yml13
-rw-r--r--.gitignore1
-rw-r--r--README.rst4
-rw-r--r--docs/reference/hil_boards.md2
-rw-r--r--docs/superpowers/followup/pr3803-hil-blindness-reporting.md3
-rw-r--r--docs/superpowers/followup/pr3836-report-single-source.md470
-rw-r--r--docs/superpowers/followup/pr3840-mret-board-result.md90
-rw-r--r--docs/superpowers/followup/pr3840-skill-md-no-boards-drift.md38
-rw-r--r--docs/superpowers/followup/pr3840-write-report-atomicity.md30
-rw-r--r--docs/superpowers/plans/2026-08-21-hil-report-module.md658
-rw-r--r--docs/superpowers/specs/2026-08-19-ci-build-family-filter-design.md67
-rw-r--r--docs/superpowers/specs/2026-08-21-hil-report-module-design.md144
-rw-r--r--examples/device/audio_4_channel_mic/skip.txt1
-rw-r--r--examples/device/audio_4_channel_mic_freertos/skip.txt1
-rw-r--r--examples/device/audio_test/skip.txt1
-rw-r--r--examples/device/audio_test_freertos/skip.txt1
-rw-r--r--examples/device/audio_test_multi_rate/skip.txt1
-rw-r--r--examples/device/cdc_msc_freertos/skip.txt1
-rw-r--r--examples/device/cdc_uac2/skip.txt1
-rw-r--r--examples/device/hid_composite_freertos/skip.txt1
-rw-r--r--examples/device/midi_test_freertos/skip.txt1
-rw-r--r--examples/device/msc_dual_lun/skip.txt1
-rw-r--r--examples/device/uac2_headset/skip.txt1
-rw-r--r--examples/device/uac2_speaker_fb/skip.txt1
-rw-r--r--test/hil/helper/hil_health.py42
-rw-r--r--test/hil/helper/hil_pool_check.py8
-rw-r--r--test/hil/helper/hil_report.py568
-rw-r--r--test/hil/helper/hil_summary.py115
-rw-r--r--test/hil/helper/hil_util.py81
-rw-r--r--test/hil/hil_ci.sh113
-rwxr-xr-xtest/hil/hil_test.py676
-rw-r--r--test/hil/test/stubs/hid.py74
-rw-r--r--test/hil/test/test_ci_metrics.py134
-rw-r--r--test/hil/test/test_ci_select.py421
-rw-r--r--test/hil/test/test_hil_bounded.py302
-rw-r--r--test/hil/test/test_hil_health.py43
-rw-r--r--test/hil/test/test_hil_report.py1162
-rw-r--r--test/hil/test/test_hil_util.py54
-rwxr-xr-xtools/build.py4
-rwxr-xr-xtools/build_utils.py40
-rwxr-xr-xtools/ci_select.py211
61 files changed, 5272 insertions, 1512 deletions
diff --git a/.claude/agents/builder.md b/.claude/agents/builder.md
index 4edb7e0d4..3f06f328a 100644
--- a/.claude/agents/builder.md
+++ b/.claude/agents/builder.md
@@ -3,6 +3,7 @@ name: builder
description: Build TinyUSB examples for one board and report structured pass/fail with first-error triage. Use for build sweeps and post-change build verification. Never edits source.
tools: Bash, Read, Grep, Glob
model: haiku
+effort: low
---
You build TinyUSB examples for exactly one board per run and report the result as machine-readable JSON. You never modify source files.
diff --git a/.claude/agents/driver-reviewer.md b/.claude/agents/code-verifier.md
index 9ce8b621f..7e529a641 100644
--- a/.claude/agents/driver-reviewer.md
+++ b/.claude/agents/code-verifier.md
@@ -1,8 +1,9 @@
---
-name: driver-reviewer
+name: code-verifier
description: Review one TinyUSB driver directory or one diff against one review dimension (correctness, ISR safety, datasheet/errata conformance, style) with coverage-first structured findings; or adversarially verify a single finding / fix. Read-only.
tools: Bash, Read, Grep, Glob, Skill
model: opus
+effort: xhigh
---
You review exactly the scope given in your prompt (one driver directory, or one git diff) for exactly the dimension(s) given. Read the code yourself; follow callers, headers, and macros as far as needed to judge correctly. You never modify files.
diff --git a/.claude/agents/port-dev.md b/.claude/agents/code-writer.md
index 77a28bafa..82e52e334 100644
--- a/.claude/agents/port-dev.md
+++ b/.claude/agents/code-writer.md
@@ -1,7 +1,8 @@
---
-name: port-dev
+name: code-writer
description: Implement one well-scoped change in one TinyUSB port or explicit file set, following repo style and .clang-format, verified by a targeted build. Use for fan-out development across ports and for fixing validated PR findings.
model: opus
+effort: xhigh
---
You implement exactly one specified change in one assigned scope (a directory under `src/portable/`, a class driver, or an explicitly listed file set). Never touch files outside the assigned scope.
diff --git a/.claude/agents/hil-operator.md b/.claude/agents/hil-operator.md
index a37501211..d128f6f53 100644
--- a/.claude/agents/hil-operator.md
+++ b/.claude/agents/hil-operator.md
@@ -3,6 +3,7 @@ name: hil-operator
description: Run TinyUSB hardware-in-the-loop actions on the physical test rig — per-board locking, firmware flash, hil_test.py runs, USB recovery. Strictly one instance at a time. Never edits source; never touches the actions-runner service.
tools: Bash, Read, Grep, Glob
model: sonnet
+effort: high
---
You operate physical USB test hardware. These repo skills are your source of truth — read the relevant one BEFORE acting:
@@ -68,12 +69,12 @@ For a board run, do NOT transcribe the report table. Run the tests, then hand ba
output verbatim:
```bash
-python3 test/hil/helper/hil_summary.py <config> -b BOARD [-b BOARD...] # from the report dir
+python3 test/hil/helper/hil_report.py <config> -b BOARD [-b BOARD...] # from the report dir
```
-`{"results": <its results array, verbatim>, "banner": <its banner, verbatim>, "wedged": ["board", ...]}`
+`{"results": <its results array, verbatim>, "banner": <its banner, verbatim>, "caveat": <its caveat, verbatim>, "wedged": ["board", ...]}`
-`results` and `banner` are copied, never retyped, reworded or re-ordered: report rows are named
+`results`, `banner` and `caveat` are copied, never retyped, reworded or re-ordered (`caveat` is the run-level notice — abandoned, aborted, no-boards — and it can say the run failed while every row says pass): report rows are named
per variant, a variant name need not start with the board name, and lock contention is a cell
rather than a phrase, so re-deriving any of it by hand is how this contract broke before.
`wedged` is yours — the boards your run left unresponsive, usually none — and the only field you
diff --git a/.claude/agents/pr-ci-watcher.md b/.claude/agents/pr-ci-watcher.md
new file mode 100644
index 000000000..10a32084b
--- /dev/null
+++ b/.claude/agents/pr-ci-watcher.md
@@ -0,0 +1,26 @@
+---
+name: pr-ci-watcher
+description: Watch one TinyUSB PR's CI — classify failures (infra flake / real / rig-side), re-run infra ones, report real ones with first error and files. CI only; never reads review comments, never edits code, never pushes.
+tools: Bash, Read, Grep, Glob
+model: sonnet
+effort: high
+---
+
+You watch CI for exactly one PR (number given in your prompt) using `gh`. You never modify source files, never commit, never push, never read review comments.
+
+## Procedure
+
+1. `gh pr checks <N>`. If checks are running and your prompt says to wait, run `gh pr checks <N> --watch` as a BACKGROUND Bash task (the foreground timeout is capped at 10 min).
+2. For each failing check, find its run and read the failure: `gh run view <run-id> --log-failed | head -150`.
+3. Classify each failure:
+ - **infra/flake**: runner lost communication, network/DNS timeouts, artifact 404, docker pull/rate-limit errors, cancelled-by-timeout with no test output. Re-run once (`gh run rerun <run-id> --failed`); record run ids in `infraRerun`.
+ - **real**: compile/link errors, test assertions, HIL failures with device output. Extract the FIRST error line and the source files involved.
+ - **rigSide=true** on a real failure NOT attributable to the PR: probe/fixture faults, byte-identical reproduction on unrelated PRs, boards outside the diff. These are reported for humans, never handed to a fixer.
+
+## Output contract
+
+Your final message is parsed by a program. Return ONLY this JSON — no prose, no code fences:
+
+{"status": "green", "infraRerun": [], "realFailures": [{"check": "...", "firstError": "...", "files": ["..."], "rigSide": false}]}
+
+status: "green" (all pass), "red" (any real failure), "running" (still pending after your wait budget).
diff --git a/.claude/agents/pr-monitor.md b/.claude/agents/pr-monitor.md
deleted file mode 100644
index 77777b0fb..000000000
--- a/.claude/agents/pr-monitor.md
+++ /dev/null
@@ -1,38 +0,0 @@
----
-name: pr-monitor
-description: Triage one TinyUSB GitHub PR — CI status + failure classification, infra re-runs, bot review harvesting (Codex/Copilot/Claude) with adversarial validation of each finding against the code. Read/triage/re-run only; never edits code, never pushes.
-tools: Bash, Read, Grep, Glob
-model: sonnet
----
-
-You triage exactly one PR (number given in your prompt) using `gh`. You never modify source files, never commit, never push.
-
-## CI triage
-
-1. `gh pr checks <N>`. If checks are running and your prompt says to wait, run `gh pr checks <N> --watch` as a BACKGROUND Bash task (the foreground timeout is capped at 10 min).
-2. For each failing check, find its run and read the failure: `gh run view <run-id> --log-failed | head -150`.
-3. Classify each failure:
- - **infra/flake**: runner lost communication, network/DNS timeouts, artifact 404, docker pull/rate-limit errors, cancelled-by-timeout with no test output.
- - **real**: compile/link errors, test assertions, HIL failures with device output.
-4. Re-run infra failures once: `gh run rerun <run-id> --failed`; record run ids in `infraRerun`.
-5. For real failures extract the FIRST error line and the source files involved (from the log paths).
-
-## Bot review harvest
-
-- Inline review comments: `gh api repos/{owner}/{repo}/pulls/<N>/comments --paginate` (use `gh repo view --json nameWithOwner -q .nameWithOwner` for owner/repo). Issue comments: `gh pr view <N> --comments`.
-- Known signals: Codex posts an issue comment when done — "Didn't find any major issues" means clean, not silence. Copilot is finished when it no longer appears in `requested_reviewers`. Bot logins differ across REST/GraphQL — match authors case-insensitively on substrings `codex`, `copilot`, `claude`.
-- For EACH unresolved bot finding: open the file at the cited line in the current checkout and judge the claim adversarially. `valid` only if the code truly has the problem; `invalid` with a concrete refutation otherwise; `stale` if the current code already fixed it.
-- Draft a courteous, technical reply for every `invalid`/`stale` finding (cite the code that refutes it). Put them in `replies` with the comment id — a later step posts the reply AND marks the inline thread resolved (via the GraphQL `resolveReviewThread` mutation); you do not post or resolve. The `commentId` must be the inline review comment's integer databaseId so the thread can be found.
-
-## done
-
-`done` = true only when CI is green (all checks pass, nothing running) AND no unresolved `valid` findings remain.
-
-## Output contract
-
-Your final message is parsed by a program. Return ONLY this JSON — no prose, no code fences:
-
-{"ci": {"status": "green", "infraRerun": [], "realFailures": [{"check": "...", "firstError": "...", "files": ["..."]}]},
- "findings": [{"source": "codex", "commentId": 123, "file": "...", "line": 1, "claim": "...", "verdict": "valid", "reason": "...", "fixHint": "..."}],
- "replies": [{"commentId": 123, "body": "..."}],
- "done": false}
diff --git a/.claude/agents/pr-review-validator.md b/.claude/agents/pr-review-validator.md
new file mode 100644
index 000000000..324e8efcb
--- /dev/null
+++ b/.claude/agents/pr-review-validator.md
@@ -0,0 +1,26 @@
+---
+name: pr-review-validator
+description: Harvest one TinyUSB PR's bot reviews (Codex/Copilot/Claude) and adversarially validate each finding against the code — verdict valid/invalid/stale, draft replies for refuted ones. Read-only; never edits code, never posts, never pushes.
+tools: Bash, Read, Grep, Glob
+model: opus
+effort: xhigh
+---
+
+You validate the bot review findings on exactly one PR (number given in your prompt) using `gh`. You never modify source files, never commit, never push, never post comments. Do not read or classify CI.
+
+## Procedure
+
+- Inline review comments: `gh api repos/{owner}/{repo}/pulls/<N>/comments --paginate` (use `gh repo view --json nameWithOwner -q .nameWithOwner` for owner/repo). Issue comments: `gh api repos/{owner}/{repo}/issues/<N>/comments --paginate` — this returns each comment's integer `id`, which `gh pr view --comments` does not print and the output contract needs.
+- Known signals: Codex posts an issue comment when done — "Didn't find any major issues" means clean, not silence. Copilot is finished when it no longer appears in `requested_reviewers`. Bot logins differ across REST/GraphQL — match authors case-insensitively on substrings `codex`, `copilot`, `claude`.
+- For EACH unresolved bot finding: open the file at the cited line in the current checkout and judge the claim adversarially. `valid` only if the code truly has the problem; `invalid` with a concrete refutation otherwise; `stale` if the current code already fixed it.
+- Draft a courteous, technical reply for every `invalid`/`stale` finding (cite the code that refutes it). Put them in `replies` with the comment id — a later step posts the reply AND resolves the thread; you do not. For a finding from an inline thread, `commentId` is the inline review comment's integer databaseId (that is how the thread is located and resolved); for one that exists only in an issue comment, use that issue comment's id — the poster falls back to a plain PR comment and skips resolving.
+
+## Output contract
+
+Your final message is parsed by a program. Return ONLY this JSON — no prose, no code fences:
+
+{"findings": [{"source": "codex", "commentId": 123, "file": "...", "line": 1, "claim": "...", "verdict": "valid", "reason": "...", "fixHint": "..."}],
+ "replies": [{"commentId": 123, "body": "..."}],
+ "done": false}
+
+done = true only when no unresolved `valid` findings remain.
diff --git a/.claude/agents/static-analyzer.md b/.claude/agents/static-analyzer.md
index 0f0b2a6e1..e7da82192 100644
--- a/.claude/agents/static-analyzer.md
+++ b/.claude/agents/static-analyzer.md
@@ -3,6 +3,7 @@ name: static-analyzer
description: Run PVS-Studio static analysis (SAST + MISRA C:2023/C++:2008) on TinyUSB for one board and report structured findings, gated on diagnostics in files changed vs a base ref. Read-only; never edits source.
tools: Bash, Read, Grep, Glob
model: sonnet
+effort: medium
---
You run PVS-Studio over the TinyUSB examples build for exactly one board per run and report machine-readable findings. You never modify source files.
diff --git a/.claude/agents/target-debugger.md b/.claude/agents/target-debugger.md
index 1b6931307..c7acca91c 100644
--- a/.claude/agents/target-debugger.md
+++ b/.claude/agents/target-debugger.md
@@ -2,6 +2,7 @@
name: target-debugger
description: Root-cause one USB misbehavior on real HIL hardware by instrumenting the TinyUSB target — device or host stack — with TU_LOG/RTT, RAM ring-buffer trace, GDB autopsy, J-Link PC-sampling, correlated with capture from the link's other end (Linux PC host, another TinyUSB board, or a Linux gadget peer) and the wire. Long serial debug loop under one held board lock; strictly one instance. Produces a diagnosis with on-target evidence (plus a candidate fix when one emerges), never a merged patch.
model: opus
+effort: xhigh
---
You debug one failing USB behavior on one physical board until you can name the
diff --git a/.claude/skills/update-sponsor/SKILL.md b/.claude/skills/update-sponsor/SKILL.md
new file mode 100644
index 000000000..8c4987c8f
--- /dev/null
+++ b/.claude/skills/update-sponsor/SKILL.md
@@ -0,0 +1,28 @@
+---
+name: update-sponsor
+description: Use when a GitHub sponsor joins, upgrades, cancels, or switches between public and private; when the README sponsor sections are stale or show "be the first!" despite active sponsors; or when new issues, PRs, or discussions still need sponsor, priority, or Adafruit triage labels.
+---
+
+# Update Sponsors
+
+Rewrite `README.rst`'s sponsor blocks and backfill triage labels from live GitHub Sponsors data.
+
+```bash
+S=.claude/skills/update-sponsor/update_sponsor.py
+python3 $S --rules # tier -> README section -> labels, and the privacy rules
+python3 $S --help # flags
+python3 $S --dry-run # preview; every run previews and asks before applying
+```
+
+Run from the repo root as `hathach` — the script refuses any other account, whose sponsors are not
+the ones this README lists. Hand-edited data lives in `config.json`, documented in that file.
+
+**Agents:** the confirmation prompt needs a terminal and a tool-call shell has none, so the script
+refuses to apply rather than guessing. Run `--dry-run`, show the maintainer the preview, get their
+answer, then re-run with `--yes`. Never `--yes` on the first call — the preview is the point.
+Applying dirties tracked files (`README.rst`, sometimes `tools/codespell/ignore-words.txt`);
+leave them unstaged for the maintainer, as `make-release` does.
+
+**`.github/workflows/labeler.yml` owns the label rules.** It applies the same labels when a ticket is
+opened; this script only backfills what that workflow cannot reach — tickets older than it, and
+private sponsors its `GITHUB_TOKEN` cannot see. Changing the policy means changing both.
diff --git a/.claude/skills/update-sponsor/config.json b/.claude/skills/update-sponsor/config.json
new file mode 100644
index 000000000..32d535be6
--- /dev/null
+++ b/.claude/skills/update-sponsor/config.json
@@ -0,0 +1,15 @@
+{
+ "_comment": "Curated data for update_sponsor.py. Committed. Edit by hand.",
+ "_exclude": "Logins that never get labels, however they qualify. The maintainer is a public member of the adafruit org, so without this every self-authored ticket would be tagged 'Reported by an Adafruit member'.",
+ "_org_members": "Sponsoring ORGANIZATIONS only. A GitHub org never opens tickets itself - its people do. List the logins that should inherit that org's tier. Curated on purpose: the public-members API misses private members and over-counts uninvolved ones.",
+ "org_members": {
+ "8086net": [
+ "burtyb"
+ ]
+ },
+ "_adafruit_members_extra": "Adafruit logins the org's PUBLIC member list omits. Unioned with `gh api orgs/adafruit/public_members`. Never use /members: it returns concealed members to an org admin, and the Adafruit label would then publish an affiliation those people deliberately hid.",
+ "adafruit_members_extra": [],
+ "exclude": [
+ "hathach"
+ ]
+}
diff --git a/.claude/skills/update-sponsor/update_sponsor.py b/.claude/skills/update-sponsor/update_sponsor.py
new file mode 100644
index 000000000..d09c8b294
--- /dev/null
+++ b/.claude/skills/update-sponsor/update_sponsor.py
@@ -0,0 +1,505 @@
+#!/usr/bin/env python3
+"""Sync README.rst sponsor sections and sponsor/priority labels from GitHub Sponsors.
+
+Reads live sponsorship data (including private sponsors) via `gh api graphql`,
+rewrites the four marker-delimited blocks in README.rst, and applies triage
+labels to OPEN issues / PRs / discussions authored by entitled logins.
+
+Every run plans first and prints what it would change, then asks before
+touching anything. --yes skips the prompt (needed when stdin is not a tty),
+--dry-run stops after the preview.
+
+State lives in state.json next to this file (gitignored): the highest ticket
+number already scanned, so later runs skip old tickets.
+"""
+
+import argparse
+import difflib
+import hashlib
+import json
+import re
+import subprocess
+import tempfile
+import sys
+from datetime import date
+from pathlib import Path
+
+HERE = Path(__file__).resolve().parent
+REPO = HERE.parents[2]
+README = REPO / "README.rst"
+CONFIG = HERE / "config.json"
+IGNORE_WORDS = REPO / "tools" / "codespell" / "ignore-words.txt"
+STATE = HERE / "state.json"
+
+OWNER, NAME = "hathach", "tinyusb"
+
+L_SPONSOR = "Sponsor \U0001f496"
+L_PRIO = "Prio \U0001f4cc"
+L_PRIO_TOP = "Prio Top \U0001f6a8"
+L_ADAFRUIT = "Adafruit \U0001f338"
+
+# Label rules mirror .github/workflows/labeler.yml, which applies the same set when
+# a ticket is opened. Keep the two in step or a ticket's labels start depending on
+# which mechanism happened to touch it.
+TIER_LABELS = {
+ "QWORD": {L_SPONSOR, L_PRIO_TOP},
+ "DWORD": {L_SPONSOR, L_PRIO_TOP},
+ "WORD": {L_SPONSOR, L_PRIO},
+ "BYTE": {L_SPONSOR},
+ "BIT": {L_SPONSOR},
+}
+ADAFRUIT_LABELS = {L_ADAFRUIT, L_SPONSOR, L_PRIO_TOP}
+
+# tier key -> (min $/month, README marker, placeholder when empty, avatar px)
+TIERS = [
+ ("QWORD", 512, "QWORD-SPONSORS", "*No QWORD sponsors yet — be the first!*", 120),
+ ("DWORD", 128, "DWORD-BACKERS", "*No backers yet — be the first!*", 80),
+ ("WORD", 32, "WORD-SUPPORTERS", "*No supporters yet — be the first!*", 40),
+ ("BYTE", 8, "BYTE-THANKS", "*No names listed yet — be the first!*", 0),
+ ("BIT", 2, None, None, 0), # no README listing
+]
+
+
+def gh(*args, **kw):
+ out = subprocess.run(["gh", *args], capture_output=True, text=True, **kw)
+ if out.returncode:
+ sys.exit(f"gh {' '.join(args[:2])} failed:\n{out.stderr.strip()}")
+ return out.stdout
+
+
+def graphql(query, **variables):
+ args = ["api", "graphql", "-f", f"query={query}"]
+ for k, v in variables.items():
+ args += ["-F", f"{k}={'null' if v is None else v}"]
+ data = json.loads(gh(*args))
+ if "errors" in data:
+ sys.exit("GraphQL errors:\n" + json.dumps(data["errors"], indent=2))
+ return data["data"]
+
+
+# ---------------------------------------------------------------- sponsors
+
+SPONSOR_Q = """
+query($cursor:String){ viewer{ login sponsorshipsAsMaintainer(first:100, includePrivate:true, activeOnly:true, after:$cursor){
+ pageInfo{hasNextPage endCursor}
+ nodes{ privacyLevel createdAt tier{monthlyPriceInDollars}
+ sponsorEntity{ __typename ... on User{login name} ... on Organization{login name} } } } } }
+"""
+
+
+def tier_of(dollars):
+ for key, floor, *_ in TIERS:
+ if dollars >= floor:
+ return key
+ return "BIT" # below the lowest published tier, but still a sponsor
+
+
+def fetch_sponsors():
+ """Active sponsorships, oldest first (chronological README order)."""
+ sponsors, cursor = [], None
+ while True:
+ viewer = graphql(SPONSOR_Q, cursor=cursor)["viewer"]
+ if not viewer:
+ sys.exit("gh is authenticated with a token that has no user identity - "
+ "it cannot see sponsorships")
+ if viewer["login"].lower() != OWNER.lower():
+ sys.exit(f"gh is authenticated as {viewer['login']}, not {OWNER} - "
+ f"its sponsors are not the ones this README lists")
+ page = viewer["sponsorshipsAsMaintainer"]
+ for n in page["nodes"]:
+ entity = n["sponsorEntity"]
+ if not entity: # private sponsor we somehow cannot resolve
+ continue
+ tier = tier_of((n["tier"] or {}).get("monthlyPriceInDollars") or 0)
+ sponsors.append({
+ "login": entity["login"],
+ "name": (entity["name"] or "").strip() or entity["login"],
+ "is_org": entity["__typename"] == "Organization",
+ "private": n["privacyLevel"] == "PRIVATE",
+ "since": n["createdAt"],
+ "tier": tier,
+ })
+ if not page["pageInfo"]["hasNextPage"]:
+ break
+ cursor = page["pageInfo"]["endCursor"]
+ sponsors.sort(key=lambda s: s["since"])
+ return sponsors
+
+
+# ------------------------------------------------------------------ README
+
+def mask(login):
+ """Private sponsor display name: first 3 chars, rest hidden behind a fixed
+ 4 stars so the real length does not leak. A login of 3 chars or fewer has no
+ `rest` to hide, so it is withheld entirely."""
+ return login[:3] + "****" if len(login) > 3 else "a private supporter"
+
+
+def rst_escape(text):
+ """A GitHub display name is free-form: backticks/angle brackets would break out
+ of the inline-link markup and could point the link anywhere."""
+ return re.sub(r"([*`<>|_\\])", r"\\\1", text)
+
+
+def render(sponsors, size, use_company_name, seen):
+ """One line of comma-separated entries, plus any avatar substitution defs."""
+ entries, defs = [], []
+ for s in sponsors:
+ if s["login"].lower() in seen: # duplicate |av-x| defs are an RST error
+ continue
+ seen.add(s["login"].lower())
+ if s["private"]:
+ entries.append(rst_escape(mask(s["login"]))) # no avatar, no link: both would out them
+ continue
+ # DWORD/QWORD perks promise a company name; Byte/Word promise a username.
+ label = rst_escape(s["name"]) if use_company_name else "@" + s["login"]
+ link = f"`{label} <https://github.com/{s['login']}>`__"
+ if size:
+ entries.append(f"|av-{s['login']}| {link}")
+ defs += [f".. |av-{s['login']}| image:: https://github.com/{s['login']}.png?size={size}",
+ f" :target: https://github.com/{s['login']}",
+ f" :alt: {s['login']}", ""]
+ else:
+ entries.append(link)
+ body = ", ".join(entries)
+ return body + ("\n\n" + "\n".join(defs).rstrip() if defs else "")
+
+
+def render_readme(sponsors, original):
+ """Return README.rst with every marker block regenerated, or None if unchanged."""
+ text, seen = original, set()
+ for key, _floor, marker, placeholder, size in TIERS:
+ if marker is None:
+ continue
+ members = [s for s in sponsors if s["tier"] == key]
+ block = render(members, size, key in ("DWORD", "QWORD"), seen) if members else placeholder
+ pattern = re.compile(rf"(^\.\. {re.escape(marker)}-START$\n)(.*?)(^\.\. {re.escape(marker)}-END$)",
+ re.M | re.S)
+ if not pattern.search(text):
+ sys.exit(f"README.rst: marker {marker}-START/-END not found")
+ text = pattern.sub(lambda m: m.group(1) + "\n" + block + "\n\n" + m.group(3), text)
+ return None if text == original else text
+
+
+def codespell_collisions(original, new_text):
+ """Logins/names the repo's auto-fixing codespell hook would rewrite in place."""
+ added = [l for l in difflib.unified_diff(original.splitlines(),
+ new_text.splitlines(), n=0) if l.startswith("+")]
+ if not added:
+ return []
+ with tempfile.NamedTemporaryFile("w", suffix=".txt", delete=True) as probe:
+ # outside the repo and not dot-prefixed: codespell skips hidden files unless
+ # .codespellrc is picked up from cwd, which would make this guard fail open
+ probe.write("\n".join(added) + "\n")
+ probe.flush()
+ try:
+ run = subprocess.run(["codespell", "--ignore-words", str(IGNORE_WORDS), probe.name],
+ capture_output=True, text=True)
+ except FileNotFoundError:
+ print(" note: codespell not on PATH - generated block is UNCHECKED")
+ return []
+ if run.returncode not in (0, 65): # 65 = typos found; anything else is a tool error
+ print(f" note: codespell failed (rc={run.returncode}) - generated block is UNCHECKED")
+ return []
+ out = run.stdout
+ # v2.2.4 (the pinned hook) lowercases dictionary keys before testing ignore-words,
+ # so a cased entry would never match and -w would rewrite the login anyway.
+ return sorted({l.split(":", 2)[2].split("==>")[0].strip().lower()
+ for l in out.splitlines() if l.count(":") >= 2})
+
+
+def readme_diff(original, new_text):
+ return "".join(difflib.unified_diff(
+ original.splitlines(keepends=True), new_text.splitlines(keepends=True),
+ fromfile="README.rst", tofile="README.rst (new)", n=2))
+
+
+# ------------------------------------------------------------------ labels
+
+def as_logins(value, where):
+ """config.json is hand-edited: a bare string here would iterate as characters and
+ label the single-letter accounts it spells."""
+ if not isinstance(value, list) or not all(isinstance(x, str) for x in value):
+ sys.exit(f"config.json: {where} must be a list of logins, got {value!r}")
+ return [x for x in value if x]
+
+
+def entitlements(sponsors, config):
+ """login -> sorted labels, first matching rule only (Adafruit, then sponsor tier)."""
+ ent = {}
+
+ # GitHub logins are case-insensitive and config.json is hand-edited, so
+ # normalise everywhere. labeler.yml compares with .toLowerCase() for this reason.
+ skip = {x.lower() for x in as_logins(config.get("exclude", []), "exclude")}
+
+ def grant(login, labels):
+ login = login.lower()
+ if login and login not in skip: # the maintainer does not triage their own tickets
+ ent.setdefault(login, set(labels)) # setdefault: first rule wins, never a union
+
+ # Adafruit is evaluated FIRST, matching labeler.yml's branch order.
+ # public_members ONLY: /members returns concealed members to an org admin, and
+ # labelling one "Reported by an Adafruit member" publishes what they hid.
+ members = set(gh("api", "orgs/adafruit/public_members", "--paginate", "-q", ".[].login").split())
+ members |= set(as_logins(config.get("adafruit_members_extra", []), "adafruit_members_extra"))
+ for m in members:
+ grant(m, ADAFRUIT_LABELS)
+
+ # Rules mirror .github/workflows/labeler.yml, which applies these same labels
+ # when a ticket is opened. This pass backfills what the workflow cannot reach:
+ # tickets older than it, and private sponsors its GITHUB_TOKEN cannot see.
+ for s in sponsors:
+ # A private sponsor gets NO label. Every candidate set was measured against the
+ # live repo and each one identifies them: withholding `Sponsor 💖` leaves a bare
+ # `Prio Top 🚨`, which nothing else in the repo emits; and a bare `Prio 📌`
+ # appears on 1 of 204 open issues and 0 of 873 discussions. A label applied only
+ # to private sponsors IS the disclosure, whichever label it is. The triage perk
+ # cannot ride on a public label - honour it off-ticket.
+ if s["private"]:
+ continue
+ labels = set(TIER_LABELS[s["tier"]])
+ if not labels:
+ continue
+ if s["is_org"]:
+ members = as_logins({k.lower(): v for k, v in config.get("org_members", {}).items()}
+ .get(s["login"].lower(), []), f"org_members[{s['login']}]")
+ if not members:
+ who = mask(s["login"]) if s["private"] else s["login"]
+ print(f"note: org sponsor {who} ({s['tier']}) has no members in config.json - "
+ f"nothing to label")
+ for m in members:
+ grant(m, labels)
+ else:
+ grant(s["login"], labels)
+
+ return {k: sorted(v) for k, v in sorted(ent.items())}
+
+
+SCAN_Q = """
+query($cursor:String){ repository(owner:"%s",name:"%s"){ %s(first:100, %sorderBy:{field:CREATED_AT,direction:DESC}, after:$cursor){
+ pageInfo{hasNextPage endCursor}
+ nodes{ number id %s author{login} labels(first:40){nodes{name}} } } } }
+"""
+
+
+def scan(kind, since):
+ """Open tickets with number > since, newest first; stops at the watermark."""
+ states = "" if kind == "discussions" else "states:OPEN, "
+ closed = "closed" if kind == "discussions" else ""
+ query = SCAN_Q % (OWNER, NAME, kind, states, closed)
+ cursor, found = None, []
+ while True:
+ page = graphql(query, cursor=cursor)["repository"][kind]
+ for n in page["nodes"]:
+ if n["number"] <= since:
+ return found
+ if n.get("closed"):
+ continue
+ found.append({"number": n["number"], "id": n["id"],
+ "author": (n["author"] or {}).get("login"),
+ "labels": {x["name"] for x in n["labels"]["nodes"]}})
+ if not page["pageInfo"]["hasNextPage"]:
+ return found
+ cursor = page["pageInfo"]["endCursor"]
+
+
+def label_ids():
+ q = '{repository(owner:"%s",name:"%s"){labels(first:100){nodes{name id}}}}' % (OWNER, NAME)
+ return {n["name"]: n["id"] for n in graphql(q)["repository"]["labels"]["nodes"]}
+
+
+def highest_number():
+ q = ('{repository(owner:"%s",name:"%s"){'
+ 'issues(first:1,orderBy:{field:CREATED_AT,direction:DESC}){nodes{number}}'
+ 'pullRequests(first:1,orderBy:{field:CREATED_AT,direction:DESC}){nodes{number}}'
+ 'discussions(first:1,orderBy:{field:CREATED_AT,direction:DESC}){nodes{number}}}}') % (OWNER, NAME)
+ r = graphql(q)["repository"]
+ return max((v["nodes"][0]["number"] for v in r.values() if v["nodes"]), default=0)
+
+
+def plan_labels(ent, since, private):
+ """Open tickets above the watermark whose author is owed labels they lack."""
+ actions = []
+ # path segment differs per type, and doubles as the "which kind is this?" hint
+ for kind, path in (("issues", "issues"), ("pullRequests", "pull"), ("discussions", "discussions")):
+ for t in scan(kind, since):
+ want = sorted(set(ent.get((t["author"] or "").lower(), [])) - t["labels"])
+ if want:
+ actions.append({"number": t["number"], "id": t["id"], "add": want,
+ "author": mask(t["author"]) if (t["author"] or "").lower() in private else t["author"],
+ "url": f"https://github.com/{OWNER}/{NAME}/{path}/{t['number']}"})
+ return sorted(actions, key=lambda a: a["number"])
+
+
+def print_label_table(actions):
+ rows = [(a["url"], a["author"], ", ".join(a["add"])) for a in actions]
+ head = ("Ticket", "Author", "Labels to add")
+ # emoji render double-width, so pad by display width, not len()
+ width = lambda t: len(t) + sum(c > "\u2100" for c in t)
+ w = [max(width(r[i]) for r in rows + [head]) for i in range(3)]
+ pad = lambda t, i: t + " " * (w[i] - width(t))
+ print(" " + " ".join(pad(head[i], i) for i in range(3)))
+ print(" " + " ".join("-" * w[i] for i in range(3)))
+ for r in rows:
+ print(" " + " ".join(pad(r[i], i) for i in range(3)))
+
+
+def apply_label_actions(actions, ids):
+ for i, a in enumerate(actions, 1):
+ # IDs inlined: `gh api graphql -F` cannot pass a list variable.
+ graphql("mutation{addLabelsToLabelable(input:{labelableId:%s,labelIds:%s})"
+ "{clientMutationId}}" % (json.dumps(a["id"]),
+ json.dumps([ids[w] for w in a["add"]])))
+ print(f" [{i}/{len(actions)}] {a['url']}") # per ticket: a mid-run failure must be legible
+ print(f"labels: {len(actions)} ticket(s) updated")
+
+
+def confirm(question):
+ if not sys.stdin.isatty():
+ sys.exit("stdin is not a terminal - re-run with --yes to apply, or --dry-run to plan only")
+ return input(f"{question} [y/N] ").strip().lower() in ("y", "yes")
+
+
+# -------------------------------------------------------------------- main
+
+def print_rules():
+ """The applied policy, read out of the constants above so it cannot drift."""
+ print("Label rules mirror .github/workflows/labeler.yml (the source of truth for new "
+ "tickets).\nThis script backfills what that workflow cannot reach: tickets older "
+ "than it, and\nprivate sponsors, which its GITHUB_TOKEN cannot see at all.\n")
+ row = " {:<9} {:>5} {:<17} {:<22} {}"
+ print(row.format("Tier", "$/mo", "README section", "Listed as", "Labels"))
+ for key, floor, marker, _placeholder, size in TIERS:
+ company = key in ("DWORD", "QWORD")
+ listed = ("not listed" if marker is None else
+ (("logo + " if company else "avatar + ") if size else "")
+ + ("company name" if company else "@username"))
+ print(row.format(key, floor, marker or "-", listed, " ".join(sorted(TIER_LABELS[key]))))
+ print(row.format("Adafruit", "-", "hand-written", "-", " ".join(sorted(ADAFRUIT_LABELS))))
+ print("\nAdafruit membership comes from orgs/adafruit/public_members, never /members:"
+ "\n an org admin sees concealed members too, and the Adafruit label would publish"
+ "\n an affiliation those people deliberately hid."
+ "\nA private sponsor is masked in the README (first 3 chars, no avatar, no link) and"
+ "\n gets NO label at all: any label applied only to private sponsors is itself the"
+ "\n disclosure. Measured live - a bare Prio Top 🚨 is emitted by nothing else in the"
+ "\n repo, and a bare Prio 📌 by 1 of 204 open issues. Honour their perk off-ticket."
+ "\nRules are first-match-only (Adafruit, then tier), never a union: two priority"
+ "\n labels on one ticket double-count it in triage."
+ "\nOnly OPEN tickets are labelled, labels are only ever added, and a ticket reopened"
+ "\n below the watermark needs --full-rescan.")
+
+
+def main():
+ p = argparse.ArgumentParser(description=__doc__,
+ formatter_class=argparse.RawDescriptionHelpFormatter)
+ p.add_argument("--rules", action="store_true",
+ help="print the tier/label policy and exit")
+ p.add_argument("--dry-run", action="store_true",
+ help="preview and stop; writes nothing, not even state.json")
+ p.add_argument("--yes", action="store_true",
+ help="skip the confirmation prompt (required when stdin is not a terminal)")
+ p.add_argument("--full-rescan", action="store_true",
+ help="ignore the watermark and scan every open ticket")
+ only = p.add_mutually_exclusive_group()
+ only.add_argument("--readme-only", action="store_true", help="skip the label pass")
+ only.add_argument("--labels-only", action="store_true", help="skip the README pass")
+ args = p.parse_args()
+
+ if args.rules:
+ return print_rules()
+
+ if REPO != Path.cwd().resolve() and REPO not in Path.cwd().resolve().parents:
+ sys.exit(f"run from inside {REPO} - this script writes that checkout, not the cwd")
+
+ config = json.loads(CONFIG.read_text(encoding="utf-8"))
+ unknown = {k for k in config if not k.startswith("_")} - {"exclude", "org_members",
+ "adafruit_members_extra"}
+ if unknown: # a mistyped key reads as absent, and `exclude` failing open means
+ sys.exit(f"config.json: unknown key(s) {sorted(unknown)}") # labelling our own tickets
+ state = json.loads(STATE.read_text(encoding="utf-8")) if STATE.exists() else {}
+ original = README.read_text(encoding="utf-8") # one snapshot, re-checked before the write
+
+ sponsors = fetch_sponsors()
+ print(f"{len(sponsors)} active sponsor(s):")
+ for s in sponsors:
+ who = mask(s["login"]) + " (private)" if s["private"] else s["login"]
+ print(f" {s['since'][:10]} {s['tier']:<5} {who}")
+
+ # ---------------------------------------------------------------- plan
+ if not args.labels_only and not sponsors:
+ # Rewriting every section back to "be the first!" is indistinguishable from a
+ # token that cannot see the sponsorships. Refuse rather than wipe.
+ sys.exit("no active sponsorships returned - refusing to rewrite README.rst")
+ new_readme = None if args.labels_only else render_readme(sponsors, original)
+
+ actions, ids, watermark, fingerprint = [], {}, None, None
+ if not args.readme_only:
+ ent = entitlements(sponsors, config)
+ fingerprint = hashlib.sha256(json.dumps(ent, sort_keys=True).encode()).hexdigest()[:16]
+ changed = bool(state) and fingerprint != state.get("fingerprint")
+ since = 0 if args.full_rescan or changed else state.get("last_ticket", 0)
+ if changed:
+ print("entitlements changed since last run - rescanning all open tickets")
+ elif args.full_rescan:
+ print("--full-rescan - ignoring the watermark")
+ print(f"scanning open tickets above #{since}")
+ ids = label_ids()
+ watermark = highest_number()
+ # checked before the prompt: an unknown name must not KeyError mid-apply
+ unknown = {n for n in (L_SPONSOR, L_PRIO, L_PRIO_TOP, L_ADAFRUIT) if n not in ids}
+ if unknown:
+ sys.exit(f"labels missing from the repo: {sorted(unknown)}")
+ private = {s["login"].lower() for s in sponsors if s["private"]}
+ actions = plan_labels(ent, since, private)
+
+ # ------------------------------------------------------------- preview
+ collisions = codespell_collisions(original, new_readme) if new_readme else []
+ if not args.labels_only:
+ print("\nREADME.rst")
+ print(readme_diff(original, new_readme) if new_readme else " no change\n")
+ if collisions:
+ print(f" note: codespell (-w) would rewrite {', '.join(collisions)} in the generated block;\n"
+ f" adding them to {IGNORE_WORDS.relative_to(REPO)} on apply\n")
+ if not args.readme_only:
+ print("Tickets")
+ if actions:
+ print_label_table(actions)
+ else:
+ print(" no change")
+ print()
+
+ if not new_readme and not actions:
+ print("nothing to do")
+ return
+ if args.dry_run:
+ print("dry run - nothing applied")
+ return
+ if not args.yes and not confirm("Apply these changes?"):
+ sys.exit("aborted - nothing applied")
+
+ # --------------------------------------------------------------- apply
+ # Remote label writes go FIRST: they cannot be undone by git, so if they fail
+ # partway the local README edit has not happened and `git status` stays honest.
+ if actions:
+ apply_label_actions(actions, ids)
+ if new_readme:
+ if README.read_text(encoding="utf-8") != original: # edited during the prompt
+ sys.exit("README.rst changed while this run was in progress - "
+ "labels are applied, re-run for the README")
+ if collisions: # before the write, so the hook cannot mangle it first
+ have = [w for w in IGNORE_WORDS.read_text(encoding="utf-8").splitlines() if w.strip()]
+ IGNORE_WORDS.write_text("\n".join(sorted(set(have) | set(collisions))) + "\n", encoding="utf-8")
+ print(f"ignore-words.txt: added {', '.join(collisions)}")
+ README.write_text(new_readme, encoding="utf-8")
+ print("README.rst: updated")
+ # Written only here: a dry run or an aborted confirmation must leave the
+ # watermark alone, or the next run would skip tickets it never labelled.
+ if watermark is not None:
+ STATE.write_text(json.dumps(
+ {"last_ticket": watermark, "fingerprint": fingerprint, "updated": date.today().isoformat()},
+ indent=2) + "\n", encoding="utf-8")
+ print(f"state.json: last_ticket={watermark}")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/.claude/workflows/driver-review.js b/.claude/workflows/driver-review.js
index 255b8ac74..3d380c7e0 100644
--- a/.claude/workflows/driver-review.js
+++ b/.claude/workflows/driver-review.js
@@ -1,9 +1,9 @@
export const meta = {
name: 'driver-review',
- description: 'Review driver directories across dimensions with driver-reviewer scanners, then adversarially verify every finding; returns only confirmed findings',
+ description: 'Review driver directories across dimensions with code-verifier scanners, then adversarially verify every finding; returns only confirmed findings',
whenToUse: 'Auditing dcd/hcd drivers for a bug class (pass question) or a full-dimension review (default dimensions)',
phases: [
- { title: 'Scan', detail: 'driver-reviewer per (dir x dimension)' },
+ { title: 'Scan', detail: 'code-verifier per (dir x dimension)' },
{ title: 'Verify', detail: 'adversarial refutation per finding' },
],
}
@@ -58,7 +58,7 @@ const results = await pipeline(
p => agent(
`Review ${p.dir} for exactly one dimension: ${p.dim}. Read the sources yourself. Coverage-first — report everything, a verifier filters.`,
- { label: `scan:${short(p.dir)}`, phase: 'Scan', agentType: 'driver-reviewer', effort: 'xhigh', schema: FINDINGS },
+ { label: `scan:${short(p.dir)}`, phase: 'Scan', agentType: 'code-verifier', schema: FINDINGS },
),
(scan, p) => {
@@ -69,7 +69,8 @@ const results = await pipeline(
`Adversarially verify ONE review finding about ${p.dir}.\nDimension: ${p.dim}\nFinding: ${JSON.stringify(f)}\n` +
'Read the cited code plus enough context (callers, ISR paths, macros, and the datasheet if register-related) to judge. ' +
'Try to REFUTE it; real=true only if it survives your best attempt. Return {"real": bool, "reason": string}.',
- { label: `verify:${short(p.dir)}:${f.line}`, phase: 'Verify', agentType: 'driver-reviewer', effort: 'xhigh', schema: VERDICT },
+ // max: one judgment-dense call per finding decides what survives - worth the top tier
+ { label: `verify:${short(p.dir)}:${f.line}`, phase: 'Verify', agentType: 'code-verifier', effort: 'max', schema: VERDICT },
).then(v => v && { ...f, verdict: v })
)).then(vs => {
const alive = vs.filter(Boolean)
diff --git a/.claude/workflows/fanout-dev.js b/.claude/workflows/fanout-dev.js
index e98a86f1d..386f6df46 100644
--- a/.claude/workflows/fanout-dev.js
+++ b/.claude/workflows/fanout-dev.js
@@ -1,11 +1,11 @@
export const meta = {
name: 'fanout-dev',
- description: 'Implement one described change across many ports/file-sets: one port-dev worker per item, independent builder verification, optional review',
+ description: 'Implement one described change across many ports/file-sets: one code-writer worker per item, independent builder verification, optional review',
whenToUse: 'Applying a fix or pattern across multiple TinyUSB ports (e.g. the same DCD bug in several drivers)',
phases: [
- { title: 'Implement', detail: 'port-dev per item (opus xhigh)' },
+ { title: 'Implement', detail: 'code-writer per item (opus xhigh)' },
{ title: 'Verify', detail: 'builder single-example check' },
- { title: 'Review', detail: 'optional driver-reviewer pass' },
+ { title: 'Review', detail: 'optional code-verifier pass' },
],
}
@@ -71,7 +71,7 @@ const results = await pipeline(
: ' Pick a verification board from hw/bsp whose family uses this scope.'),
{
label: `dev:${short(item)}`, phase: 'Implement',
- agentType: 'port-dev', effort: 'xhigh', schema: DEV,
+ agentType: 'code-writer', schema: DEV,
...(args.worktree ? { isolation: 'worktree' } : {}),
},
),
@@ -96,7 +96,7 @@ const results = await pipeline(
return agent(
`Review the uncommitted change in ${item} (inspect with: git diff -- ${item}) against this task:\n${args.task}\n` +
'Dimension: does the diff correctly and completely implement the task with no unintended side effects? Coverage-first findings.',
- { label: `review:${short(item)}`, phase: 'Review', agentType: 'driver-reviewer', effort: 'xhigh', schema: FINDINGS },
+ { label: `review:${short(item)}`, phase: 'Review', agentType: 'code-verifier', schema: FINDINGS },
).then(f => {
// review: array = findings; null = reviewer died; absent = not requested
if (!f) log(`review:${short(item)}: reviewer agent died`)
diff --git a/.claude/workflows/hil-validate.js b/.claude/workflows/hil-validate.js
index bc0bda8b9..741ba481d 100644
--- a/.claude/workflows/hil-validate.js
+++ b/.claude/workflows/hil-validate.js
@@ -11,10 +11,10 @@ if (!args || !Array.isArray(args.boards) || args.boards.length === 0) {
throw new Error('args must be { boards: string[], force? } with the boards already built')
}
-// The operator returns hil_summary.py's JSON verbatim plus its own observations. It does NOT
+// The operator returns hil_report.py's JSON verbatim plus its own observations. It does NOT
// retype the report table: rows are named per variant, a variant need not start with the board
// name, and lock contention is a cell rather than a phrase — rebuilding board identity from
-// prose produced a defect in each of four review rounds. hil_summary.py does that join against
+// prose produced a defect in each of four review rounds. hil_report.py does that join against
// the roster, so `locked` and `ran` arrive as fields and nothing here parses a detail string.
const BOARD = {
type: 'object', additionalProperties: false,
@@ -26,12 +26,16 @@ const BOARD = {
}
const HIL = {
type: 'object', additionalProperties: false,
- required: ['results', 'wedged'],
+ required: ['results', 'wedged', 'caveat'],
properties: {
results: { type: 'array', items: BOARD },
// the operator's own observation — not derivable from the report
wedged: { type: 'array', items: { type: 'string' } },
banner: { type: 'string' },
+ // the run-level caveat: abandoned / aborted / selected-no-boards. `banner` carries rig
+ // HEALTH across an --accumulate retry; `caveat` carries how THIS run ended, and every
+ // row can still say pass while it failed — so it gates `pass` in summarize() below.
+ caveat: { type: 'string' },
},
}
@@ -51,12 +55,12 @@ const runBoards = (boards, isRetry = false) => agent(
(args.force
? 'THE USER HAS EXPLICITLY AUTHORIZED FORCING: run hil_test.py with HIL_NO_BOARD_LOCK=1 in the environment (bypasses the board lock check; do NOT release or kill the existing holder). '
: 'A board whose lock is held (a dev session or concurrent CI job) fails fast inside the run without blocking the others — never force the lock. ') +
- 'If hil_test.py refuses the run with "board(s) not in <config>", re-run it WITHOUT the unknown names but keep the FULL board list on the hil_summary call below — it emits a ran:false entry for every board you name, so the unknown ones surface as "no report row" instead of costing the whole batch. ' +
+ 'If hil_test.py refuses the run with "board(s) not in <config>", re-run it WITHOUT the unknown names but keep the FULL board list on the hil_report call below — it emits a ran:false entry for every board you name, so the unknown ones surface as "no report row" instead of costing the whole batch. ' +
'Use the config for this host (hostname first). Run hil_test.py as a BACKGROUND Bash task and wait for it (a stuck fleet runs to its pool guard, 60 min by default — beyond any foreground timeout); never cancel it early. ' +
'On non-lock failures retry ONCE from the re-run spec hil_test.py just wrote — `<config>.failed`, which already begins with --accumulate — adding -v. A usbtest battery that produced per-case verdicts is NOT auto-retried, so its result already stands. ' +
'THEN, from the directory the run wrote its report to, produce the results with:\n' +
- ` python3 test/hil/helper/hil_summary.py <the config you used> ${boards.map((b) => `-b ${b}`).join(' ')}\n` +
- 'Return its `results` array and `banner` EXACTLY as printed — do not retype, reword, re-order or "correct" them, and never transcribe the markdown table instead. ' +
+ ` python3 test/hil/helper/hil_report.py <the config you used> ${boards.map((b) => `-b ${b}`).join(' ')}\n` +
+ 'Return its `results` array, `banner` and `caveat` EXACTLY as printed — do not retype, reword, re-order or "correct" them, and never transcribe the markdown table instead. ' +
'Add `wedged`: the board names whose board or fixture your run left unresponsive (usually none). That is your own observation and the one field you author; put `dmesg | tail -50` in your reply text for any board you list.',
{
label: boards.length === 1 ? `hil:${boards[0]}` : `hil:${boards.length} boards`,
@@ -64,7 +68,7 @@ const runBoards = (boards, isRetry = false) => agent(
},
)
-// A lookup, not a reconciliation: hil_summary.py emits exactly one entry per requested board,
+// A lookup, not a reconciliation: hil_report.py emits exactly one entry per requested board,
// so a missing entry means the operator dropped it rather than that the names disagree.
const byBoard = (out) => new Map((out?.results || [])
.filter((r) => r && typeof r.board === 'string')
@@ -91,7 +95,9 @@ const results = args.boards.map((b) => {
})
for (const r of results) log(`${r.board}: ${r.pass ? 'PASS' : r.locked ? 'LOCKED' : 'FAIL'}`)
if (first?.banner) log(`report banner: ${first.banner.trim().split('\n')[0]}`)
+if (first?.caveat) log(`report caveat: ${first.caveat.trim().split('\n')[0]}`)
+let runCaveat = first?.caveat || ''
// A concurrent CI job may have held some boards (its hil_test.py flock).
// CI finishes a board in minutes — retry locked boards once, at the end.
if (!args.force) {
@@ -116,19 +122,26 @@ if (!args.force) {
}
log(`${b}: retry ${results[i].pass ? 'PASS' : 'FAIL'}`)
}
+ // the retry's own run-level verdict, not the first attempt's: a retry that abandoned
+ // or aborted must sink the run even though its rows may all say pass.
+ if (again?.caveat) runCaveat = again.caveat
}
}
-// pass/wedged/locked in one place so it can be exercised without running an agent
-const summarize = (rs, force) => ({
- pass: rs.every((r) => r.pass),
+// pass/wedged/locked in one place so it can be exercised without running an agent.
+// `caveat` is a RUN-level verdict and must gate `pass`: on the abandon and no-boards
+// paths every row can legitimately say pass while the run itself failed (hil_test.py
+// os._exit(1) -- a red job), so per-row agreement alone published those runs as green.
+const summarize = (rs, force, caveat) => ({
+ pass: rs.every((r) => r.pass) && !/^\*\*HIL run (abandoned|aborted|selected no boards)/m
+ .test(caveat || ''),
wedged: rs.filter((r) => r.wedged).map((r) => r.board),
locked: force ? [] : rs.filter((r) => !r.pass && r.locked).map((r) => r.board),
})
-const { pass, wedged, locked } = summarize(results, args.force)
+const { pass, wedged, locked } = summarize(results, args.force, runCaveat)
if (wedged.length) log(`WEDGED boards needing usb-kernel-recover: ${wedged.join(', ')}`)
// Workers cannot prompt the user — surface still-locked boards for the main
// session to ask: force (re-invoke with force: true), wait, or accept.
if (locked.length) log(`still locked after retry: ${locked.join(', ')} — ask the user: force / keep waiting / accept`)
-return { pass, results, wedged, locked }
+return { pass, results, wedged, locked, caveat: runCaveat }
diff --git a/.claude/workflows/pr-babysit.js b/.claude/workflows/pr-babysit.js
index 406213a5f..2731ed254 100644
--- a/.claude/workflows/pr-babysit.js
+++ b/.claude/workflows/pr-babysit.js
@@ -1,47 +1,55 @@
export const meta = {
name: 'pr-babysit',
- description: 'Drive a PR to green: pr-monitor triage (CI + bot reviews), port-dev fixes for validated findings, driver-reviewer verification, one commit+push per cycle',
+ description: 'Drive a PR to green: a fast review lane (validate bot findings, fix, push without waiting on CI) overlapped with a CI-watch lane; code-writer fixes, code-verifier verification, at most one push per lane per cycle',
whenToUse: 'After opening a PR, from a checkout of the PR branch. Default is a dry run (fixes left uncommitted, nothing posted); passing autoPush: true is the explicit authorization for pushes and PR comments.',
phases: [{ title: 'Triage' }, { title: 'Fix' }, { title: 'Verify' }, { title: 'Push' }],
}
-// args: { pr: number, maxCycles?: number, autoPush?: boolean (default false = dry run) }
+// args: { pr: number, maxCycles?: number, autoPush?: boolean (default false = dry run),
+// checkoutDir?: string (PR branch checkout; default: the session working dir) }
if (typeof args === 'string') { try { args = JSON.parse(args) } catch { /* not JSON: shape check below reports it */ } }
if (!args || !args.pr) {
- throw new Error('args must be { pr: number, maxCycles?, autoPush? }; run from a checkout of the PR branch')
+ throw new Error('args must be { pr: number, maxCycles?, autoPush?, checkoutDir? }; run from the PR branch checkout or point checkoutDir at it')
}
args.pr = Number(args.pr)
if (!Number.isInteger(args.pr) || args.pr <= 0) {
throw new Error('args.pr must be a positive integer PR number')
}
+const checkoutDir = args.checkoutDir || '.'
+if (typeof checkoutDir !== 'string' || checkoutDir.includes("'")) {
+ throw new Error('checkoutDir must be a plain path string')
+}
+const IN_CHECKOUT = checkoutDir === '.' ? 'The working tree IS the PR checkout. '
+ : `The PR branch checkout is at ${checkoutDir} - run every git/build/file command there, not in the session directory. `
const maxCycles = args.maxCycles ?? 3
if (!Number.isInteger(maxCycles) || maxCycles < 1) {
throw new Error('maxCycles must be an integer >= 1')
}
-const TRIAGE = {
+const CI = {
type: 'object', additionalProperties: false,
- required: ['ci', 'findings', 'replies', 'done'],
+ required: ['status', 'infraRerun', 'realFailures'],
properties: {
- ci: {
- type: 'object', additionalProperties: false,
- required: ['status', 'infraRerun', 'realFailures'],
- properties: {
- status: { type: 'string', enum: ['green', 'red', 'running'] },
- infraRerun: { type: 'array', items: { type: 'string' } },
- realFailures: {
- type: 'array',
- items: {
- type: 'object', additionalProperties: false,
- required: ['check', 'firstError', 'files'],
- properties: {
- check: { type: 'string' }, firstError: { type: 'string' },
- files: { type: 'array', items: { type: 'string' } },
- },
- },
+ status: { type: 'string', enum: ['green', 'red', 'running'] },
+ infraRerun: { type: 'array', items: { type: 'string' } },
+ realFailures: {
+ type: 'array',
+ items: {
+ type: 'object', additionalProperties: false,
+ required: ['check', 'firstError', 'files', 'rigSide'],
+ properties: {
+ check: { type: 'string' }, firstError: { type: 'string' },
+ files: { type: 'array', items: { type: 'string' } },
+ rigSide: { type: 'boolean' },
},
},
},
+ },
+}
+const REVIEWS = {
+ type: 'object', additionalProperties: false,
+ required: ['findings', 'replies', 'done'],
+ properties: {
findings: {
type: 'array',
items: {
@@ -84,6 +92,19 @@ const OP = {
required: ['pass', 'detail'],
properties: { pass: { type: 'boolean' }, detail: { type: 'string' } },
}
+const SCOPE = {
+ type: 'object', additionalProperties: false,
+ required: ['files'],
+ properties: { files: { type: 'array', items: { type: 'string' } } },
+}
+const OPIDS = {
+ type: 'object', additionalProperties: false,
+ required: ['pass', 'detail', 'doneIds'],
+ properties: {
+ pass: { type: 'boolean' }, detail: { type: 'string' },
+ doneIds: { type: 'array', items: { type: 'integer' } },
+ },
+}
// Marking a review thread resolved has no REST endpoint — it needs the
// GraphQL resolveReviewThread mutation. Shared recipe handed to the posting
@@ -107,120 +128,249 @@ const postReplyRecipe = (noun) =>
const history = []
const repliedIds = new Set() // issue comments can't be thread-resolved, so they re-harvest every cycle — never reply twice
-for (let cycle = 1; cycle <= maxCycles; cycle++) {
- const t = await agent(
- `Triage PR #${args.pr}. If checks are still running, wait for them first (gh pr checks ${args.pr} --watch as a BACKGROUND Bash task; the foreground timeout is capped at 10 min). ` +
- 'Then follow your triage procedure: classify CI failures, re-run infra ones, harvest and adversarially validate bot review findings, draft replies for invalid/stale ones.',
- { label: `triage#${cycle}`, phase: 'Triage', agentType: 'pr-monitor', schema: TRIAGE },
- )
- if (!t) {
- history.push({ cycle, error: 'pr-monitor died' })
- return { pass: false, cycles: cycle, history, reason: 'pr-monitor-died' }
- }
- const entry = { cycle, triage: t }
- history.push(entry)
- // Post drafted replies to REFUTED findings as soon as triage produces them —
- // decoupled from fixing/pushing so done/unactionable cycles still post.
- // Reply AND resolve the thread. Outward-facing, so gated on autoPush.
- const freshReplies = t.replies.filter(r => !repliedIds.has(r.commentId))
- if (freshReplies.length > 0 && args.autoPush === true) {
- const posted = await agent(
- `Reply to and resolve these refuted review comments on PR #${args.pr}. For each: ${postReplyRecipe('reply')}` +
- `Replies: ${JSON.stringify(freshReplies)}. pass=true only if every reply was posted and every inline thread resolved; detail = what went where.`,
- { label: `replies#${cycle}`, phase: 'Push', model: 'sonnet', schema: OP },
- )
- // attempted counts as replied: better to drop a failed reply than spam duplicates
- freshReplies.forEach(r => repliedIds.add(r.commentId))
- if (!posted || !posted.pass) log(`cycle ${cycle}: refuted reply/resolve incomplete — ${posted ? posted.detail : 'agent died'}`)
- }
-
- if (t.done) {
- log(`cycle ${cycle}: PR is green with no unresolved valid findings`)
- return { pass: true, cycles: cycle, history }
+// Canonicalize a repo-relative path for set/collision comparison: resolve ./..
+// segments, unify separators; '' for anything that escapes the repo or uses
+// characters no repo path does (also makes the path shell-safe to interpolate).
+const canon = (p) => {
+ const s = String(p).trim().replace(/\\/g, '/')
+ // Absolute (CI-runner) paths: reject rather than corrupt into a bogus relative
+ // path — the file-less group then routes through the scoper, which recovers the
+ // real repo path and is existence-checked.
+ if (s.startsWith('/')) return ''
+ const out = []
+ for (const seg of s.split('/')) {
+ if (!seg || seg === '.') continue
+ if (seg === '..') { if (out.pop() === undefined) return '' } else out.push(seg)
}
+ const c = out.join('/')
+ return /^[A-Za-z0-9._+/-]+$/.test(c) ? c : ''
+}
- // Group actionable work by top-level scope (plain JS — no model tokens).
+// Group actionable notes by top-level scope (plain JS — no model tokens).
+const groupWork = (notes) => {
const groups = new Map()
- const groupOf = (key) => {
+ for (const n of notes) {
+ const key = (canon(n.scopeFile) || n.scopeFile).split('/').slice(0, 3).join('/')
if (!groups.has(key)) groups.set(key, { key, files: new Set(), notes: [] })
- return groups.get(key)
- }
- for (const f of t.findings.filter(x => x.verdict === 'valid')) {
- const g = groupOf(f.file.split('/').slice(0, 3).join('/'))
- g.files.add(f.file)
- g.notes.push(`${f.file}:${f.line} [${f.source}] ${f.claim} — hint: ${f.fixHint}`)
+ const g = groups.get(key)
+ n.files.forEach(f => { const c = canon(f); if (c) g.files.add(c) })
+ g.notes.push(n.text)
}
- for (const rf of t.ci.realFailures) {
- const g = groupOf((rf.files[0] || rf.check).split('/').slice(0, 3).join('/'))
- rf.files.forEach(x => g.files.add(x))
- g.notes.push(`CI ${rf.check}: ${rf.firstError}`)
- }
- const work = [...groups.values()]
+ return [...groups.values()]
+}
- if (work.length === 0) {
- if (t.ci.status === 'running' || t.ci.infraRerun.length > 0) {
- log(`cycle ${cycle}: only infra re-runs in flight — next cycle waits on them`)
- continue
+// Fix + verify one work list; returns { ok, fixes } — ok only if every group
+// was scoped, fixed by a live worker, AND passed code-verifier verification.
+const fixAndVerify = async (workIn) => {
+ // code-writer's contract needs an explicit file set: a group whose notes named no
+ // files (a CI failure whose log yielded no paths) is scoped by a dedicated agent
+ // first; if that fails too, the group is withheld (ok=false → human review) rather
+ // than dispatched with an invalid scope.
+ const fileless = workIn.filter(w => w.files.size === 0)
+ await parallel(fileless.map(w => () =>
+ agent(
+ `${IN_CHECKOUT}Determine which repo files must change to address these notes (read the code; if a note is a CI failure, read its CI log too):\n- ${w.notes.join('\n- ')}\n` +
+ 'files = repo-relative paths; empty only if genuinely undeterminable.',
+ { label: `scope:${w.key}`, phase: 'Fix', model: 'sonnet', schema: SCOPE },
+ ).then(s => s && s.files.forEach(f => { const c = canon(f); if (c) w.files.add(c) }))))
+ // Scoped paths are model output: keep only what git ls-files confirms exists.
+ // The check is executed (by a mechanical agent) and intersected here — a dead
+ // checker drops every candidate, so unconfirmed groups fall through to withheld.
+ const candidates = [...new Set(fileless.flatMap(w => [...w.files]))]
+ if (candidates.length > 0) {
+ const v = await agent(
+ `${IN_CHECKOUT}Run exactly: git ls-files -- ${candidates.join(' ')}\nReturn files = the paths that command printed, verbatim — no additions, no substitutions.`,
+ { label: 'scope:verify', phase: 'Fix', model: 'haiku', schema: SCOPE },
+ )
+ const exists = new Set((v ? v.files : []).map(canon))
+ for (const w of fileless) for (const f of [...w.files])
+ if (!exists.has(f)) { w.files.delete(f); log(`scope:${w.key}: dropped ${f} — not confirmed as a repo file`) }
+ }
+ const unscoped = workIn.filter(w => w.files.size === 0)
+ for (const w of unscoped) log(`fix for ${w.key}: no file scope determinable — withheld for human review`)
+ // Scoping can make groups overlap (two checks resolving to the same file); merge
+ // intersecting groups (to closure) so two fixers never edit one file concurrently.
+ const work = []
+ for (let g of workIn.filter(w => w.files.size > 0)) {
+ for (let i; (i = work.findIndex(m => [...g.files].some(f => m.files.has(f)))) >= 0;) {
+ const [m] = work.splice(i, 1)
+ g.files.forEach(f => m.files.add(f)); m.notes.push(...g.notes); m.key = `${m.key}+${g.key}`
+ g = m
}
- log(`cycle ${cycle}: nothing actionable`)
- return { pass: false, cycles: cycle, history, reason: 'unactionable' }
+ work.push(g)
}
-
+ const scopeOf = (w) => [...w.files].join(', ')
const fixes = await pipeline(
work,
w => agent(
- `Fix the following issues on the current PR branch (the working tree IS the PR checkout).\n` +
- `Scope: ${[...w.files].join(', ')}\nIssues:\n- ${w.notes.join('\n- ')}`,
- { label: `fix:${w.key}`, phase: 'Fix', agentType: 'port-dev', effort: 'xhigh', schema: DEV },
+ `Fix the following issues on the PR branch. ${IN_CHECKOUT}\n` +
+ `Scope: ${scopeOf(w)}\nIssues:\n- ${w.notes.join('\n- ')}`,
+ { label: `fix:${w.key}`, phase: 'Fix', agentType: 'code-writer', schema: DEV },
),
(fix, w) => fix && agent(
- `Verify the uncommitted changes for ${[...w.files].join(', ')} (use git diff -- <files>, and read any newly created untracked files directly) address these issues:\n- ${w.notes.join('\n- ')}\n` +
+ `${IN_CHECKOUT}Verify the uncommitted changes for ${scopeOf(w)} (use git diff -- <the files above>, and read any newly created untracked files directly) address these issues:\n- ${w.notes.join('\n- ')}\n` +
'Return {"addresses": bool, "reason": string}.',
- { label: `check:${w.key}`, phase: 'Verify', agentType: 'driver-reviewer', effort: 'xhigh', schema: CHECK },
+ { label: `check:${w.key}`, phase: 'Verify', agentType: 'code-verifier', schema: CHECK },
).then(v => ({ ...fix, addresses: !!(v && v.addresses), checkReason: v ? v.reason : 'verifier died' })),
)
- const aliveFixes = fixes.filter(Boolean)
- if (aliveFixes.length < work.length) log(`${work.length - aliveFixes.length} fix group(s) lost to dead workers`)
- entry.fixes = aliveFixes
-
- if (args.autoPush !== true) {
- log('autoPush not set: fixes left uncommitted in the working tree (dry run)')
- return { pass: false, cycles: cycle, history, dryRun: true }
- }
-
- // Verification gates the push: never push a cycle containing an unverified
- // fix or the partial edits of a dead worker.
- const unverified = aliveFixes.filter(f => f.addresses !== true)
- if (aliveFixes.length < work.length || unverified.length > 0) {
- for (const f of unverified) log(`fix for ${f.item}: failed verification — ${f.checkReason}`)
- log(`cycle ${cycle}: fixes left uncommitted for human review — not pushing unverified changes`)
- return { pass: false, cycles: cycle, history, reason: 'fix-verification-failed' }
- }
+ const alive = fixes.filter(Boolean)
+ if (alive.length < work.length) log(`${work.length - alive.length} fix group(s) lost to dead workers`)
+ const unverified = alive.filter(f => f.addresses !== true)
+ for (const f of unverified) log(`fix for ${f.item}: failed verification — ${f.checkReason}`)
+ return { ok: unscoped.length === 0 && alive.length === work.length && unverified.length === 0, fixes: alive }
+}
+// Verification gates every push: never push unverified or partial edits.
+const commitAndPush = async (cycle, what) => {
const push = await agent(
- `On the current PR branch: commit ALL working-tree changes as ONE commit (imperative message summarizing the cycle-${cycle} fixes for PR #${args.pr}, repo commit conventions), ` +
+ `${IN_CHECKOUT}On the PR branch: commit ALL working-tree changes as ONE commit (imperative message summarizing the cycle-${cycle} ${what} fixes for PR #${args.pr}, repo commit conventions), ` +
"then push to the PR's remote branch. pass=true only if commit AND push succeeded; detail = pushed SHA.",
- { label: `push#${cycle}`, phase: 'Push', model: 'sonnet', schema: OP },
+ { label: `push#${cycle}-${what}`, phase: 'Push', model: 'sonnet', schema: OP },
+ )
+ return push && push.pass ? push : null
+}
+
+for (let cycle = 1; cycle <= maxCycles; cycle++) {
+ // Two independent lanes, launched together. The review lane never waits on
+ // CI: it validates, fixes, and pushes while the CI lane is still watching.
+ const ciPromise = agent(
+ `Watch CI for PR #${args.pr} per your procedure; wait for pending checks.`,
+ { label: `ci#${cycle}`, phase: 'Triage', agentType: 'pr-ci-watcher', schema: CI },
+ ).catch(e => { log(`cycle ${cycle}: pr-ci-watcher errored — ${e && e.message}`); return null })
+ // Every early return below leaves the loop while the CI lane is still
+ // running: settle it first so no CI agent outlives the workflow.
+ const stopWith = async (result) => { await ciPromise; return result }
+
+ const r = await agent(
+ `Validate the bot review findings on PR #${args.pr} per your procedure. ${IN_CHECKOUT}`,
+ { label: `reviews#${cycle}`, phase: 'Triage', agentType: 'pr-review-validator', schema: REVIEWS },
)
- if (!push || !push.pass) {
- log(`cycle ${cycle}: push failed — stopping`)
- return { pass: false, cycles: cycle, history, reason: 'push-failed' }
+ if (!r) {
+ history.push({ cycle, error: 'pr-review-validator died' })
+ return await stopWith({ pass: false, cycles: cycle, history, reason: 'review-validator-died' })
+ }
+ const entry = { cycle, reviews: r }
+ history.push(entry)
+ // Outward reply/resolve attempts this cycle that did not fully complete; a green
+ // PR must not terminate the loop while any remain, or the retry never happens.
+ let pendingReplies = 0
+
+ // Post drafted replies to REFUTED findings immediately. Outward-facing,
+ // so gated on autoPush.
+ const freshReplies = r.replies.filter(x => !repliedIds.has(x.commentId))
+ if (freshReplies.length > 0 && args.autoPush === true) {
+ const posted = await agent(
+ `Reply to and resolve these refuted review comments on PR #${args.pr}. For each: ${postReplyRecipe('reply')}` +
+ 'If a thread already carries an identical reply of ours (a prior attempt that posted but failed to resolve), do not repost — just resolve it. ' +
+ `Replies: ${JSON.stringify(freshReplies)}. pass=true only if every reply was posted and every inline thread resolved; detail = what went where. ` +
+ 'doneIds = the commentIds fully handled: reply posted (or already present) AND (thread resolved, or an issue comment with no thread to resolve).',
+ { label: `replies#${cycle}`, phase: 'Push', model: 'sonnet', schema: OPIDS },
+ )
+ // Per-id accounting, matching the resolve path: only fully handled ids are marked
+ // replied; a failed reply/resolve stays fresh and retries next cycle (the prompt's
+ // already-present check keeps the retry from duplicating the reply).
+ for (const id of (posted && posted.doneIds) || []) repliedIds.add(id)
+ pendingReplies += freshReplies.filter(x => !repliedIds.has(x.commentId)).length
+ if (!posted || !posted.pass) log(`cycle ${cycle}: refuted reply/resolve incomplete — ${posted ? posted.detail : 'agent died'}`)
}
- // The valid bot findings were fixed and pushed — answer each inline comment
- // with what changed and resolve its thread. CI-failure work has no comment.
- const fixed = t.findings.filter(x => x.verdict === 'valid')
- if (fixed.length > 0) {
+ // ---- review lane: fix + push without waiting for CI ----
+ const validFindings = r.findings.filter(x => x.verdict === 'valid')
+ let reviewPushed = false
+ if (validFindings.length > 0) {
+ const work = groupWork(validFindings.map(f => ({
+ scopeFile: f.file, files: [f.file],
+ text: `${f.file}:${f.line} [${f.source}] ${f.claim} — hint: ${f.fixHint}`,
+ })))
+ const { ok, fixes } = await fixAndVerify(work)
+ entry.reviewFixes = fixes
+ if (args.autoPush !== true) {
+ log('autoPush not set: review-lane fixes left uncommitted (dry run)')
+ return await stopWith({ pass: false, cycles: cycle, history, dryRun: true })
+ }
+ if (!ok) {
+ log(`cycle ${cycle}: review-lane fixes left uncommitted for human review — not pushing unverified changes`)
+ return await stopWith({ pass: false, cycles: cycle, history, reason: 'fix-verification-failed' })
+ }
+ const push = await commitAndPush(cycle, 'review')
+ if (!push) {
+ log(`cycle ${cycle}: review-lane push failed — stopping`)
+ return await stopWith({ pass: false, cycles: cycle, history, reason: 'push-failed' })
+ }
+ reviewPushed = true
const resolved = await agent(
`The fixes for PR #${args.pr}'s valid review findings were just committed and pushed (${push.detail}). ` +
`For each finding below: ${postReplyRecipe('fix note')}` +
'Each reply states the finding is fixed in the pushed commit, with one line on the change. ' +
- `Findings: ${JSON.stringify(fixed.map(f => ({ commentId: f.commentId, file: f.file, line: f.line, claim: f.claim, fixHint: f.fixHint })))}. ` +
- 'pass=true only if every reply was posted and every thread resolved; detail = what went where.',
- { label: `resolve#${cycle}`, phase: 'Push', model: 'sonnet', schema: OP },
+ `Findings: ${JSON.stringify(validFindings.map(f => ({ commentId: f.commentId, file: f.file, line: f.line, claim: f.claim, fixHint: f.fixHint })))}. ` +
+ 'pass=true only if every reply was posted and every thread resolved; detail = what went where. ' +
+ 'doneIds = the commentIds fully handled: reply posted AND (thread resolved, or an issue comment with no thread to resolve).',
+ { label: `resolve#${cycle}`, phase: 'Push', model: 'sonnet', schema: OPIDS },
)
+ // Per-id accounting: a fully handled finding never re-replies (an issue comment
+ // has no thread to resolve, so it re-harvests as stale next cycle and would get
+ // a duplicate "fixed" note); an unfinished one stays out of repliedIds so its
+ // reply/resolve is retried next cycle instead of silently abandoned.
+ for (const id of (resolved && resolved.doneIds) || []) repliedIds.add(id)
+ pendingReplies += validFindings.filter(f => !repliedIds.has(f.commentId)).length
if (!resolved || !resolved.pass) log(`cycle ${cycle}: fixed reply/resolve incomplete — ${resolved ? resolved.detail : 'agent died'}`)
}
+
+ // ---- CI lane result ----
+ const c = await ciPromise
+ entry.ci = c
+ if (!c) {
+ log(`cycle ${cycle}: pr-ci-watcher died — re-arming`)
+ continue
+ }
+ if (reviewPushed) {
+ // The push restarted CI: this cycle's CI verdict is superseded. Re-arm;
+ // next cycle's ci#N watches the fresh run.
+ log(`cycle ${cycle}: review-lane push superseded the CI run — re-arming`)
+ continue
+ }
+ const rigSide = c.realFailures.filter(rf => rf.rigSide)
+ for (const rf of rigSide) log(`cycle ${cycle}: rig-side CI failure (not fixing): ${rf.check} — ${rf.firstError.slice(0, 120)}`)
+ const fixable = c.realFailures.filter(rf => !rf.rigSide)
+ if (fixable.length > 0) {
+ const work = groupWork(fixable.map(rf => ({
+ scopeFile: rf.files[0] || rf.check, files: rf.files,
+ text: `CI ${rf.check}: ${rf.firstError}`,
+ })))
+ const { ok, fixes } = await fixAndVerify(work)
+ entry.ciFixes = fixes
+ if (args.autoPush !== true) {
+ log('autoPush not set: CI-lane fixes left uncommitted (dry run)')
+ return { pass: false, cycles: cycle, history, dryRun: true }
+ }
+ if (!ok) {
+ log(`cycle ${cycle}: CI-lane fixes left uncommitted for human review — not pushing unverified changes`)
+ return { pass: false, cycles: cycle, history, reason: 'fix-verification-failed' }
+ }
+ if (!(await commitAndPush(cycle, 'ci'))) {
+ log(`cycle ${cycle}: CI-lane push failed — stopping`)
+ return { pass: false, cycles: cycle, history, reason: 'push-failed' }
+ }
+ continue // pushed: fresh CI run next cycle
+ }
+ if (r.done && c.status === 'green') {
+ if (pendingReplies > 0) {
+ log(`cycle ${cycle}: PR green but ${pendingReplies} reply/resolve unfinished — re-arming to retry`)
+ continue
+ }
+ log(`cycle ${cycle}: PR is green with no unresolved valid findings`)
+ return { pass: true, cycles: cycle, history }
+ }
+ if (r.done && rigSide.length > 0 && fixable.length === 0 && c.infraRerun.length === 0 && c.status !== 'running') {
+ log(`cycle ${cycle}: CI red only from rig-side failures — human/rig attention needed, nothing to fix in the PR`)
+ return { pass: false, cycles: cycle, history, reason: 'ci-red-rig-side' }
+ }
+ if (c.status === 'running' || c.infraRerun.length > 0) {
+ log(`cycle ${cycle}: CI still settling (${c.infraRerun.length} infra re-run(s)) — re-arming`)
+ continue
+ }
+ log(`cycle ${cycle}: nothing actionable`)
+ return { pass: false, cycles: cycle, history, reason: 'unactionable' }
}
return { pass: false, cycles: maxCycles, history, reason: 'maxCycles reached' }
diff --git a/.claude/workflows/test-hil-validate.mjs b/.claude/workflows/test-hil-validate.mjs
index db73095f6..e8be57cfc 100644
--- a/.claude/workflows/test-hil-validate.mjs
+++ b/.claude/workflows/test-hil-validate.mjs
@@ -4,7 +4,7 @@
// `board locked` out of a prose detail, folding rows, keeping a wedged flag alive -- produced
// a defect in each of four review rounds, including a test that asserted an invariant using
// the one input shape that could not break it. That logic now lives in
-// test/hil/helper/hil_summary.py, where the roster is, and arrives here as fields. What is
+// test/hil/helper/hil_report.py, where the roster is, and arrives here as fields. What is
// left is a lookup and a verdict, and this pins both.
//
// Run: node .claude/workflows/test-hil-validate.mjs
@@ -23,6 +23,11 @@ const cut = (start, end) => {
}
const body = cut('const byBoard =', 'const first = await runBoards')
+ cut('const summarize =', 'const { pass, wedged, locked } =')
+// more than one schema declares `required:`; pick the HIL one by its contents
+const HIL_REQUIRED = (src.match(/required: \[[^\]]*\]/g) || [])
+ .map((m) => m.replace('required: ', '').replace(/'/g, '"'))
+ .map((m) => JSON.parse(m))
+ .find((a) => a.includes('wedged')) || []
const { byBoard, summarize, wedgedFor } = new Function(`${body}; return { byBoard, summarize, wedgedFor }`)()
let failed = 0
@@ -65,5 +70,27 @@ check('wedged surfaces', summarize([R('a', false, false, true)], false).wedged,
check('a wedged board that passed still surfaces',
summarize([R('a', true, false, true)], false).wedged, ['a'])
+// A run-level caveat outranks per-row agreement: on the abandon and no-boards paths every
+// row can legitimately pass while hil_test.py exits non-zero. Row agreement alone published
+// those runs green.
+check('all rows pass and no caveat is a pass',
+ summarize([R('a', true), R('b', true)], false, '').pass, true)
+check('an abandoned run is not a pass',
+ summarize([R('a', true)], false,
+ '**HIL run abandoned: the worker pool would not shut down.** x').pass, false)
+check('an aborted run is not a pass',
+ summarize([R('a', true)], false, '**HIL run aborted: a worker raised RuntimeError**').pass,
+ false)
+check('a no-boards run is not a pass',
+ summarize([R('a', true)], false, '**HIL run selected no boards.** filters emptied').pass,
+ false)
+check('a rig-health note is NOT a caveat and does not fail the run',
+ summarize([R('a', true)], false, '> **Rig note.** 2 process(es) in D state').pass, true)
+check('a retry that abandoned sinks the run even with all rows passing',
+ summarize([R('a', true)], false,
+ '**HIL run abandoned: the worker pool would not shut down.** retry').pass, false)
+check('an omitted caveat cannot silently disable the gate (schema requires it)',
+ HIL_REQUIRED.includes('caveat'), true)
+
console.log(failed ? `\n${failed} FAILED` : '\nall checks passed')
process.exit(failed ? 1 : 0)
diff --git a/.claude/workflows/validate.js b/.claude/workflows/validate.js
index 522548ee6..dadf8a77f 100644
--- a/.claude/workflows/validate.js
+++ b/.claude/workflows/validate.js
@@ -1,11 +1,11 @@
export const meta = {
name: 'validate',
- description: 'Pre-PR software validation: unit tests + per-board build sweeps + code-size compare + PVS, in parallel, joined into one verdict',
+ description: 'Pre-PR software validation: unit tests + per-board build sweeps + code-size compare + PVS + diff reviews (claude + codex), in parallel, joined into one verdict',
whenToUse: 'Before opening or updating a PR, after any non-trivial change',
- phases: [{ title: 'Validate', detail: 'unit + builds + size + pvs in parallel' }],
+ phases: [{ title: 'Validate', detail: 'unit + builds + size + pvs + reviews in parallel' }],
}
-// args: { boards: string[], examples?: string, base?: string, skip?: ('unit'|'size'|'pvs')[] }
+// args: { boards: string[], examples?: string, base?: string, skip?: ('unit'|'size'|'pvs'|'review'|'codex')[] }
if (typeof args === 'string') { try { args = JSON.parse(args) } catch { /* not JSON: shape check below reports it */ } }
if (!args || !Array.isArray(args.boards) || args.boards.length === 0) {
throw new Error('args must be { boards: string[], examples?, base?, skip? }')
@@ -56,6 +56,26 @@ const PVS = {
},
}
+const REVIEW = {
+ type: 'object', additionalProperties: false,
+ required: ['pass', 'findings', 'detail'],
+ properties: {
+ pass: { type: 'boolean' },
+ findings: {
+ type: 'array',
+ items: {
+ type: 'object', additionalProperties: false,
+ required: ['file', 'line', 'severity', 'summary'],
+ properties: {
+ file: { type: 'string' }, line: { type: 'integer' },
+ severity: { type: 'string' }, summary: { type: 'string' },
+ },
+ },
+ },
+ detail: { type: 'string' },
+ },
+}
+
const thunks = []
if (!skip.includes('unit')) thunks.push(() =>
@@ -92,6 +112,40 @@ if (!skip.includes('pvs')) thunks.push(() =>
detail: r.pass ? r.detail : clip(`${r.detail} ${JSON.stringify(r.changedFindings)}`),
}))
+if (!skip.includes('review')) thunks.push(() =>
+ agent(
+ `Code-review this branch's diff vs ${base} (git diff ${base}...HEAD), coverage-first: walk every hunk, no spot checks. ` +
+ 'Find pass — candidate defects across all dimensions: correctness/logic, ISR & concurrency safety, ' +
+ 'memory/resource handling (bounds, leaks, no dynamic alloc), API contract & spec conformance, ' +
+ 'security of untrusted input parsing, behavior regressions; plus quality/simplification notes. ' +
+ 'Verify pass — adversarially check each candidate against the surrounding code: verdict CONFIRMED ' +
+ '(failing scenario constructed) or PLAUSIBLE (could not refute); report both, drop only refuted ones. ' +
+ 'Read-only: never apply fixes. severity = verdict plus category (e.g. "CONFIRMED correctness"). ' +
+ 'pass=false if any CONFIRMED correctness/safety/security bug survives; PLAUSIBLE and quality findings keep pass=true. ' +
+ 'detail = one-line review summary.',
+ { label: 'review', phase: 'Validate', model: 'opus', effort: 'high', schema: REVIEW },
+ ).then(r => r && {
+ stage: 'review',
+ // gate enforced here, not trusted from the agent: any CONFIRMED non-quality finding fails
+ pass: r.pass && !r.findings.some(f =>
+ /^confirmed/i.test(f.severity) && !/quality|simplification|style/i.test(f.severity)),
+ findings: r.findings, detail: r.detail,
+ }))
+
+if (!skip.includes('codex')) thunks.push(() =>
+ agent(
+ `Run a Codex review of this branch's diff vs ${base}: ` +
+ `codex review --base ${base} -c model="gpt-5.6-sol" -c model_reasoning_effort="high" ` +
+ '(Bash timeout 600000; run from the repo root). Parse its output into findings; severity = Codex\'s priority label. ' +
+ 'pass=false only if Codex reports a correctness bug (P0/P1); style-level items keep pass=true. ' +
+ 'detail = Codex\'s overall verdict line. If the codex CLI is missing or the run errors, pass=false with the error in detail.',
+ { label: 'codex', phase: 'Validate', model: 'haiku', schema: REVIEW },
+ ).then(r => r && {
+ stage: 'codex',
+ pass: r.pass && !r.findings.some(f => /\bP[01]\b/i.test(f.severity)),
+ findings: r.findings, detail: r.detail,
+ }))
+
const results = (await parallel(thunks)).filter(Boolean)
const dead = thunks.length - results.length
if (dead > 0) log(`${dead} stage agent(s) died — counted as failures`)
diff --git a/.github/scripts/ci_set_matrix.py b/.github/scripts/ci_set_matrix.py
index 79f466893..409e6dbc1 100755
--- a/.github/scripts/ci_set_matrix.py
+++ b/.github/scripts/ci_set_matrix.py
@@ -131,7 +131,13 @@ def set_matrix_json(select=None):
# a family this file does not list builds on no toolchain, so it contributes no
# leg. hw/bsp holds several CI has never built (efm32, py32f0, same7x, ...) plus
# espressif, whose boards hil-build-esp builds by name.
- unbuilt = sorted(f for f in sel_fams if f not in family_list)
+ # espressif is not a gap: its examples need the ESP-IDF environment
+ # (CLAUDE.md: `. "$IDF_PATH/export.sh"` before any build), which the cmake legs
+ # do not have - that is why it is commented out of family_list above. Its
+ # coverage comes from hil-build-esp, which builds those boards BY NAME in an IDF
+ # container, so an espressif-only PR is already validated and falling open to the
+ # full matrix would add 74 legs, none of which can compile espressif.
+ unbuilt = sorted(f for f in sel_fams if f not in family_list and f != 'espressif')
if unbuilt and not any(matrix.values()):
# NONE of the selected families is buildable here, so every leg would skip
# and the PR would go green from a build job that ran no compiler. That is
diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml
index 39a4e7afd..c26fe5cf8 100644
--- a/.github/workflows/build.yml
+++ b/.github/workflows/build.yml
@@ -68,14 +68,9 @@ jobs:
with:
fetch-depth: 0
- # The `ci-full` PR label turns the scoping off for one PR: no selection file is
- # written, so both matrices and every rig job fall back to the unscoped behaviour.
- # An escape hatch is the point - a selector bug under-selects SILENTLY, and without
- # a label the only routes back to a full matrix are accidental (touch an
- # unclassified path, or break the selector badly enough that it falls open).
- name: CI selection (PR only)
id: hil-select
- if: github.event_name == 'pull_request' && !contains(github.event.pull_request.labels.*.name, 'ci-full')
+ if: github.event_name == 'pull_request'
env:
BASE_REF: ${{ github.base_ref }}
run: |
@@ -179,29 +174,43 @@ jobs:
# treats false like null, so .build.full is compared explicitly.
EXAMPLE_MAP='{}'
BUILD_FILTERED='false'
- FAM_REGEX=''
+ FAMILY_REGEX=''
if [ -n "$BUILD_SELECT_FILE" ]; then
EXAMPLE_MAP=$(jq -c '.build.family_examples // {}' "$BUILD_SELECT_FILE") || EXAMPLE_MAP='{}'
BUILD_FILTERED=$(jq -r 'if (.build? | type) == "object" and .build.full == false then "true" else "false" end' "$BUILD_SELECT_FILE") || BUILD_FILTERED='false'
if [ "$BUILD_FILTERED" = "true" ]; then
- FAM_REGEX=$(jq -r '.build.families | join("|")' "$BUILD_SELECT_FILE") || FAM_REGEX=''
+ FAMILY_COUNT=$(jq -r '.build.families | length' "$BUILD_SELECT_FILE") || FAMILY_COUNT=0
+ FAMILY_REGEX=$(jq -r '.build.families | join("|")' "$BUILD_SELECT_FILE") || FAMILY_REGEX=''
# family names come from hw/bsp dir names, which rule 6 reads straight out
# of the PR's diff path - and this is interpolated raw into a
# `name_is_regexp` artifact pattern, so a regex metacharacter there would
# silently match another family's baseline
- case "$FAM_REGEX" in
+ FAMILY_REJECTED=0
+ case "$FAMILY_REGEX" in
*[!-A-Za-z0-9_\|]*)
echo "::warning::unexpected characters in the family list - dropping the scoping"
- FAM_REGEX='' ;;
+ FAMILY_REGEX=''; FAMILY_REJECTED=1 ;;
esac
- if [ -z "$FAM_REGEX" ]; then
- # all three drop together, as CircleCI's fall-open does. Resetting only
+ # An EMPTY families list and a REJECTED one both leave FAMILY_REGEX empty and
+ # mean opposite things, so branch on which happened. Testing `-z` alone sent
+ # every nothing-selected PR down the fall-open path: a docs/.gitignore diff
+ # (#3842) and a test/hil-only diff (#3840) each rebuilt all 74 cmake legs
+ # after the selector had correctly chosen none.
+ if [ "$FAMILY_REJECTED" = "1" ]; then
+ # unusable: fall open, and all three drop together. Resetting only
# build_filtered leaves the build scoped while code-metrics takes the
# UNSCOPED branch, diffing a 1-family run against the full averaged
# baseline and publishing that as the PR's code-size impact.
BUILD_FILTERED='false'
EXAMPLE_MAP='{}'
MATRIX_JSON=$(python .github/scripts/ci_set_matrix.py)
+ elif [ "$FAMILY_COUNT" = "0" ]; then
+ # legitimate nothing-selected. MATRIX_JSON already holds the all-empty
+ # matrix ci_set_matrix produced from this selection - keep it, so every
+ # leg skips. Nothing is built, so there is nothing to compare a baseline
+ # against: build_filtered goes false to keep code-metrics off the scoped
+ # path, and EXAMPLE_MAP stays '{}' (family_examples is empty anyway).
+ BUILD_FILTERED='false'
fi
fi
fi
@@ -210,7 +219,7 @@ jobs:
echo "matrix=$MATRIX_JSON" >> $GITHUB_OUTPUT
echo "example_map=$EXAMPLE_MAP" >> $GITHUB_OUTPUT
echo "build_filtered=$BUILD_FILTERED" >> $GITHUB_OUTPUT
- echo "build_families_regex=$FAM_REGEX" >> $GITHUB_OUTPUT
+ echo "build_families_regex=$FAMILY_REGEX" >> $GITHUB_OUTPUT
# HIL matrix (merged from tinyusb + hifiphile configs), scoped on PRs.
# Scoping is best-effort too: fall back to the unscoped (full) matrix.
diff --git a/.github/workflows/build_util.yml b/.github/workflows/build_util.yml
index 52999616d..407ed1e71 100644
--- a/.github/workflows/build_util.yml
+++ b/.github/workflows/build_util.yml
@@ -126,16 +126,11 @@ jobs:
MEMBROWSE_API_KEY: ${{ secrets.MEMBROWSE_API_KEY }}
run: |
# if code-changed is false --> there is no elf -> membrowse target upload with --identical flag
- # $EX_ARGS is passed for the BOARD it picks, not to scope the targets:
- # --one-first now chooses a board that can build the -e set (tools/build.py),
- # so omitting it here would configure a DIFFERENT, empty build dir and upload
- # --identical for a board that was never compiled. The target list is not
- # scoped by it - `examples-membrowse-upload` is not `all`, so it passes
- # through as the aggregate, which has no DEPENDS (hw/bsp/family_support.cmake):
- # it rebuilds nothing and still records every example, --identical for the
- # ones without an elf.
+ # deliberately unscoped by $EX_ARGS: keeps the size history on a stable board
+ # per family, at the cost of an --identical-only upload where that board is not
+ # the one the Build step picked (test_ci_metrics pins which families those are)
BUILD_PY_ARGS="-s ${{ inputs.build-system }} ${{ steps.setup-toolchain.outputs.build_option }} ${{ inputs.build-options }}"
- python tools/build.py $BUILD_PY_ARGS --target examples-membrowse-upload -j 1 ${{ matrix.arg }} $EX_ARGS
+ python tools/build.py $BUILD_PY_ARGS --target examples-membrowse-upload -j 1 ${{ matrix.arg }}
shell: bash
- name: Upload Artifacts for Metrics
diff --git a/.gitignore b/.gitignore
index f6c8702b6..d74f12459 100644
--- a/.gitignore
+++ b/.gitignore
@@ -62,6 +62,7 @@ README_processed.rst
docs/examples/
.worktrees
.claude/worktrees/
+.claude/skills/update-sponsor/state.json
cmake-metrics/
# Directories fetched by tools/get_deps.py - not to be committed
lib/CMSIS_5/
diff --git a/README.rst b/README.rst
index dea352532..7fa8d70f1 100644
--- a/README.rst
+++ b/README.rst
@@ -60,7 +60,7 @@ Supporters (Word)
.. WORD-SUPPORTERS-START
-*No supporters yet — be the first!*
+cee\*\*\*\*
.. WORD-SUPPORTERS-END
@@ -69,7 +69,7 @@ Thanks (Byte)
.. BYTE-THANKS-START
-*No names listed yet — be the first!*
+`@8086net <https://github.com/8086net>`__, `@GCRev <https://github.com/GCRev>`__
.. BYTE-THANKS-END
diff --git a/docs/reference/hil_boards.md b/docs/reference/hil_boards.md
index e8f364646..678f7f0ed 100644
--- a/docs/reference/hil_boards.md
+++ b/docs/reference/hil_boards.md
@@ -12,7 +12,7 @@
| espressif_s3_devkitm | device, host | esptool | espressif_s3_devkitm, espressif_s3_devkitm-DMA | Use TS3USB30 mux to test both device and host |
| feather_nrf52840_express | device | jlink | | |
| max32666fthr | device | openocd | | |
-| metro_m4_express | device, dual | jlink | | pl23x; audio_test_freertos skipped: samd51 iso-IN capture fails (arecord EIO) |
+| metro_m4_express | device, dual | jlink | metro_m4_express | pl23x; audio_test_freertos skipped: samd51 iso-IN capture fails (arecord EIO) |
| lpcxpresso11u37 | device | jlink | | |
| lpcxpresso55s28 | device | jlink | | |
| ra4m1_ek | device | jlink | | |
diff --git a/docs/superpowers/followup/pr3803-hil-blindness-reporting.md b/docs/superpowers/followup/pr3803-hil-blindness-reporting.md
index 69ff939b0..374ee62c7 100644
--- a/docs/superpowers/followup/pr3803-hil-blindness-reporting.md
+++ b/docs/superpowers/followup/pr3803-hil-blindness-reporting.md
@@ -158,7 +158,8 @@ def _board_result_on_error(name, exc):
"""A row for a board that died by exception. err_count 1, no per-test detail, but the
blindness and stray fields are still accurate -- they explain the failure more often
than the exception text does."""
- rows = [(name, {BOUNDARY_CELL: f'{REPORT_CELL["fail"]} {type(exc).__name__}'}, None)]
+ rows = [(name, {hil_report.BOUNDARY_CELL:
+ f'{hil_report.REPORT_CELL["fail"]} {type(exc).__name__}'}, None)]
return _board_result(name, 1, [], rows, 0.0, True)
```
diff --git a/docs/superpowers/followup/pr3836-report-single-source.md b/docs/superpowers/followup/pr3836-report-single-source.md
deleted file mode 100644
index f4ccc77c2..000000000
--- a/docs/superpowers/followup/pr3836-report-single-source.md
+++ /dev/null
@@ -1,470 +0,0 @@
-# One Source of Truth for the HIL Report Implementation Plan
-
-> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
-
-**Goal:** Make `hil_report.md` a rendering of `hil_report.json` rather than a second, independently written artifact, so no run can produce a table whose contents are not in the JSON.
-
-**Architecture:** `hil_report.json` gains the two fields the markdown carries but the JSON does not (`scope`, and a `caveat` for text prepended after the fact). A single `render_report(doc) -> str` turns that document into the markdown, and every writer — the normal path, the pool-guard fallback, the no-boards exit, and `_abandon_exit` — goes through `write_report(report_dir, doc)`, which writes both files from the same dict. `_abandon_exit` stops doing a text-prepend on a file it did not write and instead sets `doc['caveat']`.
-
-**Tech Stack:** Python 3.13 stdlib only (`json`, `pathlib`); existing unit suites under `test/hil/test/` run with plain `unittest`.
-
-**Spec:** none — this is a follow-up split out of the `claude/hil-doc-audit` branch. The evidence it argues from is inline below.
-
-**Origin:** split out of PR #3836 (the HIL one-run rework + `.claude` instruction audit). Delete this file when its own PR lands.
-
-## Global Constraints
-
-- **No behaviour change to the containment paths' ordering or exit codes.** `_abandon_exit` runs while the interpreter is being torn down; its own comments record that anything raising between the pool's `finally` and `os._exit` hangs the process in multiprocessing's unbounded `join()` (reproduced at rc=124/25s with SIGTERM-ignoring workers). Serialisation added there must stay inside the existing `try`/`except` and must never raise past it.
-- **The markdown stays the human artifact.** `.github/workflows/build.yml:487` uploads `hil_report.md`, `test/hil/hil_ci.sh:293` copies only it back, and `.claude/skills/hil/SKILL.md` tells the operator to paste that table verbatim. It becomes generated output, not a dropped file.
-- **Banner outranks the scope note outranks the table.** Preserve the existing order (`hil_test.py:2153-2161`): the caveat is outermost because that is where `hil/SKILL.md` tells an agent to look.
-- **`--accumulate` merges from the JSON** (`hil_test.py:2102-2115`), including carrying the prior banner forward. Adding fields must not break that merge for a sidecar written by an older version.
-- Run `python3 -m unittest discover -s test/hil/test` (115 tests, ~78 s) before each commit; `pre-commit run --files <changed>` before pushing.
-
-## Why this is worth doing
-
-Four writers produce `hil_report.md`, and three of them write no JSON at all:
-
-| Writer | JSON? | Line |
-|---|---|---|
-| `accumulate_report` — the normal path | yes | `hil_test.py:2149`, `:2162` |
-| `**HIL run selected no boards.**` | **no** | `hil_test.py:2317` |
-| pool-guard fallback → `hil_health.write_timeout_report(...)` | **no** | `hil_test.py:2469`, `hil_health.py:346` |
-| `_abandon_exit` — prepends to whatever `.md` exists | **no** | `hil_test.py:2603` |
-
-Those three are exactly the paths where the run died, so they are the cases where the artifact matters most and where a JSON consumer sees nothing. `test/hil/helper/hil_summary.py` (added on the origin branch) reads the JSON to build the per-board verdicts an agent hands back — on any of those three paths it finds no file and reports "no report row for this board" for the whole fleet, while a human reading the markdown sees the real story.
-
-Separately, `scope` exists only in the markdown (`hil_test.py:2154`, from `accumulate_report`'s `scope: str = ''` parameter at `:2092`). A PR-scoped three-board table and a full-fleet run that lost 24 boards are indistinguishable in the JSON.
-
----
-
-### Task 1: Put `scope` in the JSON
-
-**Files:**
-- Modify: `test/hil/hil_test.py:2092-2163` (`accumulate_report`)
-- Test: `test/hil/test/test_hil_bounded.py` (new class beside `CaveatSurvivesAccumulate`)
-
-**Interfaces:**
-- Produces: `hil_report.json` gains a top-level `"scope": str` (empty string when unscoped). Existing keys `rows` and `banner` are unchanged.
-
-- [ ] **Step 1: Write the failing test**
-
-```python
-class ScopeSurvivesInTheJson(unittest.TestCase):
- """A scoped run's small table is indistinguishable from a full run that lost boards.
- The markdown says so; the JSON did not, so any JSON consumer could not tell."""
-
- def _rows(self, board, cell):
- return [(board, 0, 0, [(board, {cell: 'OK'}, '1s')], 0)]
-
- def test_scope_is_recorded_in_the_sidecar(self):
- import json
- td = TemporaryDirectory()
- self.addCleanup(td.cleanup)
- rd = Path(td.name)
- hil_test.accumulate_report(self._rows('boardA', 'cdc_msc'), rd, True,
- '-b boardA', '')
- doc = json.loads((rd / 'hil_report.json').read_text())
- self.assertEqual(doc['scope'], '-b boardA')
-
- def test_an_unscoped_run_records_an_empty_scope(self):
- import json
- td = TemporaryDirectory()
- self.addCleanup(td.cleanup)
- rd = Path(td.name)
- hil_test.accumulate_report(self._rows('boardA', 'cdc_msc'), rd, True, '', '')
- self.assertEqual(json.loads((rd / 'hil_report.json').read_text())['scope'], '')
-```
-
-- [ ] **Step 2: Run it to verify it fails**
-
-Run: `python3 test/hil/test/test_hil_bounded.py ScopeSurvivesInTheJson`
-Expected: FAIL — `KeyError: 'scope'`
-
-- [ ] **Step 3: Add the field**
-
-In `accumulate_report`, change the `jpath.write_text(...)` call at `hil_test.py:2149`:
-
-```python
- jpath.write_text(json.dumps({'rows': [{'board': k, 'cells': c, 'duration': d}
- for k, (c, d) in acc.items()],
- 'banner': banner,
- 'scope': scope}, indent=2) + '\n')
-```
-
-- [ ] **Step 4: Run the tests**
-
-Run: `python3 test/hil/test/test_hil_bounded.py ScopeSurvivesInTheJson` → PASS
-Run: `python3 -m unittest discover -s test/hil/test` → 117 tests OK (the merge at `:2102` reads only `rows` and `banner`, so an older sidecar without `scope` still loads).
-
-- [ ] **Step 5: Commit**
-
-```bash
-git add test/hil/hil_test.py test/hil/test/test_hil_bounded.py
-git commit -m "hil_test: record the run's scope in hil_report.json
-
-The markdown says a scoped table is scoped; the JSON did not, so a consumer
-could not tell a three-board PR run from a full run that lost 24 boards."
-```
-
----
-
-### Task 2: Render the markdown from the document
-
-**Files:**
-- Modify: `test/hil/hil_test.py:1921` (`render_matrix`), `:2149-2163` (`accumulate_report`'s tail)
-- Test: `test/hil/test/test_hil_bounded.py`
-
-**Interfaces:**
-- Consumes: the `scope` key from Task 1.
-- Produces: `render_report(doc: dict) -> str`, where `doc` is `{'rows': [{'board','cells','duration'}], 'banner': str, 'scope': str, 'caveat': str}`. `caveat` is optional and empty by default (Task 4 sets it). Order is caveat, banner, scope note, table.
-
-- [ ] **Step 1: Write the failing test**
-
-```python
-class RenderReportIsPureFunctionOfTheDocument(unittest.TestCase):
- def _doc(self, **kw):
- d = {'rows': [{'board': 'boardA', 'cells': {'cdc_msc': 'pass'}, 'duration': '1s'}],
- 'banner': '', 'scope': '', 'caveat': ''}
- d.update(kw)
- return d
-
- def test_table_comes_from_rows(self):
- md = hil_test.render_report(self._doc())
- self.assertIn('boardA', md)
- self.assertIn('cdc_msc', md)
-
- def test_scope_note_appears_above_the_table(self):
- md = hil_test.render_report(self._doc(scope='-b boardA'))
- self.assertLess(md.index('Scoped run'), md.index('boardA'))
-
- def test_banner_outranks_the_scope_note(self):
- md = hil_test.render_report(self._doc(scope='-b boardA',
- banner='> **Rig dirty.** x\n'))
- self.assertLess(md.index('Rig dirty'), md.index('Scoped run'))
-
- def test_caveat_is_outermost(self):
- md = hil_test.render_report(self._doc(banner='> **Rig dirty.** x\n',
- caveat='**HIL run abandoned.**\n'))
- self.assertLess(md.index('abandoned'), md.index('Rig dirty'))
-
- def test_a_document_with_no_rows_still_renders(self):
- md = hil_test.render_report(self._doc(rows=[]))
- self.assertIn('No tests were run.', md)
-```
-
-- [ ] **Step 2: Run it to verify it fails**
-
-Run: `python3 test/hil/test/test_hil_bounded.py RenderReportIsPureFunctionOfTheDocument`
-Expected: FAIL — `AttributeError: module 'hil_test' has no attribute 'render_report'`
-
-- [ ] **Step 3: Add `render_report` and route `accumulate_report` through it**
-
-Add beside `render_matrix` (after `hil_test.py:1919`):
-
-```python
-def render_report(doc: dict) -> str:
- """The markdown IS a rendering of the sidecar. Every writer goes through here, so a
- table can never contain something the JSON does not."""
- md = render_matrix([(r['board'], r['cells'], r.get('duration'))
- for r in doc.get('rows', [])])
- if doc.get('scope'):
- # a scoped run's small table is otherwise indistinguishable from a full one, and
- # it replaces the previous full table in the sticky PR comment
- md = f'_Scoped run: {doc["scope"]}. Boards/tests not listed were not run._\n\n' + md
- # banner, then caveat: a rig-health caveat outranks the table AND the scope note, and an
- # abandon notice outranks even that -- the top of the report is where hil/SKILL.md tells
- # the agent to look
- if doc.get('banner'):
- md = doc['banner'] + '\n' + md
- if doc.get('caveat'):
- md = doc['caveat'] + '\n' + md
- return md
-```
-
-Then replace `accumulate_report`'s tail (`hil_test.py:2153-2163`) with:
-
-```python
- doc = {'rows': [{'board': k, 'cells': c, 'duration': d} for k, (c, d) in acc.items()],
- 'banner': banner, 'scope': scope, 'caveat': ''}
- jpath.write_text(json.dumps(doc, indent=2) + '\n')
- md = render_report(doc)
- (report_dir / REPORT_MD).write_text(md + '\n', encoding='utf-8')
- return md
-```
-
-- [ ] **Step 4: Run the tests**
-
-Run: `python3 -m unittest discover -s test/hil/test`
-Expected: 122 OK. `CaveatSurvivesAccumulate` must still pass — it asserts the banner survives a rerun, which is now the `banner` key round-tripping through the document.
-
-- [ ] **Step 5: Commit**
-
-```bash
-git add test/hil/hil_test.py test/hil/test/test_hil_bounded.py
-git commit -m "hil_test: render the markdown from the report document
-
-One function turns the sidecar into the table, so the markdown cannot carry
-anything the JSON lacks. Ordering (caveat > banner > scope > table) is pinned
-by tests rather than by the order of three string concatenations."
-```
-
----
-
-### Task 3: Give the two early-exit paths a document
-
-**Files:**
-- Modify: `test/hil/hil_test.py:2313-2320` (no-boards exit), `test/hil/helper/hil_health.py:346` (`write_timeout_report`)
-- Test: `test/hil/test/test_hil_health.py` (beside `WriteTimeoutReport`), `test/hil/test/test_hil_bounded.py`
-
-**Interfaces:**
-- Consumes: `render_report(doc)` from Task 2.
-- Produces: `write_report(report_dir: Path, doc: dict) -> None`, which writes `hil_report.json` and `hil_report.md` from one dict. Both early-exit paths call it.
-
-- [ ] **Step 1: Write the failing test**
-
-```python
-class EveryExitPathLeavesBothArtifacts(unittest.TestCase):
- """hil_summary.py builds an agent's verdicts from the JSON. A path that writes only
- markdown reports the whole fleet as 'no report row' while a human sees the real story."""
-
- def test_the_no_boards_exit_writes_json_too(self):
- import json
- td = TemporaryDirectory()
- self.addCleanup(td.cleanup)
- rd = Path(td.name)
- hil_test.write_report(rd, {'rows': [], 'banner': '', 'scope': '',
- 'caveat': '**HIL run selected no boards.** why\n'})
- self.assertIn('selected no boards', (rd / 'hil_report.md').read_text())
- doc = json.loads((rd / 'hil_report.json').read_text())
- self.assertEqual(doc['rows'], [])
- self.assertIn('selected no boards', doc['caveat'])
-```
-
-and, in `test_hil_health.py`:
-
-```python
- def test_timeout_report_writes_the_sidecar(self):
- import json
- td = TemporaryDirectory()
- self.addCleanup(td.cleanup)
- rd = Path(td.name)
- hil_health.write_timeout_report(rd, [{'name': 'boardA'}], 3600, 'hil_report.md')
- self.assertTrue((rd / 'hil_report.json').is_file())
- self.assertIn('boardA', (rd / 'hil_report.json').read_text())
-```
-
-- [ ] **Step 2: Run them to verify they fail**
-
-Run: `python3 test/hil/test/test_hil_bounded.py EveryExitPathLeavesBothArtifacts`
-Expected: FAIL — `AttributeError: module 'hil_test' has no attribute 'write_report'`
-Run: `python3 test/hil/test/test_hil_health.py WriteTimeoutReport`
-Expected: FAIL — `hil_report.json` is not a file
-
-- [ ] **Step 3: Add `write_report` and use it in both paths**
-
-Beside `render_report`:
-
-```python
-def write_report(report_dir: Path, doc: dict) -> None:
- """Write both artifacts from one document. Best-effort by design: every caller is on a
- failure path where an OSError must not replace the failure being reported."""
- try:
- report_dir.mkdir(parents=True, exist_ok=True)
- (report_dir / REPORT_JSON).write_text(json.dumps(doc, indent=2) + '\n')
- (report_dir / REPORT_MD).write_text(render_report(doc) + '\n', encoding='utf-8')
- except OSError:
- pass
-```
-
-Replace the no-boards block at `hil_test.py:2315-2320` with:
-
-```python
- rd = Path(os.environ.get('HIL_REPORT_DIR', '.'))
- write_report(rd, {'rows': [], 'banner': '', 'scope': '',
- 'caveat': f'**HIL run selected no boards.** {msg}\n'})
-```
-
-In `hil_health.write_timeout_report`, after the markdown is composed, write the sidecar next to it with a row per stuck board:
-
-```python
- json_path = report_dir / 'hil_report.json'
- json_path.write_text(json.dumps(
- {'rows': [{'board': b['name'], 'cells': {'pool-timeout': 'fail'},
- 'duration': None} for b in boards],
- 'banner': banner, 'scope': '', 'caveat': prefix}, indent=2) + '\n')
-```
-
-Keep it inside the function's existing broad `try` — a roster entry without `name` must not escape, which is what that handler exists to prevent.
-
-- [ ] **Step 4: Run the tests**
-
-Run: `python3 -m unittest discover -s test/hil/test`
-Expected: 124 OK.
-
-- [ ] **Step 5: Commit**
-
-```bash
-git add test/hil/hil_test.py test/hil/helper/hil_health.py test/hil/test/
-git commit -m "hil_test, hil_health: write the sidecar on the early-exit paths too
-
-The no-boards exit and the pool-guard fallback wrote markdown only, so a JSON
-consumer saw nothing on exactly the runs that failed. hil_summary.py reported
-the whole fleet as 'no report row' while the markdown told the real story."
-```
-
----
-
-### Task 4: Make `_abandon_exit` set a field instead of prepending text
-
-**Files:**
-- Modify: `test/hil/hil_test.py` (`_abandon_exit`, the `if report is not None:` block near `:2622`), and its call site at `:2603`
-- Test: `test/hil/test/test_hil_bounded.py`
-
-**Interfaces:**
-- Consumes: `write_report`/`render_report` from Tasks 2–3.
-- Produces: `_abandon_exit(pool, mgr, abandoned, err_count, report_dir: Path | None = None)` — the parameter becomes the **directory**, not the markdown path.
-
-- [ ] **Step 1: Write the failing test**
-
-```python
-class AbandonNoticeLandsInBothArtifacts(unittest.TestCase):
- def test_abandon_sets_the_caveat_not_just_the_markdown(self):
- import json
- td = TemporaryDirectory()
- self.addCleanup(td.cleanup)
- rd = Path(td.name)
- hil_test.accumulate_report(
- [('boardA', 0, 0, [('boardA', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
- hil_test.mark_report_abandoned(rd, 'the worker pool would not shut down.')
- doc = json.loads((rd / 'hil_report.json').read_text())
- self.assertIn('abandoned', doc['caveat'])
- self.assertEqual(len(doc['rows']), 1, 'the finished board must survive')
- md = (rd / 'hil_report.md').read_text()
- self.assertLess(md.index('abandoned'), md.index('boardA'))
-
- def test_marking_a_missing_report_is_a_no_op(self):
- td = TemporaryDirectory()
- self.addCleanup(td.cleanup)
- hil_test.mark_report_abandoned(Path(td.name), 'x') # must not raise
-```
-
-- [ ] **Step 2: Run it to verify it fails**
-
-Run: `python3 test/hil/test/test_hil_bounded.py AbandonNoticeLandsInBothArtifacts`
-Expected: FAIL — `AttributeError: module 'hil_test' has no attribute 'mark_report_abandoned'`
-
-- [ ] **Step 3: Implement it**
-
-```python
-def mark_report_abandoned(report_dir: Path, why: str) -> None:
- """Stamp an existing report as abandoned, in BOTH artifacts.
-
- Best-effort and silent: this runs while the interpreter is being torn down, and an
- exception here hangs the process in multiprocessing's unbounded join()."""
- try:
- jpath = report_dir / REPORT_JSON
- doc = json.loads(jpath.read_text()) if jpath.is_file() else None
- if doc is None:
- return
- doc['caveat'] = (f'**HIL run abandoned: {why}** The table below is this run\'s '
- f'partial result.\n')
- write_report(report_dir, doc)
- except (OSError, ValueError, TypeError):
- pass
-```
-
-Then in `_abandon_exit`, replace the read-modify-write of the markdown with `mark_report_abandoned(report, ...)` and change the call site at `:2603` from `report_dir / REPORT_MD` to `report_dir`.
-
-- [ ] **Step 4: Run the tests**
-
-Run: `python3 -m unittest discover -s test/hil/test`
-Expected: 126 OK.
-
-- [ ] **Step 5: Commit**
-
-```bash
-git add test/hil/hil_test.py test/hil/test/test_hil_bounded.py
-git commit -m "hil_test: stamp abandonment into the document, not onto the markdown
-
-_abandon_exit did a text prepend on a file it had not written, so the caveat
-never reached the JSON and an agent reading the sidecar saw a clean partial
-report under a red job. Still best-effort and still silent: it runs while the
-interpreter is being torn down."
-```
-
----
-
-### Task 5: Prove the two artifacts cannot disagree
-
-**Files:**
-- Test: `test/hil/test/test_hil_bounded.py`
-
-- [ ] **Step 1: Write the test**
-
-```python
-class MarkdownIsAlwaysARenderingOfTheJson(unittest.TestCase):
- """The property this whole change buys: whatever wrote the report, re-rendering the
- sidecar reproduces the markdown byte for byte."""
-
- def _check(self, rd):
- import json
- doc = json.loads((rd / 'hil_report.json').read_text())
- self.assertEqual((rd / 'hil_report.md').read_text(),
- hil_test.render_report(doc) + '\n')
-
- def test_normal_path(self):
- td = TemporaryDirectory(); self.addCleanup(td.cleanup); rd = Path(td.name)
- hil_test.accumulate_report(
- [('boardA', 0, 0, [('boardA', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True,
- '-b boardA', '> **Rig note.** x\n')
- self._check(rd)
-
- def test_after_an_accumulate_rerun(self):
- td = TemporaryDirectory(); self.addCleanup(td.cleanup); rd = Path(td.name)
- hil_test.accumulate_report(
- [('boardA', 0, 0, [('boardA', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
- hil_test.accumulate_report(
- [('boardB', 0, 0, [('boardB', {'cdc_msc': 'OK'}, '1s')], 0)], rd, False, '', '')
- self._check(rd)
-
- def test_after_abandonment(self):
- td = TemporaryDirectory(); self.addCleanup(td.cleanup); rd = Path(td.name)
- hil_test.accumulate_report(
- [('boardA', 0, 0, [('boardA', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
- hil_test.mark_report_abandoned(rd, 'the worker pool would not shut down.')
- self._check(rd)
-
- def test_no_boards_exit(self):
- td = TemporaryDirectory(); self.addCleanup(td.cleanup); rd = Path(td.name)
- hil_test.write_report(rd, {'rows': [], 'banner': '', 'scope': '',
- 'caveat': '**HIL run selected no boards.** why\n'})
- self._check(rd)
-```
-
-- [ ] **Step 2: Run it**
-
-Run: `python3 test/hil/test/test_hil_bounded.py MarkdownIsAlwaysARenderingOfTheJson`
-Expected: PASS on all four. A failure here means a writer still bypasses `render_report`.
-
-- [ ] **Step 3: Full gate and commit**
-
-```bash
-python3 -m unittest discover -s test/hil/test # 130 OK
-pre-commit run --files test/hil/hil_test.py test/hil/helper/hil_health.py \
- test/hil/test/test_hil_bounded.py test/hil/test/test_hil_health.py
-git add test/hil/test/test_hil_bounded.py
-git commit -m "test/hil: pin that the markdown is always a rendering of the sidecar
-
-Four writers, one renderer. This is the invariant the change exists to create,
-so it is asserted directly rather than inferred from the writers."
-```
-
----
-
-## Out of scope
-
-Deliberately not included, each its own follow-up:
-
-- **The flat `HIL_POOL_TIMEOUT`.** `hil_test.py:225` is a per-process 3600 s guard that does not scale with board count. It was per board when runs were serial; the origin branch made one run cover the fleet, so a 27-board run shares one budget. Real, and a scheduling change rather than a reporting one.
-- **`hil_ci.sh` accumulate in remote mode.** `hil_ci.sh:183` `rm -rf`s `REMOTE_DIR` every run and the copies are one-way, so a remote `--accumulate` retry has no merge base and its one-row report overwrites the local full-fleet one. Fixing that means uploading `hil_report.json` and `<config>.failed` before the run, or keeping `REMOTE_DIR` when `--accumulate` is present.
-- **Dropping `hil_report.md` entirely.** Not proposed. It is the PR artifact and what the `hil` skill tells operators to paste; this plan makes it generated, not redundant.
diff --git a/docs/superpowers/followup/pr3840-mret-board-result.md b/docs/superpowers/followup/pr3840-mret-board-result.md
new file mode 100644
index 000000000..7b8da7b9c
--- /dev/null
+++ b/docs/superpowers/followup/pr3840-mret-board-result.md
@@ -0,0 +1,90 @@
+# Give the HIL worker result a name
+
+**Origin:** split out of PR #3840 (making `hil_report.md` a rendering of `hil_report.json`).
+Delete this file when its own PR lands.
+
+## What is established
+
+`test_board()` returns a bare tuple that three producers build and fourteen call sites read
+positionally. It has grown 5 → 6 → 7 fields, and the code already works around its own
+shape:
+
+```python
+hil_test.py:1992 dirty = [(r[0], r[6]) for r in mret if len(r) > 6 and r[6]]
+hil_test.py:2014 blind = [r[0] for r in mret if len(r) > 5 and r[5]]
+hil_test.py:2386 for name, _, _, _, dur, *_ in mret:
+hil_report.py:306 for name, _, _, rows, *_ in mret:
+```
+
+Two facts make this worth closing rather than tolerating:
+
+- **The declared type is already wrong.** `hil_test.py:1711` says
+ `tuple[str, int, list[str], list, float]` — five fields — while the main return at `:1872`
+ yields seven (`+ sysfs_blind(), stray`).
+- **A wrong slot is a wrong verdict, not a crash.** Field 5 is `blind`, which decides whether
+ a board's red cells are reported as broken hardware or as "could not tell". Inserting a
+ field mid-tuple makes `r[5]` read the wrong slot and keep running.
+
+It has bitten once already: `test_hil_bounded.py`'s
+`test_both_row_widths_survive_the_report_writers` exists because the blindness flag widened
+the tuple to 6 while the pool-timeout path still synthesised 5-field rows, and *"a
+fixed-width unpack in either one raises INSIDE the containment path, which is where a raise
+costs every board's results."* That is why the unpacks end in `*_`.
+
+## What remains
+
+A `NamedTuple` with defaults. Verified to pickle across the pool boundary and to stay
+fully tuple-compatible — existing `r[0]`, `e[1]`, `for name, _, _, rows, *_` and `len(r)`
+all keep working, so it lands without touching the fourteen consumers:
+
+```python
+class BoardResult(NamedTuple):
+ """What one worker returns. Field ORDER is load-bearing: it is unpacked positionally
+ in a dozen places, and the pool-timeout path synthesises one by hand."""
+ name: str
+ err_count: int
+ failed_tests: list[str]
+ rows: list | None # None from the pool-timeout synthesis, never []
+ duration: float
+ blind: bool = False # defaults, so a synthesised result is full-width
+ stray: int = 0
+```
+
+Then a second, smaller step removes the coupling itself: `accumulate_report` takes
+`[(name, rows)]` pairs instead of `mret`, and `hil_test` does the extraction because it owns
+the shape. One line at each end; the subtle merge logic — stale lock clearing,
+`BOUNDARY_CELL`, `duration=None` preservation — is untouched.
+
+## Sizing
+
+| | Sites |
+|---|---|
+| Producers to convert | 4 (`hil_test.py:1724`, `:1872`, `:2283`, `:2327`) |
+| Arity guards deleted | 2 (`:1992`, `:2014`) |
+| Wrong annotation fixed | 1 (`:1711`) |
+| `hil_report`'s coupled line | 1 (`:306`) |
+| Positional consumers (optional migration) | 14 |
+| **Test fixtures building tuples by hand** | **34** |
+
+Production code is roughly ten changed lines. **The work is dominated by the test
+fixtures**, which is also the risk.
+
+## Do this first, or the refactor is unverifiable
+
+`test_hil_report.py` (27 sites) and `test_hil_bounded.py` (7) construct plain tuples by
+hand — `('boardA', 0, [], [], 1.0, True)`. A producer that forgot to switch to
+`BoardResult`, or a pickling regression, **passes the entire 310-test suite** and surfaces
+only on the rig. Convert the fixtures to build `BoardResult` as task 1, before touching any
+producer. This ordering is not optional.
+
+Second trap: `rows` is `None` on the pool-timeout path (`hil_test.py:2283`), never `[]`, and
+`accumulate_report` guards with `if rows and ...`. A well-meaning `rows: list = []` default
+silently changes that path. Pin it with a test before the conversion.
+
+## Why it was split out
+
+PR #3840 touches the report document. This touches `test_board`'s return and the containment
+paths, where a raise costs every board's results rather than one board's — a different blast
+radius, needing its own review and its own rig run. #3840 is twice-reviewed and dogfooded
+ten times on hardware; folding this in would reset that surface for a latent-trap cleanup
+that is not causing bugs today.
diff --git a/docs/superpowers/followup/pr3840-skill-md-no-boards-drift.md b/docs/superpowers/followup/pr3840-skill-md-no-boards-drift.md
new file mode 100644
index 000000000..a039a8c12
--- /dev/null
+++ b/docs/superpowers/followup/pr3840-skill-md-no-boards-drift.md
@@ -0,0 +1,38 @@
+# `SKILL.md` contradicts the code on no-boards tables
+
+**Origin:** split out of PR #3840, surfaced by its second review round. Delete this file
+when its own PR lands.
+
+`.claude/skills/hil/SKILL.md:150-151` tells the reading agent:
+
+> `**HIL run selected no boards.**` — the filters intersected to nothing, so there is **no
+> table at all**. Report that (and the filter shown), never `"pass": true`.
+
+That was true when the no-boards exit wrote a bare notice. It no longer is. An
+`--accumulate` no-boards run keeps the accumulated rows — deliberately, because wiping them
+destroyed real results — so the artifact now reads:
+
+```
+**HIL run selected no boards.** filters emptied
+
+**✅ 1 passed · ❌ 0 failed · ⚪ 0 skipped · blank not run**
+
+| Board | t | duration |
+...
+```
+
+The behaviour is correct; the documentation is wrong, and wrong in the direction that
+matters. An agent is told to expect no table, sees one, and has no rule for whether those
+rows are reportable. **They are not this run's** — they are a previous attempt's, carried
+forward.
+
+**What remains:** update that bullet to describe both cases — a fresh run has no table, an
+`--accumulate` run shows the previous attempt's rows under the notice and they must not be
+reported as this run's. Add a test asserting the fresh case renders no matrix, so the two
+halves cannot drift again.
+
+## Why it was split out
+
+PR #3840 fixed the findings that changed a verdict. This is a documentation drift: the
+behaviour is correct and the doc describing it is not, so it is better reviewed on its own
+than appended to a branch already carrying a module consolidation.
diff --git a/docs/superpowers/followup/pr3840-write-report-atomicity.md b/docs/superpowers/followup/pr3840-write-report-atomicity.md
new file mode 100644
index 000000000..2094207bb
--- /dev/null
+++ b/docs/superpowers/followup/pr3840-write-report-atomicity.md
@@ -0,0 +1,30 @@
+# `write_report` commits the two artifacts non-atomically
+
+**Origin:** split out of PR #3840, surfaced by its second review round. Delete this file
+when its own PR lands.
+
+```python
+md = render_report(doc) + '\n'
+report_dir.mkdir(parents=True, exist_ok=True)
+(report_dir / REPORT_JSON).write_text(json.dumps(doc, indent=2) + '\n')
+(report_dir / REPORT_MD).write_text(md, encoding='utf-8')
+```
+
+Rendering before writing closed the *render-failure* case: a raise can no longer commit a
+sidecar the markdown contradicts. It does not close the *interrupted-between-writes* case. A
+kill between those two lines leaves the pair disagreeing — and this runs on the containment
+path, on the way to `os._exit`, on a rig whose jobs get cancelled by the GitHub job ceiling.
+
+**What remains:** write both to temp files, then `os.replace` both. The window shrinks from
+two full writes to two renames, and neither file is ever observed half-written. `os.replace`
+is atomic per file on POSIX; the pair is still not transactional, which is acceptable and
+should be said in the docstring rather than implied away.
+
+Worth pairing with a test that kills between the writes — or, more practically, one that
+asserts no partial file is ever visible by checking the temp-then-rename shape directly.
+
+## Why it was split out
+
+A durability edge, not a wrong verdict. PR #3840 closed the render-failure half of this
+(nothing is written until the markdown renders); the interrupted-between-writes half needs
+a temp-then-rename and is better reviewed on its own.
diff --git a/docs/superpowers/plans/2026-08-21-hil-report-module.md b/docs/superpowers/plans/2026-08-21-hil-report-module.md
new file mode 100644
index 000000000..5a7c832d8
--- /dev/null
+++ b/docs/superpowers/plans/2026-08-21-hil-report-module.md
@@ -0,0 +1,658 @@
+# hil_report.py Module Implementation Plan
+
+> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
+
+**Goal:** Fold every function that produces, renders, merges or reads `hil_report.json`/`hil_report.md` into one module, `test/hil/helper/hil_report.py`, and take the two fixes that consolidation enables.
+
+**Architecture:** A new leaf-ish module owns the report document. `hil_test.py` and `hil_health.py` both import it, which dissolves the circular-import constraint that forced `write_timeout_report` to compose its own markdown. The duplicated cell classifier (`cell_kind` in `hil_test`, `cell_state` in `hil_summary`) collapses into one. `hil_summary.py` is deleted and its CLI moves in.
+
+**Tech Stack:** Python 3.13 stdlib only (`json`, `argparse`, `pathlib`); existing unit suites under `test/hil/test/` run with plain `unittest`.
+
+**Spec:** `docs/superpowers/specs/2026-08-21-hil-report-module-design.md`
+
+## Global Constraints
+
+- **Behaviour-preserving motion.** `hil_test.py`'s CLI, arguments, output and report format stay byte-identical. The two intended exceptions are named in the spec: the `hil_summary.py` → `hil_report.py` CLI path, and `write_timeout_report` rendering instead of concatenating.
+- **`hil_report.py` must work in two modes.** It is imported as `helper.hil_report` by `hil_test.py`, and run as a script by the operator (`python3 test/hil/helper/hil_report.py <config> -b BOARD`). A script run puts `test/hil/helper/` on `sys.path`, *not* `test/hil/`, so `from helper import hil_health` fails in that mode. Task 1 pins both modes with tests.
+- **Containment paths must never raise.** `mark_report_abandoned` and `write_timeout_report` run while the interpreter is being torn down or on the way to `os._exit`; anything escaping hangs the process in multiprocessing's unbounded `join()`. Their existing broad handlers move with them unchanged.
+- **`hil_ci.sh` stages helpers by an explicit list** (`test/hil/hil_ci.sh:222-228`). A helper module missing from it reaches the rig absent, and the run dies with `ImportError` *after* `REMOTE_DIR` has been wiped. `RemoteStaging.test_import_closure_is_staged_to_the_rig` in `test_hil_bounded.py` already enforces this from the AST import closure; Task 1 only has to add the file to the list.
+- Run `python3 -m unittest discover -s test/hil/test` (~82 s) before each commit; `pre-commit run --files <changed>` before pushing.
+
+---
+
+### Task 1: The module, the vocabulary, one classifier, and the render half
+
+**Files:**
+- Create: `test/hil/helper/hil_report.py`
+- Create: `test/hil/test/test_hil_report.py`
+- Modify: `test/hil/hil_test.py:110` (`REPORT_CELL`), `:1715` (`BOUNDARY_CELL`), `:1902-1903` (`REPORT_MD`/`REPORT_JSON`), `:1921-1978` (`render_matrix`), `:1981-2003` (`render_report`), `:67` (imports)
+- Modify: `test/hil/hil_ci.sh:222-228` (scp list)
+- Modify: `test/hil/test/test_hil_bounded.py` (move `RenderReportIsPureFunctionOfTheDocument` out)
+
+**Interfaces:**
+- Produces: `helper.hil_report` exposing `REPORT_MD`, `REPORT_JSON`, `REPORT_CELL`, `BOUNDARY_CELL`, `LOCKED_CELL`, `cell_state(v) -> str`, `render_matrix(rows_all) -> str`, `render_report(doc) -> str`.
+- `hil_test.py` re-exports nothing: call sites become `hil_report.NAME`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Create `test/hil/test/test_hil_report.py`:
+
+```python
+#!/usr/bin/env python3
+# SPDX-License-Identifier: MIT
+# Unit tests for the report document: the vocabulary, the one cell classifier, rendering,
+# the four writers, and the fold to per-board verdicts. Split out of test_hil_bounded.py
+# and test_hil_health.py when the report code moved into helper/hil_report.py.
+# Run directly:
+# python3 test/hil/test/test_hil_report.py
+import json
+import os
+import subprocess
+import sys
+import unittest
+from pathlib import Path
+from tempfile import TemporaryDirectory
+
+TEST_DIR = os.path.dirname(os.path.abspath(__file__))
+HIL_DIR = os.path.dirname(TEST_DIR)
+sys.path.insert(0, HIL_DIR)
+
+from helper import hil_report
+
+
+class OneClassifierForBothArtifacts(unittest.TestCase):
+ """The markdown tally and the agent's verdict used to classify cells with two separate
+ copies of one rule -- hil_test's cell_kind against REPORT_CELL, and hil_summary's
+ cell_state against its own re-typed '❌'/'⚪' literals. Change the icons and the table
+ and the verdict silently disagree."""
+
+ def test_bare_states(self):
+ self.assertEqual(hil_report.cell_state('fail'), 'fail')
+ self.assertEqual(hil_report.cell_state('skip'), 'skip')
+ self.assertEqual(hil_report.cell_state('pass'), 'pass')
+
+ def test_icon_prefixed_metrics_carry_their_verdict(self):
+ self.assertEqual(hil_report.cell_state(f'{hil_report.REPORT_CELL["fail"]} 29/30'), 'fail')
+ self.assertEqual(hil_report.cell_state(f'{hil_report.REPORT_CELL["skip"]} board wedged'),
+ 'skip')
+
+ def test_an_unprefixed_metric_is_a_pass(self):
+ """Load-bearing: a passing test may return a plain metric string. Classifying
+ unknown shapes as fail would publish a green table as a red verdict."""
+ self.assertEqual(hil_report.cell_state('480.0 MBps'), 'pass')
+ self.assertEqual(hil_report.cell_state('1103 KB/s'), 'pass')
+
+ def test_a_non_string_cell_does_not_raise(self):
+ """render_matrix's copy guarded with isinstance; hil_summary's did not, because its
+ caller str()'d first. The merged one keeps the guard -- it is the safer superset."""
+ self.assertEqual(hil_report.cell_state(None), 'pass')
+
+ def test_the_icons_come_from_REPORT_CELL(self):
+ """No second copy of the emoji anywhere in the module."""
+ src = (Path(HIL_DIR) / 'helper' / 'hil_report.py').read_text(encoding='utf-8')
+ for icon in ('❌', '⚪', '✅'):
+ self.assertEqual(src.count(f"'{icon}'"), 1,
+ f'{icon} is spelled as a literal more than once')
+
+
+class ModuleWorksImportedAndAsAScript(unittest.TestCase):
+ """It is imported as helper.hil_report by hil_test, and run as a script by the operator
+ (.claude/agents/hil-operator.md). A script run puts helper/ on sys.path, NOT test/hil,
+ so a plain `from helper import hil_health` breaks the CLI and only the CLI."""
+
+ def test_importable_as_a_package_module(self):
+ r = subprocess.run(
+ [sys.executable, '-c',
+ f'import sys; sys.path.insert(0, {HIL_DIR!r}); '
+ f'from helper import hil_report; print(hil_report.REPORT_JSON)'],
+ capture_output=True, text=True, timeout=60)
+ self.assertEqual(r.returncode, 0, r.stderr)
+ self.assertIn('hil_report.json', r.stdout)
+
+ def test_runnable_as_a_script(self):
+ r = subprocess.run(
+ [sys.executable, str(Path(HIL_DIR) / 'helper' / 'hil_report.py'), '--help'],
+ capture_output=True, text=True, timeout=60)
+ self.assertEqual(r.returncode, 0, r.stderr)
+
+
+class HilCiStagesEveryHelperTheRunImports(unittest.TestCase):
+ """hil_ci.sh copies helper modules by an EXPLICIT list. One missing module reaches the
+ rig absent and the run dies with ImportError -- after REMOTE_DIR has already been
+ rm -rf'd, so the previous run's report and re-run spec are gone too."""
+
+ def test_the_scp_list_covers_what_hil_test_imports(self):
+ sh = (Path(HIL_DIR) / 'hil_ci.sh').read_text(encoding='utf-8')
+ staged = {line.split('helper/')[1].rstrip('" \\\n')
+ for line in sh.splitlines() if '/test/hil/helper/' in line and '.py' in line}
+ imported = set()
+ for mod in (Path(HIL_DIR) / 'hil_test.py', Path(HIL_DIR) / 'helper' / 'hil_report.py'):
+ src = mod.read_text(encoding='utf-8')
+ for raw in src.splitlines():
+ line = raw.strip() # hil_report's own import is indented in a try
+ if line.startswith('from helper import '):
+ imported |= {f'{n.strip()}.py' for n in line.split('import', 1)[1].split(',')}
+ elif line.startswith('from helper.'):
+ imported.add(line.split('.')[1].split(' ')[0] + '.py')
+ missing = imported - staged
+ self.assertEqual(missing, set(),
+ f'hil_ci.sh does not stage {missing}; a remote run will ImportError')
+
+
+if __name__ == '__main__':
+ unittest.main()
+```
+
+Then **move** the class `RenderReportIsPureFunctionOfTheDocument` from `test/hil/test/test_hil_bounded.py` into this file verbatim, changing only `hil_test.render_report` → `hil_report.render_report` throughout.
+
+- [ ] **Step 2: Run them to verify they fail**
+
+Run: `python3 test/hil/test/test_hil_report.py`
+Expected: FAIL — `ModuleNotFoundError: No module named 'helper.hil_report'`
+
+- [ ] **Step 3: Create the module**
+
+Create `test/hil/helper/hil_report.py`:
+
+```python
+#!/usr/bin/env python3
+# SPDX-License-Identifier: MIT
+"""The HIL report document: one owner for hil_report.json and hil_report.md.
+
+The markdown IS a rendering of the sidecar -- every writer goes through render_report(),
+so a table can never contain something the JSON does not. This module owns the whole life
+of that document: the cell vocabulary, the one classifier both artifacts share, rendering,
+the four writers, and the fold to one machine-readable verdict per board.
+
+Dual-mode by design: imported as `helper.hil_report` by hil_test.py, and run as a script by
+the operator (see .claude/agents/hil-operator.md). A script run puts test/hil/helper on
+sys.path rather than test/hil, hence the guarded hil_health import below.
+"""
+import argparse
+import json
+import sys
+from pathlib import Path
+
+try: # imported as part of the helper package
+ from helper.hil_health import _p
+except ImportError: # run as a script: helper/ is sys.path[0]
+ from hil_health import _p
+
+REPORT_MD = 'hil_report.md'
+REPORT_JSON = 'hil_report.json'
+# The status vocabulary, shared by the code that WRITES a cell (hil_test's test runners) and
+# the code that reads one back (cell_state). One dict, so the human's table and the agent's
+# verdict cannot drift apart.
+REPORT_CELL = {'pass': '✅', 'fail': '❌', 'skip': '⚪'}
+BOUNDARY_CELL = 'same-PID boundary'
+LOCKED_CELL = 'board-locked'
+
+
+def cell_state(v) -> str:
+ """'pass' | 'fail' | 'skip' for one report cell.
+
+ THE classifier -- the markdown tally and the per-board verdict both call this, so they
+ cannot disagree. 'fail' or a ❌ prefix is a failure, 'skip' or a ⚪ prefix is a skip, and
+ EVERYTHING ELSE is a pass. That last arm is load-bearing: a passing test may return a
+ plain metric string ('480.0 MBps') that lands in the cell unprefixed, while failures are
+ guaranteed marked -- TestFail's docstring pins that its metric is icon-prefixed precisely
+ so render and tally treat it as a failure. Classifying unknown shapes as fail here would
+ publish a green table as a red verdict.
+
+ isinstance-guarded: cells are usually str but a caller may hand over None or a number,
+ and .startswith on those raises inside a report writer that must not raise."""
+ if v == 'fail' or (isinstance(v, str) and v.startswith(REPORT_CELL['fail'])):
+ return 'fail'
+ if v == 'skip' or (isinstance(v, str) and v.startswith(REPORT_CELL['skip'])):
+ return 'skip'
+ return 'pass'
+```
+
+Then move, verbatim, from `hil_test.py`:
+- `render_matrix` (`hil_test.py:1921-1978`) — with one change: delete its nested `cell_kind`
+ definition and call the module-level `cell_state` instead. The line
+ `kinds = [cell_kind(v) for _, cells, _ in rows_all for v in cells.values()]` becomes
+ `kinds = [cell_state(v) for _, cells, _ in rows_all for v in cells.values()]`.
+- `render_report` (`hil_test.py:1981-2003`) — unchanged.
+
+Add a placeholder CLI so `--help` works (Task 4 fills in `summarize`):
+
+```python
+def main() -> int:
+ ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
+ ap.add_argument('config_file')
+ ap.add_argument('-b', '--board', action='append', default=[],
+ help='boards to report on; default: every board in the config')
+ ap.add_argument('--report-dir', default='.', help=f'where {REPORT_JSON} lives (default: cwd)')
+ ap.parse_args()
+ raise SystemExit('hil_report: summarize() lands in Task 4')
+
+
+if __name__ == '__main__':
+ sys.exit(main())
+```
+
+- [ ] **Step 4: Point `hil_test.py` at the module**
+
+In `hil_test.py:67`, extend the import:
+
+```python
+from helper import hil_health, hil_lock, hil_report, hil_util
+```
+
+Delete `REPORT_CELL` (`:110`), `BOUNDARY_CELL` (`:1715`), `REPORT_MD`/`REPORT_JSON`
+(`:1902-1903`), `render_matrix` and `render_report` from `hil_test.py`. Then rewrite every
+reference to the moved names as `hil_report.<name>`. Find them all with:
+
+```bash
+grep -n "REPORT_CELL\|BOUNDARY_CELL\|REPORT_MD\|REPORT_JSON\|render_matrix\|render_report" \
+ test/hil/hil_test.py
+```
+
+Known sites: `:876`, `:1369`, `:1459`, `:1490`, `:1492`, `:1508`, `:1818`, `:1834`, `:2162`,
+`:2191-2192`, `:2209-2210`, `:2403`, `:2592`.
+
+- [ ] **Step 5: Stage the new module for remote runs**
+
+In `test/hil/hil_ci.sh:222-228`, add the module to the scp list (keep alphabetical-ish order
+with the rest):
+
+```bash
+scp -q "$ROOT_DIR/test/hil/helper/__init__.py" \
+ "$ROOT_DIR/test/hil/helper/hil_util.py" \
+ "$ROOT_DIR/test/hil/helper/hil_health.py" \
+ "$ROOT_DIR/test/hil/helper/hil_lock.py" \
+ "$ROOT_DIR/test/hil/helper/hil_report.py" \
+ "$ROOT_DIR/test/hil/helper/hil_summary.py" \
+ "$ROOT_DIR/test/hil/helper/hil_select.py" \
+ "$REMOTE:$REMOTE_DIR/test/hil/helper/"
+```
+
+- [ ] **Step 6: Run the tests**
+
+Run: `python3 test/hil/test/test_hil_report.py` → OK
+Run: `python3 -m unittest discover -s test/hil/test` → 274 OK (266 + 8 new: 5 classifier,
+2 dual-mode, 1 scp guard; `RenderReport…` moves rather than adds)
+
+- [ ] **Step 7: Commit**
+
+```bash
+git add test/hil/helper/hil_report.py test/hil/hil_test.py test/hil/hil_ci.sh \
+ test/hil/test/test_hil_report.py test/hil/test/test_hil_bounded.py
+git commit -m "hil_report: new module for the report vocabulary, classifier and rendering
+
+The markdown tally and the agent's verdict classified cells with two separate
+copies of one rule, the second documented as 'the EXACT classifier hil_test.py's
+own tally uses'. One cell_state now serves both, keyed off the one REPORT_CELL."
+```
+
+---
+
+### Task 2: Move the three writers
+
+**Files:**
+- Modify: `test/hil/helper/hil_report.py` (add the writers)
+- Modify: `test/hil/hil_test.py:2005-2036` (`write_report`, `mark_report_abandoned`), `:2149-2212` (`accumulate_report`)
+- Modify: `test/hil/test/test_hil_bounded.py` (move three classes out), `test/hil/test/test_hil_report.py`
+
+**Interfaces:**
+- Consumes: `render_report`, `REPORT_MD`, `REPORT_JSON`, `BOUNDARY_CELL` from Task 1.
+- Produces: `hil_report.write_report(report_dir, doc)`, `hil_report.mark_report_abandoned(report_dir, why)`, `hil_report.accumulate_report(mret, report_dir, fresh, scope='', banner='') -> str`.
+
+- [ ] **Step 1: Move the tests**
+
+Move these classes from `test/hil/test/test_hil_bounded.py` into `test/hil/test/test_hil_report.py`,
+verbatim except `hil_test.<name>` → `hil_report.<name>` for the three moved functions:
+
+- `ScopeSurvivesInTheJson`
+- `EveryExitPathLeavesBothArtifacts`
+- `AbandonNoticeLandsInBothArtifacts`
+- `CaveatSurvivesAccumulate`
+- `MarkdownIsAlwaysARenderingOfTheJson`
+
+`AbandonNoticeLandsInBothArtifacts.test_an_existing_abandon_caveat_is_not_overwritten` calls
+`hil_health.write_timeout_report`; leave that call as-is — Task 3 moves it.
+
+- [ ] **Step 2: Run them to verify they fail**
+
+Run: `python3 test/hil/test/test_hil_report.py`
+Expected: FAIL — `AttributeError: module 'helper.hil_report' has no attribute 'write_report'`
+
+- [ ] **Step 3: Move the functions**
+
+Cut `write_report` (`hil_test.py:2005-2014`), `mark_report_abandoned` (`:2016-2036`) and
+`accumulate_report` (`:2149-2212`) from `hil_test.py` and paste them into `hil_report.py`
+below `render_report`, unchanged.
+
+Add to `accumulate_report`'s docstring, after the existing text, so the wart is recorded
+where a reader meets it:
+
+```
+ `mret` is hil_test.py's worker-result shape (name, err, fts, rows, ...), so this one
+ function knows something about its caller that the rest of the module does not. Folding
+ mret into rows could live in hil_test and only the merge here, but that would rewrite
+ the subtle parts -- stale board-locked clearing, BOUNDARY_CELL dropping, duration=None
+ preservation -- for a tidier seam. Data-shape coupling, not an import cycle.
+```
+
+- [ ] **Step 4: Update the call sites**
+
+In `hil_test.py`, the three call sites become `hil_report.*`:
+
+```bash
+grep -n "accumulate_report(\|write_report(\|mark_report_abandoned(" test/hil/hil_test.py
+```
+
+Known sites: `:2260` (inside `_abandon_exit`), `:2351` (no-boards exit), `:2486`, `:2525`,
+`:2618`.
+
+- [ ] **Step 5: Run the tests**
+
+Run: `python3 -m unittest discover -s test/hil/test` → 274 OK (motion only, no count change)
+
+- [ ] **Step 6: Commit**
+
+```bash
+git add test/hil/helper/hil_report.py test/hil/hil_test.py \
+ test/hil/test/test_hil_report.py test/hil/test/test_hil_bounded.py
+git commit -m "hil_report: move the report writers off hil_test
+
+write_report, mark_report_abandoned and accumulate_report join the renderer they
+already call. Pure motion; accumulate_report's knowledge of mret's tuple shape
+moves with it and is now documented rather than implicit."
+```
+
+---
+
+### Task 3: `write_timeout_report` renders like everyone else
+
+**Files:**
+- Modify: `test/hil/helper/hil_report.py` (receive the function)
+- Modify: `test/hil/helper/hil_health.py:347-398` (remove it), `:19` (drop `import json`)
+- Modify: `test/hil/hil_test.py:2498` (call site)
+- Modify: `test/hil/test/test_hil_health.py` (move `WriteTimeoutReport` out), `test/hil/test/test_hil_report.py`
+
+**Interfaces:**
+- Consumes: `render_report`, `write_report` from Tasks 1-2.
+- Produces: `hil_report.write_timeout_report(report_dir, boards, secs, banner='', prefix='')`. The `md_name` parameter is **gone** — the module owns `REPORT_MD`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Move `WriteTimeoutReport` from `test/hil/test/test_hil_health.py` into
+`test/hil/test/test_hil_report.py`, changing `hil_health.write_timeout_report` →
+`hil_report.write_timeout_report` and dropping the `md_name` argument from every call. Two
+of its tests change substantively:
+
+```python
+ def test_the_prior_attempts_rows_survive(self):
+ """Was: the prior MARKDOWN TEXT survives below the banner. It now re-renders from
+ the merged sidecar, so the guarantee is stated against rows -- one table with the
+ stuck boards in it, rather than a banner stapled above a duplicate table."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('done', 0, 0, [('done', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600)
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertEqual([r['board'] for r in doc['rows']], ['done', 'stuck'])
+ md = (rd / hil_report.REPORT_MD).read_text()
+ self.assertIn('done', md)
+ self.assertIn('stuck', md)
+ self.assertIn('abandoned', md)
+ self.assertLess(md.index('abandoned'), md.index('done'))
+ self.assertEqual(md.count('| Board'), 1, 'the prior table was duplicated, not merged')
+
+ def test_prefix_carries_the_preflight_diagnosis(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.write_timeout_report(rd, [{'name': 'b1'}], 4200,
+ prefix='> **wedged usb_hub_wq worker.**\n')
+ out = (rd / hil_report.REPORT_MD).read_text()
+ self.assertTrue(out.startswith('> **wedged usb_hub_wq worker.**'))
+ self.assertIn('timed out after 4200s', out)
+ self.assertIn('b1', out)
+```
+
+And in `MarkdownIsAlwaysARenderingOfTheJson`, **delete**
+`test_the_pool_guard_fallback_agrees_even_if_it_does_not_render` and add the fifth case in
+its place:
+
+```python
+ def test_the_pool_guard_fallback(self):
+ """The last writer to join the invariant: it composed its own markdown only because
+ hil_health could not import the renderer."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('done', 0, 0, [('done', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600,
+ prefix='> **wedged usb_hub_wq worker.**\n')
+ self._check(rd)
+```
+
+- [ ] **Step 2: Run them to verify they fail**
+
+Run: `python3 test/hil/test/test_hil_report.py`
+Expected: FAIL — `AttributeError: module 'helper.hil_report' has no attribute 'write_timeout_report'`
+
+- [ ] **Step 3: Move it and make it render**
+
+Add to `hil_report.py`, and delete `hil_health.py:347-398` plus its now-unused
+`import json` at `hil_health.py:19`:
+
+```python
+def write_timeout_report(report_dir: Path, boards, secs: int,
+ banner: str = '', prefix: str = '') -> None:
+ """Leave a report behind when the worker pool has to be abandoned.
+
+ map_async is all-or-nothing, so a timeout loses every per-board result and the report
+ dir would stay empty with no reason for the failure. Any prior attempt's rows are kept
+ and the stuck boards are merged in beside them.
+
+ `prefix` carries the preflight rig-health verdict: the timeout aborts before
+ accumulate_report, so without it the report loses the one line saying WHY the pool never
+ finished."""
+ try:
+ # Built INSIDE the try: a roster entry without a 'name' key raises while assembling
+ # the board list, and outside the try that escaped and stranded the runner -- which
+ # is exactly what the broad handler below exists to prevent.
+ caveat = (prefix + '\n' if prefix else '') + (banner or (
+ f'**HIL run abandoned: worker pool timed out after {secs}s.**\n\n'
+ f'No per-board results could be collected for this attempt, so any rows below '
+ f'are from an earlier one. Boards dispatched:\n\n'
+ + '\n'.join(f'- {b.get("name", "?")}' for b in boards) + '\n'))
+ # Rows MERGE rather than replace: an earlier attempt's finished boards are real
+ # results and this attempt has none of its own. Own handler, because a torn sidecar
+ # must not cost the stuck rows -- losing the old table is a nicety, losing the
+ # caveat is the failure.
+ jpath = report_dir / REPORT_JSON
+ try:
+ doc = json.loads(jpath.read_text()) if jpath.is_file() else {}
+ rows = list(doc.get('rows', []))
+ except (OSError, ValueError, TypeError, AttributeError):
+ doc, rows = {}, []
+ done = {r.get('board') for r in rows if isinstance(r, dict)}
+ rows += [{'board': b.get('name', '?'), 'cells': {'pool-timeout': 'fail'},
+ 'duration': None} for b in boards if b.get('name', '?') not in done]
+ write_report(report_dir, {'rows': rows, 'banner': doc.get('banner', ''),
+ 'scope': doc.get('scope', ''), 'caveat': caveat})
+ except Exception as e: # noqa: BLE001
+ # Deliberately broad: this is the first statement of the pool-abandon path, so ANY
+ # escape skips kill_pool_children and os._exit and strands the runner.
+ _p(f'warning: cannot write {REPORT_MD} to {report_dir}: {e}', flush=True)
+```
+
+Update `hil_health.py`'s module docstring: its first line reads "Shutting a wedged HIL run
+down: kill what the workers spawned, then report." — drop ", then report".
+
+- [ ] **Step 4: Update the call site**
+
+`hil_test.py:2498` becomes:
+
+```python
+ hil_report.write_timeout_report(
+ report_dir, [b for b in config_boards
+ if b['name'] in stuck], POOL_TIMEOUT,
+ prefix=health_banner)
+```
+
+- [ ] **Step 5: Run the tests**
+
+Run: `python3 -m unittest discover -s test/hil/test` → 274 OK (one deleted, one added)
+
+- [ ] **Step 6: Commit**
+
+```bash
+git add test/hil/helper/hil_report.py test/hil/helper/hil_health.py test/hil/hil_test.py \
+ test/hil/test/test_hil_report.py test/hil/test/test_hil_health.py
+git commit -m "hil_report: the pool-guard fallback renders like every other writer
+
+It composed its own markdown for one reason: hil_health cannot import hil_test
+back, so it could not reach render_report. With the renderer in a module both
+import, that constraint is gone and all five writers are byte-identical --
+MarkdownIsAlwaysARenderingOfTheJson covers the fifth, and the weaker
+'agrees even if it does not render' promise is deleted.
+
+hil_health goes back to doing one thing: killing wedged processes."
+```
+
+---
+
+### Task 4: Fold `hil_summary.py` in and delete it
+
+**Files:**
+- Modify: `test/hil/helper/hil_report.py` (real `summarize` + CLI)
+- Delete: `test/hil/helper/hil_summary.py`
+- Modify: `test/hil/hil_ci.sh` (drop `hil_summary.py` from the scp list)
+- Modify: `.claude/agents/hil-operator.md:71`, `.claude/workflows/hil-validate.js:14,17,54,58,67`, `.claude/workflows/test-hil-validate.mjs:7`
+- Modify: `test/hil/test/test_hil_bounded.py` (move `SummaryFoldsReportToBoards` out), `test/hil/test/test_hil_report.py`
+
+**Interfaces:**
+- Consumes: `cell_state`, `LOCKED_CELL`, `REPORT_JSON` from Task 1.
+- Produces: `hil_report.variants_of(cfg, board) -> list`, `hil_report.summarize(cfg, boards, report) -> dict` returning `{'results': [...], 'banner': str, 'caveat': str}`; CLI `python3 test/hil/helper/hil_report.py <config> [-b BOARD]... [--report-dir DIR]`.
+
+- [ ] **Step 1: Move the tests**
+
+Move `SummaryFoldsReportToBoards` from `test/hil/test/test_hil_bounded.py` into
+`test/hil/test/test_hil_report.py`, changing the subprocess target from
+`helper/hil_summary.py` to `helper/hil_report.py` in both places (`test_hil_bounded.py:1675`
+and `:1757`). Add one test pinning that the old entry point is gone:
+
+```python
+ def test_the_old_entry_point_is_gone(self):
+ """hil_summary.py's CLI moved here. A leftover file would keep working while
+ drifting from the module that now owns the fold."""
+ self.assertFalse((Path(HIL_DIR) / 'helper' / 'hil_summary.py').exists())
+```
+
+- [ ] **Step 2: Run them to verify they fail**
+
+Run: `python3 test/hil/test/test_hil_report.py`
+Expected: FAIL — the subprocess exits non-zero with `hil_report: summarize() lands in Task 4`
+
+- [ ] **Step 3: Move `summarize` in and delete the old file**
+
+Copy `variants_of` (`hil_summary.py:47-52`) and `summarize` (`:54-92`) into `hil_report.py`
+verbatim, with two changes: `cell_state(str(val))` becomes `cell_state(val)` (the merged
+classifier is isinstance-guarded, so the `str()` is dead), and the module's own
+`FAIL_ICON`/`SKIP_ICON`/`LOCKED_CELL`/`cell_state` definitions are NOT copied — Task 1's
+already serve.
+
+Replace the Task 1 placeholder `main()` with the real one from `hil_summary.py:94-115`,
+changing `Path(a.report_dir) / 'hil_report.json'` to `Path(a.report_dir) / REPORT_JSON`.
+
+Then:
+
+```bash
+git rm test/hil/helper/hil_summary.py
+```
+
+- [ ] **Step 4: Update the consumers**
+
+`test/hil/hil_ci.sh` — remove the `hil_summary.py` line from the scp list added in Task 1.
+
+`.claude/agents/hil-operator.md:71`:
+
+```bash
+python3 test/hil/helper/hil_report.py <config> -b BOARD [-b BOARD...] # from the report dir
+```
+
+`.claude/workflows/hil-validate.js:58`:
+
+```javascript
+ ` python3 test/hil/helper/hil_report.py <the config you used> ${boards.map((b) => `-b ${b}`).join(' ')}\n` +
+```
+
+In `.claude/workflows/hil-validate.js` lines 14, 17, 54 and 67, and
+`.claude/workflows/test-hil-validate.mjs` line 7, replace the prose mentions of
+`hil_summary.py` with `hil_report.py`. Change nothing else in those files — the operator's
+return contract (`{results, banner, wedged}`) is untouched.
+
+- [ ] **Step 5: Run the tests**
+
+Run: `python3 test/hil/test/test_hil_report.py` → OK
+Run: `python3 -m unittest discover -s test/hil/test` → 275 OK
+Run: `node .claude/workflows/test-hil-validate.mjs` → OK
+Run: `grep -rn "hil_summary" . --include=*.py --include=*.sh --include=*.js --include=*.mjs --include=*.md | grep -v docs/superpowers` → no hits
+
+- [ ] **Step 6: Commit**
+
+```bash
+git add test/hil/helper/hil_report.py test/hil/hil_ci.sh test/hil/test/ \
+ .claude/agents/hil-operator.md .claude/workflows/hil-validate.js \
+ .claude/workflows/test-hil-validate.mjs
+git rm --cached test/hil/helper/hil_summary.py 2>/dev/null || true
+git commit -m "hil_report: fold hil_summary in; one module owns the document end to end
+
+The fold to per-board verdicts is the read half of the artifact the rest of this
+module writes, and it carried the second copy of the cell classifier. The CLI
+keeps its arguments; only its path changes, which the two harness docs that
+invoke it by name follow."
+```
+
+---
+
+## Validation
+
+- [ ] **Full gate**
+
+```bash
+python3 -m unittest discover -s test/hil/test # 275 OK
+pre-commit run --all-files
+```
+
+- [ ] **Prove the motion changed no behaviour.** Re-render the real fleet report captured
+ before the refactor and diff it against what the branch produces now:
+
+```bash
+python3 - <<'EOF'
+import json, sys
+sys.path.insert(0, 'test/hil')
+from helper import hil_report
+doc = json.load(open('hil_report.json')) # the pair the rig produced pre-refactor
+assert open('hil_report.md').read() == hil_report.render_report(doc) + '\n', 'render drifted'
+print('render is byte-identical to the pre-refactor artifact')
+EOF
+```
+
+- [ ] **Rig re-check.** `hil_report.py` must reach the rig and the CLI must run there:
+
+```bash
+bash test/hil/hil_ci.sh -b stm32f407disco -b nanoch32v203
+ssh [email protected] 'cd /tmp/tinyusb-hil && python3 test/hil/helper/hil_report.py \
+ test/hil/tinyusb.json -b stm32f407disco -b nanoch32v203'
+```
+
+Expect a two-board table, `md == render_report(json)`, and a `summarize` verdict naming both
+boards — `nanoch32v203` proving the variant fold still works through the moved code.
+
+## Out of scope
+
+Each its own follow-up, unchanged from the spec:
+
+- Splitting `accumulate_report`'s `mret` folding from its merge.
+- The flat `HIL_POOL_TIMEOUT` that does not scale with board count.
+- Carrying `caveat` through the operator/workflow return contract (`hil-validate.js:34`).
diff --git a/docs/superpowers/specs/2026-08-19-ci-build-family-filter-design.md b/docs/superpowers/specs/2026-08-19-ci-build-family-filter-design.md
index 8f77dc50a..799c83c23 100644
--- a/docs/superpowers/specs/2026-08-19-ci-build-family-filter-design.md
+++ b/docs/superpowers/specs/2026-08-19-ci-build-family-filter-design.md
@@ -46,33 +46,41 @@ never inflates one axis with another's breadth.
`FAM` = the families whose `family.cmake` references the changed path (CMake only — see below).
"roster boards" = boards on `test/hil/{tinyusb,hfp}.json`.
-| # | Changed path | Build families | Build examples | HIL boards → tests |
-| --- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------- | ----------------------------------------------------------------- | ---------------------------------------------------------------------------------------------- |
-| 1 | `docs/`, `.claude/`, `*.md`, `*.rst`, `LICENSE` | — | — | — |
-| 2 | `test/hil/**` | — | — | all boards → all tests |
-| 2b | `tools/metrics.py`, `.github/scripts/metrics_*.py` | `ALL` (unchanged — `tinyusb_metrics` runs `metrics.py` as a build target) | `ALL` | — (nothing on the rig runs it) |
-| 3 | `src/portable/<port>/dcd_*`, `*_device.[ch]` | `FAM` | `DEV`+`DUAL` | `FAM`'s device-role boards → device+dual tests |
-| 4 | `src/portable/<port>/hcd_*`, `*_host.[ch]` | `FAM` | `HOST`+`DUAL` | `FAM`'s host-role boards → host+dual tests |
-| 5 | `src/portable/<port>/**` (anything else) | `FAM` | `ALL` | `FAM`'s boards → all their tests |
-| 5b | `src/portable/<port>/**` where `FAM` is empty | — | — | — (empty resolves to nothing on BOTH axes) |
-| 6 | `hw/bsp/<family>/**` | that family | `ALL` | that family's boards → all tests (a `boards/<board>/` path narrows to that board) |
-| 7 | `hw/mcu/<vendor>/**` | `FAM` — empty resolves to nothing (maintainer ruling) | `ALL` | `FAM`'s boards → all tests; empty resolves to nothing (maintainer ruling) ⚠ *see below* |
-| 8 | `src/class/<cls>/*_device.[ch]` | `ALL` | examples enabling `CFG_TUD_<CLS>` | device-role boards → HIL tests enabling `CFG_TUD_<CLS>` |
-| 9 | `src/class/<cls>/*_host.[ch]` | `ALL` | examples enabling `CFG_TUH_<CLS>` | host-role boards → HIL tests enabling `CFG_TUH_<CLS>` |
-| 10 | `src/class/<cls>/**` (shared header) | `ALL` | either, **plus include-edge classes** | both roles → same, plus include-edge classes |
-| 11 | `src/device/**` | `ALL` | `DEV`+`DUAL` | device-role boards → device+dual tests |
-| 12 | `src/host/**` | `ALL` | `HOST`+`DUAL` | host-role boards → host+dual tests |
-| 13 | `examples/<role>/<name>/**` | `ALL` | just `<name>` | if `<name>` is a HIL test: all boards → that test; else nothing |
-| 14 | `examples/device/board_test/**` | `ALL` | just `board_test` | all boards → all tests (HIL parking firmware) |
-| 15 | `examples/build_system/**`, `examples/CMakeLists.txt`, `examples/<role>/CMakeLists.txt` | `ALL` | `ALL` | all boards → all tests |
-| 16 | `src/common/`, `src/osal/`, `src/tusb.[ch]`, `src/tusb_option.h`, `tools/build*.py`, `tools/cmake/**`, `hw/bsp/{family_support.cmake,board.c,board_api.h,ansi_escape.h}`, `.github/**` | `ALL` | `ALL` | all boards → all tests |
-| 16a | `lib/<name>/**` | `ALL` | examples whose own `CMakeLists.txt`/`Makefile` names `lib/<name>` | those examples that are HIL tests, on all boards; empty resolves to nothing |
-| 16b | `tools/get_deps.py` | families whose `deps_mandatory`/`deps_optional` entries changed | `ALL` | those families' boards → all tests; a logic change, an `'all'` entry, no base content or a changed token naming no family → full |
-| 17 | anything unclassified | `ALL` | `ALL` | all boards → all tests (fail-open) |
+| # | Changed path | Build families | Build examples | HIL boards → tests |
+| --- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------- | ----------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------- |
+| 1 | `docs/`, `.claude/`, `*.md`, `*.rst`, `LICENSE` | — | — | — |
+| 1b | `.gitignore`, `.clang-format`, `.idea/**`, `test/{fuzz,unit-test}/**`, `test/hil/test/**`, non-build `.github/**`, packaging manifests | — | — | — |
+| 2 | `test/hil/**` (not `test/hil/test/**`) | — | — | all boards → all tests |
+| 2b | `tools/metrics.py`, `.github/scripts/metrics_*.py` | `ALL` (unchanged — `tinyusb_metrics` runs `metrics.py` as a build target) | `ALL` | — (nothing on the rig runs it) |
+| 3 | `src/portable/<port>/dcd_*`, `*_device.[ch]` | `FAM` | `DEV`+`DUAL` | `FAM`'s device-role boards → device+dual tests |
+| 4 | `src/portable/<port>/hcd_*`, `*_host.[ch]` | `FAM` | `HOST`+`DUAL` | `FAM`'s host-role boards → host+dual tests |
+| 5 | `src/portable/<port>/**` (anything else) | `FAM` | `ALL` | `FAM`'s boards → all their tests |
+| 5b | `src/portable/<port>/**` where `FAM` is empty | — | — | — (empty resolves to nothing on BOTH axes) |
+| 6 | `hw/bsp/<family>/**` | that family | `ALL` | that family's boards → all tests (a `boards/<board>/` path narrows to that board) |
+| 7 | `hw/mcu/<vendor>/**` | `FAM` — empty resolves to nothing (maintainer ruling) | `ALL` | `FAM`'s boards → all tests; empty resolves to nothing (maintainer ruling) ⚠ *see below* |
+| 8 | `src/class/<cls>/*_device.[ch]` | `ALL` | examples enabling `CFG_TUD_<CLS>` | device-role boards → HIL tests enabling `CFG_TUD_<CLS>` |
+| 9 | `src/class/<cls>/*_host.[ch]` | `ALL` | examples enabling `CFG_TUH_<CLS>` | host-role boards → HIL tests enabling `CFG_TUH_<CLS>` |
+| 10 | `src/class/<cls>/**` (shared header) | `ALL` | either, **plus include-edge classes** | both roles → same, plus include-edge classes |
+| 11 | `src/device/**` | `ALL` | `DEV`+`DUAL` | device-role boards → device+dual tests |
+| 12 | `src/host/**` | `ALL` | `HOST`+`DUAL` | host-role boards → host+dual tests |
+| 12b | `src/typec/**` | `ALL` | examples enabling `CFG_TUC_ENABLED` | — (no rig board runs a typec test) |
+| 13 | `examples/<role>/<name>/**` | `ALL` | just `<name>` | if `<name>` is a HIL test: all boards → that test; else nothing |
+| 14 | `examples/device/board_test/**` | `ALL` | just `board_test` | all boards → all tests (HIL parking firmware) |
+| 15 | `examples/build_system/**`, `examples/CMakeLists.txt`, `examples/<role>/CMakeLists.txt` | `ALL` | `ALL` | all boards → all tests |
+| 16 | `src/common/`, `src/osal/`, `src/tusb.[ch]`, `src/tusb_option.h`, `tools/{build,build_utils,ci_select}.py`, `tools/cmake/**`, `src/CMakeLists.txt`, `src/tinyusb.mk`, `hw/bsp/{family_support.{cmake,mk},family_rules.mk,zephyr_board_aliases.cmake,board.c,board_api.h,ansi_escape.h}`, `.github/**`, `.circleci/**` | `ALL` | `ALL` | all boards → all tests |
+| 16a | `lib/<name>/**` | `ALL` | examples whose own `CMakeLists.txt`/`Makefile` names `lib/<name>` | those examples that are HIL tests, on all boards; empty resolves to nothing |
+| 16b | `tools/get_deps.py` | families whose `deps_mandatory`/`deps_optional` entries changed | `ALL` | those families' boards → all tests; a logic change, an `'all'` entry, no base content or a changed token naming no family → full |
+| 17 | anything unclassified (no tracked file reaches this — TestNoTrackedFileIsUnclassified) | `ALL` | `ALL` | all boards → all tests (fail-open) |
**Rule 2 is deliberately asymmetric.** A `test/hil/**` change is invisible to the family matrix
but is exactly what the rig exercises, so it builds nothing and runs everything.
+`test/hil/test/**` is carved out to rule 1b: it holds the harness's own unit tests, which
+nothing on the rig runs (pre-commit does, and `build.yml` runs `test_ci_select.py` as the
+gate before trusting a selection). A bare `test/hil/` prefix was booking the full 27-board
+rig for diffs that cannot reach it. The carve-out is a claim about that directory's
+contents, so a test pins its file list: add anything the rig reads and it fails.
+
**Rule 7 is the one HIL-side behaviour change in this design.** Today `hw/mcu/` sits in
`hil_select`'s `_FULL_RE` and forces the full HIL matrix. Since the build axis now resolves
those paths to a family through the same scan, forcing full on the rig is inconsistent. The
@@ -144,6 +152,19 @@ the existing `test_hil_util.BottomLayer` structural tests.
Fail-open survives where it belongs: an *unclassified* path or any exception widens to `ALL` on
every axis.
+### A class no example enables selects nothing
+
+`src/class/bth` is the live instance: no example's `tusb_config.h` sets `CFG_TUD_BTH`, so
+rules 8-10 resolve to no examples and a bth-only PR builds nothing and runs nothing. That is
+the empty-means-empty ruling applied to classes, and it is deliberate — nothing compiles the
+file, so nothing can validate it, and the master-push build is the net.
+
+Worth stating plainly because the exposure changed: GHA used to rebuild everything for such
+a PR by accident, through the empty-`families` bug in `build.yml`. With that fixed, both
+providers now correctly build nothing, so `tud_bt_*` can be broken by a green PR.
+`TestClassesWithNoEnablingExample` pins the set to `{bth}` so a second class cannot enter
+this state unnoticed.
+
### Why `hw/mcu/**` is rule 7 and not "full"
`hw/mcu` is overwhelmingly dependency territory — `tools/get_deps.py` has 87 entries under it,
diff --git a/docs/superpowers/specs/2026-08-21-hil-report-module-design.md b/docs/superpowers/specs/2026-08-21-hil-report-module-design.md
new file mode 100644
index 000000000..41e7000b7
--- /dev/null
+++ b/docs/superpowers/specs/2026-08-21-hil-report-module-design.md
@@ -0,0 +1,144 @@
+# hil_report.py: one owner for the HIL report document
+
+**Date:** 2026-08-21
+**Branch:** `hil-report` (continues the report-unification work already on it)
+
+## Motivation
+
+`hil_report.json` and `hil_report.md` are now one document rendered two ways, but the code that
+produces, renders, merges and reads that document is spread across three modules:
+
+| Module | Report-related content |
+|---|---|
+| `hil_test.py` | `REPORT_CELL`, `BOUNDARY_CELL`, `REPORT_MD`, `REPORT_JSON`, `render_matrix`, `render_report`, `write_report`, `mark_report_abandoned`, `accumulate_report` |
+| `helper/hil_health.py` | `write_timeout_report` — composes its own markdown |
+| `helper/hil_summary.py` | `cell_state`, `variants_of`, `summarize`, CLI |
+
+Two concrete defects follow from that spread.
+
+**One classifier, two copies.** `hil_test.py:1966` (`cell_kind`, keyed off `REPORT_CELL`) and
+`hil_summary.py:34` (`cell_state`, with its own re-typed `FAIL_ICON, SKIP_ICON = '❌', '⚪'`)
+implement the same rule. The latter's docstring says it is *"the EXACT classifier hil_test.py's own
+tally uses"* — the duplication was noticed and documented as an obligation to keep in sync, rather
+than removed. Change `REPORT_CELL` and the human's table and the agent's verdict silently disagree:
+the markdown says ❌ where the JSON says `pass`. That is the same class of defect this branch
+exists to eliminate, one layer up.
+
+**A writer that cannot render.** `hil_test.py` imports `hil_health`, so `hil_health` cannot import
+`hil_test` back. That is the only reason `write_timeout_report` composes its own markdown instead of
+calling `render_report`, and the only reason the pool-guard fallback is held to a weaker promise
+(same boards and caveat in both artifacts, not byte-identical) while the other four writers are
+exact. The constraint is structural, not essential: a leaf module both can import dissolves it.
+
+## Goal / non-goals
+
+**Goal:** `test/hil/helper/hil_report.py` becomes the single owner of the report document.
+
+**This is NOT purely code motion, and the distinction matters for review.** Measured against
+`master`, `hil_test.py` contains only `render_matrix` and `accumulate_report`. Everything else in
+the new module — `render_report`, `write_report`, `mark_report_abandoned`, `mark_report_no_boards`,
+`_load`, `_write_stuck_over_prior_md`, `cell_state`, and the `scope`/`caveat` plumbing — is NEW
+code, roughly 150 lines of it, and two rounds of review found most of their defects there. Read
+those functions as new, not as relocated. `hil_test.py`'s CLI, arguments and table format do stay
+unchanged.
+
+**Deliberate user-visible changes:**
+1. `hil_summary.py` is deleted; its CLI moves to `hil_report.py`. The documented command becomes
+ `python3 test/hil/helper/hil_report.py <config> -b BOARD [-b BOARD…]`.
+2. `write_timeout_report` re-renders from the merged sidecar instead of stapling its banner above
+ the previous attempt's markdown text. Output improves — one table containing the stuck boards,
+ rather than a fresh banner above a duplicate table — but it is a change (see Testing).
+
+**Non-goals (explicit follow-ups, not this change):**
+- Splitting `accumulate_report`'s `mret` folding from its merge (see "Deliberate wart").
+- The flat `HIL_POOL_TIMEOUT` that does not scale with board count (`hil_test.py:225`).
+
+## Resulting layout (`test/hil/`)
+
+| File | ~Lines | Role |
+|---|---|---|
+| `hil_test.py` | 2390 (−250) | tests + orchestration + CLI |
+| `helper/hil_report.py` (new) | ~400 | the report document: vocabulary, render, write, merge, fold, CLI |
+| `helper/hil_health.py` | ~345 (−53) | killing wedged processes only |
+| `helper/hil_summary.py` | deleted | superseded by `hil_report.py` |
+
+Import graph: `hil_health` is a leaf; `hil_report` → `hil_health` (for `_p`, the
+BrokenPipeError-safe print used on containment paths); `hil_test` → both. No cycles.
+
+## hil_report.py
+
+Stdlib only (`json`, `argparse`, `pathlib`) beyond that one `_p` import. Sections, in order:
+
+**Vocabulary.** `REPORT_MD`, `REPORT_JSON`, `REPORT_CELL`, `BOUNDARY_CELL`, `LOCKED_CELL`.
+`REPORT_CELL` becomes the single source of the status icons; `hil_summary.py`'s `FAIL_ICON`/
+`SKIP_ICON` literals are deleted.
+
+**Classifier.** One `cell_state(v) -> 'pass' | 'fail' | 'skip'`, replacing both `cell_kind` and the
+old `cell_state`. Keeps the surviving docstring's warning that the `pass` arm is load-bearing: a
+passing test may return an unprefixed metric string (`'480.0 MBps'`), while failures are guaranteed
+icon-marked, so classifying unknown shapes as `fail` would publish a green table as a red verdict.
+
+**Render.** `render_matrix(rows_all)`, `render_report(doc)`. Unchanged; `render_matrix`'s inline
+`cell_kind` is replaced by a call to the module-level `cell_state`.
+
+**Write.** `write_report`, `accumulate_report`, `mark_report_abandoned`, `write_timeout_report`.
+Moved verbatim except `write_timeout_report`, which loses its `md_name` parameter (the module owns
+`REPORT_MD`) and renders instead of concatenating.
+
+**Fold.** `variants_of`, `summarize`, and the `main()` CLI from `hil_summary.py`.
+
+## Deliberate wart
+
+`accumulate_report` moves wholesale, keeping its knowledge of `mret`'s worker-result tuple shape.
+The cleaner boundary would split "fold `mret` → rows" (`hil_test`'s domain) from "merge rows → doc"
+(`hil_report`'s), but that rewrites subtle, well-tested logic — stale `board-locked` clearing,
+`BOUNDARY_CELL` dropping, `duration=None` preservation — for a tidier seam. It is a data-shape
+coupling, not an import cycle. Moving it verbatim keeps the motion reviewable as motion.
+
+## The sharp edge
+
+`hil_ci.sh:222-228` stages helper modules by an **explicit scp list**. A new `helper/hil_report.py`
+that is not added there reaches the rig missing, and the run dies with `ImportError` *after*
+`REMOTE_DIR` has already been wiped — so the previous run's report and re-run spec are gone too.
+
+This is already guarded: `test_hil_bounded.py`'s `RemoteStaging.test_import_closure_is_staged_to_the_rig`
+walks the AST import closure from `hil_test.py`, `usbtest.py` and `mtp_test.py` and requires an exact
+scp entry for each file. Adding the module to the list is all this change needs; no new guard is
+warranted, and an earlier draft of this document wrongly claimed none existed.
+
+## Consumers to update
+
+| File | Change |
+|---|---|
+| `test/hil/hil_ci.sh:226` | `hil_summary.py` → `hil_report.py` in the scp list |
+| `.claude/agents/hil-operator.md:71` | the documented command |
+| `.claude/workflows/hil-validate.js:58` | the command the operator is told to run |
+| `.claude/workflows/hil-validate.js:14,17,54,67`, `test-hil-validate.mjs:7` | stale `hil_summary.py` mentions in comments |
+
+No logic in the `.claude` files changes — the operator's return contract
+(`{results, banner, wedged}`) is untouched.
+
+## Testing
+
+New `test/hil/test/test_hil_report.py`. The report-specific classes move there from
+`test_hil_bounded.py` (`CaveatSurvivesAccumulate`, `SummaryFoldsReportToBoards`,
+`ScopeSurvivesInTheJson`, `RenderReportIsPureFunctionOfTheDocument`,
+`EveryExitPathLeavesBothArtifacts`, `AbandonNoticeLandsInBothArtifacts`,
+`MarkdownIsAlwaysARenderingOfTheJson`) and from `test_hil_health.py` (`WriteTimeoutReport`).
+
+Three test changes are substantive rather than mechanical:
+
+1. `WriteTimeoutReport.test_keeps_a_previous_attempts_table` asserts the prior **markdown text**
+ survives. It becomes an assertion that the prior attempt's **rows** survive — the same guarantee
+ against the new representation.
+2. `MarkdownIsAlwaysARenderingOfTheJson` gains a fifth case for the pool-guard fallback, which now
+ satisfies the byte-identical invariant like the other four.
+3. `test_the_pool_guard_fallback_agrees_even_if_it_does_not_render` — the weaker promise — is
+ deleted, because the promise it encoded no longer applies.
+
+Gate: `python3 -m unittest discover -s test/hil/test` at 275 — the current 266, minus the one
+deleted test, plus the fifth invariant case, the scp-list guard, two dual-mode import tests,
+five classifier tests and one pinning that the old entry point is gone — then
+`pre-commit run --all-files`. Because this lands on a
+branch already validated on hardware, it closes with a rig re-check: the invariant check against a
+real report pair and a scoped `--accumulate` run, not the full fleet.
diff --git a/examples/device/audio_4_channel_mic/skip.txt b/examples/device/audio_4_channel_mic/skip.txt
index 3ca433c08..e5e74cd60 100644
--- a/examples/device/audio_4_channel_mic/skip.txt
+++ b/examples/device/audio_4_channel_mic/skip.txt
@@ -1,5 +1,4 @@
mcu:SAMD11
-mcu:SAME5X
mcu:SAMG
family:broadcom_64bit
family:espressif
diff --git a/examples/device/audio_4_channel_mic_freertos/skip.txt b/examples/device/audio_4_channel_mic_freertos/skip.txt
index 1fd6b4b8a..cfde51051 100644
--- a/examples/device/audio_4_channel_mic_freertos/skip.txt
+++ b/examples/device/audio_4_channel_mic_freertos/skip.txt
@@ -7,7 +7,6 @@ mcu:CXD56
mcu:F1C100S
mcu:GD32VF103
mcu:MCXA15
-mcu:MKL25ZXX
mcu:MSP430x5xx
mcu:FT90X
mcu:SAMD11
diff --git a/examples/device/audio_test/skip.txt b/examples/device/audio_test/skip.txt
index 42394bb11..862c91c6f 100644
--- a/examples/device/audio_test/skip.txt
+++ b/examples/device/audio_test/skip.txt
@@ -1,5 +1,4 @@
mcu:SAMD11
-mcu:SAME5X
mcu:SAMG
family:espressif
mcu:CH583
diff --git a/examples/device/audio_test_freertos/skip.txt b/examples/device/audio_test_freertos/skip.txt
index 660bacd25..3d8d43286 100644
--- a/examples/device/audio_test_freertos/skip.txt
+++ b/examples/device/audio_test_freertos/skip.txt
@@ -7,7 +7,6 @@ mcu:CXD56
mcu:F1C100S
mcu:GD32VF103
mcu:MCXA15
-mcu:MKL25ZXX
mcu:MSP430x5xx
mcu:FT90X
mcu:SAMD11
diff --git a/examples/device/audio_test_multi_rate/skip.txt b/examples/device/audio_test_multi_rate/skip.txt
index 42394bb11..862c91c6f 100644
--- a/examples/device/audio_test_multi_rate/skip.txt
+++ b/examples/device/audio_test_multi_rate/skip.txt
@@ -1,5 +1,4 @@
mcu:SAMD11
-mcu:SAME5X
mcu:SAMG
family:espressif
mcu:CH583
diff --git a/examples/device/cdc_msc_freertos/skip.txt b/examples/device/cdc_msc_freertos/skip.txt
index 48781de84..095e350c9 100644
--- a/examples/device/cdc_msc_freertos/skip.txt
+++ b/examples/device/cdc_msc_freertos/skip.txt
@@ -7,7 +7,6 @@ mcu:CXD56
mcu:F1C100S
mcu:GD32VF103
mcu:MCXA15
-mcu:MKL25ZXX
mcu:MSP430x5xx
mcu:FT90X
mcu:SAMD11
diff --git a/examples/device/cdc_uac2/skip.txt b/examples/device/cdc_uac2/skip.txt
index db1d5b80b..3159cb176 100644
--- a/examples/device/cdc_uac2/skip.txt
+++ b/examples/device/cdc_uac2/skip.txt
@@ -2,7 +2,6 @@ mcu:LPC11UXX
mcu:LPC13XX
mcu:NUC121
mcu:SAMD11
-mcu:SAME5X
mcu:SAMG
board:stm32l052dap52
family:espressif
diff --git a/examples/device/hid_composite_freertos/skip.txt b/examples/device/hid_composite_freertos/skip.txt
index 97d8e168b..0e8415d3b 100644
--- a/examples/device/hid_composite_freertos/skip.txt
+++ b/examples/device/hid_composite_freertos/skip.txt
@@ -7,7 +7,6 @@ mcu:CXD56
mcu:F1C100S
mcu:GD32VF103
mcu:MCXA15
-mcu:MKL25ZXX
mcu:MSP430x5xx
mcu:FT90X
mcu:SAMD11
diff --git a/examples/device/midi_test_freertos/skip.txt b/examples/device/midi_test_freertos/skip.txt
index 97d8e168b..0e8415d3b 100644
--- a/examples/device/midi_test_freertos/skip.txt
+++ b/examples/device/midi_test_freertos/skip.txt
@@ -7,7 +7,6 @@ mcu:CXD56
mcu:F1C100S
mcu:GD32VF103
mcu:MCXA15
-mcu:MKL25ZXX
mcu:MSP430x5xx
mcu:FT90X
mcu:SAMD11
diff --git a/examples/device/msc_dual_lun/skip.txt b/examples/device/msc_dual_lun/skip.txt
index a9e3a99b1..833fd072c 100644
--- a/examples/device/msc_dual_lun/skip.txt
+++ b/examples/device/msc_dual_lun/skip.txt
@@ -1,3 +1,2 @@
mcu:SAMD11
-mcu:MKL25ZXX
family:espressif
diff --git a/examples/device/uac2_headset/skip.txt b/examples/device/uac2_headset/skip.txt
index db1d5b80b..3159cb176 100644
--- a/examples/device/uac2_headset/skip.txt
+++ b/examples/device/uac2_headset/skip.txt
@@ -2,7 +2,6 @@ mcu:LPC11UXX
mcu:LPC13XX
mcu:NUC121
mcu:SAMD11
-mcu:SAME5X
mcu:SAMG
board:stm32l052dap52
family:espressif
diff --git a/examples/device/uac2_speaker_fb/skip.txt b/examples/device/uac2_speaker_fb/skip.txt
index 0c7339c65..88df3e549 100644
--- a/examples/device/uac2_speaker_fb/skip.txt
+++ b/examples/device/uac2_speaker_fb/skip.txt
@@ -2,7 +2,6 @@ mcu:LPC11UXX
mcu:LPC13XX
mcu:NUC121
mcu:SAMD11
-mcu:SAME5X
mcu:SAMG
board:stm32l052dap52
family:broadcom_64bit
diff --git a/test/hil/helper/hil_health.py b/test/hil/helper/hil_health.py
index 92f0accc8..b9c05c236 100644
--- a/test/hil/helper/hil_health.py
+++ b/test/hil/helper/hil_health.py
@@ -1,6 +1,6 @@
#!/usr/bin/env python3
# SPDX-License-Identifier: MIT
-"""Shutting a wedged HIL run down: kill what the workers spawned, then report.
+"""Shutting a wedged HIL run down: kill what the workers spawned.
A device whose usbfs node is held by a D-state process cannot be freed -- SIGKILL is not
delivered in uninterruptible sleep -- so the goal is never to fix the rig from here. It is
@@ -214,9 +214,6 @@ def _kill_kids(kids: dict, seen: set) -> int:
own = os.getpgid(0)
except OSError:
own = None # cannot tell our own group apart: never killpg, signal pids only
- # One list: every pid here is a DESCENDANT of one of our own workers, so it is ours by
- # construction -- no argv identity check needed, because we never signal anything we
- # did not discover through our own ppid tree.
touched: list = []
for children in kids.values():
for cpid, cpgid in children:
@@ -341,40 +338,3 @@ def kill_pool_children(pool, *extra) -> int:
# SIGKILL is asynchronous and a D-state task ignores it: only a confirmed survivor
# justifies the caller's power-cycle wording
return len(_kill_and_confirm(killed_pids)) if killed_pids else 0
-
-
-def write_timeout_report(report_dir: Path, boards, secs: int, md_name: str,
- banner: str = '', prefix: str = '') -> None:
- """Leave a report behind when the worker pool has to be abandoned.
-
- map_async is all-or-nothing, so a timeout loses every per-board result and the report
- dir would stay empty with no reason for the failure. Any prior attempt's markdown is
- kept below the banner."""
- # `prefix` carries the preflight rig-health verdict: the timeout aborts before
- # accumulate_report, so without it the report loses the one line saying WHY the pool
- # never finished. The '\n' stops Markdown lazy continuation pulling the banner into
- # the blockquote.
- try:
- # Built INSIDE the try: a roster entry without a 'name' key raises KeyError while
- # assembling the board list, and outside the try that escaped and stranded the
- # runner -- which is exactly what the broad handler below exists to prevent.
- head = (prefix + '\n' if prefix else '') + (banner or (
- f'**HIL run abandoned: worker pool timed out after {secs}s.**\n\n'
- f'No per-board results could be collected for this attempt, so the '
- f'table below (if any) is from an earlier one. Boards dispatched:\n\n'
- + '\n'.join(f'- {b.get("name", "?")}' for b in boards) + '\n'))
- report_dir.mkdir(parents=True, exist_ok=True)
- md_path = report_dir / md_name
- # Its own handler so it cannot take the write down with it: a report torn by an
- # attempt killed mid-write raises UnicodeDecodeError (a ValueError, and prior
- # reports always contain status emoji), which under a shared try skipped the write
- # entirely. Losing the old table is a nicety; losing the banner is the failure.
- try:
- prior = md_path.read_text(encoding='utf-8') if md_path.is_file() else ''
- except (OSError, ValueError):
- prior = ''
- md_path.write_text(head + (f'\n{prior}' if prior else ''), encoding='utf-8')
- except Exception as e: # noqa: BLE001
- # Deliberately broad: this is the first statement of the pool-abandon path, so ANY
- # escape skips kill_pool_children and os._exit and strands the runner.
- _p(f'warning: cannot write {md_name} to {report_dir}: {e}', flush=True)
diff --git a/test/hil/helper/hil_pool_check.py b/test/hil/helper/hil_pool_check.py
index 179a417ed..b92f0aee0 100644
--- a/test/hil/helper/hil_pool_check.py
+++ b/test/hil/helper/hil_pool_check.py
@@ -983,9 +983,13 @@ def main() -> None:
headers = ['Board', 'Probe', 'Flash', 'Device', 'Status', 'Note']
cells = [[r['name'], r['probe'], r['flash'], r['device'],
status_mark.get(r['status'], r['status']), '; '.join(r['note'])] for r in rows]
- widths = [max(len(h), *(len(c[i]) for c in cells)) if cells else len(h)
+ # display_width, not len(): ✅ / ❌ / 🔒 / ⚠ are one character and two columns, so
+ # len() pads every row holding one a column short of the header rule
+ _w = hil_util.display_width
+ widths = [max(_w(h), *(_w(c[i]) for c in cells)) if cells else _w(h)
for i, h in enumerate(headers)]
- line = lambda vals: '| ' + ' | '.join(v.ljust(w) for v, w in zip(vals, widths)) + ' |'
+ line = lambda vals: ('| ' + ' | '.join(hil_util.pad(v, w)
+ for v, w in zip(vals, widths)) + ' |')
print()
print(line(headers))
print('|' + '|'.join('-' * (w + 2) for w in widths) + '|')
diff --git a/test/hil/helper/hil_report.py b/test/hil/helper/hil_report.py
new file mode 100644
index 000000000..d059c62c9
--- /dev/null
+++ b/test/hil/helper/hil_report.py
@@ -0,0 +1,568 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: MIT
+"""The HIL report document: one owner for hil_report.json and hil_report.md.
+
+The markdown IS a rendering of the sidecar -- every writer goes through render_report(), so
+a table can never contain something the JSON does not. This module owns the whole life of
+that document: the cell vocabulary, the one classifier both artifacts share, rendering, the
+writers, and the fold to one machine-readable verdict per board.
+
+Dual-mode by design: imported as `helper.hil_report` by hil_test.py, and run as a script by
+the operator (see .claude/agents/hil-operator.md). A script run puts test/hil/helper on
+sys.path rather than test/hil, so this module imports no sibling helper at all --
+_p and the width helpers below are defined locally for that reason.
+"""
+import argparse
+import json
+import sys
+import unicodedata
+from pathlib import Path
+
+
+def _w(s: str) -> int:
+ """Terminal COLUMNS, not characters. Every status mark in REPORT_CELL is one Python
+ character and TWO columns wide, so len() pads a cell holding one a column short and
+ the pipes drift out of line with the header rule for the whole table.
+
+ Local, like _p above and for the same reason: this module is also run as a script, and
+ under PYTHONSAFEPATH=1 a sibling import dies before argparse runs. hil_util carries the
+ same pair for callers that can import it.
+ """
+ return sum(2 if unicodedata.east_asian_width(c) in 'WF' else 1 for c in s)
+
+
+def _pad(s: str, width: int, center: bool = False) -> str:
+ """str.ljust/center, measured in display columns. See _w."""
+ room = max(0, width - _w(s))
+ if not center:
+ return s + ' ' * room
+ left = room // 2
+ return ' ' * left + s + ' ' * (room - left)
+
+
+def _p(*args, **kwargs) -> None:
+ """Print that cannot raise. Defined here rather than imported from hil_health: this
+ module is ALSO run as a script (hil-operator.md invokes it by path), and under
+ PYTHONSAFEPATH=1 -- which the suite's own MTP fixtures set -- sys.path[0] is not the
+ script dir, so any sibling import dies before argparse runs. Five lines beat that."""
+ try:
+ print(*args, **kwargs)
+ except (OSError, ValueError):
+ # ValueError too: printing to a CLOSED stream raises "I/O operation on closed
+ # file", and escaping here skips the containment path's os._exit.
+ pass
+
+REPORT_MD = 'hil_report.md'
+REPORT_JSON = 'hil_report.json'
+# The status vocabulary, shared by the code that WRITES a cell (hil_test's test runners) and
+# the code that reads one back (cell_state). One dict, so the human's table and the agent's
+# verdict cannot drift apart.
+REPORT_CELL = {'pass': '✅', 'fail': '❌', 'skip': '⚪'}
+BOUNDARY_CELL = 'same-PID boundary'
+LOCKED_CELL = 'board-locked'
+# A pseudo-test column, not a real one: write_timeout_report marks the boards that were
+# still dispatched when the pool guard fired. accumulate_report clears it on a retry.
+POOL_TIMEOUT_CELL = 'pool-timeout'
+
+
+def _load(report_dir: Path) -> tuple:
+ """(doc, readable) for the sidecar, coerced to the canonical shape.
+
+ hil_ci.sh uploads a sidecar as the --accumulate merge base, so a non-conforming one is
+ reachable from OUTSIDE the harness -- and every writer here runs on a path where a
+ TypeError costs the whole report. Coerce once, at the boundary, instead of guarding
+ each use: `banner: null` used to kill a fully successful run with a traceback and no
+ artifact at all, and `cells: null` sent write_timeout_report down its fallback so a
+ board that ate the whole pool guard was published as a pass.
+
+ `readable` is False only when a sidecar EXISTS but could not be parsed, or is absent --
+ both mean its rows are unrecoverable, which callers use to avoid destroying a markdown
+ that may still hold them."""
+ jpath = report_dir / REPORT_JSON
+ if not jpath.is_file():
+ return {'rows': [], 'banner': '', 'scope': '', 'caveat': ''}, False
+ try:
+ raw = json.loads(jpath.read_text())
+ if not isinstance(raw, dict):
+ raise ValueError('sidecar is not an object')
+ except (OSError, ValueError, TypeError):
+ return {'rows': [], 'banner': '', 'scope': '', 'caveat': ''}, False
+ rows = []
+ # isinstance, not `or []`: a sidecar with `rows: 1` iterates an int and raises outside
+ # the parse handler above.
+ for r in (raw.get('rows') if isinstance(raw.get('rows'), list) else []):
+ if not isinstance(r, dict) or 'board' not in r:
+ continue
+ cells = r.get('cells')
+ dur = r.get('duration')
+ # VALUES as well as keys: render_matrix does REPORT_CELL.get(v, v), which raises
+ # TypeError on an unhashable value, and cell_state does v.startswith. A non-str
+ # cell is corrupt, and dropping it renders blank -- "not run" -- which is the
+ # honest reading. Coercing it to str would make it classify as a PASS.
+ rows.append({'board': str(r['board']),
+ 'cells': {str(k): v for k, v in cells.items() if isinstance(v, str)}
+ if isinstance(cells, dict) else {},
+ 'duration': dur if isinstance(dur, str) else None})
+ text = lambda k: raw[k] if isinstance(raw.get(k), str) else ''
+ return {'rows': rows, 'banner': text('banner'), 'scope': text('scope'),
+ 'caveat': text('caveat')}, True
+
+
+def cell_state(v) -> str:
+ """'pass' | 'fail' | 'skip' for one report cell.
+
+ THE classifier -- the markdown tally and the per-board verdict both call this, so they
+ cannot disagree. 'fail' or a fail-icon prefix is a failure, 'skip' or a skip-icon prefix
+ is a skip, and EVERYTHING ELSE is a pass. That last arm is load-bearing: a passing test
+ may return a plain metric string ('480.0 MBps') that lands in the cell unprefixed, while
+ failures are guaranteed marked -- TestFail's docstring pins that its metric is
+ icon-prefixed precisely so render and tally treat it as a failure. Classifying unknown
+ shapes as fail here would publish a green table as a red verdict.
+
+ isinstance-guarded: cells are usually str but a caller may hand over None or a number,
+ and .startswith on those raises inside a report writer that must not raise."""
+ if v == 'fail' or (isinstance(v, str) and v.startswith(REPORT_CELL['fail'])):
+ return 'fail'
+ if v == 'skip' or (isinstance(v, str) and v.startswith(REPORT_CELL['skip'])):
+ return 'skip'
+ return 'pass'
+
+
+def render_matrix(rows_all: list) -> str:
+ """Render rows (list of (row_label, {example: status}, duration)) as an aligned
+ markdown matrix: columns = tests (bare names) centered, boards left-aligned,
+ per-row duration as the trailing column."""
+ seen = set()
+ for _, cells, _ in rows_all:
+ seen.update(cells)
+ if not seen:
+ return 'No tests were run.'
+
+ # metric-bearing columns pinned first, the rest alphabetical: stable regardless of the
+ # shuffled execution order
+ pinned = ['usbtest', 'cdc_msc_throughput', 'msc_file_explorer', 'msc_file_explorer_freertos']
+
+ def col_key(t):
+ name = t.rsplit('/', 1)[-1]
+ return (pinned.index(name) if name in pinned else len(pinned), name, t)
+
+ columns = sorted(seen, key=col_key)
+ headers = [c.rsplit('/', 1)[-1] for c in columns] + ['duration'] # bare example names
+
+ def cell(cells, col):
+ v = cells.get(col)
+ if v is None:
+ return ''
+ return REPORT_CELL.get(v, v) # status symbol, or a metric string (e.g. speed) verbatim
+
+ rows_vals = [(lbl, [cell(cells, c) for c in columns] + [dur or ''])
+ for lbl, cells, dur in rows_all]
+ board_hdr = 'Board'
+ # display_width, not len(): the ✅/❌/⚪ marks are one character and two columns
+ board_w = max([_w(board_hdr)] + [_w(lbl) for lbl, _ in rows_vals])
+ col_w = [max([_w(h)] + [_w(vals[i]) for _, vals in rows_vals])
+ for i, h in enumerate(headers)]
+
+ def line(label, values):
+ padded = [_pad(label, board_w)] + [_pad(v, w, center=True)
+ for v, w in zip(values, col_w)]
+ return '| ' + ' | '.join(padded) + ' |'
+
+ header = line(board_hdr, headers)
+ sep = '| ' + '-' * board_w + ' | ' + ' | '.join(':' + '-' * (w - 2) + ':' for w in col_w) + ' |'
+ body = [line(lbl, vals) for lbl, vals in rows_vals]
+
+ # tally run cells (not-run cells are absent from the dicts). A cell is a bare status or
+ # a metric string carrying its own icon ("❌ 29/30"), so classify by the leading icon --
+ # through cell_state, the same call the per-board verdict makes.
+ kinds = [cell_state(v) for _, cells, _ in rows_all for v in cells.values()]
+ failed = kinds.count('fail')
+ skipped = kinds.count('skip')
+ passed = kinds.count('pass')
+ summary = (f'**{REPORT_CELL["pass"]} {passed} passed · {REPORT_CELL["fail"]} {failed} failed · '
+ f'{REPORT_CELL["skip"]} {skipped} skipped · blank not run**')
+
+ return summary + '\n\n' + '\n'.join([header, sep] + body)
+
+
+def render_report(doc: dict) -> str:
+ """The markdown IS a rendering of the sidecar. Every writer goes through here, so a
+ table can never contain something the JSON does not."""
+ # .get throughout, not subscripts: mark_report_abandoned renders a sidecar it did NOT
+ # write (hil_ci.sh reuses a persistent REMOTE_DIR, so it may be an older version's or
+ # a torn one) on the way to os._exit, and a KeyError there is not in its handler --
+ # it would unwind into multiprocessing's unbounded join and hang the runner it is
+ # trying to free. Same reason summarize() below reads cells as `r.get('cells') or {}`.
+ md = render_matrix([(r.get('board', '?'), r.get('cells') or {}, r.get('duration'))
+ for r in doc.get('rows') or [] if isinstance(r, dict)])
+ if doc.get('scope'):
+ # a scoped run's small table is otherwise indistinguishable from a full one, and
+ # it replaces the previous full table in the sticky PR comment
+ md = f'_Scoped run: {doc["scope"]}. Boards/tests not listed were not run._\n\n' + md
+ # banner, then caveat: a rig-health caveat outranks the table AND the scope note, and an
+ # abandon notice outranks even that -- the top of the report is where hil/SKILL.md tells
+ # the agent to look
+ if doc.get('banner'):
+ md = doc['banner'] + '\n' + md
+ if doc.get('caveat'):
+ md = doc['caveat'] + '\n' + md
+ return md
+
+
+def write_report(report_dir: Path, doc: dict) -> None:
+ """Write both artifacts from one document.
+
+ RAISES on failure, deliberately: every caller is on a path whose own handler exists to
+ report exactly this (write_timeout_report's _p warning, hil_test's fallback-of-the-
+ fallback). Swallowing OSError here made both of those dead code, so an unwritable or
+ root-owned report dir produced no artifact AND no message.
+
+ Renders BEFORE writing anything: committing the JSON first and then raising in
+ render_report left a sidecar saying "abandoned" beside a markdown still reading as a
+ clean green table -- the one invariant this module exists to hold."""
+ md = render_report(doc) + '\n'
+ report_dir.mkdir(parents=True, exist_ok=True)
+ (report_dir / REPORT_JSON).write_text(json.dumps(doc, indent=2) + '\n')
+ (report_dir / REPORT_MD).write_text(md, encoding='utf-8')
+
+
+def _abandon_notice(why: str) -> str:
+ # Wording is a CONTRACT: .claude/skills/hil/SKILL.md pins this banner as the case where
+ # "the table below IS this run's ... Report the results AND the abandonment". Calling
+ # the table partial would send the reading agent to re-run boards that already passed.
+ return (f'**HIL run abandoned: {why}** The table below was collected before the '
+ f'abandon; treat board results as unverified.\n')
+
+
+def _already_abandoned(doc: dict) -> bool:
+ """Whether THIS attempt already recorded how it ended.
+
+ `caveat` only. It used to check `banner` too, because hil_test.py folded its abandon
+ notices in there -- but banner is carried across an --accumulate retry by design, so a
+ stale notice from an earlier attempt silenced a genuinely new abandon and the run's own
+ failure went unrecorded. banner now carries rig HEALTH (which describes the conditions
+ the cells were collected under, and so must persist); caveat carries the run's OUTCOME
+ (which must not)."""
+ return '**HIL run ab' in doc.get('caveat', '')
+
+
+def _stamp_markdown(report_dir: Path, notice: str) -> None:
+ """Last line of defence: prepend the notice to the markdown itself.
+
+ pr_comment.yml cats only hil_report.md, so a path that gives up here publishes a clean
+ green table under an abandoned, non-zero job. Master did this unconditionally."""
+ mpath = report_dir / REPORT_MD
+ if not mpath.is_file():
+ return
+ # errors='replace' and catch ValueError: a torn report or a LANG=C locale raises
+ # UnicodeDecodeError -- NOT an OSError -- straight past os._exit.
+ body = mpath.read_text(encoding='utf-8', errors='replace')
+ if '**HIL run ab' not in body[:2000]:
+ mpath.write_text(notice + '\n' + body, encoding='utf-8')
+
+
+def mark_report_abandoned(report_dir: Path, why: str) -> None:
+ """Stamp an existing report as abandoned, in BOTH artifacts.
+
+ Best-effort and silent: this runs while the interpreter is being torn down, and an
+ exception here hangs the process in multiprocessing's unbounded join()."""
+ notice = _abandon_notice(why)
+ try:
+ doc, readable = _load(report_dir)
+ if readable:
+ if _already_abandoned(doc):
+ return # whoever got there first wins, WRITE included
+ doc['caveat'] = notice
+ write_report(report_dir, doc)
+ return
+ except (OSError, ValueError, TypeError, AttributeError):
+ pass # fall through -- a failure here must not cost the stamp entirely
+ # Unreadable sidecar, or the document write failed. Either way the markdown is what
+ # the PR comment reads, so stamp it directly rather than giving up.
+ try:
+ _stamp_markdown(report_dir, notice)
+ except (OSError, ValueError, TypeError, AttributeError):
+ pass
+
+
+def mark_report_no_boards(report_dir: Path, msg: str, fresh: bool = True) -> None:
+ """Record that the board filters intersected to nothing.
+
+ `fresh` mirrors hil_test's own flag, because this runs BEFORE the fresh wipe: without
+ it a fresh run whose filter emptied re-published the PREVIOUS run's green rows under
+ this run's red job -- the stale-table failure it exists to prevent. An --accumulate run
+ keeps them, since nothing this attempt did invalidates them."""
+ try:
+ doc, _ = _load(report_dir)
+ if not fresh and _already_abandoned(doc):
+ # SKILL.md gives the two notices OPPOSITE rules, and an abandon outranks a
+ # filter that matched nothing -- do not overwrite the record of a failed run.
+ # Only while ACCUMULATING, though: this runs before the fresh wipe, so guarding
+ # a fresh run would leave the previous attempt's rows AND its abandon notice
+ # published as this run's.
+ return
+ # A fresh run carries NOTHING from the prior sidecar -- rows, banner and scope
+ # alike, matching accumulate_report, which builds from an empty prior when fresh.
+ # Resetting only rows republished a stale rig-health note and a stale scope line
+ # under this run's notice, from a leftover or uploaded sidecar.
+ prior = {'rows': [], 'banner': '', 'scope': ''} if fresh else doc
+ write_report(report_dir, {'rows': prior['rows'], 'banner': prior['banner'],
+ 'scope': prior['scope'],
+ 'caveat': f'**HIL run selected no boards.** {msg}\n'})
+ except (OSError, ValueError, TypeError, AttributeError):
+ pass # loud on stdout already; the exit code is what the job reads
+
+
+def accumulate_report(mret: list, report_dir: Path, fresh: bool, scope: str = '',
+ banner: str = '', caveat: str = '') -> str:
+ """Merge this run's results into json in report_dir, then (re)write
+ the markdown matrix to md. `fresh` (a first run, no --accumulate)
+ starts a new report; otherwise a re-run accumulates so boards/tests that
+ already passed are preserved while re-run cells are updated. `scope` names the
+ board filter, if any, so a scoped table is not mistaken for a full one.
+ Returns the md.
+
+ `mret` is hil_test.py's worker-result shape (name, err, fts, rows, ...), so this one
+ function knows something about its caller that the rest of the module does not. Folding
+ mret into rows could live in hil_test and only the merge here, but that would rewrite
+ the subtle parts -- stale board-locked clearing, BOUNDARY_CELL dropping, duration=None
+ preservation -- for a tidier seam. Data-shape coupling, not an import cycle."""
+ # ONE canonical load: a sidecar reaching here may have been uploaded by hil_ci.sh as
+ # the merge base, so it is untrusted input. `banner` carries forward -- it describes
+ # the conditions the earlier cells were collected under, and the .failed spec re-runs
+ # only FAILURES so those passes are never re-earned. `caveat` does NOT: it records how
+ # a RUN ENDED, and this attempt has not ended yet. Carrying it made a clean retry
+ # publish "HIL run abandoned" over a run where nothing was abandoned.
+ prior = {'rows': [], 'banner': ''}
+ if not fresh:
+ prior, _ = _load(report_dir)
+ acc = {r['board']: [dict(r['cells']), r['duration']] for r in prior['rows']}
+ prior_banner = prior['banner']
+
+ # current cells override prior for boards/tests that ran; a filtered run reports
+ # duration None, keeping the previous full-run value
+ for name, _, _, rows, *_ in mret:
+ if rows and not any(LOCKED_CELL in cells for _, cells, _ in rows):
+ # board ran for real: clear a stale lock-failure cell (its row is keyed by
+ # board name; test rows may be variant names)
+ stale = acc.get(name)
+ if stale is not None:
+ stale[0].pop(LOCKED_CELL, None)
+ # and the pool-timeout mark: write_timeout_report stamps it on a board that
+ # never reported, and update() below MERGES, so without this a board that
+ # passed clean on the retry kept a red cell for ever.
+ stale[0].pop(POOL_TIMEOUT_CELL, None)
+ if not stale[0]:
+ # variant-keyed boards never repopulate the board-name row, so drop it
+ # or it renders as a blank ghost row
+ del acc[name]
+ for row_label, cells, dur in rows:
+ row = acc.setdefault(row_label, [{}, None])
+ # a row that ran is no longer pool-timed-out, whatever it is keyed by
+ row[0].pop(POOL_TIMEOUT_CELL, None)
+ # the boundary cell is only ever written on failure, so a re-run of this
+ # variant that cleared the boundary must drop the previous attempt's ❌
+ if BOUNDARY_CELL not in cells:
+ row[0].pop(BOUNDARY_CELL, None)
+ row[0].update(cells)
+ if dur is not None:
+ row[1] = dur
+
+ report_dir.mkdir(parents=True, exist_ok=True)
+ # by LINE, deduped: attempts repeat the same caveat far more often than they add a new
+ # one, and three copies of the D-state note reads as three incidents
+ seen, merged = set(), []
+ for line in (prior_banner + banner).splitlines():
+ if line.strip() and line not in seen:
+ seen.add(line)
+ merged.append(line)
+ banner = '\n'.join(merged) + '\n' if merged else ''
+ doc = {'rows': [{'board': k, 'cells': c, 'duration': d} for k, (c, d) in acc.items()],
+ 'banner': banner, 'scope': scope, 'caveat': caveat}
+ # through write_report, not hand-rolled: writing the JSON and only then rendering is
+ # the ordering write_report exists to forbid -- a render failure left the sidecar ahead
+ # of the markdown, which is the one invariant this module holds.
+ write_report(report_dir, doc)
+ return render_report(doc)
+
+
+def _write_stuck_over_prior_md(report_dir: Path, doc: dict) -> None:
+ """Sidecar unrecoverable: rebuild it from the stuck rows alone, but leave the
+ markdown's existing table beneath the caveat rather than throwing real results away.
+
+ The one place the md-is-a-rendering-of-the-json invariant is deliberately suspended,
+ because there is no readable json left for it to be a rendering of."""
+ try:
+ prior = (report_dir / REPORT_MD).read_text(encoding='utf-8')
+ except (OSError, ValueError):
+ prior = ''
+ # Say so explicitly: those rows exist only as rendered text, so no later --accumulate
+ # can merge them back. Claiming the sidecar represents them would be false.
+ note = ('_The table below is a previous attempt\'s rendered output. The sidecar could '
+ 'not be read, so those rows are NOT in it and will not survive another run._\n')
+ head = (doc['banner'] + '\n' if doc['banner'] else '') + doc['caveat'] + '\n' + note
+ body = prior if prior.strip() else render_matrix(
+ [(r['board'], r['cells'], r['duration']) for r in doc['rows']])
+ report_dir.mkdir(parents=True, exist_ok=True)
+ (report_dir / REPORT_JSON).write_text(json.dumps(doc, indent=2) + '\n')
+ (report_dir / REPORT_MD).write_text(head + '\n' + body, encoding='utf-8')
+
+
+def write_timeout_report(report_dir: Path, boards, secs: int,
+ banner: str = '', prefix: str = '') -> None:
+ """Leave a report behind when the worker pool has to be abandoned.
+
+ map_async is all-or-nothing, so a timeout loses every per-board result and the report
+ dir would stay empty with no reason for the failure. Any prior attempt's rows are kept
+ and each stuck board is marked with a POOL_TIMEOUT_CELL beside them.
+
+ `prefix` is the preflight rig-health verdict and goes to the BANNER, where rig health
+ lives and where an --accumulate retry carries it forward; the abandon notice goes to
+ the caveat, which does not carry. Folding both into the caveat is what made a clean
+ retry report an abandonment that had not happened."""
+ try:
+ # names INSIDE the try: a roster entry that is not a dict raises here, and outside
+ # it that escaped and stranded the runner.
+ names = [b.get('name', '?') if isinstance(b, dict) else '?' for b in boards]
+ caveat = banner or (
+ f'**HIL run abandoned: worker pool timed out after {secs}s.**\n\n'
+ f'No per-board results could be collected for this attempt. Rows other than '
+ f'the {POOL_TIMEOUT_CELL} cells below are from an earlier attempt. Boards '
+ f'dispatched:\n\n' + '\n'.join(f'- {n}' for n in names) + '\n')
+ doc, readable = _load(report_dir)
+ rows = doc['rows']
+ by_board = {r['board']: r for r in rows}
+ for name in names:
+ row = by_board.get(name)
+ if row is None:
+ rows.append({'board': name, 'cells': {POOL_TIMEOUT_CELL: 'fail'},
+ 'duration': None})
+ else:
+ # _load guarantees `cells` is a dict, so a null-cells row from an uploaded
+ # sidecar can no longer send this down the fallback and publish a board
+ # that ate the whole pool guard as a pass.
+ row['cells'][POOL_TIMEOUT_CELL] = 'fail'
+ out = {'rows': rows, 'scope': doc['scope'], 'caveat': caveat,
+ 'banner': ((doc['banner'] + prefix) if prefix not in doc['banner']
+ else doc['banner'])}
+ if not readable and (report_dir / REPORT_MD).is_file():
+ # `readable` covers ABSENT as well as torn: an absent sidecar beside an intact
+ # markdown used to re-render from the stuck row alone and destroy real results.
+ _write_stuck_over_prior_md(report_dir, out)
+ return
+ write_report(report_dir, out)
+ except Exception as e: # noqa: BLE001
+ # Deliberately broad: this is the first statement of the pool-abandon path, so ANY
+ # escape skips kill_pool_children and os._exit and strands the runner.
+ _p(f'warning: cannot write {REPORT_MD} to {report_dir}: {e}', flush=True)
+ try:
+ # Same wording as above and the same guarded name extraction -- the fallback
+ # used to re-derive b.get("name") outside any try and raise identically, so a
+ # malformed roster left NO artifact at all.
+ names = [b.get('name', '?') if isinstance(b, dict) else '?' for b in boards]
+ head = (prefix + '\n' if prefix else '') + (banner or (
+ f'**HIL run abandoned: worker pool timed out after {secs}s.**\n\n'
+ f'No per-board results could be collected for this attempt, so the table '
+ f'below (if any) is from an earlier one. Boards dispatched:\n\n'
+ + '\n'.join(f'- {n}' for n in names) + '\n'))
+ try:
+ prior = (report_dir / REPORT_MD).read_text(encoding='utf-8')
+ except (OSError, ValueError):
+ prior = ''
+ report_dir.mkdir(parents=True, exist_ok=True)
+ (report_dir / REPORT_MD).write_text(
+ head + (f'\n{prior}' if prior else ''), encoding='utf-8')
+ except Exception as e2: # noqa: BLE001
+ _p(f'warning: fallback {REPORT_MD} write failed too: {e2}', flush=True)
+
+
+def variants_of(cfg: dict, board: str) -> list:
+ for b in cfg.get('boards', []):
+ if b['name'] == board:
+ return [v['name'] for v in (b.get('variant') or [])] or [board]
+ return [board]
+
+
+def summarize(cfg: dict, boards: list, report: dict) -> dict:
+ # .get, not a subscript: this is the one reader an agent's verdict depends on, and a
+ # row without 'board' used to kill the CLI with a traceback and no results at all --
+ # hil-validate.js then reports every board as "hil-operator returned no entry".
+ rows = {r['board']: r.get('cells') or {}
+ for r in (report.get('rows') or [])
+ if isinstance(r, dict) and 'board' in r}
+ owner = {v['name']: b['name'] for b in cfg.get('boards', [])
+ for v in (b.get('variant') or [])}
+ results = []
+ for board in boards:
+ names = variants_of(cfg, board)
+ mine = {n: rows[n] for n in names if n in rows}
+ # a variant name that is neither declared nor prefixed cannot be attributed; the
+ # `<board>-` fallback only helps ad-hoc builds, it is not the primary path. It must
+ # also never steal a row DECLARED by another board: a declared variant need not start
+ # with its own board's name, so it may happen to start with this board's name plus '-'.
+ mine.update({n: c for n, c in rows.items()
+ if n.startswith(f'{board}-') and n not in mine
+ and owner.get(n, board) == board})
+ # the BOARD-name row too: hil_test writes lock contention and pool timeouts keyed
+ # by board name, but variants_of returns only DECLARED variant names -- and
+ # nanoch32v203 / ch32v307v_r1_1v0 declare none equal to their board name. Without
+ # this those rows are invisible, so a lock held by concurrent CI is published as a
+ # hardware FAIL and hil-validate.js never retries it.
+ if board in rows and board not in mine:
+ mine[board] = rows[board]
+ if not mine:
+ results.append({'board': board, 'ran': False, 'pass': False, 'locked': False,
+ 'detail': 'no report row for this board'})
+ continue
+ # a wedge outranks lock contention: `locked` short-circuits `detail` below, so a
+ # stale board-locked cell from an earlier attempt used to mask the pool-timeout
+ # cell the retry added -- publishing a board that hung the rig as LOCKED, which
+ # hil-validate.js then RE-RUNS, paying another pool guard on it.
+ wedged = any(POOL_TIMEOUT_CELL in cells for cells in mine.values())
+ locked = not wedged and any(LOCKED_CELL in cells for cells in mine.values())
+ bad = []
+ for vname, cells in sorted(mine.items()):
+ for test, val in sorted(cells.items()):
+ if test == LOCKED_CELL:
+ continue
+ if cell_state(val) == 'fail':
+ bad.append(f'{vname} {test}: {val}')
+ ok = not bad and not locked
+ if locked:
+ detail = 'held by another holder; not flashed'
+ elif bad:
+ detail = '; '.join(bad)
+ else:
+ detail = f'{len(mine)} variant(s), {sum(len(c) for c in mine.values())} cell(s) ok'
+ results.append({'board': board, 'ran': True, 'pass': ok, 'locked': locked,
+ 'detail': detail})
+ # `caveat` too: an abandoned or no-boards run says so THERE, and this JSON is all
+ # an agent gets -- leaving it in the sidecar puts it back where only a human looks.
+ return {'results': results, 'banner': report.get('banner', ''),
+ 'caveat': report.get('caveat', '')}
+
+
+def main() -> int:
+ ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
+ ap.add_argument('config_file')
+ ap.add_argument('-b', '--board', action='append', default=[],
+ help='boards to report on; default: every board in the config')
+ ap.add_argument('--report-dir', default='.', help=f'where {REPORT_JSON} lives (default: cwd)')
+ a = ap.parse_args()
+
+ cfg = json.loads(Path(a.config_file).read_text())
+ boards = a.board or [b['name'] for b in cfg.get('boards', [])]
+ jpath = Path(a.report_dir) / REPORT_JSON
+ if not jpath.is_file():
+ print(f'error: {jpath} not found -- did hil_test.py run in this directory?',
+ file=sys.stderr)
+ return 1
+ # through _load, like every writer: feeding raw JSON to summarize left the one reader an
+ # agent's verdict depends on crashing on the malformed sidecars the writers tolerate.
+ doc, _ = _load(Path(a.report_dir))
+ json.dump(summarize(cfg, boards, doc), sys.stdout, indent=2)
+ print()
+ return 0
+
+if __name__ == '__main__':
+ sys.exit(main())
diff --git a/test/hil/helper/hil_summary.py b/test/hil/helper/hil_summary.py
deleted file mode 100644
index e566bead0..000000000
--- a/test/hil/helper/hil_summary.py
+++ /dev/null
@@ -1,115 +0,0 @@
-#!/usr/bin/env python3
-# SPDX-License-Identifier: MIT
-"""Fold hil_report.json into one machine-readable verdict per BOARD.
-
-A workflow driving hil_test.py through an operator agent has no filesystem access, so the
-agent has to carry the results across. It must carry them, not retype them: the previous
-design asked the agent to transcribe the markdown table, and every defect found in four
-review rounds came from re-parsing that prose -- variant row names vs board names,
-`board locked` vs `board-locked`, folding several variant rows into one verdict, rows that
-matched no board. All of it is a join, and the join belongs here, where the roster is.
-
-Report rows are named per VARIANT (hil_test.py builds them from `vname`), and a variant name
-is not required to start with the board name -- nanoch32v203 produces only `-fsdev`/`-usbfs`,
-ch32v307v_r1_1v0 only `-usbhs`/`-usbfs`. The config is what maps them back.
-
-Emits, on stdout:
- {"results": [{"board", "ran", "pass", "locked", "detail"}...], "banner": str}
-
-`locked` is a field, not a prefix to grep for. `ran` false means the board produced no row at
-all, which is not the same as failing.
-
-Usage: hil_summary.py <config.json> [-b BOARD]... [--report-dir DIR]
-"""
-import argparse
-import json
-import sys
-from pathlib import Path
-
-FAIL_ICON, SKIP_ICON = '❌', '⚪' # a pass needs no icon: unmarked = pass
-LOCKED_CELL = 'board-locked'
-
-
-def cell_state(v: str) -> str:
- """'pass' | 'fail' | 'skip' -- the EXACT classifier hil_test.py's own tally uses
- (cell_kind in render_matrix): 'fail' or a ❌ prefix is a failure, 'skip' or a ⚪
- prefix is a skip, and EVERYTHING ELSE is a pass. That last arm is load-bearing: a
- passing test may return a plain metric string ('480.0 MBps') that lands in the cell
- unprefixed, while failures are guaranteed marked -- TestFail's docstring pins that its
- metric is icon-prefixed precisely so render/tally treat it as a failure. Classifying
- unknown shapes as fail here would publish a green table as a red verdict."""
- if v == 'fail' or v.startswith(FAIL_ICON):
- return 'fail'
- if v == 'skip' or v.startswith(SKIP_ICON):
- return 'skip'
- return 'pass'
-
-
-def variants_of(cfg: dict, board: str) -> list:
- for b in cfg.get('boards', []):
- if b['name'] == board:
- return [v['name'] for v in (b.get('variant') or [])] or [board]
- return [board]
-
-
-def summarize(cfg: dict, boards: list, report: dict) -> dict:
- rows = {r['board']: r.get('cells') or {} for r in report.get('rows', [])}
- owner = {v['name']: b['name'] for b in cfg.get('boards', [])
- for v in (b.get('variant') or [])}
- results = []
- for board in boards:
- names = variants_of(cfg, board)
- mine = {n: rows[n] for n in names if n in rows}
- # a variant name that is neither declared nor prefixed cannot be attributed; the
- # `<board>-` fallback only helps ad-hoc builds, it is not the primary path. It must
- # also never steal a row DECLARED by another board: a declared variant need not start
- # with its own board's name, so it may happen to start with this board's name plus '-'.
- mine.update({n: c for n, c in rows.items()
- if n.startswith(f'{board}-') and n not in mine
- and owner.get(n, board) == board})
- if not mine:
- results.append({'board': board, 'ran': False, 'pass': False, 'locked': False,
- 'detail': 'no report row for this board'})
- continue
- locked = any(LOCKED_CELL in cells for cells in mine.values())
- bad = []
- for vname, cells in sorted(mine.items()):
- for test, val in sorted(cells.items()):
- if test == LOCKED_CELL:
- continue
- if cell_state(str(val)) == 'fail':
- bad.append(f'{vname} {test}: {val}')
- ok = not bad and not locked
- if locked:
- detail = 'held by another holder; not flashed'
- elif bad:
- detail = '; '.join(bad)
- else:
- detail = f'{len(mine)} variant(s), {sum(len(c) for c in mine.values())} cell(s) ok'
- results.append({'board': board, 'ran': True, 'pass': ok, 'locked': locked,
- 'detail': detail})
- return {'results': results, 'banner': report.get('banner', '')}
-
-
-def main() -> int:
- ap = argparse.ArgumentParser()
- ap.add_argument('config_file')
- ap.add_argument('-b', '--board', action='append', default=[],
- help='boards to report on; default: every board in the config')
- ap.add_argument('--report-dir', default='.', help='where hil_report.json lives (default: cwd)')
- a = ap.parse_args()
-
- cfg = json.loads(Path(a.config_file).read_text())
- boards = a.board or [b['name'] for b in cfg.get('boards', [])]
- jpath = Path(a.report_dir) / 'hil_report.json'
- if not jpath.is_file():
- print(f'error: {jpath} not found -- did hil_test.py run in this directory?',
- file=sys.stderr)
- return 1
- json.dump(summarize(cfg, boards, json.loads(jpath.read_text())), sys.stdout, indent=2)
- print()
- return 0
-
-
-if __name__ == '__main__':
- sys.exit(main())
diff --git a/test/hil/helper/hil_util.py b/test/hil/helper/hil_util.py
index 0a2a13fca..03d01270f 100644
--- a/test/hil/helper/hil_util.py
+++ b/test/hil/helper/hil_util.py
@@ -11,6 +11,7 @@ import glob
import os
import signal
import subprocess
+import unicodedata
import threading
import sys
from pathlib import Path
@@ -92,6 +93,25 @@ CMD_TIMEOUT = pos_int_env('HIL_CMD_TIMEOUT', 180)
TINYUSB_ROOT = Path(__file__).resolve().parents[3] # test/hil/helper/ -> repo root
+def display_width(s: str) -> int:
+ """Terminal COLUMNS, not characters.
+
+ The status marks the reports use -- ✅ ❌ ⚪ ⚠ 🔒 -- are one Python character and TWO
+ columns wide. Measuring with len() pads every cell containing one a column short, so
+ the pipes drift out of line against the header rule for the whole table.
+ """
+ return sum(2 if unicodedata.east_asian_width(c) in 'WF' else 1 for c in s)
+
+
+def pad(s: str, width: int, center: bool = False) -> str:
+ """str.ljust/center, measured in display columns. See display_width."""
+ room = max(0, width - display_width(s))
+ if not center:
+ return s + ' ' * room
+ left = room // 2
+ return ' ' * left + s + ' ' * (room - left)
+
+
def cmd_stdout_text(out: Any) -> str:
if out is None:
return ''
@@ -479,9 +499,26 @@ def run_alongside(argv: list, work, timeout: int) -> subprocess.CompletedProcess
return _reap()
-def run_cmd(cmd: str, cwd: str | None = None, timeout: int | None = None,
+def _cmd_label(cmd) -> str:
+ """A one-line name for a banner. An argv whose payload is a `python3 -c` program would
+ otherwise dump the whole body into the CI log, where run_cmd's banners are already the
+ noisiest thing in a failing row."""
+ if isinstance(cmd, str):
+ return cmd
+ parts = [a if len(a) <= 60 else f'<{len(a)}-char program>' for a in cmd]
+ return ' '.join(parts)
+
+
+def run_cmd(cmd: str | list, cwd: str | None = None, timeout: int | None = None,
binary: bool = False, split_stderr: bool = False,
quiet: bool = False) -> subprocess.CompletedProcess:
+ """Bounded subprocess: own session, killpg on expiry, rc 124 when it had to be killed.
+
+ `cmd` is a shell STRING or an argv LIST. argv exists for a program that cannot survive
+ a trip through the shell -- a multi-line `python3 -c` body -- which is how the harness
+ runs a library call that no in-process bound can contain. A daemon thread cannot bound
+ a C call that holds the GIL, so for those the child process IS the bound.
+ """
if timeout is None:
timeout = CMD_TIMEOUT
# binary: raw bytes (text mode's errors='replace' mangles non-UTF-8 file content).
@@ -490,32 +527,29 @@ def run_cmd(cmd: str, cwd: str | None = None, timeout: int | None = None,
# still print: a killed child is always noteworthy).
popen_kwargs = {
'cwd': cwd,
- 'shell': True,
+ # a list goes straight to execve; only a string needs a shell to parse it
+ 'shell': isinstance(cmd, str),
'stdout': subprocess.PIPE,
'stderr': subprocess.PIPE if split_stderr else subprocess.STDOUT,
}
if not binary:
popen_kwargs.update({'text': True, 'encoding': 'utf-8', 'errors': 'replace'})
- if os.name != 'nt':
- # C-level setsid, same process-group semantics as preexec_fn=os.setsid but
- # safe when called from threads (pool_check runs flashes from a thread pool)
- popen_kwargs['start_new_session'] = True
+ # C-level setsid, same process-group semantics as preexec_fn=os.setsid but safe when
+ # called from threads (pool_check runs flashes from a thread pool)
+ popen_kwargs['start_new_session'] = True
p = subprocess.Popen(cmd, **popen_kwargs)
try:
out, err = p.communicate(timeout=timeout)
r = subprocess.CompletedProcess(args=cmd, returncode=p.returncode, stdout=out, stderr=err)
except subprocess.TimeoutExpired as ex:
- if os.name != 'nt':
- try:
- os.killpg(p.pid, signal.SIGKILL)
- except OSError:
- # ProcessLookupError: already gone. PermissionError: an all-root group
- # refuses the group kill -- letting either escape would skip the bounded
- # reap, the pipe close and the rc-124 return this handler exists for.
- pass
- else:
- p.kill()
+ try:
+ os.killpg(p.pid, signal.SIGKILL)
+ except OSError:
+ # ProcessLookupError: already gone. PermissionError: an all-root group refuses
+ # the group kill -- letting either escape would skip the bounded reap, the pipe
+ # close and the rc-124 return this handler exists for.
+ pass
try:
out, err = p.communicate(timeout=10)
except subprocess.TimeoutExpired:
@@ -543,7 +577,7 @@ def run_cmd(cmd: str, cwd: str | None = None, timeout: int | None = None,
timeout_err = _typed(err if err is not None else ex.stderr)
if split_stderr and timeout_err is None:
timeout_err = b'' if binary else ''
- _print_banner(f'COMMAND TIMEOUT ({timeout}s): {cmd}', timeout_out, timeout_err)
+ _print_banner(f'COMMAND TIMEOUT ({timeout}s): {_cmd_label(cmd)}', timeout_out, timeout_err)
return subprocess.CompletedProcess(args=cmd, returncode=124, stdout=timeout_out, stderr=timeout_err)
except BaseException:
# BaseException, not Exception (as in CPython's own subprocess.run):
@@ -551,18 +585,15 @@ def run_cmd(cmd: str, cwd: str | None = None, timeout: int | None = None,
# its OWN group, so it never got the terminal's SIGINT -- without this, Ctrl-C
# leaves the flasher or testusb holding the probe and its usbfs node. Kill and
# close, never wait: this path must not add a hang of its own.
- if os.name != 'nt':
- try:
- os.killpg(p.pid, signal.SIGKILL)
- except OSError:
- pass
- else:
- p.kill()
+ try:
+ os.killpg(p.pid, signal.SIGKILL)
+ except OSError:
+ pass
_close_pipes(p)
raise
if r.returncode != 0 and not quiet:
- _print_banner(f'COMMAND FAILED: {cmd}', r.stdout, r.stderr)
+ _print_banner(f'COMMAND FAILED: {_cmd_label(cmd)}', r.stdout, r.stderr)
elif verbose:
print(cmd)
print(cmd_stdout_text(r.stdout))
diff --git a/test/hil/hil_ci.sh b/test/hil/hil_ci.sh
index 514b0f174..daa787242 100644
--- a/test/hil/hil_ci.sh
+++ b/test/hil/hil_ci.sh
@@ -210,6 +210,69 @@ rm -rf -- "$1"
mkdir -p -- "$1/test/hil/helper" "$1/examples"
REMOTE
+# The --accumulate merge base. The wipe above just cleared REMOTE_DIR, and
+# accumulate_report merges onto the sidecar in the RUN's cwd (hil_test.py:2193 sets
+# `fresh = not args.accumulate`, and only a non-fresh run reads it) -- so without this a
+# remote retry starts from nothing and its one-row table REPLACES the full-fleet one it was
+# meant to extend. The copy-back at the end of this script has always existed; this is the
+# other half of it.
+#
+# Gated, not unconditional: a fresh run unlinks the sidecar anyway (hil_test.py:2244), so
+# uploading there is wasted work that also obscures what the wipe means.
+#
+# <config>.failed is deliberately NOT uploaded: hil_test.py only ever writes it, never
+# reads it -- the retry spec reaches the rig as the -b/-bt arguments the caller expanded
+# from it (`hil_ci.sh $(cat <config>.failed)`).
+# argparse decides, not a case arm: hil_test.py declares `-a, --accumulate`, so argparse
+# also accepts `-av`, `-va`, `--accum` and `--acc` -- and hil-validate.js tells the
+# operator to retry "adding -v", which makes `-av` the natural spelling. A hand-rolled
+# match missed all four: no upload, and the else-branch warning never fired either, so the
+# one-row table replaced the full-fleet one in silence.
+ACCUMULATE=$(python3 - ${ARGS[@]+"${ARGS[@]}"} <<'PY'
+import argparse, sys
+p = argparse.ArgumentParser(add_help=False)
+p.add_argument('-a', '--accumulate', action='store_true')
+p.add_argument('-v', '--verbose', action='store_true') # so -av/-va bundle as they do there
+print(1 if p.parse_known_args(sys.argv[1:])[0].accumulate else 0)
+PY
+) || ACCUMULATE=0
+if [ "$ACCUMULATE" = 1 ]; then
+ if [ -f "$ROOT_DIR/hil_report.json" ]; then
+ # Provenance: hil_report.json is not namespaced by CONFIG or REMOTE (build.yml and
+ # pr_comment.yml read that exact name), so a `REMOTE=hifiphile CONFIG=.../hfp.json`
+ # run leaves an hfp sidecar behind that a later ci.lan retry would merge, publishing
+ # boards that never ran here. Require at least one row to belong to THIS roster.
+ if python3 - "$ROOT_DIR/hil_report.json" "$CONFIG" <<'PY'
+import json, sys
+try:
+ rows = json.load(open(sys.argv[1])).get('rows') or []
+ cfg = json.load(open(sys.argv[2])).get('boards') or []
+except Exception:
+ sys.exit(1)
+known = set()
+for b in cfg:
+ known.add(b.get('name'))
+ known.update(v.get('name') for v in (b.get('variant') or []))
+sys.exit(0 if not rows or any(r.get('board') in known for r in rows if isinstance(r, dict))
+ else 1)
+PY
+ then
+ echo "==> Uploading hil_report.json as the --accumulate merge base"
+ scp -q "$ROOT_DIR/hil_report.json" "$REMOTE:$REMOTE_DIR/"
+ else
+ echo "==> warning: $ROOT_DIR/hil_report.json holds no board from $(basename "$CONFIG")" \
+ "-- it is from another rig or config, so it is NOT being uploaded; this run's" \
+ "table will REPLACE rather than extend" >&2
+ fi
+ else
+ # Loud, because this is the failure mode: the run still succeeds, and quietly
+ # publishes a small table where a full one used to be.
+ echo "==> warning: --accumulate was requested but $ROOT_DIR/hil_report.json does not" \
+ "exist, so there is nothing to merge onto -- this run's table will REPLACE the" \
+ "previous one rather than extend it" >&2
+ fi
+fi
+
# Copy HIL test script and config
echo "==> Copying test scripts"
scp -q "$ROOT_DIR/test/hil/hil_test.py" \
@@ -223,7 +286,7 @@ scp -q "$ROOT_DIR/test/hil/helper/__init__.py" \
"$ROOT_DIR/test/hil/helper/hil_util.py" \
"$ROOT_DIR/test/hil/helper/hil_health.py" \
"$ROOT_DIR/test/hil/helper/hil_lock.py" \
- "$ROOT_DIR/test/hil/helper/hil_summary.py" \
+ "$ROOT_DIR/test/hil/helper/hil_report.py" \
"$REMOTE:$REMOTE_DIR/test/hil/helper/"
# Copy only firmware binaries (elf/bin/hex) plus esptool metadata
@@ -316,19 +379,41 @@ REMOTE
# Copy the generated report back to the local checkout (best-effort; the run's
# exit code is preserved regardless of whether a report was produced).
-scp -q "$REMOTE:$REMOTE_DIR/hil_report.md" "$ROOT_DIR/hil_report.md" \
- && echo "==> Report copied to $ROOT_DIR/hil_report.md" \
- || echo "==> warning: no hil_report.md copied back" >&2
+# rm -f FIRST, exactly as the sidecar loop below does: the markdown and the JSON are two
+# halves of ONE document now, so leaving a stale table behind when the copy fails -- beside
+# a sidecar that was correctly removed -- publishes last run's green results under this
+# run's red job, and the operator's hil_report.py call exits 1 against the missing sidecar.
+# Fetch BOTH halves to temps and commit them as a pair. Separate fetch/rename meant a
+# markdown that arrived beside a sidecar that did not left the local pair failing the
+# rendering invariant, and the next --accumulate retry merging the wrong base. Deleting
+# first and then scp'ing was worse still: an ssh drop at the end of a 60-minute run
+# destroyed the report outright.
+md_ok=0; json_ok=0
+scp -q "$REMOTE:$REMOTE_DIR/hil_report.md" "$ROOT_DIR/hil_report.md.tmp" 2>/dev/null \
+ && [ -f "$ROOT_DIR/hil_report.md.tmp" ] && md_ok=1
+scp -q "$REMOTE:$REMOTE_DIR/hil_report.json" "$ROOT_DIR/hil_report.json.tmp" 2>/dev/null \
+ && [ -f "$ROOT_DIR/hil_report.json.tmp" ] && json_ok=1
+if [ "$md_ok" = 1 ] && [ "$json_ok" = 1 ]; then
+ mv -f "$ROOT_DIR/hil_report.md.tmp" "$ROOT_DIR/hil_report.md"
+ mv -f "$ROOT_DIR/hil_report.json.tmp" "$ROOT_DIR/hil_report.json"
+ echo "==> Report copied to $ROOT_DIR/hil_report.md (+ sidecar)"
+else
+ rm -f "$ROOT_DIR/hil_report.md.tmp" "$ROOT_DIR/hil_report.json.tmp"
+ # All or nothing: a half-copied pair is worse than none. The stale local markdown goes
+ # because that is what gets pasted into a PR as this run's results; the stale sidecar
+ # goes with it so the two cannot disagree.
+ rm -f "$ROOT_DIR/hil_report.md" "$ROOT_DIR/hil_report.json"
+ echo "==> warning: report copy-back incomplete (md=$md_ok json=$json_ok); removed the" \
+ "stale local pair -- an --accumulate retry has no merge base until a run succeeds" >&2
+fi
-# The re-run spec and the JSON sidecar live in the run's cwd on the rig (REMOTE_DIR), and the
-# next invocation rm -rf's it. Without copying them back, the `--accumulate` retry every doc on
-# this branch prescribes has nothing to read and nothing to merge onto. Delete the local copies
-# FIRST: a green run writes no .failed, so a silent no-op scp would leave last run's spec in
-# the checkout looking current, and "retry from the spec" would re-flash boards that passed.
-for extra in "$(basename "$CONFIG").failed" hil_report.json; do
- rm -f "$ROOT_DIR/$extra"
- scp -q "$REMOTE:$REMOTE_DIR/$extra" "$ROOT_DIR/$extra" 2>/dev/null \
- && echo "==> $extra copied to $ROOT_DIR/$extra" || true
-done
+# The re-run spec lives in the run's cwd on the rig and the next invocation rm -rf's it.
+# Delete the local copy first: a green run writes no .failed, so a silent no-op scp would
+# leave last run's spec looking current and "retry from the spec" would re-flash boards
+# that passed.
+spec="$(basename "$CONFIG").failed"
+rm -f "$ROOT_DIR/$spec"
+scp -q "$REMOTE:$REMOTE_DIR/$spec" "$ROOT_DIR/$spec" 2>/dev/null \
+ && echo "==> $spec copied to $ROOT_DIR/$spec" || true
exit $rc
diff --git a/test/hil/hil_test.py b/test/hil/hil_test.py
index fcd7c7e6f..e32998420 100755
--- a/test/hil/hil_test.py
+++ b/test/hil/hil_test.py
@@ -64,14 +64,14 @@ from multiprocessing import TimeoutError as MpTimeoutError
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) # PYTHONSAFEPATH drops it
import hil_flash
-from helper import hil_health, hil_lock, hil_util
+from helper import hil_health, hil_lock, hil_report, hil_util
from helper.hil_util import device_tests, dual_tests, host_test
# Raw Lock/Semaphore objects in Pool initargs are inheritable only under fork
# (spawn/forkserver pickle them and fail at Pool creation), so pin it against an
-# interpreter default change. Windows has no fork: fall back so it still IMPORTS there.
+# interpreter default change.
-_mp = multiprocessing.get_context('fork') if os.name != 'nt' else multiprocessing.get_context()
+_mp = multiprocessing.get_context('fork')
Pool, Lock, Semaphore, Manager = _mp.Pool, _mp.Lock, _mp.Semaphore, _mp.Manager
import string
@@ -106,9 +106,6 @@ STATUS_OK = "\033[32mOK\033[0m"
STATUS_FAILED = "\033[31mFailed\033[0m"
STATUS_SKIPPED = "\033[33mSkipped\033[0m"
-# Plain (non-ANSI) cell symbols for hil_report.md; a missing binary counts as skipped.
-REPORT_CELL = {'pass': '✅', 'fail': '❌', 'skip': '⚪'}
-
class TestFail(AssertionError):
"""Fail a test but still surface a metric string in its report cell (e.g. usbtest's '❌ 29/30'
@@ -241,6 +238,9 @@ USBTEST_BATTERY_BUDGET = hil_util.pos_int_env('HIL_USBTEST_BATTERY_BUDGET', 260)
# timeout paths = 95s. 120 leaves a margin; 75 (my first estimate, taken before checking
# dmesg_tail) was 20s SHORT and would have killed the battery mid-print.
USBTEST_OVERSHOOT = 120
+# Named, not a literal, so the unit tests can zero it: every test that drives
+# test_device_usbtest against a fake rig otherwise pays a real 3s (ten of them, 30s a run).
+USBTEST_SETTLE = 3
SERIAL_READ_TIMEOUT = hil_util.pos_float_env('HIL_SERIAL_READ_TIMEOUT', 5)
SERIAL_WRITE_TIMEOUT = hil_util.pos_float_env('HIL_SERIAL_WRITE_TIMEOUT', 10)
@@ -320,6 +320,63 @@ LP_READER = (
' buf += chunk\n'
'sys.stdout.buffer.write(buf)\n'
)
+# Runs under hil_util.run_cmd as `python3 -c`, argv so the body needs no shell quoting.
+# A PROCESS, not a thread, and not optional: cython-hidapi wraps hid_enumerate in
+# `with nogil` but calls hid_open and hid_close BARE (hidapi 0.15.0 hid.pyx), so those hold
+# the GIL for their whole blocking call. A daemon thread cannot bound that -- the waiter
+# parks off-GIL but must reacquire the GIL to return, which the stuck thread never yields
+# -- so an in-process bound is inert exactly where it is needed, and the whole worker
+# freezes rather than just the call. killpg reaches a child regardless.
+#
+# What blocks: hidapi's hidraw backend reads `manufacturer` and `product` via udev for each
+# device that reaches create_device_info_for_device, via copy_udev_string(usb_dev,
+# "manufacturer"/"product") -- both usb_string_attr, served under the device lock a wedged
+# usbfs ioctl holds (v6.12.96 sysfs.c:141-143).
+#
+# Passing BOTH ids is what keeps a wedged peer out of that path, and it does more than skip
+# non-matches: hidapi only runs the cheap pre-check `if (vendor_id != 0 || product_id != 0)`
+# (0.15.0 linux/hid.c:962), so an unfiltered walk sends EVERY device straight to the locked
+# reads. The pre-check itself is free -- parse_hid_vid_pid_from_sysfs parses
+# <sysfs_path>/device/uevent (:532) -- and both `continue`s precede
+# create_device_info_for_device (:966-970 before :976). Six examples in this tree expose a
+# HID interface under VID cafe, so a VID-only walk would stall on any of them wedged on a
+# peer. hid_open passes the same ids through to hid_enumerate internally (:1030), so the
+# filter narrows that walk too -- but a peer running THIS example still matches both ids,
+# which is why the child process, not the filter, is what bounds this.
+HID_ECHO = r"""
+import hid, random, sys, time
+
+uid, budget, want_pid = sys.argv[1], float(sys.argv[2]), int(sys.argv[3], 16)
+deadline = time.monotonic() + budget
+
+dev = None
+while dev is None:
+ for d in hid.enumerate(0xCafe, want_pid):
+ if d["serial_number"] == uid:
+ dev = d
+ break
+ if dev is not None or time.monotonic() >= deadline:
+ break
+ time.sleep(1)
+if dev is None:
+ sys.exit(f"HID device not found for {uid}")
+
+h = hid.device()
+h.open(dev["vendor_id"], dev["product_id"], uid)
+try:
+ for size in (8, 32, 63):
+ # Report ID (0) + payload, padded to 64 bytes
+ payload = bytes(random.randint(1, 255) for _ in range(size))
+ h.write(bytes([0]) + payload + bytes(64 - size))
+ echo = h.read(64, 2000)
+ if not echo or len(echo) < size:
+ sys.exit(f"HID echo timeout or short read ({size} bytes)")
+ if bytes(echo[:size]) != payload:
+ sys.exit(f"HID echo wrong data ({size} bytes): "
+ f"sent {payload.hex()} received {bytes(echo[:size]).hex()}")
+finally:
+ h.close()
+"""
MTYPE_TIMEOUT = 30 # a README-sized read is <1 s; bounds a D-state hang on a wedged device
@@ -358,6 +415,13 @@ def read_disk_file(uid: str, lun: int, fname: str) -> bytes:
# ~5 KB of transfers plus libmtp setup takes seconds, not minutes; a larger value makes a
# wedged MTP board cost that much on every retry, all charged to the pool guard.
MTP_SESSION_MARGIN = 30 # transfer budget after enumeration; past it the session is killed
+# room past the child's OWN enumeration budget for the echo exchange (3 x write + a 2000ms
+# hidapi read) and interpreter start-up, so the outer kill only fires on a real stall
+HID_ECHO_MARGIN = 30
+# hid_generic_inout's own idProduct. Pinned against the example's descriptor by
+# HidEchoRunsInAChild.test_the_pid_matches_the_example, because a silent drift here would
+# widen the walk back to every cafe: HID device without failing anything.
+HID_INOUT_PID = 0x4012
def get_printer_dev(id: str, vendor_str, product_str, ifnum: int):
@@ -871,7 +935,7 @@ def test_device_cdc_msc_throughput(board):
# payload, so an HS board reads as suspiciously slow. Say so rather than publish a green
# cell whose scale is a guess.
scale = '' if speed_known else ' FS?'
- return f'{REPORT_CELL["pass"]} C {pair(cdc_r, cdc_w)} M {pair(msc_r, msc_w)}{scale}'
+ return f'{hil_report.REPORT_CELL["pass"]} C {pair(cdc_r, cdc_w)} M {pair(msc_r, msc_w)}{scale}'
def test_device_dfu(board):
@@ -1069,8 +1133,13 @@ def test_device_printer_to_cdc(board):
ready.unlink(missing_ok=True)
# stderr, not stdout: run_alongside keeps the payload stream clean, so a traceback
# from the reader now arrives on its own pipe
- assert r.returncode == 0, (f'CDC->Printer reader failed ({size} bytes, rc '
- f'{r.returncode}): {hil_util.cmd_stdout_text(r.stderr)[:200]}')
+ # rc 124 is run_alongside's kill -- a blocked usblp_open leaves stderr EMPTY, so
+ # without the fallback this renders as 'failed (32 bytes, rc 124):' and nothing
+ rdetail = hil_util.cmd_stdout_text(r.stderr).strip()[:200]
+ assert r.returncode == 0, (
+ f'CDC->Printer reader failed ({size} bytes): {rdetail}' if rdetail else
+ f'printer: reading {lp_dev} blocked (device wedged): the reader was killed on '
+ f'its bound (rc {r.returncode})')
assert r.stdout == test_data, (f'CDC->Printer wrong data ({size} bytes):\n'
f' expected: {test_data[:64]}\n received: {r.stdout[:64]}')
time.sleep(0.2)
@@ -1237,9 +1306,6 @@ def test_device_midi_test(board):
def test_device_audio_test_freertos(board):
uid = board['uid']
- if os.name == 'nt':
- return 'skipped'
-
pcm = None
timeout = enum_timeout()
while timeout > 0:
@@ -1299,38 +1365,19 @@ def test_device_audio_test_freertos(board):
def test_device_hid_generic_inout(board):
+ # The whole exchange runs in a child (see HID_ECHO): hidapi's blocking calls hold the
+ # GIL, so nothing in-process can bound them. run_cmd's killpg can.
uid = board['uid']
- import hid # cython-hidapi (pip: hidapi, apt: python3-hid)
-
- timeout = enum_timeout()
- dev = None
- while timeout > 0:
- for d in hid.enumerate(0xCafe):
- if d['serial_number'] == uid:
- dev = d
- break
- if dev:
- break
- time.sleep(1)
- timeout -= 1
- assert dev is not None, f'HID device not found for {uid}'
-
- h = hid.device()
- h.open(dev['vendor_id'], dev['product_id'], uid)
- try:
- for size in [8, 32, 63]:
- # Report ID (0) + payload, padded to 64 bytes
- payload = bytes([random.randint(1, 255) for _ in range(size)])
- report = bytes([0]) + payload + bytes(64 - size)
- h.write(report)
- echo = h.read(64, 2000)
- assert echo and len(echo) >= size, (
- f'HID echo timeout or short read ({size} bytes)')
- assert bytes(echo[:size]) == payload, (
- f'HID echo wrong data ({size} bytes):\n'
- f' expected: {payload.hex()}\n received: {bytes(echo[:size]).hex()}')
- finally:
- h.close()
+ r = hil_util.run_cmd(
+ [sys.executable, '-c', HID_ECHO, uid, str(enum_timeout()), f'{HID_INOUT_PID:#06x}'],
+ timeout=enum_timeout() + HID_ECHO_MARGIN, split_stderr=True)
+ # rc 124 is run_cmd's kill: the child was still inside a hidapi call, which is the
+ # wedge this runs in a child FOR -- and stderr is empty there, so say so rather than
+ # render a bare trailing colon
+ detail = hil_util.cmd_stdout_text(r.stderr).strip()[:300]
+ assert r.returncode == 0, (f'hid_generic_inout: {detail}' if detail else
+ f'hid_generic_inout: the child was killed on its bound '
+ f'(rc {r.returncode}) -- a hidapi call did not return')
def test_device_usbtest(board):
@@ -1364,11 +1411,11 @@ def test_device_usbtest(board):
f'no cafe:4010 device with serial {uid}' if seen is False else
f'cannot tell whether cafe:4010 {uid} is present: the bounded sysfs reads did '
f'not answer{hil_util.sysfs_blind_note()}',
- metric=f'{REPORT_CELL["fail"]} 0/30')
+ metric=f'{hil_report.REPORT_CELL["fail"]} 0/30')
# settle: right after flashing the enumeration can bounce once (and on dual-port parts
# the other port's stale node — same serial and PID — lingers), and testusb run into
# that gap sees the device drop mid-case
- time.sleep(3)
+ time.sleep(USBTEST_SETTLE)
# --keep-binding is required for concurrent batteries: usbtest.py's cleanup unbinds
# EVERY usbtest-bound interface, killing a peer battery under USBTEST_PARALLEL > 1, and
@@ -1454,8 +1501,22 @@ def test_device_usbtest(board):
board_wedged = (f'{board["name"]}: usbtest reported a hang and was killed '
f'before it could report a verdict')
raise TestFail(f'usbtest did not run: {detail}',
- metric=f'{REPORT_CELL["fail"]} 0/30')
+ metric=f'{hil_report.REPORT_CELL["fail"]} 0/30')
+
+ return _usbtest_verdict(board, data, out, passed, failed, recovery,
+ _rec_flasher)
+
+def _usbtest_verdict(board: Board, data: dict, out: str, passed: int, failed: int,
+ recovery: bool, rec_flasher: dict) -> str:
+ """The report cell for a battery that produced JSON, or a TestFail carrying one.
+
+ Also latches board_wedged, which stops the REST of this board's examples: each would
+ flash THROUGH the poisoned usbfs node, block, survive SIGKILL and add another stray --
+ one wedge becoming one stray per remaining example, which is the convoy this whole
+ containment path exists to prevent.
+ """
+ global board_wedged
# A HUNG case that recovery could not clear leaves a D-state holder on this board's
# usbfs node. Latch it: the remaining examples would each flash THROUGH that node,
# block, survive SIGKILL and add another stray -- turning one wedge into one stray per
@@ -1464,15 +1525,15 @@ def test_device_usbtest(board):
# the reflash worked, so a convoy-safe board whose recovery failed used to come back
# unlatched and flash every remaining example through the poisoned node.
if data.get('wedged') or (not recovery and 'HUNG' in out):
- # _rec_flasher, NOT board['flasher']: recovery was decided against recover_flasher()
- # at the top of this function, and the two diverge as soon as a roster carries the
+ # rec_flasher, NOT board['flasher']: recovery was decided against recover_flasher()
+ # in the caller, and the two diverge as soon as a roster carries the
# optional `flasher_recover` key -- naming the wrong one sends the operator to the
# wrong probe. The wording stays on what usbtest actually reported ("still wedged"),
# because unrecovered_hang is also set by the ambiguous/inconclusive aborts, where
# nothing hung and the old text was false on both clauses.
board_wedged = (f'{board["name"]}: usbtest reports the device still wedged '
- + (f'after a recovery reflash via {_rec_flasher["name"]}' if recovery
- else f'and {_rec_flasher["name"]} cannot deliver a recovery reflash'))
+ + (f'after a recovery reflash via {rec_flasher["name"]}' if recovery
+ else f'and {rec_flasher["name"]} cannot deliver a recovery reflash'))
# notrun counts toward the denominator but is NOT a failure: listing cases that never
# ran as failures sends a maintainer bisecting one of them.
@@ -1485,9 +1546,9 @@ def test_device_usbtest(board):
# the re-run spec. parsed=True: a retry re-pays the whole battery to re-observe a
# wedge, and flashes through the poisoned node to do it.
raise TestFail(f'usbtest {passed}/{total} but the device wedged ({board_wedged})',
- metric=f'{REPORT_CELL["fail"]} {passed}/{total}', parsed=True)
+ metric=f'{hil_report.REPORT_CELL["fail"]} {passed}/{total}', parsed=True)
if failed == 0 and notrun == 0 and total > 0:
- return f'{REPORT_CELL["pass"]} {passed}/{total}'
+ return f'{hil_report.REPORT_CELL["pass"]} {passed}/{total}'
bad = [c.get('num') for c in data.get('cases', [])
if c.get('status') not in ('PASS', 'BUDGET')]
why = f'usbtest {passed}/{total}'
@@ -1503,7 +1564,7 @@ def test_device_usbtest(board):
why += f'; {notrun} case(s) never ran ({reason}), so this says nothing about them'
# parsed ONLY when every case ran: an aborted battery (budget expiry, kernel hang, bus
# drop) leaves BUDGET entries, and those are exactly what a reflash retry can fix.
- raise TestFail(why, metric=f'{REPORT_CELL["fail"]} {passed}/{total}',
+ raise TestFail(why, metric=f'{hil_report.REPORT_CELL["fail"]} {passed}/{total}',
parsed=(notrun == 0))
@@ -1705,11 +1766,49 @@ def build_board(board: Board) -> tuple[str, int]:
return name, failed
-# pseudo-test column for a variant boundary the park-flash could not clear (see below)
-BOUNDARY_CELL = 'same-PID boundary'
+def _tests_for(board: Board) -> list:
+ """Which examples this board runs, in roster order.
+ Three sources, most specific first: an explicit -bt list for this board, a global -t
+ list filtered against what the board can actually do, or the roster's own capability
+ flags. The -t filter is not cosmetic -- without it a device-only board runs host/dual
+ tests whose `dev_attached` roster entry does not exist.
+ """
+ name = board['name']
+ if name in board_test:
+ return list(board_test[name])
+
+ board_tests = board.get('tests', {})
+ if test_only:
+ if 'only' in board_tests:
+ allowed = set(board_tests['only'])
+ return [t for t in test_only if t in allowed]
+ return [t for t in test_only
+ if board_tests.get(t.split('/', 1)[0]) is True]
+
+ if 'tests' not in board:
+ return []
+ test_list: list = []
+ if board_tests.get('device') is True:
+ test_list += list(device_tests)
+ if board_tests.get('dual') is True:
+ test_list += dual_tests
+ if board_tests.get('host') is True:
+ test_list += host_test
+ if 'only' in board_tests:
+ test_list = list(board_tests['only'])
+ for skip in board_tests.get('skip', []):
+ if skip in test_list:
+ test_list.remove(skip)
+ log_line(f'{name:25} {skip:30} ... Skip')
+ return test_list
-def test_board(board: Board) -> tuple[str, int, list[str], list, float]:
+
+def test_board(board: Board) -> tuple:
+ # (name, err_count, failed_tests, rows, duration[, blind, strays]) -- the board-LOCKED
+ # early return is 5 wide, the normal one 7. _blind_note and _stray_note index 5 and 6
+ # behind a len() guard, so a field inserted before them reads a WRONG SLOT rather than
+ # raising: a duration would report as a stray count.
swept = False
name = board['name']
flasher = board['flasher']
@@ -1722,42 +1821,11 @@ def test_board(board: Board) -> tuple[str, int, list[str], list, float]:
log_line(f'{name:25} {STATUS_FAILED}: {e}')
# visible report row so the ❌ matches the exit code; failed-tests stays empty so a
# re-run repeats the whole board (no bogus -bt filter)
- return name, 1, [], [(name, {'board-locked': 'fail'}, None)], 0.0
+ return name, 1, [], [(name, {hil_report.LOCKED_CELL: 'fail'}, None)], 0.0
# after the lock: flock wait behind a concurrent run is not board cost
t_board = time.monotonic()
try:
- test_list = []
-
- if name in board_test:
- test_list = board_test[name]
- elif len(test_only) > 0:
- # Explicit -t: filter against the board's capabilities, or a device-only board
- # runs host/dual tests whose `dev_attached` config entry does not exist.
- board_tests = board.get('tests', {})
- if 'only' in board_tests:
- allowed = set(board_tests['only'])
- test_list = [t for t in test_only if t in allowed]
- else:
- for t in test_only:
- category = t.split('/', 1)[0]
- if board_tests.get(category) is True:
- test_list.append(t)
- else:
- if 'tests' in board:
- board_tests = board['tests']
- if board_tests.get('device') is True:
- test_list += list(device_tests)
- if board_tests.get('dual') is True:
- test_list += dual_tests
- if board_tests.get('host') is True:
- test_list += host_test
- if 'only' in board_tests:
- test_list = board_tests['only']
- if 'skip' in board_tests:
- for skip in board_tests['skip']:
- if skip in test_list:
- test_list.remove(skip)
- log_line(f'{name:25} {skip:30} ... Skip')
+ test_list = _tests_for(board)
err_count = 0
failed_tests = []
@@ -1809,7 +1877,7 @@ def test_board(board: Board) -> tuple[str, int, list[str], list, float]:
# charging again would double-count one incident in the exit code
if not wedge_skip:
err_count += 1
- cells[BOUNDARY_CELL] = 'fail'
+ cells[hil_report.BOUNDARY_CELL] = 'fail'
# blaming run_list[0] would re-run an innocent test that then passes,
# leaving the boundary unretested; re-run the whole board instead
board_wide_fail = True
@@ -1825,7 +1893,7 @@ def test_board(board: Board) -> tuple[str, int, list[str], list, float]:
# Do NOT flash through a poisoned node: each attempt enumerates into
# it, blocks uninterruptibly and leaves another stray behind. Report
# the skip so the cell is not mistaken for a pass.
- cells[test] = f'{REPORT_CELL["skip"]} board wedged'
+ cells[test] = f'{hil_report.REPORT_CELL["skip"]} board wedged'
# ...and re-run the WHOLE board, like the boundary-failure path above:
# these tests never executed, so naming them individually in the .failed
# spec is not enough -- an --accumulate re-run that fixes only the wedged
@@ -1893,8 +1961,6 @@ def test_board(board: Board) -> tuple[str, int, list[str], list, float]:
_lock_fh.close()
-REPORT_MD = 'hil_report.md'
-REPORT_JSON = 'hil_report.json'
# controller hints from previous runs: uid -> {'name', 'pci', 'duration'}. Only 'pci' is
# consumed (dispatch order and first-flash budgeting, never battery serialization). PCI
# addresses are boot-stable, so the cache survives reboots and goes stale on re-cabling.
@@ -1912,66 +1978,6 @@ def schedule_boards(boards: list, pci_of_uid: dict) -> list:
return [b for grp in itertools.zip_longest(*buckets.values()) for b in grp if b is not None]
-def render_matrix(rows_all: list) -> str:
- """Render rows (list of (row_label, {example: status}, duration)) as an aligned
- markdown matrix: columns = tests (bare names) centered, boards left-aligned,
- per-row duration as the trailing column."""
- seen = set()
- for _, cells, _ in rows_all:
- seen.update(cells)
- if not seen:
- return 'No tests were run.'
-
- # metric-bearing columns pinned first, the rest alphabetical: stable regardless of the
- # shuffled execution order
- pinned = ['usbtest', 'cdc_msc_throughput', 'msc_file_explorer', 'msc_file_explorer_freertos']
-
- def col_key(t):
- name = t.rsplit('/', 1)[-1]
- return (pinned.index(name) if name in pinned else len(pinned), name, t)
-
- columns = sorted(seen, key=col_key)
- headers = [c.rsplit('/', 1)[-1] for c in columns] + ['duration'] # bare example names
-
- def cell(cells, col):
- v = cells.get(col)
- if v is None:
- return ''
- return REPORT_CELL.get(v, v) # status symbol, or a metric string (e.g. speed) verbatim
-
- rows_vals = [(lbl, [cell(cells, c) for c in columns] + [dur or ''])
- for lbl, cells, dur in rows_all]
- board_hdr = 'Board'
- board_w = max([len(board_hdr)] + [len(lbl) for lbl, _ in rows_vals])
- col_w = [max([len(h)] + [len(vals[i]) for _, vals in rows_vals])
- for i, h in enumerate(headers)]
-
- def line(label, values):
- padded = [label.ljust(board_w)] + [v.center(w) for v, w in zip(values, col_w)]
- return '| ' + ' | '.join(padded) + ' |'
-
- header = line(board_hdr, headers)
- sep = '| ' + '-' * board_w + ' | ' + ' | '.join(':' + '-' * (w - 2) + ':' for w in col_w) + ' |'
- body = [line(lbl, vals) for lbl, vals in rows_vals]
-
- # tally run cells (not-run cells are absent from the dicts). A cell is a bare status or
- # a metric string carrying its own icon ("❌ 29/30"), so classify by the leading icon.
- def cell_kind(v):
- if v == 'fail' or (isinstance(v, str) and v.startswith(REPORT_CELL['fail'])):
- return 'fail'
- if v == 'skip' or (isinstance(v, str) and v.startswith(REPORT_CELL['skip'])):
- return 'skip'
- return 'pass'
- kinds = [cell_kind(v) for _, cells, _ in rows_all for v in cells.values()]
- failed = kinds.count('fail')
- skipped = kinds.count('skip')
- passed = kinds.count('pass')
- summary = (f'**{REPORT_CELL["pass"]} {passed} passed · {REPORT_CELL["fail"]} {failed} failed · '
- f'{REPORT_CELL["skip"]} {skipped} skipped · blank not run**')
-
- return summary + '\n\n' + '\n'.join([header, sep] + body)
-
-
def _write_failed_spec(failed_fname: Path, report_dir: Path, mret: list) -> None:
"""Re-run spec: only the failed boards (-b), each restricted to its own failed tests
(-bt); a board with failures but no test list re-runs entirely.
@@ -2083,87 +2089,13 @@ def _blind_note(mret: list) -> str:
f'{", ".join(blind)}. See the usb-kernel-recover skill.\n')
-def accumulate_report(mret: list, report_dir: Path, fresh: bool, scope: str = '',
- banner: str = '') -> str:
- """Merge this run's results into hil_report.json in report_dir, then (re)write
- the markdown matrix to hil_report.md. `fresh` (a first run, no --accumulate)
- starts a new report; otherwise a re-run accumulates so boards/tests that
- already passed are preserved while re-run cells are updated. `scope` names the
- board filter, if any, so a scoped table is not mistaken for a full one.
- Returns the md."""
- acc = {} # ordered {row_label: [cells dict, duration str|None]}
- prior_banner = ''
- jpath = report_dir / REPORT_JSON
- if not fresh and jpath.is_file():
- try:
- saved = json.loads(jpath.read_text())
- # CI keys the report dir by run id, so the sidecar is from an earlier attempt
- for entry in saved.get('rows', []):
- acc[entry['board']] = [dict(entry['cells']), entry.get('duration')]
- # ... and so is the caveat those cells were collected under. A rerun on a rig
- # that has since recovered contributes no banner, and the .failed spec reruns
- # only FAILURES -- so the earlier attempt's passes are never re-earned and
- # would be published as clean results of a rig that was not.
- prior_banner = saved.get('banner', '')
- except (ValueError, KeyError, TypeError):
- pass # corrupt/old sidecar: start fresh
-
- # current cells override prior for boards/tests that ran; a filtered run reports
- # duration None, keeping the previous full-run value
- for name, _, _, rows, *_ in mret:
- if rows and not any('board-locked' in cells for _, cells, _ in rows):
- # board ran for real: clear a stale lock-failure cell (its row is keyed by
- # board name; test rows may be variant names)
- stale = acc.get(name)
- if stale is not None:
- stale[0].pop('board-locked', None)
- if not stale[0]:
- # variant-keyed boards never repopulate the board-name row, so drop it
- # or it renders as a blank ghost row
- del acc[name]
- for row_label, cells, dur in rows:
- row = acc.setdefault(row_label, [{}, None])
- # the boundary cell is only ever written on failure, so a re-run of this
- # variant that cleared the boundary must drop the previous attempt's ❌
- if BOUNDARY_CELL not in cells:
- row[0].pop(BOUNDARY_CELL, None)
- row[0].update(cells)
- if dur is not None:
- row[1] = dur
-
- report_dir.mkdir(parents=True, exist_ok=True)
- # by LINE, deduped: attempts repeat the same caveat far more often than they add a new
- # one, and three copies of the D-state note reads as three incidents
- seen, merged = set(), []
- for line in (prior_banner + banner).splitlines():
- if line.strip() and line not in seen:
- seen.add(line)
- merged.append(line)
- banner = '\n'.join(merged) + '\n' if merged else ''
- jpath.write_text(json.dumps({'rows': [{'board': k, 'cells': c, 'duration': d}
- for k, (c, d) in acc.items()],
- 'banner': banner}, indent=2) + '\n')
-
- md = render_matrix([(k, c, d) for k, (c, d) in acc.items()])
- if scope:
- # a scoped run's small table is otherwise indistinguishable from a full one, and
- # it replaces the previous full table in the sticky PR comment
- md = f'_Scoped run: {scope}. Boards/tests not listed were not run._\n\n' + md
- # LAST, so it is outermost: a rig-health caveat outranks the table AND the scope note,
- # and the top of the report is where hil/SKILL.md tells the agent to look for it.
- if banner:
- md = banner + '\n' + md
- (report_dir / REPORT_MD).write_text(md + '\n', encoding='utf-8')
- return md
-
-
# containment paths print through hil_health._p: stdout may already be a dead pipe (a
# dropped ssh session), and a BrokenPipeError there would skip os._exit
_p = hil_health._p
def _abandon_exit(pool, mgr, abandoned: bool, err_count: int,
- report: Path | None = None) -> None:
+ report_dir: Path | None = None) -> None:
"""Free the runner when the pool could not be shut down. Returns only if not abandoned.
Must run even while an exception is propagating: multiprocessing's atexit handler
@@ -2199,24 +2131,11 @@ def _abandon_exit(pool, mgr, abandoned: bool, err_count: int,
'stay locked.', flush=True)
# A report already written by accumulate_report says nothing about the abandon, and a
# green table under a red job is how an agent ends up pasting it as this run's result.
- # Prepend the caveat; best-effort, never at the cost of exiting.
- if report is not None:
- try:
- if report.exists():
- # utf-8 explicitly (the cells are ✅/❌/⚪) and catch ValueError too: a torn
- # report or a LANG=C locale raises UnicodeDecodeError -- NOT an OSError --
- # straight past os._exit, stranding the runner.
- body = report.read_text(encoding='utf-8', errors='replace')
- # Only when no banner is there yet, searched anywhere in the head rather
- # than at char 0: write_timeout_report's banner must stay FIRST (its table
- # is a PREVIOUS attempt's) and it puts the rig-health quote above itself.
- if '**HIL run ab' not in body[:2000]:
- report.write_text(
- '**HIL run abandoned: the worker pool would not shut down.** The '
- 'table below was collected before the abandon; treat board '
- 'results as unverified.\n\n' + body, encoding='utf-8')
- except (OSError, ValueError):
- pass
+ # Set the caveat in the DOCUMENT -- prepending to the markdown alone left the sidecar,
+ # which is all hil_report.summarize() and therefore an agent ever sees, saying nothing.
+ # Best-effort, never at the cost of exiting.
+ if report_dir is not None:
+ hil_report.mark_report_abandoned(report_dir, 'the worker pool would not shut down.')
try:
sys.stdout.flush()
except OSError:
@@ -2226,6 +2145,127 @@ def _abandon_exit(pool, mgr, abandoned: bool, err_count: int,
os._exit(min(err_count, 125) if err_count else 1)
+def _load_controller_hints() -> tuple[dict, dict]:
+ """The uid -> {name, pci, duration} cache, plus the uid -> pci view scheduling wants.
+
+ Best effort throughout: a missing, hand-edited or torn cache costs dispatch ORDER,
+ never the run.
+ """
+ hints: dict = {}
+ try:
+ with CONTROLLER_CACHE.open() as f:
+ loaded = json.load(f)
+ if isinstance(loaded, dict): # keep only the expected uid -> dict shape
+ hints = {k: v for k, v in loaded.items() if isinstance(v, dict)}
+ except (OSError, ValueError):
+ pass
+ return hints, {uid: h['pci'] for uid, h in hints.items() if h.get('pci')}
+
+
+def _save_controller_hints(hints: dict, mret: list, uid_of: dict, cmap) -> None:
+ """Fold this run's PCI resolutions and durations back into the cache, atomically.
+
+ Merge-on-write: another HIL job (the esp split) may have finished since our startup
+ read, so overlay only this run's boards rather than publishing our whole view.
+ """
+ for name, _, _, _, dur, *_ in mret:
+ uid = uid_of.get(name)
+ if uid is None:
+ continue
+ h = dict(hints.get(uid) or {})
+ h['name'] = name # informational: the cache is keyed by uid
+ h['pci'] = cmap.get(f'uid:{uid}') or h.get('pci')
+ if dur > 0: # test_board reports 0.0 for filtered (partial) runs
+ h['duration'] = round(dur, 1)
+ hints[uid] = h
+ merged: dict = {}
+ try:
+ with CONTROLLER_CACHE.open() as f:
+ cur = json.load(f)
+ if isinstance(cur, dict):
+ merged = {k: v for k, v in cur.items() if isinstance(v, dict)}
+ except (OSError, ValueError):
+ pass
+ # onto what the CACHE now holds, not onto our startup snapshot: another HIL job may
+ # have written a newer duration/pci for these boards since we read it
+ for name, *_ in mret:
+ uid = uid_of.get(name)
+ if uid is not None and uid in hints:
+ merged[uid] = {**merged.get(uid, {}), **hints[uid]}
+ CONTROLLER_CACHE.parent.mkdir(parents=True, exist_ok=True)
+ tmp = CONTROLLER_CACHE.with_suffix('.json.tmp')
+ with tmp.open('w') as f:
+ json.dump(merged, f, indent=1, sort_keys=True)
+ tmp.replace(CONTROLLER_CACHE)
+
+
+def _abort_report(reason: str, mret: list, config_boards: list, failed_fname: Path,
+ report_dir: Path, fresh: bool, health_banner: str,
+ timeout_secs: int | None = None) -> None:
+ """Keep what finished, name what did not, and get a report on disk. Never raises.
+
+ Both abort paths -- the pool guard expiring and a worker raising -- need exactly this,
+ and in this order. The re-run spec goes FIRST: a fresh run already unlinked it, and
+ leaving it unwritten is what made a GitHub re-run repeat the whole fleet. Only the
+ boards that never reported go in it.
+
+ The report follows, before anything that can block, and the caller raises afterwards
+ into the one containment path. `timeout_secs` adds the pool-guard fallback: when
+ accumulate_report itself fails -- an unwritable report dir, a torn JSON --
+ _abandon_exit can only stamp a report that EXISTS, so without it the artifact upload
+ finds nothing and the sticky PR comment keeps the previous push's green table under a
+ red job.
+ """
+ stuck = [b['name'] for b in config_boards if b['name'] not in {r[0] for r in mret}]
+ try:
+ _write_failed_spec(failed_fname, report_dir,
+ [(n, 1, [], None, 0) for n in stuck]
+ + [r for r in mret if r[1] > 0])
+ except Exception as werr: # noqa: BLE001 - it mkdir()s and open()s the report dir
+ # letting it raise here REPLACES the caller's RuntimeError, so the operator never
+ # sees the 'pool timed out' line and no report is written at all
+ print(f'warning: re-run spec failed: {type(werr).__name__}: {werr}', flush=True)
+ banner = (f"**HIL run {reason}.** {len(mret)} board(s) below finished and are this "
+ f"run's; {len(stuck)} never reported and are NOT in the table: "
+ f"{', '.join(stuck)}. The re-run spec covers those.\n")
+ try:
+ hil_report.accumulate_report(mret, report_dir, fresh, '',
+ health_banner + _blind_note(mret)
+ + _stray_note(mret), caveat=banner)
+ return
+ except Exception as rerr: # noqa: BLE001 - the caller's raise must still happen
+ print(f'warning: partial report failed: {type(rerr).__name__}: {rerr}'
+ + ('; falling back to the board list' if timeout_secs else ''), flush=True)
+ if timeout_secs is None:
+ return
+ try:
+ hil_report.write_timeout_report(
+ report_dir, [b for b in config_boards if b['name'] in stuck],
+ timeout_secs, prefix=health_banner)
+ except Exception as re2: # noqa: BLE001
+ print(f'warning: fallback report failed too: {type(re2).__name__}: {re2}',
+ flush=True)
+
+
+def _start_pool(seed: str, hints_by_uid: dict):
+ """(mgr, cmap, pool). Split out so main()'s try/finally reads as one shape.
+
+ maxtasksperchild=1: a fresh worker per board makes cross-board contamination
+ structural rather than dependent on every module global being reset by hand
+ (board_wedged, _current_fw, hil_flash's warn-once sets). The extra fork is noise
+ against a flash+test cycle.
+ """
+ mgr = Manager()
+ cmap = mgr.dict()
+ initargs = (Lock(), seed,
+ hil_lock.make_permit_sems(Semaphore, hil_lock.USBTEST_PARALLEL),
+ hil_lock.make_permit_sems(Semaphore, hil_lock.FLASH_PARALLEL),
+ cmap, Lock(), hints_by_uid)
+ pool = Pool(processes=os.cpu_count() or 1, initializer=init_worker,
+ initargs=initargs, maxtasksperchild=1)
+ return mgr, cmap, pool
+
+
def main() -> None:
"""
Hardware test on specified boards
@@ -2305,13 +2345,11 @@ def main() -> None:
print(msg, flush=True)
# loud AND leaving evidence: exiting with no report at all lets the PR comment
# keep the previous push's stale table under a red job
- try:
- rd = Path(os.environ.get('HIL_REPORT_DIR', '.'))
- rd.mkdir(parents=True, exist_ok=True)
- (rd / REPORT_MD).write_text(f'**HIL run selected no boards.** {msg}\n',
- encoding='utf-8')
- except OSError:
- pass
+ rd = Path(os.environ.get('HIL_REPORT_DIR', '.'))
+ # fresh must be threaded through: this runs BEFORE the `if fresh:` wipe below, so
+ # defaulting it here wiped an --accumulate run's accumulated rows -- the exact
+ # regression the parameter exists to prevent.
+ hil_report.mark_report_no_boards(rd, msg, fresh=not args.accumulate)
sys.exit(1)
@@ -2346,9 +2384,6 @@ def main() -> None:
report_dir = Path(os.environ.get('HIL_REPORT_DIR', '.'))
failed_fname = report_dir / (config_file.name + '.failed')
fresh = not args.accumulate
- # The unlink is DEFERRED to inside the pool try/except below: wiping here leaves
- # Manager() and Pool() running with the old report gone and no report-writing path
- # armed, so an EAGAIN/ENOMEM on fork gives CI an EMPTY report dir with no reason.
seed = os.getenv('HIL_SHUFFLE_SEED') or str(int(time.time()))
log_line(f'test-order shuffle seed: {seed} (HIL_SHUFFLE_SEED={seed} to replay); '
@@ -2358,16 +2393,7 @@ def main() -> None:
# unattributable from the log alone
f'pool guard: {POOL_TIMEOUT}s')
- hints = {}
- try:
- with CONTROLLER_CACHE.open() as f:
- loaded = json.load(f)
- # tolerate a hand-edited/torn cache: keep only the expected uid -> dict shape
- if isinstance(loaded, dict):
- hints = {k: v for k, v in loaded.items() if isinstance(v, dict)}
- except (OSError, ValueError):
- pass
- hints_by_uid = {uid: h['pci'] for uid, h in hints.items() if h.get('pci')}
+ hints, hints_by_uid = _load_controller_hints()
config_boards = schedule_boards(config_boards, hints_by_uid)
log_line('dispatch order: ' + ', '.join(b['name'] for b in config_boards))
@@ -2387,31 +2413,18 @@ def main() -> None:
# BEFORE Manager()/Pool(), not inside the try: hil_ci.sh reuses a persistent REMOTE_DIR
# and scp's the report back unconditionally, so if a fork failure (OSError/EAGAIN right
# after a convoy -- the case this whole block guards) skipped the wipe, the finally's
- # _abandon_exit would prepend "HIL run abandoned" to the PREVIOUS run's table and
+ # _abandon_exit would stamp "HIL run abandoned" onto the PREVIOUS run's report and
# publish last night's board results as this run's. Nothing is live yet here, so an
# OSError from the wipe itself just exits with its traceback -- it cannot strand the
# interpreter in multiprocessing's unbounded atexit join, which is what deferring it
# was protecting against.
if fresh:
report_dir.mkdir(parents=True, exist_ok=True)
- for f in (REPORT_JSON, REPORT_MD):
+ for f in (hil_report.REPORT_JSON, hil_report.REPORT_MD):
(report_dir / f).unlink(missing_ok=True)
failed_fname.unlink(missing_ok=True)
try:
- mgr = Manager()
- cmap = mgr.dict()
- initargs = (Lock(), seed,
- hil_lock.make_permit_sems(Semaphore, hil_lock.USBTEST_PARALLEL),
- hil_lock.make_permit_sems(Semaphore, hil_lock.FLASH_PARALLEL),
- cmap, Lock(), hints_by_uid)
- # maxtasksperchild=1: the sysfs blindness latch is process-global and permanent
- # (no decrement anywhere -- see hil_util.SYSFS_STUCK_MAX), so a worker that goes
- # blind on ONE wedged board would report 0/30 and "probe missing" for the 2-3
- # healthy boards it picked up afterwards. A fresh worker per board confines the
- # damage to the board that caused it; the extra fork is noise against a
- # flash+test cycle.
- pool = Pool(processes=os.cpu_count() or 1, initializer=init_worker,
- initargs=initargs, maxtasksperchild=1)
+ mgr, cmap, pool = _start_pool(seed, hints_by_uid)
# OUTER: encloses the pool block too, not just the reporting below. An exception
# escaping async_ret.get() (a worker exception, a Ctrl-C) runs the pool finally and
# then propagates straight out of main(); with _abandon_exit in a sibling try it
@@ -2428,43 +2441,13 @@ def main() -> None:
try:
mret = drain_pool(it, config_boards, deadline, out=mret)
except MpTimeoutError as te:
- mret = te.finished
- stuck = [b['name'] for b in config_boards
- if b['name'] not in {r[0] for r in mret}]
- # The re-run spec FIRST and before the raise: a fresh run already unlinked
- # it, so leaving it unwritten is what made the GitHub re-run repeat the
- # whole fleet. Only the boards that never reported go in it.
- _write_failed_spec(failed_fname, report_dir,
- [(n, 1, [], None, 0) for n in stuck]
- + [r for r in mret if r[1] > 0])
- # Then the report, with the rows that DID finish, before anything that can
- # block. Then RAISE into the ONE containment path: the inner finally runs
+ # RAISE afterwards into the ONE containment path: the inner finally runs
# the ordered sweep (kill_worker_children BEFORE terminate, or a reaped
# worker's flasher reparents out of reach), the outer one os._exit's.
- banner = (f'**HIL run abandoned: worker pool timed out after '
- f'{POOL_TIMEOUT}s.** {len(mret)} board(s) below finished and '
- f'are this run\'s; {len(stuck)} never reported and are NOT in '
- f'the table: {", ".join(stuck)}. Re-run covers those.\n')
- try:
- accumulate_report(mret, report_dir, fresh, '',
- health_banner + _blind_note(mret)
- + _stray_note(mret) + banner)
- except Exception as rerr: # noqa: BLE001 - the raise below must still happen
- # FALL BACK, do not just warn: accumulate_report can raise on an
- # unwritable/root-owned report dir or a torn JSON, and _abandon_exit
- # only PREPENDS to a report that exists. Without this the artifact
- # upload finds nothing (if-no-files-found: ignore) and the sticky PR
- # comment keeps the previous push's green table under a red job.
- print(f'warning: partial report failed: {type(rerr).__name__}: {rerr}; '
- f'falling back to the board list', flush=True)
- try:
- hil_health.write_timeout_report(
- report_dir, [b for b in config_boards
- if b['name'] in stuck], POOL_TIMEOUT, REPORT_MD,
- prefix=health_banner)
- except Exception as re2: # noqa: BLE001
- print(f'warning: fallback report failed too: '
- f'{type(re2).__name__}: {re2}', flush=True)
+ mret = te.finished
+ _abort_report(f'abandoned: worker pool timed out after {POOL_TIMEOUT}s',
+ mret, config_boards, failed_fname, report_dir, fresh,
+ health_banner, timeout_secs=POOL_TIMEOUT)
_p(f'HIL worker pool timed out after {POOL_TIMEOUT}s; sweeping and '
f'shutting it down (abandoning it if a worker is unkillable)',
flush=True)
@@ -2472,25 +2455,11 @@ def main() -> None:
except Exception as e:
# A worker RAISED -- e.g. a flasher adapter dropping off the bus makes
# get_serial_dev raise in the worker's flash section, which no per-test
- # handler guards. Same treatment as the timeout path: the drain means
- # `mret` already holds every board that finished, so keep those rows and
- # name only the ones still in flight. (Under map_async they were all lost,
- # which is what the old banner here claimed.)
- done = {r[0] for r in mret}
- stuck = [b['name'] for b in config_boards if b['name'] not in done]
- _write_failed_spec(failed_fname, report_dir,
- [(n, 1, [], None, 0) for n in stuck]
- + [r for r in mret if r[1] > 0])
- banner = (f'**HIL run aborted: a worker raised {type(e).__name__}: {e}.** '
- f'{len(mret)} board(s) below finished and are this run\'s; '
- f'{len(stuck)} did not report: {", ".join(stuck)}.\n')
- try:
- accumulate_report(mret, report_dir, fresh, '',
- health_banner + _blind_note(mret)
- + _stray_note(mret) + banner)
- except Exception as re2: # noqa: BLE001 - the raise below must still happen
- print(f'warning: partial report failed: {type(re2).__name__}: {re2}',
- flush=True)
+ # handler guards. The drain means `mret` already holds every board that
+ # finished, so keep those rows and name only the ones still in flight.
+ _abort_report(f'aborted: a worker raised {type(e).__name__}: {e}',
+ mret, config_boards, failed_fname, report_dir, fresh,
+ health_banner)
raise
err_count = build_err + sum(e[1] for e in mret)
@@ -2537,33 +2506,8 @@ def main() -> None:
report_dir.mkdir(parents=True, exist_ok=True)
with (report_dir / 'hil_profile_ctrl.json').open('w') as f:
json.dump(dict(cmap), f, indent=1, sort_keys=True)
- uid_of = {b['name']: b['uid'] for b in config['boards']}
- for name, _, _, _, dur, *_ in mret:
- uid = uid_of.get(name)
- if uid is None:
- continue
- h = dict(hints.get(uid) or {})
- h['name'] = name # informational: cache is keyed by uid
- h['pci'] = cmap.get(f'uid:{uid}') or h.get('pci')
- if dur > 0: # test_board reports 0.0 for filtered (partial) runs
- h['duration'] = round(dur, 1)
- hints[uid] = h
- # merge-on-write: another HIL job (e.g. the esp split) may have finished since
- # our startup read, so overlay only this run's boards and replace atomically
- merged = {}
- try:
- with CONTROLLER_CACHE.open() as f:
- cur = json.load(f)
- if isinstance(cur, dict):
- merged = {k: v for k, v in cur.items() if isinstance(v, dict)}
- except (OSError, ValueError):
- pass
- merged.update({uid_of[n]: hints[uid_of[n]] for n, *_ in mret if n in uid_of})
- CONTROLLER_CACHE.parent.mkdir(parents=True, exist_ok=True)
- tmp = CONTROLLER_CACHE.with_suffix('.json.tmp')
- with tmp.open('w') as f:
- json.dump(merged, f, indent=1, sort_keys=True)
- tmp.replace(CONTROLLER_CACHE)
+ _save_controller_hints(
+ hints, mret, {b['name']: b['uid'] for b in config['boards']}, cmap)
except Exception as e:
# Deliberately broad, and it must stay that way: this best-effort refresh makes
# Manager proxy RPCs that raise EOFError / BrokenPipeError / RemoteError when
@@ -2578,12 +2522,12 @@ def main() -> None:
# looks exactly like a full run that happened to be small
scoped = sorted(set(args.board) | set(board_test))
scope = f'{len(scoped)} board(s) — {", ".join(scoped)}' if scoped else ''
- report = accumulate_report(mret, report_dir, fresh, scope,
+ report = hil_report.accumulate_report(mret, report_dir, fresh, scope,
health_banner + _blind_note(mret)
+ _stray_note(mret))
print()
print(report)
- print(f'\nReport written to {(report_dir / REPORT_MD).resolve()}')
+ print(f'\nReport written to {(report_dir / hil_report.REPORT_MD).resolve()}')
duration = time.time() - duration
print()
@@ -2594,7 +2538,7 @@ def main() -> None:
# In the finally, not after: any raise above (accumulate_report sits outside the
# OSError handler) would skip the abandon path and unwind into multiprocessing's
# unbounded atexit join, hanging the runner.
- _abandon_exit(pool, mgr, pool_abandoned, err_count, report_dir / REPORT_MD)
+ _abandon_exit(pool, mgr, pool_abandoned, err_count, report_dir)
# Same clamp: exit status is a byte either way, so 256 failures would report green.
sys.exit(min(err_count, 125))
diff --git a/test/hil/test/stubs/hid.py b/test/hil/test/stubs/hid.py
new file mode 100644
index 000000000..e72aeea57
--- /dev/null
+++ b/test/hil/test/stubs/hid.py
@@ -0,0 +1,74 @@
+# SPDX-License-Identifier: MIT
+"""Scripted stand-in for cython-hidapi, for the HID_ECHO child tests.
+
+A real wedge cannot be manufactured on demand, so the failure modes are scripted here and
+selected with FAKE_HID_MODE. Mirrors test/stubs/pymtp.py, which does the same for libmtp.
+"""
+import ctypes
+import ctypes.util
+import os
+import time
+
+_MODE = os.environ.get('FAKE_HID_MODE', 'ok')
+_UID = os.environ.get('FAKE_HID_UID', 'CAFE01')
+
+
+def _gil_stall():
+ """Block forever WITHOUT releasing the GIL -- the shape cython-hidapi's bare
+ hid_open()/hid_close() calls have, and the one an in-process bound cannot touch.
+
+ PyDLL, not CDLL: CDLL releases the GIL around the call, which would make this the
+ easy case instead of the hard one. Resolved through find_library so a non-glibc libc
+ still works; PyDLL(None) is not usable here (its `sleep` returns immediately).
+ """
+ ctypes.PyDLL(ctypes.util.find_library('c') or 'libc.so.6').sleep(3600)
+_PID = int(os.environ.get('FAKE_HID_PID', '0x4012'), 16)
+
+
+def enumerate(vid=0, pid=0):
+ """Real hid.enumerate(vid, pid) filters on both ids -- 0 means "any" -- and returns a
+ 'path' key too. The filters are applied BEFORE the locked manufacturer/product reads,
+ which is why passing both narrows what a wedged peer can stall."""
+ if _MODE == 'wedged_enumerate':
+ # hidapi's hidraw backend reads `manufacturer`/`product` for every device it
+ # lists, both served under the device lock -- this is that stall.
+ while True:
+ time.sleep(3600)
+ if _MODE == 'absent':
+ return []
+ if vid not in (0, 0xCafe) or pid not in (0, _PID):
+ return []
+ return [{'serial_number': _UID, 'vendor_id': 0xCafe, 'product_id': _PID,
+ 'path': b'/dev/hidraw0'}]
+
+
+class device:
+ def __init__(self):
+ self._last = b''
+
+ def open(self, vid, pid, serial):
+ if _MODE == 'wedged_open':
+ while True:
+ time.sleep(3600)
+ if _MODE == 'wedged_open_gil':
+ # a thread-based bound is inert against this; only killing the process works
+ _gil_stall()
+
+ def write(self, report):
+ self._last = bytes(report)
+
+ def read(self, size, timeout_ms):
+ if _MODE == 'wedged_read':
+ while True:
+ time.sleep(3600)
+ if _MODE == 'short_read':
+ return list(self._last[1:4])
+ if _MODE == 'wrong_data':
+ return list(bytes(b ^ 0xFF for b in self._last[1:]))
+ return list(self._last[1:]) # the device echoes the payload, minus report ID
+
+ def close(self):
+ if _MODE == 'wedged_close':
+ # also GIL-holding in cython-hidapi, and it runs in HID_ECHO's finally on
+ # every failure path
+ _gil_stall()
diff --git a/test/hil/test/test_ci_metrics.py b/test/hil/test/test_ci_metrics.py
index a76b6e3a0..6f1511913 100644
--- a/test/hil/test/test_ci_metrics.py
+++ b/test/hil/test/test_ci_metrics.py
@@ -447,16 +447,134 @@ class TestWorkflowSelectionHandOff(unittest.TestCase):
self.assertIn('UNSCOPED', flat[max(0, i - 200):i],
'a fall-open path without the marker build.yml greps for')
- def test_membrowse_upload_sees_the_same_board_as_the_build(self):
- # $EX_ARGS is passed for the BOARD it selects: --one-first picks a board that can
- # build the -e set, so without it membrowse configures a different, empty build
- # dir and uploads --identical for a board that was never compiled. It does NOT
- # scope the targets - `examples-membrowse-upload` is not `all`, so it passes
- # through as the aggregate, which has no DEPENDS and still records every example.
+ def _run_extras_block(self, sel):
+ """Extract the build-extras shell block from build.yml and run it for real.
+ Nothing else exercises it, which is why the empty/rejected conflation shipped."""
+ import re as _re, shlex, subprocess, tempfile, json as _json
+ repo = os.path.dirname(CIRCLECI)
+ i = self.build.index("EXAMPLE_MAP='{}'\n BUILD_FILTERED='false'")
+ i = self.build.rindex('\n', 0, i) + 1
+ j = self.build.index(' echo "matrix=$MATRIX_JSON"', i)
+ block = _re.sub(r'^ {10}', '', self.build[i:j], flags=_re.M)
+ with tempfile.TemporaryDirectory() as d:
+ selp = os.path.join(d, 'sel.json')
+ with open(selp, 'w') as fh:
+ _json.dump(sel, fh)
+ matrix = subprocess.run(
+ [sys.executable, os.path.join(repo, '.github/scripts/ci_set_matrix.py'),
+ '--select-file', selp], capture_output=True, text=True, cwd=repo).stdout.strip()
+ self.assertTrue(matrix, 'ci_set_matrix produced nothing')
+ sh = os.path.join(d, 'probe.sh')
+ with open(sh, 'w') as fh:
+ # shlex.quote, not hand-rolled quoting: a TMPDIR with a space in it
+ # made this fail for a reason that had nothing to do with the block
+ fh.write('BUILD_SELECT_FILE=' + shlex.quote(selp) + '\n')
+ fh.write('MATRIX_JSON=' + shlex.quote(matrix) + '\n')
+ fh.write(block)
+ # sentinel + newline separated: the block itself writes ::warning:: to
+ # stdout, and '|' would collide with the regex's own separator
+ fh.write('\nprintf "@@R@@\\n%s\\n%s\\n%s" "$MATRIX_JSON" "$BUILD_FILTERED" "$FAMILY_REGEX"\n')
+ r = subprocess.run(['bash', sh], capture_output=True, text=True, cwd=repo)
+ self.assertEqual(r.returncode, 0, r.stderr)
+ mj, filtered, regex = r.stdout.split('@@R@@\n', 1)[1].split('\n', 2)
+ return sum(len(v) for v in _json.loads(mj).values()), filtered, regex
+
+ def test_an_empty_family_list_is_not_treated_as_unusable(self):
+ """.build.families is read twice - as a count and as a `|`-joined regex. An EMPTY
+ list and one REJECTED by the charset guard both leave the regex empty and mean
+ opposite things, so the block has to branch on which happened.
+
+ Testing `-z "$FAMILY_REGEX"` alone sent every nothing-selected PR down the
+ fall-open path and discarded the correct all-empty matrix: #3842 (docs +
+ .gitignore) and #3840 (test/hil only) each rebuilt all 74 cmake legs after the
+ selector had correctly chosen none."""
+ legs, filtered, regex = self._run_extras_block(
+ {'build': {'full': False, 'families': [], 'family_examples': {}}})
+ self.assertEqual(legs, 0, 'an empty families list must keep the all-empty matrix')
+ self.assertEqual(filtered, 'false', 'nothing was built, so nothing to compare')
+ self.assertEqual(regex, '')
+
+ def test_a_real_family_list_stays_scoped(self):
+ legs, filtered, regex = self._run_extras_block(
+ {'build': {'full': False, 'families': ['stm32f4', 'rp2040'],
+ 'family_examples': {}}})
+ self.assertGreater(legs, 0)
+ self.assertEqual(filtered, 'true')
+ self.assertEqual(regex, 'stm32f4|rp2040')
+
+ def test_a_regex_metacharacter_in_a_family_name_falls_open(self):
+ # the name is interpolated raw into a name_is_regexp artifact pattern, so a
+ # metacharacter would match another family's baseline - reject and widen
+ legs, filtered, regex = self._run_extras_block(
+ {'build': {'full': False, 'families': ['stm32f4.*'], 'family_examples': {}}})
+ self.assertGreater(legs, 100, 'a rejected family list must fall open to full')
+ self.assertEqual(filtered, 'false')
+ self.assertEqual(regex, '')
+
+ def test_membrowse_upload_is_not_scoped_by_the_pr_filter(self):
+ # by decision, the upload runs unfiltered so the size history stays keyed on the
+ # family's preferred board whatever the PR touched. $EX_ARGS would not have
+ # scoped the targets either way - `examples-membrowse-upload` is not `all`, so
+ # resolve_example_target_groups passes it through as the aggregate - but it DID
+ # move the board, because --one-first picks one that can build the -e set.
+ #
+ # The accepted cost: on a family whose preferred board cannot build that set,
+ # the upload lands on a board the Build step never compiled and every example
+ # goes up --identical. test_the_upload_board_can_diverge_from_the_built_board
+ # keeps that consequence measured rather than assumed.
line = [l for l in self.util.splitlines()
if '--target examples-membrowse-upload' in l][0]
- self.assertIn('$EX_ARGS', line)
- self.assertNotIn('-e ', line.replace('$EX_ARGS', ''))
+ self.assertNotIn('$EX_ARGS', line)
+ self.assertNotIn('-e ', line)
+
+ def test_the_upload_board_can_diverge_from_the_built_board(self):
+ """Pins the SIZE of what the removal gave up, so it cannot grow unnoticed.
+
+ --one-first with no -e returns preferred_list[0]; with one it returns the first
+ preferred board that can build it. Where those differ, the Membrowse Upload step
+ configures a build dir the Build step never wrote.
+
+ ci=True unconditionally, as _prune_buildable does and for the same reason: the
+ answer must be the runner's, not the developer's. The CI skip lists are off by
+ default locally, which moves the pick on three families - this test asserted the
+ local set and went red on its first CI run."""
+ sys.path.insert(0, os.path.join(REPO, 'tools'))
+ import build as build_py
+ roles = ('device', 'host', 'dual')
+ exs = sorted(f'{r}/{n}' for r in roles
+ for n in os.listdir(os.path.join(REPO, 'examples', r))
+ if os.path.isdir(os.path.join(REPO, 'examples', r, n)))
+ fams = sorted(d for d in os.listdir(os.path.join(REPO, 'hw/bsp'))
+ if os.path.isdir(os.path.join(REPO, 'hw/bsp', d, 'boards')))
+ cwd = os.getcwd()
+ os.chdir(REPO)
+ try:
+ diverging = set()
+ for fam in fams:
+ try:
+ base = build_py.get_family_boards(fam, False, True, None, 'cmake',
+ (), ci=True)
+ except Exception:
+ continue
+ if not base:
+ continue
+ for e in exs:
+ try:
+ one = build_py.get_family_boards(fam, False, True, [e], 'cmake',
+ (), ci=True)
+ except Exception:
+ continue
+ if one and one[0] != base[0]:
+ diverging.add(fam)
+ break
+ finally:
+ os.chdir(cwd)
+ self.assertEqual(diverging, {'imxrt', 'lpc11', 'lpc18', 'lpc54', 'mcx', 'rx',
+ 'samd11', 'samd2x_l2x', 'samd5x_e5x', 'stm32l0',
+ 'stm32l4', 'tm4c'},
+ 'the set of families whose membrowse upload can land on an '
+ 'uncompiled board changed; re-check whether dropping $EX_ARGS '
+ 'from the upload step is still the right trade')
if __name__ == '__main__':
diff --git a/test/hil/test/test_ci_select.py b/test/hil/test/test_ci_select.py
index 8f1841531..ace230246 100644
--- a/test/hil/test/test_ci_select.py
+++ b/test/hil/test/test_ci_select.py
@@ -302,7 +302,11 @@ class TestArgsEmission(unittest.TestCase):
out = j.loads(r.stdout)
self.assertFalse(out['full'])
self.assertIn('tinyusb.json', out['args'])
- self.assertTrue(any('cdc_device' in line for line in out['reasons']))
+ # reasons are a stderr diagnostic, deliberately NOT in the payload: they were
+ # 97% of a 9.8 MB JSON on a dep bump, and every consumer re-parses that file
+ self.assertNotIn('reasons', out, 'reasons must not ride in the machine-read JSON')
+ self.assertNotIn('reasons', out['build'])
+ self.assertIn('cdc_device', r.stderr)
# A core-class diff must select boards THROUGH THE CLI: the in-process tests
# inject their own repo root, so only this subprocess path catches a broken
# repo_root derivation -- which once made every repo-relative glob match
@@ -840,6 +844,368 @@ class TestRostersDoNotOverlap(unittest.TestCase):
seen[b['name']] = b.get('tests')
+class TestTypecRule(unittest.TestCase):
+ """Rule 12b. src/typec/usbc.c is listed unconditionally by src/CMakeLists.txt and
+ src/tinyusb.mk, but its whole body is `#if CFG_TUC_ENABLED`, which only
+ examples/typec/power_delivery sets - so it is parsed by every build and compiled by
+ one. Same shape as the class rule, same answer. Before this rule it matched nothing
+ and force-fulled 82 families and all 30 rig boards."""
+
+ def test_build_axis_selects_only_the_typec_examples(self):
+ s = ci_select.classify_build(['src/typec/usbc.c'], REPO)
+ self.assertFalse(s['full'])
+ self.assertTrue(s['families'], 'typec must be compiled somewhere')
+ self.assertTrue(s['family_examples'], 'and the examples must be named')
+ for fam, exs in s['family_examples'].items():
+ self.assertTrue(exs, fam)
+ for e in exs:
+ self.assertTrue(e.startswith('typec/'), f'{fam}: {e} is not a typec example')
+
+ def test_every_typec_file_answers_the_same(self):
+ for f in ('src/typec/usbc.c', 'src/typec/usbc.h', 'src/typec/tcd.h',
+ 'src/typec/pd_types.h'):
+ s = ci_select.classify_build([f], REPO)
+ self.assertFalse(s['full'], f)
+ self.assertTrue(s['families'], f)
+
+ def test_no_rig_board_runs_typec(self):
+ # typec is not a HIL role, so the rig cannot exercise it whatever it selects
+ s = sel(['src/typec/usbc.c'])
+ self.assertFalse(s['full'])
+ self.assertEqual(s['boards'], {})
+
+ def test_it_tracks_the_enabling_config_rather_than_a_hardcoded_list(self):
+ # the answer must come from CFG_TUC_ENABLED in the example configs, so it
+ # follows a new typec example (or an old one switched off) on its own
+ want = ci_select.examples_enabling(
+ ci_select.role_examples(REPO, ('typec',)), ('CFG_TUC_ENABLED',), REPO)
+ self.assertTrue(want, 'no example enables CFG_TUC_ENABLED - rule 12b is dead')
+ got = set()
+ for exs in ci_select.classify_build(['src/typec/usbc.c'], REPO)['family_examples'].values():
+ got |= set(exs)
+ self.assertEqual(got, want)
+
+
+class TestCachesAreKeyedOnTheTree(unittest.TestCase):
+ """build_utils caches on repo-RELATIVE paths while ci_select._in_repo() chdirs
+ between trees, so the cwd has to be part of every cache key. Without it a second
+ tree gets the first tree's skip.txt/only.txt and FAMILY_MCUS - which is exactly the
+ base-vs-branch comparison the code-size skill does in one process."""
+
+ def test_a_second_tree_is_not_answered_from_the_first(self):
+ import build_utils, tempfile
+ old = os.getcwd()
+ try:
+ os.chdir(REPO)
+ self.assertFalse(build_utils.skip_example('host/bare_api', 'metro_m0_express'))
+ with tempfile.TemporaryDirectory() as d:
+ os.makedirs(os.path.join(d, 'hw/bsp'), exist_ok=True)
+ os.chdir(d)
+ # the board does not exist in this tree at all -> unknown board -> skip
+ self.assertTrue(build_utils.skip_example('host/bare_api', 'metro_m0_express'),
+ 'the empty tree was answered from the repo tree cache')
+ os.chdir(REPO)
+ self.assertFalse(build_utils.skip_example('host/bare_api', 'metro_m0_express'),
+ 'and the repo answer must survive the excursion')
+ finally:
+ os.chdir(old)
+
+
+class TestClassesWithNoEnablingExample(unittest.TestCase):
+ """The class rule is the one rule with no drift guard: ports, hw/mcu, get_deps
+ tokens and bsp families all have one. A class dir that no example config enables
+ selects NOTHING on both axes (the maintainer's empty-means-empty ruling), which is
+ right - but it must be a listed state, not a surprise, or a class added before its
+ first example silently stops being built."""
+
+ # class dirs no example's tusb_config.h turns on, for either role. Must only shrink:
+ # a new entry means a class nothing compiles, so a break in it reaches master.
+ NO_EXAMPLE = {'bth'}
+
+ def test_only_the_known_classes_select_nothing(self):
+ import glob as _glob
+ dead = set()
+ for d in sorted(_glob.glob(os.path.join(REPO, 'src/class/*'))):
+ if not os.path.isdir(d):
+ continue
+ cls = os.path.basename(d)
+ hit = False
+ for base in sorted(os.path.basename(f) for f in _glob.glob(os.path.join(d, '*.[ch]'))):
+ roles = ci_select._class_roles(base)
+ if ci_select._build_class_examples(cls, base, roles, REPO):
+ hit = True
+ break
+ if not hit:
+ dead.add(cls)
+ self.assertEqual(dead, self.NO_EXAMPLE,
+ 'a class dir enabled by no example config: it selects nothing on '
+ 'both axes, so nothing compiles it until the next master push')
+
+
+class TestTheHarnessTestsAreNotTheHarness(unittest.TestCase):
+ """test/hil/test/ selects nothing; test/hil/ itself still selects everything.
+
+ Rule 2 is a bare `test/hil/` prefix, so the harness's own unit tests were booking
+ the full 27-board rig - ~11 minutes of exclusive hardware for a diff that cannot
+ reach it. Nothing on the rig runs them: pre-commit does, and build.yml runs
+ test_ci_select.py as the gate before trusting a selection at all.
+
+ The carve-out is only safe while that directory holds nothing rig-affecting, which
+ is what the second test pins."""
+
+ def test_the_harness_own_tests_select_nothing_on_either_axis(self):
+ for p in ('test/hil/test/test_ci_select.py', 'test/hil/test/test_ci_metrics.py',
+ 'test/hil/test/test_hil_bounded.py', 'test/hil/test/stubs/pymtp.py',
+ 'test/hil/test/stubs/hid.py'):
+ s = ci_select.classify([p], REPO, ROSTERS)
+ self.assertFalse(s['full'], p)
+ self.assertFalse(s['boards'], p)
+ b = ci_select.classify_build([p], REPO)
+ self.assertFalse(b['full'], p)
+ self.assertFalse(b['families'], p)
+
+ def test_the_harness_itself_still_takes_the_whole_rig(self):
+ # the thing rule 2 exists for: these decide what the rig does, so they cannot be
+ # trusted to narrow their own blast radius
+ for p in ('test/hil/hil_test.py', 'test/hil/tinyusb.json',
+ 'test/hil/helper/hil_ci_set_matrix.py'):
+ s = ci_select.classify([p], REPO, ROSTERS)
+ self.assertTrue(s['full'], f'{p} must still force the full rig')
+
+ def test_nothing_rig_affecting_has_moved_into_the_carve_out(self):
+ """The carve-out is a claim about that directory's contents; pin them.
+
+ A new file there that the rig DOES read would silently stop selecting the rig.
+ Listing them costs one line per file and makes that a failing test instead."""
+ out = subprocess.run(['git', 'ls-files', 'test/hil/test'], cwd=REPO,
+ capture_output=True, text=True, check=True)
+ self.assertEqual(sorted(out.stdout.split()), [
+ 'test/hil/test/stubs/hid.py',
+ 'test/hil/test/stubs/pymtp.py',
+ 'test/hil/test/test_ci_metrics.py',
+ 'test/hil/test/test_ci_select.py',
+ 'test/hil/test/test_hil_bounded.py',
+ 'test/hil/test/test_hil_health.py',
+ 'test/hil/test/test_hil_report.py',
+ 'test/hil/test/test_hil_util.py',
+ ], 'test/hil/test/ gained or lost a file; it is carved out of rule 2, so confirm '
+ 'the rig still does not read anything in there before updating this list')
+
+
+class TestExampleMapOmitsFullFamilies(unittest.TestCase):
+ """A family whose selection is ALREADY everything it can build carries no -e list.
+
+ Sixth of the same shape as the class below, found the same way: a perf rewrite of
+ _prune_buildable dropped the `set(kept) != set(buildable)` test and all 216 tests
+ stayed green. The build outcome is identical either way -- build.py applies the same
+ skip_example the pruner just did -- so nothing compiled differently and only the
+ payload grew (22 families x 33 examples on one dcd_dwc2.c diff). That is exactly the
+ kind of drift no build failure ever reports."""
+
+ def test_a_device_only_port_diff_still_omits_families_it_cannot_narrow(self):
+ # dcd_dwc2.c selects device+dual examples only, but a family whose host examples
+ # are all unbuildable anyway ends up wanting its entire buildable set
+ b = ci_select.classify_build(['src/portable/synopsys/dwc2/dcd_dwc2.c'], REPO)
+ self.assertFalse(b['full'])
+ self.assertTrue(b['families'])
+ omitted = [f for f in b['families'] if f not in b['family_examples']]
+ self.assertTrue(omitted, 'no family omitted its -e list; the "already everything '
+ 'this family builds" case stopped being detected')
+ for fam in omitted:
+ self.assertNotIn(fam, b['family_examples'])
+
+ def test_a_family_that_can_build_more_than_the_diff_wants_keeps_its_list(self):
+ # the other direction: one example selects itself and nothing else, so every
+ # family it lands on must carry an explicit -e or CI builds all 46
+ b = ci_select.classify_build(['examples/device/cdc_msc/src/main.c'], REPO)
+ self.assertFalse(b['full'])
+ for fam in b['families']:
+ self.assertEqual(b['family_examples'].get(fam), ['device/cdc_msc'], fam)
+
+
+class TestSelectionBehavioursThatHadNoTest(unittest.TestCase):
+ """Five behaviours a reviewer's mutation pass proved were unpinned: break each one
+ and the whole suite stayed green. Each test here fails against its mutant.
+
+ They are grouped because they share a shape - every one is a small expression whose
+ removal silently NARROWS the selection, which is the failure direction that merges a
+ regression rather than wasting a runner."""
+
+ def test_build_defines_reach_the_prefilter(self):
+ # mutant: `defines = ()` in build.py's build_boards_list. metro_m4_express gets
+ # MAX3421_HOST=1 from its roster variant, never from its BSP, so without the
+ # defines the -e prefilter drops the rig's only MAX3421 firmware and hil-tinyusb
+ # has nothing to flash.
+ import build as build_py, build_utils, inspect
+ src = inspect.getsource(build_py.build_boards_list)
+ self.assertIn('defines = tuple(sorted(build_defines))', src,
+ 'the -D tokens must reach cmake_board/skip_example')
+ old = os.getcwd()
+ os.chdir(REPO)
+ try:
+ ex, board = 'dual/host_info_to_device_cdc', 'metro_m4_express'
+ self.assertTrue(build_utils.skip_example(ex, board),
+ 'without the define this example is correctly skipped')
+ self.assertFalse(build_utils.skip_example(ex, board, ('MAX3421_HOST=1',)),
+ 'with it, it must build - that is what the roster passes')
+ finally:
+ os.chdir(old)
+
+ def test_one_first_prefers_a_board_that_can_build_the_filter(self):
+ # mutant: buildable() -> True, i.e. back to all_boards[0]. lpc54's first board
+ # skips every msc_file_explorer example, so the leg would compile nothing.
+ import build as build_py
+ old_env, old = os.environ.get('GITHUB_ACTIONS'), os.getcwd()
+ os.environ['GITHUB_ACTIONS'] = 'true'
+ os.chdir(REPO)
+ try:
+ unfiltered = build_py.get_family_boards('lpc54', False, True)
+ filtered = build_py.get_family_boards('lpc54', False, True,
+ ['host/msc_file_explorer'])
+ self.assertEqual(unfiltered, ['lpcxpresso54114'], 'unfiltered pick must not move')
+ self.assertNotEqual(filtered, unfiltered,
+ 'the -e pick must avoid a board that skips the whole filter')
+ import build_utils
+ self.assertFalse(build_utils.skip_example('host/msc_file_explorer', filtered[0]),
+ f'{filtered[0]} must actually build the filtered example')
+ finally:
+ os.chdir(old)
+ if old_env is None:
+ os.environ.pop('GITHUB_ACTIONS', None)
+ else:
+ os.environ['GITHUB_ACTIONS'] = old_env
+
+ def test_a_class_file_selects_its_own_macro_not_just_the_directory(self):
+ # mutant: delete the _CLS_STEM_RE block. src/class/midi holds MIDI 1.0 AND 2.0;
+ # examples/device/midi2_device is the only example enabling CFG_TUD_MIDI2 and the
+ # only one that compiles midi2_device.c, but the directory macro alone misses it.
+ got = ci_select._build_class_examples('midi', 'midi2_device.c', {'device'}, REPO)
+ self.assertIn('device/midi2_device', got,
+ 'a midi2 change must select the example that compiles it')
+ host = ci_select._build_class_examples('midi', 'midi2_host.c', {'host'}, REPO)
+ self.assertIn('host/midi2_host', host)
+ # and the plain midi files must NOT drag midi2 in
+ plain = ci_select._build_class_examples('midi', 'midi_device.c', {'device'}, REPO)
+ self.assertNotIn('device/midi2_device', plain)
+
+ def test_a_port_change_selects_the_dual_examples(self):
+ # mutant: drop `+ ('dual',)`. A dcd/hcd change must build the dual examples -
+ # they exercise both stacks on one board, so a dwc2 break lands there first.
+ s = ci_select.classify_build(['src/portable/synopsys/dwc2/dcd_dwc2.c'], REPO)
+ duals = {e for exs in s['family_examples'].values() for e in exs
+ if e.startswith('dual/')}
+ self.assertTrue(duals, 'a dcd change selected no dual example')
+
+ def test_the_selector_answers_the_same_with_and_without_ci_env(self):
+ # mutant: drop ci=True from _prune_buildable. ci_skip_boards/ci_preferred_boards
+ # only apply when GITHUB_ACTIONS/CIRCLECI is set, so without the pin a laptop and
+ # a runner disagree - and /pre-pr would report a family list CI will not build.
+ files = ['examples/host/cdc_msc_hid_freertos/src/main.c']
+ old = os.environ.get('GITHUB_ACTIONS')
+ os.environ.pop('GITHUB_ACTIONS', None)
+ try:
+ local = ci_select.classify_build(files, REPO)['families']
+ os.environ['GITHUB_ACTIONS'] = 'true'
+ import importlib
+ importlib.reload(ci_select)
+ runner = ci_select.classify_build(files, REPO)['families']
+ finally:
+ if old is None:
+ os.environ.pop('GITHUB_ACTIONS', None)
+ else:
+ os.environ['GITHUB_ACTIONS'] = old
+ import importlib
+ importlib.reload(ci_select)
+ self.assertEqual(local, runner, 'the selector must not depend on the CI env vars')
+
+
+class TestRuleTableIsCarbonOfTheSpec(unittest.TestCase):
+ """ci_select's module docstring carries the rule table so a reader landing in the
+ code does not have to open the spec to learn what rule 6 is. Both are maintained by
+ hand, so this pins them cell-for-cell: edit one without the other and this fails.
+
+ It also pins the table against the CODE - every rule id the docstring claims must
+ appear as a `# rule N` marker on a branch of _classify_build_one, so a row cannot be
+ documented without a branch, or a branch renumbered without the table."""
+
+ @staticmethod
+ def _rows(text):
+ import re as _re
+ out = []
+ for l in text.splitlines():
+ if not l.startswith('| '):
+ continue
+ c = [x.strip() for x in l.strip().strip('|').split('|')]
+ if len(c) == 5 and _re.fullmatch(r'\d+[a-z]?', c[0]):
+ out.append(c)
+ return out
+
+ def test_docstring_table_matches_the_spec(self):
+ spec = open(os.path.join(
+ REPO, 'docs/superpowers/specs/2026-08-19-ci-build-family-filter-design.md')).read()
+ doc, spec_rows = self._rows(ci_select.__doc__), self._rows(spec)
+ self.assertTrue(spec_rows, 'no rule table found in the spec')
+ self.assertEqual([r[0] for r in doc], [r[0] for r in spec_rows],
+ 'rule ids differ between ci_select.__doc__ and the spec')
+ for d, s in zip(doc, spec_rows):
+ self.assertEqual(d, s, f'rule {d[0]} differs between the docstring and the spec')
+
+ def test_every_documented_rule_has_a_branch(self):
+ import re as _re
+ src = open(os.path.join(REPO, 'tools/ci_select.py')).read()
+ marked = set()
+ # handles `# rule 6`, `# rules 1, 1b` and `# rules 8-10`
+ for m in _re.finditer(r'#\s*rules?\s+([0-9a-z, -]+)', src):
+ for tok in _re.split(r',\s*', m.group(1).strip()):
+ rng = _re.fullmatch(r'(\d+)\s*-\s*(\d+)', tok.strip())
+ if rng:
+ marked.update(str(n) for n in range(int(rng.group(1)), int(rng.group(2)) + 1))
+ elif _re.fullmatch(r'\d+[a-z]?', tok.strip()):
+ marked.add(tok.strip())
+ documented = {r[0] for r in self._rows(ci_select.__doc__)}
+ missing = sorted(documented - marked, key=lambda s: (int(_re.match(r'\d+', s).group()), s))
+ self.assertEqual(missing, [], f'documented rules with no `# rule N` branch marker: {missing}')
+
+
+class TestNoTrackedFileIsUnclassified(unittest.TestCase):
+ """Rule 17 (unclassified -> full on both axes) is the fail-open net for paths nobody
+ anticipated. It must stay that way - a wrong `full` costs runner minutes and is
+ visible in the run, a wrong `empty` costs a merged regression and is invisible - but
+ nothing in the tree should REACH it. Every tracked file is classified by a rule, so
+ 17 fires only for genuinely new shapes, and this test is what tells the author to
+ write the row instead of letting the fall-through pick an answer for them.
+
+ Before this guard, 254 tracked files reached 17: .gitignore took a docs-only PR to
+ 74 cmake legs and the whole rig, while examples/<role>/CMakeLists.txt got the RIGHT
+ answer from the wrong rule - row 15 names it, the regex never matched it."""
+
+ def _unclassified(self, axis):
+ import subprocess as sp
+ r = sp.run(['git', 'ls-files'], cwd=REPO, capture_output=True, text=True)
+ if r.returncode != 0:
+ self.skipTest('not a git checkout')
+ files = r.stdout.split()
+ self.assertGreater(len(files), 1000, 'suspiciously few tracked files')
+ out = []
+ for f in files:
+ s = (ci_select.classify_build([f], REPO) if axis == 'build'
+ else ci_select.classify([f], REPO, real_rosters()))
+ if any('unclassified' in why for why in s['reasons']):
+ out.append(f)
+ return out
+
+ def test_build_axis(self):
+ left = self._unclassified('build')
+ self.assertEqual(left, [], f'{len(left)} tracked files fall through to rule 17 on '
+ f'the build axis, e.g. {left[:5]} - classify them, or '
+ f'add the pattern to _META_RE if no build reads them')
+
+ def test_hil_axis(self):
+ left = self._unclassified('hil')
+ self.assertEqual(left, [], f'{len(left)} tracked files fall through to rule 17 on '
+ f'the HIL axis, e.g. {left[:5]}')
+
+
class TestLibRule(unittest.TestCase):
"""lib/** is not a full-matrix path: only the examples that build the lib need it."""
@@ -1241,7 +1607,28 @@ class TestBuildClassifier(unittest.TestCase):
'tools/build.py', 'tools/cmake/cpu/cortex-m4.cmake',
'examples/CMakeLists.txt', 'examples/device/CMakeLists.txt',
'examples/build_system/cmake/cpu.cmake', '.github/workflows/build.yml',
- 'sonar-project.properties', 'some/unknown/path.c'):
+ '.circleci/config.yml', 'src/CMakeLists.txt', 'src/tinyusb.mk',
+ 'hw/bsp/family_support.mk', 'tools/build_utils.py',
+ 'some/unknown/path.c'):
+ self.assertTrue(self.b([p])['full'], p)
+
+ def test_repo_metadata_is_not_a_build_input(self):
+ # these used to reach `full` through rule 17: a PR touching only .gitignore and a
+ # README created 74 cmake legs and booked the whole rig. No Build step reads them.
+ for p in ('sonar-project.properties', '.gitignore', '.gitattributes',
+ '.clang-format', '.idea/misc.xml', 'version.yml', 'library.json',
+ 'examples/CMakePresets.json', 'test/fuzz/fuzz.cc',
+ 'test/unit-test/project.yml', '.github/workflows/pr_comment.yml',
+ 'tools/gen_doc.py'):
+ s = self.b([p])
+ self.assertFalse(s['full'], p)
+ self.assertEqual(s['families'], [], p)
+
+ def test_the_build_machinery_is_still_full(self):
+ # the other side of the same line: these DECIDE what gets built
+ for p in ('.circleci/config.yml', '.github/workflows/build.yml',
+ '.github/scripts/ci_set_matrix.py', 'tools/ci_select.py',
+ 'tools/build_utils.py', 'tools/metrics.py'):
self.assertTrue(self.b([p])['full'], p)
def test_mixed_diff_unions_per_family(self):
@@ -1313,15 +1700,23 @@ class TestBuildPostFilter(unittest.TestCase):
self.assertTrue(any('gone from tree' in r for r in s['reasons']), s['reasons'])
def test_class_source_selecting_nothing_selects_nothing(self):
- # synthetic class-with-no-enabling-config case (vendor_host.c was the live
- # instance until its removal): no config enables CFG_TUH_VENDOR, so
+ # a class-with-no-enabling-config case: no config enables CFG_TUH_VENDOR, so
# nothing exercises it and nothing builds - empty means empty (maintainer
# decision; the file is still parsed by every full master-push build, which is
- # the accepted net for a break outside its #if guard)
- s = ci_select.classify_build(['src/class/vendor/vendor_host.c'], REPO)
+ # the accepted net for a break outside its #if guard). src/class/bth is the
+ # live instance of this state today; TestClassesWithNoEnablingExample pins the
+ # whole set, so a new one cannot appear unnoticed.
+ # src/class/bth/bth_device.c, a file that EXISTS: the old assertion named
+ # src/class/vendor/vendor_host.c, deleted by the same branch, so any made-up
+ # path reached the same branch and the test passed vacuously.
+ real = os.path.join(REPO, 'src/class/bth/bth_device.c')
+ self.assertTrue(os.path.isfile(real), 'the case needs a file that exists')
+ s = ci_select.classify_build(['src/class/bth/bth_device.c'], REPO)
self.assertFalse(s['full'])
self.assertEqual(s['families'], [])
self.assertTrue(any('no contribution' in r for r in s['reasons']), s['reasons'])
+ # and the reason must name the class, not just any empty answer
+ self.assertTrue(any('bth' in r for r in s['reasons']), s['reasons'])
def test_class_source_with_examples_still_scopes(self):
s = ci_select.classify_build(['src/class/cdc/cdc_device.c'], REPO)
@@ -1881,14 +2276,14 @@ class TestMcuTokensResolve(unittest.TestCase):
# produce, or a rename nobody followed through. `family:samd21` was one of these
# until the nine examples/host/*/only.txt files were corrected to samd2x_l2x.
#
- # The `mcu:` entries are NOT all harmless. MIMXRT10XX/MIMXRT11XX and LPC177X_8X sit
- # beside a live token in the same file, so they gate nothing either way. MKL25ZXX
- # (device/msc_dual_lun) and SAME5X (device/audio_test) do not: those skips are dead,
- # and both examples are built today on the boards their skip file meant to exclude -
- # successfully, which is why nobody noticed. Correcting them REMOVES working build
- # coverage, so it is a maintainer call, not a drive-by fix.
+ # The remaining `mcu:` entries sit beside a live token in the same file, so they gate
+ # nothing either way. MKL25ZXX (7 files) and SAME5X (1) were dead too, but unlike
+ # these they were the ONLY token for their board - the examples were already being
+ # built on the very boards those lines meant to exclude. Dropping them is a no-op for
+ # the build (verified per example) and was chosen over re-pointing, which would have
+ # removed working coverage.
UNREACHABLE_TOKENS = {
- 'mcu': {'LPC177X_8X', 'MIMXRT10XX', 'MIMXRT11XX', 'MKL25ZXX', 'SAME5X', 'STM32U3'},
+ 'mcu': {'LPC177X_8X', 'MIMXRT10XX', 'MIMXRT11XX', 'STM32U3'},
'family': set(),
'board': set(),
}
diff --git a/test/hil/test/test_hil_bounded.py b/test/hil/test/test_hil_bounded.py
index c6d454f0e..6da65543f 100644
--- a/test/hil/test/test_hil_bounded.py
+++ b/test/hil/test/test_hil_bounded.py
@@ -39,6 +39,7 @@ serial_stub.SerialTimeoutException = type('SerialTimeoutException', (Exception,)
sys.modules.setdefault('serial', serial_stub)
import hil_flash
import hil_test
+from helper import hil_report
def write_script(path: Path, body: str) -> None:
@@ -46,6 +47,17 @@ def write_script(path: Path, body: str) -> None:
path.chmod(path.stat().st_mode | stat.S_IEXEC)
+def no_settle(case):
+ """Zero test_device_usbtest's post-flash settle for one test.
+
+ Real hardware needs it -- the enumeration can bounce once after a flash, and on
+ dual-port parts the stale same-serial node lingers. A fake rig has neither, and ten
+ tests drive that path, so leaving it real cost 30s of every suite run.
+ """
+ case.addCleanup(setattr, hil_test, 'USBTEST_SETTLE', hil_test.USBTEST_SETTLE)
+ hil_test.USBTEST_SETTLE = 0
+
+
def run_bounded(fn, timeout: float):
"""Run fn in a daemon thread; return (finished, exception). A still-running thread is
the hang under test — leave it to die with the interpreter."""
@@ -63,7 +75,6 @@ def run_bounded(fn, timeout: float):
return not t.is_alive(), exc[0] if exc else None
[email protected](os.name == 'nt', 'POSIX shell fakes')
class ReadDiskFile(unittest.TestCase):
def setUp(self):
self.tmp = TemporaryDirectory()
@@ -77,7 +88,7 @@ class ReadDiskFile(unittest.TestCase):
for name in ('get_disk_dev', '_enum_timeout', 'MTYPE_TIMEOUT'):
self.addCleanup(setattr, hil_test, name, getattr(hil_test, name))
hil_test.get_disk_dev = lambda uid, vendor, lun: str(self.dev)
- hil_test._enum_timeout = 2
+ hil_test._enum_timeout = 1 # the wait these tests must outlast; keep it small
self.bin = tmp / 'bin'
self.bin.mkdir()
self.addCleanup(os.environ.__setitem__, 'PATH', os.environ['PATH'])
@@ -112,7 +123,11 @@ class ReadDiskFile(unittest.TestCase):
t0 = time.monotonic()
with self.assertRaises(AssertionError) as cm:
hil_test.read_disk_file('uid0', 0, 'README.TXT')
- self.assertLess(time.monotonic() - t0, 1.5)
+ # BELOW one full _enum_timeout wait, not above it: "fails immediately" is the
+ # claim, and a bound of 1.5 against a 1s budget passes for code that spun the
+ # whole budget -- which is the regression this test exists to catch.
+ self.assertLess(time.monotonic() - t0, hil_test._enum_timeout,
+ 'read_disk_file spun the enumeration budget on a real answer')
self.assertIn('README.TXT', str(cm.exception))
def test_hung_mtype_cannot_hang_the_worker(self):
@@ -313,7 +328,7 @@ class _MtpFakeRig:
os.environ['PYTHONSAFEPATH'] = '1'
for name in ('_enum_timeout', 'MTP_SESSION_MARGIN'):
self.addCleanup(setattr, hil_test, name, getattr(hil_test, name))
- hil_test._enum_timeout = 2
+ hil_test._enum_timeout = 1 # the wait these tests must outlast; keep it small
# the session scratch files land in cwd
self.addCleanup(os.chdir, os.getcwd())
os.chdir(tmp)
@@ -326,7 +341,6 @@ class _MtpFakeRig:
os.environ[k] = v
[email protected](os.name == 'nt', 'POSIX shell fakes')
@unittest.skipIf(sys.version_info < (3, 11), 'fake-pymtp steering needs PYTHONSAFEPATH')
class DeviceMtp(_MtpFakeRig, unittest.TestCase):
"""test_device_mtp end to end: the real mtp_test.py subprocess under run_cmd,
@@ -426,7 +440,6 @@ class BoundedOpen(unittest.TestCase):
self.assertIsNone(self.hil_util.bounded_open(
str(Path(self.tmp.name) / 'nope'), os.O_RDONLY, 5))
- @unittest.skipIf(os.name == 'nt', 'POSIX fifo')
def test_blocking_open_gives_up_and_does_not_leak_fds(self):
"""A reader-less FIFO blocks open(O_WRONLY) forever -- the closest portable
stand-in for a wedged usblp node."""
@@ -441,7 +454,6 @@ class BoundedOpen(unittest.TestCase):
self.assertLessEqual(len(os.listdir('/proc/self/fd')) - before, 1,
'bounded_open leaked fds on the blocking path')
- @unittest.skipIf(os.name == 'nt', 'POSIX fifo')
def test_open_completing_during_the_abandon_does_not_leak(self):
"""The window the handoff lock exists for: the worker is at its store-or-close
decision when the caller gives up and drains the box.
@@ -512,7 +524,6 @@ class SysfsUnknownIsNotAbsent(unittest.TestCase):
def test_missing_attribute_is_none(self):
self.assertIsNone(self.hil_util.read_sysfs(str(Path(self.tmp.name) / 'nope')))
- @unittest.skipIf(os.name == 'nt', 'POSIX fifo')
def test_blocking_read_is_unknown_not_absent(self):
"""A reader-less FIFO stands in for the wedged device whose sysfs read never
returns; None here would read as "the board is gone"."""
@@ -806,7 +817,6 @@ class WedgedPidsFailsClosed(unittest.TestCase):
self.assertFalse(complete, 'a hidden holder was reported as absent')
[email protected](os.name == 'nt', 'POSIX shell fakes')
@unittest.skipIf(sys.version_info < (3, 11), 'fake-pymtp steering needs PYTHONSAFEPATH')
class StrandMemoRemembersUnstattablePaths(unittest.TestCase):
"""A stranded path whose inode could not be read is stored as None -- which dict.get()
@@ -899,16 +909,16 @@ class MtpGioFallthrough(unittest.TestCase):
t0 = time.monotonic()
r = subprocess.run([sys.executable,
str(Path(TEST_DIR).parents[0] / 'mtp_test.py'),
- '--uid', 'CAFE01', '--timeout', '3'],
+ '--uid', 'CAFE01', '--timeout', '1'],
capture_output=True, text=True, timeout=60, env=env)
elapsed = time.monotonic() - t0
- self.assertLess(elapsed, 30, f'did not honour --timeout 3 ({elapsed:.1f}s)')
+ self.assertLess(elapsed, 30, f'did not honour --timeout 1 ({elapsed:.1f}s)')
self.assertNotEqual(r.returncode, 0)
# The assertions above are satisfied by an immediate CRASH, which is exactly what
# shipped through this test once: `pass` left gio unbound and the next line
# dereferenced it. Assert the behaviour the docstring names -- it POLLED for the
# device (so it spent its budget) and did not die on a traceback.
- self.assertGreater(elapsed, 2.0,
+ self.assertGreater(elapsed, 0.8,
f'exited without polling ({elapsed:.1f}s) -- it crashed')
self.assertNotIn('Traceback', r.stderr)
self.assertIn('MTP device not found', r.stdout + r.stderr)
@@ -936,11 +946,17 @@ class RunWhileContract(unittest.TestCase):
def boom():
raise AssertionError('x')
+ # a duration no other process would plausibly pick: `pgrep -f` searches the WHOLE
+ # machine, so a bare `sleep 20` matched an unrelated background job -- another
+ # agent session's retry loop, in the case that exposed this -- and failed a test
+ # about our own child. Observed failing 3/3 in isolation while that loop ran.
+ sentinel = '20.0451'
with self.assertRaises(AssertionError):
- self.hil_util.run_alongside(['sleep', '20'], boom, 1)
+ self.hil_util.run_alongside(['sleep', sentinel], boom, 1)
# nothing of ours is left running: the reap ran on the error path too
import subprocess
- out = subprocess.run(['pgrep', '-f', '^sleep 20'], capture_output=True, text=True)
+ out = subprocess.run(['pgrep', '-f', f'^sleep {sentinel}'],
+ capture_output=True, text=True)
seen['strays'] = [p for p in out.stdout.split() if p]
self.assertEqual(seen['strays'], [], 'work() raising leaked the child')
@@ -1146,10 +1162,16 @@ class AbandonExitSurvivesAFailedFork(unittest.TestCase):
def test_none_pool_and_manager_still_write_the_banner(self):
# a subprocess, because _abandon_exit ends in os._exit: in-process it would take
# the test runner with it, before any assertion could run
+ import json
import subprocess
with TemporaryDirectory() as td:
- report = Path(td) / 'hil_report.md'
- report.write_text('| board | test |\n|---|---|\n', encoding='utf-8')
+ rd = Path(td)
+ # it takes the report DIRECTORY now and re-renders both artifacts from the
+ # sidecar, so seed the sidecar -- the markdown is output, not input
+ (rd / 'hil_report.json').write_text(json.dumps(
+ {'rows': [{'board': 'boardA', 'cells': {'cdc_msc': 'pass'},
+ 'duration': '1s'}],
+ 'banner': '', 'scope': '', 'caveat': ''}))
src = (
'import sys, types\n'
f'sys.path.insert(0, {str(Path(TEST_DIR).parents[0])!r})\n'
@@ -1160,12 +1182,14 @@ class AbandonExitSurvivesAFailedFork(unittest.TestCase):
'sys.modules.setdefault("serial", st)\n'
'import hil_test\n'
f'hil_test._abandon_exit(None, None, True, 1, __import__("pathlib")'
- f'.Path({str(report)!r}))\n')
+ f'.Path({str(rd)!r}))\n')
r = subprocess.run([sys.executable, '-c', src], capture_output=True,
text=True, timeout=120)
self.assertEqual(r.returncode, 1, r.stderr)
- self.assertTrue(report.read_text().startswith('**HIL run abandoned'),
- 'the abandon banner never reached the report')
+ self.assertTrue((rd / 'hil_report.md').read_text().startswith(
+ '**HIL run abandoned'), 'the abandon banner never reached the report')
+ self.assertIn('abandoned',
+ json.loads((rd / 'hil_report.json').read_text())['caveat'])
def test_kill_pool_children_tolerates_a_pool_that_never_existed(self):
from helper import hil_health
@@ -1211,6 +1235,7 @@ class UsbtestOuterBoundIsOneValue(unittest.TestCase):
# wedged-FIFO test would otherwise make every read here answer SYSFS_UNKNOWN
patch(_hu, '_sysfs_stuck', 0)
patch(_hu, '_sysfs_stranded', {})
+ patch(hil_test, 'USBTEST_SETTLE', 0) # see no_settle
patch(hil_lock, 'usbtest_permit', contextmanager(_permit))
patch(hil_test, 'skip_flash', skip_flash)
patch(hil_test, '_current_fw', '/tmp/fw.elf')
@@ -1318,6 +1343,7 @@ class UsbtestOuterKillStaysRetryable(unittest.TestCase):
from helper import hil_util as _hu
patch(_hu, 'glob', types.SimpleNamespace(glob=lambda p: [str(dev)]))
+ patch(hil_test, 'USBTEST_SETTLE', 0) # see no_settle
def _permit(uid): # a real generator: a lambda returning an iterator has
yield # no .throw(), so any raise inside the `with` would
# surface as an AttributeError from contextlib instead
@@ -1647,124 +1673,6 @@ class StagingCoversEveryBoardForm(unittest.TestCase):
"last run's re-run spec survived a green run")
-class SummaryFoldsReportToBoards(unittest.TestCase):
- """hil_summary.py replaces the agent retyping the markdown table. Report rows are named per
- VARIANT and a variant need not start with the board name, so the config is what maps them
- back -- the previous string-matching design produced a defect in each of four review rounds."""
-
- def _sum(self, boards, rows, cfg_boards=None, banner=''):
- import json
- import subprocess
- td = TemporaryDirectory()
- self.addCleanup(td.cleanup)
- d = Path(td.name)
- (d / 'hil_report.json').write_text(json.dumps(
- {'rows': [{'board': b, 'cells': c, 'duration': '1s'} for b, c in rows],
- 'banner': banner}))
- cfg = d / 'cfg.json'
- cfg.write_text(json.dumps({'boards': cfg_boards or [{'name': b} for b in boards]}))
- args = [a for b in boards for a in ('-b', b)]
- r = subprocess.run(['python3', str(Path(TEST_DIR).parents[0] / 'helper' / 'hil_summary.py'),
- str(cfg), *args, '--report-dir', str(d)],
- capture_output=True, text=True, timeout=60)
- self.assertEqual(r.returncode, 0, r.stderr)
- return json.loads(r.stdout)['results']
-
- def test_variant_rows_fold_onto_their_board(self):
- """nanoch32v203 never produces a row named after the board."""
- got = self._sum(['nanoch32v203'],
- [('nanoch32v203-fsdev', {'usbtest': 'pass'}),
- ('nanoch32v203-usbfs', {'usbtest': 'pass'})],
- cfg_boards=[{'name': 'nanoch32v203',
- 'variant': [{'name': 'nanoch32v203-fsdev'},
- {'name': 'nanoch32v203-usbfs'}]}])
- self.assertEqual([r['board'] for r in got], ['nanoch32v203'])
- self.assertTrue(got[0]['pass'])
- self.assertTrue(got[0]['ran'])
-
- def test_one_failing_variant_fails_the_board(self):
- got = self._sum(['nano'],
- [('nano-a', {'usbtest': 'pass'}), ('nano-b', {'usbtest': '❌ 29/30'})],
- cfg_boards=[{'name': 'nano', 'variant': [{'name': 'nano-a'},
- {'name': 'nano-b'}]}])
- self.assertFalse(got[0]['pass'])
- self.assertIn('29/30', got[0]['detail'])
-
- def test_lock_contention_is_a_field_not_a_prefix(self):
- got = self._sum(['alpha'], [('alpha', {'board-locked': 'fail'})])
- self.assertTrue(got[0]['locked'])
- self.assertFalse(got[0]['pass'])
-
- def test_a_board_with_no_row_is_marked_not_run(self):
- got = self._sum(['alpha', 'beta'], [('alpha', {'usbtest': 'pass'})])
- self.assertTrue(got[0]['ran'])
- self.assertFalse(got[1]['ran'])
- self.assertFalse(got[1]['pass'])
-
- def test_a_metric_cell_counts_by_its_icon(self):
- got = self._sum(['a', 'b'], [('a', {'cdc_msc_throughput': '✅ C 1.2 M 3.4'}),
- ('b', {'cdc_msc_throughput': '❌ C 0.0 M 0.0'})])
- self.assertTrue(got[0]['pass'])
- self.assertFalse(got[1]['pass'])
-
- def test_skipped_cells_do_not_fail_a_board(self):
- got = self._sum(['a'], [('a', {'usbtest': 'skip', 'cdc_msc': 'pass'})])
- self.assertTrue(got[0]['pass'])
-
- def test_a_plain_metric_cell_is_a_pass(self):
- """Mirrors hil_test.py's own tally (cell_kind): failures are ALWAYS marked -- 'fail'
- or a ❌ prefix, per TestFail's docstring -- while a passing test may return a plain
- metric string that lands in the cell unprefixed. Treating unknown shapes as fail
- would publish a green table as a red verdict."""
- got = self._sum(['a'], [('a', {'device_speed': '480.0 MBps'})])
- self.assertTrue(got[0]['pass'])
-
- def test_a_declared_variant_of_another_board_is_not_stolen(self):
- """A declared variant need not start with its own board's name, so it may start with
- a DIFFERENT board's name plus '-'. The prefix fallback must not attribute it twice."""
- got = self._sum(['alpha', 'beta'],
- [('beta-x', {'usbtest': 'fail'})],
- cfg_boards=[{'name': 'alpha', 'variant': [{'name': 'beta-x'}]},
- {'name': 'beta'}])
- self.assertTrue(got[0]['ran'])
- self.assertFalse(got[0]['pass'])
- self.assertFalse(got[1]['ran'], "beta must not inherit alpha's row")
-
-
-class CaveatSurvivesAccumulate(unittest.TestCase):
- """CI reruns with --accumulate: the sidecar keeps every earlier attempt's cells, but the
- banner was recomputed per attempt. A first attempt on a degraded rig and a clean rerun
- therefore published the degraded attempt's PASSES with no caveat on them -- and the
- generated .failed spec reruns only failures, so those cells are never re-earned."""
-
- def _rows(self, board, cell):
- return [(board, 0, 0, [(board, {cell: 'OK'}, '1s')], 0)]
-
- def test_an_earlier_attempts_caveat_is_still_on_the_report(self):
- td = TemporaryDirectory()
- self.addCleanup(td.cleanup)
- rd = Path(td.name)
- banner = '> **Rig note.** 2 process(es) in D state at start.\n'
-
- hil_test.accumulate_report(self._rows('boardA', 'cdc_msc'), rd, True, '', banner)
- self.assertIn('Rig note', (rd / hil_test.REPORT_MD).read_text())
-
- # the rerun: clean rig, so this attempt contributes no banner of its own
- md = hil_test.accumulate_report(self._rows('boardB', 'cdc_msc'), rd, False, '', '')
- self.assertIn('boardA', md) # the earlier cells are kept ...
- self.assertIn('Rig note', md,
- 'the caveat the earlier cells were collected under was dropped')
-
- def test_the_same_caveat_twice_is_not_stacked(self):
- td = TemporaryDirectory()
- self.addCleanup(td.cleanup)
- rd = Path(td.name)
- banner = '> **Rig note.** 2 process(es) in D state at start.\n'
- hil_test.accumulate_report(self._rows('boardA', 'cdc_msc'), rd, True, '', banner)
- md = hil_test.accumulate_report(self._rows('boardB', 'cdc_msc'), rd, False, '', banner)
- self.assertEqual(md.count('Rig note'), 1)
-
-
class BlindWorkerReachesTheReport(unittest.TestCase):
"""A worker that exhausts its bounded-read budget answers SYSFS_UNKNOWN for every
attribute, so its "device not found" means "could not tell". That reached the log and
@@ -1796,7 +1704,7 @@ class BlindWorkerReachesTheReport(unittest.TestCase):
wide = ('boardA', 1, ['device/cdc_msc'], [('boardA', {'cdc_msc': '❌'}, '2s')], 2.0, True)
narrow = ('stuck', 1, [], None, 0) # what the timeout path builds
hil_test._write_failed_spec(rd / 'x.failed', rd, [wide, narrow])
- md = hil_test.accumulate_report([wide], rd, True, '', hil_test._blind_note([wide]))
+ md = hil_report.accumulate_report([wide], rd, True, '', hil_test._blind_note([wide]))
self.assertIn('boardA', md)
self.assertIn('not all verdicts are evidence', md.lower())
@@ -1930,6 +1838,7 @@ class WedgeVerdictReachesTheLatch(unittest.TestCase):
def setUp(self):
self.addCleanup(setattr, hil_test, 'board_wedged', hil_test.board_wedged)
hil_test.board_wedged = ''
+ no_settle(self)
def _run(self, stdout, rc=0):
from helper import hil_lock, hil_util
@@ -1978,6 +1887,7 @@ class WedgedBoardCannotReportAPass(unittest.TestCase):
def setUp(self):
self.addCleanup(setattr, hil_test, 'board_wedged', hil_test.board_wedged)
hil_test.board_wedged = ''
+ no_settle(self)
def _cell(self, js):
"""Returns ('pass', cell) or ('fail', message)."""
@@ -2014,5 +1924,121 @@ class WedgedBoardCannotReportAPass(unittest.TestCase):
self.assertIn('30/30', cell)
+def _gil_stall_available() -> bool:
+ """Whether the hid stub can simulate a GIL-HOLDING stall on this host.
+
+ It needs a libc with sleep(3) loaded through ctypes.PyDLL. Everywhere the HIL harness
+ actually runs that is present; where it is not, the two tests that depend on it skip
+ rather than fail, because their subject is the bound, not ctypes.
+ """
+ import ctypes
+ import ctypes.util
+ try:
+ ctypes.PyDLL(ctypes.util.find_library('c') or 'libc.so.6')
+ return True
+ except OSError:
+ return False
+
+
+class HidEchoRunsInAChild(unittest.TestCase):
+ """hidapi's blocking calls hold the GIL -- cython-hidapi wraps hid_enumerate in
+ `with nogil` but calls hid_open and hid_close bare -- so a daemon thread cannot bound
+ them: the waiter parks off-GIL but must reacquire the GIL to return, which the stuck
+ thread never yields. Only a child process can be killed regardless, which is what
+ run_cmd's killpg does."""
+
+ def _run(self, mode, uid='CAFE01', budget='0', timeout=20, pid=None):
+ saved = {k: os.environ.get(k) for k in ('FAKE_HID_MODE', 'FAKE_HID_UID',
+ 'FAKE_HID_PID', 'PYTHONPATH',
+ 'PYTHONSAFEPATH')}
+
+ def restore():
+ for k, v in saved.items():
+ os.environ.pop(k, None) if v is None else os.environ.__setitem__(k, v)
+ self.addCleanup(restore)
+ os.environ['FAKE_HID_MODE'] = mode
+ os.environ['FAKE_HID_UID'] = uid
+ stubs = os.path.join(TEST_DIR, 'stubs')
+ pp = saved['PYTHONPATH']
+ os.environ['PYTHONPATH'] = stubs if not pp else f'{stubs}:{pp}'
+ # `python3 -c` puts the cwd at sys.path[0], AHEAD of PYTHONPATH, so any hid.py
+ # reachable from the suite's cwd would displace the stub and every mode-driven
+ # test below would pass or fail for the wrong reason. Safe-path mode drops it --
+ # the same practice _MtpFakeRig documents.
+ os.environ['PYTHONSAFEPATH'] = '1'
+ from helper import hil_util
+ want = pid or f'{hil_test.HID_INOUT_PID:#06x}'
+ return hil_util.run_cmd(
+ [sys.executable, '-c', hil_test.HID_ECHO, uid, budget, want],
+ timeout=timeout, split_stderr=True, quiet=True)
+
+ def _stderr(self, r):
+ from helper import hil_util
+ return hil_util.cmd_stdout_text(r.stderr)
+
+ def test_a_healthy_device_passes(self):
+ r = self._run('ok')
+ self.assertEqual(r.returncode, 0, self._stderr(r))
+
+ def test_the_pid_matches_the_example(self):
+ """The walk filters on BOTH ids, and hidapi applies them before the locked
+ manufacturer/product reads. Six examples in this tree expose a HID interface under
+ VID cafe, so a stale PID here silently widens the walk back to all of them -- and
+ nothing else would fail. Pinned against the descriptor rather than restated."""
+ import re
+ src = (Path(TEST_DIR).parents[2]
+ / 'examples/device/hid_generic_inout/src/usb_descriptors.c').read_text()
+ m = re.search(r'#define\s+USB_PID\s+(0x[0-9a-fA-F]+)', src)
+ self.assertIsNotNone(m, 'hid_generic_inout no longer defines USB_PID')
+ self.assertEqual(hil_test.HID_INOUT_PID, int(m.group(1), 16),
+ 'HID_INOUT_PID drifted from the example descriptor')
+
+ def test_a_peer_running_another_example_is_filtered_out(self):
+ """The point of the PID filter: a wedged sibling on a different example never
+ reaches the locked reads at all."""
+ r = self._run('ok', pid='0x400f') # hid_composite, not ours
+ self.assertNotEqual(r.returncode, 0)
+ self.assertIn('HID device not found', self._stderr(r))
+
+ @unittest.skipUnless(_gil_stall_available(), 'no libc for a GIL-holding stall')
+ def test_a_gil_holding_stall_is_still_killed(self):
+ """THE case an in-process bound cannot cover. hid_open is not `with nogil`, so a
+ thread-based guard is inert there; the child is killed anyway."""
+ t0 = time.monotonic()
+ r = self._run('wedged_open_gil', timeout=2)
+ self.assertEqual(r.returncode, 124,
+ 'a GIL-holding hidapi stall must still be killed on the bound')
+ self.assertLess(time.monotonic() - t0, 20, 'run_cmd did not bound the child')
+
+ def test_a_wedged_enumerate_is_killed_on_the_bound(self):
+ r = self._run('wedged_enumerate', timeout=2)
+ self.assertEqual(r.returncode, 124)
+
+ @unittest.skipUnless(_gil_stall_available(), 'no libc for a GIL-holding stall')
+ def test_a_wedged_close_is_killed_on_the_bound(self):
+ """close() runs in the child's finally on EVERY failure path and is also
+ GIL-holding; hidraw_release takes the same rwsem hidraw_open needs."""
+ r = self._run('wedged_close', timeout=3)
+ self.assertEqual(r.returncode, 124)
+
+ def test_an_absent_device_reports_why(self):
+ r = self._run('absent')
+ self.assertNotEqual(r.returncode, 0)
+ self.assertIn('HID device not found', self._stderr(r))
+
+ def test_a_bad_echo_reports_both_payloads(self):
+ r = self._run('wrong_data')
+ self.assertNotEqual(r.returncode, 0)
+ msg = self._stderr(r)
+ self.assertIn('wrong data', msg)
+ self.assertIn('sent', msg)
+ self.assertIn('received', msg)
+
+ def test_a_short_echo_is_not_read_as_a_pass(self):
+ r = self._run('short_read')
+ self.assertNotEqual(r.returncode, 0)
+ self.assertIn('short read', self._stderr(r))
+
+
if __name__ == '__main__':
unittest.main()
diff --git a/test/hil/test/test_hil_health.py b/test/hil/test/test_hil_health.py
index 5695cad6d..c7864f568 100644
--- a/test/hil/test/test_hil_health.py
+++ b/test/hil/test/test_hil_health.py
@@ -467,49 +467,6 @@ class KillPoolChildren(PatchCase):
self.assertEqual(hil_health.kill_pool_children(NoPool()), 0)
-class WriteTimeoutReport(unittest.TestCase):
- def test_prefix_carries_the_preflight_diagnosis(self):
- """The timeout aborts before accumulate_report, so without the prefix the artifact
- and the PR comment lose the one line saying WHY the pool never finished."""
- with TemporaryDirectory() as td:
- d = Path(td)
- hil_health.write_timeout_report(d, [{'name': 'b1'}], 4200, 'r.md',
- prefix='> **wedged usb_hub_wq worker.**\n')
- out = (d / 'r.md').read_text()
- self.assertTrue(out.startswith('> **wedged usb_hub_wq worker.**'))
- self.assertIn('timed out after 4200s', out)
- self.assertIn('- b1', out)
-
- def test_writes_a_report_where_there_would_be_none(self):
- with TemporaryDirectory() as td:
- hil_health.write_timeout_report(Path(td), [{'name': 'ra6m5_ek'}], 4200,
- 'hil_report.md')
- md = (Path(td) / 'hil_report.md').read_text()
- self.assertIn('4200s', md)
- self.assertIn('ra6m5_ek', md)
-
- def test_keeps_a_previous_attempts_table(self):
- with TemporaryDirectory() as td:
- path = Path(td) / 'hil_report.md'
- path.write_text('| board | cdc_msc |\n')
- hil_health.write_timeout_report(Path(td), [{'name': 'b1'}], 4200, 'hil_report.md')
- md = path.read_text()
- self.assertIn('abandoned', md)
- self.assertIn('| board | cdc_msc |', md)
- self.assertLess(md.index('abandoned'), md.index('| board |'))
-
- def test_custom_banner_is_used(self):
- with TemporaryDirectory() as td:
- hil_health.write_timeout_report(Path(td), [], 0, 'hil_report.md',
- banner='**refused to start.**\n')
- self.assertIn('refused to start', (Path(td) / 'hil_report.md').read_text())
-
- def test_unwritable_dir_does_not_raise(self):
- """The caller may be about to os._exit; losing the report must not also lose the
- exit path."""
- hil_health.write_timeout_report(Path('/proc/nonexistent/nope'), [], 0, 'x.md')
-
-
class WorkerSweepsItsOwnChildren(unittest.TestCase):
"""maxtasksperchild=1 makes a worker exit the moment its task returns, so by the time
main()'s finally sweeps, the strays have been reparented to init and are off the pool's
diff --git a/test/hil/test/test_hil_report.py b/test/hil/test/test_hil_report.py
new file mode 100644
index 000000000..9ab39bde3
--- /dev/null
+++ b/test/hil/test/test_hil_report.py
@@ -0,0 +1,1162 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: MIT
+# Unit tests for the report document: the vocabulary, the one cell classifier, rendering,
+# the four writers, and the fold to per-board verdicts. Split out of test_hil_bounded.py
+# and test_hil_health.py when the report code moved into helper/hil_report.py.
+# Run directly:
+# python3 test/hil/test/test_hil_report.py
+import json
+import os
+import subprocess
+import sys
+import unittest
+from pathlib import Path
+from tempfile import TemporaryDirectory
+
+TEST_DIR = os.path.dirname(os.path.abspath(__file__))
+HIL_DIR = os.path.dirname(TEST_DIR)
+sys.path.insert(0, HIL_DIR)
+
+from helper import hil_report
+
+
+class OneClassifierForBothArtifacts(unittest.TestCase):
+ """The markdown tally and the agent's verdict used to classify cells with two separate
+ copies of one rule -- hil_test's cell_kind against REPORT_CELL, and hil_summary's
+ cell_state against its own re-typed '❌'/'⚪' literals. Change the icons and the table
+ and the verdict silently disagree."""
+
+ def test_bare_states(self):
+ self.assertEqual(hil_report.cell_state('fail'), 'fail')
+ self.assertEqual(hil_report.cell_state('skip'), 'skip')
+ self.assertEqual(hil_report.cell_state('pass'), 'pass')
+
+ def test_icon_prefixed_metrics_carry_their_verdict(self):
+ self.assertEqual(hil_report.cell_state(f'{hil_report.REPORT_CELL["fail"]} 29/30'), 'fail')
+ self.assertEqual(hil_report.cell_state(f'{hil_report.REPORT_CELL["skip"]} board wedged'),
+ 'skip')
+
+ def test_an_unprefixed_metric_is_a_pass(self):
+ """Load-bearing: a passing test may return a plain metric string. Classifying
+ unknown shapes as fail would publish a green table as a red verdict."""
+ self.assertEqual(hil_report.cell_state('480.0 MBps'), 'pass')
+ self.assertEqual(hil_report.cell_state('1103 KB/s'), 'pass')
+
+ def test_a_non_string_cell_does_not_raise(self):
+ """render_matrix's copy guarded with isinstance; hil_summary's did not, because its
+ caller str()'d first. The merged one keeps the guard -- it is the safer superset."""
+ self.assertEqual(hil_report.cell_state(None), 'pass')
+
+ def test_the_icons_come_from_REPORT_CELL(self):
+ """No second copy of the emoji anywhere in the module."""
+ src = (Path(HIL_DIR) / 'helper' / 'hil_report.py').read_text(encoding='utf-8')
+ # CODE only: prose may quote an icon to explain a rule. The old assertion counted
+ # the single-quoted spelling `'❌'`, which a second copy written as "❌" would have
+ # sailed past.
+ code = '\n'.join(line.split('#', 1)[0] for line in src.splitlines())
+ for icon in ('❌', '⚪', '✅'):
+ self.assertEqual(code.count(icon), 1,
+ f'{icon} is spelled in code more than once; REPORT_CELL is'
+ f' meant to be the one source')
+
+
+class ModuleWorksImportedAndAsAScript(unittest.TestCase):
+ """It is imported as helper.hil_report by hil_test, and run as a script by the operator
+ (.claude/agents/hil-operator.md). A script run puts helper/ on sys.path, NOT test/hil,
+ so a plain `from helper import hil_health` breaks the CLI and only the CLI."""
+
+ def test_importable_as_a_package_module(self):
+ r = subprocess.run(
+ [sys.executable, '-c',
+ f'import sys; sys.path.insert(0, {HIL_DIR!r}); '
+ f'from helper import hil_report; print(hil_report.REPORT_JSON)'],
+ capture_output=True, text=True, timeout=60)
+ self.assertEqual(r.returncode, 0, r.stderr)
+ self.assertIn('hil_report.json', r.stdout)
+
+ def test_runnable_as_a_script(self):
+ r = subprocess.run(
+ [sys.executable, str(Path(HIL_DIR) / 'helper' / 'hil_report.py'), '--help'],
+ capture_output=True, text=True, timeout=60)
+ self.assertEqual(r.returncode, 0, r.stderr)
+
+
+class RenderReportIsPureFunctionOfTheDocument(unittest.TestCase):
+ """Four writers used to compose the markdown independently, so a table could carry
+ something the sidecar did not. One renderer, and the ordering it guarantees, is what
+ stops that -- pinned here rather than left to the order of three concatenations."""
+
+ def _doc(self, **kw):
+ d = {'rows': [{'board': 'boardA', 'cells': {'cdc_msc': 'pass'}, 'duration': '1s'}],
+ 'banner': '', 'scope': '', 'caveat': ''}
+ d.update(kw)
+ return d
+
+ def test_table_comes_from_rows(self):
+ md = hil_report.render_report(self._doc())
+ self.assertIn('boardA', md)
+ self.assertIn('cdc_msc', md)
+
+ def test_scope_note_appears_above_the_table(self):
+ md = hil_report.render_report(self._doc(scope='-b boardA'))
+ self.assertLess(md.index('Scoped run'), md.index('boardA'))
+
+ def test_banner_outranks_the_scope_note(self):
+ md = hil_report.render_report(self._doc(scope='-b boardA',
+ banner='> **Rig dirty.** x\n'))
+ self.assertLess(md.index('Rig dirty'), md.index('Scoped run'))
+
+ def test_caveat_is_outermost(self):
+ md = hil_report.render_report(self._doc(banner='> **Rig dirty.** x\n',
+ caveat='**HIL run abandoned.**\n'))
+ self.assertLess(md.index('abandoned'), md.index('Rig dirty'))
+
+ def test_a_document_with_no_rows_still_renders(self):
+ md = hil_report.render_report(self._doc(rows=[]))
+ self.assertIn('No tests were run.', md)
+
+ def test_a_malformed_row_does_not_raise(self):
+ """mark_report_abandoned renders a sidecar it did not write -- hil_ci.sh reuses a
+ persistent REMOTE_DIR, so it can be an older version's or a torn one -- and it runs
+ on the way to os._exit, where a KeyError hangs the runner in multiprocessing's
+ unbounded join() instead of freeing it."""
+ md = hil_report.render_report(self._doc(
+ rows=[{'board': 'boardA', 'cells': {'cdc_msc': 'pass'}}, {'board': 'half'},
+ {}]))
+ self.assertIn('boardA', md) # the intact row still renders ...
+ self.assertIn('half', md) # ... and a cell-less one becomes a blank row
+
+
+class ScopeSurvivesInTheJson(unittest.TestCase):
+ """A scoped run's small table is indistinguishable from a full run that lost boards.
+ The markdown says so; the JSON did not, so any JSON consumer could not tell."""
+
+ def _rows(self, board, cell):
+ return [(board, 0, 0, [(board, {cell: 'OK'}, '1s')], 0)]
+
+ def test_scope_is_recorded_in_the_sidecar(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(self._rows('boardA', 'cdc_msc'), rd, True,
+ '-b boardA', '')
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertEqual(doc['scope'], '-b boardA')
+
+ def test_an_unscoped_run_records_an_empty_scope(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(self._rows('boardA', 'cdc_msc'), rd, True, '', '')
+ self.assertEqual(json.loads((rd / hil_report.REPORT_JSON).read_text())['scope'], '')
+
+
+class EveryExitPathLeavesBothArtifacts(unittest.TestCase):
+ """summarize() builds an agent's verdicts from the JSON. A path that writes only
+ markdown reports the whole fleet as 'no report row' while a human sees the real story."""
+
+ def test_the_no_boards_exit_writes_json_too(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.write_report(rd, {'rows': [], 'banner': '', 'scope': '',
+ 'caveat': '**HIL run selected no boards.** why\n'})
+ self.assertIn('selected no boards', (rd / hil_report.REPORT_MD).read_text())
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertEqual(doc['rows'], [])
+ self.assertIn('selected no boards', doc['caveat'])
+
+ def test_write_report_raises_so_its_callers_can_report_it(self):
+ """write_report is NOT best-effort. Swallowing the OSError made
+ write_timeout_report's _p warning and hil_test's fallback-of-the-fallback dead
+ code -- an unwritable report dir produced no artifact and no message."""
+ with self.assertRaises(OSError):
+ hil_report.write_report(Path('/proc/nonexistent/nope'),
+ {'rows': [], 'banner': '', 'scope': '', 'caveat': 'x\n'})
+
+ def test_the_guarded_callers_still_do_not_raise(self):
+ """They are the ones on the way to os._exit, where a raise hangs the interpreter
+ in multiprocessing's unbounded join()."""
+ bad = Path('/proc/nonexistent/nope')
+ hil_report.mark_report_abandoned(bad, 'the worker pool would not shut down.')
+ hil_report.mark_report_no_boards(bad, 'filters intersected to nothing')
+ import io
+ from contextlib import redirect_stdout
+ with redirect_stdout(io.StringIO()):
+ hil_report.write_timeout_report(bad, [{'name': 'b1'}], 3600)
+
+
+class AbandonNoticeLandsInBothArtifacts(unittest.TestCase):
+ """_abandon_exit did a text prepend on a file it had not written, so the caveat never
+ reached the JSON and an agent reading the sidecar saw a clean partial report under a
+ red job."""
+
+ def test_abandon_sets_the_caveat_not_just_the_markdown(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('boardA', 0, 0, [('boardA', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertIn('abandoned', doc['caveat'])
+ self.assertEqual(len(doc['rows']), 1, 'the finished board must survive')
+ md = (rd / hil_report.REPORT_MD).read_text()
+ self.assertLess(md.index('abandoned'), md.index('boardA'))
+
+ def test_marking_a_missing_report_is_a_no_op(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ hil_report.mark_report_abandoned(Path(td.name), 'x') # must not raise
+
+ def test_a_sidecar_with_a_malformed_row_still_gets_stamped(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_JSON).write_text(json.dumps(
+ {'rows': [{'board': 'boardA'}], 'banner': '', 'scope': '', 'caveat': ''}))
+ hil_report.mark_report_abandoned(rd, 'x')
+ self.assertIn('abandoned',
+ json.loads((rd / hil_report.REPORT_JSON).read_text())['caveat'])
+ self.assertIn('abandoned', (rd / hil_report.REPORT_MD).read_text())
+
+ def test_a_torn_sidecar_is_a_no_op(self):
+ """This runs while the interpreter is being torn down: a raise here hangs the
+ process in multiprocessing's unbounded join()."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_JSON).write_text('{ truncated mid-')
+ hil_report.mark_report_abandoned(rd, 'x') # must not raise
+
+ def test_an_existing_abandon_caveat_is_not_overwritten(self):
+ """The pool-timeout path names the stuck boards and the rig-health verdict; this
+ one only knows the pool would not shut down. Whoever got there first wins --
+ the guard _abandon_exit used to spell as "'**HIL run ab' not in body[:2000]"."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600,
+ prefix='> **wedged usb_hub_wq worker.**\n')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertIn('timed out after 3600s', doc['caveat'])
+ self.assertIn('wedged usb_hub_wq worker', doc['banner']) # rig health, not outcome
+ self.assertNotIn('would not shut down', doc['caveat'])
+
+
+class CaveatSurvivesAccumulate(unittest.TestCase):
+ """CI reruns with --accumulate: the sidecar keeps every earlier attempt's cells, but the
+ banner was recomputed per attempt. A first attempt on a degraded rig and a clean rerun
+ therefore published the degraded attempt's PASSES with no caveat on them -- and the
+ generated .failed spec reruns only failures, so those cells are never re-earned."""
+
+ def _rows(self, board, cell):
+ return [(board, 0, 0, [(board, {cell: 'OK'}, '1s')], 0)]
+
+ def test_an_earlier_attempts_caveat_is_still_on_the_report(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ banner = '> **Rig note.** 2 process(es) in D state at start.\n'
+
+ hil_report.accumulate_report(self._rows('boardA', 'cdc_msc'), rd, True, '', banner)
+ self.assertIn('Rig note', (rd / hil_report.REPORT_MD).read_text())
+
+ # the rerun: clean rig, so this attempt contributes no banner of its own
+ md = hil_report.accumulate_report(self._rows('boardB', 'cdc_msc'), rd, False, '', '')
+ self.assertIn('boardA', md) # the earlier cells are kept ...
+ self.assertIn('Rig note', md,
+ 'the caveat the earlier cells were collected under was dropped')
+
+ def test_the_same_caveat_twice_is_not_stacked(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ banner = '> **Rig note.** 2 process(es) in D state at start.\n'
+ hil_report.accumulate_report(self._rows('boardA', 'cdc_msc'), rd, True, '', banner)
+ md = hil_report.accumulate_report(self._rows('boardB', 'cdc_msc'), rd, False, '', banner)
+ self.assertEqual(md.count('Rig note'), 1)
+
+
+class MarkdownIsAlwaysARenderingOfTheJson(unittest.TestCase):
+ """The property this whole change buys: whatever wrote the report, re-rendering the
+ sidecar reproduces the markdown byte for byte. Four writers, one renderer -- asserted
+ directly rather than inferred from the writers."""
+
+ def _check(self, rd):
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertEqual((rd / hil_report.REPORT_MD).read_text(),
+ hil_report.render_report(doc) + '\n')
+
+ def test_normal_path(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('boardA', 0, 0, [('boardA', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True,
+ '-b boardA', '> **Rig note.** x\n')
+ self._check(rd)
+
+ def test_after_an_accumulate_rerun(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('boardA', 0, 0, [('boardA', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
+ hil_report.accumulate_report(
+ [('boardB', 0, 0, [('boardB', {'cdc_msc': 'OK'}, '1s')], 0)], rd, False, '', '')
+ self._check(rd)
+
+ def test_after_abandonment(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('boardA', 0, 0, [('boardA', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ self._check(rd)
+
+ def test_no_boards_exit(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.write_report(rd, {'rows': [], 'banner': '', 'scope': '',
+ 'caveat': '**HIL run selected no boards.** why\n'})
+ self._check(rd)
+
+ def test_the_pool_guard_fallback(self):
+ """The last writer to join the invariant: it composed its own markdown only because
+ hil_health could not import the renderer."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('done', 0, 0, [('done', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600,
+ prefix='> **wedged usb_hub_wq worker.**\n')
+ self._check(rd)
+
+class WriteTimeoutReport(unittest.TestCase):
+ def test_prefix_carries_the_preflight_diagnosis(self):
+ """The timeout aborts before accumulate_report, so without the prefix the artifact
+ and the PR comment lose the one line saying WHY the pool never finished."""
+ with TemporaryDirectory() as td:
+ d = Path(td)
+ hil_report.write_timeout_report(d, [{'name': 'b1'}], 4200,
+ prefix='> **wedged usb_hub_wq worker.**\n')
+ out = (d / hil_report.REPORT_MD).read_text()
+ # the abandon notice leads (run outcome), the rig-health prefix follows in the
+ # banner -- prefix used to be folded INTO the caveat, which is what made a clean
+ # --accumulate retry inherit an abandonment that had not happened
+ self.assertTrue(out.startswith('**HIL run abandoned: worker pool timed out'), out[:80])
+ self.assertIn('> **wedged usb_hub_wq worker.**', out)
+ self.assertIn('timed out after 4200s', out)
+ self.assertIn('- b1', out)
+
+ def test_writes_a_report_where_there_would_be_none(self):
+ with TemporaryDirectory() as td:
+ hil_report.write_timeout_report(Path(td), [{'name': 'ra6m5_ek'}], 4200)
+ md = (Path(td) / hil_report.REPORT_MD).read_text()
+ self.assertIn('4200s', md)
+ self.assertIn('ra6m5_ek', md)
+
+ def test_the_prior_attempts_rows_survive(self):
+ """Was: the prior MARKDOWN TEXT survives below the banner. It now re-renders from
+ the merged sidecar, so the guarantee is stated against rows -- one table with the
+ stuck boards in it, rather than a banner stapled above a duplicate table."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('done', 0, 0, [('done', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600)
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertEqual([r['board'] for r in doc['rows']], ['done', 'stuck'])
+ md = (rd / hil_report.REPORT_MD).read_text()
+ self.assertIn('done', md)
+ self.assertIn('stuck', md)
+ self.assertIn('abandoned', md)
+ self.assertLess(md.index('abandoned'), md.index('done'))
+ self.assertEqual(md.count('| Board'), 1, 'the prior table was duplicated, not merged')
+
+ def test_custom_banner_is_used(self):
+ with TemporaryDirectory() as td:
+ hil_report.write_timeout_report(Path(td), [], 0,
+ banner='**refused to start.**\n')
+ self.assertIn('refused to start', (Path(td) / hil_report.REPORT_MD).read_text())
+
+ def test_timeout_report_writes_the_sidecar(self):
+ """This path used to write markdown only, so summarize() -- which is all an
+ agent gets -- reported the whole fleet as 'no report row' on exactly the runs
+ that failed."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.write_timeout_report(rd, [{'name': 'boardA'}], 3600)
+ self.assertTrue((rd / hil_report.REPORT_JSON).is_file())
+ self.assertIn('boardA', (rd / hil_report.REPORT_JSON).read_text())
+
+ def test_the_sidecar_keeps_a_previous_attempts_rows(self):
+ """An earlier attempt's finished boards are real results and this attempt has none
+ of its own, so the rows merge rather than replace."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_JSON).write_text(json.dumps(
+ {'rows': [{'board': 'done', 'cells': {'cdc_msc': 'pass'}, 'duration': '1s'}],
+ 'banner': '', 'scope': '', 'caveat': ''}))
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600)
+ rows = json.loads((rd / hil_report.REPORT_JSON).read_text())['rows']
+ self.assertEqual([r['board'] for r in rows], ['done', 'stuck'])
+
+ def test_a_torn_sidecar_does_not_lose_the_stuck_boards(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_JSON).write_text('{ truncated mid-')
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600)
+ rows = json.loads((rd / hil_report.REPORT_JSON).read_text())['rows']
+ self.assertEqual([r['board'] for r in rows], ['stuck'])
+
+ def test_a_roster_entry_without_a_name_does_not_escape(self):
+ """The broad handler exists to stop a KeyError here stranding the runner, but a
+ report that silently loses its only board is worse than one saying '?'."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.write_timeout_report(rd, [{}], 3600)
+ rows = json.loads((rd / hil_report.REPORT_JSON).read_text())['rows']
+ self.assertEqual([r['board'] for r in rows], ['?'])
+
+ def test_unwritable_dir_does_not_raise(self):
+ """The caller may be about to os._exit; losing the report must not also lose the
+ exit path."""
+ hil_report.write_timeout_report(Path('/proc/nonexistent/nope'), [], 0)
+
+
+class SummaryFoldsReportToBoards(unittest.TestCase):
+ """summarize() replaces the agent retyping the markdown table. Report rows are named per
+ VARIANT and a variant need not start with the board name, so the config is what maps them
+ back -- the previous string-matching design produced a defect in each of four review rounds."""
+
+ def _sum(self, boards, rows, cfg_boards=None, banner=''):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ d = Path(td.name)
+ (d / 'hil_report.json').write_text(json.dumps(
+ {'rows': [{'board': b, 'cells': c, 'duration': '1s'} for b, c in rows],
+ 'banner': banner}))
+ cfg = d / 'cfg.json'
+ cfg.write_text(json.dumps({'boards': cfg_boards or [{'name': b} for b in boards]}))
+ args = [a for b in boards for a in ('-b', b)]
+ r = subprocess.run(['python3', str(Path(TEST_DIR).parents[0] / 'helper' / 'hil_report.py'),
+ str(cfg), *args, '--report-dir', str(d)],
+ capture_output=True, text=True, timeout=60)
+ self.assertEqual(r.returncode, 0, r.stderr)
+ return json.loads(r.stdout)['results']
+
+ def test_variant_rows_fold_onto_their_board(self):
+ """nanoch32v203 never produces a row named after the board."""
+ got = self._sum(['nanoch32v203'],
+ [('nanoch32v203-fsdev', {'usbtest': 'pass'}),
+ ('nanoch32v203-usbfs', {'usbtest': 'pass'})],
+ cfg_boards=[{'name': 'nanoch32v203',
+ 'variant': [{'name': 'nanoch32v203-fsdev'},
+ {'name': 'nanoch32v203-usbfs'}]}])
+ self.assertEqual([r['board'] for r in got], ['nanoch32v203'])
+ self.assertTrue(got[0]['pass'])
+ self.assertTrue(got[0]['ran'])
+
+ def test_one_failing_variant_fails_the_board(self):
+ got = self._sum(['nano'],
+ [('nano-a', {'usbtest': 'pass'}), ('nano-b', {'usbtest': '❌ 29/30'})],
+ cfg_boards=[{'name': 'nano', 'variant': [{'name': 'nano-a'},
+ {'name': 'nano-b'}]}])
+ self.assertFalse(got[0]['pass'])
+ self.assertIn('29/30', got[0]['detail'])
+
+ def test_lock_contention_is_a_field_not_a_prefix(self):
+ got = self._sum(['alpha'], [('alpha', {'board-locked': 'fail'})])
+ self.assertTrue(got[0]['locked'])
+ self.assertFalse(got[0]['pass'])
+
+ def test_a_board_with_no_row_is_marked_not_run(self):
+ got = self._sum(['alpha', 'beta'], [('alpha', {'usbtest': 'pass'})])
+ self.assertTrue(got[0]['ran'])
+ self.assertFalse(got[1]['ran'])
+ self.assertFalse(got[1]['pass'])
+
+ def test_a_metric_cell_counts_by_its_icon(self):
+ got = self._sum(['a', 'b'], [('a', {'cdc_msc_throughput': '✅ C 1.2 M 3.4'}),
+ ('b', {'cdc_msc_throughput': '❌ C 0.0 M 0.0'})])
+ self.assertTrue(got[0]['pass'])
+ self.assertFalse(got[1]['pass'])
+
+ def test_skipped_cells_do_not_fail_a_board(self):
+ got = self._sum(['a'], [('a', {'usbtest': 'skip', 'cdc_msc': 'pass'})])
+ self.assertTrue(got[0]['pass'])
+
+ def test_a_plain_metric_cell_is_a_pass(self):
+ """Mirrors hil_test.py's own tally (cell_kind): failures are ALWAYS marked -- 'fail'
+ or a ❌ prefix, per TestFail's docstring -- while a passing test may return a plain
+ metric string that lands in the cell unprefixed. Treating unknown shapes as fail
+ would publish a green table as a red verdict."""
+ got = self._sum(['a'], [('a', {'device_speed': '480.0 MBps'})])
+ self.assertTrue(got[0]['pass'])
+
+ def test_a_declared_variant_of_another_board_is_not_stolen(self):
+ """A declared variant need not start with its own board's name, so it may start with
+ a DIFFERENT board's name plus '-'. The prefix fallback must not attribute it twice."""
+ got = self._sum(['alpha', 'beta'],
+ [('beta-x', {'usbtest': 'fail'})],
+ cfg_boards=[{'name': 'alpha', 'variant': [{'name': 'beta-x'}]},
+ {'name': 'beta'}])
+ self.assertTrue(got[0]['ran'])
+ self.assertFalse(got[0]['pass'])
+ self.assertFalse(got[1]['ran'], "beta must not inherit alpha's row")
+
+
+ def test_the_caveat_reaches_the_agents_verdict(self):
+ """The abandon/no-boards notice lives in the document now, and this JSON is all an
+ agent gets -- dropping it here puts the caveat back where only a human sees it."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ d = Path(td.name)
+ (d / 'hil_report.json').write_text(json.dumps(
+ {'rows': [{'board': 'boardA', 'cells': {'cdc_msc': 'pass'}, 'duration': '1s'}],
+ 'banner': '', 'scope': '',
+ 'caveat': '**HIL run abandoned: the worker pool would not shut down.**\n'}))
+ cfg = d / 'cfg.json'
+ cfg.write_text(json.dumps({'boards': [{'name': 'boardA'}]}))
+ r = subprocess.run(
+ ['python3', str(Path(TEST_DIR).parents[0] / 'helper' / 'hil_report.py'),
+ str(cfg), '-b', 'boardA', '--report-dir', str(d)],
+ capture_output=True, text=True, timeout=60)
+ self.assertEqual(r.returncode, 0, r.stderr)
+ self.assertIn('abandoned', json.loads(r.stdout)['caveat'])
+
+ def test_an_older_sidecar_without_a_caveat_still_summarises(self):
+ got = self._sum(['boardA'], [('boardA', {'cdc_msc': 'pass'})])
+ self.assertTrue(got[0]['pass'])
+
+ def test_the_old_entry_point_is_gone(self):
+ """hil_summary.py's CLI moved here. A leftover file would keep working while
+ drifting from the module that now owns the fold."""
+ self.assertFalse((Path(HIL_DIR) / 'helper' / 'hil_summary.py').exists())
+
+
+class AbandonStampIsNotDestructive(unittest.TestCase):
+ """mark_report_abandoned runs on the way to os._exit, on a report it did not write.
+ Every case here was a live regression found by review."""
+
+ def _doc(self, **kw):
+ d = {'rows': [{'board': 'OLD', 'cells': {'t': 'pass'}, 'duration': '9s'}],
+ 'banner': '', 'scope': '', 'caveat': ''}
+ d.update(kw)
+ return d
+
+ def test_declining_to_stamp_does_not_republish_the_markdown(self):
+ """The guard skipped the caveat assignment but write_report ran anyway, so a
+ no-op call still overwrote THIS run's table with a re-render of an older sidecar."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_JSON).write_text(json.dumps(self._doc(
+ caveat='**HIL run abandoned: worker pool timed out after 3600s.**\n')))
+ (rd / hil_report.REPORT_MD).write_text('THIS RUN table with boardX\n')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ self.assertEqual((rd / hil_report.REPORT_MD).read_text(),
+ 'THIS RUN table with boardX\n')
+
+ def test_a_banner_borne_abandon_notice_also_wins(self):
+ """The pool-timeout path puts its notice in `banner` (hil_test.py:2300), not
+ `caveat`. SKILL.md gives the two notices OPPOSITE rules, so stamping the vaguer
+ one on top tells the agent to publish rows it is meant to discard."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('a', 0, 0, [('a', {'t': 'OK'}, '1s')], 0)], rd, True, '', '',
+ caveat='**HIL run abandoned: worker pool timed out after 3600s.** 2 never'
+ ' reported.\n')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ md = (rd / hil_report.REPORT_MD).read_text()
+ self.assertTrue(md.startswith('**HIL run abandoned: worker pool timed out'), md[:80])
+ self.assertNotIn('would not shut down', md)
+
+ def test_a_missing_sidecar_still_stamps_the_markdown(self):
+ """Master read the MARKDOWN and prepended unconditionally, so it always stamped.
+ pr_comment.yml cats only hil_report.md -- giving up here publishes a clean green
+ table under an abandoned, non-zero job."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_MD).write_text('**✅ 27 passed · ❌ 0 failed**\n')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ md = (rd / hil_report.REPORT_MD).read_text(encoding='utf-8')
+ self.assertIn('abandoned', md)
+ self.assertIn('27 passed', md)
+
+ def test_a_torn_sidecar_still_stamps_the_markdown(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_JSON).write_text('{ truncated mid-')
+ (rd / hil_report.REPORT_MD).write_text('**✅ 27 passed**\n')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ self.assertIn('abandoned', (rd / hil_report.REPORT_MD).read_text(encoding='utf-8'))
+
+ def test_the_wording_matches_the_skill_contract(self):
+ """SKILL.md pins this banner as 'the table below IS this run's ... Report the
+ results AND the abandonment'. Calling it 'partial' sends the agent to re-run
+ boards that already passed."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_JSON).write_text(json.dumps(self._doc()))
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ caveat = json.loads((rd / hil_report.REPORT_JSON).read_text())['caveat']
+ self.assertNotIn('partial', caveat)
+ self.assertIn('unverified', caveat)
+
+
+class WriteReportFailsLoudly(unittest.TestCase):
+ def test_a_render_failure_does_not_leave_a_committed_json(self):
+ """It wrote the JSON, then rendered. A render raise left the sidecar saying
+ 'abandoned' beside a markdown that still read as a clean green table -- breaking
+ the one invariant this module exists to hold."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_MD).write_text('STALE GREEN TABLE\n')
+ (rd / hil_report.REPORT_JSON).write_text(json.dumps(
+ {'rows': ['boardA'], 'banner': '', 'scope': '', 'caveat': ''}))
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ md = (rd / hil_report.REPORT_MD).read_text()
+ # either both moved or neither did -- never a sidecar the markdown contradicts
+ self.assertEqual('abandoned' in doc.get('caveat', ''), 'abandoned' in md,
+ 'the sidecar was committed without its markdown')
+
+ def test_a_non_dict_row_does_not_cost_the_abandon_stamp(self):
+ """A row that is a bare string raised out of render_report, so the stamp was lost
+ entirely -- the failure mode this whole function exists to prevent."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_JSON).write_text(json.dumps(
+ {'rows': ['boardA', {'board': 'good', 'cells': {'t': 'pass'}}],
+ 'banner': '', 'scope': '', 'caveat': ''}))
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ md = (rd / hil_report.REPORT_MD).read_text()
+ self.assertIn('abandoned', md)
+ self.assertIn('good', md)
+
+ def test_an_unwritable_dir_reaches_the_callers_warning(self):
+ """write_report swallowing OSError made write_timeout_report's broad handler --
+ and hil_test's fallback-of-the-fallback -- dead code: no artifact, no message."""
+ import io
+ from contextlib import redirect_stdout
+ buf = io.StringIO()
+ with redirect_stdout(buf):
+ hil_report.write_timeout_report(Path('/proc/nonexistent/nope'),
+ [{'name': 'b1'}], 3600, prefix='x\n')
+ self.assertIn('warning', buf.getvalue().lower(), 'the failure was silent')
+
+
+class PoolTimeoutCellIsHonest(unittest.TestCase):
+ def test_a_stuck_board_with_a_prior_row_still_gets_the_cell(self):
+ """`not in done` skipped the cell for any board carrying an earlier attempt's row,
+ so a board that just ate the 60-minute guard summarized as pass:true."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('stm32f4', 0, 0, [('stm32f4', {'cdc_msc': 'OK'}, '1s')], 0)], rd, True, '', '')
+ hil_report.write_timeout_report(rd, [{'name': 'stm32f4'}], 3600)
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ verdict = hil_report.summarize({'boards': [{'name': 'stm32f4'}]}, ['stm32f4'], doc)
+ self.assertFalse(verdict['results'][0]['pass'],
+ 'a board that hung the pool was published as a pass')
+
+ def test_a_clean_retry_clears_the_cell(self):
+ """accumulate_report clears stale board-locked and BOUNDARY_CELL cells but not
+ this one, so a board that passed clean on the retry stayed red forever."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600)
+ hil_report.accumulate_report(
+ [('stuck', 0, 0, [('stuck', {'cdc_msc': 'OK'}, '2s')], 0)], rd, False, '', '')
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertNotIn('pool-timeout', doc['rows'][0]['cells'])
+ verdict = hil_report.summarize({'boards': [{'name': 'stuck'}]}, ['stuck'], doc)
+ self.assertTrue(verdict['results'][0]['pass'])
+
+ def test_a_torn_sidecar_does_not_destroy_an_intact_markdown(self):
+ """Re-rendering from an unusable sidecar threw away real results the human copy
+ still had. Master concatenated below its banner and kept them."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_MD).write_text('| Board | t |\n| a | OK |\n| b | OK |\n')
+ (rd / hil_report.REPORT_JSON).write_text('{ truncated')
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600)
+ md = (rd / hil_report.REPORT_MD).read_text()
+ self.assertIn('| a | OK |', md, "an earlier attempt's real results were destroyed")
+ self.assertIn('stuck', md)
+
+
+class SummarizeSeesEveryRow(unittest.TestCase):
+ def test_board_name_rows_reach_a_variant_boards_verdict(self):
+ """hil_test writes lock-contention and pool-timeout rows keyed by BOARD name, but
+ variants_of returns only declared variant names -- so for nanoch32v203 and
+ ch32v307v_r1_1v0 those rows were invisible and a held lock published as a
+ hardware FAIL that hil-validate.js never retried."""
+ cfg = {'boards': [{'name': 'nano',
+ 'variant': [{'name': 'nano-fsdev'}, {'name': 'nano-usbfs'}]}]}
+ doc = {'rows': [{'board': 'nano', 'cells': {'board-locked': 'fail'},
+ 'duration': None}], 'banner': '', 'scope': '', 'caveat': ''}
+ r = hil_report.summarize(cfg, ['nano'], doc)['results'][0]
+ self.assertTrue(r['ran'])
+ self.assertTrue(r['locked'], 'a held lock was published as a hardware failure')
+
+ def test_a_malformed_row_does_not_kill_the_cli(self):
+ """summarize is the one reader with no defense, and it is the only one an agent's
+ verdict depends on."""
+ out = hil_report.summarize({'boards': [{'name': 'a'}]}, ['a'],
+ {'rows': [{'cells': {}}, {'board': 'a',
+ 'cells': {'t': 'pass'}}]})
+ self.assertTrue(out['results'][0]['pass'])
+
+
+class NoBoardsExitKeepsWhatRan(unittest.TestCase):
+ def test_it_does_not_wipe_an_accumulated_sidecar(self):
+ """Master wrote only markdown here, so the sidecar survived. Writing rows:[]
+ unconditionally makes an --accumulate rerun whose filters empty erase every
+ board that had already passed."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('a', 0, 0, [('a', {'t': 'OK'}, '1s')], 0)], rd, True, '', '')
+ hil_report.mark_report_no_boards(rd, 'No boards left after the flasher filter',
+ fresh=False)
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertEqual([r['board'] for r in doc['rows']], ['a'])
+ self.assertIn('selected no boards', doc['caveat'])
+ self.assertIn('selected no boards', (rd / hil_report.REPORT_MD).read_text())
+
+
+class TheMergeBehavioursAreActuallyPinned(unittest.TestCase):
+ """accumulate_report's docstring cites these three as the reason not to split it, yet
+ deleting any of them left the whole suite green. Mutation-verified."""
+
+ def test_a_cleared_boundary_drops_the_previous_attempts_mark(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('b', 0, 0, [('b-v', {hil_report.BOUNDARY_CELL: 'fail'}, '1s')], 0)],
+ rd, True, '', '')
+ hil_report.accumulate_report(
+ [('b', 0, 0, [('b-v', {'cdc_msc': 'OK'}, '2s')], 0)], rd, False, '', '')
+ cells = json.loads((rd / hil_report.REPORT_JSON).read_text())['rows'][0]['cells']
+ self.assertNotIn(hil_report.BOUNDARY_CELL, cells)
+
+ def test_a_board_that_really_ran_drops_its_stale_lock_cell(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('b', 0, 0, [('b', {hil_report.LOCKED_CELL: 'fail'}, None)], 0)], rd, True, '', '')
+ hil_report.accumulate_report(
+ [('b', 0, 0, [('b', {'cdc_msc': 'OK'}, '2s')], 0)], rd, False, '', '')
+ rows = json.loads((rd / hil_report.REPORT_JSON).read_text())['rows']
+ self.assertEqual([r['board'] for r in rows], ['b'])
+ self.assertNotIn(hil_report.LOCKED_CELL, rows[0]['cells'])
+
+ def test_a_filtered_rerun_keeps_the_previous_duration(self):
+ """A -t-filtered re-run reports duration None; blanking the column loses the only
+ record of how long the full run took."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('b', 0, 0, [('b', {'cdc_msc': 'OK'}, '119s')], 0)], rd, True, '', '')
+ hil_report.accumulate_report(
+ [('b', 0, 0, [('b', {'cdc_msc': 'OK'}, None)], 0)], rd, False, '', '')
+ self.assertEqual(json.loads(
+ (rd / hil_report.REPORT_JSON).read_text())['rows'][0]['duration'], '119s')
+
+
+class TheFooterCountsAreNotSwapped(unittest.TestCase):
+ """SKILL.md tells the operator to paste the footer counts verbatim, and swapping the
+ failed/skipped tallies left the suite green."""
+
+ def test_each_kind_is_counted_under_its_own_label(self):
+ md = hil_report.render_matrix([
+ ('b', {'p1': 'pass', 'p2': 'pass', 'f1': 'fail',
+ 's1': f'{hil_report.REPORT_CELL["skip"]} board wedged'}, '1s')])
+ self.assertIn(f'{hil_report.REPORT_CELL["pass"]} 2 passed', md)
+ self.assertIn(f'{hil_report.REPORT_CELL["fail"]} 1 failed', md)
+ self.assertIn(f'{hil_report.REPORT_CELL["skip"]} 1 skipped', md)
+
+
+class HilCiUploadsTheAccumulateMergeBase(unittest.TestCase):
+ """hil_ci.sh rm -rf's REMOTE_DIR at the start of every run, and accumulate_report
+ merges onto the sidecar in the run's cwd -- so without an upload a remote
+ `--accumulate` retry silently starts from nothing and its one-row table REPLACES the
+ full-fleet one. The copy-back at the end has always existed; the upload did not."""
+
+ def _gate(self, *args):
+ """Run the real gate block out of hil_ci.sh and return its ACCUMULATE verdict.
+
+ Executed, not grepped: the previous pair of tests searched the source text and
+ stayed green when `if [ "$ACCUMULATE" = 1 ]` was mutated to `if true`, because the
+ comment block above it mentions --accumulate five times."""
+ sh = (Path(HIL_DIR) / 'hil_ci.sh').read_text(encoding='utf-8')
+ a = sh.index('ACCUMULATE=$(python3 -')
+ b = sh.index(') || ACCUMULATE=0', a) + len(') || ACCUMULATE=0')
+ script = 'ARGS=("$@")\n' + sh[a:b] + '\necho "$ACCUMULATE"'
+ r = subprocess.run(['bash', '-c', script, '_', *args],
+ capture_output=True, text=True, timeout=60)
+ self.assertEqual(r.returncode, 0, r.stderr)
+ return r.stdout.strip()
+
+ def test_every_spelling_argparse_accepts_is_detected(self):
+ """hil_test.py declares `-a, --accumulate`, so argparse also takes -av, -va,
+ --accum and --acc; hil-validate.js tells the operator to retry 'adding -v'."""
+ for spelling in ('--accumulate', '-a', '-av', '-va', '--accum', '--acc'):
+ self.assertEqual(self._gate(spelling), '1', f'{spelling} was not detected')
+
+ def test_a_run_without_it_is_not_treated_as_accumulate(self):
+ for spelling in ('-b', '-v', '--retry'):
+ self.assertEqual(self._gate(spelling), '0', f'{spelling} falsely detected')
+
+ def test_the_sidecar_is_uploaded_and_gated(self):
+ sh = (Path(HIL_DIR) / 'hil_ci.sh').read_text(encoding='utf-8')
+ up = [ln for ln in sh.splitlines()
+ if 'scp' in ln and 'hil_report.json' in ln and '$REMOTE:' in ln]
+ self.assertTrue(up, 'nothing uploads hil_report.json; --accumulate has no merge base')
+ self.assertIn('if [ "$ACCUMULATE" = 1 ]', sh, 'the upload is not gated')
+
+ def test_a_missing_merge_base_is_loud(self):
+ """The damage: --accumulate with nothing to merge onto succeeds and quietly
+ publishes a small table where a full one used to be."""
+ warn = [ln for ln in (Path(HIL_DIR) / 'hil_ci.sh').read_text().splitlines()
+ if 'warning' in ln.lower() and 'accumulate' in ln.lower()]
+ self.assertTrue(warn, 'no warning when --accumulate has no local sidecar')
+
+
+class RunOutcomeAndRigHealthAreSeparate(unittest.TestCase):
+ """`banner` describes the CONDITIONS cells were collected under, so it carries across a
+ retry. `caveat` describes how a RUN ENDED, so it must not: a clean retry that reports
+ an earlier attempt's abandonment tells the agent a green run failed."""
+
+ def test_a_clean_retry_drops_the_previous_abandon_notice(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report([('a', 0, 0, [('a', {'t': 'OK'}, '1s')], 0)],
+ rd, True, '', '')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ hil_report.accumulate_report([('a', 0, 0, [('a', {'t': 'OK'}, '2s')], 0)],
+ rd, False, '', '')
+ self.assertEqual(json.loads((rd / hil_report.REPORT_JSON).read_text())['caveat'], '')
+
+ def test_rig_health_still_carries_across_the_retry(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.write_timeout_report(rd, [{'name': 's'}], 3600,
+ prefix='> **Rig note.** wedged\n')
+ hil_report.accumulate_report([('a', 0, 0, [('a', {'t': 'OK'}, '1s')], 0)],
+ rd, False, '', '')
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertIn('Rig note', doc['banner'])
+
+ def test_a_second_attempts_abandon_is_recorded(self):
+ """_already_abandoned matched a notice carried forward from an EARLIER attempt, so
+ a genuinely new abandon wrote nothing and the run's own failure vanished."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report(
+ [('a', 0, 0, [('a', {'t': 'OK'}, '1s')], 0)], rd, True, '',
+ '> **Rig note.** x\n',
+ caveat='**HIL run abandoned: worker pool timed out after 3600s.**\n')
+ hil_report.accumulate_report([('a', 0, 0, [('a', {'t': 'OK'}, '2s')], 0)],
+ rd, False, '', '')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ self.assertIn('would not shut down',
+ json.loads((rd / hil_report.REPORT_JSON).read_text())['caveat'])
+
+
+class AMalformedSidecarNeverCostsTheReport(unittest.TestCase):
+ """hil_ci.sh now uploads a sidecar as the merge base, so a non-conforming one is
+ reachable from outside the harness."""
+
+ def _write(self, rd, doc):
+ (rd / hil_report.REPORT_JSON).write_text(json.dumps(doc))
+
+ def test_a_null_banner_does_not_kill_a_successful_run(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ self._write(rd, {'rows': [{'board': 'a', 'cells': {'t': 'pass'}, 'duration': '1s'}],
+ 'banner': None, 'caveat': None, 'scope': ''})
+ hil_report.accumulate_report([('b', 0, 0, [('b', {'t': 'OK'}, '1s')], 0)],
+ rd, False, '', '')
+ self.assertTrue((rd / hil_report.REPORT_MD).is_file())
+
+ def test_a_null_cells_row_still_gets_its_pool_timeout_cell(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ self._write(rd, {'rows': [{'board': 'boardA', 'cells': None, 'duration': '61s'}],
+ 'banner': '', 'caveat': '', 'scope': ''})
+ hil_report.write_timeout_report(rd, [{'name': 'boardA'}], 3600)
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ v = hil_report.summarize({'boards': [{'name': 'boardA'}]}, ['boardA'], doc)
+ self.assertFalse(v['results'][0]['pass'],
+ 'a board that ate the whole pool guard was published as a pass')
+
+ def test_an_awkward_sidecar_still_gets_the_abandon_stamp(self):
+ """Any raise inside the dict branch was swallowed and the markdown fallback was
+ unreachable, so the stamp was lost from BOTH artifacts."""
+ for bad in ({'rows': [{'board': 'a', 'cells': {'t': 'p'}, 'duration': 120}],
+ 'banner': '', 'caveat': '', 'scope': ''},
+ {'rows': [{'board': 'a', 'cells': {'t': ['x']}, 'duration': '1s'}],
+ 'banner': None, 'caveat': '', 'scope': ''}):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ self._write(rd, bad)
+ (rd / hil_report.REPORT_MD).write_text('**✅ 27 passed**\n')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ self.assertIn('abandoned', (rd / hil_report.REPORT_MD).read_text(encoding='utf-8'),
+ f'no stamp for {bad}')
+
+ def test_a_malformed_roster_entry_still_leaves_an_artifact(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ import io
+ from contextlib import redirect_stdout
+ with redirect_stdout(io.StringIO()):
+ hil_report.write_timeout_report(rd, ['plainstring'], 3600)
+ self.assertTrue((rd / hil_report.REPORT_MD).is_file(), 'no artifact at all')
+
+
+class PoolTimeoutOutranksAStaleLock(unittest.TestCase):
+ def test_a_wedge_is_not_published_as_lock_contention(self):
+ """locked was computed across every cell and short-circuited detail, so a board
+ that wedged the rig on the retry was reported as LOCKED -- and hil-validate.js
+ re-runs those, paying another pool guard on a board that just hung it."""
+ doc = {'rows': [{'board': 'boardX',
+ 'cells': {'board-locked': 'fail', 'pool-timeout': 'fail'},
+ 'duration': None}], 'banner': '', 'caveat': '', 'scope': ''}
+ r = hil_report.summarize({'boards': [{'name': 'boardX'}]}, ['boardX'], doc)['results'][0]
+ self.assertFalse(r['locked'], 'a wedge was published as lock contention')
+ self.assertFalse(r['pass'])
+
+
+class NoBoardsExitRespectsFreshness(unittest.TestCase):
+ def test_a_fresh_run_does_not_republish_the_previous_rows(self):
+ """It is called BEFORE the fresh wipe, so it re-published last run's green table
+ under this run's red job -- the stale-table failure it exists to prevent."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report([('a', 0, 0, [('a', {'t': 'OK'}, '1s')], 0)],
+ rd, True, '', '')
+ hil_report.mark_report_no_boards(rd, 'filters emptied', fresh=True)
+ self.assertEqual(json.loads((rd / hil_report.REPORT_JSON).read_text())['rows'], [])
+
+ def test_an_accumulate_run_keeps_them(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report([('a', 0, 0, [('a', {'t': 'OK'}, '1s')], 0)],
+ rd, True, '', '')
+ hil_report.mark_report_no_boards(rd, 'filters emptied', fresh=False)
+ self.assertEqual([r['board'] for r in json.loads(
+ (rd / hil_report.REPORT_JSON).read_text())['rows']], ['a'])
+
+ def test_it_does_not_overwrite_an_abandon_notice(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report([('a', 0, 0, [('a', {'t': 'OK'}, '1s')], 0)],
+ rd, True, '', '')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ hil_report.mark_report_no_boards(rd, 'filters emptied', fresh=False)
+ self.assertIn('abandoned',
+ json.loads((rd / hil_report.REPORT_JSON).read_text())['caveat'])
+
+
+class TheNoBoardsCallSiteIsWired(unittest.TestCase):
+ """The fresh/accumulate branches of mark_report_no_boards were tested by calling it
+ DIRECTLY, so both passed while hil_test.py's one real call site never passed the flag
+ at all -- an --accumulate run whose filter emptied still wiped the accumulated rows.
+ This drives hil_test.py itself; the no-boards exit needs only a config and a filter
+ that matches nothing, so it costs no hardware."""
+
+ def _run(self, *extra):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / 'cfg.json').write_text(json.dumps(
+ {'boards': [{'name': 'alpha', 'uid': '1', 'flasher': {'name': 'jlink', 'uid': '2'}}]}))
+ (rd / hil_report.REPORT_JSON).write_text(json.dumps(
+ {'rows': [{'board': 'earlier', 'cells': {'t': 'pass'}, 'duration': '1s'}],
+ 'banner': '', 'scope': '', 'caveat': ''}))
+ r = subprocess.run(
+ [sys.executable, str(Path(HIL_DIR) / 'hil_test.py'),
+ '--flasher', 'nonexistent', *extra, str(rd / 'cfg.json')],
+ capture_output=True, text=True, timeout=120,
+ env={**os.environ, 'HIL_REPORT_DIR': str(rd)})
+ self.assertEqual(r.returncode, 1, r.stdout + r.stderr)
+ return json.loads((rd / hil_report.REPORT_JSON).read_text())
+
+ def test_an_accumulate_run_keeps_the_accumulated_rows(self):
+ doc = self._run('--accumulate')
+ self.assertEqual([r['board'] for r in doc['rows']], ['earlier'],
+ "the call site did not pass fresh=not args.accumulate")
+ self.assertIn('selected no boards', doc['caveat'])
+
+ def test_a_fresh_run_does_not_republish_them(self):
+ doc = self._run()
+ self.assertEqual(doc['rows'], [])
+ self.assertIn('selected no boards', doc['caveat'])
+
+
+class EveryWriterRendersBeforeItCommits(unittest.TestCase):
+ def test_accumulate_report_does_not_commit_json_then_fail_to_render(self):
+ """accumulate_report hand-rolled the write instead of calling write_report, so a
+ render failure left the sidecar ahead of the markdown -- the exact ordering
+ write_report's docstring forbids."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_JSON).write_text(json.dumps(
+ {'rows': [{'board': 'boardA', 'cells': {'t': 'pass'}, 'duration': 119.0}],
+ 'banner': '', 'caveat': '', 'scope': ''}))
+ hil_report.accumulate_report([('boardB', 0, 0, [('boardB', {'t': 'OK'}, '1s')], 0)],
+ rd, False, '', '')
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertEqual((rd / hil_report.REPORT_MD).read_text(),
+ hil_report.render_report(doc) + '\n')
+
+
+class MissingSidecarDoesNotDestroyTheMarkdown(unittest.TestCase):
+ def test_an_absent_sidecar_keeps_the_prior_table(self):
+ """`recovered` was only cleared when the sidecar was TORN, not when it was absent
+ -- reachable from hil_ci.sh's asymmetric copy-back and build.yml's skip marker."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / hil_report.REPORT_MD).write_text('| Board | t |\n| a | OK |\n| b | OK |\n')
+ hil_report.write_timeout_report(rd, [{'name': 'stuck'}], 3600)
+ self.assertIn('| a | OK |', (rd / hil_report.REPORT_MD).read_text())
+
+
+class LoadIsTheOnlyTrustBoundary(unittest.TestCase):
+ """hil_ci.sh uploads a sidecar as the merge base, so these shapes arrive from OUTSIDE
+ the harness. Every one of these raised past a handler before."""
+
+ def _seed(self, rd, raw):
+ (rd / hil_report.REPORT_JSON).write_text(raw if isinstance(raw, str)
+ else json.dumps(raw))
+
+ def test_a_non_list_rows_does_not_raise(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ self._seed(rd, {'rows': 1, 'banner': '', 'caveat': '', 'scope': ''})
+ hil_report.accumulate_report([('a', 0, 0, [('a', {'t': 'OK'}, '1s')], 0)],
+ rd, False, '', '')
+ self.assertTrue((rd / hil_report.REPORT_MD).is_file())
+
+ def test_an_unhashable_cell_value_does_not_raise(self):
+ """render_matrix does REPORT_CELL.get(v, v); an unhashable value raised TypeError
+ on the NORMAL accumulate path."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ self._seed(rd, {'rows': [{'board': 'a', 'cells': {'t': ['x'], 'u': 'pass'},
+ 'duration': '1s'}],
+ 'banner': '', 'caveat': '', 'scope': ''})
+ hil_report.accumulate_report([('b', 0, 0, [('b', {'t': 'OK'}, '1s')], 0)],
+ rd, False, '', '')
+ cells = {r['board']: r['cells']
+ for r in json.loads((rd / hil_report.REPORT_JSON).read_text())['rows']}
+ self.assertNotIn('t', cells['a'], 'a corrupt cell must drop, not become a pass')
+ self.assertIn('u', cells['a'])
+
+ def test_summarize_survives_a_malformed_sidecar_via_load(self):
+ """The CLI is the one reader an agent's verdict depends on."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ (rd / 'cfg.json').write_text(json.dumps({'boards': [{'name': 'a'}]}))
+ self._seed(rd, {'rows': [{'board': 1, 'cells': 'notadict'},
+ {'board': 'a', 'cells': {'t': 'pass'}}],
+ 'banner': '', 'caveat': '', 'scope': ''})
+ r = subprocess.run(
+ [sys.executable, str(Path(HIL_DIR) / 'helper' / 'hil_report.py'),
+ str(rd / 'cfg.json'), '-b', 'a', '--report-dir', str(rd)],
+ capture_output=True, text=True, timeout=60)
+ self.assertEqual(r.returncode, 0, r.stderr)
+ self.assertTrue(json.loads(r.stdout)['results'][0]['pass'])
+
+
+class NoBoardsGuardOnlyAppliesWhenAccumulating(unittest.TestCase):
+ def test_a_fresh_run_carries_nothing_from_the_prior_sidecar(self):
+ """rows were reset on fresh but banner and scope were not, so a leftover or
+ uploaded sidecar republished a stale rig-health note and a stale scope line under
+ this run's notice."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report([('old', 0, 0, [('old', {'t': 'OK'}, '1s')], 0)],
+ rd, True, '3 board(s) — a, b, c',
+ '> **Rig note.** stale D-state holder\n')
+ hil_report.mark_report_no_boards(rd, 'filters emptied', fresh=True)
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertEqual(doc['rows'], [])
+ self.assertEqual(doc['banner'], '', 'a stale rig-health banner was republished')
+ self.assertEqual(doc['scope'], '', 'a stale scope note was republished')
+ self.assertNotIn('Rig note', (rd / hil_report.REPORT_MD).read_text())
+
+ def test_an_accumulate_run_keeps_banner_and_scope(self):
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report([('old', 0, 0, [('old', {'t': 'OK'}, '1s')], 0)],
+ rd, True, '3 board(s) — a, b, c',
+ '> **Rig note.** real\n')
+ hil_report.mark_report_no_boards(rd, 'filters emptied', fresh=False)
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertIn('Rig note', doc['banner'])
+ self.assertEqual([r['board'] for r in doc['rows']], ['old'])
+
+ def test_a_fresh_run_is_not_blocked_by_a_prior_abandon(self):
+ """The guard runs BEFORE the fresh wipe, so guarding a fresh run left the previous
+ attempt's rows AND its abandon notice published as this run's."""
+ td = TemporaryDirectory()
+ self.addCleanup(td.cleanup)
+ rd = Path(td.name)
+ hil_report.accumulate_report([('old', 0, 0, [('old', {'t': 'OK'}, '1s')], 0)],
+ rd, True, '', '')
+ hil_report.mark_report_abandoned(rd, 'the worker pool would not shut down.')
+ hil_report.mark_report_no_boards(rd, 'filters emptied', fresh=True)
+ doc = json.loads((rd / hil_report.REPORT_JSON).read_text())
+ self.assertEqual(doc['rows'], [])
+ self.assertIn('selected no boards', doc['caveat'])
+
+
+if __name__ == '__main__':
+ unittest.main()
diff --git a/test/hil/test/test_hil_util.py b/test/hil/test/test_hil_util.py
index c95e20b6d..e06d2ba8b 100644
--- a/test/hil/test/test_hil_util.py
+++ b/test/hil/test/test_hil_util.py
@@ -19,7 +19,6 @@ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from helper import hil_util
[email protected](os.name == 'nt', 'POSIX shell commands')
class RunCmdModes(unittest.TestCase):
def test_default_mode_unchanged(self):
r = hil_util.run_cmd('printf out; printf err >&2')
@@ -227,5 +226,58 @@ class RunAlongsideKeepsStderrOffThePayload(unittest.TestCase):
'child stderr leaked into the payload stream')
+class RunCmdCleanupShape(unittest.TestCase):
+ """run_cmd's two cleanup paths, asserted structurally.
+
+ Both must kill the process GROUP: start_new_session puts the child in its own group, so
+ a flasher run through a shell keeps children a p.kill() cannot reach, and on the
+ BaseException path the child never receives the terminal's SIGINT either.
+
+ Structural rather than behavioural on purpose. Driving a real SIGINT into a blocked
+ communicate() from a unit test is timing-dependent, and a flaky guard on this block is
+ worse than none -- while what actually breaks it is an edit that rebinds a branch. Both
+ times this block has been mis-edited, an `else:` ended up attached to the `try` instead
+ of the `if` it belonged to, so `p.kill()` ran when killpg had SUCCEEDED and its
+ ProcessLookupError masked the caller's exception. That is a shape, and shapes are
+ exactly what an AST can pin.
+ """
+
+ def _run_cmd_ast(self):
+ import ast
+ src = Path(hil_util.__file__).read_text()
+ return next(n for n in ast.walk(ast.parse(src))
+ if isinstance(n, ast.FunctionDef) and n.name == 'run_cmd')
+
+ def test_no_cleanup_try_has_an_else(self):
+ import ast
+ for n in ast.walk(self._run_cmd_ast()):
+ if isinstance(n, ast.Try) and n.orelse:
+ self.fail(f'try/else at line {n.lineno}: an else here runs when the kill '
+ f'SUCCEEDED, and its ProcessLookupError masks the caller\'s '
+ f'exception -- this block has been mis-edited that way twice')
+
+ def test_both_cleanup_paths_kill_the_group(self):
+ import ast
+ fn = self._run_cmd_ast()
+ killers = [getattr(c.func, 'attr', '') for c in ast.walk(fn)
+ if isinstance(c, ast.Call) and getattr(c.func, 'attr', '') in
+ ('killpg', 'kill')]
+ self.assertEqual(killers.count('killpg'), 2,
+ 'both the timeout and the BaseException path must killpg')
+ self.assertEqual(killers.count('kill'), 0,
+ 'p.kill() reaches only the direct child; a flasher run through a '
+ 'shell keeps grandchildren it cannot touch')
+
+ def test_the_interrupt_path_reraises(self):
+ import ast
+ fn = self._run_cmd_ast()
+ base = [h for n in ast.walk(fn) if isinstance(n, ast.Try) for h in n.handlers
+ if isinstance(h.type, ast.Name) and h.type.id == 'BaseException']
+ self.assertTrue(base, 'the BaseException cleanup path is gone')
+ for h in base:
+ self.assertTrue(any(isinstance(x, ast.Raise) for x in ast.walk(h)),
+ 'the interrupt path must re-raise, or Ctrl-C is swallowed')
+
+
if __name__ == '__main__':
unittest.main()
diff --git a/tools/build.py b/tools/build.py
index eeefca22d..0bb366e3d 100755
--- a/tools/build.py
+++ b/tools/build.py
@@ -356,11 +356,11 @@ def get_family_boards(family, one_random, one_first, examples=None, build_system
# the WHOLE preferred list, in order - stopping at entry one would abandon a
# curated list for the raw alphabetical order the moment its first board cannot
# build the filter, which also moves the board the metrics baseline is keyed on
+ # the whole preferred list, in order. Unreachable-when-unfiltered: with
+ # examples is None, buildable() is True and the loop returns on entry one.
for b in preferred_list:
if buildable(b):
return [b]
- if preferred_list and examples is None:
- return [preferred_list[0]]
candidates = [b for b in all_boards if buildable(b)] or all_boards
if one_first:
return [candidates[0]]
diff --git a/tools/build_utils.py b/tools/build_utils.py
index 1eeef0269..1b81335e0 100755
--- a/tools/build_utils.py
+++ b/tools/build_utils.py
@@ -1,5 +1,6 @@
#!/usr/bin/env python3
import functools
+import os
import subprocess
import pathlib
import re
@@ -24,7 +25,30 @@ _CMAKE_VAR_RE = re.compile(r'\$\{([A-Za-z_]\w*)\}')
_CMAKE_CASE_RE = re.compile(r'string\s*\(\s*(TOUPPER|TOLOWER)\s+(\S+)\s+([A-Za-z_]\w*)\s*\)')
[email protected]_cache(maxsize=None)
+
+def _cwd_cache(fn):
+ """lru_cache, keyed on the working directory as well as the arguments.
+
+ Every cached helper below takes repo-RELATIVE paths ('hw/bsp/<fam>',
+ 'examples/<ex>/skip.txt', or the literal 'hw/bsp' glob), while ci_select._in_repo()
+ chdirs around each call so one process can classify more than one tree - the
+ code-size skill's base-vs-branch worktrees, /pre-pr, a test pointing at a fixture.
+ Without the cwd in the key the second tree silently gets the first tree's
+ skip.txt/only.txt and FAMILY_MCUS answers. Master had no caching here, so this
+ hazard arrived with it."""
+ cache = {}
+
+ @functools.wraps(fn)
+ def wrapper(*args):
+ key = (os.getcwd(), args)
+ if key not in cache:
+ cache[key] = fn(*args)
+ return cache[key]
+
+ wrapper.cache_clear = cache.clear
+ return wrapper
+
+@_cwd_cache
def _cmake_sets(path):
"""One cmake file's variable assignments as NAME -> first definition seen, as
either a literal value or an ('TOUPPER'|'TOLOWER', source) pair. Only used to
@@ -87,7 +111,7 @@ def _cmake_expand(value, files, depth=0):
return None if '${' in out else out
[email protected]_cache(maxsize=None)
+@_cwd_cache
def _board_dirs(board):
"""(board_dir, family_dir) for a board name, or (None, None). Cached: skip_example
is asked (board x example) times - 566k lstat calls per selector run without this,
@@ -98,7 +122,7 @@ def _board_dirs(board):
return hits[0], hits[0].parent.parent
[email protected]_cache(maxsize=None)
+@_cwd_cache
def _family_mcus(family_dir, board_dir):
"""The MCU names CMake's family_filter iterates. family_support.cmake:176/190
loop `foreach(MCU IN LISTS FAMILY_MCUS)`, so a family-wide list (broadcom_64bit
@@ -175,7 +199,7 @@ def _family_mcus(family_dir, board_dir):
return frozenset(out)
[email protected]_cache(maxsize=None)
+@_cwd_cache
def _scrape_mcu(family_dir, board_dir, family):
"""(CFG_TUSB_MCU token of this board, the text it was read from), master's
algorithm verbatim: family.mk (family.cmake when there is none) first, falling
@@ -215,7 +239,7 @@ def _scrape_mcu(family_dir, board_dir, family):
return mcu, mk_contents
[email protected]_cache(maxsize=None)
+@_cwd_cache
def _board_mcu(board_dir, family_dir, family):
"""(CFG_TUSB_MCU of this board, MAX3421_HOST enabled by its cmake BSP).
@@ -254,7 +278,7 @@ def _board_mcu(board_dir, family_dir, family):
return mcu, max3421_enabled
[email protected]_cache(maxsize=None)
+@_cwd_cache
def _filter_tokens(path):
"""skip.txt / only.txt as a token set, or None when the file does not exist."""
f = pathlib.Path(path)
@@ -285,7 +309,7 @@ def skip_example(example, board, extra_defines=(), build_system='cmake'):
return _skip_example(example, board, tuple(extra_defines), build_system)
[email protected]_cache(maxsize=None)
+@_cwd_cache
def _skip_example_make(example, board):
"""master's skip_example, verbatim (tools/build_utils.py @ 9c202e8c6): the
make build's own answer, derived from family.mk/board.mk with the single
@@ -333,7 +357,7 @@ def _skip_example_make(example, board):
return False
[email protected]_cache(maxsize=None)
+@_cwd_cache
def _skip_example(example, board, extra_defines, build_system):
if build_system == 'make':
return _skip_example_make(example, board)
diff --git a/tools/ci_select.py b/tools/ci_select.py
index ced3bbbc0..ca9d54c27 100755
--- a/tools/ci_select.py
+++ b/tools/ci_select.py
@@ -13,6 +13,38 @@ JSON: full, boards (name -> 'all' | [tests]), families (bsp families the diff
touches, including ones with no rig board - build-only consumers such as /pre-pr
sample from these), args (hil_test.py args per config) and args_flasher (the same
args split by each board's flasher, for CI legs that split one rig by flasher).
+
+THE RULE TABLE. First match wins; answers union per family (build) and per board
+(HIL). A CARBON COPY of the table in the design spec above - edit both, or
+TestRuleTableIsCarbonOfTheSpec fails. `FAM` = the families whose family.cmake
+references the changed path (CMake only; make follows it). `DEV`/`HOST`/`DUAL`/
+`TYPEC`/`ALL` are the example role sets. The Build families column is PRE-PRUNE:
+_prune_buildable then intersects each family with what it can actually build.
+
+| # | Changed path | Build families | Build examples | HIL boards → tests |
+| 1 | `docs/`, `.claude/`, `*.md`, `*.rst`, `LICENSE` | — | — | — |
+| 1b | `.gitignore`, `.clang-format`, `.idea/**`, `test/{fuzz,unit-test}/**`, `test/hil/test/**`, non-build `.github/**`, packaging manifests | — | — | — |
+| 2 | `test/hil/**` (not `test/hil/test/**`) | — | — | all boards → all tests |
+| 2b | `tools/metrics.py`, `.github/scripts/metrics_*.py` | `ALL` (unchanged — `tinyusb_metrics` runs `metrics.py` as a build target) | `ALL` | — (nothing on the rig runs it) |
+| 3 | `src/portable/<port>/dcd_*`, `*_device.[ch]` | `FAM` | `DEV`+`DUAL` | `FAM`'s device-role boards → device+dual tests |
+| 4 | `src/portable/<port>/hcd_*`, `*_host.[ch]` | `FAM` | `HOST`+`DUAL` | `FAM`'s host-role boards → host+dual tests |
+| 5 | `src/portable/<port>/**` (anything else) | `FAM` | `ALL` | `FAM`'s boards → all their tests |
+| 5b | `src/portable/<port>/**` where `FAM` is empty | — | — | — (empty resolves to nothing on BOTH axes) |
+| 6 | `hw/bsp/<family>/**` | that family | `ALL` | that family's boards → all tests (a `boards/<board>/` path narrows to that board) |
+| 7 | `hw/mcu/<vendor>/**` | `FAM` — empty resolves to nothing (maintainer ruling) | `ALL` | `FAM`'s boards → all tests; empty resolves to nothing (maintainer ruling) ⚠ *see below* |
+| 8 | `src/class/<cls>/*_device.[ch]` | `ALL` | examples enabling `CFG_TUD_<CLS>` | device-role boards → HIL tests enabling `CFG_TUD_<CLS>` |
+| 9 | `src/class/<cls>/*_host.[ch]` | `ALL` | examples enabling `CFG_TUH_<CLS>` | host-role boards → HIL tests enabling `CFG_TUH_<CLS>` |
+| 10 | `src/class/<cls>/**` (shared header) | `ALL` | either, **plus include-edge classes** | both roles → same, plus include-edge classes |
+| 11 | `src/device/**` | `ALL` | `DEV`+`DUAL` | device-role boards → device+dual tests |
+| 12 | `src/host/**` | `ALL` | `HOST`+`DUAL` | host-role boards → host+dual tests |
+| 12b | `src/typec/**` | `ALL` | examples enabling `CFG_TUC_ENABLED` | — (no rig board runs a typec test) |
+| 13 | `examples/<role>/<name>/**` | `ALL` | just `<name>` | if `<name>` is a HIL test: all boards → that test; else nothing |
+| 14 | `examples/device/board_test/**` | `ALL` | just `board_test` | all boards → all tests (HIL parking firmware) |
+| 15 | `examples/build_system/**`, `examples/CMakeLists.txt`, `examples/<role>/CMakeLists.txt` | `ALL` | `ALL` | all boards → all tests |
+| 16 | `src/common/`, `src/osal/`, `src/tusb.[ch]`, `src/tusb_option.h`, `tools/{build,build_utils,ci_select}.py`, `tools/cmake/**`, `src/CMakeLists.txt`, `src/tinyusb.mk`, `hw/bsp/{family_support.{cmake,mk},family_rules.mk,zephyr_board_aliases.cmake,board.c,board_api.h,ansi_escape.h}`, `.github/**`, `.circleci/**` | `ALL` | `ALL` | all boards → all tests |
+| 16a | `lib/<name>/**` | `ALL` | examples whose own `CMakeLists.txt`/`Makefile` names `lib/<name>` | those examples that are HIL tests, on all boards; empty resolves to nothing |
+| 16b | `tools/get_deps.py` | families whose `deps_mandatory`/`deps_optional` entries changed | `ALL` | those families' boards → all tests; a logic change, an `'all'` entry, no base content or a changed token naming no family → full |
+| 17 | anything unclassified (no tracked file reaches this — TestNoTrackedFileIsUnclassified) | `ALL` | `ALL` | all boards → all tests (fail-open) |
"""
import argparse
import ast
@@ -53,7 +85,42 @@ def _read(path: str) -> str:
_NONCODE_RE = re.compile(
- r'^(docs/|\.claude/|.*\.(md|rst)$|LICENSE)')
+ # LICENSE is anchored and LICENSES/ named separately: a bare `LICENSE` alternative
+ # also swallowed anything merely STARTING with it (a future LICENSE_extra.c),
+ # which is the silent-under-selection direction
+ r'^(docs/|\.claude/|.*\.(md|rst)$|LICENSE$|LICENSES/)')
+# Repo metadata and tooling that no CI build reads. Enumerated rather than left to
+# rule 17, which widens BOTH axes: a PR touching only .gitignore and a README was
+# creating 74 cmake legs (each a runner doing checkout + toolchain + get_deps before
+# skipping the build) and booking the whole 30-board rig.
+#
+# Deliberately NOT here, and still full: .circleci/**, .github/workflows/build*.yml,
+# .github/actions/**, .github/scripts/** - those decide what gets built. The line is
+# "does any Build step read this file", not "is it source".
+#
+# test/{fuzz,unit-test} have their own jobs (cifuzz.yml, the unit-test pre-commit hook
+# and workflow); the Build matrix never compiles them, and test/hil is rule 2.
+_META_RE = re.compile(
+ r'^('
+ r'\.(gitignore|gitattributes|clang-format|codespellrc|readthedocs\.yaml)$|'
+ r'\.pre-commit-config\.yaml$|\.PVS-Studio/|\.idea/|\.vscode/|'
+ r'sonar-project\.properties$|library\.json$|pkg\.yml$|repository\.yml$|'
+ r'version\.yml$|SConscript$|'
+ r'.*CMakePresets\.json$|hw/bsp/BoardPresets\.json$|examples/west\.yml$|'
+ r'.*/[0-9]+-tinyusb[^/]*\.rules$|tools/usb_drivers/|tools/codespell/|'
+ # test/hil/test/ holds the harness's own unit tests, not the harness: nothing on
+ # the rig runs them (pre-commit does, and build.yml runs test_ci_select.py as the
+ # gate before trusting a selection), so they cannot change what the rig does.
+ # The harness itself stays under _FULL_RE's test/hil/ prefix.
+ r'test/(fuzz|unit-test)/|test/hil/test/|'
+ # .github, minus the build machinery named in _FULL_RE
+ r'\.github/(FUNDING\.yml$|labeler\.yml$|membrowse_pr_message\.j2$|ISSUE_TEMPLATE/|'
+ r'workflows/(cifuzz|claude|claude-code-review|labeler|membrowse-comment|'
+ r'membrowse-onboard|pr_comment|pre-commit|static_analysis|trigger)\.yml$)|'
+ # tools/ scripts no build invokes (tools/build*.py and metrics are handled above)
+ r'tools/(build_doc|check_example_pids|file2carray|gen_doc|gen_presets|iar_gen|'
+ r'make_release|mksunxi|pcapng_to_corpus)\.py$|tools/iar_template\.ipcf$'
+ r')')
# Build-size metrics tooling. HIL axis ONLY: nothing on the rig runs any of it, and
# without this rule these paths are unclassified, so a metrics-only PR booked an
# exclusive full 30-board sweep to validate a script no board executes.
@@ -66,9 +133,20 @@ _METRICS_RE = re.compile(
_FULL_RE = re.compile(
r'^(src/common/|src/osal/|src/tusb\.c$|src/tusb\.h$|src/tusb_option\.h$|'
r'test/hil/|\.github/workflows/build.*\.yml$|\.github/actions/|\.github/scripts/|'
- r'tools/build\.py$|tools/cmake/|'
- r'hw/bsp/(family_support\.cmake|board_api\.h|board\.c|ansi_escape\.h)$|'
+ # generates the whole CircleCI matrix, same authority as .github/**
+ r'\.circleci/|'
+ # rule 16 says `tools/build*.py`; name the two siblings the glob implies. Both
+ # decide what gets built, so neither can be trusted to narrow its own change.
+ r'tools/(build|build_utils|ci_select)\.py$|tools/cmake/|'
+ # the make twins of family_support.cmake are the same authority for the make legs
+ r'hw/bsp/(family_support\.(cmake|mk)|family_rules\.mk|zephyr_board_aliases\.cmake|'
+ r'board_api\.h|board\.c|ansi_escape\.h)$|'
+ # rule 15 lists examples/<role>/CMakeLists.txt - it registers every target in that
+ # role, so it was only ever reaching `full` through rule 17's fall-through
r'examples/build_system/|examples/CMakeLists\.txt$|'
+ r'examples/[^/]+/CMakeLists\.txt$|'
+ # every firmware compiles these unconditionally (src/CMakeLists.txt, src/tinyusb.mk)
+ r'src/CMakeLists\.txt$|src/tinyusb\.mk$|'
# board_test is HIL infrastructure, not a test: hil_test.py flashes it to park
# every board (variant boundary + end-of-board teardown), so every board depends on it
r'examples/device/board_test/)')
@@ -113,10 +191,20 @@ def board_tests(board: dict) -> list:
return [x for x in run if x not in t.get('skip', [])]
+
+def _rg(repo_root: str, *parts: str) -> str:
+ """A glob pattern rooted at repo_root, with the ROOT escaped and the parts left as
+ patterns. The root is a filesystem path, not a pattern: a checkout at
+ /w/pr[1]/tinyusb (a worktree named after a PR, a CI workspace with brackets) makes
+ an unescaped '[1]' a character class that matches nothing, and every lookup below
+ then resolves to zero - families=0 instead of 30, i.e. the selector fails CLOSED
+ and the whole matrix compiles nothing while reporting green."""
+ return os.path.join(glob.escape(repo_root), *parts)
+
# cached: called per changed file x roster board, and the tree doesn't change mid-run
@functools.lru_cache(maxsize=None)
def board_family(board_name: str, repo_root: str):
- hits = glob.glob(os.path.join(repo_root, 'hw/bsp/*/boards', board_name))
+ hits = glob.glob(_rg(repo_root, 'hw/bsp/*/boards', board_name))
return os.path.basename(os.path.dirname(os.path.dirname(hits[0]))) if hits else None
@@ -224,10 +312,10 @@ def _family_file_texts(repo_root: str) -> tuple:
CMakeLists.txt, read once. path_families is called per distinct directory in the
diff and its own cache only helps repeats: a 6,000-file hw/mcu dep bump re-read
these 84 files 99,892 times (2.2 s) before this."""
- bsp_root = os.path.join(repo_root, 'hw/bsp')
+ bsp_root = os.path.join(repo_root, 'hw/bsp') # escaped by _rg below
out = []
- for f in sorted(glob.glob(os.path.join(bsp_root, '*/family.cmake')) +
- glob.glob(os.path.join(bsp_root, '*/components/*/CMakeLists.txt'))):
+ for f in sorted(glob.glob(_rg(bsp_root, '*/family.cmake')) +
+ glob.glob(_rg(bsp_root, '*/components/*/CMakeLists.txt'))):
try:
out.append((os.path.relpath(f, bsp_root).split(os.sep, 1)[0], _read(f)))
except OSError:
@@ -347,7 +435,7 @@ def class_include_edges(repo_root: str) -> dict:
Derived from the actual #include lines rather than a hand-written table so it
cannot rot when a class picks up or drops a cross-class include."""
edges = {}
- for f in sorted(glob.glob(os.path.join(repo_root, 'src/class/*/*.[ch]'))):
+ for f in sorted(glob.glob(_rg(repo_root, 'src/class/*/*.[ch]'))):
cls = os.path.basename(os.path.dirname(f))
try:
text = _read(f)
@@ -431,11 +519,23 @@ def _class_roles(base: str) -> set:
return {'device', 'host'}
-def _config_enables(cfg_path: str, macros) -> bool:
[email protected]_cache(maxsize=None)
+def _config_text(cfg_path: str) -> str:
+ """An example's tusb_config.h, read once. Every class path re-asks the same 46
+ configs on both axes, so the reads go up with the diff: 4,240 of the same 46 files
+ for a diff touching all of src/class (0.48s -> 0.13s), and they cannot change
+ mid-run. Cached here rather than on _config_enables so the macros argument stays an
+ ordinary list at every call site."""
try:
with open(cfg_path, encoding='utf-8', errors='replace') as f:
- text = f.read()
+ return f.read()
except OSError:
+ return ''
+
+
+def _config_enables(cfg_path: str, macros) -> bool:
+ text = _config_text(cfg_path)
+ if not text:
return False
for m in macros:
for value in re.findall(_DEF_VALUE.format(m), text, re.M):
@@ -472,10 +572,13 @@ def lib_examples(lib_name: str, repo_root: str) -> set:
pat = re.compile(re.escape('lib/' + lib_name) + r'(?=[/\s"\')}]|$)', re.M)
out = set()
for ex in all_examples(repo_root):
- for f in sorted(glob.glob(os.path.join(repo_root, 'examples', ex, '**', '*'),
+ # the two filenames directly: '**/*' enumerated 489 entries per lib against a
+ # clean tree to use 107, and grows without bound once `make BOARD=... all` has
+ # written examples/<role>/<name>/_build/ - which is where /pre-pr runs
+ for f in sorted(glob.glob(_rg(repo_root, 'examples', ex, '**', 'CMakeLists.txt'),
+ recursive=True) +
+ glob.glob(_rg(repo_root, 'examples', ex, '**', 'Makefile'),
recursive=True)):
- if os.path.basename(f) not in ('CMakeLists.txt', 'Makefile'):
- continue
try:
text = _read(f)
except OSError:
@@ -538,10 +641,10 @@ class _Sel:
def _classify_one(path, repo_root, roster_boards, extras: set, s: _Sel,
get_deps_families=None):
base = os.path.basename(path)
- if _NONCODE_RE.match(path):
+ if _NONCODE_RE.match(path) or _META_RE.match(path):
s.reasons.append(f'{path}: non-code, no contribution')
return
- if _METRICS_RE.match(path):
+ if _METRICS_RE.match(path): # rule 2b
s.reasons.append(f'{path}: build-size metrics tooling, no HIL contribution')
return
if _FULL_RE.match(path):
@@ -673,6 +776,11 @@ def _classify_one(path, repo_root, roster_boards, extras: set, s: _Sel,
s.add(boards, sorted(tests), f'{path}: lib {lib} -> {sorted(tests)} on all boards')
return
+ if re.match(r'src/typec/', path):
+ # only examples/typec enables CFG_TUC_ENABLED, and no rig board runs a typec
+ # test (see _HIL_EX_ROLES) - so the build axis covers it and the rig cannot
+ s.reasons.append(f'{path}: typec, no HIL contribution')
+ return
m = _BUILD_EX_RE.match(path)
if m:
if m.group(1) not in _HIL_EX_ROLES:
@@ -859,7 +967,13 @@ def main():
print(f'ci_select[build]: {r}', file=sys.stderr)
for r in s['reasons']:
print(f'ci_select: {r}', file=sys.stderr)
- print(json.dumps(s))
+ # reasons go to stderr ONLY - they are a human diagnostic and no consumer reads them
+ # back. They are also ~97% of the payload (a whole-tree diff: 453 KB -> 12 KB), which
+ # build.yml re-parses with ci_set_matrix, hil_ci_set_matrix, an inline python and
+ # three jq calls. The in-process dicts still carry them, for the log and the tests.
+ out = {k: v for k, v in s.items() if k != 'reasons'}
+ out['build'] = {k: v for k, v in s['build'].items() if k != 'reasons'}
+ print(json.dumps(out))
# -------------------------------------------------------------
@@ -881,7 +995,7 @@ def all_examples(repo_root: str) -> tuple:
"""Every examples/<role>/<name> with a CMakeLists.txt, as 'role/name'."""
out = []
for role in _EX_ROLES:
- for d in sorted(glob.glob(os.path.join(repo_root, 'examples', role, '*/'))):
+ for d in sorted(glob.glob(_rg(repo_root, 'examples', role, '*/'))):
if os.path.isfile(os.path.join(d, 'CMakeLists.txt')):
out.append(f'{role}/{os.path.basename(d.rstrip(os.sep))}')
return tuple(out)
@@ -935,12 +1049,13 @@ class _BSel:
def _classify_build_one(path, repo_root, s: _BSel, get_deps_families=None):
base = os.path.basename(path)
- if _NONCODE_RE.match(path): # rule 1
+ if _NONCODE_RE.match(path) or _META_RE.match(path): # rules 1, 1b
+ s.reasons.append(f'{path}: non-code, no build contribution')
return
if re.match(r'test/hil/', path): # rule 2
s.reasons.append(f'{path}: HIL harness, no build contribution')
return
- if path == GET_DEPS_PATH: # get_deps rule
+ if path == GET_DEPS_PATH: # rule 16b
if get_deps_families is None:
s.force_full(f'{path}: dep changes not resolvable -> full build matrix')
return
@@ -957,6 +1072,7 @@ def _classify_build_one(path, repo_root, s: _BSel, get_deps_families=None):
roles = _port_roles(base)
exs = 'all' if roles == {'device', 'host'} else \
role_examples(repo_root, tuple(roles) + ('dual',))
+ # rule 5b: fams empty -> s.add iterates nothing -> no contribution
s.add(fams, exs, f'{path}: port {port} -> families {sorted(fams)}')
return
if re.match(r'hw/bsp/[^/]+/', path): # rule 6
@@ -1006,8 +1122,20 @@ def _classify_build_one(path, repo_root, s: _BSel, get_deps_families=None):
# CMakeLists (rule 15) is what forces the full matrix
s.reasons.append(f'{path}: not an example dir, no build contribution')
return
+ if re.match(r'src/typec/', path): # rule 12b
+ # listed unconditionally by src/CMakeLists.txt and src/tinyusb.mk, but the whole
+ # body is `#if CFG_TUC_ENABLED` - so it is PARSED by every build and COMPILED
+ # only for examples that enable it. Same shape as the class rule, same answer:
+ # the examples whose tusb_config.h turns it on, and empty means empty.
+ exs = examples_enabling(role_examples(repo_root, ('typec',)),
+ ('CFG_TUC_ENABLED',), repo_root)
+ if not exs:
+ s.reasons.append(f'{path}: typec enabled by no example config, no contribution')
+ return
+ s.add(all_bsp_families(repo_root), exs, f'{path}: typec -> {sorted(exs)}')
+ return
m = re.match(r'lib/([^/]+)/', path)
- if m: # lib rule
+ if m: # rule 16a
lib = m.group(1)
exs = lib_examples(lib, repo_root)
if not exs:
@@ -1018,7 +1146,19 @@ def _classify_build_one(path, repo_root, s: _BSel, get_deps_families=None):
return
s.add(all_bsp_families(repo_root), exs, f'{path}: lib {lib} -> {sorted(exs)}')
return
- s.force_full(f'{path}: unclassified -> full build matrix') # rules 15-17
+ if _METRICS_RE.match(path):
+ # HIL-suppressed above; on this axis they stay full - tools/metrics.py runs as
+ # the `tinyusb_metrics` build target, so a break in it fails the build
+ s.force_full(f'{path}: metrics tooling runs in the build -> full build matrix')
+ return
+ if _FULL_RE.match(path): # rules 15-16
+ # attribution, not behaviour: these already reached `full` through the
+ # fall-through below. Naming them means a future narrowing of rule 17 cannot
+ # silently change what they do. Deliberately last, so every earlier rule keeps
+ # priority - examples/device/board_test is rule 14 (just board_test), not ALL.
+ s.force_full(f'{path}: core/infra -> full build matrix')
+ return
+ s.force_full(f'{path}: unclassified -> full build matrix') # rule 17
@contextlib.contextmanager
@@ -1081,11 +1221,25 @@ def _prune_buildable(fams, fam_ex, repo_root):
# for anything else spins up CI's most expensive leg to skip every example
# it was given. Identical to the unfiltered list on all 81 other families.
pool = set(build_py.get_examples(fam))
+
+ # asked per example instead of materialising the family's whole buildable
+ # list: skip_example is by far the hottest call in the selector, and every
+ # question below short-circuits (one cdc_device.c diff: 6,883 calls -> 1,889)
+ def can_build(ex):
+ # EITHER build system: this one list gates CircleCI's make legs too, and
+ # the two answer differently (build_utils.skip_example)
+ return ex in pool and any(
+ not build_utils.skip_example(ex, b) or
+ not build_utils.skip_example(ex, b, (), 'make') for b in boards)
+
+ want = fam_ex.get(fam)
try:
- buildable = [e for e in allex if e in pool and
- any(not build_utils.skip_example(e, b) or
- not build_utils.skip_example(e, b, (), 'make')
- for b in boards)]
+ if want is None:
+ kept = None if any(can_build(e) for e in allex) else []
+ else:
+ kept = [e for e in want if can_build(e)]
+ if kept and not any(can_build(e) for e in allex if e not in want):
+ kept = None # already everything the family can build
except OSError as e:
# a family mid-bring-up (boards/ but no family.cmake/family.mk yet)
# reads as unbuildable to the scrape; keep it rather than tracebacking
@@ -1093,13 +1247,10 @@ def _prune_buildable(fams, fam_ex, repo_root):
reasons.append(f'{fam}: mcu scrape unreadable ({e}), kept unfiltered')
out_fams.append(fam)
continue
- want = fam_ex.get(fam)
- have = set(buildable)
- kept = buildable if want is None else [e for e in want if e in have]
- if not kept:
+ if kept == []:
continue # this diff builds nothing for this family
out_fams.append(fam)
- if set(kept) != set(buildable):
+ if kept is not None:
out_ex[fam] = kept
return out_fams, out_ex, reasons