From 1b34b2082e3374a1b39355dbdaace42451b93d54 Mon Sep 17 00:00:00 2001 From: kunchenguid Date: Thu, 17 Sep 2026 17:50:33 -0700 Subject: [PATCH 1/2] Improve CI reliability and rebalance full-coverage validation --- .github/workflows/ci.yml | 33 +++- CONTRIBUTING.md | 21 +- bin/fm-lint.sh | 104 +++++++--- bin/fm-mail-check.sh | 12 +- bin/fm-test-run.sh | 334 +++++++++++++++++--------------- docs/fm-test-portable-shards.md | 35 +++- tests/fm-ci-workflow.test.sh | 31 +++ tests/fm-lint.test.sh | 43 ++++ tests/fm-mail-check.test.sh | 51 +++++ tests/fm-test-run.test.sh | 8 +- tests/fm-watcher-lock.test.sh | 67 ++++++- tests/wake-helpers.sh | 16 +- 12 files changed, 542 insertions(+), 213 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index a68f49cdd02..8436d065119 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -10,7 +10,7 @@ permissions: contents: read # Per-PR supersession: a new push to the same PR replaces that PR's in-flight -# CI instead of letting superseded heads keep 13 jobs of hosted-runner work. +# CI instead of letting superseded heads keep the full hosted-runner fan-out. # The group uses the PR number for pull_request events, so every run of one PR # shares a group, and falls back to the unique run id for push events, so each # main push gets its own group and is never cancelled. Cancellation is likewise @@ -24,11 +24,14 @@ concurrency: jobs: lint: - name: Lint + name: Lint ${{ matrix.partition }} runs-on: ubuntu-latest - # Hang tripwire only: lint executions measured at 14-16 minutes in the - # September 12 starvation report, so this leaves deliberate margin. + # Keep the hang tripwire separate from the measured performance target. timeout-minutes: 25 + strategy: + fail-fast: false + matrix: + partition: [1, 2] steps: - uses: actions/checkout@v6 - name: Install pinned ShellCheck @@ -45,7 +48,19 @@ jobs: # and GitHub workflow lint). Do not re-spell the checks here; keep CI # and the pre-push gate on this script so a self-broken ci.yml still # fails locally before merge. - - run: bin/fm-lint.sh + - name: Lint canonical partition + run: | + set -eu + mkdir -p "$RUNNER_TEMP/fm-lint" + bin/fm-lint.sh --partition "${{ matrix.partition }}of${{ strategy.job-total }}" \ + --telemetry "$RUNNER_TEMP/fm-lint/partition-${{ matrix.partition }}.tsv" + - name: Upload lint telemetry + if: always() + uses: actions/upload-artifact@v4 + with: + name: fm-lint-telemetry-${{ matrix.partition }} + path: ${{ runner.temp }}/fm-lint/partition-${{ matrix.partition }}.tsv + if-no-files-found: warn # Deterministic proof that portable parallel shards + portable serial + Herdr # equal the complete tests/*.test.sh inventory with no missing or duplicates, @@ -159,15 +174,15 @@ jobs: tests-portable-serial: name: Behavior portable serial ${{ matrix.shard }} runs-on: ubuntu-latest - # Current runners can take ~20 min for a balanced shard. This 30-minute cap - # preserves the timeout as a hang tripwire while allowing runner-speed and - # job-setup margin; it is not the expected healthy end of the lane. + # Refreshed weights put the longest modeled shard near 12 minutes across + # nine runners. Preserve the existing hang tripwire until complete Linux + # measurements establish the new healthy envelope; a model is not a timer. timeout-minutes: 30 strategy: # Every shard reports so one failure never hides another shard's result. fail-fast: false matrix: - shard: [1, 2, 3, 4, 5] + shard: [1, 2, 3, 4, 5, 6, 7, 8, 9] steps: - uses: actions/checkout@v6 with: diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 5250d77e555..e2fd860cafd 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -32,6 +32,23 @@ GitHub Actions and Dependabot are exempt so their automation keeps working, but See the [no-mistakes quick start](https://kunchenguid.github.io/no-mistakes/start-here/quick-start/) for the full first-run walkthrough. +## Maintaining required checks + +GitHub required checks are configured in the repository's existing main ruleset, not activated by committing workflow YAML. +When applying this CI layout, preserve its existing pull-request, merge-method, linear-history, deletion, non-fast-forward, and administrator-bypass settings. +Add required status checks with `strict_required_status_checks_policy: false`; a main update alone must not force a branch update and retest. +Bind the checks to the GitHub Actions app already producing them, rather than accepting the same context from any integration. +No new app installation or manual runner setup is needed for that setting. + +Require the actual job contexts: `Lint 1`, `Lint 2`, `Test coverage guard`, `Repo invariants`, `Stock macOS Bash snapshot compatibility`, `Behavior portable parallel 1`, `Behavior portable parallel 2`, `Behavior portable serial 1` through `Behavior portable serial 9`, `Behavior tests (Herdr)`, `Behavior timing aggregate`, and `PR must be raised via no-mistakes`. +The last name is the compliance job context, not its workflow title; its existing automation exceptions remain unchanged. +The timing aggregate is not a substitute for individual jobs because it can succeed while collecting evidence from a failed run. + +Apply the approved rule change only after the corresponding workflow is green and landed, confirming exact names and the Actions integration id from real checks first. +Snapshot the current ruleset, amend that same rule with the authenticated GitHub API or settings UI, and read back both the ruleset and effective branch rules. +Verify missing or red checks prevent ordinary merging without creating a test merge; administrator override intentionally remains available. +Coordinate any workflow rollback with its required-check names so a retired check cannot leave ordinary merges waiting forever. + ## Repo conventions - This repo is a template for running a firstmate orchestrator agent. @@ -48,7 +65,9 @@ See the [no-mistakes quick start](https://kunchenguid.github.io/no-mistakes/star - Helper scripts in `bin/` are plain bash. Each starts with a usage header comment; keep it accurate when you change behavior. Test scripts and helpers in `tests/` are plain bash too. - `bin/fm-lint.sh` must pass: it is the single owner of the lint definition (the shellcheck file set, config, pinned shellcheck version, pinned actionlint workflow lint, and the backend-purity check rejecting direct Beads CLI calls in core `bin/` scripts), and both CI and the no-mistakes pre-push gate invoke it with no arguments. + `bin/fm-lint.sh` must pass: it is the single owner of the lint definition (the shellcheck file set, config, pinned shellcheck version, pinned actionlint workflow lint, and the backend-purity check rejecting direct Beads CLI calls in core `bin/` scripts). + CI uses its full canonical partitions; the no-mistakes pre-push gate uses its context-selected default. + `docs/fm-test-portable-shards.md` owns partition verification and performance evidence. Its header and `--help` output own the exact local lint modes, file-set selection, and analysis flags. A malformed `.github/workflows/*.yml`, including a self-broken `ci.yml`, fails that local lint path before merge because a broken workflow cannot report its own breakage. It pins one exact shellcheck version and one exact actionlint version and refuses to run under any other. diff --git a/bin/fm-lint.sh b/bin/fm-lint.sh index 9408508aff9..9886476177f 100755 --- a/bin/fm-lint.sh +++ b/bin/fm-lint.sh @@ -2,9 +2,9 @@ # fm-lint.sh - the single owner of firstmate's lint definition. # # Runs its file set with ShellCheck's default severity, extended analysis, -# ambient configuration disabled, and one exact ShellCheck version. CI and -# no-mistakes both invoke this script with no arguments, so this owner selects -# the context-appropriate rule set without duplicating lint configuration. +# ambient configuration disabled, and one exact ShellCheck version. CI selects +# canonical partitions; no-mistakes invokes the context-selected default, so +# both use this owner without duplicating lint configuration. # The explicit --fast mode is local-only and disables ShellCheck's extended # dataflow analysis while preserving ordinary shell lint checks and source # following. CI, main, and merge-base-less runs keep --norc --external-sources @@ -41,10 +41,15 @@ # invocations in the core bin/ and bin/backends/ scripts so every configured # backlog backend follows the same tasks-axi lifecycle path. # -# Canonical lint defaults to two bounded workers over two stable logical shards. -# Each shard writes separate diagnostics, and the parent replays those outputs in -# deterministic shard and root order after every worker finishes. FM_LINT_JOBS=1 -# runs the same shards serially with byte-identical diagnostics and exit selection. +# Lint defaults to two bounded workers over two stable logical shards. +# Diagnostics replay in stable shard/root order. FM_LINT_JOBS=1 changes +# concurrency, not diagnostics or exit selection. +# --partition 1of2/2of2 splits the entire canonical inventory across +# two CI runners, each with those same bounded workers. Partitions are complete, +# disjoint, and byte-weight balanced; --list-files exposes their actual roots. +# Partition mode is always full source-aware analysis, never changed-only or +# --fast, and does not accept explicit paths. Each partition also runs workflow +# lint and backend-purity checks, keeping either invocation independently useful. # # Optional quiet telemetry writes one bounded TSV snapshot of content and source # graph identity, wall/CPU/RSS, shard load, and competing ShellCheck processes. @@ -54,6 +59,7 @@ # fm-lint.sh --fast [path]... local lint with extended analysis disabled # fm-lint.sh ... lint explicit roots with the same config # fm-lint.sh --jobs <1|2> [path]... override bounded worker count +# fm-lint.sh --partition <1of2|2of2> lint one full-rigor canonical CI partition # fm-lint.sh --telemetry ... write a quiet metrics snapshot # fm-lint.sh --required-version print the ShellCheck pin # fm-lint.sh --list-files print the file set that would be linted @@ -396,6 +402,8 @@ JOBS=${FM_LINT_JOBS:-2} TELEMETRY=${FM_LINT_TELEMETRY:-} FAST=0 ANALYSIS_MODE=full +PARTITION= +PARTITION_REQUESTED=0 LIST_FILES=0 while [ "$#" -gt 0 ]; do case "$1" in @@ -417,6 +425,17 @@ while [ "$#" -gt 0 ]; do TELEMETRY=${1#*=} shift ;; + --partition) + [ "$#" -ge 2 ] || { printf 'fm-lint.sh: --partition requires 1of2 or 2of2.\n' >&2; exit 2; } + PARTITION=$2 + PARTITION_REQUESTED=1 + shift 2 + ;; + --partition=*) + PARTITION=${1#*=} + PARTITION_REQUESTED=1 + shift + ;; --fast) FAST=1 ANALYSIS_MODE=fast @@ -443,6 +462,22 @@ case "$JOBS" in *) printf 'fm-lint.sh: jobs must be 1 or 2, got %s.\n' "$JOBS" >&2; exit 2 ;; esac +case "$PARTITION" in + '') + if [ "$PARTITION_REQUESTED" -eq 1 ]; then + printf 'fm-lint.sh: --partition requires 1of2 or 2of2.\n' >&2 + exit 2 + fi + ;; + 1of2|2of2) + if [ "$FAST" -eq 1 ] || [ "$#" -gt 0 ]; then + printf 'fm-lint.sh: --partition requires full canonical lint; omit --fast and explicit paths.\n' >&2 + exit 2 + fi + ;; + *) printf 'fm-lint.sh: --partition must be 1of2 or 2of2, got %s.\n' "$PARTITION" >&2; exit 2 ;; +esac + if [ "$FAST" -eq 1 ] && { [ "${GITHUB_ACTIONS:-}" = true ] || [ "${CI:-}" = true ]; }; then printf 'fm-lint.sh: --fast is local-only; CI uses full ShellCheck analysis.\n' >&2 exit 2 @@ -492,7 +527,7 @@ if [ "$#" -gt 0 ]; then ROOTS=("$@") else full_lint=1 - if [ "${GITHUB_ACTIONS:-}" != true ] && [ "${CI:-}" != true ] \ + if [ -z "$PARTITION" ] && [ "${GITHUB_ACTIONS:-}" != true ] && [ "${CI:-}" != true ] \ && command -v git >/dev/null 2>&1 \ && git rev-parse --is-inside-work-tree >/dev/null 2>&1 \ && [ "$(git rev-parse --abbrev-ref HEAD 2>/dev/null)" != main ]; then @@ -519,6 +554,38 @@ if [ "$CHANGED_MODE" -eq 1 ] && [ "$FAST" -eq 0 ]; then EXCLUDE_CODES=$LOCAL_NOX_EXCLUDE ANALYSIS_MODE=local fi +# Stable largest-first packing is shared by cross-runner partition selection +# and the two local workers. Weights are a scheduling proxy, never a skip rule. +TAB=$(printf '\t') +fm_lint_root_weights() { + local index=1 path weight + for path in "${ROOTS[@]}"; do + case "$path" in + *"$TAB"*|*$'\n'*) + printf 'fm-lint.sh: paths containing tabs or newlines are not supported: %s\n' "$path" >&2 + return 2 + ;; + esac + weight=1 + if [ -f "$path" ]; then + weight=$(wc -c < "$path" 2>/dev/null | tr -d '[:space:]') + fi + case "$weight" in ''|*[!0-9]*) weight=1 ;; esac + printf '%s\t%s\t%s\n' "$weight" "$index" "$path" + index=$((index + 1)) + done +} + +if [ -n "$PARTITION" ]; then + PARTITION_ROOTS=() + partition_weights=$(fm_lint_root_weights) || exit $? + while IFS="$TAB" read -r index path; do + PARTITION_ROOTS+=("$path") + done < <(printf '%s\n' "$partition_weights" | LC_ALL=C sort -t "$TAB" -k1,1nr -k2,2n | awk -F '\t' -v want="${PARTITION%%of*}" ' + { shard=(load[2] < load[1]) ? 2 : 1; load[shard]+=$1; if (shard == want) print $2 "\t" $3 } + ' | LC_ALL=C sort -t "$TAB" -k1,1n) + ROOTS=("${PARTITION_ROOTS[@]}") +fi ROOT_COUNT=${#ROOTS[@]} if [ "$LIST_FILES" -eq 1 ]; then @@ -597,7 +664,6 @@ trap 'exit 129' HUP trap 'exit 130' INT trap 'exit 143' TERM -TAB=$(printf '\t') WEIGHTS="$TMP_ROOT/weights" OUTPUT_DIR="$TMP_ROOT/output" mkdir -p "$OUTPUT_DIR" @@ -608,24 +674,7 @@ while [ "$worker" -lt "$SHARD_COUNT" ]; do worker=$((worker + 1)) done -index=1 -: > "$WEIGHTS" -for path in "${ROOTS[@]}"; do - case "$path" in - *"$TAB"*|*$'\n'*) - printf 'fm-lint.sh: paths containing tabs or newlines are not supported: %s\n' "$path" >&2 - exit 2 - ;; - esac - if [ -f "$path" ]; then - weight=$(wc -c < "$path" 2>/dev/null | tr -d '[:space:]') - else - weight=1 - fi - case "$weight" in ''|*[!0-9]*) weight=1 ;; esac - printf '%s\t%s\t%s\n' "$weight" "$index" "$path" >> "$WEIGHTS" - index=$((index + 1)) -done +fm_lint_root_weights > "$WEIGHTS" || exit $? # Largest-first deterministic greedy assignment keeps the two bounded workers # balanced without affecting replay order. Direct bytes are a stable portable @@ -841,6 +890,7 @@ EOF printf 'content_cksum\t%s\n' "$content_cksum" printf 'shellcheck_version\t%s\n' "$resolved" printf 'analysis_mode\t%s\n' "$ANALYSIS_MODE" + printf 'partition\t%s\n' "${PARTITION:-all}" printf 'jobs\t%s\n' "$JOBS" printf 'root_count\t%s\n' "$ROOT_COUNT" printf 'direct_lines\t%s\n' "$direct_lines" diff --git a/bin/fm-mail-check.sh b/bin/fm-mail-check.sh index 6d73102594c..b30c642f3e3 100755 --- a/bin/fm-mail-check.sh +++ b/bin/fm-mail-check.sh @@ -126,9 +126,11 @@ fi # dropped, and an empty failure gets a truth-stating fallback. poll_summary() { local rc=$1 out=$2 line - line=$(printf '%s\n' "$out" | sed -n '/^fm-mail: woke for /d; s/^fm-mail: //p' | head -n 1) + # First-line selectors must still drain the stream: head/quiet grep can + # close a large poll's pipe early and add a Broken pipe diagnostic. + line=$(printf '%s\n' "$out" | sed -n '/^fm-mail: woke for /d; s/^fm-mail: //p' | sed -n '1p') if [ -z "$line" ]; then - line=$(printf '%s\n' "$out" | sed -n '/^fm-mail: woke for /d; /^$/d; p' | head -n 1) + line=$(printf '%s\n' "$out" | sed -n '/^fm-mail: woke for /d; /^$/d; p' | sed -n '1p') fi if [ -z "$line" ]; then line="poll failed (rc=$rc)" @@ -174,8 +176,8 @@ record_write() { poll_has_publication_evidence() { local rc=${1:-0} out=$2 woken_before=$3 [ "$rc" -eq 124 ] && return 0 - if [ -n "$out" ] && printf '%s\n' "$out" | grep -qE \ - '^fm-mail: woke for |the wake stays queued|could not clear retry for recovered' + if [ -n "$out" ] && printf '%s\n' "$out" | grep -E \ + '^fm-mail: woke for |the wake stays queued|could not clear retry for recovered' >/dev/null then return 0 fi @@ -210,7 +212,7 @@ action_check() { line="poll did not finish within the ${BUDGET_SECS}s budget" elif [ "${rc:-0}" -ne 0 ]; then line=$(poll_summary "$rc" "$out") - elif printf '%s\n' "$out" | grep -q '^fm-mail: woke for '; then + elif printf '%s\n' "$out" | grep '^fm-mail: woke for ' >/dev/null; then # A successful poll can still surface new mail: the poll itself already # appended the durable mail wake rows, but the watcher only calls wake() # when THIS check's output is non-empty. Emit one line naming a surfaced diff --git a/bin/fm-test-run.sh b/bin/fm-test-run.sh index 4819cc579b6..b939101c943 100755 --- a/bin/fm-test-run.sh +++ b/bin/fm-test-run.sh @@ -194,7 +194,7 @@ CHANGED_DEFAULT_TIMEOUT_SECS=900 # How many separate-runner shards the portable serial remainder splits into. # One owner: CI lane names carry this count and are refused when they disagree. -PORTABLE_SERIAL_SHARDS=5 +PORTABLE_SERIAL_SHARDS=9 # Balance hint for a portable-serial script with no measured duration, close to # the measured per-script mean so a newly added test neither starves nor @@ -657,163 +657,188 @@ list_portable_serial() { # Measured portable-serial script durations in milliseconds, from the CI timing # artifacts recorded in docs/fm-test-portable-shards.md. Each value is the -# slowest of several green runs, so the balance holds on a slow runner rather +# slowest successful sample in the referenced complete/partial CI runs, rather # than only on the fastest one measured. These are balance hints only: the shard # partition stays complete and disjoint whatever they say, so a stale hint costs # balance rather than coverage. That doc owns the refresh procedure. portable_serial_weight_hints() { cat <<'EOF' -tests/fm-agy-harness.test.sh 11000 -tests/fm-agy-signals-live-e2e.test.sh 23 -tests/fm-afk-contract.test.sh 3000 -tests/fm-afk-inject-e2e.test.sh 35792 -tests/fm-afk-pi-herdr-return-e2e.test.sh 100 -tests/fm-afk-return.test.sh 1837 -tests/fm-ask-user-authority.test.sh 128 +tests/fm-afk-contract.test.sh 15645 +tests/fm-afk-inject-e2e.test.sh 35889 +tests/fm-afk-pi-herdr-return-e2e.test.sh 45 +tests/fm-afk-return.test.sh 20385 +tests/fm-agy-harness.test.sh 47933 +tests/fm-agy-signals-live-e2e.test.sh 49 +tests/fm-ask-user-authority.test.sh 131 tests/fm-backend-cmux-smoke.test.sh 33 -tests/fm-backend-cmux.test.sh 3657 -tests/fm-backend-orca.test.sh 19253 -tests/fm-backend-tmux-smoke.test.sh 393 -tests/fm-backend-zellij-smoke.test.sh 23 -tests/fm-backend-zellij.test.sh 9418 -tests/fm-backend.test.sh 20061 -tests/fm-backlog-atomicity.test.sh 161989 -tests/fm-backlog-handoff.test.sh 52291 -tests/fm-bearings-board-render.test.sh 1528 -tests/fm-bearings-board.test.sh 4195 -tests/fm-bearings-snapshot.test.sh 116374 -tests/fm-bootstrap-network-parallel.test.sh 8214 -tests/fm-bootstrap.test.sh 25208 -tests/fm-branch-supervision.test.sh 5729 -tests/fm-busy-adapter-wiring.test.sh 49731 -tests/fm-busy-state.test.sh 2926 -tests/fm-calm-pi-extension.test.sh 256 -tests/fm-check-unregister.test.sh 481 -tests/fm-classify-corr-token.test.sh 38742 -tests/fm-classify-decision-key.test.sh 1167 -tests/fm-claude-stop-autoarm-live-e2e.test.sh 21 -tests/fm-claude-stop-autoarm.test.sh 60709 -tests/fm-cmux-claude-composer-live-e2e.test.sh 23 -tests/fm-codex-continuity-live-e2e.test.sh 21 -tests/fm-composer-matrix-live-e2e.test.sh 23 -tests/fm-control-relaunch.test.sh 48210 -tests/fm-control.test.sh 54301 -tests/fm-cursor-harness.test.sh 30103 -tests/fm-cursor-primary-live-e2e.test.sh 21 -tests/fm-cursor-primary.test.sh 54947 -tests/fm-dispatch-resolve.test.sh 1800 -tests/fm-daemon.test.sh 26870 -tests/fm-documentation-audiences.test.sh 732 -tests/fm-extension-binding.test.sh 7398 -tests/fm-fleet-snapshot-view.test.sh 8547 -tests/fm-fleet-sync.test.sh 37749 -tests/fm-gate-refuse.test.sh 4977 -tests/fm-gitignore-config.test.sh 62 -tests/fm-gotmp.test.sh 1310 -tests/fm-grok-continuity-live-e2e.test.sh 20 -tests/fm-grok-stop-live-e2e.test.sh 21 -tests/fm-guard-stale-banner.test.sh 32981 -tests/fm-harness-adapter-instructions-live-e2e.test.sh 20 -tests/fm-harness-adapter-references.test.sh 55 -tests/fm-harness-liveness-drift-live-e2e.test.sh 21 -tests/fm-herdr-attached-viewer-live-e2e.test.sh 19000 -tests/fm-herdr-session-cleanup.test.sh 6704 -tests/fm-herdr-submit-confirm-live-e2e.test.sh 23 -tests/fm-herdr-version-floor-live-e2e.test.sh 23 -tests/fm-home-summary-refresh.test.sh 34793 -tests/fm-inactive-reconcile.test.sh 74399 -tests/fm-kimi-harness.test.sh 18015 -tests/fm-lint-workflows.test.sh 855 -tests/fm-live-gate.test.sh 6000 -tests/fm-muse-harness.test.sh 55572 -tests/fm-muse-signals-live-e2e.test.sh 23 -tests/fm-no-mistakes-required.test.sh 370 -tests/fm-omp-harness.test.sh 59969 -tests/fm-on.test.sh 34087 -tests/fm-opencode-primary-live-e2e.test.sh 21 -tests/fm-operational-input.test.sh 231 -tests/fm-peek-remote.test.sh 1018 -tests/fm-pending-reply.test.sh 86711 -tests/fm-pi-branch-extension.test.sh 22239 -tests/fm-pi-branch-live-e2e.test.sh 56 -tests/fm-pi-branch-responsiveness-live-e2e.test.sh 21 -tests/fm-pi-primary-live-e2e.test.sh 20 -tests/fm-pi-watch-extension.test.sh 42970 +tests/fm-backend-cmux.test.sh 3498 +tests/fm-backend-orca.test.sh 23381 +tests/fm-backend-tmux-smoke.test.sh 363 +tests/fm-backend-zellij-smoke.test.sh 21 +tests/fm-backend-zellij.test.sh 9064 +tests/fm-backend.test.sh 21658 +tests/fm-backlog-atomicity.test.sh 196948 +tests/fm-backlog-handoff.test.sh 51990 +tests/fm-backlog-read-bound.test.sh 24288 +tests/fm-bearings-board-lavish-live-e2e.test.sh 48 +tests/fm-bearings-board-render.test.sh 12591 +tests/fm-bearings-board.test.sh 36490 +tests/fm-bearings-snapshot.test.sh 171176 +tests/fm-bootstrap-network-parallel.test.sh 9539 +tests/fm-bootstrap.test.sh 46634 +tests/fm-branch-supervision.test.sh 8915 +tests/fm-busy-adapter-wiring.test.sh 27817 +tests/fm-busy-state.test.sh 2990 +tests/fm-calm-claude-mod-live-e2e.test.sh 46 +tests/fm-calm-claude-mod-plugin.test.sh 172 +tests/fm-calm-claude-mod.test.sh 1252 +tests/fm-calm-pi-extension.test.sh 45128 +tests/fm-check-unregister.test.sh 464 +tests/fm-ci-workflow.test.sh 2073 +tests/fm-classify-corr-token.test.sh 49294 +tests/fm-classify-decision-key.test.sh 3336 +tests/fm-claude-stop-autoarm-live-e2e.test.sh 45 +tests/fm-claude-stop-autoarm.test.sh 60797 +tests/fm-claude-trust.test.sh 10410 +tests/fm-cmux-claude-composer-live-e2e.test.sh 47 +tests/fm-codex-continuity-live-e2e.test.sh 71 +tests/fm-codex-hook-layer-live-e2e.test.sh 47 +tests/fm-composer-codex-idle-live-e2e.test.sh 229 +tests/fm-composer-matrix-live-e2e.test.sh 47 +tests/fm-contributions.test.sh 35676 +tests/fm-control-relaunch.test.sh 137013 +tests/fm-control.test.sh 39524 +tests/fm-cursor-harness.test.sh 30212 +tests/fm-cursor-primary-live-e2e.test.sh 72 +tests/fm-cursor-primary.test.sh 52269 +tests/fm-daemon.test.sh 27262 +tests/fm-dispatch-resolve.test.sh 4397 +tests/fm-documentation-audiences.test.sh 847 +tests/fm-extension-binding.test.sh 9053 +tests/fm-fleet-snapshot-view.test.sh 17465 +tests/fm-fleet-sync.test.sh 35983 +tests/fm-gate-refuse.test.sh 5328 +tests/fm-gemini-harness.test.sh 938 +tests/fm-gitignore-config.test.sh 58 +tests/fm-gotmp.test.sh 1320 +tests/fm-grok-continuity-live-e2e.test.sh 45 +tests/fm-grok-stop-live-e2e.test.sh 46 +tests/fm-guard-stale-banner.test.sh 14968 +tests/fm-harness-adapter-instructions-live-e2e.test.sh 48 +tests/fm-harness-adapter-references.test.sh 83 +tests/fm-harness-liveness-drift-live-e2e.test.sh 881 +tests/fm-harness-precedence.test.sh 3661 +tests/fm-herdr-pi-stale-registration-live-e2e.test.sh 47 +tests/fm-herdr-session-cleanup.test.sh 6828 +tests/fm-herdr-submit-confirm-live-e2e.test.sh 46 +tests/fm-herdr-version-floor-live-e2e.test.sh 72 +tests/fm-home-summary-refresh.test.sh 37264 +tests/fm-inactive-reconcile.test.sh 53178 +tests/fm-kimi-harness.test.sh 19151 +tests/fm-lint-workflows.test.sh 785 +tests/fm-live-gate.test.sh 1755 +tests/fm-mail-check.test.sh 9162 +tests/fm-mail.test.sh 9703 +tests/fm-muse-harness.test.sh 40970 +tests/fm-muse-signals-live-e2e.test.sh 77 +tests/fm-nm-test-contract.test.sh 128 +tests/fm-no-mistakes-required.test.sh 247 +tests/fm-omp-harness.test.sh 47734 +tests/fm-omp-primary-live-e2e.test.sh 46 +tests/fm-on.test.sh 11001 +tests/fm-opencode-primary-live-e2e.test.sh 48 +tests/fm-operational-input.test.sh 221 +tests/fm-peek-remote.test.sh 964 +tests/fm-pending-reply.test.sh 28255 +tests/fm-pi-branch-extension.test.sh 60394 +tests/fm-pi-branch-live-e2e.test.sh 72 +tests/fm-pi-branch-responsiveness-live-e2e.test.sh 13121 +tests/fm-pi-codex-native.test.sh 46 +tests/fm-pi-primary-live-e2e.test.sh 47 +tests/fm-pi-watch-extension.test.sh 50637 tests/fm-pi-windows-shell-invocation.test.sh 5121 -tests/fm-pr-check-security.test.sh 172215 -tests/fm-procevent-quota.test.sh 1949 -tests/fm-procevent-when.test.sh 17392 -tests/fm-procevent.test.sh 69715 -tests/fm-project-origin.test.sh 137 -tests/fm-public-followup.test.sh 196745 -tests/fm-quota-array-dispatch-live-e2e.test.sh 21 -tests/fm-quota-choose.test.sh 1461 -tests/fm-remote-backlog-handoff.test.sh 41432 -tests/fm-remote-doctor.test.sh 5198 -tests/fm-remote-entrypoint.test.sh 132 -tests/fm-remote-herdr-guard.test.sh 1500 -tests/fm-remote-job-orphan-reap.test.sh 2972 -tests/fm-remote-job.test.sh 59603 -tests/fm-remote-reply.test.sh 101690 -tests/fm-remote-secondmate-lifecycle-e2e.test.sh 209631 -tests/fm-remote-secondmate-parent-binding.test.sh 29562 -tests/fm-remote-secondmate-trace-context.test.sh 67096 -tests/fm-remote-transport-lanes.test.sh 63976 -tests/fm-secondmate-harness.test.sh 151589 -tests/fm-secondmate-lifecycle-e2e.test.sh 8793 -tests/fm-secondmate-liveness.test.sh 18146 -tests/fm-secondmate-reconcile.test.sh 62726 -tests/fm-secondmate-restart.test.sh 119085 -tests/fm-secondmate-safety.test.sh 57689 -tests/fm-secondmate-sync.test.sh 17183 -tests/fm-send-inbox-doorbell-live-e2e.test.sh 22 -tests/fm-send-inbox.test.sh 38956 -tests/fm-send-remote-delivery.test.sh 27686 -tests/fm-send-resolve-key.test.sh 19619 -tests/fm-send-secondmate-marker-herdr-e2e.test.sh 51 -tests/fm-send-secondmate-marker.test.sh 6252 -tests/fm-session-lock-ancestry.test.sh 1414 -tests/fm-session-start.test.sh 156952 -tests/fm-sessionstart-hook-live-e2e.test.sh 20 -tests/fm-sessionstart-instruction-refresh-live-e2e.test.sh 22 -tests/fm-sessionstart-nudge.test.sh 66194 -tests/fm-shared-captain-inheritance.test.sh 6108 -tests/fm-spawn-dispatch-profile.test.sh 63996 -tests/fm-spawn-pool-base-freshen.test.sh 34920 -tests/fm-spawn-worktree-settle.test.sh 5687 -tests/fm-startup-memory-budget.test.sh 6964 -tests/fm-startup-network.test.sh 62274 -tests/fm-stow-cascade.test.sh 3101 -tests/fm-subagent-pretool-check.test.sh 1030 -tests/fm-supervision-events.test.sh 719 -tests/fm-tangle-guard.test.sh 9662 -tests/fm-task-delivery.test.sh 5952 -tests/fm-task-inbox.test.sh 25369 -tests/fm-teardown-endpoint-safety.test.sh 4620 -tests/fm-teardown.test.sh 97603 -tests/fm-test-fixture-cleanup.test.sh 915 -tests/fm-test-fixtures.test.sh 151 -tests/fm-test-isolation-proof.test.sh 2567 -tests/fm-tmux-agent-liveness.test.sh 1516 -tests/fm-turnend-foreign-owner-arm-fix.test.sh 2530 -tests/fm-tool-update-check.test.sh 14176 -tests/fm-trace-context-lib.test.sh 209 -tests/fm-trace-context-spawn.test.sh 44702 -tests/fm-turnend-guard.test.sh 42565 -tests/fm-update.test.sh 5212 -tests/fm-vendor-auth-probe.test.sh 43316 -tests/fm-voice-relay.test.sh 28699 -tests/fm-wake-daemon-lifecycle-e2e.test.sh 7381 -tests/fm-wake-drain-open-decisions-cursor.test.sh 20629 -tests/fm-wake-drain-open-decisions.test.sh 6240 -tests/fm-wake-drain-outcome-backstop.test.sh 15182 -tests/fm-wake-drain-unread-status.test.sh 35078 -tests/fm-wake-queue.test.sh 56674 -tests/fm-watch-arm.test.sh 69464 -tests/fm-watch-checkpoint.test.sh 5779 -tests/fm-watch-recovery-loop.test.sh 58731 -tests/fm-watch-triage.test.sh 262626 -tests/fm-watcher-lock.test.sh 88554 +tests/fm-pr-check-security.test.sh 226546 +tests/fm-pr-reviewers.test.sh 273 +tests/fm-pr-state-live-e2e.test.sh 45 +tests/fm-pr-state.test.sh 531 +tests/fm-procevent-quota.test.sh 1900 +tests/fm-procevent-when.test.sh 23805 +tests/fm-procevent.test.sh 221745 +tests/fm-project-origin.test.sh 136 +tests/fm-public-followup.test.sh 153508 +tests/fm-quota-array-dispatch-live-e2e.test.sh 71 +tests/fm-quota-choose.test.sh 1484 +tests/fm-remote-backlog-handoff.test.sh 73123 +tests/fm-remote-doctor.test.sh 13889 +tests/fm-remote-entrypoint.test.sh 108 +tests/fm-remote-herdr-guard.test.sh 3044 +tests/fm-remote-job-orphan-reap.test.sh 2905 +tests/fm-remote-job.test.sh 59354 +tests/fm-remote-reply.test.sh 118669 +tests/fm-remote-secondmate-lifecycle-e2e.test.sh 241208 +tests/fm-remote-secondmate-parent-binding.test.sh 32176 +tests/fm-remote-secondmate-trace-context.test.sh 59689 +tests/fm-remote-transport-lanes.test.sh 62635 +tests/fm-rovo-harness.test.sh 14322 +tests/fm-rovo-signals-live-e2e.test.sh 48 +tests/fm-secondmate-harness.test.sh 163801 +tests/fm-secondmate-lifecycle-e2e.test.sh 9633 +tests/fm-secondmate-liveness.test.sh 10402 +tests/fm-secondmate-reconcile.test.sh 97544 +tests/fm-secondmate-restart.test.sh 44488 +tests/fm-secondmate-safety.test.sh 127260 +tests/fm-secondmate-sync.test.sh 54502 +tests/fm-send-agy-confirm.test.sh 3983 +tests/fm-send-inbox-doorbell-live-e2e.test.sh 46 +tests/fm-send-inbox.test.sh 38632 +tests/fm-send-remote-delivery.test.sh 27717 +tests/fm-send-resolve-key.test.sh 28685 +tests/fm-send-secondmate-marker-herdr-e2e.test.sh 52 +tests/fm-send-secondmate-marker.test.sh 5309 +tests/fm-session-lock-ancestry.test.sh 2857 +tests/fm-session-start.test.sh 179350 +tests/fm-sessionstart-hook-live-e2e.test.sh 97 +tests/fm-sessionstart-instruction-refresh-live-e2e.test.sh 46 +tests/fm-sessionstart-nudge.test.sh 66247 +tests/fm-shared-captain-inheritance.test.sh 5687 +tests/fm-spawn-dispatch-profile.test.sh 138433 +tests/fm-spawn-pool-base-freshen.test.sh 62249 +tests/fm-spawn-worktree-settle.test.sh 8482 +tests/fm-startup-memory-budget.test.sh 7392 +tests/fm-startup-network.test.sh 61336 +tests/fm-stat-shadowing.test.sh 48 +tests/fm-stow-cascade.test.sh 3022 +tests/fm-subagent-pretool-check.test.sh 949 +tests/fm-supervision-events.test.sh 659 +tests/fm-tangle-guard.test.sh 7470 +tests/fm-task-delivery.test.sh 19784 +tests/fm-task-inbox.test.sh 30004 +tests/fm-tasks-axi.test.sh 1953 +tests/fm-teardown-endpoint-safety.test.sh 33210 +tests/fm-teardown.test.sh 145174 +tests/fm-test-fixture-cleanup.test.sh 937 +tests/fm-test-fixtures.test.sh 1562 +tests/fm-test-isolation-proof.test.sh 2692 +tests/fm-tmux-agent-liveness.test.sh 1953 +tests/fm-tool-update-check.test.sh 13832 +tests/fm-trace-context-lib.test.sh 227 +tests/fm-trace-context-spawn.test.sh 49071 +tests/fm-turnend-foreign-owner-arm-fix.test.sh 2397 +tests/fm-turnend-guard.test.sh 33450 +tests/fm-update.test.sh 11572 +tests/fm-vendor-auth-probe.test.sh 45255 +tests/fm-voice-relay.test.sh 32486 +tests/fm-wake-daemon-lifecycle-e2e.test.sh 7477 +tests/fm-wake-drain-open-decisions-cursor.test.sh 38506 +tests/fm-wake-drain-open-decisions.test.sh 6890 +tests/fm-wake-drain-outcome-backstop.test.sh 44076 +tests/fm-wake-drain-unread-status.test.sh 16169 +tests/fm-wake-queue.test.sh 85252 +tests/fm-watch-arm.test.sh 68479 +tests/fm-watch-checkpoint.test.sh 6076 +tests/fm-watch-recovery-loop.test.sh 58946 +tests/fm-watch-triage.test.sh 697969 +tests/fm-watcher-lock.test.sh 108940 EOF } @@ -2183,6 +2208,13 @@ fi # An explicit --jobs names a concurrency for exactly the selection given, so an # unproven script in it is a refusal rather than something to schedule around. if [ "$JOBS" -gt 1 ] && [ "$AUTO_CONCURRENCY" -eq 0 ]; then + # A single heavy suite can occupy a whole serial shard. Its family may have + # a separate concurrency proof, but that never changes this lane's contract. + if [ "$MODE" = lane ]; then + case "$LANE" in + portable-serial|portable-serial-*) die "--jobs $JOBS refused: portable serial lanes stay serial; use --jobs 1" ;; + esac + fi for s in "${SCRIPTS[@]}"; do if ! script_allows_concurrency "$s"; then die "--jobs $JOBS refused: $s is not in the proven-isolated set (see bin/fm-test-isolation-proof.sh --list) and its family has no recorded concurrent proof. Unproven stateful scripts stay serial." diff --git a/docs/fm-test-portable-shards.md b/docs/fm-test-portable-shards.md index a327902b99c..c7d6be38fde 100644 --- a/docs/fm-test-portable-shards.md +++ b/docs/fm-test-portable-shards.md @@ -57,8 +57,10 @@ Each shard is still strictly serial in itself, and separate runners mean no two `.github/workflows/ci.yml` derives the same `n` from `strategy.job-total` rather than a literal, so changing the shard count in either file without the other fails the lane loudly instead of leaving part of the required suite unrun. Assignment is longest-processing-time bin packing over per-script duration hints embedded in `bin/fm-test-run.sh`. -The embedded hints include the slowest measurements retained from the `fm-test-timing-portable-serial-*` artifacts of three green CI runs on 2026-09-01, [33558082172](https://github.com/kunchenguid/firstmate/actions/runs/33558082172), [33523597838](https://github.com/kunchenguid/firstmate/actions/runs/33523597838), and [33463326167](https://github.com/kunchenguid/firstmate/actions/runs/33463326167), the completed-script measurements from [run 34342484144](https://github.com/kunchenguid/firstmate/actions/runs/34342484144), plus the 5121 ms native-Windows focused runner measurement for `tests/fm-pi-windows-shell-invocation.test.sh` from 2026-09-06T21:02Z. -Taking the slowest of several CI runs rather than a single run keeps the balance honest on a slow runner. +The serial hints were refreshed from successful per-script records in the `fm-test-timing-portable-serial-*` artifacts of the complete green [run 35279383618](https://github.com/kunchenguid/firstmate/actions/runs/35279383618) and the available completed shards of [run 35282466441](https://github.com/kunchenguid/firstmate/actions/runs/35282466441) on 2026-09-17. +Together these cover all 176 serial scripts at refresh time; retain the slower successful sample where both exist. +The native-Windows-only `tests/fm-pi-windows-shell-invocation.test.sh` retains its separate 5121 ms measurement from 2026-09-06T21:02Z instead of a portable capability skip. +An unfinished or failed invocation is not a healthy duration sample. A script with no hint gets the conservative `PORTABLE_SERIAL_DEFAULT_WEIGHT_MS` default. Hints only affect balance: the coverage guard keeps the partition complete and disjoint whatever they say, so a stale hint costs a slower shard rather than lost coverage. Balance is still worth keeping current, because enough unmeasured scripts let one shard carry more than twice another shard's real work and reach the job cap while another runner sits idle. @@ -66,10 +68,12 @@ That is not hypothetical: by 2026-09-01 the lane had grown from 116 to 139 scrip `bin/fm-test-run.sh --check-coverage` now reports the unmeasured share as `serial_unhinted=` and refuses past `PORTABLE_SERIAL_MAX_UNHINTED_PERCENT`, so hint drift fails the coverage guard instead of silently pushing one shard into its job cap. Refresh the hints whenever the serial lane gains scripts, rather than waiting for that bound to trip. -`bin/fm-test-run.sh` owns the per-shard packing, so its `--check-coverage` output is the current account of lane size, shard composition, and balance rather than a copied table. -Run 34342484144 observed a shard reach about 20 minutes of passing work, so the 30-minute job cap keeps meaningful hang-tripwire margin for job setup and runner-speed spread. - -The single longest script, `tests/fm-watch-triage.test.sh` at 262626 ms, is the floor for any shard count. +`bin/fm-test-run.sh` owns the per-shard packing, so its `--check-coverage` output is the current account of lane size and coverage rather than a copied inventory. +Nine serial runners pack the refreshed measurements into a longest modeled script sum of 697969 ms (11m38s), with other shards near 10m36s. +The longest script, `tests/fm-watch-triage.test.sh`, legitimately occupies one whole shard and is the indivisible floor for this layout. +This is a packing estimate, not measured new-workflow execution or an end-to-end latency guarantee. +Existing job timeouts remain hang tripwires; they are not the desired healthy duration. +`tests/fm-ci-workflow.test.sh` compares the parsed CI matrix to the executable runner lanes, and the runner rejects parallel `--jobs` on a serial lane even when that shard has only one member. Refresh the CI-derived hints by downloading the per-shard timing artifacts from several green CI runs and replacing the `portable_serial_weight_hints` table in `bin/fm-test-run.sh` with the slowest measured `duration_ms` per `path`: @@ -77,13 +81,14 @@ Refresh the CI-derived hints by downloading the per-shard timing artifacts from for run in ; do gh run download "$run" -R kunchenguid/firstmate --pattern 'fm-test-timing-portable-serial-*' -D "/tmp/fm-serial/$run" done -jq -r '.scripts[] | [.path, .duration_ms] | @tsv' /tmp/fm-serial/*/*.json \ +jq -r '.scripts[] | select(.exit == 0) | [.path, .duration_ms] | @tsv' /tmp/fm-serial/*/*/*.json \ | awk -F'\t' '$2 > m[$1] { m[$1] = $2 } END { for (p in m) print p, m[p] }' \ | LC_ALL=C sort bin/fm-test-run.sh --check-coverage ``` -A timed-out shard uploads no artifact, so pick runs where every serial shard is green or the lane's slowest scripts go unmeasured in exactly the shard that needs them most. +A timed-out shard may upload no artifact, so include a complete green run or the slowest scripts go unmeasured in exactly the shard that needs them most. +Completed shards from a partial run can supplement that complete baseline, but never treat missing tail scripts or the timeout duration as successful samples. Measure native-Windows-only scripts through the focused Git Bash runner and retain that `duration_ms` separately, because the portable CI shards skip them. ## Coverage guard @@ -99,6 +104,18 @@ Portable shards, each portable serial shard, and the Herdr lane upload runner-ge `bin/fm-test-run.sh --aggregate-json` creates the combined summary artifact. `.github/workflows/ci.yml` owns the exact artifact names and aggregation wiring. +## Lint partitions and end-to-end latency + +`bin/fm-lint.sh` owns two canonical CI partitions, each running the same full source-aware ShellCheck analysis with two bounded workers, pinned versions, workflow validation, and backend-purity checks. +Its `--list-files` interface exposes partition membership; `tests/fm-lint.test.sh` verifies complete/disjoint executed roots and unchanged analysis flags. +The workflow uploads each partition's quiet telemetry to distinguish analysis cost, memory use, and host contention. +No fast mode, path skips, reduced checks, or paid runner provisioning is part of this layout. + +The performance objective is a complete green run under fifteen minutes including start delay: roughly twelve minutes of longest-path execution, at most two minutes of runner delay, and less than one minute of other overhead. +The candidate uses fourteen long-lived Linux jobs (nine serial, two parallel, Herdr, two lint), plus short checks and macOS; insufficient shared account capacity can erase the packing gain. +Compare complete before/after runs, preserve cancelled and partial-run evidence, and measure a representative normal-run sample before claiming a P95 improvement. +The workflow retains per-PR supersession without cancelling main pushes or changing the compliance workflow's event semantics. + ## Local entry points [CONTRIBUTING.md](../CONTRIBUTING.md) owns the local test policy and common entry points. @@ -109,7 +126,7 @@ Portable shards, each portable serial shard, and the Herdr lane upload runner-ge | Lane | Bound | Rationale | |---|---|---| | portable parallel 1/2 | See [CI workflow](../.github/workflows/ci.yml) | The workflow owns the parallel cap rationale and its evidence limits. | -| portable serial 1-5 | job `timeout-minutes: 30` | Current runners can take about 20 minutes; the 30-minute cap remains a hang tripwire while leaving margin for job setup and runner-speed spread. | +| portable serial shards | See [CI workflow](../.github/workflows/ci.yml) | Packing estimates are not healthy execution bounds; the existing cap remains a hang tripwire. | | Herdr | family-run step `timeout-minutes: 20`; job `timeout-minutes: 75` backstop | Healthy runs finished around 7 minutes before this lane gained `fm-backend-herdr-focus-flash-e2e`, which measures about 2 minutes against a real lab locally, so the step bound is still the hang tripwire (cleanup and timing artifacts still upload) while the job cap stays a last-resort backstop. Refresh this figure from the lane's uploaded timing artifact. | Timeouts are intended as hang tripwires; a passing coverage guard does not establish a healthy job duration. diff --git a/tests/fm-ci-workflow.test.sh b/tests/fm-ci-workflow.test.sh index fd2f7918493..78795eebe4b 100755 --- a/tests/fm-ci-workflow.test.sh +++ b/tests/fm-ci-workflow.test.sh @@ -151,6 +151,37 @@ CAPS pass "the already-measured lane bounds are unchanged" } +test_ci_matrices_match_executable_partitions() { + ruby -ryaml -ropen3 - "$CI_WORKFLOW" "$ROOT" <<'RUBY' || fail "CI partition contract" +jobs = YAML.load_file(ARGV[0]).fetch("jobs") +root = ARGV[1] +serial = jobs.fetch("tests-portable-serial").fetch("strategy") +raise "serial failures must not cancel other shards" unless serial.fetch("fail-fast") == false +matrix = serial.fetch("matrix") +raise "unexpected serial dimensions" unless matrix.keys == ["shard"] +shards = matrix.fetch("shard") +lanes, status = Open3.capture2(File.join(root, "bin/fm-test-run.sh"), "--list-lanes") +raise "cannot list runner lanes" unless status.success? +actual = lanes.lines.map(&:strip).select { |l| l.match?(/\Aportable-serial-\d+of\d+\z/) } +expected = shards.map { |s| "portable-serial-#{s}of#{shards.length}" } +raise "CI matrix and runner disagree" unless actual.sort == expected.sort +lint = jobs.fetch("lint").fetch("strategy") +raise "lint failures must not cancel another partition" unless lint.fetch("fail-fast") == false +matrix = lint.fetch("matrix") +raise "unexpected lint dimensions" unless matrix.keys == ["partition"] +parts = matrix.fetch("partition") +roots = parts.flat_map do |p| + output, result = Open3.capture2(File.join(root, "bin/fm-lint.sh"), "--partition", "#{p}of#{parts.length}", "--list-files") + raise "unsupported lint partition" unless result.success? + output.lines.map(&:strip) +end +canonical, result = Open3.capture2({"CI" => "true"}, File.join(root, "bin/fm-lint.sh"), "--list-files") +raise "lint matrix loses or duplicates canonical roots" unless result.success? && roots.sort == canonical.lines.map(&:strip).sort +RUBY + pass "CI matrices cover every executable serial lane and canonical lint root exactly once" +} + +test_ci_matrices_match_executable_partitions test_pr_pushes_supersede_within_one_pr test_separate_prs_do_not_cancel_each_other test_main_pushes_are_never_cancelled diff --git a/tests/fm-lint.test.sh b/tests/fm-lint.test.sh index f2fe3a528d4..63d321eb56d 100755 --- a/tests/fm-lint.test.sh +++ b/tests/fm-lint.test.sh @@ -178,6 +178,48 @@ test_list_files_reports_the_shell_inventory() { pass "fm-lint.sh --list-files reports the complete shell inventory" } +test_canonical_partitions_preserve_full_lint() { + local tmp fakebin all part selected log flags mode rc option + tmp=$(fm_test_tmproot fm-lint-partitions) + fakebin="$tmp/bin" + mkdir -p "$fakebin" + all=$(CI=true "$LINT" --list-files | LC_ALL=C sort) + : > "$tmp/union" + for part in 1of2 2of2; do + selected=$(CI=false GITHUB_ACTIONS=false "$LINT" --partition "$part" --list-files) \ + || fail "partition $part must select full canonical roots even on a local branch" + [ -n "$selected" ] || fail "empty lint partition $part" + printf '%s\n' "$selected" >> "$tmp/union" + [ "$selected" = "$("$LINT" --partition "$part" --list-files)" ] \ + || fail "partition $part is nondeterministic" + log="$tmp/$part.roots" + flags="$tmp/$part.flags" + mode="$tmp/$part.mode" + fm_lint_stub_shellcheck "$fakebin" "$log" + PATH="$fakebin:$PATH" FM_TEST_FLAG_LOG="$flags" FM_TEST_MODE_LOG="$mode" \ + "$LINT" --partition "$part" > "$tmp/$part.out" 2>&1 \ + || fail "canonical partition $part failed: $(cat "$tmp/$part.out")" + [ "$(LC_ALL=C sort "$log")" = "$(printf '%s\n' "$selected" | LC_ALL=C sort)" ] \ + || fail "partition $part executed a different root set than it listed" + [ "$(LC_ALL=C sort -u "$flags")" = "$(printf 'exclude=none\nexternal-sources=yes')" ] \ + || fail "partition $part weakened source-aware analysis" + [ "$(LC_ALL=C sort -u "$mode")" = on ] || fail "partition $part disabled full analysis" + done + [ "$(LC_ALL=C sort "$tmp/union")" = "$all" ] || fail "lint partitions lose or duplicate canonical roots" + for option in 0of2 3of2 1of3; do + rc=0 + "$LINT" --partition "$option" --list-files > "$tmp/refused" 2>&1 || rc=$? + [ "$rc" = 2 ] || fail "invalid partition $option was not refused" + done + rc=0 + "$LINT" --partition 1of2 --fast > "$tmp/refused" 2>&1 || rc=$? + [ "$rc" = 2 ] || fail "partition accepted --fast" + rc=0 + "$LINT" --partition 1of2 bin/fm-lint.sh > "$tmp/refused" 2>&1 || rc=$? + [ "$rc" = 2 ] || fail "partition accepted an explicit subset" + pass "two canonical lint partitions preserve complete source-aware coverage and reject weakened modes" +} + # fm_lint_stub_git : install a git stub for the changed-file mode # tests below. Its answers are driven by env vars the caller sets before # invoking fm-lint.sh, so those tests can steer git state without depending on @@ -1364,6 +1406,7 @@ SH test_help_reports_the_complete_interface test_list_files_reports_the_shell_inventory +test_canonical_partitions_preserve_full_lint test_fast_mode_disables_extended_analysis test_ci_defaults_to_full_analysis test_ci_rejects_explicit_fast_mode diff --git a/tests/fm-mail-check.test.sh b/tests/fm-mail-check.test.sh index 36152594924..736bd0195d5 100644 --- a/tests/fm-mail-check.test.sh +++ b/tests/fm-mail-check.test.sh @@ -256,6 +256,56 @@ test_repeated_failure_that_queued_new_mail_still_wakes() { pass "fm-mail-check: a repeated failure that queued new mail still wakes" } +test_large_poll_output_is_drained() { + # A pipe reader that exits at the first match closes before the producer has + # written this poll's output. Ignored SIGPIPE makes that race observable as + # stderr noise instead of silently terminating a pipeline subprocess. + # Exercise both summary selectors and both wake predicates via the real check. + local tmpbin home shape attempt out expected + tmpbin="$TMP_ROOT/large-poll/bin" + mkdir -p "$tmpbin" + cp "$CHECK" "$tmpbin/" + for lib in fm-timeout-lib.sh fm-pr-lib.sh fm-line-cap-lib.sh fm-check-lib.sh; do + ln -s "$ROOT/bin/$lib" "$tmpbin/$lib" + done + cat > "$tmpbin/fm-mail.sh" <<'SH' +#!/usr/bin/env bash +printf 'fm-mail: woke for 42\n' +case "$FM_TEST_POLL_SHAPE" in + success) + awk 'BEGIN { for (i=0; i<20000; i++) print "poll diagnostic padding padding padding" }' + printf 'fm-mail: woke for 43\n' + ;; + preferred) + printf 'fm-mail: connection refused\n' >&2 + awk 'BEGIN { for (i=0; i<20000; i++) print "fm-mail: later diagnostic padding padding" }' >&2 + exit 1 + ;; + fallback) + printf 'raw connection failure\n' >&2 + awk 'BEGIN { for (i=0; i<20000; i++) print "raw later diagnostic padding padding" }' >&2 + exit 1 + ;; +esac +SH + chmod +x "$tmpbin/fm-mail.sh" + for shape in success preferred fallback; do + home=$(make_home "large-$shape") + case "$shape" in + success) expected='mail: new mail: woke for 43' ;; + preferred) expected='mail: connection refused' ;; + fallback) expected='mail: raw connection failure' ;; + esac + for attempt in 1 2; do + out="$home/out-$attempt.txt" + (trap '' PIPE; run_check "$home" "$out" "$tmpbin/fm-mail-check.sh" FM_TEST_POLL_SHAPE="$shape") + [ "$(cat "$out")" = "$expected" ] || fail "large $shape poll $attempt must emit only its summary: $(cat "$out")" + [ "$(wc -l < "$out" | tr -d '[:space:]')" = 1 ] || fail "large $shape poll must be exactly one line" + done + done + pass "fm-mail-check: large repeated polls drain every reader without output noise" +} + test_repeated_timeout_still_wakes() { # A timeout can kill the poll after wake_for queued mail and before the # woke-for line is printed. Difference-record silence would then leave that @@ -418,6 +468,7 @@ test_fail_closed_poll_after_wake_reports_the_failure test_repeated_status4_fail_closed_still_wakes test_repeated_status2_stays_queued_still_wakes test_repeated_failure_that_queued_new_mail_still_wakes +test_large_poll_output_is_drained test_repeated_timeout_still_wakes test_repeated_heal_failure_stays_silent test_missing_mail_plane_is_reported \ No newline at end of file diff --git a/tests/fm-test-run.test.sh b/tests/fm-test-run.test.sh index bb7c6b3c5fa..b3100151480 100755 --- a/tests/fm-test-run.test.sh +++ b/tests/fm-test-run.test.sh @@ -1142,8 +1142,8 @@ test_portable_serial_shards_partition_the_serial_lane() { shard=1 while [ "$shard" -le "$count" ]; do listed=$("$RUNNER" --list --lane "portable-serial-${shard}of${count}" | wc -l | tr -d ' ') - [ "$listed" -ge 2 ] \ - || fail "portable-serial-${shard}of${count} holds only $listed script(s)" + # One expensive suite can legitimately occupy a whole runner. Non-empty + # coverage is asserted above; script counts are not duration weights. [ "$listed" -le "$cap" ] \ || fail "portable-serial-${shard}of${count} holds $listed of $total scripts" shard=$((shard + 1)) @@ -1223,7 +1223,7 @@ test_jobs_requires_proven_isolated() { rc=$? set -e [ "$rc" -eq 2 ] || fail "--jobs with portable-serial must refuse (exit 2), got $rc" - grep -Fq 'not in the proven-isolated set' "$tmp/err" \ + grep -Fq 'portable serial lanes stay serial' "$tmp/err" \ || fail "--jobs refusal message missing: $(cat "$tmp/err")" set +e "$RUNNER" --jobs 2 tests/fm-afk-inject-e2e.test.sh >"$tmp/out2" 2>"$tmp/err2" @@ -1237,7 +1237,7 @@ test_jobs_requires_proven_isolated() { rc=$? set -e [ "$rc" -eq 2 ] || fail "--jobs with a portable serial shard must refuse, got $rc" - grep -Fq 'not in the proven-isolated set' "$tmp/err3" \ + grep -Fq 'portable serial lanes stay serial' "$tmp/err3" \ || fail "shard --jobs refusal message missing: $(cat "$tmp/err3")" rm -rf "$tmp" pass "--jobs refuses non-proven / stateful selections" diff --git a/tests/fm-watcher-lock.test.sh b/tests/fm-watcher-lock.test.sh index c0d37f0cf61..37af8ac641e 100755 --- a/tests/fm-watcher-lock.test.sh +++ b/tests/fm-watcher-lock.test.sh @@ -34,6 +34,62 @@ drain_and_ack() { # --recovery-generation "$generation" } +test_wait_deadline_reaps_a_stopped_child() { + # A stopped TERM-resistant child cannot finish graceful cleanup. The helper waited + # forever after its nominal deadline. An outer process-group deadline keeps + # this regression finite even if that bug returns. + python3 - "$ROOT/tests/wake-helpers.sh" <<'PY' || fail "bounded child cleanup regression" +import os +import signal +import subprocess +import sys + +script = r''' +. "$1" +bash -c 'trap "" TERM; kill -STOP "$$"; exec sleep 300' & +pid=$! +for i in $(seq 1 100); do + state=$(ps -p "$pid" -o stat=) + case "$state" in *T*) break ;; esac + sleep 0.01 +done +case "$state" in *T*) ;; *) kill -KILL "$pid"; exit 23 ;; esac +wait_for_exit "$pid" 2 +rc=$? +[ "$rc" = 124 ] || exit 21 +! kill -0 "$pid" 2>/dev/null || exit 22 +''' +p = subprocess.Popen([os.environ.get("BASH", "bash"), "-c", script, "_", sys.argv[1]], + start_new_session=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) +try: + out, err = p.communicate(timeout=15) +except subprocess.TimeoutExpired: + os.killpg(p.pid, signal.SIGKILL) + p.communicate() + raise SystemExit("wait_for_exit hung after its deadline on a stopped child") +if p.returncode or "survived TERM; sending KILL" not in err: + raise SystemExit(f"cleanup rc={p.returncode}, stdout={out}, stderr={err}") +PY + pass "wait deadline diagnoses and reaps a stopped test child without hanging" +} + +# Preserve the real watcher's trap diagnostics when testing its termination. +# A termination defect should fail this case promptly, not occupy a CI runner +# until the whole job times out and hides every following test. +stop_seed_watcher() { # + local pid=$1 out=$2 status=0 + kill -TERM "$pid" 2>/dev/null || true + wait_for_exit "$pid" 100 || status=$? + if [ "$status" -eq 124 ]; then + cat "$out" >&2 + fail "seed watcher survived TERM; see bounded wait/process/trap evidence above" + fi + if grep -E 'unexpected EOF|syntax error' "$out" >/dev/null; then + cat "$out" >&2 + fail "seed watcher emitted a shell parser error during termination" + fi +} + test_singleton_start() { local dir state fakebin out1 out2 pid1 pid2 live i dir=$(make_case singleton) @@ -575,7 +631,7 @@ test_arm_attaches_and_waits_for_live_fresh_watcher() { out="$dir/watch.out" armout="$dir/arm.out" # A genuinely live watcher with a fresh beacon already holds the singleton. - PATH="$fakebin:$PATH" FM_STATE_OVERRIDE="$state" FM_POLL=5 FM_SIGNAL_GRACE=1 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + PATH="$fakebin:$PATH" FM_STATE_OVERRIDE="$state" FM_POLL=5 FM_SIGNAL_GRACE=1 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" 2>&1 & wpid=$! i=0 while [ "$i" -lt 60 ]; do @@ -600,8 +656,7 @@ test_arm_attaches_and_waits_for_live_fresh_watcher() { [ "$(cat "$state/.watch.lock/pid" 2>/dev/null || true)" = "$wpid" ] || fail "arm disturbed the healthy watcher's lock" is_live_non_zombie "$armpid" || fail "arm exited while the seed watcher was still healthy" # After the seed dies without a successor, the attached arm must fail loudly. - kill "$wpid" 2>/dev/null || true - wait "$wpid" 2>/dev/null || true + stop_seed_watcher "$wpid" "$out" wait_for_exit "$armpid" 80 status=$? [ "$status" -ne 0 ] && [ "$status" -ne 124 ] || fail "attached arm did not fail after seed died (status $status)" @@ -616,7 +671,7 @@ test_attached_arm_signal_is_recorded_in_cycle_ledger() { fakebin="$dir/fakebin" out="$dir/watch.out" armout="$dir/arm.out" - PATH="$fakebin:$PATH" FM_STATE_OVERRIDE="$state" FM_POLL=5 FM_SIGNAL_GRACE=1 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" & + PATH="$fakebin:$PATH" FM_STATE_OVERRIDE="$state" FM_POLL=5 FM_SIGNAL_GRACE=1 FM_CHECK_INTERVAL=999999 FM_HEARTBEAT=999999 "$WATCH" > "$out" 2>&1 & wpid=$! i=0 while [ "$i" -lt 60 ]; do @@ -641,8 +696,7 @@ test_attached_arm_signal_is_recorded_in_cycle_ledger() { grep -q "arm_pid=$armpid.*watcher_pid=$wpid.*origin=attached.*exit_code=143.*signal=TERM.*reason=arm-interrupted" "$state/.watch-cycle-exits.log" \ || fail "attached arm signal was not recorded in the lifecycle ledger" is_live_non_zombie "$wpid" || fail "signaling an attached arm terminated the peer watcher" - kill "$wpid" 2>/dev/null || true - wait "$wpid" 2>/dev/null || true + stop_seed_watcher "$wpid" "$out" pass "attached arm signals record a classified lifecycle entry" } @@ -1108,6 +1162,7 @@ test_msys_pid_identity_uses_proc() { pass "MSYS process identity uses compatible /proc fields" } +test_wait_deadline_reaps_a_stopped_child test_singleton_start test_pid_identity_is_locale_invariant test_proc_pid_identity_ignores_wall_clock_and_detects_pid_reuse diff --git a/tests/wake-helpers.sh b/tests/wake-helpers.sh index da83bb3dc91..8e7106d8230 100644 --- a/tests/wake-helpers.sh +++ b/tests/wake-helpers.sh @@ -296,6 +296,9 @@ SH printf '%s\n' "$dir" } +# Only pass a process owned by this test. A deadline must also bound cleanup: +# TERM can be ignored or remain pending on a stopped child, so never follow it +# with an unbounded wait. Keep process evidence before the final owned-PID kill. wait_for_exit() { local pid=$1 limit=${2:-50} i=0 while [ "$i" -lt "$limit" ]; do @@ -306,7 +309,18 @@ wait_for_exit() { sleep 0.1 i=$((i + 1)) done - kill "$pid" 2>/dev/null || true + printf 'wait_for_exit: owned pid %s exceeded %s polls; sending TERM\n' "$pid" "$limit" >&2 + ps -p "$pid" -o pid= -o ppid= -o stat= -o command= >&2 2>/dev/null || true + kill -TERM "$pid" 2>/dev/null || true + i=0 + while [ "$i" -lt 20 ] && is_live_non_zombie "$pid"; do + sleep 0.1 + i=$((i + 1)) + done + if is_live_non_zombie "$pid"; then + printf 'wait_for_exit: owned pid %s survived TERM; sending KILL\n' "$pid" >&2 + kill -KILL "$pid" 2>/dev/null || true + fi wait "$pid" 2>/dev/null || true return 124 } From ab7be8111cd63f8fdde6bf45305cb5d453c08b83 Mon Sep 17 00:00:00 2001 From: kunchenguid Date: Thu, 17 Sep 2026 18:13:35 -0700 Subject: [PATCH 2/2] no-mistakes(document): Clarify lint partition documentation --- tests/fm-lint.test.sh | 20 ++++++++++---------- 1 file changed, 10 insertions(+), 10 deletions(-) diff --git a/tests/fm-lint.test.sh b/tests/fm-lint.test.sh index 63d321eb56d..75edfe85bae 100755 --- a/tests/fm-lint.test.sh +++ b/tests/fm-lint.test.sh @@ -1,17 +1,17 @@ #!/usr/bin/env bash # Parity guard for firstmate's shell-lint definition. # -# bin/fm-lint.sh must be the single owner that BOTH CI -# (.github/workflows/ci.yml) and the pre-push gate (.no-mistakes.yaml -# commands.lint) invoke, so the local lint can never diverge from CI again. +# bin/fm-lint.sh is the single owner invoked by CI +# (.github/workflows/ci.yml) and by the pre-push gate (.no-mistakes.yaml +# commands.lint). CI runs its two full-rigor canonical partitions; the local +# gate uses its context-selected default. Their selection differs deliberately, +# while this owner keeps analysis flags, configuration, and tool versions from +# drifting. # Regression origin: with no commands.lint configured, the local no-mistakes -# lint step never ran the deterministic -# `shellcheck bin/*.sh bin/backends/*.sh tests/*.sh`, so PRs passed local -# validation yet failed that exact check in CI on info/warning findings such as -# SC2015, SC1007, and SC2034. A second axis was tool-version skew: CI's -# ShellCheck floated with the runner image and still emitted SC2015, which -# ShellCheck retired in 0.11.0. fm-lint.sh now pins one exact version and both -# gates resolve it, so command, file set, config, AND version all match. +# lint step never ran the deterministic shell lint, so PRs passed local +# validation yet failed CI on info/warning findings such as SC2015, SC1007, and +# SC2034. A second axis was tool-version skew: CI's ShellCheck floated with the +# runner image and still emitted SC2015, which ShellCheck retired in 0.11.0. set -u # shellcheck source=tests/lib.sh