From 0427aea6e448161df3f24d7b4a219d6909427588 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 23:28:24 +0200 Subject: [PATCH 01/24] Timing verdicts leave QEMU, and audio leaves QEMU entirely The owner's ruling: a QEMU guest test asserts order, completion, content and counts, and never how long something took; its only time-based ending is a harness ceiling. Audio is judged on metal and nowhere else. Deleted outright: - gate A's thorough tier (`--audio-gate`, `--slow-usb`, the nightly `audio` shards, `tests/audio-baseline.toml`, `tests/common/stats.rs`, `tests/common/hostload.rs`) and every QEMU audio test: `audio_tone`, `audio_tone_load`, `metal_sim_null_audio`, `null_sink_shipped_client`'s QEMU arm, `doom_sound_flood`, `doom_music`, `soundd_log_stall`, `desktop_audio_client`, `hda_tone`, `hda_client_stall`, `hda_two_live_refused`, the playback half of `inspect_reads_its_owners`, the wav capture (`-audiodev wav` is `none` now) and `tests/common/hda.rs`. - The pass-cost verdict of `sched_check_build` (`tests/common/passcost.rs`). - `kernel_heartbeat`'s CPU-mask and gap verdicts (`src/heartbeat.rs`). - `panic_halts_the_others_first` and `netd_stalled_peer`, whose only verdicts were a 100 ms stamp bound and a busy fraction over 2 s. Timing halves cut, the rest kept: `latency_wake`'s p99 bound, `i8042_absent`'s 300 ms A/B, `timer_calibration`'s ppm bound (metal only now), `tlb_shootdown_waits`' disarmed upper bound, the 3 s bounds and watchdogs of `exit_wait_storm` and `blocking_read_stress`, `poll_wake_pipe`'s 200 ms per-wake deadline, `netd_lookup_let_go`'s "at once", the USB settle ceiling and call/ladder upper bounds, and the flush-bound inference in `log_ring_keeps_the_owners_slots`. Metal-only rows, riding existing boots or three new ones priced in `tests/metal-profile.toml`: `wake_storm_cost`, `audio_idle_suspend`, `hda_tone`, `hda_client_stall`, `null_sink_shipped_client`, `doom_sound_flood`, `doom_music`, `soundd_log_stall`. Disabled rows and issues whose only content was a QEMU timing red go with their tests; what timing properties now have no metal arm is `issues/build/timing-verdicts-ruled-off-qemu-have-no-metal-arm.md`. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/nightly.yml | 22 +- Cargo.toml | 31 +- ...doom-sound-flood-played-full-scale-once.md | 6 +- .../gate-a-first-run-to-record-its-host.md | 56 - issues/audio/gate-a-has-no-runner-baseline.md | 135 -- ...gate-a-suspend-structure-verdict-unread.md | 48 +- issues/audio/hda-tone-phase-check.md | 7 +- .../hda-tone-red-beyond-its-exemption.md | 40 - .../stop-the-device-voice-keep-the-wake.md | 15 - .../t14-wake-lateness-is-bimodal-per-boot.md | 188 -- .../thorough-tier-reds-on-unmodified-main.md | 187 -- ...-of-500-round-trips-beside-other-guests.md | 22 - ...completed-26-of-500-beside-other-guests.md | 32 - ...ncy-wake-reds-on-the-dev-host-at-a-rate.md | 38 - ...tes-ci-sample-is-eight-days-stale-twice.md | 43 - .../there-is-no-attributed-session-ledger.md | 24 +- ...rdicts-ruled-off-qemu-have-no-metal-arm.md | 52 + ...-storm-cost-red-under-induced-host-load.md | 63 - .../the-t14-boots-toyos-unattended.md | 4 +- kernel/src/drivers/hda.rs | 4 +- src/build.rs | 11 +- src/ci.rs | 6 +- src/heartbeat.rs | 756 ------ src/lib.rs | 1 - src/redlist.rs | 7 - src/sourcegate.rs | 1 - src/testargs.rs | 49 +- src/tiers.rs | 6 +- tests/audio-baseline.toml | 495 ---- tests/common/audio.rs | 1366 ++--------- tests/common/clock.rs | 4 +- tests/common/console.rs | 12 +- tests/common/hda.rs | 332 --- tests/common/hostload.rs | 180 -- tests/common/inspect.rs | 37 +- tests/common/lane.rs | 26 +- tests/common/mod.rs | 8 - tests/common/origin.rs | 26 +- tests/common/passcost.rs | 536 ----- tests/common/qemu.rs | 143 +- tests/common/stats.rs | 120 - tests/common/usb.rs | 88 +- tests/logstallcase/system.toml | 10 +- tests/metal-profile.toml | 126 + tests/test-durations | 13 - .../src/bin/audio_idle_suspend.rs | 5 +- tests/toyos-rust-tests/src/bin/audio_tone.rs | 4 +- .../src/bin/audio_tone_load.rs | 47 - .../src/bin/blocking_read_stress.rs | 54 +- tests/toyos-rust-tests/src/bin/doom_music.rs | 3 +- .../src/bin/exit_wait_storm.rs | 58 +- .../toyos-rust-tests/src/bin/inspect_plays.rs | 52 - .../src/bin/netd_lookup_let_go.rs | 3 +- .../src/bin/netd_stalled_peer.rs | 85 - .../src/bin/panic_halts_first.rs | 48 - .../src/bin/poll_wake_pipe.rs | 28 +- .../src/bin/soundd_log_stall.rs | 7 +- .../src/bin/tlb_shootdown_waits.rs | 16 +- tests/toyos-rust-tests/src/tone.rs | 5 +- tests/toyos.rs | 2038 ++--------------- toyos-hda/src/config.rs | 3 +- toyos-mixer/Cargo.toml | 4 +- toyos-mixer/src/lib.rs | 6 +- toyos-mixer/src/shape.rs | 3 +- toyos-mixer/src/stats.rs | 8 +- toyos-sched/sim/src/scenarios.rs | 11 +- toyos-sched/sim/tests/scenarios.rs | 12 +- toyos-sched/src/cpu.rs | 15 +- userland/logd/src/policy.rs | 3 +- userland/logd/src/store.rs | 4 +- userland/soundd/src/backend.rs | 19 +- 71 files changed, 712 insertions(+), 7205 deletions(-) delete mode 100644 issues/audio/gate-a-first-run-to-record-its-host.md delete mode 100644 issues/audio/gate-a-has-no-runner-baseline.md delete mode 100644 issues/audio/hda-tone-red-beyond-its-exemption.md delete mode 100644 issues/audio/t14-wake-lateness-is-bimodal-per-boot.md delete mode 100644 issues/audio/thorough-tier-reds-on-unmodified-main.md delete mode 100644 issues/build/blocking-read-window-completed-21-of-500-round-trips-beside-other-guests.md delete mode 100644 issues/build/blocking-read-window-completed-26-of-500-beside-other-guests.md delete mode 100644 issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md delete mode 100644 issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md create mode 100644 issues/build/timing-verdicts-ruled-off-qemu-have-no-metal-arm.md delete mode 100644 issues/build/wake-storm-cost-red-under-induced-host-load.md delete mode 100644 src/heartbeat.rs delete mode 100644 tests/audio-baseline.toml delete mode 100644 tests/common/hda.rs delete mode 100644 tests/common/hostload.rs delete mode 100644 tests/common/passcost.rs delete mode 100644 tests/common/stats.rs delete mode 100644 tests/toyos-rust-tests/src/bin/audio_tone_load.rs delete mode 100644 tests/toyos-rust-tests/src/bin/inspect_plays.rs delete mode 100644 tests/toyos-rust-tests/src/bin/netd_stalled_peer.rs delete mode 100644 tests/toyos-rust-tests/src/bin/panic_halts_first.rs diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index cba134ab63e..300f1da0cdd 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -175,26 +175,6 @@ jobs: key: guest-${{ github.run_id }} - *serial - # Gate A's thorough tier, `tests/audio-baseline.toml`: N boots per config, - # strictly one at a time, as two shards of the same configs. - audio: - needs: build - runs-on: ubuntu-24.04 - timeout-minutes: 180 - strategy: - fail-fast: false - matrix: - shard: [1, 2] - container: *kvm - env: - GH_TOKEN: ${{ github.token }} - steps: - - *deps - - *checkout - - *guest-cache - - run: cargo run -- --ci audio ${{ matrix.shard }}/2 - - *serial - # "Only Rust and QEMU", run rather than argued: `cargo run -- --build-only` # from a fresh machine. `sid` as it stands, image and archive both, and no # cache — a fresh machine is the premise. @@ -253,7 +233,7 @@ jobs: # One standing issue, found by title and commented on; a dispatch is somebody # watching the run, so only the schedule files. nightly-red: - needs: [host, build, guest, tcg, audio, portability-linux, portability-macos] + needs: [host, build, guest, tcg, portability-linux, portability-macos] if: ${{ !cancelled() && github.event_name == 'schedule' }} runs-on: ubuntu-latest permissions: diff --git a/Cargo.toml b/Cargo.toml index e71be2c3637..ccc645e6201 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -105,10 +105,9 @@ default-run = "toyos-build" fatfs = "0.3.6" fontdue = "0.9" gpt = "3.1.0" -# `statvfs` for `worktree::free_bytes`; `getloadavg` and the libproc pair -# (`proc_listallpids`/`proc_pidpath`) for gate A's host-conditions line; and the -# unprivileged ICMP datagram socket `icmp::echo` asks a metal boot over, which -# is the platform call that replaces a `ping` binary. +# `statvfs` for `worktree::free_bytes`, and the unprivileged ICMP datagram +# socket `icmp::echo` asks a metal boot over, which is the platform call that +# replaces a `ping` binary. libc = "0.2" toml = "0.8" uuid = { version = "1", features = ["v4"] } @@ -171,23 +170,13 @@ toyos-userbound = { path = "toyos-userbound" } # declaration. Pure and `no_std`, so it compiles for the host as it does for # the guest. toyos-i219 = { path = "toyos-i219" } -# The scheduler core, for the one type the check build publishes and the -# harness judges: `cpu::PassCostReport` is the wire form of the pass-cost -# distribution, and its `Display`/`parse` pair is what keeps the kernel that -# writes it and `tests/common/passcost.rs` that reads it on one format. -# -# `check` because that is where the type lives, and it lives there because -# `src/build.rs`'s artifact gate asks the *shipping* kernel to carry none of the -# check instruments' literals — a linker may keep a string constant no code -# reaches, so an unconditional `Display` puts the report's prefix in every -# image. This is a dev-dependency of a host binary and `kernel/` is excluded -# from this workspace with its own lockfile, so nothing here reaches the kernel's -# resolution of the same crate. -toyos-sched = { path = "toyos-sched", features = ["check"] } -# The root-hub port machine, for the three durations `xhci_slow_connect` derives -# its bounds from: `DEBOUNCE_NS`, `EMPTY_BUS_NS` and `SLOW_CONNECT_NS` are -# declared once there and `use`d by the driver, so the window the harness -# certifies and the one the driver holds cannot drift apart. Also the Bulk-Only +# The watch's staged window, so the harness judges `blocking_read_window` +# against the bound the scheduler core declares. +toyos-sched = { path = "toyos-sched" } +# The root-hub port machine, for the two durations `xhci_slow_connect` derives +# its floor from: `DEBOUNCE_NS` and `SLOW_CONNECT_NS` are declared once there +# and `use`d by the driver, so the window the harness certifies and the one the +# driver holds cannot drift apart. Also the Bulk-Only # phases, so the harness judging a wedge reads the word `toyos_xhci::bot::Phase` # declares instead of spelling it a second time. toyos-xhci = { path = "toyos-xhci" } diff --git a/issues/audio/doom-sound-flood-played-full-scale-once.md b/issues/audio/doom-sound-flood-played-full-scale-once.md index 5349b6284c4..a12852e48dd 100644 --- a/issues/audio/doom-sound-flood-played-full-scale-once.md +++ b/issues/audio/doom-sound-flood-played-full-scale-once.md @@ -1,5 +1,5 @@ --- -status: expected-red +status: open kind: defect opened: 2026-09-04 --- @@ -36,6 +36,4 @@ whether it is the mixer's sum overflowing or the analysis reading a wrapped value, and whether a listener would hear it. Nothing in the captured line distinguishes those, and the WAV that would is not kept by a CI job. -**Exit condition.** The full-scale sample's cause is fixed, and -`doom_sound_flood` green on the KVM `guest` shards where it went red. Owner: -orchestrator. +**Exit condition.** The full-scale sample's cause is fixed. Owner: orchestrator. diff --git a/issues/audio/gate-a-first-run-to-record-its-host.md b/issues/audio/gate-a-first-run-to-record-its-host.md deleted file mode 100644 index e7a3c40970b..00000000000 --- a/issues/audio/gate-a-first-run-to-record-its-host.md +++ /dev/null @@ -1,56 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-08-07 ---- - -# The first gate A run to record its host: four of six boots outside the recorded sample, two of them harm - -2026-08-07, tree `4a0a07f`, the run that verified the host-conditions -annotation itself (`cargo test -- audio_tone`, filtered). Every line below is -the harness's own, printed beside the counters it qualifies: - -``` -audio_tone smp=1 gaps 1 [3p] underruns 3/1137 drains 8 wake_lat 86862us (3.74 pl) host: load 49.8/22.7/15.0 qemu 1 toyos-build 4 - confirm gaps none underruns 0/1136 drains 0 wake_lat 16968us (0.73 pl) host: load 49.0/23.4/15.4 qemu 1 toyos-build 4 -audio_tone smp=8 gaps 1 [1p] underruns 1/1113 drains 0 wake_lat 28118us (1.21 pl) host: load 48.3/24.1/15.7 qemu 1 toyos-build 4 - confirm gaps none underruns 0/1111 drains 0 wake_lat 8434us (0.36 pl) host: load 46.5/24.1/15.8 qemu 1 toyos-build 3 -audio_tone_load smp=1 gaps none underruns 0/1132 drains 0 wake_lat 7174us (0.31 pl) host: load 41.2/23.7/15.7 qemu 1 toyos-build 3 -audio_tone_load smp=8 gaps none underruns 0/1126 drains 0 wake_lat 23083us (0.99 pl) host: load 37.9/23.6/15.8 qemu 1 toyos-build 3 -``` - -**The invocation passed**, and correctly: harm appeared on both `audio_tone` -configs and neither reproduced on its confirming boot, which is precisely what -the two-boot rule is for. What it passed *with* is the finding. - -- `audio_tone.smp1` at **86862us — 3.74 pipeline depths, 8.6x that config's - recorded worst (10090us), and past its 56000us ceiling.** The baseline file - records `ceiling_runs = 0` across all 120 runs of the 2026-07-31 sample; this - is the first breach since. It came with `drains 8` — the ceiling exactly — - three periods of silence on the wire and a 3-period gap in the capture. -- **Two of six boots passed a whole pipeline depth** (3.74 and 1.21), which the - baseline file states no run of its 120 reached. -- `audio_tone_load.smp8` at 23083us is 2.9x its recorded worst with no harm at - all — the "bad but real" shape the ceilings exist to admit. - -**And now the conditions are on the record rather than reconstructed.** 1-minute -load 37.9-49.8 on 14 cores with three to four other `toyos-build` processes and -no other guest, against the 4.2-6.1 the 2026-07-29 ceiling derivation recorded -per run — six to twelve times it. Under the owner's ruling of 2026-08-04 that is -**not** an excuse and not grounds to re-run it away: it is a defect of the -pipeline until something shows otherwise, and it is the same shape as the -load-stall family `issues/audio/thorough-tier-reds-on-unmodified-main.md` -records, its fast-tier and 142 ms sightings included. What is new is only that -the next investigation starts from a measured host state instead of a guess. - -Whoever takes it: the thorough tier is the instrument for the rate, and it now -prints `host conditions over N runs` so its own arm's conditions can be stated. -The recorded arm's cannot — see `tests/audio-baseline.toml`. - -## Promoted 2026-08-25 - -A measured ceiling breach (86862us, 8.6x the recorded worst, past the 56000us -ceiling, with three periods of silence behind it) on a passing invocation is -harm under the audio law. Owed to whoever runs gate A's thorough tier next: -read the `host conditions over N runs` line and decide whether the ceilings -need a host-load term. diff --git a/issues/audio/gate-a-has-no-runner-baseline.md b/issues/audio/gate-a-has-no-runner-baseline.md deleted file mode 100644 index 4abe837bfc8..00000000000 --- a/issues/audio/gate-a-has-no-runner-baseline.md +++ /dev/null @@ -1,135 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-08-10 ---- - -# Gate A's thorough tier on a runner compares against the dev host's sample, and needs its own - -`tests/audio-baseline.toml`'s recorded sample was taken on the dev host under -cross-arch TCG. The thorough tier compares a fresh sample against *that*, so -`gate-a.yml` on any KVM runner is comparing two instruments and calling the -difference a regression. The gate runs on two GitHub-hosted KVM shards, so that -is what every nightly does. - -## Settled 2026-08-21: it is the instrument, and the control says so - -The earlier version of this file inferred the cross-instrument gap from level -differences. It is now measured against a same-session control, which is what -the audio law requires before a harm verdict may be set aside. - -**The experiment.** Four interleaved 15-iteration blocks of -`cargo test --test toyos-build -- --audio-gate 15 --shard 1/1 --host-slots 0` -on the T14, in the CI image of the day, `--device=/dev/kvm`, -QEMU 11.1.0 — the CI invocation, with private checkouts and a private cache root -so the runner's own state was untouched. Order A,B,A,B; 30 iterations per arm. - -* arm A = `960b96e3`, **the tree the recorded sample was taken on** -* arm B = `53101d08`, `main` - -Every block was gated on the machine being idle first and carried a witness -sampled every 10 s: no CI job container was present for any of the 240 boots, -and the 1-minute load stayed in 0.2-1.74. - -**The verdict, by the gate's own Mann-Whitney at its own alpha (1e-3, z>3.0902):** - -| config | recorded | T14 arm A | T14 arm B | A vs B (n=30) | -|---|---|---|---|---| -| `audio_tone.smp1` | 8972 | 20314 | 19994 | z=1.49 — **no difference** | -| `audio_tone.smp8` | 9249 | 14069 | 4088 | z=5.37 — B faster | -| `audio_tone_load.smp1` | 5765 | 2764 | 4352 | z=4.12 — B slower | -| `audio_tone_load.smp8` | 6097 | 12676 | 3904 | z=5.48 — B faster | - -Medians of `max_wake_lat_us`, microseconds. - -**The negative control is the whole finding: arm A fails this baseline on the -T14 and arm B does not.** Both arm-A blocks red `audio_tone.smp1` against the -recorded sample — `median 8972 -> 20438 (z=5.03)` and `8972 -> 20144 (z=4.67)` -— and both arm-B blocks print `PASS`. The tree the sample was recorded on reds -its own sample on this host, *harder* than `main` does. A level difference that -reds the recording tree is not a regression in anything. - -So gate A run `32479089989`'s verdict — -`audio_tone.smp1 wake lateness: median 8972 -> 17186 (z=4.36)`, the first -readable exit this workflow ever produced — is adjudicated: **the instrument, -not the tree.** No re-run was used to reach that; a same-session interleaved -control was. - -**Harm was null on both arms**: dropouts 0/120 and 0/120, underruns 0 in all 240 config-runs, -drains all-zero but for a handful of single events, ceiling breaches 1/120 on -arm A (a 334604 us single wake on `audio_tone.smp8`, no dropout behind it) and -0/120 on arm B. - -## What is still owed, and it is more than pasting numbers - -**Nothing here licenses replacing the recorded sample with a T14 one.** The dev -host still runs the fast tier against it, and a KVM sample would be as wrong -there as the TCG sample is on the runner. What is needed is a baseline *per -host*, and two things block writing one: - -1. **The file has no host dimension and the loader has no way to pick one.** - `AudioBaseline` is `BTreeMap>` with - `deny_unknown_fields`, and `config_baseline` (`tests/toyos.rs`) selects on - `(name, smp)` and nothing else. A `[runner]` sample is therefore a schema - change plus a selection keyed on whether the accelerator is in use — a change - to how a high-risk gate decides, not an edit to a table. -2. **The T14 distribution is bimodal per boot, and the mode's probability moves - with the tree.** Recording 30 runs of it would freeze a mixture whose mixing - weight is the thing that varies. Measured and written up in - `issues/audio/t14-wake-lateness-is-bimodal-per-boot.md`, which is what has to - be understood before any T14 number is worth recording. - - **And the mixing weight does not move only with the tree.** Four hours after - the A/B, 296 config-runs on the same host — `main` and `53101d08` - interleaved, each carrying only the wake instrument — produced the fast mode - 296 times of 296, and both 15-iteration blocks reported `[gate A] PASS` - against *this very sample*, `audio_tone.smp1` included: fresh medians 4075 - and 4143 against the recorded 8972. A T14 sample recorded on one evening - therefore would not describe the same host on another, which is a stronger - objection than the schema and the one that has to be answered first. - -The 2026-08-10 measurement this file opened with — run `31386117376`, tree -`99e47d9`, two GitHub-hosted runners of different vendors — remains the reason a -hosted sample would be a sample over two unnamed CPUs. Its `wakes` observation -still holds and the T14 reproduces it: 1185-1432 fresh against 846-905 recorded, -a KVM guest waking about 1.6x as often as the same guest under cross-arch TCG. -That artifact expired on 2026-09-16; the T14 arrays above are in this branch's -commit message. - -## Promoted 2026-08-25 - -Real, actionable work remains even though harm was null on this measurement: a -per-host baseline needs a schema change (`AudioBaseline`'s host dimension, -`tests/toyos.rs`'s `config_baseline` selection) and the T14's bimodal mixing -weight has to be understood before a sample is worth recording. Owed to -whoever owns `tests/audio-baseline.toml` and `gate-a.yml`. - -## The T14 is no longer a runner - -The gate's two shards are GitHub-hosted, and the sample they compare against is -still the dev host's under cross-arch TCG, so this record is unchanged in what -it says and only narrower in where a fresh sample can come from: a hosted -sample is a sample over unnamed CPUs, and a T14 sample is now a job of -`issues/hardware/the-t14-boots-toyos-unattended.md` rather than of a CI lane. - -## Main's nightly at 1ce71831 reds on it - -Run 36290616312, `audio (2)`: `audio_tone_load.smp1 wake lateness: median -5765 -> 6650 (Mann-Whitney z=4.03 > 3.09)`. Dropouts were 0/60, underruns 0, -ceiling breaches 0/60, and wakes 1396-1426 against the recorded 856-887, -the 1.6x KVM-over-TCG ratio this file records. The same lane on the nightly -before #527 (run 36285169430) read median 6478 with wakes 1394-1425 and -passed. On `nightly-green2` at dbf4ace5 (run 36292135439) it passed too. The -fresh samples agree with each other and differ from the dev host's TCG -sample, so this is the instrument, as above. Until a per-host baseline -exists, whether a KVM runner's gate A reds depends on where its median lands -against the TCG sample's. - -The four `audio (2)` samples of `audio_tone_load.smp1` around #527 have -medians of 6453 (before #527, passed), 6648 (main at 1ce71831, red), 6190 -(`nightly-green2` at dbf4ace5, passed) and 6520 (`nightly-green2` at -c2715880, red, run 36297455432). Mann-Whitney of each later sample against -the one before #527, on the gate's 30-value arrays: z=1.40, -1.20 and 0.84. -None is a difference at the gate's alpha. The runner's sample did not move. -The gate's verdict flips because that sample's median sits about 0.8 ms above -the TCG sample's 5765, right at the gate's edge. diff --git a/issues/audio/gate-a-suspend-structure-verdict-unread.md b/issues/audio/gate-a-suspend-structure-verdict-unread.md index b5337723c03..1c7c4024fe3 100644 --- a/issues/audio/gate-a-suspend-structure-verdict-unread.md +++ b/issues/audio/gate-a-suspend-structure-verdict-unread.md @@ -17,57 +17,13 @@ structure: no `virtio-sound: stream 0 stopped` after the last client removal — the device is still running with no clients ``` -Shard 2 of the same run failed on a statistic: - -``` -[gate A] FAILED — 1 statistic(s) regressed: - audio_tone_load.smp1 wake lateness: median 5765 -> 17684 (Mann-Whitney z=4.61 > 3.09) -``` - -**Neither was ever adjudicated, because the exit code could not tell anyone they -had happened.** That workflow's step ended in `exit "${PIPESTATUS[0]}"` under a -shell with no such array, so it reported `failure` on every run whatever the -audio said; 08-17's two FAILEDs and the PASSes on 08-16 and 08-18 arrived as the -same red. The mechanism and the full run-by-run table are in -`issues/audio/thorough-tier-reds-on-unmodified-main.md`; the exit code is fixed, -these two verdicts are not. - -**Why this one is not the cross-instrument shape.** Three of the five FAILEDs -that workflow ever printed are the runner-vs-dev-host wake-lateness level -difference `gate-a-has-no-runner-baseline` explains, and they are all from before -the 2026-08-15 re-record. These two are after it, and they sit between a -two-shard PASS night (08-16) and a two-shard PASS night (08-18) on the same -recorded sample. A level difference does not come and go for one night. - -**Shard 2's half of that is now adjudicated, and the paragraph above was half -right.** The 2026-08-21 same-session A/B on the T14 -(`issues/audio/gate-a-has-no-runner-baseline.md`) shows the level difference is -real *and* that it comes and goes: on a KVM host `max_wake_lat_us` is bimodal -per boot, ~4 ms or ~20 ms, and which mode an `smp=1` config draws moves between -runs of the same tree — see -`issues/audio/t14-wake-lateness-is-bimodal-per-boot.md`. A night that draws the -slow mode reds and a night that draws the fast one passes, on one unmodified -tree, which is exactly the 08-16 PASS / 08-17 FAILED / 08-18 PASS sequence. -`audio_tone_load.smp1` is also the one config where `main` sits worse than the -baseline tree on the T14 (z=4.12 pooled, but z=3.57 and z=2.09 across the two -interleaved block pairs), so shard 2's `5765 -> 17684` is that same unstable -config and needs no separate hunt. - -What that does **not** settle is a hosted baseline: the 08-17 run was a -GitHub-hosted runner of an unnamed vendor, and the T14 control speaks for the -T14. A hosted verdict still has no hosted sample to be read against. - -**Shard 1's is the one to take first.** `soundd: suspended` and +`soundd: suspended` and `virtio-sound: stream 0 stopped` are the two lines that say the idle path released the device; their absence says a boot left the device running with no clients, which is the subject of `stop-the-device-voice-keep-the-wake` and `idle-suspend-reds-on-a-loaded-host-and-on-main` from the other side. One -occurrence in 25 iterations is a rate nobody has, and the tier is the instrument -that produces one. +occurrence in 25 iterations is a rate nobody has. **The evidence expires.** `/tmp/gate-a.log` is uploaded per shard with `retention-days: 30`, so run `31992902784`'s artifacts go on 2026-09-16; the job logs outlive them. Everything quoted above is already here for that reason. - -Whoever takes it: `gh workflow run gate-a.yml -f iterations=30` now reports its -own verdict, so a re-dispatch is a readable experiment for the first time. diff --git a/issues/audio/hda-tone-phase-check.md b/issues/audio/hda-tone-phase-check.md index 3daec584bb9..25ca8ab47aa 100644 --- a/issues/audio/hda-tone-phase-check.md +++ b/issues/audio/hda-tone-phase-check.md @@ -1,5 +1,5 @@ --- -status: expected-red +status: open kind: defect opened: 2026-08-07 task: 88 @@ -101,6 +101,5 @@ adjacent-frame *pairs* with |period| in the hundreds, not the 118-frame clusters. And the load dependence is sharp where it used to be a correlation: 0 of 8 alone against 3 of 11 beside other guests, same tree, same hour. -**Exit condition.** The adjacent-frame-pair breaks' cause is fixed, and -`hda_tone` reads 0 phase breaks beside other guests on the dev host and on -CI's KVM shards. Owner: orchestrator. +**Exit condition.** The adjacent-frame-pair breaks' cause is fixed. Owner: +orchestrator. diff --git a/issues/audio/hda-tone-red-beyond-its-exemption.md b/issues/audio/hda-tone-red-beyond-its-exemption.md deleted file mode 100644 index 5cd53a280d9..00000000000 --- a/issues/audio/hda-tone-red-beyond-its-exemption.md +++ /dev/null @@ -1,40 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-08-07 -task: 88 ---- - -# `hda_tone` is red on `main` for a reason `#88`'s exemption does not cover - -`cargo test -- hda_tone` on `main` at `6d11938`, alone, 2026-08-07 18:5x: - -``` -FAIL hda_tone: 1 mid-tone silences in the capture: total 1 [1p×1] - FAIL hda_tone (15s) — listed against #88, and this is not that failure: - the entry covers ["the captured tone is not one sine"] -``` - -**What has changed since, and what has not.** `hda_tone` is `Tier::Nightly` for -`Why::TimerAnchored` (`src/tiers.rs`), so a plain `cargo test` no longer runs -it and a landing whose gate is `cargo test` no longer meets this red at all. - -That doesn't touch the verdict, and the verdict was owed a fresh sample: every -capture behind it had gone through QEMU's 48000→44100 resampler, since removed. - -**Re-judged 2026-08-29, and it stands.** On the current instrument (QEMU -11.1.0), `main` at `48437ca4`: alone, 8 of 8 boots clean (`gaps none`); beside -two other suites of this worktree (1-min load 5.1-21.6), **1 of 11 boots put one -mid-tone period of silence in the capture** — `1 mid-tone silences in the -capture: total 1 [1p×1]`, byte-identical to the line above, with 4 phase breaks -beside it and a confirming re-boot that came back 8 breaks and no gap. So the -red is real on the fixed pipeline, load-keyed, and roughly the shape of CI's 4 -of 5 — the re-judging that was tracked apart closed with this measurement, and -the load-stall family (`issues/audio/thorough-tier-reds-on-unmodified-main.md`) -is the standing suspect for the mechanism. - -Found while landing task #98/#12: the same test failed identically inside that -landing's gate, and the A/B against `main` in the same session is what -identified it as `main`'s. Assigning it needs whoever owns H3 — -`5fdfeb7`/`a022811` ("wip: H3, the virtio-sound stub and its userland driver") -landed hours before this measurement. diff --git a/issues/audio/stop-the-device-voice-keep-the-wake.md b/issues/audio/stop-the-device-voice-keep-the-wake.md index 0c0b296bba1..bae952f45f2 100644 --- a/issues/audio/stop-the-device-voice-keep-the-wake.md +++ b/issues/audio/stop-the-device-voice-keep-the-wake.md @@ -11,18 +11,3 @@ engine and the codec — the battery-relevant hardware — and gives up only the itself. Resume still works unchanged, because soundd keeps writing signal bytes, so it does not need the missing client→soundd message that `cpal-backend-hardcodes-the-format` is waiting on. - -**It is blocked on the audio gate, not on the fork and not on the owner.** A -mid-session device stop/restart is an audible transient plus a DLL re-lock, which -needs gate A's thorough tier on a quiet tree. That tier reds on the dev host -(`thorough-tier-reds-on-unmodified-main`), so the instrument comes first. - -2026-08-21: the sentence above used to read "that tier is itself red on `main`", -which was being sourced partly from the CI nightly — and the nightly's red was an -exit-code defect over a printed verdict, not a verdict. The dev-host red it now -names is the one that was ever measured, and the block stands on it. The same -entry records why the runner's PASSes do not lift it. - -That unblock condition is the useful part and the reason this is filed apart from -the fork-blocked cluster: it could land *first* if the quiet tree arrives before -fork access does. diff --git a/issues/audio/t14-wake-lateness-is-bimodal-per-boot.md b/issues/audio/t14-wake-lateness-is-bimodal-per-boot.md deleted file mode 100644 index 1a096837d67..00000000000 --- a/issues/audio/t14-wake-lateness-is-bimodal-per-boot.md +++ /dev/null @@ -1,188 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-08-21 ---- - -# On the T14 soundd's worst wake is bimodal per boot — ~4 ms or ~20 ms — and which mode a config lands in moves with the tree - -Measured on the self-hosted T14 (Intel i5-1135G7, 4c/8t, KVM, QEMU 11.1.0, CI -image) during the A/B that settled -`issues/audio/gate-a-has-no-runner-baseline.md`: four interleaved 15-iteration -gate A blocks, A,B,A,B, arm A `960b96e3`, arm B `53101d08`, idle machine, no CI -job container for any of the 240 boots, 1-min load 0.2-1.74. - -`max_wake_lat_us` does not vary continuously on this host. It takes one of two -values per boot: - -* a **fast mode** at 2544-4352 us (0.11-0.19 pipeline depths), and -* a **slow mode** at roughly 10000-25000 us (0.43-1.08 pl), - -with nothing much in between, and the mode is drawn per boot rather than -drifting through a block. `audio_tone.smp1`, arm B, in run order — the fast runs -are scattered, not clustered at either end: - -``` -b1 21389 18528 3945 4183 19994 20483 21544 33934 21591 21852 21139 22114 3871 16861 21481 -b2 4010 21562 4085 4028 4028 4212 14811 21692 17017 19528 22652 20878 18474 24184 11543 -``` - -Arm A, same config, has essentially no fast runs at all (one of 30, at 9880), -and arm B has eight of 30 — yet the two arms are **indistinguishable** on this -config overall (medians 20314 and 19994, z=1.49), because the slow mode -dominates both. - -Where the arms *do* differ, they differ by which mode dominates: - -| config | arm A, n=30 | arm B, n=30 | A vs B | -|---|---|---|---| -| `audio_tone.smp1` | 20314, one fast run | 19994, eight fast runs | z=1.49, same | -| `audio_tone.smp8` | 14069, mixed | 4088, **all 30 fast** (3939-4240) | z=5.37, B faster | -| `audio_tone_load.smp1` | 2764, **all 30 fast** (2544-3259) | 4352, mixed | z=4.12, B slower | -| `audio_tone_load.smp8` | 12676, mixed | 3904, **all 30 fast** (3643-4241) | z=5.48, B faster | - -## Why this matters more than the direction of any one row - -**The slow mode has no margin.** 20 ms is 0.86 of the 23219 us pipeline depth — -the point at which every buffer has drained and the device has run out of audio. -The recorded dev-host sample reaches 0.98 pl once in 120 runs and sits at 0.39 pl -in the median; on the T14 the *median* `audio_tone.smp1` boot, on both arms, is -where the dev host's worst run was. Nothing was audible in any of the 240 boots -— dropouts 0/120 and 0/120, underruns 0 in all 240 config-runs — but the -distance to harm on a slow-mode boot is one scheduling accident. - -**And a bimodal statistic is a bad thing to baseline.** The thorough tier's -Mann-Whitney is comparing mixtures, so its verdict tracks the mixing weight -rather than either mode. That is why no T14 sample should be recorded into -`tests/audio-baseline.toml` until the mode is understood: a re-record would -freeze one afternoon's mixing weight and red on any tree that moved it. - -## The one row that is a real same-host difference, and why it is not called a regression here - -`audio_tone_load.smp1` is worse on `main` than on `960b96e3` at z=4.12 pooled — -and it is the same config the never-read 2026-08-17 hosted nightly failed on -(`median 5765 -> 17684, z=4.61`, quoted in -`issues/audio/gate-a-suspend-structure-verdict-unread.md`). It is not called a -bisected regression because the block structure says it is not stable: the -per-block figures are z=3.57 (a1 vs b1) and z=2.09 (a2 vs b2), and arm B's own -two blocks differ by z=2.51 against arm A's z=0.35. Arm B is *unstable* on this -config; arm A is not. That is a change in the mixing weight, not a level shift, -and bisecting a mixing weight at n=15 per point would measure noise. - -The arrays behind every number above are in `0942f02c`'s commit message; the -gate's own logs expire with the workflow artifacts. - -## 2026-08-21, four hours later: the number is the *device's* lateness, and the slow mode did not come back - -`max_wake_lat_us` now arrives in two halves (`toyos_mixer::WorstWake`), split at -the completion interrupt's own ISR timestamp: `irq` is the device failing to -complete when the grid said it would, `pickup` is soundd failing to run once it -had. They sum to the old number exactly. `late_wakes` counts how many wakes in -the run were a whole period or more late, so the maximum can be read as one -stall or as a thousand. - -**296 config-runs on the T14 the same evening, 17:26-19:00 UTC, in the CI image -of the day, `--device=/dev/kvm --shard 1/1 --host-slots 0`, no other -container up for any boot** — each block samples `docker ps` every five seconds -and discards and retries itself whole if a CI job appears, which happened three -times and cost three blocks. Two trees, each carrying only the instrument: -`fe41dbae` (`main`, 51 runs per config) and `53101d08` (23 per config) — *the -A/B's own arm B*, the tree that produced the arrays above. - -| config | tree | n | wake_lat | irq mean | pickup mean / max | late wakes | -|---|---|---|---|---|---|---| -| `audio_tone.smp1` | main | 51 | 3875-4299 | 4005 | 76 / **113** | 12.9% | -| | `53101d08` | 23 | 3835-4197 | 3969 | 75 / **121** | 12.8% | -| `audio_tone.smp8` | main | 51 | 3977-6176 | 4025 | 139 / **206** | 14.2% | -| | `53101d08` | 23 | 3982-4341 | 3985 | 144 / **188** | 14.2% | -| `audio_tone_load.smp1` | main | 51 | 2509-2986 | 2732 | 10 / **14** | **0.0%** | -| | `53101d08` | 23 | 2542-2981 | 2737 | 10 / **12** | **0.0%** | -| `audio_tone_load.smp8` | main | 51 | 3692-4086 | 3812 | 64 / **142** | 11.1% | -| | `53101d08` | 23 | 3728-4331 | 3881 | 59 / **159** | 11.3% | - -Dropouts 0/296, underruns 0/296, drains 0/296, ceiling breaches 0/296. The two -15-iteration blocks that ran the thorough tier both reported **`[gate A] PASS` -— no statistic regressed at alpha=1e-3 per test**, on both trees, against the -same recorded sample the morning's first readable T14 run failed at -`audio_tone.smp1 median 8972 -> 17186 (z=4.36)`. Its fresh medians this evening -were 4075 (`53101d08`) and 4143 (`main`). - -Two things fall out of that table and a third out of its absence. - -**The statistic is not about the scheduler.** `pickup` never once exceeded -206 µs — 0.009 pipeline depths — on one CPU or on eight, and `irq` is 94-99.6% -of every worst wake. So *which CPU soundd lands on cannot be the mechanism*: -every interrupt lands on the boot CPU (`kernel/src/drivers/pci.rs`'s `MSG_ADDR`) -and a soundd sharing that CPU or not moves a term that is two orders of -magnitude too small to matter. The instrument is not blind to the other half — -the dev host under load 30 reported `pickup 8681us` on `audio_tone_load.smp8` -the same afternoon — the T14 simply never spends it. - -**The fast mode is a beat, not an event.** 12-14% of wakes are a whole period -late on the three idle configs, and the worst is ~1.4 periods with `2 empty -wakes` and `batch 2` on essentially every boot. That is soundd's 2.902 ms grid -against QEMU's `timer-period=5000` audio timer: soundd arms, wakes punctually -twice at a device that has produced nothing, and the batch lands ~4 ms after the -grid point. `audio_tone_load.smp1` — the config that is *always* fast — has -**zero** late wakes and `pickup 10 µs`, because a guest with work to do never -lets the beat open. - -**And the slow mode did not appear once in 74 boots per config.** Not on `main` -and not on the tree that produced it at 11 of 15 and 9 of 15 four hours earlier. -Under that afternoon's mixing weight, 0 of 74 has probability ~1e-35. So the -mode is **not a per-boot draw from a per-tree distribution**: the distribution -itself moved between two sessions of the same day, on the same host, with -nothing about the tree between them — which also means the A/B's one same-host -row (`audio_tone_load.smp1`, z=4.12) is a difference between afternoons and not -between trees. This evening the two trees are indistinguishable on it: 2509-2986 -against 2542-2981. - -The host was on AC throughout, `intel_pstate`/`powersave`/`balance_performance`, -`intel_idle` whose deepest state (`C3_ACPI`) costs 1048 µs to leave — a third of -one period, and a fortieth of the 20 ms mode. Nothing on the host was measured -*during* the earlier session, so what moved is not established; the one -difference recorded is that the slow session ran at 1-min load 0.2-1.74 and the -fast one at 1.1-4.4 — overlapping, and with the fast blocks' own quietest -stretches at 1.1-1.6, so load does not separate them either. - -## Whoever takes it next - -Do not spend the day on soundd. The next sighting of the slow mode is now one -line, and that line already answers three questions: whether it is the device or -soundd (`irq` vs `pickup`), whether it is one stall or a thousand -(`late_wakes`), and whether the guest was executing at all while it happened -(`empty` — a punctual soundd waking repeatedly at a silent device, versus a -single overlong sleep). Gate A's per-boot line also carries the two numbers this -boot drew for its clocks, which are the only per-boot draws that scale every -armed timer for the boot's whole life; on the T14 they are stable to 0.02% -(`tsc 2418-2419MHz`, `lapic 10000742-10002460 ticks/10ms` over 120 boots) and on -the dev host to 0.2%, so a slow-mode boot whose pair sits outside that is a -finding on sight and one whose pair sits inside it removes the whole class. - -What is missing is a *host-side* record taken during a slow session, because the -guest-side evidence above points there and cannot go further on its own. The -cheapest one is per-boot `/proc//task/*/schedstat` and the host's -`cpuidle` residencies sampled across a block — and it perturbs the measurement, -so it is worth taking only once a session is producing the mode. - -## Promoted 2026-08-25 - -The slow mode sits within one scheduling accident of harm (0.86 of a pipeline -depth, no margin) and its mixing weight is unexplained across two sessions on -the same host and tree. Owed to whoever next sees the slow mode, per this -entry's own "whoever takes it next" section: the host-side schedstat/cpuidle -capture during a slow session. - -## The exit is not the metal loop, and no lane produces the mode any more - -The T14 is no longer a CI runner, so nothing schedules a gate A run on it: the -slow mode can only be sighted by somebody running the gate there by hand, and -this record waits for that sighting rather than for a nightly. - -The metal loop (`issues/hardware/the-t14-boots-toyos-unattended.md`) does not -take the capture this record is owed. That capture reads a Linux host's -`schedstat` and `cpuidle` for the QEMU process the guest runs in, and the loop -boots ToyOS on the bare machine with no host under it and no QEMU process to -read. What the loop can answer is the adjacent question this record makes worth -asking — whether the two modes survive with no hypervisor at all — and that is -a job of the loop, not this exit. diff --git a/issues/audio/thorough-tier-reds-on-unmodified-main.md b/issues/audio/thorough-tier-reds-on-unmodified-main.md deleted file mode 100644 index 8d8b1803b83..00000000000 --- a/issues/audio/thorough-tier-reds-on-unmodified-main.md +++ /dev/null @@ -1,187 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-08-07 ---- - -# Gate A's *thorough* tier reds on an unmodified `main`, and that is the rate the fast-tier intermittents asked for - -The fast tier's two-boot rule failed intermittently through early August — -dropouts on the first boot *and* the confirming re-boot, four times at smp=1 in -one 2026-08-04 session on two trees at once, twice more at smp=8 on 2026-08-07 — -and those sightings asked for the rate first, naming the thorough tier -(`--audio-gate N`) as the instrument. (Closed into this entry 2026-08-29; their -run tables are in that closing commit.) H3's session got the rate, and the -instrument reds on the tree it is supposed to certify. - -`cargo test --test toyos-build -- --audio-gate 30` on `80fe031` — **main's tip, -no delta at all**, run as H3's A arm before that branch existed: - -``` -[gate A] FAILED after 15 of 30 iterations (the remaining runs cannot change this): - pooled dropout rate: 10 of 120 vs recorded 0 of 120 (Fisher p=8.03e-4 <= 1e-3) -``` - -The ten, by config and iteration: `audio_tone_load smp=1` at 4, 9, 13, 15; -`audio_tone_load smp=8` at 9, 13, 14; `audio_tone smp=8` at 8, 9; -`audio_tone smp=1` at 13. So **`audio_tone` at both widths reds too**, which -the fast-tier sightings had established only for `audio_tone_load` — and at -both of its widths there, so no config is anyone's quiet control in an A/B. - -**The load correlation is the wrong way round, and that is the finding.** The -1-minute average across the run spanned 7.2 to 19.1 on 14 cores, with one to -five other guests and six other `toyos-build` processes throughout. The clean -early iterations ran at 19.1 and 16.8; the three worst — 13, 14 and 15 — ran at -11.4, 10.6 and 11.9. Every dropout carried a wake latency of 33-117 ms against -5-17 ms on the clean runs, which is the same "soundd was not scheduled" -signature as the fast-tier sightings and as the 2026-08-03 boot that put 142 ms -of silence on the wire — 49 underruns, one drain, a 5.6x worst-wake outlier, -gaps, soundd stats and capture all agreeing (also closed into this entry -2026-08-29). That boot's nearest suspect, the ESP-log flush on the kernel's -idle path, no longer exists — the idle loop touches no filesystem — and the -nearest *measured* mechanism on file is `issues/audio/disk-wait-pins-a-cpu.md`: -a staged 2 ms disk-completion delay alone produced 165-260 ms soundd wakes and -76 silent periods. - -What this changes for anyone reading them: the intermittency is not a property -of one config, and it is **large enough to fail the thorough tier's own pooled -test on a clean tree**. Anything that gates on this tier — the nightly run, and -H3 itself — cannot presently tell its own change from this. H3 therefore -compared its two arms against *each other* rather than against the recorded -sample, and said so. - -**Three arms, all of them red, and `main` red hardest.** Measured 2026-08-07 on -`main` at `c0365ea`, one session: - -| tree | dropout runs | measured runs | verdict | -|---|---|---|---| -| `main` | 7 | 28 | `pooled dropout rate: 7 of 40 vs recorded 0 of 120 (Fisher p=4.00e-5)` | -| `wt/toyos-m3` | 5 | 12 | `5 of 40 … (p=8.02e-4)` | -| the same branch with its one new wait deleted | 5 | 40 | `5 of 40 … (p=8.02e-4)` | - -The denominators differ because the gate stops as soon as the remaining runs -cannot change the verdict. The gate's own documentation says it cannot detect a -doubling of the dropout rate at any N a human waits for, so it cannot separate -these three from each other either — what it says unambiguously is that every one -of them is far from the recorded `0 of 120`. Every gap is small and none is a -silence anyone would hear as a break: the largest is 51 periods, most are one or -two, and the fast tier — whose verdict is harm — is green on all three arms, 7 of -7 each. So this is a *rate* finding against a recorded sample, not a report that -the machine sounds wrong. Host load was 6-20 throughout and is not offered as the -explanation; the three arms ran back to back on the same host, which is what makes -them comparable to each other. **Consequence while this is open:** the thorough -tier cannot serve as a pass/fail gate — an A/B between two arms is what it can -still answer. - -**The next step is named and is still nobody's: the thorough tier on the commit -the sample was recorded against, on the dev host, in one session.** That is the -one testable half of the two readings below. Do not re-record the baseline first: -a sample re-taken now would make the disagreement disappear without anyone -learning which of the two it was, and the recorded zero is the only reason the -question is visible at all. - -The recorded sample in `tests/audio-baseline.toml` is 0/120 and was taken in a -session this host no longer resembles. **Re-recording it is not licensed by this -entry** — a baseline widened to accept the defect is the defect made permanent. -What is needed is the cause. - -**The B arm was never obtainable, and the reason is `issues/kernel/`'s shootdown deadlock.** -Two attempts on the audio branch stopped at iterations 2 and 4, both on -`audio_tone.smp8`, both with the tier's "instrument broken" verdict — which is -what a guest whose kernel double-panicked looks like from here. Those commits -landed between the two arms and `--land` merged them in, so the arms differ by -more than the change under test and no comparison between them means anything. -What H3 has instead: a full suite green at 289/289 with all four audio configs -clean, and ten standalone runs of the audio family. None of that is a rate. - -## 2026-08-21: the CI nightly's red was never this, and its verdicts were never read - -**The dev-host finding above stands unchanged.** What follows is about a -different instrument — `gate-a.yml` on a GitHub runner — and it must not be -conflated with it. Nobody re-ran the dev-host arm; nothing here re-runs an audio -verdict away. - -`gate-a.yml`'s `gate` step ended in `exit "${PIPESTATUS[0]}"`. The runner logs -the shell it picked for that container on every step — `shell: sh -e {0}` — and -that shell has no `PIPESTATUS` array. Every run answered - -``` -/__w/_temp/.sh: 4: Bad substitution -##[error]Process completed with exit code 2. -``` - -on the line *after* the gate had printed its verdict. Without `pipefail` the -pipeline's status was `tee`'s 0, so `-e` never fired on the harness's own code; -the step's exit was dash's 2 for a failed expansion, and it was 2 whatever the -audio said. `gate-a.yml` has therefore **never once reported its verdict**: -every run it has ever had is a `failure`, including the ones that passed. - -The verdict each shard actually printed, read out of the job logs (artifacts -expire at 30 days; these lines do not): - -| run | date | shard 1 | shard 2 | -|---|---|---|---| -| 31386117376 | 08-10 (dispatch, `wt/toyos-ciwave2`) | FAILED `audio_tone.smp8` wake lateness median 6658 → 8496 (z=4.27) | FAILED `audio_tone_load.smp8` wake lateness median 7134 → 9520 (z=6.05) | -| 31771577360 | 08-14 | FAILED `audio_tone.smp8` wake lateness median 6658 → 8673 (z=5.41) | PASS | -| 31862912891 | 08-15 | PASS | PASS | -| 31925451196 | 08-16 | PASS | PASS | -| 31992902784 | 08-17 | FAILED at iteration 25, `audio_tone.smp8` instrument broken — suspend structure | FAILED `audio_tone_load.smp1` wake lateness median 5765 → 17684 (z=4.61) | -| 32097206141 | 08-18 | PASS | PASS | -| 32213928799 | 08-19 | PASS | PASS | -| 32330040225 | 08-20 | PASS | PASS | -| 32445243829 | 08-21 | PASS | PASS | - -Thirteen shard-runs PASS, five FAILED, eighteen exits of 2. - -**Two consequences, and they point opposite ways.** - -The thirteen PASSes mean the standing sentence "the thorough tier reds on -`main`" was being sourced from a red that was not a verdict. It is not evidence -that the dev-host finding above is stale: `gate-a-has-no-runner-baseline` already -establishes that a runner arm compared against the dev host's sample is a -cross-instrument comparison, and since the 2026-08-15 re-record the runner's -numbers are one-sidedly *better* than the recorded ones (08-21 shard 1: -`wake_lat_us recorded 7052/8972/22744 fresh 4002/6034/9942`), which is a -comparison that cannot red. A PASS of that comparison certifies little. The -dev-host question is still open and still needs the dev host. - -The five FAILEDs are the harm. Two on 08-10 and one on 08-14 are the -cross-instrument shape `gate-a-has-no-runner-baseline` explains. **The two on -08-17 are not**, and nobody has looked at them: they sit between a PASS night and -a PASS night on the same baseline, and shard 1's is not a statistic at all but -the instrument refusing — -`no 'soundd: suspended' after the last client removal; no 'virtio-sound: stream 0 -stopped' after the last client removal — the device is still running with no -clients`. That is filed apart as `gate-a-suspend-structure-verdict-unread`. - -The exit code is fixed in `.github/workflows/gate-a.yml` (`set -o pipefail`, the -idiom every other workflow in `.github/` already uses). Nothing about how a -verdict is *reached* changed. - -## 2026-09-03: on a quiet dev host the tier passes, on both arms of an A/B - -`cargo test --test toyos-build -- --audio-gate=10`, six blocks alternating the -IOMMU audio move and its base (0c805ecf), base first; the arms differ in -`iommu_platform=on` on `virtio-sound-pci` and the sound driver's -`DeviceSpace::create` plus `attach`. 1-minute load 1.3-5.1 across the 240 runs -(block medians 2.4, 2.7, 2.3, 1.8, 2.7, 3.2, base and moved alternating) and -`qemu 1 toyos-build 1` on every one: every block `PASS` against the -recorded sample, dropouts 0 of 120 on each arm, ceiling breaches 0 of 120, no -instrument-broken iteration. The finding above was taken at loads of 7-19 -beside other guests and stands as measured; this is the quiet-host reading it -said nobody had taken, on that tip rather than on the sample's commit, so the -question of the sample's commit is still open. - -## 2026-09-25: an A/B for #492, both arms dropping, main no less often - -`cargo test --test toyos-build -- --audio-gate 30 audio_tone_load`, three blocks -per arm, alternating #492's merge of `origin/main` 3ad278c0 with a checkout of -3ad278c0 itself, branch first; one session, 1-minute load 1.5-34.2. Branch: -dropouts 1/60, 0/60, 0/60, every block `PASS`. `main`: 0/60, 1/60, 1/60, every -block `FAILED` on `audio_tone_load.smp1` wake lateness alone (medians 6042, -6803, 6473 against the recorded 5765; the branch's were 6147, 6137, 6121). The -branch's dropout (1 period) has its lateness in the interrupt half (`irq 65153us -+ pickup 874us`); main's are 28 periods in the pickup half (`irq 2973us + pickup -27401us`) and 1 period in the interrupt half (`irq 56120us + pickup 265us`). The fast tier, -three runs each: the branch's first run dropped once at load 57 and did not -reproduce on its confirming boot; every other boot on both arms was clean. diff --git a/issues/build/blocking-read-window-completed-21-of-500-round-trips-beside-other-guests.md b/issues/build/blocking-read-window-completed-21-of-500-round-trips-beside-other-guests.md deleted file mode 100644 index 3beb62541eb..00000000000 --- a/issues/build/blocking-read-window-completed-21-of-500-round-trips-beside-other-guests.md +++ /dev/null @@ -1,22 +0,0 @@ ---- -status: open -kind: finding -opened: 2026-09-26 ---- - -# `blocking_read_window` completed 21 of 500 round trips beside other guests - -Fast tier at `62ef89a4` (PR #525's branch, other worktrees' guests running): -`blocking_read_stress: only 21 of 500 round trips completed inside 3s — a wake -was not delivered`; the process's own line gave `syscall_wall=3087ms` and -`cpu=1703ms` for pid 7, and cpu0 took 169 xHCI interrupts in the window. The -harness's re-run alone was green. `cargo run -- --known-red` answers NO. - -In the same session the fast tier ran six times, three on the branch and three -on `main` at `d65446cc`, interleaved: this test was red once on the branch and -never on `main`, while `main`'s third run had five reds of its own that the -branch never showed. The branch's kernel differs from `main` in the claim -DMA paths, which this test does not reach, and in one boot log line. - -**Exit**: a cause — a lost wake, or a 3 s budget a starved host cannot meet — -and, if it is the budget, the bound derived rather than measured. diff --git a/issues/build/blocking-read-window-completed-26-of-500-beside-other-guests.md b/issues/build/blocking-read-window-completed-26-of-500-beside-other-guests.md deleted file mode 100644 index 45649d3551b..00000000000 --- a/issues/build/blocking-read-window-completed-26-of-500-beside-other-guests.md +++ /dev/null @@ -1,32 +0,0 @@ ---- -status: open -kind: finding -opened: 2026-09-26 ---- - -# `blocking_read_window` completed 26 of 500 round trips beside other guests - -Fast tier at `1e5ef5c1` (PR #524's branch; the host carried the branch's own -twelve guest slots and another worktree's suite at the same time): -`blocking_read_stress: only 26 of 500 round trips completed inside 3s — a wake -was not delivered`. The harness's re-run alone was green in 2 s -(`at least 193 held windows a post landed in (0 -> 256)`). `cargo run -- ---known-red blocking_read_window` answers NO. - -What the red run's own log says against its sentence: the two processes spent -`cpu=1639ms` and `cpu=1998ms` of the 3.5 s they ran (`syscall_wall=3535ms` and -`3599ms`), and cpu0 took 522 interrupts, 456 of them xHCI — a guest that was -running and slow, not one parked on a wake that never came. So the verdict's -cause is unread: the test waits host seconds and names a lost wake when they -run out, which a starved guest satisfies as well as a lost wake does. - -Main reproduces it. A same-session A/B, runs of main (`d65446cc`) and of the -branch started together so both arms carried one load (six suites at once): -main 16 of 17 green, the branch 17 of 18, and each arm's one red is this -sentence (`only 26 of 500` on main, `only 27 of 500` on the branch; the -branch's red spent `cpu=1568ms` and `cpu=1532ms` of its window). The rate is -the same on both arms, so the branch did not move it. - -**Exit**: the verdict tells a lost wake from a slow guest (the round trips' -progress over the window, not only the count at its end), and a cause for -this run. diff --git a/issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md b/issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md deleted file mode 100644 index 299b2482cba..00000000000 --- a/issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md +++ /dev/null @@ -1,38 +0,0 @@ ---- -status: expected-red -kind: finding -opened: 2026-09-07 ---- - -# `latency_wake` reds on the dev host at a rate - -Measured on the dev host (14 cores, macOS, TCG), same session, six runs alone, -three on `metal-suite` + the boot-deadline branch and three on the same tree -stashed back to `metal-suite`: - -| arm | p99 | verdict | -|---|---|---| -| deadline | 1606 us | PASS | -| deadline | 1634 us | PASS | -| deadline | 1785 us | PASS | -| base | 1456 us | PASS | -| base | 4096 us (floor, 902 past the histogram) | FAIL | -| base | 1428 us | PASS | - -The red is on the **base**, and the arm that touches the timer interrupt entry -did not produce one. The failure mode is always the same: the p99 lands in the -histogram's last bucket, so `4096us` is a floor and the harness refuses it as a -measurement rather than reporting a number it does not have. - -A seventh sighting, in the whole-branch review's twelve-wide `cargo test` on -7294acef: `258 past the 4096us histogram`, and the harness's own isolated re-run -green. - -What is still owed is the other reading of the same evidence: whether -cyclictest's 4,096-bucket histogram is simply too low for a TCG guest on a -loaded laptop, which is a change to the instrument and not to the kernel. - -**Exit condition.** Re-enabled when the base's p99 is shown to genuinely -exceed 4096 us and that is fixed at the timer-interrupt entry the deadline arm -already touches. Owner: the boot-deadline work (`kernel/src/sched`'s -timer-interrupt entry); held by the orchestrator. diff --git a/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md b/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md deleted file mode 100644 index b322bbd081d..00000000000 --- a/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md +++ /dev/null @@ -1,43 +0,0 @@ ---- -status: expected-red -kind: tooling -opened: 2026-09-13 ---- - -# The pass-cost gate's CI sample has been exceeded twice in eight days - -`sched_check_build`'s cost half on CI judges a boot's scheduler-pass -distribution against a recorded sample: sixteen CI runs, thirty-two CPU-runs, -2026-08-17 to 18, with zero passes over 200 000 ns and a 90th percentile of -32 768 ns (`tests/common/passcost.rs`). The line is one bucket above that -sample's worst 90th percentile. - -CI has now crossed it twice on branches that touch no scheduler code, each -time green alone in the same job: - -- `ci` run 33973213660, `guest (8)`, 2026-09-05, on `pkg-install-file`: - 1956 passes, p90 < 262144 ns, 366 over the budget. -- `ci` run 34768854940, `guest (11)`, 2026-09-13, the merge queue's run for - PR #448 (`host-bridge-abi`, the sysroot half of the host-bridge windows; - it moves no scheduler code): 3919 passes, p50 < 131072 ns, - p90 < 262144 ns, p99 < 262144 ns, max 2487630 ns, 51 over the budget. - The red dequeued the pull request and disarmed its auto-merge. - -Two sightings of a whole-distribution shift on a population the sample said -had none is not the sample returning; it is the population moving. Either -CI's runners changed under the sample (the August runs were one Azure SKU; -nothing records which SKU each shard lands on), or the tree's pass cost -moved in a way the dev host cannot see. The gate cannot tell, so every -further crossing costs a landing an hour and answers nothing. - -**Exit condition.** The sample is re-recorded from the CI runs since -2026-09-01 — every `guest` shard that ran this name, with the runner's -reported CPU model beside each reading — and the line re-derived from it -with the same rule; if the readings split by runner model, the gate names -the model it judges and refuses to judge the rest. Owner: orchestrator. - -A second signature: the fast -tier on PR #524's branch at `235c5a5b` reds `sched_check_build` on `cpu0: 85 -passes … a 90th percentile needs at least 100 samples behind it and this has -85`. The host was loaded then by another worktree's spinner at 397% CPU. The -harness's re-run alone was green. diff --git a/issues/build/there-is-no-attributed-session-ledger.md b/issues/build/there-is-no-attributed-session-ledger.md index 05d2c7f8a95..76db552988c 100644 --- a/issues/build/there-is-no-attributed-session-ledger.md +++ b/issues/build/there-is-no-attributed-session-ledger.md @@ -4,16 +4,13 @@ kind: track opened: 2026-09-01 --- -# There is no attributed session ledger, so seven flake records cannot name what overlapped them +# There is no attributed session ledger, so flake records cannot name what overlapped them -Seven open records describe a test that reds beside other work and is green +Open records describe a test that reds beside other work and is green alone, or a cost charged to the wrong artifact. Every one of them is blocked on the same missing observation: **no record joins a guest's loss of progress to -the interval that overlapped it.** They are not seven mechanisms. They are one -instrument, seven times. +the interval that overlapped it.** They are one instrument. -- `issues/audio/gate-a-has-no-runner-baseline.md` -- `issues/audio/thorough-tier-reds-on-unmodified-main.md` - `issues/audio/idle-suspend-reds-on-a-loaded-host-and-on-main.md` - `issues/boot-media/kernel-log-file-reds-beside-other-guests-and-is-green-alone.md` - `issues/boot-media/usb-short-read-reds-beside-other-guests-and-is-green-alone.md` @@ -32,16 +29,11 @@ image-build spans with their content key and cache hit or miss, QEMU and vCPU scheduling intervals where the host exposes them, and the guest progress markers the tests already emit. A sighting is then a join, not an inference. -**What exists, and why each is not the thing.** `tests/common/hostload.rs` (180 -lines) records load averages and process counts and is attached to an audio run -— a *sample*, not an interval, so it cannot say what overlapped a window. -`src/buildlock.rs` already names holders (`records_holder`, `guest_slot`, -`build_slot`) but the record is transient: it exists while the guard is held and -is gone when the question is asked. The audio baseline at `tests/toyos.rs:2107` -is keyed on `(test, smp)` and nothing else, so it cannot distinguish a CI -runner's distribution from the developer's — that key has no runner provenance -at all. The committed shard input is per-test duration only, which is -why `issues/build/the-shard-split-prices-a-boot-and-not-the-image-behind-it.md` +**What exists, and why each is not the thing.** `src/buildlock.rs` already +names holders (`records_holder`, `guest_slot`, `build_slot`) but the record is +transient: it exists while the guard is held and is gone when the question is +asked. The committed shard input is per-test duration only, which is why +`issues/build/the-shard-split-prices-a-boot-and-not-the-image-behind-it.md` charges an image build to whichever test followed it. **The probe must prove itself inert.** Collection writes files and samples the diff --git a/issues/build/timing-verdicts-ruled-off-qemu-have-no-metal-arm.md b/issues/build/timing-verdicts-ruled-off-qemu-have-no-metal-arm.md new file mode 100644 index 00000000000..2150387cbd5 --- /dev/null +++ b/issues/build/timing-verdicts-ruled-off-qemu-have-no-metal-arm.md @@ -0,0 +1,52 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# Timing verdicts ruled off QEMU have no metal arm yet + +A QEMU guest test asserts order, completion, content and counts, never how +long something took (owner ruling). The timing halves below were deleted from +the QEMU suite; each is a property of the T14, and no `METAL` row judges it +there yet. `latency_wake`'s p99, `tlb_shootdown_cost`'s tail, +`timer_calibration`'s ppm bound and `wake_storm_cost`'s linearity already have +one. + +- **Audio deadlines.** No silence on the wire while a client streams — the + capture's mid-tone gaps and soundd's `underruns` in `audio_tone`, + `soundd_log_stall` and `hda_tone`, and the same under two CPU burners + (`audio_tone_load`, deleted). The T14 carries no capture, so a metal arm reads + soundd's own `underruns` and `late_wakes` off the stick's `/log`. +- **Every CPU reaches a scheduler pass each heartbeat period, and no window + between two heartbeats is wide enough to hide a death** (`kernel_heartbeat`). +- **A fatal path halts the other CPUs before anything else**: none of their + records stamped more than 100 ms past the fatal one + (`panic_halts_the_others_first`, deleted). +- **A client out-writing a stalled peer costs netd no CPU**: under half a CPU + busy over a 2 s window (`netd_stalled_peer`, deleted). +- **A wake is delivered rather than waited out**: 500 pipe round trips and a + 24-child, 24-thread exit storm inside 3 s (`blocking_read_stress`, + `exit_wait_storm`), and every armed ring watcher woken inside 200 ms + (`poll_wake_pipe`). These binaries ride the shared metal boot, so their + bounds are gone there too. +- **With the acknowledgement delay disarmed, `munmap` returns under half the + delay** (`tlb_shootdown_waits`' control on its own instrument). +- **A scheduler pass's cost distribution** (`sched_check_build`); no metal arm + boots the check kernel. +- **A boot with no i8042 is no slower than one with it** (`i8042_absent`). +- **The USB connect settle ends on the device appearing and not at + `EMPTY_BUS_NS`** (`xhci_slow_connect`); **a disk call ends inside + `toyos_xhci::call::AFTER_BREAK`, a staged break skips its data-phase wait, and + the offline ladder ends inside 2.75 s** (`tests/common/usb.rs`). +- **A null sink drains a client at the audio rate** (`metal_sim_null_audio`, + which is `QemuOnly`). +- **init's stop waited for logd's flush answer**, read off the gap between init's + stop line and the kernel's sync (`log_ring_keeps_the_owners_slots`); only + init's own word is read now. + +Not changed, and timing-dependent below the harness: `dump_nmi_probe`'s verdict +rests on the kernel's own 1 ms NMI answer budget and 250 ms kick budget. + +**Exit**: each line above is judged by a `METAL` row, or ruled not owed. +Owner: the metal suite (`tests/toyos.rs`'s `METAL`); held by the orchestrator. diff --git a/issues/build/wake-storm-cost-red-under-induced-host-load.md b/issues/build/wake-storm-cost-red-under-induced-host-load.md deleted file mode 100644 index 5b21b250011..00000000000 --- a/issues/build/wake-storm-cost-red-under-induced-host-load.md +++ /dev/null @@ -1,63 +0,0 @@ ---- -status: open -kind: finding -opened: 2026-08-26 ---- - -# `wake_storm_cost` reds beside other guests on a deliberately loaded dev host - -One sighting, dev host, 2026-08-26, in a full 12-wide `cargo test` run with seven -pure-shell spin loops as company on a 14-core machine (load average 53). The -loop was staged to measure something else — a denominator for typed-input loss — -so the load is the instrument here rather than the tree's usual condition. - -``` -FAIL rs::wake_storm_cost: exit code 101 -wake_storm_cost: 16 waiters, 21000 cycles -wake_storm_cost: 64 waiters, 115000 cycles -quadrupling the storm from 16 waiters to 64 took the waker's own cost from -21000 cycles to 115000. `post_n` walks the waiters once and does a constant -amount per claim, so four times the waiters may cost four times the loop and -no more; past this, something in the claim grows with the size of the storm - FAIL wake_storm_cost (585ms) - ALONE wake_storm_cost: GREEN -``` - -`cargo run -- --known-red wake_storm_cost` answers **NOT ON THE LIST**, which is -why this file exists: the next reader of this name gets a sighting instead of -nothing. - -## Second sighting, hosted shard, 2026-08-26 — the other instrument - -Run 32909059602 `guest (1)`, the small-fix batch's pull request (a diff of -tracker closes and unrelated small fixes, none reaching the scheduler or -`post_n`): the same ratio assertion red, 183 of 184 names green beside it. A -hosted shard is four cores running one eight-vCPU guest, so it is loaded by -construction — which means both recorded firings are on oversubscribed hosts -and none on a quiet one. That is half the denominator the section below asks -for; the quiet-host arm is still the missing half. - -## What this is and is not evidence about - -The assertion is a *ratio* of guest TSC deltas, so a host that deschedules the -vCPU inside the 64-waiter arm and not inside the 16-waiter one moves the ratio -without anything in the kernel changing. That makes host oversubscription a -live alternative to a claim cost that grows with the storm, and this sighting -cannot separate them: it is one observation at one load. - -The tree it ran on touches `tests/` only — the harness's typed-input delivery — -and nothing in that diff reaches the scheduler, the wait queues or `post_n`. - -## What would settle it - -Either arm measured against the other in the same session: the same ratio taken -on a quiet host and on a loaded one, several runs of each, with the host's load -recorded per run. If the ratio moves with the load, the assertion needs a -denominator that host time cannot inflate — a count of claims rather than a span -of cycles. If it does not, the finding is the kernel's and this is a `defect`. - -## Third sighting, main's nightly at 1ce71831 - -Run 36290616312, one guest shard: `FAIL rs::wake_storm_cost: exit code 101` -wide, then `ALONE wake_storm_cost: GREEN` twice. This is again a hosted -four-core shard running eight-vCPU guests, so an oversubscribed host. diff --git a/issues/hardware/the-t14-boots-toyos-unattended.md b/issues/hardware/the-t14-boots-toyos-unattended.md index 4cb50310dfe..7e5de030809 100644 --- a/issues/hardware/the-t14-boots-toyos-unattended.md +++ b/issues/hardware/the-t14-boots-toyos-unattended.md @@ -112,9 +112,7 @@ What is left to build: `issues/kernel/the-split-window-tlb-cost-is-unpriced.md`, `issues/kernel/ap-control-registers-inherit-init.md`, `issues/kernel/ap-tsc-trail-is-assumed-and-never-checked.md`, - `issues/audio/hda-ring-fix-unverified-on-metal.md`, - `issues/audio/t14-wake-lateness-is-bimodal-per-boot.md`, - `issues/audio/gate-a-has-no-runner-baseline.md` (a metal sample), and the + `issues/audio/hda-ring-fix-unverified-on-metal.md`, and the IOMMU track's three hardware-only answers — isolation scopes and reserved regions, the 2× cost bar, and the compatibility-format question in `issues/kernel/qemu-passes-compatibility-format-interrupts.md` diff --git a/kernel/src/drivers/hda.rs b/kernel/src/drivers/hda.rs index daf8f95b3d6..7a87c3ef528 100644 --- a/kernel/src/drivers/hda.rs +++ b/kernel/src/drivers/hda.rs @@ -77,8 +77,8 @@ const SD_STS_FIFOE: u8 = 1 << 3; const SD_STS_DESE: u8 = 1 << 4; const SD_STS_WRITE_CLEAR: u8 = SD_STS_BCIS | SD_STS_FIFOE | SD_STS_DESE; -/// The pipeline shape; soundd's mix loop, its client ring depth and gate A's recorded counters are -/// sized against it. +// The pipeline shape; soundd's mix loop and its client ring depth are sized +/// against it. const PERIODS: usize = 8; const PERIOD_BYTES: usize = 512; diff --git a/src/build.rs b/src/build.rs index 34dba99c00e..dbf3639bed1 100644 --- a/src/build.rs +++ b/src/build.rs @@ -1667,12 +1667,11 @@ fn assert_entry_window_matches_features(features: &str, kernel: &[u8]) { /// bound and the container-versus-state-word agreement. The third is the /// pass-cost report (`cpu::PassCostReport::PREFIX`), which is a *measurement* /// and not an assert: a pass's elapsed time includes any interval a hypervisor -/// took the CPU away, so it is recorded and gated in the harness rather than -/// panicked over. Their format strings are the only part of the check build -/// with a literal the linker keeps, which is what makes the artifact answerable -/// at all — and the report's literal is kept out of the shipping kernel by -/// nothing but dead-code elimination, which the `want == false` direction below -/// is what checks. +/// took the CPU away, so it is recorded rather than panicked over. Their format +/// strings are the only part of the check build with a literal the linker keeps, +/// which is what makes the artifact answerable at all — and the report's literal +/// is kept out of the shipping kernel by nothing but dead-code elimination, +/// which the `want == false` direction below is what checks. const SCHED_CHECK_LITERALS: [&str; 3] = [ "sched-check pass-costs cpu=", "invariant T: cpu", diff --git a/src/ci.rs b/src/ci.rs index 5518b76c6b8..fecff6f36da 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -48,7 +48,6 @@ const USAGE: &str = "cargo run -- --ci , where is one of: toolchain publish this tree's toolchain if nobody has (nightly) guest / one shard of the whole guest suite, nightly tier included (nightly) tcg one test on an emulated CPU (nightly) - audio / one shard of gate A (nightly) nightly-red file or update the nightly-red issue from $NEEDS (nightly) publish put main's SDK crates on crates.io (publish.yml)"; @@ -59,7 +58,6 @@ enum Job { Toolchain, Guest(String), Tcg, - Audio(String), NightlyRed, Publish, } @@ -76,13 +74,12 @@ fn parse(words: &[String]) -> Result { Some("toolchain") => Job::Toolchain, Some("guest") => Job::Guest(shard(words.get(1))?), Some("tcg") => Job::Tcg, - Some("audio") => Job::Audio(shard(words.get(1))?), Some("nightly-red") => Job::NightlyRed, Some("publish") => Job::Publish, Some(other) => return Err(format!("no CI job is called {other:?}")), None => return Err("which job?".to_string()), }; - let takes = usize::from(matches!(job, Job::Guest(_) | Job::Audio(_))) + 1; + let takes = usize::from(matches!(job, Job::Guest(_))) + 1; if words.len() > takes { return Err(format!("{:?} takes nothing after it: {:?}", words[0], &words[takes..])); } @@ -102,7 +99,6 @@ pub fn dispatch(root: &Path, args: &[String]) { guest(root, &suite_args(&["--shard", shard, "--jobs", "1", "--nightly"])) } Job::Tcg => guest(root, &suite_args(&["--jobs", "1", "process_stats"])), - Job::Audio(shard) => guest(root, &suite_args(&["--audio-gate", "30", "--shard", shard])), Job::NightlyRed => vec![step("the nightly-red issue", nightly_red)], Job::Publish => vec![step("the SDK crates on crates.io", || publish(root))], }; diff --git a/src/heartbeat.rs b/src/heartbeat.rs deleted file mode 100644 index eddfe9d099d..00000000000 --- a/src/heartbeat.rs +++ /dev/null @@ -1,756 +0,0 @@ -//! Whether a heartbeat capture settled and what its mask then says — the -//! verdict `kernel_heartbeat` in `tests/toyos.rs` reads. Text in, a verdict -//! out, so a capture the instrument has already taken replays here against the -//! rule. -//! -//! **What a clear bit can be evidence of.** `mask=` says which CPUs reached a -//! scheduler pass in the period (`kernel/src/heartbeat.rs`), and a CPU running -//! one task with nothing to preempt it takes its timer and reaches none, as does -//! one inside a disk wait, which cannot park — so a clear bit reads "busy" and -//! "stopped" alike. The test exists for the second, on a machine that is -//! otherwise running, and a beat can say so only where the machine held that -//! state: -//! -//! - **after the boot's start-up.** That is not `Boot: complete`, which is -//! printed as `init` is spawned, and not `init`'s last spawn record either: -//! the programs `init` starts do their own start-up after it, and that work -//! is what keeps a CPU off the mask. The boot has started when every -//! `[boot] start` program has said it is done — its ready line or its exit -//! record, which `DONE` is the one table of — and the window opens at the -//! first beat whose whole period follows the last of them. -//! - **on a beat the machine ran through.** A beat later than `LATE` periods -//! is a period no CPU reached the idle loop in, and `ran=0` is a period no -//! CPU dispatched a task in. Either is the machine not running — a guest its -//! host did not schedule — and what a CPU did in it is unreadable, so such a -//! beat closes the window and the sample is [`Refused::NotRunning`], never a -//! CPU missing. -//! - **over `MIN_SETTLED` beats.** A verdict about a CPU is read from that many -//! settled beats or from none: a shorter window says only why it is short. -//! -//! Inside the window a CPU absent from [`STOPPED_BEATS`] consecutive beats has -//! stopped: `diag-tick` caps a sleep at 100 ms against a 250 ms line, so it has -//! missed five wakes. Absent from one and back on the next it has missed two, -//! which the owning instrument produces on a healthy guest. -//! -//! **The capture follows the window, not a clock.** How long a boot's start-up -//! takes is the loaded host's to decide, so an instrument that drains for a -//! fixed span hands this module whatever is left over and reds on its own -//! refusal when that is less than a verdict needs. [`window_beats`] is what a -//! capture is taken to: it says how much window the capture holds so far, and -//! [`CAPTURE_BEATS`] is enough. - -#![forbid(unsafe_code)] - -/// The line's period, `kernel/src/heartbeat.rs`'s `PERIOD_NS`. -pub const PERIOD_MS: u64 = 250; - -/// A beat whose `gap=` exceeds this many periods is one the machine did not run -/// through: eight CPUs whose longest sleep is 100 ms, and none reached the idle -/// loop for a whole period. -const LATE: u64 = 2; - -/// Consecutive settled beats a CPU is absent from before it has stopped. -pub const STOPPED_BEATS: usize = 2; - -/// The fewest settled beats a verdict is read from. -const MIN_SETTLED: usize = 4; - -/// Settled beats a capture is taken to. `MIN_SETTLED` is the floor a verdict is -/// read from and the spare is detection, not slack: a CPU whose first absence -/// is the capture's last beat is a blip and convicts nobody, so the capture -/// carries one beat past the floor — enough for [`STOPPED_BEATS`]'s second -/// absence to land in. -pub const CAPTURE_BEATS: usize = MIN_SETTLED + 1; - -/// The widest `alive=N/M` denominator that is a reading of the `mask=` beside -/// it: that mask is 64 bits, so no `M` at 64 or above describes it, and `M = 0` -/// describes no machine. -const MOST_CPUS: u32 = 63; - -/// One `heartbeat: t=… alive=… mask=… ran=… gap=…` line, read. -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct Beat { - /// Index of the line in the capture. - pub line: usize, - pub t_ms: u64, - pub cpus: u32, - pub mask: u64, - pub ran: u64, - pub gap_ms: u64, -} - -impl Beat { - /// The beat's reading of `text`, or `None` where any field is unreadable. - fn parse(line: usize, text: &str) -> Option { - let field = |key: &str| text.split(key).nth(1)?.split_whitespace().next(); - let cpus: u32 = field("alive=")?.split_once('/')?.1.parse().ok()?; - if !(1..=MOST_CPUS).contains(&cpus) { - return None; - } - Some(Beat { - line, - t_ms: millis(field("t=")?)?, - cpus, - mask: u64::from_str_radix(field("mask=0x")?, 16).ok()?, - ran: field("ran=")?.parse().ok()?, - gap_ms: millis(field("gap=")?)?, - }) - } - - fn full(&self) -> bool { - self.mask == (1u64 << self.cpus) - 1 - } - - fn absent(&self, cpu: u32) -> bool { - self.mask & (1 << cpu) == 0 - } - - /// Whether the machine ran through this beat's period. - fn ran_through(&self) -> bool { - self.gap_ms <= LATE * PERIOD_MS && self.ran > 0 - } -} - -/// `S.mmms` as milliseconds. -fn millis(field: &str) -> Option { - let (s, ms) = field.strip_suffix('s')?.split_once('.')?; - if ms.len() != 3 { - return None; - } - Some(s.parse::().ok()? * 1000 + ms.parse::().ok()?) -} - -/// `tests/metalcase`'s `[boot] start` programs and the line each says it has -/// finished starting with. The one table: [`done_lines`] holds it against the -/// config, and a caller reads it through that rather than declaring its own. -const DONE: &[(&str, &str)] = &[ - ("logd", "logd: this boot's kernel log is"), - ("compositor", "compositor: ready"), - ("soundd", "soundd: null sink idle"), - ("netd", "exit: netd pid="), - ("sshd", "exit: sshd pid="), - ("test-runner", "===READY==="), -]; - -/// The done line of each program in `start`, or the disagreement between the -/// config and [`DONE`] — a `[boot] start` program with no done line here leaves -/// its own start-up inside the window, which is the one thing the window exists -/// to exclude. -pub fn done_lines(start: &[String]) -> Result, String> { - let start: Vec<&str> = start.iter().map(String::as_str).collect(); - let known: Vec<&str> = DONE.iter().map(|(program, _)| *program).collect(); - if start != known { - return Err(format!( - "`tests/metalcase` starts {start:?} and `src/heartbeat.rs` knows the done line of \ - {known:?} — a program without one leaves its start-up inside the window" - )); - } - Ok(DONE.iter().map(|(_, line)| *line).collect()) -} - -/// Why a capture is not a claim about a settled, running machine's CPUs. -#[derive(Debug, PartialEq, Eq)] -pub enum Refused { - /// A `heartbeat: t=` line one of whose fields would not parse. - Unreadable(String), - /// A `[boot] start` program never said it was done: the line it says it with. - BootUnfinished(String), - /// Fewer than `MIN_SETTLED` beats had a whole period after the boot's - /// start-up. - Unsettled { settled: usize, beats: usize }, - /// A settled beat the machine did not run through, after `held` it did. - NotRunning { beat: Beat, held: usize }, - /// CPUs absent from [`STOPPED_BEATS`] consecutive settled beats, of `settled` - /// from the capture line `opened`. - CpuMissing { cpus: Vec, settled: usize, opened: usize }, -} - -/// The settled window, read. -#[derive(Debug, PartialEq, Eq)] -pub struct Settled { - pub beats: Vec, - /// Settled beats missing a CPU — for one line each, since none for two. - pub blips: usize, - /// The widest `gap=` anywhere in the capture, settled or not: a window - /// between two lines is wide enough to hide a death wherever it falls. - pub widest_gap_ms: u64, -} - -/// The beats whose whole period follows the last of `started`, or the done line -/// nothing in `lines` said. -fn window<'a>(beats: &'a [Beat], lines: &[&str], started: &[&str]) -> Result<&'a [Beat], String> { - let mut last = 0; - for said in started { - let Some(at) = lines.iter().position(|l| l.contains(said)) else { - return Err((*said).to_string()); - }; - last = last.max(at); - } - // The beat after the record straddles it; the one after that is the first - // whose whole period follows it. - let after = beats.iter().filter(|b| b.line > last).count(); - Ok(&beats[(beats.len() - after + 1).min(beats.len())..]) -} - -/// How much window `lines` holds so far — what a capture is taken to, against -/// [`CAPTURE_BEATS`]. Zero until every one of `started` has said it is done, and -/// a line whose fields will not parse is not a beat here; [`settle`] is what -/// refuses one. -pub fn window_beats(lines: &[&str], started: &[&str]) -> usize { - let beats: Vec = - lines.iter().enumerate().filter_map(|(i, l)| Beat::parse(i, l)).collect(); - window(&beats, lines, started).map_or(0, <[Beat]>::len) -} - -/// The verdict on `lines`, a capture whose boot's start-up ends with the last of -/// `started` — one line per `[boot] start` program, the one it says it is done -/// with. -pub fn settle(lines: &[&str], started: &[&str]) -> Result { - let beats = lines - .iter() - .enumerate() - .filter(|(_, l)| l.contains("heartbeat: t=")) - .map(|(i, l)| Beat::parse(i, l).ok_or_else(|| Refused::Unreadable(l.to_string()))) - .collect::, _>>()?; - if let Some(odd) = beats.iter().find(|b| b.cpus != beats[0].cpus) { - return Err(Refused::Unreadable(lines[odd.line].to_string())); - } - let window = window(&beats, lines, started).map_err(Refused::BootUnfinished)?; - let held = window.iter().position(|b| !b.ran_through()).unwrap_or(window.len()); - let read = &window[..held]; - // A CPU is read from `MIN_SETTLED` settled beats or from none, so what a - // short window says is only why it is short. - if held < MIN_SETTLED { - return Err(if held < window.len() { - Refused::NotRunning { beat: window[held].clone(), held } - } else { - Refused::Unsettled { settled: held, beats: beats.len() } - }); - } - let cpus: Vec = (0..beats[0].cpus) - .filter(|&c| read.windows(STOPPED_BEATS).any(|w| w.iter().all(|b| b.absent(c)))) - .collect(); - if !cpus.is_empty() { - return Err(Refused::CpuMissing { cpus, settled: read.len(), opened: read[0].line }); - } - if held < window.len() { - return Err(Refused::NotRunning { beat: window[held].clone(), held }); - } - Ok(Settled { - blips: read.iter().filter(|b| !b.full()).count(), - widest_gap_ms: beats.iter().map(|b| b.gap_ms).max().unwrap_or(0), - beats: read.to_vec(), - }) -} - -#[cfg(test)] -mod tests { - use std::path::Path; - - use super::*; - - /// `tests/metalcase`'s done lines, as the test passes them. - fn started() -> Vec<&'static str> { - done_lines(&crate::build::boot_start( - &Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/metalcase/system.toml"), - )) - .unwrap() - } - - /// Nightly `35072262489`, guest shard 8, suite run: every heartbeat line and - /// every `[boot] start` program's done line, in the capture's order and - /// verbatim, the lines between them dropped. - const SHARD_8_SUITE: &str = "\ -logd: this boot's kernel log is /log/2026-09-16-083530.log (2026-09-16 08:35:30 at UTC+0 recovered from two readings) -[kernel 1.109 cpu1] heartbeat: t=1.109s alive=8/8 mask=0xff ran=11 gap=0.260s -[kernel 1.361 cpu7] heartbeat: t=1.361s alive=7/8 mask=0xdf ran=4 gap=0.251s -[kernel 1.361 cpu7] heartbeat: cpu5 last reached one 0.307s ago -[kernel 1.618 cpu3] heartbeat: t=1.618s alive=7/8 mask=0xbf ran=5 gap=0.257s -[kernel 1.618 cpu3] heartbeat: cpu6 last reached one 0.423s ago -soundd: null sink idle -[kernel 1.762 cpu7] exit: netd pid=7 code=0 cpu=399ms -[kernel 1.934 cpu1] heartbeat: t=1.934s alive=7/8 mask=0xfe ran=9 gap=0.315s -[kernel 1.934 cpu1] heartbeat: cpu0 last reached one 0.572s ago -===READY=== -[kernel 2.021 cpu0] spawn: /system/bin/test-runner pid=9 tid=0 dst=2 base=0x10000000000 entry=0x1000001ff80 cr3=0x6ecf000 symbols=2048KiB (layout=21ms relocs=0ms deps=0ms tls=1ms total=75ms) -[kernel 2.184 cpu0] heartbeat: t=2.184s alive=7/8 mask=0xdf ran=8 gap=0.250s -[kernel 2.184 cpu0] heartbeat: cpu5 last reached one 0.339s ago -[kernel 2.434 cpu0] heartbeat: t=2.434s alive=7/8 mask=0xdf ran=15 gap=0.250s -[kernel 2.434 cpu0] heartbeat: cpu5 last reached one 0.589s ago -compositor: ready -[kernel 2.618 cpu1] exit: sshd pid=8 code=0 cpu=634ms -[kernel 2.686 cpu2] heartbeat: t=2.686s alive=8/8 mask=0xff ran=36 gap=0.251s -[kernel 2.937 cpu2] heartbeat: t=2.937s alive=8/8 mask=0xff ran=43 gap=0.251s -[kernel 3.188 cpu2] heartbeat: t=3.188s alive=8/8 mask=0xff ran=43 gap=0.251s -[kernel 3.439 cpu2] heartbeat: t=3.439s alive=8/8 mask=0xff ran=43 gap=0.251s -[kernel 3.690 cpu5] heartbeat: t=3.690s alive=8/8 mask=0xff ran=42 gap=0.250s -[kernel 3.940 cpu2] heartbeat: t=3.940s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 4.196 cpu5] heartbeat: t=4.196s alive=8/8 mask=0xff ran=44 gap=0.255s -[kernel 4.449 cpu5] heartbeat: t=4.449s alive=8/8 mask=0xff ran=44 gap=0.252s -[kernel 4.703 cpu2] heartbeat: t=4.703s alive=8/8 mask=0xff ran=43 gap=0.253s -[kernel 4.953 cpu2] heartbeat: t=4.953s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 5.203 cpu2] heartbeat: t=5.203s alive=8/8 mask=0xff ran=44 gap=0.250s -"; - - /// The same shard's ALONE re-run, the same way. - const SHARD_8_ALONE: &str = "\ -logd: this boot's kernel log is /log/2026-09-16-083541.log (2026-09-16 08:35:41 at UTC+0 recovered from two readings) -[kernel 1.029 cpu1] heartbeat: t=1.029s alive=8/8 mask=0xff ran=11 gap=0.262s -[kernel 1.290 cpu7] heartbeat: t=1.290s alive=7/8 mask=0xdf ran=4 gap=0.260s -[kernel 1.290 cpu7] heartbeat: cpu5 last reached one 0.323s ago -[kernel 1.541 cpu3] heartbeat: t=1.541s alive=7/8 mask=0xbf ran=5 gap=0.251s -[kernel 1.544 cpu3] heartbeat: cpu6 last reached one 0.429s ago -soundd: null sink idle -[kernel 1.692 cpu7] exit: netd pid=7 code=0 cpu=402ms -[kernel 1.793 cpu7] heartbeat: t=1.793s alive=7/8 mask=0xfe ran=10 gap=0.251s -[kernel 1.793 cpu7] heartbeat: cpu0 last reached one 0.499s ago -===READY=== -[kernel 1.951 cpu0] spawn: /system/bin/test-runner pid=9 tid=0 dst=2 base=0x10000000000 entry=0x1000001ff80 cr3=0x2d83000 symbols=2048KiB (layout=9ms relocs=0ms deps=0ms tls=1ms total=79ms) -[kernel 2.043 cpu0] heartbeat: t=2.043s alive=7/8 mask=0xdf ran=8 gap=0.250s -[kernel 2.043 cpu0] heartbeat: cpu5 last reached one 0.281s ago -[kernel 2.293 cpu0] heartbeat: t=2.293s alive=6/8 mask=0xdd ran=1 gap=0.250s -[kernel 2.293 cpu0] heartbeat: cpu1 last reached one 0.422s ago -[kernel 2.293 cpu0] heartbeat: cpu5 last reached one 0.531s ago -compositor: ready -[kernel 2.543 cpu0] heartbeat: t=2.543s alive=7/8 mask=0xdf ran=25 gap=0.250s -[kernel 2.543 cpu0] heartbeat: cpu5 last reached one 0.781s ago -[kernel 2.624 cpu1] exit: sshd pid=8 code=0 cpu=708ms -[kernel 2.793 cpu5] heartbeat: t=2.793s alive=8/8 mask=0xff ran=42 gap=0.250s -[kernel 3.044 cpu2] heartbeat: t=3.044s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 3.294 cpu2] heartbeat: t=3.294s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 3.545 cpu2] heartbeat: t=3.545s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 3.796 cpu2] heartbeat: t=3.796s alive=8/8 mask=0xff ran=43 gap=0.251s -[kernel 4.047 cpu2] heartbeat: t=4.047s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 4.298 cpu2] heartbeat: t=4.298s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 4.557 cpu2] heartbeat: t=4.557s alive=8/8 mask=0xff ran=44 gap=0.259s -[kernel 4.814 cpu2] heartbeat: t=4.814s alive=8/8 mask=0xff ran=43 gap=0.256s -[kernel 5.065 cpu2] heartbeat: t=5.065s alive=8/8 mask=0xff ran=43 gap=0.251s -"; - - /// A dev-host boot, up to its torn last beat: every heartbeat line, every - /// `[boot] start` program's done line and the two lines naming the wait, in - /// the capture's order and verbatim, the lines between them dropped. - const DISK_WAIT_PINS_CPU4: &str = "\ -logd: this boot's kernel log is /log/2026-09-18-135738.log (2026-09-18 13:57:38 at UTC+0 recovered from two readings) -[kernel 1.122 cpu3] heartbeat: t=1.121s alive=8/8 mask=0xff ran=11 gap=0.261s -[kernel 1.395 cpu1] heartbeat: t=1.395s alive=8/8 mask=0xff ran=4 gap=0.273s -soundd: null sink idle -[kernel 1.529 cpu7] exit: netd pid=7 code=0 cpu=94ms -[kernel 1.645 cpu0] heartbeat: t=1.645s alive=8/8 mask=0xff ran=22 gap=0.250s -===READY=== -[kernel 1.895 cpu0] heartbeat: t=1.895s alive=6/8 mask=0xdb ran=4 gap=0.250s -[kernel 2.146 cpu0] heartbeat: t=2.145s alive=6/8 mask=0xdd ran=23 gap=0.250s -[kernel 2.230 cpu1] exit: sshd pid=8 code=0 cpu=666ms -[kernel 2.397 cpu2] heartbeat: t=2.397s alive=8/8 mask=0xff ran=31 gap=0.251s -compositor: ready -[kernel 2.658 cpu2] heartbeat: t=2.658s alive=8/8 mask=0xff ran=33 gap=0.261s -[kernel 2.909 cpu0] heartbeat: t=2.909s alive=8/8 mask=0xff ran=38 gap=0.250s -[kernel 3.159 cpu0] heartbeat: t=3.159s alive=7/8 mask=0xef ran=35 gap=0.250s -[kernel 3.409 cpu0] heartbeat: t=3.409s alive=7/8 mask=0xef ran=26 gap=0.250s -[kernel 3.659 cpu0] heartbeat: t=3.659s alive=7/8 mask=0xef ran=38 gap=0.250s -[kernel 3.909 cpu0] heartbeat: t=3.909s alive=7/8 mask=0xef ran=32 gap=0.250s -[kernel 4.159 cpu0] heartbeat: t=4.159s alive=7/8 mask=0xef ran=37 gap=0.250s -[kernel 4.409 cpu0] heartbeat: t=4.409s alive=7/8 mask=0xef ran=37 gap=0.250s -[kernel 4.659 cpu0] heartbeat: t=4.659s alive=7/8 mask=0xef ran=33 gap=0.250s -[kernel 4.663 cpu4] usb-storage: 00:02.0 slot 1 transport broke on SCSI 0x2a: no answer in the status phase in 2000 ms -[kernel 4.677 cpu4] fsync: /log/2026-09-18-135738.log durable on attempt 2 after 2016ms — a refused attempt kept every page dirty and a later one delivered them -"; - - /// A CPU a disk wait pins reaches no pass, and that is what the mask says: - /// every beat of it ran through, so neither refusal excuses it, and the - /// verdict names the CPU. The red's owner is the wait that cannot park. - #[test] - fn a_cpu_pinned_by_a_disk_wait_is_reported_and_not_excused() { - let lines = lines(DISK_WAIT_PINS_CPU4); - let opened = lines.iter().position(|l| l.contains("t=2.909s")).unwrap(); - assert_eq!( - settle(&lines, &started()), - Err(Refused::CpuMissing { cpus: vec![4], settled: 8, opened }) - ); - } - - fn lines(capture: &str) -> Vec<&str> { - capture.lines().collect() - } - - fn beat(t_ms: u64, mask: u64, ran: u64, gap_ms: u64) -> String { - format!( - "[kernel {}.{:03} cpu2] heartbeat: t={}.{:03}s alive={}/8 mask={mask:#04x} ran={ran} \ - gap={}.{:03}s", - t_ms / 1000, - t_ms % 1000, - t_ms / 1000, - t_ms % 1000, - mask.count_ones(), - gap_ms / 1000, - gap_ms % 1000, - ) - } - - /// A capture whose boot's start-up ends before its first beat, then `beats`. - fn settled_capture(beats: &[String]) -> String { - let mut capture: String = started().iter().map(|s| format!("{s}\n")).collect(); - capture.push_str(&beat(2000, 0xff, 40, 250)); - capture.push('\n'); - for b in beats { - capture.push_str(b); - capture.push('\n'); - } - capture - } - - /// The started programs' own start-up outlasts `init`'s last spawn record, - /// so a window opened at a full mask holds cpu5 absent from two consecutive - /// beats; opened after the last done line it holds no clear bit at all. - #[test] - fn the_nightly_shard_8_boots_settle_after_the_last_program_finishes_starting() { - for (capture, opens_at, settled, widest) in - [(SHARD_8_SUITE, 2937, 10, 315), (SHARD_8_ALONE, 3044, 9, 262)] - { - let lines = lines(capture); - let beats: Vec = lines - .iter() - .enumerate() - .filter_map(|(i, l)| Beat::parse(i, l)) - .collect(); - assert_eq!(beats.len(), 17); - assert!(beats[0].full()); - let spawned = lines - .iter() - .position(|l| l.contains("spawn: /system/bin/test-runner")) - .unwrap(); - let after_spawn: Vec<&Beat> = beats.iter().filter(|b| b.line > spawned).collect(); - assert!(after_spawn[0].absent(5) && after_spawn[1].absent(5)); - let verdict = settle(&lines, &started()).unwrap(); - assert_eq!(verdict.beats[0].t_ms, opens_at); - assert_eq!(verdict.beats.len(), settled); - assert_eq!(verdict.blips, 0); - assert_eq!(verdict.widest_gap_ms, widest); - let exit = lines.iter().position(|l| l.contains("exit: sshd")).unwrap(); - let straddling = beats.iter().find(|b| b.line > exit).unwrap(); - assert!(straddling.t_ms < opens_at); - } - } - - /// A dev-host boot the host stopped scheduling: 37 beats, a pair of full - /// masks at t=2.654 s and 2.904 s, then seven seconds of `alive=4/8 ran=0` - /// naming cpu[0, 1, 2, 4, 5]. Reconstructed from those numbers rather than - /// captured — the boot beats and the split of the missing set across the - /// tail are this test's. - fn dev_host_capture() -> String { - let started = started(); - let mut capture = String::new(); - for said in &started[..4] { - capture.push_str(said); - capture.push('\n'); - } - for (t, mask, ran) in [(904, 0xff, 11), (1154, 0xdf, 4), (1404, 0xbf, 5), (1654, 0xfe, 9)] { - capture.push_str(&beat(t, mask, ran, 250)); - capture.push('\n'); - } - capture.push_str("===READY===\n"); - for (t, mask, ran) in [(1904, 0xdf, 8), (2154, 0xdf, 15)] { - capture.push_str(&beat(t, mask, ran, 250)); - capture.push('\n'); - } - capture.push_str("[kernel 2.3 cpu1] exit: sshd pid=8 code=0 cpu=634ms\n"); - for (t, mask, ran) in [(2404, 0xdd, 1), (2654, 0xff, 36), (2904, 0xff, 43)] { - capture.push_str(&beat(t, mask, ran, 250)); - capture.push('\n'); - } - for i in 0..28u64 { - let mask = if i < 14 { 0xe8 } else { 0xc9 }; - capture.push_str(&beat(3154 + i * 250, mask, 0, 250)); - capture.push('\n'); - } - capture - } - - #[test] - fn the_dev_host_capture_is_a_machine_that_was_not_running() { - let capture = dev_host_capture(); - let lines = lines(&capture); - let beats: Vec = - lines.iter().enumerate().filter_map(|(i, l)| Beat::parse(i, l)).collect(); - assert_eq!(beats.len(), 37); - let pair = beats.windows(2).position(|w| w[0].full() && w[1].full()).unwrap(); - assert_eq!(beats[pair].t_ms, 2654); - let stopped: Vec = (0..8) - .filter(|&c| beats[pair..].windows(2).any(|w| w[0].absent(c) && w[1].absent(c))) - .collect(); - assert_eq!(stopped, [0, 1, 2, 4, 5]); - assert_eq!( - settle(&lines, &started()), - Err(Refused::NotRunning { beat: beats[pair + 2].clone(), held: 2 }) - ); - assert_eq!(beats[pair + 2].ran, 0); - } - - /// The owning instrument's healthy shape: one line a CPU is absent from and - /// back on is two missed wakes, not a stopped CPU. - #[test] - fn one_line_absent_and_back_is_a_blip() { - let mut beats: Vec = (1..=9).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.push(beat(4500, 0xbf, 43, 250)); - beats.extend((1..=8).map(|i| beat(4500 + i * 250, 0xff, 43, 250))); - let capture = settled_capture(&beats); - let verdict = settle(&lines(&capture), &started()).unwrap(); - assert_eq!(verdict.beats.len(), 18); - assert_eq!(verdict.blips, 1); - assert_eq!(verdict.widest_gap_ms, 250); - } - - #[test] - fn a_cpu_absent_from_two_consecutive_settled_beats_has_stopped() { - let mut beats: Vec = (1..=3).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.extend((1..=6).map(|i| beat(2750 + i * 250, 0xdf, 43, 250))); - let capture = settled_capture(&beats); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::CpuMissing { cpus: vec![5], settled: 9, opened: started().len() + 1 }) - ); - } - - /// A late beat is a period no CPU reached the idle loop in; an empty one is - /// a period no CPU ran a task in. Both close the window, and what follows - /// is not read as a CPU. - #[test] - fn a_beat_the_machine_did_not_run_through_closes_the_window() { - for (gap_ms, ran) in [(600, 43), (250, 0)] { - let mut beats: Vec = - (1..=4).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - let stalled = beat(3650, 0xdf, ran, gap_ms); - beats.push(stalled.clone()); - beats.extend((1..=6).map(|i| beat(3650 + i * 250, 0xdf, 43, 250))); - let capture = settled_capture(&beats); - let lines = lines(&capture); - let at = lines.iter().position(|l| *l == stalled).unwrap(); - assert_eq!( - settle(&lines, &started()), - Err(Refused::NotRunning { beat: Beat::parse(at, &stalled).unwrap(), held: 4 }) - ); - } - } - - /// `LATE`, both sides of it, in milliseconds rather than in the constant: a - /// 0.500 s gap is two 250 ms periods and the machine ran through it, and a - /// 0.600 s gap is a period in which no CPU reached the idle loop at all — - /// five `diag-tick` wakes missed — and closes the window. - #[test] - fn two_periods_of_gap_is_run_through_and_more_than_two_is_not() { - let settled = |gap_ms| { - let mut beats: Vec = - (1..=4).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.push(beat(3250, 0xff, 43, gap_ms)); - settle(&lines(&settled_capture(&beats)), &started()).map(|s| s.beats.len()) - }; - assert_eq!(settled(500), Ok(5)); - assert!(matches!(settled(600), Err(Refused::NotRunning { held: 4, .. }))); - } - - #[test] - fn a_cpu_that_stopped_before_the_stall_is_still_the_finding() { - let mut beats: Vec = (1..=4).map(|i| beat(2000 + i * 250, 0xdf, 43, 250)).collect(); - beats.push(beat(3400, 0xdf, 43, 600)); - let capture = settled_capture(&beats); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::CpuMissing { cpus: vec![5], settled: 4, opened: started().len() + 1 }) - ); - } - - /// A verdict about a CPU is read from `MIN_SETTLED` settled beats or from - /// none: a stall two beats in, and a capture that ends three beats in, each - /// say why the window is short and neither names a CPU. - #[test] - fn a_cpu_missing_from_a_window_shorter_than_the_minimum_is_not_named() { - let mut beats: Vec = (1..=2).map(|i| beat(2000 + i * 250, 0xdf, 43, 250)).collect(); - beats.push(beat(3400, 0xdf, 43, 600)); - let capture = settled_capture(&beats); - let stalled = lines(&capture); - let at = stalled.iter().position(|l| l.contains("t=3.400s")).unwrap(); - assert_eq!( - settle(&stalled, &started()), - Err(Refused::NotRunning { beat: Beat::parse(at, stalled[at]).unwrap(), held: 2 }) - ); - - let beats: Vec = (1..=3).map(|i| beat(2000 + i * 250, 0xdf, 43, 250)).collect(); - assert_eq!( - settle(&lines(&settled_capture(&beats)), &started()), - Err(Refused::Unsettled { settled: 3, beats: 4 }) - ); - } - - #[test] - fn a_program_that_never_finishes_starting_refuses_the_whole_capture() { - let without: Vec<&str> = lines(SHARD_8_SUITE) - .into_iter() - .filter(|l| !l.contains("exit: sshd")) - .collect(); - assert_eq!( - settle(&without, &started()), - Err(Refused::BootUnfinished("exit: sshd pid=".to_string())) - ); - assert_eq!(window_beats(&without, &started()), 0); - } - - /// The beat after the last done line straddles it, so the window opens on - /// the one after that; a done line after every beat opens nothing. - #[test] - fn the_window_opens_on_the_first_whole_period_after_the_last_done_line() { - let started = started(); - let beats: Vec = (1..=6).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - let mut capture: String = started[..5].iter().map(|s| format!("{s}\n")).collect(); - capture.push_str(&beats[0]); - capture.push_str("\n===READY===\n"); - for b in &beats[1..] { - capture.push_str(b); - capture.push('\n'); - } - let verdict = settle(&lines(&capture), &started).unwrap(); - assert_eq!(verdict.beats[0].t_ms, 2750); - assert_eq!(verdict.beats.len(), 4); - assert_eq!(window_beats(&lines(&capture), &started), 4); - - let mut capture: String = started[..5].iter().map(|s| format!("{s}\n")).collect(); - for b in &beats { - capture.push_str(b); - capture.push('\n'); - } - capture.push_str("===READY===\n"); - assert_eq!( - settle(&lines(&capture), &started), - Err(Refused::Unsettled { settled: 0, beats: 6 }) - ); - } - - #[test] - fn fewer_settled_beats_than_the_minimum_is_not_a_verdict() { - let beats: Vec = (1..=3).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - let capture = settled_capture(&beats); - assert_eq!(window_beats(&lines(&capture), &started()), 3); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::Unsettled { settled: 3, beats: 4 }) - ); - } - - /// What `CAPTURE_BEATS` buys over the floor: cut at `MIN_SETTLED` the - /// capture ends on cpu5's first absence, which is a blip and names nobody; - /// taken to `CAPTURE_BEATS` the second absence lands inside it. - #[test] - fn the_capture_carries_beats_past_the_floor_for_the_convicting_one() { - let mut beats: Vec = - (1..MIN_SETTLED as u64).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.push(beat(2000 + MIN_SETTLED as u64 * 250, 0xdf, 43, 250)); - let cut = settled_capture(&beats); - assert_eq!(window_beats(&lines(&cut), &started()), MIN_SETTLED); - assert_eq!(settle(&lines(&cut), &started()).unwrap().blips, 1); - - beats.extend( - (MIN_SETTLED as u64 + 1..=CAPTURE_BEATS as u64) - .map(|i| beat(2000 + i * 250, 0xdf, 43, 250)), - ); - let whole = settled_capture(&beats); - assert_eq!(window_beats(&lines(&whole), &started()), CAPTURE_BEATS); - assert_eq!( - settle(&lines(&whole), &started()), - Err(Refused::CpuMissing { - cpus: vec![5], - settled: CAPTURE_BEATS, - opened: started().len() + 1 - }) - ); - } - - #[test] - fn a_beat_with_an_unreadable_field_is_refused_by_line() { - let torn = "[kernel 2.5 cpu2] heartbeat: t=2.500s alive=8/8 mask=0xff ran=43 gap=0.2"; - let mut capture = settled_capture(&[]); - capture.push_str(torn); - capture.push('\n'); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::Unreadable(torn.to_string())) - ); - assert_eq!(millis("0.251s"), Some(251)); - assert_eq!(millis("12.000s"), Some(12_000)); - assert_eq!(millis("0.25s"), None); - assert_eq!(millis("0.251"), None); - } - - /// The `mask=` is 64 bits, so an `alive=` denominator at 64 or above is no - /// reading of it and neither is zero — and the refusal is the contract's, - /// not a shift overflow's. - #[test] - fn a_cpu_count_the_mask_cannot_carry_is_unreadable() { - for alive in ["8/64", "8/100", "8/4294967296", "0/0"] { - let wide = format!( - "[kernel 2.5 cpu2] heartbeat: t=2.500s alive={alive} mask=0xff ran=43 gap=0.250s" - ); - let mut capture = settled_capture(&[]); - capture.push_str(&wide); - capture.push('\n'); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::Unreadable(wide.clone())), - "{alive}" - ); - } - } - - /// The bound above is reached on its own, not stood in for by the - /// differing-`cpus` refusal: every beat here reads the same wide - /// `alive=8/64`, so `cpus` never differs from `beats[0].cpus` and that - /// refusal cannot fire. Without the bound `Beat::parse` would read `64` - /// and the eight bits `mask=0xff` never sets would convict cpus 8..64 - /// instead of refusing the capture as unreadable. - #[test] - fn a_cpu_count_the_mask_cannot_carry_is_unreadable_even_when_every_beat_agrees() { - let wide = |t_ms: u64| { - format!( - "[kernel {0}.{1:03} cpu2] heartbeat: t={0}.{1:03}s alive=8/64 mask=0xff ran=40 \ - gap=0.250s", - t_ms / 1000, - t_ms % 1000, - ) - }; - let mut capture: String = started().iter().map(|s| format!("{s}\n")).collect(); - for i in 0..=CAPTURE_BEATS as u64 { - capture.push_str(&wide(2000 + i * 250)); - capture.push('\n'); - } - let lines = lines(&capture); - let head = lines.iter().position(|l| l.contains("heartbeat: t=")).unwrap(); - assert_eq!(settle(&lines, &started()), Err(Refused::Unreadable(lines[head].to_string()))); - } - - /// One capture is one machine: a beat whose `alive=` denominator is not the - /// first's describes a different one, and a mask read against the wrong - /// width is a CPU invented or a CPU dropped. - #[test] - fn a_capture_whose_cpu_count_changes_is_unreadable() { - let odd = "[kernel 3.0 cpu2] heartbeat: t=3.000s alive=7/7 mask=0x7f ran=43 gap=0.250s"; - let mut beats: Vec = (1..=2).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.push(odd.to_string()); - beats.extend((1..=4).map(|i| beat(3000 + i * 250, 0xff, 43, 250))); - let capture = settled_capture(&beats); - assert_eq!(settle(&lines(&capture), &started()), Err(Refused::Unreadable(odd.to_string()))); - } - - /// The one table, held against the config it transcribes: a `[boot] start` - /// program it does not know is refused by name, here and not at the guest. - #[test] - fn a_program_the_done_table_does_not_know_is_refused() { - let known: Vec = DONE.iter().map(|(program, _)| (*program).to_string()).collect(); - assert_eq!(done_lines(&known).unwrap(), started()); - - let mut added = known.clone(); - added.push("sniffer".to_string()); - assert!(done_lines(&added).unwrap_err().contains("sniffer")); - - let mut dropped = known; - dropped.pop(); - assert!(done_lines(&dropped).is_err()); - } -} diff --git a/src/lib.rs b/src/lib.rs index 2ef6e4d30dc..cfb0436ce23 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -13,7 +13,6 @@ pub mod compiler; pub mod fingerprint; pub mod flags; pub mod forkcheck; -pub mod heartbeat; pub mod hostws; pub mod icmp; pub mod identity; diff --git a/src/redlist.rs b/src/redlist.rs index f941c3eaccd..4b92863e408 100644 --- a/src/redlist.rs +++ b/src/redlist.rs @@ -34,15 +34,12 @@ pub const DISABLED: &[Disabled] = &[ issue: "issues/build/the-console-input-path-can-stop-after-a-ps2-overflow.md", }, Disabled { test: "desktop_window_child", issue: "issues/kernel/desktop-window-child-freeze.md" }, - Disabled { test: "doom_sound_flood", issue: "issues/audio/doom-sound-flood-played-full-scale-once.md" }, Disabled { test: "handle_kill_policy", issue: "issues/kernel/handle-kill-policy-census-grew-one-sharedmem-on-two-nightlies.md", }, Disabled { test: "handle_transfer", issue: "issues/kernel/deferred-release-outlives-its-syscall.md" }, - Disabled { test: "hda_tone", issue: "issues/audio/hda-tone-phase-check.md" }, Disabled { test: "kill_while_blocked", issue: "issues/kernel/deferred-release-outlives-its-syscall.md" }, - Disabled { test: "latency_wake", issue: "issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md" }, Disabled { test: "quiesce_dump_holds_the_stopped", issue: "issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md", @@ -51,10 +48,6 @@ pub const DISABLED: &[Disabled] = &[ test: "quiesce_wakes_on_the_last_exit", issue: "issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md", }, - Disabled { - test: "sched_check_build", - issue: "issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md", - }, Disabled { test: "screen_fatal_halt", issue: "issues/boot-media/screen-fatal-halt-reds-on-ci-with-a-usb-storage-transport-break-during-boot.md", diff --git a/src/sourcegate.rs b/src/sourcegate.rs index ba7fdb0624a..f5b1b776f0c 100644 --- a/src/sourcegate.rs +++ b/src/sourcegate.rs @@ -1472,7 +1472,6 @@ const ARCH_RULES: &[PlaceRule] = &[ ("tests/toyos-rust-tests/src/bin/log_hold.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/mmap_prot.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/nmi_window_spin.rs", USERLAND_ASM), - ("tests/toyos-rust-tests/src/bin/panic_halts_first.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/partition_claimant.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/process_lifecycle.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/syscall_cost.rs", USERLAND_ASM), diff --git a/src/testargs.rs b/src/testargs.rs index 4eaaf0e4abc..a53149b537e 100644 --- a/src/testargs.rs +++ b/src/testargs.rs @@ -43,8 +43,8 @@ impl Shard { /// property a verdict depends on and the one the gates below hold. /// /// **`load` is the run's one accumulator, not this call's.** A suite that - /// partitions several pools — the parallel tasks, the serial tail, gate A's - /// configs — is one machine's wall clock either way, so the second pool has + /// partitions several pools — the parallel tasks and the serial tail — is one + /// machine's wall clock either way, so the second pool has /// to fill the bins the first left light. Starting each call from /// [`bins`](Self::bins) makes each partition good and their sum bad, and /// the imbalances add: measured over run `31377439504`'s twelve shards it @@ -150,13 +150,11 @@ declare_flags!(pub SUITE = { pub LIST = "--list", None; pub NOCAPTURE = "--nocapture", None; pub SHOW_OUTPUT = "--show-output", None; - pub AUDIO_GATE = "--audio-gate", Next; pub JOBS = "--jobs", Next; pub JOBS_SHORT = "-j", Next; pub HOST_SLOTS = "--host-slots", Next; pub HOST_BUILDS = "--host-builds", Next; pub SHARD = "--shard", Next; - pub SLOW_USB = "--slow-usb", None; pub NIGHTLY = "--nightly", None; /// The metal profile: the registrations that run on the T14, batched into /// images and judged off the log the stick came back with. @@ -214,20 +212,6 @@ pub fn parse(args: &[String]) -> Result, String> { .to_string(), ); } - if has(&METAL) && has(&AUDIO_GATE) { - return Err( - "--metal and --audio-gate are separate tiers on separate machines and cannot be \ - combined; run one at a time" - .to_string(), - ); - } - if has(&NIGHTLY) && has(&AUDIO_GATE) { - return Err( - "--nightly and --audio-gate are separate tiers and cannot be combined; run one \ - tier at a time" - .to_string(), - ); - } Ok(filter) } @@ -255,29 +239,15 @@ mod tests { fn a_flags_value_is_not_the_filter() { assert_eq!(parse_owned(&["--jobs", "4"]).unwrap(), None); assert_eq!(parse_owned(&["-j", "4"]).unwrap(), None); - assert_eq!(parse_owned(&["--audio-gate", "30"]).unwrap(), None); assert_eq!(parse_owned(&["--host-slots", "0"]).unwrap(), None); assert_eq!(parse_owned(&["--host-builds", "0"]).unwrap(), None); } - #[test] - fn nightly_and_audio_gate_are_refused_by_the_argv_validator() { - for argv in [ - vec!["--nightly", "--audio-gate", "30"], - vec!["--audio-gate=30", "--nightly"], - ] { - let refusal = parse_owned(&argv).unwrap_err(); - assert!(refusal.contains("--nightly"), "{refusal}"); - assert!(refusal.contains("--audio-gate"), "{refusal}"); - assert!(refusal.contains("cannot be combined"), "{refusal}"); - } - } - #[test] fn the_filter_is_the_word_that_is_nobodys_value() { assert_eq!(parse_owned(&["process_stats"]).unwrap().as_deref(), Some("process_stats")); assert_eq!( - parse_owned(&["--audio-gate", "30", "audio_tone", "--nocapture"]).unwrap().as_deref(), + parse_owned(&["--jobs", "4", "audio_tone", "--nocapture"]).unwrap().as_deref(), Some("audio_tone") ); assert_eq!( @@ -363,9 +333,9 @@ mod tests { assert_eq!(totals, vec![130, 130, 130, 130], "{totals:?}"); } - /// **One run is one accumulator.** The suite partitions three pools — the - /// parallel tasks, the serial tail, gate A's configs — and a shard runs all - /// three, so the second call has to fill the bins the first left light. Two + /// **One run is one accumulator.** The suite partitions two pools — the + /// parallel tasks and the serial tail — and a shard runs both, so the second + /// call has to fill the bins the first left light. Two /// pools of `[3 s, 1 s]` across two shards is the smallest case that tells /// the two apart: threaded, both shards take 4 s; from a fresh accumulator /// each time, the heavy item lands on shard 1 twice and the widest bin is @@ -430,8 +400,8 @@ mod tests { } /// Every `None` here is a default the run then takes in silence: `--jobs` - /// the built-in width, `--audio-gate` the thorough tier off, `--host-slots` - /// the host's own budget, `--host-builds` no budget at all. + /// the built-in width, `--host-slots` the host's own budget, `--host-builds` + /// no budget at all. #[test] fn a_flag_left_without_its_value_is_refused_by_name() { for flag in SUITE.0.iter().filter(|f| !matches!(f.value, Value::None | Value::Optional)) { @@ -483,7 +453,6 @@ mod tests { vec!["process_stats"], vec!["process_stats", "--nocapture"], vec!["--list"], - vec!["--audio-gate", "30"], vec!["--jobs", "4"], vec!["--host-slots", "0"], vec!["--host-builds", "0"], @@ -505,7 +474,5 @@ mod tests { fn a_readback_directory_alone_selects_no_tier() { let refusal = parse_owned(&["--metal-readback", "target/metal"]).unwrap_err(); assert!(refusal.contains("add --metal"), "{refusal}"); - let refusal = parse_owned(&["--metal", "--audio-gate", "30"]).unwrap_err(); - assert!(refusal.contains("cannot be combined"), "{refusal}"); } } diff --git a/src/tiers.rs b/src/tiers.rs index f0c57216b9f..7d33ea42a26 100644 --- a/src/tiers.rs +++ b/src/tiers.rs @@ -1,9 +1,9 @@ //! Which run a registered test belongs to. //! //! **The table is the registration.** Every row of `tests/toyos.rs`'s -//! `MACHINE_TESTS`, `SCREEN_TESTS` and `AUDIO_TESTS` carries its [`Tier`] or -//! does not compile, and the shared boot's discovered tests share one. Moving a -//! test between tiers is editing that one word. +//! `MACHINE_TESTS` and `SCREEN_TESTS` carries its [`Tier`] or does not compile, +//! and the shared boot's discovered tests share one. Moving a test between +//! tiers is editing that one word. //! //! The fast tier is what every plain `cargo test` runs; the nightly tier is //! `--nightly`, run by `.github/workflows/nightly.yml`. No pull request boots a diff --git a/tests/audio-baseline.toml b/tests/audio-baseline.toml deleted file mode 100644 index 69132d3d6d6..00000000000 --- a/tests/audio-baseline.toml +++ /dev/null @@ -1,495 +0,0 @@ -# Audio gate (gate A) baselines — one section per (test, smp) config. -# -# Every config must appear here. A missing section is a hard error, not a -# pass: an ungated config would go green by omission. -# -# What no config here covers: `/system/bin/tone` and tests/toyos-rust-tests/src/tone.rs -# are two implementations of the same generator and gate A captures only the -# second, so nothing in the suite ever hears the tone the owner judges the -# machine by. -# -# Two independent instruments per config. What each one is allowed to decide -# differs by tier, and TWO TIERS below is where that is settled: only silence -# that reached the device fails a single run. -# -# gaps Underrun histogram from the captured wav, keyed by gap -# length in whole device periods (2.902ms). An absent -# table is the strict zero-gap bar, which is what all four -# configs carry — the quality target is no dropouts, and -# recording a non-empty histogram here would be a decision -# to accept audible ones. -# max_wake_lat_us Worst single stats window of soundd waking later than a -# predicted DMA completion it armed a timer on. Waits that -# named no wake time — the idle path, and the one wake -# after a drain before the DLL re-locks — contribute -# nothing: nothing was due, so nothing can be late. -# **It is two delays added together, and mostly not -# soundd's.** The gate's per-run line splits it at the -# completion interrupt's own timestamp — `irq` is the -# device late, `pickup` is soundd late — and on the T14 -# under KVM `pickup` was 10-206us across 176 config-runs -# while the whole number ran 2509-6176us. -# `toyos_mixer::WorstWake` is where that split is argued; -# read it before taking a movement in this number for a -# scheduling change. -# drains Cycles that found the entire DMA pipeline free *and* -# could only have got there by soundd being late. -# The idle path empties the pipeline by design, and a -# device that retires faster than it plays empties it -# without anything having been due; neither counts. See -# the count site in soundd for the two conditions. -# underruns Periods submitted with no client audio behind them. -# -# VOID AS A TIMING SAMPLE UNDER QEMU 11.1.1. `.github/qemu-version` declares -# 11.1.1 and the sample below was measured on 11.1.0, so every comparison of a -# fresh `max_wake_lat_us` or `wakes` against it is cross-instrument and decides -# nothing, on the dev host and on CI alike. No re-record is taken for the move: -# timing verdicts belong on metal, and the dev host is not a quiet instrument. -# The harm statistics are the exception, because what they compare against is -# a sample of zeros — no dropout, no underrun — which is the quality bar on any -# instrument, so a harm red still counts. CI's `audio` shards neither re-sample -# nor skip: they compare a fresh KVM sample against this one, as they already -# did across the host and the accelerator -# (`issues/audio/gate-a-has-no-runner-baseline.md`), and now across the QEMU -# version as well; a timing red from them is read against this paragraph. -# -# RE-RECORDED 2026-08-15 on tree 4c7d809, justified by an instrument change: -# the dev host's QEMU moved 11.0.3 -> 11.1.0 (owner-approved upgrade; CI's -# runner had already moved and `.github/qemu-version` declares 11.1.0), and the -# QEMU version was measured to decide verdicts — so every comparison of a fresh -# 11.1.0 arm against the 11.0.3 sample was cross-instrument and void. The -# prior sample (2026-07-31, 17a6c88 — see git log for its full narrative, -# including the CQE fan-out A/B that justified it) described the old -# instrument. -# -# The recording was taken TWICE, back to back, 30 iterations per arm, and the -# second run is the sample below — its host-conditions line is the clean one -# (1-min load 1.2-2.1). The first run, whose load briefly spiked to 10.2, -# reproduces the second's shape, which is the control that says the shift is -# the instrument and not the spike: -# -# max_wake_lat_us medians, old -> run1 / run2: -# tone.smp1 6698 -> 9123 / 8972 UP ~34% -# tone.smp8 6658 -> 8626 / 9249 UP ~30-39% -# load.smp1 7042 -> 5813 / 5765 DOWN ~18% -# load.smp8 7134 -> 6121 / 6097 DOWN ~15% -# The no-load configs got slower and the load configs got -# faster on 11.1.0 — a reshuffle only an instrument change -# produces; a kernel regression does not speed up the -# harder case. -# harm null in BOTH runs: underruns all-zero, drains all-zero, -# dropouts 0/120 and 0/120, ceiling breaches 0/120 and -# 0/120. The quality bar holds on the new instrument. -# -# This recording also takes the `wakes` re-record the fourth drift instance -# below recorded as owed-but-not-taken: the mechanism that paragraph was -# waiting for is the instrument itself. -# -# Drift note, third instance: the BASE arm itself — zero suspend-series -# commits — failed the old sample on audio_tone.smp1 wakes (median 1011 -> -# 942, z=5.28), the third consecutive fewer-wakes-all-harm-clean failure of -# that statistic across sessions. Cross-batch drift on this host is real; -# only same-session A/B numbers mean anything (CLAUDE.md). -# -# Drift note, FOURTH instance, 2026-08-01, and this one is measured against a -# same-session control rather than argued. Two N=30 arms back to back (1026 s -# and 1024 s, quiet host), differing only in whether `esp_log::install()` runs: -# -# recorded esp_log ON esp_log OFF -# audio_tone.smp1 wakes 920 902 z=4.12 908 z=3.63 -# audio_tone.smp8 wakes 872 815 z=4.79 816 z=4.73 -# -# `wakes` reds in BOTH arms, at the same magnitude, with the suspect entirely -# removed — so it is drift, not the change under test. Harm null in both: -# dropouts 0/120 and 0/120, ceiling breaches 0/120 and 0/120, underruns and -# drains all-zero. Four consecutive sessions have now failed `wakes` downward -# with every harm measure clean, so a re-record of that statistic is owed — -# but NOT taken here, because no mechanism was identified and this file's -# protocol wants one. Read the fourth instance as evidence for the owner, not -# as licence. -# -# The one statistic that separated the arms was audio_tone.smp8 wake lateness: -# 6691 with the sink off, 7346 with it on (recorded 6658). That is the ESP -# log's residual flush cost on the idle path, not drift, and it is written up -# in `issues/boot-media/` rather than recorded here. -# -# Drift note, FIFTH instance, 2026-08-15, and the first one measured against -# the sample below rather than against a superseded one — so the instrument -# argument that closed the fourth instance does not apply to it. Two N=30 arms -# back to back (904 s and 907 s), differing only in the four soundd commits of -# `wt/toyos-soundfix`; BASE is `a0729cf`, the tree this sample was recorded on: -# -# recorded soundfix BASE a0729cf -# audio_tone.smp1 wakes 882 870 z=3.47 870 z=3.14 -# audio_tone.smp8 wakes 877 849 z=5.29 854 z=5.44 -# audio_tone_load.smp1 wake_lat 5765 6051 pass 6151 z=3.54 -# -# `wakes` reds downward in BOTH arms again, and the BASE arm reds one statistic -# MORE than the branch under test. Harm null in both: underruns 0/0/0 in all -# eight config-arms, dropouts 0/120 and 0/120, ceiling breaches 0/120 and -# 0/120, drains all-zero except a single run in BASE's audio_tone_load.smp8. -# That is five consecutive sessions of fewer-wakes-all-harm-clean, two of them -# now with a same-session control. -# -# What is new here is a candidate mechanism, which is what this file's protocol -# has been waiting for and what the fourth instance could not offer. The -# conditions differ in the one figure that is exact rather than lagging: -# -# recorded qemu 3-3, toyos-build 1-1, 1-min load 1.2-2.1 (median 1.6) -# soundfix qemu 1-1, toyos-build 1-2, 1-min load 1.0-8.6 (median 2.3) -# BASE qemu 3-3, toyos-build 2-2, 1-min load 0.9-2.3 (median 1.6) -# -# `wakes` counts iterations of a loop woken by completions and by a predicted -# timeout, and the pipeline retires in batches of 8 on this host in every run of -# all three: fewer wakes for the same ~1110 periods submitted is the host -# handing soundd more completions per wake. A guest sharing the host with two -# other guests is descheduled more often, which is more wakes, not fewer — so -# the *recorded* arm being the contended one is the direction that matches. -# Not established: the BASE arm above ran at `qemu 3-3` like the recording and -# still red, so competition alone does not close it either. Recorded as the -# next thing to test, not as the answer. -# -# NO RE-RECORD IS TAKEN HERE. Not by this branch and not for this statistic: -# the owner's ruling and this file both want a mechanism first, and a candidate -# is not one. -# -# WHAT THIS SAMPLE IS A SAMPLE OF, measured 2026-08-21 rather than argued. Every -# number below was taken on the dev host under cross-arch TCG, and the thorough -# tier compares a KVM runner's sample against it — two instruments, whichever -# runner it is. That was an argument before; the control it lacked has been run -# on the T14 (Intel i5-1135G7, KVM, the same QEMU 11.1.0) while that machine was -# still a runner. Four interleaved 15-iteration blocks on an idle T14 (A,B,A,B; no CI job -# container on the machine for any of the 240 boots; 1-min load 0.2-1.74), arm A -# = `960b96e3`, the tree this sample was recorded on, arm B = `53101d08`: -# -# max_wake_lat_us median recorded T14 arm A T14 arm B A vs B, n=30 -# audio_tone.smp1 8972 20314 19994 z=1.49 same -# audio_tone.smp8 9249 14069 4088 z=5.37 B faster -# audio_tone_load.smp1 5765 2764 4352 z=4.12 B slower -# audio_tone_load.smp8 6097 12676 3904 z=5.48 B faster -# -# **Arm A fails this file on the T14 and arm B does not.** Both of arm A's -# blocks red `audio_tone.smp1` against the sample below (median 8972 -> 20438 -# z=5.03, and -> 20144 z=4.67); both of arm B's pass. So the tree the sample was -# recorded on reds its own sample on that host, harder than `main` does — which -# is the negative control saying the level difference is the instrument and not -# a regression. The first readable T14 gate A run (32479089989) failed -# `audio_tone.smp1` at 8972 -> 17186, z=4.36; that verdict is this. -# -# Harm was null on both arms: dropouts 0/120 and 0/120, underruns 0 in all 240 -# config-runs, ceiling breaches 1/120 (arm A) and 0/120 (arm B). -# -# NO RE-RECORD IS TAKEN FOR THAT EITHER, and this sample must not be replaced by -# a T14 one: it is the dev host's, and the dev host still runs the fast tier. -# What a per-host baseline needs before it can be written — a schema that has a -# host dimension at all, and a T14 distribution that stops being bimodal — is in -# `issues/audio/gate-a-has-no-runner-baseline.md`. -# -# Why the counters are here at all: the wav is a rare-event detector. One 3s -# tone samples ~1100 device periods once per run, so it only fires when a -# dropout happens to land inside the tone — a 0-10% per-run event that cannot -# resolve a change in the failure rate. soundd's counters are non-zero on -# essentially every run and are direct evidence of the same stall, so they are -# the half of this gate with statistical power. They are scoped to the -# streaming phase: soundd zeroes them when the first client arrives and -# flushes them when the last one leaves, so no number here is diluted by the -# idle path (which since the suspend-on-idle series is a stopped device and a -# parked soundd — no wakes at all). -# -# THE PHYSICAL SCALE. The DMA pipeline holds TX_INFLIGHT_MAX = 8 buffers of -# 128 frames at 44.1kHz = 23.219ms. That is soundd's entire timing budget: -# wake later than one pipeline depth and every buffer has already drained and -# the device has run out of audio. `max_wake_lat_us` limits are therefore -# quantised to whole pipeline depths (1 pl = 23219us), so the number says -# something about the hardware rather than about a particular afternoon. -# -# HOW THE NUMBERS WERE PICKED. The ceilings were derived at the 2026-07-29 -# recording (30 serial runs per config, quiet host: one QEMU at a time, -# verified before every iteration; Apple M4 Pro, QEMU 11.0.2 under cross-arch -# TCG, 1-min load 4.2-6.1 recorded per run) and deliberately KEPT at the -# 2026-07-31 re-record: they are catastrophe detectors, the new maxima are -# lower everywhere, and lowering a ceiling to track an improving sample would -# turn the catastrophe detector into a second distributional test. They now -# sit 5.5-22.5x above the observed maxima. Derivation rule, applied uniformly: -# -# limit = 2x the observed maximum, rounded up to a round number -# -# These are max-of-window order statistics with heavy right tails. At n=30 the -# observed maximum sits near the 97th percentile, so a limit AT the maximum -# would false-red on the order of once every 30 runs per config — four configs -# per invocation would make the gate red more often than the defect it is -# watching. Doubling it puts the per-config false-red rate well under 1% while -# still failing on a doubling of the tail, which is the magnitude a scheduler -# regression that broke RT preemption or the audio IRQ path would produce. -# Since 2026-08-04 what a breach can redden is the thorough tier's pooled -# `ceiling_runs` rate rather than a `cargo test`, so this is now the false-red -# argument for the recorded 0/120 that rate is compared against. The derivation -# is unchanged and so is every number below. -# -# The tails were not physics. The previous sample's worst wake lateness on -# `audio_tone_load.smp1` went 26627us at n=15 to 92989us at n=30 — one more run -# more than tripled it — and that is why the rule has a doubling factor at all. -# It is kept, but the thing it was compensating for is gone: two instrument -# faults (824dd7d scoring idle sleeps that armed no timer against a stale DLL -# estimate; 7095046 counting connect-time pre-roll as drains) and one real -# defect (069d158). The same config now spans 6057-8250us, a ratio of 1.36 -# across 30 runs. Keep the factor anyway — it costs nothing and the next tail -# will not announce itself either. -# -# WHAT THESE NUMBERS ADMIT. The spec-derived bar — never wake later than the -# pipeline you are feeding — is met on all 120 config-runs. The medians span -# 5765-9249us across the four configs (a quarter to two-fifths of a pipeline -# depth) and the worst single run reaches 0.98 pl (22744us, on -# `audio_tone.smp1`) — under a whole pipeline depth, but barely, and the -# ceilings stay where they are so a tail that grows says so. `ceiling_runs` -# is 0 everywhere. The 54/57ms suspected-suspend tail pair did not reappear -# in either 11.1.0 run. -# -# The three paragraphs this replaced said the bar was NOT met on any config, -# blamed "mode-B client misses under single-CPU load", and described -# `underruns` as having a floor of ~4 from a stream-start transient. All three -# were wrong, and each was wrong the same way: a defective counter was being -# read as a property of the system. `underruns` is now 0 on all 120 runs. If a -# statement in this file ever again explains why a counter is *expected* to be -# non-zero, suspect the counter first. -# -# Re-record deliberately, never casually, and never to make a red run green. -# Every number above must survive being asked "why that value?". -# -# =========================================================================== -# TWO TIERS, AND WHAT EACH ONE CERTIFIES -# =========================================================================== -# -# FAST TIER — `cargo test`, one boot per config, every time. -# Certifies: the instrument is alive (tone present, dither present, no clicks, -# a stats window that exists at all), no counter sits on the wrong side of a -# physical bound, and this build does not *reproducibly* put silence on the -# wire. -# -# THE VERDICT IS HARM, and harm is silence that reached the device: a -# mid-tone gap in the capture, or a period soundd submitted with no client -# audio behind it (`underruns`). The ceilings above are measured on every run, -# printed with the run's counters, and fail nothing here. `drains` past its -# ceiling with an empty histogram and zero underruns is a pipeline that -# recovered before anyone could hear it, and one boot cannot say whether it -# recovers less often than it used to — that is a rate, and the tier below is -# the instrument for it. Owner's ruling, 2026-08-04. -# -# Harm is confirmed before it fails: a run showing any is re-booted once and -# only a second failure counts. The strict zero-gap bar is unchanged and -# applies to both boots — one Bernoulli trial against a 0-7% per-config rate -# is not a verdict, and the un-confirmed version measured 12.8% red per -# invocation on an unmodified tree. The 2-of-2 rule takes that to 0.67%. The -# first occurrence is still printed and its wav still kept. -# -# The direction this moved in is not "looser". `underruns` was previously -# judged against the ceiling above (12-70 depending on config), so 40 periods -# of silence on the wire passed; it is now judged against zero, which is what -# all 120 recorded runs measured. What stopped failing a single run is the -# counters that describe timing rather than output. -# What it CANNOT certify: any statement about a rate. One run is one sample. -# -# THOROUGH TIER — `cargo test --test toyos-build -- --audio-gate 30`, ~17 minutes. -# What the nightly runs. -# N iterations of all four configs; every per-run outcome becomes a rate or a -# distribution, and the fresh sample is compared against the RECORDED SAMPLE -# below — Mann-Whitney for the counters, Fisher exact for the yes/no -# outcomes. Not against a fitted constant: a threshold derived from 30 runs -# carries the sampling error of those 30 runs, and a one-sample test against -# it claims a confidence it does not have (a sign test against the recorded -# median of `max_wake_lat_us` has a nominal false-red rate of 0.07% and a -# real one near 1.5%, purely because the reference median moves by one order -# statistic). -# -# At N=30, measured against these distributions, it detects: -# wake lateness +25% 99.9% +20% 93% +10% 4% (missed) -# underruns +50% 100% +25% 94% -# wakes -5% 99.9% (completions batched = soundd ran late) -# dropout rate 10x 100% 5x 71% 3x 5% (missed) -# False-red on a clean tree: 0.25%, over 2000 invocations simulated from -# these samples. -# -# Read the dropout row honestly: the audible symptom is the WEAKEST -# instrument here and cannot be made strong. Separating a 3% dropout rate -# from a 7% one at this confidence needs ~600 runs per config — five hours -# per config. It stays in the gate because it is the only statistic that -# says "someone would have heard it"; the counters are what actually decide -# whether a stage regressed. -# -# WHY FIXED N AND NOT A SEQUENTIAL TEST. An SPRT on the dropout rate would -# accept a clean tree after ~17 iterations instead of 30, saving about a -# third of the wall clock. Two reasons not to: the run is also the artefact -# you re-baseline from, and a variable-length sample is not comparable to the -# one recorded here; and the accept rule would have to be combined across all -# 18 statistics, which is where the honesty of the alpha would go to die. -# What is taken from the sequential idea is the cheap half — fail-side -# curtailment. A count can only rise, so once one passes the threshold it -# would face at the full N, the verdict is already decided and the run stops. -# A badly broken tree fails in minutes; a clean one pays the full 17. -# -# ALPHA IS 0.001 PER TEST, and it is a constant in `tests/common/stats.rs`, -# not a field here. A gate whose alpha can be raised is a gate that will be -# raised on the afternoon it goes red. -# -# BOTH TIERS — the physical bound (`audio::check_physical`). Every number above -# answers "did this get worse?". Nothing here answered "is this possible?", -# and the two are different questions. Stage 6's thorough tier PASSED while -# carrying `wake_lat 153766519us` — 153 seconds inside a test the harness -# kills at 30 — because a rank test is robust to exactly one absurd value, -# and it would then have been pasted back into `max_wake_lat_us` below as -# data. The bound is the wall-clock life of the QEMU process, measured by the -# harness: soundd's whole life is inside it, so no duration it reports can -# exceed it, and no count of device periods can exceed the periods that fit -# in it. Nothing is recorded here to be tuned. A violation is reported as a -# broken instrument — fatal in both tiers, and in the thorough tier it aborts -# before the value joins the sample. It is NOT a ceiling: it sits two orders -# of magnitude above the ones above, because a ceiling admits values that are -# bad but real, and a stall of even three seconds should still be reported as -# the regression it is. -# -# =========================================================================== -# THE CONDITIONS A VERDICT WAS TAKEN UNDER -# =========================================================================== -# -# Every run of either tier prints the host's concurrent load beside its -# counters — `host: load <1min>/<5min>/<15min> qemu N toyos-build N`, sampled -# once per boot — and the thorough tier prints the range over its whole sample -# beside the numbers a re-record is pasted from. NOTHING BRANCHES ON ANY OF IT. -# CLAUDE.md's 2026-08-04 ruling stands unchanged: load is not an excuse, and a -# load-coincident red is investigated as a real defect of the pipeline, never -# re-run away as noise. -# -# What moved is the premise's context, not the ruling. It was made when the only -# load was the audio test itself; the tree now runs several agents in separate -# worktrees at once. The load that prompted this was measured at 49.9 on this -# 14-core host with twelve rustc/cargo processes and a single guest live; while -# this was written the same host read 6.6/8.1/14.3. A verdict taken there and -# one taken quiet are different measurements, and nothing recorded which was -# which. -# -# The three readings answer different questions. The load average lags — no -# figure of it resolves one ~15 s boot — but the competition is other worktrees' -# builds, which last minutes, and the triple's shape says whether the host was -# ramping up or winding down. `qemu` and `toyos-build` are exact and -# instantaneous, and they name the competition as ToyOS work; the harness's own -# knowledge says nothing here, because gate A already asserts -# `live_instances() == 0` and runs at width 1, so every intra-run fact is a -# constant. The run's own guest is up when the sample is taken, so -# `qemu 1 toyos-build 1` is the quiet reading. -# -# WHAT THE RECORDED SAMPLE'S OWN CONDITIONS ARE — and the two recordings this -# file rests on are not symmetric: -# -# 2026-07-29, the ceiling derivation: "1-min load 4.2-6.1" recorded per run. -# A real measurement, and the reason the 1-minute figure leads the triple: a -# fresh reading compares directly against it. -# -# 2026-08-15, THE SAMPLE BELOW — the one both tiers compare against — closes -# the asymmetry the previous recording left: its conditions are MEASURED, -# the line the gate itself printed over the recording run: -# -# host conditions over 120 runs: 1-min load 1.2-2.1 (median 1.6), -# qemu 3-3, toyos-build 1-1 -# -# So both arms of every future thorough-tier comparison rest on measured -# conditions, which is what the previous version of this paragraph asked the -# next re-record to deliver. -# -# =========================================================================== -# THE RECORDED SAMPLE -# =========================================================================== -# -# 30 serial invocations of the audio suite, tree 4c7d809, QEMU 11.1.0, the -# second of the two back-to-back recordings described at the top of this -# file; measured conditions: 1-min load 1.2-2.1 (median 1.6), qemu 3-3, -# toyos-build 1-1, no concurrent agents. -# -# Zero parse casualties this pass: all 120 config-runs produced a full stats -# line and a gap histogram (`gap_sample` = 30 everywhere), every capture -# analyzed `gaps: none`. -# -# `ceiling_runs` is 0 everywhere: no run in 120 breached the per-run limits -# above. That is the intended relationship between the tiers — the ceilings are -# a catastrophe detector, the distributions are the sensitive instrument. - -[audio_tone.smp1] -# wake_lat median 8972us, max 22744us (0.98 pl); drains 0; underruns 0. -# Dropouts: 0 of 30. The worst run sits just under one pipeline depth — the -# widest tail of the four configs on 11.1.0, and both recording runs put it -# here (run 1's max was 14953). Ceiling kept at the 2026-07-29 derivation, -# now 2.46x the observed maximum — still above the 2x rule. -max_wake_lat_us = 56000 # 2.41 pl -drains = 8 -underruns = 40 - -[audio_tone.smp1.sample] -gap_sample = 30 -gap_runs = 0 -ceiling_runs = 0 -max_wake_lat_us = [7052, 7362, 7600, 7619, 7852, 8186, 8192, 8335, 8364, 8401, 8529, 8615, 8759, 8867, 8914, 8972, 9432, 9457, 9953, 10003, 10044, 10117, 10591, 11154, 13162, 15077, 15945, 16184, 20251, 22744] -underruns = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] -wakes = [858, 860, 868, 870, 871, 871, 873, 874, 874, 874, 875, 876, 879, 880, 882, 882, 884, 884, 885, 886, 886, 888, 889, 890, 893, 893, 895, 896, 898, 898] -drains = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] - -[audio_tone.smp8] -# wake_lat median 9249us, max 15399us (0.66 pl); drains 0; underruns 0. -# Dropouts: 0 of 30. No run's worst wake reached a pipeline depth. -# drains stays at 4 rather than dropping to the observed 0: a bar of 0 on a -# counter whose whole sample is 0 would fail on the first single event, and one -# event is not evidence of anything. The ceilings are a catastrophe detector; -# the distributional test below is what has the power here. Ceiling kept — -# 3.12x the observed maximum. -max_wake_lat_us = 48000 # 2.07 pl -drains = 4 -underruns = 12 - -[audio_tone.smp8.sample] -gap_sample = 30 -gap_runs = 0 -ceiling_runs = 0 -max_wake_lat_us = [6320, 6672, 6929, 7228, 7233, 7299, 7461, 7613, 8035, 8331, 8863, 8904, 8923, 9034, 9205, 9249, 9271, 9314, 9410, 9456, 10856, 10897, 11076, 11445, 11779, 12292, 12333, 13590, 15369, 15399] -underruns = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] -wakes = [846, 857, 859, 859, 860, 863, 863, 864, 868, 869, 870, 870, 874, 874, 877, 877, 878, 878, 879, 881, 882, 885, 889, 890, 890, 892, 892, 898, 899, 905] -drains = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] - -[audio_tone_load.smp1] -# wake_lat median 5765us, max 7064us (0.30 pl); drains 0; underruns 0. -# Dropouts: 0 of 30. No run's worst wake reached a pipeline depth. -# The first-class case — glitch-free on one CPU under load — -# stays the flattest config, and 11.1.0 made it faster: lateness spans -# 5271-7064us, a 1.34x ratio. The 186000 ceiling is now ~26.3x the observed -# maximum and is left deliberately loose: it is the catastrophe detector, and -# the distributional test on the sample below is what actually holds this -# config. -max_wake_lat_us = 186000 # 8.01 pl -drains = 26 -underruns = 70 - -[audio_tone_load.smp1.sample] -gap_sample = 30 -gap_runs = 0 -ceiling_runs = 0 -max_wake_lat_us = [5271, 5298, 5301, 5305, 5315, 5328, 5328, 5333, 5341, 5347, 5427, 5430, 5444, 5506, 5526, 5765, 5867, 5945, 5946, 5948, 6023, 6063, 6074, 6154, 6182, 6274, 6304, 6335, 6841, 7064] -underruns = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] -wakes = [856, 860, 861, 862, 863, 863, 866, 867, 868, 871, 871, 872, 872, 873, 874, 874, 875, 875, 875, 875, 875, 876, 876, 879, 880, 881, 882, 883, 887, 887] -drains = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] - -[audio_tone_load.smp8] -# wake_lat median 6097us, max 17501us (0.75 pl); drains 0; underruns 0. -# Dropouts: 0 of 30. No run's worst wake reached a pipeline depth. 29 of 30 -# runs sit in 5122-6967us (1.36x); the one outlier at 17501us has run 1's -# counterpart (15430us in that arm's load.smp1), so a rare high single wake -# under load is a property of this instrument, not of one afternoon. -# Ceiling kept — 4.57x observed. -max_wake_lat_us = 80000 # 3.45 pl -drains = 6 -underruns = 32 - -[audio_tone_load.smp8.sample] -gap_sample = 30 -gap_runs = 0 -ceiling_runs = 0 -max_wake_lat_us = [5122, 5416, 5558, 5705, 5722, 5841, 5841, 5872, 5879, 5901, 5911, 5982, 5993, 6044, 6096, 6097, 6101, 6128, 6209, 6239, 6248, 6372, 6503, 6623, 6672, 6678, 6683, 6835, 6967, 17501] -underruns = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] -wakes = [882, 883, 887, 888, 889, 893, 893, 893, 893, 894, 894, 895, 895, 895, 897, 897, 897, 898, 900, 900, 900, 900, 901, 901, 902, 902, 903, 903, 904, 904] -drains = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] diff --git a/tests/common/audio.rs b/tests/common/audio.rs index a3bd2a69bf9..adac656e780 100644 --- a/tests/common/audio.rs +++ b/tests/common/audio.rs @@ -1,715 +1,93 @@ -//! Wav capture parsing and glitch analysis for the audio integration tests. -//! -//! The QEMU wav audiodev records a continuous timeline of what the -//! virtio-sound device played. Underruns show up as stretches of digital -//! silence inside an otherwise active signal; clicks show up as -//! sample-to-sample jumps no band-limited signal could produce. -//! -//! The capture timeline is NOT wall clock. QEMU's wav backend writes only -//! while the guest voice is enabled, so the file freezes across every -//! suspended stretch and splices the next resume directly -//! onto the last stopped sample. Verified empirically: 25s of wall clock with -//! the stream stopped adds zero PCM bytes. Consequence: `analyze` reports an -//! underrun for ANY two signal regions in one capture, at ANY wall-clock gap -//! between them — the spliced silence (drain tail + resume prime) always -//! exceeds `MIN_GAP_SECS`, and `NEAR_SECS` is a proximity window into -//! adjacent *samples*, not wall time, so it can never exonerate the gap. A -//! test that plays two tones in one boot will always go red against a -//! zero-gap baseline; keep one signal region per capture. +//! soundd judged off the log the T14's stick came back with. Audio is judged on +//! metal and nowhere else: no QEMU guest test plays it. -use std::collections::BTreeMap; -use std::fs; -use std::path::Path; -use std::time::{Duration, Instant}; +use super::serial::Serial; -use super::qemu::{self, BootOptions, QemuInstance}; - -/// Largest magnitude a *silent* mix can reach on the wire, in LSB. -/// -/// Derivation from soundd's dither generator (`userland/soundd/src/main.rs`): -/// `Xorshift32::next()` returns `state / 2^32 - 0.5`, i.e. a uniform draw in -/// `[-0.5, +0.5]` LSB; TPDF dither sums two independent draws, so -/// `|dither| <= 1.0` LSB exactly. A period no client covered leaves the f32 -/// mix bus at exactly `0.0`, so the sample written to the DMA buffer is -/// `round(dither)` — one of `{-1, 0, +1}`. Hence `|s| <= 1` *is* digital -/// silence, and the bound is tight: `P(|s| = 1) = 0.25`. -/// -/// Testing `s == 0` instead would be a detector that only works against a -/// truncating quantizer, which is a defect, not a property to rely -/// on: with a correct quantizer 75% of silent samples are 0, so the longest -/// run of exact zeros in 4M silent samples measures 47 — well under the -/// `MIN_GAP_SECS` floor of 88. Such a detector reports "no dropouts" forever. -/// -/// The band is far too narrow to swallow the 440 Hz test tone: at amplitude -/// 16000 the tone slews ~1000 LSB per sample through its zero crossing, so at -/// most one sample per crossing lands inside it. -const SILENCE_MAX: i32 = 1; -/// Silent runs shorter than this are ignored (the test tone dips through the -/// silence band for a single sample at each zero crossing). -const MIN_GAP_SECS: f64 = 0.002; -/// A silent run only counts as an underrun if there is signal within this -/// window on BOTH sides — i.e. it interrupts active playback. -const NEAR_SECS: f64 = 0.25; -/// Amplitude above which a sample counts as signal rather than noise floor. -const SIGNAL_THRESHOLD: i32 = 500; -/// A single-sample jump larger than this is a click: the 440Hz test tone has -/// a max per-sample delta of ~1.1k at 44.1kHz, and any sane audio is -/// band-limited far below this. -const CLICK_DELTA: i32 = 8000; -/// Device period: 512 period_bytes at 44.1kHz stereo 16-bit = 128 frames -/// = 2.902ms (`toyos_abi::virtio_sound::PERIOD_BYTES`, and the HDA stub's is the -/// same number for the same reason). Underruns are period-quantized — the device -/// plays silence one period at a time — so gap lengths are reported in whole -/// periods. -pub const PERIOD_SECS: f64 = 128.0 / 44100.0; - -pub struct Wav { - pub sample_rate: u32, - pub channels: u16, - /// Channel 0 only — soundd mixes identical data to all channels. - pub mono: Vec, -} - -pub struct SilentRun { - pub start: usize, - pub len: usize, -} - -pub struct Click { - pub index: usize, - pub from: i32, - pub to: i32, -} - -pub struct Analysis { - /// Mid-signal silent runs >= MIN_GAP_SECS: underruns. - pub underruns: Vec, - /// Hard discontinuities not at the edges of counted zero runs. - pub clicks: Vec, - /// Samples with amplitude above SIGNAL_THRESHOLD. - pub active_samples: usize, - pub peak: i32, - /// Fraction of non-zero samples in the capture's longest silent stretch — - /// the detector's own precondition, measured. TPDF dither into a - /// round-to-nearest quantizer puts 25% of silent samples at ±1; a - /// truncating quantizer puts 0% there and collapses this detector's band - /// back onto `s == 0`, at which point the gate passes while measuring - /// nothing. `None` when the capture has no silent stretch to judge. - pub dither_ratio: Option, -} - -/// Floor on `Analysis::dither_ratio`. The expected value is 0.25; anything -/// this far below it means the dither is gone, not that it got unlucky -/// (over the ~6000-sample stretches these captures contain, the sampling -/// error on 0.25 is under 0.01). -pub const MIN_DITHER_RATIO: f64 = 0.10; - -/// Parse a 16-bit PCM RIFF wav. The QEMU wav backend leaves the RIFF/data -/// size fields at 0 until clean shutdown, so sizes are advisory: a zero data -/// size means "read to EOF". -pub fn parse_wav(path: &Path) -> Result { - let bytes = fs::read(path).map_err(|e| format!("read {}: {e}", path.display()))?; - if bytes.len() < 12 || &bytes[0..4] != b"RIFF" || &bytes[8..12] != b"WAVE" { - return Err(format!("{}: not a RIFF/WAVE file", path.display())); - } - - let mut channels: Option = None; - let mut sample_rate: Option = None; - let mut data: Option<&[u8]> = None; - - let mut pos = 12; - while pos + 8 <= bytes.len() { - let id = &bytes[pos..pos + 4]; - let size = u32::from_le_bytes(bytes[pos + 4..pos + 8].try_into().unwrap()) as usize; - let body_start = pos + 8; - match id { - b"fmt " => { - let fmt = bytes - .get(body_start..body_start + 16) - .ok_or("truncated fmt chunk")?; - let audio_format = u16::from_le_bytes(fmt[0..2].try_into().unwrap()); - let bits = u16::from_le_bytes(fmt[14..16].try_into().unwrap()); - if audio_format != 1 || bits != 16 { - return Err(format!( - "unsupported wav format: audio_format={audio_format} bits={bits}" - )); - } - channels = Some(u16::from_le_bytes(fmt[2..4].try_into().unwrap())); - sample_rate = Some(u32::from_le_bytes(fmt[4..8].try_into().unwrap())); - pos = body_start + size; - } - b"data" => { - let end = if size == 0 || body_start + size > bytes.len() { - bytes.len() - } else { - body_start + size - }; - data = Some(&bytes[body_start..end]); - pos = end; - } - other => { - return Err(format!( - "unexpected wav chunk {:?} — QEMU writes only fmt+data", - String::from_utf8_lossy(other) - )); - } - } - } - - let channels = channels.ok_or("wav has no fmt chunk")?; - let sample_rate = sample_rate.ok_or("wav has no fmt chunk")?; - let data = data.ok_or("wav has no data chunk")?; - if channels == 0 || sample_rate == 0 { - return Err(format!("degenerate wav format: {channels}ch {sample_rate}Hz")); - } - - let frame_bytes = channels as usize * 2; - let mono = data - .chunks_exact(frame_bytes) - .map(|frame| i16::from_le_bytes(frame[0..2].try_into().unwrap()) as i32) - .collect(); - - Ok(Wav { - sample_rate, - channels, - mono, - }) +/// The text of `log` from the kernel's record of `job`'s spawn up to the next +/// test binary's, which is the window soundd's lines about that job land in. +fn job_window<'a>(log: &'a str, job: &str) -> Result<&'a str, String> { + let head = format!("spawn: /system/bin/{job} "); + let at = log.find(&head).ok_or_else(|| format!("no `{head}` record: {job} never ran"))?; + let rest = &log[at + head.len()..]; + Ok(&rest[..rest.find("spawn: /system/bin/test_rs_").unwrap_or(rest.len())]) } -pub fn analyze(wav: &Wav) -> Analysis { - let mono = &wav.mono; - let rate = wav.sample_rate as f64; - let min_gap = (MIN_GAP_SECS * rate) as usize; - let near = (NEAR_SECS * rate) as usize; - - let sig_runs = silent_runs(mono, min_gap); - - let has_signal = |range: &[i32]| range.iter().any(|&s| s.abs() > SIGNAL_THRESHOLD); - let underruns = sig_runs - .iter() - .filter(|run| { - let left = &mono[run.start.saturating_sub(near)..run.start]; - let end = run.start + run.len; - let right = &mono[end..(end + near).min(mono.len())]; - !left.is_empty() && !right.is_empty() && has_signal(left) && has_signal(right) - }) - .map(|run| SilentRun { - start: run.start, - len: run.len, - }) - .collect(); - - // Jumps at the edges of counted silent runs are the underruns themselves, - // not separate clicks. - let mut run_edges = std::collections::HashSet::new(); - for run in &sig_runs { - if run.start > 0 { - run_edges.insert(run.start - 1); - } - run_edges.insert(run.start + run.len - 1); - run_edges.insert(run.start + run.len); - } - let clicks = mono - .windows(2) - .enumerate() - .filter(|(i, w)| { - (w[1] - w[0]).abs() > CLICK_DELTA - && !run_edges.contains(i) - && !run_edges.contains(&(i + 1)) - }) - .map(|(i, w)| Click { - index: i, - from: w[0], - to: w[1], - }) - .collect(); - - // The capture's leading and trailing silence are the longest silent runs - // and are not underruns, so this measures the quantizer, not the glitch. - let dither_ratio = sig_runs.iter().max_by_key(|r| r.len).map(|run| { - let span = &mono[run.start..run.start + run.len]; - span.iter().filter(|&&s| s != 0).count() as f64 / span.len() as f64 - }); - - Analysis { - underruns, - clicks, - active_samples: mono.iter().filter(|s| s.abs() > SIGNAL_THRESHOLD).count(), - peak: mono.iter().map(|s| s.abs()).max().unwrap_or(0), - dither_ratio, - } -} - -/// The test tone's frequency, which the phase check below has to know and -/// `tests/toyos-rust-tests/src/tone.rs` states. -pub const TONE_HZ: f64 = 440.0; - -/// Where the captured tone stops being one sine. -/// -/// The harm this exists for: **a cyclic DMA engine replays a period nobody -/// refilled**, and a repeat is audible harm that -/// [`analyze`]'s gap detector cannot see — the samples are not silent and the -/// seam is not a large enough single-sample jump to be a click. The -/// zero-on-complete rule is what stops a repeat happening, and it is a design -/// promise; this is the measurement beside it. -/// -/// A sampled sinusoid obeys `x[n+1] = 2·cos(ω)·x[n] − x[n−1]` exactly, so one -/// pass over the capture tests the whole signal for phase continuity with no -/// transform. A replayed 128-frame period is 1.28 cycles of 440 Hz, so the -/// tone re-enters 0.28 of a cycle out and breaks the recurrence by thousands -/// of LSB. -/// -/// The tolerance covers what a correct capture does contain: TPDF dither at -/// ±1 LSB and quantization. Only the region between the first and last strong -/// sample is examined. -pub fn phase_breaks(wav: &Wav) -> Vec { - const TOLERANCE: f64 = 400.0; - let k = 2.0 * (2.0 * std::f64::consts::PI * TONE_HZ / wav.sample_rate as f64).cos(); - let mono = &wav.mono; - let (Some(first), Some(last)) = ( - mono.iter().position(|&s| s.abs() > SIGNAL_THRESHOLD), - mono.iter().rposition(|&s| s.abs() > SIGNAL_THRESHOLD), - ) else { - return Vec::new(); - }; - (first + 1..last) - .filter(|&n| { - let predicted = k * mono[n] as f64 - mono[n - 1] as f64; - (predicted - mono[n + 1] as f64).abs() > TOLERANCE - }) - .collect() -} - -/// The captured tone's pitch, in Hz, measured off the wav. -/// -/// The rate the device *plays* at is not a fact any guest counter can state: -/// soundd generates 44100 frames per second of content and the engine consumes -/// them at whatever `SDnFMT` and the codec's `Set Converter Format` agreed on, -/// so a wrong rate field is a stream that is correct in every buffer and comes -/// out at the wrong speed. Nothing else in this file can see that — -/// [`phase_breaks`] tolerates it (an 8.8% pitch error perturbs the recurrence -/// by ~12 LSB against a 400 LSB tolerance) and the gap detector is blind to it. -/// -/// Schmitt-triggered zero crossings over the strong region, which needs no -/// transform and is exact for one sine: the dither cannot cross a ±500 LSB -/// hysteresis band and the count is two per cycle. -pub fn dominant_hz(wav: &Wav) -> Option { - let mono = &wav.mono; - let first = mono.iter().position(|&s| s.abs() > SIGNAL_THRESHOLD)?; - let last = mono.iter().rposition(|&s| s.abs() > SIGNAL_THRESHOLD)?; - let mut high = mono[first] > 0; - let mut crossings = 0u32; - let (mut start, mut end) = (None, first); - for (n, &s) in mono.iter().enumerate().take(last + 1).skip(first) { - if (high && s < -SIGNAL_THRESHOLD) || (!high && s > SIGNAL_THRESHOLD) { - high = !high; - crossings += 1; - start.get_or_insert(n); - end = n; - } - } - // Between the first and last crossing, so the partial cycles at each end - // are outside the measurement rather than rounded into it. - let span = end.checked_sub(start?)?; - if crossings < 3 || span == 0 { - return None; - } - Some((crossings - 1) as f64 * wav.sample_rate as f64 / (2.0 * span as f64)) -} - -/// The complaint, when the capture did not come back at [`TONE_HZ`]. -/// -/// The band is half the distance between the two rates an HDA stream format can -/// name — 44.1 and 48 kHz are 8.8% apart — which is the coarsest error this can -/// be asked about and the widest band that still separates every pair the field -/// can express. Nothing narrower is wanted: this is a rate check, not a -/// frequency meter, and the estimator is a crossing count over a ramped tone. -pub fn wrong_pitch(wav: &Wav) -> Option { - const TOLERANCE: f64 = 0.044; - let Some(hz) = dominant_hz(wav) else { - return Some("the capture has no tone to measure the pitch of".to_string()); - }; - if (hz - TONE_HZ).abs() <= TONE_HZ * TOLERANCE { - return None; - } - Some(format!( - "the capture came back at {hz:.1} Hz for a {TONE_HZ} Hz tone — the device consumed the \ - buffers at {:+.1}% of the rate soundd generated them for", - (hz / TONE_HZ - 1.0) * 100.0, - )) +/// The tone client played to its end through the HDA controller soundd drives +/// itself, and soundd never fell back to the null sink. +pub fn tone_on_metal(log: &Serial) -> Result<(), String> { + log.must_say("soundd: hda path configured in")?; + log.must_not_say(NULL_SINK)?; + Ok(()) } -/// Underrun histogram keyed by gap length in device periods (rounded, -/// min 1): `gaps[n]` = number of mid-signal silent runs of ~n×2.902ms. This is -/// the unit gate A's thorough tier compares against the recorded sample in -/// `tests/audio-baseline.toml`. -pub fn gap_histogram(analysis: &Analysis, sample_rate: u32) -> BTreeMap { - let mut gaps = BTreeMap::new(); - for run in &analysis.underruns { - let secs = run.len as f64 / sample_rate as f64; - let n = (secs / PERIOD_SECS).round().max(1.0) as u32; - *gaps.entry(n).or_insert(0u32) += 1; +/// soundd's own word for the sink it took when the machine has none. +const NULL_SINK: &str = "soundd: no audio device, presenting a null sink"; + +/// The T14's panic, staged on its own HDA ring: a client that stops producing +/// for longer than the DMA ring takes to come round. The engine replays every +/// buffer it completes, so soundd has to fill the periods the client did not +/// cover (`underruns`) and may hold none of them back (`deferred`), across a +/// suspend and a resume. +pub fn client_stall_on_metal(log: &Serial) -> Result<(), String> { + log.must_not_say("repeated completion for free buffer")?; + let window = job_window(log.text(), "test_rs_hda_client_stall")?; + let resumes = window.matches("soundd: resumed").count(); + if resumes < 2 { + return Err(format!( + "soundd resumed {resumes} time(s) — the second stream did not find a suspended \ + daemon, so nothing here tests a resume:\n{window}" + )); } - gaps -} - -/// Render a histogram as e.g. `total 3 [1p×2 4p×1]`, or `none`. -pub fn format_histogram(gaps: &BTreeMap) -> String { - if gaps.is_empty() { - return "none".to_string(); + if !window.contains("soundd: wakes=") { + return Err(format!("soundd reported no stats window while the client ran:\n{window}")); } - let total: u32 = gaps.values().sum(); - let entries: Vec = gaps.iter().map(|(n, c)| format!("{n}p×{c}")).collect(); - format!("total {total} [{}]", entries.join(" ")) -} - -/// No-regression gate against a recorded baseline histogram: neither the -/// total gap count nor the longest gap class may exceed the baseline. An -/// empty baseline is the strict zero-gap gate. -pub fn check_gap_regression( - measured: &BTreeMap, - baseline: &BTreeMap, -) -> Result<(), String> { - let m_total: u32 = measured.values().sum(); - let b_total: u32 = baseline.values().sum(); - if m_total > b_total { + if sum_field(window, "underruns") == 0 { return Err(format!( - "underrun regression: {m_total} gaps vs baseline {b_total}" + "soundd filled no period the stalled client had not covered, so this boot staged \ + nothing:\n{window}" )); } - let m_max = measured.keys().next_back().copied().unwrap_or(0); - let b_max = baseline.keys().next_back().copied().unwrap_or(0); - if m_max > b_max { + let deferred = sum_field(window, "deferred"); + if deferred != 0 { return Err(format!( - "underrun regression: longest gap {m_max} periods vs baseline {b_max}" + "soundd deferred {deferred} period(s) on a ring that replays every one and completes \ + it again, which is the panic this exists for:\n{window}" )); } Ok(()) } -/// The DMA pipeline depth: `TX_INFLIGHT_MAX` = 8 buffers of one device period. -/// This is soundd's entire timing budget — wake later than this and every -/// buffer has already drained, so the device has run out of audio to play. -pub const PIPELINE_DEPTH_US: u64 = (8.0 * PERIOD_SECS * 1e6) as u64; - -/// One `soundd: wakes=...` stats line. soundd emits one every 2s, but only -/// while it has clients, so every line describes streaming, not idle. -#[derive(Debug, Clone, Copy)] -pub struct SounddWindow { - pub wakes: u32, - pub completions: u32, - pub submitted: u32, - pub underruns: u32, - pub drains: u32, - pub max_wake_lat_us: u64, - pub max_batch: u32, - pub clients: u32, - /// The worst wake taken apart, as `toyos_mixer::WorstWake` describes it. - /// `worst_irq_late_us + worst_pickup_us == max_wake_lat_us`, up to the - /// truncation each half takes on its way to microseconds. - pub worst: WorstWake, - /// Wakes in this window a whole device period or more past their grid - /// point — how many stalls the maximum is the maximum *of*. - pub late_wakes: u32, -} - -/// soundd's decomposition of its worst wake, carried through the harness so a -/// per-run line can say *which* half the number is. -#[derive(Debug, Default, Clone, Copy)] -pub struct WorstWake { - pub irq_late_us: u64, - pub pickup_us: u64, - pub empty: u32, - pub batch: u32, -} - -/// Worst/total over every stats window of one run. -#[derive(Debug, Default, Clone, Copy)] -pub struct SounddCounters { - pub windows: usize, - /// Worst single-window wake lateness — the sharpest instrument here. - pub max_wake_lat_us: u64, - /// The decomposition belonging to *that* window's worst wake. Taken from - /// the window that set the maximum rather than maximised on its own: two - /// independently-worst halves describe a wake that never happened. - pub worst: WorstWake, - /// Cycles that found the whole DMA pipeline free. - pub drains: u32, - /// Periods submitted with no client audio behind them: silence that - /// actually went on the wire while a client was streaming. - pub underruns: u32, - pub submitted: u32, - pub wakes: u32, - /// Summed over the run's windows: how many wakes were a whole period or - /// more late, which is what separates one stall from a thousand. - pub late_wakes: u32, - pub max_batch: u32, -} - -/// The two numbers this boot drew for its clocks, off the kernel's own boot -/// lines: the TSC period against the HPET (`kernel/src/clock.rs`) and the LAPIC -/// timer's tick rate against that (`kernel/src/arch/x86_64/apic.rs`). -/// -/// **They are here because they are the only per-boot draws that scale every -/// armed timer for the boot's whole life**, which is the shape -/// `issues/audio/t14-wake-lateness-is-bimodal-per-boot.md` is looking for: a -/// wake latency that is one of two values, decided at boot and steady inside -/// it, cannot come from anything re-decided per wake. Both calibrations are -/// busy-wait windows on a virtual machine, so both are exactly the kind of -/// number a host that stalls the guest mid-window would move. -/// -/// Printed, never asserted: what a correct pair looks like on a given host is -/// not something this harness knows, and a threshold nobody measured is the -/// problem `tests/audio-baseline.toml` exists to avoid. -pub fn boot_clocks(boot_log: &str) -> String { - let field = |marker: &str, upto: char| { - boot_log.find(marker).map(|at| { - let rest = &boot_log[at + marker.len()..]; - let end = rest.find(upto).unwrap_or(rest.len()); - rest[..end].trim().to_string() - }) - }; - format!( - "tsc {} lapic {}", - field("TSC: ", ' ').unwrap_or_else(|| "?".into()), - field("LAPIC timer: ", '\n').unwrap_or_else(|| "?".into()), - ) -} - -/// Kernel logging shares the virtio-console with userspace and is not -/// line-atomic, so a kernel message lands wherever it lands — including -/// mid-word inside soundd's stats line, which pushes that line's tail onto -/// the next serial line. A kernel message always runs from `[kernel ` to the -/// end of its line, so deleting exactly that span splices the interrupted -/// line back together and leaves standalone kernel lines simply removed. -fn strip_kernel_logging(serial: &str) -> String { - let mut out = String::with_capacity(serial.len()); - let mut rest = serial; - while let Some(start) = rest.find("[kernel ") { - out.push_str(&rest[..start]); - rest = match rest[start..].find('\n') { - Some(nl) => &rest[start + nl + 1..], - None => "", - }; +/// Two clients through soundd, and what soundd said about each leaving: every +/// removal names a departure soundd established, and none claims a death. +pub fn departures_on_metal(log: &Serial) -> Result<(), String> { + const CLIENTS: usize = 2; + let window = job_window(log.text(), "test_rs_null_sink_client_exits")?; + let problems = check_departures(window, CLIENTS); + if problems.is_empty() { + return Ok(()); } - out.push_str(rest); - out + Err(format!("{}\n{window}", problems.join("\n"))) } -const STATS_MARKER: &str = "soundd: wakes="; -/// soundd's stats fields, in the order it prints them. -const STATS_KEYS: [&str; 13] = [ - "wakes", - "completions", - "submitted", - "underruns", - "drains", - "max_wake_lat_us", - "max_batch", - "clients", - // `deferred` and `starve_max` sit between these and `clients` on the wire - // and are read by nothing here; the scan is forward-only from the previous - // key, so a printed field this list omits is simply stepped over. - "worst_irq_late_us", - "worst_pickup_us", - "worst_empty", - "worst_batch", - "late_wakes", -]; - -/// Read `key=` at or after `from`, tolerating a foreign line spliced -/// in between. Any writer sharing the console can land in the middle of the -/// value — the kernel is stripped beforehand, but the tone client's own -/// `println!` does it too — and such a write always ends at a newline, so -/// when the value is interrupted it resumes on the following line. -fn stats_field(window: &str, key: &str, from: usize) -> Option<(u64, usize)> { - let pat = format!("{key}="); - let mut at = from + window[from..].find(&pat)? + pat.len(); - loop { - let digits: String = window[at..].chars().take_while(|c| c.is_ascii_digit()).collect(); - if !digits.is_empty() { - return Some((digits.parse().ok()?, at)); - } - at += window[at..].find('\n')? + 1; - } -} - -/// Pull soundd's stats windows out of a serial capture. An unreadable window -/// is an error rather than a skip: silently dropping one would under-count -/// `drains` and `underruns`, which is a gate passing because it failed to -/// look. -pub fn parse_soundd_counters(serial: &str) -> Result { - let text = strip_kernel_logging(serial); - // A window's fields can be split across lines, so it extends to the next - // window marker rather than to the next newline. - let starts: Vec = text.match_indices(STATS_MARKER).map(|(i, _)| i).collect(); - let mut out = SounddCounters::default(); - for (n, &start) in starts.iter().enumerate() { - let end = starts.get(n + 1).copied().unwrap_or(text.len()); - let window = &text[start..end]; - let mut vals = [0u64; STATS_KEYS.len()]; - let mut cursor = 0; - for (i, key) in STATS_KEYS.iter().enumerate() { - let (v, at) = stats_field(window, key, cursor).ok_or_else(|| { - format!("unreadable soundd stats window (no {key}=): {window:?}") - })?; - vals[i] = v; - cursor = at; - } - let w = SounddWindow { - wakes: vals[0] as u32, - completions: vals[1] as u32, - submitted: vals[2] as u32, - underruns: vals[3] as u32, - drains: vals[4] as u32, - max_wake_lat_us: vals[5], - max_batch: vals[6] as u32, - clients: vals[7] as u32, - worst: WorstWake { - irq_late_us: vals[8], - pickup_us: vals[9], - empty: vals[10] as u32, - batch: vals[11] as u32, - }, - late_wakes: vals[12] as u32, - }; - out.windows += 1; - // The decomposition travels with the maximum it decomposes: `>=` so - // the first window still sets one, and so a later window that ties - // hands over its own halves rather than leaving stale ones behind. - if w.max_wake_lat_us >= out.max_wake_lat_us { - out.max_wake_lat_us = w.max_wake_lat_us; - out.worst = w.worst; - } - out.max_batch = out.max_batch.max(w.max_batch); - out.drains += w.drains; - out.underruns += w.underruns; - out.submitted += w.submitted; - out.wakes += w.wakes; - out.late_wakes += w.late_wakes; - } - Ok(out) -} - -/// Per-config ceilings on soundd's counters. Every number is justified in -/// `tests/audio-baseline.toml`; there are no defaults, because an unjustified -/// threshold is the same problem as an unmeasured baseline. -#[derive(Debug, Clone, Copy)] -pub struct CounterLimits { - pub max_wake_lat_us: u64, - pub drains: u32, - pub underruns: u32, -} - -/// Which of this config's per-run ceilings this run sits outside, one message -/// each. Unlike the wav histogram — a rare-event detector that samples ~1000 -/// periods once per run — these counters are non-zero on nearly every run, so -/// the *rate* at which they breach can resolve a change the histogram cannot. -/// -/// A breach is not by itself audible: a pipeline that drained and recovered put -/// no silence on the wire. So this decides nothing on its own — the thorough -/// tier counts breaches and compares the rate, and the fast tier prints them -/// and judges harm. -pub fn check_counters(counters: &SounddCounters, limits: &CounterLimits) -> Vec { - let mut problems = Vec::new(); - if counters.max_wake_lat_us > limits.max_wake_lat_us { - problems.push(format!( - "wake lateness {}us > limit {}us ({:.1} vs {:.1} pipeline depths)", - counters.max_wake_lat_us, - limits.max_wake_lat_us, - counters.max_wake_lat_us as f64 / PIPELINE_DEPTH_US as f64, - limits.max_wake_lat_us as f64 / PIPELINE_DEPTH_US as f64, - )); - } - if counters.drains > limits.drains { - problems.push(format!( - "pipeline drains {} > limit {}", - counters.drains, limits.drains - )); - } - if counters.underruns > limits.underruns { - problems.push(format!( - "client underruns {} > limit {} ({} periods submitted total)", - counters.underruns, limits.underruns, counters.submitted - )); - } - problems -} - -/// Structural suspend assertions, per-run and yes/no: the device stream -/// must start only for a client and must be stopped — with soundd suspended -/// and silent — once the last client is gone. TCG-immune, so a violation is -/// categorical, never a rare event to be averaged. -/// -/// Positions are byte offsets in the RAW serial. That is sound because every -/// pattern below lands as one atomic chunk on the shared console: each pattern -/// sits inside a single format piece of its `eprintln!`, so one `write` syscall -/// carries it. Writers interleave BETWEEN chunks — whole foreign lines can land -/// inside a soundd line — but never inside these patterns, and chunk order is -/// emission order for soundd's mix thread, which emits every marker here. All -/// four are soundd's own since H3 moved the driver into it; the two stream -/// markers used to come from the kernel, from inside the submit syscall, and -/// the order they appear in is unchanged. -/// -/// `serial` is expected to carry `qemu.boot_log()` prepended ahead of the -/// test window, so a restored boot prime — the exact code deleted in -/// 465bc22, which would open the voice, play 8 periods, drain and suspend -/// entirely before ===TEST_START — lands inside these patterns rather than -/// in a discarded prefix. `audio_idle_suspend` reads `result.serial` alone -/// and stays blind to that window; this function is where it is caught. -pub fn check_suspend_structure(serial: &str) -> Vec { - const STARTED: &str = "virtio-sound: stream 0 started"; - const STOPPED: &str = "virtio-sound: stream 0 stopped"; - const CONNECTED: &str = " connected (id="; - const SUSPENDED: &str = "soundd: suspended"; - - let mut problems = Vec::new(); - - let Some(first_connect) = serial.find(CONNECTED) else { - problems.push("suspend structure: no client connect in capture".to_string()); - return problems; - }; - match serial.find(STARTED) { - None => problems.push( - "suspend structure: the stream never started inside the test window — \ - either the device was already running at boot (the boot state is \ - SUSPENDED) or the resume path is broken" - .to_string(), - ), - Some(at) if at < first_connect => problems.push( - "suspend structure: stream started before the first client connect".to_string(), - ), - Some(_) => {} - } - - let Some(last_removed) = last_client_removed(serial) else { - problems.push("suspend structure: no client removal in capture".to_string()); - return problems; - }; - if !serial[last_removed..].contains(SUSPENDED) { - problems.push( - "suspend structure: no `soundd: suspended` after the last client removal" - .to_string(), - ); - } - if !serial[last_removed..].contains(STOPPED) { - problems.push( - "suspend structure: no `virtio-sound: stream 0 stopped` after the last \ - client removal — the device is still running with no clients" - .to_string(), - ); - } - problems +/// Sum one `soundd:` counter across every stats window. +fn sum_field(serial: &str, key: &str) -> u32 { + let needle = format!(" {key}="); + serial + .match_indices(&needle) + .filter_map(|(at, _)| { + let rest = &serial[at + needle.len()..]; + let digits: String = rest.chars().take_while(char::is_ascii_digit).collect(); + digits.parse::().ok() + }) + .sum() } /// Every way a client left, as soundd reported it: one entry per /// `soundd: client {id} removed ({how})` in `serial`. /// -/// Anchored on ` removed` with the `soundd: client ` prefix discipline -/// [`last_client_removed`] documents, and the reason is read from the same line -/// — a removal that names none yields the empty string, which is what -/// [`check_departures`] reds on. -pub fn departures(serial: &str) -> Vec { +/// The reason is read from the same line — a removal that names none yields the +/// empty string, which is what [`check_departures`] reds on. +fn departures(serial: &str) -> Vec { serial .lines() .filter(|l| l.contains("soundd: client ") && l.contains(" removed")) @@ -735,7 +113,7 @@ pub fn departures(serial: &str) -> Vec { /// /// `expect` is how many removals the window must carry: a capture where no /// client ever left would otherwise satisfy every check above it vacuously. -pub fn check_departures(serial: &str, expect: usize) -> Vec { +fn check_departures(serial: &str, expect: usize) -> Vec { const KNOWN: [&str; 4] = ["closed", "refused", "disconnected", "signal pipe gone"]; let mut problems = Vec::new(); @@ -767,295 +145,19 @@ pub fn check_departures(serial: &str, expect: usize) -> Vec { problems } -/// Offset of the last `soundd: client {id} removed`, the anchor the two -/// after-the-last-client assertions above are relative to. -/// -/// ` removed` alone is an eight-character substring that any future line in -/// any component could carry; landing after soundd's markers it would move the -/// anchor past them and red all four configs at once, with a message accusing -/// soundd of the bug it does not have. Requiring the `soundd: client ` prefix -/// makes the anchor soundd's by construction rather than by a tree-wide -/// absence of other emitters. -/// -/// The two halves are matched separately because they are separate console -/// writes: `eprintln!("soundd: client {} removed", id)` emits three format -/// pieces, so a whole foreign line can land between the prefix and the suffix -/// (see the module doc on interleaving). A ` removed` qualifies when some -/// `soundd: client ` precedes it with no other ` removed` in between — true -/// for soundd's own, false for a foreign line printed after it. -fn last_client_removed(serial: &str) -> Option { - const CLIENT: &str = "soundd: client "; - const REMOVED: &str = " removed"; - serial - .match_indices(REMOVED) - .filter(|(at, _)| { - let before = &serial[..*at]; - before.rfind(CLIENT).is_some_and(|c| !before[c..].contains(REMOVED)) - }) - .map(|(at, _)| at) - .last() -} - -/// Bounds derived from the device's clock, not from any recorded run: values -/// on the wrong side of one did not happen, whatever the counter says. -/// -/// `check_counters` asks whether a run got *worse*; this asks whether it -/// happened *at all*. A violation is reported as a **broken instrument**, never -/// as a regression: fatal in both tiers, and in the thorough tier it aborts the -/// run before the value can enter the sample or the re-baselining output. That -/// separation is what the thorough tier cannot provide for itself — it applies -/// no per-run ceiling, its Mann-Whitney test is rank-based so one absurd value -/// moves no median, and it prints its own sample as the next baseline. -/// -/// The reference is the wall-clock life of the QEMU process, timed by the -/// harness. It is *outside* the guest, so no guest-side defect can inflate it -/// in step with the counter it bounds; the wav capture cannot serve, because -/// its timeline is the stream soundd submitted, so a stall that submits nothing -/// does not lengthen it. And it needs no recorded number, so there is nothing -/// to tune when a run goes red. The one assumption is that the guest and host -/// clocks agree to within a large factor — they are the same TSC up to -/// calibration error, and a calibration wrong by a percent would break the DLL -/// long before it reached these margins. -/// -/// These bounds sit far above every per-run ceiling in -/// `tests/audio-baseline.toml` (2.07-8.01 pipeline depths), and they have to: a -/// ceiling admits values that are bad but real, so a bound firing anywhere near -/// one would be answering the regression question again. "A few pipeline -/// depths" is a health threshold, not a physical limit. -pub fn check_physical(counters: &SounddCounters, run_secs: f64) -> Vec { - let mut faults = Vec::new(); - - // Lateness is the distance between two instants on the guest clock, both - // inside the life of the soundd process, which is inside the life of the - // QEMU process. A larger value does not fit, whatever it would mean. - if counters.max_wake_lat_us as f64 > run_secs * 1e6 { - faults.push(format!( - "wake lateness {}us ({:.1} pipeline depths) exceeds the whole {run_secs:.2}s run \ - it was measured inside — the instrument is broken, not the scheduler", - counters.max_wake_lat_us, - counters.max_wake_lat_us as f64 / PIPELINE_DEPTH_US as f64, - )); - } - - // The device is a fixed-rate DAC: it retires exactly one period every - // PERIOD_SECS and frees the DMA slot soundd then refills, so it cannot - // have taken more periods in the run than the run had room for, plus the - // pipeline still in flight at the end. - let room = (run_secs / PERIOD_SECS) as u32 + 8; - if counters.submitted > room { - faults.push(format!( - "{} periods submitted, but a {run_secs:.2}s run holds at most {room} \ - — the instrument is broken", - counters.submitted - )); - } - - // Definitional: soundd counts an underrun on a subset of the periods it - // counts as submitted, in the same branch. Violating it means the counter - // or the parser is wrong. - if counters.underruns > counters.submitted { - faults.push(format!( - "{} underruns out of {} periods submitted — underruns are a subset of \ - submitted, so one of the two counters is wrong", - counters.underruns, counters.submitted - )); - } - - faults -} - -fn silent_runs(mono: &[i32], min_len: usize) -> Vec { - let mut runs = Vec::new(); - let mut start = None; - for (i, &s) in mono.iter().enumerate() { - match (s.abs() <= SILENCE_MAX, start) { - (true, None) => start = Some(i), - (false, Some(s0)) => { - if i - s0 >= min_len { - runs.push(SilentRun { - start: s0, - len: i - s0, - }); - } - start = None; - } - _ => {} - } - } - if let Some(s0) = start { - if mono.len() - s0 >= min_len { - runs.push(SilentRun { - start: s0, - len: mono.len() - s0, - }); - } - } - runs -} - -/// soundd's own word for the sink it took when the machine has none. -pub const NULL_SINK: &str = "soundd: no audio device, presenting a null sink"; - -/// The kernel's word for the other outcome: soundd is not there any more. -/// -/// Read by every wait on something soundd owes, so that a dead mixer ends the -/// wait with the caller's own sentence instead of a guard expiring. -pub const SOUNDD_GONE: &str = "exit: soundd"; -/// Wait until soundd has said which sink it took, bounded by the guest. +/// soundd's mix thread never waits on the log: `tests/logstallcase`'s `logd` +/// reads nothing of soundd's until the job says the tone has played, and the +/// job fills soundd's ring with soundd's own refusals before it plays. Judged +/// off `/log`: /// -/// init spawns its programs without waiting, so the ready marker is one child's -/// first line and orders nothing about another's — which of soundd and the test -/// runner speaks first is a race the guest never promised to win. Both callers -/// used to bound it with a span of host wall clock, and on a KVM runner soundd -/// lost that race: `metal_sim_null_audio` was red 5 of 5 with its line arriving -/// 64 ms past a 500 ms window (run `31258202923`, and the probe that timed it). -/// -/// [`SOUNDD_GONE`] ends the wait too, because the regression this gates is -/// soundd *exiting* on a device-less machine — it has to red with the caller's -/// own sentence and not fifteen seconds later as a stall. -pub fn await_null_sink(qemu: &mut QemuInstance, log: &mut String) -> Result<(), String> { - qemu::await_guest(qemu, log, "soundd to say which sink it took", |seen| { - seen.contains(NULL_SINK) || seen.contains(SOUNDD_GONE) - }) -} - -/// Gate: on a machine with **no audio hardware** (`Profile::Metal`, the T14's -/// shape and the one the `tone` panic reproduced on), an audio-producing client -/// runs to completion — exit 0, no panic — and does so at the real audio rate, -/// neither instantly nor stalled. -/// -/// This is the whole point of the null sink: hardware absence is a routing -/// state. Before it, soundd exited on a device-less machine and released its -/// service name, so cpal's `build_output_stream` failed `NotFound` and `tone` -/// panicked in `.expect("failed to build audio stream")`. That pre-null tree is -/// the negative control — reverting soundd's null-sink path reds this test on -/// the first assertion, with the kernel's `exit: soundd` in the capture. -/// -/// Three host-side assertions, and there is no wav here (this machine has no -/// device to capture from), so ground truth is the client's exit, the host wall -/// clock around it, and soundd's own counters: -/// -/// 1. **No crash.** The client exits 0. -/// 2. **Real rate.** A 3 s tone takes ~3 s of wall clock. Instant discard would -/// finish in a fraction of a second (the client fills its ring and races to -/// the end); a stalled sink would time the run out. soundd's `submitted` -/// counter — periods it drained — is the in-guest cross-check: ~3 s / 2.9 ms -/// ≈ 1034. -/// 3. **Not silent about being silenced.** soundd reports the discarded stream -/// in its stats windows (clients ≥ 1), the same accounting a real sink emits -/// and what #106's status tool will read. -pub fn null_sink_real_rate( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - ..Default::default() - }, - ); - - // soundd must present the null sink rather than exit. - let mut early = qemu.boot_log().to_string(); - let stalled = await_null_sink(&mut qemu, &mut early).err(); - if !early.contains(NULL_SINK) { - return Err(format!( - "{}soundd did not present a null sink on a device-less machine:\n{early}", - stalled.map(|why| format!("{why}\n")).unwrap_or_default() - )); - } - - // The tone is DURATION_SECS = 3.0 s of samples (tests/toyos-rust-tests/ - // src/tone.rs). At the real audio rate it cannot finish before then. - let start = Instant::now(); - let result = qemu.run_test("test_rs_audio_tone", Duration::from_secs(30)); - let elapsed = start.elapsed().as_secs_f64(); - - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - match result.exit_code { - Some(0) => {} - Some(code) => { - return Err(format!( - "tone exited {code} on a device-less machine (a no-device machine must \ - still play to completion):\n{}", - result.stdout - )) - } - None => return Err(format!("tone produced no exit code:\n{}", result.stdout)), - } - - // Real rate, host-measured. The 3 s tone plus its ~0.2 s drain tail and a - // little spawn overhead lands near 3.3 s; the bounds catch the two failure - // shapes the null sink exists to avoid — instant discard below, a sink - // draining slower than the audio rate above. - const MIN_SECS: f64 = 2.5; - const MAX_SECS: f64 = 8.0; - if !(MIN_SECS..=MAX_SECS).contains(&elapsed) { - return Err(format!( - "null sink drained a 3 s tone in {elapsed:.2} s (expected {MIN_SECS}..={MAX_SECS} s): \ - a client that writes N seconds of audio must take ~N seconds\nstdout:\n{}", - result.stdout - )); - } - - // soundd's own accounting of the discarded stream: a real-rate cross-check - // and the proof it is not silent about the silencing. The final window - // races the client's exit, so collect a little more serial first. - let serial = result.serial.clone() + &qemu.drain_serial(Duration::from_millis(500)); - let counters = parse_soundd_counters(&serial)?; - if counters.windows == 0 { - return Err(format!( - "soundd reported no stats window with a client — the tone never reached the null sink:\n{serial}" - )); - } - // ~3 s / 2.902 ms ≈ 1034 periods, plus the disconnect ramp. Wide enough to - // absorb the window boundaries, tight enough that instant discard (a handful - // of periods) or a half-rate drain (~520) both fail. - const MIN_SUBMITTED: u32 = 700; - const MAX_SUBMITTED: u32 = 1500; - if !(MIN_SUBMITTED..=MAX_SUBMITTED).contains(&counters.submitted) { - return Err(format!( - "null sink submitted {} periods for a 3 s tone (expected {MIN_SUBMITTED}..={MAX_SUBMITTED}): \ - the drain rate is not the audio rate\nstdout:\n{}", - counters.submitted, result.stdout - )); - } - - eprintln!( - " [metal-sim] null sink drained a 3 s tone in {elapsed:.2} s, {} periods, \ - {} stats window(s) — real rate, no device", - counters.submitted, counters.windows - ); - Ok(()) -} - -/// Gate: soundd's mix thread never waits on the log. -/// -/// `tests/logstallcase` boots a `logd` that reads nothing of soundd's until the -/// guest says the tone has played. The guest fills soundd's log ring with -/// soundd's own refusals and then plays the tone, so every line soundd says -/// while it plays is said to a full ring. Judged off `/log`, the sink of -/// record, after `run shutdown`: -/// -/// 1. **The tone played whole**: the capture carries it with no underrun and no -/// click. A mix thread that waited on its log would stop at its first line, -/// and the tone never plays at all. -/// 2. **The ring was full** — the premise: `logd` found every one of its slots -/// waiting when the stall ended, and had read none of them before. -/// 3. **Nothing went unwritten silently**: every line soundd's control thread +/// 1. **The ring was full** — the premise: `logd` found every one of its slots +/// waiting when the stall ended. +/// 2. **Nothing went unwritten silently**: every line soundd's control thread /// said after the boot is in `/log` or among the records `logd` counted /// unwritten, exactly, and some were counted — the flood is larger than the /// ring. -pub fn soundd_log_stall(rust_bins: &[(String, Vec)]) -> Result<(), String> { - use std::cell::Cell; - +pub fn log_stall_on_metal(log: &Serial) -> Result<(), String> { const REFUSAL: &str = "soundd: refusing connection,"; // `logd`'s counts of soundd's shared-ring records it could not write: its // lanes are the mix thread's, counted apart. @@ -1074,260 +176,90 @@ pub fn soundd_log_stall(rust_bins: &[(String, Vec)]) -> Result<(), String> { let rest = &line[line.find(marker)? + marker.len()..]; rest.split(' ').next()?.parse().ok() }; - - let staged = super::logstream::stage("tests/logstallcase", "soundd-log-stall", &[], rust_bins)?; - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/logstallcase"); - let options = BootOptions { - boot_image: Some(qemu::Staged::Written(staged.image.clone())), - ..Default::default() - }; - let mut qemu = QemuInstance::boot_with_options(&config, &[], rust_bins, options); - let result = qemu.run_test("test_rs_soundd_log_stall", Duration::from_secs(240)); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - if result.exit_code != Some(0) { - return Err(format!("the guest failed (exit {:?}):\n{}", result.exit_code, result.stdout)); - } - let said_by_guest = |marker: &str| -> Result { - result - .stdout - .lines() + let text = log.text(); + let said_by_job = |marker: &str| -> Result { + text.lines() .find_map(|l| number_after(l, marker)) - .ok_or_else(|| format!("the guest never said {marker:?}:\n{}", result.stdout)) + .ok_or_else(|| format!("the job never said {marker:?}:\n{text}")) }; - // Every refusal the guest heard or provoked is one line soundd said. - let owed = said_by_guest("flooded soundd with ")? + said_by_guest("soundd refused ")?; + // Every refusal the job heard or provoked is one line soundd said. + let owed = said_by_job("flooded soundd with ")? + said_by_job("soundd refused ")?; - // What the ledger reads over a log so far: refusals present, records - // counted unwritten, the other control lines present, and `logd`'s word - // on the ring: how many of its slots were waiting, of how many. - struct Ledger { - refusals: Cell, - unsaid: Cell, - others: Cell, - waiting: Cell>, - } - let read = |ledger: &Ledger, line: &str| { + let (mut refusals, mut unsaid, mut others, mut waiting) = (0u64, 0u64, 0u64, None); + let soundd = toyos_build::bootlog::lines_of(text, "soundd"); + let logd = toyos_build::bootlog::lines_of(text, "logd"); + for line in soundd.lines().chain(logd.lines()) { if line.contains(REFUSAL) { - ledger.refusals.set(ledger.refusals.get() + 1); + refusals += 1; } - let unwritten: u64 = UNWRITTEN.iter().filter_map(|m| number_before(line, m)).sum(); - ledger.unsaid.set(ledger.unsaid.get() + unwritten); + unsaid += UNWRITTEN.iter().filter_map(|m| number_before(line, m)).sum::(); if AFTER_FLOOD.iter().any(|m| line.contains(m)) { - ledger.others.set(ledger.others.get() + 1); + others += 1; } - if let (Some(waiting), Some(slots)) = (number_after(line, RELEASED), number_after(line, OF_SLOTS)) { - ledger.waiting.set(Some((waiting, slots))); + if let (Some(held), Some(slots)) = (number_after(line, RELEASED), number_after(line, OF_SLOTS)) { + waiting = Some((held, slots)); } - }; - let new_ledger = - || Ledger { refusals: Cell::new(0), unsaid: Cell::new(0), others: Cell::new(0), waiting: Cell::new(None) }; - let closed = |l: &Ledger| { - l.waiting.get().is_some() - && l.refusals.get() + l.others.get() + l.unsaid.get() >= owed + AFTER_FLOOD.len() as u64 - }; - - // The console carries the same lines as `/log`; it is only what says when - // to shut the guest down. - let console = new_ledger(); - for line in result.before.lines().chain(result.serial.lines()) { - read(&console, line); } - if !closed(&console) { - let _ = qemu.drain_until(Duration::from_secs(120), |line| { - read(&console, line); - closed(&console) - }); - } - - // Read once QEMU has exited, which is when it has written the capture's - // tail, and before the guest is dropped, which deletes it. - let mut shutdown = String::new(); - let (file, wav) = super::logstream::shut_down_keeping(qemu, &mut shutdown, &staged, |qemu| { - parse_wav(qemu.audio_wav_path()) - })?; - let (file, wav) = (file.concat(), wav?); - let analysis = analyze(&wav); - let _ = std::fs::remove_file(&staged.image); - - let log = new_ledger(); - for line in toyos_build::bootlog::lines_of(&file, "soundd").lines() { - read(&log, line); - } - for line in toyos_build::bootlog::lines_of(&file, "logd").lines() { - read(&log, line); - } - - let slots = match log.waiting.get() { - Some((waiting, slots)) if waiting == slots && slots > 0 => slots, - Some((waiting, slots)) => { + match waiting { + Some((held, slots)) if held == slots && slots > 0 => {} + Some((held, slots)) => { return Err(format!( - "logd found {waiting} of soundd's {slots} ring slots waiting when its stall \ - ended: the tone was not played to a full ring" + "logd found {held} of soundd's {slots} ring slots waiting when its stall ended: \ + the tone was not played to a full ring" )) } None => return Err("/log never says logd's stall on soundd ended".to_string()), - }; + } let said = owed + AFTER_FLOOD.len() as u64; - let accounted = log.refusals.get() + log.others.get() + log.unsaid.get(); - if accounted != said || log.unsaid.get() == 0 { + if refusals + others + unsaid != said || unsaid == 0 { return Err(format!( - "soundd's control thread said {said} lines after the boot ({owed} refusals and \ - {} others); /log holds {} refusals and {} of the others, and logd counted {} \ - unwritten", + "soundd's control thread said {said} lines after the boot ({owed} refusals and {} \ + others); /log holds {refusals} refusals and {others} of the others, and logd \ + counted {unsaid} unwritten", AFTER_FLOOD.len(), - log.refusals.get(), - log.others.get(), - log.unsaid.get() )); } - - let signal_secs = analysis.active_samples as f64 / wav.sample_rate as f64; - // The tone is 3 s, and a sine of amplitude 16000 is under SIGNAL_THRESHOLD - // for 2 asin(500 / 16000) / pi = 2% of its samples. - const MIN_SIGNAL_SECS: f64 = 2.5; - if signal_secs < MIN_SIGNAL_SECS || !analysis.underruns.is_empty() || !analysis.clicks.is_empty() - { - return Err(format!( - "the tone played to a stalled log is not whole: {signal_secs:.2} s of signal \ - (expected at least {MIN_SIGNAL_SECS}), {} underrun(s), {} click(s)", - analysis.underruns.len(), - analysis.clicks.len() - )); - } - - eprintln!( - " [logstallcase] {said} lines said to a full ring of {slots} records: {} in /log, {} \ - counted unwritten; the tone played {signal_secs:.2} s with no underrun", - log.refusals.get() + log.others.get(), - log.unsaid.get() - ); Ok(()) } -/// Gate: doom's sound producer outruns its audio callback and the game lives. -/// -/// The T14 report this exists for: about five seconds into playing doom the -/// machine froze, and the first thing to go was doom itself — -/// `sound command ring overflow: audio callback stalled` inside -/// `I_UpdateSoundParams`, an `extern "C"` frame with no unwind path, so the -/// panic became `abort`. The kernel then panicked retiring the task and the -/// compositor died granting shared memory to a pid that was no longer there. -/// This is the first domino. +/// Doom's sound producer outruns its audio callback and the game lives — the +/// first domino of the T14's freeze. `/system/bin/doom --sound-stress` parks the +/// callback and requires its own period counter to stand still across the +/// burst, so "the producer outran the consumer" is a fact about the two of them. /// -/// Nothing in the flood is timed. The actuator parks the audio callback with -/// `cpal::Stream::pause` and requires the callback's own period counter to -/// stand still across the burst, so "the producer outran the consumer" is a -/// fact about the two of them rather than about how busy this host was — a -/// distinction the harness owes the suspended-laptop case, where a wall-clock -/// burst is satisfied by stopping both sides at once. -/// -/// Four assertions, three of them in-guest facts the host reads back and one -/// on the wire: -/// -/// 1. **The game lives.** `/system/bin/doom --sound-stress` exits 0. On the tree this -/// replaced the same burst aborts on its 65th command. -/// 2. **The burst was real.** `stalled_burst` commands were issued with the -/// callback's period count unchanged — 4096 of them, 64x the retired ring. -/// 3. **The callback converged, and did so at the audio rate.** The sound the -/// last command started plays to completion, and the periods that took -/// match its length: a mixer that lost the command never finishes, and one -/// that restarted it takes longer. -/// 4. **The last command is what reached the device.** Every superseded update -/// in the burst carries `QUIET_VOLUME`, which mixes to 251 LSB — under -/// `SIGNAL_THRESHOLD`, so it is not signal. The final one carries full -/// volume, 16000 * 127/255 = 7968. A capture with signal in it is a capture -/// in which the last write won. -pub fn doom_sound_flood(rust_bins: &[(String, Vec)]) -> Result<(), String> { - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/doomcase"); - let mut qemu = QemuInstance::boot_with_options(&config, &[], rust_bins, BootOptions::default()); - - let result = qemu.run_test("test_rs_doom_sound_flood", Duration::from_secs(60)); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - if result.exit_code != Some(0) { - return Err(format!( - "doom did not survive its own sound producer (exit {:?}):\n{}", - result.exit_code, result.stdout - )); - } - - let counters = parse_stress_line(&result.stdout)?; - +/// 1. **The burst was real.** More commands were issued with the callback's +/// period count unchanged than the retired 64-entry ring held. +/// 2. **The callback converged.** The sound the last command started plays to +/// completion, in no fewer periods than its length: a mixer that lost the +/// command never finishes. +pub fn sound_flood_on_metal(log: &Serial) -> Result<(), String> { + let counters = parse_stress_line(log.text())?; // The retired ring held 64 commands and asserted on the 65th. const RETIRED_RING_CAP: u64 = 64; if counters.stalled_burst <= RETIRED_RING_CAP { return Err(format!( - "the flood was {} commands against a callback that had stopped, which the \ - retired 64-entry ring would have swallowed — the actuator proved nothing", + "the flood was {} commands against a callback that had stopped, which the retired \ + 64-entry ring would have swallowed — the actuator proved nothing", counters.stalled_burst )); } - - // A period is 128 frames, so a sound of N frames occupies ceil(N/128) of - // them. The ceiling is loose on purpose: it is a liveness bound on the game - // thread noticing the sound ended, and the verdict is the floor — a mixer - // that restarted or skipped the sound cannot land on its exact length. check_playback("tone", counters.tone_periods, counters.tone_frames)?; - check_playback("probe", counters.probe_periods, counters.probe_frames)?; - - // Let the tail of the capture reach the file before reading it. - let _ = qemu.drain_serial(Duration::from_millis(500)); - let wav = parse_wav(qemu.audio_wav_path())?; - let analysis = analyze(&wav); - - // 16000 * 127/255 = 7968 at full volume against 251 for every superseded - // update, so the band excludes the second outcome by a factor of eight and - // is wide enough that soundd's mix path is not being measured here. - const MIN_PEAK: i32 = 4000; - const MAX_PEAK: i32 = 12000; - if !(MIN_PEAK..=MAX_PEAK).contains(&analysis.peak) { - return Err(format!( - "the device played a peak of {} (expected {MIN_PEAK}..={MAX_PEAK}): the volume \ - the last command named is not the volume that reached the wire", - analysis.peak - )); - } - // The tone is TONE_FRAMES long and 96% of a sine at this amplitude clears - // SIGNAL_THRESHOLD; a third of it is a floor no partial application meets. - let min_active = counters.tone_frames as usize / 3; - if analysis.active_samples < min_active { - return Err(format!( - "only {} samples of signal reached the device for a {}-frame tone (expected \ - at least {min_active})", - analysis.active_samples, counters.tone_frames - )); - } - - eprintln!( - " [doomcase] {} commands issued with the callback parked, tone converged in {} \ - periods for {} frames, {} concurrent commands, {} samples of signal at peak {}", - counters.stalled_burst, - counters.tone_periods, - counters.tone_frames, - counters.concurrent_cmds, - analysis.active_samples, - analysis.peak, - ); - Ok(()) + check_playback("probe", counters.probe_periods, counters.probe_frames) } struct StressCounters { stalled_burst: u64, tone_periods: u64, tone_frames: u64, - concurrent_cmds: u64, probe_periods: u64, probe_frames: u64, } -fn parse_stress_line(stdout: &str) -> Result { - let line = stdout +fn parse_stress_line(text: &str) -> Result { + let line = text .lines() .find(|l| l.contains("[sound-stress] stalled_burst=")) - .ok_or_else(|| format!("doom printed no [sound-stress] line:\n{stdout}"))?; + .ok_or_else(|| format!("doom printed no [sound-stress] line:\n{text}"))?; let field = |name: &str| -> Result { let prefix = format!("{name}="); line.split_whitespace() @@ -1338,61 +270,45 @@ fn parse_stress_line(stdout: &str) -> Result { stalled_burst: field("stalled_burst")?, tone_periods: field("tone_periods")?, tone_frames: field("tone_frames")?, - concurrent_cmds: field("concurrent_cmds")?, probe_periods: field("probe_periods")?, probe_frames: field("probe_frames")?, }) } +/// A period is 128 frames, so a sound of N frames occupies ceil(N/128) of them, +/// and a mixer that skipped part of it lands under that. fn check_playback(what: &str, periods: u64, frames: u64) -> Result<(), String> { const PERIOD_FRAMES: u64 = 128; let exact = frames.div_ceil(PERIOD_FRAMES); - if !(exact..=exact * 4).contains(&periods) { + if periods < exact { return Err(format!( - "the {what} took {periods} periods to play {frames} frames (expected \ - {exact}..={}): the mixer did not apply the last command as written", - exact * 4 + "the {what} took {periods} periods to play {frames} frames (expected at least \ + {exact}): the mixer did not apply the last command as written" )); } Ok(()) } -/// The shipped audio client must finish and exit on a machine with no device. -/// -/// `metal_sim_null_audio` already asserts a client drains at the real rate, and -/// it passed on every boot while the T14 hung — because it runs this crate's -/// own tone, which reaches soundd through the SDK, and the program a user runs -/// reaches it through `cpal`. Same sink, same period grid, different client. -/// -/// Two clients in series, because the T14 log shows the *second* connect never -/// being applied: the control thread accepts and prints `opening stream`, and -/// no `client N connected` follows it. -pub fn null_sink_shipped_client( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { profile: qemu::Profile::Metal, ..Default::default() }, - ); - - let result = qemu.run_test("test_rs_null_sink_client_exits", Duration::from_secs(60)); - if let Some(err) = &result.error { - return Err(format!("{err}\nstdout:\n{}\nserial:\n{}", result.stdout, result.serial)); - } - match result.exit_code { - Some(0) => {} - Some(code) => { - return Err(format!( - "the shipped tone exited {code} on a device-less machine:\n{}", - result.stdout - )) - } - None => return Err(format!("no exit code:\n{}", result.stdout)), +/// Doom's music reaches the device with the SoundFont this tree ships: doom +/// opened the committed file — its byte count against `assets/soundfont.sf2` on +/// the host — and played to the end of the check. +pub fn music_on_metal(log: &Serial) -> Result<(), String> { + let root = super::compile::repo_root(); + let shipped = std::fs::metadata(root.join(toyos_build::soundfont::SOUNDFONT_PATH)) + .map_err(|e| format!("{}: {e}", toyos_build::soundfont::SOUNDFONT_PATH))? + .len(); + let opened = log.must_say("[doom-sound] /system/share/soundfont.sf2:")?; + let bytes: u64 = opened + .split_whitespace() + .find_map(|token| token.parse().ok()) + .ok_or_else(|| format!("no byte count in {opened:?}"))?; + if bytes != shipped { + return Err(format!( + "doom opened a {bytes}-byte SoundFont and this tree ships {shipped} bytes: the image \ + is not carrying {}", + toyos_build::soundfont::SOUNDFONT_PATH + )); } - eprintln!(" [null-sink] {}", result.stdout.trim()); + log.must_say("[music-check] lump=")?; Ok(()) } diff --git a/tests/common/clock.rs b/tests/common/clock.rs index bb581e95e24..83ce1876797 100644 --- a/tests/common/clock.rs +++ b/tests/common/clock.rs @@ -3,8 +3,8 @@ //! The dev machine is a laptop and the owner closes the lid. A run that spans //! that is not a slow run, it is an **invalid measurement**: QEMU's virtual //! clock, the guest's own millisecond stamps and every device timing in it jump -//! by however long the machine was away, and every wall-clock verdict in the -//! serial tail and in gate A is taken against one of those. CLAUDE.md already +//! by however long the machine was away, and every wall-clock ceiling in the +//! suite is taken against one of those. CLAUDE.md already //! documents the signature — a tight cluster of durations plus a few enormous //! outliers — and documents it as something an agent must check *before* //! recording a finding, which is to say the harness has never been able to. diff --git a/tests/common/console.rs b/tests/common/console.rs index 31fd198640d..254b8bdc3c0 100644 --- a/tests/common/console.rs +++ b/tests/common/console.rs @@ -360,12 +360,12 @@ pub struct Verdict<'a> { /// /// **The scope boundary, and it is the whole safety argument.** This is the C /// family's stdout comparison and nothing else. Every other reader of a -/// daemon's line — the audio gates counting soundd's stats, `netd_*` waiting on -/// `netd: ready`, the sshd tests reading its host identity, the log gates — -/// reads `TestResult::serial` or a boot log, which this never touches. Those -/// tests *assert on* a daemon's line; this family is the one for which a -/// daemon's line is by construction not the subject, because the subject is a -/// C program's own stdout against a file recorded from it. +/// daemon's line — `netd_*` waiting on `netd: ready`, the sshd tests reading +/// its host identity, the log gates — reads `TestResult::serial` or a boot log, +/// which this never touches. Those tests *assert on* a daemon's line; this +/// family is the one for which a daemon's line is by construction not the +/// subject, because the subject is a C program's own stdout against a file +/// recorded from it. /// /// Which is also why the filter cannot make a broken case pass: a tinycc case's /// output is decided by its source, so the only way `soundd: …` appears in one diff --git a/tests/common/hda.rs b/tests/common/hda.rs deleted file mode 100644 index ce635e00798..00000000000 --- a/tests/common/hda.rs +++ /dev/null @@ -1,332 +0,0 @@ -//! H4's audio arm, in the harness: soundd driving a real Intel HDA controller -//! itself, read back off the device rather than off the guest's opinion. -//! H0's own feasibility diagnostic was built ahead of this, then deleted once -//! the driver above answered every question it was asked for. - -use std::path::Path; -use std::time::Duration; - -use crate::common::audio::{await_null_sink, NULL_SINK}; -use crate::common::qemu::{BootOptions, Profile, QemuInstance}; -use crate::common::serial::Serial; - -/// H4's gate: a 440 Hz tone out of an Intel HDA controller soundd drives -/// itself, read back off the device rather than off the guest's opinion. -/// -/// `-audiodev wav` is the same ground truth gate A's four recorded configs use, -/// and the machine differs from them in the sound card alone -/// ([`Profile::Hda`]), so the capture is comparable by construction. What is -/// asserted here is **harm** — the tone is present, continuous, and dithered — -/// which is the fast tier's verdict. This is not a gate-A arm: it has no -/// recorded distribution behind it, and the four HDA sections a gate-A arm -/// would need in `tests/audio-baseline.toml` — `audio_tone_hda.smp1`, `.smp8`, -/// `audio_tone_hda_load.smp1`, `.smp8` — are unrecorded. -pub fn hda_tone( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: Profile::Hda, - kernel_params: &["hda-allowlist-selftest"], - ..Default::default() - }, - ); - // soundd claims and configures the controller the instant it starts, which - // is after the ready marker and before any test command — a window neither - // `boot_log` nor `run_test`'s own capture covers. - let mut log = Serial::boot(&qemu); - log.push(&qemu.drain_serial(Duration::from_millis(500))); - - let result = qemu.run_test("test_rs_audio_tone", Duration::from_secs(30)); - if let Some(err) = &result.error { - return Err(err.to_string()); - } - if result.exit_code != Some(0) { - return Err(format!("the tone did not play: {:?}\n{}", result.exit_code, result.stdout)); - } - log.push(&result.serial); - log.push(&qemu.drain_serial(Duration::from_millis(500))); - let serial = log.text().to_string(); - log.must_say("hda: 00:")?; - log.must_say("bound, statests=")?; - log.must_say("soundd: hda codec0 vendor=1af4")?; - log.must_say("-> pin 0x03 (line-out)")?; - log.must_say("soundd: hda path configured in")?; - if serial.contains("presenting a null sink") { - return Err(format!("soundd fell back to the null sink:\n{serial}")); - } - - // The allow-list, every arm, on the one caller that can reach it. - for want in [ - "hda: selftest write ICW written", - "hda: selftest write SDnFMT written", - "hda: selftest write SDnCTL written", - "hda: selftest write SDnCTL-tag written", - "hda: selftest write SDnBDPL refused", - "hda: selftest write SDnBDPU refused", - "hda: selftest write SDnCBL refused", - "hda: selftest write SDnLVI refused", - "hda: selftest write SDnSTS refused", - "hda: selftest write SDnCTL-srst refused", - "hda: selftest write SDnCTL-wide refused", - "hda: selftest write INTCTL refused", - "hda: selftest write GCTL refused", - "hda: selftest read ICS read", - "hda: selftest read IRR read", - "hda: selftest read SDnLPIB refused", - "hda: selftest read STATESTS refused", - ] { - log.must_say(want)?; - } - - let wav = crate::common::audio::parse_wav(qemu.audio_wav_path())?; - let analysis = crate::common::audio::analyze(&wav); - if analysis.peak < 8000 { - return Err(format!( - "the capture peaks at {} — the tone plays at 16000 and nothing reached the device", - analysis.peak - )); - } - let gaps = crate::common::audio::gap_histogram(&analysis, wav.sample_rate); - let dropouts: u32 = gaps.values().sum(); - let breaks = crate::common::audio::phase_breaks(&wav); - let pitch = crate::common::audio::dominant_hz(&wav); - eprintln!( - " [hda] {} frames at {} Hz {} ch, peak {} active {:.2}s dither {:.1}% pitch {:.1}Hz \ - gaps {} phase-breaks {}", - wav.mono.len(), - wav.sample_rate, - wav.channels, - analysis.peak, - analysis.active_samples as f64 / wav.sample_rate as f64, - analysis.dither_ratio.unwrap_or(0.0) * 100.0, - pitch.unwrap_or(0.0), - crate::common::audio::format_histogram(&gaps), - breaks.len(), - ); - if dropouts > 0 { - return Err(format!( - "{dropouts} mid-tone silences in the capture: {}", - crate::common::audio::format_histogram(&gaps) - )); - } - // The rate the engine plays at is soundd's decision on this machine and - // nothing else here can see it: a stream format naming the wrong base is - // eight buffers of correct audio a second played 8.8% fast, which every - // other assertion in this file passes. - if let Some(complaint) = crate::common::audio::wrong_pitch(&wav) { - return Err(complaint); - } - // The instrument the gap detector cannot be: an engine that replays a - // period nobody refilled puts the tone back 0.28 of a cycle out, and - // nothing about that is silent. Zero here and zero on - // all four virtio configs, measured — so the check has a calibration and - // not just a threshold. - // - // `dither_ratio` is deliberately not asserted, and is printed above so the - // difference is visible rather than hidden. It measures the longest - // *silent* run, and QEMU's two device models put different silence there: - // virtio-sound's capture opens before the stream does, so its longest - // silent run is soundd's own dithered output (24.6% on this host), while - // `intel-hda`'s wav voice runs only while the stream does and the longest - // silent run is host padding at the ends of the file. The virtio arm still - // asserts it, over a stretch that is soundd's. - if !breaks.is_empty() { - let where_ = |n: &usize| { - format!("{n} (period {:.1}, {:?})", *n as f64 / 128.0, &wav.mono[n - 1..=n + 1]) - }; - return Err(format!( - "the captured tone is not one sine: {} phase breaks at {}", - breaks.len(), - breaks.iter().take(8).map(where_).collect::>().join(", ") - )); - } - log.must_be_clean() -} - -/// The T14's panic, staged: a client that stops producing mid-stream. -/// -/// **Ground truth is which of two things soundd did with the periods the client -/// did not cover**, and the two machines must answer differently. HDA's engine -/// is a cyclic ring — it plays buffer `i` again `num_buffers` periods after -/// completing it, whatever soundd put there — so a period held back for a -/// client is played as silence anyway and then completed a second time, which -/// is a completion for a buffer soundd still holds. virtio-sound's queue plays -/// nothing soundd has not submitted, so holding one costs nothing and the -/// deferral is exactly right there. -/// -/// So: the ring arm must report `underruns` (soundd filled the periods and had -/// no client audio for them) and the queue arm must report `deferred` (soundd -/// held them). Asserting both is what stops the two obvious wrong fixes — -/// deleting the deferral, which reds the queue arm, and letting the ring hold a -/// period, which reds the ring arm with the panic this exists for. -/// -/// Nothing in the tone clients reaches this state: they keep their rings full, -/// so `hda_tone` measured `deferred=0` on every run. The stall is the actuator, -/// and it has to outlast one lap of the ring — 8 periods, 23.2 ms — or the -/// engine never comes back round to a period soundd is holding. -pub fn hda_client_stall( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let ring = stall_run(test_config, c_bins, rust_bins, "ring", Profile::Hda)?; - let queue = stall_run(test_config, c_bins, rust_bins, "queue", Profile::Headless)?; - - if ring.underruns == 0 { - return Err(format!( - "the ring arm reports no underrun: soundd never filled a period the stalled client \ - had not covered, so this run staged nothing\n{}", - ring.serial - )); - } - if ring.deferred != 0 { - return Err(format!( - "the ring arm deferred {} period(s): the engine replays every one of them and then \ - completes it again, which is the panic this test exists for\n{}", - ring.deferred, ring.serial - )); - } - if queue.deferred == 0 { - return Err(format!( - "the queue arm deferred nothing: deferral is what a stalled client is supposed to buy on \ - a device that plays only what it is given\n{}", - queue.serial - )); - } - eprintln!( - " [hda] stalled client: ring filled {} period(s) it had no audio for and held none; \ - queue held {}", - ring.underruns, queue.deferred - ); - Ok(()) -} - -struct StallRun { - underruns: u32, - deferred: u32, - serial: String, -} - -/// One boot of the stalling client, and what soundd did with the periods. -/// -/// soundd's liveness is the first verdict and it is not implied by the client's -/// exit: the client talks to soundd over IPC and a soundd that died mid-stream -/// leaves it blocked, so the run times out rather than reporting a code. The -/// panic line is checked by name anyway — `must_be_clean` would catch it, but -/// not say which of soundd's assertions it was. -fn stall_run( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], - arm: &str, - profile: Profile, -) -> Result { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { profile, ..Default::default() }, - ); - let mut log = Serial::boot(&qemu); - let result = qemu.run_test("test_rs_hda_client_stall", Duration::from_secs(60)); - if let Some(err) = &result.error { - return Err(format!("the {arm} arm: {err}\n{}\n{}", result.stdout, result.serial)); - } - log.push(&result.serial); - log.push(&qemu.drain_serial(Duration::from_millis(500))); - if result.exit_code != Some(0) { - return Err(format!( - "the stalling client exited {:?} on the {arm} arm:\n{}\n{}", - result.exit_code, - result.stdout, - log.text() - )); - } - log.must_not_say("repeated completion for free buffer")?; - log.must_be_clean()?; - - let serial = log.text().to_string(); - // The client plays twice with a suspend between, so a resume is under test - // as much as the stall is: on a ring the drain gives its periods up rather - // than holding them, and what the second prime fills and where in the ring - // it starts are both what the first stream left behind. - let resumes = serial.matches("soundd: resumed").count(); - if resumes < 2 { - return Err(format!( - "soundd resumed {resumes} time(s) on the {arm} arm — the second stream did not find a \ - suspended daemon, so nothing here tests a resume:\n{serial}" - )); - } - let counters = crate::common::audio::parse_soundd_counters(&serial)?; - if counters.windows == 0 { - return Err(format!("soundd reported no stats window on the {arm} arm:\n{serial}")); - } - Ok(StallRun { underruns: counters.underruns, deferred: sum_field(&serial, "deferred"), serial }) -} - -/// Sum one `soundd:` counter across every stats window. -/// -/// `parse_soundd_counters` stops at the fields gate A's baseline records, and -/// `deferred` is not one of them — it is an activity signal with no ceiling. It -/// is read here because it is the whole difference between the two arms. -fn sum_field(serial: &str, key: &str) -> u32 { - let needle = format!(" {key}="); - serial - .match_indices(&needle) - .filter_map(|(at, _)| { - let rest = &serial[at + needle.len()..]; - let digits: String = rest.chars().take_while(char::is_ascii_digit).collect(); - digits.parse::().ok() - }) - .sum() -} - -/// Two controllers, both with a codec that answers. -/// -/// The kernel binds neither and names both. A first-match bind would go green -/// on every other test in this file, and it is the defect `pci.rs` records one -/// layer down — so this is the arm that makes the rule tested rather than -/// merely written. -pub fn hda_two_live_refused( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { profile: Profile::HdaTwoLive, ..Default::default() }, - ); - // The refusal is a kernel boot line and is in the capture already; soundd's - // answer to it is a userland line that races the ready marker, so it is - // waited for on the guest's clock rather than on a span of the host's. - let mut text = qemu.boot_log().to_string(); - let stalled = await_null_sink(&mut qemu, &mut text).err(); - let log = Serial::named("boot console", text); - - log.must_say("hda: 00:")?; - log.must_say("has a live link (statests=")?; - log.must_say("controllers answer on this machine")?; - log.must_say("refused by name, no HDA audio")?; - log.must_not_say("bound, statests=")?; - // The machine still boots and still has a sink: absence of hardware is a - // routing state, and a refusal must not be a machine that will not run. - // **And it is the bind's absence and not a second spelling of the line - // above**: init claims each class before it spawns, and soundd reaches the - // null sink only where the endowment is missing — so this line requires - // `device::try_claim(HdaAudio)` to have answered `Absent`. - log.must_say(NULL_SINK) - .map_err(|why| match stalled { - Some(stall) => format!("{stall}\n{why}"), - None => why, - })?; - log.must_say("Boot: complete")?; - log.must_be_clean() -} diff --git a/tests/common/hostload.rs b/tests/common/hostload.rs deleted file mode 100644 index f7033afbdae..00000000000 --- a/tests/common/hostload.rs +++ /dev/null @@ -1,180 +0,0 @@ -//! What else the host was doing when a gate A verdict was taken. -//! -//! CLAUDE.md's 2026-08-04 ruling stands and nothing here touches it: load is -//! not an excuse, no threshold branches on any number below, and a -//! load-coincident red is investigated as a real defect of the pipeline. What -//! this adds is that the investigation starts from a recorded fact, and that an -//! A/B can say whether its two arms were taken under comparable conditions — -//! which `tests/audio-baseline.toml`'s recorded sample can only claim in prose. -//! -//! Three readings, because they fail in different directions: -//! -//! - **The load average triple**, 1/5/15 minutes. No figure of it resolves a -//! single ~15 s boot; the 1-minute one lags by design. But the competition is -//! other worktrees' builds, which last minutes, and the triple's *shape* says -//! whether the host was ramping up or winding down where one figure cannot. -//! The 1-minute figure leads because it is the one the baseline's ceiling -//! derivation recorded per run, so a fresh reading compares directly to it. -//! - **QEMU processes machine-wide.** Exact and instantaneous where the load -//! average is neither, and it counts the guests the *host* has rather than -//! the ones this run started — `qemu::live_instances()` is asserted zero -//! before gate A, so the harness's own knowledge is a constant here. This -//! run's own guest is still up when the sample is taken, so 1 is quiet. -//! - **`toyos-build` processes machine-wide**, drivers and harnesses alike: the -//! count that names the competition as ToyOS work rather than as whatever -//! else a laptop is doing. 1 is this run alone. -//! -//! A reading that cannot be taken is reported as unknown. A gate A verdict must -//! not turn on whether the process table answered. - -use std::fmt; - -#[derive(Clone, Copy)] -pub struct HostLoad { - /// 1, 5 and 15-minute load averages. - pub load: Option<[f64; 3]>, - pub qemu: Option, - pub toyos_build: Option, -} - -impl HostLoad { - pub fn sample() -> Self { - let names = process_names(); - HostLoad { - load: load_average(), - qemu: names.as_deref().map(|n| count(n, |name| name.starts_with("qemu-system"))), - toyos_build: names.as_deref().map(|n| count(n, is_toyos_build)), - } - } -} - -impl fmt::Display for HostLoad { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - match self.load { - Some([one, five, fifteen]) => { - write!(f, "host: load {one:.1}/{five:.1}/{fifteen:.1}")? - } - None => write!(f, "host: load ?")?, - } - write!(f, " qemu {} toyos-build {}", show(self.qemu), show(self.toyos_build)) - } -} - -/// The conditions a whole sample was taken under, for the re-record that sample -/// becomes. The thorough tier prints its own numbers in a form meant to be -/// pasted into `tests/audio-baseline.toml`; this is the sentence that has to go -/// beside them, and its absence is why the recorded sample's own conditions are -/// a claim rather than a measurement. -pub fn summarise(runs: &[HostLoad]) -> String { - let load: Vec = runs.iter().filter_map(|r| r.load.map(|l| l[0])).collect(); - let load = match span_f64(&load) { - Some((lo, med, hi)) => format!("1-min load {lo:.1}-{hi:.1} (median {med:.1})"), - None => "1-min load unreadable".to_string(), - }; - format!( - "host conditions over {} runs: {load}, qemu {}, toyos-build {}", - runs.len(), - span_usize(runs.iter().filter_map(|r| r.qemu)), - span_usize(runs.iter().filter_map(|r| r.toyos_build)), - ) -} - -fn span_f64(values: &[f64]) -> Option<(f64, f64, f64)> { - let mut sorted = values.to_vec(); - sorted.sort_by(|a, b| a.partial_cmp(b).unwrap()); - Some((*sorted.first()?, sorted[sorted.len() / 2], *sorted.last()?)) -} - -fn span_usize(values: impl Iterator) -> String { - let mut sorted: Vec = values.collect(); - sorted.sort_unstable(); - match (sorted.first(), sorted.last()) { - (Some(lo), Some(hi)) => format!("{lo}-{hi}"), - _ => "?".to_string(), - } -} - -fn show(count: Option) -> String { - count.map_or_else(|| "?".to_string(), |n| n.to_string()) -} - -fn load_average() -> Option<[f64; 3]> { - let mut out = [0.0f64; 3]; - // SAFETY: getloadavg writes at most `nelem` doubles through the pointer. - let filled = unsafe { libc::getloadavg(out.as_mut_ptr(), out.len() as i32) }; - (filled == out.len() as i32).then_some(out) -} - -/// Every process on the host, by executable basename. -/// -/// A pid that exits between the sizing call and the read, or refuses its path -/// (a zombie, another user's), contributes no name rather than failing the -/// sample — the counts above are of processes that could be named. -#[cfg(target_os = "macos")] -fn process_names() -> Option> { - let count = unsafe { libc::proc_listallpids(std::ptr::null_mut(), 0) }; - if count <= 0 { - return None; - } - // Slack for processes spawned between the sizing call and the fill. - let mut pids = vec![0 as libc::pid_t; count as usize + 16]; - let bytes = i32::try_from(std::mem::size_of_val(&pids[..])).ok()?; - let filled = unsafe { libc::proc_listallpids(pids.as_mut_ptr().cast(), bytes) }; - if filled <= 0 { - return None; - } - pids.truncate(filled as usize); - let mut names = Vec::with_capacity(pids.len()); - for pid in pids { - let mut buf = [0u8; libc::PROC_PIDPATHINFO_MAXSIZE as usize]; - let len = unsafe { libc::proc_pidpath(pid, buf.as_mut_ptr().cast(), buf.len() as u32) }; - if len <= 0 { - continue; - } - let path = String::from_utf8_lossy(&buf[..len as usize]); - names.push(path.rsplit('/').next().unwrap_or(&path).to_string()); - } - Some(names) -} - -/// Every process on the host, by executable basename: `/proc//exe`'s -/// target where that link is readable, the kernel's 15-byte `comm` otherwise — -/// every prefix `sample` matches on fits in either. -#[cfg(target_os = "linux")] -fn process_names() -> Option> { - let mut names = Vec::new(); - for entry in std::fs::read_dir("/proc").ok()?.flatten() { - let is_pid = entry - .file_name() - .to_str() - .is_some_and(|n| !n.is_empty() && n.bytes().all(|b| b.is_ascii_digit())); - if !is_pid { - continue; - } - if let Ok(path) = std::fs::read_link(entry.path().join("exe")) { - if let Some(name) = path.file_name() { - names.push(name.to_string_lossy().into_owned()); - continue; - } - } - if let Ok(comm) = std::fs::read_to_string(entry.path().join("comm")) { - names.push(comm.trim().to_string()); - } - } - Some(names) -} - -fn count(names: &[String], matches: impl Fn(&str) -> bool) -> usize { - names.iter().filter(|name| matches(name)).count() -} - -/// Cargo spells the two halves of this build system differently — the driver is -/// the package's bin target `toyos-build`, the harness is the test target's -/// `toyos_build-` — so a single prefix misses whichever one is asking. -fn is_toyos_build(name: &str) -> bool { - name.starts_with("toyos-build") || name.starts_with("toyos_build") -} - -// A third host OS gets a named gap, not a missing-function error at a distance. -#[cfg(not(any(target_os = "macos", target_os = "linux")))] -compile_error!("process_names reads the process table per-OS, and this OS has no arm yet"); diff --git a/tests/common/inspect.rs b/tests/common/inspect.rs index 767c623acd3..99349bbe8d5 100644 --- a/tests/common/inspect.rs +++ b/tests/common/inspect.rs @@ -6,9 +6,8 @@ //! matches too much is a path in the answer this file did not name, and one //! that matches too little is a named path missing from it. Values are judged //! where the machine fixes them — QEMU's user network leases `10.0.2.15/24`, -//! nothing plays audio until `inspect_plays` does, and the USB stick this file -//! crafts has one partition free and one init grants — and read only for shape -//! elsewhere. +//! nothing plays audio, and the USB stick this file crafts has one partition +//! free and one init grants — and read only for shape elsewhere. use std::collections::BTreeMap; use std::path::Path; @@ -23,8 +22,6 @@ pub const CONFIG: &str = "tests/inspectcase"; /// The guest binary that holds the negative control. pub const DENIED: &str = "inspect_denied"; -/// The guest binary that plays periods through soundd. -pub const PLAYS: &str = "inspect_plays"; /// The guest binary that sends `SYS_DEVICE_INVENTORY` its edges. pub const BOUNDS: &str = "inventory_bounds"; @@ -55,9 +52,9 @@ const NET: &[&str] = &[ pub fn boot(rust_bins: &[(String, Vec)]) -> Result { let bins: Vec<(String, Vec)> = - rust_bins.iter().filter(|(name, _)| [DENIED, PLAYS, BOUNDS].contains(&name.as_str())).cloned().collect(); - if bins.len() != 3 { - return Err(format!("{DENIED}, {PLAYS} and {BOUNDS} were not all built")); + rust_bins.iter().filter(|(name, _)| [DENIED, BOUNDS].contains(&name.as_str())).cloned().collect(); + if bins.len() != 2 { + return Err(format!("{DENIED} and {BOUNDS} were not both built")); } let stick = super::lane::dir().join("inspect-stick.img"); let mib = 1024 * 1024; @@ -200,30 +197,6 @@ pub fn reads_its_owners(qemu: &mut QemuInstance) -> Result<(), String> { expect(line, &got, "sound.stream.state", "suspended")?; expect(line, &got, "sound.stream.clients", "0")?; - // Periods through soundd, and the running sums grow by at least what it - // took: each frame soundd took went out in a period it submitted. - let result = job(qemu, &format!("test_rs_{PLAYS}"), 0)?; - let said = "inspect plays: soundd took "; - let taken: u64 = result - .stdout - .lines() - .find_map(|l| l.split_once(said).map(|(_, rest)| rest.trim_end_matches(" frames"))) - .and_then(|n| n.trim().parse().ok()) - .ok_or_else(|| format!("{PLAYS} did not say how much soundd took:\n{}", result.stdout))?; - if taken == 0 { - return Err(format!("{PLAYS} proved soundd took nothing:\n{}", result.stdout)); - } - let line = "inspect sound.*"; - let got = answer(&job(qemu, line, 0)?); - let submitted = number(line, &got, "sound.periods.submitted")?; - let period = number(line, &got, "sound.period_frames")?; - if submitted.saturating_mul(period) < taken { - return Err(format!( - "`{line}`: {submitted} periods of {period} frames submitted, and soundd took {taken} \ - frames" - )); - } - let line = "inspect log.*"; let got = answer(&job(qemu, line, 0)?); exactly( diff --git a/tests/common/lane.rs b/tests/common/lane.rs index 9046d269855..3c28cbe121b 100644 --- a/tests/common/lane.rs +++ b/tests/common/lane.rs @@ -66,8 +66,7 @@ static RUN: OnceLock = OnceLock::new(); /// it, and it is gone when the run is, green or red (`toyos_tmpdir` is the /// policy, and what reclaims the directory of a run that was killed). /// -/// A red run's serial logs, and any suspect audio capture already renamed to -/// keep (`audio-*-smp*.wav`), are the parts of it read afterwards — by an agent +/// A red run's serial logs are the parts of it read afterwards — by an agent /// and by the nightly's artifact — so they are copied to a directory of their /// own under [`RED_RUN_SERIAL`] first: megabytes, where the images are /// gigabytes. Named for this run's own root — unique across every process a @@ -125,31 +124,18 @@ fn kept_root(run: &Path) -> Result { Ok(super::compile::repo_root().join(RED_RUN_SERIAL).join(name)) } -/// Where `path`, somewhere under this run's directory, ends up if the run ends -/// red: mirrored under [`kept_root`], as [`keep_serial`] would copy it there. -pub fn kept_path(path: &Path) -> PathBuf { - let run = RUN.get().expect("`Run::begin` comes before any scratch"); - let rel = path - .strip_prefix(run) - .unwrap_or_else(|_| panic!("{} is not under this run's directory {}", path.display(), run.display())); - kept_root(run).unwrap_or_else(|e| panic!("{e}")).join(rel) -} - -/// Copy every `uart-*.log` and suspect `audio-*-smp*.wav` under `run` to a -/// directory of its own under [`RED_RUN_SERIAL`], keeping each one's path -/// below the run. +/// Copy every `uart-*.log` under `run` to a directory of its own under +/// [`RED_RUN_SERIAL`], keeping each one's path below the run. fn keep_serial(run: &Path) -> Result { let kept = kept_root(run)?; copy_serial(run, &kept)?; Ok(kept) } -/// A serial log, or a suspect audio capture already renamed for keeping -/// (`tests/toyos.rs`'s `measure_audio_run`) — everything [`keep_serial`] -/// rescues from a run's scratch before it goes. +/// A serial log — everything [`keep_serial`] rescues from a run's scratch +/// before it goes. fn worth_keeping(name: &str) -> bool { - (name.starts_with("uart-") && name.ends_with(".log")) - || (name.starts_with("audio-") && name.contains("-smp") && name.ends_with(".wav")) + name.starts_with("uart-") && name.ends_with(".log") } fn copy_serial(dir: &Path, into: &Path) -> Result<(), String> { diff --git a/tests/common/mod.rs b/tests/common/mod.rs index 0c64deee567..d8867c00237 100644 --- a/tests/common/mod.rs +++ b/tests/common/mod.rs @@ -20,10 +20,6 @@ pub mod faults; #[allow(dead_code)] pub mod gpt; #[allow(dead_code)] -pub mod hda; -#[allow(dead_code)] -pub mod hostload; -#[allow(dead_code)] pub mod https; #[allow(dead_code)] pub mod iommu; @@ -44,8 +40,6 @@ pub mod origin; #[allow(dead_code)] pub mod partclaim; #[allow(dead_code)] -pub mod passcost; -#[allow(dead_code)] pub mod pkg; #[allow(dead_code)] pub mod power; @@ -61,8 +55,6 @@ pub mod serial; #[allow(dead_code)] pub mod ssh; #[allow(dead_code)] -pub mod stats; -#[allow(dead_code)] pub mod storage; pub mod swap; #[allow(dead_code)] diff --git a/tests/common/origin.rs b/tests/common/origin.rs index 2a3cd88b136..26e416498ce 100644 --- a/tests/common/origin.rs +++ b/tests/common/origin.rs @@ -349,12 +349,8 @@ pub fn refused_stop(c_bins: &[(String, Vec)], rust_bins: &[(String, Vec) Ok(()) } -/// What init says when it stops the machine without `logd`'s answer, and how -/// long it waits first (`userland/init`'s `FLUSH_BOUND`). +/// What init says when it stops the machine without `logd`'s answer. const FLUSH_WAITED_OUT: &str = "init: logd did not answer the flush in"; -const FLUSH_BOUND_MS: u64 = 5_000; -/// The kernel's record as a stop begins its sync, after every thread stopped. -const SYNCING: &str = "Syncing filesystems..."; /// The milliseconds since boot a program's line in `/log` carries: the one /// field of its head with a decimal point. @@ -469,23 +465,12 @@ pub fn keeps_the_owners_slots(rust_bins: &[(String, Vec)]) -> Result<(), Str let _ = std::fs::remove_file(&staged.image); // A stop that went ahead before the flush answered cuts `/log` short // whatever the slots did, so that is its own verdict and not this one's. - // init's word is in its ring when the machine stops, so the console - // rarely carries it. An answered flush wrote init's stop line, which is - // stamped before the flush was asked; and the kernel's sync starting a - // flush bound or more after that line is the wait on two records of one - // clock. - let sync = tail - .lines() - .find(|l| l.contains(SYNCING)) - .and_then(bootlog::record_millis) - .ok_or_else(|| format!("the console carries no {SYNCING:?} with a time\n{tail}"))?; + // An answered flush wrote init's stop line. let stop = bootlog::stopping_line(&log).and_then(program_millis); let unanswered = match stop { _ if tail.contains(FLUSH_WAITED_OUT) => Some("init said so".to_string()), None => Some("init's stop line never reached /log".to_string()), - Some(stop) => (sync.saturating_sub(stop) >= FLUSH_BOUND_MS).then(|| { - format!("the kernel synced {} ms after init's stop line", sync.saturating_sub(stop)) - }), + Some(_) => None, }; if let Some(why) = unanswered { return Err(format!( @@ -493,7 +478,6 @@ pub fn keeps_the_owners_slots(rust_bins: &[(String, Vec)]) -> Result<(), Str slots\n{tail}" )); } - let stop = stop.expect("an answered flush's stop line has a time"); let runner = bootlog::lines_of(&log, RUNNER); // Non-vacuity: the flood met a full ring, so the slots were contested. let flooded = runner.lines().filter(|l| l.starts_with("flood ")).count(); @@ -507,9 +491,7 @@ pub fn keeps_the_owners_slots(rust_bins: &[(String, Vec)]) -> Result<(), Str )); } eprintln!( - " [origin] {flooded} flood lines filled the ring, and test-runner's {ENDED:?} is in /log; \ - the kernel synced {} ms after init's stop line", - sync.saturating_sub(stop) + " [origin] {flooded} flood lines filled the ring, and test-runner's {ENDED:?} is in /log" ); Ok(()) } diff --git a/tests/common/passcost.rs b/tests/common/passcost.rs deleted file mode 100644 index ff741280c06..00000000000 --- a/tests/common/passcost.rs +++ /dev/null @@ -1,536 +0,0 @@ -//! The harness half of the scheduler's pass-cost instrument: read the check -//! build's published distribution and judge it against **what this accelerator -//! has been recorded producing**, not against an absolute line. -//! -//! **Why the judgement is here and not in the kernel.** A pass is measured with -//! the only clock either world has across one — wall clock, `rdtsc` in the -//! kernel — and a guest's wall clock advances while the host has taken its vCPU -//! away. The elapsed time of a pass is therefore a *composed* quantity: the -//! scheduler's own work plus an interval the host's scheduler sets, which this -//! CPU neither observes nor controls and which no constant bounds. A panic may -//! assert only what its own site observes and what no workload scales, so the -//! kernel records the distribution ([`toyos_sched::cpu::PassCostReport`]) and -//! this file decides what it means. -//! -//! **What that costs, stated rather than implied.** A single pass that really -//! ran long and a single pass whose CPU was descheduled are the same sample, -//! and nothing — here or in the kernel — can separate them. So the maximum is -//! printed and never gated, and a rare long pass is reported rather than -//! caught. That is the honest limit of this instrument and it has not changed. -//! -//! # Why `MAX_PASS_NS` is not the line any more -//! -//! This gate used to read the budget directly: nine passes in ten provably -//! under 200 000 ns. Its argument was that a host which deschedules a vCPU -//! touches the handful of passes it lands in, so it moves the maximum and not -//! the 90th percentile — *mass is the scheduler, extremes are the machine*. -//! That argument was an observed rate rather than a bound, and it was put to -//! the experiment on 2026-08-18. -//! -//! **It is false.** Six repetitions per arm, quiet and loaded, strictly -//! interleaved in one session so both arms share the ambient host; twelve -//! CPU-runs per arm at `smp: 2`. The load was fourteen pure-shell spin loops, -//! one per logical CPU (`sysctl -n hw.logicalcpu` answers 14 on the dev host), -//! measured standing alone at 90.0–94.9 % CPU each and taking the 1-minute load -//! average from 9.45 to 28.02. The arms separate on the harness's own boot-width -//! instrument with no overlap: 1.74x–2.34x quiet against 2.66x–2.78x loaded. -//! -//! | | quiet, 12 CPU-runs | loaded, 12 CPU-runs | -//! |---|---|---| -//! | p50 | 8 192 ×1, 16 384 ×1, 65 536 ×10 | 8 192 ×1, 32 768 ×1, 131 072 ×10 | -//! | p90 | 65 536 ×2, 131 072 ×10 | 131 072 ×3, 262 144 ×9 | -//! | max | 1 508 330 – 1 983 355 ns | 1 723 825 – 3 914 718 ns | -//! | over budget | 95 of 2 732 (3.48 %) | 187 of 2 619 (7.14 %) | -//! | p90 over budget | **0 of 12** | **9 of 12** | -//! -//! **The whole distribution translates up by one power-of-two bucket under host -//! load, and 200 000 ns sits between the quiet p90 and the loaded one** — which -//! is the entirety of why the verdict flips. Every order statistic moved, the -//! median as much as the tail, so no fraction chosen instead of nine-in-ten -//! would have survived either. There is no quantile of this quantity that host -//! load leaves alone. -//! -//! # What replaces it: the recorded sample of the accelerator in hand -//! -//! The same session recovered the other half. This file's line has printed on -//! every run whatever the verdict, so CI's own logs already held a KVM sample — -//! sixteen `ci` runs, 32 CPU-runs, 7 612 passes, **not one of them over the -//! budget**, p90 at 32 768 ns and the largest single pass in the whole set -//! 173 906 ns. The two accelerators are not one instrument: p90 differs by four -//! between KVM and a *quiet* dev host and by eight against a loaded one, and the -//! maxima by twenty. An absolute line cannot be right for both, and 200 000 ns -//! is right for neither — six times looser than anything KVM produces, and -//! inside the range host load alone sweeps on TCG. -//! -//! So a run is judged against its own accelerator's recorded sample -//! ([`KVM`], [`TCG`]) — the shape `tests/audio-baseline.toml` uses, made -//! environment-relative, which needs no separation of steal from work at all. -//! **And where a recorded sample cannot support a verdict it says so and does -//! not take one**: TCG's own sample spans four buckets on one unchanged tree, -//! so no line drawn from it would be a statement about the scheduler. That is -//! `tests/CLAUDE.md`'s standing rule for this host — what only CI's KVM shards -//! can decide is decided there — applied to a cost rather than to a vendor. -//! -//! **What was not done, so nobody re-derives it.** Reading KVM's paravirtual -//! steal-time MSR would let the guest gate a quantity it observes rather than -//! one it infers, and it is closed by owner ruling: a hypervisor-specific -//! facility cannot be the basis of a gate in a tree whose north star is -//! self-hosting on metal. Steal accounting may be a diagnostic and never a gate. - -use toyos_sched::cpu::{PassCostReport, MAX_PASS_NS}; - -/// Samples a report must carry before its quantiles mean anything: a 90th -/// percentile over this many has ten above it. -/// -/// **The margin is 21 %, and it is stated rather than assumed.** Eighteen -/// CPU-runs on the dev host, 2026-08-17, ranged from 121 to 346 passes with a -/// median around 148; 121 is the closest any of them came to this floor. The -/// two recorded samples below bear it out from the other side — their smallest -/// counts are 135 (KVM) and 143 (TCG). If a run ever falls below it the gate -/// reds *saying so*: a report that cannot answer must not answer, and a -/// percentile computed from forty samples is a sample. -/// -/// **The floor holds in both judgement modes**, because it is not about the -/// quantile — it is about the instrument having run at all, and a report of -/// forty passes is a broken instrument on any accelerator. -pub const MIN_SAMPLES: u64 = 100; - -/// What a recorded sample supports. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum Judgement { - /// Gate one bucket above the sample's own worst 90th percentile. - Ceiling, - /// Print the distribution and take no verdict on its magnitude. The string - /// is why: a mode that judges nothing must say what stopped it. - Report(&'static str), -} - -/// One accelerator's recorded pass-cost sample, and the verdict it supports. -/// -/// The sample is the observations themselves rather than a summary of them, for -/// `tests/toyos.rs`'s reason at `BaselineSample`: a summary cannot be re-read by -/// the next person to ask whether the line drawn from it was fair. -#[derive(Clone, Copy)] -pub struct Baseline { - /// The accelerator, as the transcript names it. - pub accelerator: &'static str, - /// What was sampled, when, and with what — the sentence that has to survive - /// beside the numbers, or they are a claim rather than a measurement. - pub recorded: &'static str, - /// Every per-CPU-run 90th percentile in the recorded sample, in ns. These - /// are bucket ends, so every one is a power of two, and [`self_check`] - /// refuses a sample where one is not: a number that is not a bucket end was - /// not read off a run. - pub p90_ns: &'static [u64], - pub judgement: Judgement, -} - -impl Baseline { - /// The worst 90th percentile anywhere in the recorded sample. - pub fn worst_p90_ns(&self) -> u64 { - self.p90_ns.iter().copied().max().expect("a recorded sample with no observations in it") - } - - /// The line a fresh run is held to, or `None` where the sample supports - /// none. - /// - /// **One bucket above the sample's worst, and that margin is the smallest - /// this instrument can express.** A quantile is answered as a bucket end, - /// so the resolution is one power of two; a ceiling *at* the sample's own - /// worst has no margin at all against the next draw from the same - /// distribution, and that worst is itself a thin observation — 3 of 32 on - /// KVM. One bucket is therefore the least that is not a coin flip. - /// - /// **What it costs, said here rather than discovered later.** A regression - /// that moves the 90th percentile by less than four times the sample's mode - /// passes. The instrument has no finer unit to do better with. - pub fn ceiling_ns(&self) -> Option { - match self.judgement { - Judgement::Ceiling => Some(self.worst_p90_ns() * 2), - Judgement::Report(_) => None, - } - } -} - -/// KVM on native x86-64 — CI's `guest` shards, and the accelerator every claim -/// about what a scheduler pass costs is really about. -/// -/// Harvested with `gh run view --job --log` over every `ci` workflow run -/// between the instrument landing (#113, 2026-08-17) and 2026-08-18: sixteen -/// runs, 32 CPU-runs, 135–1 533 passes each and 7 612 pooled. **Not one pass in -/// any of them reached `MAX_PASS_NS`**, and the largest single pass in the whole -/// set was 173 906 ns — so this sample is also, independently, the strongest -/// statement anything has made about the budget on the machine it was derived -/// for. -pub const KVM: Baseline = Baseline { - accelerator: "KVM, native x86-64", - recorded: "16 CI runs / 32 CPU-runs, 2026-08-17..18, runs 32043101865 32043658251 \ - 32044008591 32044311347 32044665350 32044756253 32044758468 32044760536 \ - 32044762195 32044763748 32045857575 32047352064 32050586046 32096188866 \ - 32116842348 32117779110; 7612 passes pooled, 0 over MAX_PASS_NS, max 173906 ns", - // Two per run, cpu0 then cpu1, in the run order above. - p90_ns: &[ - 32_768, 32_768, // 32043101865 - 32_768, 32_768, // 32043658251 - 32_768, 32_768, // 32044008591 - 32_768, 32_768, // 32044311347 - 32_768, 32_768, // 32044665350 - 32_768, 32_768, // 32044756253 - 65_536, 65_536, // 32044758468 - 32_768, 32_768, // 32044760536 - 32_768, 32_768, // 32044762195 - 32_768, 32_768, // 32044763748 - 32_768, 32_768, // 32045857575 - 32_768, 32_768, // 32047352064 - 32_768, 32_768, // 32050586046 - 65_536, 32_768, // 32096188866 - 32_768, 32_768, // 32116842348 - 32_768, 32_768, // 32117779110 - ], - judgement: Judgement::Ceiling, -}; - -/// Cross-arch TCG on the arm64 dev host, emulating x86-64 instruction by -/// instruction while the guest TSC advances with host wall clock. -/// -/// The quiet arm then the loaded arm of the 2026-08-18 experiment — **one -/// unchanged tree, one session, twelve CPU-runs each.** The spread is the whole -/// content of this entry: 65 536 to 262 144 ns, a factor of four, with nothing -/// but the host's other processes separating the halves. A ceiling at the quiet -/// arm's worst reds every loaded run; one at the loaded arm's worst accepts a -/// fourfold regression. Neither is a statement about the scheduler, so this -/// accelerator takes no verdict on magnitude at all. -/// -/// The sample is kept rather than dropped because it *is* that argument, and -/// because the next person to propose a dev-host pass-cost gate should have to -/// read it first. **It also understates the range**, which is the confirming -/// run rather than a caveat: the same test under the same load at 4.09x boot -/// width — harder than either arm — reported `p50 < 131072 ns, p90 < 524288 ns, -/// max 22666022 ns` on a tree whose `sched_stress` passed and whose boot was -/// clean. That is 2.6 times the budget at the 90th percentile and a hundred -/// times it at the maximum, from nothing but fourteen shell loops. -pub const TCG: Baseline = Baseline { - accelerator: "cross-arch TCG, arm64 dev host", - recorded: "12 quiet + 12 loaded CPU-runs, 2026-08-18, one interleaved session; the load is \ - 14 pure-shell spin loops, one per logical CPU, and the arms separate on boot \ - width 1.74x-2.34x against 2.66x-2.78x with no overlap", - p90_ns: &[ - // quiet, six runs, cpu0 then cpu1 - 65_536, 131_072, 131_072, 131_072, 65_536, 131_072, 131_072, 131_072, 131_072, 131_072, - 131_072, 131_072, // loaded, six runs, cpu0 then cpu1 - 262_144, 262_144, 262_144, 262_144, 262_144, 262_144, 131_072, 262_144, 131_072, 262_144, - 131_072, 262_144, - ], - judgement: Judgement::Report( - "the recorded sample spans four buckets on one unchanged tree, moved by nothing but \ - what else the host was running, so no line drawn from it separates a scheduler that \ - grew from a laptop that was busy. What this accelerator still gates is everything \ - else `sched_check_build` asks: a clean boot, the three check-build asserts not \ - firing, and `sched_stress` running to completion", - ), -}; - -/// The recorded sample for the accelerator this run is actually using. -pub fn baseline() -> &'static Baseline { - if super::qemu::SUITE_ARCH.accel().is_hardware() { - &KVM - } else { - &TCG - } -} - -/// The last report each CPU published in `capture`, ordered by CPU. -/// -/// The counters are cumulative since boot, so the last line is the whole run -/// and the earlier ones are prefixes of it. -pub fn reports(capture: &str) -> Vec { - let mut last: Vec = Vec::new(); - for report in capture.lines().filter_map(PassCostReport::parse) { - match last.iter().position(|r| r.cpu == report.cpu) { - Some(at) => last[at] = report, - None => last.push(report), - } - } - last.sort_by_key(|r| r.cpu.0); - last -} - -/// One line per CPU, for the test's own transcript. Printed whatever the -/// verdict: a green run that stops publishing what it measured is how a gate -/// goes quiet. `over` is still counted against `MAX_PASS_NS` — the budget stays -/// the kernel's policy number and stays reported; it has only stopped deciding. -pub fn describe(report: &PassCostReport) -> String { - format!( - "cpu{}: {} passes, p50 < {} ns, p90 < {} ns, p99 < {} ns, max {} ns, \ - {} over the {} ns budget", - report.cpu.0, - report.count, - report.quantile_upper_ns(1, 2), - report.quantile_upper_ns(9, 10), - report.quantile_upper_ns(99, 100), - report.max_ns, - report.over, - MAX_PASS_NS, - ) -} - -/// The one line that says which sample judged this run and how, printed before -/// the per-CPU lines and whatever the verdict. -/// -/// A verdict taken against a recorded sample is unreadable without naming the -/// sample, and a run that judged nothing has to say *that* loudest of all. -pub fn judgement_line(baseline: &Baseline) -> String { - match baseline.judgement { - Judgement::Ceiling => format!( - "judged against the {} sample: 9 passes in 10 under {} ns, one bucket above that \ - sample's own worst of {} ns. Sample: {}", - baseline.accelerator, - baseline.ceiling_ns().expect("Judgement::Ceiling has a ceiling"), - baseline.worst_p90_ns(), - baseline.recorded, - ), - Judgement::Report(why) => format!( - "NOT judged on magnitude — the {} sample supports no line: {}. Sample: {}", - baseline.accelerator, why, baseline.recorded, - ), - } -} - -/// Judge one CPU's distribution against `baseline`. -/// -/// Two claims, and every term in both comes from somewhere: -/// -/// - **The sample floor holds on every accelerator.** A report of fewer than -/// [`MIN_SAMPLES`] passes is a broken instrument, not a distribution, and -/// nothing below it is worth reading. -/// - **The magnitude claim is the recorded sample's, one bucket up** — see -/// [`Baseline::ceiling_ns`] — and only where the sample supports one. -/// -/// The **fraction is still nine in ten and not ninety-nine in a hundred**, -/// because a whole boot plus `sched_stress` is about 150 passes per CPU: a 90th -/// percentile over 150 samples has fifteen above it; a 99th has one and a half, -/// which is its own largest sample wearing a percentile's name. Moving the line -/// from a policy number to a recorded one does not touch that reasoning, and -/// both recorded samples confirm the count from below. -/// -/// `quantile_upper_ns` answers with a bucket's upper bound, so "this fraction -/// cost less than X ns" is exact and X ≤ ceiling is a proof. A quantile landing -/// in the bucket that straddles the ceiling reds, which makes the gate strictly -/// conservative rather than approximately right. -/// -/// **What it does not catch, said here rather than discovered later.** The -/// maximum is not read at all, for the reason the module header gives. A -/// regression smaller than the instrument's own resolution — one power of two — -/// passes on any accelerator. And on a report-only accelerator nothing about -/// magnitude is caught at all, which is the price of not making a claim the -/// environment cannot support. -pub fn verdict(report: &PassCostReport, baseline: &Baseline) -> Result<(), String> { - if report.count < MIN_SAMPLES { - return Err(format!( - "{} — a 90th percentile needs at least {MIN_SAMPLES} samples behind it and this \ - has {}, so the gate would be reading its own largest sample", - describe(report), - report.count, - )); - } - let Some(ceiling) = baseline.ceiling_ns() else { - return Ok(()); - }; - let p90 = report.quantile_upper_ns(9, 10); - if p90 > ceiling { - return Err(format!( - "{} — this distribution has mass the {} sample never showed: nine passes in ten \ - must be provably under {ceiling} ns and it cannot show that. That line is one \ - bucket above the worst 90th percentile in the recorded sample ({}), and the \ - budget itself ({} ns) is deliberately *not* the line — it is six times looser \ - than anything that sample contains", - describe(report), - baseline.accelerator, - baseline.recorded, - MAX_PASS_NS, - )); - } - Ok(()) -} - -/// Prove the instrument in both directions, with no guest. -/// -/// `serial::self_check`'s shape and its reason: a gate nothing checks is a gate -/// nobody knows is broken. Two of the cases below are the two this design turns -/// on — a host that took the CPU away must pass, and a scheduler whose passes -/// grew must not — and neither can be staged on a booted machine. **They are the -/// same distribution to a maximum and to an `over` count**, which is the point: -/// the pair is what says this gate reads mass rather than extremes, and any -/// repair that starts gating the maximum again fails one of them. -/// -/// The recorded samples are checked against themselves as well, because a -/// ceiling that has drifted from the sample it claims to come from is the one -/// failure this design can suffer silently. -pub fn self_check() -> Result<(), String> { - check_recorded(&KVM)?; - check_recorded(&TCG)?; - - // The line KVM's sample draws, spelled out so that re-recording the sample - // has to move this number deliberately rather than as a side effect. - if KVM.ceiling_ns() != Some(131_072) { - return Err(format!( - "pass-cost gate self-check: the KVM sample's ceiling is {:?} and it was 131072 ns \ - when this gate was written — re-record deliberately or not at all", - KVM.ceiling_ns(), - )); - } - // And the fact that made TCG report-only: its own sample spans four - // buckets. A future sample narrow enough to draw a line from would have to - // come through here to do it. - let tcg_span = TCG.worst_p90_ns() / TCG.p90_ns.iter().copied().min().unwrap_or(1); - if tcg_span < 4 { - return Err(format!( - "pass-cost gate self-check: TCG takes no verdict because host load alone sweeps \ - its recorded sample, and that sample now spans only {tcg_span}x — the reason and \ - the evidence have come apart" - )); - } - - let mut bulk = PassCostReport::empty(toyos_sched::hw::CpuId(0)); - // 100 000 passes at 2048..4096 ns — the shape a scheduler doing scheduling - // rather than work produces. - bulk.buckets[12] = 100_000; - bulk.count = 100_000; - - // The case removing the panic exists for: the same scheduler on a host that - // took the vCPU away, at 1–2 ms a time — ten times the budget, and about - // what a whole cross-arch TCG pass cost when invariant P last fired on the - // dev host. - let mut stolen = bulk; - stolen.buckets[12] -= 2_000; - stolen.buckets[21] = 2_000; - stolen.max_ns = 2_000_000; - stolen.over = 2_000; - - // The case it must still catch, and it is `stolen` with the mass moved: - // one pass in five is now over. Same maximum, same shape of outlier, ten - // times the mass — and only the quantile can tell them apart. - let mut grown = bulk; - grown.buckets[12] -= 20_000; - grown.buckets[21] = 20_000; - grown.max_ns = 2_000_000; - grown.over = 20_000; - - // A limit, asserted on purpose: every pass at 32..64 µs is the worst 90th - // percentile KVM's whole recorded sample contains, and it is accepted - // because the ceiling sits one bucket above it. - let mut sample_worst = PassCostReport::empty(toyos_sched::hw::CpuId(0)); - sample_worst.buckets[16] = 100_000; - sample_worst.count = 100_000; - sample_worst.max_ns = 65_000; - - // Two buckets above that, and still four times *under* `MAX_PASS_NS`. - // **This is the case the budget-shaped gate passed and this one refuses**, - // and it is the whole gain from making the line the sample's rather than - // the policy number's. - let mut over_recorded = PassCostReport::empty(toyos_sched::hw::CpuId(0)); - over_recorded.buckets[18] = 100_000; // every pass in 131 072..262 144 ns - over_recorded.count = 100_000; - over_recorded.max_ns = 260_000; - - let mut short = bulk; - short.buckets[12] = 99; - short.count = 99; - - let cases: &[(&str, bool, PassCostReport)] = &[ - ("a scheduler doing scheduling", true, bulk), - ("the same one on a host that stole its vCPU", true, stolen), - ("the same maximum with one pass in five over the budget", false, grown), - ("every pass at the worst 90th percentile KVM has shown", true, sample_worst), - ("every pass two buckets over that, and still under the budget", false, over_recorded), - ("too few samples to have a 90th percentile", false, short), - ]; - for (name, want_ok, report) in cases { - let got = verdict(report, &KVM); - if got.is_ok() != *want_ok { - return Err(format!( - "pass-cost gate self-check: `{name}` should have been {} against the KVM \ - sample, and it was {}: {got:?}", - if *want_ok { "accepted" } else { "refused" }, - if got.is_ok() { "accepted" } else { "refused" }, - )); - } - } - - // Report-only judges no magnitude and still refuses a broken instrument. - // Both halves, because a mode that refused nothing would hide a kernel that - // had stopped measuring. - if verdict(&grown, &TCG).is_err() { - return Err( - "pass-cost gate self-check: a report-only accelerator refused a distribution on \ - its magnitude, which is the one thing it must not do" - .to_string(), - ); - } - if verdict(&short, &TCG).is_ok() { - return Err( - "pass-cost gate self-check: a report-only accelerator accepted a report of 99 \ - passes — the sample floor is about the instrument having run, not about the \ - quantile, and it holds in both modes" - .to_string(), - ); - } - - // The reader, on the wire form the kernel actually prints: the last line - // per CPU wins, and a CPU that never reported is absent rather than zero. - let mut other = bulk; - other.cpu = toyos_sched::hw::CpuId(1); - let capture = format!( - "[kernel 1.001 cpu0] {}\n\ - [kernel 1.002 cpu1] {}\n\ - [kernel 1.003 cpu0] {}\n\ - hello from userland\n", - short, other, bulk, - ); - let read = reports(&capture); - if read.len() != 2 || read[0] != bulk || read[1] != other { - return Err(format!( - "pass-cost gate self-check: the reader took {} report(s) off a capture with two \ - CPUs and three lines: {read:?}", - read.len(), - )); - } - if !reports("hello from userland\n").is_empty() { - return Err("pass-cost gate self-check: a capture with no report yielded one".to_string()); - } - Ok(()) -} - -/// A recorded sample must be in the instrument's own units, and must pass the -/// line it is used to draw. Either failure makes every verdict below it -/// meaningless while looking exactly like a working gate. -fn check_recorded(baseline: &Baseline) -> Result<(), String> { - if baseline.p90_ns.is_empty() { - return Err(format!( - "pass-cost gate self-check: the {} sample has no observations in it", - baseline.accelerator, - )); - } - for &p90 in baseline.p90_ns { - if !p90.is_power_of_two() { - return Err(format!( - "pass-cost gate self-check: the {} sample records a 90th percentile of \ - {p90} ns, and a quantile off this instrument is a bucket end — so that \ - number was not read off a run", - baseline.accelerator, - )); - } - } - if let Some(ceiling) = baseline.ceiling_ns() { - if baseline.worst_p90_ns() > ceiling { - return Err(format!( - "pass-cost gate self-check: the {} sample's own worst 90th percentile ({} ns) \ - is over the ceiling drawn from it ({ceiling} ns), so the recorded runs would \ - red the gate they justify", - baseline.accelerator, - baseline.worst_p90_ns(), - )); - } - } - Ok(()) -} diff --git a/tests/common/qemu.rs b/tests/common/qemu.rs index ed205a58079..cf5e41831ce 100644 --- a/tests/common/qemu.rs +++ b/tests/common/qemu.rs @@ -20,24 +20,10 @@ pub const SUITE_ARCH: Arch = Arch::X86_64; pub static VERBOSE: AtomicBool = AtomicBool::new(false); /// Distinguishes every file one QEMU boot owns from every other boot's within -/// one test process — the wav capture, the UART log, the QMP socket, the -/// screendump, and the bootable image itself. +/// one test process — the UART log, the QMP socket, the screendump, and the +/// bootable image itself. static BOOT_SEQ: AtomicU32 = AtomicU32::new(0); -/// Guests that have been booted and not yet dropped. -/// -/// Gate A's numbers were recorded with one QEMU on the host and nothing else -/// (`tests/audio-baseline.toml`), so "the parallel phase has drained" is a -/// precondition of the audio block rather than a property of where it sits in -/// `main`. This is what lets it be asserted instead of arranged — see -/// [`live_instances`]. -static LIVE: AtomicU32 = AtomicU32::new(0); - -/// How many guests are up right now, across every thread. -pub fn live_instances() -> u32 { - LIVE.load(Ordering::SeqCst) -} - /// The NVMe backing files live guests are holding open. /// /// A lane reuses one image across its boots on purpose ([`super::lane`]), so @@ -196,8 +182,7 @@ pub fn boot_census() -> (u32, u32, Vec) { /// `kernel/Cargo.toml` has forwarded `sched-check = ["toyos-sched/check"]` since /// the check build was written, and nothing in `src/` or `tests/` ever asked for /// it, so `cpu::MAX_PASS_NS`, the pass-cost recorder and `invariants::check_cpu` -/// were compiled by no CI run at all. `sched_check_build` is the test that asks, -/// and `common::passcost` is what judges the half of it that is a measurement. +/// were compiled by no CI run at all. `sched_check_build` is the test that asks. /// /// A fifth entry is that decision again, and it gets this paragraph's argument /// made afresh. Interactive debug mode is separate: it builds @@ -1331,24 +1316,9 @@ pub enum Profile { /// 32-bit-destination entry format rather than the 8-bit one. IommuEim, /// [`Profile::Headless`] with its virtio sound card replaced by an Intel - /// HDA controller and one codec — the machine soundd drives itself. - /// - /// Everything else is held still on purpose. The console is still - /// virtio-serial, the NIC is still there, the disks are the same: what - /// differs from the machine gate A's four recorded configs run on is the - /// sound card, so a difference in the capture is a difference in the audio - /// path. It is not the T14's literal shape and does not try to be — this is - /// the audio arm, not a PCI-topology one. H0's diagnostic staged that - /// comparison and is deleted now that the - /// driver above answers every question it was asked for. + /// HDA controller and one codec — the machine soundd drives itself, and + /// the class-0403 function the IOMMU tests aim. Hda, - /// [`Profile::Hda`] with a second controller that also has a codec. - /// - /// Two live links, which the kernel refuses by name rather than binding - /// the first: choosing between them means walking their codec graphs, and - /// that is the driver's work. The negative control on the whole bind path - /// — a first-match kernel would go green on every other HDA test. - HdaTwoLive, /// QEMU `virt` on AArch64 (GICv3, AAVMF): a GOP from `ramfb`, the boot /// stick on an xHCI, the PL011, and nothing else — no virtio, NIC, NVMe or /// IOMMU. The machine the AArch64 port reaches its console on, and the only @@ -1399,8 +1369,7 @@ impl Profile { | Self::IommuNarrow | Self::IommuNoIntremap | Self::IommuEim - | Self::Hda - | Self::HdaTwoLive => Arch::X86_64, + | Self::Hda => Arch::X86_64, } } @@ -1479,23 +1448,11 @@ const XHCI_MSI_ONLY: &str = "nec-usb-xhci,id=xhci1,msix=off"; const XHCI_NO_IRQ_FIRST: &str = "nec-usb-xhci,id=xhci,msix=off,msi=off"; const XHCI_NO_IRQ_SECOND: &str = "nec-usb-xhci,id=xhci1,msix=off,msi=off"; -/// One controller with one codec: the ordinary machine, and the one an audio -/// arm needs. `hda-output` because it is a playback-only codec — the driver +/// One controller with one codec. `hda-output` because it is a playback-only codec — the driver /// configures no input path and a duplex codec would only add widgets nothing /// walks. const HDA_ONE: &[&str] = &["intel-hda,id=hda0", "hda-output,bus=hda0.0,cad=0,audiodev=hdaaud"]; -/// Two controllers, each with a codec that answers. -/// -/// The state the kernel refuses: it can tell which links are alive and cannot -/// tell which one a human is wired to, so binding either would be a guess. -const HDA_TWO_LIVE: &[&str] = &[ - "intel-hda,id=hda0", - "hda-output,bus=hda0.0,cad=0,audiodev=hdaaud", - "intel-hda,id=hda1", - "hda-output,bus=hda1.0,cad=0,audiodev=hdaaud", -]; - /// Whether a machine has the virtio console and sound block. Which NIC it has /// is [`Nic`]. #[derive(Clone, Copy, PartialEq, Eq)] @@ -1507,10 +1464,7 @@ enum Virtio { /// /// Not a lesser [`Virtio::Present`]: soundd claims a kernel-driven card /// before it looks for a controller to drive itself, so a machine carrying - /// both would exercise the virtio path and nothing else. This is what makes - /// an HDA arm of gate A a *different machine* rather than a different flag, - /// and it keeps the console, the NIC and the timing of the recorded audio - /// configs so the two arms differ in the sound card and not in the machine. + /// both would exercise the virtio path and nothing else. WithoutSound, } @@ -1614,8 +1568,7 @@ struct Shape { usb_disks: &'static [UsbDisk], /// Every Intel HDA controller on the machine and the codecs behind each, /// as `-device` arguments in the order QEMU is to create them. Empty is - /// what every profile but [`Profile::Hda`] and [`Profile::HdaTwoLive`] - /// declares, and it is the machine this kernel has always booted: audio + /// what every profile but [`Profile::Hda`] declares, and it is the machine this kernel has always booted: audio /// through virtio-sound or through nothing at all. /// /// Presence of a class-0403 *function* is the shape dimension, and it is @@ -2235,12 +2188,6 @@ impl Profile { hda: HDA_ONE, ..Self::Headless.shape() }, - Self::HdaTwoLive => Shape { - virtio: Virtio::WithoutSound, - nic: Nic::Virtio, - hda: HDA_TWO_LIVE, - ..Self::Headless.shape() - }, } } @@ -2570,7 +2517,7 @@ pub struct TestResult { /// /// A caller that reads a daemon's startup out of a boot appends this to its /// capture. It is separate from `serial` because `serial` means "while this - /// test ran" and audio gates count lines in it. + /// test ran". pub before: String, /// Why the run did not finish, when it did not. /// @@ -2641,7 +2588,6 @@ pub struct QemuInstance { rx: Receiver, console: ConsoleStream, _reader_thread: thread::JoinHandle, - audio_wav: PathBuf, uart_log: PathBuf, nvme: NvmeClaim, usb_images: Vec, @@ -3151,18 +3097,15 @@ impl QemuInstance { }) .collect(); - let audio_wav = test_dir.join(format!("audio-{seq}.wav")); - let _ = fs::remove_file(&audio_wav); - let qmp_socket = options.qmp.then(|| test_dir.join(format!("qmp-{seq}.sock"))); if let Some(path) = &qmp_socket { let _ = fs::remove_file(path); } let screendump = test_dir.join(format!("screen-{seq}.ppm")); - // Per-instance, not a fixed /tmp path: the audio gate boots dozens of - // guests and a screen test waits on this file, so a shared one would - // let instances read each other's early boot. + // Per-instance, not a fixed /tmp path: a screen test waits on this + // file, so a shared one would let instances read each other's early + // boot. let uart_log = test_dir.join(format!("uart-{seq}.log")); let _ = fs::remove_file(&uart_log); let console_file = options.console_file.then(|| ConsoleFile::of(&uart_log).made()); @@ -3171,7 +3114,6 @@ impl QemuInstance { &boot_image, nvme.path(), &usb_images, - &audio_wav, &uart_log, qmp_socket.as_deref(), &options, @@ -3181,7 +3123,6 @@ impl QemuInstance { &options, Files { seq, - audio_wav, uart_log, nvme, usb_images, @@ -3392,16 +3333,10 @@ impl QemuInstance { self.i8042_trace } - /// The wav file the virtio-sound device records into for this boot. - /// The RIFF size fields stay 0 until QEMU exits cleanly — parse to EOF. - pub fn audio_wav_path(&self) -> &Path { - &self.audio_wav - } - /// Wait for QEMU to exit within `by`: its console closing is the event, and /// the process is reaped after it. Answers what the guest said on the way. - /// A file QEMU finishes only at its exit, the wav among them, is whole once - /// this answers, and is still there until this instance is dropped. + /// A file QEMU finishes only at its exit is whole once this answers, and is + /// still there until this instance is dropped. pub fn await_exit(&mut self, by: Duration) -> Result { let deadline = Instant::now() + by; let mut said = String::new(); @@ -3465,9 +3400,6 @@ impl QemuInstance { } /// Keep collecting serial output for `dur` after a test has returned. - /// soundd flushes its final stats window when the last client leaves, - /// which races the client process's exit — so the line the audio gate - /// reads lands on either side of `===TEST_END===`. /// **Not scaled by the width**, and it is the one duration in this file that /// is not. Callers use it to *pace* — "let the guest run for 400 ms and tell /// me what it said" — so multiplying it does not buy a slow guest more room, @@ -3800,7 +3732,6 @@ impl Drop for QemuInstance { // hopeful is that the process whose descriptors hold QEMU's write lock // on the image is gone by the time it happens. let _ = self.child.wait(); - let _ = fs::remove_file(&self.audio_wav); // **The 16550's log outlives the guest, because it is the one channel // that exists before the console does.** Firmware, the bootloader and // the kernel up to the backend switch write here and nowhere else, so a @@ -3824,7 +3755,6 @@ impl Drop for QemuInstance { if let Some(image) = &self.own_boot_image { let _ = fs::remove_file(image); } - LIVE.fetch_sub(1, Ordering::SeqCst); } } @@ -4329,7 +4259,7 @@ impl QmpDevices { pub fn profile_argv(options: &BootOptions) -> Vec { let p = Path::new("/nonexistent"); let usb: Vec = options.profile.usb_disks().iter().map(|_| p.to_path_buf()).collect(); - qemu_command(p, p, &usb, p, p, None, options) + qemu_command(p, p, &usb, p, None, options) .get_args() .map(|a| a.to_string_lossy().into_owned()) .collect() @@ -4353,7 +4283,6 @@ fn qemu_command( boot_image: &Path, nvme_image: &Path, usb_images: &[PathBuf], - audio_wav: &Path, uart_log: &Path, qmp_socket: Option<&Path>, options: &BootOptions, @@ -4648,26 +4577,9 @@ fn qemu_command( } if !shape.hda.is_empty() { - // The same wav backend virtio-sound gets, so gate A's ground truth — - // what the *device* received — transfers with no new instrument. A boot - // that plays nothing leaves an empty file and costs nothing. - // - // **`timer-period` is 1000 µs here and 5000 for virtio-sound, and that - // is an instrument repair rather than a difference in the audio path.** - // At 5000 the capture of a 3 s 440 Hz tone comes back with eight phase - // discontinuities, at frames 2703-2705, 2821-2823 and 2939-2940 — - // *identical positions across six runs whose audio content differed*, - // which is a capture that drops samples on a fixed cadence and not a - // guest that plays them wrong. QEMU's `hda-codec` holds its own output - // ring and discards what overruns it, and shortening the host's drain - // interval is what stops the overrun. Measured on this host, QEMU - // 11.0.3: 8 breaks at 5000, 0 at 1000, with the guest's own counters - // (1127 periods submitted, no underruns, no drains) identical either - // way and identical to the virtio arm's. - qemu.arg("-audiodev").arg(format!( - "wav,id=hdaaud,path={},timer-period=1000", - audio_wav.display() - )); + // No guest test plays audio: the device is here as a DMA master and a + // claim, so its audio goes nowhere. + qemu.arg("-audiodev").arg("none,id=hdaaud"); for dev in shape.hda { qemu.arg("-device").arg(*dev); } @@ -4745,14 +4657,10 @@ fn qemu_command( if shape.virtio.present() { if shape.virtio.sound() { - // virtio-sound records everything the guest plays into a per-boot - // wav for glitch analysis; timer-period matches the interactive - // config in src/qemu.rs so test timing represents what users hear. + // No guest test plays audio: the device is here as a DMA master and + // a claim, so its audio goes nowhere. qemu.arg("-audiodev") - .arg(format!( - "wav,id=audio0,path={},timer-period=5000", - audio_wav.display() - )) + .arg("none,id=audio0") .arg("-device") .arg(format!("virtio-sound-pci,audiodev=audio0,streams=1{platform}")); } @@ -4801,7 +4709,6 @@ fn qemu_command( /// parameter list eight paths long. struct Files { seq: u32, - audio_wav: PathBuf, uart_log: PathBuf, nvme: NvmeClaim, usb_images: Vec, @@ -4875,7 +4782,6 @@ impl Read for Followed { fn spawn_and_wait_ready(mut qemu: Command, options: &BootOptions, files: Files) -> QemuInstance { let Files { seq, - audio_wav, uart_log, nvme, usb_images, @@ -4966,16 +4872,11 @@ fn spawn_and_wait_ready(mut qemu: Command, options: &BootOptions, files: Files) wait_for_ready(&mut child, &rx, options, &uart_log) }; - // Counted from here rather than from the spawn: every panic inside - // `wait_for_ready` kills the child on its way out and never builds a value - // to drop, so a guest that failed to come up must not be left on the books. - LIVE.fetch_add(1, Ordering::SeqCst); QemuInstance { child, stdin, rx, _reader_thread: reader_thread, - audio_wav, uart_log, nvme, usb_images, diff --git a/tests/common/stats.rs b/tests/common/stats.rs deleted file mode 100644 index 7f4a90527b9..00000000000 --- a/tests/common/stats.rs +++ /dev/null @@ -1,120 +0,0 @@ -//! Two-sample tests for the audio gate's thorough tier. -//! -//! Every thorough-tier decision compares the **fresh sample** against the -//! **recorded baseline sample** — never against a fitted constant. That matters -//! more than it sounds: a threshold derived from 30 clean runs carries the -//! sampling error of those 30 runs, and a one-sample test against it states a -//! confidence it does not have. Measured on this tree, a sign test against the -//! recorded median of `max_wake_lat_us` has a nominal false-red rate of 0.07% -//! and a real one near 1.5%, purely because the reference median moves by an -//! order statistic. A two-sample test carries that uncertainty in the maths. -//! -//! Two tests, one per kind of observation: -//! -//! * `mann_whitney_z` for the counters (continuous, heavy right tails, no -//! distributional assumption survives them). -//! * `fisher_greater` for the yes/no outcomes (did this run drop audio, did -//! it breach a per-run ceiling, did it fail to complete). -//! -//! Both are one-sided in the direction of "worse". A scheduler change that -//! improves audio timing must not fail the gate. - -/// Per-test significance level. One in a thousand, so the ~19 tests a thorough -/// run performs still union-bound under 2%, and the measured family-wise -/// false-red rate on a clean tree is 0.25% (2000 simulated runs against the -/// recorded distributions). Deliberately not a tunable: a gate whose alpha can -/// be raised is a gate that will be raised. -pub const ALPHA: f64 = 0.001; - -/// One-sided standard-normal critical value for `ALPHA`. -pub const Z_CRIT: f64 = 3.0902; - -/// Mann-Whitney U, one-sided, as a normal-approximation z score: how strongly -/// `test` is stochastically *greater* than `base`. Ties get midranks and the -/// tie-corrected variance, which matters here — `underruns` and `drains` are -/// small integers with many repeats. -/// -/// The normal approximation is used rather than the exact permutation -/// distribution because at n1 = n2 = 30 it is already accurate and, measured by -/// bootstrap against the recorded samples, conservative: 3 rejections in 240000 -/// trials under H0 against a nominal 0.1%. -pub fn mann_whitney_z(base: &[f64], test: &[f64]) -> f64 { - let n1 = base.len() as f64; - let n2 = test.len() as f64; - assert!(n1 >= 2.0 && n2 >= 2.0, "Mann-Whitney needs both samples"); - - let mut all: Vec = base.iter().chain(test.iter()).copied().collect(); - all.sort_by(|a, b| a.partial_cmp(b).expect("audio counters are never NaN")); - - // Midrank of each distinct value, and the tie-group sizes. - let mut rank_of: Vec<(f64, f64)> = Vec::new(); - let mut tie_term = 0.0; - let mut i = 0; - while i < all.len() { - let mut j = i; - while j + 1 < all.len() && all[j + 1] == all[i] { - j += 1; - } - let group = (j - i + 1) as f64; - rank_of.push((all[i], (i + j) as f64 / 2.0 + 1.0)); - tie_term += group * group * group - group; - i = j + 1; - } - let rank = |v: f64| { - rank_of - .binary_search_by(|(k, _)| k.partial_cmp(&v).unwrap()) - .map(|idx| rank_of[idx].1) - .expect("value came from the pooled sample") - }; - - let r2: f64 = test.iter().map(|&v| rank(v)).sum(); - let u2 = r2 - n2 * (n2 + 1.0) / 2.0; - let n = n1 + n2; - let var = n1 * n2 / 12.0 * ((n + 1.0) - tie_term / (n * (n - 1.0))); - if var <= 0.0 { - // Every observation identical: no evidence of anything. - return 0.0; - } - (u2 - n1 * n2 / 2.0) / var.sqrt() -} - -/// Fisher's exact test, one-sided: the probability of seeing at least `k1` -/// events in `n1` fresh runs when the fresh runs share the baseline's rate, -/// given `k0` of `n0` in the baseline. Exact, so it stays honest at the tiny -/// counts these rates produce (4 dropped runs in 117). -pub fn fisher_greater(k1: u32, n1: u32, k0: u32, n0: u32) -> f64 { - let (k1, n1, k0, n0) = (k1 as usize, n1 as usize, k0 as usize, n0 as usize); - assert!(k1 <= n1 && k0 <= n0); - let total = n1 + n0; - let events = k1 + k0; - let ln_fact = ln_factorials(total); - let ln_choose = |n: usize, k: usize| ln_fact[n] - ln_fact[k] - ln_fact[n - k]; - let denom = ln_choose(total, events); - let hi = n1.min(events); - let mut p = 0.0; - for x in k1..=hi { - if events - x > n0 { - continue; - } - p += (ln_choose(n1, x) + ln_choose(n0, events - x) - denom).exp(); - } - p.min(1.0) -} - -/// Smallest event count in `n1` fresh runs that Fisher rejects at `ALPHA`. -/// `None` when even `n1` of `n1` would not reject — which is itself worth -/// printing, because it means the sample is too small to test that rate at all. -pub fn fisher_reject_at(n1: u32, k0: u32, n0: u32) -> Option { - (0..=n1).find(|&k| fisher_greater(k, n1, k0, n0) <= ALPHA) -} - -fn ln_factorials(n: usize) -> Vec { - let mut out = Vec::with_capacity(n + 1); - out.push(0.0); - let mut acc = 0.0; - for i in 1..=n { - acc += (i as f64).ln(); - out.push(acc); - } - out -} diff --git a/tests/common/usb.rs b/tests/common/usb.rs index 333a93fd168..7571a0bac3c 100644 --- a/tests/common/usb.rs +++ b/tests/common/usb.rs @@ -1260,8 +1260,8 @@ pub fn xhci_slow_connect( const PARAMS: &[&str] = &["usb-storage-gate", "xhci-slow-connect"]; // The driver's own durations, from where the driver reads them: each is // declared once in `toyos-xhci` and `use`d by `kernel/src/drivers/xhci`, so - // neither bound below can be a copy that drifted. - use toyos_xhci::port::{DEBOUNCE_NS, EMPTY_BUS_NS, SLOW_CONNECT_NS}; + // the floor below cannot be a copy that drifted. + use toyos_xhci::port::{DEBOUNCE_NS, SLOW_CONNECT_NS}; /// Nanoseconds per second, to put those in the units the log's stamps are in. const PER_S: f64 = 1_000_000_000.0; /// How long after port power the driver can first name a port. The register @@ -1269,11 +1269,6 @@ pub fn xhci_slow_connect( /// ports, and `await_connect_settle` then wants `DEBOUNCE_NS` of a connect /// set that has held still and is non-empty. const FIRST_CONNECT_S: f64 = (SLOW_CONNECT_NS + DEBOUNCE_NS) as f64 / PER_S; - /// How late the first port line may be: halfway between the two settles this - /// test tells apart, so one that ends on the device appearing cannot reach - /// it and one that ends at `EMPTY_BUS_NS` cannot stay under it. - const SETTLE_CEILING_S: f64 = - FIRST_CONNECT_S + (EMPTY_BUS_NS as f64 / PER_S - FIRST_CONNECT_S) / 2.0; let (bytes, lba) = Profile::UsbDisk.usb_disk().expect("UsbDisk declares a disk"); let image = test_dir().join("usb-slow-connect.img"); @@ -1295,28 +1290,15 @@ pub fn xhci_slow_connect( // would be green on a driver that never waits and a QEMU that answers // instantly, which is exactly the pair that shipped. // - // `powered_at` is not logged; the two lines that bracket it are. - // `controller started` is taken before it, so it is the anchor that can only - // make the floor generous; the port-power line is printed after it, so it is - // the anchor that can only make the ceiling generous. Neither bound can be - // false because of where in the bracket the instant actually fell. + // `powered_at` is not logged; `controller started` is taken before it, so it + // is the anchor that can only make the floor generous. let started = stamp_of(&log, "xHCI: controller started")?; - let powered = stamp_of(&log, "root-hub ports powered")?; // The first line this driver prints about any port at all. Every other // per-port line is preceded by that port's connect line, so the first match // is the first connect whichever port register it lands on — which the // profile does not fix, since a SuperSpeed stick appears on a high one. let first_seen = stamp_of(&log, "xHCI: port ")?; - // What makes the bracket the settle's own: `await_connect_settle` anchors on - // the greatest `powered_at` across controllers, which on a second controller - // is not the instant these two lines bracket. - if !log.contains("xHCI: 1 controller(s),") { - return Err(format!( - "this profile grew a second controller, so the settle no longer anchors on the \ - `powered_at` these lines bracket\n{log}" - )); - } // The floor, and the non-vacuity with it: a driver that did not wait, or an // injection that did not land, names a port within a millisecond of the scan // rather than after the held-empty window and the debounce behind it. @@ -1328,17 +1310,6 @@ pub fn xhci_slow_connect( injection did not reach the driver\n{log}" )); } - // The ceiling. - let after_power = first_seen - powered; - if after_power > SETTLE_CEILING_S { - return Err(format!( - "the first port was named {after_power:.3} s after the ports were powered, {:.3} s \ - after the connect became visible — the settle did not end on the device \ - appearing\n{log}", - after_power - FIRST_CONNECT_S - )); - } - // And it found everything, and the bytes are the host's. if !log.contains("usb-storage: 2 device(s)") { return Err(format!("the driver did not bind both sticks after the wait\n{log}")); @@ -1376,9 +1347,8 @@ pub fn xhci_slow_connect( let _ = std::fs::remove_file(&image); eprintln!( - " [usb] controller started at {started:.3} s and powered its ports at {powered:.3} s; \ - first port named at {first_seen:.3} s, {after_start:.3} s after the start and \ - {after_power:.3} s after the power, both sticks bound, host bytes verified host-side; \ + " [usb] controller started at {started:.3} s; first port named at {first_seen:.3} s, \ + {after_start:.3} s after the start, both sticks bound, host bytes verified host-side; \ Boot: complete at {boot_ms} ms" ); Ok(()) @@ -1604,19 +1574,6 @@ pub fn usb_transport_break( // `SCSI 0x35`, slot 1, 2.3 s after the gate had swept — pushed the total // from the injected disk's real 2 to 3 and reddened a run in which the // disk under test never left its budget. - // The kernel's own clock against the claim, on a channel the message does - // not write: every record carries `[kernel ]`, so a wait that had - // really spent `USB_TIMEOUT_NS` would put two seconds between the break and - // the record before it. Measured here rather than asserted from the wording. - let waited = elapsed_before(&log, staged[0])?; - if waited >= 2.0 { - return Err(format!( - "the staged break took {waited:.3} s, which is the transfer budget — the wait it \ - is supposed to skip really ran\n{log}" - )); - } - eprintln!(" [usb] the staged break waited {waited:.3} s, not the 2 s it used to claim"); - let under_test = broke_on(staged[0])?; // And the driver got over it. Two attempts are explained by the fault — the @@ -1893,17 +1850,10 @@ enum Moved { FlushedStick, /// The stick itself, which `usb-return-silent` has come back late in the /// held call and answer nothing on the operation sent again on it: every - /// wait of it spins to its end, and the call still ends inside its bound, - /// timed from the break. + /// wait of it spins to its end. SilentReturn, } -/// What one disk call may spin for from the wait its transport broke on, in -/// seconds: the bound `toyos_xhci::call` keeps every path of a call inside. -fn call_after_break_secs() -> f64 { - toyos_xhci::call::AFTER_BREAK.whole() as f64 / 1e9 -} - /// The boot stick's first WRITE(10) is abandoned, the ladder resets its port, /// and the device leaves that port and binds on another. **QEMU cannot move a device on a reset**, so `usb-reset-moves` /// holds the port rung's reset until the port reads empty and the host makes @@ -1917,8 +1867,7 @@ fn call_after_break_secs() -> f64 { /// device left was before. /// /// **The call held for the stick only waits.** It spins with `IF` clear, so -/// the bind is another CPU's, and the call ends inside [`call_after_break_secs`] of -/// the break however slow the bind is — both read off the kernel's own stamps. +/// the bind is another CPU's, read off the kernel's own stamps. fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { const HELD: &str = "is held empty for the host to move its device (usb-reset-moves)"; const MOVE_NOW: &str = "usb-reset-moves: move the device now"; @@ -2048,13 +1997,6 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { let line = held_end; let ended = stamp_of(line, "[kernel ")?; let took = ended - stamp_of(&log, staged)?; - let bound = call_after_break_secs(); - if took > bound { - return Err(format!( - "{moved:?}: the call its transport broke in ended {took:.3} s after the break, past \ - the {bound} s a call may spin for\n{log}" - )); - } let (call_cpu, bind_cpu) = (cpu_of(line)?, cpu_of(line_with(&log, bind)?)?); let during = stamp_of(&log, bind)? <= ended; if during && call_cpu == bind_cpu { @@ -2334,14 +2276,13 @@ fn abandoned_write_is_taken_offline( return Err(format!("slot {slot} never went back after the give-up\n{log}")); } - // The clock, against the staging's claim: each rung really spent its bound - // before the next began, so the last one ran with the others' all gone. + // The clock, against the staging's claim: the port reset's rung really + // spent its bound, so the last rung ran with it gone. let stamp = |l: &str| -> Option { l.split_once("[kernel ")?.1.split_whitespace().next()?.parse().ok() }; let at = |needle: &str| log.lines().find(|l| l.contains(needle)).and_then(stamp); - let (Some(broke), Some(unverified), Some(ended)) = - (stamp(staged), at(rungs[0].as_str()), stamp(said)) + let (Some(broke), Some(unverified)) = (stamp(staged), at(rungs[0].as_str())) else { return Err(format!("a rung's record carries no kernel timestamp\n{log}")); }; @@ -2352,13 +2293,6 @@ fn abandoned_write_is_taken_offline( unverified - broke )); } - if ended - broke >= 2.75 { - return Err(format!( - "the ladder took {:.3} s from the break to the offline line; the two bounds it \ - ran on sum to 2 s\n{log}", - ended - broke - )); - } no_command_was_refused(&log)?; if !log.contains("Boot: complete") { return Err(format!("the boot did not finish after the give-up\n{log}")); diff --git a/tests/logstallcase/system.toml b/tests/logstallcase/system.toml index 7ad28116336..d6abb204fee 100644 --- a/tests/logstallcase/system.toml +++ b/tests/logstallcase/system.toml @@ -1,8 +1,8 @@ # The shared boot's shape with `/system/bin/logd` leaving soundd's log ring # unread until `test_rs_soundd_log_stall` says the tone has played: the ring # fills and stays full while it plays, which is the state its mix thread must -# never wait in. `soundd_log_stall` runs it, and reads the verdict off `/log` after -# `run shutdown`. +# never wait in. The `soundd_log_stall` metal row runs it, and reads the verdict +# off `/log`. [boot] start = ["logd", "soundd", "test-runner"] @@ -21,12 +21,8 @@ serves = ["soundd"] devices = ["hda-audio", "virtio-sound"] syscap = ["rt"] -# `power` is what `run shutdown` asks init through. [programs.test-runner] -receives = ["soundd", "power"] +receives = ["soundd"] [programs.toybox] receives = ["soundd"] - -[symlinks] -"bin/shutdown" = "/system/bin/toybox" diff --git a/tests/metal-profile.toml b/tests/metal-profile.toml index fa0ece05879..857681a6024 100644 --- a/tests/metal-profile.toml +++ b/tests/metal-profile.toml @@ -361,6 +361,24 @@ ceiling = 60000 ceiling_from = "toyos_tco::JOB_BOUND_MS — as boot.testcases.complete_ms" measured = 1254 +[[number]] +name = "boot.doomcase.complete_ms" +unit = "ms" +ceiling = 60000 +ceiling_from = "as boot.foreignrecord.complete_ms" + +[[number]] +name = "boot.doommusiccase.complete_ms" +unit = "ms" +ceiling = 60000 +ceiling_from = "as boot.foreignrecord.complete_ms" + +[[number]] +name = "boot.logstallcase.complete_ms" +unit = "ms" +ceiling = 60000 +ceiling_from = "as boot.foreignrecord.complete_ms" + [[number]] name = "boot.foreignrecord.back_secs" unit = "s" @@ -368,6 +386,24 @@ ceiling = 420 ceiling_from = "toyos_build::metal::return_secs" measured = 44 +[[number]] +name = "boot.doomcase.back_secs" +unit = "s" +ceiling = 420 +ceiling_from = "as boot.foreignrecord.back_secs" + +[[number]] +name = "boot.doommusiccase.back_secs" +unit = "s" +ceiling = 420 +ceiling_from = "as boot.foreignrecord.back_secs" + +[[number]] +name = "boot.logstallcase.back_secs" +unit = "s" +ceiling = 420 +ceiling_from = "as boot.foreignrecord.back_secs" + [[number]] name = "boot.foreignrecord.stick_secs" unit = "s" @@ -382,6 +418,24 @@ measured = 0 # that bound rather than against the runner's, because the runner never reaches # its own. +[[number]] +name = "boot.doomcase.stick_secs" +unit = "s" +ceiling = 30 +ceiling_from = "as boot.testcases.stick_secs" + +[[number]] +name = "boot.doommusiccase.stick_secs" +unit = "s" +ceiling = 30 +ceiling_from = "as boot.testcases.stick_secs" + +[[number]] +name = "boot.logstallcase.stick_secs" +unit = "s" +ceiling = 30 +ceiling_from = "as boot.testcases.stick_secs" + [[number]] name = "boot.deadlinewedge.complete_ms" unit = "ms" @@ -624,6 +678,24 @@ unit = "ms" ceiling = 1400 ceiling_from = "as list.testcases.job_ms" +[[number]] +name = "list.doomcase.job_ms" +unit = "ms" +ceiling = 1400 +ceiling_from = "as list.testcases.job_ms" + +[[number]] +name = "list.doommusiccase.job_ms" +unit = "ms" +ceiling = 1400 +ceiling_from = "as list.testcases.job_ms" + +[[number]] +name = "list.logstallcase.job_ms" +unit = "ms" +ceiling = 1400 +ceiling_from = "as list.testcases.job_ms" + [[number]] name = "list.deadlinewedge.job_ms" unit = "ms" @@ -862,6 +934,24 @@ unit = "us" ceiling = 16667 ceiling_from = "as boot.testcases.panel_max_us" +[[number]] +name = "boot.doomcase.panel_max_us" +unit = "us" +ceiling = 16667 +ceiling_from = "as boot.testcases.panel_max_us" + +[[number]] +name = "boot.doommusiccase.panel_max_us" +unit = "us" +ceiling = 16667 +ceiling_from = "as boot.testcases.panel_max_us" + +[[number]] +name = "boot.logstallcase.panel_max_us" +unit = "us" +ceiling = 16667 +ceiling_from = "as boot.testcases.panel_max_us" + [[number]] name = "boot.deadlinewedge.panel_max_us" unit = "us" @@ -985,6 +1075,24 @@ unit = "us" ceiling = 100000 ceiling_from = "as boot.testcases.panel_us" +[[number]] +name = "boot.doomcase.panel_us" +unit = "us" +ceiling = 100000 +ceiling_from = "as boot.testcases.panel_us" + +[[number]] +name = "boot.doommusiccase.panel_us" +unit = "us" +ceiling = 100000 +ceiling_from = "as boot.testcases.panel_us" + +[[number]] +name = "boot.logstallcase.panel_us" +unit = "us" +ceiling = 100000 +ceiling_from = "as boot.testcases.panel_us" + [[number]] name = "boot.deadlinewedge.panel_us" unit = "us" @@ -1100,6 +1208,24 @@ unit = "operations" ceiling = 0 ceiling_from = "as boot.testcases.park_open_operations" +[[number]] +name = "boot.doomcase.park_open_operations" +unit = "operations" +ceiling = 0 +ceiling_from = "as boot.testcases.park_open_operations" + +[[number]] +name = "boot.doommusiccase.park_open_operations" +unit = "operations" +ceiling = 0 +ceiling_from = "as boot.testcases.park_open_operations" + +[[number]] +name = "boot.logstallcase.park_open_operations" +unit = "operations" +ceiling = 0 +ceiling_from = "as boot.testcases.park_open_operations" + [[number]] name = "boot.usbbreak.park_open_operations" unit = "operations" diff --git a/tests/test-durations b/tests/test-durations index d76f46dec0c..6aa79b4dc5d 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -136,9 +136,6 @@ abuse_tls_alloc 15 acpi_table_inventory 4601 allocator_stress 787 apps_and_home_are_one_filesystem 7263 -audio_idle_suspend 1013 -audio_tone (smp=1) 8450 -audio_tone_load (smp=1) 18139 bar_placement_is_proven 3297 blackbox_done_chain 4119 blackbox_early_panic_sealed 2470 @@ -166,7 +163,6 @@ debug_float 11 debug_trap 25 demand_paging_sse 12 demand_window_race 54 -desktop_audio_client 13791 desktop_locale_detect 4510 desktop_typing_damage 14338 desktop_window_child 40312 @@ -174,8 +170,6 @@ device_claim_lifetime 21 disk_backtrace 85 diskless_boot 4475 dlopen_dedup 40 -doom_music 7514 -doom_sound_flood 5888 double_fault_stack 5139 double_panic_names_the_fault 9120 driver_wait_refused 17739 @@ -206,9 +200,6 @@ handle_transfer 66 hang_bounded_by_the_stick 3717 hard_lockup_ends_a_deaf_cpu 20277 hash_seed_precedes_every_map 4976 -hda_client_stall 19047 -hda_tone 13389 -hda_two_live_refused 4848 heap_ceiling_recovery 10371 hierarchy_paths 87 home_backing_revoked 663 @@ -287,7 +278,6 @@ metal_sim_compositor 8986 metal_sim_compositor_stall 11390 metal_sim_input 5065 metal_sim_ipc_hostile_peer 226 -metal_sim_null_audio 8311 metal_sim_pointer_churn 16658 metal_sim_scanout_wc 0 metal_sim_window_caps 148 @@ -302,8 +292,6 @@ netd_gone_mid_bind 32 netd_hostile_peer 4509 netd_listener_forgery 2649 nightly_tier_is_announced 0 -null_sink_client_exits 2167 -null_sink_shipped_client 7095 nvme_home_roundtrip 13 nvme_image_is_held_by_one_guest 0 nvme_large_device 6052 @@ -418,7 +406,6 @@ va_exhaustion 6159 virtio_net_no_msix 2039 virtio_used_ring 4189 volume_from_another_disk 7122 -wake_storm_cost 337 wall_clock_century_register 9030 wall_clock_file 4642 wall_clock_no_century 6987 diff --git a/tests/toyos-rust-tests/src/bin/audio_idle_suspend.rs b/tests/toyos-rust-tests/src/bin/audio_idle_suspend.rs index 67ef61de860..5475f6bdafa 100644 --- a/tests/toyos-rust-tests/src/bin/audio_idle_suspend.rs +++ b/tests/toyos-rust-tests/src/bin/audio_idle_suspend.rs @@ -2,10 +2,7 @@ //! connects, soundd's CPU cost is exactly zero. Two sysinfo samples ~1s apart //! must show no cpu_ns movement on any soundd thread — a suspended soundd //! holds no timer and takes no wakes, so any nonzero delta is the mix or -//! control loop running without a reason. This is the one idle-suspend claim gate A -//! structurally cannot see: its counters are streaming-scoped, and its boots -//! always connect a client. No wav analysis — there is no signal, and the -//! capture freezes while the voice is stopped anyway. +//! control loop running without a reason. //! //! Another process's threads is the process roster, which costs //! `Rights::ROSTER` on a `SysCap` — `tests/testcases` names `roster` on the diff --git a/tests/toyos-rust-tests/src/bin/audio_tone.rs b/tests/toyos-rust-tests/src/bin/audio_tone.rs index 9403deedfcd..f28fe525b2d 100644 --- a/tests/toyos-rust-tests/src/bin/audio_tone.rs +++ b/tests/toyos-rust-tests/src/bin/audio_tone.rs @@ -1,6 +1,4 @@ -//! Audio glitch regression test, idle variant: play a deterministic 440Hz -//! sine on an otherwise idle system. The host harness asserts the wav the -//! virtio-sound device captured is glitch-free. +//! Play a deterministic 440Hz sine to the end. #[path = "../tone.rs"] mod tone; diff --git a/tests/toyos-rust-tests/src/bin/audio_tone_load.rs b/tests/toyos-rust-tests/src/bin/audio_tone_load.rs deleted file mode 100644 index d096a77f092..00000000000 --- a/tests/toyos-rust-tests/src/bin/audio_tone_load.rs +++ /dev/null @@ -1,47 +0,0 @@ -//! Audio glitch regression test, load variant: play the tone while two pure -//! busy-spin processes saturate the (single) CPU. Glitch-free playback under -//! load is what the scheduler's audio priority handling must guarantee. - -#[path = "../tone.rs"] -mod tone; - -use std::process::Command; -use std::time::{Duration, Instant}; - -/// Outlives tone startup + 3s playback + drain even under heavy contention. -const BURN_SECS: u64 = 6; - -fn main() { - if std::env::args().nth(1).as_deref() == Some("burn") { - burn(Duration::from_secs(BURN_SECS)); - return; - } - - let burners: Vec<_> = (0..2) - .map(|_| { - Command::new("/system/bin/test_rs_audio_tone_load") - .arg("burn") - .spawn() - .expect("spawn burner") - }) - .collect(); - - tone::play_tone(); - println!("tone done"); - - for mut burner in burners { - burner.wait().expect("wait burner"); - } -} - -fn burn(duration: Duration) { - let start = Instant::now(); - let mut i = 0u64; - loop { - i = i.wrapping_add(1); - // Check the clock rarely so the load stays pure CPU, not syscalls. - if i % (1 << 22) == 0 && start.elapsed() >= duration { - return; - } - } -} diff --git a/tests/toyos-rust-tests/src/bin/blocking_read_stress.rs b/tests/toyos-rust-tests/src/bin/blocking_read_stress.rs index 4297aab87cc..474a5c5bbd3 100644 --- a/tests/toyos-rust-tests/src/bin/blocking_read_stress.rs +++ b/tests/toyos-rust-tests/src/bin/blocking_read_stress.rs @@ -2,32 +2,19 @@ //! //! Every round trip here is two parks and two posts on the completion core — //! the reader blocks in `sys_read` on an empty pipe, the writer's post wakes -//! it, and the same happens back the other way. **The verdict is a count -//! inside a wall-clock bound**, never a hang: a stall is disqualified as a -//! verdict, because the harness prints "the guard expired, so this says -//! nothing about the tree" beside one and tells nobody to bisect it. A -//! dropped completion reds as a number that is short of `ROUNDS`, with the -//! round it stopped at named. +//! it, and the same happens back the other way. **The verdict is the count of +//! round trips**: a dropped completion parks this process, and the harness's +//! ceiling turns that into a red. //! //! The echo half is this same binary with an argument, so the two ends are one //! file and the child's own parks are the same parks the parent's are. use std::io::{Read, Write}; use std::process::{Command, Stdio}; -use std::sync::atomic::{AtomicU32, Ordering}; -use std::thread; -use std::time::{Duration, Instant}; -/// Round trips. Enough that a wake lost at some rate shows up, small enough -/// that the whole thing fits inside the harness's five seconds with room — -/// measured at 500 rounds in 68 ms on the dev host, so this is two orders of -/// magnitude of headroom. +/// Round trips. Enough that a wake lost at some rate shows up. const ROUNDS: u32 = 500; -/// What the round trips must fit in. Not a latency assertion: it is what turns -/// a lost wake into a *number* rather than into a stall the suite names apart. -const BOUND: Duration = Duration::from_secs(3); - fn echo() -> ! { let mut stdin = std::io::stdin(); let mut stdout = std::io::stdout(); @@ -59,30 +46,6 @@ fn main() { let mut to_child = child.stdin.take().expect("piped stdin"); let mut from_child = child.stdout.take().expect("piped stdout"); - // **The watchdog is what makes a lost wake a number.** Without it a - // dropped completion parks this process for ever, the harness's guard - // expires, and the suite prints `STALL` — which is disqualified as a - // verdict, because it is the one class the harness names apart and tells - // nobody to bisect. With it, the round the machine stopped at is the - // failure message. - static DONE: AtomicU32 = AtomicU32::new(0); - thread::spawn(|| { - thread::sleep(BOUND); - // **Ends the process, rather than panicking this thread.** A thread - // panic leaves `main` parked in the read that never came back and the - // harness times the guest out — which is the stall this watchdog - // exists to replace. Verified both ways: with the pipe's readable post - // dropped, `panic!` here produced `timed out after 8s` and this - // produces the count. - eprintln!( - "blocking_read_stress: only {} of {ROUNDS} round trips completed inside {BOUND:?} \ - — a wake was not delivered", - DONE.load(Ordering::Relaxed), - ); - std::process::exit(1); - }); - - let started = Instant::now(); let mut completed = 0u32; for round in 0..ROUNDS { let sent = [(round % 251) as u8; 1]; @@ -94,9 +57,7 @@ fn main() { .unwrap_or_else(|e| panic!("round {round} of {ROUNDS} never came back: {e}")); assert_eq!(got, sent, "round {round} came back as another byte"); completed += 1; - DONE.store(completed, Ordering::Relaxed); } - let elapsed = started.elapsed(); drop(to_child); let status = child.wait().expect("wait for the echo half"); @@ -106,10 +67,5 @@ fn main() { completed, ROUNDS, "only {completed} of {ROUNDS} round trips completed", ); - assert!( - elapsed < BOUND, - "{ROUNDS} round trips took {elapsed:?}, past the {BOUND:?} bound — a wake is being \ - waited out rather than delivered", - ); - println!("blocking_read_stress: {completed} round trips in {elapsed:?}"); + println!("blocking_read_stress: {completed} round trips"); } diff --git a/tests/toyos-rust-tests/src/bin/doom_music.rs b/tests/toyos-rust-tests/src/bin/doom_music.rs index a9603dfbdde..aa23d1b1e15 100644 --- a/tests/toyos-rust-tests/src/bin/doom_music.rs +++ b/tests/toyos-rust-tests/src/bin/doom_music.rs @@ -3,8 +3,7 @@ //! The playing is `sound::music_check` in `userland/doom`: it opens the shipped //! SoundFont, converts one of the WAD's own MUS lumps and pushes the result at //! the audio device. Only that binary can reach any of it. This side starts it -//! and answers the question a serial log cannot — whether it exited or died — -//! while the verdict on the sound is the host's capture of the device. +//! and answers the question a log line cannot — whether it exited or died. use std::process::Command; diff --git a/tests/toyos-rust-tests/src/bin/exit_wait_storm.rs b/tests/toyos-rust-tests/src/bin/exit_wait_storm.rs index d25dbe313d8..6ea2d307430 100644 --- a/tests/toyos-rust-tests/src/bin/exit_wait_storm.rs +++ b/tests/toyos-rust-tests/src/bin/exit_wait_storm.rs @@ -8,10 +8,8 @@ //! ordering rather than volume — `process_lifecycle` has one arm on the wake //! and `std_threading` joins four threads. //! -//! **The verdict is a count of collected exit codes inside a bound**, the same -//! shape `blocking_read_stress` takes and for the same reason: a lost publish -//! must red as a number rather than as a stall the suite names apart and -//! nobody bisects. +//! **The verdict is a count of collected exit codes**; a lost publish parks +//! this process, and the harness's ceiling turns that into a red. //! //! **A child parks until the parent releases it, and that is what makes the //! parent's wait a park.** A child on its own schedule has published its exit @@ -22,9 +20,8 @@ use std::io::Read; use std::process::{Command, Stdio}; -use std::sync::atomic::{AtomicU32, Ordering}; use std::thread; -use std::time::{Duration, Instant}; +use std::time::Duration; /// Children spawned, held on their own stdin, and released together. const CHILDREN: u32 = 24; @@ -34,20 +31,6 @@ const CHILDREN: u32 = 24; /// one of those parks ended. const THREADS: u32 = 24; -/// What the storm must fit in. The spawns that set it up are outside it: they -/// are ELF loads, and a bound over them measures the loader. -const BOUND: Duration = Duration::from_secs(3); - -/// A liveness allowance for the spawns and never a measurement of them: it -/// turns a wedged setup into a number instead of the harness's stall. -const SETUP: Duration = Duration::from_secs(30); - -/// Which phase the watchdog names, and how far each got. -static SPAWNING: AtomicU32 = AtomicU32::new(1); -static SPAWNED: AtomicU32 = AtomicU32::new(0); -static COLLECTED: AtomicU32 = AtomicU32::new(0); -static JOINED: AtomicU32 = AtomicU32::new(0); - fn main() { if let Some(code) = std::env::args().nth(1) { // The child half: park in `read` until the parent drops the write end, @@ -59,28 +42,6 @@ fn main() { let exe = std::env::current_exe().expect("current_exe failed"); - thread::spawn(|| { - thread::sleep(SETUP + BOUND); - // Ends the process rather than this thread, for - // `blocking_read_stress`'s reason: a thread panic leaves `main` parked - // on the publish that never came, and the harness reports a stall. - if SPAWNING.load(Ordering::Relaxed) == 1 { - eprintln!( - "exit_wait_storm: {} of {CHILDREN} children spawned in {SETUP:?} — the storm \ - never started", - SPAWNED.load(Ordering::Relaxed), - ); - } else { - eprintln!( - "exit_wait_storm: {} of {CHILDREN} exits collected and {} of {THREADS} threads \ - joined inside {BOUND:?} — a publish was not delivered", - COLLECTED.load(Ordering::Relaxed), - JOINED.load(Ordering::Relaxed), - ); - } - std::process::exit(1); - }); - let mut children = Vec::new(); let mut held = Vec::new(); for i in 0..CHILDREN { @@ -91,7 +52,6 @@ fn main() { .unwrap_or_else(|e| panic!("spawn child {i}: {e}")); held.push(child.stdin.take().expect("the spawn was asked for a pipe")); children.push((i, child)); - SPAWNED.store(i + 1, Ordering::Relaxed); } // The premise, asserted rather than left to timing: nothing has released a @@ -103,8 +63,6 @@ fn main() { ); } - SPAWNING.store(0, Ordering::Relaxed); - let started = Instant::now(); drop(held); let mut collected = 0u32; @@ -116,7 +74,6 @@ fn main() { "child {i} answered with another process's code", ); collected += 1; - COLLECTED.store(collected, Ordering::Relaxed); } // The thread half: each thread returns its own number, and the join is the @@ -134,16 +91,9 @@ fn main() { let got = handle.join().unwrap_or_else(|_| panic!("join thread {i}")); assert_eq!(got, i as u32, "thread {i} answered for another one"); joined += 1; - JOINED.store(joined, Ordering::Relaxed); } - let elapsed = started.elapsed(); assert_eq!(collected, CHILDREN, "only {collected} of {CHILDREN} exits were collected"); assert_eq!(joined, THREADS, "only {joined} of {THREADS} threads were joined"); - assert!( - elapsed < BOUND, - "the storm took {elapsed:?}, past the {BOUND:?} bound — a publish is being waited out \ - rather than delivered", - ); - println!("exit_wait_storm: {collected} exits collected and {joined} threads joined in {elapsed:?}"); + println!("exit_wait_storm: {collected} exits collected and {joined} threads joined"); } diff --git a/tests/toyos-rust-tests/src/bin/inspect_plays.rs b/tests/toyos-rust-tests/src/bin/inspect_plays.rs deleted file mode 100644 index 7a230b7f52d..00000000000 --- a/tests/toyos-rust-tests/src/bin/inspect_plays.rs +++ /dev/null @@ -1,52 +0,0 @@ -//! Plays periods through soundd and says how many frames soundd provably took, -//! for `inspect_reads_its_owners` to hold `sound.periods.*` against. -//! -//! **What is proven taken is what the ring says, not what was written.** A -//! slot can be filled only once soundd has emptied it, so of `fills` slots -//! filled into a ring of `slots`, at least `fills - slots` were taken, each in -//! a mix pass that submitted the periods it went out in and then published its -//! counters. soundd signals every client at the top of every pass and a read -//! drains every signal waiting, so the second signal read after the last -//! counted fill was written by a pass that began after that fill: every pass -//! that took a counted slot had published by then. -//! -//! At the device's own rate and channel count, so a client frame is a device -//! frame and no resampler stands between the two counts. - -use toyos::audio::{AudioStream, FORMAT_S16LE}; - -const RATE: u32 = 44_100; -const CHANNELS: u16 = 2; -/// Slots filled past the ring's first fill: every one is a slot soundd took. -const TAKEN_SLOTS: u64 = 32; - -fn main() { - let mut stream = AudioStream::open(RATE, CHANNELS, FORMAT_S16LE).expect("open a stream on soundd"); - assert_eq!(stream.device_sample_rate(), RATE, "the device is not at the client's rate"); - assert_eq!(stream.device_channels(), CHANNELS, "the device is not the client's channel count"); - let period_frames = u64::from(stream.period_frames()); - - // The first signal comes with the stream and finds the ring empty: that - // fill is the ring's own size, and says nothing about what soundd took. - let mut slots = 0u64; - stream - .wait_and_fill(|buf| { - buf.fill(0x11); - slots += 1; - }) - .expect("the first fill"); - let mut fills = slots; - while fills < slots + TAKEN_SLOTS { - stream - .wait_and_fill(|buf| { - buf.fill(0x11); - fills += 1; - }) - .expect("soundd went away mid-stream"); - } - for _ in 0..2 { - stream.wait_and_fill(|buf| buf.fill(0)).expect("soundd went away before publishing"); - } - stream.close(); - println!("inspect plays: soundd took {} frames", (fills - slots) * period_frames); -} diff --git a/tests/toyos-rust-tests/src/bin/netd_lookup_let_go.rs b/tests/toyos-rust-tests/src/bin/netd_lookup_let_go.rs index c93fec198fa..91564feb21c 100644 --- a/tests/toyos-rust-tests/src/bin/netd_lookup_let_go.rs +++ b/tests/toyos-rust-tests/src/bin/netd_lookup_let_go.rs @@ -13,7 +13,7 @@ //! //! **Spoke again.** A connection carries one request, so a client that says //! more while its lookup is in flight is dropped: its connection closes with -//! no answer, long before the schedule would have answered it. +//! no answer, where the schedule would have answered it timed out. //! //! `netd_lookup_let_go: ok` is the only success line. @@ -83,7 +83,6 @@ fn spoke_again() { let answered = closed_or_answered(&chatty); let took = spoke.elapsed(); assert_eq!(answered, 0, "a client that spoke again while its lookup ran was answered, after {took:?}"); - assert!(took < SCHEDULE, "a client that spoke again was dropped after {took:?}, not at once"); println!("netd_lookup_let_go: a client that spoke again was dropped after {took:?}, unanswered"); drop(held); } diff --git a/tests/toyos-rust-tests/src/bin/netd_stalled_peer.rs b/tests/toyos-rust-tests/src/bin/netd_stalled_peer.rs deleted file mode 100644 index 5f0e5b180c5..00000000000 --- a/tests/toyos-rust-tests/src/bin/netd_stalled_peer.rs +++ /dev/null @@ -1,85 +0,0 @@ -//! A client that out-writes a peer which has stopped reading costs netd no -//! CPU. -//! -//! The host reads this connection's ask and nothing after it. This program -//! writes until the whole path — its send pipe, netd's socket buffer, slirp and -//! the host's socket — has stopped taking bytes, and then measures how busy the -//! machine is while nothing can move. A netd that watches the send pipe -//! whether or not its socket has room is woken by the pipe's bytes on every -//! pass, which is one whole CPU for as long as the peer stays stalled. -//! -//! argv[1] is the port of the harness's host server on `HOST`. -//! `netd_stalled_peer: ok` is the only success line. - -#[path = "../netd_stream.rs"] -mod netd_stream; - -use std::time::{Duration, Instant}; - -use netd_stream::{ask, Ask, HOST}; -use toyos::poller::{Poller, WRITABLE}; -use toyos_abi::syscall::{self, SyscallError, SysinfoHeader}; - -/// How long the send pipe must stay full, with the peer reading nothing, for -/// the path to count as stalled. Policy: every hop drains in far less. -const SETTLE: Duration = Duration::from_secs(2); - -/// The window the machine's busy time is measured over. -const WINDOW: Duration = Duration::from_secs(2); - -/// Liveness guard on reaching the stall at all. -const STALL_BOUND: Duration = Duration::from_secs(60); - -fn sysinfo() -> SysinfoHeader { - let mut buf = [0u8; toyos::system::SYSINFO_HEADER_SIZE]; - let n = toyos::system::sysinfo(&mut buf); - assert!(n >= toyos::system::SYSINFO_HEADER_SIZE, "sysinfo returned {n} bytes"); - SysinfoHeader::decode(&buf) -} - -fn main() { - let port: u16 = std::env::args() - .nth(1) - .and_then(|p| p.parse().ok()) - .expect("usage: netd_stalled_peer "); - let conn = toyos::net::tcp_connect(HOST, port, 30_000).expect("connect to the host server"); - ask(&conn.tx, Ask::Held(0)); - - let chunk = [0u8; 65536]; - let poller = Poller::new(1); - let started = Instant::now(); - let mut written = 0u64; - loop { - match conn.tx.write_nonblock(&chunk) { - Ok(n) => { - written += n as u64; - continue; - } - Err(SyscallError::WouldBlock) => {} - Err(e) => panic!("writing to the stalled peer after {written} bytes: {e:?}"), - } - assert!(started.elapsed() < STALL_BOUND, "the send path still took bytes after {STALL_BOUND:?}"); - // The pipe is full: room within `SETTLE` is the path still moving. - poller.watch(&conn.tx, WRITABLE, 0); - let mut roomed = false; - poller.wait(1, SETTLE.as_nanos() as u64, |_| roomed = true); - if !roomed { - break; - } - } - println!("netd_stalled_peer: the path stopped taking bytes after {written}"); - - let before = sysinfo(); - // An interval, not a wait: a rate is what is measured, and nothing is - // expected to happen in it. - syscall::nanosleep(WINDOW.as_nanos() as u64); - let after = sysinfo(); - let busy_ns = after.total_cpu_ns - before.total_cpu_ns; - let wall_ns = after.uptime_ns - before.uptime_ns; - let busy_cpus = busy_ns as f64 / wall_ns as f64; - println!("netd_stalled_peer: {busy_cpus:.3} CPUs busy over {wall_ns} ns, of {}", after.cpus); - // A netd polling a pipe it cannot drain is one whole CPU; half of one is - // far above an idle machine and far below that. - assert!(busy_cpus < 0.5, "the machine was {busy_cpus:.3} CPUs busy with every connection stalled"); - println!("netd_stalled_peer: ok, {busy_cpus:.3} CPUs busy while stalled"); -} diff --git a/tests/toyos-rust-tests/src/bin/panic_halts_first.rs b/tests/toyos-rust-tests/src/bin/panic_halts_first.rs deleted file mode 100644 index 64580a5bdf4..00000000000 --- a/tests/toyos-rust-tests/src/bin/panic_halts_first.rs +++ /dev/null @@ -1,48 +0,0 @@ -//! Threads that make kernel records as fast as they can while this one has -//! the kernel go fatal (`SYS_DEBUG` action 3). A sibling still running after -//! the fatal path began is a record stamped after the fatal one; -//! `panic_halts_the_others_first` reads the console for one. - -use std::sync::atomic::{AtomicUsize, Ordering}; -use std::sync::Arc; -use std::time::{Duration, Instant}; - -/// A retired syscall's number: each call is refused and is one kernel record -/// naming it. -const RETIRED: u64 = 26; -/// One per other CPU of the boot that runs this. -const SIBLINGS: usize = 3; -/// A liveness guard on the siblings' first records. -const STARTED_WITHIN: Duration = Duration::from_secs(10); - -fn retired() { - let ret: u64; - // SAFETY: a register-only `syscall` whose number the kernel refuses - // without reading any argument; nothing in this process is touched. - unsafe { - core::arch::asm!("syscall", in("rdi") RETIRED, lateout("rax") ret, out("rcx") _, out("r11") _); - } - assert_ne!(ret, 0, "syscall {RETIRED} answered as if it were live"); -} - -fn main() { - let started = Arc::new(AtomicUsize::new(0)); - for _ in 0..SIBLINGS { - let started = Arc::clone(&started); - std::thread::spawn(move || { - retired(); - started.fetch_add(1, Ordering::Release); - loop { - retired(); - } - }); - } - let until = Instant::now() + STARTED_WITHIN; - while started.load(Ordering::Acquire) < SIBLINGS { - assert!(Instant::now() < until, "the siblings made no record in {STARTED_WITHIN:?}"); - std::hint::spin_loop(); - } - let rc = toyos_abi::syscall::debug(toyos_abi::syscall::debug_action::FATAL_HALT); - eprintln!("ERROR: SYS_DEBUG FATAL_HALT returned {rc:#x}"); - std::process::exit(1); -} diff --git a/tests/toyos-rust-tests/src/bin/poll_wake_pipe.rs b/tests/toyos-rust-tests/src/bin/poll_wake_pipe.rs index 01408546985..7d7ab0897aa 100644 --- a/tests/toyos-rust-tests/src/bin/poll_wake_pipe.rs +++ b/tests/toyos-rust-tests/src/bin/poll_wake_pipe.rs @@ -4,30 +4,22 @@ //! and the 7a cutover once deleted the second for two sources undetected. A //! watcher arms `POLL_ADD` READABLE on a pipe read end and blocks in `wait`, a //! writer writes each round, and every readable edge must wake the ring: a -//! dropped ring wake reds as a short count, which is the whole verdict. +//! dropped ring wake parks the watcher, and the harness's ceiling turns that +//! into a red. //! `blocking_read_stress` is the same canary for the other half. use std::sync::atomic::{AtomicU32, Ordering}; use std::thread; -use std::time::{Duration, Instant}; use toyos::pipe_pair; use toyos::poller::{Poller, READABLE}; const ROUNDS: u32 = 300; -/// The pacing spin's escape, so a lost wake reds as a short count, not a hang. -/// Past it the writer stops pacing, so the edges are no longer distinct and the verdict is only the count. -const PACE_ESCAPE: Duration = Duration::from_secs(3); - -/// Per-round patience, far above a live wake's latency, so only an absent completion trips it. -const WAIT_NANOS: u64 = 200_000_000; - fn main() { let (reader, writer) = pipe_pair().expect("a pipe"); let woken = AtomicU32::new(0); - let began = Instant::now(); thread::scope(|s| { s.spawn(|| { let poller = Poller::new(1); @@ -35,10 +27,8 @@ fn main() { for _ in 0..ROUNDS { poller.watch(&reader, READABLE, 0); let mut got = false; - poller.wait(1, WAIT_NANOS, |_| got = true); - if !got { - return; // the completion never arrived — the ring half was lost - } + poller.wait(1, u64::MAX, |_| got = true); + assert!(got, "a wait with no deadline returned with no completion"); woken.fetch_add(1, Ordering::Relaxed); let _ = reader.read(&mut buf); // drain so the next round arms empty } @@ -47,9 +37,7 @@ fn main() { // One byte per round, paced behind the watcher so every write is a distinct edge. s.spawn(|| { for round in 0..ROUNDS { - while woken.load(Ordering::Relaxed) < round - && began.elapsed() < PACE_ESCAPE - { + while woken.load(Ordering::Relaxed) < round { std::hint::spin_loop(); } if writer.write(&[0x5A]).is_err() { @@ -60,11 +48,9 @@ fn main() { }); let woken = woken.load(Ordering::Relaxed); - let elapsed = began.elapsed(); assert_eq!( woken, ROUNDS, - "the poll completed {woken} of {ROUNDS} readable edges in {elapsed:?} — a ring \ - watcher's wake was lost", + "the poll completed {woken} of {ROUNDS} readable edges — a ring watcher's wake was lost", ); - println!("poll_wake_pipe: {ROUNDS} readable edges each woke the armed ring in {elapsed:?}"); + println!("poll_wake_pipe: {ROUNDS} readable edges each woke the armed ring"); } diff --git a/tests/toyos-rust-tests/src/bin/soundd_log_stall.rs b/tests/toyos-rust-tests/src/bin/soundd_log_stall.rs index 20b7c2e6e63..60d8136a60b 100644 --- a/tests/toyos-rust-tests/src/bin/soundd_log_stall.rs +++ b/tests/toyos-rust-tests/src/bin/soundd_log_stall.rs @@ -9,10 +9,9 @@ //! is said where nobody reads. A mix thread that waits on its output stops //! there, and the tone with it. //! -//! The verdicts are the host's: the capture carries the tone without a gap, and -//! once `logd` reads again soundd's lines account for every connection — each -//! refusal said, or counted by `logd` among the records that found the ring -//! full. +//! The verdict is `/log`'s: once `logd` reads again soundd's lines account for +//! every connection — each refusal said, or counted by `logd` among the records +//! that found the ring full. #[path = "../tone.rs"] mod tone; diff --git a/tests/toyos-rust-tests/src/bin/tlb_shootdown_waits.rs b/tests/toyos-rust-tests/src/bin/tlb_shootdown_waits.rs index 51c3a4ee9b4..71b04e28cd2 100644 --- a/tests/toyos-rust-tests/src/bin/tlb_shootdown_waits.rs +++ b/tests/toyos-rust-tests/src/bin/tlb_shootdown_waits.rs @@ -30,7 +30,7 @@ const DELAY_NANOS: u64 = 20_000_000; /// Half the delay. Every number compared against it is a lower bound on a spin /// the kernel or this process performs, so it cannot come out short for /// scheduling reasons — but the clock reads bracketing it are syscalls, and the -/// margin is there so a slow host cannot turn a pass into a fail either way. +/// margin is there so a slow host cannot turn a pass into a fail. const FLOOR_NANOS: u64 = DELAY_NANOS / 2; /// How many measured `munmap`s must *all* return fast before that is the @@ -163,19 +163,5 @@ fn main() { disarm(); - // 3. And the delay is what produced every number above, not the machine: - // disarmed, the same operation is back to microseconds. Without this the - // assertions above would still pass on a kernel that happened to be slow - // for some other reason. - let quiet = map(PAGE_2M); - let elapsed = timed(|| { - unsafe { syscall::munmap(quiet, PAGE_2M) }.expect("munmap"); - }); - assert!( - elapsed < FLOOR_NANOS, - "munmap still took {elapsed}ns with the delay disarmed, so the numbers above \ - measured something other than the wait", - ); - println!("a shootdown waits for every other CPU, and munmap and a fixed mmap wait for it"); } diff --git a/tests/toyos-rust-tests/src/tone.rs b/tests/toyos-rust-tests/src/tone.rs index c12e29851b0..72e0e5f7e38 100644 --- a/tests/toyos-rust-tests/src/tone.rs +++ b/tests/toyos-rust-tests/src/tone.rs @@ -1,7 +1,4 @@ -//! Shared tone playback for the audio glitch tests (audio_tone, -//! audio_tone_load). The host-side harness records what the virtio-sound -//! device plays into a wav and asserts the tone contains no mid-signal -//! silence (underruns) and no hard discontinuities (clicks). +//! Shared tone playback: a deterministic sine. use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::sync::Arc; diff --git a/tests/toyos.rs b/tests/toyos.rs index f2ed0e90973..d282e7eb544 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -12,11 +12,10 @@ use common::qemu::{ STALLED, }; use common::{ - audio, compile, devices, faults, hostload, lan, metal, partclaim, pkg, power, screen, serial, - stats, storage, usb, + audio, compile, devices, faults, lan, metal, partclaim, pkg, power, screen, serial, storage, + usb, }; use toyos_build::bootlog::{self, boot_millis}; -use toyos_build::heartbeat; use toyos_build::testargs::{self, Shard, SUITE}; use toyos_build::redlist; use toyos_build::tiers::Tier; @@ -208,6 +207,9 @@ const RUST_SKIP: &[&str] = &[ // that measured anything. It also needs the real-time band, which only // `tests/latencycase` endows. `latency_wake` runs it there. "cyclictest", + // Its verdict is a ratio of cycle counts, which a guest's host moves: the + // `wake_storm_cost` metal row runs it on the T14. + "wake_storm_cost", // Its verdict is a property of the *console capture*, which only a boot of // its own can hold: in the shared boot every other binary's output is in the // same stream. `console_line_atomicity` runs it. @@ -262,8 +264,6 @@ const RUST_SKIP: &[&str] = &[ // `gsbase_locked`'s probe child; its #UD must kill the child, not the run. "gsbase_probe", "test_panic_child", - // It takes the machine down; `panic_halts_the_others_first` runs it. - "panic_halts_first", // A binary that panics at once, sent over ssh as a service's replacement; // `swap_crash_rolls_back` stages it from the host and never runs it as a job. "swap_crash", @@ -306,13 +306,12 @@ const RUST_SKIP: &[&str] = &[ // `netd_listener_forgery` runs it there. "netd_listener_forgery", // Needs a NIC in front of netd and a host server behind it. - // `netd_slow_reader`, `netd_held_open`, `netd_stalled_peer`, - // `netd_udp_refused`, `netd_udp_any_address` and `netd_refused_pipes` run - // them on `tests/netcase`, and `netd_lookup_let_go` on it with its frames + // `netd_slow_reader`, `netd_held_open`, `netd_udp_refused`, + // `netd_udp_any_address` and `netd_refused_pipes` run them on + // `tests/netcase`, and `netd_lookup_let_go` on it with its frames // held. "netd_slow_reader", "netd_held_open", - "netd_stalled_peer", "netd_udp_refused", "netd_udp_any_address", "netd_refused_pipes", @@ -368,9 +367,8 @@ const RUST_SKIP: &[&str] = &[ // Needs netd with a NIC. `netd_connection_caps` runs it on tests/netcase. "netd_caps", // Need every owner `inspect` reads, which only tests/inspectcase runs. - // `inspect_reads_its_owners` runs all three there. + // `inspect_reads_its_owners` runs both there. "inspect_denied", - "inspect_plays", "inventory_bounds", // Same reason, same config: `netd_hostile_peer` runs it there. "netd_hostile_peer", @@ -414,18 +412,6 @@ const RUST_SKIP: &[&str] = &[ // printed four hundred lines to a console nothing was reading and passed // on its exit code. "test_screen_churn", - // Spawns `/system/bin/doom`, which `tests/testcases` does not carry — doom is - // 4 MiB and every other test boots that config. `doom_sound_flood` runs it - // on `tests/doomcase`. - "doom_sound_flood", - // Same, plus the WAD and the SoundFont doom's music is made of, which no - // other config should pay 19 MiB of ROOT for. `doom_music` runs it on - // `tests/doommusiccase`. - "doom_music", - // Needs a `logd` that leaves soundd's ring unread until it says so, and a - // capture of the tone it plays into that. `soundd_log_stall` runs it on - // `tests/logstallcase`. - "soundd_log_stall", // Its failure mode is a CPU that never runs anything again, so on the // shared boot it would be reported against whichever test came next — and // every one after that. `short_sleep_livelock` gives it a boot of its own. @@ -461,13 +447,16 @@ const RUST_SKIP: &[&str] = &[ // Stages real directories on `/log` for `fs_dirs_durable` to read back // off the image. "fs_dirs_durable", - // Needs an HDA controller, which `tests/testcases` has none of. + // Audio is judged on the T14 and nowhere else: the `hda_client_stall`, + // `hda_tone`, `audio_idle_suspend`, `null_sink_shipped_client`, + // `doom_sound_flood`, `doom_music` and `soundd_log_stall` metal rows run these. "hda_client_stall", - // Gate A's two, whose verdict is the wav the device captured — which the - // shared boot takes no capture of. The comment below has claimed since it - // was written that they are excluded from this boot; now they are. "audio_tone", - "audio_tone_load", + "audio_idle_suspend", + "null_sink_client_exits", + "doom_sound_flood", + "doom_music", + "soundd_log_stall", // Its whole subject is a page of a file the host wrote onto the volume // before the machine existed; the shared boot stages nothing, so it prints // `did not open` and passes on its exit code. `log_backing_read_error` @@ -535,7 +524,6 @@ const DRIVEN_AND_SHARED: &[&str] = &[ // device was refused at the unit — and stages nothing for it. "handle_basic", "hierarchy_paths", - "null_sink_client_exits", "nvme_home_roundtrip", "sched_stress", "std_alloc", @@ -543,16 +531,6 @@ const DRIVEN_AND_SHARED: &[&str] = &[ "wall_clock_now", ]; -// Audio glitch tests. Each runs in its own QEMU boot per SMP config and -// asserts on the wav the virtio-sound device captured, so they are excluded -// from the shared multi-test boot. -const AUDIO_TESTS: &[(&str, Tier)] = - &[("audio_tone", Tier::Nightly), ("audio_tone_load", Tier::Nightly)]; - -// Scheduler-core gate A covers both SMP configs: smp=1 is the audio spec's -// first-class single-CPU case, smp=8 the full-SMP case. -const AUDIO_SMP: &[u32] = &[1, 8]; - /// What `test-early-panic` panics with (`kernel/src/main.rs`): the last line its /// report puts on serial. const EARLY_PANIC_MESSAGE: &str = "test-early-panic: on-screen console check"; @@ -682,11 +660,11 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // CPU the firmware named came up and none of their timestamp counters // trails the BSP's; the physical memory manager's accounting against the // firmware map balances to the byte; every ACPI table this kernel goes on - // to decode checksummed; the TSC the whole machine is timed by agrees with - // the frequency the part itself states; the PCI inventory's function count - // matches its rows; and one machine-wide TLB shootdown's cost is a - // distribution rather than one boot's average. Every verdict is arithmetic - // over records, with no clock in the judging, so all six are Parallel. + // to decode checksummed; the LAPIC timer and the TSC calibrated to a + // frequency; the PCI inventory's function count matches its rows; and one + // machine-wide TLB shootdown's cost is a distribution rather than one + // boot's average. Every verdict is arithmetic over records, with no clock + // in the judging, so all six are Parallel. ("smp_roster_and_tsc_trail", Sched::Parallel, Tier::Fast), ("pmm_accounting", Sched::Parallel, Tier::Fast), // ROOT is the loader's image in memory: the kernel says it mounted it @@ -712,12 +690,9 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // latency against a programmed timer: soundd's figure is a maximum over a // window, taken against a DLL's prediction of a DMA completion and needing // a sound card to exist at all, and `toyos-sched`'s bound on the same - // quantity runs in a simulator where no IPI is ever delivered. - // - // Serial: it is the one registration here whose verdict is a *time*, and a - // wake latency measured beside eleven other guests is the host's schedule. - // Nightly for that same reason. - ("latency_wake", Sched::Serial, Tier::Nightly), + // quantity runs in a simulator where no IPI is ever delivered. The number is + // judged on the T14; here the verdict is that both channels carry it. + ("latency_wake", Sched::Parallel, Tier::Nightly), ("smp_failed_ap_leaves_no_hole", Sched::Parallel, Tier::Fast), ("input_merge", Sched::Parallel, Tier::Fast), ("metal_sim_input", Sched::Parallel, Tier::Fast), @@ -752,27 +727,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // slow on purpose. Its own boot too: it leaves the pointer somewhere else // and the window in a different place than it found them. ("metal_sim_window_drag", Sched::Serial, Tier::Nightly), - // A host-measured drain rate with an 8 s ceiling on a 3.3 s expectation. - // Not gate A, but the same instrument: what it measures is how fast a - // client's audio leaves the machine. - ("metal_sim_null_audio", Sched::Serial, Tier::Nightly), - ("null_sink_shipped_client", Sched::Serial, Tier::Nightly), - // Parallel, and this one is argued rather than assumed: not a verdict in it - // is a wall-clock margin. The flood's size is asserted against the audio - // callback's own period counter standing still, both playback checks are - // counted in periods, and the capture is read for amplitude and never for - // timing. Its own boot, its own config, and the only client its soundd has. - ("doom_sound_flood", Sched::Parallel, Tier::Nightly), - // Reads a device capture and requires at least MIN_SIGNAL_SECS = 0.8 s of - // it to carry signal at peak >= 6000 — an absolute seconds-of-signal - // floor on audio recorded in real time, not a fraction of the capture and - // not compute-bound: timer-anchored, and Nightly for that reason. - ("doom_music", Sched::Parallel, Tier::Nightly), - // A tone played while soundd's pipe to a stalled log is full. The verdicts - // are the capture's gaps and a count of lines; the clocks are liveness - // guards. Nightly for the two megabytes of refusals the log reads back - // through a TCG guest's volume. - ("soundd_log_stall", Sched::Serial, Tier::Nightly), // ureq and rustls from crates.io, fetching over TLS 1.3 from a host this // test mints a CA for. Every verdict is a printed line or a digest; the // only clock is `run_test`'s ceiling. @@ -921,10 +875,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // verdict is the guest's byte-for-byte comparison; its clocks are // liveness guards. ("netd_held_open", Sched::Parallel, Tier::Fast), - // The netcase boot again: a client out-writing a peer that stopped reading - // costs netd no CPU. Nightly: its verdict is the machine's busy time over - // a window of real time. - ("netd_stalled_peer", Sched::Parallel, Tier::Nightly), // The netcase boot again, beside a host UDP echo: a datagram the client's // pipe will not take whole ends that socket by name and no other. The // verdict is each socket's answer; its clocks are liveness guards. @@ -1146,9 +1096,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // One boot, and its verdict is a line the kernel printed before any device // was brought up. No clock and no device in it. ("virtio_used_ring", Sched::Parallel, Tier::Fast), - // A fatal path with other CPUs running userland that makes kernel records: - // none is stamped past the fatal record by more than an IPI takes. - ("panic_halts_the_others_first", Sched::Parallel, Tier::Fast), // A kernel log line from PCI enumeration; no clock and no real device in it. ("pci_capability_walk", Sched::Parallel, Tier::Fast), // What QEMU was told to create against what the guest enumerated: two @@ -1203,22 +1150,8 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("gsbase_locked", Sched::Parallel, Tier::Nightly), // The fourth declared kernel build, booted so that the scheduler core's // `feature = "check"` instruments are compiled and executed by a CI run at - // all. One of its verdicts is a *quantile* of the guest's published - // pass-cost distribution, which is wall clock across a scheduler pass. - // - // **Serial, and it used to say `Parallel` for a reason that was wrong.** - // The old note read "a bound the guest measures against its own TSC inside - // a single scheduler pass, which no amount of host load lengthens. A pass - // is preempt-off by construction." Preempt-off stops the *guest's* - // scheduler and stops nothing above it: the guest's TSC advances while the - // host has the vCPU, which is why invariant P panicked on a KVM shard at - // 200569 ns and why it is a measurement now. Measured here, 2026-08-17, one - // suite: alone on a quiet host (1.02x the reference boot) cpu0 reports - // `168 passes, p50 < 16384 ns, p90 < 131072 ns` and passes; in the same - // run's 12-wide phase, `134 passes, p50 < 131072 ns, p90 < 262144 ns` and - // it reds. Host contention moves this guest's median by a factor of eight, - // which is the definition of a test that must have the machine to itself. - ("sched_check_build", Sched::Serial, Tier::Fast), + // all. + ("sched_check_build", Sched::Parallel, Tier::Fast), // What nesting a `scheduler::Operation` may and may not do, which is a law // with no host-side reader: the type reaches `percpu::cpu_id` and // `driver::current_handle`, so nothing outside a booted machine can @@ -1376,20 +1309,12 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("toolkit_window_wake", Sched::Parallel, Tier::Nightly), ("toolkit_winit_loop", Sched::Parallel, Tier::Nightly), ("toolkit_winit_pace", Sched::Parallel, Tier::Nightly), - // The same desktop with soundd behind it: an audio client spawned by a - // shell, which is the only place all three of its descriptors are pipes to - // a surface. Parallel — every verdict is a marker with its own ceiling, and - // none of them reads a clock. - ("desktop_audio_client", Sched::Parallel, Tier::Nightly), // Ctrl+Alt+D on the same machine. Parallel: it waits for a marker and its // verdicts are counts the report has to agree with itself about, not a // wall-clock margin — the one duration in it is the dump's own 250 ms // ceiling, which the guest spends and the host never measures. ("blocked_dump", Sched::Parallel, Tier::Fast), - // Two boots of one machine compared on the guest's own `Boot: complete` - // with a 300 ms allowance, which is the whole assertion — a real-time - // verdict, so Nightly. - ("i8042_absent", Sched::Serial, Tier::Nightly), + ("i8042_absent", Sched::Parallel, Tier::Nightly), // The fault quarantines (masks) the controller's GSI within milliseconds // of readiness — confirmed from the serial log, before a host round trip // could land anything — so no sentinel can ever reach the guest and the @@ -1439,12 +1364,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("usb_storage_write_error", Sched::Parallel, Tier::Fast), ("usb_flush_optional", Sched::Parallel, Tier::Nightly), ("xhci_deaf_registers", Sched::Parallel, Tier::Nightly), - // The window is anchored on the controller's own port-power stamp now, not - // boot, so a slow boot no longer eats it — but the bound is still a fixed - // span of the guest's own TSC clock (`SLOW_CONNECT_NS`/`DEBOUNCE_NS`), and - // a host running several other guests can still stall this one's vCPU past - // that span for reasons that are not the defect. - ("xhci_slow_connect", Sched::Serial, Tier::Nightly), + ("xhci_slow_connect", Sched::Parallel, Tier::Nightly), ("xhci_portsc_rw1c", Sched::Parallel, Tier::Fast), // One staged break and no other, which puts the driver's recovery finishing // on its first try in the verdict: a retried command that reaches an @@ -1478,10 +1398,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("log_flush_retry", Sched::Parallel, Tier::Nightly), ("toybox_cp_volume", Sched::Parallel, Tier::Nightly), ("kernel_log_file", Sched::Parallel, Tier::Nightly), - // Serial: its verdict is a cadence — heartbeats against a 250 ms period — - // and a guest sharing the host with eleven others reaches its idle loop - // late for reasons that are not the defect. - ("kernel_heartbeat", Sched::Serial, Tier::Nightly), + ("kernel_heartbeat", Sched::Parallel, Tier::Nightly), // Both own their images and their lanes, and neither verdict is a // wall-clock margin: the guest's clock starts from an instant the host set // and the only duration either measures is how long a boot takes to reach @@ -1593,15 +1510,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // standing; then, on a second boot, a function no release resets is never // lent where it was left aimed. ("blockd_lends_within_its_bound", Sched::Parallel, Tier::Nightly), - // H4: soundd driving an Intel HDA controller itself, read back off the - // device. Serial — its verdict is a wav capture, and one taken while eleven - // other guests contend for the host measures the host. - ("hda_tone", Sched::Serial, Tier::Nightly), - // The T14's panic, staged: a client that stops producing for longer than - // the DMA ring takes to come round. The verdict is soundd's own liveness - // and its counters rather than a capture, so it runs wide. - ("hda_client_stall", Sched::Parallel, Tier::Nightly), - ("hda_two_live_refused", Sched::Parallel, Tier::Fast), ("serial_vocabulary", Sched::Parallel, Tier::Fast), // Host-side, no guest: the harness asking whether it can still tell a // suspended machine from a slow one, and whether it reports one as a @@ -1684,7 +1592,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("netd_slow_reader", &["test_rs_netd_slow_reader"]), ("netd_refused_pipes", &["test_rs_netd_refused_pipes"]), ("netd_held_open", &["test_rs_netd_held_open"]), - ("netd_stalled_peer", &["test_rs_netd_stalled_peer"]), ("netd_udp_refused", &["test_rs_netd_udp_refused"]), ("netd_udp_any_address", &["test_rs_netd_udp_any_address"]), ("netd_lookup_let_go", &["test_rs_netd_lookup_let_go"]), @@ -1702,7 +1609,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("blockd_lends_within_its_bound", &["test_rs_blockd_io"]), ( "inspect_reads_its_owners", - &["test_rs_inspect_denied", "test_rs_inspect_plays", "test_rs_inventory_bounds"], + &["test_rs_inspect_denied", "test_rs_inventory_bounds"], ), ("metal_sim_compositor", METAL_SIM_CLIENTS), ("metal_sim_scanout_wc", METAL_SIM_CLIENTS), @@ -1715,13 +1622,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("toolkit_window_wake", &["test_rs_window_wake"]), ("toolkit_winit_loop", &["test_rs_winit_loop"]), ("toolkit_winit_pace", &["test_rs_winit_pace"]), - ("doom_sound_flood", &["test_rs_doom_sound_flood"]), - ("doom_music", &["test_rs_doom_music"]), - ("soundd_log_stall", &["test_rs_soundd_log_stall"]), - ("metal_sim_null_audio", &["test_rs_audio_tone"]), - ("null_sink_shipped_client", &["test_rs_null_sink_client_exits"]), - ("hda_tone", &["test_rs_audio_tone"]), - ("hda_client_stall", &["test_rs_hda_client_stall"]), ("latency_wake", &["test_rs_cyclictest", "test_rs_sched_stress"]), ("smp_failed_ap_leaves_no_hole", &["test_rs_smp_hole_shootdown"]), ("sshd_exec", &["test_rs_empty_dir_stat"]), @@ -1793,7 +1693,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("screen_console_scroll", &["test_rs_test_screen_churn"]), ("screen_console_panic", &["test_rs_test_panic_child"]), ("screen_fatal_halt", &["test_rs_test_panic_child"]), - ("panic_halts_the_others_first", &["test_rs_panic_halts_first"]), ("screen_recoverable_untouched", &["test_rs_test_panic_child"]), ("screen_survived_panic_not_blamed", &["test_rs_test_panic_child"]), ]; @@ -1922,14 +1821,81 @@ const METAL: &[(&str, metal::Metal)] = &[ }, ), ( - // The shipped tone client plays to completion and exits 0. Its QEMU - // registration calls the sink a null one; on the T14 whether soundd - // binds the laptop's own HDA controller is unmeasured, so what this - // asserts here is the weaker and truer thing — the client came back. + "wake_storm_cost", + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| b[0].job_passed("test_rs_wake_storm_cost"), + }, + ), + ( + // The shipped tone client, twice in series, plays to completion and + // exits 0, and soundd names how each left. "null_sink_shipped_client", metal::Metal::Runs { arms: TESTCASES, - judge: |b| b[0].job_passed("test_rs_null_sink_client_exits"), + judge: |b| { + b[0].job_passed("test_rs_null_sink_client_exits")?; + audio::departures_on_metal(&b[0].log()) + }, + }, + ), + ( + // soundd with no client costs no CPU, before any client has connected. + "audio_idle_suspend", + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| b[0].job_passed("test_rs_audio_idle_suspend"), + }, + ), + ( + "hda_tone", + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| { + b[0].job_passed("test_rs_audio_tone")?; + audio::tone_on_metal(&b[0].log()) + }, + }, + ), + ( + "hda_client_stall", + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| { + b[0].job_passed("test_rs_hda_client_stall")?; + audio::client_stall_on_metal(&b[0].log()) + }, + }, + ), + // ---- the three audio boots of their own ---- + ( + "doom_sound_flood", + metal::Metal::Runs { + arms: DOOMCASE, + judge: |b| { + b[0].job_passed("test_rs_doom_sound_flood")?; + audio::sound_flood_on_metal(&b[0].log()) + }, + }, + ), + ( + "doom_music", + metal::Metal::Runs { + arms: DOOMMUSICCASE, + judge: |b| { + b[0].job_passed("test_rs_doom_music")?; + audio::music_on_metal(&b[0].log()) + }, + }, + ), + ( + "soundd_log_stall", + metal::Metal::Runs { + arms: LOGSTALLCASE, + judge: |b| { + b[0].job_passed("test_rs_soundd_log_stall")?; + audio::log_stall_on_metal(&b[0].log()) + }, }, ), // ---- one image: tests/testcases armed with the chipset watchdog ---- @@ -1979,7 +1945,13 @@ const METAL: &[(&str, metal::Metal)] = &[ ), ( "timer_calibration", - metal::Metal::Runs { arms: TESTCASES, judge: |b| timer_calibration(b[0].kernel().text()) }, + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| { + timer_calibration(b[0].kernel().text())?; + tsc_agrees_with_cpuid(b[0].kernel().text()) + }, + }, ), ( "pci_inventory", @@ -2196,24 +2168,33 @@ const METAL: &[(&str, metal::Metal)] = &[ judge: |b| operation_nesting_log(b[0].kernel().text()), }, ), - // ---- named, looked at, and not run there ---- - ( - "metal_sim_null_audio", - metal::Metal::QemuOnly( - "its subject is soundd's null sink, and whether the T14's own HDA controller binds \ - is unmeasured; both halves of the verdict — soundd's counters and a host-timed \ - drain — are console text and a host clock, neither of which the stick carries", - ), - ), ]; /// **The [`METAL`] rows no QEMU registration answers for**, each with why none /// can: a verdict under a name only the T14 reports. -const METAL_ONLY: &[(&str, &str)] = &[( - "lan_message_delivery", - "whether the T14's own I219 delivers a message through that machine's interrupt remapping \ - is a fact of that part and that path; QEMU's e1000e is another part behind another path", -)]; +const METAL_ONLY: &[(&str, &str)] = &[ + ( + "lan_message_delivery", + "whether the T14's own I219 delivers a message through that machine's interrupt \ + remapping is a fact of that part and that path; QEMU's e1000e is another part behind \ + another path", + ), + ( + "wake_storm_cost", + "its verdict is that a wake storm's cost grows linearly with the waiters, read off the \ + TSC around the syscall, and a guest's TSC runs while its host has the vCPU", + ), + ("null_sink_shipped_client", AUDIO_ON_METAL_ONLY), + ("audio_idle_suspend", AUDIO_ON_METAL_ONLY), + ("hda_tone", AUDIO_ON_METAL_ONLY), + ("hda_client_stall", AUDIO_ON_METAL_ONLY), + ("doom_sound_flood", AUDIO_ON_METAL_ONLY), + ("doom_music", AUDIO_ON_METAL_ONLY), + ("soundd_log_stall", AUDIO_ON_METAL_ONLY), +]; + +/// Why an audio row has no QEMU arm. +const AUDIO_ON_METAL_ONLY: &str = "audio is judged on the T14 and in no QEMU guest (owner ruling)"; /// The boot most of the first tranche rides: the plain `tests/testcases` shape /// with a job list that ends it. @@ -2226,7 +2207,16 @@ const TESTCASES: &[metal::Arm] = &[metal::once( "testcases", "tests/testcases", &[], - &["test_rs_abuse_short_sleep", "test_rs_null_sink_client_exits", "log-close"], + &[ + "test_rs_wake_storm_cost", + // Before any client connects, which is `audio_idle_suspend`'s premise. + "test_rs_audio_idle_suspend", + "test_rs_audio_tone", + "test_rs_hda_client_stall", + "test_rs_abuse_short_sleep", + "test_rs_null_sink_client_exits", + "log-close", + ], )]; /// **Two boots of one config, because these two cannot share one.** Each fills @@ -2242,6 +2232,18 @@ const TESTCASES_READDIR: &[metal::Arm] = const JOBCASE: &[metal::Arm] = &[metal::once("jobcase", "tests/jobcase", &[], &[])]; +/// doom beside soundd: `doom --sound-stress` outrunning its parked callback. +const DOOMCASE: &[metal::Arm] = + &[metal::once("doomcase", "tests/doomcase", &[], &["test_rs_doom_sound_flood"])]; + +/// doom with its WAD and the SoundFont its music is made of. +const DOOMMUSICCASE: &[metal::Arm] = + &[metal::once("doommusiccase", "tests/doommusiccase", &[], &["test_rs_doom_music"])]; + +/// A `logd` that leaves soundd's ring unread until the job says the tone played. +const LOGSTALLCASE: &[metal::Arm] = + &[metal::once("logstallcase", "tests/logstallcase", &[], &["test_rs_soundd_log_stall"])]; + /// The two boots the reset ruling is judged on, and both are boots this suite /// already flashes: the device boot for a reset with megabytes behind it, and /// `jobcase` for one with nothing. @@ -2849,9 +2851,7 @@ fn discover_rust_tests(bins: &[(String, Vec)]) -> Vec { if name.ends_with(".so") { return None; } - if RUST_SKIP.contains(&name.as_str()) - || AUDIO_TESTS.iter().any(|(audio, _)| *audio == name) - { + if RUST_SKIP.contains(&name.as_str()) { return None; } Some(name.clone()) @@ -3227,65 +3227,6 @@ fn check_tripwire_attribution(serial: &str) -> Result<(), String> { Ok(()) } -/// A zero CPU delta is the signature of a suspended soundd and equally of one -/// wedged with the device running, so the counter the test reads cannot tell -/// them apart on its own. The serial can: in a window where no audio client -/// ever connects, the PCM stream has no business starting. -/// -/// This reads only `result.serial`, which begins at ===TEST_START, so a -/// device started before then (a restored boot prime) is invisible to this -/// particular check — not because the harness cannot see it: `qemu.boot_log()` -/// holds it, which is what `audio::check_suspend_structure` concatenates in -/// ahead of its own window. What this one does catch is a start inside its -/// window with no client to justify it: soundd's `!streams.is_empty()` -/// fill-loop gate going away, or a resume fired by anything other than a -/// connect. -fn check_audio_idle_suspend(result: &TestResult) -> bool { - if !check_rust_result(result) { - return false; - } - const STARTED: &str = "virtio-sound: stream 0 started"; - if result.serial.contains(STARTED) { - eprintln!( - "FAIL rs::audio_idle_suspend: `{STARTED}` with no client connected — \ - soundd's zero CPU is the device left running, not a suspend\nserial:\n{}", - result.serial - ); - return false; - } - true -} - -/// Two clients through the null sink, and what soundd said about each leaving. -/// -/// The exit code already says both `/system/bin/tone` runs finished cleanly — that is -/// the test's own assertion — so this window is exactly the case soundd used to -/// misreport: `client N died` for a process that exited `code=0`, because the -/// mix loop's signal pipe broke before the control thread read the peer. What -/// it asserts is that neither outcome of that race is worded as a death, and -/// that both removals name a departure soundd actually established. -/// -/// **The count is per removal and stays exact**, because the vocabulary is -/// asserted per removal: a capture where no client ever left would satisfy -/// every check above it vacuously, and a range would let the second removal go -/// missing again. What used to make that count a race was the window and not -/// the number — see [`settle_null_sink_client_exits`], which is what closes it. -fn check_null_sink_client_exits(result: &TestResult) -> bool { - if !check_rust_result(result) { - return false; - } - let problems = audio::check_departures(&result.serial, NULL_SINK_CLIENTS); - if !problems.is_empty() { - eprintln!( - "FAIL rs::null_sink_client_exits: {}\nserial:\n{}", - problems.join("; "), - result.serial - ); - return false; - } - true -} - /// The exit code says the child died; only the serial says *why*. /// /// A #DE with no gate escalates to #DF, and `double_fault_handler` halts every @@ -3389,45 +3330,6 @@ fn check_debug_trap(result: &TestResult) -> bool { ok } -/// The clients `test_rs_null_sink_client_exits` runs in series, and so the -/// number of removals soundd owes. One constant, because the wait below and the -/// count above have to be the same number or the wait is for something else. -const NULL_SINK_CLIENTS: usize = 2; - -/// Wait for soundd to report the second client leaving, on the guest's liveness. -/// -/// **The last removal arrives after the process whose exit produced it**, and -/// that process exiting is what ends the capture: round 1's line makes it in -/// because a whole second round follows it, and round 2's has nothing behind it -/// but `===TEST_END===`. Counting two removals over that window is an assertion -/// about scheduling, and it went red on CI twice on documentation-only branches -/// — `soundd reported 1 client removals, expected 2` — with the capture showing -/// the line never arriving rather than arriving wrong. -/// -/// The wait is [`await_guest`]'s: it ends when the removals are there, or when -/// the guest stops making progress, and never on a span of host wall clock. It -/// costs nothing on a run that already had both lines — the predicate is checked -/// before anything is drained, which was true of 6 of 6 measured runs on the dev -/// host — and its expiry is not a verdict, which is why the error is dropped: -/// what fails this test is still the count, in `check_departures`'s own words. -/// -/// [`audio::SOUNDD_GONE`] ends it too, for the same reason `await_null_sink` -/// reads that line: soundd exiting is a removal that is never coming, and the -/// test should say so in its own sentence rather than wait out the guard. The -/// guard is the whole of [`qemu::GUEST_WEDGED`] here and not the quiet bound — -/// this boot's kernel prints on a 10 s cadence, so the machine is never silent -/// for the 15 s that would end the wait early (measured: an unreachable -/// predicate takes 302 s). That price is paid only by a run where soundd is -/// alive and has genuinely stopped reporting departures, which is the defect -/// this test exists for. -fn settle_null_sink_client_exits(qemu: &mut QemuInstance, result: &mut TestResult) { - let mut serial = std::mem::take(&mut result.serial); - let _ = await_guest(qemu, &mut serial, "soundd to report both clients leaving", |seen| { - audio::departures(seen).len() >= NULL_SINK_CLIENTS || seen.contains(audio::SOUNDD_GONE) - }); - result.serial = serial; -} - /// Nothing to wait for: the test's own window carries everything its check /// reads. Every name but two. fn no_settle(_: &mut QemuInstance, _: &mut TestResult) {} @@ -3471,7 +3373,6 @@ fn accounting_of(pid: u32) -> String { /// selects the check. fn settle_for(name: &str) -> fn(&mut QemuInstance, &mut TestResult) { match name { - "null_sink_client_exits" => settle_null_sink_client_exits, "syscall_cost" => settle_syscall_cost, "exit_wait_storm" => settle_exit_wait_storm, _ => no_settle, @@ -3483,8 +3384,6 @@ fn check_for(name: &str) -> fn(&TestResult) -> bool { match name { "panic_recovery" => check_panic_recovery, "disk_backtrace" => check_disk_backtrace, - "audio_idle_suspend" => check_audio_idle_suspend, - "null_sink_client_exits" => check_null_sink_client_exits, "fault_gates" => check_fault_gates, "debug_trap" => check_debug_trap, "dlopen_dedup" => check_dlopen_dedup, @@ -3753,736 +3652,12 @@ fn check_exit_wait_storm(result: &TestResult) -> bool { ); return false; } - } - true -} - -/// Minimum active (non-silent) playback the 3s test tone must produce. -/// Guards against a vacuous pass when nothing plays at all. -const TONE_MIN_ACTIVE_SECS: f64 = 2.5; -/// The tone is generated at amplitude 16000; a far lower peak proves the -/// signal path is broken even if technically "active". -const TONE_MIN_PEAK: i32 = 4000; - -/// Recorded per-(test, smp) baselines — gate A's thorough tier. -/// Two independent instruments per config: -/// the wav underrun histogram (`gaps`, keyed by gap length in device periods) -/// and ceilings on soundd's own counters. The wav is a rare-event detector; -/// the counters fire on nearly every run and carry the statistical power. Both -/// must hold. Re-record deliberately, never casually — and justify every -/// number in `tests/audio-baseline.toml` itself. -#[derive(serde::Deserialize)] -#[serde(deny_unknown_fields)] -struct AudioBaselineEntry { - #[serde(default)] - gaps: BTreeMap, - max_wake_lat_us: u64, - drains: u32, - underruns: u32, - sample: BaselineSample, -} - -/// The recorded clean-tree *sample* for one config, not a summary of it. The -/// thorough tier compares a fresh sample against this one, so it needs the -/// observations themselves — see `tests/common/stats.rs` for why a summary -/// would understate the false-red rate. -#[derive(serde::Deserialize)] -#[serde(deny_unknown_fields)] -struct BaselineSample { - /// Runs whose wav was analysed (the counter arrays can be longer: a run - /// can lose its histogram and still report counters). - gap_sample: u32, - /// Of `gap_sample`, how many showed at least one mid-tone dropout. - gap_runs: u32, - /// Of the counter runs, how many breached this config's per-run ceilings. - ceiling_runs: u32, - max_wake_lat_us: Vec, - underruns: Vec, - wakes: Vec, - /// Recorded for re-baselining the per-run ceiling only. Deliberately not - /// tested distributionally: it is zero on 50-90% of runs, and the ties - /// leave a rank test with no power (measured: 0.00-0.21 against a tripling). - drains: Vec, -} - -type AudioBaseline = BTreeMap>; - -struct ConfigBaseline<'a> { - gaps: BTreeMap, - counters: audio::CounterLimits, - sample: &'a BaselineSample, -} - -fn load_audio_baseline() -> AudioBaseline { - let path = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/audio-baseline.toml"); - let text = fs::read_to_string(&path) - .unwrap_or_else(|e| panic!("read {}: {e}", path.display())); - toml::from_str(&text).unwrap_or_else(|e| panic!("parse {}: {e}", path.display())) -} - -/// Baseline for one (test, smp) config. Every config must be recorded: an -/// ungated config would pass by omission. -fn config_baseline<'a>(baseline: &'a AudioBaseline, name: &str, smp: u32) -> ConfigBaseline<'a> { - let entry = baseline - .get(name) - .and_then(|per_smp| per_smp.get(&format!("smp{smp}"))) - .unwrap_or_else(|| panic!("audio-baseline.toml: no [{name}.smp{smp}] section")); - ConfigBaseline { - sample: &entry.sample, - gaps: entry - .gaps - .iter() - .map(|(k, &count)| { - let periods: u32 = k.parse().unwrap_or_else(|_| { - panic!("audio-baseline.toml: bad gap key {k:?} for {name} smp{smp}") - }); - (periods, count) - }) - .collect(), - counters: audio::CounterLimits { - max_wake_lat_us: entry.max_wake_lat_us, - drains: entry.drains, - underruns: entry.underruns, - }, - } -} - -/// What one audio boot measured. Both tiers are computed from this; they -/// differ only in how many they collect and what decision they take on the -/// collection. -struct AudioRun { - gaps: BTreeMap, - counters: audio::SounddCounters, - /// The instrument itself is untrustworthy on this run (no tone, no dither, - /// clicks, no stats window). Never a rare-event judgement — always fatal, - /// in both tiers. - broken: Vec, - /// soundd counters past this config's per-run ceilings. A counted rate in - /// the thorough tier; printed but not a verdict in the fast tier, which - /// judges `harm` instead. - breaches: Vec, - /// What else the host was doing while this boot was measured. Annotation - /// only — nothing above or below branches on it. - host: hostload::HostLoad, -} - -impl AudioRun { - /// The capture's verdict alone. The thorough tier's dropout *rate* is - /// defined on this and nothing else, because that is what the recorded - /// sample counted. - fn dropped_audio(&self) -> bool { - !self.gaps.is_empty() - } - - /// Silence that reached the device on this run: a mid-tone gap in the - /// capture, or a period soundd put on the wire with no client audio behind - /// it. Both are audio someone would have heard drop out, and together they - /// are the fast tier's whole verdict — a counter past a ceiling says the - /// pipeline came close, and how close is a question for a distribution. - fn harm(&self) -> Option { - let mut evidence = Vec::new(); - if self.dropped_audio() { - evidence.push(format!("dropout {}", audio::format_histogram(&self.gaps))); - } - if self.counters.underruns > 0 { - evidence.push(format!( - "{} of {} periods submitted with no client audio", - self.counters.underruns, self.counters.submitted - )); - } - (!evidence.is_empty()).then(|| evidence.join(", ")) - } -} - -/// `--slow-usb`: give every audio boot a USB stick that answers a bulk transfer -/// in 2 ms instead of microseconds — what a real stick's erase block does, and -/// what the T14's audio pops are made of. -/// -/// A switch and not a test of its own, because it changes no verdict: it makes -/// the four audio configs measure a machine the host cannot otherwise present, -/// and what it produces is an A/B against the same command without it in the -/// same session. `issues/kernel/every-wait-in-this-kernel-is-a-spin.md` is what -/// the numbers are for. -static SLOW_USB: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); - -/// Boot a fresh QEMU with the given CPU count, run one in-guest audio test, -/// and measure it: soundd's in-guest counters (wake lateness, pipeline drains, -/// periods of silence submitted) and the captured wav (mid-signal silence, hard -/// sample-to-sample discontinuities, and the dither the detector needs to see -/// anything at all). -/// -/// `Err` means the run produced no measurement — a boot failure, a timeout, an -/// unreadable capture. That is never a rare-event judgement call; it is fatal -/// in both tiers. -fn measure_audio_run( - name: &str, - smp: u32, - baseline: &ConfigBaseline, - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], - // Distinguishes this boot from the others of the same config in the log - // and in the kept capture's filename; empty for a plain single boot. - tag: &str, -) -> Result { - let label = if tag.is_empty() { - String::new() - } else { - format!("{tag}: ") - }; - // Bounds every duration soundd can report: its whole life is inside this - // process's. See `audio::check_physical`. - let job = format!("test_rs_{name}"); - let carried = qemu::carrying(c_bins, rust_bins, [job.as_str()]); - let run_start = std::time::Instant::now(); - let mut qemu = QemuInstance::boot_with_options( - test_config, - &carried.c, - &carried.rust, - BootOptions { - smp, - kernel_params: if SLOW_USB.load(std::sync::atomic::Ordering::Relaxed) { - &["usb-slow-device"] - } else { - &[] - }, - ..Default::default() - }, - ); - - let result = qemu.run_test(&job, Duration::from_secs(30)); - if let Some(err) = &result.error { - return Err(err.to_string()); - } - match result.exit_code { - Some(0) => {} - Some(code) => return Err(format!("exit code {code}\nstdout:\n{}", result.stdout)), - None => return Err(format!("no exit code\nstdout:\n{}", result.stdout)), - } - - // The wav timeline advances in real time; give the tone tail and its - // trailing silence context time to reach the file before reading it. The - // same wait collects soundd's final stats flush, which races the client's - // exit and so can arrive after ===TEST_END===. - // - // Boot prepended so `check_suspend_structure` can see a device started - // before ===TEST_START — the boot capture exists (`qemu.boot_log()`), - // where its doc comment used to say it did not. - let serial = - qemu.boot_log().to_string() + &result.serial + &qemu.drain_serial(Duration::from_millis(500)); - - let wav = audio::parse_wav(qemu.audio_wav_path())?; - let analysis = audio::analyze(&wav); - let rate = wav.sample_rate as f64; - let secs = |samples: usize| samples as f64 / rate; - - // Always printed, so every run leaves comparable numbers in the log. - let gaps = audio::gap_histogram(&analysis, wav.sample_rate); - let counters = audio::parse_soundd_counters(&serial)?; - // Sampled here rather than before the boot because the load averages are - // trailing: a reading taken now covers the run, one taken before it covers - // only what preceded it. This run's own guest is still up, so `qemu 1` is - // the quiet reading. - let host = hostload::HostLoad::sample(); - eprintln!( - " {label}{name} smp={smp} gaps: {} (baseline {}) peak {} active {:.2}s dither {:.1}% \ - pitch {:.1}Hz phase-breaks {}", - audio::format_histogram(&gaps), - audio::format_histogram(&baseline.gaps), - analysis.peak, - secs(analysis.active_samples), - analysis.dither_ratio.unwrap_or(0.0) * 100.0, - audio::dominant_hz(&wav).unwrap_or(0.0), - audio::phase_breaks(&wav).len(), - ); - eprintln!( - " {label}{name} smp={smp} soundd: wake_lat {}us ({:.2} pipelines, limit {}us) \ - [irq {}us + pickup {}us, {} empty wakes, batch {}, {} late of {}] \ - drains {}/{} underruns {}/{} submitted {} wakes {} batch {} windows {} — {} — {host}", - counters.max_wake_lat_us, - counters.max_wake_lat_us as f64 / audio::PIPELINE_DEPTH_US as f64, - baseline.counters.max_wake_lat_us, - counters.worst.irq_late_us, - counters.worst.pickup_us, - counters.worst.empty, - counters.worst.batch, - counters.late_wakes, - counters.wakes, - counters.drains, - baseline.counters.drains, - counters.underruns, - baseline.counters.underruns, - counters.submitted, - counters.wakes, - counters.max_batch, - counters.windows, - audio::boot_clocks(qemu.boot_log()), - ); - - let breaches = audio::check_counters(&counters, &baseline.counters); - if !breaches.is_empty() { - eprintln!( - " {label}{name} smp={smp} over ceiling: {} — recorded; the fast tier's \ - verdict is harm, the rate of these is the thorough tier's", - breaches.join("; ") - ); - } - - // A counter past a physical bound is the instrument failing, so it belongs - // here with the other instrument checks rather than among the ceilings: it - // must fail loudly in both tiers, and it must never be ranked against the - // recorded sample or printed into the next baseline. - let mut problems = audio::check_physical(&counters, run_start.elapsed().as_secs_f64()); - // soundd counts only while it has clients, so a run with no window reports - // zero for every counter — the best numbers this gate can see, from a run - // that measured nothing. That is the instrument dead, not a ceiling held. - if counters.windows == 0 { - problems.push( - "soundd printed no stats window with clients — the tone never reached the mixer" - .to_string(), - ); - } - if secs(analysis.active_samples) < TONE_MIN_ACTIVE_SECS { - problems.push(format!( - "tone missing: only {:.2}s of active signal (expected >= {TONE_MIN_ACTIVE_SECS}s)", - secs(analysis.active_samples) - )); - } - if analysis.peak < TONE_MIN_PEAK { - problems.push(format!( - "tone too quiet: peak {} (expected >= {TONE_MIN_PEAK})", - analysis.peak - )); - } - // Present, loud and continuous is not the same as right: a device consuming - // the buffers at a rate soundd did not ask for satisfies all three and plays - // the whole session off pitch. - if let Some(complaint) = audio::wrong_pitch(&wav) { - problems.push(complaint); - } - // Without this the gate can go green while measuring nothing: the underrun - // detector's silence band is derived from soundd applying TPDF dither into - // a rounding quantizer. Lose the dither and silence becomes - // exact zero everywhere, the band collapses, and dropouts stop being - // visible — the exact failure this instrument was rebuilt to remove. - match analysis.dither_ratio { - Some(ratio) if ratio < audio::MIN_DITHER_RATIO => problems.push(format!( - "dither missing: only {:.1}% of silent samples are non-zero (expected ~25%, \ - floor {:.0}%) — soundd is not dithering, so the underrun detector is blind", - ratio * 100.0, - audio::MIN_DITHER_RATIO * 100.0 - )), - Some(_) => {} - None => problems.push("no silent stretch in capture to verify dither against".to_string()), - } - if audio::check_gap_regression(&gaps, &baseline.gaps).is_err() { - let mut msg = format!( - "{} mid-signal underruns (silence >= 2ms inside the tone):", - analysis.underruns.len() - ); - for run in analysis.underruns.iter().take(20) { - msg.push_str(&format!( - "\n at {:8.3}s len {:6.2}ms", - secs(run.start), - secs(run.len) * 1000.0 - )); - } - if analysis.underruns.len() > 20 { - msg.push_str(&format!("\n ... and {} more", analysis.underruns.len() - 20)); - } - eprintln!(" {label}{name} smp={smp} {msg}"); - } - if !analysis.clicks.is_empty() { - let mut msg = format!("{} hard discontinuities (|delta| > 8000):", analysis.clicks.len()); - for click in analysis.clicks.iter().take(10) { - msg.push_str(&format!( - "\n at {:8.3}s {} -> {}", - secs(click.index), - click.from, - click.to - )); - } - if analysis.clicks.len() > 10 { - msg.push_str(&format!("\n ... and {} more", analysis.clicks.len() - 10)); - } - problems.push(msg); - } - - // Suspend structure — categorical per-run assertions, so they belong - // with the instrument checks: fatal in both tiers, never a counted rate. - problems.extend(audio::check_suspend_structure(&serial)); - - // Keep every capture that shows something, so a dropout can be listened to - // even when the tier's rule says one occurrence is not yet a verdict. - if !problems.is_empty() || !breaches.is_empty() || !gaps.is_empty() { - let suffix = if tag.is_empty() { - String::new() - } else { - format!("-{tag}") - }; - // Renamed in the lane so `keep_serial` recognises and copies it out if - // this run ends red; the lane itself is gone on every exit, so the path - // worth printing is where that copy lands, not this one. - let suspect = qemu - .audio_wav_path() - .with_file_name(format!("audio-{name}-smp{smp}{suffix}.wav")); - match fs::rename(qemu.audio_wav_path(), &suspect) { - Ok(()) => eprintln!( - " {label}{name} smp={smp} wav kept at {} if this run ends red", - common::lane::kept_path(&suspect).display() - ), - Err(e) => eprintln!( - " {label}{name} smp={smp} could not keep {}: {e}", - suspect.display() - ), - } - } - - Ok(AudioRun { - gaps, - counters, - broken: problems, - breaches, - host, - }) -} - -/// Fast tier — one boot per config, run on every `cargo test`. -/// -/// Certifies: the instrument is alive, no counter is on the wrong side of a -/// physical bound, and this build does not *reproducibly* put silence on the -/// wire. It cannot certify a *rate*; one run is one Bernoulli trial against a -/// per-config dropout rate measured at 0-7%, which discriminates nothing. That -/// is what `--audio-gate` is for. -/// -/// **The verdict is harm** — a mid-tone gap in the capture, or a period soundd -/// submitted with no client audio behind it. The per-run ceilings are measured, -/// printed and kept, and fail nothing here: `drains` past its ceiling with an -/// empty histogram and zero underruns is a pipeline that recovered before -/// anyone could hear it, and one boot cannot say whether it recovers less often -/// than it used to. That question has an instrument with power, and it is the -/// thorough tier's `ceiling_runs` rate. -/// -/// Harm is confirmed before it fails: a run that shows any is re-booted once, -/// and only a second failure counts. No bar is widened by this — the zero-gap -/// bar is strict on both boots. Without the confirmation the per-config dropout -/// rate alone reds one invocation in eight on a clean tree, and a gate -/// developers see every day cannot cry wolf that often. The first occurrence is -/// still printed and its capture still kept. -fn run_audio_test( - name: &str, - smp: u32, - baseline: &ConfigBaseline, - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let run = measure_audio_run(name, smp, baseline, test_config, c_bins, rust_bins, "")?; - - if !run.broken.is_empty() { - return Err(run.broken.join("\n ")); - } - let Some(harm) = run.harm() else { - return Ok(()); - }; - - let silent_runs = baseline.sample.underruns.iter().filter(|&&u| u > 0.0).count(); - eprintln!( - " {name} smp={smp} HARM {harm} — rare on this tree ({} of {} recorded runs \ - dropped audio, {silent_runs} of {} submitted a silent period); re-booting once \ - to confirm", - baseline.sample.gap_runs, - baseline.sample.gap_sample, - baseline.sample.underruns.len(), - ); - let again = measure_audio_run(name, smp, baseline, test_config, c_bins, rust_bins, "confirm")?; - if !again.broken.is_empty() { - return Err(again.broken.join("\n ")); - } - match again.harm() { - Some(again_harm) => Err(format!( - "audio dropped out on two consecutive boots: {harm} then {again_harm}" - )), - None => { - eprintln!(" {name} smp={smp} not reproduced on the confirming boot"); - Ok(()) - } - } -} - -// Thorough tier: `cargo test --test toyos-build -- --audio-gate N` - -/// One config's fresh sample, accumulated over the N iterations. -#[derive(Default)] -struct GateSamples { - max_wake_lat_us: Vec, - underruns: Vec, - wakes: Vec, - drains: Vec, - gap_runs: u32, - ceiling_runs: u32, -} - -/// A rejected statistic, ready to print. -struct Rejection { - config: String, - statistic: String, - detail: String, -} - -fn mwu_verdict( - config: &str, - statistic: &str, - base: &[f64], - fresh: &[f64], - worse_is_lower: bool, -) -> Option { - let z = stats::mann_whitney_z(base, fresh); - let z = if worse_is_lower { -z } else { z }; - let med = |v: &[f64]| { - let mut v = v.to_vec(); - v.sort_by(|a, b| a.partial_cmp(b).unwrap()); - v[v.len() / 2] - }; - (z > stats::Z_CRIT).then(|| Rejection { - config: config.to_string(), - statistic: statistic.to_string(), - detail: format!( - "median {:.0} -> {:.0} (Mann-Whitney z={z:.2} > {:.2})", - med(base), - med(fresh), - stats::Z_CRIT - ), - }) -} - -fn rate_verdict( - config: &str, - statistic: &str, - k1: u32, - n1: u32, - k0: u32, - n0: u32, -) -> Option { - let p = stats::fisher_greater(k1, n1, k0, n0); - (p <= stats::ALPHA).then(|| Rejection { - config: config.to_string(), - statistic: statistic.to_string(), - detail: format!( - "{k1} of {n1} vs recorded {k0} of {n0} (Fisher p={p:.2e} <= {:.0e})", - stats::ALPHA - ), - }) -} - -/// Thorough tier — N iterations of all four configs, gating on *rates* and -/// *distributions* rather than on single outcomes. The nightly runs it. -/// -/// Certifies, at N=30 and the measured clean-tree distributions: -/// * wake lateness has not shifted by 25% (detected 99.9% of the time) or -/// 20% (93%). A 10% shift is missed (4%). -/// * periods of silence on the wire have not risen 25% (94%) or 50% (100%). -/// * soundd is not being woken less often — the signature of completions -/// being batched because it ran late. A 5% drop is caught 99.9% of the -/// time. -/// * the mid-tone dropout *rate* has not risen 10x (100%) or 5x (71%). -/// A doubling is NOT detectable at this N and never will be at any N a -/// human waits for: separating 3% from 7% at this confidence needs ~600 -/// runs per config. The counters above are the instrument with power; the -/// dropout rate is the audible symptom, kept because it is the only -/// statistic here that says "someone would have heard it". -/// -/// False-red rate on a clean tree: 0.25%, measured over 2000 invocations -/// simulated from the recorded distributions. -fn run_audio_gate( - iterations: u32, - audio_baseline: &AudioBaseline, - audio_to_run: &[&str], - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> bool { - let configs: Vec<(&str, u32)> = audio_to_run - .iter() - .flat_map(|name| AUDIO_SMP.iter().map(move |&smp| (*name, smp))) - .collect(); - let mut samples: BTreeMap = BTreeMap::new(); - // Session-wide rather than per-config: the host is one host, and this is - // the sentence a re-record has to carry beside the numbers below. - let mut host: Vec = Vec::new(); - let start = std::time::Instant::now(); - - eprintln!( - "\n[gate A] {iterations} iterations x {} configs, serial. Every per-run outcome \ - becomes a rate; the verdict is on the collection, not on any one run.", - configs.len() - ); - - for iter in 1..=iterations { - eprintln!(" --- iteration {iter}/{iterations} ---"); - for &(name, smp) in &configs { - let key = format!("{name}.smp{smp}"); - let baseline = config_baseline(audio_baseline, name, smp); - let tag = format!("iter{iter:03}"); - let run = match measure_audio_run( - name, smp, &baseline, test_config, c_bins, rust_bins, &tag, - ) { - Ok(run) => run, - Err(err) => { - eprintln!("\n[gate A] FAILED on iteration {iter}: {key} produced no measurement: {err}"); - eprintln!("[gate A] A run that does not complete is not a rare event to be \ - averaged away — every known cause of one has been fixed."); - return false; - } - }; - if !run.broken.is_empty() { - eprintln!("\n[gate A] FAILED on iteration {iter}: {key} instrument broken: {}", - run.broken.join("; ")); - return false; - } - host.push(run.host); - let s = samples.entry(key).or_default(); - s.max_wake_lat_us.push(run.counters.max_wake_lat_us as f64); - s.underruns.push(run.counters.underruns as f64); - s.wakes.push(run.counters.wakes as f64); - s.drains.push(run.counters.drains as f64); - s.gap_runs += u32::from(run.dropped_audio()); - s.ceiling_runs += u32::from(!run.breaches.is_empty()); - } - - // Fail-side curtailment. Adding runs can only raise a count, so once a - // count passes the threshold for the *full* N the final verdict is - // already decided — stopping early costs no confidence. - if let Some(v) = curtail(&samples, audio_baseline, &configs, iterations) { - eprintln!("\n[gate A] FAILED after {iter} of {iterations} iterations (the remaining \ - runs cannot change this):"); - eprintln!(" {} {}: {}", v.config, v.statistic, v.detail); - return false; - } - } - - let mut rejected: Vec = Vec::new(); - let (mut pooled_gap_k, mut pooled_gap_n) = (0, 0); - let (mut pooled_ceil_k, mut pooled_ceil_n) = (0, 0); - let (mut base_gap_k, mut base_gap_n) = (0, 0); - let (mut base_ceil_k, mut base_ceil_n) = (0, 0); - - eprintln!("\n[gate A] {iterations} iterations in {:.0?}. Fresh sample vs recorded sample:\n", start.elapsed()); - eprintln!(" {}\n", hostload::summarise(&host)); - for &(name, smp) in &configs { - let key = format!("{name}.smp{smp}"); - let base = config_baseline(audio_baseline, name, smp).sample; - let s = &samples[&key]; - - rejected.extend(mwu_verdict(&key, "wake lateness", &base.max_wake_lat_us, &s.max_wake_lat_us, false)); - rejected.extend(mwu_verdict(&key, "underruns", &base.underruns, &s.underruns, false)); - rejected.extend(mwu_verdict(&key, "wakes", &base.wakes, &s.wakes, true)); - rejected.extend(rate_verdict(&key, "dropout rate", s.gap_runs, iterations, base.gap_runs, base.gap_sample)); - - pooled_gap_k += s.gap_runs; - pooled_gap_n += iterations; - pooled_ceil_k += s.ceiling_runs; - pooled_ceil_n += iterations; - base_gap_k += base.gap_runs; - base_gap_n += base.gap_sample; - base_ceil_k += base.ceiling_runs; - base_ceil_n += base.max_wake_lat_us.len() as u32; - - report_config(&key, base, s, iterations); - } - rejected.extend(rate_verdict("pooled", "dropout rate", pooled_gap_k, pooled_gap_n, base_gap_k, base_gap_n)); - rejected.extend(rate_verdict("pooled", "per-run ceiling breaches", pooled_ceil_k, pooled_ceil_n, base_ceil_k, base_ceil_n)); - - eprintln!( - " pooled dropouts {pooled_gap_k}/{pooled_gap_n} (recorded {base_gap_k}/{base_gap_n}), \ - ceiling breaches {pooled_ceil_k}/{pooled_ceil_n} (recorded {base_ceil_k}/{base_ceil_n})" - ); - - if rejected.is_empty() { - eprintln!("\n[gate A] PASS — no statistic regressed at alpha={:.0e} per test.", stats::ALPHA); - true - } else { - eprintln!("\n[gate A] FAILED — {} statistic(s) regressed:", rejected.len()); - for v in &rejected { - eprintln!(" {} {}: {}", v.config, v.statistic, v.detail); - } - false - } -} - -/// Whether a count has already passed the threshold it would face at the full -/// iteration count. Only the yes/no statistics curtail: a rank test's outcome -/// is not monotone in the sample, so there is no honest early exit for it. -fn curtail( - samples: &BTreeMap, - audio_baseline: &AudioBaseline, - configs: &[(&str, u32)], - iterations: u32, -) -> Option { - let mut pooled_gap = 0; - let mut pooled_ceil = 0; - let (mut base_gap_k, mut base_gap_n) = (0, 0); - let (mut base_ceil_k, mut base_ceil_n) = (0, 0); - for &(name, smp) in configs { - let key = format!("{name}.smp{smp}"); - let base = config_baseline(audio_baseline, name, smp).sample; - let Some(s) = samples.get(&key) else { continue }; - if let Some(v) = rate_verdict(&key, "dropout rate", s.gap_runs, iterations, base.gap_runs, base.gap_sample) { - return Some(v); - } - pooled_gap += s.gap_runs; - pooled_ceil += s.ceiling_runs; - base_gap_k += base.gap_runs; - base_gap_n += base.gap_sample; - base_ceil_k += base.ceiling_runs; - base_ceil_n += base.max_wake_lat_us.len() as u32; - } - let n = iterations * configs.len() as u32; - rate_verdict("pooled", "dropout rate", pooled_gap, n, base_gap_k, base_gap_n) - .or_else(|| rate_verdict("pooled", "per-run ceiling breaches", pooled_ceil, n, base_ceil_k, base_ceil_n)) -} - -/// Print one config's fresh sample next to the recorded one, in a form that can -/// be pasted straight back into `tests/audio-baseline.toml` when a re-baseline -/// is deliberate. The gate's output *is* the next baseline. -fn report_config(key: &str, base: &BaselineSample, s: &GateSamples, iterations: u32) { - let stat = |v: &[f64]| { - let mut v = v.to_vec(); - v.sort_by(|a, b| a.partial_cmp(b).unwrap()); - (v[0], v[v.len() / 2], v[v.len() - 1]) - }; - eprintln!(" {key} (n={iterations}, recorded n={})", base.max_wake_lat_us.len()); - for (label, b, f) in [ - ("wake_lat_us", &base.max_wake_lat_us, &s.max_wake_lat_us), - ("underruns ", &base.underruns, &s.underruns), - ("wakes ", &base.wakes, &s.wakes), - ("drains ", &base.drains, &s.drains), - ] { - let (bl, bm, bh) = stat(b); - let (fl, fm, fh) = stat(f); - eprintln!( - " {label} recorded {bl:.0}/{bm:.0}/{bh:.0} fresh {fl:.0}/{fm:.0}/{fh:.0} (min/median/max)" - ); - } - eprintln!( - " dropouts recorded {}/{} fresh {}/{iterations}", - base.gap_runs, base.gap_sample, s.gap_runs - ); - let fmt = |v: &[f64]| { - let mut v = v.to_vec(); - v.sort_by(|a, b| a.partial_cmp(b).unwrap()); - let v: Vec = v.iter().map(|x| format!("{x:.0}")).collect(); - format!("[{}]", v.join(", ")) - }; - eprintln!(" toml: max_wake_lat_us = {}", fmt(&s.max_wake_lat_us)); - eprintln!(" toml: underruns = {}", fmt(&s.underruns)); - eprintln!(" toml: wakes = {}", fmt(&s.wakes)); - eprintln!(" toml: drains = {}", fmt(&s.drains)); + } + true } /// Echo what the guest actually put on screen, under `--nocapture` only — -/// it is the measurement these tests are built on, and the audio gate prints -/// its numbers for the same reason. +/// it is the measurement these tests are built on. fn print_screen(name: &str, text: &str) { if !qemu::VERBOSE.load(std::sync::atomic::Ordering::Relaxed) { return; @@ -8646,9 +7821,7 @@ fn shell_answers(qemu: &mut QemuInstance, log: &mut String, ack: &Drained) -> Re /// **Two waits, because the two ways this fails are different questions.** The /// first is "has the terminal come up", and it used to be answered by retyping /// against `qemu.budget(20 s)` — a guess at how long a desktop takes to come up -/// on the host of the day, which is exactly the shape `issues/design-debt/` -/// bills for: `desktop_audio_client` 385 s wide against 13 s alone, and a -/// landing gate that is a coin toss. The terminal knows when it is up and now +/// on the host of the day. The terminal knows when it is up and now /// says so, so this asks it and waits on the guest's own liveness. The second is /// "does a keystroke reach the shell", and it starts from a machine that is /// demonstrably up — a ceiling on *that* is a claim about the guest. @@ -8766,121 +7939,6 @@ const SNAKE_ROUNDS: usize = 3; /// a program that has been running and drawing rather than one a second old. const SNAKE_TURNS: usize = 8; -/// Gate: doom's music reaches the device, with the SoundFont this tree ships. -/// -/// **The wiring is all this measures, and the wiring is the part nothing else -/// can.** `src/soundfont.rs`'s host tests say the committed bank covers every -/// instrument `assets/DOOM1.WAD` selects, and the subset was measured to render -/// bit-exact against the full bank through this same -/// `mus2mid.c` and this same rustysynth. Neither can say the file got into an -/// image, that doom opened it, or that what came out reached an audio device. -/// Those three are what `b8b0749` broke for a cycle with the suite green. -/// -/// Three verdicts, none of them a clock: -/// -/// 1. **doom opened the file this tree committed.** The guest prints the byte -/// count it read, and the host compares it against `assets/soundfont.sf2` on -/// disk. A stale image, a truncated asset and a second SoundFont from -/// somewhere else all fail here rather than turning into quiet silence. -/// 2. **It played to the end of the check.** The actuator counts the audio -/// callback's own periods, so a host that stopped this guest cannot shorten -/// what the capture is judged on. -/// 3. **Music reached the wire.** The device capture carries signal across most -/// of its length — which separates music from the one thing a broken -/// soundfont path still produces, a stream of zeroes. -fn doom_music(rust_bins: &[(String, Vec)]) -> Result<(), String> { - let root = Path::new(env!("CARGO_MANIFEST_DIR")); - let shipped = fs::metadata(root.join(toyos_build::soundfont::SOUNDFONT_PATH)) - .map_err(|e| format!("{}: {e}", toyos_build::soundfont::SOUNDFONT_PATH))? - .len(); - - let config = root.join("tests/doommusiccase"); - let mut qemu = QemuInstance::boot_with_options(&config, &[], rust_bins, BootOptions::default()); - - let result = qemu.run_test("test_rs_doom_music", Duration::from_secs(120)); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - if result.exit_code != Some(0) { - return Err(format!( - "doom could not play its own music (exit {:?}):\n{}", - result.exit_code, result.stdout - )); - } - - let opened = result - .stdout - .lines() - .find(|line| line.contains("[doom-sound] /system/share/soundfont.sf2:")) - .ok_or_else(|| { - format!( - "doom said nothing about the SoundFont, so this image has none:\n{}", - result.stdout - ) - })?; - let bytes: u64 = opened - .split_whitespace() - .find_map(|token| token.parse().ok()) - .ok_or_else(|| format!("no byte count in {opened:?}"))?; - if bytes != shipped { - return Err(format!( - "doom opened a {bytes}-byte SoundFont and this tree ships {shipped} bytes: the \ - image is not carrying {}", - toyos_build::soundfont::SOUNDFONT_PATH - )); - } - - let played = result - .stdout - .lines() - .find(|line| line.contains("[music-check] lump=")) - .ok_or_else(|| format!("doom printed no [music-check] line:\n{}", result.stdout))? - .to_string(); - - let _ = qemu.drain_serial(Duration::from_millis(500)); - let wav = audio::parse_wav(qemu.audio_wav_path())?; - let analysis = audio::analyze(&wav); - - // Seconds of signal, not a fraction of the capture: the capture runs from - // soundd opening the stream to the harness closing the file, so a fraction - // measures the harness as much as the music. Three of these four runs - // measured 1.19 s over a 3.11 s capture — the material's own dynamics, not - // a shortfall, since 500 LSB is -36 dBFS and E1M1's riff drops through it - // between notes. - const MIN_SIGNAL_SECS: f64 = 0.8; - let signal = analysis.active_samples as f64 / wav.sample_rate as f64; - if signal < MIN_SIGNAL_SECS { - return Err(format!( - "{signal:.2} s of the capture carries signal, under {MIN_SIGNAL_SECS} s: what doom \ - rendered is not what the device played\n{played}" - )); - } - // A floor and not a band: how loud E1M1 is at a given moment is the - // arrangement's business, and what this excludes is a dither floor being - // read as music. Measured 13547. - const MIN_PEAK: i32 = 6000; - if analysis.peak < MIN_PEAK { - return Err(format!( - "the device peaked at {} (expected at least {MIN_PEAK}): the music is inaudible\ - \n{played}", - analysis.peak - )); - } - - // Underruns are reported and fail nothing: whether music *stutters* is gate - // A's question and it has the statistics to ask it, where one boot of one - // track is one sample of an intermittent. - eprintln!( - " [doommusiccase] {}, {signal:.2} s of signal in a {:.2} s capture at peak {}, \ - {} underrun(s)", - played.trim(), - wav.mono.len() as f64 / wav.sample_rate as f64, - analysis.peak, - analysis.underruns.len(), - ); - Ok(()) -} - fn desktop_window_child(rust_bins: &[(String, Vec)]) -> Result<(), String> { let bins: Vec<(String, Vec)> = rust_bins.iter().filter(|(name, _)| name == "window_child").cloned().collect(); @@ -9845,165 +8903,6 @@ fn desktop_locale_detect() -> Result<(), String> { Ok(()) } -/// A shell-spawned audio client on a device-less desktop, and the desktop -/// afterwards. -/// -/// The machine `metal_sim_null_audio` and `null_sink_shipped_client` both miss: -/// they spawn the client from a test binary whose stdio is the console, and the -/// T14 spawns it from a shell inside a terminal inside the compositor, so every -/// one of the client's three descriptors is a pipe to a surface. Three verdicts -/// on one boot, in the order the T14 lost them: -/// -/// 1. **A client finishes.** `tone` writes a second of audio to the null sink -/// and prints its own completion line. -/// 2. **A second client connects while the first is streaming.** The T14's log -/// shows soundd's control thread printing `opening stream` for the second -/// with no `client N connected` behind it, so the connect is what has to be -/// observed, not just the exit. -/// 3. **The desktop survives them.** A terminal opened afterwards reaches a -/// shell that answers — the verdict the owner's machine failed while the -/// compositor was still painting, which is why nothing that reads pixels or -/// counts frames would have caught it. -fn desktop_audio_client() -> Result<(), String> { - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/desktopaudiocase"); - let options = BootOptions { - profile: qemu::Profile::Metal, - // The T14's core count: the suite's default of two serialises threads - // this shape is about the wakes between. - smp: 8, - qmp: true, - ready_marker: "compositor: ready", - // `Drained::Bytes`; off the shipping kernel, and implies fast-health - // and edge-race. - kernel_params: &["i8042-trace"], - ..Default::default() - }; - metal_sim_argv_check(&qemu::profile_argv(&options))?; - let mut qemu = QemuInstance::boot_with_options(&config, &[], &[], options); - let mut log = qemu.boot_log().to_string(); - - const NULL_LINE: &str = "soundd: no audio device, presenting a null sink"; - await_marker(&mut qemu, &mut log, NULL_LINE, "soundd to present a null sink") - .map_err(|why| format!("{why}\n{log}"))?; - // No panel row under a compositor; the kernel's drain report is the answer. - let ack = Drained::Bytes; - if let Err(why) = shell_answers(&mut qemu, &mut log, &ack) { - return Err(format!( - "{why}\nnothing typed at the terminal window reached a shell:\n{log}" - )); - } - - // One client, start to finish. `tone: done` is the client's own last line, - // so it is the client saying it got its callbacks and left — not the shell - // saying it launched something. - shell_type_line(&mut qemu, "tone 440 1", &ack)?; - await_marker( - &mut qemu, - &mut log, - "tone: done", - "a shell-spawned tone to finish on a device-less desktop", - ) - .map_err(|why| format!("{why}\n{log}"))?; - - // Two clients overlapping, each under its own shell in its own terminal. - // The shell has no job control, so the long tone holds its terminal and the - // second one has to be typed somewhere else — which is exactly how the T14 - // reached two live clients, and why the second terminal is part of the - // stimulus rather than only part of the verdict. - let before_second = log.len(); - shell_type_line(&mut qemu, "tone 660 8", &ack)?; - await_marker_new( - &mut qemu, - &mut log, - "tone: 660Hz", - before_second, - "the long tone to start", - ) - .map_err(|why| format!("{why}\n{}", &log[before_second..]))?; - open_terminal(&mut qemu, &mut log, "overlap-terminal-jc4t", &ack)?; - shell_type_line(&mut qemu, "tone 440 1", &ack)?; - // **The count is the verdict and the wait is not.** Both of these used to be - // `budget(60 s)`, which is a claim that a desktop with two audio clients on - // it finishes inside a minute times the width — and at 385 s wide against - // 13 s alone it was the single most expensive entry in `issues/design-debt/`. What - // ends the wait now is soundd going quiet, and what fails it is still the - // number of connects. - if let Err(why) = await_guest(&mut qemu, &mut log, "soundd to take up both connects", |log| { - connects_since(log, before_second) >= 2 - }) { - return Err(format!( - "{why}\nsoundd applied {} of the two connects — a client that opened a stream \ - was never taken up by the mixer:\n{}", - connects_since(&log, before_second), - &log[before_second..] - )); - } - // Both of them out again, counted in the same window. Waiting for `null - // sink idle` would not do: that line is already in the log from the first - // client, and a marker an earlier phase produced is not a verdict about - // this one. - if let Err(why) = await_guest(&mut qemu, &mut log, "both clients to leave the mixer", |log| { - removals_since(log, before_second) >= 2 - }) { - return Err(format!( - "{why}\n{} of the two overlapping clients left the mixer — the other one is \ - still streaming to a sink that stopped draining it:\n{}", - removals_since(&log, before_second), - &log[before_second..] - )); - } - - // The desktop afterwards: a process created after every one of the clients - // above, focused the moment it maps its window. This is the verdict the - // owner's machine failed while the compositor was still painting, which is - // why nothing that reads pixels or counts frames would have caught it. - open_terminal(&mut qemu, &mut log, "post-audio-desktop-vqmz", &ack)?; - eprintln!(" [desktop] three shell-spawned audio clients ran and the desktop still answers"); - Ok(()) -} - -/// Ctrl+N at the compositor, and a shell in the window it opens that answers. -/// -/// The nonce is per call because the verdict is that *this* terminal answered: -/// a marker an earlier one already produced would pass on a window that never -/// came up. [`shell_echoes`]'s split applies for the same reason it does there, -/// and `terminal: ready` is looked for after `before` rather than anywhere, -/// because every terminal already up has printed one. -fn open_terminal( - qemu: &mut QemuInstance, - log: &mut String, - nonce: &str, - ack: &Drained, -) -> Result<(), String> { - let before = log.len(); - { - let mut input = qemu::QmpInput::open(qemu.qmp_socket()); - input.keys(&[("ctrl", true), ("n", true), ("n", false), ("ctrl", false)]); - } - await_marker_new(qemu, log, "terminal: ready", before, "Ctrl+N to open a terminal") - .map_err(|why| format!("{why}\n{}", &log[before..]))?; - - // The same count-of-attempts as [`shell_echoes`], and for the same reason. - const TRIES: usize = 10; - let mut lost = String::new(); - for _ in 0..TRIES { - if let Err(said) = - shell_type_once(qemu, &format!("echo {nonce}"), round_trip(ECHO_TRY), ack) - { - lost = said; - continue; - } - if serial_until(qemu, log, nonce, round_trip(Duration::from_secs(2))) { - return Ok(()); - } - } - Err(format!( - "a terminal opened with Ctrl+N never reached a shell that answers in {TRIES} typed \ - lines:\n{lost}\n{}", - &log[before..] - )) -} - /// The host server behind the netd stream tests, where slirp's `10.0.2.2` /// lands. Each connection asks in nine bytes — a mode, then a little-endian /// length — and is served by the mode (`Ask` in @@ -10298,17 +9197,6 @@ fn netd_held_open(rust_bins: &[(String, Vec)]) -> Result<(), String> { Ok(()) } -/// A client that out-writes a peer which has stopped reading costs netd no -/// CPU: the send pipe is not watched while the socket has no room for it. -fn netd_stalled_peer(rust_bins: &[(String, Vec)]) -> Result<(), String> { - let HostRun { result, .. } = netcase_against_host(rust_bins, "netd_stalled_peer", false, "")?; - let Some(line) = result.stdout.lines().find(|l| l.contains("netd_stalled_peer: ok")) else { - return Err(format!("the guest never said it was done:\n{}", result.stdout)); - }; - eprintln!(" [netcase] {}", line.trim_end()); - Ok(()) -} - /// A UDP datagram the client's pipe will not take whole ends that socket by /// name, and nothing else: another socket still gets its datagram. fn netd_udp_refused(rust_bins: &[(String, Vec)]) -> Result<(), String> { @@ -10705,25 +9593,6 @@ fn dump_field(report: &str, marker: &str, word: &str) -> Result { .ok_or_else(|| format!("no number before {word:?} on {line:?}")) } -/// Connects the mixer has *applied* since `from`, which is a different event -/// from the control thread's `opening stream` — the T14 log carries the second -/// without the first. -fn connects_since(log: &str, from: usize) -> usize { - soundd_clients_since(log, from, " connected") -} - -/// Clients the mixer has ramped out and dropped since `from`. -fn removals_since(log: &str, from: usize) -> usize { - soundd_clients_since(log, from, " removed") -} - -fn soundd_clients_since(log: &str, from: usize, verb: &str) -> usize { - log[from..] - .lines() - .filter(|l| l.contains("soundd: client ") && l.contains(verb)) - .count() -} - /// The direct regression for the readiness defect: a stimulus that produces /// bytes and no events must produce no wake. Pause is that stimulus — six /// bytes, deliberately swallowed. @@ -11605,7 +10474,7 @@ fn run_machine_test( // The lost-wake canary with the window it guards held open: every pipe // wait reads its condition, waits for a post to land, then parks, so // the ping-pong's posts land between the two. A commit that ignored the - // notified bit parks for good and the canary counts it short. + // notified bit parks for good, and the run's ceiling reds it. "blocking_read_window" => { let options = BootOptions { kernel_params: &["watch-window"], @@ -11712,30 +10581,11 @@ fn run_machine_test( "kernel_heartbeat" => { // The instrument for a machine whose log cannot say whether it was // alive: ten of the owner's boots are byte-identical between the - // ones that froze and the ones that did not. The gate has to prove - // three things a `must_say` cannot — that the lines *keep coming*, - // that no CPU drops out of the mask on a machine with nothing to - // do, and that no window between two lines is wide enough to hide a - // death. - // - // The second is why this asserts a *constant full* mask where the - // old gate asserted a *varying* one — and the old gate's assertion - // was satisfied by the defect, so it certified it. Same guest with - // the tick removed: 10 of 11 lines below `alive=8/8`, six of them at - // `alive=2/8`, 56 lines naming a silent CPU and one silent for - // 2.811 s. Every line is `8/8` with the tick. - // - // The gap bound below is the one with no demonstrated teeth here: - // without the tick the widest gap was still 0.260 s, because QEMU's - // devices keep waking *someone* even when they wake nobody in - // particular. It is carried for the metal log, where the same code - // left gaps of 14 s to 102 s. - // - // **What none of it establishes**: a QEMU guest is never as quiet as - // the owner's laptop. This proves the tick arms, fires and re-arms, - // and that the instrument reads a full mask when nothing is wrong. - // It cannot prove the T14's LAPIC keeps counting through whatever - // its firmware does with a halted core. + // ones that froze and the ones that did not. What a guest can prove + // of it is that the lines *keep coming*, that each carries the + // pin's state beside it, and that the pin's state is read off the + // chip. Whether a CPU drops out of the mask, and how wide a window + // between two lines is, are the T14's to judge. let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/metalcase"); let options = BootOptions { profile: qemu::Profile::Metal, @@ -11745,12 +10595,11 @@ fn run_machine_test( }; // **A heartbeat and the `i8042: line` under it are one reading**, // and `heartbeat::poll` emits them as two `log!`s — so a capture can - // end between them. Run `31273373928` on `main` did: twelve beats, - // eleven pin readings, and the last beat was the last line of the - // log. Counting the two kinds against each other reads that as a pin - // whose state was unreadable, which is the one thing this pairing - // exists to detect. So the unit is the pair, and a beat with nothing - // after it at all is a reading this capture does not hold. + // end between them. Counting the two kinds against each other reads + // that as a pin whose state was unreadable, which is the one thing + // this pairing exists to detect. So the unit is the pair, and a beat + // with nothing after it at all is a reading this capture does not + // hold. fn whole(log: &str) -> (Vec<&str>, Vec) { let captured: Vec<&str> = log.lines().collect(); let at: Vec = captured @@ -11763,51 +10612,31 @@ fn run_machine_test( let kept = captured.len() - usize::from(torn); (captured[..kept].to_vec(), at[..at.len() - usize::from(torn)].to_vec()) } - - // The mask is a claim only about a settled machine that is running, - // and `toyos_build::heartbeat` is where that is decided — including - // which line each `[boot] start` program says it has finished - // starting with, held there against this config's own list. - let said = heartbeat::done_lines(&toyos_build::build::boot_start( - &config.join("system.toml"), - ))?; + /// Whole beats the verdict is read from. + const BEATS: usize = 5; let mut qemu = QemuInstance::boot_with_options(&config, &[], &[], options); let mut log = qemu.boot_log().to_string(); - // **The capture follows the window, not the clock.** How long the - // started programs take is the loaded host's to decide, and a - // capture cut a fixed span after `===READY===` hands the predicate - // whatever start-up left over — so a slow boot reds the test on the - // predicate's own refusal. The drain ends when the window holds - // `CAPTURE_BEATS`, which at a 250 ms period is under two seconds of - // settled machine; the bound is the liveness ceiling and not the - // capture's length, and it is counted in *steps* rather than - // measured in wall time, because a guest that has exited - // disconnects the reader and a step then returns at once. + // The bound is the liveness ceiling and not the capture's length, + // and it is counted in *steps* rather than measured in wall time, + // because a guest that has exited disconnects the reader and a step + // then returns at once. const DRAIN_STEP: Duration = Duration::from_millis(500); const DRAIN_STEPS: u32 = 40; - let drained = Instant::now(); - let mut held = 0; for _ in 0..DRAIN_STEPS { - log.push_str(&qemu.drain_serial(DRAIN_STEP)); - held = heartbeat::window_beats(&whole(&log).0, &said); - if held >= heartbeat::CAPTURE_BEATS { + if whole(&log).1.len() >= BEATS { break; } + log.push_str(&qemu.drain_serial(DRAIN_STEP)); } - if held < heartbeat::CAPTURE_BEATS { + let (captured, at) = whole(&log); + if at.len() < BEATS { return Err(format!( - "{held} heartbeat(s) with a whole period after the boot's start-up against \ - the {} a verdict is taken from, after {} drains of {} ms at a 250 ms period \ - ({:.1}s) — the instrument has to keep reporting past the start-up, and a log \ - that stops, or a boot that never finishes starting, says nothing\n{log}", - heartbeat::CAPTURE_BEATS, - DRAIN_STEPS, - DRAIN_STEP.as_millis(), - drained.elapsed().as_secs_f64(), + "{} whole heartbeat(s) against the {BEATS} a verdict is taken from, after \ + {DRAIN_STEPS} drains — the instrument has to keep reporting\n{log}", + at.len(), )); } - let (captured, at) = whole(&log); let beats: Vec<&str> = at.iter().map(|&i| captured[i]).collect(); // Each pair, positionally: `report_line` is the statement after the // heartbeat's `log!`, and another CPU's line may land between the @@ -11832,79 +10661,6 @@ fn run_machine_test( unpaired.iter().take(4).cloned().collect::>().join("\n"), )); } - let settled = match heartbeat::settle(&captured, &said) { - Ok(settled) => settled, - Err(heartbeat::Refused::Unreadable(line)) => { - return Err(format!( - "a heartbeat carries no readable t=, alive=, mask=, ran= or gap= — the \ - fields that say which CPU stopped and whether the machine ran: \ - {line}\n{log}" - )); - } - Err(heartbeat::Refused::BootUnfinished(line)) => { - return Err(format!( - "the boot never finished starting: nothing said {line:?}, so every \ - heartbeat is inside the start-up and none is a claim about a settled \ - machine\n{log}" - )); - } - Err(heartbeat::Refused::Unsettled { settled, beats }) => { - return Err(format!( - "too few heartbeats have a whole period after the last `[boot] start` \ - program finished starting — the machine did not settle inside this \ - capture, and a clear bit before it settles says nothing\n\ - {settled} of {beats}\n{log}" - )); - } - Err(heartbeat::Refused::NotRunning { beat, held }) => { - return Err(format!( - "the machine was not running: a settled heartbeat came late or dispatched \ - nothing — a guest its host did not schedule, and what a CPU did in a \ - period the machine did not run is unreadable\n\ - t={}.{:03}s gap={}.{:03}s ran={} against a {} ms period, after {held} \ - settled heartbeat(s) it ran through\n{}\n{log}", - beat.t_ms / 1000, - beat.t_ms % 1000, - beat.gap_ms / 1000, - beat.gap_ms % 1000, - beat.ran, - heartbeat::PERIOD_MS, - captured[beat.line], - )); - } - Err(heartbeat::Refused::CpuMissing { cpus, settled, opened }) => { - return Err(format!( - "cpu{cpus:?} missing from {} consecutive heartbeats on a settled guest \ - that was running — a CPU that misses two lines has missed five \ - `diag-tick` wakes, so a clear bit does not mean that CPU stopped, which \ - is the whole of the field\n{settled} settled heartbeats\n{}\n{log}", - heartbeat::STOPPED_BEATS, - captured[opened..] - .iter() - .filter(|l| l.contains("heartbeat: ")) - .take(16) - .copied() - .collect::>() - .join("\n"), - )); - } - }; - let quiet = &settled.beats; - let blips = settled.blips; - // And no window between two lines may be wide enough to hide a - // death. The metal boots this exists for went quiet for between 14 s - // and 102 s; four times the period is far below any of them and far - // above anything a loaded host does to a 250 ms cadence. - const MAX_GAP_MS: u64 = 1000; - let worst = settled.widest_gap_ms; - if worst > MAX_GAP_MS { - return Err(format!( - "the widest window between two heartbeats was {}.{:03}s against a 250 ms \ - period — the machine stopped reporting for long enough to have died in\n{log}", - worst / 1000, - worst % 1000, - )); - } // And the clock in the line advances, or the timestamp cannot // localise a death. let stamps: Vec<&str> = beats @@ -11986,17 +10742,10 @@ fn run_machine_test( )); } eprintln!( - " [heartbeat] {} whole lines in {:.1}s, each with its own pin reading, {} \ - before the machine settled and {} after, {blips} of those missing a CPU for one \ - line and none for two, widest gap {}.{:03}s, t={} → t={}; \ - {} i8042 line reading(s), vec 0x{vector} on gsi {kbd_gsi}, none masked, none with \ + " [heartbeat] {} whole lines, each with its own pin reading, t={} → t={}; {} \ + i8042 line reading(s), vec 0x{vector} on gsi {kbd_gsi}, none masked, none with \ OBF set", beats.len(), - drained.elapsed().as_secs_f64(), - beats.len() - quiet.len(), - quiet.len(), - worst / 1000, - worst % 1000, stamps.first().unwrap_or(&"?"), stamps.last().unwrap_or(&"?"), lines.len(), @@ -12142,12 +10891,6 @@ fn run_machine_test( "blockd_lends_within_its_bound" => { common::blockd::blockd_lends_within_its_bound(test_config, c_bins, rust_bins) } - // Body in `tests/common/hda.rs`, same reason. - "hda_tone" => common::hda::hda_tone(test_config, c_bins, rust_bins), - "hda_client_stall" => common::hda::hda_client_stall(test_config, c_bins, rust_bins), - "hda_two_live_refused" => { - common::hda::hda_two_live_refused(test_config, c_bins, rust_bins) - } "double_fault_stack" => faults::double_fault_stack(test_config, c_bins, rust_bins), "syscall_window_nmi" => faults::syscall_window_nmi(test_config, c_bins, rust_bins), "syscall_window_nmi_controls" => { @@ -12161,12 +10904,6 @@ fn run_machine_test( "diskless_boot" => faults::diskless_boot(test_config, c_bins, rust_bins), "virtio_net_no_msix" => faults::virtio_net_no_msix(), "pci_claim_caps_truncated" => faults::claim_caps_truncated(), - // Body in `tests/common/audio.rs`, so the hunk here stays one line. - "metal_sim_null_audio" => audio::null_sink_real_rate(test_config, c_bins, rust_bins), - "null_sink_shipped_client" => audio::null_sink_shipped_client(test_config, c_bins, rust_bins), - "doom_sound_flood" => audio::doom_sound_flood(rust_bins), - "doom_music" => doom_music(rust_bins), - "soundd_log_stall" => audio::soundd_log_stall(rust_bins), "metal_sim_compositor" => { metal_sim_compositor(group_boot(held, METAL_SIM_DESKTOP, || { boot_metal_sim_desktop(rust_bins) @@ -12244,7 +10981,6 @@ fn run_machine_test( "toolkit_window_wake" => toolkit_window_wake(rust_bins), "toolkit_winit_loop" => toolkit_winit_loop(rust_bins), "toolkit_winit_pace" => toolkit_winit_pace(rust_bins), - "desktop_audio_client" => desktop_audio_client(), "blocked_dump" => blocked_dump(), "xhci_many_devices" => { // The T14's internal controller carries a camera, Bluetooth and a @@ -13332,27 +12068,6 @@ fn run_machine_test( // budget were each caught by nothing on hardware, however green the // simulator was. // - // **What the simulator cannot say.** Two of the three instruments - // are asserts about state, and the sim checks those globally and - // better. The third is a measurement of *cost*, and the sim's clock - // does not advance inside a step — `scenarios::overlong_pass` feeds - // the recorder a modelled pass cost, which proves the recorder - // compiles and counts, not what a real pass on real silicon costs. - // Only a booted kernel reads a TSC. - // - // **And the cost half is gated here rather than in the kernel, and - // against a recorded sample rather than against the budget.** What - // a pass measures is wall clock across the pass, and a guest's wall - // clock runs while the host has taken its vCPU away — so the - // quantity includes a term the host's scheduler sets. Measured - // 2026-08-18, that term moves *every* order statistic and not only - // the tail, so `common::passcost` judges each accelerator against - // what that accelerator has been recorded producing, and takes no - // verdict at all where the recorded sample supports none. Its own - // two-directions self-check runs first, because a gate that must - // stay green under host descheduling has to be shown doing so on a - // case no booted machine can stage. - // // **The workload is `sched_stress`** because the asserts are dense // on exactly what it does: it spawns burners that drive vruntime, // blocks and wakes across io_uring and ports, and forces a @@ -13368,7 +12083,6 @@ fn run_machine_test( // 0 of 3 assert texts in the shipping kernel, 3 of 3 in this one. // This half is the other one: on a machine that really carries them, // honest work does not trip them. - common::passcost::self_check()?; let mut qemu = QemuInstance::boot_with_options( test_config, c_bins, @@ -13403,35 +12117,6 @@ fn run_machine_test( for line in result.stdout.lines() { eprintln!(" [sched-check] {}", line.trim()); } - // The whole boot, in the three pieces a capture comes in: the ready - // marker, the hole after it, and the test window. The counters are - // cumulative since boot, so the last line each CPU published is the - // whole of that CPU's run. - let mut capture = serial::Serial::boot(&qemu); - capture.push(&result.before); - capture.push(&result.serial); - let reports = common::passcost::reports(capture.text()); - if reports.is_empty() { - return Err(format!( - "the check build published no pass-cost report at all, so nothing above \ - gated what a pass costs — every pass on this boot went unmeasured or \ - unspoken. `{}` is the prefix that never appeared:\n{}", - toyos_sched::cpu::PassCostReport::PREFIX, - capture.text(), - )); - } - // Which recorded sample this run is judged against, before the - // numbers it judges: a verdict taken against a sample is - // unreadable without naming the sample, and a run that judged - // nothing has to say so where a reader cannot miss it. - let baseline = common::passcost::baseline(); - eprintln!(" [sched-check] {}", common::passcost::judgement_line(baseline)); - for report in &reports { - eprintln!(" [sched-check] {}", common::passcost::describe(report)); - } - for report in &reports { - common::passcost::verdict(report, baseline)?; - } Ok(()) } "klogd_hosted" => { @@ -14654,81 +13339,6 @@ fn run_machine_test( ); lapic_vectors(qemu.boot_log()) } - "panic_halts_the_others_first" => { - // **A kernel that has declared itself corrupt runs nothing else.** - // `halt_all_cpus` sends the halt IPI before anything else it does; - // a fatal path that waited first — for a log, a drain, anything — - // would leave every other CPU running userland under it. - // `test_rs_panic_halts_first` keeps three siblings making kernel - // records while its main thread goes fatal, and every record is - // stamped on the kernel's clock with its CPU: none of another CPU - // may be stamped past the fatal record by more than a sibling can - // take to reach its next instruction boundary with `IF` set. - // - // **The bound is 100 ms, against the derivation**: an IPI is - // taken at the sibling's next instruction boundary with `IF` set, - // so a sibling runs past the fatal record by at most the longest - // window this kernel holds `IF` clear, and every such window is - // bounded in milliseconds. - const BOUND_MS: u64 = 100; - const RECORD: &str = "syscall 26 is retired"; - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { smp: 4, kernel_features: ACTUATOR_KERNEL, ..Default::default() }, - ); - writeln!(qemu.stdin_mut(), "run test_rs_panic_halts_first").map_err(|e| format!("stdin: {e}"))?; - qemu.flush_stdin(); - let mut console = - qemu.drain_until(Duration::from_secs(30), |l| l.contains(FATAL_HALT_NONCE)); - if !console.contains(FATAL_HALT_NONCE) { - return Err(format!("{FATAL_HALT_NONCE:?} never reached the console\n{console}")); - } - // What the fatal path flushes after the nonce. The machine is - // halted and says nothing more, so this is a pace, not a guard. - console.push_str(&qemu.drain_serial(Duration::from_secs(3))); - let stamp = |line: &str| -> Option<(u64, u32)> { - let head = line.split_once("[kernel ")?.1.split_once(']')?.0; - let (secs, cpu) = head.split_once(" cpu")?; - let (s, ms) = secs.split_once('.')?; - let cpu = cpu.split(' ').next()?; - Some((s.parse::().ok()? * 1000 + ms.parse::().ok()?, cpu.parse().ok()?)) - }; - let Some((fatal_ms, fatal_cpu)) = - console.lines().find(|l| l.contains(FATAL_HALT_NONCE)).and_then(stamp) - else { - return Err(format!("no stamped {FATAL_HALT_NONCE:?} record on the console\n{console}")); - }; - let siblings: Vec<(u64, u32, &str)> = console - .lines() - .filter(|l| l.contains(RECORD)) - .filter_map(|l| stamp(l).map(|(ms, cpu)| (ms, cpu, l))) - .filter(|&(_, cpu, _)| cpu != fatal_cpu) - .collect(); - // Non-vacuity: another CPU was making records up to the fatal one. - if !siblings.iter().any(|&(ms, _, _)| ms + 1000 >= fatal_ms) { - return Err(format!( - "no other CPU's record in the second before the fatal one at {fatal_ms} ms, so \ - nothing was running to be halted\n{console}" - )); - } - if let Some(&(ms, cpu, line)) = siblings.iter().max_by_key(|&&(ms, _, _)| ms) { - if ms > fatal_ms + BOUND_MS { - return Err(format!( - "cpu{cpu} made a record {} ms after the fatal one on cpu{fatal_cpu}: the \ - fatal path let it run\n {line}", - ms - fatal_ms - )); - } - } - eprintln!( - " [panic] {} record(s) of other CPUs; the last {} ms after the fatal one", - siblings.len(), - siblings.iter().map(|&(ms, _, _)| ms.saturating_sub(fatal_ms)).max().unwrap_or(0) - ); - Ok(()) - } "virtio_used_ring" => { // Both fields of a virtqueue used-ring element are written by the // device, and on virtio-sound's control and event queues the ring @@ -15237,21 +13847,6 @@ fn run_machine_test( Ok(()) } "i8042_absent" => { - // A/B in one session: the guest's own `Boot: complete (Nms)` is - // the instrument, because host-side timing here is dominated by - // image builds. A wait-loop bug that costs a second on a machine - // with a controller costs a minute on one without. - let with = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { profile: qemu::Profile::Metal, ..Default::default() }, - ); - let with_log = with.boot_log().to_string(); - let with_ms = boot_millis(&with_log) - .ok_or_else(|| format!("no `Boot: complete` line:\n{with_log}"))?; - drop(with); - let without = QemuInstance::boot_with_options( test_config, c_bins, @@ -15280,29 +13875,19 @@ fn run_machine_test( )); } // The floating bus, not any of the sixteen handshake refusals: on a - // machine with nothing there the probe must cost one `inb`, and - // that is also what makes the timing assertion below tight. + // machine with nothing there the probe must cost one `inb`. let want = "i8042: absent — port 0x64 reads 0xff"; if !log.contains(want) { return Err(format!("no `{want}` line on a machine with no i8042:\n{log}")); } - let without_ms = boot_millis(&log) - .ok_or_else(|| format!("no `Boot: complete` line:\n{log}"))?; - // The regression this guards is 2100 ms: with no floating-bus test - // the very first `wait_writable` sees IBF set in 0xff and waits out - // the whole init budget. The allowance is for boot-to-boot noise - // between two QEMU launches in one session, nothing else. - if without_ms > with_ms + 300 { - return Err(format!( - "boot took {without_ms}ms without an i8042 and {with_ms}ms with one — a wait is not bounded" - )); + if boot_millis(&log).is_none() { + return Err(format!("no `Boot: complete` line:\n{log}")); } eprintln!(" [i8042] firmware: {}", claim.trim()); eprintln!( " [i8042] {}", log.lines().find(|l| l.contains(want)).unwrap_or_default().trim() ); - eprintln!(" [i8042] boot {without_ms}ms without vs {with_ms}ms with"); Ok(()) } "i8042_quarantine" => { @@ -16066,7 +14651,6 @@ fn run_machine_test( "netd_slow_reader" => netd_slow_reader(rust_bins), "netd_refused_pipes" => netd_refused_pipes(rust_bins), "netd_held_open" => netd_held_open(rust_bins), - "netd_stalled_peer" => netd_stalled_peer(rust_bins), "netd_udp_refused" => netd_udp_refused(rust_bins), "netd_udp_any_address" => netd_udp_any_address(rust_bins), "dns_resolve" => dns_resolve(), @@ -17609,6 +16193,20 @@ fn acpi_table_inventory(log: &str) -> Result<(), String> { Ok(()) } +/// The LAPIC timer and the TSC each calibrated to a frequency. +fn timer_calibration(log: &str) -> Result<(), String> { + let lapic_hz = number_between(log, "ticks/10ms, so ", "Hz")?; + if lapic_hz == 0 { + return Err("the LAPIC timer calibrated to no frequency at all".to_string()); + } + let measured = number_between(log, "clock: TSC measured ", "Hz against the HPET")?; + if measured == 0 { + return Err("the TSC calibrated to no frequency at all".to_string()); + } + eprintln!(" [timer] TSC {measured}Hz measured; LAPIC {lapic_hz}Hz"); + Ok(()) +} + /// The TSC the whole machine is timed by, against the frequency the part itself /// states. /// @@ -17616,30 +16214,24 @@ fn acpi_table_inventory(log: &str) -> Result<(), String> { /// is derived from the HPET calibration, so it can only agree with itself; /// CPUID leaf 15H's crystal ratio and leaf 16H's base frequency are the CPU's /// own statement, arrived at by neither the HPET nor the counting loop. A part -/// that states neither is a fact about the part, not a failure — `qemu64`, this -/// host's guest CPU, is one — so the ppm bound is asserted only where a -/// statement exists. -fn timer_calibration(log: &str) -> Result<(), String> { +/// that states neither is a fact about the part, not a failure, so the ppm +/// bound is asserted only where a statement exists. Judged on metal only: the +/// calibration is a span of a clock, and a guest's clock runs while its host +/// has the vCPU. +fn tsc_agrees_with_cpuid(log: &str) -> Result<(), String> { /// One percent, which is the widest two timebases can differ and still be /// counting the same second. A refusal and not a measurement: it catches a /// machine whose HPET and CPUID have stopped agreeing at all, and nothing /// narrower is true of every part this kernel may boot on. const CEILING_PPM: u64 = 10_000; - let lapic_hz = number_between(log, "ticks/10ms, so ", "Hz")?; - if lapic_hz == 0 { - return Err("the LAPIC timer calibrated to no frequency at all".to_string()); - } let measured = number_between(log, "clock: TSC measured ", "Hz against the HPET")?; - if measured == 0 { - return Err("the TSC calibrated to no frequency at all".to_string()); - } let Ok(stated) = number_between(log, "CPUID states ", "Hz,") else { let why = log .lines() .find(|l| l.contains("CPUID leaves 15H and 16H")) .ok_or("neither a stated frequency nor the record saying there is none")?; - eprintln!(" [timer] TSC {measured}Hz, LAPIC {lapic_hz}Hz — {}", why.trim()); + eprintln!(" [timer] TSC {measured}Hz — {}", why.trim()); return Ok(()); }; let ppm = number_between(log, "Hz, ", "ppm apart")?; @@ -17649,9 +16241,7 @@ fn timer_calibration(log: &str) -> Result<(), String> { {ppm}ppm apart, over the {CEILING_PPM}ppm this bound allows" )); } - eprintln!( - " [timer] TSC {measured}Hz measured, {stated}Hz stated, {ppm}ppm apart; LAPIC {lapic_hz}Hz" - ); + eprintln!(" [timer] TSC {measured}Hz measured, {stated}Hz stated, {ppm}ppm apart"); Ok(()) } @@ -17744,12 +16334,6 @@ fn tlb_shootdown_cost(log: &str, cpus: u32) -> Result<(u64, u64), String> { /// a program gets off that machine. **They must carry the same number**, or the /// metal readback is reporting something the guest did not measure. fn latency_wake(rust_bins: &[(String, Vec)]) -> Result<(), String> { - /// **Derived from the instrument, not from this host.** A p99 at the - /// histogram's last bucket is a floor and not a measurement, so what is - /// asserted here is that the figure is one — TCG under a twelve-wide suite - /// is no latency instrument, and the number this measures on hardware is - /// the T14's, priced as `latency.p99_us` in `tests/metal-profile.toml`. - const HISTOGRAM_US: u64 = 4096; const WAIT: Duration = Duration::from_secs(60); let config = compile::repo_root().join("tests/latencycase/system.toml"); @@ -17822,12 +16406,6 @@ fn latency_wake(rust_bins: &[(String, Vec)]) -> Result<(), String> { } bootlog::verdict(&text).map_err(|unfit| format!("{name}: {unfit}\n{text}"))?; - if printed as u64 >= HISTOGRAM_US { - return Err(format!( - "the p99 landed in the histogram's last bucket, so {printed}us is a floor and not a \ - measurement: {distribution}" - )); - } eprintln!(" [latency] {}", distribution.trim()); eprintln!(" [latency] p99 {printed}us, off the stick's own `exit: cyclictest` record too"); Ok(()) @@ -19680,7 +18258,7 @@ fn read_durations(path: &Path, out: &mut BTreeMap) { let Ok(text) = fs::read_to_string(path) else { return }; for line in text.lines() { // `