diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index e09e03cc118..c5409f917fd 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -178,26 +178,6 @@ jobs: key: guest-${{ github.run_id }} - *serial - # Gate A's thorough tier, `tests/audio-baseline.toml`: N boots per config, - # strictly one at a time, as two shards of the same configs. - audio: - needs: build - runs-on: ubuntu-24.04 - timeout-minutes: 180 - strategy: - fail-fast: false - matrix: - shard: [1, 2] - container: *kvm - env: - GH_TOKEN: ${{ github.token }} - steps: - - *deps - - *checkout - - *guest-cache - - run: cargo run -- --ci audio ${{ matrix.shard }}/2 - - *serial - # "Only Rust and QEMU", run rather than argued: `cargo run -- --build-only` # from a fresh machine. `sid` as it stands, image and archive both, and no # cache — a fresh machine is the premise. @@ -255,7 +235,7 @@ jobs: # One standing issue, found by title and commented on; a dispatch is somebody # watching the run, so only the schedule files. nightly-red: - needs: [host, build, guest, tcg, audio, portability-linux, portability-macos] + needs: [host, build, guest, tcg, portability-linux, portability-macos] if: ${{ !cancelled() && github.event_name == 'schedule' }} runs-on: ubuntu-latest permissions: diff --git a/CLAUDE.md b/CLAUDE.md index 386aa11acb3..5e0d25ff176 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -92,7 +92,7 @@ The root `Cargo.toml`'s `[workspace]` `members` and `exclude` lists account for - **Never rewrite history, and never touch `main`.** No `--amend`, no `rebase`, no `--force` — on your own branch as much as anywhere: a pushed hash may already be cited. `main` is protected — PR required, no force-push, no deletion, no bypass. - **A red test is a defect unless `src/redlist.rs` disables it with its issue (`cargo run -- --known-red `); a flaky test is disabled at once, never re-run.** - **A high-risk change names its two checks.** Security boundaries, the scheduler, the ABI, filesystems, devices, memory management, concurrency primitives: the PR names the negative control or mutation that fails if the implementation is wrong, and one epistemically independent oracle — an external specification, a differential implementation, real hardware, a third-party checker, a formal model, or a recorded real failure. A second agent is not independence: five artifacts from one wrong model still agree. A mutation is a negative control only if it reverts the *whole* change onto the base the green arm was measured on — a one-line revert of a change that moved two things measures neither. -- **Host load is not an excuse.** A load-coincident audio failure is investigated as a real defect, never re-run away as noise; evidence against that assumption goes to the owner, not into quiet workarounds. +- **Timing and audio verdicts come only from metal.** A QEMU test asserts order, completion, content and counts, never how long something took, and plays no audio; its only clock is a hang ceiling. - **Subagents wait in the foreground** — background notifications do not reliably re-wake them: explicit `timeout`s, and for longer work background once and block with a few long foreground waits, polling before each sleep. - **An agent never waits on CI.** It arms auto-merge, reports, and exits. Sequencing across landings belongs to the orchestrator, done in passes on its own wake-ups; several finished branches land as one batch PR rather than as one agent babysitting N cycles. - **Subagents get an explicit model, never the session default.** The orchestrator scopes, dispatches and verifies; it edits nothing. Match the tier to the judgment in the task: judgment-bearing coding gets a frontier model, mechanical execution from an exact brief a mid tier, and non-coding mechanical work the cheapest. Never encode a temporary usage circumstance as a rule. diff --git a/Cargo.toml b/Cargo.toml index aff7f48c5ef..661c2ad735b 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -107,10 +107,9 @@ default-run = "toyos-build" fatfs = "0.3.6" fontdue = "0.9" gpt = "3.1.0" -# `statvfs` for `worktree::free_bytes`; `getloadavg` and the libproc pair -# (`proc_listallpids`/`proc_pidpath`) for gate A's host-conditions line; and the -# unprivileged ICMP datagram socket `icmp::echo` asks a metal boot over, which -# is the platform call that replaces a `ping` binary. +# `statvfs` for `worktree::free_bytes`, and the unprivileged ICMP datagram +# socket `icmp::echo` asks a metal boot over, which is the platform call that +# replaces a `ping` binary. libc = "0.2" toml = "0.8" uuid = { version = "1", features = ["v4"] } @@ -172,25 +171,18 @@ toyos-userbound = { path = "toyos-userbound" } # declaration. Pure and `no_std`, so it compiles for the host as it does for # the guest. toyos-i219 = { path = "toyos-i219" } -# The scheduler core, for the one type the check build publishes and the -# harness judges: `cpu::PassCostReport` is the wire form of the pass-cost -# distribution, and its `Display`/`parse` pair is what keeps the kernel that -# writes it and `tests/common/passcost.rs` that reads it on one format. -# -# `check` because that is where the type lives, and it lives there because -# `src/build.rs`'s artifact gate asks the *shipping* kernel to carry none of the -# check instruments' literals — a linker may keep a string constant no code -# reaches, so an unconditional `Display` puts the report's prefix in every -# image. This is a dev-dependency of a host binary and `kernel/` is excluded -# from this workspace with its own lockfile, so nothing here reaches the kernel's -# resolution of the same crate. +# The scheduler core, for two things the harness reads the kernel's words by: +# the watch window's count (`watch::window`) that `blocking_read_window`'s +# metal row judges, and `cpu::PassCostReport`, whose `Display`/`parse` pair +# keeps the check build that publishes the pass-cost report and +# `sched_check_build` that reads it on one format. `check` because that is +# where the report lives: `src/build.rs`'s artifact gate asks the shipping +# kernel to carry none of the check instruments' literals. `kernel/` is +# excluded from this workspace with its own lockfile, so nothing here reaches +# the kernel's resolution of the same crate. toyos-sched = { path = "toyos-sched", features = ["check"] } -# The root-hub port machine, for the three durations `xhci_slow_connect` derives -# its bounds from: `DEBOUNCE_NS`, `EMPTY_BUS_NS` and `SLOW_CONNECT_NS` are -# declared once there and `use`d by the driver, so the window the harness -# certifies and the one the driver holds cannot drift apart. Also the Bulk-Only -# phases, so the harness judging a wedge reads the word `toyos_xhci::bot::Phase` -# declares instead of spelling it a second time. +# The Bulk-Only phases, so the harness judging a wedge reads the word +# `toyos_xhci::bot::Phase` declares instead of spelling it a second time. toyos-xhci = { path = "toyos-xhci" } # The second reader `pkg_install_gbae` is judged against: what is read back off # the guest's volume is compared with a third party's decoding of the committed diff --git a/issues/README.md b/issues/README.md index 1ee48f15ede..bcf697a0fb6 100644 --- a/issues/README.md +++ b/issues/README.md @@ -102,7 +102,7 @@ renamed in the commit that corrects the body, with every citation moved. ## Pointing at one -**Name the file, not the directory.** `issues/audio/hda-tone-phase-check.md` +**Name the file, not the directory.** `issues/audio/null-sink-applies-one-connect.md` is a claim something can check; `issues/audio/` is a claim that an area exists, which says nothing about whether the entry you meant is still there. diff --git a/issues/audio/a-megabyte-written-to-the-stick-starves-a-tone-beside-it.md b/issues/audio/a-megabyte-written-to-the-stick-starves-a-tone-beside-it.md index 1b05756dd65..832568b9c9d 100644 --- a/issues/audio/a-megabyte-written-to-the-stick-starves-a-tone-beside-it.md +++ b/issues/audio/a-megabyte-written-to-the-stick-starves-a-tone-beside-it.md @@ -25,5 +25,6 @@ the stick while a tone plays is the same stimulus. ## Exit condition -`audio_tone_load` at eight CPUs green across repeated boots while a guest -program writes a mebibyte to `/log` during the tone. +On the T14, a `METAL` row plays `hda_tone`'s tone while a program writes a +mebibyte to `/log`, and soundd reports `underruns=0` over the tone's window +across repeated boots. Owner: the metal suite (`tests/toyos.rs`'s `METAL`). diff --git a/issues/audio/client-ring-depth-is-the-devices-pipeline-depth.md b/issues/audio/client-ring-depth-is-the-devices-pipeline-depth.md index 3dd66132a2f..4d746b85f67 100644 --- a/issues/audio/client-ring-depth-is-the-devices-pipeline-depth.md +++ b/issues/audio/client-ring-depth-is-the-devices-pipeline-depth.md @@ -22,11 +22,6 @@ negotiation, with soundd clamping to `[num_buffers.next_power_of_two(), MAX_SLOTS]`. A client that asks for nothing gets today's value, so the default is unchanged. `num_buffers` stays a device property and stops leaking. -**Blocked on the audio gate, not on anything in soundd.** This changes the -latency and wake pattern gate A measures, so it lands in a quiet window with the -thorough tier behind it as a same-session A/B — the fast tier cannot see it, and -these counters drift between batches on one host with no code change at all. - Half of the original coupling is already gone: the `num_buffers > 5` startup panic became `deferral_floor_nanos` returning `None`, and an unrenderable shape is refused by name rather than asserted. diff --git a/issues/audio/desktop-session-put-26ms-of-silence.md b/issues/audio/desktop-session-put-26ms-of-silence.md index 0038434237f..f054d5545e7 100644 --- a/issues/audio/desktop-session-put-26ms-of-silence.md +++ b/issues/audio/desktop-session-put-26ms-of-silence.md @@ -21,9 +21,7 @@ soundd: wakes=389 completions=690 submitted=690 underruns=1 drains=2 max_wake_la Nine periods — 26 ms — submitted with a client streaming and no client audio behind them (`MixStats::period` in `toyos-mixer/src/stats.rs`, which is where that counter moved when the mixer's decisions became a pure crate). -`tests/audio-baseline.toml` records `underruns` 0 -on all 120 runs of its sample, and the fast tier's verdict is exactly this -counter. There is no capture to corroborate it: `--dump-audio` was not on. +There is no capture to corroborate it: `--dump-audio` was not on. **And it is not confined to those two windows.** Across the whole run, 15 of 119 windows report `drains` (22 events; recorded sample 0/120), and the @@ -37,8 +35,7 @@ audio_tone sample 30 5666 — — 10090 (baseline file) ``` The tone phase is 88 windows, none below 18116 us, against a recorded sample -whose *worst of 30* is 10090. The two distributions are disjoint. 106654 us is -past the `audio_tone_load.smp8` ceiling of 80000 (this guest is `--smp 8`). +whose *worst of 30* is 10090. The two distributions are disjoint. **Whose lateness it is, is not the same question in the two phases, and the `deferred` column separates them.** `deferred` counts a mix cycle declining to @@ -115,6 +112,11 @@ underruns. The counters print every 2 s while a client exists. Measured harm — 26 ms / 9 periods of silence with a client streaming, plus a tone-phase wake-lateness cluster the recorded sample never reaches — makes -this a defect under the audio law rather than a comparability note. Owed to -whoever next gives gate A a doom-plus-resume workload and settles `armed_on` -against `target` in `signal_clients`. +this a defect under the audio law rather than a comparability note. + +## Exit condition + +`armed_on` is settled against `target` in `signal_clients`, and on the T14 a +`METAL` row streams a client while `tone` restarts beside it, as the session +above did, with soundd reporting `underruns=0` in every window a client holds, +across repeated boots. Owner: the metal suite (`tests/toyos.rs`'s `METAL`). diff --git a/issues/audio/disk-wait-pins-a-cpu.md b/issues/audio/disk-wait-pins-a-cpu.md index 12259691990..a02bb9b617f 100644 --- a/issues/audio/disk-wait-pins-a-cpu.md +++ b/issues/audio/disk-wait-pins-a-cpu.md @@ -21,17 +21,6 @@ the log sink; it is that this kernel cannot touch a disk without pinning a CPU for the whole device round trip. The log sink is only the writer that runs continuously, which is why it is the one gate A sees. -`usb-slow-device` (kernel feature) holds every mass-storage bulk completion back -2 ms, which is what a USB stick's erase block does to a 4 KiB write and what -QEMU's `usb-storage` has no device, drive or machine property to express. -`cargo test --test toyos-build -- audio_tone --slow-usb` stages the T14's harm -on this host: soundd's worst wake goes 7,117 → 165,948 µs at smp=1 and -10,632 → 259,706 µs at smp=8 — 7 to 11 whole 23.2 ms pipelines — drains appear -on three of the four configs, and one boot of three submitted 76 silent periods -and tripped gate A's own harm verdict. Baseline arm at host load 5.0–6.6, slow -arm at 1.3–1.5, so the direction is not the host's. Both arms, one session, one -tree. - The log-flush deferral fix — whose affordability heuristic left the kernel with the log architecture, so no CPU takes a flush now — and [`stop-the-device-voice-keep-the-wake.md`](stop-the-device-voice-keep-the-wake.md) diff --git a/issues/audio/doom-audio-callback-stalled-on-the-t14.md b/issues/audio/doom-audio-callback-stalled-on-the-t14.md index b3a1df388e9..6a6c684ca81 100644 --- a/issues/audio/doom-audio-callback-stalled-on-the-t14.md +++ b/issues/audio/doom-audio-callback-stalled-on-the-t14.md @@ -7,8 +7,7 @@ opened: 2026-08-05 # Nothing explains why doom's audio callback stalled on the T14 doom's sound producer can no longer kill the game when its callback stops -consuming — `doom_sound_flood` stages exactly that and the game lives — and -that is the whole of what is now known. **Why the callback stopped on the +consuming, and that is the whole of what is now known. **Why the callback stopped on the owner's machine, about five seconds into play, is not.** The evidence is one abort message, because the process that would have carried the answer is the one that died. @@ -29,19 +28,5 @@ where any syscall on that thread can become the USB driver's engine for as long as a second; and plain scheduling pressure from a game thread and a compositor that never yield to it. -**One runner sighting is filed under this name and is not this defect.** In run -`31247206462`, `doom_sound_flood` was `timed out after 88s` when re-run alone, -against 4–26 s on the dev host, and 0 of 5 in the rate probe five days later — -a sighting without a rate, carried as a row in `src/redlist.rs`. It was one of -four reds in that run, three of them soundd's, which is why they were read -together at the time; of the other three, `metal_sim_null_audio` was a test -reading a boot line through a span of host wall clock and is closed, -`hda_client_stall` was a `DEADLOCK` between the idle loop's log-file flush and -the xHCI disk lock and is no longer reachable, and `sshd_fail_closed` is -undiagnosed and has its own row. Nothing in that run's capture names the -callback, so it neither supports the mechanism above nor rules it out. - What would decide it is the callback's own period count against wall clock, on -that machine. doom now keeps that counter (`MIXED_PERIODS`) and now survives -the stall, so the next T14 session can be asked the question instead of losing -the process that would have answered it. +that machine. diff --git a/issues/audio/doom-sound-flood-played-full-scale-once.md b/issues/audio/doom-sound-flood-played-full-scale-once.md deleted file mode 100644 index 5349b6284c4..00000000000 --- a/issues/audio/doom-sound-flood-played-full-scale-once.md +++ /dev/null @@ -1,41 +0,0 @@ ---- -status: expected-red -kind: defect -opened: 2026-09-04 ---- - -# `doom_sound_flood` played full scale once, and the volume that reached the wire was not the one the last command named - -Nightly `ci` run `33728852421`, job `100563991283` (`guest (11)`), 2026-09-03, -KVM, one guest per machine: - -``` -FAIL doom_sound_flood: the device played a peak of 32768 (expected 4000..=12000): the volume the last command named is not the volume that reached the wire - FAIL doom_sound_flood (7s) -``` - -The re-run alone in the same job was green and printed what a healthy run -prints: - -``` - [doomcase] 4096 commands issued with the callback parked, tone converged in 173 periods for 22050 frames, 7963136 concurrent commands, 21160 samples of signal at peak 7969 - PASS doom_sound_flood (8s) - ALONE doom_sound_flood: GREEN, and it was alone both times — nothing the harness controls differed, so it failed once and passed once. That is a rate and not a classification. -``` - -**Why the number matters.** `tests/common/audio.rs` derives the band from the -two outcomes the actuator can produce: 16000 × 127/255 = 7968 if the last -volume command is the one applied, and 251 if every superseded update is. 32768 -is neither. It is `-i16::MIN` — the magnitude of the most negative sample an -`i16` holds — so the capture carried a full-scale sample, eight times the -amplitude the mixer was asked for and above anything a superseded command -explains. The alone re-run's `peak 7969` is the expected outcome to one count. - -**What is not known.** Whether the full-scale sample is one sample or a span, -whether it is the mixer's sum overflowing or the analysis reading a wrapped -value, and whether a listener would hear it. Nothing in the captured line -distinguishes those, and the WAV that would is not kept by a CI job. - -**Exit condition.** The full-scale sample's cause is fixed, and -`doom_sound_flood` green on the KVM `guest` shards where it went red. Owner: -orchestrator. diff --git a/issues/audio/gate-a-first-run-to-record-its-host.md b/issues/audio/gate-a-first-run-to-record-its-host.md deleted file mode 100644 index e7a3c40970b..00000000000 --- a/issues/audio/gate-a-first-run-to-record-its-host.md +++ /dev/null @@ -1,56 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-08-07 ---- - -# The first gate A run to record its host: four of six boots outside the recorded sample, two of them harm - -2026-08-07, tree `4a0a07f`, the run that verified the host-conditions -annotation itself (`cargo test -- audio_tone`, filtered). Every line below is -the harness's own, printed beside the counters it qualifies: - -``` -audio_tone smp=1 gaps 1 [3p] underruns 3/1137 drains 8 wake_lat 86862us (3.74 pl) host: load 49.8/22.7/15.0 qemu 1 toyos-build 4 - confirm gaps none underruns 0/1136 drains 0 wake_lat 16968us (0.73 pl) host: load 49.0/23.4/15.4 qemu 1 toyos-build 4 -audio_tone smp=8 gaps 1 [1p] underruns 1/1113 drains 0 wake_lat 28118us (1.21 pl) host: load 48.3/24.1/15.7 qemu 1 toyos-build 4 - confirm gaps none underruns 0/1111 drains 0 wake_lat 8434us (0.36 pl) host: load 46.5/24.1/15.8 qemu 1 toyos-build 3 -audio_tone_load smp=1 gaps none underruns 0/1132 drains 0 wake_lat 7174us (0.31 pl) host: load 41.2/23.7/15.7 qemu 1 toyos-build 3 -audio_tone_load smp=8 gaps none underruns 0/1126 drains 0 wake_lat 23083us (0.99 pl) host: load 37.9/23.6/15.8 qemu 1 toyos-build 3 -``` - -**The invocation passed**, and correctly: harm appeared on both `audio_tone` -configs and neither reproduced on its confirming boot, which is precisely what -the two-boot rule is for. What it passed *with* is the finding. - -- `audio_tone.smp1` at **86862us — 3.74 pipeline depths, 8.6x that config's - recorded worst (10090us), and past its 56000us ceiling.** The baseline file - records `ceiling_runs = 0` across all 120 runs of the 2026-07-31 sample; this - is the first breach since. It came with `drains 8` — the ceiling exactly — - three periods of silence on the wire and a 3-period gap in the capture. -- **Two of six boots passed a whole pipeline depth** (3.74 and 1.21), which the - baseline file states no run of its 120 reached. -- `audio_tone_load.smp8` at 23083us is 2.9x its recorded worst with no harm at - all — the "bad but real" shape the ceilings exist to admit. - -**And now the conditions are on the record rather than reconstructed.** 1-minute -load 37.9-49.8 on 14 cores with three to four other `toyos-build` processes and -no other guest, against the 4.2-6.1 the 2026-07-29 ceiling derivation recorded -per run — six to twelve times it. Under the owner's ruling of 2026-08-04 that is -**not** an excuse and not grounds to re-run it away: it is a defect of the -pipeline until something shows otherwise, and it is the same shape as the -load-stall family `issues/audio/thorough-tier-reds-on-unmodified-main.md` -records, its fast-tier and 142 ms sightings included. What is new is only that -the next investigation starts from a measured host state instead of a guess. - -Whoever takes it: the thorough tier is the instrument for the rate, and it now -prints `host conditions over N runs` so its own arm's conditions can be stated. -The recorded arm's cannot — see `tests/audio-baseline.toml`. - -## Promoted 2026-08-25 - -A measured ceiling breach (86862us, 8.6x the recorded worst, past the 56000us -ceiling, with three periods of silence behind it) on a passing invocation is -harm under the audio law. Owed to whoever runs gate A's thorough tier next: -read the `host conditions over N runs` line and decide whether the ceilings -need a host-load term. diff --git a/issues/audio/gate-a-has-no-runner-baseline.md b/issues/audio/gate-a-has-no-runner-baseline.md deleted file mode 100644 index 4abe837bfc8..00000000000 --- a/issues/audio/gate-a-has-no-runner-baseline.md +++ /dev/null @@ -1,135 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-08-10 ---- - -# Gate A's thorough tier on a runner compares against the dev host's sample, and needs its own - -`tests/audio-baseline.toml`'s recorded sample was taken on the dev host under -cross-arch TCG. The thorough tier compares a fresh sample against *that*, so -`gate-a.yml` on any KVM runner is comparing two instruments and calling the -difference a regression. The gate runs on two GitHub-hosted KVM shards, so that -is what every nightly does. - -## Settled 2026-08-21: it is the instrument, and the control says so - -The earlier version of this file inferred the cross-instrument gap from level -differences. It is now measured against a same-session control, which is what -the audio law requires before a harm verdict may be set aside. - -**The experiment.** Four interleaved 15-iteration blocks of -`cargo test --test toyos-build -- --audio-gate 15 --shard 1/1 --host-slots 0` -on the T14, in the CI image of the day, `--device=/dev/kvm`, -QEMU 11.1.0 — the CI invocation, with private checkouts and a private cache root -so the runner's own state was untouched. Order A,B,A,B; 30 iterations per arm. - -* arm A = `960b96e3`, **the tree the recorded sample was taken on** -* arm B = `53101d08`, `main` - -Every block was gated on the machine being idle first and carried a witness -sampled every 10 s: no CI job container was present for any of the 240 boots, -and the 1-minute load stayed in 0.2-1.74. - -**The verdict, by the gate's own Mann-Whitney at its own alpha (1e-3, z>3.0902):** - -| config | recorded | T14 arm A | T14 arm B | A vs B (n=30) | -|---|---|---|---|---| -| `audio_tone.smp1` | 8972 | 20314 | 19994 | z=1.49 — **no difference** | -| `audio_tone.smp8` | 9249 | 14069 | 4088 | z=5.37 — B faster | -| `audio_tone_load.smp1` | 5765 | 2764 | 4352 | z=4.12 — B slower | -| `audio_tone_load.smp8` | 6097 | 12676 | 3904 | z=5.48 — B faster | - -Medians of `max_wake_lat_us`, microseconds. - -**The negative control is the whole finding: arm A fails this baseline on the -T14 and arm B does not.** Both arm-A blocks red `audio_tone.smp1` against the -recorded sample — `median 8972 -> 20438 (z=5.03)` and `8972 -> 20144 (z=4.67)` -— and both arm-B blocks print `PASS`. The tree the sample was recorded on reds -its own sample on this host, *harder* than `main` does. A level difference that -reds the recording tree is not a regression in anything. - -So gate A run `32479089989`'s verdict — -`audio_tone.smp1 wake lateness: median 8972 -> 17186 (z=4.36)`, the first -readable exit this workflow ever produced — is adjudicated: **the instrument, -not the tree.** No re-run was used to reach that; a same-session interleaved -control was. - -**Harm was null on both arms**: dropouts 0/120 and 0/120, underruns 0 in all 240 config-runs, -drains all-zero but for a handful of single events, ceiling breaches 1/120 on -arm A (a 334604 us single wake on `audio_tone.smp8`, no dropout behind it) and -0/120 on arm B. - -## What is still owed, and it is more than pasting numbers - -**Nothing here licenses replacing the recorded sample with a T14 one.** The dev -host still runs the fast tier against it, and a KVM sample would be as wrong -there as the TCG sample is on the runner. What is needed is a baseline *per -host*, and two things block writing one: - -1. **The file has no host dimension and the loader has no way to pick one.** - `AudioBaseline` is `BTreeMap>` with - `deny_unknown_fields`, and `config_baseline` (`tests/toyos.rs`) selects on - `(name, smp)` and nothing else. A `[runner]` sample is therefore a schema - change plus a selection keyed on whether the accelerator is in use — a change - to how a high-risk gate decides, not an edit to a table. -2. **The T14 distribution is bimodal per boot, and the mode's probability moves - with the tree.** Recording 30 runs of it would freeze a mixture whose mixing - weight is the thing that varies. Measured and written up in - `issues/audio/t14-wake-lateness-is-bimodal-per-boot.md`, which is what has to - be understood before any T14 number is worth recording. - - **And the mixing weight does not move only with the tree.** Four hours after - the A/B, 296 config-runs on the same host — `main` and `53101d08` - interleaved, each carrying only the wake instrument — produced the fast mode - 296 times of 296, and both 15-iteration blocks reported `[gate A] PASS` - against *this very sample*, `audio_tone.smp1` included: fresh medians 4075 - and 4143 against the recorded 8972. A T14 sample recorded on one evening - therefore would not describe the same host on another, which is a stronger - objection than the schema and the one that has to be answered first. - -The 2026-08-10 measurement this file opened with — run `31386117376`, tree -`99e47d9`, two GitHub-hosted runners of different vendors — remains the reason a -hosted sample would be a sample over two unnamed CPUs. Its `wakes` observation -still holds and the T14 reproduces it: 1185-1432 fresh against 846-905 recorded, -a KVM guest waking about 1.6x as often as the same guest under cross-arch TCG. -That artifact expired on 2026-09-16; the T14 arrays above are in this branch's -commit message. - -## Promoted 2026-08-25 - -Real, actionable work remains even though harm was null on this measurement: a -per-host baseline needs a schema change (`AudioBaseline`'s host dimension, -`tests/toyos.rs`'s `config_baseline` selection) and the T14's bimodal mixing -weight has to be understood before a sample is worth recording. Owed to -whoever owns `tests/audio-baseline.toml` and `gate-a.yml`. - -## The T14 is no longer a runner - -The gate's two shards are GitHub-hosted, and the sample they compare against is -still the dev host's under cross-arch TCG, so this record is unchanged in what -it says and only narrower in where a fresh sample can come from: a hosted -sample is a sample over unnamed CPUs, and a T14 sample is now a job of -`issues/hardware/the-t14-boots-toyos-unattended.md` rather than of a CI lane. - -## Main's nightly at 1ce71831 reds on it - -Run 36290616312, `audio (2)`: `audio_tone_load.smp1 wake lateness: median -5765 -> 6650 (Mann-Whitney z=4.03 > 3.09)`. Dropouts were 0/60, underruns 0, -ceiling breaches 0/60, and wakes 1396-1426 against the recorded 856-887, -the 1.6x KVM-over-TCG ratio this file records. The same lane on the nightly -before #527 (run 36285169430) read median 6478 with wakes 1394-1425 and -passed. On `nightly-green2` at dbf4ace5 (run 36292135439) it passed too. The -fresh samples agree with each other and differ from the dev host's TCG -sample, so this is the instrument, as above. Until a per-host baseline -exists, whether a KVM runner's gate A reds depends on where its median lands -against the TCG sample's. - -The four `audio (2)` samples of `audio_tone_load.smp1` around #527 have -medians of 6453 (before #527, passed), 6648 (main at 1ce71831, red), 6190 -(`nightly-green2` at dbf4ace5, passed) and 6520 (`nightly-green2` at -c2715880, red, run 36297455432). Mann-Whitney of each later sample against -the one before #527, on the gate's 30-value arrays: z=1.40, -1.20 and 0.84. -None is a difference at the gate's alpha. The runner's sample did not move. -The gate's verdict flips because that sample's median sits about 0.8 ms above -the TCG sample's 5765, right at the gate's edge. diff --git a/issues/audio/gate-a-suspend-structure-verdict-unread.md b/issues/audio/gate-a-suspend-structure-verdict-unread.md index b5337723c03..df25938229f 100644 --- a/issues/audio/gate-a-suspend-structure-verdict-unread.md +++ b/issues/audio/gate-a-suspend-structure-verdict-unread.md @@ -17,57 +17,18 @@ structure: no `virtio-sound: stream 0 stopped` after the last client removal — the device is still running with no clients ``` -Shard 2 of the same run failed on a statistic: - -``` -[gate A] FAILED — 1 statistic(s) regressed: - audio_tone_load.smp1 wake lateness: median 5765 -> 17684 (Mann-Whitney z=4.61 > 3.09) -``` - -**Neither was ever adjudicated, because the exit code could not tell anyone they -had happened.** That workflow's step ended in `exit "${PIPESTATUS[0]}"` under a -shell with no such array, so it reported `failure` on every run whatever the -audio said; 08-17's two FAILEDs and the PASSes on 08-16 and 08-18 arrived as the -same red. The mechanism and the full run-by-run table are in -`issues/audio/thorough-tier-reds-on-unmodified-main.md`; the exit code is fixed, -these two verdicts are not. - -**Why this one is not the cross-instrument shape.** Three of the five FAILEDs -that workflow ever printed are the runner-vs-dev-host wake-lateness level -difference `gate-a-has-no-runner-baseline` explains, and they are all from before -the 2026-08-15 re-record. These two are after it, and they sit between a -two-shard PASS night (08-16) and a two-shard PASS night (08-18) on the same -recorded sample. A level difference does not come and go for one night. - -**Shard 2's half of that is now adjudicated, and the paragraph above was half -right.** The 2026-08-21 same-session A/B on the T14 -(`issues/audio/gate-a-has-no-runner-baseline.md`) shows the level difference is -real *and* that it comes and goes: on a KVM host `max_wake_lat_us` is bimodal -per boot, ~4 ms or ~20 ms, and which mode an `smp=1` config draws moves between -runs of the same tree — see -`issues/audio/t14-wake-lateness-is-bimodal-per-boot.md`. A night that draws the -slow mode reds and a night that draws the fast one passes, on one unmodified -tree, which is exactly the 08-16 PASS / 08-17 FAILED / 08-18 PASS sequence. -`audio_tone_load.smp1` is also the one config where `main` sits worse than the -baseline tree on the T14 (z=4.12 pooled, but z=3.57 and z=2.09 across the two -interleaved block pairs), so shard 2's `5765 -> 17684` is that same unstable -config and needs no separate hunt. - -What that does **not** settle is a hosted baseline: the 08-17 run was a -GitHub-hosted runner of an unnamed vendor, and the T14 control speaks for the -T14. A hosted verdict still has no hosted sample to be read against. - -**Shard 1's is the one to take first.** `soundd: suspended` and +`soundd: suspended` and `virtio-sound: stream 0 stopped` are the two lines that say the idle path released the device; their absence says a boot left the device running with no -clients, which is the subject of `stop-the-device-voice-keep-the-wake` and -`idle-suspend-reds-on-a-loaded-host-and-on-main` from the other side. One -occurrence in 25 iterations is a rate nobody has, and the tier is the instrument -that produces one. +clients, which is the subject of `stop-the-device-voice-keep-the-wake` from the other side. One +occurrence in 25 iterations is a rate nobody has. **The evidence expires.** `/tmp/gate-a.log` is uploaded per shard with `retention-days: 30`, so run `31992902784`'s artifacts go on 2026-09-16; the job logs outlive them. Everything quoted above is already here for that reason. -Whoever takes it: `gh workflow run gate-a.yml -f iterations=30` now reports its -own verdict, so a re-dispatch is a readable experiment for the first time. +## Exit condition + +A `METAL` row reads `soundd: suspended` after each last client's removal on the +T14's HDA ring, and it holds across the T14's boots. Owner: the metal suite +(`tests/toyos.rs`'s `METAL`). diff --git a/issues/audio/hda-tone-phase-check.md b/issues/audio/hda-tone-phase-check.md deleted file mode 100644 index 3daec584bb9..00000000000 --- a/issues/audio/hda-tone-phase-check.md +++ /dev/null @@ -1,106 +0,0 @@ ---- -status: expected-red -kind: defect -opened: 2026-08-07 -task: 88 ---- - -# HDA: the captured tone is not one sine - -`hda_tone` plays the same 3.0 s 440 Hz tone the virtio arm plays, out of an -`intel-hda` controller soundd drives itself, and the capture comes back with -**8 to 16 phase discontinuities** where the virtio arm has none. Every other -assertion that test makes still reds the run. - -What is *not* wrong, measured on this host (QEMU 11.0.3, 2026-08-07): the tone -is present at full amplitude, there is **no mid-tone silence at all** (`gaps -none`), and soundd's own counters match the virtio arm's — **1127 periods -submitted, 0 underruns, 0 drains** on both. The guest put the same audio on the -wire; something between soundd's buffer and the wav file did not carry it. - -The instrument is new and calibrated: `audio::phase_breaks` tests the recurrence -`x[n+1] = 2·cos(ω)·x[n] − x[n−1]` that a sampled sinusoid obeys exactly, and it -reads **0 on all four recorded virtio configs** (`audio_tone` and -`audio_tone_load`, smp 1 and 8). It is the second of two guards on the same -property, built because the first — the driver zeroing a period's buffer once -that period has played, so an unfilled period sounds like silence on both -backends and one gap detector stays valid for each — is a design promise with -no measurement behind it. - -Evidence, and why it does not yet name a cause: - -- Six consecutive runs at `timer-period=5000` gave **8 breaks at identical frame - positions** — 2703-2705, 2821-2823, 2939-2940 — across runs whose audio - content differed (the dither seed is clock-derived, and the sample values at - those positions differ run to run). Identical positions with different content - is a capture dropping samples on a cadence, not a guest playing them wrong. -- Shortening the host's drain interval to `timer-period=1000` moved the - positions rather than removing them: one run at 0 breaks, the next at 16, - clustered around frame 95725 instead. **So it is intermittent and the host - cadence is not the whole story.** -- Within a cluster the breaks sit at multiples of **118 frames**, which is - neither the device period (128) nor either backend timer's frame count - (44.1 or 220.5). That number is unexplained and is the sharpest thing here. -- The capture also holds **2.756 s of tone where the virtio arm holds 2.94 s**, - with no seam accounting for the difference on the runs that show none. Either - the capture opens late or QEMU's `hda-codec` discards on its own ring - overrun; both are host-side and neither is established. - -Where to start: QEMU's `hw/audio/hda-codec.c` output ring against soundd's -eight-period pipeline. **Do not weaken the check to make it green**; the virtio -zero is what says it has teeth. - -**2026-08-07: one guest-side cause found and fixed, and it is not the whole of -it.** soundd filled a completion batch lowest-index-first, so a batch that -wrapped the ring — `{6,7,0,1}` — was filled 0, 1, 6, 7 and played 6, 7, 0, 1. -That is a splice with no silence in it, which is this signature exactly. Six -`hda_tone` runs in one session, instrumented with a counter of fills that were -not in the engine's order, separated cleanly: **one run at `out_of_order=2`, -`max_batch=7`, 9 phase breaks; five at `out_of_order=0`, `max_batch=4`, 0 -breaks.** It is fixed (`hda-ring-fix-unverified-on-metal` is what that fix still -owes) and the fill order is now the engine's by construction. - -**The breaks survive it.** Two runs on the fixed tree gave 8 and 6, with -`deferred=0` and an ordering that can no longer be wrong. So the ordering was a -contributor and not the cause, and the remaining one is still unnamed. What the -new numbers add: - -- The break count tracks **soundd's own wake lateness**, which on this host is - the host descheduling QEMU: 0 breaks at `max_wake_lat_us` 8626–8752 (five - runs), 8 at 22525, 9 at 16730, 6 at 50837. Nothing in the guest changed - between them. -- soundd put **1127 periods = 144,256 frames** on the wire and the capture holds - **131,061** — 9.1% of what was submitted is not in the file, with `gaps none` - and `underruns 0`. On the run at 6 breaks it is 10.8%. A capture missing a - tenth of its samples has phase breaks whatever the guest did. - -So the next step is the host side. `isr_complete` is a weaker candidate than it -was: `stream::decode` now refuses any mask that is not a walk of the ring, and -soundd fills in the order that walk names rather than in index order. - -## Re-judged on the current instrument, 2026-08-29 — it stands, and it is load-keyed - -Every capture above predates QEMU 11.1.0, the instrument change that forced the -baseline re-record (`960b96e3`). Fresh samples, `main` at `48437ca4` plus only -the idle-wake tripwire (corpus-certified to leave the audible path -bit-identical): - -- **Dev host TCG, alone: 8 of 8 clean.** `gaps none phase-breaks 0` on every - run, active 2.94 s — the virtio arm's own figure, where the broken captures - held 2.756 s — dither 24.8-26.2%. -- **Dev host TCG, beside two other suites (1-min load 5.1-21.6): 3 of 11 boots - fire** — 2, 4 and 8 breaks; the 4-break boot also put one mid-tone period of - silence in the capture, the other red's line exactly. -- **CI, hosted KVM: fires** — the 2026-08-29 scheduled run printed 4 breaks at - 2047/2048 and 2559/2560. - -Three things move under the verdict. The captures are no longer starved — every -fresh one holds 139,253-142,325 of the 144,256 submitted frames, so the -9.1%-missing mechanism above is gone. The signature changed: breaks now come as -adjacent-frame *pairs* with |period| in the hundreds, not the 118-frame -clusters. And the load dependence is sharp where it used to be a correlation: -0 of 8 alone against 3 of 11 beside other guests, same tree, same hour. - -**Exit condition.** The adjacent-frame-pair breaks' cause is fixed, and -`hda_tone` reads 0 phase breaks beside other guests on the dev host and on -CI's KVM shards. Owner: orchestrator. diff --git a/issues/audio/hda-tone-red-beyond-its-exemption.md b/issues/audio/hda-tone-red-beyond-its-exemption.md deleted file mode 100644 index 5cd53a280d9..00000000000 --- a/issues/audio/hda-tone-red-beyond-its-exemption.md +++ /dev/null @@ -1,40 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-08-07 -task: 88 ---- - -# `hda_tone` is red on `main` for a reason `#88`'s exemption does not cover - -`cargo test -- hda_tone` on `main` at `6d11938`, alone, 2026-08-07 18:5x: - -``` -FAIL hda_tone: 1 mid-tone silences in the capture: total 1 [1p×1] - FAIL hda_tone (15s) — listed against #88, and this is not that failure: - the entry covers ["the captured tone is not one sine"] -``` - -**What has changed since, and what has not.** `hda_tone` is `Tier::Nightly` for -`Why::TimerAnchored` (`src/tiers.rs`), so a plain `cargo test` no longer runs -it and a landing whose gate is `cargo test` no longer meets this red at all. - -That doesn't touch the verdict, and the verdict was owed a fresh sample: every -capture behind it had gone through QEMU's 48000→44100 resampler, since removed. - -**Re-judged 2026-08-29, and it stands.** On the current instrument (QEMU -11.1.0), `main` at `48437ca4`: alone, 8 of 8 boots clean (`gaps none`); beside -two other suites of this worktree (1-min load 5.1-21.6), **1 of 11 boots put one -mid-tone period of silence in the capture** — `1 mid-tone silences in the -capture: total 1 [1p×1]`, byte-identical to the line above, with 4 phase breaks -beside it and a confirming re-boot that came back 8 breaks and no gap. So the -red is real on the fixed pipeline, load-keyed, and roughly the shape of CI's 4 -of 5 — the re-judging that was tracked apart closed with this measurement, and -the load-stall family (`issues/audio/thorough-tier-reds-on-unmodified-main.md`) -is the standing suspect for the mechanism. - -Found while landing task #98/#12: the same test failed identically inside that -landing's gate, and the A/B against `main` in the same session is what -identified it as `main`'s. Assigning it needs whoever owns H3 — -`5fdfeb7`/`a022811` ("wip: H3, the virtio-sound stub and its userland driver") -landed hours before this measurement. diff --git a/issues/audio/idle-suspend-reds-on-a-loaded-host-and-on-main.md b/issues/audio/idle-suspend-reds-on-a-loaded-host-and-on-main.md deleted file mode 100644 index aa6a51815fd..00000000000 --- a/issues/audio/idle-suspend-reds-on-a-loaded-host-and-on-main.md +++ /dev/null @@ -1,107 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-08-20 ---- - -# `audio_idle_suspend` reds on a loaded host, on `main` as much as anywhere - -`tests/toyos-rust-tests/src/bin/audio_idle_suspend.rs` asserts §5.8's strongest -claim: on a boot where no client ever connects, soundd's summed `cpu_ns` across -both its threads is **exactly** unchanged over ~1 s. It reds routinely on the -dev host, with a delta of one to three milliseconds. - -It is **not** anybody's diff. Same-session A/B, 2026-08-20, one interleaved -session on the dev host, `cargo test --test toyos-build -- --nightly -audio_idle_suspend`, magnitudes in nanoseconds of the reported delta: - -| tree | n | min | median | mean | max | -|---|---|---|---|---|---| -| `625afce1` (`main`) | 9 | 1,457,204 | 1,942,387 | 1,939,260 | 2,240,925 | -| `cf72c3dc` (`toyos-mixer` extraction) | 14 | 1,227,121 | 1,856,260 | 1,782,107 | 2,596,139 | - -One population, and the branch's is if anything the lower of the two. The -extraction that prompted this measurement did not touch the idle path. - -**The harness already names a diagnosis, twice.** In two of five fast-tier runs -the re-run alone went green and the suite reported: - - ALONE audio_idle_suspend: GREEN — it fails only beside other guests, so its - Sched::Parallel is wrong. The run stays red on the classification. - -That is the shape of the whole thing: the delta is **exactly zero** when the -guest runs with the host to itself, and one to three milliseconds when it does -not. A soundd that genuinely spun would never produce an exact zero. - -It is also not selection-dependent, which was the first hypothesis and is wrong: -five consecutive fast-tier runs of the single test red 5/5, while the *full* -272-test fast run on the same commit passed it. Fewer guests is not quieter -here — a one-test run and a full suite differ in more than load. - -**Two candidate causes, and the experiment that separates them.** - -1. soundd takes real wakes while suspended that are cheap enough to round to - zero on a quiet host. The device path watches the audio handle `READABLE` - with `timeout = u64::MAX`; if that handle is ever readable with the stream - stopped, the mix loop turns over. -2. The guest's `cpu_ns` accounting charges a blocked thread under contention — - the delta is an artifact of when the scheduler samples, not of work done. - -They separate on **whether soundd's own wake counter moves**: `MixStats::wakes` -is zeroed when the first client arrives and never reported while idle, so -nothing could see an idle wake when this was filed. The 2026-08-29 section -below is that instrument existing; a red taken since then decides the split by -itself. - -Gate A is unaffected and green throughout: `audio_tone` and `audio_tone_load` at -smp=1 and smp=8 all pass on `cf72c3dc` with 440.0 Hz, phase-breaks 0, gaps none, -0 underruns and 0 drains. - -## 2026-08-21: it also fails by *position in the session*, which is what makes a two-arm reading of it worthless below n≈10 - -Confirmed again on the dev host against `fe41dbae` (`main`), 42 invocations in -one afternoon across three interleaved A/Bs, `cargo test --test toyos-build -- ---nightly audio_idle_suspend`. The magnitudes are one population on both arms — -every reported delta in 1,231,172-2,561,459 ns, the band this file already -records. - -The **rate** moves with where in a session an invocation falls, not with the -arm. Two A/Bs that held the within-round order fixed gave one arm the early -slots and produced 6 reds of 9 against 2 of 9; a third — twelve rounds with the -order alternating, so neither arm keeps the early slots — produced **4 of 12 -against 3 of 12**, no difference at all. In all three the reds cluster in the -first four invocations of the session whichever arm holds them, which is the -same quiet-host dependence as the `ALONE:` line above rather than a property of -any diff. - -Two consequences. A rate difference in this test at n<10 per arm is not evidence -about a diff, and an A/B of it must alternate the within-round order or it -measures session position instead. And a third failure mode belongs on the -record: one `main` invocation of the 42 failed differently — `expected soundd's -mix and control threads in sysinfo, found 1` — so this test also races soundd's -control thread into `sysinfo`, and a red carrying that sentence is not this -issue at all. - -## 2026-08-29: the instrument is in, and a 40-invocation session could not raise the red on either arm - -The decisive instrument exists now. `mix_thread` and `null_sink_thread` name -every wake that began with no client, found the device already stopped, and -ended with none — `soundd: idle wake (cmd=.. records=..)`, decided by -`toyos_mixer::wake_left_idle` and pinned by its truth-table test — so cause 1 -prints a line per wake and cause 2 cannot print one. The test now waits for -both soundd threads before its first sample, which removes the `found 1` race -above, and a red names each thread's delta apart, so the next red separates -mix loop from control thread from accounting on sight. - -What a red did not do today: 40 invocations in one dev-host session — 12 -sequential, 16 beside two other suites of this worktree, then 12 of the -*unmodified* test the same way (1-min load 5.1-21.6 across the loaded rounds) -— were green 40 of 40, on both arms. The same loaded rounds raised `hda_tone`'s -load-keyed reds on 3 boots of 11, so the load was real; this red's conditions -were not met, which is the same session-to-session movement the 08-21 section -records. The rate question stays open; the next red answers the cause question -by itself — with one stated exception: `wake_left_idle` exempts every wake -carrying a command byte, so an idle wake genuinely caused by the command pipe -(a byte stranded between the drain and the ring pop) prints no line and would -read as cause 2. A red with no idle-wake line therefore rules out every wake -source except the command pipe, not every source. diff --git a/issues/audio/null-sink-applies-one-connect.md b/issues/audio/null-sink-applies-one-connect.md index 88dac4b3a73..9d4ab0bb7b6 100644 --- a/issues/audio/null-sink-applies-one-connect.md +++ b/issues/audio/null-sink-applies-one-connect.md @@ -58,8 +58,3 @@ not fire; parked with an absurd one says the timeout was computed wrong; absent from every parked list says nothing ever held it, which would move this into #142's family rather than audio's. The report paints the panel, so the machine with no serial port answers on glass and a photograph is enough. - -Until that press happens, the three gates this task landed are what stands -between the milestone and a silent recurrence: a client through the null sink -must exit, a second must be taken up while the first streams, and the desktop -must still answer afterwards. diff --git a/issues/audio/stop-the-device-voice-keep-the-wake.md b/issues/audio/stop-the-device-voice-keep-the-wake.md index 0c0b296bba1..bae952f45f2 100644 --- a/issues/audio/stop-the-device-voice-keep-the-wake.md +++ b/issues/audio/stop-the-device-voice-keep-the-wake.md @@ -11,18 +11,3 @@ engine and the codec — the battery-relevant hardware — and gives up only the itself. Resume still works unchanged, because soundd keeps writing signal bytes, so it does not need the missing client→soundd message that `cpal-backend-hardcodes-the-format` is waiting on. - -**It is blocked on the audio gate, not on the fork and not on the owner.** A -mid-session device stop/restart is an audible transient plus a DLL re-lock, which -needs gate A's thorough tier on a quiet tree. That tier reds on the dev host -(`thorough-tier-reds-on-unmodified-main`), so the instrument comes first. - -2026-08-21: the sentence above used to read "that tier is itself red on `main`", -which was being sourced partly from the CI nightly — and the nightly's red was an -exit-code defect over a printed verdict, not a verdict. The dev-host red it now -names is the one that was ever measured, and the block stands on it. The same -entry records why the runner's PASSes do not lift it. - -That unblock condition is the useful part and the reason this is filed apart from -the fork-blocked cluster: it could land *first* if the quiet tree arrives before -fork access does. diff --git a/issues/audio/t14-wake-lateness-is-bimodal-per-boot.md b/issues/audio/t14-wake-lateness-is-bimodal-per-boot.md deleted file mode 100644 index 1a096837d67..00000000000 --- a/issues/audio/t14-wake-lateness-is-bimodal-per-boot.md +++ /dev/null @@ -1,188 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-08-21 ---- - -# On the T14 soundd's worst wake is bimodal per boot — ~4 ms or ~20 ms — and which mode a config lands in moves with the tree - -Measured on the self-hosted T14 (Intel i5-1135G7, 4c/8t, KVM, QEMU 11.1.0, CI -image) during the A/B that settled -`issues/audio/gate-a-has-no-runner-baseline.md`: four interleaved 15-iteration -gate A blocks, A,B,A,B, arm A `960b96e3`, arm B `53101d08`, idle machine, no CI -job container for any of the 240 boots, 1-min load 0.2-1.74. - -`max_wake_lat_us` does not vary continuously on this host. It takes one of two -values per boot: - -* a **fast mode** at 2544-4352 us (0.11-0.19 pipeline depths), and -* a **slow mode** at roughly 10000-25000 us (0.43-1.08 pl), - -with nothing much in between, and the mode is drawn per boot rather than -drifting through a block. `audio_tone.smp1`, arm B, in run order — the fast runs -are scattered, not clustered at either end: - -``` -b1 21389 18528 3945 4183 19994 20483 21544 33934 21591 21852 21139 22114 3871 16861 21481 -b2 4010 21562 4085 4028 4028 4212 14811 21692 17017 19528 22652 20878 18474 24184 11543 -``` - -Arm A, same config, has essentially no fast runs at all (one of 30, at 9880), -and arm B has eight of 30 — yet the two arms are **indistinguishable** on this -config overall (medians 20314 and 19994, z=1.49), because the slow mode -dominates both. - -Where the arms *do* differ, they differ by which mode dominates: - -| config | arm A, n=30 | arm B, n=30 | A vs B | -|---|---|---|---| -| `audio_tone.smp1` | 20314, one fast run | 19994, eight fast runs | z=1.49, same | -| `audio_tone.smp8` | 14069, mixed | 4088, **all 30 fast** (3939-4240) | z=5.37, B faster | -| `audio_tone_load.smp1` | 2764, **all 30 fast** (2544-3259) | 4352, mixed | z=4.12, B slower | -| `audio_tone_load.smp8` | 12676, mixed | 3904, **all 30 fast** (3643-4241) | z=5.48, B faster | - -## Why this matters more than the direction of any one row - -**The slow mode has no margin.** 20 ms is 0.86 of the 23219 us pipeline depth — -the point at which every buffer has drained and the device has run out of audio. -The recorded dev-host sample reaches 0.98 pl once in 120 runs and sits at 0.39 pl -in the median; on the T14 the *median* `audio_tone.smp1` boot, on both arms, is -where the dev host's worst run was. Nothing was audible in any of the 240 boots -— dropouts 0/120 and 0/120, underruns 0 in all 240 config-runs — but the -distance to harm on a slow-mode boot is one scheduling accident. - -**And a bimodal statistic is a bad thing to baseline.** The thorough tier's -Mann-Whitney is comparing mixtures, so its verdict tracks the mixing weight -rather than either mode. That is why no T14 sample should be recorded into -`tests/audio-baseline.toml` until the mode is understood: a re-record would -freeze one afternoon's mixing weight and red on any tree that moved it. - -## The one row that is a real same-host difference, and why it is not called a regression here - -`audio_tone_load.smp1` is worse on `main` than on `960b96e3` at z=4.12 pooled — -and it is the same config the never-read 2026-08-17 hosted nightly failed on -(`median 5765 -> 17684, z=4.61`, quoted in -`issues/audio/gate-a-suspend-structure-verdict-unread.md`). It is not called a -bisected regression because the block structure says it is not stable: the -per-block figures are z=3.57 (a1 vs b1) and z=2.09 (a2 vs b2), and arm B's own -two blocks differ by z=2.51 against arm A's z=0.35. Arm B is *unstable* on this -config; arm A is not. That is a change in the mixing weight, not a level shift, -and bisecting a mixing weight at n=15 per point would measure noise. - -The arrays behind every number above are in `0942f02c`'s commit message; the -gate's own logs expire with the workflow artifacts. - -## 2026-08-21, four hours later: the number is the *device's* lateness, and the slow mode did not come back - -`max_wake_lat_us` now arrives in two halves (`toyos_mixer::WorstWake`), split at -the completion interrupt's own ISR timestamp: `irq` is the device failing to -complete when the grid said it would, `pickup` is soundd failing to run once it -had. They sum to the old number exactly. `late_wakes` counts how many wakes in -the run were a whole period or more late, so the maximum can be read as one -stall or as a thousand. - -**296 config-runs on the T14 the same evening, 17:26-19:00 UTC, in the CI image -of the day, `--device=/dev/kvm --shard 1/1 --host-slots 0`, no other -container up for any boot** — each block samples `docker ps` every five seconds -and discards and retries itself whole if a CI job appears, which happened three -times and cost three blocks. Two trees, each carrying only the instrument: -`fe41dbae` (`main`, 51 runs per config) and `53101d08` (23 per config) — *the -A/B's own arm B*, the tree that produced the arrays above. - -| config | tree | n | wake_lat | irq mean | pickup mean / max | late wakes | -|---|---|---|---|---|---|---| -| `audio_tone.smp1` | main | 51 | 3875-4299 | 4005 | 76 / **113** | 12.9% | -| | `53101d08` | 23 | 3835-4197 | 3969 | 75 / **121** | 12.8% | -| `audio_tone.smp8` | main | 51 | 3977-6176 | 4025 | 139 / **206** | 14.2% | -| | `53101d08` | 23 | 3982-4341 | 3985 | 144 / **188** | 14.2% | -| `audio_tone_load.smp1` | main | 51 | 2509-2986 | 2732 | 10 / **14** | **0.0%** | -| | `53101d08` | 23 | 2542-2981 | 2737 | 10 / **12** | **0.0%** | -| `audio_tone_load.smp8` | main | 51 | 3692-4086 | 3812 | 64 / **142** | 11.1% | -| | `53101d08` | 23 | 3728-4331 | 3881 | 59 / **159** | 11.3% | - -Dropouts 0/296, underruns 0/296, drains 0/296, ceiling breaches 0/296. The two -15-iteration blocks that ran the thorough tier both reported **`[gate A] PASS` -— no statistic regressed at alpha=1e-3 per test**, on both trees, against the -same recorded sample the morning's first readable T14 run failed at -`audio_tone.smp1 median 8972 -> 17186 (z=4.36)`. Its fresh medians this evening -were 4075 (`53101d08`) and 4143 (`main`). - -Two things fall out of that table and a third out of its absence. - -**The statistic is not about the scheduler.** `pickup` never once exceeded -206 µs — 0.009 pipeline depths — on one CPU or on eight, and `irq` is 94-99.6% -of every worst wake. So *which CPU soundd lands on cannot be the mechanism*: -every interrupt lands on the boot CPU (`kernel/src/drivers/pci.rs`'s `MSG_ADDR`) -and a soundd sharing that CPU or not moves a term that is two orders of -magnitude too small to matter. The instrument is not blind to the other half — -the dev host under load 30 reported `pickup 8681us` on `audio_tone_load.smp8` -the same afternoon — the T14 simply never spends it. - -**The fast mode is a beat, not an event.** 12-14% of wakes are a whole period -late on the three idle configs, and the worst is ~1.4 periods with `2 empty -wakes` and `batch 2` on essentially every boot. That is soundd's 2.902 ms grid -against QEMU's `timer-period=5000` audio timer: soundd arms, wakes punctually -twice at a device that has produced nothing, and the batch lands ~4 ms after the -grid point. `audio_tone_load.smp1` — the config that is *always* fast — has -**zero** late wakes and `pickup 10 µs`, because a guest with work to do never -lets the beat open. - -**And the slow mode did not appear once in 74 boots per config.** Not on `main` -and not on the tree that produced it at 11 of 15 and 9 of 15 four hours earlier. -Under that afternoon's mixing weight, 0 of 74 has probability ~1e-35. So the -mode is **not a per-boot draw from a per-tree distribution**: the distribution -itself moved between two sessions of the same day, on the same host, with -nothing about the tree between them — which also means the A/B's one same-host -row (`audio_tone_load.smp1`, z=4.12) is a difference between afternoons and not -between trees. This evening the two trees are indistinguishable on it: 2509-2986 -against 2542-2981. - -The host was on AC throughout, `intel_pstate`/`powersave`/`balance_performance`, -`intel_idle` whose deepest state (`C3_ACPI`) costs 1048 µs to leave — a third of -one period, and a fortieth of the 20 ms mode. Nothing on the host was measured -*during* the earlier session, so what moved is not established; the one -difference recorded is that the slow session ran at 1-min load 0.2-1.74 and the -fast one at 1.1-4.4 — overlapping, and with the fast blocks' own quietest -stretches at 1.1-1.6, so load does not separate them either. - -## Whoever takes it next - -Do not spend the day on soundd. The next sighting of the slow mode is now one -line, and that line already answers three questions: whether it is the device or -soundd (`irq` vs `pickup`), whether it is one stall or a thousand -(`late_wakes`), and whether the guest was executing at all while it happened -(`empty` — a punctual soundd waking repeatedly at a silent device, versus a -single overlong sleep). Gate A's per-boot line also carries the two numbers this -boot drew for its clocks, which are the only per-boot draws that scale every -armed timer for the boot's whole life; on the T14 they are stable to 0.02% -(`tsc 2418-2419MHz`, `lapic 10000742-10002460 ticks/10ms` over 120 boots) and on -the dev host to 0.2%, so a slow-mode boot whose pair sits outside that is a -finding on sight and one whose pair sits inside it removes the whole class. - -What is missing is a *host-side* record taken during a slow session, because the -guest-side evidence above points there and cannot go further on its own. The -cheapest one is per-boot `/proc//task/*/schedstat` and the host's -`cpuidle` residencies sampled across a block — and it perturbs the measurement, -so it is worth taking only once a session is producing the mode. - -## Promoted 2026-08-25 - -The slow mode sits within one scheduling accident of harm (0.86 of a pipeline -depth, no margin) and its mixing weight is unexplained across two sessions on -the same host and tree. Owed to whoever next sees the slow mode, per this -entry's own "whoever takes it next" section: the host-side schedstat/cpuidle -capture during a slow session. - -## The exit is not the metal loop, and no lane produces the mode any more - -The T14 is no longer a CI runner, so nothing schedules a gate A run on it: the -slow mode can only be sighted by somebody running the gate there by hand, and -this record waits for that sighting rather than for a nightly. - -The metal loop (`issues/hardware/the-t14-boots-toyos-unattended.md`) does not -take the capture this record is owed. That capture reads a Linux host's -`schedstat` and `cpuidle` for the QEMU process the guest runs in, and the loop -boots ToyOS on the bare machine with no host under it and no QEMU process to -read. What the loop can answer is the adjacent question this record makes worth -asking — whether the two modes survive with no hypervisor at all — and that is -a job of the loop, not this exit. diff --git a/issues/audio/thorough-tier-reds-on-unmodified-main.md b/issues/audio/thorough-tier-reds-on-unmodified-main.md deleted file mode 100644 index 575001d5335..00000000000 --- a/issues/audio/thorough-tier-reds-on-unmodified-main.md +++ /dev/null @@ -1,187 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-08-07 ---- - -# Gate A's *thorough* tier reds on an unmodified `main`, and that is the rate the fast-tier intermittents asked for - -The fast tier's two-boot rule failed intermittently through early August — -dropouts on the first boot *and* the confirming re-boot, four times at smp=1 in -one 2026-08-04 session on two trees at once, twice more at smp=8 on 2026-08-07 — -and those sightings asked for the rate first, naming the thorough tier -(`--audio-gate N`) as the instrument. (Closed into this entry 2026-08-29; their -run tables are in that closing commit.) H3's session got the rate, and the -instrument reds on the tree it is supposed to certify. - -`cargo test --test toyos-build -- --audio-gate 30` on `80fe031` — **main's tip, -no delta at all**, run as H3's A arm before that branch existed: - -``` -[gate A] FAILED after 15 of 30 iterations (the remaining runs cannot change this): - pooled dropout rate: 10 of 120 vs recorded 0 of 120 (Fisher p=8.03e-4 <= 1e-3) -``` - -The ten, by config and iteration: `audio_tone_load smp=1` at 4, 9, 13, 15; -`audio_tone_load smp=8` at 9, 13, 14; `audio_tone smp=8` at 8, 9; -`audio_tone smp=1` at 13. So **`audio_tone` at both widths reds too**, which -the fast-tier sightings had established only for `audio_tone_load` — and at -both of its widths there, so no config is anyone's quiet control in an A/B. - -**The load correlation is the wrong way round, and that is the finding.** The -1-minute average across the run spanned 7.2 to 19.1 on 14 cores, with one to -five other guests and six other `toyos-build` processes throughout. The clean -early iterations ran at 19.1 and 16.8; the three worst — 13, 14 and 15 — ran at -11.4, 10.6 and 11.9. Every dropout carried a wake latency of 33-117 ms against -5-17 ms on the clean runs, which is the same "soundd was not scheduled" -signature as the fast-tier sightings and as the 2026-08-03 boot that put 142 ms -of silence on the wire — 49 underruns, one drain, a 5.6x worst-wake outlier, -gaps, soundd stats and capture all agreeing (also closed into this entry -2026-08-29). That boot's nearest suspect, the ESP-log flush on the kernel's -idle path, no longer exists — the idle loop touches no filesystem — and the -nearest *measured* mechanism on file is `issues/audio/disk-wait-pins-a-cpu.md`: -a staged 2 ms disk-completion delay alone produced 165-260 ms soundd wakes and -76 silent periods. - -What this changes for anyone reading them: the intermittency is not a property -of one config, and it is **large enough to fail the thorough tier's own pooled -test on a clean tree**. Anything that gates on this tier — the nightly run, and -H3 itself — cannot presently tell its own change from this. H3 therefore -compared its two arms against *each other* rather than against the recorded -sample, and said so. - -**Three arms, all of them red, and `main` red hardest.** Measured 2026-08-07 on -`main` at `c0365ea`, one session: - -| tree | dropout runs | measured runs | verdict | -|---|---|---|---| -| `main` | 7 | 28 | `pooled dropout rate: 7 of 40 vs recorded 0 of 120 (Fisher p=4.00e-5)` | -| `wt/toyos-m3` | 5 | 12 | `5 of 40 … (p=8.02e-4)` | -| the same branch with its one new wait deleted | 5 | 40 | `5 of 40 … (p=8.02e-4)` | - -The denominators differ because the gate stops as soon as the remaining runs -cannot change the verdict. The gate's own documentation says it cannot detect a -doubling of the dropout rate at any N a human waits for, so it cannot separate -these three from each other either — what it says unambiguously is that every one -of them is far from the recorded `0 of 120`. Every gap is small and none is a -silence anyone would hear as a break: the largest is 51 periods, most are one or -two, and the fast tier — whose verdict is harm — is green on all three arms, 7 of -7 each. So this is a *rate* finding against a recorded sample, not a report that -the machine sounds wrong. Host load was 6-20 throughout and is not offered as the -explanation; the three arms ran back to back on the same host, which is what makes -them comparable to each other. **Consequence while this is open:** the thorough -tier cannot serve as a pass/fail gate — an A/B between two arms is what it can -still answer. - -**The next step is named and is still nobody's: the thorough tier on the commit -the sample was recorded against, on the dev host, in one session.** That is the -one testable half of the two readings below. Do not re-record the baseline first: -a sample re-taken now would make the disagreement disappear without anyone -learning which of the two it was, and the recorded zero is the only reason the -question is visible at all. - -The recorded sample in `tests/audio-baseline.toml` is 0/120 and was taken in a -session this host no longer resembles. **Re-recording it is not licensed by this -entry** — a baseline widened to accept the defect is the defect made permanent. -What is needed is the cause. - -**The B arm was never obtainable, and the reason is `issues/kernel/`'s shootdown deadlock.** -Two attempts on the audio branch stopped at iterations 2 and 4, both on -`audio_tone.smp8`, both with the tier's "instrument broken" verdict — which is -what a guest whose kernel double-panicked looks like from here. Those commits -landed between the two arms and `--land` merged them in, so the arms differ by -more than the change under test and no comparison between them means anything. -What H3 has instead: a full suite green at 289/289 with all four audio configs -clean, and ten standalone runs of the audio family. None of that is a rate. - -## 2026-08-21: the CI nightly's red was never this, and its verdicts were never read - -**The dev-host finding above stands unchanged.** What follows is about a -different instrument — `gate-a.yml` on a GitHub runner — and it must not be -conflated with it. Nobody re-ran the dev-host arm; nothing here re-runs an audio -verdict away. - -`gate-a.yml`'s `gate` step ended in `exit "${PIPESTATUS[0]}"`. The runner logs -the shell it picked for that container on every step — `shell: sh -e {0}` — and -that shell has no `PIPESTATUS` array. Every run answered - -``` -/__w/_temp/.sh: 4: Bad substitution -##[error]Process completed with exit code 2. -``` - -on the line *after* the gate had printed its verdict. Without `pipefail` the -pipeline's status was `tee`'s 0, so `-e` never fired on the harness's own code; -the step's exit was dash's 2 for a failed expansion, and it was 2 whatever the -audio said. `gate-a.yml` has therefore **never once reported its verdict**: -every run it has ever had is a `failure`, including the ones that passed. - -The verdict each shard actually printed, read out of the job logs (artifacts -expire at 30 days; these lines do not): - -| run | date | shard 1 | shard 2 | -|---|---|---|---| -| 31386117376 | 08-10 (dispatch, `wt/toyos-ciwave2`) | FAILED `audio_tone.smp8` wake lateness median 6658 → 8496 (z=4.27) | FAILED `audio_tone_load.smp8` wake lateness median 7134 → 9520 (z=6.05) | -| 31771577360 | 08-14 | FAILED `audio_tone.smp8` wake lateness median 6658 → 8673 (z=5.41) | PASS | -| 31862912891 | 08-15 | PASS | PASS | -| 31925451196 | 08-16 | PASS | PASS | -| 31992902784 | 08-17 | FAILED at iteration 25, `audio_tone.smp8` instrument broken — suspend structure | FAILED `audio_tone_load.smp1` wake lateness median 5765 → 17684 (z=4.61) | -| 32097206141 | 08-18 | PASS | PASS | -| 32213928799 | 08-19 | PASS | PASS | -| 32330040225 | 08-20 | PASS | PASS | -| 32445243829 | 08-21 | PASS | PASS | - -Thirteen shard-runs PASS, five FAILED, eighteen exits of 2. - -**Two consequences, and they point opposite ways.** - -The thirteen PASSes mean the standing sentence "the thorough tier reds on -`main`" was being sourced from a red that was not a verdict. It is not evidence -that the dev-host finding above is stale: `gate-a-has-no-runner-baseline` already -establishes that a runner arm compared against the dev host's sample is a -cross-instrument comparison, and since the 2026-08-15 re-record the runner's -numbers are one-sidedly *better* than the recorded ones (08-21 shard 1: -`wake_lat_us recorded 7052/8972/22744 fresh 4002/6034/9942`), which is a -comparison that cannot red. A PASS of that comparison certifies little. The -dev-host question is still open and still needs the dev host. - -The five FAILEDs are the harm. Two on 08-10 and one on 08-14 are the -cross-instrument shape `gate-a-has-no-runner-baseline` explains. **The two on -08-17 are not**, and nobody has looked at them: they sit between a PASS night and -a PASS night on the same baseline, and shard 1's is not a statistic at all but -the instrument refusing — -`no 'soundd: suspended' after the last client removal; no 'virtio-sound: stream 0 -stopped' after the last client removal — the device is still running with no -clients`. That is filed apart as `gate-a-suspend-structure-verdict-unread`. - -The exit code is fixed in `35383398^:.github/workflows/gate-a.yml` (`set -o pipefail`, the -idiom every other workflow in `.github/` already uses). Nothing about how a -verdict is *reached* changed. - -## 2026-09-03: on a quiet dev host the tier passes, on both arms of an A/B - -`cargo test --test toyos-build -- --audio-gate=10`, six blocks alternating the -IOMMU audio move and its base (0c805ecf), base first; the arms differ in -`iommu_platform=on` on `virtio-sound-pci` and the sound driver's -`DeviceSpace::create` plus `attach`. 1-minute load 1.3-5.1 across the 240 runs -(block medians 2.4, 2.7, 2.3, 1.8, 2.7, 3.2, base and moved alternating) and -`qemu 1 toyos-build 1` on every one: every block `PASS` against the -recorded sample, dropouts 0 of 120 on each arm, ceiling breaches 0 of 120, no -instrument-broken iteration. The finding above was taken at loads of 7-19 -beside other guests and stands as measured; this is the quiet-host reading it -said nobody had taken, on that tip rather than on the sample's commit, so the -question of the sample's commit is still open. - -## 2026-09-25: an A/B for #492, both arms dropping, main no less often - -`cargo test --test toyos-build -- --audio-gate 30 audio_tone_load`, three blocks -per arm, alternating #492's merge of `origin/main` 3ad278c0 with a checkout of -3ad278c0 itself, branch first; one session, 1-minute load 1.5-34.2. Branch: -dropouts 1/60, 0/60, 0/60, every block `PASS`. `main`: 0/60, 1/60, 1/60, every -block `FAILED` on `audio_tone_load.smp1` wake lateness alone (medians 6042, -6803, 6473 against the recorded 5765; the branch's were 6147, 6137, 6121). The -branch's dropout (1 period) has its lateness in the interrupt half (`irq 65153us -+ pickup 874us`); main's are 28 periods in the pickup half (`irq 2973us + pickup -27401us`) and 1 period in the interrupt half (`irq 56120us + pickup 265us`). The fast tier, -three runs each: the branch's first run dropped once at load 57 and did not -reproduce on its confirming boot; every other boot on both arms was clean. diff --git a/issues/build/a-machine-tests-row-with-no-dispatch-arm-is-found-only-by-booting-it.md b/issues/build/a-machine-tests-row-with-no-dispatch-arm-is-found-only-by-booting-it.md new file mode 100644 index 00000000000..e10ae2c0f7a --- /dev/null +++ b/issues/build/a-machine-tests-row-with-no-dispatch-arm-is-found-only-by-booting-it.md @@ -0,0 +1,31 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# A `MACHINE_TESTS` (or `SCREEN_TESTS`) row with no matching arm in `run_machine_test` / `run_screen_test` compiles, lists and clips clean + +`run_machine_test`'s and `run_screen_test`'s final arm is a catch-all — +`other => Err(format!("unknown input test {other}"))` — so a name registered +in `MACHINE_TESTS` or `SCREEN_TESTS` with no corresponding `match` arm is not a +compile error: it is a `String` that falls through to that arm the one time +something schedules it. `cargo test --test toyos-build -- --list`, `cargo run +-- --clippy` and CI `host` never schedule a machine test, so none of them call +it, and `65511a24` (a partial revert left by a merge's conflict resolution +restoring `i8042_health_cadence`'s registration alone) passed all three at +`6c8ff3f4`; the gap surfaced only when a nightly boot actually ran the name and +got "unknown input test i8042_health_cadence" back. + +## Exit condition + +A host check — run in `host` or `--list`, not only on a boot — that fails on +an orphan either way: a registration with no arm, or an arm with no +registration. `check_metal_registration`'s `metal_rows_are_registered` +already does this comparison for `METAL`/`METAL_ONLY` against +`MACHINE_TESTS`/`SCREEN_TESTS`; the same shape of check, run against every +`match` arm's own pattern literal instead of a second table, closes this. + +## Owner + +`tests/toyos.rs`, `run_machine_test` and `run_screen_test`. Nobody holds it. diff --git a/issues/build/ftruncate-flush-race-reds-intermittently-and-nothing-says-why.md b/issues/build/ftruncate-flush-race-reds-intermittently-and-nothing-says-why.md index 995fff63ed2..dbf7f0f3e37 100644 --- a/issues/build/ftruncate-flush-race-reds-intermittently-and-nothing-says-why.md +++ b/issues/build/ftruncate-flush-race-reds-intermittently-and-nothing-says-why.md @@ -63,3 +63,9 @@ deleted. The `ftruncate-flush-stall` actuator, `tests/common/volumes.rs`; held by the orchestrator. + +**Unrun since it was disabled**: PR #562 deleted the guest's 150 ms verdict, +whose panic and harness message the reds above quote. The guest races its ten +truncates and settles; the host still requires `HELD`, refuses `BROKEN` and +reads the settled size off the volume. That change has never run: the test's +first run back is also that change's. diff --git a/issues/build/harness-reads-after-a-flat-half-second-drain.md b/issues/build/harness-reads-after-a-flat-half-second-drain.md deleted file mode 100644 index b59b83d4235..00000000000 --- a/issues/build/harness-reads-after-a-flat-half-second-drain.md +++ /dev/null @@ -1,24 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-09-25 ---- - -# Harness verdicts read after a flat half-second drain - -Several guest tests drain the console for a fixed 500 ms and then judge what -arrived, or read a file QEMU finishes only at its exit — a wait on nothing, -which a slow host turns into a wrong verdict: `tests/common/audio.rs` -(`doom_sound_flood`'s wav read), `tests/common/hda.rs` (three sites), -`tests/common/iommu.rs` (two), `tests/common/pkg.rs`, `tests/common/usb.rs` -and `tests/common/volumes.rs` — every `drain_serial(Duration::from_millis(500))` -in `tests/common/`. - -`soundd_log_stall` reads its wav after `QemuInstance::await_exit`, which waits -on QEMU's console closing; that is the pattern. - -## Exit condition - -`rg 'drain_serial\(Duration::from_millis' tests/common` finds nothing: each -site waits on the event it needs — a line, or QEMU's exit — bounded by a -ceiling that fails loudly. diff --git a/issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md b/issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md deleted file mode 100644 index 299b2482cba..00000000000 --- a/issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md +++ /dev/null @@ -1,38 +0,0 @@ ---- -status: expected-red -kind: finding -opened: 2026-09-07 ---- - -# `latency_wake` reds on the dev host at a rate - -Measured on the dev host (14 cores, macOS, TCG), same session, six runs alone, -three on `metal-suite` + the boot-deadline branch and three on the same tree -stashed back to `metal-suite`: - -| arm | p99 | verdict | -|---|---|---| -| deadline | 1606 us | PASS | -| deadline | 1634 us | PASS | -| deadline | 1785 us | PASS | -| base | 1456 us | PASS | -| base | 4096 us (floor, 902 past the histogram) | FAIL | -| base | 1428 us | PASS | - -The red is on the **base**, and the arm that touches the timer interrupt entry -did not produce one. The failure mode is always the same: the p99 lands in the -histogram's last bucket, so `4096us` is a floor and the harness refuses it as a -measurement rather than reporting a number it does not have. - -A seventh sighting, in the whole-branch review's twelve-wide `cargo test` on -7294acef: `258 past the 4096us histogram`, and the harness's own isolated re-run -green. - -What is still owed is the other reading of the same evidence: whether -cyclictest's 4,096-bucket histogram is simply too low for a TCG guest on a -loaded laptop, which is a change to the instrument and not to the kernel. - -**Exit condition.** Re-enabled when the base's p99 is shown to genuinely -exceed 4096 us and that is fixed at the timer-interrupt entry the deadline arm -already touches. Owner: the boot-deadline work (`kernel/src/sched`'s -timer-interrupt entry); held by the orchestrator. diff --git a/issues/build/no-device-class-answers-for-a-block-device.md b/issues/build/no-device-class-answers-for-a-block-device.md index 9935dc83f32..083bdf45331 100644 --- a/issues/build/no-device-class-answers-for-a-block-device.md +++ b/issues/build/no-device-class-answers-for-a-block-device.md @@ -49,5 +49,4 @@ requiring only that they were named. The sibling gate `hda_two_live_refused` is **not** in this record: init claims `hda-audio` before it spawns soundd, and soundd reaches the null sink only where that endowment is missing, so its existing `must_say(NULL_SINK)` already requires -`try_claim(HdaAudio)` to have answered `Absent`. That transitivity is now stated -at the assertion in `tests/common/hda.rs`. +`try_claim(HdaAudio)` to have answered `Absent`. diff --git a/issues/build/quiesce-leaves-the-volume-whole-needs-its-flush-to-close-inside-the-stops-budget.md b/issues/build/quiesce-leaves-the-volume-whole-needs-its-flush-to-close-inside-the-stops-budget.md index a1a8d73fced..7fce59a45be 100644 --- a/issues/build/quiesce-leaves-the-volume-whole-needs-its-flush-to-close-inside-the-stops-budget.md +++ b/issues/build/quiesce-leaves-the-volume-whole-needs-its-flush-to-close-inside-the-stops-budget.md @@ -30,10 +30,7 @@ Attempt 9 was due at about 1944 ms (a 640 ms park after 1304). It closed at 2705 ms, two milliseconds after the stop's own deadline wake, which fired 27 ms late. A single stall of the whole guest from before 1944 ms to past 2676 ms explains both late wakes firing together. Nothing in the guest was -waiting on the other. Nothing here is PR #562's doing either: on that branch -the kernel paths this boot runs (`quiesce.rs`, `block.rs`, `fat32_adapter.rs`) -are the same as on `main`. The one change to `quiesce_fsync.rs` removes a -deadline the guest only reached on a hang. +waiting on the other. ## Exit condition diff --git a/issues/build/the-harness-carries-three-helpers-nothing-calls.md b/issues/build/the-harness-carries-three-helpers-nothing-calls.md index 7cf97d13bd4..5444c1d0acb 100644 --- a/issues/build/the-harness-carries-three-helpers-nothing-calls.md +++ b/issues/build/the-harness-carries-three-helpers-nothing-calls.md @@ -16,10 +16,6 @@ branch deleted: - `tests/common/qemu.rs`: the field `usb_images` and its method `usb_images`; - `tests/common/qemu.rs`: `QmpDevices::set_link`. -The same pass names `tests/common/audio.rs`'s `completions` and `clients` and -`tests/common/stats.rs`'s `fisher_reject_at`, which the audio branch -(`wt/toyos-notiming`) deletes with their modules. - **Exit**: the three deleted, and the module-wide `allow(dead_code)` replaced by nothing, so the next orphan is a warning the host gate denies. Owner: orchestrator. diff --git a/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md b/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md deleted file mode 100644 index b322bbd081d..00000000000 --- a/issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md +++ /dev/null @@ -1,43 +0,0 @@ ---- -status: expected-red -kind: tooling -opened: 2026-09-13 ---- - -# The pass-cost gate's CI sample has been exceeded twice in eight days - -`sched_check_build`'s cost half on CI judges a boot's scheduler-pass -distribution against a recorded sample: sixteen CI runs, thirty-two CPU-runs, -2026-08-17 to 18, with zero passes over 200 000 ns and a 90th percentile of -32 768 ns (`tests/common/passcost.rs`). The line is one bucket above that -sample's worst 90th percentile. - -CI has now crossed it twice on branches that touch no scheduler code, each -time green alone in the same job: - -- `ci` run 33973213660, `guest (8)`, 2026-09-05, on `pkg-install-file`: - 1956 passes, p90 < 262144 ns, 366 over the budget. -- `ci` run 34768854940, `guest (11)`, 2026-09-13, the merge queue's run for - PR #448 (`host-bridge-abi`, the sysroot half of the host-bridge windows; - it moves no scheduler code): 3919 passes, p50 < 131072 ns, - p90 < 262144 ns, p99 < 262144 ns, max 2487630 ns, 51 over the budget. - The red dequeued the pull request and disarmed its auto-merge. - -Two sightings of a whole-distribution shift on a population the sample said -had none is not the sample returning; it is the population moving. Either -CI's runners changed under the sample (the August runs were one Azure SKU; -nothing records which SKU each shard lands on), or the tree's pass cost -moved in a way the dev host cannot see. The gate cannot tell, so every -further crossing costs a landing an hour and answers nothing. - -**Exit condition.** The sample is re-recorded from the CI runs since -2026-09-01 — every `guest` shard that ran this name, with the runner's -reported CPU model beside each reading — and the line re-derived from it -with the same rule; if the readings split by runner model, the gate names -the model it judges and refuses to judge the rest. Owner: orchestrator. - -A second signature: the fast -tier on PR #524's branch at `235c5a5b` reds `sched_check_build` on `cpu0: 85 -passes … a 90th percentile needs at least 100 samples behind it and this has -85`. The host was loaded then by another worktree's spinner at 397% CPU. The -harness's re-run alone was green. diff --git a/issues/build/there-is-no-attributed-session-ledger.md b/issues/build/there-is-no-attributed-session-ledger.md index 4c5fcab23e4..bead95e959c 100644 --- a/issues/build/there-is-no-attributed-session-ledger.md +++ b/issues/build/there-is-no-attributed-session-ledger.md @@ -4,17 +4,13 @@ kind: track opened: 2026-09-01 --- -# There is no attributed session ledger, so seven flake records cannot name what overlapped them +# There is no attributed session ledger, so flake records cannot name what overlapped them -Seven open records describe a test that reds beside other work and is green +Open records describe a test that reds beside other work and is green alone, or a cost charged to the wrong artifact. Every one of them is blocked on the same missing observation: **no record joins a guest's loss of progress to -the interval that overlapped it.** They are not seven mechanisms. They are one -instrument, seven times. +the interval that overlapped it.** They are one instrument. -- `issues/audio/gate-a-has-no-runner-baseline.md` -- `issues/audio/thorough-tier-reds-on-unmodified-main.md` -- `issues/audio/idle-suspend-reds-on-a-loaded-host-and-on-main.md` - `issues/boot-media/kernel-log-file-reds-beside-other-guests-and-is-green-alone.md` - `issues/boot-media/usb-short-read-reds-beside-other-guests-and-is-green-alone.md` - `issues/build/parallel-tests-red-under-other-suites.md` @@ -32,16 +28,11 @@ image-build spans with their content key and cache hit or miss, QEMU and vCPU scheduling intervals where the host exposes them, and the guest progress markers the tests already emit. A sighting is then a join, not an inference. -**What exists, and why each is not the thing.** `tests/common/hostload.rs` (180 -lines) records load averages and process counts and is attached to an audio run -— a *sample*, not an interval, so it cannot say what overlapped a window. -`src/buildlock.rs` already names holders (`records_holder`) -but the record is transient: it exists while the guard is held and -is gone when the question is asked. The audio baseline at `tests/toyos.rs:2107` -is keyed on `(test, smp)` and nothing else, so it cannot distinguish a CI -runner's distribution from the developer's — that key has no runner provenance -at all. The committed shard input is per-test duration only, which is -why `issues/build/the-shard-split-prices-a-boot-and-not-the-image-behind-it.md` +**What exists, and why each is not the thing.** `src/buildlock.rs` already +names holders (`records_holder`) but the record is transient: it exists while +the guard is held and is gone when the question is asked. The committed shard +input is per-test duration only, which is why +`issues/build/the-shard-split-prices-a-boot-and-not-the-image-behind-it.md` charges an image build to whichever test followed it. **The probe must prove itself inert.** Collection writes files and samples the diff --git a/issues/build/there-is-no-network-gate.md b/issues/build/there-is-no-network-gate.md index edaef91aaef..50231c703b1 100644 --- a/issues/build/there-is-no-network-gate.md +++ b/issues/build/there-is-no-network-gate.md @@ -11,9 +11,6 @@ idle→packet wake fix shipped with no regression coverage at all. The gate was scheduled after the first bare-metal attempt by the owner on 2026-07-31; that attempt has since happened many times. -**It mirrors gate A deliberately**, because four things about gate A were worth -copying and one capability the audio gate never had is available here. - - **Device-side ground truth.** QEMU's `-object filter-dump` writes a pcap of the virtual wire; the harness parses it offline. Byte-exact payloads, checksum and length validity of every frame ToyOS emits, ARP sanity, and @@ -28,8 +25,7 @@ copying and one capability the audio gate never had is available here. the analyser against known-good and known-bad captures before trusting it. **The new capability is `-netdev socket`: the harness owns the other end of the -Ethernet link at frame level.** That makes three things possible that gate A -could not do — adversarial frames (truncated headers, wrong length fields, giant +Ethernet link at frame level.** That makes three things possible — adversarial frames (truncated headers, wrong length fields, giant and zero-length, garbage: the kernel must return errors and never panic, netd may drop but must not wedge), deterministic impairment (seeded loss, reorder, delay, duplication, so smoltcp's retransmission becomes reproducible rather than diff --git a/issues/build/timing-verdicts-ruled-off-qemu-have-no-metal-arm.md b/issues/build/timing-verdicts-ruled-off-qemu-have-no-metal-arm.md new file mode 100644 index 00000000000..41e7c2d7def --- /dev/null +++ b/issues/build/timing-verdicts-ruled-off-qemu-have-no-metal-arm.md @@ -0,0 +1,91 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# Timing properties QEMU no longer judges, and the clocks it still waits on + +A QEMU test measures no time; the harness's hang ceiling is its only clock +(owner ruling). The properties below were deleted from the QEMU suite with +their clocks, and no `METAL` row judges them. + +**Owed a metal row:** + +- **Audio.** soundd's `underruns` stay 0 under two CPU burners + (`audio_tone_load`). `desktop_audio_client`'s three verdicts: a desktop client + through the null sink exits, a second is taken up while the first streams, + and the desktop still answers afterwards. A null sink drains a client at the + audio rate (`metal_sim_null_audio`). soundd submits at least a period for + every period's worth of frames a client handed it + (`inspect_reads_its_owners`: `sound.periods.submitted × sound.period_frames` + against what `inspect_plays` proved taken). +- **Boot.** The loader's and the kernel's TSC account of a boot agrees with the + host's clock over the same boot (`boot_from_power_on`). +- **Scheduling.** Every CPU reaches a scheduler pass each heartbeat period, and + no window between two heartbeats hides a death (`kernel_heartbeat`). A + scheduler pass's cost distribution (`sched_check_build`; no metal arm boots + the check kernel). A stop is woken by its last park before its budget, + rather than giving up on it (`quiesce_wakes_on_the_last_park`). +- **Wakes delivered rather than waited out**: 500 pipe round trips and the + 24-child, 24-thread exit storm inside 3 s, every armed ring watcher woken + inside 200 ms (`blocking_read_stress`, `exit_wait_storm`, `poll_wake_pipe`). + These ride the shared metal boot without their bounds. A kept, closed window + takes no wakes while its application idles (`toolkit_winit_loop`'s stage 6). +- **Accounting.** A refused syscall is charged to its caller's CPU time + (`process_stats`). A client out-writing a stalled peer costs netd under half + a CPU (`netd_stalled_peer`). blockd's bench throughput (`blockd_io`). +- **Input.** A boot with no i8042 is no slower than one with it + (`i8042_absent`). The i8042 counters repeat at most once per 10 s and only + when the pin asserted (`i8042_health_cadence`), and the idle loop does not + spin on the controller (`i8042_health`'s idle trips). The fatal path's panel + holds while a key is held (`panic_key_holds`). +- **USB.** The connect settle ends on the device appearing and not at + `EMPTY_BUS_NS` (`xhci_slow_connect`). A disk call ends inside + `toyos_xhci::call::AFTER_BREAK`, a staged break skips its data-phase wait, the + offline ladder ends inside 2.75 s, and a port reset that never verified spent + its rung (`tests/common/usb.rs`). A controller that will not halt and a port + that will not reset are refused only after the 2 s budget was waited out + (`xhci_deaf_registers`). +- **Network.** netd's connection cap counts connects still pending alongside + established ones (`netd_connection_caps`, which now fills the cap with + established connections only). A burst of silent connections is bounded: + netd keeps some and not all (`netd_hostile_peer`). +- **Logging.** init's stop waited for logd's flush answer + (`log_ring_keeps_the_owners_slots`; only init's own word is read now). logd + lets a network reader that stopped reading go within its `STALLED` + (`log_stream_stalled_reader`). logd writes `/log` promptly while the machine + runs (`kernel_log_file`, and `tests/common/usb.rs`'s stick with no write + cache). +- **Panel and storage.** A dump survives the compositor's next repaint + (`screen_blocked_dump`). A same-length overwrite of a pinned `/home` file + reads back whole through its displaced file's teardown: one read is taken + now, and it may come before the teardown (`home_overwrite_reads_back`). + +**Premises: a QEMU test still waits on a clock that decides no verdict**, where +the event it stands for has no word a test can read: + +- Paces that give a stimulus time to land: the input pacing of + `metal_sim_window_drag` and of the i8042 keyboard arms, + `metal_sim_pointer_churn`'s `SETTLE`, the xHCI hotplug and flap sleeps in + `tests/common/usb.rs`, `sched_stress`'s connect-storm pace, `locale_gate`'s + poll, the compositor and IPC probes' paces, `ftruncate_flush_race`'s + `INTO_WINDOW`, and `redirty_mid_flush`'s swept delay. +- Drains of work handed to another thread: the 200 ms `iod` drains in + `writeback_durability` and `home_backing_revoked`, `fat_backing_revoked`'s + drain, and `log_gate`'s `STORM_SETTLE` and `QUIET_READS`. +- Product clocks a QEMU boot races: `boot_deadline_ends_a_wedge`'s 15 s + deadline against the job list reaching its shutdown, + `hard_lockup_ends_a_deaf_cpu`'s 30 s, `loader_watchdog_arms` reading + `timeout=0` before the TCO expires, and the stop's 2010 ms budget (a stop + that gives up is not judged in QEMU). + +**A clock that decides a verdict**: `census_wait::settled` takes the census +once two readings 10 ms apart agree. On a starved host the first can be read +mid-release, so a leak check anchored on it (`handle_kill_policy`, +`handle_lifetime`, `shm_release_reclaims`) can pass wrongly. + +**Exit**: each owed line is judged by a `METAL` row or ruled not owed by the +owner, each premise waits on the event it stands for or is ruled acceptable, +and the census is read on the release it stands for. Owner: the metal suite +(`tests/toyos.rs`'s `METAL`); held by the orchestrator. diff --git a/issues/build/wake-storm-cost-red-under-induced-host-load.md b/issues/build/wake-storm-cost-red-under-induced-host-load.md deleted file mode 100644 index 5b21b250011..00000000000 --- a/issues/build/wake-storm-cost-red-under-induced-host-load.md +++ /dev/null @@ -1,63 +0,0 @@ ---- -status: open -kind: finding -opened: 2026-08-26 ---- - -# `wake_storm_cost` reds beside other guests on a deliberately loaded dev host - -One sighting, dev host, 2026-08-26, in a full 12-wide `cargo test` run with seven -pure-shell spin loops as company on a 14-core machine (load average 53). The -loop was staged to measure something else — a denominator for typed-input loss — -so the load is the instrument here rather than the tree's usual condition. - -``` -FAIL rs::wake_storm_cost: exit code 101 -wake_storm_cost: 16 waiters, 21000 cycles -wake_storm_cost: 64 waiters, 115000 cycles -quadrupling the storm from 16 waiters to 64 took the waker's own cost from -21000 cycles to 115000. `post_n` walks the waiters once and does a constant -amount per claim, so four times the waiters may cost four times the loop and -no more; past this, something in the claim grows with the size of the storm - FAIL wake_storm_cost (585ms) - ALONE wake_storm_cost: GREEN -``` - -`cargo run -- --known-red wake_storm_cost` answers **NOT ON THE LIST**, which is -why this file exists: the next reader of this name gets a sighting instead of -nothing. - -## Second sighting, hosted shard, 2026-08-26 — the other instrument - -Run 32909059602 `guest (1)`, the small-fix batch's pull request (a diff of -tracker closes and unrelated small fixes, none reaching the scheduler or -`post_n`): the same ratio assertion red, 183 of 184 names green beside it. A -hosted shard is four cores running one eight-vCPU guest, so it is loaded by -construction — which means both recorded firings are on oversubscribed hosts -and none on a quiet one. That is half the denominator the section below asks -for; the quiet-host arm is still the missing half. - -## What this is and is not evidence about - -The assertion is a *ratio* of guest TSC deltas, so a host that deschedules the -vCPU inside the 64-waiter arm and not inside the 16-waiter one moves the ratio -without anything in the kernel changing. That makes host oversubscription a -live alternative to a claim cost that grows with the storm, and this sighting -cannot separate them: it is one observation at one load. - -The tree it ran on touches `tests/` only — the harness's typed-input delivery — -and nothing in that diff reaches the scheduler, the wait queues or `post_n`. - -## What would settle it - -Either arm measured against the other in the same session: the same ratio taken -on a quiet host and on a loaded one, several runs of each, with the host's load -recorded per run. If the ratio moves with the load, the assertion needs a -denominator that host time cannot inflate — a count of claims rather than a span -of cycles. If it does not, the finding is the kernel's and this is a `defect`. - -## Third sighting, main's nightly at 1ce71831 - -Run 36290616312, one guest shard: `FAIL rs::wake_storm_cost: exit code 101` -wide, then `ALONE wake_storm_cost: GREEN` twice. This is again a hosted -four-core shard running eight-vCPU guests, so an oversubscribed host. diff --git a/issues/diagnostics/no-cyclictest.md b/issues/diagnostics/no-cyclictest.md index 1cc9432eb77..c8ee0a7250b 100644 --- a/issues/diagnostics/no-cyclictest.md +++ b/issues/diagnostics/no-cyclictest.md @@ -26,9 +26,7 @@ soundd and measuring a different machine. Number 96 is retired, and **What exists is not a substitute, and each instrument fails differently.** soundd's `max_wake_lat_ns` (`toyos-mixer/src/stats.rs`, printed by -`userland/soundd/src/mix.rs`, read by gate A in `tests/common/audio.rs` and -baselined in `tests/audio-baseline.toml`; the thorough tier runs Mann-Whitney on -`max_wake_lat_us`) is a **max over a ~2 s window, not a distribution** — no +`userland/soundd/src/mix.rs`) is a **max over a ~2 s window, not a distribution** — no percentiles, no sample count; it measures against a DLL's *prediction of a DMA completion* rather than against a programmed timer, so it folds in the device model; and it needs soundd plus a sound card to exist at all, which is exactly diff --git a/issues/diagnostics/the-deaf-cpu-actuator-arms-on-three-seconds-of-guest-clock.md b/issues/diagnostics/the-deaf-cpu-actuator-arms-on-three-seconds-of-guest-clock.md index bd1d9357747..b21623dcf70 100644 --- a/issues/diagnostics/the-deaf-cpu-actuator-arms-on-three-seconds-of-guest-clock.md +++ b/issues/diagnostics/the-deaf-cpu-actuator-arms-on-three-seconds-of-guest-clock.md @@ -7,18 +7,14 @@ opened: 2026-09-24 # The deaf-CPU actuator arms on three seconds of guest clock `kernel/src/sched/dump.rs`'s `deaf_window` (`dump-deaf-cpu`, driven by the -nightly `dump_nmi_probe`) does nothing until `nanos_since_boot()` reaches its +`dump_nmi_probe`) does nothing until `nanos_since_boot()` reaches its `ARM_AT_NS`, 3 s, "late enough that the machine is up and every CPU has joined". That is a sleep standing in for an event: a boot slower than 3 s puts -the deafening inside it, and a fast one idles until the clock passes. The -harness (`tests/common/faults.rs`'s `dump_nmi_probe`) prices its 20 s drain on -the same number. +the deafening inside it, and a fast one idles until the clock passes. The same shape in `dump-in-blocking-pass` was a review BLOCKER on the branch that filed this and was replaced by arming on `smp::is_ready()`, the SMP release that is the machine's own word that every CPU has joined. This one was off that branch's path. -**Exit:** `deaf_window` arms on an event, the harness reads the boot log for -whatever it prints first, and `dump_nmi_probe` passes on a boot slowed past -3 s. +**Exit:** `deaf_window` arms on an event. diff --git a/issues/hardware/eleven-names-red-on-ci.md b/issues/hardware/eleven-names-red-on-ci.md index 77b5456abb6..b3ee56ca5e8 100644 --- a/issues/hardware/eleven-names-red-on-ci.md +++ b/issues/hardware/eleven-names-red-on-ci.md @@ -59,13 +59,3 @@ none of them**: the retry loop was written for the parallel phase and branched o the *run's* width. Half the list had no second sample at all, which is most of why the earlier lists looked like they rotated. -**Twelve names came off the list when the wall-clock work landed**, and the -previous write-up's samples predate it. Run `31252989653` was `ab7f5d6`, which -does not contain `wt/toyos-clock` (`5b6e192`, and `1cf7fee`, `c546335`, -`02a3bc9`, `d50a8c9` under it). `metal_sim_client_death`, `metal_sim_window_drag`, -`metal_sim_pointer_churn`, `metal_sim_compositor_stall`, `desktop_audio_client`, -`desktop_typing_damage`, `doom_sound_flood`, `i8042_health_cadence`, -`sshd_fail_closed`, `xhci_hotplug`, `xhci_hid_break` and `screen_pager_keys` are -all **0 of 5** now, and the "a guest stops making progress and pays its whole -ceiling" shape with them. Anything read off a run older than `31254054628` is -about a different tree. diff --git a/issues/hardware/i8042-health-cadence-counted-three-lines-once-on-mains-nightly.md b/issues/hardware/i8042-health-cadence-counted-three-lines-once-on-mains-nightly.md deleted file mode 100644 index 852c68d8865..00000000000 --- a/issues/hardware/i8042-health-cadence-counted-three-lines-once-on-mains-nightly.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -status: open -kind: finding -opened: 2026-09-27 ---- - -# `i8042_health_cadence` counted three counter lines for two keystrokes once on main's nightly - -Main's nightly at 1ce71831 (run 36290616312), one guest shard, wide: -`FAIL i8042_health_cadence: two keystrokes three seconds apart, 3 counter -lines — the report is on a timer rather than on the pin`. `ALONE -i8042_health_cadence: GREEN` twice. -`cargo run -- --known-red i8042_health_cadence` answers NO. - -**Exit**: the red run's three counter lines matched to the edges that -produced them, or a rate with enough runs to call it gone. diff --git a/issues/hardware/t14-lost-every-integrated-input.md b/issues/hardware/t14-lost-every-integrated-input.md index 728a691a7a7..36e1fe61034 100644 --- a/issues/hardware/t14-lost-every-integrated-input.md +++ b/issues/hardware/t14-lost-every-integrated-input.md @@ -52,9 +52,7 @@ The cadence fix is what makes the next session decisive rather than a guess: after the verdict the counters repeat, at most once per 10 s and **only when the pin has asserted since the last line**. That gating is the point — past the first repeat, no line means no interrupt, so silence becomes evidence instead of -absence of evidence. `i8042_health_cadence` gates it, and reverting either half -(fire on the timer, or make `HEALTH_DONE` terminal again) reds it at 9 lines and -0 lines respectively against the required 2. +absence of evidence. **What the next boot should capture.** A repeat line dated after 6.6 s, or none. If bytes are arriving, `undecoded`/`discarded` name the fault in this driver. If diff --git a/issues/hardware/the-t14-boots-toyos-unattended.md b/issues/hardware/the-t14-boots-toyos-unattended.md index 4cb50310dfe..7e5de030809 100644 --- a/issues/hardware/the-t14-boots-toyos-unattended.md +++ b/issues/hardware/the-t14-boots-toyos-unattended.md @@ -112,9 +112,7 @@ What is left to build: `issues/kernel/the-split-window-tlb-cost-is-unpriced.md`, `issues/kernel/ap-control-registers-inherit-init.md`, `issues/kernel/ap-tsc-trail-is-assumed-and-never-checked.md`, - `issues/audio/hda-ring-fix-unverified-on-metal.md`, - `issues/audio/t14-wake-lateness-is-bimodal-per-boot.md`, - `issues/audio/gate-a-has-no-runner-baseline.md` (a metal sample), and the + `issues/audio/hda-ring-fix-unverified-on-metal.md`, and the IOMMU track's three hardware-only answers — isolation scopes and reserved regions, the 2× cost bar, and the compatibility-format question in `issues/kernel/qemu-passes-compatibility-format-interrupts.md` diff --git a/issues/kernel/a-held-disk-waits-for-a-pass-no-cpu-takes-when-every-cpu-is-in-a-call-on-it.md b/issues/kernel/a-held-disk-waits-for-a-pass-no-cpu-takes-when-every-cpu-is-in-a-call-on-it.md index 4cd36d5b146..0bfb676062c 100644 --- a/issues/kernel/a-held-disk-waits-for-a-pass-no-cpu-takes-when-every-cpu-is-in-a-call-on-it.md +++ b/issues/kernel/a-held-disk-waits-for-a-pass-no-cpu-takes-when-every-cpu-is-in-a-call-on-it.md @@ -67,3 +67,8 @@ Related records — `usb_transport_break`'s three other open red modes, not this one: `issues/kernel/a-shutdown-on-a-held-usb-disk-left-a-cpu-deaf-to-a-tlb-shootdown.md`, `issues/build/usb-transport-break-flushedstick-can-break-after-the-reboot.md`, and `issues/boot-media/a-disk-whose-port-went-away-panics-the-boot-at-roots-hold.md`. + +**Unrun since it was disabled**: PR #562 deleted this test's timing check, +that the staged break's record came less than 2 s of kernel clock after the +record before it. That change has never run: the test's first run back is +also that change's. diff --git a/issues/kernel/a-log-rings-owner-is-named-only-when-logd-reads-its-registration.md b/issues/kernel/a-log-rings-owner-is-named-only-when-logd-reads-its-registration.md index c759e91b593..ff1629b7111 100644 --- a/issues/kernel/a-log-rings-owner-is-named-only-when-logd-reads-its-registration.md +++ b/issues/kernel/a-log-rings-owner-is-named-only-when-logd-reads-its-registration.md @@ -44,3 +44,10 @@ No test today covers that owner decision in `Ring::push` `ring.own(pid)`, passes every test in the tree, so the exit's test must turn both mutations red. Owner: `toyos/src/log/region.rs`'s `Ring::push`; held by the orchestrator. + +**Unrun since it was disabled**: PR #562 changed this test's verdict on the +flush. An answered flush is no longer inferred from the kernel's sync starting +a flush bound after init's stop line; the flush is unanswered when init says +it waited one out (`FLUSH_WAITED_OUT`, on the console or in `/log`) or its stop +line never reached `/log`. That change has never run: the test's first run +back is also that change's. diff --git a/issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md b/issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md index a9d1a0de752..2da298ed258 100644 --- a/issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md +++ b/issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md @@ -7,10 +7,7 @@ opened: 2026-09-25 # A `quiesce_writers` writer's first write-and-fsync pass outlasts the job's 5 s spin-up It asks for the reset only once each of its six writers has -finished one pass: a create, 64 KiB of writes and an fsync. If a writer is -still in its first pass after 5 s, the job prints `quiesce_writers: of 6 -writers reached their loop in 5s` and exits 1 without asking, so no stop -begins. Every sighting below then reads `QEMU never reported stopping: the +finished one pass: a create, 64 KiB of writes and an fsync. Every sighting below then reads `QEMU never reported stopping: the guest asked for a reboot and stayed up`; that misreport is `issues/build/a-stopped-boot-whose-job-never-asked-waits-out-the-reset-budget-and-says-it-asked.md`. @@ -71,3 +68,9 @@ first write-and-fsync pass for over 5 s while another writer passes in under a third of a second, and the fix is shown against it. Owner: the `/log` write and sync path `tests/toyos-rust-tests/src/bin/quiesce_writers.rs` drives; held by the orchestrator. + +**Unrun since it was disabled**: PR #562 deleted the job's 5 s spin-up. +`quiesce_writers` waits for every writer's first pass with no deadline, so a +first pass as slow as those above delays the reset and no longer ends the boot +unasked. That change has never run: the test's first run back is also that +change's. diff --git a/issues/kernel/a-root-metadata-read-refused-on-budget-is-not-retried.md b/issues/kernel/a-root-metadata-read-refused-on-budget-is-not-retried.md index 38bb78398ee..75609cc6c51 100644 --- a/issues/kernel/a-root-metadata-read-refused-on-budget-is-not-retried.md +++ b/issues/kernel/a-root-metadata-read-refused-on-budget-is-not-retried.md @@ -23,9 +23,7 @@ twelve guests sharing one host spent the budget waiting for the xHCI controller lock and demand-paged executables faulted at `_start+0x0`. **Reproduction.** Not reached in a suite yet: the data path was, and this one -shares its mechanism and its device. `usb-slow-device` holds every mass-storage -completion back, and a boot armed with it that also spawns under load is the -shape to try. +shares its mechanism and its device. **Exit condition.** A refused metadata read is retried on a fresh budget above the cache lock, bounded by `block::DEADMAN`, or the mount reports it as the diff --git a/issues/kernel/a-stick-that-answers-late-is-broken-by-the-two-second-abandon.md b/issues/kernel/a-stick-that-answers-late-is-broken-by-the-two-second-abandon.md index 793d3453b20..8d9b97b12aa 100644 --- a/issues/kernel/a-stick-that-answers-late-is-broken-by-the-two-second-abandon.md +++ b/issues/kernel/a-stick-that-answers-late-is-broken-by-the-two-second-abandon.md @@ -95,6 +95,4 @@ recording a mass-storage completion nobody is waiting on. ## Exit condition A T14 boot in which a CSW arrives more than two seconds after its CBW and the -command completes on the caller's retry with no `transport broke` record, or a -QEMU boot with `usb-slow-device` widened past the operation budget doing the -same. +command completes on the caller's retry with no `transport broke` record. diff --git a/issues/kernel/logging-records-from-every-producer-and-a-kernel-that-waits-on-nobody.md b/issues/kernel/logging-records-from-every-producer-and-a-kernel-that-waits-on-nobody.md index a65294736f8..95296355cf6 100644 --- a/issues/kernel/logging-records-from-every-producer-and-a-kernel-that-waits-on-nobody.md +++ b/issues/kernel/logging-records-from-every-producer-and-a-kernel-that-waits-on-nobody.md @@ -58,8 +58,8 @@ audio-latency A/B, neither of which is taken today. metadata on an append, but the zeros are a burst of writes to the stick, and that burst starves a tone playing beside it (`issues/audio/a-megabyte-written-to-the-stick-starves-a-tone-beside-it.md`). -*Exit:* that issue closed, and parts preallocated with `audio_tone_load` -green at eight CPUs. +*Exit:* that issue closed, and on the T14 a `METAL` row plays a tone at +`underruns=0` while parts are preallocated beside it. Constraints a reader would otherwise re-derive: diff --git a/issues/kernel/netd-never-reaches-its-loop-under-iommu-userdev-foreign-dma.md b/issues/kernel/netd-never-reaches-its-loop-under-iommu-userdev-foreign-dma.md deleted file mode 100644 index b99cc89a644..00000000000 --- a/issues/kernel/netd-never-reaches-its-loop-under-iommu-userdev-foreign-dma.md +++ /dev/null @@ -1,24 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-25 ---- - -# netd never reaches its loop on the `iommu-userdev-foreign-dma` boot, and no userland line reaches that boot's console - -`tests/common/iommu.rs`'s `USERDEV_FOREIGN` arm boots `tests/netcase` with -netd's first grant answered with an NVMe address. With a kernel that logged -every `pcidev::take_record` call (a throwaway diagnostic on `wt/toyos-logd`), -netd made **no** interrupt read in the 20 s after the fault — its loop's first -act — and the boot's console carried no userland line at all, before the fault -or after it: no `init: started`, no `netd: VirtIO: PCI`, nothing from logd. -The kernel's own lines ran on, and no `exit:` for netd was ever printed. - -So netd is parked somewhere between its grant and its loop on that boot, and -`userdev_dma_fault` passes without asking where: its verdict is the machine, -not the driver. Where netd is parked, and why this config's userland output -never reaches the console, are unmeasured. - -**Exit condition**: the place netd waits on that boot named — a blocked-task -dump or a line of its own — and either the wait removed or `userdev_dma_fault` -asserting what netd does after the fault. diff --git a/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md b/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md index 4261892d060..db35aca27b8 100644 --- a/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md +++ b/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md @@ -47,8 +47,6 @@ records cannot exclude, and it sits on the same `/log` fsync path the writers issue owns. **What no enabled guest test checks while this is disabled.** -`woken_by_its_threads` (`tests/common/power.rs`) has no enabled caller. So no -enabled test checks any of these: - that a band, a park or an exit wakes the stop, rather than its deadline; - `in_flight == 0` with `begun > 0`; @@ -58,9 +56,7 @@ enabled test checks any of these: `quiesce_wakes_on_the_last_teardown`'s only claim, and the only enabled guest check of it, disabled by this same issue. -`quiesce_refuses_a_second_shutdown` stays green over a lost post. It judges the -stop only by `stopped_the_machine`, so a stop that spends its budget and then -finds everything stopped passes it. +`quiesce_refuses_a_second_shutdown` stays green over a lost post. **Exit**: @@ -73,3 +69,12 @@ finds everything stopped passes it. - an enabled guest test checking each claim listed above. Owner: the stop path, `kernel/src/quiesce.rs`; held by the orchestrator. + +**Unrun since it was disabled**: PR #562 deleted the stop's completion from +every QEMU verdict: `stopped_boot`'s `stopped_the_machine` check, whose message +the sightings above quote, and `woken_by_its_threads`. The stop gives up at a +budget of the kernel's own clock, so on metal alone +(`metal::Readback::stop_completed`) does a stop that gave up red. This test now +judges the held thread before the sync and more than one sweep, and the failure +quoted above no longer reds it. That change has never run: the test's first run +back is also that change's. diff --git a/issues/kernel/scheduler-pass-blocks-in-xhci.md b/issues/kernel/scheduler-pass-blocks-in-xhci.md index 9478a91fdd2..8fc7d5c0f3b 100644 --- a/issues/kernel/scheduler-pass-blocks-in-xhci.md +++ b/issues/kernel/scheduler-pass-blocks-in-xhci.md @@ -52,13 +52,3 @@ prologue is outside the window the budget covers, so the pass-cost histogram records a microsecond pass while the CPU had been in the driver for two seconds. **The measured window has to start where the scheduler entry starts, and that half of this issue is still open.** - -The other half — that the gate ran nowhere — is closed. `sched_check_build` -(`tests/toyos.rs`) boots the `sched-check` kernel and `tests/common/passcost.rs` -judges what it publishes; the second sentence of the paragraph this replaced, -that invariant P "has never executed against the kernel in any image or any test -run", was true when it was written and has not been since. Invariant P itself no -longer exists: a pass's elapsed time is wall clock and a guest's wall clock -advances while a hypervisor holds its vCPU, so the budget is measured and gated -in the harness rather than asserted in the kernel (`tests/common/passcost.rs`). -Widening the window is untouched by that and is what this file still wants. diff --git a/issues/kernel/soundd-past-due-wake-max-1.md b/issues/kernel/soundd-past-due-wake-max-1.md index 13adaa94156..7cf16c2191a 100644 --- a/issues/kernel/soundd-past-due-wake-max-1.md +++ b/issues/kernel/soundd-past-due-wake-max-1.md @@ -16,6 +16,5 @@ cpu1 both stopped within 100 ms of `soundd: resumed`. Not fixed here, and deliberately: it is one line in the mix loop, `apic::OneShot`'s floor makes it safe, and what it changes is when soundd wakes on a late period — -audio timing, which is the owner's call and which gate A's thorough tier cannot -currently adjudicate (`issues/audio/`). With the floor, a past-due grid point now costs up to +audio timing, which is the owner's call. With the floor, a past-due grid point now costs up to 10 µs of extra lateness against a 2.9 ms period. diff --git a/issues/kernel/syscall-preemption-is-incidental.md b/issues/kernel/syscall-preemption-is-incidental.md index 23c12f0933b..e1a7fa60e56 100644 --- a/issues/kernel/syscall-preemption-is-incidental.md +++ b/issues/kernel/syscall-preemption-is-incidental.md @@ -36,5 +36,4 @@ This was masked until the preempt count was made conserved across a context switch (the scheduler's own baselines needed it): before that the count drifted, so a lock drop inside a syscall reached zero at random and preempted at random. The behaviour is now deterministic, and deterministically weaker than the model -assumes. Whether that matters is measurable — gate A's wake-lateness -distribution is the instrument — and it did not move at N=8. +assumes. diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 1c79d5b12f0..cd2f2cd8bae 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -89,9 +89,6 @@ actuators! { /// Cap the i8042 ISR at 4 bytes and answer empty until the mute verdict is out; `service` then polls the rest, so the verdict beats the sequence on every boot instead of on a loaded shard's luck. i8042_split_burst = "i8042-split-burst"; - /// Shorten the idle loop's health/PMM snapshot cadence from 10s to 200ms. - sched_fast_health = "sched-fast-health"; - /// Script the input core directly at end of boot. test_input_merge = "test-input-merge"; @@ -269,9 +266,6 @@ actuators! { /// mid-wait does. usb_port_gone = "usb-port-gone"; - /// Hold every mass-storage bulk completion back 2ms before the driver may see it. - usb_slow_device = "usb-slow-device"; - /// Report the preempt depth and backtrace at the deepest point of a disk transfer; it stages nothing, only measures. io_depth_probe = "io-depth-probe"; diff --git a/kernel/src/drivers/hda.rs b/kernel/src/drivers/hda.rs index f8676aa5a4f..579c341cbed 100644 --- a/kernel/src/drivers/hda.rs +++ b/kernel/src/drivers/hda.rs @@ -77,8 +77,6 @@ const SD_STS_FIFOE: u8 = 1 << 3; const SD_STS_DESE: u8 = 1 << 4; const SD_STS_WRITE_CLEAR: u8 = SD_STS_BCIS | SD_STS_FIFOE | SD_STS_DESE; -/// The pipeline shape; soundd's mix loop, its client ring depth and gate A's recorded counters are -/// sized against it. const PERIODS: usize = 8; const PERIOD_BYTES: usize = 512; diff --git a/kernel/src/drivers/xhci/mod.rs b/kernel/src/drivers/xhci/mod.rs index ef82aa6be07..dd235f04649 100644 --- a/kernel/src/drivers/xhci/mod.rs +++ b/kernel/src/drivers/xhci/mod.rs @@ -344,11 +344,6 @@ fn port_answers() -> bool { !crate::actuator::xhci_deaf_port() } -/// How long a mass-storage bulk transfer's completion is held back before the driver may see it. -/// -/// A kernel feature: QEMU cannot stage a slow-answering drive, only a failing one. -const SLOW_TRANSFER_NS: u64 = 2_000_000; - /// The boot-time connect settle reads the same interval the per-port machine uses. use portmachine::DEBOUNCE_NS as PORT_DEBOUNCE_NS; @@ -767,9 +762,6 @@ pub struct XhciController { /// Replaces the register, not a verdict: the port reads PED clear for every reader until reset (§4.19.1.1.3). software_disabled: PortMask, - /// The event ring slot a slow device's completion is held in, and when it was first seen. See [`SLOW_TRANSFER_NS`]. - held_event: Option<(u16, u64)>, - /// What the disk call now inside this controller may still spend, once its transport has broken; closed between calls. after_break: AfterBreak, @@ -893,9 +885,6 @@ impl XhciController { } barrier::dma_rmb(); let event: Trb = self.event_ring.read(at); - if crate::actuator::usb_slow_device() && !self.slow_device_would_have_answered(&event) { - return None; - } // A controller that has not answered yet, which QEMU cannot be: it // posts a command's completion inside the write to the doorbell. #[cfg(feature = "boot-actuators")] @@ -906,29 +895,6 @@ impl XhciController { Some(event) } - /// Whether a stick this slow would have answered yet; `true` for anything that is not a bound disk's bulk completion. See [`SLOW_TRANSFER_NS`]. - /// - /// Keyed on ring position: the head does not advance while an event is held, so a second look finds the same first-seen time. - fn slow_device_would_have_answered(&mut self, event: &Trb) -> bool { - let slot = ((event.control >> 24) & 0xFF) as u8; - let dci = ((event.control >> 16) & 0x1F) as u8; - let is_disk_bulk = (event.control >> 10) & 0x3F == EVENT_TRANSFER - && dci >= 2 - && self.msc.iter().any(|b| b.disk.is_some_and(|d| d.dev.slot_id() == slot)); - if !is_disk_bulk { - return true; - } - let now = crate::clock::nanos_since_boot(); - let since = match self.held_event { - Some((head, at)) if head == self.event_head => at, - _ => { - self.held_event = Some((self.event_head, now)); - now - } - }; - now.saturating_sub(since) >= SLOW_TRANSFER_NS - } - /// Give an event to the device it names; an interrupt completion dropped here leaves the device's ring empty for the life of the boot. /// /// A code other than Success or Short Packet is recorded rather than dropped, so [`Self::recover_endpoints`] can act on it. diff --git a/kernel/src/drivers/xhci/wait/boot.rs b/kernel/src/drivers/xhci/wait/boot.rs index d56866fa656..266465e1119 100644 --- a/kernel/src/drivers/xhci/wait/boot.rs +++ b/kernel/src/drivers/xhci/wait/boot.rs @@ -441,7 +441,6 @@ fn init_one(pci_dev: &PciDevice) -> Option { ports_dirty: false, outstanding: Outstanding::EMPTY, software_disabled: [0u64; 4], - held_event: None, after_break: toyos_xhci::call::AfterBreak::CLOSED, bulk_began: 0, stopped: None, diff --git a/kernel/src/drivers/xhci/wait/msc.rs b/kernel/src/drivers/xhci/wait/msc.rs index 8f1fe5421b7..01cae267558 100644 --- a/kernel/src/drivers/xhci/wait/msc.rs +++ b/kernel/src/drivers/xhci/wait/msc.rs @@ -142,11 +142,6 @@ impl MscDevice { self.no_write_cache } - /// Which slot this disk is on. - pub fn slot_id(&self) -> u8 { - self.slot_id - } - /// The port whose slot is owed back to the controller, once: the disk was /// taken offline with its endpoints Stopped, and its slot is still enabled. pub(in crate::drivers::xhci) fn take_slot_owed(&mut self) -> Option { diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index 14b41910db0..e35e2301e9c 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -560,33 +560,15 @@ const SNAPSHOT_INTERVAL: Cadence = Cadence::every( "one clock read and one relaxed compare per idle trip, on a CPU already awake", ); -/// `sched-fast-health`'s cadence: no guest test program this suite runs lives -/// past [`SNAPSHOT_INTERVAL`] once, let alone the two prints a comparison needs. -const FAST_SNAPSHOT_INTERVAL: Cadence = Cadence::every( - Duration::from_millis(200), - "an actuator no boot arms; a test that needs two prints buys them for one boot", -); - -/// Which of the two cadences this boot took, read once per idle trip. -fn snapshot_interval_ns() -> u64 { - if crate::actuator::sched_fast_health() { - FAST_SNAPSHOT_INTERVAL.nanos() - } else { - SNAPSHOT_INTERVAL.nanos() - } -} - /// When each CPU may next print its own line: per CPU, not global, so no /// single CPU speaks for all of them. static NEXT_HEALTH: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; -/// How many times each CPU has passed through idle since boot, counted on -/// every trip rather than only the ones that print: `i8042_quarantine` needs -/// the raw rate to tell a halting CPU from a spinning one. +/// How many times each CPU has passed through idle since boot. static IDLE_TRIPS: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; /// A snapshot of this CPU's run queues, at most once per -/// [`snapshot_interval_ns`], plus the machine's page pools on the same +/// [`SNAPSHOT_INTERVAL`], plus the machine's page pools on the same /// cadence. Called from the idle loop on every trip; the cadence is wall /// clock rather than per-trip because a CPU that declines to sleep loops at /// memory speed. Not a heartbeat: a busy CPU prints nothing, so a gap here @@ -600,7 +582,7 @@ pub fn log_health() { .get(cpu as usize) .map_or(0, |t| t.fetch_add(1, Ordering::Relaxed) + 1); if now >= next_health.load(Ordering::Relaxed) { - next_health.store(now + snapshot_interval_ns(), Ordering::Relaxed); + next_health.store(now + SNAPSHOT_INTERVAL.nanos(), Ordering::Relaxed); let ready = driver::ready_len() + usize::from(percpu::current_tid().is_some()); let parked = driver::parked_len(); let dying = driver::dying_len(); @@ -620,12 +602,12 @@ pub fn log_health() { static NEXT_PMM_DUMP: AtomicU64 = AtomicU64::new(0); let next = NEXT_PMM_DUMP.load(Ordering::Relaxed); if next == 0 { - NEXT_PMM_DUMP.store(now + snapshot_interval_ns(), Ordering::Relaxed); + NEXT_PMM_DUMP.store(now + SNAPSHOT_INTERVAL.nanos(), Ordering::Relaxed); } else if now >= next && NEXT_PMM_DUMP .compare_exchange( next, - now + snapshot_interval_ns(), + now + SNAPSHOT_INTERVAL.nanos(), Ordering::Relaxed, Ordering::Relaxed, ) diff --git a/src/build.rs b/src/build.rs index c5b98186a8b..d775d1d105f 100644 --- a/src/build.rs +++ b/src/build.rs @@ -1675,12 +1675,11 @@ fn assert_entry_window_matches_features(features: &str, kernel: &[u8]) { /// bound and the container-versus-state-word agreement. The third is the /// pass-cost report (`cpu::PassCostReport::PREFIX`), which is a *measurement* /// and not an assert: a pass's elapsed time includes any interval a hypervisor -/// took the CPU away, so it is recorded and gated in the harness rather than -/// panicked over. Their format strings are the only part of the check build -/// with a literal the linker keeps, which is what makes the artifact answerable -/// at all — and the report's literal is kept out of the shipping kernel by -/// nothing but dead-code elimination, which the `want == false` direction below -/// is what checks. +/// took the CPU away, so it is recorded rather than panicked over. Their format +/// strings are the only part of the check build with a literal the linker keeps, +/// which is what makes the artifact answerable at all — and the report's literal +/// is kept out of the shipping kernel by nothing but dead-code elimination, +/// which the `want == false` direction below is what checks. const SCHED_CHECK_LITERALS: [&str; 3] = [ "sched-check pass-costs cpu=", "invariant T: cpu", @@ -3108,7 +3107,6 @@ mod tests { "tests/blockdcase/system.toml", "tests/desktopcase/system.toml", "tests/desktopaudiocase/system.toml", - "tests/doomcase/system.toml", "tests/doommusiccase/system.toml", "tests/e1000case/system.toml", "tests/e1000leasecase/system.toml", diff --git a/src/ci.rs b/src/ci.rs index c120d590b03..89f11b2ed56 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -53,7 +53,6 @@ const USAGE: &str = "cargo run -- --ci , where is one of: toolchain publish this tree's toolchain if nobody has (nightly) guest / one shard of the guest suite at the reach its schedule names (nightly) tcg one test on an emulated CPU (nightly) - audio / one shard of gate A (nightly) nightly-red file or update the nightly-red issue from $NEEDS (nightly) publish put main's SDK crates on crates.io (publish.yml)"; @@ -64,7 +63,6 @@ enum Job { Toolchain, Guest(String), Tcg, - Audio(String), NightlyRed, Publish, } @@ -81,13 +79,12 @@ fn parse(words: &[String]) -> Result { Some("toolchain") => Job::Toolchain, Some("guest") => Job::Guest(shard(words.get(1))?), Some("tcg") => Job::Tcg, - Some("audio") => Job::Audio(shard(words.get(1))?), Some("nightly-red") => Job::NightlyRed, Some("publish") => Job::Publish, Some(other) => return Err(format!("no CI job is called {other:?}")), None => return Err("which job?".to_string()), }; - let takes = usize::from(matches!(job, Job::Guest(_) | Job::Audio(_))) + 1; + let takes = usize::from(matches!(job, Job::Guest(_))) + 1; if words.len() > takes { return Err(format!("{:?} takes nothing after it: {:?}", words[0], &words[takes..])); } @@ -108,7 +105,6 @@ pub fn dispatch(root: &Path, args: &[String]) { Err(refusal) => vec![step("the reach", || Err(refusal))], }, Job::Tcg => guest(root, &suite_args(&["--jobs", "1", "process_stats"])), - Job::Audio(shard) => guest(root, &suite_args(&["--audio-gate", "30", "--shard", shard])), Job::NightlyRed => vec![step("the nightly-red issue", nightly_red)], Job::Publish => vec![step("the SDK crates on crates.io", || publish(root))], }; diff --git a/src/heartbeat.rs b/src/heartbeat.rs deleted file mode 100644 index eddfe9d099d..00000000000 --- a/src/heartbeat.rs +++ /dev/null @@ -1,756 +0,0 @@ -//! Whether a heartbeat capture settled and what its mask then says — the -//! verdict `kernel_heartbeat` in `tests/toyos.rs` reads. Text in, a verdict -//! out, so a capture the instrument has already taken replays here against the -//! rule. -//! -//! **What a clear bit can be evidence of.** `mask=` says which CPUs reached a -//! scheduler pass in the period (`kernel/src/heartbeat.rs`), and a CPU running -//! one task with nothing to preempt it takes its timer and reaches none, as does -//! one inside a disk wait, which cannot park — so a clear bit reads "busy" and -//! "stopped" alike. The test exists for the second, on a machine that is -//! otherwise running, and a beat can say so only where the machine held that -//! state: -//! -//! - **after the boot's start-up.** That is not `Boot: complete`, which is -//! printed as `init` is spawned, and not `init`'s last spawn record either: -//! the programs `init` starts do their own start-up after it, and that work -//! is what keeps a CPU off the mask. The boot has started when every -//! `[boot] start` program has said it is done — its ready line or its exit -//! record, which `DONE` is the one table of — and the window opens at the -//! first beat whose whole period follows the last of them. -//! - **on a beat the machine ran through.** A beat later than `LATE` periods -//! is a period no CPU reached the idle loop in, and `ran=0` is a period no -//! CPU dispatched a task in. Either is the machine not running — a guest its -//! host did not schedule — and what a CPU did in it is unreadable, so such a -//! beat closes the window and the sample is [`Refused::NotRunning`], never a -//! CPU missing. -//! - **over `MIN_SETTLED` beats.** A verdict about a CPU is read from that many -//! settled beats or from none: a shorter window says only why it is short. -//! -//! Inside the window a CPU absent from [`STOPPED_BEATS`] consecutive beats has -//! stopped: `diag-tick` caps a sleep at 100 ms against a 250 ms line, so it has -//! missed five wakes. Absent from one and back on the next it has missed two, -//! which the owning instrument produces on a healthy guest. -//! -//! **The capture follows the window, not a clock.** How long a boot's start-up -//! takes is the loaded host's to decide, so an instrument that drains for a -//! fixed span hands this module whatever is left over and reds on its own -//! refusal when that is less than a verdict needs. [`window_beats`] is what a -//! capture is taken to: it says how much window the capture holds so far, and -//! [`CAPTURE_BEATS`] is enough. - -#![forbid(unsafe_code)] - -/// The line's period, `kernel/src/heartbeat.rs`'s `PERIOD_NS`. -pub const PERIOD_MS: u64 = 250; - -/// A beat whose `gap=` exceeds this many periods is one the machine did not run -/// through: eight CPUs whose longest sleep is 100 ms, and none reached the idle -/// loop for a whole period. -const LATE: u64 = 2; - -/// Consecutive settled beats a CPU is absent from before it has stopped. -pub const STOPPED_BEATS: usize = 2; - -/// The fewest settled beats a verdict is read from. -const MIN_SETTLED: usize = 4; - -/// Settled beats a capture is taken to. `MIN_SETTLED` is the floor a verdict is -/// read from and the spare is detection, not slack: a CPU whose first absence -/// is the capture's last beat is a blip and convicts nobody, so the capture -/// carries one beat past the floor — enough for [`STOPPED_BEATS`]'s second -/// absence to land in. -pub const CAPTURE_BEATS: usize = MIN_SETTLED + 1; - -/// The widest `alive=N/M` denominator that is a reading of the `mask=` beside -/// it: that mask is 64 bits, so no `M` at 64 or above describes it, and `M = 0` -/// describes no machine. -const MOST_CPUS: u32 = 63; - -/// One `heartbeat: t=… alive=… mask=… ran=… gap=…` line, read. -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct Beat { - /// Index of the line in the capture. - pub line: usize, - pub t_ms: u64, - pub cpus: u32, - pub mask: u64, - pub ran: u64, - pub gap_ms: u64, -} - -impl Beat { - /// The beat's reading of `text`, or `None` where any field is unreadable. - fn parse(line: usize, text: &str) -> Option { - let field = |key: &str| text.split(key).nth(1)?.split_whitespace().next(); - let cpus: u32 = field("alive=")?.split_once('/')?.1.parse().ok()?; - if !(1..=MOST_CPUS).contains(&cpus) { - return None; - } - Some(Beat { - line, - t_ms: millis(field("t=")?)?, - cpus, - mask: u64::from_str_radix(field("mask=0x")?, 16).ok()?, - ran: field("ran=")?.parse().ok()?, - gap_ms: millis(field("gap=")?)?, - }) - } - - fn full(&self) -> bool { - self.mask == (1u64 << self.cpus) - 1 - } - - fn absent(&self, cpu: u32) -> bool { - self.mask & (1 << cpu) == 0 - } - - /// Whether the machine ran through this beat's period. - fn ran_through(&self) -> bool { - self.gap_ms <= LATE * PERIOD_MS && self.ran > 0 - } -} - -/// `S.mmms` as milliseconds. -fn millis(field: &str) -> Option { - let (s, ms) = field.strip_suffix('s')?.split_once('.')?; - if ms.len() != 3 { - return None; - } - Some(s.parse::().ok()? * 1000 + ms.parse::().ok()?) -} - -/// `tests/metalcase`'s `[boot] start` programs and the line each says it has -/// finished starting with. The one table: [`done_lines`] holds it against the -/// config, and a caller reads it through that rather than declaring its own. -const DONE: &[(&str, &str)] = &[ - ("logd", "logd: this boot's kernel log is"), - ("compositor", "compositor: ready"), - ("soundd", "soundd: null sink idle"), - ("netd", "exit: netd pid="), - ("sshd", "exit: sshd pid="), - ("test-runner", "===READY==="), -]; - -/// The done line of each program in `start`, or the disagreement between the -/// config and [`DONE`] — a `[boot] start` program with no done line here leaves -/// its own start-up inside the window, which is the one thing the window exists -/// to exclude. -pub fn done_lines(start: &[String]) -> Result, String> { - let start: Vec<&str> = start.iter().map(String::as_str).collect(); - let known: Vec<&str> = DONE.iter().map(|(program, _)| *program).collect(); - if start != known { - return Err(format!( - "`tests/metalcase` starts {start:?} and `src/heartbeat.rs` knows the done line of \ - {known:?} — a program without one leaves its start-up inside the window" - )); - } - Ok(DONE.iter().map(|(_, line)| *line).collect()) -} - -/// Why a capture is not a claim about a settled, running machine's CPUs. -#[derive(Debug, PartialEq, Eq)] -pub enum Refused { - /// A `heartbeat: t=` line one of whose fields would not parse. - Unreadable(String), - /// A `[boot] start` program never said it was done: the line it says it with. - BootUnfinished(String), - /// Fewer than `MIN_SETTLED` beats had a whole period after the boot's - /// start-up. - Unsettled { settled: usize, beats: usize }, - /// A settled beat the machine did not run through, after `held` it did. - NotRunning { beat: Beat, held: usize }, - /// CPUs absent from [`STOPPED_BEATS`] consecutive settled beats, of `settled` - /// from the capture line `opened`. - CpuMissing { cpus: Vec, settled: usize, opened: usize }, -} - -/// The settled window, read. -#[derive(Debug, PartialEq, Eq)] -pub struct Settled { - pub beats: Vec, - /// Settled beats missing a CPU — for one line each, since none for two. - pub blips: usize, - /// The widest `gap=` anywhere in the capture, settled or not: a window - /// between two lines is wide enough to hide a death wherever it falls. - pub widest_gap_ms: u64, -} - -/// The beats whose whole period follows the last of `started`, or the done line -/// nothing in `lines` said. -fn window<'a>(beats: &'a [Beat], lines: &[&str], started: &[&str]) -> Result<&'a [Beat], String> { - let mut last = 0; - for said in started { - let Some(at) = lines.iter().position(|l| l.contains(said)) else { - return Err((*said).to_string()); - }; - last = last.max(at); - } - // The beat after the record straddles it; the one after that is the first - // whose whole period follows it. - let after = beats.iter().filter(|b| b.line > last).count(); - Ok(&beats[(beats.len() - after + 1).min(beats.len())..]) -} - -/// How much window `lines` holds so far — what a capture is taken to, against -/// [`CAPTURE_BEATS`]. Zero until every one of `started` has said it is done, and -/// a line whose fields will not parse is not a beat here; [`settle`] is what -/// refuses one. -pub fn window_beats(lines: &[&str], started: &[&str]) -> usize { - let beats: Vec = - lines.iter().enumerate().filter_map(|(i, l)| Beat::parse(i, l)).collect(); - window(&beats, lines, started).map_or(0, <[Beat]>::len) -} - -/// The verdict on `lines`, a capture whose boot's start-up ends with the last of -/// `started` — one line per `[boot] start` program, the one it says it is done -/// with. -pub fn settle(lines: &[&str], started: &[&str]) -> Result { - let beats = lines - .iter() - .enumerate() - .filter(|(_, l)| l.contains("heartbeat: t=")) - .map(|(i, l)| Beat::parse(i, l).ok_or_else(|| Refused::Unreadable(l.to_string()))) - .collect::, _>>()?; - if let Some(odd) = beats.iter().find(|b| b.cpus != beats[0].cpus) { - return Err(Refused::Unreadable(lines[odd.line].to_string())); - } - let window = window(&beats, lines, started).map_err(Refused::BootUnfinished)?; - let held = window.iter().position(|b| !b.ran_through()).unwrap_or(window.len()); - let read = &window[..held]; - // A CPU is read from `MIN_SETTLED` settled beats or from none, so what a - // short window says is only why it is short. - if held < MIN_SETTLED { - return Err(if held < window.len() { - Refused::NotRunning { beat: window[held].clone(), held } - } else { - Refused::Unsettled { settled: held, beats: beats.len() } - }); - } - let cpus: Vec = (0..beats[0].cpus) - .filter(|&c| read.windows(STOPPED_BEATS).any(|w| w.iter().all(|b| b.absent(c)))) - .collect(); - if !cpus.is_empty() { - return Err(Refused::CpuMissing { cpus, settled: read.len(), opened: read[0].line }); - } - if held < window.len() { - return Err(Refused::NotRunning { beat: window[held].clone(), held }); - } - Ok(Settled { - blips: read.iter().filter(|b| !b.full()).count(), - widest_gap_ms: beats.iter().map(|b| b.gap_ms).max().unwrap_or(0), - beats: read.to_vec(), - }) -} - -#[cfg(test)] -mod tests { - use std::path::Path; - - use super::*; - - /// `tests/metalcase`'s done lines, as the test passes them. - fn started() -> Vec<&'static str> { - done_lines(&crate::build::boot_start( - &Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/metalcase/system.toml"), - )) - .unwrap() - } - - /// Nightly `35072262489`, guest shard 8, suite run: every heartbeat line and - /// every `[boot] start` program's done line, in the capture's order and - /// verbatim, the lines between them dropped. - const SHARD_8_SUITE: &str = "\ -logd: this boot's kernel log is /log/2026-09-16-083530.log (2026-09-16 08:35:30 at UTC+0 recovered from two readings) -[kernel 1.109 cpu1] heartbeat: t=1.109s alive=8/8 mask=0xff ran=11 gap=0.260s -[kernel 1.361 cpu7] heartbeat: t=1.361s alive=7/8 mask=0xdf ran=4 gap=0.251s -[kernel 1.361 cpu7] heartbeat: cpu5 last reached one 0.307s ago -[kernel 1.618 cpu3] heartbeat: t=1.618s alive=7/8 mask=0xbf ran=5 gap=0.257s -[kernel 1.618 cpu3] heartbeat: cpu6 last reached one 0.423s ago -soundd: null sink idle -[kernel 1.762 cpu7] exit: netd pid=7 code=0 cpu=399ms -[kernel 1.934 cpu1] heartbeat: t=1.934s alive=7/8 mask=0xfe ran=9 gap=0.315s -[kernel 1.934 cpu1] heartbeat: cpu0 last reached one 0.572s ago -===READY=== -[kernel 2.021 cpu0] spawn: /system/bin/test-runner pid=9 tid=0 dst=2 base=0x10000000000 entry=0x1000001ff80 cr3=0x6ecf000 symbols=2048KiB (layout=21ms relocs=0ms deps=0ms tls=1ms total=75ms) -[kernel 2.184 cpu0] heartbeat: t=2.184s alive=7/8 mask=0xdf ran=8 gap=0.250s -[kernel 2.184 cpu0] heartbeat: cpu5 last reached one 0.339s ago -[kernel 2.434 cpu0] heartbeat: t=2.434s alive=7/8 mask=0xdf ran=15 gap=0.250s -[kernel 2.434 cpu0] heartbeat: cpu5 last reached one 0.589s ago -compositor: ready -[kernel 2.618 cpu1] exit: sshd pid=8 code=0 cpu=634ms -[kernel 2.686 cpu2] heartbeat: t=2.686s alive=8/8 mask=0xff ran=36 gap=0.251s -[kernel 2.937 cpu2] heartbeat: t=2.937s alive=8/8 mask=0xff ran=43 gap=0.251s -[kernel 3.188 cpu2] heartbeat: t=3.188s alive=8/8 mask=0xff ran=43 gap=0.251s -[kernel 3.439 cpu2] heartbeat: t=3.439s alive=8/8 mask=0xff ran=43 gap=0.251s -[kernel 3.690 cpu5] heartbeat: t=3.690s alive=8/8 mask=0xff ran=42 gap=0.250s -[kernel 3.940 cpu2] heartbeat: t=3.940s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 4.196 cpu5] heartbeat: t=4.196s alive=8/8 mask=0xff ran=44 gap=0.255s -[kernel 4.449 cpu5] heartbeat: t=4.449s alive=8/8 mask=0xff ran=44 gap=0.252s -[kernel 4.703 cpu2] heartbeat: t=4.703s alive=8/8 mask=0xff ran=43 gap=0.253s -[kernel 4.953 cpu2] heartbeat: t=4.953s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 5.203 cpu2] heartbeat: t=5.203s alive=8/8 mask=0xff ran=44 gap=0.250s -"; - - /// The same shard's ALONE re-run, the same way. - const SHARD_8_ALONE: &str = "\ -logd: this boot's kernel log is /log/2026-09-16-083541.log (2026-09-16 08:35:41 at UTC+0 recovered from two readings) -[kernel 1.029 cpu1] heartbeat: t=1.029s alive=8/8 mask=0xff ran=11 gap=0.262s -[kernel 1.290 cpu7] heartbeat: t=1.290s alive=7/8 mask=0xdf ran=4 gap=0.260s -[kernel 1.290 cpu7] heartbeat: cpu5 last reached one 0.323s ago -[kernel 1.541 cpu3] heartbeat: t=1.541s alive=7/8 mask=0xbf ran=5 gap=0.251s -[kernel 1.544 cpu3] heartbeat: cpu6 last reached one 0.429s ago -soundd: null sink idle -[kernel 1.692 cpu7] exit: netd pid=7 code=0 cpu=402ms -[kernel 1.793 cpu7] heartbeat: t=1.793s alive=7/8 mask=0xfe ran=10 gap=0.251s -[kernel 1.793 cpu7] heartbeat: cpu0 last reached one 0.499s ago -===READY=== -[kernel 1.951 cpu0] spawn: /system/bin/test-runner pid=9 tid=0 dst=2 base=0x10000000000 entry=0x1000001ff80 cr3=0x2d83000 symbols=2048KiB (layout=9ms relocs=0ms deps=0ms tls=1ms total=79ms) -[kernel 2.043 cpu0] heartbeat: t=2.043s alive=7/8 mask=0xdf ran=8 gap=0.250s -[kernel 2.043 cpu0] heartbeat: cpu5 last reached one 0.281s ago -[kernel 2.293 cpu0] heartbeat: t=2.293s alive=6/8 mask=0xdd ran=1 gap=0.250s -[kernel 2.293 cpu0] heartbeat: cpu1 last reached one 0.422s ago -[kernel 2.293 cpu0] heartbeat: cpu5 last reached one 0.531s ago -compositor: ready -[kernel 2.543 cpu0] heartbeat: t=2.543s alive=7/8 mask=0xdf ran=25 gap=0.250s -[kernel 2.543 cpu0] heartbeat: cpu5 last reached one 0.781s ago -[kernel 2.624 cpu1] exit: sshd pid=8 code=0 cpu=708ms -[kernel 2.793 cpu5] heartbeat: t=2.793s alive=8/8 mask=0xff ran=42 gap=0.250s -[kernel 3.044 cpu2] heartbeat: t=3.044s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 3.294 cpu2] heartbeat: t=3.294s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 3.545 cpu2] heartbeat: t=3.545s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 3.796 cpu2] heartbeat: t=3.796s alive=8/8 mask=0xff ran=43 gap=0.251s -[kernel 4.047 cpu2] heartbeat: t=4.047s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 4.298 cpu2] heartbeat: t=4.298s alive=8/8 mask=0xff ran=43 gap=0.250s -[kernel 4.557 cpu2] heartbeat: t=4.557s alive=8/8 mask=0xff ran=44 gap=0.259s -[kernel 4.814 cpu2] heartbeat: t=4.814s alive=8/8 mask=0xff ran=43 gap=0.256s -[kernel 5.065 cpu2] heartbeat: t=5.065s alive=8/8 mask=0xff ran=43 gap=0.251s -"; - - /// A dev-host boot, up to its torn last beat: every heartbeat line, every - /// `[boot] start` program's done line and the two lines naming the wait, in - /// the capture's order and verbatim, the lines between them dropped. - const DISK_WAIT_PINS_CPU4: &str = "\ -logd: this boot's kernel log is /log/2026-09-18-135738.log (2026-09-18 13:57:38 at UTC+0 recovered from two readings) -[kernel 1.122 cpu3] heartbeat: t=1.121s alive=8/8 mask=0xff ran=11 gap=0.261s -[kernel 1.395 cpu1] heartbeat: t=1.395s alive=8/8 mask=0xff ran=4 gap=0.273s -soundd: null sink idle -[kernel 1.529 cpu7] exit: netd pid=7 code=0 cpu=94ms -[kernel 1.645 cpu0] heartbeat: t=1.645s alive=8/8 mask=0xff ran=22 gap=0.250s -===READY=== -[kernel 1.895 cpu0] heartbeat: t=1.895s alive=6/8 mask=0xdb ran=4 gap=0.250s -[kernel 2.146 cpu0] heartbeat: t=2.145s alive=6/8 mask=0xdd ran=23 gap=0.250s -[kernel 2.230 cpu1] exit: sshd pid=8 code=0 cpu=666ms -[kernel 2.397 cpu2] heartbeat: t=2.397s alive=8/8 mask=0xff ran=31 gap=0.251s -compositor: ready -[kernel 2.658 cpu2] heartbeat: t=2.658s alive=8/8 mask=0xff ran=33 gap=0.261s -[kernel 2.909 cpu0] heartbeat: t=2.909s alive=8/8 mask=0xff ran=38 gap=0.250s -[kernel 3.159 cpu0] heartbeat: t=3.159s alive=7/8 mask=0xef ran=35 gap=0.250s -[kernel 3.409 cpu0] heartbeat: t=3.409s alive=7/8 mask=0xef ran=26 gap=0.250s -[kernel 3.659 cpu0] heartbeat: t=3.659s alive=7/8 mask=0xef ran=38 gap=0.250s -[kernel 3.909 cpu0] heartbeat: t=3.909s alive=7/8 mask=0xef ran=32 gap=0.250s -[kernel 4.159 cpu0] heartbeat: t=4.159s alive=7/8 mask=0xef ran=37 gap=0.250s -[kernel 4.409 cpu0] heartbeat: t=4.409s alive=7/8 mask=0xef ran=37 gap=0.250s -[kernel 4.659 cpu0] heartbeat: t=4.659s alive=7/8 mask=0xef ran=33 gap=0.250s -[kernel 4.663 cpu4] usb-storage: 00:02.0 slot 1 transport broke on SCSI 0x2a: no answer in the status phase in 2000 ms -[kernel 4.677 cpu4] fsync: /log/2026-09-18-135738.log durable on attempt 2 after 2016ms — a refused attempt kept every page dirty and a later one delivered them -"; - - /// A CPU a disk wait pins reaches no pass, and that is what the mask says: - /// every beat of it ran through, so neither refusal excuses it, and the - /// verdict names the CPU. The red's owner is the wait that cannot park. - #[test] - fn a_cpu_pinned_by_a_disk_wait_is_reported_and_not_excused() { - let lines = lines(DISK_WAIT_PINS_CPU4); - let opened = lines.iter().position(|l| l.contains("t=2.909s")).unwrap(); - assert_eq!( - settle(&lines, &started()), - Err(Refused::CpuMissing { cpus: vec![4], settled: 8, opened }) - ); - } - - fn lines(capture: &str) -> Vec<&str> { - capture.lines().collect() - } - - fn beat(t_ms: u64, mask: u64, ran: u64, gap_ms: u64) -> String { - format!( - "[kernel {}.{:03} cpu2] heartbeat: t={}.{:03}s alive={}/8 mask={mask:#04x} ran={ran} \ - gap={}.{:03}s", - t_ms / 1000, - t_ms % 1000, - t_ms / 1000, - t_ms % 1000, - mask.count_ones(), - gap_ms / 1000, - gap_ms % 1000, - ) - } - - /// A capture whose boot's start-up ends before its first beat, then `beats`. - fn settled_capture(beats: &[String]) -> String { - let mut capture: String = started().iter().map(|s| format!("{s}\n")).collect(); - capture.push_str(&beat(2000, 0xff, 40, 250)); - capture.push('\n'); - for b in beats { - capture.push_str(b); - capture.push('\n'); - } - capture - } - - /// The started programs' own start-up outlasts `init`'s last spawn record, - /// so a window opened at a full mask holds cpu5 absent from two consecutive - /// beats; opened after the last done line it holds no clear bit at all. - #[test] - fn the_nightly_shard_8_boots_settle_after_the_last_program_finishes_starting() { - for (capture, opens_at, settled, widest) in - [(SHARD_8_SUITE, 2937, 10, 315), (SHARD_8_ALONE, 3044, 9, 262)] - { - let lines = lines(capture); - let beats: Vec = lines - .iter() - .enumerate() - .filter_map(|(i, l)| Beat::parse(i, l)) - .collect(); - assert_eq!(beats.len(), 17); - assert!(beats[0].full()); - let spawned = lines - .iter() - .position(|l| l.contains("spawn: /system/bin/test-runner")) - .unwrap(); - let after_spawn: Vec<&Beat> = beats.iter().filter(|b| b.line > spawned).collect(); - assert!(after_spawn[0].absent(5) && after_spawn[1].absent(5)); - let verdict = settle(&lines, &started()).unwrap(); - assert_eq!(verdict.beats[0].t_ms, opens_at); - assert_eq!(verdict.beats.len(), settled); - assert_eq!(verdict.blips, 0); - assert_eq!(verdict.widest_gap_ms, widest); - let exit = lines.iter().position(|l| l.contains("exit: sshd")).unwrap(); - let straddling = beats.iter().find(|b| b.line > exit).unwrap(); - assert!(straddling.t_ms < opens_at); - } - } - - /// A dev-host boot the host stopped scheduling: 37 beats, a pair of full - /// masks at t=2.654 s and 2.904 s, then seven seconds of `alive=4/8 ran=0` - /// naming cpu[0, 1, 2, 4, 5]. Reconstructed from those numbers rather than - /// captured — the boot beats and the split of the missing set across the - /// tail are this test's. - fn dev_host_capture() -> String { - let started = started(); - let mut capture = String::new(); - for said in &started[..4] { - capture.push_str(said); - capture.push('\n'); - } - for (t, mask, ran) in [(904, 0xff, 11), (1154, 0xdf, 4), (1404, 0xbf, 5), (1654, 0xfe, 9)] { - capture.push_str(&beat(t, mask, ran, 250)); - capture.push('\n'); - } - capture.push_str("===READY===\n"); - for (t, mask, ran) in [(1904, 0xdf, 8), (2154, 0xdf, 15)] { - capture.push_str(&beat(t, mask, ran, 250)); - capture.push('\n'); - } - capture.push_str("[kernel 2.3 cpu1] exit: sshd pid=8 code=0 cpu=634ms\n"); - for (t, mask, ran) in [(2404, 0xdd, 1), (2654, 0xff, 36), (2904, 0xff, 43)] { - capture.push_str(&beat(t, mask, ran, 250)); - capture.push('\n'); - } - for i in 0..28u64 { - let mask = if i < 14 { 0xe8 } else { 0xc9 }; - capture.push_str(&beat(3154 + i * 250, mask, 0, 250)); - capture.push('\n'); - } - capture - } - - #[test] - fn the_dev_host_capture_is_a_machine_that_was_not_running() { - let capture = dev_host_capture(); - let lines = lines(&capture); - let beats: Vec = - lines.iter().enumerate().filter_map(|(i, l)| Beat::parse(i, l)).collect(); - assert_eq!(beats.len(), 37); - let pair = beats.windows(2).position(|w| w[0].full() && w[1].full()).unwrap(); - assert_eq!(beats[pair].t_ms, 2654); - let stopped: Vec = (0..8) - .filter(|&c| beats[pair..].windows(2).any(|w| w[0].absent(c) && w[1].absent(c))) - .collect(); - assert_eq!(stopped, [0, 1, 2, 4, 5]); - assert_eq!( - settle(&lines, &started()), - Err(Refused::NotRunning { beat: beats[pair + 2].clone(), held: 2 }) - ); - assert_eq!(beats[pair + 2].ran, 0); - } - - /// The owning instrument's healthy shape: one line a CPU is absent from and - /// back on is two missed wakes, not a stopped CPU. - #[test] - fn one_line_absent_and_back_is_a_blip() { - let mut beats: Vec = (1..=9).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.push(beat(4500, 0xbf, 43, 250)); - beats.extend((1..=8).map(|i| beat(4500 + i * 250, 0xff, 43, 250))); - let capture = settled_capture(&beats); - let verdict = settle(&lines(&capture), &started()).unwrap(); - assert_eq!(verdict.beats.len(), 18); - assert_eq!(verdict.blips, 1); - assert_eq!(verdict.widest_gap_ms, 250); - } - - #[test] - fn a_cpu_absent_from_two_consecutive_settled_beats_has_stopped() { - let mut beats: Vec = (1..=3).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.extend((1..=6).map(|i| beat(2750 + i * 250, 0xdf, 43, 250))); - let capture = settled_capture(&beats); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::CpuMissing { cpus: vec![5], settled: 9, opened: started().len() + 1 }) - ); - } - - /// A late beat is a period no CPU reached the idle loop in; an empty one is - /// a period no CPU ran a task in. Both close the window, and what follows - /// is not read as a CPU. - #[test] - fn a_beat_the_machine_did_not_run_through_closes_the_window() { - for (gap_ms, ran) in [(600, 43), (250, 0)] { - let mut beats: Vec = - (1..=4).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - let stalled = beat(3650, 0xdf, ran, gap_ms); - beats.push(stalled.clone()); - beats.extend((1..=6).map(|i| beat(3650 + i * 250, 0xdf, 43, 250))); - let capture = settled_capture(&beats); - let lines = lines(&capture); - let at = lines.iter().position(|l| *l == stalled).unwrap(); - assert_eq!( - settle(&lines, &started()), - Err(Refused::NotRunning { beat: Beat::parse(at, &stalled).unwrap(), held: 4 }) - ); - } - } - - /// `LATE`, both sides of it, in milliseconds rather than in the constant: a - /// 0.500 s gap is two 250 ms periods and the machine ran through it, and a - /// 0.600 s gap is a period in which no CPU reached the idle loop at all — - /// five `diag-tick` wakes missed — and closes the window. - #[test] - fn two_periods_of_gap_is_run_through_and_more_than_two_is_not() { - let settled = |gap_ms| { - let mut beats: Vec = - (1..=4).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.push(beat(3250, 0xff, 43, gap_ms)); - settle(&lines(&settled_capture(&beats)), &started()).map(|s| s.beats.len()) - }; - assert_eq!(settled(500), Ok(5)); - assert!(matches!(settled(600), Err(Refused::NotRunning { held: 4, .. }))); - } - - #[test] - fn a_cpu_that_stopped_before_the_stall_is_still_the_finding() { - let mut beats: Vec = (1..=4).map(|i| beat(2000 + i * 250, 0xdf, 43, 250)).collect(); - beats.push(beat(3400, 0xdf, 43, 600)); - let capture = settled_capture(&beats); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::CpuMissing { cpus: vec![5], settled: 4, opened: started().len() + 1 }) - ); - } - - /// A verdict about a CPU is read from `MIN_SETTLED` settled beats or from - /// none: a stall two beats in, and a capture that ends three beats in, each - /// say why the window is short and neither names a CPU. - #[test] - fn a_cpu_missing_from_a_window_shorter_than_the_minimum_is_not_named() { - let mut beats: Vec = (1..=2).map(|i| beat(2000 + i * 250, 0xdf, 43, 250)).collect(); - beats.push(beat(3400, 0xdf, 43, 600)); - let capture = settled_capture(&beats); - let stalled = lines(&capture); - let at = stalled.iter().position(|l| l.contains("t=3.400s")).unwrap(); - assert_eq!( - settle(&stalled, &started()), - Err(Refused::NotRunning { beat: Beat::parse(at, stalled[at]).unwrap(), held: 2 }) - ); - - let beats: Vec = (1..=3).map(|i| beat(2000 + i * 250, 0xdf, 43, 250)).collect(); - assert_eq!( - settle(&lines(&settled_capture(&beats)), &started()), - Err(Refused::Unsettled { settled: 3, beats: 4 }) - ); - } - - #[test] - fn a_program_that_never_finishes_starting_refuses_the_whole_capture() { - let without: Vec<&str> = lines(SHARD_8_SUITE) - .into_iter() - .filter(|l| !l.contains("exit: sshd")) - .collect(); - assert_eq!( - settle(&without, &started()), - Err(Refused::BootUnfinished("exit: sshd pid=".to_string())) - ); - assert_eq!(window_beats(&without, &started()), 0); - } - - /// The beat after the last done line straddles it, so the window opens on - /// the one after that; a done line after every beat opens nothing. - #[test] - fn the_window_opens_on_the_first_whole_period_after_the_last_done_line() { - let started = started(); - let beats: Vec = (1..=6).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - let mut capture: String = started[..5].iter().map(|s| format!("{s}\n")).collect(); - capture.push_str(&beats[0]); - capture.push_str("\n===READY===\n"); - for b in &beats[1..] { - capture.push_str(b); - capture.push('\n'); - } - let verdict = settle(&lines(&capture), &started).unwrap(); - assert_eq!(verdict.beats[0].t_ms, 2750); - assert_eq!(verdict.beats.len(), 4); - assert_eq!(window_beats(&lines(&capture), &started), 4); - - let mut capture: String = started[..5].iter().map(|s| format!("{s}\n")).collect(); - for b in &beats { - capture.push_str(b); - capture.push('\n'); - } - capture.push_str("===READY===\n"); - assert_eq!( - settle(&lines(&capture), &started), - Err(Refused::Unsettled { settled: 0, beats: 6 }) - ); - } - - #[test] - fn fewer_settled_beats_than_the_minimum_is_not_a_verdict() { - let beats: Vec = (1..=3).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - let capture = settled_capture(&beats); - assert_eq!(window_beats(&lines(&capture), &started()), 3); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::Unsettled { settled: 3, beats: 4 }) - ); - } - - /// What `CAPTURE_BEATS` buys over the floor: cut at `MIN_SETTLED` the - /// capture ends on cpu5's first absence, which is a blip and names nobody; - /// taken to `CAPTURE_BEATS` the second absence lands inside it. - #[test] - fn the_capture_carries_beats_past_the_floor_for_the_convicting_one() { - let mut beats: Vec = - (1..MIN_SETTLED as u64).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.push(beat(2000 + MIN_SETTLED as u64 * 250, 0xdf, 43, 250)); - let cut = settled_capture(&beats); - assert_eq!(window_beats(&lines(&cut), &started()), MIN_SETTLED); - assert_eq!(settle(&lines(&cut), &started()).unwrap().blips, 1); - - beats.extend( - (MIN_SETTLED as u64 + 1..=CAPTURE_BEATS as u64) - .map(|i| beat(2000 + i * 250, 0xdf, 43, 250)), - ); - let whole = settled_capture(&beats); - assert_eq!(window_beats(&lines(&whole), &started()), CAPTURE_BEATS); - assert_eq!( - settle(&lines(&whole), &started()), - Err(Refused::CpuMissing { - cpus: vec![5], - settled: CAPTURE_BEATS, - opened: started().len() + 1 - }) - ); - } - - #[test] - fn a_beat_with_an_unreadable_field_is_refused_by_line() { - let torn = "[kernel 2.5 cpu2] heartbeat: t=2.500s alive=8/8 mask=0xff ran=43 gap=0.2"; - let mut capture = settled_capture(&[]); - capture.push_str(torn); - capture.push('\n'); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::Unreadable(torn.to_string())) - ); - assert_eq!(millis("0.251s"), Some(251)); - assert_eq!(millis("12.000s"), Some(12_000)); - assert_eq!(millis("0.25s"), None); - assert_eq!(millis("0.251"), None); - } - - /// The `mask=` is 64 bits, so an `alive=` denominator at 64 or above is no - /// reading of it and neither is zero — and the refusal is the contract's, - /// not a shift overflow's. - #[test] - fn a_cpu_count_the_mask_cannot_carry_is_unreadable() { - for alive in ["8/64", "8/100", "8/4294967296", "0/0"] { - let wide = format!( - "[kernel 2.5 cpu2] heartbeat: t=2.500s alive={alive} mask=0xff ran=43 gap=0.250s" - ); - let mut capture = settled_capture(&[]); - capture.push_str(&wide); - capture.push('\n'); - assert_eq!( - settle(&lines(&capture), &started()), - Err(Refused::Unreadable(wide.clone())), - "{alive}" - ); - } - } - - /// The bound above is reached on its own, not stood in for by the - /// differing-`cpus` refusal: every beat here reads the same wide - /// `alive=8/64`, so `cpus` never differs from `beats[0].cpus` and that - /// refusal cannot fire. Without the bound `Beat::parse` would read `64` - /// and the eight bits `mask=0xff` never sets would convict cpus 8..64 - /// instead of refusing the capture as unreadable. - #[test] - fn a_cpu_count_the_mask_cannot_carry_is_unreadable_even_when_every_beat_agrees() { - let wide = |t_ms: u64| { - format!( - "[kernel {0}.{1:03} cpu2] heartbeat: t={0}.{1:03}s alive=8/64 mask=0xff ran=40 \ - gap=0.250s", - t_ms / 1000, - t_ms % 1000, - ) - }; - let mut capture: String = started().iter().map(|s| format!("{s}\n")).collect(); - for i in 0..=CAPTURE_BEATS as u64 { - capture.push_str(&wide(2000 + i * 250)); - capture.push('\n'); - } - let lines = lines(&capture); - let head = lines.iter().position(|l| l.contains("heartbeat: t=")).unwrap(); - assert_eq!(settle(&lines, &started()), Err(Refused::Unreadable(lines[head].to_string()))); - } - - /// One capture is one machine: a beat whose `alive=` denominator is not the - /// first's describes a different one, and a mask read against the wrong - /// width is a CPU invented or a CPU dropped. - #[test] - fn a_capture_whose_cpu_count_changes_is_unreadable() { - let odd = "[kernel 3.0 cpu2] heartbeat: t=3.000s alive=7/7 mask=0x7f ran=43 gap=0.250s"; - let mut beats: Vec = (1..=2).map(|i| beat(2000 + i * 250, 0xff, 43, 250)).collect(); - beats.push(odd.to_string()); - beats.extend((1..=4).map(|i| beat(3000 + i * 250, 0xff, 43, 250))); - let capture = settled_capture(&beats); - assert_eq!(settle(&lines(&capture), &started()), Err(Refused::Unreadable(odd.to_string()))); - } - - /// The one table, held against the config it transcribes: a `[boot] start` - /// program it does not know is refused by name, here and not at the guest. - #[test] - fn a_program_the_done_table_does_not_know_is_refused() { - let known: Vec = DONE.iter().map(|(program, _)| (*program).to_string()).collect(); - assert_eq!(done_lines(&known).unwrap(), started()); - - let mut added = known.clone(); - added.push("sniffer".to_string()); - assert!(done_lines(&added).unwrap_err().contains("sniffer")); - - let mut dropped = known; - dropped.pop(); - assert!(done_lines(&dropped).is_err()); - } -} diff --git a/src/lib.rs b/src/lib.rs index b10a2a5f703..2ba408bdda0 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -15,7 +15,6 @@ pub mod fingerprint; pub mod firmware; pub mod flags; pub mod forkcheck; -pub mod heartbeat; pub mod hostws; pub mod icmp; pub mod identity; diff --git a/src/metal.rs b/src/metal.rs index e1f7d6d4bd7..b71edc5cd51 100644 --- a/src/metal.rs +++ b/src/metal.rs @@ -803,6 +803,15 @@ pub const FLASHABLE: &[(&str, Flash)] = &[ // hold, reaches no firmware state, and the worst // it leaves is a stick a replug clears — the defect the arm exists to stage. ("usb-transport-break", Flash::Ok), + // It deafens one CPU for a window of its own clock and has the blocked-task + // dump kick it and probe it with an NMI. It reaches no device register and + // writes no firmware state; the CPU rejoins, and the boot goes on to + // userland and ends the way an unarmed one does. + ("dump-deaf-cpu", Flash::Ok), + // It holds each pipe waiter up to a short budget of its own clock for a post + // to land between its condition and its park. It reaches no device and + // writes no firmware state, and a hold nothing posts into lapses. + ("watch-window", Flash::Ok), ( "quiesce-late-word", Flash::Never( diff --git a/src/redlist.rs b/src/redlist.rs index 5f251f369f7..1b6eff2b320 100644 --- a/src/redlist.rs +++ b/src/redlist.rs @@ -30,7 +30,6 @@ pub const DISABLED: &[Disabled] = &[ issue: "issues/build/the-console-input-path-can-stop-after-a-ps2-overflow.md", }, Disabled { test: "desktop_window_child", issue: "issues/kernel/desktop-window-child-freeze.md" }, - Disabled { test: "doom_sound_flood", issue: "issues/audio/doom-sound-flood-played-full-scale-once.md" }, Disabled { test: "ftruncate_flush_race", issue: "issues/build/ftruncate-flush-race-reds-intermittently-and-nothing-says-why.md", @@ -41,13 +40,11 @@ pub const DISABLED: &[Disabled] = &[ issue: "issues/kernel/handle-kill-policy-census-grew-one-sharedmem-on-two-nightlies.md", }, Disabled { test: "handle_transfer", issue: "issues/kernel/deferred-release-outlives-its-syscall.md" }, - Disabled { test: "hda_tone", issue: "issues/audio/hda-tone-phase-check.md" }, Disabled { test: "i8042_mouse", issue: "issues/hardware/i8042-mouse-ends-four-packets-short-with-a-clean-exit.md", }, Disabled { test: "kill_while_blocked", issue: "issues/kernel/deferred-release-outlives-its-syscall.md" }, - Disabled { test: "latency_wake", issue: "issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md" }, Disabled { test: "log_ring_keeps_the_owners_slots", issue: "issues/kernel/a-log-rings-owner-is-named-only-when-logd-reads-its-registration.md", @@ -80,10 +77,6 @@ pub const DISABLED: &[Disabled] = &[ test: "root_chunk_refused_on_a_usb_stick", issue: "issues/boot-media/an-unreadable-sector-on-a-usb-boot-stick-hangs-the-loader-past-the-firmware-watchdog.md", }, - Disabled { - test: "sched_check_build", - issue: "issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md", - }, Disabled { test: "screen_fatal_halt", issue: "issues/boot-media/screen-fatal-halt-reds-on-ci-with-a-usb-storage-transport-break-during-boot.md", diff --git a/src/testargs.rs b/src/testargs.rs index fd11308331e..4125fe188e0 100644 --- a/src/testargs.rs +++ b/src/testargs.rs @@ -42,8 +42,8 @@ impl Shard { /// property a verdict depends on and the one the gates below hold. /// /// **`load` is the run's one accumulator, not this call's.** A suite that - /// partitions several pools — the parallel tasks, the serial tail, gate A's - /// configs — is one machine's wall clock either way, so the second pool has + /// partitions several pools — the parallel tasks and the serial tail — is one + /// machine's wall clock either way, so the second pool has /// to fill the bins the first left light. Starting each call from /// [`bins`](Self::bins) makes each partition good and their sum bad, and /// the imbalances add: measured over run `31377439504`'s twelve shards it @@ -149,11 +149,9 @@ declare_flags!(pub SUITE = { pub LIST = "--list", None; pub NOCAPTURE = "--nocapture", None; pub SHOW_OUTPUT = "--show-output", None; - pub AUDIO_GATE = "--audio-gate", Next; pub JOBS = "--jobs", Next; pub JOBS_SHORT = "-j", Next; pub SHARD = "--shard", Next; - pub SLOW_USB = "--slow-usb", None; pub NIGHTLY = "--nightly", None; pub WEEKLY = "--weekly", None; /// The metal profile: the registrations that run on the T14, batched into @@ -212,13 +210,6 @@ pub fn parse(args: &[String]) -> Result, String> { .to_string(), ); } - if has(&METAL) && has(&AUDIO_GATE) { - return Err( - "--metal and --audio-gate are separate tiers on separate machines and cannot be \ - combined; run one at a time" - .to_string(), - ); - } if has(&NIGHTLY) && has(&WEEKLY) { return Err( "--weekly runs the nightly tier too, so a --nightly beside it would be read by \ @@ -227,13 +218,11 @@ pub fn parse(args: &[String]) -> Result, String> { ); } for reach in [&NIGHTLY, &WEEKLY] { - for other in [&AUDIO_GATE, &METAL] { - if has(reach) && has(other) { - return Err(format!( - "{} and {} are separate tiers and cannot be combined; run one tier at a time", - reach.name, other.name - )); - } + if has(reach) && has(&METAL) { + return Err(format!( + "{} and {} are separate tiers and cannot be combined; run one tier at a time", + reach.name, METAL.name + )); } } Ok(filter) @@ -267,20 +256,15 @@ mod tests { fn a_flags_value_is_not_the_filter() { assert_eq!(parse_owned(&["--jobs", "4"]).unwrap(), None); assert_eq!(parse_owned(&["-j", "4"]).unwrap(), None); - assert_eq!(parse_owned(&["--audio-gate", "30"]).unwrap(), None); } #[test] - fn a_reach_and_another_tier_are_refused_by_the_argv_validator() { - for (argv, reach, other) in [ - (vec!["--nightly", "--audio-gate", "30"], "--nightly", "--audio-gate"), - (vec!["--audio-gate=30", "--nightly"], "--nightly", "--audio-gate"), - (vec!["--weekly", "--audio-gate", "30"], "--weekly", "--audio-gate"), - (vec!["--metal", "--nightly"], "--nightly", "--metal"), - (vec!["--weekly", "--metal"], "--weekly", "--metal"), - ] { + fn a_reach_and_the_metal_tier_are_refused_by_the_argv_validator() { + for (argv, reach) in + [(vec!["--metal", "--nightly"], "--nightly"), (vec!["--weekly", "--metal"], "--weekly")] + { let refusal = parse_owned(&argv).unwrap_err(); - assert!(refusal.contains(reach) && refusal.contains(other), "{argv:?}: {refusal}"); + assert!(refusal.contains(reach) && refusal.contains("--metal"), "{argv:?}: {refusal}"); assert!(refusal.contains("cannot be combined"), "{argv:?}: {refusal}"); } } @@ -296,7 +280,7 @@ mod tests { fn the_filter_is_the_word_that_is_nobodys_value() { assert_eq!(parse_owned(&["process_stats"]).unwrap().as_deref(), Some("process_stats")); assert_eq!( - parse_owned(&["--audio-gate", "30", "audio_tone", "--nocapture"]).unwrap().as_deref(), + parse_owned(&["--jobs", "4", "audio_tone", "--nocapture"]).unwrap().as_deref(), Some("audio_tone") ); assert_eq!( @@ -382,9 +366,9 @@ mod tests { assert_eq!(totals, vec![130, 130, 130, 130], "{totals:?}"); } - /// **One run is one accumulator.** The suite partitions three pools — the - /// parallel tasks, the serial tail, gate A's configs — and a shard runs all - /// three, so the second call has to fill the bins the first left light. Two + /// **One run is one accumulator.** The suite partitions two pools — the + /// parallel tasks and the serial tail — and a shard runs both, so the second + /// call has to fill the bins the first left light. Two /// pools of `[3 s, 1 s]` across two shards is the smallest case that tells /// the two apart: threaded, both shards take 4 s; from a fresh accumulator /// each time, the heavy item lands on shard 1 twice and the widest bin is @@ -449,7 +433,7 @@ mod tests { } /// Every `None` here is a default the run then takes in silence: `--jobs` - /// the built-in width, `--audio-gate` the thorough tier off. + /// the built-in width. #[test] fn a_flag_left_without_its_value_is_refused_by_name() { for flag in SUITE.0.iter().filter(|f| !matches!(f.value, Value::None | Value::Optional)) { @@ -501,7 +485,6 @@ mod tests { vec!["process_stats"], vec!["process_stats", "--nocapture"], vec!["--list"], - vec!["--audio-gate", "30"], vec!["--jobs", "4"], vec!["--shard", "2/4"], vec!["--nightly"], @@ -522,7 +505,5 @@ mod tests { fn a_readback_directory_alone_selects_no_tier() { let refusal = parse_owned(&["--metal-readback", "target/metal"]).unwrap_err(); assert!(refusal.contains("add --metal"), "{refusal}"); - let refusal = parse_owned(&["--metal", "--audio-gate", "30"]).unwrap_err(); - assert!(refusal.contains("cannot be combined"), "{refusal}"); } } diff --git a/src/tiers.rs b/src/tiers.rs index 93a170134c0..770d746a95c 100644 --- a/src/tiers.rs +++ b/src/tiers.rs @@ -1,10 +1,10 @@ //! Which run a registered test belongs to. //! //! **The table is the registration.** Every row of `tests/toyos.rs`'s -//! `MACHINE_TESTS`, `SCREEN_TESTS` and `AUDIO_TESTS` carries its [`Tier`] or -//! does not compile, and the shared boot's discovered tests share one. A -//! [`Schedule`] is every one of those names at its one tier. Moving a test -//! between tiers is editing that one word. +//! `MACHINE_TESTS` and `SCREEN_TESTS` carries its [`Tier`] or does not compile, +//! and the shared boot's discovered tests share one. A [`Schedule`] is every +//! one of those names at its one tier. Moving a test between tiers is editing +//! that one word. //! //! The tiers nest: a plain `cargo test` reaches `Fast`, `--nightly` adds //! `Nightly`, and `--weekly` adds `Weekly` to that. diff --git a/tests/CLAUDE.md b/tests/CLAUDE.md index 3d5d594383b..ad6ce0ebdb0 100644 --- a/tests/CLAUDE.md +++ b/tests/CLAUDE.md @@ -7,7 +7,6 @@ The mechanics live where the work is: profiles and shapes in `tests/common/`, re - **The dev host is a laptop that sleeps mid-session, and the suite says so** — a run whose wall clock jumped against the monotonic one reports `INVL` per test and exits 2: re-run. A wild outlier *not* marked that way is a real finding. - **A machine-wide kernel panic reds whichever test was running** — that red's name is the workload, never the cause. `QEMU died before ===READY=== (status 0)` is the same thing said silently: a guest that reset itself is a kernel death, and its evidence is the boot log. - **A machine-wide death during boot is reproduced by boots, not by suites** — parallel `bootable.img` guests, each waited on its completion marker and never on a fixed timer; the baseline is measured in the same session as the arm; a death counts whether or not a marker printed, so run guests with `-action reboot=shutdown -action shutdown=pause` and read a silent one's registers over QMP; a T14 block that gained a CI container mid-run is discarded, never corrected; a defect whose rate is set by interrupts per unit of guest work is measured on the *slowest* instrument — TCG can be the stronger oracle. -- **Gate A's thorough tier reds on the dev host, and the fast tier intermittently** — a plain `cargo test` boots no audio config at all; read the nightly's verdict line, not its check status. - **A C test's capture has other processes' lines removed before comparison**, on the boot config's list of who may speak (`tests/common/console.rs`); a line without a trailing newline is unjoined from the next writer's there too. - **`/system/bin/init` speaks in every program's name before that program runs** — a predicate keyed on a `: ` prefix is satisfied by the wrong speaker; wait for the whole line, in the constant the assertion also reads. - **A guest binary cannot ask what a handle it does not hold does** — the probe ends its caller with exit 139, so it runs in a child, one fault per child; `handle_kill_policy` is the pattern. diff --git a/tests/audio-baseline.toml b/tests/audio-baseline.toml deleted file mode 100644 index 69132d3d6d6..00000000000 --- a/tests/audio-baseline.toml +++ /dev/null @@ -1,495 +0,0 @@ -# Audio gate (gate A) baselines — one section per (test, smp) config. -# -# Every config must appear here. A missing section is a hard error, not a -# pass: an ungated config would go green by omission. -# -# What no config here covers: `/system/bin/tone` and tests/toyos-rust-tests/src/tone.rs -# are two implementations of the same generator and gate A captures only the -# second, so nothing in the suite ever hears the tone the owner judges the -# machine by. -# -# Two independent instruments per config. What each one is allowed to decide -# differs by tier, and TWO TIERS below is where that is settled: only silence -# that reached the device fails a single run. -# -# gaps Underrun histogram from the captured wav, keyed by gap -# length in whole device periods (2.902ms). An absent -# table is the strict zero-gap bar, which is what all four -# configs carry — the quality target is no dropouts, and -# recording a non-empty histogram here would be a decision -# to accept audible ones. -# max_wake_lat_us Worst single stats window of soundd waking later than a -# predicted DMA completion it armed a timer on. Waits that -# named no wake time — the idle path, and the one wake -# after a drain before the DLL re-locks — contribute -# nothing: nothing was due, so nothing can be late. -# **It is two delays added together, and mostly not -# soundd's.** The gate's per-run line splits it at the -# completion interrupt's own timestamp — `irq` is the -# device late, `pickup` is soundd late — and on the T14 -# under KVM `pickup` was 10-206us across 176 config-runs -# while the whole number ran 2509-6176us. -# `toyos_mixer::WorstWake` is where that split is argued; -# read it before taking a movement in this number for a -# scheduling change. -# drains Cycles that found the entire DMA pipeline free *and* -# could only have got there by soundd being late. -# The idle path empties the pipeline by design, and a -# device that retires faster than it plays empties it -# without anything having been due; neither counts. See -# the count site in soundd for the two conditions. -# underruns Periods submitted with no client audio behind them. -# -# VOID AS A TIMING SAMPLE UNDER QEMU 11.1.1. `.github/qemu-version` declares -# 11.1.1 and the sample below was measured on 11.1.0, so every comparison of a -# fresh `max_wake_lat_us` or `wakes` against it is cross-instrument and decides -# nothing, on the dev host and on CI alike. No re-record is taken for the move: -# timing verdicts belong on metal, and the dev host is not a quiet instrument. -# The harm statistics are the exception, because what they compare against is -# a sample of zeros — no dropout, no underrun — which is the quality bar on any -# instrument, so a harm red still counts. CI's `audio` shards neither re-sample -# nor skip: they compare a fresh KVM sample against this one, as they already -# did across the host and the accelerator -# (`issues/audio/gate-a-has-no-runner-baseline.md`), and now across the QEMU -# version as well; a timing red from them is read against this paragraph. -# -# RE-RECORDED 2026-08-15 on tree 4c7d809, justified by an instrument change: -# the dev host's QEMU moved 11.0.3 -> 11.1.0 (owner-approved upgrade; CI's -# runner had already moved and `.github/qemu-version` declares 11.1.0), and the -# QEMU version was measured to decide verdicts — so every comparison of a fresh -# 11.1.0 arm against the 11.0.3 sample was cross-instrument and void. The -# prior sample (2026-07-31, 17a6c88 — see git log for its full narrative, -# including the CQE fan-out A/B that justified it) described the old -# instrument. -# -# The recording was taken TWICE, back to back, 30 iterations per arm, and the -# second run is the sample below — its host-conditions line is the clean one -# (1-min load 1.2-2.1). The first run, whose load briefly spiked to 10.2, -# reproduces the second's shape, which is the control that says the shift is -# the instrument and not the spike: -# -# max_wake_lat_us medians, old -> run1 / run2: -# tone.smp1 6698 -> 9123 / 8972 UP ~34% -# tone.smp8 6658 -> 8626 / 9249 UP ~30-39% -# load.smp1 7042 -> 5813 / 5765 DOWN ~18% -# load.smp8 7134 -> 6121 / 6097 DOWN ~15% -# The no-load configs got slower and the load configs got -# faster on 11.1.0 — a reshuffle only an instrument change -# produces; a kernel regression does not speed up the -# harder case. -# harm null in BOTH runs: underruns all-zero, drains all-zero, -# dropouts 0/120 and 0/120, ceiling breaches 0/120 and -# 0/120. The quality bar holds on the new instrument. -# -# This recording also takes the `wakes` re-record the fourth drift instance -# below recorded as owed-but-not-taken: the mechanism that paragraph was -# waiting for is the instrument itself. -# -# Drift note, third instance: the BASE arm itself — zero suspend-series -# commits — failed the old sample on audio_tone.smp1 wakes (median 1011 -> -# 942, z=5.28), the third consecutive fewer-wakes-all-harm-clean failure of -# that statistic across sessions. Cross-batch drift on this host is real; -# only same-session A/B numbers mean anything (CLAUDE.md). -# -# Drift note, FOURTH instance, 2026-08-01, and this one is measured against a -# same-session control rather than argued. Two N=30 arms back to back (1026 s -# and 1024 s, quiet host), differing only in whether `esp_log::install()` runs: -# -# recorded esp_log ON esp_log OFF -# audio_tone.smp1 wakes 920 902 z=4.12 908 z=3.63 -# audio_tone.smp8 wakes 872 815 z=4.79 816 z=4.73 -# -# `wakes` reds in BOTH arms, at the same magnitude, with the suspect entirely -# removed — so it is drift, not the change under test. Harm null in both: -# dropouts 0/120 and 0/120, ceiling breaches 0/120 and 0/120, underruns and -# drains all-zero. Four consecutive sessions have now failed `wakes` downward -# with every harm measure clean, so a re-record of that statistic is owed — -# but NOT taken here, because no mechanism was identified and this file's -# protocol wants one. Read the fourth instance as evidence for the owner, not -# as licence. -# -# The one statistic that separated the arms was audio_tone.smp8 wake lateness: -# 6691 with the sink off, 7346 with it on (recorded 6658). That is the ESP -# log's residual flush cost on the idle path, not drift, and it is written up -# in `issues/boot-media/` rather than recorded here. -# -# Drift note, FIFTH instance, 2026-08-15, and the first one measured against -# the sample below rather than against a superseded one — so the instrument -# argument that closed the fourth instance does not apply to it. Two N=30 arms -# back to back (904 s and 907 s), differing only in the four soundd commits of -# `wt/toyos-soundfix`; BASE is `a0729cf`, the tree this sample was recorded on: -# -# recorded soundfix BASE a0729cf -# audio_tone.smp1 wakes 882 870 z=3.47 870 z=3.14 -# audio_tone.smp8 wakes 877 849 z=5.29 854 z=5.44 -# audio_tone_load.smp1 wake_lat 5765 6051 pass 6151 z=3.54 -# -# `wakes` reds downward in BOTH arms again, and the BASE arm reds one statistic -# MORE than the branch under test. Harm null in both: underruns 0/0/0 in all -# eight config-arms, dropouts 0/120 and 0/120, ceiling breaches 0/120 and -# 0/120, drains all-zero except a single run in BASE's audio_tone_load.smp8. -# That is five consecutive sessions of fewer-wakes-all-harm-clean, two of them -# now with a same-session control. -# -# What is new here is a candidate mechanism, which is what this file's protocol -# has been waiting for and what the fourth instance could not offer. The -# conditions differ in the one figure that is exact rather than lagging: -# -# recorded qemu 3-3, toyos-build 1-1, 1-min load 1.2-2.1 (median 1.6) -# soundfix qemu 1-1, toyos-build 1-2, 1-min load 1.0-8.6 (median 2.3) -# BASE qemu 3-3, toyos-build 2-2, 1-min load 0.9-2.3 (median 1.6) -# -# `wakes` counts iterations of a loop woken by completions and by a predicted -# timeout, and the pipeline retires in batches of 8 on this host in every run of -# all three: fewer wakes for the same ~1110 periods submitted is the host -# handing soundd more completions per wake. A guest sharing the host with two -# other guests is descheduled more often, which is more wakes, not fewer — so -# the *recorded* arm being the contended one is the direction that matches. -# Not established: the BASE arm above ran at `qemu 3-3` like the recording and -# still red, so competition alone does not close it either. Recorded as the -# next thing to test, not as the answer. -# -# NO RE-RECORD IS TAKEN HERE. Not by this branch and not for this statistic: -# the owner's ruling and this file both want a mechanism first, and a candidate -# is not one. -# -# WHAT THIS SAMPLE IS A SAMPLE OF, measured 2026-08-21 rather than argued. Every -# number below was taken on the dev host under cross-arch TCG, and the thorough -# tier compares a KVM runner's sample against it — two instruments, whichever -# runner it is. That was an argument before; the control it lacked has been run -# on the T14 (Intel i5-1135G7, KVM, the same QEMU 11.1.0) while that machine was -# still a runner. Four interleaved 15-iteration blocks on an idle T14 (A,B,A,B; no CI job -# container on the machine for any of the 240 boots; 1-min load 0.2-1.74), arm A -# = `960b96e3`, the tree this sample was recorded on, arm B = `53101d08`: -# -# max_wake_lat_us median recorded T14 arm A T14 arm B A vs B, n=30 -# audio_tone.smp1 8972 20314 19994 z=1.49 same -# audio_tone.smp8 9249 14069 4088 z=5.37 B faster -# audio_tone_load.smp1 5765 2764 4352 z=4.12 B slower -# audio_tone_load.smp8 6097 12676 3904 z=5.48 B faster -# -# **Arm A fails this file on the T14 and arm B does not.** Both of arm A's -# blocks red `audio_tone.smp1` against the sample below (median 8972 -> 20438 -# z=5.03, and -> 20144 z=4.67); both of arm B's pass. So the tree the sample was -# recorded on reds its own sample on that host, harder than `main` does — which -# is the negative control saying the level difference is the instrument and not -# a regression. The first readable T14 gate A run (32479089989) failed -# `audio_tone.smp1` at 8972 -> 17186, z=4.36; that verdict is this. -# -# Harm was null on both arms: dropouts 0/120 and 0/120, underruns 0 in all 240 -# config-runs, ceiling breaches 1/120 (arm A) and 0/120 (arm B). -# -# NO RE-RECORD IS TAKEN FOR THAT EITHER, and this sample must not be replaced by -# a T14 one: it is the dev host's, and the dev host still runs the fast tier. -# What a per-host baseline needs before it can be written — a schema that has a -# host dimension at all, and a T14 distribution that stops being bimodal — is in -# `issues/audio/gate-a-has-no-runner-baseline.md`. -# -# Why the counters are here at all: the wav is a rare-event detector. One 3s -# tone samples ~1100 device periods once per run, so it only fires when a -# dropout happens to land inside the tone — a 0-10% per-run event that cannot -# resolve a change in the failure rate. soundd's counters are non-zero on -# essentially every run and are direct evidence of the same stall, so they are -# the half of this gate with statistical power. They are scoped to the -# streaming phase: soundd zeroes them when the first client arrives and -# flushes them when the last one leaves, so no number here is diluted by the -# idle path (which since the suspend-on-idle series is a stopped device and a -# parked soundd — no wakes at all). -# -# THE PHYSICAL SCALE. The DMA pipeline holds TX_INFLIGHT_MAX = 8 buffers of -# 128 frames at 44.1kHz = 23.219ms. That is soundd's entire timing budget: -# wake later than one pipeline depth and every buffer has already drained and -# the device has run out of audio. `max_wake_lat_us` limits are therefore -# quantised to whole pipeline depths (1 pl = 23219us), so the number says -# something about the hardware rather than about a particular afternoon. -# -# HOW THE NUMBERS WERE PICKED. The ceilings were derived at the 2026-07-29 -# recording (30 serial runs per config, quiet host: one QEMU at a time, -# verified before every iteration; Apple M4 Pro, QEMU 11.0.2 under cross-arch -# TCG, 1-min load 4.2-6.1 recorded per run) and deliberately KEPT at the -# 2026-07-31 re-record: they are catastrophe detectors, the new maxima are -# lower everywhere, and lowering a ceiling to track an improving sample would -# turn the catastrophe detector into a second distributional test. They now -# sit 5.5-22.5x above the observed maxima. Derivation rule, applied uniformly: -# -# limit = 2x the observed maximum, rounded up to a round number -# -# These are max-of-window order statistics with heavy right tails. At n=30 the -# observed maximum sits near the 97th percentile, so a limit AT the maximum -# would false-red on the order of once every 30 runs per config — four configs -# per invocation would make the gate red more often than the defect it is -# watching. Doubling it puts the per-config false-red rate well under 1% while -# still failing on a doubling of the tail, which is the magnitude a scheduler -# regression that broke RT preemption or the audio IRQ path would produce. -# Since 2026-08-04 what a breach can redden is the thorough tier's pooled -# `ceiling_runs` rate rather than a `cargo test`, so this is now the false-red -# argument for the recorded 0/120 that rate is compared against. The derivation -# is unchanged and so is every number below. -# -# The tails were not physics. The previous sample's worst wake lateness on -# `audio_tone_load.smp1` went 26627us at n=15 to 92989us at n=30 — one more run -# more than tripled it — and that is why the rule has a doubling factor at all. -# It is kept, but the thing it was compensating for is gone: two instrument -# faults (824dd7d scoring idle sleeps that armed no timer against a stale DLL -# estimate; 7095046 counting connect-time pre-roll as drains) and one real -# defect (069d158). The same config now spans 6057-8250us, a ratio of 1.36 -# across 30 runs. Keep the factor anyway — it costs nothing and the next tail -# will not announce itself either. -# -# WHAT THESE NUMBERS ADMIT. The spec-derived bar — never wake later than the -# pipeline you are feeding — is met on all 120 config-runs. The medians span -# 5765-9249us across the four configs (a quarter to two-fifths of a pipeline -# depth) and the worst single run reaches 0.98 pl (22744us, on -# `audio_tone.smp1`) — under a whole pipeline depth, but barely, and the -# ceilings stay where they are so a tail that grows says so. `ceiling_runs` -# is 0 everywhere. The 54/57ms suspected-suspend tail pair did not reappear -# in either 11.1.0 run. -# -# The three paragraphs this replaced said the bar was NOT met on any config, -# blamed "mode-B client misses under single-CPU load", and described -# `underruns` as having a floor of ~4 from a stream-start transient. All three -# were wrong, and each was wrong the same way: a defective counter was being -# read as a property of the system. `underruns` is now 0 on all 120 runs. If a -# statement in this file ever again explains why a counter is *expected* to be -# non-zero, suspect the counter first. -# -# Re-record deliberately, never casually, and never to make a red run green. -# Every number above must survive being asked "why that value?". -# -# =========================================================================== -# TWO TIERS, AND WHAT EACH ONE CERTIFIES -# =========================================================================== -# -# FAST TIER — `cargo test`, one boot per config, every time. -# Certifies: the instrument is alive (tone present, dither present, no clicks, -# a stats window that exists at all), no counter sits on the wrong side of a -# physical bound, and this build does not *reproducibly* put silence on the -# wire. -# -# THE VERDICT IS HARM, and harm is silence that reached the device: a -# mid-tone gap in the capture, or a period soundd submitted with no client -# audio behind it (`underruns`). The ceilings above are measured on every run, -# printed with the run's counters, and fail nothing here. `drains` past its -# ceiling with an empty histogram and zero underruns is a pipeline that -# recovered before anyone could hear it, and one boot cannot say whether it -# recovers less often than it used to — that is a rate, and the tier below is -# the instrument for it. Owner's ruling, 2026-08-04. -# -# Harm is confirmed before it fails: a run showing any is re-booted once and -# only a second failure counts. The strict zero-gap bar is unchanged and -# applies to both boots — one Bernoulli trial against a 0-7% per-config rate -# is not a verdict, and the un-confirmed version measured 12.8% red per -# invocation on an unmodified tree. The 2-of-2 rule takes that to 0.67%. The -# first occurrence is still printed and its wav still kept. -# -# The direction this moved in is not "looser". `underruns` was previously -# judged against the ceiling above (12-70 depending on config), so 40 periods -# of silence on the wire passed; it is now judged against zero, which is what -# all 120 recorded runs measured. What stopped failing a single run is the -# counters that describe timing rather than output. -# What it CANNOT certify: any statement about a rate. One run is one sample. -# -# THOROUGH TIER — `cargo test --test toyos-build -- --audio-gate 30`, ~17 minutes. -# What the nightly runs. -# N iterations of all four configs; every per-run outcome becomes a rate or a -# distribution, and the fresh sample is compared against the RECORDED SAMPLE -# below — Mann-Whitney for the counters, Fisher exact for the yes/no -# outcomes. Not against a fitted constant: a threshold derived from 30 runs -# carries the sampling error of those 30 runs, and a one-sample test against -# it claims a confidence it does not have (a sign test against the recorded -# median of `max_wake_lat_us` has a nominal false-red rate of 0.07% and a -# real one near 1.5%, purely because the reference median moves by one order -# statistic). -# -# At N=30, measured against these distributions, it detects: -# wake lateness +25% 99.9% +20% 93% +10% 4% (missed) -# underruns +50% 100% +25% 94% -# wakes -5% 99.9% (completions batched = soundd ran late) -# dropout rate 10x 100% 5x 71% 3x 5% (missed) -# False-red on a clean tree: 0.25%, over 2000 invocations simulated from -# these samples. -# -# Read the dropout row honestly: the audible symptom is the WEAKEST -# instrument here and cannot be made strong. Separating a 3% dropout rate -# from a 7% one at this confidence needs ~600 runs per config — five hours -# per config. It stays in the gate because it is the only statistic that -# says "someone would have heard it"; the counters are what actually decide -# whether a stage regressed. -# -# WHY FIXED N AND NOT A SEQUENTIAL TEST. An SPRT on the dropout rate would -# accept a clean tree after ~17 iterations instead of 30, saving about a -# third of the wall clock. Two reasons not to: the run is also the artefact -# you re-baseline from, and a variable-length sample is not comparable to the -# one recorded here; and the accept rule would have to be combined across all -# 18 statistics, which is where the honesty of the alpha would go to die. -# What is taken from the sequential idea is the cheap half — fail-side -# curtailment. A count can only rise, so once one passes the threshold it -# would face at the full N, the verdict is already decided and the run stops. -# A badly broken tree fails in minutes; a clean one pays the full 17. -# -# ALPHA IS 0.001 PER TEST, and it is a constant in `tests/common/stats.rs`, -# not a field here. A gate whose alpha can be raised is a gate that will be -# raised on the afternoon it goes red. -# -# BOTH TIERS — the physical bound (`audio::check_physical`). Every number above -# answers "did this get worse?". Nothing here answered "is this possible?", -# and the two are different questions. Stage 6's thorough tier PASSED while -# carrying `wake_lat 153766519us` — 153 seconds inside a test the harness -# kills at 30 — because a rank test is robust to exactly one absurd value, -# and it would then have been pasted back into `max_wake_lat_us` below as -# data. The bound is the wall-clock life of the QEMU process, measured by the -# harness: soundd's whole life is inside it, so no duration it reports can -# exceed it, and no count of device periods can exceed the periods that fit -# in it. Nothing is recorded here to be tuned. A violation is reported as a -# broken instrument — fatal in both tiers, and in the thorough tier it aborts -# before the value joins the sample. It is NOT a ceiling: it sits two orders -# of magnitude above the ones above, because a ceiling admits values that are -# bad but real, and a stall of even three seconds should still be reported as -# the regression it is. -# -# =========================================================================== -# THE CONDITIONS A VERDICT WAS TAKEN UNDER -# =========================================================================== -# -# Every run of either tier prints the host's concurrent load beside its -# counters — `host: load <1min>/<5min>/<15min> qemu N toyos-build N`, sampled -# once per boot — and the thorough tier prints the range over its whole sample -# beside the numbers a re-record is pasted from. NOTHING BRANCHES ON ANY OF IT. -# CLAUDE.md's 2026-08-04 ruling stands unchanged: load is not an excuse, and a -# load-coincident red is investigated as a real defect of the pipeline, never -# re-run away as noise. -# -# What moved is the premise's context, not the ruling. It was made when the only -# load was the audio test itself; the tree now runs several agents in separate -# worktrees at once. The load that prompted this was measured at 49.9 on this -# 14-core host with twelve rustc/cargo processes and a single guest live; while -# this was written the same host read 6.6/8.1/14.3. A verdict taken there and -# one taken quiet are different measurements, and nothing recorded which was -# which. -# -# The three readings answer different questions. The load average lags — no -# figure of it resolves one ~15 s boot — but the competition is other worktrees' -# builds, which last minutes, and the triple's shape says whether the host was -# ramping up or winding down. `qemu` and `toyos-build` are exact and -# instantaneous, and they name the competition as ToyOS work; the harness's own -# knowledge says nothing here, because gate A already asserts -# `live_instances() == 0` and runs at width 1, so every intra-run fact is a -# constant. The run's own guest is up when the sample is taken, so -# `qemu 1 toyos-build 1` is the quiet reading. -# -# WHAT THE RECORDED SAMPLE'S OWN CONDITIONS ARE — and the two recordings this -# file rests on are not symmetric: -# -# 2026-07-29, the ceiling derivation: "1-min load 4.2-6.1" recorded per run. -# A real measurement, and the reason the 1-minute figure leads the triple: a -# fresh reading compares directly against it. -# -# 2026-08-15, THE SAMPLE BELOW — the one both tiers compare against — closes -# the asymmetry the previous recording left: its conditions are MEASURED, -# the line the gate itself printed over the recording run: -# -# host conditions over 120 runs: 1-min load 1.2-2.1 (median 1.6), -# qemu 3-3, toyos-build 1-1 -# -# So both arms of every future thorough-tier comparison rest on measured -# conditions, which is what the previous version of this paragraph asked the -# next re-record to deliver. -# -# =========================================================================== -# THE RECORDED SAMPLE -# =========================================================================== -# -# 30 serial invocations of the audio suite, tree 4c7d809, QEMU 11.1.0, the -# second of the two back-to-back recordings described at the top of this -# file; measured conditions: 1-min load 1.2-2.1 (median 1.6), qemu 3-3, -# toyos-build 1-1, no concurrent agents. -# -# Zero parse casualties this pass: all 120 config-runs produced a full stats -# line and a gap histogram (`gap_sample` = 30 everywhere), every capture -# analyzed `gaps: none`. -# -# `ceiling_runs` is 0 everywhere: no run in 120 breached the per-run limits -# above. That is the intended relationship between the tiers — the ceilings are -# a catastrophe detector, the distributions are the sensitive instrument. - -[audio_tone.smp1] -# wake_lat median 8972us, max 22744us (0.98 pl); drains 0; underruns 0. -# Dropouts: 0 of 30. The worst run sits just under one pipeline depth — the -# widest tail of the four configs on 11.1.0, and both recording runs put it -# here (run 1's max was 14953). Ceiling kept at the 2026-07-29 derivation, -# now 2.46x the observed maximum — still above the 2x rule. -max_wake_lat_us = 56000 # 2.41 pl -drains = 8 -underruns = 40 - -[audio_tone.smp1.sample] -gap_sample = 30 -gap_runs = 0 -ceiling_runs = 0 -max_wake_lat_us = [7052, 7362, 7600, 7619, 7852, 8186, 8192, 8335, 8364, 8401, 8529, 8615, 8759, 8867, 8914, 8972, 9432, 9457, 9953, 10003, 10044, 10117, 10591, 11154, 13162, 15077, 15945, 16184, 20251, 22744] -underruns = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] -wakes = [858, 860, 868, 870, 871, 871, 873, 874, 874, 874, 875, 876, 879, 880, 882, 882, 884, 884, 885, 886, 886, 888, 889, 890, 893, 893, 895, 896, 898, 898] -drains = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] - -[audio_tone.smp8] -# wake_lat median 9249us, max 15399us (0.66 pl); drains 0; underruns 0. -# Dropouts: 0 of 30. No run's worst wake reached a pipeline depth. -# drains stays at 4 rather than dropping to the observed 0: a bar of 0 on a -# counter whose whole sample is 0 would fail on the first single event, and one -# event is not evidence of anything. The ceilings are a catastrophe detector; -# the distributional test below is what has the power here. Ceiling kept — -# 3.12x the observed maximum. -max_wake_lat_us = 48000 # 2.07 pl -drains = 4 -underruns = 12 - -[audio_tone.smp8.sample] -gap_sample = 30 -gap_runs = 0 -ceiling_runs = 0 -max_wake_lat_us = [6320, 6672, 6929, 7228, 7233, 7299, 7461, 7613, 8035, 8331, 8863, 8904, 8923, 9034, 9205, 9249, 9271, 9314, 9410, 9456, 10856, 10897, 11076, 11445, 11779, 12292, 12333, 13590, 15369, 15399] -underruns = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] -wakes = [846, 857, 859, 859, 860, 863, 863, 864, 868, 869, 870, 870, 874, 874, 877, 877, 878, 878, 879, 881, 882, 885, 889, 890, 890, 892, 892, 898, 899, 905] -drains = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] - -[audio_tone_load.smp1] -# wake_lat median 5765us, max 7064us (0.30 pl); drains 0; underruns 0. -# Dropouts: 0 of 30. No run's worst wake reached a pipeline depth. -# The first-class case — glitch-free on one CPU under load — -# stays the flattest config, and 11.1.0 made it faster: lateness spans -# 5271-7064us, a 1.34x ratio. The 186000 ceiling is now ~26.3x the observed -# maximum and is left deliberately loose: it is the catastrophe detector, and -# the distributional test on the sample below is what actually holds this -# config. -max_wake_lat_us = 186000 # 8.01 pl -drains = 26 -underruns = 70 - -[audio_tone_load.smp1.sample] -gap_sample = 30 -gap_runs = 0 -ceiling_runs = 0 -max_wake_lat_us = [5271, 5298, 5301, 5305, 5315, 5328, 5328, 5333, 5341, 5347, 5427, 5430, 5444, 5506, 5526, 5765, 5867, 5945, 5946, 5948, 6023, 6063, 6074, 6154, 6182, 6274, 6304, 6335, 6841, 7064] -underruns = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] -wakes = [856, 860, 861, 862, 863, 863, 866, 867, 868, 871, 871, 872, 872, 873, 874, 874, 875, 875, 875, 875, 875, 876, 876, 879, 880, 881, 882, 883, 887, 887] -drains = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] - -[audio_tone_load.smp8] -# wake_lat median 6097us, max 17501us (0.75 pl); drains 0; underruns 0. -# Dropouts: 0 of 30. No run's worst wake reached a pipeline depth. 29 of 30 -# runs sit in 5122-6967us (1.36x); the one outlier at 17501us has run 1's -# counterpart (15430us in that arm's load.smp1), so a rare high single wake -# under load is a property of this instrument, not of one afternoon. -# Ceiling kept — 4.57x observed. -max_wake_lat_us = 80000 # 3.45 pl -drains = 6 -underruns = 32 - -[audio_tone_load.smp8.sample] -gap_sample = 30 -gap_runs = 0 -ceiling_runs = 0 -max_wake_lat_us = [5122, 5416, 5558, 5705, 5722, 5841, 5841, 5872, 5879, 5901, 5911, 5982, 5993, 6044, 6096, 6097, 6101, 6128, 6209, 6239, 6248, 6372, 6503, 6623, 6672, 6678, 6683, 6835, 6967, 17501] -underruns = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] -wakes = [882, 883, 887, 888, 889, 893, 893, 893, 893, 894, 894, 895, 895, 895, 897, 897, 897, 898, 900, 900, 900, 900, 901, 901, 902, 902, 903, 903, 904, 904] -drains = [0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0] diff --git a/tests/checks.rs b/tests/checks.rs index bf9369983ba..09675cbed19 100644 --- a/tests/checks.rs +++ b/tests/checks.rs @@ -125,7 +125,7 @@ mod checks { Ok(()) } - /// A blown guard stays red, and stops reading as an answer. + /// A blown ceiling stays red, and is named apart from a failed assertion. /// /// Both halves, because each fails the other's way round. An implementation /// that made a stall its own non-red status would hide a guest that genuinely @@ -134,12 +134,15 @@ mod checks { /// than against the marker on its own, because a caller prefixes its own /// sentence to [`await_marker`]'s and the classification has to survive that. #[test] - fn stall_is_not_a_verdict() -> Result<(), String> { + fn a_stall_stays_red() -> Result<(), String> { // Built from the marker rather than copied, so a rename cannot leave the // gate asserting against a string nothing produces any more. let real = format!("{STALLED} waiting for the long tone to start — it went quiet"); let under_a_sentence = format!("the compositor stopped painting\n{real}"); - let cases: [(&str, Option<&str>, bool); 4] = [ + let past = qemu::GUEST_WEDGED + Duration::from_secs(1); + let backstop = qemu::ceiling_verdict(None, past, qemu::GUEST_WEDGED, Duration::from_secs(1), 900) + .ok_or("a guest talking past the backstop was given no verdict")?; + let cases: [(&str, Option<&str>, bool); 5] = [ ("an ordinary red", Some("the pointer never moved right"), false), ("a wait that expired", Some(real.as_str()), true), ( @@ -147,6 +150,7 @@ mod checks { Some(under_a_sentence.as_str()), true, ), + ("the backstop on a guest still talking", Some(backstop.as_str()), true), ("a pass", None, false), ]; for (what, reason, want_stall) in cases { @@ -193,7 +197,7 @@ mod checks { )); } let summary = tally.summary(2, Duration::from_secs(2), Duration::ZERO); - if !summary.contains("1 of those reds are blown liveness guards") { + if !summary.contains("1 of those reds are the ceiling") { return Err(format!("the summary does not separate the two kinds of red:\n{summary}")); } Ok(()) @@ -326,52 +330,6 @@ mod checks { Ok(()) } - /// [`idle_is_spinning`] against a healthy trace and a crafted one shaped like - /// the regression it exists to catch, with no guest — the same split - /// `control_regs`/`control_regs_verdict` use, and for the same reason: a - /// gate's own teeth are a claim a live boot cannot demonstrate on the - /// negative side, because nothing in this tree can stage a CPU into spinning - /// through idle on purpose. - #[test] - fn i8042_quarantine_verdict() -> Result<(), String> { - let healthy = "\ - [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ - [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ - [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=3\n\ - [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n"; - if let Some((cpu, delta)) = idle_is_spinning(healthy) { - return Err(format!("a healthy trace was refused: cpu{cpu} moved by {delta}")); - } - - // The regression's own shape: one CPU quarantines cleanly and stays - // quiet, the other's undrained ring never lets it halt. - let spinning = "\ - [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ - [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=4\n\ - [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n\ - [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=2685004\n"; - match idle_is_spinning(spinning) { - Some((1, delta)) if delta > MAX_IDLE_TRIP_DELTA => {} - Some((cpu, delta)) => { - return Err(format!("refused the wrong CPU or by the wrong margin: cpu{cpu} delta {delta}")) - } - None => return Err("a spinning CPU's trace was accepted".to_string()), - } - - // And the line the old, count-of-lines check would have been fooled by: - // the same number of `sched: cpu=` lines either way, because the print - // itself is rate-limited regardless of what is underneath it — which is - // exactly the vacuity this replaces. - assert_eq!( - healthy.matches("sched: cpu=").count(), - spinning.matches("sched: cpu=").count(), - "the crafted traces must differ only in trips=, not in line count — otherwise this proves nothing about the old check's blindness" - ); - - eprintln!(" [i8042] the idle-trip verdict accepts a healthy trace and refuses a spinning one"); - Ok(()) - } - /// What the declaration itself has to be, before any of it means anything. /// Which shared-boot binaries need `SYS_DEBUG`, asked of their source. /// @@ -650,4 +608,9 @@ mod checks { fn screen_decoder() { screen::self_test(); } + + #[test] + fn metal_audio_judges() -> Result<(), String> { + audio::judges_verdict() + } } diff --git a/tests/common/audio.rs b/tests/common/audio.rs index a3bd2a69bf9..40f62b94a54 100644 --- a/tests/common/audio.rs +++ b/tests/common/audio.rs @@ -1,715 +1,153 @@ -//! Wav capture parsing and glitch analysis for the audio integration tests. -//! -//! The QEMU wav audiodev records a continuous timeline of what the -//! virtio-sound device played. Underruns show up as stretches of digital -//! silence inside an otherwise active signal; clicks show up as -//! sample-to-sample jumps no band-limited signal could produce. -//! -//! The capture timeline is NOT wall clock. QEMU's wav backend writes only -//! while the guest voice is enabled, so the file freezes across every -//! suspended stretch and splices the next resume directly -//! onto the last stopped sample. Verified empirically: 25s of wall clock with -//! the stream stopped adds zero PCM bytes. Consequence: `analyze` reports an -//! underrun for ANY two signal regions in one capture, at ANY wall-clock gap -//! between them — the spliced silence (drain tail + resume prime) always -//! exceeds `MIN_GAP_SECS`, and `NEAR_SECS` is a proximity window into -//! adjacent *samples*, not wall time, so it can never exonerate the gap. A -//! test that plays two tones in one boot will always go red against a -//! zero-gap baseline; keep one signal region per capture. - -use std::collections::BTreeMap; -use std::fs; -use std::path::Path; -use std::time::{Duration, Instant}; - -use super::qemu::{self, BootOptions, QemuInstance}; - -/// Largest magnitude a *silent* mix can reach on the wire, in LSB. -/// -/// Derivation from soundd's dither generator (`userland/soundd/src/main.rs`): -/// `Xorshift32::next()` returns `state / 2^32 - 0.5`, i.e. a uniform draw in -/// `[-0.5, +0.5]` LSB; TPDF dither sums two independent draws, so -/// `|dither| <= 1.0` LSB exactly. A period no client covered leaves the f32 -/// mix bus at exactly `0.0`, so the sample written to the DMA buffer is -/// `round(dither)` — one of `{-1, 0, +1}`. Hence `|s| <= 1` *is* digital -/// silence, and the bound is tight: `P(|s| = 1) = 0.25`. -/// -/// Testing `s == 0` instead would be a detector that only works against a -/// truncating quantizer, which is a defect, not a property to rely -/// on: with a correct quantizer 75% of silent samples are 0, so the longest -/// run of exact zeros in 4M silent samples measures 47 — well under the -/// `MIN_GAP_SECS` floor of 88. Such a detector reports "no dropouts" forever. -/// -/// The band is far too narrow to swallow the 440 Hz test tone: at amplitude -/// 16000 the tone slews ~1000 LSB per sample through its zero crossing, so at -/// most one sample per crossing lands inside it. -const SILENCE_MAX: i32 = 1; -/// Silent runs shorter than this are ignored (the test tone dips through the -/// silence band for a single sample at each zero crossing). -const MIN_GAP_SECS: f64 = 0.002; -/// A silent run only counts as an underrun if there is signal within this -/// window on BOTH sides — i.e. it interrupts active playback. -const NEAR_SECS: f64 = 0.25; -/// Amplitude above which a sample counts as signal rather than noise floor. -const SIGNAL_THRESHOLD: i32 = 500; -/// A single-sample jump larger than this is a click: the 440Hz test tone has -/// a max per-sample delta of ~1.1k at 44.1kHz, and any sane audio is -/// band-limited far below this. -const CLICK_DELTA: i32 = 8000; -/// Device period: 512 period_bytes at 44.1kHz stereo 16-bit = 128 frames -/// = 2.902ms (`toyos_abi::virtio_sound::PERIOD_BYTES`, and the HDA stub's is the -/// same number for the same reason). Underruns are period-quantized — the device -/// plays silence one period at a time — so gap lengths are reported in whole -/// periods. -pub const PERIOD_SECS: f64 = 128.0 / 44100.0; - -pub struct Wav { - pub sample_rate: u32, - pub channels: u16, - /// Channel 0 only — soundd mixes identical data to all channels. - pub mono: Vec, -} - -pub struct SilentRun { - pub start: usize, - pub len: usize, -} - -pub struct Click { - pub index: usize, - pub from: i32, - pub to: i32, -} - -pub struct Analysis { - /// Mid-signal silent runs >= MIN_GAP_SECS: underruns. - pub underruns: Vec, - /// Hard discontinuities not at the edges of counted zero runs. - pub clicks: Vec, - /// Samples with amplitude above SIGNAL_THRESHOLD. - pub active_samples: usize, - pub peak: i32, - /// Fraction of non-zero samples in the capture's longest silent stretch — - /// the detector's own precondition, measured. TPDF dither into a - /// round-to-nearest quantizer puts 25% of silent samples at ±1; a - /// truncating quantizer puts 0% there and collapses this detector's band - /// back onto `s == 0`, at which point the gate passes while measuring - /// nothing. `None` when the capture has no silent stretch to judge. - pub dither_ratio: Option, -} - -/// Floor on `Analysis::dither_ratio`. The expected value is 0.25; anything -/// this far below it means the dither is gone, not that it got unlucky -/// (over the ~6000-sample stretches these captures contain, the sampling -/// error on 0.25 is under 0.01). -pub const MIN_DITHER_RATIO: f64 = 0.10; - -/// Parse a 16-bit PCM RIFF wav. The QEMU wav backend leaves the RIFF/data -/// size fields at 0 until clean shutdown, so sizes are advisory: a zero data -/// size means "read to EOF". -pub fn parse_wav(path: &Path) -> Result { - let bytes = fs::read(path).map_err(|e| format!("read {}: {e}", path.display()))?; - if bytes.len() < 12 || &bytes[0..4] != b"RIFF" || &bytes[8..12] != b"WAVE" { - return Err(format!("{}: not a RIFF/WAVE file", path.display())); - } - - let mut channels: Option = None; - let mut sample_rate: Option = None; - let mut data: Option<&[u8]> = None; - - let mut pos = 12; - while pos + 8 <= bytes.len() { - let id = &bytes[pos..pos + 4]; - let size = u32::from_le_bytes(bytes[pos + 4..pos + 8].try_into().unwrap()) as usize; - let body_start = pos + 8; - match id { - b"fmt " => { - let fmt = bytes - .get(body_start..body_start + 16) - .ok_or("truncated fmt chunk")?; - let audio_format = u16::from_le_bytes(fmt[0..2].try_into().unwrap()); - let bits = u16::from_le_bytes(fmt[14..16].try_into().unwrap()); - if audio_format != 1 || bits != 16 { - return Err(format!( - "unsupported wav format: audio_format={audio_format} bits={bits}" - )); - } - channels = Some(u16::from_le_bytes(fmt[2..4].try_into().unwrap())); - sample_rate = Some(u32::from_le_bytes(fmt[4..8].try_into().unwrap())); - pos = body_start + size; - } - b"data" => { - let end = if size == 0 || body_start + size > bytes.len() { - bytes.len() - } else { - body_start + size - }; - data = Some(&bytes[body_start..end]); - pos = end; - } - other => { - return Err(format!( - "unexpected wav chunk {:?} — QEMU writes only fmt+data", - String::from_utf8_lossy(other) - )); - } - } - } - - let channels = channels.ok_or("wav has no fmt chunk")?; - let sample_rate = sample_rate.ok_or("wav has no fmt chunk")?; - let data = data.ok_or("wav has no data chunk")?; - if channels == 0 || sample_rate == 0 { - return Err(format!("degenerate wav format: {channels}ch {sample_rate}Hz")); - } - - let frame_bytes = channels as usize * 2; - let mono = data - .chunks_exact(frame_bytes) - .map(|frame| i16::from_le_bytes(frame[0..2].try_into().unwrap()) as i32) - .collect(); - - Ok(Wav { - sample_rate, - channels, - mono, - }) -} - -pub fn analyze(wav: &Wav) -> Analysis { - let mono = &wav.mono; - let rate = wav.sample_rate as f64; - let min_gap = (MIN_GAP_SECS * rate) as usize; - let near = (NEAR_SECS * rate) as usize; - - let sig_runs = silent_runs(mono, min_gap); - - let has_signal = |range: &[i32]| range.iter().any(|&s| s.abs() > SIGNAL_THRESHOLD); - let underruns = sig_runs - .iter() - .filter(|run| { - let left = &mono[run.start.saturating_sub(near)..run.start]; - let end = run.start + run.len; - let right = &mono[end..(end + near).min(mono.len())]; - !left.is_empty() && !right.is_empty() && has_signal(left) && has_signal(right) - }) - .map(|run| SilentRun { - start: run.start, - len: run.len, - }) - .collect(); - - // Jumps at the edges of counted silent runs are the underruns themselves, - // not separate clicks. - let mut run_edges = std::collections::HashSet::new(); - for run in &sig_runs { - if run.start > 0 { - run_edges.insert(run.start - 1); - } - run_edges.insert(run.start + run.len - 1); - run_edges.insert(run.start + run.len); - } - let clicks = mono - .windows(2) - .enumerate() - .filter(|(i, w)| { - (w[1] - w[0]).abs() > CLICK_DELTA - && !run_edges.contains(i) - && !run_edges.contains(&(i + 1)) - }) - .map(|(i, w)| Click { - index: i, - from: w[0], - to: w[1], - }) - .collect(); - - // The capture's leading and trailing silence are the longest silent runs - // and are not underruns, so this measures the quantizer, not the glitch. - let dither_ratio = sig_runs.iter().max_by_key(|r| r.len).map(|run| { - let span = &mono[run.start..run.start + run.len]; - span.iter().filter(|&&s| s != 0).count() as f64 / span.len() as f64 - }); - - Analysis { - underruns, - clicks, - active_samples: mono.iter().filter(|s| s.abs() > SIGNAL_THRESHOLD).count(), - peak: mono.iter().map(|s| s.abs()).max().unwrap_or(0), - dither_ratio, - } -} - -/// The test tone's frequency, which the phase check below has to know and -/// `tests/toyos-rust-tests/src/tone.rs` states. -pub const TONE_HZ: f64 = 440.0; - -/// Where the captured tone stops being one sine. -/// -/// The harm this exists for: **a cyclic DMA engine replays a period nobody -/// refilled**, and a repeat is audible harm that -/// [`analyze`]'s gap detector cannot see — the samples are not silent and the -/// seam is not a large enough single-sample jump to be a click. The -/// zero-on-complete rule is what stops a repeat happening, and it is a design -/// promise; this is the measurement beside it. -/// -/// A sampled sinusoid obeys `x[n+1] = 2·cos(ω)·x[n] − x[n−1]` exactly, so one -/// pass over the capture tests the whole signal for phase continuity with no -/// transform. A replayed 128-frame period is 1.28 cycles of 440 Hz, so the -/// tone re-enters 0.28 of a cycle out and breaks the recurrence by thousands -/// of LSB. -/// -/// The tolerance covers what a correct capture does contain: TPDF dither at -/// ±1 LSB and quantization. Only the region between the first and last strong -/// sample is examined. -pub fn phase_breaks(wav: &Wav) -> Vec { - const TOLERANCE: f64 = 400.0; - let k = 2.0 * (2.0 * std::f64::consts::PI * TONE_HZ / wav.sample_rate as f64).cos(); - let mono = &wav.mono; - let (Some(first), Some(last)) = ( - mono.iter().position(|&s| s.abs() > SIGNAL_THRESHOLD), - mono.iter().rposition(|&s| s.abs() > SIGNAL_THRESHOLD), - ) else { - return Vec::new(); - }; - (first + 1..last) - .filter(|&n| { - let predicted = k * mono[n] as f64 - mono[n - 1] as f64; - (predicted - mono[n + 1] as f64).abs() > TOLERANCE - }) - .collect() -} - -/// The captured tone's pitch, in Hz, measured off the wav. -/// -/// The rate the device *plays* at is not a fact any guest counter can state: -/// soundd generates 44100 frames per second of content and the engine consumes -/// them at whatever `SDnFMT` and the codec's `Set Converter Format` agreed on, -/// so a wrong rate field is a stream that is correct in every buffer and comes -/// out at the wrong speed. Nothing else in this file can see that — -/// [`phase_breaks`] tolerates it (an 8.8% pitch error perturbs the recurrence -/// by ~12 LSB against a 400 LSB tolerance) and the gap detector is blind to it. -/// -/// Schmitt-triggered zero crossings over the strong region, which needs no -/// transform and is exact for one sine: the dither cannot cross a ±500 LSB -/// hysteresis band and the count is two per cycle. -pub fn dominant_hz(wav: &Wav) -> Option { - let mono = &wav.mono; - let first = mono.iter().position(|&s| s.abs() > SIGNAL_THRESHOLD)?; - let last = mono.iter().rposition(|&s| s.abs() > SIGNAL_THRESHOLD)?; - let mut high = mono[first] > 0; - let mut crossings = 0u32; - let (mut start, mut end) = (None, first); - for (n, &s) in mono.iter().enumerate().take(last + 1).skip(first) { - if (high && s < -SIGNAL_THRESHOLD) || (!high && s > SIGNAL_THRESHOLD) { - high = !high; - crossings += 1; - start.get_or_insert(n); - end = n; - } - } - // Between the first and last crossing, so the partial cycles at each end - // are outside the measurement rather than rounded into it. - let span = end.checked_sub(start?)?; - if crossings < 3 || span == 0 { - return None; - } - Some((crossings - 1) as f64 * wav.sample_rate as f64 / (2.0 * span as f64)) -} - -/// The complaint, when the capture did not come back at [`TONE_HZ`]. -/// -/// The band is half the distance between the two rates an HDA stream format can -/// name — 44.1 and 48 kHz are 8.8% apart — which is the coarsest error this can -/// be asked about and the widest band that still separates every pair the field -/// can express. Nothing narrower is wanted: this is a rate check, not a -/// frequency meter, and the estimator is a crossing count over a ramped tone. -pub fn wrong_pitch(wav: &Wav) -> Option { - const TOLERANCE: f64 = 0.044; - let Some(hz) = dominant_hz(wav) else { - return Some("the capture has no tone to measure the pitch of".to_string()); - }; - if (hz - TONE_HZ).abs() <= TONE_HZ * TOLERANCE { - return None; +//! soundd judged off the log the T14's stick came back with. Audio is judged on +//! metal and nowhere else: no QEMU guest test plays it. + +use super::serial::Serial; + +/// The text of `log` from the kernel's record of `job`'s spawn to the end of +/// what soundd said about the `sessions` streams it played, one after another. +/// +/// A session ends at soundd's flush on its last client leaving — the stats line +/// that carries `clients=0`, which soundd writes after the removal — and not at +/// the next job's spawn, which can land before it. A job that plays nothing +/// (`sessions` of 0) ends at the next test binary's spawn. +fn job_window<'a>(log: &'a str, job: &str, sessions: usize) -> Result<&'a str, String> { + let head = format!("spawn: /system/bin/{job} "); + let at = log.find(&head).ok_or_else(|| format!("no `{head}` record: {job} never ran"))?; + let rest = &log[at + head.len()..]; + if sessions == 0 { + return Ok(&rest[..rest.find("spawn: /system/bin/test_rs_").unwrap_or(rest.len())]); + } + let mut end = 0; + for session in 1..=sessions { + let flushed = rest[end..] + .match_indices(SESSION_ENDED) + .find(|&(line_at, _)| is_soundd_stats(&rest[end..], line_at)) + .map(|(line_at, _)| end + line_at) + .ok_or_else(|| { + format!( + "soundd never said session {session} of {job}'s {sessions} ended (a stats \ + line with `{}`):\n{rest}", + SESSION_ENDED.trim() + ) + })?; + end = rest[flushed..].find('\n').map_or(rest.len(), |nl| flushed + nl + 1); } - Some(format!( - "the capture came back at {hz:.1} Hz for a {TONE_HZ} Hz tone — the device consumed the \ - buffers at {:+.1}% of the rate soundd generated them for", - (hz / TONE_HZ - 1.0) * 100.0, - )) + Ok(&rest[..end]) } -/// Underrun histogram keyed by gap length in device periods (rounded, -/// min 1): `gaps[n]` = number of mid-signal silent runs of ~n×2.902ms. This is -/// the unit gate A's thorough tier compares against the recorded sample in -/// `tests/audio-baseline.toml`. -pub fn gap_histogram(analysis: &Analysis, sample_rate: u32) -> BTreeMap { - let mut gaps = BTreeMap::new(); - for run in &analysis.underruns { - let secs = run.len as f64 / sample_rate as f64; - let n = (secs / PERIOD_SECS).round().max(1.0) as u32; - *gaps.entry(n).or_insert(0u32) += 1; - } - gaps -} +/// What soundd's flush on its last client leaving carries, and no other stats +/// line does. +const SESSION_ENDED: &str = " clients=0 "; -/// Render a histogram as e.g. `total 3 [1p×2 4p×1]`, or `none`. -pub fn format_histogram(gaps: &BTreeMap) -> String { - if gaps.is_empty() { - return "none".to_string(); - } - let total: u32 = gaps.values().sum(); - let entries: Vec = gaps.iter().map(|(n, c)| format!("{n}p×{c}")).collect(); - format!("total {total} [{}]", entries.join(" ")) +/// Whether the match at `at` in `text` is on one of soundd's stats lines. +fn is_soundd_stats(text: &str, at: usize) -> bool { + let line_start = text[..at].rfind('\n').map_or(0, |nl| nl + 1); + text[line_start..at].contains("soundd: wakes=") } -/// No-regression gate against a recorded baseline histogram: neither the -/// total gap count nor the longest gap class may exceed the baseline. An -/// empty baseline is the strict zero-gap gate. -pub fn check_gap_regression( - measured: &BTreeMap, - baseline: &BTreeMap, -) -> Result<(), String> { - let m_total: u32 = measured.values().sum(); - let b_total: u32 = baseline.values().sum(); - if m_total > b_total { - return Err(format!( - "underrun regression: {m_total} gaps vs baseline {b_total}" - )); - } - let m_max = measured.keys().next_back().copied().unwrap_or(0); - let b_max = baseline.keys().next_back().copied().unwrap_or(0); - if m_max > b_max { +/// The tone client played to its end through the HDA controller soundd drives +/// itself, soundd never fell back to the null sink, and no period of it went +/// unfilled. +pub fn tone_on_metal(log: &Serial) -> Result<(), String> { + log.must_say("soundd: hda path configured in")?; + log.must_not_say(NULL_SINK)?; + let window = job_window(log.text(), "test_rs_audio_tone", 1)?; + let underruns = sum_field(window, "underruns"); + if underruns != 0 { return Err(format!( - "underrun regression: longest gap {m_max} periods vs baseline {b_max}" + "soundd filled {underruns} period(s) the tone client had not covered, on a client \ + that keeps its ring full:\n{window}" )); } Ok(()) } -/// The DMA pipeline depth: `TX_INFLIGHT_MAX` = 8 buffers of one device period. -/// This is soundd's entire timing budget — wake later than this and every -/// buffer has already drained, so the device has run out of audio to play. -pub const PIPELINE_DEPTH_US: u64 = (8.0 * PERIOD_SECS * 1e6) as u64; - -/// One `soundd: wakes=...` stats line. soundd emits one every 2s, but only -/// while it has clients, so every line describes streaming, not idle. -#[derive(Debug, Clone, Copy)] -pub struct SounddWindow { - pub wakes: u32, - pub completions: u32, - pub submitted: u32, - pub underruns: u32, - pub drains: u32, - pub max_wake_lat_us: u64, - pub max_batch: u32, - pub clients: u32, - /// The worst wake taken apart, as `toyos_mixer::WorstWake` describes it. - /// `worst_irq_late_us + worst_pickup_us == max_wake_lat_us`, up to the - /// truncation each half takes on its way to microseconds. - pub worst: WorstWake, - /// Wakes in this window a whole device period or more past their grid - /// point — how many stalls the maximum is the maximum *of*. - pub late_wakes: u32, -} - -/// soundd's decomposition of its worst wake, carried through the harness so a -/// per-run line can say *which* half the number is. -#[derive(Debug, Default, Clone, Copy)] -pub struct WorstWake { - pub irq_late_us: u64, - pub pickup_us: u64, - pub empty: u32, - pub batch: u32, -} - -/// Worst/total over every stats window of one run. -#[derive(Debug, Default, Clone, Copy)] -pub struct SounddCounters { - pub windows: usize, - /// Worst single-window wake lateness — the sharpest instrument here. - pub max_wake_lat_us: u64, - /// The decomposition belonging to *that* window's worst wake. Taken from - /// the window that set the maximum rather than maximised on its own: two - /// independently-worst halves describe a wake that never happened. - pub worst: WorstWake, - /// Cycles that found the whole DMA pipeline free. - pub drains: u32, - /// Periods submitted with no client audio behind them: silence that - /// actually went on the wire while a client was streaming. - pub underruns: u32, - pub submitted: u32, - pub wakes: u32, - /// Summed over the run's windows: how many wakes were a whole period or - /// more late, which is what separates one stall from a thousand. - pub late_wakes: u32, - pub max_batch: u32, -} - -/// The two numbers this boot drew for its clocks, off the kernel's own boot -/// lines: the TSC period against the HPET (`kernel/src/clock.rs`) and the LAPIC -/// timer's tick rate against that (`kernel/src/arch/x86_64/apic.rs`). -/// -/// **They are here because they are the only per-boot draws that scale every -/// armed timer for the boot's whole life**, which is the shape -/// `issues/audio/t14-wake-lateness-is-bimodal-per-boot.md` is looking for: a -/// wake latency that is one of two values, decided at boot and steady inside -/// it, cannot come from anything re-decided per wake. Both calibrations are -/// busy-wait windows on a virtual machine, so both are exactly the kind of -/// number a host that stalls the guest mid-window would move. -/// -/// Printed, never asserted: what a correct pair looks like on a given host is -/// not something this harness knows, and a threshold nobody measured is the -/// problem `tests/audio-baseline.toml` exists to avoid. -pub fn boot_clocks(boot_log: &str) -> String { - let field = |marker: &str, upto: char| { - boot_log.find(marker).map(|at| { - let rest = &boot_log[at + marker.len()..]; - let end = rest.find(upto).unwrap_or(rest.len()); - rest[..end].trim().to_string() - }) - }; - format!( - "tsc {} lapic {}", - field("TSC: ", ' ').unwrap_or_else(|| "?".into()), - field("LAPIC timer: ", '\n').unwrap_or_else(|| "?".into()), - ) -} - -/// Kernel logging shares the virtio-console with userspace and is not -/// line-atomic, so a kernel message lands wherever it lands — including -/// mid-word inside soundd's stats line, which pushes that line's tail onto -/// the next serial line. A kernel message always runs from `[kernel ` to the -/// end of its line, so deleting exactly that span splices the interrupted -/// line back together and leaves standalone kernel lines simply removed. -fn strip_kernel_logging(serial: &str) -> String { - let mut out = String::with_capacity(serial.len()); - let mut rest = serial; - while let Some(start) = rest.find("[kernel ") { - out.push_str(&rest[..start]); - rest = match rest[start..].find('\n') { - Some(nl) => &rest[start + nl + 1..], - None => "", - }; - } - out.push_str(rest); - out -} - -const STATS_MARKER: &str = "soundd: wakes="; -/// soundd's stats fields, in the order it prints them. -const STATS_KEYS: [&str; 13] = [ - "wakes", - "completions", - "submitted", - "underruns", - "drains", - "max_wake_lat_us", - "max_batch", - "clients", - // `deferred` and `starve_max` sit between these and `clients` on the wire - // and are read by nothing here; the scan is forward-only from the previous - // key, so a printed field this list omits is simply stepped over. - "worst_irq_late_us", - "worst_pickup_us", - "worst_empty", - "worst_batch", - "late_wakes", -]; - -/// Read `key=` at or after `from`, tolerating a foreign line spliced -/// in between. Any writer sharing the console can land in the middle of the -/// value — the kernel is stripped beforehand, but the tone client's own -/// `println!` does it too — and such a write always ends at a newline, so -/// when the value is interrupted it resumes on the following line. -fn stats_field(window: &str, key: &str, from: usize) -> Option<(u64, usize)> { - let pat = format!("{key}="); - let mut at = from + window[from..].find(&pat)? + pat.len(); - loop { - let digits: String = window[at..].chars().take_while(|c| c.is_ascii_digit()).collect(); - if !digits.is_empty() { - return Some((digits.parse().ok()?, at)); - } - at += window[at..].find('\n')? + 1; - } -} +/// soundd's own word for the sink it took when the machine has none. +pub const NULL_SINK: &str = "soundd: no audio device, presenting a null sink"; -/// Pull soundd's stats windows out of a serial capture. An unreadable window -/// is an error rather than a skip: silently dropping one would under-count -/// `drains` and `underruns`, which is a gate passing because it failed to -/// look. -pub fn parse_soundd_counters(serial: &str) -> Result { - let text = strip_kernel_logging(serial); - // A window's fields can be split across lines, so it extends to the next - // window marker rather than to the next newline. - let starts: Vec = text.match_indices(STATS_MARKER).map(|(i, _)| i).collect(); - let mut out = SounddCounters::default(); - for (n, &start) in starts.iter().enumerate() { - let end = starts.get(n + 1).copied().unwrap_or(text.len()); - let window = &text[start..end]; - let mut vals = [0u64; STATS_KEYS.len()]; - let mut cursor = 0; - for (i, key) in STATS_KEYS.iter().enumerate() { - let (v, at) = stats_field(window, key, cursor).ok_or_else(|| { - format!("unreadable soundd stats window (no {key}=): {window:?}") - })?; - vals[i] = v; - cursor = at; - } - let w = SounddWindow { - wakes: vals[0] as u32, - completions: vals[1] as u32, - submitted: vals[2] as u32, - underruns: vals[3] as u32, - drains: vals[4] as u32, - max_wake_lat_us: vals[5], - max_batch: vals[6] as u32, - clients: vals[7] as u32, - worst: WorstWake { - irq_late_us: vals[8], - pickup_us: vals[9], - empty: vals[10] as u32, - batch: vals[11] as u32, - }, - late_wakes: vals[12] as u32, - }; - out.windows += 1; - // The decomposition travels with the maximum it decomposes: `>=` so - // the first window still sets one, and so a later window that ties - // hands over its own halves rather than leaving stale ones behind. - if w.max_wake_lat_us >= out.max_wake_lat_us { - out.max_wake_lat_us = w.max_wake_lat_us; - out.worst = w.worst; - } - out.max_batch = out.max_batch.max(w.max_batch); - out.drains += w.drains; - out.underruns += w.underruns; - out.submitted += w.submitted; - out.wakes += w.wakes; - out.late_wakes += w.late_wakes; +/// soundd with no client costs no CPU, and the device is not what is running: +/// in a window no client connects in, soundd never started the stream. A zero +/// CPU delta is the signature of a suspended soundd and equally of one wedged +/// with the device running; the start line tells them apart. +pub fn idle_suspend_on_metal(log: &Serial) -> Result<(), String> { + let window = job_window(log.text(), "test_rs_audio_idle_suspend", 0)?; + if window.contains(DEVICE_STARTED) { + return Err(format!( + "`{DEVICE_STARTED}` with no client connected — soundd's zero CPU is the device left \ + running, not a suspend:\n{window}" + )); } - Ok(out) + Ok(()) } -/// Per-config ceilings on soundd's counters. Every number is justified in -/// `tests/audio-baseline.toml`; there are no defaults, because an unjustified -/// threshold is the same problem as an unmeasured baseline. -#[derive(Debug, Clone, Copy)] -pub struct CounterLimits { - pub max_wake_lat_us: u64, - pub drains: u32, - pub underruns: u32, -} +/// What soundd says as it starts the stream, before the first submit. +const DEVICE_STARTED: &str = "soundd: resumed"; -/// Which of this config's per-run ceilings this run sits outside, one message -/// each. Unlike the wav histogram — a rare-event detector that samples ~1000 -/// periods once per run — these counters are non-zero on nearly every run, so -/// the *rate* at which they breach can resolve a change the histogram cannot. -/// -/// A breach is not by itself audible: a pipeline that drained and recovered put -/// no silence on the wire. So this decides nothing on its own — the thorough -/// tier counts breaches and compares the rate, and the fast tier prints them -/// and judges harm. -pub fn check_counters(counters: &SounddCounters, limits: &CounterLimits) -> Vec { - let mut problems = Vec::new(); - if counters.max_wake_lat_us > limits.max_wake_lat_us { - problems.push(format!( - "wake lateness {}us > limit {}us ({:.1} vs {:.1} pipeline depths)", - counters.max_wake_lat_us, - limits.max_wake_lat_us, - counters.max_wake_lat_us as f64 / PIPELINE_DEPTH_US as f64, - limits.max_wake_lat_us as f64 / PIPELINE_DEPTH_US as f64, +/// The T14's panic, staged on its own HDA ring: a client that stops producing +/// for longer than the DMA ring takes to come round. The engine replays every +/// buffer it completes, so soundd has to fill the periods the client did not +/// cover (`underruns`) and may hold none of them back (`deferred`), across a +/// suspend and a resume. +pub fn client_stall_on_metal(log: &Serial) -> Result<(), String> { + log.must_not_say("repeated completion for free buffer")?; + let window = job_window(log.text(), "test_rs_hda_client_stall", 2)?; + let resumes = window.matches("soundd: resumed").count(); + if resumes < 2 { + return Err(format!( + "soundd resumed {resumes} time(s) — the second stream did not find a suspended \ + daemon, so nothing here tests a resume:\n{window}" )); } - if counters.drains > limits.drains { - problems.push(format!( - "pipeline drains {} > limit {}", - counters.drains, limits.drains + if !window.contains("soundd: wakes=") { + return Err(format!("soundd reported no stats window while the client ran:\n{window}")); + } + if sum_field(window, "underruns") == 0 { + return Err(format!( + "soundd filled no period the stalled client had not covered, so this boot staged \ + nothing:\n{window}" )); } - if counters.underruns > limits.underruns { - problems.push(format!( - "client underruns {} > limit {} ({} periods submitted total)", - counters.underruns, limits.underruns, counters.submitted + let deferred = sum_field(window, "deferred"); + if deferred != 0 { + return Err(format!( + "soundd deferred {deferred} period(s) on a ring that replays every one and completes \ + it again, which is the panic this exists for:\n{window}" )); } - problems + Ok(()) } -/// Structural suspend assertions, per-run and yes/no: the device stream -/// must start only for a client and must be stopped — with soundd suspended -/// and silent — once the last client is gone. TCG-immune, so a violation is -/// categorical, never a rare event to be averaged. -/// -/// Positions are byte offsets in the RAW serial. That is sound because every -/// pattern below lands as one atomic chunk on the shared console: each pattern -/// sits inside a single format piece of its `eprintln!`, so one `write` syscall -/// carries it. Writers interleave BETWEEN chunks — whole foreign lines can land -/// inside a soundd line — but never inside these patterns, and chunk order is -/// emission order for soundd's mix thread, which emits every marker here. All -/// four are soundd's own since H3 moved the driver into it; the two stream -/// markers used to come from the kernel, from inside the submit syscall, and -/// the order they appear in is unchanged. -/// -/// `serial` is expected to carry `qemu.boot_log()` prepended ahead of the -/// test window, so a restored boot prime — the exact code deleted in -/// 465bc22, which would open the voice, play 8 periods, drain and suspend -/// entirely before ===TEST_START — lands inside these patterns rather than -/// in a discarded prefix. `audio_idle_suspend` reads `result.serial` alone -/// and stays blind to that window; this function is where it is caught. -pub fn check_suspend_structure(serial: &str) -> Vec { - const STARTED: &str = "virtio-sound: stream 0 started"; - const STOPPED: &str = "virtio-sound: stream 0 stopped"; - const CONNECTED: &str = " connected (id="; - const SUSPENDED: &str = "soundd: suspended"; - - let mut problems = Vec::new(); - - let Some(first_connect) = serial.find(CONNECTED) else { - problems.push("suspend structure: no client connect in capture".to_string()); - return problems; - }; - match serial.find(STARTED) { - None => problems.push( - "suspend structure: the stream never started inside the test window — \ - either the device was already running at boot (the boot state is \ - SUSPENDED) or the resume path is broken" - .to_string(), - ), - Some(at) if at < first_connect => problems.push( - "suspend structure: stream started before the first client connect".to_string(), - ), - Some(_) => {} +/// Two clients through soundd, and what soundd said about each leaving: every +/// removal names a departure soundd established, and none claims a death. +pub fn departures_on_metal(log: &Serial) -> Result<(), String> { + const CLIENTS: usize = 2; + let window = job_window(log.text(), "test_rs_null_sink_client_exits", CLIENTS)?; + let problems = check_departures(window, CLIENTS); + if problems.is_empty() { + return Ok(()); } + Err(format!("{}\n{window}", problems.join("\n"))) +} - let Some(last_removed) = last_client_removed(serial) else { - problems.push("suspend structure: no client removal in capture".to_string()); - return problems; - }; - if !serial[last_removed..].contains(SUSPENDED) { - problems.push( - "suspend structure: no `soundd: suspended` after the last client removal" - .to_string(), - ); - } - if !serial[last_removed..].contains(STOPPED) { - problems.push( - "suspend structure: no `virtio-sound: stream 0 stopped` after the last \ - client removal — the device is still running with no clients" - .to_string(), - ); - } - problems +/// Sum one `soundd:` counter across every stats window. +fn sum_field(serial: &str, key: &str) -> u32 { + let needle = format!(" {key}="); + serial + .match_indices(&needle) + .filter_map(|(at, _)| { + let rest = &serial[at + needle.len()..]; + let digits: String = rest.chars().take_while(char::is_ascii_digit).collect(); + digits.parse::().ok() + }) + .sum() } /// Every way a client left, as soundd reported it: one entry per /// `soundd: client {id} removed ({how})` in `serial`. /// -/// Anchored on ` removed` with the `soundd: client ` prefix discipline -/// [`last_client_removed`] documents, and the reason is read from the same line -/// — a removal that names none yields the empty string, which is what -/// [`check_departures`] reds on. -pub fn departures(serial: &str) -> Vec { +/// The reason is read from the same line — a removal that names none yields the +/// empty string, which is what [`check_departures`] reds on. +fn departures(serial: &str) -> Vec { serial .lines() .filter(|l| l.contains("soundd: client ") && l.contains(" removed")) @@ -735,7 +173,7 @@ pub fn departures(serial: &str) -> Vec { /// /// `expect` is how many removals the window must carry: a capture where no /// client ever left would otherwise satisfy every check above it vacuously. -pub fn check_departures(serial: &str, expect: usize) -> Vec { +fn check_departures(serial: &str, expect: usize) -> Vec { const KNOWN: [&str; 4] = ["closed", "refused", "disconnected", "signal pipe gone"]; let mut problems = Vec::new(); @@ -767,295 +205,18 @@ pub fn check_departures(serial: &str, expect: usize) -> Vec { problems } -/// Offset of the last `soundd: client {id} removed`, the anchor the two -/// after-the-last-client assertions above are relative to. -/// -/// ` removed` alone is an eight-character substring that any future line in -/// any component could carry; landing after soundd's markers it would move the -/// anchor past them and red all four configs at once, with a message accusing -/// soundd of the bug it does not have. Requiring the `soundd: client ` prefix -/// makes the anchor soundd's by construction rather than by a tree-wide -/// absence of other emitters. -/// -/// The two halves are matched separately because they are separate console -/// writes: `eprintln!("soundd: client {} removed", id)` emits three format -/// pieces, so a whole foreign line can land between the prefix and the suffix -/// (see the module doc on interleaving). A ` removed` qualifies when some -/// `soundd: client ` precedes it with no other ` removed` in between — true -/// for soundd's own, false for a foreign line printed after it. -fn last_client_removed(serial: &str) -> Option { - const CLIENT: &str = "soundd: client "; - const REMOVED: &str = " removed"; - serial - .match_indices(REMOVED) - .filter(|(at, _)| { - let before = &serial[..*at]; - before.rfind(CLIENT).is_some_and(|c| !before[c..].contains(REMOVED)) - }) - .map(|(at, _)| at) - .last() -} - -/// Bounds derived from the device's clock, not from any recorded run: values -/// on the wrong side of one did not happen, whatever the counter says. -/// -/// `check_counters` asks whether a run got *worse*; this asks whether it -/// happened *at all*. A violation is reported as a **broken instrument**, never -/// as a regression: fatal in both tiers, and in the thorough tier it aborts the -/// run before the value can enter the sample or the re-baselining output. That -/// separation is what the thorough tier cannot provide for itself — it applies -/// no per-run ceiling, its Mann-Whitney test is rank-based so one absurd value -/// moves no median, and it prints its own sample as the next baseline. -/// -/// The reference is the wall-clock life of the QEMU process, timed by the -/// harness. It is *outside* the guest, so no guest-side defect can inflate it -/// in step with the counter it bounds; the wav capture cannot serve, because -/// its timeline is the stream soundd submitted, so a stall that submits nothing -/// does not lengthen it. And it needs no recorded number, so there is nothing -/// to tune when a run goes red. The one assumption is that the guest and host -/// clocks agree to within a large factor — they are the same TSC up to -/// calibration error, and a calibration wrong by a percent would break the DLL -/// long before it reached these margins. -/// -/// These bounds sit far above every per-run ceiling in -/// `tests/audio-baseline.toml` (2.07-8.01 pipeline depths), and they have to: a -/// ceiling admits values that are bad but real, so a bound firing anywhere near -/// one would be answering the regression question again. "A few pipeline -/// depths" is a health threshold, not a physical limit. -pub fn check_physical(counters: &SounddCounters, run_secs: f64) -> Vec { - let mut faults = Vec::new(); - - // Lateness is the distance between two instants on the guest clock, both - // inside the life of the soundd process, which is inside the life of the - // QEMU process. A larger value does not fit, whatever it would mean. - if counters.max_wake_lat_us as f64 > run_secs * 1e6 { - faults.push(format!( - "wake lateness {}us ({:.1} pipeline depths) exceeds the whole {run_secs:.2}s run \ - it was measured inside — the instrument is broken, not the scheduler", - counters.max_wake_lat_us, - counters.max_wake_lat_us as f64 / PIPELINE_DEPTH_US as f64, - )); - } - - // The device is a fixed-rate DAC: it retires exactly one period every - // PERIOD_SECS and frees the DMA slot soundd then refills, so it cannot - // have taken more periods in the run than the run had room for, plus the - // pipeline still in flight at the end. - let room = (run_secs / PERIOD_SECS) as u32 + 8; - if counters.submitted > room { - faults.push(format!( - "{} periods submitted, but a {run_secs:.2}s run holds at most {room} \ - — the instrument is broken", - counters.submitted - )); - } - - // Definitional: soundd counts an underrun on a subset of the periods it - // counts as submitted, in the same branch. Violating it means the counter - // or the parser is wrong. - if counters.underruns > counters.submitted { - faults.push(format!( - "{} underruns out of {} periods submitted — underruns are a subset of \ - submitted, so one of the two counters is wrong", - counters.underruns, counters.submitted - )); - } - - faults -} - -fn silent_runs(mono: &[i32], min_len: usize) -> Vec { - let mut runs = Vec::new(); - let mut start = None; - for (i, &s) in mono.iter().enumerate() { - match (s.abs() <= SILENCE_MAX, start) { - (true, None) => start = Some(i), - (false, Some(s0)) => { - if i - s0 >= min_len { - runs.push(SilentRun { - start: s0, - len: i - s0, - }); - } - start = None; - } - _ => {} - } - } - if let Some(s0) = start { - if mono.len() - s0 >= min_len { - runs.push(SilentRun { - start: s0, - len: mono.len() - s0, - }); - } - } - runs -} - -/// soundd's own word for the sink it took when the machine has none. -pub const NULL_SINK: &str = "soundd: no audio device, presenting a null sink"; - -/// The kernel's word for the other outcome: soundd is not there any more. -/// -/// Read by every wait on something soundd owes, so that a dead mixer ends the -/// wait with the caller's own sentence instead of a guard expiring. -pub const SOUNDD_GONE: &str = "exit: soundd"; - -/// Wait until soundd has said which sink it took, bounded by the guest. -/// -/// init spawns its programs without waiting, so the ready marker is one child's -/// first line and orders nothing about another's — which of soundd and the test -/// runner speaks first is a race the guest never promised to win. Both callers -/// used to bound it with a span of host wall clock, and on a KVM runner soundd -/// lost that race: `metal_sim_null_audio` was red 5 of 5 with its line arriving -/// 64 ms past a 500 ms window (run `31258202923`, and the probe that timed it). -/// -/// [`SOUNDD_GONE`] ends the wait too, because the regression this gates is -/// soundd *exiting* on a device-less machine — it has to red with the caller's -/// own sentence and not fifteen seconds later as a stall. -pub fn await_null_sink(qemu: &mut QemuInstance, log: &mut String) -> Result<(), String> { - qemu::await_guest(qemu, log, "soundd to say which sink it took", |seen| { - seen.contains(NULL_SINK) || seen.contains(SOUNDD_GONE) - }) -} - -/// Gate: on a machine with **no audio hardware** (`Profile::Metal`, the T14's -/// shape and the one the `tone` panic reproduced on), an audio-producing client -/// runs to completion — exit 0, no panic — and does so at the real audio rate, -/// neither instantly nor stalled. -/// -/// This is the whole point of the null sink: hardware absence is a routing -/// state. Before it, soundd exited on a device-less machine and released its -/// service name, so cpal's `build_output_stream` failed `NotFound` and `tone` -/// panicked in `.expect("failed to build audio stream")`. That pre-null tree is -/// the negative control — reverting soundd's null-sink path reds this test on -/// the first assertion, with the kernel's `exit: soundd` in the capture. -/// -/// Three host-side assertions, and there is no wav here (this machine has no -/// device to capture from), so ground truth is the client's exit, the host wall -/// clock around it, and soundd's own counters: -/// -/// 1. **No crash.** The client exits 0. -/// 2. **Real rate.** A 3 s tone takes ~3 s of wall clock. Instant discard would -/// finish in a fraction of a second (the client fills its ring and races to -/// the end); a stalled sink would time the run out. soundd's `submitted` -/// counter — periods it drained — is the in-guest cross-check: ~3 s / 2.9 ms -/// ≈ 1034. -/// 3. **Not silent about being silenced.** soundd reports the discarded stream -/// in its stats windows (clients ≥ 1), the same accounting a real sink emits -/// and what #106's status tool will read. -pub fn null_sink_real_rate( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - ..Default::default() - }, - ); - - // soundd must present the null sink rather than exit. - let mut early = qemu.boot_log().to_string(); - let stalled = await_null_sink(&mut qemu, &mut early).err(); - if !early.contains(NULL_SINK) { - return Err(format!( - "{}soundd did not present a null sink on a device-less machine:\n{early}", - stalled.map(|why| format!("{why}\n")).unwrap_or_default() - )); - } - - // The tone is DURATION_SECS = 3.0 s of samples (tests/toyos-rust-tests/ - // src/tone.rs). At the real audio rate it cannot finish before then. - let start = Instant::now(); - let result = qemu.run_test("test_rs_audio_tone", Duration::from_secs(30)); - let elapsed = start.elapsed().as_secs_f64(); - - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - match result.exit_code { - Some(0) => {} - Some(code) => { - return Err(format!( - "tone exited {code} on a device-less machine (a no-device machine must \ - still play to completion):\n{}", - result.stdout - )) - } - None => return Err(format!("tone produced no exit code:\n{}", result.stdout)), - } - - // Real rate, host-measured. The 3 s tone plus its ~0.2 s drain tail and a - // little spawn overhead lands near 3.3 s; the bounds catch the two failure - // shapes the null sink exists to avoid — instant discard below, a sink - // draining slower than the audio rate above. - const MIN_SECS: f64 = 2.5; - const MAX_SECS: f64 = 8.0; - if !(MIN_SECS..=MAX_SECS).contains(&elapsed) { - return Err(format!( - "null sink drained a 3 s tone in {elapsed:.2} s (expected {MIN_SECS}..={MAX_SECS} s): \ - a client that writes N seconds of audio must take ~N seconds\nstdout:\n{}", - result.stdout - )); - } - - // soundd's own accounting of the discarded stream: a real-rate cross-check - // and the proof it is not silent about the silencing. The final window - // races the client's exit, so collect a little more serial first. - let serial = result.serial.clone() + &qemu.drain_serial(Duration::from_millis(500)); - let counters = parse_soundd_counters(&serial)?; - if counters.windows == 0 { - return Err(format!( - "soundd reported no stats window with a client — the tone never reached the null sink:\n{serial}" - )); - } - // ~3 s / 2.902 ms ≈ 1034 periods, plus the disconnect ramp. Wide enough to - // absorb the window boundaries, tight enough that instant discard (a handful - // of periods) or a half-rate drain (~520) both fail. - const MIN_SUBMITTED: u32 = 700; - const MAX_SUBMITTED: u32 = 1500; - if !(MIN_SUBMITTED..=MAX_SUBMITTED).contains(&counters.submitted) { - return Err(format!( - "null sink submitted {} periods for a 3 s tone (expected {MIN_SUBMITTED}..={MAX_SUBMITTED}): \ - the drain rate is not the audio rate\nstdout:\n{}", - counters.submitted, result.stdout - )); - } - - eprintln!( - " [metal-sim] null sink drained a 3 s tone in {elapsed:.2} s, {} periods, \ - {} stats window(s) — real rate, no device", - counters.submitted, counters.windows - ); - Ok(()) -} - -/// Gate: soundd's mix thread never waits on the log. -/// -/// `tests/logstallcase` boots a `logd` that reads nothing of soundd's until the -/// guest says the tone has played. The guest fills soundd's log ring with -/// soundd's own refusals and then plays the tone, so every line soundd says -/// while it plays is said to a full ring. Judged off `/log`, the sink of -/// record, after `run shutdown`: +/// soundd's mix thread never waits on the log: `tests/logstallcase`'s `logd` +/// reads nothing of soundd's until the job says the tone has played, and the +/// job fills soundd's ring with soundd's own refusals before it plays. Judged +/// off `/log`: /// -/// 1. **The tone played whole**: the capture carries it with no underrun and no -/// click. A mix thread that waited on its log would stop at its first line, -/// and the tone never plays at all. -/// 2. **The ring was full** — the premise: `logd` found every one of its slots -/// waiting when the stall ended, and had read none of them before. -/// 3. **Nothing went unwritten silently**: every line soundd's control thread +/// 1. **The ring was full** — the premise: `logd` found every one of its slots +/// waiting when the stall ended. +/// 2. **Nothing went unwritten silently**: every line soundd's control thread /// said after the boot is in `/log` or among the records `logd` counted /// unwritten, exactly, and some were counted — the flood is larger than the /// ring. -pub fn soundd_log_stall(rust_bins: &[(String, Vec)]) -> Result<(), String> { - use std::cell::Cell; - +pub fn log_stall_on_metal(log: &Serial) -> Result<(), String> { const REFUSAL: &str = "soundd: refusing connection,"; // `logd`'s counts of soundd's shared-ring records it could not write: its // lanes are the mix thread's, counted apart. @@ -1074,325 +235,202 @@ pub fn soundd_log_stall(rust_bins: &[(String, Vec)]) -> Result<(), String> { let rest = &line[line.find(marker)? + marker.len()..]; rest.split(' ').next()?.parse().ok() }; - - let staged = super::logstream::stage("tests/logstallcase", "soundd-log-stall", &[], rust_bins)?; - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/logstallcase"); - let options = BootOptions { - boot_image: Some(qemu::Staged::Written(staged.image.clone())), - ..Default::default() - }; - let mut qemu = QemuInstance::boot_with_options(&config, &[], rust_bins, options); - let result = qemu.run_test("test_rs_soundd_log_stall", Duration::from_secs(240)); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - if result.exit_code != Some(0) { - return Err(format!("the guest failed (exit {:?}):\n{}", result.exit_code, result.stdout)); - } - let said_by_guest = |marker: &str| -> Result { - result - .stdout - .lines() + let text = log.text(); + let said_by_job = |marker: &str| -> Result { + text.lines() .find_map(|l| number_after(l, marker)) - .ok_or_else(|| format!("the guest never said {marker:?}:\n{}", result.stdout)) + .ok_or_else(|| format!("the job never said {marker:?}:\n{text}")) }; - // Every refusal the guest heard or provoked is one line soundd said. - let owed = said_by_guest("flooded soundd with ")? + said_by_guest("soundd refused ")?; + // Every refusal the job heard or provoked is one line soundd said. + let owed = said_by_job("flooded soundd with ")? + said_by_job("soundd refused ")?; - // What the ledger reads over a log so far: refusals present, records - // counted unwritten, the other control lines present, and `logd`'s word - // on the ring: how many of its slots were waiting, of how many. - struct Ledger { - refusals: Cell, - unsaid: Cell, - others: Cell, - waiting: Cell>, - } - let read = |ledger: &Ledger, line: &str| { + let (mut refusals, mut unsaid, mut others, mut waiting) = (0u64, 0u64, 0u64, None); + let soundd = toyos_build::bootlog::lines_of(text, "soundd"); + let logd = toyos_build::bootlog::lines_of(text, "logd"); + for line in soundd.lines().chain(logd.lines()) { if line.contains(REFUSAL) { - ledger.refusals.set(ledger.refusals.get() + 1); + refusals += 1; } - let unwritten: u64 = UNWRITTEN.iter().filter_map(|m| number_before(line, m)).sum(); - ledger.unsaid.set(ledger.unsaid.get() + unwritten); + unsaid += UNWRITTEN.iter().filter_map(|m| number_before(line, m)).sum::(); if AFTER_FLOOD.iter().any(|m| line.contains(m)) { - ledger.others.set(ledger.others.get() + 1); + others += 1; } - if let (Some(waiting), Some(slots)) = (number_after(line, RELEASED), number_after(line, OF_SLOTS)) { - ledger.waiting.set(Some((waiting, slots))); + if let (Some(held), Some(slots)) = (number_after(line, RELEASED), number_after(line, OF_SLOTS)) { + waiting = Some((held, slots)); } - }; - let new_ledger = - || Ledger { refusals: Cell::new(0), unsaid: Cell::new(0), others: Cell::new(0), waiting: Cell::new(None) }; - let closed = |l: &Ledger| { - l.waiting.get().is_some() - && l.refusals.get() + l.others.get() + l.unsaid.get() >= owed + AFTER_FLOOD.len() as u64 - }; - - // The console carries the same lines as `/log`; it is only what says when - // to shut the guest down. - let console = new_ledger(); - for line in result.before.lines().chain(result.serial.lines()) { - read(&console, line); - } - if !closed(&console) { - let _ = qemu.drain_until(Duration::from_secs(120), |line| { - read(&console, line); - closed(&console) - }); } - - // Read once QEMU has exited, which is when it has written the capture's - // tail, and before the guest is dropped, which deletes it. - let mut shutdown = String::new(); - let (file, wav) = super::logstream::shut_down_keeping(qemu, &mut shutdown, &staged, |qemu| { - parse_wav(qemu.audio_wav_path()) - })?; - let (file, wav) = (file.concat(), wav?); - let analysis = analyze(&wav); - let _ = std::fs::remove_file(&staged.image); - - let log = new_ledger(); - for line in toyos_build::bootlog::lines_of(&file, "soundd").lines() { - read(&log, line); - } - for line in toyos_build::bootlog::lines_of(&file, "logd").lines() { - read(&log, line); - } - - let slots = match log.waiting.get() { - Some((waiting, slots)) if waiting == slots && slots > 0 => slots, - Some((waiting, slots)) => { + match waiting { + Some((held, slots)) if held == slots && slots > 0 => {} + Some((held, slots)) => { return Err(format!( - "logd found {waiting} of soundd's {slots} ring slots waiting when its stall \ - ended: the tone was not played to a full ring" + "logd found {held} of soundd's {slots} ring slots waiting when its stall ended: \ + the tone was not played to a full ring" )) } None => return Err("/log never says logd's stall on soundd ended".to_string()), - }; + } let said = owed + AFTER_FLOOD.len() as u64; - let accounted = log.refusals.get() + log.others.get() + log.unsaid.get(); - if accounted != said || log.unsaid.get() == 0 { + if refusals + others + unsaid != said || unsaid == 0 { return Err(format!( - "soundd's control thread said {said} lines after the boot ({owed} refusals and \ - {} others); /log holds {} refusals and {} of the others, and logd counted {} \ - unwritten", + "soundd's control thread said {said} lines after the boot ({owed} refusals and {} \ + others); /log holds {refusals} refusals and {others} of the others, and logd \ + counted {unsaid} unwritten", AFTER_FLOOD.len(), - log.refusals.get(), - log.others.get(), - log.unsaid.get() )); } - - let signal_secs = analysis.active_samples as f64 / wav.sample_rate as f64; - // The tone is 3 s, and a sine of amplitude 16000 is under SIGNAL_THRESHOLD - // for 2 asin(500 / 16000) / pi = 2% of its samples. - const MIN_SIGNAL_SECS: f64 = 2.5; - if signal_secs < MIN_SIGNAL_SECS || !analysis.underruns.is_empty() || !analysis.clicks.is_empty() - { - return Err(format!( - "the tone played to a stalled log is not whole: {signal_secs:.2} s of signal \ - (expected at least {MIN_SIGNAL_SECS}), {} underrun(s), {} click(s)", - analysis.underruns.len(), - analysis.clicks.len() - )); - } - - eprintln!( - " [logstallcase] {said} lines said to a full ring of {slots} records: {} in /log, {} \ - counted unwritten; the tone played {signal_secs:.2} s with no underrun", - log.refusals.get() + log.others.get(), - log.unsaid.get() - ); Ok(()) } -/// Gate: doom's sound producer outruns its audio callback and the game lives. -/// -/// The T14 report this exists for: about five seconds into playing doom the -/// machine froze, and the first thing to go was doom itself — -/// `sound command ring overflow: audio callback stalled` inside -/// `I_UpdateSoundParams`, an `extern "C"` frame with no unwind path, so the -/// panic became `abort`. The kernel then panicked retiring the task and the -/// compositor died granting shared memory to a pid that was no longer there. -/// This is the first domino. -/// -/// Nothing in the flood is timed. The actuator parks the audio callback with -/// `cpal::Stream::pause` and requires the callback's own period counter to -/// stand still across the burst, so "the producer outran the consumer" is a -/// fact about the two of them rather than about how busy this host was — a -/// distinction the harness owes the suspended-laptop case, where a wall-clock -/// burst is satisfied by stopping both sides at once. -/// -/// Four assertions, three of them in-guest facts the host reads back and one -/// on the wire: -/// -/// 1. **The game lives.** `/system/bin/doom --sound-stress` exits 0. On the tree this -/// replaced the same burst aborts on its 65th command. -/// 2. **The burst was real.** `stalled_burst` commands were issued with the -/// callback's period count unchanged — 4096 of them, 64x the retired ring. -/// 3. **The callback converged, and did so at the audio rate.** The sound the -/// last command started plays to completion, and the periods that took -/// match its length: a mixer that lost the command never finishes, and one -/// that restarted it takes longer. -/// 4. **The last command is what reached the device.** Every superseded update -/// in the burst carries `QUIET_VOLUME`, which mixes to 251 LSB — under -/// `SIGNAL_THRESHOLD`, so it is not signal. The final one carries full -/// volume, 16000 * 127/255 = 7968. A capture with signal in it is a capture -/// in which the last write won. -pub fn doom_sound_flood(rust_bins: &[(String, Vec)]) -> Result<(), String> { - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/doomcase"); - let mut qemu = QemuInstance::boot_with_options(&config, &[], rust_bins, BootOptions::default()); - let result = qemu.run_test("test_rs_doom_sound_flood", Duration::from_secs(60)); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - if result.exit_code != Some(0) { - return Err(format!( - "doom did not survive its own sound producer (exit {:?}):\n{}", - result.exit_code, result.stdout - )); - } - - let counters = parse_stress_line(&result.stdout)?; - - // The retired ring held 64 commands and asserted on the 65th. - const RETIRED_RING_CAP: u64 = 64; - if counters.stalled_burst <= RETIRED_RING_CAP { - return Err(format!( - "the flood was {} commands against a callback that had stopped, which the \ - retired 64-entry ring would have swallowed — the actuator proved nothing", - counters.stalled_burst - )); - } - - // A period is 128 frames, so a sound of N frames occupies ceil(N/128) of - // them. The ceiling is loose on purpose: it is a liveness bound on the game - // thread noticing the sound ended, and the verdict is the floor — a mixer - // that restarted or skipped the sound cannot land on its exact length. - check_playback("tone", counters.tone_periods, counters.tone_frames)?; - check_playback("probe", counters.probe_periods, counters.probe_frames)?; - - // Let the tail of the capture reach the file before reading it. - let _ = qemu.drain_serial(Duration::from_millis(500)); - let wav = parse_wav(qemu.audio_wav_path())?; - let analysis = analyze(&wav); - - // 16000 * 127/255 = 7968 at full volume against 251 for every superseded - // update, so the band excludes the second outcome by a factor of eight and - // is wide enough that soundd's mix path is not being measured here. - const MIN_PEAK: i32 = 4000; - const MAX_PEAK: i32 = 12000; - if !(MIN_PEAK..=MAX_PEAK).contains(&analysis.peak) { - return Err(format!( - "the device played a peak of {} (expected {MIN_PEAK}..={MAX_PEAK}): the volume \ - the last command named is not the volume that reached the wire", - analysis.peak - )); - } - // The tone is TONE_FRAMES long and 96% of a sine at this amplitude clears - // SIGNAL_THRESHOLD; a third of it is a floor no partial application meets. - let min_active = counters.tone_frames as usize / 3; - if analysis.active_samples < min_active { - return Err(format!( - "only {} samples of signal reached the device for a {}-frame tone (expected \ - at least {min_active})", - analysis.active_samples, counters.tone_frames - )); - } - - eprintln!( - " [doomcase] {} commands issued with the callback parked, tone converged in {} \ - periods for {} frames, {} concurrent commands, {} samples of signal at peak {}", - counters.stalled_burst, - counters.tone_periods, - counters.tone_frames, - counters.concurrent_cmds, - analysis.active_samples, - analysis.peak, - ); - Ok(()) -} - -struct StressCounters { - stalled_burst: u64, - tone_periods: u64, - tone_frames: u64, - concurrent_cmds: u64, - probe_periods: u64, - probe_frames: u64, -} - -fn parse_stress_line(stdout: &str) -> Result { - let line = stdout - .lines() - .find(|l| l.contains("[sound-stress] stalled_burst=")) - .ok_or_else(|| format!("doom printed no [sound-stress] line:\n{stdout}"))?; - let field = |name: &str| -> Result { - let prefix = format!("{name}="); - line.split_whitespace() - .find_map(|tok| tok.strip_prefix(&prefix)?.parse().ok()) - .ok_or_else(|| format!("no {name} in {line:?}")) +/// Every judge above against logs crafted to pass it and logs crafted to fail +/// it, one way each: the judges read a stick only the T14 comes back with, so +/// this is where each is shown able to red at all. +pub fn judges_verdict() -> Result<(), String> { + let judged = |what: &str, judge: fn(&Serial) -> Result<(), String>, log: &str, green: bool| { + match (judge(&Serial::named(what, log)), green) { + (Ok(()), true) | (Err(_), false) => Ok(()), + (Ok(()), false) => Err(format!("{what} passed a log it has to refuse:\n{log}")), + (Err(why), true) => Err(format!("{what} refused a log it has to pass: {why}\n{log}")), + } }; - Ok(StressCounters { - stalled_burst: field("stalled_burst")?, - tone_periods: field("tone_periods")?, - tone_frames: field("tone_frames")?, - concurrent_cmds: field("concurrent_cmds")?, - probe_periods: field("probe_periods")?, - probe_frames: field("probe_frames")?, - }) -} - -fn check_playback(what: &str, periods: u64, frames: u64) -> Result<(), String> { - const PERIOD_FRAMES: u64 = 128; - let exact = frames.div_ceil(PERIOD_FRAMES); - if !(exact..=exact * 4).contains(&periods) { - return Err(format!( - "the {what} took {periods} periods to play {frames} frames (expected \ - {exact}..={}): the mixer did not apply the last command as written", - exact * 4 - )); - } - Ok(()) -} - -/// The shipped audio client must finish and exit on a machine with no device. -/// -/// `metal_sim_null_audio` already asserts a client drains at the real rate, and -/// it passed on every boot while the T14 hung — because it runs this crate's -/// own tone, which reaches soundd through the SDK, and the program a user runs -/// reaches it through `cpal`. Same sink, same period grid, different client. -/// -/// Two clients in series, because the T14 log shows the *second* connect never -/// being applied: the control thread accepts and prints `opening stream`, and -/// no `client N connected` follows it. -pub fn null_sink_shipped_client( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { profile: qemu::Profile::Metal, ..Default::default() }, - ); + let stats = |underruns: u32, clients: u32, deferred: u32| { + format!( + "{{2.000 soundd}} soundd: wakes=9 completions=9 submitted=9 underruns={underruns} \ + drains=0 max_wake_lat_us=9 max_batch=1 clients={clients} deferred={deferred} \ + starve_max=0 worst_irq_late_us=0 worst_pickup_us=0 worst_empty=0 worst_batch=1 \ + late_wakes=0\n" + ) + }; + let spawn = |job: &str| format!("[kernel 1.000 cpu0] spawn: /system/bin/{job} pid=7\n"); + // One stream through soundd: `playing` is the stats line while it plays, + // and `ended` the flush once it has left. + let session = |playing: String, how: &str, ended: String| { + format!( + "{{1.000 soundd}} soundd: client 0 connected (id=1)\n{{1.000 soundd}} soundd: \ + resumed\n{playing}{{3.000 soundd}} soundd: client 1 removed ({how})\n{ended}\ + {{3.000 soundd}} soundd: suspended\n" + ) + }; + let next = spawn("test_rs_next"); - let result = qemu.run_test("test_rs_null_sink_client_exits", Duration::from_secs(60)); - if let Some(err) = &result.error { - return Err(format!("{err}\nstdout:\n{}\nserial:\n{}", result.stdout, result.serial)); - } - match result.exit_code { - Some(0) => {} - Some(code) => { - return Err(format!( - "the shipped tone exited {code} on a device-less machine:\n{}", - result.stdout - )) - } - None => return Err(format!("no exit code:\n{}", result.stdout)), - } - eprintln!(" [null-sink] {}", result.stdout.trim()); + let configured = "{0.500 soundd} soundd: hda path configured in 3 ms\n"; + let tone = |ended: String| { + format!("{configured}{}{}", spawn("test_rs_audio_tone"), session(stats(0, 1, 0), "closed", ended)) + }; + judged("the tone", tone_on_metal, &format!("{}{next}", tone(stats(0, 0, 0))), true)?; + judged("a tone short of periods", tone_on_metal, &format!("{}{next}", tone(stats(2, 0, 0))), false)?; + judged( + "a tone off no hda path", + tone_on_metal, + &format!("{}{next}", tone(stats(0, 0, 0))).replace(configured, ""), + false, + )?; + judged( + "a tone on the null sink", + tone_on_metal, + &format!("{}{{0.600 soundd}} {NULL_SINK}\n{next}", tone(stats(0, 0, 0))), + false, + )?; + + let stall = |second: String, rest: &str| { + format!( + "{}{}{second}{rest}", + spawn("test_rs_hda_client_stall"), + session(stats(3, 1, 0), "closed", stats(0, 0, 0)) + ) + }; + let second = session(stats(1, 1, 0), "closed", stats(0, 0, 0)); + judged("the stalled client", client_stall_on_metal, &stall(second.clone(), &next), true)?; + judged( + "a stalled client whose second stream never resumed soundd", + client_stall_on_metal, + &stall(second.replace("soundd: resumed\n", "soundd: client 0 streaming\n"), &next), + false, + )?; + judged( + "a stalled client soundd filled no period for", + client_stall_on_metal, + &stall(second.clone(), &next).replace("underruns=3", "underruns=0").replace("underruns=1", "underruns=0"), + false, + )?; + // The next job spawned ahead of soundd's last word on the second stream. + let (playing, tail) = second.split_at(second.find("{3.000 soundd} soundd: client 1 removed").expect("staged")); + judged( + "a stalled client soundd deferred for after the next job began", + client_stall_on_metal, + &stall(format!("{playing}{next}{}", tail.replace("deferred=0", "deferred=1")), ""), + false, + )?; + judged( + "a stalled client over a repeated completion", + client_stall_on_metal, + &stall(second.clone(), &format!("{{2.500 soundd}} soundd: repeated completion for free buffer\n{next}")), + false, + )?; + + let departures = |second: &str| { + format!( + "{}{}{}{next}", + spawn("test_rs_null_sink_client_exits"), + session(stats(0, 1, 0), "closed", stats(0, 0, 0)), + session(stats(0, 1, 0), second, stats(0, 0, 0)) + ) + }; + judged("two departures", departures_on_metal, &departures("signal pipe gone"), true)?; + judged("a departure soundd did not establish", departures_on_metal, &departures("died"), false)?; + judged( + "a death soundd claimed beside a departure it established", + departures_on_metal, + &departures("closed").replacen( + "{3.000 soundd} soundd: client 1 removed", + "{3.000 soundd} soundd: client 1 died\n{3.000 soundd} soundd: client 1 removed", + 1, + ), + false, + )?; + judged( + "a departure soundd never reported", + departures_on_metal, + &departures("closed").replacen("{3.000 soundd} soundd: client 1 removed (closed)\n", "", 1), + false, + )?; + judged( + "one departure of two", + departures_on_metal, + &format!( + "{}{}{next}", + spawn("test_rs_null_sink_client_exits"), + session(stats(0, 1, 0), "closed", stats(0, 0, 0)) + ), + false, + )?; + + let idle = |said: &str| format!("{}{said}{next}", spawn("test_rs_audio_idle_suspend")); + judged("an idle soundd", idle_suspend_on_metal, &idle("{1.000 soundd} soundd: suspended\n"), true)?; + judged( + "an idle soundd that started the device", + idle_suspend_on_metal, + &idle(&format!("{{1.000 soundd}} {DEVICE_STARTED}\n")), + false, + )?; + + // Eleven lines owed after the boot and the two others: six refusals and + // both others in `/log`, and five counted unwritten. + let stalled = |held: u32, unwritten: u32, refusals: usize| { + format!( + "{{1.000 test_rs_soundd_log_stall}} flooded soundd with 10 refusals\n\ + {{1.000 test_rs_soundd_log_stall}} soundd refused 1 probe\n\ + {}{{1.100 soundd}} soundd: protocol violation (msg 7)\n\ + {{1.200 soundd}} soundd: opening stream: 48000 Hz\n\ + {{2.000 logd}} logd: reading soundd again, as `--stall-until` asked, with {held} of \ + its ring's 64 slots waiting\n\ + {{2.100 logd}} logd: {unwritten} record(s) of soundd's found its ring full\n", + "{1.050 soundd} soundd: refusing connection, too many\n".repeat(refusals) + ) + }; + judged("the stalled log", log_stall_on_metal, &stalled(64, 5, 6), true)?; + judged("a log stall that never filled the ring", log_stall_on_metal, &stalled(40, 5, 6), false)?; + judged("a log stall that lost a line silently", log_stall_on_metal, &stalled(64, 4, 6), false)?; + judged("a log stall that counted nothing unwritten", log_stall_on_metal, &stalled(64, 0, 11), false)?; Ok(()) } diff --git a/tests/common/clock.rs b/tests/common/clock.rs index bb581e95e24..83ce1876797 100644 --- a/tests/common/clock.rs +++ b/tests/common/clock.rs @@ -3,8 +3,8 @@ //! The dev machine is a laptop and the owner closes the lid. A run that spans //! that is not a slow run, it is an **invalid measurement**: QEMU's virtual //! clock, the guest's own millisecond stamps and every device timing in it jump -//! by however long the machine was away, and every wall-clock verdict in the -//! serial tail and in gate A is taken against one of those. CLAUDE.md already +//! by however long the machine was away, and every wall-clock ceiling in the +//! suite is taken against one of those. CLAUDE.md already //! documents the signature — a tight cluster of durations plus a few enormous //! outliers — and documents it as something an agent must check *before* //! recording a finding, which is to say the harness has never been able to. diff --git a/tests/common/console.rs b/tests/common/console.rs index 159bea96488..f81a2ed935e 100644 --- a/tests/common/console.rs +++ b/tests/common/console.rs @@ -129,12 +129,12 @@ pub struct Verdict<'a> { /// /// **The scope boundary, and it is the whole safety argument.** This is the C /// family's stdout comparison and nothing else. Every other reader of a -/// daemon's line — the audio gates counting soundd's stats, `netd_*` waiting on -/// `netd: ready`, the sshd tests reading its host identity, the log gates — -/// reads `TestResult::serial` or a boot log, which this never touches. Those -/// tests *assert on* a daemon's line; this family is the one for which a -/// daemon's line is by construction not the subject, because the subject is a -/// C program's own stdout against a file recorded from it. +/// daemon's line — `netd_*` waiting on `netd: ready`, the sshd tests reading +/// its host identity, the log gates — reads `TestResult::serial` or a boot log, +/// which this never touches. Those tests *assert on* a daemon's line; this +/// family is the one for which a daemon's line is by construction not the +/// subject, because the subject is a C program's own stdout against a file +/// recorded from it. /// /// Which is also why the filter cannot make a broken case pass: a tinycc case's /// output is decided by its source, so the only way `soundd: …` appears in one diff --git a/tests/common/devices.rs b/tests/common/devices.rs index 7bafdc6efc5..40c5a1f4e6c 100644 --- a/tests/common/devices.rs +++ b/tests/common/devices.rs @@ -118,8 +118,7 @@ pub fn metal_device_probe( .and_then(|(_, rest)| rest.split_whitespace().next()) .and_then(|v| v.parse().ok()) .ok_or_else(|| format!("unreadable exit record: {line:?}"))?; - let span = measurement(job, code)?; - eprintln!(" [devices] {job}: {span} us"); + measurement(job, code)?; } // **The positive control for the safety instrument**, and this machine is diff --git a/tests/common/faults.rs b/tests/common/faults.rs index 5b331d9c82e..21bdb039902 100644 --- a/tests/common/faults.rs +++ b/tests/common/faults.rs @@ -1099,41 +1099,13 @@ fn parse(line: &str) -> Option<(usize, usize)> { /// reported *some* address would satisfy every other line here; only resolving /// it against the kernel's own symbols says the report points at where the CPU /// actually was. -pub fn dump_nmi_probe( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - kernel_params: &["dump-deaf-cpu"], - ..Default::default() - }, - ); - // 3 s is the actuator's earliest arming, not its schedule: cpu0 only looks - // once per idle-loop iteration, and on a settled guest the next thing that - // wakes it is the 10 s health tick. Add 400 ms of deafness and the dump's - // 250 ms kick budget, and 20 s is the first round number that clears it. - // - // **A ceiling now rather than the run.** The guest neither exits nor halts - // here — an NMI interrupts a CPU, it does not kill it — so a plain drain - // paid the whole twenty seconds on every green run, against a guest that - // was done in about a third of it. Both markers, and neither implies the - // other's order: the dump is requested while the victim is deaf and the - // victim announces its own return when the 400 ms window closes, so which - // of the two lands last is a fact about how long the report takes rather - // than about the machine. - let dumped = std::cell::Cell::new(false); - let rejoined = std::cell::Cell::new(false); - let log = qemu.drain_until(Duration::from_secs(20), |line| { - dumped.set(dumped.get() || line.contains("=== end of dump ===")); - rejoined.set(rejoined.get() || line.contains("rejoined after")); - dumped.get() && rejoined.get() - }); - +/// +/// **Judged on the T14 and in no QEMU guest.** The deafness is a window of the +/// actuator's own clock and the dump's kick and NMI budgets are the kernel's, +/// so a guest the host starves misses the window and reads exactly like the +/// defect this hunts. +pub fn dump_nmi_probe_on_metal(kernel: &Serial) -> Result<(), String> { + let log = kernel.text(); if !log.contains("=== blocked-task dump:") { return Err(format!("the dump never ran — is `dump-deaf-cpu` on?\n{log}")); } diff --git a/tests/common/gpt.rs b/tests/common/gpt.rs index 5b470c94a0d..5e25eb42b96 100644 --- a/tests/common/gpt.rs +++ b/tests/common/gpt.rs @@ -206,7 +206,7 @@ fn boot( boot_image: &Path, nvme_image: &Path, ) -> Result { - let mut qemu = QemuInstance::boot_with_options( + let qemu = QemuInstance::boot_with_options( config.parent().expect("system.toml has a directory"), &[], &[], @@ -222,8 +222,7 @@ fn boot( ..Default::default() }, ); - let mut log = qemu.boot_log().to_string(); - log.push_str(&qemu.drain_serial(std::time::Duration::from_millis(500))); + let log = qemu.boot_log().to_string(); for bad in ["PANIC:", "panicked at"] { if log.contains(bad) { return Err(format!("{bad:?} during the boot:\n{log}")); diff --git a/tests/common/hda.rs b/tests/common/hda.rs deleted file mode 100644 index ce635e00798..00000000000 --- a/tests/common/hda.rs +++ /dev/null @@ -1,332 +0,0 @@ -//! H4's audio arm, in the harness: soundd driving a real Intel HDA controller -//! itself, read back off the device rather than off the guest's opinion. -//! H0's own feasibility diagnostic was built ahead of this, then deleted once -//! the driver above answered every question it was asked for. - -use std::path::Path; -use std::time::Duration; - -use crate::common::audio::{await_null_sink, NULL_SINK}; -use crate::common::qemu::{BootOptions, Profile, QemuInstance}; -use crate::common::serial::Serial; - -/// H4's gate: a 440 Hz tone out of an Intel HDA controller soundd drives -/// itself, read back off the device rather than off the guest's opinion. -/// -/// `-audiodev wav` is the same ground truth gate A's four recorded configs use, -/// and the machine differs from them in the sound card alone -/// ([`Profile::Hda`]), so the capture is comparable by construction. What is -/// asserted here is **harm** — the tone is present, continuous, and dithered — -/// which is the fast tier's verdict. This is not a gate-A arm: it has no -/// recorded distribution behind it, and the four HDA sections a gate-A arm -/// would need in `tests/audio-baseline.toml` — `audio_tone_hda.smp1`, `.smp8`, -/// `audio_tone_hda_load.smp1`, `.smp8` — are unrecorded. -pub fn hda_tone( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: Profile::Hda, - kernel_params: &["hda-allowlist-selftest"], - ..Default::default() - }, - ); - // soundd claims and configures the controller the instant it starts, which - // is after the ready marker and before any test command — a window neither - // `boot_log` nor `run_test`'s own capture covers. - let mut log = Serial::boot(&qemu); - log.push(&qemu.drain_serial(Duration::from_millis(500))); - - let result = qemu.run_test("test_rs_audio_tone", Duration::from_secs(30)); - if let Some(err) = &result.error { - return Err(err.to_string()); - } - if result.exit_code != Some(0) { - return Err(format!("the tone did not play: {:?}\n{}", result.exit_code, result.stdout)); - } - log.push(&result.serial); - log.push(&qemu.drain_serial(Duration::from_millis(500))); - let serial = log.text().to_string(); - log.must_say("hda: 00:")?; - log.must_say("bound, statests=")?; - log.must_say("soundd: hda codec0 vendor=1af4")?; - log.must_say("-> pin 0x03 (line-out)")?; - log.must_say("soundd: hda path configured in")?; - if serial.contains("presenting a null sink") { - return Err(format!("soundd fell back to the null sink:\n{serial}")); - } - - // The allow-list, every arm, on the one caller that can reach it. - for want in [ - "hda: selftest write ICW written", - "hda: selftest write SDnFMT written", - "hda: selftest write SDnCTL written", - "hda: selftest write SDnCTL-tag written", - "hda: selftest write SDnBDPL refused", - "hda: selftest write SDnBDPU refused", - "hda: selftest write SDnCBL refused", - "hda: selftest write SDnLVI refused", - "hda: selftest write SDnSTS refused", - "hda: selftest write SDnCTL-srst refused", - "hda: selftest write SDnCTL-wide refused", - "hda: selftest write INTCTL refused", - "hda: selftest write GCTL refused", - "hda: selftest read ICS read", - "hda: selftest read IRR read", - "hda: selftest read SDnLPIB refused", - "hda: selftest read STATESTS refused", - ] { - log.must_say(want)?; - } - - let wav = crate::common::audio::parse_wav(qemu.audio_wav_path())?; - let analysis = crate::common::audio::analyze(&wav); - if analysis.peak < 8000 { - return Err(format!( - "the capture peaks at {} — the tone plays at 16000 and nothing reached the device", - analysis.peak - )); - } - let gaps = crate::common::audio::gap_histogram(&analysis, wav.sample_rate); - let dropouts: u32 = gaps.values().sum(); - let breaks = crate::common::audio::phase_breaks(&wav); - let pitch = crate::common::audio::dominant_hz(&wav); - eprintln!( - " [hda] {} frames at {} Hz {} ch, peak {} active {:.2}s dither {:.1}% pitch {:.1}Hz \ - gaps {} phase-breaks {}", - wav.mono.len(), - wav.sample_rate, - wav.channels, - analysis.peak, - analysis.active_samples as f64 / wav.sample_rate as f64, - analysis.dither_ratio.unwrap_or(0.0) * 100.0, - pitch.unwrap_or(0.0), - crate::common::audio::format_histogram(&gaps), - breaks.len(), - ); - if dropouts > 0 { - return Err(format!( - "{dropouts} mid-tone silences in the capture: {}", - crate::common::audio::format_histogram(&gaps) - )); - } - // The rate the engine plays at is soundd's decision on this machine and - // nothing else here can see it: a stream format naming the wrong base is - // eight buffers of correct audio a second played 8.8% fast, which every - // other assertion in this file passes. - if let Some(complaint) = crate::common::audio::wrong_pitch(&wav) { - return Err(complaint); - } - // The instrument the gap detector cannot be: an engine that replays a - // period nobody refilled puts the tone back 0.28 of a cycle out, and - // nothing about that is silent. Zero here and zero on - // all four virtio configs, measured — so the check has a calibration and - // not just a threshold. - // - // `dither_ratio` is deliberately not asserted, and is printed above so the - // difference is visible rather than hidden. It measures the longest - // *silent* run, and QEMU's two device models put different silence there: - // virtio-sound's capture opens before the stream does, so its longest - // silent run is soundd's own dithered output (24.6% on this host), while - // `intel-hda`'s wav voice runs only while the stream does and the longest - // silent run is host padding at the ends of the file. The virtio arm still - // asserts it, over a stretch that is soundd's. - if !breaks.is_empty() { - let where_ = |n: &usize| { - format!("{n} (period {:.1}, {:?})", *n as f64 / 128.0, &wav.mono[n - 1..=n + 1]) - }; - return Err(format!( - "the captured tone is not one sine: {} phase breaks at {}", - breaks.len(), - breaks.iter().take(8).map(where_).collect::>().join(", ") - )); - } - log.must_be_clean() -} - -/// The T14's panic, staged: a client that stops producing mid-stream. -/// -/// **Ground truth is which of two things soundd did with the periods the client -/// did not cover**, and the two machines must answer differently. HDA's engine -/// is a cyclic ring — it plays buffer `i` again `num_buffers` periods after -/// completing it, whatever soundd put there — so a period held back for a -/// client is played as silence anyway and then completed a second time, which -/// is a completion for a buffer soundd still holds. virtio-sound's queue plays -/// nothing soundd has not submitted, so holding one costs nothing and the -/// deferral is exactly right there. -/// -/// So: the ring arm must report `underruns` (soundd filled the periods and had -/// no client audio for them) and the queue arm must report `deferred` (soundd -/// held them). Asserting both is what stops the two obvious wrong fixes — -/// deleting the deferral, which reds the queue arm, and letting the ring hold a -/// period, which reds the ring arm with the panic this exists for. -/// -/// Nothing in the tone clients reaches this state: they keep their rings full, -/// so `hda_tone` measured `deferred=0` on every run. The stall is the actuator, -/// and it has to outlast one lap of the ring — 8 periods, 23.2 ms — or the -/// engine never comes back round to a period soundd is holding. -pub fn hda_client_stall( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let ring = stall_run(test_config, c_bins, rust_bins, "ring", Profile::Hda)?; - let queue = stall_run(test_config, c_bins, rust_bins, "queue", Profile::Headless)?; - - if ring.underruns == 0 { - return Err(format!( - "the ring arm reports no underrun: soundd never filled a period the stalled client \ - had not covered, so this run staged nothing\n{}", - ring.serial - )); - } - if ring.deferred != 0 { - return Err(format!( - "the ring arm deferred {} period(s): the engine replays every one of them and then \ - completes it again, which is the panic this test exists for\n{}", - ring.deferred, ring.serial - )); - } - if queue.deferred == 0 { - return Err(format!( - "the queue arm deferred nothing: deferral is what a stalled client is supposed to buy on \ - a device that plays only what it is given\n{}", - queue.serial - )); - } - eprintln!( - " [hda] stalled client: ring filled {} period(s) it had no audio for and held none; \ - queue held {}", - ring.underruns, queue.deferred - ); - Ok(()) -} - -struct StallRun { - underruns: u32, - deferred: u32, - serial: String, -} - -/// One boot of the stalling client, and what soundd did with the periods. -/// -/// soundd's liveness is the first verdict and it is not implied by the client's -/// exit: the client talks to soundd over IPC and a soundd that died mid-stream -/// leaves it blocked, so the run times out rather than reporting a code. The -/// panic line is checked by name anyway — `must_be_clean` would catch it, but -/// not say which of soundd's assertions it was. -fn stall_run( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], - arm: &str, - profile: Profile, -) -> Result { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { profile, ..Default::default() }, - ); - let mut log = Serial::boot(&qemu); - let result = qemu.run_test("test_rs_hda_client_stall", Duration::from_secs(60)); - if let Some(err) = &result.error { - return Err(format!("the {arm} arm: {err}\n{}\n{}", result.stdout, result.serial)); - } - log.push(&result.serial); - log.push(&qemu.drain_serial(Duration::from_millis(500))); - if result.exit_code != Some(0) { - return Err(format!( - "the stalling client exited {:?} on the {arm} arm:\n{}\n{}", - result.exit_code, - result.stdout, - log.text() - )); - } - log.must_not_say("repeated completion for free buffer")?; - log.must_be_clean()?; - - let serial = log.text().to_string(); - // The client plays twice with a suspend between, so a resume is under test - // as much as the stall is: on a ring the drain gives its periods up rather - // than holding them, and what the second prime fills and where in the ring - // it starts are both what the first stream left behind. - let resumes = serial.matches("soundd: resumed").count(); - if resumes < 2 { - return Err(format!( - "soundd resumed {resumes} time(s) on the {arm} arm — the second stream did not find a \ - suspended daemon, so nothing here tests a resume:\n{serial}" - )); - } - let counters = crate::common::audio::parse_soundd_counters(&serial)?; - if counters.windows == 0 { - return Err(format!("soundd reported no stats window on the {arm} arm:\n{serial}")); - } - Ok(StallRun { underruns: counters.underruns, deferred: sum_field(&serial, "deferred"), serial }) -} - -/// Sum one `soundd:` counter across every stats window. -/// -/// `parse_soundd_counters` stops at the fields gate A's baseline records, and -/// `deferred` is not one of them — it is an activity signal with no ceiling. It -/// is read here because it is the whole difference between the two arms. -fn sum_field(serial: &str, key: &str) -> u32 { - let needle = format!(" {key}="); - serial - .match_indices(&needle) - .filter_map(|(at, _)| { - let rest = &serial[at + needle.len()..]; - let digits: String = rest.chars().take_while(char::is_ascii_digit).collect(); - digits.parse::().ok() - }) - .sum() -} - -/// Two controllers, both with a codec that answers. -/// -/// The kernel binds neither and names both. A first-match bind would go green -/// on every other test in this file, and it is the defect `pci.rs` records one -/// layer down — so this is the arm that makes the rule tested rather than -/// merely written. -pub fn hda_two_live_refused( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { profile: Profile::HdaTwoLive, ..Default::default() }, - ); - // The refusal is a kernel boot line and is in the capture already; soundd's - // answer to it is a userland line that races the ready marker, so it is - // waited for on the guest's clock rather than on a span of the host's. - let mut text = qemu.boot_log().to_string(); - let stalled = await_null_sink(&mut qemu, &mut text).err(); - let log = Serial::named("boot console", text); - - log.must_say("hda: 00:")?; - log.must_say("has a live link (statests=")?; - log.must_say("controllers answer on this machine")?; - log.must_say("refused by name, no HDA audio")?; - log.must_not_say("bound, statests=")?; - // The machine still boots and still has a sink: absence of hardware is a - // routing state, and a refusal must not be a machine that will not run. - // **And it is the bind's absence and not a second spelling of the line - // above**: init claims each class before it spawns, and soundd reaches the - // null sink only where the endowment is missing — so this line requires - // `device::try_claim(HdaAudio)` to have answered `Absent`. - log.must_say(NULL_SINK) - .map_err(|why| match stalled { - Some(stall) => format!("{stall}\n{why}"), - None => why, - })?; - log.must_say("Boot: complete")?; - log.must_be_clean() -} diff --git a/tests/common/hostload.rs b/tests/common/hostload.rs deleted file mode 100644 index f7033afbdae..00000000000 --- a/tests/common/hostload.rs +++ /dev/null @@ -1,180 +0,0 @@ -//! What else the host was doing when a gate A verdict was taken. -//! -//! CLAUDE.md's 2026-08-04 ruling stands and nothing here touches it: load is -//! not an excuse, no threshold branches on any number below, and a -//! load-coincident red is investigated as a real defect of the pipeline. What -//! this adds is that the investigation starts from a recorded fact, and that an -//! A/B can say whether its two arms were taken under comparable conditions — -//! which `tests/audio-baseline.toml`'s recorded sample can only claim in prose. -//! -//! Three readings, because they fail in different directions: -//! -//! - **The load average triple**, 1/5/15 minutes. No figure of it resolves a -//! single ~15 s boot; the 1-minute one lags by design. But the competition is -//! other worktrees' builds, which last minutes, and the triple's *shape* says -//! whether the host was ramping up or winding down where one figure cannot. -//! The 1-minute figure leads because it is the one the baseline's ceiling -//! derivation recorded per run, so a fresh reading compares directly to it. -//! - **QEMU processes machine-wide.** Exact and instantaneous where the load -//! average is neither, and it counts the guests the *host* has rather than -//! the ones this run started — `qemu::live_instances()` is asserted zero -//! before gate A, so the harness's own knowledge is a constant here. This -//! run's own guest is still up when the sample is taken, so 1 is quiet. -//! - **`toyos-build` processes machine-wide**, drivers and harnesses alike: the -//! count that names the competition as ToyOS work rather than as whatever -//! else a laptop is doing. 1 is this run alone. -//! -//! A reading that cannot be taken is reported as unknown. A gate A verdict must -//! not turn on whether the process table answered. - -use std::fmt; - -#[derive(Clone, Copy)] -pub struct HostLoad { - /// 1, 5 and 15-minute load averages. - pub load: Option<[f64; 3]>, - pub qemu: Option, - pub toyos_build: Option, -} - -impl HostLoad { - pub fn sample() -> Self { - let names = process_names(); - HostLoad { - load: load_average(), - qemu: names.as_deref().map(|n| count(n, |name| name.starts_with("qemu-system"))), - toyos_build: names.as_deref().map(|n| count(n, is_toyos_build)), - } - } -} - -impl fmt::Display for HostLoad { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - match self.load { - Some([one, five, fifteen]) => { - write!(f, "host: load {one:.1}/{five:.1}/{fifteen:.1}")? - } - None => write!(f, "host: load ?")?, - } - write!(f, " qemu {} toyos-build {}", show(self.qemu), show(self.toyos_build)) - } -} - -/// The conditions a whole sample was taken under, for the re-record that sample -/// becomes. The thorough tier prints its own numbers in a form meant to be -/// pasted into `tests/audio-baseline.toml`; this is the sentence that has to go -/// beside them, and its absence is why the recorded sample's own conditions are -/// a claim rather than a measurement. -pub fn summarise(runs: &[HostLoad]) -> String { - let load: Vec = runs.iter().filter_map(|r| r.load.map(|l| l[0])).collect(); - let load = match span_f64(&load) { - Some((lo, med, hi)) => format!("1-min load {lo:.1}-{hi:.1} (median {med:.1})"), - None => "1-min load unreadable".to_string(), - }; - format!( - "host conditions over {} runs: {load}, qemu {}, toyos-build {}", - runs.len(), - span_usize(runs.iter().filter_map(|r| r.qemu)), - span_usize(runs.iter().filter_map(|r| r.toyos_build)), - ) -} - -fn span_f64(values: &[f64]) -> Option<(f64, f64, f64)> { - let mut sorted = values.to_vec(); - sorted.sort_by(|a, b| a.partial_cmp(b).unwrap()); - Some((*sorted.first()?, sorted[sorted.len() / 2], *sorted.last()?)) -} - -fn span_usize(values: impl Iterator) -> String { - let mut sorted: Vec = values.collect(); - sorted.sort_unstable(); - match (sorted.first(), sorted.last()) { - (Some(lo), Some(hi)) => format!("{lo}-{hi}"), - _ => "?".to_string(), - } -} - -fn show(count: Option) -> String { - count.map_or_else(|| "?".to_string(), |n| n.to_string()) -} - -fn load_average() -> Option<[f64; 3]> { - let mut out = [0.0f64; 3]; - // SAFETY: getloadavg writes at most `nelem` doubles through the pointer. - let filled = unsafe { libc::getloadavg(out.as_mut_ptr(), out.len() as i32) }; - (filled == out.len() as i32).then_some(out) -} - -/// Every process on the host, by executable basename. -/// -/// A pid that exits between the sizing call and the read, or refuses its path -/// (a zombie, another user's), contributes no name rather than failing the -/// sample — the counts above are of processes that could be named. -#[cfg(target_os = "macos")] -fn process_names() -> Option> { - let count = unsafe { libc::proc_listallpids(std::ptr::null_mut(), 0) }; - if count <= 0 { - return None; - } - // Slack for processes spawned between the sizing call and the fill. - let mut pids = vec![0 as libc::pid_t; count as usize + 16]; - let bytes = i32::try_from(std::mem::size_of_val(&pids[..])).ok()?; - let filled = unsafe { libc::proc_listallpids(pids.as_mut_ptr().cast(), bytes) }; - if filled <= 0 { - return None; - } - pids.truncate(filled as usize); - let mut names = Vec::with_capacity(pids.len()); - for pid in pids { - let mut buf = [0u8; libc::PROC_PIDPATHINFO_MAXSIZE as usize]; - let len = unsafe { libc::proc_pidpath(pid, buf.as_mut_ptr().cast(), buf.len() as u32) }; - if len <= 0 { - continue; - } - let path = String::from_utf8_lossy(&buf[..len as usize]); - names.push(path.rsplit('/').next().unwrap_or(&path).to_string()); - } - Some(names) -} - -/// Every process on the host, by executable basename: `/proc//exe`'s -/// target where that link is readable, the kernel's 15-byte `comm` otherwise — -/// every prefix `sample` matches on fits in either. -#[cfg(target_os = "linux")] -fn process_names() -> Option> { - let mut names = Vec::new(); - for entry in std::fs::read_dir("/proc").ok()?.flatten() { - let is_pid = entry - .file_name() - .to_str() - .is_some_and(|n| !n.is_empty() && n.bytes().all(|b| b.is_ascii_digit())); - if !is_pid { - continue; - } - if let Ok(path) = std::fs::read_link(entry.path().join("exe")) { - if let Some(name) = path.file_name() { - names.push(name.to_string_lossy().into_owned()); - continue; - } - } - if let Ok(comm) = std::fs::read_to_string(entry.path().join("comm")) { - names.push(comm.trim().to_string()); - } - } - Some(names) -} - -fn count(names: &[String], matches: impl Fn(&str) -> bool) -> usize { - names.iter().filter(|name| matches(name)).count() -} - -/// Cargo spells the two halves of this build system differently — the driver is -/// the package's bin target `toyos-build`, the harness is the test target's -/// `toyos_build-` — so a single prefix misses whichever one is asking. -fn is_toyos_build(name: &str) -> bool { - name.starts_with("toyos-build") || name.starts_with("toyos_build") -} - -// A third host OS gets a named gap, not a missing-function error at a distance. -#[cfg(not(any(target_os = "macos", target_os = "linux")))] -compile_error!("process_names reads the process table per-OS, and this OS has no arm yet"); diff --git a/tests/common/inspect.rs b/tests/common/inspect.rs index 767c623acd3..99349bbe8d5 100644 --- a/tests/common/inspect.rs +++ b/tests/common/inspect.rs @@ -6,9 +6,8 @@ //! matches too much is a path in the answer this file did not name, and one //! that matches too little is a named path missing from it. Values are judged //! where the machine fixes them — QEMU's user network leases `10.0.2.15/24`, -//! nothing plays audio until `inspect_plays` does, and the USB stick this file -//! crafts has one partition free and one init grants — and read only for shape -//! elsewhere. +//! nothing plays audio, and the USB stick this file crafts has one partition +//! free and one init grants — and read only for shape elsewhere. use std::collections::BTreeMap; use std::path::Path; @@ -23,8 +22,6 @@ pub const CONFIG: &str = "tests/inspectcase"; /// The guest binary that holds the negative control. pub const DENIED: &str = "inspect_denied"; -/// The guest binary that plays periods through soundd. -pub const PLAYS: &str = "inspect_plays"; /// The guest binary that sends `SYS_DEVICE_INVENTORY` its edges. pub const BOUNDS: &str = "inventory_bounds"; @@ -55,9 +52,9 @@ const NET: &[&str] = &[ pub fn boot(rust_bins: &[(String, Vec)]) -> Result { let bins: Vec<(String, Vec)> = - rust_bins.iter().filter(|(name, _)| [DENIED, PLAYS, BOUNDS].contains(&name.as_str())).cloned().collect(); - if bins.len() != 3 { - return Err(format!("{DENIED}, {PLAYS} and {BOUNDS} were not all built")); + rust_bins.iter().filter(|(name, _)| [DENIED, BOUNDS].contains(&name.as_str())).cloned().collect(); + if bins.len() != 2 { + return Err(format!("{DENIED} and {BOUNDS} were not both built")); } let stick = super::lane::dir().join("inspect-stick.img"); let mib = 1024 * 1024; @@ -200,30 +197,6 @@ pub fn reads_its_owners(qemu: &mut QemuInstance) -> Result<(), String> { expect(line, &got, "sound.stream.state", "suspended")?; expect(line, &got, "sound.stream.clients", "0")?; - // Periods through soundd, and the running sums grow by at least what it - // took: each frame soundd took went out in a period it submitted. - let result = job(qemu, &format!("test_rs_{PLAYS}"), 0)?; - let said = "inspect plays: soundd took "; - let taken: u64 = result - .stdout - .lines() - .find_map(|l| l.split_once(said).map(|(_, rest)| rest.trim_end_matches(" frames"))) - .and_then(|n| n.trim().parse().ok()) - .ok_or_else(|| format!("{PLAYS} did not say how much soundd took:\n{}", result.stdout))?; - if taken == 0 { - return Err(format!("{PLAYS} proved soundd took nothing:\n{}", result.stdout)); - } - let line = "inspect sound.*"; - let got = answer(&job(qemu, line, 0)?); - let submitted = number(line, &got, "sound.periods.submitted")?; - let period = number(line, &got, "sound.period_frames")?; - if submitted.saturating_mul(period) < taken { - return Err(format!( - "`{line}`: {submitted} periods of {period} frames submitted, and soundd took {taken} \ - frames" - )); - } - let line = "inspect log.*"; let got = answer(&job(qemu, line, 0)?); exactly( diff --git a/tests/common/iommu.rs b/tests/common/iommu.rs index a463d7e3155..a07825a63ec 100644 --- a/tests/common/iommu.rs +++ b/tests/common/iommu.rs @@ -911,7 +911,8 @@ pub fn iommu_context_absent( c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - let (log, blocked) = fault_boot(test_config, c_bins, rust_bins, &["iommu-context-absent"])?; + let (log, blocked) = + fault_boot(test_config, c_bins, rust_bins, &["iommu-context-absent", "panic-reboot-fast"])?; // Which function the actuator left out is decided in the guest by class // code; which function that *is* on this machine is read here from the PCI @@ -964,7 +965,8 @@ pub fn iommu_empty_domain( c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - let (log, blocked) = fault_boot(test_config, c_bins, rust_bins, &["iommu-empty-domain"])?; + let (log, blocked) = + fault_boot(test_config, c_bins, rust_bins, &["iommu-empty-domain", "panic-reboot-fast"])?; let nvme = class_function(&log, "0108").ok_or_else(|| { format!("this machine enumerated no NVMe controller to strand\n{}", log.text()) @@ -1189,9 +1191,8 @@ fn foreign_fault( unit_is_first(&qemu::profile_argv(&options), arm.name)?; // An arm whose device is driven by a process boots that process's config. let config = arm.driver.config(test_config); - let mut qemu = QemuInstance::boot_with_options(&config, c_bins, rust_bins, options); - let mut log = Serial::boot(&qemu); - log.push(&qemu.drain_serial(Duration::from_secs(2))); + let qemu = QemuInstance::boot_with_options(&config, c_bins, rust_bins, options); + let log = Serial::boot(&qemu); let socket = qemu.qmp_socket(); let blocked = blocked_on(log.must_say(FAULT)?)?; @@ -1335,7 +1336,6 @@ pub fn iommu_gpu_scanout_swap( )); } log.push(&result.serial); - log.push(&qemu.drain_serial(Duration::from_millis(500))); log.must_not_say(FAULT)?; log.must_be_clean()?; let shown = qemu.screendump(); @@ -1801,9 +1801,16 @@ fn fault_boot( ); let mut log = Serial::boot(&qemu); // Past the fault, because the claim is that the machine stopped there: the - // handler halts every CPU, so anything the boot would have gone on to do - // has to be absent from a window that stays open after it. - log.push(&qemu.drain_serial(Duration::from_secs(2))); + // handler takes the fatal path, and the capture is judged once its reset + // has ended QEMU. + let mut after = String::new(); + qemu::await_reset( + &mut qemu, + &mut after, + "the fault's fatal path to reset the machine", + &["Boot: complete", qemu::DEFAULT_READY], + )?; + log.push(&after); log.must_not_say("Boot: complete")?; log.must_not_say(qemu::DEFAULT_READY)?; @@ -2074,6 +2081,16 @@ pub fn userdev_dma_fault( )); } + // netd's answer to the refusal is its own end: `Card::begin_pass` panics + // on the claim's `Io`, and it exits 101. Awaited so that the capture below + // is the machine's after netd, and a claim that stops refusing reds here. + let mut end = String::new(); + qemu::await_guest(&mut qemu, &mut end, "netd's end on its refused claim", |end| { + end.contains("netd: this NIC's claim refused an interrupt read: Io") + && end.lines().any(|l| l.contains("exit: netd pid=") && l.contains(" code=101 ")) + }) + .map_err(|e| format!("{e}\n{end}\n{}", log.text()))?; + // And the machine is running. This is the assertion the whole stage is // for: a guest that answers here is one whose scheduler, spawn path and // IPC all survived a device being refused mid-flight. @@ -2091,13 +2108,30 @@ pub fn userdev_dma_fault( result.exit_code, result.stdout )); } - // Nothing panicked on the way, and the staged fault happened **once**: - // clearing the function's Bus Master Enable is what bounds a storm, and a - // second line would say it did not. Every other boot in the estate reds on - // this line through `must_be_clean`; this is the one that staged it. + // `end` is the window from the fault to netd's exit, and nothing else + // judges it — it goes into the check below rather than staying read only + // for the two needles `await_guest` waited on. netd's own panic is + // staged, so its location line, immediately above the message already + // matched above, is the one line this capture may hold; a second panic, + // netd's or anyone else's, has no line here to hide behind. + let message_at = end + .lines() + .position(|l| l.contains("netd: this NIC's claim refused an interrupt read: Io")) + .ok_or_else(|| format!("netd's panic message vanished between the wait and the check:\n{end}"))?; + let mut lines: Vec<&str> = end.lines().collect(); + if message_at == 0 || !lines[message_at - 1].contains("panicked at") { + return Err(format!("netd's panic message arrived without its location line:\n{end}")); + } + lines.remove(message_at - 1); + let end = lines.join("\n"); + + // The staged fault happened **once**: clearing the function's Bus Master + // Enable is what bounds a storm, and a second line would say it did not. + // Every other boot in the estate reds on this line through + // `must_be_clean`; this is the one that staged it. let mut after = log; + after.push(&end); after.push(&result.serial); - after.push(&qemu.drain_serial(Duration::from_millis(500))); after.must_be_clean_apart_from("iommu: DMA FAULT owner=slot", 1)?; eprintln!( " [iommu] the NIC's driver was refused an address it was handed, and the machine ran on" diff --git a/tests/common/lan.rs b/tests/common/lan.rs index 6e9968a0bbb..1c64312111b 100644 --- a/tests/common/lan.rs +++ b/tests/common/lan.rs @@ -470,8 +470,7 @@ pub fn lan_dhcp_lease( } let mut guest = QemuInstance::boot_with_options(&case, &[], &[], options); let mut console = guest.boot_log().to_string(); - qemu::await_marker(&mut guest, &mut console, READY, "netd to take an address")?; - console.push_str(&guest.drain_serial(std::time::Duration::from_millis(500))); + qemu::await_marker(&mut guest, &mut console, LEASE, "netd to take an address")?; // QEMU owns the pcap while it runs, and every refusal below is a return: // the frames are taken once the machine is gone and the file removed here. drop(guest); @@ -499,12 +498,6 @@ pub fn lan_dhcp_lease( // The order, and not merely the presence of both. log.must_say_after(LEASE, READY)?; log.must_say(LINK_UP)?; - let ms = link_up_ms(log.text())?; - eprintln!( - " [lan] the emulated link came up in {ms} ms and the lease landed {} ms after netd \ - started", - lease.ms - ); log.must_be_clean()?; asked_under_its_own_name(&frames)?; eprintln!(" [lan] the client asked under its own name on the wire"); @@ -525,17 +518,26 @@ pub fn lan_no_lease( let options = BootOptions { profile: qemu::Profile::E1000eNoServer, ..Default::default() }; let mut guest = QemuInstance::boot_with_options(&case, &[], &[], options); let mut console = guest.boot_log().to_string(); - // Drained rather than waited on: netd owes its line inside its own bound - // and the guest says nothing at all until then, which every wait in this - // harness reads as a machine that stopped. - console.push_str( - &guest.drain_serial(std::time::Duration::from_millis(toyos_tco::LEASE_BOUND_MS + 10_000)), - ); + // Drained until netd's `ready` after its give-up, and not awaited: the + // guest says nothing at all until netd gives up on its own clock, which + // every wait in this harness reads as a machine that stopped. The ceiling + // is the harness's. + let gave_up = format!("{NO_LEASE}{HOSTNAME} in "); + let given_up = std::cell::Cell::new(false); + let served = std::cell::Cell::new(false); + console.push_str(&guest.drain_until(qemu::GUEST_WEDGED, |line| { + given_up.set(given_up.get() || line.contains(&gave_up)); + served.set(given_up.get() && line.contains(READY)); + served.get() + })); + if !served.get() { + return Err(format!("{} waiting for {gave_up:?} and then {READY:?}\n{console}", qemu::STALLED)); + } let log = serial::Serial::named("the lan boot with no server", console.as_str()); if let Ok(lease) = lease_in(log.text()) { return Err(format!("a wire with no server leased {lease:?}")); } - log.must_say_after(&format!("{NO_LEASE}{HOSTNAME} in "), READY)?; + log.must_say_after(&gave_up, READY)?; eprintln!(" [lan] no server answered and netd said so, then served anyway"); Ok(()) } diff --git a/tests/common/lane.rs b/tests/common/lane.rs index fbe4a98240f..526a0669856 100644 --- a/tests/common/lane.rs +++ b/tests/common/lane.rs @@ -66,8 +66,7 @@ static RUN: OnceLock = OnceLock::new(); /// it, and it is gone when the run is, green or red (`toyos_tmpdir` is the /// policy, and what reclaims the directory of a run that was killed). /// -/// A red run's serial logs, and any suspect audio capture already renamed to -/// keep (`audio-*-smp*.wav`), are the parts of it read afterwards — by an agent +/// A red run's serial logs are the parts of it read afterwards — by an agent /// and by the nightly's artifact — so they are copied to a directory of their /// own under [`RED_RUN_SERIAL`] first: megabytes, where the images are /// gigabytes. Named for this run's own root — unique across every process a @@ -125,31 +124,18 @@ fn kept_root(run: &Path) -> Result { Ok(super::compile::repo_root().join(RED_RUN_SERIAL).join(name)) } -/// Where `path`, somewhere under this run's directory, ends up if the run ends -/// red: mirrored under [`kept_root`], as [`keep_serial`] would copy it there. -pub fn kept_path(path: &Path) -> PathBuf { - let run = RUN.get().expect("`Run::begin` comes before any scratch"); - let rel = path - .strip_prefix(run) - .unwrap_or_else(|_| panic!("{} is not under this run's directory {}", path.display(), run.display())); - kept_root(run).unwrap_or_else(|e| panic!("{e}")).join(rel) -} - -/// Copy every `uart-*.log` and suspect `audio-*-smp*.wav` under `run` to a -/// directory of its own under [`RED_RUN_SERIAL`], keeping each one's path -/// below the run. +/// Copy every `uart-*.log` under `run` to a directory of its own under +/// [`RED_RUN_SERIAL`], keeping each one's path below the run. fn keep_serial(run: &Path) -> Result { let kept = kept_root(run)?; copy_serial(run, &kept)?; Ok(kept) } -/// A serial log, or a suspect audio capture already renamed for keeping -/// (`tests/toyos.rs`'s `measure_audio_run`) — everything [`keep_serial`] -/// rescues from a run's scratch before it goes. +/// A serial log — everything [`keep_serial`] rescues from a run's scratch +/// before it goes. fn worth_keeping(name: &str) -> bool { - (name.starts_with("uart-") && name.ends_with(".log")) - || (name.starts_with("audio-") && name.contains("-smp") && name.ends_with(".wav")) + name.starts_with("uart-") && name.ends_with(".log") } fn copy_serial(dir: &Path, into: &Path) -> Result<(), String> { diff --git a/tests/common/logread.rs b/tests/common/logread.rs index e438887b250..8bba44eb139 100644 --- a/tests/common/logread.rs +++ b/tests/common/logread.rs @@ -22,9 +22,7 @@ use super::qemu::{BootOptions, QemuInstance}; /// either way. const GATE: &str = "log-gate"; -/// The whole run's ceiling. A liveness guard and never a verdict: the guest has -/// a ceiling of its own and reports what it had when it gave up, so this only -/// catches a guest that stopped answering at all. +/// The whole run's ceiling: a gate that never finishes is what it reds. const CEILING: Duration = Duration::from_secs(60); /// One boot's storm, as the guest reported it. diff --git a/tests/common/logstream.rs b/tests/common/logstream.rs index ec621d8cc67..e991aab264d 100644 --- a/tests/common/logstream.rs +++ b/tests/common/logstream.rs @@ -247,10 +247,6 @@ const NETWORK_READERS: usize = 8; /// TCG guest's netd, as long as the flood job itself is given. const FLOOD_CEILING: Duration = Duration::from_secs(300); -/// How long `logd` lets a reader take no bytes before it lets it go -/// (`serve.rs`'s `STALLED`). -const STALLED_SECS: u64 = 10; - /// What `logd` says as it lets a reader go that took no bytes it was owed. const LET_GO: &str = "logd: letting "; @@ -280,12 +276,10 @@ pub fn stalled_reader( // Flooded until every stalled reader is let go, not by an amount: a // reader is owed only what `logd` took of the flood, which is `logd`'s // pace and not this test's, so a fixed flood outruns every buffer between - // them only on a host fast enough. Each let-go is owed `STALLED_SECS` - // after its reader's writes stop being taken; the flood's own ceiling, - // widened by this host, is that promise judged, and a guest past it has - // broken it, which is this test's verdict and not a stall of the harness. - let began = Instant::now(); - let deadline = began + guest.budget(FLOOD_CEILING); + // them only on a host fast enough. When `logd` lets each one go is its own + // clock's business and no verdict here: the ceiling is the harness's, and + // a `logd` that never lets a reader go is a hang it reds. + let deadline = Instant::now() + guest.budget(FLOOD_CEILING); let mut floods = 0usize; while seen(&console) < NETWORK_READERS { let left = deadline.saturating_duration_since(Instant::now()); @@ -307,15 +301,14 @@ pub fn stalled_reader( } let let_go = seen(&console); if let_go < NETWORK_READERS { - let waited = began.elapsed().as_secs(); let said = match shut_down(guest, &mut console, &staged) { Ok(file) => file.iter().filter(|l| l.contains("logd: ")).cloned().collect::(), Err(why) => format!("none read: {}", why.lines().next().unwrap_or("")), }; return Err(format!( - "logd let {let_go} of the {NETWORK_READERS} readers that stopped reading go in {waited} s \ - of {floods} flood(s), and owes each one {STALLED_SECS} s after its writes stop being \ - taken; /log's logd lines:\n{said}" + "{} flooding until logd let the readers that stopped reading go: {let_go} of \ + {NETWORK_READERS} after {floods} flood(s); /log's logd lines:\n{said}", + qemu::STALLED )); } let second = reader(port, "logstream-stalled-second.txt")?; diff --git a/tests/common/mod.rs b/tests/common/mod.rs index aaeea0cc903..2958914faeb 100644 --- a/tests/common/mod.rs +++ b/tests/common/mod.rs @@ -24,10 +24,6 @@ pub mod faults; #[allow(dead_code)] pub mod gpt; #[allow(dead_code)] -pub mod hda; -#[allow(dead_code)] -pub mod hostload; -#[allow(dead_code)] pub mod https; #[allow(dead_code)] pub mod iommu; @@ -48,8 +44,6 @@ pub mod origin; #[allow(dead_code)] pub mod partclaim; #[allow(dead_code)] -pub mod passcost; -#[allow(dead_code)] pub mod pkg; #[allow(dead_code)] pub mod power; @@ -65,8 +59,6 @@ pub mod serial; #[allow(dead_code)] pub mod ssh; #[allow(dead_code)] -pub mod stats; -#[allow(dead_code)] pub mod storage; pub mod swap; #[allow(dead_code)] diff --git a/tests/common/origin.rs b/tests/common/origin.rs index 3b3df827d5c..fcc8cd09531 100644 --- a/tests/common/origin.rs +++ b/tests/common/origin.rs @@ -349,22 +349,8 @@ pub fn refused_stop(c_bins: &[(String, Vec)], rust_bins: &[(String, Vec) Ok(()) } -/// What init says when it stops the machine without `logd`'s answer, and how -/// long it waits first (`userland/init`'s `FLUSH_BOUND`). +/// What init says when it stops the machine without `logd`'s answer. const FLUSH_WAITED_OUT: &str = "init: logd did not answer the flush in"; -const FLUSH_BOUND_MS: u64 = 5_000; -/// The kernel's record as a stop begins its sync, after every thread stopped. -const SYNCING: &str = "Syncing filesystems..."; - -/// The milliseconds since boot a program's line in `/log` carries: the one -/// field of its head with a decimal point. -fn program_millis(line: &str) -> Option { - let (head, _) = line.strip_prefix(toyos_logstream::OPEN)?.split_once(toyos_logstream::CLOSE)?; - head.split_whitespace().find_map(|field| { - let (secs, millis) = field.split_once('.')?; - secs.parse::().ok()?.checked_mul(1_000)?.checked_add(millis.parse().ok()?) - }) -} /// **A resume that reaches `logd` with its flush unrun answers that flush.** /// `tests/logflushcase` holds `logd`'s first flush until init speaks again, so @@ -469,23 +455,16 @@ pub fn keeps_the_owners_slots(rust_bins: &[(String, Vec)]) -> Result<(), Str let _ = std::fs::remove_file(&staged.image); // A stop that went ahead before the flush answered cuts `/log` short // whatever the slots did, so that is its own verdict and not this one's. - // init's word is in its ring when the machine stops, so the console - // rarely carries it. An answered flush wrote init's stop line, which is - // stamped before the flush was asked; and the kernel's sync starting a - // flush bound or more after that line is the wait on two records of one - // clock. - let sync = tail - .lines() - .find(|l| l.contains(SYNCING)) - .and_then(bootlog::record_millis) - .ok_or_else(|| format!("the console carries no {SYNCING:?} with a time\n{tail}"))?; - let stop = bootlog::stopping_line(&log).and_then(program_millis); - let unanswered = match stop { - _ if tail.contains(FLUSH_WAITED_OUT) => Some("init said so".to_string()), - None => Some("init's stop line never reached /log".to_string()), - Some(stop) => (sync.saturating_sub(stop) >= FLUSH_BOUND_MS).then(|| { - format!("the kernel synced {} ms after init's stop line", sync.saturating_sub(stop)) - }), + // init's stop line in /log does not say the flush was answered: init says + // it waited one out, on the console or in /log, and neither may carry that. + let unanswered = if tail.contains(FLUSH_WAITED_OUT) + || bootlog::lines_of(&log, "init").contains(FLUSH_WAITED_OUT) + { + Some("init said so") + } else if bootlog::stopping_line(&log).is_none() { + Some("init's stop line never reached /log") + } else { + None }; if let Some(why) = unanswered { return Err(format!( @@ -493,7 +472,6 @@ pub fn keeps_the_owners_slots(rust_bins: &[(String, Vec)]) -> Result<(), Str slots\n{tail}" )); } - let stop = stop.expect("an answered flush's stop line has a time"); let runner = bootlog::lines_of(&log, RUNNER); // Non-vacuity: the flood met a full ring, so the slots were contested. let flooded = runner.lines().filter(|l| l.starts_with("flood ")).count(); @@ -507,9 +485,7 @@ pub fn keeps_the_owners_slots(rust_bins: &[(String, Vec)]) -> Result<(), Str )); } eprintln!( - " [origin] {flooded} flood lines filled the ring, and test-runner's {ENDED:?} is in /log; \ - the kernel synced {} ms after init's stop line", - sync.saturating_sub(stop) + " [origin] {flooded} flood lines filled the ring, and test-runner's {ENDED:?} is in /log" ); Ok(()) } @@ -683,7 +659,6 @@ pub fn mdns(c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)]) -> Re const LOOPBACK_ID: u16 = 0x7f01; const OTHER_ID: u16 = 0x0bad; const OWN_ID: u16 = 0x5eed; - let asked = Instant::now(); wire.send(&ask([127, 0, 0, 1], &query(LOOPBACK_ID, host)))?; wire.send(&ask(NEIGHBOUR, &query(OTHER_ID, "some-other-host")))?; wire.send(&ask(NEIGHBOUR, &query(OWN_ID, host)))?; @@ -707,7 +682,6 @@ pub fn mdns(c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)]) -> Re _ => {} } }; - let took = asked.elapsed(); let mut want = query(OWN_ID, host); // QR and AA, one question, one answer. want[2..8].copy_from_slice(&[0x84, 0x00, 0, 1, 0, 1]); @@ -723,9 +697,8 @@ pub fn mdns(c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)]) -> Re drop(guest); serial::Serial::named("the boot", console.as_str()).must_be_clean()?; eprintln!( - " [mdns] {host}.local answered {GUEST:?} to an on-link neighbour in {} ms; 127.0.0.1 and \ - another name, nothing", - took.as_millis() + " [mdns] {host}.local answered {GUEST:?} to an on-link neighbour; 127.0.0.1 and another \ + name, nothing" ); Ok(()) } diff --git a/tests/common/passcost.rs b/tests/common/passcost.rs deleted file mode 100644 index ff741280c06..00000000000 --- a/tests/common/passcost.rs +++ /dev/null @@ -1,536 +0,0 @@ -//! The harness half of the scheduler's pass-cost instrument: read the check -//! build's published distribution and judge it against **what this accelerator -//! has been recorded producing**, not against an absolute line. -//! -//! **Why the judgement is here and not in the kernel.** A pass is measured with -//! the only clock either world has across one — wall clock, `rdtsc` in the -//! kernel — and a guest's wall clock advances while the host has taken its vCPU -//! away. The elapsed time of a pass is therefore a *composed* quantity: the -//! scheduler's own work plus an interval the host's scheduler sets, which this -//! CPU neither observes nor controls and which no constant bounds. A panic may -//! assert only what its own site observes and what no workload scales, so the -//! kernel records the distribution ([`toyos_sched::cpu::PassCostReport`]) and -//! this file decides what it means. -//! -//! **What that costs, stated rather than implied.** A single pass that really -//! ran long and a single pass whose CPU was descheduled are the same sample, -//! and nothing — here or in the kernel — can separate them. So the maximum is -//! printed and never gated, and a rare long pass is reported rather than -//! caught. That is the honest limit of this instrument and it has not changed. -//! -//! # Why `MAX_PASS_NS` is not the line any more -//! -//! This gate used to read the budget directly: nine passes in ten provably -//! under 200 000 ns. Its argument was that a host which deschedules a vCPU -//! touches the handful of passes it lands in, so it moves the maximum and not -//! the 90th percentile — *mass is the scheduler, extremes are the machine*. -//! That argument was an observed rate rather than a bound, and it was put to -//! the experiment on 2026-08-18. -//! -//! **It is false.** Six repetitions per arm, quiet and loaded, strictly -//! interleaved in one session so both arms share the ambient host; twelve -//! CPU-runs per arm at `smp: 2`. The load was fourteen pure-shell spin loops, -//! one per logical CPU (`sysctl -n hw.logicalcpu` answers 14 on the dev host), -//! measured standing alone at 90.0–94.9 % CPU each and taking the 1-minute load -//! average from 9.45 to 28.02. The arms separate on the harness's own boot-width -//! instrument with no overlap: 1.74x–2.34x quiet against 2.66x–2.78x loaded. -//! -//! | | quiet, 12 CPU-runs | loaded, 12 CPU-runs | -//! |---|---|---| -//! | p50 | 8 192 ×1, 16 384 ×1, 65 536 ×10 | 8 192 ×1, 32 768 ×1, 131 072 ×10 | -//! | p90 | 65 536 ×2, 131 072 ×10 | 131 072 ×3, 262 144 ×9 | -//! | max | 1 508 330 – 1 983 355 ns | 1 723 825 – 3 914 718 ns | -//! | over budget | 95 of 2 732 (3.48 %) | 187 of 2 619 (7.14 %) | -//! | p90 over budget | **0 of 12** | **9 of 12** | -//! -//! **The whole distribution translates up by one power-of-two bucket under host -//! load, and 200 000 ns sits between the quiet p90 and the loaded one** — which -//! is the entirety of why the verdict flips. Every order statistic moved, the -//! median as much as the tail, so no fraction chosen instead of nine-in-ten -//! would have survived either. There is no quantile of this quantity that host -//! load leaves alone. -//! -//! # What replaces it: the recorded sample of the accelerator in hand -//! -//! The same session recovered the other half. This file's line has printed on -//! every run whatever the verdict, so CI's own logs already held a KVM sample — -//! sixteen `ci` runs, 32 CPU-runs, 7 612 passes, **not one of them over the -//! budget**, p90 at 32 768 ns and the largest single pass in the whole set -//! 173 906 ns. The two accelerators are not one instrument: p90 differs by four -//! between KVM and a *quiet* dev host and by eight against a loaded one, and the -//! maxima by twenty. An absolute line cannot be right for both, and 200 000 ns -//! is right for neither — six times looser than anything KVM produces, and -//! inside the range host load alone sweeps on TCG. -//! -//! So a run is judged against its own accelerator's recorded sample -//! ([`KVM`], [`TCG`]) — the shape `tests/audio-baseline.toml` uses, made -//! environment-relative, which needs no separation of steal from work at all. -//! **And where a recorded sample cannot support a verdict it says so and does -//! not take one**: TCG's own sample spans four buckets on one unchanged tree, -//! so no line drawn from it would be a statement about the scheduler. That is -//! `tests/CLAUDE.md`'s standing rule for this host — what only CI's KVM shards -//! can decide is decided there — applied to a cost rather than to a vendor. -//! -//! **What was not done, so nobody re-derives it.** Reading KVM's paravirtual -//! steal-time MSR would let the guest gate a quantity it observes rather than -//! one it infers, and it is closed by owner ruling: a hypervisor-specific -//! facility cannot be the basis of a gate in a tree whose north star is -//! self-hosting on metal. Steal accounting may be a diagnostic and never a gate. - -use toyos_sched::cpu::{PassCostReport, MAX_PASS_NS}; - -/// Samples a report must carry before its quantiles mean anything: a 90th -/// percentile over this many has ten above it. -/// -/// **The margin is 21 %, and it is stated rather than assumed.** Eighteen -/// CPU-runs on the dev host, 2026-08-17, ranged from 121 to 346 passes with a -/// median around 148; 121 is the closest any of them came to this floor. The -/// two recorded samples below bear it out from the other side — their smallest -/// counts are 135 (KVM) and 143 (TCG). If a run ever falls below it the gate -/// reds *saying so*: a report that cannot answer must not answer, and a -/// percentile computed from forty samples is a sample. -/// -/// **The floor holds in both judgement modes**, because it is not about the -/// quantile — it is about the instrument having run at all, and a report of -/// forty passes is a broken instrument on any accelerator. -pub const MIN_SAMPLES: u64 = 100; - -/// What a recorded sample supports. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum Judgement { - /// Gate one bucket above the sample's own worst 90th percentile. - Ceiling, - /// Print the distribution and take no verdict on its magnitude. The string - /// is why: a mode that judges nothing must say what stopped it. - Report(&'static str), -} - -/// One accelerator's recorded pass-cost sample, and the verdict it supports. -/// -/// The sample is the observations themselves rather than a summary of them, for -/// `tests/toyos.rs`'s reason at `BaselineSample`: a summary cannot be re-read by -/// the next person to ask whether the line drawn from it was fair. -#[derive(Clone, Copy)] -pub struct Baseline { - /// The accelerator, as the transcript names it. - pub accelerator: &'static str, - /// What was sampled, when, and with what — the sentence that has to survive - /// beside the numbers, or they are a claim rather than a measurement. - pub recorded: &'static str, - /// Every per-CPU-run 90th percentile in the recorded sample, in ns. These - /// are bucket ends, so every one is a power of two, and [`self_check`] - /// refuses a sample where one is not: a number that is not a bucket end was - /// not read off a run. - pub p90_ns: &'static [u64], - pub judgement: Judgement, -} - -impl Baseline { - /// The worst 90th percentile anywhere in the recorded sample. - pub fn worst_p90_ns(&self) -> u64 { - self.p90_ns.iter().copied().max().expect("a recorded sample with no observations in it") - } - - /// The line a fresh run is held to, or `None` where the sample supports - /// none. - /// - /// **One bucket above the sample's worst, and that margin is the smallest - /// this instrument can express.** A quantile is answered as a bucket end, - /// so the resolution is one power of two; a ceiling *at* the sample's own - /// worst has no margin at all against the next draw from the same - /// distribution, and that worst is itself a thin observation — 3 of 32 on - /// KVM. One bucket is therefore the least that is not a coin flip. - /// - /// **What it costs, said here rather than discovered later.** A regression - /// that moves the 90th percentile by less than four times the sample's mode - /// passes. The instrument has no finer unit to do better with. - pub fn ceiling_ns(&self) -> Option { - match self.judgement { - Judgement::Ceiling => Some(self.worst_p90_ns() * 2), - Judgement::Report(_) => None, - } - } -} - -/// KVM on native x86-64 — CI's `guest` shards, and the accelerator every claim -/// about what a scheduler pass costs is really about. -/// -/// Harvested with `gh run view --job --log` over every `ci` workflow run -/// between the instrument landing (#113, 2026-08-17) and 2026-08-18: sixteen -/// runs, 32 CPU-runs, 135–1 533 passes each and 7 612 pooled. **Not one pass in -/// any of them reached `MAX_PASS_NS`**, and the largest single pass in the whole -/// set was 173 906 ns — so this sample is also, independently, the strongest -/// statement anything has made about the budget on the machine it was derived -/// for. -pub const KVM: Baseline = Baseline { - accelerator: "KVM, native x86-64", - recorded: "16 CI runs / 32 CPU-runs, 2026-08-17..18, runs 32043101865 32043658251 \ - 32044008591 32044311347 32044665350 32044756253 32044758468 32044760536 \ - 32044762195 32044763748 32045857575 32047352064 32050586046 32096188866 \ - 32116842348 32117779110; 7612 passes pooled, 0 over MAX_PASS_NS, max 173906 ns", - // Two per run, cpu0 then cpu1, in the run order above. - p90_ns: &[ - 32_768, 32_768, // 32043101865 - 32_768, 32_768, // 32043658251 - 32_768, 32_768, // 32044008591 - 32_768, 32_768, // 32044311347 - 32_768, 32_768, // 32044665350 - 32_768, 32_768, // 32044756253 - 65_536, 65_536, // 32044758468 - 32_768, 32_768, // 32044760536 - 32_768, 32_768, // 32044762195 - 32_768, 32_768, // 32044763748 - 32_768, 32_768, // 32045857575 - 32_768, 32_768, // 32047352064 - 32_768, 32_768, // 32050586046 - 65_536, 32_768, // 32096188866 - 32_768, 32_768, // 32116842348 - 32_768, 32_768, // 32117779110 - ], - judgement: Judgement::Ceiling, -}; - -/// Cross-arch TCG on the arm64 dev host, emulating x86-64 instruction by -/// instruction while the guest TSC advances with host wall clock. -/// -/// The quiet arm then the loaded arm of the 2026-08-18 experiment — **one -/// unchanged tree, one session, twelve CPU-runs each.** The spread is the whole -/// content of this entry: 65 536 to 262 144 ns, a factor of four, with nothing -/// but the host's other processes separating the halves. A ceiling at the quiet -/// arm's worst reds every loaded run; one at the loaded arm's worst accepts a -/// fourfold regression. Neither is a statement about the scheduler, so this -/// accelerator takes no verdict on magnitude at all. -/// -/// The sample is kept rather than dropped because it *is* that argument, and -/// because the next person to propose a dev-host pass-cost gate should have to -/// read it first. **It also understates the range**, which is the confirming -/// run rather than a caveat: the same test under the same load at 4.09x boot -/// width — harder than either arm — reported `p50 < 131072 ns, p90 < 524288 ns, -/// max 22666022 ns` on a tree whose `sched_stress` passed and whose boot was -/// clean. That is 2.6 times the budget at the 90th percentile and a hundred -/// times it at the maximum, from nothing but fourteen shell loops. -pub const TCG: Baseline = Baseline { - accelerator: "cross-arch TCG, arm64 dev host", - recorded: "12 quiet + 12 loaded CPU-runs, 2026-08-18, one interleaved session; the load is \ - 14 pure-shell spin loops, one per logical CPU, and the arms separate on boot \ - width 1.74x-2.34x against 2.66x-2.78x with no overlap", - p90_ns: &[ - // quiet, six runs, cpu0 then cpu1 - 65_536, 131_072, 131_072, 131_072, 65_536, 131_072, 131_072, 131_072, 131_072, 131_072, - 131_072, 131_072, // loaded, six runs, cpu0 then cpu1 - 262_144, 262_144, 262_144, 262_144, 262_144, 262_144, 131_072, 262_144, 131_072, 262_144, - 131_072, 262_144, - ], - judgement: Judgement::Report( - "the recorded sample spans four buckets on one unchanged tree, moved by nothing but \ - what else the host was running, so no line drawn from it separates a scheduler that \ - grew from a laptop that was busy. What this accelerator still gates is everything \ - else `sched_check_build` asks: a clean boot, the three check-build asserts not \ - firing, and `sched_stress` running to completion", - ), -}; - -/// The recorded sample for the accelerator this run is actually using. -pub fn baseline() -> &'static Baseline { - if super::qemu::SUITE_ARCH.accel().is_hardware() { - &KVM - } else { - &TCG - } -} - -/// The last report each CPU published in `capture`, ordered by CPU. -/// -/// The counters are cumulative since boot, so the last line is the whole run -/// and the earlier ones are prefixes of it. -pub fn reports(capture: &str) -> Vec { - let mut last: Vec = Vec::new(); - for report in capture.lines().filter_map(PassCostReport::parse) { - match last.iter().position(|r| r.cpu == report.cpu) { - Some(at) => last[at] = report, - None => last.push(report), - } - } - last.sort_by_key(|r| r.cpu.0); - last -} - -/// One line per CPU, for the test's own transcript. Printed whatever the -/// verdict: a green run that stops publishing what it measured is how a gate -/// goes quiet. `over` is still counted against `MAX_PASS_NS` — the budget stays -/// the kernel's policy number and stays reported; it has only stopped deciding. -pub fn describe(report: &PassCostReport) -> String { - format!( - "cpu{}: {} passes, p50 < {} ns, p90 < {} ns, p99 < {} ns, max {} ns, \ - {} over the {} ns budget", - report.cpu.0, - report.count, - report.quantile_upper_ns(1, 2), - report.quantile_upper_ns(9, 10), - report.quantile_upper_ns(99, 100), - report.max_ns, - report.over, - MAX_PASS_NS, - ) -} - -/// The one line that says which sample judged this run and how, printed before -/// the per-CPU lines and whatever the verdict. -/// -/// A verdict taken against a recorded sample is unreadable without naming the -/// sample, and a run that judged nothing has to say *that* loudest of all. -pub fn judgement_line(baseline: &Baseline) -> String { - match baseline.judgement { - Judgement::Ceiling => format!( - "judged against the {} sample: 9 passes in 10 under {} ns, one bucket above that \ - sample's own worst of {} ns. Sample: {}", - baseline.accelerator, - baseline.ceiling_ns().expect("Judgement::Ceiling has a ceiling"), - baseline.worst_p90_ns(), - baseline.recorded, - ), - Judgement::Report(why) => format!( - "NOT judged on magnitude — the {} sample supports no line: {}. Sample: {}", - baseline.accelerator, why, baseline.recorded, - ), - } -} - -/// Judge one CPU's distribution against `baseline`. -/// -/// Two claims, and every term in both comes from somewhere: -/// -/// - **The sample floor holds on every accelerator.** A report of fewer than -/// [`MIN_SAMPLES`] passes is a broken instrument, not a distribution, and -/// nothing below it is worth reading. -/// - **The magnitude claim is the recorded sample's, one bucket up** — see -/// [`Baseline::ceiling_ns`] — and only where the sample supports one. -/// -/// The **fraction is still nine in ten and not ninety-nine in a hundred**, -/// because a whole boot plus `sched_stress` is about 150 passes per CPU: a 90th -/// percentile over 150 samples has fifteen above it; a 99th has one and a half, -/// which is its own largest sample wearing a percentile's name. Moving the line -/// from a policy number to a recorded one does not touch that reasoning, and -/// both recorded samples confirm the count from below. -/// -/// `quantile_upper_ns` answers with a bucket's upper bound, so "this fraction -/// cost less than X ns" is exact and X ≤ ceiling is a proof. A quantile landing -/// in the bucket that straddles the ceiling reds, which makes the gate strictly -/// conservative rather than approximately right. -/// -/// **What it does not catch, said here rather than discovered later.** The -/// maximum is not read at all, for the reason the module header gives. A -/// regression smaller than the instrument's own resolution — one power of two — -/// passes on any accelerator. And on a report-only accelerator nothing about -/// magnitude is caught at all, which is the price of not making a claim the -/// environment cannot support. -pub fn verdict(report: &PassCostReport, baseline: &Baseline) -> Result<(), String> { - if report.count < MIN_SAMPLES { - return Err(format!( - "{} — a 90th percentile needs at least {MIN_SAMPLES} samples behind it and this \ - has {}, so the gate would be reading its own largest sample", - describe(report), - report.count, - )); - } - let Some(ceiling) = baseline.ceiling_ns() else { - return Ok(()); - }; - let p90 = report.quantile_upper_ns(9, 10); - if p90 > ceiling { - return Err(format!( - "{} — this distribution has mass the {} sample never showed: nine passes in ten \ - must be provably under {ceiling} ns and it cannot show that. That line is one \ - bucket above the worst 90th percentile in the recorded sample ({}), and the \ - budget itself ({} ns) is deliberately *not* the line — it is six times looser \ - than anything that sample contains", - describe(report), - baseline.accelerator, - baseline.recorded, - MAX_PASS_NS, - )); - } - Ok(()) -} - -/// Prove the instrument in both directions, with no guest. -/// -/// `serial::self_check`'s shape and its reason: a gate nothing checks is a gate -/// nobody knows is broken. Two of the cases below are the two this design turns -/// on — a host that took the CPU away must pass, and a scheduler whose passes -/// grew must not — and neither can be staged on a booted machine. **They are the -/// same distribution to a maximum and to an `over` count**, which is the point: -/// the pair is what says this gate reads mass rather than extremes, and any -/// repair that starts gating the maximum again fails one of them. -/// -/// The recorded samples are checked against themselves as well, because a -/// ceiling that has drifted from the sample it claims to come from is the one -/// failure this design can suffer silently. -pub fn self_check() -> Result<(), String> { - check_recorded(&KVM)?; - check_recorded(&TCG)?; - - // The line KVM's sample draws, spelled out so that re-recording the sample - // has to move this number deliberately rather than as a side effect. - if KVM.ceiling_ns() != Some(131_072) { - return Err(format!( - "pass-cost gate self-check: the KVM sample's ceiling is {:?} and it was 131072 ns \ - when this gate was written — re-record deliberately or not at all", - KVM.ceiling_ns(), - )); - } - // And the fact that made TCG report-only: its own sample spans four - // buckets. A future sample narrow enough to draw a line from would have to - // come through here to do it. - let tcg_span = TCG.worst_p90_ns() / TCG.p90_ns.iter().copied().min().unwrap_or(1); - if tcg_span < 4 { - return Err(format!( - "pass-cost gate self-check: TCG takes no verdict because host load alone sweeps \ - its recorded sample, and that sample now spans only {tcg_span}x — the reason and \ - the evidence have come apart" - )); - } - - let mut bulk = PassCostReport::empty(toyos_sched::hw::CpuId(0)); - // 100 000 passes at 2048..4096 ns — the shape a scheduler doing scheduling - // rather than work produces. - bulk.buckets[12] = 100_000; - bulk.count = 100_000; - - // The case removing the panic exists for: the same scheduler on a host that - // took the vCPU away, at 1–2 ms a time — ten times the budget, and about - // what a whole cross-arch TCG pass cost when invariant P last fired on the - // dev host. - let mut stolen = bulk; - stolen.buckets[12] -= 2_000; - stolen.buckets[21] = 2_000; - stolen.max_ns = 2_000_000; - stolen.over = 2_000; - - // The case it must still catch, and it is `stolen` with the mass moved: - // one pass in five is now over. Same maximum, same shape of outlier, ten - // times the mass — and only the quantile can tell them apart. - let mut grown = bulk; - grown.buckets[12] -= 20_000; - grown.buckets[21] = 20_000; - grown.max_ns = 2_000_000; - grown.over = 20_000; - - // A limit, asserted on purpose: every pass at 32..64 µs is the worst 90th - // percentile KVM's whole recorded sample contains, and it is accepted - // because the ceiling sits one bucket above it. - let mut sample_worst = PassCostReport::empty(toyos_sched::hw::CpuId(0)); - sample_worst.buckets[16] = 100_000; - sample_worst.count = 100_000; - sample_worst.max_ns = 65_000; - - // Two buckets above that, and still four times *under* `MAX_PASS_NS`. - // **This is the case the budget-shaped gate passed and this one refuses**, - // and it is the whole gain from making the line the sample's rather than - // the policy number's. - let mut over_recorded = PassCostReport::empty(toyos_sched::hw::CpuId(0)); - over_recorded.buckets[18] = 100_000; // every pass in 131 072..262 144 ns - over_recorded.count = 100_000; - over_recorded.max_ns = 260_000; - - let mut short = bulk; - short.buckets[12] = 99; - short.count = 99; - - let cases: &[(&str, bool, PassCostReport)] = &[ - ("a scheduler doing scheduling", true, bulk), - ("the same one on a host that stole its vCPU", true, stolen), - ("the same maximum with one pass in five over the budget", false, grown), - ("every pass at the worst 90th percentile KVM has shown", true, sample_worst), - ("every pass two buckets over that, and still under the budget", false, over_recorded), - ("too few samples to have a 90th percentile", false, short), - ]; - for (name, want_ok, report) in cases { - let got = verdict(report, &KVM); - if got.is_ok() != *want_ok { - return Err(format!( - "pass-cost gate self-check: `{name}` should have been {} against the KVM \ - sample, and it was {}: {got:?}", - if *want_ok { "accepted" } else { "refused" }, - if got.is_ok() { "accepted" } else { "refused" }, - )); - } - } - - // Report-only judges no magnitude and still refuses a broken instrument. - // Both halves, because a mode that refused nothing would hide a kernel that - // had stopped measuring. - if verdict(&grown, &TCG).is_err() { - return Err( - "pass-cost gate self-check: a report-only accelerator refused a distribution on \ - its magnitude, which is the one thing it must not do" - .to_string(), - ); - } - if verdict(&short, &TCG).is_ok() { - return Err( - "pass-cost gate self-check: a report-only accelerator accepted a report of 99 \ - passes — the sample floor is about the instrument having run, not about the \ - quantile, and it holds in both modes" - .to_string(), - ); - } - - // The reader, on the wire form the kernel actually prints: the last line - // per CPU wins, and a CPU that never reported is absent rather than zero. - let mut other = bulk; - other.cpu = toyos_sched::hw::CpuId(1); - let capture = format!( - "[kernel 1.001 cpu0] {}\n\ - [kernel 1.002 cpu1] {}\n\ - [kernel 1.003 cpu0] {}\n\ - hello from userland\n", - short, other, bulk, - ); - let read = reports(&capture); - if read.len() != 2 || read[0] != bulk || read[1] != other { - return Err(format!( - "pass-cost gate self-check: the reader took {} report(s) off a capture with two \ - CPUs and three lines: {read:?}", - read.len(), - )); - } - if !reports("hello from userland\n").is_empty() { - return Err("pass-cost gate self-check: a capture with no report yielded one".to_string()); - } - Ok(()) -} - -/// A recorded sample must be in the instrument's own units, and must pass the -/// line it is used to draw. Either failure makes every verdict below it -/// meaningless while looking exactly like a working gate. -fn check_recorded(baseline: &Baseline) -> Result<(), String> { - if baseline.p90_ns.is_empty() { - return Err(format!( - "pass-cost gate self-check: the {} sample has no observations in it", - baseline.accelerator, - )); - } - for &p90 in baseline.p90_ns { - if !p90.is_power_of_two() { - return Err(format!( - "pass-cost gate self-check: the {} sample records a 90th percentile of \ - {p90} ns, and a quantile off this instrument is a bucket end — so that \ - number was not read off a run", - baseline.accelerator, - )); - } - } - if let Some(ceiling) = baseline.ceiling_ns() { - if baseline.worst_p90_ns() > ceiling { - return Err(format!( - "pass-cost gate self-check: the {} sample's own worst 90th percentile ({} ns) \ - is over the ceiling drawn from it ({ceiling} ns), so the recorded runs would \ - red the gate they justify", - baseline.accelerator, - baseline.worst_p90_ns(), - )); - } - } - Ok(()) -} diff --git a/tests/common/power.rs b/tests/common/power.rs index 34ccec04762..7b16b9f3a93 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -222,27 +222,10 @@ pub fn quiesce_stops_the_machine( WRITERS + OTHERS, )); } - woken_by_its_threads(&record)?; - eprintln!(" [power] the machine stopped before it claimed anything: {record}"); Ok(()) } -/// **The stop is woken by its threads' own transitions**, read off the -/// record's own budget: a stop that stopped everything only once that budget -/// was spent was parked while its threads stopped, and nothing they did woke -/// it. -fn woken_by_its_threads(record: &toyos_quiesce::Record) -> Result<(), String> { - if record.spent_its_budget() { - return Err(format!( - "the stop took its whole {} ms budget to see a machine it had stopped, so nothing \ - its threads did woke it:\n {record}", - record.budget_ms, - )); - } - Ok(()) -} - /// Every [`stopped_boot`] arms it, for the reason `usb_reset_hands_devices_back`'s /// deadline arm does: QEMU has no window between the boot's last word and the /// reset and hardware does, so without it the last-word judge is green whether @@ -314,21 +297,17 @@ fn stopped_boot( .iter() .find(|line| line.contains(toyos_quiesce::STOPPED)) .ok_or_else(|| format!("the kernel wrote no stop record\n{whole}"))?; + // Whether it stopped every thread is not judged here: the stop gives up at + // a budget of the kernel's own clock, so a starved guest reads as the + // defect. Every metal boot is held to it (`metal::Readback::stop_completed`). let record = toyos_quiesce::Record::parse(said) .ok_or_else(|| format!("the kernel's stop record did not read back as one:\n {said}"))?; - if !record.stopped_the_machine() { - return Err(format!( - "the stop gave up on {} thread(s) that never reached a safe point:\n {record}\n{whole}", - record.sweep.running, - )); - } Ok((whole, record)) } -/// **A park that is the stop's last transition wakes it.** `quiesce-last-park` -/// holds a thread inside `SYS_NANOSLEEP` until the stop's latest sweep counts -/// it as the one thread still running, so the park it then makes is the last -/// thing the stop can be woken by. +/// `quiesce-last-park` holds a thread inside `SYS_NANOSLEEP` until the stop's +/// latest sweep counts it as the one thread still running, so the park it then +/// makes is the stop's last transition. pub fn quiesce_wakes_on_the_last_park( _test_config: &Path, _c_bins: &[(String, Vec)], @@ -402,8 +381,7 @@ fn woken_by_the_held_thread( )); } } - woken_by_its_threads(&record)?; - eprintln!(" [power] {actuator}: the held thread's transition woke the stop: {record}"); + eprintln!(" [power] {actuator}: the stop waited on the held thread's transition: {record}"); Ok(()) } @@ -545,10 +523,9 @@ fn loader_window(console: &str) -> Result, String> { Ok(lines[first..=last].iter().map(|line| (*line).to_string()).collect()) } -/// The chipset resets a machine whose kernel stops feeding its watchdog, and -/// `watchdog_fed` is the same guest with the feed on. Starvation begins well -/// after boot, so what is measured is a reset inside this guest's own scaled -/// ceiling once it has, never an arm-to-ready race. +/// The chipset resets a machine whose kernel stops feeding its watchdog. +/// Starvation begins well after boot, so what is waited for is the reset, with +/// this guest's own scaled ceiling behind it, never an arm-to-ready race. pub fn watchdog_resets( test_config: &Path, c_bins: &[(String, Vec)], @@ -569,36 +546,6 @@ pub fn watchdog_resets( Ok(()) } -/// The control: the same guest, feeding, runs past the bound and is still there. -pub fn watchdog_fed( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let options = BootOptions { kernel_params: &["watchdog", "tco-fast"], ..starved() }; - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - let boot = serial::Serial::boot(&qemu); - boot.must_be_clean()?; - // Without this the control cannot tell a fed watchdog from none at all. - boot.must_say(ARMED)?; - - let mut stop = qemu::QmpShutdown::open(qemu.qmp_socket(), FED_FOR); - if let Some(seen) = stop.reason() { - let tail = qemu.drain_serial(WAIT); - return Err(format!( - "QEMU stopped a guest that was feeding its watchdog, for {seen:?}\n{tail}" - )); - } - - let result = qemu.run_test("pwd", Duration::from_secs(30)); - if result.exit_code != Some(0) { - return Err(format!("the guest stopped answering after {FED_FOR:?}: {result:?}")); - } - - eprintln!(" [power] a fed guest ran {FED_FOR:?}, several bounds, and still answers"); - Ok(()) -} - /// The kernel's read-back above its own arm, in /// `kernel/src/arch/x86_64/watchdog.rs`: whole clauses, one per branch. const ARMED_ON_ARRIVAL: &str = "so the bootloader had already armed the timer"; @@ -764,7 +711,7 @@ fn starved() -> BootOptions { } } -/// The line `arm` logs on q35 at the fast bound; both tests demand it first. +/// The line `arm` logs on q35 at the fast bound, demanded before anything is judged. /// /// The tail is what makes it the kernel's: on every guest that passes the /// parameter the loader prints the same port and a `TCO_TMR=` of its own, which @@ -773,14 +720,12 @@ const ARMED: &str = "watchdog: 8086:2918 TCO at 0x660 TCO_TMR=2 — this machine resets if no scheduler pass runs \ for 2400ms"; -const FED_FOR: Duration = Duration::from_secs(20); - /// The kernel's fast panic bound in seconds — `kernel/src/panic_reboot.rs`'s /// `FAST_BOUND`, which `panic-reboot-fast` swaps in for the shipped minute. /// /// A kernel constant does not cross into the harness, so it is written here and /// then *read back*: [`panic_armed`] is the whole arm line including this -/// number, and both tests demand it before they judge anything. A bound that +/// number, demanded before anything is judged. A bound that /// moved in the kernel and not here reds on that line rather than on a stop /// reason nobody could attribute. const PANIC_FAST_SECS: u64 = 5; @@ -815,11 +760,10 @@ const PANICKED_AND_STAYED_UP: &str = anyway"; /// A panicked kernel nobody is at returns the machine to firmware itself. -/// [`panic_key_holds`] is the same guest with a key pressed inside the bound. /// -/// The verdict is QEMU's stop reason arriving inside the bound the arm line -/// names, which is what the budget below is: a reset that had to wait longer -/// than the bound plus what a reset costs is not this bound firing. +/// The verdict is QEMU's stop reason and the panic path's own line saying the +/// bound ran out; the budget below is the ceiling on that reset, never a bound +/// it is held to. pub fn panic_reboots( test_config: &Path, c_bins: &[(String, Vec)], @@ -857,7 +801,7 @@ fn resets_inside_the_bound( returned_to_firmware(reason, never, &tail)?; let drain = serial::Serial::named("panic reboot drain", tail.as_str()); - drain.must_say(PANIC_REBOOTING)?; + drain.must_say(qemu::PANIC_REBOOTING)?; Ok((budget, tail)) } @@ -980,104 +924,11 @@ pub fn panic_before_peripherals_reboots( /// [`panic_reboot::arm`]: kernel/src/panic_reboot.rs const PANIC_HELD_HEAD: &str = "panic: holding this panel"; -/// What the reset itself is allowed to cost on top of the bound: the flush the -/// reset path makes before it writes the register, and the host seeing QEMU's -/// event. Scaled by [`QemuInstance::budget`] at the call site. +/// The ceiling on the reset past the bound: the flush the reset path makes +/// before it writes the register, and the host seeing QEMU's event. Scaled by +/// [`QemuInstance::budget`] at the call site. const RESET_ALLOWANCE: Duration = Duration::from_secs(20); -/// The panic path's second line, written raw because the log is already drained -/// by then (`kernel/src/panic_reboot.rs`'s `reboot_now`). -const PANIC_REBOOTING: &str = "panic: no key inside the bound, so nobody is here"; - -/// The control on [`panic_reboots`]: the same guest with one key pressed inside -/// the bound holds its panel and is still there several bounds later. -/// -/// The key is `a`, not a page key: what retires the bound is that somebody is at -/// the machine, and `screen_pager_keys` is where the pager's own two keys are judged. -pub fn panic_key_holds( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, panicked()); - let boot = serial::Serial::boot(&qemu); - // Demanded before the key, so a control that never armed a bound cannot - // pass by holding a panel nothing was counting down. - boot.must_say(&panic_armed())?; - - let socket = qemu.qmp_socket().to_path_buf(); - qemu::qmp_send_keys(&socket, &[("a", true), ("a", false)]); - - let mut stop = qemu::QmpShutdown::open(qemu.qmp_socket(), qemu.budget(PANEL_HELD_FOR)); - if let Some(seen) = stop.reason() { - let tail = qemu.drain_serial(WAIT); - return Err(format!( - "a key was pressed inside the bound and QEMU stopped this guest anyway, for \ - {seen:?}\n{tail}" - )); - } - - // One monitor per `-qmp` socket, so the shutdown watch is given up before - // the screendump connects. - drop(stop); - // QEMU not exiting is not the claim in this test's name; the panel still - // carrying the report is. The fill and not a line of it, because a key that - // is not a page key leaves the pager unsteered and which page is up when the - // dump is taken is nobody's to say. - let fill = qemu.screendump().fill(); - if fill != FILL_FATAL { - return Err(format!( - "the guest is still up {PANEL_HELD_FOR:?} after the key, but its panel fills \ - {fill:?} and not the fatal {FILL_FATAL:?}: whatever it holds is not the report" - )); - } - - eprintln!(" [power] a key retired the bound and the report held the panel {PANEL_HELD_FOR:?}"); - drop(qemu); - release_alone_does_not_retire(test_config, c_bins, rust_bins) -} - -/// The negative arm on [`panic_key_holds`]: a byte that is not a key **press** -/// leaves the bound armed, and the machine still resets itself. -/// -/// A bare break code is what this injects, because it is the one byte of that -/// class QMP can deliver — a controller's own ACK reaches the port through no -/// monitor command. The claim is the same either way: `read_key` retires on a -/// make code and on nothing else, so a panel nobody pressed a key at is a panel -/// nobody is reading, whatever else the controller said. -fn release_alone_does_not_retire( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, panicked()); - let boot = serial::Serial::boot(&qemu); - boot.must_say(&panic_armed())?; - - let socket = qemu.qmp_socket().to_path_buf(); - qemu::qmp_send_keys(&socket, &[("a", false)]); - - let budget = qemu.budget(Duration::from_secs(PANIC_FAST_SECS) + RESET_ALLOWANCE); - let mut stop = qemu::QmpShutdown::open(qemu.qmp_socket(), budget); - let reason = stop.reason(); - let tail = qemu.drain_serial(WAIT); - returned_to_firmware( - reason, - "a key release retired a bound only a key press may retire, and the guest held its panel", - &tail, - )?; - - eprintln!(" [power] a release alone left the bound armed and the guest reset itself"); - Ok(()) -} - -/// What `panic_console`'s `Fill::Fatal` paints behind a report — the one thing -/// on that panel which does not depend on the page the pager has up. -const FILL_FATAL: [u8; 3] = [0x60, 0x00, 0x00]; - -/// Several bounds, so the control is not a race the guest won once. -const PANEL_HELD_FOR: Duration = Duration::from_secs(PANIC_FAST_SECS * 4); - /// A line of the first boot's own report, which has to come back out of DRAM on /// the boot after it: the panic's message, so what is recovered is the crash /// and not merely a page that checksummed. @@ -1859,8 +1710,8 @@ pub fn done_chain(after: &serial::Serial) -> Result<(), String> { } /// The tail the stop sealed under the seal: the one channel for what the -/// kernel said after `logd` stopped — its stop record, which stopped every -/// thread, the panel's census, and the last word. Read after the seal, +/// kernel said after `logd` stopped — its stop record, the panel's census, and +/// the last word. Read after the seal, /// because the same lines are in the capture on the first boot's console. fn the_tail_is_the_stops(after: &serial::Serial) -> Result<(), String> { let head = after.must_say_after(&done_line(), bootlog::LOG_TAIL_HEAD)?.to_string(); @@ -1870,15 +1721,8 @@ fn the_tail_is_the_stops(after: &serial::Serial) -> Result<(), String> { .skip_while(|line| !line.contains(&head)) .filter(|line| line.contains(bootlog::LOG_TAIL)) .collect(); - let record = tail - .iter() - .find_map(|line| toyos_quiesce::Record::parse(line)) - .ok_or_else(|| format!("the page's tail carries no stop record:\n{}", tail.join("\n")))?; - if !record.stopped_the_machine() { - return Err(format!( - "the stop left {} thread(s) running into the reset:\n {record}", - record.sweep.running, - )); + if !tail.iter().any(|line| toyos_quiesce::Record::parse(line).is_some()) { + return Err(format!("the page's tail carries no stop record:\n{}", tail.join("\n"))); } for owed in [bootlog::PANEL_CENSUS, REBOOTING] { if !tail.iter().any(|line| line.contains(owed)) { diff --git a/tests/common/qemu.rs b/tests/common/qemu.rs index f8a6d81435a..e0c89ebe094 100644 --- a/tests/common/qemu.rs +++ b/tests/common/qemu.rs @@ -21,24 +21,10 @@ pub const SUITE_ARCH: Arch = Arch::X86_64; pub static VERBOSE: AtomicBool = AtomicBool::new(false); /// Distinguishes every file one QEMU boot owns from every other boot's within -/// one test process — the wav capture, the UART log, the QMP socket, the -/// screendump, and the bootable image itself. +/// one test process — the UART log, the QMP socket, the screendump, and the +/// bootable image itself. static BOOT_SEQ: AtomicU32 = AtomicU32::new(0); -/// Guests that have been booted and not yet dropped. -/// -/// Gate A's numbers were recorded with one QEMU on the host and nothing else -/// (`tests/audio-baseline.toml`), so "the parallel phase has drained" is a -/// precondition of the audio block rather than a property of where it sits in -/// `main`. This is what lets it be asserted instead of arranged — see -/// [`live_instances`]. -static LIVE: AtomicU32 = AtomicU32::new(0); - -/// How many guests are up right now, across every thread. -pub fn live_instances() -> u32 { - LIVE.load(Ordering::SeqCst) -} - /// The NVMe backing files live guests are holding open. /// /// A lane reuses one image across its boots on purpose ([`super::lane`]), so @@ -197,8 +183,7 @@ pub fn boot_census() -> (u32, u32, Vec) { /// `kernel/Cargo.toml` has forwarded `sched-check = ["toyos-sched/check"]` since /// the check build was written, and nothing in `src/` or `tests/` ever asked for /// it, so `cpu::MAX_PASS_NS`, the pass-cost recorder and `invariants::check_cpu` -/// were compiled by no CI run at all. `sched_check_build` is the test that asks, -/// and `common::passcost` is what judges the half of it that is a measurement. +/// were compiled by no CI run at all. `sched_check_build` is the test that asks. /// /// A fifth entry is that decision again, and it gets this paragraph's argument /// made afresh. Interactive debug mode is separate: it builds @@ -494,6 +479,10 @@ impl Liveness { /// second producer: a test's own ceiling is a guard of exactly this kind. pub const STALLED: &str = "STALLED:"; +/// The backstop's red: a guest still talking past [`GUEST_WEDGED`] that never +/// finished. The ceiling too, and counted with [`STALLED`]'s. +pub const TIMED_OUT: &str = "timed out after"; + /// How long a guest may say nothing before a wait on it is a stall. /// /// Every config these waits run on has something on a periodic interval — the @@ -630,12 +619,8 @@ impl std::fmt::Display for WaitVerdict { /// `dying` is the line on which the kernel said it was dying, if it ever did, /// and `quiet` is how long the guest has said nothing. **The first arm is the /// whole point.** A Rust `panic!` in the kernel prints `PANIC:` and then -/// `halt_all_cpus` stops every CPU, so the guest goes silent and the ceiling — -/// which is a liveness guard and never a verdict — expired on a machine that -/// had been dead since the first second. `sched_check_build` in run -/// `31946183485` was reported `STALLED: 382s of guard expired` with the panic -/// and its full backtrace four lines above that sentence, on a guest that died -/// at 1.450 s of its own uptime. +/// `halt_all_cpus` stops every CPU, so the guest goes silent and the ceiling +/// expires on a machine that has been dead since the panic. /// /// **The wall clock is not the wedge; silence is.** A test's `ceiling` is the /// budgeted wall clock (`budget_smp`-scaled, so it already carries #256's @@ -685,7 +670,7 @@ pub fn ceiling_verdict( let backstop = ceiling.max(GUEST_WEDGED); if elapsed > backstop { return Some(format!( - "timed out after {}s, with the guest still talking {quiet:.0?} ago ({lines} \ + "{TIMED_OUT} {}s, with the guest still talking {quiet:.0?} ago ({lines} \ console line(s) while it ran) — it was working and did not finish", backstop.as_secs() )); @@ -959,11 +944,6 @@ pub fn ceiling_self_check() -> Result<(), String> { /// `doing` is what the guest was asked to do, in the caller's own words. The /// caller keeps its assertion; what this owns is the difference between "it did /// the wrong thing" and "it never got there". -/// -/// It lives beside [`Liveness`] rather than in the test list because a test in -/// `tests/common/` could not reach it there, and the two that could not — -/// `metal_sim_null_audio` and `hda_two_live_refused` — each reached for a span -/// of host wall clock instead and lost the race on a runner. pub fn await_guest( qemu: &mut QemuInstance, log: &mut String, @@ -1026,6 +1006,72 @@ pub fn await_marker_new( await_guest(qemu, log, doing, |log| log[from.min(log.len())..].contains(marker)) } +/// Per vCPU in `info registers -a`, whether it is halted with interrupts off: +/// the stop's `cli; hlt`, which no interrupt ends. An idle CPU halts with `IF` +/// set, and a running one is not halted, so neither is this. +pub fn stopped_cpus(registers: &str) -> Vec { + registers + .split("CPU#") + .skip(1) + .map(|cpu| { + let field = |name: &str| -> Option { + Some(cpu.split(name).nth(1)?.chars().take_while(char::is_ascii_hexdigit).collect()) + }; + let flags = field("RFL=").and_then(|f| u64::from_str_radix(&f, 16).ok()); + field("HLT=").as_deref() == Some("1") && flags.is_some_and(|f| f & (1 << 9) == 0) + }) + .collect() +} + +/// The fatal path's last line, which `panic_reboot::reboot_now` writes to the +/// 16550 raw just before it resets the machine. +pub const PANIC_REBOOTING: &str = "panic: no key inside the bound, so nobody is here"; + +/// Drain the console into `log` until QEMU exits on the fatal path's reset, or +/// the console says one of `refused`: a line this guest must never write ends +/// the wait at once, and the exit closes a capture that is then whole. +/// +/// **The reset and not a halt**: the CPU that went fatal never halts. It holds +/// its panel under `panic_reboot`'s bound, and the bound's reset is the path's +/// last act, which `-no-reboot` turns into QEMU's exit. So the boot passes +/// `panic-reboot-fast`: its five seconds of silence sit inside [`GUEST_QUIET`], +/// and the shipped minute does not. +pub fn await_reset( + qemu: &mut QemuInstance, + log: &mut String, + doing: &str, + refused: &[&str], +) -> Result<(), String> { + let from = log.len(); + let mut live = guest_liveness(); + loop { + if refused.iter().any(|line| log[from..].contains(line)) { + return Ok(()); + } + match qemu.rx.recv_timeout(Duration::from_millis(200)) { + Ok(line) => { + log.push_str(&line); + log.push('\n'); + } + Err(RecvTimeoutError::Timeout) => {} + Err(RecvTimeoutError::Disconnected) => break, + } + if !live.working(log) { + return Err(format!("{STALLED} waiting for {doing} — {}", live.why())); + } + } + let status = qemu.child.wait().map_err(|e| format!("QEMU could not be waited for: {e}"))?; + // On a machine with a console the line is only in the 16550's own log. + let said = format!("{}{}", &log[from..], qemu.uart_log()); + if !status.success() || !said.contains(PANIC_REBOOTING) { + return Err(format!( + "QEMU exited {status} waiting for {doing}, and not on the fatal path's reset: no \ + {PANIC_REBOOTING:?}\n{said}" + )); + } + Ok(()) +} + /// The hardware shape QEMU presents to the guest. /// /// Not a display setting: each variant is a whole machine. `Headless` is the @@ -1316,16 +1362,8 @@ pub enum Profile { /// 32-bit-destination entry format rather than the 8-bit one. IommuEim, /// [`Profile::Headless`] with its virtio sound card replaced by an Intel - /// HDA controller and one codec — the machine soundd drives itself. - /// - /// Everything else is held still on purpose. The console is still - /// virtio-serial, the NIC is still there, the disks are the same: what - /// differs from the machine gate A's four recorded configs run on is the - /// sound card, so a difference in the capture is a difference in the audio - /// path. It is not the T14's literal shape and does not try to be — this is - /// the audio arm, not a PCI-topology one. H0's diagnostic staged that - /// comparison and is deleted now that the - /// driver above answers every question it was asked for. + /// HDA controller and one codec — the machine soundd drives itself, and + /// the class-0403 function the IOMMU tests aim. Hda, /// [`Profile::Hda`] with a second controller that also has a codec. /// @@ -1464,8 +1502,7 @@ const XHCI_MSI_ONLY: &str = "nec-usb-xhci,id=xhci1,msix=off"; const XHCI_NO_IRQ_FIRST: &str = "nec-usb-xhci,id=xhci,msix=off,msi=off"; const XHCI_NO_IRQ_SECOND: &str = "nec-usb-xhci,id=xhci1,msix=off,msi=off"; -/// One controller with one codec: the ordinary machine, and the one an audio -/// arm needs. `hda-output` because it is a playback-only codec — the driver +/// One controller with one codec. `hda-output` because it is a playback-only codec — the driver /// configures no input path and a duplex codec would only add widgets nothing /// walks. const HDA_ONE: &[&str] = &["intel-hda,id=hda0", "hda-output,bus=hda0.0,cad=0,audiodev=hdaaud"]; @@ -1492,10 +1529,7 @@ enum Virtio { /// /// Not a lesser [`Virtio::Present`]: soundd claims a kernel-driven card /// before it looks for a controller to drive itself, so a machine carrying - /// both would exercise the virtio path and nothing else. This is what makes - /// an HDA arm of gate A a *different machine* rather than a different flag, - /// and it keeps the console, the NIC and the timing of the recorded audio - /// configs so the two arms differ in the sound card and not in the machine. + /// both would exercise the virtio path and nothing else. WithoutSound, } @@ -2557,7 +2591,7 @@ pub struct TestResult { /// /// A caller that reads a daemon's startup out of a boot appends this to its /// capture. It is separate from `serial` because `serial` means "while this - /// test ran" and audio gates count lines in it. + /// test ran". pub before: String, /// Why the run did not finish, when it did not. /// @@ -2628,7 +2662,6 @@ pub struct QemuInstance { rx: Receiver, console: ConsoleStream, _reader_thread: thread::JoinHandle, - audio_wav: PathBuf, uart_log: PathBuf, nvme: NvmeClaim, usb_images: Vec, @@ -3141,15 +3174,12 @@ impl QemuInstance { }) .collect(); - let audio_wav = test_dir.join(format!("audio-{seq}.wav")); - let _ = fs::remove_file(&audio_wav); - let sockets = Sockets::new(&options); let screendump = test_dir.join(format!("screen-{seq}.ppm")); - // Per-instance, not a fixed /tmp path: the audio gate boots dozens of - // guests and a screen test waits on this file, so a shared one would - // let instances read each other's early boot. + // Per-instance, not a fixed /tmp path: a screen test waits on this + // file, so a shared one would let instances read each other's early + // boot. let uart_log = test_dir.join(format!("uart-{seq}.log")); let _ = fs::remove_file(&uart_log); let console_file = options.console_file.then(|| ConsoleFile::of(&uart_log).made()); @@ -3169,7 +3199,6 @@ impl QemuInstance { &boot_image, nvme.path(), &usb_images, - &audio_wav, &uart_log, &sockets.dir, &firmware_vars, @@ -3180,7 +3209,6 @@ impl QemuInstance { &options, Files { seq, - audio_wav, uart_log, nvme, usb_images, @@ -3389,16 +3417,10 @@ impl QemuInstance { self.i8042_trace } - /// The wav file the virtio-sound device records into for this boot. - /// The RIFF size fields stay 0 until QEMU exits cleanly — parse to EOF. - pub fn audio_wav_path(&self) -> &Path { - &self.audio_wav - } - /// Wait for QEMU to exit within `by`: its console closing is the event, and /// the process is reaped after it. Answers what the guest said on the way. - /// A file QEMU finishes only at its exit, the wav among them, is whole once - /// this answers, and is still there until this instance is dropped. + /// A file QEMU finishes only at its exit is whole once this answers, and is + /// still there until this instance is dropped. pub fn await_exit(&mut self, by: Duration) -> Result { let deadline = Instant::now() + by; let mut said = String::new(); @@ -3462,9 +3484,6 @@ impl QemuInstance { } /// Keep collecting serial output for `dur` after a test has returned. - /// soundd flushes its final stats window when the last client leaves, - /// which races the client process's exit — so the line the audio gate - /// reads lands on either side of `===TEST_END===`. /// **Not scaled by the width**, and it is the one duration in this file that /// is not. Callers use it to *pace* — "let the guest run for 400 ms and tell /// me what it said" — so multiplying it does not buy a slow guest more room, @@ -3629,15 +3648,9 @@ impl QemuInstance { // dropped — `TestResult::before` is the argument. let mut before = String::new(); let mut in_test = false; - // **Which of the two things the ceiling caught.** A test's `timeout` is - // a liveness guard and never a verdict, and until now its expiry said - // only how many seconds had passed — `metal_sim_client_death` 364 s, - // `metal_sim_window_drag` 355 s, `desktop_audio_client` 354 s and - // `blocked_dump` 329 s in run `31250706113`, four reds indistinguishable - // from four slow tests. The console tells them apart for free, and the - // fix `1cf7fee` made to the waits *inside* a test never reached this - // one: a guest that has said nothing for [`GUEST_QUIET`] has stopped, - // and one still talking at the ceiling has not. + // **Which of the two things the ceiling caught**: a guest that has said + // nothing for [`GUEST_QUIET`] has stopped, and one still talking at the + // ceiling has not. let mut last_line = Instant::now(); let mut lines = 0usize; // **The line on which the kernel said it was dying, if it ever did.** @@ -3799,7 +3812,6 @@ impl Drop for QemuInstance { // hopeful is that the process whose descriptors hold QEMU's write lock // on the image is gone by the time it happens. let _ = self.child.wait(); - let _ = fs::remove_file(&self.audio_wav); // **The 16550's log outlives the guest, because it is the one channel // that exists before the console does.** 1.4 KB on a // healthy `tests/testcases` boot, measured, against the hundreds of @@ -3817,7 +3829,6 @@ impl Drop for QemuInstance { let _ = fs::remove_file(own); } // `sockets` goes with the fields, after QEMU is reaped. - LIVE.fetch_sub(1, Ordering::SeqCst); } } @@ -4322,7 +4333,7 @@ impl QmpDevices { pub fn profile_argv(options: &BootOptions) -> Vec { let p = Path::new("/nonexistent"); let usb: Vec = options.profile.usb_disks().iter().map(|_| p.to_path_buf()).collect(); - qemu_command(p, p, &usb, p, p, p, p, options) + qemu_command(p, p, &usb, p, p, p, options) .get_args() .map(|a| a.to_string_lossy().into_owned()) .collect() @@ -4342,12 +4353,10 @@ fn stick_file(image: &Path, read_error: Option) -> String { } } -#[allow(clippy::too_many_arguments)] fn qemu_command( boot_image: &Path, nvme_image: &Path, usb_images: &[PathBuf], - audio_wav: &Path, uart_log: &Path, socket_dir: &Path, firmware_vars: &Path, @@ -4642,26 +4651,9 @@ fn qemu_command( } if !shape.hda.is_empty() { - // The same wav backend virtio-sound gets, so gate A's ground truth — - // what the *device* received — transfers with no new instrument. A boot - // that plays nothing leaves an empty file and costs nothing. - // - // **`timer-period` is 1000 µs here and 5000 for virtio-sound, and that - // is an instrument repair rather than a difference in the audio path.** - // At 5000 the capture of a 3 s 440 Hz tone comes back with eight phase - // discontinuities, at frames 2703-2705, 2821-2823 and 2939-2940 — - // *identical positions across six runs whose audio content differed*, - // which is a capture that drops samples on a fixed cadence and not a - // guest that plays them wrong. QEMU's `hda-codec` holds its own output - // ring and discards what overruns it, and shortening the host's drain - // interval is what stops the overrun. Measured on this host, QEMU - // 11.0.3: 8 breaks at 5000, 0 at 1000, with the guest's own counters - // (1127 periods submitted, no underruns, no drains) identical either - // way and identical to the virtio arm's. - qemu.arg("-audiodev").arg(format!( - "wav,id=hdaaud,path={},timer-period=1000", - audio_wav.display() - )); + // No guest test plays audio: the device is here as a DMA master and a + // claim, so its audio goes nowhere. + qemu.arg("-audiodev").arg("none,id=hdaaud"); for dev in shape.hda { qemu.arg("-device").arg(*dev); } @@ -4739,14 +4731,10 @@ fn qemu_command( if shape.virtio.present() { if shape.virtio.sound() { - // virtio-sound records everything the guest plays into a per-boot - // wav for glitch analysis; timer-period matches the interactive - // config in src/qemu.rs so test timing represents what users hear. + // No guest test plays audio: the device is here as a DMA master and + // a claim, so its audio goes nowhere. qemu.arg("-audiodev") - .arg(format!( - "wav,id=audio0,path={},timer-period=5000", - audio_wav.display() - )) + .arg("none,id=audio0") .arg("-device") .arg(format!("virtio-sound-pci,audiodev=audio0,streams=1{platform}")); } @@ -4821,7 +4809,6 @@ fn socket_names( /// parameter list eight paths long. struct Files { seq: u32, - audio_wav: PathBuf, uart_log: PathBuf, nvme: NvmeClaim, usb_images: Vec, @@ -4896,7 +4883,6 @@ impl Read for Followed { fn spawn_and_wait_ready(mut qemu: Command, options: &BootOptions, files: Files) -> QemuInstance { let Files { seq, - audio_wav, uart_log, nvme, usb_images, @@ -5000,16 +4986,11 @@ fn spawn_and_wait_ready(mut qemu: Command, options: &BootOptions, files: Files) wait_for_ready(&mut child, &rx, options, &uart_log) }; - // Counted from here rather than from the spawn: every panic inside - // `wait_for_ready` kills the child on its way out and never builds a value - // to drop, so a guest that failed to come up must not be left on the books. - LIVE.fetch_add(1, Ordering::SeqCst); QemuInstance { child, stdin, rx, _reader_thread: reader_thread, - audio_wav, uart_log, nvme, usb_images, diff --git a/tests/common/stats.rs b/tests/common/stats.rs deleted file mode 100644 index 7f4a90527b9..00000000000 --- a/tests/common/stats.rs +++ /dev/null @@ -1,120 +0,0 @@ -//! Two-sample tests for the audio gate's thorough tier. -//! -//! Every thorough-tier decision compares the **fresh sample** against the -//! **recorded baseline sample** — never against a fitted constant. That matters -//! more than it sounds: a threshold derived from 30 clean runs carries the -//! sampling error of those 30 runs, and a one-sample test against it states a -//! confidence it does not have. Measured on this tree, a sign test against the -//! recorded median of `max_wake_lat_us` has a nominal false-red rate of 0.07% -//! and a real one near 1.5%, purely because the reference median moves by an -//! order statistic. A two-sample test carries that uncertainty in the maths. -//! -//! Two tests, one per kind of observation: -//! -//! * `mann_whitney_z` for the counters (continuous, heavy right tails, no -//! distributional assumption survives them). -//! * `fisher_greater` for the yes/no outcomes (did this run drop audio, did -//! it breach a per-run ceiling, did it fail to complete). -//! -//! Both are one-sided in the direction of "worse". A scheduler change that -//! improves audio timing must not fail the gate. - -/// Per-test significance level. One in a thousand, so the ~19 tests a thorough -/// run performs still union-bound under 2%, and the measured family-wise -/// false-red rate on a clean tree is 0.25% (2000 simulated runs against the -/// recorded distributions). Deliberately not a tunable: a gate whose alpha can -/// be raised is a gate that will be raised. -pub const ALPHA: f64 = 0.001; - -/// One-sided standard-normal critical value for `ALPHA`. -pub const Z_CRIT: f64 = 3.0902; - -/// Mann-Whitney U, one-sided, as a normal-approximation z score: how strongly -/// `test` is stochastically *greater* than `base`. Ties get midranks and the -/// tie-corrected variance, which matters here — `underruns` and `drains` are -/// small integers with many repeats. -/// -/// The normal approximation is used rather than the exact permutation -/// distribution because at n1 = n2 = 30 it is already accurate and, measured by -/// bootstrap against the recorded samples, conservative: 3 rejections in 240000 -/// trials under H0 against a nominal 0.1%. -pub fn mann_whitney_z(base: &[f64], test: &[f64]) -> f64 { - let n1 = base.len() as f64; - let n2 = test.len() as f64; - assert!(n1 >= 2.0 && n2 >= 2.0, "Mann-Whitney needs both samples"); - - let mut all: Vec = base.iter().chain(test.iter()).copied().collect(); - all.sort_by(|a, b| a.partial_cmp(b).expect("audio counters are never NaN")); - - // Midrank of each distinct value, and the tie-group sizes. - let mut rank_of: Vec<(f64, f64)> = Vec::new(); - let mut tie_term = 0.0; - let mut i = 0; - while i < all.len() { - let mut j = i; - while j + 1 < all.len() && all[j + 1] == all[i] { - j += 1; - } - let group = (j - i + 1) as f64; - rank_of.push((all[i], (i + j) as f64 / 2.0 + 1.0)); - tie_term += group * group * group - group; - i = j + 1; - } - let rank = |v: f64| { - rank_of - .binary_search_by(|(k, _)| k.partial_cmp(&v).unwrap()) - .map(|idx| rank_of[idx].1) - .expect("value came from the pooled sample") - }; - - let r2: f64 = test.iter().map(|&v| rank(v)).sum(); - let u2 = r2 - n2 * (n2 + 1.0) / 2.0; - let n = n1 + n2; - let var = n1 * n2 / 12.0 * ((n + 1.0) - tie_term / (n * (n - 1.0))); - if var <= 0.0 { - // Every observation identical: no evidence of anything. - return 0.0; - } - (u2 - n1 * n2 / 2.0) / var.sqrt() -} - -/// Fisher's exact test, one-sided: the probability of seeing at least `k1` -/// events in `n1` fresh runs when the fresh runs share the baseline's rate, -/// given `k0` of `n0` in the baseline. Exact, so it stays honest at the tiny -/// counts these rates produce (4 dropped runs in 117). -pub fn fisher_greater(k1: u32, n1: u32, k0: u32, n0: u32) -> f64 { - let (k1, n1, k0, n0) = (k1 as usize, n1 as usize, k0 as usize, n0 as usize); - assert!(k1 <= n1 && k0 <= n0); - let total = n1 + n0; - let events = k1 + k0; - let ln_fact = ln_factorials(total); - let ln_choose = |n: usize, k: usize| ln_fact[n] - ln_fact[k] - ln_fact[n - k]; - let denom = ln_choose(total, events); - let hi = n1.min(events); - let mut p = 0.0; - for x in k1..=hi { - if events - x > n0 { - continue; - } - p += (ln_choose(n1, x) + ln_choose(n0, events - x) - denom).exp(); - } - p.min(1.0) -} - -/// Smallest event count in `n1` fresh runs that Fisher rejects at `ALPHA`. -/// `None` when even `n1` of `n1` would not reject — which is itself worth -/// printing, because it means the sample is too small to test that rate at all. -pub fn fisher_reject_at(n1: u32, k0: u32, n0: u32) -> Option { - (0..=n1).find(|&k| fisher_greater(k, n1, k0, n0) <= ALPHA) -} - -fn ln_factorials(n: usize) -> Vec { - let mut out = Vec::with_capacity(n + 1); - out.push(0.0); - let mut acc = 0.0; - for i in 1..=n { - acc += (i as f64).ln(); - out.push(acc); - } - out -} diff --git a/tests/common/swap.rs b/tests/common/swap.rs index 8ca0a32fab2..2c50ba7d871 100644 --- a/tests/common/swap.rs +++ b/tests/common/swap.rs @@ -53,20 +53,10 @@ struct Rig { } impl Rig { - /// `binary` sent as netd's replacement once sshd answers, and the - /// machine's word on it, let go at once: `logd` serves its port before - /// netd leases, and sshd may still be binding then. - fn swap_once_sshd_answers(&self, binary: &Path, digest: &toyos_swap::Digest) -> Result { - let asked = std::time::Instant::now(); - loop { - match self.ssh.swap(self.forward, "netd", binary, digest).and_then(|answered| answered.go()) { - Err(why) if asked.elapsed() < Duration::from_secs(30) => { - eprintln!(" [swap] not taken yet: {why}"); - std::thread::sleep(Duration::from_secs(1)); - } - answered => return answered.map(|a| a.said.clone()), - } - } + /// `binary` sent as netd's replacement, and the machine's word on it, let + /// go at once. + fn swap_netd(&self, binary: &Path, digest: &toyos_swap::Digest) -> Result { + self.ssh.swap(self.forward, "netd", binary, digest).and_then(|answered| answered.go()).map(|a| a.said.clone()) } fn boot(name: &str, bench: Bench) -> Result { @@ -80,7 +70,12 @@ impl Rig { let options = BootOptions { ssh_port: Some(ssh_port), ..staged.options() }; let mut guest = QemuInstance::boot_with_options(&staged.case, &[], &[], options); let mut console = guest.boot_log().to_string(); - qemu::await_marker(&mut guest, &mut console, super::logstream::SERVING, "logd to open its port")?; + // `logd` serves its port before netd leases, and sshd binds only after. + for (marker, doing) in + [(super::logstream::SERVING, "logd to open its port"), ("sshd: listening on port 22", "sshd to listen")] + { + qemu::await_marker(&mut guest, &mut console, marker, doing)?; + } let stream = super::logstream::reader(staged.log_port, &format!("{name}-stream.txt"))?; let ssh = Ssh::at(&super::compile::repo_root(), staged.identity.private().to_path_buf())?; let forward = SocketAddr::from((Ipv4Addr::LOCALHOST, ssh_port)); @@ -411,21 +406,18 @@ const RESET_NOTHING: &[&str] = &["pcidev-reset-nothing"]; /// last of them. const KNOCKS: usize = 25; -/// A liveness guard on those connects, never a verdict. -const KNOCKS_WITHIN: Duration = Duration::from_secs(30); - /// netd swapped for `replacement` (the test binary `name`), and from the moment /// it says `holding` — its part mastering — [`KNOCKS`] SYNs sent through slirp -/// at the guest's address. Answers how many slirp completed and how long they -/// took; `Err` is a why the caller fails the rig with. -fn swap_and_knock(rig: &mut Rig, rust_bins: &[(String, Vec)], name: &str, holding: &str) -> Result<(usize, Duration), String> { +/// at the guest's address. Answers how many slirp completed; `Err` is a why the +/// caller fails the rig with. +fn swap_and_knock(rig: &mut Rig, rust_bins: &[(String, Vec)], name: &str, holding: &str) -> Result { use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; use std::sync::Arc; let replacement = test_binary(rust_bins, name)?; let binary = rig.staged.scratch.join(name); std::fs::write(&binary, replacement).map_err(|e| format!("{}: {e}", binary.display()))?; - let answer = rig.swap_once_sshd_answers(&binary, &toyos_swap::digest(replacement)); + let answer = rig.swap_netd(&binary, &toyos_swap::digest(replacement)); eprintln!(" [swap] the swap was answered {answer:?}"); init_accepted(&answer)?; qemu::await_marker(&mut rig.guest, &mut rig.console, holding, "the replacement holding the part mastering")?; @@ -443,21 +435,13 @@ fn swap_and_knock(rig: &mut Rig, rust_bins: &[(String, Vec)], name: &str, ho } }) }; - let asked = std::time::Instant::now(); let knocked = qemu::await_guest(&mut rig.guest, &mut rig.console, "the host's frames at the part", |_| { - taken.load(Ordering::SeqCst) >= KNOCKS || asked.elapsed() > KNOCKS_WITHIN + taken.load(Ordering::SeqCst) >= KNOCKS }); stop.store(true, Ordering::SeqCst); let _ = knocking.join(); knocked?; - let taken = taken.load(Ordering::SeqCst); - if taken < KNOCKS { - return Err(format!( - "slirp took {taken} of {KNOCKS} connects in {KNOCKS_WITHIN:?}, so the window held too few \ - frames for a clean console to mean anything" - )); - } - Ok((taken, asked.elapsed())) + Ok(taken.load(Ordering::SeqCst)) } /// **The part keeps running across a release, and the next holder stops it @@ -479,7 +463,7 @@ pub fn swap_quiets_the_function( rust_bins: &[(String, Vec)], ) -> Result<(), String> { let mut rig = Rig::boot("swap-quiet", super::lan::TALK_BENCH)?; - let (taken, took) = match swap_and_knock(&mut rig, rust_bins, IDLE, HOLDING) { + let taken = match swap_and_knock(&mut rig, rust_bins, IDLE, HOLDING) { Ok(knocked) => knocked, Err(why) => return Err(rig.fail(why)), }; @@ -491,7 +475,7 @@ pub fn swap_quiets_the_function( return Err(rig.fail(format!("{why}\n the release said: {}", released.trim_end()))); } eprintln!( - " [swap] {}; {}; the next holder stopped it, mastered it through {taken} SYNs in {took:?}, and \ + " [swap] {}; {}; the next holder stopped it, mastered it through {taken} SYNs, and \ the unit saw no fault", released.trim_end(), inherited.trim_end(), @@ -523,7 +507,7 @@ pub fn swap_keeps_what_nothing_reset( rust_bins: &[(String, Vec)], ) -> Result<(), String> { let mut rig = Rig::boot_armed("swap-residue", super::lan::TALK_BENCH, RESET_NOTHING)?; - let (taken, took) = match swap_and_knock(&mut rig, rust_bins, RUNNING, RUNNING_HOLDING) { + let taken = match swap_and_knock(&mut rig, rust_bins, RUNNING, RUNNING_HOLDING) { Ok(knocked) => knocked, Err(why) => return Err(rig.fail(why)), }; @@ -547,7 +531,7 @@ pub fn swap_keeps_what_nothing_reset( console.must_be_clean()?; let taken_over = console.must_say("pcidev: slot 0 holds 1 range(s)")?; eprintln!( - " [swap] {}; {}; {}; mastered through {taken} SYNs in {took:?}, and the unit saw no fault", + " [swap] {}; {}; {}; mastered through {taken} SYNs, and the unit saw no fault", released.trim_end(), inherited.trim_end(), taken_over.trim_end(), @@ -586,15 +570,15 @@ const HOLDER_FAULT: &str = "iommu: DMA FAULT owner=slot"; /// this host's SYNs make the part fetch a descriptor there, and the unit /// refuses it. Nothing but that fault can wake the program. The verdict is its /// own line: the claim refused the interrupt read with `Io`. Without the -/// refusal and the wake it earns, the program waits out its bound and says it -/// was told nothing. +/// refusal and the wake it earns, the program waits, and the harness ceiling reds +/// the boot. pub fn swap_fault_tells_its_holder( _test_config: &Path, _c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { let mut rig = Rig::boot("swap-astray", super::lan::TALK_BENCH)?; - let (taken, took) = match swap_and_knock(&mut rig, rust_bins, ASTRAY, ASTRAY_HOLDING) { + let taken = match swap_and_knock(&mut rig, rust_bins, ASTRAY, ASTRAY_HOLDING) { Ok(knocked) => knocked, Err(why) => return Err(rig.fail(why)), }; @@ -615,7 +599,7 @@ pub fn swap_fault_tells_its_holder( let faults = text.matches(HOLDER_FAULT).count(); console.must_be_clean_apart_from(HOLDER_FAULT, faults)?; eprintln!( - " [swap] {}; {}; after {taken} SYNs in {took:?}", + " [swap] {}; {}; after {taken} SYNs", fault.trim_end(), told.trim_end(), ); @@ -672,7 +656,7 @@ pub fn swap_resets_the_function( let probe = test_binary(rust_bins, FLR_PROBE)?; let binary = rig.staged.scratch.join(FLR_PROBE); std::fs::write(&binary, probe).map_err(|e| format!("{}: {e}", binary.display()))?; - let answer = rig.swap_once_sshd_answers(&binary, &toyos_swap::digest(probe)); + let answer = rig.swap_netd(&binary, &toyos_swap::digest(probe)); eprintln!(" [swap] the swap was answered {answer:?}"); if let Err(why) = init_accepted(&answer) { return Err(rig.fail(why)); @@ -756,7 +740,7 @@ fn refused_device_fails(name: &str, actuators: &'static [&'static str]) -> Resul let mut rig = Rig::boot_armed(name, IGB_BENCH, actuators)?; let binary = rebuilt("netd", &rig.staged.scratch)?; let digest = toyos_swap::digest(&std::fs::read(&binary).map_err(|e| e.to_string())?); - let answer = rig.swap_once_sshd_answers(&binary, &digest); + let answer = rig.swap_netd(&binary, &digest); eprintln!(" [swap] the swap was answered {answer:?}"); if let Err(why) = init_accepted(&answer) { return Err(rig.fail(why)); @@ -829,16 +813,8 @@ pub fn swap_not_inherited( let probe = test_binary(rust_bins, PROBE)?; let remote = format!("/tmp/{PROBE}"); let (host, port) = (super::ssh::HOST, rig.forward.port()); - let asked = std::time::Instant::now(); - loop { - match super::ssh::ssh_put(host, port, &rig.staged.identity, &remote, probe) { - Err(why) if asked.elapsed() < Duration::from_secs(30) => { - eprintln!(" [swap] not taken yet: {why}"); - std::thread::sleep(Duration::from_secs(1)); - } - Err(why) => return Err(rig.fail(why)), - Ok(()) => break, - } + if let Err(why) = super::ssh::ssh_put(host, port, &rig.staged.identity, &remote, probe) { + return Err(rig.fail(why)); } let command = format!("{remote} {} {} {}", toyos_swap::PORT, toyos_swap::LABEL, toyos_swap::MSG_SWAP); let ran = match super::ssh::ssh_exec(host, port, &rig.staged.identity, &command) { diff --git a/tests/common/update.rs b/tests/common/update.rs index a2944254c1d..bb0c95e01f2 100644 --- a/tests/common/update.rs +++ b/tests/common/update.rs @@ -300,28 +300,20 @@ pub fn update_boots_the_new_kernel(_: &Path, _: &[(String, Vec)], _: &[(Stri let (mut guest, mut console) = rig.boot()?; owed(&console, 0, &format!("{SLOT_RECORD} A, the one the slot table marks"))?; - let asked = Instant::now(); let (status, said) = rig.install(&next)?; - let installed = asked.elapsed(); if status != Some(0) || !said.contains(&format!("update: installed version {NEXT} in slot B")) { return Err(format!("`update` ended {status:?} saying {said:?}")); } let (from, uart) = rig.reboot_until(&mut guest, &mut console, &format!("{SLOT_RECORD} B, the one the slot table marks"))?; await_machine(&mut guest, &mut console, "the new slot's ready marker", |c| c[from..].contains(DEFAULT_READY))?; - let booted = asked.elapsed(); loader_said(&guest, uart, &format!("Anti-rollback floor: {BASE}, raised from 0 by the boot that proved it"))?; loader_said(&guest, uart, &format!("Slot B: {VERIFIED}"))?; loader_said(&guest, uart, &format!("Kernel: {next_kernel} bytes"))?; - for line in guest.uart_log()[uart..].lines().filter(|l| l.contains("TSC cycles") || l.contains("Loader TSC")) { - eprintln!(" [update] loader: {line}"); - } eprintln!( - " [update] {} bytes installed in {} ms; slot B's kernel ({next_kernel} bytes, the base's {}) \ - at its ready marker {} ms after `update` was asked", + " [update] {} bytes installed; slot B's kernel ({next_kernel} bytes, the base's {}) at its \ + ready marker", std::fs::metadata(&next).map(|m| m.len()).unwrap_or(0), - installed.as_millis(), rig.base_kernel, - booted.as_millis() ); // **A record the running system forged proves nothing**: slot B's own diff --git a/tests/common/usb.rs b/tests/common/usb.rs index 60e3e22a1c9..dd21fa8dac3 100644 --- a/tests/common/usb.rs +++ b/tests/common/usb.rs @@ -12,7 +12,6 @@ use std::io::{Read, Seek, SeekFrom, Write}; use std::path::{Path, PathBuf}; -use std::thread; use std::time::{Duration, SystemTime, UNIX_EPOCH}; use super::qemu::{self, BootOptions, Profile, QemuInstance}; @@ -817,17 +816,25 @@ fn optional_flush_keeps_the_log( // Mid-run and polled, exactly as `kernel_log_file` does it: the claim is that // the sink is still running, and the only place that is visible is the - // device while the machine is up. - let deadline = std::time::Instant::now() + Duration::from_secs(10); - let mut on_device; + // device while the machine is up. The ceiling is the harness's: a stick + // with no write cache that cost the machine its log is a hang here. + let give_up = std::time::Instant::now() + qemu.budget(qemu::GUEST_WEDGED); loop { - on_device = String::from_utf8_lossy( + let on_device = String::from_utf8_lossy( &super::volumes::newest_log(&image_path, start, len)?.1, ) .into_owned(); - if on_device.contains("Boot: complete") || std::time::Instant::now() >= deadline { + if on_device.contains("Boot: complete") { break; } + if std::time::Instant::now() >= give_up { + return Err(format!( + "{} waiting for `Boot: complete` in the log on a stick with no write cache: {} \ + bytes there", + qemu::STALLED, + on_device.len() + )); + } std::thread::sleep(Duration::from_millis(50)); } @@ -861,14 +868,6 @@ fn optional_flush_keeps_the_log( if log.contains("logd: /log has not answered") { return Err(format!("logd gave up on a stick that is working\n{log}")); } - if !on_device.contains("Boot: complete") { - return Err(format!( - "the log on the device stops before `Boot: complete` at {} bytes — a stick with no \ - write cache cost the machine its log", - on_device.len() - )); - } - let after = super::volumes::newest_log(&image_path, start, len)?.1; let after = String::from_utf8_lossy(&after).into_owned(); // `/log` ends at init's stop line: the kernel's own last word comes after @@ -889,6 +888,33 @@ fn optional_flush_keeps_the_log( Ok(()) } +/// Ask test-runner to run `name`, a binary no image carries, and wait for its +/// answer. The spawn's refusal is a kernel record +/// (`spawn: /system/bin/: not found`), which is the probe's load; any other +/// answer ends the wait red, naming it. +fn absent_probe( + qemu: &mut QemuInstance, + console: &mut String, + name: &str, + after: &str, +) -> Result<(), String> { + let asked = console.len(); + writeln!(qemu.stdin_mut(), "run {name}").expect("write to QEMU stdin"); + qemu.flush_stdin(); + let answer = format!("===TEST_END {name} "); + let refused = format!("===TEST_END {name} error=entity not found==="); + qemu::await_guest(qemu, console, &format!("test-runner's answer to {name} {after}"), |c| { + c[asked..].contains(&answer) + })?; + match console[asked..].lines().find(|line| line.contains(&answer)) { + Some(line) if line.contains(&refused) => Ok(()), + line => Err(format!( + "test-runner answered {name} {after} with {line:?}, and a name no image carries is \ + answered {refused:?}" + )), + } +} + /// Boot with a stick whose flush genuinely fails. The writer says so once and /// stops, rather than writing the device that just refused it. /// @@ -924,6 +950,8 @@ fn failed_flush_stops_once( /// failure is a kernel record, and the record is something logd then tries /// to write. The loop is in the coupling and not in either half. const BOUND: usize = 4; + /// logd's word as it gives up, naming the sync as the call that refused it. + const GAVE_UP: &str = "logd: /log has not answered (the sync"; let mut qemu = QemuInstance::boot_with_options( test_config, @@ -936,13 +964,12 @@ fn failed_flush_stops_once( }, ); let mut boot = qemu.boot_log().to_string(); - // Long enough for the give-up to be reachable, and *driven* rather than - // waited out: each probe names a binary that is not there, which commits a - // kernel record, which is what gives logd something to fail to write. + // The give-up first, then the probes after it, each *driven* and awaited: + // each names a binary that is not there, which commits a kernel record, + // which is what gives logd something to fail to write. + qemu::await_marker(&mut qemu, &mut boot, GAVE_UP, "logd to give up on the sync")?; for i in 0..PROBES { - let _ = writeln!(qemu.stdin_mut(), "run flush-probe-{i}"); - qemu.flush_stdin(); - boot.push_str(&qemu.drain_serial(Duration::from_millis(500))); + absent_probe(&mut qemu, &mut boot, &format!("flush-probe-{i}"), "after the give-up")?; } writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); qemu.flush_stdin(); @@ -960,7 +987,7 @@ fn failed_flush_stops_once( // By step and not by code alone: logd names which of the two calls refused // it, and the one this test stages is the sync rather than the append ahead // of it. - let gave_up = log.matches("logd: /log has not answered (the sync").count(); + let gave_up = log.matches(GAVE_UP).count(); if gave_up != 1 { return Err(format!( "logd gave up {gave_up} times, wanted exactly one — a failed `SYS_FSYNC` has to \ @@ -995,24 +1022,6 @@ fn stamp_of(log: &str, needle: &str) -> Result { secs.parse::().map_err(|e| format!("timestamp {secs:?}: {e}")) } -/// That the wait between `from` and `refusal` really was the transfer budget. -/// -/// Without this the gate would stay green for a `settles` that gave up on its -/// first read: QEMU answers every one of these registers before the deadline is -/// ever consulted, so no other test in the suite would notice either, and a -/// driver that refused every controller after zero nanoseconds would ship. -fn waited_out_the_budget(log: &str, from: &str, refusal: &str) -> Result { - let waited = stamp_of(log, refusal)? - stamp_of(log, from)?; - // The budget is 2 s and the serial stamps have millisecond resolution. - if waited < 1.5 { - return Err(format!( - "the refusal came {waited:.3} s after {from:?}; the wait is supposed to be the 2 s \ - transfer budget, so this driver gave up without waiting" - )); - } - Ok(waited) -} - /// A controller and a port that stop answering, which on the machine this is /// for is a silent hang and nothing else. /// @@ -1052,8 +1061,6 @@ pub fn xhci_deaf_registers( if !log.contains("Boot: complete") { return Err(format!("the boot did not finish without its USB controller\n{log}")); } - let controller_wait = waited_out_the_budget(&log, "xHCI: found at PCI", "it never halted") - .map_err(|e| format!("{e}\n{log}"))?; // And the port, which is the wait an ordinary machine can actually reach: // a device pulled between the port scan and the reset lands here. @@ -1082,12 +1089,9 @@ pub fn xhci_deaf_registers( if !log.contains("Boot: complete") { return Err(format!("the boot did not finish past a port that would not reset\n{log}")); } - let port_wait = waited_out_the_budget(&log, " connected", "never finished its reset") - .map_err(|e| format!("{e}\n{log}"))?; eprintln!( - " [usb] a controller that will not halt is refused by name after {controller_wait:.3} s; \ - {skipped} port(s) that will not reset are skipped after {port_wait:.3} s; both machines \ - reach `Boot: complete`" + " [usb] a controller that will not halt is refused by name; {skipped} port(s) that will \ + not reset are skipped; both machines reach `Boot: complete`" ); Ok(()) } @@ -1123,23 +1127,6 @@ pub fn xhci_slow_connect( rust_bins: &[(String, Vec)], ) -> Result<(), String> { const PARAMS: &[&str] = &["usb-storage-gate", "xhci-slow-connect"]; - // The driver's own durations, from where the driver reads them: each is - // declared once in `toyos-xhci` and `use`d by `kernel/src/drivers/xhci`, so - // neither bound below can be a copy that drifted. - use toyos_xhci::port::{DEBOUNCE_NS, EMPTY_BUS_NS, SLOW_CONNECT_NS}; - /// Nanoseconds per second, to put those in the units the log's stamps are in. - const PER_S: f64 = 1_000_000_000.0; - /// How long after port power the driver can first name a port. The register - /// reads empty for `SLOW_CONNECT_NS` after this controller powered its - /// ports, and `await_connect_settle` then wants `DEBOUNCE_NS` of a connect - /// set that has held still and is non-empty. - const FIRST_CONNECT_S: f64 = (SLOW_CONNECT_NS + DEBOUNCE_NS) as f64 / PER_S; - /// How late the first port line may be: halfway between the two settles this - /// test tells apart, so one that ends on the device appearing cannot reach - /// it and one that ends at `EMPTY_BUS_NS` cannot stay under it. - const SETTLE_CEILING_S: f64 = - FIRST_CONNECT_S + (EMPTY_BUS_NS as f64 / PER_S - FIRST_CONNECT_S) / 2.0; - let (bytes, lba) = Profile::UsbDisk.usb_disk().expect("UsbDisk declares a disk"); let image = test_dir().join("usb-slow-connect.img"); let nonce = stage(&image, bytes); @@ -1156,54 +1143,6 @@ pub fn xhci_slow_connect( }, )?; - // The driver looked at an empty bus and kept looking. Without this the test - // would be green on a driver that never waits and a QEMU that answers - // instantly, which is exactly the pair that shipped. - // - // `powered_at` is not logged; the two lines that bracket it are. - // `controller started` is taken before it, so it is the anchor that can only - // make the floor generous; the port-power line is printed after it, so it is - // the anchor that can only make the ceiling generous. Neither bound can be - // false because of where in the bracket the instant actually fell. - let started = stamp_of(&log, "xHCI: controller started")?; - let powered = stamp_of(&log, "root-hub ports powered")?; - // The first line this driver prints about any port at all. Every other - // per-port line is preceded by that port's connect line, so the first match - // is the first connect whichever port register it lands on — which the - // profile does not fix, since a SuperSpeed stick appears on a high one. - let first_seen = stamp_of(&log, "xHCI: port ")?; - - // What makes the bracket the settle's own: `await_connect_settle` anchors on - // the greatest `powered_at` across controllers, which on a second controller - // is not the instant these two lines bracket. - if !log.contains("xHCI: 1 controller(s),") { - return Err(format!( - "this profile grew a second controller, so the settle no longer anchors on the \ - `powered_at` these lines bracket\n{log}" - )); - } - // The floor, and the non-vacuity with it: a driver that did not wait, or an - // injection that did not land, names a port within a millisecond of the scan - // rather than after the held-empty window and the debounce behind it. - let after_start = first_seen - started; - if after_start < FIRST_CONNECT_S { - return Err(format!( - "the first port was named {after_start:.3} s after the controller started, inside the \ - {FIRST_CONNECT_S} s the held-empty window and the debounce behind it come to — the \ - injection did not reach the driver\n{log}" - )); - } - // The ceiling. - let after_power = first_seen - powered; - if after_power > SETTLE_CEILING_S { - return Err(format!( - "the first port was named {after_power:.3} s after the ports were powered, {:.3} s \ - after the connect became visible — the settle did not end on the device \ - appearing\n{log}", - after_power - FIRST_CONNECT_S - )); - } - // And it found everything, and the bytes are the host's. if !log.contains("usb-storage: 2 device(s)") { return Err(format!("the driver did not bind both sticks after the wait\n{log}")); @@ -1214,38 +1153,13 @@ pub fn xhci_slow_connect( return Err(format!("the guest did not report a clean pass\n{log}")); } verify(&image, bytes, nonce)?; - // The guest's own boot stamp, printed rather than asserted on. - // - // **This is the log architecture's own named boot instrument for the - // producer path's cost, and until 2026-08-15 it could not be read off the - // test that *is* it.** What it is for is an interleaved A/B of that cost - // against this boot's `Boot: complete`, and the stamp reached only the - // per-run UART file, which goes when the guest does. So the measurement had - // to instrument something, and **the reading taken on an instrumented build - // is the one that misleads**: measuring the 2026-08-08 log-ring regression, - // an uninstrumented A/B reproduced it 5 times out of 5 (497-504 ms against - // 812-839), and a third build — the same source with a timing `log!` added - // to `boot_checkpoint` — measured 500 ms and no regression at all, because - // the extra call changed inlining enough to move the cost where the - // instrument could not see it. **Bisect the source when that happens, and - // interleave the arms**: the first uncontrolled A/B there ran all of one arm - // and then all of the other, the host settled in between, and a reproducible - // 350 ms regression read as host noise. One line of output, - // `i8042_absent`'s arrangement, and the obligation is re-runnable by - // anybody. It decides nothing: what is asserted is that the boot finished, - // which is the `else` below. - let Some(boot_ms) = toyos_build::bootlog::boot_millis(&log) else { + if toyos_build::bootlog::boot_millis(&log).is_none() { return Err(format!("the boot did not finish\n{log}")); - }; + } serial::Serial::named("boot console", log.as_str()).must_be_clean()?; let _ = std::fs::remove_file(&image); - eprintln!( - " [usb] controller started at {started:.3} s and powered its ports at {powered:.3} s; \ - first port named at {first_seen:.3} s, {after_start:.3} s after the start and \ - {after_power:.3} s after the power, both sticks bound, host bytes verified host-side; \ - Boot: complete at {boot_ms} ms" - ); + eprintln!(" [usb] both sticks bound after the held-empty window, host bytes verified host-side"); Ok(()) } @@ -1469,19 +1383,6 @@ pub fn usb_transport_break( // `SCSI 0x35`, slot 1, 2.3 s after the gate had swept — pushed the total // from the injected disk's real 2 to 3 and reddened a run in which the // disk under test never left its budget. - // The kernel's own clock against the claim, on a channel the message does - // not write: every record carries `[kernel ]`, so a wait that had - // really spent `USB_TIMEOUT_NS` would put two seconds between the break and - // the record before it. Measured here rather than asserted from the wording. - let waited = elapsed_before(&log, staged[0])?; - if waited >= 2.0 { - return Err(format!( - "the staged break took {waited:.3} s, which is the transfer budget — the wait it \ - is supposed to skip really ran\n{log}" - )); - } - eprintln!(" [usb] the staged break waited {waited:.3} s, not the 2 s it used to claim"); - let under_test = broke_on(staged[0])?; // And the driver got over it. Two attempts are explained by the fault — the @@ -1666,14 +1567,6 @@ fn a_read_whose_first_wait_spent_its_budget_goes_out_again( } let broke = line_with(&log, "transport broke on SCSI 0x28: no answer in the ")?; let under_test = broke_on(broke)?; - // The staged wait really spent the operation's two seconds. - let waited = elapsed_before(&log, broke)?; - if waited < 1.9 { - return Err(format!( - "{broke:?} came {waited:.3} s after the record before it: the staged wait did not \ - spend the operation's budget\n{log}" - )); - } let (_, after) = log.split_once(broke).expect("the line came from this text"); let mut rest = after; for needle in [ @@ -1693,8 +1586,8 @@ fn a_read_whose_first_wait_spent_its_budget_goes_out_again( serial::Serial::named("boot console", log.as_str()).must_be_clean()?; let _ = std::fs::remove_file(&image); eprintln!( - " [usb] {under_test}: a READ whose wait spent {waited:.3} s went out again after the port \ - reset took, and returned the host's bytes" + " [usb] {under_test}: a READ whose wait spent the operation's budget went out again \ + after the port reset took, and returned the host's bytes" ); Ok(()) } @@ -1758,17 +1651,10 @@ enum Moved { FlushedStick, /// The stick itself, which `usb-return-silent` has come back late in the /// held call and answer nothing on the operation sent again on it: every - /// wait of it spins to its end, and the call still ends inside its bound, - /// timed from the break. + /// wait of it spins to its end. SilentReturn, } -/// What one disk call may spin for from the wait its transport broke on, in -/// seconds: the bound `toyos_xhci::call` keeps every path of a call inside. -fn call_after_break_secs() -> f64 { - toyos_xhci::call::AFTER_BREAK.whole() as f64 / 1e9 -} - /// The boot stick's first WRITE(10) is abandoned, the ladder resets its port, /// and the device leaves that port and binds on another. **QEMU cannot move a device on a reset**, so `usb-reset-moves` /// holds the port rung's reset until the port reads empty and the host makes @@ -1782,8 +1668,7 @@ fn call_after_break_secs() -> f64 { /// device left was before. /// /// **The call held for the stick only waits.** It spins with `IF` clear, so -/// the bind is another CPU's, and the call ends inside [`call_after_break_secs`] of -/// the break however slow the bind is — both read off the kernel's own stamps. +/// the bind is another CPU's, read off the kernel's own stamps. fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { const HELD: &str = "is held empty for the host to move its device (usb-reset-moves)"; const MOVE_NOW: &str = "usb-reset-moves: move the device now"; @@ -1849,17 +1734,21 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { devices.blockdev_add_again("moved", &image); devices.add("usb-storage", "xhci.0", "movedstick", &[("drive", "moved"), ("port", "3"), ("serial", serial)]); drop(devices); - let ends = |l: &str| match moved { - Moved::SameStick | Moved::SlowStick | Moved::FlushedStick => { - l.contains(toyos_build::bootlog::REBOOTING) - } - Moved::AnotherStick => l.contains(" did not come back within "), - Moved::OwedFlush => l.contains(FLUSH_LOST), - Moved::SilentReturn => l.contains(WENT_OUT_AGAIN) || l.contains(STILL_HELD), + // The last line each shape's verdict reads, and the end of the held call, + // which may come on either side of it. + let ends = |c: &str| { + let last = match moved { + Moved::SameStick | Moved::SlowStick | Moved::FlushedStick => { + c.contains(toyos_build::bootlog::REBOOTING) + } + Moved::AnotherStick => c.contains(" did not come back within "), + Moved::OwedFlush => c.contains(FLUSH_LOST), + Moved::SilentReturn => c.contains(&format!("{WENT_OUT_AGAIN}: it failed")), + }; + last && (c.contains(WENT_OUT_AGAIN) || c.contains(STILL_HELD)) }; - log.push_str(&qemu.drain_until(Duration::from_secs(60), ends)); - // The rest of the boot: the job's reset, or the lost disk's failures. - log.push_str(&qemu.drain_serial(Duration::from_secs(5))); + qemu::await_guest(&mut qemu, &mut log, "the moved stick's last line", ends) + .map_err(|why| format!("{moved:?}: {why}\n{log}"))?; drop(qemu); let _ = std::fs::remove_file(&image); @@ -1909,17 +1798,9 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { .ok_or_else(|| format!("{moved:?}: the call held on cpu{staged_cpu} never ended\n{log}"))?; // Where it ended, and when. A bind while it held ran on another CPU; one // after it ended may run on any. - let held_call = |bind: &str| -> Result { + let held_call = |bind: &str| -> Result<(), String> { let line = held_end; let ended = stamp_of(line, "[kernel ")?; - let took = ended - stamp_of(&log, staged)?; - let bound = call_after_break_secs(); - if took > bound { - return Err(format!( - "{moved:?}: the call its transport broke in ended {took:.3} s after the break, past \ - the {bound} s a call may spin for\n{log}" - )); - } let (call_cpu, bind_cpu) = (cpu_of(line)?, cpu_of(line_with(&log, bind)?)?); let during = stamp_of(&log, bind)? <= ended; if during && call_cpu == bind_cpu { @@ -1929,11 +1810,11 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { )); } eprintln!( - " [usb] {moved:?}: the held call ended {took:.3} s after the break on cpu{call_cpu}; \ - the stick was bound on cpu{bind_cpu}, {} it ended", + " [usb] {moved:?}: the held call ended on cpu{call_cpu}; the stick was bound on \ + cpu{bind_cpu}, {} it ended", if during { "before" } else { "after" } ); - Ok(took) + Ok(()) }; let came_back = "usb-storage: disk 0 came back on port 3 slot "; match moved { @@ -2039,20 +1920,10 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { operation again on the stick that came back\n{log}" )); } - let took = held_call(came_back)?; - let bounds = toyos_xhci::call::AFTER_BREAK; - let spun = (bounds.whole() - bounds.offline) as f64 / 1e9; - if took < spun { - return Err(format!( - "{moved:?}: the call ended {took:.3} s after the break, before the {spun} s its \ - waits may reach: the operation sent again did not spend what the call had \ - left\n{log}" - )); - } + held_call(came_back)?; eprintln!( " [usb] a stick that came back and answered nothing on the operation sent again \ - on it spun what the held call had left, went offline on the last rung, and the \ - call ended {took:.3} s after the break" + on it went offline on the last rung" ); } Moved::AnotherStick => { @@ -2199,31 +2070,6 @@ fn abandoned_write_is_taken_offline( return Err(format!("slot {slot} never went back after the give-up\n{log}")); } - // The clock, against the staging's claim: each rung really spent its bound - // before the next began, so the last one ran with the others' all gone. - let stamp = |l: &str| -> Option { - l.split_once("[kernel ")?.1.split_whitespace().next()?.parse().ok() - }; - let at = |needle: &str| log.lines().find(|l| l.contains(needle)).and_then(stamp); - let (Some(broke), Some(unverified), Some(ended)) = - (stamp(staged), at(rungs[0].as_str()), stamp(said)) - else { - return Err(format!("a rung's record carries no kernel timestamp\n{log}")); - }; - if unverified - broke < 1.4 { - return Err(format!( - "the port reset's rung took {:.3} s, so the staging did not spend its bound and the \ - last rung was not starved of anything\n{log}", - unverified - broke - )); - } - if ended - broke >= 2.75 { - return Err(format!( - "the ladder took {:.3} s from the break to the offline line; the two bounds it \ - ran on sum to 2 s\n{log}", - ended - broke - )); - } no_command_was_refused(&log)?; if !log.contains("Boot: complete") { return Err(format!("the boot did not finish after the give-up\n{log}")); @@ -2231,9 +2077,7 @@ fn abandoned_write_is_taken_offline( serial::Serial::named("boot console", log.as_str()).must_be_clean()?; let _ = std::fs::remove_file(&image); eprintln!( - " [usb] {under_test}: a port reset that spent {:.3} s and never verified left the last \ - rung whole: {}", - unverified - broke, + " [usb] {under_test}: a port reset that never verified left the last rung whole: {}", said.split_once("is offline: ").map_or("", |(_, rest)| rest) ); Ok(()) @@ -2589,7 +2433,10 @@ fn transport_gives_up( // the line that ends it. wait_for(&mut qemu, &mut plugged, "would not answer INQUIRY"); // And the poll after it, which is where the refused slot goes back. - plugged.push_str(&qemu.drain_serial(Duration::from_millis(1200))); + let inquiry = plugged.find("would not answer INQUIRY").unwrap_or(0); + qemu::await_guest(&mut qemu, &mut plugged, "the refused slot to go back", |c| { + c[inquiry..].lines().any(|l| l.contains("xHCI: slot ") && l.ends_with(" disabled")) + })?; writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); qemu.flush_stdin(); let log = format!("{boot}{plugged}{}", qemu.drain_serial(Duration::from_secs(20))); @@ -2930,31 +2777,6 @@ fn transport_gives_up( /// transport break" is not recoverable from a line that does not say which disk /// it was, and matching every disk's line instead is how this test came to red /// on a boot stick's own clean recovery. -/// How long the guest's own clock says passed between `line` and the kernel -/// record before it. -/// -/// The timestamp in a record's `[kernel ]` prefix is a channel the -/// message text does not write, which is the whole point of reading it: a claim -/// about how long a wait took is checkable against the clock that timed it. -fn elapsed_before(log: &str, line: &str) -> Result { - let stamp = |l: &str| -> Option { - l.split_once("[kernel ")?.1.split_whitespace().next()?.parse().ok() - }; - let lines: Vec<&str> = log.lines().collect(); - let at = lines - .iter() - .position(|l| *l == line) - .ok_or_else(|| format!("{line:?} is not a line of the log it came from"))?; - let now = stamp(line) - .ok_or_else(|| format!("{line:?} carries no kernel timestamp to measure against"))?; - let before = lines[..at] - .iter() - .rev() - .find_map(|l| stamp(l)) - .ok_or_else(|| format!("no kernel record precedes {line:?}, so nothing times it"))?; - Ok(now - before) -} - fn broke_on(line: &str) -> Result<&str, String> { line.split_once("usb-storage: ") .and_then(|(_, rest)| rest.split_once(" transport broke")) @@ -3520,23 +3342,30 @@ pub fn usb_refused_disk_first( let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); let boot = qemu.boot_log().to_string(); + let mut log = boot.clone(); // Pull the disk the driver refused, and put a disk it can use where it was. // Nothing else on this machine can free that pool block, so the bind below // is the whole assertion. + let pulled = log.len(); let mut devices = qemu::QmpDevices::open(qemu.qmp_socket()); devices.del(&qemu::usb_device_id(0)); drop(devices); - thread::sleep(Duration::from_millis(1200)); + qemu::await_guest(&mut qemu, &mut log, "the refused disk's port to disconnect", |c| { + c[pulled..].contains(" disconnected") + })?; + let plugged = log.len(); let mut devices = qemu::QmpDevices::open(qemu.qmp_socket()); devices.blockdev_add("replacement", &replacement); devices.add("usb-storage", "xhci.0", "replacement0", &[("drive", "replacement")]); drop(devices); - thread::sleep(Duration::from_millis(1200)); + qemu::await_guest(&mut qemu, &mut log, "the replacement disk to bind or be refused", |c| { + c[plugged..].contains("usb-storage: disk 1 ready on slot") || c[plugged..].contains("this driver serves 2") + })?; writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); qemu.flush_stdin(); - let log = format!("{boot}{}", qemu.drain_serial(Duration::from_secs(20))); + log.push_str(&qemu.drain_serial(Duration::from_secs(20))); drop(qemu); for bad in ["PANIC:", "panicked at"] { if log.contains(bad) { @@ -3739,17 +3568,14 @@ pub fn usb_boot_stick_pulled( _rust_bins: &[(String, Vec)], ) -> Result<(), String> { let _ = test_config; - /// Probes sent before the pull, and after it. The drumbeat is the liveness - /// signal as well as the load: each one is a userland `println!` into the - /// ring the log sink drains to the stick, and a VFS walk for a binary that - /// is not there. + /// Probes sent before the pull, and after it, each once the one before it + /// was answered. The drumbeat is the liveness signal as well as the load: + /// each one is a userland `println!` into the ring the log sink drains to + /// the stick, and a VFS walk for a binary that is not there. A machine that + /// pauses while the driver tears the port down and then carries on answers + /// every one; one that never carries on is what the ceiling reds. const BEFORE: usize = 12; const AFTER: usize = 40; - /// How many of the probes after the pull have to come back. Not all of - /// them: a machine that pauses while the driver tears the port down and - /// then carries on has survived, and this gate's subject is a machine that - /// never carries on. - const ANSWERED: usize = 8; // metalcase's machine shape with `/system/bin/logd` rotating at 256 bytes rather // than a mebibyte, so the log writer is not just appending when the device @@ -3780,13 +3606,8 @@ pub fn usb_boot_stick_pulled( // The machine has to be drawing before it is asked to survive anything, or // a green run is a boot that never started. - let deadline = std::time::Instant::now() + Duration::from_secs(20); - while std::time::Instant::now() < deadline && frames(&console) < 1 { - console.push_str(&qemu.drain_serial(Duration::from_millis(250))); - } - if frames(&console) < 1 { - return Err(format!("the compositor never composited a frame:\n{console}")); - } + qemu::await_guest(&mut qemu, &mut console, "the compositor's first frame", |c| frames(c) >= 1) + .map_err(|why| format!("{why}\n{console}"))?; // And it has to be writing to the stick, or the pull is a disconnect with // nothing in flight — which is not the state the owner's machine is in. @@ -3799,22 +3620,32 @@ pub fn usb_boot_stick_pulled( )); } - let probe = |qemu: &mut QemuInstance, console: &mut String, i: usize| { - let _ = writeln!(qemu.stdin_mut(), "run pull-probe-{i}"); - qemu.flush_stdin(); - console.push_str(&qemu.drain_serial(Duration::from_millis(120))); - }; - - for i in 0..BEFORE { - probe(&mut qemu, &mut console, i); - } - let answered_before = console.matches("===TEST_END pull-probe-").count(); - if answered_before < BEFORE / 2 { - return Err(format!( - "only {answered_before} of {BEFORE} probes came back *before* the pull, so the \ - drumbeat this gate measures does not work on a healthy machine:\n{console}" - )); + /// Every probe answered, and the compositor two frame batches on, or the + /// ceiling's red with QEMU's account of the vCPUs. + fn drumbeat( + qemu: &mut QemuInstance, + console: &mut String, + probes: std::ops::Range, + after: &str, + ) -> Result<(), String> { + let frames = |text: &str| text.matches("compositor: frames=").count(); + let from = console.len(); + for i in probes { + if let Err(why) = absent_probe(qemu, console, &format!("pull-probe-{i}"), after) { + let report = crate::freeze_report(qemu, console); + return Err(format!("{why}\n{console}\n{report}")); + } + } + if let Err(why) = qemu::await_guest(qemu, console, &format!("two frame batches {after}"), |c| { + frames(&c[from..]) >= 2 + }) { + let report = crate::freeze_report(qemu, console); + return Err(format!("{why}\n{console}\n{report}")); + } + Ok(()) } + + drumbeat(&mut qemu, &mut console, 0..BEFORE, "before the pull")?; // The rotation actually ran, so "the writer was busy" is a fact rather than // a manifest row that might have been dropped. if !console.contains("logd: /log/") || !console.contains("and this boot continues in") { @@ -3824,52 +3655,26 @@ pub fn usb_boot_stick_pulled( )); } - let pulled_at = console.len(); let mut devices = qemu::QmpDevices::open(&socket); devices.del(qemu::BOOT_STICK_ID); drop(devices); - - for i in BEFORE..BEFORE + AFTER { - probe(&mut qemu, &mut console, i); - } - let after = &console[pulled_at.min(console.len())..]; - let answered = after.matches("===TEST_END pull-probe-").count(); - let drawn = frames(after); - if answered < ANSWERED || drawn < 2 { - return Err(format!( - "the boot stick was pulled and the machine answered {answered} of {AFTER} console \ - probes and composited {drawn} frame batches after it (want {ANSWERED} and 2) — it \ - stopped:\n{console}\n{}", - crate::freeze_report(&mut qemu, &mut console) - )); - } + drumbeat(&mut qemu, &mut console, BEFORE..BEFORE + AFTER, "after the boot stick was pulled")?; // And the same stick put back. The owner reports the freeze from a replug // as well as from a pull, and the two are different states: a replug binds // a new disk under mounts that still name the old one. let replug = test_dir().join("usb-replug.img"); drop(sparse(&replug, 512 * 1024 * 1024)); - let replugged_at = console.len(); let mut devices = qemu::QmpDevices::open(&socket); devices.blockdev_add("replug", &replug); devices.add("usb-storage", "xhci.0", "replug0", &[("drive", "replug")]); drop(devices); - - for i in BEFORE + AFTER..BEFORE + 2 * AFTER { - probe(&mut qemu, &mut console, i); - } - let after_replug = &console[replugged_at.min(console.len())..]; - let answered_replug = after_replug.matches("===TEST_END pull-probe-").count(); - let drawn_replug = frames(after_replug); - if answered_replug < ANSWERED || drawn_replug < 2 { - return Err(format!( - "a stick was plugged back into the port the boot stick was pulled from and the \ - machine answered {answered_replug} of {AFTER} console probes and composited \ - {drawn_replug} frame batches after it (want {ANSWERED} and 2) — it stopped:\ - \n{console}\n{}", - crate::freeze_report(&mut qemu, &mut console) - )); - } + drumbeat( + &mut qemu, + &mut console, + BEFORE + AFTER..BEFORE + 2 * AFTER, + "after a stick went back into the port the boot stick was pulled from", + )?; for bad in ["PANIC:", "panicked at"] { if console.contains(bad) { @@ -3880,8 +3685,8 @@ pub fn usb_boot_stick_pulled( eprintln!( " [usb] the boot stick was pulled out from under a running desktop with the log sink \ - rotating: {answered}/{AFTER} console probes answered and {drawn} frame batches after \ - the pull, {answered_replug}/{AFTER} and {drawn_replug} after a stick went back in" + rotating: all {AFTER} console probes answered and the desktop drew on after the pull, \ + and again after a stick went back in" ); Ok(()) } diff --git a/tests/common/volumes.rs b/tests/common/volumes.rs index ace7fd983a0..1e3938ebf0e 100644 --- a/tests/common/volumes.rs +++ b/tests/common/volumes.rs @@ -549,26 +549,30 @@ pub fn kernel_log_file( // Mid-run, with the guest still up and nothing shut down. Whatever is here // was put there by `/system/bin/logd` while the machine was running. // - // Polled rather than read once, because the claim is "promptly", not - // "instantly": the ready marker is printed by a userland process and logd - // is another one, so a single read races a window the design does not - // promise to close. Ten seconds is three orders of magnitude above what a - // working logd needs — the measurement below says what it actually took — - // so a broken one still reds. - let deadline = std::time::Instant::now() + Duration::from_secs(10); - let began = std::time::Instant::now(); + // Polled until logd has written through `Boot: complete`, with the + // harness's ceiling and no deadline of this test's own: when logd writes + // is its own business, and one that never does is a hang. + let give_up = std::time::Instant::now() + qemu.budget(qemu::GUEST_WEDGED); let mut running; let mut running_text; let mut running_name; loop { (running_name, running) = newest_log(&image_path, start, len)?; running_text = String::from_utf8_lossy(&running).into_owned(); - if running_text.contains("Boot: complete") || std::time::Instant::now() >= deadline { + if running_text.contains("Boot: complete") { break; } + if std::time::Instant::now() >= give_up { + return Err(format!( + "{} waiting for logd to write `Boot: complete` to the device: {} bytes there, \ + starting {:?}", + qemu::STALLED, + running.len(), + running_text.chars().take(120).collect::() + )); + } std::thread::sleep(Duration::from_millis(50)); } - let took = began.elapsed(); if !running_text.contains(&nonce) { return Err(format!( "the log on the device does not carry this boot's partition GUID ({nonce:?}); it is \ @@ -577,21 +581,13 @@ pub fn kernel_log_file( running_text.chars().take(120).collect::() )); } - if !running_text.contains("Boot: complete") { - return Err(format!( - "the log on the device stops before `Boot: complete` at {} bytes — logd wrote \ - once when it opened the file and never again", - running.len() - )); - } if running_text.contains("Shutting down.") { return Err("the guest shut down before the mid-run read".to_string()); } eprintln!( - " [log] {running_name}: {} bytes on the device {} ms after the ready marker, with the \ - machine still running and through `Boot: complete`", + " [log] {running_name}: {} bytes on the device, with the machine still running and \ + through `Boot: complete`", running.len(), - took.as_millis() ); writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); @@ -1483,8 +1479,8 @@ pub fn redirty_mid_flush( /// A truncate staged inside a flush's `update_metadata` window /// (`ftruncate-flush-stall`), which `SYS_FTRUNCATE`'s lockless resize once -/// landed in. The guest makes the race and asserts its truncate serialised; -/// the shut-down volume is re-judged by the FAT reader and the fatgen103 +/// landed in. The guest makes the race and the kernel says whether a truncate +/// landed inside the window; the shut-down volume is re-judged by the FAT reader and the fatgen103 /// checker — `fs_rename_durable`'s oracle. pub fn ftruncate_flush_race( test_config: &Path, @@ -1526,8 +1522,7 @@ pub fn ftruncate_flush_race( } if result.exit_code != Some(0) { return Err(format!( - "ftruncate_flush_race guest failed — the truncate did not serialise with the \ - stalled flush:\n{}\nkernel log while it ran:\n{}{}", + "ftruncate_flush_race guest failed:\n{}\nkernel log while it ran:\n{}{}", result.stdout, result.before, result.serial )); } @@ -1578,8 +1573,8 @@ pub fn ftruncate_flush_race( let _ = std::fs::remove_file(&image_path); eprintln!( - " [ftruncate] the truncate waited out the stalled window in-guest; {SHORT} bytes on \ - the device by the host's own reader, checker silent" + " [ftruncate] the stalled window held; {SHORT} bytes on the device by the host's own \ + reader, checker silent" ); Ok(()) } @@ -2734,7 +2729,6 @@ pub fn log_partition_identity( }, ); let mut log = qemu.boot_log().to_string(); - log.push_str(&qemu.drain_serial(Duration::from_millis(500))); writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); qemu.flush_stdin(); log.push_str(&qemu.drain_serial(Duration::from_secs(20))); diff --git a/tests/common/wallclock.rs b/tests/common/wallclock.rs index a8a18503cd6..d04bb3416ea 100644 --- a/tests/common/wallclock.rs +++ b/tests/common/wallclock.rs @@ -42,22 +42,21 @@ use super::volumes::{self, Entry}; const RTC_BASE: &str = "2033-03-07T09:14:25"; const RTC_BASE_SECS: i64 = 1_993_799_665; /// What a file named for that instant begins with. The date and not the time, -/// because the seconds move on while the machine boots and -/// [`MAX_BOOT_DRIFT_SECS`] is what bounds that — the timestamp inside the entry -/// is where this is checked to the second. +/// because the seconds move on while the machine boots and [`after_the_base`] +/// is what bounds that — the timestamp inside the entry is where this is +/// checked to the second. const RTC_BASE_DATE: &str = "2033-03-07-"; -/// How far past [`RTC_BASE`] the guest may have got by the time it names its -/// log file. -/// -/// The guest's clock runs from the base at host speed, so this is a boot's own -/// wall-clock duration and nothing else — a machine that reaches its log sink -/// in under a second on a quiet host, and in a good deal more on a busy one. -/// Five minutes is three orders of magnitude above the former, which keeps this -/// from being a wall-clock margin the way the serial-tail tests' verdicts are: -/// nothing in the phase can slow a boot by that much, and a kernel that got the -/// century wrong is out by a hundred years rather than by seconds. -const MAX_BOOT_DRIFT_SECS: i64 = 300; +/// Whether `secs` past [`RTC_BASE`] is a time this guest's clock can have read: +/// not before the instant the host staged, and not after the RTC — which runs +/// from that instant at the host's own pace from the moment QEMU starts — had +/// got to by the time the boot was read back, `lived` after the launch. Both +/// ends are causality and neither is a margin: a slower host only widens the +/// second, and a kernel that got the century or a zone wrong is out by years or +/// hours. +fn after_the_base(secs: i64, lived: Duration) -> bool { + (0..=lived.as_secs_f64().ceil() as i64).contains(&secs) +} /// What [`boot_and_read`] makes the guest print into the window between the /// ready marker and the first test it runs. @@ -116,7 +115,7 @@ fn boot_and_read( image_name: &str, params: &'static [&'static str], stage: &[(String, Vec)], -) -> Result<(Vec, String), String> { +) -> Result<(Vec, String, Duration), String> { let image_path = super::lane::dir().join(image_name); let mut image = qemu::build_boot_image(test_config, c_bins, rust_bins, params); std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; @@ -127,6 +126,8 @@ fn boot_and_read( std::fs::write(&image_path, &image).map_err(|e| format!("rewrite the boot image: {e}"))?; } + // Before the launch, so the RTC the guest reads has run no longer than this. + let launched = std::time::Instant::now(); let mut qemu = QemuInstance::boot_with_options( test_config, c_bins, @@ -185,7 +186,7 @@ fn boot_and_read( } let entries = volumes::root_entries(&after[start..start + len])?; let _ = std::fs::remove_file(&image_path); - Ok((entries, log)) + Ok((entries, log, launched.elapsed())) } /// A firmware-named zone separates local time from UTC, in the direction UEFI @@ -206,7 +207,7 @@ pub fn zone_from_firmware( /// What `clock::init_wall` stages, in seconds: two hours east of UTC. const OFFSET_SECS: i64 = -120 * 60; - let (entries, log) = + let (entries, log, lived) = boot_and_read(test_config, c_bins, rust_bins, "wall-clock-zone.img", PARAMS, &[])?; let logs = logs(&entries); @@ -223,7 +224,7 @@ pub fn zone_from_firmware( )); } let stamp_drift = only.modified - RTC_BASE_SECS; - if !(0..=MAX_BOOT_DRIFT_SECS).contains(&stamp_drift) { + if !after_the_base(stamp_drift, lived) { return Err(format!( "a zone moved this boot's FAT timestamp by {stamp_drift}s, and FAT stores local time" )); @@ -236,7 +237,7 @@ pub fn zone_from_firmware( )); }; let drift = epoch - (RTC_BASE_SECS + OFFSET_SECS); - if !(0..=MAX_BOOT_DRIFT_SECS).contains(&drift) { + if !after_the_base(drift, lived) { let unshifted = epoch - RTC_BASE_SECS; return Err(format!( "with firmware naming -120 minutes, `SYS_CLOCK_EPOCH` answered {epoch}: {drift}s from \ @@ -263,7 +264,7 @@ pub fn undated( params: &'static [&'static str], because: &str, ) -> Result<(), String> { - let (entries, log) = boot_and_read(test_config, c_bins, rust_bins, image_name, params, &[])?; + let (entries, log, _) = boot_and_read(test_config, c_bins, rust_bins, image_name, params, &[])?; let logs = logs(&entries); // The refusal, by name and with its reason. A kernel that silently took @@ -329,7 +330,7 @@ pub fn no_century( // *next* century separates them: honouring the table gives 2033 and // ignoring it gives 2133. const PARAMS: &[&str] = &["rtc-no-century", "rtc-century-next"]; - let (entries, log) = + let (entries, log, _) = boot_and_read(test_config, c_bins, rust_bins, "wall-clock-no-century.img", PARAMS, &[])?; let logs = logs(&entries); @@ -370,7 +371,7 @@ pub fn century_from_the_register( rust_bins: &[(String, Vec)], ) -> Result<(), String> { const PARAMS: &[&str] = &["rtc-century-next"]; - let (entries, log) = + let (entries, log, _) = boot_and_read(test_config, c_bins, rust_bins, "wall-clock-century.img", PARAMS, &[])?; let logs = logs(&entries); diff --git a/tests/doomcase/system.toml b/tests/doomcase/system.toml deleted file mode 100644 index f9b7b2411c6..00000000000 --- a/tests/doomcase/system.toml +++ /dev/null @@ -1,43 +0,0 @@ -# doom beside soundd, and nothing else that makes noise. -# -# `doom_sound_flood` drives `/system/bin/doom --sound-stress`, whose producer is the -# game's own sound module — the only thing that can reach it is the binary that -# owns it, so the binary has to be in the image. It is in this config rather -# than in `tests/testcases` because doom is 4 MiB and every other test boots -# that one. -# -# No `assets`: the actuator synthesises its own sound and never opens the WAD -# or the soundfont, so neither belongs on this ROOT either. - -[boot] -start = ["logd", "soundd", "test-runner"] - -# **Every image that carries a `TOYOS-LOG` partition runs this**, and every -# image does. The kernel keeps the record ring and writes no file at all, so a -# boot config without `logd` is a boot whose `/log` is empty — -# `every_boot_config_runs_logd` is what refuses one. -# It claims no device and serves no port: its row's authority is `logread`, -# which is `Rights::LOG | Rights::WAIT` on a `SysCap` duplicate, and init hands -# it every program's output beside that. -[programs.logd] -service = true -syscap = ["logread"] - -[programs.soundd] -service = true -serves = ["soundd"] -devices = ["hda-audio", "virtio-sound"] -syscap = ["rt"] - -# test-runner passes its whole namespace to the binaries it spawns, so -# it holds what doom needs. No compositor here — doom's `--sound-stress` opens -# no window. -# `logread` is the log gate's own: a spawned test binary does not inherit a -# `SysCap` dup, so the gate that reads the kernel's records runs inside -# `test-runner` itself. -[programs.test-runner] -receives = ["soundd"] -syscap = ["logread"] - -[programs.doom] -receives = ["soundd"] diff --git a/tests/doommusiccase/system.toml b/tests/doommusiccase/system.toml index c2e5ecfe924..f75d630351f 100644 --- a/tests/doommusiccase/system.toml +++ b/tests/doommusiccase/system.toml @@ -1,14 +1,7 @@ # doom, soundd and the assets doom's music is made of — and the WAD whose demo # `doom_frames` replays, hashing every tic's frame. # -# `tests/doomcase` is deliberately asset-free — its actuator synthesises its own -# sound and opens neither the WAD nor the SoundFont — and this one is the -# opposite: what it exists to check is that `assets/soundfont.sf2` and -# `assets/DOOM1.WAD` reach ROOT and that doom finds both. So the two cannot -# share a config, and this one pays 19 MiB of ROOT that no other test should. -# -# No compositor: `/system/bin/doom --music-check` and `--frame-check` never open -# a window. +# No compositor: `/system/bin/doom --frame-check` never opens a window. assets = ["assets"] @@ -32,8 +25,7 @@ serves = ["soundd"] devices = ["hda-audio", "virtio-sound"] syscap = ["rt"] -# No compositor: `/system/bin/doom --music-check` opens no window. test-runner passes -# its namespace to doom. +# test-runner passes its namespace to doom. # `logread` is the log gate's own: a spawned test binary does not inherit a # `SysCap` dup, so the gate that reads the kernel's records runs inside # `test-runner` itself. diff --git a/tests/logstallcase/system.toml b/tests/logstallcase/system.toml index 7ad28116336..d6abb204fee 100644 --- a/tests/logstallcase/system.toml +++ b/tests/logstallcase/system.toml @@ -1,8 +1,8 @@ # The shared boot's shape with `/system/bin/logd` leaving soundd's log ring # unread until `test_rs_soundd_log_stall` says the tone has played: the ring # fills and stays full while it plays, which is the state its mix thread must -# never wait in. `soundd_log_stall` runs it, and reads the verdict off `/log` after -# `run shutdown`. +# never wait in. The `soundd_log_stall` metal row runs it, and reads the verdict +# off `/log`. [boot] start = ["logd", "soundd", "test-runner"] @@ -21,12 +21,8 @@ serves = ["soundd"] devices = ["hda-audio", "virtio-sound"] syscap = ["rt"] -# `power` is what `run shutdown` asks init through. [programs.test-runner] -receives = ["soundd", "power"] +receives = ["soundd"] [programs.toybox] receives = ["soundd"] - -[symlinks] -"bin/shutdown" = "/system/bin/toybox" diff --git a/tests/metal-profile.toml b/tests/metal-profile.toml index c7636091b61..8118a7fd0fc 100644 --- a/tests/metal-profile.toml +++ b/tests/metal-profile.toml @@ -361,6 +361,30 @@ ceiling = 60000 ceiling_from = "toyos_tco::JOB_BOUND_MS — as boot.testcases.complete_ms" measured = 1254 +[[number]] +name = "boot.logstallcase.complete_ms" +unit = "ms" +ceiling = 60000 +ceiling_from = "as boot.foreignrecord.complete_ms" + +[[number]] +name = "boot.testcases-deaf.complete_ms" +unit = "ms" +ceiling = 60000 +ceiling_from = "as boot.foreignrecord.complete_ms" + +[[number]] +name = "boot.testcases-window.complete_ms" +unit = "ms" +ceiling = 60000 +ceiling_from = "as boot.foreignrecord.complete_ms" + +[[number]] +name = "boot.testcases-debug.complete_ms" +unit = "ms" +ceiling = 60000 +ceiling_from = "as boot.foreignrecord.complete_ms" + [[number]] name = "boot.foreignrecord.back_secs" unit = "s" @@ -368,6 +392,30 @@ ceiling = 420 ceiling_from = "toyos_build::metal::return_secs" measured = 44 +[[number]] +name = "boot.logstallcase.back_secs" +unit = "s" +ceiling = 420 +ceiling_from = "as boot.foreignrecord.back_secs" + +[[number]] +name = "boot.testcases-deaf.back_secs" +unit = "s" +ceiling = 420 +ceiling_from = "as boot.foreignrecord.back_secs" + +[[number]] +name = "boot.testcases-window.back_secs" +unit = "s" +ceiling = 420 +ceiling_from = "as boot.foreignrecord.back_secs" + +[[number]] +name = "boot.testcases-debug.back_secs" +unit = "s" +ceiling = 420 +ceiling_from = "as boot.foreignrecord.back_secs" + [[number]] name = "boot.foreignrecord.stick_secs" unit = "s" @@ -382,6 +430,30 @@ measured = 0 # that bound rather than against the runner's, because the runner never reaches # its own. +[[number]] +name = "boot.logstallcase.stick_secs" +unit = "s" +ceiling = 30 +ceiling_from = "as boot.testcases.stick_secs" + +[[number]] +name = "boot.testcases-deaf.stick_secs" +unit = "s" +ceiling = 30 +ceiling_from = "as boot.testcases.stick_secs" + +[[number]] +name = "boot.testcases-window.stick_secs" +unit = "s" +ceiling = 30 +ceiling_from = "as boot.testcases.stick_secs" + +[[number]] +name = "boot.testcases-debug.stick_secs" +unit = "s" +ceiling = 30 +ceiling_from = "as boot.testcases.stick_secs" + [[number]] name = "boot.deadlinewedge.complete_ms" unit = "ms" @@ -624,6 +696,30 @@ unit = "ms" ceiling = 1400 ceiling_from = "as list.testcases.job_ms" +[[number]] +name = "list.logstallcase.job_ms" +unit = "ms" +ceiling = 1400 +ceiling_from = "as list.testcases.job_ms" + +[[number]] +name = "list.testcases-deaf.job_ms" +unit = "ms" +ceiling = 22000 +ceiling_from = "as list.lancase.job_ms — the same one job" + +[[number]] +name = "list.testcases-window.job_ms" +unit = "ms" +ceiling = 27000 +ceiling_from = "as list.latencycase.job_ms: one job each of whose pipe waits the actuator may hold for its window" + +[[number]] +name = "list.testcases-debug.job_ms" +unit = "ms" +ceiling = 2900 +ceiling_from = "as list.shared-debug.job_ms: one of the same eight binaries" + [[number]] name = "list.deadlinewedge.job_ms" unit = "ms" @@ -862,6 +958,30 @@ unit = "us" ceiling = 16667 ceiling_from = "as boot.testcases.panel_max_us" +[[number]] +name = "boot.logstallcase.panel_max_us" +unit = "us" +ceiling = 16667 +ceiling_from = "as boot.testcases.panel_max_us" + +[[number]] +name = "boot.testcases-deaf.panel_max_us" +unit = "us" +ceiling = 16667 +ceiling_from = "as boot.testcases.panel_max_us" + +[[number]] +name = "boot.testcases-window.panel_max_us" +unit = "us" +ceiling = 16667 +ceiling_from = "as boot.testcases.panel_max_us" + +[[number]] +name = "boot.testcases-debug.panel_max_us" +unit = "us" +ceiling = 16667 +ceiling_from = "as boot.testcases.panel_max_us" + [[number]] name = "boot.deadlinewedge.panel_max_us" unit = "us" @@ -985,6 +1105,30 @@ unit = "us" ceiling = 100000 ceiling_from = "as boot.testcases.panel_us" +[[number]] +name = "boot.logstallcase.panel_us" +unit = "us" +ceiling = 100000 +ceiling_from = "as boot.testcases.panel_us" + +[[number]] +name = "boot.testcases-deaf.panel_us" +unit = "us" +ceiling = 100000 +ceiling_from = "as boot.testcases.panel_us" + +[[number]] +name = "boot.testcases-window.panel_us" +unit = "us" +ceiling = 100000 +ceiling_from = "as boot.testcases.panel_us" + +[[number]] +name = "boot.testcases-debug.panel_us" +unit = "us" +ceiling = 100000 +ceiling_from = "as boot.testcases.panel_us" + [[number]] name = "boot.deadlinewedge.panel_us" unit = "us" @@ -1100,6 +1244,30 @@ unit = "operations" ceiling = 0 ceiling_from = "as boot.testcases.park_open_operations" +[[number]] +name = "boot.logstallcase.park_open_operations" +unit = "operations" +ceiling = 0 +ceiling_from = "as boot.testcases.park_open_operations" + +[[number]] +name = "boot.testcases-deaf.park_open_operations" +unit = "operations" +ceiling = 0 +ceiling_from = "as boot.testcases.park_open_operations" + +[[number]] +name = "boot.testcases-window.park_open_operations" +unit = "operations" +ceiling = 0 +ceiling_from = "as boot.testcases.park_open_operations" + +[[number]] +name = "boot.testcases-debug.park_open_operations" +unit = "operations" +ceiling = 0 +ceiling_from = "as boot.testcases.park_open_operations" + [[number]] name = "boot.usbbreak.park_open_operations" unit = "operations" diff --git a/tests/test-durations b/tests/test-durations index c32248eab77..895ee3c3f96 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -136,9 +136,6 @@ abuse_tls_alloc 15 acpi_table_inventory 4601 allocator_stress 787 apps_and_home_are_one_filesystem 7263 -audio_idle_suspend 1013 -audio_tone (smp=1) 8450 -audio_tone_load (smp=1) 18139 bar_placement_is_proven 3297 blackbox_done_chain 4119 blackbox_early_panic_sealed 2470 @@ -164,7 +161,6 @@ debug_float 11 debug_trap 25 demand_paging_sse 12 demand_window_race 54 -desktop_audio_client 13791 desktop_locale_detect 4510 desktop_typing_damage 14338 desktop_window_child 40312 @@ -172,12 +168,9 @@ device_claim_lifetime 21 disk_backtrace 85 diskless_boot 4475 dlopen_dedup 40 -doom_music 7514 -doom_sound_flood 5888 double_fault_stack 5139 double_panic_names_the_fault 9120 driver_wait_refused 17739 -dump_nmi_probe 8098 empty_dir_stat 12 endowment_denied 10 esp_filesystem 10123 @@ -203,8 +196,6 @@ handle_transfer 66 hang_bounded_by_the_stick 3717 hard_lockup_ends_a_deaf_cpu 20277 hash_seed_precedes_every_map 4976 -hda_client_stall 19047 -hda_tone 13389 hda_two_live_refused 4848 hierarchy_paths 87 home_backing_revoked 663 @@ -215,7 +206,6 @@ i8042_absent 9221 i8042_budget_expiry 3135 i8042_fadt_denial 5736 i8042_health 9509 -i8042_health_cadence 10236 i8042_kbd_echo 4770 i8042_mouse 4310 i8042_no_spurious_wake 146 @@ -248,7 +238,6 @@ lan_dhcp_lease 3029 lan_no_lease 32709 lapic_spurious_vector 6829 late_storage_connect 6455 -latency_wake 8790 launcher_refusals 2106 leak_rollback_selftest 5423 loader_watchdog_arms 10064 @@ -277,7 +266,6 @@ metal_sim_compositor 8986 metal_sim_compositor_stall 11390 metal_sim_input 5065 metal_sim_ipc_hostile_peer 226 -metal_sim_null_audio 8311 metal_sim_pointer_churn 16658 metal_sim_scanout_wc 0 metal_sim_window_caps 148 @@ -291,15 +279,12 @@ netd_connection_caps 6538 netd_gone_mid_bind 32 netd_hostile_peer 4509 netd_listener_forgery 2649 -null_sink_client_exits 2167 -null_sink_shipped_client 7095 nvme_home_roundtrip 13 nvme_large_device 6052 nvme_wide_sector 3500 operation_nesting 5075 page_cache_partition_offset 6984 panic_before_peripherals_reboots 3227 -panic_key_holds 42033 panic_reboots 9768 pci_capability_walk 5775 pci_claim_caps_truncated 2583 @@ -368,13 +353,10 @@ std_tls_multi_crate 13 std_unwind 18 std_unwind_so 20 swiss_german_layout 5355 -syscall_cost 92 syscall_window_nmi 6825 syscall_window_nmi_controls 13090 sysret_ss_reload 6453 timer_calibration 6722 -tlb_shootdown_cost 5046 -tlb_shootdown_waits 107 toybox_cp_volume 18616 toybox_file_tools 205 usb_boot_stick_pulled 15650 @@ -393,14 +375,12 @@ va_exhaustion 6159 virtio_net_no_msix 2039 virtio_used_ring 4189 volume_from_another_disk 7122 -wake_storm_cost 337 wall_clock_century_register 9030 wall_clock_no_century 6987 wall_clock_now 13 wall_clock_rtc_dead 8070 wall_clock_rtc_unstable 7805 wall_clock_zone 9347 -watchdog_fed 23405 watchdog_resets 9393 window_refusal 16 writeback_durability 8888 diff --git a/tests/toyos-rust-tests/src/bin/audio_idle_suspend.rs b/tests/toyos-rust-tests/src/bin/audio_idle_suspend.rs index 67ef61de860..5475f6bdafa 100644 --- a/tests/toyos-rust-tests/src/bin/audio_idle_suspend.rs +++ b/tests/toyos-rust-tests/src/bin/audio_idle_suspend.rs @@ -2,10 +2,7 @@ //! connects, soundd's CPU cost is exactly zero. Two sysinfo samples ~1s apart //! must show no cpu_ns movement on any soundd thread — a suspended soundd //! holds no timer and takes no wakes, so any nonzero delta is the mix or -//! control loop running without a reason. This is the one idle-suspend claim gate A -//! structurally cannot see: its counters are streaming-scoped, and its boots -//! always connect a client. No wav analysis — there is no signal, and the -//! capture freezes while the voice is stopped anyway. +//! control loop running without a reason. //! //! Another process's threads is the process roster, which costs //! `Rights::ROSTER` on a `SysCap` — `tests/testcases` names `roster` on the diff --git a/tests/toyos-rust-tests/src/bin/audio_tone.rs b/tests/toyos-rust-tests/src/bin/audio_tone.rs index 9403deedfcd..f28fe525b2d 100644 --- a/tests/toyos-rust-tests/src/bin/audio_tone.rs +++ b/tests/toyos-rust-tests/src/bin/audio_tone.rs @@ -1,6 +1,4 @@ -//! Audio glitch regression test, idle variant: play a deterministic 440Hz -//! sine on an otherwise idle system. The host harness asserts the wav the -//! virtio-sound device captured is glitch-free. +//! Play a deterministic 440Hz sine to the end. #[path = "../tone.rs"] mod tone; diff --git a/tests/toyos-rust-tests/src/bin/audio_tone_load.rs b/tests/toyos-rust-tests/src/bin/audio_tone_load.rs deleted file mode 100644 index d096a77f092..00000000000 --- a/tests/toyos-rust-tests/src/bin/audio_tone_load.rs +++ /dev/null @@ -1,47 +0,0 @@ -//! Audio glitch regression test, load variant: play the tone while two pure -//! busy-spin processes saturate the (single) CPU. Glitch-free playback under -//! load is what the scheduler's audio priority handling must guarantee. - -#[path = "../tone.rs"] -mod tone; - -use std::process::Command; -use std::time::{Duration, Instant}; - -/// Outlives tone startup + 3s playback + drain even under heavy contention. -const BURN_SECS: u64 = 6; - -fn main() { - if std::env::args().nth(1).as_deref() == Some("burn") { - burn(Duration::from_secs(BURN_SECS)); - return; - } - - let burners: Vec<_> = (0..2) - .map(|_| { - Command::new("/system/bin/test_rs_audio_tone_load") - .arg("burn") - .spawn() - .expect("spawn burner") - }) - .collect(); - - tone::play_tone(); - println!("tone done"); - - for mut burner in burners { - burner.wait().expect("wait burner"); - } -} - -fn burn(duration: Duration) { - let start = Instant::now(); - let mut i = 0u64; - loop { - i = i.wrapping_add(1); - // Check the clock rarely so the load stays pure CPU, not syscalls. - if i % (1 << 22) == 0 && start.elapsed() >= duration { - return; - } - } -} diff --git a/tests/toyos-rust-tests/src/bin/blockd_io.rs b/tests/toyos-rust-tests/src/bin/blockd_io.rs index 0fea755af1b..c0d3c657b0a 100644 --- a/tests/toyos-rust-tests/src/bin/blockd_io.rs +++ b/tests/toyos-rust-tests/src/bin/blockd_io.rs @@ -36,8 +36,7 @@ use std::io::{BufRead, BufReader, Write}; use std::os::toyos::process::{ChildExt, CommandExt}; use std::process::{Child, Command, Stdio}; use std::sync::atomic::Ordering; -use std::sync::mpsc::{self, Receiver, RecvTimeoutError}; -use std::time::{Duration, Instant}; +use std::sync::mpsc::{self, Receiver}; use blockd::nvme::{Controller, Owner}; use blockd::region::Region; @@ -80,12 +79,6 @@ const FILE_BYTES: usize = 48 * 1024; /// Mirrored: the file written after the restart, on the same mount. const AFTER: &str = "/AFTER.BIN"; -/// How long a restart waits for the claim of the process it stopped to come -/// back. **A compromise over a kernel defect, the one init makes for the same -/// reason** (`issues/kernel/deferred-release-outlives-its-syscall.md`): a -/// process's end is published before the release its handles queued has run. -const CLAIM_RETURN: Duration = Duration::from_secs(2); - /// Mirrored in `tests/common/blockd.rs`: the most a claim may hold across its /// grants and what it lends (`pcidev::MAX_GRANT_TOTAL`), one region, and the /// addresses a device domain has under `iommu-domain-narrow` @@ -94,14 +87,6 @@ const GRANT_TOTAL: u64 = 32 * 1024 * 1024; const REGION: usize = 2 * 1024 * 1024; const NARROW: u64 = 128 * 1024 * 1024; -/// How long a read this process aims may go unanswered before the role fails: -/// a liveness bound, far past what one block takes. -const AIMED: Duration = Duration::from_secs(10); - -/// How long `hostile-head` waits for blockd to withhold its write's answer, and -/// then for the reset that ends its session: a liveness bound. -const SILENCE_ENDS: Duration = Duration::from_secs(30); - fn guid(text: &str) -> [u8; 16] { PartGuid::parse(text).unwrap_or_else(|| panic!("{text} is no GUID")).0 } @@ -158,16 +143,7 @@ impl Blockd { /// write's answer: the write is done on the device, and its session never /// hears. fn spawn(&mut self, args: &[&str], kill_on_withheld: bool) { - let asked = Instant::now(); - let claim: toyos::Device = loop { - match self.syscap.claim_pci(BLOCKD) { - Err(SyscallError::AlreadyExists) if asked.elapsed() < CLAIM_RETURN => { - std::thread::sleep(Duration::from_millis(1)); - } - Ok(claim) => break claim, - Err(e) => fail(format!("the controller's claim was refused: {e:?}")), - } - }; + let claim: toyos::Device = claim_when_free(&self.syscap); let acceptor = toyos_abi::syscall::dup(self.acceptor.as_handle()) .unwrap_or_else(|e| fail(format!("the acceptor would not duplicate: {e:?}"))); let mut command = Command::new("/system/bin/blockd"); @@ -198,18 +174,15 @@ impl Blockd { self.said = Some(said); } - /// Wait, at most `bound`, for the running blockd to say a line holding + /// Wait, with no deadline, for the running blockd to say a line holding /// `needle`. - fn says(&self, needle: &str, bound: Duration) { + fn says(&self, needle: &str) { let said = self.said.as_ref().expect("spawned"); - let asked = Instant::now(); loop { - let left = bound.saturating_sub(asked.elapsed()); - match said.recv_timeout(left) { + match said.recv() { Ok(line) if line.contains(needle) => return, Ok(_) => {} - Err(RecvTimeoutError::Timeout) => fail(format!("blockd did not say {needle:?} in {bound:?}")), - Err(RecvTimeoutError::Disconnected) => fail(format!("blockd ended before it said {needle:?}")), + Err(_) => fail(format!("blockd ended before it said {needle:?}")), } } } @@ -248,15 +221,14 @@ fn chunks(blocks: u64, salt: u8) -> Vec> { } /// Write `chunks` from block 0, `in_flight` requests at a time; flush. Answers -/// how long the flushes took and how many there were: an acknowledged write -/// holds its arena blocks until a flush covers it, so a full arena is where -/// one is asked. -fn write_all(s: &mut Session, chunks: &[Vec], in_flight: usize) -> (Duration, u32) { +/// how many flushes there were: an acknowledged write holds its arena blocks +/// until a flush covers it, so a full arena is where one is asked. +fn write_all(s: &mut Session, chunks: &[Vec], in_flight: usize) -> u32 { let mut next = 0usize; let mut lba = 0u64; let mut outstanding = 0usize; let mut full = false; - let (mut flushing, mut flushes) = (Duration::ZERO, 0u32); + let mut flushes = 0u32; while next < chunks.len() || outstanding > 0 { while next < chunks.len() && outstanding < in_flight && !full { match s.submit_write(lba, &chunks[next]) { @@ -282,16 +254,13 @@ fn write_all(s: &mut Session, chunks: &[Vec], in_flight: usize) -> (Duration } } if full && outstanding == 0 { - let started = Instant::now(); flushed(s); - flushing += started.elapsed(); flushes += 1; full = false; } } - let started = Instant::now(); flushed(s); - (flushing + started.elapsed(), flushes + 1) + flushes + 1 } fn flushed(s: &mut Session) { @@ -429,13 +398,8 @@ fn holder_role(expect: &str) { println!("blockd_io: a second client of the slot refused with {got}, as expected"); } -fn mb_per_s(blocks: u64, took: Duration) -> f64 { - (blocks * BLOCK_BYTES as u64) as f64 / (1024.0 * 1024.0) / took.as_secs_f64() -} - -/// The same bytes through the kernel's driver and through blockd, the data -/// built before and checked after what is timed, so each number is the -/// driver's path and nothing of this binary's. +/// The same bytes through the kernel's driver and through blockd, one request +/// at a time and then as many as the arena holds. fn bench() { // The kernel's driver, through a partition claim on the first controller: // one request at a time, as it moves them. @@ -448,21 +412,17 @@ fn bench() { .iter() .map(|c| c.chunks(BLOCK_BYTES).map(|b| b.try_into().expect("a block")).collect()) .collect(); - let started = Instant::now(); let mut lba = 0u64; for chunk in &blocks { part.write(lba, chunk).unwrap_or_else(|e| fail(format!("a kernel write: {e:?}"))); lba += chunk.len() as u64; } part.sync().unwrap_or_else(|e| fail(format!("the kernel's fsync: {e:?}"))); - let kernel_write = started.elapsed(); let mut read = vec![[0u8; BLOCK_BYTES]; BENCH_BLOCKS as usize]; let per = toyos_abi::part::MAX_BLOCKS_PER_CALL; - let started = Instant::now(); for (i, chunk) in read.chunks_mut(per).enumerate() { part.read((i * per) as u64, chunk).unwrap_or_else(|e| fail(format!("a kernel read: {e:?}"))); } - let kernel_read = started.elapsed(); holds(&read.concat(), &written, "the kernel's bench partition"); drop(part); @@ -472,26 +432,15 @@ fn bench() { let mut runs = Vec::new(); for (salt, in_flight) in [(0x3D, 1usize), (0x3C, 15)] { let written = chunks(BENCH_BLOCKS, salt); - let started = Instant::now(); - let (flushing, flushes) = write_all(&mut s, &written, in_flight); - let write = started.elapsed(); - let started = Instant::now(); + let flushes = write_all(&mut s, &written, in_flight); let read = read_all(&mut s, BENCH_BLOCKS, in_flight); - let took = started.elapsed(); holds(&read, &written, "blockd's bench partition"); - runs.push(format!( - "{in_flight} in flight: write {:.1} MiB/s ({flushes} Flushes, {} ms of it) read {:.1} MiB/s", - mb_per_s(BENCH_BLOCKS, write), - flushing.as_millis(), - mb_per_s(BENCH_BLOCKS, took), - )); + runs.push(format!("{in_flight} in flight with {flushes} Flushes")); } println!( - "blockd_io: bench {} MiB each way: kernel driver write {:.1} MiB/s (its one fsync issues no \ - Flush) read {:.1} MiB/s; blockd {}; at most {} requests on the wire", + "blockd_io: bench {} MiB each way through the kernel driver and through blockd {}; at \ + most {} requests on the wire", BENCH_BLOCKS * BLOCK_BYTES as u64 / (1024 * 1024), - mb_per_s(BENCH_BLOCKS, kernel_write), - mb_per_s(BENCH_BLOCKS, kernel_read), runs.join("; "), s.peak_on_the_wire() ); @@ -506,15 +455,11 @@ fn reset() { Ok(Outcome::Done) => {} other => fail(format!("the first write was answered {other:?}")), } - let started = Instant::now(); match s.write(1, &pattern(0x71, 1)) { Ok(Outcome::Device) => {} other => fail(format!("the withheld write was answered {other:?}, not Device")), } - println!( - "blockd_io: the withheld write was answered Device after {} ms", - started.elapsed().as_millis() - ); + println!("blockd_io: the withheld write was answered Device"); // The reset may have dropped the device's cache: the write acknowledged // before it goes out again, inside this flush. flushed(&mut s); @@ -555,12 +500,12 @@ fn hostile_head() { } words[SQ_TAIL].store(1, Ordering::Release); conn.write_nonblock(&[1]).unwrap_or_else(|e| fail(format!("the doorbell: {e:?}"))); - blockd.says("WITHHELD", SILENCE_ENDS); + blockd.says("WITHHELD"); let tail = words[CQ_TAIL].load(Ordering::Acquire); words[CQ_HEAD].store(tail.wrapping_sub(DEPTH), Ordering::Release); conn.write_nonblock(&[1]).unwrap_or_else(|e| fail(format!("the doorbell: {e:?}"))); println!("blockd_io: with a write on the device, the client moved its completion head {DEPTH} behind the tail"); - blockd.says("closed after", SILENCE_ENDS); + blockd.says("closed after"); println!("blockd_io: blockd ended the session and runs on"); let mut next = open(blockd.names(), TARGET); let block = pattern(0x6B, 0); @@ -815,9 +760,7 @@ fn dma(role: &str) { fn refused(ctrl: &mut Controller, region: &SharedMemory, at: u64, what: &str) { let answered = transfer(ctrl, 0, at); println!("blockd_io: the device answered a read aimed {what} with {answered:?}"); - if !refused_within(ctrl, Duration::from_secs(5)) { - fail(format!("a read aimed {what} left the claim answering; the unit did not refuse it")); - } + await_refusal(ctrl); if !region.as_slice().iter().all(|b| *b == 0xA5) { fail(format!("a read aimed {what} changed the lent region")); } @@ -827,46 +770,39 @@ fn refused(ctrl: &mut Controller, region: &SharedMemory, at: u64, what: &str) { ); } -/// One read of device block `block` into device address `at`, waited for on -/// the claim's interrupt, with nothing else on the device: whether the device -/// did it, or the claim's refusal once the unit refused the function an -/// access — which is how a read aimed outside the function's domain ends. +/// One read of device block `block` into device address `at`, waited for with +/// no deadline on the claim's interrupt, with nothing else on the device: +/// whether the device did it, or the claim's refusal once the unit refused the +/// function an access — which is how a read aimed outside the function's +/// domain ends. fn transfer(ctrl: &mut Controller, block: u64, at: u64) -> Result { if ctrl.busy() != 0 { fail("a waited read beside other commands".into()); } ctrl.submit_io(false, block, 1, at, Owner::Driver); let poller = Poller::new(1); - let start = Instant::now(); let mut done = Vec::new(); loop { ctrl.reap(&mut done); if let Some(d) = done.pop() { return Ok(d.ok); } - let Some(left) = AIMED.checked_sub(start.elapsed()) else { - fail(format!("a read aimed at {at:#x} was not answered in {AIMED:?}")); - }; poller.watch(ctrl.claim(), READABLE, 0); - poller.wait(1, left.as_nanos() as u64, |_| {}); + poller.wait(1, u64::MAX, |_| {}); ctrl.take_interrupt()?; } } -/// Wait, at most `bound`, for the claim to answer with the unit's refusal — -/// what every call on a claim answers once its function was refused an access; -/// whether it did. -fn refused_within(ctrl: &Controller, bound: Duration) -> bool { +/// Wait, with no deadline, for the claim to answer with the unit's refusal — +/// what every call on a claim answers once its function was refused an access. +/// A unit that never refuses leaves this waiting, and the harness ceiling reds +/// it. +fn await_refusal(ctrl: &Controller) { let poller = Poller::new(1); - let start = Instant::now(); - while start.elapsed() < bound { - if ctrl.take_interrupt() == Err(SyscallError::Io) { - return true; - } + while ctrl.take_interrupt() != Err(SyscallError::Io) { poller.watch(ctrl.claim(), READABLE, 0); - poller.wait(1, bound.saturating_sub(start.elapsed()).as_nanos() as u64, |_| {}); + poller.wait(1, u64::MAX, |_| {}); } - ctrl.take_interrupt() == Err(SyscallError::Io) } /// A kernel driver's own pool is not the holder's to lend, though it is @@ -993,14 +929,15 @@ fn dma_churn() { println!("blockd_io: PASS dma-churn"); } -/// The controller's claim once the last holder's release has run. -fn claim_when_free(syscap: &SysCap) -> toyos::PciDev { - let asked = Instant::now(); +/// The controller's claim once the last holder's release has run, waited for with +/// no deadline: a process's end is published before the release its handles +/// queued has run (`issues/kernel/deferred-release-outlives-its-syscall.md`). +fn claim_when_free(syscap: &SysCap) -> T { loop { match syscap.claim_pci(BLOCKD) { - Err(SyscallError::AlreadyExists) if asked.elapsed() < CLAIM_RETURN => { - std::thread::sleep(Duration::from_millis(1)); - } + // A pace, so the CPU this runs on can reach the idle loop that + // drains the release. + Err(SyscallError::AlreadyExists) => std::thread::sleep(std::time::Duration::from_millis(1)), Ok(claim) => return claim, Err(e) => fail(format!("the controller's claim was refused: {e:?}")), } @@ -1013,7 +950,7 @@ fn claim_when_free(syscap: &SysCap) -> toyos::PciDev { fn dma_residue() { let syscap = capability(); let first = { - let dev = claim_when_free(&syscap); + let dev: toyos::PciDev = claim_when_free(&syscap); let mut ctrl = Controller::open(dev, None).unwrap_or_else(|e| fail(format!("the controller: {e}"))); let mut region = SharedMemory::create(REGION).unwrap_or_else(|e| fail(format!("a region: {e:?}"))); region.as_mut_slice().fill(0xA5); @@ -1026,7 +963,7 @@ fn dma_residue() { }; println!("blockd_io: claim 1 lent a region at {first:#x}, the device read into it, and the claim ended holding it"); for n in [2, 3] { - let dev = claim_when_free(&syscap); + let dev: toyos::PciDev = claim_when_free(&syscap); let region = SharedMemory::create(REGION).unwrap_or_else(|e| fail(format!("a region: {e:?}"))); let mapping = dev.dma_map(region.as_handle()).unwrap_or_else(|e| fail(format!("claim {n}'s dma_map: {e:?}"))); if mapping.device_addr == first { diff --git a/tests/toyos-rust-tests/src/bin/blocking_read_stress.rs b/tests/toyos-rust-tests/src/bin/blocking_read_stress.rs index 4297aab87cc..474a5c5bbd3 100644 --- a/tests/toyos-rust-tests/src/bin/blocking_read_stress.rs +++ b/tests/toyos-rust-tests/src/bin/blocking_read_stress.rs @@ -2,32 +2,19 @@ //! //! Every round trip here is two parks and two posts on the completion core — //! the reader blocks in `sys_read` on an empty pipe, the writer's post wakes -//! it, and the same happens back the other way. **The verdict is a count -//! inside a wall-clock bound**, never a hang: a stall is disqualified as a -//! verdict, because the harness prints "the guard expired, so this says -//! nothing about the tree" beside one and tells nobody to bisect it. A -//! dropped completion reds as a number that is short of `ROUNDS`, with the -//! round it stopped at named. +//! it, and the same happens back the other way. **The verdict is the count of +//! round trips**: a dropped completion parks this process, and the harness's +//! ceiling turns that into a red. //! //! The echo half is this same binary with an argument, so the two ends are one //! file and the child's own parks are the same parks the parent's are. use std::io::{Read, Write}; use std::process::{Command, Stdio}; -use std::sync::atomic::{AtomicU32, Ordering}; -use std::thread; -use std::time::{Duration, Instant}; -/// Round trips. Enough that a wake lost at some rate shows up, small enough -/// that the whole thing fits inside the harness's five seconds with room — -/// measured at 500 rounds in 68 ms on the dev host, so this is two orders of -/// magnitude of headroom. +/// Round trips. Enough that a wake lost at some rate shows up. const ROUNDS: u32 = 500; -/// What the round trips must fit in. Not a latency assertion: it is what turns -/// a lost wake into a *number* rather than into a stall the suite names apart. -const BOUND: Duration = Duration::from_secs(3); - fn echo() -> ! { let mut stdin = std::io::stdin(); let mut stdout = std::io::stdout(); @@ -59,30 +46,6 @@ fn main() { let mut to_child = child.stdin.take().expect("piped stdin"); let mut from_child = child.stdout.take().expect("piped stdout"); - // **The watchdog is what makes a lost wake a number.** Without it a - // dropped completion parks this process for ever, the harness's guard - // expires, and the suite prints `STALL` — which is disqualified as a - // verdict, because it is the one class the harness names apart and tells - // nobody to bisect. With it, the round the machine stopped at is the - // failure message. - static DONE: AtomicU32 = AtomicU32::new(0); - thread::spawn(|| { - thread::sleep(BOUND); - // **Ends the process, rather than panicking this thread.** A thread - // panic leaves `main` parked in the read that never came back and the - // harness times the guest out — which is the stall this watchdog - // exists to replace. Verified both ways: with the pipe's readable post - // dropped, `panic!` here produced `timed out after 8s` and this - // produces the count. - eprintln!( - "blocking_read_stress: only {} of {ROUNDS} round trips completed inside {BOUND:?} \ - — a wake was not delivered", - DONE.load(Ordering::Relaxed), - ); - std::process::exit(1); - }); - - let started = Instant::now(); let mut completed = 0u32; for round in 0..ROUNDS { let sent = [(round % 251) as u8; 1]; @@ -94,9 +57,7 @@ fn main() { .unwrap_or_else(|e| panic!("round {round} of {ROUNDS} never came back: {e}")); assert_eq!(got, sent, "round {round} came back as another byte"); completed += 1; - DONE.store(completed, Ordering::Relaxed); } - let elapsed = started.elapsed(); drop(to_child); let status = child.wait().expect("wait for the echo half"); @@ -106,10 +67,5 @@ fn main() { completed, ROUNDS, "only {completed} of {ROUNDS} round trips completed", ); - assert!( - elapsed < BOUND, - "{ROUNDS} round trips took {elapsed:?}, past the {BOUND:?} bound — a wake is being \ - waited out rather than delivered", - ); - println!("blocking_read_stress: {completed} round trips in {elapsed:?}"); + println!("blocking_read_stress: {completed} round trips"); } diff --git a/tests/toyos-rust-tests/src/bin/compositor_stall.rs b/tests/toyos-rust-tests/src/bin/compositor_stall.rs index 00dcdb306c2..3c4232e2aa3 100644 --- a/tests/toyos-rust-tests/src/bin/compositor_stall.rs +++ b/tests/toyos-rust-tests/src/bin/compositor_stall.rs @@ -9,15 +9,14 @@ //! accepted connection. //! //! Each case sets its stall up and leaves it standing, then asks the -//! compositor a question **with a deadline**. That is the shape the assertion -//! has to have: a frozen compositor turns any unbounded call into a hung boot, -//! and a hung boot names no defect. The host side asserts the other half — -//! that the desktop is still *painting*, and that every client dropped along -//! the way was named in the log. +//! compositor a question. No wait here has a deadline: a compositor frozen on a +//! client never answers, and the harness ceiling reds it. The host side asserts +//! the other half — that the desktop is still *painting*, and that every client +//! dropped along the way was named in the log. use std::process::exit; use std::thread; -use std::time::{Duration, Instant}; +use std::sync::atomic::{AtomicBool, Ordering}; use toyos::endow; use toyos::AsHandle; @@ -25,26 +24,15 @@ use toyos::{ipc, Connection}; use toyos_abi::syscall::{self, SyscallError}; use window::Window; -/// `MSG_GET_RESOLUTION` is answered from the compositor's dispatch, so a reply -/// proves the event loop reached the end of a pass rather than merely that the -/// process exists. -const PROBE_POLLS: u32 = 500; -const PROBE_POLL_NS: u64 = 10_000_000; - -/// Past the compositor's own `HANDSHAKE_TIMEOUT`, so the three connections -/// that never finish a first frame have been ruled on by the time the run -/// ends and the host can require the lines that say so. -const HANDSHAKE_WAIT: Duration = Duration::from_secs(3); +/// Between two looks at a connection the compositor is expected to drop. A +/// pace and never a verdict. +const POLL_NS: u64 = 10_000_000; /// A message type no protocol here defines: the compositor's dispatch ignores /// it, so a stream of them is pure event-loop load with nothing to draw. That /// is what makes it a starvation case rather than a redraw case. const UNKNOWN_MSG: u32 = 0x7FFF_0001; -/// Long enough to contain the compositor's 2 s reporting interval whichever -/// side of one it starts on. -const STREAM: Duration = Duration::from_secs(5); - /// One `MSG_GET_RESOLUTION` costs the client 8 bytes and the compositor 16, so /// filling a client's 2,097,088-byte receive ring from the far side takes /// 131,068 answers. This is that with margin, and the requests themselves are @@ -52,13 +40,6 @@ const STREAM: Duration = Duration::from_secs(5); /// the *client* instead, which would prove the wrong thing. const REQUESTS: usize = 140_000; -/// How long the compositor gets to reach the end of that ring and say so. -/// -/// Measured at roughly a second on the metal-sim boot; this is an order of -/// magnitude of slack, and it is a bound on the machine rather than on the -/// answer — the answer is the connection closing. -const REFUSAL_POLLS: u32 = 1_500; - fn main() { // Held to the end of the run: a dropped `Connection` closes the handle, and // a closed handle is a peer that hung up rather than one that went quiet. @@ -79,8 +60,11 @@ fn main() { probe("header without payload"); // The three above are handshakes that never complete. Nothing the client - // does ends them; the compositor's own deadline does. - thread::sleep(HANDSHAKE_WAIT); + // does ends them; the compositor's own deadline does, and each one's close + // is what is waited for. + for conn in &held { + await_hang_up(conn.as_handle()); + } probe("after the handshake deadline"); // A window that stops in the middle of a message it already declared. The @@ -100,37 +84,59 @@ fn main() { requests.extend_from_slice(&header(window::MSG_GET_RESOLUTION, 0)); } write_handle(deaf.handle(), &requests, "window that will not read"); - await_refusal(&deaf); + await_hang_up(deaf.handle()); probe("window that will not read"); // A window with something to send on every pass. Nothing here is // unanswerable — the loop simply never runs out of work, and a drain that - // ends only when nothing is ready never reaches the screen. The assertion - // is the host's: frames, between these two markers. + // ends only when nothing is ready never reaches the screen. So a second + // window presents while it streams, and the stream runs until that present + // is composited: a drain loop the stream starves never gets to `redraw`, and + // the frame never comes. let noisy = Window::create(64, 64).expect("a window to stream from"); let handle = noisy.handle(); - println!("compositor stall: stream start"); - let streamer = thread::spawn(move || { - let frame = header(UNKNOWN_MSG, 0); - let until = Instant::now() + STREAM; + let mut watcher = Window::create(64, 64).expect("a window to composite under the stream"); + let (streaming, framed) = (AtomicBool::new(false), AtomicBool::new(false)); + let presenter = thread::current(); + thread::scope(|s| { + s.spawn(|| { + let frame = header(UNKNOWN_MSG, 0); + while !framed.load(Ordering::Acquire) { + // Fill the ring, not merely feed it. The compositor takes one + // frame per client per pass, so a client that keeps up with only + // that lets the drain run dry and the screen get painted — which + // is the thing this case is supposed to prevent. + // + // Never a torn frame: both ends move this ring in multiples of + // eight bytes and its capacity is one too, so a write of a header + // either fits whole or finds no room at all. + while matches!(syscall::write_nonblock(handle, &frame), Ok(8)) {} + streaming.store(true, Ordering::Release); + presenter.unpark(); + syscall::nanosleep(1_000_000); + } + }); + // Presented once the ring is full, so the frame is composited under + // the stream and not ahead of it. Parked until then, never spinning: a + // thread yielding beside the writer held it below the compositor's + // drain rate, and the ring never filled. + while !streaming.load(Ordering::Acquire) { + thread::park(); + } + println!("compositor stall: the ring is full; a second window presents under it"); + watcher.present(); loop { - // Fill the ring, not merely feed it. The compositor takes one - // frame per client per pass, so a client that keeps up with only - // that lets the drain run dry and the screen get painted — which - // is the thing this case is supposed to prevent. - // - // Never a torn frame: both ends move this ring in multiples of - // eight bytes and its capacity is one too, so a write of a header - // either fits whole or finds no room at all. - while matches!(syscall::write_nonblock(handle, &frame), Ok(8)) {} - if Instant::now() >= until { - break; + match watcher.recv_event() { + window::Event::Frame => break, + window::Event::Close => fail( + "[window that never stops sending] the window presented under the stream \ + was closed", + ), + _ => {} } - syscall::nanosleep(1_000_000); } + framed.store(true, Ordering::Release); }); - streamer.join().expect("the streaming thread"); - println!("compositor stall: stream end"); probe("window that never stops sending"); println!("compositor stall: 6 stalls survived, compositor still serving"); @@ -164,7 +170,7 @@ fn write_handle(handle: toyos_abi::RawHandle, bytes: &[u8], what: &str) { } } -/// Wait for the compositor to hang up on the window that stopped reading. +/// Wait, with no deadline, until nothing holds the other end of `handle`. /// /// **Without draining a byte**, which is the whole difficulty: this client's /// receive ring has to stay full for the compositor to reach the end of it, @@ -173,21 +179,15 @@ fn write_handle(handle: toyos_abi::RawHandle, bytes: &[u8], what: &str) { /// still holding the read end — so the refusal is observed rather than slept /// through. A compositor parked in `write` instead has its handle open and /// answers `Ok` here forever. -fn await_refusal(deaf: &Window) { - for _ in 0..REFUSAL_POLLS { - if let Err(SyscallError::Gone) = syscall::write_nonblock(deaf.handle(), &[]) { - return; - } - syscall::nanosleep(PROBE_POLL_NS); +fn await_hang_up(handle: toyos_abi::RawHandle) { + while syscall::write_nonblock(handle, &[]) != Err(SyscallError::Gone) { + syscall::nanosleep(POLL_NS); } - fail(&format!( - "[window that will not read] {} bytes of unread answers and the compositor never \ - dropped the connection — it is waiting for this client to read its mail", - REQUESTS * 16, - )); } -/// Ask the compositor something it always answers, and give it a deadline. +/// Ask the compositor something it always answers from its dispatch, so a reply +/// proves the event loop reached the end of a pass. No deadline: a compositor +/// parked on a client never answers, and the harness ceiling reds it. fn probe(what: &str) { let conn = connect(what); if let Err(e) = ipc::signal(conn.as_handle(), window::MSG_GET_RESOLUTION) { @@ -195,23 +195,13 @@ fn probe(what: &str) { } let mut buf = [0u8; 16]; let mut got = 0; - for _ in 0..PROBE_POLLS { - match conn.read_nonblock(&mut buf[got..]) { + while got < buf.len() { + match syscall::read(conn.as_handle(), &mut buf[got..]) { Ok(0) => fail(&format!("[{what}] the compositor closed the probe unanswered")), - Ok(n) => { - got += n; - if got == buf.len() { - return; - } - } - Err(SyscallError::WouldBlock) => syscall::nanosleep(PROBE_POLL_NS), + Ok(n) => got += n, Err(e) => fail(&format!("[{what}] the probe could not be read: {e:?}")), } } - fail(&format!( - "[{what}] the compositor did not answer in {} ms — its event loop is parked on a client", - PROBE_POLLS as u64 * PROBE_POLL_NS / 1_000_000, - )); } fn fail(msg: &str) -> ! { diff --git a/tests/toyos-rust-tests/src/bin/copy_out_races_munmap.rs b/tests/toyos-rust-tests/src/bin/copy_out_races_munmap.rs index 28aa079b76e..edbb013fb81 100644 --- a/tests/toyos-rust-tests/src/bin/copy_out_races_munmap.rs +++ b/tests/toyos-rust-tests/src/bin/copy_out_races_munmap.rs @@ -15,7 +15,6 @@ use std::sync::atomic::{AtomicU64, Ordering}; use std::sync::Arc; use std::thread; -use std::time::{Duration, Instant}; use toyos_abi::syscall::{self, MmapFlags, MmapProt, OpenFlags, SYS_FSTAT}; @@ -27,9 +26,6 @@ const SELF_PATH: &str = "/system/bin/test_rs_copy_out_races_munmap"; const PAGE_2M: usize = 2 * 1024 * 1024; /// Past `Stat`, which is what `fstat` stores. const CHECKED: usize = 64; -/// A liveness bound on the kernel's cue, never a pace: the held thread is -/// already inside its syscall when the main thread starts waiting. -const CUE_BOUND: Duration = Duration::from_secs(10); /// `fstat` into an address, which the typed wrapper cannot take. fn fstat_into(handle: u64, addr: u64) -> u64 { @@ -79,12 +75,9 @@ fn main() { thread::spawn(move || ret.store(fstat_into(fd, addr), Ordering::SeqCst)) }; - let began = Instant::now(); + // No deadline: a kernel that never holds the marked copy leaves this spinning, + // and the harness ceiling reds it. while unsafe { words.add(1).read_volatile() } != HELD { - assert!( - began.elapsed() < CUE_BOUND, - "the kernel never held the marked copy in {CUE_BOUND:?}: is copy-meets-a-remap armed?" - ); std::hint::spin_loop(); } unsafe { syscall::munmap(victim, PAGE_2M) }.expect("munmap the copy's destination"); diff --git a/tests/toyos-rust-tests/src/bin/demand_window_race.rs b/tests/toyos-rust-tests/src/bin/demand_window_race.rs index d1255857c47..0049526ff44 100644 --- a/tests/toyos-rust-tests/src/bin/demand_window_race.rs +++ b/tests/toyos-rust-tests/src/bin/demand_window_race.rs @@ -294,11 +294,9 @@ fn run(exe: &std::path::Path, arg: &str) -> ProcessStats { let status = child.wait().unwrap_or_else(|e| panic!("wait {arg}: {e}")); let stats = stats_of(&child); println!( - " {arg}: {} fills, {} kept, {} ns in faults ({} ns/fill)", + " {arg}: {} fills, {} kept", fills(&stats), stats.alloc_count, - stats.fault_ns, - stats.fault_ns / fills(&stats).max(1), ); assert!(status.success(), "the {arg} child exited with {status}"); stats diff --git a/tests/toyos-rust-tests/src/bin/doom_music.rs b/tests/toyos-rust-tests/src/bin/doom_music.rs deleted file mode 100644 index a9603dfbdde..00000000000 --- a/tests/toyos-rust-tests/src/bin/doom_music.rs +++ /dev/null @@ -1,19 +0,0 @@ -//! Runs doom's music actuator and reports whether the process lived. -//! -//! The playing is `sound::music_check` in `userland/doom`: it opens the shipped -//! SoundFont, converts one of the WAD's own MUS lumps and pushes the result at -//! the audio device. Only that binary can reach any of it. This side starts it -//! and answers the question a serial log cannot — whether it exited or died — -//! while the verdict on the sound is the host's capture of the device. - -use std::process::Command; - -fn main() { - let mut child = Command::new("/system/bin/doom") - .arg("--music-check") - .spawn() - .expect("spawn /system/bin/doom --music-check"); - let status = child.wait().expect("wait for /system/bin/doom"); - assert!(status.success(), "doom could not play its own music: {status:?}"); - println!("doom played its music"); -} diff --git a/tests/toyos-rust-tests/src/bin/doom_sound_flood.rs b/tests/toyos-rust-tests/src/bin/doom_sound_flood.rs deleted file mode 100644 index f844524bafa..00000000000 --- a/tests/toyos-rust-tests/src/bin/doom_sound_flood.rs +++ /dev/null @@ -1,21 +0,0 @@ -//! Runs doom's stalled-consumer actuator and reports whether the game lived. -//! -//! The flood itself is `sound::sound_stress` in `userland/doom`: the producer -//! is the game thread inside its own sound module, so nothing out here can -//! drive it. This side only starts it and answers the one question the host -//! cannot ask from a serial log alone — whether the process exited or died. - -use std::process::Command; - -fn main() { - let mut child = Command::new("/system/bin/doom") - .arg("--sound-stress") - .spawn() - .expect("spawn /system/bin/doom --sound-stress"); - let status = child.wait().expect("wait for /system/bin/doom"); - assert!( - status.success(), - "doom did not survive its own sound producer: {status:?}" - ); - println!("doom survived the sound flood"); -} diff --git a/tests/toyos-rust-tests/src/bin/exit_wait_storm.rs b/tests/toyos-rust-tests/src/bin/exit_wait_storm.rs index d25dbe313d8..6ea2d307430 100644 --- a/tests/toyos-rust-tests/src/bin/exit_wait_storm.rs +++ b/tests/toyos-rust-tests/src/bin/exit_wait_storm.rs @@ -8,10 +8,8 @@ //! ordering rather than volume — `process_lifecycle` has one arm on the wake //! and `std_threading` joins four threads. //! -//! **The verdict is a count of collected exit codes inside a bound**, the same -//! shape `blocking_read_stress` takes and for the same reason: a lost publish -//! must red as a number rather than as a stall the suite names apart and -//! nobody bisects. +//! **The verdict is a count of collected exit codes**; a lost publish parks +//! this process, and the harness's ceiling turns that into a red. //! //! **A child parks until the parent releases it, and that is what makes the //! parent's wait a park.** A child on its own schedule has published its exit @@ -22,9 +20,8 @@ use std::io::Read; use std::process::{Command, Stdio}; -use std::sync::atomic::{AtomicU32, Ordering}; use std::thread; -use std::time::{Duration, Instant}; +use std::time::Duration; /// Children spawned, held on their own stdin, and released together. const CHILDREN: u32 = 24; @@ -34,20 +31,6 @@ const CHILDREN: u32 = 24; /// one of those parks ended. const THREADS: u32 = 24; -/// What the storm must fit in. The spawns that set it up are outside it: they -/// are ELF loads, and a bound over them measures the loader. -const BOUND: Duration = Duration::from_secs(3); - -/// A liveness allowance for the spawns and never a measurement of them: it -/// turns a wedged setup into a number instead of the harness's stall. -const SETUP: Duration = Duration::from_secs(30); - -/// Which phase the watchdog names, and how far each got. -static SPAWNING: AtomicU32 = AtomicU32::new(1); -static SPAWNED: AtomicU32 = AtomicU32::new(0); -static COLLECTED: AtomicU32 = AtomicU32::new(0); -static JOINED: AtomicU32 = AtomicU32::new(0); - fn main() { if let Some(code) = std::env::args().nth(1) { // The child half: park in `read` until the parent drops the write end, @@ -59,28 +42,6 @@ fn main() { let exe = std::env::current_exe().expect("current_exe failed"); - thread::spawn(|| { - thread::sleep(SETUP + BOUND); - // Ends the process rather than this thread, for - // `blocking_read_stress`'s reason: a thread panic leaves `main` parked - // on the publish that never came, and the harness reports a stall. - if SPAWNING.load(Ordering::Relaxed) == 1 { - eprintln!( - "exit_wait_storm: {} of {CHILDREN} children spawned in {SETUP:?} — the storm \ - never started", - SPAWNED.load(Ordering::Relaxed), - ); - } else { - eprintln!( - "exit_wait_storm: {} of {CHILDREN} exits collected and {} of {THREADS} threads \ - joined inside {BOUND:?} — a publish was not delivered", - COLLECTED.load(Ordering::Relaxed), - JOINED.load(Ordering::Relaxed), - ); - } - std::process::exit(1); - }); - let mut children = Vec::new(); let mut held = Vec::new(); for i in 0..CHILDREN { @@ -91,7 +52,6 @@ fn main() { .unwrap_or_else(|e| panic!("spawn child {i}: {e}")); held.push(child.stdin.take().expect("the spawn was asked for a pipe")); children.push((i, child)); - SPAWNED.store(i + 1, Ordering::Relaxed); } // The premise, asserted rather than left to timing: nothing has released a @@ -103,8 +63,6 @@ fn main() { ); } - SPAWNING.store(0, Ordering::Relaxed); - let started = Instant::now(); drop(held); let mut collected = 0u32; @@ -116,7 +74,6 @@ fn main() { "child {i} answered with another process's code", ); collected += 1; - COLLECTED.store(collected, Ordering::Relaxed); } // The thread half: each thread returns its own number, and the join is the @@ -134,16 +91,9 @@ fn main() { let got = handle.join().unwrap_or_else(|_| panic!("join thread {i}")); assert_eq!(got, i as u32, "thread {i} answered for another one"); joined += 1; - JOINED.store(joined, Ordering::Relaxed); } - let elapsed = started.elapsed(); assert_eq!(collected, CHILDREN, "only {collected} of {CHILDREN} exits were collected"); assert_eq!(joined, THREADS, "only {joined} of {THREADS} threads were joined"); - assert!( - elapsed < BOUND, - "the storm took {elapsed:?}, past the {BOUND:?} bound — a publish is being waited out \ - rather than delivered", - ); - println!("exit_wait_storm: {collected} exits collected and {joined} threads joined in {elapsed:?}"); + println!("exit_wait_storm: {collected} exits collected and {joined} threads joined"); } diff --git a/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs b/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs index ee0e124b5b8..908ad50f12d 100644 --- a/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs +++ b/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs @@ -23,7 +23,7 @@ use std::fs; use std::io::{Read, Write}; use std::thread; -use std::time::{Duration, Instant}; +use std::time::Duration; /// Mirrored in `tests/common/volumes.rs::fat_backing_revoked`. Two halves of one /// fixture; a change to either alone fails loudly rather than passing quietly. @@ -40,24 +40,19 @@ const LEN: usize = 8 * 4096; const VICTIM_BYTE: u8 = 0xA7; const ATTACKER_BYTE: u8 = 0x5C; -/// Five of the 2 s operation budgets behind the kernel's `WouldBlock` refusal -/// (`kernel/src/block.rs::OPERATION`): device patience is not what this test -/// is about, so its setup asks again the way logd's flush policy does. -const SETUP_PATIENCE: Duration = Duration::from_secs(10); +/// Between two asks of a setup step the kernel refused with `WouldBlock` +/// (`kernel/src/block.rs::OPERATION`). A pace and never a verdict: device +/// patience is not what this test is about, so its setup asks again the way +/// logd's flush policy does. const SETUP_PAUSE: Duration = Duration::from_millis(200); -/// One idempotent setup step, asked again on `WouldBlock` until -/// [`SETUP_PATIENCE`] is spent; anything else panics with the step's message. +/// One idempotent setup step, asked again on `WouldBlock` with no deadline; +/// anything else panics with the step's message. fn patient(what: &str, mut op: impl FnMut() -> std::io::Result) -> T { - let start = Instant::now(); loop { match op() { Ok(v) => return v, - Err(e) if e.kind() == std::io::ErrorKind::WouldBlock - && start.elapsed() < SETUP_PATIENCE => - { - thread::sleep(SETUP_PAUSE); - } + Err(e) if e.kind() == std::io::ErrorKind::WouldBlock => thread::sleep(SETUP_PAUSE), Err(e) => panic!("{what}: {e}"), } } diff --git a/tests/toyos-rust-tests/src/bin/ftruncate_flush_race.rs b/tests/toyos-rust-tests/src/bin/ftruncate_flush_race.rs index 707daa26078..55415aebc41 100644 --- a/tests/toyos-rust-tests/src/bin/ftruncate_flush_race.rs +++ b/tests/toyos-rust-tests/src/bin/ftruncate_flush_race.rs @@ -3,14 +3,12 @@ //! `SYS_FTRUNCATE`'s resize once took no VFS lock, so a truncate could land //! between a flush's two steps and record the older size. `ftruncate-flush-stall` //! holds every flush of this file open for 400ms; this binary races a truncate -//! against it. Looped, since the stall spins preemption-off: one sleep can -//! overshoot into a lock-free gap, and a lockless resize can never make a -//! contended attempt — the loop is one-sided. +//! against it, [`ATTEMPTS`] times, and the host reads the size the volume kept. use std::fs::OpenOptions; use std::io::{Seek, SeekFrom, Write}; use std::thread; -use std::time::{Duration, Instant}; +use std::time::Duration; /// Mirrored in `tests/common/volumes.rs::ftruncate_flush_race`, and in the /// actuator's own path filter (`kernel/src/vfs.rs::stalled_metadata_window`). @@ -18,16 +16,15 @@ const PATH: &str = "/log/truncate-race.bin"; const FULL: usize = 3 * 4096; const SHORT: u64 = 5000; +/// Into the stalled window before the truncate: a pace that aims the race, and +/// never a verdict. const INTO_WINDOW: Duration = Duration::from_millis(50); -/// Only a serialised truncate waits this long against the 400ms stall; a lockless one returns in microseconds. -const CONTENDED: Duration = Duration::from_millis(150); const ATTEMPTS: u32 = 10; fn main() { let mut f = OpenOptions::new().create(true).write(true).open(PATH).expect("create"); - let mut contended = None; - for attempt in 0..ATTEMPTS { + for _ in 0..ATTEMPTS { f.seek(SeekFrom::Start(0)).expect("rewind"); f.write_all(&vec![0xB6u8; FULL]).expect("fill"); @@ -36,24 +33,11 @@ fn main() { thread::spawn(move || f.sync_all().expect("the stalled fsync")) }; thread::sleep(INTO_WINDOW); - let began = Instant::now(); f.set_len(SHORT).expect("truncate"); - let waited = began.elapsed(); flusher.join().expect("flusher panicked"); - - if waited >= CONTENDED { - contended = Some((attempt, waited)); - break; - } } - let Some((attempt, waited)) = contended else { - panic!( - "in {ATTEMPTS} attempts the truncate never once waited for the stalled flush — \ - the resize does not serialise with the metadata window", - ); - }; // The truncated size durable before the host reads the shut-down volume. f.sync_all().expect("the settling fsync"); - println!("attempt {attempt}: the truncate waited {waited:?} for the stalled flush; {SHORT} bytes settled"); + println!("{ATTEMPTS} truncates raced the stalled flush; {SHORT} bytes settled"); } diff --git a/tests/toyos-rust-tests/src/bin/futex_wake_counts.rs b/tests/toyos-rust-tests/src/bin/futex_wake_counts.rs index 6e383ea8c99..906ed0b545e 100644 --- a/tests/toyos-rust-tests/src/bin/futex_wake_counts.rs +++ b/tests/toyos-rust-tests/src/bin/futex_wake_counts.rs @@ -51,9 +51,14 @@ use std::sync::atomic::{AtomicU32, Ordering}; use std::thread; use std::time::Duration; +use toyos::endow::{Endowments, SYSCAP_LABEL}; +use toyos::syscap::SysCap; use toyos_abi::syscall::{self, MmapFlags, MmapProt}; use toyos_abi::RawHandle; +#[path = "../roster.rs"] +mod roster; + /// Two futex words in one page, `FUTEX_BUCKETS * 4` bytes apart, so /// `(phys >> 2) % 64` is the same for both. #[repr(C, align(4096))] @@ -69,15 +74,10 @@ static WORDS: SameBucket = SameBucket { sibling: AtomicU32::new(0), }; -/// How long the waiters are given to reach their `futex_wait` before the first -/// wake. **A margin and not a bound**: every assertion in [`counts`] is about a -/// *number returned*, so a waiter that had not parked yet makes that arm -/// weaker rather than wrong — it would count one fewer, and the count-limit -/// assertions would fail loudly rather than pass vacuously. -const PARK_MARGIN: Duration = Duration::from_millis(300); - static WORD_RETURNED: AtomicU32 = AtomicU32::new(0); static SIBLING_RETURNED: AtomicU32 = AtomicU32::new(0); +/// [`counts`]' waiters that have reached their `futex_wait`. +static COUNTS_ARMED: AtomicU32 = AtomicU32::new(0); /// This binary's own path. The sweeper is another *process* asking about /// frames this one freed, which is the whole of what the third arm is about, @@ -110,9 +110,12 @@ fn main() { Some(LET_GO) => return let_go(), _ => {} } - counts(); - claim_semantics(); - orphaned_by_unmap(); + let cap: SysCap = Endowments::get() + .take(SYSCAP_LABEL) + .expect("test-runner endows every binary it spawns a system capability"); + counts(&cap); + claim_semantics(&cap); + orphaned_by_unmap(&cap); revoked_while_still_mapped(); timeout_is_its_own_answer(); println!("futex_wake respects its count, names its word, says how many it woke, and ends"); @@ -144,7 +147,7 @@ fn timeout_is_its_own_answer() { assert_eq!(changed, 0, "a futex wait whose word did not match answered {changed}"); let waiter = thread::spawn(|| unsafe { syscall::futex_wait(WOKEN.as_ptr(), 7, None) }); - thread::sleep(PARK_MARGIN); + wait_until_parked(&WOKEN, 1); WOKEN.store(8, Ordering::SeqCst); unsafe { syscall::futex_wake(WOKEN.as_ptr(), 1) }; let woken = waiter.join().expect("the woken waiter panicked"); @@ -152,10 +155,15 @@ fn timeout_is_its_own_answer() { } /// A count-limited wake names its word and answers how many it woke. -fn counts() { +/// +/// **Parked by the kernel's roster, not by a probe**: a probe is a wake, and +/// the waiters it claims are still on their way back to the word when the +/// first wake below looks for them. +fn counts(cap: &SysCap) { let waiters: Vec<_> = (0..2) .map(|_| { thread::spawn(|| { + COUNTS_ARMED.fetch_add(1, Ordering::SeqCst); // Returns only once the word has actually changed: the kernel's // `futex_wait` re-reads it after every wake, which is what makes // "was this thread told" observable at all. @@ -165,10 +173,11 @@ fn counts() { }) .collect(); let sibling = thread::spawn(|| { + COUNTS_ARMED.fetch_add(1, Ordering::SeqCst); unsafe { syscall::futex_wait(WORDS.sibling.as_ptr(), 0, None) }; SIBLING_RETURNED.fetch_add(1, Ordering::SeqCst); }); - thread::sleep(PARK_MARGIN); + await_armed_and_blocked(cap, &COUNTS_ARMED, 3); // The word changes first, so a waiter that is told goes home instead of // re-parking — otherwise it would re-arm and be counted twice. @@ -180,7 +189,11 @@ fn counts() { "futex_wake(count=1) with two waiters answered {one}, and the ABI's answer is \ the number of threads woken", ); - thread::sleep(PARK_MARGIN); + // No deadline: the woken waiter's return is the event, and one that never + // comes is a hang the harness ceiling reds. + while WORD_RETURNED.load(Ordering::SeqCst) == 0 { + thread::yield_now(); + } let returned = WORD_RETURNED.load(Ordering::SeqCst); assert_eq!( returned, 1, @@ -214,11 +227,12 @@ fn counts() { static CLAIM_WORD: AtomicU32 = AtomicU32::new(0); static CLAIM_RETURNED: AtomicU32 = AtomicU32::new(0); +/// [`claim_semantics`]' waiters that have reached their `futex_wait`. +static CLAIM_ARMED: AtomicU32 = AtomicU32::new(0); +/// [`orphaned_by_unmap`]'s waiters that have reached their `futex_wait`. +static ORPHANS_ARMED: AtomicU32 = AtomicU32::new(0); static SPINNERS_STOP: AtomicU32 = AtomicU32::new(0); -/// How long the machine is left alone for two waiters to park on an idle CPU, -/// and for the spinners to reach their loops afterwards. -const SETTLE: Duration = Duration::from_millis(120); /// How many times the arrangement below is rebuilt before its own failure to /// hold is the verdict. const ATTEMPTS: usize = 6; @@ -242,37 +256,43 @@ const ATTEMPTS: usize = 6; /// dispatched until a quantum ends — 10 ms, against the microseconds the three /// wakes below take between them. /// -/// The waiters park **before** the spinners start, on an otherwise idle -/// machine, and are proved parked by a probe that answers how many claims it -/// won. Nothing posts to their word between that proof and the first wake -/// below, so "both are parked" is not a guess. What is left is the one thing +/// The waiters park **before** the spinners start, and are proved parked by the +/// roster, as [`counts`]' are: a probe is a wake, and the waiters it claims are +/// still on the bucket when the first wake below walks it. Nothing posts to +/// their word between that proof and the first wake below, so "both are +/// parked" is not a guess. What is left is the one thing /// the arrangement cannot make impossible — a quantum boundary landing inside /// those microseconds — and that is *checked* rather than assumed: a waiter /// that got out shows up in [`CLAIM_RETURNED`], and the attempt is rebuilt /// instead of asserted on. -fn claim_semantics() { +fn claim_semantics(cap: &SysCap) { for attempt in 1..=ATTEMPTS { CLAIM_WORD.store(0, Ordering::SeqCst); CLAIM_RETURNED.store(0, Ordering::SeqCst); + CLAIM_ARMED.store(0, Ordering::SeqCst); let waiters: Vec<_> = (0..2) .map(|_| { thread::spawn(|| { + CLAIM_ARMED.fetch_add(1, Ordering::SeqCst); unsafe { syscall::futex_wait(CLAIM_WORD.as_ptr(), 0, None) }; CLAIM_RETURNED.fetch_add(1, Ordering::SeqCst); }) }) .collect(); - - // Parked, and proved so: the probe answers the number of claims it won, - // and a waiter whose word has not changed re-parks. The machine is idle - // here, so the re-park is immediate and the settle below is generous. - wait_until_parked(&CLAIM_WORD, 2); - thread::sleep(SETTLE); + await_armed_and_blocked(cap, &CLAIM_ARMED, 2); // Now make every CPU busy, so nothing the three wakes claim can be - // dispatched before the last of them has run. + // dispatched before the last of them has run: this thread runs on one, + // and the roster shows a spinner running on every other. let spinners = start_spinners(); - thread::sleep(SETTLE); + let others = syscall::cpu_count() as usize - 1; + roster::await_true(|| { + roster::my_threads(cap) + .iter() + .filter(|&&(is_thread, state)| is_thread && state == roster::RUNNING) + .count() + == others + }); // Changed first, so a waiter that does get out goes home and says so // rather than re-parking invisibly. @@ -326,7 +346,8 @@ fn claim_semantics() { } } -/// Wake `word` until the kernel answers that it claimed `want` waiters. +/// Wake `word` until the kernel answers that it claimed `want` waiters, with no +/// deadline: waiters that never park are a hang the harness ceiling reds. /// /// The word must still hold what the waiters are waiting for, so every waiter /// this probe wakes re-checks and re-parks; what it costs is a round trip, and @@ -335,13 +356,22 @@ fn claim_semantics() { /// parked and unclaimed waiters answers `want` under every arithmetic this file /// is about. fn wait_until_parked(word: &AtomicU32, want: u64) { - for _ in 0..200 { - if unsafe { syscall::futex_wake(word.as_ptr(), 10) } == want { - return; - } - thread::sleep(Duration::from_millis(10)); - } - panic!("{want} waiters never parked on the word"); + wait_until_parked_raw(word.as_ptr(), want); +} + +/// Until `armed` reads `want` and the roster shows `want` of this process's +/// threads blocked, with no deadline. Each waiter counts itself into `armed` +/// with nothing but its `futex_wait` after, and they are the only threads of +/// this process that block, so a blocked one is blocked there. +fn await_armed_and_blocked(cap: &SysCap, armed: &AtomicU32, want: u32) { + roster::await_true(|| { + armed.load(Ordering::SeqCst) == want + && roster::my_threads(cap) + .iter() + .filter(|&&(is_thread, state)| is_thread && state == roster::BLOCKED) + .count() + == want as usize + }); } /// One never-yielding thread per CPU, so the kernel has no sleeping target to @@ -366,8 +396,6 @@ fn stop_spinners(spinners: Vec>) { } } -static ORPHAN_RETURNED: AtomicU32 = AtomicU32::new(0); - /// A wait whose word is unmapped ends there, and the frame carries no claim /// away with it. /// @@ -379,7 +407,7 @@ static ORPHAN_RETURNED: AtomicU32 = AtomicU32::new(0); /// stale nodes either has one of them stolen by the sweeper — a wake reported /// to a process with nothing parked — or, if no frame came back its way, leaves /// every waiter here parked on memory that is gone. -fn orphaned_by_unmap() { +fn orphaned_by_unmap(cap: &SysCap) { let mut sweeper = Command::new(SELF) .arg(SWEEP) .stdin(Stdio::piped()) @@ -409,15 +437,14 @@ fn orphaned_by_unmap() { .iter() .map(|&addr| { thread::spawn(move || { + ORPHANS_ARMED.fetch_add(1, Ordering::SeqCst); unsafe { syscall::futex_wait(addr as *const u32, 0, None) }; - ORPHAN_RETURNED.fetch_add(1, Ordering::SeqCst); }) }) .collect(); - for &addr in ®ions { - wait_until_parked_raw(addr as *const u32, 1); - } - thread::sleep(SETTLE); + // By the roster and not by a probe: a waiter a probe woke is still on its + // way back to its word when the unmap below looks for it. + await_armed_and_blocked(cap, &ORPHANS_ARMED, STALE_FRAMES as u32); // The unmap. Every frame behind these regions goes back to the PMM here, // and `mmap` hands the lowest free one to the next asker. @@ -435,19 +462,9 @@ fn orphaned_by_unmap() { status.code().unwrap_or(-1), ); - for _ in 0..100 { - if ORPHAN_RETURNED.load(Ordering::SeqCst) as usize == STALE_FRAMES { - break; - } - thread::sleep(Duration::from_millis(10)); - } - let home = ORPHAN_RETURNED.load(Ordering::SeqCst) as usize; - assert_eq!( - home, STALE_FRAMES, - "{home} of {STALE_FRAMES} waiters came back after the word each was parked on was \ - unmapped — nothing can ever post to a frame that has gone back to the PMM, so a wait \ - an unmap orphans is a wait nothing will end", - ); + // No deadline: nothing can ever post to a frame that has gone back to the + // PMM, so a wait an unmap orphans and does not end is a hang the harness + // ceiling reds. for waiter in waiters { waiter.join().expect("an orphaned waiter panicked"); } @@ -456,17 +473,13 @@ fn orphaned_by_unmap() { /// [`wait_until_parked`] for a word that is not a `static`. fn wait_until_parked_raw(word: *const u32, want: u64) { - for _ in 0..200 { - if unsafe { syscall::futex_wake(word, 10) } == want { - return; - } + // A pace, and never a verdict: the probe's wake re-parks the waiters it + // finds, so each look waits for them to be back. + while unsafe { syscall::futex_wake(word, 10) } != want { thread::sleep(Duration::from_millis(10)); } - panic!("{want} waiters never parked on a mapped word"); } -static REVOKED_RETURNED: AtomicU32 = AtomicU32::new(0); - /// A wait a revoke ended stays ended while its word still reads `expected`. /// /// The revoke is another process letting go of its own window onto the same @@ -482,11 +495,7 @@ fn revoked_while_still_mapped() { let word = window.wrapping_add(4096) as usize; let expected = unsafe { (word as *const u32).read_volatile() }; - let waiter = thread::spawn(move || { - let answer = unsafe { syscall::futex_wait(word as *const u32, expected, None) }; - REVOKED_RETURNED.store(1, Ordering::SeqCst); - answer - }); + let waiter = thread::spawn(move || unsafe { syscall::futex_wait(word as *const u32, expected, None) }); // A wake to an unchanged word re-parks inside the call, so the // registration the revoke must find is still there. wait_until_parked_raw(word as *const u32, 1); @@ -506,18 +515,9 @@ fn revoked_while_still_mapped() { out.status.code(), ); - for _ in 0..500 { - if REVOKED_RETURNED.load(Ordering::SeqCst) == 1 { - break; - } - thread::sleep(Duration::from_millis(10)); - } - assert_eq!( - REVOKED_RETURNED.load(Ordering::SeqCst), - 1, - "a waiter whose registration a revoke ended is still in futex_wait 5 s after it, \ - its word still mapped and still {expected:#x}: nothing can ever post to that wait", - ); + // No deadline: a waiter whose registration the revoke ended and that goes + // round its phase 1 for ever, its word still mapped and unchanged, is a + // hang the harness ceiling reds. waiter.join().expect("the revoked waiter panicked"); syscall::close(ends.read); syscall::close(ends.write); diff --git a/tests/toyos-rust-tests/src/bin/gpu_set_resolution.rs b/tests/toyos-rust-tests/src/bin/gpu_set_resolution.rs index ad64c6e91ad..9fa872fd3be 100644 --- a/tests/toyos-rust-tests/src/bin/gpu_set_resolution.rs +++ b/tests/toyos-rust-tests/src/bin/gpu_set_resolution.rs @@ -66,22 +66,17 @@ fn main() { /// The claim again, once the release the last close *queued* has run. /// `issues/kernel/deferred-release-outlives-its-syscall.md` is the kernel half: -/// `AlreadyExists` here is that tracked defect, not another holder. +/// `AlreadyExists` here is that tracked defect, not another holder, and a claim +/// never released is a hang the harness ceiling reds. fn reclaim(cap: &SysCap) -> FramebufferDev { - for _ in 0..RECLAIM_TRIES { + loop { match cap.claim(DeviceType::Framebuffer) { Ok(fb) => return fb, Err(SyscallError::AlreadyExists) => std::thread::sleep(RECLAIM_STEP), Err(e) => panic!("the second claim was refused {e:?}"), } } - panic!( - "the framebuffer claim was still held {:?} after its handle closed", - RECLAIM_STEP * RECLAIM_TRIES, - ); } -/// Five seconds over a queue drained at every syscall exit: exhausting it is a -/// defect, not a slow host. -const RECLAIM_TRIES: u32 = 500; +/// A pace and never a verdict. const RECLAIM_STEP: Duration = Duration::from_millis(10); diff --git a/tests/toyos-rust-tests/src/bin/handle_kill_policy.rs b/tests/toyos-rust-tests/src/bin/handle_kill_policy.rs index c28ad9cd22b..954d514ee52 100644 --- a/tests/toyos-rust-tests/src/bin/handle_kill_policy.rs +++ b/tests/toyos-rust-tests/src/bin/handle_kill_policy.rs @@ -56,30 +56,21 @@ use std::io::Write; use std::process::{Command, Stdio}; -use std::time::{Duration, Instant}; -use toyos::census::Census; use toyos::poller::{Poller, READABLE, WRITABLE}; use toyos::AsHandle; use toyos_abi::handle::Rights; use toyos_abi::syscall::{self, debug_action, MmapFlags, MmapProt, SpawnArgs, SyscallError}; use toyos_abi::RawHandle; +#[path = "../census_wait.rs"] +mod census_wait; + /// Where the child starts: `SpawnArgs` names a working directory or the spawn is refused. const CWD: &str = "/"; const SELF_PATH: &str = "/system/bin/test_rs_handle_kill_policy"; -/// How long a `POLL_ADD` that cannot ever fire is given to say so. -/// -/// **A liveness bound and never a verdict.** Every arm below submits a poll the -/// kernel can answer without waiting for anything — a refusal, or an object -/// with no readiness in the direction asked for — so the answer is due in the -/// submitting syscall itself. What this number separates is "answered" from -/// "waits forever", which is the whole defect, and four orders of magnitude -/// above a syscall is enough to do that on any machine this suite runs on. -const POLL_ANSWER: Duration = Duration::from_secs(2); - /// `process::HANDLE_FAULT_EXIT_CODE`. The shell convention for "died on /// SIGSEGV", which is the same class of mistake with a pointer instead of a /// handle. @@ -93,10 +84,6 @@ const UNHELD_SLOT: u32 = 3000; /// enough that one leaked object per round is a number no drain lag can hide. const CHURN_ROUNDS: usize = 16; -/// The most 10 ms census samples `settled_census` takes before answering with -/// what it last saw. `handle_lifetime`'s bound, for the same deferred queues. -const SETTLE_SAMPLES: usize = 100; - /// The three kinds that end the caller. Each is a role this binary runs as, and /// the description is what the kernel is being asked to refuse. const FATAL: &[(&str, &str)] = &[ @@ -181,13 +168,8 @@ fn test() { /// narrowed here to stage it. fn a_poll_without_wait_is_a_word() { let region = toyos::shm::SharedMemory::create(4096).expect("a region to poll"); - let answered = answered_within(POLL_ANSWER, region.as_handle(), READABLE); - assert!( - answered.is_some(), - "a POLL_ADD on a handle carrying no WAIT was neither answered nor refused in \ - {POLL_ANSWER:?} — it was registered on nothing", - ); - println!(" poll without WAIT: answered in {:?}, and the process is still here", answered.unwrap()); + answered(region.as_handle(), READABLE); + println!(" poll without WAIT: answered, and the process is still here"); } /// A `POLL_ADD` in a direction the object has no readiness in is answered. @@ -200,27 +182,22 @@ fn a_poll_without_wait_is_a_word() { /// kernel gave was silence. fn a_poll_with_no_source_is_answered() { let (read, _write) = toyos::pipe_pair().expect("a pipe with a read end"); - let answered = answered_within(POLL_ANSWER, read.as_handle(), WRITABLE); - assert!( - answered.is_some(), - "a POLL_ADD for writability on a pipe's read end went unanswered for \ - {POLL_ANSWER:?} — the poll was pushed with no source behind it", - ); - println!(" poll with no source: answered in {:?}", answered.unwrap()); + answered(read.as_handle(), WRITABLE); + println!(" poll with no source: answered"); } -/// How long a poll on `handle` took to complete, or `None` if it never did. +/// Wait, with no deadline, for a poll on `handle` to complete: a poll pushed +/// with no source behind it never does, and the harness ceiling reds it. /// /// The result word is deliberately not asserted on: `Poller::wait` hands back /// the token and not the CQE, and what these arms are about is that an answer /// arrives at all. Which word it is belongs to the kernel's own matrix. -fn answered_within(bound: Duration, handle: RawHandle, flags: u32) -> Option { +fn answered(handle: RawHandle, flags: u32) { let poller = Poller::new(1); poller.watch_raw(handle, flags, 0); - let started = Instant::now(); let mut seen = 0usize; - poller.wait(1, bound.as_nanos() as u64, |_| seen += 1); - (seen > 0).then(|| started.elapsed()) + poller.wait(1, u64::MAX, |_| seen += 1); + assert!(seen > 0, "a wait with no deadline returned with no completion"); } /// A right the handle does not carry is refused and the process carries on. @@ -284,31 +261,17 @@ fn a_full_table_is_a_word() { /// lag — it accumulates — so no *kind* may be higher after the second round of /// rounds than after the first. Per kind and not in total, because a total /// hides a leak of one kind behind churn in another. -/// -/// And each sample is a *settled* census, because the two-sample design alone -/// still loses to the lag it describes: on a loaded CI shard the last corpse's -/// own `Process` object outlived the parent's `wait` into the second census — -/// `[("Process", 6, 7)]`, twice on one shard, green alone both times (PR #141 -/// run 32307331537, the same deferral `handle_lifetime` measured decaying across -/// eight back-to-back reads). The kernel half is -/// `issues/kernel/deferred-release-outlives-its-syscall.md`; here it is a lag -/// and not a leak exactly when settling converges, which is what -/// `settled_census` requires before it answers. fn the_kills_release_what_they_held() { - let after_first = churn(CHURN_ROUNDS); - let after_second = churn(CHURN_ROUNDS); - let grown: Vec<_> = after_second.grown_since(&after_first).collect(); - assert!( - grown.is_empty(), - "{CHURN_ROUNDS} more killed processes left more live objects behind: \ - {grown:?} — first {after_first}, then {after_second}", - ); + churn(CHURN_ROUNDS); + let after_first = census_wait::settled(); + churn(CHURN_ROUNDS); + let after_second = census_wait::released_to(&after_first); println!(" census: {} live objects, then {}", after_first.total(), after_second.total()); } -/// `CHURN_ROUNDS` processes that each hold a pipe and a region and then die on a -/// bad handle, and what the kernel holds when they are all gone. -fn churn(rounds: usize) -> Census { +/// `rounds` processes that each hold a pipe and a region and then die on a bad +/// handle. +fn churn(rounds: usize) { for _ in 0..rounds { let status = Command::new(SELF_PATH) .arg("holder") @@ -317,26 +280,6 @@ fn churn(rounds: usize) -> Census { .expect("spawn a holder"); assert_eq!(status.code(), Some(HANDLE_FAULT), "a holder did not die on its bad handle"); } - settled_census() -} - -/// The census once the deferred queues have finished giving back what the -/// kills released. `handle_lifetime`'s `settled_free_bytes`, for object counts: -/// sample until two readings ten milliseconds apart agree, which is the -/// machine saying it has finished. **A liveness bound and not a margin** — a -/// kernel that leaks holds a stable, elevated census, is quiescent on the -/// first pair, and the grown-kinds assertion above reds exactly as before. -fn settled_census() -> Census { - let mut last = Census::now(); - for _ in 0..SETTLE_SAMPLES { - std::thread::sleep(Duration::from_millis(10)); - let next = Census::now(); - if next == last { - return next; - } - last = next; - } - last } /// Fill every slot and require the refusal to be a word. Exits 0, which is the @@ -444,7 +387,7 @@ fn fatal_role(role: &str) -> ! { let poller = Poller::new(1); poller.watch_raw(RawHandle(UNHELD_SLOT), READABLE, 0); let mut seen = 0usize; - poller.wait(1, POLL_ANSWER.as_nanos() as u64, |_| seen += 1); + poller.wait(1, u64::MAX, |_| seen += 1); panic!("a POLL_ADD on a slot this process never held left it running ({seen} CQEs)"); } // And of `stale`. The pipe is closed before the poll is submitted, so @@ -457,7 +400,7 @@ fn fatal_role(role: &str) -> ! { let poller = Poller::new(1); poller.watch_raw(closed, READABLE, 0); let mut seen = 0usize; - poller.wait(1, POLL_ANSWER.as_nanos() as u64, |_| seen += 1); + poller.wait(1, u64::MAX, |_| seen += 1); panic!("a POLL_ADD on a handle this process closed left it running ({seen} CQEs)"); } // No capability needed: the handle never resolves, so the call is @@ -510,12 +453,14 @@ fn fatal_role(role: &str) -> ! { // any thread ends every thread. Asserted from the exit code, which the // main thread never reaches to set. "faulting-thread" => { - std::thread::spawn(|| { + let faulting = std::thread::spawn(|| { let mut buf = [0u8; 8]; let n = syscall::read_nonblock(RawHandle(UNHELD_SLOT), &mut buf); panic!("a slot no thread of this process held answered {n:?}"); }); - std::thread::sleep(std::time::Duration::from_secs(10)); + // A kill ends this thread inside the join; a fault answered as a word + // ends the other one in its panic, and the join returns. + let _ = faulting.join(); panic!("a thread's handle fault left the process running"); } // Holds two objects the kernel has to give back, then dies where diff --git a/tests/toyos-rust-tests/src/bin/handle_lifetime.rs b/tests/toyos-rust-tests/src/bin/handle_lifetime.rs index 2ccddcbe2c0..b9ac13ca687 100644 --- a/tests/toyos-rust-tests/src/bin/handle_lifetime.rs +++ b/tests/toyos-rust-tests/src/bin/handle_lifetime.rs @@ -31,6 +31,9 @@ use toyos::{namespace, port, AsHandle}; use toyos_abi::inbox::RingLayout; use toyos_abi::syscall::{self, OpenFlags, SeekFrom, SyscallError, SERVE_PREFIX}; +#[path = "../census_wait.rs"] +mod census_wait; + const SELF_PATH: &str = "/system/bin/test_rs_handle_lifetime"; /// The name this test's own namespaces map to the port under test. Private to /// this process and its children, which is the whole of what a namespace is. @@ -215,10 +218,10 @@ fn kill_releases_acceptor() { /// a green run. `Census::grown_since` is the comparison the census header asks /// for. /// -/// **Both readings are [`settled_census`] and not [`Census::now`], because the -/// release does not finish inside the killing syscall** — see that function. +/// **The second reading is a wait and not a sample**: the release does not +/// finish inside the killing syscall (`census_wait`). fn kill_releases_ring() { - let before = settled_census(); + let before = census_wait::settled(); let (mut child, _) = spawn_holder("ring"); let held = Census::now(); @@ -239,54 +242,7 @@ fn kill_releases_ring() { // bound to `_` and is gone at the call. drop(child); - let after = settled_census(); - let grown: Vec<_> = after.grown_since(&before).collect(); - assert!( - grown.is_empty(), - "a killed process kept what it held: {grown:?} — first {before}, then {after}" - ); -} - -/// How many 10 ms samples [`settled_census`] will take before it stops asking. -/// Reaching it is not a failure — the last reading is handed back and the -/// caller's assertion is still the whole verdict. -const SETTLE_SAMPLES: usize = 100; - -/// The live-object census once the machine has stopped giving objects back. -/// -/// **A killed process's rings are not released by the syscall that killed it, -/// and the first reading after `wait` is therefore not the reading this test -/// is about.** The kill drains the victim's handle table on the killer's CPU, -/// which drops each ring's last handle onto the object layer's zero-handle -/// queue; the *release* happens when some CPU drains that queue. -/// `object::drain_zero_handles` clears its pending flag before it runs the -/// hooks, so the killer's own drain site — its syscall exit — can find the -/// queue empty while another CPU is still working through the batch, and every -/// ring still unreleased at that moment is released outside the killing -/// syscall altogether. -/// -/// Measured on this tree, 2026-08-19, alone in the guest: the deficit after -/// `wait` decays 2 MiB at a time across consecutive `SYS_SYSINFO` calls — -/// `[12, 10, 10, 10, 8, 6, 4, 2]` MiB over eight back-to-back reads — and over -/// twenty kill rounds free memory returned to its starting value every single -/// time. Nothing is lost; the first reading is simply early. The kernel half is -/// `issues/kernel/deferred-release-outlives-its-syscall.md`. -/// -/// So this samples until two readings ten milliseconds apart agree, which is -/// the machine saying it has finished. **It is a liveness bound and not a -/// margin**: a kernel that releases nothing holds a stable, elevated census, is -/// quiescent on the first pair, and reds at once. -fn settled_census() -> Census { - let mut last = Census::now(); - for _ in 0..SETTLE_SAMPLES { - std::thread::sleep(std::time::Duration::from_millis(10)); - let next = Census::now(); - if next == last { - return next; - } - last = next; - } - last + census_wait::released_to(&before); } /// A killed process's dirty file must still reach the filesystem: that flush diff --git a/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs b/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs index 48df97c4791..254cdf06fa6 100644 --- a/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs +++ b/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs @@ -11,17 +11,12 @@ use std::fs::{self, File}; use std::io::{Read, Write}; -use std::thread; -use std::time::Duration; /// Mirrored in `tests/common/storage.rs`, whose reader sees these names without /// the mount point: `/home` is a directory of DATA. const PINNED: &str = "/home/overwrite-pinned.bin"; const LOOPED: &str = "/home/overwrite-looped.bin"; const LEN: usize = 1_902_104; -const POLL_STEP: Duration = Duration::from_millis(10); -/// The window is the green arm's whole cost, and a step of it is the resolution. -const POLL_STEPS: u32 = 30; fn payload(seed: u8) -> Vec { (0..LEN).map(|i| (i.wrapping_mul(131) ^ seed as usize) as u8).collect() @@ -54,19 +49,10 @@ fn pinned_overwrite(first: &[u8], second: &[u8]) { writer.write_all(second).unwrap_or_else(|e| panic!("overwrite {PINNED}: {e}")); drop(held); - // Every length over the window, not one after a wait; `fs::metadata` is the `file_cache::size` a read bounds itself by. - let mut lowest = u64::MAX; - let mut steps = 0; - for _ in 0..POLL_STEPS { - let len = fs::metadata(PINNED).unwrap_or_else(|e| panic!("stat {PINNED}: {e}")).len(); - lowest = lowest.min(len); - steps += 1; - if lowest != LEN as u64 { - break; - } - thread::sleep(POLL_STEP); - } - println!(" poll: {steps} step(s) of {POLL_STEP:?}, lowest {lowest}"); + // One reading and no window: the device the host reads after the shutdown's + // drain is the time-free judge, and this length is what it is held against. + // `fs::metadata` is the `file_cache::size` a read bounds itself by. + let mut lowest = fs::metadata(PINNED).unwrap_or_else(|e| panic!("stat {PINNED}: {e}")).len(); let mut got = Vec::new(); File::open(PINNED) .unwrap_or_else(|e| panic!("re-open {PINNED}: {e}")) @@ -81,5 +67,5 @@ fn pinned_overwrite(first: &[u8], second: &[u8]) { assert_eq!(lowest, LEN as u64, "the same-length overwrite read back short"); assert!(got == second, "{PINNED} read back bytes that are not the overwrite's"); - println!(" PASS {PINNED} answered {LEN} bytes at every step of the poll"); + println!(" PASS {PINNED} answered {LEN} bytes at its stat and its read"); } diff --git a/tests/toyos-rust-tests/src/bin/i8042_keyboard.rs b/tests/toyos-rust-tests/src/bin/i8042_keyboard.rs index 7decbc89a71..8e0fc3cd31e 100644 --- a/tests/toyos-rust-tests/src/bin/i8042_keyboard.rs +++ b/tests/toyos-rust-tests/src/bin/i8042_keyboard.rs @@ -1,6 +1,6 @@ //! Claims the keyboard and prints what arrives, one line per event. //! -//! Driven by eight host tests that boot a guest with no USB HID at all and +//! Driven by host tests that boot a guest with no USB HID at all and //! inject through QMP once the ready line appears. Not a standalone test: on //! its own it would time out with nothing to report, which is why it is in //! RUST_SKIP. @@ -11,7 +11,7 @@ //! every window client make, so `tr=` below is what a real surface would put //! on a real shell's stdin. -use std::time::{Duration, Instant}; +use std::time::Duration; use toyos::device::Keyboard; use toyos::endow::Endowments; use toyos::syscap::SysCap; @@ -21,11 +21,10 @@ use toyos_abi::input::RawKeyEvent; const EVENT_SIZE: usize = std::mem::size_of::(); /// The host's end-of-run marker: the HID usage for the End key. None of this -/// binary's eight callers' own injections presses it, so its release is +/// binary's callers' own injections presses it, so its release is /// unambiguous — the same shape as `input_events.rs`'s right-button release -/// and `i8042_mouse.rs`'s own `ended`. The one caller whose verdict is a -/// report cadence rather than a delivered key (`i8042_health_cadence`) sends -/// no sentinel and runs out the deadline below instead. +/// and `i8042_mouse.rs`'s own `ended`. No deadline: a lost sentinel is a hang +/// the host's ceiling reds. const SENTINEL: u8 = 0x4D; fn main() { @@ -34,13 +33,10 @@ fn main() { let mut translator = window::configured_translator(); println!("===I8042_READY==="); - // A liveness ceiling, not the measurement: the normal path exits on the - // sentinel below, and this only bounds a run that lost it. - let deadline = Instant::now() + Duration::from_secs(5); let mut buf = [0u8; 512]; let mut seen = 0; let mut ended = false; - while !ended && Instant::now() < deadline { + while !ended { let n = keyboard.read_nonblock(&mut buf).unwrap_or(0); if n == 0 { std::thread::sleep(Duration::from_millis(5)); diff --git a/tests/toyos-rust-tests/src/bin/i8042_mouse.rs b/tests/toyos-rust-tests/src/bin/i8042_mouse.rs index 70e26e334e3..7fedab1456a 100644 --- a/tests/toyos-rust-tests/src/bin/i8042_mouse.rs +++ b/tests/toyos-rust-tests/src/bin/i8042_mouse.rs @@ -5,7 +5,7 @@ //! host's clock as well as its evidence. Not a standalone test — in RUST_SKIP //! for that reason. -use std::time::{Duration, Instant}; +use std::time::Duration; use toyos::device::Mouse; use toyos::endow::Endowments; use toyos::syscap::SysCap; @@ -17,21 +17,18 @@ const EVENT_SIZE: usize = 6; /// PS/2 bit 1 is right and so is HID boot-mouse bit 1. const RIGHT: u8 = 0x02; -/// A liveness ceiling, not a duration: the run ends on the marker's release, -/// so this only bounds a machine that lost it. -const RUN_CEILING: Duration = Duration::from_secs(60); - fn main() { let mouse: Mouse = capability().claim(DeviceType::Mouse).expect("i8042_mouse: no mouse device"); println!("===I8042_MOUSE_READY==="); - let deadline = Instant::now() + RUN_CEILING; let mut buf = [0u8; 1024]; let mut seen = 0; let mut right_down = false; let mut ended = false; - while !ended && Instant::now() < deadline { + // No deadline: the run ends on the marker's release, and a machine that + // lost it is a hang the host's ceiling reds. + while !ended { let n = mouse.read_nonblock(&mut buf).unwrap_or(0); if n == 0 { std::thread::sleep(Duration::from_millis(2)); diff --git a/tests/toyos-rust-tests/src/bin/inbox_cancel_wakes.rs b/tests/toyos-rust-tests/src/bin/inbox_cancel_wakes.rs index 5d6a033270d..1bacce60881 100644 --- a/tests/toyos-rust-tests/src/bin/inbox_cancel_wakes.rs +++ b/tests/toyos-rust-tests/src/bin/inbox_cancel_wakes.rs @@ -4,79 +4,62 @@ //! stdio leaves behind, and closing one of them is the whole stimulus. The //! pipe keeps a reader either way, so `close_read`'s own wake path is not //! involved and cannot mask the missing one. +//! +//! **The close waits for the park, and nothing here waits on a clock.** The +//! defect is only visible with the waiter parked — a close that lands first +//! leaves the completion sitting in the ring and the wait returns at once — so +//! the close is made once the kernel's roster says the waiter is blocked, and a +//! waiter the cancellation leaves parked is a hang the harness ceiling reds. use std::sync::atomic::{AtomicBool, Ordering}; use std::thread; -use std::time::{Duration, Instant}; +use toyos::endow::{Endowments, SYSCAP_LABEL}; use toyos::poller::{Poller, READABLE}; +use toyos::syscap::SysCap; use toyos_abi::syscall; -const TOKEN: u64 = 7; - -/// How long the waiter's own `wait` may take before this is a hang and not a -/// slow guest. The cancellation is posted the instant `close` runs, so a -/// healthy kernel returns in microseconds; seconds of margin cost nothing. -const PATIENCE: Duration = Duration::from_secs(10); +#[path = "../roster.rs"] +mod roster; -/// Between the waiter saying it is about to park and the close that cancels -/// it. Only the ordering matters, and it is benign in both directions — a -/// close that lands before the park leaves the completion sitting in the ring -/// and the wait returns at once — but the defect is only visible with the -/// thread genuinely parked, so this is wide. -const PARK_MARGIN: Duration = Duration::from_millis(500); - -static REGISTERED: AtomicBool = AtomicBool::new(false); -static RETURNED: AtomicBool = AtomicBool::new(false); +const TOKEN: u64 = 7; fn main() { + let cap: SysCap = Endowments::get() + .take(SYSCAP_LABEL) + .expect("test-runner endows every binary it spawns a system capability"); + let registered = AtomicBool::new(false); let pipe = syscall::pipe().expect("the pipe the waiter parks on"); // The second descriptor. Closing this one is what `ops::close` answers polls for, // while `pipe.read` keeps the pipe's reader count above zero. let dup = syscall::dup(pipe.read).expect("dup the read end"); - let waiter = thread::spawn(move || { - let poller = Poller::new(4); - poller.watch_raw(pipe.read, READABLE, TOKEN); - // A non-blocking enter, so the poll is registered in the kernel before - // anything is closed. Without it the close could reach a ring with - // nothing pending in it and cancel nothing at all, which is a - // different test that would pass on a broken kernel. - poller.wait(0, 0, |token| panic!("nothing is ready yet, got token {token}")); + let tokens = thread::scope(|s| { + let waiter = s.spawn(|| { + let poller = Poller::new(4); + poller.watch_raw(pipe.read, READABLE, TOKEN); + // A non-blocking enter, so the poll is registered in the kernel before + // anything is closed. Without it the close could reach a ring with + // nothing pending in it and cancel nothing at all, which is a + // different test that would pass on a broken kernel. + poller.wait(0, 0, |token| panic!("nothing is ready yet, got token {token}")); - REGISTERED.store(true, Ordering::Release); - let start = Instant::now(); - let mut tokens = Vec::new(); - poller.wait(1, u64::MAX, |token| tokens.push(token)); - RETURNED.store(true, Ordering::Release); - (start.elapsed(), tokens) - }); - - while !REGISTERED.load(Ordering::Acquire) { - thread::sleep(Duration::from_millis(10)); - } - thread::sleep(PARK_MARGIN); - syscall::close(dup); + registered.store(true, Ordering::Release); + let mut tokens = Vec::new(); + poller.wait(1, u64::MAX, |token| tokens.push(token)); + tokens + }); - let deadline = Instant::now() + PATIENCE; - while !RETURNED.load(Ordering::Acquire) && Instant::now() < deadline { - thread::sleep(Duration::from_millis(20)); - } - if !RETURNED.load(Ordering::Acquire) { - // Nothing can release the waiter now: its poll was cancelled, so the - // watcher list no longer names its ring and a write to the pipe would - // complete nothing. Report and take the process down rather than hand - // the harness a timeout with no message in it. - println!( - "the waiter is still parked {PATIENCE:?} after its poll was cancelled — \ - the cancellation posted a completion and woke nobody" - ); - std::process::exit(1); - } - - let (took, tokens) = waiter.join().expect("the waiter thread panicked"); + // The roster first: its syscall is the loop's preemption point. + roster::await_true(|| { + roster::my_threads(&cap).iter().any(|&(is_thread, state)| is_thread && state == roster::BLOCKED) + && registered.load(Ordering::Acquire) + }); + syscall::close(dup); + waiter.join().expect("the waiter thread panicked") + }); assert_eq!(tokens, [TOKEN], "the wait returned with the wrong completions"); - println!("a cancelled poll woke its waiter in {took:.2?}"); + println!("a cancelled poll woke its parked waiter"); syscall::close(pipe.read); syscall::close(pipe.write); diff --git a/tests/toyos-rust-tests/src/bin/input_events.rs b/tests/toyos-rust-tests/src/bin/input_events.rs index f4f6f24a2b3..8822cb51199 100644 --- a/tests/toyos-rust-tests/src/bin/input_events.rs +++ b/tests/toyos-rust-tests/src/bin/input_events.rs @@ -7,7 +7,7 @@ //! parses them with the same two functions. Not a standalone test: on its own //! it would report nothing, which is why it is in RUST_SKIP. -use std::time::{Duration, Instant}; +use std::time::Duration; use toyos::device::{Keyboard, Mouse}; use toyos::endow::Endowments; use toyos::syscap::SysCap; @@ -29,15 +29,14 @@ fn main() { let mut translator = window::configured_translator(); println!("===INPUT_READY==="); - // A liveness ceiling, not a duration: the host's sequence ends on the - // release of the right button, which nothing else in it produces, so a - // path that delivers nothing fails rather than hangs. - let deadline = Instant::now() + Duration::from_secs(30); + // No deadline: the host's sequence ends on the release of the right + // button, which nothing else in it produces, and a path that delivers + // nothing is a hang the host's ceiling reds. let mut buf = [0u8; 1024]; let (mut keys, mut pointer) = (0, 0); let mut right_down = false; let mut ended = false; - while !ended && Instant::now() < deadline { + while !ended { let mut idle = true; let n = keyboard.read_nonblock(&mut buf).unwrap_or(0); diff --git a/tests/toyos-rust-tests/src/bin/inspect_plays.rs b/tests/toyos-rust-tests/src/bin/inspect_plays.rs deleted file mode 100644 index 7a230b7f52d..00000000000 --- a/tests/toyos-rust-tests/src/bin/inspect_plays.rs +++ /dev/null @@ -1,52 +0,0 @@ -//! Plays periods through soundd and says how many frames soundd provably took, -//! for `inspect_reads_its_owners` to hold `sound.periods.*` against. -//! -//! **What is proven taken is what the ring says, not what was written.** A -//! slot can be filled only once soundd has emptied it, so of `fills` slots -//! filled into a ring of `slots`, at least `fills - slots` were taken, each in -//! a mix pass that submitted the periods it went out in and then published its -//! counters. soundd signals every client at the top of every pass and a read -//! drains every signal waiting, so the second signal read after the last -//! counted fill was written by a pass that began after that fill: every pass -//! that took a counted slot had published by then. -//! -//! At the device's own rate and channel count, so a client frame is a device -//! frame and no resampler stands between the two counts. - -use toyos::audio::{AudioStream, FORMAT_S16LE}; - -const RATE: u32 = 44_100; -const CHANNELS: u16 = 2; -/// Slots filled past the ring's first fill: every one is a slot soundd took. -const TAKEN_SLOTS: u64 = 32; - -fn main() { - let mut stream = AudioStream::open(RATE, CHANNELS, FORMAT_S16LE).expect("open a stream on soundd"); - assert_eq!(stream.device_sample_rate(), RATE, "the device is not at the client's rate"); - assert_eq!(stream.device_channels(), CHANNELS, "the device is not the client's channel count"); - let period_frames = u64::from(stream.period_frames()); - - // The first signal comes with the stream and finds the ring empty: that - // fill is the ring's own size, and says nothing about what soundd took. - let mut slots = 0u64; - stream - .wait_and_fill(|buf| { - buf.fill(0x11); - slots += 1; - }) - .expect("the first fill"); - let mut fills = slots; - while fills < slots + TAKEN_SLOTS { - stream - .wait_and_fill(|buf| { - buf.fill(0x11); - fills += 1; - }) - .expect("soundd went away mid-stream"); - } - for _ in 0..2 { - stream.wait_and_fill(|buf| buf.fill(0)).expect("soundd went away before publishing"); - } - stream.close(); - println!("inspect plays: soundd took {} frames", (fills - slots) * period_frames); -} diff --git a/tests/toyos-rust-tests/src/bin/lan_hold.rs b/tests/toyos-rust-tests/src/bin/lan_hold.rs index 40514f01848..feb9350eada 100644 --- a/tests/toyos-rust-tests/src/bin/lan_hold.rs +++ b/tests/toyos-rust-tests/src/bin/lan_hold.rs @@ -1,6 +1,7 @@ //! Hold the boot open for as long as the host needs to reach this machine over //! the cable, and exit. It asserts nothing; the host judges the records netd -//! wrote inside this window. +//! wrote inside this window, and `dump_nmi_probe`'s metal row the dump its +//! actuator stages inside it. use std::thread::sleep; use std::time::Duration; diff --git a/tests/toyos-rust-tests/src/bin/launcher_refusals.rs b/tests/toyos-rust-tests/src/bin/launcher_refusals.rs index ee7ccb2c392..cbf2f97973f 100644 --- a/tests/toyos-rust-tests/src/bin/launcher_refusals.rs +++ b/tests/toyos-rust-tests/src/bin/launcher_refusals.rs @@ -34,22 +34,13 @@ //! holder of the connector — the compositor, every terminal, every shell, sshd //! — parked the machine's only way to start a process for ever, with init alive //! and looking healthy. -//! -//! **Every answer this file waits for is bounded, and that is not decoration.** -//! A test that hangs instead of failing is worse than no test: a harness -//! timeout is a liveness guard rather than a verdict, and a guest that never -//! returns takes its whole shared boot down with it. So the launcher's replies -//! are read with [`answer_within`], and a launcher that has stopped answering -//! is an assertion with a name on it. The one arm that cannot be bounded from -//! here — `Command`, which blocks inside `std` — runs after a bounded launch -//! has already proved init is answering. use std::process::Command; -use std::time::{Duration, Instant}; use toyos::census::Census; use toyos::ipc::{Connection, FrameRx, RxStep}; use toyos::launch::{self, Launch}; +use toyos::poller::{Poller, READABLE}; use toyos::{namespace, port, AsHandle}; use toyos_abi::handle::Rights; use toyos_abi::syscall::{self, SyscallError}; @@ -59,14 +50,6 @@ use toyos_abi::RawHandle; /// a number no drain lag can hide. const ROUNDS: usize = 16; -/// How long init may take to answer a launch before this file calls it wedged. -/// -/// Generous by two orders of magnitude: a refusal is a frame decode and a -/// namespace build, and a grant is one `SYS_SPAWN`. What this bounds is the -/// launcher that answers *never*, and the number only decides how long the red -/// takes to arrive. -const ANSWER_BUDGET: Duration = Duration::from_secs(5); - /// Clients that connect to the launcher and then say nothing, held open across /// the launch that must still be answered. /// @@ -104,13 +87,10 @@ fn launcher() -> Connection { toyos::endow::service("launcher").expect("this process was endowed a launcher connector") } -/// Read the launcher's reply without ever blocking on it. -/// -/// `Err` is the verdict this file exists to be able to reach: a launcher that -/// has not answered inside the budget is one no `recv_header` would ever come -/// back from. -fn answer_within(conn: &Connection, budget: Duration) -> Result { - let deadline = Instant::now() + budget; +/// The launcher's reply, waited for with no deadline: a launcher that never +/// answers is a hang the harness ceiling reds. +fn answer(conn: &Connection) -> Result { + let poller = Poller::new(1); // Only the reply's type is judged here, so nothing of a payload is kept. let mut rx = FrameRx::<0>::new(); loop { @@ -119,10 +99,8 @@ fn answer_within(conn: &Connection, budget: Duration) -> Result return Err("the launcher dropped the connection"), RxStep::Malformed => return Err("the launcher sent a frame this protocol cannot describe"), RxStep::Idle => { - if Instant::now() >= deadline { - return Err("the launcher never answered"); - } - std::thread::sleep(Duration::from_millis(5)); + poller.watch(conn, READABLE, 0); + poller.wait(1, u64::MAX, |_| {}); } } } @@ -158,7 +136,7 @@ fn a_quiet_client_does_not_wedge_the_launcher() { conn.send_bytes_with_handles(&[], launch::MSG_LAUNCH, &buf[..len]) .expect("the launcher took the frame"); - match answer_within(&conn, ANSWER_BUDGET) { + match answer(&conn) { Ok(launch::MSG_NOT_DECLARED) => {} Ok(other) => panic!("the launcher answered {other} for a program nothing declares"), Err(why) => panic!( @@ -249,7 +227,7 @@ fn answer_to(cwd: &str, extras: &[(&str, RawHandle)]) -> u32 { let conn = launcher(); conn.send_bytes_with_handles(&handles[..count], launch::MSG_LAUNCH, &buf[..len]) .expect("the launcher took the frame"); - answer_within(&conn, ANSWER_BUDGET).expect("init answered the launch") + answer(&conn).expect("init answered the launch") } /// A cwd the launch does not state absolutely. std would join it onto init's diff --git a/tests/toyos-rust-tests/src/bin/locale_gate.rs b/tests/toyos-rust-tests/src/bin/locale_gate.rs index 3f0b3a9b207..cde9042ea41 100644 --- a/tests/toyos-rust-tests/src/bin/locale_gate.rs +++ b/tests/toyos-rust-tests/src/bin/locale_gate.rs @@ -30,7 +30,7 @@ use std::io::{BufRead, BufReader, Read}; use std::process::{Child, Command, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::Arc; -use std::time::{Duration, Instant}; +use std::time::Duration; use std::os::toyos::process::CommandExt; use toyos::device::Keyboard; use toyos::endow::Endowments; @@ -47,9 +47,7 @@ const EVENT_SIZE: usize = std::mem::size_of::(); /// The host's end-of-run marker for `layout`: the HID usage for the End key. /// The same sentinel and the same reason as `i8042_keyboard.rs` — nothing -/// `swiss_german_layout` injects presses End, so its release is unambiguous, -/// and without one the mode below spends its whole fallback deadline on a -/// keyboard that has nothing left to say. +/// `swiss_german_layout` injects presses End, so its release is unambiguous. const SENTINEL: u8 = 0x4D; const TOKEN_KEYBOARD: u64 = 1; @@ -161,16 +159,11 @@ fn layout(mut surface: Surface, connector: &Connector) { surface.drain_notices(); println!("===SWISS_READY==="); - // A liveness ceiling, not the measurement: the host's sequence ends on - // [`SENTINEL`]'s release, so the normal path leaves as soon as the last - // transition it typed has been reported, and only a run that lost the - // sentinel pays this. It used to be the whole run — eight seconds of an - // idle keyboard on every green `swiss_german_layout`, against half a second - // of injection. - let deadline = Instant::now() + Duration::from_secs(8); + // No deadline: the host's sequence ends on [`SENTINEL`]'s release, and a + // run that lost it is a hang the host's ceiling reds. let mut seen = 0; let mut ended = false; - while !ended && Instant::now() < deadline { + while !ended { surface.drain_keyboard(|key, text| { println!("kev usage=0x{:02x} mods=0x{:02x} tr={:?}", key.keycode, key.modifiers, text); seen += 1; @@ -214,8 +207,8 @@ fn detect(mut surface: Surface, connector: &Connector) { }); let poller = Poller::new(1 + Host::POLL_HANDLES); - let deadline = Instant::now() + Duration::from_secs(25); - while !wizard_done.load(Ordering::Relaxed) && Instant::now() < deadline { + // No deadline: a wizard that never ends is a hang the host's ceiling reds. + while !wizard_done.load(Ordering::Relaxed) { poller.watch(&surface.keyboard, READABLE, TOKEN_KEYBOARD); poller.watch_raw(surface.host.acceptor_handle(), READABLE, TOKEN_LISTEN); for client in surface.host.client_handles() { @@ -223,6 +216,8 @@ fn detect(mut surface: Surface, connector: &Connector) { } let mut ready = [false; 4]; + // A pace, and never a verdict: `wizard_done` is a flag and not a handle, + // so it is looked at again at least this often. poller.wait(1, 50_000_000, |token| { if (token as usize) < ready.len() { ready[token as usize] = true; @@ -239,9 +234,6 @@ fn detect(mut surface: Surface, connector: &Connector) { // was granted would be translated into nothing anyone is reading. surface.drain_keyboard(|_, _| {}); } - if !wizard_done.load(Ordering::Relaxed) { - println!("locale_gate: the wizard was still running after 25s"); - } println!("===DETECT_DRAINED==="); let mut stderr = child.stderr.take().expect("locale_gate: no stderr pipe"); diff --git a/tests/toyos-rust-tests/src/bin/log_flood.rs b/tests/toyos-rust-tests/src/bin/log_flood.rs index 934548bd6b1..1c6186e6e6c 100644 --- a/tests/toyos-rust-tests/src/bin/log_flood.rs +++ b/tests/toyos-rust-tests/src/bin/log_flood.rs @@ -7,7 +7,6 @@ //! the host's; `log_program_flood` runs it. use std::io::Write; -use std::time::Instant; /// Lines written, many times the records one log ring holds, each [`WIDTH`] /// bytes: what fills a ring is its slots, and a narrow line keeps the stop's @@ -17,20 +16,15 @@ const WIDTH: usize = 64; fn main() { let mut out = std::io::stdout().lock(); - let began = Instant::now(); - let mut slowest = 0u128; for i in 0..LINES { let head = format!("flood {i:06} "); let line = format!("{head}{}\n", "x".repeat(WIDTH - head.len() - 1)); - let at = Instant::now(); out.write_all(line.as_bytes()).expect("a write to the log never fails"); out.flush().expect("a write to the log never fails"); - slowest = slowest.max(at.elapsed().as_micros()); } let _ = writeln!( out, - "flood done lines={LINES} bytes={} ms={} slowest_write_us={slowest}", + "flood done lines={LINES} bytes={}", LINES * WIDTH, - began.elapsed().as_millis() ); } diff --git a/tests/toyos-rust-tests/src/bin/mmap_prot.rs b/tests/toyos-rust-tests/src/bin/mmap_prot.rs index db3d44cc61b..e592f4e7038 100644 --- a/tests/toyos-rust-tests/src/bin/mmap_prot.rs +++ b/tests/toyos-rust-tests/src/bin/mmap_prot.rs @@ -24,7 +24,6 @@ use std::process::{Command, Stdio}; use std::sync::mpsc; use std::thread; -use std::time::Duration; use toyos_abi::syscall::{mmap, munmap, MmapFlags, MmapProt, SyscallError, SYS_MMAP}; @@ -171,9 +170,6 @@ const UNPLACEABLE: [u64; 4] = [u64::MAX - PAGE_2M, u64::MAX - (PAGE_2M - 1), u64::MAX, 1 << 47]; const PAGE_2M: u64 = 2 * 1024 * 1024; -/// A liveness bound on the sibling's round, never a pace: it maps 4 KiB. -const SIBLING_BOUND: Duration = Duration::from_secs(10); - /// The sibling keeps taking both address-space locks while the main thread /// asks for every [`UNPLACEABLE`] length under both prots; told to stop, it /// finishes one whole round more, so at least one map and unmap begins after @@ -203,7 +199,7 @@ fn a_length_no_window_holds_is_refused_and_the_space_still_answers() { } done_tx.send(rounds).expect("the main thread waits for the last round"); }); - started_rx.recv_timeout(SIBLING_BOUND).expect("the sibling never finished its first round"); + started_rx.recv().expect("the sibling never finished its first round"); let mut answers = Vec::with_capacity(2 * UNPLACEABLE.len()); for prot in [MmapProt::NONE, MmapProt::READ | MmapProt::WRITE] { @@ -213,9 +209,8 @@ fn a_length_no_window_holds_is_refused_and_the_space_still_answers() { } stop_tx.send(()).expect("the sibling is still running"); - let rounds = done_rx.recv_timeout(SIBLING_BOUND).unwrap_or_else(|e| { - panic!("the sibling's address space stopped answering after the unplaceable lengths: {e:?}") - }); + // No deadline: a round that never answers is a hang the harness ceiling reds. + let rounds = done_rx.recv().expect("the sibling sends its count before it ends"); sibling.join().expect("the sibling thread"); for (prot, size, ret) in answers { diff --git a/tests/toyos-rust-tests/src/bin/munmap_reissues_read_window.rs b/tests/toyos-rust-tests/src/bin/munmap_reissues_read_window.rs index 03509d99054..ac6f41418a0 100644 --- a/tests/toyos-rust-tests/src/bin/munmap_reissues_read_window.rs +++ b/tests/toyos-rust-tests/src/bin/munmap_reissues_read_window.rs @@ -11,15 +11,16 @@ //! copy overwrites B's mapping — one process reading a pipe corrupts another //! allocation's memory. //! -//! The assertion is that B's mapping still holds B's byte, asked only of an -//! attempt that actually staged (A's `read` returned the whole buffer); a run -//! that never stages fails rather than passing, so nothing reads green for free. +//! The assertion is that B's mapping still holds B's byte. B unmaps only once +//! the kernel's roster says A is parked, so every run stages; A's `read` +//! returning the whole buffer is asserted, so nothing reads green for free. use std::sync::atomic::{AtomicI64, AtomicU32, Ordering}; use std::sync::Arc; use std::thread; -use std::time::Duration; +use toyos::endow::{Endowments, SYSCAP_LABEL}; +use toyos::syscap::SysCap; use toyos_abi::syscall::{close, mmap, munmap, pipe, read, write, MmapFlags, MmapProt}; const PAGE_2M: usize = 2 * 1024 * 1024; @@ -33,11 +34,6 @@ const PATTERN_A: u8 = 0xA1; /// What B writes into its own frame, and a safe kernel leaves there. const PATTERN_B: u8 = 0xB2; -/// How many times the sequence is attempted before "never staged" is the -/// verdict. Each attempt is a few tens of milliseconds; a staged one exits the -/// loop, and the only reason to retry is a victim unmapped before it parked. -const ATTEMPTS: usize = 12; - fn map_2m() -> *mut u8 { let p = unsafe { mmap( @@ -51,80 +47,72 @@ fn map_2m() -> *mut u8 { p } +#[path = "../roster.rs"] +mod roster; + fn main() { - let mut staged = false; - for attempt in 1..=ATTEMPTS { - let ends = pipe().expect("a pipe"); - let read_end = ends.read; - let write_end = ends.write; - - let victim = map_2m(); - let victim_addr = victim as usize; - - let ready = Arc::new(AtomicU32::new(0)); - let result = Arc::new(AtomicI64::new(i64::MIN)); - - let a = { - let ready = Arc::clone(&ready); - let result = Arc::clone(&result); - thread::spawn(move || { - // The buffer is valid when the reference is formed and when the - // read begins; the kernel owns the pointer once it parks, and B - // unmaps only then. Nothing in this thread dereferences it. - let buf = unsafe { core::slice::from_raw_parts_mut(victim_addr as *mut u8, LEN) }; - ready.store(1, Ordering::SeqCst); - let n = match read(read_end, buf) { - Ok(n) => n as i64, - Err(_) => -1, - }; - result.store(n, Ordering::SeqCst); - }) - }; - - while ready.load(Ordering::SeqCst) == 0 { - std::hint::spin_loop(); - } - // The victim set `ready` immediately before the read syscall, so this - // is the whole distance to the park. - thread::sleep(Duration::from_millis(30)); - - unsafe { munmap(victim_addr as *mut u8, PAGE_2M) }.expect("munmap the victim's buffer"); - - let sibling = map_2m(); - unsafe { core::ptr::write_bytes(sibling, PATTERN_B, LEN) }; - - let payload = [PATTERN_A; LEN]; - write(write_end, &payload).expect("write to wake the reader"); - - a.join().expect("the victim thread panicked"); - close(read_end); - close(write_end); - - let n = result.load(Ordering::SeqCst); - if n != LEN as i64 { - // Unmapped before the window was built: nothing was staged. - unsafe { munmap(sibling, PAGE_2M) }.ok(); - continue; - } - staged = true; - - let got = unsafe { core::slice::from_raw_parts(sibling, LEN) }; - let bad = got.iter().position(|&b| b != PATTERN_B); - assert!( - bad.is_none(), - "attempt {attempt}: a parked reader's copy reached a sibling's reissued frame — \ - byte {} of it is {:#x}, not {PATTERN_B:#x}", - bad.unwrap(), - got[bad.unwrap()], - ); - unsafe { munmap(sibling, PAGE_2M) }.expect("munmap the sibling's buffer"); - println!("munmap_reissues_read_window: staged on attempt {attempt}; the frame held under the copy"); - break; - } + let cap: SysCap = Endowments::get() + .take(SYSCAP_LABEL) + .expect("test-runner endows every binary it spawns a system capability"); + let ends = pipe().expect("a pipe"); + let read_end = ends.read; + let write_end = ends.write; + + let victim = map_2m(); + let victim_addr = victim as usize; + + let ready = Arc::new(AtomicU32::new(0)); + let result = Arc::new(AtomicI64::new(i64::MIN)); + + let a = { + let ready = Arc::clone(&ready); + let result = Arc::clone(&result); + thread::spawn(move || { + // The buffer is valid when the reference is formed and when the + // read begins; the kernel owns the pointer once it parks, and B + // unmaps only then. Nothing in this thread dereferences it. + let buf = unsafe { core::slice::from_raw_parts_mut(victim_addr as *mut u8, LEN) }; + ready.store(1, Ordering::SeqCst); + let n = match read(read_end, buf) { + Ok(n) => n as i64, + Err(_) => -1, + }; + result.store(n, Ordering::SeqCst); + }) + }; + + // The victim sets `ready` immediately before the read syscall, and its + // only park is that read's. The roster first: its syscall is the loop's + // preemption point. + roster::await_true(|| { + roster::my_threads(&cap).iter().any(|&(is_thread, state)| is_thread && state == roster::BLOCKED) + && ready.load(Ordering::SeqCst) == 1 + }); + + unsafe { munmap(victim_addr as *mut u8, PAGE_2M) }.expect("munmap the victim's buffer"); + + let sibling = map_2m(); + unsafe { core::ptr::write_bytes(sibling, PATTERN_B, LEN) }; + + let payload = [PATTERN_A; LEN]; + write(write_end, &payload).expect("write to wake the reader"); + + a.join().expect("the victim thread panicked"); + close(read_end); + close(write_end); + + let n = result.load(Ordering::SeqCst); + assert_eq!(n, LEN as i64, "the parked reader's read answered {n}, so the copy the test is about never ran"); + let got = unsafe { core::slice::from_raw_parts(sibling, LEN) }; + let bad = got.iter().position(|&b| b != PATTERN_B); assert!( - staged, - "no attempt parked a read across the sibling's munmap in {ATTEMPTS} tries — the test \ - staged nothing and proved nothing", + bad.is_none(), + "a parked reader's copy reached a sibling's reissued frame — \ + byte {} of it is {:#x}, not {PATTERN_B:#x}", + bad.unwrap(), + got[bad.unwrap()], ); + unsafe { munmap(sibling, PAGE_2M) }.expect("munmap the sibling's buffer"); + println!("munmap_reissues_read_window: the frame held under the copy"); } diff --git a/tests/toyos-rust-tests/src/bin/netd_caps.rs b/tests/toyos-rust-tests/src/bin/netd_caps.rs index fb8a5ef8d37..c53d0d4c78b 100644 --- a/tests/toyos-rust-tests/src/bin/netd_caps.rs +++ b/tests/toyos-rust-tests/src/bin/netd_caps.rs @@ -1,189 +1,93 @@ //! netd's piped-connection cap, from the client side. //! -//! Needs netd with a NIC in front of it, which only `tests/netcase` provides — -//! it is in `RUST_SKIP` and `netd_connection_caps` runs it there. +//! Needs netd with a NIC in front of it and the harness's host server behind +//! it, which only `tests/netcase` provides — it is in `RUST_SKIP` and +//! `netd_connection_caps` runs it there. //! -//! Every request goes to a destination that cannot answer, so each one netd -//! accepts stays in `pending_piped_connects` for the whole burst. That is -//! deliberate: the cap counts pending connects alongside established ones, so -//! a burst of connects is the shortest path to the boundary, and it needs no -//! peer at all. The requests are issued without reading any reply — `NetdConn` -//! splits send from receive — because reading would block on a connect that is -//! working exactly as intended. -//! -//! The argument is the burst size, not the answer. The host passes the cap -//! netd announced so this does not have to guess how many requests are enough; -//! where the boundary actually falls is measured here and compared there. +//! Every connect goes to the host server and is answered before the next is +//! asked, and every one it grants is held open: the cap counts established +//! connections, so the first `ResourceExhausted` is the boundary, and no clock +//! decides where it falls. Where the boundary falls is measured here and +//! compared with the cap netd announced by the host. + +#[path = "../netd_stream.rs"] +mod netd_stream; +use netd_stream::{HOST, NO_DEADLINE}; use toyos::net::{ - MsgType, NetError, NetdConn, PendingResponse, TcpConnectPipedRequest, TcpConnectResponse, - DATA_FROM_CLIENT, DATA_HANDLES, DATA_TO_CLIENT, + MsgType, NetError, NetdConn, TcpConnectPipedRequest, TcpConnectResponse, DATA_FROM_CLIENT, + DATA_HANDLES, DATA_TO_CLIENT, }; +use toyos::Pipe; use toyos_abi::syscall; -use toyos_abi::RawHandle; - -/// TEST-NET-1 (RFC 5737), reserved for documentation and guaranteed not to be -/// a real host. What matters is only that it does not complete a handshake. -const BLACK_HOLE: [u8; 4] = [192, 0, 2, 1]; -const PORT: u16 = 80; - -/// Long enough that netd is still holding every accepted connect when it -/// reaches the end of the burst — if the first ones expired on the way, the -/// count would never reach the cap and no refusal would ever be sent. -const TIMEOUT_MS: u32 = 4000; -/// How far past the announced cap to keep asking. Small: the point is to cross -/// the boundary, and every request costs netd an IPC connection. +/// How far past the boundary to keep asking. Small: the point is to cross the +/// boundary, and every request costs netd an IPC connection. const MARGIN: usize = 4; -/// How many times one connect may be refused by the *kernel* before this gives -/// up on it. -/// -/// **A count of attempts, not a span of time**, and it is measuring something -/// other than netd's cap: the port's queue holds `MAX_PENDING_CONNECTIONS` -/// connections netd has not accepted yet, and a burst issued as fast as a -/// single-threaded client can issue it outruns one accept per event-loop pass. -/// That is backpressure from the kernel and is retryable against the same peer; -/// netd's own cap is the `ResourceExhausted` in a *response*, which is what -/// this file is about and what it must not be confused with. -/// -/// The retry used to be invisible: `NetdConn::connect_blocking` spun a hundred -/// times at ten milliseconds over any error at all, so this test never saw the -/// queue at the same time as it never saw a boot race. That loop is deleted -/// because a boot race is unrepresentable now; this is the part of it that was -/// doing a different job. -const CONNECT_ATTEMPTS: usize = 200; - -/// One netd event-loop pass, which is what an attempt is waiting for. -/// -/// netd polls at 1 ms while it holds piped connections and accepts one -/// connection per pass, so this paces the client to the rate the queue drains -/// at. **The bound is [`CONNECT_ATTEMPTS`], not this** — a slow host makes the -/// attempts cheaper, never fewer. +/// One netd event-loop pass, which is what a connect the kernel's queue +/// refused waits for before it asks again. A pace and never a verdict. const PASS_NANOS: u64 = 1_000_000; fn main() { - let announced: usize = std::env::args() + let port: u16 = std::env::args() .nth(1) - .expect("netd_caps needs the announced cap as its argument") - .parse() - .expect("the announced cap must be a number"); - let burst = announced + MARGIN; - - let request = TcpConnectPipedRequest { - addr: BLACK_HOLE, - port: PORT, - _pad: 0, - timeout_ms: TIMEOUT_MS, - }; - - let mut sent: Vec = Vec::with_capacity(burst); - for i in 0..burst { - let conn = connect_past_the_queue(i); - sent.push( - conn.request_with_handles(&data_path(), MsgType::TcpConnectPiped, &request) - .unwrap_or_else(|e| panic!("request {i}: netd would not take it: {e:?}")), - ); - } - - let outcomes: Vec> = sent - .into_iter() - .map(|p| p.response::().err()) - .collect(); - - let Some(granted) = outcomes - .iter() - .position(|o| *o == Some(NetError::ResourceExhausted)) - else { - panic!( - "{burst} connects and netd never reported ResourceExhausted; outcomes: {}", - summarise(&outcomes) - ); + .and_then(|p| p.parse().ok()) + .expect("usage: netd_caps "); + let request = TcpConnectPipedRequest { addr: HOST, port, _pad: 0, timeout_ms: NO_DEADLINE }; + + let mut held: Vec<[Pipe; DATA_HANDLES]> = Vec::new(); + let granted = loop { + match connect(&request) { + Ok(kept) => held.push(kept), + Err(NetError::ResourceExhausted) => break held.len(), + Err(e) => panic!("connect {}: {e:?}, not a capacity refusal", held.len()), + } }; - // Both sides of the boundary, because "a refusal happened" is also true of // a netd that refused everything, and of one that refused at random. - for (i, outcome) in outcomes.iter().enumerate() { - if i < granted { - assert!( - outcome.is_some(), - "connect {i} to a black hole succeeded — the burst never filled netd" - ); - assert_ne!( - *outcome, - Some(NetError::ResourceExhausted), - "connect {i} was refused before the boundary at {granted}" - ); - } else { - assert_eq!( - *outcome, - Some(NetError::ResourceExhausted), - "connect {i} past the boundary at {granted} was not a capacity refusal" - ); - } + for past in 1..=MARGIN { + assert_eq!( + connect(&request).err(), + Some(NetError::ResourceExhausted), + "connect {} past the boundary at {granted} was not a capacity refusal", + granted + past, + ); } - assert!( - granted >= 2, - "only {granted} connects were accepted; netd is refusing, not bounding" - ); - - println!( - "netd caps: {granted} connections accepted then refused (accepted ones ended as {})", - summarise(&outcomes[..granted]) - ); + assert!(granted >= 2, "only {granted} connects were accepted; netd is refusing, not bounding"); + println!("netd caps: {granted} connections accepted then refused"); + drop(held); } -/// The two ends one request hands netd. -/// -/// **A fresh pair per request, where one pair used to serve the whole burst.** -/// The ends travel with the request now rather than being named by an id netd -/// would reopen later, and a handle that has been moved cannot be moved again. -/// This side's own two ends are dropped where they are made: nothing here will -/// read or write them, and each pipe stays alive on the end netd holds — which -/// is exactly where the cap is counting it. -fn data_path() -> [RawHandle; DATA_HANDLES] { +/// One connect, answered before this returns. What a granted one answers is +/// this side's two ends of its data path: while they are held netd holds the +/// connection, which is exactly where the cap is counting it. +fn connect(request: &TcpConnectPipedRequest) -> Result<[Pipe; DATA_HANDLES], NetError> { let (to_client_read, to_client_write) = toyos::pipe_pair().expect("the pipe netd writes into"); let (from_client_read, from_client_write) = toyos::pipe_pair().expect("the pipe netd reads from"); let mut handles = [toyos_abi::HANDLE_INVALID; DATA_HANDLES]; handles[DATA_TO_CLIENT] = to_client_write.into_raw(); handles[DATA_FROM_CLIENT] = from_client_read.into_raw(); - drop(to_client_read); - drop(from_client_write); - handles -} - -/// The distinct outcomes, in first-seen order. Printed so the log says what -/// the accepted connects actually died of rather than only how many there were. -fn summarise(outcomes: &[Option]) -> String { - let mut kinds: Vec = Vec::new(); - for outcome in outcomes { - let kind = format!("{outcome:?}"); - if !kinds.contains(&kind) { - kinds.push(kind); - } - } - kinds.join(", ") + netd() + .request_with_handles(&handles, MsgType::TcpConnectPiped, request) + .unwrap_or_else(|e| panic!("netd would not take a connect: {e:?}")) + .response::()?; + Ok([to_client_read, from_client_write]) } -/// One connection, retrying only the kernel's queue-full answer. +/// A connection to netd, asking again with no bound while the kernel's queue +/// of connections netd has not accepted yet is full. /// -/// Every other refusal ends the test where it happens: "netd is not reachable" -/// and "netd has not drained its accept queue yet" are different facts and only -/// the second one is worth another attempt. -fn connect_past_the_queue(request: usize) -> NetdConn { - for attempt in 0..CONNECT_ATTEMPTS { +/// That refusal is backpressure from the kernel, retryable against the same +/// peer; netd's own cap is the `ResourceExhausted` in a *response*, which is +/// what this file is about and what it must not be confused with. +fn netd() -> NetdConn { + loop { match NetdConn::connect() { Ok(conn) => return conn, - Err(NetError::ResourceExhausted) => { - let _ = attempt; - syscall::nanosleep(PASS_NANOS); - } - Err(e) => panic!("request {request}: could not reach netd: {e:?}"), + Err(NetError::ResourceExhausted) => syscall::nanosleep(PASS_NANOS), + Err(e) => panic!("could not reach netd: {e:?}"), } } - panic!( - "request {request}: the port queue stayed full for {CONNECT_ATTEMPTS} attempts — \ - netd is not accepting" - ) } diff --git a/tests/toyos-rust-tests/src/bin/netd_held_open.rs b/tests/toyos-rust-tests/src/bin/netd_held_open.rs index 6d5698e6681..d6d851b6de1 100644 --- a/tests/toyos-rust-tests/src/bin/netd_held_open.rs +++ b/tests/toyos-rust-tests/src/bin/netd_held_open.rs @@ -17,9 +17,7 @@ #[path = "../netd_stream.rs"] mod netd_stream; -use std::time::{Duration, Instant}; - -use netd_stream::{ask, await_ring_full, await_ring_holds, read_pattern_from, ring_capacity, Ask, HOST}; +use netd_stream::{ask, await_ring_full, await_ring_holds, read_pattern_from, ring_capacity, Ask, HOST, NO_DEADLINE}; /// Bytes the host sends past the ring's capacity: less than the socket's /// 64 KiB buffer, so the window never closes on the peer. @@ -28,11 +26,6 @@ const PAST_THE_RING: u64 = 32 * 1024; /// Room made at a time, and the bytes each step waits to see arrive. const STEP: u64 = 1024; -/// Liveness guards, said by name: the host has all of it to send the moment -/// this connects, and each step is one pass of netd's. -const FILL_BOUND: Duration = Duration::from_secs(60); -const STEP_BOUND: Duration = Duration::from_secs(20); - fn main() { let port: u16 = std::env::args() .nth(1) @@ -42,23 +35,18 @@ fn main() { let total = capacity + PAST_THE_RING; println!("netd_held_open: ring capacity {capacity}, expecting {total} bytes and no FIN"); - let conn = toyos::net::tcp_connect(HOST, port, 30_000).expect("connect to the host server"); + let conn = toyos::net::tcp_connect(HOST, port, NO_DEADLINE).expect("connect to the host server"); ask(&conn.tx, Ask::Held(total)); - await_ring_full(&conn.rx, capacity, FILL_BOUND); + await_ring_full(&conn.rx, capacity); println!("netd_held_open: the ring is full at {capacity} bytes unread; making room a step at a time"); let what = format!("netd_held_open (ring capacity {capacity}, the peer silent)"); - let started = Instant::now(); let mut at = 0; while at < PAST_THE_RING { - at = read_pattern_from(&conn.rx, at, at + STEP, STEP_BOUND, &what); - await_ring_holds(&conn.rx, capacity, capacity + at - 16, STEP_BOUND); + at = read_pattern_from(&conn.rx, at, at + STEP, &what); + await_ring_holds(&conn.rx, capacity, capacity + at - 16); } - let stepped = started.elapsed(); - let at = read_pattern_from(&conn.rx, at, total, STEP_BOUND, &what); + let at = read_pattern_from(&conn.rx, at, total, &what); assert_eq!(at, total, "the stream ended after {at} of {total} bytes, every one of them right"); - println!( - "netd_held_open: ok bytes={at}, the last {PAST_THE_RING} moved a {STEP}-byte step at a time in {} ms", - stepped.as_millis() - ); + println!("netd_held_open: ok bytes={at}, the last {PAST_THE_RING} moved a {STEP}-byte step at a time"); } diff --git a/tests/toyos-rust-tests/src/bin/netd_hostile_peer.rs b/tests/toyos-rust-tests/src/bin/netd_hostile_peer.rs index 2794d335bc8..d8bde50319b 100644 --- a/tests/toyos-rust-tests/src/bin/netd_hostile_peer.rs +++ b/tests/toyos-rust-tests/src/bin/netd_hostile_peer.rs @@ -11,32 +11,19 @@ //! Needs netd with a NIC in front of it, which only `tests/netcase` provides — //! it is in `RUST_SKIP` and `netd_hostile_peer` runs it there. //! -//! **Every wait here is bounded and every failure is named.** The obvious way -//! to ask netd a question is `toyos::net::dns_lookup`, and against a netd that -//! has stopped serving it blocks in `recv_header` forever — which is the exact -//! failure under test, so it cannot also be how the test waits. +//! **No wait here has a deadline.** A netd parked on a client never answers, and +//! the harness ceiling reds the hang. use std::process::exit; -use std::time::{Duration, Instant}; use toyos::endow; use toyos::AsHandle; use toyos::ipc::{self, RxStep}; +use toyos::poller::{Poller, READABLE}; use toyos::net::{MsgType, RespType}; use toyos::Connection; use toyos_abi::syscall; -/// How long netd may take to answer a question it answers from the request -/// itself. Four orders of magnitude over the real thing, and it still fails in -/// two seconds rather than hanging the boot. -const ANSWER_TIMEOUT: Duration = Duration::from_secs(2); -/// How long netd may take to rule on a frame it cannot act on. -const RULING_TIMEOUT: Duration = Duration::from_secs(2); -/// netd's own `HANDSHAKE_TIMEOUT` is 2 s. This is the ceiling on observing it, -/// with enough slack that a busy host is not a failure. -const HANDSHAKE_CEILING: Duration = Duration::from_secs(10); -const POLL: Duration = Duration::from_millis(10); - /// A literal address, which netd parses out of the request and answers without /// a packet — so this asks whether the daemon is *serving*, on a machine whose /// NIC has nobody on the other end of it. @@ -44,17 +31,6 @@ const LITERAL: &[u8] = b"192.0.2.7"; /// What netd sends back for [`LITERAL`]: one address, four bytes, the octets. const LITERAL_REPLY: [u8; 6] = [1, 4, 192, 0, 2, 7]; -/// Hostile connections opened in the run that fills netd's pending table. Past -/// its cap netd refuses by name, so this has to be comfortably over it. -const SILENT_BURST: usize = 48; -/// One netd pass per connection, so the table fills instead of the *kernel's* -/// 32-deep unaccepted queue refusing the connects before netd ever sees them. -const BURST_PACE: Duration = Duration::from_millis(1); -/// Long enough for netd to take one pass over a poller full of ready handles. Two -/// orders of magnitude over the pass itself, and far inside netd's own 2 s -/// handshake deadline, which is what would make the survivors miscount. -const SETTLE: Duration = Duration::from_millis(100); - /// A frame netd cannot act on, and what it must do about it. struct Case { name: &'static str, @@ -110,79 +86,45 @@ fn main() { exit(1); } - if case.ruled && !ruled_on(&conn, RULING_TIMEOUT) { - eprintln!( - "[{}] netd neither answered the frame nor closed the connection", - case.name - ); - exit(1); + if case.ruled { + await_ruling(&conn); } drop(conn); } - // A connection that never says anything must not be netd's to hold forever. + // A connection that never says anything must not be netd's to hold forever: + // its handshake deadline is netd's own, and the close is what is waited for. let silent = endow::service("netd").expect("netd is not serving"); - let opened = Instant::now(); - if !closed_by_peer(&silent, HANDSHAKE_CEILING) { - eprintln!("netd held a silent connection for {:?} without dropping it", opened.elapsed()); - exit(1); - } - let handshake = opened.elapsed(); + await_close(&silent); drop(silent); - - // And a burst of them must be bounded rather than accumulated. Liveness is - // *not* asserted while the table is full: refusing a new connection there - // is the bound doing its job, and a test that called that a stall would be - // asserting the opposite of the design. - let mut held = Vec::new(); - for _ in 0..SILENT_BURST { - if let Ok(conn) = endow::service("netd") { - held.push(conn); - } - syscall::nanosleep(BURST_PACE.as_nanos() as u64); - } - syscall::nanosleep(SETTLE.as_nanos() as u64); - let opened = held.len(); - // netd drops what it will not hold, so the survivors are the ones it took. - held.retain(|c| !closed_by_peer(c, Duration::ZERO)); - let kept = held.len(); - // Both sides of the boundary. "It bounded them" is also true of a netd that - // refused every one, which would be a broken daemon reported as a working - // bound — the same non-vacuity check `netd_caps` makes. - assert!(kept < opened, "netd held all {opened} unidentified connections — nothing bounded them"); - assert!(kept >= 2, "netd held only {kept} unidentified connections; it is refusing, not bounding"); - - // Released, netd must come back — a bound that is really a wedge would - // stay refusing after the thing it was bounding is gone. - drop(held); - syscall::nanosleep(SETTLE.as_nanos() as u64); if let Err(e) = still_serving() { - eprintln!("[after the burst] {e}"); + eprintln!("[after the silent connection] {e}"); exit(1); } println!( - "netd hostile peer: {} malformed frames refused, {kept} of {opened} unidentified \ - connections held, silent one dropped after {} ms, netd alive", + "netd hostile peer: {} malformed frames refused, silent one dropped, netd alive", cases.len(), - handshake.as_millis(), ); } /// Ask netd something it answers from the request itself, without blocking. /// /// [`ipc::FrameRx`] is the SDK's non-blocking framing — the same type netd now -/// reads its clients with — so the deadline is this test's and not the peer's. +/// reads its clients with — waited on with no deadline. fn still_serving() -> Result<(), String> { let conn = endow::service("netd").map_err(|e| format!("netd refused a connection: {e:?}"))?; conn.try_send_bytes(MsgType::DnsLookup as u32, LITERAL) .map_err(|e| format!("netd would not take a request: {e:?}"))?; let mut rx = ipc::FrameRx::<16>::new(); - let deadline = Instant::now() + ANSWER_TIMEOUT; - while Instant::now() < deadline { + let poller = Poller::new(1); + loop { match rx.pump(&conn) { - RxStep::Idle => syscall::nanosleep(POLL.as_nanos() as u64), + RxStep::Idle => { + poller.watch(&conn, READABLE, 0); + poller.wait(1, u64::MAX, |_| {}); + } RxStep::Eof => return Err("netd closed a request without answering it".to_string()), RxStep::Malformed => return Err("netd sent a frame the SDK cannot read".to_string()), RxStep::Frame { msg_type, payload_len } => { @@ -197,47 +139,47 @@ fn still_serving() -> Result<(), String> { } } } - Err(format!( - "netd did not answer a request in {ANSWER_TIMEOUT:?} — it is parked on a client" - )) } -/// Did netd either answer this connection or close it? +/// Wait, with no deadline, until netd has either answered this connection or +/// closed it. /// /// Both are rulings. An answer is the better one — a client learns that its /// frame was refused — and a close is what a frame with no locatable next -/// message boundary gets. -fn ruled_on(conn: &Connection, within: Duration) -> bool { +/// message boundary gets, and what a connection that never finished its +/// request gets at netd's handshake deadline. +fn await_ruling(conn: &Connection) { let mut buf = [0u8; 8]; - let deadline = Instant::now() + within; - while Instant::now() < deadline { + let poller = Poller::new(1); + loop { match conn.read_nonblock(&mut buf) { - Ok(_) => return true, - Err(syscall::SyscallError::WouldBlock) => syscall::nanosleep(POLL.as_nanos() as u64), - // The connection itself is gone, which is the same hang-up seen from the - // other end of the same race. - Err(_) => return true, + Err(syscall::SyscallError::WouldBlock) => { + poller.watch(conn, READABLE, 0); + poller.wait(1, u64::MAX, |_| {}); + } + // Bytes, EOF, or the connection itself gone: each is netd's word. + _ => return, } } - false } -/// Did netd hang up? `read_nonblock` returning 0 is EOF; `WouldBlock` is "not -/// yet"; anything else it sent means the connection is alive, which is not what -/// was asked. -fn closed_by_peer(conn: &Connection, within: Duration) -> bool { +/// Wait, with no deadline, for netd to close `conn`, which never asked anything: +/// bytes on it are an answer to nothing. +fn await_close(conn: &Connection) { let mut buf = [0u8; 8]; - let deadline = Instant::now() + within; + let poller = Poller::new(1); loop { match conn.read_nonblock(&mut buf) { - Ok(0) => return true, - Ok(_) => return false, - Err(syscall::SyscallError::WouldBlock) => {} - Err(_) => return true, - } - if Instant::now() >= deadline { - return false; + Err(syscall::SyscallError::WouldBlock) => { + poller.watch(conn, READABLE, 0); + poller.wait(1, u64::MAX, |_| {}); + } + Ok(n) if n > 0 => { + eprintln!("netd answered a connection that never asked anything with {n} bytes"); + exit(1); + } + // EOF, or the connection itself gone. + _ => return, } - syscall::nanosleep(POLL.as_nanos() as u64); } } diff --git a/tests/toyos-rust-tests/src/bin/netd_lookup_let_go.rs b/tests/toyos-rust-tests/src/bin/netd_lookup_let_go.rs index c93fec198fa..3eaf4b52edd 100644 --- a/tests/toyos-rust-tests/src/bin/netd_lookup_let_go.rs +++ b/tests/toyos-rust-tests/src/bin/netd_lookup_let_go.rs @@ -13,13 +13,10 @@ //! //! **Spoke again.** A connection carries one request, so a client that says //! more while its lookup is in flight is dropped: its connection closes with -//! no answer, long before the schedule would have answered it. +//! no answer, where the schedule would have answered it timed out. //! //! `netd_lookup_let_go: ok` is the only success line. -use std::sync::mpsc; -use std::time::{Duration, Instant}; - use toyos::net::{dns_lookup, MsgType, NetError, NetdConn, PendingResponse}; use toyos::poller::{Poller, READABLE}; use toyos::Connection; @@ -31,14 +28,6 @@ const CAP: usize = toyos_dns::MAX_LOOKUPS; /// Any name: nothing on this network answers one. const NAME: &str = "unanswered.example"; -/// How long a lookup's whole schedule runs with the one resolver the lease -/// names. -const SCHEDULE: Duration = Duration::from_millis(toyos_dns::ROUNDS as u64 * toyos_dns::WAIT_MS); - -/// How long anything here may take before this program says so by name. A -/// bound, not a pace. -const WITHIN: Duration = Duration::from_secs(60); - fn main() { hung_up(); spoke_again(); @@ -48,24 +37,21 @@ fn main() { fn hung_up() { let held: Vec = (0..CAP).map(|_| ask()).collect(); assert_eq!( - lookup_within(), + lookup(), Err(NetError::ResourceExhausted), "lookup {} was not refused as exhausted, so the {CAP} before it are not all in flight", CAP + 1 ); println!("netd_lookup_let_go: {CAP} lookups in flight, and the next refused"); drop(held); - let asked = Instant::now(); - let answer = lookup_within(); - let took = asked.elapsed(); + let answer = lookup(); assert_ne!( answer, Err(NetError::ResourceExhausted), "{CAP} lookups hung up and the next was refused as exhausted: netd did not let them go" ); assert_eq!(answer, Err(NetError::TimedOut), "a lookup nothing answers"); - assert!(took >= SCHEDULE, "a lookup nothing answers ended timed out after {took:?}, before its {SCHEDULE:?}"); - println!("netd_lookup_let_go: {CAP} hung up, and the next was asked and timed out after {took:?}"); + println!("netd_lookup_let_go: {CAP} hung up, and the next was asked and timed out"); } fn spoke_again() { @@ -73,18 +59,15 @@ fn spoke_again() { chatty.send_bytes(MsgType::DnsLookup as u32, NAME.as_bytes()).expect("netd takes a lookup"); let held: Vec = (1..CAP).map(|_| ask()).collect(); assert_eq!( - lookup_within(), + lookup(), Err(NetError::ResourceExhausted), "lookup {} was not refused as exhausted, so the {CAP} before it are not all in flight", CAP + 1 ); - let spoke = Instant::now(); chatty.send_bytes(MsgType::DnsLookup as u32, NAME.as_bytes()).expect("netd's end is still open"); let answered = closed_or_answered(&chatty); - let took = spoke.elapsed(); - assert_eq!(answered, 0, "a client that spoke again while its lookup ran was answered, after {took:?}"); - assert!(took < SCHEDULE, "a client that spoke again was dropped after {took:?}, not at once"); - println!("netd_lookup_let_go: a client that spoke again was dropped after {took:?}, unanswered"); + assert_eq!(answered, 0, "a client that spoke again while its lookup ran was answered"); + println!("netd_lookup_let_go: a client that spoke again was dropped, unanswered"); drop(held); } @@ -96,18 +79,16 @@ fn ask() -> PendingResponse { .expect("netd takes a lookup") } -/// A lookup of [`NAME`] to its end, which has to come within [`WITHIN`]. -fn lookup_within() -> Result { - let (answered, answer) = mpsc::channel(); - std::thread::spawn(move || answered.send(dns_lookup(NAME, &mut [[0; 4]; 4]))); - answer.recv_timeout(WITHIN).unwrap_or_else(|_| panic!("netd did not end a lookup within {WITHIN:?}")) +/// A lookup of [`NAME`] to its end, with no deadline: a lookup netd never ends +/// is a hang the harness ceiling reds. +fn lookup() -> Result { + dns_lookup(NAME, &mut [[0; 4]; 4]) } -/// Bytes netd wrote on `conn` before it closed, once it has closed, within -/// [`WITHIN`]. +/// Bytes netd wrote on `conn` before it closed, once it has closed, with no +/// deadline. fn closed_or_answered(conn: &Connection) -> usize { let poller = Poller::new(1); - let deadline = Instant::now() + WITHIN; let mut got = 0; let mut buf = [0u8; 64]; loop { @@ -115,10 +96,8 @@ fn closed_or_answered(conn: &Connection) -> usize { Ok(0) => return got, Ok(n) => got += n, Err(SyscallError::WouldBlock) => { - let left = deadline.saturating_duration_since(Instant::now()); - assert!(!left.is_zero(), "netd neither answered nor closed a connection within {WITHIN:?}"); poller.watch(conn, READABLE, 0); - poller.wait(1, left.as_nanos() as u64, |_| {}); + poller.wait(1, u64::MAX, |_| {}); } Err(e) => panic!("reading netd's end of a lookup's connection: {e:?}"), } diff --git a/tests/toyos-rust-tests/src/bin/netd_refused_pipes.rs b/tests/toyos-rust-tests/src/bin/netd_refused_pipes.rs index ab9ded031a2..d501a0a3b31 100644 --- a/tests/toyos-rust-tests/src/bin/netd_refused_pipes.rs +++ b/tests/toyos-rust-tests/src/bin/netd_refused_pipes.rs @@ -31,11 +31,9 @@ #[path = "../netd_stream.rs"] mod netd_stream; -use std::time::{Duration, Instant}; - use netd_stream::{ ask, ask_bytes, await_ring_full, await_until, keep_full_until_released, read_pattern, ring_capacity, Ask, - FORWARDED_PORT, HOST, + FORWARDED_PORT, HOST, NO_DEADLINE, }; use toyos::net::{ MsgType, NetError, NetdConn, TcpBindPipedRequest, TcpBindResponse, TcpConnectPipedRequest, @@ -46,10 +44,6 @@ use toyos::Pipe; use toyos_abi::syscall::{self, OpenFlags, SeekFrom, SyscallError}; use toyos_abi::RawHandle; -/// How long netd may take to act on something it has been handed. Orders of -/// magnitude over a pass; a bound, said by name, not a pace. -const WITHIN: Duration = Duration::from_secs(20); - /// Bytes a round trip asks for: more than one pipe write and one TCP segment, /// far less than a ring. const ROUND_TRIP: u64 = 256 * 1024; @@ -83,10 +77,9 @@ fn main() { ("a notify end at its size limit, owed a wake", &|| notify_end_is_a_full_file(port)), ]; for (case, run) in cases { - let started = Instant::now(); run(); round_trip(port, &format!("after {case}")); - println!("netd_refused_pipes: {case}, and a round trip after it, in {} ms", started.elapsed().as_millis()); + println!("netd_refused_pipes: {case}, and a round trip after it"); } println!("netd_refused_pipes: ok"); } @@ -99,7 +92,7 @@ fn connect_with(port: u16, to_client: RawHandle, from_client: Pipe) { .request_with_handles( &[to_client, from_client.into_raw()], MsgType::TcpConnectPiped, - &TcpConnectPipedRequest { addr: HOST, port, _pad: 0, timeout_ms: 30_000 }, + &TcpConnectPipedRequest { addr: HOST, port, _pad: 0, timeout_ms: NO_DEADLINE }, ) .expect("netd takes the request") .response() @@ -154,7 +147,7 @@ fn receive_end_is_a_read_end(port: u16) { Ok(n) => assert_eq!(n, asked.len(), "telling the host what to send"), Err(e) => assert_eq!(e, SyscallError::Gone, "telling the host what to send"), } - keep_full_until_released(&watch, WITHIN, "a read end handed over as the receive pipe"); + keep_full_until_released(&watch, "a read end handed over as the receive pipe"); } fn send_end_is_a_write_end(port: u16) { @@ -165,7 +158,7 @@ fn send_end_is_a_write_end(port: u16) { // this program reads as EOF. let what = "a write end handed over as the send pipe"; let mut buf = [0u8; 64]; - let got = await_until(&rx, READABLE, WITHIN, what, || match rx.read_nonblock(&mut buf) { + let got = await_until(&rx, READABLE, || match rx.read_nonblock(&mut buf) { Err(SyscallError::WouldBlock) => None, other => Some(other), }); @@ -175,13 +168,13 @@ fn send_end_is_a_write_end(port: u16) { fn notify_end_is_a_read_end() { let (read_end, watch) = toyos::pipe_pair().expect("a pipe"); bind_with(0, read_end.into_raw()); - keep_full_until_released(&watch, WITHIN, "a read end handed over as the notify pipe"); + keep_full_until_released(&watch, "a read end handed over as the notify pipe"); } fn receive_end_dropped_while_held(port: u16, capacity: u64) { - let conn = toyos::net::tcp_connect(HOST, port, 30_000).expect("connect to the host server"); + let conn = toyos::net::tcp_connect(HOST, port, NO_DEADLINE).expect("connect to the host server"); ask(&conn.tx, Ask::Stream(capacity + PAST_THE_RING)); - await_ring_full(&conn.rx, capacity, WITHIN); + await_ring_full(&conn.rx, capacity); drop(conn.rx); println!("netd_refused_pipes: dropped a full receive end with the host still sending"); } @@ -191,7 +184,7 @@ fn receive_end_is_a_full_file(port: u16) { connect_with(port, file_at_its_limit(), from_client); ask(&tx, Ask::Stream(64)); // netd ends the connection by closing the send pipe's read end. - keep_full_until_released(&tx, WITHIN, "a file at its size limit handed over as the receive pipe"); + keep_full_until_released(&tx, "a file at its size limit handed over as the receive pipe"); std::fs::remove_file(FILE_PATH).expect("remove the file"); } @@ -200,10 +193,10 @@ fn notify_end_is_a_full_file(port: u16) { // The host dials the listener and writes into the connection until it is // refused, which a listener netd has closed does at the host's next // segment; it then ends this connection. - let dial = toyos::net::tcp_connect(HOST, port, 30_000).expect("connect to the host server"); + let dial = toyos::net::tcp_connect(HOST, port, NO_DEADLINE).expect("connect to the host server"); ask(&dial.tx, Ask::Dial); let what = "a file at its size limit handed over as the notify pipe, and the host dialling in"; - assert_eq!(read_pattern(&dial.rx, WITHIN, what), 0, "{what}: the host sent bytes, not its FIN"); + assert_eq!(read_pattern(&dial.rx, what), 0, "{what}: the host sent bytes, not its FIN"); assert_eq!( toyos::net::tcp_accept(listener).err(), Some(NetError::NotConnected), @@ -213,10 +206,10 @@ fn notify_end_is_a_full_file(port: u16) { } fn round_trip(port: u16, what: &str) { - let conn = toyos::net::tcp_connect(HOST, port, 30_000) + let conn = toyos::net::tcp_connect(HOST, port, NO_DEADLINE) .unwrap_or_else(|e| panic!("{what}: netd did not connect: {e:?}")); ask(&conn.tx, Ask::Stream(ROUND_TRIP)); - let got = read_pattern(&conn.rx, WITHIN, what); + let got = read_pattern(&conn.rx, what); assert_eq!(got, ROUND_TRIP, "{what}: the stream ended after {got} of {ROUND_TRIP} bytes"); println!("netd_refused_pipes: round trip {what}: {got} bytes"); } diff --git a/tests/toyos-rust-tests/src/bin/netd_slow_reader.rs b/tests/toyos-rust-tests/src/bin/netd_slow_reader.rs index 9f8ef8332a4..03a56121688 100644 --- a/tests/toyos-rust-tests/src/bin/netd_slow_reader.rs +++ b/tests/toyos-rust-tests/src/bin/netd_slow_reader.rs @@ -20,20 +20,12 @@ #[path = "../netd_stream.rs"] mod netd_stream; -use std::time::Duration; - -use netd_stream::{ask, await_ring_full, read_pattern, ring_capacity, Ask, HOST}; +use netd_stream::{ask, await_ring_full, read_pattern, ring_capacity, Ask, HOST, NO_DEADLINE}; /// Bytes the host sends past the ring's capacity. Anything over the socket's /// own buffer makes a pipe that drops bytes when full drop some. const PAST_THE_RING: u64 = 1024 * 1024; -/// Liveness guards, said by name rather than left to the runner's ceiling: the -/// host has more than a capacity to send the moment this connects, so a ring -/// that never fills or a stream that stops is a netd that stopped moving bytes. -const FILL_BOUND: Duration = Duration::from_secs(60); -const READ_BOUND: Duration = Duration::from_secs(30); - fn main() { let port: u16 = std::env::args() .nth(1) @@ -43,15 +35,15 @@ fn main() { let total = capacity + PAST_THE_RING; println!("netd_slow_reader: ring capacity {capacity}, expecting {total} bytes"); - let conn = toyos::net::tcp_connect(HOST, port, 30_000).expect("connect to the host server"); + let conn = toyos::net::tcp_connect(HOST, port, NO_DEADLINE).expect("connect to the host server"); // The host learns how much to send from here, before it sends anything, // so the two ends cannot disagree about the length being judged. ask(&conn.tx, Ask::Stream(total)); - await_ring_full(&conn.rx, capacity, FILL_BOUND); + await_ring_full(&conn.rx, capacity); println!("netd_slow_reader: the ring is full at {capacity} bytes unread; reading"); let what = format!("netd_slow_reader (ring capacity {capacity})"); - let at = read_pattern(&conn.rx, READ_BOUND, &what); + let at = read_pattern(&conn.rx, &what); assert_eq!(at, total, "the stream ended after {at} of {total} bytes, every one of them right"); println!("netd_slow_reader: ok bytes={at}"); } diff --git a/tests/toyos-rust-tests/src/bin/netd_stalled_peer.rs b/tests/toyos-rust-tests/src/bin/netd_stalled_peer.rs deleted file mode 100644 index 5f0e5b180c5..00000000000 --- a/tests/toyos-rust-tests/src/bin/netd_stalled_peer.rs +++ /dev/null @@ -1,85 +0,0 @@ -//! A client that out-writes a peer which has stopped reading costs netd no -//! CPU. -//! -//! The host reads this connection's ask and nothing after it. This program -//! writes until the whole path — its send pipe, netd's socket buffer, slirp and -//! the host's socket — has stopped taking bytes, and then measures how busy the -//! machine is while nothing can move. A netd that watches the send pipe -//! whether or not its socket has room is woken by the pipe's bytes on every -//! pass, which is one whole CPU for as long as the peer stays stalled. -//! -//! argv[1] is the port of the harness's host server on `HOST`. -//! `netd_stalled_peer: ok` is the only success line. - -#[path = "../netd_stream.rs"] -mod netd_stream; - -use std::time::{Duration, Instant}; - -use netd_stream::{ask, Ask, HOST}; -use toyos::poller::{Poller, WRITABLE}; -use toyos_abi::syscall::{self, SyscallError, SysinfoHeader}; - -/// How long the send pipe must stay full, with the peer reading nothing, for -/// the path to count as stalled. Policy: every hop drains in far less. -const SETTLE: Duration = Duration::from_secs(2); - -/// The window the machine's busy time is measured over. -const WINDOW: Duration = Duration::from_secs(2); - -/// Liveness guard on reaching the stall at all. -const STALL_BOUND: Duration = Duration::from_secs(60); - -fn sysinfo() -> SysinfoHeader { - let mut buf = [0u8; toyos::system::SYSINFO_HEADER_SIZE]; - let n = toyos::system::sysinfo(&mut buf); - assert!(n >= toyos::system::SYSINFO_HEADER_SIZE, "sysinfo returned {n} bytes"); - SysinfoHeader::decode(&buf) -} - -fn main() { - let port: u16 = std::env::args() - .nth(1) - .and_then(|p| p.parse().ok()) - .expect("usage: netd_stalled_peer "); - let conn = toyos::net::tcp_connect(HOST, port, 30_000).expect("connect to the host server"); - ask(&conn.tx, Ask::Held(0)); - - let chunk = [0u8; 65536]; - let poller = Poller::new(1); - let started = Instant::now(); - let mut written = 0u64; - loop { - match conn.tx.write_nonblock(&chunk) { - Ok(n) => { - written += n as u64; - continue; - } - Err(SyscallError::WouldBlock) => {} - Err(e) => panic!("writing to the stalled peer after {written} bytes: {e:?}"), - } - assert!(started.elapsed() < STALL_BOUND, "the send path still took bytes after {STALL_BOUND:?}"); - // The pipe is full: room within `SETTLE` is the path still moving. - poller.watch(&conn.tx, WRITABLE, 0); - let mut roomed = false; - poller.wait(1, SETTLE.as_nanos() as u64, |_| roomed = true); - if !roomed { - break; - } - } - println!("netd_stalled_peer: the path stopped taking bytes after {written}"); - - let before = sysinfo(); - // An interval, not a wait: a rate is what is measured, and nothing is - // expected to happen in it. - syscall::nanosleep(WINDOW.as_nanos() as u64); - let after = sysinfo(); - let busy_ns = after.total_cpu_ns - before.total_cpu_ns; - let wall_ns = after.uptime_ns - before.uptime_ns; - let busy_cpus = busy_ns as f64 / wall_ns as f64; - println!("netd_stalled_peer: {busy_cpus:.3} CPUs busy over {wall_ns} ns, of {}", after.cpus); - // A netd polling a pipe it cannot drain is one whole CPU; half of one is - // far above an idle machine and far below that. - assert!(busy_cpus < 0.5, "the machine was {busy_cpus:.3} CPUs busy with every connection stalled"); - println!("netd_stalled_peer: ok, {busy_cpus:.3} CPUs busy while stalled"); -} diff --git a/tests/toyos-rust-tests/src/bin/netd_udp_any_address.rs b/tests/toyos-rust-tests/src/bin/netd_udp_any_address.rs index b38b16dd1af..a87fde0647e 100644 --- a/tests/toyos-rust-tests/src/bin/netd_udp_any_address.rs +++ b/tests/toyos-rust-tests/src/bin/netd_udp_any_address.rs @@ -14,16 +14,9 @@ mod netd_stream; use std::net::{Ipv4Addr, SocketAddr, UdpSocket}; -use std::sync::mpsc; -use std::time::Duration; use netd_stream::HOST; -/// How long netd may take to answer a receive whose datagram is on its way. -/// A bound, said by name; not a pace. std's `UdpSocket` does not honour a read -/// timeout, so the receive runs on a thread and this bounds the wait for it. -const WITHIN: Duration = Duration::from_secs(20); - fn main() { let echo: u16 = std::env::args() .nth(2) @@ -35,17 +28,11 @@ fn main() { let datagram: Vec = (0..200u8).collect(); assert_eq!(socket.send_to(&datagram, to).expect("send to the echo"), datagram.len()); - let (answered, answer) = mpsc::channel(); - let receiver = socket.try_clone().expect("a second handle for the receive"); - std::thread::spawn(move || { - let mut buf = [0u8; 512]; - answered.send(receiver.recv_from(&mut buf).map(|(n, from)| (buf[..n].to_vec(), from))) - }); - let (got, from) = answer - .recv_timeout(WITHIN) - .unwrap_or_else(|_| panic!("a socket bound to 0.0.0.0 received no reply within {WITHIN:?}")) - .expect("the receive"); + // No deadline: a reply that never comes is a hang the harness ceiling reds. + let mut buf = [0u8; 512]; + let (n, from) = socket.recv_from(&mut buf).expect("the receive"); + let got = &buf[..n]; assert_eq!(from, to, "the reply came from somewhere other than the echo"); - assert_eq!(got, datagram, "the echo's reply is not the datagram sent"); + assert_eq!(got, &datagram[..], "the echo's reply is not the datagram sent"); println!("netd_udp_any_address: ok"); } diff --git a/tests/toyos-rust-tests/src/bin/netd_udp_refused.rs b/tests/toyos-rust-tests/src/bin/netd_udp_refused.rs index c2e9384b938..e65750478fe 100644 --- a/tests/toyos-rust-tests/src/bin/netd_udp_refused.rs +++ b/tests/toyos-rust-tests/src/bin/netd_udp_refused.rs @@ -31,9 +31,6 @@ #[path = "../netd_stream.rs"] mod netd_stream; -use std::sync::mpsc; -use std::time::{Duration, Instant}; - use netd_stream::{fill, HOST}; use toyos::ipc::{FrameRx, RxStep}; use toyos::net::{ @@ -54,10 +51,6 @@ const ROOM: usize = 100; /// Bytes in each datagram: more than [`ROOM`], less than one Ethernet frame. const DATAGRAM: usize = 1000; -/// How long netd may take to answer a receive. A bound, said by name; not a -/// pace. -const WITHIN: Duration = Duration::from_secs(20); - fn main() { let echo: u16 = std::env::args() .nth(2) @@ -153,7 +146,7 @@ fn send(socket: UdpSocketId, tx: &Pipe, echo: u16, byte: u8) { } /// The sockets netd's stack holds that no table entry names, as its own -/// `inspect` answers, within [`WITHIN`]. Every program's sockets are in the +/// `inspect` answers, waited for with no deadline. Every program's sockets are in the /// table and the resolver's are left out, so only netd's own and one that /// outlived its entry move it. fn untabled() -> u64 { @@ -161,7 +154,6 @@ fn untabled() -> u64 { conn.signal(MSG_INSPECT).expect("netd takes an inspect request"); let poller = Poller::new(1); let mut rx: Box> = Box::new(FrameRx::new()); - let deadline = Instant::now() + WITHIN; loop { match rx.pump(&conn) { RxStep::Frame { msg_type: MSG_SNAPSHOT, payload_len } => { @@ -175,19 +167,13 @@ fn untabled() -> u64 { RxStep::Idle => {} other => panic!("netd answered inspect with {other:?}, not a snapshot"), } - let left = deadline.saturating_duration_since(Instant::now()); - assert!(!left.is_zero(), "netd did not answer inspect within {WITHIN:?}"); poller.watch(&conn, READABLE, 0); - poller.wait(1, left.as_nanos() as u64, |_| {}); + poller.wait(1, u64::MAX, |_| {}); } } -/// Ask netd for `socket`'s next datagram, and panic by name if no answer -/// comes within [`WITHIN`]. +/// Ask netd for `socket`'s next datagram, with no deadline: an answer that +/// never comes is a hang the harness ceiling reds. fn recv(socket: UdpSocketId) -> Result { - let (answered, answer) = mpsc::channel(); - std::thread::spawn(move || answered.send(udp_recv_from(socket, DATAGRAM as u32))); - answer - .recv_timeout(WITHIN) - .unwrap_or_else(|_| panic!("netd did not answer a receive on socket {} within {WITHIN:?}", socket.0)) + udp_recv_from(socket, DATAGRAM as u32) } diff --git a/tests/toyos-rust-tests/src/bin/nmi_window_spin.rs b/tests/toyos-rust-tests/src/bin/nmi_window_spin.rs index bd6c2f21cd2..7dd722641c3 100644 --- a/tests/toyos-rust-tests/src/bin/nmi_window_spin.rs +++ b/tests/toyos-rust-tests/src/bin/nmi_window_spin.rs @@ -35,19 +35,14 @@ fn main() { println!("nmi-window-spin: spinning on SYS_GETPID for {secs}s"); - let started = clock_nanos(); - let until = started + secs * 1_000_000_000; + let until = clock_nanos() + secs * 1_000_000_000; let mut done: u64 = 0; while clock_nanos() < until { chunk(); done += CHUNK; } - let elapsed = clock_nanos() - started; - println!( - "nmi-window-spin: {done} syscalls in {elapsed} ns ({} ns each)", - elapsed / done.max(1), - ); + println!("nmi-window-spin: {done} syscalls"); } /// [`CHUNK`] iterations of exactly four instructions. diff --git a/tests/toyos-rust-tests/src/bin/null_sink_client_exits.rs b/tests/toyos-rust-tests/src/bin/null_sink_client_exits.rs index c6b3c19c3bc..dd0d0f07a23 100644 --- a/tests/toyos-rust-tests/src/bin/null_sink_client_exits.rs +++ b/tests/toyos-rust-tests/src/bin/null_sink_client_exits.rs @@ -1,10 +1,9 @@ //! A client that plays through the null sink must finish and exit. //! //! `/system/bin/tone` and not this crate's own tone: the T14 hangs on the shipped -//! binary, which reaches soundd through `cpal`, while the raw-API tone that -//! `metal_sim_null_audio` runs drains perfectly on the same sink. Whatever the -//! defect is, only the path a user actually takes shows it — which is why this -//! spawns the program the shell spawns rather than linking the SDK directly. +//! binary, which reaches soundd through `cpal`. Whatever the defect is, only the +//! path a user actually takes shows it — which is why this spawns the program the +//! shell spawns rather than linking the SDK directly. //! //! Two of them in series, because one hung client is what the T14 log shows //! blocking the *next* connect: if the first exits and the second hangs, the diff --git a/tests/toyos-rust-tests/src/bin/panic_halts_first.rs b/tests/toyos-rust-tests/src/bin/panic_halts_first.rs index 64580a5bdf4..0dd2fd7d685 100644 --- a/tests/toyos-rust-tests/src/bin/panic_halts_first.rs +++ b/tests/toyos-rust-tests/src/bin/panic_halts_first.rs @@ -1,19 +1,17 @@ //! Threads that make kernel records as fast as they can while this one has //! the kernel go fatal (`SYS_DEBUG` action 3). A sibling still running after -//! the fatal path began is a record stamped after the fatal one; -//! `panic_halts_the_others_first` reads the console for one. +//! the fatal path stopped the other CPUs is a record past the fatal path's own +//! line after the stop; `panic_halts_the_others_first` reads the console for +//! one. use std::sync::atomic::{AtomicUsize, Ordering}; use std::sync::Arc; -use std::time::{Duration, Instant}; /// A retired syscall's number: each call is refused and is one kernel record /// naming it. const RETIRED: u64 = 26; /// One per other CPU of the boot that runs this. const SIBLINGS: usize = 3; -/// A liveness guard on the siblings' first records. -const STARTED_WITHIN: Duration = Duration::from_secs(10); fn retired() { let ret: u64; @@ -37,9 +35,9 @@ fn main() { } }); } - let until = Instant::now() + STARTED_WITHIN; + // No deadline: siblings that never make a record are a hang the harness + // ceiling reds. while started.load(Ordering::Acquire) < SIBLINGS { - assert!(Instant::now() < until, "the siblings made no record in {STARTED_WITHIN:?}"); std::hint::spin_loop(); } let rc = toyos_abi::syscall::debug(toyos_abi::syscall::debug_action::FATAL_HALT); diff --git a/tests/toyos-rust-tests/src/bin/poll_wake_pipe.rs b/tests/toyos-rust-tests/src/bin/poll_wake_pipe.rs index 01408546985..7d7ab0897aa 100644 --- a/tests/toyos-rust-tests/src/bin/poll_wake_pipe.rs +++ b/tests/toyos-rust-tests/src/bin/poll_wake_pipe.rs @@ -4,30 +4,22 @@ //! and the 7a cutover once deleted the second for two sources undetected. A //! watcher arms `POLL_ADD` READABLE on a pipe read end and blocks in `wait`, a //! writer writes each round, and every readable edge must wake the ring: a -//! dropped ring wake reds as a short count, which is the whole verdict. +//! dropped ring wake parks the watcher, and the harness's ceiling turns that +//! into a red. //! `blocking_read_stress` is the same canary for the other half. use std::sync::atomic::{AtomicU32, Ordering}; use std::thread; -use std::time::{Duration, Instant}; use toyos::pipe_pair; use toyos::poller::{Poller, READABLE}; const ROUNDS: u32 = 300; -/// The pacing spin's escape, so a lost wake reds as a short count, not a hang. -/// Past it the writer stops pacing, so the edges are no longer distinct and the verdict is only the count. -const PACE_ESCAPE: Duration = Duration::from_secs(3); - -/// Per-round patience, far above a live wake's latency, so only an absent completion trips it. -const WAIT_NANOS: u64 = 200_000_000; - fn main() { let (reader, writer) = pipe_pair().expect("a pipe"); let woken = AtomicU32::new(0); - let began = Instant::now(); thread::scope(|s| { s.spawn(|| { let poller = Poller::new(1); @@ -35,10 +27,8 @@ fn main() { for _ in 0..ROUNDS { poller.watch(&reader, READABLE, 0); let mut got = false; - poller.wait(1, WAIT_NANOS, |_| got = true); - if !got { - return; // the completion never arrived — the ring half was lost - } + poller.wait(1, u64::MAX, |_| got = true); + assert!(got, "a wait with no deadline returned with no completion"); woken.fetch_add(1, Ordering::Relaxed); let _ = reader.read(&mut buf); // drain so the next round arms empty } @@ -47,9 +37,7 @@ fn main() { // One byte per round, paced behind the watcher so every write is a distinct edge. s.spawn(|| { for round in 0..ROUNDS { - while woken.load(Ordering::Relaxed) < round - && began.elapsed() < PACE_ESCAPE - { + while woken.load(Ordering::Relaxed) < round { std::hint::spin_loop(); } if writer.write(&[0x5A]).is_err() { @@ -60,11 +48,9 @@ fn main() { }); let woken = woken.load(Ordering::Relaxed); - let elapsed = began.elapsed(); assert_eq!( woken, ROUNDS, - "the poll completed {woken} of {ROUNDS} readable edges in {elapsed:?} — a ring \ - watcher's wake was lost", + "the poll completed {woken} of {ROUNDS} readable edges — a ring watcher's wake was lost", ); - println!("poll_wake_pipe: {ROUNDS} readable edges each woke the armed ring in {elapsed:?}"); + println!("poll_wake_pipe: {ROUNDS} readable edges each woke the armed ring"); } diff --git a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs index aa98fa60e9a..f47f83c593c 100644 --- a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs +++ b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs @@ -29,7 +29,6 @@ use std::os::toyos::process::{ChildExt, CommandExt}; use std::process::{Child, ChildStdin, Command, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::OnceLock; -use std::time::{Duration, Instant}; use toyos::endow::{Endowments, SYSCAP_LABEL}; use toyos::AsHandle; @@ -38,6 +37,9 @@ use toyos::syscap::SysCap; use toyos_abi::syscall::{self, SyscallError}; use toyos_abi::RawHandle; +#[path = "../roster.rs"] +mod roster; + const SELF_PATH: &str = "/system/bin/test_rs_process_lifecycle"; /// The label the `waiter` role finds its subject under. A local name in one @@ -139,8 +141,10 @@ fn a_wait_before_the_exit_is_woken_by_it() { "a process that cannot have exited reported an exit code", ); + // Released once the kernel says the wait is parked, so the wake is the + // exit's and not a code already there. let releaser = std::thread::spawn(move || { - std::thread::sleep(std::time::Duration::from_millis(200)); + roster::await_true(main_thread_is_parked); drop(release); }); assert_eq!(child.wait().expect("wait").code(), Some(7), "the woken wait"); @@ -174,13 +178,13 @@ fn an_unrelated_wake_does_not_end_the_wait() { let (mut child, release) = start(5); let poker = std::thread::spawn(|| { - await_true("the main thread never parked", main_thread_is_parked); + roster::await_true(main_thread_is_parked); POKED.store(true, Ordering::Release); // Returning is the poke: nothing else in this closure matters, because // `thread_exit` is what wakes the main thread. }); let releaser = std::thread::spawn(move || { - await_true("the poking thread never exited", a_thread_of_mine_has_exited); + roster::await_true(a_thread_of_mine_has_exited); RELEASED.store(true, Ordering::Release); drop(release); }); @@ -196,32 +200,12 @@ fn an_unrelated_wake_does_not_end_the_wait() { println!(" a wake meant for something else does not end a wait"); } -/// Poll until `cond` holds. The bound is a hang guard and not a timing -/// assumption: both callers wait for a state the kernel has already decided and -/// reaches in microseconds, and the `sysinfo` call inside `cond` is the loop's -/// preemption point (`thread::yield_now` is a spin hint on this platform). -fn await_true(what: &str, cond: fn() -> bool) { - let give_up = Instant::now() + Duration::from_secs(5); - while !cond() { - assert!(Instant::now() < give_up, "{what}"); - } -} - -/// `sched::payload::SCHED_BLOCKED` — the state column `ps` prints. -const BLOCKED: u8 = 2; - -/// `SCHED_UNKNOWN`, which `sys_sysinfo` also answers for a thread whose entry -/// is a zombie. A live thread's scheduler record is installed under the same -/// table lock that inserts its entry, so a thread of ours reading this has -/// exited and nothing else. -const ZOMBIE: u8 = 3; - fn main_thread_is_parked() -> bool { - my_threads().iter().any(|&(is_thread, state)| !is_thread && state == BLOCKED) + roster::my_threads(cap()).iter().any(|&(is_thread, state)| !is_thread && state == roster::BLOCKED) } fn a_thread_of_mine_has_exited() -> bool { - my_threads().iter().any(|&(is_thread, state)| is_thread && state == ZOMBIE) + roster::my_threads(cap()).iter().any(|&(is_thread, state)| is_thread && state == roster::ZOMBIE) } /// The estate's system capability, taken once. @@ -238,27 +222,6 @@ fn cap() -> &'static SysCap { }) } -/// This process's threads as the kernel publishes them: `(is a child thread, -/// scheduler state)`. -/// -/// Even one's own threads arrive in the machine-wide roster, which is -/// `Rights::ROSTER` on a `SysCap` — there is no narrower question in the ABI, -/// and `tests/testcases` names `roster` on the test-runner row for this. -fn my_threads() -> Vec<(bool, u8)> { - const HEADER: usize = toyos::system::SYSINFO_HEADER_SIZE; - const ENTRY: usize = toyos::system::SYSINFO_ENTRY_SIZE; - let mut buf = vec![0u8; HEADER + ENTRY * 256]; - let n = cap().roster(&mut buf); - assert!((HEADER..=buf.len()).contains(&n), "sysinfo answered {n}"); - let me = syscall::getpid().raw(); - (HEADER..) - .step_by(ENTRY) - .take_while(|pos| pos + ENTRY <= n) - .filter(|&pos| u32::from_le_bytes(buf[pos..pos + 4].try_into().unwrap()) == me) - .map(|pos| (buf[pos + 9] != 0, buf[pos + 8])) - .collect() -} - /// A second handle is a second name for one object, and the object is where the /// code is. fn two_handles_answer_the_same() { diff --git a/tests/toyos-rust-tests/src/bin/process_stats.rs b/tests/toyos-rust-tests/src/bin/process_stats.rs index 5a84025450e..d83c47832a9 100644 --- a/tests/toyos-rust-tests/src/bin/process_stats.rs +++ b/tests/toyos-rust-tests/src/bin/process_stats.rs @@ -18,10 +18,15 @@ use std::io::{BufRead, BufReader, Read, Write}; use std::os::toyos::process::ChildExt; use std::process::{Command, Stdio}; +use toyos::endow::{Endowments, SYSCAP_LABEL}; use toyos::process::Process; +use toyos::syscap::SysCap; use toyos_abi::handle::Rights; use toyos_abi::syscall::{self, ProcessStats, SyscallError}; +#[path = "../roster.rs"] +mod roster; + const SELF_PATH: &str = "/system/bin/test_rs_process_stats"; fn main() { @@ -35,16 +40,12 @@ fn main() { blocked_time_names_what_it_waited_on(); repeatable(); refused_without_read(); - refused_calls_are_timed(); + refused_calls_are_counted(); println!("all process_stats tests passed"); } const REFUSED_CALLS: u64 = 500_000; -/// A refused call is timed, so each one the kernel counts adds at least this much -/// to `syscall_total_ns`; one returned past the clock adds nothing. -const MIN_NS_PER_REFUSED_CALL: u64 = 5; - /// Says it is ready, waits for the parent's byte, then issues nothing but refusals. fn refused_child() { const NAME: [u8; 64] = [0u8; 64]; // past THREAD_NAME_LEN (28): refused before any pointer read @@ -58,9 +59,9 @@ fn refused_child() { } } -/// A refused syscall is counted *and* timed, read across the child's refusal loop -/// alone so its startup's timed calls cannot stand in for the refusals'. -fn refused_calls_are_timed() { +/// A refused syscall is counted, read across the child's refusal loop alone so +/// its startup's calls cannot stand in for the refusals'. +fn refused_calls_are_counted() { let mut child = Command::new(SELF_PATH) .arg("refused") .stdin(Stdio::piped()) @@ -83,21 +84,11 @@ fn refused_calls_are_timed() { let after = stats_of(&child).expect("the exited child answers"); let calls = after.syscall_total.saturating_sub(before.syscall_total); - let timed_ns = after.syscall_total_ns.saturating_sub(before.syscall_total_ns); assert!( calls >= REFUSED_CALLS, "the child made {REFUSED_CALLS} refused calls but only {calls} were counted", ); - assert!( - timed_ns >= calls * MIN_NS_PER_REFUSED_CALL, - "{calls} calls counted across the refusal loop added {timed_ns} ns to syscall_total_ns, \ - under {MIN_NS_PER_REFUSED_CALL} ns each — refused calls counted but not timed", - ); - println!( - " refused calls timed: ok (calls={calls} timed_ns={timed_ns} whole child: total={} \ - total_ns={} cpu_ns={})", - after.syscall_total, after.syscall_total_ns, after.cpu_ns, - ); + println!(" refused calls counted: ok (calls={calls} whole child: total={})", after.syscall_total); } /// Says it is running, then blocks until it is killed. The marker is flushed, so @@ -218,10 +209,12 @@ fn blocked_time_names_what_it_waited_on() { out.read_line(&mut line).expect("the held child's marker"); assert_eq!(line.trim(), "running", "the held child said {line:?}"); - // Long enough that the park is measurable at the accounting's resolution, - // and short enough that it is a margin rather than a bound: what is - // asserted is which counter moved, never how far. - std::thread::sleep(std::time::Duration::from_millis(200)); + // Parked, by the kernel's own roster: the counter has a park to charge. + let cap: SysCap = Endowments::get() + .take(SYSCAP_LABEL) + .expect("test-runner endows every binary it spawns a system capability"); + let pid = stats_of(&child).expect("the held child answers").pid; + roster::await_true(|| roster::threads_of(&cap, pid).iter().any(|&(_, state)| state == roster::BLOCKED)); // Ending the park is what charges it. The child's `read` returns and it // exits; its object keeps answering, which is this file's first arm. child diff --git a/tests/toyos-rust-tests/src/bin/quiesce_fsync.rs b/tests/toyos-rust-tests/src/bin/quiesce_fsync.rs index 956330352f2..7907804ab47 100644 --- a/tests/toyos-rust-tests/src/bin/quiesce_fsync.rs +++ b/tests/toyos-rust-tests/src/bin/quiesce_fsync.rs @@ -12,7 +12,6 @@ use std::fs::File; use std::io::Write; -use std::time::{Duration, Instant}; use toyos::endow::{Endowments, SYSCAP_LABEL}; use toyos::log::{LogTail, Record, MAX_LOG_SHARDS}; @@ -37,9 +36,6 @@ const CHUNKS: usize = 8; const BATCH: usize = 4 * MAX_LOG_SHARDS as usize; const LOG_TOKEN: u64 = 1; -/// How long the fsync gets to be refused the second time. -const PARKS_WITHIN: Duration = Duration::from_secs(5); - fn main() { let Some(cap) = Endowments::get().take::(SYSCAP_LABEL) else { eprintln!("quiesce_fsync: this program was endowed no system capability"); @@ -66,7 +62,6 @@ fn main() { let mut tail = LogTail::new(); let mut buf = [Record::EMPTY; BATCH]; let poller = Poller::new(1); - let give_up = Instant::now() + PARKS_WITHIN; poller.watch(&cap, READABLE, LOG_TOKEN); poller.wait(0, 0, |_| {}); loop { @@ -80,11 +75,9 @@ fn main() { if !batch.is_empty() { continue; } - let Some(left) = give_up.checked_duration_since(Instant::now()) else { - eprintln!("quiesce_fsync: the fsync of {STAGED} was not refused twice in {PARKS_WITHIN:?}"); - std::process::exit(1); - }; - poller.wait(1, left.as_nanos() as u64, |_| {}); + // No deadline: an fsync never refused twice is a hang the harness + // ceiling reds. + poller.wait(1, u64::MAX, |_| {}); poller.watch(&cap, READABLE, LOG_TOKEN); poller.wait(0, 0, |_| {}); } diff --git a/tests/toyos-rust-tests/src/bin/quiesce_twice.rs b/tests/toyos-rust-tests/src/bin/quiesce_twice.rs index f178eacdfc4..2303b11af97 100644 --- a/tests/toyos-rust-tests/src/bin/quiesce_twice.rs +++ b/tests/toyos-rust-tests/src/bin/quiesce_twice.rs @@ -14,7 +14,7 @@ use std::fs::File; use std::io::Write; -use std::time::{Duration, Instant}; +use std::time::Duration; use toyos::endow::{Endowments, SYSCAP_LABEL}; use toyos::log::{LogTail, Record, MAX_LOG_SHARDS}; @@ -39,14 +39,6 @@ const BATCH: usize = 4 * MAX_LOG_SHARDS as usize; /// The poll's one token. const LOG_TOKEN: u64 = 1; -/// How long the first call gets to reach its wait before this boot says it -/// never did: inside the kernel's own bound on that wait, so this line is the -/// one that says why. -const WAITS_WITHIN: Duration = Duration::from_secs(5); - -/// How long the machine gets to stop this process once the held thread runs. -const STOPPED_WITHIN: Duration = Duration::from_secs(20); - fn main() { let Some(cap) = Endowments::get().take::(SYSCAP_LABEL) else { eprintln!("quiesce_twice: this program was endowed no system capability"); @@ -78,7 +70,6 @@ fn main() { let mut tail = LogTail::new(); let mut buf = [Record::EMPTY; BATCH]; let poller = Poller::new(1); - let give_up = Instant::now() + WAITS_WITHIN; poller.watch(&cap, READABLE, LOG_TOKEN); poller.wait(0, 0, |_| {}); loop { @@ -92,11 +83,9 @@ fn main() { if !batch.is_empty() { continue; } - let Some(left) = give_up.checked_duration_since(Instant::now()) else { - eprintln!("quiesce_twice: the first call never waited for {LAST_THREAD}"); - std::process::exit(1); - }; - poller.wait(1, left.as_nanos() as u64, |_| {}); + // No deadline: a first call that never waits is a hang the harness + // ceiling reds. + poller.wait(1, u64::MAX, |_| {}); poller.watch(&cap, READABLE, LOG_TOKEN); poller.wait(0, 0, |_| {}); } @@ -109,9 +98,18 @@ fn main() { } std::thread::Builder::new() .name(LAST_THREAD.into()) - .spawn(|| std::thread::sleep(STOPPED_WITHIN)) + .spawn(asleep_until_stopped) .expect("spawn the thread the stop waits for"); - std::thread::sleep(STOPPED_WITHIN); - eprintln!("quiesce_twice: the machine did not stop this process in {STOPPED_WITHIN:?}"); - std::process::exit(1); + // No deadline: a machine that never stops this process is a hang the + // harness ceiling reds. + asleep_until_stopped() +} + +/// Asleep until the machine stops, which is the only thing that ends it. A +/// sleep and not a park because `nanosleep` is the syscall `quiesce-last-park` +/// holds the named thread in; its span is never reached. +fn asleep_until_stopped() -> ! { + loop { + std::thread::sleep(Duration::MAX); + } } diff --git a/tests/toyos-rust-tests/src/bin/quiesce_writers.rs b/tests/toyos-rust-tests/src/bin/quiesce_writers.rs index 7ebc41c17f2..67cffc0cdd6 100644 --- a/tests/toyos-rust-tests/src/bin/quiesce_writers.rs +++ b/tests/toyos-rust-tests/src/bin/quiesce_writers.rs @@ -14,7 +14,6 @@ use std::fs::File; use std::io::Write; use std::sync::mpsc; -use std::time::{Duration, Instant}; use toyos::power::Stop; @@ -26,13 +25,6 @@ const WRITERS: usize = 6; /// rather than a page-cache touch. const CHUNK: usize = 8192; -/// How long the writers get to reach their loop before this boot gives up on -/// being a machine with anything to stop. -/// -/// Inside the host's own wait for this guest to stop, so the line below reaches -/// the console it is read from rather than a budget ending the boot first. -const SPIN_UP: Duration = Duration::from_secs(5); - /// What a writer says every pass. const WRITING: &str = "quiesce-writer:"; @@ -82,12 +74,15 @@ fn main() { }) .expect("spawn a writer"); } + // The writers hold the only senders: a writer that failed is a closed + // channel here once it is the last. + drop(in_the_loop); - let give_up = Instant::now() + SPIN_UP; + // No deadline: a writer that never reaches its loop is a hang the harness + // ceiling reds. for reached in 0..WRITERS { - let left = give_up.saturating_duration_since(Instant::now()); - if first_passes.recv_timeout(left).is_err() { - eprintln!("quiesce_writers: {reached} of {WRITERS} writers reached their loop in {SPIN_UP:?}"); + if first_passes.recv().is_err() { + eprintln!("quiesce_writers: {reached} of {WRITERS} writers reached their loop"); std::process::exit(1); } } diff --git a/tests/toyos-rust-tests/src/bin/sched_stress.rs b/tests/toyos-rust-tests/src/bin/sched_stress.rs index 45f7009fdd3..70f7bba32fa 100644 --- a/tests/toyos-rust-tests/src/bin/sched_stress.rs +++ b/tests/toyos-rust-tests/src/bin/sched_stress.rs @@ -1,4 +1,3 @@ -use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::Arc; use std::thread; use std::time::Duration; @@ -10,10 +9,7 @@ use toyos::{namespace, port}; fn main() { match std::env::args().nth(1).as_deref() { - Some("burn") => { - let ms: u64 = std::env::args().nth(2).unwrap().parse().unwrap(); - child_burn(ms); - } + Some("burn") => child_burn(), Some("sched-info") => child_sched_info(), _ => run_tests(), } @@ -26,15 +22,11 @@ fn run_tests() { println!("all sched_stress tests passed"); } -fn child_burn(ms: u64) { - let mut count = 0u64; - let start = std::time::Instant::now(); - let dur = Duration::from_millis(ms); - while start.elapsed() < dur { - count += 1; - if count % 1000 == 0 { thread::yield_now(); } +/// Burns until it is killed. +fn child_burn() -> ! { + loop { + std::hint::spin_loop(); } - println!("{count}"); } fn child_sched_info() { @@ -55,53 +47,34 @@ fn child_sched_info() { /// readiness signal keyed on "some acceptor" instead of on the object wakes /// every waiter in the system, which froze the compositor. fn test_acceptor_isolation_io_uring() { - let a_ready = Arc::new(AtomicBool::new(false)); - let b_ready = Arc::new(AtomicBool::new(false)); - let a_ready2 = Arc::clone(&a_ready); - let b_ready2 = Arc::clone(&b_ready); - let (acc_a, con_a) = port::create().expect("port a"); let (acc_b, _con_b) = port::create().expect("port b"); - // Thread A: watch its acceptor, report whether the poll completed - let a = thread::spawn(move || -> bool { + // Port a's poll, waited on with no deadline: its connection is the event. + let a = thread::spawn(move || { let handle = acc_a.into_raw(); - a_ready2.store(true, Ordering::Release); let poller = Poller::new(1); poller.watch_raw(handle, READABLE, 0); - let mut ready = false; - poller.wait(1, 500_000_000, |_| ready = true); + poller.wait(1, u64::MAX, |_| ()); syscall::close(handle); - ready }); - // Thread B: a different port, watched the same way - let b = thread::spawn(move || -> bool { - let handle = acc_b.into_raw(); - b_ready2.store(true, Ordering::Release); - let poller = Poller::new(1); - poller.watch_raw(handle, READABLE, 0); - let mut ready = false; - poller.wait(1, 200_000_000, |_| ready = true); - syscall::close(handle); - ready - }); - - // Wait for both to be watching - while !a_ready.load(Ordering::Acquire) || !b_ready.load(Ordering::Acquire) { - thread::yield_now(); - } - thread::sleep(Duration::from_millis(20)); + // Port b's, registered before the connection and asked only after a's poll + // completed: the order, not a window, is what makes its silence a verdict. + let b = acc_b.into_raw(); + let b_poller = Poller::new(1); + b_poller.watch_raw(b, READABLE, 0); // Open a connection through port A's connector only. let ns = namespace::build().add("a", &con_a).finish().expect("a namespace naming port a"); let client = ns.open("a").expect("open a"); drop(client); - let a_poll_ready = a.join().unwrap(); - let b_poll_ready = b.join().unwrap(); + a.join().expect("port a's watcher panicked"); + let mut b_poll_ready = false; + b_poller.wait(0, 0, |_| b_poll_ready = true); + syscall::close(b); - assert!(a_poll_ready, "port a's poll should have completed (connection pending)"); assert!(!b_poll_ready, "port b's poll completed spuriously — acceptor isolation broken!"); println!(" acceptor isolation (io_uring): ok"); @@ -113,38 +86,38 @@ fn test_acceptor_isolation_io_uring() { fn test_min_vruntime_invariant() { let me = "/system/bin/test_rs_sched_stress"; - // Spawn 3 CPU burners to drive min_vruntime forward. + // Spawn 3 CPU burners to drive min_vruntime forward, until they are killed. let mut burners = Vec::new(); for _ in 0..3 { - burners.push(Command::new(me).arg("burn").arg("1000") - .stdout(Stdio::piped()).spawn().expect("spawn burner")); + burners.push(Command::new(me).arg("burn").spawn().expect("spawn burner")); } - // Let them run for 500ms to accumulate vruntime. - thread::sleep(Duration::from_millis(500)); - - // Spawn a process that sleeps briefly (forces Runnable→NonRunnable→ - // Runnable) and reports lag immediately after wake. - let info_child = Command::new(me).arg("sched-info") - .stdout(Stdio::piped()).spawn().expect("spawn sched-info"); - let info_out = info_child.wait_with_output().expect("wait sched-info"); - let output = String::from_utf8_lossy(&info_out.stdout); - let parts: Vec<&str> = output.trim().split_whitespace().collect(); - assert_eq!(parts.len(), 3, "sched-info output should be 'vruntime min_vruntime lag', got: {output:?}"); - let vruntime: u64 = parts[0].parse().expect("parse vruntime"); - let min_vruntime: u64 = parts[1].parse().expect("parse min_vruntime"); - let lag: i64 = parts[2].parse().expect("parse lag"); - - // Clean up burners. - for child in burners { - let _ = child.wait_with_output(); + // A process that sleeps briefly (forces Runnable→NonRunnable→Runnable) and + // reports lag immediately after wake, asked again until the burners have + // moved min_vruntime off zero: a min_vruntime never updated is a hang the + // harness ceiling reds. + let (vruntime, min_vruntime, lag) = loop { + let info_out = Command::new(me).arg("sched-info") + .stdout(Stdio::piped()).spawn().expect("spawn sched-info") + .wait_with_output().expect("wait sched-info"); + let output = String::from_utf8_lossy(&info_out.stdout); + let parts: Vec<&str> = output.trim().split_whitespace().collect(); + assert_eq!(parts.len(), 3, "sched-info output should be 'vruntime min_vruntime lag', got: {output:?}"); + let vruntime: u64 = parts[0].parse().expect("parse vruntime"); + let min_vruntime: u64 = parts[1].parse().expect("parse min_vruntime"); + let lag: i64 = parts[2].parse().expect("parse lag"); + if min_vruntime > 0 { + break (vruntime, min_vruntime, lag); + } + }; + + for mut child in burners { + child.kill().expect("kill a burner"); + child.wait().expect("reap a burner"); } println!(" sched_info: vruntime={vruntime} min_vruntime={min_vruntime} lag={lag}"); - assert!(min_vruntime > 0, - "min_vruntime is still 0 after 500ms of CPU-bound work — not being updated!"); - let max_lag_ns: i64 = 50_000_000; assert!(lag.abs() <= max_lag_ns, "post-wake lag ({lag}) exceeds ±MAX_VRUNTIME_LAG_NS ({max_lag_ns}) — \ diff --git a/tests/toyos-rust-tests/src/bin/shm_release_reclaims.rs b/tests/toyos-rust-tests/src/bin/shm_release_reclaims.rs index f58bdb36c62..ac01ca24151 100644 --- a/tests/toyos-rust-tests/src/bin/shm_release_reclaims.rs +++ b/tests/toyos-rust-tests/src/bin/shm_release_reclaims.rs @@ -21,6 +21,9 @@ use toyos::shm::SharedMemory; use toyos::{namespace, port, AsHandle}; use toyos_abi::syscall::{self, SVC_LABEL}; +#[path = "../census_wait.rs"] +mod census_wait; + const SELF_PATH: &str = "/system/bin/test_rs_shm_release_reclaims"; const PAYLOAD: &[u8] = b"sent-before-the-maker-let-go"; /// Sixteen rather than one because the arrival check has to be able to fail: a @@ -30,39 +33,6 @@ const ROUNDS: usize = 16; const REGION: usize = 4096; const SERVICE: &str = "region"; -/// How many 10 ms samples [`settled_census`] takes before it stops asking. -/// Reaching it is not a failure — the last reading is handed back and the -/// caller's assertion is still the whole verdict. -const SETTLE_SAMPLES: usize = 100; - -/// The live-object census once the machine has stopped giving objects back. -/// -/// **A region's pages are not released by the `close` that dropped its last -/// handle.** The drop queues the region on the object layer's zero-handle -/// queue and the release happens when some CPU drains that queue; -/// `object::drain_zero_handles` clears its pending flag before it runs the -/// hooks, so the CPU that queued them can find the queue empty while another -/// CPU is still working through the batch, and the release then escapes the -/// syscall that caused it. `handle_lifetime` carries the measurement and -/// `issues/kernel/deferred-release-outlives-its-syscall.md` the kernel -/// half. -/// -/// **A liveness bound and not a margin**: a kernel that releases nothing holds -/// a stable, elevated census, is quiescent on the first pair of samples, and -/// reds immediately. -fn settled_census() -> Census { - let mut last = Census::now(); - for _ in 0..SETTLE_SAMPLES { - std::thread::sleep(std::time::Duration::from_millis(10)); - let next = Census::now(); - if next == last { - return next; - } - last = next; - } - last -} - fn main() { if std::env::args().nth(1).as_deref() == Some("donor") { return donor(); @@ -73,7 +43,7 @@ fn main() { // nothing else in the guest holds or releases a page across the window, and // nothing orders that. A live object count moves only when somebody makes // or releases one, and it is exact: a leak of one region is `+1`. - let start = settled_census(); + let start = census_wait::settled(); let mut regions = Vec::new(); for _ in 0..ROUNDS { @@ -94,13 +64,7 @@ fn main() { ); drop(regions); - let after = settled_census(); - let grown: Vec<_> = after.grown_since(&start).collect(); - assert!( - grown.is_empty(), - "{ROUNDS} regions were allocated, mapped and dropped, and this was not released: \ - {grown:?} --- first {start}, then {after}" - ); + census_wait::released_to(&start); // The other direction. The donor makes a region, sends it here, and drops // its own handle before this process has mapped anything — nobody has the // region mapped at that moment, and it is still this process's. diff --git a/tests/toyos-rust-tests/src/bin/soundd_log_stall.rs b/tests/toyos-rust-tests/src/bin/soundd_log_stall.rs index 20b7c2e6e63..60d8136a60b 100644 --- a/tests/toyos-rust-tests/src/bin/soundd_log_stall.rs +++ b/tests/toyos-rust-tests/src/bin/soundd_log_stall.rs @@ -9,10 +9,9 @@ //! is said where nobody reads. A mix thread that waits on its output stops //! there, and the tone with it. //! -//! The verdicts are the host's: the capture carries the tone without a gap, and -//! once `logd` reads again soundd's lines account for every connection — each -//! refusal said, or counted by `logd` among the records that found the ring -//! full. +//! The verdict is `/log`'s: once `logd` reads again soundd's lines account for +//! every connection — each refusal said, or counted by `logd` among the records +//! that found the ring full. #[path = "../tone.rs"] mod tone; diff --git a/tests/toyos-rust-tests/src/bin/spawn_cwd.rs b/tests/toyos-rust-tests/src/bin/spawn_cwd.rs index daf96d422bd..22479663530 100644 --- a/tests/toyos-rust-tests/src/bin/spawn_cwd.rs +++ b/tests/toyos-rust-tests/src/bin/spawn_cwd.rs @@ -40,8 +40,7 @@ const LOG_DIR: &str = "/log/spawn-cwd/dir"; const LOG_FILE: &str = "/log/spawn-cwd/file"; const HOME_DIR: &str = "/home/spawn-cwd/dir"; const HOME_FILE: &str = "/home/spawn-cwd/file"; -/// Spawns timed from each cwd, so the figure is an average and not one outlier. -const TIMED: u32 = 8; +const SPAWNS: u32 = 8; fn main() { let args: Vec = std::env::args().collect(); @@ -199,17 +198,13 @@ fn refusals() { println!("spawn-cwd: every refusal arrived under its own name"); } -/// A cwd with more beneath it than one listing may hold is still a cwd. The -/// spawns it times are all from `/` or a tmpfs, so they price tmpfs's `is_dir` -/// and no other filesystem's. +/// A cwd with more beneath it than one listing may hold is still a cwd. fn a_cwd_is_judged_in_its_depth() { // Never made by `mkdir`: a directory the VFS carries is answered from its own // set, and this one has to be judged by the filesystem its files are on. std::fs::File::create(format!("{BIG}/0")).expect("/tmp is writable"); - // Unmeasured, so the first figure is not the one that pays for a cold start. - mean_spawn("/"); - let from_root = mean_spawn("/"); - let from_named = mean_spawn(NAMED); + spawns_from("/"); + spawns_from(NAMED); // Entered while small, grown after: the process that `cd`s into a build // directory is not asked again when the build fills it. std::env::set_current_dir(BIG).expect("chdir into a directory this made"); @@ -217,26 +212,18 @@ fn a_cwd_is_judged_in_its_depth() { std::fs::File::create(format!("{BIG}/{i}")).expect("/tmp is writable"); } said("own cwd over a large subtree, direct", Command::new(SELF).arg("pwd").output(), BIG); - let from_big = mean_spawn(BIG); + spawns_from(BIG); std::env::set_current_dir("/").expect("chdir to /"); - println!( - "spawn-cwd: a spawn and wait, mean of {TIMED} after {TIMED} unmeasured, no cwd on \ - bcachefs or FAT32: from / {} us, from {NAMED} {} us, from {BIG} ({BIG_FILES} files) {} us", - from_root.as_micros(), - from_named.as_micros(), - from_big.as_micros(), - ); + println!("spawn-cwd: {SPAWNS} spawns each from /, {NAMED} and {BIG} ({BIG_FILES} files)"); // Removed by name: a listing of this directory is the very thing it outgrew. for i in 0..BIG_FILES { std::fs::remove_file(format!("{BIG}/{i}")).expect("remove a file this made"); } } -/// The mean of [`TIMED`] direct spawns-and-waits from `cwd`. -fn mean_spawn(cwd: &str) -> std::time::Duration { - let start = std::time::Instant::now(); - for _ in 0..TIMED { - assert_eq!(spawn_in(cwd), Ok(0), "a timed spawn into {cwd}"); +/// [`SPAWNS`] direct spawns-and-waits from `cwd`, each answered. +fn spawns_from(cwd: &str) { + for _ in 0..SPAWNS { + assert_eq!(spawn_in(cwd), Ok(0), "a spawn into {cwd}"); } - start.elapsed() / TIMED } diff --git a/tests/toyos-rust-tests/src/bin/std_threading.rs b/tests/toyos-rust-tests/src/bin/std_threading.rs index ea89f9ad56e..05c73513162 100644 --- a/tests/toyos-rust-tests/src/bin/std_threading.rs +++ b/tests/toyos-rust-tests/src/bin/std_threading.rs @@ -1,19 +1,16 @@ +use std::io::Read; +use std::process::{Command, Stdio}; use std::thread; use std::time::Duration; -/// The child's life on the first attempt; a lost race quadruples it. **A host -/// fact, not a bound** — winning it means being asked before the child ends. -const LINGER: Duration = Duration::from_millis(200); -/// Attempts at the running answer. Three covers a 64x slower host. -const TRIES: u32 = 3; -/// The poll for the exited answer, and its ceiling — a liveness margin. +/// Between two asks of the exited answer. A pace and never a verdict: a child +/// `try_wait` never sees exit is a hang the harness ceiling reds. const POLL: Duration = Duration::from_millis(10); -const POLLS: u32 = 500; fn main() { - // The `try_wait` target, told how long to stay up. - if let Some(ms) = std::env::args().nth(1).and_then(|a| a.strip_prefix("linger=").map(String::from)) { - thread::sleep(Duration::from_millis(ms.parse().expect("linger= wants milliseconds"))); + // The `try_wait` target, up until its stdin closes. + if std::env::args().nth(1).as_deref() == Some("linger") { + std::io::stdin().read_to_end(&mut Vec::new()).expect("the linger child reads its stdin"); return; } @@ -38,40 +35,24 @@ fn main() { assert_eq!(total, expected, "partial sums mismatch: {total} != {expected}"); // Both answers, because a `try_wait` stuck on either satisfies the other. - // The running answer re-arms with a longer child rather than reding a lost race. + // The child cannot end before its stdin closes, so the running answer is + // asked of a child that is running whatever the clock says. let exe = std::env::current_exe().expect("current_exe failed"); - let mut running = None; - for attempt in 0..TRIES { - let mut child = std::process::Command::new(&exe) - .arg(format!("linger={}", (LINGER * 4u32.pow(attempt)).as_millis())) - .spawn() - .expect("spawn child failed"); - match child.try_wait().expect("try_wait failed") { - None => { - running = Some(child); - break; - } - Some(_) => child.wait().map(|_| ()).expect("reap the raced child"), - } - } - let mut child = running.unwrap_or_else(|| { - panic!( - "try_wait reported an exited child on all {TRIES} attempts, the last a {:?} sleep", - LINGER * 4u32.pow(TRIES - 1) - ) - }); - - let mut exited = None; - for _ in 0..POLLS { + let mut child = Command::new(&exe) + .arg("linger") + .stdin(Stdio::piped()) + .spawn() + .expect("spawn child failed"); + let running = child.try_wait().expect("try_wait failed"); + assert!(running.is_none(), "try_wait reported {running:?} for a child still reading its stdin"); + + drop(child.stdin.take().expect("the child's piped stdin")); + let status = loop { if let Some(status) = child.try_wait().expect("try_wait failed") { - exited = Some(status); - break; + break status; } thread::sleep(POLL); - } - let status = exited.unwrap_or_else(|| { - panic!("try_wait never reported the exited child within {:?}", POLL * POLLS) - }); + }; assert!(status.success(), "child exited with {status}"); println!("all threading tests passed"); diff --git a/tests/toyos-rust-tests/src/bin/swap_claim_astray.rs b/tests/toyos-rust-tests/src/bin/swap_claim_astray.rs index aec99029cf4..9e43afd0b51 100644 --- a/tests/toyos-rust-tests/src/bin/swap_claim_astray.rs +++ b/tests/toyos-rust-tests/src/bin/swap_claim_astray.rs @@ -7,11 +7,9 @@ //! faults. The part's interrupts stay masked, so nothing but that fault can //! wake this program; what it then reads from the claim is the verdict. //! -//! It exits 1 on the refusal it waits for, and 2 when [`TOLD_WITHIN`] passed -//! or a wait on the claim ended with nothing ready: a refusal read after an -//! unwoken wait is one nobody was woken for. - -use std::time::{Duration, Instant}; +//! It exits 1 on the refusal it waits for, with no deadline, and 2 when a wait +//! on the claim ended with nothing ready: a refusal read after an unwoken wait +//! is one nobody was woken for. use toyos::poller::{Poller, READABLE}; use toyos_abi::syscall::{PciId, SyscallError}; @@ -24,10 +22,6 @@ const E82574: PciId = PciId { vendor: 0x8086, device: 0x10d3 }; /// granted in total, so nothing this claim holds is there. const ASTRAY: u64 = 64 * 1024 * 1024; -/// A liveness bound on the host's frames and the kernel's wake, never a -/// measurement. -const TOLD_WITHIN: Duration = Duration::from_secs(20); - /// The register window, as volatile 32-bit accesses. struct Bar(*mut u8); @@ -74,7 +68,6 @@ fn main() { println!("swap_claim_astray: holding the NIC mastering, its receive ring at {ring:#x}, outside its grant"); let poller = Poller::new(1); - let asked = Instant::now(); loop { match dev.irq() { Ok(_) | Err(SyscallError::WouldBlock) => {} @@ -83,15 +76,11 @@ fn main() { std::process::exit(1); } } - let Some(left) = TOLD_WITHIN.checked_sub(asked.elapsed()) else { - println!("swap_claim_astray: its claim refused nothing in {TOLD_WITHIN:?}"); - std::process::exit(2); - }; poller.watch(&dev, READABLE, 0); let mut woken = false; - poller.wait(1, left.as_nanos() as u64, |_| woken = true); + poller.wait(1, u64::MAX, |_| woken = true); if !woken { - println!("swap_claim_astray: its claim refused nothing: no wake in {TOLD_WITHIN:?}"); + println!("swap_claim_astray: its claim refused nothing: a wait with no deadline ended unwoken"); std::process::exit(2); } } diff --git a/tests/toyos-rust-tests/src/bin/tlb_shootdown_waits.rs b/tests/toyos-rust-tests/src/bin/tlb_shootdown_waits.rs index 51c3a4ee9b4..71b04e28cd2 100644 --- a/tests/toyos-rust-tests/src/bin/tlb_shootdown_waits.rs +++ b/tests/toyos-rust-tests/src/bin/tlb_shootdown_waits.rs @@ -30,7 +30,7 @@ const DELAY_NANOS: u64 = 20_000_000; /// Half the delay. Every number compared against it is a lower bound on a spin /// the kernel or this process performs, so it cannot come out short for /// scheduling reasons — but the clock reads bracketing it are syscalls, and the -/// margin is there so a slow host cannot turn a pass into a fail either way. +/// margin is there so a slow host cannot turn a pass into a fail. const FLOOR_NANOS: u64 = DELAY_NANOS / 2; /// How many measured `munmap`s must *all* return fast before that is the @@ -163,19 +163,5 @@ fn main() { disarm(); - // 3. And the delay is what produced every number above, not the machine: - // disarmed, the same operation is back to microseconds. Without this the - // assertions above would still pass on a kernel that happened to be slow - // for some other reason. - let quiet = map(PAGE_2M); - let elapsed = timed(|| { - unsafe { syscall::munmap(quiet, PAGE_2M) }.expect("munmap"); - }); - assert!( - elapsed < FLOOR_NANOS, - "munmap still took {elapsed}ns with the delay disarmed, so the numbers above \ - measured something other than the wait", - ); - println!("a shootdown waits for every other CPU, and munmap and a fixed mmap wait for it"); } diff --git a/tests/toyos-rust-tests/src/bin/tls_dtv_race.rs b/tests/toyos-rust-tests/src/bin/tls_dtv_race.rs index dfda028af3e..255eecaddcc 100644 --- a/tests/toyos-rust-tests/src/bin/tls_dtv_race.rs +++ b/tests/toyos-rust-tests/src/bin/tls_dtv_race.rs @@ -15,7 +15,6 @@ //! not; `tls_rebase_window` runs this there and reads which. use std::sync::atomic::{AtomicBool, AtomicU64, Ordering::SeqCst}; -use std::time::{Duration, Instant}; use toyos_abi::syscall; /// `kernel/src/loader/tls.rs`'s `rebase_window::MARK`. @@ -27,8 +26,6 @@ const BLOCK: u64 = 2 * 1024 * 1024; const DTV_SLOT0: u64 = 16; /// Spawn/retire rounds; every one after the first is watched. const ROUNDS: u64 = 16; -/// A liveness bound on another thread's store, never a pace. -const BOUND: Duration = Duration::from_secs(10); /// The block base the sibling stores into. static TARGET: AtomicU64 = AtomicU64::new(0); @@ -51,11 +48,10 @@ thread_local! { static ANCHOR: u8 = const { 0 }; } -/// Spin until `done`, or panic naming `what` past [`BOUND`]. -fn until(what: &str, done: impl Fn() -> bool) { - let deadline = Instant::now() + BOUND; +/// Spin until `done`, with no deadline: a store that never comes is a hang the +/// harness ceiling reds. +fn until(done: impl Fn() -> bool) { while !done() { - assert!(Instant::now() < deadline, "tls_dtv_race: {what} did not come within {BOUND:?}"); core::hint::spin_loop(); } } @@ -66,7 +62,7 @@ extern "C" fn worker(_arg: u64) { let block = ANCHOR.with(|anchor| anchor as *const u8 as u64) & !(BLOCK - 1); TARGET.store(block, SeqCst); READY.store(true, SeqCst); - until("leave to exit", || GO_EXIT.load(SeqCst)); + until(|| GO_EXIT.load(SeqCst)); syscall::thread_exit(0); } @@ -122,12 +118,12 @@ fn main() { // SAFETY: `worker` is a valid entry; `top`/`base` describe the leaked stack. let tid = unsafe { syscall::thread_spawn(entry, top, arg, base) }; assert!(syscall::SyscallError::from_u64(tid).is_none(), "thread_spawn failed: {tid}"); - until("the worker's start", || READY.load(SeqCst)); - until("the sibling's store into the block", || ENGAGED.load(SeqCst) > round); + until(|| READY.load(SeqCst)); + until(|| ENGAGED.load(SeqCst) > round); // Stood down and acknowledged before the join frees the block, so no // store is in flight against a free. PAUSE.store(true, SeqCst); - until("the sibling standing down", || PAUSED_ACK.load(SeqCst)); + until(|| PAUSED_ACK.load(SeqCst)); GO_EXIT.store(true, SeqCst); assert_eq!(syscall::thread_join(tid), 0, "round {round}: join"); } diff --git a/tests/toyos-rust-tests/src/bin/wall_clock_now.rs b/tests/toyos-rust-tests/src/bin/wall_clock_now.rs index e5fc7629f3a..08c6d3417e0 100644 --- a/tests/toyos-rust-tests/src/bin/wall_clock_now.rs +++ b/tests/toyos-rust-tests/src/bin/wall_clock_now.rs @@ -11,18 +11,8 @@ //! broken. A machine with no wall clock is not a failure here — printing that //! it has none is the answer the host is checking for. -use std::time::Instant; - -/// Enough calls that a per-call port handshake would be unmistakable, and few -/// enough that the loop itself is free. The kernel used to read the CMOS on -/// every one of these — two port accesses per register, after a wait on the -/// update-in-progress flag that could take a second. const CALLS: u32 = 1000; -/// Both readings come from `SYS_CLOCK_EPOCH` a call apart, so this is a tick -/// rather than a margin. -const MAX_CLOCK_SKEW_SECS: u64 = 2; - fn main() { let epoch = toyos::system::clock_epoch(); let time = toyos::system::clock_realtime(); @@ -53,19 +43,20 @@ fn main() { .expect("std put the wall clock before the epoch") .as_secs(); println!("wall-clock: std_epoch={std_epoch}"); - assert!( - std_epoch.abs_diff(epoch) <= MAX_CLOCK_SKEW_SECS, - "std's SystemTime::now says {std_epoch} and SYS_CLOCK_EPOCH says {epoch}", - ); - let began = Instant::now(); let mut last = 0; for _ in 0..CALLS { last = toyos::system::clock_epoch().expect("the clock answered once and then stopped"); } - let elapsed = began.elapsed(); - println!("wall-clock: {CALLS} calls in {}us, last={last}", elapsed.as_micros()); + println!("wall-clock: {CALLS} calls, last={last}"); - // Monotonic-plus-offset cannot go backwards inside a boot. + // Monotonic-plus-offset cannot go backwards inside a boot, and std's + // reading sits between the two the kernel gave either side of it: order, + // with no margin in it. assert!(last >= epoch, "the wall clock went backwards: {epoch} then {last}"); + assert!( + (epoch..=last).contains(&std_epoch), + "std's SystemTime::now says {std_epoch}, outside the {epoch}..={last} SYS_CLOCK_EPOCH \ + answered either side of it", + ); } diff --git a/tests/toyos-rust-tests/src/bin/window_child.rs b/tests/toyos-rust-tests/src/bin/window_child.rs index 37809000f3e..feaa9c356ac 100644 --- a/tests/toyos-rust-tests/src/bin/window_child.rs +++ b/tests/toyos-rust-tests/src/bin/window_child.rs @@ -18,18 +18,11 @@ //! away, which is the owner's case: the process is alive when its window is //! closed. -use std::time::{Duration, Instant}; - use window::{Color, Event, Window}; const WIDTH: u32 = 320; const HEIGHT: u32 = 200; -/// A liveness ceiling, not a duration: it costs nothing when the close -/// arrives, and bounds a run where nothing ever closes this window. -const RUN_CEILING: Duration = Duration::from_secs(30); -const POLL_NS: u64 = 200_000_000; - fn main() { let leave_now = std::env::args().nth(1).as_deref() == Some("exit"); @@ -45,19 +38,10 @@ fn main() { window.present(); println!("WINDOW-CHILD-UP"); + // No deadline: a window nothing ever closes is a hang the host's ceiling + // reds. if !leave_now { - let deadline = Instant::now() + RUN_CEILING; - loop { - match window.poll_event(POLL_NS) { - Some(Event::Close) => break, - Some(_) => {} - None if Instant::now() >= deadline => { - println!("WINDOW-CHILD-TIMEOUT"); - break; - } - None => {} - } - } + while !matches!(window.poll_event(u64::MAX), Some(Event::Close)) {} } println!("WINDOW-CHILD-GONE"); diff --git a/tests/toyos-rust-tests/src/bin/window_drag.rs b/tests/toyos-rust-tests/src/bin/window_drag.rs index e7a9772dd4d..08a63520477 100644 --- a/tests/toyos-rust-tests/src/bin/window_drag.rs +++ b/tests/toyos-rust-tests/src/bin/window_drag.rs @@ -17,8 +17,6 @@ //! area of the window, so a window that filled the screen would pass a gate //! about screen-sized damage without meaning anything. -use std::time::{Duration, Instant}; - use window::{Color, Event, Window, MOUSE_PRESS}; /// Short enough that the middle of the screen and the title bar are a known @@ -35,11 +33,6 @@ const HEIGHT: u32 = 160; /// it, so the second is the end of the sequence. const PRESSES: usize = 2; -/// A liveness ceiling, not a duration: it costs nothing when the presses -/// arrive, and bounds a run where the injected pointer never reached this -/// window at all. -const RUN_CEILING: Duration = Duration::from_secs(30); - fn main() { let mut window = Window::create_with_title(WIDTH, HEIGHT, "drag") .unwrap_or_else(|e| panic!("the compositor would not give this a window: {e}")); @@ -51,10 +44,11 @@ fn main() { println!("drag probe: {WIDTH}x{HEIGHT} window up"); println!("===DRAG_READY==="); - let deadline = Instant::now() + RUN_CEILING; + // No deadline: an injected pointer that never reaches this window is a hang + // the host's ceiling reds. let mut presses = 0; - while presses < PRESSES && Instant::now() < deadline { - match window.poll_event(Duration::from_millis(200).as_nanos() as u64) { + while presses < PRESSES { + match window.poll_event(u64::MAX) { Some(Event::MouseInput(ev)) if ev.event_type == MOUSE_PRESS && ev.changed == 1 => { presses += 1; println!("drag probe: press at {},{}", ev.x, ev.y); diff --git a/tests/toyos-rust-tests/src/bin/window_wake.rs b/tests/toyos-rust-tests/src/bin/window_wake.rs index fe3c38b54a5..41f9e587c96 100644 --- a/tests/toyos-rust-tests/src/bin/window_wake.rs +++ b/tests/toyos-rust-tests/src/bin/window_wake.rs @@ -18,20 +18,15 @@ use std::sync::mpsc; use std::thread; -use std::time::{Duration, Instant}; use toyos_abi::syscall::SyscallError; -use window::{Waiter, Window, Woke}; +use window::{Waiter, Window}; const ROUNDS: u32 = 200; /// More than a new `Waiter` has room for once the wake is counted. const WINDOWS: usize = 4; -/// A liveness ceiling, not a duration: a delivered wake ends the wait at once, -/// and only a lost one reaches this. -const CEILING: Duration = Duration::from_secs(10); - fn main() { let mut windows: Vec = (0..WINDOWS) .map(|i| match Window::create_with_title(160, 100, &format!("wake {i}")) { @@ -57,13 +52,12 @@ fn main() { let pending_first = round % 2 == 0; go.send(()).expect("the helper is alive until `go` drops"); if pending_first { - raised.recv_timeout(CEILING).expect("the helper raised the wake"); + raised.recv().expect("the helper raised the wake"); } + // No deadline: a lost wake leaves this wait blocked, and the host's + // ceiling reds it. loop { - if waiter.wait(windows.iter().map(Window::handle), Some(CEILING)) == Woke::TimedOut { - println!("WINDOW-WAKE-LOST round={round}"); - std::process::exit(1); - } + waiter.wait(windows.iter().map(Window::handle), None); if waiter.take_wake() { break; } @@ -73,7 +67,7 @@ fn main() { } } if !pending_first { - raised.recv_timeout(CEILING).expect("the helper raised the wake"); + raised.recv().expect("the helper raised the wake"); } } drop(go); @@ -81,14 +75,11 @@ fn main() { let flood = pipe_capacity() + 1; let waker = waiter.waker(); - let raising = Instant::now(); for _ in 0..flood { waker.wake(); } - let raised_in = raising.elapsed(); - if waiter.wait(windows.iter().map(Window::handle), Some(CEILING)) == Woke::TimedOut - || !waiter.take_wake() - { + waiter.wait(windows.iter().map(Window::handle), None); + if !waiter.take_wake() { println!("WINDOW-WAKE-LOST after a flood of {flood} wakes"); std::process::exit(1); } @@ -96,7 +87,7 @@ fn main() { println!("WINDOW-WAKE-LOST a flood of {flood} wakes was not taken by one take"); std::process::exit(1); } - println!("WINDOW-WAKE-OK rounds={ROUNDS} windows={WINDOWS} flood={flood} raised in {raised_in:?}"); + println!("WINDOW-WAKE-OK rounds={ROUNDS} windows={WINDOWS} flood={flood}"); } /// The bytes a fresh pipe holds before a write would block. diff --git a/tests/toyos-rust-tests/src/bin/winit_loop.rs b/tests/toyos-rust-tests/src/bin/winit_loop.rs index 42d65245f48..f7fecbd1784 100644 --- a/tests/toyos-rust-tests/src/bin/winit_loop.rs +++ b/tests/toyos-rust-tests/src/bin/winit_loop.rs @@ -3,8 +3,8 @@ //! is waiting, and a window its application dropped hears its `Destroyed` and //! nothing else from the drop on, which every stage checks. //! -//! Six stages, one after another, each ended by what it waits for or failed -//! by [`CEILING`], a liveness bound only a lost wake reaches: +//! Six stages, one after another, each ended by what it waits for; a lost wake +//! is a hang the harness ceiling reds: //! //! 1. A user event sent from `AboutToWait`, with no window open: the wake it //! raises is taken by the next iteration, never by the one that sent it. @@ -19,10 +19,9 @@ //! `Resized`, delivers nothing for it. //! 5. A window asked for a redraw and dropped in a `user_event`, which runs //! after the loop's destroy step: no `RedrawRequested` reaches it. -//! 6. A window whose close the application ignores: once the compositor has -//! closed it, the loop does not wake for it while it waits out -//! [`IDLE_WINDOW`]. The harness closes it with GUI+Q when told -//! `WINIT-LOOP CLOSE-ME`. +//! 6. A window whose close the application ignores: the compositor's close +//! reaches it as `CloseRequested`. The harness closes it with GUI+Q when +//! told `WINIT-LOOP CLOSE-ME`. //! //! Every window but stage 6's is closed by the application's drop, which the //! harness reads off the compositor's close lines. @@ -31,35 +30,19 @@ use std::num::NonZeroU32; use std::sync::mpsc; use std::sync::Arc; use std::thread; -use std::time::{Duration, Instant}; use softbuffer::{Context, Surface}; use winit::application::ApplicationHandler; -use winit::event::{StartCause, WindowEvent}; +use winit::event::WindowEvent; use winit::event_loop::{ActiveEventLoop, ControlFlow, EventLoop, EventLoopProxy, OwnedDisplayHandle}; use winit::window::{Window, WindowAttributes, WindowId}; -/// A liveness ceiling, not a duration: a delivered event ends each wait at -/// once, and only a lost wake reaches it. -const CEILING: Duration = Duration::from_secs(20); - const HELPER_EVENTS: u32 = 100; /// Rounds, because a push that lands before the loop has looked at its queues /// needs no wake. const HELPER_ROUNDS: u32 = 20; -/// How long stage 6 watches a loop with nothing to deliver. A measurement -/// window rather than a wait for an event: what is counted is what happens -/// when nothing does. -const IDLE_WINDOW: Duration = Duration::from_secs(1); - -/// The iterations an idle loop may start in [`IDLE_WINDOW`]: the one that -/// ends it at its deadline. A loop that waits on a closed connection wakes for -/// as long as the window lasts, and one the app dropped but the loop still -/// holds is sent an event by the close. -const IDLE_WAKES: usize = 1; - enum Ev { Ping, Seq(u32), @@ -75,7 +58,7 @@ enum Stage { Handed { round: u32, step: Handed }, Dropped { id: WindowId, destroyed: bool, waits: u32 }, UserDrop { window: Option>, id: WindowId, destroyed: bool, waits: u32 }, - Close { window: Option>, surface: Option>>, idle: Option<(Instant, Vec)> }, + Close { window: Option>, surface: Option>>, closed: bool }, } enum Handed { @@ -142,41 +125,11 @@ impl App { let window = self.create(event_loop, "ignores its close"); let surface = Surface::new(&self.context, window.clone()) .unwrap_or_else(|e| fail(&format!("surface: {e}"))); - self.stage = Stage::Close { window: Some(window), surface: Some(surface), idle: None }; + self.stage = Stage::Close { window: Some(window), surface: Some(surface), closed: false }; } } impl ApplicationHandler for App { - fn new_events(&mut self, _event_loop: &ActiveEventLoop, cause: StartCause) { - if let Stage::Close { idle: Some((deadline, wakes)), .. } = &mut self.stage { - let left = deadline.saturating_duration_since(Instant::now()); - wakes.push(format!("{cause:?} with {left:?} of the window left")); - return; - } - if let StartCause::ResumeTimeReached { .. } = cause { - let at = match &self.stage { - Stage::Ping { .. } => "the user event sent from AboutToWait".to_string(), - Stage::Helper { next, .. } => format!("helper event {next}"), - Stage::Handed { round, step: Handed::Created(_) } => format!("round {round}'s first redraw"), - Stage::Handed { round, step: Handed::RedrawAsked(_) } => { - format!("round {round}'s redraw asked from the helper") - } - Stage::Handed { round, step: Handed::DropAsked(_) } => { - format!("round {round}'s window dropped on the helper") - } - Stage::Dropped { .. } => "the Destroyed of a window dropped in its handler".to_string(), - Stage::UserDrop { window: Some(_), .. } => { - "the first redraw of the window a user event drops".to_string() - } - Stage::UserDrop { window: None, .. } => { - "the Destroyed of a window dropped in a user event".to_string() - } - Stage::Close { .. } => "the compositor's close".to_string(), - }; - fail(&format!("LOST: the loop waited out its ceiling for {at}")); - } - } - fn resumed(&mut self, _event_loop: &ActiveEventLoop) {} fn user_event(&mut self, event_loop: &ActiveEventLoop, event: Ev) { @@ -191,8 +144,8 @@ impl ApplicationHandler for App { if proxy.send_event(Ev::Seq(i)).is_err() { fail("the loop closed under the helper"); } - if acked.recv_timeout(CEILING).is_err() { - fail(&format!("LOST: helper event {i} was never delivered")); + if acked.recv().is_err() { + fail(&format!("helper event {i}'s ack channel closed")); } } })) @@ -279,7 +232,7 @@ impl ApplicationHandler for App { *destroyed = true; } } - Stage::Close { window: Some(window), surface: Some(surface), idle } if window.id() == id => { + Stage::Close { window: Some(window), surface: Some(surface), closed } if window.id() == id => { match event { WindowEvent::RedrawRequested => { let size = window.inner_size(); @@ -293,10 +246,8 @@ impl ApplicationHandler for App { buffer.present().unwrap_or_else(|e| fail(&format!("present: {e}"))); println!("WINIT-LOOP CLOSE-ME"); } - WindowEvent::CloseRequested if idle.is_none() => { - // Ignored, as an application that asks "save first?" does. - *idle = Some((Instant::now() + IDLE_WINDOW, Vec::new())); - } + // Ignored, as an application that asks "save first?" does. + WindowEvent::CloseRequested => *closed = true, _ => {} } } @@ -344,29 +295,14 @@ impl ApplicationHandler for App { self.jobs.send(job).expect("the helper outlives the loop"); } match &mut self.stage { - Stage::Close { idle: Some((deadline, wakes)), window, surface } => { - if Instant::now() < *deadline { - event_loop.set_control_flow(ControlFlow::WaitUntil(*deadline)); - return; - } - let count = wakes.len(); - if count > IDLE_WAKES { - let first: Vec<&String> = wakes.iter().take(8).collect(); - fail(&format!( - "a closed window the application kept woke the loop {count} times in \ - {IDLE_WINDOW:?}, first {first:?}" - )); - } - println!( - "WINIT-LOOP stage 6: a closed window the application kept woke the loop {count} \ - time(s) in {IDLE_WINDOW:?}: {wakes:?}" - ); + Stage::Close { closed: true, window, surface } => { + println!("WINIT-LOOP stage 6: a closed window the application kept got CloseRequested"); surface.take(); window.take(); println!("WINIT-LOOP-OK"); event_loop.exit(); } - _ => event_loop.set_control_flow(ControlFlow::WaitUntil(Instant::now() + CEILING)), + _ => event_loop.set_control_flow(ControlFlow::Wait), } } } diff --git a/tests/toyos-rust-tests/src/census_wait.rs b/tests/toyos-rust-tests/src/census_wait.rs new file mode 100644 index 00000000000..fba76532e8f --- /dev/null +++ b/tests/toyos-rust-tests/src/census_wait.rs @@ -0,0 +1,43 @@ +//! The live-object census, waited on rather than sampled against a clock. +//! +//! **A release is not finished when the syscall that caused it returns.** The +//! last handle's drop queues the object on the object layer's zero-handle +//! queue, and `object::drain_zero_handles` clears its pending flag before it +//! runs the hooks, so the release can land on another CPU after the caller's +//! own drain site found the queue empty +//! (`issues/kernel/deferred-release-outlives-its-syscall.md`). So a reading is +//! a wait for an event, and a leak is the event that never comes: the harness +//! ceiling reds it. + +use std::time::Duration; + +use toyos::census::Census; + +/// Between two readings. A pace and never a verdict: the zero-handle queue is +/// drained on the idle loop among other places, and a reader that never +/// sleeps keeps its CPU from ever reaching it. +const PACE: Duration = Duration::from_millis(10); + +/// The census once two readings a pace apart agree. +pub fn settled() -> Census { + let mut last = Census::now(); + loop { + std::thread::sleep(PACE); + let next = Census::now(); + if next == last { + return next; + } + last = next; + } +} + +/// The census once no kind has grown since `before`. +pub fn released_to(before: &Census) -> Census { + loop { + let now = Census::now(); + if now.grown_since(before).next().is_none() { + return now; + } + std::thread::sleep(PACE); + } +} diff --git a/tests/toyos-rust-tests/src/netd_stream.rs b/tests/toyos-rust-tests/src/netd_stream.rs index 02f86c036b8..996fdd6c319 100644 --- a/tests/toyos-rust-tests/src/netd_stream.rs +++ b/tests/toyos-rust-tests/src/netd_stream.rs @@ -5,7 +5,7 @@ //! Each netd stream test includes this file whole and uses its own part of it. #![allow(dead_code)] -use std::time::{Duration, Instant}; +use std::time::Duration; use toyos::poller::{Poller, READABLE, WRITABLE}; use toyos::{AsHandle, Pipe}; @@ -47,8 +47,9 @@ pub fn ring_capacity() -> u64 { fill(&write) } -/// Wait until `check` answers, re-asking it each time `handle` reports ready -/// for `flags`, and panic by name if `within` passes first. +/// Wait, with no deadline, until `check` answers, re-asking it each time +/// `handle` reports ready for `flags`: an answer that never comes is a hang the +/// harness ceiling reds. /// /// **A readiness completion is a reason to look again, not an answer**: a /// zero-byte write still wakes the other end's watch, and netd's liveness @@ -56,20 +57,15 @@ pub fn ring_capacity() -> u64 { pub fn await_until( handle: &impl AsHandle, flags: u32, - within: Duration, - what: &str, mut check: impl FnMut() -> Option, ) -> T { - let deadline = Instant::now() + within; let poller = Poller::new(1); loop { if let Some(answer) = check() { return answer; } - let left = deadline.saturating_duration_since(Instant::now()); - assert!(!left.is_zero(), "{what}: not within {within:?}"); poller.watch(handle, flags, 0); - poller.wait(1, left.as_nanos() as u64, |_| {}); + poller.wait(1, u64::MAX, |_| {}); } } @@ -89,6 +85,10 @@ pub enum Ask { /// connects to. pub const FORWARDED_PORT: u16 = 22; +/// A connect's `timeout_ms` that sets no deadline: netd times a connect out +/// only on a non-zero one. +pub const NO_DEADLINE: u32 = 0; + /// `what` on the wire: a mode byte, then eight little-endian bytes of length. pub fn ask_bytes(what: Ask) -> [u8; 9] { let (mode, total) = match what { @@ -108,14 +108,13 @@ pub fn ask(tx: &Pipe, what: Ask) { } /// Keep the pipe `write` feeds full until its reader is gone, looking again -/// each time the pipe reports room, and panic by name if `within` passes -/// first. +/// each time the pipe reports room. /// /// **A reader's departure is an event only for a full pipe**: one with room is /// writable already, so its watch would complete at once, every time. -pub fn keep_full_until_released(write: &Pipe, within: Duration, what: &str) { +pub fn keep_full_until_released(write: &Pipe, what: &str) { let chunk = [0u8; 65536]; - await_until(write, WRITABLE, within, what, || loop { + await_until(write, WRITABLE, || loop { match write.write_nonblock(&chunk) { Ok(_) => {} Err(SyscallError::WouldBlock) => return None, @@ -130,16 +129,17 @@ pub fn keep_full_until_released(write: &Pipe, within: Duration, what: &str) { /// the pattern's group at `capacity - 16`. /// /// **A poll, because nothing announces a full ring to its reader**: readiness -/// fires on the first byte. `within` is a liveness guard, said by name. -pub fn await_ring_full(rx: &Pipe, capacity: u64, within: Duration) { - await_ring_holds(rx, capacity, capacity - 16, within) +/// fires on the first byte. +pub fn await_ring_full(rx: &Pipe, capacity: u64) { + await_ring_holds(rx, capacity, capacity - 16) } /// Wait until `rx`'s ring holds the pattern's 16-byte group at stream position /// `group_start`, in the slot a ring of `capacity` bytes puts it: a group of an /// earlier lap in that slot carries another stamp. A poll, as /// [`await_ring_full`] is. -pub fn await_ring_holds(rx: &Pipe, capacity: u64, group_start: u64, within: Duration) { +pub fn await_ring_holds(rx: &Pipe, capacity: u64, group_start: u64) { + /// A pace and never a verdict: no deadline follows it. const POLL: Duration = Duration::from_millis(1); let page = rx.pipe_map().expect("map the receive pipe") as *const u8; // SAFETY: `pipe_map` returned the base of this pipe's mapped page, whose @@ -148,35 +148,26 @@ pub fn await_ring_holds(rx: &Pipe, capacity: u64, group_start: u64, within: Dura let group = unsafe { page.add(core::mem::size_of::() + (group_start % capacity) as usize) }; // SAFETY: inside the data region, as `group` above. let at = |i: u64| unsafe { group.add(i as usize).read_volatile() }; - let started = Instant::now(); while !(0..16).all(|i| at(i) == stream_byte(group_start + i)) { - assert!( - started.elapsed() < within, - "the receive ring never held stream byte {group_start} in {within:?}: its group reads {:02x?}, want {:02x?}", - (0..16).map(at).collect::>(), - (0..16).map(|i| stream_byte(group_start + i)).collect::>(), - ); syscall::nanosleep(POLL.as_nanos() as u64); } } -/// Read `rx` to its end, each wait for more bounded by `within`, and panic by -/// name at the first byte that is not the pattern's. Answers how many bytes -/// came. -pub fn read_pattern(rx: &Pipe, within: Duration, what: &str) -> u64 { - read_pattern_from(rx, 0, u64::MAX, within, what) +/// Read `rx` to its end, and panic by name at the first byte that is not the +/// pattern's. Answers how many bytes came. +pub fn read_pattern(rx: &Pipe, what: &str) -> u64 { + read_pattern_from(rx, 0, u64::MAX, what) } /// [`read_pattern`] from stream position `from`, ending at position `until`. /// Answers the position reached. -pub fn read_pattern_from(rx: &Pipe, from: u64, until: u64, within: Duration, what: &str) -> u64 { +pub fn read_pattern_from(rx: &Pipe, from: u64, until: u64, what: &str) -> u64 { let mut buf = vec![0u8; 65536]; let mut at = from; while at < until { let want = buf.len().min((until - at) as usize); - let waiting = format!("{what}: the stream after byte {at}"); - let n = await_until(rx, READABLE, within, &waiting, || match rx.read_nonblock(&mut buf[..want]) { + let n = await_until(rx, READABLE, || match rx.read_nonblock(&mut buf[..want]) { Ok(n) => Some(n), Err(SyscallError::WouldBlock) => None, Err(e) => panic!("{what}: reading the stream at {at}: {e:?}"), diff --git a/tests/toyos-rust-tests/src/roster.rs b/tests/toyos-rust-tests/src/roster.rs new file mode 100644 index 00000000000..cbd5612e11d --- /dev/null +++ b/tests/toyos-rust-tests/src/roster.rs @@ -0,0 +1,51 @@ +//! This process's threads, as the kernel's roster publishes them: the one +//! place a process can ask whether a thread of its own is parked. +//! +//! Each binary includes this file whole and uses its own part of it. +#![allow(dead_code)] + +use toyos::syscap::SysCap; +use toyos_abi::syscall; + +/// `sched::payload::SCHED_RUNNING` — the state column `ps` prints. +pub const RUNNING: u8 = 0; + +/// `sched::payload::SCHED_BLOCKED`. +pub const BLOCKED: u8 = 2; + +/// `SCHED_UNKNOWN`, which `sys_sysinfo` also answers for a thread whose entry +/// is a zombie. A live thread's scheduler record is installed under the same +/// table lock that inserts its entry, so a thread of ours reading this has +/// exited and nothing else. +pub const ZOMBIE: u8 = 3; + +/// This process's threads as the kernel publishes them: `(is a child thread, +/// scheduler state)`. +pub fn my_threads(cap: &SysCap) -> Vec<(bool, u8)> { + threads_of(cap, syscall::getpid().raw()) +} + +/// Process `pid`'s threads, as [`my_threads`] reads this one's. +/// +/// Even one's own threads arrive in the machine-wide roster, which is +/// `Rights::ROSTER` on a `SysCap` — there is no narrower question in the ABI, +/// and `tests/testcases` names `roster` on the test-runner row for this. +pub fn threads_of(cap: &SysCap, pid: u32) -> Vec<(bool, u8)> { + const HEADER: usize = toyos::system::SYSINFO_HEADER_SIZE; + const ENTRY: usize = toyos::system::SYSINFO_ENTRY_SIZE; + let mut buf = vec![0u8; HEADER + ENTRY * 256]; + let n = cap.roster(&mut buf); + assert!((HEADER..=buf.len()).contains(&n), "sysinfo answered {n}"); + (HEADER..) + .step_by(ENTRY) + .take_while(|pos| pos + ENTRY <= n) + .filter(|&pos| u32::from_le_bytes(buf[pos..pos + 4].try_into().unwrap()) == pid) + .map(|pos| (buf[pos + 9] != 0, buf[pos + 8])) + .collect() +} + +/// Poll until `cond` holds, with no deadline: a state the kernel never reaches +/// is a hang the harness ceiling reds. +pub fn await_true(cond: impl Fn() -> bool) { + while !cond() {} +} diff --git a/tests/toyos-rust-tests/src/tone.rs b/tests/toyos-rust-tests/src/tone.rs index c12e29851b0..72e0e5f7e38 100644 --- a/tests/toyos-rust-tests/src/tone.rs +++ b/tests/toyos-rust-tests/src/tone.rs @@ -1,7 +1,4 @@ -//! Shared tone playback for the audio glitch tests (audio_tone, -//! audio_tone_load). The host-side harness records what the virtio-sound -//! device plays into a wav and asserts the tone contains no mid-signal -//! silence (underruns) and no hard discontinuities (clicks). +//! Shared tone playback: a deterministic sine. use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::sync::Arc; diff --git a/tests/toyos.rs b/tests/toyos.rs index ff90ef5aae8..fe340049e57 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -9,14 +9,13 @@ use std::time::{Duration, Instant}; use common::qemu::{ self, await_guest, await_marker, await_marker_new, BootOptions, QemuInstance, TestResult, - STALLED, + STALLED, TIMED_OUT, }; use common::{ - audio, compile, devices, faults, hostload, lan, metal, partclaim, pkg, power, screen, serial, - stats, storage, usb, + audio, compile, devices, faults, lan, metal, partclaim, pkg, power, screen, serial, storage, + usb, }; use toyos_build::bootlog::{self, boot_millis}; -use toyos_build::heartbeat; use toyos_build::testargs::{self, Shard, SUITE}; use toyos_build::redlist; use toyos_build::tiers::{Reach, Schedule, Tier}; @@ -122,10 +121,6 @@ const ACTUATOR_TESTS: &[&str] = &[ // kernel's address space, so without them a kernel that still made the write // under test answers a userland that cannot notice. "abuse_kernel_addr", - // Actions 12 and 13: hold each other CPU's shootdown acknowledgement back in - // turn, so that whether the initiator waits for every one of them becomes a - // duration userland can read. - "tlb_shootdown_waits", // Action 16, the live-object census per kind, and 17 and 18 for the idle // stack the deferred release path runs on. A leak is two readings and a // comparison, so on a kernel that answers `InvalidArgument` both readings @@ -144,12 +139,6 @@ const ACTUATOR_TESTS: &[&str] = &[ /// no actuator armed in it. const ACTUATOR_KERNEL: &[&str] = toyos_build::build::TEST_KERNEL; -/// `tlb_shootdown_waits` asserts a claim about every *other* CPU, and two vCPUs -/// leave exactly one — a width at which a wait narrowed to a single sibling and -/// a wait for the whole set are the same measurement. Four is the smallest -/// width where they are not. -const ACTUATOR_SMP: u32 = 4; - /// How many times a shared block will answer a dead guest with a new one. /// /// Bounded because a block whose every member kills the guest must not boot one @@ -169,6 +158,16 @@ const RUST_SKIP: &[&str] = &[ // that measured anything. It also needs the real-time band, which only // `tests/latencycase` endows. `latency_wake` runs it there. "cyclictest", + // Its verdict is a ratio of cycle counts, which a guest's host moves: the + // `wake_storm_cost` metal row runs it on the T14. + "wake_storm_cost", + // Its product is a cycle count per syscall and the clock rate beside it, + // which a guest's host sets: the `syscall_cost` metal row runs it. + "syscall_cost", + // Its verdict is a duration: whether a shootdown waits for every other CPU + // is read off the clock around the syscall. The `tlb_shootdown_waits` + // metal row runs it on the T14. + "tlb_shootdown_waits", // **It reboots the machine**, so in the shared block it would end the boot // under whichever member came next; and its verdict is the order of the // console after that reset, which only its own boot holds. @@ -260,13 +259,12 @@ const RUST_SKIP: &[&str] = &[ // `netd_listener_forgery` runs it there. "netd_listener_forgery", // Needs a NIC in front of netd and a host server behind it. - // `netd_slow_reader`, `netd_held_open`, `netd_stalled_peer`, - // `netd_udp_refused`, `netd_udp_any_address`, `netd_refused_pipes` and - // `netd_refused_accept` run them on `tests/netcase`, and - // `netd_lookup_let_go` on it with its frames held. + // `netd_slow_reader`, `netd_held_open`, `netd_udp_refused`, + // `netd_udp_any_address`, `netd_refused_pipes` and `netd_refused_accept` + // run them on `tests/netcase`, and `netd_lookup_let_go` on it with its + // frames held. "netd_slow_reader", "netd_held_open", - "netd_stalled_peer", "netd_udp_refused", "netd_udp_any_address", "netd_refused_pipes", @@ -325,9 +323,8 @@ const RUST_SKIP: &[&str] = &[ // Needs netd with a NIC. `netd_connection_caps` runs it on tests/netcase. "netd_caps", // Need every owner `inspect` reads, which only tests/inspectcase runs. - // `inspect_reads_its_owners` runs all three there. + // `inspect_reads_its_owners` runs both there. "inspect_denied", - "inspect_plays", "inventory_bounds", // Same reason, same config: `netd_hostile_peer` runs it there. "netd_hostile_peer", @@ -371,20 +368,9 @@ const RUST_SKIP: &[&str] = &[ // printed four hundred lines to a console nothing was reading and passed // on its exit code. "test_screen_churn", - // Spawns `/system/bin/doom`, which `tests/testcases` does not carry — doom is - // 4 MiB and every other test boots that config. `doom_sound_flood` runs it - // on `tests/doomcase`. - "doom_sound_flood", - // Same, plus the WAD and the SoundFont doom's music is made of, which no - // other config should pay 19 MiB of ROOT for. `doom_music` runs it on - // `tests/doommusiccase`. - "doom_music", - // Same WAD, same config: `doom_frames` runs it on `tests/doommusiccase`. + // Spawns `/system/bin/doom` and reads the WAD, which `tests/testcases` does + // not carry. `doom_frames` runs it on `tests/doommusiccase`. "doom_frames", - // Needs a `logd` that leaves soundd's ring unread until it says so, and a - // capture of the tone it plays into that. `soundd_log_stall` runs it on - // `tests/logstallcase`. - "soundd_log_stall", // Its failure mode is a CPU that never runs anything again, so on the // shared boot it would be reported against whichever test came next — and // every one after that. `short_sleep_livelock` gives it a boot of its own. @@ -410,13 +396,14 @@ const RUST_SKIP: &[&str] = &[ // Stages real directories on `/log` for `fs_dirs_durable` to read back // off the image. "fs_dirs_durable", - // Needs an HDA controller, which `tests/testcases` has none of. + // Audio is judged on the T14 and nowhere else: the `hda_client_stall`, + // `hda_tone`, `audio_idle_suspend`, `shipped_client_departures` and + // `soundd_log_stall` metal rows run these. "hda_client_stall", - // Gate A's two, whose verdict is the wav the device captured — which the - // shared boot takes no capture of. The comment below has claimed since it - // was written that they are excluded from this boot; now they are. "audio_tone", - "audio_tone_load", + "audio_idle_suspend", + "null_sink_client_exits", + "soundd_log_stall", // Its whole subject is a page of a file the host wrote onto the volume // before the machine existed; the shared boot stages nothing, so it prints // `did not open` and passes on its exit code. `log_backing_read_error` @@ -475,7 +462,6 @@ const DRIVEN_AND_SHARED: &[&str] = &[ // not for anything it does: it is the cheapest process this tree starts. "empty_dir_stat", "hierarchy_paths", - "null_sink_client_exits", "nvme_home_roundtrip", "sched_stress", "std_alloc", @@ -483,16 +469,6 @@ const DRIVEN_AND_SHARED: &[&str] = &[ "wall_clock_now", ]; -// Audio glitch tests. Each runs in its own QEMU boot per SMP config and -// asserts on the wav the virtio-sound device captured, so they are excluded -// from the shared multi-test boot. -const AUDIO_TESTS: &[(&str, Tier)] = - &[("audio_tone", Tier::Nightly), ("audio_tone_load", Tier::Nightly)]; - -// Scheduler-core gate A covers both SMP configs: smp=1 is the audio spec's -// first-class single-CPU case, smp=8 the full-SMP case. -const AUDIO_SMP: &[u32] = &[1, 8]; - /// What `test-early-panic` panics with (`kernel/src/main.rs`): the last line its /// report puts on serial. const EARLY_PANIC_MESSAGE: &str = "test-early-panic: on-screen console check"; @@ -515,10 +491,8 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ // same boot's console; no clock is in either. ("screen_loader_lines", Sched::Parallel, Tier::Nightly), ("screen_gop_firmware_mode", Sched::Parallel, Tier::Weekly), - // `thread::sleep(5 s)` is the measurement, not a ceiling: the assertion is - // literally that the log is still on the panel five seconds after the boot - // finished, so a 2x slower machine changes nothing about the wait but the - // wait is the verdict either way — timer-anchored. + // The log is still on the panel once every program the image starts has + // run: an event, and no clock in it. ("screen_diag_boot", Sched::Parallel, Tier::Nightly), // A guest halted in the window, so the panel is read where only the repaint // under test can have painted it. @@ -529,10 +503,7 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ ("screen_console_scroll", Sched::Parallel, Tier::Fast), ("screen_i8042_health", Sched::Parallel, Tier::Weekly), // Ctrl+Alt+D with no console at all: the panel is the whole channel, and a - // compositor is holding it. A fixed 2 s settle sits inside the dump's own - // guest-timed 15 s hold, and the verdict is whether the report survived - // the desktop's next repaint — which only where that wait lands decides, - // so it is timer-anchored despite being a screendump-content check. + // compositor is holding it. The verdict is the report on the panel. ("screen_blocked_dump", Sched::Parallel, Tier::Nightly), ("screen_late_panic", Sched::Parallel, Tier::Fast), ("screen_paged_scrollback", Sched::Parallel, Tier::Weekly), @@ -549,7 +520,9 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ // `screen_blocked_dump` has one but paints through `paint_report` rather // than through `halt_all_cpus`. ("screen_fatal_halt_composited", Sched::Parallel, Tier::Nightly), - ("screen_pager_keys", Sched::Serial, Tier::Nightly), + // Every PageUp moves the page one back, which the unattended deadline + // never does: order, and no clock in it. + ("screen_pager_keys", Sched::Parallel, Tier::Nightly), // AArch64 guests on QEMU `virt`: local, because no CI runner boots one yet. ("virt_early_panic", Sched::Parallel, Tier::Local), ("virt_early_fault", Sched::Parallel, Tier::Local), @@ -613,11 +586,11 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // CPU the firmware named came up and none of their timestamp counters // trails the BSP's; the physical memory manager's accounting against the // firmware map balances to the byte; every ACPI table this kernel goes on - // to decode checksummed; the TSC the whole machine is timed by agrees with - // the frequency the part itself states; the PCI inventory's function count - // matches its rows; and one machine-wide TLB shootdown's cost is a - // distribution rather than one boot's average. Every verdict is arithmetic - // over records, with no clock in the judging, so all six are Parallel. + // to decode checksummed; the LAPIC timer and the TSC calibrated to a + // frequency; the PCI inventory's function count matches its rows; and one + // machine-wide TLB shootdown's cost is a distribution rather than one + // boot's average. Every verdict is arithmetic over records, with no clock + // in the judging, so all six are Parallel. ("smp_roster_and_tsc_trail", Sched::Parallel, Tier::Nightly), ("pmm_accounting", Sched::Parallel, Tier::Nightly), // ROOT is the loader's image in memory: the kernel says it mounted it @@ -628,26 +601,11 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("root_from_memory", Sched::Parallel, Tier::Nightly), ("root_withheld_refused", Sched::Parallel, Tier::Nightly), // The boot from power-on, as the kernel converts the loader's TSC readings: - // judged against the loader's raw counts and the kernel's own rate, and - // bounded above by the host's clock, which is Parallel-safe because load - // only widens that bound. + // judged against the loader's raw counts and the kernel's own rate. ("boot_from_power_on", Sched::Parallel, Tier::Nightly), ("acpi_table_inventory", Sched::Parallel, Tier::Nightly), ("timer_calibration", Sched::Parallel, Tier::Nightly), ("pci_inventory", Sched::Parallel, Tier::Nightly), - ("tlb_shootdown_cost", Sched::Parallel, Tier::Nightly), - // What a waiter in the real-time band pays to be woken, as a distribution - // over ten thousand programmed wakes — and, beside it, that the number - // reaches a machine with no serial port at all, through the kernel's own - // `exit:` record on the log volume. Nothing else in the tree measures wake - // latency against a programmed timer: soundd's figure is a maximum over a - // window, taken against a DLL's prediction of a DMA completion and needing - // a sound card to exist at all, and `toyos-sched`'s bound on the same - // quantity runs in a simulator where no IPI is ever delivered. - // - // Serial: it is the one registration here whose verdict is a *time*, and a - // wake latency measured beside eleven other guests is the host's schedule. - ("latency_wake", Sched::Serial, Tier::Nightly), ("smp_failed_ap_leaves_no_hole", Sched::Parallel, Tier::Weekly), ("input_merge", Sched::Parallel, Tier::Weekly), ("metal_sim_input", Sched::Parallel, Tier::Weekly), @@ -685,22 +643,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // Its own boot: the compositor it abuses has to be one nothing else has // touched. ("metal_sim_hostile_clipboard", Sched::Parallel, Tier::Fast), - // A host-measured drain rate with an 8 s ceiling on a 3.3 s expectation. - // Not gate A, but the same instrument: what it measures is how fast a - // client's audio leaves the machine. - ("metal_sim_null_audio", Sched::Serial, Tier::Nightly), - ("null_sink_shipped_client", Sched::Serial, Tier::Nightly), - // Parallel, and this one is argued rather than assumed: not a verdict in it - // is a wall-clock margin. The flood's size is asserted against the audio - // callback's own period counter standing still, both playback checks are - // counted in periods, and the capture is read for amplitude and never for - // timing. Its own boot, its own config, and the only client its soundd has. - ("doom_sound_flood", Sched::Parallel, Tier::Nightly), - // Reads a device capture and requires at least MIN_SIGNAL_SECS = 0.8 s of - // it to carry signal at peak >= 6000 — an absolute seconds-of-signal - // floor on audio recorded in real time, not a fraction of the capture and - // not compute-bound: timer-anchored. - ("doom_music", Sched::Parallel, Tier::Nightly), // One C program through the toolchain's clang, read by the loader's decoder, // and one boot to run it. ("c_hello", Sched::Parallel, Tier::Fast), @@ -708,10 +650,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // counted in game tics, whatever the host's speed. Fast, because it is the // gate on what the C compiler makes of doom. ("doom_frames", Sched::Parallel, Tier::Fast), - // A tone played while soundd's pipe to a stalled log is full. The verdicts - // are the capture's gaps and a count of lines; the clocks are liveness - // guards. - ("soundd_log_stall", Sched::Serial, Tier::Nightly), // ureq and rustls from crates.io, fetching over TLS 1.3 from a host this // test mints a CA for. Every verdict is a printed line or a digest; the // only clock is `run_test`'s ceiling. @@ -862,9 +800,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // verdict is the guest's byte-for-byte comparison; its clocks are // liveness guards. ("netd_held_open", Sched::Parallel, Tier::Nightly), - // The netcase boot again: a client out-writing a peer that stopped reading - // costs netd no CPU. - ("netd_stalled_peer", Sched::Parallel, Tier::Nightly), // The netcase boot again, beside a host UDP echo: a datagram the client's // pipe will not take whole ends that socket by name and no other. The // verdict is each socket's answer; its clocks are liveness guards. @@ -881,7 +816,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // The netcase boot, every frame it sends held once it has its lease: // lookups whose clients hung up or spoke again are let go at once, and // one nobody answers ends when its schedule does. The verdict is netd's - // answers; the schedule's end is a bound derived from it. + // answers. ("netd_lookup_let_go", Sched::Parallel, Tier::Nightly), // Two netcase boots, each frame put on the wire kept: the first DHCP // transaction ID of each differs, because netd seeds smoltcp's random @@ -907,11 +842,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("sshd_exec", Sched::Parallel, Tier::Nightly), ("sshd_files", Sched::Parallel, Tier::Nightly), ("sshd_key_auth", Sched::Parallel, Tier::Nightly), - // Serial: it measures netd's 2 s handshake deadline against the host's - // clock, and counts how many connections survived a 48 ms paced burst - // before that deadline could expire any of them. Both are wall-clock - // margins, which is the definition of [`Sched::Serial`]. - ("netd_hostile_peer", Sched::Serial, Tier::Weekly), + ("netd_hostile_peer", Sched::Parallel, Tier::Weekly), ("launcher_refusals", Sched::Parallel, Tier::Weekly), // Paths a child prints and the kernel's refusals by name; no clock in any of them. ("spawn_cwd", Sched::Parallel, Tier::Nightly), @@ -976,17 +907,13 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("loader_watchdog_arms", Sched::Parallel, Tier::Nightly), // Its own boot, and the verdict is QEMU's stop reason inside the bound. ("watchdog_resets", Sched::Parallel, Tier::Nightly), - // Serial: its verdict is that nothing happened for a span of host clock. - ("watchdog_fed", Sched::Serial, Tier::Weekly), // The panicked kernel's own bound, which is what ends a boot on a machine - // whose chipset timer does not count. Both verdicts are QEMU's stop reason - // against a bound the guest printed, so a slower machine moves both. + // whose chipset timer does not count. The verdict is QEMU's stop reason and + // the line saying the bound ran out. ("panic_reboots", Sched::Parallel, Tier::Nightly), // The same verdict from inside `percpu::init_bsp`: the earliest point a // panic is reportable, and the window the owner's T14 stops in. ("panic_before_peripherals_reboots", Sched::Parallel, Tier::Nightly), - // Serial like `watchdog_fed`: its verdict is that nothing happened for a span of host clock. - ("panic_key_holds", Sched::Serial, Tier::Weekly), // The boot chain's three answers. The two chain names each watch a guest // take its own reset and read the pass after it, so both are anchored to // the bound the first boot counts down. @@ -1045,15 +972,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // Its own boot, its own feature, and it drives the guest only through // stdin — nothing it touches is shared with another test. ("idle_stack_guard", Sched::Parallel, Tier::Weekly), - // Its own boot and its own feature, and it deafens one CPU for 400 ms — - // but the deafening is a *window*, and the verdict is whether the NMI is - // answered inside `NMI_BUDGET_NS`, which is one millisecond. That is a - // wall-clock margin on the host as much as on the guest: at width 12 the - // probe missed the window and reported the NMI as never delivered, which - // reads exactly like the defect it hunts, and it was green alone in the - // same run and three times after it. Serial by the default rule — a - // verdict that is a duration does not go in the parallel phase. - ("dump_nmi_probe", Sched::Serial, Tier::Weekly), // The same dump asked for inside the passes that may not serve it, on one // CPU. Parallel: every verdict is a line the guest prints or a count the guest // keeps, and no duration is in any of them. @@ -1069,7 +987,8 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // was brought up. No clock and no device in it. ("virtio_used_ring", Sched::Parallel, Tier::Weekly), // A fatal path with other CPUs running userland that makes kernel records: - // none is stamped past the fatal record by more than an IPI takes. + // none of theirs follows the fatal path's own line after the stop beyond + // the one each may have had in flight. Order, and no clock in it. ("panic_halts_the_others_first", Sched::Parallel, Tier::Nightly), // A kernel log line from PCI enumeration; no clock and no real device in it. ("pci_capability_walk", Sched::Parallel, Tier::Weekly), @@ -1121,22 +1040,8 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("gsbase_locked", Sched::Parallel, Tier::Weekly), // The fourth declared kernel build, booted so that the scheduler core's // `feature = "check"` instruments are compiled and executed by a CI run at - // all. One of its verdicts is a *quantile* of the guest's published - // pass-cost distribution, which is wall clock across a scheduler pass. - // - // **Serial, and it used to say `Parallel` for a reason that was wrong.** - // The old note read "a bound the guest measures against its own TSC inside - // a single scheduler pass, which no amount of host load lengthens. A pass - // is preempt-off by construction." Preempt-off stops the *guest's* - // scheduler and stops nothing above it: the guest's TSC advances while the - // host has the vCPU, which is why invariant P panicked on a KVM shard at - // 200569 ns and why it is a measurement now. Measured here, 2026-08-17, one - // suite: alone on a quiet host (1.02x the reference boot) cpu0 reports - // `168 passes, p50 < 16384 ns, p90 < 131072 ns` and passes; in the same - // run's 12-wide phase, `134 passes, p50 < 131072 ns, p90 < 262144 ns` and - // it reds. Host contention moves this guest's median by a factor of eight, - // which is the definition of a test that must have the machine to itself. - ("sched_check_build", Sched::Serial, Tier::Fast), + // all. + ("sched_check_build", Sched::Parallel, Tier::Fast), // What nesting a `scheduler::Operation` may and may not do, which is a law // with no host-side reader: the type reaches `percpu::cpu_id` and // `driver::current_handle`, so nothing outside a booted machine can @@ -1254,36 +1159,19 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("toolkit_window_wake", Sched::Parallel, Tier::Nightly), ("toolkit_winit_loop", Sched::Parallel, Tier::Nightly), ("toolkit_winit_pace", Sched::Parallel, Tier::Nightly), - // The same desktop with soundd behind it: an audio client spawned by a - // shell, which is the only place all three of its descriptors are pipes to - // a surface. Parallel — every verdict is a marker with its own ceiling, and - // none of them reads a clock. - ("desktop_audio_client", Sched::Parallel, Tier::Nightly), // Ctrl+Alt+D on the same machine. Parallel: it waits for a marker and its // verdicts are counts the report has to agree with itself about, not a // wall-clock margin — the one duration in it is the dump's own 250 ms // ceiling, which the guest spends and the host never measures. ("blocked_dump", Sched::Parallel, Tier::Nightly), - // Two boots of one machine compared on the guest's own `Boot: complete` - // with a 300 ms allowance, which is the whole assertion. - ("i8042_absent", Sched::Serial, Tier::Nightly), - // The fault quarantines (masks) the controller's GSI within milliseconds - // of readiness — confirmed from the serial log, before a host round trip - // could land anything — so no sentinel can ever reach the guest and the - // run necessarily pays `test_rs_i8042_keyboard`'s full fallback deadline. - // A fixed wall-clock window is the verdict's floor, not its cost: - // timer-anchored, and its price straddles the ceiling run to run (9,355 / - // 10,568 / 11,073 ms across three measurements) for exactly that reason. + ("i8042_absent", Sched::Parallel, Tier::Nightly), + // The fault quarantines (masks) the controller's GSI: the line and its + // count are the verdict, and no program runs. ("i8042_quarantine", Sched::Parallel, Tier::Nightly), ("i8042_budget_expiry", Sched::Parallel, Tier::Nightly), ("i8042_fadt_denial", Sched::Parallel, Tier::Weekly), ("i8042_kbd_echo", Sched::Parallel, Tier::Nightly), ("i8042_undecoded_bytes", Sched::Parallel, Tier::Fast), - // Its verdict is a cadence, and its absence is the assertion — both read - // off the guest's own `last byte at Nms` stamps. The gap it injects is - // 3 s against a 500 ms period, so six periods of margin decide whether the - // report is on the pin or on a timer. - ("i8042_health_cadence", Sched::Parallel, Tier::Nightly), ("xhci_xecp_walk", Sched::Parallel, Tier::Weekly), ("xhci_slot_exhaustion", Sched::Parallel, Tier::Weekly), ("usb_storage_gate", Sched::Parallel, Tier::Weekly), @@ -1301,12 +1189,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("usb_storage_write_error", Sched::Parallel, Tier::Weekly), ("usb_flush_optional", Sched::Parallel, Tier::Nightly), ("xhci_deaf_registers", Sched::Parallel, Tier::Weekly), - // The window is anchored on the controller's own port-power stamp now, not - // boot, so a slow boot no longer eats it — but the bound is still a fixed - // span of the guest's own TSC clock (`SLOW_CONNECT_NS`/`DEBOUNCE_NS`), and - // a host running several other guests can still stall this one's vCPU past - // that span for reasons that are not the defect. - ("xhci_slow_connect", Sched::Serial, Tier::Nightly), + ("xhci_slow_connect", Sched::Parallel, Tier::Nightly), ("xhci_portsc_rw1c", Sched::Parallel, Tier::Weekly), // One staged break and no other, which puts the driver's recovery finishing // on its first try in the verdict: a retried command that reaches an @@ -1333,10 +1216,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("log_flush_retry", Sched::Parallel, Tier::Nightly), ("toybox_cp_volume", Sched::Parallel, Tier::Weekly), ("kernel_log_file", Sched::Parallel, Tier::Nightly), - // Serial: its verdict is a cadence — heartbeats against a 250 ms period — - // and a guest sharing the host with eleven others reaches its idle loop - // late for reasons that are not the defect. - ("kernel_heartbeat", Sched::Serial, Tier::Nightly), + ("kernel_heartbeat", Sched::Parallel, Tier::Nightly), // The five RTC/firmware shapes, one kernel build and one boot each. Five // registrations because the artifact memo builds one kernel per feature // set anyway, so the split costs nothing and the parallel phase gets five @@ -1442,14 +1322,8 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // standing; then, on a second boot, a function no release resets is never // lent where it was left aimed. ("blockd_lends_within_its_bound", Sched::Parallel, Tier::Weekly), - // H4: soundd driving an Intel HDA controller itself, read back off the - // device. Serial — its verdict is a wav capture, and one taken while eleven - // other guests contend for the host measures the host. - ("hda_tone", Sched::Serial, Tier::Nightly), - // The T14's panic, staged: a client that stops producing for longer than - // the DMA ring takes to come round. The verdict is soundd's own liveness - // and its counters rather than a capture, so it runs wide. - ("hda_client_stall", Sched::Parallel, Tier::Nightly), + // Two live HDA links, refused by name: the negative control on the kernel's + // bind path. No sample is played; lines are the verdict. ("hda_two_live_refused", Sched::Parallel, Tier::Weekly), ]; @@ -1493,12 +1367,10 @@ const CARRIES: &[(&str, &[&str])] = &[ ("heap_ceiling_bounds", &["test_rs_heap_ceiling"]), ("cache_eviction", &["test_rs_cache_eviction"]), ("irq_census_conservation", &["test_rs_std_mmap"]), - ("i8042_health_cadence", &["test_rs_i8042_keyboard"]), ("i8042_health", &["test_rs_i8042_keyboard"]), ("i8042_fadt_denial", &["test_rs_i8042_keyboard"]), ("i8042_kbd_echo", &["test_rs_i8042_keyboard"]), ("i8042_undecoded_bytes", &["test_rs_i8042_keyboard"]), - ("i8042_quarantine", &["test_rs_i8042_keyboard"]), ("i8042_no_spurious_wake", &["test_rs_i8042_keyboard"]), ("i8042_mouse", &["test_rs_i8042_mouse"]), ("swiss_german_layout", &["test_rs_locale_gate"]), @@ -1510,7 +1382,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("netd_refused_pipes", &["test_rs_netd_refused_pipes"]), ("netd_refused_accept", &["test_rs_netd_refused_accept"]), ("netd_held_open", &["test_rs_netd_held_open"]), - ("netd_stalled_peer", &["test_rs_netd_stalled_peer"]), ("netd_udp_refused", &["test_rs_netd_udp_refused"]), ("netd_udp_any_address", &["test_rs_netd_udp_any_address"]), ("netd_lookup_let_go", &["test_rs_netd_lookup_let_go"]), @@ -1528,7 +1399,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("blockd_lends_within_its_bound", &["test_rs_blockd_io"]), ( "inspect_reads_its_owners", - &["test_rs_inspect_denied", "test_rs_inspect_plays", "test_rs_inventory_bounds"], + &["test_rs_inspect_denied", "test_rs_inventory_bounds"], ), ("metal_sim_compositor", METAL_SIM_CLIENTS), ("metal_sim_scanout_wc", METAL_SIM_CLIENTS), @@ -1542,15 +1413,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("toolkit_window_wake", &["test_rs_window_wake"]), ("toolkit_winit_loop", &["test_rs_winit_loop"]), ("toolkit_winit_pace", &["test_rs_winit_pace"]), - ("doom_sound_flood", &["test_rs_doom_sound_flood"]), - ("doom_music", &["test_rs_doom_music"]), ("doom_frames", &["test_rs_doom_frames"]), - ("soundd_log_stall", &["test_rs_soundd_log_stall"]), - ("metal_sim_null_audio", &["test_rs_audio_tone"]), - ("null_sink_shipped_client", &["test_rs_null_sink_client_exits"]), - ("hda_tone", &["test_rs_audio_tone"]), - ("hda_client_stall", &["test_rs_hda_client_stall"]), - ("latency_wake", &["test_rs_cyclictest", "test_rs_sched_stress"]), ("smp_failed_ap_leaves_no_hole", &["test_rs_smp_hole_shootdown"]), ("sshd_exec", &["test_rs_empty_dir_stat"]), ("sshd_files", &["test_rs_empty_dir_stat"]), @@ -1582,7 +1445,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("lock_across_switch_halts", &["test_rs_test_panic_child"]), ("heap_over_ceiling_halts", &["test_rs_test_panic_child"]), ("dump_left_pending_is_owed", &["test_rs_dump_stage_load"]), - ("dump_nmi_probe", &["test_rs_dump_stage_load"]), ("syscall_window_nmi", &["test_rs_nmi_window_spin"]), ("syscall_window_nmi_controls", &["test_rs_nmi_window_spin"]), ("partition_claim", &["test_rs_partition_claimant"]), @@ -1747,14 +1609,64 @@ const METAL: &[(&str, metal::Metal)] = &[ }, ), ( - // The shipped tone client plays to completion and exits 0. Its QEMU - // registration calls the sink a null one; on the T14 whether soundd - // binds the laptop's own HDA controller is unmeasured, so what this - // asserts here is the weaker and truer thing — the client came back. - "null_sink_shipped_client", + "wake_storm_cost", metal::Metal::Runs { arms: TESTCASES, - judge: |b| b[0].job_passed("test_rs_null_sink_client_exits"), + judge: |b| b[0].job_passed("test_rs_wake_storm_cost"), + }, + ), + ( + // The shipped tone client, twice in series, plays to completion and + // exits 0, and soundd names how each left. + "shipped_client_departures", + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| { + b[0].job_passed("test_rs_null_sink_client_exits")?; + audio::departures_on_metal(&b[0].log()) + }, + }, + ), + ( + // soundd with no client costs no CPU, before any client has connected. + "audio_idle_suspend", + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| { + b[0].job_passed("test_rs_audio_idle_suspend")?; + audio::idle_suspend_on_metal(&b[0].log()) + }, + }, + ), + ( + "hda_tone", + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| { + b[0].job_passed("test_rs_audio_tone")?; + audio::tone_on_metal(&b[0].log()) + }, + }, + ), + ( + "hda_client_stall", + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| { + b[0].job_passed("test_rs_hda_client_stall")?; + audio::client_stall_on_metal(&b[0].log()) + }, + }, + ), + // ---- the audio boot of its own ---- + ( + "soundd_log_stall", + metal::Metal::Runs { + arms: LOGSTALLCASE, + judge: |b| { + b[0].job_passed("test_rs_soundd_log_stall")?; + audio::log_stall_on_metal(&b[0].log()) + }, }, ), // ---- one image: tests/testcases armed with the chipset watchdog ---- @@ -1777,6 +1689,75 @@ const METAL: &[(&str, metal::Metal)] = &[ }, }, ), + ( + // The two numbers `syscall_cost` measures, printed by the job and read + // off the stick: that it ran, and what it said, are the verdict. + "syscall_cost", + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| { + b[0].job_passed("test_rs_syscall_cost")?; + b[0].log().must_say("syscall_cost: tsc ")?; + b[0].log().must_say(" cycles/syscall over ")?; + Ok(()) + }, + }, + ), + ( + // `SYS_DEBUG` holds each other CPU's shootdown acknowledgement back, + // so the kernel that carries it; the verdict is the guest's own exit. + "tlb_shootdown_waits", + metal::Metal::Runs { + arms: &[metal::Arm { + features: toyos_build::build::TEST_KERNEL, + ..metal::once("testcases-debug", "tests/testcases", &[], &["test_rs_tlb_shootdown_waits"]) + }], + judge: |b| b[0].job_passed("test_rs_tlb_shootdown_waits"), + }, + ), + ( + // The canary beside the `watch-window` actuator, and the count of + // windows a post landed in while it ran — the staging that is + // timing, and so metal's. + "blocking_read_window", + metal::Metal::Runs { + arms: &[metal::once( + "testcases-window", + "tests/testcases", + &["watch-window"], + &["test_rs_blocking_read_stress"], + )], + judge: |b| { + b[0].job_passed("test_rs_blocking_read_stress")?; + window_held_on_metal(&b[0].kernel()) + }, + }, + ), + ( + // One CPU deafened by the actuator, named by the blocked-task dump and + // found by its NMI where it spins. + "dump_nmi_probe", + metal::Metal::Runs { + // Held open by a job, because an empty list ends the boot before the + // actuator arms. + arms: &[metal::once( + "testcases-deaf", + "tests/testcases", + &["dump-deaf-cpu"], + &["test_rs_lan_hold"], + )], + judge: |b| faults::dump_nmi_probe_on_metal(&b[0].kernel()), + }, + ), + ( + // A fed watchdog: the armed boot runs its whole list to its own stop, + // which a chipset reset anywhere in it would have cut short. + "watchdog_fed", + metal::Metal::Runs { + arms: &[metal::once("testcases-watchdog", "tests/testcases", &["watchdog"], &[])], + judge: |b| b[0].log_reached_the_stick(), + }, + ), // ---- the boot facts, riding the same tests/testcases image ---- // // **Six names and no extra minute of the machine.** Every needle below is a @@ -1804,7 +1785,13 @@ const METAL: &[(&str, metal::Metal)] = &[ ), ( "timer_calibration", - metal::Metal::Runs { arms: TESTCASES, judge: |b| timer_calibration(b[0].kernel().text()) }, + metal::Metal::Runs { + arms: TESTCASES, + judge: |b| { + timer_calibration(b[0].kernel().text())?; + tsc_agrees_with_cpuid(b[0].kernel().text()) + }, + }, ), ( "pci_inventory", @@ -1827,9 +1814,6 @@ const METAL: &[(&str, metal::Metal)] = &[ }, ), ( - // The console half of the QEMU registration does not exist here, so - // what is judged is the half that does: the exit record carries the - // p99, and the stress suite beside it carries a verdict. "latency_wake", metal::Metal::Runs { arms: LATENCYCASE, @@ -2021,24 +2005,60 @@ const METAL: &[(&str, metal::Metal)] = &[ judge: |b| operation_nesting_log(b[0].kernel().text()), }, ), - // ---- named, looked at, and not run there ---- - ( - "metal_sim_null_audio", - metal::Metal::QemuOnly( - "its subject is soundd's null sink, and whether the T14's own HDA controller binds \ - is unmeasured; both halves of the verdict — soundd's counters and a host-timed \ - drain — are console text and a host clock, neither of which the stick carries", - ), - ), ]; /// **The [`METAL`] rows no QEMU registration answers for**, each with why none /// can: a verdict under a name only the T14 reports. -const METAL_ONLY: &[(&str, &str)] = &[( - "lan_message_delivery", - "whether the T14's own I219 delivers a message through that machine's interrupt remapping \ - is a fact of that part and that path; QEMU's e1000e is another part behind another path", -)]; +const METAL_ONLY: &[(&str, &str)] = &[ + ( + "lan_message_delivery", + "whether the T14's own I219 delivers a message through that machine's interrupt \ + remapping is a fact of that part and that path; QEMU's e1000e is another part behind \ + another path", + ), + ( + "wake_storm_cost", + "its verdict is that a wake storm's cost grows linearly with the waiters, read off the \ + TSC around the syscall, and a guest's TSC runs while its host has the vCPU", + ), + ("shipped_client_departures", AUDIO_ON_METAL_ONLY), + ("audio_idle_suspend", AUDIO_ON_METAL_ONLY), + ("hda_tone", AUDIO_ON_METAL_ONLY), + ("hda_client_stall", AUDIO_ON_METAL_ONLY), + ("soundd_log_stall", AUDIO_ON_METAL_ONLY), + ( + "tlb_shootdown_cost", + "its product is how long a machine-wide shootdown takes, a span a guest's host \ + sets", + ), + ( + "latency_wake", + "its product is how late a programmed wake lands, a span a guest's host sets", + ), + ( + "syscall_cost", + "its product is a cycle count per syscall and the clock rate beside it, which a \ + guest's host sets", + ), + ( + "tlb_shootdown_waits", + "its verdict is a lower bound on a span the guest reads off its own clock around the \ + syscall", + ), + ( + "dump_nmi_probe", + "its verdict rests on a CPU the actuator deafens for a window of its own clock being \ + kicked and probed inside it, which a starved guest misses", + ), + ( + "watchdog_fed", + "its verdict is that a feeding kernel is not reset, which only a span of the \ + machine's own time can show", + ), +]; + +/// Why an audio row has no QEMU arm. +const AUDIO_ON_METAL_ONLY: &str = "audio is judged on the T14 and in no QEMU guest (owner ruling)"; /// The boot most of the first tranche rides: the plain `tests/testcases` shape /// with a job list that ends it. @@ -2051,7 +2071,17 @@ const TESTCASES: &[metal::Arm] = &[metal::once( "testcases", "tests/testcases", &[], - &["test_rs_abuse_short_sleep", "test_rs_null_sink_client_exits", "log-close"], + &[ + "test_rs_wake_storm_cost", + // Before any client connects, which is `audio_idle_suspend`'s premise. + "test_rs_audio_idle_suspend", + "test_rs_audio_tone", + "test_rs_hda_client_stall", + "test_rs_abuse_short_sleep", + "test_rs_syscall_cost", + "test_rs_null_sink_client_exits", + "log-close", + ], )]; /// **Two boots of one config, because these two cannot share one.** Each fills @@ -2067,6 +2097,10 @@ const TESTCASES_READDIR: &[metal::Arm] = const JOBCASE: &[metal::Arm] = &[metal::once("jobcase", "tests/jobcase", &[], &[])]; +/// A `logd` that leaves soundd's ring unread until the job says the tone played. +const LOGSTALLCASE: &[metal::Arm] = + &[metal::once("logstallcase", "tests/logstallcase", &[], &["test_rs_soundd_log_stall"])]; + /// The two boots the reset ruling is judged on, and both are boots this suite /// already flashes: the device boot for a reset with megabytes behind it, and /// `jobcase` for one with nothing. @@ -2644,9 +2678,7 @@ fn discover_rust_tests(bins: &[(String, Vec)]) -> Vec { if name.ends_with(".so") { return None; } - if RUST_SKIP.contains(&name.as_str()) - || AUDIO_TESTS.iter().any(|(audio, _)| *audio == name) - { + if RUST_SKIP.contains(&name.as_str()) { return None; } Some(name.clone()) @@ -2897,33 +2929,18 @@ fn check_c_result(result: &TestResult) -> bool { } } +/// Every red prints the test's own lines, a ceiling's too: a hang is named by +/// the last thing the test said. fn check_rust_result(result: &TestResult) -> bool { let test_name = result.name.strip_prefix("test_rs_").unwrap_or(&result.name); - - if let Some(err) = &result.error { - eprintln!("FAIL rs::{test_name}: {err}\nstdout:\n{}{}", result.stdout, kernel_account(result)); - return false; - } - - match result.exit_code { - Some(0) => true, - Some(code) => { - eprintln!( - "FAIL rs::{test_name}: exit code {code}\nstdout:\n{}{}", - result.stdout, - kernel_account(result) - ); - false - } - None => { - eprintln!( - "FAIL rs::{test_name}: no exit code\nstdout:\n{}{}", - result.stdout, - kernel_account(result) - ); - false - } - } + let why = match (&result.error, result.exit_code) { + (None, Some(0)) => return true, + (Some(err), _) => err.to_string(), + (None, Some(code)) => format!("exit code {code}"), + (None, None) => "no exit code".to_string(), + }; + eprintln!("FAIL rs::{test_name}: {why}\nstdout:\n{}{}", result.stdout, kernel_account(result)); + false } /// The kernel names the frames of a process it loaded off a **disk**. @@ -3033,65 +3050,6 @@ fn check_ring0_read_unmapped(serial: &str) -> Result<(), String> { Ok(()) } -/// A zero CPU delta is the signature of a suspended soundd and equally of one -/// wedged with the device running, so the counter the test reads cannot tell -/// them apart on its own. The serial can: in a window where no audio client -/// ever connects, the PCM stream has no business starting. -/// -/// This reads only `result.serial`, which begins at ===TEST_START, so a -/// device started before then (a restored boot prime) is invisible to this -/// particular check — not because the harness cannot see it: `qemu.boot_log()` -/// holds it, which is what `audio::check_suspend_structure` concatenates in -/// ahead of its own window. What this one does catch is a start inside its -/// window with no client to justify it: soundd's `!streams.is_empty()` -/// fill-loop gate going away, or a resume fired by anything other than a -/// connect. -fn check_audio_idle_suspend(result: &TestResult) -> bool { - if !check_rust_result(result) { - return false; - } - const STARTED: &str = "virtio-sound: stream 0 started"; - if result.serial.contains(STARTED) { - eprintln!( - "FAIL rs::audio_idle_suspend: `{STARTED}` with no client connected — \ - soundd's zero CPU is the device left running, not a suspend\nserial:\n{}", - result.serial - ); - return false; - } - true -} - -/// Two clients through the null sink, and what soundd said about each leaving. -/// -/// The exit code already says both `/system/bin/tone` runs finished cleanly — that is -/// the test's own assertion — so this window is exactly the case soundd used to -/// misreport: `client N died` for a process that exited `code=0`, because the -/// mix loop's signal pipe broke before the control thread read the peer. What -/// it asserts is that neither outcome of that race is worded as a death, and -/// that both removals name a departure soundd actually established. -/// -/// **The count is per removal and stays exact**, because the vocabulary is -/// asserted per removal: a capture where no client ever left would satisfy -/// every check above it vacuously, and a range would let the second removal go -/// missing again. What used to make that count a race was the window and not -/// the number — see [`settle_null_sink_client_exits`], which is what closes it. -fn check_null_sink_client_exits(result: &TestResult) -> bool { - if !check_rust_result(result) { - return false; - } - let problems = audio::check_departures(&result.serial, NULL_SINK_CLIENTS); - if !problems.is_empty() { - eprintln!( - "FAIL rs::null_sink_client_exits: {}\nserial:\n{}", - problems.join("; "), - result.serial - ); - return false; - } - true -} - /// The exit code says the child died; only the serial says *why*. /// /// A #DE with no gate escalates to #DF, and `double_fault_handler` halts every @@ -3197,79 +3155,10 @@ fn check_debug_trap(result: &TestResult) -> bool { ok } -/// The clients `test_rs_null_sink_client_exits` runs in series, and so the -/// number of removals soundd owes. One constant, because the wait below and the -/// count above have to be the same number or the wait is for something else. -const NULL_SINK_CLIENTS: usize = 2; - -/// Wait for soundd to report the second client leaving, on the guest's liveness. -/// -/// **The last removal arrives after the process whose exit produced it**, and -/// that process exiting is what ends the capture: round 1's line makes it in -/// because a whole second round follows it, and round 2's has nothing behind it -/// but `===TEST_END===`. Counting two removals over that window is an assertion -/// about scheduling, and it went red on CI twice on documentation-only branches -/// — `soundd reported 1 client removals, expected 2` — with the capture showing -/// the line never arriving rather than arriving wrong. -/// -/// The wait is [`await_guest`]'s: it ends when the removals are there, or when -/// the guest stops making progress, and never on a span of host wall clock. It -/// costs nothing on a run that already had both lines — the predicate is checked -/// before anything is drained, which was true of 6 of 6 measured runs on the dev -/// host — and its expiry is not a verdict, which is why the error is dropped: -/// what fails this test is still the count, in `check_departures`'s own words. -/// -/// [`audio::SOUNDD_GONE`] ends it too, for the same reason `await_null_sink` -/// reads that line: soundd exiting is a removal that is never coming, and the -/// test should say so in its own sentence rather than wait out the guard. The -/// guard is the whole of [`qemu::GUEST_WEDGED`] here and not the quiet bound — -/// this boot's kernel prints on a 10 s cadence, so the machine is never silent -/// for the 15 s that would end the wait early (measured: an unreachable -/// predicate takes 302 s). That price is paid only by a run where soundd is -/// alive and has genuinely stopped reporting departures, which is the defect -/// this test exists for. -fn settle_null_sink_client_exits(qemu: &mut QemuInstance, result: &mut TestResult) { - let mut serial = std::mem::take(&mut result.serial); - let _ = await_guest(qemu, &mut serial, "soundd to report both clients leaving", |seen| { - audio::departures(seen).len() >= NULL_SINK_CLIENTS || seen.contains(audio::SOUNDD_GONE) - }); - result.serial = serial; -} - /// Nothing to wait for: the test's own window carries everything its check -/// reads. Every name but two. +/// reads. Every name but one. fn no_settle(_: &mut QemuInstance, _: &mut TestResult) {} -fn spawned_pid(log: &str, name: &str) -> Option { - let want = format!("/system/bin/test_rs_{name} "); - log.lines() - .rev() - .find(|l| l.contains("spawn: ") && l.contains(&want))? - .split("pid=") - .nth(1)? - .split_whitespace() - .next()? - .parse() - .ok() -} - -/// Wait for the kernel to account the process: the capture closes when the guest -/// runner reaps the child, while `syscalls: pid=N` comes from -/// `teardown_resources`, which `deferred-release-outlives-its-syscall.md` -/// records as able to run after the syscall that caused it returned. -fn settle_syscall_cost(qemu: &mut QemuInstance, result: &mut TestResult) { - /// A liveness ceiling and never a verdict. - const ACCOUNTED: Duration = Duration::from_secs(5); - - let Some(pid) = spawned_pid(&result.serial, "syscall_cost") else { return }; - let want = accounting_of(pid); - if result.serial.contains(&want) { - return; - } - let more = qemu.drain_until(ACCOUNTED, |l| l.contains(&want)); - result.serial.push_str(&more); -} - /// The trailing space is what keeps `pid=21` from matching `pid=212`. fn accounting_of(pid: u32) -> String { format!("syscalls: pid={pid} ") @@ -3279,8 +3168,6 @@ fn accounting_of(pid: u32) -> String { /// selects the check. fn settle_for(name: &str) -> fn(&mut QemuInstance, &mut TestResult) { match name { - "null_sink_client_exits" => settle_null_sink_client_exits, - "syscall_cost" => settle_syscall_cost, "exit_wait_storm" => settle_exit_wait_storm, _ => no_settle, } @@ -3290,13 +3177,10 @@ fn settle_for(name: &str) -> fn(&mut QemuInstance, &mut TestResult) { fn check_for(name: &str) -> fn(&TestResult) -> bool { match name { "disk_backtrace" => check_disk_backtrace, - "audio_idle_suspend" => check_audio_idle_suspend, - "null_sink_client_exits" => check_null_sink_client_exits, "fault_gates" => check_fault_gates, "debug_trap" => check_debug_trap, "dlopen_dedup" => check_dlopen_dedup, "abuse_elf_loader" => check_abuse_elf_loader, - "syscall_cost" => check_syscall_cost, "exit_wait_storm" => check_exit_wait_storm, _ => check_rust_result, } @@ -3363,81 +3247,6 @@ fn check_dlopen_dedup(result: &TestResult) -> bool { true } -/// The two numbers `syscall_cost` measures, read back and recorded. No -/// threshold — a TCG cycle count prices nothing; what is owed is that the -/// measurement happened, which is what its guest header states. -fn check_syscall_cost(result: &TestResult) -> bool { - if !check_rust_result(result) { - return false; - } - let field = |prefix: &str| -> Option { - result.stdout.lines().find_map(|l| { - l.trim().strip_prefix(prefix)?.split_whitespace().next()?.parse::().ok() - }) - }; - let (Some(cycles), Some(mhz)) = (field("syscall_cost: "), field("syscall_cost: tsc ")) else { - eprintln!( - "FAIL rs::syscall_cost: the run reported no cycles/syscall and MHz pair\nstdout:\n{}", - result.stdout - ); - return false; - }; - // A counter that did not move is what a measurement can be wrong about silently. - if cycles == 0 || mhz == 0 { - eprintln!( - "FAIL rs::syscall_cost: {cycles} cycles/syscall at {mhz} MHz — twenty thousand \ - transitions cost no cycles on a clock that did not tick\nstdout:\n{}", - result.stdout - ); - return false; - } - // And the workload happened, against a counter this test cannot reach: - // `SYS_GETPID` is 51 in the kernel's per-syscall accounting at process exit. - let Some(claimed) = result.stdout.lines().find_map(|l| { - let (reps, per) = l.split_once(" over ")?.1.split_once('x')?; - Some(reps.trim().parse::().ok()? * per.trim().parse::().ok()?) - }) else { - eprintln!( - "FAIL rs::syscall_cost: the run did not say how many syscalls it made\nstdout:\n{}", - result.stdout - ); - return false; - }; - // By pid, and absence is its own failure: read over every line with - // `unwrap_or(0)` behind it, a late line was a process that made no calls. - let want = spawned_pid(&result.serial, "syscall_cost").map(accounting_of); - let line = want - .as_ref() - .and_then(|w| result.serial.lines().find(|l| l.contains(w.as_str()))); - let Some(line) = line else { - eprintln!( - "FAIL rs::syscall_cost: the kernel never accounted the process — no `{}` line \ - reached the capture, so nothing here says whether it made the calls it claims{}", - want.as_deref().unwrap_or("syscalls: pid="), - kernel_account(result) - ); - return false; - }; - // Absent `51=` is a process that made no `SYS_GETPID` calls: a real zero. - let counted = line - .split(" 51=") - .nth(1) - .and_then(|r| r.split_whitespace().next()) - .and_then(|n| n.parse::().ok()) - .unwrap_or(0); - if counted < claimed { - eprintln!( - "FAIL rs::syscall_cost: the run claims {claimed} SYS_GETPID transitions and the \ - kernel counted {counted}\nstdout:\n{}{}", - result.stdout, - kernel_account(result) - ); - return false; - } - eprintln!(" [syscall] {cycles} cycles per SYS_GETPID over {counted} of them, tsc {mhz} MHz"); - true -} - /// The name, formatted into every needle below rather than written beside a /// `test_rs_` literal: `suite_split` reads that spelling as a machine test /// *driving* the binary, and these only read console lines about it. @@ -3463,8 +3272,7 @@ fn storm_log(result: &TestResult) -> String { } /// The parent is the lowest pid among the storm's `spawn:` lines: pids are -/// never reused and it is made before any child. `spawned_pid` reads the last -/// and would answer with a child. +/// never reused and it is made before any child. fn storm_parent(log: &str) -> Option { let want = format!("/system/bin/test_rs_{STORM} "); log.lines() @@ -3477,16 +3285,18 @@ fn storm_parent(log: &str) -> Option { /// teardown line was emitted before it, so its arrival is what says the /// capture this check reads is whole. fn settle_exit_wait_storm(qemu: &mut QemuInstance, result: &mut TestResult) { - /// A liveness ceiling and never a verdict. - const ACCOUNTED: Duration = Duration::from_secs(5); - let Some(pid) = storm_parent(&storm_log(result)) else { return }; let want = accounting_of(pid); if storm_log(result).contains(&want) { return; } - let more = qemu.drain_until(ACCOUNTED, |l| l.contains(&want)); - result.serial.push_str(&more); + // A line that never comes is the ceiling's red, with its reason on the + // result the check then answers with. + if let Err(why) = await_guest(qemu, &mut result.serial, "the storm parent's accounting line", |c| { + c.contains(&want) + }) { + result.error = Some(qemu::WaitVerdict::new(why, &[&result.before, &result.serial])); + } } /// `exit_wait_storm` against the kernel's own record of the same run: the codes @@ -3564,732 +3374,8 @@ fn check_exit_wait_storm(result: &TestResult) -> bool { true } -/// Minimum active (non-silent) playback the 3s test tone must produce. -/// Guards against a vacuous pass when nothing plays at all. -const TONE_MIN_ACTIVE_SECS: f64 = 2.5; -/// The tone is generated at amplitude 16000; a far lower peak proves the -/// signal path is broken even if technically "active". -const TONE_MIN_PEAK: i32 = 4000; - -/// Recorded per-(test, smp) baselines — gate A's thorough tier. -/// Two independent instruments per config: -/// the wav underrun histogram (`gaps`, keyed by gap length in device periods) -/// and ceilings on soundd's own counters. The wav is a rare-event detector; -/// the counters fire on nearly every run and carry the statistical power. Both -/// must hold. Re-record deliberately, never casually — and justify every -/// number in `tests/audio-baseline.toml` itself. -#[derive(serde::Deserialize)] -#[serde(deny_unknown_fields)] -struct AudioBaselineEntry { - #[serde(default)] - gaps: BTreeMap, - max_wake_lat_us: u64, - drains: u32, - underruns: u32, - sample: BaselineSample, -} - -/// The recorded clean-tree *sample* for one config, not a summary of it. The -/// thorough tier compares a fresh sample against this one, so it needs the -/// observations themselves — see `tests/common/stats.rs` for why a summary -/// would understate the false-red rate. -#[derive(serde::Deserialize)] -#[serde(deny_unknown_fields)] -struct BaselineSample { - /// Runs whose wav was analysed (the counter arrays can be longer: a run - /// can lose its histogram and still report counters). - gap_sample: u32, - /// Of `gap_sample`, how many showed at least one mid-tone dropout. - gap_runs: u32, - /// Of the counter runs, how many breached this config's per-run ceilings. - ceiling_runs: u32, - max_wake_lat_us: Vec, - underruns: Vec, - wakes: Vec, - /// Recorded for re-baselining the per-run ceiling only. Deliberately not - /// tested distributionally: it is zero on 50-90% of runs, and the ties - /// leave a rank test with no power (measured: 0.00-0.21 against a tripling). - drains: Vec, -} - -type AudioBaseline = BTreeMap>; - -struct ConfigBaseline<'a> { - gaps: BTreeMap, - counters: audio::CounterLimits, - sample: &'a BaselineSample, -} - -fn load_audio_baseline() -> AudioBaseline { - let path = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/audio-baseline.toml"); - let text = fs::read_to_string(&path) - .unwrap_or_else(|e| panic!("read {}: {e}", path.display())); - toml::from_str(&text).unwrap_or_else(|e| panic!("parse {}: {e}", path.display())) -} - -/// Baseline for one (test, smp) config. Every config must be recorded: an -/// ungated config would pass by omission. -fn config_baseline<'a>(baseline: &'a AudioBaseline, name: &str, smp: u32) -> ConfigBaseline<'a> { - let entry = baseline - .get(name) - .and_then(|per_smp| per_smp.get(&format!("smp{smp}"))) - .unwrap_or_else(|| panic!("audio-baseline.toml: no [{name}.smp{smp}] section")); - ConfigBaseline { - sample: &entry.sample, - gaps: entry - .gaps - .iter() - .map(|(k, &count)| { - let periods: u32 = k.parse().unwrap_or_else(|_| { - panic!("audio-baseline.toml: bad gap key {k:?} for {name} smp{smp}") - }); - (periods, count) - }) - .collect(), - counters: audio::CounterLimits { - max_wake_lat_us: entry.max_wake_lat_us, - drains: entry.drains, - underruns: entry.underruns, - }, - } -} - -/// What one audio boot measured. Both tiers are computed from this; they -/// differ only in how many they collect and what decision they take on the -/// collection. -struct AudioRun { - gaps: BTreeMap, - counters: audio::SounddCounters, - /// The instrument itself is untrustworthy on this run (no tone, no dither, - /// clicks, no stats window). Never a rare-event judgement — always fatal, - /// in both tiers. - broken: Vec, - /// soundd counters past this config's per-run ceilings. A counted rate in - /// the thorough tier; printed but not a verdict in the fast tier, which - /// judges `harm` instead. - breaches: Vec, - /// What else the host was doing while this boot was measured. Annotation - /// only — nothing above or below branches on it. - host: hostload::HostLoad, -} - -impl AudioRun { - /// The capture's verdict alone. The thorough tier's dropout *rate* is - /// defined on this and nothing else, because that is what the recorded - /// sample counted. - fn dropped_audio(&self) -> bool { - !self.gaps.is_empty() - } - - /// Silence that reached the device on this run: a mid-tone gap in the - /// capture, or a period soundd put on the wire with no client audio behind - /// it. Both are audio someone would have heard drop out, and together they - /// are the fast tier's whole verdict — a counter past a ceiling says the - /// pipeline came close, and how close is a question for a distribution. - fn harm(&self) -> Option { - let mut evidence = Vec::new(); - if self.dropped_audio() { - evidence.push(format!("dropout {}", audio::format_histogram(&self.gaps))); - } - if self.counters.underruns > 0 { - evidence.push(format!( - "{} of {} periods submitted with no client audio", - self.counters.underruns, self.counters.submitted - )); - } - (!evidence.is_empty()).then(|| evidence.join(", ")) - } -} - -/// `--slow-usb`: give every audio boot a USB stick that answers a bulk transfer -/// in 2 ms instead of microseconds — what a real stick's erase block does, and -/// what the T14's audio pops are made of. -/// -/// A switch and not a test of its own, because it changes no verdict: it makes -/// the four audio configs measure a machine the host cannot otherwise present, -/// and what it produces is an A/B against the same command without it in the -/// same session. `issues/kernel/every-wait-in-this-kernel-is-a-spin.md` is what -/// the numbers are for. -static SLOW_USB: std::sync::atomic::AtomicBool = std::sync::atomic::AtomicBool::new(false); - -/// Boot a fresh QEMU with the given CPU count, run one in-guest audio test, -/// and measure it: soundd's in-guest counters (wake lateness, pipeline drains, -/// periods of silence submitted) and the captured wav (mid-signal silence, hard -/// sample-to-sample discontinuities, and the dither the detector needs to see -/// anything at all). -/// -/// `Err` means the run produced no measurement — a boot failure, a timeout, an -/// unreadable capture. That is never a rare-event judgement call; it is fatal -/// in both tiers. -fn measure_audio_run( - name: &str, - smp: u32, - baseline: &ConfigBaseline, - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], - // Distinguishes this boot from the others of the same config in the log - // and in the kept capture's filename; empty for a plain single boot. - tag: &str, -) -> Result { - let label = if tag.is_empty() { - String::new() - } else { - format!("{tag}: ") - }; - // Bounds every duration soundd can report: its whole life is inside this - // process's. See `audio::check_physical`. - let job = format!("test_rs_{name}"); - let carried = qemu::carrying(c_bins, rust_bins, [job.as_str()]); - let run_start = std::time::Instant::now(); - let mut qemu = QemuInstance::boot_with_options( - test_config, - &carried.c, - &carried.rust, - BootOptions { - smp, - kernel_params: if SLOW_USB.load(std::sync::atomic::Ordering::Relaxed) { - &["usb-slow-device"] - } else { - &[] - }, - ..Default::default() - }, - ); - - let result = qemu.run_test(&job, Duration::from_secs(30)); - if let Some(err) = &result.error { - return Err(err.to_string()); - } - match result.exit_code { - Some(0) => {} - Some(code) => return Err(format!("exit code {code}\nstdout:\n{}", result.stdout)), - None => return Err(format!("no exit code\nstdout:\n{}", result.stdout)), - } - - // The wav timeline advances in real time; give the tone tail and its - // trailing silence context time to reach the file before reading it. The - // same wait collects soundd's final stats flush, which races the client's - // exit and so can arrive after ===TEST_END===. - // - // Boot prepended so `check_suspend_structure` can see a device started - // before ===TEST_START — the boot capture exists (`qemu.boot_log()`), - // where its doc comment used to say it did not. - let serial = - qemu.boot_log().to_string() + &result.serial + &qemu.drain_serial(Duration::from_millis(500)); - - let wav = audio::parse_wav(qemu.audio_wav_path())?; - let analysis = audio::analyze(&wav); - let rate = wav.sample_rate as f64; - let secs = |samples: usize| samples as f64 / rate; - - // Always printed, so every run leaves comparable numbers in the log. - let gaps = audio::gap_histogram(&analysis, wav.sample_rate); - let counters = audio::parse_soundd_counters(&serial)?; - // Sampled here rather than before the boot because the load averages are - // trailing: a reading taken now covers the run, one taken before it covers - // only what preceded it. This run's own guest is still up, so `qemu 1` is - // the quiet reading. - let host = hostload::HostLoad::sample(); - eprintln!( - " {label}{name} smp={smp} gaps: {} (baseline {}) peak {} active {:.2}s dither {:.1}% \ - pitch {:.1}Hz phase-breaks {}", - audio::format_histogram(&gaps), - audio::format_histogram(&baseline.gaps), - analysis.peak, - secs(analysis.active_samples), - analysis.dither_ratio.unwrap_or(0.0) * 100.0, - audio::dominant_hz(&wav).unwrap_or(0.0), - audio::phase_breaks(&wav).len(), - ); - eprintln!( - " {label}{name} smp={smp} soundd: wake_lat {}us ({:.2} pipelines, limit {}us) \ - [irq {}us + pickup {}us, {} empty wakes, batch {}, {} late of {}] \ - drains {}/{} underruns {}/{} submitted {} wakes {} batch {} windows {} — {} — {host}", - counters.max_wake_lat_us, - counters.max_wake_lat_us as f64 / audio::PIPELINE_DEPTH_US as f64, - baseline.counters.max_wake_lat_us, - counters.worst.irq_late_us, - counters.worst.pickup_us, - counters.worst.empty, - counters.worst.batch, - counters.late_wakes, - counters.wakes, - counters.drains, - baseline.counters.drains, - counters.underruns, - baseline.counters.underruns, - counters.submitted, - counters.wakes, - counters.max_batch, - counters.windows, - audio::boot_clocks(qemu.boot_log()), - ); - - let breaches = audio::check_counters(&counters, &baseline.counters); - if !breaches.is_empty() { - eprintln!( - " {label}{name} smp={smp} over ceiling: {} — recorded; the fast tier's \ - verdict is harm, the rate of these is the thorough tier's", - breaches.join("; ") - ); - } - - // A counter past a physical bound is the instrument failing, so it belongs - // here with the other instrument checks rather than among the ceilings: it - // must fail loudly in both tiers, and it must never be ranked against the - // recorded sample or printed into the next baseline. - let mut problems = audio::check_physical(&counters, run_start.elapsed().as_secs_f64()); - // soundd counts only while it has clients, so a run with no window reports - // zero for every counter — the best numbers this gate can see, from a run - // that measured nothing. That is the instrument dead, not a ceiling held. - if counters.windows == 0 { - problems.push( - "soundd printed no stats window with clients — the tone never reached the mixer" - .to_string(), - ); - } - if secs(analysis.active_samples) < TONE_MIN_ACTIVE_SECS { - problems.push(format!( - "tone missing: only {:.2}s of active signal (expected >= {TONE_MIN_ACTIVE_SECS}s)", - secs(analysis.active_samples) - )); - } - if analysis.peak < TONE_MIN_PEAK { - problems.push(format!( - "tone too quiet: peak {} (expected >= {TONE_MIN_PEAK})", - analysis.peak - )); - } - // Present, loud and continuous is not the same as right: a device consuming - // the buffers at a rate soundd did not ask for satisfies all three and plays - // the whole session off pitch. - if let Some(complaint) = audio::wrong_pitch(&wav) { - problems.push(complaint); - } - // Without this the gate can go green while measuring nothing: the underrun - // detector's silence band is derived from soundd applying TPDF dither into - // a rounding quantizer. Lose the dither and silence becomes - // exact zero everywhere, the band collapses, and dropouts stop being - // visible — the exact failure this instrument was rebuilt to remove. - match analysis.dither_ratio { - Some(ratio) if ratio < audio::MIN_DITHER_RATIO => problems.push(format!( - "dither missing: only {:.1}% of silent samples are non-zero (expected ~25%, \ - floor {:.0}%) — soundd is not dithering, so the underrun detector is blind", - ratio * 100.0, - audio::MIN_DITHER_RATIO * 100.0 - )), - Some(_) => {} - None => problems.push("no silent stretch in capture to verify dither against".to_string()), - } - if audio::check_gap_regression(&gaps, &baseline.gaps).is_err() { - let mut msg = format!( - "{} mid-signal underruns (silence >= 2ms inside the tone):", - analysis.underruns.len() - ); - for run in analysis.underruns.iter().take(20) { - msg.push_str(&format!( - "\n at {:8.3}s len {:6.2}ms", - secs(run.start), - secs(run.len) * 1000.0 - )); - } - if analysis.underruns.len() > 20 { - msg.push_str(&format!("\n ... and {} more", analysis.underruns.len() - 20)); - } - eprintln!(" {label}{name} smp={smp} {msg}"); - } - if !analysis.clicks.is_empty() { - let mut msg = format!("{} hard discontinuities (|delta| > 8000):", analysis.clicks.len()); - for click in analysis.clicks.iter().take(10) { - msg.push_str(&format!( - "\n at {:8.3}s {} -> {}", - secs(click.index), - click.from, - click.to - )); - } - if analysis.clicks.len() > 10 { - msg.push_str(&format!("\n ... and {} more", analysis.clicks.len() - 10)); - } - problems.push(msg); - } - - // Suspend structure — categorical per-run assertions, so they belong - // with the instrument checks: fatal in both tiers, never a counted rate. - problems.extend(audio::check_suspend_structure(&serial)); - - // Keep every capture that shows something, so a dropout can be listened to - // even when the tier's rule says one occurrence is not yet a verdict. - if !problems.is_empty() || !breaches.is_empty() || !gaps.is_empty() { - let suffix = if tag.is_empty() { - String::new() - } else { - format!("-{tag}") - }; - // Renamed in the lane so `keep_serial` recognises and copies it out if - // this run ends red; the lane itself is gone on every exit, so the path - // worth printing is where that copy lands, not this one. - let suspect = qemu - .audio_wav_path() - .with_file_name(format!("audio-{name}-smp{smp}{suffix}.wav")); - match fs::rename(qemu.audio_wav_path(), &suspect) { - Ok(()) => eprintln!( - " {label}{name} smp={smp} wav kept at {} if this run ends red", - common::lane::kept_path(&suspect).display() - ), - Err(e) => eprintln!( - " {label}{name} smp={smp} could not keep {}: {e}", - suspect.display() - ), - } - } - - Ok(AudioRun { - gaps, - counters, - broken: problems, - breaches, - host, - }) -} - -/// Fast tier — one boot per config, run on every `cargo test`. -/// -/// Certifies: the instrument is alive, no counter is on the wrong side of a -/// physical bound, and this build does not *reproducibly* put silence on the -/// wire. It cannot certify a *rate*; one run is one Bernoulli trial against a -/// per-config dropout rate measured at 0-7%, which discriminates nothing. That -/// is what `--audio-gate` is for. -/// -/// **The verdict is harm** — a mid-tone gap in the capture, or a period soundd -/// submitted with no client audio behind it. The per-run ceilings are measured, -/// printed and kept, and fail nothing here: `drains` past its ceiling with an -/// empty histogram and zero underruns is a pipeline that recovered before -/// anyone could hear it, and one boot cannot say whether it recovers less often -/// than it used to. That question has an instrument with power, and it is the -/// thorough tier's `ceiling_runs` rate. -/// -/// Harm is confirmed before it fails: a run that shows any is re-booted once, -/// and only a second failure counts. No bar is widened by this — the zero-gap -/// bar is strict on both boots. Without the confirmation the per-config dropout -/// rate alone reds one invocation in eight on a clean tree, and a gate -/// developers see every day cannot cry wolf that often. The first occurrence is -/// still printed and its capture still kept. -fn run_audio_test( - name: &str, - smp: u32, - baseline: &ConfigBaseline, - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let run = measure_audio_run(name, smp, baseline, test_config, c_bins, rust_bins, "")?; - - if !run.broken.is_empty() { - return Err(run.broken.join("\n ")); - } - let Some(harm) = run.harm() else { - return Ok(()); - }; - - let silent_runs = baseline.sample.underruns.iter().filter(|&&u| u > 0.0).count(); - eprintln!( - " {name} smp={smp} HARM {harm} — rare on this tree ({} of {} recorded runs \ - dropped audio, {silent_runs} of {} submitted a silent period); re-booting once \ - to confirm", - baseline.sample.gap_runs, - baseline.sample.gap_sample, - baseline.sample.underruns.len(), - ); - let again = measure_audio_run(name, smp, baseline, test_config, c_bins, rust_bins, "confirm")?; - if !again.broken.is_empty() { - return Err(again.broken.join("\n ")); - } - match again.harm() { - Some(again_harm) => Err(format!( - "audio dropped out on two consecutive boots: {harm} then {again_harm}" - )), - None => { - eprintln!(" {name} smp={smp} not reproduced on the confirming boot"); - Ok(()) - } - } -} - -// Thorough tier: `cargo test --test toyos-build -- --audio-gate N` - -/// One config's fresh sample, accumulated over the N iterations. -#[derive(Default)] -struct GateSamples { - max_wake_lat_us: Vec, - underruns: Vec, - wakes: Vec, - drains: Vec, - gap_runs: u32, - ceiling_runs: u32, -} - -/// A rejected statistic, ready to print. -struct Rejection { - config: String, - statistic: String, - detail: String, -} - -fn mwu_verdict( - config: &str, - statistic: &str, - base: &[f64], - fresh: &[f64], - worse_is_lower: bool, -) -> Option { - let z = stats::mann_whitney_z(base, fresh); - let z = if worse_is_lower { -z } else { z }; - let med = |v: &[f64]| { - let mut v = v.to_vec(); - v.sort_by(|a, b| a.partial_cmp(b).unwrap()); - v[v.len() / 2] - }; - (z > stats::Z_CRIT).then(|| Rejection { - config: config.to_string(), - statistic: statistic.to_string(), - detail: format!( - "median {:.0} -> {:.0} (Mann-Whitney z={z:.2} > {:.2})", - med(base), - med(fresh), - stats::Z_CRIT - ), - }) -} - -fn rate_verdict( - config: &str, - statistic: &str, - k1: u32, - n1: u32, - k0: u32, - n0: u32, -) -> Option { - let p = stats::fisher_greater(k1, n1, k0, n0); - (p <= stats::ALPHA).then(|| Rejection { - config: config.to_string(), - statistic: statistic.to_string(), - detail: format!( - "{k1} of {n1} vs recorded {k0} of {n0} (Fisher p={p:.2e} <= {:.0e})", - stats::ALPHA - ), - }) -} - -/// Thorough tier — N iterations of all four configs, gating on *rates* and -/// *distributions* rather than on single outcomes. The nightly runs it. -/// -/// Certifies, at N=30 and the measured clean-tree distributions: -/// * wake lateness has not shifted by 25% (detected 99.9% of the time) or -/// 20% (93%). A 10% shift is missed (4%). -/// * periods of silence on the wire have not risen 25% (94%) or 50% (100%). -/// * soundd is not being woken less often — the signature of completions -/// being batched because it ran late. A 5% drop is caught 99.9% of the -/// time. -/// * the mid-tone dropout *rate* has not risen 10x (100%) or 5x (71%). -/// A doubling is NOT detectable at this N and never will be at any N a -/// human waits for: separating 3% from 7% at this confidence needs ~600 -/// runs per config. The counters above are the instrument with power; the -/// dropout rate is the audible symptom, kept because it is the only -/// statistic here that says "someone would have heard it". -/// -/// False-red rate on a clean tree: 0.25%, measured over 2000 invocations -/// simulated from the recorded distributions. -fn run_audio_gate( - iterations: u32, - audio_baseline: &AudioBaseline, - audio_to_run: &[&str], - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> bool { - let configs: Vec<(&str, u32)> = audio_to_run - .iter() - .flat_map(|name| AUDIO_SMP.iter().map(move |&smp| (*name, smp))) - .collect(); - let mut samples: BTreeMap = BTreeMap::new(); - // Session-wide rather than per-config: the host is one host, and this is - // the sentence a re-record has to carry beside the numbers below. - let mut host: Vec = Vec::new(); - let start = std::time::Instant::now(); - - eprintln!( - "\n[gate A] {iterations} iterations x {} configs, serial. Every per-run outcome \ - becomes a rate; the verdict is on the collection, not on any one run.", - configs.len() - ); - - for iter in 1..=iterations { - eprintln!(" --- iteration {iter}/{iterations} ---"); - for &(name, smp) in &configs { - let key = format!("{name}.smp{smp}"); - let baseline = config_baseline(audio_baseline, name, smp); - let tag = format!("iter{iter:03}"); - let run = match measure_audio_run( - name, smp, &baseline, test_config, c_bins, rust_bins, &tag, - ) { - Ok(run) => run, - Err(err) => { - eprintln!("\n[gate A] FAILED on iteration {iter}: {key} produced no measurement: {err}"); - eprintln!("[gate A] A run that does not complete is not a rare event to be \ - averaged away — every known cause of one has been fixed."); - return false; - } - }; - if !run.broken.is_empty() { - eprintln!("\n[gate A] FAILED on iteration {iter}: {key} instrument broken: {}", - run.broken.join("; ")); - return false; - } - host.push(run.host); - let s = samples.entry(key).or_default(); - s.max_wake_lat_us.push(run.counters.max_wake_lat_us as f64); - s.underruns.push(run.counters.underruns as f64); - s.wakes.push(run.counters.wakes as f64); - s.drains.push(run.counters.drains as f64); - s.gap_runs += u32::from(run.dropped_audio()); - s.ceiling_runs += u32::from(!run.breaches.is_empty()); - } - - // Fail-side curtailment. Adding runs can only raise a count, so once a - // count passes the threshold for the *full* N the final verdict is - // already decided — stopping early costs no confidence. - if let Some(v) = curtail(&samples, audio_baseline, &configs, iterations) { - eprintln!("\n[gate A] FAILED after {iter} of {iterations} iterations (the remaining \ - runs cannot change this):"); - eprintln!(" {} {}: {}", v.config, v.statistic, v.detail); - return false; - } - } - - let mut rejected: Vec = Vec::new(); - let (mut pooled_gap_k, mut pooled_gap_n) = (0, 0); - let (mut pooled_ceil_k, mut pooled_ceil_n) = (0, 0); - let (mut base_gap_k, mut base_gap_n) = (0, 0); - let (mut base_ceil_k, mut base_ceil_n) = (0, 0); - - eprintln!("\n[gate A] {iterations} iterations in {:.0?}. Fresh sample vs recorded sample:\n", start.elapsed()); - eprintln!(" {}\n", hostload::summarise(&host)); - for &(name, smp) in &configs { - let key = format!("{name}.smp{smp}"); - let base = config_baseline(audio_baseline, name, smp).sample; - let s = &samples[&key]; - - rejected.extend(mwu_verdict(&key, "wake lateness", &base.max_wake_lat_us, &s.max_wake_lat_us, false)); - rejected.extend(mwu_verdict(&key, "underruns", &base.underruns, &s.underruns, false)); - rejected.extend(mwu_verdict(&key, "wakes", &base.wakes, &s.wakes, true)); - rejected.extend(rate_verdict(&key, "dropout rate", s.gap_runs, iterations, base.gap_runs, base.gap_sample)); - - pooled_gap_k += s.gap_runs; - pooled_gap_n += iterations; - pooled_ceil_k += s.ceiling_runs; - pooled_ceil_n += iterations; - base_gap_k += base.gap_runs; - base_gap_n += base.gap_sample; - base_ceil_k += base.ceiling_runs; - base_ceil_n += base.max_wake_lat_us.len() as u32; - - report_config(&key, base, s, iterations); - } - rejected.extend(rate_verdict("pooled", "dropout rate", pooled_gap_k, pooled_gap_n, base_gap_k, base_gap_n)); - rejected.extend(rate_verdict("pooled", "per-run ceiling breaches", pooled_ceil_k, pooled_ceil_n, base_ceil_k, base_ceil_n)); - - eprintln!( - " pooled dropouts {pooled_gap_k}/{pooled_gap_n} (recorded {base_gap_k}/{base_gap_n}), \ - ceiling breaches {pooled_ceil_k}/{pooled_ceil_n} (recorded {base_ceil_k}/{base_ceil_n})" - ); - - if rejected.is_empty() { - eprintln!("\n[gate A] PASS — no statistic regressed at alpha={:.0e} per test.", stats::ALPHA); - true - } else { - eprintln!("\n[gate A] FAILED — {} statistic(s) regressed:", rejected.len()); - for v in &rejected { - eprintln!(" {} {}: {}", v.config, v.statistic, v.detail); - } - false - } -} - -/// Whether a count has already passed the threshold it would face at the full -/// iteration count. Only the yes/no statistics curtail: a rank test's outcome -/// is not monotone in the sample, so there is no honest early exit for it. -fn curtail( - samples: &BTreeMap, - audio_baseline: &AudioBaseline, - configs: &[(&str, u32)], - iterations: u32, -) -> Option { - let mut pooled_gap = 0; - let mut pooled_ceil = 0; - let (mut base_gap_k, mut base_gap_n) = (0, 0); - let (mut base_ceil_k, mut base_ceil_n) = (0, 0); - for &(name, smp) in configs { - let key = format!("{name}.smp{smp}"); - let base = config_baseline(audio_baseline, name, smp).sample; - let Some(s) = samples.get(&key) else { continue }; - if let Some(v) = rate_verdict(&key, "dropout rate", s.gap_runs, iterations, base.gap_runs, base.gap_sample) { - return Some(v); - } - pooled_gap += s.gap_runs; - pooled_ceil += s.ceiling_runs; - base_gap_k += base.gap_runs; - base_gap_n += base.gap_sample; - base_ceil_k += base.ceiling_runs; - base_ceil_n += base.max_wake_lat_us.len() as u32; - } - let n = iterations * configs.len() as u32; - rate_verdict("pooled", "dropout rate", pooled_gap, n, base_gap_k, base_gap_n) - .or_else(|| rate_verdict("pooled", "per-run ceiling breaches", pooled_ceil, n, base_ceil_k, base_ceil_n)) -} - -/// Print one config's fresh sample next to the recorded one, in a form that can -/// be pasted straight back into `tests/audio-baseline.toml` when a re-baseline -/// is deliberate. The gate's output *is* the next baseline. -fn report_config(key: &str, base: &BaselineSample, s: &GateSamples, iterations: u32) { - let stat = |v: &[f64]| { - let mut v = v.to_vec(); - v.sort_by(|a, b| a.partial_cmp(b).unwrap()); - (v[0], v[v.len() / 2], v[v.len() - 1]) - }; - eprintln!(" {key} (n={iterations}, recorded n={})", base.max_wake_lat_us.len()); - for (label, b, f) in [ - ("wake_lat_us", &base.max_wake_lat_us, &s.max_wake_lat_us), - ("underruns ", &base.underruns, &s.underruns), - ("wakes ", &base.wakes, &s.wakes), - ("drains ", &base.drains, &s.drains), - ] { - let (bl, bm, bh) = stat(b); - let (fl, fm, fh) = stat(f); - eprintln!( - " {label} recorded {bl:.0}/{bm:.0}/{bh:.0} fresh {fl:.0}/{fm:.0}/{fh:.0} (min/median/max)" - ); - } - eprintln!( - " dropouts recorded {}/{} fresh {}/{iterations}", - base.gap_runs, base.gap_sample, s.gap_runs - ); - let fmt = |v: &[f64]| { - let mut v = v.to_vec(); - v.sort_by(|a, b| a.partial_cmp(b).unwrap()); - let v: Vec = v.iter().map(|x| format!("{x:.0}")).collect(); - format!("[{}]", v.join(", ")) - }; - eprintln!(" toml: max_wake_lat_us = {}", fmt(&s.max_wake_lat_us)); - eprintln!(" toml: underruns = {}", fmt(&s.underruns)); - eprintln!(" toml: wakes = {}", fmt(&s.wakes)); - eprintln!(" toml: drains = {}", fmt(&s.drains)); -} - /// Echo what the guest actually put on screen, under `--nocapture` only — -/// it is the measurement these tests are built on, and the audio gate prints -/// its numbers for the same reason. +/// it is the measurement these tests are built on. fn print_screen(name: &str, text: &str) { if !qemu::VERBOSE.load(std::sync::atomic::Ordering::Relaxed) { return; @@ -4723,14 +3809,12 @@ fn run_screen_test( }; metal_sim_argv_check(&qemu::profile_argv(&options))?; let mut qemu = QemuInstance::boot_with_options(&config, &[], &[], options); - let console = qemu.boot_log().to_string(); - qemu.screendump_until("Boot: complete", Duration::from_secs(30)); - + let mut console = qemu.boot_log().to_string(); // The window the mode exists to close: on the flashed image the // compositor's first output landed 48 ms after `Boot: complete`. - // Holding two orders of magnitude longer than that is what makes - // "indefinitely" a measurement rather than a claim. - thread::sleep(Duration::from_secs(5)); + // So the panel is read once the last program this image starts has + // run and exited, when anything that would claim it already has. + await_marker(&mut qemu, &mut console, "exit: toybox", "the last program the image starts")?; let dump = qemu.screendump(); let text = dump.text(); print_screen(name, &text); @@ -4752,8 +3836,8 @@ fn run_screen_test( { if !text.contains(want) { return Err(format!( - "{want:?} is not on screen five seconds after the boot \ - finished\ndecoded screen:\n{text}" + "{want:?} is not on screen once every program the image starts had \ + run\ndecoded screen:\n{text}" )); } } @@ -5802,7 +4886,6 @@ fn run_screen_test( // drop and declaration, the PL011 SPCR names, the boot's survey of // the machine, and a panic on both channels, before the kernel // reaches the AArch64 userland its ROOT carries. - let started = std::time::Instant::now(); let mut qemu = QemuInstance::boot_with_options( test_config, &[], @@ -5818,7 +4901,6 @@ fn run_screen_test( let dump = qemu.screendump_until("EARLY PANIC:", Duration::from_secs(30)); let rest = qemu.drain_until(Duration::from_secs(10), |l| l.contains(EARLY_PANIC_MESSAGE)); let serial = format!("{}\n{rest}", qemu.boot_log()); - eprintln!(" [virt] the panel is up {} ms after the boot began", started.elapsed().as_millis()); // What stage 3 prints before it panics: every item is a record // only the AArch64 side of the loader or the kernel writes. for want in [ @@ -6095,7 +5177,7 @@ fn run_screen_test( Ok(()) } "screen_pager_keys" => { - // The halted pager takes PageDown off the i8042 with every + // The halted pager takes PageUp off the i8042 with every // CPU stopped, and this is the only place that claim can be made: // the decode is `toyos-ps2`'s and host-tested, but that a keystroke // reaches a machine which has stopped scheduling is a fact about @@ -6145,59 +5227,45 @@ fn run_screen_test( } }; - // How long the unattended deadline actually takes to move the page, - // measured before a key is pressed because the first key retires it - // for good. This is what stops the last phase passing vacuously: a - // guest too slow to have paged in its window would prove nothing by - // not paging, and this is the window measured on *this* guest. - let timing_from = Instant::now(); - let unattended_move = loop { + // The unattended deadline moves the page on its own, which is + // waited for before a key is pressed because the first key retires + // it for good. + loop { let Some(now) = footer(&mut qemu) else { return Err(format!( - "{STALLED} the footer vanished while timing the unattended deadline" + "{STALLED} the footer vanished while waiting for the unattended deadline" )); }; if now != last { last = now; - break timing_from.elapsed(); + break; } if Instant::now() >= deadline { return Err(format!( - "{STALLED} the pager did not advance on its own in {:.1}s against a 3s \ - deadline — nothing here can say whether a keystroke stops it", - timing_from.elapsed().as_secs_f64() + "{STALLED} the pager did not advance on its own — nothing here can say \ + whether a keystroke stops it" )); } - }; + } - // **One keystroke, then its page, then the next keystroke.** The - // verdict is that every one of them moved the page, and there is no - // clock of the host's in it: a guest that is slow costs this run - // wall clock and never a move. - // - // It used to inject all thirty at the host's own speed and compare - // the moves it saw against what a 3 s deadline could have produced - // in the elapsed time — `moved >= elapsed/3 + 1` times three. That - // arithmetic asks a guest which has not been given time to repaint - // once for three moves, so on a host that got through the thirty in - // 0.3 s it demanded 3.3 of them and reported `0 page moves over 30 - // keystrokes in 0.3s`: the symptom, where the fact was that nothing - // had run. Two agents bisected that as a kernel regression on one - // day. Unpaced it was wrong about the wire as well — thirty - // press/release pairs is sixty scancodes into QEMU's 16-byte - // `PS2_QUEUE_SIZE` (`hw/input/ps2.c`), so the keys a full-panel - // repaint had no room for were never delivered at all. - // - // What makes "every key moved it" the whole claim, with no rate - // beside it, is the phase below: after the first keystroke the - // deadline is retired for good, so it contributes no move to this - // loop, and if it were still running the steered page would not hold. + // **One keystroke, then its page, then the next keystroke**, and + // every key is PageUp: the unattended deadline only ever moves the + // page forward, so a page one back is the key's and nothing else's. + // The first key races the deadline it retires, so its page may come + // after one more forward move; from the second on, every move has to + // be exactly one page back, and a forward one is the deadline still + // running under a reader who has taken the wheel. No clock of the + // host's is in any of it: a guest that is slow costs this run wall + // clock and never a move. const SAMPLES: usize = 30; - let started = Instant::now(); + let page_of = |footer: &str| -> Option<(usize, usize)> { + let (n, m) = footer.strip_prefix("[page ")?.split_once(']')?.0.split_once('/')?; + Some((n.trim().parse().ok()?, m.trim().parse().ok()?)) + }; for key in 1..=SAMPLES { - qemu::qmp_send_keys(&socket, &[("pgdn", true), ("pgdn", false)]); + qemu::qmp_send_keys(&socket, &[("pgup", true), ("pgup", false)]); let by = Instant::now() + qemu.budget(Duration::from_secs(20)); - loop { + let now = loop { let Some(now) = footer(&mut qemu) else { return Err(format!( "{STALLED} the footer vanished after {} of {SAMPLES} keystrokes", @@ -6205,78 +5273,29 @@ fn run_screen_test( )); }; if now != last { - last = now; - break; + break now; } if Instant::now() >= by { return Err(format!( - "keystroke {key} of {SAMPLES} left the pager on {last:?}: a PageDown \ - reached a halted machine and no page came of it" + "{STALLED} keystroke {key} of {SAMPLES} left the pager on {last:?}: a \ + PageUp reached a halted machine and no page came of it" )); } - } - } - let elapsed = started.elapsed(); - - // Nothing is in flight — the loop above did not send a key until the - // page the one before it moved was on the screen — so this asks only - // that the panel is not mid-repaint before the watch starts. - const SETTLED: Duration = Duration::from_secs(1); - let settle_by = Instant::now() + qemu.budget(Duration::from_secs(20)); - let mut held = last; - let mut stable_since = Instant::now(); - loop { - let Some(now) = footer(&mut qemu) else { - return Err(format!( - "{STALLED} the footer vanished while the last page settled" - )); - }; - if now != held { - held = now; - stable_since = Instant::now(); - } else if stable_since.elapsed() >= SETTLED { - break; - } - if Instant::now() >= settle_by { - return Err(format!( - "the pager never held one page for {}s after the last keystroke, so \ - something is still moving it", - SETTLED.as_secs() - )); - } - } - - // And now the owner's complaint, which is the other half: a page he - // steered to must stay up. The window is twice what the unattended - // deadline was measured to need above, so a pager still running it - // moves at least twice inside this and a slow guest cannot pass by - // being slow. - let quiet = unattended_move * 2 + Duration::from_secs(1); - let watching_from = Instant::now(); - while watching_from.elapsed() < quiet { - let Some(now) = footer(&mut qemu) else { - return Err("the footer vanished while watching a steered page".into()); }; - if now != held { + let ((was, pages), (is, _)) = page_of(&last) + .zip(page_of(&now)) + .ok_or_else(|| format!("unreadable footers {last:?} and {now:?}"))?; + let back = if was == 1 { pages } else { was - 1 }; + if key > 1 && is != back { return Err(format!( - "the page moved from {held:?} to {now:?} on its own {:.1}s into a {:.1}s \ - watch after the last keystroke — the deadline is still running under a \ - reader who has taken the wheel, which is what it must not do", - watching_from.elapsed().as_secs_f64(), - quiet.as_secs_f64() + "keystroke {key} of {SAMPLES} moved the page from {last:?} to {now:?}, \ + not one back — the deadline is still running under a reader who has \ + taken the wheel, which is what it must not do" )); } + last = now; } - print_screen( - name, - &format!( - "every one of {SAMPLES} keystrokes moved the page, in {:.1}s; unattended it \ - moved once in {:.1}s, and after a keystroke it held {held} for {:.1}s", - elapsed.as_secs_f64(), - unattended_move.as_secs_f64(), - quiet.as_secs_f64(), - ), - ); + print_screen(name, &format!("every one of {SAMPLES} PageUps moved the page one back")); Ok(()) } "screen_fatal_halt" => { @@ -6526,11 +5545,6 @@ fn run_screen_test( // tells the three states apart, and a photograph that has it has // the answer. // - // Twice, and the second time is the half that matters. A photograph - // is taken seconds after the key, by a person, of a machine whose - // userland may still be composing — so the assertion is not that - // the paint happened but that the panel still carries it once the - // desktop has had its turn. let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/desktopaudiocase"); let options = BootOptions { profile: qemu::Profile::Metal, @@ -6595,67 +5609,17 @@ fn run_screen_test( dump = qemu.screendump_while( Duration::from_secs(4), Duration::from_millis(100), - |d| report_is_photographable(d, "").is_ok(), - ); - } - let text = dump.text(); - print_screen(name, &text); - report_is_photographable(&dump, "the report the keystroke painted")?; - - // **A single paint is not a report.** Whoever owns the screen goes - // on composing and has no idea the kernel drew, so the panel the - // owner photographs is the one that survived the next client frame - // — not the one the dump painted. Typing is the actuator: the shell - // echoes and the terminal repaints its whole window, which is - // exactly what was measured blanking every row of the report that - // lay under it, inside 100 ms, leaving the four rows below the - // window and a 40-pixel strip beside it. - // - // **This guest is muted, so the actuator is bounded rather than - // confirmed.** No console carries the shell's echo back and no - // window's font decodes, so what is left is arithmetic: this line - // plus the one chord the loop above may still have outstanding fits - // the device queue, so nothing here can be dropped. The premise of - // the assertion below is still the measured idle repaint. - { - let mut input = qemu::QmpInput::open(qemu.qmp_socket()); - let actuator = "echo\n"; - let bytes: usize = actuator.chars().map(qemu::scancode_bytes).sum(); - assert!( - bytes + CHORD_BYTES <= QEMU_PS2_QUEUE, - "{actuator:?} is {bytes} set-1 bytes behind a {CHORD_BYTES}-byte chord that \ - may still be queued, against a {QEMU_PS2_QUEUE}-byte device queue" + |d| report_is_photographable(d, "").is_ok(), ); - input.type_burst(actuator); } - // Userland's turn, and the wait is the assertion's premise rather - // than padding: measured with the hold compiled out, an idle - // desktop changes this panel inside 1.5 s on its own, and without - // this the check below answered on its first capture — before the - // compositor had composed once — and passed vacuously. - // - // **Deliberately not `qemu::budget`.** That pays a *liveness - // ceiling* out per guest, and this is not one: it is a settle - // inside a window the guest itself is timing, so scaling it by the - // phase width walks out of the window it has to stay inside. - // Twelve wide it became 36 s against a 15 s hold, and the check - // then measured a machine that had already given the screen back. - // The window is guest time and a loaded guest's is *longer* in host - // seconds, so a fixed wait errs inwards from both sides. - std::thread::sleep(Duration::from_secs(2)); - let back = qemu.screendump_while( - Duration::from_secs(5), - Duration::from_millis(100), - |d| report_is_photographable(d, "").is_ok(), - ); - print_screen(&format!("{name} after a client repaint"), &back.text()); - report_is_photographable(&back, "the report after a client repainted over it")?; + let text = dump.text(); + print_screen(name, &text); + report_is_photographable(&dump, "the report the keystroke painted")?; - let row = back.row_index("== VERDICT:").expect("checked above"); + let row = dump.row_index("== VERDICT:").expect("checked above"); eprintln!( - " [dump] on the panel of a guest with no console, and still there after the \ - desktop repainted: {}", - back.rows()[row].trim() + " [dump] on the panel of a guest with no console: {}", + dump.rows()[row].trim() ); Ok(()) } @@ -6707,20 +5671,11 @@ const SSHD_LOGIN: &str = "sshd login"; /// The line `tests/toyos-rust-tests/src/bin/i8042_keyboard.rs` prints once it /// holds the keyboard claim, and the line every injection into that binary is -/// timed off. Eight callers wait for it, and one — `i8042_undecoded_bytes` — +/// timed off. Its callers wait for it, and one — `i8042_undecoded_bytes` — /// also reads its capture *from* it: it is the boundary between what the /// machine did on its own and what this test staged. const I8042_READY: &str = "===I8042_READY==="; -impl Boot { - /// Drain for `dur` into the group's console, and hand back the whole of it. - fn drain(&mut self, dur: Duration) -> &str { - let more = self.qemu.drain_serial(dur); - self.console.push_str(&more); - &self.console - } -} - /// The shared boot this machine test runs on, or `None` if it owns its own. /// /// **Two conditions decide membership and neither is cost.** No member may kill @@ -6966,9 +5921,6 @@ fn metal_sim_compositor(boot: &mut Boot) -> Result<(), String> { boot.console )); } - // One more drain so the tail of that line cannot still be in - // flight when it is parsed. - boot.drain(Duration::from_millis(250)); let console = &boot.console; // The compositor reports the mode it was handed, which is the // proof it claimed a real firmware framebuffer rather than @@ -7194,7 +6146,6 @@ fn metal_sim_window_drag(rust_bins: &[(String, Vec)]) -> Result<(), String> let mut boot_log = qemu.boot_log().to_string(); await_marker(&mut qemu, &mut boot_log, "compositor: frames=", "the boot's own repaint interval") .map_err(|why| format!("{why}\n{boot_log}"))?; - boot_log.push_str(&qemu.drain_serial(Duration::from_millis(250))); let screen_px = compositor_screen_px(&boot_log)?; let (screen_w, screen_h) = compositor_screen_size(&boot_log)?; let ppc = px_per_count(screen_w, screen_h); @@ -7468,12 +6419,10 @@ fn compositor_screen_size(console: &str) -> Result<(u32, u32), String> { /// End's release: the sentinel `test_rs_i8042_keyboard` exits on /// (`tests/toyos-rust-tests/src/bin/i8042_keyboard.rs`). Every caller that -/// injects through a fresh connection sends this after its last injection -/// instead of running out the binary's fallback deadline, except -/// `i8042_health_cadence` — whose verdict is a report cadence over a real span, -/// not a delivered key. [`i8042_no_spurious_wake`], which holds one connection -/// open for the whole run, sends the same two transitions as the last group of -/// its own script: a `-qmp …,server` socket +/// injects through a fresh connection sends this after its last injection. +/// [`i8042_no_spurious_wake`], which holds one connection open for the whole +/// run, sends the same two transitions as the last group of its own script: a +/// `-qmp …,server` socket /// serves one monitor at a time, so a second one opened here would block. fn send_i8042_sentinel(socket: &Path) { qemu::qmp_send_keys(socket, &[("end", true), ("end", false)]); @@ -8200,9 +7149,7 @@ fn shell_answers(qemu: &mut QemuInstance, log: &mut String, ack: &Drained) -> Re /// **Two waits, because the two ways this fails are different questions.** The /// first is "has the terminal come up", and it used to be answered by retyping /// against `qemu.budget(20 s)` — a guess at how long a desktop takes to come up -/// on the host of the day, which is exactly the shape `issues/design-debt/` -/// bills for: `desktop_audio_client` 385 s wide against 13 s alone, and a -/// landing gate that is a coin toss. The terminal knows when it is up and now +/// on the host of the day. The terminal knows when it is up and now /// says so, so this asks it and waits on the guest's own liveness. The second is /// "does a keystroke reach the shell", and it starts from a machine that is /// demonstrably up — a ceiling on *that* is a claim about the guest. @@ -8364,121 +7311,6 @@ fn doom_frames(rust_bins: &[(String, Vec)]) -> Result<(), String> { Ok(()) } -/// Gate: doom's music reaches the device, with the SoundFont this tree ships. -/// -/// **The wiring is all this measures, and the wiring is the part nothing else -/// can.** `src/soundfont.rs`'s host tests say the committed bank covers every -/// instrument `assets/DOOM1.WAD` selects, and the subset was measured to render -/// bit-exact against the full bank through this same -/// `mus2mid.c` and this same rustysynth. Neither can say the file got into an -/// image, that doom opened it, or that what came out reached an audio device. -/// Those three are what `b8b0749` broke for a cycle with the suite green. -/// -/// Three verdicts, none of them a clock: -/// -/// 1. **doom opened the file this tree committed.** The guest prints the byte -/// count it read, and the host compares it against `assets/soundfont.sf2` on -/// disk. A stale image, a truncated asset and a second SoundFont from -/// somewhere else all fail here rather than turning into quiet silence. -/// 2. **It played to the end of the check.** The actuator counts the audio -/// callback's own periods, so a host that stopped this guest cannot shorten -/// what the capture is judged on. -/// 3. **Music reached the wire.** The device capture carries signal across most -/// of its length — which separates music from the one thing a broken -/// soundfont path still produces, a stream of zeroes. -fn doom_music(rust_bins: &[(String, Vec)]) -> Result<(), String> { - let root = Path::new(env!("CARGO_MANIFEST_DIR")); - let shipped = fs::metadata(root.join(toyos_build::soundfont::SOUNDFONT_PATH)) - .map_err(|e| format!("{}: {e}", toyos_build::soundfont::SOUNDFONT_PATH))? - .len(); - - let config = root.join("tests/doommusiccase"); - let mut qemu = QemuInstance::boot_with_options(&config, &[], rust_bins, BootOptions::default()); - - let result = qemu.run_test("test_rs_doom_music", Duration::from_secs(120)); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - if result.exit_code != Some(0) { - return Err(format!( - "doom could not play its own music (exit {:?}):\n{}", - result.exit_code, result.stdout - )); - } - - let opened = result - .stdout - .lines() - .find(|line| line.contains("[doom-sound] /system/share/soundfont.sf2:")) - .ok_or_else(|| { - format!( - "doom said nothing about the SoundFont, so this image has none:\n{}", - result.stdout - ) - })?; - let bytes: u64 = opened - .split_whitespace() - .find_map(|token| token.parse().ok()) - .ok_or_else(|| format!("no byte count in {opened:?}"))?; - if bytes != shipped { - return Err(format!( - "doom opened a {bytes}-byte SoundFont and this tree ships {shipped} bytes: the \ - image is not carrying {}", - toyos_build::soundfont::SOUNDFONT_PATH - )); - } - - let played = result - .stdout - .lines() - .find(|line| line.contains("[music-check] lump=")) - .ok_or_else(|| format!("doom printed no [music-check] line:\n{}", result.stdout))? - .to_string(); - - let _ = qemu.drain_serial(Duration::from_millis(500)); - let wav = audio::parse_wav(qemu.audio_wav_path())?; - let analysis = audio::analyze(&wav); - - // Seconds of signal, not a fraction of the capture: the capture runs from - // soundd opening the stream to the harness closing the file, so a fraction - // measures the harness as much as the music. Three of these four runs - // measured 1.19 s over a 3.11 s capture — the material's own dynamics, not - // a shortfall, since 500 LSB is -36 dBFS and E1M1's riff drops through it - // between notes. - const MIN_SIGNAL_SECS: f64 = 0.8; - let signal = analysis.active_samples as f64 / wav.sample_rate as f64; - if signal < MIN_SIGNAL_SECS { - return Err(format!( - "{signal:.2} s of the capture carries signal, under {MIN_SIGNAL_SECS} s: what doom \ - rendered is not what the device played\n{played}" - )); - } - // A floor and not a band: how loud E1M1 is at a given moment is the - // arrangement's business, and what this excludes is a dither floor being - // read as music. Measured 13547. - const MIN_PEAK: i32 = 6000; - if analysis.peak < MIN_PEAK { - return Err(format!( - "the device peaked at {} (expected at least {MIN_PEAK}): the music is inaudible\ - \n{played}", - analysis.peak - )); - } - - // Underruns are reported and fail nothing: whether music *stutters* is gate - // A's question and it has the statistics to ask it, where one boot of one - // track is one sample of an intermittent. - eprintln!( - " [doommusiccase] {}, {signal:.2} s of signal in a {:.2} s capture at peak {}, \ - {} underrun(s)", - played.trim(), - wav.mono.len() as f64 / wav.sample_rate as f64, - analysis.peak, - analysis.underruns.len(), - ); - Ok(()) -} - fn desktop_window_child(rust_bins: &[(String, Vec)]) -> Result<(), String> { let bins: Vec<(String, Vec)> = rust_bins.iter().filter(|(name, _)| name == "window_child").cloned().collect(); @@ -8778,7 +7610,6 @@ fn toolkit_iced() -> Result<(), String> { // The rectangle holds the app's pixels only while its window is open, so // the close has to be of that same window, by this keystroke: an app that // died after it opened leaves the rectangle to whatever is behind it. - log.push_str(&qemu.drain_serial(Duration::from_millis(200))); if log[launched..].contains(&exited) { return Err(format!("{app} left before its window was judged:\n{}", &log[launched..])); } @@ -8872,7 +7703,7 @@ fn toolkit_launch( /// /// `test_rs_window_wake` is launched from the desktop's shell, so it holds the /// shell's compositor, and says OK only if every one of its rounds was ended by -/// the wake; a lost one ends its wait on a ten-second ceiling and says so. +/// the wake. fn toolkit_window_wake(rust_bins: &[(String, Vec)]) -> Result<(), String> { let (mut qemu, mut log, launched) = toolkit_launch(rust_bins, "window_wake", "test_rs_window_wake")?; let log = &mut log; @@ -8907,8 +7738,6 @@ fn toolkit_window_wake(rust_bins: &[(String, Vec)]) -> Result<(), String> { fn toolkit_winit_loop(rust_bins: &[(String, Vec)]) -> Result<(), String> { let (mut qemu, mut log, launched) = toolkit_launch(rust_bins, "winit_loop", "test_rs_winit_loop")?; let log = &mut log; - // Past the app's own twenty-second ceiling, so a lost wake is its verdict - // and not this one's. let mut live = qemu::Liveness::new(Duration::from_secs(40), Duration::from_secs(240)); let mut closed = false; while live.working(log) { @@ -9102,12 +7931,6 @@ fn window_child_probes(qemu: &mut QemuInstance, log: &mut String) -> Result<(), &log[before.min(log.len())..] )); } - if log[before..].contains("WINDOW-CHILD-TIMEOUT") { - return Err(format!( - "the client left on its own deadline, so it never saw the close:\n{}", - &log[before..] - )); - } if let Err(why) = shell_echoes(qemu, log, "after-window-closed-zqjxk", &ack) { return Err(format!( "{why}\nthe compositor closed a child's window and the shell never answered again \ @@ -9208,11 +8031,14 @@ fn desktop_typing_damage() -> Result<(), String> { let screen_px = compositor_screen_px(&log)?; // Let the interval carrying the boot's full-screen repaint and the - // terminal's first paint close before anything here is measured. Those are - // real frames and they are not what this is about. - await_marker(&mut qemu, &mut log, "compositor: frames=", "the compositor to report an interval") - .map_err(|why| format!("{why}\n{log}"))?; - log.push_str(&qemu.drain_serial(Duration::from_secs(3))); + // terminal's first paint close before anything here is measured: the + // report after the one that follows the shell's answer. Those are real + // frames and they are not what this is about. + let from = log.len(); + await_guest(&mut qemu, &mut log, "two more compositor intervals", |c| { + c[from..].matches("compositor: frames=").count() >= 2 + }) + .map_err(|why| format!("{why}\n{log}"))?; let before = log.len(); // Eight lines, each typed a character at a time — the shell's echo of each @@ -9243,7 +8069,12 @@ fn desktop_typing_damage() -> Result<(), String> { log.push_str(&seen); } } - log.push_str(&qemu.drain_serial(Duration::from_secs(3))); + // The interval holding the last keystrokes is the one reported after them. + let last = log.len(); + await_guest(&mut qemu, &mut log, "the interval after the last line", |c| { + c[last..].contains("compositor: frames=") + }) + .map_err(|why| format!("{why}\n{log}"))?; let typed = &log[before..]; // Sixteen: the shell echoes the command as it is typed and again as its @@ -9443,165 +8274,6 @@ fn desktop_locale_detect() -> Result<(), String> { Ok(()) } -/// A shell-spawned audio client on a device-less desktop, and the desktop -/// afterwards. -/// -/// The machine `metal_sim_null_audio` and `null_sink_shipped_client` both miss: -/// they spawn the client from a test binary whose stdio is the console, and the -/// T14 spawns it from a shell inside a terminal inside the compositor, so every -/// one of the client's three descriptors is a pipe to a surface. Three verdicts -/// on one boot, in the order the T14 lost them: -/// -/// 1. **A client finishes.** `tone` writes a second of audio to the null sink -/// and prints its own completion line. -/// 2. **A second client connects while the first is streaming.** The T14's log -/// shows soundd's control thread printing `opening stream` for the second -/// with no `client N connected` behind it, so the connect is what has to be -/// observed, not just the exit. -/// 3. **The desktop survives them.** A terminal opened afterwards reaches a -/// shell that answers — the verdict the owner's machine failed while the -/// compositor was still painting, which is why nothing that reads pixels or -/// counts frames would have caught it. -fn desktop_audio_client() -> Result<(), String> { - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/desktopaudiocase"); - let options = BootOptions { - profile: qemu::Profile::Metal, - // The T14's core count: the suite's default of two serialises threads - // this shape is about the wakes between. - smp: 8, - qmp: true, - ready_marker: "compositor: ready", - // `Drained::Bytes`; off the shipping kernel, and implies fast-health - // and edge-race. - kernel_params: &["i8042-trace"], - ..Default::default() - }; - metal_sim_argv_check(&qemu::profile_argv(&options))?; - let mut qemu = QemuInstance::boot_with_options(&config, &[], &[], options); - let mut log = qemu.boot_log().to_string(); - - const NULL_LINE: &str = "soundd: no audio device, presenting a null sink"; - await_marker(&mut qemu, &mut log, NULL_LINE, "soundd to present a null sink") - .map_err(|why| format!("{why}\n{log}"))?; - // No panel row under a compositor; the kernel's drain report is the answer. - let ack = Drained::Bytes; - if let Err(why) = shell_answers(&mut qemu, &mut log, &ack) { - return Err(format!( - "{why}\nnothing typed at the terminal window reached a shell:\n{log}" - )); - } - - // One client, start to finish. `tone: done` is the client's own last line, - // so it is the client saying it got its callbacks and left — not the shell - // saying it launched something. - shell_type_line(&mut qemu, "tone 440 1", &ack)?; - await_marker( - &mut qemu, - &mut log, - "tone: done", - "a shell-spawned tone to finish on a device-less desktop", - ) - .map_err(|why| format!("{why}\n{log}"))?; - - // Two clients overlapping, each under its own shell in its own terminal. - // The shell has no job control, so the long tone holds its terminal and the - // second one has to be typed somewhere else — which is exactly how the T14 - // reached two live clients, and why the second terminal is part of the - // stimulus rather than only part of the verdict. - let before_second = log.len(); - shell_type_line(&mut qemu, "tone 660 8", &ack)?; - await_marker_new( - &mut qemu, - &mut log, - "tone: 660Hz", - before_second, - "the long tone to start", - ) - .map_err(|why| format!("{why}\n{}", &log[before_second..]))?; - open_terminal(&mut qemu, &mut log, "overlap-terminal-jc4t", &ack)?; - shell_type_line(&mut qemu, "tone 440 1", &ack)?; - // **The count is the verdict and the wait is not.** Both of these used to be - // `budget(60 s)`, which is a claim that a desktop with two audio clients on - // it finishes inside a minute times the width — and at 385 s wide against - // 13 s alone it was the single most expensive entry in `issues/design-debt/`. What - // ends the wait now is soundd going quiet, and what fails it is still the - // number of connects. - if let Err(why) = await_guest(&mut qemu, &mut log, "soundd to take up both connects", |log| { - connects_since(log, before_second) >= 2 - }) { - return Err(format!( - "{why}\nsoundd applied {} of the two connects — a client that opened a stream \ - was never taken up by the mixer:\n{}", - connects_since(&log, before_second), - &log[before_second..] - )); - } - // Both of them out again, counted in the same window. Waiting for `null - // sink idle` would not do: that line is already in the log from the first - // client, and a marker an earlier phase produced is not a verdict about - // this one. - if let Err(why) = await_guest(&mut qemu, &mut log, "both clients to leave the mixer", |log| { - removals_since(log, before_second) >= 2 - }) { - return Err(format!( - "{why}\n{} of the two overlapping clients left the mixer — the other one is \ - still streaming to a sink that stopped draining it:\n{}", - removals_since(&log, before_second), - &log[before_second..] - )); - } - - // The desktop afterwards: a process created after every one of the clients - // above, focused the moment it maps its window. This is the verdict the - // owner's machine failed while the compositor was still painting, which is - // why nothing that reads pixels or counts frames would have caught it. - open_terminal(&mut qemu, &mut log, "post-audio-desktop-vqmz", &ack)?; - eprintln!(" [desktop] three shell-spawned audio clients ran and the desktop still answers"); - Ok(()) -} - -/// Ctrl+N at the compositor, and a shell in the window it opens that answers. -/// -/// The nonce is per call because the verdict is that *this* terminal answered: -/// a marker an earlier one already produced would pass on a window that never -/// came up. [`shell_echoes`]'s split applies for the same reason it does there, -/// and `terminal: ready` is looked for after `before` rather than anywhere, -/// because every terminal already up has printed one. -fn open_terminal( - qemu: &mut QemuInstance, - log: &mut String, - nonce: &str, - ack: &Drained, -) -> Result<(), String> { - let before = log.len(); - { - let mut input = qemu::QmpInput::open(qemu.qmp_socket()); - input.keys(&[("ctrl", true), ("n", true), ("n", false), ("ctrl", false)]); - } - await_marker_new(qemu, log, "terminal: ready", before, "Ctrl+N to open a terminal") - .map_err(|why| format!("{why}\n{}", &log[before..]))?; - - // The same count-of-attempts as [`shell_echoes`], and for the same reason. - const TRIES: usize = 10; - let mut lost = String::new(); - for _ in 0..TRIES { - if let Err(said) = - shell_type_once(qemu, &format!("echo {nonce}"), round_trip(ECHO_TRY), ack) - { - lost = said; - continue; - } - if serial_until(qemu, log, nonce, round_trip(Duration::from_secs(2))) { - return Ok(()); - } - } - Err(format!( - "a terminal opened with Ctrl+N never reached a shell that answers in {TRIES} typed \ - lines:\n{lost}\n{}", - &log[before..] - )) -} - /// The host server behind the netd stream tests, where slirp's `10.0.2.2` /// lands. Each connection asks in nine bytes — a mode, then a little-endian /// length — and is served by the mode (`Ask` in @@ -9896,17 +8568,6 @@ fn netd_held_open(rust_bins: &[(String, Vec)]) -> Result<(), String> { Ok(()) } -/// A client that out-writes a peer which has stopped reading costs netd no -/// CPU: the send pipe is not watched while the socket has no room for it. -fn netd_stalled_peer(rust_bins: &[(String, Vec)]) -> Result<(), String> { - let HostRun { result, .. } = netcase_against_host(rust_bins, "netd_stalled_peer", false, "")?; - let Some(line) = result.stdout.lines().find(|l| l.contains("netd_stalled_peer: ok")) else { - return Err(format!("the guest never said it was done:\n{}", result.stdout)); - }; - eprintln!(" [netcase] {}", line.trim_end()); - Ok(()) -} - /// A UDP datagram the client's pipe will not take whole ends that socket by /// name, and nothing else: another socket still gets its datagram. fn netd_udp_refused(rust_bins: &[(String, Vec)]) -> Result<(), String> { @@ -10025,7 +8686,7 @@ const DNS_NO_ADDRESS: &str = "no results"; /// again, and ends one nobody answers when its schedule does: the netcase boot /// once it has its lease, with every frame it sends from then on held by /// QEMU, so no query reaches its resolver. The verdict is the guest's, from -/// netd's answers and their times. +/// netd's answers. fn netd_lookup_let_go(rust_bins: &[(String, Vec)]) -> Result<(), String> { const NAME: &str = "netd_lookup_let_go"; let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/netcase"); @@ -10126,7 +8787,7 @@ fn netd_refused_pipes(rust_bins: &[(String, Vec)]) -> Result<(), String> { } } serial::Serial::named("boot console", console.as_str()).must_be_clean()?; - for line in result.stdout.lines().filter(|l| l.contains(", and a round trip after it, in ")) { + for line in result.stdout.lines().filter(|l| l.contains(", and a round trip after it")) { eprintln!(" [netcase] {}", line.trim_end()); } eprintln!(" [netcase] six refused client handles cost netd nothing, and each was named"); @@ -10315,25 +8976,6 @@ fn dump_field(report: &str, marker: &str, word: &str) -> Result { .ok_or_else(|| format!("no number before {word:?} on {line:?}")) } -/// Connects the mixer has *applied* since `from`, which is a different event -/// from the control thread's `opening stream` — the T14 log carries the second -/// without the first. -fn connects_since(log: &str, from: usize) -> usize { - soundd_clients_since(log, from, " connected") -} - -/// Clients the mixer has ramped out and dropped since `from`. -fn removals_since(log: &str, from: usize) -> usize { - soundd_clients_since(log, from, " removed") -} - -fn soundd_clients_since(log: &str, from: usize, verb: &str) -> usize { - log[from..] - .lines() - .filter(|l| l.contains("soundd: client ") && l.contains(verb)) - .count() -} - /// The direct regression for the readiness defect: a stimulus that produces /// bytes and no events must produce no wake. Pause is that stimulus — six /// bytes, deliberately swallowed. @@ -10485,14 +9127,6 @@ fn i8042_no_spurious_wake(boot: &mut Boot) -> Result<(), String> { /// guest's own report rather than against a wall clock. const QEMU_PS2_QUEUE: usize = 16; -/// What a three-key chord costs on the wire, in set-1 bytes. -/// -/// Ctrl+Alt+D and Ctrl+N and GUI+Q are all this or less: six transitions at -/// most, none of them `0xE0`-prefixed on the qcodes this suite injects. A -/// caller that types behind a chord it cannot prove has been consumed budgets -/// this much of [`QEMU_PS2_QUEUE`] for it. -const CHORD_BYTES: usize = 6; - /// A PS/2 pointer packet. Three bytes, because the driver's aux init sends no /// IntelliMouse knock and QEMU therefore frames a plain mouse. const MOUSE_PACKET: usize = 3; @@ -10844,14 +9478,9 @@ fn metal_sim_ipc_hostile_peer(boot: &mut Boot) -> Result<(), String> { /// A client that stops talking, stops listening, or never stops. /// -/// The guest carries the "is it still answering" half, because only it can put -/// a deadline on the answer; the host carries the half the guest cannot see — -/// whether the desktop is still *painting*, and whether every client the -/// compositor got rid of was named. -/// -/// The two halves are not redundant. A compositor parked on one client answers -/// nobody, so the guest catches that; a compositor livelocked on one client -/// answers everybody and draws nothing, which only the frame counter shows. +/// The guest carries the "is it still answering" half; the host carries the half +/// the guest cannot see — whether the desktop is still *painting*, and whether +/// every client the compositor got rid of was named. /// /// Last in its group: it is the one that abuses the compositor hardest, and /// its own final assertion is that the desktop is still compositing after it. @@ -10882,27 +9511,6 @@ fn metal_sim_compositor_stall(boot: &mut Boot) -> Result<(), String> { let frames = |text: &str| text.matches("compositor: frames=").count(); - // Starvation, which is the one shape the guest cannot see. Between - // these two markers one window is sending on every pass; a drain - // loop that ends only when nothing is ready never gets to `redraw` - // and this window holds zero frames. - let Some(stream) = result - .stdout - .split("compositor stall: stream start") - .nth(1) - .and_then(|rest| rest.split("compositor stall: stream end").next()) - else { - return Err(format!( - "the guest never bracketed its streaming window:\n{}", - result.stdout - )); - }; - if frames(stream) == 0 { - return Err(format!( - "the compositor composited nothing while one client streamed:\n{stream}" - )); - } - // Dropped by name, never silently. Three connections never finish // a first frame, and one window stops reading its mail. const TIMED_OUT: &str = "it never finished its first message"; @@ -11108,13 +9716,11 @@ fn run_machine_test( "quiesce_wakes_on_the_last_park" => power::quiesce_wakes_on_the_last_park(test_config, c_bins, rust_bins), "quiesce_wakes_on_the_last_teardown" => power::quiesce_wakes_on_the_last_teardown(test_config, c_bins, rust_bins), "watchdog_resets" => power::watchdog_resets(test_config, c_bins, rust_bins), - "watchdog_fed" => power::watchdog_fed(test_config, c_bins, rust_bins), "loader_watchdog_arms" => power::loader_watchdog_arms(test_config, c_bins, rust_bins), "panic_reboots" => power::panic_reboots(test_config, c_bins, rust_bins), "panic_before_peripherals_reboots" => { power::panic_before_peripherals_reboots(test_config, c_bins, rust_bins) } - "panic_key_holds" => power::panic_key_holds(test_config, c_bins, rust_bins), "blackbox_panic_chain" => power::blackbox_panic_chain(test_config, c_bins, rust_bins), "panic_outlives_the_deadline" => { power::panic_outlives_the_deadline(test_config, c_bins, rust_bins) @@ -11199,7 +9805,7 @@ fn run_machine_test( // The lost-wake canary with the window it guards held open: every pipe // wait reads its condition, waits for a post to land, then parks, so // the ping-pong's posts land between the two. A commit that ignored the - // notified bit parks for good and the canary counts it short. + // notified bit parks for good, and the run's ceiling reds it. "blocking_read_window" => { let options = BootOptions { kernel_params: &["watch-window"], @@ -11216,7 +9822,7 @@ fn run_machine_test( result.stdout, result.before, result.serial )); } - window_held(&(boot + &result.before), &result.serial) + Ok(()) } // Two CPUs: the held copy spins in the kernel while its sibling unmaps // and maps on the other. @@ -11306,30 +9912,10 @@ fn run_machine_test( "kernel_heartbeat" => { // The instrument for a machine whose log cannot say whether it was // alive: ten of the owner's boots are byte-identical between the - // ones that froze and the ones that did not. The gate has to prove - // three things a `must_say` cannot — that the lines *keep coming*, - // that no CPU drops out of the mask on a machine with nothing to - // do, and that no window between two lines is wide enough to hide a - // death. - // - // The second is why this asserts a *constant full* mask where the - // old gate asserted a *varying* one — and the old gate's assertion - // was satisfied by the defect, so it certified it. Same guest with - // the tick removed: 10 of 11 lines below `alive=8/8`, six of them at - // `alive=2/8`, 56 lines naming a silent CPU and one silent for - // 2.811 s. Every line is `8/8` with the tick. - // - // The gap bound below is the one with no demonstrated teeth here: - // without the tick the widest gap was still 0.260 s, because QEMU's - // devices keep waking *someone* even when they wake nobody in - // particular. It is carried for the metal log, where the same code - // left gaps of 14 s to 102 s. - // - // **What none of it establishes**: a QEMU guest is never as quiet as - // the owner's laptop. This proves the tick arms, fires and re-arms, - // and that the instrument reads a full mask when nothing is wrong. - // It cannot prove the T14's LAPIC keeps counting through whatever - // its firmware does with a halted core. + // ones that froze and the ones that did not. What a guest can prove + // of it is that the lines *keep coming*, that each carries the + // pin's state beside it, and that the pin's state is read off the + // chip. let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/metalcase"); let options = BootOptions { profile: qemu::Profile::Metal, @@ -11339,12 +9925,11 @@ fn run_machine_test( }; // **A heartbeat and the `i8042: line` under it are one reading**, // and `heartbeat::poll` emits them as two `log!`s — so a capture can - // end between them. Run `31273373928` on `main` did: twelve beats, - // eleven pin readings, and the last beat was the last line of the - // log. Counting the two kinds against each other reads that as a pin - // whose state was unreadable, which is the one thing this pairing - // exists to detect. So the unit is the pair, and a beat with nothing - // after it at all is a reading this capture does not hold. + // end between them. Counting the two kinds against each other reads + // that as a pin whose state was unreadable, which is the one thing + // this pairing exists to detect. So the unit is the pair, and a beat + // with nothing after it at all is a reading this capture does not + // hold. fn whole(log: &str) -> (Vec<&str>, Vec) { let captured: Vec<&str> = log.lines().collect(); let at: Vec = captured @@ -11357,49 +9942,17 @@ fn run_machine_test( let kept = captured.len() - usize::from(torn); (captured[..kept].to_vec(), at[..at.len() - usize::from(torn)].to_vec()) } - - // The mask is a claim only about a settled machine that is running, - // and `toyos_build::heartbeat` is where that is decided — including - // which line each `[boot] start` program says it has finished - // starting with, held there against this config's own list. - let said = heartbeat::done_lines(&toyos_build::build::boot_start( - &config.join("system.toml"), - ))?; + /// Whole beats the verdict is read from. + const BEATS: usize = 5; let mut qemu = QemuInstance::boot_with_options(&config, &[], &[], options); let mut log = qemu.boot_log().to_string(); - // **The capture follows the window, not the clock.** How long the - // started programs take is the loaded host's to decide, and a - // capture cut a fixed span after `===READY===` hands the predicate - // whatever start-up left over — so a slow boot reds the test on the - // predicate's own refusal. The drain ends when the window holds - // `CAPTURE_BEATS`, which at a 250 ms period is under two seconds of - // settled machine; the bound is the liveness ceiling and not the - // capture's length, and it is counted in *steps* rather than - // measured in wall time, because a guest that has exited - // disconnects the reader and a step then returns at once. - const DRAIN_STEP: Duration = Duration::from_millis(500); - const DRAIN_STEPS: u32 = 40; - let drained = Instant::now(); - let mut held = 0; - for _ in 0..DRAIN_STEPS { - log.push_str(&qemu.drain_serial(DRAIN_STEP)); - held = heartbeat::window_beats(&whole(&log).0, &said); - if held >= heartbeat::CAPTURE_BEATS { - break; - } - } - if held < heartbeat::CAPTURE_BEATS { - return Err(format!( - "{held} heartbeat(s) with a whole period after the boot's start-up against \ - the {} a verdict is taken from, after {} drains of {} ms at a 250 ms period \ - ({:.1}s) — the instrument has to keep reporting past the start-up, and a log \ - that stops, or a boot that never finishes starting, says nothing\n{log}", - heartbeat::CAPTURE_BEATS, - DRAIN_STEPS, - DRAIN_STEP.as_millis(), - drained.elapsed().as_secs_f64(), - )); + // The instrument has to keep reporting: a guest that stops is what + // the harness ceiling reds. + if let Err(why) = + await_guest(&mut qemu, &mut log, "five whole heartbeats", |c| whole(c).1.len() >= BEATS) + { + return Err(format!("{why}\n{log}")); } let (captured, at) = whole(&log); let beats: Vec<&str> = at.iter().map(|&i| captured[i]).collect(); @@ -11426,79 +9979,6 @@ fn run_machine_test( unpaired.iter().take(4).cloned().collect::>().join("\n"), )); } - let settled = match heartbeat::settle(&captured, &said) { - Ok(settled) => settled, - Err(heartbeat::Refused::Unreadable(line)) => { - return Err(format!( - "a heartbeat carries no readable t=, alive=, mask=, ran= or gap= — the \ - fields that say which CPU stopped and whether the machine ran: \ - {line}\n{log}" - )); - } - Err(heartbeat::Refused::BootUnfinished(line)) => { - return Err(format!( - "the boot never finished starting: nothing said {line:?}, so every \ - heartbeat is inside the start-up and none is a claim about a settled \ - machine\n{log}" - )); - } - Err(heartbeat::Refused::Unsettled { settled, beats }) => { - return Err(format!( - "too few heartbeats have a whole period after the last `[boot] start` \ - program finished starting — the machine did not settle inside this \ - capture, and a clear bit before it settles says nothing\n\ - {settled} of {beats}\n{log}" - )); - } - Err(heartbeat::Refused::NotRunning { beat, held }) => { - return Err(format!( - "the machine was not running: a settled heartbeat came late or dispatched \ - nothing — a guest its host did not schedule, and what a CPU did in a \ - period the machine did not run is unreadable\n\ - t={}.{:03}s gap={}.{:03}s ran={} against a {} ms period, after {held} \ - settled heartbeat(s) it ran through\n{}\n{log}", - beat.t_ms / 1000, - beat.t_ms % 1000, - beat.gap_ms / 1000, - beat.gap_ms % 1000, - beat.ran, - heartbeat::PERIOD_MS, - captured[beat.line], - )); - } - Err(heartbeat::Refused::CpuMissing { cpus, settled, opened }) => { - return Err(format!( - "cpu{cpus:?} missing from {} consecutive heartbeats on a settled guest \ - that was running — a CPU that misses two lines has missed five \ - `diag-tick` wakes, so a clear bit does not mean that CPU stopped, which \ - is the whole of the field\n{settled} settled heartbeats\n{}\n{log}", - heartbeat::STOPPED_BEATS, - captured[opened..] - .iter() - .filter(|l| l.contains("heartbeat: ")) - .take(16) - .copied() - .collect::>() - .join("\n"), - )); - } - }; - let quiet = &settled.beats; - let blips = settled.blips; - // And no window between two lines may be wide enough to hide a - // death. The metal boots this exists for went quiet for between 14 s - // and 102 s; four times the period is far below any of them and far - // above anything a loaded host does to a 250 ms cadence. - const MAX_GAP_MS: u64 = 1000; - let worst = settled.widest_gap_ms; - if worst > MAX_GAP_MS { - return Err(format!( - "the widest window between two heartbeats was {}.{:03}s against a 250 ms \ - period — the machine stopped reporting for long enough to have died in\n{log}", - worst / 1000, - worst % 1000, - )); - } // And the clock in the line advances, or the timestamp cannot // localise a death. let stamps: Vec<&str> = beats @@ -11580,17 +10060,10 @@ fn run_machine_test( )); } eprintln!( - " [heartbeat] {} whole lines in {:.1}s, each with its own pin reading, {} \ - before the machine settled and {} after, {blips} of those missing a CPU for one \ - line and none for two, widest gap {}.{:03}s, t={} → t={}; \ - {} i8042 line reading(s), vec 0x{vector} on gsi {kbd_gsi}, none masked, none with \ + " [heartbeat] {} whole lines, each with its own pin reading, t={} → t={}; {} \ + i8042 line reading(s), vec 0x{vector} on gsi {kbd_gsi}, none masked, none with \ OBF set", beats.len(), - drained.elapsed().as_secs_f64(), - beats.len() - quiet.len(), - quiet.len(), - worst / 1000, - worst % 1000, stamps.first().unwrap_or(&"?"), stamps.last().unwrap_or(&"?"), lines.len(), @@ -11726,34 +10199,21 @@ fn run_machine_test( "blockd_lends_within_its_bound" => { common::blockd::blockd_lends_within_its_bound(test_config, c_bins, rust_bins) } - // Body in `tests/common/hda.rs`, same reason. - "hda_tone" => common::hda::hda_tone(test_config, c_bins, rust_bins), - "hda_client_stall" => common::hda::hda_client_stall(test_config, c_bins, rust_bins), - "hda_two_live_refused" => { - common::hda::hda_two_live_refused(test_config, c_bins, rust_bins) - } "double_fault_stack" => faults::double_fault_stack(test_config, c_bins, rust_bins), "syscall_window_nmi" => faults::syscall_window_nmi(test_config, c_bins, rust_bins), "syscall_window_nmi_controls" => { faults::syscall_window_nmi_controls(test_config, c_bins, rust_bins) } "idle_stack_guard" => faults::idle_stack_guard(test_config, c_bins, rust_bins), - "dump_nmi_probe" => faults::dump_nmi_probe(test_config, c_bins, rust_bins), "dump_left_pending_is_owed" => { faults::dump_left_pending_is_owed(test_config, c_bins, rust_bins) } "diskless_boot" => faults::diskless_boot(test_config, c_bins, rust_bins), "virtio_net_no_msix" => faults::virtio_net_no_msix(), "pci_claim_caps_truncated" => faults::claim_caps_truncated(), - // Body in `tests/common/audio.rs`, so the hunk here stays one line. - "metal_sim_null_audio" => audio::null_sink_real_rate(test_config, c_bins, rust_bins), - "null_sink_shipped_client" => audio::null_sink_shipped_client(test_config, c_bins, rust_bins), - "doom_sound_flood" => audio::doom_sound_flood(rust_bins), - "doom_music" => doom_music(rust_bins), // Body in `tests/common/clang.rs`. "c_hello" => common::clang::c_hello(rust_bins), "doom_frames" => doom_frames(rust_bins), - "soundd_log_stall" => audio::soundd_log_stall(rust_bins), "metal_sim_compositor" => { metal_sim_compositor(group_boot(held, METAL_SIM_DESKTOP, || { boot_metal_sim_desktop(rust_bins) @@ -11828,7 +10288,6 @@ fn run_machine_test( "toolkit_window_wake" => toolkit_window_wake(rust_bins), "toolkit_winit_loop" => toolkit_winit_loop(rust_bins), "toolkit_winit_pace" => toolkit_winit_pace(rust_bins), - "desktop_audio_client" => desktop_audio_client(), "blocked_dump" => blocked_dump(), "xhci_many_devices" => { // The T14's internal controller carries a camera, Bluetooth and a @@ -12620,14 +11079,11 @@ fn run_machine_test( ready_marker: REFUSAL, ..Default::default() }; - /// A liveness margin over the work that follows the refusal, never a bound. - const AFTER_REFUSAL: Duration = Duration::from_secs(2); - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); + let qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); // This profile has no virtio-serial, so stdio *is* the 16550 and - // `boot_log` is the whole record. It ends at the refusal, and the - // drain is the rest of the window the downstream work would be in. - let mut log = serial::Serial::boot(&qemu); - log.push(&qemu.drain_serial(AFTER_REFUSAL)); + // `boot_log` is the whole record. It ends at the refusal, which is + // the driver's `assert!`: nothing downstream of it can run. + let log = serial::Serial::boot(&qemu); // Named, not just refused: the value the device reported is the // whole diagnostic on a machine that will not boot again without @@ -12895,27 +11351,6 @@ fn run_machine_test( // budget were each caught by nothing on hardware, however green the // simulator was. // - // **What the simulator cannot say.** Two of the three instruments - // are asserts about state, and the sim checks those globally and - // better. The third is a measurement of *cost*, and the sim's clock - // does not advance inside a step — `scenarios::overlong_pass` feeds - // the recorder a modelled pass cost, which proves the recorder - // compiles and counts, not what a real pass on real silicon costs. - // Only a booted kernel reads a TSC. - // - // **And the cost half is gated here rather than in the kernel, and - // against a recorded sample rather than against the budget.** What - // a pass measures is wall clock across the pass, and a guest's wall - // clock runs while the host has taken its vCPU away — so the - // quantity includes a term the host's scheduler sets. Measured - // 2026-08-18, that term moves *every* order statistic and not only - // the tail, so `common::passcost` judges each accelerator against - // what that accelerator has been recorded producing, and takes no - // verdict at all where the recorded sample supports none. Its own - // two-directions self-check runs first, because a gate that must - // stay green under host descheduling has to be shown doing so on a - // case no booted machine can stage. - // // **The workload is `sched_stress`** because the asserts are dense // on exactly what it does: it spawns burners that drive vruntime, // blocks and wakes across io_uring and ports, and forces a @@ -12931,7 +11366,6 @@ fn run_machine_test( // 0 of 3 assert texts in the shipping kernel, 3 of 3 in this one. // This half is the other one: on a machine that really carries them, // honest work does not trip them. - common::passcost::self_check()?; let mut qemu = QemuInstance::boot_with_options( test_config, c_bins, @@ -12966,35 +11400,28 @@ fn run_machine_test( for line in result.stdout.lines() { eprintln!(" [sched-check] {}", line.trim()); } - // The whole boot, in the three pieces a capture comes in: the ready - // marker, the hole after it, and the test window. The counters are - // cumulative since boot, so the last line each CPU published is the - // whole of that CPU's run. + // The instrument published, read back through the parser its format + // is held to: a report from every CPU, over the whole boot in the + // three pieces a capture comes in. What a report says is not judged + // here — a pass's cost is a duration. let mut capture = serial::Serial::boot(&qemu); capture.push(&result.before); capture.push(&result.serial); - let reports = common::passcost::reports(capture.text()); - if reports.is_empty() { + let reported: BTreeSet = capture + .text() + .lines() + .filter_map(toyos_sched::cpu::PassCostReport::parse) + .map(|report| report.cpu.0) + .collect(); + let cpus = BootOptions::default().smp; + if reported.len() != cpus as usize { return Err(format!( - "the check build published no pass-cost report at all, so nothing above \ - gated what a pass costs — every pass on this boot went unmeasured or \ - unspoken. `{}` is the prefix that never appeared:\n{}", + "the check build published pass-cost reports from cpus {reported:?} of the \ + {cpus} it boots: `{}` is the prefix the rest never printed\n{}", toyos_sched::cpu::PassCostReport::PREFIX, capture.text(), )); } - // Which recorded sample this run is judged against, before the - // numbers it judges: a verdict taken against a sample is - // unreadable without naming the sample, and a run that judged - // nothing has to say so where a reader cannot miss it. - let baseline = common::passcost::baseline(); - eprintln!(" [sched-check] {}", common::passcost::judgement_line(baseline)); - for report in &reports { - eprintln!(" [sched-check] {}", common::passcost::describe(report)); - } - for report in &reports { - common::passcost::verdict(report, baseline)?; - } Ok(()) } "klogd_hosted" => { @@ -13223,14 +11650,12 @@ fn run_machine_test( // the CPU: the branch printed no `RECURSIVE` and ran the whole // second report. `test-late-panic` is the first crash and // `fault-in-report` is the wild read inside its report. - /// A liveness margin over the report that follows the alert, never a bound. - const AFTER_RECURSIVE: Duration = Duration::from_secs(1); let mut qemu = QemuInstance::boot_with_options( test_config, c_bins, rust_bins, BootOptions { - kernel_params: &["test-late-panic", "fault-in-report"], + kernel_params: &["test-late-panic", "fault-in-report", "panic-reboot-fast"], ready_marker: "RECURSIVE", ..Default::default() }, @@ -13250,10 +11675,14 @@ fn run_machine_test( // first panic's report never ran either — the wild read is at its // head. `KERNEL PANIC:` is `crash_report_exception`'s own header; // the stack scan is `double_fault_handler`'s, which is the - // escalation and not the report. Drained past the marker first: - // `boot_log` stops at `RECURSIVE` and the report follows it there. - nested.push(&qemu.drain_serial(AFTER_RECURSIVE)); - for report in ["KERNEL PANIC:", "Scanning kernel stack at"] { + // escalation and not the report. Judged past the marker, over the + // capture the fatal path's reset closes: `boot_log` stops at + // `RECURSIVE` and the report would follow it there. + const REPORTS: [&str; 2] = ["KERNEL PANIC:", "Scanning kernel stack at"]; + let mut tail = String::new(); + qemu::await_reset(&mut qemu, &mut tail, "the fatal path to reset the machine", &REPORTS)?; + nested.push(&tail); + for report in REPORTS { nested.must_not_say(report)?; } eprintln!(" [nested] the recursive arm bounded the report: no second crash report"); @@ -13273,13 +11702,10 @@ fn run_machine_test( // for the whole boot, so the end of the log is now where the // machine stopped rather than where it last drained. // - // The verdict is content and not a duration: what is asserted is - // which lines arrived, from the first phase to the wedge, and that - // the phase after it never did. + // The verdict is which lines arrived, from the first phase to the + // wedge; nothing follows it because the wedge never returns. const WEDGE: &str = "pre-idle-wedge: the boot stops here"; - /// A liveness margin over the phase after the wedge line, never a bound. - const STAYED_WEDGED: Duration = Duration::from_millis(1500); - let mut qemu = QemuInstance::boot_with_options( + let qemu = QemuInstance::boot_with_options( test_config, c_bins, rust_bins, @@ -13299,7 +11725,7 @@ fn run_machine_test( ..Default::default() }, ); - let mut boot = serial::Serial::boot(&qemu); + let boot = serial::Serial::boot(&qemu); // Every phase up to the wedge, oldest first — the first line the // machine ever logs, a line from between the first two checkpoints, // and the storage phase the wedge follows. @@ -13313,11 +11739,6 @@ fn run_machine_test( ] { boot.must_say(needle)?; } - // And nothing from after it, which is what says the machine really - // is wedged rather than slow — over a window the later phases could - // have reached, the marker here being the wedge line itself. - boot.push(&qemu.drain_serial(STAYED_WEDGED)); - boot.must_not_say("Boot: complete")?; eprintln!( " [wedge] {} kernel line(s) reached the console from a machine that never \ reached a scheduler pass", @@ -13715,11 +12136,9 @@ fn run_machine_test( root_from_memory(qemu.boot_log()) } "boot_from_power_on" => { - let asked = std::time::Instant::now(); let qemu = QemuInstance::boot(test_config, c_bins, rust_bins); - let host = asked.elapsed(); // The loader speaks on the firmware's serial, the kernel on the console. - boot_from_power_on(&format!("{}{}", qemu.uart_log(), qemu.boot_log()), host) + boot_from_power_on(&format!("{}{}", qemu.uart_log(), qemu.boot_log())) } "root_withheld_refused" => { let qemu = QemuInstance::boot_with_options( @@ -13747,21 +12166,6 @@ fn run_machine_test( let qemu = QemuInstance::boot(test_config, c_bins, rust_bins); pci_inventory(qemu.boot_log()) } - "tlb_shootdown_cost" => { - const CPUS: u32 = 8; - let qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - smp: CPUS, - kernel_params: &["tlb-shootdown-bench"], - ..Default::default() - }, - ); - tlb_shootdown_cost(qemu.boot_log(), CPUS).map(|_| ()) - } - "latency_wake" => latency_wake(rust_bins), "smp_failed_ap_leaves_no_hole" => { smp_failed_ap_leaves_no_hole(test_config, c_bins, rust_bins) } @@ -13770,102 +12174,15 @@ fn run_machine_test( // failure arrives as a dead boot; the marker is the only proof it // ran at all. let qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - kernel_params: &["test-input-merge"], - ..Default::default() - }, - ); - input_merge_ok(qemu.boot_log()) - } - "i8042_health_cadence" => { - // The T14 lost keyboard, TrackPoint and touchpad — all three behind - // this controller — 6.6 s into a session, and the driver's last - // word on the subject was printed 15 ms *before* it happened. The - // verdict was terminal, so for the remaining 54 s the log cannot - // distinguish "the pin stopped asserting" from "bytes kept arriving - // and decoded to nothing". Those are opposite defects in opposite - // subsystems and the counters that separate them were read once. - // - // What is under test is not that a line appears. It is that its - // *absence* means something: the report fires whenever the pin has - // asserted since the last one, so no line means no interrupt. A - // report that fired on a timer would satisfy every "is it alive" - // search and answer nothing. - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - qmp: true, - kernel_params: &["i8042-fast-health"], - ..Default::default() - }, - ); - if !qemu.boot_log().contains("i8042: kbd set2+xlat") { - return Err(format!("the PS/2 keyboard never came up:\n{}", qemu.boot_log())); - } - // One key, a silence several periods long, then one more key. The - // guest program holds the keyboard claim for 5 s and the period is - // 500 ms, so the quiet stretch is nine periods with nothing to say. - let result = qemu.run_test_hooked( - "test_rs_i8042_keyboard", - Duration::from_secs(30), - I8042_READY, - |socket| { - qemu::qmp_send_keys(socket, &[("a", true), ("a", false)]); - thread::sleep(Duration::from_millis(3000)); - qemu::qmp_send_keys(socket, &[("b", true), ("b", false)]); - thread::sleep(Duration::from_millis(1000)); - }, - ); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - let lines: Vec<&str> = - result.serial.lines().filter(|l| l.contains("last byte at")).collect(); - // Two keystrokes, two lines. Not one — the verdict is not the - // report and a driver that only ever spoke at boot would give one. - // Not ten — a line per period through the quiet stretch is the - // failure that makes silence unreadable, and it is the reason this - // test injects a gap at all. - if lines.len() != 2 { - return Err(format!( - "two keystrokes three seconds apart, {} counter lines — the report is on a \ - timer rather than on the pin:\n{}", - lines.len(), - lines.join("\n") - )); - } - let last_byte_ms = |line: &str| -> Option { - line.rsplit_once("last byte at ")?.1.trim_end_matches("ms").parse().ok() - }; - let first = last_byte_ms(lines[0]) - .ok_or_else(|| format!("unreadable counter line: {}", lines[0]))?; - let second = last_byte_ms(lines[1]) - .ok_or_else(|| format!("unreadable counter line: {}", lines[1]))?; - // The second line is about the second keystroke, not a rerun of the - // first. This is what dates the freeze on a machine whose log is - // read hours later. - if second <= first { - return Err(format!( - "the second report dates the last byte at {second}ms, not after {first}ms — \ - it is repeating a stale reading:\n{}", - lines.join("\n") - )); - } - // A working keyboard owes none of the four fault counters. - for want in ["0 discarded", "0 overruns", "0 dropped", "0 lost edges"] { - if !lines[1].contains(want) { - return Err(format!("a healthy keyboard reports {want:?} wrong: {}", lines[1])); - } - } - eprintln!(" [i8042] {}", lines[0].trim()); - eprintln!(" [i8042] {}", lines[1].trim()); - Ok(()) + test_config, + c_bins, + rust_bins, + BootOptions { + kernel_params: &["test-input-merge"], + ..Default::default() + }, + ); + input_merge_ok(qemu.boot_log()) } "i8042_health" => { // The failure mode that had no line at all: `init` arms the pin, @@ -13954,17 +12271,6 @@ fn run_machine_test( "the alive line reports {irqs} interrupts, {bytes} bytes, {keys} keys: {line}" )); } - // `verdict_due` keeps a CPU awake for one pass. If it ever failed to - // self-clear, that CPU would spin instead of halting — the exact - // failure the quarantine path already had once. `log_health` - // prints at a fixed rate regardless, so the trip-delta check is - // read from the counter inside the line rather than a count of - // the lines themselves. - if let Some((cpu, delta)) = idle_is_spinning(&result.serial) { - return Err(format!( - "cpu{cpu}'s idle-trip counter moved by {delta} within the capture — spinning, not halting" - )); - } // Boot three: the arming edge staged — the vector delivered once // with no byte behind it, because init consumed the byte itself. @@ -14136,81 +12442,8 @@ fn run_machine_test( ); lapic_vectors(qemu.boot_log()) } - "panic_halts_the_others_first" => { - // **A kernel that has declared itself corrupt runs nothing else.** - // `halt_all_cpus` sends the halt IPI before anything else it does; - // a fatal path that waited first — for a log, a drain, anything — - // would leave every other CPU running userland under it. - // `test_rs_panic_halts_first` keeps three siblings making kernel - // records while its main thread goes fatal, and every record is - // stamped on the kernel's clock with its CPU: none of another CPU - // may be stamped past the fatal record by more than a sibling can - // take to reach its next instruction boundary with `IF` set. - // - // **The bound is 100 ms, against the derivation**: an IPI is - // taken at the sibling's next instruction boundary with `IF` set, - // so a sibling runs past the fatal record by at most the longest - // window this kernel holds `IF` clear, and every such window is - // bounded in milliseconds. - const BOUND_MS: u64 = 100; - const RECORD: &str = "syscall 26 is retired"; - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { smp: 4, kernel_features: ACTUATOR_KERNEL, ..Default::default() }, - ); - writeln!(qemu.stdin_mut(), "run test_rs_panic_halts_first").map_err(|e| format!("stdin: {e}"))?; - qemu.flush_stdin(); - let mut console = - qemu.drain_until(Duration::from_secs(30), |l| l.contains(FATAL_HALT_NONCE)); - if !console.contains(FATAL_HALT_NONCE) { - return Err(format!("{FATAL_HALT_NONCE:?} never reached the console\n{console}")); - } - // What the fatal path flushes after the nonce. The machine is - // halted and says nothing more, so this is a pace, not a guard. - console.push_str(&qemu.drain_serial(Duration::from_secs(3))); - let stamp = |line: &str| -> Option<(u64, u32)> { - let head = line.split_once("[kernel ")?.1.split_once(']')?.0; - let (secs, cpu) = head.split_once(" cpu")?; - let (s, ms) = secs.split_once('.')?; - let cpu = cpu.split(' ').next()?; - Some((s.parse::().ok()? * 1000 + ms.parse::().ok()?, cpu.parse().ok()?)) - }; - let Some((fatal_ms, fatal_cpu)) = - console.lines().find(|l| l.contains(FATAL_HALT_NONCE)).and_then(stamp) - else { - return Err(format!("no stamped {FATAL_HALT_NONCE:?} record on the console\n{console}")); - }; - let siblings: Vec<(u64, u32, &str)> = console - .lines() - .filter(|l| l.contains(RECORD)) - .filter_map(|l| stamp(l).map(|(ms, cpu)| (ms, cpu, l))) - .filter(|&(_, cpu, _)| cpu != fatal_cpu) - .collect(); - // Non-vacuity: another CPU was making records up to the fatal one. - if !siblings.iter().any(|&(ms, _, _)| ms + 1000 >= fatal_ms) { - return Err(format!( - "no other CPU's record in the second before the fatal one at {fatal_ms} ms, so \ - nothing was running to be halted\n{console}" - )); - } - if let Some(&(ms, cpu, line)) = siblings.iter().max_by_key(|&&(ms, _, _)| ms) { - if ms > fatal_ms + BOUND_MS { - return Err(format!( - "cpu{cpu} made a record {} ms after the fatal one on cpu{fatal_cpu}: the \ - fatal path let it run\n {line}", - ms - fatal_ms - )); - } - } - eprintln!( - " [panic] {} record(s) of other CPUs; the last {} ms after the fatal one", - siblings.len(), - siblings.iter().map(|&(ms, _, _)| ms.saturating_sub(fatal_ms)).max().unwrap_or(0) - ); - Ok(()) - } + "panic_halts_the_others_first" => panic_halts_the_others_first(test_config, c_bins, rust_bins), + "hda_two_live_refused" => hda_two_live_refused(test_config, c_bins, rust_bins), "virtio_used_ring" => { // Both fields of a virtqueue used-ring element are written by the // device, and on virtio-sound's control and event queues the ring @@ -14719,21 +12952,6 @@ fn run_machine_test( Ok(()) } "i8042_absent" => { - // A/B in one session: the guest's own `Boot: complete (Nms)` is - // the instrument, because host-side timing here is dominated by - // image builds. A wait-loop bug that costs a second on a machine - // with a controller costs a minute on one without. - let with = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { profile: qemu::Profile::Metal, ..Default::default() }, - ); - let with_log = with.boot_log().to_string(); - let with_ms = boot_millis(&with_log) - .ok_or_else(|| format!("no `Boot: complete` line:\n{with_log}"))?; - drop(with); - let without = QemuInstance::boot_with_options( test_config, c_bins, @@ -14762,29 +12980,19 @@ fn run_machine_test( )); } // The floating bus, not any of the sixteen handshake refusals: on a - // machine with nothing there the probe must cost one `inb`, and - // that is also what makes the timing assertion below tight. + // machine with nothing there the probe must cost one `inb`. let want = "i8042: absent — port 0x64 reads 0xff"; if !log.contains(want) { return Err(format!("no `{want}` line on a machine with no i8042:\n{log}")); } - let without_ms = boot_millis(&log) - .ok_or_else(|| format!("no `Boot: complete` line:\n{log}"))?; - // The regression this guards is 2100 ms: with no floating-bus test - // the very first `wait_writable` sees IBF set in 0xff and waits out - // the whole init budget. The allowance is for boot-to-boot noise - // between two QEMU launches in one session, nothing else. - if without_ms > with_ms + 300 { - return Err(format!( - "boot took {without_ms}ms without an i8042 and {with_ms}ms with one — a wait is not bounded" - )); + if boot_millis(&log).is_none() { + return Err(format!("no `Boot: complete` line:\n{log}")); } eprintln!(" [i8042] firmware: {}", claim.trim()); eprintln!( " [i8042] {}", log.lines().find(|l| l.contains(want)).unwrap_or_default().trim() ); - eprintln!(" [i8042] boot {without_ms}ms without vs {with_ms}ms with"); Ok(()) } "i8042_quarantine" => { @@ -14798,12 +13006,7 @@ fn run_machine_test( BootOptions { profile: qemu::Profile::Metal, qmp: true, - // `sched-fast-health` shortens the idle-trip print from - // 10 s to 200 ms: comparing two samples is how a spinning - // CPU is told from a halting one, and this test's whole - // capture is a handful of seconds — shorter than one - // shipped period, let alone two. - kernel_params: &["i8042-fault", "sched-fast-health"], + kernel_params: &["i8042-fault"], ..Default::default() }, ); @@ -14813,30 +13016,18 @@ fn run_machine_test( qemu.boot_log() )); } - // The in-guest reader keeps a CPU doing work, so a livelocked - // one is visible as a dead test rather than as a quiet pass. - // - // No sentinel here: the log below shows quarantine landing within - // milliseconds of `===I8042_READY===`, before a host round trip - // could possibly deliver anything, and quarantine masks the GSI — - // so nothing sent afterward, sentinel included, ever reaches the - // guest. This is `test_rs_i8042_keyboard`'s fallback deadline by - // design, not a lost sentinel. - let result = qemu.run_test_hooked( - "test_rs_i8042_keyboard", - Duration::from_secs(30), - I8042_READY, - |socket| { - qemu::qmp_send_keys(socket, &[("a", true), ("a", false)]); - }, - ); - if let Some(err) = &result.error { - return Err(format!("the guest did not survive the wedge: {err}")); - } - let Some(line) = result.serial.lines().find(|l| l.contains("i8042: quarantined")) - else { - return Err(format!("no quarantine line:\n{}", result.serial)); - }; + // One key, and the flood behind it: what ends the wait is the + // driver's own line. + let socket = qemu.qmp_socket().to_path_buf(); + qemu::qmp_send_keys(&socket, &[("a", true), ("a", false)]); + let mut serial = String::new(); + await_guest(&mut qemu, &mut serial, "the quarantine line", |c| { + c.contains("i8042: quarantined") + })?; + let line = serial + .lines() + .find(|l| l.contains("i8042: quarantined")) + .expect("the wait ended on this line"); // The count the driver actually achieved, not the word "masked" // in a format string: a quarantine that does not take the line // down leaves the CPU exposed to the next flood. @@ -14849,22 +13040,7 @@ fn run_machine_test( if masked == 0 { return Err(format!("quarantined without masking any line: {line}")); } - // "A keyboard, not a CPU" is the claim, so measure the CPU. The - // first version of this driver left the `irq_ring` record - // undrained after quarantine and produced 2685 idle-health lines - // in 5 s, against 1 on a healthy run — a regression this exact - // shape would no longer trip a *count of lines* now that - // `log_health` prints at a fixed rate whether the CPU behind it - // is halting or spinning — the vacuity the closed line-count - // entry named. What still moves at two different speeds is the - // `trips=` counter inside each line, which is not rate-limited. - if let Some((cpu, delta)) = idle_is_spinning(&result.serial) { - return Err(format!( - "cpu{cpu}'s idle-trip counter moved by {delta} within the capture — spinning, not halting" - )); - } eprintln!(" [i8042] {}", line.trim()); - eprintln!(" [i8042] idle-trip counters stayed sane — the CPU still halts"); Ok(()) } "metal_sim_window_drag" => metal_sim_window_drag(rust_bins), @@ -14960,20 +13136,21 @@ fn run_machine_test( } // And the motion reached the compositor, or the churn was against - // a pointer nobody was reading. An idle desktop composites twice - // per reporting interval (the taskbar's clock); anything above - // that is the cursor being moved. + // a pointer nobody was reading: a frame that drew the software + // cursor is one whose damage met it, and the idle desktop's only + // damage is the taskbar's clock, away from where the cursor starts. let moved = console .lines() - .filter_map(|l| l.split("compositor: frames=").nth(1)) + .filter(|l| l.contains("compositor: frames=")) + .filter_map(|l| l.split(" cursor=").nth(1)) .filter_map(|rest| rest.split_whitespace().next()) .filter_map(|n| n.parse::().ok()) - .any(|frames| frames > 2); + .any(|draws| draws > 0); if !moved { return Err(format!( - "no reporting interval composited more than the taskbar's two frames — the \ - injected motion never reached the compositor, so the churn was against a \ - pointer it was not reading:\n{console}" + "no composited frame drew the cursor — the injected motion never reached \ + the compositor, so the churn was against a pointer it was not \ + reading:\n{console}" )); } @@ -14987,8 +13164,8 @@ fn run_machine_test( } if frames(&after) < 2 { return Err(format!( - "the compositor composited {before} frame batches before {CYCLES} pointer \ - plug/unplug cycles and {} in the 20 s after them — the desktop stopped:\ + "{STALLED} the compositor composited {before} frame batches before {CYCLES} \ + pointer plug/unplug cycles and {} after them — the desktop stopped:\ \n{console}\n--- after ---\n{after}", frames(&after) )); @@ -15144,31 +13321,10 @@ fn run_machine_test( // the daemon's bound had no evidence behind it whatsoever. // // Same assertion design as `metal_sim_window_caps`: netd announces - // the cap it derived, the guest measures where the refusals start, - // and these must be the same number. - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/netcase"); - let bins: Vec<(String, Vec)> = rust_bins - .iter() - .filter(|(name, _)| name == "netd_caps") - .cloned() - .collect(); - if bins.is_empty() { - return Err("netd_caps was not built".to_string()); - } - // Headless is the profile with virtio-net; without a NIC netd - // exits before reaching anything this test is about. - let options = BootOptions { - profile: qemu::Profile::Headless, - ..Default::default() - }; - if !qemu::profile_argv(&options).iter().any(|a| a.contains("virtio-net")) { - return Err("this test needs a NIC and the profile has none".to_string()); - } - - let mut qemu = QemuInstance::boot_with_options(&config, &[], &bins, options); - - let mut console = qemu.boot_log().to_string(); - let _ = await_marker(&mut qemu, &mut console, "netd: ready, at most ", "netd to come up"); + // the cap it derived, the guest measures where the refusals start + // against the host server, and these must be the same number. + let HostRun { result, console, .. } = + netcase_against_host(rust_bins, "netd_caps", false, "")?; let Some(declared) = console .lines() .find_map(|l| l.split("netd: ready, at most ").nth(1)) @@ -15183,22 +13339,6 @@ fn run_machine_test( return Err("netd derived a cap of zero connections".to_string()); } - // The cap is passed as the burst size, not as the answer: the - // guest still measures the boundary itself. - let result = qemu.run_test( - &format!("test_rs_netd_caps {declared}"), - Duration::from_secs(120), - ); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - if result.exit_code != Some(0) { - return Err(format!( - "netd_caps exited {:?}:\n{}", - result.exit_code, result.stdout - )); - } - let Some(granted) = result .stdout .split("netd caps: ") @@ -15235,10 +13375,16 @@ fn run_machine_test( } let mut qemu = QemuInstance::boot_with_options(&config, &[], &[], options); let mut console = qemu.boot_log().to_string(); - // netd holds the claim only once it has come up on it; waiting for - // that is what makes the count below the settled one. - let _ = await_marker(&mut qemu, &mut console, "netd: ready, at most ", "netd to come up"); - console.push_str(&qemu.drain_serial(Duration::from_millis(500))); + // Both claimants have acted once netd is up on the claim and init has + // told test-runner it lost, which is what makes the count below the + // settled one. + await_marker(&mut qemu, &mut console, "netd: ready, at most ", "netd to come up")?; + await_marker( + &mut qemu, + &mut console, + "init: test-runner: pci:1af4:1041 is already claimed", + "init to refuse the second claim", + )?; let log = serial::Serial::named("boot console", console.as_str()); // Exactly one hand-over of that function. Two would be the defect @@ -15277,9 +13423,7 @@ fn run_machine_test( } let mut qemu = QemuInstance::boot_with_options(&config, &[], &[], options); let mut console = qemu.boot_log().to_string(); - let _ = - await_marker(&mut qemu, &mut console, "netd: ready, at most ", "netd to come up"); - console.push_str(&qemu.drain_serial(Duration::from_millis(500))); + await_marker(&mut qemu, &mut console, "netd: ready, at most ", "netd to come up")?; let mtree = qemu::QmpMonitor::open(qemu.qmp_socket()).human("info mtree"); let log = serial::Serial::named("boot console", console.as_str()); @@ -15490,7 +13634,6 @@ fn run_machine_test( "netd_refused_pipes" => netd_refused_pipes(rust_bins), "netd_refused_accept" => netd_refused_accept(rust_bins), "netd_held_open" => netd_held_open(rust_bins), - "netd_stalled_peer" => netd_stalled_peer(rust_bins), "netd_udp_refused" => netd_udp_refused(rust_bins), "netd_udp_any_address" => netd_udp_any_address(rust_bins), "dns_resolve" => dns_resolve(), @@ -15501,12 +13644,11 @@ fn run_machine_test( // returns on a machine with no NIC, so this is the only config // where there is a daemon to be hostile to. // - // The guest carries every verdict that needs a deadline on it — - // only it can tell a netd that answered from a netd that never - // did. The host carries the half the guest cannot see: whether - // netd *named* what it got rid of. A daemon that drops clients - // silently is one this machine cannot be asked about afterwards, - // which is the whole argument for the log lines. + // The guest carries whether netd answered. The host carries the + // half the guest cannot see: whether netd *named* what it got rid + // of. A daemon that drops clients silently is one this machine + // cannot be asked about afterwards, which is the whole argument for + // the log lines. let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/netcase"); let bins: Vec<(String, Vec)> = rust_bins .iter() @@ -15569,13 +13711,12 @@ fn run_machine_test( // share one window (`issues/build/`), which here is what makes the // daemon's side of the story readable at all. console.push_str(&result.serial); - for named in ["netd: dropping client", "netd: refusing client"] { - if !console.contains(named) { - return Err(format!( - "netd got rid of clients without a `{named}` line — a daemon that \ - drops peers silently cannot be asked what happened:\n{console}" - )); - } + let named = "netd: dropping client"; + if !console.contains(named) { + return Err(format!( + "netd got rid of clients without a `{named}` line — a daemon that drops \ + peers silently cannot be asked what happened:\n{console}" + )); } serial::Serial::named("boot console", console.as_str()).must_be_clean()?; eprintln!(" [netcase] {refused} hostile frames refused, netd named every peer it dropped"); @@ -15679,10 +13820,6 @@ fn run_machine_test( console.push_str(&result.serial); serial::Serial::named("boot console", console.as_str()).must_be_clean()?; eprintln!(" [netcase] every child started in the directory its spawn named"); - // The spawn latency by cwd, which is the guest's to measure and nobody's to assert. - for line in result.stdout.lines().filter(|l| l.contains("a spawn and wait")) { - eprintln!(" [netcase] {line}"); - } Ok(()) } "input_claim_absent" => { @@ -16458,6 +14595,21 @@ fn window_count(log: &str) -> u64 { .unwrap_or(0) } +/// Whether posts landed in held windows while the canary ran, not only before, +/// off a metal boot's kernel log split at the canary's spawn record. +/// +/// **Judged on the T14 and in no QEMU guest**: the actuator holds each window +/// for a budget of its own clock, so how many a post lands in is how much of the +/// host the guest had. +fn window_held_on_metal(kernel: &serial::Serial) -> Result<(), String> { + let text = kernel.text(); + let spawned = "spawn: /system/bin/test_rs_blocking_read_stress "; + let at = text + .find(spawned) + .ok_or_else(|| format!("no `{spawned}` record: the canary never ran\n{text}"))?; + window_held(&text[..at], &text[at..]) +} + /// Whether posts landed in held windows while the canary ran, not only before. /// /// The holds during the run are at least the last count said during it, less @@ -17020,6 +15172,20 @@ fn acpi_table_inventory(log: &str) -> Result<(), String> { Ok(()) } +/// The LAPIC timer and the TSC each calibrated to a frequency. +fn timer_calibration(log: &str) -> Result<(), String> { + let lapic_hz = number_between(log, "ticks/10ms, so ", "Hz")?; + if lapic_hz == 0 { + return Err("the LAPIC timer calibrated to no frequency at all".to_string()); + } + let measured = number_between(log, "clock: TSC measured ", "Hz against the HPET")?; + if measured == 0 { + return Err("the TSC calibrated to no frequency at all".to_string()); + } + eprintln!(" [timer] TSC {measured}Hz measured; LAPIC {lapic_hz}Hz"); + Ok(()) +} + /// The TSC the whole machine is timed by, against the frequency the part itself /// states. /// @@ -17027,30 +15193,24 @@ fn acpi_table_inventory(log: &str) -> Result<(), String> { /// is derived from the HPET calibration, so it can only agree with itself; /// CPUID leaf 15H's crystal ratio and leaf 16H's base frequency are the CPU's /// own statement, arrived at by neither the HPET nor the counting loop. A part -/// that states neither is a fact about the part, not a failure — `qemu64`, this -/// host's guest CPU, is one — so the ppm bound is asserted only where a -/// statement exists. -fn timer_calibration(log: &str) -> Result<(), String> { +/// that states neither is a fact about the part, not a failure, so the ppm +/// bound is asserted only where a statement exists. Judged on metal only: the +/// calibration is a span of a clock, and a guest's clock runs while its host +/// has the vCPU. +fn tsc_agrees_with_cpuid(log: &str) -> Result<(), String> { /// One percent, which is the widest two timebases can differ and still be /// counting the same second. A refusal and not a measurement: it catches a /// machine whose HPET and CPUID have stopped agreeing at all, and nothing /// narrower is true of every part this kernel may boot on. const CEILING_PPM: u64 = 10_000; - let lapic_hz = number_between(log, "ticks/10ms, so ", "Hz")?; - if lapic_hz == 0 { - return Err("the LAPIC timer calibrated to no frequency at all".to_string()); - } let measured = number_between(log, "clock: TSC measured ", "Hz against the HPET")?; - if measured == 0 { - return Err("the TSC calibrated to no frequency at all".to_string()); - } let Ok(stated) = number_between(log, "CPUID states ", "Hz,") else { let why = log .lines() .find(|l| l.contains("CPUID leaves 15H and 16H")) .ok_or("neither a stated frequency nor the record saying there is none")?; - eprintln!(" [timer] TSC {measured}Hz, LAPIC {lapic_hz}Hz — {}", why.trim()); + eprintln!(" [timer] TSC {measured}Hz — {}", why.trim()); return Ok(()); }; let ppm = number_between(log, "Hz, ", "ppm apart")?; @@ -17060,9 +15220,7 @@ fn timer_calibration(log: &str) -> Result<(), String> { {ppm}ppm apart, over the {CEILING_PPM}ppm this bound allows" )); } - eprintln!( - " [timer] TSC {measured}Hz measured, {stated}Hz stated, {ppm}ppm apart; LAPIC {lapic_hz}Hz" - ); + eprintln!(" [timer] TSC {measured}Hz measured, {stated}Hz stated, {ppm}ppm apart"); Ok(()) } @@ -17144,107 +15302,7 @@ fn tlb_shootdown_cost(log: &str, cpus: u32) -> Result<(u64, u64), String> { Ok((p50, p99)) } -/// What a waiter in the real-time band pays to be woken, and — the half this -/// suite exists for — that the number reaches a machine with no console. -/// -/// `tests/latencycase` is a job-list boot: no host types at it, the runner runs -/// `cyclictest` and then the scheduler stress suite, and the last job hands the -/// machine back to firmware. That is the T14's own shape, so both channels are -/// read here: the console, which the T14 does not have, and the kernel's -/// `exit: pid=N code=N` record on the log volume, which is the only word -/// a program gets off that machine. **They must carry the same number**, or the -/// metal readback is reporting something the guest did not measure. -fn latency_wake(rust_bins: &[(String, Vec)]) -> Result<(), String> { - /// **Derived from the instrument, not from this host.** A p99 at the - /// histogram's last bucket is a floor and not a measurement, so what is - /// asserted here is that the figure is one — TCG under a twelve-wide suite - /// is no latency instrument, and the number this measures on hardware is - /// the T14's, priced as `latency.p99_us` in `tests/metal-profile.toml`. - const HISTOGRAM_US: u64 = 4096; - const WAIT: Duration = Duration::from_secs(60); - - let config = compile::repo_root().join("tests/latencycase/system.toml"); - let case = config.parent().expect("system.toml has a directory"); - // The two the config's job list names, and not the whole set: latencycase's - // ROOT is `logd`, the runner and `toybox`, and every unnamed binary staged - // beside them is image the boot pays to write. - const JOBS: &[&str] = &["cyclictest", "sched_stress"]; - let bins: Vec<(String, Vec)> = - rust_bins.iter().filter(|(name, _)| JOBS.contains(&name.as_str())).cloned().collect(); - if bins.len() != JOBS.len() { - return Err(format!("the suite built {} of {JOBS:?}", bins.len())); - } - - // Built here and written out, because the boot deletes the image it built - // and the log volume is read after the guest is gone. - let image_path = common::lane::dir().join("latencycase-boot.img"); - let image = common::qemu::build_boot_image(case, &[], &bins, &[]); - fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - let (start, len) = common::volumes::log_extent(&image, &image_path)?; - - let mut qemu = QemuInstance::boot_with_options( - case, - &[], - &bins, - BootOptions { qmp: true, boot_image: Some(qemu::Staged::Written(image_path.clone())), ..Default::default() }, - ); - let mut stop = common::qemu::QmpShutdown::open(qemu.qmp_socket(), qemu.budget(WAIT)); - let reason = stop.reason(); - let console = qemu.drain_serial(WAIT); - drop(qemu); - - if reason.as_deref() != Some("guest-reset") { - return Err(format!( - "the job list did not hand the machine back to firmware ({reason:?})\n{console}" - )); - } - - // The console's copy: the marker the runner writes around every job. - let distribution = console - .lines() - .find(|l| l.contains("cyclictest: ")) - .ok_or("cyclictest printed no distribution")?; - let printed: i64 = field_between(&console, "===TEST_END test_rs_cyclictest exit=", "===")? - .parse() - .map_err(|_| format!("cyclictest's exit marker carries no number:\n{console}"))?; - if printed < 0 { - return Err(format!("cyclictest refused rather than measuring: {distribution}")); - } - if !console.contains("===TEST_END test_rs_sched_stress exit=0===") { - return Err(format!("the scheduler stress suite did not pass:\n{console}")); - } - - // The volume's copy, which is the whole of what a machine with no serial - // port gives back. - let (name, log) = common::volumes::newest_log(&image_path, start, len)?; - let text = String::from_utf8_lossy(&log); - let record = text - .lines() - .find(|l| l.contains("exit: test_rs_cyclictest pid=")) - .ok_or_else(|| format!("{name} carries no `exit: test_rs_cyclictest` record\n{text}"))?; - let recorded: i64 = field_between(record, " code=", " ")? - .parse() - .map_err(|_| format!("the exit record carries no number: {record}"))?; - if recorded != printed { - return Err(format!( - "the console says cyclictest exited {printed} and {name}'s kernel record says \ - {recorded} — the metal channel and the QEMU one disagree" - )); - } - bootlog::verdict(&text).map_err(|unfit| format!("{name}: {unfit}\n{text}"))?; - - if printed as u64 >= HISTOGRAM_US { - return Err(format!( - "the p99 landed in the histogram's last bucket, so {printed}us is a floor and not a \ - measurement: {distribution}" - )); - } - eprintln!(" [latency] {}", distribution.trim()); - eprintln!(" [latency] p99 {printed}us, off the stick's own `exit: cyclictest` record too"); - Ok(()) -} - -/// [`latency_wake`]'s half that exists on a machine with no console: the p99, +/// `latency_wake` on a machine with no console: the p99, /// off the kernel's own exit record, against the ceiling the metal profile /// holds. /// @@ -17409,54 +15467,6 @@ fn smp_failed_ap_leaves_no_hole( Ok(()) } -/// The largest an idle-trip counter (`kernel/src/scheduler.rs`'s -/// `IDLE_TRIPS`, printed as `trips=` on `sched: cpu=`'s now rate-limited -/// line) may move for one CPU across a captured serial before -/// [`idle_is_spinning`] calls it spinning rather than halting. -/// -/// Two orders of magnitude above the worst real trip delta this suite has -/// measured on a healthy `i8042_quarantine` run on this host (`cargo test -- -/// i8042_quarantine`, 2026-08-17: cpu1 moved by 2 within the capture — one -/// print at readiness, one roughly ten seconds later, the rate limit's own -/// cadence) and well under the shape of the regression this gate exists for: -/// the first quarantine driver's undrained `irq_ring` produced 2685 printed -/// lines in 5 s under the *old*, unthrottled-per-1000-trips counter — at -/// least 2,685,000 trips in that one window alone. -const MAX_IDLE_TRIP_DELTA: u64 = 100_000; - -/// Whether any CPU's idle-trip counter moved by more than [`MAX_IDLE_TRIP_DELTA`] -/// across `serial`, and which one if so. -/// -/// **Not the same question a count of `sched: cpu=` lines answers**, and -/// deliberately not: `log_health` prints at most once per -/// `SNAPSHOT_INTERVAL_NS` now, so a CPU that spins through idle and one that -/// halts cleanly between rare wakes produce the same number of *lines* — -/// only the counter inside each line still moves at the two different -/// speeds, which is what made the line count vacuous. -/// Per CPU, and the worst offender rather than every one, because a spin on -/// one CPU must not be hidden by averaging it against another CPU's healthy -/// rate. -fn idle_is_spinning(serial: &str) -> Option<(u32, u64)> { - let mut spread: BTreeMap = BTreeMap::new(); - for line in serial.lines() { - let Some(rest) = line.split("sched: cpu=").nth(1) else { continue }; - let Some((id, rest)) = rest.split_once(' ') else { continue }; - let Some(trips) = rest.split("trips=").nth(1).and_then(|t| t.split_whitespace().next()) - else { - continue; - }; - let (Ok(id), Ok(trips)) = (id.parse::(), trips.parse::()) else { continue }; - spread - .entry(id) - .and_modify(|(min, max)| { - *min = (*min).min(trips); - *max = (*max).max(trips); - }) - .or_insert((trips, trips)); - } - spread.into_iter().map(|(id, (min, max))| (id, max - min)).find(|&(_, delta)| delta > MAX_IDLE_TRIP_DELTA) -} - /// Everything the driver derived its DMA pool from, off the two lines it /// prints. Reading these is what makes a test see a *derivation* rather than /// the fact that some number was printed: every fixed cap that ever stood @@ -17821,10 +15831,6 @@ fn build_test_registry( // Writes the child's whole image through bcachefs before it can run // it, which is the only thing here that is not a spawn. "disk_backtrace" => Duration::from_secs(15), - // Its verdict is that a parked waiter woke, so the failing run is - // the slow one: it spends its own patience before reporting, and - // the report is worth more than the harness's timeout message. - "inbox_cancel_wakes" => Duration::from_secs(30), _ => Duration::from_secs(5), }; tests.push(TestDef { @@ -17993,9 +15999,10 @@ impl Outcome { } } - /// Whether this red is a blown liveness guard rather than an answer. + /// Whether this red is a blown liveness guard or the backstop rather than an + /// answer. fn stalled(&self) -> bool { - self.reason.as_deref().is_some_and(|r| r.contains(STALLED)) + self.reason.as_deref().is_some_and(|r| r.contains(STALLED) || r.contains(TIMED_OUT)) } } @@ -18009,6 +16016,139 @@ fn headline(reason: Option<&str>) -> String { reason.unwrap_or("check failed").lines().next().unwrap_or("check failed").to_string() } +/// Two controllers, both with a codec that answers. +/// +/// The kernel binds neither and names both. A first-match bind would go green +/// on every test that has one controller, so this is the arm that makes the +/// rule tested rather than merely written. +fn hda_two_live_refused( + test_config: &Path, + c_bins: &[(String, Vec)], + rust_bins: &[(String, Vec)], +) -> Result<(), String> { + /// The kernel's word for soundd ending, and soundd's for taking the + /// controller, which end the wait with this test's own sentence rather than + /// the ceiling. + const SOUNDD_GONE: &str = "exit: soundd"; + const HDA_PATH: &str = "soundd: hda path configured in"; + let mut qemu = QemuInstance::boot_with_options( + test_config, + c_bins, + rust_bins, + BootOptions { profile: qemu::Profile::HdaTwoLive, ..Default::default() }, + ); + // The refusal is a kernel boot line and is in the capture already; soundd's + // answer to it is a userland line that races the ready marker. + let mut text = qemu.boot_log().to_string(); + let stalled = await_guest(&mut qemu, &mut text, "soundd to say which sink it took", |seen| { + [common::audio::NULL_SINK, SOUNDD_GONE, HDA_PATH].iter().any(|said| seen.contains(said)) + }) + .err(); + let log = serial::Serial::named("boot console", text.as_str()); + log.must_say("hda: 00:")?; + log.must_say("has a live link (statests=")?; + log.must_say("controllers answer on this machine")?; + log.must_say("refused by name, no HDA audio")?; + log.must_not_say("bound, statests=")?; + // The machine still boots and still has a sink: absence of hardware is a + // routing state, and a refusal must not be a machine that will not run. + // **And it is the bind's absence and not a second spelling of the line + // above**: init claims each class before it spawns, and soundd reaches the + // null sink only where the endowment is missing — so this line requires + // `device::try_claim(HdaAudio)` to have answered `Absent`. + log.must_say(common::audio::NULL_SINK).map_err(|why| match stalled { + Some(stall) => format!("{stall}\n{why}"), + None => why, + })?; + log.must_say("Boot: complete")?; + log.must_be_clean() +} + +/// **A kernel that has declared itself corrupt runs nothing else.** +/// `halt_all_cpus` stops the other CPUs before anything else it does; a fatal +/// path that waited first — for a log, a drain, anything — would leave every +/// other CPU running userland under it. `test_rs_panic_halts_first` keeps three +/// siblings making kernel records while its main thread goes fatal. +/// +/// **QEMU is the judge, and no clock is in it.** Once the fatal path has said +/// its line past the stop, `panic_reboot::arm`'s, every vCPU but the one that +/// went fatal must show [`qemu::stopped_cpus`]' `cli; hlt`. The fatal one holds +/// its panel under the shipped minute, so the machine is still there to ask. +fn panic_halts_the_others_first( + test_config: &Path, + c_bins: &[(String, Vec)], + rust_bins: &[(String, Vec)], +) -> Result<(), String> { + const SMP: usize = 4; + let mut qemu = QemuInstance::boot_with_options( + test_config, + c_bins, + rust_bins, + BootOptions { + smp: SMP as u32, + kernel_features: ACTUATOR_KERNEL, + qmp: true, + ..Default::default() + }, + ); + writeln!(qemu.stdin_mut(), "run test_rs_panic_halts_first").map_err(|e| format!("stdin: {e}"))?; + qemu.flush_stdin(); + let mut console = String::new(); + await_guest(&mut qemu, &mut console, "the fatal path's line past the stop", |c| { + fatal_past_the_stop(c).is_some() + })?; + let fatal = fatal_past_the_stop(&console).expect("awaited above"); + let before = &console[..console.find(FATAL_HALT_NONCE).expect("awaited above")]; + // Non-vacuity: another CPU was making records up to the fatal one. + if !before.lines().any(|l| l.contains(RETIRED_RECORD) && record_cpu(l).is_some_and(|cpu| cpu != fatal)) { + return Err(format!( + "no other CPU made a record before the fatal one on cpu{fatal}, so nothing was running \ + to be stopped\n{console}" + )); + } + let mut monitor = qemu::QmpMonitor::open(qemu.qmp_socket()); + // A guard and never a verdict: a vCPU the host has not run yet has not + // taken the stop, and this is how long it is waited for. + let give_up = Instant::now() + qemu::GUEST_QUIET; + loop { + let stopped = qemu::stopped_cpus(&monitor.human("info registers -a")); + if stopped.len() == SMP && stopped.iter().filter(|&&cpu| cpu).count() >= SMP - 1 { + break; + } + if Instant::now() >= give_up { + return Err(format!( + "{STALLED} waiting for the other CPUs to halt after the fatal path on cpu{fatal} \ + stopped them — QEMU shows each vCPU in `cli; hlt` as {stopped:?}\n{console}" + )); + } + console.push_str(&qemu.drain_serial(Duration::from_millis(200))); + } + eprintln!(" [panic] the fatal path on cpu{fatal} left every other CPU in `cli; hlt`"); + Ok(()) +} + +/// The record each sibling of `test_rs_panic_halts_first` makes, over and over. +const RETIRED_RECORD: &str = "syscall 26 is retired"; + +/// The CPU a kernel record is stamped with: `[kernel cpu]`. +fn record_cpu(line: &str) -> Option { + let head = line.split_once("[kernel ")?.1.split_once(']')?.0; + head.split_once(" cpu")?.1.split(' ').next()?.parse().ok() +} + +/// The CPU that went fatal, once the console carries its line past +/// `stop_other_cpus` — `panic_reboot::arm`'s, in either of its two words. +fn fatal_past_the_stop(console: &str) -> Option { + const PAST_THE_STOP: [&str; 2] = ["panic: rebooting in", "panic: holding this panel"]; + let lines: Vec<&str> = console.lines().collect(); + let nonce = lines.iter().position(|l| l.contains(FATAL_HALT_NONCE))?; + let fatal = record_cpu(lines[nonce])?; + lines[nonce..] + .iter() + .any(|l| record_cpu(l) == Some(fatal) && PAST_THE_STOP.iter().any(|word| l.contains(word))) + .then_some(fatal) +} + /// What a run has established, as it establishes it. /// /// One place rather than five counters in `main`, because the interesting part @@ -18018,9 +16158,8 @@ fn headline(reason: Option<&str>) -> String { struct Tally { passed: usize, failures: Vec<(String, String)>, - /// The subset of `failures` whose guard expired rather than whose assertion - /// failed, by name. Red like any other — and named apart, because a run - /// that never got the guest going has measured the host and not the tree. + /// The subset of `failures` whose ceiling expired rather than whose + /// assertion failed, by name. Red like any other, and named apart. stalls: Vec, invalid: Vec<(String, Duration)>, /// What each tier held back, under the flag that runs it. Not a verdict and @@ -18126,7 +16265,7 @@ impl Tally { } if !self.stalls.is_empty() { say(format!( - "{} of those reds are blown liveness guards, not answers: {}", + "{} of those reds are the ceiling: {}", self.stalls.len(), self.stalls.join(", ") )); @@ -18197,14 +16336,6 @@ fn shared_kernel(name: &str) -> &'static [&'static str] { } } -/// The other half of the same question: how wide that boot is. -fn shared_smp(name: &str) -> u32 { - if ACTUATOR_TESTS.contains(&name) { - ACTUATOR_SMP - } else { - BootOptions::default().smp - } -} /// The binaries and config every task boots with. struct Bins<'a> { @@ -18281,7 +16412,6 @@ fn run_task(task: Task<'_>, bins: &Bins<'_>, report: &std::sync::mpsc::Sender, bins: &Bins<'_>, report: &std::sync::mpsc::Sender std::path::PathBuf { fn read_durations(path: &Path, out: &mut BTreeMap) { let Ok(text) = fs::read_to_string(path) else { return }; for line in text.lines() { - // `