From 4505f872d080e6b9485aeea523f3c02169b484ce Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:06:40 +0200 Subject: [PATCH 01/18] =?UTF-8?q?tests:=20the=20measured=20schedule=20?= =?UTF-8?q?=E2=80=94=20Fast=20is=20every=20PR,=20Nightly,=20Weekly;=2017?= =?UTF-8?q?=20never-caught=20tests=20deleted?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The owner adopted a schedule from what each guest test has caught: 426 of 524 CI-scheduled guest tests never caught a real defect and were 87% of guest time. - `src/tiers.rs`: `Tier` gains `Weekly`, and a run's `Reach` (plain, `--nightly`, `--weekly`) selects its own tier and every narrower one. A `Schedule` is every registered name at its one tier: a name registered twice is refused by name, and asking the tier of an unregistered one is refused by name. The harness builds it from the shared boot's discovered binaries and the three declared registries, which replaces `check_no_collisions` and `check_registration`'s duplicate loop; the redlist asks it what is registered, and `--list` prints every test at its tier. - `src/testargs.rs`: `--weekly`; `--nightly --weekly` is refused (the second would be read by nothing), and either reach beside `--audio-gate` or `--metal` is refused — `--metal --nightly` used to be accepted and its `--nightly` dropped. - `nightly.yml` runs Monday to Saturday at 03:00 and Sunday at 03:00; the guest job reads which schedule started it (`src/ci.rs`'s `guest_reach`) and runs `--nightly` or `--weekly`, and a schedule the file does not declare is a red step. Same jobs, one workflow. - Every row's tier is its bucket's: Fast is the shared boot and the 22 machine and screen tests of the every-PR bucket. A boot group takes its most frequent member's tier, so six weekly-bucket riders stay Nightly on the boot they share. The audio and timing rows `wt/toyos-notiming` rewrites keep the tier they had, and the fix-first rows are untouched. - Deleted with their harness code, guest binaries, duration rows, redlist rows and the issues that existed only for them: console_line_atomicity, fat_backing_revoked, home_budget_refusal_retried, i8042_keyboard, klogd_panic_halts, launcher_refusals, log_conservation_smp4, log_conservation_smp8, quiesce_dump_holds_the_stopped, quiesce_wakes_on_the_last_exit, screen_early_panic, so_cache_refusals, sshd_fail_closed, usb_disk_index_stable, wall_clock_file, xhci_hid_break, xhci_hotplug. `quiesce_last` loses the exiting thread only the last-exit test held. - `syscall_window_nmi`, `syscall_window_nmi_controls` and `dump_nmi_probe` stay, Weekly: what they guard is the frame an x86 CPU builds for an NMI at CPL 0 on a user `rsp`, which only a CPU produces. - Stale tier reasons in the registration comments are deleted. Filed: the seven kernel actuators no test arms any more, and three harness helpers nothing calls. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/nightly.yml | 10 +- issues/README.md | 4 +- ...ty-loses-five-of-a-thousand-lines-on-ci.md | 25 - ...st-its-serial-ready-beside-other-guests.md | 34 - ...st-its-serial-ready-beside-other-guests.md | 4 - ...ess-carries-three-helpers-nothing-calls.md | 24 + ...tors-lost-the-only-test-that-armed-them.md | 30 + ...refusal-retried-is-red-on-every-nightly.md | 31 - ...hing-enumerates-on-the-first-controller.md | 20 - ...ped-reds-wide-with-usb-transport-breaks.md | 47 - ...sals-saw-the-kernel-refuse-nothing-once.md | 19 - src/ci.rs | 64 +- src/redlist.rs | 20 - src/testargs.rs | 42 +- src/tiers.rs | 165 +- tests/common/console.rs | 257 +--- tests/common/faults.rs | 7 +- tests/common/iommu.rs | 5 - tests/common/lan.rs | 3 +- tests/common/logread.rs | 25 - tests/common/power.rs | 87 -- tests/common/storage.rs | 203 --- tests/common/usb.rs | 686 --------- tests/common/volumes.rs | 153 +- tests/common/wallclock.rs | 143 -- tests/netcase/system.toml | 12 +- tests/quiescelastcase/system.toml | 4 +- tests/test-durations | 15 - .../src/bin/abuse_listener_hijack.rs | 3 +- .../src/bin/console_line_atomicity.rs | 204 --- .../src/bin/fat_backing_revoked.rs | 153 -- .../src/bin/home_fsync_budget.rs | 30 - .../src/bin/launcher_refusals.rs | 291 ---- .../toyos-rust-tests/src/bin/quiesce_last.rs | 28 +- .../src/bin/so_cache_policy.rs | 160 -- tests/toyos.rs | 1339 +++++------------ 36 files changed, 705 insertions(+), 3642 deletions(-) delete mode 100644 issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md delete mode 100644 issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md create mode 100644 issues/build/the-harness-carries-three-helpers-nothing-calls.md create mode 100644 issues/design-debt/seven-kernel-actuators-lost-the-only-test-that-armed-them.md delete mode 100644 issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md delete mode 100644 issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md delete mode 100644 issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md delete mode 100644 issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md delete mode 100644 tests/toyos-rust-tests/src/bin/console_line_atomicity.rs delete mode 100644 tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs delete mode 100644 tests/toyos-rust-tests/src/bin/home_fsync_budget.rs delete mode 100644 tests/toyos-rust-tests/src/bin/launcher_refusals.rs delete mode 100644 tests/toyos-rust-tests/src/bin/so_cache_policy.rs diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index cba134ab63e..cb177f7d0fa 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -1,13 +1,15 @@ name: nightly # Everything that boots a guest, the host gate again to write the cache the -# merge queue restores, and portability: on main every night, and on any branch by -# `gh workflow run nightly.yml --ref `. Every job's logic is -# `cargo run -- --ci ` (src/ci.rs); this file says where each one runs. +# merge queue restores, and portability: on main every night, the weekly tier +# on Sundays, and on any branch by `gh workflow run nightly.yml --ref `. +# Every job's logic is `cargo run -- --ci ` (src/ci.rs); this file says +# where each one runs. on: schedule: - - cron: '0 3 * * *' + - cron: '0 3 * * 1-6' + - cron: '0 3 * * 0' workflow_dispatch: # Never cancelled: `build` may be an hour into a bootstrap. diff --git a/issues/README.md b/issues/README.md index 646c3154d8f..1ee48f15ede 100644 --- a/issues/README.md +++ b/issues/README.md @@ -155,5 +155,5 @@ owns "every policy about files — where they go, what they are called, how many there are, what happens when the stick stops answering" (`userland/logd/src/main.rs:1-10`). Gated by `esp_filesystem`, `kernel_log_file`, `log_backing_read_error`, -`boot_volume_metadata_error`, `log_partition_layout`, `log_partition_identity` -and `wall_clock_file`, plus `toybox_cp_volume`. +`boot_volume_metadata_error`, `log_partition_layout` and +`log_partition_identity`, plus `toybox_cp_volume`. diff --git a/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md b/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md deleted file mode 100644 index 432fda64247..00000000000 --- a/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md +++ /dev/null @@ -1,25 +0,0 @@ ---- -status: expected-red -kind: defect -opened: 2026-08-20 ---- - -# `console_line_atomicity` loses five of a thousand lines on CI, one guest per machine - -First sighting on the CI instrument: PR #166 run `32364721784`, `guest (10)`, -`writer A declared 1000 whole lines and the capture carries 995`, `ALONE: -GREEN` in the same job. The diff it rode on is an issues-and-prose audit, so -the tree is not a suspect. - -CI runs one guest per machine with `--jobs 1`, so whatever loses five of a -writer's thousand lines there is not host contention between suites — which -sharpens the question the test was built to ask rather than settling it: the -loss is inside one guest's own console path. - -Split out of `issues/build/parallel-tests-red-under-other-suites.md`, whose -rate table was never this test's — CI's single-guest-per-machine shards rule -out the contention shape that file is about. - -**Exit condition.** The lost lines' cause is fixed, and -`console_line_atomicity` green on CI's `guest` shards, one guest per machine. -Owner: orchestrator. diff --git a/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md b/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md deleted file mode 100644 index 03940d2f0fe..00000000000 --- a/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -status: expected-red -kind: finding -opened: 2026-09-26 ---- - -# `quiesce_wakes_on_the_last_exit` lost its serial READY beside other guests - -Fast tier at `a4b44a91` (PR #510's branch; `toyos-sshupload`'s suite held build -slots at the same time): `QEMU died before ===READY=== (status: exit 0)`. The -guest did everything the test reads: `quiesce-last-exit: quiesce-last is held`, -`Syncing filesystems...`, `stop: 4 of 4 userland thread(s) stopped ... over 3 -sweep(s)`, `usb-quiesce: disk 0 SYNCHRONIZE CACHE ok`, `Rebooting.` But the uart -captured `nothing at all`. Before it, `usb-storage: 00:02.0 slot 1 transport -broke on SCSI 0x28: no answer in the data phase in 2000 ms`, and the -test-runner's spawn reported `layout=2069ms`. The harness's re-run alone was -green in 3 s, 2 sweeps. - -Not shown: why the uart saw none of the test-runner's output in a boot whose -console carried the whole stop. The READY wait reports the missing marker, not -its cause. - -Again in the fast tier at `8846c021` (PR #524's branch, alone on the host): -the same `QEMU died before ===READY=== (status: exit 0)` with `uart: nothing -at all`, after `stop: 5 of 5 userland thread(s) stopped`, `usb-quiesce: disk 0 -SYNCHRONIZE CACHE ok` and `Rebooting.`; this time no usb-storage transport -break before it. The re-run alone was green. - -On main's lineage, with `the stopped-boot drain carried no kernel output at all -(38 bytes)`: f231c43e (run 36280285913, `guest (3)`), 67a430c8 (run -36285169430) and a55d62c6 (run 36287592139). - -**Exit**: the marker waited for where the boot's reboot cannot race it, and the -test green in a fast tier beside other guests. Owner: orchestrator. diff --git a/issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md b/issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md index 8af3bd54ce3..b89ef302104 100644 --- a/issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md +++ b/issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md @@ -15,10 +15,6 @@ ok` and `Rebooting.`, then `shutdown: /log did not answer in 2000ms`; the uart captured `nothing at all`. The harness's re-run alone was green in 2 s. `cargo run -- --known-red` answers NO. -The same shape as -`issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md`, -on the sibling arm. - **Exit**: a cause for the empty uart on a boot that rebooted as designed, or the marker waited for where the boot's reboot cannot race it. diff --git a/issues/build/the-harness-carries-three-helpers-nothing-calls.md b/issues/build/the-harness-carries-three-helpers-nothing-calls.md new file mode 100644 index 00000000000..3986bd290c1 --- /dev/null +++ b/issues/build/the-harness-carries-three-helpers-nothing-calls.md @@ -0,0 +1,24 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# The harness carries three helpers nothing calls + +`tests/common/mod.rs` puts `#[allow(dead_code)]` on nearly every module, so +the compiler says nothing about a harness helper whose last caller went. With +those attributes taken off, `cargo check --tests` on `wt/toyos-schedule` names +three that no test reaches, and none of the three was called by anything that +branch deleted: + +- `tests/common/storage.rs`: `FileBlocks::whole`; +- `tests/common/qemu.rs`: the field `usb_images` and its method `usb_images`; +- `tests/common/qemu.rs`: `QmpDevices::set_link`. + +The same pass names `tests/common/audio.rs`'s `completions` and `clients` and +`tests/common/stats.rs`'s `fisher_reject_at`, which the audio branch +(`wt/toyos-notiming`) deletes with their modules. + +**Exit**: the three deleted, and the module-wide `allow(dead_code)` replaced by +nothing, so the next orphan is a warning the host gate denies. diff --git a/issues/design-debt/seven-kernel-actuators-lost-the-only-test-that-armed-them.md b/issues/design-debt/seven-kernel-actuators-lost-the-only-test-that-armed-them.md new file mode 100644 index 00000000000..ed22a9a792c --- /dev/null +++ b/issues/design-debt/seven-kernel-actuators-lost-the-only-test-that-armed-them.md @@ -0,0 +1,30 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# Seven kernel actuators lost the only test that armed them + +The test schedule deleted the guest tests that never caught a defect, and seven +actuators in `kernel/src/actuator.rs` were armed by nothing else: no row of +`tests/toyos.rs`, `tests/common/` or `tests/metal-profile.toml` names them now. + +| actuator | the deleted test | the kernel site it guards | +|---|---|---| +| `klogd-panic` | `klogd_panic_halts` | `kernel/src/log/console.rs` | +| `usbd-panic` | `klogd_panic_halts` | `kernel/src/drivers/xhci/usbd.rs` | +| `quiesce-dump` | `quiesce_dump_holds_the_stopped` | `kernel/src/sched/dump.rs` | +| `quiesce-last-exit` | `quiesce_wakes_on_the_last_exit` | `kernel/src/quiesce.rs`, `toyos-quiesce/src/lib.rs` | +| `so-cache-tiny` | `so_cache_refusals` | `kernel/src/elf/cache.rs` | +| `xhci-hid-break-first` | `xhci_hid_break` | the xHCI HID completion path | +| `xhci-hid-break-late` | `xhci_hid_break` | the xHCI HID completion path | + +An actuator nothing arms is kernel code no run executes: the `test-actuators` +kernel still compiles each arm, and no verdict reads what it stages. + +Found by: `git grep` for each actuator's name over `tests/` and `src/` after the +deletions on `wt/toyos-schedule`, every count 0. + +**Exit**: each actuator and the code only it reaches deleted from the kernel, +or a test that arms it registered again. Owner: orchestrator. diff --git a/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md b/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md deleted file mode 100644 index 25d447d835b..00000000000 --- a/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md +++ /dev/null @@ -1,31 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# `home_budget_refusal_retried` is red on every nightly - -`home_budget_refusal_retried` (nightly tier) is red, and red alone, on the -nightlies of three trees, each time with the same two shapes: -- main at e8d7c9c0 (run 36228604597); -- PR #524's branch at 8c5be843 (run 36273557690); -- main at fd62f567 (run 36278449733, guest (12)). - -The two shapes: `no fsync: /home/... durable on attempt line — the retry -never ran`, and `Boot timed out waiting for ===READY===`. In the second, the -console ends at the loader's `Applied 5483 relocations`, with -`fsync-budget-spent` on the boot parameter line. It was green at 3f46a019 -(run 36111884575). - -It is green alone on a dev host at f231c43e (`cargo test --test toyos-build --- --nightly home_budget_refusal_retried` EXIT=0). -`cargo run -- --known-red home_budget_refusal_retried` answers NO. - -`fsync-budget-spent` is the machine-wide actuator that -issues/boot-media/partition-claim-gives-up-reds-beside-other-guests-and-is-green-alone.md -names as racing whichever fsync the boot reaches first. Not shown: whether -that race is this test's cause. - -**Exit**: the cause shown on a red run's log, and the test green on a -nightly. diff --git a/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md b/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md deleted file mode 100644 index 8ad470d8c29..00000000000 --- a/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md +++ /dev/null @@ -1,20 +0,0 @@ ---- -status: expected-red -kind: defect -opened: 2026-08-08 ---- - -# `usb_disk_index_stable` reds 1 of 5 on CI: nothing enumerated on the first controller - -From the twelve-shard CI probe (`probe-rate.yml` run `31258202923`, tree -`f8f73e1`, five reps of the exact `ci.yml` configuration): `usb_disk_index_stable` -red 1 of 5, shard 2, `Sched::Parallel`, `nothing enumerated on the first -controller`. Re-taken 2026-09-04 against the last three nightly `ci` runs on -`main` (`33485669019`, `33603832656`, `33728852421`): of the eleven names that -probe found, only this one still reds. - -Split out of `issues/hardware/eleven-names-red-on-ci.md`, which covers eleven -names and has no exit condition for this one in particular. - -**Exit condition.** The cause of the empty first controller is fixed, and -`usb_disk_index_stable` green on CI's `guest` shards. Owner: orchestrator. diff --git a/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md b/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md deleted file mode 100644 index 25c97e68686..00000000000 --- a/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md +++ /dev/null @@ -1,47 +0,0 @@ ---- -status: expected-red -kind: defect -opened: 2026-09-25 ---- - -# quiesce_dump_holds_the_stopped reds wide, with two USB transport breaks, and is green alone - -Seen once, in the fast tier on the logd branch (PR #492), dev host: red wide -with `QEMU never reported stopping: the guest asked for a reboot and stayed -up`, then `ALONE ... GREEN`. - -The boot's console shows the stick's transport breaking twice on `SCSI 0x2a` -(`no answer in the status phase in 2000 ms`, recovered each time), then -`quiesce_writers: 4 of 6 writers reached their loop in 5s` and -`test_rs_quiesce_writers exit=1`. The branch touches logd, netd and the -console, not the USB stack, quiesce or its config. - -**Recurrence, PR #524's fast tier at `396f5b4d`, dev host.** Red wide after -325 s with the same `QEMU never reported stopping: the guest asked for a -reboot and stayed up`, then `ALONE ... GREEN` in 4 s. The capture has the same -`quiesce_writers: 4 of 6 writers reached their loop in 5s` and -`test_rs_quiesce_writers exit=1`, and this time no transport break on the -stick. While it ran, the host's 12 guest slots were full, shared with the -suites of five other worktrees (`toyos-lld`, `toyos-tcp`, `toyos-update`, -`toyos-logtrack`, `toyos-guiplat`). -The branch reorders xHCI ring writes (`dma_wmb`) but touches no quiesce code. -A same-session A/B then gave 5 of 5 green on each arm, main at `e48604c0` and -the branch: the red did not come back alone or beside the other arm, so it -is still not shown to be anything but load-bound. - -**Recurrence, PR #542's nightly at 059c5de7 (run 36328646395, `guest (7)`), -KVM, one guest on the runner.** The same `QEMU never reported stopping: the -guest asked for a reboot and stayed up`, the same `quiesce_writers: 4 of 6 -writers reached their loop in 5s` and `test_rs_quiesce_writers exit=1`, and no -transport break in the capture. A writer's first `quiesce-writer: 1` line -is its first pass done: writer 1 at 0.681 s, 0 at 1.061 s, 3 at 1.844 s, 5 at -2.014 s, and writers 2 and 4 only at 6.610 s and 6.778 s, 5.8 s and 5.9 s -after their ` 0` lines. Writer 1 printed pass 33 at 6.905 s, its passes 93 -to 306 ms apart. The branch changes no kernel or guest code. A red on a -one-guest lane is not the other suites' load. - -**Exit condition.** Re-enabled when a reproduction names what holds a writer's -first write-and-fsync pass for over 5 s while another writer passes in under a -third of a second, and the fix is shown against it. Owner: the `/log` write and -sync path `tests/toyos-rust-tests/src/bin/quiesce_writers.rs` drives; held by -the orchestrator. diff --git a/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md b/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md deleted file mode 100644 index f21bdbf3331..00000000000 --- a/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md +++ /dev/null @@ -1,19 +0,0 @@ ---- -status: expected-red -kind: defect -opened: 2026-09-03 ---- - -# `so_cache_refusals` saw the kernel refuse nothing, once, on CI - -`ci` run 33756442944, `guest (5)`, 2026-09-03, on `w5b15-ready`, whose diff -touches no loader or cache file; `ALONE so_cache_refusals: GREEN, and it was -alone both times` in the same job. - -The verdict: no "byte budget; refused" line — the kernel refused nothing: -twelve 2 MiB images entered a cache whose test budget refuses at the second. - -Owed: a mechanism. Nobody has one. - -**Exit condition.** The cause of the missing refusal is fixed, and -`so_cache_refusals` green on CI's KVM `guest` shards. Owner: orchestrator. diff --git a/src/ci.rs b/src/ci.rs index 3c56f355009..927020bd63e 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -30,7 +30,7 @@ use std::path::Path; use std::process::Command; use crate::arch::Arch; -use crate::{flags, pr, release, sdkversion}; +use crate::{flags, pr, release, sdkversion, testargs}; /// The checks `main`'s ruleset must require, as `gate-stage` reads them back: /// a minimum, never an equality, so a name GitHub requires and this does not @@ -40,13 +40,18 @@ pub(crate) const REQUIRED_CHECKS: &[&str] = &["host"]; /// The one issue a red nightly files or comments on, found by title. const NIGHTLY_RED: &str = "nightly is red"; +/// `nightly.yml`'s two schedules: the nightly reach six nights a week, the weekly +/// reach on the seventh. +const NIGHTLY_CRON: &str = "0 3 * * 1-6"; +const WEEKLY_CRON: &str = "0 3 * * 0"; + const USAGE: &str = "cargo run -- --ci , where is one of: host every host test: the build system, the host workspace, the licences of what ships, clippy, the model controls, userland and the SDK (ci.yml, nightly) gate-stage what protects main, read back from GitHub (ci.yml) toolchain publish this tree's toolchain if nobody has (nightly) - guest / one shard of the whole guest suite, nightly tier included (nightly) + guest / one shard of the guest suite at the reach its schedule names (nightly) tcg one test on an emulated CPU (nightly) audio / one shard of gate A (nightly) nightly-red file or update the nightly-red issue from $NEEDS (nightly) @@ -98,9 +103,10 @@ pub fn dispatch(root: &Path, args: &[String]) { Job::Host => host(root), Job::GateStage => vec![step("what protects main", || gate_stage(root))], Job::Toolchain => vec![step("the toolchain release", || release::ensure_published(root))], - Job::Guest(shard) => { - guest(root, &suite_args(&["--shard", shard, "--jobs", "1", "--nightly"])) - } + Job::Guest(shard) => match guest_reach() { + Ok(reach) => guest(root, &suite_args(&["--shard", shard, "--jobs", "1", reach])), + Err(refusal) => vec![step("the reach", || Err(refusal))], + }, Job::Tcg => guest(root, &suite_args(&["--jobs", "1", "process_stats"])), Job::Audio(shard) => guest(root, &suite_args(&["--audio-gate", "30", "--shard", shard])), Job::NightlyRed => vec![step("the nightly-red issue", nightly_red)], @@ -602,6 +608,31 @@ fn protection(rules: &serde_json::Value) -> (Vec, Vec) { // --- The guest jobs ------------------------------------------------------------ +/// The reach flag of the run that started this job: the weekly one on the weekly +/// schedule, and the nightly one on the other schedule, on a dispatch and off a +/// runner. +fn guest_reach() -> Result<&'static str, String> { + if std::env::var("GITHUB_EVENT_NAME").as_deref() != Ok("schedule") { + return Ok(testargs::NIGHTLY.name); + } + let path = std::env::var("GITHUB_EVENT_PATH") + .map_err(|_| "a scheduled run with no $GITHUB_EVENT_PATH names no schedule".to_string())?; + let payload = std::fs::read_to_string(&path).map_err(|e| format!("{path}: {e}"))?; + let event: serde_json::Value = + serde_json::from_str(&payload).map_err(|e| format!("{path}: {e}"))?; + reach_of_schedule(event["schedule"].as_str()) +} + +fn reach_of_schedule(cron: Option<&str>) -> Result<&'static str, String> { + match cron { + Some(NIGHTLY_CRON) => Ok(testargs::NIGHTLY.name), + Some(WEEKLY_CRON) => Ok(testargs::WEEKLY.name), + other => Err(format!( + "a scheduled run of {other:?}, which is neither schedule nightly.yml declares" + )), + } +} + /// The harness's arguments for a CI lane: a runner is a whole host with one /// suite on it, so the host's guest slots arbitrate nothing there. fn suite_args(args: &[&str]) -> Vec { @@ -912,6 +943,29 @@ mod tests { assert!(parse(&[]).is_err()); } + #[test] + fn each_schedule_names_its_reach_and_another_is_refused() { + assert_eq!(reach_of_schedule(Some(NIGHTLY_CRON)), Ok("--nightly")); + assert_eq!(reach_of_schedule(Some(WEEKLY_CRON)), Ok("--weekly")); + for stray in [Some("0 4 * * *"), None] { + let refusal = reach_of_schedule(stray).unwrap_err(); + assert!(refusal.contains(&format!("{stray:?}")), "{refusal}"); + } + } + + /// The schedules `nightly.yml` declares are exactly the two a reach is + /// named for, so no scheduled run reaches the refusal above. + #[test] + fn nightly_yml_declares_the_two_schedules() { + let text = std::fs::read_to_string(repo_root().join(".github/workflows/nightly.yml")) + .expect("nightly.yml is readable"); + let crons: Vec<&str> = text + .lines() + .filter_map(|l| l.trim().strip_prefix("- cron: '")?.strip_suffix('\'')) + .collect(); + assert_eq!(crons, [NIGHTLY_CRON, WEEKLY_CRON]); + } + /// Teeth for the controls' judge: a green negative control, a control that /// never reached its verdict, and a self-catching case that failed are all /// red. diff --git a/src/redlist.rs b/src/redlist.rs index f941c3eaccd..bf82d560d6c 100644 --- a/src/redlist.rs +++ b/src/redlist.rs @@ -25,10 +25,6 @@ pub struct Disabled { /// Every disabled test. pub const DISABLED: &[Disabled] = &[ - Disabled { - test: "console_line_atomicity", - issue: "issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md", - }, Disabled { test: "console_locale_detect", issue: "issues/build/the-console-input-path-can-stop-after-a-ps2-overflow.md", @@ -43,14 +39,6 @@ pub const DISABLED: &[Disabled] = &[ Disabled { test: "hda_tone", issue: "issues/audio/hda-tone-phase-check.md" }, Disabled { test: "kill_while_blocked", issue: "issues/kernel/deferred-release-outlives-its-syscall.md" }, Disabled { test: "latency_wake", issue: "issues/build/latency-wake-reds-on-the-dev-host-at-a-rate.md" }, - Disabled { - test: "quiesce_dump_holds_the_stopped", - issue: "issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md", - }, - Disabled { - test: "quiesce_wakes_on_the_last_exit", - issue: "issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md", - }, Disabled { test: "sched_check_build", issue: "issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md", @@ -63,14 +51,6 @@ pub const DISABLED: &[Disabled] = &[ test: "short_sleep_livelock", issue: "issues/kernel/short-sleep-livelock-stalls-on-ci-with-one-sleeper-never-returning.md", }, - Disabled { - test: "so_cache_refusals", - issue: "issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md", - }, - Disabled { - test: "usb_disk_index_stable", - issue: "issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md", - }, Disabled { test: "xhci_flap", issue: "issues/hardware/a-collapsed-replug-is-enumerated-only-when-another-port-event-arrives.md", diff --git a/src/testargs.rs b/src/testargs.rs index 4eaaf0e4abc..9646a991136 100644 --- a/src/testargs.rs +++ b/src/testargs.rs @@ -158,6 +158,7 @@ declare_flags!(pub SUITE = { pub SHARD = "--shard", Next; pub SLOW_USB = "--slow-usb", None; pub NIGHTLY = "--nightly", None; + pub WEEKLY = "--weekly", None; /// The metal profile: the registrations that run on the T14, batched into /// images and judged off the log the stick came back with. pub METAL = "--metal", None; @@ -221,13 +222,23 @@ pub fn parse(args: &[String]) -> Result, String> { .to_string(), ); } - if has(&NIGHTLY) && has(&AUDIO_GATE) { + if has(&NIGHTLY) && has(&WEEKLY) { return Err( - "--nightly and --audio-gate are separate tiers and cannot be combined; run one \ - tier at a time" + "--weekly runs the nightly tier too, so a --nightly beside it would be read by \ + nothing; write one" .to_string(), ); } + for reach in [&NIGHTLY, &WEEKLY] { + for other in [&AUDIO_GATE, &METAL] { + if has(reach) && has(other) { + return Err(format!( + "{} and {} are separate tiers and cannot be combined; run one tier at a time", + reach.name, other.name + )); + } + } + } Ok(filter) } @@ -261,18 +272,26 @@ mod tests { } #[test] - fn nightly_and_audio_gate_are_refused_by_the_argv_validator() { - for argv in [ - vec!["--nightly", "--audio-gate", "30"], - vec!["--audio-gate=30", "--nightly"], + fn a_reach_and_another_tier_are_refused_by_the_argv_validator() { + for (argv, reach, other) in [ + (vec!["--nightly", "--audio-gate", "30"], "--nightly", "--audio-gate"), + (vec!["--audio-gate=30", "--nightly"], "--nightly", "--audio-gate"), + (vec!["--weekly", "--audio-gate", "30"], "--weekly", "--audio-gate"), + (vec!["--metal", "--nightly"], "--nightly", "--metal"), + (vec!["--weekly", "--metal"], "--weekly", "--metal"), ] { let refusal = parse_owned(&argv).unwrap_err(); - assert!(refusal.contains("--nightly"), "{refusal}"); - assert!(refusal.contains("--audio-gate"), "{refusal}"); - assert!(refusal.contains("cannot be combined"), "{refusal}"); + assert!(refusal.contains(reach) && refusal.contains(other), "{argv:?}: {refusal}"); + assert!(refusal.contains("cannot be combined"), "{argv:?}: {refusal}"); } } + #[test] + fn the_weekly_reach_refuses_a_nightly_it_already_runs() { + let refusal = parse_owned(&["--nightly", "--weekly"]).unwrap_err(); + assert!(refusal.contains("read by nothing"), "{refusal}"); + } + #[test] fn the_filter_is_the_word_that_is_nobodys_value() { assert_eq!(parse_owned(&["process_stats"]).unwrap().as_deref(), Some("process_stats")); @@ -489,10 +508,11 @@ mod tests { vec!["--host-builds", "0"], vec!["--shard", "2/4"], vec!["--nightly"], + vec!["--weekly"], + vec!["--weekly", "--shard", "2/12", "--jobs", "1"], vec!["--debug"], vec!["--metal"], vec!["--metal", "--metal-readback", "target/metal"], - vec!["--metal", "--nightly"], ] { assert!(parse_owned(&argv).is_ok(), "{argv:?}"); } diff --git a/src/tiers.rs b/src/tiers.rs index f0c57216b9f..9bbe674d3b1 100644 --- a/src/tiers.rs +++ b/src/tiers.rs @@ -2,26 +2,36 @@ //! //! **The table is the registration.** Every row of `tests/toyos.rs`'s //! `MACHINE_TESTS`, `SCREEN_TESTS` and `AUDIO_TESTS` carries its [`Tier`] or -//! does not compile, and the shared boot's discovered tests share one. Moving a -//! test between tiers is editing that one word. +//! does not compile, and the shared boot's discovered tests share one. A +//! [`Schedule`] is every one of those names at its one tier. Moving a test +//! between tiers is editing that one word. //! -//! The fast tier is what every plain `cargo test` runs; the nightly tier is -//! `--nightly`, run by `.github/workflows/nightly.yml`. No pull request boots a -//! guest. The line between the two is 10 s on CI's hosted shards, and nothing -//! enforces it: a name that is Nightly because its verdict is anchored to real -//! time, or because it shares a boot with one that is slow, stays where it is -//! whatever it measures. +//! The tiers nest: a plain `cargo test` reaches `Fast`, `--nightly` adds +//! `Nightly`, and `--weekly` adds `Weekly` to that. `.github/workflows/nightly.yml` +//! runs the nightly reach six nights a week and the weekly reach on the seventh +//! (`src/ci.rs`). No pull request boots a guest. //! -//! The local tier is the third, and the only one CI never runs: its guests are +//! A test's tier is what it has caught: `Fast` is the shared boot and the tests +//! that caught a real defect for at most 5 s of guest time, `Nightly` the other +//! tests that caught one, and `Weekly` those that never did; a group sharing one +//! boot takes its most frequent member's tier. A new test enters `Nightly`. +//! +//! The local tier is the fourth, and the only one CI never runs: its guests are //! of an architecture no hosted runner has been measured to boot. +use std::collections::BTreeMap; + +use crate::testargs::{NIGHTLY, SUITE, WEEKLY}; + /// Which run a registered test belongs to. #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub enum Tier { /// Every `cargo test`. Fast, - /// `cargo test --test toyos-build -- --nightly`. + /// `cargo test --test toyos-build -- --nightly`, and every weekly run. Nightly, + /// `cargo test --test toyos-build -- --weekly`, and before a release. + Weekly, /// Every `cargo test` on a developer's machine and no sharded run: a guest /// of an architecture no CI runner boots yet. The AArch64 port's stage 8 /// (`issues/kernel/toyos-runs-on-arm64.md`) measures the runners and @@ -29,14 +39,141 @@ pub enum Tier { Local, } +/// How far down the nested tiers a run reaches. +#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Debug)] +pub enum Reach { + Fast, + Nightly, + Weekly, +} + +impl Reach { + /// The reach `args` ask for; `testargs::parse` refuses both flags at once. + pub fn of(args: &[String]) -> Self { + if SUITE.present(args, &WEEKLY) { + Self::Weekly + } else if SUITE.present(args, &NIGHTLY) { + Self::Nightly + } else { + Self::Fast + } + } +} + impl Tier { - /// Whether a run selects this tier: `nightly` is `--nightly`, `sharded` - /// a `--shard`, which only CI's jobs pass. - pub fn selected(self, nightly: bool, sharded: bool) -> bool { + /// Whether a run that reaches `reach` selects this tier; `sharded` is a + /// `--shard`, which only CI's jobs pass. + pub fn selected(self, reach: Reach, sharded: bool) -> bool { match self { Self::Fast => true, - Self::Nightly => nightly, + Self::Nightly => reach >= Reach::Nightly, + Self::Weekly => reach >= Reach::Weekly, Self::Local => !sharded, } } + + /// The flag of the narrowest run that selects this tier, for a run that + /// held it back to name; `None` for the tiers every unsharded run takes. + pub fn flag(self) -> Option<&'static str> { + match self { + Self::Nightly => Some(NIGHTLY.name), + Self::Weekly => Some(WEEKLY.name), + Self::Fast | Self::Local => None, + } + } +} + +/// Every registered test at its one tier. +pub struct Schedule<'a>(BTreeMap<&'a str, Tier>); + +impl<'a> Schedule<'a> { + /// Refuses a name registered twice, by name, at one tier or two: two rows + /// are two verdicts under one name. + pub fn new(rows: impl IntoIterator) -> Result { + let mut tiers = BTreeMap::new(); + for (name, tier) in rows { + if let Some(first) = tiers.insert(name, tier) { + return Err(format!( + "{name} is registered twice, at {first:?} and at {tier:?}: one name is one \ + verdict at one tier" + )); + } + } + Ok(Self(tiers)) + } + + /// `name`'s tier, refused by name when nothing registers it. + pub fn tier(&self, name: &str) -> Result { + self.0 + .get(name) + .copied() + .ok_or_else(|| format!("{name} is not a registered test, so it has no tier")) + } + + /// Every registered name and its tier, by name. + pub fn iter(&self) -> impl Iterator + '_ { + self.0.iter().map(|(name, tier)| (*name, *tier)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const EVERY: [Tier; 4] = [Tier::Fast, Tier::Nightly, Tier::Weekly, Tier::Local]; + const NAMES: [&str; 4] = ["fast_one", "nightly_one", "weekly_one", "local_one"]; + + #[test] + fn every_registered_test_has_exactly_one_tier() { + let schedule = Schedule::new(NAMES.into_iter().zip(EVERY)).unwrap(); + for (name, tier) in NAMES.into_iter().zip(EVERY) { + assert_eq!(schedule.tier(name), Ok(tier), "{name}"); + } + assert_eq!(schedule.iter().count(), EVERY.len()); + for second in EVERY { + let refusal = + Schedule::new([("twice", Tier::Fast), ("once", Tier::Weekly), ("twice", second)]) + .err() + .expect("a second row for one name is refused"); + assert!(refusal.starts_with("twice is registered twice"), "{refusal}"); + } + } + + #[test] + fn an_unregistered_test_is_refused_by_name() { + let schedule = Schedule::new([("registered", Tier::Weekly)]).unwrap(); + let refusal = schedule.tier("never_registered").unwrap_err(); + assert!(refusal.starts_with("never_registered is not a registered test"), "{refusal}"); + assert!(schedule.tier("registere").is_err(), "a prefix is not the name"); + } + + /// Each reach selects its own tier and every narrower one, and nothing + /// wider; a shard alone decides `Local`. + #[test] + fn the_tiers_nest() { + let selected = |reach| -> Vec { + EVERY.into_iter().filter(|tier| tier.selected(reach, true)).collect() + }; + assert_eq!(selected(Reach::Fast), [Tier::Fast]); + assert_eq!(selected(Reach::Nightly), [Tier::Fast, Tier::Nightly]); + assert_eq!(selected(Reach::Weekly), [Tier::Fast, Tier::Nightly, Tier::Weekly]); + for reach in [Reach::Fast, Reach::Nightly, Reach::Weekly] { + assert!(Tier::Local.selected(reach, false) && !Tier::Local.selected(reach, true)); + } + } + + /// The flag a held-back tier names is the one whose reach selects it. + #[test] + fn a_held_tiers_flag_selects_it() { + for tier in EVERY { + let Some(flag) = tier.flag() else { + assert!(tier.selected(Reach::Fast, false), "{tier:?} names no flag, so a plain run takes it"); + continue; + }; + let reach = Reach::of(&[flag.to_string()]); + assert!(tier.selected(reach, true), "{flag} does not select {tier:?}"); + assert!(!tier.selected(Reach::Fast, true), "{tier:?} names {flag} and needs none"); + } + assert_eq!(Reach::of(&[]), Reach::Fast); + } } diff --git a/tests/common/console.rs b/tests/common/console.rs index 31fd198640d..159bea96488 100644 --- a/tests/common/console.rs +++ b/tests/common/console.rs @@ -1,31 +1,13 @@ -//! The console's line atomicity, and the one thing the harness may conclude -//! from it. +//! The one thing the harness may conclude from the console's line atomicity: +//! [`verdict`] — a line's *first bytes are its writer's own*, so the C family +//! can tell a daemon's line from the program under test's by reading it, and +//! stop failing on output that is not its own. //! -//! Two subjects, in this order because the second rests on the first: -//! -//! 1. [`console_line_atomicity`] — every line on the console is one writer's, -//! whole. -//! 2. [`verdict`] — therefore a line's *first bytes are its writer's own*, so -//! the C family can tell a daemon's line from the program under test's by -//! reading it, and stop failing on output that is not its own. -//! -//! The order is the argument, and it is the whole of what closed task #84. -//! Before L5 a daemon's `println!` and a test's could reach the backend in -//! pieces and arrive spliced into one line, and no rule over lines could -//! separate them — so the standing write-up said there was no cheap honest fix -//! and left a choice between giving each child a capture channel and tagging -//! every console write with its writer. Both are built now: a program's line -//! is a record in its own log ring, assembled by the process that wrote it, -//! and `logd` puts it on the console whole under the program's name — so a -//! line begins with its writer's first bytes or `console_line_atomicity` is -//! red. -//! -//! **L5's guarantee is about flushes, not about newlines, and that is the -//! second half.** A program that writes without a trailing newline has its -//! bytes joined to the next writer's line by the *host's* splitter — see -//! [`speaker_at`], which is where that is written down. Both halves are -//! [`c_capture_ignores_daemon_lines`]'s, and every one of its verdicts carries -//! the control that says it has teeth. +//! **L5's guarantee is about flushes, not about newlines.** A program that +//! writes without a trailing newline has its bytes joined to the next writer's +//! line by the *host's* splitter — see [`speaker_at`], which is where that is +//! written down. Both are [`c_capture_ignores_daemon_lines`]'s, and every one +//! of its verdicts carries the control that says it has teeth. use std::collections::BTreeSet; use std::path::{Path, PathBuf}; @@ -33,225 +15,12 @@ use std::sync::OnceLock; use std::time::Duration; use super::qemu::{BootOptions, QemuInstance}; -use super::serial::Serial; - -/// The guest binary's name in the `run ` protocol. -const WRITER: &str = "test_rs_console_line_atomicity"; -/// A liveness guard and never the verdict: two thousand 200-byte lines is a -/// fraction of a second of virtio-console, and this only catches a guest that +/// A liveness guard and never the verdict: it only catches a guest that /// stopped answering. const CEILING: Duration = Duration::from_secs(60); -/// What the guest's binary declares, so the host is not carrying a second copy -/// of the numbers. -struct Declared { - writers: usize, - lines: usize, - width: usize, - /// Bytes the third writer said in two `write`s and never ended with a - /// newline, which only its own exit can put on the wire. - midline: usize, - /// Digits of the sequence number after each line's leading tag byte — - /// what tells a gap in a writer's own run from a capture that ends early. - seq: usize, -} - -pub fn console_line_atomicity( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - // **Two CPUs, because one writer preempting another is the stimulus.** At - // `--smp 1` the two processes still interleave — they are preempted, not - // parallel — but at two the gap between one writer's two `write`s can be - // filled by a genuinely concurrent one, which is the harder case and the - // one a laptop has. - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { smp: 2, ..Default::default() }, - ); - let result = qemu.run_test(WRITER, CEILING); - if let Some(err) = &result.error { - return Err(format!("{err}\nstdout:\n{}", tail(&result.stdout))); - } - if result.exit_code != Some(0) { - return Err(format!( - "the writers exited {:?}\n{}", - result.exit_code, - tail(&result.stdout) - )); - } - let declared = declared(&result.stdout)?; - // **The writers' lines and the runner's `===TEST_START` reach the console - // through `logd` from two rings' reads**, so a writer's first lines may - // arrive before the marker opens the window: every line since the command - // was typed is read. - let capture = format!("{}{}", result.before, result.stdout); - - let mut pure: [BTreeSet; 2] = [BTreeSet::new(), BTreeSet::new()]; - let mut duplicated: usize = 0; - let mut mixed: Vec<&str> = Vec::new(); - let mut short: usize = 0; - for line in capture.lines() { - let a = line.bytes().filter(|b| *b == b'A').count(); - let b = line.bytes().filter(|b| *b == b'B').count(); - // A writer's line is one tag byte, its sequence digits, and the tag - // repeated to the width — so a line that is mostly one tag is one of - // its lines however it ended up. - let (tag, other) = if a >= b { (a, b) } else { (b, a) }; - if tag * 2 < declared.width - 1 { - continue; // not a writer's line at all - } - if other > 0 { - mixed.push(line); - continue; - } - let writer = usize::from(b > a); - let bytes = line.as_bytes(); - let tag_byte = b"AB"[writer]; - let whole = bytes.len() == declared.width - 1 - && bytes[0] == tag_byte - && bytes[1..1 + declared.seq].iter().all(u8::is_ascii_digit) - && bytes[1 + declared.seq..].iter().all(|c| *c == tag_byte); - let seq = whole - .then(|| std::str::from_utf8(&bytes[1..1 + declared.seq]).expect("ascii digits")) - .and_then(|digits| digits.parse::().ok()) - .filter(|seq| *seq < declared.lines); - let Some(seq) = seq else { - short += 1; - continue; - }; - if !pure[writer].insert(seq) { - duplicated += 1; - } - } - - if !mixed.is_empty() { - let sample: Vec = mixed - .iter() - .take(3) - .map(|l| l.chars().take(80).collect::()) - .collect(); - return Err(format!( - "{} of {} console lines carry both writers' bytes — a `write` syscall is still the \ - unit of interleaving, so half a line reaches the backend and another process's \ - half follows it. First three, truncated to 80 columns:\n{}", - mixed.len(), - declared.writers * declared.lines, - sample.join("\n") - )); - } - if short != 0 { - return Err(format!( - "{short} console lines are a writer's bytes at the wrong width; a whole line is one \ - unit and these were cut" - )); - } - if duplicated != 0 { - return Err(format!( - "{duplicated} console lines repeat a sequence number a writer used once — the \ - capture duplicated lines, which no writer and no buffer can do" - )); - } - // Non-vacuity: a capture that lost the writers' output entirely would count - // zero mixed lines and prove nothing. The sequence numbers say what *kind* - // of loss it was, so a red here is not misread as the line buffer breaking: - // a gap inside a writer's own run is lines lost mid-stream, a contiguous - // run that stops early is a capture missing its tail, and neither is a - // mixed or short line — the mechanism's own verdicts are above. - for (i, tag) in ["A", "B"].iter().enumerate() { - let seen = &pure[i]; - if seen.len() == declared.lines { - continue; - } - let top = seen.iter().next_back().map_or(0, |s| s + 1); - let gaps = top - seen.len(); - if gaps > 0 { - let first_gap = (0..top).find(|s| !seen.contains(s)).unwrap_or(0); - return Err(format!( - "writer {tag} declared {} whole lines and the capture carries {}: {gaps} gap(s) \ - inside the writer's own numbered run (first at #{first_gap}, run ends at \ - #{}) — lines were lost mid-stream, between the guest's console and this \ - capture, not by the line buffer", - declared.lines, - seen.len(), - top - 1, - )); - } - return Err(format!( - "writer {tag} declared {} whole lines and the capture carries {}: the numbered run \ - is contiguous and simply stops at #{} — a short capture missing its tail, not a \ - line lost by the buffer", - declared.lines, - seen.len(), - top.saturating_sub(1), - )); - } - // **The line's other half: a process that exits mid-line.** The third - // writer says `midline` bytes in two `write`s, ends them with nothing and - // exits; the only thing that can make them a line is its exit ending what - // its stream held. A tree without that loses them silently, which drops a - // dying process's last words — so the assertion is the run's *length*, - // and it is exact on both sides: shorter means bytes were lost, longer - // means something else was acquired inside them. - let longest = capture - .split(|c| c != 'C') - .map(str::len) - .max() - .unwrap_or(0); - if longest != declared.midline { - return Err(format!( - "a process exited having written {} unterminated bytes and the longest run of them on \ - the console is {longest} — the process's exit is what turns a partial line into all \ - there will ever be, and this capture says it went nowhere", - declared.midline - )); - } - - // The kernel-into-userland half, on the same capture. A kernel record can - // only land inside a userland line if the line reached the backend in - // pieces, so this reds on exactly the coupling the count above reds on and - // observes it from the other side. - let console = Serial::named("console", &capture); - if let Some(spliced) = console.interleaved() { - return Err(format!( - "a kernel record landed inside a userland line: {:?}", - spliced.chars().take(160).collect::() - )); - } - eprintln!( - " [console] {} writers x {} lines of {} bytes, 0 mixed; {} unterminated bytes flushed by \ - an exit", - declared.writers, declared.lines, declared.width, declared.midline - ); - Ok(()) -} - -fn declared(stdout: &str) -> Result { - let line = stdout - .lines() - .find(|l| l.contains("console-atomicity: writers=")) - .ok_or_else(|| format!("the guest never declared its run\n{}", tail(stdout)))?; - let field = |key: &str| -> Result { - line.split_whitespace() - .find_map(|w| w.strip_prefix(key)) - .and_then(|v| v.parse::().ok()) - .ok_or_else(|| format!("the guest's declaration has no `{key}`: {line:?}")) - }; - Ok(Declared { - writers: field("writers=")?, - lines: field("lines=")?, - width: field("width=")?, - midline: field("midline=")?, - seq: field("seq=")?, - }) -} - -/// The last of a capture, for a failure message. Two thousand 200-byte lines is -/// not something to put in an assertion message whole. +/// The last of a capture, for a failure message. fn tail(text: &str) -> String { let lines: Vec<&str> = text.lines().collect(); lines[lines.len().saturating_sub(20)..] @@ -704,8 +473,8 @@ pub fn c_capture_ignores_daemon_lines( /// /// **`Profile::Metal` because the keystroke has to arrive.** Its i8042 is the /// only keyboard on the machine — no USB HID, no virtio — which is the shape -/// `i8042_keyboard` and `swiss_german_layout` already inject through, and the -/// mouse the middle arm claims is the PS/2 one beside it. +/// `swiss_german_layout` already injects through, and the mouse the middle arm +/// claims is the PS/2 one beside it. /// /// **One CPU, because the keystroke outlives the probe.** Nothing holds the /// keyboard once the claim is released, so the key stays queued while the diff --git a/tests/common/faults.rs b/tests/common/faults.rs index a00c2e86b56..5b331d9c82e 100644 --- a/tests/common/faults.rs +++ b/tests/common/faults.rs @@ -579,12 +579,7 @@ fn hex(line: &str, field: &str) -> Option { /// that carries on. /// /// **One boot, and the two negative controls are `syscall_window_nmi_controls`'s -/// two.** All three used to be one name and it priced at 19,740 ms on the hosted -/// lane against a 10,000 ms ceiling — three Metal boots of 3,000 NMIs each. What -/// belongs per pull request is the property: the window is reachable, arrivals -/// land in it, and the machine survives them. What the controls establish is that -/// the property is not vacuous, which is a claim about the *instrument* and moves -/// to the nightly tier. +/// two.** /// /// **The window.** `SYSCALL` switches no stack, so `arch::syscall`'s entry runs /// three instructions at CPL 0 with the user's `rsp` and its exit one more diff --git a/tests/common/iommu.rs b/tests/common/iommu.rs index cb3b49349e9..57d48b12be3 100644 --- a/tests/common/iommu.rs +++ b/tests/common/iommu.rs @@ -259,11 +259,6 @@ fn table_word(socket: &Path, base: u64, index: u16) -> Result<(u64, u64), String /// /// [`Profile::Headless`] carries the most sources of both kinds — the i8042's /// two pins, and xHCI, virtio-net and virtio-sound over MSI-X. -/// -/// **Per pull request this is the one machine that runs.** The machines that -/// make the verdict move — `intremap=off`, and `eim=on` for the other entry -/// format — are [`iommu_discovery`]'s, and that is nightly, so a change that -/// broke only the not-remapping arm would land and be caught the next night. pub fn iommu_interrupt_remapping( test_config: &Path, c_bins: &[(String, Vec)], diff --git a/tests/common/lan.rs b/tests/common/lan.rs index 8ce0a3be67b..6e9968a0bbb 100644 --- a/tests/common/lan.rs +++ b/tests/common/lan.rs @@ -616,8 +616,7 @@ pub fn lan_talk( let (start, len) = super::volumes::log_extent(&bytes, &image)?; // **Its own disk.** sshd mints its identity under `/home`, and the lane's - // shared image would hand that identity to the next boot of the lane, - // which `sshd_fail_closed` asserts it mints itself. + // shared image would hand that identity to the next boot of the lane. let data = super::lane::dir().join("lan-talk-data.img"); toyos_build::build::create_sparse(&data, qemu::NVME_SMALL); let (ssh_port, log_port) = (qemu::free_host_port(), qemu::free_host_port()); diff --git a/tests/common/logread.rs b/tests/common/logread.rs index 731f349b003..e438887b250 100644 --- a/tests/common/logread.rs +++ b/tests/common/logread.rs @@ -58,15 +58,6 @@ struct Contaminated { } /// The conservation law, at one width. -/// -/// **Three registered names and not one, and the reason is the fast tier's -/// line.** What the law is about is concurrent producers, so a machine with one -/// CPU and a machine with eight are different subjects rather than one subject -/// measured three times: `--smp 1` is where the reader and the one producer -/// share a CPU, `--smp 4` and `--smp 8` are where they do not. One name over -/// all three boots measured 17,112 ms in CI — over the fast tier's 10 s line, and the -/// gate the whole design turns on may not sit in the nightly tier — while each -/// boot on its own is comfortably under it. fn conservation( test_config: &Path, c_bins: &[(String, Vec)], @@ -112,22 +103,6 @@ pub fn log_conservation_smp1( conservation(test_config, c_bins, rust_bins, 1) } -pub fn log_conservation_smp4( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - conservation(test_config, c_bins, rust_bins, 4) -} - -pub fn log_conservation_smp8( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - conservation(test_config, c_bins, rust_bins, 8) -} - /// The nested-`emit` gate: an interrupt that logs, inside another `emit`, on one CPU. /// /// **The one case loom cannot express and the host cannot stage.** The diff --git a/tests/common/power.rs b/tests/common/power.rs index 198952e0448..e06617e6c70 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -337,17 +337,6 @@ pub fn quiesce_wakes_on_the_last_park( woken_by_the_held_thread(&["quiesce-last-park", LATE_WORD], rust_bins) } -/// **An exit that is the stop's last transition wakes it.** The same, with -/// `quiesce-last-exit` holding the thread inside `SYS_THREAD_EXIT`. -pub fn quiesce_wakes_on_the_last_exit( - _test_config: &Path, - _c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - woken_by_the_held_thread(&["quiesce-last-exit", LATE_WORD], rust_bins) -} - -/// One of the two `quiesce-last-*` boots: its actuator first, the late word beside it. fn woken_by_the_held_thread( armed: &'static [&'static str; 2], rust_bins: &[(String, Vec)], @@ -386,82 +375,6 @@ fn woken_by_the_held_thread( Ok(()) } -/// **A dump served during the stop says every thread it stopped is held.** -/// `quiesce-dump` serves Ctrl+Alt+D's report from the shutdown once its first -/// stage has stopped the writers, the moment the owner presses it on a -/// shutdown stuck there. A stopped thread's state word reads `Ready`, and a -/// report that counted it nowhere else would call it claimed and not held. -pub fn quiesce_dump_holds_the_stopped( - _test_config: &Path, - _c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let (whole, record) = stopped_boot( - "tests/quiescecase/system.toml", - "quiesce_writers", - &["quiesce-dump", LATE_WORD], - rust_bins, - )?; - let lines: Vec<&str> = whole.lines().collect(); - let at = |needle: &str| lines.iter().position(|line| line.contains(needle)); - let (Some(began), Some(ended), Some(synced)) = ( - at("=== blocked-task dump:"), - at("=== end of dump ==="), - at("Syncing filesystems..."), - ) else { - return Err(format!("no whole dump and sync in this boot\n{whole}")); - }; - if !(began < ended && ended < synced) { - return Err(format!( - "the dump (lines {began} to {ended}) did not finish inside the stop, which ends at \ - line {synced}\n{whole}" - )); - } - let report = &lines[began..=ended]; - let number_before = |marker: &str, word: &str| -> Result { - let line = report - .iter() - .find(|line| line.contains(marker)) - .ok_or_else(|| format!("no {marker:?} line in the report:\n{}", report.join("\n")))?; - let (head, _) = - line.split_once(word).ok_or_else(|| format!("no {word:?} on {line:?}"))?; - head.split_whitespace() - .next_back() - .and_then(|n| n.parse().ok()) - .ok_or_else(|| format!("no number before {word:?} on {line:?}")) - }; - // The harm first: a report that has no word for a stopped thread still - // writes this verdict, and calls each one it cannot place unheld. - let unheld = number_before("== VERDICT:", " unheld,")?; - if unheld != 0 { - return Err(format!( - "a dump served during the stop called {unheld} thread(s) claimed and not held:\n{}", - report.join("\n"), - )); - } - // Then the premise: the report was served over a machine with banded - // threads, or the zero above is about nothing. - let stopped = number_before("== sched:", " stopped,")?; - if stopped == 0 { - return Err(format!( - "the report found no stopped thread on any cpu, so it says nothing about how it \ - counts one:\n{}", - report.join("\n"), - )); - } - // And the count is the threads it names: this stop bands fewer than one - // cpu's line cap, so each one it counts is a line of its own. - let named = report.iter().filter(|line| line.contains("stopped (the machine is stopping)")).count(); - if named != stopped as usize { - return Err(format!( - "the report counted {stopped} stopped thread(s) and named {named}:\n{}", - report.join("\n"), - )); - } - eprintln!(" [power] the dump inside the stop held all {stopped} stopped thread(s): {record}"); - Ok(()) -} - /// **The machine has one shutdown, and the second caller is refused while the /// first holds it.** init makes the first call; `quiesce-last-park` holds it /// after it has claimed the stop and before it stops anything, while every diff --git a/tests/common/storage.rs b/tests/common/storage.rs index fe93ca226d9..0ac5fa85ea0 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -446,209 +446,6 @@ fn front(path: &Path, at: u64, n: usize) -> Vec { head } -/// The shared-object cache's two refusals, judged in -/// `tests/toyos-rust-tests/src/bin/so_cache_policy.rs`. The independent oracle is -/// the NVMe image: the replaced library's bytes are read off the device after -/// the shutdown, so the claim rests on nothing the guest says. -pub fn so_cache_refusals( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - /// Without it the budget arm would have to load 256 MiB of libraries. - const PARAMS: &[&str] = &["so-cache-tiny"]; - /// Mirrored in the guest binary; `/home` is a directory of DATA, so that is - /// the name the host reader sees on the volume. - const STALE: &str = "home/so-cache-stale.so"; - const SECOND: &str = "libtls_dlopen_lib.so"; - - let want = rust_bins - .iter() - .find(|(name, _)| name == SECOND) - .map(|(_, data)| data.clone()) - .ok_or_else(|| format!("{SECOND} was not built, so there is nothing to compare against"))?; - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::MetalDisk, - kernel_params: PARAMS, - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - if boot.contains("are a tmpfs") { - return Err(format!( - "/apps and /home fell back to tmpfs, so the readback below would judge no device:\n{boot}" - )); - } - - let result = qemu.run_test("test_rs_so_cache_policy", Duration::from_secs(60)); - let log = format!("{boot}\n{}{}{}", result.before, result.stdout, result.serial); - if result.exit_code != Some(0) { - return Err(format!( - "so_cache_policy guest failed:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - // Stated by the kernel too: an arm reporting a refusal nobody made would pass. - for said in ["the cached image is stale", "byte budget; refused"] { - if !log.contains(said) { - return Err(format!("no {said:?} line — the kernel refused nothing:\n{log}")); - } - } - - let image = qemu.nvme_image().to_path_buf(); - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - - let io = FileBlocks::open(&image)?; - let fs = bcachefs::Mounted::<_, bcachefs::ReadOnly>::open(io) - .map_err(|e| format!("the NVMe image does not mount on the host: {e:?}"))?; - let got = fs - .read_file(STALE) - .map_err(|e| format!("reading {STALE} off the image: {e:?}"))?; - if got != want { - let at = got.iter().zip(&want).position(|(a, b)| a != b); - return Err(format!( - "{STALE} on the device is {} bytes against {SECOND}'s {}, first differing at {at:?} \ - — the guest's second write did not reach the device, so the refusal above was \ - about a file that had not changed", - got.len(), - want.len() - )); - } - - eprintln!( - " [so-cache] {} bytes of {SECOND} byte-identical at {STALE} off the NVMe image via the \ - host's own bcachefs reader", - want.len() - ); - Ok(()) -} - -/// F9's negative control: an fsync on `/home` whose first attempt is -/// budget-refused (`fsync-budget-spent`) must be retried to durable — on the -/// erasing adapter the `BudgetExpired` came back as `Io` and the guest's -/// `sync_all` failed on attempt 1. The independent oracle is the NVMe image -/// itself: after the shutdown the file's bytes are read off it on the host, -/// through this crate's own build of the `bcachefs` reader over a plain -/// seek-and-read device — nothing the guest kernel executed. -pub fn home_budget_refusal_retried( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - const PARAMS: &[&str] = &["fsync-budget-spent"]; - /// Mirrored in `tests/toyos-rust-tests/src/bin/home_fsync_budget.rs`, under - /// the `/home` directory of DATA the host reader sees. - const PATH: &str = "home/f9-budget.bin"; - const LEN: usize = 3 * 4096 + 41; - fn pattern() -> Vec { - (0..LEN).map(|i| (i.wrapping_mul(151) ^ 0x3C) as u8).collect() - } - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::MetalDisk, - kernel_params: PARAMS, - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - if boot.contains("are a tmpfs") { - return Err(format!( - "/apps and /home fell back to tmpfs, so nothing below touches the NVMe path:\n{boot}" - )); - } - - let result = qemu.run_test("test_rs_home_fsync_budget", Duration::from_secs(30)); - let log = format!("{boot}\n{}{}{}", result.before, result.stdout, result.serial); - if result.exit_code != Some(0) { - return Err(format!( - "home_fsync_budget guest failed — a budget-refused /home fsync was not retried \ - to durable:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - // Both halves of the staging, or the arm proved nothing: the refusal at the - // shipped NVMe site, and the fsync loop's own retry verdict. - if !log.contains("not issued") { - return Err(format!( - "no `not issued` line, so `fsync-budget-spent` staged no NVMe refusal:\n{log}" - )); - } - let retried = log - .lines() - .find(|l| l.contains("fsync: /home/") && l.contains("durable on attempt")) - .ok_or_else(|| { - format!("no `fsync: /home/... durable on attempt` line — the retry never ran:\n{log}") - })? - .trim() - .to_string(); - // And the refusal was one per file, not one per flush: `logd` flushes every - // round it wrote a line in, and a refused flush is lines of its own, so a - // refusal on every flush keeps `/log` retrying for as long as the machine - // runs and rotates the boot's own log away. An absence has no event to wait - // on, so it is judged over a fixed window: the storm retries there without - // pause, the once-per-file refusal never. - let after = qemu.drain_serial(Duration::from_secs(2)); - let again: Vec<&str> = - after.lines().filter(|l| l.contains("fsync: ") && l.contains("durable on attempt")).collect(); - if !again.is_empty() { - return Err(format!( - "{} flush(es) retried in the 2 s after the guest's, first {:?}: every flush is \ - being refused, and each refusal's records are the next flush\n{after}", - again.len(), - again[0] - )); - } - - let image = qemu.nvme_image().to_path_buf(); - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - - let io = FileBlocks::open(&image)?; - let fs = bcachefs::Mounted::<_, bcachefs::ReadOnly>::open(io) - .map_err(|e| format!("the NVMe image does not mount on the host: {e:?}"))?; - let got = fs - .read_file(PATH) - .map_err(|e| format!("reading {PATH} off the image: {e:?}"))?; - if got != pattern() { - let at = got.iter().zip(pattern()).position(|(a, b)| *a != b); - return Err(format!( - "{PATH} on the device is {} bytes, first differing at {at:?} — the retried fsync \ - reported durable over bytes the device does not hold", - got.len() - )); - } - - eprintln!(" [f9] {retried}"); - eprintln!( - " [f9] {LEN} bytes byte-identical off the NVMe image via the host's own bcachefs reader" - ); - Ok(()) -} - /// A same-length overwrite on `/home` read back through the name it rebound. /// /// The oracle is outside the guest and outside the kernel: with the machine diff --git a/tests/common/usb.rs b/tests/common/usb.rs index 333a93fd168..370b6243f70 100644 --- a/tests/common/usb.rs +++ b/tests/common/usb.rs @@ -506,141 +506,6 @@ pub fn usb_short_read( Ok(()) } -/// A disk plugged into a **different controller** must not renumber the one a -/// mount is holding. -/// -/// The machine-wide disk index was `storage.len()` summed across controllers, -/// and that vector grows on every bind — hot-plug included. The T14 has two -/// xHCIs, the Thunderbolt block's at 00:0d.0 ahead of the PCH's at 00:14.0, and -/// it boots off a stick in a PCH port: with nothing on the first controller the -/// boot stick is disk 0, and plugging any USB storage into the USB-C side made -/// the *new* drive disk 0 and the boot stick disk 1. `FatDevice` holds its -/// `UsbBlockDevice` for the life of the mount and that handle is an index, so -/// every later `/log` append went into the middle of the new drive and every -/// `/boot` read served its bytes as the ESP's. -/// -/// **The actuator is QEMU's own `device_add` and nothing about the driver is -/// modified.** Both verdicts are host-side and neither is a log line: the disk -/// that arrives late is a file the harness staged as zeros and must find as -/// zeros, and `/log/kernel.log` is read out of the boot image's own partition -/// and must carry a line the guest printed *after* the plug. -pub fn usb_disk_index_stable( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - /// The disk that arrives late: 48 GiB, sparse, and no size any other - /// device in this suite reports. - const LATE_BYTES: u64 = 48 * 1024 * 1024 * 1024; - /// The line the guest prints once the late disk is up, which is therefore a - /// line `/log` can only carry if it was still reaching the boot stick after - /// the plug. - const LATE_READY: &str = "usb-storage: disk 1 ready on slot"; - - // The boot stick is on the second controller and the first carries - // nothing, which is the laptop exactly — and the arrangement in which the - // boot stick is disk 0 with a free controller ahead of it. - let options = BootOptions { - profile: Profile::MetalXhciSecond, - qmp: true, - ..Default::default() - }; - let argv = qemu::profile_argv(&options); - if !argv.iter().any(|a| a.contains("usb-storage,bus=xhci1.0")) { - return Err(format!("the boot stick is not on the second controller: {argv:?}")); - } - if argv.iter().any(|a| a.starts_with("usb-storage,bus=xhci.0")) { - return Err(format!("the first controller already carries storage: {argv:?}")); - } - - // Built here rather than by the boot, because `/log` has to be read off the - // partition afterwards and the image gets a fresh GUID every time it is - // built. - let image_path = test_dir().join("usb-index-stable.img"); - let image = qemu::build_boot_image(test_config, c_bins, rust_bins, &[]); - std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - let (start, len) = super::volumes::log_extent(&image, &image_path)?; - - let late = test_dir().join("usb-index-late.img"); - drop(sparse(&late, LATE_BYTES)); - let before = fingerprint(&late, LATE_BYTES); - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - boot_image: Some(qemu::Staged::Written(image_path.clone())), - ..options - }, - ); - let boot = qemu.boot_log().to_string(); - if !boot.contains("usb-storage: disk 0 ready on slot") { - return Err(format!("the boot stick did not come up as disk 0\n{boot}")); - } - - let mut devices = qemu::QmpDevices::open(qemu.qmp_socket()); - devices.blockdev_add("latedisk", &late); - devices.add("usb-storage", "xhci.0", "latedisk0", &[("drive", "latedisk")]); - drop(devices); - // The driver's debounce is 100 ms and the enumeration behind it is - // microseconds under TCG; this is that with room. - thread::sleep(Duration::from_millis(1200)); - - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let log = format!("{boot}{}", qemu.drain_serial(Duration::from_secs(20))); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if log.contains(bad) { - return Err(format!("{bad:?} after a disk arrived on the other controller\n{log}")); - } - } - - // The plug happened at all. Without this both host-side claims below hold - // trivially on a boot where nothing was added. - if !log.contains(LATE_READY) { - return Err(format!( - "nothing enumerated on the first controller; there is no renumbering to survive\n{log}" - )); - } - - // **The disk that arrived is not the disk anything was mounted on.** The - // harness made this file and the guest was never told it was writable, so - // a single changed byte is `/boot` or `/log` writing through a handle that - // now names the wrong device. - if fingerprint(&late, LATE_BYTES) != before { - return Err( - "the guest wrote to the disk plugged into the other controller — the index a mount \ - was holding moved onto it" - .to_string(), - ); - } - - // **And the log kept reaching the stick.** Read off the boot image's own - // `/log` partition, so this is the device's view and not the guest's. The - // sink names one file per boot, so the newest on the volume is this one's. - let (name, on_device) = super::volumes::newest_log(&image_path, start, len)?; - let on_device = String::from_utf8_lossy(&on_device).into_owned(); - if !on_device.contains(LATE_READY) { - return Err(format!( - "/log/{name} stops at {} bytes and never carries {LATE_READY:?} — the appends after \ - the plug went somewhere else\n{log}", - on_device.len() - )); - } - let _ = std::fs::remove_file(&late); - let _ = std::fs::remove_file(&image_path); - - eprintln!( - " [usb] a {LATE_BYTES} B disk plugged into the empty first controller: it comes back \ - byte-identical, and {} bytes of /log/{name} on the boot stick carry the lines printed \ - after it", - on_device.len() - ); - Ok(()) -} - /// More disks on one controller than its DMA pool has blocks for. /// /// `MSC_BLOCKS` is 2 and the boot stick takes one, so the second data disk on @@ -3347,531 +3212,6 @@ pub fn xhci_full_speed_device( Ok(()) } -/// A HID interrupt endpoint whose transfer completes with a code the driver did -/// not expect. -/// -/// `dispatch_event` requeued a bound device's interrupt TRB only for Success and -/// Short Packet. **Every other code was dropped where it was read** — no log -/// line, no requeue, no fault — and that endpoint carries exactly one TRB, so -/// the device went silent for the rest of the boot with every bind-time line -/// reading perfectly. A Logitech mouse hot-plugged into the T14 did exactly -/// that: `HID mouse ready on slot 6 … merges as source 1` at 30.485 s and not -/// one motion event until it was unplugged at 58.659 s. -/// -/// Both timings, because they are different states of the driver and neither is -/// a weaker version of the other: the **fourth** completion is a device that has -/// been delivering and stops, and the **first** is a freshly configured endpoint -/// that never delivered at all — which is the shape the T14 showed and the one -/// whose recovery has to work before any report has ever arrived. -/// -/// The actuator is a boot parameter and `xhci/hid.rs`'s `stage_break` says why -/// nothing on the host side can reach it. What it replaces is the completion -/// code **and the report that transfer delivered**: QEMU really moved a mouse -/// report into the buffer, so a driver that dispatched it despite the error -/// would publish a delta it never earned and this gate would pass against the -/// defect it names. Everything the recovery reads is the controller's own — the -/// Endpoint State out of the output device context, and three commands the -/// controller really answers. -/// -/// Ground truth is host-side: the keys and the pointer delta injected **after** -/// the staged failure arrive in the guest's own event stream, on a machine -/// (`i8042=off`, boot-time HID absolute-only) where no other device can produce -/// either. -pub fn xhci_hid_break( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - for (features, which) in [ - ( - &["xhci-hid-break-first"][..], - "the very first completion, before the device ever delivered", - ), - ( - &["xhci-hid-break-late"][..], - "the fourth completion, after the device had been delivering", - ), - ] { - hid_break_boot(test_config, c_bins, rust_bins, features, which)?; - } - Ok(()) -} - -/// One boot of [`xhci_hid_break`], with the break staged at whichever -/// completion `feature` names. -fn hid_break_boot( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], - params: &'static [&'static str], - which: &str, -) -> Result<(), String> { - /// The delta the assertion is about, injected after the break is spent. - /// Neither component is a number any other move in this boot produces. - const DX: i32 = 40; - const DY: i32 = -30; - /// Typed after the break is spent, and nowhere else in the boot. - const WORD: [&str; 5] = ["h", "e", "l", "l", "o"]; - /// Both QEMU HID devices carry one IN interrupt endpoint at address 1, so - /// this is the device's own number for it and the controller's, and a line - /// naming either wrongly stops matching. - const ENDPOINT: &str = "interrupt endpoint 0x81 (dci 3)"; - - let options = BootOptions { - profile: Profile::MetalHotplug, - qmp: true, - // The only keyboard on the machine has to be the one plugged in, or - // QEMU delivers the keystrokes over PS/2 and every assertion below - // passes with the interrupt endpoint dead. - i8042: false, - kernel_params: params, - ..Default::default() - }; - let argv = qemu::profile_argv(&options); - let usb = crate::usb_argv(&argv); - for absent in ["usb-kbd", "usb-mouse"] { - if usb.iter().any(|d| d.starts_with(absent)) { - return Err(format!("{absent} is on the bus at boot; argv has {usb:?}")); - } - } - if !argv.iter().any(|a| a.contains("i8042=off")) { - return Err("the i8042 is on; a PS/2 keyboard could deliver instead".to_string()); - } - - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - let boot = qemu.boot_log().to_string(); - let Some((scale_x, scale_y)) = crate::parse_rel_scale(&boot) else { - return Err(format!("the kernel never said what pointer scale it used:\n{boot}")); - }; - - let result = qemu.run_test_hooked( - "test_rs_input_events", - Duration::from_secs(60), - "===INPUT_READY===", - move |socket| { - let mut devices = qemu::QmpDevices::open(socket); - devices.add("usb-mouse", "xhci1.0", "hidmouse", &[]); - devices.add("usb-kbd", "xhci1.0", "hidkbd", &[]); - drop(devices); - thread::sleep(Duration::from_millis(800)); - - // Spend the break. Ten pointer completions and six keyboard ones, - // against an injection that strikes the first or the fourth: the - // margin is for QEMU coalescing rel events it has not been polled - // for, which can only make the count smaller. - let mut input = qemu::QmpInput::open(socket); - for _ in 0..10 { - input.mouse(4, 4, None); - thread::sleep(Duration::from_millis(60)); - } - for key in ["a", "b", "c"] { - input.keys(&[(key, true), (key, false)]); - thread::sleep(Duration::from_millis(60)); - } - drop(input); - thread::sleep(Duration::from_millis(300)); - - // And the measured phase, every event of which is after the break. - let mut input = qemu::QmpInput::open(socket); - // The accumulated position clamps at 0, so a move up or left from - // the origin is invisible — and with the first completion eaten the - // pointer may still be sitting there. - input.mouse(200, 200, None); - thread::sleep(Duration::from_millis(150)); - input.mouse(DX, DY, None); - thread::sleep(Duration::from_millis(150)); - for key in WORD { - input.keys(&[(key, true), (key, false)]); - thread::sleep(Duration::from_millis(30)); - } - crate::input_events_end(&mut input); - drop(input); - thread::sleep(Duration::from_millis(200)); - }, - ); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}\n{}", result.serial, result.stdout)); - } - let log = format!("{boot}{}", result.serial); - for bad in ["PANIC:", "panicked at"] { - if log.contains(bad) { - return Err(format!("{bad:?} with the break staged at {which}\n{log}")); - } - } - - // **Delivery first**, because it is what the gate is about and what the - // pre-fix driver cannot do: an endpoint whose completion was dropped holds - // no TRB and nothing ever puts one back, so everything injected after the - // break stays on the host side of the wire. - let word: String = WORD.concat(); - hotplug_delivered(&result.stdout, &word, (DX * scale_x, DY * scale_y)).map_err(|e| { - format!("with the break staged at {which}, input never came back: {e}\n{log}") - })?; - - // Then that a break was staged at all, and that the line names the device, - // the endpoint and the code. Without this the boot above is one where - // nothing failed — and the line itself is the instrument: the T14's log - // cannot name the code its mouse died of, because the driver discarded it. - let named: Vec<&str> = log.lines().filter(|l| l.contains(ENDPOINT)).collect(); - let want = format!("{ENDPOINT} completed with code 6 (Stall Error); failure 1 of 8"); - let staged: Vec<&&str> = named.iter().filter(|l| l.contains(want.as_str())).collect(); - if staged.len() != 2 { - return Err(format!( - "{} endpoint(s) reported a broken completion, want the mouse and the \ - keyboard: {named:?}\n{log}", - staged.len() - )); - } - // Which two devices those were, by the controller they are on and the slot - // they hold on it — and both halves are needed. **A slot id is one - // controller's numbering and this machine has two**: the boot disk is slot 1 - // on `00:02.0` and the mouse plugged in below is slot 1 on `00:03.0`. - let mut broken: Vec<(&str, &str)> = Vec::new(); - for line in &staged { - broken.push(hid_broke_on(line)?); - } - for kind in ["pointer", "keyboard"] { - if !broken.iter().any(|(k, _)| *k == kind) { - return Err(format!("no {kind:?} among the broken completions {broken:?}\n{log}")); - } - } - if broken[0].1 == broken[1].1 { - return Err(format!( - "both broken completions are on {}, so this boot broke one device twice rather \ - than the mouse and the keyboard once each\n{log}", - broken[0].1 - )); - } - - // The endpoint state the recovery had to be chosen for, read out of the - // controller's own output device context. The transfer really completed, so - // the endpoint really is Running — `Halted` here would mean the injection - // staged a shape this boot cannot produce and everything above proves - // something else. - // - // **Once for each of the two devices the injection struck, and no longer a - // count over the whole boot.** `endpoint 3` is the first IN endpoint of - // *every* USB device — the boot disk's bulk IN as much as a HID interrupt - // endpoint — so one transport recovery on the boot disk anywhere in the boot - // used to red this test with a failure about HID: three CI runs did exactly - // that (`31405969578` shard 10, `31424496450`, `31601325987`), and in the - // first the disk's own `slot 1 endpoint 3` and `slot 1 endpoint 4` at - // 2.639 s — a `SCSI 0x35` status-phase break on a shard measured at 2.16x - // boot width — were counted beside the mouse's and the keyboard's. - let mut recovered: Vec<(&str, usize)> = Vec::new(); - for (_, who) in &broken { - let running = format!("xHCI: {who} endpoint 3 is Running, recovering"); - recovered.push((who, log.matches(running.as_str()).count())); - } - if recovered.iter().any(|(_, n)| *n != 1) { - let states: Vec<&str> = log.lines().filter(|l| l.contains(", recovering")).collect(); - return Err(format!( - "the two devices the injection struck were found Running {recovered:?} time(s), \ - want once each; every recovery this boot: {states:?}\n{log}" - )); - } - - // `run_command` logs only refusals, so each of these is the controller - // declining a command the endpoint's state did not permit. - for illegal in [ - "Reset Endpoint failed", - "Stop Endpoint failed", - "Set TR Dequeue failed", - "would not clear the halt", - "is being let go", - ] { - if log.contains(illegal) { - return Err(format!("{illegal:?} after a single staged failure\n{log}")); - } - } - serial::Serial::named("boot console", log.as_str()).must_be_clean()?; - - eprintln!( - " [xhci] a HID interrupt endpoint broken at {which}: {broken:?} named the code, were \ - each found Running once and restarted (of {} recoveries in the boot), and {word:?} \ - plus a {:?} pointer delta crossed them afterwards", - log.matches("is Running, recovering").count(), - (DX * scale_x, DY * scale_y) - ); - Ok(()) -} - -/// Which device an `xHCI: USB on slot : interrupt endpoint …` -/// line is about, as the kind and the device. -/// -/// **Refused rather than widened if the line stops naming one.** Recovery lines -/// carry ` slot ` and nothing else that identifies a device, so a test -/// that cannot read this pair off the completion has no way to tell its own -/// device's recovery from another device's — and the only alternative to -/// refusing is the count over every device that reddened this test three times. -fn hid_broke_on(line: &str) -> Result<(&str, &str), String> { - line.split_once("xHCI: USB ") - .and_then(|(_, rest)| rest.split_once(": interrupt endpoint")) - .and_then(|(who, _)| who.split_once(" on ")) - .ok_or_else(|| { - format!("{line:?} does not name the device and the controller its endpoint broke \ - on, so nothing can tell that device's recovery from another's") - }) -} - -/// Devices plugged in **after** the machine has booted. -/// -/// The driver enumerated once, from `init`, and `dispatch_event` advanced past -/// every TRB that was not a transfer completion — Port Status Change Events -/// included. So the set of USB devices was whatever was connected at boot, -/// forever, and a keyboard plugged into a machine with no input did nothing at -/// all: no port line, no slot, no event, and a compositor already holding a -/// keyboard claim that would never produce anything. That machine is -/// indistinguishable from hung, and it is the first thing a person tries. -/// -/// **The actuator is QEMU's own `device_add`, and nothing about the driver is -/// modified to run this.** A USB device attached at runtime goes through the -/// same `usb_device_attach` → `xhci_port_update` → `xhci_port_notify` path a -/// device attached at startup does, so what the guest sees is a real Port -/// Status Change Event with a real device behind it. That is the whole -/// difference from `xhci_slow_connect`, which needs an actuator because -/// it has to aim at a window the boot opens and closes in milliseconds; here -/// the window is the entire life of the machine. -/// -/// Every claim below is host-side in the sense that matters: -/// -/// - **the keyboard** is the only one on the machine — `i8042=off`, and the -/// profile's boot-time HID is a tablet — so a keystroke that reaches -/// userland can only have crossed a device that was added after the boot; -/// - **the pointer** is the only *relative* one, so QEMU has no handler for an -/// injected `rel` event until it is plugged in, and the boot-time tablet -/// cannot stand in for it; -/// - **the disk's block count** is the size of a file the harness made, which -/// the guest can only have learned by running READ CAPACITY over the wire. -pub fn xhci_hotplug( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - /// The disk that arrives late: 48 GiB, which is no other device in this - /// suite and no round number the driver could have printed by accident. - /// Sparse, so the host pays for nothing. - const HOT_DISK_BYTES: u64 = 48 * 1024 * 1024 * 1024; - /// What the boot-time tablet takes, so what a late pointer must not. - const BOOT_SOURCE: u32 = 1; - const LATE_SOURCE: u32 = 2; - const DX: i32 = 40; - const DY: i32 = -30; - - let options = BootOptions { - profile: Profile::MetalHotplug, - qmp: true, - // With an i8042 on the machine QEMU would deliver the injected - // keystrokes over PS/2 and every assertion below would pass with the - // hot-plug path dead. - i8042: false, - ..Default::default() - }; - // The claim is about what is *not* on the machine at boot, and argv is the - // only place absence is visible: no console line distinguishes "the driver - // never enumerated a keyboard" from "there was never one to enumerate". - let argv = qemu::profile_argv(&options); - let usb = crate::usb_argv(&argv); - for absent in ["usb-kbd", "usb-mouse"] { - if usb.iter().any(|d| d.starts_with(absent)) { - return Err(format!("{absent} is on the bus at boot; argv has {usb:?}")); - } - } - if !usb.iter().any(|d| d.starts_with("usb-tablet")) { - return Err(format!("this gate needs the boot-time tablet, argv has {usb:?}")); - } - if !argv.iter().any(|a| a.contains("i8042=off")) { - return Err("the i8042 is on; a PS/2 keyboard could deliver instead".to_string()); - } - - let image = test_dir().join("usb-hotplug.img"); - drop(sparse(&image, HOT_DISK_BYTES)); - - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - let boot = qemu.boot_log().to_string(); - - // Both controllers came up and exactly one of them found nothing — the - // T14's Thunderbolt xHC exactly, and the controller everything below is - // plugged into. Without this the test could not tell a device enumerated - // late from one enumerated at boot on a port nobody looked at. - let found = boot.matches("xHCI: found at PCI ").count(); - if found != 2 { - return Err(format!("{found} controller(s) initialised, want 2:\n{boot}")); - } - let empty = boot.matches("xHCI: no HID devices on the controller").count(); - if empty != 1 { - return Err(format!( - "{empty} controller(s) reported an empty bus at boot, want the second one only:\n{boot}" - )); - } - let at_boot = crate::parse_xhci_binds(&boot); - if at_boot.len() != 1 || at_boot[0].kind != "tablet" { - return Err(format!("want exactly the tablet bound at boot, got {at_boot:?}\n{boot}")); - } - let booted: Vec = crate::parse_pointer_sources(&boot).iter().map(|(_, s)| *s).collect(); - if booted != vec![BOOT_SOURCE] { - return Err(format!( - "the boot-time tablet did not take source {BOOT_SOURCE} alone: {booted:?}\n{boot}" - )); - } - let Some((scale_x, scale_y)) = crate::parse_rel_scale(&boot) else { - return Err(format!("the kernel never said what pointer scale it used:\n{boot}")); - }; - - let hot_image = image.clone(); - let result = qemu.run_test_hooked( - "test_rs_input_events", - Duration::from_secs(60), - "===INPUT_READY===", - move |socket| { - // One monitor at a time: a `-qmp unix:…,server` socket serves one - // connection, so each phase opens, does its work and closes. - let mut devices = qemu::QmpDevices::open(socket); - devices.blockdev_add("hotdisk", &hot_image); - devices.add("usb-mouse", "xhci1.0", "hotmouse", &[]); - devices.add("usb-kbd", "xhci1.0", "hotkbd", &[]); - devices.add("usb-storage", "xhci1.0", "hotdisk0", &[("drive", "hotdisk")]); - drop(devices); - // The driver's own debounce is 100 ms and the enumeration behind it - // is microseconds under TCG; this is that with room, not a settling - // time the assertions depend on. - thread::sleep(Duration::from_millis(800)); - - let mut input = qemu::QmpInput::open(socket); - // Off the origin first: the accumulated position clamps at 0, so a - // move up or left from there is invisible. - input.mouse(100, 100, None); - thread::sleep(Duration::from_millis(100)); - input.mouse(DX, DY, None); - thread::sleep(Duration::from_millis(100)); - for key in ["h", "e", "l", "l", "o"] { - input.keys(&[(key, true), (key, false)]); - thread::sleep(Duration::from_millis(20)); - } - drop(input); - thread::sleep(Duration::from_millis(200)); - - let mut devices = qemu::QmpDevices::open(socket); - devices.del("hotmouse"); - devices.del("hotdisk0"); - drop(devices); - thread::sleep(Duration::from_millis(800)); - - // The keyboard is still there, and still the only one. - let mut input = qemu::QmpInput::open(socket); - for key in ["w", "o", "r", "l", "d"] { - input.keys(&[(key, true), (key, false)]); - thread::sleep(Duration::from_millis(20)); - } - drop(input); - - // And a pointer plugged in where the last one was unplugged, which - // is the only thing that can show the button-table entry came back: - // a driver that leaked it binds this one as source 3. - let mut devices = qemu::QmpDevices::open(socket); - devices.add("usb-mouse", "xhci1.0", "hotmouse2", &[]); - drop(devices); - thread::sleep(Duration::from_millis(800)); - crate::input_events_end(&mut qemu::QmpInput::open(socket)); - }, - ); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}\n{}", result.serial, result.stdout)); - } - let log = format!("{boot}{}", result.serial); - for bad in ["PANIC:", "panicked at"] { - if log.contains(bad) { - return Err(format!("{bad:?} while devices came and went\n{log}")); - } - } - - hotplug_bound(&log)?; - hotplug_delivered(&result.stdout, "hello", (DX * scale_x, DY * scale_y))?; - hotplug_unbound(&log)?; - // The keyboard was untouched by the mouse's teardown, and the merge it - // shares with the pointer that went away still works. `world` is typed - // after the unplug and nothing else in this boot can produce it. - let typed: String = crate::parse_key_events(&result.stdout) - .iter() - .filter(|e| e.modifiers & 0x10 == 0) - .map(|e| e.translated.as_str()) - .collect(); - if !typed.contains("world") { - return Err(format!( - "typed {typed:?} — the keyboard beside the unplugged pointer stopped delivering\n{}", - result.stdout - )); - } - - // The replug. Source 2 twice is the assertion: an entry that is leaked and - // one that is handed back read the same from every other angle. - let sources: Vec = crate::parse_pointer_sources(&log).iter().map(|(_, s)| *s).collect(); - if sources != vec![BOOT_SOURCE, LATE_SOURCE, LATE_SOURCE] { - return Err(format!( - "pointer sources were {sources:?}, want the boot tablet's {BOOT_SOURCE} and then \ - {LATE_SOURCE} twice — the second late pointer took a fresh entry, so the first \ - one's was never given back\n{log}" - )); - } - - let _ = std::fs::remove_file(&image); - eprintln!( - " [xhci] after boot: a keyboard, a pointer and a {HOT_DISK_BYTES} B disk enumerated on a \ - controller that had nothing on it; typed keys and a {:?} pointer delta delivered \ - host-side; unplug released source {LATE_SOURCE}, disabled the slots and took the disk \ - offline; the replug took source {LATE_SOURCE} again", - (DX * scale_x, DY * scale_y) - ); - Ok(()) -} - -/// Everything the guest has to say about three devices that were not there -/// when it booted. -fn hotplug_bound(log: &str) -> Result<(), String> { - // 48 GiB in the 4 KiB blocks the driver counts in, which is a number that - // exists only because the guest asked the device. - let geometry = format!("{} blocks of 512 B", 48u64 * 1024 * 1024 * 1024 / 4096); - for want in [ - "xHCI: USB keyboard ready on slot", - "xHCI: USB mouse ready on slot", - "usb-storage: disk 1 ready on slot", - geometry.as_str(), - ] { - if !log.contains(want) { - return Err(format!("nothing enumerated after the boot: no {want:?}\n{log}")); - } - } - Ok(()) -} - -/// That the devices which arrived late are the ones delivering. -fn hotplug_delivered(stdout: &str, word: &str, want: (i32, i32)) -> Result<(), String> { - let typed: String = crate::parse_key_events(stdout) - .iter() - .filter(|e| e.modifiers & 0x10 == 0) - .map(|e| e.translated.as_str()) - .collect(); - if !typed.contains(word) { - return Err(format!( - "typed {typed:?}, want it to contain {word:?} — this machine has no keyboard but the \ - one plugged in after it booted\n{stdout}" - )); - } - let pointer = crate::parse_mouse_events(stdout); - let deltas: Vec<(i32, i32)> = pointer - .windows(2) - .map(|w| (w[1].x as i32 - w[0].x as i32, w[1].y as i32 - w[0].y as i32)) - .collect(); - if !deltas.contains(&want) { - return Err(format!( - "no pointer event moved by {want:?}; deltas seen: {deltas:?} — the boot-time tablet \ - is absolute, so a relative move can only have come from the mouse that was plugged \ - in\n{stdout}" - )); - } - Ok(()) -} - /// A device pulled and pushed back before the driver has looked at the port /// twice — which is what a person replugging a mouse does. /// @@ -4106,32 +3446,6 @@ fn take_one(pool: &mut Vec, id: u8) -> bool { } } -/// Everything a device that has been pulled has to leave behind. -fn hotplug_unbound(log: &str) -> Result<(), String> { - for want in [ - "xHCI: port ", - "disconnected", - "unplugged from port", - "source 2 released", - "xHCI: slot ", - "disabled", - "usb-storage: disk 1 unplugged", - "it is offline", - ] { - if !log.contains(want) { - return Err(format!("an unplugged device left {want:?} unsaid\n{log}")); - } - } - // The teardown ran the commands the controller's state permits. Every one - // of these lines is `run_command` reporting one it refused. - for illegal in ["Disable Slot failed", "Disable Slot timed out"] { - if log.contains(illegal) { - return Err(format!("{illegal:?} during the teardown\n{log}")); - } - } - Ok(()) -} - /// A disk the driver refuses, on the port the controller enumerates *first*. /// /// `bind` claims a 64 KiB DMA pool block, issues Configure Endpoint — which diff --git a/tests/common/volumes.rs b/tests/common/volumes.rs index 3273dc85603..550f20d03ad 100644 --- a/tests/common/volumes.rs +++ b/tests/common/volumes.rs @@ -781,8 +781,7 @@ fn rotation( // log — the sink drains everything pending before it looks at the size — // so a metal-sim boot makes a handful, measured at four. That is under the // retention bound, which is why this only requires the count to stay inside - // it; deleting the oldest is `wall_clock_file`'s claim, staged with a full - // volume rather than hoped for here. + // it. if logs.len() < 2 || logs.len() > super::wallclock::MAX_LOG_FILES { return Err(format!( "the volume holds {} log files, wanted 2..={}: {}", @@ -1160,153 +1159,6 @@ pub fn writeback_durability( Ok(()) } -/// A `FatBacking` handed out before an unlink reads nothing after it, and the -/// delete-and-reallocate cycle leaves the volume a volume. -/// -/// `FatFs::delete` frees the file's clusters, and FSInfo's `next_free` is walked -/// down to the lowest one freed — so the next allocation on the volume takes -/// them. Until `FatFs::revoke` existed, a process holding a descriptor across -/// somebody else's `rm` demand-paged whatever went into those clusters next. -/// The guest (`test_rs_fat_backing_revoked`) stages exactly that and asserts the -/// read is **refused**. -/// -/// This is the **independent oracle**, and it answers the two questions the -/// guest cannot ask about itself: -/// -/// - **were the clusters really reissued?** The attacker file is read off the -/// image by the `fatfs` crate and must hold its own bytes end to end, and the -/// victim's name must be gone from the directory. A run in which the volume -/// simply had room elsewhere is not a run in which the refusal proved -/// anything, and only the host's own FAT implementation can say. -/// - **did the cycle break the format?** `toyos-fat32-check` (fatgen103) reads -/// the volume after it, against a partition asserted clean before the boot. -/// A revocation that also corrupted the FAT would pass every assertion the -/// guest can make and leave a stick that does not boot. -pub fn fat_backing_revoked( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - /// Mirrored in `tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs`. - const VICTIM: &str = "fat-revoke-victim.bin"; - const ATTACKER: &str = "fat-revoke-attacker.bin"; - const CONTROL: &str = "fat-revoke-control.bin"; - const LEN: usize = 8 * 4096; - const VICTIM_BYTE: u8 = 0xA7; - const ATTACKER_BYTE: u8 = 0x5C; - - let image_path = test_dir().join("fat-backing-revoked.img"); - let image = qemu::build_boot_image(test_config, c_bins, rust_bins, &[]); - std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - let (start, len) = log_extent(&image, &image_path)?; - - // Born clean, asserted rather than assumed, so a complaint after the run is - // the guest's and not one it inherited. - let complaints_before = check(&image[start..start + len]); - if !complaints_before.is_empty() { - return Err(format!( - "the log partition was not born clean, so this gate cannot tell a complaint the \ - guest caused from one it inherited:\n{}", - describe(&complaints_before) - )); - } - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - boot_image: Some(qemu::Staged::Written(image_path.clone())), - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; - if !boot.contains("log-volume: partition mounted") { - return Err(format!( - "the log partition did not mount, so the guest had nowhere to stage the unlink:\n{}", - volume_lines(&boot) - )); - } - - let result = qemu.run_test("test_rs_fat_backing_revoked", Duration::from_secs(60)); - if let Some(err) = &result.error { - return Err(format!("the guest stopped answering: {err}\nserial:\n{}", result.serial)); - } - if result.exit_code != Some(0) { - // The kernel's own lines too: the refusal reaches userland as one `Io`, - // and which layer refused is only in a `log!`. - return Err(format!( - "fat_backing_revoked guest failed:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - serial::Serial::named("test serial", result.serial.as_str()).must_be_clean()?; - - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - - let after = std::fs::read(&image_path).map_err(|e| format!("read the image back: {e}"))?; - if after.len() != image.len() { - return Err(format!("the image is {} bytes, was {}", after.len(), image.len())); - } - let volume = &after[start..start + len]; - - // The strongest claim first: the volume is still a volume. A revocation that - // freed the chain wrongly would pass every byte comparison below and leave a - // stick that cannot boot. - let complaints_after = check(volume); - if !complaints_after.is_empty() { - return Err(format!( - "the unlink-and-reallocate cycle left the log volume breaking the format:\n{}", - describe(&complaints_after) - )); - } - - let mut files = read_files(volume, &[VICTIM, ATTACKER, CONTROL])?; - let control = need(files.pop().flatten(), CONTROL)?; - let attacker = need(files.pop().flatten(), ATTACKER)?; - if files.pop().flatten().is_some() { - return Err(format!( - "{VICTIM} is still on the volume after the guest unlinked it — the delete never \ - reached the directory, so nothing about a reissued cluster was staged" - )); - } - - for (name, bytes, want) in - [(ATTACKER, &attacker, ATTACKER_BYTE), (CONTROL, &control, VICTIM_BYTE)] - { - if bytes.len() != LEN { - return Err(format!( - "{name} is {} bytes on the volume; the guest wrote {LEN}", - bytes.len() - )); - } - if let Some(at) = bytes.iter().position(|&b| b != want) { - return Err(format!( - "{name} holds {:#04x} at byte {at} on the volume, not {want:#04x} — the host's \ - own FAT implementation does not see the file the guest wrote", - bytes[at] - )); - } - } - - let _ = std::fs::remove_file(&image_path); - eprintln!( - " [fat] the victim is gone from the volume, the {LEN}-byte file written into its place \ - holds {ATTACKER_BYTE:#04x} end to end on the host's own reader, and the checker is silent" - ); - Ok(()) -} - /// F5's negative control: under `usb-flush-fails` a second `fsync` must refuse /// like the first, because the mount's device commit is still owed — the /// pre-generation kernel answered the second call with success and issued @@ -1586,8 +1438,7 @@ pub fn ftruncate_flush_race( } /// A rename whose source is absent leaves the destination on the FAT `/log` -/// volume, and `rename(p, p)` leaves the file — the **independent oracle** for -/// the same reason `fat_backing_revoked` is one. The guest +/// volume, and `rename(p, p)` leaves the file — the **independent oracle**. The guest /// (`test_rs_fs_rename_durable`) stages both; after the shutdown drain the two /// files are read off the image by the `fatfs` crate and must hold their bytes /// end to end, and `toyos-fat32-check` reads the whole volume against a diff --git a/tests/common/wallclock.rs b/tests/common/wallclock.rs index 33d50a1e032..a8a18503cd6 100644 --- a/tests/common/wallclock.rs +++ b/tests/common/wallclock.rs @@ -71,24 +71,6 @@ const WINDOW_MARKER: &str = "between-tests-window-is-captured"; /// quietly weakening the gate. pub const MAX_LOG_FILES: usize = 16; -/// The name of the boot the guest must delete, and the ones it must not. -/// -/// Sixteen staged files, so the volume is exactly at the bound before the guest -/// starts and this boot's own file is the one that puts it over. They are dated -/// well before [`RTC_BASE`], so the one that has to go is unambiguous — and -/// they are *not* consecutive days, so an implementation deleting "the first one -/// it listed" or "the lowest index" rather than the oldest lands on a different -/// file than the one asserted. -fn staged_logs() -> Vec<(String, Vec)> { - (0..MAX_LOG_FILES) - .map(|i| { - let name = format!("2019-12-{:02}-120000.log", 1 + i * 2); - let body = format!("staged by the host as boot {i}\n").into_bytes(); - (name, body) - }) - .collect() -} - /// The log files this module's kernel wrote or was given, oldest first. fn logs(entries: &[Entry]) -> Vec<&Entry> { entries.iter().filter(|e| toyos_build::bootlog::is_logd_file(&e.name)).collect() @@ -106,13 +88,6 @@ fn probed_epoch(log: &str) -> Option { rest.split_whitespace().next()?.parse().ok() } -/// What the same probe printed for `std`'s `SystemTime::now`. -fn probed_std_epoch(log: &str) -> Option { - let line = log.lines().find(|l| l.contains("wall-clock: std_epoch="))?; - let rest = line.split("std_epoch=").nth(1)?; - rest.split_whitespace().next()?.parse().ok() -} - fn clock_lines(log: &str) -> String { let lines: Vec<&str> = log .lines() @@ -213,124 +188,6 @@ fn boot_and_read( Ok((entries, log)) } -/// One file per boot, named and stamped from the wall clock, with the oldest -/// deleted once the volume is at its bound. -pub fn wall_clock_file( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let staged = staged_logs(); - let (entries, log) = - boot_and_read(test_config, c_bins, rust_bins, "wall-clock-boot.img", &[], &staged)?; - let logs = logs(&entries); - - // The bound, on the volume rather than in the guest's account of it. - if logs.len() != MAX_LOG_FILES { - return Err(format!( - "the volume holds {} logs, not the {MAX_LOG_FILES} the bound allows: {}\n{}", - logs.len(), - names(&logs), - clock_lines(&log) - )); - } - - // The oldest staged file and no other. `staged_logs` dates them two days - // apart, so deleting by list order or by index would take a different one. - let oldest = &staged[0].0; - if logs.iter().any(|e| &e.name == oldest) { - return Err(format!( - "the oldest log {oldest} is still on the volume, so the bound was met by deleting \ - something else: {}", - names(&logs) - )); - } - for (name, _) in &staged[1..] { - if !logs.iter().any(|e| &e.name == name) { - return Err(format!( - "{name} was deleted and it is not the oldest: {}\n{}", - names(&logs), - clock_lines(&log) - )); - } - } - if !log.contains(&format!("/log/{oldest} was deleted")) { - return Err(format!( - "nothing in the log names {oldest} as the file that was deleted to make room\n{}", - clock_lines(&log) - )); - } - - // This boot's own file: named for the staged instant, which needs the - // century register to come out as 2101 rather than 2001. - let Some(mine) = logs.iter().find(|e| e.name.starts_with(RTC_BASE_DATE)) else { - return Err(format!( - "no log named for the day the host staged ({RTC_BASE_DATE}*): {}\n{}", - names(&logs), - clock_lines(&log) - )); - }; - let drift = mine.modified - RTC_BASE_SECS; - if !(0..=MAX_BOOT_DRIFT_SECS).contains(&drift) { - return Err(format!( - "{} carries a timestamp {drift}s from the {RTC_BASE} the host set, outside \ - 0..={MAX_BOOT_DRIFT_SECS}\n{}", - mine.name, - clock_lines(&log) - )); - } - if mine.len == 0 { - return Err(format!("{} is on the volume and empty", mine.name)); - } - - // The other end of the same instant. Firmware named no zone on this - // machine — OVMF ships `EFI_UNSPECIFIED_TIMEZONE` — so the kernel takes the - // RTC as UTC and the epoch syscall must answer the staged instant itself. - // A kernel serving 1970, or serving local time shifted by a zone it - // invented, lands outside this window. - let Some(epoch) = probed_epoch(&log) else { - return Err(format!( - "the guest never printed what `SYS_CLOCK_EPOCH` answered\n{}", - clock_lines(&log) - )); - }; - let epoch_drift = epoch - RTC_BASE_SECS; - if !(0..=MAX_BOOT_DRIFT_SECS).contains(&epoch_drift) { - return Err(format!( - "`SYS_CLOCK_EPOCH` answered {epoch}, {epoch_drift}s from the {RTC_BASE} the host set \ - and outside 0..={MAX_BOOT_DRIFT_SECS}\n{}", - clock_lines(&log) - )); - } - - // Judged against `RTC_BASE` and not against the syscall above, so a std - // agreeing with a wrong kernel still lands outside this window. - let Some(std_epoch) = probed_std_epoch(&log) else { - return Err(format!( - "the guest never printed what std's `SystemTime::now` answered\n{}", - clock_lines(&log) - )); - }; - let std_drift = std_epoch - RTC_BASE_SECS; - if !(0..=MAX_BOOT_DRIFT_SECS).contains(&std_drift) { - return Err(format!( - "std's `SystemTime::now` answered {std_epoch}, {std_drift}s from the {RTC_BASE} the \ - host set and outside 0..={MAX_BOOT_DRIFT_SECS}. A std that never asks the kernel \ - answers the epoch, which is {}s out\n{}", - -RTC_BASE_SECS, - clock_lines(&log) - )); - } - - eprintln!( - " [clock] {} carries {} bytes, stamped {drift}s after the {RTC_BASE} the host set, epoch \ - {epoch_drift}s after it and std {std_drift}s after it; {} deleted for the \ - {MAX_LOG_FILES}-log bound", - mine.name, mine.len, oldest - ); - Ok(()) -} - /// A firmware-named zone separates local time from UTC, in the direction UEFI /// defines. /// diff --git a/tests/netcase/system.toml b/tests/netcase/system.toml index 1ae7dc9ddde..c0a2cc97dd6 100644 --- a/tests/netcase/system.toml +++ b/tests/netcase/system.toml @@ -29,11 +29,9 @@ service = true serves = ["netd"] devices = ["pci:1af4:1041"] -# `launcher` for `launcher_refusals`, whose subject is what a client can make -# `/system/bin/init` do — so the test estate has to be able to reach it somewhere. -# This is also the only config whose test binaries take the launcher path to -# `Command::spawn` at all: everywhere else test-runner holds no connector and -# every spawn is direct. +# `launcher`: this is the only config whose test binaries take the launcher +# path to `Command::spawn` at all; everywhere else test-runner holds no +# connector and every spawn is direct. # `logread` is the log gate's own: a spawned test binary does not inherit a # `SysCap` dup, so the gate that reads the kernel's records runs inside # `test-runner` itself. @@ -47,9 +45,7 @@ syscap = ["logread"] # `every_device_class_has_at_most_one_claimant`. devices = ["pci:1af4:1041"] -# What `launcher_refusals` asks init to start. A declared program that serves -# nothing and provides nothing, so a refused launch of it takes no acceptor -# with it and a granted one is a clean round trip. +# A declared program `spawn_cwd` spawns through init's launcher. [programs.toybox] # What `spawn_cwd` asks for the owner's `cd` and a launch from it: the shell diff --git a/tests/quiescelastcase/system.toml b/tests/quiescelastcase/system.toml index 7a4510d7c06..5145a13dd2a 100644 --- a/tests/quiescelastcase/system.toml +++ b/tests/quiescelastcase/system.toml @@ -1,5 +1,5 @@ -# The boot `quiesce_wakes_on_the_last_park` and `quiesce_wakes_on_the_last_exit` -# judge: its one job starts the two threads the kernel may hold and reboots. +# The boot `quiesce_wakes_on_the_last_park` judges: its one job starts the +# thread the kernel may hold and reboots. [boot] start = ["logd", "test-runner"] diff --git a/tests/test-durations b/tests/test-durations index d76f46dec0c..0db621526dc 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -157,7 +157,6 @@ boot_volume_metadata_error 4968 c_capture_ignores_daemon_lines 4578 cache_eviction 8165 connect_before_serve 23 -console_line_atomicity 8925 console_locale_detect 3922 control_regs 7243 control_regs_negative 3872 @@ -185,7 +184,6 @@ endowment_denied 10 esp_filesystem 10123 exit_wait_storm 178 run_exit_status 0 -fat_backing_revoked 8226 fault_gates 31 foreign_disk_untouched 4688 fpu_isolation 14662 @@ -212,7 +210,6 @@ hda_two_live_refused 4848 heap_ceiling_recovery 10371 hierarchy_paths 87 home_backing_revoked 663 -home_budget_refusal_retried 18661 home_overwrite_reads_back 4835 https_tls13 5166 https_tls13_e1000e 6072 @@ -222,7 +219,6 @@ i8042_fadt_denial 5736 i8042_health 9509 i8042_health_cadence 10236 i8042_kbd_echo 4770 -i8042_keyboard 4550 i8042_mouse 4310 i8042_no_spurious_wake 146 i8042_quarantine 10068 @@ -251,21 +247,17 @@ kernel_log_file 13152 keyboard_claim_close_spares_stdin 4580 kill_while_blocked 44 klogd_hosted 5674 -klogd_panic_halts 16658 lan_dhcp_lease 3029 lan_no_lease 32709 lapic_spurious_vector 6829 late_storage_connect 6455 latency_wake 8790 -launcher_refusals 2106 leak_rollback_selftest 5423 loader_watchdog_arms 10064 locale_detect 9959 locale_detect_unrecognized 160 log_backing_read_error 4887 log_conservation_smp1 4686 -log_conservation_smp4 8248 -log_conservation_smp8 5112 log_flush_retry 20526 log_nested_emit 5008 log_partition_identity 9516 @@ -349,7 +341,6 @@ screen_console_shell 2604 screen_decoder 6 screen_diag_boot 7330 screen_early_panel 5779 -screen_early_panic 4441 screen_fatal_halt 4970 screen_fatal_halt_composited 6908 screen_gop_firmware_mode 12624 @@ -368,9 +359,7 @@ shm_release_reclaims 29 short_sleep_livelock 4823 smp_failed_ap_leaves_no_hole 6672 smp_roster_and_tsc_trail 4926 -so_cache_refusals 7784 sshd_exec 5110 -sshd_fail_closed 2307 sshd_files 535 sshd_key_auth 1428 stall_is_not_a_verdict 0 @@ -402,7 +391,6 @@ tlb_shootdown_waits 107 toybox_cp_volume 18616 toybox_file_tools 205 usb_boot_stick_pulled 15650 -usb_disk_index_stable 6267 usb_flush_optional 18056 usb_pool_exhausted 5237 usb_refused_disk_first 7260 @@ -420,7 +408,6 @@ virtio_used_ring 4189 volume_from_another_disk 7122 wake_storm_cost 337 wall_clock_century_register 9030 -wall_clock_file 4642 wall_clock_no_century 6987 wall_clock_now 13 wall_clock_rtc_dead 8070 @@ -436,8 +423,6 @@ xhci_deaf_registers 21244 xhci_descriptor_walk 5186 xhci_flap 8220 xhci_full_speed_device 8833 -xhci_hid_break 15290 -xhci_hotplug 7687 xhci_many_devices 6247 xhci_msi_only 5757 xhci_no_interrupt 5114 diff --git a/tests/toyos-rust-tests/src/bin/abuse_listener_hijack.rs b/tests/toyos-rust-tests/src/bin/abuse_listener_hijack.rs index 4f565f67c7b..cb81350f675 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_listener_hijack.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_listener_hijack.rs @@ -33,8 +33,7 @@ //! the only wrong-typed handle in the ABI that does**: an `add` entry's //! connector is routinely one a *peer* transferred — a `provides` name is //! exactly that — so presenting the wrong one may be reporting a peer's bug -//! rather than your own. `/system/bin/init`'s launcher is why, and -//! `launcher_refusals` is the other end of the same property. +//! rather than your own. `/system/bin/init`'s launcher is why. //! 3. **A handle number from another process's table names nothing here.** The //! victim prints the raw number of a live acceptor of its own; the thief //! presenting it is ended, and the victim's port is still serving afterwards. diff --git a/tests/toyos-rust-tests/src/bin/console_line_atomicity.rs b/tests/toyos-rust-tests/src/bin/console_line_atomicity.rs deleted file mode 100644 index 8ed0479c3ca..00000000000 --- a/tests/toyos-rust-tests/src/bin/console_line_atomicity.rs +++ /dev/null @@ -1,204 +0,0 @@ -//! Two processes, two `write`s per line, and not one line that belongs to -//! both — nor one a program put on the console itself. -//! -//! **The defect this is aimed at is a `write` being the unit of -//! interleaving.** `println!` is a `LineWriter`: it issues `flush_buf()` and -//! then `inner.write(rest)`, so a line leaves the program in two pieces with an -//! arbitrary gap between them, and anything else written in that gap could -//! land inside the line. Each process assembles its own line before it is a -//! record (`toyos::log::stdio`), so two *processes* is the shape that tests it -//! and two threads would not. -//! -//! The two writes are made by hand rather than through `println!` because the -//! split has to be the subject rather than an implementation detail of `std`: -//! the line is a fixed width, the gap is exactly in the middle, and the newline -//! is on the second write, which is the piece the line waits for. -//! -//! **And the console is not a program's to write.** A program's console is -//! read-only: its lines reach the console only through `logd`, under its name, -//! so each writer's first act is a line on its console that must be refused. -//! -//! The verdict is the host's — a line is mixed or it is not, and only the -//! console capture can say. What this binary owes the host is that both writers -//! ran, said how much, and agreed about it. - -use std::io::Write; -use std::process::{exit, Command}; - -use toyos::log::stdio::{self, Stream}; -use toyos_abi::syscall::{self, SyscallError}; -use toyos_abi::RawHandle; - -const SELF_PATH: &str = "/system/bin/test_rs_console_line_atomicity"; - -/// Lines each writer emits. -/// -/// **A count and not a duration**, so the gate's verdict does not move with the -/// host: the assertion is zero mixed lines out of `2 * LINES`, and a run that -/// produced fewer lines than that has failed the non-vacuity check rather than -/// passed a weaker version of the same test. Both writers' lines fit the -/// runner's log ring unread, so a line missing is one lost, never one the ring -/// had no room for. -const LINES: usize = 400; - -/// Bytes in one line, newline included. -/// -/// Two hundred, which is comfortably inside `MAX_CONSOLE_LINE`'s 1024 — the -/// claim under test is that a whole line is one unit, and a line past that -/// bound is deliberately emitted in pieces of it, which is a different -/// sentence. -const WIDTH: usize = 200; - -/// Bytes the third writer says, in two `write`s, never ending them with a -/// newline. -/// -/// **The other half of the same line.** A line leaves on the `\n` that ends -/// it; the one moment a partial line stops being "not finished yet" and becomes -/// "all there will ever be" is the process exiting, which ends it. Without that -/// these bytes are dropped on the floor — a dying process's last words lost — -/// and a hundred of them arriving whole is also proof the line accumulated -/// across two `write`s to get there. -const MIDLINE: usize = 100; - -/// The byte the third writer repeats. Not `A` or `B`, so the whole-line count -/// cannot see it, and a hundred of them in a row is nothing an ordinary console -/// line contains. -const MIDLINE_BYTE: u8 = b'C'; - -/// Digits of the per-writer sequence number each line carries after its -/// leading tag byte, zero-padded so every line stays exactly [`WIDTH`] bytes. -const SEQ_DIGITS: usize = 6; - -/// The console, at stdin's slot: the runner starts this job holding its own -/// there (`CONSOLE_JOBS`), and each writer inherits it, read-only. -const CONSOLE: RawHandle = RawHandle(0); - -/// A line in the kernel's own shape, which a program on the console could -/// pass off as a record. -const FORGED: &[u8] = b"[kernel 0.000 cpu0] console-atomicity: a record no kernel wrote\n"; - -fn main() { - let mut args = std::env::args(); - let _ = args.next(); - match args.next().as_deref() { - Some("C") => exit_mid_line(), - Some(tag) => { - let byte = tag.as_bytes().first().copied().unwrap_or(b'?'); - write_lines(byte); - } - None => parent(), - } -} - -/// Say half of something, say the other half, and exit without ever ending it. -/// -/// No newline anywhere, so nothing in the write path makes these bytes a -/// line: what does is this process exiting. Through `std`, whose exit is the -/// one that ends it, with a flush between the halves, so the first leaves as -/// a piece the line goes on from. -fn exit_mid_line() { - let partial = [MIDLINE_BYTE; MIDLINE]; - let (head, tail) = partial.split_at(MIDLINE / 2); - let mut out = std::io::stdout(); - let wrote = out.write_all(head).and_then(|()| out.flush()).and_then(|()| out.write_all(tail)); - if let Err(e) = wrote { - eprintln!("console-atomicity: the mid-line writer's write failed: {e}"); - exit(1); - } - exit(0); -} - -/// A piece taken whole, or this process ends saying what it got instead. -fn written(got: Result, len: usize, who: &str) { - if got != Ok(len) { - eprintln!("console-atomicity: {who} wrote {got:?} of {len}"); - exit(1); - } -} - -/// Spawn the two writers and wait for both. -/// -/// Two processes and not two threads: the buffer is per console object and a -/// process gets its own, so two threads of one process share one buffer and -/// would prove nothing about the property the object exists to have. -fn parent() { - let mut children = Vec::new(); - for tag in ["A", "B"] { - match Command::new(SELF_PATH).arg(tag).spawn() { - Ok(child) => children.push((tag, child)), - Err(e) => { - eprintln!("console-atomicity: writer {tag} would not start: {e}"); - exit(1); - } - } - } - for (tag, mut child) in children { - match child.wait() { - Ok(status) if status.code() == Some(0) => {} - Ok(status) => { - eprintln!("console-atomicity: writer {tag} exited {:?}", status.code()); - exit(1); - } - Err(e) => { - eprintln!("console-atomicity: writer {tag} would not be waited for: {e}"); - exit(1); - } - } - } - // **The mid-line writer runs after the other two are gone**, so the bytes - // its exit flushes cannot land inside a line the count above is reading — - // they are on the wire on their own, which is what lets the host look for - // them as a run rather than as a line. - match Command::new(SELF_PATH).arg("C").spawn().and_then(|mut c| c.wait()) { - Ok(status) if status.code() == Some(0) => {} - Ok(status) => { - eprintln!("console-atomicity: the mid-line writer exited {:?}", status.code()); - exit(1); - } - Err(e) => { - eprintln!("console-atomicity: the mid-line writer would not run: {e}"); - exit(1); - } - } - // After all three, so the count the host checks against is a claim about a - // run that finished rather than one still going. It follows the mid-line - // writer's unterminated bytes on the wire, which is why the host finds this - // declaration with a substring search rather than a whole-line one. - println!( - "console-atomicity: writers=2 lines={LINES} width={WIDTH} midline={MIDLINE} \ - seq={SEQ_DIGITS}" - ); -} - -/// One writer: `LINES` numbered lines of one repeated byte, each in two -/// `write`s. -/// -/// The sequence number after the leading tag byte is what lets the host tell -/// a gap in a writer's own run from a capture that ends early — a count alone -/// reads both as the same missing-lines number. -fn write_lines(tag: u8) { - match syscall::write(CONSOLE, FORGED) { - Err(SyscallError::PermissionDenied) => {} - other => { - eprintln!("console-atomicity: writer {} put a line on its console: {other:?}", tag as char); - exit(1); - } - } - let mut line = [tag; WIDTH]; - line[WIDTH - 1] = b'\n'; - for seq in 0..LINES { - let mut digits = [0u8; SEQ_DIGITS]; - let mut rest = seq; - for d in digits.iter_mut().rev() { - *d = b'0' + (rest % 10) as u8; - rest /= 10; - } - line[1..1 + SEQ_DIGITS].copy_from_slice(&digits); - let (head, tail) = line.split_at(WIDTH / 2); - // Refused rather than retried: a short write here is the sink taking - // half a line, which is the defect and not an error to paper over. - for piece in [head, tail] { - written(stdio::write(Stream::Out, piece), piece.len(), &format!("writer {}", tag as char)); - } - } -} diff --git a/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs b/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs deleted file mode 100644 index ee0e124b5b8..00000000000 --- a/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs +++ /dev/null @@ -1,153 +0,0 @@ -//! A FAT32 file backing must not outlive the file it reads. -//! -//! `/log` hands every open file a `FatBacking` holding the volume byte ranges -//! its data lives in. Unlink the file and `Fat32::remove` puts those clusters -//! back in the FAT — and FSInfo's `next_free` is walked *down* to the lowest one -//! freed, so the very next allocation on the volume is the one that takes them. -//! A backing that still names them reads that file's contents: an information -//! disclosure through `open`, `rm` and a write, with nothing crafted about it -//! and no privilege needed. -//! -//! The same defect `home_backing_revoked` covers for `/home`, on the other -//! filesystem and the other allocator. `FatFs::delete` already dropped the -//! *write* handle, so the destructive half was closed and the read half was not. -//! -//! Staged rather than reasoned about: the victim's clusters are freed and then -//! deliberately handed to a file whose bytes are nothing like the victim's, and -//! the still-open descriptor is read afterwards. The host half -//! (`tests/common/volumes.rs::fat_backing_revoked`) shuts the machine down and -//! reads the volume back with an independent FAT implementation and the -//! fatgen103 checker, so what the delete-and-reallocate cycle left on the stick -//! is judged by something that is not the kernel. - -use std::fs; -use std::io::{Read, Write}; -use std::thread; -use std::time::{Duration, Instant}; - -/// Mirrored in `tests/common/volumes.rs::fat_backing_revoked`. Two halves of one -/// fixture; a change to either alone fails loudly rather than passing quietly. -const VICTIM: &str = "/log/fat-revoke-victim.bin"; -const ATTACKER: &str = "/log/fat-revoke-attacker.bin"; -const CONTROL: &str = "/log/fat-revoke-control.bin"; - -/// Eight pages. More than one so the read crosses pages, and — at the 512-byte -/// clusters a 34 MiB FAT32 volume gets — sixty-four clusters, so each page is -/// several extents and the multi-run half of `FatBacking::read_page` is the one -/// under test. -const LEN: usize = 8 * 4096; - -const VICTIM_BYTE: u8 = 0xA7; -const ATTACKER_BYTE: u8 = 0x5C; - -/// Five of the 2 s operation budgets behind the kernel's `WouldBlock` refusal -/// (`kernel/src/block.rs::OPERATION`): device patience is not what this test -/// is about, so its setup asks again the way logd's flush policy does. -const SETUP_PATIENCE: Duration = Duration::from_secs(10); -const SETUP_PAUSE: Duration = Duration::from_millis(200); - -/// One idempotent setup step, asked again on `WouldBlock` until -/// [`SETUP_PATIENCE`] is spent; anything else panics with the step's message. -fn patient(what: &str, mut op: impl FnMut() -> std::io::Result) -> T { - let start = Instant::now(); - loop { - match op() { - Ok(v) => return v, - Err(e) if e.kind() == std::io::ErrorKind::WouldBlock - && start.elapsed() < SETUP_PATIENCE => - { - thread::sleep(SETUP_PAUSE); - } - Err(e) => panic!("{what}: {e}"), - } - } -} - -fn write_file(path: &str, byte: u8) { - { - let mut f = patient(&format!("create {path}"), || fs::File::create(path)); - // Not `patient`: a `write_all` refused partway has advanced the cursor, - // so asking again blind would double bytes; it lands in the cache anyway. - f.write_all(&vec![byte; LEN]).unwrap_or_else(|e| panic!("write {path}: {e}")); - patient(&format!("fsync {path}"), || f.sync_all()); - } // close: the last handle drops here. - // The last close no longer drops the file from the cache on this thread — - // it pins it and hands the teardown to `iod` (`kernel::writeback`). Let that - // drain, so a later open of this name is served by the backing rather than - // adopting the pages this write left cached; the revocation this test checks - // lives in the backing, and a cache-served read would never reach it. The - // margin is enormous: the drain is microseconds of work. - thread::sleep(Duration::from_millis(200)); -} - -fn read_all(f: &mut fs::File) -> std::io::Result> { - let mut got = Vec::new(); - f.read_to_end(&mut got)?; - Ok(got) -} - -fn main() { - // The control. `write_file` drains the write-back so the file leaves the - // cache, so this open is served by the backing and not by pages the write - // left cached — if it were not, the attack below would prove nothing about - // that path. - write_file(CONTROL, VICTIM_BYTE); - let control = read_all(&mut fs::File::open(CONTROL).expect("open the control")) - .expect("read the control"); - assert_eq!(control.len(), LEN, "the control read short"); - assert!( - control.iter().all(|&b| b == VICTIM_BYTE), - "the backing did not serve the control file's own bytes", - ); - - write_file(VICTIM, VICTIM_BYTE); - - // Held open, and deliberately not read: `write_file` drained the victim out - // of the cache, so every page is absent and each one is a fault the backing - // has to answer. - let mut held = fs::File::open(VICTIM).expect("open the victim"); - - fs::remove_file(VICTIM).expect("unlink the victim"); - - // The victim's clusters are the lowest free ones now, so this takes them. - write_file(ATTACKER, ATTACKER_BYTE); - - // **Refused, not zeroed.** A revoked backing has no bytes to serve and has - // to say so; the byte checks below are kept for the case where the refusal - // does not come, because a backing that still resolves serves either - // {ATTACKER_BYTE:#04x} or {VICTIM_BYTE:#04x} and neither is zero. - let refused = match read_all(&mut held) { - Err(e) => e, - Ok(got) => { - if let Some(at) = got.iter().position(|&b| b == ATTACKER_BYTE) { - panic!( - "byte {at} read through the deleted file's descriptor is \ - {ATTACKER_BYTE:#04x} — the backing served another file's data", - ); - } - if let Some(at) = got.iter().position(|&b| b != 0) { - panic!( - "byte {at} read through the deleted file's descriptor is {:#04x}, not \ - zero — the backing still resolves clusters the FAT has taken back", - got[at], - ); - } - panic!( - "the read through the deleted file's descriptor returned {} bytes and \ - succeeded; a revoked backing has no bytes to serve and has to say so", - got.len(), - ); - } - }; - - // The name is gone as well as the bytes, and a fresh open says so with the - // error a missing file gets rather than the one a revoked backing gets. - assert!(fs::File::open(VICTIM).is_err(), "the unlinked victim still opens by name"); - - // Left on the volume on purpose: the host reads both back off the image - // with its own FAT implementation after the shutdown. - println!( - "a read through a backing whose file was deleted was refused ({refused}) rather than \ - serving any of the next file's {LEN} bytes" - ); -} diff --git a/tests/toyos-rust-tests/src/bin/home_fsync_budget.rs b/tests/toyos-rust-tests/src/bin/home_fsync_budget.rs deleted file mode 100644 index 36cdf5854b7..00000000000 --- a/tests/toyos-rust-tests/src/bin/home_fsync_budget.rs +++ /dev/null @@ -1,30 +0,0 @@ -//! An fsync on `/home` whose first attempt is budget-refused (`fsync-budget-spent`) -//! must retry on a fresh budget and succeed — a `BudgetExpired` reaching the -//! bcachefs adapter as `Io` ends the syscall on attempt 1 instead (F9). -//! `tests/common/storage.rs::home_budget_refusal_retried` boots and judges this. - -use std::fs::File; -use std::io::{Read, Write}; - -/// Mirrored in `tests/common/storage.rs::home_budget_refusal_retried`. -const PATH: &str = "/home/f9-budget.bin"; -const LEN: usize = 3 * 4096 + 41; - -fn pattern() -> Vec { - (0..LEN).map(|i| (i.wrapping_mul(151) ^ 0x3C) as u8).collect() -} - -fn main() { - let want = pattern(); - let mut f = File::create(PATH).expect("create on /home"); - f.write_all(&want).expect("write"); - f.sync_all().expect( - "fsync refused: a budget-expired first attempt must be retried on a fresh budget, \ - never returned as the device's word", - ); - - let mut got = Vec::new(); - File::open(PATH).expect("re-open").read_to_end(&mut got).expect("read back"); - assert_eq!(got, want, "the bytes changed across the refused-then-retried fsync"); - println!("budget-refused fsync retried to durable: {LEN} bytes on /home"); -} diff --git a/tests/toyos-rust-tests/src/bin/launcher_refusals.rs b/tests/toyos-rust-tests/src/bin/launcher_refusals.rs deleted file mode 100644 index ee7ccb2c392..00000000000 --- a/tests/toyos-rust-tests/src/bin/launcher_refusals.rs +++ /dev/null @@ -1,291 +0,0 @@ -//! What a client can make `/system/bin/init` do by sending it a bad launch. -//! -//! **init is the one process the machine cannot lose.** It holds the only -//! `SysCap`, every unhanded acceptor and the `launcher` port, and nothing -//! restarts it — so a client that can end it, panic it or grow its handle -//! table without bound takes the machine's ability to start a process with it. -//! Every field of `MSG_LAUNCH` and every handle in its batch is a client's -//! claim about itself, and the launcher connector is held by the compositor, -//! every terminal, every shell and sshd. -//! -//! Three shapes, each of which was reachable before this gate existed: -//! -//! 1. **A frame whose handle count is not the batch's.** The handles are -//! already in init's table when the count is checked, so a refusal that -//! returns without closing them leaks one per attempt — and a client picks -//! how often it attempts. Measured against the kernel's live-object census -//! rather than believed: the batch is a duplicate of a pipe end this -//! process then drops, so the object survives exactly if init kept it. -//! 2. **An extra that is not a connector.** init has no way to ask what a -//! handle it received names, so it hands it to `SYS_NAMESPACE_BUILD` — and -//! a wrong type there used to end the caller, which is init. -//! 3. **A connector the client narrowed `DUP` away from.** init duplicates a -//! provided connector so the namespace and the label can both carry one; a -//! duplicate that is refused used to be an `.expect`. -//! -//! The fourth arm is what stops the other three passing on a dead launcher: an -//! ordinary spawn, which goes through init because this process holds a -//! `launcher` connector and `/system/bin/toybox` is a declared program. It runs last, -//! so it also asserts init survived all three. -//! -//! **A fourth shape, and it is the one init could not survive at all: a client -//! that connects and says nothing.** `serve_launch`'s first statement was a -//! blocking `recv_header` on the fresh connection, so two syscalls from any -//! holder of the connector — the compositor, every terminal, every shell, sshd -//! — parked the machine's only way to start a process for ever, with init alive -//! and looking healthy. -//! -//! **Every answer this file waits for is bounded, and that is not decoration.** -//! A test that hangs instead of failing is worse than no test: a harness -//! timeout is a liveness guard rather than a verdict, and a guest that never -//! returns takes its whole shared boot down with it. So the launcher's replies -//! are read with [`answer_within`], and a launcher that has stopped answering -//! is an assertion with a name on it. The one arm that cannot be bounded from -//! here — `Command`, which blocks inside `std` — runs after a bounded launch -//! has already proved init is answering. - -use std::process::Command; -use std::time::{Duration, Instant}; - -use toyos::census::Census; -use toyos::ipc::{Connection, FrameRx, RxStep}; -use toyos::launch::{self, Launch}; -use toyos::{namespace, port, AsHandle}; -use toyos_abi::handle::Rights; -use toyos_abi::syscall::{self, SyscallError}; -use toyos_abi::RawHandle; - -/// Rounds per census sample. Large enough that one leaked handle per round is -/// a number no drain lag can hide. -const ROUNDS: usize = 16; - -/// How long init may take to answer a launch before this file calls it wedged. -/// -/// Generous by two orders of magnitude: a refusal is a frame decode and a -/// namespace build, and a grant is one `SYS_SPAWN`. What this bounds is the -/// launcher that answers *never*, and the number only decides how long the red -/// takes to arrive. -const ANSWER_BUDGET: Duration = Duration::from_secs(5); - -/// Clients that connect to the launcher and then say nothing, held open across -/// the launch that must still be answered. -/// -/// Well under init's own `MAX_PENDING_LAUNCHES`, because what is under test is -/// that a silent client costs a slot rather than the event loop — not the bound -/// on how many slots there are. -const QUIET_CLIENTS: usize = 8; - -/// A program `tests/netcase` declares that serves nothing and provides -/// nothing, so a refused launch of it takes no acceptor with it. -const DECLARED: &str = "/system/bin/toybox"; - -fn main() { - the_kernel_answers_rather_than_faults(); - a_quiet_client_does_not_wedge_the_launcher(); - not_a_connector(); - a_connector_it_cannot_duplicate(); - a_working_directory_that_is_not_absolute(); - - let before = churn(); - let after = churn(); - let grown: Vec<_> = after.grown_since(&before).collect(); - assert!( - grown.is_empty(), - "{ROUNDS} more refused launches left more live objects behind: {grown:?} — \ - first {before}, then {after}: init is keeping the handles a refusal took", - ); - println!(" census: {} live objects, then {}", before.total(), after.total()); - - the_launcher_still_works(); - println!("a bad launch is refused, and init is still the launcher"); -} - -fn launcher() -> Connection { - toyos::endow::service("launcher").expect("this process was endowed a launcher connector") -} - -/// Read the launcher's reply without ever blocking on it. -/// -/// `Err` is the verdict this file exists to be able to reach: a launcher that -/// has not answered inside the budget is one no `recv_header` would ever come -/// back from. -fn answer_within(conn: &Connection, budget: Duration) -> Result { - let deadline = Instant::now() + budget; - // Only the reply's type is judged here, so nothing of a payload is kept. - let mut rx = FrameRx::<0>::new(); - loop { - match rx.pump(conn) { - RxStep::Frame { msg_type, .. } => return Ok(msg_type), - RxStep::Eof => return Err("the launcher dropped the connection"), - RxStep::Malformed => return Err("the launcher sent a frame this protocol cannot describe"), - RxStep::Idle => { - if Instant::now() >= deadline { - return Err("the launcher never answered"); - } - std::thread::sleep(Duration::from_millis(5)); - } - } - } -} - -/// Clients that connect and go quiet, and a launch that must be answered anyway. -/// -/// Two silences, because they park a server at different statements: a -/// connection that never writes a byte, and one that writes half a header and -/// stops. The first is what `accept` used to be fused to; the second is what a -/// frame read in one blocking call used to wait out. -fn a_quiet_client_does_not_wedge_the_launcher() { - let quiet: Vec = (0..QUIET_CLIENTS).map(|_| launcher()).collect(); - let half = launcher(); - half.write_nonblock(&[0u8; 4]).expect("half a frame header"); - - // **A launch init answers and does not grant.** What is under test is that - // the event loop reaches a frame at all while other connections are silent; - // a *granted* launch would put a spawned process's output on init's own - // stdio, which is this boot's console, and hand back a `Process` handle for - // the census arm below to account for. - let conn = launcher(); - let mut buf = [0u8; 512]; - let request = Launch { - program: "/system/bin/no-such-program", - argv: b"", - env: b"", - cwd: "/", - extras: &[], - slots: &[], - }; - let len = request.encode(&mut buf).expect("encode a launch"); - conn.send_bytes_with_handles(&[], launch::MSG_LAUNCH, &buf[..len]) - .expect("the launcher took the frame"); - - match answer_within(&conn, ANSWER_BUDGET) { - Ok(launch::MSG_NOT_DECLARED) => {} - Ok(other) => panic!("the launcher answered {other} for a program nothing declares"), - Err(why) => panic!( - "{QUIET_CLIENTS} clients that said nothing and one that said half a header \ - left the machine unable to start a process: {why}", - ), - } - drop(quiet); - drop(half); - println!(" quiet clients: {QUIET_CLIENTS} silent and one half-spoken, and a launch still ran"); -} - -/// One frame that promises no handles, with two in the batch beside it. -/// -/// init cannot answer this — it does not know which handle was for what — so -/// there is no reply to read. What it must do is close them. -fn a_frame_that_lies() { - let (read, write) = toyos::pipe_pair().expect("a pipe of our own"); - let first = syscall::dup(write.as_handle()).expect("a duplicate to send"); - let second = syscall::dup(write.as_handle()).expect("a second duplicate to send"); - - let mut buf = [0u8; 512]; - let request = - Launch { program: DECLARED, argv: b"", env: b"", cwd: "/", extras: &[], slots: &[] }; - let len = request.encode(&mut buf).expect("encode a launch"); - - let conn = launcher(); - conn.send_bytes_with_handles(&[first, second], launch::MSG_LAUNCH, &buf[..len]) - .expect("the launcher took the frame"); - drop(conn); - // Both ends go, so the pipe's objects are alive after this only if init - // still holds one of the duplicates. - drop(read); - drop(write); -} - -fn churn() -> Census { - for _ in 0..ROUNDS { - a_frame_that_lies(); - } - // **A launch init answers, and deliberately not one it grants.** init is - // single-threaded and serves connections in the order they queued, so an - // answer to a request sent after the sixteen is the proof it has served - // every one of them. A *granted* launch would put a process exit inside - // the sampled window, and an exiting process leaves objects on the - // deferred release queue — the sample would be reading that lag. - assert_eq!(a_launch_it_refuses(), launch::MSG_REFUSED, "the synchronising launch was granted"); - Census::now() -} - -/// A launch init answers and does not grant: an extra naming a pipe where a -/// connector belongs. -fn a_launch_it_refuses() -> u32 { - let (_read, write) = toyos::pipe_pair().expect("a pipe of our own"); - let handle = syscall::dup(write.as_handle()).expect("a duplicate to send"); - refused_with(&[("surface", handle)]) -} - -fn not_a_connector() { - assert_eq!(a_launch_it_refuses(), launch::MSG_REFUSED); - println!(" not a connector: refused, and init is still here"); -} - -/// A real connector, narrowed so init cannot duplicate it. -fn a_connector_it_cannot_duplicate() { - let (_acceptor, connector) = port::create().expect("a port of our own"); - // Everything `SYS_NAMESPACE_BUILD` asks for and nothing `dup` does, so - // init gets past the namespace and fails on the label. - let narrowed = syscall::dup_narrowed(connector.as_handle(), Rights::TRANSFER) - .expect("a connector carrying only TRANSFER"); - assert_eq!(refused_with(&[("surface", narrowed)]), launch::MSG_REFUSED); - println!(" a connector it cannot duplicate: refused, and init is still here"); -} - -/// Send one launch carrying `extras` and answer the message type init replied -/// with. The reply is the liveness proof as much as the verdict. -fn refused_with(extras: &[(&str, RawHandle)]) -> u32 { - answer_to("/", extras) -} - -/// Send one launch from `cwd` carrying `extras`, and answer init's reply. -fn answer_to(cwd: &str, extras: &[(&str, RawHandle)]) -> u32 { - let mut buf = [0u8; 512]; - let request = Launch { program: DECLARED, argv: b"", env: b"", cwd, extras, slots: &[] }; - let (handles, count) = request.handles(); - let len = request.encode(&mut buf).expect("encode a launch"); - - let conn = launcher(); - conn.send_bytes_with_handles(&handles[..count], launch::MSG_LAUNCH, &buf[..len]) - .expect("the launcher took the frame"); - answer_within(&conn, ANSWER_BUDGET).expect("init answered the launch") -} - -/// A cwd the launch does not state absolutely. std would join it onto init's -/// own, so a grant here is the child started in a directory nobody named. -fn a_working_directory_that_is_not_absolute() { - for cwd in ["", "tmp"] { - assert_eq!(answer_to(cwd, &[]), launch::MSG_REFUSED, "a launch from cwd {cwd:?} was not refused"); - } - println!(" a working directory that is not absolute: refused, and init is still here"); -} - -/// The non-vacuity arm. `/system/bin/toybox` is a `[programs]` key, so a caller -/// holding a `launcher` connector reaches it through init and not through -/// `SYS_SPAWN` — this only passes while init is alive and still launching. -fn the_launcher_still_works() { - let out = Command::new(DECLARED) - .arg("pwd") - .output() - .expect("the launcher started a declared program"); - assert!(out.status.success(), "the launched program exited {:?}", out.status.code()); -} - -/// **The half of it that is the kernel's**, asserted from a process that can -/// afford to die so that init does not have to. -/// -/// `SYS_NAMESPACE_BUILD`'s added connector is the one handle argument in the -/// ABI that routinely crossed a trust boundary — a `provides` name is exactly -/// a connector somebody else made — so a wrong type there answers a word. Every -/// other `WrongType` in the table still ends the caller, and if this one goes -/// back to doing that, this arm never returns and the test reds on exit 139. -fn the_kernel_answers_rather_than_faults() { - let (_read, write) = toyos::pipe_pair().expect("a pipe of our own"); - // SAFETY: it is not a connector, which is the point — the call must answer - // a word rather than end this process. - let pretend = unsafe { port::Connector::from_raw(write.as_handle()) }; - let refused = namespace::build().add("surface", &pretend).finish(); - let _ = pretend.into_raw(); - assert_eq!(refused.err(), Some(SyscallError::InvalidArgument)); -} diff --git a/tests/toyos-rust-tests/src/bin/quiesce_last.rs b/tests/toyos-rust-tests/src/bin/quiesce_last.rs index 1f240479ee6..581501a31a9 100644 --- a/tests/toyos-rust-tests/src/bin/quiesce_last.rs +++ b/tests/toyos-rust-tests/src/bin/quiesce_last.rs @@ -1,14 +1,12 @@ -//! Two threads the kernel's `quiesce-last-*` actuators can hold, and a reboot. +//! A thread the kernel's `quiesce-last-park` actuator can hold, and a reboot. //! -//! Both carry [`toyos_quiesce::LAST_THREAD`]'s name. One parks for longer than -//! any stop's budget and the other exits at once. `quiesce-last-park` holds -//! the first inside its `SYS_NANOSLEEP`, and `quiesce-last-exit` holds the -//! second inside its `SYS_THREAD_EXIT`. The kernel holds the reset itself until -//! one of them is held, so this program orders nothing. +//! It carries [`toyos_quiesce::LAST_THREAD`]'s name and parks for longer than +//! any stop's budget; `quiesce-last-park` holds it inside its `SYS_NANOSLEEP`. +//! The kernel holds the reset itself until it is held, so this program orders +//! nothing. //! -//! Nothing here asserts: `common::power::quiesce_wakes_on_the_last_park` and -//! `quiesce_wakes_on_the_last_exit` read the kernel's hold line and its `stop:` -//! record. +//! Nothing here asserts: `common::power::quiesce_wakes_on_the_last_park` reads +//! the kernel's hold line and its `stop:` record. use std::time::Duration; @@ -21,12 +19,10 @@ use toyos_quiesce::LAST_THREAD; const PARKED_FOR: Duration = Duration::from_secs(3_600); fn main() { - for body in [park as fn(), exit] { - std::thread::Builder::new() - .name(LAST_THREAD.into()) - .spawn(body) - .expect("spawn a thread the kernel may hold"); - } + std::thread::Builder::new() + .name(LAST_THREAD.into()) + .spawn(park) + .expect("spawn a thread the kernel may hold"); // Comes back only refused. let refused = toyos::power::stop(Stop::Reboot); @@ -37,5 +33,3 @@ fn main() { fn park() { std::thread::sleep(PARKED_FOR); } - -fn exit() {} diff --git a/tests/toyos-rust-tests/src/bin/so_cache_policy.rs b/tests/toyos-rust-tests/src/bin/so_cache_policy.rs deleted file mode 100644 index a922011ccff..00000000000 --- a/tests/toyos-rust-tests/src/bin/so_cache_policy.rs +++ /dev/null @@ -1,160 +0,0 @@ -//! The two refusals the shared-object cache owes, because it never removes an -//! entry: a path whose file changed, and a load past its byte budget. -//! `tests/common/storage.rs::so_cache_refusals` boots and judges this. -//! -//! **Every arm needs a second process.** `SYS_DLOPEN` answers a name this -//! process holds out of its own `lib_paths` before the cache is consulted, so -//! one process can never ask the cache twice about one path. - -use std::process::{Command, Stdio}; - -use toyos_abi::syscall::{self, SyscallError}; - -/// Mirrored in `so_cache_refusals`, which reads its bytes off the device. -const STALE: &str = "/home/so-cache-stale.so"; -const SAME_SIZE: &str = "/home/so-cache-same-size.so"; -const FIRST: &str = "/system/lib/libtls_lib.so"; -const SECOND: &str = "/system/lib/libtls_dlopen_lib.so"; -/// A symbol `FIRST` exports and `SECOND` does not: the verdict is a name. -const ONLY_IN_FIRST: &[u8] = b"tls_get_label"; - -/// 8 MiB of budget against 2 MiB images, so a kernel that never refuses runs -/// out of attempts and says so rather than looping. -const BUDGET_ATTEMPTS: usize = 12; - -const SELF_PATH: &str = "/system/bin/test_rs_so_cache_policy"; - -fn main() { - match std::env::args().nth(1) { - Some(path) => load_and_report(&path), - None => test(), - } -} - -/// Both arms run whatever the first answers: on a kernel with neither refusal -/// both are red, and stopping at the first would hide one control. -fn test() { - let arms = [ - ("stale-size", a_changed_file_is_refused()), - ("stale-mtime", a_same_size_rewrite_is_refused()), - ("budget", the_budget_is_refused()), - ]; - let mut failed = false; - for (name, outcome) in arms { - match outcome { - Ok(said) => println!(" {name}: {said}"), - Err(why) => { - println!(" {name} FAILED: {why}"); - failed = true; - } - } - } - if failed { - std::process::exit(1); - } - println!("the shared-object cache refuses a stale path and a full budget"); -} - -/// One library, loaded; a different one over the same path, loaded again. The -/// second must be refused: the first image is mapped into the child that loaded -/// it, so serving it again is a lie and reloading would map the library twice. -fn a_changed_file_is_refused() -> Result { - let only = String::from_utf8_lossy(ONLY_IN_FIRST).into_owned(); - copy(FIRST, STALE); - let first = load_in_child(STALE); - if !(first.contains("LOADED") && first.contains("SYMBOL-FOUND")) { - return Err(format!("the first load of {STALE} did not resolve {only}: {first}")); - } - - copy(SECOND, STALE); - let second = load_in_child(STALE); - if second.contains("SYMBOL-FOUND") { - return Err(format!( - "{STALE} now holds {SECOND}, which exports no {only} — a kernel that resolved it \ - served the image {FIRST} left in the cache: {second}" - )); - } - if !second.contains("REFUSED NotSupported") { - return Err(format!( - "the load of a path whose file changed was not refused by name: {second}" - )); - } - Ok(format!("{STALE} became {SECOND} and the second load was refused")) -} - -/// The same library written over itself: same bytes, same length, new mtime. -/// **This is the arm a partial fix fails** — the one above also changes the -/// size, so an identity keyed on size alone passes it. No symbol can tell these -/// two writes apart, so the refusal itself is the whole verdict. -fn a_same_size_rewrite_is_refused() -> Result { - copy(FIRST, SAME_SIZE); - let first = load_in_child(SAME_SIZE); - if !first.contains("LOADED") { - return Err(format!("the first load of {SAME_SIZE} did not happen: {first}")); - } - - copy(FIRST, SAME_SIZE); - let second = load_in_child(SAME_SIZE); - if !second.contains("REFUSED NotSupported") { - return Err(format!( - "{SAME_SIZE} was rewritten with the same bytes at the same length and the load was \ - not refused: {second} — an identity keyed on size alone cannot see this write" - )); - } - Ok(format!("a same-size rewrite of {SAME_SIZE} was refused")) -} - -/// Distinct paths are distinct entries, so copies of one library fill the -/// budget as surely as distinct libraries would. -fn the_budget_is_refused() -> Result { - for attempt in 0..BUDGET_ATTEMPTS { - let path = format!("/home/so-cache-fill-{attempt}.so"); - copy(FIRST, &path); - let said = load_in_child(&path); - if said.contains("REFUSED ResourceExhausted") { - return Ok(format!("refused at copy {attempt} of {BUDGET_ATTEMPTS}")); - } - if !said.contains("LOADED") { - return Err(format!("copy {attempt} neither loaded nor refused: {said}")); - } - } - Err(format!( - "{BUDGET_ATTEMPTS} distinct 2 MiB images entered the cache unrefused — this kernel \ - holds no byte budget over it at all" - )) -} - -fn copy(from: &str, to: &str) { - let bytes = std::fs::read(from).unwrap_or_else(|e| panic!("read {from}: {e}")); - std::fs::write(to, &bytes).unwrap_or_else(|e| panic!("write {to}: {e}")); -} - -fn load_in_child(path: &str) -> String { - let out = Command::new(SELF_PATH) - .arg(path) - .stdout(Stdio::piped()) - .spawn() - .unwrap_or_else(|e| panic!("spawn a loader for {path}: {e}")) - .wait_with_output() - .unwrap_or_else(|e| panic!("wait for the loader of {path}: {e}")); - String::from_utf8_lossy(&out.stdout).trim().to_string() -} - -fn load_in_this_process(path: &str) { - match syscall::dl_open(path.as_bytes()) { - Ok(handle) => { - println!("LOADED"); - if unsafe { syscall::dl_sym(handle, ONLY_IN_FIRST) }.is_ok() { - println!("SYMBOL-FOUND"); - } - } - Err(SyscallError::NotSupported) => println!("REFUSED NotSupported"), - Err(SyscallError::ResourceExhausted) => println!("REFUSED ResourceExhausted"), - Err(other) => println!("REFUSED {other:?}"), - } -} - -fn load_and_report(path: &str) -> ! { - load_in_this_process(path); - std::process::exit(0) -} diff --git a/tests/toyos.rs b/tests/toyos.rs index f2ed0e90973..4bc5eecf66e 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -19,7 +19,7 @@ use toyos_build::bootlog::{self, boot_millis}; use toyos_build::heartbeat; use toyos_build::testargs::{self, Shard, SUITE}; use toyos_build::redlist; -use toyos_build::tiers::Tier; +use toyos_build::tiers::{Reach, Schedule, Tier}; struct TestDef { name: String, @@ -208,10 +208,6 @@ const RUST_SKIP: &[&str] = &[ // that measured anything. It also needs the real-time band, which only // `tests/latencycase` endows. `latency_wake` runs it there. "cyclictest", - // Its verdict is a property of the *console capture*, which only a boot of - // its own can hold: in the shared boot every other binary's output is in the - // same stream. `console_line_atomicity` runs it. - "console_line_atomicity", // **It reboots the machine**, so in the shared block it would end the boot // under whichever member came next; and its verdict is the order of the // console after that reset, which only its own boot holds. @@ -221,7 +217,7 @@ const RUST_SKIP: &[&str] = &[ // `quiesce_refuses_a_second_shutdown` runs it. "quiesce_twice", // The same, and its verdict is the stop record of a boot staged around it. - // `quiesce_wakes_on_the_last_park` and `quiesce_wakes_on_the_last_exit` run it. + // `quiesce_wakes_on_the_last_park` runs it. "quiesce_last", // The same, and its verdict is the log volume the stop leaves. // `quiesce_leaves_the_volume_whole` runs it. @@ -374,10 +370,6 @@ const RUST_SKIP: &[&str] = &[ "inventory_bounds", // Same reason, same config: `netd_hostile_peer` runs it there. "netd_hostile_peer", - // Needs a `launcher` connector, which `tests/testcases`'s test-runner has - // no reason to hold. `launcher_refusals` runs it on tests/netcase, whose - // test-runner receives one for exactly this. - "launcher_refusals", // Needs a launcher to tell its two roads apart, and a declared shell and // toybox to take it: `spawn_cwd` runs it on tests/netcase. "spawn_cwd", @@ -430,16 +422,6 @@ const RUST_SKIP: &[&str] = &[ // shared boot it would be reported against whichever test came next — and // every one after that. `short_sleep_livelock` gives it a boot of its own. "abuse_short_sleep", - // The four below were **running twice under one name**, once here on the - // plain boot and once as the test that owns the name, and the collision was - // invisible: `check_registration` compared the three declared lists against - // each other and never against the binaries the registry discovers. - // `check_no_collisions` closes that, and this is what it found. - // - // What the shared copy adds is the binary exiting 0 on a boot that gives it - // nothing to measure — `cache_eviction` in 132 ms against the 22.5 s its - // own device shape costs (run `31247206462`). - // // `cache_eviction` needs the small NVMe that makes the cache evict at all. "cache_eviction", // `writeback_reopen` and `writeback_spawn` each need their own boot with @@ -449,12 +431,6 @@ const RUST_SKIP: &[&str] = &[ "writeback_reopen", "writeback_spawn", "writeback_durability", - // Same shape as `writeback_durability`: what it stages on `/log` — a file - // unlinked out from under a held descriptor, its clusters handed to the next - // writer — is only half the claim, and the other half is the volume read - // back off the image after a shutdown by a FAT implementation that is not - // the kernel's. `fat_backing_revoked` runs it. - "fat_backing_revoked", // Stages a rename with an absent source on `/log` and leaves the // destination for `fs_rename_durable` to read back off the image. "fs_rename_durable", @@ -477,11 +453,6 @@ const RUST_SKIP: &[&str] = &[ // succeeds and its two must-refuse assertions red for an honest reason. // `fsync_failed_commit` boots it with the arm. "fsync_flush_failed", - // Needs `fsync-budget-spent` and the NVMe `/home`; unstaged it passes - // vacuously. `home_budget_refusal_retried` boots it with both. - "home_fsync_budget", - // Needs `so-cache-tiny` and the NVMe `/home`. `so_cache_refusals` gives both. - "so_cache_policy", // Needs the NVMe `/home` and a boot of its own for the readback it is judged against; `home_overwrite_reads_back` runs it. "home_overwrite_zero", // Needs a boot where the DATA volume is ours and absent; on the shared @@ -572,12 +543,12 @@ const EARLY_PANIC_MESSAGE: &str = "test-early-panic: on-screen console check"; /// kept because these are read the way they are /// written. const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ - ("screen_decoder", Sched::Parallel, Tier::Fast), + ("screen_decoder", Sched::Parallel, Tier::Weekly), // Two boots, each ended at a loader line rather than at the kernel's ready // marker. Every verdict is a count of rows against a count of lines off the // same boot's console; no clock is in either. - ("screen_loader_lines", Sched::Parallel, Tier::Fast), - ("screen_gop_firmware_mode", Sched::Parallel, Tier::Nightly), + ("screen_loader_lines", Sched::Parallel, Tier::Nightly), + ("screen_gop_firmware_mode", Sched::Parallel, Tier::Weekly), // `thread::sleep(5 s)` is the measurement, not a ceiling: the assertion is // literally that the log is still on the panel five seconds after the boot // finished, so a 2x slower machine changes nothing about the wait but the @@ -585,33 +556,32 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ ("screen_diag_boot", Sched::Parallel, Tier::Nightly), // A guest halted in the window, so the panel is read where only the repaint // under test can have painted it. - ("screen_early_panel", Sched::Parallel, Tier::Fast), + ("screen_early_panel", Sched::Parallel, Tier::Nightly), ("screen_log_absent", Sched::Parallel, Tier::Fast), ("screen_console_shell", Sched::Parallel, Tier::Fast), ("screen_console_clear", Sched::Parallel, Tier::Fast), - ("screen_console_scroll", Sched::Parallel, Tier::Nightly), - ("screen_i8042_health", Sched::Parallel, Tier::Fast), + ("screen_console_scroll", Sched::Parallel, Tier::Fast), + ("screen_i8042_health", Sched::Parallel, Tier::Weekly), // Ctrl+Alt+D with no console at all: the panel is the whole channel, and a // compositor is holding it. A fixed 2 s settle sits inside the dump's own // guest-timed 15 s hold, and the verdict is whether the report survived // the desktop's next repaint — which only where that wait lands decides, // so it is timer-anchored despite being a screendump-content check. ("screen_blocked_dump", Sched::Parallel, Tier::Nightly), - ("screen_recoverable_untouched", Sched::Parallel, Tier::Fast), + ("screen_recoverable_untouched", Sched::Parallel, Tier::Weekly), // The other half of the recovery branch: the test above reads the screen // either side of a survived panic, which holds whether or not the discard // did anything. - ("screen_survived_panic_not_blamed", Sched::Parallel, Tier::Nightly), - ("screen_early_panic", Sched::Parallel, Tier::Fast), + ("screen_survived_panic_not_blamed", Sched::Parallel, Tier::Weekly), ("screen_late_panic", Sched::Parallel, Tier::Fast), - ("screen_paged_scrollback", Sched::Parallel, Tier::Nightly), - ("screen_panic_muted", Sched::Parallel, Tier::Fast), + ("screen_paged_scrollback", Sched::Parallel, Tier::Weekly), + ("screen_panic_muted", Sched::Parallel, Tier::Weekly), ("screen_console_panic", Sched::Parallel, Tier::Fast), ("screen_fatal_halt", Sched::Parallel, Tier::Fast), // The same fatal path from inside Ctrl+Alt+D's report painter, holding the // panel's latch it will never give back: the report has to take the screen // anyway, and its CPU has to go on to watch the reset bound. - ("screen_fatal_behind_a_painter", Sched::Parallel, Tier::Fast), + ("screen_fatal_behind_a_painter", Sched::Parallel, Tier::Nightly), // The same fatal path with a compositor holding the panel, which is the // only configuration the owner's laptop is ever in and the one no screen // test covered: `screen_fatal_halt` boots a config with no compositor, and @@ -675,7 +645,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // and the first number against a number. Parallel: every verdict is // arithmetic over counters the guest printed, and there is no clock in any // of it. - ("irq_census_conservation", Sched::Parallel, Tier::Fast), + ("irq_census_conservation", Sched::Parallel, Tier::Weekly), ("control_regs", Sched::Parallel, Tier::Fast), ("control_regs_negative", Sched::Parallel, Tier::Fast), // The boot facts the metal suite reads off a machine's own records: every @@ -687,24 +657,24 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // matches its rows; and one machine-wide TLB shootdown's cost is a // distribution rather than one boot's average. Every verdict is arithmetic // over records, with no clock in the judging, so all six are Parallel. - ("smp_roster_and_tsc_trail", Sched::Parallel, Tier::Fast), - ("pmm_accounting", Sched::Parallel, Tier::Fast), + ("smp_roster_and_tsc_trail", Sched::Parallel, Tier::Nightly), + ("pmm_accounting", Sched::Parallel, Tier::Nightly), // ROOT is the loader's image in memory: the kernel says it mounted it // from memory, and that init was spawned with no storage command issued; // and a loader that hands no image is a boot refused by name, never one // that goes to a disk for ROOT. Both are records of one boot each, with no // clock in the verdict. - ("root_from_memory", Sched::Parallel, Tier::Fast), - ("root_withheld_refused", Sched::Parallel, Tier::Fast), + ("root_from_memory", Sched::Parallel, Tier::Nightly), + ("root_withheld_refused", Sched::Parallel, Tier::Nightly), // The boot from power-on, as the kernel converts the loader's TSC readings: // judged against the loader's raw counts and the kernel's own rate, and // bounded above by the host's clock, which is Parallel-safe because load // only widens that bound. - ("boot_from_power_on", Sched::Parallel, Tier::Fast), - ("acpi_table_inventory", Sched::Parallel, Tier::Fast), - ("timer_calibration", Sched::Parallel, Tier::Fast), - ("pci_inventory", Sched::Parallel, Tier::Fast), - ("tlb_shootdown_cost", Sched::Parallel, Tier::Fast), + ("boot_from_power_on", Sched::Parallel, Tier::Nightly), + ("acpi_table_inventory", Sched::Parallel, Tier::Nightly), + ("timer_calibration", Sched::Parallel, Tier::Nightly), + ("pci_inventory", Sched::Parallel, Tier::Nightly), + ("tlb_shootdown_cost", Sched::Parallel, Tier::Nightly), // What a waiter in the real-time band pays to be woken, as a distribution // over ten thousand programmed wakes — and, beside it, that the number // reaches a machine with no serial port at all, through the kernel's own @@ -718,13 +688,13 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // wake latency measured beside eleven other guests is the host's schedule. // Nightly for that same reason. ("latency_wake", Sched::Serial, Tier::Nightly), - ("smp_failed_ap_leaves_no_hole", Sched::Parallel, Tier::Fast), - ("input_merge", Sched::Parallel, Tier::Fast), - ("metal_sim_input", Sched::Parallel, Tier::Fast), - ("input_claim_absent", Sched::Parallel, Tier::Fast), + ("smp_failed_ap_leaves_no_hole", Sched::Parallel, Tier::Weekly), + ("input_merge", Sched::Parallel, Tier::Weekly), + ("metal_sim_input", Sched::Parallel, Tier::Weekly), + ("input_claim_absent", Sched::Parallel, Tier::Weekly), // One boot; every verdict is a PPM header field or a console line, and no - // clock is in any of them, so it is Nightly for its price and nothing else. - ("gpu_set_resolution", Sched::Parallel, Tier::Nightly), + // clock is in any of them. + ("gpu_set_resolution", Sched::Parallel, Tier::Fast), // One boot from here to `metal_sim_compositor_stall` (`METAL_SIM_DESKTOP`). ("metal_sim_compositor", Sched::Parallel, Tier::Nightly), // Reads the boot log this group already has, after the member above has @@ -751,7 +721,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // verdict rather than a slower one. Watched to happen, on a compositor made // slow on purpose. Its own boot too: it leaves the pointer somewhere else // and the window in a different place than it found them. - ("metal_sim_window_drag", Sched::Serial, Tier::Nightly), + ("metal_sim_window_drag", Sched::Serial, Tier::Weekly), // A host-measured drain rate with an 8 s ceiling on a 3.3 s expectation. // Not gate A, but the same instrument: what it measures is how fast a // client's audio leaves the machine. @@ -776,7 +746,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // ureq and rustls from crates.io, fetching over TLS 1.3 from a host this // test mints a CA for. Every verdict is a printed line or a digest; the // only clock is `run_test`'s ceiling. - ("https_tls13", Sched::Parallel, Tier::Fast), + ("https_tls13", Sched::Parallel, Tier::Weekly), // The same judge over netd's Intel driver instead of the virtio one: the // 82574L QEMU models has the register file the T14's I219 has, so this is // where that driver moves real frames before the laptop does. @@ -792,55 +762,54 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // only machine in reach that runs that driver. ("log_stream_e1000e", Sched::Parallel, Tier::Nightly), // A reader that never reads, beside a flood: the file and a second reader - // are whole regardless. Nightly for the flood's megabytes. + // are whole regardless. ("log_stream_stalled_reader", Sched::Parallel, Tier::Nightly), // A program's line in `/log`, on the served log and on the console, under // the name of the pipe it came out of. Lines and a comparison; no clock. - ("log_program_line", Sched::Parallel, Tier::Fast), + ("log_program_line", Sched::Parallel, Tier::Nightly), // A program writing the kernel's words and another program's head: every // judge of `/log` reads the truth. Lines, and the exit judge; no clock. - ("log_program_forgery", Sched::Parallel, Tier::Fast), + ("log_program_forgery", Sched::Parallel, Tier::Nightly), // A stop the kernel refuses after logd flushed for it: a line said after // it is in `/log`. Lines; no clock. - ("log_after_a_refused_stop", Sched::Parallel, Tier::Fast), + ("log_after_a_refused_stop", Sched::Parallel, Tier::Nightly), // The same stop with logd's flush held until init resumes it: logd runs // the flush, then the resume, and lives. Lines; init's flush bound. - ("log_resume_meets_its_flush", Sched::Parallel, Tier::Fast), + ("log_resume_meets_its_flush", Sched::Parallel, Tier::Nightly), // A child flooding its parent's log ring while logd reads none of it: the // parent's next line is in `/log`. Lines; its clock is a guard. - ("log_ring_keeps_the_owners_slots", Sched::Parallel, Tier::Fast), + ("log_ring_keeps_the_owners_slots", Sched::Parallel, Tier::Nightly), // A program's line said after three batches of records, read before them: // `/log` carries it after every one. Lines and positions; no clock. - ("log_program_line_after_its_records", Sched::Parallel, Tier::Fast), + ("log_program_line_after_its_records", Sched::Parallel, Tier::Nightly), // A program printing init's word accepting a swap of netd: logd turns // nobody away, and a reader after it is admitted. Lines; no clock. - ("log_carrier_forgery", Sched::Parallel, Tier::Fast), + ("log_carrier_forgery", Sched::Parallel, Tier::Nightly), // A flood many times its ring: every line in `/log` in order or counted. - // Nightly for its megabytes through a TCG guest's volume. ("log_program_flood", Sched::Parallel, Tier::Nightly), // netd taking this machine's address from the network instead of carrying // one written down. The DHCP server it is judged against is QEMU's own, an // implementation of RFC 2131 this repository did not write, and its lease // is known field by field. The verdicts are records and a lease's fields; // no clock in it. - ("lan_dhcp_lease", Sched::Parallel, Tier::Fast), + ("lan_dhcp_lease", Sched::Parallel, Tier::Nightly), // The lease probe on the same part: netd serves its window, exits with the // lease's verdict, and the report it leaves on the log volume names the // backend's lease with frames counted both ways and the lease kept across // a link flap. Records and a file; its clocks are the flap's hold and the // drain that outlasts netd's window, and a slower machine moves neither. - ("lan_lease_report", Sched::Parallel, Tier::Fast), + ("lan_lease_report", Sched::Parallel, Tier::Weekly), // The T14's talking boot rehearsed on the same part: the log the guest // serves, read from its first line through a forward, one command over ssh // answered byte for byte, and `reboot` over ssh ending the guest. Lines, // bytes and a reset; its clocks are liveness guards on a guest that // stopped talking. - ("lan_talk", Sched::Parallel, Tier::Fast), + ("lan_talk", Sched::Parallel, Tier::Nightly), // netd answering for its name: a resolver's query through a forward onto // the guest's multicast DNS port is answered with the lease's address, and // one for another name is not answered. Bytes; its clocks are a guard on // the answer and the window the unanswered query is given. - ("lan_mdns_answer", Sched::Parallel, Tier::Fast), + ("lan_mdns_answer", Sched::Parallel, Tier::Nightly), // A running service's binary replaced with no reboot: netd swapped for its // rebuild over ssh while the stream runs, on virtio-net and on the 82574 // the T14's I219 shares a register file with; a wrong digest and a @@ -848,79 +817,78 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // once answered by the old binary running again. Records, bytes and the // guest's own `/log`; every clock is a liveness guard on a guest that // stopped talking, and probation is init's. - ("swap_netd", Sched::Parallel, Tier::Fast), + ("swap_netd", Sched::Parallel, Tier::Nightly), // The machine updates itself: an image over `ssh … update` is written to // the idle slot and is the kernel the next boot runs; every slot the loader // must refuse is refused by name and the other boots; and a slot whose // kernel dies falls back on its own, its death in the next boot's `/log`. // The same machine's floor, grant and hang: a floor is its key's and its // image's, init grants nothing the slot table names but an idle slot's - // partition, and a hang of an unproven image is a death. Each is 25 s or - // more of boots, past the fast tier's line, so they are nightly. - ("update_boots_the_new_kernel", Sched::Parallel, Tier::Nightly), - ("update_refusals_boot_the_other_slot", Sched::Parallel, Tier::Nightly), - ("update_falls_back_from_a_dying_kernel", Sched::Parallel, Tier::Nightly), + // partition, and a hang of an unproven image is a death. + ("update_boots_the_new_kernel", Sched::Parallel, Tier::Weekly), + ("update_refusals_boot_the_other_slot", Sched::Parallel, Tier::Weekly), + ("update_falls_back_from_a_dying_kernel", Sched::Parallel, Tier::Weekly), ("update_hang_kills_an_unproven_image", Sched::Parallel, Tier::Nightly), ("update_grant_refuses_a_stray_partition", Sched::Parallel, Tier::Nightly), - ("update_floor_is_the_images_own", Sched::Parallel, Tier::Nightly), + ("update_floor_is_the_images_own", Sched::Parallel, Tier::Weekly), ("update_refused_pass_credits_no_image", Sched::Parallel, Tier::Nightly), - ("lan_swap", Sched::Parallel, Tier::Fast), - ("swap_refusals", Sched::Parallel, Tier::Fast), - ("swap_crash_rolls_back", Sched::Parallel, Tier::Fast), + ("lan_swap", Sched::Parallel, Tier::Nightly), + ("swap_refusals", Sched::Parallel, Tier::Nightly), + ("swap_crash_rolls_back", Sched::Parallel, Tier::Nightly), // The 82574 swapped to a holder that stops it and masters it while the // host sends it frames. The verdict is the kernel's console; its clocks // are liveness guards. - ("swap_quiets_the_function", Sched::Parallel, Tier::Fast), + ("swap_quiets_the_function", Sched::Parallel, Tier::Nightly), // The same part released as though nothing could reset it, the way the // T14's I219 is, and swapped to a holder that masters it with its receive // unit still on. The verdict is the kernel's console; its clocks are // liveness guards. - ("swap_keeps_what_nothing_reset", Sched::Parallel, Tier::Fast), + ("swap_keeps_what_nothing_reset", Sched::Parallel, Tier::Nightly), // The same part swapped to a holder that aims it outside its grant and // waits on its claim. The verdict is the holder's own word on what the // faulted claim answered; its clocks are liveness guards. - ("swap_fault_tells_its_holder", Sched::Parallel, Tier::Fast), + ("swap_fault_tells_its_holder", Sched::Parallel, Tier::Nightly), // An `igb` netd held, released by an Express function level reset and // claimed again by netd's replacement, which reads it through the window // its claim maps. The verdict is what the function answers there. - ("swap_resets_the_function", Sched::Parallel, Tier::Fast), + ("swap_resets_the_function", Sched::Parallel, Tier::Nightly), // The same `igb`, its window left where its reset put it: the replacement // is refused it, and the verdict is init failing the swap by the // device's name rather than putting netd in service without it. ("swap_refused_device_fails", Sched::Parallel, Tier::Fast), // The same, its window moved inside the cut rather than lost: the kernel // reads the address the register holds, not only that it holds one. - ("swap_moved_device_fails", Sched::Parallel, Tier::Fast), + ("swap_moved_device_fails", Sched::Parallel, Tier::Nightly), // A program sshd runs undeclared asks for the swap port. The verdict is // the program's own exit: the port is not in the namespace it inherited. - ("swap_not_inherited", Sched::Parallel, Tier::Fast), + ("swap_not_inherited", Sched::Parallel, Tier::Nightly), // The same client on a wire with no server: it says it has no address and // announces itself anyway. Its verdict waits out netd's own lease bound, so // a slower machine moves it. - ("lan_no_lease", Sched::Parallel, Tier::Nightly), - ("netd_connection_caps", Sched::Parallel, Tier::Fast), + ("lan_no_lease", Sched::Parallel, Tier::Weekly), + ("netd_connection_caps", Sched::Parallel, Tier::Nightly), // `inspect` against the four owners on the one boot that runs them all, // with the negative control's binary run on it too. Every verdict is the // set of lines a selector printed; no clock in any. - ("inspect_reads_its_owners", Sched::Parallel, Tier::Fast), + ("inspect_reads_its_owners", Sched::Parallel, Tier::Nightly), // The netcase boot again: netd must not abort a listener on a ring flag its // own client forged. Its verdict is a kernel-reported EOF or its absence; // no clock in it. - ("netd_listener_forgery", Sched::Parallel, Tier::Fast), + ("netd_listener_forgery", Sched::Parallel, Tier::Weekly), // The netcase boot again: a receiver that stops reading until its pipe // is full still gets every byte of a stream past it. The verdict is the // guest's byte-for-byte comparison; its clocks are liveness guards. - ("netd_slow_reader", Sched::Parallel, Tier::Fast), + ("netd_slow_reader", Sched::Parallel, Tier::Nightly), // The netcase boot again: client pipes netd cannot use, or loses under // it, cost that client its connection and never netd. The verdict is a // round trip after each case, a named line per refusal and a clean // console; its clocks are liveness guards. - ("netd_refused_pipes", Sched::Parallel, Tier::Fast), + ("netd_refused_pipes", Sched::Parallel, Tier::Nightly), // The netcase boot again: bytes held back past a full pipe move on the // pipe's room alone, the peer holding the connection open and silent. The // verdict is the guest's byte-for-byte comparison; its clocks are // liveness guards. - ("netd_held_open", Sched::Parallel, Tier::Fast), + ("netd_held_open", Sched::Parallel, Tier::Nightly), // The netcase boot again: a client out-writing a peer that stopped reading // costs netd no CPU. Nightly: its verdict is the machine's busy time over // a window of real time. @@ -928,66 +896,61 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // The netcase boot again, beside a host UDP echo: a datagram the client's // pipe will not take whole ends that socket by name and no other. The // verdict is each socket's answer; its clocks are liveness guards. - ("netd_udp_refused", Sched::Parallel, Tier::Fast), + ("netd_udp_refused", Sched::Parallel, Tier::Nightly), // The netcase boot again, beside a host UDP echo: a socket bound through // std to 0.0.0.0 receives the echo's unicast reply. The verdict is the // reply's bytes; its clock is a liveness guard. - ("netd_udp_any_address", Sched::Parallel, Tier::Fast), + ("netd_udp_any_address", Sched::Parallel, Tier::Nightly), // The netcase boot, whose user network forwards 10.0.2.3 to this host's // resolver: `host` resolves a real name to the addresses this host's // resolver gives it, and a `.invalid` name to none. The verdict is the - // two answers; its clocks are liveness guards. Nightly, because both - // answers rest on this host's network, which no change to the tree moves. + // two answers; its clocks are liveness guards. ("dns_resolve", Sched::Parallel, Tier::Nightly), // The netcase boot, every frame it sends held once it has its lease: // lookups whose clients hung up or spoke again are let go at once, and // one nobody answers ends when its schedule does. The verdict is netd's // answers; the schedule's end is a bound derived from it. - ("netd_lookup_let_go", Sched::Parallel, Tier::Fast), + ("netd_lookup_let_go", Sched::Parallel, Tier::Nightly), // Two netcase boots, each frame put on the wire kept: the first DHCP // transaction ID of each differs, because netd seeds smoltcp's random // source from the kernel's. The verdict is two numbers off the wire. - ("netd_seeds_its_stack", Sched::Parallel, Tier::Fast), + ("netd_seeds_its_stack", Sched::Parallel, Tier::Nightly), // The netcase boot with two programs naming one PCI function: the verdict // is which of them the kernel let have it. Console lines only, no clock. - ("pci_function_is_exclusive", Sched::Parallel, Tier::Fast), + ("pci_function_is_exclusive", Sched::Parallel, Tier::Nightly), // The same boot again, read for what the kernel asked the machine before it // moved that function's BAR. Its own boot rather than a second assertion in // the row above, because that one's subject is exclusivity and a test that // reds tells a reader which of the two it is about. It waits out a drain for // the message record, so its price carries a fixed span of host wall clock. - ("bar_placement_is_proven", Sched::Parallel, Tier::Fast), - // Its own boot with a NIC under it, because sshd leaves at the bind on - // every other config. Every verdict is a line of text; no clock in any. - ("sshd_fail_closed", Sched::Parallel, Tier::Fast), + ("bar_placement_is_proven", Sched::Parallel, Tier::Nightly), // Its own boot with a blank DATA volume, asked over ssh where everything // it wrote went. Every verdict is a listing or a line; no clock in any. - ("layout_fresh_boot", Sched::Parallel, Tier::Fast), + ("layout_fresh_boot", Sched::Parallel, Tier::Nightly), // One `SSHD_LOGIN` boot for the three, driven by `tests/ssh-client-host`. // Adjacent because `group_of` makes adjacency load-bearing, and one tier // because one boot cannot be in two. Every verdict is bytes or an exit // status; the client's own ceiling is a liveness guard and no assertion // reads a clock. - ("sshd_exec", Sched::Parallel, Tier::Fast), - ("sshd_files", Sched::Parallel, Tier::Fast), - ("sshd_key_auth", Sched::Parallel, Tier::Fast), + ("sshd_exec", Sched::Parallel, Tier::Nightly), + ("sshd_files", Sched::Parallel, Tier::Nightly), + ("sshd_key_auth", Sched::Parallel, Tier::Nightly), // Serial: it measures netd's 2 s handshake deadline against the host's // clock, and counts how many connections survived a 48 ms paced burst // before that deadline could expire any of them. Both are wall-clock // margins, which is the definition of [`Sched::Serial`]. - ("netd_hostile_peer", Sched::Serial, Tier::Nightly), - ("launcher_refusals", Sched::Parallel, Tier::Fast), + ("netd_hostile_peer", Sched::Serial, Tier::Weekly), // Paths a child prints and the kernel's refusals by name; no clock in any of them. - ("spawn_cwd", Sched::Parallel, Tier::Fast), - ("foreign_disk_untouched", Sched::Parallel, Tier::Fast), - ("volume_from_another_disk", Sched::Parallel, Tier::Fast), - ("broken_data_volume_is_absent", Sched::Parallel, Tier::Fast), - ("data_candidate_with_bad_geometry_is_absent", Sched::Parallel, Tier::Fast), + ("spawn_cwd", Sched::Parallel, Tier::Nightly), + ("foreign_disk_untouched", Sched::Parallel, Tier::Weekly), + ("volume_from_another_disk", Sched::Parallel, Tier::Weekly), + ("broken_data_volume_is_absent", Sched::Parallel, Tier::Nightly), + ("data_candidate_with_bad_geometry_is_absent", Sched::Parallel, Tier::Nightly), // Four kernel lines and a file read off the image once the guest is gone; no clock in any of them. - ("internal_disk_boot", Sched::Parallel, Tier::Fast), + ("internal_disk_boot", Sched::Parallel, Tier::Weekly), // One boot each, kernel lines and image bytes for verdicts, no clock in either. - ("block_duplicate_id", Sched::Parallel, Tier::Fast), - ("page_cache_partition_offset", Sched::Parallel, Tier::Fast), + ("block_duplicate_id", Sched::Parallel, Tier::Weekly), + ("page_cache_partition_offset", Sched::Parallel, Tier::Weekly), // A partition claimed as a device: one boot, every refusal in the guest, // the neighbours and the target judged off the image. Body in // `tests/common/partclaim.rs`, as are the two below. @@ -995,83 +958,72 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // Three boots: a disk that does not answer a read of its table, every // attempt refused until the deadman, and ROOT's source withheld from every // claim once its disk did not answer the boot's hold. - ("partition_claim_gives_up", Sched::Parallel, Tier::Fast), + ("partition_claim_gives_up", Sched::Parallel, Tier::Nightly), // Three boots, a USB stick's device leaving owing one claim's write and // coming back on another port each time: each partition's fsync answers // for its own writes, across a close, after another's flush, and at the // shutdown when nobody asked. - ("partition_claim_departure", Sched::Parallel, Tier::Fast), - // F9's negative control: a budget-refused /home fsync retried to durable, - // its bytes then read off the NVMe image by the host's own bcachefs - // reader. Body in `tests/common/storage.rs`. - ("home_budget_refusal_retried", Sched::Parallel, Tier::Nightly), - // The shared-object cache's two refusals. Body in `tests/common/storage.rs`. - ("so_cache_refusals", Sched::Parallel, Tier::Fast), + ("partition_claim_departure", Sched::Parallel, Tier::Nightly), // A same-length overwrite on /home, the guest's read held against the image. Body in `tests/common/storage.rs`. - ("home_overwrite_reads_back", Sched::Parallel, Tier::Fast), + ("home_overwrite_reads_back", Sched::Parallel, Tier::Weekly), // One filesystem under two paths: the guest writes under each of /apps and // /home, the host finds both in one volume on the image. Body in // `tests/common/storage.rs`. - ("apps_and_home_are_one_filesystem", Sched::Parallel, Tier::Fast), + ("apps_and_home_are_one_filesystem", Sched::Parallel, Tier::Weekly), // `pkg install ` from a local archive and gbae's first run: the whole // package path in one boot, judged off the DATA volume once the guest is // gone. Body in `tests/common/pkg.rs`. - ("pkg_install_gbae", Sched::Parallel, Tier::Fast), - ("boot_partition_identity", Sched::Parallel, Tier::Fast), + ("pkg_install_gbae", Sched::Parallel, Tier::Nightly), + ("boot_partition_identity", Sched::Parallel, Tier::Nightly), // One boot of its own, because it ends the machine. Every verdict is a // kernel line or the stop reason QEMU reported; no clock is in either. - ("machine_reboot", Sched::Parallel, Tier::Fast), + ("machine_reboot", Sched::Parallel, Tier::Weekly), // Its own boot: every verdict is a console line, QEMU's stop reason or a record off the image. ("metal_job_reboot", Sched::Parallel, Tier::Fast), // Its own boot, and the one whose numbers are the T14's: what it judges // here is the plumbing, since every span on an emulated device is a fact // about TCG. - ("metal_device_probe", Sched::Parallel, Tier::Fast), + ("metal_device_probe", Sched::Parallel, Tier::Nightly), // Its verdict waits out a staged window. - ("job_deadline_reboots", Sched::Parallel, Tier::Fast), + ("job_deadline_reboots", Sched::Parallel, Tier::Nightly), // Its own boot, and its verdict waits out the same staged window. ("quiesce_stops_the_machine", Sched::Parallel, Tier::Fast), // Its own boot: it ends the machine, and its verdict is the order of // kernel lines. - ("quiesce_refuses_a_second_shutdown", Sched::Parallel, Tier::Fast), + ("quiesce_refuses_a_second_shutdown", Sched::Parallel, Tier::Nightly), // Its own boot: it ends the machine, and its verdict is the volume that // boot leaves. - ("quiesce_leaves_the_volume_whole", Sched::Parallel, Tier::Fast), - // Its own boot each: it ends the machine, and its verdict is the stop - // record that boot writes. - ("quiesce_wakes_on_the_last_park", Sched::Parallel, Tier::Fast), - ("quiesce_wakes_on_the_last_exit", Sched::Parallel, Tier::Fast), - // Its own boot: it ends the machine, and its verdict is a dump served - // inside that boot's stop. - ("quiesce_dump_holds_the_stopped", Sched::Parallel, Tier::Fast), + ("quiesce_leaves_the_volume_whole", Sched::Parallel, Tier::Nightly), + // Its own boot: it ends the machine, and its verdict is the stop record + // that boot writes. + ("quiesce_wakes_on_the_last_park", Sched::Parallel, Tier::Nightly), // Two reads of `TCO_RLD` straddling a real-time stall, so a slower machine // changes the verdict. ("loader_watchdog_arms", Sched::Parallel, Tier::Nightly), // Its own boot, and the verdict is QEMU's stop reason inside the bound. ("watchdog_resets", Sched::Parallel, Tier::Nightly), // Serial: its verdict is that nothing happened for a span of host clock. - ("watchdog_fed", Sched::Serial, Tier::Nightly), + ("watchdog_fed", Sched::Serial, Tier::Weekly), // The panicked kernel's own bound, which is what ends a boot on a machine // whose chipset timer does not count. Both verdicts are QEMU's stop reason // against a bound the guest printed, so a slower machine moves both. ("panic_reboots", Sched::Parallel, Tier::Nightly), // The same verdict from inside `percpu::init_bsp`: the earliest point a // panic is reportable, and the window the owner's T14 stops in. - ("panic_before_peripherals_reboots", Sched::Parallel, Tier::Fast), + ("panic_before_peripherals_reboots", Sched::Parallel, Tier::Nightly), // Serial like `watchdog_fed`: its verdict is that nothing happened for a span of host clock. - ("panic_key_holds", Sched::Serial, Tier::Nightly), + ("panic_key_holds", Sched::Serial, Tier::Weekly), // The boot chain's three answers. The two chain names each watch a guest // take its own reset and read the pass after it, so both are anchored to // the bound the first boot counts down. ("blackbox_panic_chain", Sched::Parallel, Tier::Nightly), - ("blackbox_done_chain", Sched::Parallel, Tier::Fast), + ("blackbox_done_chain", Sched::Parallel, Tier::Nightly), // The one bound in this tree that ends a machine nothing else can: a boot // whose every CPU has stopped taking scheduler passes. Its verdict is a - // bound counted down in the guest, so it is Nightly. + // bound counted down in the guest. ("boot_deadline_ends_a_wedge", Sched::Parallel, Tier::Nightly), - // The same bound ending the same machine with a device in its hands, and - // Nightly for the same reason. - ("usb_reset_records_the_phase_it_cut", Sched::Parallel, Tier::Nightly), + // The same bound ending the same machine with a device in its hands. + ("usb_reset_records_the_phase_it_cut", Sched::Parallel, Tier::Weekly), // The other half of that same parameter, and the state its poll cannot // reach: one CPU with interrupts off, which no running CPU can see. Two // bounds counted down in the guest, so it belongs beside the row above. @@ -1082,36 +1034,32 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("panic_outlives_the_deadline", Sched::Parallel, Tier::Nightly), // Four chained boots, one per way this kernel reaches a reset, each // anchored to the bound its own first boot counts down. - ("usb_reset_hands_devices_back", Sched::Parallel, Tier::Nightly), + ("usb_reset_hands_devices_back", Sched::Parallel, Tier::Weekly), // The control on the chain: a record another image left in the same memory // is cleared and its pass boots a kernel, where a real predecessor's ends // the chain. One boot, one actuator. ("blackbox_foreign_record", Sched::Parallel, Tier::Fast), // Three launches of one image file, and the third is the one that makes it // a bound: a hang costs the machine one boot and never traps it. - ("hang_bounded_by_the_stick", Sched::Parallel, Tier::Fast), + ("hang_bounded_by_the_stick", Sched::Parallel, Tier::Nightly), // Its own boot, and every verdict is a line: no host clock in any of it. - ("blackbox_unclaimed_page", Sched::Parallel, Tier::Fast), + ("blackbox_unclaimed_page", Sched::Parallel, Tier::Nightly), // The seal read off the page's own bytes by QEMU, after a panic earlier than // anything the kernel used to learn the page's address from. Its own boot, // and no clock in the verdict. - ("blackbox_early_panic_sealed", Sched::Parallel, Tier::Fast), + ("blackbox_early_panic_sealed", Sched::Parallel, Tier::Nightly), // The same crash on the owner's machine's own shape — no serial port at all // — where the panel and the page are the only two channels there are. - ("blackbox_early_panic_sealed_muted", Sched::Parallel, Tier::Fast), + ("blackbox_early_panic_sealed_muted", Sched::Parallel, Tier::Nightly), // The exception entry's own seal, off the page's bytes on an ordinary boot. - ("blackbox_fault_sealed", Sched::Parallel, Tier::Fast), - ("double_fault_stack", Sched::Parallel, Tier::Fast), + ("blackbox_fault_sealed", Sched::Parallel, Tier::Nightly), + ("double_fault_stack", Sched::Parallel, Tier::Nightly), // One boot of its own, ten seconds of Ring 3 spinning, and every verdict is // a count the kernel printed or a line it printed: how many NMIs landed at // CPL 0 with a user `rsp`, against how many landed in Ring 3, both off the // same storm. No host clock is in any of it — the ten seconds are how long // the victim spins, not a margin anything is measured against — so Parallel. - // - // **It was three boots and priced at 19,740 ms** on the hosted lane (run - // 32580794553), twice the Fast ceiling. The two negative controls are the - // name below. - ("syscall_window_nmi", Sched::Parallel, Tier::Fast), + ("syscall_window_nmi", Sched::Parallel, Tier::Weekly), // The two controls on the name above: the kernel with vector 2's IST index // taken off, which must double fault at the entry on the NMI aimed at the // CPU the storm holds inside it, with `cr2 = rsp - 8` at the held `rsp`, and @@ -1119,10 +1067,10 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // path. Both boots end in a halted machine that has to be drained past its // own report, which is where the price is. Nothing in either verdict is a // duration. - ("syscall_window_nmi_controls", Sched::Parallel, Tier::Nightly), + ("syscall_window_nmi_controls", Sched::Parallel, Tier::Weekly), // Its own boot, its own feature, and it drives the guest only through // stdin — nothing it touches is shared with another test. - ("idle_stack_guard", Sched::Parallel, Tier::Nightly), + ("idle_stack_guard", Sched::Parallel, Tier::Weekly), // Its own boot and its own feature, and it deafens one CPU for 400 ms — // but the deafening is a *window*, and the verdict is whether the NMI is // answered inside `NMI_BUDGET_NS`, which is one millisecond. That is a @@ -1131,30 +1079,30 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // reads exactly like the defect it hunts, and it was green alone in the // same run and three times after it. Serial by the default rule — a // verdict that is a duration does not go in the parallel phase. - ("dump_nmi_probe", Sched::Serial, Tier::Nightly), + ("dump_nmi_probe", Sched::Serial, Tier::Weekly), // The same dump asked for inside the passes that may not serve it, on one // CPU. Parallel: every verdict is a line the guest prints or a count the guest // keeps, and no duration is in any of them. - ("dump_left_pending_is_owed", Sched::Parallel, Tier::Fast), - ("diskless_boot", Sched::Parallel, Tier::Fast), + ("dump_left_pending_is_owed", Sched::Parallel, Tier::Nightly), + ("diskless_boot", Sched::Parallel, Tier::Nightly), // Every verdict is a line of text or a device property, and no clock is in // any of them. ("virtio_net_no_msix", Sched::Parallel, Tier::Fast), // One boot of the NIC config with a staged capability list, and every // verdict a console line. No clock in any of them. - ("pci_claim_caps_truncated", Sched::Parallel, Tier::Fast), + ("pci_claim_caps_truncated", Sched::Parallel, Tier::Nightly), // One boot, and its verdict is a line the kernel printed before any device // was brought up. No clock and no device in it. - ("virtio_used_ring", Sched::Parallel, Tier::Fast), + ("virtio_used_ring", Sched::Parallel, Tier::Weekly), // A fatal path with other CPUs running userland that makes kernel records: // none is stamped past the fatal record by more than an IPI takes. ("panic_halts_the_others_first", Sched::Parallel, Tier::Fast), // A kernel log line from PCI enumeration; no clock and no real device in it. - ("pci_capability_walk", Sched::Parallel, Tier::Fast), + ("pci_capability_walk", Sched::Parallel, Tier::Weekly), // What QEMU was told to create against what the guest enumerated: two // accounts of one bus from two independent readers. One boot, every // verdict a set comparison. - ("query_pci_agreement", Sched::Parallel, Tier::Fast), + ("query_pci_agreement", Sched::Parallel, Tier::Weekly), // The root `system.toml`, booted rather than read: the shipped init list // and program namespaces have no other gate a harness run can reach. // Every verdict is a console line. @@ -1162,45 +1110,41 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // One boot whose verdict is three lines of kernel log and a census column. // The two waits inside the guest are bounded and report rather than hang, so // no host clock decides anything. - ("lapic_spurious_vector", Sched::Parallel, Tier::Fast), + ("lapic_spurious_vector", Sched::Parallel, Tier::Weekly), // One boot with both stuck-device actuators armed. - ("driver_wait_refused", Sched::Parallel, Tier::Nightly), + ("driver_wait_refused", Sched::Parallel, Tier::Weekly), // One boot; the leak-rollback controls' two verdict lines. - ("leak_rollback_selftest", Sched::Parallel, Tier::Fast), + ("leak_rollback_selftest", Sched::Parallel, Tier::Weekly), // One boot; the reopen control's one verdict line. - ("process_reopen_selftest", Sched::Parallel, Tier::Fast), + ("process_reopen_selftest", Sched::Parallel, Tier::Weekly), // One boot; three read-fault control verdicts. - ("read_fault_selftests", Sched::Parallel, Tier::Fast), - ("xhci_many_devices", Sched::Parallel, Tier::Fast), + ("read_fault_selftests", Sched::Parallel, Tier::Weekly), + ("xhci_many_devices", Sched::Parallel, Tier::Weekly), // Its whole assertion is that a keystroke injected from the host crossed a // USB keyboard on the *second* controller, and `input_events_run` sends // each one only after the guest has printed the last — so a key the host // never got to send is a stall it names, and never a key the driver lost. - ("xhci_second_controller", Sched::Parallel, Tier::Fast), - ("xhci_two_controllers", Sched::Parallel, Tier::Fast), + ("xhci_second_controller", Sched::Parallel, Tier::Weekly), + ("xhci_two_controllers", Sched::Parallel, Tier::Weekly), // **Returned 2026-08-17**, on the same `input_events_run` the two names // above it run: it had `xhci_second_controller`'s sequence written out again // on fixed sleeps, and nothing sent the right-button release // `test_rs_input_events` exits on — so 30 s of its 35.2 s CI price was a // client waiting out a fallback deadline with every assertion already // satisfied. - ("xhci_msi_only", Sched::Parallel, Tier::Fast), - ("xhci_no_interrupt", Sched::Parallel, Tier::Fast), - ("nvme_large_device", Sched::Parallel, Tier::Fast), - ("nvme_wide_sector", Sched::Parallel, Tier::Fast), - ("iommu_discovery", Sched::Parallel, Tier::Nightly), - ("readdir_bound", Sched::Parallel, Tier::Nightly), + ("xhci_msi_only", Sched::Parallel, Tier::Weekly), + ("xhci_no_interrupt", Sched::Parallel, Tier::Weekly), + ("nvme_large_device", Sched::Parallel, Tier::Nightly), + ("nvme_wide_sector", Sched::Parallel, Tier::Weekly), + ("iommu_discovery", Sched::Parallel, Tier::Weekly), + ("readdir_bound", Sched::Parallel, Tier::Weekly), // Its own boot: it fills the VFS `created_dirs` cap and leaves it there. - ("mkdir_cap", Sched::Parallel, Tier::Fast), + ("mkdir_cap", Sched::Parallel, Tier::Weekly), // Two boots, and the verdict is that they answer differently. Nothing in it - // is timed: every arm is a process exit code or a byte comparison — still - // compute-bound, still Nightly: 11,075 ms in the sweep's final shard - // packing (run 31705986758) is a Cost row, the same shape - // `desktop_window_child` carries, not a reclassification. + // is timed: every arm is a process exit code or a byte comparison. ("fpu_isolation", Sched::Parallel, Tier::Nightly), - // Two boots, exit codes only (no clock, Parallel). Nightly for `Cost`: its - // second (`user-writable-gsbase`) kernel double-boots it over the ceiling. - ("gsbase_locked", Sched::Parallel, Tier::Nightly), + // Two boots, exit codes only (no clock, Parallel). + ("gsbase_locked", Sched::Parallel, Tier::Weekly), // The fourth declared kernel build, booted so that the scheduler core's // `feature = "check"` instruments are compiled and executed by a CI run at // all. One of its verdicts is a *quantile* of the guest's published @@ -1225,112 +1169,75 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // construct one. Parallel, and nothing in it is a duration — every verdict // is a comparison between two numbers the kernel printed, both of them // offsets it chose itself. - ("operation_nesting", Sched::Parallel, Tier::Fast), + ("operation_nesting", Sched::Parallel, Tier::Weekly), ("short_sleep_livelock", Sched::Parallel, Tier::Fast), // The spawn half alone: one headless boot whose verdict is three kernel // log lines. - ("klogd_hosted", Sched::Parallel, Tier::Fast), - // The two actuator boots (`klogd-panic`, `usbd-panic`), split off so the - // spawn half is per-PR again; alone they still price over the ceiling, - // and sit Nightly. - ("klogd_panic_halts", Sched::Parallel, Tier::Nightly), + ("klogd_hosted", Sched::Parallel, Tier::Weekly), // The two dead ends of the panic path, each staged on purpose and read for - // what the machine manages to say on its way out. **Two names because one - // over two boots measured 12 s twelve-wide on the dev host**, against - // `screen_late_panic`'s 5 s there for one boot of the same shape and its - // 3,782 ms in CI — a single name would have arrived at the fast tier's line - // with nothing to spare. Each boot dies inside the boot phases at the marker - // the harness waits for, so neither pays for a userland. Parallel and Fast: + // what the machine manages to say on its way out. Each boot dies inside the + // boot phases at the marker the harness waits for, so neither pays for a + // userland. Parallel: // every verdict is a substring of a report the guest wrote, and there is no // clock in any of it. - ("reentry_names_the_first_panic", Sched::Parallel, Tier::Fast), + ("reentry_names_the_first_panic", Sched::Parallel, Tier::Weekly), // The kernel hasher's boot-order obligation, in the row above's shape and // for its reasons. - ("hash_seed_precedes_every_map", Sched::Parallel, Tier::Fast), - ("double_panic_names_the_fault", Sched::Parallel, Tier::Nightly), + ("hash_seed_precedes_every_map", Sched::Parallel, Tier::Weekly), + ("double_panic_names_the_fault", Sched::Parallel, Tier::Weekly), // The third shape: a `#PF` inside a panic, which is the one // `fatal_exception`'s recursive short-circuit exists for and the one it // never classified. Same boot shape as its two neighbours — dies inside the - // boot phases at the marker, no userland — so Parallel and, pending its - // first measured run, Fast. - ("nested_fault_is_recursive", Sched::Parallel, Tier::Fast), - // The conservation law across `SYS_LOG_READ`, one registered name per - // width, and the nesting gate at one CPU. **Three names because one over - // three boots measured 17,112 ms in CI** — over the fast tier's line, and - // the gate the whole design turns on may not sit in the nightly tier — and - // because the three widths are different subjects rather than one subject - // measured three times. Parallel: every verdict is a ledger the + // boot phases at the marker, no userland — so Parallel. + ("nested_fault_is_recursive", Sched::Parallel, Tier::Weekly), + // The conservation law across `SYS_LOG_READ`, and the nesting gate at one + // CPU. Parallel: every verdict is a ledger the // guest computes over its own records — every sequence number read or // counted lost, every payload regenerated byte for byte — and not one of // them reads a clock. A loaded host makes the producers outrun the reader // further, which moves records from `read` into `lost` and leaves the law // exactly where it was. - ("log_conservation_smp1", Sched::Parallel, Tier::Fast), - ("log_conservation_smp4", Sched::Parallel, Tier::Nightly), - ("log_conservation_smp8", Sched::Parallel, Tier::Fast), - ("log_nested_emit", Sched::Parallel, Tier::Fast), + ("log_conservation_smp1", Sched::Parallel, Tier::Weekly), + ("log_nested_emit", Sched::Parallel, Tier::Weekly), // The same interrupt one window earlier — between a record's shard-pointer // read and its `xadd` — and its negative control, which is the only reader - // `log-unbracketed-reserve` has ever had. Parallel and Fast for + // `log-unbracketed-reserve` has ever had. Parallel for // `log_nested_emit`'s reasons: both verdicts are the guest's ledger over its // own records, one saying the shard kept a single order and the other that // it lost it by name, and no clock is in either. - ("log_reserve_window", Sched::Parallel, Tier::Fast), - ("log_reserve_window_negative", Sched::Parallel, Tier::Fast), - // Two processes building a fixed-width line out of two `write`s each, and a - // count of the lines that carry both of them. Parallel and Fast: the verdict - // is a count over a fixed number of lines the guest declares, so a loaded - // host changes when the writers run and not whether a line is whole. It boots - // its own machine because what it reads is the console capture, which a - // shared boot fills with everything else. - ("console_line_atomicity", Sched::Parallel, Tier::Nightly), - // What the C family is allowed to conclude from the line above being whole: - // a guest writes a daemon-shaped line into a real capture window on purpose + ("log_reserve_window", Sched::Parallel, Tier::Weekly), + ("log_reserve_window_negative", Sched::Parallel, Tier::Weekly), + // A guest writes a daemon-shaped line into a real capture window on purpose // and the real comparison ignores it, with the filter turned off as the // control. One boot, two `echo`s, and every verdict is a string comparison - // the host makes over a capture — no clock in it; Nightly because its - // *wall* clock is whatever the partition co-schedules, and it straddles the - // fast line run to run. - ("c_capture_ignores_daemon_lines", Sched::Parallel, Tier::Nightly), - // A poll on the machine's log against a *handle* going away. Parallel and - // Fast: both halves are verdicts the guest computes — a completion count + // the host makes over a capture — no clock in it. + ("c_capture_ignores_daemon_lines", Sched::Parallel, Tier::Weekly), + // A poll on the machine's log against a *handle* going away. Parallel: both + // halves are verdicts the guest computes — a completion count // immediately after a close, retried against a record arriving in the same // microseconds, and a completion afterwards bounded far above the two // scheduler passes it needs. - ("log_poll_outlives_a_close", Sched::Parallel, Tier::Fast), + ("log_poll_outlives_a_close", Sched::Parallel, Tier::Weekly), // The same question asked of the keyboard, where two *kinds* of object name // one source: a poll on stdin against the keyboard claim going away, a poll // on the mouse claim against its own, and an injected keystroke to show the - // first was still armed. Parallel and Fast: two of the three verdicts are + // first was still armed. Parallel: two of the three verdicts are // counts the guest takes immediately after a close on its own thread, and // the third is bounded far above the one interrupt it waits for. ("keyboard_claim_close_spares_stdin", Sched::Parallel, Tier::Fast), // One boot that stops dead in phase 3, read for what it managed to say. - ("pre_idle_wedge_speaks", Sched::Parallel, Tier::Fast), + ("pre_idle_wedge_speaks", Sched::Parallel, Tier::Weekly), ("i8042_health", Sched::Parallel, Tier::Nightly), - // And one from here to `i8042_mouse` (`I8042_TRACE`), which is why all - // three carry the answer the last of them needs. - // - // None of the three measures a rate. All three keep fewer bytes in flight - // than QEMU's PS/2 device holds, and all three do it the same way: nothing - // goes out until the guest has reported what the injection before it - // produced — `i8042_mouse` within [`MOUSE_LEAD`], the two keyboard ones a - // group at a time. So a guest with less of the host is a longer run and not - // a smaller count. **A wall clock cannot buy that bound**: the two keyboard - // tests spaced their injections with `thread::sleep` and put 26 and 20 - // bytes against a sixteen-byte device queue, which held only as long as the - // guest kept draining, and a stalled guest lost bytes with nothing anywhere - // reporting a loss. `i8042_keyboard` itself held the group Nightly on a cost - // that was really the fixed 5 s collection deadline in - // `test_rs_i8042_keyboard`; now that the binary exits on a sentinel instead, - // all three return. - ("i8042_keyboard", Sched::Parallel, Tier::Fast), - ("i8042_no_spurious_wake", Sched::Parallel, Tier::Fast), - ("i8042_mouse", Sched::Parallel, Tier::Fast), + // And one from here to `i8042_mouse` (`I8042_TRACE`). Neither measures a + // rate: nothing goes out until the guest has reported what the injection + // before it produced — `i8042_mouse` within [`MOUSE_LEAD`], the keyboard one + // a group at a time — so a guest with less of the host is a longer run and + // not a smaller count. + ("i8042_no_spurious_wake", Sched::Parallel, Tier::Nightly), + ("i8042_mouse", Sched::Parallel, Tier::Nightly), // A boot each, and deliberately not a group: every one of them changes - // the machine's layout, which `i8042_keyboard` asserts against, and a - // wizard that exits the instant it has its answer leaves the guest with - // nothing to run — so a later member reads a console the previous one is + // the machine's layout, and a wizard that exits the instant it has its + // answer leaves the guest with nothing to run — so a later member reads a console the previous one is // still draining into. // // Each is a wizard conversation typed from the host, and that used to make @@ -1342,19 +1249,11 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // against a marker with a twenty-second ceiling, so a slower guest is a // slower test and not a different verdict — which is the same argument // `i8042_kbd_echo` has run on at width 4 since the phase landed. - // - // **Returned 2026-08-17.** Eight of its 12.6 s CI price were - // `test_rs_locale_gate layout` holding an idle keyboard open until a fixed - // deadline expired, against half a second of injection; it exits on the End - // key's release now, which is `i8042_keyboard`'s own sentinel and the fix - // made for that whole family. - ("swiss_german_layout", Sched::Parallel, Tier::Fast), + ("swiss_german_layout", Sched::Parallel, Tier::Nightly), // One `LOCALE_WIZARD` boot for the pair since the drainer was made // runnable at commit — the boot-apiece and the injected drain keys it took // to share one were both the closed log-ring lag. Adjacent because - // `group_of` makes adjacency load-bearing; the carrier straddles the - // fast line run to run, so the pair is Nightly — the rider by its - // `RidesTheBootOf` row, the collateral that record exists to name. + // `group_of` makes adjacency load-bearing. ("locale_detect", Sched::Parallel, Tier::Nightly), ("locale_detect_unrecognized", Sched::Parallel, Tier::Nightly), // The wizard on the two surfaces the machine actually has, rather than on @@ -1370,7 +1269,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // An unmodified iced app on the desktop, launched from the shell: the // window, its text in the system font, no redraw it did not ask for, and // a clean exit when the compositor closes it. - ("toolkit_iced", Sched::Parallel, Tier::Nightly), + ("toolkit_iced", Sched::Parallel, Tier::Weekly), // The wait every winit loop blocks in, the loop itself through winit's // API, and an animation held to the compositor's frame events. ("toolkit_window_wake", Sched::Parallel, Tier::Nightly), @@ -1385,7 +1284,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // verdicts are counts the report has to agree with itself about, not a // wall-clock margin — the one duration in it is the dump's own 250 ms // ceiling, which the guest spends and the host never measures. - ("blocked_dump", Sched::Parallel, Tier::Fast), + ("blocked_dump", Sched::Parallel, Tier::Nightly), // Two boots of one machine compared on the guest's own `Boot: complete` // with a 300 ms allowance, which is the whole assertion — a real-time // verdict, so Nightly. @@ -1398,54 +1297,44 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // timer-anchored, and its price straddles the ceiling run to run (9,355 / // 10,568 / 11,073 ms across three measurements) for exactly that reason. ("i8042_quarantine", Sched::Parallel, Tier::Nightly), - // The negative-direction half of the same gate, and the one that runs on - // every PR: no QEMU can stage a CPU into spinning through idle on + // The negative-direction half of the same gate: no QEMU can stage a CPU into spinning through idle on // purpose, so this is `idle_is_spinning` proving its teeth against a // crafted trace shaped like the regression, the way // `control_regs`/`control_regs_verdict` split the same question. - ("i8042_quarantine_verdict", Sched::Parallel, Tier::Fast), - ("i8042_budget_expiry", Sched::Parallel, Tier::Fast), - ("i8042_fadt_denial", Sched::Parallel, Tier::Fast), - ("i8042_kbd_echo", Sched::Parallel, Tier::Fast), - // Returned 2026-08-13: relegated on this branch for the same 5 s fixed - // collection deadline the rest of the family crossed on, now fixed at - // the source (`test_rs_i8042_keyboard` exits on a sentinel). + ("i8042_quarantine_verdict", Sched::Parallel, Tier::Weekly), + ("i8042_budget_expiry", Sched::Parallel, Tier::Nightly), + ("i8042_fadt_denial", Sched::Parallel, Tier::Weekly), + ("i8042_kbd_echo", Sched::Parallel, Tier::Nightly), ("i8042_undecoded_bytes", Sched::Parallel, Tier::Fast), // Its verdict is a cadence, and its absence is the assertion — both read // off the guest's own `last byte at Nms` stamps. The gap it injects is // 3 s against a 500 ms period, so six periods of margin decide whether the // report is on the pin or on a timer. ("i8042_health_cadence", Sched::Parallel, Tier::Nightly), - ("xhci_xecp_walk", Sched::Parallel, Tier::Fast), - ("xhci_slot_exhaustion", Sched::Parallel, Tier::Nightly), - ("usb_storage_gate", Sched::Parallel, Tier::Nightly), - ("usb_storage_shapes", Sched::Parallel, Tier::Nightly), - ("usb_refused_disk_first", Sched::Parallel, Tier::Nightly), - ("xhci_scan_hands_over_a_free_slot", Sched::Parallel, Tier::Fast), + ("xhci_xecp_walk", Sched::Parallel, Tier::Weekly), + ("xhci_slot_exhaustion", Sched::Parallel, Tier::Weekly), + ("usb_storage_gate", Sched::Parallel, Tier::Weekly), + ("usb_storage_shapes", Sched::Parallel, Tier::Weekly), + ("usb_refused_disk_first", Sched::Parallel, Tier::Weekly), + ("xhci_scan_hands_over_a_free_slot", Sched::Parallel, Tier::Nightly), // The owner's freeze, staged: `device_del` on the stick carrying `/boot` // and `/log` while the desktop draws. Serial because both verdicts are // liveness ceilings — two 2 s compositor reporting intervals inside 20 s, // and a console round trip inside 20 s — and a guest sharing the host with // eleven others answers those late for reasons that are not the defect. ("usb_boot_stick_pulled", Sched::Serial, Tier::Nightly), - ("usb_pool_exhausted", Sched::Parallel, Tier::Fast), - ("usb_short_read", Sched::Parallel, Tier::Nightly), - // A plug over QMP and two host-side verdicts, neither of them a byte - // comparison alone: the fixed 1.2 s wait against a 100 ms debounce is a - // staged latency window the LATE_READY assertion is waited out before - // being read, which is timer-anchored regardless of how comfortable the - // margin looks under TCG. - ("usb_disk_index_stable", Sched::Parallel, Tier::Nightly), - ("usb_storage_write_error", Sched::Parallel, Tier::Fast), + ("usb_pool_exhausted", Sched::Parallel, Tier::Weekly), + ("usb_short_read", Sched::Parallel, Tier::Weekly), + ("usb_storage_write_error", Sched::Parallel, Tier::Weekly), ("usb_flush_optional", Sched::Parallel, Tier::Nightly), - ("xhci_deaf_registers", Sched::Parallel, Tier::Nightly), + ("xhci_deaf_registers", Sched::Parallel, Tier::Weekly), // The window is anchored on the controller's own port-power stamp now, not // boot, so a slow boot no longer eats it — but the bound is still a fixed // span of the guest's own TSC clock (`SLOW_CONNECT_NS`/`DEBOUNCE_NS`), and // a host running several other guests can still stall this one's vCPU past // that span for reasons that are not the defect. ("xhci_slow_connect", Sched::Serial, Tier::Nightly), - ("xhci_portsc_rw1c", Sched::Parallel, Tier::Fast), + ("xhci_portsc_rw1c", Sched::Parallel, Tier::Weekly), // One staged break and no other, which puts the driver's recovery finishing // on its first try in the verdict: a retried command that reaches an // endpoint still halted from the staged break logs a second `transport @@ -1453,14 +1342,8 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // had — its own doc says one break under KVM and two under TCG off the // same tree, which is the race timer-anchored, not a margin, describes. ("usb_transport_break", Sched::Serial, Tier::Nightly), - ("xhci_full_speed_device", Sched::Parallel, Tier::Nightly), - ("xhci_superspeed_ports", Sched::Parallel, Tier::Fast), - // Two of the three below stage plug and unplug with fixed waits, 600-800 ms - // against a 100 ms debounce, plus 20-200 ms sleeps pacing the input pokes - // that follow — staged latency windows gating the verdict, so timer- - // anchored even though every individual check is a count of what the - // guest logged. - ("xhci_hotplug", Sched::Parallel, Tier::Nightly), + ("xhci_full_speed_device", Sched::Parallel, Tier::Weekly), + ("xhci_superspeed_ports", Sched::Parallel, Tier::Weekly), // `xhci_flap` is the one that genuinely races the host against the guest: // its two QMP writes have to land inside *one* 100 ms debounce or the state // under test never happens, and it says so — `no replug collapsed inside a @@ -1468,56 +1351,48 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // second write past 100 ms turns a green machine red with that sentence, // which is indistinguishable from the driver defect it hunts. ("xhci_flap", Sched::Serial, Tier::Nightly), - ("xhci_hid_break", Sched::Parallel, Tier::Nightly), - ("xhci_descriptor_walk", Sched::Parallel, Tier::Fast), + ("xhci_descriptor_walk", Sched::Parallel, Tier::Weekly), ("esp_filesystem", Sched::Parallel, Tier::Nightly), // Three boots: a budget-refused flush retried and kept, the deadman's // declared death, and a hung device's failed reset escalation — the three // exits of `object/ops.rs`'s fsync loop. Every verdict is line presence // and host-side bytes, never a wall-clock margin. ("log_flush_retry", Sched::Parallel, Tier::Nightly), - ("toybox_cp_volume", Sched::Parallel, Tier::Nightly), + ("toybox_cp_volume", Sched::Parallel, Tier::Weekly), ("kernel_log_file", Sched::Parallel, Tier::Nightly), // Serial: its verdict is a cadence — heartbeats against a 250 ms period — // and a guest sharing the host with eleven others reaches its idle loop // late for reasons that are not the defect. ("kernel_heartbeat", Sched::Serial, Tier::Nightly), - // Both own their images and their lanes, and neither verdict is a - // wall-clock margin: the guest's clock starts from an instant the host set - // and the only duration either measures is how long a boot takes to reach - // its log sink, against a bound five minutes wide. A host so loaded that - // this failed would have failed every timed test in the phase first. - ("wall_clock_file", Sched::Parallel, Tier::Fast), // The five RTC/firmware shapes, one kernel build and one boot each. Five // registrations because the artifact memo builds one kernel per feature // set anyway, so the split costs nothing and the parallel phase gets five - // jobs it can place instead of one serial five-boot job it cannot. Three - // priced near the line across two runs and sit Nightly. - ("wall_clock_rtc_dead", Sched::Parallel, Tier::Nightly), - ("wall_clock_rtc_unstable", Sched::Parallel, Tier::Fast), - ("wall_clock_no_century", Sched::Parallel, Tier::Fast), - ("wall_clock_century_register", Sched::Parallel, Tier::Nightly), - ("wall_clock_zone", Sched::Parallel, Tier::Nightly), + // jobs it can place instead of one serial five-boot job it cannot. + ("wall_clock_rtc_dead", Sched::Parallel, Tier::Weekly), + ("wall_clock_rtc_unstable", Sched::Parallel, Tier::Weekly), + ("wall_clock_no_century", Sched::Parallel, Tier::Weekly), + ("wall_clock_century_register", Sched::Parallel, Tier::Weekly), + ("wall_clock_zone", Sched::Parallel, Tier::Weekly), // `xhci_slow_connect`'s shape against the disk's port, but its actuator // masks the port until `BOOT_SCAN_DONE` — a kernel event, not a duration — // so what it stages is an ordering with no wall-clock margin on either // side: nothing here needs the serial tail. - ("late_storage_connect", Sched::Parallel, Tier::Nightly), - ("log_backing_read_error", Sched::Parallel, Tier::Fast), - ("boot_volume_metadata_error", Sched::Parallel, Tier::Fast), - ("log_partition_layout", Sched::Parallel, Tier::Fast), + ("late_storage_connect", Sched::Parallel, Tier::Weekly), + ("log_backing_read_error", Sched::Parallel, Tier::Weekly), + ("boot_volume_metadata_error", Sched::Parallel, Tier::Weekly), + ("log_partition_layout", Sched::Parallel, Tier::Weekly), // What the loader does with a slot's ROOT: bytes its signature does not // cover, a parameter naming another, an overlapping partition and an // unreadable chunk refused by name, and a twin on the boot disk or on // another disk never read. Serial, not by association: each stages a whole // boot image, one a second 32 GiB stick beside it. - ("root_candidate_malformed", Sched::Serial, Tier::Fast), - ("root_named_but_absent", Sched::Serial, Tier::Fast), - ("root_chunk_refused", Sched::Serial, Tier::Fast), - ("root_candidate_overlaps", Sched::Serial, Tier::Fast), - ("root_named_twice_on_the_boot_disk", Sched::Serial, Tier::Fast), - ("root_named_twice", Sched::Serial, Tier::Nightly), - ("log_partition_identity", Sched::Parallel, Tier::Nightly), + ("root_candidate_malformed", Sched::Serial, Tier::Weekly), + ("root_named_but_absent", Sched::Serial, Tier::Weekly), + ("root_chunk_refused", Sched::Serial, Tier::Nightly), + ("root_candidate_overlaps", Sched::Serial, Tier::Nightly), + ("root_named_twice_on_the_boot_disk", Sched::Serial, Tier::Nightly), + ("root_named_twice", Sched::Serial, Tier::Weekly), + ("log_partition_identity", Sched::Parallel, Tier::Weekly), ("cache_eviction", Sched::Parallel, Tier::Nightly), // The write-back queue's three negative controls (wall 4 of // `issues/kernel/every-wait-in-this-kernel-is-a-spin.md`). `writeback_reopen` @@ -1529,70 +1404,65 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // The watch's lost-wake window, staged: `watch-window` holds every pipe // waiter between reading its condition and parking, so the peer's post lands where // only the notified bit carries it to the commit. - ("blocking_read_window", Sched::Parallel, Tier::Fast), + ("blocking_read_window", Sched::Parallel, Tier::Nightly), // A sibling's munmap and mmap staged between a typed copy's translation // and its store (`copy-meets-a-remap`): the store never reaches the region // mapped after it. - ("user_copy_races_munmap", Sched::Parallel, Tier::Fast), + ("user_copy_races_munmap", Sched::Parallel, Tier::Nightly), // A sibling's store staged between a thread's TLS block being placed and // its rebase (`tls-rebase-window`): the block is never reachable there. - ("tls_rebase_window", Sched::Parallel, Tier::Fast), - ("writeback_reopen", Sched::Parallel, Tier::Fast), - ("writeback_spawn", Sched::Parallel, Tier::Nightly), - ("writeback_durability", Sched::Parallel, Tier::Nightly), + ("tls_rebase_window", Sched::Parallel, Tier::Nightly), + ("writeback_reopen", Sched::Parallel, Tier::Weekly), + ("writeback_spawn", Sched::Parallel, Tier::Weekly), + ("writeback_durability", Sched::Parallel, Tier::Fast), // `KernelHw::switch`'s SS reload (AMD `X86_BUG_SYSRET_SS_ATTRS`) observed the // one way a guest can, since its `SYSRET` does not reproduce the erratum. Reds // the day that `mov ss` leaves the switch. - ("sysret_ss_reload", Sched::Parallel, Tier::Fast), - // The FAT32 read side's revocation gate, and a host-side volume oracle for - // the same reason `writeback_durability` is one: whether the clusters the - // unlink freed were really reissued, and whether the cycle left a volume, are - // both questions the guest that staged them cannot answer about itself. - ("fat_backing_revoked", Sched::Parallel, Tier::Nightly), + ("sysret_ss_reload", Sched::Parallel, Tier::Weekly), // F5 and F6's negative controls: an fsync that must keep refusing while the // device refuses its cache flush, and a mid-flush redirty raced for real and // re-read off the image. Both bodies in `tests/common/volumes.rs`. - ("fsync_failed_commit", Sched::Parallel, Tier::Nightly), - ("redirty_mid_flush", Sched::Parallel, Tier::Nightly), + ("fsync_failed_commit", Sched::Parallel, Tier::Weekly), + ("redirty_mid_flush", Sched::Parallel, Tier::Weekly), // A truncate staged inside a flush's metadata window, re-read off the image. - ("ftruncate_flush_race", Sched::Parallel, Tier::Nightly), - // The rename gate's FAT arm, a host-side volume oracle like `fat_backing_revoked`. - ("fs_rename_durable", Sched::Parallel, Tier::Nightly), + ("ftruncate_flush_race", Sched::Parallel, Tier::Weekly), + // The rename gate's FAT arm, a host-side volume oracle. + ("fs_rename_durable", Sched::Parallel, Tier::Weekly), // The directory work's FAT arm, `fs_rename_durable`'s oracle shape. - ("fs_dirs_durable", Sched::Parallel, Tier::Fast), - ("va_exhaustion", Sched::Parallel, Tier::Fast), - ("heap_ceiling_recovery", Sched::Parallel, Tier::Nightly), - ("iommu_context_absent", Sched::Parallel, Tier::Fast), - ("iommu_empty_domain", Sched::Parallel, Tier::Fast), - ("iommu_interrupt_remapping", Sched::Parallel, Tier::Fast), + ("fs_dirs_durable", Sched::Parallel, Tier::Weekly), + ("va_exhaustion", Sched::Parallel, Tier::Weekly), + ("heap_ceiling_recovery", Sched::Parallel, Tier::Weekly), + ("iommu_context_absent", Sched::Parallel, Tier::Weekly), + ("iommu_empty_domain", Sched::Parallel, Tier::Weekly), + ("iommu_interrupt_remapping", Sched::Parallel, Tier::Weekly), ("iommu_virtio_platform", Sched::Parallel, Tier::Nightly), - ("iommu_domain_isolation", Sched::Parallel, Tier::Nightly), - ("iommu_gpu_scanout_swap", Sched::Parallel, Tier::Fast), - ("iommu_gpu_foreign_backing", Sched::Parallel, Tier::Fast), - ("iommu_hda_foreign_bdl", Sched::Parallel, Tier::Fast), - ("iommu_sound_foreign_dma", Sched::Parallel, Tier::Fast), + ("iommu_domain_isolation", Sched::Parallel, Tier::Weekly), + ("iommu_gpu_scanout_swap", Sched::Parallel, Tier::Nightly), + ("iommu_gpu_foreign_backing", Sched::Parallel, Tier::Weekly), + ("iommu_hda_foreign_bdl", Sched::Parallel, Tier::Weekly), + ("iommu_sound_foreign_dma", Sched::Parallel, Tier::Weekly), // The one arm of that family whose device is driven by a *process*, and // the only one whose verdict is that the machine is still running. - ("userdev_dma_fault", Sched::Parallel, Tier::Fast), + ("userdev_dma_fault", Sched::Parallel, Tier::Nightly), // Two claims of a function nothing resets, the first closed with its grant // still mapped: the verdict is the first holder's own grant, read in the // guest after the second holder wrote its own. - ("userdev_residue_is_its_own", Sched::Parallel, Tier::Fast), + ("userdev_residue_is_its_own", Sched::Parallel, Tier::Nightly), // blockd, the NVMe driver in userland, on a second controller beside the // kernel's: partitions served and timed against the kernel's driver; a // controller reset and its own death, each survived by the client and // judged off the image by the host's readers; and a transfer outside what // its function was lent, which is a fault record. Each boot runs several // blockd lifetimes and one waits out a ten-second silence. - ("blockd_serves_partitions", Sched::Parallel, Tier::Nightly), - ("blockd_survives_its_death", Sched::Parallel, Tier::Nightly), - ("blockd_dma_outside_the_lent", Sched::Parallel, Tier::Nightly), + ("blockd_serves_partitions", Sched::Parallel, Tier::Weekly), + ("blockd_survives_its_death", Sched::Parallel, Tier::Weekly), + ("blockd_dma_outside_the_lent", Sched::Parallel, Tier::Weekly), // What a claim may lend: a kernel driver's pool refused, the claim's bound // refusing the next region at the count it leaves room for, and lending and // taking back ten narrowed domains' worth of addresses with the kernel // standing; then, on a second boot, a function no release resets is never // lent where it was left aimed. - ("blockd_lends_within_its_bound", Sched::Parallel, Tier::Nightly), + ("blockd_lends_within_its_bound", Sched::Parallel, Tier::Weekly), // H4: soundd driving an Intel HDA controller itself, read back off the // device. Serial — its verdict is a wav capture, and one taken while eleven // other guests contend for the host measures the host. @@ -1602,27 +1472,27 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // and its counters rather than a capture, so it runs wide. ("hda_client_stall", Sched::Parallel, Tier::Nightly), ("hda_two_live_refused", Sched::Parallel, Tier::Fast), - ("serial_vocabulary", Sched::Parallel, Tier::Fast), + ("serial_vocabulary", Sched::Parallel, Tier::Weekly), // Host-side, no guest: the harness asking whether it can still tell a // suspended machine from a slow one, and whether it reports one as a // verdict it does not have. - ("suspend_detector", Sched::Parallel, Tier::Fast), - ("suspend_invalidates_a_verdict", Sched::Parallel, Tier::Fast), - ("stall_is_not_a_verdict", Sched::Parallel, Tier::Fast), + ("suspend_detector", Sched::Parallel, Tier::Weekly), + ("suspend_invalidates_a_verdict", Sched::Parallel, Tier::Weekly), + ("stall_is_not_a_verdict", Sched::Parallel, Tier::Weekly), // Same: whether two guests can still be handed one lane's NVMe image, which // is what a shared-boot reboot did to itself. - ("nvme_image_is_held_by_one_guest", Sched::Parallel, Tier::Fast), + ("nvme_image_is_held_by_one_guest", Sched::Parallel, Tier::Weekly), // Same: what a whole run exits with. - ("run_exit_status", Sched::Parallel, Tier::Fast), + ("run_exit_status", Sched::Parallel, Tier::Nightly), // Same: the control-register verdict, against the machine this tree // actually booted before `arch/x86_64/control_regs.rs`. - ("control_regs_verdict", Sched::Parallel, Tier::Fast), + ("control_regs_verdict", Sched::Parallel, Tier::Weekly), // Same: which of the two shared boots each binary belongs on, asked of the // binaries rather than of the list that claims to name them. - ("suite_split", Sched::Parallel, Tier::Fast), + ("suite_split", Sched::Parallel, Tier::Weekly), // Same: whether a run that did not attempt most of the suite's measured cost says // so where its verdict is read. - ("nightly_tier_is_announced", Sched::Parallel, Tier::Fast), + ("nightly_tier_is_announced", Sched::Parallel, Tier::Weekly), ]; /// The test binaries a [`MACHINE_TESTS`] or [`SCREEN_TESTS`] entry runs, which @@ -1654,8 +1524,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("xhci_msi_only", &["test_rs_input_events"]), ("metal_sim_input", &["test_rs_input_events"]), ("xhci_flap", &["test_rs_input_events"]), - ("xhci_hotplug", &["test_rs_input_events"]), - ("xhci_hid_break", &["test_rs_input_events"]), ("nvme_large_device", &["test_rs_nvme_home_roundtrip"]), ("va_exhaustion", &["test_rs_va_exhaustion"]), ("readdir_bound", &["test_rs_readdir_bound"]), @@ -1673,7 +1541,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("i8042_kbd_echo", &["test_rs_i8042_keyboard"]), ("i8042_undecoded_bytes", &["test_rs_i8042_keyboard"]), ("i8042_quarantine", &["test_rs_i8042_keyboard"]), - ("i8042_keyboard", &["test_rs_i8042_keyboard"]), ("i8042_no_spurious_wake", &["test_rs_i8042_keyboard"]), ("i8042_mouse", &["test_rs_i8042_mouse"]), ("swiss_german_layout", &["test_rs_locale_gate"]), @@ -1689,7 +1556,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("netd_udp_any_address", &["test_rs_netd_udp_any_address"]), ("netd_lookup_let_go", &["test_rs_netd_lookup_let_go"]), ("netd_hostile_peer", &["test_rs_netd_hostile_peer"]), - ("launcher_refusals", &["test_rs_launcher_refusals"]), ("spawn_cwd", &["test_rs_spawn_cwd"]), ("input_claim_absent", &["test_rs_input_absent"]), ("gpu_set_resolution", &["test_rs_gpu_set_resolution"]), @@ -1734,13 +1600,10 @@ const CARRIES: &[(&str, &[&str])] = &[ ("layout_fresh_boot", &["test_rs_layout_paths"]), ("broken_data_volume_is_absent", &["test_rs_home_absent"]), ("data_candidate_with_bad_geometry_is_absent", &["test_rs_home_absent"]), - ("home_budget_refusal_retried", &["test_rs_home_fsync_budget"]), ("home_overwrite_reads_back", &["test_rs_home_overwrite_zero"]), - ("so_cache_refusals", &["test_rs_so_cache_policy"]), ("boot_volume_metadata_error", &["test_rs_boot_volume_metadata_error"]), ("esp_filesystem", &["test_rs_esp_files"]), ("log_flush_retry", &["test_rs_esp_files"]), - ("fat_backing_revoked", &["test_rs_fat_backing_revoked"]), ("fs_dirs_durable", &["test_rs_fs_dirs_durable"]), ("fs_rename_durable", &["test_rs_fs_rename_durable", "test_rs_fs_dirs_durable"]), ("fsync_failed_commit", &["test_rs_fsync_flush_failed"]), @@ -1769,13 +1632,10 @@ const CARRIES: &[(&str, &[&str])] = &[ ("log_stream", &["test_rs_log_origin", "test_rs_empty_dir_stat"]), ("log_stream_e1000e", &["test_rs_log_origin", "test_rs_empty_dir_stat"]), ("log_stream_stalled_reader", &["test_rs_log_flood"]), - ("console_line_atomicity", &["test_rs_console_line_atomicity"]), ("c_capture_ignores_daemon_lines", &["test_c_71_macro_empty_arg"]), ("quiesce_stops_the_machine", &["test_rs_quiesce_writers"]), ("quiesce_refuses_a_second_shutdown", &["test_rs_quiesce_twice"]), ("quiesce_wakes_on_the_last_park", &["test_rs_quiesce_last"]), - ("quiesce_wakes_on_the_last_exit", &["test_rs_quiesce_last"]), - ("quiesce_dump_holds_the_stopped", &["test_rs_quiesce_writers"]), ("quiesce_leaves_the_volume_whole", &["test_rs_quiesce_fsync"]), ("swap_crash_rolls_back", &["test_rs_swap_crash"]), ("swap_quiets_the_function", &["test_rs_swap_claim_idle"]), @@ -1783,7 +1643,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("swap_fault_tells_its_holder", &["test_rs_swap_claim_astray"]), ("swap_resets_the_function", &["test_rs_swap_flr_probe"]), ("swap_not_inherited", &["test_rs_swap_probe"]), - ("wall_clock_file", &["test_rs_wall_clock_now"]), ("wall_clock_rtc_dead", &["test_rs_wall_clock_now"]), ("wall_clock_rtc_unstable", &["test_rs_wall_clock_now"]), ("wall_clock_no_century", &["test_rs_wall_clock_now"]), @@ -6123,39 +5982,6 @@ fn run_screen_test( } Ok(()) } - "screen_early_panic" => { - // The window the console exists for: percpu is not up, mm::init - // has not run, and on a machine with no UART nothing else can - // report at all. The alert! that is the ready marker precedes - // capture(), panic_flush() and render(), so the screen is polled. - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Gop, - qmp: true, - kernel_params: &["test-early-panic"], - ready_marker: "EARLY PANIC:", - ..Default::default() - }, - ); - let dump = qemu.screendump_until("EARLY PANIC:", Duration::from_secs(30)); - let text = dump.text(); - print_screen(name, &text); - for want in ["EARLY PANIC:", "test-early-panic: on-screen console check"] { - if !text.contains(want) { - return Err(format!("{want:?} not on screen\ndecoded screen:\n{text}")); - } - } - check_colors( - &dump, - FILL_FATAL, - &["EARLY PANIC:", "test-early-panic: on-screen console check"], - "PAT:", - )?; - Ok(()) - } "screen_late_panic" => { // The ordinary fatal panic, which no userland process can produce: // crash_report, capture, panic_flush, halt_all_cpus, render. The @@ -7111,7 +6937,7 @@ fn group_of(name: &str) -> Option<&'static str> { | "metal_sim_ipc_hostile_peer" | "metal_sim_compositor_stall" | "metal_sim_client_death" => Some(METAL_SIM_DESKTOP), - "i8042_keyboard" | "i8042_no_spurious_wake" | "i8042_mouse" => Some(I8042_TRACE), + "i8042_no_spurious_wake" | "i8042_mouse" => Some(I8042_TRACE), // The positive wizard first: it applies a layout, and the negative // member reads only its own window, so the order is the argument that // no member reads state another left. @@ -7759,171 +7585,14 @@ fn compositor_screen_size(console: &str) -> Result<(u32, u32), String> { /// injects through a fresh connection sends this after its last injection /// instead of running out the binary's fallback deadline, except /// `i8042_health_cadence` — whose verdict is a report cadence over a real span, -/// not a delivered key. The two callers that hold one connection open for the -/// whole run ([`i8042_keyboard`], [`i8042_no_spurious_wake`]) send the same two -/// transitions as the last group of their own script: a `-qmp …,server` socket +/// not a delivered key. [`i8042_no_spurious_wake`], which holds one connection +/// open for the whole run, sends the same two transitions as the last group of +/// its own script: a `-qmp …,server` socket /// serves one monitor at a time, so a second one opened here would block. fn send_i8042_sentinel(socket: &Path) { qemu::qmp_send_keys(socket, &[("end", true), ("end", false)]); } -/// What [`i8042_keyboard`] types, as the groups it may have in flight at once, -/// each with the number of `kev` lines the guest owes for it. -/// -/// **A group is the unit of pacing, and its size is bounded by -/// [`QEMU_PS2_QUEUE`].** The device holds sixteen set-1 bytes and -/// `ps2_queue()` drops the seventeenth *silently and one byte at a time*; the -/// kernel never learns of it, so its `dropped`/`lost edges`/`overruns` -/// counters all read zero on a stream with a hole in it. What a lost byte -/// costs is not one transition: a lost make leaves its break to be filtered by -/// `handle_key` — a break for a usage nothing holds queues nothing — so the -/// whole key disappears, and a lost `0xE0` leaves the break to decode as an -/// unrelated keypad code that also holds nothing, so a press survives with no -/// release. Sending the script on a wall clock and hoping the guest keeps up -/// is what put 26 bytes against those 16 and produced both of those shapes on -/// CI. The largest group below is four bytes. -const KEYBOARD_SCRIPT: &[(&[(&str, bool)], usize)] = &[ - (&[("h", true), ("h", false)], 2), - (&[("e", true), ("e", false)], 2), - (&[("l", true), ("l", false)], 2), - (&[("l", true), ("l", false)], 2), - (&[("o", true), ("o", false)], 2), - // One command, so the chord arrives as a chord rather than as a race. - (&[("shift", true), ("b", true), ("b", false), ("shift", false)], 4), - (&[("left", true), ("left", false)], 2), - (&[("esc", true), ("esc", false)], 2), - // A modifier on its own, so a stuck one is visible. - (&[("shift", true)], 1), - (&[("shift", false)], 1), - // The sentinel the guest exits on; see [`send_i8042_sentinel`]. - (&[("end", true), ("end", false)], 2), -]; - -/// A key injected at the controller, decoded, mapped and delivered to a -/// userland process — IRQ delivery, set-1 decode, the HID mapping and the -/// shared translate/layout path, in one run. -/// -/// **Paced against the guest's own report**, for [`i8042_mouse`]'s reason and -/// [`KEYBOARD_SCRIPT`]'s: a group goes out only once every `kev` line the one -/// before it owed has come back, so at most four bytes are ever outstanding at -/// a device that holds sixteen. A guest that stalls costs this test wall clock -/// and never a verdict. -fn i8042_keyboard(boot: &mut Boot) -> Result<(), String> { - let qemu = &mut boot.qemu; - let boot = qemu.boot_log().to_string(); - if !boot.contains("i8042: kbd set2+xlat (readback 0x41)") { - return Err(format!("the PS/2 keyboard never came up:\n{boot}")); - } - - let sent = std::cell::Cell::new(0usize); - let seen = std::cell::Cell::new(0usize); - let result = { - let mut input: Option = None; - qemu.run_test_paced( - "test_rs_i8042_keyboard", - Duration::from_secs(20), - |socket, line| { - if line.contains(I8042_READY) { - input = Some(qemu::QmpInput::open( - socket.expect("i8042_keyboard needs BootOptions { qmp }"), - )); - } - if line.contains("kev usage=") { - seen.set(seen.get() + 1); - } - let Some(input) = input.as_mut() else { return }; - // What everything already sent owes. Nothing new goes out - // until the guest has reported all of it, which is what bounds - // the bytes outstanding at the device to one group's worth. - let owed: usize = KEYBOARD_SCRIPT[..sent.get()].iter().map(|(_, n)| n).sum(); - if seen.get() < owed { - return; - } - if let Some((keys, _)) = KEYBOARD_SCRIPT.get(sent.get()) { - input.keys(keys); - sent.set(sent.get() + 1); - } - }, - ) - }; - let (sent, seen) = (sent.get(), seen.get()); - if let Some(err) = &result.error { - // The guard, not the verdict: under the pacing the host is *waiting* - // for the guest when this fires, so what it establishes is that the run - // stopped and never that the machine dropped a key. - let owed: usize = KEYBOARD_SCRIPT.iter().map(|(_, n)| n).sum(); - return Err(format!( - "{STALLED} {err} — {sent} of {} groups sent and {seen} of {owed} key events back \ - when the host gave up waiting for the next\n{}", - KEYBOARD_SCRIPT.len(), - result.stdout - )); - } - - let events = parse_key_events(&result.stdout); - if events.is_empty() { - return Err(format!("no key event reached userland:\n{}", result.stdout)); - } - // Presses spell the injected text: IRQ delivery, set-1 decode, - // the HID mapping, the shared translate/layout path, and arrival - // in a userland process, in one assertion. - let typed: String = events - .iter() - .filter(|e| e.modifiers & 0x10 == 0) - .map(|e| e.translated.as_str()) - .collect(); - if !typed.contains("hello") { - return Err(format!("typed {typed:?}, want it to contain \"hello\"")); - } - if !typed.contains('B') { - return Err(format!("typed {typed:?} — Shift+b did not produce a capital")); - } - if !typed.contains("\u{1b}[D") { - return Err(format!("typed {typed:?} — Left arrow produced no escape sequence")); - } - for want in [0x29u8, 0x50, 0xE1] { - if !events.iter().any(|e| e.usage == want) { - return Err(format!("no event for HID usage {want:#04x} in {events:?}")); - } - } - // Every press is matched by a release — **the sentinel's included**. `0x4D` - // is End, and the guest exits on its release, so a run that never receives - // it runs out that binary's own five-second fallback instead and every - // assertion above still passes: a green test six seconds slower than its - // price, which is what a lost sentinel used to look like. - for usage in [0x0Bu8, 0x08, 0x0F, 0x12, 0x05, 0x29, 0x50, 0xE1, 0x4D] { - let presses = events.iter().filter(|e| e.usage == usage && e.modifiers & 0x10 == 0).count(); - let releases = events.iter().filter(|e| e.usage == usage && e.modifiers & 0x10 != 0).count(); - if presses == 0 || presses != releases { - return Err(format!( - "usage {usage:#04x}: {presses} presses, {releases} releases" - )); - } - } - // Nothing is left held: the bare Shift came back up. - let last = events.last().unwrap(); - if last.modifiers & !0x10 != 0 { - return Err(format!("a modifier is stuck down: last event {last:?}")); - } - // And they came from the i8042, not from somewhere else. - let drained: usize = qemu - .boot_log() - .lines() - .chain(result.serial.lines()) - .filter_map(trace_keys) - .filter(|&k| k > 0) - .sum(); - if drained == 0 { - return Err("no i8042 drain reported a key event".to_string()); - } - eprintln!( - " [i8042] {} events to userland, {drained} from the driver; {sent} groups, none sent \ - before the one before it came back", - events.len() - ); - Ok(()) -} - /// The line `tests/toyos-rust-tests/src/bin/locale_gate.rs` prints in `layout` /// mode once the surface holds the keyboard and the wizard's child has gone — /// the moment a key injected at this machine reaches a translator, and the one @@ -7934,8 +7603,7 @@ const SWISS_READY: &str = "===SWISS_READY==="; /// once, each with the number of `kev` lines the guest owes for it. /// /// **A group is the unit of pacing, and its size is bounded by -/// [`QEMU_PS2_QUEUE`]**, for [`KEYBOARD_SCRIPT`]'s reason and by the same -/// arithmetic: the widest group here is four transitions, eight set-1 bytes even +/// [`QEMU_PS2_QUEUE`]**: the widest group here is four transitions, eight set-1 bytes even /// if every one were `0xE0`-prefixed, against a device holding sixteen. The /// whole string is far more than the queue, so a host sending it on a wall clock /// loses its tail to a guest that stops draining. @@ -7984,8 +7652,9 @@ const SWISS_SCRIPT: &[(&[(&str, bool)], usize)] = &[ (&[("end", true), ("end", false)], 2), ]; -/// No group of [`SWISS_SCRIPT`] may outrun the device queue, on the same -/// worst-case width [`KEYBOARD_SCRIPT`] is held to. +/// No group of [`SWISS_SCRIPT`] may outrun the device queue even if every +/// transition in it is an `0xE0`-prefixed two-byte one, which is the widest a +/// non-Pause set-1 transition gets. const _: () = { let mut i = 0; while i < SWISS_SCRIPT.len() { @@ -8008,8 +7677,7 @@ const _: () = { /// on the table, the modifier levels, the ISO key and the dead-key machine at /// once. /// -/// **Paced against the guest's own report**, for [`i8042_keyboard`]'s reason and -/// [`SWISS_SCRIPT`]'s: a group goes out only once every `kev` line the one +/// **Paced against the guest's own report**, for [`SWISS_SCRIPT`]'s reason: a group goes out only once every `kev` line the one /// before it owed has come back, so at most one group's bytes are ever /// outstanding at a device that holds sixteen. A guest that stalls costs this /// test wall clock and never a verdict. @@ -10728,8 +10396,7 @@ fn soundd_clients_since(log: &str, from: usize, verb: &str) -> usize { /// bytes and no events must produce no wake. Pause is that stimulus — six /// bytes, deliberately swallowed. /// -/// It drives the same in-guest reader as [`i8042_keyboard`], for the userland -/// half of the assertion. +/// It drives `test_rs_i8042_keyboard`, for the userland half of the assertion. /// /// **The zero-event drain is arranged, not hoped for.** What a drain carries is /// whatever the ISR found in the ring, so a host that injects on a wall clock @@ -10884,21 +10551,6 @@ const QEMU_PS2_QUEUE: usize = 16; /// this much of [`QEMU_PS2_QUEUE`] for it. const CHORD_BYTES: usize = 6; -/// No group of [`KEYBOARD_SCRIPT`] may outrun the device queue even if every -/// transition in it is an `0xE0`-prefixed two-byte one, which is the widest a -/// non-Pause set-1 transition gets. -const _: () = { - let mut i = 0; - while i < KEYBOARD_SCRIPT.len() { - assert!( - KEYBOARD_SCRIPT[i].0.len() * 2 <= QEMU_PS2_QUEUE, - "an i8042_keyboard group can outrun QEMU's PS/2 queue, which drops what it \ - cannot hold one byte at a time and says nothing" - ); - i += 1; - } -}; - /// A PS/2 pointer packet. Three bytes, because the driver's aux init sends no /// IntelliMouse knock and QEMU therefore frames a plain mouse. const MOUSE_PACKET: usize = 3; @@ -11490,10 +11142,6 @@ fn run_machine_test( "data_candidate_with_bad_geometry_is_absent" => { storage::data_candidate_with_bad_geometry_is_absent(test_config, c_bins, rust_bins) } - "home_budget_refusal_retried" => { - storage::home_budget_refusal_retried(test_config, c_bins, rust_bins) - } - "so_cache_refusals" => storage::so_cache_refusals(test_config, c_bins, rust_bins), "home_overwrite_reads_back" => { storage::home_overwrite_reads_back(test_config, c_bins, rust_bins) } @@ -11510,8 +11158,6 @@ fn run_machine_test( "quiesce_stops_the_machine" => power::quiesce_stops_the_machine(test_config, c_bins, rust_bins), "quiesce_refuses_a_second_shutdown" => power::quiesce_refuses_a_second_shutdown(test_config, c_bins, rust_bins), "quiesce_wakes_on_the_last_park" => power::quiesce_wakes_on_the_last_park(test_config, c_bins, rust_bins), - "quiesce_wakes_on_the_last_exit" => power::quiesce_wakes_on_the_last_exit(test_config, c_bins, rust_bins), - "quiesce_dump_holds_the_stopped" => power::quiesce_dump_holds_the_stopped(test_config, c_bins, rust_bins), "watchdog_resets" => power::watchdog_resets(test_config, c_bins, rust_bins), "watchdog_fed" => power::watchdog_fed(test_config, c_bins, rust_bins), "loader_watchdog_arms" => power::loader_watchdog_arms(test_config, c_bins, rust_bins), @@ -11565,7 +11211,6 @@ fn run_machine_test( } "usb_pool_exhausted" => usb::usb_pool_exhausted(test_config, c_bins, rust_bins), "usb_short_read" => usb::usb_short_read(test_config, c_bins, rust_bins), - "usb_disk_index_stable" => usb::usb_disk_index_stable(test_config, c_bins, rust_bins), // Body in `tests/common/volumes.rs`, same reason. "esp_filesystem" => common::volumes::esp_filesystem(test_config, c_bins, rust_bins), "log_flush_retry" => common::volumes::log_flush_retry(test_config, c_bins, rust_bins), @@ -11575,9 +11220,6 @@ fn run_machine_test( // Body in `tests/common/volumes.rs`, same reason: the host-side oracle // shuts the guest down and reads `/log` back with `toyos-fat32-check`. "writeback_durability" => common::volumes::writeback_durability(test_config, c_bins, rust_bins), - // Same again: the FAT32 read side's revocation, judged off the volume the - // guest's unlink-and-reallocate cycle left behind. - "fat_backing_revoked" => common::volumes::fat_backing_revoked(test_config, c_bins, rust_bins), // `sysret-ss-probe` has iod null SS, force a switch, and log whether the // switch reloaded it; a missing `mov ss` turns `reloaded` into `NOT`. "sysret_ss_reload" => { @@ -12003,8 +11645,6 @@ fn run_machine_test( ); Ok(()) } - // Body in `tests/common/wallclock.rs`, same reason. - "wall_clock_file" => common::wallclock::wall_clock_file(test_config, c_bins, rust_bins), "wall_clock_rtc_dead" => common::wallclock::undated( test_config, c_bins, @@ -12065,21 +11705,13 @@ fn run_machine_test( usb::xhci_full_speed_device(test_config, c_bins, rust_bins) } "xhci_superspeed_ports" => usb::xhci_superspeed_ports(test_config, c_bins, rust_bins), - "xhci_hotplug" => usb::xhci_hotplug(test_config, c_bins, rust_bins), "xhci_flap" => usb::xhci_flap(test_config, c_bins, rust_bins), - "xhci_hid_break" => usb::xhci_hid_break(test_config, c_bins, rust_bins), // Body in `tests/common/iommu.rs`, same reason. "iommu_discovery" => common::iommu::iommu_discovery(test_config, c_bins, rust_bins), // Body in `tests/common/logread.rs`, so the hunk here stays one line. "log_conservation_smp1" => { common::logread::log_conservation_smp1(test_config, c_bins, rust_bins) } - "log_conservation_smp4" => { - common::logread::log_conservation_smp4(test_config, c_bins, rust_bins) - } - "log_conservation_smp8" => { - common::logread::log_conservation_smp8(test_config, c_bins, rust_bins) - } "log_nested_emit" => common::logread::log_nested_emit(test_config, c_bins, rust_bins), "log_reserve_window" => { common::logread::log_reserve_window(test_config, c_bins, rust_bins) @@ -12094,9 +11726,6 @@ fn run_machine_test( "c_capture_ignores_daemon_lines" => { common::console::c_capture_ignores_daemon_lines(test_config, c_bins, rust_bins) } - "console_line_atomicity" => { - common::console::console_line_atomicity(test_config, c_bins, rust_bins) - } "keyboard_claim_close_spares_stdin" => { common::console::keyboard_claim_close_spares_stdin(test_config, c_bins, rust_bins) } @@ -12197,9 +11826,6 @@ fn run_machine_test( boot_metal_sim_desktop(rust_bins) })) } - "i8042_keyboard" => i8042_keyboard(group_boot(held, I8042_TRACE, || { - boot_i8042_trace(test_config, c_bins, rust_bins) - })), "i8042_no_spurious_wake" => i8042_no_spurious_wake(group_boot(held, I8042_TRACE, || { boot_i8042_trace(test_config, c_bins, rust_bins) })), @@ -13442,8 +13068,7 @@ fn run_machine_test( // trampoline that never issues an `iretq`. It gets a process-table // entry rather than a bare task, and that is what makes it // nameable: without one a crash report would print a pid nothing - // in the machine resolves. What each row *means* when the panic - // really fires is `klogd_panic_halts`' two actuator boots. + // in the machine resolves. let qemu = QemuInstance::boot_with_options( test_config, c_bins, @@ -13452,85 +13077,6 @@ fn run_machine_test( ); klogd_hosted(&serial::Serial::boot(&qemu)) } - "klogd_panic_halts" => { - // **A kernel thread's panic is not recoverable by accident.** - // `syscall_rip` is never cleared, so the ordinary recovery - // predicate reads whatever user thread last ran on that CPU, and - // a kernel task would recover or halt by accident of work - // stealing. The row in `sched::kthread` replaces the accident - // with an answer; these two actuator boots walk both branches. - // - // The marker is a line of the crash *report* rather than `PANIC:` - // itself, because `boot_log` stops at the marker and the name is - // printed after the header — a boot stopped at the header would - // have nothing left to assert the process table against. - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - kernel_params: &["klogd-panic"], - ready_marker: "Process: klogd", - ..Default::default() - }, - ); - let mut dead = serial::Serial::boot(&qemu); - dead.must_say("PANIC:")?; - dead.must_say("klogd-panic: the console drainer died")?; - // The process table answered for a task with no *user* address - // space — since C6 it names the kernel's, which is what let - // `KernelPayload.address_space` stop being an `Option`. - dead.must_say("Process: klogd")?; - - // The verdict. A *recovered* panic kills the thread and lets the - // machine carry on into userland, which announces itself; the - // fatal branch halts every CPU. The window is a liveness margin - // and not a threshold: `klogd` panics as the scheduler starts, and - // the arm this must never become reaches the marker a few hundred - // milliseconds later — so three seconds is a tenfold margin over - // the state it refuses, and it is the whole of this test's fixed - // cost against the Fast ceiling. - const CARRIED_ON: Duration = Duration::from_secs(3); - dead.push(&qemu.drain_serial(CARRIED_ON)); - dead.must_not_say(qemu::DEFAULT_READY)?; - eprintln!(" [klogd] a kernel thread's panic halted the machine rather than recovering"); - - drop(qemu); - - // **The same panic on the other row, and it is the direction - // nothing had ever taken.** Two rows in one table are one row - // until both branches have been walked: before this arm, every - // kernel-thread panic this tree had ever run took `OnPanic::Halt`, - // so `Recover` was a value rather than a path — and the path it - // names goes through `poison_tid`, the idle loop's `reap_poisoned` - // and `zombify_poisoned`, none of which had ever seen a task with - // no user address space. A row that quietly halted the machine - // would make `usbd` and `iod` worse than the thread they were - // split off from. - // - // The verdict is content in the same window and never a timeout: - // the boot returns at the crash report's own line, and what the - // three seconds after it must contain is the ready marker the - // arm above must *not*. - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - kernel_params: &["usbd-panic"], - ready_marker: "Process: usbd", - ..Default::default() - }, - ); - let mut survived = serial::Serial::boot(&qemu); - survived.must_say("PANIC:")?; - survived.must_say("usbd-panic: the device thread died")?; - survived.must_say("Process: usbd")?; - survived.push(&qemu.drain_serial(CARRIED_ON)); - survived.must_say(qemu::DEFAULT_READY)?; - eprintln!(" [usbd] a kernel thread's panic killed the thread and the machine booted"); - Ok(()) - } "hash_seed_precedes_every_map" => { // `kernel/src/hasher.rs`'s `UNSEEDED`, as a prefix: the wrong seed // the compiler cannot reach, because the container works. Its other @@ -15519,66 +15065,6 @@ fn run_machine_test( ); Ok(()) } - "sshd_fail_closed" => { - // sshd with a network under it — the only boot that gets past its - // bind. What that reaches for the first time is the daemon's own - // state on disk: the identity it mints under `/state/sshd`, and the file - // it authenticates against. - // - // The verdict is that it authenticates nobody and says which file - // left it that way. A daemon that cannot accept any key must not - // be holding port 22, so "never listened" is asserted too — that - // is the half a missing-file check would still pass without. - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/sshdcase"); - let options = BootOptions { - profile: qemu::Profile::Headless, - ..Default::default() - }; - if !qemu::profile_argv(&options).iter().any(|a| a.contains("virtio-net")) { - return Err("this test needs a NIC and the profile has none".to_string()); - } - - let mut qemu = QemuInstance::boot_with_options(&config, &[], &[], options); - let mut console = qemu.boot_log().to_string(); - - // Minting proves init made `/state/sshd` and set it as the daemon's - // `HOME`; the fingerprint proves the key it wrote reads back. - // - // **Both files are named, because either one alone authorizes.** - // This boot stages neither, so the daemon has to report both as - // unreadable — a check on only the writable one would pass a - // machine whose image file was silently never consulted. - const WANT: [&str; 5] = [ - "sshd: minted a new host identity at /state/sshd/host_ed25519", - "sshd: host identity SHA256:", - "sshd: cannot read /state/sshd/authorized_keys", - "sshd: cannot read /system/etc/ssh_authorized_keys", - "sshd: no file names a usable key", - ]; - let stalled = - await_guest(&mut qemu, &mut console, "every line sshd owes", |c| { - WANT.iter().all(|w| c.contains(w)) - }) - .err(); - if let Some(why) = stalled { - eprintln!(" [sshd] {why}"); - } - for want in WANT { - if !console.contains(want) { - return Err(format!("{want:?} never reached the console:\n{console}")); - } - } - if console.contains("sshd: listening on port 22") { - return Err(format!( - "sshd listened on port 22 with no key it could ever accept:\n{console}" - )); - } - eprintln!( - " [sshd] host identity minted under /state/sshd, and neither authorized_keys file \ - left it refusing to listen at all" - ); - Ok(()) - } "layout_fresh_boot" => { // The layout as ruled, on a boot of its own with a blank DATA // volume, asked over the cable because sshd is the service that @@ -16157,66 +15643,6 @@ fn run_machine_test( eprintln!(" [netcase] {refused} hostile frames refused, netd named every peer it dropped"); Ok(()) } - "launcher_refusals" => { - // **`/system/bin/init` is the one process the machine cannot lose**, and - // every launcher client — the compositor, every terminal, every - // shell, sshd — can send it whatever it likes. The guest carries - // the verdicts: init answered, init is still launching, and the - // kernel's live-object count did not grow across sixteen refused - // launches. The host carries the one the guest cannot see — - // whether init said anything about what it refused. - // - // `tests/netcase` because its test-runner is the only one that - // receives a `launcher` connector, and because two boot programs - // is the smallest blast radius for a test whose whole subject is - // making init misbehave. - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/netcase"); - let bins: Vec<(String, Vec)> = rust_bins - .iter() - .filter(|(name, _)| name == "launcher_refusals") - .cloned() - .collect(); - if bins.is_empty() { - return Err("launcher_refusals was not built".to_string()); - } - let mut qemu = QemuInstance::boot_with_options( - &config, - &[], - &bins, - BootOptions { - profile: qemu::Profile::Headless, - // The live-object count is a `SYS_DEBUG` action, and a - // shipping kernel has none: both readings would be the same - // `InvalidArgument` and the leak arm would pass having - // counted nothing. - kernel_features: ACTUATOR_KERNEL, - ..Default::default() - }, - ); - let mut console = qemu.boot_log().to_string(); - let _ = await_marker(&mut qemu, &mut console, "===READY===", "test-runner to come up"); - - let result = qemu.run_test("test_rs_launcher_refusals", Duration::from_secs(120)); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}", result.stdout)); - } - if result.exit_code != Some(0) { - return Err(format!( - "launcher_refusals exited {:?}:\n{}", - result.exit_code, result.stdout - )); - } - console.push_str(&result.serial); - if !console.contains("init: launcher: cannot start") { - return Err(format!( - "init refused a launch without a line saying so — a launcher that \ - drops requests silently cannot be asked what happened:\n{console}" - )); - } - serial::Serial::named("boot console", console.as_str()).must_be_clean()?; - eprintln!(" [netcase] init refused three bad launches, named them, and kept launching"); - Ok(()) - } "spawn_cwd" => { // A child starts in the directory its spawn names — through the // launcher from the shell's `cd`, through the launcher from @@ -18390,8 +17816,7 @@ enum Poke { /// The right button, which no sequence driving that client produces for any /// other reason, and the release rather than the press so the pointer is left /// with nothing held. Every caller owes it one: without it the client waits out -/// its liveness ceiling, and `xhci_hid_break`, `xhci_hotplug` and `xhci_flap` -/// each paid 30 s for the omission. +/// its liveness ceiling. pub(crate) fn input_events_end(input: &mut qemu::QmpInput) { input.mouse(0, 0, Some(("right", true))); input.mouse(0, 0, Some(("right", false))); @@ -18866,7 +18291,7 @@ fn stall_is_not_a_verdict() -> Result<(), String> { /// /// **The failure mode the tier introduces is silence, not a wrong answer.** A /// green run holding back 60 tests and a green run holding back none print the -/// same word, and the difference between them is the whole reason `--nightly` +/// same word, and the difference between them is the whole reason a reach flag /// exists. Nothing else can see whether the *run* mentions it, and a run nobody /// can tell apart from a full one is how a temporary measure becomes permanent. /// @@ -18874,21 +18299,29 @@ fn stall_is_not_a_verdict() -> Result<(), String> { /// that ran everything must not claim to have held anything back either, or the /// line stops carrying information the day somebody makes it unconditional. fn nightly_tier_is_announced() -> Result<(), String> { - let held: [&str; 2] = ["desktop_window_child", "sshd_fail_closed"]; - let announced = - Tally::new().holding_back(&held).summary(1, Duration::ZERO, Duration::ZERO); + let held = vec![ + (Tier::Nightly, vec!["desktop_window_child".to_string(), "sshd_exec".to_string()]), + (Tier::Weekly, vec!["iommu_empty_domain".to_string()]), + ]; + let announced = Tally::new().holding_back(held).summary(1, Duration::ZERO, Duration::ZERO); for want in [ - "not run — the nightly tier", - "desktop_window_child, sshd_fail_closed", + "not run without --nightly:", + "desktop_window_child, sshd_exec", "`cargo test --test toyos-build -- --nightly` runs them", - "2 held back for the nightly tier", + "not run without --weekly:", + "`cargo test --test toyos-build -- --weekly` runs them", + "2 held back for --nightly, 1 held back for --weekly", ] { if !announced.contains(want) { return Err(format!("a run holding tests back never says {want:?}:\n{announced}")); } } - let whole = Tally::new().summary(1, Duration::ZERO, Duration::ZERO); - if whole.contains("nightly") || whole.contains("held back") { + let whole = Tally::new().holding_back(vec![(Tier::Weekly, Vec::new())]).summary( + 1, + Duration::ZERO, + Duration::ZERO, + ); + if whole.contains("--") || whole.contains("held back") { return Err(format!("a run that held nothing back says it did:\n{whole}")); } Ok(()) @@ -18945,11 +18378,11 @@ struct Tally { /// that never got the guest going has measured the host and not the tree. stalls: Vec, invalid: Vec<(String, Duration)>, - /// What the tier held back, by name. Not a verdict and never red — it is the - /// one thing a reader of the last line cannot infer from anything else in - /// it, because a run that skipped a third of its cost looks exactly like a - /// run that had nothing to do. - relegated: Vec<&'static str>, + /// What each tier held back, under the flag that runs it. Not a verdict and + /// never red — it is the one thing a reader of the last line cannot infer + /// from anything else in it, because a run that skipped a third of its cost + /// looks exactly like a run that had nothing to do. + relegated: Vec<(&'static str, Vec)>, } impl Tally { @@ -18964,8 +18397,12 @@ impl Tally { } /// The names this run's tier filter took out, for the summary to say so. - fn holding_back(mut self, names: &[&'static str]) -> Self { - self.relegated = names.to_vec(); + fn holding_back(mut self, held: Vec<(Tier, Vec)>) -> Self { + self.relegated = held + .into_iter() + .filter(|(_, names)| !names.is_empty()) + .map(|(tier, names)| (tier.flag().expect("only a flag's tier is held back"), names)) + .collect(); self } @@ -19060,21 +18497,21 @@ impl Tally { // Above the result line and not below it, so the pointer is the last // thing before the verdict rather than an afterthought under it. - if !self.relegated.is_empty() { - say("not run — the nightly tier:".to_string()); - say(format!(" {}", self.relegated.join(", "))); - say(" `cargo test --test toyos-build -- --nightly` runs them.".to_string()); + for (flag, names) in &self.relegated { + say(format!("not run without {flag}:")); + say(format!(" {}", names.join(", "))); + say(format!(" `cargo test --test toyos-build -- {flag}` runs them.")); say(String::new()); } // **In the result line, because that is the line a shard's job summary // extracts and the line anybody reads.** A count of what ran means // something different depending on how much was not attempted. - let held = if self.relegated.is_empty() { - String::new() - } else { - format!(", {} held back for the nightly tier", self.relegated.len()) - }; + let held: String = self + .relegated + .iter() + .map(|(flag, names)| format!(", {} held back for {flag}", names.len())) + .collect(); match self.exit_code() { 1 => say(format!( "test result: FAILED. {} passed, {} failed, {} invalidated, \ @@ -19333,7 +18770,7 @@ fn suite_split() -> Result<(), String> { each entry — coverage of the binary an image ships is what it costs." )); } - // **The other shape `check_no_collisions` cannot see**: a binary a machine + // **The other shape [`schedule`] cannot see**: a binary a machine // test drives under a *different* name is still discovered here, still runs // on the shared boot, and there passes on its exit code with nothing staged // for it to act on. @@ -19739,9 +19176,7 @@ fn save_durations(mut known: BTreeMap, timed: &[(String, Durat /// A phase's wall clock is `max(sum / width, longest job)`, and FIFO reaches the /// first term only if no long job is dispatched late. Declaration order puts the /// feature-carrying tests last — deliberately, to keep the kernel rebuilds -/// together — which is exactly the worst order for a wide phase: `xhci_hid_break` -/// and `xhci_deaf_registers` are two of the three longest jobs in the suite and -/// both sit in the last quarter of `MACHINE_TESTS`. +/// together — which is exactly the worst order for a wide phase. /// /// **The profile is measured, not declared**, because the alternative is a /// hand-maintained list of long tests — a second registration to keep true, and @@ -19908,9 +19343,9 @@ fn build_tasks<'a>( /// the width CI actually runs. fn check_shard_partition(all_tests: &[TestDef]) { let pricing = shard_pricing(); - for &nightly in &[false, true] { + for reach in [Reach::Fast, Reach::Nightly, Reach::Weekly] { // Sharded, because every run this partition is for is one. - let in_tier = |tier: Tier| tier.selected(nightly, true); + let in_tier = |tier: Tier| tier.selected(reach, true); let tests_to_run: Vec<&TestDef> = all_tests.iter().filter(|_| in_tier(SHARED_TIER)).collect(); let machine_to_run: Vec<(&str, Sched)> = MACHINE_TESTS @@ -19961,28 +19396,28 @@ fn check_shard_partition(all_tests: &[TestDef]) { for name in mine_p.iter().chain(&mine_s).flat_map(Task::names) { assert!( seen.insert(name.to_string()), - "nightly={nightly}: {name} lands in shard {index}/{COUNT} and at least \ + "{reach:?}: {name} lands in shard {index}/{COUNT} and at least \ one earlier shard too — every execution label must belong to exactly one" ); } for name in mine_a { assert!( audio_seen.insert(name.to_string()), - "nightly={nightly}: audio config {name} lands in shard {index}/{COUNT} \ + "{reach:?}: audio config {name} lands in shard {index}/{COUNT} \ and at least one earlier shard too" ); } } assert_eq!( seen, want, - "nightly={nightly}: the twelve shards together do not equal the full selection — \ + "{reach:?}: the twelve shards together do not equal the full selection — \ {:?} present in the selection and missing from every shard", want.difference(&seen).collect::>() ); let want_audio: BTreeSet = audio_names.iter().map(|s| s.to_string()).collect(); assert_eq!( audio_seen, want_audio, - "nightly={nightly}: the twelve shards' audio configs do not equal the full \ + "{reach:?}: the twelve shards' audio configs do not equal the full \ selection" ); } @@ -20216,16 +19651,6 @@ fn check_registration() { "CARRIES has a row for {test}, which no MACHINE_TESTS or SCREEN_TESTS entry registers" ); } - let mut seen: BTreeSet<&str> = BTreeSet::new(); - for name in MACHINE_TESTS - .iter() - .chain(SCREEN_TESTS) - .map(|(n, _, _)| *n) - .chain(AUDIO_TESTS.iter().map(|(n, _)| *n)) - { - assert!(seen.insert(name), "{name} is registered twice"); - } - let mut groups: BTreeMap<&str, (usize, usize, usize)> = BTreeMap::new(); for (i, (name, _, _)) in MACHINE_TESTS.iter().enumerate() { let Some(group) = group_of(name) else { continue }; @@ -20253,48 +19678,30 @@ fn check_registration() { } } -/// The half [`check_registration`] could not ask: the shared boot's tests are -/// *discovered* from the binaries in `tests/toyos-rust-tests` and `tests/c`, so -/// nothing declared can be compared against them until they exist. -fn check_no_collisions(shared: &[TestDef]) { - let mut shared_seen = BTreeSet::new(); - let shared_twice: Vec<&str> = shared - .iter() - .map(|test| test.name.as_str()) - .filter(|name| !shared_seen.insert(*name)) - .collect(); - assert!( - shared_twice.is_empty(), - "{shared_twice:?} name two binaries on the shared boot; every verdict and duration \ - label must identify exactly one execution" - ); - let declared: BTreeSet<&str> = MACHINE_TESTS - .iter() - .map(|(n, _, _)| *n) - .chain(SCREEN_TESTS.iter().map(|(n, _, _)| *n)) - .chain(AUDIO_TESTS.iter().map(|(name, _)| *name)) - .collect(); - let clash: Vec<&str> = - shared.iter().map(|t| t.name.as_str()).filter(|n| declared.contains(n)).collect(); - assert!( - clash.is_empty(), - "{clash:?} name both a binary on the shared boot and a test that declares its own \ - machine — two verdicts under one name. Add each to RUST_SKIP with the reason its \ - own test exists, or rename one of the two." - ); +/// Every name this suite can produce a verdict for, at its one tier: the shared +/// boot's discovered binaries and the three declared registries, so a name two +/// of them give is refused before anything boots. +fn schedule(shared: &[TestDef]) -> Schedule<'_> { + Schedule::new( + shared + .iter() + .map(|t| (t.name.as_str(), SHARED_TIER)) + .chain(AUDIO_TESTS.iter().copied()) + .chain(SCREEN_TESTS.iter().chain(MACHINE_TESTS).map(|(n, _, tier)| (*n, *tier))), + ) + .unwrap_or_else(|refusal| { + panic!( + "{refusal}. A binary a machine test drives goes on RUST_SKIP with the reason its own \ + test exists, or one of the two is renamed." + ) + }) } -/// `redlist::DISABLED` against every name `all_tests` plus the three declared -/// registries could produce a verdict for, before any boot on any entry point. -fn check_redlist(all_tests: &[TestDef]) -> Result<(), String> { - let runnable: BTreeSet<&str> = all_tests - .iter() - .map(|t| t.name.as_str()) - .chain(AUDIO_TESTS.iter().map(|(name, _)| *name)) - .chain(SCREEN_TESTS.iter().map(|(n, _, _)| *n)) - .chain(MACHINE_TESTS.iter().map(|(n, _, _)| *n)) - .collect(); - redlist::check(redlist::DISABLED, |name| runnable.contains(name), &compile::repo_root()) +/// `redlist::DISABLED` against every name the schedule holds, before any boot +/// on any entry point. +fn check_redlist(schedule: &Schedule<'_>) -> Result<(), String> { + redlist::check(redlist::DISABLED, |name| schedule.tier(name).is_ok(), &compile::repo_root()) + } fn main() { @@ -20321,10 +19728,11 @@ fn main() { let debug_mode = SUITE.present(&args, &testargs::DEBUG); let list_mode = SUITE.present(&args, &testargs::LIST); - // The nightly tier, on. A flag and not an env var for `--audio-gate`'s - // reason: an env var is invisible in the command line and easy to leave set, - // and the whole point of the split is that a run says what it ran. - let nightly = SUITE.present(&args, &testargs::NIGHTLY); + // How far down the tiers this run reaches. Flags and not an env var for + // `--audio-gate`'s reason: an env var is invisible in the command line and + // easy to leave set, and the whole point of the split is that a run says what + // it ran. + let reach = Reach::of(&args); // The metal profile, and where its images and readbacks live. Naming the // directory means the machine is not touched — see `common::metal::Mode`. let metal_mode = SUITE.present(&args, &testargs::METAL); @@ -20422,7 +19830,7 @@ fn main() { let c_bins = compile_c_tests(&c_names); check_metal_only_unshared(&rust_bins, &c_bins); let c_compiled: Vec = c_bins.iter().map(|(n, _)| n.clone()).collect(); - if let Err(refusal) = check_redlist(&build_test_registry(&rust_bins, &c_compiled)) { + if let Err(refusal) = check_redlist(&schedule(&build_test_registry(&rust_bins, &c_compiled))) { eprintln!("[toyos] src/redlist.rs: {refusal}"); run.exit(1); } @@ -20482,24 +19890,16 @@ fn main() { // Every name this process could produce a verdict for, before `--list`, // `--debug` and `--audio-gate` can return without ever reaching it. let all_tests = build_test_registry(&rust_bins, &c_compiled); - if let Err(refusal) = check_redlist(&all_tests) { + let schedule = schedule(&all_tests); + if let Err(refusal) = check_redlist(&schedule) { eprintln!("[toyos] src/redlist.rs: {refusal}"); run.exit(1); } - // --list: print test names and exit + // --list: every test at its tier, and exit if list_mode { - for t in &all_tests { - println!("{}", t.name); - } - for (name, _) in AUDIO_TESTS { - println!("{name}"); - } - for (name, _, _) in SCREEN_TESTS { - println!("{name}"); - } - for (name, _, _) in MACHINE_TESTS { - println!("{name}"); + for (name, tier) in schedule.iter() { + println!("{tier:?} {name}"); } return; } @@ -20553,7 +19953,6 @@ fn main() { return; } - check_no_collisions(&all_tests); check_metal_only_unshared(&rust_bins, &c_bins); // Every row against the catalogue before any boot, so a name the suite did // not build is refused here and not by whichever worker reaches it. @@ -20567,7 +19966,7 @@ fn main() { // nobody remembers. `cargo test -- screen_diag_boot` refuses below and // says what to type instead, which is the same information a silent skip // would have withheld. - let in_tier = |tier: Tier| tier.selected(nightly, shard.is_some()); + let in_tier = |tier: Tier| tier.selected(reach, shard.is_some()); let tests_to_run: Vec<&TestDef> = all_tests .iter() .filter(|t| keep(t.name.as_str()) && in_tier(SHARED_TIER)) @@ -20593,21 +19992,13 @@ fn main() { // introduces, so the names are printed rather than counted, and the line // carries both the command that runs them and the record that says what each // one guarded. - let held = |which: Tier| -> Vec<&str> { - MACHINE_TESTS + let held = |which: Tier| -> Vec { + schedule .iter() - .chain(SCREEN_TESTS) - .filter(|(n, _, tier)| keep(n) && *tier == which && !in_tier(*tier)) - .map(|(n, _, _)| *n) - .chain( - AUDIO_TESTS - .iter() - .filter(|(name, tier)| keep(name) && *tier == which && !in_tier(*tier)) - .map(|(name, _)| *name), - ) + .filter(|(n, tier)| keep(n) && *tier == which && !in_tier(*tier)) + .map(|(n, _)| n.to_string()) .collect() }; - let held_back = held(Tier::Nightly); let held_local = held(Tier::Local); if !held_local.is_empty() { eprintln!( @@ -20617,14 +20008,18 @@ fn main() { ); eprintln!("[toyos] {}", held_local.join(", ")); } - if !held_back.is_empty() { + let held_back: Vec<(Tier, Vec)> = + [Tier::Nightly, Tier::Weekly].into_iter().map(|tier| (tier, held(tier))).collect(); + let widest_held = held_back.iter().rev().find(|(_, names)| !names.is_empty()).map(|(tier, _)| *tier); + for (tier, names) in held_back.iter().filter(|(_, names)| !names.is_empty()) { + let flag = tier.flag().expect("a held tier is a reach's"); eprintln!( - "[toyos] nightly tier: {} test(s) NOT run. \ - `cargo test --test toyos-build -- --nightly` runs them manually; \ - .github/workflows/nightly.yml runs them every night at 03:00 UTC.", - held_back.len(), + "[toyos] {} test(s) NOT run without {flag}, which \ + `cargo test --test toyos-build -- {flag}` and .github/workflows/nightly.yml's \ + schedule for it run.", + names.len(), ); - eprintln!("[toyos] {}", held_back.join(", ")); + eprintln!("[toyos] {}", names.join(", ")); } if tests_to_run.is_empty() @@ -20632,19 +20027,19 @@ fn main() { && screen_to_run.is_empty() && machine_to_run.is_empty() { - if !held_back.is_empty() { - eprintln!( - "[toyos] filter {filter:?} matches only tests in the nightly tier. Add \ - --nightly to run them." - ); - } else { - eprintln!("No enabled test matches filter {filter:?}"); + match widest_held.and_then(Tier::flag) { + Some(flag) => eprintln!( + "[toyos] filter {filter:?} matches only tests a wider reach runs. Add {flag} to \ + run them." + ), + None => eprintln!("No enabled test matches filter {filter:?}"), } run.exit(1); } let test_config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/testcases"); - let mut tally = Tally::new().holding_back(&held_back); + let mut tally = Tally::new().holding_back(held_back); + let suite_start = common::clock::mark(); let bins = Bins { From cb43477db94a5f4473809e7e1434ef9847927c0b Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:06:52 +0200 Subject: [PATCH 02/18] host tests for two guest catches: a join keeps its answer, a dump waits for a bare pass Two defects only a guest caught, now caught on the host: - B23, #513's `thread_join` answering NotFound: the wait's predicate collected the zombie and the syscall collected again. The kept answer moves into `toyos_proclife::join::Join`, which asks the table until one ask settles and returns that ask after it; `sys_thread_join` asks it under the table lock through `process::ask_join`, which replaces `wait_thread_zombie`. `a_join_asked_after_it_collected_keeps_its_answer` reds on the re-collecting shape. - B09, the blocked-task dump served on a syscall-driven pass: `Entered` and its decision, `may_serve`, move into `sched/dump_request.rs`, the file `kernel-loom` already compiles, and `dump.rs` mints its proof from it. `only_a_pass_entered_at_depth_zero_takes_the_request` reds on a pass entered above zero taking the request. Co-Authored-By: Claude Opus 5.5 --- kernel-loom/tests/dump_request.rs | 14 ++++++++++- kernel/src/process.rs | 8 +++---- kernel/src/sched/dump.rs | 21 ++-------------- kernel/src/sched/dump_request.rs | 27 +++++++++++++++++++++ kernel/src/syscall/proc.rs | 33 ++++++++++--------------- toyos-proclife/src/join.rs | 40 +++++++++++++++++++++++++++++++ 6 files changed, 98 insertions(+), 45 deletions(-) diff --git a/kernel-loom/tests/dump_request.rs b/kernel-loom/tests/dump_request.rs index dd7584ea5a1..29ebab527f7 100644 --- a/kernel-loom/tests/dump_request.rs +++ b/kernel-loom/tests/dump_request.rs @@ -13,7 +13,7 @@ //! ``` #![cfg(feature = "loom")] -use kernel_loom::dump_request::{DumpRequest, Left}; +use kernel_loom::dump_request::{DumpRequest, Entered, Left}; use loom::cell::UnsafeCell; use loom::sync::Arc; @@ -186,3 +186,15 @@ fn a_request_left_during_a_report_is_still_taken_by_its_end() { assert_eq!(machine.reports(), 2); }); } + +/// Only a pass entered at depth zero takes the request. A syscall's pass — `yield_now`, `exit_current` — +/// is entered above zero and a blocking pass is inside a wait ticket, and `dump::request` asserts it runs +/// with neither under it: a report served from one of them is a kernel panic. Not an interleaving question, +/// so no model. +#[test] +fn only_a_pass_entered_at_depth_zero_takes_the_request() { + assert!(Entered::Pass { depth: 0 }.may_serve()); + for entered in [Entered::Blocking, Entered::Pass { depth: 1 }, Entered::Pass { depth: 2 }] { + assert!(!entered.may_serve(), "{entered} took the request, and its report runs above a bare pass"); + } +} diff --git a/kernel/src/process.rs b/kernel/src/process.rs index b89fa88ae70..8a91c941c2e 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -1298,12 +1298,10 @@ pub fn wake_pipe_writers(pipe_id: pipe::PipeId) { scheduler::wake_pipe_writers(pipe_id); } -/// Atomically validate the parent-thread relationship and collect a zombie thread; the table lock is the atomicity. `Err(())`: this caller may not join `tid`/`pid`. -pub fn wait_thread_zombie(tid: Tid, parent_pid: Pid) -> Result, ()> { +/// Ask `join` once more under the table lock, which is what makes validating the parent-thread relationship and collecting the zombie one act. +pub fn ask_join(join: &mut join::Join, tid: Tid, parent_pid: Pid) -> Option> { let mut guard = PROCESS_TABLE.lock(); - let table = guard.as_mut().unwrap(); - // Both refusals are one answer here (`SyscallError::NotFound`); `JoinRefused` keeps them apart because they aren't the same fact. - join::collect_zombie(table, parent_pid, tid).map_err(|_| ()) + join.ask(guard.as_mut().unwrap(), parent_pid, tid) } /// Handle a page fault at `fault_addr` by looking up the current process's VMAs. Returns whether the fault was resolved. diff --git a/kernel/src/sched/dump.rs b/kernel/src/sched/dump.rs index 10cdc57dab1..051612c3742 100644 --- a/kernel/src/sched/dump.rs +++ b/kernel/src/sched/dump.rs @@ -15,6 +15,7 @@ use core::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +pub use super::dump_request::Entered; use super::dump_request::{DumpRequest, Left}; use crate::arch::{irqchip, percpu, smp}; @@ -142,27 +143,9 @@ fn online_cpus() -> usize { (smp::cpu_count() as usize).min(MAX_CPUS) } -/// How the pass that met a request was entered: the whole of what decides whether it may serve. -#[derive(Clone, Copy)] -pub enum Entered { - /// `driver::pass`, by the preempt depth it was entered at, before it raised its own level. - Pass { depth: u32 }, - /// `driver::pass_block`, inside a wait ticket's registration window at every depth. - Blocking, -} - impl Entered { fn under_nothing(self) -> Option { - matches!(self, Self::Pass { depth: 0 }).then_some(UnderNothing(())) - } -} - -impl core::fmt::Display for Entered { - fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::Blocking => write!(f, "a blocking pass"), - Self::Pass { depth } => write!(f, "a pass entered at preempt depth {depth}"), - } + self.may_serve().then_some(UnderNothing(())) } } diff --git a/kernel/src/sched/dump_request.rs b/kernel/src/sched/dump_request.rs index 02ec98d156c..a7538136c38 100644 --- a/kernel/src/sched/dump_request.rs +++ b/kernel/src/sched/dump_request.rs @@ -1,6 +1,7 @@ //! Ctrl+Alt+D's request and the report it asks for, as one word: what is pending, and whether a report runs. //! Every transition is one atomic update of that word, so a request is announced once and taken once, no //! report begins inside another, and a request filed during a report is taken by that report's end. +//! [`Entered`] is which pass may take a request. //! No `crate::` references: `kernel-loom` compiles this file directly under `feature = "loom"`. #[cfg(not(feature = "loom"))] @@ -115,3 +116,29 @@ impl DumpRequest { was & PENDING != NONE } } + +/// How the pass that met a request was entered: the whole of what decides whether it may serve. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Entered { + /// `driver::pass`, by the preempt depth it was entered at, before it raised its own level. + Pass { depth: u32 }, + /// `driver::pass_block`, inside a wait ticket's registration window at every depth. + Blocking, +} + +impl Entered { + /// Only a pass entered at depth zero has nothing under it; a syscall's pass has its trap frame and a + /// blocking pass its wait ticket, and a report served inside either panics on its own depth assertion. + pub fn may_serve(self) -> bool { + matches!(self, Self::Pass { depth: 0 }) + } +} + +impl core::fmt::Display for Entered { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Blocking => write!(f, "a blocking pass"), + Self::Pass { depth } => write!(f, "a pass entered at preempt depth {depth}"), + } + } +} diff --git a/kernel/src/syscall/proc.rs b/kernel/src/syscall/proc.rs index 1fc6adcd132..0e80649e38b 100644 --- a/kernel/src/syscall/proc.rs +++ b/kernel/src/syscall/proc.rs @@ -153,26 +153,20 @@ pub(super) fn sys_thread_join(tid: u64) -> u64 { // None means never existed or already collected; the predicate below answers both. let target = process::thread_sched(caller, tid); let parkable = crate::scheduler::Parkable::at_entry(); - // Collecting takes the zombie out of the table, so the first answer is kept: asked - // again, the same join finds no such thread. - let answer = core::cell::Cell::new(None); - let settled = || { - answer.get().is_some() - || match process::wait_thread_zombie(tid, caller) { - Ok(None) => false, - Ok(Some(_)) => { - answer.set(Some(0)); - true - } - Err(()) => { - answer.set(Some(SyscallError::NotFound.to_u64())); - true - } - } + let join = core::cell::Cell::new(toyos_proclife::join::Join::default()); + let ask = || { + let mut asked = join.get(); + let answer = process::ask_join(&mut asked, tid, caller); + join.set(asked); + answer }; - while !settled() { + loop { + // Both refusals are one answer here; `JoinRefused` keeps them apart because they aren't the same fact. + if let Some(answer) = ask() { + return answer.map_or(SyscallError::NotFound.to_u64(), |_| 0); + } let Some(sched) = target.as_ref() else { - // Nothing to arm on and no zombie: wait_thread_zombie will never answer differently. + // Nothing to arm on and no zombie: the join will never answer differently. return SyscallError::NotFound.to_u64(); }; // Arms on the target thread's own watch, not a wake-by-name to the main thread. @@ -182,14 +176,13 @@ pub(super) fn sys_thread_join(tid: u64) -> u64 { tid.raw() as u64, WaitClass::Other, Deadline::never(), - settled, + || ask().is_some(), ) .is_err() { return cancelled(); } } - answer.get().expect("a settled join holds its answer") } pub(super) fn sys_nanosleep(nanos: u64) -> u64 { diff --git a/toyos-proclife/src/join.rs b/toyos-proclife/src/join.rs index 666055ba7e7..5505b4bed02 100644 --- a/toyos-proclife/src/join.rs +++ b/toyos-proclife/src/join.rs @@ -48,6 +48,25 @@ pub fn collect_zombie( } } +/// One join's answer, kept from the ask that settled it. +/// +/// Collecting takes the zombie out of the table, so an ask after the one that +/// collected finds [`JoinRefused::NoSuchThread`]: a join whose wait asks again +/// after every wake answers with its first settled ask, never its last. +#[derive(Clone, Copy, Default, PartialEq, Eq, Debug)] +pub struct Join(Option>); + +impl Join { + /// Ask the table, unless an earlier ask settled; `None` is still waiting. + #[must_use = "a settled join is the answer the syscall returns"] + pub fn ask(&mut self, table: &mut T, pid: Pid, tid: Tid) -> Option> { + if self.0.is_none() { + self.0 = collect_zombie(table, pid, tid).transpose(); + } + self.0 + } +} + #[cfg(test)] mod tests { use super::*; @@ -73,6 +92,27 @@ mod tests { assert_eq!(collect_zombie(&mut world, pid, t1), Err(JoinRefused::NoSuchThread)); } + /// The wait's predicate collects the zombie, and the syscall asks again once + /// the wait returns: the second ask is the answer the caller gets. + #[test] + fn a_join_asked_after_it_collected_keeps_its_answer() { + let mut world = World::new(); + let pid = world.spawn_process(); + let t1 = world.spawn_thread(pid); + let mut join = Join::default(); + assert_eq!(join.ask(&mut world, pid, t1), None); + world.set_location(pid, t1, ThreadLocation::Zombie(11)); + assert_eq!(join.ask(&mut world, pid, t1), Some(Ok(11))); + assert_eq!( + join.ask(&mut world, pid, t1), + Some(Ok(11)), + "a join that collected its thread answered the ask after it with no such thread", + ); + let mut refused = Join::default(); + assert_eq!(refused.ask(&mut world, pid, Tid(9)), Some(Err(JoinRefused::NoSuchThread))); + assert_eq!(refused.ask(&mut world, pid, Tid(9)), Some(Err(JoinRefused::NoSuchThread))); + } + #[test] fn the_two_refusals_are_told_apart() { let mut world = World::new(); From eb647912d1f0c453ad2f70c76bf301fac0231b90 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:08:00 +0200 Subject: [PATCH 03/18] schedule tests: each assertion names the reason it reds for Co-Authored-By: Claude Opus 5.5 --- src/ci.rs | 6 ++++-- src/testargs.rs | 3 ++- src/tiers.rs | 17 ++++++++++++----- 3 files changed, 18 insertions(+), 8 deletions(-) diff --git a/src/ci.rs b/src/ci.rs index 927020bd63e..a5a151e4fb5 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -945,8 +945,10 @@ mod tests { #[test] fn each_schedule_names_its_reach_and_another_is_refused() { - assert_eq!(reach_of_schedule(Some(NIGHTLY_CRON)), Ok("--nightly")); - assert_eq!(reach_of_schedule(Some(WEEKLY_CRON)), Ok("--weekly")); + let nightly = reach_of_schedule(Some(NIGHTLY_CRON)); + let weekly = reach_of_schedule(Some(WEEKLY_CRON)); + assert_eq!(nightly, Ok("--nightly"), "the nightly schedule's reach"); + assert_eq!(weekly, Ok("--weekly"), "the weekly schedule's reach"); for stray in [Some("0 4 * * *"), None] { let refusal = reach_of_schedule(stray).unwrap_err(); assert!(refusal.contains(&format!("{stray:?}")), "{refusal}"); diff --git a/src/testargs.rs b/src/testargs.rs index 9646a991136..6ee0ff4b44f 100644 --- a/src/testargs.rs +++ b/src/testargs.rs @@ -288,7 +288,8 @@ mod tests { #[test] fn the_weekly_reach_refuses_a_nightly_it_already_runs() { - let refusal = parse_owned(&["--nightly", "--weekly"]).unwrap_err(); + let refusal = parse_owned(&["--nightly", "--weekly"]) + .expect_err("a --nightly beside --weekly was accepted and read by nothing"); assert!(refusal.contains("read by nothing"), "{refusal}"); } diff --git a/src/tiers.rs b/src/tiers.rs index 9bbe674d3b1..b31b9c78ae8 100644 --- a/src/tiers.rs +++ b/src/tiers.rs @@ -142,7 +142,8 @@ mod tests { #[test] fn an_unregistered_test_is_refused_by_name() { let schedule = Schedule::new([("registered", Tier::Weekly)]).unwrap(); - let refusal = schedule.tier("never_registered").unwrap_err(); + let refusal = + schedule.tier("never_registered").expect_err("an unregistered name was given a tier"); assert!(refusal.starts_with("never_registered is not a registered test"), "{refusal}"); assert!(schedule.tier("registere").is_err(), "a prefix is not the name"); } @@ -154,9 +155,14 @@ mod tests { let selected = |reach| -> Vec { EVERY.into_iter().filter(|tier| tier.selected(reach, true)).collect() }; - assert_eq!(selected(Reach::Fast), [Tier::Fast]); - assert_eq!(selected(Reach::Nightly), [Tier::Fast, Tier::Nightly]); - assert_eq!(selected(Reach::Weekly), [Tier::Fast, Tier::Nightly, Tier::Weekly]); + let nested = [ + (Reach::Fast, &[Tier::Fast][..]), + (Reach::Nightly, &[Tier::Fast, Tier::Nightly]), + (Reach::Weekly, &[Tier::Fast, Tier::Nightly, Tier::Weekly]), + ]; + for (reach, tiers) in nested { + assert_eq!(selected(reach), tiers, "a {reach:?} run selects the wrong tiers"); + } for reach in [Reach::Fast, Reach::Nightly, Reach::Weekly] { assert!(Tier::Local.selected(reach, false) && !Tier::Local.selected(reach, true)); } @@ -167,7 +173,8 @@ mod tests { fn a_held_tiers_flag_selects_it() { for tier in EVERY { let Some(flag) = tier.flag() else { - assert!(tier.selected(Reach::Fast, false), "{tier:?} names no flag, so a plain run takes it"); + let plain = tier.selected(Reach::Fast, false); + assert!(plain, "{tier:?} names no flag, so a plain run takes it"); continue; }; let reach = Reach::of(&[flag.to_string()]); From c956cb4e4d2be5c6217e7cb72a30c4ed8a0dee1d Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:12:08 +0200 Subject: [PATCH 04/18] kernel-loom: the i8042 tally models explored one execution; they explore now MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The two models over `Tally` spawned the ISR, and `Tally::record` begins with its saturation check's load. Loom 0.7 explores only the first schedule when a spawned thread's first operation is a load: measured on a bare loom `AtomicU64`, two adds after a leading load gave 1 execution against 10 without it. So neither model checked an interleaving of the real word, and A16's shape — an empty interrupt counted as one that carried until the burst corrects it — passed both. Both now spawn the reader and run the ISR on the model's own thread, and `explored` refuses a model that ran one execution. Co-Authored-By: Claude Opus 5.5 --- kernel-loom/tests/i8042_tally.rs | 49 +++++++++++++++++++------------- 1 file changed, 30 insertions(+), 19 deletions(-) diff --git a/kernel-loom/tests/i8042_tally.rs b/kernel-loom/tests/i8042_tally.rs index b145d20e3d9..9318c089391 100644 --- a/kernel-loom/tests/i8042_tally.rs +++ b/kernel-loom/tests/i8042_tally.rs @@ -41,12 +41,26 @@ //! does; the guest suite can supply neither, because a rate is not falsified by //! a green boot. -use std::sync::atomic::{AtomicBool, Ordering as StdOrdering}; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as StdOrdering}; use kernel_loom::i8042_tally::{Carried, Counts, Tally}; use loom::sync::atomic::{AtomicU32, Ordering}; use loom::sync::Arc; +/// `loom::model`, refusing a model that explored one execution: loom 0.7 explores nothing but the +/// first schedule when the spawned thread's first operation is a load, which `Tally::record`'s +/// saturation check is, so every model here spawns the reader and runs the ISR on its own thread. +fn explored(f: impl Fn() + Sync + Send + 'static) { + let runs = std::sync::Arc::new(AtomicUsize::new(0)); + let counted = runs.clone(); + loom::model(move || { + counted.fetch_add(1, StdOrdering::Relaxed); + f(); + }); + let runs = runs.load(StdOrdering::Relaxed); + assert!(runs > 1, "loom explored {runs} execution, so no interleaving was checked"); +} + /// An interrupt the ISR found nothing behind is never counted as one that /// carried a byte — at the settled end, and at every instant on the way there. /// @@ -55,16 +69,16 @@ use loom::sync::Arc; /// is not synchronized against it by anything. #[test] fn an_empty_interrupt_is_never_counted_as_one_that_carried() { - loom::model(|| { + explored(|| { let tally = Arc::new(Tally::new()); - let isr = { + // The verdict's own question: "did anything arrive to decode?" + let reader = { let tally = tally.clone(); - loom::thread::spawn(move || tally.record(Carried::Nothing)) + loom::thread::spawn(move || tally.read()) }; - - // The verdict's own question: "did anything arrive to decode?" - let mid = tally.read(); + tally.record(Carried::Nothing); + let mid = reader.join().unwrap(); assert_eq!( mid.carried, 0, "a reader saw {mid:?} while the only interrupt on the machine carried nothing — \ @@ -72,8 +86,6 @@ fn an_empty_interrupt_is_never_counted_as_one_that_carried() { ); // And the total it prints alongside can never exceed what happened. assert!(mid.irqs() <= 1, "a reader saw {mid:?}, which is more interrupts than were taken"); - - isr.join().unwrap(); assert_eq!( tally.read(), Counts { carried: 0, empty: 1 }, @@ -102,30 +114,29 @@ fn an_empty_interrupt_is_never_counted_as_one_that_carried() { /// ask the question either. #[test] fn a_counted_interrupt_carries_its_bytes_with_it() { - loom::model(|| { + explored(|| { let tally = Arc::new(Tally::new()); let published = Arc::new(AtomicU32::new(0)); - let isr = { + let reader = { let (tally, published) = (tally.clone(), published.clone()); loom::thread::spawn(move || { - // Everything the interrupt did, before it says it happened. - published.store(1, Ordering::Relaxed); - tally.record(Carried::Bytes); + let seen = tally.read(); + (seen, published.load(Ordering::Relaxed)) }) }; + // Everything the interrupt did, before it says it happened. + published.store(1, Ordering::Relaxed); + tally.record(Carried::Bytes); - let seen = tally.read(); + let (seen, behind) = reader.join().unwrap(); if seen.carried > 0 { assert_eq!( - published.load(Ordering::Relaxed), - 1, + behind, 1, "a reader counted an interrupt as having delivered a byte and could not see the \ byte: the report would say `0 bytes` about one that had arrived", ); } - - isr.join().unwrap(); assert_eq!( tally.read(), Counts { carried: 1, empty: 0 }, From fae90977ccc563ead03e8a10cf3d7221b5db9d3c Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:12:46 +0200 Subject: [PATCH 05/18] kernel-loom: the tally model gates the release/acquire pair; the claim that it could not goes With the models exploring, `a_counted_interrupt_carries_its_bytes_with_it` reds with `record`'s release and `read`'s acquire both weakened to `Relaxed`: the measurement that said loom could not see the weakening was taken on a model that ran one execution. Co-Authored-By: Claude Opus 5.5 --- kernel-loom/tests/i8042_tally.rs | 11 ++--------- 1 file changed, 2 insertions(+), 9 deletions(-) diff --git a/kernel-loom/tests/i8042_tally.rs b/kernel-loom/tests/i8042_tally.rs index 9318c089391..d340aeecf75 100644 --- a/kernel-loom/tests/i8042_tally.rs +++ b/kernel-loom/tests/i8042_tally.rs @@ -103,15 +103,8 @@ fn an_empty_interrupt_is_never_counted_as_one_that_carried() { /// so a count visible ahead of its own evidence is a report on a ring that looks /// empty. **This is the direction that reds on the old shape**: with the two /// counters back in `tally.rs` it failed here too, `published` still 0 under a -/// count that already said a byte had arrived. -/// -/// **It is not a gate on the release/acquire pair, and saying so is the point.** -/// Measured: with `record`'s release and `read`'s acquire both weakened to -/// `Relaxed`, all three models still passed. Loom 0.7 does not distinguish the -/// weakening — which is why this crate's other negative control removes `SeqCst` -/// *fences* rather than weakening an ordering. The pair rests on the argument in -/// `tally.rs`, and on x86 it is the same instruction either way, so no guest can -/// ask the question either. +/// count that already said a byte had arrived. It reds too with `record`'s +/// release and `read`'s acquire both weakened to `Relaxed`. #[test] fn a_counted_interrupt_carries_its_bytes_with_it() { explored(|| { From 0fc0f5a88eab8dbfc202f34afa66a17d1ba3926f Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:12:54 +0200 Subject: [PATCH 06/18] kernel-loom: the tally test states what it is, not a measurement of it Co-Authored-By: Claude Opus 5.5 --- kernel-loom/tests/i8042_tally.rs | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/kernel-loom/tests/i8042_tally.rs b/kernel-loom/tests/i8042_tally.rs index d340aeecf75..3d3500deca9 100644 --- a/kernel-loom/tests/i8042_tally.rs +++ b/kernel-loom/tests/i8042_tally.rs @@ -103,8 +103,7 @@ fn an_empty_interrupt_is_never_counted_as_one_that_carried() { /// so a count visible ahead of its own evidence is a report on a ring that looks /// empty. **This is the direction that reds on the old shape**: with the two /// counters back in `tally.rs` it failed here too, `published` still 0 under a -/// count that already said a byte had arrived. It reds too with `record`'s -/// release and `read`'s acquire both weakened to `Relaxed`. +/// count that already said a byte had arrived. #[test] fn a_counted_interrupt_carries_its_bytes_with_it() { explored(|| { From 5e64f5a41c290ad42299145c2ba476a8054ee3e4 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:13:14 +0200 Subject: [PATCH 07/18] issues: a loom model whose spawned thread opens with a load explores one execution Co-Authored-By: Claude Opus 5.5 --- ...pens-with-a-load-explores-one-execution.md | 25 +++++++++++++++++++ 1 file changed, 25 insertions(+) create mode 100644 issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md diff --git a/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md b/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md new file mode 100644 index 00000000000..d952174b253 --- /dev/null +++ b/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md @@ -0,0 +1,25 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# A loom model whose spawned thread opens with a load explores one execution + +Loom 0.7, as `kernel-loom` pins it, ran a model's closure **once** when the +spawned thread's first operation was a load: a bare loom `AtomicU64`, a +spawned thread doing two `fetch_add`s and the model's thread one `load`, +explored 10 executions; the same with a `Relaxed` load at the top of the +spawned thread explored 1. A reader spawned beside an adder on the model's own +thread explored 19. The probes were throwaway test files in `kernel-loom/tests/` +on `wt/toyos-schedule`, counting closure runs in a `std` atomic. + +`kernel-loom/tests/i8042_tally.rs` was this shape — `Tally::record` opens with +its saturation check's load — so its two models checked no interleaving of +the real word, and a mutation that counted an empty interrupt as a carrying +one passed them. They now spawn the reader and refuse a model that ran one +execution (`explored`). No other model in `kernel-loom/` or +`toyos-sched/loom` has been counted. + +**Exit**: every loom model in the tree shown to run more than one execution — +the same refusal `explored` makes, applied to each . From 21005ec85acbdfcea28c4926fbefb561c1777258 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:13:19 +0200 Subject: [PATCH 08/18] issues: a loom model whose spawned thread opens with a load explores one execution Co-Authored-By: Claude Opus 5.5 --- ...e-spawned-thread-opens-with-a-load-explores-one-execution.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md b/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md index d952174b253..d85c497287f 100644 --- a/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md +++ b/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md @@ -22,4 +22,4 @@ execution (`explored`). No other model in `kernel-loom/` or `toyos-sched/loom` has been counted. **Exit**: every loom model in the tree shown to run more than one execution — -the same refusal `explored` makes, applied to each . +the same refusal `explored` makes, applied to each. From 6f9a317fc373f4d95f6980873c7a2eaceb719ac4 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:20:14 +0200 Subject: [PATCH 09/18] kernel/CLAUDE.md: console_line_atomicity is deleted, so it gates nothing Co-Authored-By: Claude Opus 5.5 --- kernel/CLAUDE.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kernel/CLAUDE.md b/kernel/CLAUDE.md index 6fffea753de..eb0254cff8a 100644 --- a/kernel/CLAUDE.md +++ b/kernel/CLAUDE.md @@ -16,7 +16,7 @@ The module header at the site owns its subsystem — read it before changing a m - **A block-layer `BudgetExpired` is not-durable-yet and never a loss** — it is retried on a fresh budget above every lock; a flush that discards its pages on one splits a FAT mirror. - **A decision the process table makes lives in `toyos-proclife`, never in `process.rs`** — its defects are interleavings and that crate is the only machine that can enumerate one. - **A task holds at most one watch registration** — a standing registration across a loop must not call anything that registers again. A double registration panics only at attempt ≥ 2, so the contention depth is the coverage. -- **A console is per holder, minted at spawn** — the object *is* the line buffer; `console_line_atomicity` is the gate. +- **A console is per holder, minted at spawn** — the object *is* the line buffer. - **`ops::close` cancels a poll only for a source its object really ends** — `Watch::cancel_polls` answers every ring's poll on that watch; `ops::close_ends_polls` is where a new object kind answers. - **A page shared with userland is never reached through a Rust reference** — the words a protocol shares are `&AtomicU32` one at a time, everything else is a volatile copy of the whole value, and a page is laid out *before* it is mapped (`SharedMemObject::phys_before_mapping`). The kernel, `toyos-abi` and the SDK each hold one end of this rule. - **A device a process drives reaches only the memory its grants map** — the domain is the gate, not the descriptor; a device address outside a grant is a `DMA FAULT` record and never a crash, and `userdev_dma_fault` is the registration. From 088ed7fec20e634ba56c5bcca17a871aed7f5afc Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 07:36:02 +0200 Subject: [PATCH 10/18] tests: no boot names a deleted binary, and --list refuses one that does `writeback_durability`'s CARRIES row still named `test_rs_fat_backing_revoked`, which this branch deleted, so every run panicked in the catalogue check before its first boot. That check ran after `--list` returned, so the list was green over it; it now runs before `--list`, `--debug` and `--audio-gate`. The test runner's `CONSOLE_JOBS` held only `test_rs_console_line_atomicity`, also deleted, so it goes and every job's stdin is a pipe. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- tests/toyos.rs | 14 ++++++++------ userland/test-runner/src/main.rs | 15 ++------------- 2 files changed, 10 insertions(+), 19 deletions(-) diff --git a/tests/toyos.rs b/tests/toyos.rs index 4bc5eecf66e..eead7408676 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -1610,7 +1610,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("ftruncate_flush_race", &["test_rs_ftruncate_flush_race", "test_rs_fs_rename_durable"]), ("log_backing_read_error", &["test_rs_log_volume_reread"]), ("redirty_mid_flush", &["test_rs_redirty_mid_flush"]), - ("writeback_durability", &["test_rs_writeback_durability", "test_rs_fat_backing_revoked"]), + ("writeback_durability", &["test_rs_writeback_durability"]), ("kernel_log_file", &["test_rs_writeback_durability"]), ("double_fault_stack", &["test_rs_test_panic_child"]), ("idle_stack_guard", &["test_rs_test_panic_child"]), @@ -19896,6 +19896,13 @@ fn main() { run.exit(1); } + // Every row against the catalogue before `--list` and before any boot, so a + // name the suite did not build is refused here and not by whichever worker + // reaches it. + for (_, names) in CARRIES { + qemu::carrying(&c_bins, &rust_bins, names.iter().copied()); + } + // --list: every test at its tier, and exit if list_mode { for (name, tier) in schedule.iter() { @@ -19954,11 +19961,6 @@ fn main() { } check_metal_only_unshared(&rust_bins, &c_bins); - // Every row against the catalogue before any boot, so a name the suite did - // not build is refused here and not by whichever worker reaches it. - for (_, names) in CARRIES { - qemu::carrying(&c_bins, &rust_bins, names.iter().copied()); - } check_shard_partition(&all_tests); // The tier filter, and it is not conditional on the name filter: a rule with diff --git a/userland/test-runner/src/main.rs b/userland/test-runner/src/main.rs index 84af3af8c79..85407cdcdf0 100644 --- a/userland/test-runner/src/main.rs +++ b/userland/test-runner/src/main.rs @@ -33,13 +33,6 @@ const BUILTINS: &[(&str, fn(Option<&SysCap>) -> i32)] = &[ ("kbd-close", kbd_close::run), ]; -/// Jobs started holding this program's console as stdin rather than a pipe: -/// a job whose subject is the console object has no other way to hold one, its -/// stdout being this program's log ring. The kernel mints the job a console of its -/// own from this one, and a job that never reads it takes none of the serial -/// commands. -const CONSOLE_JOBS: &[&str] = &["test_rs_console_line_atomicity"]; - /// The job the runner is inside, and whether the list got through: written by /// the loop, read by the deadline watching it. static RUNNING: Mutex = Mutex::new(String::new()); @@ -278,13 +271,9 @@ fn run_one(name: &str, args: &[&str], cap: Option<&SysCap>) -> Ran { return Ran::Builtin(code); } - // Piped stdin so the child does not consume the serial commands, but for - // a job in `CONSOLE_JOBS`. + // Piped stdin so the child does not consume the serial commands. let mut command = Command::new(&path); - command.args(args); - if !CONSOLE_JOBS.contains(&name) { - command.stdin(Stdio::piped()); - } + command.args(args).stdin(Stdio::piped()); // **A refused dup is an answer and not a failure — but only one // refusal is.** `duplicate` needs `DUP` on the capability, which a // manifest grants by name, so `PermissionDenied` says this cap is one From e527bee719f7f81b0598d56ce8a4e7bfaafea830 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 07:43:20 +0200 Subject: [PATCH 11/18] issues: loom misses a store racing a load only when the storer loaded the word first The filed claim, that a spawned thread opening with a load explores one execution, is false: `dump_request`'s spawned threads open with `DumpRequest::update`'s load and explore 16 to 5777 executions. Loom 0.7.2 keeps one last access per atomic, and a thread's own load overwrites it, so a store is never raced against another thread's earlier load of the same word. The issue now says that, and lists every loom model whose spawned thread opens with a load with its measured execution count; `explored`'s doc in `i8042_tally.rs` says the same. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...pens-with-a-load-explores-one-execution.md | 25 ------- ...before-a-load-another-thread-made-first.md | 74 +++++++++++++++++++ kernel-loom/tests/i8042_tally.rs | 7 +- 3 files changed, 78 insertions(+), 28 deletions(-) delete mode 100644 issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md create mode 100644 issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md diff --git a/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md b/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md deleted file mode 100644 index d85c497287f..00000000000 --- a/issues/build/a-loom-model-whose-spawned-thread-opens-with-a-load-explores-one-execution.md +++ /dev/null @@ -1,25 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-09-28 ---- - -# A loom model whose spawned thread opens with a load explores one execution - -Loom 0.7, as `kernel-loom` pins it, ran a model's closure **once** when the -spawned thread's first operation was a load: a bare loom `AtomicU64`, a -spawned thread doing two `fetch_add`s and the model's thread one `load`, -explored 10 executions; the same with a `Relaxed` load at the top of the -spawned thread explored 1. A reader spawned beside an adder on the model's own -thread explored 19. The probes were throwaway test files in `kernel-loom/tests/` -on `wt/toyos-schedule`, counting closure runs in a `std` atomic. - -`kernel-loom/tests/i8042_tally.rs` was this shape — `Tally::record` opens with -its saturation check's load — so its two models checked no interleaving of -the real word, and a mutation that counted an empty interrupt as a carrying -one passed them. They now spawn the reader and refuse a model that ran one -execution (`explored`). No other model in `kernel-loom/` or -`toyos-sched/loom` has been counted. - -**Exit**: every loom model in the tree shown to run more than one execution — -the same refusal `explored` makes, applied to each. diff --git a/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md b/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md new file mode 100644 index 00000000000..892ee10ca1c --- /dev/null +++ b/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md @@ -0,0 +1,74 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# Loom never moves a store before a load another thread made first, when the storer loaded the word too + +Loom 0.7.2, as `Cargo.lock` pins it, keeps one last access per atomic +(`src/rt/atomic.rs`: `last_access`, `last_non_load_access`). Its DPOR step +(`src/rt/execution.rs`, `schedule`) races a pending load only against the last +store or RMW, and a pending store or RMW against the last access of any kind, +which the storing thread's own earlier load of the word has overwritten. So when +thread A loads a word and thread B then loads and writes it, no execution runs +B's write before A's load. + +Measured on a bare loom `AtomicU64`, counting closure runs in a `std` atomic: a +spawned thread doing two `fetch_add`s beside one `load` on the model's thread +explored 10 executions; the same with a `load` at the top of the spawned thread +explored 1; the reader spawned beside the adder on the model's thread explored +19. + +`kernel-loom/tests/i8042_tally.rs` had that shape: the model's thread read the +tally, and the spawned ISR's `Tally::record` opens with its saturation check's +load. Its models ran one execution, and a mutation that counted an empty +interrupt as a carrying one passed them. They now spawn the reader and run the +ISR on the model's thread, so the reader's pending load is raced against the +ISR's write, and they refuse a run of one execution (`explored`). + +A spawned thread that opens with a load is not the trigger on its own. Every +model below has one, and each explores more than one execution unless it joins +the spawned thread before its own thread races it. None has been read for an +A-then-B pair on one word. + +Executions per model, from `LOOM_LOG=loom::model=info` and a throwaway count in +the two `Builder::check` helpers: + +| model | the spawned thread's first atomic access | executions | +|---|---|---| +| `kernel-loom` `dump_request` `a_request_filed_during_a_report_is_reported` | `DumpRequest::update`'s load | 16 | +| `dump_request` `one_request_is_taken_once` | `update`'s load | 28 | +| `dump_request` `the_asker_owns_the_report_it_asks_for` | `update`'s load | 19 | +| `dump_request` `a_request_is_announced_at_most_once` | `update`'s load | 5777 | +| `dump_request` `a_request_left_during_a_report_is_still_taken_by_its_end` | `update`'s load; joined at once | 1 | +| `log_ring` `a_published_record_is_whole_and_read_once` | `push`'s `tail_then_head` | 1442 | +| `log_ring` `a_slot_is_reused_only_after_its_record_was_read` | `push`'s `tail_then_head` | 43 | +| `log_ring` `a_lane_publishes_whole_and_reuses_only_after_a_read` | `Reader::next_lane`'s `head` | 51 | +| `log_wake` `exactly_one_producer_owns_a_park` | `signal_after_commit`'s load | 7 | +| `panic_capture` `a_reader_and_refresh_never_overlap_on_the_snapshot` | `CaptureAccess::read`, a `fetch_update`, which loom runs as a load then a CAS | 351 | +| `panic_capture` `discard_cannot_admit_a_writer_under_a_fatal_reader` | `read`; `CaptureLatch::owned_by` | 116870 | +| `panic_console_publish` `a_snapshot_is_one_publication_whole` | `publish`'s `seq` | 23613 | +| `reap_gate` `one_raise_is_claimed_once` | `ReapGate::take`'s load | 7 | +| `sleep_lock` `try_lock_observes_the_previous_holders_writes` | `try_lock`'s `now` | 13 | +| `sleep_lock` `a_parking_contender_observes_the_holders_writes`, `a_queued_contender_is_served_and_named`, `two_holders_never_overlap` | `lock`'s `holder` | 225 each | +| `smp_bringup` `a_committed_count_never_outruns_its_slot` | `commit`'s `debug_assert` load; `count` | 45 | +| `smp_bringup` `a_released_machine_is_answering` | `released` | 37 | +| `ticket_lock` `try_lock_observes_the_previous_owners_writes`, `two_try_locks_do_not_both_succeed` | `try_lock`'s `now` | 13 each | +| `tlb_shootdown` `an_acknowledged_flush_postdates_the_page_table_write` | `Shootdown::serve`'s `requested` | 95 | +| `tlb_shootdown` `one_serve_answers_two_concurrent_shootdowns` | `serve`'s `requested` | 70855 | +| `toyos-sched/loom` `loom_park`: all three | `TaskShared::post`'s `state`, through `park::notify` | 79, 18, 18 | +| `loom_retire` `a_wake_and_a_retire_ride_distinct_nodes` | `post`'s `state` | 61 | +| `loom_retire` `a_retire_and_a_wake_never_both_claim_a_parked_task` | `post`'s `state` | 165 | +| `loom_retire` `the_retire_arm_never_loses_a_parked_task_to_a_racing_wake` | `post`'s `state` | 1193 | +| `loom_retire` `the_retire_chase_reuses_one_node_under_a_racing_migration` | `transition`'s `state` | 4 | +| `loom_retire` `an_adopting_cpu_always_observes_the_kill_bit` | `transition`'s `state` | 24 | +| `loom_watch` `a_bounded_post_racing_a_timeout_reaches_a_live_waiter` | `claim_wake`'s `state` | 39 | +| `loom_watch` `a_revoke_racing_the_wait_loop_ends_it_in_every_arm` | `begin_commit`'s `state`, through `prepare` | 199 | +| `loom_watch` `a_transition_racing_an_opening_gate_is_never_missed` | `begin_commit`'s `state` | 2449 | +| `toyos-transport` `loom` `a_published_entry_is_read_whole` | the consumer's `published` | 39 | + +**Exit**: each model above read for a word one thread only loads while another +loads then writes it, and every such pair driven with the writer on the model's +thread, as `i8042_tally.rs` now is; or a loom that races a store against every +thread's last load. diff --git a/kernel-loom/tests/i8042_tally.rs b/kernel-loom/tests/i8042_tally.rs index 3d3500deca9..1716df032b8 100644 --- a/kernel-loom/tests/i8042_tally.rs +++ b/kernel-loom/tests/i8042_tally.rs @@ -47,9 +47,10 @@ use kernel_loom::i8042_tally::{Carried, Counts, Tally}; use loom::sync::atomic::{AtomicU32, Ordering}; use loom::sync::Arc; -/// `loom::model`, refusing a model that explored one execution: loom 0.7 explores nothing but the -/// first schedule when the spawned thread's first operation is a load, which `Tally::record`'s -/// saturation check is, so every model here spawns the reader and runs the ISR on its own thread. +/// `loom::model`, refusing a model that explored one execution: loom races a pending store only +/// against the word's last access, which `Tally::record`'s own saturation load overwrites, so a +/// read made before the ISR ran is never moved after its write. Every model here spawns the +/// reader and runs the ISR on its own thread. fn explored(f: impl Fn() + Sync + Send + 'static) { let runs = std::sync::Arc::new(AtomicUsize::new(0)); let counted = runs.clone(); From 0c03f40743eb8277b2ddd241bae1eb9a715a51c8 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 10:14:59 +0200 Subject: [PATCH 12/18] tests: the review's round-2 findings on the measured schedule The six actuators nothing arms go, with every line only they reached: quiesce-last-exit (the exit hold in SYS_THREAD_EXIT, and `Last` collapses to the one park it still stages), quiesce-dump (`serve_for_the_stop`, `DumpRequest::file_and_take` and its loom model), so-cache-tiny (the tiny budget), xhci-hid-break-first/-late (`stage_break`, `break_at` and the completion count only they read) and klogd-panic. The design-debt issue that recorded them goes with them. The five defect records the deletions closed are restored: deleting a test is no evidence about a defect. Each is `open`, since no redlist row disables anything for them now, and the exit clauses that named a deleted test as the witness are deleted. `fat_backing_revoked` and `launcher_refusals` come back as Weekly guest rows, with their binaries, the FAT oracle in `tests/common/volumes.rs`, their skips, carries, dispatch arms and duration rows. Neither claim can be held on the host without a new abstraction: the FAT revocation is `FatFs::delete`/`revoke` and `FatBacking::read_page` in the kernel adapter, which no host test compiles, and the launcher's refusals are init's syscalls against the kernel's `SYS_NAMESPACE_BUILD` answer, with init built only for the ToyOS target. The ten harness self-checks that boot nothing leave the guest tiers and become libtest tests: `tests/checks.rs` includes `tests/toyos.rs` under the test harness, where its `cfg(test)` module runs them, and the host CI job runs that target. `nvme_image_is_held_by_one_guest` names its claims under CARGO_TARGET_TMPDIR, since a claim is a name and no run directory exists outside a suite run. Smaller findings: `Schedule::tier` becomes `contains`, `guest_reach`'s event decision is a function a host test drives (a run no schedule started reaches nightly), `woken_by_the_held_thread` is inlined into its one caller, `panic_halts_the_others_first` is Nightly and `hda_two_live_refused` Weekly by the data, the two filed issues name an owner, and the REMOVEd sentences are deleted. Three issues are filed: the two mutations only a nightly guest catches (`drain_irqs`'s `Entered`, the join's write-back) and the settled join that still locks the process table. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- Cargo.toml | 4 + ...answer-is-gated-only-by-a-nightly-guest.md | 19 ++ ...ty-loses-five-of-a-thousand-lines-on-ci.md | 24 ++ ...before-a-load-another-thread-made-first.md | 2 +- ...ess-carries-three-helpers-nothing-calls.md | 3 +- ...d-from-is-gated-only-by-a-nightly-guest.md | 18 ++ ...tors-lost-the-only-test-that-armed-them.md | 29 -- ...refusal-retried-is-red-on-every-nightly.md | 30 ++ ...hing-enumerates-on-the-first-controller.md | 19 ++ ...in-takes-the-process-table-to-ask-again.md | 16 + ...ped-reds-wide-with-usb-transport-breaks.md | 47 +++ ...sals-saw-the-kernel-refuse-nothing-once.md | 18 ++ kernel-loom/tests/dump_request.rs | 29 +- kernel-loom/tests/i8042_tally.rs | 5 +- kernel/src/actuator.rs | 18 -- kernel/src/drivers/xhci/device.rs | 1 - kernel/src/drivers/xhci/hid.rs | 31 -- kernel/src/drivers/xhci/mod.rs | 2 - kernel/src/elf/cache.rs | 13 +- kernel/src/log/console.rs | 5 - kernel/src/quiesce.rs | 59 +--- kernel/src/sched/dump.rs | 21 +- kernel/src/sched/dump_request.rs | 7 - kernel/src/syscall/machine.rs | 4 - kernel/src/syscall/proc.rs | 4 +- src/ci.rs | 41 ++- src/tiers.rs | 34 +- tests/CLAUDE.md | 2 +- tests/checks.rs | 3 + tests/common/power.rs | 15 +- tests/common/volumes.rs | 150 ++++++++- tests/netcase/system.toml | 12 +- tests/test-durations | 12 +- .../src/bin/fat_backing_revoked.rs | 153 +++++++++ .../src/bin/launcher_refusals.rs | 291 ++++++++++++++++++ tests/toyos.rs | 217 +++++++++---- toyos-quiesce/src/lib.rs | 7 +- 37 files changed, 1030 insertions(+), 335 deletions(-) create mode 100644 issues/build/a-thread-join-keeping-its-answer-is-gated-only-by-a-nightly-guest.md create mode 100644 issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md create mode 100644 issues/build/which-pass-drain-irqs-is-entered-from-is-gated-only-by-a-nightly-guest.md delete mode 100644 issues/design-debt/six-kernel-actuators-lost-the-only-test-that-armed-them.md create mode 100644 issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md create mode 100644 issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md create mode 100644 issues/kernel/a-settled-thread-join-takes-the-process-table-to-ask-again.md create mode 100644 issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md create mode 100644 issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md create mode 100644 tests/checks.rs create mode 100644 tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs create mode 100644 tests/toyos-rust-tests/src/bin/launcher_refusals.rs diff --git a/Cargo.toml b/Cargo.toml index ccc0d35460f..d49bf5802ef 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -289,5 +289,9 @@ name = "toyos-build" path = "tests/toyos.rs" harness = false +[[test]] +name = "toyos-checks" +path = "tests/checks.rs" + [patch.crates-io] target-lexicon = { git = "https://github.com/ToyOSOrg/target-lexicon", branch = "toyos" } diff --git a/issues/build/a-thread-join-keeping-its-answer-is-gated-only-by-a-nightly-guest.md b/issues/build/a-thread-join-keeping-its-answer-is-gated-only-by-a-nightly-guest.md new file mode 100644 index 00000000000..375fb06cd17 --- /dev/null +++ b/issues/build/a-thread-join-keeping-its-answer-is-gated-only-by-a-nightly-guest.md @@ -0,0 +1,19 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# A thread join keeping its answer is gated only by a nightly guest + +`toyos_proclife::join::Join` keeps the answer of the ask that collected the +zombie, and `a_join_asked_after_it_collected_keeps_its_answer` holds that. +`kernel/src/syscall/proc.rs`'s `sys_thread_join` asks through a copy of it held +in a `Cell` and writes the copy back after each ask. PR #564's round-2 review +dropped that write-back (`join.set(asked)`), so the ask after the wait collects +again and finds no such thread: every host test and the fast tier stayed +green, and only the nightly `fpu_isolation` went red, as it does for the same +mutation of the base. + +**Exit**: dropping the write-back reds a host test or a fast-tier test. Owner: +orchestrator. diff --git a/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md b/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md new file mode 100644 index 00000000000..afc9395912b --- /dev/null +++ b/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md @@ -0,0 +1,24 @@ +--- +status: open +kind: defect +opened: 2026-08-20 +--- + +# `console_line_atomicity` loses five of a thousand lines on CI, one guest per machine + +First sighting on the CI instrument: PR #166 run `32364721784`, `guest (10)`, +`writer A declared 1000 whole lines and the capture carries 995`, `ALONE: +GREEN` in the same job. The diff it rode on is an issues-and-prose audit, so +the tree is not a suspect. + +CI runs one guest per machine with `--jobs 1`, so whatever loses five of a +writer's thousand lines there is not host contention between suites — which +sharpens the question the test was built to ask rather than settling it: the +loss is inside one guest's own console path. + +Split out of `issues/build/parallel-tests-red-under-other-suites.md`, whose +rate table was never this test's — CI's single-guest-per-machine shards rule +out the contention shape that file is about. + +**Exit condition.** The lost lines' cause is fixed. +Owner: orchestrator. diff --git a/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md b/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md index 892ee10ca1c..46b05b8ffdd 100644 --- a/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md +++ b/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md @@ -71,4 +71,4 @@ the two `Builder::check` helpers: **Exit**: each model above read for a word one thread only loads while another loads then writes it, and every such pair driven with the writer on the model's thread, as `i8042_tally.rs` now is; or a loom that races a store against every -thread's last load. +thread's last load. Owner: orchestrator. diff --git a/issues/build/the-harness-carries-three-helpers-nothing-calls.md b/issues/build/the-harness-carries-three-helpers-nothing-calls.md index 3986bd290c1..7cf97d13bd4 100644 --- a/issues/build/the-harness-carries-three-helpers-nothing-calls.md +++ b/issues/build/the-harness-carries-three-helpers-nothing-calls.md @@ -21,4 +21,5 @@ The same pass names `tests/common/audio.rs`'s `completions` and `clients` and (`wt/toyos-notiming`) deletes with their modules. **Exit**: the three deleted, and the module-wide `allow(dead_code)` replaced by -nothing, so the next orphan is a warning the host gate denies. +nothing, so the next orphan is a warning the host gate denies. Owner: +orchestrator. diff --git a/issues/build/which-pass-drain-irqs-is-entered-from-is-gated-only-by-a-nightly-guest.md b/issues/build/which-pass-drain-irqs-is-entered-from-is-gated-only-by-a-nightly-guest.md new file mode 100644 index 00000000000..748adf28e0c --- /dev/null +++ b/issues/build/which-pass-drain-irqs-is-entered-from-is-gated-only-by-a-nightly-guest.md @@ -0,0 +1,18 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# Which pass `drain_irqs` is told it is entered from is gated only by a nightly guest + +`kernel-loom`'s `only_a_pass_entered_at_depth_zero_takes_the_request` holds +`Entered::may_serve`, the decision. The two call sites that choose its argument, +`kernel/src/sched/driver.rs`'s `pass` (`Entered::Pass` at the depth it was +entered at) and `pass_block` (`Entered::Blocking`), are in no crate a host test +compiles. PR #564's round-2 review handed `pass_block`'s drain +`Entered::Pass { depth: 0 }`: every host test and the fast tier stayed green, +and only the nightly `blocked_dump` went red. + +**Exit**: a mutation of either call site's `Entered` reds a host test or a +fast-tier test. Owner: orchestrator. diff --git a/issues/design-debt/six-kernel-actuators-lost-the-only-test-that-armed-them.md b/issues/design-debt/six-kernel-actuators-lost-the-only-test-that-armed-them.md deleted file mode 100644 index c82bd1eafd4..00000000000 --- a/issues/design-debt/six-kernel-actuators-lost-the-only-test-that-armed-them.md +++ /dev/null @@ -1,29 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# Six kernel actuators lost the only test that armed them - -The test schedule deleted the guest tests that never caught a defect, and six -actuators in `kernel/src/actuator.rs` were armed by nothing else: no row of -`tests/toyos.rs`, `tests/common/` or `tests/metal-profile.toml` names them now. - -| actuator | the deleted test | the kernel site it guards | -|---|---|---| -| `klogd-panic` | `klogd_panic_halts` | `kernel/src/log/console.rs` | -| `quiesce-dump` | `quiesce_dump_holds_the_stopped` | `kernel/src/sched/dump.rs` | -| `quiesce-last-exit` | `quiesce_wakes_on_the_last_exit` | `kernel/src/quiesce.rs`, `toyos-quiesce/src/lib.rs` | -| `so-cache-tiny` | `so_cache_refusals` | `kernel/src/elf/cache.rs` | -| `xhci-hid-break-first` | `xhci_hid_break` | the xHCI HID completion path | -| `xhci-hid-break-late` | `xhci_hid_break` | the xHCI HID completion path | - -An actuator nothing arms is kernel code no run executes: the `test-actuators` -kernel still compiles each arm, and no verdict reads what it stages. - -Found by: `git grep` for each actuator's name over `tests/` and `src/` after the -deletions on `wt/toyos-schedule`, every count 0. - -**Exit**: each actuator and the code only it reaches deleted from the kernel, -or a test that arms it registered again. Owner: orchestrator. diff --git a/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md b/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md new file mode 100644 index 00000000000..8e9c210e81f --- /dev/null +++ b/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md @@ -0,0 +1,30 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# `home_budget_refusal_retried` is red on every nightly + +`home_budget_refusal_retried` (nightly tier) is red, and red alone, on the +nightlies of three trees, each time with the same two shapes: +- main at e8d7c9c0 (run 36228604597); +- PR #524's branch at 8c5be843 (run 36273557690); +- main at fd62f567 (run 36278449733, guest (12)). + +The two shapes: `no fsync: /home/... durable on attempt line — the retry +never ran`, and `Boot timed out waiting for ===READY===`. In the second, the +console ends at the loader's `Applied 5483 relocations`, with +`fsync-budget-spent` on the boot parameter line. It was green at 3f46a019 +(run 36111884575). + +It is green alone on a dev host at f231c43e (`cargo test --test toyos-build +-- --nightly home_budget_refusal_retried` EXIT=0). +`cargo run -- --known-red home_budget_refusal_retried` answers NO. + +`fsync-budget-spent` is the machine-wide actuator that +issues/boot-media/partition-claim-gives-up-reds-beside-other-guests-and-is-green-alone.md +names as racing whichever fsync the boot reaches first. Not shown: whether +that race is this test's cause. + +**Exit**: the cause shown on a red run's log. diff --git a/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md b/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md new file mode 100644 index 00000000000..3ebab4d84c2 --- /dev/null +++ b/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md @@ -0,0 +1,19 @@ +--- +status: open +kind: defect +opened: 2026-08-08 +--- + +# `usb_disk_index_stable` reds 1 of 5 on CI: nothing enumerated on the first controller + +From the twelve-shard CI probe (`probe-rate.yml` run `31258202923`, tree +`f8f73e1`, five reps of the exact `ci.yml` configuration): `usb_disk_index_stable` +red 1 of 5, shard 2, `Sched::Parallel`, `nothing enumerated on the first +controller`. Re-taken 2026-09-04 against the last three nightly `ci` runs on +`main` (`33485669019`, `33603832656`, `33728852421`): of the eleven names that +probe found, only this one still reds. + +Split out of `issues/hardware/eleven-names-red-on-ci.md`, which covers eleven +names and has no exit condition for this one in particular. + +**Exit condition.** The cause of the empty first controller is fixed. Owner: orchestrator. diff --git a/issues/kernel/a-settled-thread-join-takes-the-process-table-to-ask-again.md b/issues/kernel/a-settled-thread-join-takes-the-process-table-to-ask-again.md new file mode 100644 index 00000000000..6fc42171c42 --- /dev/null +++ b/issues/kernel/a-settled-thread-join-takes-the-process-table-to-ask-again.md @@ -0,0 +1,16 @@ +--- +status: open +kind: defect +opened: 2026-09-28 +--- + +# A settled thread join takes the process table to ask again + +`sys_thread_join` (`kernel/src/syscall/proc.rs`) asks through +`process::ask_join`, which locks `PROCESS_TABLE` before `Join::ask` looks at +whether an earlier ask settled. A join whose wait's predicate collected the +zombie then asks once more at the top of its loop, so it takes the machine-wide +table lock for an answer it already holds. Found by PR #564's round-2 review. + +**Exit**: a settled join answers without taking `PROCESS_TABLE`. Owner: +orchestrator. diff --git a/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md b/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md new file mode 100644 index 00000000000..a0431dd9991 --- /dev/null +++ b/issues/kernel/quiesce-dump-holds-the-stopped-reds-wide-with-usb-transport-breaks.md @@ -0,0 +1,47 @@ +--- +status: open +kind: defect +opened: 2026-09-25 +--- + +# quiesce_dump_holds_the_stopped reds wide, with two USB transport breaks, and is green alone + +Seen once, in the fast tier on the logd branch (PR #492), dev host: red wide +with `QEMU never reported stopping: the guest asked for a reboot and stayed +up`, then `ALONE ... GREEN`. + +The boot's console shows the stick's transport breaking twice on `SCSI 0x2a` +(`no answer in the status phase in 2000 ms`, recovered each time), then +`quiesce_writers: 4 of 6 writers reached their loop in 5s` and +`test_rs_quiesce_writers exit=1`. The branch touches logd, netd and the +console, not the USB stack, quiesce or its config. + +**Recurrence, PR #524's fast tier at `396f5b4d`, dev host.** Red wide after +325 s with the same `QEMU never reported stopping: the guest asked for a +reboot and stayed up`, then `ALONE ... GREEN` in 4 s. The capture has the same +`quiesce_writers: 4 of 6 writers reached their loop in 5s` and +`test_rs_quiesce_writers exit=1`, and this time no transport break on the +stick. While it ran, the host's 12 guest slots were full, shared with the +suites of five other worktrees (`toyos-lld`, `toyos-tcp`, `toyos-update`, +`toyos-logtrack`, `toyos-guiplat`). +The branch reorders xHCI ring writes (`dma_wmb`) but touches no quiesce code. +A same-session A/B then gave 5 of 5 green on each arm, main at `e48604c0` and +the branch: the red did not come back alone or beside the other arm, so it +is still not shown to be anything but load-bound. + +**Recurrence, PR #542's nightly at 059c5de7 (run 36328646395, `guest (7)`), +KVM, one guest on the runner.** The same `QEMU never reported stopping: the +guest asked for a reboot and stayed up`, the same `quiesce_writers: 4 of 6 +writers reached their loop in 5s` and `test_rs_quiesce_writers exit=1`, and no +transport break in the capture. A writer's first `quiesce-writer: 1` line +is its first pass done: writer 1 at 0.681 s, 0 at 1.061 s, 3 at 1.844 s, 5 at +2.014 s, and writers 2 and 4 only at 6.610 s and 6.778 s, 5.8 s and 5.9 s +after their ` 0` lines. Writer 1 printed pass 33 at 6.905 s, its passes 93 +to 306 ms apart. The branch changes no kernel or guest code. A red on a +one-guest lane is not the other suites' load. + +**Exit condition.** A reproduction names what holds a writer's +first write-and-fsync pass for over 5 s while another writer passes in under a +third of a second, and the fix is shown against it. Owner: the `/log` write and +sync path `tests/toyos-rust-tests/src/bin/quiesce_writers.rs` drives; held by +the orchestrator. diff --git a/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md b/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md new file mode 100644 index 00000000000..30f92133053 --- /dev/null +++ b/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md @@ -0,0 +1,18 @@ +--- +status: open +kind: defect +opened: 2026-09-03 +--- + +# `so_cache_refusals` saw the kernel refuse nothing, once, on CI + +`ci` run 33756442944, `guest (5)`, 2026-09-03, on `w5b15-ready`, whose diff +touches no loader or cache file; `ALONE so_cache_refusals: GREEN, and it was +alone both times` in the same job. + +The verdict: no "byte budget; refused" line — the kernel refused nothing: +twelve 2 MiB images entered a cache whose test budget refuses at the second. + +Owed: a mechanism. Nobody has one. + +**Exit condition.** The cause of the missing refusal is fixed. Owner: orchestrator. diff --git a/kernel-loom/tests/dump_request.rs b/kernel-loom/tests/dump_request.rs index 29ebab527f7..dde1309fd04 100644 --- a/kernel-loom/tests/dump_request.rs +++ b/kernel-loom/tests/dump_request.rs @@ -39,16 +39,10 @@ impl Machine { if !self.request.take() { return false; } - self.report_until_nothing_pending(); - true - } - - /// `sched::dump::report_until_nothing_pending`, once the caller took the request. - fn report_until_nothing_pending(&self) { loop { self.report(); if !self.request.end_report() { - return; + return true; } } } @@ -111,27 +105,6 @@ fn one_request_is_taken_once() { }); } -/// The stop asks while a sibling's pass may serve: filed and taken in one exchange, the request is never -/// pending for the sibling to take, and the one report is the asker's. -#[test] -fn the_asker_owns_the_report_it_asks_for() { - loom::model(|| { - let machine = Machine::new(); - - let sibling = { - let machine = machine.clone(); - loom::thread::spawn(move || machine.serve()) - }; - assert!(machine.request.file_and_take(), "no report ran, and the asker did not take its own request"); - machine.report_until_nothing_pending(); - let sibling_took = sibling.join().unwrap(); - - assert!(!sibling_took, "a sibling's pass took the request the asker filed"); - assert!(!machine.request.pending(), "the request outlived its report"); - assert_eq!(machine.reports(), 1, "one request was reported other than once"); - }); -} - /// Two passes that may not serve and one that may: the request is announced at /// most once, and by nobody once it has been taken unannounced. #[test] diff --git a/kernel-loom/tests/i8042_tally.rs b/kernel-loom/tests/i8042_tally.rs index 1716df032b8..aad97802f74 100644 --- a/kernel-loom/tests/i8042_tally.rs +++ b/kernel-loom/tests/i8042_tally.rs @@ -178,10 +178,7 @@ impl TornTally { /// Asserted by collecting rather than by failing, so this stays a passing test /// that proves a failure is reachable — the flag is an ordinary `std` atomic and /// therefore outlives loom's executions, and the assertion after `loom::model` -/// is the whole verdict. If a future loom, or a future edit to the model's -/// shape, stops scheduling a reader inside the ISR's window, **this reds** and -/// says so: at that point the two models above are passing because nothing is -/// being explored, which is the one failure a gate must not have. +/// is the whole verdict. #[test] fn the_split_counters_this_replaced_are_read_torn() { static TORN: AtomicBool = AtomicBool::new(false); diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 23b8aa6e8ee..5bf2de80cc3 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -119,12 +119,6 @@ actuators! { /// Hold the thread named `toyos_quiesce::LAST_THREAD` inside `SYS_NANOSLEEP`, and the shutdown until it is held there, until the stop waits on it alone: its park is then the stop's last transition. quiesce_last_park = "quiesce-last-park"; - /// The same inside `SYS_THREAD_EXIT`: its exit is then the stop's last transition. - quiesce_last_exit = "quiesce-last-exit"; - - /// Serve a blocked-task dump from the shutdown once its first stage has stopped the machine: the report Ctrl+Alt+D gives on a shutdown stuck in its stop. - quiesce_dump = "quiesce-dump"; - /// Refuse the second directory-entry write of the file `writeback_durability` stages for the retry gate — the first is that file's own seed being made durable — as a budget expiry, so a flush fails at its metadata write with its pages already written and settled. fat_flush_meta_refuse = "fat-flush-meta-refuse"; @@ -248,9 +242,6 @@ actuators! { /// `usb_reset_records_the_phase_it_cut`. usb_reset_under_load = "usb-reset-under-load"; - /// Put the shared-object cache's byte budget within reach of the libraries a guest can build, so the shipped refusal runs at all. - so_cache_tiny = "so-cache-tiny"; - /// Run the first attempt of each run `object::ops::until_answered` retries — /// a file's `SYS_FSYNC`, a claimed partition's read, write or flush — under an /// operation that is already over, once per file and per partition and kind. @@ -326,12 +317,6 @@ actuators! { /// Give PORTSC's PED bit the RW1CS meaning xHCI 1.2 §5.4.8 gives it. xhci_portsc_rw1c = "xhci-portsc-rw1c"; - /// Take a bound HID device's first completion away and hand back a stall. - xhci_hid_break_first = "xhci-hid-break-first"; - - /// The same at its fourth completion. - xhci_hid_break_late = "xhci-hid-break-late"; - /// Run `parse_config` over nine crafted configuration descriptors at init. xhci_descriptor_selftest = "xhci-descriptor-selftest"; @@ -462,9 +447,6 @@ actuators! { /// Let a handle close cancel every poll on the keyboard's watch in the machine. keyboard_close_cancels_every_console = "keyboard-close-cancels-every-console"; - /// Panic inside `klogd` on its first instruction. - klogd_panic = "klogd-panic"; - /// Read address zero inside `klogd` on its first instruction. klogd_fault = "klogd-fault"; diff --git a/kernel/src/drivers/xhci/device.rs b/kernel/src/drivers/xhci/device.rs index 30abbf5cfd7..5e3f52798b5 100644 --- a/kernel/src/drivers/xhci/device.rs +++ b/kernel/src/drivers/xhci/device.rs @@ -784,7 +784,6 @@ fn bind_hid( prev_report: [0; 8], broke_with: None, failures: 0, - completions: 0, }; dev.requeue(&ctrl.db_base); diff --git a/kernel/src/drivers/xhci/hid.rs b/kernel/src/drivers/xhci/hid.rs index 49d205fb544..a92aa589f51 100644 --- a/kernel/src/drivers/xhci/hid.rs +++ b/kernel/src/drivers/xhci/hid.rs @@ -45,10 +45,6 @@ pub struct HidDevice { pub broke_with: Option, /// Consecutive failures; a delivered report clears it — see [`super::MAX_HID_FAILURES`]. pub failures: u8, - /// Completions this endpoint has produced. - /// Counted unconditionally so the `xhci-hid-break-*` actuators aren't a second code path. - #[cfg_attr(not(feature = "boot-actuators"), allow(dead_code))] - pub completions: u32, } impl HidDevice { @@ -106,30 +102,3 @@ impl HidDevice { db_base.write_u32(self.slot_id as u64 * 4, self.int_ep_dci as u32); } } - -/// Takes one completion away from the device that earned it and hands the driver a stall in its place. -// QEMU's usb-hid has no path to USB_RET_STALL for an interrupt IN token, so nothing on the host side can stage this. -// Replaces the completion code and the delivered report, not the TRB/ring/transfer-event/output-context chain, so a dispatched "success" can only be real. -#[cfg(feature = "boot-actuators")] -impl HidDevice { - // The first completion is a never-delivered endpoint; the fourth is one that was working and stopped — different driver states, not degrees of one. - fn break_at() -> Option { - if crate::actuator::xhci_hid_break_first() { - Some(1) - } else if crate::actuator::xhci_hid_break_late() { - Some(4) - } else { - None - } - } - - pub fn stage_break(&mut self, code: u32) -> u32 { - self.completions += 1; - if Self::break_at() != Some(self.completions) { - return code; - } - // Zeroing leaves the slot as a stalled endpoint would have; runs before requeue, so nothing else touches the buffer. - self.report.subview(0, self.report_size as usize).zero(); - super::CC_STALL - } -} diff --git a/kernel/src/drivers/xhci/mod.rs b/kernel/src/drivers/xhci/mod.rs index dc9fb0edf63..8c224612c25 100644 --- a/kernel/src/drivers/xhci/mod.rs +++ b/kernel/src/drivers/xhci/mod.rs @@ -973,8 +973,6 @@ impl XhciController { } return; }; - #[cfg(feature = "boot-actuators")] - let code = self.devices[at].stage_break(code); let dev = &mut self.devices[at]; if code == CC_SUCCESS || code == CC_SHORT_PACKET { dev.failures = 0; diff --git a/kernel/src/elf/cache.rs b/kernel/src/elf/cache.rs index b7322d5b73b..7af3d804e90 100644 --- a/kernel/src/elf/cache.rs +++ b/kernel/src/elf/cache.rs @@ -154,17 +154,6 @@ static SO_CACHE: Lock> = Lock::new(Vec::new()); /// this admits one of those and refuses a second. const BUDGET_BYTES: usize = 256 * 1024 * 1024; -/// `so-cache-tiny`'s number, in reach of a guest. Only the magnitude moves. -const TINY_BUDGET_BYTES: usize = 8 * 1024 * 1024; - -fn budget_bytes() -> usize { - if crate::actuator::so_cache_tiny() { - TINY_BUDGET_BYTES - } else { - BUDGET_BYTES - } -} - /// Every cached image's allocation, summed. The caller holds the lock. fn held_bytes(cache: &[(String, CachedLib)]) -> usize { cache.iter().map(|(_, c)| c.alloc.size()).sum() @@ -239,7 +228,7 @@ pub fn cache_loaded_lib( return outcome.map(|cloned| cloned.unwrap_or_else(|| owned(alloc))); } // Under the lock that publishes: two concurrent loads must not both find room. - let budget = budget_bytes(); + let budget = BUDGET_BYTES; let (held, entries) = (held_bytes(&cache), cache.len()); let Some(after) = held.checked_add(alloc.size()).filter(|b| *b <= budget) else { drop(cache); diff --git a/kernel/src/log/console.rs b/kernel/src/log/console.rs index 8b49f3bff0d..cfb47556423 100644 --- a/kernel/src/log/console.rs +++ b/kernel/src/log/console.rs @@ -417,11 +417,6 @@ fn left_to_the_stop() -> bool { } extern "C" fn body(_arg: u64) -> ! { - // First, before any drain: stages a panic inside a kernel thread to test the panic handler's branch. - #[cfg(feature = "boot-actuators")] - if crate::actuator::klogd_panic() { - panic!("klogd-panic: the console drainer died"); - } #[cfg(feature = "boot-actuators")] if crate::actuator::klogd_fault() { // SAFETY: unsound by design — a staged Ring 0 null read, only on this actuator's boot. diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 42a598524b7..a851a8fff3d 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -245,11 +245,11 @@ fn sweep(caller: ThreadId) -> Sweep { out } -/// `quiesce-last-park` and `quiesce-last-exit`: one thread, named -/// [`toyos_quiesce::LAST_THREAD`], held inside its syscall until the stop's -/// latest sweep counts it as the one thread still running, so the park or the -/// exit it makes next is the last transition the stop sees. Without them no -/// boot can tell whether that transition's post is what wakes the stop. +/// `quiesce-last-park`: one thread, named [`toyos_quiesce::LAST_THREAD`], +/// held inside its `SYS_NANOSLEEP` until the stop's latest sweep counts it as +/// the one thread still running, so the park it makes next is the last +/// transition the stop sees. Without it no boot can tell whether that park's +/// post is what wakes the stop. #[cfg(feature = "boot-actuators")] pub mod last { use core::sync::atomic::{ @@ -262,28 +262,7 @@ pub mod last { use crate::watch::{self, Watch}; use crate::time::{Budget, Deadline, Duration}; - /// The transition the held thread makes once it is released. - #[derive(Clone, Copy)] - pub enum Last { - Park, - Exit, - } - - impl Last { - fn armed(self) -> bool { - match self { - Last::Park => crate::actuator::quiesce_last_park(), - Last::Exit => crate::actuator::quiesce_last_exit(), - } - } - - fn name(self) -> &'static str { - match self { - Last::Park => "quiesce-last-park", - Last::Exit => "quiesce-last-exit", - } - } - } + const NAME: &str = "quiesce-last-park"; /// How long either side waits for the other before the boot dies by name. const STAGED: Budget = Budget::of( @@ -304,18 +283,13 @@ pub mod last { RUNNING.store(swept.running, Release); } - fn armed() -> Option { - [Last::Park, Last::Exit].into_iter().find(|last| last.armed()) - } - - /// Hold the running thread here if it is the one `last` stages. - pub fn hold(last: Last) { - if !last.armed() || !is_the_named_thread() || HELD.swap(true, AcqRel) { + /// Hold the running thread here if it is the one this staging names. + pub fn hold() { + if !crate::actuator::quiesce_last_park() || !is_the_named_thread() || HELD.swap(true, AcqRel) { return; } crate::log!( - "{}: {} is held until the stop waits on it alone", - last.name(), + "{NAME}: {} is held until the stop waits on it alone", toyos_quiesce::LAST_THREAD, ); ARRIVED.post(); @@ -325,8 +299,7 @@ pub mod last { while RUNNING.load(Acquire) != 1 { assert!( !deadline.reached(crate::clock::now()), - "{}: the stop never came down to this thread alone in {} ms", - last.name(), + "{NAME}: the stop never came down to this thread alone in {} ms", STAGED.nanos() / 1_000_000, ); crate::scheduler::yield_now(); @@ -336,10 +309,11 @@ pub mod last { /// Called by the shutdown before it stops anything: the stop is staged /// only once the thread it is staged around is inside its syscall. pub fn await_the_held_thread() { - let Some(last) = armed() else { return }; + if !crate::actuator::quiesce_last_park() { + return; + } crate::log!( - "{}: the stop waits for {} to reach its syscall", - last.name(), + "{NAME}: the stop waits for {} to reach its syscall", toyos_quiesce::LAST_THREAD, ); let deadline = Deadline::at(crate::clock::now() + STAGED.duration()); @@ -354,8 +328,7 @@ pub mod last { ); assert!( HELD.load(Acquire), - "{}: no thread named {} reached its syscall in {} ms", - last.name(), + "{NAME}: no thread named {} reached its syscall in {} ms", toyos_quiesce::LAST_THREAD, STAGED.nanos() / 1_000_000, ); diff --git a/kernel/src/sched/dump.rs b/kernel/src/sched/dump.rs index 655bc42f00e..aed0ffd54d0 100644 --- a/kernel/src/sched/dump.rs +++ b/kernel/src/sched/dump.rs @@ -157,21 +157,6 @@ pub fn file_request() { REQUEST.file(); } -/// `quiesce-dump`: the report a keystroke asks for, served by the thread -/// running the shutdown once its stop holds every thread it named. That -/// thread is at its syscall's entry depth with no lock under it, which is -/// what a pass entered at zero proves and what `Parkable::at_entry` asserts. -#[cfg(feature = "boot-actuators")] -pub fn serve_for_the_stop() { - let _nothing_under_it = crate::scheduler::Parkable::at_entry(); - // Filed and taken in one exchange: a sibling's pass entered at zero would take a request filed alone. - assert!( - REQUEST.file_and_take(), - "quiesce-dump: a report was already running when the stop asked for its own" - ); - report_until_nothing_pending(&UnderNothing(())); -} - /// Ctrl+Alt+D's request, from `drain_irqs` on every pass. pub fn serve_request(entered: Entered) { #[cfg(feature = "boot-actuators")] @@ -190,11 +175,7 @@ fn serve(proof: &UnderNothing) { if !REQUEST.pending() || !REQUEST.take() { return; } - report_until_nothing_pending(proof); -} - -/// The caller took the request: report until a report ends with nothing filed during it. -fn report_until_nothing_pending(proof: &UnderNothing) { + // Until a report ends with nothing filed during it. loop { report(proof); if !REQUEST.end_report() { diff --git a/kernel/src/sched/dump_request.rs b/kernel/src/sched/dump_request.rs index a7538136c38..c6a18b705ad 100644 --- a/kernel/src/sched/dump_request.rs +++ b/kernel/src/sched/dump_request.rs @@ -75,13 +75,6 @@ impl DumpRequest { }); } - /// Ask and begin the report in one update, unless a report runs: no pass can take the request between - /// its filing and its taking, so the asker owns the report it asked for, and it serves anything pending. - #[cfg(any(feature = "boot-actuators", feature = "loom"))] - pub fn file_and_take(&self) -> bool { - self.update(HANDOFF.0, HANDOFF.1, |word| (word & REPORTING == 0).then_some(REPORTING)).is_ok() - } - /// Whether a request is pending. A read, for the pass that has nothing to take. pub fn pending(&self) -> bool { self.word.load(Ordering::Relaxed) & PENDING != NONE diff --git a/kernel/src/syscall/machine.rs b/kernel/src/syscall/machine.rs index 541cf8fee06..5cbb532931d 100644 --- a/kernel/src/syscall/machine.rs +++ b/kernel/src/syscall/machine.rs @@ -101,10 +101,6 @@ fn quiesce(last: &str) -> Result<(), SyscallError> { assert!(queued, "console-queue-at-the-stop: the queue had no room for its one line"); } crate::log::console::drain_for_the_stop(); - #[cfg(feature = "boot-actuators")] - if crate::actuator::quiesce_dump() { - crate::sched::dump::serve_for_the_stop(); - } log!("Syncing filesystems..."); // drain_all before sync_all: a closed-but-undrained file's dirty pages are only in the cache, which sync_all would miss. crate::writeback::drain_all(); diff --git a/kernel/src/syscall/proc.rs b/kernel/src/syscall/proc.rs index 0e80649e38b..6b76f554652 100644 --- a/kernel/src/syscall/proc.rs +++ b/kernel/src/syscall/proc.rs @@ -22,8 +22,6 @@ use super::cancelled; use super::handles::{demand_syscap, handle_result}; pub(super) fn sys_thread_exit(code: i32) -> u64 { - #[cfg(feature = "boot-actuators")] - crate::quiesce::last::hold(crate::quiesce::last::Last::Exit); process::thread_exit(code); } @@ -187,7 +185,7 @@ pub(super) fn sys_thread_join(tid: u64) -> u64 { pub(super) fn sys_nanosleep(nanos: u64) -> u64 { #[cfg(feature = "boot-actuators")] - crate::quiesce::last::hold(crate::quiesce::last::Last::Park); + crate::quiesce::last::hold(); // The ABI's relative span becomes an absolute Deadline here, and only here. let deadline = Deadline::at(crate::clock::now() + Duration::from_nanos(nanos)); // Armed on its own thread with no subject: nothing posts, only the deadline fires it. diff --git a/src/ci.rs b/src/ci.rs index e5b1f2cc3f6..2e035232c13 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -46,9 +46,9 @@ const NIGHTLY_CRON: &str = "0 3 * * 1-6"; const WEEKLY_CRON: &str = "0 3 * * 0"; const USAGE: &str = "cargo run -- --ci , where is one of: - host every host test: the build system, the host workspace, the - licences of what ships, clippy, the model controls, userland - and the SDK (ci.yml, nightly) + host every host test: the build system, the harness's own checks, + the host workspace, the licences of what ships, clippy, the + model controls, userland and the SDK (ci.yml, nightly) gate-stage what protects main, read back from GitHub (ci.yml) toolchain publish this tree's toolchain if nobody has (nightly) guest / one shard of the guest suite at the reach its schedule names (nightly) @@ -439,6 +439,7 @@ fn host(root: &Path) -> Vec { let host_triple = crate::toolchain::host_triple(); let mut steps = vec![ step("the build system", || cargo(root, &["test", "--lib"])), + step("the harness's own checks", || cargo(root, &["test", "--test", "toyos-checks"])), step("the host workspace", || { cargo(root, &["test", "--workspace", "--exclude", "toyos-build"]) }), @@ -609,14 +610,25 @@ fn protection(rules: &serde_json::Value) -> (Vec, Vec) { /// schedule, and the nightly one on the other schedule, on a dispatch and off a /// runner. fn guest_reach() -> Result<&'static str, String> { - if std::env::var("GITHUB_EVENT_NAME").as_deref() != Ok("schedule") { + reach_of_event(std::env::var("GITHUB_EVENT_NAME").ok().as_deref(), || { + let path = std::env::var("GITHUB_EVENT_PATH").map_err(|_| { + "a scheduled run with no $GITHUB_EVENT_PATH names no schedule".to_string() + })?; + std::fs::read_to_string(&path).map_err(|e| format!("{path}: {e}")) + }) +} + +/// [`guest_reach`] over the event's name and a reader of its payload, which +/// only a scheduled run asks for. +fn reach_of_event( + name: Option<&str>, + payload: impl FnOnce() -> Result, +) -> Result<&'static str, String> { + if name != Some("schedule") { return Ok(testargs::NIGHTLY.name); } - let path = std::env::var("GITHUB_EVENT_PATH") - .map_err(|_| "a scheduled run with no $GITHUB_EVENT_PATH names no schedule".to_string())?; - let payload = std::fs::read_to_string(&path).map_err(|e| format!("{path}: {e}"))?; let event: serde_json::Value = - serde_json::from_str(&payload).map_err(|e| format!("{path}: {e}"))?; + serde_json::from_str(&payload()?).map_err(|e| format!("the schedule event: {e}"))?; reach_of_schedule(event["schedule"].as_str()) } @@ -947,6 +959,19 @@ mod tests { } } + /// Only a scheduled run reads its payload, and every other run is a nightly one. + #[test] + fn a_run_no_schedule_started_reaches_nightly() { + for name in [None, Some("workflow_dispatch"), Some("push")] { + let reach = reach_of_event(name, || panic!("{name:?} read a schedule payload")); + assert_eq!(reach, Ok("--nightly"), "a {name:?} run"); + } + let weekly = format!(r#"{{"schedule":"{WEEKLY_CRON}"}}"#); + assert_eq!(reach_of_event(Some("schedule"), || Ok(weekly)), Ok("--weekly")); + let unread = reach_of_event(Some("schedule"), || Err("no payload".into())); + assert_eq!(unread, Err("no payload".into())); + } + /// The schedules `nightly.yml` declares are exactly the two a reach is /// named for, so no scheduled run reaches the refusal above. #[test] diff --git a/src/tiers.rs b/src/tiers.rs index b31b9c78ae8..93a170134c0 100644 --- a/src/tiers.rs +++ b/src/tiers.rs @@ -7,14 +7,7 @@ //! between tiers is editing that one word. //! //! The tiers nest: a plain `cargo test` reaches `Fast`, `--nightly` adds -//! `Nightly`, and `--weekly` adds `Weekly` to that. `.github/workflows/nightly.yml` -//! runs the nightly reach six nights a week and the weekly reach on the seventh -//! (`src/ci.rs`). No pull request boots a guest. -//! -//! A test's tier is what it has caught: `Fast` is the shared boot and the tests -//! that caught a real defect for at most 5 s of guest time, `Nightly` the other -//! tests that caught one, and `Weekly` those that never did; a group sharing one -//! boot takes its most frequent member's tier. A new test enters `Nightly`. +//! `Nightly`, and `--weekly` adds `Weekly` to that. //! //! The local tier is the fourth, and the only one CI never runs: its guests are //! of an architecture no hosted runner has been measured to boot. @@ -102,12 +95,9 @@ impl<'a> Schedule<'a> { Ok(Self(tiers)) } - /// `name`'s tier, refused by name when nothing registers it. - pub fn tier(&self, name: &str) -> Result { - self.0 - .get(name) - .copied() - .ok_or_else(|| format!("{name} is not a registered test, so it has no tier")) + /// Whether anything registers `name`. + pub fn contains(&self, name: &str) -> bool { + self.0.contains_key(name) } /// Every registered name and its tier, by name. @@ -126,10 +116,9 @@ mod tests { #[test] fn every_registered_test_has_exactly_one_tier() { let schedule = Schedule::new(NAMES.into_iter().zip(EVERY)).unwrap(); - for (name, tier) in NAMES.into_iter().zip(EVERY) { - assert_eq!(schedule.tier(name), Ok(tier), "{name}"); - } - assert_eq!(schedule.iter().count(), EVERY.len()); + let mut want: Vec<_> = NAMES.into_iter().zip(EVERY).collect(); + want.sort_by_key(|(name, _)| *name); + assert_eq!(schedule.iter().collect::>(), want); for second in EVERY { let refusal = Schedule::new([("twice", Tier::Fast), ("once", Tier::Weekly), ("twice", second)]) @@ -140,12 +129,11 @@ mod tests { } #[test] - fn an_unregistered_test_is_refused_by_name() { + fn only_a_registered_name_is_contained() { let schedule = Schedule::new([("registered", Tier::Weekly)]).unwrap(); - let refusal = - schedule.tier("never_registered").expect_err("an unregistered name was given a tier"); - assert!(refusal.starts_with("never_registered is not a registered test"), "{refusal}"); - assert!(schedule.tier("registere").is_err(), "a prefix is not the name"); + assert!(schedule.contains("registered")); + assert!(!schedule.contains("never_registered")); + assert!(!schedule.contains("registere"), "a prefix is not the name"); } /// Each reach selects its own tier and every narrower one, and nothing diff --git a/tests/CLAUDE.md b/tests/CLAUDE.md index 4e5d5e966cb..3d5d594383b 100644 --- a/tests/CLAUDE.md +++ b/tests/CLAUDE.md @@ -1,6 +1,6 @@ # Tests -The mechanics live where the work is: profiles and shapes in `tests/common/`, registration and tiers in `tests/toyos.rs`, the fast tier's line in `src/tiers.rs` — read those, not this file, for how the harness works. +The mechanics live where the work is: profiles and shapes in `tests/common/`, registration and tiers in `tests/toyos.rs` — read those, not this file, for how the harness works. ## Caveats that bite every agent diff --git a/tests/checks.rs b/tests/checks.rs new file mode 100644 index 00000000000..a82abcfd7dd --- /dev/null +++ b/tests/checks.rs @@ -0,0 +1,3 @@ +//! `tests/toyos.rs` under the libtest harness, which is where its `cfg(test)` +//! checks run: the suite's own crate root, so every path in it resolves here. +include!("toyos.rs"); diff --git a/tests/common/power.rs b/tests/common/power.rs index c2ca6be8f57..3f46d456272 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -334,25 +334,18 @@ pub fn quiesce_wakes_on_the_last_park( _c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - woken_by_the_held_thread(&["quiesce-last-park", LATE_WORD], rust_bins) -} - -fn woken_by_the_held_thread( - armed: &'static [&'static str; 2], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let actuator = armed[0]; + const ACTUATOR: &str = "quiesce-last-park"; let (whole, record) = stopped_boot( "tests/quiescelastcase/system.toml", "quiesce_last", - armed, + &[ACTUATOR, LATE_WORD], rust_bins, )?; // **The premise, by the kernel's own word**: the thread was held, and // held before the stop claimed anything. Without it the boot below is // one whose last transition was anything at all. let held = format!( - "{actuator}: {} is held until the stop waits on it alone", + "{ACTUATOR}: {} is held until the stop waits on it alone", toyos_quiesce::LAST_THREAD, ); let at = |needle: &str| whole.lines().position(|line| line.contains(needle)); @@ -371,7 +364,7 @@ fn woken_by_the_held_thread( )); } woken_by_its_threads(&record)?; - eprintln!(" [power] {actuator}: the held thread's transition woke the stop: {record}"); + eprintln!(" [power] {ACTUATOR}: the held thread's transition woke the stop: {record}"); Ok(()) } diff --git a/tests/common/volumes.rs b/tests/common/volumes.rs index 550f20d03ad..89c369a97ab 100644 --- a/tests/common/volumes.rs +++ b/tests/common/volumes.rs @@ -1159,6 +1159,153 @@ pub fn writeback_durability( Ok(()) } +/// A `FatBacking` handed out before an unlink reads nothing after it, and the +/// delete-and-reallocate cycle leaves the volume a volume. +/// +/// `FatFs::delete` frees the file's clusters, and FSInfo's `next_free` is walked +/// down to the lowest one freed — so the next allocation on the volume takes +/// them. Until `FatFs::revoke` existed, a process holding a descriptor across +/// somebody else's `rm` demand-paged whatever went into those clusters next. +/// The guest (`test_rs_fat_backing_revoked`) stages exactly that and asserts the +/// read is **refused**. +/// +/// This is the **independent oracle**, and it answers the two questions the +/// guest cannot ask about itself: +/// +/// - **were the clusters really reissued?** The attacker file is read off the +/// image by the `fatfs` crate and must hold its own bytes end to end, and the +/// victim's name must be gone from the directory. A run in which the volume +/// simply had room elsewhere is not a run in which the refusal proved +/// anything, and only the host's own FAT implementation can say. +/// - **did the cycle break the format?** `toyos-fat32-check` (fatgen103) reads +/// the volume after it, against a partition asserted clean before the boot. +/// A revocation that also corrupted the FAT would pass every assertion the +/// guest can make and leave a stick that does not boot. +pub fn fat_backing_revoked( + test_config: &Path, + c_bins: &[(String, Vec)], + rust_bins: &[(String, Vec)], +) -> Result<(), String> { + /// Mirrored in `tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs`. + const VICTIM: &str = "fat-revoke-victim.bin"; + const ATTACKER: &str = "fat-revoke-attacker.bin"; + const CONTROL: &str = "fat-revoke-control.bin"; + const LEN: usize = 8 * 4096; + const VICTIM_BYTE: u8 = 0xA7; + const ATTACKER_BYTE: u8 = 0x5C; + + let image_path = test_dir().join("fat-backing-revoked.img"); + let image = qemu::build_boot_image(test_config, c_bins, rust_bins, &[]); + std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; + let (start, len) = log_extent(&image, &image_path)?; + + // Born clean, asserted rather than assumed, so a complaint after the run is + // the guest's and not one it inherited. + let complaints_before = check(&image[start..start + len]); + if !complaints_before.is_empty() { + return Err(format!( + "the log partition was not born clean, so this gate cannot tell a complaint the \ + guest caused from one it inherited:\n{}", + describe(&complaints_before) + )); + } + + let mut qemu = QemuInstance::boot_with_options( + test_config, + c_bins, + rust_bins, + BootOptions { + profile: qemu::Profile::Metal, + boot_image: Some(qemu::Staged::Written(image_path.clone())), + ..Default::default() + }, + ); + let boot = qemu.boot_log().to_string(); + serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; + if !boot.contains("log-volume: partition mounted") { + return Err(format!( + "the log partition did not mount, so the guest had nowhere to stage the unlink:\n{}", + volume_lines(&boot) + )); + } + + let result = qemu.run_test("test_rs_fat_backing_revoked", Duration::from_secs(60)); + if let Some(err) = &result.error { + return Err(format!("the guest stopped answering: {err}\nserial:\n{}", result.serial)); + } + if result.exit_code != Some(0) { + // The kernel's own lines too: the refusal reaches userland as one `Io`, + // and which layer refused is only in a `log!`. + return Err(format!( + "fat_backing_revoked guest failed:\n{}\nkernel log while it ran:\n{}{}", + result.stdout, result.before, result.serial + )); + } + serial::Serial::named("test serial", result.serial.as_str()).must_be_clean()?; + + writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); + qemu.flush_stdin(); + let tail = qemu.drain_serial(Duration::from_secs(20)); + drop(qemu); + for bad in ["PANIC:", "panicked at"] { + if tail.contains(bad) { + return Err(format!("{bad:?} on the way down\n{tail}")); + } + } + + let after = std::fs::read(&image_path).map_err(|e| format!("read the image back: {e}"))?; + if after.len() != image.len() { + return Err(format!("the image is {} bytes, was {}", after.len(), image.len())); + } + let volume = &after[start..start + len]; + + // The strongest claim first: the volume is still a volume. A revocation that + // freed the chain wrongly would pass every byte comparison below and leave a + // stick that cannot boot. + let complaints_after = check(volume); + if !complaints_after.is_empty() { + return Err(format!( + "the unlink-and-reallocate cycle left the log volume breaking the format:\n{}", + describe(&complaints_after) + )); + } + + let mut files = read_files(volume, &[VICTIM, ATTACKER, CONTROL])?; + let control = need(files.pop().flatten(), CONTROL)?; + let attacker = need(files.pop().flatten(), ATTACKER)?; + if files.pop().flatten().is_some() { + return Err(format!( + "{VICTIM} is still on the volume after the guest unlinked it — the delete never \ + reached the directory, so nothing about a reissued cluster was staged" + )); + } + + for (name, bytes, want) in + [(ATTACKER, &attacker, ATTACKER_BYTE), (CONTROL, &control, VICTIM_BYTE)] + { + if bytes.len() != LEN { + return Err(format!( + "{name} is {} bytes on the volume; the guest wrote {LEN}", + bytes.len() + )); + } + if let Some(at) = bytes.iter().position(|&b| b != want) { + return Err(format!( + "{name} holds {:#04x} at byte {at} on the volume, not {want:#04x} — the host's \ + own FAT implementation does not see the file the guest wrote", + bytes[at] + )); + } + } + + let _ = std::fs::remove_file(&image_path); + eprintln!( + " [fat] the victim is gone from the volume, the {LEN}-byte file written into its place \ + holds {ATTACKER_BYTE:#04x} end to end on the host's own reader, and the checker is silent" + ); + Ok(()) +} + /// F5's negative control: under `usb-flush-fails` a second `fsync` must refuse /// like the first, because the mount's device commit is still owed — the /// pre-generation kernel answered the second call with success and issued @@ -1438,7 +1585,8 @@ pub fn ftruncate_flush_race( } /// A rename whose source is absent leaves the destination on the FAT `/log` -/// volume, and `rename(p, p)` leaves the file — the **independent oracle**. The guest +/// volume, and `rename(p, p)` leaves the file — the **independent oracle** for +/// the same reason `fat_backing_revoked` is one. The guest /// (`test_rs_fs_rename_durable`) stages both; after the shutdown drain the two /// files are read off the image by the `fatfs` crate and must hold their bytes /// end to end, and `toyos-fat32-check` reads the whole volume against a diff --git a/tests/netcase/system.toml b/tests/netcase/system.toml index c0a2cc97dd6..1ae7dc9ddde 100644 --- a/tests/netcase/system.toml +++ b/tests/netcase/system.toml @@ -29,9 +29,11 @@ service = true serves = ["netd"] devices = ["pci:1af4:1041"] -# `launcher`: this is the only config whose test binaries take the launcher -# path to `Command::spawn` at all; everywhere else test-runner holds no -# connector and every spawn is direct. +# `launcher` for `launcher_refusals`, whose subject is what a client can make +# `/system/bin/init` do — so the test estate has to be able to reach it somewhere. +# This is also the only config whose test binaries take the launcher path to +# `Command::spawn` at all: everywhere else test-runner holds no connector and +# every spawn is direct. # `logread` is the log gate's own: a spawned test binary does not inherit a # `SysCap` dup, so the gate that reads the kernel's records runs inside # `test-runner` itself. @@ -45,7 +47,9 @@ syscap = ["logread"] # `every_device_class_has_at_most_one_claimant`. devices = ["pci:1af4:1041"] -# A declared program `spawn_cwd` spawns through init's launcher. +# What `launcher_refusals` asks init to start. A declared program that serves +# nothing and provides nothing, so a refused launch of it takes no acceptor +# with it and a granted one is a clean round trip. [programs.toybox] # What `spawn_cwd` asks for the owner's `cd` and a launch from it: the shell diff --git a/tests/test-durations b/tests/test-durations index 49d0ef5ec14..f8dcc8577c4 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -160,7 +160,6 @@ connect_before_serve 23 console_locale_detect 3922 control_regs 7243 control_regs_negative 3872 -control_regs_verdict 0 debug_float 11 debug_trap 25 demand_paging_sse 12 @@ -184,6 +183,7 @@ endowment_denied 10 esp_filesystem 10123 exit_wait_storm 178 run_exit_status 0 +fat_backing_revoked 8226 fault_gates 31 foreign_disk_untouched 4688 fpu_isolation 14662 @@ -221,7 +221,6 @@ i8042_kbd_echo 4770 i8042_mouse 4310 i8042_no_spurious_wake 146 i8042_quarantine 10068 -i8042_quarantine_verdict 0 i8042_undecoded_bytes 4960 idle_stack_guard 9601 inbox_cancel_wakes 543 @@ -251,6 +250,7 @@ lan_no_lease 32709 lapic_spurious_vector 6829 late_storage_connect 6455 latency_wake 8790 +launcher_refusals 2106 leak_rollback_selftest 5423 loader_watchdog_arms 10064 locale_detect 9959 @@ -292,11 +292,9 @@ netd_connection_caps 6538 netd_gone_mid_bind 32 netd_hostile_peer 4509 netd_listener_forgery 2649 -nightly_tier_is_announced 0 null_sink_client_exits 2167 null_sink_shipped_client 7095 nvme_home_roundtrip 13 -nvme_image_is_held_by_one_guest 0 nvme_large_device 6052 nvme_wide_sector 3500 operation_nesting 5075 @@ -336,7 +334,6 @@ screen_console_clear 5756 screen_console_panic 6788 screen_console_scroll 12310 screen_console_shell 2604 -screen_decoder 6 screen_diag_boot 7330 screen_early_panel 5779 screen_fatal_halt 4970 @@ -349,7 +346,6 @@ screen_log_absent 1823 screen_paged_scrollback 7384 screen_pager_keys 16152 screen_panic_muted 4416 -serial_vocabulary 2 shipped_config_boots 3024 shm_release_reclaims 29 short_sleep_livelock 4823 @@ -358,7 +354,6 @@ smp_roster_and_tsc_trail 4926 sshd_exec 5110 sshd_files 535 sshd_key_auth 1428 -stall_is_not_a_verdict 0 std_alloc 12 std_fs 17 std_fs_write 12 @@ -373,9 +368,6 @@ std_tls_dlopen 18 std_tls_multi_crate 13 std_unwind 18 std_unwind_so 20 -suite_split 1 -suspend_detector 20 -suspend_invalidates_a_verdict 0 swiss_german_layout 5355 syscall_cost 92 syscall_window_nmi 6825 diff --git a/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs b/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs new file mode 100644 index 00000000000..ee0e124b5b8 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs @@ -0,0 +1,153 @@ +//! A FAT32 file backing must not outlive the file it reads. +//! +//! `/log` hands every open file a `FatBacking` holding the volume byte ranges +//! its data lives in. Unlink the file and `Fat32::remove` puts those clusters +//! back in the FAT — and FSInfo's `next_free` is walked *down* to the lowest one +//! freed, so the very next allocation on the volume is the one that takes them. +//! A backing that still names them reads that file's contents: an information +//! disclosure through `open`, `rm` and a write, with nothing crafted about it +//! and no privilege needed. +//! +//! The same defect `home_backing_revoked` covers for `/home`, on the other +//! filesystem and the other allocator. `FatFs::delete` already dropped the +//! *write* handle, so the destructive half was closed and the read half was not. +//! +//! Staged rather than reasoned about: the victim's clusters are freed and then +//! deliberately handed to a file whose bytes are nothing like the victim's, and +//! the still-open descriptor is read afterwards. The host half +//! (`tests/common/volumes.rs::fat_backing_revoked`) shuts the machine down and +//! reads the volume back with an independent FAT implementation and the +//! fatgen103 checker, so what the delete-and-reallocate cycle left on the stick +//! is judged by something that is not the kernel. + +use std::fs; +use std::io::{Read, Write}; +use std::thread; +use std::time::{Duration, Instant}; + +/// Mirrored in `tests/common/volumes.rs::fat_backing_revoked`. Two halves of one +/// fixture; a change to either alone fails loudly rather than passing quietly. +const VICTIM: &str = "/log/fat-revoke-victim.bin"; +const ATTACKER: &str = "/log/fat-revoke-attacker.bin"; +const CONTROL: &str = "/log/fat-revoke-control.bin"; + +/// Eight pages. More than one so the read crosses pages, and — at the 512-byte +/// clusters a 34 MiB FAT32 volume gets — sixty-four clusters, so each page is +/// several extents and the multi-run half of `FatBacking::read_page` is the one +/// under test. +const LEN: usize = 8 * 4096; + +const VICTIM_BYTE: u8 = 0xA7; +const ATTACKER_BYTE: u8 = 0x5C; + +/// Five of the 2 s operation budgets behind the kernel's `WouldBlock` refusal +/// (`kernel/src/block.rs::OPERATION`): device patience is not what this test +/// is about, so its setup asks again the way logd's flush policy does. +const SETUP_PATIENCE: Duration = Duration::from_secs(10); +const SETUP_PAUSE: Duration = Duration::from_millis(200); + +/// One idempotent setup step, asked again on `WouldBlock` until +/// [`SETUP_PATIENCE`] is spent; anything else panics with the step's message. +fn patient(what: &str, mut op: impl FnMut() -> std::io::Result) -> T { + let start = Instant::now(); + loop { + match op() { + Ok(v) => return v, + Err(e) if e.kind() == std::io::ErrorKind::WouldBlock + && start.elapsed() < SETUP_PATIENCE => + { + thread::sleep(SETUP_PAUSE); + } + Err(e) => panic!("{what}: {e}"), + } + } +} + +fn write_file(path: &str, byte: u8) { + { + let mut f = patient(&format!("create {path}"), || fs::File::create(path)); + // Not `patient`: a `write_all` refused partway has advanced the cursor, + // so asking again blind would double bytes; it lands in the cache anyway. + f.write_all(&vec![byte; LEN]).unwrap_or_else(|e| panic!("write {path}: {e}")); + patient(&format!("fsync {path}"), || f.sync_all()); + } // close: the last handle drops here. + // The last close no longer drops the file from the cache on this thread — + // it pins it and hands the teardown to `iod` (`kernel::writeback`). Let that + // drain, so a later open of this name is served by the backing rather than + // adopting the pages this write left cached; the revocation this test checks + // lives in the backing, and a cache-served read would never reach it. The + // margin is enormous: the drain is microseconds of work. + thread::sleep(Duration::from_millis(200)); +} + +fn read_all(f: &mut fs::File) -> std::io::Result> { + let mut got = Vec::new(); + f.read_to_end(&mut got)?; + Ok(got) +} + +fn main() { + // The control. `write_file` drains the write-back so the file leaves the + // cache, so this open is served by the backing and not by pages the write + // left cached — if it were not, the attack below would prove nothing about + // that path. + write_file(CONTROL, VICTIM_BYTE); + let control = read_all(&mut fs::File::open(CONTROL).expect("open the control")) + .expect("read the control"); + assert_eq!(control.len(), LEN, "the control read short"); + assert!( + control.iter().all(|&b| b == VICTIM_BYTE), + "the backing did not serve the control file's own bytes", + ); + + write_file(VICTIM, VICTIM_BYTE); + + // Held open, and deliberately not read: `write_file` drained the victim out + // of the cache, so every page is absent and each one is a fault the backing + // has to answer. + let mut held = fs::File::open(VICTIM).expect("open the victim"); + + fs::remove_file(VICTIM).expect("unlink the victim"); + + // The victim's clusters are the lowest free ones now, so this takes them. + write_file(ATTACKER, ATTACKER_BYTE); + + // **Refused, not zeroed.** A revoked backing has no bytes to serve and has + // to say so; the byte checks below are kept for the case where the refusal + // does not come, because a backing that still resolves serves either + // {ATTACKER_BYTE:#04x} or {VICTIM_BYTE:#04x} and neither is zero. + let refused = match read_all(&mut held) { + Err(e) => e, + Ok(got) => { + if let Some(at) = got.iter().position(|&b| b == ATTACKER_BYTE) { + panic!( + "byte {at} read through the deleted file's descriptor is \ + {ATTACKER_BYTE:#04x} — the backing served another file's data", + ); + } + if let Some(at) = got.iter().position(|&b| b != 0) { + panic!( + "byte {at} read through the deleted file's descriptor is {:#04x}, not \ + zero — the backing still resolves clusters the FAT has taken back", + got[at], + ); + } + panic!( + "the read through the deleted file's descriptor returned {} bytes and \ + succeeded; a revoked backing has no bytes to serve and has to say so", + got.len(), + ); + } + }; + + // The name is gone as well as the bytes, and a fresh open says so with the + // error a missing file gets rather than the one a revoked backing gets. + assert!(fs::File::open(VICTIM).is_err(), "the unlinked victim still opens by name"); + + // Left on the volume on purpose: the host reads both back off the image + // with its own FAT implementation after the shutdown. + println!( + "a read through a backing whose file was deleted was refused ({refused}) rather than \ + serving any of the next file's {LEN} bytes" + ); +} diff --git a/tests/toyos-rust-tests/src/bin/launcher_refusals.rs b/tests/toyos-rust-tests/src/bin/launcher_refusals.rs new file mode 100644 index 00000000000..ee7ccb2c392 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/launcher_refusals.rs @@ -0,0 +1,291 @@ +//! What a client can make `/system/bin/init` do by sending it a bad launch. +//! +//! **init is the one process the machine cannot lose.** It holds the only +//! `SysCap`, every unhanded acceptor and the `launcher` port, and nothing +//! restarts it — so a client that can end it, panic it or grow its handle +//! table without bound takes the machine's ability to start a process with it. +//! Every field of `MSG_LAUNCH` and every handle in its batch is a client's +//! claim about itself, and the launcher connector is held by the compositor, +//! every terminal, every shell and sshd. +//! +//! Three shapes, each of which was reachable before this gate existed: +//! +//! 1. **A frame whose handle count is not the batch's.** The handles are +//! already in init's table when the count is checked, so a refusal that +//! returns without closing them leaks one per attempt — and a client picks +//! how often it attempts. Measured against the kernel's live-object census +//! rather than believed: the batch is a duplicate of a pipe end this +//! process then drops, so the object survives exactly if init kept it. +//! 2. **An extra that is not a connector.** init has no way to ask what a +//! handle it received names, so it hands it to `SYS_NAMESPACE_BUILD` — and +//! a wrong type there used to end the caller, which is init. +//! 3. **A connector the client narrowed `DUP` away from.** init duplicates a +//! provided connector so the namespace and the label can both carry one; a +//! duplicate that is refused used to be an `.expect`. +//! +//! The fourth arm is what stops the other three passing on a dead launcher: an +//! ordinary spawn, which goes through init because this process holds a +//! `launcher` connector and `/system/bin/toybox` is a declared program. It runs last, +//! so it also asserts init survived all three. +//! +//! **A fourth shape, and it is the one init could not survive at all: a client +//! that connects and says nothing.** `serve_launch`'s first statement was a +//! blocking `recv_header` on the fresh connection, so two syscalls from any +//! holder of the connector — the compositor, every terminal, every shell, sshd +//! — parked the machine's only way to start a process for ever, with init alive +//! and looking healthy. +//! +//! **Every answer this file waits for is bounded, and that is not decoration.** +//! A test that hangs instead of failing is worse than no test: a harness +//! timeout is a liveness guard rather than a verdict, and a guest that never +//! returns takes its whole shared boot down with it. So the launcher's replies +//! are read with [`answer_within`], and a launcher that has stopped answering +//! is an assertion with a name on it. The one arm that cannot be bounded from +//! here — `Command`, which blocks inside `std` — runs after a bounded launch +//! has already proved init is answering. + +use std::process::Command; +use std::time::{Duration, Instant}; + +use toyos::census::Census; +use toyos::ipc::{Connection, FrameRx, RxStep}; +use toyos::launch::{self, Launch}; +use toyos::{namespace, port, AsHandle}; +use toyos_abi::handle::Rights; +use toyos_abi::syscall::{self, SyscallError}; +use toyos_abi::RawHandle; + +/// Rounds per census sample. Large enough that one leaked handle per round is +/// a number no drain lag can hide. +const ROUNDS: usize = 16; + +/// How long init may take to answer a launch before this file calls it wedged. +/// +/// Generous by two orders of magnitude: a refusal is a frame decode and a +/// namespace build, and a grant is one `SYS_SPAWN`. What this bounds is the +/// launcher that answers *never*, and the number only decides how long the red +/// takes to arrive. +const ANSWER_BUDGET: Duration = Duration::from_secs(5); + +/// Clients that connect to the launcher and then say nothing, held open across +/// the launch that must still be answered. +/// +/// Well under init's own `MAX_PENDING_LAUNCHES`, because what is under test is +/// that a silent client costs a slot rather than the event loop — not the bound +/// on how many slots there are. +const QUIET_CLIENTS: usize = 8; + +/// A program `tests/netcase` declares that serves nothing and provides +/// nothing, so a refused launch of it takes no acceptor with it. +const DECLARED: &str = "/system/bin/toybox"; + +fn main() { + the_kernel_answers_rather_than_faults(); + a_quiet_client_does_not_wedge_the_launcher(); + not_a_connector(); + a_connector_it_cannot_duplicate(); + a_working_directory_that_is_not_absolute(); + + let before = churn(); + let after = churn(); + let grown: Vec<_> = after.grown_since(&before).collect(); + assert!( + grown.is_empty(), + "{ROUNDS} more refused launches left more live objects behind: {grown:?} — \ + first {before}, then {after}: init is keeping the handles a refusal took", + ); + println!(" census: {} live objects, then {}", before.total(), after.total()); + + the_launcher_still_works(); + println!("a bad launch is refused, and init is still the launcher"); +} + +fn launcher() -> Connection { + toyos::endow::service("launcher").expect("this process was endowed a launcher connector") +} + +/// Read the launcher's reply without ever blocking on it. +/// +/// `Err` is the verdict this file exists to be able to reach: a launcher that +/// has not answered inside the budget is one no `recv_header` would ever come +/// back from. +fn answer_within(conn: &Connection, budget: Duration) -> Result { + let deadline = Instant::now() + budget; + // Only the reply's type is judged here, so nothing of a payload is kept. + let mut rx = FrameRx::<0>::new(); + loop { + match rx.pump(conn) { + RxStep::Frame { msg_type, .. } => return Ok(msg_type), + RxStep::Eof => return Err("the launcher dropped the connection"), + RxStep::Malformed => return Err("the launcher sent a frame this protocol cannot describe"), + RxStep::Idle => { + if Instant::now() >= deadline { + return Err("the launcher never answered"); + } + std::thread::sleep(Duration::from_millis(5)); + } + } + } +} + +/// Clients that connect and go quiet, and a launch that must be answered anyway. +/// +/// Two silences, because they park a server at different statements: a +/// connection that never writes a byte, and one that writes half a header and +/// stops. The first is what `accept` used to be fused to; the second is what a +/// frame read in one blocking call used to wait out. +fn a_quiet_client_does_not_wedge_the_launcher() { + let quiet: Vec = (0..QUIET_CLIENTS).map(|_| launcher()).collect(); + let half = launcher(); + half.write_nonblock(&[0u8; 4]).expect("half a frame header"); + + // **A launch init answers and does not grant.** What is under test is that + // the event loop reaches a frame at all while other connections are silent; + // a *granted* launch would put a spawned process's output on init's own + // stdio, which is this boot's console, and hand back a `Process` handle for + // the census arm below to account for. + let conn = launcher(); + let mut buf = [0u8; 512]; + let request = Launch { + program: "/system/bin/no-such-program", + argv: b"", + env: b"", + cwd: "/", + extras: &[], + slots: &[], + }; + let len = request.encode(&mut buf).expect("encode a launch"); + conn.send_bytes_with_handles(&[], launch::MSG_LAUNCH, &buf[..len]) + .expect("the launcher took the frame"); + + match answer_within(&conn, ANSWER_BUDGET) { + Ok(launch::MSG_NOT_DECLARED) => {} + Ok(other) => panic!("the launcher answered {other} for a program nothing declares"), + Err(why) => panic!( + "{QUIET_CLIENTS} clients that said nothing and one that said half a header \ + left the machine unable to start a process: {why}", + ), + } + drop(quiet); + drop(half); + println!(" quiet clients: {QUIET_CLIENTS} silent and one half-spoken, and a launch still ran"); +} + +/// One frame that promises no handles, with two in the batch beside it. +/// +/// init cannot answer this — it does not know which handle was for what — so +/// there is no reply to read. What it must do is close them. +fn a_frame_that_lies() { + let (read, write) = toyos::pipe_pair().expect("a pipe of our own"); + let first = syscall::dup(write.as_handle()).expect("a duplicate to send"); + let second = syscall::dup(write.as_handle()).expect("a second duplicate to send"); + + let mut buf = [0u8; 512]; + let request = + Launch { program: DECLARED, argv: b"", env: b"", cwd: "/", extras: &[], slots: &[] }; + let len = request.encode(&mut buf).expect("encode a launch"); + + let conn = launcher(); + conn.send_bytes_with_handles(&[first, second], launch::MSG_LAUNCH, &buf[..len]) + .expect("the launcher took the frame"); + drop(conn); + // Both ends go, so the pipe's objects are alive after this only if init + // still holds one of the duplicates. + drop(read); + drop(write); +} + +fn churn() -> Census { + for _ in 0..ROUNDS { + a_frame_that_lies(); + } + // **A launch init answers, and deliberately not one it grants.** init is + // single-threaded and serves connections in the order they queued, so an + // answer to a request sent after the sixteen is the proof it has served + // every one of them. A *granted* launch would put a process exit inside + // the sampled window, and an exiting process leaves objects on the + // deferred release queue — the sample would be reading that lag. + assert_eq!(a_launch_it_refuses(), launch::MSG_REFUSED, "the synchronising launch was granted"); + Census::now() +} + +/// A launch init answers and does not grant: an extra naming a pipe where a +/// connector belongs. +fn a_launch_it_refuses() -> u32 { + let (_read, write) = toyos::pipe_pair().expect("a pipe of our own"); + let handle = syscall::dup(write.as_handle()).expect("a duplicate to send"); + refused_with(&[("surface", handle)]) +} + +fn not_a_connector() { + assert_eq!(a_launch_it_refuses(), launch::MSG_REFUSED); + println!(" not a connector: refused, and init is still here"); +} + +/// A real connector, narrowed so init cannot duplicate it. +fn a_connector_it_cannot_duplicate() { + let (_acceptor, connector) = port::create().expect("a port of our own"); + // Everything `SYS_NAMESPACE_BUILD` asks for and nothing `dup` does, so + // init gets past the namespace and fails on the label. + let narrowed = syscall::dup_narrowed(connector.as_handle(), Rights::TRANSFER) + .expect("a connector carrying only TRANSFER"); + assert_eq!(refused_with(&[("surface", narrowed)]), launch::MSG_REFUSED); + println!(" a connector it cannot duplicate: refused, and init is still here"); +} + +/// Send one launch carrying `extras` and answer the message type init replied +/// with. The reply is the liveness proof as much as the verdict. +fn refused_with(extras: &[(&str, RawHandle)]) -> u32 { + answer_to("/", extras) +} + +/// Send one launch from `cwd` carrying `extras`, and answer init's reply. +fn answer_to(cwd: &str, extras: &[(&str, RawHandle)]) -> u32 { + let mut buf = [0u8; 512]; + let request = Launch { program: DECLARED, argv: b"", env: b"", cwd, extras, slots: &[] }; + let (handles, count) = request.handles(); + let len = request.encode(&mut buf).expect("encode a launch"); + + let conn = launcher(); + conn.send_bytes_with_handles(&handles[..count], launch::MSG_LAUNCH, &buf[..len]) + .expect("the launcher took the frame"); + answer_within(&conn, ANSWER_BUDGET).expect("init answered the launch") +} + +/// A cwd the launch does not state absolutely. std would join it onto init's +/// own, so a grant here is the child started in a directory nobody named. +fn a_working_directory_that_is_not_absolute() { + for cwd in ["", "tmp"] { + assert_eq!(answer_to(cwd, &[]), launch::MSG_REFUSED, "a launch from cwd {cwd:?} was not refused"); + } + println!(" a working directory that is not absolute: refused, and init is still here"); +} + +/// The non-vacuity arm. `/system/bin/toybox` is a `[programs]` key, so a caller +/// holding a `launcher` connector reaches it through init and not through +/// `SYS_SPAWN` — this only passes while init is alive and still launching. +fn the_launcher_still_works() { + let out = Command::new(DECLARED) + .arg("pwd") + .output() + .expect("the launcher started a declared program"); + assert!(out.status.success(), "the launched program exited {:?}", out.status.code()); +} + +/// **The half of it that is the kernel's**, asserted from a process that can +/// afford to die so that init does not have to. +/// +/// `SYS_NAMESPACE_BUILD`'s added connector is the one handle argument in the +/// ABI that routinely crossed a trust boundary — a `provides` name is exactly +/// a connector somebody else made — so a wrong type there answers a word. Every +/// other `WrongType` in the table still ends the caller, and if this one goes +/// back to doing that, this arm never returns and the test reds on exit 139. +fn the_kernel_answers_rather_than_faults() { + let (_read, write) = toyos::pipe_pair().expect("a pipe of our own"); + // SAFETY: it is not a connector, which is the point — the call must answer + // a word rather than end this process. + let pretend = unsafe { port::Connector::from_raw(write.as_handle()) }; + let refused = namespace::build().add("surface", &pretend).finish(); + let _ = pretend.into_raw(); + assert_eq!(refused.err(), Some(SyscallError::InvalidArgument)); +} diff --git a/tests/toyos.rs b/tests/toyos.rs index 598b0e1d91a..cbb4e821260 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -330,6 +330,10 @@ const RUST_SKIP: &[&str] = &[ "inventory_bounds", // Same reason, same config: `netd_hostile_peer` runs it there. "netd_hostile_peer", + // Needs a `launcher` connector, which `tests/testcases`'s test-runner has + // no reason to hold. `launcher_refusals` runs it on tests/netcase, whose + // test-runner receives one for exactly this. + "launcher_refusals", // Needs a launcher to tell its two roads apart, and a declared shell and // toybox to take it: `spawn_cwd` runs it on tests/netcase. "spawn_cwd", @@ -391,6 +395,12 @@ const RUST_SKIP: &[&str] = &[ "writeback_reopen", "writeback_spawn", "writeback_durability", + // Same shape as `writeback_durability`: what it stages on `/log` — a file + // unlinked out from under a held descriptor, its clusters handed to the next + // writer — is only half the claim, and the other half is the volume read + // back off the image after a shutdown by a FAT implementation that is not + // the kernel's. `fat_backing_revoked` runs it. + "fat_backing_revoked", // Stages a rename with an absent source on `/log` and leaves the // destination for `fs_rename_durable` to read back off the image. "fs_rename_durable", @@ -503,7 +513,6 @@ const EARLY_PANIC_MESSAGE: &str = "test-early-panic: on-screen console check"; /// kept because these are read the way they are /// written. const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ - ("screen_decoder", Sched::Parallel, Tier::Weekly), // Two boots, each ended at a loader line rather than at the kernel's ready // marker. Every verdict is a count of rows against a count of lines off the // same boot's console; no clock is in either. @@ -641,7 +650,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // // Serial: it is the one registration here whose verdict is a *time*, and a // wake latency measured beside eleven other guests is the host's schedule. - // Nightly for that same reason. ("latency_wake", Sched::Serial, Tier::Nightly), ("smp_failed_ap_leaves_no_hole", Sched::Parallel, Tier::Weekly), ("input_merge", Sched::Parallel, Tier::Weekly), @@ -691,12 +699,11 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // Reads a device capture and requires at least MIN_SIGNAL_SECS = 0.8 s of // it to carry signal at peak >= 6000 — an absolute seconds-of-signal // floor on audio recorded in real time, not a fraction of the capture and - // not compute-bound: timer-anchored, and Nightly for that reason. + // not compute-bound: timer-anchored. ("doom_music", Sched::Parallel, Tier::Nightly), // A tone played while soundd's pipe to a stalled log is full. The verdicts // are the capture's gaps and a count of lines; the clocks are liveness - // guards. Nightly for the two megabytes of refusals the log reads back - // through a TCG guest's volume. + // guards. ("soundd_log_stall", Sched::Serial, Tier::Nightly), // ureq and rustls from crates.io, fetching over TLS 1.3 from a host this // test mints a CA for. Every verdict is a printed line or a digest; the @@ -849,8 +856,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // liveness guards. ("netd_held_open", Sched::Parallel, Tier::Nightly), // The netcase boot again: a client out-writing a peer that stopped reading - // costs netd no CPU. Nightly: its verdict is the machine's busy time over - // a window of real time. + // costs netd no CPU. ("netd_stalled_peer", Sched::Parallel, Tier::Nightly), // The netcase boot again, beside a host UDP echo: a datagram the client's // pipe will not take whole ends that socket by name and no other. The @@ -899,6 +905,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // before that deadline could expire any of them. Both are wall-clock // margins, which is the definition of [`Sched::Serial`]. ("netd_hostile_peer", Sched::Serial, Tier::Weekly), + ("launcher_refusals", Sched::Parallel, Tier::Weekly), // Paths a child prints and the kernel's refusals by name; no clock in any of them. ("spawn_cwd", Sched::Parallel, Tier::Nightly), ("foreign_disk_untouched", Sched::Parallel, Tier::Weekly), @@ -1055,7 +1062,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("virtio_used_ring", Sched::Parallel, Tier::Weekly), // A fatal path with other CPUs running userland that makes kernel records: // none is stamped past the fatal record by more than an IPI takes. - ("panic_halts_the_others_first", Sched::Parallel, Tier::Fast), + ("panic_halts_the_others_first", Sched::Parallel, Tier::Nightly), // A kernel log line from PCI enumeration; no clock and no real device in it. ("pci_capability_walk", Sched::Parallel, Tier::Weekly), // What QEMU was told to create against what the guest enumerated: two @@ -1250,8 +1257,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // ceiling, which the guest spends and the host never measures. ("blocked_dump", Sched::Parallel, Tier::Nightly), // Two boots of one machine compared on the guest's own `Boot: complete` - // with a 300 ms allowance, which is the whole assertion — a real-time - // verdict, so Nightly. + // with a 300 ms allowance, which is the whole assertion. ("i8042_absent", Sched::Serial, Tier::Nightly), // The fault quarantines (masks) the controller's GSI within milliseconds // of readiness — confirmed from the serial log, before a host round trip @@ -1261,11 +1267,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // timer-anchored, and its price straddles the ceiling run to run (9,355 / // 10,568 / 11,073 ms across three measurements) for exactly that reason. ("i8042_quarantine", Sched::Parallel, Tier::Nightly), - // The negative-direction half of the same gate: no QEMU can stage a CPU into spinning through idle on - // purpose, so this is `idle_is_spinning` proving its teeth against a - // crafted trace shaped like the regression, the way - // `control_regs`/`control_regs_verdict` split the same question. - ("i8042_quarantine_verdict", Sched::Parallel, Tier::Weekly), ("i8042_budget_expiry", Sched::Parallel, Tier::Nightly), ("i8042_fadt_denial", Sched::Parallel, Tier::Weekly), ("i8042_kbd_echo", Sched::Parallel, Tier::Nightly), @@ -1383,6 +1384,11 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // one way a guest can, since its `SYSRET` does not reproduce the erratum. Reds // the day that `mov ss` leaves the switch. ("sysret_ss_reload", Sched::Parallel, Tier::Weekly), + // The FAT32 read side's revocation gate, and a host-side volume oracle for + // the same reason `writeback_durability` is one: whether the clusters the + // unlink freed were really reissued, and whether the cycle left a volume, are + // both questions the guest that staged them cannot answer about itself. + ("fat_backing_revoked", Sched::Parallel, Tier::Weekly), // F5 and F6's negative controls: an fsync that must keep refusing while the // device refuses its cache flush, and a mid-flush redirty raced for real and // re-read off the image. Both bodies in `tests/common/volumes.rs`. @@ -1390,7 +1396,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("redirty_mid_flush", Sched::Parallel, Tier::Weekly), // A truncate staged inside a flush's metadata window, re-read off the image. ("ftruncate_flush_race", Sched::Parallel, Tier::Weekly), - // The rename gate's FAT arm, a host-side volume oracle. + // The rename gate's FAT arm, a host-side volume oracle like `fat_backing_revoked`. ("fs_rename_durable", Sched::Parallel, Tier::Weekly), // The directory work's FAT arm, `fs_rename_durable`'s oracle shape. ("fs_dirs_durable", Sched::Parallel, Tier::Weekly), @@ -1435,28 +1441,9 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // the DMA ring takes to come round. The verdict is soundd's own liveness // and its counters rather than a capture, so it runs wide. ("hda_client_stall", Sched::Parallel, Tier::Nightly), - ("hda_two_live_refused", Sched::Parallel, Tier::Fast), - ("serial_vocabulary", Sched::Parallel, Tier::Weekly), - // Host-side, no guest: the harness asking whether it can still tell a - // suspended machine from a slow one, and whether it reports one as a - // verdict it does not have. - ("suspend_detector", Sched::Parallel, Tier::Weekly), - ("suspend_invalidates_a_verdict", Sched::Parallel, Tier::Weekly), - ("stall_is_not_a_verdict", Sched::Parallel, Tier::Weekly), - // Same: whether two guests can still be handed one lane's NVMe image, which - // is what a shared-boot reboot did to itself. - ("nvme_image_is_held_by_one_guest", Sched::Parallel, Tier::Weekly), - // Same: what a whole run exits with. + ("hda_two_live_refused", Sched::Parallel, Tier::Weekly), + // Host-side, no guest: what a whole run exits with. ("run_exit_status", Sched::Parallel, Tier::Nightly), - // Same: the control-register verdict, against the machine this tree - // actually booted before `arch/x86_64/control_regs.rs`. - ("control_regs_verdict", Sched::Parallel, Tier::Weekly), - // Same: which of the two shared boots each binary belongs on, asked of the - // binaries rather than of the list that claims to name them. - ("suite_split", Sched::Parallel, Tier::Weekly), - // Same: whether a run that did not attempt most of the suite's measured cost says - // so where its verdict is read. - ("nightly_tier_is_announced", Sched::Parallel, Tier::Weekly), ]; /// The test binaries a [`MACHINE_TESTS`] or [`SCREEN_TESTS`] entry runs, which @@ -1521,6 +1508,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("netd_udp_any_address", &["test_rs_netd_udp_any_address"]), ("netd_lookup_let_go", &["test_rs_netd_lookup_let_go"]), ("netd_hostile_peer", &["test_rs_netd_hostile_peer"]), + ("launcher_refusals", &["test_rs_launcher_refusals"]), ("spawn_cwd", &["test_rs_spawn_cwd"]), ("input_claim_absent", &["test_rs_input_absent"]), ("gpu_set_resolution", &["test_rs_gpu_set_resolution"]), @@ -1569,6 +1557,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("boot_volume_metadata_error", &["test_rs_boot_volume_metadata_error"]), ("esp_filesystem", &["test_rs_esp_files"]), ("log_flush_retry", &["test_rs_esp_files"]), + ("fat_backing_revoked", &["test_rs_fat_backing_revoked"]), ("fs_dirs_durable", &["test_rs_fs_dirs_durable"]), ("fs_rename_durable", &["test_rs_fs_rename_durable", "test_rs_fs_dirs_durable"]), ("fsync_failed_commit", &["test_rs_fsync_flush_failed"]), @@ -4473,10 +4462,6 @@ fn run_screen_test( rust_bins: &[(String, Vec)], ) -> Result<(), String> { match name { - "screen_decoder" => { - screen::self_test(); - Ok(()) - } "screen_loader_lines" => { // An EXCLUSIVE open of `GraphicsOutput` calls `Stop` on the // firmware's graphics console, so with one the panel stops at the @@ -11043,6 +11028,9 @@ fn run_machine_test( // Body in `tests/common/volumes.rs`, same reason: the host-side oracle // shuts the guest down and reads `/log` back with `toyos-fat32-check`. "writeback_durability" => common::volumes::writeback_durability(test_config, c_bins, rust_bins), + // Same again: the FAT32 read side's revocation, judged off the volume the + // guest's unlink-and-reallocate cycle left behind. + "fat_backing_revoked" => common::volumes::fat_backing_revoked(test_config, c_bins, rust_bins), // `sysret-ss-probe` has iod null SS, force a switch, and log whether the // switch reloaded it; a missing `mov ss` turns `reloaded` into `NOT`. "sysret_ss_reload" => { @@ -12465,27 +12453,7 @@ fn run_machine_test( ); Ok(()) } - // No guest: the instrument itself, in both directions. `screen_decoder` - // is the same idea for the framebuffer decoder. - // - // Three of them under one name, because they are one subject: what a - // console line says died, what a wait does about it, and the fact that - // only one place in the harness is allowed to answer either. - "serial_vocabulary" => { - serial::self_check()?; - qemu::ceiling_self_check()?; - qemu::host_scale_self_check()?; - one_vocabulary() - } - "suspend_detector" => common::clock::self_check(), - "suspend_invalidates_a_verdict" => suspend_invalidates_a_verdict(), - "stall_is_not_a_verdict" => stall_is_not_a_verdict(), - "nvme_image_is_held_by_one_guest" => nvme_image_is_held_by_one_guest(), "run_exit_status" => run_exit_status(), - "control_regs_verdict" => control_regs_verdict(), - "i8042_quarantine_verdict" => idle_trip_verdict(), - "suite_split" => suite_split(), - "nightly_tier_is_announced" => nightly_tier_is_announced(), "nvme_wide_sector" => { // The other half of "a device's size is a shape dimension": not how // many sectors, but how big one is. `lba_ds` is an 8-bit @@ -15466,6 +15434,66 @@ fn run_machine_test( eprintln!(" [netcase] {refused} hostile frames refused, netd named every peer it dropped"); Ok(()) } + "launcher_refusals" => { + // **`/system/bin/init` is the one process the machine cannot lose**, and + // every launcher client — the compositor, every terminal, every + // shell, sshd — can send it whatever it likes. The guest carries + // the verdicts: init answered, init is still launching, and the + // kernel's live-object count did not grow across sixteen refused + // launches. The host carries the one the guest cannot see — + // whether init said anything about what it refused. + // + // `tests/netcase` because its test-runner is the only one that + // receives a `launcher` connector, and because two boot programs + // is the smallest blast radius for a test whose whole subject is + // making init misbehave. + let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/netcase"); + let bins: Vec<(String, Vec)> = rust_bins + .iter() + .filter(|(name, _)| name == "launcher_refusals") + .cloned() + .collect(); + if bins.is_empty() { + return Err("launcher_refusals was not built".to_string()); + } + let mut qemu = QemuInstance::boot_with_options( + &config, + &[], + &bins, + BootOptions { + profile: qemu::Profile::Headless, + // The live-object count is a `SYS_DEBUG` action, and a + // shipping kernel has none: both readings would be the same + // `InvalidArgument` and the leak arm would pass having + // counted nothing. + kernel_features: ACTUATOR_KERNEL, + ..Default::default() + }, + ); + let mut console = qemu.boot_log().to_string(); + let _ = await_marker(&mut qemu, &mut console, "===READY===", "test-runner to come up"); + + let result = qemu.run_test("test_rs_launcher_refusals", Duration::from_secs(120)); + if let Some(err) = &result.error { + return Err(format!("{err}\n{}", result.stdout)); + } + if result.exit_code != Some(0) { + return Err(format!( + "launcher_refusals exited {:?}:\n{}", + result.exit_code, result.stdout + )); + } + console.push_str(&result.serial); + if !console.contains("init: launcher: cannot start") { + return Err(format!( + "init refused a launch without a line saying so — a launcher that \ + drops requests silently cannot be asked what happened:\n{console}" + )); + } + serial::Serial::named("boot console", console.as_str()).must_be_clean()?; + eprintln!(" [netcase] init refused three bad launches, named them, and kept launching"); + Ok(()) + } "spawn_cwd" => { // A child starts in the directory its spawn names — through the // launcher from the shell's `cd`, through the launcher from @@ -17976,7 +18004,7 @@ fn headline(reason: Option<&str>) -> String { fn nvme_image_is_held_by_one_guest() -> Result<(), String> { // Names, not files: a claim is a hold on a path and touches no disk, so // nothing here has to create or delete a hundred megabytes to ask. - let dir = common::lane::dir(); + let dir = Path::new(env!("CARGO_TARGET_TMPDIR")); let image = dir.join("nvme-claim-gate.img"); let other = dir.join("nvme-claim-gate-other.img"); @@ -19499,8 +19527,7 @@ fn schedule(shared: &[TestDef]) -> Schedule<'_> { /// `redlist::DISABLED` against every name the schedule holds, before any boot /// on any entry point. fn check_redlist(schedule: &Schedule<'_>) -> Result<(), String> { - redlist::check(redlist::DISABLED, |name| schedule.tier(name).is_ok(), &compile::repo_root()) - + redlist::check(redlist::DISABLED, |name| schedule.contains(name), &compile::repo_root()) } fn main() { @@ -20122,3 +20149,65 @@ fn root_withheld_refused(log: &str) -> Result<(), String> { eprintln!(" [root] a handoff with no ROOT image refused the boot by name"); Ok(()) } + +/// The harness's own checks: none boots a guest, so each is a libtest test of +/// `tests/checks.rs` and runs on every host gate rather than in a guest tier. +#[cfg(test)] +mod checks { + use super::*; + + /// One subject: what a console line says died, what a wait does about it, + /// and that only one place in the harness answers either. + #[test] + fn serial_vocabulary() -> Result<(), String> { + serial::self_check()?; + qemu::ceiling_self_check()?; + qemu::host_scale_self_check()?; + one_vocabulary() + } + + #[test] + fn suspend_detector() -> Result<(), String> { + common::clock::self_check() + } + + #[test] + fn suspend_invalidates_a_verdict() -> Result<(), String> { + super::suspend_invalidates_a_verdict() + } + + #[test] + fn stall_is_not_a_verdict() -> Result<(), String> { + super::stall_is_not_a_verdict() + } + + #[test] + fn nvme_image_is_held_by_one_guest() -> Result<(), String> { + super::nvme_image_is_held_by_one_guest() + } + + #[test] + fn control_regs_verdict() -> Result<(), String> { + super::control_regs_verdict() + } + + #[test] + fn i8042_quarantine_verdict() -> Result<(), String> { + idle_trip_verdict() + } + + #[test] + fn suite_split() -> Result<(), String> { + super::suite_split() + } + + #[test] + fn nightly_tier_is_announced() -> Result<(), String> { + super::nightly_tier_is_announced() + } + + #[test] + fn screen_decoder() { + screen::self_test(); + } +} diff --git a/toyos-quiesce/src/lib.rs b/toyos-quiesce/src/lib.rs index 6397d07b861..2d93fe75ab4 100644 --- a/toyos-quiesce/src/lib.rs +++ b/toyos-quiesce/src/lib.rs @@ -67,10 +67,9 @@ impl Sweep { } } -/// The thread the kernel's `quiesce-last-park` and `quiesce-last-exit` -/// actuators hold, by the name its program gives it: held in its syscall -/// until it is the one thread the stop still waits on, so the transition it -/// makes next is the stop's last. +/// The thread the kernel's `quiesce-last-park` actuator holds, by the name its +/// program gives it: held in its `SYS_NANOSLEEP` until it is the one thread the +/// stop still waits on, so the park it makes next is the stop's last. pub const LAST_THREAD: &str = "quiesce-last"; /// What a line of kernel log carrying a [`Record`] begins with. From a9dfb9c89cc21afef89d7b4613355d433cc0e85b Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 10:22:49 +0200 Subject: [PATCH 13/18] tests: the harness's own checks live in the target that runs them `tests/toyos.rs` is a `harness = false` target, so the ten checks' bodies left there were dead in it, and clippy's `cfg(test)` pass over it drops a `#[test]` without a harness. The bodies, and `hand_rolled_deaths`, `needs_actuators`, `driven_binaries` and `DRIVEN_AND_SHARED`, which only they read, move into `tests/checks.rs`'s `checks` module, which only the libtest target compiles. `idle_trip_verdict` takes its test's name, `i8042_quarantine_verdict`. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- tests/checks.rs | 675 +++++++++++++++++++++++++++++++++++++++++++++- tests/toyos.rs | 700 ------------------------------------------------ 2 files changed, 673 insertions(+), 702 deletions(-) diff --git a/tests/checks.rs b/tests/checks.rs index a82abcfd7dd..56804e5db1a 100644 --- a/tests/checks.rs +++ b/tests/checks.rs @@ -1,3 +1,674 @@ -//! `tests/toyos.rs` under the libtest harness, which is where its `cfg(test)` -//! checks run: the suite's own crate root, so every path in it resolves here. +//! The harness's own checks, none of which boots a guest: `tests/toyos.rs` +//! included under the libtest harness, so every path in it resolves here, and +//! the checks beside it in a module only this target compiles. + include!("toyos.rs"); + +mod checks { + use super::*; + + /// One subject: what a console line says died, what a wait does about it, + /// and that only one place in the harness answers either. + #[test] + fn serial_vocabulary() -> Result<(), String> { + serial::self_check()?; + qemu::ceiling_self_check()?; + qemu::host_scale_self_check()?; + one_vocabulary() + } + + /// A wait that hands a death spelling to a scan of its own, asked of the + /// harness's source. + /// + /// **The one place the vocabulary lives is the whole of the fix, so this is + /// what keeps it the one place.** `tests/common/qemu.rs` holds three waits on a + /// guest and they used to disagree: the boot half ended on three spellings, the + /// test half on one and `await_guest` on none at all — so a Rust `panic!` in the + /// kernel matched nothing while a test was running, the machine halted every + /// CPU, and the guard expired onto a verdict saying the guest had stopped + /// answering. All three ask `serial::died` now, which is the only thing in the + /// harness that knows the words and the only thing that knows the prefix decides + /// whose death they report. + /// + /// The way that comes back is the obvious patch: one more spelling handed + /// straight to a `contains` beside the call. It would match a *program's* panic + /// as readily as the kernel's and take the run down with a guest binary that + /// was expected to die — so the shape is refused by name rather than left to a + /// reviewer. Comment lines go first: this file argues about these words at + /// length, and prose is not a second answer. + fn hand_rolled_deaths(text: &str) -> Vec { + let mut found = Vec::new(); + for (n, line) in text.lines().enumerate() { + if line.trim_start().starts_with("//") { + continue; + } + for word in serial::spellings() { + // The shape is the spelling as somebody's first argument — + // `contains`, `starts_with`, `find`, any of them. A spelling + // *inside* a longer staged line is how this file's own gates build + // their inputs, and those are not scans. + if line.contains(&format!("(\"{word}")) { + found.push(format!("{}:{}: {}", n + 1, word, line.trim())); + } + } + } + found + } + + /// [`hand_rolled_deaths`] over the file that has to stay clean, with its own + /// bad input beside it so a check that stopped finding anything says so. + fn one_vocabulary() -> Result<(), String> { + const FILE: &str = "tests/common/qemu.rs"; + let path = Path::new(env!("CARGO_MANIFEST_DIR")).join(FILE); + let text = fs::read_to_string(&path).map_err(|e| format!("read {}: {e}", path.display()))?; + let found = hand_rolled_deaths(&text); + if !found.is_empty() { + return Err(format!( + "{FILE} scans for a death spelling itself, and `serial::died` is where that is \ + decided for every wait at once — a second answer here is what let a kernel panic \ + read as a stall, and it matches a program's own panic besides:\n {}", + found.join("\n ") + )); + } + // The negative control. Every line of it is a shape this must name, and the + // last two are shapes it must not: prose, and a staged capture built out of + // the same words. + let staged = "\ + } else if line.contains(\"KERNEL PANIC\") {\n\ + if line.starts_with(\"SEGFAULT\") {\n\ + // ends on `PANIC:` and nothing else, which is the defect\n\ + const KERNEL: &str = \"[kernel 1.450 cpu3] PANIC: panicked at reserve.rs:812:9:\";\n"; + let named = hand_rolled_deaths(staged); + if named.len() != 2 { + return Err(format!( + "the check names {} of the two hand-rolled scans staged for it: {named:?}", + named.len() + )); + } + eprintln!(" [vocabulary] {FILE} asks `serial::died` and nothing else"); + Ok(()) + } + + #[test] + fn suspend_detector() -> Result<(), String> { + common::clock::self_check() + } + + /// What a suspend is worth to a verdict, staged rather than reasoned about. + /// + /// `common::clock::self_check` gates the detector; this gates what the suite + /// does with what it detects. Both halves are needed and neither implies the + /// other: **a suspend that silently passes is as bad as one that silently + /// fails**, and here the two are one line apart. + #[test] + fn suspend_invalidates_a_verdict() -> Result<(), String> { + let slept = common::clock::SUSPENDED_AT_LEAST + Duration::from_secs(120); + let awake = Duration::ZERO; + // Under the threshold on purpose: two clock reads jitter against each other + // by microseconds, and a run must not be thrown away for that. + let jitter = common::clock::SUSPENDED_AT_LEAST + .checked_sub(Duration::from_millis(1)) + .expect("SUSPENDED_AT_LEAST must be at least 1ms for this case to mean anything"); + let cases: [(&str, Option<&str>, Duration, Verdict); 6] = [ + ("a pass on a host that stayed up", None, awake, Verdict::Pass), + ("a fail on a host that stayed up", Some("the guest said no"), awake, Verdict::Fail), + ("a pass across a suspend", None, slept, Verdict::Invalid), + ("a fail across a suspend", Some("timed out"), slept, Verdict::Invalid), + ("a pass across clock jitter", None, jitter, Verdict::Pass), + ("a fail across clock jitter", Some("the guest said no"), jitter, Verdict::Fail), + ]; + for (what, reason, suspended, want) in cases { + let outcome = Outcome { + name: what.to_string(), + reason: reason.map(str::to_string), + elapsed: Duration::from_secs(3), + suspended, + }; + let got = outcome.verdict(); + if got != want { + return Err(format!("{what} is {got:?}, and it has to be {want:?}")); + } + } + Ok(()) + } + + /// A blown guard stays red, and stops reading as an answer. + /// + /// Both halves, because each fails the other's way round. An implementation + /// that made a stall its own non-red status would hide a guest that genuinely + /// stops; one that only renamed the line would leave the summary saying a test + /// found something. Staged against the strings a wait actually produces rather + /// than against the marker on its own, because a caller prefixes its own + /// sentence to [`await_marker`]'s and the classification has to survive that. + #[test] + fn stall_is_not_a_verdict() -> Result<(), String> { + // Built from the marker rather than copied, so a rename cannot leave the + // gate asserting against a string nothing produces any more. + let real = format!("{STALLED} waiting for the long tone to start — it went quiet"); + let under_a_sentence = format!("the compositor stopped painting\n{real}"); + let cases: [(&str, Option<&str>, bool); 4] = [ + ("an ordinary red", Some("the pointer never moved right"), false), + ("a wait that expired", Some(real.as_str()), true), + ( + "a wait that expired under a caller's own sentence", + Some(under_a_sentence.as_str()), + true, + ), + ("a pass", None, false), + ]; + for (what, reason, want_stall) in cases { + let outcome = Outcome { + name: what.to_string(), + reason: reason.map(str::to_string), + elapsed: Duration::from_secs(1), + suspended: Duration::ZERO, + }; + if outcome.stalled() != want_stall { + return Err(format!( + "{what} reads as stalled={}, and it has to be {want_stall}", + outcome.stalled() + )); + } + // Red is red. A stall that stopped failing the run would be a gate that + // reports and enforces nothing. + let red = outcome.verdict() == Verdict::Fail; + if red != reason.is_some() { + return Err(format!("{what} is red={red}, and a reason is always red")); + } + } + + let mut tally = Tally::new(); + tally.record(Outcome { + name: "a_stalled_test".to_string(), + reason: Some(format!("{STALLED} waiting for nothing at all — it went quiet")), + elapsed: Duration::from_secs(1), + suspended: Duration::ZERO, + }); + tally.record(Outcome { + name: "a_wrong_answer".to_string(), + reason: Some("the pointer never moved right".to_string()), + elapsed: Duration::from_secs(1), + suspended: Duration::ZERO, + }); + if tally.exit_code() != 1 { + return Err(format!("two reds exited {}, and they have to red", tally.exit_code())); + } + if tally.stalls != ["a_stalled_test"] { + return Err(format!( + "the run named {:?} as blown guards; it has to name exactly the one that was", + tally.stalls + )); + } + let summary = tally.summary(2, Duration::from_secs(2), Duration::ZERO); + if !summary.contains("1 of those reds are blown liveness guards") { + return Err(format!("the summary does not separate the two kinds of red:\n{summary}")); + } + Ok(()) + } + + /// One live guest holds its lane's NVMe image, and the next one may not. + /// + /// **The overlap this stages is the one the shared-boot reboot used to + /// produce.** `qemu = boot()` evaluates its right-hand side first, so the + /// replacement was launched while the guest it replaced still held the lane's + /// `test-nvme-*.img` open for write; QEMU's second process exited 1 on its own + /// image lock, `wait_for_ready` panicked, and the panic escaped the shared + /// block — 129 of one run's 131 reds on one sentence, 2026-08-17. + /// + /// The ordering itself is now the type's: `boot` takes a [`qemu::LaneFree`] and + /// the only thing that makes one out of a guest is `QemuInstance::shutdown`, + /// which takes it by value. What is left to check at runtime is the claim + /// underneath — that a hold is real while a guest is up and gone once it is + /// not — and this checks it on the harness's own registry, in both directions, + /// with no guest. + #[test] + fn nvme_image_is_held_by_one_guest() -> Result<(), String> { + // Names, not files: a claim is a hold on a path and touches no disk, so + // nothing here has to create or delete a hundred megabytes to ask. + let dir = Path::new(env!("CARGO_TARGET_TMPDIR")); + let image = dir.join("nvme-claim-gate.img"); + let other = dir.join("nvme-claim-gate-other.img"); + + let held = qemu::NvmeClaim::take(&image).map_err(|why| { + format!("a free image refused its first guest: {why}") + })?; + + // The overlap. This is the direction that must red, and it is what the + // reboot produced. + match qemu::NvmeClaim::take(&image) { + Ok(_) => { + return Err(format!( + "a second guest took {}, which a live one is holding — two QEMUs are then \ + handed one image and the second dies on its lock", + image.display() + )) + } + Err(why) => { + // The refusal has to name the image, or it cannot be acted on: a + // run makes dozens of guests and the message is all a reader gets. + if !why.contains(&image.display().to_string()) { + return Err(format!("the refusal does not name the image it is about: {why}")); + } + } + } + + // A different image is not a conflict, or every lane would refuse every + // other lane's boot the moment this gate had teeth. + let elsewhere = qemu::NvmeClaim::take(&other) + .map_err(|why| format!("an unheld image was refused: {why}"))?; + drop(elsewhere); + + // And the ordinary reboot: the replacement takes the image the guest it + // replaces released. Green, and it is the half a fix that simply refused + // every second boot would break. + drop(held); + let replacement = qemu::NvmeClaim::take(&image).map_err(|why| { + format!("a replacement was refused the image its predecessor released: {why}") + })?; + drop(replacement); + Ok(()) + } + + /// [`control_regs`] against machines this host cannot boot, with no guest. + /// + /// [`control_regs_negative`] runs the real defective machine and is the link + /// between this verdict and a kernel; what is here is the states no actuator + /// reaches — a CPU that differs from three others, a bit set uniformly on all + /// four, an AP that never printed. Every value is one this tree has printed or + /// one bit away from it. + #[test] + fn control_regs_verdict() -> Result<(), String> { + /// The pre-fix machine, `smp=4`, TCG, read off this tree on 2026-08-08: + /// firmware's registers on the BSP and INIT's on every AP. + const AP_BEFORE: (u64, u64) = (0xe000_0011, 0x0031_0620); + const DECLARED: (u64, u64) = (0x8001_0033, 0x0030_0668); + + fn log(cpus: &[(u64, u64)]) -> String { + cpus.iter() + .enumerate() + .map(|(i, (cr0, cr4))| { + format!("[kernel 0.1 cpu{i}] control_regs: cpu{i} cr0={cr0:#010x} cr4={cr4:#010x}\n") + }) + .collect() + } + + let refused = |what: &str, cpus: &[(u64, u64)], says: &str| match control_regs(&log(cpus), 4) { + Ok(()) => Err(format!("{what} was accepted")), + Err(e) if e.contains(says) => Ok(()), + Err(e) => Err(format!("{what} was refused for the wrong reason: {e}")), + }; + + // Positive control first: a verdict that refuses everything refuses the + // defect too, and would prove nothing below. + control_regs(&log(&[DECLARED; 4]), 4) + .map_err(|e| format!("the declared machine was refused: {e}"))?; + + refused("the machine this tree booted", &[DECLARED, AP_BEFORE, AP_BEFORE, AP_BEFORE], "CD")?; + // The case a "do all the CPUs agree?" test passes: they agree, on INIT's + // value. Nothing about uniformity says caching is on. + refused("four CPUs agreeing on INIT's CR0", &[AP_BEFORE; 4], "CD")?; + refused( + "one CPU without WP", + &[DECLARED, (DECLARED.0 & !(1 << 16), DECLARED.1), DECLARED, DECLARED], + "WP", + )?; + refused( + "one CPU without NE", + &[DECLARED, DECLARED, (DECLARED.0 & !(1 << 5), DECLARED.1), DECLARED], + "NE", + )?; + // The bit that must be *absent*: with it set, XCR0 can name components + // FXSAVE64 does not save. + refused("OSXSAVE set", &[(DECLARED.0, DECLARED.1 | (1 << 18)); 4], "OSXSAVE")?; + // Two bits a machine could hold uniformly, each one line of kernel diff + // away, and neither reachable by an actuator. `AM` is named clear above and + // answers by name; `PGE` is named nowhere, which is the case the whole + // never-named rule exists for — `TSD` and `PKE` are the same case. `UMIP` + // used to be this file's example of the same thing, until it joined + // `CR4_MAY` — a bit that moves from unnamed to optional is exactly the + // migration this gate exists to force a diff for. + refused("every CPU with AM set", &[(DECLARED.0 | (1 << 18), DECLARED.1); 4], "AM")?; + refused( + "every CPU with PGE set", + &[(DECLARED.0, DECLARED.1 | (1 << 7)); 4], + "never named", + )?; + // The bit that was on before this and asserted nowhere, so deleting `+smep` + // from the launcher or breaking the CPUID gate in `control_regs::supported` + // reddened nothing at all. + refused("every CPU without SMEP", &[(DECLARED.0, DECLARED.1 & !(1 << 20)); 4], "SMEP")?; + // A CPU that agrees about every named bit and differs in one the CPU is + // allowed to withhold, so nothing above it can object. + refused("one CPU with PCID and three without", &[DECLARED, DECLARED, DECLARED, (DECLARED.0, DECLARED.1 | (1 << 17))], "cpu3")?; + // And an AP that never printed at all, which is what a machine whose AP + // died before the check looks like. + refused("three lines for four CPUs", &[DECLARED; 3], "{0, 1, 2, 3}")?; + + eprintln!(" [control_regs] the verdict refuses 10 machines and accepts the declared one"); + Ok(()) + } + + /// [`idle_is_spinning`] against a healthy trace and a crafted one shaped like + /// the regression it exists to catch, with no guest — the same split + /// `control_regs`/`control_regs_verdict` use, and for the same reason: a + /// gate's own teeth are a claim a live boot cannot demonstrate on the + /// negative side, because nothing in this tree can stage a CPU into spinning + /// through idle on purpose. + /// + /// This is the demonstration the closed vacuous-line-count entry + /// asked for: proof the restored assertion still fails when the condition it + /// names is violated, not just that it still passes when it is not. + #[test] + fn i8042_quarantine_verdict() -> Result<(), String> { + let healthy = "\ + [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ + [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ + [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=3\n\ + [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n"; + if let Some((cpu, delta)) = idle_is_spinning(healthy) { + return Err(format!("a healthy trace was refused: cpu{cpu} moved by {delta}")); + } + + // The regression's own shape: one CPU quarantines cleanly and stays + // quiet, the other's undrained ring never lets it halt. + let spinning = "\ + [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ + [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=4\n\ + [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n\ + [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=2685004\n"; + match idle_is_spinning(spinning) { + Some((1, delta)) if delta > MAX_IDLE_TRIP_DELTA => {} + Some((cpu, delta)) => { + return Err(format!("refused the wrong CPU or by the wrong margin: cpu{cpu} delta {delta}")) + } + None => return Err("a spinning CPU's trace was accepted".to_string()), + } + + // And the line the old, count-of-lines check would have been fooled by: + // the same number of `sched: cpu=` lines either way, because the print + // itself is rate-limited regardless of what is underneath it — which is + // exactly the vacuity this replaces. + assert_eq!( + healthy.matches("sched: cpu=").count(), + spinning.matches("sched: cpu=").count(), + "the crafted traces must differ only in trips=, not in line count — otherwise this proves nothing about the old check's blindness" + ); + + eprintln!(" [i8042] the idle-trip verdict accepts a healthy trace and refuses a spinning one"); + Ok(()) + } + + /// Binaries a machine test drives that the shared boot also runs on purpose. + /// + /// **A binary a machine test drives under a different name is still discovered + /// by [`discover_rust_tests`]**, still runs on the shared boot, and there + /// passes on its exit code with nothing staged for it to act on. `RUST_SKIP` is + /// one answer to that; this list is the other, for the binaries whose shared + /// run asserts something of its own. Every driven name is on one list or the + /// other, so neither answer is silence — `suite_split` is the gate. + /// `sched_stress` is the one whose two runs differ by *kernel* rather than by + /// what the host staged: the shipping build here, `sched_check_build`'s + /// assert-carrying build there. + const DRIVEN_AND_SHARED: &[&str] = &[ + // The lost-wake canary: its shared run is the count on the shipping + // kernel with nothing staged, and `blocking_read_window` drives it again + // with the watch's window held open. + "blocking_read_stress", + // The log-stream arms drive it for the kernel's `exit:` record about it, + // not for anything it does: it is the cheapest process this tree starts. + "empty_dir_stat", + // Its shared run is a whole handle-lifecycle gate with its own census; + // `userdev_dma_fault` drives the same binary for a different reason + // entirely — as the proof the machine still schedules and spawns after a + // device was refused at the unit — and stages nothing for it. + "handle_basic", + "hierarchy_paths", + "null_sink_client_exits", + "nvme_home_roundtrip", + "sched_stress", + "std_alloc", + "std_mmap", + "wall_clock_now", + ]; + + /// What the declaration itself has to be, before any of it means anything. + /// Which shared-boot binaries need `SYS_DEBUG`, asked of their source. + /// + /// A name reaches the syscall directly, or through a child it spawns. + fn needs_actuators(sources: &[(String, String)], registry: &[&str]) -> BTreeSet { + // The fourth spelling is the argument-taking form: every action that + // carries a payload (TLB_ACK_DELAY_ARM, CENSUS_KIND, LOWER_SYSINFO_BOUND, + // SLOT_TO_LAST_GENERATION) is reached through `debug_with`, never + // `debug`. The third is the SDK's: `toyos::census` calls `debug_with` on + // the caller's behalf, so a binary whose leak assertion is a census names + // no syscall of its own and reads as innocent to the others. + let calls = |text: &str| { + text.contains("SYS_DEBUG") + || text.contains("syscall::debug(") + || text.contains("syscall::debug_with(") + || text.contains("census::Census") + }; + let direct: BTreeSet<&str> = + sources.iter().filter(|(_, t)| calls(t)).map(|(n, _)| n.as_str()).collect(); + let mut out = BTreeSet::new(); + for (name, text) in sources { + if !registry.contains(&name.as_str()) { + continue; + } + let spawns = direct.iter().any(|d| text.contains(&format!("test_rs_{d}"))); + if direct.contains(name.as_str()) || spawns { + out.insert(name.clone()); + } + } + out + } + + /// [`ACTUATOR_TESTS`] is exactly the shared-boot binaries that reach + /// `SYS_DEBUG`, and the binaries are what is asked. + /// + /// **What this does not cover, stated because the hole is real:** a machine or + /// screen test that *drives* one of those binaries on a boot of its own. No + /// static rule here can say which `BootOptions` a `run_test` call belongs to. + /// What answers it instead is the guest: `test_panic_child` names + /// `InvalidArgument` as *this kernel carries no actuators* rather than reporting + /// a kernel that failed to stop, so the red says what is wrong wherever it + /// happens. + /// + /// **Both directions are the point.** A binary that gains a `debug()` call and + /// no entry would run on the shipping kernel, where the syscall answers + /// `InvalidArgument` — and a test whose verdict is that a process died would + /// then fail for a reason with nothing to do with what it is about. An entry + /// whose binary no longer calls it is a test kept off the shipping kernel for + /// nothing, which is the erosion this split exists to stop. + #[test] + fn suite_split() -> Result<(), String> { + let dir = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/toyos-rust-tests/src/bin"); + let mut sources: Vec<(String, String)> = Vec::new(); + for entry in fs::read_dir(&dir).map_err(|e| format!("read {}: {e}", dir.display()))? { + let path = entry.map_err(|e| e.to_string())?.path(); + if path.extension().is_none_or(|e| e != "rs") { + continue; + } + let name = path.file_stem().unwrap().to_string_lossy().into_owned(); + let text = fs::read_to_string(&path).map_err(|e| format!("read {name}: {e}"))?; + sources.push((name, text)); + } + let registry: Vec<&str> = + sources.iter().map(|(n, _)| n.as_str()).filter(|n| !RUST_SKIP.contains(n)).collect(); + + // The negative control, and it carries its own bad input: a binary that + // calls the syscall and is on no list must be named, or the check above is + // a spelling of `true`. + let staged = vec![ + ("a_listed_one".to_string(), "syscall::debug(3)".to_string()), + ("an_unlisted_one".to_string(), "SYS_DEBUG".to_string()), + ("its_parent".to_string(), "Command::new(\"/system/bin/test_rs_an_unlisted_one\")".to_string()), + ("a_censor".to_string(), "use toyos::census::Census;".to_string()), + ("a_debug_with_user".to_string(), "syscall::debug_with(3, 4)".to_string()), + ("innocent".to_string(), "println!()".to_string()), + ]; + let staged_registry = [ + "a_listed_one", + "an_unlisted_one", + "its_parent", + "a_censor", + "a_debug_with_user", + "innocent", + ]; + let found = needs_actuators(&staged, &staged_registry); + let want: BTreeSet = + ["a_listed_one", "an_unlisted_one", "its_parent", "a_censor", "a_debug_with_user"] + .iter() + .map(|s| s.to_string()) + .collect(); + if found != want { + return Err(format!("the check does not work: on staged input it named {found:?}")); + } + + let want: BTreeSet = needs_actuators(&sources, ®istry); + let listed: BTreeSet = ACTUATOR_TESTS.iter().map(|s| s.to_string()).collect(); + let missing: Vec<&String> = want.difference(&listed).collect(); + if !missing.is_empty() { + return Err(format!( + "{missing:?} reach SYS_DEBUG and are on the shipping boot, where the syscall answers \ + InvalidArgument. Add each to ACTUATOR_TESTS, or to RUST_SKIP if it is driven rather \ + than run." + )); + } + let stale: Vec<&String> = listed.difference(&want).collect(); + if !stale.is_empty() { + return Err(format!( + "{stale:?} are held off the shipping kernel and no longer reach SYS_DEBUG. Delete \ + each entry — coverage of the binary an image ships is what it costs." + )); + } + // **The other shape [`schedule`] cannot see**: a binary a machine + // test drives under a *different* name is still discovered here, still runs + // on the shared boot, and there passes on its exit code with nothing staged + // for it to act on. + let staged_driven = [ + String::from("qemu.run_test(\"test_rs_a_driven_one\", Duration::from_secs(30))"), + String::from("Command::new(\"/system/bin/test_rs_another_driven\")"), + String::from("qemu.run_test(&format!(\"test_rs_{name}\"), ceiling)"), + ]; + let found = driven_binaries(&staged_driven); + let want: BTreeSet = + ["a_driven_one", "another_driven"].iter().map(|s| s.to_string()).collect(); + if found != want { + return Err(format!("the driven-name reader does not work: it named {found:?}")); + } + + let root = Path::new(env!("CARGO_MANIFEST_DIR")); + let mut harness = vec![fs::read_to_string(root.join("tests/toyos.rs")) + .map_err(|e| format!("read tests/toyos.rs: {e}"))?]; + let common = root.join("tests/common"); + for entry in fs::read_dir(&common).map_err(|e| format!("read {}: {e}", common.display()))? { + let path = entry.map_err(|e| e.to_string())?.path(); + if path.extension().is_some_and(|e| e == "rs") { + harness.push(fs::read_to_string(&path).map_err(|e| e.to_string())?); + } + } + let shared: BTreeSet<&str> = registry.iter().copied().collect(); + let both: BTreeSet = driven_binaries(&harness) + .into_iter() + .filter(|name| shared.contains(name.as_str())) + .collect(); + let declared: BTreeSet = DRIVEN_AND_SHARED.iter().map(|s| s.to_string()).collect(); + let undeclared: Vec<&String> = both.difference(&declared).collect(); + if !undeclared.is_empty() { + return Err(format!( + "{undeclared:?} are driven by a machine test and also run on the shared boot, where \ + nothing stages what they need — so each passes on its exit code with no verdict. Add \ + each to RUST_SKIP with the reason its driver exists, or to DRIVEN_AND_SHARED if its \ + shared run asserts something of its own." + )); + } + let stale: Vec<&String> = declared.difference(&both).collect(); + if !stale.is_empty() { + return Err(format!( + "DRIVEN_AND_SHARED names {stale:?}, which no machine test drives or the shared boot \ + no longer runs. Delete each entry — a declaration nothing is true of is what makes \ + the rest of the list unreadable." + )); + } + + println!( + " [split] {} shared binaries on the shipping kernel, {} on the actuator one, {} of them \ + driven elsewhere and declared", + registry.len() - listed.len(), + listed.len(), + both.len() + ); + Ok(()) + } + + /// Every guest binary the harness drives by name, read out of its own sources. + /// + /// A driver reaches a binary as the literal `test_rs_`, so that is what + /// says a binary has one; a `format!` over a variable name yields no literal + /// and is not a driver of any particular binary. Pure, and the sources are a + /// parameter, so `suite_split` stages its own before trusting it on the tree. + fn driven_binaries(sources: &[String]) -> BTreeSet { + const MARK: &str = "test_rs_"; + let mut found = BTreeSet::new(); + for text in sources { + let mut rest = text.as_str(); + while let Some(at) = rest.find(MARK) { + rest = &rest[at + MARK.len()..]; + let end = rest + .find(|c: char| !c.is_ascii_alphanumeric() && c != '_') + .unwrap_or(rest.len()); + if end > 0 { + found.insert(rest[..end].to_string()); + } + } + } + found + } + + /// Whether a run that did not attempt most of the suite's measured cost says so. + /// + /// **The failure mode the tier introduces is silence, not a wrong answer.** A + /// green run holding back 60 tests and a green run holding back none print the + /// same word, and the difference between them is the whole reason a reach flag + /// exists. Nothing else can see whether the *run* mentions it, and a run nobody + /// can tell apart from a full one is how a temporary measure becomes permanent. + /// + /// Both directions, because the second is the one that rots quietly: a suite + /// that ran everything must not claim to have held anything back either, or the + /// line stops carrying information the day somebody makes it unconditional. + #[test] + fn nightly_tier_is_announced() -> Result<(), String> { + let held = vec![ + (Tier::Nightly, vec!["desktop_window_child".to_string(), "sshd_exec".to_string()]), + (Tier::Weekly, vec!["iommu_empty_domain".to_string()]), + ]; + let announced = Tally::new().holding_back(held).summary(1, Duration::ZERO, Duration::ZERO); + for want in [ + "not run without --nightly:", + "desktop_window_child, sshd_exec", + "`cargo test --test toyos-build -- --nightly` runs them", + "not run without --weekly:", + "`cargo test --test toyos-build -- --weekly` runs them", + "2 held back for --nightly, 1 held back for --weekly", + ] { + if !announced.contains(want) { + return Err(format!("a run holding tests back never says {want:?}:\n{announced}")); + } + } + let whole = Tally::new().holding_back(vec![(Tier::Weekly, Vec::new())]).summary( + 1, + Duration::ZERO, + Duration::ZERO, + ); + if whole.contains("--") || whole.contains("held back") { + return Err(format!("a run that held nothing back says it did:\n{whole}")); + } + Ok(()) + } + + #[test] + fn screen_decoder() { + screen::self_test(); + } +} diff --git a/tests/toyos.rs b/tests/toyos.rs index 8c433561fe2..638e78fc261 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -453,39 +453,6 @@ const RUST_SKIP: &[&str] = &[ "pkg_launch_gbae", ]; -/// Binaries a machine test drives that the shared boot also runs on purpose. -/// -/// **A binary a machine test drives under a different name is still discovered -/// by [`discover_rust_tests`]**, still runs on the shared boot, and there -/// passes on its exit code with nothing staged for it to act on. `RUST_SKIP` is -/// one answer to that; this list is the other, for the binaries whose shared -/// run asserts something of its own. Every driven name is on one list or the -/// other, so neither answer is silence — `suite_split` is the gate. -/// `sched_stress` is the one whose two runs differ by *kernel* rather than by -/// what the host staged: the shipping build here, `sched_check_build`'s -/// assert-carrying build there. -const DRIVEN_AND_SHARED: &[&str] = &[ - // The lost-wake canary: its shared run is the count on the shipping - // kernel with nothing staged, and `blocking_read_window` drives it again - // with the watch's window held open. - "blocking_read_stress", - // The log-stream arms drive it for the kernel's `exit:` record about it, - // not for anything it does: it is the cheapest process this tree starts. - "empty_dir_stat", - // Its shared run is a whole handle-lifecycle gate with its own census; - // `userdev_dma_fault` drives the same binary for a different reason - // entirely — as the proof the machine still schedules and spawns after a - // device was refused at the unit — and stages nothing for it. - "handle_basic", - "hierarchy_paths", - "null_sink_client_exits", - "nvme_home_roundtrip", - "sched_stress", - "std_alloc", - "std_mmap", - "wall_clock_now", -]; - // Audio glitch tests. Each runs in its own QEMU boot per SMP config and // asserts on the wav the virtio-sound device captured, so they are excluded // from the shared multi-test boot. @@ -17170,84 +17137,6 @@ fn wake_latency_recorded(boot: &metal::Readback) -> Result<(), String> { boot.number("latency.p99_us", u64::try_from(code).expect("a non-negative code")) } -/// [`control_regs`] against machines this host cannot boot, with no guest. -/// -/// [`control_regs_negative`] runs the real defective machine and is the link -/// between this verdict and a kernel; what is here is the states no actuator -/// reaches — a CPU that differs from three others, a bit set uniformly on all -/// four, an AP that never printed. Every value is one this tree has printed or -/// one bit away from it. -fn control_regs_verdict() -> Result<(), String> { - /// The pre-fix machine, `smp=4`, TCG, read off this tree on 2026-08-08: - /// firmware's registers on the BSP and INIT's on every AP. - const AP_BEFORE: (u64, u64) = (0xe000_0011, 0x0031_0620); - const DECLARED: (u64, u64) = (0x8001_0033, 0x0030_0668); - - fn log(cpus: &[(u64, u64)]) -> String { - cpus.iter() - .enumerate() - .map(|(i, (cr0, cr4))| { - format!("[kernel 0.1 cpu{i}] control_regs: cpu{i} cr0={cr0:#010x} cr4={cr4:#010x}\n") - }) - .collect() - } - - let refused = |what: &str, cpus: &[(u64, u64)], says: &str| match control_regs(&log(cpus), 4) { - Ok(()) => Err(format!("{what} was accepted")), - Err(e) if e.contains(says) => Ok(()), - Err(e) => Err(format!("{what} was refused for the wrong reason: {e}")), - }; - - // Positive control first: a verdict that refuses everything refuses the - // defect too, and would prove nothing below. - control_regs(&log(&[DECLARED; 4]), 4) - .map_err(|e| format!("the declared machine was refused: {e}"))?; - - refused("the machine this tree booted", &[DECLARED, AP_BEFORE, AP_BEFORE, AP_BEFORE], "CD")?; - // The case a "do all the CPUs agree?" test passes: they agree, on INIT's - // value. Nothing about uniformity says caching is on. - refused("four CPUs agreeing on INIT's CR0", &[AP_BEFORE; 4], "CD")?; - refused( - "one CPU without WP", - &[DECLARED, (DECLARED.0 & !(1 << 16), DECLARED.1), DECLARED, DECLARED], - "WP", - )?; - refused( - "one CPU without NE", - &[DECLARED, DECLARED, (DECLARED.0 & !(1 << 5), DECLARED.1), DECLARED], - "NE", - )?; - // The bit that must be *absent*: with it set, XCR0 can name components - // FXSAVE64 does not save. - refused("OSXSAVE set", &[(DECLARED.0, DECLARED.1 | (1 << 18)); 4], "OSXSAVE")?; - // Two bits a machine could hold uniformly, each one line of kernel diff - // away, and neither reachable by an actuator. `AM` is named clear above and - // answers by name; `PGE` is named nowhere, which is the case the whole - // never-named rule exists for — `TSD` and `PKE` are the same case. `UMIP` - // used to be this file's example of the same thing, until it joined - // `CR4_MAY` — a bit that moves from unnamed to optional is exactly the - // migration this gate exists to force a diff for. - refused("every CPU with AM set", &[(DECLARED.0 | (1 << 18), DECLARED.1); 4], "AM")?; - refused( - "every CPU with PGE set", - &[(DECLARED.0, DECLARED.1 | (1 << 7)); 4], - "never named", - )?; - // The bit that was on before this and asserted nowhere, so deleting `+smep` - // from the launcher or breaking the CPUID gate in `control_regs::supported` - // reddened nothing at all. - refused("every CPU without SMEP", &[(DECLARED.0, DECLARED.1 & !(1 << 20)); 4], "SMEP")?; - // A CPU that agrees about every named bit and differs in one the CPU is - // allowed to withhold, so nothing above it can object. - refused("one CPU with PCID and three without", &[DECLARED, DECLARED, DECLARED, (DECLARED.0, DECLARED.1 | (1 << 17))], "cpu3")?; - // And an AP that never printed at all, which is what a machine whose AP - // died before the check looks like. - refused("three lines for four CPUs", &[DECLARED; 3], "{0, 1, 2, 3}")?; - - eprintln!(" [control_regs] the verdict refuses 10 machines and accepts the declared one"); - Ok(()) -} - /// The negative control, executed: an AP left holding what `INIT` gave it, and /// [`control_regs`] refusing the machine that produces. /// @@ -17443,55 +17332,6 @@ fn idle_is_spinning(serial: &str) -> Option<(u32, u64)> { spread.into_iter().map(|(id, (min, max))| (id, max - min)).find(|&(_, delta)| delta > MAX_IDLE_TRIP_DELTA) } -/// [`idle_is_spinning`] against a healthy trace and a crafted one shaped like -/// the regression it exists to catch, with no guest — the same split -/// `control_regs`/`control_regs_verdict` use, and for the same reason: a -/// gate's own teeth are a claim a live boot cannot demonstrate on the -/// negative side, because nothing in this tree can stage a CPU into spinning -/// through idle on purpose. -/// -/// This is the demonstration the closed vacuous-line-count entry -/// asked for: proof the restored assertion still fails when the condition it -/// names is violated, not just that it still passes when it is not. -fn idle_trip_verdict() -> Result<(), String> { - let healthy = "\ -[kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ -[kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ -[kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=3\n\ -[kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n"; - if let Some((cpu, delta)) = idle_is_spinning(healthy) { - return Err(format!("a healthy trace was refused: cpu{cpu} moved by {delta}")); - } - - // The regression's own shape: one CPU quarantines cleanly and stays - // quiet, the other's undrained ring never lets it halt. - let spinning = "\ -[kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ -[kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=4\n\ -[kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n\ -[kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=2685004\n"; - match idle_is_spinning(spinning) { - Some((1, delta)) if delta > MAX_IDLE_TRIP_DELTA => {} - Some((cpu, delta)) => { - return Err(format!("refused the wrong CPU or by the wrong margin: cpu{cpu} delta {delta}")) - } - None => return Err("a spinning CPU's trace was accepted".to_string()), - } - - // And the line the old, count-of-lines check would have been fooled by: - // the same number of `sched: cpu=` lines either way, because the print - // itself is rate-limited regardless of what is underneath it — which is - // exactly the vacuity this replaces. - assert_eq!( - healthy.matches("sched: cpu=").count(), - spinning.matches("sched: cpu=").count(), - "the crafted traces must differ only in trips=, not in line count — otherwise this proves nothing about the old check's blindness" - ); - - eprintln!(" [i8042] the idle-trip verdict accepts a healthy trace and refuses a spinning one"); - Ok(()) -} - /// Everything the driver derived its DMA pool from, off the two lines it /// prints. Reading these is what makes a test see a *derivation* rather than /// the fact that some number was printed: every fixed cap that ever stood @@ -18044,218 +17884,6 @@ fn headline(reason: Option<&str>) -> String { reason.unwrap_or("check failed").lines().next().unwrap_or("check failed").to_string() } -/// One live guest holds its lane's NVMe image, and the next one may not. -/// -/// **The overlap this stages is the one the shared-boot reboot used to -/// produce.** `qemu = boot()` evaluates its right-hand side first, so the -/// replacement was launched while the guest it replaced still held the lane's -/// `test-nvme-*.img` open for write; QEMU's second process exited 1 on its own -/// image lock, `wait_for_ready` panicked, and the panic escaped the shared -/// block — 129 of one run's 131 reds on one sentence, 2026-08-17. -/// -/// The ordering itself is now the type's: `boot` takes a [`qemu::LaneFree`] and -/// the only thing that makes one out of a guest is `QemuInstance::shutdown`, -/// which takes it by value. What is left to check at runtime is the claim -/// underneath — that a hold is real while a guest is up and gone once it is -/// not — and this checks it on the harness's own registry, in both directions, -/// with no guest. -fn nvme_image_is_held_by_one_guest() -> Result<(), String> { - // Names, not files: a claim is a hold on a path and touches no disk, so - // nothing here has to create or delete a hundred megabytes to ask. - let dir = Path::new(env!("CARGO_TARGET_TMPDIR")); - let image = dir.join("nvme-claim-gate.img"); - let other = dir.join("nvme-claim-gate-other.img"); - - let held = qemu::NvmeClaim::take(&image).map_err(|why| { - format!("a free image refused its first guest: {why}") - })?; - - // The overlap. This is the direction that must red, and it is what the - // reboot produced. - match qemu::NvmeClaim::take(&image) { - Ok(_) => { - return Err(format!( - "a second guest took {}, which a live one is holding — two QEMUs are then \ - handed one image and the second dies on its lock", - image.display() - )) - } - Err(why) => { - // The refusal has to name the image, or it cannot be acted on: a - // run makes dozens of guests and the message is all a reader gets. - if !why.contains(&image.display().to_string()) { - return Err(format!("the refusal does not name the image it is about: {why}")); - } - } - } - - // A different image is not a conflict, or every lane would refuse every - // other lane's boot the moment this gate had teeth. - let elsewhere = qemu::NvmeClaim::take(&other) - .map_err(|why| format!("an unheld image was refused: {why}"))?; - drop(elsewhere); - - // And the ordinary reboot: the replacement takes the image the guest it - // replaces released. Green, and it is the half a fix that simply refused - // every second boot would break. - drop(held); - let replacement = qemu::NvmeClaim::take(&image).map_err(|why| { - format!("a replacement was refused the image its predecessor released: {why}") - })?; - drop(replacement); - Ok(()) -} - -/// A blown guard stays red, and stops reading as an answer. -/// -/// Both halves, because each fails the other's way round. An implementation -/// that made a stall its own non-red status would hide a guest that genuinely -/// stops; one that only renamed the line would leave the summary saying a test -/// found something. Staged against the strings a wait actually produces rather -/// than against the marker on its own, because a caller prefixes its own -/// sentence to [`await_marker`]'s and the classification has to survive that. -fn stall_is_not_a_verdict() -> Result<(), String> { - // Built from the marker rather than copied, so a rename cannot leave the - // gate asserting against a string nothing produces any more. - let real = format!("{STALLED} waiting for the long tone to start — it went quiet"); - let under_a_sentence = format!("the compositor stopped painting\n{real}"); - let cases: [(&str, Option<&str>, bool); 4] = [ - ("an ordinary red", Some("the pointer never moved right"), false), - ("a wait that expired", Some(real.as_str()), true), - ( - "a wait that expired under a caller's own sentence", - Some(under_a_sentence.as_str()), - true, - ), - ("a pass", None, false), - ]; - for (what, reason, want_stall) in cases { - let outcome = Outcome { - name: what.to_string(), - reason: reason.map(str::to_string), - elapsed: Duration::from_secs(1), - suspended: Duration::ZERO, - }; - if outcome.stalled() != want_stall { - return Err(format!( - "{what} reads as stalled={}, and it has to be {want_stall}", - outcome.stalled() - )); - } - // Red is red. A stall that stopped failing the run would be a gate that - // reports and enforces nothing. - let red = outcome.verdict() == Verdict::Fail; - if red != reason.is_some() { - return Err(format!("{what} is red={red}, and a reason is always red")); - } - } - - let mut tally = Tally::new(); - tally.record(Outcome { - name: "a_stalled_test".to_string(), - reason: Some(format!("{STALLED} waiting for nothing at all — it went quiet")), - elapsed: Duration::from_secs(1), - suspended: Duration::ZERO, - }); - tally.record(Outcome { - name: "a_wrong_answer".to_string(), - reason: Some("the pointer never moved right".to_string()), - elapsed: Duration::from_secs(1), - suspended: Duration::ZERO, - }); - if tally.exit_code() != 1 { - return Err(format!("two reds exited {}, and they have to red", tally.exit_code())); - } - if tally.stalls != ["a_stalled_test"] { - return Err(format!( - "the run named {:?} as blown guards; it has to name exactly the one that was", - tally.stalls - )); - } - let summary = tally.summary(2, Duration::from_secs(2), Duration::ZERO); - if !summary.contains("1 of those reds are blown liveness guards") { - return Err(format!("the summary does not separate the two kinds of red:\n{summary}")); - } - Ok(()) -} - -/// Whether a run that did not attempt most of the suite's measured cost says so. -/// -/// **The failure mode the tier introduces is silence, not a wrong answer.** A -/// green run holding back 60 tests and a green run holding back none print the -/// same word, and the difference between them is the whole reason a reach flag -/// exists. Nothing else can see whether the *run* mentions it, and a run nobody -/// can tell apart from a full one is how a temporary measure becomes permanent. -/// -/// Both directions, because the second is the one that rots quietly: a suite -/// that ran everything must not claim to have held anything back either, or the -/// line stops carrying information the day somebody makes it unconditional. -fn nightly_tier_is_announced() -> Result<(), String> { - let held = vec![ - (Tier::Nightly, vec!["desktop_window_child".to_string(), "sshd_exec".to_string()]), - (Tier::Weekly, vec!["iommu_empty_domain".to_string()]), - ]; - let announced = Tally::new().holding_back(held).summary(1, Duration::ZERO, Duration::ZERO); - for want in [ - "not run without --nightly:", - "desktop_window_child, sshd_exec", - "`cargo test --test toyos-build -- --nightly` runs them", - "not run without --weekly:", - "`cargo test --test toyos-build -- --weekly` runs them", - "2 held back for --nightly, 1 held back for --weekly", - ] { - if !announced.contains(want) { - return Err(format!("a run holding tests back never says {want:?}:\n{announced}")); - } - } - let whole = Tally::new().holding_back(vec![(Tier::Weekly, Vec::new())]).summary( - 1, - Duration::ZERO, - Duration::ZERO, - ); - if whole.contains("--") || whole.contains("held back") { - return Err(format!("a run that held nothing back says it did:\n{whole}")); - } - Ok(()) -} - -/// What a suspend is worth to a verdict, staged rather than reasoned about. -/// -/// `common::clock::self_check` gates the detector; this gates what the suite -/// does with what it detects. Both halves are needed and neither implies the -/// other: **a suspend that silently passes is as bad as one that silently -/// fails**, and here the two are one line apart. -fn suspend_invalidates_a_verdict() -> Result<(), String> { - let slept = common::clock::SUSPENDED_AT_LEAST + Duration::from_secs(120); - let awake = Duration::ZERO; - // Under the threshold on purpose: two clock reads jitter against each other - // by microseconds, and a run must not be thrown away for that. - let jitter = common::clock::SUSPENDED_AT_LEAST - .checked_sub(Duration::from_millis(1)) - .expect("SUSPENDED_AT_LEAST must be at least 1ms for this case to mean anything"); - let cases: [(&str, Option<&str>, Duration, Verdict); 6] = [ - ("a pass on a host that stayed up", None, awake, Verdict::Pass), - ("a fail on a host that stayed up", Some("the guest said no"), awake, Verdict::Fail), - ("a pass across a suspend", None, slept, Verdict::Invalid), - ("a fail across a suspend", Some("timed out"), slept, Verdict::Invalid), - ("a pass across clock jitter", None, jitter, Verdict::Pass), - ("a fail across clock jitter", Some("the guest said no"), jitter, Verdict::Fail), - ]; - for (what, reason, suspended, want) in cases { - let outcome = Outcome { - name: what.to_string(), - reason: reason.map(str::to_string), - elapsed: Duration::from_secs(3), - suspended, - }; - let got = outcome.verdict(); - if got != want { - return Err(format!("{what} is {got:?}, and it has to be {want:?}")); - } - } - Ok(()) -} - /// What a run has established, as it establishes it. /// /// One place rather than five counters in `main`, because the interesting part @@ -18475,272 +18103,6 @@ fn run_exit_status() -> Result<(), String> { Ok(()) } -/// A wait that hands a death spelling to a scan of its own, asked of the -/// harness's source. -/// -/// **The one place the vocabulary lives is the whole of the fix, so this is -/// what keeps it the one place.** `tests/common/qemu.rs` holds three waits on a -/// guest and they used to disagree: the boot half ended on three spellings, the -/// test half on one and `await_guest` on none at all — so a Rust `panic!` in the -/// kernel matched nothing while a test was running, the machine halted every -/// CPU, and the guard expired onto a verdict saying the guest had stopped -/// answering. All three ask `serial::died` now, which is the only thing in the -/// harness that knows the words and the only thing that knows the prefix decides -/// whose death they report. -/// -/// The way that comes back is the obvious patch: one more spelling handed -/// straight to a `contains` beside the call. It would match a *program's* panic -/// as readily as the kernel's and take the run down with a guest binary that -/// was expected to die — so the shape is refused by name rather than left to a -/// reviewer. Comment lines go first: this file argues about these words at -/// length, and prose is not a second answer. -fn hand_rolled_deaths(text: &str) -> Vec { - let mut found = Vec::new(); - for (n, line) in text.lines().enumerate() { - if line.trim_start().starts_with("//") { - continue; - } - for word in serial::spellings() { - // The shape is the spelling as somebody's first argument — - // `contains`, `starts_with`, `find`, any of them. A spelling - // *inside* a longer staged line is how this file's own gates build - // their inputs, and those are not scans. - if line.contains(&format!("(\"{word}")) { - found.push(format!("{}:{}: {}", n + 1, word, line.trim())); - } - } - } - found -} - -/// [`hand_rolled_deaths`] over the file that has to stay clean, with its own -/// bad input beside it so a check that stopped finding anything says so. -fn one_vocabulary() -> Result<(), String> { - const FILE: &str = "tests/common/qemu.rs"; - let path = Path::new(env!("CARGO_MANIFEST_DIR")).join(FILE); - let text = fs::read_to_string(&path).map_err(|e| format!("read {}: {e}", path.display()))?; - let found = hand_rolled_deaths(&text); - if !found.is_empty() { - return Err(format!( - "{FILE} scans for a death spelling itself, and `serial::died` is where that is \ - decided for every wait at once — a second answer here is what let a kernel panic \ - read as a stall, and it matches a program's own panic besides:\n {}", - found.join("\n ") - )); - } - // The negative control. Every line of it is a shape this must name, and the - // last two are shapes it must not: prose, and a staged capture built out of - // the same words. - let staged = "\ - } else if line.contains(\"KERNEL PANIC\") {\n\ - if line.starts_with(\"SEGFAULT\") {\n\ - // ends on `PANIC:` and nothing else, which is the defect\n\ - const KERNEL: &str = \"[kernel 1.450 cpu3] PANIC: panicked at reserve.rs:812:9:\";\n"; - let named = hand_rolled_deaths(staged); - if named.len() != 2 { - return Err(format!( - "the check names {} of the two hand-rolled scans staged for it: {named:?}", - named.len() - )); - } - eprintln!(" [vocabulary] {FILE} asks `serial::died` and nothing else"); - Ok(()) -} - -/// What the declaration itself has to be, before any of it means anything. -/// Which shared-boot binaries need `SYS_DEBUG`, asked of their source. -/// -/// A name reaches the syscall directly, or through a child it spawns. -fn needs_actuators(sources: &[(String, String)], registry: &[&str]) -> BTreeSet { - // The fourth spelling is the argument-taking form: every action that - // carries a payload (TLB_ACK_DELAY_ARM, CENSUS_KIND, LOWER_SYSINFO_BOUND, - // SLOT_TO_LAST_GENERATION) is reached through `debug_with`, never - // `debug`. The third is the SDK's: `toyos::census` calls `debug_with` on - // the caller's behalf, so a binary whose leak assertion is a census names - // no syscall of its own and reads as innocent to the others. - let calls = |text: &str| { - text.contains("SYS_DEBUG") - || text.contains("syscall::debug(") - || text.contains("syscall::debug_with(") - || text.contains("census::Census") - }; - let direct: BTreeSet<&str> = - sources.iter().filter(|(_, t)| calls(t)).map(|(n, _)| n.as_str()).collect(); - let mut out = BTreeSet::new(); - for (name, text) in sources { - if !registry.contains(&name.as_str()) { - continue; - } - let spawns = direct.iter().any(|d| text.contains(&format!("test_rs_{d}"))); - if direct.contains(name.as_str()) || spawns { - out.insert(name.clone()); - } - } - out -} - -/// [`ACTUATOR_TESTS`] is exactly the shared-boot binaries that reach -/// `SYS_DEBUG`, and the binaries are what is asked. -/// -/// **What this does not cover, stated because the hole is real:** a machine or -/// screen test that *drives* one of those binaries on a boot of its own. No -/// static rule here can say which `BootOptions` a `run_test` call belongs to. -/// What answers it instead is the guest: `test_panic_child` names -/// `InvalidArgument` as *this kernel carries no actuators* rather than reporting -/// a kernel that failed to stop, so the red says what is wrong wherever it -/// happens. -/// -/// **Both directions are the point.** A binary that gains a `debug()` call and -/// no entry would run on the shipping kernel, where the syscall answers -/// `InvalidArgument` — and a test whose verdict is that a process died would -/// then fail for a reason with nothing to do with what it is about. An entry -/// whose binary no longer calls it is a test kept off the shipping kernel for -/// nothing, which is the erosion this split exists to stop. -fn suite_split() -> Result<(), String> { - let dir = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/toyos-rust-tests/src/bin"); - let mut sources: Vec<(String, String)> = Vec::new(); - for entry in fs::read_dir(&dir).map_err(|e| format!("read {}: {e}", dir.display()))? { - let path = entry.map_err(|e| e.to_string())?.path(); - if path.extension().is_none_or(|e| e != "rs") { - continue; - } - let name = path.file_stem().unwrap().to_string_lossy().into_owned(); - let text = fs::read_to_string(&path).map_err(|e| format!("read {name}: {e}"))?; - sources.push((name, text)); - } - let registry: Vec<&str> = - sources.iter().map(|(n, _)| n.as_str()).filter(|n| !RUST_SKIP.contains(n)).collect(); - - // The negative control, and it carries its own bad input: a binary that - // calls the syscall and is on no list must be named, or the check above is - // a spelling of `true`. - let staged = vec![ - ("a_listed_one".to_string(), "syscall::debug(3)".to_string()), - ("an_unlisted_one".to_string(), "SYS_DEBUG".to_string()), - ("its_parent".to_string(), "Command::new(\"/system/bin/test_rs_an_unlisted_one\")".to_string()), - ("a_censor".to_string(), "use toyos::census::Census;".to_string()), - ("a_debug_with_user".to_string(), "syscall::debug_with(3, 4)".to_string()), - ("innocent".to_string(), "println!()".to_string()), - ]; - let staged_registry = [ - "a_listed_one", - "an_unlisted_one", - "its_parent", - "a_censor", - "a_debug_with_user", - "innocent", - ]; - let found = needs_actuators(&staged, &staged_registry); - let want: BTreeSet = - ["a_listed_one", "an_unlisted_one", "its_parent", "a_censor", "a_debug_with_user"] - .iter() - .map(|s| s.to_string()) - .collect(); - if found != want { - return Err(format!("the check does not work: on staged input it named {found:?}")); - } - - let want: BTreeSet = needs_actuators(&sources, ®istry); - let listed: BTreeSet = ACTUATOR_TESTS.iter().map(|s| s.to_string()).collect(); - let missing: Vec<&String> = want.difference(&listed).collect(); - if !missing.is_empty() { - return Err(format!( - "{missing:?} reach SYS_DEBUG and are on the shipping boot, where the syscall answers \ - InvalidArgument. Add each to ACTUATOR_TESTS, or to RUST_SKIP if it is driven rather \ - than run." - )); - } - let stale: Vec<&String> = listed.difference(&want).collect(); - if !stale.is_empty() { - return Err(format!( - "{stale:?} are held off the shipping kernel and no longer reach SYS_DEBUG. Delete \ - each entry — coverage of the binary an image ships is what it costs." - )); - } - // **The other shape [`schedule`] cannot see**: a binary a machine - // test drives under a *different* name is still discovered here, still runs - // on the shared boot, and there passes on its exit code with nothing staged - // for it to act on. - let staged_driven = [ - String::from("qemu.run_test(\"test_rs_a_driven_one\", Duration::from_secs(30))"), - String::from("Command::new(\"/system/bin/test_rs_another_driven\")"), - String::from("qemu.run_test(&format!(\"test_rs_{name}\"), ceiling)"), - ]; - let found = driven_binaries(&staged_driven); - let want: BTreeSet = - ["a_driven_one", "another_driven"].iter().map(|s| s.to_string()).collect(); - if found != want { - return Err(format!("the driven-name reader does not work: it named {found:?}")); - } - - let root = Path::new(env!("CARGO_MANIFEST_DIR")); - let mut harness = vec![fs::read_to_string(root.join("tests/toyos.rs")) - .map_err(|e| format!("read tests/toyos.rs: {e}"))?]; - let common = root.join("tests/common"); - for entry in fs::read_dir(&common).map_err(|e| format!("read {}: {e}", common.display()))? { - let path = entry.map_err(|e| e.to_string())?.path(); - if path.extension().is_some_and(|e| e == "rs") { - harness.push(fs::read_to_string(&path).map_err(|e| e.to_string())?); - } - } - let shared: BTreeSet<&str> = registry.iter().copied().collect(); - let both: BTreeSet = driven_binaries(&harness) - .into_iter() - .filter(|name| shared.contains(name.as_str())) - .collect(); - let declared: BTreeSet = DRIVEN_AND_SHARED.iter().map(|s| s.to_string()).collect(); - let undeclared: Vec<&String> = both.difference(&declared).collect(); - if !undeclared.is_empty() { - return Err(format!( - "{undeclared:?} are driven by a machine test and also run on the shared boot, where \ - nothing stages what they need — so each passes on its exit code with no verdict. Add \ - each to RUST_SKIP with the reason its driver exists, or to DRIVEN_AND_SHARED if its \ - shared run asserts something of its own." - )); - } - let stale: Vec<&String> = declared.difference(&both).collect(); - if !stale.is_empty() { - return Err(format!( - "DRIVEN_AND_SHARED names {stale:?}, which no machine test drives or the shared boot \ - no longer runs. Delete each entry — a declaration nothing is true of is what makes \ - the rest of the list unreadable." - )); - } - - println!( - " [split] {} shared binaries on the shipping kernel, {} on the actuator one, {} of them \ - driven elsewhere and declared", - registry.len() - listed.len(), - listed.len(), - both.len() - ); - Ok(()) -} - -/// Every guest binary the harness drives by name, read out of its own sources. -/// -/// A driver reaches a binary as the literal `test_rs_`, so that is what -/// says a binary has one; a `format!` over a variable name yields no literal -/// and is not a driver of any particular binary. Pure, and the sources are a -/// parameter, so `suite_split` stages its own before trusting it on the tree. -fn driven_binaries(sources: &[String]) -> BTreeSet { - const MARK: &str = "test_rs_"; - let mut found = BTreeSet::new(); - for text in sources { - let mut rest = text.as_str(); - while let Some(at) = rest.find(MARK) { - rest = &rest[at + MARK.len()..]; - let end = rest - .find(|c: char| !c.is_ascii_alphanumeric() && c != '_') - .unwrap_or(rest.len()); - if end > 0 { - found.insert(rest[..end].to_string()); - } - } - } - found -} - /// Which of the two shared boots a name belongs on — a *kernel build*, because /// `SYS_DEBUG` is compiled in or it is not, and never a boot parameter. fn shared_kernel(name: &str) -> &'static [&'static str] { @@ -20207,65 +19569,3 @@ fn root_withheld_refused(log: &str) -> Result<(), String> { eprintln!(" [root] a handoff with no ROOT image refused the boot by name"); Ok(()) } - -/// The harness's own checks: none boots a guest, so each is a libtest test of -/// `tests/checks.rs` and runs on every host gate rather than in a guest tier. -#[cfg(test)] -mod checks { - use super::*; - - /// One subject: what a console line says died, what a wait does about it, - /// and that only one place in the harness answers either. - #[test] - fn serial_vocabulary() -> Result<(), String> { - serial::self_check()?; - qemu::ceiling_self_check()?; - qemu::host_scale_self_check()?; - one_vocabulary() - } - - #[test] - fn suspend_detector() -> Result<(), String> { - common::clock::self_check() - } - - #[test] - fn suspend_invalidates_a_verdict() -> Result<(), String> { - super::suspend_invalidates_a_verdict() - } - - #[test] - fn stall_is_not_a_verdict() -> Result<(), String> { - super::stall_is_not_a_verdict() - } - - #[test] - fn nvme_image_is_held_by_one_guest() -> Result<(), String> { - super::nvme_image_is_held_by_one_guest() - } - - #[test] - fn control_regs_verdict() -> Result<(), String> { - super::control_regs_verdict() - } - - #[test] - fn i8042_quarantine_verdict() -> Result<(), String> { - idle_trip_verdict() - } - - #[test] - fn suite_split() -> Result<(), String> { - super::suite_split() - } - - #[test] - fn nightly_tier_is_announced() -> Result<(), String> { - super::nightly_tier_is_announced() - } - - #[test] - fn screen_decoder() { - screen::self_test(); - } -} From 3b94e0667a1991875081656bdd8dd3d9cf1a245b Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 10:27:37 +0200 Subject: [PATCH 14/18] issues: the loom table loses the row of the model that went with its actuator `the_asker_owns_the_report_it_asks_for` modelled `DumpRequest::file_and_take`, which went with `quiesce-dump`, so its execution count describes nothing. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...ever-moves-a-store-before-a-load-another-thread-made-first.md | 1 - 1 file changed, 1 deletion(-) diff --git a/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md b/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md index 46b05b8ffdd..10cf1825b79 100644 --- a/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md +++ b/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md @@ -39,7 +39,6 @@ the two `Builder::check` helpers: |---|---|---| | `kernel-loom` `dump_request` `a_request_filed_during_a_report_is_reported` | `DumpRequest::update`'s load | 16 | | `dump_request` `one_request_is_taken_once` | `update`'s load | 28 | -| `dump_request` `the_asker_owns_the_report_it_asks_for` | `update`'s load | 19 | | `dump_request` `a_request_is_announced_at_most_once` | `update`'s load | 5777 | | `dump_request` `a_request_left_during_a_report_is_still_taken_by_its_end` | `update`'s load; joined at once | 1 | | `log_ring` `a_published_record_is_whole_and_read_once` | `push`'s `tail_then_head` | 1442 | From 97f53b581dca3fb624988e535db70526f4b6898e Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 12:51:48 +0200 Subject: [PATCH 15/18] tests: run_exit_status is a host test; a settled join takes no table lock `run_exit_status` boots nothing and gates `Tally::exit_code` and the result line this branch rewrote, so it becomes a `#[test]` in `tests/checks.rs` and its guest row, dispatch arm and duration line go. Under a patch making `exit_code` return 0 for a suspended-only tally it reds with "a suspended run exits 0, and it has to be 2". `Join::ask` takes the collect as a closure and calls it only while the join is unsettled; `process::ask_join` takes `PROCESS_TABLE` inside that closure, so a settled join answers without the lock, and the decision is `toyos-proclife`'s where a host test reaches it. `a_settled_join_does_not_take_the_table_again` reds when the collect runs on a settled join, with the answer still kept. The issue that recorded the regression is closed. `DRIVEN_AND_SHARED` returns beside `RUST_SKIP` in `tests/toyos.rs`, the partition's two halves in one file, byte-identical to main's block; the libtest-free target does not read it, so it carries the one `allow`. `elf/cache.rs` uses `BUDGET_BYTES` itself. The restored defect issues each name the reproducer at `1808fb8d` their fix is shown against, and home-budget names its owner. Chronology in `tests/checks.rs`, the `screen_decoder` note under `SCREEN_TESTS` and the fix's story in the loom issue are deleted. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...ty-loses-five-of-a-thousand-lines-on-ci.md | 5 +- ...before-a-load-another-thread-made-first.md | 4 +- ...refusal-retried-is-red-on-every-nightly.md | 5 +- ...in-takes-the-process-table-to-ask-again.md | 16 --- ...sals-saw-the-kernel-refuse-nothing-once.md | 4 +- kernel/src/elf/cache.rs | 7 +- kernel/src/process.rs | 5 +- tests/checks.rs | 105 ++++++++---------- tests/test-durations | 1 - tests/toyos.rs | 80 ++++++------- toyos-proclife/src/join.rs | 40 +++++-- 11 files changed, 127 insertions(+), 145 deletions(-) delete mode 100644 issues/kernel/a-settled-thread-join-takes-the-process-table-to-ask-again.md diff --git a/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md b/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md index afc9395912b..08152da71bb 100644 --- a/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md +++ b/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md @@ -20,5 +20,8 @@ Split out of `issues/build/parallel-tests-red-under-other-suites.md`, whose rate table was never this test's — CI's single-guest-per-machine shards rule out the contention shape that file is about. -**Exit condition.** The lost lines' cause is fixed. +**Exit condition.** The lost lines' cause is fixed, shown against +`console_line_atomicity` as it stands at `1808fb8d` (its binary, the test +runner's `CONSOLE_JOBS` stdin and its harness arm), restored and green on CI's +`guest` shards, one guest per machine. Owner: orchestrator. diff --git a/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md b/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md index 10cf1825b79..aec7607822d 100644 --- a/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md +++ b/issues/build/loom-never-moves-a-store-before-a-load-another-thread-made-first.md @@ -23,9 +23,7 @@ explored 1; the reader spawned beside the adder on the model's thread explored `kernel-loom/tests/i8042_tally.rs` had that shape: the model's thread read the tally, and the spawned ISR's `Tally::record` opens with its saturation check's load. Its models ran one execution, and a mutation that counted an empty -interrupt as a carrying one passed them. They now spawn the reader and run the -ISR on the model's thread, so the reader's pending load is raced against the -ISR's write, and they refuse a run of one execution (`explored`). +interrupt as a carrying one passed them. A spawned thread that opens with a load is not the trigger on its own. Every model below has one, and each explores more than one execution unless it joins diff --git a/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md b/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md index 8e9c210e81f..8732d0b086d 100644 --- a/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md +++ b/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md @@ -27,4 +27,7 @@ issues/boot-media/partition-claim-gives-up-reds-beside-other-guests-and-is-green names as racing whichever fsync the boot reaches first. Not shown: whether that race is this test's cause. -**Exit**: the cause shown on a red run's log. +**Exit**: the cause shown on the log of a red run of `home_budget_refusal_retried` +as it stands at `1808fb8d` (its binary `test_rs_home_fsync_budget` and its +`storage` body), restored and run on CI's nightly `guest` shards; and that +test green on a nightly after the fix. Owner: orchestrator. diff --git a/issues/kernel/a-settled-thread-join-takes-the-process-table-to-ask-again.md b/issues/kernel/a-settled-thread-join-takes-the-process-table-to-ask-again.md deleted file mode 100644 index 6fc42171c42..00000000000 --- a/issues/kernel/a-settled-thread-join-takes-the-process-table-to-ask-again.md +++ /dev/null @@ -1,16 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-28 ---- - -# A settled thread join takes the process table to ask again - -`sys_thread_join` (`kernel/src/syscall/proc.rs`) asks through -`process::ask_join`, which locks `PROCESS_TABLE` before `Join::ask` looks at -whether an earlier ask settled. A join whose wait's predicate collected the -zombie then asks once more at the top of its loop, so it takes the machine-wide -table lock for an answer it already holds. Found by PR #564's round-2 review. - -**Exit**: a settled join answers without taking `PROCESS_TABLE`. Owner: -orchestrator. diff --git a/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md b/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md index 30f92133053..16c73122407 100644 --- a/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md +++ b/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md @@ -15,4 +15,6 @@ twelve 2 MiB images entered a cache whose test budget refuses at the second. Owed: a mechanism. Nobody has one. -**Exit condition.** The cause of the missing refusal is fixed. Owner: orchestrator. +**Exit condition.** The cause of the missing refusal is fixed, shown against +`so_cache_refusals` and the `so-cache-tiny` budget it arms, as both stand at +`1808fb8d`, restored and green on CI's KVM `guest` shards. Owner: orchestrator. diff --git a/kernel/src/elf/cache.rs b/kernel/src/elf/cache.rs index 7af3d804e90..7eb03a5d926 100644 --- a/kernel/src/elf/cache.rs +++ b/kernel/src/elf/cache.rs @@ -228,15 +228,14 @@ pub fn cache_loaded_lib( return outcome.map(|cloned| cloned.unwrap_or_else(|| owned(alloc))); } // Under the lock that publishes: two concurrent loads must not both find room. - let budget = BUDGET_BYTES; let (held, entries) = (held_bytes(&cache), cache.len()); - let Some(after) = held.checked_add(alloc.size()).filter(|b| *b <= budget) else { + let Some(after) = held.checked_add(alloc.size()).filter(|b| *b <= BUDGET_BYTES) else { drop(cache); // Both allocations drop here, so the refusal gives back what the load took. log!( "dlopen: {} would take the shared-object cache to {} bytes over {} entries, past its \ {}-byte budget; refused, and nothing is evicted for it", - path, held.saturating_add(alloc.size()), entries + 1, budget + path, held.saturating_add(alloc.size()), entries + 1, BUDGET_BYTES ); return Err(SyscallError::ResourceExhausted); }; @@ -248,7 +247,7 @@ pub fn cache_loaded_lib( log!( "dlopen: cached {} with {} bind + {} tpoff64 + {} tpoff32 + {} dtpmod64 + {} dtpoff64 pre-scanned relocs, cache now {} of {} bytes", path, relocs.bind.len(), relocs.tpoff64.len(), relocs.tpoff32.len(), - relocs.dtpmod64.len(), relocs.dtpoff64.len(), after, budget + relocs.dtpmod64.len(), relocs.dtpoff64.len(), after, BUDGET_BYTES ); Ok(snapshot.into_lib( diff --git a/kernel/src/process.rs b/kernel/src/process.rs index 63f7ddc028d..c35fcad6070 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -1271,10 +1271,9 @@ pub fn wake_pipe_writers(pipe_id: pipe::PipeId) { scheduler::wake_pipe_writers(pipe_id); } -/// Ask `join` once more under the table lock, which is what makes validating the parent-thread relationship and collecting the zombie one act. +/// Ask `join` once more; an unsettled one collects under the table lock, which is what makes validating the parent-thread relationship and collecting the zombie one act. pub fn ask_join(join: &mut join::Join, tid: Tid, parent_pid: Pid) -> Option> { - let mut guard = PROCESS_TABLE.lock(); - join.ask(guard.as_mut().unwrap(), parent_pid, tid) + join.ask(|| join::collect_zombie(PROCESS_TABLE.lock().as_mut().unwrap(), parent_pid, tid)) } /// Handle a page fault at `fault_addr` by looking up the current process's VMAs. Returns whether the fault was resolved. diff --git a/tests/checks.rs b/tests/checks.rs index 56804e5db1a..4a231af7629 100644 --- a/tests/checks.rs +++ b/tests/checks.rs @@ -21,14 +21,7 @@ mod checks { /// harness's source. /// /// **The one place the vocabulary lives is the whole of the fix, so this is - /// what keeps it the one place.** `tests/common/qemu.rs` holds three waits on a - /// guest and they used to disagree: the boot half ended on three spellings, the - /// test half on one and `await_guest` on none at all — so a Rust `panic!` in the - /// kernel matched nothing while a test was running, the machine halted every - /// CPU, and the guard expired onto a verdict saying the guest had stopped - /// answering. All three ask `serial::died` now, which is the only thing in the - /// harness that knows the words and the only thing that knows the prefix decides - /// whose death they report. + /// what keeps it the one place.** /// /// The way that comes back is the obvious patch: one more spelling handed /// straight to a `contains` beside the call. It would match a *program's* panic @@ -208,13 +201,6 @@ mod checks { /// One live guest holds its lane's NVMe image, and the next one may not. /// - /// **The overlap this stages is the one the shared-boot reboot used to - /// produce.** `qemu = boot()` evaluates its right-hand side first, so the - /// replacement was launched while the guest it replaced still held the lane's - /// `test-nvme-*.img` open for write; QEMU's second process exited 1 on its own - /// image lock, `wait_for_ready` panicked, and the panic escaped the shared - /// block — 129 of one run's 131 reds on one sentence, 2026-08-17. - /// /// The ordering itself is now the type's: `boot` takes a [`qemu::LaneFree`] and /// the only thing that makes one out of a guest is `QemuInstance::shutdown`, /// which takes it by value. What is left to check at runtime is the claim @@ -278,8 +264,6 @@ mod checks { /// one bit away from it. #[test] fn control_regs_verdict() -> Result<(), String> { - /// The pre-fix machine, `smp=4`, TCG, read off this tree on 2026-08-08: - /// firmware's registers on the BSP and INIT's on every AP. const AP_BEFORE: (u64, u64) = (0xe000_0011, 0x0031_0620); const DECLARED: (u64, u64) = (0x8001_0033, 0x0030_0668); @@ -323,19 +307,13 @@ mod checks { // Two bits a machine could hold uniformly, each one line of kernel diff // away, and neither reachable by an actuator. `AM` is named clear above and // answers by name; `PGE` is named nowhere, which is the case the whole - // never-named rule exists for — `TSD` and `PKE` are the same case. `UMIP` - // used to be this file's example of the same thing, until it joined - // `CR4_MAY` — a bit that moves from unnamed to optional is exactly the - // migration this gate exists to force a diff for. + // never-named rule exists for — `TSD` and `PKE` are the same case. refused("every CPU with AM set", &[(DECLARED.0 | (1 << 18), DECLARED.1); 4], "AM")?; refused( "every CPU with PGE set", &[(DECLARED.0, DECLARED.1 | (1 << 7)); 4], "never named", )?; - // The bit that was on before this and asserted nowhere, so deleting `+smep` - // from the launcher or breaking the CPUID gate in `control_regs::supported` - // reddened nothing at all. refused("every CPU without SMEP", &[(DECLARED.0, DECLARED.1 & !(1 << 20)); 4], "SMEP")?; // A CPU that agrees about every named bit and differs in one the CPU is // allowed to withhold, so nothing above it can object. @@ -354,10 +332,6 @@ mod checks { /// gate's own teeth are a claim a live boot cannot demonstrate on the /// negative side, because nothing in this tree can stage a CPU into spinning /// through idle on purpose. - /// - /// This is the demonstration the closed vacuous-line-count entry - /// asked for: proof the restored assertion still fails when the condition it - /// names is violated, not just that it still passes when it is not. #[test] fn i8042_quarantine_verdict() -> Result<(), String> { let healthy = "\ @@ -398,39 +372,6 @@ mod checks { Ok(()) } - /// Binaries a machine test drives that the shared boot also runs on purpose. - /// - /// **A binary a machine test drives under a different name is still discovered - /// by [`discover_rust_tests`]**, still runs on the shared boot, and there - /// passes on its exit code with nothing staged for it to act on. `RUST_SKIP` is - /// one answer to that; this list is the other, for the binaries whose shared - /// run asserts something of its own. Every driven name is on one list or the - /// other, so neither answer is silence — `suite_split` is the gate. - /// `sched_stress` is the one whose two runs differ by *kernel* rather than by - /// what the host staged: the shipping build here, `sched_check_build`'s - /// assert-carrying build there. - const DRIVEN_AND_SHARED: &[&str] = &[ - // The lost-wake canary: its shared run is the count on the shipping - // kernel with nothing staged, and `blocking_read_window` drives it again - // with the watch's window held open. - "blocking_read_stress", - // The log-stream arms drive it for the kernel's `exit:` record about it, - // not for anything it does: it is the cheapest process this tree starts. - "empty_dir_stat", - // Its shared run is a whole handle-lifecycle gate with its own census; - // `userdev_dma_fault` drives the same binary for a different reason - // entirely — as the proof the machine still schedules and spawns after a - // device was refused at the unit — and stages nothing for it. - "handle_basic", - "hierarchy_paths", - "null_sink_client_exits", - "nvme_home_roundtrip", - "sched_stress", - "std_alloc", - "std_mmap", - "wall_clock_now", - ]; - /// What the declaration itself has to be, before any of it means anything. /// Which shared-boot binaries need `SYS_DEBUG`, asked of their source. /// @@ -626,6 +567,48 @@ mod checks { found } + /// What a whole run exits with, and what its last line says. + /// + /// Driven through [`Tally`] rather than asserted about it: the property that + /// matters is what `--land`'s gate reads off the process, and that is the exit + /// code after `record` has seen every outcome. + #[test] + fn run_exit_status() -> Result<(), String> { + let outcome = |name: &str, reason: Option<&str>, suspended: Duration| Outcome { + name: name.to_string(), + reason: reason.map(str::to_string), + elapsed: Duration::from_secs(3), + suspended, + }; + let slept = common::clock::SUSPENDED_AT_LEAST + Duration::from_secs(120); + + let mut red = Tally::new(); + red.record(outcome("a_red", Some("the disk came back short"), Duration::ZERO)); + red.record(outcome("a_suspended_one", None, slept)); + if red.exit_code() != 1 { + return Err(format!("a run with a red exits {}, and it has to be 1", red.exit_code())); + } + + let mut suspended = Tally::new(); + suspended.record(outcome("a_suspended_one", None, slept)); + if suspended.exit_code() != 2 { + return Err(format!("a suspended run exits {}, and it has to be 2", suspended.exit_code())); + } + + // The clean case, so that none of the above is passing because everything + // reds. + let mut clean = Tally::new(); + clean.record(outcome("a_green", None, Duration::ZERO)); + let text = clean.summary(1, Duration::from_secs(9), Duration::ZERO); + if clean.exit_code() != 0 { + return Err(format!("a clean run exits {}, and it has to be 0", clean.exit_code())); + } + if !text.lines().last().unwrap_or_default().starts_with("test result: ok.") { + return Err(format!("a clean run does not say so plainly:\n{text}")); + } + Ok(()) + } + /// Whether a run that did not attempt most of the suite's measured cost says so. /// /// **The failure mode the tier introduces is silence, not a wrong answer.** A diff --git a/tests/test-durations b/tests/test-durations index f8dcc8577c4..c32248eab77 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -182,7 +182,6 @@ empty_dir_stat 12 endowment_denied 10 esp_filesystem 10123 exit_wait_storm 178 -run_exit_status 0 fat_backing_revoked 8226 fault_gates 31 foreign_disk_untouched 4688 diff --git a/tests/toyos.rs b/tests/toyos.rs index 638e78fc261..b6411179d31 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -453,6 +453,40 @@ const RUST_SKIP: &[&str] = &[ "pkg_launch_gbae", ]; +/// Binaries a machine test drives that the shared boot also runs on purpose. +/// +/// **A binary a machine test drives under a different name is still discovered +/// by [`discover_rust_tests`]**, still runs on the shared boot, and there +/// passes on its exit code with nothing staged for it to act on. `RUST_SKIP` is +/// one answer to that; this list is the other, for the binaries whose shared +/// run asserts something of its own. Every driven name is on one list or the +/// other, so neither answer is silence — `suite_split` is the gate. +/// `sched_stress` is the one whose two runs differ by *kernel* rather than by +/// what the host staged: the shipping build here, `sched_check_build`'s +/// assert-carrying build there. +#[allow(dead_code, reason = "`suite_split` reads it, in `toyos-checks` alone")] +const DRIVEN_AND_SHARED: &[&str] = &[ + // The lost-wake canary: its shared run is the count on the shipping + // kernel with nothing staged, and `blocking_read_window` drives it again + // with the watch's window held open. + "blocking_read_stress", + // The log-stream arms drive it for the kernel's `exit:` record about it, + // not for anything it does: it is the cheapest process this tree starts. + "empty_dir_stat", + // Its shared run is a whole handle-lifecycle gate with its own census; + // `userdev_dma_fault` drives the same binary for a different reason + // entirely — as the proof the machine still schedules and spawns after a + // device was refused at the unit — and stages nothing for it. + "handle_basic", + "hierarchy_paths", + "null_sink_client_exits", + "nvme_home_roundtrip", + "sched_stress", + "std_alloc", + "std_mmap", + "wall_clock_now", +]; + // Audio glitch tests. Each runs in its own QEMU boot per SMP config and // asserts on the wav the virtio-sound device captured, so they are excluded // from the shared multi-test boot. @@ -474,8 +508,6 @@ const EARLY_PANIC_MESSAGE: &str = "test-early-panic: on-screen console check"; // used to read a screendump now reads the console instead — a screenshot is a // poor way to ask "did the right process come up", and thresholds over a live // desktop are how those tests passed vacuously twice. -// `screen_decoder` needs no guest at all; it proves the decoder against a -// bitmap it rendered itself, before anything points it at a real screen. /// The order was once about kernel rebuilds — every actuator was a build, and a /// feature-carrying test last left the plain-kernel ones above it untouched by /// the thrash. There are two kernels now and nothing to thrash; the order is @@ -1418,8 +1450,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // and its counters rather than a capture, so it runs wide. ("hda_client_stall", Sched::Parallel, Tier::Nightly), ("hda_two_live_refused", Sched::Parallel, Tier::Weekly), - // Host-side, no guest: what a whole run exits with. - ("run_exit_status", Sched::Parallel, Tier::Nightly), ]; /// The test binaries a [`MACHINE_TESTS`] or [`SCREEN_TESTS`] entry runs, which @@ -12478,7 +12508,6 @@ fn run_machine_test( ); Ok(()) } - "run_exit_status" => run_exit_status(), "nvme_wide_sector" => { // The other half of "a device's size is a shape dimension": not how // many sectors, but how big one is. `lba_ds` is an 8-bit @@ -18062,47 +18091,6 @@ impl Tally { } } -/// What a whole run exits with, and what its last line says. -/// -/// Driven through [`Tally`] rather than asserted about it: the property that -/// matters is what `--land`'s gate reads off the process, and that is the exit -/// code after `record` has seen every outcome. -fn run_exit_status() -> Result<(), String> { - let outcome = |name: &str, reason: Option<&str>, suspended: Duration| Outcome { - name: name.to_string(), - reason: reason.map(str::to_string), - elapsed: Duration::from_secs(3), - suspended, - }; - let slept = common::clock::SUSPENDED_AT_LEAST + Duration::from_secs(120); - - let mut red = Tally::new(); - red.record(outcome("a_red", Some("the disk came back short"), Duration::ZERO)); - red.record(outcome("a_suspended_one", None, slept)); - if red.exit_code() != 1 { - return Err(format!("a run with a red exits {}, and it has to be 1", red.exit_code())); - } - - let mut suspended = Tally::new(); - suspended.record(outcome("a_suspended_one", None, slept)); - if suspended.exit_code() != 2 { - return Err(format!("a suspended run exits {}, and it has to be 2", suspended.exit_code())); - } - - // The clean case, so that none of the above is passing because everything - // reds. - let mut clean = Tally::new(); - clean.record(outcome("a_green", None, Duration::ZERO)); - let text = clean.summary(1, Duration::from_secs(9), Duration::ZERO); - if clean.exit_code() != 0 { - return Err(format!("a clean run exits {}, and it has to be 0", clean.exit_code())); - } - if !text.lines().last().unwrap_or_default().starts_with("test result: ok.") { - return Err(format!("a clean run does not say so plainly:\n{text}")); - } - Ok(()) -} - /// Which of the two shared boots a name belongs on — a *kernel build*, because /// `SYS_DEBUG` is compiled in or it is not, and never a boot parameter. fn shared_kernel(name: &str) -> &'static [&'static str] { diff --git a/toyos-proclife/src/join.rs b/toyos-proclife/src/join.rs index 5505b4bed02..dfc66b8f327 100644 --- a/toyos-proclife/src/join.rs +++ b/toyos-proclife/src/join.rs @@ -57,11 +57,16 @@ pub fn collect_zombie( pub struct Join(Option>); impl Join { - /// Ask the table, unless an earlier ask settled; `None` is still waiting. + /// Ask `collect` — [`collect_zombie`] under the table lock — unless an + /// earlier ask settled; `None` is still waiting. A settled join never calls + /// `collect`, so it never takes the table lock again. #[must_use = "a settled join is the answer the syscall returns"] - pub fn ask(&mut self, table: &mut T, pid: Pid, tid: Tid) -> Option> { + pub fn ask( + &mut self, + collect: impl FnOnce() -> Result, JoinRefused>, + ) -> Option> { if self.0.is_none() { - self.0 = collect_zombie(table, pid, tid).transpose(); + self.0 = collect().transpose(); } self.0 } @@ -100,17 +105,36 @@ mod tests { let pid = world.spawn_process(); let t1 = world.spawn_thread(pid); let mut join = Join::default(); - assert_eq!(join.ask(&mut world, pid, t1), None); + assert_eq!(join.ask(|| collect_zombie(&mut world, pid, t1)), None); world.set_location(pid, t1, ThreadLocation::Zombie(11)); - assert_eq!(join.ask(&mut world, pid, t1), Some(Ok(11))); + assert_eq!(join.ask(|| collect_zombie(&mut world, pid, t1)), Some(Ok(11))); assert_eq!( - join.ask(&mut world, pid, t1), + join.ask(|| collect_zombie(&mut world, pid, t1)), Some(Ok(11)), "a join that collected its thread answered the ask after it with no such thread", ); let mut refused = Join::default(); - assert_eq!(refused.ask(&mut world, pid, Tid(9)), Some(Err(JoinRefused::NoSuchThread))); - assert_eq!(refused.ask(&mut world, pid, Tid(9)), Some(Err(JoinRefused::NoSuchThread))); + assert_eq!(refused.ask(|| collect_zombie(&mut world, pid, Tid(9))), Some(Err(JoinRefused::NoSuchThread))); + assert_eq!(refused.ask(|| collect_zombie(&mut world, pid, Tid(9))), Some(Err(JoinRefused::NoSuchThread))); + } + + /// The syscall's wait asks after every wake, and `collect` is where the + /// kernel takes `PROCESS_TABLE`: a settled join answers without it. + #[test] + fn a_settled_join_does_not_take_the_table_again() { + let mut world = World::new(); + let pid = world.spawn_process(); + let t1 = world.spawn_thread(pid); + world.set_location(pid, t1, ThreadLocation::Zombie(11)); + let mut join = Join::default(); + assert_eq!(join.ask(|| collect_zombie(&mut world, pid, t1)), Some(Ok(11))); + assert_eq!(join.ask(|| panic!("a settled join took the table lock to ask again")), Some(Ok(11))); + let mut refused = Join::default(); + assert_eq!(refused.ask(|| collect_zombie(&mut world, pid, Tid(9))), Some(Err(JoinRefused::NoSuchThread))); + assert_eq!( + refused.ask(|| panic!("a refused join took the table lock to ask again")), + Some(Err(JoinRefused::NoSuchThread)), + ); } #[test] From b1d847266373e3344791d0aeec902643eb2d923c Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 15:21:31 +0200 Subject: [PATCH 16/18] A join holds its own answer; DRIVEN_AND_SHARED exists only where it is read `toyos_proclife::join::Join` keeps its answer in a `Cell` and is asked through `&self`, so `sys_thread_join` holds one join and its `ask` closure borrows it: there is no copy to take out and write back after each ask. `a_join_asked_after_it_collected_keeps_its_answer` now runs the syscall's shape, one join shared by the first ask, the wait's predicate and the ask after the wait, so the mutation that keeps no answer reds a host test and `issues/build/a-thread-join-keeping-its-answer-is-gated-only-by-a-nightly-guest.md` meets its exit and goes. `issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md` records that whether `ask_join` takes `PROCESS_TABLE` inside `collect` is held by no test. `DRIVEN_AND_SHARED` is `#[cfg(test)]` rather than allowed dead, so it is compiled only into `toyos-checks`, and the lint names it if `suite_split` stops reading it. `run_exit_status`'s doc loses the sentence about `--land`'s gate, which is retired. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...he-table-lock-again-is-gated-by-nothing.md | 19 +++++++ ...answer-is-gated-only-by-a-nightly-guest.md | 19 ------- kernel/src/process.rs | 2 +- kernel/src/syscall/proc.rs | 9 +-- tests/checks.rs | 4 -- tests/toyos.rs | 2 +- toyos-proclife/src/join.rs | 55 +++++++++++-------- 7 files changed, 56 insertions(+), 54 deletions(-) create mode 100644 issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md delete mode 100644 issues/build/a-thread-join-keeping-its-answer-is-gated-only-by-a-nightly-guest.md diff --git a/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md b/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md new file mode 100644 index 00000000000..ed9257bf515 --- /dev/null +++ b/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md @@ -0,0 +1,19 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# A settled thread join taking the table lock again is gated by nothing + +`toyos_proclife::join::Join::ask` calls its `collect` only while the join is +unsettled, and `a_settled_join_does_not_take_the_table_again` holds that. +Whether `collect` is where the kernel takes `PROCESS_TABLE` is +`kernel/src/process.rs`'s `ask_join`, in no crate a host test compiles. PR +#564's round-4 review wrote it as +`let mut g = PROCESS_TABLE.lock(); join.ask(|| join::collect_zombie(g.as_mut().unwrap(), parent_pid, tid))`: +every test passed. A settled join then takes the lock on every wake of its +wait, which costs contention on `PROCESS_TABLE` and not a wrong answer. + +**Exit**: taking `PROCESS_TABLE` outside `ask_join`'s `collect` reds a host +test or a fast-tier test. Owner: orchestrator. diff --git a/issues/build/a-thread-join-keeping-its-answer-is-gated-only-by-a-nightly-guest.md b/issues/build/a-thread-join-keeping-its-answer-is-gated-only-by-a-nightly-guest.md deleted file mode 100644 index 375fb06cd17..00000000000 --- a/issues/build/a-thread-join-keeping-its-answer-is-gated-only-by-a-nightly-guest.md +++ /dev/null @@ -1,19 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-09-28 ---- - -# A thread join keeping its answer is gated only by a nightly guest - -`toyos_proclife::join::Join` keeps the answer of the ask that collected the -zombie, and `a_join_asked_after_it_collected_keeps_its_answer` holds that. -`kernel/src/syscall/proc.rs`'s `sys_thread_join` asks through a copy of it held -in a `Cell` and writes the copy back after each ask. PR #564's round-2 review -dropped that write-back (`join.set(asked)`), so the ask after the wait collects -again and finds no such thread: every host test and the fast tier stayed -green, and only the nightly `fpu_isolation` went red, as it does for the same -mutation of the base. - -**Exit**: dropping the write-back reds a host test or a fast-tier test. Owner: -orchestrator. diff --git a/kernel/src/process.rs b/kernel/src/process.rs index c35fcad6070..8f1ab13a0bb 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -1272,7 +1272,7 @@ pub fn wake_pipe_writers(pipe_id: pipe::PipeId) { } /// Ask `join` once more; an unsettled one collects under the table lock, which is what makes validating the parent-thread relationship and collecting the zombie one act. -pub fn ask_join(join: &mut join::Join, tid: Tid, parent_pid: Pid) -> Option> { +pub fn ask_join(join: &join::Join, tid: Tid, parent_pid: Pid) -> Option> { join.ask(|| join::collect_zombie(PROCESS_TABLE.lock().as_mut().unwrap(), parent_pid, tid)) } diff --git a/kernel/src/syscall/proc.rs b/kernel/src/syscall/proc.rs index 6b76f554652..1ac63701348 100644 --- a/kernel/src/syscall/proc.rs +++ b/kernel/src/syscall/proc.rs @@ -151,13 +151,8 @@ pub(super) fn sys_thread_join(tid: u64) -> u64 { // None means never existed or already collected; the predicate below answers both. let target = process::thread_sched(caller, tid); let parkable = crate::scheduler::Parkable::at_entry(); - let join = core::cell::Cell::new(toyos_proclife::join::Join::default()); - let ask = || { - let mut asked = join.get(); - let answer = process::ask_join(&mut asked, tid, caller); - join.set(asked); - answer - }; + let join = toyos_proclife::join::Join::default(); + let ask = || process::ask_join(&join, tid, caller); loop { // Both refusals are one answer here; `JoinRefused` keeps them apart because they aren't the same fact. if let Some(answer) = ask() { diff --git a/tests/checks.rs b/tests/checks.rs index 4a231af7629..bf9369983ba 100644 --- a/tests/checks.rs +++ b/tests/checks.rs @@ -568,10 +568,6 @@ mod checks { } /// What a whole run exits with, and what its last line says. - /// - /// Driven through [`Tally`] rather than asserted about it: the property that - /// matters is what `--land`'s gate reads off the process, and that is the exit - /// code after `record` has seen every outcome. #[test] fn run_exit_status() -> Result<(), String> { let outcome = |name: &str, reason: Option<&str>, suspended: Duration| Outcome { diff --git a/tests/toyos.rs b/tests/toyos.rs index 1f9f7088171..a6953f9a913 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -463,7 +463,7 @@ const RUST_SKIP: &[&str] = &[ /// `sched_stress` is the one whose two runs differ by *kernel* rather than by /// what the host staged: the shipping build here, `sched_check_build`'s /// assert-carrying build there. -#[allow(dead_code, reason = "`suite_split` reads it, in `toyos-checks` alone")] +#[cfg(test)] const DRIVEN_AND_SHARED: &[&str] = &[ // The lost-wake canary: its shared run is the count on the shipping // kernel with nothing staged, and `blocking_read_window` drives it again diff --git a/toyos-proclife/src/join.rs b/toyos-proclife/src/join.rs index dfc66b8f327..3bd16df2d79 100644 --- a/toyos-proclife/src/join.rs +++ b/toyos-proclife/src/join.rs @@ -13,6 +13,8 @@ //! is gone stays gone, and a tid this process never had never appears. //! `sys_thread_join` reads it as the terminal answer for exactly that reason. +use core::cell::Cell; + use crate::table::{Lifecycle, Processes}; use crate::{Pid, Tid}; @@ -52,9 +54,11 @@ pub fn collect_zombie( /// /// Collecting takes the zombie out of the table, so an ask after the one that /// collected finds [`JoinRefused::NoSuchThread`]: a join whose wait asks again -/// after every wake answers with its first settled ask, never its last. -#[derive(Clone, Copy, Default, PartialEq, Eq, Debug)] -pub struct Join(Option>); +/// after every wake answers with its first settled ask, never its last. The +/// answer lives here, behind `&self`, so every ask of one join reads the same +/// answer and no caller holds a copy to write back. +#[derive(Default, Debug)] +pub struct Join(Cell>>); impl Join { /// Ask `collect` — [`collect_zombie`] under the table lock — unless an @@ -62,19 +66,21 @@ impl Join { /// `collect`, so it never takes the table lock again. #[must_use = "a settled join is the answer the syscall returns"] pub fn ask( - &mut self, + &self, collect: impl FnOnce() -> Result, JoinRefused>, ) -> Option> { - if self.0.is_none() { - self.0 = collect().transpose(); + if self.0.get().is_none() { + self.0.set(collect().transpose()); } - self.0 + self.0.get() } } #[cfg(test)] mod tests { use super::*; + use core::cell::RefCell; + use crate::model::World; use crate::ThreadLocation; @@ -97,25 +103,30 @@ mod tests { assert_eq!(collect_zombie(&mut world, pid, t1), Err(JoinRefused::NoSuchThread)); } - /// The wait's predicate collects the zombie, and the syscall asks again once - /// the wait returns: the second ask is the answer the caller gets. + /// `sys_thread_join`'s path: one join, one shared `ask`, asked before the + /// wait, by the wait's predicate on the wake that finds the zombie, and again + /// once the wait returns. The last ask is the answer the caller gets. #[test] fn a_join_asked_after_it_collected_keeps_its_answer() { - let mut world = World::new(); - let pid = world.spawn_process(); - let t1 = world.spawn_thread(pid); - let mut join = Join::default(); - assert_eq!(join.ask(|| collect_zombie(&mut world, pid, t1)), None); - world.set_location(pid, t1, ThreadLocation::Zombie(11)); - assert_eq!(join.ask(|| collect_zombie(&mut world, pid, t1)), Some(Ok(11))); + let world = RefCell::new(World::new()); + let pid = world.borrow_mut().spawn_process(); + let t1 = world.borrow_mut().spawn_thread(pid); + let join = Join::default(); + let ask = || join.ask(|| collect_zombie(&mut *world.borrow_mut(), pid, t1)); + let settled = || ask().is_some(); + assert_eq!(ask(), None); + assert!(!settled()); + world.borrow_mut().set_location(pid, t1, ThreadLocation::Zombie(11)); + assert!(settled(), "the wake that found the zombie did not settle the join"); assert_eq!( - join.ask(|| collect_zombie(&mut world, pid, t1)), + ask(), Some(Ok(11)), "a join that collected its thread answered the ask after it with no such thread", ); - let mut refused = Join::default(); - assert_eq!(refused.ask(|| collect_zombie(&mut world, pid, Tid(9))), Some(Err(JoinRefused::NoSuchThread))); - assert_eq!(refused.ask(|| collect_zombie(&mut world, pid, Tid(9))), Some(Err(JoinRefused::NoSuchThread))); + let refused = Join::default(); + let ask = || refused.ask(|| collect_zombie(&mut *world.borrow_mut(), pid, Tid(9))); + assert_eq!(ask(), Some(Err(JoinRefused::NoSuchThread))); + assert_eq!(ask(), Some(Err(JoinRefused::NoSuchThread))); } /// The syscall's wait asks after every wake, and `collect` is where the @@ -126,10 +137,10 @@ mod tests { let pid = world.spawn_process(); let t1 = world.spawn_thread(pid); world.set_location(pid, t1, ThreadLocation::Zombie(11)); - let mut join = Join::default(); + let join = Join::default(); assert_eq!(join.ask(|| collect_zombie(&mut world, pid, t1)), Some(Ok(11))); assert_eq!(join.ask(|| panic!("a settled join took the table lock to ask again")), Some(Ok(11))); - let mut refused = Join::default(); + let refused = Join::default(); assert_eq!(refused.ask(|| collect_zombie(&mut world, pid, Tid(9))), Some(Err(JoinRefused::NoSuchThread))); assert_eq!( refused.ask(|| panic!("a refused join took the table lock to ask again")), From 9b2645e1b43b8fb33a02d9281e4d6dc83e482e7d Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 15:30:16 +0200 Subject: [PATCH 17/18] DRIVEN_AND_SHARED is allowed dead again: clippy compiles toyos-build under cfg(test) `cargo run -- --clippy` runs `cargo clippy --workspace --all-targets`, which compiles the `harness = false` target `toyos-build` with `cfg(test)` set too. Under `#[cfg(test)]` the list is compiled there with no reader, and the gate fails with "constant `DRIVEN_AND_SHARED` is never used" (`-D dead-code`), so `cfg(test)` does not tell the two targets apart and the allow comes back. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- tests/toyos.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/toyos.rs b/tests/toyos.rs index a6953f9a913..1f9f7088171 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -463,7 +463,7 @@ const RUST_SKIP: &[&str] = &[ /// `sched_stress` is the one whose two runs differ by *kernel* rather than by /// what the host staged: the shipping build here, `sched_check_build`'s /// assert-carrying build there. -#[cfg(test)] +#[allow(dead_code, reason = "`suite_split` reads it, in `toyos-checks` alone")] const DRIVEN_AND_SHARED: &[&str] = &[ // The lost-wake canary: its shared run is the count on the shipping // kernel with nothing staged, and `blocking_read_window` drives it again From e30e7d7f1b01f29e521a9e0cd923b78534d05f41 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 17:52:25 +0200 Subject: [PATCH 18/18] Round 5 review fixes: the join-per-ask gap recorded on the syscall's issue, two false test-replica claims removed The host test only replicates sys_thread_join's shape, so the syscall's answer-keeping stays gated by nothing but the nightly guest fpu_isolation; record that on the existing table-lock issue rather than filing a new one. Delete the comment claiming to describe "sys_thread_join's path" (it describes the test's replica of it) and the claim that no caller holds a copy to write back (describes the deleted implementation). Drop the issue's two unsupported claims ("every test passed", "costs contention ... not a wrong answer") that no run backs. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...aking-the-table-lock-again-is-gated-by-nothing.md | 12 +++++++++--- toyos-proclife/src/join.rs | 5 +---- 2 files changed, 10 insertions(+), 7 deletions(-) diff --git a/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md b/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md index ed9257bf515..c0f11b06a82 100644 --- a/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md +++ b/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md @@ -11,9 +11,15 @@ unsettled, and `a_settled_join_does_not_take_the_table_again` holds that. Whether `collect` is where the kernel takes `PROCESS_TABLE` is `kernel/src/process.rs`'s `ask_join`, in no crate a host test compiles. PR #564's round-4 review wrote it as -`let mut g = PROCESS_TABLE.lock(); join.ask(|| join::collect_zombie(g.as_mut().unwrap(), parent_pid, tid))`: -every test passed. A settled join then takes the lock on every wake of its -wait, which costs contention on `PROCESS_TABLE` and not a wrong answer. +`let mut g = PROCESS_TABLE.lock(); join.ask(|| join::collect_zombie(g.as_mut().unwrap(), parent_pid, tid))`. +A settled join then takes the lock on every wake of its wait. **Exit**: taking `PROCESS_TABLE` outside `ask_join`'s `collect` reds a host test or a fast-tier test. Owner: orchestrator. + +The syscall's answer-keeping is the same gap: `sys_thread_join` +(`kernel/src/syscall/proc.rs:154`) holds one `Join` across every ask, but +`a_join_asked_after_it_collected_keeps_its_answer` only replicates that shape +rather than calling the syscall, so a `Join::ask` built fresh per ask +(`join-per-ask`) stays host-green and reds only on the nightly guest +`fpu_isolation` (`thread_join failed`, `left: 18446744073709551614`). diff --git a/toyos-proclife/src/join.rs b/toyos-proclife/src/join.rs index 3bd16df2d79..a76af7d1e47 100644 --- a/toyos-proclife/src/join.rs +++ b/toyos-proclife/src/join.rs @@ -56,7 +56,7 @@ pub fn collect_zombie( /// collected finds [`JoinRefused::NoSuchThread`]: a join whose wait asks again /// after every wake answers with its first settled ask, never its last. The /// answer lives here, behind `&self`, so every ask of one join reads the same -/// answer and no caller holds a copy to write back. +/// answer. #[derive(Default, Debug)] pub struct Join(Cell>>); @@ -103,9 +103,6 @@ mod tests { assert_eq!(collect_zombie(&mut world, pid, t1), Err(JoinRefused::NoSuchThread)); } - /// `sys_thread_join`'s path: one join, one shared `ask`, asked before the - /// wait, by the wait's predicate on the wake that finds the zombie, and again - /// once the wait returns. The last ask is the answer the caller gets. #[test] fn a_join_asked_after_it_collected_keeps_its_answer() { let world = RefCell::new(World::new());