diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index 19aa5ada208..a0cddada720 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -1,13 +1,15 @@ name: nightly # Everything that boots a guest, the host gate again to write the cache the -# merge queue restores, and portability: on main every night, and on any branch by -# `gh workflow run nightly.yml --ref `. Every job's logic is -# `cargo run -- --ci ` (src/ci.rs); this file says where each one runs. +# merge queue restores, and portability: on main every night, the weekly tier +# on Sundays, and on any branch by `gh workflow run nightly.yml --ref `. +# Every job's logic is `cargo run -- --ci ` (src/ci.rs); this file says +# where each one runs. on: schedule: - - cron: '0 3 * * *' + - cron: '0 3 * * 1-6' + - cron: '0 3 * * 0' workflow_dispatch: # Never cancelled: `build` may be an hour into a bootstrap. diff --git a/Cargo.toml b/Cargo.toml index ba38e041fae..aff7f48c5ef 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -274,3 +274,7 @@ strip = "debuginfo" name = "toyos-build" path = "tests/toyos.rs" harness = false + +[[test]] +name = "toyos-checks" +path = "tests/checks.rs" diff --git a/issues/README.md b/issues/README.md index 646c3154d8f..1ee48f15ede 100644 --- a/issues/README.md +++ b/issues/README.md @@ -155,5 +155,5 @@ owns "every policy about files — where they go, what they are called, how many there are, what happens when the stick stops answering" (`userland/logd/src/main.rs:1-10`). Gated by `esp_filesystem`, `kernel_log_file`, `log_backing_read_error`, -`boot_volume_metadata_error`, `log_partition_layout`, `log_partition_identity` -and `wall_clock_file`, plus `toybox_cp_volume`. +`boot_volume_metadata_error`, `log_partition_layout` and +`log_partition_identity`, plus `toybox_cp_volume`. diff --git a/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md b/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md new file mode 100644 index 00000000000..c0f11b06a82 --- /dev/null +++ b/issues/build/a-settled-thread-join-taking-the-table-lock-again-is-gated-by-nothing.md @@ -0,0 +1,25 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# A settled thread join taking the table lock again is gated by nothing + +`toyos_proclife::join::Join::ask` calls its `collect` only while the join is +unsettled, and `a_settled_join_does_not_take_the_table_again` holds that. +Whether `collect` is where the kernel takes `PROCESS_TABLE` is +`kernel/src/process.rs`'s `ask_join`, in no crate a host test compiles. PR +#564's round-4 review wrote it as +`let mut g = PROCESS_TABLE.lock(); join.ask(|| join::collect_zombie(g.as_mut().unwrap(), parent_pid, tid))`. +A settled join then takes the lock on every wake of its wait. + +**Exit**: taking `PROCESS_TABLE` outside `ask_join`'s `collect` reds a host +test or a fast-tier test. Owner: orchestrator. + +The syscall's answer-keeping is the same gap: `sys_thread_join` +(`kernel/src/syscall/proc.rs:154`) holds one `Join` across every ask, but +`a_join_asked_after_it_collected_keeps_its_answer` only replicates that shape +rather than calling the syscall, so a `Join::ask` built fresh per ask +(`join-per-ask`) stays host-green and reds only on the nightly guest +`fpu_isolation` (`thread_join failed`, `left: 18446744073709551614`). diff --git a/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md b/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md index 432fda64247..08152da71bb 100644 --- a/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md +++ b/issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md @@ -1,5 +1,5 @@ --- -status: expected-red +status: open kind: defect opened: 2026-08-20 --- @@ -20,6 +20,8 @@ Split out of `issues/build/parallel-tests-red-under-other-suites.md`, whose rate table was never this test's — CI's single-guest-per-machine shards rule out the contention shape that file is about. -**Exit condition.** The lost lines' cause is fixed, and -`console_line_atomicity` green on CI's `guest` shards, one guest per machine. +**Exit condition.** The lost lines' cause is fixed, shown against +`console_line_atomicity` as it stands at `1808fb8d` (its binary, the test +runner's `CONSOLE_JOBS` stdin and its harness arm), restored and green on CI's +`guest` shards, one guest per machine. Owner: orchestrator. diff --git a/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md b/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md deleted file mode 100644 index 03940d2f0fe..00000000000 --- a/issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md +++ /dev/null @@ -1,34 +0,0 @@ ---- -status: expected-red -kind: finding -opened: 2026-09-26 ---- - -# `quiesce_wakes_on_the_last_exit` lost its serial READY beside other guests - -Fast tier at `a4b44a91` (PR #510's branch; `toyos-sshupload`'s suite held build -slots at the same time): `QEMU died before ===READY=== (status: exit 0)`. The -guest did everything the test reads: `quiesce-last-exit: quiesce-last is held`, -`Syncing filesystems...`, `stop: 4 of 4 userland thread(s) stopped ... over 3 -sweep(s)`, `usb-quiesce: disk 0 SYNCHRONIZE CACHE ok`, `Rebooting.` But the uart -captured `nothing at all`. Before it, `usb-storage: 00:02.0 slot 1 transport -broke on SCSI 0x28: no answer in the data phase in 2000 ms`, and the -test-runner's spawn reported `layout=2069ms`. The harness's re-run alone was -green in 3 s, 2 sweeps. - -Not shown: why the uart saw none of the test-runner's output in a boot whose -console carried the whole stop. The READY wait reports the missing marker, not -its cause. - -Again in the fast tier at `8846c021` (PR #524's branch, alone on the host): -the same `QEMU died before ===READY=== (status: exit 0)` with `uart: nothing -at all`, after `stop: 5 of 5 userland thread(s) stopped`, `usb-quiesce: disk 0 -SYNCHRONIZE CACHE ok` and `Rebooting.`; this time no usb-storage transport -break before it. The re-run alone was green. - -On main's lineage, with `the stopped-boot drain carried no kernel output at all -(38 bytes)`: f231c43e (run 36280285913, `guest (3)`), 67a430c8 (run -36285169430) and a55d62c6 (run 36287592139). - -**Exit**: the marker waited for where the boot's reboot cannot race it, and the -test green in a fast tier beside other guests. Owner: orchestrator. diff --git a/issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md b/issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md index 54d8ef7f738..56a1ac4a4a6 100644 --- a/issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md +++ b/issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md @@ -14,10 +14,6 @@ until the stop waits on it alone`, `stop: 4 of 7 userland thread(s) stopped ok` and `Rebooting.`, then `shutdown: /log did not answer in 2000ms`; the uart captured `nothing at all`. The harness's re-run alone was green in 2 s. -The same shape as -`issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md`, -on the sibling arm. - **Exit**: a cause for the empty uart on a boot that rebooted as designed, or the marker waited for where the boot's reboot cannot race it. diff --git a/issues/build/the-harness-carries-three-helpers-nothing-calls.md b/issues/build/the-harness-carries-three-helpers-nothing-calls.md new file mode 100644 index 00000000000..7cf97d13bd4 --- /dev/null +++ b/issues/build/the-harness-carries-three-helpers-nothing-calls.md @@ -0,0 +1,25 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# The harness carries three helpers nothing calls + +`tests/common/mod.rs` puts `#[allow(dead_code)]` on nearly every module, so +the compiler says nothing about a harness helper whose last caller went. With +those attributes taken off, `cargo check --tests` on `wt/toyos-schedule` names +three that no test reaches, and none of the three was called by anything that +branch deleted: + +- `tests/common/storage.rs`: `FileBlocks::whole`; +- `tests/common/qemu.rs`: the field `usb_images` and its method `usb_images`; +- `tests/common/qemu.rs`: `QmpDevices::set_link`. + +The same pass names `tests/common/audio.rs`'s `completions` and `clients` and +`tests/common/stats.rs`'s `fisher_reject_at`, which the audio branch +(`wt/toyos-notiming`) deletes with their modules. + +**Exit**: the three deleted, and the module-wide `allow(dead_code)` replaced by +nothing, so the next orphan is a warning the host gate denies. Owner: +orchestrator. diff --git a/issues/build/which-pass-drain-irqs-is-entered-from-is-gated-only-by-a-nightly-guest.md b/issues/build/which-pass-drain-irqs-is-entered-from-is-gated-only-by-a-nightly-guest.md new file mode 100644 index 00000000000..748adf28e0c --- /dev/null +++ b/issues/build/which-pass-drain-irqs-is-entered-from-is-gated-only-by-a-nightly-guest.md @@ -0,0 +1,18 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# Which pass `drain_irqs` is told it is entered from is gated only by a nightly guest + +`kernel-loom`'s `only_a_pass_entered_at_depth_zero_takes_the_request` holds +`Entered::may_serve`, the decision. The two call sites that choose its argument, +`kernel/src/sched/driver.rs`'s `pass` (`Entered::Pass` at the depth it was +entered at) and `pass_block` (`Entered::Blocking`), are in no crate a host test +compiles. PR #564's round-2 review handed `pass_block`'s drain +`Entered::Pass { depth: 0 }`: every host test and the fast tier stayed green, +and only the nightly `blocked_dump` went red. + +**Exit**: a mutation of either call site's `Entered` reds a host test or a +fast-tier test. Owner: orchestrator. diff --git a/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md b/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md index 25d447d835b..8732d0b086d 100644 --- a/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md +++ b/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md @@ -27,5 +27,7 @@ issues/boot-media/partition-claim-gives-up-reds-beside-other-guests-and-is-green names as racing whichever fsync the boot reaches first. Not shown: whether that race is this test's cause. -**Exit**: the cause shown on a red run's log, and the test green on a -nightly. +**Exit**: the cause shown on the log of a red run of `home_budget_refusal_retried` +as it stands at `1808fb8d` (its binary `test_rs_home_fsync_budget` and its +`storage` body), restored and run on CI's nightly `guest` shards; and that +test green on a nightly after the fix. Owner: orchestrator. diff --git a/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md b/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md index 8ad470d8c29..3ebab4d84c2 100644 --- a/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md +++ b/issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md @@ -1,5 +1,5 @@ --- -status: expected-red +status: open kind: defect opened: 2026-08-08 --- @@ -16,5 +16,4 @@ probe found, only this one still reds. Split out of `issues/hardware/eleven-names-red-on-ci.md`, which covers eleven names and has no exit condition for this one in particular. -**Exit condition.** The cause of the empty first controller is fixed, and -`usb_disk_index_stable` green on CI's `guest` shards. Owner: orchestrator. +**Exit condition.** The cause of the empty first controller is fixed. Owner: orchestrator. diff --git a/issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md b/issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md index af7028ec9a1..a9d1a0de752 100644 --- a/issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md +++ b/issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md @@ -6,8 +6,7 @@ opened: 2026-09-25 # A `quiesce_writers` writer's first write-and-fsync pass outlasts the job's 5 s spin-up -`quiesce_dump_holds_the_stopped` and `quiesce_stops_the_machine` both boot -`quiesce_writers`. It asks for the reset only once each of its six writers has +It asks for the reset only once each of its six writers has finished one pass: a create, 64 KiB of writes and an fsync. If a writer is still in its first pass after 5 s, the job prints `quiesce_writers: of 6 writers reached their loop in 5s` and exits 1 without asking, so no stop diff --git a/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md b/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md index f21bdbf3331..16c73122407 100644 --- a/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md +++ b/issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md @@ -1,5 +1,5 @@ --- -status: expected-red +status: open kind: defect opened: 2026-09-03 --- @@ -15,5 +15,6 @@ twelve 2 MiB images entered a cache whose test budget refuses at the second. Owed: a mechanism. Nobody has one. -**Exit condition.** The cause of the missing refusal is fixed, and -`so_cache_refusals` green on CI's KVM `guest` shards. Owner: orchestrator. +**Exit condition.** The cause of the missing refusal is fixed, shown against +`so_cache_refusals` and the `so-cache-tiny` budget it arms, as both stand at +`1808fb8d`, restored and green on CI's KVM `guest` shards. Owner: orchestrator. diff --git a/kernel-loom/tests/dump_request.rs b/kernel-loom/tests/dump_request.rs index dd7584ea5a1..dde1309fd04 100644 --- a/kernel-loom/tests/dump_request.rs +++ b/kernel-loom/tests/dump_request.rs @@ -13,7 +13,7 @@ //! ``` #![cfg(feature = "loom")] -use kernel_loom::dump_request::{DumpRequest, Left}; +use kernel_loom::dump_request::{DumpRequest, Entered, Left}; use loom::cell::UnsafeCell; use loom::sync::Arc; @@ -39,16 +39,10 @@ impl Machine { if !self.request.take() { return false; } - self.report_until_nothing_pending(); - true - } - - /// `sched::dump::report_until_nothing_pending`, once the caller took the request. - fn report_until_nothing_pending(&self) { loop { self.report(); if !self.request.end_report() { - return; + return true; } } } @@ -111,27 +105,6 @@ fn one_request_is_taken_once() { }); } -/// The stop asks while a sibling's pass may serve: filed and taken in one exchange, the request is never -/// pending for the sibling to take, and the one report is the asker's. -#[test] -fn the_asker_owns_the_report_it_asks_for() { - loom::model(|| { - let machine = Machine::new(); - - let sibling = { - let machine = machine.clone(); - loom::thread::spawn(move || machine.serve()) - }; - assert!(machine.request.file_and_take(), "no report ran, and the asker did not take its own request"); - machine.report_until_nothing_pending(); - let sibling_took = sibling.join().unwrap(); - - assert!(!sibling_took, "a sibling's pass took the request the asker filed"); - assert!(!machine.request.pending(), "the request outlived its report"); - assert_eq!(machine.reports(), 1, "one request was reported other than once"); - }); -} - /// Two passes that may not serve and one that may: the request is announced at /// most once, and by nobody once it has been taken unannounced. #[test] @@ -186,3 +159,15 @@ fn a_request_left_during_a_report_is_still_taken_by_its_end() { assert_eq!(machine.reports(), 2); }); } + +/// Only a pass entered at depth zero takes the request. A syscall's pass — `yield_now`, `exit_current` — +/// is entered above zero and a blocking pass is inside a wait ticket, and `dump::request` asserts it runs +/// with neither under it: a report served from one of them is a kernel panic. Not an interleaving question, +/// so no model. +#[test] +fn only_a_pass_entered_at_depth_zero_takes_the_request() { + assert!(Entered::Pass { depth: 0 }.may_serve()); + for entered in [Entered::Blocking, Entered::Pass { depth: 1 }, Entered::Pass { depth: 2 }] { + assert!(!entered.may_serve(), "{entered} took the request, and its report runs above a bare pass"); + } +} diff --git a/kernel/CLAUDE.md b/kernel/CLAUDE.md index 6fffea753de..eb0254cff8a 100644 --- a/kernel/CLAUDE.md +++ b/kernel/CLAUDE.md @@ -16,7 +16,7 @@ The module header at the site owns its subsystem — read it before changing a m - **A block-layer `BudgetExpired` is not-durable-yet and never a loss** — it is retried on a fresh budget above every lock; a flush that discards its pages on one splits a FAT mirror. - **A decision the process table makes lives in `toyos-proclife`, never in `process.rs`** — its defects are interleavings and that crate is the only machine that can enumerate one. - **A task holds at most one watch registration** — a standing registration across a loop must not call anything that registers again. A double registration panics only at attempt ≥ 2, so the contention depth is the coverage. -- **A console is per holder, minted at spawn** — the object *is* the line buffer; `console_line_atomicity` is the gate. +- **A console is per holder, minted at spawn** — the object *is* the line buffer. - **`ops::close` cancels a poll only for a source its object really ends** — `Watch::cancel_polls` answers every ring's poll on that watch; `ops::close_ends_polls` is where a new object kind answers. - **A page shared with userland is never reached through a Rust reference** — the words a protocol shares are `&AtomicU32` one at a time, everything else is a volatile copy of the whole value, and a page is laid out *before* it is mapped (`SharedMemObject::phys_before_mapping`). The kernel, `toyos-abi` and the SDK each hold one end of this rule. - **A device a process drives reaches only the memory its grants map** — the domain is the gate, not the descriptor; a device address outside a grant is a `DMA FAULT` record and never a crash, and `userdev_dma_fault` is the registration. diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 23b8aa6e8ee..5bf2de80cc3 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -119,12 +119,6 @@ actuators! { /// Hold the thread named `toyos_quiesce::LAST_THREAD` inside `SYS_NANOSLEEP`, and the shutdown until it is held there, until the stop waits on it alone: its park is then the stop's last transition. quiesce_last_park = "quiesce-last-park"; - /// The same inside `SYS_THREAD_EXIT`: its exit is then the stop's last transition. - quiesce_last_exit = "quiesce-last-exit"; - - /// Serve a blocked-task dump from the shutdown once its first stage has stopped the machine: the report Ctrl+Alt+D gives on a shutdown stuck in its stop. - quiesce_dump = "quiesce-dump"; - /// Refuse the second directory-entry write of the file `writeback_durability` stages for the retry gate — the first is that file's own seed being made durable — as a budget expiry, so a flush fails at its metadata write with its pages already written and settled. fat_flush_meta_refuse = "fat-flush-meta-refuse"; @@ -248,9 +242,6 @@ actuators! { /// `usb_reset_records_the_phase_it_cut`. usb_reset_under_load = "usb-reset-under-load"; - /// Put the shared-object cache's byte budget within reach of the libraries a guest can build, so the shipped refusal runs at all. - so_cache_tiny = "so-cache-tiny"; - /// Run the first attempt of each run `object::ops::until_answered` retries — /// a file's `SYS_FSYNC`, a claimed partition's read, write or flush — under an /// operation that is already over, once per file and per partition and kind. @@ -326,12 +317,6 @@ actuators! { /// Give PORTSC's PED bit the RW1CS meaning xHCI 1.2 §5.4.8 gives it. xhci_portsc_rw1c = "xhci-portsc-rw1c"; - /// Take a bound HID device's first completion away and hand back a stall. - xhci_hid_break_first = "xhci-hid-break-first"; - - /// The same at its fourth completion. - xhci_hid_break_late = "xhci-hid-break-late"; - /// Run `parse_config` over nine crafted configuration descriptors at init. xhci_descriptor_selftest = "xhci-descriptor-selftest"; @@ -462,9 +447,6 @@ actuators! { /// Let a handle close cancel every poll on the keyboard's watch in the machine. keyboard_close_cancels_every_console = "keyboard-close-cancels-every-console"; - /// Panic inside `klogd` on its first instruction. - klogd_panic = "klogd-panic"; - /// Read address zero inside `klogd` on its first instruction. klogd_fault = "klogd-fault"; diff --git a/kernel/src/drivers/xhci/device.rs b/kernel/src/drivers/xhci/device.rs index 30abbf5cfd7..5e3f52798b5 100644 --- a/kernel/src/drivers/xhci/device.rs +++ b/kernel/src/drivers/xhci/device.rs @@ -784,7 +784,6 @@ fn bind_hid( prev_report: [0; 8], broke_with: None, failures: 0, - completions: 0, }; dev.requeue(&ctrl.db_base); diff --git a/kernel/src/drivers/xhci/hid.rs b/kernel/src/drivers/xhci/hid.rs index 49d205fb544..a92aa589f51 100644 --- a/kernel/src/drivers/xhci/hid.rs +++ b/kernel/src/drivers/xhci/hid.rs @@ -45,10 +45,6 @@ pub struct HidDevice { pub broke_with: Option, /// Consecutive failures; a delivered report clears it — see [`super::MAX_HID_FAILURES`]. pub failures: u8, - /// Completions this endpoint has produced. - /// Counted unconditionally so the `xhci-hid-break-*` actuators aren't a second code path. - #[cfg_attr(not(feature = "boot-actuators"), allow(dead_code))] - pub completions: u32, } impl HidDevice { @@ -106,30 +102,3 @@ impl HidDevice { db_base.write_u32(self.slot_id as u64 * 4, self.int_ep_dci as u32); } } - -/// Takes one completion away from the device that earned it and hands the driver a stall in its place. -// QEMU's usb-hid has no path to USB_RET_STALL for an interrupt IN token, so nothing on the host side can stage this. -// Replaces the completion code and the delivered report, not the TRB/ring/transfer-event/output-context chain, so a dispatched "success" can only be real. -#[cfg(feature = "boot-actuators")] -impl HidDevice { - // The first completion is a never-delivered endpoint; the fourth is one that was working and stopped — different driver states, not degrees of one. - fn break_at() -> Option { - if crate::actuator::xhci_hid_break_first() { - Some(1) - } else if crate::actuator::xhci_hid_break_late() { - Some(4) - } else { - None - } - } - - pub fn stage_break(&mut self, code: u32) -> u32 { - self.completions += 1; - if Self::break_at() != Some(self.completions) { - return code; - } - // Zeroing leaves the slot as a stalled endpoint would have; runs before requeue, so nothing else touches the buffer. - self.report.subview(0, self.report_size as usize).zero(); - super::CC_STALL - } -} diff --git a/kernel/src/drivers/xhci/mod.rs b/kernel/src/drivers/xhci/mod.rs index 5b9dbb4185d..ef82aa6be07 100644 --- a/kernel/src/drivers/xhci/mod.rs +++ b/kernel/src/drivers/xhci/mod.rs @@ -973,8 +973,6 @@ impl XhciController { } return; }; - #[cfg(feature = "boot-actuators")] - let code = self.devices[at].stage_break(code); let dev = &mut self.devices[at]; if code == CC_SUCCESS || code == CC_SHORT_PACKET { dev.failures = 0; diff --git a/kernel/src/elf/cache.rs b/kernel/src/elf/cache.rs index b7322d5b73b..7eb03a5d926 100644 --- a/kernel/src/elf/cache.rs +++ b/kernel/src/elf/cache.rs @@ -154,17 +154,6 @@ static SO_CACHE: Lock> = Lock::new(Vec::new()); /// this admits one of those and refuses a second. const BUDGET_BYTES: usize = 256 * 1024 * 1024; -/// `so-cache-tiny`'s number, in reach of a guest. Only the magnitude moves. -const TINY_BUDGET_BYTES: usize = 8 * 1024 * 1024; - -fn budget_bytes() -> usize { - if crate::actuator::so_cache_tiny() { - TINY_BUDGET_BYTES - } else { - BUDGET_BYTES - } -} - /// Every cached image's allocation, summed. The caller holds the lock. fn held_bytes(cache: &[(String, CachedLib)]) -> usize { cache.iter().map(|(_, c)| c.alloc.size()).sum() @@ -239,15 +228,14 @@ pub fn cache_loaded_lib( return outcome.map(|cloned| cloned.unwrap_or_else(|| owned(alloc))); } // Under the lock that publishes: two concurrent loads must not both find room. - let budget = budget_bytes(); let (held, entries) = (held_bytes(&cache), cache.len()); - let Some(after) = held.checked_add(alloc.size()).filter(|b| *b <= budget) else { + let Some(after) = held.checked_add(alloc.size()).filter(|b| *b <= BUDGET_BYTES) else { drop(cache); // Both allocations drop here, so the refusal gives back what the load took. log!( "dlopen: {} would take the shared-object cache to {} bytes over {} entries, past its \ {}-byte budget; refused, and nothing is evicted for it", - path, held.saturating_add(alloc.size()), entries + 1, budget + path, held.saturating_add(alloc.size()), entries + 1, BUDGET_BYTES ); return Err(SyscallError::ResourceExhausted); }; @@ -259,7 +247,7 @@ pub fn cache_loaded_lib( log!( "dlopen: cached {} with {} bind + {} tpoff64 + {} tpoff32 + {} dtpmod64 + {} dtpoff64 pre-scanned relocs, cache now {} of {} bytes", path, relocs.bind.len(), relocs.tpoff64.len(), relocs.tpoff32.len(), - relocs.dtpmod64.len(), relocs.dtpoff64.len(), after, budget + relocs.dtpmod64.len(), relocs.dtpoff64.len(), after, BUDGET_BYTES ); Ok(snapshot.into_lib( diff --git a/kernel/src/log/console.rs b/kernel/src/log/console.rs index 8b49f3bff0d..cfb47556423 100644 --- a/kernel/src/log/console.rs +++ b/kernel/src/log/console.rs @@ -417,11 +417,6 @@ fn left_to_the_stop() -> bool { } extern "C" fn body(_arg: u64) -> ! { - // First, before any drain: stages a panic inside a kernel thread to test the panic handler's branch. - #[cfg(feature = "boot-actuators")] - if crate::actuator::klogd_panic() { - panic!("klogd-panic: the console drainer died"); - } #[cfg(feature = "boot-actuators")] if crate::actuator::klogd_fault() { // SAFETY: unsound by design — a staged Ring 0 null read, only on this actuator's boot. diff --git a/kernel/src/process.rs b/kernel/src/process.rs index 3c6fcc9b66b..8f1ab13a0bb 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -1271,12 +1271,9 @@ pub fn wake_pipe_writers(pipe_id: pipe::PipeId) { scheduler::wake_pipe_writers(pipe_id); } -/// Atomically validate the parent-thread relationship and collect a zombie thread; the table lock is the atomicity. `Err(())`: this caller may not join `tid`/`pid`. -pub fn wait_thread_zombie(tid: Tid, parent_pid: Pid) -> Result, ()> { - let mut guard = PROCESS_TABLE.lock(); - let table = guard.as_mut().unwrap(); - // Both refusals are one answer here (`SyscallError::NotFound`); `JoinRefused` keeps them apart because they aren't the same fact. - join::collect_zombie(table, parent_pid, tid).map_err(|_| ()) +/// Ask `join` once more; an unsettled one collects under the table lock, which is what makes validating the parent-thread relationship and collecting the zombie one act. +pub fn ask_join(join: &join::Join, tid: Tid, parent_pid: Pid) -> Option> { + join.ask(|| join::collect_zombie(PROCESS_TABLE.lock().as_mut().unwrap(), parent_pid, tid)) } /// Handle a page fault at `fault_addr` by looking up the current process's VMAs. Returns whether the fault was resolved. diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 42a598524b7..a851a8fff3d 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -245,11 +245,11 @@ fn sweep(caller: ThreadId) -> Sweep { out } -/// `quiesce-last-park` and `quiesce-last-exit`: one thread, named -/// [`toyos_quiesce::LAST_THREAD`], held inside its syscall until the stop's -/// latest sweep counts it as the one thread still running, so the park or the -/// exit it makes next is the last transition the stop sees. Without them no -/// boot can tell whether that transition's post is what wakes the stop. +/// `quiesce-last-park`: one thread, named [`toyos_quiesce::LAST_THREAD`], +/// held inside its `SYS_NANOSLEEP` until the stop's latest sweep counts it as +/// the one thread still running, so the park it makes next is the last +/// transition the stop sees. Without it no boot can tell whether that park's +/// post is what wakes the stop. #[cfg(feature = "boot-actuators")] pub mod last { use core::sync::atomic::{ @@ -262,28 +262,7 @@ pub mod last { use crate::watch::{self, Watch}; use crate::time::{Budget, Deadline, Duration}; - /// The transition the held thread makes once it is released. - #[derive(Clone, Copy)] - pub enum Last { - Park, - Exit, - } - - impl Last { - fn armed(self) -> bool { - match self { - Last::Park => crate::actuator::quiesce_last_park(), - Last::Exit => crate::actuator::quiesce_last_exit(), - } - } - - fn name(self) -> &'static str { - match self { - Last::Park => "quiesce-last-park", - Last::Exit => "quiesce-last-exit", - } - } - } + const NAME: &str = "quiesce-last-park"; /// How long either side waits for the other before the boot dies by name. const STAGED: Budget = Budget::of( @@ -304,18 +283,13 @@ pub mod last { RUNNING.store(swept.running, Release); } - fn armed() -> Option { - [Last::Park, Last::Exit].into_iter().find(|last| last.armed()) - } - - /// Hold the running thread here if it is the one `last` stages. - pub fn hold(last: Last) { - if !last.armed() || !is_the_named_thread() || HELD.swap(true, AcqRel) { + /// Hold the running thread here if it is the one this staging names. + pub fn hold() { + if !crate::actuator::quiesce_last_park() || !is_the_named_thread() || HELD.swap(true, AcqRel) { return; } crate::log!( - "{}: {} is held until the stop waits on it alone", - last.name(), + "{NAME}: {} is held until the stop waits on it alone", toyos_quiesce::LAST_THREAD, ); ARRIVED.post(); @@ -325,8 +299,7 @@ pub mod last { while RUNNING.load(Acquire) != 1 { assert!( !deadline.reached(crate::clock::now()), - "{}: the stop never came down to this thread alone in {} ms", - last.name(), + "{NAME}: the stop never came down to this thread alone in {} ms", STAGED.nanos() / 1_000_000, ); crate::scheduler::yield_now(); @@ -336,10 +309,11 @@ pub mod last { /// Called by the shutdown before it stops anything: the stop is staged /// only once the thread it is staged around is inside its syscall. pub fn await_the_held_thread() { - let Some(last) = armed() else { return }; + if !crate::actuator::quiesce_last_park() { + return; + } crate::log!( - "{}: the stop waits for {} to reach its syscall", - last.name(), + "{NAME}: the stop waits for {} to reach its syscall", toyos_quiesce::LAST_THREAD, ); let deadline = Deadline::at(crate::clock::now() + STAGED.duration()); @@ -354,8 +328,7 @@ pub mod last { ); assert!( HELD.load(Acquire), - "{}: no thread named {} reached its syscall in {} ms", - last.name(), + "{NAME}: no thread named {} reached its syscall in {} ms", toyos_quiesce::LAST_THREAD, STAGED.nanos() / 1_000_000, ); diff --git a/kernel/src/sched/dump.rs b/kernel/src/sched/dump.rs index 81e23cffb06..aed0ffd54d0 100644 --- a/kernel/src/sched/dump.rs +++ b/kernel/src/sched/dump.rs @@ -15,6 +15,7 @@ use core::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +pub use super::dump_request::Entered; use super::dump_request::{DumpRequest, Left}; use crate::arch::{irqchip, percpu, smp}; @@ -142,27 +143,9 @@ fn online_cpus() -> usize { (smp::cpu_count() as usize).min(MAX_CPUS) } -/// How the pass that met a request was entered: the whole of what decides whether it may serve. -#[derive(Clone, Copy)] -pub enum Entered { - /// `driver::pass`, by the preempt depth it was entered at, before it raised its own level. - Pass { depth: u32 }, - /// `driver::pass_block`, inside a wait ticket's registration window at every depth. - Blocking, -} - impl Entered { fn under_nothing(self) -> Option { - matches!(self, Self::Pass { depth: 0 }).then_some(UnderNothing(())) - } -} - -impl core::fmt::Display for Entered { - fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::Blocking => write!(f, "a blocking pass"), - Self::Pass { depth } => write!(f, "a pass entered at preempt depth {depth}"), - } + self.may_serve().then_some(UnderNothing(())) } } @@ -174,21 +157,6 @@ pub fn file_request() { REQUEST.file(); } -/// `quiesce-dump`: the report a keystroke asks for, served by the thread -/// running the shutdown once its stop holds every thread it named. That -/// thread is at its syscall's entry depth with no lock under it, which is -/// what a pass entered at zero proves and what `Parkable::at_entry` asserts. -#[cfg(feature = "boot-actuators")] -pub fn serve_for_the_stop() { - let _nothing_under_it = crate::scheduler::Parkable::at_entry(); - // Filed and taken in one exchange: a sibling's pass entered at zero would take a request filed alone. - assert!( - REQUEST.file_and_take(), - "quiesce-dump: a report was already running when the stop asked for its own" - ); - report_until_nothing_pending(&UnderNothing(())); -} - /// Ctrl+Alt+D's request, from `drain_irqs` on every pass. pub fn serve_request(entered: Entered) { #[cfg(feature = "boot-actuators")] @@ -207,11 +175,7 @@ fn serve(proof: &UnderNothing) { if !REQUEST.pending() || !REQUEST.take() { return; } - report_until_nothing_pending(proof); -} - -/// The caller took the request: report until a report ends with nothing filed during it. -fn report_until_nothing_pending(proof: &UnderNothing) { + // Until a report ends with nothing filed during it. loop { report(proof); if !REQUEST.end_report() { diff --git a/kernel/src/sched/dump_request.rs b/kernel/src/sched/dump_request.rs index 02ec98d156c..c6a18b705ad 100644 --- a/kernel/src/sched/dump_request.rs +++ b/kernel/src/sched/dump_request.rs @@ -1,6 +1,7 @@ //! Ctrl+Alt+D's request and the report it asks for, as one word: what is pending, and whether a report runs. //! Every transition is one atomic update of that word, so a request is announced once and taken once, no //! report begins inside another, and a request filed during a report is taken by that report's end. +//! [`Entered`] is which pass may take a request. //! No `crate::` references: `kernel-loom` compiles this file directly under `feature = "loom"`. #[cfg(not(feature = "loom"))] @@ -74,13 +75,6 @@ impl DumpRequest { }); } - /// Ask and begin the report in one update, unless a report runs: no pass can take the request between - /// its filing and its taking, so the asker owns the report it asked for, and it serves anything pending. - #[cfg(any(feature = "boot-actuators", feature = "loom"))] - pub fn file_and_take(&self) -> bool { - self.update(HANDOFF.0, HANDOFF.1, |word| (word & REPORTING == 0).then_some(REPORTING)).is_ok() - } - /// Whether a request is pending. A read, for the pass that has nothing to take. pub fn pending(&self) -> bool { self.word.load(Ordering::Relaxed) & PENDING != NONE @@ -115,3 +109,29 @@ impl DumpRequest { was & PENDING != NONE } } + +/// How the pass that met a request was entered: the whole of what decides whether it may serve. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Entered { + /// `driver::pass`, by the preempt depth it was entered at, before it raised its own level. + Pass { depth: u32 }, + /// `driver::pass_block`, inside a wait ticket's registration window at every depth. + Blocking, +} + +impl Entered { + /// Only a pass entered at depth zero has nothing under it; a syscall's pass has its trap frame and a + /// blocking pass its wait ticket, and a report served inside either panics on its own depth assertion. + pub fn may_serve(self) -> bool { + matches!(self, Self::Pass { depth: 0 }) + } +} + +impl core::fmt::Display for Entered { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Blocking => write!(f, "a blocking pass"), + Self::Pass { depth } => write!(f, "a pass entered at preempt depth {depth}"), + } + } +} diff --git a/kernel/src/syscall/machine.rs b/kernel/src/syscall/machine.rs index 541cf8fee06..5cbb532931d 100644 --- a/kernel/src/syscall/machine.rs +++ b/kernel/src/syscall/machine.rs @@ -101,10 +101,6 @@ fn quiesce(last: &str) -> Result<(), SyscallError> { assert!(queued, "console-queue-at-the-stop: the queue had no room for its one line"); } crate::log::console::drain_for_the_stop(); - #[cfg(feature = "boot-actuators")] - if crate::actuator::quiesce_dump() { - crate::sched::dump::serve_for_the_stop(); - } log!("Syncing filesystems..."); // drain_all before sync_all: a closed-but-undrained file's dirty pages are only in the cache, which sync_all would miss. crate::writeback::drain_all(); diff --git a/kernel/src/syscall/proc.rs b/kernel/src/syscall/proc.rs index 1fc6adcd132..1ac63701348 100644 --- a/kernel/src/syscall/proc.rs +++ b/kernel/src/syscall/proc.rs @@ -22,8 +22,6 @@ use super::cancelled; use super::handles::{demand_syscap, handle_result}; pub(super) fn sys_thread_exit(code: i32) -> u64 { - #[cfg(feature = "boot-actuators")] - crate::quiesce::last::hold(crate::quiesce::last::Last::Exit); process::thread_exit(code); } @@ -153,26 +151,15 @@ pub(super) fn sys_thread_join(tid: u64) -> u64 { // None means never existed or already collected; the predicate below answers both. let target = process::thread_sched(caller, tid); let parkable = crate::scheduler::Parkable::at_entry(); - // Collecting takes the zombie out of the table, so the first answer is kept: asked - // again, the same join finds no such thread. - let answer = core::cell::Cell::new(None); - let settled = || { - answer.get().is_some() - || match process::wait_thread_zombie(tid, caller) { - Ok(None) => false, - Ok(Some(_)) => { - answer.set(Some(0)); - true - } - Err(()) => { - answer.set(Some(SyscallError::NotFound.to_u64())); - true - } - } - }; - while !settled() { + let join = toyos_proclife::join::Join::default(); + let ask = || process::ask_join(&join, tid, caller); + loop { + // Both refusals are one answer here; `JoinRefused` keeps them apart because they aren't the same fact. + if let Some(answer) = ask() { + return answer.map_or(SyscallError::NotFound.to_u64(), |_| 0); + } let Some(sched) = target.as_ref() else { - // Nothing to arm on and no zombie: wait_thread_zombie will never answer differently. + // Nothing to arm on and no zombie: the join will never answer differently. return SyscallError::NotFound.to_u64(); }; // Arms on the target thread's own watch, not a wake-by-name to the main thread. @@ -182,19 +169,18 @@ pub(super) fn sys_thread_join(tid: u64) -> u64 { tid.raw() as u64, WaitClass::Other, Deadline::never(), - settled, + || ask().is_some(), ) .is_err() { return cancelled(); } } - answer.get().expect("a settled join holds its answer") } pub(super) fn sys_nanosleep(nanos: u64) -> u64 { #[cfg(feature = "boot-actuators")] - crate::quiesce::last::hold(crate::quiesce::last::Last::Park); + crate::quiesce::last::hold(); // The ABI's relative span becomes an absolute Deadline here, and only here. let deadline = Deadline::at(crate::clock::now() + Duration::from_nanos(nanos)); // Armed on its own thread with no subject: nothing posts, only the deadline fires it. diff --git a/src/ci.rs b/src/ci.rs index 19a6b3c951a..4abafde7532 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -30,7 +30,7 @@ use std::path::{Path, PathBuf}; use std::process::Command; use crate::arch::Arch; -use crate::{flags, pr, release, sdkversion}; +use crate::{flags, pr, release, sdkversion, testargs}; /// The checks `main`'s ruleset must require, as `gate-stage` reads them back: /// a minimum, never an equality, so a name GitHub requires and this does not @@ -40,13 +40,18 @@ pub(crate) const REQUIRED_CHECKS: &[&str] = &["host"]; /// The one issue a red nightly files or comments on, found by title. const NIGHTLY_RED: &str = "nightly is red"; +/// `nightly.yml`'s two schedules: the nightly reach six nights a week, the weekly +/// reach on the seventh. +const NIGHTLY_CRON: &str = "0 3 * * 1-6"; +const WEEKLY_CRON: &str = "0 3 * * 0"; + const USAGE: &str = "cargo run -- --ci , where is one of: - host every host test: the build system, the host workspace, the - licences of what ships, clippy, the model controls, userland - and the SDK (ci.yml, nightly) + host every host test: the build system, the harness's own checks, + the host workspace, the licences of what ships, clippy, the + model controls, userland and the SDK (ci.yml, nightly) gate-stage what protects main, read back from GitHub (ci.yml) toolchain publish this tree's toolchain if nobody has (nightly) - guest / one shard of the whole guest suite, nightly tier included (nightly) + guest / one shard of the guest suite at the reach its schedule names (nightly) tcg one test on an emulated CPU (nightly) audio / one shard of gate A (nightly) nightly-red file or update the nightly-red issue from $NEEDS (nightly) @@ -98,9 +103,10 @@ pub fn dispatch(root: &Path, args: &[String]) { Job::Host => host(root), Job::GateStage => vec![step("what protects main", || gate_stage(root))], Job::Toolchain => vec![step("the toolchain release", || release::ensure_published(root))], - Job::Guest(shard) => { - guest(root, &suite_args(&["--shard", shard, "--jobs", "1", "--nightly"])) - } + Job::Guest(shard) => match guest_reach() { + Ok(reach) => guest(root, &suite_args(&["--shard", shard, "--jobs", "1", reach])), + Err(refusal) => vec![step("the reach", || Err(refusal))], + }, Job::Tcg => guest(root, &suite_args(&["--jobs", "1", "process_stats"])), Job::Audio(shard) => guest(root, &suite_args(&["--audio-gate", "30", "--shard", shard])), Job::NightlyRed => vec![step("the nightly-red issue", nightly_red)], @@ -435,6 +441,7 @@ fn host(root: &Path) -> Vec { let host_triple = crate::toolchain::host_triple(); let mut steps = vec![ step("the build system", || cargo(root, &["test", "--lib"])), + step("the harness's own checks", || cargo(root, &["test", "--test", "toyos-checks"])), step("the host workspace", || { cargo(root, &["test", "--workspace", "--exclude", "toyos-build"]) }), @@ -615,6 +622,42 @@ fn protection(rules: &serde_json::Value) -> (Vec, Vec) { // --- The guest jobs ------------------------------------------------------------ +/// The reach flag of the run that started this job: the weekly one on the weekly +/// schedule, and the nightly one on the other schedule, on a dispatch and off a +/// runner. +fn guest_reach() -> Result<&'static str, String> { + reach_of_event(std::env::var("GITHUB_EVENT_NAME").ok().as_deref(), || { + let path = std::env::var("GITHUB_EVENT_PATH").map_err(|_| { + "a scheduled run with no $GITHUB_EVENT_PATH names no schedule".to_string() + })?; + std::fs::read_to_string(&path).map_err(|e| format!("{path}: {e}")) + }) +} + +/// [`guest_reach`] over the event's name and a reader of its payload, which +/// only a scheduled run asks for. +fn reach_of_event( + name: Option<&str>, + payload: impl FnOnce() -> Result, +) -> Result<&'static str, String> { + if name != Some("schedule") { + return Ok(testargs::NIGHTLY.name); + } + let event: serde_json::Value = + serde_json::from_str(&payload()?).map_err(|e| format!("the schedule event: {e}"))?; + reach_of_schedule(event["schedule"].as_str()) +} + +fn reach_of_schedule(cron: Option<&str>) -> Result<&'static str, String> { + match cron { + Some(NIGHTLY_CRON) => Ok(testargs::NIGHTLY.name), + Some(WEEKLY_CRON) => Ok(testargs::WEEKLY.name), + other => Err(format!( + "a scheduled run of {other:?}, which is neither schedule nightly.yml declares" + )), + } +} + fn suite_args(args: &[&str]) -> Vec { let mut all = vec!["test", "--test", "toyos-build", "--"]; all.extend(args); @@ -941,6 +984,44 @@ mod tests { assert!(parse(&[]).is_err()); } + #[test] + fn each_schedule_names_its_reach_and_another_is_refused() { + let nightly = reach_of_schedule(Some(NIGHTLY_CRON)); + let weekly = reach_of_schedule(Some(WEEKLY_CRON)); + assert_eq!(nightly, Ok("--nightly"), "the nightly schedule's reach"); + assert_eq!(weekly, Ok("--weekly"), "the weekly schedule's reach"); + for stray in [Some("0 4 * * *"), None] { + let refusal = reach_of_schedule(stray).unwrap_err(); + assert!(refusal.contains(&format!("{stray:?}")), "{refusal}"); + } + } + + /// Only a scheduled run reads its payload, and every other run is a nightly one. + #[test] + fn a_run_no_schedule_started_reaches_nightly() { + for name in [None, Some("workflow_dispatch"), Some("push")] { + let reach = reach_of_event(name, || panic!("{name:?} read a schedule payload")); + assert_eq!(reach, Ok("--nightly"), "a {name:?} run"); + } + let weekly = format!(r#"{{"schedule":"{WEEKLY_CRON}"}}"#); + assert_eq!(reach_of_event(Some("schedule"), || Ok(weekly)), Ok("--weekly")); + let unread = reach_of_event(Some("schedule"), || Err("no payload".into())); + assert_eq!(unread, Err("no payload".into())); + } + + /// The schedules `nightly.yml` declares are exactly the two a reach is + /// named for, so no scheduled run reaches the refusal above. + #[test] + fn nightly_yml_declares_the_two_schedules() { + let text = std::fs::read_to_string(repo_root().join(".github/workflows/nightly.yml")) + .expect("nightly.yml is readable"); + let crons: Vec<&str> = text + .lines() + .filter_map(|l| l.trim().strip_prefix("- cron: '")?.strip_suffix('\'')) + .collect(); + assert_eq!(crons, [NIGHTLY_CRON, WEEKLY_CRON]); + } + /// Teeth for the controls' judge: a green negative control, a control that /// never reached its verdict, and a self-catching case that failed are all /// red. diff --git a/src/redlist.rs b/src/redlist.rs index 884f0760969..88d243fd6fc 100644 --- a/src/redlist.rs +++ b/src/redlist.rs @@ -25,10 +25,6 @@ pub struct Disabled { /// Every disabled test. pub const DISABLED: &[Disabled] = &[ - Disabled { - test: "console_line_atomicity", - issue: "issues/build/console-line-atomicity-loses-five-of-a-thousand-lines-on-ci.md", - }, Disabled { test: "console_locale_detect", issue: "issues/build/the-console-input-path-can-stop-after-a-ps2-overflow.md", @@ -64,18 +60,10 @@ pub const DISABLED: &[Disabled] = &[ test: "partition_claim_departure", issue: "issues/boot-media/partition-claim-departure-exits-clean-with-none-of-its-refusals-said.md", }, - Disabled { - test: "quiesce_dump_holds_the_stopped", - issue: "issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md", - }, Disabled { test: "quiesce_stops_the_machine", issue: "issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md", }, - Disabled { - test: "quiesce_wakes_on_the_last_exit", - issue: "issues/build/quiesce-wakes-on-the-last-exit-lost-its-serial-ready-beside-other-guests.md", - }, Disabled { test: "quiesce_wakes_on_the_last_park", issue: "issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md", @@ -92,10 +80,6 @@ pub const DISABLED: &[Disabled] = &[ test: "short_sleep_livelock", issue: "issues/kernel/short-sleep-livelock-stalls-on-ci-with-one-sleeper-never-returning.md", }, - Disabled { - test: "so_cache_refusals", - issue: "issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md", - }, Disabled { test: "swap_crash_rolls_back", issue: "issues/build/a-swaps-redial-races-a-hard-dial-ceiling-against-an-unbounded-guest-gap.md", @@ -104,10 +88,6 @@ pub const DISABLED: &[Disabled] = &[ test: "swap_netd", issue: "issues/build/a-swaps-redial-races-a-hard-dial-ceiling-against-an-unbounded-guest-gap.md", }, - Disabled { - test: "usb_disk_index_stable", - issue: "issues/hardware/usb-disk-index-stable-nothing-enumerates-on-the-first-controller.md", - }, Disabled { test: "usb_transport_break", issue: "issues/kernel/a-held-disk-waits-for-a-pass-no-cpu-takes-when-every-cpu-is-in-a-call-on-it.md", diff --git a/src/testargs.rs b/src/testargs.rs index 58d0fa8c519..fd11308331e 100644 --- a/src/testargs.rs +++ b/src/testargs.rs @@ -155,6 +155,7 @@ declare_flags!(pub SUITE = { pub SHARD = "--shard", Next; pub SLOW_USB = "--slow-usb", None; pub NIGHTLY = "--nightly", None; + pub WEEKLY = "--weekly", None; /// The metal profile: the registrations that run on the T14, batched into /// images and judged off the log the stick came back with. pub METAL = "--metal", None; @@ -218,13 +219,23 @@ pub fn parse(args: &[String]) -> Result, String> { .to_string(), ); } - if has(&NIGHTLY) && has(&AUDIO_GATE) { + if has(&NIGHTLY) && has(&WEEKLY) { return Err( - "--nightly and --audio-gate are separate tiers and cannot be combined; run one \ - tier at a time" + "--weekly runs the nightly tier too, so a --nightly beside it would be read by \ + nothing; write one" .to_string(), ); } + for reach in [&NIGHTLY, &WEEKLY] { + for other in [&AUDIO_GATE, &METAL] { + if has(reach) && has(other) { + return Err(format!( + "{} and {} are separate tiers and cannot be combined; run one tier at a time", + reach.name, other.name + )); + } + } + } Ok(filter) } @@ -260,18 +271,27 @@ mod tests { } #[test] - fn nightly_and_audio_gate_are_refused_by_the_argv_validator() { - for argv in [ - vec!["--nightly", "--audio-gate", "30"], - vec!["--audio-gate=30", "--nightly"], + fn a_reach_and_another_tier_are_refused_by_the_argv_validator() { + for (argv, reach, other) in [ + (vec!["--nightly", "--audio-gate", "30"], "--nightly", "--audio-gate"), + (vec!["--audio-gate=30", "--nightly"], "--nightly", "--audio-gate"), + (vec!["--weekly", "--audio-gate", "30"], "--weekly", "--audio-gate"), + (vec!["--metal", "--nightly"], "--nightly", "--metal"), + (vec!["--weekly", "--metal"], "--weekly", "--metal"), ] { let refusal = parse_owned(&argv).unwrap_err(); - assert!(refusal.contains("--nightly"), "{refusal}"); - assert!(refusal.contains("--audio-gate"), "{refusal}"); - assert!(refusal.contains("cannot be combined"), "{refusal}"); + assert!(refusal.contains(reach) && refusal.contains(other), "{argv:?}: {refusal}"); + assert!(refusal.contains("cannot be combined"), "{argv:?}: {refusal}"); } } + #[test] + fn the_weekly_reach_refuses_a_nightly_it_already_runs() { + let refusal = parse_owned(&["--nightly", "--weekly"]) + .expect_err("a --nightly beside --weekly was accepted and read by nothing"); + assert!(refusal.contains("read by nothing"), "{refusal}"); + } + #[test] fn the_filter_is_the_word_that_is_nobodys_value() { assert_eq!(parse_owned(&["process_stats"]).unwrap().as_deref(), Some("process_stats")); @@ -485,10 +505,11 @@ mod tests { vec!["--jobs", "4"], vec!["--shard", "2/4"], vec!["--nightly"], + vec!["--weekly"], + vec!["--weekly", "--shard", "2/12", "--jobs", "1"], vec!["--debug"], vec!["--metal"], vec!["--metal", "--metal-readback", "target/metal"], - vec!["--metal", "--nightly"], ] { assert!(parse_owned(&argv).is_ok(), "{argv:?}"); } diff --git a/src/tiers.rs b/src/tiers.rs index f0c57216b9f..93a170134c0 100644 --- a/src/tiers.rs +++ b/src/tiers.rs @@ -2,26 +2,29 @@ //! //! **The table is the registration.** Every row of `tests/toyos.rs`'s //! `MACHINE_TESTS`, `SCREEN_TESTS` and `AUDIO_TESTS` carries its [`Tier`] or -//! does not compile, and the shared boot's discovered tests share one. Moving a -//! test between tiers is editing that one word. +//! does not compile, and the shared boot's discovered tests share one. A +//! [`Schedule`] is every one of those names at its one tier. Moving a test +//! between tiers is editing that one word. //! -//! The fast tier is what every plain `cargo test` runs; the nightly tier is -//! `--nightly`, run by `.github/workflows/nightly.yml`. No pull request boots a -//! guest. The line between the two is 10 s on CI's hosted shards, and nothing -//! enforces it: a name that is Nightly because its verdict is anchored to real -//! time, or because it shares a boot with one that is slow, stays where it is -//! whatever it measures. +//! The tiers nest: a plain `cargo test` reaches `Fast`, `--nightly` adds +//! `Nightly`, and `--weekly` adds `Weekly` to that. //! -//! The local tier is the third, and the only one CI never runs: its guests are +//! The local tier is the fourth, and the only one CI never runs: its guests are //! of an architecture no hosted runner has been measured to boot. +use std::collections::BTreeMap; + +use crate::testargs::{NIGHTLY, SUITE, WEEKLY}; + /// Which run a registered test belongs to. #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub enum Tier { /// Every `cargo test`. Fast, - /// `cargo test --test toyos-build -- --nightly`. + /// `cargo test --test toyos-build -- --nightly`, and every weekly run. Nightly, + /// `cargo test --test toyos-build -- --weekly`, and before a release. + Weekly, /// Every `cargo test` on a developer's machine and no sharded run: a guest /// of an architecture no CI runner boots yet. The AArch64 port's stage 8 /// (`issues/kernel/toyos-runs-on-arm64.md`) measures the runners and @@ -29,14 +32,143 @@ pub enum Tier { Local, } +/// How far down the nested tiers a run reaches. +#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Debug)] +pub enum Reach { + Fast, + Nightly, + Weekly, +} + +impl Reach { + /// The reach `args` ask for; `testargs::parse` refuses both flags at once. + pub fn of(args: &[String]) -> Self { + if SUITE.present(args, &WEEKLY) { + Self::Weekly + } else if SUITE.present(args, &NIGHTLY) { + Self::Nightly + } else { + Self::Fast + } + } +} + impl Tier { - /// Whether a run selects this tier: `nightly` is `--nightly`, `sharded` - /// a `--shard`, which only CI's jobs pass. - pub fn selected(self, nightly: bool, sharded: bool) -> bool { + /// Whether a run that reaches `reach` selects this tier; `sharded` is a + /// `--shard`, which only CI's jobs pass. + pub fn selected(self, reach: Reach, sharded: bool) -> bool { match self { Self::Fast => true, - Self::Nightly => nightly, + Self::Nightly => reach >= Reach::Nightly, + Self::Weekly => reach >= Reach::Weekly, Self::Local => !sharded, } } + + /// The flag of the narrowest run that selects this tier, for a run that + /// held it back to name; `None` for the tiers every unsharded run takes. + pub fn flag(self) -> Option<&'static str> { + match self { + Self::Nightly => Some(NIGHTLY.name), + Self::Weekly => Some(WEEKLY.name), + Self::Fast | Self::Local => None, + } + } +} + +/// Every registered test at its one tier. +pub struct Schedule<'a>(BTreeMap<&'a str, Tier>); + +impl<'a> Schedule<'a> { + /// Refuses a name registered twice, by name, at one tier or two: two rows + /// are two verdicts under one name. + pub fn new(rows: impl IntoIterator) -> Result { + let mut tiers = BTreeMap::new(); + for (name, tier) in rows { + if let Some(first) = tiers.insert(name, tier) { + return Err(format!( + "{name} is registered twice, at {first:?} and at {tier:?}: one name is one \ + verdict at one tier" + )); + } + } + Ok(Self(tiers)) + } + + /// Whether anything registers `name`. + pub fn contains(&self, name: &str) -> bool { + self.0.contains_key(name) + } + + /// Every registered name and its tier, by name. + pub fn iter(&self) -> impl Iterator + '_ { + self.0.iter().map(|(name, tier)| (*name, *tier)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const EVERY: [Tier; 4] = [Tier::Fast, Tier::Nightly, Tier::Weekly, Tier::Local]; + const NAMES: [&str; 4] = ["fast_one", "nightly_one", "weekly_one", "local_one"]; + + #[test] + fn every_registered_test_has_exactly_one_tier() { + let schedule = Schedule::new(NAMES.into_iter().zip(EVERY)).unwrap(); + let mut want: Vec<_> = NAMES.into_iter().zip(EVERY).collect(); + want.sort_by_key(|(name, _)| *name); + assert_eq!(schedule.iter().collect::>(), want); + for second in EVERY { + let refusal = + Schedule::new([("twice", Tier::Fast), ("once", Tier::Weekly), ("twice", second)]) + .err() + .expect("a second row for one name is refused"); + assert!(refusal.starts_with("twice is registered twice"), "{refusal}"); + } + } + + #[test] + fn only_a_registered_name_is_contained() { + let schedule = Schedule::new([("registered", Tier::Weekly)]).unwrap(); + assert!(schedule.contains("registered")); + assert!(!schedule.contains("never_registered")); + assert!(!schedule.contains("registere"), "a prefix is not the name"); + } + + /// Each reach selects its own tier and every narrower one, and nothing + /// wider; a shard alone decides `Local`. + #[test] + fn the_tiers_nest() { + let selected = |reach| -> Vec { + EVERY.into_iter().filter(|tier| tier.selected(reach, true)).collect() + }; + let nested = [ + (Reach::Fast, &[Tier::Fast][..]), + (Reach::Nightly, &[Tier::Fast, Tier::Nightly]), + (Reach::Weekly, &[Tier::Fast, Tier::Nightly, Tier::Weekly]), + ]; + for (reach, tiers) in nested { + assert_eq!(selected(reach), tiers, "a {reach:?} run selects the wrong tiers"); + } + for reach in [Reach::Fast, Reach::Nightly, Reach::Weekly] { + assert!(Tier::Local.selected(reach, false) && !Tier::Local.selected(reach, true)); + } + } + + /// The flag a held-back tier names is the one whose reach selects it. + #[test] + fn a_held_tiers_flag_selects_it() { + for tier in EVERY { + let Some(flag) = tier.flag() else { + let plain = tier.selected(Reach::Fast, false); + assert!(plain, "{tier:?} names no flag, so a plain run takes it"); + continue; + }; + let reach = Reach::of(&[flag.to_string()]); + assert!(tier.selected(reach, true), "{flag} does not select {tier:?}"); + assert!(!tier.selected(Reach::Fast, true), "{tier:?} names {flag} and needs none"); + } + assert_eq!(Reach::of(&[]), Reach::Fast); + } } diff --git a/tests/CLAUDE.md b/tests/CLAUDE.md index 4e5d5e966cb..3d5d594383b 100644 --- a/tests/CLAUDE.md +++ b/tests/CLAUDE.md @@ -1,6 +1,6 @@ # Tests -The mechanics live where the work is: profiles and shapes in `tests/common/`, registration and tiers in `tests/toyos.rs`, the fast tier's line in `src/tiers.rs` — read those, not this file, for how the harness works. +The mechanics live where the work is: profiles and shapes in `tests/common/`, registration and tiers in `tests/toyos.rs` — read those, not this file, for how the harness works. ## Caveats that bite every agent diff --git a/tests/checks.rs b/tests/checks.rs new file mode 100644 index 00000000000..bf9369983ba --- /dev/null +++ b/tests/checks.rs @@ -0,0 +1,653 @@ +//! The harness's own checks, none of which boots a guest: `tests/toyos.rs` +//! included under the libtest harness, so every path in it resolves here, and +//! the checks beside it in a module only this target compiles. + +include!("toyos.rs"); + +mod checks { + use super::*; + + /// One subject: what a console line says died, what a wait does about it, + /// and that only one place in the harness answers either. + #[test] + fn serial_vocabulary() -> Result<(), String> { + serial::self_check()?; + qemu::ceiling_self_check()?; + qemu::host_scale_self_check()?; + one_vocabulary() + } + + /// A wait that hands a death spelling to a scan of its own, asked of the + /// harness's source. + /// + /// **The one place the vocabulary lives is the whole of the fix, so this is + /// what keeps it the one place.** + /// + /// The way that comes back is the obvious patch: one more spelling handed + /// straight to a `contains` beside the call. It would match a *program's* panic + /// as readily as the kernel's and take the run down with a guest binary that + /// was expected to die — so the shape is refused by name rather than left to a + /// reviewer. Comment lines go first: this file argues about these words at + /// length, and prose is not a second answer. + fn hand_rolled_deaths(text: &str) -> Vec { + let mut found = Vec::new(); + for (n, line) in text.lines().enumerate() { + if line.trim_start().starts_with("//") { + continue; + } + for word in serial::spellings() { + // The shape is the spelling as somebody's first argument — + // `contains`, `starts_with`, `find`, any of them. A spelling + // *inside* a longer staged line is how this file's own gates build + // their inputs, and those are not scans. + if line.contains(&format!("(\"{word}")) { + found.push(format!("{}:{}: {}", n + 1, word, line.trim())); + } + } + } + found + } + + /// [`hand_rolled_deaths`] over the file that has to stay clean, with its own + /// bad input beside it so a check that stopped finding anything says so. + fn one_vocabulary() -> Result<(), String> { + const FILE: &str = "tests/common/qemu.rs"; + let path = Path::new(env!("CARGO_MANIFEST_DIR")).join(FILE); + let text = fs::read_to_string(&path).map_err(|e| format!("read {}: {e}", path.display()))?; + let found = hand_rolled_deaths(&text); + if !found.is_empty() { + return Err(format!( + "{FILE} scans for a death spelling itself, and `serial::died` is where that is \ + decided for every wait at once — a second answer here is what let a kernel panic \ + read as a stall, and it matches a program's own panic besides:\n {}", + found.join("\n ") + )); + } + // The negative control. Every line of it is a shape this must name, and the + // last two are shapes it must not: prose, and a staged capture built out of + // the same words. + let staged = "\ + } else if line.contains(\"KERNEL PANIC\") {\n\ + if line.starts_with(\"SEGFAULT\") {\n\ + // ends on `PANIC:` and nothing else, which is the defect\n\ + const KERNEL: &str = \"[kernel 1.450 cpu3] PANIC: panicked at reserve.rs:812:9:\";\n"; + let named = hand_rolled_deaths(staged); + if named.len() != 2 { + return Err(format!( + "the check names {} of the two hand-rolled scans staged for it: {named:?}", + named.len() + )); + } + eprintln!(" [vocabulary] {FILE} asks `serial::died` and nothing else"); + Ok(()) + } + + #[test] + fn suspend_detector() -> Result<(), String> { + common::clock::self_check() + } + + /// What a suspend is worth to a verdict, staged rather than reasoned about. + /// + /// `common::clock::self_check` gates the detector; this gates what the suite + /// does with what it detects. Both halves are needed and neither implies the + /// other: **a suspend that silently passes is as bad as one that silently + /// fails**, and here the two are one line apart. + #[test] + fn suspend_invalidates_a_verdict() -> Result<(), String> { + let slept = common::clock::SUSPENDED_AT_LEAST + Duration::from_secs(120); + let awake = Duration::ZERO; + // Under the threshold on purpose: two clock reads jitter against each other + // by microseconds, and a run must not be thrown away for that. + let jitter = common::clock::SUSPENDED_AT_LEAST + .checked_sub(Duration::from_millis(1)) + .expect("SUSPENDED_AT_LEAST must be at least 1ms for this case to mean anything"); + let cases: [(&str, Option<&str>, Duration, Verdict); 6] = [ + ("a pass on a host that stayed up", None, awake, Verdict::Pass), + ("a fail on a host that stayed up", Some("the guest said no"), awake, Verdict::Fail), + ("a pass across a suspend", None, slept, Verdict::Invalid), + ("a fail across a suspend", Some("timed out"), slept, Verdict::Invalid), + ("a pass across clock jitter", None, jitter, Verdict::Pass), + ("a fail across clock jitter", Some("the guest said no"), jitter, Verdict::Fail), + ]; + for (what, reason, suspended, want) in cases { + let outcome = Outcome { + name: what.to_string(), + reason: reason.map(str::to_string), + elapsed: Duration::from_secs(3), + suspended, + }; + let got = outcome.verdict(); + if got != want { + return Err(format!("{what} is {got:?}, and it has to be {want:?}")); + } + } + Ok(()) + } + + /// A blown guard stays red, and stops reading as an answer. + /// + /// Both halves, because each fails the other's way round. An implementation + /// that made a stall its own non-red status would hide a guest that genuinely + /// stops; one that only renamed the line would leave the summary saying a test + /// found something. Staged against the strings a wait actually produces rather + /// than against the marker on its own, because a caller prefixes its own + /// sentence to [`await_marker`]'s and the classification has to survive that. + #[test] + fn stall_is_not_a_verdict() -> Result<(), String> { + // Built from the marker rather than copied, so a rename cannot leave the + // gate asserting against a string nothing produces any more. + let real = format!("{STALLED} waiting for the long tone to start — it went quiet"); + let under_a_sentence = format!("the compositor stopped painting\n{real}"); + let cases: [(&str, Option<&str>, bool); 4] = [ + ("an ordinary red", Some("the pointer never moved right"), false), + ("a wait that expired", Some(real.as_str()), true), + ( + "a wait that expired under a caller's own sentence", + Some(under_a_sentence.as_str()), + true, + ), + ("a pass", None, false), + ]; + for (what, reason, want_stall) in cases { + let outcome = Outcome { + name: what.to_string(), + reason: reason.map(str::to_string), + elapsed: Duration::from_secs(1), + suspended: Duration::ZERO, + }; + if outcome.stalled() != want_stall { + return Err(format!( + "{what} reads as stalled={}, and it has to be {want_stall}", + outcome.stalled() + )); + } + // Red is red. A stall that stopped failing the run would be a gate that + // reports and enforces nothing. + let red = outcome.verdict() == Verdict::Fail; + if red != reason.is_some() { + return Err(format!("{what} is red={red}, and a reason is always red")); + } + } + + let mut tally = Tally::new(); + tally.record(Outcome { + name: "a_stalled_test".to_string(), + reason: Some(format!("{STALLED} waiting for nothing at all — it went quiet")), + elapsed: Duration::from_secs(1), + suspended: Duration::ZERO, + }); + tally.record(Outcome { + name: "a_wrong_answer".to_string(), + reason: Some("the pointer never moved right".to_string()), + elapsed: Duration::from_secs(1), + suspended: Duration::ZERO, + }); + if tally.exit_code() != 1 { + return Err(format!("two reds exited {}, and they have to red", tally.exit_code())); + } + if tally.stalls != ["a_stalled_test"] { + return Err(format!( + "the run named {:?} as blown guards; it has to name exactly the one that was", + tally.stalls + )); + } + let summary = tally.summary(2, Duration::from_secs(2), Duration::ZERO); + if !summary.contains("1 of those reds are blown liveness guards") { + return Err(format!("the summary does not separate the two kinds of red:\n{summary}")); + } + Ok(()) + } + + /// One live guest holds its lane's NVMe image, and the next one may not. + /// + /// The ordering itself is now the type's: `boot` takes a [`qemu::LaneFree`] and + /// the only thing that makes one out of a guest is `QemuInstance::shutdown`, + /// which takes it by value. What is left to check at runtime is the claim + /// underneath — that a hold is real while a guest is up and gone once it is + /// not — and this checks it on the harness's own registry, in both directions, + /// with no guest. + #[test] + fn nvme_image_is_held_by_one_guest() -> Result<(), String> { + // Names, not files: a claim is a hold on a path and touches no disk, so + // nothing here has to create or delete a hundred megabytes to ask. + let dir = Path::new(env!("CARGO_TARGET_TMPDIR")); + let image = dir.join("nvme-claim-gate.img"); + let other = dir.join("nvme-claim-gate-other.img"); + + let held = qemu::NvmeClaim::take(&image).map_err(|why| { + format!("a free image refused its first guest: {why}") + })?; + + // The overlap. This is the direction that must red, and it is what the + // reboot produced. + match qemu::NvmeClaim::take(&image) { + Ok(_) => { + return Err(format!( + "a second guest took {}, which a live one is holding — two QEMUs are then \ + handed one image and the second dies on its lock", + image.display() + )) + } + Err(why) => { + // The refusal has to name the image, or it cannot be acted on: a + // run makes dozens of guests and the message is all a reader gets. + if !why.contains(&image.display().to_string()) { + return Err(format!("the refusal does not name the image it is about: {why}")); + } + } + } + + // A different image is not a conflict, or every lane would refuse every + // other lane's boot the moment this gate had teeth. + let elsewhere = qemu::NvmeClaim::take(&other) + .map_err(|why| format!("an unheld image was refused: {why}"))?; + drop(elsewhere); + + // And the ordinary reboot: the replacement takes the image the guest it + // replaces released. Green, and it is the half a fix that simply refused + // every second boot would break. + drop(held); + let replacement = qemu::NvmeClaim::take(&image).map_err(|why| { + format!("a replacement was refused the image its predecessor released: {why}") + })?; + drop(replacement); + Ok(()) + } + + /// [`control_regs`] against machines this host cannot boot, with no guest. + /// + /// [`control_regs_negative`] runs the real defective machine and is the link + /// between this verdict and a kernel; what is here is the states no actuator + /// reaches — a CPU that differs from three others, a bit set uniformly on all + /// four, an AP that never printed. Every value is one this tree has printed or + /// one bit away from it. + #[test] + fn control_regs_verdict() -> Result<(), String> { + const AP_BEFORE: (u64, u64) = (0xe000_0011, 0x0031_0620); + const DECLARED: (u64, u64) = (0x8001_0033, 0x0030_0668); + + fn log(cpus: &[(u64, u64)]) -> String { + cpus.iter() + .enumerate() + .map(|(i, (cr0, cr4))| { + format!("[kernel 0.1 cpu{i}] control_regs: cpu{i} cr0={cr0:#010x} cr4={cr4:#010x}\n") + }) + .collect() + } + + let refused = |what: &str, cpus: &[(u64, u64)], says: &str| match control_regs(&log(cpus), 4) { + Ok(()) => Err(format!("{what} was accepted")), + Err(e) if e.contains(says) => Ok(()), + Err(e) => Err(format!("{what} was refused for the wrong reason: {e}")), + }; + + // Positive control first: a verdict that refuses everything refuses the + // defect too, and would prove nothing below. + control_regs(&log(&[DECLARED; 4]), 4) + .map_err(|e| format!("the declared machine was refused: {e}"))?; + + refused("the machine this tree booted", &[DECLARED, AP_BEFORE, AP_BEFORE, AP_BEFORE], "CD")?; + // The case a "do all the CPUs agree?" test passes: they agree, on INIT's + // value. Nothing about uniformity says caching is on. + refused("four CPUs agreeing on INIT's CR0", &[AP_BEFORE; 4], "CD")?; + refused( + "one CPU without WP", + &[DECLARED, (DECLARED.0 & !(1 << 16), DECLARED.1), DECLARED, DECLARED], + "WP", + )?; + refused( + "one CPU without NE", + &[DECLARED, DECLARED, (DECLARED.0 & !(1 << 5), DECLARED.1), DECLARED], + "NE", + )?; + // The bit that must be *absent*: with it set, XCR0 can name components + // FXSAVE64 does not save. + refused("OSXSAVE set", &[(DECLARED.0, DECLARED.1 | (1 << 18)); 4], "OSXSAVE")?; + // Two bits a machine could hold uniformly, each one line of kernel diff + // away, and neither reachable by an actuator. `AM` is named clear above and + // answers by name; `PGE` is named nowhere, which is the case the whole + // never-named rule exists for — `TSD` and `PKE` are the same case. + refused("every CPU with AM set", &[(DECLARED.0 | (1 << 18), DECLARED.1); 4], "AM")?; + refused( + "every CPU with PGE set", + &[(DECLARED.0, DECLARED.1 | (1 << 7)); 4], + "never named", + )?; + refused("every CPU without SMEP", &[(DECLARED.0, DECLARED.1 & !(1 << 20)); 4], "SMEP")?; + // A CPU that agrees about every named bit and differs in one the CPU is + // allowed to withhold, so nothing above it can object. + refused("one CPU with PCID and three without", &[DECLARED, DECLARED, DECLARED, (DECLARED.0, DECLARED.1 | (1 << 17))], "cpu3")?; + // And an AP that never printed at all, which is what a machine whose AP + // died before the check looks like. + refused("three lines for four CPUs", &[DECLARED; 3], "{0, 1, 2, 3}")?; + + eprintln!(" [control_regs] the verdict refuses 10 machines and accepts the declared one"); + Ok(()) + } + + /// [`idle_is_spinning`] against a healthy trace and a crafted one shaped like + /// the regression it exists to catch, with no guest — the same split + /// `control_regs`/`control_regs_verdict` use, and for the same reason: a + /// gate's own teeth are a claim a live boot cannot demonstrate on the + /// negative side, because nothing in this tree can stage a CPU into spinning + /// through idle on purpose. + #[test] + fn i8042_quarantine_verdict() -> Result<(), String> { + let healthy = "\ + [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ + [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ + [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=3\n\ + [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n"; + if let Some((cpu, delta)) = idle_is_spinning(healthy) { + return Err(format!("a healthy trace was refused: cpu{cpu} moved by {delta}")); + } + + // The regression's own shape: one CPU quarantines cleanly and stays + // quiet, the other's undrained ring never lets it halt. + let spinning = "\ + [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ + [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=4\n\ + [kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n\ + [kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=2685004\n"; + match idle_is_spinning(spinning) { + Some((1, delta)) if delta > MAX_IDLE_TRIP_DELTA => {} + Some((cpu, delta)) => { + return Err(format!("refused the wrong CPU or by the wrong margin: cpu{cpu} delta {delta}")) + } + None => return Err("a spinning CPU's trace was accepted".to_string()), + } + + // And the line the old, count-of-lines check would have been fooled by: + // the same number of `sched: cpu=` lines either way, because the print + // itself is rate-limited regardless of what is underneath it — which is + // exactly the vacuity this replaces. + assert_eq!( + healthy.matches("sched: cpu=").count(), + spinning.matches("sched: cpu=").count(), + "the crafted traces must differ only in trips=, not in line count — otherwise this proves nothing about the old check's blindness" + ); + + eprintln!(" [i8042] the idle-trip verdict accepts a healthy trace and refuses a spinning one"); + Ok(()) + } + + /// What the declaration itself has to be, before any of it means anything. + /// Which shared-boot binaries need `SYS_DEBUG`, asked of their source. + /// + /// A name reaches the syscall directly, or through a child it spawns. + fn needs_actuators(sources: &[(String, String)], registry: &[&str]) -> BTreeSet { + // The fourth spelling is the argument-taking form: every action that + // carries a payload (TLB_ACK_DELAY_ARM, CENSUS_KIND, LOWER_SYSINFO_BOUND, + // SLOT_TO_LAST_GENERATION) is reached through `debug_with`, never + // `debug`. The third is the SDK's: `toyos::census` calls `debug_with` on + // the caller's behalf, so a binary whose leak assertion is a census names + // no syscall of its own and reads as innocent to the others. + let calls = |text: &str| { + text.contains("SYS_DEBUG") + || text.contains("syscall::debug(") + || text.contains("syscall::debug_with(") + || text.contains("census::Census") + }; + let direct: BTreeSet<&str> = + sources.iter().filter(|(_, t)| calls(t)).map(|(n, _)| n.as_str()).collect(); + let mut out = BTreeSet::new(); + for (name, text) in sources { + if !registry.contains(&name.as_str()) { + continue; + } + let spawns = direct.iter().any(|d| text.contains(&format!("test_rs_{d}"))); + if direct.contains(name.as_str()) || spawns { + out.insert(name.clone()); + } + } + out + } + + /// [`ACTUATOR_TESTS`] is exactly the shared-boot binaries that reach + /// `SYS_DEBUG`, and the binaries are what is asked. + /// + /// **What this does not cover, stated because the hole is real:** a machine or + /// screen test that *drives* one of those binaries on a boot of its own. No + /// static rule here can say which `BootOptions` a `run_test` call belongs to. + /// What answers it instead is the guest: `test_panic_child` names + /// `InvalidArgument` as *this kernel carries no actuators* rather than reporting + /// a kernel that failed to stop, so the red says what is wrong wherever it + /// happens. + /// + /// **Both directions are the point.** A binary that gains a `debug()` call and + /// no entry would run on the shipping kernel, where the syscall answers + /// `InvalidArgument` — and a test whose verdict is that a process died would + /// then fail for a reason with nothing to do with what it is about. An entry + /// whose binary no longer calls it is a test kept off the shipping kernel for + /// nothing, which is the erosion this split exists to stop. + #[test] + fn suite_split() -> Result<(), String> { + let dir = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/toyos-rust-tests/src/bin"); + let mut sources: Vec<(String, String)> = Vec::new(); + for entry in fs::read_dir(&dir).map_err(|e| format!("read {}: {e}", dir.display()))? { + let path = entry.map_err(|e| e.to_string())?.path(); + if path.extension().is_none_or(|e| e != "rs") { + continue; + } + let name = path.file_stem().unwrap().to_string_lossy().into_owned(); + let text = fs::read_to_string(&path).map_err(|e| format!("read {name}: {e}"))?; + sources.push((name, text)); + } + let registry: Vec<&str> = + sources.iter().map(|(n, _)| n.as_str()).filter(|n| !RUST_SKIP.contains(n)).collect(); + + // The negative control, and it carries its own bad input: a binary that + // calls the syscall and is on no list must be named, or the check above is + // a spelling of `true`. + let staged = vec![ + ("a_listed_one".to_string(), "syscall::debug(3)".to_string()), + ("an_unlisted_one".to_string(), "SYS_DEBUG".to_string()), + ("its_parent".to_string(), "Command::new(\"/system/bin/test_rs_an_unlisted_one\")".to_string()), + ("a_censor".to_string(), "use toyos::census::Census;".to_string()), + ("a_debug_with_user".to_string(), "syscall::debug_with(3, 4)".to_string()), + ("innocent".to_string(), "println!()".to_string()), + ]; + let staged_registry = [ + "a_listed_one", + "an_unlisted_one", + "its_parent", + "a_censor", + "a_debug_with_user", + "innocent", + ]; + let found = needs_actuators(&staged, &staged_registry); + let want: BTreeSet = + ["a_listed_one", "an_unlisted_one", "its_parent", "a_censor", "a_debug_with_user"] + .iter() + .map(|s| s.to_string()) + .collect(); + if found != want { + return Err(format!("the check does not work: on staged input it named {found:?}")); + } + + let want: BTreeSet = needs_actuators(&sources, ®istry); + let listed: BTreeSet = ACTUATOR_TESTS.iter().map(|s| s.to_string()).collect(); + let missing: Vec<&String> = want.difference(&listed).collect(); + if !missing.is_empty() { + return Err(format!( + "{missing:?} reach SYS_DEBUG and are on the shipping boot, where the syscall answers \ + InvalidArgument. Add each to ACTUATOR_TESTS, or to RUST_SKIP if it is driven rather \ + than run." + )); + } + let stale: Vec<&String> = listed.difference(&want).collect(); + if !stale.is_empty() { + return Err(format!( + "{stale:?} are held off the shipping kernel and no longer reach SYS_DEBUG. Delete \ + each entry — coverage of the binary an image ships is what it costs." + )); + } + // **The other shape [`schedule`] cannot see**: a binary a machine + // test drives under a *different* name is still discovered here, still runs + // on the shared boot, and there passes on its exit code with nothing staged + // for it to act on. + let staged_driven = [ + String::from("qemu.run_test(\"test_rs_a_driven_one\", Duration::from_secs(30))"), + String::from("Command::new(\"/system/bin/test_rs_another_driven\")"), + String::from("qemu.run_test(&format!(\"test_rs_{name}\"), ceiling)"), + ]; + let found = driven_binaries(&staged_driven); + let want: BTreeSet = + ["a_driven_one", "another_driven"].iter().map(|s| s.to_string()).collect(); + if found != want { + return Err(format!("the driven-name reader does not work: it named {found:?}")); + } + + let root = Path::new(env!("CARGO_MANIFEST_DIR")); + let mut harness = vec![fs::read_to_string(root.join("tests/toyos.rs")) + .map_err(|e| format!("read tests/toyos.rs: {e}"))?]; + let common = root.join("tests/common"); + for entry in fs::read_dir(&common).map_err(|e| format!("read {}: {e}", common.display()))? { + let path = entry.map_err(|e| e.to_string())?.path(); + if path.extension().is_some_and(|e| e == "rs") { + harness.push(fs::read_to_string(&path).map_err(|e| e.to_string())?); + } + } + let shared: BTreeSet<&str> = registry.iter().copied().collect(); + let both: BTreeSet = driven_binaries(&harness) + .into_iter() + .filter(|name| shared.contains(name.as_str())) + .collect(); + let declared: BTreeSet = DRIVEN_AND_SHARED.iter().map(|s| s.to_string()).collect(); + let undeclared: Vec<&String> = both.difference(&declared).collect(); + if !undeclared.is_empty() { + return Err(format!( + "{undeclared:?} are driven by a machine test and also run on the shared boot, where \ + nothing stages what they need — so each passes on its exit code with no verdict. Add \ + each to RUST_SKIP with the reason its driver exists, or to DRIVEN_AND_SHARED if its \ + shared run asserts something of its own." + )); + } + let stale: Vec<&String> = declared.difference(&both).collect(); + if !stale.is_empty() { + return Err(format!( + "DRIVEN_AND_SHARED names {stale:?}, which no machine test drives or the shared boot \ + no longer runs. Delete each entry — a declaration nothing is true of is what makes \ + the rest of the list unreadable." + )); + } + + println!( + " [split] {} shared binaries on the shipping kernel, {} on the actuator one, {} of them \ + driven elsewhere and declared", + registry.len() - listed.len(), + listed.len(), + both.len() + ); + Ok(()) + } + + /// Every guest binary the harness drives by name, read out of its own sources. + /// + /// A driver reaches a binary as the literal `test_rs_`, so that is what + /// says a binary has one; a `format!` over a variable name yields no literal + /// and is not a driver of any particular binary. Pure, and the sources are a + /// parameter, so `suite_split` stages its own before trusting it on the tree. + fn driven_binaries(sources: &[String]) -> BTreeSet { + const MARK: &str = "test_rs_"; + let mut found = BTreeSet::new(); + for text in sources { + let mut rest = text.as_str(); + while let Some(at) = rest.find(MARK) { + rest = &rest[at + MARK.len()..]; + let end = rest + .find(|c: char| !c.is_ascii_alphanumeric() && c != '_') + .unwrap_or(rest.len()); + if end > 0 { + found.insert(rest[..end].to_string()); + } + } + } + found + } + + /// What a whole run exits with, and what its last line says. + #[test] + fn run_exit_status() -> Result<(), String> { + let outcome = |name: &str, reason: Option<&str>, suspended: Duration| Outcome { + name: name.to_string(), + reason: reason.map(str::to_string), + elapsed: Duration::from_secs(3), + suspended, + }; + let slept = common::clock::SUSPENDED_AT_LEAST + Duration::from_secs(120); + + let mut red = Tally::new(); + red.record(outcome("a_red", Some("the disk came back short"), Duration::ZERO)); + red.record(outcome("a_suspended_one", None, slept)); + if red.exit_code() != 1 { + return Err(format!("a run with a red exits {}, and it has to be 1", red.exit_code())); + } + + let mut suspended = Tally::new(); + suspended.record(outcome("a_suspended_one", None, slept)); + if suspended.exit_code() != 2 { + return Err(format!("a suspended run exits {}, and it has to be 2", suspended.exit_code())); + } + + // The clean case, so that none of the above is passing because everything + // reds. + let mut clean = Tally::new(); + clean.record(outcome("a_green", None, Duration::ZERO)); + let text = clean.summary(1, Duration::from_secs(9), Duration::ZERO); + if clean.exit_code() != 0 { + return Err(format!("a clean run exits {}, and it has to be 0", clean.exit_code())); + } + if !text.lines().last().unwrap_or_default().starts_with("test result: ok.") { + return Err(format!("a clean run does not say so plainly:\n{text}")); + } + Ok(()) + } + + /// Whether a run that did not attempt most of the suite's measured cost says so. + /// + /// **The failure mode the tier introduces is silence, not a wrong answer.** A + /// green run holding back 60 tests and a green run holding back none print the + /// same word, and the difference between them is the whole reason a reach flag + /// exists. Nothing else can see whether the *run* mentions it, and a run nobody + /// can tell apart from a full one is how a temporary measure becomes permanent. + /// + /// Both directions, because the second is the one that rots quietly: a suite + /// that ran everything must not claim to have held anything back either, or the + /// line stops carrying information the day somebody makes it unconditional. + #[test] + fn nightly_tier_is_announced() -> Result<(), String> { + let held = vec![ + (Tier::Nightly, vec!["desktop_window_child".to_string(), "sshd_exec".to_string()]), + (Tier::Weekly, vec!["iommu_empty_domain".to_string()]), + ]; + let announced = Tally::new().holding_back(held).summary(1, Duration::ZERO, Duration::ZERO); + for want in [ + "not run without --nightly:", + "desktop_window_child, sshd_exec", + "`cargo test --test toyos-build -- --nightly` runs them", + "not run without --weekly:", + "`cargo test --test toyos-build -- --weekly` runs them", + "2 held back for --nightly, 1 held back for --weekly", + ] { + if !announced.contains(want) { + return Err(format!("a run holding tests back never says {want:?}:\n{announced}")); + } + } + let whole = Tally::new().holding_back(vec![(Tier::Weekly, Vec::new())]).summary( + 1, + Duration::ZERO, + Duration::ZERO, + ); + if whole.contains("--") || whole.contains("held back") { + return Err(format!("a run that held nothing back says it did:\n{whole}")); + } + Ok(()) + } + + #[test] + fn screen_decoder() { + screen::self_test(); + } +} diff --git a/tests/common/console.rs b/tests/common/console.rs index 31fd198640d..159bea96488 100644 --- a/tests/common/console.rs +++ b/tests/common/console.rs @@ -1,31 +1,13 @@ -//! The console's line atomicity, and the one thing the harness may conclude -//! from it. +//! The one thing the harness may conclude from the console's line atomicity: +//! [`verdict`] — a line's *first bytes are its writer's own*, so the C family +//! can tell a daemon's line from the program under test's by reading it, and +//! stop failing on output that is not its own. //! -//! Two subjects, in this order because the second rests on the first: -//! -//! 1. [`console_line_atomicity`] — every line on the console is one writer's, -//! whole. -//! 2. [`verdict`] — therefore a line's *first bytes are its writer's own*, so -//! the C family can tell a daemon's line from the program under test's by -//! reading it, and stop failing on output that is not its own. -//! -//! The order is the argument, and it is the whole of what closed task #84. -//! Before L5 a daemon's `println!` and a test's could reach the backend in -//! pieces and arrive spliced into one line, and no rule over lines could -//! separate them — so the standing write-up said there was no cheap honest fix -//! and left a choice between giving each child a capture channel and tagging -//! every console write with its writer. Both are built now: a program's line -//! is a record in its own log ring, assembled by the process that wrote it, -//! and `logd` puts it on the console whole under the program's name — so a -//! line begins with its writer's first bytes or `console_line_atomicity` is -//! red. -//! -//! **L5's guarantee is about flushes, not about newlines, and that is the -//! second half.** A program that writes without a trailing newline has its -//! bytes joined to the next writer's line by the *host's* splitter — see -//! [`speaker_at`], which is where that is written down. Both halves are -//! [`c_capture_ignores_daemon_lines`]'s, and every one of its verdicts carries -//! the control that says it has teeth. +//! **L5's guarantee is about flushes, not about newlines.** A program that +//! writes without a trailing newline has its bytes joined to the next writer's +//! line by the *host's* splitter — see [`speaker_at`], which is where that is +//! written down. Both are [`c_capture_ignores_daemon_lines`]'s, and every one +//! of its verdicts carries the control that says it has teeth. use std::collections::BTreeSet; use std::path::{Path, PathBuf}; @@ -33,225 +15,12 @@ use std::sync::OnceLock; use std::time::Duration; use super::qemu::{BootOptions, QemuInstance}; -use super::serial::Serial; - -/// The guest binary's name in the `run ` protocol. -const WRITER: &str = "test_rs_console_line_atomicity"; -/// A liveness guard and never the verdict: two thousand 200-byte lines is a -/// fraction of a second of virtio-console, and this only catches a guest that +/// A liveness guard and never the verdict: it only catches a guest that /// stopped answering. const CEILING: Duration = Duration::from_secs(60); -/// What the guest's binary declares, so the host is not carrying a second copy -/// of the numbers. -struct Declared { - writers: usize, - lines: usize, - width: usize, - /// Bytes the third writer said in two `write`s and never ended with a - /// newline, which only its own exit can put on the wire. - midline: usize, - /// Digits of the sequence number after each line's leading tag byte — - /// what tells a gap in a writer's own run from a capture that ends early. - seq: usize, -} - -pub fn console_line_atomicity( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - // **Two CPUs, because one writer preempting another is the stimulus.** At - // `--smp 1` the two processes still interleave — they are preempted, not - // parallel — but at two the gap between one writer's two `write`s can be - // filled by a genuinely concurrent one, which is the harder case and the - // one a laptop has. - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { smp: 2, ..Default::default() }, - ); - let result = qemu.run_test(WRITER, CEILING); - if let Some(err) = &result.error { - return Err(format!("{err}\nstdout:\n{}", tail(&result.stdout))); - } - if result.exit_code != Some(0) { - return Err(format!( - "the writers exited {:?}\n{}", - result.exit_code, - tail(&result.stdout) - )); - } - let declared = declared(&result.stdout)?; - // **The writers' lines and the runner's `===TEST_START` reach the console - // through `logd` from two rings' reads**, so a writer's first lines may - // arrive before the marker opens the window: every line since the command - // was typed is read. - let capture = format!("{}{}", result.before, result.stdout); - - let mut pure: [BTreeSet; 2] = [BTreeSet::new(), BTreeSet::new()]; - let mut duplicated: usize = 0; - let mut mixed: Vec<&str> = Vec::new(); - let mut short: usize = 0; - for line in capture.lines() { - let a = line.bytes().filter(|b| *b == b'A').count(); - let b = line.bytes().filter(|b| *b == b'B').count(); - // A writer's line is one tag byte, its sequence digits, and the tag - // repeated to the width — so a line that is mostly one tag is one of - // its lines however it ended up. - let (tag, other) = if a >= b { (a, b) } else { (b, a) }; - if tag * 2 < declared.width - 1 { - continue; // not a writer's line at all - } - if other > 0 { - mixed.push(line); - continue; - } - let writer = usize::from(b > a); - let bytes = line.as_bytes(); - let tag_byte = b"AB"[writer]; - let whole = bytes.len() == declared.width - 1 - && bytes[0] == tag_byte - && bytes[1..1 + declared.seq].iter().all(u8::is_ascii_digit) - && bytes[1 + declared.seq..].iter().all(|c| *c == tag_byte); - let seq = whole - .then(|| std::str::from_utf8(&bytes[1..1 + declared.seq]).expect("ascii digits")) - .and_then(|digits| digits.parse::().ok()) - .filter(|seq| *seq < declared.lines); - let Some(seq) = seq else { - short += 1; - continue; - }; - if !pure[writer].insert(seq) { - duplicated += 1; - } - } - - if !mixed.is_empty() { - let sample: Vec = mixed - .iter() - .take(3) - .map(|l| l.chars().take(80).collect::()) - .collect(); - return Err(format!( - "{} of {} console lines carry both writers' bytes — a `write` syscall is still the \ - unit of interleaving, so half a line reaches the backend and another process's \ - half follows it. First three, truncated to 80 columns:\n{}", - mixed.len(), - declared.writers * declared.lines, - sample.join("\n") - )); - } - if short != 0 { - return Err(format!( - "{short} console lines are a writer's bytes at the wrong width; a whole line is one \ - unit and these were cut" - )); - } - if duplicated != 0 { - return Err(format!( - "{duplicated} console lines repeat a sequence number a writer used once — the \ - capture duplicated lines, which no writer and no buffer can do" - )); - } - // Non-vacuity: a capture that lost the writers' output entirely would count - // zero mixed lines and prove nothing. The sequence numbers say what *kind* - // of loss it was, so a red here is not misread as the line buffer breaking: - // a gap inside a writer's own run is lines lost mid-stream, a contiguous - // run that stops early is a capture missing its tail, and neither is a - // mixed or short line — the mechanism's own verdicts are above. - for (i, tag) in ["A", "B"].iter().enumerate() { - let seen = &pure[i]; - if seen.len() == declared.lines { - continue; - } - let top = seen.iter().next_back().map_or(0, |s| s + 1); - let gaps = top - seen.len(); - if gaps > 0 { - let first_gap = (0..top).find(|s| !seen.contains(s)).unwrap_or(0); - return Err(format!( - "writer {tag} declared {} whole lines and the capture carries {}: {gaps} gap(s) \ - inside the writer's own numbered run (first at #{first_gap}, run ends at \ - #{}) — lines were lost mid-stream, between the guest's console and this \ - capture, not by the line buffer", - declared.lines, - seen.len(), - top - 1, - )); - } - return Err(format!( - "writer {tag} declared {} whole lines and the capture carries {}: the numbered run \ - is contiguous and simply stops at #{} — a short capture missing its tail, not a \ - line lost by the buffer", - declared.lines, - seen.len(), - top.saturating_sub(1), - )); - } - // **The line's other half: a process that exits mid-line.** The third - // writer says `midline` bytes in two `write`s, ends them with nothing and - // exits; the only thing that can make them a line is its exit ending what - // its stream held. A tree without that loses them silently, which drops a - // dying process's last words — so the assertion is the run's *length*, - // and it is exact on both sides: shorter means bytes were lost, longer - // means something else was acquired inside them. - let longest = capture - .split(|c| c != 'C') - .map(str::len) - .max() - .unwrap_or(0); - if longest != declared.midline { - return Err(format!( - "a process exited having written {} unterminated bytes and the longest run of them on \ - the console is {longest} — the process's exit is what turns a partial line into all \ - there will ever be, and this capture says it went nowhere", - declared.midline - )); - } - - // The kernel-into-userland half, on the same capture. A kernel record can - // only land inside a userland line if the line reached the backend in - // pieces, so this reds on exactly the coupling the count above reds on and - // observes it from the other side. - let console = Serial::named("console", &capture); - if let Some(spliced) = console.interleaved() { - return Err(format!( - "a kernel record landed inside a userland line: {:?}", - spliced.chars().take(160).collect::() - )); - } - eprintln!( - " [console] {} writers x {} lines of {} bytes, 0 mixed; {} unterminated bytes flushed by \ - an exit", - declared.writers, declared.lines, declared.width, declared.midline - ); - Ok(()) -} - -fn declared(stdout: &str) -> Result { - let line = stdout - .lines() - .find(|l| l.contains("console-atomicity: writers=")) - .ok_or_else(|| format!("the guest never declared its run\n{}", tail(stdout)))?; - let field = |key: &str| -> Result { - line.split_whitespace() - .find_map(|w| w.strip_prefix(key)) - .and_then(|v| v.parse::().ok()) - .ok_or_else(|| format!("the guest's declaration has no `{key}`: {line:?}")) - }; - Ok(Declared { - writers: field("writers=")?, - lines: field("lines=")?, - width: field("width=")?, - midline: field("midline=")?, - seq: field("seq=")?, - }) -} - -/// The last of a capture, for a failure message. Two thousand 200-byte lines is -/// not something to put in an assertion message whole. +/// The last of a capture, for a failure message. fn tail(text: &str) -> String { let lines: Vec<&str> = text.lines().collect(); lines[lines.len().saturating_sub(20)..] @@ -704,8 +473,8 @@ pub fn c_capture_ignores_daemon_lines( /// /// **`Profile::Metal` because the keystroke has to arrive.** Its i8042 is the /// only keyboard on the machine — no USB HID, no virtio — which is the shape -/// `i8042_keyboard` and `swiss_german_layout` already inject through, and the -/// mouse the middle arm claims is the PS/2 one beside it. +/// `swiss_german_layout` already injects through, and the mouse the middle arm +/// claims is the PS/2 one beside it. /// /// **One CPU, because the keystroke outlives the probe.** Nothing holds the /// keyboard once the claim is released, so the key stays queued while the diff --git a/tests/common/faults.rs b/tests/common/faults.rs index a00c2e86b56..5b331d9c82e 100644 --- a/tests/common/faults.rs +++ b/tests/common/faults.rs @@ -579,12 +579,7 @@ fn hex(line: &str, field: &str) -> Option { /// that carries on. /// /// **One boot, and the two negative controls are `syscall_window_nmi_controls`'s -/// two.** All three used to be one name and it priced at 19,740 ms on the hosted -/// lane against a 10,000 ms ceiling — three Metal boots of 3,000 NMIs each. What -/// belongs per pull request is the property: the window is reachable, arrivals -/// land in it, and the machine survives them. What the controls establish is that -/// the property is not vacuous, which is a claim about the *instrument* and moves -/// to the nightly tier. +/// two.** /// /// **The window.** `SYSCALL` switches no stack, so `arch::syscall`'s entry runs /// three instructions at CPL 0 with the user's `rsp` and its exit one more diff --git a/tests/common/iommu.rs b/tests/common/iommu.rs index 7a50b1ce2bc..a463d7e3155 100644 --- a/tests/common/iommu.rs +++ b/tests/common/iommu.rs @@ -259,11 +259,6 @@ fn table_word(socket: &Path, base: u64, index: u16) -> Result<(u64, u64), String /// /// [`Profile::Headless`] carries the most sources of both kinds — the i8042's /// two pins, and xHCI, virtio-net and virtio-sound over MSI-X. -/// -/// **Per pull request this is the one machine that runs.** The machines that -/// make the verdict move — `intremap=off`, and `eim=on` for the other entry -/// format — are [`iommu_discovery`]'s, and that is nightly, so a change that -/// broke only the not-remapping arm would land and be caught the next night. pub fn iommu_interrupt_remapping( test_config: &Path, c_bins: &[(String, Vec)], diff --git a/tests/common/lan.rs b/tests/common/lan.rs index 8ce0a3be67b..6e9968a0bbb 100644 --- a/tests/common/lan.rs +++ b/tests/common/lan.rs @@ -616,8 +616,7 @@ pub fn lan_talk( let (start, len) = super::volumes::log_extent(&bytes, &image)?; // **Its own disk.** sshd mints its identity under `/home`, and the lane's - // shared image would hand that identity to the next boot of the lane, - // which `sshd_fail_closed` asserts it mints itself. + // shared image would hand that identity to the next boot of the lane. let data = super::lane::dir().join("lan-talk-data.img"); toyos_build::build::create_sparse(&data, qemu::NVME_SMALL); let (ssh_port, log_port) = (qemu::free_host_port(), qemu::free_host_port()); diff --git a/tests/common/logread.rs b/tests/common/logread.rs index 731f349b003..e438887b250 100644 --- a/tests/common/logread.rs +++ b/tests/common/logread.rs @@ -58,15 +58,6 @@ struct Contaminated { } /// The conservation law, at one width. -/// -/// **Three registered names and not one, and the reason is the fast tier's -/// line.** What the law is about is concurrent producers, so a machine with one -/// CPU and a machine with eight are different subjects rather than one subject -/// measured three times: `--smp 1` is where the reader and the one producer -/// share a CPU, `--smp 4` and `--smp 8` are where they do not. One name over -/// all three boots measured 17,112 ms in CI — over the fast tier's 10 s line, and the -/// gate the whole design turns on may not sit in the nightly tier — while each -/// boot on its own is comfortably under it. fn conservation( test_config: &Path, c_bins: &[(String, Vec)], @@ -112,22 +103,6 @@ pub fn log_conservation_smp1( conservation(test_config, c_bins, rust_bins, 1) } -pub fn log_conservation_smp4( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - conservation(test_config, c_bins, rust_bins, 4) -} - -pub fn log_conservation_smp8( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - conservation(test_config, c_bins, rust_bins, 8) -} - /// The nested-`emit` gate: an interrupt that logs, inside another `emit`, on one CPU. /// /// **The one case loom cannot express and the host cannot stage.** The diff --git a/tests/common/power.rs b/tests/common/power.rs index 9b481909542..3f46d456272 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -334,36 +334,18 @@ pub fn quiesce_wakes_on_the_last_park( _c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - woken_by_the_held_thread(&["quiesce-last-park", LATE_WORD], rust_bins) -} - -/// **An exit that is the stop's last transition wakes it.** The same, with -/// `quiesce-last-exit` holding the thread inside `SYS_THREAD_EXIT`. -pub fn quiesce_wakes_on_the_last_exit( - _test_config: &Path, - _c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - woken_by_the_held_thread(&["quiesce-last-exit", LATE_WORD], rust_bins) -} - -/// One of the two `quiesce-last-*` boots: its actuator first, the late word beside it. -fn woken_by_the_held_thread( - armed: &'static [&'static str; 2], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let actuator = armed[0]; + const ACTUATOR: &str = "quiesce-last-park"; let (whole, record) = stopped_boot( "tests/quiescelastcase/system.toml", "quiesce_last", - armed, + &[ACTUATOR, LATE_WORD], rust_bins, )?; // **The premise, by the kernel's own word**: the thread was held, and // held before the stop claimed anything. Without it the boot below is // one whose last transition was anything at all. let held = format!( - "{actuator}: {} is held until the stop waits on it alone", + "{ACTUATOR}: {} is held until the stop waits on it alone", toyos_quiesce::LAST_THREAD, ); let at = |needle: &str| whole.lines().position(|line| line.contains(needle)); @@ -382,83 +364,7 @@ fn woken_by_the_held_thread( )); } woken_by_its_threads(&record)?; - eprintln!(" [power] {actuator}: the held thread's transition woke the stop: {record}"); - Ok(()) -} - -/// **A dump served during the stop says every thread it stopped is held.** -/// `quiesce-dump` serves Ctrl+Alt+D's report from the shutdown once its first -/// stage has stopped the writers, the moment the owner presses it on a -/// shutdown stuck there. A stopped thread's state word reads `Ready`, and a -/// report that counted it nowhere else would call it claimed and not held. -pub fn quiesce_dump_holds_the_stopped( - _test_config: &Path, - _c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let (whole, record) = stopped_boot( - "tests/quiescecase/system.toml", - "quiesce_writers", - &["quiesce-dump", LATE_WORD], - rust_bins, - )?; - let lines: Vec<&str> = whole.lines().collect(); - let at = |needle: &str| lines.iter().position(|line| line.contains(needle)); - let (Some(began), Some(ended), Some(synced)) = ( - at("=== blocked-task dump:"), - at("=== end of dump ==="), - at("Syncing filesystems..."), - ) else { - return Err(format!("no whole dump and sync in this boot\n{whole}")); - }; - if !(began < ended && ended < synced) { - return Err(format!( - "the dump (lines {began} to {ended}) did not finish inside the stop, which ends at \ - line {synced}\n{whole}" - )); - } - let report = &lines[began..=ended]; - let number_before = |marker: &str, word: &str| -> Result { - let line = report - .iter() - .find(|line| line.contains(marker)) - .ok_or_else(|| format!("no {marker:?} line in the report:\n{}", report.join("\n")))?; - let (head, _) = - line.split_once(word).ok_or_else(|| format!("no {word:?} on {line:?}"))?; - head.split_whitespace() - .next_back() - .and_then(|n| n.parse().ok()) - .ok_or_else(|| format!("no number before {word:?} on {line:?}")) - }; - // The harm first: a report that has no word for a stopped thread still - // writes this verdict, and calls each one it cannot place unheld. - let unheld = number_before("== VERDICT:", " unheld,")?; - if unheld != 0 { - return Err(format!( - "a dump served during the stop called {unheld} thread(s) claimed and not held:\n{}", - report.join("\n"), - )); - } - // Then the premise: the report was served over a machine with banded - // threads, or the zero above is about nothing. - let stopped = number_before("== sched:", " stopped,")?; - if stopped == 0 { - return Err(format!( - "the report found no stopped thread on any cpu, so it says nothing about how it \ - counts one:\n{}", - report.join("\n"), - )); - } - // And the count is the threads it names: this stop bands fewer than one - // cpu's line cap, so each one it counts is a line of its own. - let named = report.iter().filter(|line| line.contains("stopped (the machine is stopping)")).count(); - if named != stopped as usize { - return Err(format!( - "the report counted {stopped} stopped thread(s) and named {named}:\n{}", - report.join("\n"), - )); - } - eprintln!(" [power] the dump inside the stop held all {stopped} stopped thread(s): {record}"); + eprintln!(" [power] {ACTUATOR}: the held thread's transition woke the stop: {record}"); Ok(()) } diff --git a/tests/common/storage.rs b/tests/common/storage.rs index fe93ca226d9..0ac5fa85ea0 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -446,209 +446,6 @@ fn front(path: &Path, at: u64, n: usize) -> Vec { head } -/// The shared-object cache's two refusals, judged in -/// `tests/toyos-rust-tests/src/bin/so_cache_policy.rs`. The independent oracle is -/// the NVMe image: the replaced library's bytes are read off the device after -/// the shutdown, so the claim rests on nothing the guest says. -pub fn so_cache_refusals( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - /// Without it the budget arm would have to load 256 MiB of libraries. - const PARAMS: &[&str] = &["so-cache-tiny"]; - /// Mirrored in the guest binary; `/home` is a directory of DATA, so that is - /// the name the host reader sees on the volume. - const STALE: &str = "home/so-cache-stale.so"; - const SECOND: &str = "libtls_dlopen_lib.so"; - - let want = rust_bins - .iter() - .find(|(name, _)| name == SECOND) - .map(|(_, data)| data.clone()) - .ok_or_else(|| format!("{SECOND} was not built, so there is nothing to compare against"))?; - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::MetalDisk, - kernel_params: PARAMS, - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - if boot.contains("are a tmpfs") { - return Err(format!( - "/apps and /home fell back to tmpfs, so the readback below would judge no device:\n{boot}" - )); - } - - let result = qemu.run_test("test_rs_so_cache_policy", Duration::from_secs(60)); - let log = format!("{boot}\n{}{}{}", result.before, result.stdout, result.serial); - if result.exit_code != Some(0) { - return Err(format!( - "so_cache_policy guest failed:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - // Stated by the kernel too: an arm reporting a refusal nobody made would pass. - for said in ["the cached image is stale", "byte budget; refused"] { - if !log.contains(said) { - return Err(format!("no {said:?} line — the kernel refused nothing:\n{log}")); - } - } - - let image = qemu.nvme_image().to_path_buf(); - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - - let io = FileBlocks::open(&image)?; - let fs = bcachefs::Mounted::<_, bcachefs::ReadOnly>::open(io) - .map_err(|e| format!("the NVMe image does not mount on the host: {e:?}"))?; - let got = fs - .read_file(STALE) - .map_err(|e| format!("reading {STALE} off the image: {e:?}"))?; - if got != want { - let at = got.iter().zip(&want).position(|(a, b)| a != b); - return Err(format!( - "{STALE} on the device is {} bytes against {SECOND}'s {}, first differing at {at:?} \ - — the guest's second write did not reach the device, so the refusal above was \ - about a file that had not changed", - got.len(), - want.len() - )); - } - - eprintln!( - " [so-cache] {} bytes of {SECOND} byte-identical at {STALE} off the NVMe image via the \ - host's own bcachefs reader", - want.len() - ); - Ok(()) -} - -/// F9's negative control: an fsync on `/home` whose first attempt is -/// budget-refused (`fsync-budget-spent`) must be retried to durable — on the -/// erasing adapter the `BudgetExpired` came back as `Io` and the guest's -/// `sync_all` failed on attempt 1. The independent oracle is the NVMe image -/// itself: after the shutdown the file's bytes are read off it on the host, -/// through this crate's own build of the `bcachefs` reader over a plain -/// seek-and-read device — nothing the guest kernel executed. -pub fn home_budget_refusal_retried( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - const PARAMS: &[&str] = &["fsync-budget-spent"]; - /// Mirrored in `tests/toyos-rust-tests/src/bin/home_fsync_budget.rs`, under - /// the `/home` directory of DATA the host reader sees. - const PATH: &str = "home/f9-budget.bin"; - const LEN: usize = 3 * 4096 + 41; - fn pattern() -> Vec { - (0..LEN).map(|i| (i.wrapping_mul(151) ^ 0x3C) as u8).collect() - } - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::MetalDisk, - kernel_params: PARAMS, - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - if boot.contains("are a tmpfs") { - return Err(format!( - "/apps and /home fell back to tmpfs, so nothing below touches the NVMe path:\n{boot}" - )); - } - - let result = qemu.run_test("test_rs_home_fsync_budget", Duration::from_secs(30)); - let log = format!("{boot}\n{}{}{}", result.before, result.stdout, result.serial); - if result.exit_code != Some(0) { - return Err(format!( - "home_fsync_budget guest failed — a budget-refused /home fsync was not retried \ - to durable:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - // Both halves of the staging, or the arm proved nothing: the refusal at the - // shipped NVMe site, and the fsync loop's own retry verdict. - if !log.contains("not issued") { - return Err(format!( - "no `not issued` line, so `fsync-budget-spent` staged no NVMe refusal:\n{log}" - )); - } - let retried = log - .lines() - .find(|l| l.contains("fsync: /home/") && l.contains("durable on attempt")) - .ok_or_else(|| { - format!("no `fsync: /home/... durable on attempt` line — the retry never ran:\n{log}") - })? - .trim() - .to_string(); - // And the refusal was one per file, not one per flush: `logd` flushes every - // round it wrote a line in, and a refused flush is lines of its own, so a - // refusal on every flush keeps `/log` retrying for as long as the machine - // runs and rotates the boot's own log away. An absence has no event to wait - // on, so it is judged over a fixed window: the storm retries there without - // pause, the once-per-file refusal never. - let after = qemu.drain_serial(Duration::from_secs(2)); - let again: Vec<&str> = - after.lines().filter(|l| l.contains("fsync: ") && l.contains("durable on attempt")).collect(); - if !again.is_empty() { - return Err(format!( - "{} flush(es) retried in the 2 s after the guest's, first {:?}: every flush is \ - being refused, and each refusal's records are the next flush\n{after}", - again.len(), - again[0] - )); - } - - let image = qemu.nvme_image().to_path_buf(); - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - - let io = FileBlocks::open(&image)?; - let fs = bcachefs::Mounted::<_, bcachefs::ReadOnly>::open(io) - .map_err(|e| format!("the NVMe image does not mount on the host: {e:?}"))?; - let got = fs - .read_file(PATH) - .map_err(|e| format!("reading {PATH} off the image: {e:?}"))?; - if got != pattern() { - let at = got.iter().zip(pattern()).position(|(a, b)| *a != b); - return Err(format!( - "{PATH} on the device is {} bytes, first differing at {at:?} — the retried fsync \ - reported durable over bytes the device does not hold", - got.len() - )); - } - - eprintln!(" [f9] {retried}"); - eprintln!( - " [f9] {LEN} bytes byte-identical off the NVMe image via the host's own bcachefs reader" - ); - Ok(()) -} - /// A same-length overwrite on `/home` read back through the name it rebound. /// /// The oracle is outside the guest and outside the kernel: with the machine diff --git a/tests/common/usb.rs b/tests/common/usb.rs index 66ab8568508..60e3e22a1c9 100644 --- a/tests/common/usb.rs +++ b/tests/common/usb.rs @@ -506,141 +506,6 @@ pub fn usb_short_read( Ok(()) } -/// A disk plugged into a **different controller** must not renumber the one a -/// mount is holding. -/// -/// The machine-wide disk index was `storage.len()` summed across controllers, -/// and that vector grows on every bind — hot-plug included. The T14 has two -/// xHCIs, the Thunderbolt block's at 00:0d.0 ahead of the PCH's at 00:14.0, and -/// it boots off a stick in a PCH port: with nothing on the first controller the -/// boot stick is disk 0, and plugging any USB storage into the USB-C side made -/// the *new* drive disk 0 and the boot stick disk 1. `FatDevice` holds its -/// `UsbBlockDevice` for the life of the mount and that handle is an index, so -/// every later `/log` append went into the middle of the new drive and every -/// `/boot` read served its bytes as the ESP's. -/// -/// **The actuator is QEMU's own `device_add` and nothing about the driver is -/// modified.** Both verdicts are host-side and neither is a log line: the disk -/// that arrives late is a file the harness staged as zeros and must find as -/// zeros, and `/log/kernel.log` is read out of the boot image's own partition -/// and must carry a line the guest printed *after* the plug. -pub fn usb_disk_index_stable( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - /// The disk that arrives late: 48 GiB, sparse, and no size any other - /// device in this suite reports. - const LATE_BYTES: u64 = 48 * 1024 * 1024 * 1024; - /// The line the guest prints once the late disk is up, which is therefore a - /// line `/log` can only carry if it was still reaching the boot stick after - /// the plug. - const LATE_READY: &str = "usb-storage: disk 1 ready on slot"; - - // The boot stick is on the second controller and the first carries - // nothing, which is the laptop exactly — and the arrangement in which the - // boot stick is disk 0 with a free controller ahead of it. - let options = BootOptions { - profile: Profile::MetalXhciSecond, - qmp: true, - ..Default::default() - }; - let argv = qemu::profile_argv(&options); - if !argv.iter().any(|a| a.contains("usb-storage,bus=xhci1.0")) { - return Err(format!("the boot stick is not on the second controller: {argv:?}")); - } - if argv.iter().any(|a| a.starts_with("usb-storage,bus=xhci.0")) { - return Err(format!("the first controller already carries storage: {argv:?}")); - } - - // Built here rather than by the boot, because `/log` has to be read off the - // partition afterwards and the image gets a fresh GUID every time it is - // built. - let image_path = test_dir().join("usb-index-stable.img"); - let image = qemu::build_boot_image(test_config, c_bins, rust_bins, &[]); - std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - let (start, len) = super::volumes::log_extent(&image, &image_path)?; - - let late = test_dir().join("usb-index-late.img"); - drop(sparse(&late, LATE_BYTES)); - let before = fingerprint(&late, LATE_BYTES); - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - boot_image: Some(qemu::Staged::Written(image_path.clone())), - ..options - }, - ); - let boot = qemu.boot_log().to_string(); - if !boot.contains("usb-storage: disk 0 ready on slot") { - return Err(format!("the boot stick did not come up as disk 0\n{boot}")); - } - - let mut devices = qemu::QmpDevices::open(qemu.qmp_socket()); - devices.blockdev_add("latedisk", &late); - devices.add("usb-storage", "xhci.0", "latedisk0", &[("drive", "latedisk")]); - drop(devices); - // The driver's debounce is 100 ms and the enumeration behind it is - // microseconds under TCG; this is that with room. - thread::sleep(Duration::from_millis(1200)); - - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let log = format!("{boot}{}", qemu.drain_serial(Duration::from_secs(20))); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if log.contains(bad) { - return Err(format!("{bad:?} after a disk arrived on the other controller\n{log}")); - } - } - - // The plug happened at all. Without this both host-side claims below hold - // trivially on a boot where nothing was added. - if !log.contains(LATE_READY) { - return Err(format!( - "nothing enumerated on the first controller; there is no renumbering to survive\n{log}" - )); - } - - // **The disk that arrived is not the disk anything was mounted on.** The - // harness made this file and the guest was never told it was writable, so - // a single changed byte is `/boot` or `/log` writing through a handle that - // now names the wrong device. - if fingerprint(&late, LATE_BYTES) != before { - return Err( - "the guest wrote to the disk plugged into the other controller — the index a mount \ - was holding moved onto it" - .to_string(), - ); - } - - // **And the log kept reaching the stick.** Read off the boot image's own - // `/log` partition, so this is the device's view and not the guest's. The - // sink names one file per boot, so the newest on the volume is this one's. - let (name, on_device) = super::volumes::newest_log(&image_path, start, len)?; - let on_device = String::from_utf8_lossy(&on_device).into_owned(); - if !on_device.contains(LATE_READY) { - return Err(format!( - "/log/{name} stops at {} bytes and never carries {LATE_READY:?} — the appends after \ - the plug went somewhere else\n{log}", - on_device.len() - )); - } - let _ = std::fs::remove_file(&late); - let _ = std::fs::remove_file(&image_path); - - eprintln!( - " [usb] a {LATE_BYTES} B disk plugged into the empty first controller: it comes back \ - byte-identical, and {} bytes of /log/{name} on the boot stick carry the lines printed \ - after it", - on_device.len() - ); - Ok(()) -} - /// More disks on one controller than its DMA pool has blocks for. /// /// `MSC_BLOCKS` is 2 and the boot stick takes one, so the second data disk on @@ -3347,531 +3212,6 @@ pub fn xhci_full_speed_device( Ok(()) } -/// A HID interrupt endpoint whose transfer completes with a code the driver did -/// not expect. -/// -/// `dispatch_event` requeued a bound device's interrupt TRB only for Success and -/// Short Packet. **Every other code was dropped where it was read** — no log -/// line, no requeue, no fault — and that endpoint carries exactly one TRB, so -/// the device went silent for the rest of the boot with every bind-time line -/// reading perfectly. A Logitech mouse hot-plugged into the T14 did exactly -/// that: `HID mouse ready on slot 6 … merges as source 1` at 30.485 s and not -/// one motion event until it was unplugged at 58.659 s. -/// -/// Both timings, because they are different states of the driver and neither is -/// a weaker version of the other: the **fourth** completion is a device that has -/// been delivering and stops, and the **first** is a freshly configured endpoint -/// that never delivered at all — which is the shape the T14 showed and the one -/// whose recovery has to work before any report has ever arrived. -/// -/// The actuator is a boot parameter and `xhci/hid.rs`'s `stage_break` says why -/// nothing on the host side can reach it. What it replaces is the completion -/// code **and the report that transfer delivered**: QEMU really moved a mouse -/// report into the buffer, so a driver that dispatched it despite the error -/// would publish a delta it never earned and this gate would pass against the -/// defect it names. Everything the recovery reads is the controller's own — the -/// Endpoint State out of the output device context, and three commands the -/// controller really answers. -/// -/// Ground truth is host-side: the keys and the pointer delta injected **after** -/// the staged failure arrive in the guest's own event stream, on a machine -/// (`i8042=off`, boot-time HID absolute-only) where no other device can produce -/// either. -pub fn xhci_hid_break( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - for (features, which) in [ - ( - &["xhci-hid-break-first"][..], - "the very first completion, before the device ever delivered", - ), - ( - &["xhci-hid-break-late"][..], - "the fourth completion, after the device had been delivering", - ), - ] { - hid_break_boot(test_config, c_bins, rust_bins, features, which)?; - } - Ok(()) -} - -/// One boot of [`xhci_hid_break`], with the break staged at whichever -/// completion `feature` names. -fn hid_break_boot( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], - params: &'static [&'static str], - which: &str, -) -> Result<(), String> { - /// The delta the assertion is about, injected after the break is spent. - /// Neither component is a number any other move in this boot produces. - const DX: i32 = 40; - const DY: i32 = -30; - /// Typed after the break is spent, and nowhere else in the boot. - const WORD: [&str; 5] = ["h", "e", "l", "l", "o"]; - /// Both QEMU HID devices carry one IN interrupt endpoint at address 1, so - /// this is the device's own number for it and the controller's, and a line - /// naming either wrongly stops matching. - const ENDPOINT: &str = "interrupt endpoint 0x81 (dci 3)"; - - let options = BootOptions { - profile: Profile::MetalHotplug, - qmp: true, - // The only keyboard on the machine has to be the one plugged in, or - // QEMU delivers the keystrokes over PS/2 and every assertion below - // passes with the interrupt endpoint dead. - i8042: false, - kernel_params: params, - ..Default::default() - }; - let argv = qemu::profile_argv(&options); - let usb = crate::usb_argv(&argv); - for absent in ["usb-kbd", "usb-mouse"] { - if usb.iter().any(|d| d.starts_with(absent)) { - return Err(format!("{absent} is on the bus at boot; argv has {usb:?}")); - } - } - if !argv.iter().any(|a| a.contains("i8042=off")) { - return Err("the i8042 is on; a PS/2 keyboard could deliver instead".to_string()); - } - - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - let boot = qemu.boot_log().to_string(); - let Some((scale_x, scale_y)) = crate::parse_rel_scale(&boot) else { - return Err(format!("the kernel never said what pointer scale it used:\n{boot}")); - }; - - let result = qemu.run_test_hooked( - "test_rs_input_events", - Duration::from_secs(60), - "===INPUT_READY===", - move |socket| { - let mut devices = qemu::QmpDevices::open(socket); - devices.add("usb-mouse", "xhci1.0", "hidmouse", &[]); - devices.add("usb-kbd", "xhci1.0", "hidkbd", &[]); - drop(devices); - thread::sleep(Duration::from_millis(800)); - - // Spend the break. Ten pointer completions and six keyboard ones, - // against an injection that strikes the first or the fourth: the - // margin is for QEMU coalescing rel events it has not been polled - // for, which can only make the count smaller. - let mut input = qemu::QmpInput::open(socket); - for _ in 0..10 { - input.mouse(4, 4, None); - thread::sleep(Duration::from_millis(60)); - } - for key in ["a", "b", "c"] { - input.keys(&[(key, true), (key, false)]); - thread::sleep(Duration::from_millis(60)); - } - drop(input); - thread::sleep(Duration::from_millis(300)); - - // And the measured phase, every event of which is after the break. - let mut input = qemu::QmpInput::open(socket); - // The accumulated position clamps at 0, so a move up or left from - // the origin is invisible — and with the first completion eaten the - // pointer may still be sitting there. - input.mouse(200, 200, None); - thread::sleep(Duration::from_millis(150)); - input.mouse(DX, DY, None); - thread::sleep(Duration::from_millis(150)); - for key in WORD { - input.keys(&[(key, true), (key, false)]); - thread::sleep(Duration::from_millis(30)); - } - crate::input_events_end(&mut input); - drop(input); - thread::sleep(Duration::from_millis(200)); - }, - ); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}\n{}", result.serial, result.stdout)); - } - let log = format!("{boot}{}", result.serial); - for bad in ["PANIC:", "panicked at"] { - if log.contains(bad) { - return Err(format!("{bad:?} with the break staged at {which}\n{log}")); - } - } - - // **Delivery first**, because it is what the gate is about and what the - // pre-fix driver cannot do: an endpoint whose completion was dropped holds - // no TRB and nothing ever puts one back, so everything injected after the - // break stays on the host side of the wire. - let word: String = WORD.concat(); - hotplug_delivered(&result.stdout, &word, (DX * scale_x, DY * scale_y)).map_err(|e| { - format!("with the break staged at {which}, input never came back: {e}\n{log}") - })?; - - // Then that a break was staged at all, and that the line names the device, - // the endpoint and the code. Without this the boot above is one where - // nothing failed — and the line itself is the instrument: the T14's log - // cannot name the code its mouse died of, because the driver discarded it. - let named: Vec<&str> = log.lines().filter(|l| l.contains(ENDPOINT)).collect(); - let want = format!("{ENDPOINT} completed with code 6 (Stall Error); failure 1 of 8"); - let staged: Vec<&&str> = named.iter().filter(|l| l.contains(want.as_str())).collect(); - if staged.len() != 2 { - return Err(format!( - "{} endpoint(s) reported a broken completion, want the mouse and the \ - keyboard: {named:?}\n{log}", - staged.len() - )); - } - // Which two devices those were, by the controller they are on and the slot - // they hold on it — and both halves are needed. **A slot id is one - // controller's numbering and this machine has two**: the boot disk is slot 1 - // on `00:02.0` and the mouse plugged in below is slot 1 on `00:03.0`. - let mut broken: Vec<(&str, &str)> = Vec::new(); - for line in &staged { - broken.push(hid_broke_on(line)?); - } - for kind in ["pointer", "keyboard"] { - if !broken.iter().any(|(k, _)| *k == kind) { - return Err(format!("no {kind:?} among the broken completions {broken:?}\n{log}")); - } - } - if broken[0].1 == broken[1].1 { - return Err(format!( - "both broken completions are on {}, so this boot broke one device twice rather \ - than the mouse and the keyboard once each\n{log}", - broken[0].1 - )); - } - - // The endpoint state the recovery had to be chosen for, read out of the - // controller's own output device context. The transfer really completed, so - // the endpoint really is Running — `Halted` here would mean the injection - // staged a shape this boot cannot produce and everything above proves - // something else. - // - // **Once for each of the two devices the injection struck, and no longer a - // count over the whole boot.** `endpoint 3` is the first IN endpoint of - // *every* USB device — the boot disk's bulk IN as much as a HID interrupt - // endpoint — so one transport recovery on the boot disk anywhere in the boot - // used to red this test with a failure about HID: three CI runs did exactly - // that (`31405969578` shard 10, `31424496450`, `31601325987`), and in the - // first the disk's own `slot 1 endpoint 3` and `slot 1 endpoint 4` at - // 2.639 s — a `SCSI 0x35` status-phase break on a shard measured at 2.16x - // boot width — were counted beside the mouse's and the keyboard's. - let mut recovered: Vec<(&str, usize)> = Vec::new(); - for (_, who) in &broken { - let running = format!("xHCI: {who} endpoint 3 is Running, recovering"); - recovered.push((who, log.matches(running.as_str()).count())); - } - if recovered.iter().any(|(_, n)| *n != 1) { - let states: Vec<&str> = log.lines().filter(|l| l.contains(", recovering")).collect(); - return Err(format!( - "the two devices the injection struck were found Running {recovered:?} time(s), \ - want once each; every recovery this boot: {states:?}\n{log}" - )); - } - - // `run_command` logs only refusals, so each of these is the controller - // declining a command the endpoint's state did not permit. - for illegal in [ - "Reset Endpoint failed", - "Stop Endpoint failed", - "Set TR Dequeue failed", - "would not clear the halt", - "is being let go", - ] { - if log.contains(illegal) { - return Err(format!("{illegal:?} after a single staged failure\n{log}")); - } - } - serial::Serial::named("boot console", log.as_str()).must_be_clean()?; - - eprintln!( - " [xhci] a HID interrupt endpoint broken at {which}: {broken:?} named the code, were \ - each found Running once and restarted (of {} recoveries in the boot), and {word:?} \ - plus a {:?} pointer delta crossed them afterwards", - log.matches("is Running, recovering").count(), - (DX * scale_x, DY * scale_y) - ); - Ok(()) -} - -/// Which device an `xHCI: USB on slot : interrupt endpoint …` -/// line is about, as the kind and the device. -/// -/// **Refused rather than widened if the line stops naming one.** Recovery lines -/// carry ` slot ` and nothing else that identifies a device, so a test -/// that cannot read this pair off the completion has no way to tell its own -/// device's recovery from another device's — and the only alternative to -/// refusing is the count over every device that reddened this test three times. -fn hid_broke_on(line: &str) -> Result<(&str, &str), String> { - line.split_once("xHCI: USB ") - .and_then(|(_, rest)| rest.split_once(": interrupt endpoint")) - .and_then(|(who, _)| who.split_once(" on ")) - .ok_or_else(|| { - format!("{line:?} does not name the device and the controller its endpoint broke \ - on, so nothing can tell that device's recovery from another's") - }) -} - -/// Devices plugged in **after** the machine has booted. -/// -/// The driver enumerated once, from `init`, and `dispatch_event` advanced past -/// every TRB that was not a transfer completion — Port Status Change Events -/// included. So the set of USB devices was whatever was connected at boot, -/// forever, and a keyboard plugged into a machine with no input did nothing at -/// all: no port line, no slot, no event, and a compositor already holding a -/// keyboard claim that would never produce anything. That machine is -/// indistinguishable from hung, and it is the first thing a person tries. -/// -/// **The actuator is QEMU's own `device_add`, and nothing about the driver is -/// modified to run this.** A USB device attached at runtime goes through the -/// same `usb_device_attach` → `xhci_port_update` → `xhci_port_notify` path a -/// device attached at startup does, so what the guest sees is a real Port -/// Status Change Event with a real device behind it. That is the whole -/// difference from `xhci_slow_connect`, which needs an actuator because -/// it has to aim at a window the boot opens and closes in milliseconds; here -/// the window is the entire life of the machine. -/// -/// Every claim below is host-side in the sense that matters: -/// -/// - **the keyboard** is the only one on the machine — `i8042=off`, and the -/// profile's boot-time HID is a tablet — so a keystroke that reaches -/// userland can only have crossed a device that was added after the boot; -/// - **the pointer** is the only *relative* one, so QEMU has no handler for an -/// injected `rel` event until it is plugged in, and the boot-time tablet -/// cannot stand in for it; -/// - **the disk's block count** is the size of a file the harness made, which -/// the guest can only have learned by running READ CAPACITY over the wire. -pub fn xhci_hotplug( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - /// The disk that arrives late: 48 GiB, which is no other device in this - /// suite and no round number the driver could have printed by accident. - /// Sparse, so the host pays for nothing. - const HOT_DISK_BYTES: u64 = 48 * 1024 * 1024 * 1024; - /// What the boot-time tablet takes, so what a late pointer must not. - const BOOT_SOURCE: u32 = 1; - const LATE_SOURCE: u32 = 2; - const DX: i32 = 40; - const DY: i32 = -30; - - let options = BootOptions { - profile: Profile::MetalHotplug, - qmp: true, - // With an i8042 on the machine QEMU would deliver the injected - // keystrokes over PS/2 and every assertion below would pass with the - // hot-plug path dead. - i8042: false, - ..Default::default() - }; - // The claim is about what is *not* on the machine at boot, and argv is the - // only place absence is visible: no console line distinguishes "the driver - // never enumerated a keyboard" from "there was never one to enumerate". - let argv = qemu::profile_argv(&options); - let usb = crate::usb_argv(&argv); - for absent in ["usb-kbd", "usb-mouse"] { - if usb.iter().any(|d| d.starts_with(absent)) { - return Err(format!("{absent} is on the bus at boot; argv has {usb:?}")); - } - } - if !usb.iter().any(|d| d.starts_with("usb-tablet")) { - return Err(format!("this gate needs the boot-time tablet, argv has {usb:?}")); - } - if !argv.iter().any(|a| a.contains("i8042=off")) { - return Err("the i8042 is on; a PS/2 keyboard could deliver instead".to_string()); - } - - let image = test_dir().join("usb-hotplug.img"); - drop(sparse(&image, HOT_DISK_BYTES)); - - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - let boot = qemu.boot_log().to_string(); - - // Both controllers came up and exactly one of them found nothing — the - // T14's Thunderbolt xHC exactly, and the controller everything below is - // plugged into. Without this the test could not tell a device enumerated - // late from one enumerated at boot on a port nobody looked at. - let found = boot.matches("xHCI: found at PCI ").count(); - if found != 2 { - return Err(format!("{found} controller(s) initialised, want 2:\n{boot}")); - } - let empty = boot.matches("xHCI: no HID devices on the controller").count(); - if empty != 1 { - return Err(format!( - "{empty} controller(s) reported an empty bus at boot, want the second one only:\n{boot}" - )); - } - let at_boot = crate::parse_xhci_binds(&boot); - if at_boot.len() != 1 || at_boot[0].kind != "tablet" { - return Err(format!("want exactly the tablet bound at boot, got {at_boot:?}\n{boot}")); - } - let booted: Vec = crate::parse_pointer_sources(&boot).iter().map(|(_, s)| *s).collect(); - if booted != vec![BOOT_SOURCE] { - return Err(format!( - "the boot-time tablet did not take source {BOOT_SOURCE} alone: {booted:?}\n{boot}" - )); - } - let Some((scale_x, scale_y)) = crate::parse_rel_scale(&boot) else { - return Err(format!("the kernel never said what pointer scale it used:\n{boot}")); - }; - - let hot_image = image.clone(); - let result = qemu.run_test_hooked( - "test_rs_input_events", - Duration::from_secs(60), - "===INPUT_READY===", - move |socket| { - // One monitor at a time: a `-qmp unix:…,server` socket serves one - // connection, so each phase opens, does its work and closes. - let mut devices = qemu::QmpDevices::open(socket); - devices.blockdev_add("hotdisk", &hot_image); - devices.add("usb-mouse", "xhci1.0", "hotmouse", &[]); - devices.add("usb-kbd", "xhci1.0", "hotkbd", &[]); - devices.add("usb-storage", "xhci1.0", "hotdisk0", &[("drive", "hotdisk")]); - drop(devices); - // The driver's own debounce is 100 ms and the enumeration behind it - // is microseconds under TCG; this is that with room, not a settling - // time the assertions depend on. - thread::sleep(Duration::from_millis(800)); - - let mut input = qemu::QmpInput::open(socket); - // Off the origin first: the accumulated position clamps at 0, so a - // move up or left from there is invisible. - input.mouse(100, 100, None); - thread::sleep(Duration::from_millis(100)); - input.mouse(DX, DY, None); - thread::sleep(Duration::from_millis(100)); - for key in ["h", "e", "l", "l", "o"] { - input.keys(&[(key, true), (key, false)]); - thread::sleep(Duration::from_millis(20)); - } - drop(input); - thread::sleep(Duration::from_millis(200)); - - let mut devices = qemu::QmpDevices::open(socket); - devices.del("hotmouse"); - devices.del("hotdisk0"); - drop(devices); - thread::sleep(Duration::from_millis(800)); - - // The keyboard is still there, and still the only one. - let mut input = qemu::QmpInput::open(socket); - for key in ["w", "o", "r", "l", "d"] { - input.keys(&[(key, true), (key, false)]); - thread::sleep(Duration::from_millis(20)); - } - drop(input); - - // And a pointer plugged in where the last one was unplugged, which - // is the only thing that can show the button-table entry came back: - // a driver that leaked it binds this one as source 3. - let mut devices = qemu::QmpDevices::open(socket); - devices.add("usb-mouse", "xhci1.0", "hotmouse2", &[]); - drop(devices); - thread::sleep(Duration::from_millis(800)); - crate::input_events_end(&mut qemu::QmpInput::open(socket)); - }, - ); - if let Some(err) = &result.error { - return Err(format!("{err}\n{}\n{}", result.serial, result.stdout)); - } - let log = format!("{boot}{}", result.serial); - for bad in ["PANIC:", "panicked at"] { - if log.contains(bad) { - return Err(format!("{bad:?} while devices came and went\n{log}")); - } - } - - hotplug_bound(&log)?; - hotplug_delivered(&result.stdout, "hello", (DX * scale_x, DY * scale_y))?; - hotplug_unbound(&log)?; - // The keyboard was untouched by the mouse's teardown, and the merge it - // shares with the pointer that went away still works. `world` is typed - // after the unplug and nothing else in this boot can produce it. - let typed: String = crate::parse_key_events(&result.stdout) - .iter() - .filter(|e| e.modifiers & 0x10 == 0) - .map(|e| e.translated.as_str()) - .collect(); - if !typed.contains("world") { - return Err(format!( - "typed {typed:?} — the keyboard beside the unplugged pointer stopped delivering\n{}", - result.stdout - )); - } - - // The replug. Source 2 twice is the assertion: an entry that is leaked and - // one that is handed back read the same from every other angle. - let sources: Vec = crate::parse_pointer_sources(&log).iter().map(|(_, s)| *s).collect(); - if sources != vec![BOOT_SOURCE, LATE_SOURCE, LATE_SOURCE] { - return Err(format!( - "pointer sources were {sources:?}, want the boot tablet's {BOOT_SOURCE} and then \ - {LATE_SOURCE} twice — the second late pointer took a fresh entry, so the first \ - one's was never given back\n{log}" - )); - } - - let _ = std::fs::remove_file(&image); - eprintln!( - " [xhci] after boot: a keyboard, a pointer and a {HOT_DISK_BYTES} B disk enumerated on a \ - controller that had nothing on it; typed keys and a {:?} pointer delta delivered \ - host-side; unplug released source {LATE_SOURCE}, disabled the slots and took the disk \ - offline; the replug took source {LATE_SOURCE} again", - (DX * scale_x, DY * scale_y) - ); - Ok(()) -} - -/// Everything the guest has to say about three devices that were not there -/// when it booted. -fn hotplug_bound(log: &str) -> Result<(), String> { - // 48 GiB in the 4 KiB blocks the driver counts in, which is a number that - // exists only because the guest asked the device. - let geometry = format!("{} blocks of 512 B", 48u64 * 1024 * 1024 * 1024 / 4096); - for want in [ - "xHCI: USB keyboard ready on slot", - "xHCI: USB mouse ready on slot", - "usb-storage: disk 1 ready on slot", - geometry.as_str(), - ] { - if !log.contains(want) { - return Err(format!("nothing enumerated after the boot: no {want:?}\n{log}")); - } - } - Ok(()) -} - -/// That the devices which arrived late are the ones delivering. -fn hotplug_delivered(stdout: &str, word: &str, want: (i32, i32)) -> Result<(), String> { - let typed: String = crate::parse_key_events(stdout) - .iter() - .filter(|e| e.modifiers & 0x10 == 0) - .map(|e| e.translated.as_str()) - .collect(); - if !typed.contains(word) { - return Err(format!( - "typed {typed:?}, want it to contain {word:?} — this machine has no keyboard but the \ - one plugged in after it booted\n{stdout}" - )); - } - let pointer = crate::parse_mouse_events(stdout); - let deltas: Vec<(i32, i32)> = pointer - .windows(2) - .map(|w| (w[1].x as i32 - w[0].x as i32, w[1].y as i32 - w[0].y as i32)) - .collect(); - if !deltas.contains(&want) { - return Err(format!( - "no pointer event moved by {want:?}; deltas seen: {deltas:?} — the boot-time tablet \ - is absolute, so a relative move can only have come from the mouse that was plugged \ - in\n{stdout}" - )); - } - Ok(()) -} - /// A device pulled and pushed back before the driver has looked at the port /// twice — which is what a person replugging a mouse does. /// @@ -4108,32 +3448,6 @@ fn take_one(pool: &mut Vec, id: u8) -> bool { } } -/// Everything a device that has been pulled has to leave behind. -fn hotplug_unbound(log: &str) -> Result<(), String> { - for want in [ - "xHCI: port ", - "disconnected", - "unplugged from port", - "source 2 released", - "xHCI: slot ", - "disabled", - "usb-storage: disk 1 unplugged", - "it is offline", - ] { - if !log.contains(want) { - return Err(format!("an unplugged device left {want:?} unsaid\n{log}")); - } - } - // The teardown ran the commands the controller's state permits. Every one - // of these lines is `run_command` reporting one it refused. - for illegal in ["Disable Slot failed", "Disable Slot timed out"] { - if log.contains(illegal) { - return Err(format!("{illegal:?} during the teardown\n{log}")); - } - } - Ok(()) -} - /// A disk the driver refuses, on the port the controller enumerates *first*. /// /// `bind` claims a 64 KiB DMA pool block, issues Configure Endpoint — which diff --git a/tests/common/volumes.rs b/tests/common/volumes.rs index 3273dc85603..89c369a97ab 100644 --- a/tests/common/volumes.rs +++ b/tests/common/volumes.rs @@ -781,8 +781,7 @@ fn rotation( // log — the sink drains everything pending before it looks at the size — // so a metal-sim boot makes a handful, measured at four. That is under the // retention bound, which is why this only requires the count to stay inside - // it; deleting the oldest is `wall_clock_file`'s claim, staged with a full - // volume rather than hoped for here. + // it. if logs.len() < 2 || logs.len() > super::wallclock::MAX_LOG_FILES { return Err(format!( "the volume holds {} log files, wanted 2..={}: {}", diff --git a/tests/common/wallclock.rs b/tests/common/wallclock.rs index 33d50a1e032..a8a18503cd6 100644 --- a/tests/common/wallclock.rs +++ b/tests/common/wallclock.rs @@ -71,24 +71,6 @@ const WINDOW_MARKER: &str = "between-tests-window-is-captured"; /// quietly weakening the gate. pub const MAX_LOG_FILES: usize = 16; -/// The name of the boot the guest must delete, and the ones it must not. -/// -/// Sixteen staged files, so the volume is exactly at the bound before the guest -/// starts and this boot's own file is the one that puts it over. They are dated -/// well before [`RTC_BASE`], so the one that has to go is unambiguous — and -/// they are *not* consecutive days, so an implementation deleting "the first one -/// it listed" or "the lowest index" rather than the oldest lands on a different -/// file than the one asserted. -fn staged_logs() -> Vec<(String, Vec)> { - (0..MAX_LOG_FILES) - .map(|i| { - let name = format!("2019-12-{:02}-120000.log", 1 + i * 2); - let body = format!("staged by the host as boot {i}\n").into_bytes(); - (name, body) - }) - .collect() -} - /// The log files this module's kernel wrote or was given, oldest first. fn logs(entries: &[Entry]) -> Vec<&Entry> { entries.iter().filter(|e| toyos_build::bootlog::is_logd_file(&e.name)).collect() @@ -106,13 +88,6 @@ fn probed_epoch(log: &str) -> Option { rest.split_whitespace().next()?.parse().ok() } -/// What the same probe printed for `std`'s `SystemTime::now`. -fn probed_std_epoch(log: &str) -> Option { - let line = log.lines().find(|l| l.contains("wall-clock: std_epoch="))?; - let rest = line.split("std_epoch=").nth(1)?; - rest.split_whitespace().next()?.parse().ok() -} - fn clock_lines(log: &str) -> String { let lines: Vec<&str> = log .lines() @@ -213,124 +188,6 @@ fn boot_and_read( Ok((entries, log)) } -/// One file per boot, named and stamped from the wall clock, with the oldest -/// deleted once the volume is at its bound. -pub fn wall_clock_file( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - let staged = staged_logs(); - let (entries, log) = - boot_and_read(test_config, c_bins, rust_bins, "wall-clock-boot.img", &[], &staged)?; - let logs = logs(&entries); - - // The bound, on the volume rather than in the guest's account of it. - if logs.len() != MAX_LOG_FILES { - return Err(format!( - "the volume holds {} logs, not the {MAX_LOG_FILES} the bound allows: {}\n{}", - logs.len(), - names(&logs), - clock_lines(&log) - )); - } - - // The oldest staged file and no other. `staged_logs` dates them two days - // apart, so deleting by list order or by index would take a different one. - let oldest = &staged[0].0; - if logs.iter().any(|e| &e.name == oldest) { - return Err(format!( - "the oldest log {oldest} is still on the volume, so the bound was met by deleting \ - something else: {}", - names(&logs) - )); - } - for (name, _) in &staged[1..] { - if !logs.iter().any(|e| &e.name == name) { - return Err(format!( - "{name} was deleted and it is not the oldest: {}\n{}", - names(&logs), - clock_lines(&log) - )); - } - } - if !log.contains(&format!("/log/{oldest} was deleted")) { - return Err(format!( - "nothing in the log names {oldest} as the file that was deleted to make room\n{}", - clock_lines(&log) - )); - } - - // This boot's own file: named for the staged instant, which needs the - // century register to come out as 2101 rather than 2001. - let Some(mine) = logs.iter().find(|e| e.name.starts_with(RTC_BASE_DATE)) else { - return Err(format!( - "no log named for the day the host staged ({RTC_BASE_DATE}*): {}\n{}", - names(&logs), - clock_lines(&log) - )); - }; - let drift = mine.modified - RTC_BASE_SECS; - if !(0..=MAX_BOOT_DRIFT_SECS).contains(&drift) { - return Err(format!( - "{} carries a timestamp {drift}s from the {RTC_BASE} the host set, outside \ - 0..={MAX_BOOT_DRIFT_SECS}\n{}", - mine.name, - clock_lines(&log) - )); - } - if mine.len == 0 { - return Err(format!("{} is on the volume and empty", mine.name)); - } - - // The other end of the same instant. Firmware named no zone on this - // machine — OVMF ships `EFI_UNSPECIFIED_TIMEZONE` — so the kernel takes the - // RTC as UTC and the epoch syscall must answer the staged instant itself. - // A kernel serving 1970, or serving local time shifted by a zone it - // invented, lands outside this window. - let Some(epoch) = probed_epoch(&log) else { - return Err(format!( - "the guest never printed what `SYS_CLOCK_EPOCH` answered\n{}", - clock_lines(&log) - )); - }; - let epoch_drift = epoch - RTC_BASE_SECS; - if !(0..=MAX_BOOT_DRIFT_SECS).contains(&epoch_drift) { - return Err(format!( - "`SYS_CLOCK_EPOCH` answered {epoch}, {epoch_drift}s from the {RTC_BASE} the host set \ - and outside 0..={MAX_BOOT_DRIFT_SECS}\n{}", - clock_lines(&log) - )); - } - - // Judged against `RTC_BASE` and not against the syscall above, so a std - // agreeing with a wrong kernel still lands outside this window. - let Some(std_epoch) = probed_std_epoch(&log) else { - return Err(format!( - "the guest never printed what std's `SystemTime::now` answered\n{}", - clock_lines(&log) - )); - }; - let std_drift = std_epoch - RTC_BASE_SECS; - if !(0..=MAX_BOOT_DRIFT_SECS).contains(&std_drift) { - return Err(format!( - "std's `SystemTime::now` answered {std_epoch}, {std_drift}s from the {RTC_BASE} the \ - host set and outside 0..={MAX_BOOT_DRIFT_SECS}. A std that never asks the kernel \ - answers the epoch, which is {}s out\n{}", - -RTC_BASE_SECS, - clock_lines(&log) - )); - } - - eprintln!( - " [clock] {} carries {} bytes, stamped {drift}s after the {RTC_BASE} the host set, epoch \ - {epoch_drift}s after it and std {std_drift}s after it; {} deleted for the \ - {MAX_LOG_FILES}-log bound", - mine.name, mine.len, oldest - ); - Ok(()) -} - /// A firmware-named zone separates local time from UTC, in the direction UEFI /// defines. /// diff --git a/tests/quiescelastcase/system.toml b/tests/quiescelastcase/system.toml index 7a4510d7c06..5145a13dd2a 100644 --- a/tests/quiescelastcase/system.toml +++ b/tests/quiescelastcase/system.toml @@ -1,5 +1,5 @@ -# The boot `quiesce_wakes_on_the_last_park` and `quiesce_wakes_on_the_last_exit` -# judge: its one job starts the two threads the kernel may hold and reboots. +# The boot `quiesce_wakes_on_the_last_park` judges: its one job starts the +# thread the kernel may hold and reboots. [boot] start = ["logd", "test-runner"] diff --git a/tests/test-durations b/tests/test-durations index 4af45c58707..c32248eab77 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -157,11 +157,9 @@ boot_volume_metadata_error 4968 c_capture_ignores_daemon_lines 4578 cache_eviction 8165 connect_before_serve 23 -console_line_atomicity 8925 console_locale_detect 3922 control_regs 7243 control_regs_negative 3872 -control_regs_verdict 0 debug_float 11 debug_trap 25 demand_paging_sse 12 @@ -184,7 +182,6 @@ empty_dir_stat 12 endowment_denied 10 esp_filesystem 10123 exit_wait_storm 178 -run_exit_status 0 fat_backing_revoked 8226 fault_gates 31 foreign_disk_untouched 4688 @@ -211,7 +208,6 @@ hda_tone 13389 hda_two_live_refused 4848 hierarchy_paths 87 home_backing_revoked 663 -home_budget_refusal_retried 18661 home_overwrite_reads_back 4835 https_tls13 5166 https_tls13_e1000e 6072 @@ -221,11 +217,9 @@ i8042_fadt_denial 5736 i8042_health 9509 i8042_health_cadence 10236 i8042_kbd_echo 4770 -i8042_keyboard 4550 i8042_mouse 4310 i8042_no_spurious_wake 146 i8042_quarantine 10068 -i8042_quarantine_verdict 0 i8042_undecoded_bytes 4960 idle_stack_guard 9601 inbox_cancel_wakes 543 @@ -262,8 +256,6 @@ locale_detect 9959 locale_detect_unrecognized 160 log_backing_read_error 4887 log_conservation_smp1 4686 -log_conservation_smp4 8248 -log_conservation_smp8 5112 log_flush_retry 20526 log_nested_emit 5008 log_partition_identity 9516 @@ -299,11 +291,9 @@ netd_connection_caps 6538 netd_gone_mid_bind 32 netd_hostile_peer 4509 netd_listener_forgery 2649 -nightly_tier_is_announced 0 null_sink_client_exits 2167 null_sink_shipped_client 7095 nvme_home_roundtrip 13 -nvme_image_is_held_by_one_guest 0 nvme_large_device 6052 nvme_wide_sector 3500 operation_nesting 5075 @@ -343,10 +333,8 @@ screen_console_clear 5756 screen_console_panic 6788 screen_console_scroll 12310 screen_console_shell 2604 -screen_decoder 6 screen_diag_boot 7330 screen_early_panel 5779 -screen_early_panic 4441 screen_fatal_halt 4970 screen_fatal_halt_composited 6908 screen_gop_firmware_mode 12624 @@ -357,18 +345,14 @@ screen_log_absent 1823 screen_paged_scrollback 7384 screen_pager_keys 16152 screen_panic_muted 4416 -serial_vocabulary 2 shipped_config_boots 3024 shm_release_reclaims 29 short_sleep_livelock 4823 smp_failed_ap_leaves_no_hole 6672 smp_roster_and_tsc_trail 4926 -so_cache_refusals 7784 sshd_exec 5110 -sshd_fail_closed 2307 sshd_files 535 sshd_key_auth 1428 -stall_is_not_a_verdict 0 std_alloc 12 std_fs 17 std_fs_write 12 @@ -383,9 +367,6 @@ std_tls_dlopen 18 std_tls_multi_crate 13 std_unwind 18 std_unwind_so 20 -suite_split 1 -suspend_detector 20 -suspend_invalidates_a_verdict 0 swiss_german_layout 5355 syscall_cost 92 syscall_window_nmi 6825 @@ -397,7 +378,6 @@ tlb_shootdown_waits 107 toybox_cp_volume 18616 toybox_file_tools 205 usb_boot_stick_pulled 15650 -usb_disk_index_stable 6267 usb_flush_optional 18056 usb_pool_exhausted 5237 usb_refused_disk_first 7260 @@ -415,7 +395,6 @@ virtio_used_ring 4189 volume_from_another_disk 7122 wake_storm_cost 337 wall_clock_century_register 9030 -wall_clock_file 4642 wall_clock_no_century 6987 wall_clock_now 13 wall_clock_rtc_dead 8070 @@ -431,8 +410,6 @@ xhci_deaf_registers 21244 xhci_descriptor_walk 5186 xhci_flap 8220 xhci_full_speed_device 8833 -xhci_hid_break 15290 -xhci_hotplug 7687 xhci_many_devices 6247 xhci_msi_only 5757 xhci_no_interrupt 5114 diff --git a/tests/toyos-rust-tests/src/bin/abuse_listener_hijack.rs b/tests/toyos-rust-tests/src/bin/abuse_listener_hijack.rs index 4f565f67c7b..cb81350f675 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_listener_hijack.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_listener_hijack.rs @@ -33,8 +33,7 @@ //! the only wrong-typed handle in the ABI that does**: an `add` entry's //! connector is routinely one a *peer* transferred — a `provides` name is //! exactly that — so presenting the wrong one may be reporting a peer's bug -//! rather than your own. `/system/bin/init`'s launcher is why, and -//! `launcher_refusals` is the other end of the same property. +//! rather than your own. `/system/bin/init`'s launcher is why. //! 3. **A handle number from another process's table names nothing here.** The //! victim prints the raw number of a live acceptor of its own; the thief //! presenting it is ended, and the victim's port is still serving afterwards. diff --git a/tests/toyos-rust-tests/src/bin/console_line_atomicity.rs b/tests/toyos-rust-tests/src/bin/console_line_atomicity.rs deleted file mode 100644 index 8ed0479c3ca..00000000000 --- a/tests/toyos-rust-tests/src/bin/console_line_atomicity.rs +++ /dev/null @@ -1,204 +0,0 @@ -//! Two processes, two `write`s per line, and not one line that belongs to -//! both — nor one a program put on the console itself. -//! -//! **The defect this is aimed at is a `write` being the unit of -//! interleaving.** `println!` is a `LineWriter`: it issues `flush_buf()` and -//! then `inner.write(rest)`, so a line leaves the program in two pieces with an -//! arbitrary gap between them, and anything else written in that gap could -//! land inside the line. Each process assembles its own line before it is a -//! record (`toyos::log::stdio`), so two *processes* is the shape that tests it -//! and two threads would not. -//! -//! The two writes are made by hand rather than through `println!` because the -//! split has to be the subject rather than an implementation detail of `std`: -//! the line is a fixed width, the gap is exactly in the middle, and the newline -//! is on the second write, which is the piece the line waits for. -//! -//! **And the console is not a program's to write.** A program's console is -//! read-only: its lines reach the console only through `logd`, under its name, -//! so each writer's first act is a line on its console that must be refused. -//! -//! The verdict is the host's — a line is mixed or it is not, and only the -//! console capture can say. What this binary owes the host is that both writers -//! ran, said how much, and agreed about it. - -use std::io::Write; -use std::process::{exit, Command}; - -use toyos::log::stdio::{self, Stream}; -use toyos_abi::syscall::{self, SyscallError}; -use toyos_abi::RawHandle; - -const SELF_PATH: &str = "/system/bin/test_rs_console_line_atomicity"; - -/// Lines each writer emits. -/// -/// **A count and not a duration**, so the gate's verdict does not move with the -/// host: the assertion is zero mixed lines out of `2 * LINES`, and a run that -/// produced fewer lines than that has failed the non-vacuity check rather than -/// passed a weaker version of the same test. Both writers' lines fit the -/// runner's log ring unread, so a line missing is one lost, never one the ring -/// had no room for. -const LINES: usize = 400; - -/// Bytes in one line, newline included. -/// -/// Two hundred, which is comfortably inside `MAX_CONSOLE_LINE`'s 1024 — the -/// claim under test is that a whole line is one unit, and a line past that -/// bound is deliberately emitted in pieces of it, which is a different -/// sentence. -const WIDTH: usize = 200; - -/// Bytes the third writer says, in two `write`s, never ending them with a -/// newline. -/// -/// **The other half of the same line.** A line leaves on the `\n` that ends -/// it; the one moment a partial line stops being "not finished yet" and becomes -/// "all there will ever be" is the process exiting, which ends it. Without that -/// these bytes are dropped on the floor — a dying process's last words lost — -/// and a hundred of them arriving whole is also proof the line accumulated -/// across two `write`s to get there. -const MIDLINE: usize = 100; - -/// The byte the third writer repeats. Not `A` or `B`, so the whole-line count -/// cannot see it, and a hundred of them in a row is nothing an ordinary console -/// line contains. -const MIDLINE_BYTE: u8 = b'C'; - -/// Digits of the per-writer sequence number each line carries after its -/// leading tag byte, zero-padded so every line stays exactly [`WIDTH`] bytes. -const SEQ_DIGITS: usize = 6; - -/// The console, at stdin's slot: the runner starts this job holding its own -/// there (`CONSOLE_JOBS`), and each writer inherits it, read-only. -const CONSOLE: RawHandle = RawHandle(0); - -/// A line in the kernel's own shape, which a program on the console could -/// pass off as a record. -const FORGED: &[u8] = b"[kernel 0.000 cpu0] console-atomicity: a record no kernel wrote\n"; - -fn main() { - let mut args = std::env::args(); - let _ = args.next(); - match args.next().as_deref() { - Some("C") => exit_mid_line(), - Some(tag) => { - let byte = tag.as_bytes().first().copied().unwrap_or(b'?'); - write_lines(byte); - } - None => parent(), - } -} - -/// Say half of something, say the other half, and exit without ever ending it. -/// -/// No newline anywhere, so nothing in the write path makes these bytes a -/// line: what does is this process exiting. Through `std`, whose exit is the -/// one that ends it, with a flush between the halves, so the first leaves as -/// a piece the line goes on from. -fn exit_mid_line() { - let partial = [MIDLINE_BYTE; MIDLINE]; - let (head, tail) = partial.split_at(MIDLINE / 2); - let mut out = std::io::stdout(); - let wrote = out.write_all(head).and_then(|()| out.flush()).and_then(|()| out.write_all(tail)); - if let Err(e) = wrote { - eprintln!("console-atomicity: the mid-line writer's write failed: {e}"); - exit(1); - } - exit(0); -} - -/// A piece taken whole, or this process ends saying what it got instead. -fn written(got: Result, len: usize, who: &str) { - if got != Ok(len) { - eprintln!("console-atomicity: {who} wrote {got:?} of {len}"); - exit(1); - } -} - -/// Spawn the two writers and wait for both. -/// -/// Two processes and not two threads: the buffer is per console object and a -/// process gets its own, so two threads of one process share one buffer and -/// would prove nothing about the property the object exists to have. -fn parent() { - let mut children = Vec::new(); - for tag in ["A", "B"] { - match Command::new(SELF_PATH).arg(tag).spawn() { - Ok(child) => children.push((tag, child)), - Err(e) => { - eprintln!("console-atomicity: writer {tag} would not start: {e}"); - exit(1); - } - } - } - for (tag, mut child) in children { - match child.wait() { - Ok(status) if status.code() == Some(0) => {} - Ok(status) => { - eprintln!("console-atomicity: writer {tag} exited {:?}", status.code()); - exit(1); - } - Err(e) => { - eprintln!("console-atomicity: writer {tag} would not be waited for: {e}"); - exit(1); - } - } - } - // **The mid-line writer runs after the other two are gone**, so the bytes - // its exit flushes cannot land inside a line the count above is reading — - // they are on the wire on their own, which is what lets the host look for - // them as a run rather than as a line. - match Command::new(SELF_PATH).arg("C").spawn().and_then(|mut c| c.wait()) { - Ok(status) if status.code() == Some(0) => {} - Ok(status) => { - eprintln!("console-atomicity: the mid-line writer exited {:?}", status.code()); - exit(1); - } - Err(e) => { - eprintln!("console-atomicity: the mid-line writer would not run: {e}"); - exit(1); - } - } - // After all three, so the count the host checks against is a claim about a - // run that finished rather than one still going. It follows the mid-line - // writer's unterminated bytes on the wire, which is why the host finds this - // declaration with a substring search rather than a whole-line one. - println!( - "console-atomicity: writers=2 lines={LINES} width={WIDTH} midline={MIDLINE} \ - seq={SEQ_DIGITS}" - ); -} - -/// One writer: `LINES` numbered lines of one repeated byte, each in two -/// `write`s. -/// -/// The sequence number after the leading tag byte is what lets the host tell -/// a gap in a writer's own run from a capture that ends early — a count alone -/// reads both as the same missing-lines number. -fn write_lines(tag: u8) { - match syscall::write(CONSOLE, FORGED) { - Err(SyscallError::PermissionDenied) => {} - other => { - eprintln!("console-atomicity: writer {} put a line on its console: {other:?}", tag as char); - exit(1); - } - } - let mut line = [tag; WIDTH]; - line[WIDTH - 1] = b'\n'; - for seq in 0..LINES { - let mut digits = [0u8; SEQ_DIGITS]; - let mut rest = seq; - for d in digits.iter_mut().rev() { - *d = b'0' + (rest % 10) as u8; - rest /= 10; - } - line[1..1 + SEQ_DIGITS].copy_from_slice(&digits); - let (head, tail) = line.split_at(WIDTH / 2); - // Refused rather than retried: a short write here is the sink taking - // half a line, which is the defect and not an error to paper over. - for piece in [head, tail] { - written(stdio::write(Stream::Out, piece), piece.len(), &format!("writer {}", tag as char)); - } - } -} diff --git a/tests/toyos-rust-tests/src/bin/home_fsync_budget.rs b/tests/toyos-rust-tests/src/bin/home_fsync_budget.rs deleted file mode 100644 index 36cdf5854b7..00000000000 --- a/tests/toyos-rust-tests/src/bin/home_fsync_budget.rs +++ /dev/null @@ -1,30 +0,0 @@ -//! An fsync on `/home` whose first attempt is budget-refused (`fsync-budget-spent`) -//! must retry on a fresh budget and succeed — a `BudgetExpired` reaching the -//! bcachefs adapter as `Io` ends the syscall on attempt 1 instead (F9). -//! `tests/common/storage.rs::home_budget_refusal_retried` boots and judges this. - -use std::fs::File; -use std::io::{Read, Write}; - -/// Mirrored in `tests/common/storage.rs::home_budget_refusal_retried`. -const PATH: &str = "/home/f9-budget.bin"; -const LEN: usize = 3 * 4096 + 41; - -fn pattern() -> Vec { - (0..LEN).map(|i| (i.wrapping_mul(151) ^ 0x3C) as u8).collect() -} - -fn main() { - let want = pattern(); - let mut f = File::create(PATH).expect("create on /home"); - f.write_all(&want).expect("write"); - f.sync_all().expect( - "fsync refused: a budget-expired first attempt must be retried on a fresh budget, \ - never returned as the device's word", - ); - - let mut got = Vec::new(); - File::open(PATH).expect("re-open").read_to_end(&mut got).expect("read back"); - assert_eq!(got, want, "the bytes changed across the refused-then-retried fsync"); - println!("budget-refused fsync retried to durable: {LEN} bytes on /home"); -} diff --git a/tests/toyos-rust-tests/src/bin/quiesce_last.rs b/tests/toyos-rust-tests/src/bin/quiesce_last.rs index 1f240479ee6..581501a31a9 100644 --- a/tests/toyos-rust-tests/src/bin/quiesce_last.rs +++ b/tests/toyos-rust-tests/src/bin/quiesce_last.rs @@ -1,14 +1,12 @@ -//! Two threads the kernel's `quiesce-last-*` actuators can hold, and a reboot. +//! A thread the kernel's `quiesce-last-park` actuator can hold, and a reboot. //! -//! Both carry [`toyos_quiesce::LAST_THREAD`]'s name. One parks for longer than -//! any stop's budget and the other exits at once. `quiesce-last-park` holds -//! the first inside its `SYS_NANOSLEEP`, and `quiesce-last-exit` holds the -//! second inside its `SYS_THREAD_EXIT`. The kernel holds the reset itself until -//! one of them is held, so this program orders nothing. +//! It carries [`toyos_quiesce::LAST_THREAD`]'s name and parks for longer than +//! any stop's budget; `quiesce-last-park` holds it inside its `SYS_NANOSLEEP`. +//! The kernel holds the reset itself until it is held, so this program orders +//! nothing. //! -//! Nothing here asserts: `common::power::quiesce_wakes_on_the_last_park` and -//! `quiesce_wakes_on_the_last_exit` read the kernel's hold line and its `stop:` -//! record. +//! Nothing here asserts: `common::power::quiesce_wakes_on_the_last_park` reads +//! the kernel's hold line and its `stop:` record. use std::time::Duration; @@ -21,12 +19,10 @@ use toyos_quiesce::LAST_THREAD; const PARKED_FOR: Duration = Duration::from_secs(3_600); fn main() { - for body in [park as fn(), exit] { - std::thread::Builder::new() - .name(LAST_THREAD.into()) - .spawn(body) - .expect("spawn a thread the kernel may hold"); - } + std::thread::Builder::new() + .name(LAST_THREAD.into()) + .spawn(park) + .expect("spawn a thread the kernel may hold"); // Comes back only refused. let refused = toyos::power::stop(Stop::Reboot); @@ -37,5 +33,3 @@ fn main() { fn park() { std::thread::sleep(PARKED_FOR); } - -fn exit() {} diff --git a/tests/toyos-rust-tests/src/bin/so_cache_policy.rs b/tests/toyos-rust-tests/src/bin/so_cache_policy.rs deleted file mode 100644 index a922011ccff..00000000000 --- a/tests/toyos-rust-tests/src/bin/so_cache_policy.rs +++ /dev/null @@ -1,160 +0,0 @@ -//! The two refusals the shared-object cache owes, because it never removes an -//! entry: a path whose file changed, and a load past its byte budget. -//! `tests/common/storage.rs::so_cache_refusals` boots and judges this. -//! -//! **Every arm needs a second process.** `SYS_DLOPEN` answers a name this -//! process holds out of its own `lib_paths` before the cache is consulted, so -//! one process can never ask the cache twice about one path. - -use std::process::{Command, Stdio}; - -use toyos_abi::syscall::{self, SyscallError}; - -/// Mirrored in `so_cache_refusals`, which reads its bytes off the device. -const STALE: &str = "/home/so-cache-stale.so"; -const SAME_SIZE: &str = "/home/so-cache-same-size.so"; -const FIRST: &str = "/system/lib/libtls_lib.so"; -const SECOND: &str = "/system/lib/libtls_dlopen_lib.so"; -/// A symbol `FIRST` exports and `SECOND` does not: the verdict is a name. -const ONLY_IN_FIRST: &[u8] = b"tls_get_label"; - -/// 8 MiB of budget against 2 MiB images, so a kernel that never refuses runs -/// out of attempts and says so rather than looping. -const BUDGET_ATTEMPTS: usize = 12; - -const SELF_PATH: &str = "/system/bin/test_rs_so_cache_policy"; - -fn main() { - match std::env::args().nth(1) { - Some(path) => load_and_report(&path), - None => test(), - } -} - -/// Both arms run whatever the first answers: on a kernel with neither refusal -/// both are red, and stopping at the first would hide one control. -fn test() { - let arms = [ - ("stale-size", a_changed_file_is_refused()), - ("stale-mtime", a_same_size_rewrite_is_refused()), - ("budget", the_budget_is_refused()), - ]; - let mut failed = false; - for (name, outcome) in arms { - match outcome { - Ok(said) => println!(" {name}: {said}"), - Err(why) => { - println!(" {name} FAILED: {why}"); - failed = true; - } - } - } - if failed { - std::process::exit(1); - } - println!("the shared-object cache refuses a stale path and a full budget"); -} - -/// One library, loaded; a different one over the same path, loaded again. The -/// second must be refused: the first image is mapped into the child that loaded -/// it, so serving it again is a lie and reloading would map the library twice. -fn a_changed_file_is_refused() -> Result { - let only = String::from_utf8_lossy(ONLY_IN_FIRST).into_owned(); - copy(FIRST, STALE); - let first = load_in_child(STALE); - if !(first.contains("LOADED") && first.contains("SYMBOL-FOUND")) { - return Err(format!("the first load of {STALE} did not resolve {only}: {first}")); - } - - copy(SECOND, STALE); - let second = load_in_child(STALE); - if second.contains("SYMBOL-FOUND") { - return Err(format!( - "{STALE} now holds {SECOND}, which exports no {only} — a kernel that resolved it \ - served the image {FIRST} left in the cache: {second}" - )); - } - if !second.contains("REFUSED NotSupported") { - return Err(format!( - "the load of a path whose file changed was not refused by name: {second}" - )); - } - Ok(format!("{STALE} became {SECOND} and the second load was refused")) -} - -/// The same library written over itself: same bytes, same length, new mtime. -/// **This is the arm a partial fix fails** — the one above also changes the -/// size, so an identity keyed on size alone passes it. No symbol can tell these -/// two writes apart, so the refusal itself is the whole verdict. -fn a_same_size_rewrite_is_refused() -> Result { - copy(FIRST, SAME_SIZE); - let first = load_in_child(SAME_SIZE); - if !first.contains("LOADED") { - return Err(format!("the first load of {SAME_SIZE} did not happen: {first}")); - } - - copy(FIRST, SAME_SIZE); - let second = load_in_child(SAME_SIZE); - if !second.contains("REFUSED NotSupported") { - return Err(format!( - "{SAME_SIZE} was rewritten with the same bytes at the same length and the load was \ - not refused: {second} — an identity keyed on size alone cannot see this write" - )); - } - Ok(format!("a same-size rewrite of {SAME_SIZE} was refused")) -} - -/// Distinct paths are distinct entries, so copies of one library fill the -/// budget as surely as distinct libraries would. -fn the_budget_is_refused() -> Result { - for attempt in 0..BUDGET_ATTEMPTS { - let path = format!("/home/so-cache-fill-{attempt}.so"); - copy(FIRST, &path); - let said = load_in_child(&path); - if said.contains("REFUSED ResourceExhausted") { - return Ok(format!("refused at copy {attempt} of {BUDGET_ATTEMPTS}")); - } - if !said.contains("LOADED") { - return Err(format!("copy {attempt} neither loaded nor refused: {said}")); - } - } - Err(format!( - "{BUDGET_ATTEMPTS} distinct 2 MiB images entered the cache unrefused — this kernel \ - holds no byte budget over it at all" - )) -} - -fn copy(from: &str, to: &str) { - let bytes = std::fs::read(from).unwrap_or_else(|e| panic!("read {from}: {e}")); - std::fs::write(to, &bytes).unwrap_or_else(|e| panic!("write {to}: {e}")); -} - -fn load_in_child(path: &str) -> String { - let out = Command::new(SELF_PATH) - .arg(path) - .stdout(Stdio::piped()) - .spawn() - .unwrap_or_else(|e| panic!("spawn a loader for {path}: {e}")) - .wait_with_output() - .unwrap_or_else(|e| panic!("wait for the loader of {path}: {e}")); - String::from_utf8_lossy(&out.stdout).trim().to_string() -} - -fn load_in_this_process(path: &str) { - match syscall::dl_open(path.as_bytes()) { - Ok(handle) => { - println!("LOADED"); - if unsafe { syscall::dl_sym(handle, ONLY_IN_FIRST) }.is_ok() { - println!("SYMBOL-FOUND"); - } - } - Err(SyscallError::NotSupported) => println!("REFUSED NotSupported"), - Err(SyscallError::ResourceExhausted) => println!("REFUSED ResourceExhausted"), - Err(other) => println!("REFUSED {other:?}"), - } -} - -fn load_and_report(path: &str) -> ! { - load_in_this_process(path); - std::process::exit(0) -} diff --git a/tests/toyos.rs b/tests/toyos.rs index 29f1ca34a8d..493f00dbb1f 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -19,7 +19,7 @@ use toyos_build::bootlog::{self, boot_millis}; use toyos_build::heartbeat; use toyos_build::testargs::{self, Shard, SUITE}; use toyos_build::redlist; -use toyos_build::tiers::Tier; +use toyos_build::tiers::{Reach, Schedule, Tier}; struct TestDef { name: String, @@ -169,10 +169,6 @@ const RUST_SKIP: &[&str] = &[ // that measured anything. It also needs the real-time band, which only // `tests/latencycase` endows. `latency_wake` runs it there. "cyclictest", - // Its verdict is a property of the *console capture*, which only a boot of - // its own can hold: in the shared boot every other binary's output is in the - // same stream. `console_line_atomicity` runs it. - "console_line_atomicity", // **It reboots the machine**, so in the shared block it would end the boot // under whichever member came next; and its verdict is the order of the // console after that reset, which only its own boot holds. @@ -182,7 +178,7 @@ const RUST_SKIP: &[&str] = &[ // `quiesce_refuses_a_second_shutdown` runs it. "quiesce_twice", // The same, and its verdict is the stop record of a boot staged around it. - // `quiesce_wakes_on_the_last_park` and `quiesce_wakes_on_the_last_exit` run it. + // `quiesce_wakes_on_the_last_park` runs it. "quiesce_last", // The same, and its verdict is the log volume the stop leaves. // `quiesce_leaves_the_volume_whole` runs it. @@ -394,16 +390,6 @@ const RUST_SKIP: &[&str] = &[ // shared boot it would be reported against whichever test came next — and // every one after that. `short_sleep_livelock` gives it a boot of its own. "abuse_short_sleep", - // The four below were **running twice under one name**, once here on the - // plain boot and once as the test that owns the name, and the collision was - // invisible: `check_registration` compared the three declared lists against - // each other and never against the binaries the registry discovers. - // `check_no_collisions` closes that, and this is what it found. - // - // What the shared copy adds is the binary exiting 0 on a boot that gives it - // nothing to measure — `cache_eviction` in 132 ms against the 22.5 s its - // own device shape costs (run `31247206462`). - // // `cache_eviction` needs the small NVMe that makes the cache evict at all. "cache_eviction", // `writeback_reopen` and `writeback_spawn` each need their own boot with @@ -441,11 +427,6 @@ const RUST_SKIP: &[&str] = &[ // succeeds and its two must-refuse assertions red for an honest reason. // `fsync_failed_commit` boots it with the arm. "fsync_flush_failed", - // Needs `fsync-budget-spent` and the NVMe `/home`; unstaged it passes - // vacuously. `home_budget_refusal_retried` boots it with both. - "home_fsync_budget", - // Needs `so-cache-tiny` and the NVMe `/home`. `so_cache_refusals` gives both. - "so_cache_policy", // Needs the NVMe `/home` and a boot of its own for the readback it is judged against; `home_overwrite_reads_back` runs it. "home_overwrite_zero", // Needs a boot where the DATA volume is ours and absent; on the shared @@ -485,6 +466,7 @@ const RUST_SKIP: &[&str] = &[ /// `sched_stress` is the one whose two runs differ by *kernel* rather than by /// what the host staged: the shipping build here, `sched_check_build`'s /// assert-carrying build there. +#[allow(dead_code, reason = "`suite_split` reads it, in `toyos-checks` alone")] const DRIVEN_AND_SHARED: &[&str] = &[ // The lost-wake canary: its shared run is the count on the shipping // kernel with nothing staged, and `blocking_read_window` drives it again @@ -523,20 +505,17 @@ const EARLY_PANIC_MESSAGE: &str = "test-early-panic: on-screen console check"; // used to read a screendump now reads the console instead — a screenshot is a // poor way to ask "did the right process come up", and thresholds over a live // desktop are how those tests passed vacuously twice. -// `screen_decoder` needs no guest at all; it proves the decoder against a -// bitmap it rendered itself, before anything points it at a real screen. /// The order was once about kernel rebuilds — every actuator was a build, and a /// feature-carrying test last left the plain-kernel ones above it untouched by /// the thrash. There are two kernels now and nothing to thrash; the order is /// kept because these are read the way they are /// written. const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ - ("screen_decoder", Sched::Parallel, Tier::Fast), // Two boots, each ended at a loader line rather than at the kernel's ready // marker. Every verdict is a count of rows against a count of lines off the // same boot's console; no clock is in either. - ("screen_loader_lines", Sched::Parallel, Tier::Fast), - ("screen_gop_firmware_mode", Sched::Parallel, Tier::Nightly), + ("screen_loader_lines", Sched::Parallel, Tier::Nightly), + ("screen_gop_firmware_mode", Sched::Parallel, Tier::Weekly), // `thread::sleep(5 s)` is the measurement, not a ceiling: the assertion is // literally that the log is still on the panel five seconds after the boot // finished, so a 2x slower machine changes nothing about the wait but the @@ -544,28 +523,27 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ ("screen_diag_boot", Sched::Parallel, Tier::Nightly), // A guest halted in the window, so the panel is read where only the repaint // under test can have painted it. - ("screen_early_panel", Sched::Parallel, Tier::Fast), + ("screen_early_panel", Sched::Parallel, Tier::Nightly), ("screen_log_absent", Sched::Parallel, Tier::Fast), ("screen_console_shell", Sched::Parallel, Tier::Fast), ("screen_console_clear", Sched::Parallel, Tier::Fast), - ("screen_console_scroll", Sched::Parallel, Tier::Nightly), - ("screen_i8042_health", Sched::Parallel, Tier::Fast), + ("screen_console_scroll", Sched::Parallel, Tier::Fast), + ("screen_i8042_health", Sched::Parallel, Tier::Weekly), // Ctrl+Alt+D with no console at all: the panel is the whole channel, and a // compositor is holding it. A fixed 2 s settle sits inside the dump's own // guest-timed 15 s hold, and the verdict is whether the report survived // the desktop's next repaint — which only where that wait lands decides, // so it is timer-anchored despite being a screendump-content check. ("screen_blocked_dump", Sched::Parallel, Tier::Nightly), - ("screen_early_panic", Sched::Parallel, Tier::Fast), ("screen_late_panic", Sched::Parallel, Tier::Fast), - ("screen_paged_scrollback", Sched::Parallel, Tier::Nightly), - ("screen_panic_muted", Sched::Parallel, Tier::Fast), + ("screen_paged_scrollback", Sched::Parallel, Tier::Weekly), + ("screen_panic_muted", Sched::Parallel, Tier::Weekly), ("screen_console_panic", Sched::Parallel, Tier::Fast), ("screen_fatal_halt", Sched::Parallel, Tier::Fast), // The same fatal path from inside Ctrl+Alt+D's report painter, holding the // panel's latch it will never give back: the report has to take the screen // anyway, and its CPU has to go on to watch the reset bound. - ("screen_fatal_behind_a_painter", Sched::Parallel, Tier::Fast), + ("screen_fatal_behind_a_painter", Sched::Parallel, Tier::Nightly), // The same fatal path with a compositor holding the panel, which is the // only configuration the owner's laptop is ever in and the one no screen // test covered: `screen_fatal_halt` boots a config with no compositor, and @@ -629,7 +607,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // and the first number against a number. Parallel: every verdict is // arithmetic over counters the guest printed, and there is no clock in any // of it. - ("irq_census_conservation", Sched::Parallel, Tier::Fast), + ("irq_census_conservation", Sched::Parallel, Tier::Weekly), ("control_regs", Sched::Parallel, Tier::Fast), ("control_regs_negative", Sched::Parallel, Tier::Fast), // The boot facts the metal suite reads off a machine's own records: every @@ -641,24 +619,24 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // matches its rows; and one machine-wide TLB shootdown's cost is a // distribution rather than one boot's average. Every verdict is arithmetic // over records, with no clock in the judging, so all six are Parallel. - ("smp_roster_and_tsc_trail", Sched::Parallel, Tier::Fast), - ("pmm_accounting", Sched::Parallel, Tier::Fast), + ("smp_roster_and_tsc_trail", Sched::Parallel, Tier::Nightly), + ("pmm_accounting", Sched::Parallel, Tier::Nightly), // ROOT is the loader's image in memory: the kernel says it mounted it // from memory, and that init was spawned with no storage command issued; // and a loader that hands no image is a boot refused by name, never one // that goes to a disk for ROOT. Both are records of one boot each, with no // clock in the verdict. - ("root_from_memory", Sched::Parallel, Tier::Fast), - ("root_withheld_refused", Sched::Parallel, Tier::Fast), + ("root_from_memory", Sched::Parallel, Tier::Nightly), + ("root_withheld_refused", Sched::Parallel, Tier::Nightly), // The boot from power-on, as the kernel converts the loader's TSC readings: // judged against the loader's raw counts and the kernel's own rate, and // bounded above by the host's clock, which is Parallel-safe because load // only widens that bound. - ("boot_from_power_on", Sched::Parallel, Tier::Fast), - ("acpi_table_inventory", Sched::Parallel, Tier::Fast), - ("timer_calibration", Sched::Parallel, Tier::Fast), - ("pci_inventory", Sched::Parallel, Tier::Fast), - ("tlb_shootdown_cost", Sched::Parallel, Tier::Fast), + ("boot_from_power_on", Sched::Parallel, Tier::Nightly), + ("acpi_table_inventory", Sched::Parallel, Tier::Nightly), + ("timer_calibration", Sched::Parallel, Tier::Nightly), + ("pci_inventory", Sched::Parallel, Tier::Nightly), + ("tlb_shootdown_cost", Sched::Parallel, Tier::Nightly), // What a waiter in the real-time band pays to be woken, as a distribution // over ten thousand programmed wakes — and, beside it, that the number // reaches a machine with no serial port at all, through the kernel's own @@ -670,15 +648,14 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // // Serial: it is the one registration here whose verdict is a *time*, and a // wake latency measured beside eleven other guests is the host's schedule. - // Nightly for that same reason. ("latency_wake", Sched::Serial, Tier::Nightly), - ("smp_failed_ap_leaves_no_hole", Sched::Parallel, Tier::Fast), - ("input_merge", Sched::Parallel, Tier::Fast), - ("metal_sim_input", Sched::Parallel, Tier::Fast), - ("input_claim_absent", Sched::Parallel, Tier::Fast), + ("smp_failed_ap_leaves_no_hole", Sched::Parallel, Tier::Weekly), + ("input_merge", Sched::Parallel, Tier::Weekly), + ("metal_sim_input", Sched::Parallel, Tier::Weekly), + ("input_claim_absent", Sched::Parallel, Tier::Weekly), // One boot; every verdict is a PPM header field or a console line, and no - // clock is in any of them, so it is Nightly for its price and nothing else. - ("gpu_set_resolution", Sched::Parallel, Tier::Nightly), + // clock is in any of them. + ("gpu_set_resolution", Sched::Parallel, Tier::Fast), // One boot from here to `metal_sim_compositor_stall` (`METAL_SIM_DESKTOP`). ("metal_sim_compositor", Sched::Parallel, Tier::Nightly), // Reads the boot log this group already has, after the member above has @@ -705,7 +682,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // verdict rather than a slower one. Watched to happen, on a compositor made // slow on purpose. Its own boot too: it leaves the pointer somewhere else // and the window in a different place than it found them. - ("metal_sim_window_drag", Sched::Serial, Tier::Nightly), + ("metal_sim_window_drag", Sched::Serial, Tier::Weekly), // Its own boot: the compositor it abuses has to be one nothing else has // touched. ("metal_sim_hostile_clipboard", Sched::Parallel, Tier::Fast), @@ -723,7 +700,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // Reads a device capture and requires at least MIN_SIGNAL_SECS = 0.8 s of // it to carry signal at peak >= 6000 — an absolute seconds-of-signal // floor on audio recorded in real time, not a fraction of the capture and - // not compute-bound: timer-anchored, and Nightly for that reason. + // not compute-bound: timer-anchored. ("doom_music", Sched::Parallel, Tier::Nightly), // One C program through the toolchain's clang, read by the loader's decoder, // and one boot to run it. @@ -734,13 +711,12 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("doom_frames", Sched::Parallel, Tier::Fast), // A tone played while soundd's pipe to a stalled log is full. The verdicts // are the capture's gaps and a count of lines; the clocks are liveness - // guards. Nightly for the two megabytes of refusals the log reads back - // through a TCG guest's volume. + // guards. ("soundd_log_stall", Sched::Serial, Tier::Nightly), // ureq and rustls from crates.io, fetching over TLS 1.3 from a host this // test mints a CA for. Every verdict is a printed line or a digest; the // only clock is `run_test`'s ceiling. - ("https_tls13", Sched::Parallel, Tier::Fast), + ("https_tls13", Sched::Parallel, Tier::Weekly), // The same judge over netd's Intel driver instead of the virtio one: the // 82574L QEMU models has the register file the T14's I219 has, so this is // where that driver moves real frames before the laptop does. @@ -756,55 +732,54 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // only machine in reach that runs that driver. ("log_stream_e1000e", Sched::Parallel, Tier::Nightly), // A reader that never reads, beside a flood: the file and a second reader - // are whole regardless. Nightly for the flood's megabytes. + // are whole regardless. ("log_stream_stalled_reader", Sched::Parallel, Tier::Nightly), // A program's line in `/log`, on the served log and on the console, under // the name of the pipe it came out of. Lines and a comparison; no clock. - ("log_program_line", Sched::Parallel, Tier::Fast), + ("log_program_line", Sched::Parallel, Tier::Nightly), // A program writing the kernel's words and another program's head: every // judge of `/log` reads the truth. Lines, and the exit judge; no clock. - ("log_program_forgery", Sched::Parallel, Tier::Fast), + ("log_program_forgery", Sched::Parallel, Tier::Nightly), // A stop the kernel refuses after logd flushed for it: a line said after // it is in `/log`. Lines; no clock. - ("log_after_a_refused_stop", Sched::Parallel, Tier::Fast), + ("log_after_a_refused_stop", Sched::Parallel, Tier::Nightly), // The same stop with logd's flush held until init resumes it: logd runs // the flush, then the resume, and lives. Lines; init's flush bound. - ("log_resume_meets_its_flush", Sched::Parallel, Tier::Fast), + ("log_resume_meets_its_flush", Sched::Parallel, Tier::Nightly), // A child flooding its parent's log ring while logd reads none of it: the // parent's next line is in `/log`. Lines; its clock is a guard. - ("log_ring_keeps_the_owners_slots", Sched::Parallel, Tier::Fast), + ("log_ring_keeps_the_owners_slots", Sched::Parallel, Tier::Nightly), // A program's line said after three batches of records, read before them: // `/log` carries it after every one. Lines and positions; no clock. - ("log_program_line_after_its_records", Sched::Parallel, Tier::Fast), + ("log_program_line_after_its_records", Sched::Parallel, Tier::Nightly), // A program printing init's word accepting a swap of netd: logd turns // nobody away, and a reader after it is admitted. Lines; no clock. - ("log_carrier_forgery", Sched::Parallel, Tier::Fast), + ("log_carrier_forgery", Sched::Parallel, Tier::Nightly), // A flood many times its ring: every line in `/log` in order or counted. - // Nightly for its megabytes through a TCG guest's volume. ("log_program_flood", Sched::Parallel, Tier::Nightly), // netd taking this machine's address from the network instead of carrying // one written down. The DHCP server it is judged against is QEMU's own, an // implementation of RFC 2131 this repository did not write, and its lease // is known field by field. The verdicts are records and a lease's fields; // no clock in it. - ("lan_dhcp_lease", Sched::Parallel, Tier::Fast), + ("lan_dhcp_lease", Sched::Parallel, Tier::Nightly), // The lease probe on the same part: netd serves its window, exits with the // lease's verdict, and the report it leaves on the log volume names the // backend's lease with frames counted both ways and the lease kept across // a link flap. Records and a file; its clocks are the flap's hold and the // drain that outlasts netd's window, and a slower machine moves neither. - ("lan_lease_report", Sched::Parallel, Tier::Fast), + ("lan_lease_report", Sched::Parallel, Tier::Weekly), // The T14's talking boot rehearsed on the same part: the log the guest // serves, read from its first line through a forward, one command over ssh // answered byte for byte, and `reboot` over ssh ending the guest. Lines, // bytes and a reset; its clocks are liveness guards on a guest that // stopped talking. - ("lan_talk", Sched::Parallel, Tier::Fast), + ("lan_talk", Sched::Parallel, Tier::Nightly), // netd answering for its name: a resolver's query through a forward onto // the guest's multicast DNS port is answered with the lease's address, and // one for another name is not answered. Bytes; its clocks are a guard on // the answer and the window the unanswered query is given. - ("lan_mdns_answer", Sched::Parallel, Tier::Fast), + ("lan_mdns_answer", Sched::Parallel, Tier::Nightly), // A running service's binary replaced with no reboot: netd swapped for its // rebuild over ssh while the stream runs, on virtio-net and on the 82574 // the T14's I219 shares a register file with; a wrong digest and a @@ -812,74 +787,73 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // once answered by the old binary running again. Records, bytes and the // guest's own `/log`; every clock is a liveness guard on a guest that // stopped talking, and probation is init's. - ("swap_netd", Sched::Parallel, Tier::Fast), + ("swap_netd", Sched::Parallel, Tier::Nightly), // The machine updates itself: an image over `ssh … update` is written to // the idle slot and is the kernel the next boot runs; every slot the loader // must refuse is refused by name and the other boots; and a slot whose // kernel dies falls back on its own, its death in the next boot's `/log`. // The same machine's floor, grant and hang: a floor is its key's and its // image's, init grants nothing the slot table names but an idle slot's - // partition, and a hang of an unproven image is a death. Each is 25 s or - // more of boots, past the fast tier's line, so they are nightly. - ("update_boots_the_new_kernel", Sched::Parallel, Tier::Nightly), - ("update_refusals_boot_the_other_slot", Sched::Parallel, Tier::Nightly), - ("update_falls_back_from_a_dying_kernel", Sched::Parallel, Tier::Nightly), + // partition, and a hang of an unproven image is a death. + ("update_boots_the_new_kernel", Sched::Parallel, Tier::Weekly), + ("update_refusals_boot_the_other_slot", Sched::Parallel, Tier::Weekly), + ("update_falls_back_from_a_dying_kernel", Sched::Parallel, Tier::Weekly), ("update_hang_kills_an_unproven_image", Sched::Parallel, Tier::Nightly), ("update_grant_refuses_a_stray_partition", Sched::Parallel, Tier::Nightly), - ("update_floor_is_the_images_own", Sched::Parallel, Tier::Nightly), + ("update_floor_is_the_images_own", Sched::Parallel, Tier::Weekly), ("update_refused_pass_credits_no_image", Sched::Parallel, Tier::Nightly), - ("lan_swap", Sched::Parallel, Tier::Fast), - ("swap_refusals", Sched::Parallel, Tier::Fast), - ("swap_crash_rolls_back", Sched::Parallel, Tier::Fast), + ("lan_swap", Sched::Parallel, Tier::Nightly), + ("swap_refusals", Sched::Parallel, Tier::Nightly), + ("swap_crash_rolls_back", Sched::Parallel, Tier::Nightly), // The 82574 swapped to a holder that stops it and masters it while the // host sends it frames. The verdict is the kernel's console; its clocks // are liveness guards. - ("swap_quiets_the_function", Sched::Parallel, Tier::Fast), + ("swap_quiets_the_function", Sched::Parallel, Tier::Nightly), // The same part released as though nothing could reset it, the way the // T14's I219 is, and swapped to a holder that masters it with its receive // unit still on. The verdict is the kernel's console; its clocks are // liveness guards. - ("swap_keeps_what_nothing_reset", Sched::Parallel, Tier::Fast), + ("swap_keeps_what_nothing_reset", Sched::Parallel, Tier::Nightly), // The same part swapped to a holder that aims it outside its grant and // waits on its claim. The verdict is the holder's own word on what the // faulted claim answered; its clocks are liveness guards. - ("swap_fault_tells_its_holder", Sched::Parallel, Tier::Fast), + ("swap_fault_tells_its_holder", Sched::Parallel, Tier::Nightly), // An `igb` netd held, released by an Express function level reset and // claimed again by netd's replacement, which reads it through the window // its claim maps. The verdict is what the function answers there. - ("swap_resets_the_function", Sched::Parallel, Tier::Fast), + ("swap_resets_the_function", Sched::Parallel, Tier::Nightly), // The same `igb`, its window left where its reset put it: the replacement // is refused it, and the verdict is init failing the swap by the // device's name rather than putting netd in service without it. ("swap_refused_device_fails", Sched::Parallel, Tier::Fast), // The same, its window moved inside the cut rather than lost: the kernel // reads the address the register holds, not only that it holds one. - ("swap_moved_device_fails", Sched::Parallel, Tier::Fast), + ("swap_moved_device_fails", Sched::Parallel, Tier::Nightly), // A program sshd runs undeclared asks for the swap port. The verdict is // the program's own exit: the port is not in the namespace it inherited. - ("swap_not_inherited", Sched::Parallel, Tier::Fast), + ("swap_not_inherited", Sched::Parallel, Tier::Nightly), // The same client on a wire with no server: it says it has no address and // announces itself anyway. Its verdict waits out netd's own lease bound, so // a slower machine moves it. - ("lan_no_lease", Sched::Parallel, Tier::Nightly), - ("netd_connection_caps", Sched::Parallel, Tier::Fast), + ("lan_no_lease", Sched::Parallel, Tier::Weekly), + ("netd_connection_caps", Sched::Parallel, Tier::Nightly), // `inspect` against the four owners on the one boot that runs them all, // with the negative control's binary run on it too. Every verdict is the // set of lines a selector printed; no clock in any. - ("inspect_reads_its_owners", Sched::Parallel, Tier::Fast), + ("inspect_reads_its_owners", Sched::Parallel, Tier::Nightly), // The netcase boot again: netd must not abort a listener on a ring flag its // own client forged. Its verdict is a kernel-reported EOF or its absence; // no clock in it. - ("netd_listener_forgery", Sched::Parallel, Tier::Fast), + ("netd_listener_forgery", Sched::Parallel, Tier::Weekly), // The netcase boot again: a receiver that stops reading until its pipe // is full still gets every byte of a stream past it. The verdict is the // guest's byte-for-byte comparison; its clocks are liveness guards. - ("netd_slow_reader", Sched::Parallel, Tier::Fast), + ("netd_slow_reader", Sched::Parallel, Tier::Nightly), // The netcase boot again: client pipes netd cannot use, or loses under // it, cost that client its connection and never netd. The verdict is a // round trip after each case, a named line per refusal and a clean // console; its clocks are liveness guards. - ("netd_refused_pipes", Sched::Parallel, Tier::Fast), + ("netd_refused_pipes", Sched::Parallel, Tier::Nightly), // The netcase boot again: an accept netd refuses for room leaves its owner // a wake for the connection it left, once room returns. The verdict // is the guest's wake or its absence. @@ -888,74 +862,69 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // pipe's room alone, the peer holding the connection open and silent. The // verdict is the guest's byte-for-byte comparison; its clocks are // liveness guards. - ("netd_held_open", Sched::Parallel, Tier::Fast), + ("netd_held_open", Sched::Parallel, Tier::Nightly), // The netcase boot again: a client out-writing a peer that stopped reading - // costs netd no CPU. Nightly: its verdict is the machine's busy time over - // a window of real time. + // costs netd no CPU. ("netd_stalled_peer", Sched::Parallel, Tier::Nightly), // The netcase boot again, beside a host UDP echo: a datagram the client's // pipe will not take whole ends that socket by name and no other. The // verdict is each socket's answer; its clocks are liveness guards. - ("netd_udp_refused", Sched::Parallel, Tier::Fast), + ("netd_udp_refused", Sched::Parallel, Tier::Nightly), // The netcase boot again, beside a host UDP echo: a socket bound through // std to 0.0.0.0 receives the echo's unicast reply. The verdict is the // reply's bytes; its clock is a liveness guard. - ("netd_udp_any_address", Sched::Parallel, Tier::Fast), + ("netd_udp_any_address", Sched::Parallel, Tier::Nightly), // The netcase boot, whose user network forwards 10.0.2.3 to this host's // resolver: `host` resolves a real name to the addresses this host's // resolver gives it, and a `.invalid` name to none. The verdict is the - // two answers; its clocks are liveness guards. Nightly, because both - // answers rest on this host's network, which no change to the tree moves. + // two answers; its clocks are liveness guards. ("dns_resolve", Sched::Parallel, Tier::Nightly), // The netcase boot, every frame it sends held once it has its lease: // lookups whose clients hung up or spoke again are let go at once, and // one nobody answers ends when its schedule does. The verdict is netd's // answers; the schedule's end is a bound derived from it. - ("netd_lookup_let_go", Sched::Parallel, Tier::Fast), + ("netd_lookup_let_go", Sched::Parallel, Tier::Nightly), // Two netcase boots, each frame put on the wire kept: the first DHCP // transaction ID of each differs, because netd seeds smoltcp's random // source from the kernel's. The verdict is two numbers off the wire. - ("netd_seeds_its_stack", Sched::Parallel, Tier::Fast), + ("netd_seeds_its_stack", Sched::Parallel, Tier::Nightly), // The netcase boot with two programs naming one PCI function: the verdict // is which of them the kernel let have it. Console lines only, no clock. - ("pci_function_is_exclusive", Sched::Parallel, Tier::Fast), + ("pci_function_is_exclusive", Sched::Parallel, Tier::Nightly), // The same boot again, read for what the kernel asked the machine before it // moved that function's BAR. Its own boot rather than a second assertion in // the row above, because that one's subject is exclusivity and a test that // reds tells a reader which of the two it is about. It waits out a drain for // the message record, so its price carries a fixed span of host wall clock. - ("bar_placement_is_proven", Sched::Parallel, Tier::Fast), - // Its own boot with a NIC under it, because sshd leaves at the bind on - // every other config. Every verdict is a line of text; no clock in any. - ("sshd_fail_closed", Sched::Parallel, Tier::Fast), + ("bar_placement_is_proven", Sched::Parallel, Tier::Nightly), // Its own boot with a blank DATA volume, asked over ssh where everything // it wrote went. Every verdict is a listing or a line; no clock in any. - ("layout_fresh_boot", Sched::Parallel, Tier::Fast), + ("layout_fresh_boot", Sched::Parallel, Tier::Nightly), // One `SSHD_LOGIN` boot for the three, driven by `tests/ssh-client-host`. // Adjacent because `group_of` makes adjacency load-bearing, and one tier // because one boot cannot be in two. Every verdict is bytes or an exit // status; the client's own ceiling is a liveness guard and no assertion // reads a clock. - ("sshd_exec", Sched::Parallel, Tier::Fast), - ("sshd_files", Sched::Parallel, Tier::Fast), - ("sshd_key_auth", Sched::Parallel, Tier::Fast), + ("sshd_exec", Sched::Parallel, Tier::Nightly), + ("sshd_files", Sched::Parallel, Tier::Nightly), + ("sshd_key_auth", Sched::Parallel, Tier::Nightly), // Serial: it measures netd's 2 s handshake deadline against the host's // clock, and counts how many connections survived a 48 ms paced burst // before that deadline could expire any of them. Both are wall-clock // margins, which is the definition of [`Sched::Serial`]. - ("netd_hostile_peer", Sched::Serial, Tier::Nightly), - ("launcher_refusals", Sched::Parallel, Tier::Fast), + ("netd_hostile_peer", Sched::Serial, Tier::Weekly), + ("launcher_refusals", Sched::Parallel, Tier::Weekly), // Paths a child prints and the kernel's refusals by name; no clock in any of them. - ("spawn_cwd", Sched::Parallel, Tier::Fast), - ("foreign_disk_untouched", Sched::Parallel, Tier::Fast), - ("volume_from_another_disk", Sched::Parallel, Tier::Fast), - ("broken_data_volume_is_absent", Sched::Parallel, Tier::Fast), - ("data_candidate_with_bad_geometry_is_absent", Sched::Parallel, Tier::Fast), + ("spawn_cwd", Sched::Parallel, Tier::Nightly), + ("foreign_disk_untouched", Sched::Parallel, Tier::Weekly), + ("volume_from_another_disk", Sched::Parallel, Tier::Weekly), + ("broken_data_volume_is_absent", Sched::Parallel, Tier::Nightly), + ("data_candidate_with_bad_geometry_is_absent", Sched::Parallel, Tier::Nightly), // Four kernel lines and a file read off the image once the guest is gone; no clock in any of them. - ("internal_disk_boot", Sched::Parallel, Tier::Fast), + ("internal_disk_boot", Sched::Parallel, Tier::Weekly), // One boot each, kernel lines and image bytes for verdicts, no clock in either. - ("block_duplicate_id", Sched::Parallel, Tier::Fast), - ("page_cache_partition_offset", Sched::Parallel, Tier::Fast), + ("block_duplicate_id", Sched::Parallel, Tier::Weekly), + ("page_cache_partition_offset", Sched::Parallel, Tier::Weekly), // A partition claimed as a device: one boot, every refusal in the guest, // the neighbours and the target judged off the image. Body in // `tests/common/partclaim.rs`, as are the two below. @@ -963,83 +932,72 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // Three boots: a disk that does not answer a read of its table, every // attempt refused until the deadman, and ROOT's source withheld from every // claim once its disk did not answer the boot's hold. - ("partition_claim_gives_up", Sched::Parallel, Tier::Fast), + ("partition_claim_gives_up", Sched::Parallel, Tier::Nightly), // Three boots, a USB stick's device leaving owing one claim's write and // coming back on another port each time: each partition's fsync answers // for its own writes, across a close, after another's flush, and at the // shutdown when nobody asked. - ("partition_claim_departure", Sched::Parallel, Tier::Fast), - // F9's negative control: a budget-refused /home fsync retried to durable, - // its bytes then read off the NVMe image by the host's own bcachefs - // reader. Body in `tests/common/storage.rs`. - ("home_budget_refusal_retried", Sched::Parallel, Tier::Nightly), - // The shared-object cache's two refusals. Body in `tests/common/storage.rs`. - ("so_cache_refusals", Sched::Parallel, Tier::Fast), + ("partition_claim_departure", Sched::Parallel, Tier::Nightly), // A same-length overwrite on /home, the guest's read held against the image. Body in `tests/common/storage.rs`. - ("home_overwrite_reads_back", Sched::Parallel, Tier::Fast), + ("home_overwrite_reads_back", Sched::Parallel, Tier::Weekly), // One filesystem under two paths: the guest writes under each of /apps and // /home, the host finds both in one volume on the image. Body in // `tests/common/storage.rs`. - ("apps_and_home_are_one_filesystem", Sched::Parallel, Tier::Fast), + ("apps_and_home_are_one_filesystem", Sched::Parallel, Tier::Weekly), // `pkg install ` from a local archive and gbae's first run: the whole // package path in one boot, judged off the DATA volume once the guest is // gone. Body in `tests/common/pkg.rs`. - ("pkg_install_gbae", Sched::Parallel, Tier::Fast), - ("boot_partition_identity", Sched::Parallel, Tier::Fast), + ("pkg_install_gbae", Sched::Parallel, Tier::Nightly), + ("boot_partition_identity", Sched::Parallel, Tier::Nightly), // One boot of its own, because it ends the machine. Every verdict is a // kernel line or the stop reason QEMU reported; no clock is in either. - ("machine_reboot", Sched::Parallel, Tier::Fast), + ("machine_reboot", Sched::Parallel, Tier::Weekly), // Its own boot: every verdict is a console line, QEMU's stop reason or a record off the image. ("metal_job_reboot", Sched::Parallel, Tier::Fast), // Its own boot, and the one whose numbers are the T14's: what it judges // here is the plumbing, since every span on an emulated device is a fact // about TCG. - ("metal_device_probe", Sched::Parallel, Tier::Fast), + ("metal_device_probe", Sched::Parallel, Tier::Nightly), // Its verdict waits out a staged window. - ("job_deadline_reboots", Sched::Parallel, Tier::Fast), + ("job_deadline_reboots", Sched::Parallel, Tier::Nightly), // Its own boot, and its verdict waits out the same staged window. ("quiesce_stops_the_machine", Sched::Parallel, Tier::Fast), // Its own boot: it ends the machine, and its verdict is the order of // kernel lines. - ("quiesce_refuses_a_second_shutdown", Sched::Parallel, Tier::Fast), + ("quiesce_refuses_a_second_shutdown", Sched::Parallel, Tier::Nightly), // Its own boot: it ends the machine, and its verdict is the volume that // boot leaves. - ("quiesce_leaves_the_volume_whole", Sched::Parallel, Tier::Fast), - // Its own boot each: it ends the machine, and its verdict is the stop - // record that boot writes. - ("quiesce_wakes_on_the_last_park", Sched::Parallel, Tier::Fast), - ("quiesce_wakes_on_the_last_exit", Sched::Parallel, Tier::Fast), - // Its own boot: it ends the machine, and its verdict is a dump served - // inside that boot's stop. - ("quiesce_dump_holds_the_stopped", Sched::Parallel, Tier::Fast), + ("quiesce_leaves_the_volume_whole", Sched::Parallel, Tier::Nightly), + // Its own boot: it ends the machine, and its verdict is the stop record + // that boot writes. + ("quiesce_wakes_on_the_last_park", Sched::Parallel, Tier::Nightly), // Two reads of `TCO_RLD` straddling a real-time stall, so a slower machine // changes the verdict. ("loader_watchdog_arms", Sched::Parallel, Tier::Nightly), // Its own boot, and the verdict is QEMU's stop reason inside the bound. ("watchdog_resets", Sched::Parallel, Tier::Nightly), // Serial: its verdict is that nothing happened for a span of host clock. - ("watchdog_fed", Sched::Serial, Tier::Nightly), + ("watchdog_fed", Sched::Serial, Tier::Weekly), // The panicked kernel's own bound, which is what ends a boot on a machine // whose chipset timer does not count. Both verdicts are QEMU's stop reason // against a bound the guest printed, so a slower machine moves both. ("panic_reboots", Sched::Parallel, Tier::Nightly), // The same verdict from inside `percpu::init_bsp`: the earliest point a // panic is reportable, and the window the owner's T14 stops in. - ("panic_before_peripherals_reboots", Sched::Parallel, Tier::Fast), + ("panic_before_peripherals_reboots", Sched::Parallel, Tier::Nightly), // Serial like `watchdog_fed`: its verdict is that nothing happened for a span of host clock. - ("panic_key_holds", Sched::Serial, Tier::Nightly), + ("panic_key_holds", Sched::Serial, Tier::Weekly), // The boot chain's three answers. The two chain names each watch a guest // take its own reset and read the pass after it, so both are anchored to // the bound the first boot counts down. ("blackbox_panic_chain", Sched::Parallel, Tier::Nightly), - ("blackbox_done_chain", Sched::Parallel, Tier::Fast), + ("blackbox_done_chain", Sched::Parallel, Tier::Nightly), // The one bound in this tree that ends a machine nothing else can: a boot // whose every CPU has stopped taking scheduler passes. Its verdict is a - // bound counted down in the guest, so it is Nightly. + // bound counted down in the guest. ("boot_deadline_ends_a_wedge", Sched::Parallel, Tier::Nightly), - // The same bound ending the same machine with a device in its hands, and - // Nightly for the same reason. - ("usb_reset_records_the_phase_it_cut", Sched::Parallel, Tier::Nightly), + // The same bound ending the same machine with a device in its hands. + ("usb_reset_records_the_phase_it_cut", Sched::Parallel, Tier::Weekly), // The other half of that same parameter, and the state its poll cannot // reach: one CPU with interrupts off, which no running CPU can see. Two // bounds counted down in the guest, so it belongs beside the row above. @@ -1050,36 +1008,32 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("panic_outlives_the_deadline", Sched::Parallel, Tier::Nightly), // Four chained boots, one per way this kernel reaches a reset, each // anchored to the bound its own first boot counts down. - ("usb_reset_hands_devices_back", Sched::Parallel, Tier::Nightly), + ("usb_reset_hands_devices_back", Sched::Parallel, Tier::Weekly), // The control on the chain: a record another image left in the same memory // is cleared and its pass boots a kernel, where a real predecessor's ends // the chain. One boot, one actuator. ("blackbox_foreign_record", Sched::Parallel, Tier::Fast), // Three launches of one image file, and the third is the one that makes it // a bound: a hang costs the machine one boot and never traps it. - ("hang_bounded_by_the_stick", Sched::Parallel, Tier::Fast), + ("hang_bounded_by_the_stick", Sched::Parallel, Tier::Nightly), // Its own boot, and every verdict is a line: no host clock in any of it. - ("blackbox_unclaimed_page", Sched::Parallel, Tier::Fast), + ("blackbox_unclaimed_page", Sched::Parallel, Tier::Nightly), // The seal read off the page's own bytes by QEMU, after a panic earlier than // anything the kernel used to learn the page's address from. Its own boot, // and no clock in the verdict. - ("blackbox_early_panic_sealed", Sched::Parallel, Tier::Fast), + ("blackbox_early_panic_sealed", Sched::Parallel, Tier::Nightly), // The same crash on the owner's machine's own shape — no serial port at all // — where the panel and the page are the only two channels there are. - ("blackbox_early_panic_sealed_muted", Sched::Parallel, Tier::Fast), + ("blackbox_early_panic_sealed_muted", Sched::Parallel, Tier::Nightly), // The exception entry's own seal, off the page's bytes on an ordinary boot. - ("blackbox_fault_sealed", Sched::Parallel, Tier::Fast), - ("double_fault_stack", Sched::Parallel, Tier::Fast), + ("blackbox_fault_sealed", Sched::Parallel, Tier::Nightly), + ("double_fault_stack", Sched::Parallel, Tier::Nightly), // One boot of its own, ten seconds of Ring 3 spinning, and every verdict is // a count the kernel printed or a line it printed: how many NMIs landed at // CPL 0 with a user `rsp`, against how many landed in Ring 3, both off the // same storm. No host clock is in any of it — the ten seconds are how long // the victim spins, not a margin anything is measured against — so Parallel. - // - // **It was three boots and priced at 19,740 ms** on the hosted lane (run - // 32580794553), twice the Fast ceiling. The two negative controls are the - // name below. - ("syscall_window_nmi", Sched::Parallel, Tier::Fast), + ("syscall_window_nmi", Sched::Parallel, Tier::Weekly), // The two controls on the name above: the kernel with vector 2's IST index // taken off, which must double fault at the entry on the NMI aimed at the // CPU the storm holds inside it, with `cr2 = rsp - 8` at the held `rsp`, and @@ -1087,10 +1041,10 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // path. Both boots end in a halted machine that has to be drained past its // own report, which is where the price is. Nothing in either verdict is a // duration. - ("syscall_window_nmi_controls", Sched::Parallel, Tier::Nightly), + ("syscall_window_nmi_controls", Sched::Parallel, Tier::Weekly), // Its own boot, its own feature, and it drives the guest only through // stdin — nothing it touches is shared with another test. - ("idle_stack_guard", Sched::Parallel, Tier::Nightly), + ("idle_stack_guard", Sched::Parallel, Tier::Weekly), // Its own boot and its own feature, and it deafens one CPU for 400 ms — // but the deafening is a *window*, and the verdict is whether the NMI is // answered inside `NMI_BUDGET_NS`, which is one millisecond. That is a @@ -1099,30 +1053,30 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // reads exactly like the defect it hunts, and it was green alone in the // same run and three times after it. Serial by the default rule — a // verdict that is a duration does not go in the parallel phase. - ("dump_nmi_probe", Sched::Serial, Tier::Nightly), + ("dump_nmi_probe", Sched::Serial, Tier::Weekly), // The same dump asked for inside the passes that may not serve it, on one // CPU. Parallel: every verdict is a line the guest prints or a count the guest // keeps, and no duration is in any of them. - ("dump_left_pending_is_owed", Sched::Parallel, Tier::Fast), - ("diskless_boot", Sched::Parallel, Tier::Fast), + ("dump_left_pending_is_owed", Sched::Parallel, Tier::Nightly), + ("diskless_boot", Sched::Parallel, Tier::Nightly), // Every verdict is a line of text or a device property, and no clock is in // any of them. ("virtio_net_no_msix", Sched::Parallel, Tier::Fast), // One boot of the NIC config with a staged capability list, and every // verdict a console line. No clock in any of them. - ("pci_claim_caps_truncated", Sched::Parallel, Tier::Fast), + ("pci_claim_caps_truncated", Sched::Parallel, Tier::Nightly), // One boot, and its verdict is a line the kernel printed before any device // was brought up. No clock and no device in it. - ("virtio_used_ring", Sched::Parallel, Tier::Fast), + ("virtio_used_ring", Sched::Parallel, Tier::Weekly), // A fatal path with other CPUs running userland that makes kernel records: // none is stamped past the fatal record by more than an IPI takes. - ("panic_halts_the_others_first", Sched::Parallel, Tier::Fast), + ("panic_halts_the_others_first", Sched::Parallel, Tier::Nightly), // A kernel log line from PCI enumeration; no clock and no real device in it. - ("pci_capability_walk", Sched::Parallel, Tier::Fast), + ("pci_capability_walk", Sched::Parallel, Tier::Weekly), // What QEMU was told to create against what the guest enumerated: two // accounts of one bus from two independent readers. One boot, every // verdict a set comparison. - ("query_pci_agreement", Sched::Parallel, Tier::Fast), + ("query_pci_agreement", Sched::Parallel, Tier::Weekly), // The root `system.toml`, booted rather than read: the shipped init list // and program namespaces have no other gate a harness run can reach. // Every verdict is a console line. @@ -1130,45 +1084,41 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // One boot whose verdict is three lines of kernel log and a census column. // The two waits inside the guest are bounded and report rather than hang, so // no host clock decides anything. - ("lapic_spurious_vector", Sched::Parallel, Tier::Fast), + ("lapic_spurious_vector", Sched::Parallel, Tier::Weekly), // One boot with both stuck-device actuators armed. - ("driver_wait_refused", Sched::Parallel, Tier::Nightly), + ("driver_wait_refused", Sched::Parallel, Tier::Weekly), // One boot; the leak-rollback controls' two verdict lines. - ("leak_rollback_selftest", Sched::Parallel, Tier::Fast), + ("leak_rollback_selftest", Sched::Parallel, Tier::Weekly), // One boot; the reopen control's one verdict line. - ("process_reopen_selftest", Sched::Parallel, Tier::Fast), + ("process_reopen_selftest", Sched::Parallel, Tier::Weekly), // One boot; three read-fault control verdicts. - ("read_fault_selftests", Sched::Parallel, Tier::Fast), - ("xhci_many_devices", Sched::Parallel, Tier::Fast), + ("read_fault_selftests", Sched::Parallel, Tier::Weekly), + ("xhci_many_devices", Sched::Parallel, Tier::Weekly), // Its whole assertion is that a keystroke injected from the host crossed a // USB keyboard on the *second* controller, and `input_events_run` sends // each one only after the guest has printed the last — so a key the host // never got to send is a stall it names, and never a key the driver lost. - ("xhci_second_controller", Sched::Parallel, Tier::Fast), - ("xhci_two_controllers", Sched::Parallel, Tier::Fast), + ("xhci_second_controller", Sched::Parallel, Tier::Weekly), + ("xhci_two_controllers", Sched::Parallel, Tier::Weekly), // **Returned 2026-08-17**, on the same `input_events_run` the two names // above it run: it had `xhci_second_controller`'s sequence written out again // on fixed sleeps, and nothing sent the right-button release // `test_rs_input_events` exits on — so 30 s of its 35.2 s CI price was a // client waiting out a fallback deadline with every assertion already // satisfied. - ("xhci_msi_only", Sched::Parallel, Tier::Fast), - ("xhci_no_interrupt", Sched::Parallel, Tier::Fast), - ("nvme_large_device", Sched::Parallel, Tier::Fast), - ("nvme_wide_sector", Sched::Parallel, Tier::Fast), - ("iommu_discovery", Sched::Parallel, Tier::Nightly), - ("readdir_bound", Sched::Parallel, Tier::Nightly), + ("xhci_msi_only", Sched::Parallel, Tier::Weekly), + ("xhci_no_interrupt", Sched::Parallel, Tier::Weekly), + ("nvme_large_device", Sched::Parallel, Tier::Nightly), + ("nvme_wide_sector", Sched::Parallel, Tier::Weekly), + ("iommu_discovery", Sched::Parallel, Tier::Weekly), + ("readdir_bound", Sched::Parallel, Tier::Weekly), // Its own boot: it fills the VFS `created_dirs` cap and leaves it there. - ("mkdir_cap", Sched::Parallel, Tier::Fast), + ("mkdir_cap", Sched::Parallel, Tier::Weekly), // Two boots, and the verdict is that they answer differently. Nothing in it - // is timed: every arm is a process exit code or a byte comparison — still - // compute-bound, still Nightly: 11,075 ms in the sweep's final shard - // packing (run 31705986758) is a Cost row, the same shape - // `desktop_window_child` carries, not a reclassification. + // is timed: every arm is a process exit code or a byte comparison. ("fpu_isolation", Sched::Parallel, Tier::Nightly), - // Two boots, exit codes only (no clock, Parallel). Nightly for `Cost`: its - // second (`user-writable-gsbase`) kernel double-boots it over the ceiling. - ("gsbase_locked", Sched::Parallel, Tier::Nightly), + // Two boots, exit codes only (no clock, Parallel). + ("gsbase_locked", Sched::Parallel, Tier::Weekly), // The fourth declared kernel build, booted so that the scheduler core's // `feature = "check"` instruments are compiled and executed by a CI run at // all. One of its verdicts is a *quantile* of the guest's published @@ -1193,114 +1143,80 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // construct one. Parallel, and nothing in it is a duration — every verdict // is a comparison between two numbers the kernel printed, both of them // offsets it chose itself. - ("operation_nesting", Sched::Parallel, Tier::Fast), + ("operation_nesting", Sched::Parallel, Tier::Weekly), ("short_sleep_livelock", Sched::Parallel, Tier::Fast), // The spawn half alone: one headless boot whose verdict is kernel log // lines. - ("klogd_hosted", Sched::Parallel, Tier::Fast), - ("klogd_panic_halts", Sched::Parallel, Tier::Nightly), + ("klogd_hosted", Sched::Parallel, Tier::Weekly), ("klogd_fault_halts", Sched::Parallel, Tier::Nightly), ("syscall_panic_halts", Sched::Parallel, Tier::Nightly), ("syscall_fault_halts", Sched::Parallel, Tier::Nightly), ("lock_across_switch_halts", Sched::Parallel, Tier::Nightly), ("heap_over_ceiling_halts", Sched::Parallel, Tier::Nightly), // The two dead ends of the panic path, each staged on purpose and read for - // what the machine manages to say on its way out. **Two names because one - // over two boots measured 12 s twelve-wide on the dev host**, against - // `screen_late_panic`'s 5 s there for one boot of the same shape and its - // 3,782 ms in CI — a single name would have arrived at the fast tier's line - // with nothing to spare. Each boot dies inside the boot phases at the marker - // the harness waits for, so neither pays for a userland. Parallel and Fast: + // what the machine manages to say on its way out. Each boot dies inside the + // boot phases at the marker the harness waits for, so neither pays for a + // userland. Parallel: // every verdict is a substring of a report the guest wrote, and there is no // clock in any of it. - ("reentry_names_the_first_panic", Sched::Parallel, Tier::Fast), + ("reentry_names_the_first_panic", Sched::Parallel, Tier::Weekly), // The kernel hasher's boot-order obligation, in the row above's shape and // for its reasons. - ("hash_seed_precedes_every_map", Sched::Parallel, Tier::Fast), - ("double_panic_names_the_fault", Sched::Parallel, Tier::Nightly), + ("hash_seed_precedes_every_map", Sched::Parallel, Tier::Weekly), + ("double_panic_names_the_fault", Sched::Parallel, Tier::Weekly), // The third shape: a `#PF` inside a panic, which is the one // `fatal_exception`'s recursive short-circuit exists for and the one it // never classified. Same boot shape as its two neighbours — dies inside the - // boot phases at the marker, no userland — so Parallel and, pending its - // first measured run, Fast. - ("nested_fault_is_recursive", Sched::Parallel, Tier::Fast), - // The conservation law across `SYS_LOG_READ`, one registered name per - // width, and the nesting gate at one CPU. **Three names because one over - // three boots measured 17,112 ms in CI** — over the fast tier's line, and - // the gate the whole design turns on may not sit in the nightly tier — and - // because the three widths are different subjects rather than one subject - // measured three times. Parallel: every verdict is a ledger the + // boot phases at the marker, no userland — so Parallel. + ("nested_fault_is_recursive", Sched::Parallel, Tier::Weekly), + // The conservation law across `SYS_LOG_READ`, and the nesting gate at one + // CPU. Parallel: every verdict is a ledger the // guest computes over its own records — every sequence number read or // counted lost, every payload regenerated byte for byte — and not one of // them reads a clock. A loaded host makes the producers outrun the reader // further, which moves records from `read` into `lost` and leaves the law // exactly where it was. - ("log_conservation_smp1", Sched::Parallel, Tier::Fast), - ("log_conservation_smp4", Sched::Parallel, Tier::Nightly), - ("log_conservation_smp8", Sched::Parallel, Tier::Fast), - ("log_nested_emit", Sched::Parallel, Tier::Fast), + ("log_conservation_smp1", Sched::Parallel, Tier::Weekly), + ("log_nested_emit", Sched::Parallel, Tier::Weekly), // The same interrupt one window earlier — between a record's shard-pointer // read and its `xadd` — and its negative control, which is the only reader - // `log-unbracketed-reserve` has ever had. Parallel and Fast for + // `log-unbracketed-reserve` has ever had. Parallel for // `log_nested_emit`'s reasons: both verdicts are the guest's ledger over its // own records, one saying the shard kept a single order and the other that // it lost it by name, and no clock is in either. - ("log_reserve_window", Sched::Parallel, Tier::Fast), - ("log_reserve_window_negative", Sched::Parallel, Tier::Fast), - // Two processes building a fixed-width line out of two `write`s each, and a - // count of the lines that carry both of them. Parallel and Fast: the verdict - // is a count over a fixed number of lines the guest declares, so a loaded - // host changes when the writers run and not whether a line is whole. It boots - // its own machine because what it reads is the console capture, which a - // shared boot fills with everything else. - ("console_line_atomicity", Sched::Parallel, Tier::Nightly), - // What the C family is allowed to conclude from the line above being whole: - // a guest writes a daemon-shaped line into a real capture window on purpose + ("log_reserve_window", Sched::Parallel, Tier::Weekly), + ("log_reserve_window_negative", Sched::Parallel, Tier::Weekly), + // A guest writes a daemon-shaped line into a real capture window on purpose // and the real comparison ignores it, with the filter turned off as the // control. One boot, two `echo`s, and every verdict is a string comparison - // the host makes over a capture — no clock in it; Nightly because its - // *wall* clock is whatever the partition co-schedules, and it straddles the - // fast line run to run. - ("c_capture_ignores_daemon_lines", Sched::Parallel, Tier::Nightly), - // A poll on the machine's log against a *handle* going away. Parallel and - // Fast: both halves are verdicts the guest computes — a completion count + // the host makes over a capture — no clock in it. + ("c_capture_ignores_daemon_lines", Sched::Parallel, Tier::Weekly), + // A poll on the machine's log against a *handle* going away. Parallel: both + // halves are verdicts the guest computes — a completion count // immediately after a close, retried against a record arriving in the same // microseconds, and a completion afterwards bounded far above the two // scheduler passes it needs. - ("log_poll_outlives_a_close", Sched::Parallel, Tier::Fast), + ("log_poll_outlives_a_close", Sched::Parallel, Tier::Weekly), // The same question asked of the keyboard, where two *kinds* of object name // one source: a poll on stdin against the keyboard claim going away, a poll // on the mouse claim against its own, and an injected keystroke to show the - // first was still armed. Parallel and Fast: two of the three verdicts are + // first was still armed. Parallel: two of the three verdicts are // counts the guest takes immediately after a close on its own thread, and // the third is bounded far above the one interrupt it waits for. ("keyboard_claim_close_spares_stdin", Sched::Parallel, Tier::Fast), // One boot that stops dead in phase 3, read for what it managed to say. - ("pre_idle_wedge_speaks", Sched::Parallel, Tier::Fast), + ("pre_idle_wedge_speaks", Sched::Parallel, Tier::Weekly), ("i8042_health", Sched::Parallel, Tier::Nightly), - // And one from here to `i8042_mouse` (`I8042_TRACE`), which is why all - // three carry the answer the last of them needs. - // - // None of the three measures a rate. All three keep fewer bytes in flight - // than QEMU's PS/2 device holds, and all three do it the same way: nothing - // goes out until the guest has reported what the injection before it - // produced — `i8042_mouse` within [`MOUSE_LEAD`], the two keyboard ones a - // group at a time. So a guest with less of the host is a longer run and not - // a smaller count. **A wall clock cannot buy that bound**: the two keyboard - // tests spaced their injections with `thread::sleep` and put 26 and 20 - // bytes against a sixteen-byte device queue, which held only as long as the - // guest kept draining, and a stalled guest lost bytes with nothing anywhere - // reporting a loss. `i8042_keyboard` itself held the group Nightly on a cost - // that was really the fixed 5 s collection deadline in - // `test_rs_i8042_keyboard`; now that the binary exits on a sentinel instead, - // all three return. - ("i8042_keyboard", Sched::Parallel, Tier::Fast), - ("i8042_no_spurious_wake", Sched::Parallel, Tier::Fast), - ("i8042_mouse", Sched::Parallel, Tier::Fast), + // And one from here to `i8042_mouse` (`I8042_TRACE`). Neither measures a + // rate: nothing goes out until the guest has reported what the injection + // before it produced — `i8042_mouse` within [`MOUSE_LEAD`], the keyboard one + // a group at a time — so a guest with less of the host is a longer run and + // not a smaller count. + ("i8042_no_spurious_wake", Sched::Parallel, Tier::Nightly), + ("i8042_mouse", Sched::Parallel, Tier::Nightly), // A boot each, and deliberately not a group: every one of them changes - // the machine's layout, which `i8042_keyboard` asserts against, and a - // wizard that exits the instant it has its answer leaves the guest with - // nothing to run — so a later member reads a console the previous one is + // the machine's layout, and a wizard that exits the instant it has its + // answer leaves the guest with nothing to run — so a later member reads a console the previous one is // still draining into. // // Each is a wizard conversation typed from the host, and that used to make @@ -1312,19 +1228,11 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // against a marker with a twenty-second ceiling, so a slower guest is a // slower test and not a different verdict — which is the same argument // `i8042_kbd_echo` has run on at width 4 since the phase landed. - // - // **Returned 2026-08-17.** Eight of its 12.6 s CI price were - // `test_rs_locale_gate layout` holding an idle keyboard open until a fixed - // deadline expired, against half a second of injection; it exits on the End - // key's release now, which is `i8042_keyboard`'s own sentinel and the fix - // made for that whole family. - ("swiss_german_layout", Sched::Parallel, Tier::Fast), + ("swiss_german_layout", Sched::Parallel, Tier::Nightly), // One `LOCALE_WIZARD` boot for the pair since the drainer was made // runnable at commit — the boot-apiece and the injected drain keys it took // to share one were both the closed log-ring lag. Adjacent because - // `group_of` makes adjacency load-bearing; the carrier straddles the - // fast line run to run, so the pair is Nightly — the rider by its - // `RidesTheBootOf` row, the collateral that record exists to name. + // `group_of` makes adjacency load-bearing. ("locale_detect", Sched::Parallel, Tier::Nightly), ("locale_detect_unrecognized", Sched::Parallel, Tier::Nightly), // The wizard on the two surfaces the machine actually has, rather than on @@ -1340,7 +1248,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // An unmodified iced app on the desktop, launched from the shell: the // window, its text in the system font, no redraw it did not ask for, and // a clean exit when the compositor closes it. - ("toolkit_iced", Sched::Parallel, Tier::Nightly), + ("toolkit_iced", Sched::Parallel, Tier::Weekly), // The wait every winit loop blocks in, the loop itself through winit's // API, and an animation held to the compositor's frame events. ("toolkit_window_wake", Sched::Parallel, Tier::Nightly), @@ -1355,10 +1263,9 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // verdicts are counts the report has to agree with itself about, not a // wall-clock margin — the one duration in it is the dump's own 250 ms // ceiling, which the guest spends and the host never measures. - ("blocked_dump", Sched::Parallel, Tier::Fast), + ("blocked_dump", Sched::Parallel, Tier::Nightly), // Two boots of one machine compared on the guest's own `Boot: complete` - // with a 300 ms allowance, which is the whole assertion — a real-time - // verdict, so Nightly. + // with a 300 ms allowance, which is the whole assertion. ("i8042_absent", Sched::Serial, Tier::Nightly), // The fault quarantines (masks) the controller's GSI within milliseconds // of readiness — confirmed from the serial log, before a host round trip @@ -1368,54 +1275,39 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // timer-anchored, and its price straddles the ceiling run to run (9,355 / // 10,568 / 11,073 ms across three measurements) for exactly that reason. ("i8042_quarantine", Sched::Parallel, Tier::Nightly), - // The negative-direction half of the same gate, and the one that runs on - // every PR: no QEMU can stage a CPU into spinning through idle on - // purpose, so this is `idle_is_spinning` proving its teeth against a - // crafted trace shaped like the regression, the way - // `control_regs`/`control_regs_verdict` split the same question. - ("i8042_quarantine_verdict", Sched::Parallel, Tier::Fast), - ("i8042_budget_expiry", Sched::Parallel, Tier::Fast), - ("i8042_fadt_denial", Sched::Parallel, Tier::Fast), - ("i8042_kbd_echo", Sched::Parallel, Tier::Fast), - // Returned 2026-08-13: relegated on this branch for the same 5 s fixed - // collection deadline the rest of the family crossed on, now fixed at - // the source (`test_rs_i8042_keyboard` exits on a sentinel). + ("i8042_budget_expiry", Sched::Parallel, Tier::Nightly), + ("i8042_fadt_denial", Sched::Parallel, Tier::Weekly), + ("i8042_kbd_echo", Sched::Parallel, Tier::Nightly), ("i8042_undecoded_bytes", Sched::Parallel, Tier::Fast), // Its verdict is a cadence, and its absence is the assertion — both read // off the guest's own `last byte at Nms` stamps. The gap it injects is // 3 s against a 500 ms period, so six periods of margin decide whether the // report is on the pin or on a timer. ("i8042_health_cadence", Sched::Parallel, Tier::Nightly), - ("xhci_xecp_walk", Sched::Parallel, Tier::Fast), - ("xhci_slot_exhaustion", Sched::Parallel, Tier::Nightly), - ("usb_storage_gate", Sched::Parallel, Tier::Nightly), - ("usb_storage_shapes", Sched::Parallel, Tier::Nightly), - ("usb_refused_disk_first", Sched::Parallel, Tier::Nightly), - ("xhci_scan_hands_over_a_free_slot", Sched::Parallel, Tier::Fast), + ("xhci_xecp_walk", Sched::Parallel, Tier::Weekly), + ("xhci_slot_exhaustion", Sched::Parallel, Tier::Weekly), + ("usb_storage_gate", Sched::Parallel, Tier::Weekly), + ("usb_storage_shapes", Sched::Parallel, Tier::Weekly), + ("usb_refused_disk_first", Sched::Parallel, Tier::Weekly), + ("xhci_scan_hands_over_a_free_slot", Sched::Parallel, Tier::Nightly), // The owner's freeze, staged: `device_del` on the stick carrying `/boot` // and `/log` while the desktop draws. Serial because both verdicts are // liveness ceilings — two 2 s compositor reporting intervals inside 20 s, // and a console round trip inside 20 s — and a guest sharing the host with // eleven others answers those late for reasons that are not the defect. ("usb_boot_stick_pulled", Sched::Serial, Tier::Nightly), - ("usb_pool_exhausted", Sched::Parallel, Tier::Fast), - ("usb_short_read", Sched::Parallel, Tier::Nightly), - // A plug over QMP and two host-side verdicts, neither of them a byte - // comparison alone: the fixed 1.2 s wait against a 100 ms debounce is a - // staged latency window the LATE_READY assertion is waited out before - // being read, which is timer-anchored regardless of how comfortable the - // margin looks under TCG. - ("usb_disk_index_stable", Sched::Parallel, Tier::Nightly), - ("usb_storage_write_error", Sched::Parallel, Tier::Fast), + ("usb_pool_exhausted", Sched::Parallel, Tier::Weekly), + ("usb_short_read", Sched::Parallel, Tier::Weekly), + ("usb_storage_write_error", Sched::Parallel, Tier::Weekly), ("usb_flush_optional", Sched::Parallel, Tier::Nightly), - ("xhci_deaf_registers", Sched::Parallel, Tier::Nightly), + ("xhci_deaf_registers", Sched::Parallel, Tier::Weekly), // The window is anchored on the controller's own port-power stamp now, not // boot, so a slow boot no longer eats it — but the bound is still a fixed // span of the guest's own TSC clock (`SLOW_CONNECT_NS`/`DEBOUNCE_NS`), and // a host running several other guests can still stall this one's vCPU past // that span for reasons that are not the defect. ("xhci_slow_connect", Sched::Serial, Tier::Nightly), - ("xhci_portsc_rw1c", Sched::Parallel, Tier::Fast), + ("xhci_portsc_rw1c", Sched::Parallel, Tier::Weekly), // One staged break and no other, which puts the driver's recovery finishing // on its first try in the verdict: a retried command that reaches an // endpoint still halted from the staged break logs a second `transport @@ -1423,14 +1315,8 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // had — its own doc says one break under KVM and two under TCG off the // same tree, which is the race timer-anchored, not a margin, describes. ("usb_transport_break", Sched::Serial, Tier::Nightly), - ("xhci_full_speed_device", Sched::Parallel, Tier::Nightly), - ("xhci_superspeed_ports", Sched::Parallel, Tier::Fast), - // Two of the three below stage plug and unplug with fixed waits, 600-800 ms - // against a 100 ms debounce, plus 20-200 ms sleeps pacing the input pokes - // that follow — staged latency windows gating the verdict, so timer- - // anchored even though every individual check is a count of what the - // guest logged. - ("xhci_hotplug", Sched::Parallel, Tier::Nightly), + ("xhci_full_speed_device", Sched::Parallel, Tier::Weekly), + ("xhci_superspeed_ports", Sched::Parallel, Tier::Weekly), // `xhci_flap` is the one that genuinely races the host against the guest: // its two QMP writes have to land inside *one* 100 ms debounce or the state // under test never happens, and it says so — `no replug collapsed inside a @@ -1438,56 +1324,48 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // second write past 100 ms turns a green machine red with that sentence, // which is indistinguishable from the driver defect it hunts. ("xhci_flap", Sched::Serial, Tier::Nightly), - ("xhci_hid_break", Sched::Parallel, Tier::Nightly), - ("xhci_descriptor_walk", Sched::Parallel, Tier::Fast), + ("xhci_descriptor_walk", Sched::Parallel, Tier::Weekly), ("esp_filesystem", Sched::Parallel, Tier::Nightly), // Three boots: a budget-refused flush retried and kept, the deadman's // declared death, and a hung device's failed reset escalation — the three // exits of `object/ops.rs`'s fsync loop. Every verdict is line presence // and host-side bytes, never a wall-clock margin. ("log_flush_retry", Sched::Parallel, Tier::Nightly), - ("toybox_cp_volume", Sched::Parallel, Tier::Nightly), + ("toybox_cp_volume", Sched::Parallel, Tier::Weekly), ("kernel_log_file", Sched::Parallel, Tier::Nightly), // Serial: its verdict is a cadence — heartbeats against a 250 ms period — // and a guest sharing the host with eleven others reaches its idle loop // late for reasons that are not the defect. ("kernel_heartbeat", Sched::Serial, Tier::Nightly), - // Both own their images and their lanes, and neither verdict is a - // wall-clock margin: the guest's clock starts from an instant the host set - // and the only duration either measures is how long a boot takes to reach - // its log sink, against a bound five minutes wide. A host so loaded that - // this failed would have failed every timed test in the phase first. - ("wall_clock_file", Sched::Parallel, Tier::Fast), // The five RTC/firmware shapes, one kernel build and one boot each. Five // registrations because the artifact memo builds one kernel per feature // set anyway, so the split costs nothing and the parallel phase gets five - // jobs it can place instead of one serial five-boot job it cannot. Three - // priced near the line across two runs and sit Nightly. - ("wall_clock_rtc_dead", Sched::Parallel, Tier::Nightly), - ("wall_clock_rtc_unstable", Sched::Parallel, Tier::Fast), - ("wall_clock_no_century", Sched::Parallel, Tier::Fast), - ("wall_clock_century_register", Sched::Parallel, Tier::Nightly), - ("wall_clock_zone", Sched::Parallel, Tier::Nightly), + // jobs it can place instead of one serial five-boot job it cannot. + ("wall_clock_rtc_dead", Sched::Parallel, Tier::Weekly), + ("wall_clock_rtc_unstable", Sched::Parallel, Tier::Weekly), + ("wall_clock_no_century", Sched::Parallel, Tier::Weekly), + ("wall_clock_century_register", Sched::Parallel, Tier::Weekly), + ("wall_clock_zone", Sched::Parallel, Tier::Weekly), // `xhci_slow_connect`'s shape against the disk's port, but its actuator // masks the port until `BOOT_SCAN_DONE` — a kernel event, not a duration — // so what it stages is an ordering with no wall-clock margin on either // side: nothing here needs the serial tail. - ("late_storage_connect", Sched::Parallel, Tier::Nightly), - ("log_backing_read_error", Sched::Parallel, Tier::Fast), - ("boot_volume_metadata_error", Sched::Parallel, Tier::Fast), - ("log_partition_layout", Sched::Parallel, Tier::Fast), + ("late_storage_connect", Sched::Parallel, Tier::Weekly), + ("log_backing_read_error", Sched::Parallel, Tier::Weekly), + ("boot_volume_metadata_error", Sched::Parallel, Tier::Weekly), + ("log_partition_layout", Sched::Parallel, Tier::Weekly), // What the loader does with a slot's ROOT: bytes its signature does not // cover, a parameter naming another, an overlapping partition and an // unreadable chunk refused by name, and a twin on the boot disk or on // another disk never read. Serial, not by association: each stages a whole // boot image, one a second 32 GiB stick beside it. - ("root_candidate_malformed", Sched::Serial, Tier::Fast), - ("root_named_but_absent", Sched::Serial, Tier::Fast), - ("root_chunk_refused", Sched::Serial, Tier::Fast), - ("root_candidate_overlaps", Sched::Serial, Tier::Fast), - ("root_named_twice_on_the_boot_disk", Sched::Serial, Tier::Fast), - ("root_named_twice", Sched::Serial, Tier::Nightly), - ("log_partition_identity", Sched::Parallel, Tier::Nightly), + ("root_candidate_malformed", Sched::Serial, Tier::Weekly), + ("root_named_but_absent", Sched::Serial, Tier::Weekly), + ("root_chunk_refused", Sched::Serial, Tier::Nightly), + ("root_candidate_overlaps", Sched::Serial, Tier::Nightly), + ("root_named_twice_on_the_boot_disk", Sched::Serial, Tier::Nightly), + ("root_named_twice", Sched::Serial, Tier::Weekly), + ("log_partition_identity", Sched::Parallel, Tier::Weekly), ("cache_eviction", Sched::Parallel, Tier::Nightly), // The write-back queue's three negative controls (wall 4 of // `issues/kernel/every-wait-in-this-kernel-is-a-spin.md`). `writeback_reopen` @@ -1499,70 +1377,70 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // The watch's lost-wake window, staged: `watch-window` holds every pipe // waiter between reading its condition and parking, so the peer's post lands where // only the notified bit carries it to the commit. - ("blocking_read_window", Sched::Parallel, Tier::Fast), + ("blocking_read_window", Sched::Parallel, Tier::Nightly), // A sibling's munmap and mmap staged between a typed copy's translation // and its store (`copy-meets-a-remap`): the store never reaches the region // mapped after it. - ("user_copy_races_munmap", Sched::Parallel, Tier::Fast), + ("user_copy_races_munmap", Sched::Parallel, Tier::Nightly), // A sibling's store staged between a thread's TLS block being placed and // its rebase (`tls-rebase-window`): the block is never reachable there. - ("tls_rebase_window", Sched::Parallel, Tier::Fast), - ("writeback_reopen", Sched::Parallel, Tier::Fast), - ("writeback_spawn", Sched::Parallel, Tier::Nightly), - ("writeback_durability", Sched::Parallel, Tier::Nightly), + ("tls_rebase_window", Sched::Parallel, Tier::Nightly), + ("writeback_reopen", Sched::Parallel, Tier::Weekly), + ("writeback_spawn", Sched::Parallel, Tier::Weekly), + ("writeback_durability", Sched::Parallel, Tier::Fast), // `KernelHw::switch`'s SS reload (AMD `X86_BUG_SYSRET_SS_ATTRS`) observed the // one way a guest can, since its `SYSRET` does not reproduce the erratum. Reds // the day that `mov ss` leaves the switch. - ("sysret_ss_reload", Sched::Parallel, Tier::Fast), + ("sysret_ss_reload", Sched::Parallel, Tier::Weekly), // The FAT32 read side's revocation gate, and a host-side volume oracle for // the same reason `writeback_durability` is one: whether the clusters the // unlink freed were really reissued, and whether the cycle left a volume, are // both questions the guest that staged them cannot answer about itself. - ("fat_backing_revoked", Sched::Parallel, Tier::Nightly), + ("fat_backing_revoked", Sched::Parallel, Tier::Weekly), // F5 and F6's negative controls: an fsync that must keep refusing while the // device refuses its cache flush, and a mid-flush redirty raced for real and // re-read off the image. Both bodies in `tests/common/volumes.rs`. - ("fsync_failed_commit", Sched::Parallel, Tier::Nightly), - ("redirty_mid_flush", Sched::Parallel, Tier::Nightly), + ("fsync_failed_commit", Sched::Parallel, Tier::Weekly), + ("redirty_mid_flush", Sched::Parallel, Tier::Weekly), // A truncate staged inside a flush's metadata window, re-read off the image. - ("ftruncate_flush_race", Sched::Parallel, Tier::Nightly), + ("ftruncate_flush_race", Sched::Parallel, Tier::Weekly), // The rename gate's FAT arm, a host-side volume oracle like `fat_backing_revoked`. - ("fs_rename_durable", Sched::Parallel, Tier::Nightly), + ("fs_rename_durable", Sched::Parallel, Tier::Weekly), // The directory work's FAT arm, `fs_rename_durable`'s oracle shape. - ("fs_dirs_durable", Sched::Parallel, Tier::Fast), - ("va_exhaustion", Sched::Parallel, Tier::Fast), + ("fs_dirs_durable", Sched::Parallel, Tier::Weekly), + ("va_exhaustion", Sched::Parallel, Tier::Weekly), ("heap_ceiling_bounds", Sched::Parallel, Tier::Nightly), - ("iommu_context_absent", Sched::Parallel, Tier::Fast), - ("iommu_empty_domain", Sched::Parallel, Tier::Fast), - ("iommu_interrupt_remapping", Sched::Parallel, Tier::Fast), + ("iommu_context_absent", Sched::Parallel, Tier::Weekly), + ("iommu_empty_domain", Sched::Parallel, Tier::Weekly), + ("iommu_interrupt_remapping", Sched::Parallel, Tier::Weekly), ("iommu_virtio_platform", Sched::Parallel, Tier::Nightly), - ("iommu_domain_isolation", Sched::Parallel, Tier::Nightly), - ("iommu_gpu_scanout_swap", Sched::Parallel, Tier::Fast), - ("iommu_gpu_foreign_backing", Sched::Parallel, Tier::Fast), - ("iommu_hda_foreign_bdl", Sched::Parallel, Tier::Fast), - ("iommu_sound_foreign_dma", Sched::Parallel, Tier::Fast), + ("iommu_domain_isolation", Sched::Parallel, Tier::Weekly), + ("iommu_gpu_scanout_swap", Sched::Parallel, Tier::Nightly), + ("iommu_gpu_foreign_backing", Sched::Parallel, Tier::Weekly), + ("iommu_hda_foreign_bdl", Sched::Parallel, Tier::Weekly), + ("iommu_sound_foreign_dma", Sched::Parallel, Tier::Weekly), // The one arm of that family whose device is driven by a *process*, and // the only one whose verdict is that the machine is still running. - ("userdev_dma_fault", Sched::Parallel, Tier::Fast), + ("userdev_dma_fault", Sched::Parallel, Tier::Nightly), // Two claims of a function nothing resets, the first closed with its grant // still mapped: the verdict is the first holder's own grant, read in the // guest after the second holder wrote its own. - ("userdev_residue_is_its_own", Sched::Parallel, Tier::Fast), + ("userdev_residue_is_its_own", Sched::Parallel, Tier::Nightly), // blockd, the NVMe driver in userland, on a second controller beside the // kernel's: partitions served and timed against the kernel's driver; a // controller reset and its own death, each survived by the client and // judged off the image by the host's readers; and a transfer outside what // its function was lent, which is a fault record. Each boot runs several // blockd lifetimes and one waits out a ten-second silence. - ("blockd_serves_partitions", Sched::Parallel, Tier::Nightly), - ("blockd_survives_its_death", Sched::Parallel, Tier::Nightly), - ("blockd_dma_outside_the_lent", Sched::Parallel, Tier::Nightly), + ("blockd_serves_partitions", Sched::Parallel, Tier::Weekly), + ("blockd_survives_its_death", Sched::Parallel, Tier::Weekly), + ("blockd_dma_outside_the_lent", Sched::Parallel, Tier::Weekly), // What a claim may lend: a kernel driver's pool refused, the claim's bound // refusing the next region at the count it leaves room for, and lending and // taking back ten narrowed domains' worth of addresses with the kernel // standing; then, on a second boot, a function no release resets is never // lent where it was left aimed. - ("blockd_lends_within_its_bound", Sched::Parallel, Tier::Nightly), + ("blockd_lends_within_its_bound", Sched::Parallel, Tier::Weekly), // H4: soundd driving an Intel HDA controller itself, read back off the // device. Serial — its verdict is a wav capture, and one taken while eleven // other guests contend for the host measures the host. @@ -1571,28 +1449,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // the DMA ring takes to come round. The verdict is soundd's own liveness // and its counters rather than a capture, so it runs wide. ("hda_client_stall", Sched::Parallel, Tier::Nightly), - ("hda_two_live_refused", Sched::Parallel, Tier::Fast), - ("serial_vocabulary", Sched::Parallel, Tier::Fast), - // Host-side, no guest: the harness asking whether it can still tell a - // suspended machine from a slow one, and whether it reports one as a - // verdict it does not have. - ("suspend_detector", Sched::Parallel, Tier::Fast), - ("suspend_invalidates_a_verdict", Sched::Parallel, Tier::Fast), - ("stall_is_not_a_verdict", Sched::Parallel, Tier::Fast), - // Same: whether two guests can still be handed one lane's NVMe image, which - // is what a shared-boot reboot did to itself. - ("nvme_image_is_held_by_one_guest", Sched::Parallel, Tier::Fast), - // Same: what a whole run exits with. - ("run_exit_status", Sched::Parallel, Tier::Fast), - // Same: the control-register verdict, against the machine this tree - // actually booted before `arch/x86_64/control_regs.rs`. - ("control_regs_verdict", Sched::Parallel, Tier::Fast), - // Same: which of the two shared boots each binary belongs on, asked of the - // binaries rather than of the list that claims to name them. - ("suite_split", Sched::Parallel, Tier::Fast), - // Same: whether a run that did not attempt most of the suite's measured cost says - // so where its verdict is read. - ("nightly_tier_is_announced", Sched::Parallel, Tier::Fast), + ("hda_two_live_refused", Sched::Parallel, Tier::Weekly), ]; /// The test binaries a [`MACHINE_TESTS`] or [`SCREEN_TESTS`] entry runs, which @@ -1624,8 +1481,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("xhci_msi_only", &["test_rs_input_events"]), ("metal_sim_input", &["test_rs_input_events"]), ("xhci_flap", &["test_rs_input_events"]), - ("xhci_hotplug", &["test_rs_input_events"]), - ("xhci_hid_break", &["test_rs_input_events"]), ("nvme_large_device", &["test_rs_nvme_home_roundtrip"]), ("va_exhaustion", &["test_rs_va_exhaustion"]), ("readdir_bound", &["test_rs_readdir_bound"]), @@ -1643,7 +1498,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("i8042_kbd_echo", &["test_rs_i8042_keyboard"]), ("i8042_undecoded_bytes", &["test_rs_i8042_keyboard"]), ("i8042_quarantine", &["test_rs_i8042_keyboard"]), - ("i8042_keyboard", &["test_rs_i8042_keyboard"]), ("i8042_no_spurious_wake", &["test_rs_i8042_keyboard"]), ("i8042_mouse", &["test_rs_i8042_mouse"]), ("swiss_german_layout", &["test_rs_locale_gate"]), @@ -1707,9 +1561,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("layout_fresh_boot", &["test_rs_layout_paths"]), ("broken_data_volume_is_absent", &["test_rs_home_absent"]), ("data_candidate_with_bad_geometry_is_absent", &["test_rs_home_absent"]), - ("home_budget_refusal_retried", &["test_rs_home_fsync_budget"]), ("home_overwrite_reads_back", &["test_rs_home_overwrite_zero"]), - ("so_cache_refusals", &["test_rs_so_cache_policy"]), ("boot_volume_metadata_error", &["test_rs_boot_volume_metadata_error"]), ("esp_filesystem", &["test_rs_esp_files"]), ("log_flush_retry", &["test_rs_esp_files"]), @@ -1720,7 +1572,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("ftruncate_flush_race", &["test_rs_ftruncate_flush_race", "test_rs_fs_rename_durable"]), ("log_backing_read_error", &["test_rs_log_volume_reread"]), ("redirty_mid_flush", &["test_rs_redirty_mid_flush"]), - ("writeback_durability", &["test_rs_writeback_durability", "test_rs_fat_backing_revoked"]), + ("writeback_durability", &["test_rs_writeback_durability"]), ("kernel_log_file", &["test_rs_writeback_durability"]), ("double_fault_stack", &["test_rs_test_panic_child"]), ("idle_stack_guard", &["test_rs_test_panic_child"]), @@ -1746,13 +1598,10 @@ const CARRIES: &[(&str, &[&str])] = &[ ("log_stream", &["test_rs_log_origin", "test_rs_empty_dir_stat"]), ("log_stream_e1000e", &["test_rs_log_origin", "test_rs_empty_dir_stat"]), ("log_stream_stalled_reader", &["test_rs_log_flood"]), - ("console_line_atomicity", &["test_rs_console_line_atomicity"]), ("c_capture_ignores_daemon_lines", &["test_c_71_macro_empty_arg"]), ("quiesce_stops_the_machine", &["test_rs_quiesce_writers"]), ("quiesce_refuses_a_second_shutdown", &["test_rs_quiesce_twice"]), ("quiesce_wakes_on_the_last_park", &["test_rs_quiesce_last"]), - ("quiesce_wakes_on_the_last_exit", &["test_rs_quiesce_last"]), - ("quiesce_dump_holds_the_stopped", &["test_rs_quiesce_writers"]), ("quiesce_leaves_the_volume_whole", &["test_rs_quiesce_fsync"]), ("swap_crash_rolls_back", &["test_rs_swap_crash"]), ("swap_quiets_the_function", &["test_rs_swap_claim_idle"]), @@ -1760,7 +1609,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("swap_fault_tells_its_holder", &["test_rs_swap_claim_astray"]), ("swap_resets_the_function", &["test_rs_swap_flr_probe"]), ("swap_not_inherited", &["test_rs_swap_probe"]), - ("wall_clock_file", &["test_rs_wall_clock_now"]), ("wall_clock_rtc_dead", &["test_rs_wall_clock_now"]), ("wall_clock_rtc_unstable", &["test_rs_wall_clock_now"]), ("wall_clock_no_century", &["test_rs_wall_clock_now"]), @@ -4623,10 +4471,6 @@ fn run_screen_test( rust_bins: &[(String, Vec)], ) -> Result<(), String> { match name { - "screen_decoder" => { - screen::self_test(); - Ok(()) - } "screen_loader_lines" => { // An EXCLUSIVE open of `GraphicsOutput` calls `Stop` on the // firmware's graphics console, so with one the panel stops at the @@ -6080,39 +5924,6 @@ fn run_screen_test( } Ok(()) } - "screen_early_panic" => { - // The window the console exists for: percpu is not up, mm::init - // has not run, and on a machine with no UART nothing else can - // report at all. The alert! that is the ready marker precedes - // capture(), panic_flush() and render(), so the screen is polled. - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Gop, - qmp: true, - kernel_params: &["test-early-panic"], - ready_marker: "EARLY PANIC:", - ..Default::default() - }, - ); - let dump = qemu.screendump_until("EARLY PANIC:", Duration::from_secs(30)); - let text = dump.text(); - print_screen(name, &text); - for want in ["EARLY PANIC:", "test-early-panic: on-screen console check"] { - if !text.contains(want) { - return Err(format!("{want:?} not on screen\ndecoded screen:\n{text}")); - } - } - check_colors( - &dump, - FILL_FATAL, - &["EARLY PANIC:", "test-early-panic: on-screen console check"], - "PAT:", - )?; - Ok(()) - } "screen_late_panic" => { // The ordinary fatal panic, which no userland process can produce: // crash_report, capture, panic_flush, halt_all_cpus, render. The @@ -6931,7 +6742,7 @@ fn group_of(name: &str) -> Option<&'static str> { | "metal_sim_ipc_hostile_peer" | "metal_sim_compositor_stall" | "metal_sim_client_death" => Some(METAL_SIM_DESKTOP), - "i8042_keyboard" | "i8042_no_spurious_wake" | "i8042_mouse" => Some(I8042_TRACE), + "i8042_no_spurious_wake" | "i8042_mouse" => Some(I8042_TRACE), // The positive wizard first: it applies a layout, and the negative // member reads only its own window, so the order is the argument that // no member reads state another left. @@ -7658,171 +7469,14 @@ fn compositor_screen_size(console: &str) -> Result<(u32, u32), String> { /// injects through a fresh connection sends this after its last injection /// instead of running out the binary's fallback deadline, except /// `i8042_health_cadence` — whose verdict is a report cadence over a real span, -/// not a delivered key. The two callers that hold one connection open for the -/// whole run ([`i8042_keyboard`], [`i8042_no_spurious_wake`]) send the same two -/// transitions as the last group of their own script: a `-qmp …,server` socket +/// not a delivered key. [`i8042_no_spurious_wake`], which holds one connection +/// open for the whole run, sends the same two transitions as the last group of +/// its own script: a `-qmp …,server` socket /// serves one monitor at a time, so a second one opened here would block. fn send_i8042_sentinel(socket: &Path) { qemu::qmp_send_keys(socket, &[("end", true), ("end", false)]); } -/// What [`i8042_keyboard`] types, as the groups it may have in flight at once, -/// each with the number of `kev` lines the guest owes for it. -/// -/// **A group is the unit of pacing, and its size is bounded by -/// [`QEMU_PS2_QUEUE`].** The device holds sixteen set-1 bytes and -/// `ps2_queue()` drops the seventeenth *silently and one byte at a time*; the -/// kernel never learns of it, so its `dropped`/`lost edges`/`overruns` -/// counters all read zero on a stream with a hole in it. What a lost byte -/// costs is not one transition: a lost make leaves its break to be filtered by -/// `handle_key` — a break for a usage nothing holds queues nothing — so the -/// whole key disappears, and a lost `0xE0` leaves the break to decode as an -/// unrelated keypad code that also holds nothing, so a press survives with no -/// release. Sending the script on a wall clock and hoping the guest keeps up -/// is what put 26 bytes against those 16 and produced both of those shapes on -/// CI. The largest group below is four bytes. -const KEYBOARD_SCRIPT: &[(&[(&str, bool)], usize)] = &[ - (&[("h", true), ("h", false)], 2), - (&[("e", true), ("e", false)], 2), - (&[("l", true), ("l", false)], 2), - (&[("l", true), ("l", false)], 2), - (&[("o", true), ("o", false)], 2), - // One command, so the chord arrives as a chord rather than as a race. - (&[("shift", true), ("b", true), ("b", false), ("shift", false)], 4), - (&[("left", true), ("left", false)], 2), - (&[("esc", true), ("esc", false)], 2), - // A modifier on its own, so a stuck one is visible. - (&[("shift", true)], 1), - (&[("shift", false)], 1), - // The sentinel the guest exits on; see [`send_i8042_sentinel`]. - (&[("end", true), ("end", false)], 2), -]; - -/// A key injected at the controller, decoded, mapped and delivered to a -/// userland process — IRQ delivery, set-1 decode, the HID mapping and the -/// shared translate/layout path, in one run. -/// -/// **Paced against the guest's own report**, for [`i8042_mouse`]'s reason and -/// [`KEYBOARD_SCRIPT`]'s: a group goes out only once every `kev` line the one -/// before it owed has come back, so at most four bytes are ever outstanding at -/// a device that holds sixteen. A guest that stalls costs this test wall clock -/// and never a verdict. -fn i8042_keyboard(boot: &mut Boot) -> Result<(), String> { - let qemu = &mut boot.qemu; - let boot = qemu.boot_log().to_string(); - if !boot.contains("i8042: kbd set2+xlat (readback 0x41)") { - return Err(format!("the PS/2 keyboard never came up:\n{boot}")); - } - - let sent = std::cell::Cell::new(0usize); - let seen = std::cell::Cell::new(0usize); - let result = { - let mut input: Option = None; - qemu.run_test_paced( - "test_rs_i8042_keyboard", - Duration::from_secs(20), - |socket, line| { - if line.contains(I8042_READY) { - input = Some(qemu::QmpInput::open( - socket.expect("i8042_keyboard needs BootOptions { qmp }"), - )); - } - if line.contains("kev usage=") { - seen.set(seen.get() + 1); - } - let Some(input) = input.as_mut() else { return }; - // What everything already sent owes. Nothing new goes out - // until the guest has reported all of it, which is what bounds - // the bytes outstanding at the device to one group's worth. - let owed: usize = KEYBOARD_SCRIPT[..sent.get()].iter().map(|(_, n)| n).sum(); - if seen.get() < owed { - return; - } - if let Some((keys, _)) = KEYBOARD_SCRIPT.get(sent.get()) { - input.keys(keys); - sent.set(sent.get() + 1); - } - }, - ) - }; - let (sent, seen) = (sent.get(), seen.get()); - if let Some(err) = &result.error { - // The guard, not the verdict: under the pacing the host is *waiting* - // for the guest when this fires, so what it establishes is that the run - // stopped and never that the machine dropped a key. - let owed: usize = KEYBOARD_SCRIPT.iter().map(|(_, n)| n).sum(); - return Err(format!( - "{STALLED} {err} — {sent} of {} groups sent and {seen} of {owed} key events back \ - when the host gave up waiting for the next\n{}", - KEYBOARD_SCRIPT.len(), - result.stdout - )); - } - - let events = parse_key_events(&result.stdout); - if events.is_empty() { - return Err(format!("no key event reached userland:\n{}", result.stdout)); - } - // Presses spell the injected text: IRQ delivery, set-1 decode, - // the HID mapping, the shared translate/layout path, and arrival - // in a userland process, in one assertion. - let typed: String = events - .iter() - .filter(|e| e.modifiers & 0x10 == 0) - .map(|e| e.translated.as_str()) - .collect(); - if !typed.contains("hello") { - return Err(format!("typed {typed:?}, want it to contain \"hello\"")); - } - if !typed.contains('B') { - return Err(format!("typed {typed:?} — Shift+b did not produce a capital")); - } - if !typed.contains("\u{1b}[D") { - return Err(format!("typed {typed:?} — Left arrow produced no escape sequence")); - } - for want in [0x29u8, 0x50, 0xE1] { - if !events.iter().any(|e| e.usage == want) { - return Err(format!("no event for HID usage {want:#04x} in {events:?}")); - } - } - // Every press is matched by a release — **the sentinel's included**. `0x4D` - // is End, and the guest exits on its release, so a run that never receives - // it runs out that binary's own five-second fallback instead and every - // assertion above still passes: a green test six seconds slower than its - // price, which is what a lost sentinel used to look like. - for usage in [0x0Bu8, 0x08, 0x0F, 0x12, 0x05, 0x29, 0x50, 0xE1, 0x4D] { - let presses = events.iter().filter(|e| e.usage == usage && e.modifiers & 0x10 == 0).count(); - let releases = events.iter().filter(|e| e.usage == usage && e.modifiers & 0x10 != 0).count(); - if presses == 0 || presses != releases { - return Err(format!( - "usage {usage:#04x}: {presses} presses, {releases} releases" - )); - } - } - // Nothing is left held: the bare Shift came back up. - let last = events.last().unwrap(); - if last.modifiers & !0x10 != 0 { - return Err(format!("a modifier is stuck down: last event {last:?}")); - } - // And they came from the i8042, not from somewhere else. - let drained: usize = qemu - .boot_log() - .lines() - .chain(result.serial.lines()) - .filter_map(trace_keys) - .filter(|&k| k > 0) - .sum(); - if drained == 0 { - return Err("no i8042 drain reported a key event".to_string()); - } - eprintln!( - " [i8042] {} events to userland, {drained} from the driver; {sent} groups, none sent \ - before the one before it came back", - events.len() - ); - Ok(()) -} - /// The line `tests/toyos-rust-tests/src/bin/locale_gate.rs` prints in `layout` /// mode once the surface holds the keyboard and the wizard's child has gone — /// the moment a key injected at this machine reaches a translator, and the one @@ -7833,8 +7487,7 @@ const SWISS_READY: &str = "===SWISS_READY==="; /// once, each with the number of `kev` lines the guest owes for it. /// /// **A group is the unit of pacing, and its size is bounded by -/// [`QEMU_PS2_QUEUE`]**, for [`KEYBOARD_SCRIPT`]'s reason and by the same -/// arithmetic: the widest group here is four transitions, eight set-1 bytes even +/// [`QEMU_PS2_QUEUE`]**: the widest group here is four transitions, eight set-1 bytes even /// if every one were `0xE0`-prefixed, against a device holding sixteen. The /// whole string is far more than the queue, so a host sending it on a wall clock /// loses its tail to a guest that stops draining. @@ -7883,8 +7536,9 @@ const SWISS_SCRIPT: &[(&[(&str, bool)], usize)] = &[ (&[("end", true), ("end", false)], 2), ]; -/// No group of [`SWISS_SCRIPT`] may outrun the device queue, on the same -/// worst-case width [`KEYBOARD_SCRIPT`] is held to. +/// No group of [`SWISS_SCRIPT`] may outrun the device queue even if every +/// transition in it is an `0xE0`-prefixed two-byte one, which is the widest a +/// non-Pause set-1 transition gets. const _: () = { let mut i = 0; while i < SWISS_SCRIPT.len() { @@ -7907,8 +7561,7 @@ const _: () = { /// on the table, the modifier levels, the ISO key and the dead-key machine at /// once. /// -/// **Paced against the guest's own report**, for [`i8042_keyboard`]'s reason and -/// [`SWISS_SCRIPT`]'s: a group goes out only once every `kev` line the one +/// **Paced against the guest's own report**, for [`SWISS_SCRIPT`]'s reason: a group goes out only once every `kev` line the one /// before it owed has come back, so at most one group's bytes are ever /// outstanding at a device that holds sixteen. A guest that stalls costs this /// test wall clock and never a verdict. @@ -10683,8 +10336,7 @@ fn soundd_clients_since(log: &str, from: usize, verb: &str) -> usize { /// bytes and no events must produce no wake. Pause is that stimulus — six /// bytes, deliberately swallowed. /// -/// It drives the same in-guest reader as [`i8042_keyboard`], for the userland -/// half of the assertion. +/// It drives `test_rs_i8042_keyboard`, for the userland half of the assertion. /// /// **The zero-event drain is arranged, not hoped for.** What a drain carries is /// whatever the ISR found in the ring, so a host that injects on a wall clock @@ -10839,21 +10491,6 @@ const QEMU_PS2_QUEUE: usize = 16; /// this much of [`QEMU_PS2_QUEUE`] for it. const CHORD_BYTES: usize = 6; -/// No group of [`KEYBOARD_SCRIPT`] may outrun the device queue even if every -/// transition in it is an `0xE0`-prefixed two-byte one, which is the widest a -/// non-Pause set-1 transition gets. -const _: () = { - let mut i = 0; - while i < KEYBOARD_SCRIPT.len() { - assert!( - KEYBOARD_SCRIPT[i].0.len() * 2 <= QEMU_PS2_QUEUE, - "an i8042_keyboard group can outrun QEMU's PS/2 queue, which drops what it \ - cannot hold one byte at a time and says nothing" - ); - i += 1; - } -}; - /// A PS/2 pointer packet. Three bytes, because the driver's aux init sends no /// IntelliMouse knock and QEMU therefore frames a plain mouse. const MOUSE_PACKET: usize = 3; @@ -11451,10 +11088,6 @@ fn run_machine_test( "data_candidate_with_bad_geometry_is_absent" => { storage::data_candidate_with_bad_geometry_is_absent(test_config, c_bins, rust_bins) } - "home_budget_refusal_retried" => { - storage::home_budget_refusal_retried(test_config, c_bins, rust_bins) - } - "so_cache_refusals" => storage::so_cache_refusals(test_config, c_bins, rust_bins), "home_overwrite_reads_back" => { storage::home_overwrite_reads_back(test_config, c_bins, rust_bins) } @@ -11471,8 +11104,6 @@ fn run_machine_test( "quiesce_stops_the_machine" => power::quiesce_stops_the_machine(test_config, c_bins, rust_bins), "quiesce_refuses_a_second_shutdown" => power::quiesce_refuses_a_second_shutdown(test_config, c_bins, rust_bins), "quiesce_wakes_on_the_last_park" => power::quiesce_wakes_on_the_last_park(test_config, c_bins, rust_bins), - "quiesce_wakes_on_the_last_exit" => power::quiesce_wakes_on_the_last_exit(test_config, c_bins, rust_bins), - "quiesce_dump_holds_the_stopped" => power::quiesce_dump_holds_the_stopped(test_config, c_bins, rust_bins), "watchdog_resets" => power::watchdog_resets(test_config, c_bins, rust_bins), "watchdog_fed" => power::watchdog_fed(test_config, c_bins, rust_bins), "loader_watchdog_arms" => power::loader_watchdog_arms(test_config, c_bins, rust_bins), @@ -11526,7 +11157,6 @@ fn run_machine_test( } "usb_pool_exhausted" => usb::usb_pool_exhausted(test_config, c_bins, rust_bins), "usb_short_read" => usb::usb_short_read(test_config, c_bins, rust_bins), - "usb_disk_index_stable" => usb::usb_disk_index_stable(test_config, c_bins, rust_bins), // Body in `tests/common/volumes.rs`, same reason. "esp_filesystem" => common::volumes::esp_filesystem(test_config, c_bins, rust_bins), "log_flush_retry" => common::volumes::log_flush_retry(test_config, c_bins, rust_bins), @@ -11964,8 +11594,6 @@ fn run_machine_test( ); Ok(()) } - // Body in `tests/common/wallclock.rs`, same reason. - "wall_clock_file" => common::wallclock::wall_clock_file(test_config, c_bins, rust_bins), "wall_clock_rtc_dead" => common::wallclock::undated( test_config, c_bins, @@ -12026,21 +11654,13 @@ fn run_machine_test( usb::xhci_full_speed_device(test_config, c_bins, rust_bins) } "xhci_superspeed_ports" => usb::xhci_superspeed_ports(test_config, c_bins, rust_bins), - "xhci_hotplug" => usb::xhci_hotplug(test_config, c_bins, rust_bins), "xhci_flap" => usb::xhci_flap(test_config, c_bins, rust_bins), - "xhci_hid_break" => usb::xhci_hid_break(test_config, c_bins, rust_bins), // Body in `tests/common/iommu.rs`, same reason. "iommu_discovery" => common::iommu::iommu_discovery(test_config, c_bins, rust_bins), // Body in `tests/common/logread.rs`, so the hunk here stays one line. "log_conservation_smp1" => { common::logread::log_conservation_smp1(test_config, c_bins, rust_bins) } - "log_conservation_smp4" => { - common::logread::log_conservation_smp4(test_config, c_bins, rust_bins) - } - "log_conservation_smp8" => { - common::logread::log_conservation_smp8(test_config, c_bins, rust_bins) - } "log_nested_emit" => common::logread::log_nested_emit(test_config, c_bins, rust_bins), "log_reserve_window" => { common::logread::log_reserve_window(test_config, c_bins, rust_bins) @@ -12055,9 +11675,6 @@ fn run_machine_test( "c_capture_ignores_daemon_lines" => { common::console::c_capture_ignores_daemon_lines(test_config, c_bins, rust_bins) } - "console_line_atomicity" => { - common::console::console_line_atomicity(test_config, c_bins, rust_bins) - } "keyboard_claim_close_spares_stdin" => { common::console::keyboard_claim_close_spares_stdin(test_config, c_bins, rust_bins) } @@ -12161,9 +11778,6 @@ fn run_machine_test( boot_metal_sim_desktop(rust_bins) })) } - "i8042_keyboard" => i8042_keyboard(group_boot(held, I8042_TRACE, || { - boot_i8042_trace(test_config, c_bins, rust_bins) - })), "i8042_no_spurious_wake" => i8042_no_spurious_wake(group_boot(held, I8042_TRACE, || { boot_i8042_trace(test_config, c_bins, rust_bins) })), @@ -12980,27 +12594,6 @@ fn run_machine_test( ); Ok(()) } - // No guest: the instrument itself, in both directions. `screen_decoder` - // is the same idea for the framebuffer decoder. - // - // Three of them under one name, because they are one subject: what a - // console line says died, what a wait does about it, and the fact that - // only one place in the harness is allowed to answer either. - "serial_vocabulary" => { - serial::self_check()?; - qemu::ceiling_self_check()?; - qemu::host_scale_self_check()?; - one_vocabulary() - } - "suspend_detector" => common::clock::self_check(), - "suspend_invalidates_a_verdict" => suspend_invalidates_a_verdict(), - "stall_is_not_a_verdict" => stall_is_not_a_verdict(), - "nvme_image_is_held_by_one_guest" => nvme_image_is_held_by_one_guest(), - "run_exit_status" => run_exit_status(), - "control_regs_verdict" => control_regs_verdict(), - "i8042_quarantine_verdict" => idle_trip_verdict(), - "suite_split" => suite_split(), - "nightly_tier_is_announced" => nightly_tier_is_announced(), "nvme_wide_sector" => { // The other half of "a device's size is a shape dimension": not how // many sectors, but how big one is. `lba_ds` is an 8-bit @@ -13415,13 +13008,6 @@ fn run_machine_test( ); klogd_hosted(&serial::Serial::boot(&qemu)) } - "klogd_panic_halts" => power::klogd_death_resets( - test_config, - c_bins, - rust_bins, - &["klogd-panic", "panic-reboot-fast"], - &["klogd-panic: the console drainer died", "Process: klogd"], - ), "klogd_fault_halts" => power::klogd_death_resets( test_config, c_bins, @@ -15410,66 +14996,6 @@ fn run_machine_test( ); Ok(()) } - "sshd_fail_closed" => { - // sshd with a network under it — the only boot that gets past its - // bind. What that reaches for the first time is the daemon's own - // state on disk: the identity it mints under `/state/sshd`, and the file - // it authenticates against. - // - // The verdict is that it authenticates nobody and says which file - // left it that way. A daemon that cannot accept any key must not - // be holding port 22, so "never listened" is asserted too — that - // is the half a missing-file check would still pass without. - let config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/sshdcase"); - let options = BootOptions { - profile: qemu::Profile::Headless, - ..Default::default() - }; - if !qemu::profile_argv(&options).iter().any(|a| a.contains("virtio-net")) { - return Err("this test needs a NIC and the profile has none".to_string()); - } - - let mut qemu = QemuInstance::boot_with_options(&config, &[], &[], options); - let mut console = qemu.boot_log().to_string(); - - // Minting proves init made `/state/sshd` and set it as the daemon's - // `HOME`; the fingerprint proves the key it wrote reads back. - // - // **Both files are named, because either one alone authorizes.** - // This boot stages neither, so the daemon has to report both as - // unreadable — a check on only the writable one would pass a - // machine whose image file was silently never consulted. - const WANT: [&str; 5] = [ - "sshd: minted a new host identity at /state/sshd/host_ed25519", - "sshd: host identity SHA256:", - "sshd: cannot read /state/sshd/authorized_keys", - "sshd: cannot read /system/etc/ssh_authorized_keys", - "sshd: no file names a usable key", - ]; - let stalled = - await_guest(&mut qemu, &mut console, "every line sshd owes", |c| { - WANT.iter().all(|w| c.contains(w)) - }) - .err(); - if let Some(why) = stalled { - eprintln!(" [sshd] {why}"); - } - for want in WANT { - if !console.contains(want) { - return Err(format!("{want:?} never reached the console:\n{console}")); - } - } - if console.contains("sshd: listening on port 22") { - return Err(format!( - "sshd listened on port 22 with no key it could ever accept:\n{console}" - )); - } - eprintln!( - " [sshd] host identity minted under /state/sshd, and neither authorized_keys file \ - left it refusing to listen at all" - ); - Ok(()) - } "layout_fresh_boot" => { // The layout as ruled, on a boot of its own with a blank DATA // volume, asked over the cable because sshd is the service that @@ -17727,84 +17253,6 @@ fn wake_latency_recorded(boot: &metal::Readback) -> Result<(), String> { boot.number("latency.p99_us", u64::try_from(code).expect("a non-negative code")) } -/// [`control_regs`] against machines this host cannot boot, with no guest. -/// -/// [`control_regs_negative`] runs the real defective machine and is the link -/// between this verdict and a kernel; what is here is the states no actuator -/// reaches — a CPU that differs from three others, a bit set uniformly on all -/// four, an AP that never printed. Every value is one this tree has printed or -/// one bit away from it. -fn control_regs_verdict() -> Result<(), String> { - /// The pre-fix machine, `smp=4`, TCG, read off this tree on 2026-08-08: - /// firmware's registers on the BSP and INIT's on every AP. - const AP_BEFORE: (u64, u64) = (0xe000_0011, 0x0031_0620); - const DECLARED: (u64, u64) = (0x8001_0033, 0x0030_0668); - - fn log(cpus: &[(u64, u64)]) -> String { - cpus.iter() - .enumerate() - .map(|(i, (cr0, cr4))| { - format!("[kernel 0.1 cpu{i}] control_regs: cpu{i} cr0={cr0:#010x} cr4={cr4:#010x}\n") - }) - .collect() - } - - let refused = |what: &str, cpus: &[(u64, u64)], says: &str| match control_regs(&log(cpus), 4) { - Ok(()) => Err(format!("{what} was accepted")), - Err(e) if e.contains(says) => Ok(()), - Err(e) => Err(format!("{what} was refused for the wrong reason: {e}")), - }; - - // Positive control first: a verdict that refuses everything refuses the - // defect too, and would prove nothing below. - control_regs(&log(&[DECLARED; 4]), 4) - .map_err(|e| format!("the declared machine was refused: {e}"))?; - - refused("the machine this tree booted", &[DECLARED, AP_BEFORE, AP_BEFORE, AP_BEFORE], "CD")?; - // The case a "do all the CPUs agree?" test passes: they agree, on INIT's - // value. Nothing about uniformity says caching is on. - refused("four CPUs agreeing on INIT's CR0", &[AP_BEFORE; 4], "CD")?; - refused( - "one CPU without WP", - &[DECLARED, (DECLARED.0 & !(1 << 16), DECLARED.1), DECLARED, DECLARED], - "WP", - )?; - refused( - "one CPU without NE", - &[DECLARED, DECLARED, (DECLARED.0 & !(1 << 5), DECLARED.1), DECLARED], - "NE", - )?; - // The bit that must be *absent*: with it set, XCR0 can name components - // FXSAVE64 does not save. - refused("OSXSAVE set", &[(DECLARED.0, DECLARED.1 | (1 << 18)); 4], "OSXSAVE")?; - // Two bits a machine could hold uniformly, each one line of kernel diff - // away, and neither reachable by an actuator. `AM` is named clear above and - // answers by name; `PGE` is named nowhere, which is the case the whole - // never-named rule exists for — `TSD` and `PKE` are the same case. `UMIP` - // used to be this file's example of the same thing, until it joined - // `CR4_MAY` — a bit that moves from unnamed to optional is exactly the - // migration this gate exists to force a diff for. - refused("every CPU with AM set", &[(DECLARED.0 | (1 << 18), DECLARED.1); 4], "AM")?; - refused( - "every CPU with PGE set", - &[(DECLARED.0, DECLARED.1 | (1 << 7)); 4], - "never named", - )?; - // The bit that was on before this and asserted nowhere, so deleting `+smep` - // from the launcher or breaking the CPUID gate in `control_regs::supported` - // reddened nothing at all. - refused("every CPU without SMEP", &[(DECLARED.0, DECLARED.1 & !(1 << 20)); 4], "SMEP")?; - // A CPU that agrees about every named bit and differs in one the CPU is - // allowed to withhold, so nothing above it can object. - refused("one CPU with PCID and three without", &[DECLARED, DECLARED, DECLARED, (DECLARED.0, DECLARED.1 | (1 << 17))], "cpu3")?; - // And an AP that never printed at all, which is what a machine whose AP - // died before the check looks like. - refused("three lines for four CPUs", &[DECLARED; 3], "{0, 1, 2, 3}")?; - - eprintln!(" [control_regs] the verdict refuses 10 machines and accepts the declared one"); - Ok(()) -} - /// The negative control, executed: an AP left holding what `INIT` gave it, and /// [`control_regs`] refusing the machine that produces. /// @@ -18000,55 +17448,6 @@ fn idle_is_spinning(serial: &str) -> Option<(u32, u64)> { spread.into_iter().map(|(id, (min, max))| (id, max - min)).find(|&(_, delta)| delta > MAX_IDLE_TRIP_DELTA) } -/// [`idle_is_spinning`] against a healthy trace and a crafted one shaped like -/// the regression it exists to catch, with no guest — the same split -/// `control_regs`/`control_regs_verdict` use, and for the same reason: a -/// gate's own teeth are a claim a live boot cannot demonstrate on the -/// negative side, because nothing in this tree can stage a CPU into spinning -/// through idle on purpose. -/// -/// This is the demonstration the closed vacuous-line-count entry -/// asked for: proof the restored assertion still fails when the condition it -/// names is violated, not just that it still passes when it is not. -fn idle_trip_verdict() -> Result<(), String> { - let healthy = "\ -[kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ -[kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ -[kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=3\n\ -[kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n"; - if let Some((cpu, delta)) = idle_is_spinning(healthy) { - return Err(format!("a healthy trace was refused: cpu{cpu} moved by {delta}")); - } - - // The regression's own shape: one CPU quarantines cleanly and stays - // quiet, the other's undrained ring never lets it halt. - let spinning = "\ -[kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=1\n\ -[kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=4\n\ -[kernel 0.1 cpu0] sched: cpu=0 ready=0 dying=0 stopped=0 parked=0 current=None trips=2\n\ -[kernel 0.1 cpu1] sched: cpu=1 ready=0 dying=0 stopped=0 parked=0 current=None trips=2685004\n"; - match idle_is_spinning(spinning) { - Some((1, delta)) if delta > MAX_IDLE_TRIP_DELTA => {} - Some((cpu, delta)) => { - return Err(format!("refused the wrong CPU or by the wrong margin: cpu{cpu} delta {delta}")) - } - None => return Err("a spinning CPU's trace was accepted".to_string()), - } - - // And the line the old, count-of-lines check would have been fooled by: - // the same number of `sched: cpu=` lines either way, because the print - // itself is rate-limited regardless of what is underneath it — which is - // exactly the vacuity this replaces. - assert_eq!( - healthy.matches("sched: cpu=").count(), - spinning.matches("sched: cpu=").count(), - "the crafted traces must differ only in trips=, not in line count — otherwise this proves nothing about the old check's blindness" - ); - - eprintln!(" [i8042] the idle-trip verdict accepts a healthy trace and refuses a spinning one"); - Ok(()) -} - /// Everything the driver derived its DMA pool from, off the two lines it /// prints. Reading these is what makes a test see a *derivation* rather than /// the fact that some number was printed: every fixed cap that ever stood @@ -18266,8 +17665,7 @@ enum Poke { /// The right button, which no sequence driving that client produces for any /// other reason, and the release rather than the press so the pointer is left /// with nothing held. Every caller owes it one: without it the client waits out -/// its liveness ceiling, and `xhci_hid_break`, `xhci_hotplug` and `xhci_flap` -/// each paid 30 s for the omission. +/// its liveness ceiling. pub(crate) fn input_events_end(input: &mut qemu::QmpInput) { input.mouse(0, 0, Some(("right", true))); input.mouse(0, 0, Some(("right", false))); @@ -18602,210 +18000,6 @@ fn headline(reason: Option<&str>) -> String { reason.unwrap_or("check failed").lines().next().unwrap_or("check failed").to_string() } -/// One live guest holds its lane's NVMe image, and the next one may not. -/// -/// **The overlap this stages is the one the shared-boot reboot used to -/// produce.** `qemu = boot()` evaluates its right-hand side first, so the -/// replacement was launched while the guest it replaced still held the lane's -/// `test-nvme-*.img` open for write; QEMU's second process exited 1 on its own -/// image lock, `wait_for_ready` panicked, and the panic escaped the shared -/// block — 129 of one run's 131 reds on one sentence, 2026-08-17. -/// -/// The ordering itself is now the type's: `boot` takes a [`qemu::LaneFree`] and -/// the only thing that makes one out of a guest is `QemuInstance::shutdown`, -/// which takes it by value. What is left to check at runtime is the claim -/// underneath — that a hold is real while a guest is up and gone once it is -/// not — and this checks it on the harness's own registry, in both directions, -/// with no guest. -fn nvme_image_is_held_by_one_guest() -> Result<(), String> { - // Names, not files: a claim is a hold on a path and touches no disk, so - // nothing here has to create or delete a hundred megabytes to ask. - let dir = common::lane::dir(); - let image = dir.join("nvme-claim-gate.img"); - let other = dir.join("nvme-claim-gate-other.img"); - - let held = qemu::NvmeClaim::take(&image).map_err(|why| { - format!("a free image refused its first guest: {why}") - })?; - - // The overlap. This is the direction that must red, and it is what the - // reboot produced. - match qemu::NvmeClaim::take(&image) { - Ok(_) => { - return Err(format!( - "a second guest took {}, which a live one is holding — two QEMUs are then \ - handed one image and the second dies on its lock", - image.display() - )) - } - Err(why) => { - // The refusal has to name the image, or it cannot be acted on: a - // run makes dozens of guests and the message is all a reader gets. - if !why.contains(&image.display().to_string()) { - return Err(format!("the refusal does not name the image it is about: {why}")); - } - } - } - - // A different image is not a conflict, or every lane would refuse every - // other lane's boot the moment this gate had teeth. - let elsewhere = qemu::NvmeClaim::take(&other) - .map_err(|why| format!("an unheld image was refused: {why}"))?; - drop(elsewhere); - - // And the ordinary reboot: the replacement takes the image the guest it - // replaces released. Green, and it is the half a fix that simply refused - // every second boot would break. - drop(held); - let replacement = qemu::NvmeClaim::take(&image).map_err(|why| { - format!("a replacement was refused the image its predecessor released: {why}") - })?; - drop(replacement); - Ok(()) -} - -/// A blown guard stays red, and stops reading as an answer. -/// -/// Both halves, because each fails the other's way round. An implementation -/// that made a stall its own non-red status would hide a guest that genuinely -/// stops; one that only renamed the line would leave the summary saying a test -/// found something. Staged against the strings a wait actually produces rather -/// than against the marker on its own, because a caller prefixes its own -/// sentence to [`await_marker`]'s and the classification has to survive that. -fn stall_is_not_a_verdict() -> Result<(), String> { - // Built from the marker rather than copied, so a rename cannot leave the - // gate asserting against a string nothing produces any more. - let real = format!("{STALLED} waiting for the long tone to start — it went quiet"); - let under_a_sentence = format!("the compositor stopped painting\n{real}"); - let cases: [(&str, Option<&str>, bool); 4] = [ - ("an ordinary red", Some("the pointer never moved right"), false), - ("a wait that expired", Some(real.as_str()), true), - ( - "a wait that expired under a caller's own sentence", - Some(under_a_sentence.as_str()), - true, - ), - ("a pass", None, false), - ]; - for (what, reason, want_stall) in cases { - let outcome = Outcome { - name: what.to_string(), - reason: reason.map(str::to_string), - elapsed: Duration::from_secs(1), - suspended: Duration::ZERO, - }; - if outcome.stalled() != want_stall { - return Err(format!( - "{what} reads as stalled={}, and it has to be {want_stall}", - outcome.stalled() - )); - } - // Red is red. A stall that stopped failing the run would be a gate that - // reports and enforces nothing. - let red = outcome.verdict() == Verdict::Fail; - if red != reason.is_some() { - return Err(format!("{what} is red={red}, and a reason is always red")); - } - } - - let mut tally = Tally::new(); - tally.record(Outcome { - name: "a_stalled_test".to_string(), - reason: Some(format!("{STALLED} waiting for nothing at all — it went quiet")), - elapsed: Duration::from_secs(1), - suspended: Duration::ZERO, - }); - tally.record(Outcome { - name: "a_wrong_answer".to_string(), - reason: Some("the pointer never moved right".to_string()), - elapsed: Duration::from_secs(1), - suspended: Duration::ZERO, - }); - if tally.exit_code() != 1 { - return Err(format!("two reds exited {}, and they have to red", tally.exit_code())); - } - if tally.stalls != ["a_stalled_test"] { - return Err(format!( - "the run named {:?} as blown guards; it has to name exactly the one that was", - tally.stalls - )); - } - let summary = tally.summary(2, Duration::from_secs(2), Duration::ZERO); - if !summary.contains("1 of those reds are blown liveness guards") { - return Err(format!("the summary does not separate the two kinds of red:\n{summary}")); - } - Ok(()) -} - -/// Whether a run that did not attempt most of the suite's measured cost says so. -/// -/// **The failure mode the tier introduces is silence, not a wrong answer.** A -/// green run holding back 60 tests and a green run holding back none print the -/// same word, and the difference between them is the whole reason `--nightly` -/// exists. Nothing else can see whether the *run* mentions it, and a run nobody -/// can tell apart from a full one is how a temporary measure becomes permanent. -/// -/// Both directions, because the second is the one that rots quietly: a suite -/// that ran everything must not claim to have held anything back either, or the -/// line stops carrying information the day somebody makes it unconditional. -fn nightly_tier_is_announced() -> Result<(), String> { - let held: [&str; 2] = ["desktop_window_child", "sshd_fail_closed"]; - let announced = - Tally::new().holding_back(&held).summary(1, Duration::ZERO, Duration::ZERO); - for want in [ - "not run — the nightly tier", - "desktop_window_child, sshd_fail_closed", - "`cargo test --test toyos-build -- --nightly` runs them", - "2 held back for the nightly tier", - ] { - if !announced.contains(want) { - return Err(format!("a run holding tests back never says {want:?}:\n{announced}")); - } - } - let whole = Tally::new().summary(1, Duration::ZERO, Duration::ZERO); - if whole.contains("nightly") || whole.contains("held back") { - return Err(format!("a run that held nothing back says it did:\n{whole}")); - } - Ok(()) -} - -/// What a suspend is worth to a verdict, staged rather than reasoned about. -/// -/// `common::clock::self_check` gates the detector; this gates what the suite -/// does with what it detects. Both halves are needed and neither implies the -/// other: **a suspend that silently passes is as bad as one that silently -/// fails**, and here the two are one line apart. -fn suspend_invalidates_a_verdict() -> Result<(), String> { - let slept = common::clock::SUSPENDED_AT_LEAST + Duration::from_secs(120); - let awake = Duration::ZERO; - // Under the threshold on purpose: two clock reads jitter against each other - // by microseconds, and a run must not be thrown away for that. - let jitter = common::clock::SUSPENDED_AT_LEAST - .checked_sub(Duration::from_millis(1)) - .expect("SUSPENDED_AT_LEAST must be at least 1ms for this case to mean anything"); - let cases: [(&str, Option<&str>, Duration, Verdict); 6] = [ - ("a pass on a host that stayed up", None, awake, Verdict::Pass), - ("a fail on a host that stayed up", Some("the guest said no"), awake, Verdict::Fail), - ("a pass across a suspend", None, slept, Verdict::Invalid), - ("a fail across a suspend", Some("timed out"), slept, Verdict::Invalid), - ("a pass across clock jitter", None, jitter, Verdict::Pass), - ("a fail across clock jitter", Some("the guest said no"), jitter, Verdict::Fail), - ]; - for (what, reason, suspended, want) in cases { - let outcome = Outcome { - name: what.to_string(), - reason: reason.map(str::to_string), - elapsed: Duration::from_secs(3), - suspended, - }; - let got = outcome.verdict(); - if got != want { - return Err(format!("{what} is {got:?}, and it has to be {want:?}")); - } - } - Ok(()) -} - /// What a run has established, as it establishes it. /// /// One place rather than five counters in `main`, because the interesting part @@ -18820,11 +18014,11 @@ struct Tally { /// that never got the guest going has measured the host and not the tree. stalls: Vec, invalid: Vec<(String, Duration)>, - /// What the tier held back, by name. Not a verdict and never red — it is the - /// one thing a reader of the last line cannot infer from anything else in - /// it, because a run that skipped a third of its cost looks exactly like a - /// run that had nothing to do. - relegated: Vec<&'static str>, + /// What each tier held back, under the flag that runs it. Not a verdict and + /// never red — it is the one thing a reader of the last line cannot infer + /// from anything else in it, because a run that skipped a third of its cost + /// looks exactly like a run that had nothing to do. + relegated: Vec<(&'static str, Vec)>, } impl Tally { @@ -18839,8 +18033,12 @@ impl Tally { } /// The names this run's tier filter took out, for the summary to say so. - fn holding_back(mut self, names: &[&'static str]) -> Self { - self.relegated = names.to_vec(); + fn holding_back(mut self, held: Vec<(Tier, Vec)>) -> Self { + self.relegated = held + .into_iter() + .filter(|(_, names)| !names.is_empty()) + .map(|(tier, names)| (tier.flag().expect("only a flag's tier is held back"), names)) + .collect(); self } @@ -18935,21 +18133,21 @@ impl Tally { // Above the result line and not below it, so the pointer is the last // thing before the verdict rather than an afterthought under it. - if !self.relegated.is_empty() { - say("not run — the nightly tier:".to_string()); - say(format!(" {}", self.relegated.join(", "))); - say(" `cargo test --test toyos-build -- --nightly` runs them.".to_string()); + for (flag, names) in &self.relegated { + say(format!("not run without {flag}:")); + say(format!(" {}", names.join(", "))); + say(format!(" `cargo test --test toyos-build -- {flag}` runs them.")); say(String::new()); } // **In the result line, because that is the line a shard's job summary // extracts and the line anybody reads.** A count of what ran means // something different depending on how much was not attempted. - let held = if self.relegated.is_empty() { - String::new() - } else { - format!(", {} held back for the nightly tier", self.relegated.len()) - }; + let held: String = self + .relegated + .iter() + .map(|(flag, names)| format!(", {} held back for {flag}", names.len())) + .collect(); match self.exit_code() { 1 => say(format!( "test result: FAILED. {} passed, {} failed, {} invalidated, \ @@ -18980,313 +18178,6 @@ impl Tally { } } -/// What a whole run exits with, and what its last line says. -/// -/// Driven through [`Tally`] rather than asserted about it: the property that -/// matters is what `--land`'s gate reads off the process, and that is the exit -/// code after `record` has seen every outcome. -fn run_exit_status() -> Result<(), String> { - let outcome = |name: &str, reason: Option<&str>, suspended: Duration| Outcome { - name: name.to_string(), - reason: reason.map(str::to_string), - elapsed: Duration::from_secs(3), - suspended, - }; - let slept = common::clock::SUSPENDED_AT_LEAST + Duration::from_secs(120); - - let mut red = Tally::new(); - red.record(outcome("a_red", Some("the disk came back short"), Duration::ZERO)); - red.record(outcome("a_suspended_one", None, slept)); - if red.exit_code() != 1 { - return Err(format!("a run with a red exits {}, and it has to be 1", red.exit_code())); - } - - let mut suspended = Tally::new(); - suspended.record(outcome("a_suspended_one", None, slept)); - if suspended.exit_code() != 2 { - return Err(format!("a suspended run exits {}, and it has to be 2", suspended.exit_code())); - } - - // The clean case, so that none of the above is passing because everything - // reds. - let mut clean = Tally::new(); - clean.record(outcome("a_green", None, Duration::ZERO)); - let text = clean.summary(1, Duration::from_secs(9), Duration::ZERO); - if clean.exit_code() != 0 { - return Err(format!("a clean run exits {}, and it has to be 0", clean.exit_code())); - } - if !text.lines().last().unwrap_or_default().starts_with("test result: ok.") { - return Err(format!("a clean run does not say so plainly:\n{text}")); - } - Ok(()) -} - -/// A wait that hands a death spelling to a scan of its own, asked of the -/// harness's source. -/// -/// **The one place the vocabulary lives is the whole of the fix, so this is -/// what keeps it the one place.** `tests/common/qemu.rs` holds three waits on a -/// guest and they used to disagree: the boot half ended on three spellings, the -/// test half on one and `await_guest` on none at all — so a Rust `panic!` in the -/// kernel matched nothing while a test was running, the machine halted every -/// CPU, and the guard expired onto a verdict saying the guest had stopped -/// answering. All three ask `serial::died` now, which is the only thing in the -/// harness that knows the words and the only thing that knows the prefix decides -/// whose death they report. -/// -/// The way that comes back is the obvious patch: one more spelling handed -/// straight to a `contains` beside the call. It would match a *program's* panic -/// as readily as the kernel's and take the run down with a guest binary that -/// was expected to die — so the shape is refused by name rather than left to a -/// reviewer. Comment lines go first: this file argues about these words at -/// length, and prose is not a second answer. -fn hand_rolled_deaths(text: &str) -> Vec { - let mut found = Vec::new(); - for (n, line) in text.lines().enumerate() { - if line.trim_start().starts_with("//") { - continue; - } - for word in serial::spellings() { - // The shape is the spelling as somebody's first argument — - // `contains`, `starts_with`, `find`, any of them. A spelling - // *inside* a longer staged line is how this file's own gates build - // their inputs, and those are not scans. - if line.contains(&format!("(\"{word}")) { - found.push(format!("{}:{}: {}", n + 1, word, line.trim())); - } - } - } - found -} - -/// [`hand_rolled_deaths`] over the file that has to stay clean, with its own -/// bad input beside it so a check that stopped finding anything says so. -fn one_vocabulary() -> Result<(), String> { - const FILE: &str = "tests/common/qemu.rs"; - let path = Path::new(env!("CARGO_MANIFEST_DIR")).join(FILE); - let text = fs::read_to_string(&path).map_err(|e| format!("read {}: {e}", path.display()))?; - let found = hand_rolled_deaths(&text); - if !found.is_empty() { - return Err(format!( - "{FILE} scans for a death spelling itself, and `serial::died` is where that is \ - decided for every wait at once — a second answer here is what let a kernel panic \ - read as a stall, and it matches a program's own panic besides:\n {}", - found.join("\n ") - )); - } - // The negative control. Every line of it is a shape this must name, and the - // last two are shapes it must not: prose, and a staged capture built out of - // the same words. - let staged = "\ - } else if line.contains(\"KERNEL PANIC\") {\n\ - if line.starts_with(\"SEGFAULT\") {\n\ - // ends on `PANIC:` and nothing else, which is the defect\n\ - const KERNEL: &str = \"[kernel 1.450 cpu3] PANIC: panicked at reserve.rs:812:9:\";\n"; - let named = hand_rolled_deaths(staged); - if named.len() != 2 { - return Err(format!( - "the check names {} of the two hand-rolled scans staged for it: {named:?}", - named.len() - )); - } - eprintln!(" [vocabulary] {FILE} asks `serial::died` and nothing else"); - Ok(()) -} - -/// What the declaration itself has to be, before any of it means anything. -/// Which shared-boot binaries need `SYS_DEBUG`, asked of their source. -/// -/// A name reaches the syscall directly, or through a child it spawns. -fn needs_actuators(sources: &[(String, String)], registry: &[&str]) -> BTreeSet { - // The fourth spelling is the argument-taking form: every action that - // carries a payload (TLB_ACK_DELAY_ARM, CENSUS_KIND, LOWER_SYSINFO_BOUND, - // SLOT_TO_LAST_GENERATION) is reached through `debug_with`, never - // `debug`. The third is the SDK's: `toyos::census` calls `debug_with` on - // the caller's behalf, so a binary whose leak assertion is a census names - // no syscall of its own and reads as innocent to the others. - let calls = |text: &str| { - text.contains("SYS_DEBUG") - || text.contains("syscall::debug(") - || text.contains("syscall::debug_with(") - || text.contains("census::Census") - }; - let direct: BTreeSet<&str> = - sources.iter().filter(|(_, t)| calls(t)).map(|(n, _)| n.as_str()).collect(); - let mut out = BTreeSet::new(); - for (name, text) in sources { - if !registry.contains(&name.as_str()) { - continue; - } - let spawns = direct.iter().any(|d| text.contains(&format!("test_rs_{d}"))); - if direct.contains(name.as_str()) || spawns { - out.insert(name.clone()); - } - } - out -} - -/// [`ACTUATOR_TESTS`] is exactly the shared-boot binaries that reach -/// `SYS_DEBUG`, and the binaries are what is asked. -/// -/// **What this does not cover, stated because the hole is real:** a machine or -/// screen test that *drives* one of those binaries on a boot of its own. No -/// static rule here can say which `BootOptions` a `run_test` call belongs to. -/// What answers it instead is the guest: `test_panic_child` names -/// `InvalidArgument` as *this kernel carries no actuators* rather than reporting -/// a kernel that failed to stop, so the red says what is wrong wherever it -/// happens. -/// -/// **Both directions are the point.** A binary that gains a `debug()` call and -/// no entry would run on the shipping kernel, where the syscall answers -/// `InvalidArgument` — and a test whose verdict is that a process died would -/// then fail for a reason with nothing to do with what it is about. An entry -/// whose binary no longer calls it is a test kept off the shipping kernel for -/// nothing, which is the erosion this split exists to stop. -fn suite_split() -> Result<(), String> { - let dir = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/toyos-rust-tests/src/bin"); - let mut sources: Vec<(String, String)> = Vec::new(); - for entry in fs::read_dir(&dir).map_err(|e| format!("read {}: {e}", dir.display()))? { - let path = entry.map_err(|e| e.to_string())?.path(); - if path.extension().is_none_or(|e| e != "rs") { - continue; - } - let name = path.file_stem().unwrap().to_string_lossy().into_owned(); - let text = fs::read_to_string(&path).map_err(|e| format!("read {name}: {e}"))?; - sources.push((name, text)); - } - let registry: Vec<&str> = - sources.iter().map(|(n, _)| n.as_str()).filter(|n| !RUST_SKIP.contains(n)).collect(); - - // The negative control, and it carries its own bad input: a binary that - // calls the syscall and is on no list must be named, or the check above is - // a spelling of `true`. - let staged = vec![ - ("a_listed_one".to_string(), "syscall::debug(3)".to_string()), - ("an_unlisted_one".to_string(), "SYS_DEBUG".to_string()), - ("its_parent".to_string(), "Command::new(\"/system/bin/test_rs_an_unlisted_one\")".to_string()), - ("a_censor".to_string(), "use toyos::census::Census;".to_string()), - ("a_debug_with_user".to_string(), "syscall::debug_with(3, 4)".to_string()), - ("innocent".to_string(), "println!()".to_string()), - ]; - let staged_registry = [ - "a_listed_one", - "an_unlisted_one", - "its_parent", - "a_censor", - "a_debug_with_user", - "innocent", - ]; - let found = needs_actuators(&staged, &staged_registry); - let want: BTreeSet = - ["a_listed_one", "an_unlisted_one", "its_parent", "a_censor", "a_debug_with_user"] - .iter() - .map(|s| s.to_string()) - .collect(); - if found != want { - return Err(format!("the check does not work: on staged input it named {found:?}")); - } - - let want: BTreeSet = needs_actuators(&sources, ®istry); - let listed: BTreeSet = ACTUATOR_TESTS.iter().map(|s| s.to_string()).collect(); - let missing: Vec<&String> = want.difference(&listed).collect(); - if !missing.is_empty() { - return Err(format!( - "{missing:?} reach SYS_DEBUG and are on the shipping boot, where the syscall answers \ - InvalidArgument. Add each to ACTUATOR_TESTS, or to RUST_SKIP if it is driven rather \ - than run." - )); - } - let stale: Vec<&String> = listed.difference(&want).collect(); - if !stale.is_empty() { - return Err(format!( - "{stale:?} are held off the shipping kernel and no longer reach SYS_DEBUG. Delete \ - each entry — coverage of the binary an image ships is what it costs." - )); - } - // **The other shape `check_no_collisions` cannot see**: a binary a machine - // test drives under a *different* name is still discovered here, still runs - // on the shared boot, and there passes on its exit code with nothing staged - // for it to act on. - let staged_driven = [ - String::from("qemu.run_test(\"test_rs_a_driven_one\", Duration::from_secs(30))"), - String::from("Command::new(\"/system/bin/test_rs_another_driven\")"), - String::from("qemu.run_test(&format!(\"test_rs_{name}\"), ceiling)"), - ]; - let found = driven_binaries(&staged_driven); - let want: BTreeSet = - ["a_driven_one", "another_driven"].iter().map(|s| s.to_string()).collect(); - if found != want { - return Err(format!("the driven-name reader does not work: it named {found:?}")); - } - - let root = Path::new(env!("CARGO_MANIFEST_DIR")); - let mut harness = vec![fs::read_to_string(root.join("tests/toyos.rs")) - .map_err(|e| format!("read tests/toyos.rs: {e}"))?]; - let common = root.join("tests/common"); - for entry in fs::read_dir(&common).map_err(|e| format!("read {}: {e}", common.display()))? { - let path = entry.map_err(|e| e.to_string())?.path(); - if path.extension().is_some_and(|e| e == "rs") { - harness.push(fs::read_to_string(&path).map_err(|e| e.to_string())?); - } - } - let shared: BTreeSet<&str> = registry.iter().copied().collect(); - let both: BTreeSet = driven_binaries(&harness) - .into_iter() - .filter(|name| shared.contains(name.as_str())) - .collect(); - let declared: BTreeSet = DRIVEN_AND_SHARED.iter().map(|s| s.to_string()).collect(); - let undeclared: Vec<&String> = both.difference(&declared).collect(); - if !undeclared.is_empty() { - return Err(format!( - "{undeclared:?} are driven by a machine test and also run on the shared boot, where \ - nothing stages what they need — so each passes on its exit code with no verdict. Add \ - each to RUST_SKIP with the reason its driver exists, or to DRIVEN_AND_SHARED if its \ - shared run asserts something of its own." - )); - } - let stale: Vec<&String> = declared.difference(&both).collect(); - if !stale.is_empty() { - return Err(format!( - "DRIVEN_AND_SHARED names {stale:?}, which no machine test drives or the shared boot \ - no longer runs. Delete each entry — a declaration nothing is true of is what makes \ - the rest of the list unreadable." - )); - } - - println!( - " [split] {} shared binaries on the shipping kernel, {} on the actuator one, {} of them \ - driven elsewhere and declared", - registry.len() - listed.len(), - listed.len(), - both.len() - ); - Ok(()) -} - -/// Every guest binary the harness drives by name, read out of its own sources. -/// -/// A driver reaches a binary as the literal `test_rs_`, so that is what -/// says a binary has one; a `format!` over a variable name yields no literal -/// and is not a driver of any particular binary. Pure, and the sources are a -/// parameter, so `suite_split` stages its own before trusting it on the tree. -fn driven_binaries(sources: &[String]) -> BTreeSet { - const MARK: &str = "test_rs_"; - let mut found = BTreeSet::new(); - for text in sources { - let mut rest = text.as_str(); - while let Some(at) = rest.find(MARK) { - rest = &rest[at + MARK.len()..]; - let end = rest - .find(|c: char| !c.is_ascii_alphanumeric() && c != '_') - .unwrap_or(rest.len()); - if end > 0 { - found.insert(rest[..end].to_string()); - } - } - } - found -} - /// Which of the two shared boots a name belongs on — a *kernel build*, because /// `SYS_DEBUG` is compiled in or it is not, and never a boot parameter. fn shared_kernel(name: &str) -> &'static [&'static str] { @@ -19609,9 +18500,7 @@ fn save_durations(mut known: BTreeMap, timed: &[(String, Durat /// A phase's wall clock is `max(sum / width, longest job)`, and FIFO reaches the /// first term only if no long job is dispatched late. Declaration order puts the /// feature-carrying tests last — deliberately, to keep the kernel rebuilds -/// together — which is exactly the worst order for a wide phase: `xhci_hid_break` -/// and `xhci_deaf_registers` are two of the three longest jobs in the suite and -/// both sit in the last quarter of `MACHINE_TESTS`. +/// together — which is exactly the worst order for a wide phase. /// /// **The profile is measured, not declared**, because the alternative is a /// hand-maintained list of long tests — a second registration to keep true, and @@ -19776,9 +18665,9 @@ fn build_tasks<'a>( /// the width CI actually runs. fn check_shard_partition(all_tests: &[TestDef]) { let pricing = shard_pricing(); - for &nightly in &[false, true] { + for reach in [Reach::Fast, Reach::Nightly, Reach::Weekly] { // Sharded, because every run this partition is for is one. - let in_tier = |tier: Tier| tier.selected(nightly, true); + let in_tier = |tier: Tier| tier.selected(reach, true); let tests_to_run: Vec<&TestDef> = all_tests.iter().filter(|_| in_tier(SHARED_TIER)).collect(); let machine_to_run: Vec<(&str, Sched)> = MACHINE_TESTS @@ -19829,28 +18718,28 @@ fn check_shard_partition(all_tests: &[TestDef]) { for name in mine_p.iter().chain(&mine_s).flat_map(Task::names) { assert!( seen.insert(name.to_string()), - "nightly={nightly}: {name} lands in shard {index}/{COUNT} and at least \ + "{reach:?}: {name} lands in shard {index}/{COUNT} and at least \ one earlier shard too — every execution label must belong to exactly one" ); } for name in mine_a { assert!( audio_seen.insert(name.to_string()), - "nightly={nightly}: audio config {name} lands in shard {index}/{COUNT} \ + "{reach:?}: audio config {name} lands in shard {index}/{COUNT} \ and at least one earlier shard too" ); } } assert_eq!( seen, want, - "nightly={nightly}: the twelve shards together do not equal the full selection — \ + "{reach:?}: the twelve shards together do not equal the full selection — \ {:?} present in the selection and missing from every shard", want.difference(&seen).collect::>() ); let want_audio: BTreeSet = audio_names.iter().map(|s| s.to_string()).collect(); assert_eq!( audio_seen, want_audio, - "nightly={nightly}: the twelve shards' audio configs do not equal the full \ + "{reach:?}: the twelve shards' audio configs do not equal the full \ selection" ); } @@ -20084,16 +18973,6 @@ fn check_registration() { "CARRIES has a row for {test}, which no MACHINE_TESTS or SCREEN_TESTS entry registers" ); } - let mut seen: BTreeSet<&str> = BTreeSet::new(); - for name in MACHINE_TESTS - .iter() - .chain(SCREEN_TESTS) - .map(|(n, _, _)| *n) - .chain(AUDIO_TESTS.iter().map(|(n, _)| *n)) - { - assert!(seen.insert(name), "{name} is registered twice"); - } - let mut groups: BTreeMap<&str, (usize, usize, usize)> = BTreeMap::new(); for (i, (name, _, _)) in MACHINE_TESTS.iter().enumerate() { let Some(group) = group_of(name) else { continue }; @@ -20121,48 +19000,29 @@ fn check_registration() { } } -/// The half [`check_registration`] could not ask: the shared boot's tests are -/// *discovered* from the binaries in `tests/toyos-rust-tests` and `tests/c`, so -/// nothing declared can be compared against them until they exist. -fn check_no_collisions(shared: &[TestDef]) { - let mut shared_seen = BTreeSet::new(); - let shared_twice: Vec<&str> = shared - .iter() - .map(|test| test.name.as_str()) - .filter(|name| !shared_seen.insert(*name)) - .collect(); - assert!( - shared_twice.is_empty(), - "{shared_twice:?} name two binaries on the shared boot; every verdict and duration \ - label must identify exactly one execution" - ); - let declared: BTreeSet<&str> = MACHINE_TESTS - .iter() - .map(|(n, _, _)| *n) - .chain(SCREEN_TESTS.iter().map(|(n, _, _)| *n)) - .chain(AUDIO_TESTS.iter().map(|(name, _)| *name)) - .collect(); - let clash: Vec<&str> = - shared.iter().map(|t| t.name.as_str()).filter(|n| declared.contains(n)).collect(); - assert!( - clash.is_empty(), - "{clash:?} name both a binary on the shared boot and a test that declares its own \ - machine — two verdicts under one name. Add each to RUST_SKIP with the reason its \ - own test exists, or rename one of the two." - ); +/// Every name this suite can produce a verdict for, at its one tier: the shared +/// boot's discovered binaries and the three declared registries, so a name two +/// of them give is refused before anything boots. +fn schedule(shared: &[TestDef]) -> Schedule<'_> { + Schedule::new( + shared + .iter() + .map(|t| (t.name.as_str(), SHARED_TIER)) + .chain(AUDIO_TESTS.iter().copied()) + .chain(SCREEN_TESTS.iter().chain(MACHINE_TESTS).map(|(n, _, tier)| (*n, *tier))), + ) + .unwrap_or_else(|refusal| { + panic!( + "{refusal}. A binary a machine test drives goes on RUST_SKIP with the reason its own \ + test exists, or one of the two is renamed." + ) + }) } -/// `redlist::DISABLED` against every name `all_tests` plus the three declared -/// registries could produce a verdict for, before any boot on any entry point. -fn check_redlist(all_tests: &[TestDef]) -> Result<(), String> { - let runnable: BTreeSet<&str> = all_tests - .iter() - .map(|t| t.name.as_str()) - .chain(AUDIO_TESTS.iter().map(|(name, _)| *name)) - .chain(SCREEN_TESTS.iter().map(|(n, _, _)| *n)) - .chain(MACHINE_TESTS.iter().map(|(n, _, _)| *n)) - .collect(); - redlist::check(redlist::DISABLED, |name| runnable.contains(name), &compile::repo_root()) +/// `redlist::DISABLED` against every name the schedule holds, before any boot +/// on any entry point. +fn check_redlist(schedule: &Schedule<'_>) -> Result<(), String> { + redlist::check(redlist::DISABLED, |name| schedule.contains(name), &compile::repo_root()) } fn main() { @@ -20189,10 +19049,11 @@ fn main() { let debug_mode = SUITE.present(&args, &testargs::DEBUG); let list_mode = SUITE.present(&args, &testargs::LIST); - // The nightly tier, on. A flag and not an env var for `--audio-gate`'s - // reason: an env var is invisible in the command line and easy to leave set, - // and the whole point of the split is that a run says what it ran. - let nightly = SUITE.present(&args, &testargs::NIGHTLY); + // How far down the tiers this run reaches. Flags and not an env var for + // `--audio-gate`'s reason: an env var is invisible in the command line and + // easy to leave set, and the whole point of the split is that a run says what + // it ran. + let reach = Reach::of(&args); // The metal profile, and where its images and readbacks live. Naming the // directory means the machine is not touched — see `common::metal::Mode`. let metal_mode = SUITE.present(&args, &testargs::METAL); @@ -20263,7 +19124,7 @@ fn main() { let c_bins = compile_c_tests(&c_names); check_metal_only_unshared(&rust_bins, &c_bins); let c_compiled: Vec = c_bins.iter().map(|(n, _)| n.clone()).collect(); - if let Err(refusal) = check_redlist(&build_test_registry(&rust_bins, &c_compiled)) { + if let Err(refusal) = check_redlist(&schedule(&build_test_registry(&rust_bins, &c_compiled))) { eprintln!("[toyos] src/redlist.rs: {refusal}"); run.exit(1); } @@ -20323,24 +19184,23 @@ fn main() { // Every name this process could produce a verdict for, before `--list`, // `--debug` and `--audio-gate` can return without ever reaching it. let all_tests = build_test_registry(&rust_bins, &c_compiled); - if let Err(refusal) = check_redlist(&all_tests) { + let schedule = schedule(&all_tests); + if let Err(refusal) = check_redlist(&schedule) { eprintln!("[toyos] src/redlist.rs: {refusal}"); run.exit(1); } - // --list: print test names and exit + // Every row against the catalogue before `--list` and before any boot, so a + // name the suite did not build is refused here and not by whichever worker + // reaches it. + for (_, names) in CARRIES { + qemu::carrying(&c_bins, &rust_bins, names.iter().copied()); + } + + // --list: every test at its tier, and exit if list_mode { - for t in &all_tests { - println!("{}", t.name); - } - for (name, _) in AUDIO_TESTS { - println!("{name}"); - } - for (name, _, _) in SCREEN_TESTS { - println!("{name}"); - } - for (name, _, _) in MACHINE_TESTS { - println!("{name}"); + for (name, tier) in schedule.iter() { + println!("{tier:?} {name}"); } return; } @@ -20388,13 +19248,7 @@ fn main() { return; } - check_no_collisions(&all_tests); check_metal_only_unshared(&rust_bins, &c_bins); - // Every row against the catalogue before any boot, so a name the suite did - // not build is refused here and not by whichever worker reaches it. - for (_, names) in CARRIES { - qemu::carrying(&c_bins, &rust_bins, names.iter().copied()); - } check_shard_partition(&all_tests); // The tier filter, and it is not conditional on the name filter: a rule with @@ -20402,7 +19256,7 @@ fn main() { // nobody remembers. `cargo test -- screen_diag_boot` refuses below and // says what to type instead, which is the same information a silent skip // would have withheld. - let in_tier = |tier: Tier| tier.selected(nightly, shard.is_some()); + let in_tier = |tier: Tier| tier.selected(reach, shard.is_some()); let tests_to_run: Vec<&TestDef> = all_tests .iter() .filter(|t| keep(t.name.as_str()) && in_tier(SHARED_TIER)) @@ -20428,21 +19282,13 @@ fn main() { // introduces, so the names are printed rather than counted, and the line // carries both the command that runs them and the record that says what each // one guarded. - let held = |which: Tier| -> Vec<&str> { - MACHINE_TESTS + let held = |which: Tier| -> Vec { + schedule .iter() - .chain(SCREEN_TESTS) - .filter(|(n, _, tier)| keep(n) && *tier == which && !in_tier(*tier)) - .map(|(n, _, _)| *n) - .chain( - AUDIO_TESTS - .iter() - .filter(|(name, tier)| keep(name) && *tier == which && !in_tier(*tier)) - .map(|(name, _)| *name), - ) + .filter(|(n, tier)| keep(n) && *tier == which && !in_tier(*tier)) + .map(|(n, _)| n.to_string()) .collect() }; - let held_back = held(Tier::Nightly); let held_local = held(Tier::Local); if !held_local.is_empty() { eprintln!( @@ -20452,14 +19298,18 @@ fn main() { ); eprintln!("[toyos] {}", held_local.join(", ")); } - if !held_back.is_empty() { + let held_back: Vec<(Tier, Vec)> = + [Tier::Nightly, Tier::Weekly].into_iter().map(|tier| (tier, held(tier))).collect(); + let widest_held = held_back.iter().rev().find(|(_, names)| !names.is_empty()).map(|(tier, _)| *tier); + for (tier, names) in held_back.iter().filter(|(_, names)| !names.is_empty()) { + let flag = tier.flag().expect("a held tier is a reach's"); eprintln!( - "[toyos] nightly tier: {} test(s) NOT run. \ - `cargo test --test toyos-build -- --nightly` runs them manually; \ - .github/workflows/nightly.yml runs them every night at 03:00 UTC.", - held_back.len(), + "[toyos] {} test(s) NOT run without {flag}, which \ + `cargo test --test toyos-build -- {flag}` and .github/workflows/nightly.yml's \ + schedule for it run.", + names.len(), ); - eprintln!("[toyos] {}", held_back.join(", ")); + eprintln!("[toyos] {}", names.join(", ")); } if tests_to_run.is_empty() @@ -20467,19 +19317,19 @@ fn main() { && screen_to_run.is_empty() && machine_to_run.is_empty() { - if !held_back.is_empty() { - eprintln!( - "[toyos] filter {filter:?} matches only tests in the nightly tier. Add \ - --nightly to run them." - ); - } else { - eprintln!("No enabled test matches filter {filter:?}"); + match widest_held.and_then(Tier::flag) { + Some(flag) => eprintln!( + "[toyos] filter {filter:?} matches only tests a wider reach runs. Add {flag} to \ + run them." + ), + None => eprintln!("No enabled test matches filter {filter:?}"), } run.exit(1); } let test_config = Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/testcases"); - let mut tally = Tally::new().holding_back(&held_back); + let mut tally = Tally::new().holding_back(held_back); + let suite_start = common::clock::mark(); let bins = Bins { diff --git a/toyos-proclife/src/join.rs b/toyos-proclife/src/join.rs index 666055ba7e7..a76af7d1e47 100644 --- a/toyos-proclife/src/join.rs +++ b/toyos-proclife/src/join.rs @@ -13,6 +13,8 @@ //! is gone stays gone, and a tid this process never had never appears. //! `sys_thread_join` reads it as the terminal answer for exactly that reason. +use core::cell::Cell; + use crate::table::{Lifecycle, Processes}; use crate::{Pid, Tid}; @@ -48,9 +50,37 @@ pub fn collect_zombie( } } +/// One join's answer, kept from the ask that settled it. +/// +/// Collecting takes the zombie out of the table, so an ask after the one that +/// collected finds [`JoinRefused::NoSuchThread`]: a join whose wait asks again +/// after every wake answers with its first settled ask, never its last. The +/// answer lives here, behind `&self`, so every ask of one join reads the same +/// answer. +#[derive(Default, Debug)] +pub struct Join(Cell>>); + +impl Join { + /// Ask `collect` — [`collect_zombie`] under the table lock — unless an + /// earlier ask settled; `None` is still waiting. A settled join never calls + /// `collect`, so it never takes the table lock again. + #[must_use = "a settled join is the answer the syscall returns"] + pub fn ask( + &self, + collect: impl FnOnce() -> Result, JoinRefused>, + ) -> Option> { + if self.0.get().is_none() { + self.0.set(collect().transpose()); + } + self.0.get() + } +} + #[cfg(test)] mod tests { use super::*; + use core::cell::RefCell; + use crate::model::World; use crate::ThreadLocation; @@ -73,6 +103,48 @@ mod tests { assert_eq!(collect_zombie(&mut world, pid, t1), Err(JoinRefused::NoSuchThread)); } + #[test] + fn a_join_asked_after_it_collected_keeps_its_answer() { + let world = RefCell::new(World::new()); + let pid = world.borrow_mut().spawn_process(); + let t1 = world.borrow_mut().spawn_thread(pid); + let join = Join::default(); + let ask = || join.ask(|| collect_zombie(&mut *world.borrow_mut(), pid, t1)); + let settled = || ask().is_some(); + assert_eq!(ask(), None); + assert!(!settled()); + world.borrow_mut().set_location(pid, t1, ThreadLocation::Zombie(11)); + assert!(settled(), "the wake that found the zombie did not settle the join"); + assert_eq!( + ask(), + Some(Ok(11)), + "a join that collected its thread answered the ask after it with no such thread", + ); + let refused = Join::default(); + let ask = || refused.ask(|| collect_zombie(&mut *world.borrow_mut(), pid, Tid(9))); + assert_eq!(ask(), Some(Err(JoinRefused::NoSuchThread))); + assert_eq!(ask(), Some(Err(JoinRefused::NoSuchThread))); + } + + /// The syscall's wait asks after every wake, and `collect` is where the + /// kernel takes `PROCESS_TABLE`: a settled join answers without it. + #[test] + fn a_settled_join_does_not_take_the_table_again() { + let mut world = World::new(); + let pid = world.spawn_process(); + let t1 = world.spawn_thread(pid); + world.set_location(pid, t1, ThreadLocation::Zombie(11)); + let join = Join::default(); + assert_eq!(join.ask(|| collect_zombie(&mut world, pid, t1)), Some(Ok(11))); + assert_eq!(join.ask(|| panic!("a settled join took the table lock to ask again")), Some(Ok(11))); + let refused = Join::default(); + assert_eq!(refused.ask(|| collect_zombie(&mut world, pid, Tid(9))), Some(Err(JoinRefused::NoSuchThread))); + assert_eq!( + refused.ask(|| panic!("a refused join took the table lock to ask again")), + Some(Err(JoinRefused::NoSuchThread)), + ); + } + #[test] fn the_two_refusals_are_told_apart() { let mut world = World::new(); diff --git a/toyos-quiesce/src/lib.rs b/toyos-quiesce/src/lib.rs index 6397d07b861..2d93fe75ab4 100644 --- a/toyos-quiesce/src/lib.rs +++ b/toyos-quiesce/src/lib.rs @@ -67,10 +67,9 @@ impl Sweep { } } -/// The thread the kernel's `quiesce-last-park` and `quiesce-last-exit` -/// actuators hold, by the name its program gives it: held in its syscall -/// until it is the one thread the stop still waits on, so the transition it -/// makes next is the stop's last. +/// The thread the kernel's `quiesce-last-park` actuator holds, by the name its +/// program gives it: held in its `SYS_NANOSLEEP` until it is the one thread the +/// stop still waits on, so the park it makes next is the stop's last. pub const LAST_THREAD: &str = "quiesce-last"; /// What a line of kernel log carrying a [`Record`] begins with. diff --git a/userland/test-runner/src/main.rs b/userland/test-runner/src/main.rs index 84af3af8c79..85407cdcdf0 100644 --- a/userland/test-runner/src/main.rs +++ b/userland/test-runner/src/main.rs @@ -33,13 +33,6 @@ const BUILTINS: &[(&str, fn(Option<&SysCap>) -> i32)] = &[ ("kbd-close", kbd_close::run), ]; -/// Jobs started holding this program's console as stdin rather than a pipe: -/// a job whose subject is the console object has no other way to hold one, its -/// stdout being this program's log ring. The kernel mints the job a console of its -/// own from this one, and a job that never reads it takes none of the serial -/// commands. -const CONSOLE_JOBS: &[&str] = &["test_rs_console_line_atomicity"]; - /// The job the runner is inside, and whether the list got through: written by /// the loop, read by the deadline watching it. static RUNNING: Mutex = Mutex::new(String::new()); @@ -278,13 +271,9 @@ fn run_one(name: &str, args: &[&str], cap: Option<&SysCap>) -> Ran { return Ran::Builtin(code); } - // Piped stdin so the child does not consume the serial commands, but for - // a job in `CONSOLE_JOBS`. + // Piped stdin so the child does not consume the serial commands. let mut command = Command::new(&path); - command.args(args); - if !CONSOLE_JOBS.contains(&name) { - command.stdin(Stdio::piped()); - } + command.args(args).stdin(Stdio::piped()); // **A refused dup is an answer and not a failure — but only one // refusal is.** `duplicate` needs `DUP` on the capability, which a // manifest grants by name, so `PermissionDenied` says this cap is one