From 88643211c8c3845ca896f92d232d0f7f87600ff5 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 17:12:13 +0200 Subject: [PATCH 01/19] kernel: a kill posts its victim's retires and returns; the reaper finishes it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two processes holding each other's handle and killing at once deadlocked: each killer sat in `retire_task` on a thread that was itself inside the other kill, reached no safe point, and the kernel panicked at the 10 s tripwire (audit K1). TRANSFER makes it reachable from userland. `kill_process` now claims, posts every retire (`scheduler::post_retire`) and hands the rest to `reaper`, a kernel thread no process can kill, which waits the threads out (`scheduler::await_released`) and runs the same teardown tail as exit. The reaper stops with userland at a shutdown (`quiesce::exempt`): it runs userland's teardowns, so a stop waits for one in progress and stops it when parked, as it did the killer. A process's end is pollable: its handle's read watch is the object's watch and it is readable once the exit is published; closing one handle ends no poll. `toyos::process::Process::watch_end` is the SDK side. toyos-proclife's model now makes a retire a wait — a thread inside a scripted op reaches no safe point until it returns — adds the reaper op, a deadlock check and L6 (every claimed teardown publishes), and the KillEachOther cases. `mutate-kill-waits-for-its-victims` restores the base's inline kill and reds them. Guest tests: `mutual_kill`, and a poll arm in `process_lifecycle`. Co-Authored-By: Claude Opus 5.5 --- kernel/src/log/console.rs | 2 +- kernel/src/main.rs | 2 + kernel/src/object/ops.rs | 11 +- kernel/src/object/process.rs | 8 +- kernel/src/process.rs | 48 ++- kernel/src/quiesce.rs | 9 +- kernel/src/reaper.rs | 52 +++ kernel/src/sched/dump.rs | 2 +- kernel/src/sched/kthread.rs | 12 +- kernel/src/scheduler.rs | 28 +- kernel/src/syscall/dispatch.rs | 3 +- src/ci.rs | 4 + .../src/bin/kill_while_blocked.rs | 41 +-- tests/toyos-rust-tests/src/bin/mutual_kill.rs | 129 ++++++++ .../src/bin/process_lifecycle.rs | 33 ++ toyos-proclife/Cargo.toml | 5 + toyos-proclife/src/interleave.rs | 310 ++++++++++++++---- toyos-proclife/src/model.rs | 60 +++- toyos/src/process.rs | 8 + 19 files changed, 630 insertions(+), 137 deletions(-) create mode 100644 kernel/src/reaper.rs create mode 100644 tests/toyos-rust-tests/src/bin/mutual_kill.rs diff --git a/kernel/src/log/console.rs b/kernel/src/log/console.rs index 4e9b1a23451..d7325ed7b3a 100644 --- a/kernel/src/log/console.rs +++ b/kernel/src/log/console.rs @@ -67,7 +67,7 @@ static DRAINED: Published = Published::new(); /// Start the thread. Called once, from `kernel_main`, before the scheduler starts. /// Placement matters: APs spin until the machine is released, so an earlier spawn could not run while the machine has no console. pub fn start() { - let sched = kthread::spawn(NAME, body, 0, OnPanic::Halt); + let (_, sched) = kthread::spawn(NAME, body, 0, OnPanic::Halt); // Leaked: `klogd` never exits, and a producer reading this pointer under lock may not touch a refcount. let shared: &'static Arc = alloc::boxed::Box::leak(alloc::boxed::Box::new(sched.shared)); KLOGD.store(shared as *const _ as *mut _, Ordering::Release); diff --git a/kernel/src/main.rs b/kernel/src/main.rs index bada52f0578..8bb7a9eb9f9 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -83,6 +83,7 @@ mod clock; mod watch; mod iod; +mod reaper; mod object; mod inbox; mod pipe; @@ -669,6 +670,7 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { // After klogd so their own spawn logs have a drainer. drivers::xhci::usbd::start(); iod::start(); + reaper::start(); smp::set_ready(); diff --git a/kernel/src/object/ops.rs b/kernel/src/object/ops.rs index 3808154fb29..f06f86c101b 100644 --- a/kernel/src/object/ops.rs +++ b/kernel/src/object/ops.rs @@ -266,9 +266,11 @@ pub fn read_watch(object: &KObjectRef) -> Option { }, // Named unconditionally: the watch alone cannot enforce rights. KObjectRef::SysCap(_) => Some(WatchRef::Static(&crate::log::user::WATCH)), + // Readable once its exit is published. + KObjectRef::Process(p) => Some(WatchRef::Shared(p.watch().clone())), KObjectRef::PipeWrite(_) | KObjectRef::File(_) | KObjectRef::Inbox(_) | KObjectRef::Connector(_) | KObjectRef::Namespace(_) - | KObjectRef::SharedMem(_) | KObjectRef::Process(_) => None, + | KObjectRef::SharedMem(_) => None, } } @@ -306,10 +308,12 @@ fn close_ends_polls(object: &KObjectRef) -> bool { | device_registry::DeviceType::Framebuffer | device_registry::DeviceType::Partition => true, }, + // One handle closing ends no process. + KObjectRef::Process(_) => false, KObjectRef::PipeRead(_) | KObjectRef::PipeWrite(_) | KObjectRef::Connection(_) | KObjectRef::Acceptor(_) | KObjectRef::File(_) | KObjectRef::Inbox(_) | KObjectRef::Connector(_) | KObjectRef::Namespace(_) - | KObjectRef::SharedMem(_) | KObjectRef::Process(_) => true, + | KObjectRef::SharedMem(_) => true, } } @@ -845,9 +849,10 @@ pub fn has_data(object: &KObjectRef) -> bool { !d.info_read() || crate::drivers::virtio_sound::has_pending() } }, + KObjectRef::Process(p) => p.finished(), KObjectRef::PipeWrite(_) | KObjectRef::Inbox(_) | KObjectRef::SysCap(_) | KObjectRef::Connector(_) | KObjectRef::Namespace(_) - | KObjectRef::SharedMem(_) | KObjectRef::Process(_) => false, + | KObjectRef::SharedMem(_) => false, } } diff --git a/kernel/src/object/process.rs b/kernel/src/object/process.rs index d8438b6ca7d..e076b0308cc 100644 --- a/kernel/src/object/process.rs +++ b/kernel/src/object/process.rs @@ -29,8 +29,8 @@ pub struct ProcessObject { exit: Lock>, /// The same fact, without the lock, for a waiter's per-wake predicate. finished: AtomicBool, - /// What `SYS_PROCESS_WAIT` arms on; holding the `Arc` across the park keeps the watch from outliving its subject. - watch: Watch, + /// What `SYS_PROCESS_WAIT` and a poll arm on; holding the `Arc` across the park keeps the watch from outliving its subject. + watch: Arc, } impl ProcessObject { @@ -40,7 +40,7 @@ impl ProcessObject { pid, exit: Lock::new(None), finished: AtomicBool::new(false), - watch: Watch::new(), + watch: Arc::new(Watch::new()), }) } @@ -61,7 +61,7 @@ impl ProcessObject { self.exit.lock().as_ref().map(|e| e.stats) } - pub fn watch(&self) -> &Watch { + pub fn watch(&self) -> &Arc { &self.watch } diff --git a/kernel/src/process.rs b/kernel/src/process.rs index 072cba886a7..17d6cd9cf48 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -1045,18 +1045,22 @@ pub fn stats_from( } } +/// The scheduler records of `tids`, cloned out of the table. +fn scheds(pid: Pid, tids: Vec) -> Vec<(Tid, ThreadSched)> { + tids.into_iter().filter_map(|t| Some((t, thread_sched(pid, t)?))).collect() +} + /// Retire a set of threads, folding their scheduler accounting into the process's; returns the main thread's CPU time if it was among them (else 0). -/// `retire_task` proves each thread fully off the scheduler before returning — the ordering the whole teardown's memory-freeing safety rests on. +/// `retire` returns only once its thread is fully off the scheduler — the ordering the whole teardown's memory-freeing safety rests on. fn retire_threads( - pid: Pid, - tids: impl IntoIterator, + threads: Vec<(Tid, ThreadSched)>, main_tid: Tid, process_data_arc: &Arc>, + retire: fn(&ThreadSched), ) -> u64 { let mut main_cpu_ns = 0u64; - for t in tids { - let Some(sched) = thread_sched(pid, t) else { continue }; - scheduler::retire_task(&sched); + for (t, sched) in threads { + retire(&sched); let cpu_ns = scheduler::task_cpu_ns(&sched); let mut pdata = process_data_arc.lock(); sched.handle.merge_into(&mut pdata.accounting); @@ -1130,7 +1134,8 @@ fn release_process(code: i32) { crate::mm::paging::activate_kernel(); // Phase 2: retire every *other* thread — the current thread can't retire itself. - let mut main_cpu_ns = retire_threads(process_pid, set.others, main_tid, &process_data_arc); + let others = scheds(process_pid, set.others); + let mut main_cpu_ns = retire_threads(others, main_tid, &process_data_arc, scheduler::retire_task); // Filtered out of the retire set above, so its time is picked up here if it's the main thread. if set.current_is_main { main_cpu_ns = thread_sched(process_pid, tid) @@ -1631,11 +1636,12 @@ fn with_current_symbols(f: impl FnOnce(&crate::symbols::SymbolTable) -> bool) -> /// Kill the process an object names. /// The handle is the whole authorization, not the parent relationship: a `Process` handle carrying `Rights::MANAGE` says who may, and it can be narrowed away or handed on. `Ok` for an already-gone process: the caller asked for it to be dead and it is. +/// Returns once the victim's retires are posted; the reaper finishes the teardown, and the object's exit is published when it has. pub fn kill_process(object: &crate::object::process::ProcessObject) -> u64 { let target_pid = object.pid(); // Phase 1: claim teardown (brief table lock) - let (process_data_arc, thread_data_arc, main_tid, tids) = { + let (process_data, thread_data, main_tid, tids) = { let mut guard = PROCESS_TABLE.lock(); let table = guard.as_mut().unwrap(); @@ -1650,13 +1656,29 @@ pub fn kill_process(object: &crate::object::process::ProcessObject) -> u64 { (Arc::clone(&proc.process_data), Arc::clone(&main_thread.thread_data), proc.main_tid, tids) }; - // Phase 2: retire every thread; a running target is forced to a scheduling boundary and dropped there — never refused. - let main_cpu_ns = retire_threads(target_pid, tids, main_tid, &process_data_arc); + // Phase 2, posted and never awaited here: the victim may be killing this caller. + let threads = scheds(target_pid, tids); + for (_, sched) in &threads { + scheduler::post_retire(sched); + } + crate::reaper::owe(Killed { pid: target_pid, process_data, thread_data, main_tid, threads }); + 0 +} - // Phases 3-5: the same teardown tail as exit. - teardown_tail(&process_data_arc, &thread_data_arc, target_pid, KILLED_EXIT_CODE, main_cpu_ns); +/// A claimed kill whose every retire is posted, waiting for the reaper. +pub struct Killed { + pid: Pid, + process_data: Arc>, + thread_data: Arc>, + main_tid: Tid, + threads: Vec<(Tid, ThreadSched)>, +} - 0 +/// The reaper's half of a kill: wait out every thread, then phases 3-5, the same teardown tail as exit. +pub fn finish_kill(killed: Killed) { + let Killed { pid, process_data, thread_data, main_tid, threads } = killed; + let main_cpu_ns = retire_threads(threads, main_tid, &process_data, scheduler::await_released); + teardown_tail(&process_data, &thread_data, pid, KILLED_EXIT_CODE, main_cpu_ns); } /// The shell convention for "died on SIGKILL"; kept because every test that reads one already spells it. diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 234fa2911fe..6871fd5a859 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -101,12 +101,17 @@ fn stops(stage: u32) -> bool { // A kernel thread reaches this boundary on its first dispatch and has no // Ring 3 to be stopped from; `iod` is also what carries the sync to its // volume while userland is being stopped around it. - if crate::sched::kthread::is_kernel_task(TaskId(pid, tid)) { + if exempt(TaskId(pid, tid)) { return false; } must_stop(ThreadId { pid: pid.raw(), tid: tid.raw() }, caller()) } +/// A kernel thread is not stopped, except the reaper: the teardowns it runs are userland's. +fn exempt(id: TaskId) -> bool { + crate::sched::kthread::is_kernel_task(id) && !crate::reaper::is(id) +} + /// Whether this thread is the one shutdown this boot gets: a second caller's /// sweep would band the first where it parks inside its own sync. pub fn claim_the_shutdown() -> bool { @@ -227,7 +232,7 @@ fn sweep(caller: ThreadId) -> Sweep { if matches!(thread.state(), process::ThreadLocation::Zombie(_)) { continue; } - if crate::sched::kthread::is_kernel_task(TaskId(pid, tid)) { + if exempt(TaskId(pid, tid)) { continue; } let sched = thread diff --git a/kernel/src/reaper.rs b/kernel/src/reaper.rs new file mode 100644 index 00000000000..8d27b05991b --- /dev/null +++ b/kernel/src/reaper.rs @@ -0,0 +1,52 @@ +//! Finishes the teardowns kills hand it. +//! +//! A kill posts its victim's retires and returns, because the victim may be +//! killing the killer, and a killer that waited for its victim would wait on a +//! thread that is waiting on it. This thread does the waiting instead: no +//! process can kill it, and the only threads it waits on are killed ones, +//! whose release waits on nothing it does. It stops with userland +//! (`quiesce::exempt`), since the teardowns it runs are userland's. + +use alloc::collections::VecDeque; +use core::sync::atomic::{AtomicU64, Ordering}; + +use crate::process::{self, Killed}; +use crate::sched::kthread::{self, OnPanic}; +use crate::scheduler::{Parkable, TaskId}; +use crate::sync::Lock; +use crate::watch::{self, Watch}; + +const NAME: &str = "reaper"; + +static OWED: Lock> = Lock::new(VecDeque::new()); + +static WORK: Watch = Watch::new(); + +/// The reaper's packed `TaskId`, stored before the machine is released. +static REAPER: AtomicU64 = AtomicU64::new(u64::MAX); + +/// Spawns the reaper; call once, from `kernel_main`. +pub fn start() { + // Halts: a dead reaper leaves every later kill unpublished and nothing to say so. + let (id, _) = kthread::spawn(NAME, body, 0, OnPanic::Halt); + REAPER.store(id.pack(), Ordering::Release); +} + +pub fn is(id: TaskId) -> bool { + REAPER.load(Ordering::Acquire) == id.pack() +} + +/// Hand a claimed kill, its retires already posted, to the reaper. +pub fn owe(killed: Killed) { + OWED.lock().push_back(killed); + WORK.post(); +} + +extern "C" fn body(_arg: u64) -> ! { + let parkable = Parkable::at_entry(); + loop { + watch::wait_uncancellable_until(&parkable, &WORK, 0, || !OWED.lock().is_empty()); + let killed = OWED.lock().pop_front().expect("the one consumer saw an entry"); + process::finish_kill(killed); + } +} diff --git a/kernel/src/sched/dump.rs b/kernel/src/sched/dump.rs index 10cdc57dab1..3a9a85273ce 100644 --- a/kernel/src/sched/dump.rs +++ b/kernel/src/sched/dump.rs @@ -705,7 +705,7 @@ fn census() -> Census { // Blocked and running threads are already the CPUs' lines; skip them here. let Some(tag) = tag else { return }; // Kernel threads don't count against the budget: `MAX_KERNEL_TASKS` - // bounds them at three, so counting them can't push these lines off the page. + // bounds them, so counting them can't push these lines off the page. if !kernel { printed += 1; if printed > CENSUS_LINES { diff --git a/kernel/src/sched/kthread.rs b/kernel/src/sched/kthread.rs index 2faf66b2ca9..8e400597571 100644 --- a/kernel/src/sched/kthread.rs +++ b/kernel/src/sched/kthread.rs @@ -19,11 +19,11 @@ use crate::sync::Lock; use super::payload::ThreadSched; -/// `klogd`, `usbd` and `iod`, plus one `log-storm` thread per shard in the actuator build. +/// `klogd`, `usbd`, `iod` and `reaper`, plus one `log-storm` thread per shard in the actuator build. #[cfg(not(feature = "boot-actuators"))] -const MAX_KERNEL_TASKS: usize = 3; +const MAX_KERNEL_TASKS: usize = 4; #[cfg(feature = "boot-actuators")] -const MAX_KERNEL_TASKS: usize = 3 + toyos_abi::log::MAX_LOG_SHARDS; +const MAX_KERNEL_TASKS: usize = 4 + toyos_abi::log::MAX_LOG_SHARDS; /// Collides with no packed id: neither id map issues `u32::MAX`. const NO_TASK: u64 = u64::MAX; @@ -114,8 +114,8 @@ pub fn panic_recovers_here() -> Option { Some(current_row()?.recoverable.load(Ordering::Relaxed) != 0) } -/// Start a kernel thread running `body(arg)` on its own kernel stack and return its scheduler faces. -pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64, on_panic: OnPanic) -> ThreadSched { +/// Start a kernel thread running `body(arg)` on its own kernel stack and return its identity and scheduler faces. +pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64, on_panic: OnPanic) -> (TaskId, ThreadSched) { let (stack, entry_rsp) = crate::loader::alloc_kernel_stack( crate::loader::kernel_start, body as usize as u64, @@ -174,7 +174,7 @@ pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64, on_panic: OnPa OnPanic::Recover => "kills the thread", } ); - sched + (TaskId(pid, tid), sched) } /// Every field is the empty value: a kernel thread has no user half. diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index ec0f4c03c26..c6347273242 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -501,14 +501,31 @@ pub fn futex_wake(phys_addr: DirectMap, count: usize) -> u64 { futex::watch_of(phys_addr).post_n(phys_addr.phys(), count) as u64 } -/// Retire a thread and wait until its record — kernel stack and -/// address-space reference — is released. The state word reading `Dead` is -/// not enough: that payload is freed by the pass after the one that publishes it. +/// Retire a thread and wait until it is released. #[track_caller] pub fn retire_task(sched: &ThreadSched) { + post_retire(sched); + await_released(sched); +} + +/// Set a thread's kill bit and ask its CPU for a safe point; returns at once. +pub fn post_retire(sched: &ThreadSched) { + if sched.handle.released() { + return; + } + preempt_off(|p| { + toyos_sched::retire::begin(&sched.shared).post(cpus(), &HW, p); + }); +} + +/// Wait until a retired thread's record — kernel stack and address-space +/// reference — is released. The state word reading `Dead` is not enough: that +/// payload is freed by the pass after the one that publishes it. +#[track_caller] +pub fn await_released(sched: &ThreadSched) { // Also on the early-return path below, where no park happens and the two // asserts inside the wait would never run. - assert_baseline(BASELINE_TRAP); + assert_baseline(blocking_baseline()); if let (Some(pid), Some(tid)) = (percpu::current_pid(), percpu::current_tid()) { if let Some(handle) = driver::current_shared() { assert!( @@ -521,9 +538,6 @@ pub fn retire_task(sched: &ThreadSched) { if sched.handle.released() { return; } - preempt_off(|p| { - toyos_sched::retire::begin(&sched.shared).post(cpus(), &HW, p); - }); /// Re-poll rate for the liveness backstop; the release wake is what /// actually ends the wait. const RECHECK: Cadence = Cadence::every( diff --git a/kernel/src/syscall/dispatch.rs b/kernel/src/syscall/dispatch.rs index b70b3206b39..06a68fad7a1 100644 --- a/kernel/src/syscall/dispatch.rs +++ b/kernel/src/syscall/dispatch.rs @@ -240,8 +240,7 @@ pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> .get::(RawHandle(a1 as u32), Rights::MANAGE) }) { Ok(object) => { - // Killing yourself is exiting, not killing: kill_process retires every - // thread of its target, and retire_task asserts a CPU never retires itself. + // Killing yourself is exiting, not killing. // Reachable: TRANSFER lets a parent hand a child a handle to itself. if object.pid() == process::current_process() { // exit() never returns, so the held clone is dropped while it still can be. diff --git a/src/ci.rs b/src/ci.rs index fd9512444cf..13b0c0e5a6c 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -363,6 +363,10 @@ pub(crate) const CONTROLS: &[Control] = &[ red(PROCLIFE, "mutate-claim-teardown-always-wins", None, &[ "an_exit_and_a_kill_never_both_tear_a_process_down ... FAILED", ]), + red(PROCLIFE, "mutate-kill-waits-for-its-victims", None, &[ + "two_processes_killing_each_other_both_end ... FAILED", + "a_kill_chain_of_three_ends ... FAILED", + ]), red(SCHED_SIM, "placement-ignores-staleness", Some("policy"), &[ "a_stopped_cpu_stops_taking_work ... FAILED", ]), diff --git a/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs b/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs index 650d9e6ece8..c035f6bde2d 100644 --- a/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs +++ b/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs @@ -25,9 +25,7 @@ //! and no syscall to cancel. The other three ask what a kill *releases*; //! this one asks whether it **ends** — "a killed thread is never dispatched //! into userland again", the claim about a kill that no other test in this -//! tree executes. It is the one arm that does not issue -//! its own kill: a `kill` that does not end its target does not return -//! either, so the killer is a process of its own and this one only watches. +//! tree executes. //! //! **No census assertion anywhere in this file, and that is the point rather //! than an omission.** A live-object count is the one instrument that cannot @@ -79,10 +77,7 @@ const VICTIM_LABEL: &str = "victim"; /// window's several scheduler dispatches sit on top of. /// /// **It has to be smaller than `scheduler::retire_task`'s own tripwire, and -/// that ordering is the whole of the arm's ability to report.** The process -/// doing the killing is parked inside that tripwire for as long as the victim -/// stays in userland, and when it blows the kernel panics and takes the machine -/// with it — so an arm that had not spoken by then never speaks. Nothing here +/// that ordering is the whole of the arm's ability to report.** Nothing here /// reads that constant or restates its value: what this one needs is only to be /// first, and a bound derived from device timeouts and scheduler quanta is not /// going to come in under two seconds. @@ -222,29 +217,11 @@ fn an_acceptor_killed_in_the_accept() { /// all. With either miss the child here is preempted, queued in the dying list, /// picked straight back off it and returned to Ring 3, once per tick, forever. /// -/// **Three processes, because the killer cannot report.** `Process::kill` is -/// `scheduler::retire_task`, which parks until the victim's record is released -/// and panics the kernel at its own tripwire when it never is. On exactly the -/// tree this arm exists to catch, then, the call does not come back: an arm -/// that killed and then timed its own `kill` would be *inside* the panic it is -/// meant to name, and the version of this arm that did so could only ever have -/// reported on a tree that was already fixed. So the kill is a child of its -/// own, and the deadline below is held by a process with nothing of the kernel -/// under it. -/// /// **What it watches is the victim's own stdout.** That pipe's write end is in /// the victim's handle table and in no other, so the read end this process /// holds reaches EOF when — and only when — the victim's handles are drained. -/// That is `kill_process`'s phase 3, which runs after `retire_task` has seen -/// the victim released, which the victim publishes from its own pass out of -/// `leave_ring3_if_due` — the last exit boundary itself. EOF is downstream of the -/// boundary by a teardown and by nothing that can wait, and it is unreachable -/// without it. /// -/// **The clock starts before the kill and not after it**, which is the second -/// half of what was wrong here: the previous shape took its `Instant` once -/// `kill()` had already returned, and a returned `kill` has already published -/// the exit — so what it timed was the reap of a zombie and never the boundary. +/// **The clock starts before the kill and not after it.** /// What this one bounds is the whole of [go byte → boundary → teardown], so the /// number is an upper bound on the boundary rather than a measurement of it. fn a_ring_three_spinner_ends_at_its_next_exit_boundary() { @@ -281,16 +258,11 @@ fn a_ring_three_spinner_ends_at_its_next_exit_boundary() { println!( "a child killed while spinning in Ring 3 still held its handles {:?} later — \ it is being re-dispatched into userland, so nothing on the return path reads \ - the kill bit, and the process that killed it is parked in retire_task with \ - nothing it can say", + the kill bit", started.elapsed(), ); std::process::exit(1); } - // Bounded by the line above and not a second deadline: EOF is phase 3, and - // phases 4 and 5 behind it are straight-line kernel code with no wait in - // them, so a killer that produced the EOF has already returned from its - // `kill`. What this adds is the killer's own verdict on that call. assert!( killer.wait().expect("wait for the killer").success(), "the spinner ended, but the process that killed it did not agree", @@ -315,11 +287,6 @@ fn child(role: &str) -> ! { std::hint::spin_loop(); } } - // Arm 4's killer. It is a process rather than a thread because - // `Process::kill` does not return on the tree that arm is about — it - // is `retire_task`, which parks until the victim's record is released - // and panics the kernel at its own tripwire when it never is. Nothing - // the observer does is behind this call. "kill" => { let victim: Process = endow::Endowments::get() .take(VICTIM_LABEL) diff --git a/tests/toyos-rust-tests/src/bin/mutual_kill.rs b/tests/toyos-rust-tests/src/bin/mutual_kill.rs new file mode 100644 index 00000000000..d66f4a9e4f3 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/mutual_kill.rs @@ -0,0 +1,129 @@ +//! Two processes, each holding the other's handle, kill each other at once, and +//! both end with the kernel still up. +//! +//! Each round the parent hands each child a handle to the other and one shared +//! word; both spin on the word, the parent sets it, and both call +//! `SYS_PROCESS_KILL` inside the same few microseconds. A kill that waited for +//! its victim's threads to leave every CPU waited on a thread that was inside +//! the other kill, and the kernel panicked at `retire_task`'s tripwire. + +use std::io::{Read, Write}; +use std::os::toyos::process::{ChildExt, CommandExt}; +use std::process::{Child, Command, Stdio}; +use std::sync::atomic::{AtomicU32, Ordering}; + +use toyos::ipc::Connection; +use toyos::process::Process; +use toyos::shm::SharedMemory; +use toyos::{endow, namespace, port, AsHandle}; +use toyos_abi::syscall::{self, SVC_LABEL}; +use toyos_abi::RawHandle; + +const SELF_PATH: &str = "/system/bin/test_rs_mutual_kill"; + +/// The name the child's namespace carries the connector under. +const SERVICE: &str = "mutual"; + +const ROUNDS: usize = 64; + +/// The frame that says the batch — the shared word, then the victim — is queued. +const ARM: u32 = 1; + +const WORD_BYTES: usize = 4096; + +/// `process::KILLED_EXIT_CODE`. +const KILLED: i32 = 137; + +fn main() { + match std::env::args().nth(1).as_deref() { + Some("killer") => killer(), + Some(other) => panic!("unknown role {other:?}"), + None => test(), + } +} + +fn test() { + for round in 0..ROUNDS { + let go = SharedMemory::create(WORD_BYTES).expect("a shared word"); + let (a_conn, mut a) = killer_child(); + let (b_conn, mut b) = killer_child(); + let a_handle = RawHandle(a.as_raw_handle()); + let b_handle = RawHandle(b.as_raw_handle()); + arm(&a_conn, &go, b_handle); + arm(&b_conn, &go, a_handle); + armed(&mut a); + armed(&mut b); + + word(&go).store(1, Ordering::Release); + + let codes = [a.wait().expect("wait a").code(), b.wait().expect("wait b").code()]; + assert!( + codes.iter().all(|c| *c == Some(KILLED) || *c == Some(0)) + && codes.contains(&Some(KILLED)), + "round {round}: two processes that killed each other ended with {codes:?}", + ); + } + println!("mutual_kill: {ROUNDS} rounds of two processes killing each other, every one ended"); +} + +/// Spawn a killer holding a connector for a fresh port, and accept it. +fn killer_child() -> (Connection, Child) { + let (acceptor, connector) = port::create().expect("a port of our own"); + let ns = namespace::build() + .add(SERVICE, &connector) + .finish() + .expect("a namespace carrying one connector"); + let child = Command::new(SELF_PATH) + .arg("killer") + .endow(SVC_LABEL, ns.into_raw().0) + .stdout(Stdio::piped()) + .spawn() + .expect("spawn a killer"); + let conn = acceptor.accept().expect("the killer connected"); + (conn, child) +} + +fn arm(conn: &Connection, go: &SharedMemory, victim: RawHandle) { + let victim = syscall::dup(victim).expect("a handle to send"); + syscall::handle_send(conn.as_handle(), &[go.share().expect("the word to send"), victim]) + .expect("send the batch"); + conn.signal(ARM).expect("say the batch is queued"); +} + +/// Wait for the killer's one line: it holds its victim and is spinning. +fn armed(child: &mut Child) { + let mut line = String::new(); + let out = child.stdout.as_mut().expect("killer stdout"); + let mut byte = [0u8; 1]; + while out.read(&mut byte).expect("read the killer's marker") == 1 && byte[0] != b'\n' { + line.push(byte[0] as char); + } + assert_eq!(line, "armed", "the killer never armed"); +} + +fn word(go: &SharedMemory) -> &AtomicU32 { + // SAFETY: the region is mapped for as long as `go` lives and 2 MiB-aligned, + // and every access to this word, here and in the peer, is atomic. + unsafe { AtomicU32::from_ptr(go.as_ptr().cast()) } +} + +fn killer() -> ! { + let ns = endow::namespace().expect("the parent endowed a namespace"); + let conn = ns.open(SERVICE).expect("connect through the endowed connector"); + let header = conn.recv_header().expect("the arming frame"); + assert_eq!(header.msg_type, ARM, "an unexpected frame"); + let [word_handle, victim] = conn.recv_handles_exact::<2>().expect("the word and the victim"); + let go = SharedMemory::adopt(word_handle, WORD_BYTES).expect("map the word"); + // SAFETY: the parent sent a handle to the other killer, which nothing else here owns. + let victim = unsafe { Process::from_raw(victim) }; + + let mut out = std::io::stdout(); + out.write_all(b"armed\n").expect("say armed"); + out.flush().expect("flush"); + + while word(&go).load(Ordering::Acquire) == 0 { + std::hint::spin_loop(); + } + victim.kill().expect("kill the other killer"); + std::process::exit(0); +} diff --git a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs index 717686fb9a7..45bb0ef0010 100644 --- a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs +++ b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs @@ -32,6 +32,7 @@ use std::sync::OnceLock; use std::time::{Duration, Instant}; use toyos::endow::{Endowments, SYSCAP_LABEL}; +use toyos::poller::Poller; use toyos::AsHandle; use toyos::process::Process; use toyos::syscap::SysCap; @@ -62,6 +63,7 @@ fn test() { an_unrelated_wake_does_not_end_the_wait(); two_handles_answer_the_same(); a_kill_publishes_like_an_exit(); + its_end_completes_a_poll(); a_handle_is_the_whole_of_the_right(); a_pid_is_not_authority(); an_undefined_wait_flag_bit_is_refused(); @@ -284,6 +286,37 @@ fn a_kill_publishes_like_an_exit() { println!(" a kill publishes {KILLED} once, and a second kill changes nothing"); } +/// A process's end is an event: its handle turns readable when the exit is +/// published, never before, and a poll registered after the end completes at +/// once. +fn its_end_completes_a_poll() { + const END: u64 = 1; + const LATE: u64 = 2; + let (mut child, release) = start(4); + let handle = syscall::dup(RawHandle(child.as_raw_handle())).expect("a handle to poll"); + // SAFETY: the duplicate is this function's alone. + let process = unsafe { Process::from_raw(handle) }; + let poller = Poller::new(1); + + process.watch_end(&poller, END); + let mut early = Vec::new(); + poller.wait(0, 0, |token| early.push(token)); + assert!(early.is_empty(), "a running process's handle completed a poll: {early:?}"); + + drop(release); + let mut ended = Vec::new(); + poller.wait(1, u64::MAX, |token| ended.push(token)); + assert_eq!(ended, [END], "the end completed no poll"); + assert_eq!(process.try_wait(), Ok(4), "the poll completed before the exit was published"); + + process.watch_end(&poller, LATE); + let mut late = Vec::new(); + poller.wait(1, u64::MAX, |token| late.push(token)); + assert_eq!(late, [LATE], "a poll on an ended process did not complete"); + assert_eq!(child.wait().expect("wait").code(), Some(4)); + println!(" a process's end completes a poll on its handle, and not before"); +} + /// **The arm the pid-keyed shape could not have.** The waiter did not spawn the /// subject, is not its parent by any spelling, and holds nothing but a handle /// somebody moved into its table — and that is enough. diff --git a/toyos-proclife/Cargo.toml b/toyos-proclife/Cargo.toml index 9e5fca535d2..1e3fc6d4bad 100644 --- a/toyos-proclife/Cargo.toml +++ b/toyos-proclife/Cargo.toml @@ -43,6 +43,11 @@ mutate-claim-teardown-always-wins = [] # under the very lock the insert takes. `interleave::tests::a_published_exit_ # leaves_no_unretired_thread` must red under this. mutate-spawn-skips-the-insert-recheck = [] +# The model's kill finishes the teardown itself, retiring each victim thread +# and waiting for it, as `process::kill_process` did before the reaper. +# `interleave::tests::two_processes_killing_each_other_both_end` must red under +# this: each killer waits on a thread that is inside the other's kill. +mutate-kill-waits-for-its-victims = [] [dependencies] toyos-abi = { path = "../toyos-abi" } diff --git a/toyos-proclife/src/interleave.rs b/toyos-proclife/src/interleave.rs index 9eca5031a0b..232773f7bdd 100644 --- a/toyos-proclife/src/interleave.rs +++ b/toyos-proclife/src/interleave.rs @@ -35,8 +35,12 @@ pub enum Op { /// `process::release_process` + `teardown_tail`: a thread ending its own /// process. Exit { pid: Pid, tid: Tid, code: i32, pc: u32, others: Vec, next: usize }, - /// `process::kill_process`: a handle holder ending somebody else's. - Kill { pid: Pid, code: i32, pc: u32, tids: Vec, next: usize }, + /// `process::kill_process`: a handle holder ending somebody else's. `by` is + /// the killing thread when the model holds it; `inline` is the base's kill, + /// which finished the teardown itself. + Kill { pid: Pid, code: i32, pc: u32, tids: Vec, by: Option<(Pid, Tid)>, inline: Option }, + /// The reaper kthread, finishing what kills handed it. + Reap { job: Option }, /// `process::spawn_thread`: two lock sections with the whole of a thread /// built between them; `block` is the mapped TLS the build carries across. Spawn { pid: Pid, pc: u32, block: Option }, @@ -48,12 +52,85 @@ pub enum Op { IdlePass { pc: u32 }, } +/// A claimed kill's teardown past its claim: wait out every thread's release, +/// mark, publish. +#[derive(Clone, Debug)] +pub struct Finish { + pid: Pid, + code: i32, + tids: Vec, + next: usize, + pc: u32, +} + +impl Finish { + fn new((pid, code, tids): (Pid, i32, Vec)) -> Self { + Finish { pid, code, tids, next: 0, pc: 0 } + } + + fn enabled(&self, world: &World) -> bool { + self.pc != 0 || self.next == self.tids.len() || retire_ready(world, self.pid, self.tids[self.next]) + } + + /// One section; `true` once the exit is published. + fn step(&mut self, world: &mut World) -> bool { + match self.pc { + 0 => { + if self.next < self.tids.len() { + if retire_step(world, self.pid, self.tids[self.next]) { + self.next += 1; + } + } else { + self.pc = 1; + } + false + } + 1 => { + if let Some(proc) = world.get_mut(self.pid) { + teardown::mark_all_zombie(proc, self.code); + } + self.pc = 2; + false + } + _ => { + world.publish_exit(self.pid, self.code); + true + } + } + } +} + +/// `retire_task`, one section: post the retire, then — blocked until then — +/// find the thread released. `true` once it is. +fn retire_step(world: &mut World, pid: Pid, tid: Tid) -> bool { + if world.is_retired(pid, tid) { + return true; + } + if !world.is_killed(pid, tid) { + world.post_retire(pid, tid); + } + false +} + +/// Whether [`retire_step`] can move: nothing posted yet, or the thread gone. +fn retire_ready(world: &World, pid: Pid, tid: Tid) -> bool { + !world.is_killed(pid, tid) || world.is_retired(pid, tid) +} + impl Op { pub fn exit(pid: Pid, tid: Tid, code: i32) -> Self { Op::Exit { pid, tid, code, pc: 0, others: Vec::new(), next: 0 } } pub fn kill(pid: Pid, code: i32) -> Self { - Op::Kill { pid, code, pc: 0, tids: Vec::new(), next: 0 } + Op::Kill { pid, code, pc: 0, tids: Vec::new(), by: None, inline: None } + } + /// A kill issued by a thread the model holds, which is in the kernel until + /// the kill returns. + pub fn kill_by(pid: Pid, code: i32, by: (Pid, Tid)) -> Self { + Op::Kill { pid, code, pc: 0, tids: Vec::new(), by: Some(by), inline: None } + } + pub fn reap() -> Self { + Op::Reap { job: None } } pub fn spawn(pid: Pid) -> Self { Op::Spawn { pid, pc: 0, block: None } @@ -68,20 +145,37 @@ impl Op { Op::IdlePass { pc: 0 } } - fn done(&self) -> bool { - let (Op::Exit { pc, .. } - | Op::Kill { pc, .. } - | Op::Spawn { pc, .. } - | Op::ThreadExit { pc, .. } - | Op::Join { pc, .. } - | Op::IdlePass { pc, .. }) = self; - *pc == DONE + /// Whether the op has nothing left to do: the reaper is done whenever + /// nothing is owed to it. + fn done(&self, world: &World) -> bool { + match self { + Op::Reap { job } => job.is_none() && world.nothing_owed(), + Op::Exit { pc, .. } + | Op::Kill { pc, .. } + | Op::Spawn { pc, .. } + | Op::ThreadExit { pc, .. } + | Op::Join { pc, .. } + | Op::IdlePass { pc, .. } => *pc == DONE, + } + } + + /// Whether the op's next section can run, rather than wait. + fn enabled(&self, world: &World) -> bool { + match self { + Op::Exit { pid, pc: 1, others, next, .. } => { + *next == others.len() || retire_ready(world, *pid, others[*next]) + } + Op::Kill { inline: Some(finish), .. } => finish.enabled(world), + Op::Reap { job: Some(finish) } => finish.enabled(world), + _ => true, + } } fn label(&self) -> &'static str { match self { Op::Exit { .. } => "exit", Op::Kill { .. } => "kill", + Op::Reap { .. } => "reaper", Op::Spawn { .. } => "spawn_thread", Op::ThreadExit { .. } => "thread_exit", Op::Join { .. } => "thread_join", @@ -116,8 +210,9 @@ impl Op { // the table lock given up. 1 => { if *next < others.len() { - world.retire(*pid, others[*next]); - *next += 1; + if retire_step(world, *pid, others[*next]) { + *next += 1; + } } else { *pc = 2; } @@ -143,32 +238,43 @@ impl Op { } } } - Op::Kill { pid, code, pc, tids, next } => match *pc { - 0 => { - if !teardown::claim_teardown(world, *pid) { + Op::Kill { pid, code, pc, tids, by, inline } => { + if let Some(finish) = inline { + if finish.step(world) { *pc = DONE; - return; } - *tids = teardown::kill_set(world.get(*pid).expect("just claimed")); - *pc = 1; - } - 1 => { - if *next < tids.len() { - world.retire(*pid, tids[*next]); - *next += 1; + } else if *pc == 0 { + if !teardown::claim_teardown(world, *pid) { + *pc = DONE; } else { - *pc = 2; + *tids = teardown::kill_set(world.get(*pid).expect("just claimed")); + if cfg!(feature = "mutate-kill-waits-for-its-victims") { + *inline = Some(Finish::new((*pid, *code, tids.clone()))); + } else { + *pc = 1; + } } + } else { + // Every retire posted, the wait handed on, and the kill + // returns. + for &tid in tids.iter() { + world.post_retire(*pid, tid); + } + world.owe(*pid, *code, core::mem::take(tids)); + *pc = DONE; } - 2 => { - if let Some(proc) = world.get_mut(*pid) { - teardown::mark_all_zombie(proc, *code); + if *pc == DONE { + if let Some(by) = *by { + world.leave_kernel(by); } - *pc = 3; } - _ => { - world.publish_exit(*pid, *code); - *pc = DONE; + } + Op::Reap { job } => match job { + None => *job = Some(Finish::new(world.take_owed().expect("stepped with nothing owed"))), + Some(finish) => { + if finish.step(world) { + *job = None; + } } }, Op::Spawn { pid, pc, block } => match *pc { @@ -258,25 +364,39 @@ impl Op { const DONE: u32 = u32::MAX; -/// The first schedule that breaks a law, or `None`. +/// How many schedules ran, or the first one that breaks a law. /// /// Depth-first over "which op runs its next lock section", checking -/// `World::faults` at every state and `World::final_faults` at every leaf. The -/// returned string is the schedule that produced it, in the order the ops ran. -pub fn explore(initial: &World, ops: &[Op]) -> Option { +/// `World::faults` at every state and `World::final_faults` at every leaf; a +/// state where no unfinished op can move is a deadlock. The returned string is +/// the schedule that produced it, in the order the ops ran. +pub fn explore(initial: &World, ops: &[Op]) -> Result { + let mut world = initial.clone(); + for op in ops { + if let Op::Kill { by: Some(by), .. } = op { + world.enter_kernel(*by); + } + } let mut trace = Vec::new(); - walk(initial.clone(), ops.to_vec(), &mut trace) + let mut schedules = 0; + match walk(world, ops.to_vec(), &mut trace, &mut schedules) { + Some(found) => Err(found), + None => Ok(schedules), + } } -fn walk(world: World, ops: Vec, trace: &mut Vec) -> Option { - if ops.iter().all(Op::done) { +fn walk(world: World, ops: Vec, trace: &mut Vec, schedules: &mut u64) -> Option { + if ops.iter().all(|op| op.done(&world)) { + *schedules += 1; let faults = world.final_faults(); return report(&faults, trace); } + let mut moved = false; for i in 0..ops.len() { - if ops[i].done() { + if ops[i].done(&world) || !ops[i].enabled(&world) { continue; } + moved = true; let mut next_world = world.clone(); let mut next_ops = ops.clone(); let before = alloc::format!("{}#{i}", next_ops[i].label()); @@ -287,12 +407,17 @@ fn walk(world: World, ops: Vec, trace: &mut Vec) -> Option { trace.pop(); return Some(found); } - if let Some(found) = walk(next_world, next_ops, trace) { + if let Some(found) = walk(next_world, next_ops, trace, schedules) { trace.pop(); return Some(found); } trace.pop(); } + if !moved { + let stuck: Vec<&str> = + ops.iter().filter(|op| !op.done(&world)).map(Op::label).collect(); + return report(&[alloc::format!("deadlock: {} each wait and none can move", stuck.join(", "))], trace); + } None } @@ -328,7 +453,7 @@ mod tests { ); let ops = vec![Op::exit(pid, main, 0), Op::spawn(pid)]; - if let Some(found) = explore(&world, &ops) { + if let Err(found) = explore(&world, &ops) { panic!("a lifecycle law broke:\n{found}"); } } @@ -339,8 +464,8 @@ mod tests { fn a_kill_racing_a_spawn_leaves_no_unretired_thread() { let mut world = World::new(); let pid = world.spawn_process(); - let ops = vec![Op::kill(pid, 137), Op::spawn(pid)]; - if let Some(found) = explore(&world, &ops) { + let ops = vec![Op::kill(pid, 137), Op::spawn(pid), Op::reap()]; + if let Err(found) = explore(&world, &ops) { panic!("a lifecycle law broke:\n{found}"); } } @@ -360,8 +485,8 @@ mod tests { let pid = world.spawn_process(); let main = world.main_tid(pid); world.spawn_thread(pid); - let ops = vec![Op::exit(pid, main, 0), Op::kill(pid, 137)]; - if let Some(found) = explore(&world, &ops) { + let ops = vec![Op::exit(pid, main, 0), Op::kill(pid, 137), Op::reap()]; + if let Err(found) = explore(&world, &ops) { panic!("a lifecycle law broke:\n{found}"); } } @@ -377,7 +502,7 @@ mod tests { let waiter = world.spawn_thread(pid); let dying = world.spawn_thread(pid); let ops = vec![Op::thread_exit(pid, dying, 3), Op::join(pid, dying, waiter)]; - if let Some(found) = explore(&world, &ops) { + if let Err(found) = explore(&world, &ops) { panic!("a lifecycle law broke:\n{found}"); } } @@ -399,7 +524,7 @@ mod tests { Op::idle_pass(), Op::join(pid, dying, joiner), ]; - if let Some(found) = explore(&world, &ops) { + if let Err(found) = explore(&world, &ops) { panic!("a lifecycle law broke:\n{found}"); } } @@ -412,8 +537,8 @@ mod tests { fn a_spawn_racing_a_kill_and_the_pass_that_reaps_it() { let mut world = World::new(); let pid = world.spawn_process(); - let ops = vec![Op::kill(pid, 137), Op::spawn(pid), Op::idle_pass()]; - if let Some(found) = explore(&world, &ops) { + let ops = vec![Op::kill(pid, 137), Op::spawn(pid), Op::idle_pass(), Op::reap()]; + if let Err(found) = explore(&world, &ops) { panic!("a lifecycle law broke:\n{found}"); } } @@ -434,7 +559,7 @@ mod tests { Op::thread_exit(pid, sibling, 3), Op::spawn(pid), ]; - if let Some(found) = explore(&world, &ops) { + if let Err(found) = explore(&world, &ops) { panic!("a lifecycle law broke:\n{found}"); } } @@ -451,7 +576,7 @@ mod tests { let pid = world.spawn_process(); let main = world.main_tid(pid); let ops = vec![Op::exit(pid, main, 0), Op::spawn(pid), Op::spawn(pid)]; - if let Some(found) = explore(&world, &ops) { + if let Err(found) = explore(&world, &ops) { panic!("a lifecycle law broke:\n{found}"); } } @@ -470,7 +595,7 @@ mod tests { spawning.step(&mut world); // phase 2: the block is mapped let mut exit = Op::exit(pid, main, 0); - while !exit.done() { + while !exit.done(&world) { exit.step(&mut world); } assert!( @@ -495,8 +620,8 @@ mod tests { let pid = world.spawn_process(); let target = world.spawn_thread(pid); let waiter = world.spawn_thread(pid); - let ops = vec![Op::kill(pid, 137), Op::join(pid, target, waiter)]; - if let Some(found) = explore(&world, &ops) { + let ops = vec![Op::kill(pid, 137), Op::join(pid, target, waiter), Op::reap()]; + if let Err(found) = explore(&world, &ops) { panic!("a lifecycle law broke:\n{found}"); } } @@ -510,12 +635,17 @@ mod tests { let pid = world.spawn_process(); let sibling = world.spawn_thread(pid); - // The schedule, run by hand rather than searched for: a kill that runs - // to its publish, an idle pass, and only then the sibling's own exit. + // The schedule, run by hand rather than searched for: a kill and the + // reaper that runs it to its publish, an idle pass, and only then the + // sibling's own exit. let mut kill = Op::kill(pid, 137); - while !kill.done() { + while !kill.done(&world) { kill.step(&mut world); } + let mut reaper = Op::reap(); + while !reaper.done(&world) { + reaper.step(&mut world); + } let mut idle = Op::idle_pass(); idle.step(&mut world); assert!(world.was_reaped(pid), "the idle pass took the published entry"); @@ -532,13 +662,73 @@ mod tests { let mut exit = Op::thread_exit(pid, sibling, 0); exit.step(&mut world); assert!( - !exit.done(), + !exit.done(&world), "the exit ended at its routing section: a thread whose entry went under it \ still has to post and retire, and one that stops here is the machine \ stopping with it", ); - while !exit.done() { + while !exit.done(&world) { exit.step(&mut world); } } + + /// **`KillEachOther`, K1's shape**: two processes, each one's thread inside + /// `SYS_PROCESS_KILL` on the other. A thread in the kernel reaches no safe + /// point until its kill returns, so a kill that waited for its victim waits + /// on a thread that is waiting on it. + /// + /// Reds under `mutate-kill-waits-for-its-victims`, the base's kill. + #[test] + fn two_processes_killing_each_other_both_end() { + let mut world = World::new(); + let p = world.spawn_process(); + let c = world.spawn_process(); + let ops = vec![ + Op::kill_by(c, 137, (p, world.main_tid(p))), + Op::kill_by(p, 137, (c, world.main_tid(c))), + Op::reap(), + ]; + match explore(&world, &ops) { + Ok(schedules) => std::println!("KillEachOther: {schedules} schedules, every one ends"), + Err(found) => panic!("a lifecycle law broke:\n{found}"), + } + } + + /// The cycle at length three: A kills B, B kills C, C kills A. + #[test] + fn a_kill_chain_of_three_ends() { + let mut world = World::new(); + let a = world.spawn_process(); + let b = world.spawn_process(); + let c = world.spawn_process(); + let ops = vec![ + Op::kill_by(b, 137, (a, world.main_tid(a))), + Op::kill_by(c, 137, (b, world.main_tid(b))), + Op::kill_by(a, 137, (c, world.main_tid(c))), + Op::reap(), + ]; + if let Err(found) = explore(&world, &ops) { + panic!("a lifecycle law broke:\n{found}"); + } + } + + /// A sibling in the cycle and an exit racing it: P's main thread exits + /// while P's second thread kills C and C kills P — so the exit's own + /// retire waits on a thread that is inside a kill. + #[test] + fn an_exit_whose_sibling_kills_the_process_killing_it() { + let mut world = World::new(); + let p = world.spawn_process(); + let killer = world.spawn_thread(p); + let c = world.spawn_process(); + let ops = vec![ + Op::exit(p, world.main_tid(p), 0), + Op::kill_by(c, 137, (p, killer)), + Op::kill_by(p, 137, (c, world.main_tid(c))), + Op::reap(), + ]; + if let Err(found) = explore(&world, &ops) { + panic!("a lifecycle law broke:\n{found}"); + } + } } diff --git a/toyos-proclife/src/model.rs b/toyos-proclife/src/model.rs index cd22e5a0232..dc4a6edea83 100644 --- a/toyos-proclife/src/model.rs +++ b/toyos-proclife/src/model.rs @@ -12,7 +12,7 @@ //! depends on the order, and a model whose counter-example is different every //! run is a model nobody can bisect. -use alloc::collections::{BTreeMap, BTreeSet}; +use alloc::collections::{BTreeMap, BTreeSet, VecDeque}; use alloc::string::String; use alloc::vec::Vec; @@ -79,6 +79,12 @@ pub struct World { /// pass: a claimant inside its own teardown, which cannot retire itself and /// will not reach Ring 3 again either. leaving: BTreeSet<(Pid, Tid)>, + /// Threads a retire was posted on: the kill bit. + killed: BTreeSet<(Pid, Tid)>, + /// Threads inside a scripted operation, which reach no safe point until it returns. + in_kernel: BTreeSet<(Pid, Tid)>, + /// Teardowns a kill handed to the reaper, oldest first: the process, its code, its threads. + owed: VecDeque<(Pid, i32, Vec)>, /// Entries an idle pass has taken out of the table. reaped: BTreeSet, /// TLS blocks a spawn's phase 2 mapped and no thread owns yet. @@ -115,6 +121,9 @@ impl World { released: BTreeSet::new(), retired: BTreeSet::new(), leaving: BTreeSet::new(), + killed: BTreeSet::new(), + in_kernel: BTreeSet::new(), + owed: VecDeque::new(), reaped: BTreeSet::new(), tls_mapped: BTreeSet::new(), next_tls: 0, @@ -236,6 +245,49 @@ impl World { self.retired.contains(&(pid, tid)) } + /// `scheduler::post_retire`: the kill bit, and a thread at no safe point + /// dies only when its own operation returns. + pub fn post_retire(&mut self, pid: Pid, tid: Tid) { + if self.is_retired(pid, tid) { + return; + } + assert!(self.killed.insert((pid, tid)), "a second retirer for pid {pid} tid {tid}"); + if !self.in_kernel.contains(&(pid, tid)) { + self.retire(pid, tid); + } + } + + pub fn is_killed(&self, pid: Pid, tid: Tid) -> bool { + self.killed.contains(&(pid, tid)) + } + + /// A thread starting a scripted operation. + pub fn enter_kernel(&mut self, by: (Pid, Tid)) { + self.in_kernel.insert(by); + } + + /// Its operation returned: the safe point, where a killed thread dies. + pub fn leave_kernel(&mut self, by: (Pid, Tid)) { + self.in_kernel.remove(&by); + if self.killed.contains(&by) { + self.retire(by.0, by.1); + } + } + + /// `reaper::owe`. + pub fn owe(&mut self, pid: Pid, code: i32, tids: Vec) { + self.owed.push_back((pid, code, tids)); + } + + /// The reaper's pop. + pub fn take_owed(&mut self) -> Option<(Pid, i32, Vec)> { + self.owed.pop_front() + } + + pub fn nothing_owed(&self) -> bool { + self.owed.is_empty() + } + /// Whether this thread can still execute user code. fn runnable(&self, pid: Pid, tid: Tid) -> bool { !self.retired.contains(&(pid, tid)) @@ -311,8 +363,14 @@ impl World { /// to its end. A join that has not been answered *yet* is the ordinary /// case, so checking it at every state would report every schedule. /// And **L5** — every TLS block a spawn mapped ends owned or released. + /// And **L6** — every claimed teardown ends in a published exit. pub fn final_faults(&self) -> Vec { let mut out = self.faults(); + for (&pid, proc) in &self.procs { + if proc.claims > 0 && !self.published.contains_key(&pid) { + out.push(alloc::format!("pid {pid} was claimed for teardown and never published an exit")); + } + } for &block in &self.tls_mapped { out.push(alloc::format!( "TLS block {block} is mapped and owned by nobody — a refused spawn dropped it \ diff --git a/toyos/src/process.rs b/toyos/src/process.rs index bd807e146ca..c1c6510bf76 100644 --- a/toyos/src/process.rs +++ b/toyos/src/process.rs @@ -10,6 +10,7 @@ use toyos_abi::handle::Rights; use toyos_abi::syscall::{self, ProcessStats, SyscallError}; use crate::endow::FromHandle; +use crate::poller::{Poller, READABLE}; use crate::{AsHandle, OwnedHandle, RawHandle}; pub struct Process(pub(crate) OwnedHandle); @@ -30,10 +31,17 @@ impl Process { } /// Kill it. `Ok` for one already dead: the caller asked for it to be gone. + /// Returns before it is gone; its end is what [`wait`](Self::wait) and + /// [`watch_end`](Self::watch_end) report. pub fn kill(&self) -> Result<(), SyscallError> { syscall::process_kill(self.0.raw()) } + /// Complete `token` on `poller` once its exit is published. + pub fn watch_end(&self, poller: &Poller, token: u64) { + poller.watch(self, READABLE, token); + } + pub fn stats(&self) -> Result { let mut stats = ProcessStats::default(); syscall::process_stats(self.0.raw(), &mut stats)?; From b88bf98770920188eef642085b6abe8d9c405497 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 17:23:27 +0200 Subject: [PATCH 02/19] proclife: KillEachOther gives each process a second thread, and a sibling is a killer The audit's K1 names the sibling form too (A2 kills B while B1 kills A). 19 schedules, every one ends; red under mutate-kill-waits-for-its-victims. Co-Authored-By: Claude Opus 5.5 --- toyos-proclife/src/interleave.rs | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/toyos-proclife/src/interleave.rs b/toyos-proclife/src/interleave.rs index 232773f7bdd..2939755faf5 100644 --- a/toyos-proclife/src/interleave.rs +++ b/toyos-proclife/src/interleave.rs @@ -675,16 +675,19 @@ mod tests { /// **`KillEachOther`, K1's shape**: two processes, each one's thread inside /// `SYS_PROCESS_KILL` on the other. A thread in the kernel reaches no safe /// point until its kill returns, so a kill that waited for its victim waits - /// on a thread that is waiting on it. + /// on a thread that is waiting on it. Each process has a second thread, and + /// one of the killers is it. /// /// Reds under `mutate-kill-waits-for-its-victims`, the base's kill. #[test] fn two_processes_killing_each_other_both_end() { let mut world = World::new(); let p = world.spawn_process(); + let p_sibling = world.spawn_thread(p); let c = world.spawn_process(); + world.spawn_thread(c); let ops = vec![ - Op::kill_by(c, 137, (p, world.main_tid(p))), + Op::kill_by(c, 137, (p, p_sibling)), Op::kill_by(p, 137, (c, world.main_tid(c))), Op::reap(), ]; From db57dea2043c94fcdc672b50223bd79f51c27118 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 17:32:07 +0200 Subject: [PATCH 03/19] tests: quiesce_stops_the_machine counts the reaper among the threads the stop names The reaper stops with userland (quiesce::exempt), so the stop's exact thread count is one higher. Measured: 11 of 11 before this, and green after. Co-Authored-By: Claude Opus 5.5 --- tests/common/power.rs | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/tests/common/power.rs b/tests/common/power.rs index 198952e0448..4793e3311fc 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -172,6 +172,8 @@ pub fn quiesce_stops_the_machine( // thread, parked on init's answer; `test-runner`'s main and deadline // threads; and `logd`'s. `init` asked for the stop and is its caller. const OTHERS: u32 = 4; + // The kernel's `reaper`, which stops with userland because it runs userland's kills. + const REAPER: u32 = 1; /// Mirrored in `kernel/src/syscall/machine.rs`, which queues it. const QUEUED: &str = "console: a holder's line, queued once the stop had stopped every holder"; let (whole, record) = stopped_boot( @@ -213,13 +215,13 @@ pub fn quiesce_stops_the_machine( // harness is told, or lost one to an I/O error before the reset, has fewer // threads to stop and would pass every judge above over a machine that was // not the one described. - if record.sweep.total() != WRITERS + OTHERS { + if record.sweep.total() != WRITERS + OTHERS + REAPER { return Err(format!( "this boot's stop named {} userland thread(s); {WRITERS} writers plus the {OTHERS} \ - of the job, test-runner and logd make {}, so this is not the machine the writers \ - were on:\n {record}\n{whole}", + of the job, test-runner and logd and the reaper make {}, so this is not the machine \ + the writers were on:\n {record}\n{whole}", record.sweep.total(), - WRITERS + OTHERS, + WRITERS + OTHERS + REAPER, )); } woken_by_its_threads(&record)?; From 1ae6558544b49fe4e0cffaf7ec646eaaa4806a8d Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 18:35:54 +0200 Subject: [PATCH 04/19] kernel: the reaper takes whichever kill is released, and the teardown asserts it Answers the review of #549 at 61c0a7f3. The reaper no longer pops its queue in order and waits for the head. Every thread release now also posts `sched::payload::ANY_RELEASED`, and the reaper waits on that (or on its own `WORK` when nothing is owed). It takes the oldest owed kill whose threads are all released, so a slow victim no longer delays another kill's release. The tripwire is per kill and counts from when the kill was owed (`scheduler::RETIRE_GIVE_UP`, the same 10 s). The teardowns still run one at a time on one thread. That is filed as issues/kernel/the-reaper-runs-every-kills-teardown-one-at-a-time.md. `fold_retired` (was `retire_threads`) asserts that every thread it folds is released before any teardown frees what they ran on. Both paths wait before they reach it: exit through `await_released`, and the reaper through its choice. Mutation: the reaper takes a kill whose threads are not all released (`killed.released() || true`). mutual_kill then goes red with "teardown: a thread of the process is still on the scheduler", both wide and alone. Deleted: - `scheduler::retire_task`. The exit path posts every sibling's retire first, then awaits each one, so siblings die concurrently. - `retire_threads`' `retire` parameter. - `reaper::REAPER`/`is` and the `kthread::spawn` return change. A kernel thread now declares `OnStop::Runs` or `OnStop::Stops` in its row at spawn, and `quiesce` reads the row (`kthread::runs_through_the_stop`). - The self-kill branch in dispatch. A self-kill posts its own retire, dies at its Ring 3 boundary, and the reaper publishes 137. mutual_kill now checks that. Also: - `block::counted` reads the same row, so a block operation the reaper opens is counted as the stop's. - The stop record's thread count loses "userland": it now counts the reaper too. - `Shootdown::serve` raises `flushed` with `fetch_max`, so a serve nested inside another cannot move it backwards (audit K2, now reachable from the reaper's IF=1 teardown). kernel-loom's `a_nested_serve_is_not_undone_by_the_one_it_interrupted` reds under `store`. - process_lifecycle: closing another handle to a running process leaves a poll on it empty. It reds under `KObjectRef::Process(_) => true`. - kill_while_blocked arm 4 kills from the observer. The separate killer existed only because a kill used to wait. - The model's reaper takes a released kill, and its exit posts every retire before it waits. - Stale `retire_task` citations in toyos-sched follow the rename. The sim paragraph that said the kernel never batches one process's retires is deleted, because it now does. Filed: issues/kernel/a-manage-holder-can-kill-a-kernel-thread.md. Co-Authored-By: Claude Opus 5.5 --- ...-manage-holder-can-kill-a-kernel-thread.md | 25 +++++++ ...runs-every-kills-teardown-one-at-a-time.md | 32 ++++++++ kernel-loom/tests/tlb_shootdown.rs | 26 +++++++ kernel/src/block.rs | 7 +- kernel/src/drivers/xhci/usbd.rs | 4 +- kernel/src/iod.rs | 4 +- kernel/src/log/console.rs | 4 +- kernel/src/log/nested.rs | 4 +- kernel/src/log/storm.rs | 4 +- kernel/src/object/process.rs | 1 - kernel/src/process.rs | 39 +++++++--- kernel/src/quiesce.rs | 16 +--- kernel/src/reaper.rs | 75 +++++++++++++------ kernel/src/sched/kthread.rs | 45 ++++++++--- kernel/src/sched/payload.rs | 4 + kernel/src/scheduler.rs | 37 ++++----- kernel/src/shootdown.rs | 3 +- kernel/src/syscall/dispatch.rs | 11 +-- tests/common/metal.rs | 2 +- tests/common/power.rs | 17 ++--- .../src/bin/kill_while_blocked.rs | 44 ++--------- tests/toyos-rust-tests/src/bin/mutual_kill.rs | 16 +++- .../src/bin/process_lifecycle.rs | 3 + toyos-proclife/src/interleave.rs | 38 ++++++---- toyos-proclife/src/model.rs | 21 ++++-- toyos-proclife/src/teardown.rs | 2 +- toyos-quiesce/src/lib.rs | 15 ++-- toyos-sched/sim/src/explore.rs | 2 +- toyos-sched/sim/src/invariants.rs | 26 ++----- toyos-sched/sim/src/vm.rs | 2 +- toyos-sched/sim/src/workload.rs | 2 +- toyos-sched/sim/tests/scenarios.rs | 4 +- toyos-sched/src/cpu.rs | 14 ++-- 33 files changed, 334 insertions(+), 215 deletions(-) create mode 100644 issues/kernel/a-manage-holder-can-kill-a-kernel-thread.md create mode 100644 issues/kernel/the-reaper-runs-every-kills-teardown-one-at-a-time.md diff --git a/issues/kernel/a-manage-holder-can-kill-a-kernel-thread.md b/issues/kernel/a-manage-holder-can-kill-a-kernel-thread.md new file mode 100644 index 00000000000..97ddefc72e0 --- /dev/null +++ b/issues/kernel/a-manage-holder-can-kill-a-kernel-thread.md @@ -0,0 +1,25 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A `MANAGE` holder can kill a kernel thread + +Every kernel thread (`kernel/src/sched/kthread.rs`) holds a process-table +entry and a pid. `SYS_PROCESS_OPEN` (`kernel/src/syscall/proc.rs`'s +`sys_process_open`) mints a `Process` handle for any pid a `SysCap` carrying +`Rights::MANAGE` names, a kernel thread's included, and `SYS_PROCESS_KILL` +then claims its teardown and posts its retire. Not run yet: a kernel thread has no +Ring 3 boundary to die at, so the expected end is the reaper's tripwire halting the machine +(`reaper: pid N not released after 10s`); killing the reaper itself leaves it +owing its own teardown to itself, with the same end. + +Reachable only through init's capability today: init is the one `MANAGE` +holder. It is still userland crashing the kernel. + +## Exit condition + +A kernel thread's pid mints no `Process` handle — `sys_process_open` refuses it +as it refuses a pid that is gone — with a guest test holding `MANAGE` that +opens each kernel thread's pid from the boot's `kthread:` lines and is refused. diff --git a/issues/kernel/the-reaper-runs-every-kills-teardown-one-at-a-time.md b/issues/kernel/the-reaper-runs-every-kills-teardown-one-at-a-time.md new file mode 100644 index 00000000000..e641aab1305 --- /dev/null +++ b/issues/kernel/the-reaper-runs-every-kills-teardown-one-at-a-time.md @@ -0,0 +1,32 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# The reaper runs every kill's teardown one at a time + +`kernel/src/reaper.rs` is one kernel thread. It takes whichever owed kill has +every thread released, so one victim that is slow to leave the CPU holds up no +other kill; but the teardown it then runs (`process::finish_kill`, the same +`teardown_tail` as exit: handle drops, mapping frees, the zombie marks, the +publish) runs on that one thread. A kill's exit is published only after every +teardown the reaper took before it has finished. Before the reaper, each killer +ran its own victim's teardown and waited on nothing else. + +Who waits on a publish: `/system/bin/init`'s cut-over (`old.kill(); old.wait()`), +sshd's session end, and test-runner's deadline kill. + +What is measured: with blockd's `--silence-write` holding a write unanswered, +killing the stuck client (or blockd itself) and then an idle process B, B's +kill-to-publish was median 1.04 ms, max 6.4 ms over eight rounds, and in +several rounds B was published before the process killed first. No long +teardown has been measured; the one PR #549's review names is a file's flush +on a USB stick. + +## Exit condition + +A kill's publish waits on no other process's teardown — teardowns run +concurrently, or on the thread that owes them — with a guest test that holds +one victim's teardown (a file whose flush the device does not answer) and +shows a second victim's exit published within its own teardown's time. diff --git a/kernel-loom/tests/tlb_shootdown.rs b/kernel-loom/tests/tlb_shootdown.rs index 100edae7225..f6e6fbbd31a 100644 --- a/kernel-loom/tests/tlb_shootdown.rs +++ b/kernel-loom/tests/tlb_shootdown.rs @@ -234,3 +234,29 @@ fn an_initiator_answers_while_it_waits() { assert!(s.wait_turn(0, 1, g0, || {})); }); } + +/// A serve that an interrupt nests inside another publishes the later +/// generation, and the outer serve, finishing after it, must not take it back. +/// +/// Not an interleaving question either: the nesting is one CPU's own schedule, +/// written out. Reds when `serve` stores what it owes instead of raising to it. +#[test] +fn a_nested_serve_is_not_undone_by_the_one_it_interrupted() { + model(|| { + let s = Shootdown::new(); + let first = s.issue(); + let mut later = None; + s.serve(1, || { + let g = s.issue(); + s.serve(1, || {}); + later = Some(g); + }); + let later = later.expect("the nested serve ran"); + assert!(s.served(1, first)); + assert!( + s.served(1, later), + "cpu 1 flushed for the nested shootdown, and the serve it interrupted \ + published an older generation over it", + ); + }); +} diff --git a/kernel/src/block.rs b/kernel/src/block.rs index 41dcdefd388..247cfb1b000 100644 --- a/kernel/src/block.rs +++ b/kernel/src/block.rs @@ -132,9 +132,12 @@ pub fn userland_operations() -> (u32, u64) { } /// Whether the stop this count is read for stops the thread opening an -/// operation now: it stops every userland thread and no kernel thread. +/// operation now. fn counted() -> bool { - !crate::sched::kthread::current_is_kernel_thread() && crate::arch::percpu::current_pid().is_some() + let (Some(pid), Some(tid)) = (crate::arch::percpu::current_pid(), crate::arch::percpu::current_tid()) else { + return false; + }; + !crate::sched::kthread::runs_through_the_stop(crate::scheduler::TaskId(pid, tid)) } #[must_use = "the operation lasts exactly as long as this guard"] diff --git a/kernel/src/drivers/xhci/usbd.rs b/kernel/src/drivers/xhci/usbd.rs index ab36080e598..b96c1cea525 100644 --- a/kernel/src/drivers/xhci/usbd.rs +++ b/kernel/src/drivers/xhci/usbd.rs @@ -3,7 +3,7 @@ use toyos_sched::task::WaitClass; use crate::watch; -use crate::sched::kthread::{self, OnPanic}; +use crate::sched::kthread::{self, OnPanic, OnStop}; use crate::scheduler; use crate::time::Deadline; @@ -13,7 +13,7 @@ const NAME: &str = "usbd"; /// Start the thread. Called once, from `kernel_main`, beside `klogd`'s. pub fn start() { // Unconditional: gating this on a controller would give the kernel's thread count a second answer. - let _ = kthread::spawn(NAME, body, 0, OnPanic::Recover); + let _ = kthread::spawn(NAME, body, 0, OnPanic::Recover, OnStop::Runs); } extern "C" fn body(_arg: u64) -> ! { diff --git a/kernel/src/iod.rs b/kernel/src/iod.rs index 600d9e32cc5..ab99bfc24e2 100644 --- a/kernel/src/iod.rs +++ b/kernel/src/iod.rs @@ -3,7 +3,7 @@ use toyos_sched::task::WaitClass; use crate::watch; -use crate::sched::kthread::{self, OnPanic}; +use crate::sched::kthread::{self, OnPanic, OnStop}; use crate::scheduler; use crate::time::Deadline; @@ -13,7 +13,7 @@ const NAME: &str = "iod"; /// Spawns the `iod` kthread; call once, from `kernel_main`. pub fn start() { // Recoverable: unlike klogd's silent loss, a killed iod's stalled write-back is visible to SYS_FSYNC and logd. - let _ = kthread::spawn(NAME, body, 0, OnPanic::Recover); + let _ = kthread::spawn(NAME, body, 0, OnPanic::Recover, OnStop::Runs); } extern "C" fn body(_arg: u64) -> ! { diff --git a/kernel/src/log/console.rs b/kernel/src/log/console.rs index d7325ed7b3a..5c21bb58e3b 100644 --- a/kernel/src/log/console.rs +++ b/kernel/src/log/console.rs @@ -25,7 +25,7 @@ use toyos_sched::park::notify; use crate::drivers::serial::{self, BackendGuard, MAX_CONSOLE_LINE}; use crate::hw::HW; use crate::sched::driver::{cpus, irq_off}; -use crate::sched::kthread::{self, OnPanic}; +use crate::sched::kthread::{self, OnPanic, OnStop}; use crate::sleeplock::SleepGuard; use crate::watch; use crate::sched::payload::KShared; @@ -67,7 +67,7 @@ static DRAINED: Published = Published::new(); /// Start the thread. Called once, from `kernel_main`, before the scheduler starts. /// Placement matters: APs spin until the machine is released, so an earlier spawn could not run while the machine has no console. pub fn start() { - let (_, sched) = kthread::spawn(NAME, body, 0, OnPanic::Halt); + let sched = kthread::spawn(NAME, body, 0, OnPanic::Halt, OnStop::Runs); // Leaked: `klogd` never exits, and a producer reading this pointer under lock may not touch a refcount. let shared: &'static Arc = alloc::boxed::Box::leak(alloc::boxed::Box::new(sched.shared)); KLOGD.store(shared as *const _ as *mut _, Ordering::Release); diff --git a/kernel/src/log/nested.rs b/kernel/src/log/nested.rs index 5e80f015d60..699bbc4a12f 100644 --- a/kernel/src/log/nested.rs +++ b/kernel/src/log/nested.rs @@ -16,7 +16,7 @@ mod armed { use core::sync::atomic::{AtomicBool, Ordering}; use crate::log::shard::SHARD_RECORDS; - use crate::sched::kthread::{self, OnPanic}; + use crate::sched::kthread::{self, OnPanic, OnStop}; /// One-shot for the body-copy injection point, consumed by `mid_body` or, under `log-shared-reservation`, by the outer `inject`. static ARMED: AtomicBool = AtomicBool::new(false); @@ -44,7 +44,7 @@ mod armed { crate::log!("lognest start records={SHARD_RECORDS}"); // A kernel thread, not the syscall that arms it: `IF` is clear for a whole syscall, so injecting there would never test the guard. // `Halt`: this thread carries the whole stimulus; surviving its death would answer the gate having injected nothing. - kthread::spawn("lognest", body, 0, OnPanic::Halt); + kthread::spawn("lognest", body, 0, OnPanic::Halt, OnStop::Runs); } extern "C" fn body(_arg: u64) -> ! { diff --git a/kernel/src/log/storm.rs b/kernel/src/log/storm.rs index 0b70d0b0062..e18f38657dc 100644 --- a/kernel/src/log/storm.rs +++ b/kernel/src/log/storm.rs @@ -2,7 +2,7 @@ use core::sync::atomic::{AtomicBool, Ordering}; -use crate::sched::kthread::{self, OnPanic}; +use crate::sched::kthread::{self, OnPanic, OnStop}; // Exceeds a shard's capacity, so the drop path under test is reached at every `--smp` count. const STORM_RECORDS: u64 = 1024; @@ -47,7 +47,7 @@ pub fn start_once() { crate::log!("logstorm start threads={threads} records={STORM_RECORDS}"); for thread in 0..threads { // `Halt`: a panicked storm thread invalidates the gate's conservation law, so continuing would answer over an incomplete storm. - kthread::spawn("logstorm", body, thread as u64, OnPanic::Halt); + kthread::spawn("logstorm", body, thread as u64, OnPanic::Halt, OnStop::Runs); } } diff --git a/kernel/src/object/process.rs b/kernel/src/object/process.rs index e076b0308cc..40490ab3b8e 100644 --- a/kernel/src/object/process.rs +++ b/kernel/src/object/process.rs @@ -29,7 +29,6 @@ pub struct ProcessObject { exit: Lock>, /// The same fact, without the lock, for a waiter's per-wake predicate. finished: AtomicBool, - /// What `SYS_PROCESS_WAIT` and a poll arm on; holding the `Arc` across the park keeps the watch from outliving its subject. watch: Arc, } diff --git a/kernel/src/process.rs b/kernel/src/process.rs index 17d6cd9cf48..ad545a1c1fc 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -993,7 +993,7 @@ fn teardown_bookkeeping(table: &mut ProcessTable, process_pid: Pid, code: i32, // Dropping this `Arc` is the release; every other thread is already retired, so the thread running this line holds the last clone — which is why a lock-free crash-report read can never see a table whose owner is off every CPU. proc.symbols = Arc::new(SymbolTable::empty()); - // Whole-process total: retire_threads already folded sibling time into child_threads_cpu_ns. + // Whole-process total: fold_retired already folded sibling time into child_threads_cpu_ns. let cpu_ms = (main_cpu_ns + child_threads_cpu_ns) / 1_000_000; let name = proc.name_str(); log!("exit: {name} pid={process_pid} code={code} cpu={cpu_ms}ms"); @@ -1001,7 +1001,7 @@ fn teardown_bookkeeping(table: &mut ProcessTable, process_pid: Pid, code: i32, Arc::clone(&proc.object) } -/// The accounting a process leaves behind, for `SYS_PROCESS_STATS`. Must run after [`retire_threads`], which folds retired threads' accounting into `ProcessData`. +/// The accounting a process leaves behind, for `SYS_PROCESS_STATS`. Must run after [`fold_retired`], which folds retired threads' accounting into `ProcessData`. fn final_stats( process_data_arc: &Arc>, pid: Pid, @@ -1050,17 +1050,19 @@ fn scheds(pid: Pid, tids: Vec) -> Vec<(Tid, ThreadSched)> { tids.into_iter().filter_map(|t| Some((t, thread_sched(pid, t)?))).collect() } -/// Retire a set of threads, folding their scheduler accounting into the process's; returns the main thread's CPU time if it was among them (else 0). -/// `retire` returns only once its thread is fully off the scheduler — the ordering the whole teardown's memory-freeing safety rests on. -fn retire_threads( +/// Fold released threads' scheduler accounting into the process's; returns the main thread's CPU time if it was among them (else 0). +fn fold_retired( threads: Vec<(Tid, ThreadSched)>, main_tid: Tid, process_data_arc: &Arc>, - retire: fn(&ThreadSched), ) -> u64 { + // The whole teardown's memory-freeing safety rests on this: what follows frees what these threads ran on. + assert!( + threads.iter().all(|(_, sched)| sched.handle.released()), + "teardown: a thread of the process is still on the scheduler", + ); let mut main_cpu_ns = 0u64; for (t, sched) in threads { - retire(&sched); let cpu_ns = scheduler::task_cpu_ns(&sched); let mut pdata = process_data_arc.lock(); sched.handle.merge_into(&mut pdata.accounting); @@ -1135,7 +1137,13 @@ fn release_process(code: i32) { // Phase 2: retire every *other* thread — the current thread can't retire itself. let others = scheds(process_pid, set.others); - let mut main_cpu_ns = retire_threads(others, main_tid, &process_data_arc, scheduler::retire_task); + for (_, sched) in &others { + scheduler::post_retire(sched); + } + for (_, sched) in &others { + scheduler::await_released(sched); + } + let mut main_cpu_ns = fold_retired(others, main_tid, &process_data_arc); // Filtered out of the retire set above, so its time is picked up here if it's the main thread. if set.current_is_main { main_cpu_ns = thread_sched(process_pid, tid) @@ -1674,10 +1682,21 @@ pub struct Killed { threads: Vec<(Tid, ThreadSched)>, } -/// The reaper's half of a kill: wait out every thread, then phases 3-5, the same teardown tail as exit. +impl Killed { + /// Whether every thread is off the scheduler, so [`finish_kill`] may run. + pub fn released(&self) -> bool { + self.threads.iter().all(|(_, sched)| sched.handle.released()) + } + + pub fn pid(&self) -> Pid { + self.pid + } +} + +/// The reaper's half of a kill, once [`Killed::released`]: phases 3-5, the same teardown tail as exit. pub fn finish_kill(killed: Killed) { let Killed { pid, process_data, thread_data, main_tid, threads } = killed; - let main_cpu_ns = retire_threads(threads, main_tid, &process_data, scheduler::await_released); + let main_cpu_ns = fold_retired(threads, main_tid, &process_data); teardown_tail(&process_data, &thread_data, pid, KILLED_EXIT_CODE, main_cpu_ns); } diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 6871fd5a859..7db355b271f 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -25,10 +25,6 @@ //! whatever lock the thread on that CPU was holding, and `sync_all` is the //! first thing that would wait on it. //! -//! Kernel threads are exempt by identity, not by accident: `klogd`, `iod` and -//! `usbd` are in the process table like anything else, and -//! [`crate::sched::kthread::is_kernel_task`] is what tells them apart. -//! //! # What the stop waits on //! //! [`PROGRESS`], posted by [`note_progress`] from the three transitions that @@ -98,20 +94,12 @@ fn stops(stage: u32) -> bool { let (Some(pid), Some(tid)) = (percpu::current_pid(), percpu::current_tid()) else { return false; }; - // A kernel thread reaches this boundary on its first dispatch and has no - // Ring 3 to be stopped from; `iod` is also what carries the sync to its - // volume while userland is being stopped around it. - if exempt(TaskId(pid, tid)) { + if crate::sched::kthread::runs_through_the_stop(TaskId(pid, tid)) { return false; } must_stop(ThreadId { pid: pid.raw(), tid: tid.raw() }, caller()) } -/// A kernel thread is not stopped, except the reaper: the teardowns it runs are userland's. -fn exempt(id: TaskId) -> bool { - crate::sched::kthread::is_kernel_task(id) && !crate::reaper::is(id) -} - /// Whether this thread is the one shutdown this boot gets: a second caller's /// sweep would band the first where it parks inside its own sync. pub fn claim_the_shutdown() -> bool { @@ -232,7 +220,7 @@ fn sweep(caller: ThreadId) -> Sweep { if matches!(thread.state(), process::ThreadLocation::Zombie(_)) { continue; } - if exempt(TaskId(pid, tid)) { + if crate::sched::kthread::runs_through_the_stop(TaskId(pid, tid)) { continue; } let sched = thread diff --git a/kernel/src/reaper.rs b/kernel/src/reaper.rs index 8d27b05991b..260e0d2c32d 100644 --- a/kernel/src/reaper.rs +++ b/kernel/src/reaper.rs @@ -2,51 +2,82 @@ //! //! A kill posts its victim's retires and returns, because the victim may be //! killing the killer, and a killer that waited for its victim would wait on a -//! thread that is waiting on it. This thread does the waiting instead: no -//! process can kill it, and the only threads it waits on are killed ones, -//! whose release waits on nothing it does. It stops with userland -//! (`quiesce::exempt`), since the teardowns it runs are userland's. +//! thread that is waiting on it. This thread does the waiting instead: the +//! only threads it waits on are killed ones, whose release waits on nothing it +//! does. It takes whichever owed kill is released first, so a slow victim holds +//! up no other kill's release; the teardowns themselves run one at a time. -use alloc::collections::VecDeque; -use core::sync::atomic::{AtomicU64, Ordering}; +use alloc::vec::Vec; + +use toyos_sched::task::WaitClass; use crate::process::{self, Killed}; -use crate::sched::kthread::{self, OnPanic}; -use crate::scheduler::{Parkable, TaskId}; +use crate::sched::kthread::{self, OnPanic, OnStop}; +use crate::sched::payload::ANY_RELEASED; +use crate::scheduler::{Parkable, RETIRE_GIVE_UP}; use crate::sync::Lock; +use crate::time::{Deadline, Instant}; use crate::watch::{self, Watch}; const NAME: &str = "reaper"; -static OWED: Lock> = Lock::new(VecDeque::new()); +/// Every owed kill with the instant it was owed, oldest first. +static OWED: Lock> = Lock::new(Vec::new()); +/// What an idle reaper waits on. static WORK: Watch = Watch::new(); -/// The reaper's packed `TaskId`, stored before the machine is released. -static REAPER: AtomicU64 = AtomicU64::new(u64::MAX); - /// Spawns the reaper; call once, from `kernel_main`. pub fn start() { // Halts: a dead reaper leaves every later kill unpublished and nothing to say so. - let (id, _) = kthread::spawn(NAME, body, 0, OnPanic::Halt); - REAPER.store(id.pack(), Ordering::Release); -} - -pub fn is(id: TaskId) -> bool { - REAPER.load(Ordering::Acquire) == id.pack() + kthread::spawn(NAME, body, 0, OnPanic::Halt, OnStop::Stops); } /// Hand a claimed kill, its retires already posted, to the reaper. pub fn owe(killed: Killed) { - OWED.lock().push_back(killed); + OWED.lock().push((crate::clock::now(), killed)); + // Both: an idle reaper waits on `WORK`, a busy one on `ANY_RELEASED`, and + // this kill's threads may all be released already. WORK.post(); + ANY_RELEASED.post(); } extern "C" fn body(_arg: u64) -> ! { let parkable = Parkable::at_entry(); loop { - watch::wait_uncancellable_until(&parkable, &WORK, 0, || !OWED.lock().is_empty()); - let killed = OWED.lock().pop_front().expect("the one consumer saw an entry"); - process::finish_kill(killed); + process::finish_kill(next_released(&parkable)); + } +} + +/// The oldest owed kill whose every thread is released, once there is one. +fn next_released(parkable: &Parkable) -> Killed { + loop { + let idle = OWED.lock().is_empty(); + let armed = watch::arm(if idle { &WORK } else { &ANY_RELEASED }, 0, WaitClass::Other) + .expect("the reaper is a task"); + let oldest = { + let mut owed = OWED.lock(); + if let Some(at) = owed.iter().position(|(_, killed)| killed.released()) { + return owed.remove(at).1; + } + // Armed on the watch that no longer wakes this state. + if owed.is_empty() != idle { + continue; + } + owed.first().map(|(since, killed)| (*since, killed.pid())) + }; + let deadline = match oldest { + None => Deadline::never(), + Some((since, pid)) => { + let give_up = Deadline::at(since + RETIRE_GIVE_UP.duration()); + assert!( + !give_up.reached(crate::clock::now()), + "reaper: pid {pid} not released after {}", + RETIRE_GIVE_UP.duration(), + ); + give_up + } + }; + watch::wait_uncancellable(parkable, &armed, deadline); } } diff --git a/kernel/src/sched/kthread.rs b/kernel/src/sched/kthread.rs index 8e400597571..dcb381be306 100644 --- a/kernel/src/sched/kthread.rs +++ b/kernel/src/sched/kthread.rs @@ -7,7 +7,7 @@ use alloc::string::String; use alloc::sync::Arc; use alloc::vec::Vec; -use core::sync::atomic::{AtomicU64, Ordering}; +use core::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use crate::process::{ ElfInfo, Endowments, PageFaultTrace, ProcessAccounting, ProcessData, ProcessEntry, ThreadData, @@ -19,7 +19,6 @@ use crate::sync::Lock; use super::payload::ThreadSched; -/// `klogd`, `usbd`, `iod` and `reaper`, plus one `log-storm` thread per shard in the actuator build. #[cfg(not(feature = "boot-actuators"))] const MAX_KERNEL_TASKS: usize = 4; #[cfg(feature = "boot-actuators")] @@ -32,13 +31,14 @@ const NO_TASK: u64 = u64::MAX; /// after the policy word, without another claimant matching the same row. const CLAIMING: u64 = u64::MAX - 1; -/// `task` is stored `Release` after `recoverable` and loaded `Acquire` before it, +/// `task` is stored `Release` after `recoverable` and `stops` and loaded `Acquire` before them, /// so a row found by identity never answers with an unwritten policy. struct Row { task: AtomicU64, // `recoverable` exists because `percpu::in_syscall()` is never true for a kernel // thread, which would otherwise make every kernel-thread panic halt the machine by default. recoverable: AtomicU64, + stops: AtomicBool, } /// A row reserved before the table lock and published before `enqueue_new`. @@ -60,18 +60,28 @@ impl Claim { } /// Payload first, then the identity `Release`. - fn publish(self, id: TaskId, on_panic: OnPanic) { + fn publish(self, id: TaskId, on_panic: OnPanic, on_stop: OnStop) { self.0 .recoverable .store(u64::from(on_panic == OnPanic::Recover), Ordering::Relaxed); + self.0.stops.store(on_stop == OnStop::Stops, Ordering::Relaxed); self.0.task.store(id.pack(), Ordering::Release); } } /// Registered at spawn and never cleared: a dead `Recover` thread's row stays. -static ROWS: [Row; MAX_KERNEL_TASKS] = - [const { Row { task: AtomicU64::new(NO_TASK), recoverable: AtomicU64::new(0) } }; - MAX_KERNEL_TASKS]; +static ROWS: [Row; MAX_KERNEL_TASKS] = [const { + Row { task: AtomicU64::new(NO_TASK), recoverable: AtomicU64::new(0), stops: AtomicBool::new(false) } +}; MAX_KERNEL_TASKS]; + +/// Whether the machine's stop (`quiesce`) stops a kernel thread. +#[derive(Clone, Copy, PartialEq, Eq)] +pub enum OnStop { + /// It carries the stop's own work, as `iod` carries the sync to its volume. + Runs, + /// The work it does is userland's, so it stops with userland. + Stops, +} /// What a kernel thread's panic does. #[derive(Clone, Copy, PartialEq, Eq)] @@ -109,13 +119,26 @@ pub fn is_kernel_task(id: TaskId) -> bool { ROWS.iter().any(|row| row.task.load(Ordering::Acquire) == packed) } +/// Is `id` a kernel thread declared [`OnStop::Runs`]? +pub fn runs_through_the_stop(id: TaskId) -> bool { + let packed = id.pack(); + ROWS.iter() + .any(|row| row.task.load(Ordering::Acquire) == packed && !row.stops.load(Ordering::Relaxed)) +} + /// Whether a panic on the running task recovers; `None` unless it is a kernel thread. pub fn panic_recovers_here() -> Option { Some(current_row()?.recoverable.load(Ordering::Relaxed) != 0) } -/// Start a kernel thread running `body(arg)` on its own kernel stack and return its identity and scheduler faces. -pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64, on_panic: OnPanic) -> (TaskId, ThreadSched) { +/// Start a kernel thread running `body(arg)` on its own kernel stack and return its scheduler faces. +pub fn spawn( + name: &str, + body: extern "C" fn(u64) -> !, + arg: u64, + on_panic: OnPanic, + on_stop: OnStop, +) -> ThreadSched { let (stack, entry_rsp) = crate::loader::alloc_kernel_stack( crate::loader::kernel_start, body as usize as u64, @@ -148,7 +171,7 @@ pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64, on_panic: OnPa }); let tid = table.get(pid).expect("kthread: the entry just inserted is gone").main_tid(); // Before `enqueue_new`: from that call the task can run and panic. - claim.publish(TaskId(pid, tid), on_panic); + claim.publish(TaskId(pid, tid), on_panic, on_stop); // The kernel address space, named so one declaration decides every task's `cr3`. let (sched, _dst) = scheduler::enqueue_new( TaskId(pid, tid), @@ -174,7 +197,7 @@ pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64, on_panic: OnPa OnPanic::Recover => "kills the thread", } ); - (TaskId(pid, tid), sched) + sched } /// Every field is the empty value: a kernel thread has no user half. diff --git a/kernel/src/sched/payload.rs b/kernel/src/sched/payload.rs index 7e1cad4fabd..fd3fb48594a 100644 --- a/kernel/src/sched/payload.rs +++ b/kernel/src/sched/payload.rs @@ -77,6 +77,9 @@ pub const SCHED_READY: u8 = 1; pub const SCHED_BLOCKED: u8 = 2; pub const SCHED_UNKNOWN: u8 = 3; +/// Posted after every thread's release, for a waiter on whichever of many threads goes first. +pub static ANY_RELEASED: Watch = Watch::new(); + /// What a thread other than the one running can be asked about; published here since a `CpuSched` is `!Sync` and unreachable remotely. pub struct TaskHandle { cpu_ns: AtomicU64, @@ -124,6 +127,7 @@ impl TaskHandle { self.released.store(true, Ordering::Release); // The retirer arms on this thread's own watch, the same subject a joiner uses. self.watch.post(); + ANY_RELEASED.post(); } /// Has `Hw::release` run for this thread? The retire wait's condition. diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index c6347273242..2c673756886 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -501,13 +501,6 @@ pub fn futex_wake(phys_addr: DirectMap, count: usize) -> u64 { futex::watch_of(phys_addr).post_n(phys_addr.phys(), count) as u64 } -/// Retire a thread and wait until it is released. -#[track_caller] -pub fn retire_task(sched: &ThreadSched) { - post_retire(sched); - await_released(sched); -} - /// Set a thread's kill bit and ask its CPU for a safe point; returns at once. pub fn post_retire(sched: &ThreadSched) { if sched.handle.released() { @@ -518,6 +511,16 @@ pub fn post_retire(sched: &ThreadSched) { }); } +/// How long a retired thread may take to be released before the kernel panics. +/// Superseded by the scheduling-reservations design; kept because a +/// known-wrong constant is still what this kernel runs +/// (`issues/kernel/scheduler-pass-blocks-in-xhci.md`). +pub const RETIRE_GIVE_UP: Tripwire = Tripwire::absurd( + Duration::from_secs(10), + "four pass prologues on xHCI's own 2 s deadline, two quanta, and an unwind \ + the real-time band may stretch elevenfold; past this the wake was lost", +); + /// Wait until a retired thread's record — kernel stack and address-space /// reference — is released. The state word reading `Dead` is not enough: that /// payload is freed by the pass after the one that publishes it. @@ -525,12 +528,12 @@ pub fn post_retire(sched: &ThreadSched) { pub fn await_released(sched: &ThreadSched) { // Also on the early-return path below, where no park happens and the two // asserts inside the wait would never run. - assert_baseline(blocking_baseline()); + assert_baseline(BASELINE_TRAP); if let (Some(pid), Some(tid)) = (percpu::current_pid(), percpu::current_tid()) { if let Some(handle) = driver::current_shared() { assert!( !Arc::ptr_eq(&handle, &sched.shared), - "retire_task: cannot retire self ({})", + "await_released: cannot retire self ({})", TaskId(pid, tid), ); } @@ -544,15 +547,7 @@ pub fn await_released(sched: &ThreadSched) { Duration::from_millis(50), "two hundred re-polls inside the tripwire, on a thread that is otherwise parked", ); - /// Superseded by the scheduling-reservations design; kept because a - /// known-wrong constant is still what this kernel runs - /// (`issues/kernel/scheduler-pass-blocks-in-xhci.md`). - const GIVE_UP: Tripwire = Tripwire::absurd( - Duration::from_secs(10), - "four pass prologues on xHCI's own 2 s deadline, two quanta, and an unwind \ - the real-time band may stretch elevenfold; past this the wake was lost", - ); - let give_up = Deadline::at(crate::clock::now() + GIVE_UP.duration()); + let give_up = Deadline::at(crate::clock::now() + RETIRE_GIVE_UP.duration()); let parkable = Parkable::at_entry(); // Uncancellable: a killed retirer cannot propagate a cancel with the // retire half done; the tripwire above bounds it instead. @@ -561,13 +556,13 @@ pub fn await_released(sched: &ThreadSched) { sched.shared.key().0, WaitClass::Other, ) else { - panic!("retire_task: no current task to park"); + panic!("await_released: no current task to park"); }; while !sched.handle.released() { if give_up.reached(crate::clock::now()) { panic!( - "retire_task: task not released after {}: {:?}", - GIVE_UP.duration(), + "await_released: task not released after {}: {:?}", + RETIRE_GIVE_UP.duration(), sched.shared.state() ); } diff --git a/kernel/src/shootdown.rs b/kernel/src/shootdown.rs index 3e8f1408464..6c453a99c17 100644 --- a/kernel/src/shootdown.rs +++ b/kernel/src/shootdown.rs @@ -58,7 +58,8 @@ impl Shootdown { pub fn serve(&self, cpu: usize, flush: impl FnOnce()) { let owed = self.requested.load(OWED); flush(); - self.flushed[cpu].store(owed, Ordering::Release); + // A max, not a store: a serve nested inside this one may already have published a later generation. + self.flushed[cpu].fetch_max(owed, Ordering::Release); } /// Does `cpu` owe anyone a flush? diff --git a/kernel/src/syscall/dispatch.rs b/kernel/src/syscall/dispatch.rs index 06a68fad7a1..b02db40c7d9 100644 --- a/kernel/src/syscall/dispatch.rs +++ b/kernel/src/syscall/dispatch.rs @@ -239,16 +239,7 @@ pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> data.handles .get::(RawHandle(a1 as u32), Rights::MANAGE) }) { - Ok(object) => { - // Killing yourself is exiting, not killing. - // Reachable: TRANSFER lets a parent hand a child a handle to itself. - if object.pid() == process::current_process() { - // exit() never returns, so the held clone is dropped while it still can be. - drop(object); - process::exit(process::KILLED_EXIT_CODE); - } - process::kill_process(&object) - } + Ok(object) => process::kill_process(&object), Err(e) => e.refuse(), } } diff --git a/tests/common/metal.rs b/tests/common/metal.rs index 09db138cd58..d4f9a27cdee 100644 --- a/tests/common/metal.rs +++ b/tests/common/metal.rs @@ -397,7 +397,7 @@ impl Readback { pub fn stop_completed(&self) -> Result<(), String> { match self.stop_record() { Some(park) if !park.stopped_the_machine() => Err(format!( - "{}'s stop gave up on {} userland thread(s) that never reached a safe point, so \ + "{}'s stop gave up on {} thread(s) that never reached a safe point, so \ this boot's sync and its last word are claims about a machine that was still \ running:\n {park}", self.label, diff --git a/tests/common/power.rs b/tests/common/power.rs index 4793e3311fc..43eea24d270 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -170,10 +170,9 @@ pub fn quiesce_stops_the_machine( const WRITERS: u32 = 6; // The threads the stop names besides the writers: the job's own main // thread, parked on init's answer; `test-runner`'s main and deadline - // threads; and `logd`'s. `init` asked for the stop and is its caller. - const OTHERS: u32 = 4; - // The kernel's `reaper`, which stops with userland because it runs userland's kills. - const REAPER: u32 = 1; + // threads; `logd`'s; and the kernel's `reaper`, which stops with userland + // because it runs userland's kills. `init` asked for the stop and is its caller. + const OTHERS: u32 = 5; /// Mirrored in `kernel/src/syscall/machine.rs`, which queues it. const QUEUED: &str = "console: a holder's line, queued once the stop had stopped every holder"; let (whole, record) = stopped_boot( @@ -215,13 +214,13 @@ pub fn quiesce_stops_the_machine( // harness is told, or lost one to an I/O error before the reset, has fewer // threads to stop and would pass every judge above over a machine that was // not the one described. - if record.sweep.total() != WRITERS + OTHERS + REAPER { + if record.sweep.total() != WRITERS + OTHERS { return Err(format!( - "this boot's stop named {} userland thread(s); {WRITERS} writers plus the {OTHERS} \ - of the job, test-runner and logd and the reaper make {}, so this is not the machine \ - the writers were on:\n {record}\n{whole}", + "this boot's stop named {} thread(s); {WRITERS} writers plus the {OTHERS} \ + of the job, test-runner, logd and the reaper make {}, so this is not the machine the \ + writers were on:\n {record}\n{whole}", record.sweep.total(), - WRITERS + OTHERS + REAPER, + WRITERS + OTHERS, )); } woken_by_its_threads(&record)?; diff --git a/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs b/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs index c035f6bde2d..59826e64aea 100644 --- a/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs +++ b/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs @@ -42,24 +42,18 @@ //! anywhere, which is the case that is *not* under test. use std::io::{Read, Write}; -use std::os::toyos::process::{ChildExt, CommandExt}; +use std::os::toyos::process::CommandExt; use std::process::{Command, Stdio}; use std::sync::atomic::{AtomicU64, Ordering}; use std::thread; use std::time::{Duration, Instant}; -use toyos::process::Process; use toyos::{endow, namespace, port, AsHandle}; use toyos_abi::syscall::{self, SyscallError, SERVE_PREFIX, SVC_LABEL}; -use toyos_abi::RawHandle; const SELF_PATH: &str = "/system/bin/test_rs_kill_while_blocked"; const SERVICE: &str = "blocked"; -/// The label arm 4's killer finds the spinner under. A local name in one -/// process's own table, and it names nothing anywhere else. -const VICTIM_LABEL: &str = "victim"; - /// How long arm 4 gives a killed Ring 3 spinner to reach its last exit /// boundary, watched from outside the kill. /// @@ -75,12 +69,6 @@ const VICTIM_LABEL: &str = "victim"; /// headroom", which would be true of the interrupt delivery alone — precisely /// the part this arm cannot observe from outside the kill, and the part the /// window's several scheduler dispatches sit on top of. -/// -/// **It has to be smaller than `scheduler::retire_task`'s own tripwire, and -/// that ordering is the whole of the arm's ability to report.** Nothing here -/// reads that constant or restates its value: what this one needs is only to be -/// first, and a bound derived from device timeouts and scheduler quanta is not -/// going to come in under two seconds. const ENDS_WITHIN: Duration = Duration::from_secs(2); fn main() { @@ -101,8 +89,7 @@ fn test() { /// Spawn a child in `role` and wait for it to say it has parked. /// /// The child's stdin is a pipe this process holds the write end of, which is -/// what arms 1 and 2 measure afterwards and what releases arm 4's killer at a -/// moment this process picks. +/// what arms 1 and 2 measure afterwards. fn parked(role: &str, extra: Option<(String, u32)>) -> std::process::Child { let mut command = Command::new(SELF_PATH); command.arg(role).stdin(Stdio::piped()).stdout(Stdio::piped()); @@ -222,7 +209,7 @@ fn an_acceptor_killed_in_the_accept() { /// holds reaches EOF when — and only when — the victim's handles are drained. /// /// **The clock starts before the kill and not after it.** -/// What this one bounds is the whole of [go byte → boundary → teardown], so the +/// What this one bounds is the whole of [kill → boundary → teardown], so the /// number is an upper bound on the boundary rather than a measurement of it. fn a_ring_three_spinner_ends_at_its_next_exit_boundary() { let mut victim = parked("spin", None); @@ -231,14 +218,7 @@ fn a_ring_three_spinner_ends_at_its_next_exit_boundary() { // thing that can ever arrive on it is the end of the victim. let mut spun = victim.stdout.take().expect("the spinner's stdout"); - // A duplicate, because `endow` moves what it is handed: this process keeps - // its own handle so that dropping the `Child` stays its business. - let for_killer = - syscall::dup(RawHandle(victim.as_raw_handle())).expect("a second handle to the spinner"); - let mut killer = parked("kill", Some((VICTIM_LABEL.to_string(), for_killer.0))); - let mut go = killer.stdin.take().expect("the killer's stdin"); - - /// Nanoseconds from the go byte to EOF, and `u64::MAX` until there is one. + /// Nanoseconds from the kill to EOF, and `u64::MAX` until there is one. /// Stored by the reader so the answer is the instant it saw rather than the /// poll that noticed. static GONE_AFTER_NS: AtomicU64 = AtomicU64::new(u64::MAX); @@ -248,7 +228,7 @@ fn a_ring_three_spinner_ends_at_its_next_exit_boundary() { while spun.read(&mut byte).expect("read the spinner's stdout") != 0 {} GONE_AFTER_NS.store(started.elapsed().as_nanos() as u64, Ordering::Release); }); - go.write_all(b"g").expect("release the killer"); + victim.kill().expect("kill the spinning child"); while GONE_AFTER_NS.load(Ordering::Acquire) == u64::MAX && started.elapsed() < ENDS_WITHIN { thread::sleep(Duration::from_millis(1)); @@ -263,10 +243,6 @@ fn a_ring_three_spinner_ends_at_its_next_exit_boundary() { ); std::process::exit(1); } - assert!( - killer.wait().expect("wait for the killer").success(), - "the spinner ended, but the process that killed it did not agree", - ); println!( " ring 3: a killed spinner reached its last exit boundary in {:?}, with no syscall \ to cancel", @@ -287,16 +263,6 @@ fn child(role: &str) -> ! { std::hint::spin_loop(); } } - "kill" => { - let victim: Process = endow::Endowments::get() - .take(VICTIM_LABEL) - .expect("the parent endowed a handle to the spinner"); - say("parked in kill"); - let mut go = [0u8; 1]; - std::io::stdin().read_exact(&mut go).expect("wait for the parent's go"); - victim.kill().expect("kill the spinning child"); - std::process::exit(0); - } "pipe-read" => { say("parked in pipe-read"); let mut buf = [0u8; 64]; diff --git a/tests/toyos-rust-tests/src/bin/mutual_kill.rs b/tests/toyos-rust-tests/src/bin/mutual_kill.rs index d66f4a9e4f3..9cd1218a799 100644 --- a/tests/toyos-rust-tests/src/bin/mutual_kill.rs +++ b/tests/toyos-rust-tests/src/bin/mutual_kill.rs @@ -3,9 +3,8 @@ //! //! Each round the parent hands each child a handle to the other and one shared //! word; both spin on the word, the parent sets it, and both call -//! `SYS_PROCESS_KILL` inside the same few microseconds. A kill that waited for -//! its victim's threads to leave every CPU waited on a thread that was inside -//! the other kill, and the kernel panicked at `retire_task`'s tripwire. +//! `SYS_PROCESS_KILL` inside the same few microseconds. Last, a killer handed +//! its own handle kills itself, and ends killed rather than exiting. use std::io::{Read, Write}; use std::os::toyos::process::{ChildExt, CommandExt}; @@ -64,6 +63,15 @@ fn test() { ); } println!("mutual_kill: {ROUNDS} rounds of two processes killing each other, every one ended"); + + let go = SharedMemory::create(WORD_BYTES).expect("a shared word"); + let (conn, mut own) = killer_child(); + arm(&conn, &go, RawHandle(own.as_raw_handle())); + armed(&mut own); + word(&go).store(1, Ordering::Release); + let code = own.wait().expect("wait the self-killer").code(); + assert_eq!(code, Some(KILLED), "a process that killed itself ended with {code:?}"); + println!("mutual_kill: a process that killed itself ended {KILLED}"); } /// Spawn a killer holding a connector for a fresh port, and accept it. @@ -114,7 +122,7 @@ fn killer() -> ! { assert_eq!(header.msg_type, ARM, "an unexpected frame"); let [word_handle, victim] = conn.recv_handles_exact::<2>().expect("the word and the victim"); let go = SharedMemory::adopt(word_handle, WORD_BYTES).expect("map the word"); - // SAFETY: the parent sent a handle to the other killer, which nothing else here owns. + // SAFETY: the parent sent one handle to the victim, which nothing else here owns. let victim = unsafe { Process::from_raw(victim) }; let mut out = std::io::stdout(); diff --git a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs index 45bb0ef0010..c4d9aca14de 100644 --- a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs +++ b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs @@ -301,6 +301,9 @@ fn its_end_completes_a_poll() { process.watch_end(&poller, END); let mut early = Vec::new(); poller.wait(0, 0, |token| early.push(token)); + // Another handle's close ends no process, so it cancels no poll. + syscall::close(syscall::dup(RawHandle(child.as_raw_handle())).expect("a handle to close")); + poller.wait(0, 0, |token| early.push(token)); assert!(early.is_empty(), "a running process's handle completed a poll: {early:?}"); drop(release); diff --git a/toyos-proclife/src/interleave.rs b/toyos-proclife/src/interleave.rs index 2939755faf5..894446f1b35 100644 --- a/toyos-proclife/src/interleave.rs +++ b/toyos-proclife/src/interleave.rs @@ -68,6 +68,11 @@ impl Finish { Finish { pid, code, tids, next: 0, pc: 0 } } + /// The reaper's: it takes a kill only once every thread is released. + fn released((pid, code, tids): (Pid, i32, Vec)) -> Self { + Finish { pid, code, next: tids.len(), tids, pc: 1 } + } + fn enabled(&self, world: &World) -> bool { self.pc != 0 || self.next == self.tids.len() || retire_ready(world, self.pid, self.tids[self.next]) } @@ -100,7 +105,7 @@ impl Finish { } } -/// `retire_task`, one section: post the retire, then — blocked until then — +/// The base's `retire_task`, one section: post the retire, then — blocked until then — /// find the thread released. `true` once it is. fn retire_step(world: &mut World, pid: Pid, tid: Tid) -> bool { if world.is_retired(pid, tid) { @@ -162,11 +167,12 @@ impl Op { /// Whether the op's next section can run, rather than wait. fn enabled(&self, world: &World) -> bool { match self { - Op::Exit { pid, pc: 1, others, next, .. } => { - *next == others.len() || retire_ready(world, *pid, others[*next]) + Op::Exit { pid, pc: 2, others, next, .. } => { + *next == others.len() || world.is_retired(*pid, others[*next]) } Op::Kill { inline: Some(finish), .. } => finish.enabled(world), Op::Reap { job: Some(finish) } => finish.enabled(world), + Op::Reap { job: None } => world.owes_a_released(), _ => true, } } @@ -206,28 +212,32 @@ impl Op { world.leaving(*pid, *tid); *pc = 1; } - // Phase 2: one `retire_task` per other thread, each with - // the table lock given up. + // Phase 2, with the table lock given up: every other + // thread's retire posted, then each one's release awaited. 1 => { + for &other in others.iter() { + world.post_retire(*pid, other); + } + *pc = 2; + } + 2 => { if *next < others.len() { - if retire_step(world, *pid, others[*next]) { - *next += 1; - } + *next += 1; } else { - *pc = 2; + *pc = 3; } } // Phase 4: the zombie marks, under the table lock. - 2 => { + 3 => { if let Some(proc) = world.get_mut(*pid) { teardown::mark_all_zombie(proc, *code); } - *pc = 3; + *pc = 4; } // Phase 5: publish, with the table lock given up. - 3 => { + 4 => { world.publish_exit(*pid, *code); - *pc = 4; + *pc = 5; } // `exit_current`: the exit pass, one pass later, drops this // thread's payload and `publish_released` posts on its own @@ -270,7 +280,7 @@ impl Op { } } Op::Reap { job } => match job { - None => *job = Some(Finish::new(world.take_owed().expect("stepped with nothing owed"))), + None => *job = Some(Finish::released(world.take_released().expect("enabled with none released"))), Some(finish) => { if finish.step(world) { *job = None; diff --git a/toyos-proclife/src/model.rs b/toyos-proclife/src/model.rs index dc4a6edea83..ef4daf0b3cd 100644 --- a/toyos-proclife/src/model.rs +++ b/toyos-proclife/src/model.rs @@ -3,7 +3,7 @@ //! //! `#[cfg(test)]`, so none of it reaches a kernel build. What it adds beyond //! the two traits is the *consequences* a decision hands back and the kernel -//! performs — a watch's post, a `retire_task`, a `publish_exit`, an idle +//! performs — a watch's post, a retire, a `publish_exit`, an idle //! pass taking an entry — because the laws worth checking are about the order //! those happen in, and a model that only held the two states could not see //! one. @@ -72,7 +72,7 @@ pub struct World { waiters: BTreeSet<(Watch, Pid, Tid)>, /// Waiters a post has released. released: BTreeSet<(Watch, Pid, Tid)>, - /// Threads `scheduler::retire_task` has taken off every CPU. A thread not + /// Threads released off every CPU. A thread not /// in here may still be picked and run. retired: BTreeSet<(Pid, Tid)>, /// Threads past the point of no return but not yet dropped by an exit @@ -217,7 +217,7 @@ impl World { self.waiters.difference(&self.released).copied().collect() } - /// `scheduler::retire_task` — the thread is provably off every CPU, its + /// `Hw::release` — the thread is provably off every CPU, its /// payload dropped, and `publish_released` has posted on its own /// watch. /// @@ -279,9 +279,18 @@ impl World { self.owed.push_back((pid, code, tids)); } - /// The reaper's pop. - pub fn take_owed(&mut self) -> Option<(Pid, i32, Vec)> { - self.owed.pop_front() + /// The reaper's take: the oldest owed kill whose every thread is retired. + pub fn take_released(&mut self) -> Option<(Pid, i32, Vec)> { + let at = self.owed.iter().position(|owed| self.all_retired(owed))?; + self.owed.remove(at) + } + + pub fn owes_a_released(&self) -> bool { + self.owed.iter().any(|owed| self.all_retired(owed)) + } + + fn all_retired(&self, (pid, _, tids): &(Pid, i32, Vec)) -> bool { + tids.iter().all(|&tid| self.is_retired(*pid, tid)) } pub fn nothing_owed(&self) -> bool { diff --git a/toyos-proclife/src/teardown.rs b/toyos-proclife/src/teardown.rs index ac9ca688e2c..13fec0d9605 100644 --- a/toyos-proclife/src/teardown.rs +++ b/toyos-proclife/src/teardown.rs @@ -48,7 +48,7 @@ pub fn claim_teardown(table: &mut T, pid: Pid) -> bool { /// that exit is the main one. /// /// The current thread is **not** in `others`, and cannot be: it is executing -/// the teardown, and `retire_task` returns only when its subject is provably +/// the teardown, and `await_released` returns only when its subject is provably /// off every CPU. Its own CPU time is read separately, which is what /// [`ExitSet::current_is_main`] is for — a main thread filtered out of the /// retire set would otherwise leave `cpu=0ms` on its own exit line. diff --git a/toyos-quiesce/src/lib.rs b/toyos-quiesce/src/lib.rs index 6397d07b861..cd730f173d6 100644 --- a/toyos-quiesce/src/lib.rs +++ b/toyos-quiesce/src/lib.rs @@ -38,7 +38,6 @@ pub fn must_stop(thread: ThreadId, caller: ThreadId) -> bool { thread != caller } -/// What one sweep of the machine's userland threads found. #[derive(Clone, Copy, PartialEq, Eq, Debug, Default)] pub struct Sweep { /// Threads that must stop and have: banded at a safe point, or parked and @@ -77,7 +76,7 @@ pub const LAST_THREAD: &str = "quiesce-last"; pub const STOPPED: &str = "stop: "; const OF: &str = " of "; -const THREADS: &str = " userland thread(s) stopped across "; +const THREADS: &str = " thread(s) stopped across "; const CPUS: &str = " cpu(s) in "; const BUDGET: &str = " ms of a "; const MS: &str = " ms budget over "; @@ -249,12 +248,12 @@ mod tests { fn the_record_names_the_shortfall_only_when_there_is_one() { assert_eq!( alloc::format!("{WHOLE}"), - "stop: 6 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 ms \ + "stop: 6 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 ms \ budget over 3 sweep(s), 0 of 4812 userland block operation(s) still open", ); assert_eq!( alloc::format!("{}", short()), - "stop: 4 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 ms \ + "stop: 4 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 ms \ budget over 3 sweep(s), 1 of 4812 userland block operation(s) still open; this \ reset lands wherever the other 2 are", ); @@ -288,17 +287,17 @@ mod tests { fn a_line_that_is_not_a_record_is_refused_rather_than_half_read() { for line in [ "[stamp] Rebooting.", - "[stamp] stop: 6 of 6 userland thread(s) stopped across 8 cpu(s)", + "[stamp] stop: 6 of 6 thread(s) stopped across 8 cpu(s)", // The shortfall clause disagreeing with the counts it restates. - "[stamp] stop: 4 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ + "[stamp] stop: 4 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ ms budget over 3 sweep(s), 1 of 4812 userland block operation(s) still open; this \ reset lands wherever the other 9 are", // A shortfall clause on a record that claims to have stopped. - "[stamp] stop: 6 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ + "[stamp] stop: 6 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ ms budget over 3 sweep(s), 0 of 4812 userland block operation(s) still open; this \ reset lands wherever the other 2 are", // More threads stopped than there were. - "[stamp] stop: 7 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ + "[stamp] stop: 7 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ ms budget over 3 sweep(s), 0 of 4812 userland block operation(s) still open", ] { assert_eq!(Record::parse(line), None, "{line}"); diff --git a/toyos-sched/sim/src/explore.rs b/toyos-sched/sim/src/explore.rs index a0dc4bb21ab..1d90dbc97a1 100644 --- a/toyos-sched/sim/src/explore.rs +++ b/toyos-sched/sim/src/explore.rs @@ -49,7 +49,7 @@ pub struct Outcome { pub killed_at_park: u64, /// Invariant I14's measurement: the longest a retire went unfinalized, and /// the bound in force. A number as well as a verdict, because the kernel's - /// `retire_task` states the same property with a wall clock and a panic, and + /// `await_released` states the same property with a wall clock and a panic, and /// how much of that budget the protocol spends is what says whether the wall /// clock is a backstop or a coin flip. pub retire_latency: u64, diff --git a/toyos-sched/sim/src/invariants.rs b/toyos-sched/sim/src/invariants.rs index 1531ef92324..63686abef81 100644 --- a/toyos-sched/sim/src/invariants.rs +++ b/toyos-sched/sim/src/invariants.rs @@ -36,7 +36,7 @@ use crate::vm::{FairEpoch, Vm, IPI_LATENCY_NS, RUN_CHUNK_NS, UNWIND_NS}; /// bound into a statement about how long a kernel teardown takes. The second /// attempt served it strictly *after* `rq` and had no term at all — at the price /// of a corpse that never runs under a saturated RT band, which is -/// `scheduler::retire_task`'s tripwire and a kernel panic from a legal +/// `scheduler::await_released`'s tripwire and a kernel panic from a legal /// `Rights::RT` workload. /// /// What ships is neither absolute. `CpuSched::pick` takes the dying list ahead @@ -65,7 +65,7 @@ fn rt_latency_bound(max_kernel_section: u64) -> u64 { } /// How long a retire may take to reach `Hw::release` (invariant I14), measured -/// on the **wall clock** — the one `scheduler::retire_task`'s own tripwire +/// on the **wall clock** — the one `scheduler::await_released`'s own tripwire /// reads, and see [`crate::vm::Killed`] for why there is no second clock any /// more. /// @@ -86,7 +86,7 @@ fn rt_latency_bound(max_kernel_section: u64) -> u64 { /// CPU's *zombie*, because a pass cannot free the stack it is standing on; /// the payload is released by the **next** pass on that CPU /// (`SchedPass::begin`), and if that CPU dispatched another task its next -/// pass is that task's quantum expiry. `retire_task`'s own doc states the +/// pass is that task's quantum expiry. `await_released`'s own doc states the /// same hop from the other side, and the wait is for the release, not for /// the word. /// @@ -99,18 +99,6 @@ fn rt_latency_bound(max_kernel_section: u64) -> u64 { /// otherwise would price the machine rather than the protocol. `peers` is the /// greatest number of *other* corpses that CPU has held since this retire was /// claimed. -/// -/// **Where the shape comes from, in the model and in the kernel, and they are -/// not the same.** This model's `Vm::teardown` posts a retire for every sibling -/// of a torn-down process in one op with no wait between them, so a batched -/// single-process teardown is what drives `peers` above zero here. The kernel -/// cannot produce that: both of its teardown loops call `retire_task` per tid -/// and it blocks until the victim is released, so one process teardown holds at -/// most one corpse at a time. What produces `peers > 0` there is *concurrent -/// independent retirers* — separate killer threads retiring separate victims -/// that share a CPU — and nothing bounds how many. The model's shape is the -/// cheaper way to reach the same queue depth, and the bound is about the depth -/// rather than about who made it. fn retire_latency_bound(max_kernel_section: u64, peers: usize) -> u64 { 2 * QUANTUM_NS + IPI_LATENCY_NS @@ -130,7 +118,7 @@ fn retire_latency_bound(max_kernel_section: u64, peers: usize) -> u64 { /// *finite* factor rather than the unbounded term the previous form of this /// derivation declined to price at all. /// -/// The same factor is a term of `scheduler::retire_task`'s `GIVE_UP` +/// The same factor is a term of `scheduler::RETIRE_GIVE_UP` /// derivation, and `toyos_sched`'s own /// `an_unwind_under_saturated_rt_is_stretched_by_the_age_ratio` is what stops /// the two drifting apart. @@ -312,7 +300,7 @@ fn check_sleeping_cpus(vm: &mut Vm<'_>) { /// — so a CPU that hands on a task it knows is dead trades an unwind it could /// start in this pass for a wait on another CPU's next voluntary one. **And a /// retire completes within [`retire_latency_bound`]**, which is the statement -/// the kernel's `retire_task` makes with a wall clock and a panic. +/// the kernel's `await_released` makes with a wall clock and a panic. /// /// **Both halves survive the cancellable kill and only one of them /// changed.** It makes a killed task *run* rather than be reaped where it @@ -393,7 +381,7 @@ fn check_retires(vm: &mut Vm<'_>) { if elapsed > bound { problems.push(format!( "I14: {key:?} was retired {elapsed} ns ago and is still {:?} \ - (bound {bound} ns, on the wall clock `retire_task` reads)", + (bound {bound} ns, on the wall clock `await_released` reads)", vm.shared[&key].state(), )); } @@ -421,7 +409,7 @@ fn owner_of(state: TaskState) -> Option { } /// How long this retire has been outstanding, on the clock -/// `scheduler::retire_task`'s own tripwire reads: the wall clock, every CPU's +/// `scheduler::await_released`'s own tripwire reads: the wall clock, every CPU's /// and no CPU's, with nothing subtracted from it. /// /// [`crate::vm::Killed`] carries why there is no second clock any more. diff --git a/toyos-sched/sim/src/vm.rs b/toyos-sched/sim/src/vm.rs index 198d4f759ea..2b6fd594e45 100644 --- a/toyos-sched/sim/src/vm.rs +++ b/toyos-sched/sim/src/vm.rs @@ -152,7 +152,7 @@ pub struct ProcState { /// I14 should be read on a clock with the RT band's service subtracted out. /// /// Every step of that was true and the conclusion was a blindfold. The kernel -/// does not wait on that clock: `scheduler::retire_task` blocks behind a +/// does not wait on that clock: `scheduler::await_released` blocks behind a /// **wall-clock** tripwire and panics when it expires. A model measuring the /// same wait on a clock the kernel cannot read is a model that cannot see the /// panic — and the unbounded quantity the paragraph named is exactly the defect diff --git a/toyos-sched/sim/src/workload.rs b/toyos-sched/sim/src/workload.rs index 135eeb734b6..7d087294384 100644 --- a/toyos-sched/sim/src/workload.rs +++ b/toyos-sched/sim/src/workload.rs @@ -187,7 +187,7 @@ pub enum AgeShape { BoundedDeferral, /// The shape this branch shipped between the two fixes: `pick` asks only /// `rq.has_rt()`, so a permanently-RT thread that never parks holds the - /// dying list closed for ever and `scheduler::retire_task`'s tripwire + /// dying list closed for ever and `scheduler::await_released`'s tripwire /// panics the kernel. See `scenarios::old_rt_starved_the_corpse`. RtOutranksEveryCorpse, } diff --git a/toyos-sched/sim/tests/scenarios.rs b/toyos-sched/sim/tests/scenarios.rs index 861f3b6f3ed..552a1212d29 100644 --- a/toyos-sched/sim/tests/scenarios.rs +++ b/toyos-sched/sim/tests/scenarios.rs @@ -276,7 +276,7 @@ fn old_migrate_keeping_the_corpse_is_caught() { /// permanently-RT thread that never parks answered yes for ever. /// /// That shape shipped on this branch between the two fixes, and its failure is -/// not a slow retire: `scheduler::retire_task` blocks behind a wall-clock +/// not a slow retire: `scheduler::await_released` blocks behind a wall-clock /// tripwire and **panics the kernel**, from a workload that only needs /// `Rights::RT` — which `soundd` holds and `SYS_RT_ENTER` never gives back. /// @@ -321,7 +321,7 @@ fn rt_starving_the_corpse_is_caught() { /// that workload puts a corpse in transit, and every retire completes well /// inside the derived bound. /// -/// The measurement is the point. `retire_task`'s guard is a wall clock two +/// The measurement is the point. `await_released`'s guard is a wall clock two /// orders of magnitude wider than [`toyos_sched_sim::explore::Outcome::retire_bound`], /// so what this reports is how much of that budget the protocol actually spends /// — and a change that starts spending it shows up here as a number long before diff --git a/toyos-sched/src/cpu.rs b/toyos-sched/src/cpu.rs index 4abb0582bff..3e752b664d8 100644 --- a/toyos-sched/src/cpu.rs +++ b/toyos-sched/src/cpu.rs @@ -623,7 +623,7 @@ impl CpuSched { /// /// One permanently-RT thread that never parks then holds a CPU's dying list /// closed for ever, no sibling can rescue the corpse, and - /// `scheduler::retire_task`'s wall-clock tripwire panics the kernel from a + /// `scheduler::await_released`'s wall-clock tripwire panics the kernel from a /// legal `Rights::RT` workload. Invariant I14 must catch it; /// `scenarios::old_rt_starved_the_corpse` is the gate that proves it does, /// and it is the *other* direction of the pair @@ -1299,7 +1299,7 @@ pub const MAX_PASS_NS: u64 = 200_000; /// under one permanently-RT thread that never parks — `Rights::RT` is /// capability-gated but `soundd` holds it and `SYS_RT_ENTER` has no /// revocation, so a killed thread of an RT process on that CPU never reaches -/// `Hw::release`, and `scheduler::retire_task`'s tripwire panics the kernel +/// `Hw::release`, and `scheduler::await_released`'s tripwire panics the kernel /// from a legal workload. That is the kernel crashing from userland. /// /// So the corpse ages. Once its head has stood in the dying list for this long, @@ -1324,7 +1324,7 @@ pub const DYING_AGE_NS: u64 = QUANTUM_NS; /// **This is the number invariant I4's bound grows by**, so it is as small as /// the other side can afford. Under saturated RT an unwind is delivered at one /// chunk per `DYING_AGE_NS + DYING_CHUNK_NS`, so the corpse's release is -/// stretched by 11× — which is a term of `scheduler::retire_task`'s `GIVE_UP` +/// stretched by 11× — which is a term of `scheduler::RETIRE_GIVE_UP` /// derivation. A larger chunk buys that term /// back and spends it on RT latency; `soundd` is the process that pays, and 1 ms /// of added worst-case jitter once per 10 ms is the trade this picks. @@ -1858,7 +1858,7 @@ impl SchedPass<'_, '_, H, P, Disposed> { /// its failure is a kernel panic: one permanently-RT thread that never parks /// holds this CPU's dying list closed for ever, no sibling CPU can rescue a /// corpse (`hand_off` refuses to migrate a killed task, `pop_surplus` reads - /// `fair` only), and `scheduler::retire_task`'s tripwire fires. That is + /// `fair` only), and `scheduler::await_released`'s tripwire fires. That is /// reachable from a legal `Rights::RT` workload — `soundd` holds the right /// and `SYS_RT_ENTER` has no revocation — so it is the kernel crashing from /// userland. @@ -3411,7 +3411,7 @@ mod tests { /// corpse never starves the RT band. /// This one says the RT band never starves the corpse, because unqualified /// RT precedence over the dying list ends in a kernel panic: - /// `scheduler::retire_task` blocks on `Hw::release` behind a tripwire, and + /// `scheduler::await_released` blocks on `Hw::release` behind a tripwire, and /// one permanently-RT thread that never parks holds this CPU's dying list /// closed for ever. `hand_off` refuses to migrate a killed task and /// `pop_surplus` reads `fair` only, so no sibling CPU can rescue it. @@ -3522,7 +3522,7 @@ mod tests { /// stretched by that factor and no more. /// /// It is a separate test because it is the term - /// `scheduler::retire_task`'s `GIVE_UP` derivation carries, and a change + /// `scheduler::RETIRE_GIVE_UP` derivation carries, and a change /// that quietly widened `DYING_AGE_NS` would leave the gate above green /// while making that tripwire wrong. #[test] @@ -3534,7 +3534,7 @@ mod tests { let stretch = (DYING_AGE_NS + DYING_CHUNK_NS) / DYING_CHUNK_NS; assert_eq!( stretch, 11, - "`retire_task`'s GIVE_UP prices the unwind at {stretch}x its own CPU \ + "`RETIRE_GIVE_UP` prices the unwind at {stretch}x its own CPU \ time; the constants say {}", DYING_AGE_NS + DYING_CHUNK_NS, ); From 1c5ca17b7367ffe77d9f0e3247c7258619680626 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 19:14:21 +0200 Subject: [PATCH 05/19] review #549 round 2: no kernel thread's pid opens, and the reaper's choice is the model's A kernel thread's pid names no process a handle could hold: `process::process_object`, the one door from a pid to a `Process` object, refuses it, so `SYS_PROCESS_OPEN` answers NotFound and a MANAGE holder can no longer kill iod, usbd, klogd or the reaper into the tripwire's halt. The control is a second verdict under `process-reopen-selftest`: after the last kthread is spawned, every row's pid is asked for its object. The reaper's choice of owed kill is `toyos_proclife::teardown::next_released`, which the kernel and the model both call; the model's own copy is deleted. The model's reaper now parks on ANY_RELEASED the way the kernel's does, a kill's posts, its owe and its return are separate sections, and a new law (L7) holds at every state: the reaper never sleeps while a released kill is owed. The explorer skips a state it has reached before, since every law is a property of the state alone; without that the three-kill chain took 78 s. ANY_RELEASED is posted only on a killed thread's release. The kill mark and the release share one word on the TaskHandle, so a release and a kill posted at once decide by one read-modify-write each: `Hw::release` holds no `TaskShared` whose kill bit it could read. The idle reaper waits on ANY_RELEASED too, which deletes `WORK`, the idle branch and the re-check. The teardown issue is deleted: every handle drop in a teardown either only queues work (a file's write-back) or is a deferred row released from the zero queue, so no stimulus can hold one. The kthread issue is deleted with its fix. Co-Authored-By: Claude Opus 5.5 --- ...-manage-holder-can-kill-a-kernel-thread.md | 25 ------- ...runs-every-kills-teardown-one-at-a-time.md | 32 --------- kernel/src/actuator.rs | 2 +- kernel/src/main.rs | 6 ++ kernel/src/process.rs | 8 ++- kernel/src/reaper.rs | 71 ++++++++----------- kernel/src/sched/kthread.rs | 14 ++++ kernel/src/sched/payload.rs | 30 +++++--- kernel/src/scheduler.rs | 2 +- tests/toyos.rs | 17 +++-- toyos-proclife/src/interleave.rs | 49 +++++++------ toyos-proclife/src/lib.rs | 4 +- toyos-proclife/src/model.rs | 34 ++++++--- toyos-proclife/src/teardown.rs | 7 ++ 14 files changed, 150 insertions(+), 151 deletions(-) delete mode 100644 issues/kernel/a-manage-holder-can-kill-a-kernel-thread.md delete mode 100644 issues/kernel/the-reaper-runs-every-kills-teardown-one-at-a-time.md diff --git a/issues/kernel/a-manage-holder-can-kill-a-kernel-thread.md b/issues/kernel/a-manage-holder-can-kill-a-kernel-thread.md deleted file mode 100644 index 97ddefc72e0..00000000000 --- a/issues/kernel/a-manage-holder-can-kill-a-kernel-thread.md +++ /dev/null @@ -1,25 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# A `MANAGE` holder can kill a kernel thread - -Every kernel thread (`kernel/src/sched/kthread.rs`) holds a process-table -entry and a pid. `SYS_PROCESS_OPEN` (`kernel/src/syscall/proc.rs`'s -`sys_process_open`) mints a `Process` handle for any pid a `SysCap` carrying -`Rights::MANAGE` names, a kernel thread's included, and `SYS_PROCESS_KILL` -then claims its teardown and posts its retire. Not run yet: a kernel thread has no -Ring 3 boundary to die at, so the expected end is the reaper's tripwire halting the machine -(`reaper: pid N not released after 10s`); killing the reaper itself leaves it -owing its own teardown to itself, with the same end. - -Reachable only through init's capability today: init is the one `MANAGE` -holder. It is still userland crashing the kernel. - -## Exit condition - -A kernel thread's pid mints no `Process` handle — `sys_process_open` refuses it -as it refuses a pid that is gone — with a guest test holding `MANAGE` that -opens each kernel thread's pid from the boot's `kthread:` lines and is refused. diff --git a/issues/kernel/the-reaper-runs-every-kills-teardown-one-at-a-time.md b/issues/kernel/the-reaper-runs-every-kills-teardown-one-at-a-time.md deleted file mode 100644 index e641aab1305..00000000000 --- a/issues/kernel/the-reaper-runs-every-kills-teardown-one-at-a-time.md +++ /dev/null @@ -1,32 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# The reaper runs every kill's teardown one at a time - -`kernel/src/reaper.rs` is one kernel thread. It takes whichever owed kill has -every thread released, so one victim that is slow to leave the CPU holds up no -other kill; but the teardown it then runs (`process::finish_kill`, the same -`teardown_tail` as exit: handle drops, mapping frees, the zombie marks, the -publish) runs on that one thread. A kill's exit is published only after every -teardown the reaper took before it has finished. Before the reaper, each killer -ran its own victim's teardown and waited on nothing else. - -Who waits on a publish: `/system/bin/init`'s cut-over (`old.kill(); old.wait()`), -sshd's session end, and test-runner's deadline kill. - -What is measured: with blockd's `--silence-write` holding a write unanswered, -killing the stuck client (or blockd itself) and then an idle process B, B's -kill-to-publish was median 1.04 ms, max 6.4 ms over eight rounds, and in -several rounds B was published before the process killed first. No long -teardown has been measured; the one PR #549's review names is a file's flush -on a USB stick. - -## Exit condition - -A kill's publish waits on no other process's teardown — teardowns run -concurrently, or on the thread that owes them — with a guest test that holds -one victim's teardown (a file whose flush the device does not answer) and -shows a second victim's exit published within its own teardown's time. diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 20fe2123af2..d27375f0e3a 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -535,7 +535,7 @@ actuators! { /// Judged by `partition_claim_gives_up`. partclaim_root_withheld = "partclaim-root-withheld"; - /// Reopen init by pid once it is spawned, the way `SYS_PROCESS_OPEN` does. + /// Reopen init by pid once it is spawned, and open every kernel thread's pid, the way `SYS_PROCESS_OPEN` does. process_reopen_selftest = "process-reopen-selftest"; /// Offer the block layer a second device claiming a registered `DeviceId`, and report what it did with it. diff --git a/kernel/src/main.rs b/kernel/src/main.rs index 8bb7a9eb9f9..28445ae40b4 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -672,6 +672,12 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { iod::start(); reaper::start(); + // Here: the last kernel thread is spawned. + #[cfg(feature = "boot-actuators")] + if actuator::process_reopen_selftest() { + sched::kthread::open_selftest(); + } + smp::set_ready(); // After the release, because a shootdown waits on CPUs that are not diff --git a/kernel/src/process.rs b/kernel/src/process.rs index ad545a1c1fc..911593137bf 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -669,10 +669,14 @@ pub fn try_for_each_thread(mut f: impl FnMut(ThreadCensus<'_>)) -> bool { } /// The object a handle to `pid` would name, for a process still in the table. +/// A kernel thread's pid names none: it has no Ring 3 boundary a kill could end it at. pub fn process_object(pid: Pid) -> Option> { let guard = PROCESS_TABLE.lock(); - let table = guard.as_ref()?; - table.get(pid).map(|proc| Arc::clone(proc.object())) + let proc = guard.as_ref()?.get(pid)?; + if crate::sched::kthread::is_kernel_task(TaskId(pid, proc.main_tid())) { + return None; + } + Some(Arc::clone(proc.object())) } /// Accounting for a process. `None` only in the window between a live process and its published exit (the process being torn down right now). diff --git a/kernel/src/reaper.rs b/kernel/src/reaper.rs index 260e0d2c32d..01e48a3d067 100644 --- a/kernel/src/reaper.rs +++ b/kernel/src/reaper.rs @@ -4,11 +4,12 @@ //! killing the killer, and a killer that waited for its victim would wait on a //! thread that is waiting on it. This thread does the waiting instead: the //! only threads it waits on are killed ones, whose release waits on nothing it -//! does. It takes whichever owed kill is released first, so a slow victim holds -//! up no other kill's release; the teardowns themselves run one at a time. +//! does. Which owed kill it takes is `toyos_proclife::teardown::next_released`; +//! the teardowns themselves run one at a time. use alloc::vec::Vec; +use toyos_proclife::teardown::next_released; use toyos_sched::task::WaitClass; use crate::process::{self, Killed}; @@ -17,67 +18,51 @@ use crate::sched::payload::ANY_RELEASED; use crate::scheduler::{Parkable, RETIRE_GIVE_UP}; use crate::sync::Lock; use crate::time::{Deadline, Instant}; -use crate::watch::{self, Watch}; - -const NAME: &str = "reaper"; +use crate::watch; /// Every owed kill with the instant it was owed, oldest first. static OWED: Lock> = Lock::new(Vec::new()); -/// What an idle reaper waits on. -static WORK: Watch = Watch::new(); - /// Spawns the reaper; call once, from `kernel_main`. pub fn start() { // Halts: a dead reaper leaves every later kill unpublished and nothing to say so. - kthread::spawn(NAME, body, 0, OnPanic::Halt, OnStop::Stops); + kthread::spawn("reaper", body, 0, OnPanic::Halt, OnStop::Stops); } /// Hand a claimed kill, its retires already posted, to the reaper. pub fn owe(killed: Killed) { OWED.lock().push((crate::clock::now(), killed)); - // Both: an idle reaper waits on `WORK`, a busy one on `ANY_RELEASED`, and - // this kill's threads may all be released already. - WORK.post(); + // Its threads may all be released already, and then no release posts again. ANY_RELEASED.post(); } extern "C" fn body(_arg: u64) -> ! { let parkable = Parkable::at_entry(); loop { - process::finish_kill(next_released(&parkable)); - } -} - -/// The oldest owed kill whose every thread is released, once there is one. -fn next_released(parkable: &Parkable) -> Killed { - loop { - let idle = OWED.lock().is_empty(); - let armed = watch::arm(if idle { &WORK } else { &ANY_RELEASED }, 0, WaitClass::Other) - .expect("the reaper is a task"); - let oldest = { + let armed = watch::arm(&ANY_RELEASED, 0, WaitClass::Other).expect("the reaper is a task"); + let (next, oldest) = { let mut owed = OWED.lock(); - if let Some(at) = owed.iter().position(|(_, killed)| killed.released()) { - return owed.remove(at).1; - } - // Armed on the watch that no longer wakes this state. - if owed.is_empty() != idle { - continue; - } - owed.first().map(|(since, killed)| (*since, killed.pid())) + let next = next_released(owed.iter().map(|(_, killed)| killed.released())); + (next.map(|at| owed.remove(at).1), owed.first().map(|(since, killed)| (*since, killed.pid()))) }; - let deadline = match oldest { - None => Deadline::never(), - Some((since, pid)) => { - let give_up = Deadline::at(since + RETIRE_GIVE_UP.duration()); - assert!( - !give_up.reached(crate::clock::now()), - "reaper: pid {pid} not released after {}", - RETIRE_GIVE_UP.duration(), - ); - give_up + match next { + Some(killed) => { + drop(armed); + process::finish_kill(killed); } - }; - watch::wait_uncancellable(parkable, &armed, deadline); + None => { + // The oldest's deadline is the earliest, so it is the only one to check. + let deadline = oldest.map_or(Deadline::never(), |(since, pid)| { + let give_up = Deadline::at(since + RETIRE_GIVE_UP.duration()); + assert!( + !give_up.reached(crate::clock::now()), + "reaper: pid {pid} not released after {}", + RETIRE_GIVE_UP.duration(), + ); + give_up + }); + watch::wait_uncancellable(&parkable, &armed, deadline); + } + } } } diff --git a/kernel/src/sched/kthread.rs b/kernel/src/sched/kthread.rs index dcb381be306..92af04371cd 100644 --- a/kernel/src/sched/kthread.rs +++ b/kernel/src/sched/kthread.rs @@ -126,6 +126,20 @@ pub fn runs_through_the_stop(id: TaskId) -> bool { .any(|row| row.task.load(Ordering::Acquire) == packed && !row.stops.load(Ordering::Relaxed)) } +/// Control for `process::process_object`'s refusal: no kernel thread's pid names a process a handle could hold. +#[cfg(feature = "boot-actuators")] +pub fn open_selftest() { + let pids: Vec<_> = ROWS + .iter() + .map(|row| row.task.load(Ordering::Acquire)) + .filter(|&task| task != NO_TASK && task != CLAIMING) + .map(|task| TaskId::unpack(task).0) + .collect(); + let opened = pids.iter().filter(|&&pid| crate::process::process_object(pid).is_some()).count(); + let verdict = if !pids.is_empty() && opened == 0 { "PASS" } else { "FAIL" }; + crate::log!("process-open-kthread: {verdict} ({} kernel threads, {opened} opened)", pids.len()); +} + /// Whether a panic on the running task recovers; `None` unless it is a kernel thread. pub fn panic_recovers_here() -> Option { Some(current_row()?.recoverable.load(Ordering::Relaxed) != 0) diff --git a/kernel/src/sched/payload.rs b/kernel/src/sched/payload.rs index fd3fb48594a..5d0773f473a 100644 --- a/kernel/src/sched/payload.rs +++ b/kernel/src/sched/payload.rs @@ -3,7 +3,7 @@ //! released exactly once, by `Hw::release`. use alloc::sync::Arc; -use core::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, Ordering}; +use core::sync::atomic::{AtomicU8, AtomicU32, AtomicU64, Ordering}; use toyos_sched::fair::{FairShare, ShareState}; use toyos_sched::hw::Nanos; @@ -77,17 +77,24 @@ pub const SCHED_READY: u8 = 1; pub const SCHED_BLOCKED: u8 = 2; pub const SCHED_UNKNOWN: u8 = 3; -/// Posted after every thread's release, for a waiter on whichever of many threads goes first. +/// Posted after a killed thread's release, for a waiter on whichever of many threads goes first. pub static ANY_RELEASED: Watch = Watch::new(); +/// [`TaskHandle`]'s `fate` bits. +const RELEASED: u8 = 1; +const KILLED: u8 = 2; + /// What a thread other than the one running can be asked about; published here since a `CpuSched` is `!Sync` and unreachable remotely. pub struct TaskHandle { cpu_ns: AtomicU64, /// Dispatch timestamp while running, 0 otherwise; a reader adds the live slice itself. running_since: AtomicU64, acct: Lock, - /// Set by `Hw::release`; the one fact a retirer needs, that the thread is off every CPU. - released: AtomicBool, + /// `RELEASED`, set by `Hw::release`: the one fact a retirer needs, that the thread is off + /// every CPU. `KILLED`, set by `scheduler::post_retire`. One word, so the two decide by one + /// read-modify-write each whether the release posts [`ANY_RELEASED`]: `Hw::release` holds + /// no `TaskShared` whose kill bit it could read. + fate: AtomicU8, /// What another thread arms on to be told this one moved: its exit, for `SYS_THREAD_JOIN`, and its release, for the retirer. watch: Watch, /// Cancels reported to this thread; a second one means a caller swallowed the first, so it panics rather than spinning. @@ -102,7 +109,7 @@ impl TaskHandle { cpu_ns: AtomicU64::new(0), running_since: AtomicU64::new(0), acct: Lock::new(TaskAccounting::default()), - released: AtomicBool::new(false), + fate: AtomicU8::new(0), watch: Watch::new(), cancels: AtomicU32::new(0), operation: OperationSlot::new(), @@ -124,15 +131,22 @@ impl TaskHandle { /// Announces the death only after the payload is dropped; that ordering is the guarantee a retirer's park buys. pub(crate) fn publish_released(&self) { - self.released.store(true, Ordering::Release); + let fate = self.fate.fetch_or(RELEASED, Ordering::AcqRel); // The retirer arms on this thread's own watch, the same subject a joiner uses. self.watch.post(); - ANY_RELEASED.post(); + if fate & KILLED != 0 { + ANY_RELEASED.post(); + } } /// Has `Hw::release` run for this thread? The retire wait's condition. pub fn released(&self) -> bool { - self.released.load(Ordering::Acquire) + self.fate.load(Ordering::Acquire) & RELEASED != 0 + } + + /// Mark the thread killed; `false` if it is already released, and so has no retire to post. + pub fn mark_killed(&self) -> bool { + self.fate.fetch_or(KILLED, Ordering::AcqRel) & RELEASED == 0 } /// Where this thread's establishment lives; `scheduler::Operation` owns every rule about it. diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index 2c673756886..a8a697d3a83 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -503,7 +503,7 @@ pub fn futex_wake(phys_addr: DirectMap, count: usize) -> u64 { /// Set a thread's kill bit and ask its CPU for a safe point; returns at once. pub fn post_retire(sched: &ThreadSched) { - if sched.handle.released() { + if !sched.handle.mark_killed() { return; } preempt_off(|p| { diff --git a/tests/toyos.rs b/tests/toyos.rs index 1e413c4835a..d55a9d1a707 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -16810,18 +16810,21 @@ fn pci_cap_selftest(log: &str) -> Result<(), String> { Ok(()) } -/// The kernel reopens init by pid after the last handle to it has gone. +/// The kernel reopens init by pid after the last handle to it has gone, and +/// no kernel thread's pid opens. /// /// Text in, a verdict out: every line it reads is a kernel record, so the /// T14's readback and a QEMU boot log are judged by this one predicate. fn process_reopen(log: &str) -> Result<(), String> { - let Some(verdict) = log.lines().find(|l| l.contains("process-reopen:")) else { - return Err(format!("the reopen control never ran:\n{log}")); - }; - if !verdict.contains("PASS") { - return Err(format!("{}\n{log}", verdict.trim())); + for control in ["process-reopen:", "process-open-kthread:"] { + let Some(verdict) = log.lines().find(|l| l.contains(control)) else { + return Err(format!("{control} never ran:\n{log}")); + }; + if !verdict.contains("PASS") { + return Err(format!("{}\n{log}", verdict.trim())); + } + eprintln!(" [process] {}", verdict.trim()); } - eprintln!(" [process] {}", verdict.trim()); Ok(()) } diff --git a/toyos-proclife/src/interleave.rs b/toyos-proclife/src/interleave.rs index 894446f1b35..30f832d6fdf 100644 --- a/toyos-proclife/src/interleave.rs +++ b/toyos-proclife/src/interleave.rs @@ -20,6 +20,7 @@ use alloc::string::String; use alloc::vec; use alloc::vec::Vec; +use std::collections::HashSet; use crate::model::World; use crate::table::{Lifecycle, Processes}; @@ -30,7 +31,7 @@ use crate::{join, reap, spawn, teardown, Pid, ThreadLocation, Tid, Watch}; /// Each variant's `pc` is the number of lock sections it has completed, and /// every field beside it is a value the real path carries in a local across a /// lock release — which is the whole reason a window exists to explore. -#[derive(Clone, Debug)] +#[derive(Clone, Debug, PartialEq, Eq, Hash)] pub enum Op { /// `process::release_process` + `teardown_tail`: a thread ending its own /// process. @@ -54,7 +55,7 @@ pub enum Op { /// A claimed kill's teardown past its claim: wait out every thread's release, /// mark, publish. -#[derive(Clone, Debug)] +#[derive(Clone, Debug, PartialEq, Eq, Hash)] pub struct Finish { pid: Pid, code: i32, @@ -150,11 +151,11 @@ impl Op { Op::IdlePass { pc: 0 } } - /// Whether the op has nothing left to do: the reaper is done whenever - /// nothing is owed to it. + /// Whether the op has nothing left to do: the reaper is done once it is + /// parked with nothing owed to it. fn done(&self, world: &World) -> bool { match self { - Op::Reap { job } => job.is_none() && world.nothing_owed(), + Op::Reap { job } => job.is_none() && world.reaper_asleep() && world.nothing_owed(), Op::Exit { pc, .. } | Op::Kill { pc, .. } | Op::Spawn { pc, .. } @@ -172,7 +173,7 @@ impl Op { } Op::Kill { inline: Some(finish), .. } => finish.enabled(world), Op::Reap { job: Some(finish) } => finish.enabled(world), - Op::Reap { job: None } => world.owes_a_released(), + Op::Reap { job: None } => !world.reaper_asleep(), _ => true, } } @@ -264,13 +265,16 @@ impl Op { *pc = 1; } } - } else { - // Every retire posted, the wait handed on, and the kill - // returns. + } else if *pc == 1 { for &tid in tids.iter() { world.post_retire(*pid, tid); } + *pc = 2; + } else if *pc == 2 { world.owe(*pid, *code, core::mem::take(tids)); + *pc = if by.is_some() { 3 } else { DONE }; + } else { + // The kill returned, and its thread reaches its safe point. *pc = DONE; } if *pc == DONE { @@ -280,7 +284,7 @@ impl Op { } } Op::Reap { job } => match job { - None => *job = Some(Finish::released(world.take_released().expect("enabled with none released"))), + None => *job = world.reaper_takes().map(Finish::released), Some(finish) => { if finish.step(world) { *job = None; @@ -374,13 +378,14 @@ impl Op { const DONE: u32 = u32::MAX; -/// How many schedules ran, or the first one that breaks a law. +/// How many distinct states every schedule reaches, or the first schedule that breaks a law. /// /// Depth-first over "which op runs its next lock section", checking /// `World::faults` at every state and `World::final_faults` at every leaf; a -/// state where no unfinished op can move is a deadlock. The returned string is -/// the schedule that produced it, in the order the ops ran. -pub fn explore(initial: &World, ops: &[Op]) -> Result { +/// state where no unfinished op can move is a deadlock. A state reached before +/// is not walked again: every law is a property of the state alone. The +/// returned string is the schedule that produced it, in the order the ops ran. +pub fn explore(initial: &World, ops: &[Op]) -> Result { let mut world = initial.clone(); for op in ops { if let Op::Kill { by: Some(by), .. } = op { @@ -388,16 +393,18 @@ pub fn explore(initial: &World, ops: &[Op]) -> Result { } } let mut trace = Vec::new(); - let mut schedules = 0; - match walk(world, ops.to_vec(), &mut trace, &mut schedules) { + let mut seen = HashSet::new(); + match walk(world, ops.to_vec(), &mut trace, &mut seen) { Some(found) => Err(found), - None => Ok(schedules), + None => Ok(seen.len()), } } -fn walk(world: World, ops: Vec, trace: &mut Vec, schedules: &mut u64) -> Option { +fn walk(world: World, ops: Vec, trace: &mut Vec, seen: &mut HashSet<(World, Vec)>) -> Option { + if !seen.insert((world.clone(), ops.clone())) { + return None; + } if ops.iter().all(|op| op.done(&world)) { - *schedules += 1; let faults = world.final_faults(); return report(&faults, trace); } @@ -417,7 +424,7 @@ fn walk(world: World, ops: Vec, trace: &mut Vec, schedules: &mut u64 trace.pop(); return Some(found); } - if let Some(found) = walk(next_world, next_ops, trace, schedules) { + if let Some(found) = walk(next_world, next_ops, trace, seen) { trace.pop(); return Some(found); } @@ -702,7 +709,7 @@ mod tests { Op::reap(), ]; match explore(&world, &ops) { - Ok(schedules) => std::println!("KillEachOther: {schedules} schedules, every one ends"), + Ok(states) => std::println!("KillEachOther: {states} states, every schedule ends"), Err(found) => panic!("a lifecycle law broke:\n{found}"), } } diff --git a/toyos-proclife/src/lib.rs b/toyos-proclife/src/lib.rs index 4c5a7d38c53..8db12919ff7 100644 --- a/toyos-proclife/src/lib.rs +++ b/toyos-proclife/src/lib.rs @@ -77,7 +77,7 @@ pub use toyos_abi::{Pid, Tid}; /// /// For a live thread the scheduler is authoritative about running, ready or /// blocked — `scheduler::task_sched_state()` has that detail. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] +#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)] pub enum ThreadLocation { /// Alive: running, ready, or blocked. The scheduler owns the detail. Scheduled, @@ -107,7 +107,7 @@ impl ThreadLocation { /// whole of the distinction a real defect turned on: `process::thread_exit` /// posted one wake and it was always the process's main thread, so a non-main /// thread joining a sibling was owed a wake nobody sent. -#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Debug)] +#[derive(Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Debug)] pub enum Watch { /// One thread's exit. `SYS_THREAD_JOIN` arms here. Thread(Pid, Tid), diff --git a/toyos-proclife/src/model.rs b/toyos-proclife/src/model.rs index ef4daf0b3cd..413d994618b 100644 --- a/toyos-proclife/src/model.rs +++ b/toyos-proclife/src/model.rs @@ -17,10 +17,10 @@ use alloc::string::String; use alloc::vec::Vec; use crate::table::{Lifecycle, Processes}; -use crate::{Pid, ThreadLocation, Tid, Watch}; +use crate::{teardown, Pid, ThreadLocation, Tid, Watch}; /// One process, as its lifecycle sees it. -#[derive(Clone)] +#[derive(Clone, PartialEq, Eq, Hash)] pub struct ModelProc { main_tid: Tid, tearing_down: bool, @@ -61,7 +61,7 @@ impl Lifecycle for ModelProc { } /// The table, plus the effects the kernel would have performed. -#[derive(Clone)] +#[derive(Clone, PartialEq, Eq, Hash)] pub struct World { procs: BTreeMap, next_pid: Pid, @@ -85,6 +85,8 @@ pub struct World { in_kernel: BTreeSet<(Pid, Tid)>, /// Teardowns a kill handed to the reaper, oldest first: the process, its code, its threads. owed: VecDeque<(Pid, i32, Vec)>, + /// The reaper is parked on `ANY_RELEASED` and nothing has posted it since it armed. + reaper_asleep: bool, /// Entries an idle pass has taken out of the table. reaped: BTreeSet, /// TLS blocks a spawn's phase 2 mapped and no thread owns yet. @@ -124,6 +126,7 @@ impl World { killed: BTreeSet::new(), in_kernel: BTreeSet::new(), owed: VecDeque::new(), + reaper_asleep: true, reaped: BTreeSet::new(), tls_mapped: BTreeSet::new(), next_tls: 0, @@ -227,9 +230,14 @@ impl World { /// — "the same subject a joiner uses, and the reason the release no longer /// needs a queue of its own" (`kernel/src/sched/payload.rs`). A model /// without it strands joiners the kernel releases. + /// + /// A killed thread's release also posts `ANY_RELEASED`, the reaper's watch. pub fn retire(&mut self, pid: Pid, tid: Tid) { self.retired.insert((pid, tid)); self.post(Watch::Thread(pid, tid)); + if self.killed.contains(&(pid, tid)) { + self.reaper_asleep = false; + } } /// A thread that has committed to leaving but whose payload the exit pass @@ -277,16 +285,18 @@ impl World { /// `reaper::owe`. pub fn owe(&mut self, pid: Pid, code: i32, tids: Vec) { self.owed.push_back((pid, code, tids)); + self.reaper_asleep = false; } - /// The reaper's take: the oldest owed kill whose every thread is retired. - pub fn take_released(&mut self) -> Option<(Pid, i32, Vec)> { - let at = self.owed.iter().position(|owed| self.all_retired(owed))?; - self.owed.remove(at) + /// The reaper's check, after it armed: the kill it takes, or `None` and it parks. + pub fn reaper_takes(&mut self) -> Option<(Pid, i32, Vec)> { + let at = teardown::next_released(self.owed.iter().map(|owed| self.all_retired(owed))); + self.reaper_asleep = at.is_none(); + self.owed.remove(at?) } - pub fn owes_a_released(&self) -> bool { - self.owed.iter().any(|owed| self.all_retired(owed)) + pub fn reaper_asleep(&self) -> bool { + self.reaper_asleep } fn all_retired(&self, (pid, _, tids): &(Pid, i32, Vec)) -> bool { @@ -364,6 +374,12 @@ impl World { } } } + // L7. The reaper never sleeps while a kill it could finish is owed. + if self.reaper_asleep { + if let Some((pid, _, _)) = self.owed.iter().find(|owed| self.all_retired(owed)) { + out.push(alloc::format!("the reaper sleeps while pid {pid}'s released kill is owed")); + } + } out } diff --git a/toyos-proclife/src/teardown.rs b/toyos-proclife/src/teardown.rs index 13fec0d9605..f936c6c1520 100644 --- a/toyos-proclife/src/teardown.rs +++ b/toyos-proclife/src/teardown.rs @@ -83,6 +83,13 @@ pub fn kill_set(proc: &P) -> Vec { tids } +/// Which owed kill the reaper finishes next, given whether each one's threads +/// are all released, oldest first: the oldest released one, so a victim slow to +/// leave the CPU holds up no other kill. +pub fn next_released(released: impl IntoIterator) -> Option { + released.into_iter().position(|r| r) +} + /// Where a retired thread's CPU time is charged. /// /// The two are not interchangeable: the main thread's is what a process's exit From fa1e3254de3d4d462dad96a3c2b22c78863f9fed Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 20:13:13 +0200 Subject: [PATCH 06/19] kernel: the last thread out tears its process down; no reaper, no retire wait A kill claims its victim, posts every thread's retire and returns. Each thread leaves by its own hand (`process::leave`): at its Ring 3 exit boundary when killed, in `exit` or `thread_exit` otherwise. It leaves its address space, folds its blocked-time accounting into the process's and marks itself out under the table lock, and the one whose leaving empties a claimed process frees the process and publishes its exit on its own stack. Its kernel stack goes as every stack does, in `Hw::release` on the pass after the switch. Nobody waits for another thread's release, so a kill cannot deadlock on its victim by construction. Deleted: `kernel/src/reaper.rs` and its kernel thread, `ANY_RELEASED`, the owed queue, `next_released`, the `OnStop` row, `scheduler::await_released` and `RETIRE_GIVE_UP` with the reaper's tripwire, `TaskHandle`'s released flag and accounting copy, `ProcessAccounting::child_threads_cpu_ns`, and the teardown's thread-by-thread fold. `issues/kernel/retire-tripwire-is-not- queue-shaped.md` closes: its constant is deleted and nothing replaces it. The decisions are `toyos-proclife`'s and the kernel calls them: `claim_teardown` now carries the exit code, `retire_set` is the claim's retire set, and `leave` answers whether the thread leaving is the last one out. A poisoned thread's death is its leaving, done for it by the idle loop: a poisoned main thread claims its process and retires the rest, and a poisoned last thread publishes the claim's code. A join does not collect a thread of a process being torn down: that thread's TLS is still mapped where its siblings may run. The model drops the reaper and scripts every thread's way out as its own steps. Laws at every state: one claim and one teardown per process, nothing freed or published while a thread is still in, a stack freed only after the switch, no mapped TLS dropped under a running sibling, and no kill or exit waiting on another process's teardown; at every leaf, every claimed process published and every killed thread gone. New controls: `mutate-first-out-tears-down` and `mutate-join-collects-in-a-teardown`. `kill_ends_every_wait` kills a child parked in a futex, a poll ring, a process wait, a thread join and a sleep, and waits for each to end. Co-Authored-By: Claude Opus 5.5 --- ...ller-panics-on-its-second-drain-backoff.md | 22 + ...reed-mapped-when-its-joiner-collects-it.md | 26 + .../deferred-release-outlives-its-syscall.md | 6 +- .../retire-tripwire-is-not-queue-shaped.md | 92 --- .../kernel/scheduler-pass-blocks-in-xhci.md | 11 +- kernel/src/arch/x86_64/hw.rs | 8 +- kernel/src/block.rs | 7 +- kernel/src/drivers/xhci/usbd.rs | 4 +- kernel/src/iod.rs | 4 +- kernel/src/log/console.rs | 4 +- kernel/src/log/nested.rs | 4 +- kernel/src/log/storm.rs | 4 +- kernel/src/main.rs | 2 - kernel/src/process.rs | 328 +++------ kernel/src/quiesce.rs | 11 +- kernel/src/reaper.rs | 68 -- kernel/src/sched/dump.rs | 2 +- kernel/src/sched/kthread.rs | 52 +- kernel/src/sched/payload.rs | 44 +- kernel/src/scheduler.rs | 115 +-- kernel/src/time.rs | 4 - src/ci.rs | 6 + tests/common/metal.rs | 2 +- tests/common/power.rs | 11 +- .../src/bin/kill_ends_every_wait.rs | 122 ++++ toyos-proclife/Cargo.toml | 15 +- toyos-proclife/src/interleave.rs | 663 +++++++----------- toyos-proclife/src/join.rs | 18 + toyos-proclife/src/model.rs | 333 +++++---- toyos-proclife/src/poison.rs | 118 ++-- toyos-proclife/src/spawn.rs | 12 +- toyos-proclife/src/table.rs | 13 +- toyos-proclife/src/teardown.rs | 296 +++----- toyos-quiesce/src/lib.rs | 15 +- toyos-sched/sim/src/explore.rs | 5 +- toyos-sched/sim/src/invariants.rs | 26 +- toyos-sched/sim/src/vm.rs | 13 +- toyos-sched/sim/src/workload.rs | 3 +- toyos-sched/sim/tests/scenarios.rs | 11 +- toyos-sched/src/cpu.rs | 38 +- toyos-sched/src/park.rs | 3 +- 41 files changed, 1052 insertions(+), 1489 deletions(-) create mode 100644 issues/kernel/a-killed-shutdown-caller-panics-on-its-second-drain-backoff.md create mode 100644 issues/kernel/a-poisoned-threads-tls-is-freed-mapped-when-its-joiner-collects-it.md delete mode 100644 issues/kernel/retire-tripwire-is-not-queue-shaped.md delete mode 100644 kernel/src/reaper.rs create mode 100644 tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs diff --git a/issues/kernel/a-killed-shutdown-caller-panics-on-its-second-drain-backoff.md b/issues/kernel/a-killed-shutdown-caller-panics-on-its-second-drain-backoff.md new file mode 100644 index 00000000000..26e9490cc79 --- /dev/null +++ b/issues/kernel/a-killed-shutdown-caller-panics-on-its-second-drain-backoff.md @@ -0,0 +1,22 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A killed shutdown caller panics on its second drain backoff + +`writeback::drain_all` (`kernel/src/writeback.rs`) retries a refused drain +through `block::between_attempts`, whose park is cancellable, and discards what +it answers. A thread whose kill bit is set gets `Cancelled` from that park at +once, retries, and parks again; `TaskHandle::take_cancel` asserts on the second +cancel reported to one thread, so the kernel panics. `ops::until_answered` +reads the kill bit before it parks and does not have this; `drain_all` does not. + +The caller is the shutdown syscall (`syscall/machine.rs`), which a kill posted +before `quiesce::stop` can still reach, and it takes a second backoff only when +the device refuses two drain attempts on budget. Not reproduced. + +Exit condition: `drain_all`'s retry either stops parking once its caller is +killed or parks uncancellably, with a test that kills a caller parked in a +refused drain. diff --git a/issues/kernel/a-poisoned-threads-tls-is-freed-mapped-when-its-joiner-collects-it.md b/issues/kernel/a-poisoned-threads-tls-is-freed-mapped-when-its-joiner-collects-it.md new file mode 100644 index 00000000000..b3c79772394 --- /dev/null +++ b/issues/kernel/a-poisoned-threads-tls-is-freed-mapped-when-its-joiner-collects-it.md @@ -0,0 +1,26 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A poisoned thread's TLS is freed still mapped when its joiner collects it + +A thread that dies in panic recovery is taken out of its process by +`poison::zombify_poisoned` from the idle loop, which frees nothing: its +`ThreadData`, and the `MappedPages` of its TLS block in it, stay in its +`ThreadEntry`. When the process is not being torn down, a sibling's +`SYS_THREAD_JOIN` collects that zombie (`join::collect_zombie`) and drops the +entry, and `MappedPages` dropped without its unmap returns the frames to the +PMM while the process's page tables still map them and its other threads still +run. A thread's own `SYS_THREAD_EXIT` unmaps first (`process::release_thread`); +the poisoned one never runs again to do so. + +Reached only through a kernel panic recovered inside a multi-threaded process's +syscall. `toyos-proclife`'s model reports it as law L8 for any schedule that +collects a mapped thread with a sibling still in the process; no test scripts a +poisoned sibling and its joiner. + +Exit condition: a poisoned thread's mappings are released before its entry can +be collected, or its entry is not collectable, with a model test that scripts +the poisoned sibling and its joiner. diff --git a/issues/kernel/deferred-release-outlives-its-syscall.md b/issues/kernel/deferred-release-outlives-its-syscall.md index 3ed48138262..26e80bed5ef 100644 --- a/issues/kernel/deferred-release-outlives-its-syscall.md +++ b/issues/kernel/deferred-release-outlives-its-syscall.md @@ -196,11 +196,7 @@ at all.** It belongs with that track, not beside it. ### What "give the batch an owner" costs, worked out 2026-08-20 -An owner has to be the *thread*, not the CPU. `kill_process` phase 2 calls -`scheduler::retire_task`, which parks the killer until the victim's record is -released, so the killing thread can be moved to another CPU between the -`close_all` that queues its objects and the syscall exit that would drain them — -a per-CPU list would strand exactly the batch it was added to own. +An owner has to be the *thread*, not the CPU. A per-thread list cannot live behind `ThreadData`'s lock either: `teardown_resources` holds `ProcessData` across `close_all`, and its own first diff --git a/issues/kernel/retire-tripwire-is-not-queue-shaped.md b/issues/kernel/retire-tripwire-is-not-queue-shaped.md deleted file mode 100644 index 625c71bb77e..00000000000 --- a/issues/kernel/retire-tripwire-is-not-queue-shaped.md +++ /dev/null @@ -1,92 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-08-16 ---- - -# `retire_task`'s tripwire is a constant against a term the workload sets - -**This waits on a track, not on a decision:** the shape that closes it is -`issues/kernel/cpu-time-is-a-band-and-not-a-reservation.md`'s dying-server -chunk (`:32`, with the kill path's report and its one fixed-hop tripwire at -`:51`), so no audit should list this as ready to work. - -`kernel/src/scheduler.rs`'s `GIVE_UP` is a `Tripwire` — a constant whose expiry -is a kernel panic. Its own derivation carries a term that is not constant: - -> Times `1 + peers`, because one CPU runs one unwind at a time and this victim -> waits out the corpses queued ahead of it. Priced at `peers = 8`. - -`peers` is the number of *other* killed tasks in `CpuSched::dying` on the -victim's CPU. Nothing bounds it, and every corpse past the priced eight adds -another `(1 + peers)`-th of the unwind term — 110 ms each on a saturated -real-time band, 10 ms on an idle one. The panic is reachable from a workload -that broke no rule. - -**Two things this file previously said about that term are wrong, and both are -corrected here rather than left for the next reader to re-derive.** - -*The trigger.* The shape named was "one process's threads torn down together -onto one CPU", and this kernel cannot produce it. `kill_process` and the exit -path both loop over a process's tids calling `scheduler::retire_task`, which -blocks until the victim has been released — so one process teardown holds at -most one corpse at a time on any CPU, however many threads the process has. The -producer of `peers > 0` is *concurrent independent retirers*: separate killer -threads retiring separate victims that happen to share a CPU. That is unbounded -in exactly the way the term needs, so the defect stands; only its stated cause -was wrong, and none of the remedies below is aimed differently because of it, -since all three bound the depth rather than the producer. - -*The crossing point.* With fixed terms of 8.02 s and 110 ms per additional -corpse, the sum is 8.02 + 0.110 × N seconds: it equals the derivation's own -priced 9.01 s at N = 9, and first reaches the 10 s constant at N = 18. Nine is -the number of *further* corpses the 990 ms margin buys, not the total count at -which the constant is crossed — the two readings were conflated, understating -the crossing point by roughly a factor of two, and the earlier text stated both -readings in adjacent sentences. - -The simulator states the same term honestly, because it can: `invariants.rs`'s -`retire_latency_bound` takes `peers` as a parameter and reads it off the run, the -way invariant I5's bound takes the runnable thread count. A wall clock in the -kernel has nothing to read it off. - -**This predates bounded deferral and is not caused by it.** The -`(1 + peers) × UNWIND_NS` term entered the sim's I14 in the completion work's -first wave and the kernel-side derivation never priced `peers` at all until the -second. Aging multiplies the term by 11 under a saturated RT band, which makes -the crossing -point closer but does not create it. - -Three shapes have been considered and none is this chunk's to choose: - -1. **Bound the dying list.** A CPU that already holds *k* corpses refuses the - *k+1*-th and the retire places it elsewhere. `hand_off` currently refuses to - migrate a killed task for invariant 7's promptness reason, so this is a - change to invariant 7 and not to a constant. -2. **Make the wait queue-shaped.** `retire_task` reads the victim CPU's - `dying_len()` at arm time and scales its own deadline. That turns a - `Tripwire` into something `kernel/src/time.rs` has no kind for — the type - deliberately forbids a magnitude with a derivation attached. -3. **Stop waiting.** The wait exists because process teardown frees memory the - dead thread's page tables still map. Revisiting that belongs to the - completion architecture, which kills every wait. - -**Remedy 2 was chosen, then withdrawn, and what replaced it was withdrawn -too.** The 2026-08-16 five-lens review of the reservation design -proved the arm-time depth read is the same defect again: the snapshot is taken -before the victim reaches the queue, so k concurrent retirers all read a depth of -zero and the k-th victim legally outlives a deadline whose expiry was a panic — -and the read itself is one scheduler-core invariant 2 forbids. The revision put -two assertions on the victim's CPU in its place; the 2026-08-17 second pass -proved both reachable from legal userland (a corpse parked in a 2 s transfer -against a queue-occupancy condition, and a 32 MiB dirty file against a progress -cadence), and the design now asserts **nothing** at the victim's end. What the -wait rests on instead is the dying server's admitted reservation — one entity, -one invariant, no term any workload scales — plus a report when a corpse's tenure -exceeds a derived expectation, which is loud and never fatal. None of the three -remedies above is the shape that landed. **This file closes when that change -lands, for the strongest reason available: the constant is deleted and nothing -was put in its place.** - -Until then the constant is honest about what it does not cover, which is the -whole of what this file records. diff --git a/issues/kernel/scheduler-pass-blocks-in-xhci.md b/issues/kernel/scheduler-pass-blocks-in-xhci.md index c1885c4899c..9478a91fdd2 100644 --- a/issues/kernel/scheduler-pass-blocks-in-xhci.md +++ b/issues/kernel/scheduler-pass-blocks-in-xhci.md @@ -23,9 +23,7 @@ sits on the wrong side of the boundary the budget describes. What a CPU inside that recovery holds is *every message addressed to it*: an `Adopt` carrying a task, a `Wake` for a parked thread, a `Retire`. Nothing in the scheduler can shorten it — every reap and every wake is bounded by the owning -CPU's pass latency by design, which is exactly why the design is sound. The one -thing in the tree that notices is `scheduler::retire_task`'s 1 s guard, and it -notices by panicking: +CPU's pass latency by design, which is exactly why the design is sound. ``` retire_task: task not released after 1s: InTransit(CpuId(1)) @@ -33,10 +31,7 @@ retire_task: task not released after 1s: InTransit(CpuId(1)) That panic fired on the owner's T14 at 949.792 s of uptime with doom exiting. The *balance*-path half of it is fixed: `hand_off` reaps a killed task rather than -handing it on, gated by simulator invariant I14. This half is not, -and it would produce the same panic with `Blocked(CpuId(n))` in the message -instead — the guard cannot tell a lost message from a busy CPU, which is what it -is written as if it could. +handing it on, gated by simulator invariant I14. This half is not. The second instance of the same shape used to be the idle loop's log flush; the idle loop touches no filesystem now — the log is logd's file — and the @@ -48,7 +43,7 @@ that `drain_irqs` only ever does work it can finish: drain the event ring, dispatch HID reports, note that a port or an endpoint owes work. The debounce and the port reset were already moved off this path for exactly this reason (CLAUDE.md, USB hotplug); the control transfers inside `configure` and `recover_endpoints` -were not. Until then, `retire_task`'s bound is measuring the USB bus. +were not. **And the budget cannot see it.** `cpu::MAX_PASS_NS` is measured against by `SchedPass::finish`, from the `now` the pass was entered with to the end of diff --git a/kernel/src/arch/x86_64/hw.rs b/kernel/src/arch/x86_64/hw.rs index 902433dee72..09fc58478f3 100644 --- a/kernel/src/arch/x86_64/hw.rs +++ b/kernel/src/arch/x86_64/hw.rs @@ -472,12 +472,8 @@ impl Hw for KernelHw { } /// Reached once per task, from a later pass running on another stack, so dropping `payload` - /// here never frees the stack this call stands on; `publish_released` must be last — a - /// retirer's wait ends only once this drop has happened. + /// here never frees the stack this call stands on. fn release(&self, _key: TaskKey, payload: KernelPayload, acct: TaskAccounting) { - let handle = payload.handle.clone(); - handle.finalize(acct); - drop(payload); - handle.publish_released(); + payload.handle.finalize(acct); } } diff --git a/kernel/src/block.rs b/kernel/src/block.rs index 247cfb1b000..41dcdefd388 100644 --- a/kernel/src/block.rs +++ b/kernel/src/block.rs @@ -132,12 +132,9 @@ pub fn userland_operations() -> (u32, u64) { } /// Whether the stop this count is read for stops the thread opening an -/// operation now. +/// operation now: it stops every userland thread and no kernel thread. fn counted() -> bool { - let (Some(pid), Some(tid)) = (crate::arch::percpu::current_pid(), crate::arch::percpu::current_tid()) else { - return false; - }; - !crate::sched::kthread::runs_through_the_stop(crate::scheduler::TaskId(pid, tid)) + !crate::sched::kthread::current_is_kernel_thread() && crate::arch::percpu::current_pid().is_some() } #[must_use = "the operation lasts exactly as long as this guard"] diff --git a/kernel/src/drivers/xhci/usbd.rs b/kernel/src/drivers/xhci/usbd.rs index b96c1cea525..ab36080e598 100644 --- a/kernel/src/drivers/xhci/usbd.rs +++ b/kernel/src/drivers/xhci/usbd.rs @@ -3,7 +3,7 @@ use toyos_sched::task::WaitClass; use crate::watch; -use crate::sched::kthread::{self, OnPanic, OnStop}; +use crate::sched::kthread::{self, OnPanic}; use crate::scheduler; use crate::time::Deadline; @@ -13,7 +13,7 @@ const NAME: &str = "usbd"; /// Start the thread. Called once, from `kernel_main`, beside `klogd`'s. pub fn start() { // Unconditional: gating this on a controller would give the kernel's thread count a second answer. - let _ = kthread::spawn(NAME, body, 0, OnPanic::Recover, OnStop::Runs); + let _ = kthread::spawn(NAME, body, 0, OnPanic::Recover); } extern "C" fn body(_arg: u64) -> ! { diff --git a/kernel/src/iod.rs b/kernel/src/iod.rs index ab99bfc24e2..600d9e32cc5 100644 --- a/kernel/src/iod.rs +++ b/kernel/src/iod.rs @@ -3,7 +3,7 @@ use toyos_sched::task::WaitClass; use crate::watch; -use crate::sched::kthread::{self, OnPanic, OnStop}; +use crate::sched::kthread::{self, OnPanic}; use crate::scheduler; use crate::time::Deadline; @@ -13,7 +13,7 @@ const NAME: &str = "iod"; /// Spawns the `iod` kthread; call once, from `kernel_main`. pub fn start() { // Recoverable: unlike klogd's silent loss, a killed iod's stalled write-back is visible to SYS_FSYNC and logd. - let _ = kthread::spawn(NAME, body, 0, OnPanic::Recover, OnStop::Runs); + let _ = kthread::spawn(NAME, body, 0, OnPanic::Recover); } extern "C" fn body(_arg: u64) -> ! { diff --git a/kernel/src/log/console.rs b/kernel/src/log/console.rs index 5c21bb58e3b..4e9b1a23451 100644 --- a/kernel/src/log/console.rs +++ b/kernel/src/log/console.rs @@ -25,7 +25,7 @@ use toyos_sched::park::notify; use crate::drivers::serial::{self, BackendGuard, MAX_CONSOLE_LINE}; use crate::hw::HW; use crate::sched::driver::{cpus, irq_off}; -use crate::sched::kthread::{self, OnPanic, OnStop}; +use crate::sched::kthread::{self, OnPanic}; use crate::sleeplock::SleepGuard; use crate::watch; use crate::sched::payload::KShared; @@ -67,7 +67,7 @@ static DRAINED: Published = Published::new(); /// Start the thread. Called once, from `kernel_main`, before the scheduler starts. /// Placement matters: APs spin until the machine is released, so an earlier spawn could not run while the machine has no console. pub fn start() { - let sched = kthread::spawn(NAME, body, 0, OnPanic::Halt, OnStop::Runs); + let sched = kthread::spawn(NAME, body, 0, OnPanic::Halt); // Leaked: `klogd` never exits, and a producer reading this pointer under lock may not touch a refcount. let shared: &'static Arc = alloc::boxed::Box::leak(alloc::boxed::Box::new(sched.shared)); KLOGD.store(shared as *const _ as *mut _, Ordering::Release); diff --git a/kernel/src/log/nested.rs b/kernel/src/log/nested.rs index 699bbc4a12f..5e80f015d60 100644 --- a/kernel/src/log/nested.rs +++ b/kernel/src/log/nested.rs @@ -16,7 +16,7 @@ mod armed { use core::sync::atomic::{AtomicBool, Ordering}; use crate::log::shard::SHARD_RECORDS; - use crate::sched::kthread::{self, OnPanic, OnStop}; + use crate::sched::kthread::{self, OnPanic}; /// One-shot for the body-copy injection point, consumed by `mid_body` or, under `log-shared-reservation`, by the outer `inject`. static ARMED: AtomicBool = AtomicBool::new(false); @@ -44,7 +44,7 @@ mod armed { crate::log!("lognest start records={SHARD_RECORDS}"); // A kernel thread, not the syscall that arms it: `IF` is clear for a whole syscall, so injecting there would never test the guard. // `Halt`: this thread carries the whole stimulus; surviving its death would answer the gate having injected nothing. - kthread::spawn("lognest", body, 0, OnPanic::Halt, OnStop::Runs); + kthread::spawn("lognest", body, 0, OnPanic::Halt); } extern "C" fn body(_arg: u64) -> ! { diff --git a/kernel/src/log/storm.rs b/kernel/src/log/storm.rs index e18f38657dc..0b70d0b0062 100644 --- a/kernel/src/log/storm.rs +++ b/kernel/src/log/storm.rs @@ -2,7 +2,7 @@ use core::sync::atomic::{AtomicBool, Ordering}; -use crate::sched::kthread::{self, OnPanic, OnStop}; +use crate::sched::kthread::{self, OnPanic}; // Exceeds a shard's capacity, so the drop path under test is reached at every `--smp` count. const STORM_RECORDS: u64 = 1024; @@ -47,7 +47,7 @@ pub fn start_once() { crate::log!("logstorm start threads={threads} records={STORM_RECORDS}"); for thread in 0..threads { // `Halt`: a panicked storm thread invalidates the gate's conservation law, so continuing would answer over an incomplete storm. - kthread::spawn("logstorm", body, thread as u64, OnPanic::Halt, OnStop::Runs); + kthread::spawn("logstorm", body, thread as u64, OnPanic::Halt); } } diff --git a/kernel/src/main.rs b/kernel/src/main.rs index 28445ae40b4..823a78dab33 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -83,7 +83,6 @@ mod clock; mod watch; mod iod; -mod reaper; mod object; mod inbox; mod pipe; @@ -670,7 +669,6 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { // After klogd so their own spawn logs have a drainer. drivers::xhci::usbd::start(); iod::start(); - reaper::start(); // Here: the last kernel thread is spawned. #[cfg(feature = "boot-actuators")] diff --git a/kernel/src/process.rs b/kernel/src/process.rs index 911593137bf..09cf4a49111 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -290,8 +290,8 @@ pub struct ProcessEntry { symbols: Arc, main_tid: Tid, threads: crate::id_map::IdMap, - /// Set once by the exit/kill path that owns teardown; checked by `spawn_thread` so no thread appears after the retire sweep. - tearing_down: bool, + /// Set once, with the exit's code, by the exit/kill/poison path that claims teardown; checked by `spawn_thread` so no thread appears after the retire set. + teardown_code: Option, } impl ProcessEntry { @@ -313,7 +313,7 @@ impl ProcessEntry { symbols, main_tid, threads, - tearing_down: false, + teardown_code: None, } } pub fn pid(&self) -> Pid { self.pid } @@ -330,14 +330,14 @@ impl ProcessEntry { impl ProcessEntry { /// Mirrors [`Lifecycle::tearing_down`], usable without the trait in scope. - pub fn tearing_down(&self) -> bool { self.tearing_down } + pub fn tearing_down(&self) -> bool { self.teardown_code.is_some() } } /// The lifecycle face of an entry: only the fields `toyos-proclife` decides against, and nothing else this type carries. impl Lifecycle for ProcessEntry { fn main_tid(&self) -> Tid { self.main_tid } - fn tearing_down(&self) -> bool { self.tearing_down } - fn begin_teardown(&mut self) { self.tearing_down = true; } + fn teardown_code(&self) -> Option { self.teardown_code } + fn begin_teardown(&mut self, code: i32) { self.teardown_code = Some(code); } fn location(&self, tid: Tid) -> Option { self.threads.get(tid).map(|t| t.state) } @@ -556,7 +556,6 @@ pub struct ProcessAccounting { pub blocked_pipe_ns: u64, pub blocked_ipc_ns: u64, pub blocked_other_ns: u64, - pub child_threads_cpu_ns: u64, pub runqueue_wait_ns: u64, } @@ -709,35 +708,27 @@ pub fn stats_of( Some(stats_from(&data, pid, cpu_ns, syscall_total, syscall_total_ns)) } -/// Mark one thread dead; see `toyos_proclife::teardown::mark_zombie` for idempotency and silence. -pub fn mark_thread_zombie(table: &mut ProcessTable, pid: Pid, tid: Tid, code: i32) { - proclife::mark_zombie(table, pid, tid, code); -} - -/// What a thread that died in panic recovery leaves to be cleaned up; the panic path itself may hold any lock the faulted thread held, so this only records the thread, and the idle loop runs it later. +/// What a thread that died in panic recovery leaves to be done once the table lock is given up; the panic path itself may hold any lock the faulted thread held, so it only records the thread, and the idle loop runs this later. #[must_use = "a poisoned thread's waiter must be woken"] -pub enum PoisonWake { - /// A child thread died; the pair is the subject `thread_join` arms on (not the process's main thread). - Joiner(Pid, Tid), - /// The main thread died, so the process is over; publish outside the table lock. - Process(Arc), +pub struct PoisonWake { + /// The rest of a process whose main thread died. + pub retire: Vec, + /// A thread that is not the main one: what its joiner armed on. + pub joiner: Option, + /// It was the last thread out: the object to publish its exit on, and the code. + pub exit: Option<(Arc, i32)>, } -/// Mark a poisoned thread dead and say what the idle loop must wake for it. `None`: nothing to do — the entry is gone, or another path already owns teardown. +/// Take a poisoned thread out of its process and say what the idle loop must do for it. /// Resources are freed with the table entry rather than before it: every release below wants a lock the faulted thread may still hold. -#[must_use = "a poisoned thread's waiter must be woken"] -pub fn zombify_poisoned(table: &mut ProcessTable, pid: Pid, tid: Tid) -> Option { - match poison::zombify_poisoned(table, pid, tid) { - poison::PoisonOutcome::Nothing => None, - poison::PoisonOutcome::Joiner(watch) => { - let (pid, tid) = watch.thread()?; - Some(PoisonWake::Joiner(pid, tid)) - } - poison::PoisonOutcome::Process(pid) => { - let proc = Processes::get(table, pid) - .expect("zombify_poisoned: the entry it just claimed and marked"); - Some(PoisonWake::Process(Arc::clone(&proc.object))) - } +pub fn zombify_poisoned(table: &mut ProcessTable, pid: Pid, tid: Tid) -> PoisonWake { + let owed = poison::zombify_poisoned(table, pid, tid); + let proc = Processes::get(table, pid); + let sched = |tid: Tid| proc.and_then(|p| p.threads.get(tid)).and_then(ThreadEntry::sched).cloned(); + PoisonWake { + retire: owed.retire.into_iter().map(|t| sched(t).expect("zombify_poisoned: a thread still in its process has a scheduler record")).collect(), + joiner: owed.joiner.and_then(|watch| sched(watch.thread()?.1)), + exit: owed.exit.map(|code| (Arc::clone(&proc.expect("zombify_poisoned: the entry its last thread left").object), code)), } } @@ -769,7 +760,7 @@ pub fn current_data() -> Arc> { Some(thread) => Arc::clone(&thread.thread_data), None => { drop(guard); - scheduler::exit_current(-1); + scheduler::exit_current(); } } } @@ -793,7 +784,7 @@ pub fn process_data() -> Arc> { Some(proc) => Arc::clone(&proc.process_data), None => { drop(guard); - scheduler::exit_current(-1); + scheduler::exit_current(); } } } @@ -940,10 +931,6 @@ fn teardown_resources( let mut data = process_data_arc.lock(); - if percpu::current_pid() == Some(pid) { - scheduler::flush_current_stats(&mut data.accounting); - } - if syscall_total > 0 { use alloc::string::String; use core::fmt::Write; @@ -983,38 +970,18 @@ fn teardown_resources( (syscall_total, syscall_total_ns) } -/// Table-side teardown bookkeeping: mark remaining threads zombie, drop the symbol table. -/// Caller must hold `PROCESS_TABLE`, have claimed teardown, retired every other thread and freed resources. Returns the object whose exit the caller publishes once the table lock is given up. +/// Table-side teardown bookkeeping: drop the symbol table and total the CPU time of every thread still in the table. +/// Caller must hold `PROCESS_TABLE`, be the last thread out and have freed the resources. Returns the object whose exit the caller publishes once the table lock is given up, with that total. #[must_use = "the exit must be published on the object returned"] -fn teardown_bookkeeping(table: &mut ProcessTable, process_pid: Pid, code: i32, - main_cpu_ns: u64, child_threads_cpu_ns: u64) - -> Arc { +fn teardown_bookkeeping(table: &mut ProcessTable, process_pid: Pid, code: i32) + -> (Arc, u64) { let proc = table.get_mut(process_pid) .expect("teardown_bookkeeping: process not found"); - - proclife::mark_all_zombie(proc, code); - - // Dropping this `Arc` is the release; every other thread is already retired, so the thread running this line holds the last clone — which is why a lock-free crash-report read can never see a table whose owner is off every CPU. proc.symbols = Arc::new(SymbolTable::empty()); - - // Whole-process total: fold_retired already folded sibling time into child_threads_cpu_ns. - let cpu_ms = (main_cpu_ns + child_threads_cpu_ns) / 1_000_000; + let cpu_ns: u64 = proc.threads.iter().map(|(_, t)| t.sched().map_or(0, scheduler::task_cpu_ns)).sum(); let name = proc.name_str(); - log!("exit: {name} pid={process_pid} code={code} cpu={cpu_ms}ms"); - - Arc::clone(&proc.object) -} - -/// The accounting a process leaves behind, for `SYS_PROCESS_STATS`. Must run after [`fold_retired`], which folds retired threads' accounting into `ProcessData`. -fn final_stats( - process_data_arc: &Arc>, - pid: Pid, - syscall_total: u64, - syscall_total_ns: u64, - main_cpu_ns: u64, -) -> toyos_abi::syscall::ProcessStats { - let data = process_data_arc.lock(); - stats_from(&data, pid, main_cpu_ns, syscall_total, syscall_total_ns) + log!("exit: {name} pid={process_pid} code={code} cpu={}ms", cpu_ns / 1_000_000); + (Arc::clone(&proc.object), cpu_ns) } /// One `ProcessStats`, from a process's own data; written once, since `SYS_PROCESS_STATS` samples a live process through the same fields the teardown snapshots. @@ -1029,7 +996,7 @@ pub fn stats_from( let acct = &data.accounting; ProcessStats { wall_ns: crate::clock::nanos_since_boot().saturating_sub(data.spawn_ns), - cpu_ns: cpu_ns + acct.child_threads_cpu_ns, + cpu_ns, syscall_total, syscall_total_ns, fault_demand_count: acct.fault_demand_count, @@ -1049,116 +1016,68 @@ pub fn stats_from( } } -/// The scheduler records of `tids`, cloned out of the table. -fn scheds(pid: Pid, tids: Vec) -> Vec<(Tid, ThreadSched)> { - tids.into_iter().filter_map(|t| Some((t, thread_sched(pid, t)?))).collect() -} - -/// Fold released threads' scheduler accounting into the process's; returns the main thread's CPU time if it was among them (else 0). -fn fold_retired( - threads: Vec<(Tid, ThreadSched)>, - main_tid: Tid, - process_data_arc: &Arc>, -) -> u64 { - // The whole teardown's memory-freeing safety rests on this: what follows frees what these threads ran on. - assert!( - threads.iter().all(|(_, sched)| sched.handle.released()), - "teardown: a thread of the process is still on the scheduler", - ); - let mut main_cpu_ns = 0u64; - for (t, sched) in threads { - let cpu_ns = scheduler::task_cpu_ns(&sched); - let mut pdata = process_data_arc.lock(); - sched.handle.merge_into(&mut pdata.accounting); - match proclife::charge(t, main_tid) { - proclife::CpuCharge::MainThread => main_cpu_ns = cpu_ns, - proclife::CpuCharge::ChildThreads => { - pdata.accounting.child_threads_cpu_ns += cpu_ns; - } - } - } - main_cpu_ns -} - -/// Phases 3-5 of a process teardown, shared by exit and kill: free resources, do the table-locked bookkeeping, then publish the exit. The ordering is load-bearing: the caller has already retired every other thread, so none of this process can run. +/// The last thread out's teardown of its process, on its own stack: free what the process holds, then publish its exit. Every thread has left, so none of this process can run in what is freed. /// Publish happens after the table lock is released — `teardown_bookkeeping`'s wake needs that — and once published the entry is reapable, so nothing may read the table for this pid after. -fn teardown_tail( - process_data_arc: &Arc>, - thread_data_arc: &Arc>, - pid: Pid, - code: i32, - main_cpu_ns: u64, -) { - // Phase 3: free resources — no other thread of this process can run. - let (syscall_total, syscall_total_ns) = - teardown_resources(process_data_arc, thread_data_arc, pid); - - let child_threads_cpu_ns = process_data_arc.lock().accounting.child_threads_cpu_ns; - - // Phase 4: table bookkeeping (thread zombie marks, symbols released). - let object = { +fn teardown(pid: Pid, code: i32, process_data: &Arc>) { + let main_thread_data = { + let guard = PROCESS_TABLE.lock(); + let proc = guard.as_ref().unwrap().get(pid).expect("teardown: the process its last thread just left"); + Arc::clone(&proc.threads.get(proc.main_tid).expect("teardown: a claimed process gives up no thread").thread_data) + }; + let (syscall_total, syscall_total_ns) = teardown_resources(process_data, &main_thread_data, pid); + let (object, cpu_ns) = { let mut guard = PROCESS_TABLE.lock(); - let table = guard.as_mut().unwrap(); - teardown_bookkeeping(table, pid, code, main_cpu_ns, child_threads_cpu_ns) + teardown_bookkeeping(guard.as_mut().unwrap(), pid, code) }; - - // Phase 5: publish; once published the entry is reapable, so nothing may read the table for this pid after. - let stats = final_stats(process_data_arc, pid, syscall_total, syscall_total_ns, main_cpu_ns); + let stats = stats_from(&process_data.lock(), pid, cpu_ns, syscall_total, syscall_total_ns); object.publish_exit(crate::object::process::Exit { code, stats }); } -pub fn exit(code: i32) -> ! { - release_process(code); - scheduler::exit_current(code); -} - -/// Everything a process's own teardown gives back; returns rather than diverging because `exit_current` never comes back, and three live `Arc`s (`ProcessObject`, `ProcessData`, `ThreadData`) would leak with no scope left to drop them. -fn release_process(code: i32) { - let process_pid = current_process(); - let tid = current_tid(); - - // Phase 1: claim teardown. Exactly one exit/kill path wins; later arrivals just exit their own thread, and the claimant's retire sweep accounts for them like any other thread. - let (process_data_arc, thread_data_arc, main_tid, set) = { +/// Take the running thread out of its process: out of its address space, its accounting into the process's, and its own mark in the table. The last thread out of a claimed process tears the process down here. +/// Returns rather than diverging: the exit pass never comes back, and the live `Arc`s here would leak with no scope left to drop them. +pub fn leave(chosen: Option) { + let (pid, tid) = (current_process(), current_tid()); + crate::mm::paging::activate_kernel(); + let process_data = process_data(); + scheduler::flush_current_stats(&mut process_data.lock().accounting); + let out = { let mut guard = PROCESS_TABLE.lock(); - let table = guard.as_mut().unwrap(); - // Checked before the claim: a thread not in its own process's entry has nothing to tear down. - let present = Processes::get(table, process_pid) - .is_some_and(|proc| Lifecycle::location(proc, tid).is_some()); - if !present || !proclife::claim_teardown(table, process_pid) { - drop(guard); - crate::mm::paging::activate_kernel(); - return; - } - let proc = Processes::get(table, process_pid) - .expect("release_process: the entry the claim just succeeded on"); - let set = proclife::exit_set(proc, tid); - let thread = proc.threads.get(tid).expect("checked present above"); - (Arc::clone(&proc.process_data), Arc::clone(&thread.thread_data), - proc.main_tid, set) + proclife::leave(guard.as_mut().unwrap(), pid, tid, chosen) }; - - crate::mm::paging::activate_kernel(); - - // Phase 2: retire every *other* thread — the current thread can't retire itself. - let others = scheds(process_pid, set.others); - for (_, sched) in &others { - scheduler::post_retire(sched); + // The table says zombie now, which a sweep counts as nothing left to stop. + crate::quiesce::note_progress(); + if let proclife::Leave::Last { code } = out { + teardown(pid, code, &process_data); } - for (_, sched) in &others { - scheduler::await_released(sched); - } - let mut main_cpu_ns = fold_retired(others, main_tid, &process_data_arc); - // Filtered out of the retire set above, so its time is picked up here if it's the main thread. - if set.current_is_main { - main_cpu_ns = thread_sched(process_pid, tid) - .map_or(0, |s| scheduler::task_cpu_ns(&s)); +} + +/// Claim `pid`'s teardown for `code`: the winner gets the threads it retires, every one still in the process but `caller`; a loser, or a process already gone, gets none. +fn claim(pid: Pid, code: i32, caller: Option) -> Vec { + let mut guard = PROCESS_TABLE.lock(); + let table = guard.as_mut().unwrap(); + if !proclife::claim_teardown(table, pid, code) { + return Vec::new(); } + let proc = Processes::get(table, pid).expect("claim: the entry the claim just succeeded on"); + proclife::retire_set(proc, caller) + .into_iter() + .map(|tid| { + proc.threads.get(tid).and_then(ThreadEntry::sched).cloned() + .expect("claim: a thread in the table has its scheduler record") + }) + .collect() +} - // Phases 3-5: free resources, table bookkeeping, publish the exit. - teardown_tail(&process_data_arc, &thread_data_arc, process_pid, code, main_cpu_ns); +pub fn exit(code: i32) -> ! { + let tid = current_tid(); + for sched in claim(current_process(), code, Some(tid)) { + scheduler::post_retire(&sched); + } + leave(None); + scheduler::exit_current(); } -/// Exit the current thread. If this is the main thread, tears down the entire process via `exit()`. For child threads, frees thread resources and zombifies. +/// Exit the current thread. If this is the main thread, it is the process's exit. For a child thread, frees its own mappings and leaves. pub fn thread_exit(code: i32) -> ! { let process_pid = current_process(); let tid = current_tid(); @@ -1167,18 +1086,13 @@ pub fn thread_exit(code: i32) -> ! { let table = guard.as_ref().unwrap(); proclife::route_thread_exit(table, process_pid, tid) }; - let post = match route { proclife::ThreadExit::Process => exit(code), proclife::ThreadExit::Sibling { post } => post, - // The entry went under this thread on the way here (the race `mark_thread_zombie` already tolerates); it leaves by the sibling door. - proclife::ThreadExit::Gone { post } => { - log!("exit: pid={process_pid} tid={tid} outlived its process-table entry"); - post - } }; release_thread(process_pid, tid, code); + leave(Some(code)); // Whoever joined this thread armed on it; post before the exit pass — after it this thread never runs again. if let Some(handle) = crate::sched::driver::current_handle() { // Held together by assertion, not shared state: the decision names the subject, this performs it via the CPU's own handle. @@ -1190,11 +1104,10 @@ pub fn thread_exit(code: i32) -> ! { ); handle.watch().post(); } - scheduler::exit_current(code); + scheduler::exit_current(); } -/// A child thread's own teardown; returns rather than diverging for [`release_process`]'s reason (a live `Arc` nothing would drop). -/// Every table write here is silent about a gone entry: the process may already have been reaped by another CPU's kill. +/// A child thread's own mappings, released before it leaves; returns rather than diverging for [`leave`]'s reason. fn release_thread(process_pid: Pid, tid: Tid, code: i32) { let addr_space = current_address_space(); crate::mm::paging::activate_kernel(); @@ -1212,18 +1125,11 @@ fn release_thread(process_pid: Pid, tid: Tid, code: i32) { // After the block: dropping waits for every other CPU, and the page-fault handler takes this same lock with IF clear. drop(released); - let mut guard = PROCESS_TABLE.lock(); - let table = guard.as_mut().unwrap(); - let cpu_ms = table.get(process_pid).and_then(|p| p.threads.get(tid)) - .and_then(|t| t.sched()) - .map_or(0, scheduler::task_cpu_ns) / 1_000_000; - if let Some(thread) = table.get_mut(process_pid).and_then(|p| p.threads.get_mut(tid)) { - thread.state = ThreadLocation::Zombie(code); - } - if let Some(proc) = table.get(process_pid) { - let name = proc.name_str(); - crate::log_limited!("exit: {name} tid={tid} code={code} cpu={cpu_ms}ms"); - } + let guard = PROCESS_TABLE.lock(); + let proc = guard.as_ref().unwrap().get(process_pid).expect("release_thread: a thread's process is not in the table"); + let cpu_ms = proc.threads.get(tid).and_then(|t| t.sched()).map_or(0, scheduler::task_cpu_ns) / 1_000_000; + let name = proc.name_str(); + crate::log_limited!("exit: {name} tid={tid} code={code} cpu={cpu_ms}ms"); } /// A thread's scheduler record, cloned out of the table so wake/retire never hold the table lock while they post. @@ -1648,62 +1554,14 @@ fn with_current_symbols(f: impl FnOnce(&crate::symbols::SymbolTable) -> bool) -> /// Kill the process an object names. /// The handle is the whole authorization, not the parent relationship: a `Process` handle carrying `Rights::MANAGE` says who may, and it can be narrowed away or handed on. `Ok` for an already-gone process: the caller asked for it to be dead and it is. -/// Returns once the victim's retires are posted; the reaper finishes the teardown, and the object's exit is published when it has. +/// Returns once the victim's retires are posted, never waiting on it: the victim may be killing this caller. Its last thread out publishes the object's exit. pub fn kill_process(object: &crate::object::process::ProcessObject) -> u64 { - let target_pid = object.pid(); - - // Phase 1: claim teardown (brief table lock) - let (process_data, thread_data, main_tid, tids) = { - let mut guard = PROCESS_TABLE.lock(); - let table = guard.as_mut().unwrap(); - - // Gone, and already-tearing-down, are the same answer here: the caller asked for it to be dead. - if !proclife::claim_teardown(table, target_pid) { - return 0; - } - let proc = Processes::get(table, target_pid) - .expect("kill_process: the entry the claim just succeeded on"); - let tids = proclife::kill_set(proc); - let main_thread = proc.threads.get(proc.main_tid).unwrap(); - (Arc::clone(&proc.process_data), Arc::clone(&main_thread.thread_data), proc.main_tid, tids) - }; - - // Phase 2, posted and never awaited here: the victim may be killing this caller. - let threads = scheds(target_pid, tids); - for (_, sched) in &threads { - scheduler::post_retire(sched); + for sched in claim(object.pid(), KILLED_EXIT_CODE, None) { + scheduler::post_retire(&sched); } - crate::reaper::owe(Killed { pid: target_pid, process_data, thread_data, main_tid, threads }); 0 } -/// A claimed kill whose every retire is posted, waiting for the reaper. -pub struct Killed { - pid: Pid, - process_data: Arc>, - thread_data: Arc>, - main_tid: Tid, - threads: Vec<(Tid, ThreadSched)>, -} - -impl Killed { - /// Whether every thread is off the scheduler, so [`finish_kill`] may run. - pub fn released(&self) -> bool { - self.threads.iter().all(|(_, sched)| sched.handle.released()) - } - - pub fn pid(&self) -> Pid { - self.pid - } -} - -/// The reaper's half of a kill, once [`Killed::released`]: phases 3-5, the same teardown tail as exit. -pub fn finish_kill(killed: Killed) { - let Killed { pid, process_data, thread_data, main_tid, threads } = killed; - let main_cpu_ns = fold_retired(threads, main_tid, &process_data); - teardown_tail(&process_data, &thread_data, pid, KILLED_EXIT_CODE, main_cpu_ns); -} - /// The shell convention for "died on SIGKILL"; kept because every test that reads one already spells it. pub const KILLED_EXIT_CODE: i32 = 137; diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 7db355b271f..234fa2911fe 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -25,6 +25,10 @@ //! whatever lock the thread on that CPU was holding, and `sync_all` is the //! first thing that would wait on it. //! +//! Kernel threads are exempt by identity, not by accident: `klogd`, `iod` and +//! `usbd` are in the process table like anything else, and +//! [`crate::sched::kthread::is_kernel_task`] is what tells them apart. +//! //! # What the stop waits on //! //! [`PROGRESS`], posted by [`note_progress`] from the three transitions that @@ -94,7 +98,10 @@ fn stops(stage: u32) -> bool { let (Some(pid), Some(tid)) = (percpu::current_pid(), percpu::current_tid()) else { return false; }; - if crate::sched::kthread::runs_through_the_stop(TaskId(pid, tid)) { + // A kernel thread reaches this boundary on its first dispatch and has no + // Ring 3 to be stopped from; `iod` is also what carries the sync to its + // volume while userland is being stopped around it. + if crate::sched::kthread::is_kernel_task(TaskId(pid, tid)) { return false; } must_stop(ThreadId { pid: pid.raw(), tid: tid.raw() }, caller()) @@ -220,7 +227,7 @@ fn sweep(caller: ThreadId) -> Sweep { if matches!(thread.state(), process::ThreadLocation::Zombie(_)) { continue; } - if crate::sched::kthread::runs_through_the_stop(TaskId(pid, tid)) { + if crate::sched::kthread::is_kernel_task(TaskId(pid, tid)) { continue; } let sched = thread diff --git a/kernel/src/reaper.rs b/kernel/src/reaper.rs deleted file mode 100644 index 01e48a3d067..00000000000 --- a/kernel/src/reaper.rs +++ /dev/null @@ -1,68 +0,0 @@ -//! Finishes the teardowns kills hand it. -//! -//! A kill posts its victim's retires and returns, because the victim may be -//! killing the killer, and a killer that waited for its victim would wait on a -//! thread that is waiting on it. This thread does the waiting instead: the -//! only threads it waits on are killed ones, whose release waits on nothing it -//! does. Which owed kill it takes is `toyos_proclife::teardown::next_released`; -//! the teardowns themselves run one at a time. - -use alloc::vec::Vec; - -use toyos_proclife::teardown::next_released; -use toyos_sched::task::WaitClass; - -use crate::process::{self, Killed}; -use crate::sched::kthread::{self, OnPanic, OnStop}; -use crate::sched::payload::ANY_RELEASED; -use crate::scheduler::{Parkable, RETIRE_GIVE_UP}; -use crate::sync::Lock; -use crate::time::{Deadline, Instant}; -use crate::watch; - -/// Every owed kill with the instant it was owed, oldest first. -static OWED: Lock> = Lock::new(Vec::new()); - -/// Spawns the reaper; call once, from `kernel_main`. -pub fn start() { - // Halts: a dead reaper leaves every later kill unpublished and nothing to say so. - kthread::spawn("reaper", body, 0, OnPanic::Halt, OnStop::Stops); -} - -/// Hand a claimed kill, its retires already posted, to the reaper. -pub fn owe(killed: Killed) { - OWED.lock().push((crate::clock::now(), killed)); - // Its threads may all be released already, and then no release posts again. - ANY_RELEASED.post(); -} - -extern "C" fn body(_arg: u64) -> ! { - let parkable = Parkable::at_entry(); - loop { - let armed = watch::arm(&ANY_RELEASED, 0, WaitClass::Other).expect("the reaper is a task"); - let (next, oldest) = { - let mut owed = OWED.lock(); - let next = next_released(owed.iter().map(|(_, killed)| killed.released())); - (next.map(|at| owed.remove(at).1), owed.first().map(|(since, killed)| (*since, killed.pid()))) - }; - match next { - Some(killed) => { - drop(armed); - process::finish_kill(killed); - } - None => { - // The oldest's deadline is the earliest, so it is the only one to check. - let deadline = oldest.map_or(Deadline::never(), |(since, pid)| { - let give_up = Deadline::at(since + RETIRE_GIVE_UP.duration()); - assert!( - !give_up.reached(crate::clock::now()), - "reaper: pid {pid} not released after {}", - RETIRE_GIVE_UP.duration(), - ); - give_up - }); - watch::wait_uncancellable(&parkable, &armed, deadline); - } - } - } -} diff --git a/kernel/src/sched/dump.rs b/kernel/src/sched/dump.rs index 3a9a85273ce..10cdc57dab1 100644 --- a/kernel/src/sched/dump.rs +++ b/kernel/src/sched/dump.rs @@ -705,7 +705,7 @@ fn census() -> Census { // Blocked and running threads are already the CPUs' lines; skip them here. let Some(tag) = tag else { return }; // Kernel threads don't count against the budget: `MAX_KERNEL_TASKS` - // bounds them, so counting them can't push these lines off the page. + // bounds them at three, so counting them can't push these lines off the page. if !kernel { printed += 1; if printed > CENSUS_LINES { diff --git a/kernel/src/sched/kthread.rs b/kernel/src/sched/kthread.rs index 92af04371cd..c6fdc566e7e 100644 --- a/kernel/src/sched/kthread.rs +++ b/kernel/src/sched/kthread.rs @@ -7,7 +7,7 @@ use alloc::string::String; use alloc::sync::Arc; use alloc::vec::Vec; -use core::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use core::sync::atomic::{AtomicU64, Ordering}; use crate::process::{ ElfInfo, Endowments, PageFaultTrace, ProcessAccounting, ProcessData, ProcessEntry, ThreadData, @@ -19,10 +19,11 @@ use crate::sync::Lock; use super::payload::ThreadSched; +/// `klogd`, `usbd` and `iod`, plus one `log-storm` thread per shard in the actuator build. #[cfg(not(feature = "boot-actuators"))] -const MAX_KERNEL_TASKS: usize = 4; +const MAX_KERNEL_TASKS: usize = 3; #[cfg(feature = "boot-actuators")] -const MAX_KERNEL_TASKS: usize = 4 + toyos_abi::log::MAX_LOG_SHARDS; +const MAX_KERNEL_TASKS: usize = 3 + toyos_abi::log::MAX_LOG_SHARDS; /// Collides with no packed id: neither id map issues `u32::MAX`. const NO_TASK: u64 = u64::MAX; @@ -31,14 +32,13 @@ const NO_TASK: u64 = u64::MAX; /// after the policy word, without another claimant matching the same row. const CLAIMING: u64 = u64::MAX - 1; -/// `task` is stored `Release` after `recoverable` and `stops` and loaded `Acquire` before them, +/// `task` is stored `Release` after `recoverable` and loaded `Acquire` before it, /// so a row found by identity never answers with an unwritten policy. struct Row { task: AtomicU64, // `recoverable` exists because `percpu::in_syscall()` is never true for a kernel // thread, which would otherwise make every kernel-thread panic halt the machine by default. recoverable: AtomicU64, - stops: AtomicBool, } /// A row reserved before the table lock and published before `enqueue_new`. @@ -60,28 +60,18 @@ impl Claim { } /// Payload first, then the identity `Release`. - fn publish(self, id: TaskId, on_panic: OnPanic, on_stop: OnStop) { + fn publish(self, id: TaskId, on_panic: OnPanic) { self.0 .recoverable .store(u64::from(on_panic == OnPanic::Recover), Ordering::Relaxed); - self.0.stops.store(on_stop == OnStop::Stops, Ordering::Relaxed); self.0.task.store(id.pack(), Ordering::Release); } } /// Registered at spawn and never cleared: a dead `Recover` thread's row stays. -static ROWS: [Row; MAX_KERNEL_TASKS] = [const { - Row { task: AtomicU64::new(NO_TASK), recoverable: AtomicU64::new(0), stops: AtomicBool::new(false) } -}; MAX_KERNEL_TASKS]; - -/// Whether the machine's stop (`quiesce`) stops a kernel thread. -#[derive(Clone, Copy, PartialEq, Eq)] -pub enum OnStop { - /// It carries the stop's own work, as `iod` carries the sync to its volume. - Runs, - /// The work it does is userland's, so it stops with userland. - Stops, -} +static ROWS: [Row; MAX_KERNEL_TASKS] = + [const { Row { task: AtomicU64::new(NO_TASK), recoverable: AtomicU64::new(0) } }; + MAX_KERNEL_TASKS]; /// What a kernel thread's panic does. #[derive(Clone, Copy, PartialEq, Eq)] @@ -119,21 +109,17 @@ pub fn is_kernel_task(id: TaskId) -> bool { ROWS.iter().any(|row| row.task.load(Ordering::Acquire) == packed) } -/// Is `id` a kernel thread declared [`OnStop::Runs`]? -pub fn runs_through_the_stop(id: TaskId) -> bool { - let packed = id.pack(); - ROWS.iter() - .any(|row| row.task.load(Ordering::Acquire) == packed && !row.stops.load(Ordering::Relaxed)) -} - /// Control for `process::process_object`'s refusal: no kernel thread's pid names a process a handle could hold. #[cfg(feature = "boot-actuators")] pub fn open_selftest() { let pids: Vec<_> = ROWS .iter() .map(|row| row.task.load(Ordering::Acquire)) - .filter(|&task| task != NO_TASK && task != CLAIMING) - .map(|task| TaskId::unpack(task).0) + .filter(|&task| task != NO_TASK) + .map(|task| { + assert_ne!(task, CLAIMING, "kthread: a row is still being claimed after the last spawn"); + TaskId::unpack(task).0 + }) .collect(); let opened = pids.iter().filter(|&&pid| crate::process::process_object(pid).is_some()).count(); let verdict = if !pids.is_empty() && opened == 0 { "PASS" } else { "FAIL" }; @@ -146,13 +132,7 @@ pub fn panic_recovers_here() -> Option { } /// Start a kernel thread running `body(arg)` on its own kernel stack and return its scheduler faces. -pub fn spawn( - name: &str, - body: extern "C" fn(u64) -> !, - arg: u64, - on_panic: OnPanic, - on_stop: OnStop, -) -> ThreadSched { +pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64, on_panic: OnPanic) -> ThreadSched { let (stack, entry_rsp) = crate::loader::alloc_kernel_stack( crate::loader::kernel_start, body as usize as u64, @@ -185,7 +165,7 @@ pub fn spawn( }); let tid = table.get(pid).expect("kthread: the entry just inserted is gone").main_tid(); // Before `enqueue_new`: from that call the task can run and panic. - claim.publish(TaskId(pid, tid), on_panic, on_stop); + claim.publish(TaskId(pid, tid), on_panic); // The kernel address space, named so one declaration decides every task's `cr3`. let (sched, _dst) = scheduler::enqueue_new( TaskId(pid, tid), diff --git a/kernel/src/sched/payload.rs b/kernel/src/sched/payload.rs index 5d0773f473a..0e2aa7dfbcf 100644 --- a/kernel/src/sched/payload.rs +++ b/kernel/src/sched/payload.rs @@ -3,7 +3,7 @@ //! released exactly once, by `Hw::release`. use alloc::sync::Arc; -use core::sync::atomic::{AtomicU8, AtomicU32, AtomicU64, Ordering}; +use core::sync::atomic::{AtomicU32, AtomicU64, Ordering}; use toyos_sched::fair::{FairShare, ShareState}; use toyos_sched::hw::Nanos; @@ -77,25 +77,12 @@ pub const SCHED_READY: u8 = 1; pub const SCHED_BLOCKED: u8 = 2; pub const SCHED_UNKNOWN: u8 = 3; -/// Posted after a killed thread's release, for a waiter on whichever of many threads goes first. -pub static ANY_RELEASED: Watch = Watch::new(); - -/// [`TaskHandle`]'s `fate` bits. -const RELEASED: u8 = 1; -const KILLED: u8 = 2; - /// What a thread other than the one running can be asked about; published here since a `CpuSched` is `!Sync` and unreachable remotely. pub struct TaskHandle { cpu_ns: AtomicU64, /// Dispatch timestamp while running, 0 otherwise; a reader adds the live slice itself. running_since: AtomicU64, - acct: Lock, - /// `RELEASED`, set by `Hw::release`: the one fact a retirer needs, that the thread is off - /// every CPU. `KILLED`, set by `scheduler::post_retire`. One word, so the two decide by one - /// read-modify-write each whether the release posts [`ANY_RELEASED`]: `Hw::release` holds - /// no `TaskShared` whose kill bit it could read. - fate: AtomicU8, - /// What another thread arms on to be told this one moved: its exit, for `SYS_THREAD_JOIN`, and its release, for the retirer. + /// What another thread arms on to be told this one moved: its exit, for `SYS_THREAD_JOIN`. watch: Watch, /// Cancels reported to this thread; a second one means a caller swallowed the first, so it panics rather than spinning. cancels: AtomicU32, @@ -108,8 +95,6 @@ impl TaskHandle { Self { cpu_ns: AtomicU64::new(0), running_since: AtomicU64::new(0), - acct: Lock::new(TaskAccounting::default()), - fate: AtomicU8::new(0), watch: Watch::new(), cancels: AtomicU32::new(0), operation: OperationSlot::new(), @@ -126,27 +111,6 @@ impl TaskHandle { pub(crate) fn finalize(&self, acct: TaskAccounting) { self.cpu_ns.store(acct.cpu_ns, Ordering::Relaxed); self.running_since.store(0, Ordering::Relaxed); - *self.acct.lock() = acct; - } - - /// Announces the death only after the payload is dropped; that ordering is the guarantee a retirer's park buys. - pub(crate) fn publish_released(&self) { - let fate = self.fate.fetch_or(RELEASED, Ordering::AcqRel); - // The retirer arms on this thread's own watch, the same subject a joiner uses. - self.watch.post(); - if fate & KILLED != 0 { - ANY_RELEASED.post(); - } - } - - /// Has `Hw::release` run for this thread? The retire wait's condition. - pub fn released(&self) -> bool { - self.fate.load(Ordering::Acquire) & RELEASED != 0 - } - - /// Mark the thread killed; `false` if it is already released, and so has no retire to post. - pub fn mark_killed(&self) -> bool { - self.fate.fetch_or(KILLED, Ordering::AcqRel) & RELEASED == 0 } /// Where this thread's establishment lives; `scheduler::Operation` owns every rule about it. @@ -182,10 +146,6 @@ impl TaskHandle { } } - pub fn merge_into(&self, target: &mut ProcessAccounting) { - let acct = self.acct.lock(); - merge_accounting(&acct, target); - } } /// A thread's two scheduler-visible faces, kept by the process table; created at different instants. diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index a8a697d3a83..3e87b2c8a2e 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -22,7 +22,7 @@ use crate::sched::payload::{KShare, KernelLock, TaskHandle, ThreadSched}; use crate::sched::reap_gate::ReapGate; use crate::sched::futex; use crate::sync::Lock; -use crate::time::{Cadence, Deadline, Duration, Tripwire}; +use crate::time::{Cadence, Deadline, Duration}; use crate::DirectMap; pub use crate::sched::driver::{ @@ -382,25 +382,20 @@ pub fn leave_ring3_if_due() { unreachable!("leave_ring3_if_due: a stopped task was dispatched again"); } SafePoint::Exit => { - // The retirer owns teardown; a mark_thread_zombie here would race it. + // A syscall's own depth: the last thread out tears its process down here, and that parks. + crate::preempt::disable(); + process::leave(None); + crate::preempt::enable_no_resched(); driver::pass(Dispose::Exit); unreachable!("leave_ring3_if_due: returned from the exit pass"); } } } +/// The exit pass of a thread that has left its process (`process::leave`). #[track_caller] -pub fn exit_current(code: i32) -> ! { +pub fn exit_current() -> ! { assert_baseline(BASELINE_TRAP); - { - let mut guard = process::PROCESS_TABLE.lock(); - let table = guard.as_mut().unwrap(); - let tid = percpu::current_tid().unwrap(); - let pid = percpu::current_pid().unwrap(); - process::mark_thread_zombie(table, pid, tid, code); - } - // The table says zombie now, which a sweep counts as nothing left to stop. - crate::quiesce::note_progress(); driver::pass(Dispose::Exit); unreachable!("exit_current: returned from the exit pass"); } @@ -502,78 +497,13 @@ pub fn futex_wake(phys_addr: DirectMap, count: usize) -> u64 { } /// Set a thread's kill bit and ask its CPU for a safe point; returns at once. +/// The thread leaves at that safe point: `process::leave`. pub fn post_retire(sched: &ThreadSched) { - if !sched.handle.mark_killed() { - return; - } preempt_off(|p| { toyos_sched::retire::begin(&sched.shared).post(cpus(), &HW, p); }); } -/// How long a retired thread may take to be released before the kernel panics. -/// Superseded by the scheduling-reservations design; kept because a -/// known-wrong constant is still what this kernel runs -/// (`issues/kernel/scheduler-pass-blocks-in-xhci.md`). -pub const RETIRE_GIVE_UP: Tripwire = Tripwire::absurd( - Duration::from_secs(10), - "four pass prologues on xHCI's own 2 s deadline, two quanta, and an unwind \ - the real-time band may stretch elevenfold; past this the wake was lost", -); - -/// Wait until a retired thread's record — kernel stack and address-space -/// reference — is released. The state word reading `Dead` is not enough: that -/// payload is freed by the pass after the one that publishes it. -#[track_caller] -pub fn await_released(sched: &ThreadSched) { - // Also on the early-return path below, where no park happens and the two - // asserts inside the wait would never run. - assert_baseline(BASELINE_TRAP); - if let (Some(pid), Some(tid)) = (percpu::current_pid(), percpu::current_tid()) { - if let Some(handle) = driver::current_shared() { - assert!( - !Arc::ptr_eq(&handle, &sched.shared), - "await_released: cannot retire self ({})", - TaskId(pid, tid), - ); - } - } - if sched.handle.released() { - return; - } - /// Re-poll rate for the liveness backstop; the release wake is what - /// actually ends the wait. - const RECHECK: Cadence = Cadence::every( - Duration::from_millis(50), - "two hundred re-polls inside the tripwire, on a thread that is otherwise parked", - ); - let give_up = Deadline::at(crate::clock::now() + RETIRE_GIVE_UP.duration()); - let parkable = Parkable::at_entry(); - // Uncancellable: a killed retirer cannot propagate a cancel with the - // retire half done; the tripwire above bounds it instead. - let Some(armed) = watch::arm( - sched.handle.watch(), - sched.shared.key().0, - WaitClass::Other, - ) else { - panic!("await_released: no current task to park"); - }; - while !sched.handle.released() { - if give_up.reached(crate::clock::now()) { - panic!( - "await_released: task not released after {}: {:?}", - RETIRE_GIVE_UP.duration(), - sched.shared.state() - ); - } - watch::wait_uncancellable( - &parkable, - &armed, - Deadline::at(crate::clock::now() + RECHECK.duration()), - ); - } -} - /// Per-CPU hand-off bank for threads that died in panic recovery: the panic /// path may hold any lock, so it can only store here. static POISONED: [poison::PoisonSet; MAX_CPUS] = [const { poison::PoisonSet::new() }; MAX_CPUS]; @@ -629,27 +559,26 @@ pub(crate) fn reap_poisoned() { for bank in POISONED.iter() { bank.drain(|raw| { let id = TaskId::unpack(raw); - wakes[next] = process::zombify_poisoned(table, id.0, id.1); + wakes[next] = Some(process::zombify_poisoned(table, id.0, id.1)); next += 1; }); } } drop(reaped); for wake in wakes.into_iter().flatten() { - match wake { - process::PoisonWake::Joiner(pid, tid) => { - if let Some(sched) = process::thread_sched(pid, tid) { - sched.handle.watch().post(); - } - } - // -1: nobody asked for this exit, and teardown never ran to account it. - process::PoisonWake::Process(object) => { - let stats = toyos_abi::syscall::ProcessStats { - pid: object.pid().raw(), - ..Default::default() - }; - object.publish_exit(crate::object::process::Exit { code: -1, stats }) - } + for sched in &wake.retire { + post_retire(sched); + } + if let Some(sched) = wake.joiner { + sched.handle.watch().post(); + } + // No teardown ran to account this exit: the resources go with the entry. + if let Some((object, code)) = wake.exit { + let stats = toyos_abi::syscall::ProcessStats { + pid: object.pid().raw(), + ..Default::default() + }; + object.publish_exit(crate::object::process::Exit { code, stats }); } } } diff --git a/kernel/src/time.rs b/kernel/src/time.rs index 85cf1c1479a..48a3f2d1a9f 100644 --- a/kernel/src/time.rs +++ b/kernel/src/time.rs @@ -202,10 +202,6 @@ impl Tripwire { Self { limit, absurd_because } } - pub const fn duration(self) -> Duration { - self.limit - } - pub const fn nanos(self) -> u64 { self.limit.nanos() } diff --git a/src/ci.rs b/src/ci.rs index 93a1b34eb2f..0eb37e99d94 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -356,6 +356,12 @@ pub(crate) const CONTROLS: &[Control] = &[ "two_processes_killing_each_other_both_end ... FAILED", "a_kill_chain_of_three_ends ... FAILED", ]), + red(PROCLIFE, "mutate-first-out-tears-down", None, &[ + "an_exit_and_a_kill_never_both_tear_a_process_down ... FAILED", + ]), + red(PROCLIFE, "mutate-join-collects-in-a-teardown", None, &[ + "a_join_racing_the_kill_that_takes_its_target ... FAILED", + ]), red(SCHED_SIM, "placement-ignores-staleness", Some("policy"), &[ "a_stopped_cpu_stops_taking_work ... FAILED", ]), diff --git a/tests/common/metal.rs b/tests/common/metal.rs index d4f9a27cdee..09db138cd58 100644 --- a/tests/common/metal.rs +++ b/tests/common/metal.rs @@ -397,7 +397,7 @@ impl Readback { pub fn stop_completed(&self) -> Result<(), String> { match self.stop_record() { Some(park) if !park.stopped_the_machine() => Err(format!( - "{}'s stop gave up on {} thread(s) that never reached a safe point, so \ + "{}'s stop gave up on {} userland thread(s) that never reached a safe point, so \ this boot's sync and its last word are claims about a machine that was still \ running:\n {park}", self.label, diff --git a/tests/common/power.rs b/tests/common/power.rs index 43eea24d270..198952e0448 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -170,9 +170,8 @@ pub fn quiesce_stops_the_machine( const WRITERS: u32 = 6; // The threads the stop names besides the writers: the job's own main // thread, parked on init's answer; `test-runner`'s main and deadline - // threads; `logd`'s; and the kernel's `reaper`, which stops with userland - // because it runs userland's kills. `init` asked for the stop and is its caller. - const OTHERS: u32 = 5; + // threads; and `logd`'s. `init` asked for the stop and is its caller. + const OTHERS: u32 = 4; /// Mirrored in `kernel/src/syscall/machine.rs`, which queues it. const QUEUED: &str = "console: a holder's line, queued once the stop had stopped every holder"; let (whole, record) = stopped_boot( @@ -216,9 +215,9 @@ pub fn quiesce_stops_the_machine( // not the one described. if record.sweep.total() != WRITERS + OTHERS { return Err(format!( - "this boot's stop named {} thread(s); {WRITERS} writers plus the {OTHERS} \ - of the job, test-runner, logd and the reaper make {}, so this is not the machine the \ - writers were on:\n {record}\n{whole}", + "this boot's stop named {} userland thread(s); {WRITERS} writers plus the {OTHERS} \ + of the job, test-runner and logd make {}, so this is not the machine the writers \ + were on:\n {record}\n{whole}", record.sweep.total(), WRITERS + OTHERS, )); diff --git a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs new file mode 100644 index 00000000000..ad0a8cc408e --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs @@ -0,0 +1,122 @@ +//! A thread killed inside any wait the kernel has for it leaves, and its +//! process ends. +//! +//! One child per wait, each killed after it says it is about to park: a futex, +//! a poll ring, a process's end, a thread's end and a sleep. A wait the kill +//! cannot end keeps the child's last thread in its process for ever, and +//! `wait` below never returns; the harness's deadline is what says so. +//! `kill_while_blocked` holds the pipe, connection and accept waits, and +//! `mutual_kill` a kill inside a kill. +//! +//! The marker is printed immediately before the wait, so a kill that lands in +//! the few instructions between the two ends the child at that syscall's exit +//! instead, and the arm passes without its wait under test. + +use std::io::{Read, Write}; +use std::os::toyos::process::{ChildExt, CommandExt}; +use std::process::{Child, Command, Stdio}; +use std::sync::atomic::AtomicU32; +use std::time::Duration; + +use toyos::endow::Endowments; +use toyos::poller::Poller; +use toyos::process::Process; +use toyos::AsHandle; +use toyos_abi::syscall; +use toyos_abi::RawHandle; + +const SELF_PATH: &str = "/system/bin/test_rs_kill_ends_every_wait"; + +/// The label the process-wait child finds the process it waits on under. +const WAITED: &str = "waited"; + +/// `process::KILLED_EXIT_CODE`. +const KILLED: i32 = 137; + +const WAITS: [&str; 5] = ["futex", "poll", "process-wait", "thread-join", "sleep"]; + +fn main() { + match std::env::args().nth(1).as_deref() { + Some(role) => child(role), + None => test(), + } +} + +fn test() { + for role in WAITS { + // The process the process-wait child waits on, which nothing ends but this. + let mut waited = (role == "process-wait").then(|| spawn("sleep", None)); + let endow = waited.as_ref().map(|w| { + let dup = syscall::dup(RawHandle(w.as_raw_handle())).expect("a handle to endow"); + (WAITED.to_string(), dup.0) + }); + let mut child = spawn(role, endow); + child.kill().expect("kill the parked child"); + let code = child.wait().expect("wait for the killed child").code(); + assert_eq!(code, Some(KILLED), "a child killed in its {role} wait ended with {code:?}"); + if let Some(mut waited) = waited.take() { + waited.kill().expect("kill the waited process"); + waited.wait().expect("wait for the waited process"); + } + println!(" {role}: a kill ended it"); + } + println!("kill_ends_every_wait: every wait a kill reached, it ended"); +} + +/// Spawn a child in `role` and read its marker: it is about to park. +fn spawn(role: &str, endow: Option<(String, u32)>) -> Child { + let mut command = Command::new(SELF_PATH); + command.arg(role).stdout(Stdio::piped()); + if let Some((label, handle)) = endow { + command.endow(&label, handle); + } + let mut child = command.spawn().unwrap_or_else(|e| panic!("spawn {role}: {e}")); + let out = child.stdout.as_mut().expect("child stdout"); + let mut line = Vec::new(); + let mut byte = [0u8; 1]; + while out.read(&mut byte).expect("read the child's marker") == 1 && byte[0] != b'\n' { + line.push(byte[0]); + } + assert_eq!(String::from_utf8_lossy(&line), format!("parked in {role}"), "{role} never reached its wait"); + child +} + +fn child(role: &str) -> ! { + match role { + "futex" => { + static WORD: AtomicU32 = AtomicU32::new(0); + say(role); + // SAFETY: `WORD` is a live, aligned `u32` for the whole program. + unsafe { syscall::futex_wait(WORD.as_ptr(), 0, None) }; + } + "poll" => { + let poller = Poller::new(1); + say(role); + poller.wait(1, u64::MAX, |_| {}); + } + "process-wait" => { + let waited: Process = Endowments::get().take(WAITED).expect("the parent endowed a process"); + say(role); + let _ = syscall::process_wait(waited.as_handle()); + } + "thread-join" => { + let parked = std::thread::spawn(|| loop { + std::thread::sleep(Duration::from_secs(3600)); + }); + say(role); + let _ = parked.join(); + } + "sleep" => { + say(role); + syscall::nanosleep(u64::MAX); + } + other => panic!("unknown role {other:?}"), + } + panic!("{role} came back from a wait nothing ends"); +} + +fn say(role: &str) { + let mut out = std::io::stdout(); + out.write_all(format!("parked in {role}\n").as_bytes()).expect("say"); + out.flush().expect("flush"); +} diff --git a/toyos-proclife/Cargo.toml b/toyos-proclife/Cargo.toml index 1e3fc6d4bad..55616d639d6 100644 --- a/toyos-proclife/Cargo.toml +++ b/toyos-proclife/Cargo.toml @@ -43,11 +43,20 @@ mutate-claim-teardown-always-wins = [] # under the very lock the insert takes. `interleave::tests::a_published_exit_ # leaves_no_unretired_thread` must red under this. mutate-spawn-skips-the-insert-recheck = [] -# The model's kill finishes the teardown itself, retiring each victim thread -# and waiting for it, as `process::kill_process` did before the reaper. +# The model's kill waits for its victim's exit before it returns. # `interleave::tests::two_processes_killing_each_other_both_end` must red under -# this: each killer waits on a thread that is inside the other's kill. +# this: each killer waits on a process whose thread is inside the other's kill. mutate-kill-waits-for-its-victims = [] +# `teardown::leave` answers `Last` to every thread that leaves a claimed +# process, so the first one out tears it down under its siblings. +# `interleave::tests::an_exit_and_a_kill_never_both_tear_a_process_down` must +# red under this. +mutate-first-out-tears-down = [] +# `join::collect_zombie` gives up a thread of a process being torn down, whose +# mappings its siblings may still run in. +# `interleave::tests::a_join_racing_the_kill_that_takes_its_target` must red +# under this. +mutate-join-collects-in-a-teardown = [] [dependencies] toyos-abi = { path = "../toyos-abi" } diff --git a/toyos-proclife/src/interleave.rs b/toyos-proclife/src/interleave.rs index 30f832d6fdf..4af6c7c5aee 100644 --- a/toyos-proclife/src/interleave.rs +++ b/toyos-proclife/src/interleave.rs @@ -3,9 +3,9 @@ //! **This is the file the crate exists for.** `kernel/src/process.rs` gives the //! process table lock up between every phase of a teardown and between both //! halves of a spawn, and its comments say what each window costs — "a thread -//! enqueued now would be invisible to its retire sweep", "the current thread is -//! running this and cannot retire itself", "once it is published the entry is -//! reapable, so nothing may read the table for this pid after this point". Not +//! enqueued now would be invisible to its retire sweep", "once it is published +//! the entry is reapable, so nothing may read the table for this pid after this +//! point". Not //! one of those sentences was checkable by anything but a booted guest with a //! race that had to land the wrong way, which is why //! `issues/kernel/spawned-process-never-starts.md` has been open since August @@ -23,158 +23,96 @@ use alloc::vec::Vec; use std::collections::HashSet; use crate::model::World; -use crate::table::{Lifecycle, Processes}; -use crate::{join, reap, spawn, teardown, Pid, ThreadLocation, Tid, Watch}; +use crate::poison::{self, Poisoned}; +use crate::table::Processes; +use crate::{join, reap, spawn, teardown, Pid, Tid, Watch}; /// One kernel path, mid-flight. /// /// Each variant's `pc` is the number of lock sections it has completed, and /// every field beside it is a value the real path carries in a local across a -/// lock release — which is the whole reason a window exists to explore. +/// lock release — which is the whole reason a window exists to explore. A +/// thread's own way out after its operation is the [`World`]'s: `depart_step`. #[derive(Clone, Debug, PartialEq, Eq, Hash)] pub enum Op { - /// `process::release_process` + `teardown_tail`: a thread ending its own - /// process. - Exit { pid: Pid, tid: Tid, code: i32, pc: u32, others: Vec, next: usize }, - /// `process::kill_process`: a handle holder ending somebody else's. `by` is - /// the killing thread when the model holds it; `inline` is the base's kill, - /// which finished the teardown itself. - Kill { pid: Pid, code: i32, pc: u32, tids: Vec, by: Option<(Pid, Tid)>, inline: Option }, - /// The reaper kthread, finishing what kills handed it. - Reap { job: Option }, + /// `process::exit`: a thread ending its own process — claim, retire the + /// rest, leave. + Exit { pid: Pid, tid: Tid, code: i32, pc: u32, retire: Vec }, + /// `process::kill_process`: claim, retire every thread, return. `by` is the + /// killing thread when the model holds it. + Kill { pid: Pid, code: i32, pc: u32, retire: Vec, by: Option<(Pid, Tid)> }, /// `process::spawn_thread`: two lock sections with the whole of a thread /// built between them; `block` is the mapped TLS the build carries across. Spawn { pid: Pid, pc: u32, block: Option }, /// `process::thread_exit` on a thread that is not the main one. - ThreadExit { pid: Pid, tid: Tid, code: i32, pc: u32, post: Option }, + ThreadExit { pid: Pid, tid: Tid, code: i32, pc: u32 }, /// `sys_thread_join`: collect or arm, then re-check. Join { pid: Pid, target: Tid, waiter: Tid, pc: u32 }, + /// A thread dying in panic recovery inside a syscall, and the idle loop's + /// `reap_poisoned` that takes it out of its process. + Poison { pid: Pid, tid: Tid, pc: u32, owed: Poisoned }, /// The idle loop's `reap_poisoned`, reap half. IdlePass { pc: u32 }, } -/// A claimed kill's teardown past its claim: wait out every thread's release, -/// mark, publish. -#[derive(Clone, Debug, PartialEq, Eq, Hash)] -pub struct Finish { - pid: Pid, - code: i32, - tids: Vec, - next: usize, - pc: u32, -} - -impl Finish { - fn new((pid, code, tids): (Pid, i32, Vec)) -> Self { - Finish { pid, code, tids, next: 0, pc: 0 } - } - - /// The reaper's: it takes a kill only once every thread is released. - fn released((pid, code, tids): (Pid, i32, Vec)) -> Self { - Finish { pid, code, next: tids.len(), tids, pc: 1 } - } - - fn enabled(&self, world: &World) -> bool { - self.pc != 0 || self.next == self.tids.len() || retire_ready(world, self.pid, self.tids[self.next]) - } - - /// One section; `true` once the exit is published. - fn step(&mut self, world: &mut World) -> bool { - match self.pc { - 0 => { - if self.next < self.tids.len() { - if retire_step(world, self.pid, self.tids[self.next]) { - self.next += 1; - } - } else { - self.pc = 1; - } - false - } - 1 => { - if let Some(proc) = world.get_mut(self.pid) { - teardown::mark_all_zombie(proc, self.code); - } - self.pc = 2; - false - } - _ => { - world.publish_exit(self.pid, self.code); - true - } - } - } -} - -/// The base's `retire_task`, one section: post the retire, then — blocked until then — -/// find the thread released. `true` once it is. -fn retire_step(world: &mut World, pid: Pid, tid: Tid) -> bool { - if world.is_retired(pid, tid) { - return true; - } - if !world.is_killed(pid, tid) { - world.post_retire(pid, tid); - } - false -} - -/// Whether [`retire_step`] can move: nothing posted yet, or the thread gone. -fn retire_ready(world: &World, pid: Pid, tid: Tid) -> bool { - !world.is_killed(pid, tid) || world.is_retired(pid, tid) -} - impl Op { pub fn exit(pid: Pid, tid: Tid, code: i32) -> Self { - Op::Exit { pid, tid, code, pc: 0, others: Vec::new(), next: 0 } + Op::Exit { pid, tid, code, pc: 0, retire: Vec::new() } } pub fn kill(pid: Pid, code: i32) -> Self { - Op::Kill { pid, code, pc: 0, tids: Vec::new(), by: None, inline: None } + Op::Kill { pid, code, pc: 0, retire: Vec::new(), by: None } } /// A kill issued by a thread the model holds, which is in the kernel until /// the kill returns. pub fn kill_by(pid: Pid, code: i32, by: (Pid, Tid)) -> Self { - Op::Kill { pid, code, pc: 0, tids: Vec::new(), by: Some(by), inline: None } - } - pub fn reap() -> Self { - Op::Reap { job: None } + Op::Kill { pid, code, pc: 0, retire: Vec::new(), by: Some(by) } } pub fn spawn(pid: Pid) -> Self { Op::Spawn { pid, pc: 0, block: None } } pub fn thread_exit(pid: Pid, tid: Tid, code: i32) -> Self { - Op::ThreadExit { pid, tid, code, pc: 0, post: None } + Op::ThreadExit { pid, tid, code, pc: 0 } } pub fn join(pid: Pid, target: Tid, waiter: Tid) -> Self { Op::Join { pid, target, waiter, pc: 0 } } + pub fn poison(pid: Pid, tid: Tid) -> Self { + Op::Poison { pid, tid, pc: 0, owed: Poisoned::default() } + } pub fn idle_pass() -> Self { Op::IdlePass { pc: 0 } } - /// Whether the op has nothing left to do: the reaper is done once it is - /// parked with nothing owed to it. - fn done(&self, world: &World) -> bool { + /// The thread running the op, in the kernel until it returns. + fn actor(&self) -> Option<(Pid, Tid)> { + match *self { + Op::Exit { pid, tid, .. } | Op::ThreadExit { pid, tid, .. } | Op::Poison { pid, tid, .. } => { + Some((pid, tid)) + } + Op::Join { pid, waiter, .. } => Some((pid, waiter)), + Op::Kill { by, .. } => by, + Op::Spawn { .. } | Op::IdlePass { .. } => None, + } + } + + fn done(&self) -> bool { match self { - Op::Reap { job } => job.is_none() && world.reaper_asleep() && world.nothing_owed(), Op::Exit { pc, .. } | Op::Kill { pc, .. } | Op::Spawn { pc, .. } | Op::ThreadExit { pc, .. } | Op::Join { pc, .. } + | Op::Poison { pc, .. } | Op::IdlePass { pc, .. } => *pc == DONE, } } - /// Whether the op's next section can run, rather than wait. - fn enabled(&self, world: &World) -> bool { - match self { - Op::Exit { pid, pc: 2, others, next, .. } => { - *next == others.len() || world.is_retired(*pid, others[*next]) - } - Op::Kill { inline: Some(finish), .. } => finish.enabled(world), - Op::Reap { job: Some(finish) } => finish.enabled(world), - Op::Reap { job: None } => !world.reaper_asleep(), - _ => true, + /// The process whose teardown the op is waiting on, if its next section + /// cannot run until that process publishes its exit. + fn awaits_teardown(&self, world: &World) -> Option { + match *self { + Op::Kill { pid, pc: 2, .. } if world.published(pid).is_none() => Some(pid), + _ => None, } } @@ -182,10 +120,10 @@ impl Op { match self { Op::Exit { .. } => "exit", Op::Kill { .. } => "kill", - Op::Reap { .. } => "reaper", Op::Spawn { .. } => "spawn_thread", Op::ThreadExit { .. } => "thread_exit", Op::Join { .. } => "thread_join", + Op::Poison { .. } => "poison", Op::IdlePass { .. } => "idle pass", } } @@ -193,89 +131,41 @@ impl Op { /// Run one lock section. fn step(&mut self, world: &mut World) { match self { - Op::Exit { pid, tid, code, pc, others, next } => { + Op::Exit { pid, tid, code, pc, retire } => match *pc { + // The claim and the threads its winner retires, under the table lock. + 0 => { + if teardown::claim_teardown(world, *pid, *code) { + *retire = teardown::retire_set(world.get(*pid).expect("just claimed"), Some(*tid)); + } + *pc = 1; + } + // With the lock given up: the retires, then this thread's own way out. + _ => { + for &other in retire.iter() { + world.post_retire(*pid, other); + } + world.depart(*pid, *tid, None); + *pc = DONE; + } + }, + Op::Kill { pid, code, pc, retire, by } => { match *pc { - // Phase 1: claim the teardown and collect the threads to - // retire, under the table lock. 0 => { - let present = world.get(*pid).is_some_and(|p| p.location(*tid).is_some()); - if !present || !teardown::claim_teardown(world, *pid) { + if teardown::claim_teardown(world, *pid, *code) { + *retire = teardown::retire_set(world.get(*pid).expect("just claimed"), None); + *pc = 1; + } else { *pc = DONE; - return; } - let set = teardown::exit_set(world.get(*pid).expect("just claimed"), *tid); - *others = set.others; - // The claimant never returns to Ring 3 — `exit_current` - // is where it leaves — so from here it can no more - // touch a freed mapping than a retired thread can. It - // has released nobody yet, which is the difference and - // why this is not `retire`. - world.leaving(*pid, *tid); - *pc = 1; } - // Phase 2, with the table lock given up: every other - // thread's retire posted, then each one's release awaited. 1 => { - for &other in others.iter() { - world.post_retire(*pid, other); - } - *pc = 2; - } - 2 => { - if *next < others.len() { - *next += 1; - } else { - *pc = 3; - } - } - // Phase 4: the zombie marks, under the table lock. - 3 => { - if let Some(proc) = world.get_mut(*pid) { - teardown::mark_all_zombie(proc, *code); + for &tid in retire.iter() { + world.post_retire(*pid, tid); } - *pc = 4; - } - // Phase 5: publish, with the table lock given up. - 4 => { - world.publish_exit(*pid, *code); - *pc = 5; - } - // `exit_current`: the exit pass, one pass later, drops this - // thread's payload and `publish_released` posts on its own - // watch. - _ => { - world.retire(*pid, *tid); - *pc = DONE; + *pc = if cfg!(feature = "mutate-kill-waits-for-its-victims") { 2 } else { DONE }; } - } - } - Op::Kill { pid, code, pc, tids, by, inline } => { - if let Some(finish) = inline { - if finish.step(world) { - *pc = DONE; - } - } else if *pc == 0 { - if !teardown::claim_teardown(world, *pid) { - *pc = DONE; - } else { - *tids = teardown::kill_set(world.get(*pid).expect("just claimed")); - if cfg!(feature = "mutate-kill-waits-for-its-victims") { - *inline = Some(Finish::new((*pid, *code, tids.clone()))); - } else { - *pc = 1; - } - } - } else if *pc == 1 { - for &tid in tids.iter() { - world.post_retire(*pid, tid); - } - *pc = 2; - } else if *pc == 2 { - world.owe(*pid, *code, core::mem::take(tids)); - *pc = if by.is_some() { 3 } else { DONE }; - } else { - // The kill returned, and its thread reaches its safe point. - *pc = DONE; + // The mutation's wait, which `enabled` holds until the victim's exit is published. + _ => *pc = DONE, } if *pc == DONE { if let Some(by) = *by { @@ -283,14 +173,6 @@ impl Op { } } } - Op::Reap { job } => match job { - None => *job = world.reaper_takes().map(Finish::released), - Some(finish) => { - if finish.step(world) { - *job = None; - } - } - }, Op::Spawn { pid, pc, block } => match *pc { // Phase 1, under the table lock. 0 => { @@ -316,52 +198,69 @@ impl Op { *pc = DONE; } }, - Op::ThreadExit { pid, tid, code, pc, post } => match *pc { + Op::ThreadExit { pid, tid, code, pc } => match *pc { 0 => match teardown::route_thread_exit(world, *pid, *tid) { - teardown::ThreadExit::Sibling { post: on } => { - *post = Some(on); - *pc = 1; - } - // The entry went under this thread; the zombie mark has - // nothing to write and the rest is a sibling's exit. - teardown::ThreadExit::Gone { post: on } => { - *post = Some(on); - *pc = 2; - } + teardown::ThreadExit::Sibling { .. } => *pc = 1, // The explorer scripts a sibling; a main thread's exit is // `Op::Exit`. teardown::ThreadExit::Process => *pc = DONE, }, - // `release_thread`: the mappings go, then the zombie mark under - // the table lock. - 1 => { - world.set_location(*pid, *tid, ThreadLocation::Zombie(*code)); - *pc = 2; - } - // The post, with the table lock given up, before the exit pass - // — after it this thread does not run again. + // `release_thread`'s unmap, then this thread's own way out. _ => { - if let Some(on) = *post { - world.post(on); - } - world.retire(*pid, *tid); + world.unmap_own(*pid, *tid); + world.depart(*pid, *tid, Some(*code)); *pc = DONE; } }, - Op::Join { pid, target, waiter, pc } => match *pc { - 0 => match join::collect_zombie(world, *pid, *target) { - Ok(Some(_)) | Err(_) => *pc = DONE, - Ok(None) => { - world.arm(Watch::Thread(*pid, *target), (*pid, *waiter)); - *pc = 1; + Op::Join { pid, target, waiter, pc } => { + match *pc { + 0 => match join::collect_zombie(world, *pid, *target) { + Ok(Some(_)) => { + world.drop_collected(*pid, *target); + *pc = DONE; + } + Err(_) => *pc = DONE, + Ok(None) => { + world.arm(Watch::Thread(*pid, *target), (*pid, *waiter)); + *pc = 1; + } + }, + // `watch::wait_until` re-checks its predicate after the + // arm, so a zombie that appeared in the window is collected + // rather than waited for. + _ => { + if let Ok(Some(_)) = join::collect_zombie(world, *pid, *target) { + world.drop_collected(*pid, *target); + world.post(Watch::Thread(*pid, *target)); + } + *pc = DONE; } - }, - // `watch::wait_until` re-checks its predicate after the - // arm, so a zombie that appeared in the window is collected - // rather than waited for. + } + if *pc == DONE { + world.leave_kernel((*pid, *waiter)); + } + } + Op::Poison { pid, tid, pc, owed } => match *pc { + // The panic: the thread's exit pass, with every lock it held. + 0 => { + world.poison(*pid, *tid); + *pc = 1; + } + // The idle loop, under the table lock. + 1 => { + *owed = poison::zombify_poisoned(world, *pid, *tid); + *pc = 2; + } + // Its posts, with the lock given up. _ => { - if let Ok(Some(_)) = join::collect_zombie(world, *pid, *target) { - world.post(Watch::Thread(*pid, *target)); + for &other in &owed.retire { + world.post_retire(*pid, other); + } + if let Some(on) = owed.joiner { + world.post(on); + } + if let Some(code) = owed.exit { + world.publish_exit(*pid, code); } *pc = DONE; } @@ -378,18 +277,27 @@ impl Op { const DONE: u32 = u32::MAX; +/// One move the explorer can make: an op's next section, or a thread's next +/// step out. +#[derive(Clone, Copy)] +enum Move { + Op(usize), + Out(Pid, Tid), +} + /// How many distinct states every schedule reaches, or the first schedule that breaks a law. /// -/// Depth-first over "which op runs its next lock section", checking -/// `World::faults` at every state and `World::final_faults` at every leaf; a -/// state where no unfinished op can move is a deadlock. A state reached before -/// is not walked again: every law is a property of the state alone. The -/// returned string is the schedule that produced it, in the order the ops ran. +/// Depth-first over "which op runs its next lock section, or which thread +/// takes its next step out", checking `World::faults` and that no op waits on +/// another process's teardown at every state, and `World::final_faults` at +/// every leaf; a state where nothing can move is a deadlock. A state reached +/// before is not walked again: every law is a property of the state alone. The +/// returned string is the schedule that produced it, in the order it ran. pub fn explore(initial: &World, ops: &[Op]) -> Result { let mut world = initial.clone(); for op in ops { - if let Op::Kill { by: Some(by), .. } = op { - world.enter_kernel(*by); + if let Some(actor) = op.actor() { + world.enter_kernel(actor); } } let mut trace = Vec::new(); @@ -404,37 +312,48 @@ fn walk(world: World, ops: Vec, trace: &mut Vec, seen: &mut HashSet< if !seen.insert((world.clone(), ops.clone())) { return None; } - if ops.iter().all(|op| op.done(&world)) { - let faults = world.final_faults(); - return report(&faults, trace); - } - let mut moved = false; - for i in 0..ops.len() { - if ops[i].done(&world) || !ops[i].enabled(&world) { - continue; + let mut moves: Vec = (0..ops.len()) + .filter(|&i| !ops[i].done() && ops[i].awaits_teardown(&world).is_none()) + .map(Move::Op) + .collect(); + moves.extend(world.ready_departures().into_iter().map(|(pid, tid)| Move::Out(pid, tid))); + if moves.is_empty() { + let stuck: Vec<&str> = ops.iter().filter(|op| !op.done()).map(Op::label).collect(); + if stuck.is_empty() { + return report(&world.final_faults(), trace); } - moved = true; + return report(&[alloc::format!("deadlock: {} each wait and none can move", stuck.join(", "))], trace); + } + for m in moves { let mut next_world = world.clone(); let mut next_ops = ops.clone(); - let before = alloc::format!("{}#{i}", next_ops[i].label()); - next_ops[i].step(&mut next_world); - trace.push(before); - let faults = next_world.faults(); - if let Some(found) = report(&faults, trace) { - trace.pop(); - return Some(found); + let label = match m { + Move::Op(i) => { + let label = alloc::format!("{}#{i}", next_ops[i].label()); + next_ops[i].step(&mut next_world); + label + } + Move::Out(pid, tid) => { + next_world.depart_step(pid, tid); + alloc::format!("out {pid}/{tid}") + } + }; + trace.push(label); + let mut faults = next_world.faults(); + // L7. No kill or exit waits on another process's teardown. + for op in &next_ops { + if let Some(pid) = op.awaits_teardown(&next_world) { + if op.actor().map(|(own, _)| own) != Some(pid) { + faults.push(alloc::format!("a {} waits on pid {pid}'s teardown", op.label())); + } + } } - if let Some(found) = walk(next_world, next_ops, trace, seen) { + if let Some(found) = report(&faults, trace).or_else(|| walk(next_world, next_ops, trace, seen)) { trace.pop(); return Some(found); } trace.pop(); } - if !moved { - let stuck: Vec<&str> = - ops.iter().filter(|op| !op.done(&world)).map(Op::label).collect(); - return report(&[alloc::format!("deadlock: {} each wait and none can move", stuck.join(", "))], trace); - } None } @@ -449,6 +368,13 @@ fn report(faults: &[String], trace: &[String]) -> Option { mod tests { use super::*; + fn holds(world: &World, ops: Vec) -> usize { + match explore(world, &ops) { + Ok(states) => states, + Err(found) => panic!("a lifecycle law broke:\n{found}"), + } + } + /// **The negative control's subject, and #142's shape.** A `SYS_EXIT` on /// one thread and a `SYS_THREAD_SPAWN` on another, every ordering: the /// spawn's two lock sections have the whole build of a thread between them, @@ -456,23 +382,14 @@ mod tests { /// /// Reds under `mutate-spawn-skips-the-insert-recheck`, where the second /// question is not asked — the thread lands in the table behind the retire - /// sweep, its process publishes an exit without it, and the schedule that - /// did it is printed. + /// set, nothing retires it, and its process never ends. #[test] fn a_published_exit_leaves_no_unretired_thread() { let mut world = World::new(); let pid = world.spawn_process(); let main = world.main_tid(pid); - let sibling = world.spawn_thread(pid); - assert!( - !world.is_retired(pid, sibling), - "the sibling starts unretired, which is what the exit's sweep is for", - ); - - let ops = vec![Op::exit(pid, main, 0), Op::spawn(pid)]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + world.spawn_thread(pid); + holds(&world, vec![Op::exit(pid, main, 0), Op::spawn(pid)]); } /// The same window with the teardown coming from outside the process: a @@ -481,31 +398,26 @@ mod tests { fn a_kill_racing_a_spawn_leaves_no_unretired_thread() { let mut world = World::new(); let pid = world.spawn_process(); - let ops = vec![Op::kill(pid, 137), Op::spawn(pid), Op::reap()]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + holds(&world, vec![Op::kill(pid, 137), Op::spawn(pid)]); } /// **The second negative control's subject.** Two teardowns and one /// process: a `SYS_EXIT` on the process's own main thread and a /// `SYS_PROCESS_KILL` from a handle holder, in every ordering. Exactly one - /// claim may succeed and exactly one exit may be published — - /// `World::publish_exit` asserts the second half exactly as - /// `ProcessObject::publish_exit` does, `World::faults` the first. + /// claim may succeed, one teardown run and one exit be published — + /// `World::publish_exit` asserts the last exactly as + /// `ProcessObject::publish_exit` does, `World::faults` the first two. /// /// Reds under `mutate-claim-teardown-always-wins`, where both paths retire - /// the same threads and both publish. + /// the same threads, and under `mutate-first-out-tears-down`, where the + /// first thread out frees what its sibling still runs in. #[test] fn an_exit_and_a_kill_never_both_tear_a_process_down() { let mut world = World::new(); let pid = world.spawn_process(); let main = world.main_tid(pid); world.spawn_thread(pid); - let ops = vec![Op::exit(pid, main, 0), Op::kill(pid, 137), Op::reap()]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + holds(&world, vec![Op::exit(pid, main, 0), Op::kill(pid, 137)]); } /// A sibling exits while another sibling — not the main thread — is joining @@ -518,10 +430,7 @@ mod tests { let pid = world.spawn_process(); let waiter = world.spawn_thread(pid); let dying = world.spawn_thread(pid); - let ops = vec![Op::thread_exit(pid, dying, 3), Op::join(pid, dying, waiter)]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + holds(&world, vec![Op::thread_exit(pid, dying, 3), Op::join(pid, dying, waiter)]); } /// A teardown, an idle pass taking the entry, and a third thread joining a @@ -536,32 +445,20 @@ mod tests { let main = world.main_tid(pid); let dying = world.spawn_thread(pid); let joiner = world.spawn_thread(pid); - let ops = vec![ - Op::exit(pid, main, 0), - Op::idle_pass(), - Op::join(pid, dying, joiner), - ]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + holds(&world, vec![Op::exit(pid, main, 0), Op::idle_pass(), Op::join(pid, dying, joiner)]); } /// **The three-way form of #142's window**: a kill, a spawn racing its - /// sweep, and the idle pass that takes the entry out of the table. Two - /// operations were what the crate landed with, and the sighting is a machine - /// where the reaper is a third CPU rather than a phase of one of them. + /// retire set, and the idle pass that takes the entry out of the table. #[test] fn a_spawn_racing_a_kill_and_the_pass_that_reaps_it() { let mut world = World::new(); let pid = world.spawn_process(); - let ops = vec![Op::kill(pid, 137), Op::spawn(pid), Op::idle_pass(), Op::reap()]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + holds(&world, vec![Op::kill(pid, 137), Op::spawn(pid), Op::idle_pass()]); } /// A sibling leaving through `SYS_THREAD_EXIT` while the main thread's own - /// exit sweeps the process and a third thread is being built. + /// exit claims the process and a third thread is being built. /// /// The three doors out of a process at once, which is what the T14 log /// shows: `/system/bin/ls` spawning while a shell reaps and a terminal exits. @@ -571,18 +468,11 @@ mod tests { let pid = world.spawn_process(); let main = world.main_tid(pid); let sibling = world.spawn_thread(pid); - let ops = vec![ - Op::exit(pid, main, 0), - Op::thread_exit(pid, sibling, 3), - Op::spawn(pid), - ]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + holds(&world, vec![Op::exit(pid, main, 0), Op::thread_exit(pid, sibling, 3), Op::spawn(pid)]); } /// Two spawns and one teardown: the window admits at most one thread behind - /// the sweep, and the second question is asked of each of them separately. + /// the claim, and the second question is asked of each of them separately. /// /// One spawn cannot show that. The insert recheck reads a flag rather than /// a count, so a rule that admitted *the first* arrival after the claim and @@ -592,10 +482,7 @@ mod tests { let mut world = World::new(); let pid = world.spawn_process(); let main = world.main_tid(pid); - let ops = vec![Op::exit(pid, main, 0), Op::spawn(pid), Op::spawn(pid)]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + holds(&world, vec![Op::exit(pid, main, 0), Op::spawn(pid), Op::spawn(pid)]); } /// The teeth behind L5: the shipped shape — a refused insert dropping the @@ -612,7 +499,7 @@ mod tests { spawning.step(&mut world); // phase 2: the block is mapped let mut exit = Op::exit(pid, main, 0); - while !exit.done(&world) { + while !exit.done() { exit.step(&mut world); } assert!( @@ -628,74 +515,27 @@ mod tests { ); } - /// A `SYS_THREAD_JOIN` armed on a thread the killer is about to retire, in - /// every ordering — the waiter that L4 is about and the one shape where it - /// is answered by a teardown rather than by the thread it named. + /// A `SYS_THREAD_JOIN` armed on a thread the kill is about to retire, in + /// every ordering — the waiter that L4 is about, and a joiner that must not + /// take its target's entry while a sibling still runs in its mappings. + /// + /// Reds under `mutate-join-collects-in-a-teardown`. #[test] fn a_join_racing_the_kill_that_takes_its_target() { let mut world = World::new(); let pid = world.spawn_process(); let target = world.spawn_thread(pid); let waiter = world.spawn_thread(pid); - let ops = vec![Op::kill(pid, 137), Op::join(pid, target, waiter), Op::reap()]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } - } - - /// **A thread whose entry went under it still finishes its own exit**: a - /// kill publishes, an idle pass takes the entry, and a sibling already on - /// its way into `SYS_THREAD_EXIT` arrives after both. - #[test] - fn a_thread_exit_that_outlived_its_entry_still_leaves() { - let mut world = World::new(); - let pid = world.spawn_process(); - let sibling = world.spawn_thread(pid); - - // The schedule, run by hand rather than searched for: a kill and the - // reaper that runs it to its publish, an idle pass, and only then the - // sibling's own exit. - let mut kill = Op::kill(pid, 137); - while !kill.done(&world) { - kill.step(&mut world); - } - let mut reaper = Op::reap(); - while !reaper.done(&world) { - reaper.step(&mut world); - } - let mut idle = Op::idle_pass(); - idle.step(&mut world); - assert!(world.was_reaped(pid), "the idle pass took the published entry"); - - assert_eq!( - teardown::route_thread_exit(&world, pid, sibling), - teardown::ThreadExit::Gone { post: Watch::Thread(pid, sibling) }, - "a thread whose process was reaped under it leaves by the sibling door", - ); - - // The route is not the end of the exit: a thread routed to a state its - // caller does not survive stops here, and one out of the sibling door - // still has its post and its retire to run. - let mut exit = Op::thread_exit(pid, sibling, 0); - exit.step(&mut world); - assert!( - !exit.done(&world), - "the exit ended at its routing section: a thread whose entry went under it \ - still has to post and retire, and one that stops here is the machine \ - stopping with it", - ); - while !exit.done(&world) { - exit.step(&mut world); - } + holds(&world, vec![Op::kill(pid, 137), Op::join(pid, target, waiter)]); } /// **`KillEachOther`, K1's shape**: two processes, each one's thread inside /// `SYS_PROCESS_KILL` on the other. A thread in the kernel reaches no safe /// point until its kill returns, so a kill that waited for its victim waits - /// on a thread that is waiting on it. Each process has a second thread, and - /// one of the killers is it. + /// on a process that is waiting on it. Each process has a second thread, + /// and one of the killers is it. /// - /// Reds under `mutate-kill-waits-for-its-victims`, the base's kill. + /// Reds under `mutate-kill-waits-for-its-victims`. #[test] fn two_processes_killing_each_other_both_end() { let mut world = World::new(); @@ -703,15 +543,11 @@ mod tests { let p_sibling = world.spawn_thread(p); let c = world.spawn_process(); world.spawn_thread(c); - let ops = vec![ - Op::kill_by(c, 137, (p, p_sibling)), - Op::kill_by(p, 137, (c, world.main_tid(c))), - Op::reap(), - ]; - match explore(&world, &ops) { - Ok(states) => std::println!("KillEachOther: {states} states, every schedule ends"), - Err(found) => panic!("a lifecycle law broke:\n{found}"), - } + let states = holds( + &world, + vec![Op::kill_by(c, 137, (p, p_sibling)), Op::kill_by(p, 137, (c, world.main_tid(c)))], + ); + std::println!("KillEachOther: {states} states, every schedule ends"); } /// The cycle at length three: A kills B, B kills C, C kills A. @@ -721,34 +557,65 @@ mod tests { let a = world.spawn_process(); let b = world.spawn_process(); let c = world.spawn_process(); - let ops = vec![ - Op::kill_by(b, 137, (a, world.main_tid(a))), - Op::kill_by(c, 137, (b, world.main_tid(b))), - Op::kill_by(a, 137, (c, world.main_tid(c))), - Op::reap(), - ]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + holds( + &world, + vec![ + Op::kill_by(b, 137, (a, world.main_tid(a))), + Op::kill_by(c, 137, (b, world.main_tid(b))), + Op::kill_by(a, 137, (c, world.main_tid(c))), + ], + ); } /// A sibling in the cycle and an exit racing it: P's main thread exits - /// while P's second thread kills C and C kills P — so the exit's own - /// retire waits on a thread that is inside a kill. + /// while P's second thread kills C and C kills P. #[test] fn an_exit_whose_sibling_kills_the_process_killing_it() { let mut world = World::new(); let p = world.spawn_process(); let killer = world.spawn_thread(p); let c = world.spawn_process(); - let ops = vec![ - Op::exit(p, world.main_tid(p), 0), - Op::kill_by(c, 137, (p, killer)), - Op::kill_by(p, 137, (c, world.main_tid(c))), - Op::reap(), - ]; - if let Err(found) = explore(&world, &ops) { - panic!("a lifecycle law broke:\n{found}"); - } + holds( + &world, + vec![ + Op::exit(p, world.main_tid(p), 0), + Op::kill_by(c, 137, (p, killer)), + Op::kill_by(p, 137, (c, world.main_tid(c))), + ], + ); + } + + /// A thread killing its own process, with a sibling: it retires itself and + /// leaves when its kill returns. + #[test] + fn a_process_that_kills_itself_ends() { + let mut world = World::new(); + let p = world.spawn_process(); + let sibling = world.spawn_thread(p); + holds(&world, vec![Op::kill_by(p, 137, (p, sibling))]); + } + + /// A thread of a process being killed dies in panic recovery instead of + /// leaving: the idle loop takes it out, and whichever of it and its sibling + /// is last publishes the kill's exit. + #[test] + fn a_kill_racing_a_poisoned_thread_ends() { + let mut world = World::new(); + let p = world.spawn_process(); + let main = world.main_tid(p); + world.spawn_thread(p); + holds(&world, vec![Op::kill(p, 137), Op::poison(p, main)]); + } + + /// A poisoned main thread ends its process: its sibling is retired and the + /// last one out publishes, racing an exit on the sibling itself. + #[test] + fn a_poisoned_main_thread_ends_its_siblings() { + let mut world = World::new(); + let p = world.spawn_process(); + let main = world.main_tid(p); + let sibling = world.spawn_thread(p); + let states = holds(&world, vec![Op::poison(p, main), Op::exit(p, sibling, 3)]); + assert!(states > 0); } } diff --git a/toyos-proclife/src/join.rs b/toyos-proclife/src/join.rs index 666055ba7e7..2e26826ebf4 100644 --- a/toyos-proclife/src/join.rs +++ b/toyos-proclife/src/join.rs @@ -39,6 +39,12 @@ pub fn collect_zombie( ) -> Result, JoinRefused> { let proc = table.get_mut(pid).ok_or(JoinRefused::NoSuchProcess)?; let at = proc.location(tid).ok_or(JoinRefused::NoSuchThread)?; + // A thread that left a claimed process still holds mappings its siblings + // may run in until the last one out; the joiner leaves with it instead. + #[cfg(not(feature = "mutate-join-collects-in-a-teardown"))] + if proc.tearing_down() { + return Ok(None); + } match at.zombie_code() { Some(code) => { proc.forget_thread(tid); @@ -73,6 +79,18 @@ mod tests { assert_eq!(collect_zombie(&mut world, pid, t1), Err(JoinRefused::NoSuchThread)); } + #[cfg(not(feature = "mutate-join-collects-in-a-teardown"))] + #[test] + fn a_process_being_torn_down_gives_up_no_thread() { + let mut world = World::new(); + let pid = world.spawn_process(); + let t1 = world.spawn_thread(pid); + assert!(crate::teardown::claim_teardown(&mut world, pid, 137)); + let _ = crate::teardown::leave(&mut world, pid, t1, None); + assert_eq!(collect_zombie(&mut world, pid, t1), Ok(None)); + assert!(world.get(pid).unwrap().location(t1).is_some(), "the entry keeps the thread"); + } + #[test] fn the_two_refusals_are_told_apart() { let mut world = World::new(); diff --git a/toyos-proclife/src/model.rs b/toyos-proclife/src/model.rs index 413d994618b..dc48fba3ebd 100644 --- a/toyos-proclife/src/model.rs +++ b/toyos-proclife/src/model.rs @@ -12,18 +12,19 @@ //! depends on the order, and a model whose counter-example is different every //! run is a model nobody can bisect. -use alloc::collections::{BTreeMap, BTreeSet, VecDeque}; +use alloc::collections::{BTreeMap, BTreeSet}; use alloc::string::String; use alloc::vec::Vec; use crate::table::{Lifecycle, Processes}; -use crate::{teardown, Pid, ThreadLocation, Tid, Watch}; +use crate::teardown::{self, Leave}; +use crate::{Pid, ThreadLocation, Tid, Watch}; /// One process, as its lifecycle sees it. #[derive(Clone, PartialEq, Eq, Hash)] pub struct ModelProc { main_tid: Tid, - tearing_down: bool, + teardown_code: Option, threads: BTreeMap, next_tid: Tid, /// How many times a claim was raised on this process. Counted here because @@ -35,11 +36,11 @@ impl Lifecycle for ModelProc { fn main_tid(&self) -> Tid { self.main_tid } - fn tearing_down(&self) -> bool { - self.tearing_down + fn teardown_code(&self) -> Option { + self.teardown_code } - fn begin_teardown(&mut self) { - self.tearing_down = true; + fn begin_teardown(&mut self, code: i32) { + self.teardown_code = Some(code); self.claims += 1; } fn location(&self, tid: Tid) -> Option { @@ -60,6 +61,31 @@ impl Lifecycle for ModelProc { } } +/// Where a thread is on its way out, one lock section or pass per step. +#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)] +enum Out { + /// `process::leave`'s table section. + Leave, + /// The last one out: `teardown_resources`, the process's mappings and handles. + Free { code: i32 }, + /// The last one out: `ProcessObject::publish_exit`. + Publish { code: i32 }, + /// `thread_exit`'s post on its own watch. + Post, + /// The exit pass switches away from the thread. + Switch, + /// The next pass on that CPU frees the payload, kernel stack and all. + Release, + Gone, +} + +/// A thread leaving: the code it chose, and how far it is. +#[derive(Clone, Copy, PartialEq, Eq, Hash, Debug)] +struct Departure { + chosen: Option, + at: Out, +} + /// The table, plus the effects the kernel would have performed. #[derive(Clone, PartialEq, Eq, Hash)] pub struct World { @@ -72,23 +98,25 @@ pub struct World { waiters: BTreeSet<(Watch, Pid, Tid)>, /// Waiters a post has released. released: BTreeSet<(Watch, Pid, Tid)>, - /// Threads released off every CPU. A thread not - /// in here may still be picked and run. - retired: BTreeSet<(Pid, Tid)>, - /// Threads past the point of no return but not yet dropped by an exit - /// pass: a claimant inside its own teardown, which cannot retire itself and - /// will not reach Ring 3 again either. - leaving: BTreeSet<(Pid, Tid)>, /// Threads a retire was posted on: the kill bit. killed: BTreeSet<(Pid, Tid)>, /// Threads inside a scripted operation, which reach no safe point until it returns. in_kernel: BTreeSet<(Pid, Tid)>, - /// Teardowns a kill handed to the reaper, oldest first: the process, its code, its threads. - owed: VecDeque<(Pid, i32, Vec)>, - /// The reaper is parked on `ANY_RELEASED` and nothing has posted it since it armed. - reaper_asleep: bool, - /// Entries an idle pass has taken out of the table. - reaped: BTreeSet, + /// Threads on their way out, and how far each is. + departing: BTreeMap<(Pid, Tid), Departure>, + /// Threads the scheduler has switched away from for the last time. + switched: BTreeSet<(Pid, Tid)>, + /// Threads whose kernel stack has been freed. + stacks_freed: BTreeSet<(Pid, Tid)>, + /// Threads that died in panic recovery and never run again. + poisoned: BTreeSet<(Pid, Tid)>, + /// Threads whose own TLS block is still mapped in their process. + mapped: BTreeSet<(Pid, Tid)>, + /// Threads whose entry went, and its mapped TLS block with it, while a + /// sibling was still in the process. + unmapped_under_siblings: BTreeSet<(Pid, Tid)>, + /// How many times each process's resources were freed. + frees: BTreeMap, /// TLS blocks a spawn's phase 2 mapped and no thread owns yet. tls_mapped: BTreeSet, next_tls: u32, @@ -121,13 +149,15 @@ impl World { published: BTreeMap::new(), waiters: BTreeSet::new(), released: BTreeSet::new(), - retired: BTreeSet::new(), - leaving: BTreeSet::new(), killed: BTreeSet::new(), in_kernel: BTreeSet::new(), - owed: VecDeque::new(), - reaper_asleep: true, - reaped: BTreeSet::new(), + departing: BTreeMap::new(), + switched: BTreeSet::new(), + stacks_freed: BTreeSet::new(), + poisoned: BTreeSet::new(), + mapped: BTreeSet::new(), + unmapped_under_siblings: BTreeSet::new(), + frees: BTreeMap::new(), tls_mapped: BTreeSet::new(), next_tls: 0, } @@ -162,22 +192,24 @@ impl World { pid, ModelProc { main_tid: Tid(0), - tearing_down: false, + teardown_code: None, threads, next_tid: Tid(1), claims: 0, }, ); + self.mapped.insert((pid, Tid(0))); pid } /// Insert a thread the way `spawn_thread`'s phase 3 does — in the table and - /// enqueued in the scheduler, so it is alive and unretired. + /// enqueued in the scheduler, so it is alive and in the process. pub fn spawn_thread(&mut self, pid: Pid) -> Tid { let proc = self.procs.get_mut(&pid).expect("spawn_thread on a live process"); let tid = proc.next_tid; proc.next_tid = Tid(tid.0 + 1); proc.threads.insert(tid, ThreadLocation::Scheduled); + self.mapped.insert((pid, tid)); tid } @@ -197,6 +229,23 @@ impl World { } } + /// The `ThreadEntry` a join collected is dropped, and whatever its + /// `ThreadData` still maps goes back with it. + pub fn drop_collected(&mut self, pid: Pid, tid: Tid) { + let mut still_in = false; + if let Some(proc) = self.procs.get(&pid) { + proc.each_thread(&mut |_, at| still_in |= !at.is_zombie()); + } + if self.mapped.remove(&(pid, tid)) && still_in { + self.unmapped_under_siblings.insert((pid, tid)); + } + } + + /// `release_thread`: a thread's own exit unmaps its TLS block before it leaves. + pub fn unmap_own(&mut self, pid: Pid, tid: Tid) { + self.mapped.remove(&(pid, tid)); + } + /// `watch::wait_until` — a waiter registered on a subject's watch. pub fn arm(&mut self, on: Watch, waiter: (Pid, Tid)) { self.waiters.insert((on, waiter.0, waiter.1)); @@ -220,53 +269,10 @@ impl World { self.waiters.difference(&self.released).copied().collect() } - /// `Hw::release` — the thread is provably off every CPU, its - /// payload dropped, and `publish_released` has posted on its own - /// watch. - /// - /// **The post is not an extra the model added.** `KernelPayload`'s release - /// sink runs exactly once per task and ends with - /// `TaskHandle::publish_released`, whose post is on the thread's own watch - /// — "the same subject a joiner uses, and the reason the release no longer - /// needs a queue of its own" (`kernel/src/sched/payload.rs`). A model - /// without it strands joiners the kernel releases. - /// - /// A killed thread's release also posts `ANY_RELEASED`, the reaper's watch. - pub fn retire(&mut self, pid: Pid, tid: Tid) { - self.retired.insert((pid, tid)); - self.post(Watch::Thread(pid, tid)); - if self.killed.contains(&(pid, tid)) { - self.reaper_asleep = false; - } - } - - /// A thread that has committed to leaving but whose payload the exit pass - /// has not dropped yet — a teardown claimant, running the teardown on its - /// own kernel stack. It will not reach Ring 3 again, so it is as unable to - /// touch a freed mapping as a retired thread; what it has not done yet is - /// release anybody. - pub fn leaving(&mut self, pid: Pid, tid: Tid) { - self.leaving.insert((pid, tid)); - } - - pub fn is_retired(&self, pid: Pid, tid: Tid) -> bool { - self.retired.contains(&(pid, tid)) - } - - /// `scheduler::post_retire`: the kill bit, and a thread at no safe point - /// dies only when its own operation returns. + /// `scheduler::post_retire`: the kill bit. A thread in Ring 3 leaves at its + /// next exit boundary, one in a scripted operation when the operation returns. pub fn post_retire(&mut self, pid: Pid, tid: Tid) { - if self.is_retired(pid, tid) { - return; - } assert!(self.killed.insert((pid, tid)), "a second retirer for pid {pid} tid {tid}"); - if !self.in_kernel.contains(&(pid, tid)) { - self.retire(pid, tid); - } - } - - pub fn is_killed(&self, pid: Pid, tid: Tid) -> bool { - self.killed.contains(&(pid, tid)) } /// A thread starting a scripted operation. @@ -274,48 +280,97 @@ impl World { self.in_kernel.insert(by); } - /// Its operation returned: the safe point, where a killed thread dies. + /// Its operation returned. pub fn leave_kernel(&mut self, by: (Pid, Tid)) { self.in_kernel.remove(&by); - if self.killed.contains(&by) { - self.retire(by.0, by.1); - } } - /// `reaper::owe`. - pub fn owe(&mut self, pid: Pid, code: i32, tids: Vec) { - self.owed.push_back((pid, code, tids)); - self.reaper_asleep = false; - } - - /// The reaper's check, after it armed: the kill it takes, or `None` and it parks. - pub fn reaper_takes(&mut self) -> Option<(Pid, i32, Vec)> { - let at = teardown::next_released(self.owed.iter().map(|owed| self.all_retired(owed))); - self.reaper_asleep = at.is_none(); - self.owed.remove(at?) + /// A thread in a scripted operation starting its own way out. + pub fn depart(&mut self, pid: Pid, tid: Tid, chosen: Option) { + assert!( + self.departing.insert((pid, tid), Departure { chosen, at: Out::Leave }).is_none(), + "pid {pid} tid {tid} started out twice", + ); + self.in_kernel.remove(&(pid, tid)); + } + + /// Every thread with a step of its way out it can take now: one already on + /// its way, and a killed one at its exit boundary. + pub fn ready_departures(&self) -> Vec<(Pid, Tid)> { + let mut ready: Vec<(Pid, Tid)> = self + .departing + .iter() + .filter(|(_, d)| d.at != Out::Gone) + .map(|(&id, _)| id) + .collect(); + for &(pid, tid) in &self.killed { + let at_boundary = !self.in_kernel.contains(&(pid, tid)) + && !self.departing.contains_key(&(pid, tid)) + && !self.poisoned.contains(&(pid, tid)) + && self.procs.get(&pid).and_then(|p| p.location(tid)) == Some(ThreadLocation::Scheduled); + if at_boundary { + ready.push((pid, tid)); + } + } + ready } - pub fn reaper_asleep(&self) -> bool { - self.reaper_asleep + /// One step of a thread's way out. + pub fn depart_step(&mut self, pid: Pid, tid: Tid) { + let departure = self.departing.entry((pid, tid)).or_insert(Departure { chosen: None, at: Out::Leave }); + let chosen = departure.chosen; + let next = match departure.at { + Out::Leave => match teardown::leave(self, pid, tid, chosen) { + Leave::Last { code } => Out::Free { code }, + Leave::NotLast => Out::Post, + }, + Out::Free { code } => { + *self.frees.entry(pid).or_insert(0) += 1; + self.mapped.retain(|&(p, _)| p != pid); + Out::Publish { code } + } + Out::Publish { code } => { + self.publish_exit(pid, code); + Out::Post + } + Out::Post => { + if chosen.is_some() { + self.post(Watch::Thread(pid, tid)); + } + Out::Switch + } + Out::Switch => { + self.switched.insert((pid, tid)); + Out::Release + } + Out::Release => { + self.free_stack(pid, tid); + Out::Gone + } + Out::Gone => unreachable!("a thread that is gone has no step left"), + }; + self.departing.get_mut(&(pid, tid)).expect("inserted above").at = next; } - fn all_retired(&self, (pid, _, tids): &(Pid, i32, Vec)) -> bool { - tids.iter().all(|&tid| self.is_retired(*pid, tid)) + /// `Hw::release`: the payload goes, and the kernel stack with it. + pub fn free_stack(&mut self, pid: Pid, tid: Tid) { + self.stacks_freed.insert((pid, tid)); } - pub fn nothing_owed(&self) -> bool { - self.owed.is_empty() + /// A thread dying in panic recovery: `schedule_no_return`'s exit pass, and + /// the release after it. + pub fn poison(&mut self, pid: Pid, tid: Tid) { + self.in_kernel.remove(&(pid, tid)); + self.poisoned.insert((pid, tid)); + self.switched.insert((pid, tid)); + self.free_stack(pid, tid); } /// Whether this thread can still execute user code. fn runnable(&self, pid: Pid, tid: Tid) -> bool { - !self.retired.contains(&(pid, tid)) - && !self.leaving.contains(&(pid, tid)) - && self - .procs - .get(&pid) - .and_then(|p| p.location(tid)) - .is_some_and(|at| !at.is_zombie()) + !self.stacks_freed.contains(&(pid, tid)) + && !self.departing.contains_key(&(pid, tid)) + && !self.killed.contains(&(pid, tid)) } /// `ProcessObject::publish_exit`, assertion and all: two publishes mean two @@ -328,15 +383,14 @@ impl World { self.post(Watch::Process(pid)); } + pub fn published(&self, pid: Pid) -> Option { + self.published.get(&pid).copied() + } + /// The idle pass taking an entry, which is what `reap_finished` returns for /// the caller to drop. pub fn reap(&mut self, pid: Pid) { self.procs.remove(&pid); - self.reaped.insert(pid); - } - - pub fn was_reaped(&self, pid: Pid) -> bool { - self.reaped.contains(&pid) } /// **The laws, checked at every state a step leaves behind.** @@ -346,39 +400,37 @@ impl World { pub fn faults(&self) -> Vec { let mut out = Vec::new(); for (&pid, proc) in &self.procs { - // L1. One process, one teardown, one exit. `publish_exit` asserts - // the other half of this; this is the half that can be seen before - // the publish happens. + // L1. One process, one claim, one teardown. `publish_exit` asserts + // the publish half. if proc.claims > 1 { out.push(alloc::format!("pid {pid}: {} teardown claims succeeded", proc.claims)); } - if self.published.contains_key(&pid) { + let frees = self.frees.get(&pid).copied().unwrap_or(0); + if frees > 1 { + out.push(alloc::format!("pid {pid}: its resources were freed {frees} times")); + } + // L2. Nothing a thread can still run in is freed or published + // before every thread has left. + if frees > 0 || self.published.contains_key(&pid) { for (&tid, &at) in &proc.threads { - // L2. An exit is published only once every thread of the - // process is dead. if !at.is_zombie() { out.push(alloc::format!( - "pid {pid} published its exit with tid {tid} still Scheduled", - )); - } - // L3. **And only once every thread is provably off every - // CPU.** The stronger half, and the one a thread inserted - // behind a retire sweep breaks: the marks are a table write - // and reach a thread the sweep never named, the retire is - // what stops it running. - if !self.retired.contains(&(pid, tid)) && !self.leaving.contains(&(pid, tid)) { - out.push(alloc::format!( - "pid {pid} published its exit with tid {tid} never retired", + "pid {pid} was torn down with tid {tid} still in it", )); } } } } - // L7. The reaper never sleeps while a kill it could finish is owed. - if self.reaper_asleep { - if let Some((pid, _, _)) = self.owed.iter().find(|owed| self.all_retired(owed)) { - out.push(alloc::format!("the reaper sleeps while pid {pid}'s released kill is owed")); - } + // L3. A thread's stack is freed only once the scheduler has switched + // away from it. + for id in self.stacks_freed.difference(&self.switched) { + out.push(alloc::format!("pid {} tid {}: its stack was freed while it ran on it", id.0, id.1)); + } + // L8. No thread's mappings go while a sibling can still run in them. + for (pid, tid) in &self.unmapped_under_siblings { + out.push(alloc::format!( + "pid {pid} tid {tid}: its entry and its mapped TLS went while a sibling was still in the process", + )); } out } @@ -388,7 +440,8 @@ impl World { /// to its end. A join that has not been answered *yet* is the ordinary /// case, so checking it at every state would report every schedule. /// And **L5** — every TLS block a spawn mapped ends owned or released. - /// And **L6** — every claimed teardown ends in a published exit. + /// And **L6** — every claimed process is torn down: its exit published, + /// every killed thread gone. pub fn final_faults(&self) -> Vec { let mut out = self.faults(); for (&pid, proc) in &self.procs { @@ -396,6 +449,11 @@ impl World { out.push(alloc::format!("pid {pid} was claimed for teardown and never published an exit")); } } + for &(pid, tid) in &self.killed { + if !self.stacks_freed.contains(&(pid, tid)) && self.procs.get(&pid).is_some_and(|p| p.location(tid).is_some()) { + out.push(alloc::format!("pid {pid} tid {tid} was killed and never left")); + } + } for &block in &self.tls_mapped { out.push(alloc::format!( "TLS block {block} is mapped and owned by nobody — a refused spawn dropped it \ @@ -429,3 +487,22 @@ impl World { out } } + +#[cfg(test)] +mod tests { + use super::*; + + /// The teeth behind L3: a stack freed before the switch is what it reports. + /// Run by hand, since no kernel path the model scripts frees one early. + #[test] + fn a_stack_freed_before_the_switch_is_what_l3_reports() { + let mut world = World::new(); + let pid = world.spawn_process(); + world.free_stack(pid, world.main_tid(pid)); + let faults = world.faults(); + assert!( + faults.iter().any(|f| f.contains("freed while it ran on it")), + "L3 cannot see a stack freed under its thread: {faults:?}", + ); + } +} diff --git a/toyos-proclife/src/poison.rs b/toyos-proclife/src/poison.rs index 6581ee5f3cd..ff8c2b2f7da 100644 --- a/toyos-proclife/src/poison.rs +++ b/toyos-proclife/src/poison.rs @@ -5,67 +5,55 @@ //! the idle loop runs this later, which is the one context that provably holds //! none of them. //! -//! **It is a teardown like any other and takes the same claim.** A poisoned -//! main thread ends its process, so it competes with a `SYS_EXIT` on another -//! thread and with a `SYS_PROCESS_KILL` from a handle holder, and exactly one -//! of the three publishes the exit. What makes this path different is only what -//! it *cannot* do: no resources are released, because every release below it -//! wants a lock the faulted thread may still be recorded as holding, so the -//! process's mappings and handles go with the table entry rather than before -//! it. +//! **It is the dead thread's leaving, done for it.** A poisoned main thread +//! ends its process, so it takes the same claim an exit or a kill takes and +//! retires the rest; any poisoned thread is out of its process, and when it is +//! the last one out the exit is published here. What makes this path different +//! is only what it *cannot* do: no resources are released, because every +//! release below it wants a lock the faulted thread may still be recorded as +//! holding, so the process's mappings and handles go with the table entry +//! rather than before it. + +use alloc::vec::Vec; use crate::table::{Lifecycle, Processes}; -use crate::teardown; +use crate::teardown::{self, Leave}; use crate::{Pid, ThreadLocation, Tid, Watch, TORN_DOWN_THREAD_CODE}; -/// What must be woken for a poisoned thread, once the table lock is given up. -/// -/// Both wakes are carried out by the caller rather than performed here, because -/// both must happen with that lock released. +/// What must be done for a poisoned thread, once the table lock is given up. #[must_use = "a poisoned thread's waiter must be woken"] -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum PoisonOutcome { - /// Nothing to do: the entry is gone, or another path already owns this - /// process's teardown and will publish its exit. - Nothing, - /// A child thread died. **The subject names the thread that died**, which - /// is what a `thread_join` arms on — it used to name the process's main - /// thread, because the wake was by name into a shared parking lot and - /// whoever was woken re-checked. - Joiner(Watch), - /// The main thread died, so the process is over. The exit is published on - /// the object — outside the table lock, like every other publish — and - /// whoever holds a handle reads it there. - Process(Pid), +#[derive(Clone, PartialEq, Eq, Hash, Debug, Default)] +pub struct Poisoned { + /// The threads its process's claim retires: a main thread's death is its + /// process's. + pub retire: Vec, + /// A thread that is not the main one: what its joiner armed on. + pub joiner: Option, + /// It was the last thread out: the exit to publish, with nothing torn down. + pub exit: Option, } -/// Mark a poisoned thread dead and name what must be woken for it. -pub fn zombify_poisoned(table: &mut T, pid: Pid, tid: Tid) -> PoisonOutcome { - let Some(proc) = table.get(pid) else { return PoisonOutcome::Nothing }; - let main_tid = proc.main_tid(); - - if tid != main_tid { - if proc.location(tid).is_none() { - return PoisonOutcome::Nothing; - } - let proc = table.get_mut(pid).expect("the entry answered a line ago"); - if proc.location(tid).is_some_and(|l| !l.is_zombie()) { - proc.set_location(tid, ThreadLocation::Zombie(TORN_DOWN_THREAD_CODE)); - } - return PoisonOutcome::Joiner(Watch::Thread(pid, tid)); +/// Take a poisoned thread out of its process and say what that leaves to do. +pub fn zombify_poisoned(table: &mut T, pid: Pid, tid: Tid) -> Poisoned { + let mut owed = Poisoned::default(); + let Some(proc) = table.get_mut(pid) else { return owed }; + let Some(at) = proc.location(tid) else { return owed }; + let main = tid == proc.main_tid(); + if !main { + owed.joiner = Some(Watch::Thread(pid, tid)); } - - // The same claim every exit and kill takes, for the same reason: exactly - // one path publishes one exit. - if !teardown::claim_teardown(table, pid) { - return PoisonOutcome::Nothing; + // Left already, and died on the way out. + if at != ThreadLocation::Scheduled { + return owed; } - let proc = table.get_mut(pid).expect("the claim just succeeded on this entry"); - if proc.location(tid).is_none() { - return PoisonOutcome::Nothing; + if main && teardown::claim_teardown(table, pid, TORN_DOWN_THREAD_CODE) { + let proc = table.get(pid).expect("the claim just succeeded on this entry"); + owed.retire = teardown::retire_set(proc, Some(tid)); } - proc.set_location(tid, ThreadLocation::Zombie(TORN_DOWN_THREAD_CODE)); - PoisonOutcome::Process(pid) + if let Leave::Last { code } = teardown::leave(table, pid, tid, None) { + owed.exit = Some(code); + } + owed } #[cfg(test)] @@ -78,7 +66,8 @@ mod tests { let mut world = World::new(); let pid = world.spawn_process(); let t1 = world.spawn_thread(pid); - assert_eq!(zombify_poisoned(&mut world, pid, t1), PoisonOutcome::Joiner(Watch::Thread(pid, t1))); + let owed = zombify_poisoned(&mut world, pid, t1); + assert_eq!(owed, Poisoned { joiner: Some(Watch::Thread(pid, t1)), ..Poisoned::default() }); assert_eq!( world.get(pid).unwrap().location(t1), Some(ThreadLocation::Zombie(TORN_DOWN_THREAD_CODE)), @@ -90,32 +79,43 @@ mod tests { } #[test] - fn a_poisoned_main_thread_takes_the_claim_and_ends_the_process() { + fn a_poisoned_main_thread_claims_its_process_and_retires_the_rest() { let mut world = World::new(); let pid = world.spawn_process(); let main = world.main_tid(pid); - assert_eq!(zombify_poisoned(&mut world, pid, main), PoisonOutcome::Process(pid)); + let t1 = world.spawn_thread(pid); + let owed = zombify_poisoned(&mut world, pid, main); + assert_eq!(owed, Poisoned { retire: alloc::vec![t1], ..Poisoned::default() }); assert!(world.get(pid).unwrap().tearing_down()); } + #[test] + fn a_poisoned_main_thread_alone_publishes_its_exit() { + let mut world = World::new(); + let pid = world.spawn_process(); + let main = world.main_tid(pid); + let owed = zombify_poisoned(&mut world, pid, main); + assert_eq!(owed, Poisoned { exit: Some(TORN_DOWN_THREAD_CODE), ..Poisoned::default() }); + } + /// Not under `mutate-claim-teardown-always-wins`: this asserts the very /// exclusion that control removes, so it would red for the mutation rather /// than for a law, and the step that reads the arm's verdict lines could /// not tell the two apart. #[cfg(not(feature = "mutate-claim-teardown-always-wins"))] #[test] - fn a_process_another_path_already_claimed_is_left_alone() { + fn the_last_poisoned_thread_of_a_claimed_process_publishes_the_claims_code() { let mut world = World::new(); let pid = world.spawn_process(); let main = world.main_tid(pid); - assert!(teardown::claim_teardown(&mut world, pid)); - assert_eq!(zombify_poisoned(&mut world, pid, main), PoisonOutcome::Nothing); + assert!(teardown::claim_teardown(&mut world, pid, 137)); + assert_eq!(zombify_poisoned(&mut world, pid, main), Poisoned { exit: Some(137), ..Poisoned::default() }); } #[test] fn a_reaped_entry_is_nothing_to_do_rather_than_a_panic() { let mut world = World::new(); - assert_eq!(zombify_poisoned(&mut world, Pid(6), Tid(0)), PoisonOutcome::Nothing); + assert_eq!(zombify_poisoned(&mut world, Pid(6), Tid(0)), Poisoned::default()); } #[test] @@ -124,6 +124,6 @@ mod tests { let pid = world.spawn_process(); let t1 = world.spawn_thread(pid); world.forget_thread(pid, t1); - assert_eq!(zombify_poisoned(&mut world, pid, t1), PoisonOutcome::Nothing); + assert_eq!(zombify_poisoned(&mut world, pid, t1), Poisoned::default()); } } diff --git a/toyos-proclife/src/spawn.rs b/toyos-proclife/src/spawn.rs index 6de6a2e217d..6fbb29a042b 100644 --- a/toyos-proclife/src/spawn.rs +++ b/toyos-proclife/src/spawn.rs @@ -11,10 +11,8 @@ //! A teardown claims the process under that same lock //! ([`crate::teardown::claim_teardown`]), collects the tids it is going to //! retire, and gives the lock up to retire them. A thread inserted after that -//! collection is a thread the retire sweep never names — so it is enqueued in -//! the scheduler, its process's exit is published without it, its entry is -//! reaped, and the address space its page tables map is freed while it is still -//! runnable. `interleave::tests::a_published_exit_leaves_no_unretired_thread` +//! collection is a thread the retire sweep never names. +//! `interleave::tests::a_published_exit_leaves_no_unretired_thread` //! is that whole sentence as a host test, and //! `mutate-spawn-skips-the-insert-recheck` is what it reds under. @@ -106,7 +104,7 @@ mod tests { fn a_claimed_process_refuses_at_the_start_in_both_arms() { let mut world = World::new(); let pid = world.spawn_process(); - assert!(teardown::claim_teardown(&mut world, pid)); + assert!(teardown::claim_teardown(&mut world, pid, 137)); assert_eq!(admit_thread_start(&world, pid), Admit::TearingDown); } @@ -115,7 +113,7 @@ mod tests { fn a_claimed_process_refuses_at_the_insert_too() { let mut world = World::new(); let pid = world.spawn_process(); - assert!(teardown::claim_teardown(&mut world, pid)); + assert!(teardown::claim_teardown(&mut world, pid, 137)); assert_eq!(admit_thread_insert(&world, pid), Admit::TearingDown); } @@ -124,7 +122,7 @@ mod tests { fn the_mutation_really_admits_into_a_claimed_teardown() { let mut world = World::new(); let pid = world.spawn_process(); - assert!(teardown::claim_teardown(&mut world, pid)); + assert!(teardown::claim_teardown(&mut world, pid, 137)); assert_eq!( admit_thread_insert(&world, pid), Admit::Yes, diff --git a/toyos-proclife/src/table.rs b/toyos-proclife/src/table.rs index c8de029a8a0..15780b940d5 100644 --- a/toyos-proclife/src/table.rs +++ b/toyos-proclife/src/table.rs @@ -12,7 +12,7 @@ //! this crate's model is a `BTreeMap`: an associated iterator type would put //! both spellings in the trait for no decision's benefit. A caller that needs //! an order sorts what it collected, and the two that do -//! ([`crate::teardown::exit_set`] and [`crate::reap::finished_pids`]) say so. +//! ([`crate::teardown::retire_set`] and [`crate::reap::finished_pids`]) say so. use crate::{Pid, ThreadLocation, Tid}; @@ -23,13 +23,18 @@ pub trait Lifecycle { /// builds, but nothing here relies on that. fn main_tid(&self) -> Tid; + /// The exit code of the teardown some path has claimed, or `None`. + fn teardown_code(&self) -> Option; + /// Whether some path has claimed this process's teardown. - fn tearing_down(&self) -> bool; + fn tearing_down(&self) -> bool { + self.teardown_code().is_some() + } - /// Raise the teardown claim. [`crate::teardown::claim_teardown`] is the one + /// Raise the teardown claim, with the code the exit publishes. [`crate::teardown::claim_teardown`] is the one /// caller — the flag exists to be claimed exactly once, so raising it /// anywhere else is what that function is written to prevent. - fn begin_teardown(&mut self); + fn begin_teardown(&mut self, code: i32); /// Where `tid` is, or `None` for a thread this process does not have. fn location(&self, tid: Tid) -> Option; diff --git a/toyos-proclife/src/teardown.rs b/toyos-proclife/src/teardown.rs index f936c6c1520..4fc22bc7ec7 100644 --- a/toyos-proclife/src/teardown.rs +++ b/toyos-proclife/src/teardown.rs @@ -1,37 +1,33 @@ -//! Who ends a process, which threads they must retire, and what a thread's own -//! exit is. +//! Who ends a process, which threads they retire, and which thread tears it +//! down. //! -//! **Exactly one path publishes exactly one exit.** Three of them can arrive at -//! once — a `SYS_EXIT` on the process's own main thread, a `SYS_PROCESS_KILL` -//! from a holder of a `Process` handle, and the idle loop's sweep of threads -//! that died in panic recovery — and the whole arrangement rests on -//! [`claim_teardown`] answering `true` to one of them. A second publish is an -//! assertion failure in `ProcessObject::publish_exit`, by design: it means two -//! teardowns claimed one process, and a kernel that tolerated it would free one -//! address space twice. +//! **Exactly one path claims a process.** Three of them can arrive at once — a +//! `SYS_EXIT` on one of its threads, a `SYS_PROCESS_KILL` from a holder of a +//! `Process` handle, and the idle loop's sweep of a main thread that died in +//! panic recovery — and [`claim_teardown`] answers `true` to one of them. The +//! claim fixes the exit code, and its winner posts a retire to every thread +//! still in the process ([`retire_set`]) and waits on none of them: a kill's +//! victim may be killing the killer. //! -//! **The claimant retires every other thread before it frees anything.** A -//! thread that is still schedulable when its process's mappings go writes -//! through stale page tables into 2 MiB frames the PMM has already re-issued, -//! so the order — claim, collect, retire, free, mark, publish — is the whole of -//! the teardown's soundness. What this module owns is the *decisions* in that -//! order: which threads are in the set, where each one's CPU time is charged, -//! what code each is marked with. The retire itself is `toyos-sched`'s and the -//! free is the kernel's. +//! **The last thread out tears the process down.** Every thread leaves by its +//! own hand ([`leave`]), and the one whose leaving empties a claimed process +//! frees what the process holds and publishes its exit, on its own stack. +//! Nothing a thread can still run in is freed before then, and no thread waits +//! for another to leave. A second publish is an assertion failure in +//! `ProcessObject::publish_exit`, by design. use alloc::vec::Vec; use crate::table::{Lifecycle, Processes}; use crate::{Pid, ThreadLocation, Tid, Watch, TORN_DOWN_THREAD_CODE}; -/// Claim exclusive teardown of a process. +/// Claim exclusive teardown of a process, for `code`. /// -/// Exactly one exit/kill/poison path wins; a later caller must simply exit its -/// own thread — the claimant's retire sweep handles it like any other thread. -/// `false` also covers a process that is not in the table at all, because there -/// is nothing left for a second claimant to do either way. -#[must_use = "a caller that did not win the claim must not tear anything down"] -pub fn claim_teardown(table: &mut T, pid: Pid) -> bool { +/// Exactly one exit/kill/poison path wins; a later caller's thread simply +/// leaves. `false` also covers a process that is not in the table at all, +/// because there is nothing left for a second claimant to do either way. +#[must_use = "a caller that did not win the claim must not retire anything"] +pub fn claim_teardown(table: &mut T, pid: Pid, code: i32) -> bool { let Some(proc) = table.get_mut(pid) else { return false }; // The mutation this feature stages is the whole of the exclusion: the flag // is still raised and still readable, and every arrival is still told it @@ -40,108 +36,65 @@ pub fn claim_teardown(table: &mut T, pid: Pid) -> bool { if proc.tearing_down() { return false; } - proc.begin_teardown(); + proc.begin_teardown(code); true } -/// The threads a process's own exit must retire, and whether the thread running -/// that exit is the main one. -/// -/// The current thread is **not** in `others`, and cannot be: it is executing -/// the teardown, and `await_released` returns only when its subject is provably -/// off every CPU. Its own CPU time is read separately, which is what -/// [`ExitSet::current_is_main`] is for — a main thread filtered out of the -/// retire set would otherwise leave `cpu=0ms` on its own exit line. -pub struct ExitSet { - /// Every thread of the process except the one calling. Sorted, so the - /// retire order does not depend on a hash seed. - pub others: Vec, - /// Whether the calling thread is this process's main thread. - pub current_is_main: bool, -} - -pub fn exit_set(proc: &P, current: Tid) -> ExitSet { - let mut others = Vec::new(); - proc.each_thread(&mut |tid, _| { - if tid != current { - others.push(tid); +/// The threads a claim's winner retires: every one still in the process but +/// `caller`, which leaves by its own exit. Sorted, so the retire order does not +/// depend on a hash seed. +pub fn retire_set(proc: &P, caller: Option) -> Vec { + let mut tids = Vec::new(); + proc.each_thread(&mut |tid, at| { + if !at.is_zombie() && Some(tid) != caller { + tids.push(tid); } }); - others.sort_unstable(); - ExitSet { others, current_is_main: proc.main_tid() == current } -} - -/// Every thread of a process being killed from outside, in retire order. -/// -/// Unlike [`exit_set`] this includes the main thread and holds no exception: -/// every thread belongs to another process, so none of them is the one running -/// the kill. -pub fn kill_set(proc: &P) -> Vec { - let mut tids = Vec::new(); - proc.each_thread(&mut |tid, _| tids.push(tid)); tids.sort_unstable(); tids } -/// Which owed kill the reaper finishes next, given whether each one's threads -/// are all released, oldest first: the oldest released one, so a victim slow to -/// leave the CPU holds up no other kill. -pub fn next_released(released: impl IntoIterator) -> Option { - released.into_iter().position(|r| r) -} - -/// Where a retired thread's CPU time is charged. -/// -/// The two are not interchangeable: the main thread's is what a process's exit -/// line reports and what `ProcessStats::cpu_ns` is built from, while a -/// sibling's is folded into `child_threads_cpu_ns` and added to it. Charging -/// one as the other double-counts or loses the whole of a process's CPU time. +/// What a thread's leaving makes it. +#[must_use = "the last thread out owes its process's teardown"] #[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum CpuCharge { - /// The process's own `cpu_ns`. - MainThread, - /// `ProcessAccounting::child_threads_cpu_ns`. - ChildThreads, -} - -pub fn charge(tid: Tid, main_tid: Tid) -> CpuCharge { - if tid == main_tid { - CpuCharge::MainThread - } else { - CpuCharge::ChildThreads - } +pub enum Leave { + /// It emptied a claimed process: it tears the process down and publishes + /// `code`. + Last { code: i32 }, + /// Some thread is still in the process, or nobody has claimed it. + NotLast, } -/// Mark one thread dead. +/// Take `tid` out of `pid`: dead with the code it chose, else its process's +/// for the main thread and [`TORN_DOWN_THREAD_CODE`] for any other. /// -/// Idempotent, and silent about an entry that has gone: a main thread reaches -/// this after its own process published its exit, by which point any idle pass -/// may already have reaped the entry. A thread already dead keeps the code it -/// died with — the second mark is a teardown arriving behind the thread's own -/// exit, and the code the thread chose is the true one. -pub fn mark_zombie(table: &mut T, pid: Pid, tid: Tid, code: i32) { - let Some(proc) = table.get_mut(pid) else { return }; - if proc.location(tid).is_some_and(|l| !l.is_zombie()) { - proc.set_location(tid, ThreadLocation::Zombie(code)); - } -} - -/// Mark every thread of a terminating process dead: the main thread with the -/// process's code, every sibling with [`TORN_DOWN_THREAD_CODE`]. -/// -/// Runs under the table lock at the end of a teardown, after every thread has -/// been retired — so nothing it marks can still be running, and a thread that -/// is already a zombie has already answered for itself and keeps its code. -pub fn mark_all_zombie(proc: &mut P, code: i32) { - let main_tid = proc.main_tid(); - let mut pending: Vec<(Tid, i32)> = Vec::new(); - proc.each_thread(&mut |tid, at| { - if !at.is_zombie() { - pending.push((tid, if tid == main_tid { code } else { TORN_DOWN_THREAD_CODE })); +/// Panics for a thread that is not in its process's entry or has left already: +/// an entry outlives every thread that has not left it. +pub fn leave(table: &mut T, pid: Pid, tid: Tid, chosen: Option) -> Leave { + let proc = table.get_mut(pid).expect("leave: a thread's process is not in the table"); + assert_eq!( + proc.location(tid), + Some(ThreadLocation::Scheduled), + "leave: pid {pid} tid {tid} is not a thread still in its process", + ); + let claimed = proc.teardown_code(); + let code = match chosen { + Some(code) => code, + None if tid == proc.main_tid() => { + claimed.expect("leave: a main thread leaves only a process somebody claimed") } - }); - for (tid, code) in pending { - proc.set_location(tid, ThreadLocation::Zombie(code)); + None => TORN_DOWN_THREAD_CODE, + }; + proc.set_location(tid, ThreadLocation::Zombie(code)); + let mut still_in = false; + proc.each_thread(&mut |_, at| still_in |= !at.is_zombie()); + // The mutation this feature stages: every thread that leaves a claimed + // process tears it down, the first one out included. + #[cfg(feature = "mutate-first-out-tears-down")] + let still_in = false; + match claimed { + Some(code) if !still_in => Leave::Last { code }, + _ => Leave::NotLast, } } @@ -149,12 +102,10 @@ pub fn mark_all_zombie(proc: &mut P, code: i32) { #[must_use = "a thread exit that is not routed is a thread that never dies"] #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub enum ThreadExit { - /// The main thread: this is the process's exit, and the whole teardown - /// runs. + /// The main thread: this is the process's exit. Process, - /// A sibling: release its own mappings, mark it `Zombie(code)`, and post - /// on `post` before the exit pass — after it, this thread does not - /// run again. + /// A sibling: release its own mappings, leave, and post on `post` before + /// the exit pass — after it, this thread does not run again. Sibling { /// **The subject a joiner armed on, which is the exiting thread's own /// watch.** It was the process's main thread until `1bfe4e5b`, because @@ -164,22 +115,12 @@ pub enum ThreadExit { /// reach it. post: Watch, }, - /// The process has no entry — another CPU's kill reaped it while this - /// thread was on its way here. **The same exit a sibling takes**, on the - /// same [`Watch`]: nothing here is the main thread any more, so the - /// teardown branch is skipped and every table write is a no-op. - Gone { post: Watch }, } -/// Route a thread's own exit. -/// -/// **A missing entry is [`ThreadExit::Gone`] and not a panic**, which is -/// [`mark_zombie`]'s rule one function along: a thread that arrives to find -/// nothing has to leave, not take the machine with it. +/// Route a thread's own exit. Panics for a process not in the table: an entry +/// outlives every thread that has not left it. pub fn route_thread_exit(table: &T, pid: Pid, tid: Tid) -> ThreadExit { - let Some(proc) = table.get(pid) else { - return ThreadExit::Gone { post: Watch::Thread(pid, tid) }; - }; + let proc = table.get(pid).expect("route_thread_exit: a thread's process is not in the table"); if proc.main_tid() == tid { return ThreadExit::Process; } @@ -196,9 +137,10 @@ mod tests { fn exactly_one_claimant_wins_however_many_arrive() { let mut world = World::new(); let pid = world.spawn_process(); - assert!(claim_teardown(&mut world, pid)); - assert!(!claim_teardown(&mut world, pid)); - assert!(!claim_teardown(&mut world, pid)); + assert!(claim_teardown(&mut world, pid, 0)); + assert!(!claim_teardown(&mut world, pid, 137)); + assert!(!claim_teardown(&mut world, pid, 137)); + assert_eq!(world.get(pid).unwrap().teardown_code(), Some(0), "the winner's code stands"); } /// Teeth for the control itself: under the mutation a second claimant @@ -209,9 +151,9 @@ mod tests { fn the_mutation_really_grants_a_second_claim() { let mut world = World::new(); let pid = world.spawn_process(); - assert!(claim_teardown(&mut world, pid)); + assert!(claim_teardown(&mut world, pid, 0)); assert!( - claim_teardown(&mut world, pid), + claim_teardown(&mut world, pid, 137), "the control is inert: the claim is still exclusive, so whatever the \ model reds on under it is not this mutation", ); @@ -220,79 +162,54 @@ mod tests { #[test] fn a_process_that_is_gone_grants_no_claim() { let mut world = World::new(); - assert!(!claim_teardown(&mut world, Pid(9))); + assert!(!claim_teardown(&mut world, Pid(9), 137)); } #[test] - fn the_exit_set_leaves_out_the_thread_running_the_exit() { + fn the_retire_set_is_every_thread_still_in_but_the_caller() { let mut world = World::new(); let pid = world.spawn_process(); let main = world.main_tid(pid); let t1 = world.spawn_thread(pid); let t2 = world.spawn_thread(pid); + let t3 = world.spawn_thread(pid); + world.set_location(pid, t3, ThreadLocation::Zombie(0)); - let from_main = exit_set(world.get(pid).unwrap(), main); - assert_eq!(from_main.others, [t1, t2]); - assert!(from_main.current_is_main); - - let from_sibling = exit_set(world.get(pid).unwrap(), t1); - assert_eq!(from_sibling.others, [main, t2]); - assert!(!from_sibling.current_is_main); + let proc = world.get(pid).unwrap(); + assert_eq!(retire_set(proc, None), [main, t1, t2]); + assert_eq!(retire_set(proc, Some(main)), [t1, t2]); + assert_eq!(retire_set(proc, Some(t1)), [main, t2]); } + #[cfg(not(feature = "mutate-first-out-tears-down"))] #[test] - fn a_kill_set_holds_every_thread_including_the_main_one() { + fn only_the_thread_that_empties_a_claimed_process_tears_it_down() { let mut world = World::new(); let pid = world.spawn_process(); let main = world.main_tid(pid); let t1 = world.spawn_thread(pid); - assert_eq!(kill_set(world.get(pid).unwrap()), [main, t1]); - } + let t2 = world.spawn_thread(pid); - #[test] - fn cpu_time_is_charged_to_the_main_thread_only_for_the_main_thread() { - assert_eq!(charge(Tid(0), Tid(0)), CpuCharge::MainThread); - assert_eq!(charge(Tid(1), Tid(0)), CpuCharge::ChildThreads); - // A process whose main thread is not tid 0 — nothing in this crate - // assumes the loader's numbering. - assert_eq!(charge(Tid(3), Tid(3)), CpuCharge::MainThread); - assert_eq!(charge(Tid(0), Tid(3)), CpuCharge::ChildThreads); - } + // Unclaimed: a sibling's own exit, with its own code. + assert_eq!(leave(&mut world, pid, t2, Some(5)), Leave::NotLast); + assert!(claim_teardown(&mut world, pid, 42)); + assert_eq!(leave(&mut world, pid, main, None), Leave::NotLast); + assert_eq!(leave(&mut world, pid, t1, None), Leave::Last { code: 42 }); - #[test] - fn a_second_mark_never_moves_a_code_a_thread_chose() { - let mut world = World::new(); - let pid = world.spawn_process(); - let t1 = world.spawn_thread(pid); - mark_zombie(&mut world, pid, t1, 7); - mark_zombie(&mut world, pid, t1, TORN_DOWN_THREAD_CODE); - assert_eq!(world.get(pid).unwrap().location(t1), Some(ThreadLocation::Zombie(7))); - } - - #[test] - fn marking_a_reaped_process_is_silent() { - let mut world = World::new(); - mark_zombie(&mut world, Pid(4), Tid(0), 0); + let proc = world.get(pid).unwrap(); + assert_eq!(proc.location(main), Some(ThreadLocation::Zombie(42))); + assert_eq!(proc.location(t1), Some(ThreadLocation::Zombie(TORN_DOWN_THREAD_CODE))); + assert_eq!(proc.location(t2), Some(ThreadLocation::Zombie(5))); } #[test] - fn the_teardown_sweep_gives_the_main_thread_the_code_and_the_rest_minus_one() { + #[should_panic(expected = "is not a thread still in its process")] + fn a_thread_leaves_once() { let mut world = World::new(); let pid = world.spawn_process(); - let main = world.main_tid(pid); let t1 = world.spawn_thread(pid); - let t2 = world.spawn_thread(pid); - mark_zombie(&mut world, pid, t2, 5); - - mark_all_zombie(world.get_mut(pid).unwrap(), 42); - let proc = world.get(pid).unwrap(); - assert_eq!(proc.location(main), Some(ThreadLocation::Zombie(42))); - assert_eq!(proc.location(t1), Some(ThreadLocation::Zombie(TORN_DOWN_THREAD_CODE))); - assert_eq!( - proc.location(t2), - Some(ThreadLocation::Zombie(5)), - "a thread that had already answered for itself kept its own code", - ); + let _ = leave(&mut world, pid, t1, Some(0)); + let _ = leave(&mut world, pid, t1, Some(0)); } #[test] @@ -303,15 +220,6 @@ mod tests { assert_eq!(route_thread_exit(&world, pid, main), ThreadExit::Process); } - #[test] - fn an_exit_on_a_reaped_process_leaves_by_the_sibling_door() { - let world = World::new(); - assert_eq!( - route_thread_exit(&world, Pid(3), Tid(1)), - ThreadExit::Gone { post: Watch::Thread(Pid(3), Tid(1)) }, - ); - } - /// Two siblings, and the one that is waiting is not the main thread — the /// exact shape the wake-by-name lost, when `thread_exit` posted one wake /// and it was always `TaskId(pid, proc.main_tid)`. diff --git a/toyos-quiesce/src/lib.rs b/toyos-quiesce/src/lib.rs index cd730f173d6..6397d07b861 100644 --- a/toyos-quiesce/src/lib.rs +++ b/toyos-quiesce/src/lib.rs @@ -38,6 +38,7 @@ pub fn must_stop(thread: ThreadId, caller: ThreadId) -> bool { thread != caller } +/// What one sweep of the machine's userland threads found. #[derive(Clone, Copy, PartialEq, Eq, Debug, Default)] pub struct Sweep { /// Threads that must stop and have: banded at a safe point, or parked and @@ -76,7 +77,7 @@ pub const LAST_THREAD: &str = "quiesce-last"; pub const STOPPED: &str = "stop: "; const OF: &str = " of "; -const THREADS: &str = " thread(s) stopped across "; +const THREADS: &str = " userland thread(s) stopped across "; const CPUS: &str = " cpu(s) in "; const BUDGET: &str = " ms of a "; const MS: &str = " ms budget over "; @@ -248,12 +249,12 @@ mod tests { fn the_record_names_the_shortfall_only_when_there_is_one() { assert_eq!( alloc::format!("{WHOLE}"), - "stop: 6 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 ms \ + "stop: 6 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 ms \ budget over 3 sweep(s), 0 of 4812 userland block operation(s) still open", ); assert_eq!( alloc::format!("{}", short()), - "stop: 4 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 ms \ + "stop: 4 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 ms \ budget over 3 sweep(s), 1 of 4812 userland block operation(s) still open; this \ reset lands wherever the other 2 are", ); @@ -287,17 +288,17 @@ mod tests { fn a_line_that_is_not_a_record_is_refused_rather_than_half_read() { for line in [ "[stamp] Rebooting.", - "[stamp] stop: 6 of 6 thread(s) stopped across 8 cpu(s)", + "[stamp] stop: 6 of 6 userland thread(s) stopped across 8 cpu(s)", // The shortfall clause disagreeing with the counts it restates. - "[stamp] stop: 4 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ + "[stamp] stop: 4 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ ms budget over 3 sweep(s), 1 of 4812 userland block operation(s) still open; this \ reset lands wherever the other 9 are", // A shortfall clause on a record that claims to have stopped. - "[stamp] stop: 6 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ + "[stamp] stop: 6 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ ms budget over 3 sweep(s), 0 of 4812 userland block operation(s) still open; this \ reset lands wherever the other 2 are", // More threads stopped than there were. - "[stamp] stop: 7 of 6 thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ + "[stamp] stop: 7 of 6 userland thread(s) stopped across 8 cpu(s) in 11 ms of a 2010 \ ms budget over 3 sweep(s), 0 of 4812 userland block operation(s) still open", ] { assert_eq!(Record::parse(line), None, "{line}"); diff --git a/toyos-sched/sim/src/explore.rs b/toyos-sched/sim/src/explore.rs index 1d90dbc97a1..d8e5f6a9a06 100644 --- a/toyos-sched/sim/src/explore.rs +++ b/toyos-sched/sim/src/explore.rs @@ -48,10 +48,7 @@ pub struct Outcome { /// registration window. Reported for the same reason. pub killed_at_park: u64, /// Invariant I14's measurement: the longest a retire went unfinalized, and - /// the bound in force. A number as well as a verdict, because the kernel's - /// `await_released` states the same property with a wall clock and a panic, and - /// how much of that budget the protocol spends is what says whether the wall - /// clock is a backstop or a coin flip. + /// the bound in force. pub retire_latency: u64, pub retire_bound: u64, /// The longest one CPU's unwind gate held every other CPU still — see diff --git a/toyos-sched/sim/src/invariants.rs b/toyos-sched/sim/src/invariants.rs index 63686abef81..f0ae80c4810 100644 --- a/toyos-sched/sim/src/invariants.rs +++ b/toyos-sched/sim/src/invariants.rs @@ -35,9 +35,7 @@ use crate::vm::{FairEpoch, Vm, IPI_LATENCY_NS, RUN_CHUNK_NS, UNWIND_NS}; /// been the whole unwind, once per killed task, turning this /// bound into a statement about how long a kernel teardown takes. The second /// attempt served it strictly *after* `rq` and had no term at all — at the price -/// of a corpse that never runs under a saturated RT band, which is -/// `scheduler::await_released`'s tripwire and a kernel panic from a legal -/// `Rights::RT` workload. +/// of a corpse that never runs under a saturated RT band. /// /// What ships is neither absolute. `CpuSched::pick` takes the dying list ahead /// of the RT band only once its head has waited @@ -65,9 +63,8 @@ fn rt_latency_bound(max_kernel_section: u64) -> u64 { } /// How long a retire may take to reach `Hw::release` (invariant I14), measured -/// on the **wall clock** — the one `scheduler::await_released`'s own tripwire -/// reads, and see [`crate::vm::Killed`] for why there is no second clock any -/// more. +/// on the **wall clock** — see [`crate::vm::Killed`] for why there is no +/// second clock any more. /// /// Hop by hop, from the claim to the call: /// @@ -86,9 +83,7 @@ fn rt_latency_bound(max_kernel_section: u64) -> u64 { /// CPU's *zombie*, because a pass cannot free the stack it is standing on; /// the payload is released by the **next** pass on that CPU /// (`SchedPass::begin`), and if that CPU dispatched another task its next -/// pass is that task's quantum expiry. `await_released`'s own doc states the -/// same hop from the other side, and the wait is for the release, not for -/// the word. +/// pass is that task's quantum expiry. /// /// plus `2 × RUN_CHUNK_NS`, invariant I4's granularity term: the model observes /// each hop up to one execution chunk late. @@ -117,11 +112,6 @@ fn retire_latency_bound(max_kernel_section: u64, peers: usize) -> u64 { /// It is charged unconditionally because a bound is a worst case, and it is a /// *finite* factor rather than the unbounded term the previous form of this /// derivation declined to price at all. -/// -/// The same factor is a term of `scheduler::RETIRE_GIVE_UP` -/// derivation, and `toyos_sched`'s own -/// `an_unwind_under_saturated_rt_is_stretched_by_the_age_ratio` is what stops -/// the two drifting apart. fn rt_deferral_stretch() -> u64 { (DYING_AGE_NS + DYING_CHUNK_NS) / DYING_CHUNK_NS } @@ -299,8 +289,7 @@ fn check_sleeping_cpus(vm: &mut Vm<'_>) { /// adopt carries `Urgency::Normal`, which by design sends no IPI to a busy CPU /// — so a CPU that hands on a task it knows is dead trades an unwind it could /// start in this pass for a wait on another CPU's next voluntary one. **And a -/// retire completes within [`retire_latency_bound`]**, which is the statement -/// the kernel's `await_released` makes with a wall clock and a panic. +/// retire completes within [`retire_latency_bound`]**. /// /// **Both halves survive the cancellable kill and only one of them /// changed.** It makes a killed task *run* rather than be reaped where it @@ -381,7 +370,7 @@ fn check_retires(vm: &mut Vm<'_>) { if elapsed > bound { problems.push(format!( "I14: {key:?} was retired {elapsed} ns ago and is still {:?} \ - (bound {bound} ns, on the wall clock `await_released` reads)", + (bound {bound} ns, on the wall clock)", vm.shared[&key].state(), )); } @@ -408,8 +397,7 @@ fn owner_of(state: TaskState) -> Option { } } -/// How long this retire has been outstanding, on the clock -/// `scheduler::await_released`'s own tripwire reads: the wall clock, every CPU's +/// How long this retire has been outstanding: the wall clock, every CPU's /// and no CPU's, with nothing subtracted from it. /// /// [`crate::vm::Killed`] carries why there is no second clock any more. diff --git a/toyos-sched/sim/src/vm.rs b/toyos-sched/sim/src/vm.rs index 2b6fd594e45..f57f056b105 100644 --- a/toyos-sched/sim/src/vm.rs +++ b/toyos-sched/sim/src/vm.rs @@ -151,19 +151,12 @@ pub struct ProcState { /// the retire-to-release interval therefore contains an unbounded quantity — so /// I14 should be read on a clock with the RT band's service subtracted out. /// -/// Every step of that was true and the conclusion was a blindfold. The kernel -/// does not wait on that clock: `scheduler::await_released` blocks behind a -/// **wall-clock** tripwire and panics when it expires. A model measuring the -/// same wait on a clock the kernel cannot read is a model that cannot see the -/// panic — and the unbounded quantity the paragraph named is exactly the defect -/// that panic was reachable through, declared as a modelling convenience one -/// file away from the invariant that would have caught it. +/// Every step of that was true and the conclusion was a blindfold. /// /// So the quantity is bounded at its source instead: `CpuSched::pick`'s /// [`toyos_sched::cpu::DYING_AGE_NS`] makes the RT band's precedence over a -/// corpse a bounded deferral, and I14 is read on the clock the kernel's own -/// guard reads. The RT service the victim's CPU owed is *in* the number, which -/// is the only way the number means anything. +/// corpse a bounded deferral. The RT service the victim's CPU owed is *in* the +/// number, which is the only way the number means anything. /// /// (I5 still stops measuring service while the RT band is occupied. That /// exclusion is about *fairness*, which the RT band exists to be unfair to; diff --git a/toyos-sched/sim/src/workload.rs b/toyos-sched/sim/src/workload.rs index 7d087294384..43f1875fa23 100644 --- a/toyos-sched/sim/src/workload.rs +++ b/toyos-sched/sim/src/workload.rs @@ -187,8 +187,7 @@ pub enum AgeShape { BoundedDeferral, /// The shape this branch shipped between the two fixes: `pick` asks only /// `rq.has_rt()`, so a permanently-RT thread that never parks holds the - /// dying list closed for ever and `scheduler::await_released`'s tripwire - /// panics the kernel. See `scenarios::old_rt_starved_the_corpse`. + /// dying list closed for ever. See `scenarios::old_rt_starved_the_corpse`. RtOutranksEveryCorpse, } diff --git a/toyos-sched/sim/tests/scenarios.rs b/toyos-sched/sim/tests/scenarios.rs index 552a1212d29..137aa3e71b5 100644 --- a/toyos-sched/sim/tests/scenarios.rs +++ b/toyos-sched/sim/tests/scenarios.rs @@ -275,10 +275,7 @@ fn old_migrate_keeping_the_corpse_is_caught() { /// never dispatched at all, because the pick asked only `rq.has_rt()` and one /// permanently-RT thread that never parks answered yes for ever. /// -/// That shape shipped on this branch between the two fixes, and its failure is -/// not a slow retire: `scheduler::await_released` blocks behind a wall-clock -/// tripwire and **panics the kernel**, from a workload that only needs -/// `Rights::RT` — which `soundd` holds and `SYS_RT_ENTER` never gives back. +/// That shape shipped on this branch between the two fixes. /// /// It must be caught by **I14**, on every seed: nothing in this scenario is a /// race. The corpse is queued, the RT thread runs, and the only question is @@ -320,12 +317,6 @@ fn rt_starving_the_corpse_is_caught() { /// The positive half of the same pair: with the kill bit read, no schedule of /// that workload puts a corpse in transit, and every retire completes well /// inside the derived bound. -/// -/// The measurement is the point. `await_released`'s guard is a wall clock two -/// orders of magnitude wider than [`toyos_sched_sim::explore::Outcome::retire_bound`], -/// so what this reports is how much of that budget the protocol actually spends -/// — and a change that starts spending it shows up here as a number long before -/// it shows up on the owner's laptop as a panic. #[test] fn a_retire_completes_inside_its_derived_bound() { let scenario = scenarios::retire_under_balance(); diff --git a/toyos-sched/src/cpu.rs b/toyos-sched/src/cpu.rs index 3e752b664d8..22fd3976053 100644 --- a/toyos-sched/src/cpu.rs +++ b/toyos-sched/src/cpu.rs @@ -622,9 +622,7 @@ impl CpuSched { /// does not exist. /// /// One permanently-RT thread that never parks then holds a CPU's dying list - /// closed for ever, no sibling can rescue the corpse, and - /// `scheduler::await_released`'s wall-clock tripwire panics the kernel from a - /// legal `Rights::RT` workload. Invariant I14 must catch it; + /// closed for ever, and no sibling can rescue the corpse. Invariant I14 must catch it; /// `scenarios::old_rt_starved_the_corpse` is the gate that proves it does, /// and it is the *other* direction of the pair /// `scenarios::old_migrate_kept_the_corpse` opens. @@ -1299,8 +1297,7 @@ pub const MAX_PASS_NS: u64 = 200_000; /// under one permanently-RT thread that never parks — `Rights::RT` is /// capability-gated but `soundd` holds it and `SYS_RT_ENTER` has no /// revocation, so a killed thread of an RT process on that CPU never reaches -/// `Hw::release`, and `scheduler::await_released`'s tripwire panics the kernel -/// from a legal workload. That is the kernel crashing from userland. +/// `Hw::release`. /// /// So the corpse ages. Once its head has stood in the dying list for this long, /// the next pick takes it ahead of the RT band for one [`DYING_CHUNK_NS`], then @@ -1324,8 +1321,7 @@ pub const DYING_AGE_NS: u64 = QUANTUM_NS; /// **This is the number invariant I4's bound grows by**, so it is as small as /// the other side can afford. Under saturated RT an unwind is delivered at one /// chunk per `DYING_AGE_NS + DYING_CHUNK_NS`, so the corpse's release is -/// stretched by 11× — which is a term of `scheduler::RETIRE_GIVE_UP` -/// derivation. A larger chunk buys that term +/// stretched by 11×. A larger chunk buys that /// back and spends it on RT latency; `soundd` is the process that pays, and 1 ms /// of added worst-case jitter once per 10 ms is the trade this picks. pub const DYING_CHUNK_NS: u64 = QUANTUM_NS / 10; @@ -1854,14 +1850,12 @@ impl SchedPass<'_, '_, H, P, Disposed> { /// scheduler states as law — a ready real-time task preempts the normal /// band — outright. /// - /// Asking only `rq.has_rt()` is the other absolute and it is worse, because - /// its failure is a kernel panic: one permanently-RT thread that never parks - /// holds this CPU's dying list closed for ever, no sibling CPU can rescue a - /// corpse (`hand_off` refuses to migrate a killed task, `pop_surplus` reads - /// `fair` only), and `scheduler::await_released`'s tripwire fires. That is + /// Asking only `rq.has_rt()` is the other absolute and it is worse: one + /// permanently-RT thread that never parks holds this CPU's dying list closed + /// for ever, and no sibling CPU can rescue a corpse (`hand_off` refuses to + /// migrate a killed task, `pop_surplus` reads `fair` only). That is /// reachable from a legal `Rights::RT` workload — `soundd` holds the right - /// and `SYS_RT_ENTER` has no revocation — so it is the kernel crashing from - /// userland. + /// and `SYS_RT_ENTER` has no revocation. /// /// So the question asked here is `rq.has_rt()` **unless the head of the /// dying list has waited [`DYING_AGE_NS`]**, and an aged corpse is @@ -3409,11 +3403,9 @@ mod tests { /// **The other direction of the same law.** The three tests above say a /// corpse never starves the RT band. - /// This one says the RT band never starves the corpse, because unqualified - /// RT precedence over the dying list ends in a kernel panic: - /// `scheduler::await_released` blocks on `Hw::release` behind a tripwire, and - /// one permanently-RT thread that never parks holds this CPU's dying list - /// closed for ever. `hand_off` refuses to migrate a killed task and + /// This one says the RT band never starves the corpse: under unqualified RT + /// precedence over the dying list, one permanently-RT thread that never + /// parks holds this CPU's dying list closed for ever. `hand_off` refuses to migrate a killed task and /// `pop_surplus` reads `fair` only, so no sibling CPU can rescue it. /// /// The workload is legal: `Rights::RT` is capability-gated, `soundd` holds @@ -3520,11 +3512,6 @@ mod tests { /// saturated RT band the corpse gets `DYING_CHUNK_NS` out of every /// `DYING_AGE_NS + DYING_CHUNK_NS`, so an unwind's wall-clock length is /// stretched by that factor and no more. - /// - /// It is a separate test because it is the term - /// `scheduler::RETIRE_GIVE_UP` derivation carries, and a change - /// that quietly widened `DYING_AGE_NS` would leave the gate above green - /// while making that tripwire wrong. #[test] // Deliberately asserts on constants: the test exists to state the // relations between them, with messages a `const` block's bare assert @@ -3534,8 +3521,7 @@ mod tests { let stretch = (DYING_AGE_NS + DYING_CHUNK_NS) / DYING_CHUNK_NS; assert_eq!( stretch, 11, - "`RETIRE_GIVE_UP` prices the unwind at {stretch}x its own CPU \ - time; the constants say {}", + "the unwind is stretched {stretch}x its own CPU time; the constants say {}", DYING_AGE_NS + DYING_CHUNK_NS, ); assert!( diff --git a/toyos-sched/src/park.rs b/toyos-sched/src/park.rs index 882b70e7288..c8f28976609 100644 --- a/toyos-sched/src/park.rs +++ b/toyos-sched/src/park.rs @@ -155,8 +155,7 @@ pub enum Cancel { /// it unwinds. Answers, /// The park a kill may not end. The caller cannot propagate a cancel and - /// is bounded by something else — the retirer waiting for its victim's - /// release is the one. + /// is bounded by something else. Ignores, } From a6aaf3e4afa41dc6750a717753c0aa41dbac718d Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 08:16:33 +0200 Subject: [PATCH 07/19] kernel: the last thread out is in its process until its teardown is done The machine's stop counts a zombie as nothing left to stop. The last thread out was marked a zombie by `proclife::leave`, and `note_progress` posted, before its teardown ran; a stop taken then returned while that teardown's `close_all` still dropped files, and the shutdown's `drain_all` and `sync_all` ran beside it. A dropped file's write-back was queued after the drain, and its dirty pages were not durable at power-off. `proclife::leave` now marks every thread but the last one out, and answers `Last { code, mark }`; `proclife::torn_down` marks that one in `teardown_bookkeeping`, after its records and releases, and `note_progress` posts after that mark. The model scripts the mark as its own step, and its L3 now reads: a teardown in flight keeps its thread in the process. L3's old sentence, a stack freed only after the switch, checked the model's own step order; toyos-sched-sim's I11 is that check, and the model's copy and its hand-run teeth test go. `mutate-last-out-leaves-before-its-teardown` marks the last one out before its teardown, as the head before this did; it reds `the_last_one_out_is_in_its_process_until_its_teardown_is_done` on L3 and `only_the_thread_that_empties_a_claimed_process_tears_it_down` on its location assertion. `quiesce-last-teardown` holds the last thread out of a child of `quiesce_last` between its leaving and its teardown until the stop counts it alone; `quiesce_wakes_on_the_last_teardown` judges that boot. The exit boundary calls `process::leave` at its own preempt depth: nothing the teardown runs parks, and a park there asserts. `current_data` and `process_data` panic on a missing entry rather than ending the thread without leaving: an entry outlives every thread that has not left it. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- kernel/src/actuator.rs | 3 + kernel/src/process.rs | 61 +++++++------ kernel/src/quiesce.rs | 19 ++-- kernel/src/scheduler.rs | 3 - src/ci.rs | 4 + tests/common/power.rs | 15 +++- .../toyos-rust-tests/src/bin/quiesce_last.rs | 28 ++++-- tests/toyos.rs | 6 +- toyos-proclife/Cargo.toml | 5 ++ toyos-proclife/src/interleave.rs | 12 +++ toyos-proclife/src/model.rs | 90 +++++++------------ toyos-proclife/src/teardown.rs | 52 ++++++++--- 12 files changed, 179 insertions(+), 119 deletions(-) diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index b09612bcbcd..139d651b80f 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -122,6 +122,9 @@ actuators! { /// The same inside `SYS_THREAD_EXIT`: its exit is then the stop's last transition. quiesce_last_exit = "quiesce-last-exit"; + /// The same for the last thread out of its process, between its leaving and its teardown: that teardown is then the stop's last transition. + quiesce_last_teardown = "quiesce-last-teardown"; + /// Serve a blocked-task dump from the shutdown once its first stage has stopped the machine: the report Ctrl+Alt+D gives on a shutdown stuck in its stop. quiesce_dump = "quiesce-dump"; diff --git a/kernel/src/process.rs b/kernel/src/process.rs index b40f4d9b0db..a2055a5168c 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -767,17 +767,14 @@ pub fn current_address_space() -> PageTables { } -/// Get the current thread's ThreadData Arc (brief table lock); exits silently if the entry is gone. +/// Get the current thread's ThreadData Arc (brief table lock). pub fn current_data() -> Arc> { let guard = PROCESS_TABLE.lock(); - let table = guard.as_ref().unwrap(); - match table.get(current_process()).and_then(|p| p.threads.get(current_tid())) { - Some(thread) => Arc::clone(&thread.thread_data), - None => { - drop(guard); - scheduler::exit_current(); - } - } + let thread = guard.as_ref().unwrap() + .get(current_process()) + .and_then(|p| p.threads.get(current_tid())) + .expect("current_data: an entry outlives every thread that has not left it"); + Arc::clone(&thread.thread_data) } /// Set the name of the currently running thread. @@ -794,14 +791,10 @@ pub fn set_current_thread_name(name: &[u8]) { /// Get the process-level ProcessData Arc (brief table lock); shared by every thread of the process. pub fn process_data() -> Arc> { let guard = PROCESS_TABLE.lock(); - let table = guard.as_ref().unwrap(); - match table.get(current_process()) { - Some(proc) => Arc::clone(&proc.process_data), - None => { - drop(guard); - scheduler::exit_current(); - } - } + let proc = guard.as_ref().unwrap() + .get(current_process()) + .expect("process_data: an entry outlives every thread that has not left it"); + Arc::clone(&proc.process_data) } /// Access the current thread's ThreadData mutably; the table lock is not held during the closure. @@ -979,10 +972,10 @@ fn teardown_resources( (syscall_total, syscall_total_ns) } -/// Table-side teardown bookkeeping: drop the symbol table and total the CPU time of every thread still in the table. -/// Caller must hold `PROCESS_TABLE`, be the last thread out and have freed the resources. Returns the object whose exit the caller publishes once the table lock is given up, with that total. +/// Table-side teardown bookkeeping: drop the symbol table, total the CPU time of every thread still in the table, and mark the last thread out dead with `mark`. +/// Caller must hold `PROCESS_TABLE`, be that thread and have freed the resources. Returns the object whose exit the caller publishes once the table lock is given up, with that total. #[must_use = "the exit must be published on the object returned"] -fn teardown_bookkeeping(table: &mut ProcessTable, process_pid: Pid, code: i32) +fn teardown_bookkeeping(table: &mut ProcessTable, process_pid: Pid, tid: Tid, code: i32, mark: i32) -> (Arc, u64) { let proc = table.get_mut(process_pid) .expect("teardown_bookkeeping: process not found"); @@ -990,7 +983,9 @@ fn teardown_bookkeeping(table: &mut ProcessTable, process_pid: Pid, code: i32) let cpu_ns: u64 = proc.threads.iter().map(|(_, t)| t.sched().map_or(0, scheduler::task_cpu_ns)).sum(); let name = proc.name_str(); log!("exit: {name} pid={process_pid} code={code} cpu={}ms", cpu_ns / 1_000_000); - (Arc::clone(&proc.object), cpu_ns) + let object = Arc::clone(&proc.object); + proclife::torn_down(table, process_pid, tid, mark); + (object, cpu_ns) } /// One `ProcessStats`, from a process's own data; written once, since `SYS_PROCESS_STATS` samples a live process through the same fields the teardown snapshots. @@ -1025,19 +1020,23 @@ pub fn stats_from( } } -/// The last thread out's teardown of its process, on its own stack: free what the process holds, then publish its exit. Every thread has left, so none of this process can run in what is freed. +/// The last thread out's teardown of its process, on its own stack: free what the process holds, then publish its exit. Every other thread has left, so none can run in what is freed. +/// The thread is in the process until the bookkeeping marks it, so a machine stop waits for every record and release here. +/// Waits on nothing: it runs on a killed thread whose one cancel may be spent, and at the exit boundary's preempt depth, where a park asserts. /// Publish happens after the table lock is released — `teardown_bookkeeping`'s wake needs that — and once published the entry is reapable, so nothing may read the table for this pid after. -fn teardown(pid: Pid, code: i32, process_data: &Arc>) { +fn teardown(pid: Pid, tid: Tid, code: i32, mark: i32, process_data: &Arc>) { let main_thread_data = { let guard = PROCESS_TABLE.lock(); - let proc = guard.as_ref().unwrap().get(pid).expect("teardown: the process its last thread just left"); + let proc = guard.as_ref().unwrap().get(pid).expect("teardown: the process its last thread is in"); Arc::clone(&proc.threads.get(proc.main_tid).expect("teardown: a claimed process gives up no thread").thread_data) }; let (syscall_total, syscall_total_ns) = teardown_resources(process_data, &main_thread_data, pid); let (object, cpu_ns) = { let mut guard = PROCESS_TABLE.lock(); - teardown_bookkeeping(guard.as_mut().unwrap(), pid, code) + teardown_bookkeeping(guard.as_mut().unwrap(), pid, tid, code, mark) }; + // The table says zombie now, which a sweep counts as nothing left to stop. + crate::quiesce::note_progress(); let stats = stats_from(&process_data.lock(), pid, cpu_ns, syscall_total, syscall_total_ns); object.publish_exit(crate::object::process::Exit { code, stats }); } @@ -1053,10 +1052,14 @@ pub fn leave(chosen: Option) { let mut guard = PROCESS_TABLE.lock(); proclife::leave(guard.as_mut().unwrap(), pid, tid, chosen) }; - // The table says zombie now, which a sweep counts as nothing left to stop. - crate::quiesce::note_progress(); - if let proclife::Leave::Last { code } = out { - teardown(pid, code, &process_data); + match out { + // The table says zombie now, which a sweep counts as nothing left to stop. + proclife::Leave::NotLast => crate::quiesce::note_progress(), + proclife::Leave::Last { code, mark } => { + #[cfg(feature = "boot-actuators")] + crate::quiesce::last::hold(crate::quiesce::last::Last::Teardown); + teardown(pid, tid, code, mark, &process_data); + } } } diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 42a598524b7..1ca68d0c6de 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -221,8 +221,6 @@ fn sweep(caller: ThreadId) -> Sweep { if !must_stop(ThreadId { pid: pid.raw(), tid: tid.raw() }, caller) { continue; } - // A zombie has already written its `exit:` record and holds no - // task; there is nothing left of it to stop. if matches!(thread.state(), process::ThreadLocation::Zombie(_)) { continue; } @@ -245,11 +243,12 @@ fn sweep(caller: ThreadId) -> Sweep { out } -/// `quiesce-last-park` and `quiesce-last-exit`: one thread, named -/// [`toyos_quiesce::LAST_THREAD`], held inside its syscall until the stop's -/// latest sweep counts it as the one thread still running, so the park or the -/// exit it makes next is the last transition the stop sees. Without them no -/// boot can tell whether that transition's post is what wakes the stop. +/// `quiesce-last-park`, `quiesce-last-exit` and `quiesce-last-teardown`: one +/// thread, named [`toyos_quiesce::LAST_THREAD`], held inside its syscall until +/// the stop's latest sweep counts it as the one thread still running, so the +/// park, the exit or the process teardown it makes next is the last transition +/// the stop sees. Without them no boot can tell whether that transition's post +/// is what wakes the stop. #[cfg(feature = "boot-actuators")] pub mod last { use core::sync::atomic::{ @@ -267,6 +266,8 @@ pub mod last { pub enum Last { Park, Exit, + /// The last thread out of its process, between its leaving and its teardown. + Teardown, } impl Last { @@ -274,6 +275,7 @@ pub mod last { match self { Last::Park => crate::actuator::quiesce_last_park(), Last::Exit => crate::actuator::quiesce_last_exit(), + Last::Teardown => crate::actuator::quiesce_last_teardown(), } } @@ -281,6 +283,7 @@ pub mod last { match self { Last::Park => "quiesce-last-park", Last::Exit => "quiesce-last-exit", + Last::Teardown => "quiesce-last-teardown", } } } @@ -305,7 +308,7 @@ pub mod last { } fn armed() -> Option { - [Last::Park, Last::Exit].into_iter().find(|last| last.armed()) + [Last::Park, Last::Exit, Last::Teardown].into_iter().find(|last| last.armed()) } /// Hold the running thread here if it is the one `last` stages. diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index 0e05667f83c..6bfec8571be 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -380,10 +380,7 @@ pub fn leave_ring3_if_due() { unreachable!("leave_ring3_if_due: a stopped task was dispatched again"); } SafePoint::Exit => { - // A syscall's own depth: the last thread out tears its process down here, and that parks. - crate::preempt::disable(); process::leave(None); - crate::preempt::enable_no_resched(); driver::pass(Dispose::Exit); unreachable!("leave_ring3_if_due: returned from the exit pass"); } diff --git a/src/ci.rs b/src/ci.rs index e3a97c5ddc3..960d63d8ccc 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -360,6 +360,10 @@ pub(crate) const CONTROLS: &[Control] = &[ red(PROCLIFE, "mutate-join-collects-in-a-teardown", None, &[ "a_join_racing_the_kill_that_takes_its_target ... FAILED", ]), + red(PROCLIFE, "mutate-last-out-leaves-before-its-teardown", None, &[ + "the_last_one_out_is_in_its_process_until_its_teardown_is_done ... FAILED", + "only_the_thread_that_empties_a_claimed_process_tears_it_down ... FAILED", + ]), red(SCHED_SIM, "placement-ignores-staleness", Some("policy"), &[ "a_stopped_cpu_stops_taking_work ... FAILED", ]), diff --git a/tests/common/power.rs b/tests/common/power.rs index 9b481909542..9ee880f78fb 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -347,7 +347,20 @@ pub fn quiesce_wakes_on_the_last_exit( woken_by_the_held_thread(&["quiesce-last-exit", LATE_WORD], rust_bins) } -/// One of the two `quiesce-last-*` boots: its actuator first, the late word beside it. +/// **A process teardown that is the stop's last transition is waited for.** +/// The same, with `quiesce-last-teardown` holding the last thread out of a +/// process between its leaving and its teardown: a stop that counted it as +/// gone would return at its first sweep while that teardown still frees and +/// logs. +pub fn quiesce_wakes_on_the_last_teardown( + _test_config: &Path, + _c_bins: &[(String, Vec)], + rust_bins: &[(String, Vec)], +) -> Result<(), String> { + woken_by_the_held_thread(&["quiesce-last-teardown", LATE_WORD], rust_bins) +} + +/// One of the `quiesce-last-*` boots: its actuator first, the late word beside it. fn woken_by_the_held_thread( armed: &'static [&'static str; 2], rust_bins: &[(String, Vec)], diff --git a/tests/toyos-rust-tests/src/bin/quiesce_last.rs b/tests/toyos-rust-tests/src/bin/quiesce_last.rs index 1f240479ee6..b5bccd67d88 100644 --- a/tests/toyos-rust-tests/src/bin/quiesce_last.rs +++ b/tests/toyos-rust-tests/src/bin/quiesce_last.rs @@ -1,32 +1,42 @@ -//! Two threads the kernel's `quiesce-last-*` actuators can hold, and a reboot. +//! Three threads the kernel's `quiesce-last-*` actuators can hold, and a reboot. //! -//! Both carry [`toyos_quiesce::LAST_THREAD`]'s name. One parks for longer than -//! any stop's budget and the other exits at once. `quiesce-last-park` holds -//! the first inside its `SYS_NANOSLEEP`, and `quiesce-last-exit` holds the -//! second inside its `SYS_THREAD_EXIT`. The kernel holds the reset itself until -//! one of them is held, so this program orders nothing. +//! All carry [`toyos_quiesce::LAST_THREAD`]'s name. One parks for longer than +//! any stop's budget, one exits at once, and one is the only thread of a child +//! that exits at once. `quiesce-last-park` holds the first inside its +//! `SYS_NANOSLEEP`, `quiesce-last-exit` the second inside its +//! `SYS_THREAD_EXIT`, and `quiesce-last-teardown` the third between leaving its +//! process and tearing it down. The kernel holds the reset itself until one of +//! them is held, so this program orders nothing. //! -//! Nothing here asserts: `common::power::quiesce_wakes_on_the_last_park` and -//! `quiesce_wakes_on_the_last_exit` read the kernel's hold line and its `stop:` -//! record. +//! Nothing here asserts: `common::power::quiesce_wakes_on_the_last_park`, +//! `quiesce_wakes_on_the_last_exit` and `quiesce_wakes_on_the_last_teardown` +//! read the kernel's hold line and its `stop:` record. +use std::process::Command; use std::time::Duration; use toyos::power::Stop; use toyos_quiesce::LAST_THREAD; +const SELF_PATH: &str = "/system/bin/test_rs_quiesce_last"; + /// Longer than any stop's budget: a park the timer ends inside the stop would /// come back to Ring 3 and be banded there, and that band's post would wake the /// stop in place of the park's. const PARKED_FOR: Duration = Duration::from_secs(3_600); fn main() { + if std::env::args().nth(1).as_deref() == Some("teardown") { + toyos_abi::syscall::set_thread_name(LAST_THREAD.as_bytes()); + std::process::exit(0); + } for body in [park as fn(), exit] { std::thread::Builder::new() .name(LAST_THREAD.into()) .spawn(body) .expect("spawn a thread the kernel may hold"); } + let _child = Command::new(SELF_PATH).arg("teardown").spawn().expect("spawn a process the kernel may hold"); // Comes back only refused. let refused = toyos::power::stop(Stop::Reboot); diff --git a/tests/toyos.rs b/tests/toyos.rs index 95e8806e7da..109e2af1ac0 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -182,7 +182,8 @@ const RUST_SKIP: &[&str] = &[ // `quiesce_refuses_a_second_shutdown` runs it. "quiesce_twice", // The same, and its verdict is the stop record of a boot staged around it. - // `quiesce_wakes_on_the_last_park` and `quiesce_wakes_on_the_last_exit` run it. + // `quiesce_wakes_on_the_last_park`, `quiesce_wakes_on_the_last_exit` and + // `quiesce_wakes_on_the_last_teardown` run it. "quiesce_last", // The same, and its verdict is the log volume the stop leaves. // `quiesce_leaves_the_volume_whole` runs it. @@ -1000,6 +1001,7 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // record that boot writes. ("quiesce_wakes_on_the_last_park", Sched::Parallel, Tier::Fast), ("quiesce_wakes_on_the_last_exit", Sched::Parallel, Tier::Fast), + ("quiesce_wakes_on_the_last_teardown", Sched::Parallel, Tier::Fast), // Its own boot: it ends the machine, and its verdict is a dump served // inside that boot's stop. ("quiesce_dump_holds_the_stopped", Sched::Parallel, Tier::Fast), @@ -1741,6 +1743,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("quiesce_refuses_a_second_shutdown", &["test_rs_quiesce_twice"]), ("quiesce_wakes_on_the_last_park", &["test_rs_quiesce_last"]), ("quiesce_wakes_on_the_last_exit", &["test_rs_quiesce_last"]), + ("quiesce_wakes_on_the_last_teardown", &["test_rs_quiesce_last"]), ("quiesce_dump_holds_the_stopped", &["test_rs_quiesce_writers"]), ("quiesce_leaves_the_volume_whole", &["test_rs_quiesce_fsync"]), ("swap_crash_rolls_back", &["test_rs_swap_crash"]), @@ -11331,6 +11334,7 @@ fn run_machine_test( "quiesce_refuses_a_second_shutdown" => power::quiesce_refuses_a_second_shutdown(test_config, c_bins, rust_bins), "quiesce_wakes_on_the_last_park" => power::quiesce_wakes_on_the_last_park(test_config, c_bins, rust_bins), "quiesce_wakes_on_the_last_exit" => power::quiesce_wakes_on_the_last_exit(test_config, c_bins, rust_bins), + "quiesce_wakes_on_the_last_teardown" => power::quiesce_wakes_on_the_last_teardown(test_config, c_bins, rust_bins), "quiesce_dump_holds_the_stopped" => power::quiesce_dump_holds_the_stopped(test_config, c_bins, rust_bins), "watchdog_resets" => power::watchdog_resets(test_config, c_bins, rust_bins), "watchdog_fed" => power::watchdog_fed(test_config, c_bins, rust_bins), diff --git a/toyos-proclife/Cargo.toml b/toyos-proclife/Cargo.toml index 5583a8f07ed..8a1e7811fe0 100644 --- a/toyos-proclife/Cargo.toml +++ b/toyos-proclife/Cargo.toml @@ -53,6 +53,11 @@ mutate-kill-waits-for-its-victims = [] # `interleave::tests::an_exit_and_a_kill_never_both_tear_a_process_down` must # red under this. mutate-first-out-tears-down = [] +# `teardown::leave` marks the last one out dead before its teardown runs, so +# the machine's stop finds nothing running while that teardown frees and logs. +# `interleave::tests::the_last_one_out_is_in_its_process_until_its_teardown_is_done` +# must red under this. +mutate-last-out-leaves-before-its-teardown = [] # `join::collect_zombie` gives up a thread of a process being torn down, whose # mappings its siblings may still run in. # `interleave::tests::a_join_racing_the_kill_that_takes_its_target` must red diff --git a/toyos-proclife/src/interleave.rs b/toyos-proclife/src/interleave.rs index 818fae519cf..0edb88edb84 100644 --- a/toyos-proclife/src/interleave.rs +++ b/toyos-proclife/src/interleave.rs @@ -549,6 +549,18 @@ mod tests { ); } + /// A process's one thread exits: it claims, leaves and tears the process + /// down, and it is in the process for all of the teardown. + /// + /// Reds under `mutate-last-out-leaves-before-its-teardown`. + #[test] + fn the_last_one_out_is_in_its_process_until_its_teardown_is_done() { + let mut world = World::new(); + let pid = world.spawn_process(); + let main = world.main_tid(pid); + holds(&world, vec![Op::exit(pid, main, 0)]); + } + /// A thread killing its own process, with a sibling: it retires itself and /// leaves when its kill returns. #[test] diff --git a/toyos-proclife/src/model.rs b/toyos-proclife/src/model.rs index d7b52b223b2..1a8704691f1 100644 --- a/toyos-proclife/src/model.rs +++ b/toyos-proclife/src/model.rs @@ -67,15 +67,14 @@ enum Out { /// `process::leave`'s table section. Leave, /// The last one out: `teardown_resources`, the process's mappings and handles. - Free { code: i32 }, + Free { code: i32, mark: i32 }, + /// The last one out: `teardown_bookkeeping`'s table section, which marks it dead. + Mark { code: i32, mark: i32 }, /// The last one out: `ProcessObject::publish_exit`. Publish { code: i32 }, /// `thread_exit`'s post on its own watch. Post, - /// The exit pass switches away from the thread. - Switch, - /// The next pass on that CPU frees the payload, kernel stack and all. - Release, + /// The exit pass: the thread never runs again. Gone, } @@ -104,10 +103,6 @@ pub struct World { in_kernel: BTreeSet<(Pid, Tid)>, /// Threads on their way out, and how far each is. departing: BTreeMap<(Pid, Tid), Departure>, - /// Threads the scheduler has switched away from for the last time. - switched: BTreeSet<(Pid, Tid)>, - /// Threads whose kernel stack has been freed. - stacks_freed: BTreeSet<(Pid, Tid)>, /// Threads whose own TLS block is still mapped in their process. mapped: BTreeSet<(Pid, Tid)>, /// Threads whose entry went, and its mapped TLS block with it, while a @@ -150,8 +145,6 @@ impl World { killed: BTreeSet::new(), in_kernel: BTreeSet::new(), departing: BTreeMap::new(), - switched: BTreeSet::new(), - stacks_freed: BTreeSet::new(), mapped: BTreeSet::new(), unmapped_under_siblings: BTreeSet::new(), frees: BTreeMap::new(), @@ -311,12 +304,16 @@ impl World { let chosen = departure.chosen; let next = match departure.at { Out::Leave => match teardown::leave(self, pid, tid, chosen) { - Leave::Last { code } => Out::Free { code }, + Leave::Last { code, mark } => Out::Free { code, mark }, Leave::NotLast => Out::Post, }, - Out::Free { code } => { + Out::Free { code, mark } => { *self.frees.entry(pid).or_insert(0) += 1; self.mapped.retain(|&(p, _)| p != pid); + Out::Mark { code, mark } + } + Out::Mark { code, mark } => { + teardown::torn_down(self, pid, tid, mark); Out::Publish { code } } Out::Publish { code } => { @@ -327,14 +324,6 @@ impl World { if chosen.is_some() { self.post(Watch::Thread(pid, tid)); } - Out::Switch - } - Out::Switch => { - self.switched.insert((pid, tid)); - Out::Release - } - Out::Release => { - self.free_stack(pid, tid); Out::Gone } Out::Gone => unreachable!("a thread that is gone has no step left"), @@ -342,16 +331,17 @@ impl World { self.departing.get_mut(&(pid, tid)).expect("inserted above").at = next; } - /// `Hw::release`: the payload goes, and the kernel stack with it. - pub fn free_stack(&mut self, pid: Pid, tid: Tid) { - self.stacks_freed.insert((pid, tid)); - } - /// Whether this thread can still execute user code. fn runnable(&self, pid: Pid, tid: Tid) -> bool { - !self.stacks_freed.contains(&(pid, tid)) - && !self.departing.contains_key(&(pid, tid)) - && !self.killed.contains(&(pid, tid)) + !self.departing.contains_key(&(pid, tid)) && !self.killed.contains(&(pid, tid)) + } + + /// The thread tearing `pid` down, from its leaving to its mark. + fn tearing_down(&self, pid: Pid) -> Option { + self.departing + .iter() + .find(|(&(p, _), d)| p == pid && matches!(d.at, Out::Free { .. } | Out::Mark { .. })) + .map(|(&(_, tid), _)| tid) } /// `ProcessObject::publish_exit`, assertion and all: two publishes mean two @@ -391,21 +381,27 @@ impl World { out.push(alloc::format!("pid {pid}: its resources were freed {frees} times")); } // L2. Nothing a thread can still run in is freed or published - // before every thread has left. + // before every thread but the one tearing it down has left. + let tearing = self.tearing_down(pid); if frees > 0 || self.published.contains_key(&pid) { for (&tid, &at) in &proc.threads { - if !at.is_zombie() { + if !at.is_zombie() && Some(tid) != tearing { out.push(alloc::format!( "pid {pid} was torn down with tid {tid} still in it", )); } } } - } - // L3. A thread's stack is freed only once the scheduler has switched - // away from it. - for id in self.stacks_freed.difference(&self.switched) { - out.push(alloc::format!("pid {} tid {}: its stack was freed while it ran on it", id.0, id.1)); + // L3. A teardown in flight keeps its thread in the process, where + // the machine's stop counts it as running. + if let Some(tid) = tearing { + if proc.location(tid).is_none_or(ThreadLocation::is_zombie) { + out.push(alloc::format!( + "pid {pid} tid {tid} is tearing its process down and the table has it out, \ + so a stop would not wait for it", + )); + } + } } // L8. No thread's mappings go while a sibling can still run in them. for (pid, tid) in &self.unmapped_under_siblings { @@ -431,7 +427,8 @@ impl World { } } for &(pid, tid) in &self.killed { - if !self.stacks_freed.contains(&(pid, tid)) && self.procs.get(&pid).is_some_and(|p| p.location(tid).is_some()) { + let gone = self.departing.get(&(pid, tid)).is_some_and(|d| d.at == Out::Gone); + if !gone && self.procs.get(&pid).is_some_and(|p| p.location(tid).is_some()) { out.push(alloc::format!("pid {pid} tid {tid} was killed and never left")); } } @@ -468,22 +465,3 @@ impl World { out } } - -#[cfg(test)] -mod tests { - use super::*; - - /// The teeth behind L3: a stack freed before the switch is what it reports. - /// Run by hand, since no kernel path the model scripts frees one early. - #[test] - fn a_stack_freed_before_the_switch_is_what_l3_reports() { - let mut world = World::new(); - let pid = world.spawn_process(); - world.free_stack(pid, world.main_tid(pid)); - let faults = world.faults(); - assert!( - faults.iter().any(|f| f.contains("freed while it ran on it")), - "L3 cannot see a stack freed under its thread: {faults:?}", - ); - } -} diff --git a/toyos-proclife/src/teardown.rs b/toyos-proclife/src/teardown.rs index b8ddde891f5..61397cdddb8 100644 --- a/toyos-proclife/src/teardown.rs +++ b/toyos-proclife/src/teardown.rs @@ -57,15 +57,17 @@ pub fn retire_set(proc: &P, caller: Option) -> Vec { #[must_use = "the last thread out owes its process's teardown"] #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub enum Leave { - /// It emptied a claimed process: it tears the process down and publishes - /// `code`. - Last { code: i32 }, + /// It is emptying a claimed process: it tears the process down, publishes + /// `code`, and stays in the process until [`torn_down`] marks it dead with + /// `mark`, so the machine's stop counts its teardown as running. + Last { code: i32, mark: i32 }, /// Some thread is still in the process, or nobody has claimed it. NotLast, } /// Take `tid` out of `pid`: dead with the code it chose, else its process's -/// for the main thread and [`TORN_DOWN_THREAD_CODE`] for any other. +/// for the main thread and [`TORN_DOWN_THREAD_CODE`] for any other. The last +/// one out of a claimed process is not marked here: [`torn_down`] marks it. /// /// Panics for a thread that is not in its process's entry or has left already: /// an entry outlives every thread that has not left it. @@ -77,26 +79,46 @@ pub fn leave(table: &mut T, pid: Pid, tid: Tid, chosen: Option code, None if tid == proc.main_tid() => { claimed.expect("leave: a main thread leaves only a process somebody claimed") } None => TORN_DOWN_THREAD_CODE, }; - proc.set_location(tid, ThreadLocation::Zombie(code)); - let mut still_in = false; - proc.each_thread(&mut |_, at| still_in |= !at.is_zombie()); + let mut others_in = false; + proc.each_thread(&mut |other, at| others_in |= other != tid && !at.is_zombie()); // The mutation this feature stages: every thread that leaves a claimed // process tears it down, the first one out included. #[cfg(feature = "mutate-first-out-tears-down")] - let still_in = false; + let others_in = false; match claimed { - Some(code) if !still_in => Leave::Last { code }, - _ => Leave::NotLast, + Some(code) if !others_in => { + // The mutation this feature stages: the last one out is marked + // dead before its teardown runs. + #[cfg(feature = "mutate-last-out-leaves-before-its-teardown")] + proc.set_location(tid, ThreadLocation::Zombie(mark)); + Leave::Last { code, mark } + } + _ => { + proc.set_location(tid, ThreadLocation::Zombie(mark)); + Leave::NotLast + } } } +/// The last one out has torn its process down: it is dead with `mark`. +pub fn torn_down(table: &mut T, pid: Pid, tid: Tid, mark: i32) { + let proc = table.get_mut(pid).expect("torn_down: a process is in the table until its exit is published"); + #[cfg(not(feature = "mutate-last-out-leaves-before-its-teardown"))] + assert_eq!( + proc.location(tid), + Some(ThreadLocation::Scheduled), + "torn_down: pid {pid} tid {tid} is not the last one out", + ); + proc.set_location(tid, ThreadLocation::Zombie(mark)); +} + /// Which of the two exits a `SYS_THREAD_EXIT` is. #[must_use = "a thread exit that is not routed is a thread that never dies"] #[derive(Clone, Copy, PartialEq, Eq, Debug)] @@ -193,7 +215,13 @@ mod tests { assert_eq!(leave(&mut world, pid, t2, Some(5)), Leave::NotLast); assert!(claim_teardown(&mut world, pid, 42)); assert_eq!(leave(&mut world, pid, main, None), Leave::NotLast); - assert_eq!(leave(&mut world, pid, t1, None), Leave::Last { code: 42 }); + assert_eq!(leave(&mut world, pid, t1, None), Leave::Last { code: 42, mark: TORN_DOWN_THREAD_CODE }); + assert_eq!( + world.get(pid).unwrap().location(t1), + Some(ThreadLocation::Scheduled), + "the last one out is in its process until its teardown is done", + ); + torn_down(&mut world, pid, t1, TORN_DOWN_THREAD_CODE); let proc = world.get(pid).unwrap(); assert_eq!(proc.location(main), Some(ThreadLocation::Zombie(42))); From 57c89b5507d2a1ef49d7254787605f7897a2f3ea Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 08:16:49 +0200 Subject: [PATCH 08/19] kernel: a process's end is not pollable Nothing but `process_lifecycle`'s own arm polled it: the `Process` arms of `read_watch`, `has_data` and `close_ends_polls`, `ProcessObject`'s `Arc`, `toyos::process::Process::watch_end` and `its_end_completes_a_poll` go back to main's. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- kernel/src/object/ops.rs | 11 ++---- kernel/src/object/process.rs | 7 ++-- .../src/bin/process_lifecycle.rs | 36 ------------------- toyos/src/process.rs | 8 ----- 4 files changed, 7 insertions(+), 55 deletions(-) diff --git a/kernel/src/object/ops.rs b/kernel/src/object/ops.rs index f06f86c101b..3808154fb29 100644 --- a/kernel/src/object/ops.rs +++ b/kernel/src/object/ops.rs @@ -266,11 +266,9 @@ pub fn read_watch(object: &KObjectRef) -> Option { }, // Named unconditionally: the watch alone cannot enforce rights. KObjectRef::SysCap(_) => Some(WatchRef::Static(&crate::log::user::WATCH)), - // Readable once its exit is published. - KObjectRef::Process(p) => Some(WatchRef::Shared(p.watch().clone())), KObjectRef::PipeWrite(_) | KObjectRef::File(_) | KObjectRef::Inbox(_) | KObjectRef::Connector(_) | KObjectRef::Namespace(_) - | KObjectRef::SharedMem(_) => None, + | KObjectRef::SharedMem(_) | KObjectRef::Process(_) => None, } } @@ -308,12 +306,10 @@ fn close_ends_polls(object: &KObjectRef) -> bool { | device_registry::DeviceType::Framebuffer | device_registry::DeviceType::Partition => true, }, - // One handle closing ends no process. - KObjectRef::Process(_) => false, KObjectRef::PipeRead(_) | KObjectRef::PipeWrite(_) | KObjectRef::Connection(_) | KObjectRef::Acceptor(_) | KObjectRef::File(_) | KObjectRef::Inbox(_) | KObjectRef::Connector(_) | KObjectRef::Namespace(_) - | KObjectRef::SharedMem(_) => true, + | KObjectRef::SharedMem(_) | KObjectRef::Process(_) => true, } } @@ -849,10 +845,9 @@ pub fn has_data(object: &KObjectRef) -> bool { !d.info_read() || crate::drivers::virtio_sound::has_pending() } }, - KObjectRef::Process(p) => p.finished(), KObjectRef::PipeWrite(_) | KObjectRef::Inbox(_) | KObjectRef::SysCap(_) | KObjectRef::Connector(_) | KObjectRef::Namespace(_) - | KObjectRef::SharedMem(_) => false, + | KObjectRef::SharedMem(_) | KObjectRef::Process(_) => false, } } diff --git a/kernel/src/object/process.rs b/kernel/src/object/process.rs index 40490ab3b8e..d8438b6ca7d 100644 --- a/kernel/src/object/process.rs +++ b/kernel/src/object/process.rs @@ -29,7 +29,8 @@ pub struct ProcessObject { exit: Lock>, /// The same fact, without the lock, for a waiter's per-wake predicate. finished: AtomicBool, - watch: Arc, + /// What `SYS_PROCESS_WAIT` arms on; holding the `Arc` across the park keeps the watch from outliving its subject. + watch: Watch, } impl ProcessObject { @@ -39,7 +40,7 @@ impl ProcessObject { pid, exit: Lock::new(None), finished: AtomicBool::new(false), - watch: Arc::new(Watch::new()), + watch: Watch::new(), }) } @@ -60,7 +61,7 @@ impl ProcessObject { self.exit.lock().as_ref().map(|e| e.stats) } - pub fn watch(&self) -> &Arc { + pub fn watch(&self) -> &Watch { &self.watch } diff --git a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs index 7e940d38b66..aa98fa60e9a 100644 --- a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs +++ b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs @@ -32,7 +32,6 @@ use std::sync::OnceLock; use std::time::{Duration, Instant}; use toyos::endow::{Endowments, SYSCAP_LABEL}; -use toyos::poller::Poller; use toyos::AsHandle; use toyos::process::Process; use toyos::syscap::SysCap; @@ -63,7 +62,6 @@ fn test() { an_unrelated_wake_does_not_end_the_wait(); two_handles_answer_the_same(); a_kill_publishes_like_an_exit(); - its_end_completes_a_poll(); a_handle_is_the_whole_of_the_right(); a_pid_is_not_authority(); an_undefined_wait_flag_bit_is_refused(); @@ -286,40 +284,6 @@ fn a_kill_publishes_like_an_exit() { println!(" a kill publishes {KILLED} once, and a second kill changes nothing"); } -/// A process's end is an event: its handle turns readable when the exit is -/// published, never before, and a poll registered after the end completes at -/// once. -fn its_end_completes_a_poll() { - const END: u64 = 1; - const LATE: u64 = 2; - let (mut child, release) = start(4); - let handle = syscall::dup(RawHandle(child.as_raw_handle())).expect("a handle to poll"); - // SAFETY: the duplicate is this function's alone. - let process = unsafe { Process::from_raw(handle) }; - let poller = Poller::new(1); - - process.watch_end(&poller, END); - let mut early = Vec::new(); - poller.wait(0, 0, |token| early.push(token)); - // Another handle's close ends no process, so it cancels no poll. - syscall::close(syscall::dup(RawHandle(child.as_raw_handle())).expect("a handle to close")); - poller.wait(0, 0, |token| early.push(token)); - assert!(early.is_empty(), "a running process's handle completed a poll: {early:?}"); - - drop(release); - let mut ended = Vec::new(); - poller.wait(1, u64::MAX, |token| ended.push(token)); - assert_eq!(ended, [END], "the end completed no poll"); - assert_eq!(process.try_wait(), Ok(4), "the poll completed before the exit was published"); - - process.watch_end(&poller, LATE); - let mut late = Vec::new(); - poller.wait(1, u64::MAX, |token| late.push(token)); - assert_eq!(late, [LATE], "a poll on an ended process did not complete"); - assert_eq!(child.wait().expect("wait").code(), Some(4)); - println!(" a process's end completes a poll on its handle, and not before"); -} - /// **The arm the pid-keyed shape could not have.** The waiter did not spawn the /// subject, is not its parent by any spelling, and holds nothing but a handle /// somebody moved into its table — and that is enough. diff --git a/toyos/src/process.rs b/toyos/src/process.rs index c1c6510bf76..bd807e146ca 100644 --- a/toyos/src/process.rs +++ b/toyos/src/process.rs @@ -10,7 +10,6 @@ use toyos_abi::handle::Rights; use toyos_abi::syscall::{self, ProcessStats, SyscallError}; use crate::endow::FromHandle; -use crate::poller::{Poller, READABLE}; use crate::{AsHandle, OwnedHandle, RawHandle}; pub struct Process(pub(crate) OwnedHandle); @@ -31,17 +30,10 @@ impl Process { } /// Kill it. `Ok` for one already dead: the caller asked for it to be gone. - /// Returns before it is gone; its end is what [`wait`](Self::wait) and - /// [`watch_end`](Self::watch_end) report. pub fn kill(&self) -> Result<(), SyscallError> { syscall::process_kill(self.0.raw()) } - /// Complete `token` on `poller` once its exit is published. - pub fn watch_end(&self, poller: &Poller, token: u64) { - poller.watch(self, READABLE, token); - } - pub fn stats(&self) -> Result { let mut stats = ProcessStats::default(); syscall::process_stats(self.0.raw(), &mut stats)?; From 179f41efd1274d86e59731e47314d761d5849ec9 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 08:16:49 +0200 Subject: [PATCH 09/19] kill_while_blocked: arm 4 waits for its spinner's exit, with no clock A kill returns before its victim ends, so the killer can observe the end itself: arm 4 kills the spinner and waits for its exit code. The in-guest `ENDS_WITHIN` deadline and the timed stdout reader go; a spinner the exit boundary misses never ends, and the harness's hang ceiling says so. The deferred-release issue loses the sentence whose only reason was `retire_task`'s park. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- .../deferred-release-outlives-its-syscall.md | 2 - .../src/bin/kill_while_blocked.rs | 68 +++---------------- 2 files changed, 8 insertions(+), 62 deletions(-) diff --git a/issues/kernel/deferred-release-outlives-its-syscall.md b/issues/kernel/deferred-release-outlives-its-syscall.md index 002065f6f8b..a517a8fce19 100644 --- a/issues/kernel/deferred-release-outlives-its-syscall.md +++ b/issues/kernel/deferred-release-outlives-its-syscall.md @@ -183,8 +183,6 @@ at all.** It belongs with that track, not beside it. ### What "give the batch an owner" costs, worked out 2026-08-20 -An owner has to be the *thread*, not the CPU. - A per-thread list cannot live behind `ThreadData`'s lock either: `teardown_resources` holds `ProcessData` across `close_all`, and its own first line is that the two locks are never held together. So the list has to be on the diff --git a/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs b/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs index 59826e64aea..e576843cefb 100644 --- a/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs +++ b/tests/toyos-rust-tests/src/bin/kill_while_blocked.rs @@ -44,9 +44,6 @@ use std::io::{Read, Write}; use std::os::toyos::process::CommandExt; use std::process::{Command, Stdio}; -use std::sync::atomic::{AtomicU64, Ordering}; -use std::thread; -use std::time::{Duration, Instant}; use toyos::{endow, namespace, port, AsHandle}; use toyos_abi::syscall::{self, SyscallError, SERVE_PREFIX, SVC_LABEL}; @@ -54,22 +51,8 @@ use toyos_abi::syscall::{self, SyscallError, SERVE_PREFIX, SVC_LABEL}; const SELF_PATH: &str = "/system/bin/test_rs_kill_while_blocked"; const SERVICE: &str = "blocked"; -/// How long arm 4 gives a killed Ring 3 spinner to reach its last exit -/// boundary, watched from outside the kill. -/// -/// **A number rather than a hang**: a guest that stops making progress reds -/// as `STALL`, which the harness prints apart and tells nobody to bisect. -/// What it buys is a failure that names itself. -/// -/// **Priced against the quantity it actually bounds**, which is not one -/// interrupt delivery: what this constant covers is the whole of -/// [go byte → boundary → teardown], and the only measurement of that window is -/// this arm's own recorded green run, 4.972884 ms. Two seconds against it is -/// 402×. The sentence that used to stand here said "four orders of magnitude of -/// headroom", which would be true of the interrupt delivery alone — precisely -/// the part this arm cannot observe from outside the kill, and the part the -/// window's several scheduler dispatches sit on top of. -const ENDS_WITHIN: Duration = Duration::from_secs(2); +/// `process::KILLED_EXIT_CODE`. +const KILLED: i32 = 137; fn main() { match std::env::args().nth(1).as_deref() { @@ -204,50 +187,15 @@ fn an_acceptor_killed_in_the_accept() { /// all. With either miss the child here is preempted, queued in the dying list, /// picked straight back off it and returned to Ring 3, once per tick, forever. /// -/// **What it watches is the victim's own stdout.** That pipe's write end is in -/// the victim's handle table and in no other, so the read end this process -/// holds reaches EOF when — and only when — the victim's handles are drained. -/// -/// **The clock starts before the kill and not after it.** -/// What this one bounds is the whole of [kill → boundary → teardown], so the -/// number is an upper bound on the boundary rather than a measurement of it. +/// **What it watches is the victim's exit.** Its one thread publishes that +/// exit only once it has left at its exit boundary, so a spinner the boundary +/// misses never ends, and the harness's hang ceiling is what says so. fn a_ring_three_spinner_ends_at_its_next_exit_boundary() { let mut victim = parked("spin", None); - // Taken out of the `Child` because the observation is the pipe and not the - // process: `parked` has already read the marker line off it, so the next - // thing that can ever arrive on it is the end of the victim. - let mut spun = victim.stdout.take().expect("the spinner's stdout"); - - /// Nanoseconds from the kill to EOF, and `u64::MAX` until there is one. - /// Stored by the reader so the answer is the instant it saw rather than the - /// poll that noticed. - static GONE_AFTER_NS: AtomicU64 = AtomicU64::new(u64::MAX); - let started = Instant::now(); - thread::spawn(move || { - let mut byte = [0u8; 1]; - while spun.read(&mut byte).expect("read the spinner's stdout") != 0 {} - GONE_AFTER_NS.store(started.elapsed().as_nanos() as u64, Ordering::Release); - }); victim.kill().expect("kill the spinning child"); - - while GONE_AFTER_NS.load(Ordering::Acquire) == u64::MAX && started.elapsed() < ENDS_WITHIN { - thread::sleep(Duration::from_millis(1)); - } - let gone = GONE_AFTER_NS.load(Ordering::Acquire); - if gone == u64::MAX { - println!( - "a child killed while spinning in Ring 3 still held its handles {:?} later — \ - it is being re-dispatched into userland, so nothing on the return path reads \ - the kill bit", - started.elapsed(), - ); - std::process::exit(1); - } - println!( - " ring 3: a killed spinner reached its last exit boundary in {:?}, with no syscall \ - to cancel", - Duration::from_nanos(gone), - ); + let code = victim.wait().expect("wait for the killed spinner").code(); + assert_eq!(code, Some(KILLED), "a child killed while spinning in Ring 3 ended with {code:?}"); + println!(" ring 3: a killed spinner left at its exit boundary, with no syscall to cancel"); } fn child(role: &str) -> ! { From 0c48340f3afba6c224a1184530ecf6dff88c397d Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 08:16:50 +0200 Subject: [PATCH 10/19] kernel-loom: the nested serve is a thread of its own `a_nested_serve_is_not_undone_by_the_one_it_interrupted` ran the nesting as one written-out schedule, one execution. The nested issue and serve now run on a spawned thread, so loom places them at every point of the outer serve: 11 executions at a preemption bound of 2, and the model reds when `serve` stores what it owes instead of raising to it. Filed: `sync.rs`'s lock spin says it runs with `IF` clear, and kernel threads and the idle loop spin there with it set, which is where one serve nests inside another. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-clear-and-two-callers-spin-with-it-set.md | 22 +++++++++++++++++ kernel-loom/tests/tlb_shootdown.rs | 24 +++++++++++-------- 2 files changed, 36 insertions(+), 10 deletions(-) create mode 100644 issues/kernel/the-lock-spins-shootdown-poll-says-if-is-clear-and-two-callers-spin-with-it-set.md diff --git a/issues/kernel/the-lock-spins-shootdown-poll-says-if-is-clear-and-two-callers-spin-with-it-set.md b/issues/kernel/the-lock-spins-shootdown-poll-says-if-is-clear-and-two-callers-spin-with-it-set.md new file mode 100644 index 00000000000..5aa7a0976a8 --- /dev/null +++ b/issues/kernel/the-lock-spins-shootdown-poll-says-if-is-clear-and-two-callers-spin-with-it-set.md @@ -0,0 +1,22 @@ +--- +status: open +kind: defect +opened: 2026-09-28 +--- + +# The lock spin's shootdown poll says `IF` is clear, and two of its callers spin with it set + +`Lock::lock` (`kernel/src/sync.rs`) polls TLB shootdowns inside its spin with the +comment "this spin runs with `IF` clear". Two callers spin there with `IF` set: + +- a kernel thread, which `kernel_start` (`arch/x86_64/entry.rs`) enters with `sti`; +- the idle loop, whose `reap_finished` takes `PROCESS_TABLE.lock()`. + +There the 0xFE IPI can land inside `tlb::poll`'s own `serve_if_owed`, so one +CPU runs a serve nested inside another. `Shootdown::serve` raises `flushed` with +`fetch_max` for exactly that case, and `kernel-loom`'s +`a_nested_serve_is_not_undone_by_the_one_it_interrupted` reds when it stores. +The comment gives the poll a reason that holds only for syscall context. + +**Exit**: the comment states when the spin runs with `IF` set and when clear, and +why the poll is needed in each. diff --git a/kernel-loom/tests/tlb_shootdown.rs b/kernel-loom/tests/tlb_shootdown.rs index f6e6fbbd31a..8827622f38f 100644 --- a/kernel-loom/tests/tlb_shootdown.rs +++ b/kernel-loom/tests/tlb_shootdown.rs @@ -238,20 +238,24 @@ fn an_initiator_answers_while_it_waits() { /// A serve that an interrupt nests inside another publishes the later /// generation, and the outer serve, finishing after it, must not take it back. /// -/// Not an interleaving question either: the nesting is one CPU's own schedule, -/// written out. Reds when `serve` stores what it owes instead of raising to it. +/// The nested serve runs on a thread of its own, so loom places it at every +/// point of the outer one, the nesting among them. Reds when `serve` stores +/// what it owes instead of raising to it. #[test] fn a_nested_serve_is_not_undone_by_the_one_it_interrupted() { model(|| { - let s = Shootdown::new(); + let s = Arc::new(Shootdown::new()); let first = s.issue(); - let mut later = None; - s.serve(1, || { - let g = s.issue(); - s.serve(1, || {}); - later = Some(g); - }); - let later = later.expect("the nested serve ran"); + let nested = { + let s = Arc::clone(&s); + loom::thread::spawn(move || { + let later = s.issue(); + s.serve(1, || {}); + later + }) + }; + s.serve(1, || {}); + let later = nested.join().unwrap(); assert!(s.served(1, first)); assert!( s.served(1, later), From 63d0b08aa91a1e871ca155dfde95f348948b2a12 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 11:38:42 +0200 Subject: [PATCH 11/19] quiesce-last: the hold ends on a sweep that counted the held thread alone `hold` released on the first sweep that counted one thread running, which need not be the held one. With the last thread out marked a zombie before its teardown, no sweep counts it; a sweep that counted some other thread still in Ring 3 released the hold, and `quiesce_wakes_on_the_last_teardown`, which judged on `sweeps >= 2`, passed with the teardown unwaited. `hold` now claims the held thread's id, `sweep` notes whether it counted that thread as running, and the hold ends only on a sweep that counted exactly one running thread, the held one. It then logs `: the stop counts quiesce-last alone`. The harness requires that line before the `stop:` record for all three `quiesce-last-*` boots, and for the teardown boot also the held process's `exit: test_rs_quiesce_last pid=... code=0` record between the two. The Teardown hold sits in `process::leave`, which the Ring 3 exit boundary also reaches, at depth 0, where `yield_now` asserts. `hold` refuses by name there (`scheduler::may_yield`), logs that the thread left outside a syscall, and lets the stop go on unheld; the harness then reds on the missing line. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- kernel/src/quiesce.rs | 72 +++++++++++++++++++++++++++++------------ kernel/src/scheduler.rs | 6 ++++ tests/common/power.rs | 42 +++++++++++++++++------- 3 files changed, 87 insertions(+), 33 deletions(-) diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 1ca68d0c6de..95a924899af 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -181,8 +181,6 @@ pub fn stop() -> Record { loop { let swept = sweep(caller); sweeps += 1; - #[cfg(feature = "boot-actuators")] - last::note_sweep(swept); let elapsed = (crate::clock::now() - began).nanos(); if swept.keep_waiting(elapsed, PARK.nanos()) { // Uncancellable: the claim is taken, and a caller that left here @@ -213,6 +211,8 @@ pub fn stop() -> Record { /// on another CPU finishing its own teardown takes this same lock. fn sweep(caller: ThreadId) -> Sweep { let mut out = Sweep::default(); + #[cfg(feature = "boot-actuators")] + let mut held_running = false; let guard = process::PROCESS_TABLE.lock(); let Some(table) = guard.as_ref() else { return out }; for (_, proc) in table.iter() { @@ -237,9 +237,15 @@ fn sweep(caller: ThreadId) -> Sweep { out.stopped += 1; } else { out.running += 1; + #[cfg(feature = "boot-actuators")] + { + held_running |= last::is_held(ThreadId { pid: pid.raw(), tid: tid.raw() }); + } } } } + #[cfg(feature = "boot-actuators")] + last::note_sweep(out, held_running); out } @@ -252,10 +258,10 @@ fn sweep(caller: ThreadId) -> Sweep { #[cfg(feature = "boot-actuators")] pub mod last { use core::sync::atomic::{ - AtomicBool, AtomicU32, Ordering::AcqRel, Ordering::Acquire, Ordering::Release, + AtomicBool, AtomicU64, Ordering::AcqRel, Ordering::Acquire, Ordering::Release, }; - use toyos_quiesce::Sweep; + use toyos_quiesce::{Sweep, ThreadId}; use toyos_sched::task::WaitClass; use crate::watch::{self, Watch}; @@ -294,17 +300,26 @@ pub mod last { "the boot panics naming the side of the staging that never arrived", ); - /// No sweep yet. - const UNSWEPT: u32 = u32::MAX; + /// No thread claimed yet: `percpu`'s spelling of idle, which no thread has. + const NOBODY: u64 = u64::MAX; - /// The latest sweep's running count. - static RUNNING: AtomicU32 = AtomicU32::new(UNSWEPT); - static HELD: AtomicBool = AtomicBool::new(false); + /// The one thread this boot holds, `pid` high and `tid` low. + static HELD: AtomicU64 = AtomicU64::new(NOBODY); + /// Whether the latest sweep counted one thread running, and that one [`HELD`]. + static ALONE: AtomicBool = AtomicBool::new(false); /// What the shutdown's caller parks on until a thread is held. static ARRIVED: Watch = Watch::new(); - pub(super) fn note_sweep(swept: Sweep) { - RUNNING.store(swept.running, Release); + fn word(thread: ThreadId) -> u64 { + (u64::from(thread.pid) << 32) | u64::from(thread.tid) + } + + pub(super) fn is_held(thread: ThreadId) -> bool { + HELD.load(Acquire) == word(thread) + } + + pub(super) fn note_sweep(swept: Sweep, held_running: bool) { + ALONE.store(swept.running == 1 && held_running, Release); } fn armed() -> Option { @@ -313,7 +328,22 @@ pub mod last { /// Hold the running thread here if it is the one `last` stages. pub fn hold(last: Last) { - if !last.armed() || !is_the_named_thread() || HELD.swap(true, AcqRel) { + if !last.armed() { + return; + } + let Some(thread) = the_named_thread() else { return }; + if HELD.compare_exchange(NOBODY, word(thread), AcqRel, Acquire).is_err() { + return; + } + if !crate::scheduler::may_yield() { + // A thread killed in Ring 3 leaves at its exit boundary, at a depth + // where a yield asserts: refused there, and the stop goes on unheld. + crate::log!( + "{}: {} left outside a syscall and is not held", + last.name(), + toyos_quiesce::LAST_THREAD, + ); + ARRIVED.post(); return; } crate::log!( @@ -325,7 +355,7 @@ pub mod last { let deadline = Deadline::at(crate::clock::now() + STAGED.duration()); // Yields and never parks: a park is the transition this hold exists to // place, and a sweep would stop this thread at the first one. - while RUNNING.load(Acquire) != 1 { + while !ALONE.load(Acquire) { assert!( !deadline.reached(crate::clock::now()), "{}: the stop never came down to this thread alone in {} ms", @@ -334,6 +364,7 @@ pub mod last { ); crate::scheduler::yield_now(); } + crate::log!("{}: the stop counts {} alone", last.name(), toyos_quiesce::LAST_THREAD); } /// Called by the shutdown before it stops anything: the stop is staged @@ -353,10 +384,10 @@ pub mod last { 0, WaitClass::Other, deadline, - || HELD.load(Acquire), + || HELD.load(Acquire) != NOBODY, ); assert!( - HELD.load(Acquire), + HELD.load(Acquire) != NOBODY, "{}: no thread named {} reached its syscall in {} ms", last.name(), toyos_quiesce::LAST_THREAD, @@ -364,17 +395,16 @@ pub mod last { ); } - fn is_the_named_thread() -> bool { - let (Some(pid), Some(tid)) = - (crate::arch::percpu::current_pid(), crate::arch::percpu::current_tid()) - else { - return false; - }; + /// The running thread, if it carries [`toyos_quiesce::LAST_THREAD`]'s name. + fn the_named_thread() -> Option { + let pid = crate::arch::percpu::current_pid()?; + let tid = crate::arch::percpu::current_tid()?; let guard = crate::process::PROCESS_TABLE.lock(); guard .as_ref() .and_then(|table| table.get(pid)) .and_then(|proc| proc.threads().get(tid)) .is_some_and(|thread| thread.name_str() == toyos_quiesce::LAST_THREAD) + .then_some(ThreadId { pid: pid.raw(), tid: tid.raw() }) } } diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index 6bfec8571be..3f05fdd4f2a 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -342,6 +342,12 @@ pub fn yield_now() { driver::pass(Dispose::Yield); } +/// Whether [`yield_now`] may be called where the running context stands. +#[cfg(feature = "boot-actuators")] +pub fn may_yield() -> bool { + crate::preempt::count() == blocking_baseline() +} + /// Unified preempt entry: the Ring 3 timer path, `kernel_exit_to_user_check` /// and the `preempt::enable` slow path all funnel through here. #[track_caller] diff --git a/tests/common/power.rs b/tests/common/power.rs index 9ee880f78fb..700e56c727b 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -334,7 +334,7 @@ pub fn quiesce_wakes_on_the_last_park( _c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - woken_by_the_held_thread(&["quiesce-last-park", LATE_WORD], rust_bins) + woken_by_the_held_thread(&["quiesce-last-park", LATE_WORD], None, rust_bins) } /// **An exit that is the stop's last transition wakes it.** The same, with @@ -344,25 +344,29 @@ pub fn quiesce_wakes_on_the_last_exit( _c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - woken_by_the_held_thread(&["quiesce-last-exit", LATE_WORD], rust_bins) + woken_by_the_held_thread(&["quiesce-last-exit", LATE_WORD], None, rust_bins) } /// **A process teardown that is the stop's last transition is waited for.** /// The same, with `quiesce-last-teardown` holding the last thread out of a -/// process between its leaving and its teardown: a stop that counted it as -/// gone would return at its first sweep while that teardown still frees and -/// logs. +/// process between its leaving and its teardown. pub fn quiesce_wakes_on_the_last_teardown( _test_config: &Path, _c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - woken_by_the_held_thread(&["quiesce-last-teardown", LATE_WORD], rust_bins) + woken_by_the_held_thread( + &["quiesce-last-teardown", LATE_WORD], + Some("test_rs_quiesce_last"), + rust_bins, + ) } -/// One of the `quiesce-last-*` boots: its actuator first, the late word beside it. +/// One of the `quiesce-last-*` boots: its actuator first, the late word beside it; +/// `torn_down` names the process whose teardown the held thread runs. fn woken_by_the_held_thread( armed: &'static [&'static str; 2], + torn_down: Option<&str>, rust_bins: &[(String, Vec)], ) -> Result<(), String> { let actuator = armed[0]; @@ -386,13 +390,27 @@ fn woken_by_the_held_thread( if held_at > synced_at { return Err(format!("the thread was held after the stop was over\n{whole}")); } - // More than one sweep: the stop found the held thread running and had - // to be woken to see it stop. - if record.sweeps < 2 { + // **The claim**: a sweep counted the held thread, and it alone, as + // running before the stop wrote its record, so what the held thread did + // next is what the stop waited for. + let alone = format!("{actuator}: the stop counts {} alone", toyos_quiesce::LAST_THREAD); + let stopped_at = at(toyos_quiesce::STOPPED) + .ok_or_else(|| format!("the kernel wrote no stop record\n{whole}"))?; + let Some(alone_at) = at(&alone).filter(|&line| line < stopped_at) else { return Err(format!( - "the stop saw nothing running at its first sweep, so no transition of the held \ - thread was waited for:\n {record}" + "no {alone:?} line before the stop's record, so no transition of the held thread \ + was waited for:\n {record}\n{whole}" )); + }; + if let Some(process) = torn_down { + let exit = format!("exit: {process} pid="); + let mut torn = whole.lines().skip(alone_at).take(stopped_at - alone_at); + if !torn.any(|line| line.contains(&exit) && line.contains(" code=0 ")) { + return Err(format!( + "no `{exit}… code=0` record between {alone:?} and the stop's record, so the \ + stop did not wait for that teardown\n{whole}" + )); + } } woken_by_its_threads(&record)?; eprintln!(" [power] {actuator}: the held thread's transition woke the stop: {record}"); From a5924980a2d11d3f6b2998e1e2388f4c6389de06 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 11:39:33 +0200 Subject: [PATCH 12/19] scheduler: a killed thread's teardown at the Ring 3 boundary runs with IF set `kernel_exit_to_user_check` is entered with IF clear, so `process::leave` at the exit boundary ran a killed process's whole teardown -- `close_all`, the address-space drop, the records -- with interrupts masked on that CPU. A syscall's exit runs the same teardown with IF set. `leave_ring3_if_due` now sets IF around `leave` and clears it again before the exit pass, the bracket `kernel_exit_to_user_check` already puts around `do_preempt`. The preempt depth is unchanged: 0, `BASELINE_IRQ_EXIT`, which is `do_preempt`'s own. What IF adds is a nested interrupt, which on a Ring 0 return reaches no scheduler entry, and a `preempt::enable` at depth 0 calling `do_preempt`, which asserts that same baseline. A park asserts at depth 0 with IF either way, and nothing the teardown runs parks. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- kernel/src/scheduler.rs | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index 3f05fdd4f2a..14b41910db0 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -386,7 +386,12 @@ pub fn leave_ring3_if_due() { unreachable!("leave_ring3_if_due: a stopped task was dispatched again"); } SafePoint::Exit => { + // `IF` set across the teardown, as a syscall's exit runs it: its + // closes and address-space drop are no interrupt latency. The + // depth stays this boundary's, which is `do_preempt`'s own. + crate::arch::cpu::enable_interrupts(); process::leave(None); + crate::arch::cpu::disable_interrupts(); driver::pass(Dispose::Exit); unreachable!("leave_ring3_if_due: returned from the exit pass"); } From 751e36d98f822052a403e97727a06c6782593a9a Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 11:39:43 +0200 Subject: [PATCH 13/19] kill_ends_every_wait: a ceiling red names the arm; the arm race is filed `check_rust_result` printed only the kernel's account when the harness ended a test at its ceiling, though the streamed stdout is in the result; it now prints it, as the exit-code branches do. `kill_ends_every_wait` prints `: killing` before each kill, so a wait the kill cannot end is the last line of that red. The race by which an arm passes without reaching its wait moves out of the module doc into `issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md`, and `quiesce::last`'s in-guest `STAGED` deadline is filed as `issues/diagnostics/the-quiesce-last-staging-dies-on-ten-seconds-of-guest-clock.md`. `teardown`'s doc loses a clause naming a wake `teardown_bookkeeping` does not make. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...ging-dies-on-ten-seconds-of-guest-clock.md | 19 +++++++++++++++++++ ...-arm-can-pass-without-reaching-its-wait.md | 18 ++++++++++++++++++ kernel/src/process.rs | 2 +- .../src/bin/kill_ends_every_wait.rs | 5 +---- tests/toyos.rs | 2 +- 5 files changed, 40 insertions(+), 6 deletions(-) create mode 100644 issues/diagnostics/the-quiesce-last-staging-dies-on-ten-seconds-of-guest-clock.md create mode 100644 issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md diff --git a/issues/diagnostics/the-quiesce-last-staging-dies-on-ten-seconds-of-guest-clock.md b/issues/diagnostics/the-quiesce-last-staging-dies-on-ten-seconds-of-guest-clock.md new file mode 100644 index 00000000000..6d67eed1a25 --- /dev/null +++ b/issues/diagnostics/the-quiesce-last-staging-dies-on-ten-seconds-of-guest-clock.md @@ -0,0 +1,19 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# The `quiesce-last-*` staging dies on ten seconds of guest clock + +`kernel/src/quiesce.rs`'s `last::STAGED` is a 10 s in-guest deadline. `hold` +panics the boot when the stop has not counted the held thread alone within it, +and `await_the_held_thread` panics when no thread named `quiesce-last` has +reached its syscall within it. `quiesce-last-park`, `quiesce-last-exit` and +`quiesce-last-teardown` all rest on it, and `quiesce_twice`'s `WAITS_WITHIN` is +priced inside it. A QEMU test's only clock is the harness's ceiling: a guest +slower than the deadline dies with a staging panic that names no defect. + +Owner: orchestrator. Exit condition: both sides of the staging wait on the +other's event with no deadline, and a staging that never arrives is the +harness's ceiling. diff --git a/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md b/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md new file mode 100644 index 00000000000..4888b36bb42 --- /dev/null +++ b/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md @@ -0,0 +1,18 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# A `kill_ends_every_wait` arm can pass without reaching its wait + +`tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs` kills each child once +it reads the child's `parked in ` marker, which the child prints just +before the syscall that parks. A kill that lands between the marker and the +park ends the child at that syscall's exit boundary instead, and the arm passes +without the wait it names ever being killed. The per-arm mutations that make one +wait uncancellable each went red when measured, so each arm reached its wait in +those runs; nothing makes it reach it in every run. + +Owner: orchestrator. Exit condition: the parent sees the child parked in the +named wait, by the kernel's word, before it kills it. diff --git a/kernel/src/process.rs b/kernel/src/process.rs index a2055a5168c..c8fde7f09c8 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -1023,7 +1023,7 @@ pub fn stats_from( /// The last thread out's teardown of its process, on its own stack: free what the process holds, then publish its exit. Every other thread has left, so none can run in what is freed. /// The thread is in the process until the bookkeeping marks it, so a machine stop waits for every record and release here. /// Waits on nothing: it runs on a killed thread whose one cancel may be spent, and at the exit boundary's preempt depth, where a park asserts. -/// Publish happens after the table lock is released — `teardown_bookkeeping`'s wake needs that — and once published the entry is reapable, so nothing may read the table for this pid after. +/// Publish happens after the table lock is released, and once published the entry is reapable, so nothing may read the table for this pid after. fn teardown(pid: Pid, tid: Tid, code: i32, mark: i32, process_data: &Arc>) { let main_thread_data = { let guard = PROCESS_TABLE.lock(); diff --git a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs index ad0a8cc408e..fccce9aa619 100644 --- a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs +++ b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs @@ -7,10 +7,6 @@ //! `wait` below never returns; the harness's deadline is what says so. //! `kill_while_blocked` holds the pipe, connection and accept waits, and //! `mutual_kill` a kill inside a kill. -//! -//! The marker is printed immediately before the wait, so a kill that lands in -//! the few instructions between the two ends the child at that syscall's exit -//! instead, and the arm passes without its wait under test. use std::io::{Read, Write}; use std::os::toyos::process::{ChildExt, CommandExt}; @@ -51,6 +47,7 @@ fn test() { (WAITED.to_string(), dup.0) }); let mut child = spawn(role, endow); + println!(" {role}: killing"); child.kill().expect("kill the parked child"); let code = child.wait().expect("wait for the killed child").code(); assert_eq!(code, Some(KILLED), "a child killed in its {role} wait ended with {code:?}"); diff --git a/tests/toyos.rs b/tests/toyos.rs index 81e8c161517..ccf50633c4f 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -3053,7 +3053,7 @@ fn check_rust_result(result: &TestResult) -> bool { let test_name = result.name.strip_prefix("test_rs_").unwrap_or(&result.name); if let Some(err) = &result.error { - eprintln!("FAIL rs::{test_name}: {err}{}", kernel_account(result)); + eprintln!("FAIL rs::{test_name}: {err}\nstdout:\n{}{}", result.stdout, kernel_account(result)); return false; } From 009fa2a891565a067ff9116c7b33a08f587fef99 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 13:41:52 +0200 Subject: [PATCH 14/19] A refused hold releases its slot, a gave-up stop names its threads, and sleep leads the wait arms `quiesce::hold` claimed the one hold slot before checking whether the thread may yield, and left it claimed on refusal, so the thread a boot actually stages could never take it. The slot is now released on refusal. `kill_ends_every_wait`'s `process-wait` and `thread-join` arms both sleep underneath (the waited process's `nanosleep`, the joined thread's `std::thread::sleep`), so a mutation that breaks sleep surfaced under one of their names. `sleep` now runs first among the three. `stopped_boot`'s gave-up error was the one error in the function that omitted the boot's whole log, so a sighting of a stop giving up named no threads. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...nds-every-wait-arm-can-pass-without-reaching-its-wait.md | 5 ++--- kernel/src/quiesce.rs | 3 +++ tests/common/power.rs | 2 +- tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs | 6 +++++- 4 files changed, 11 insertions(+), 5 deletions(-) diff --git a/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md b/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md index 4888b36bb42..a7cde124dc4 100644 --- a/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md +++ b/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md @@ -10,9 +10,8 @@ opened: 2026-09-28 it reads the child's `parked in ` marker, which the child prints just before the syscall that parks. A kill that lands between the marker and the park ends the child at that syscall's exit boundary instead, and the arm passes -without the wait it names ever being killed. The per-arm mutations that make one -wait uncancellable each went red when measured, so each arm reached its wait in -those runs; nothing makes it reach it in every run. +without the wait it names ever being killed. Nothing makes it reach it in every +run. Owner: orchestrator. Exit condition: the parent sees the child parked in the named wait, by the kernel's word, before it kills it. diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 95a924899af..2425e4228cd 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -338,6 +338,9 @@ pub mod last { if !crate::scheduler::may_yield() { // A thread killed in Ring 3 leaves at its exit boundary, at a depth // where a yield asserts: refused there, and the stop goes on unheld. + // The slot this thread claimed above is freed, or the thread the + // boot stages could never take it. + HELD.store(NOBODY, Release); crate::log!( "{}: {} left outside a syscall and is not held", last.name(), diff --git a/tests/common/power.rs b/tests/common/power.rs index 700e56c727b..63488f05353 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -318,7 +318,7 @@ fn stopped_boot( .ok_or_else(|| format!("the kernel's stop record did not read back as one:\n {said}"))?; if !record.stopped_the_machine() { return Err(format!( - "the stop gave up on {} thread(s) that never reached a safe point:\n {record}", + "the stop gave up on {} thread(s) that never reached a safe point:\n {record}\n{whole}", record.sweep.running, )); } diff --git a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs index fccce9aa619..bbf63d8bd5b 100644 --- a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs +++ b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs @@ -29,7 +29,11 @@ const WAITED: &str = "waited"; /// `process::KILLED_EXIT_CODE`. const KILLED: i32 = 137; -const WAITS: [&str; 5] = ["futex", "poll", "process-wait", "thread-join", "sleep"]; +// `sleep` runs before `process-wait` and `thread-join`: both of those arms also +// sleep underneath (the waited process's `nanosleep`, the joined thread's +// `std::thread::sleep`), so a mutation that breaks sleep would otherwise surface +// under one of their names instead of its own. +const WAITS: [&str; 5] = ["futex", "poll", "sleep", "process-wait", "thread-join"]; fn main() { match std::env::args().nth(1).as_deref() { From 4d612d6d1dae74315fc4ffa8e73bd098a55c0def Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 14:19:16 +0200 Subject: [PATCH 15/19] quiesce_wakes_on_the_last_teardown shares quiesce_wakes_on_the_last_park's disabled issue Its Fast-tier sighting on PR #549 at 751e36d9 is the same shape: two threads the stop never counts down to, one of them the held thread by construction. The teardown boot's own extra thread beyond quiesce_last's usual five is the defect the issue already tracks, not this branch's teardown change. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...he-last-park-gave-up-on-one-thread-beside-the-held-one.md | 5 +++++ src/redlist.rs | 4 ++++ 2 files changed, 9 insertions(+) diff --git a/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md b/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md index 5a9246a9d9b..8ade75ff0fe 100644 --- a/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md +++ b/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md @@ -17,6 +17,11 @@ FAIL quiesce_wakes_on_the_last_park: the stop gave up on 2 thread(s) that never - PR #555 at `d2656765`. - PR #559 at `ac948e6a`. +`quiesce_wakes_on_the_last_teardown`'s boot shares the same shape, in the Fast +tier: PR #549 at `751e36d9`, `4 of 6 … 2010 ms of a 2010 ms budget over 2 +sweep(s)`. That boot normally counts 5 userland threads; the sixth is the +defect this issue is open for, not this PR's teardown change. + The earliest is PR #510 at `98e803cb`, recorded in `issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md`: `stop: 4 of 7 userland thread(s) stopped ... in 2010 ms of a 2010 ms budget`. diff --git a/src/redlist.rs b/src/redlist.rs index ec48d2a75f5..a9f473754e1 100644 --- a/src/redlist.rs +++ b/src/redlist.rs @@ -76,6 +76,10 @@ pub const DISABLED: &[Disabled] = &[ test: "quiesce_wakes_on_the_last_park", issue: "issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md", }, + Disabled { + test: "quiesce_wakes_on_the_last_teardown", + issue: "issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md", + }, Disabled { test: "sched_check_build", issue: "issues/build/the-pass-cost-gates-ci-sample-is-eight-days-stale-twice.md", From 0f28ea0ac460e0ebb4cd536c2e360b78ba4ea48f Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 16:24:26 +0200 Subject: [PATCH 16/19] kill_ends_every_wait kills a child only once the roster shows it parked Under the sleep mutation (`sys_nanosleep` parking uncancellably) the sleep arm still passed: its child exited 137, and the ceiling hang came only from the process-wait arm, whose waited process runs the same `nanosleep(u64::MAX)` through the same `sys_nanosleep`. The arm and the mutation were right; the kill was early. The parent killed on reading the child's `parked in ` marker, which the child writes before the syscall that parks, so a kill that lands while the child is still returning from that write is honoured at the write's `kernel_exit_to_user_check` and the named wait is never entered. The waited process of the process-wait arm had a whole second spawn's time to park, and hung as the mutation says it should. `spawn` now also polls the `SysCap` roster until the child's main thread is `SCHED_BLOCKED`, under a five-second hang guard. Between the marker's write and the named wait's syscall the child runs nothing else that parks: its code is paged from `/system`, which is the memory image. That is the filed issue's exit condition, so it is closed. `tests/testcases/system.toml` no longer counts the binaries that read the roster; this one makes five and the next landing moves it again. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-arm-can-pass-without-reaching-its-wait.md | 17 ------- tests/testcases/system.toml | 2 +- .../src/bin/kill_ends_every_wait.rs | 44 ++++++++++++++++--- 3 files changed, 40 insertions(+), 23 deletions(-) delete mode 100644 issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md diff --git a/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md b/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md deleted file mode 100644 index a7cde124dc4..00000000000 --- a/issues/kernel/a-kill-ends-every-wait-arm-can-pass-without-reaching-its-wait.md +++ /dev/null @@ -1,17 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-09-28 ---- - -# A `kill_ends_every_wait` arm can pass without reaching its wait - -`tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs` kills each child once -it reads the child's `parked in ` marker, which the child prints just -before the syscall that parks. A kill that lands between the marker and the -park ends the child at that syscall's exit boundary instead, and the arm passes -without the wait it names ever being killed. Nothing makes it reach it in every -run. - -Owner: orchestrator. Exit condition: the parent sees the child parked in the -named wait, by the kernel's word, before it kills it. diff --git a/tests/testcases/system.toml b/tests/testcases/system.toml index 4936ac577c0..d82230a57c0 100644 --- a/tests/testcases/system.toml +++ b/tests/testcases/system.toml @@ -31,7 +31,7 @@ syscap = ["rt"] # carried it would make vacuous. `run shutdown` does not use it: the applet asks # init through the `power` connector, and init has `logd` make the log whole # before it stops the machine. -# `roster` because four guest binaries read `SYS_SYSINFO`'s per-thread entries — +# `roster` because guest binaries read `SYS_SYSINFO`'s per-thread entries — # soundd's, for the idle-suspend certification, and their own, which arrive in # the same machine-wide answer and have no narrower question in the ABI. It is # also what `endowment_denied` narrows *away* to prove the refusal, so an estate diff --git a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs index bbf63d8bd5b..2a33e21458f 100644 --- a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs +++ b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs @@ -1,8 +1,9 @@ //! A thread killed inside any wait the kernel has for it leaves, and its //! process ends. //! -//! One child per wait, each killed after it says it is about to park: a futex, -//! a poll ring, a process's end, a thread's end and a sleep. A wait the kill +//! One child per wait, each killed once it has said it is about to park and the +//! kernel's roster shows its main thread parked: a futex, a poll ring, a +//! process's end, a thread's end and a sleep. A wait the kill //! cannot end keeps the child's last thread in its process for ever, and //! `wait` below never returns; the harness's deadline is what says so. //! `kill_while_blocked` holds the pipe, connection and accept waits, and @@ -12,11 +13,13 @@ use std::io::{Read, Write}; use std::os::toyos::process::{ChildExt, CommandExt}; use std::process::{Child, Command, Stdio}; use std::sync::atomic::AtomicU32; -use std::time::Duration; +use std::sync::OnceLock; +use std::time::{Duration, Instant}; -use toyos::endow::Endowments; +use toyos::endow::{Endowments, SYSCAP_LABEL}; use toyos::poller::Poller; use toyos::process::Process; +use toyos::syscap::SysCap; use toyos::AsHandle; use toyos_abi::syscall; use toyos_abi::RawHandle; @@ -29,6 +32,9 @@ const WAITED: &str = "waited"; /// `process::KILLED_EXIT_CODE`. const KILLED: i32 = 137; +/// `sched::payload::SCHED_BLOCKED`, the state column of the roster. +const BLOCKED: u8 = 2; + // `sleep` runs before `process-wait` and `thread-join`: both of those arms also // sleep underneath (the waited process's `nanosleep`, the joined thread's // `std::thread::sleep`), so a mutation that breaks sleep would otherwise surface @@ -64,7 +70,7 @@ fn test() { println!("kill_ends_every_wait: every wait a kill reached, it ended"); } -/// Spawn a child in `role` and read its marker: it is about to park. +/// Spawn a child in `role`, read its marker, and return once it is parked. fn spawn(role: &str, endow: Option<(String, u32)>) -> Child { let mut command = Command::new(SELF_PATH); command.arg(role).stdout(Stdio::piped()); @@ -79,9 +85,37 @@ fn spawn(role: &str, endow: Option<(String, u32)>) -> Child { line.push(byte[0]); } assert_eq!(String::from_utf8_lossy(&line), format!("parked in {role}"), "{role} never reached its wait"); + // The marker says only that the park is next: a kill landing before it ends + // the child at the marker's own syscall, and the wait the arm names is never + // killed. The bound is a hang guard, not a timing assumption. + let pid = child.id(); + assert_ne!(pid, 0, "{role}: the kernel no longer answers for the child"); + let give_up = Instant::now() + Duration::from_secs(5); + while !main_thread_parked(pid) { + assert!(Instant::now() < give_up, "{role}: the roster never showed the child parked"); + } child } +/// Whether the roster shows `pid`'s main thread parked; the `sysinfo` call is +/// the polling loop's preemption point. +fn main_thread_parked(pid: u32) -> bool { + const HEADER: usize = toyos::system::SYSINFO_HEADER_SIZE; + const ENTRY: usize = toyos::system::SYSINFO_ENTRY_SIZE; + static CAP: OnceLock = OnceLock::new(); + let cap = CAP.get_or_init(|| { + Endowments::get() + .take(SYSCAP_LABEL) + .expect("test-runner endows every binary it spawns a system capability") + }); + let mut buf = vec![0u8; HEADER + ENTRY * 256]; + let n = cap.roster(&mut buf); + assert!((HEADER..=buf.len()).contains(&n), "sysinfo answered {n}"); + buf[HEADER..n].chunks_exact(ENTRY).any(|entry| { + u32::from_le_bytes(entry[0..4].try_into().unwrap()) == pid && entry[9] == 0 && entry[8] == BLOCKED + }) +} + fn child(role: &str) -> ! { match role { "futex" => { From 5163e9bfdcf630887391e5cbc871912ec20a3535 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 16:38:02 +0200 Subject: [PATCH 17/19] kill_ends_every_wait's roster wait has no clock of its own A QEMU test's only clock is the harness's hang ceiling, so the five-second in-guest guard on the roster poll goes. The poll sleeps 10 ms between reads, as futex_wake_counts' wait_until_parked does, and prints `: waiting for the roster to show it parked` before it starts, so a ceiling red names where it stopped. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- .../toyos-rust-tests/src/bin/kill_ends_every_wait.rs | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs index 2a33e21458f..cad940de43c 100644 --- a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs +++ b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs @@ -14,7 +14,7 @@ use std::os::toyos::process::{ChildExt, CommandExt}; use std::process::{Child, Command, Stdio}; use std::sync::atomic::AtomicU32; use std::sync::OnceLock; -use std::time::{Duration, Instant}; +use std::time::Duration; use toyos::endow::{Endowments, SYSCAP_LABEL}; use toyos::poller::Poller; @@ -87,18 +87,17 @@ fn spawn(role: &str, endow: Option<(String, u32)>) -> Child { assert_eq!(String::from_utf8_lossy(&line), format!("parked in {role}"), "{role} never reached its wait"); // The marker says only that the park is next: a kill landing before it ends // the child at the marker's own syscall, and the wait the arm names is never - // killed. The bound is a hang guard, not a timing assumption. + // killed. Unbounded here: the harness's ceiling is the only clock. let pid = child.id(); assert_ne!(pid, 0, "{role}: the kernel no longer answers for the child"); - let give_up = Instant::now() + Duration::from_secs(5); + println!(" {role}: waiting for the roster to show it parked"); while !main_thread_parked(pid) { - assert!(Instant::now() < give_up, "{role}: the roster never showed the child parked"); + std::thread::sleep(Duration::from_millis(10)); } child } -/// Whether the roster shows `pid`'s main thread parked; the `sysinfo` call is -/// the polling loop's preemption point. +/// Whether the roster shows `pid`'s main thread parked. fn main_thread_parked(pid: u32) -> bool { const HEADER: usize = toyos::system::SYSINFO_HEADER_SIZE; const ENTRY: usize = toyos::system::SYSINFO_ENTRY_SIZE; From 82f917b9fe2c95e90a2c500eda94ba81045c064f Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 18:20:12 +0200 Subject: [PATCH 18/19] Round 8 review fixes: the roster poll fails by name, not by the harness ceiling, on a header over 256 entries or a main thread gone/zombied; file the missing roster-entry decoder as its own issue; name the teardown claim the quiesce issue leaves dark. - kill_ends_every_wait.rs asserts the roster header's entry count is at most the 256-entry buffer, and its poll panics by name when the child's main thread is absent from the roster or zombied, rather than spinning to the harness's ceiling. - issues/design-debt/toyos-abi-decodes-only-the-roster-header-not-its-entries.md: toyos-abi decodes the sysinfo header but not an entry, and four readers hand-spell its offsets 8 and 9 instead. - issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md: quiesce_wakes_on_the_last_teardown is the only guest check that a stop waits for a teardown in flight, named in both Exit and what no enabled test checks; the unproven sixth-thread attribution is removed. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-only-the-roster-header-not-its-entries.md | 32 ++++++++++++++ ...ve-up-on-one-thread-beside-the-held-one.md | 10 +++-- .../src/bin/kill_ends_every_wait.rs | 44 ++++++++++++++++--- 3 files changed, 76 insertions(+), 10 deletions(-) create mode 100644 issues/design-debt/toyos-abi-decodes-only-the-roster-header-not-its-entries.md diff --git a/issues/design-debt/toyos-abi-decodes-only-the-roster-header-not-its-entries.md b/issues/design-debt/toyos-abi-decodes-only-the-roster-header-not-its-entries.md new file mode 100644 index 00000000000..addae708100 --- /dev/null +++ b/issues/design-debt/toyos-abi-decodes-only-the-roster-header-not-its-entries.md @@ -0,0 +1,32 @@ +--- +status: open +kind: defect +opened: 2026-09-28 +--- + +# `toyos-abi` decodes only the roster header, not its entries + +`SysinfoHeader::decode` (`toyos-abi/src/syscall.rs:85`) is "the one spelling of +the header's offsets, so a reader takes a field rather than an index" — but +that one spelling stops at the header. `SYSINFO_ENTRY_SIZE` +(`toyos-abi/src/syscall.rs:103`) documents the entry's byte layout in prose +only, and every reader of an entry hand-spells its offsets instead of calling +a decoder: + +- `tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs:140-141` — + `entry[9] == 0`, `entry[8]`. +- `tests/toyos-rust-tests/src/bin/process_lifecycle.rs:258` — + `buf[pos + 9] != 0, buf[pos + 8]`. +- `tests/toyos-rust-tests/src/bin/abuse_thread_name.rs:62` — `entry[9]`. +- `userland/toybox/src/ps.rs:64-65` — `buf[pos + 8]`, `buf[pos + 9] != 0`. + +Four readers, each free to walk off by a field the way the header's own doc +comment warns against. `kernel/src/syscall/machine.rs:309-310` is the one +writer (`entry[8] = state; entry[9] = is_thread;`), so the four are already +one silent renumbering away from reading the wrong column. + +**Exit**: a `SysinfoEntry::decode` (or equivalent) in `toyos-abi`, the four +readers above converted to call it, and no roster-entry offset left +hand-spelled outside that one function. + +Owner: `toyos-abi/src/syscall.rs`, whoever next touches the roster ABI. diff --git a/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md b/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md index 8ade75ff0fe..ee2beaaa238 100644 --- a/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md +++ b/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md @@ -19,8 +19,7 @@ FAIL quiesce_wakes_on_the_last_park: the stop gave up on 2 thread(s) that never `quiesce_wakes_on_the_last_teardown`'s boot shares the same shape, in the Fast tier: PR #549 at `751e36d9`, `4 of 6 … 2010 ms of a 2010 ms budget over 2 -sweep(s)`. That boot normally counts 5 userland threads; the sixth is the -defect this issue is open for, not this PR's teardown change. +sweep(s)`. That boot normally counts 5 userland threads. The earliest is PR #510 at `98e803cb`, recorded in `issues/build/quiesce-wakes-on-the-last-park-lost-its-serial-ready-beside-other-guests.md`: @@ -54,7 +53,10 @@ enabled test checks any of these: - that a band, a park or an exit wakes the stop, rather than its deadline; - `in_flight == 0` with `begun > 0`; - the thread census; -- the `console-queue-at-the-stop` drain. +- the `console-queue-at-the-stop` drain; +- that the stop waits for a teardown in flight — + `quiesce_wakes_on_the_last_teardown`'s only claim, and the only enabled + guest check of it, disabled by this same issue. `quiesce_refuses_a_second_shutdown` stays green over a lost post. It judges the stop only by `stopped_the_machine`, so a stop that spends its budget and then @@ -66,6 +68,8 @@ finds everything stopped passes it. with its name, tid, cpu and scheduler state; - the mechanism it names fixed; - this test green in a Fast tier beside other guests; +- `quiesce_wakes_on_the_last_teardown` — the stop waits for a teardown in + flight — green in a Fast tier beside other guests; - an enabled guest test checking each claim listed above. Owner: the stop path, `kernel/src/quiesce.rs`; held by the orchestrator. diff --git a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs index cad940de43c..35f5aef7709 100644 --- a/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs +++ b/tests/toyos-rust-tests/src/bin/kill_ends_every_wait.rs @@ -35,6 +35,9 @@ const KILLED: i32 = 137; /// `sched::payload::SCHED_BLOCKED`, the state column of the roster. const BLOCKED: u8 = 2; +/// `sched::payload::SCHED_UNKNOWN`, the state a zombied thread's entry carries. +const ZOMBIE: u8 = 3; + // `sleep` runs before `process-wait` and `thread-join`: both of those arms also // sleep underneath (the waited process's `nanosleep`, the joined thread's // `std::thread::sleep`), so a mutation that breaks sleep would otherwise surface @@ -91,14 +94,30 @@ fn spawn(role: &str, endow: Option<(String, u32)>) -> Child { let pid = child.id(); assert_ne!(pid, 0, "{role}: the kernel no longer answers for the child"); println!(" {role}: waiting for the roster to show it parked"); - while !main_thread_parked(pid) { - std::thread::sleep(Duration::from_millis(10)); + loop { + match main_thread_status(pid) { + RosterStatus::Parked => break, + RosterStatus::NotParked => std::thread::sleep(Duration::from_millis(10)), + RosterStatus::Gone => { + panic!("{role}: the main thread is absent from the roster or zombied before the wait under test parked it") + } + } } child } -/// Whether the roster shows `pid`'s main thread parked. -fn main_thread_parked(pid: u32) -> bool { +/// What the roster says about `pid`'s main thread. +enum RosterStatus { + /// In the roster, blocked: parked in the wait under test. + Parked, + /// In the roster, not blocked yet. + NotParked, + /// Not in the roster, or in it as a zombie: no wait under test can still be ahead of it. + Gone, +} + +/// What the roster says about `pid`'s main thread. +fn main_thread_status(pid: u32) -> RosterStatus { const HEADER: usize = toyos::system::SYSINFO_HEADER_SIZE; const ENTRY: usize = toyos::system::SYSINFO_ENTRY_SIZE; static CAP: OnceLock = OnceLock::new(); @@ -110,9 +129,20 @@ fn main_thread_parked(pid: u32) -> bool { let mut buf = vec![0u8; HEADER + ENTRY * 256]; let n = cap.roster(&mut buf); assert!((HEADER..=buf.len()).contains(&n), "sysinfo answered {n}"); - buf[HEADER..n].chunks_exact(ENTRY).any(|entry| { - u32::from_le_bytes(entry[0..4].try_into().unwrap()) == pid && entry[9] == 0 && entry[8] == BLOCKED - }) + let header = toyos_abi::syscall::SysinfoHeader::decode(buf[..HEADER].try_into().unwrap()); + assert!( + header.entries as usize <= 256, + "the roster holds {} threads, more than the 256-entry buffer this test reads can carry", + header.entries + ); + buf[HEADER..n] + .chunks_exact(ENTRY) + .find(|entry| u32::from_le_bytes(entry[0..4].try_into().unwrap()) == pid && entry[9] == 0) + .map_or(RosterStatus::Gone, |entry| match entry[8] { + BLOCKED => RosterStatus::Parked, + ZOMBIE => RosterStatus::Gone, + _ => RosterStatus::NotParked, + }) } fn child(role: &str) -> ! { From 4c2e052ec1ca5542f7a0d1c2681402dfd34658fb Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 19:18:54 +0200 Subject: [PATCH 19/19] Round of #549 review: delete four stale lists the merge rewrote instead of removing The merge onto 807f4561 rewrote comments that named specific tests or actuators where a canonical list already exists elsewhere (kernel/src/actuator.rs, tests/common/power.rs, CARRIES in tests/toyos.rs): toyos-quiesce's LAST_THREAD doc, quiescelastcase/system.toml's header, quiesce_last.rs's module doc, and tests/toyos.rs's RUST_SKIP entry all drop their repeated lists rather than carry them forward again. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- tests/quiescelastcase/system.toml | 5 ++--- tests/toyos-rust-tests/src/bin/quiesce_last.rs | 9 ++++----- tests/toyos.rs | 2 -- toyos-quiesce/src/lib.rs | 7 +++---- 4 files changed, 9 insertions(+), 14 deletions(-) diff --git a/tests/quiescelastcase/system.toml b/tests/quiescelastcase/system.toml index 96942e7e060..992225957c2 100644 --- a/tests/quiescelastcase/system.toml +++ b/tests/quiescelastcase/system.toml @@ -1,6 +1,5 @@ -# The boot `quiesce_wakes_on_the_last_park` and `quiesce_wakes_on_the_last_teardown` -# judge: its one job starts the thread and the child the kernel may hold and -# reboots. +# This boot's one job starts the thread and the child the kernel may hold and +# reboots; it asserts nothing itself. [boot] start = ["logd", "test-runner"] diff --git a/tests/toyos-rust-tests/src/bin/quiesce_last.rs b/tests/toyos-rust-tests/src/bin/quiesce_last.rs index 8d9b377b8f5..6500834ae55 100644 --- a/tests/toyos-rust-tests/src/bin/quiesce_last.rs +++ b/tests/toyos-rust-tests/src/bin/quiesce_last.rs @@ -1,15 +1,14 @@ -//! Two threads the kernel's `quiesce-last-*` actuators can hold, and a reboot. +//! The threads the kernel's `quiesce-last-*` actuators can hold, and a reboot. //! -//! Both carry [`toyos_quiesce::LAST_THREAD`]'s name. One parks for longer than +//! They carry [`toyos_quiesce::LAST_THREAD`]'s name. One parks for longer than //! any stop's budget, and one is the only thread of a child that exits at once. //! `quiesce-last-park` holds the first inside its `SYS_NANOSLEEP`, and //! `quiesce-last-teardown` the second between leaving its process and tearing //! it down. The kernel holds the reset itself until one of them is held, so //! this program orders nothing. //! -//! Nothing here asserts: `common::power::quiesce_wakes_on_the_last_park` and -//! `quiesce_wakes_on_the_last_teardown` read the kernel's hold line and its -//! `stop:` record. +//! Nothing here asserts: the kernel's hold line and its `stop:` record are +//! read elsewhere. use std::process::Command; use std::time::Duration; diff --git a/tests/toyos.rs b/tests/toyos.rs index 03be1161a5c..f10a8365c1e 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -178,8 +178,6 @@ const RUST_SKIP: &[&str] = &[ // `quiesce_refuses_a_second_shutdown` runs it. "quiesce_twice", // The same, and its verdict is the stop record of a boot staged around it. - // `quiesce_wakes_on_the_last_park` and `quiesce_wakes_on_the_last_teardown` - // run it. "quiesce_last", // The same, and its verdict is the log volume the stop leaves. // `quiesce_leaves_the_volume_whole` runs it. diff --git a/toyos-quiesce/src/lib.rs b/toyos-quiesce/src/lib.rs index 426a6d8e8ca..545e4925931 100644 --- a/toyos-quiesce/src/lib.rs +++ b/toyos-quiesce/src/lib.rs @@ -67,10 +67,9 @@ impl Sweep { } } -/// The thread the kernel's `quiesce-last-park` and `quiesce-last-teardown` -/// actuators hold, by the name its program gives it: held until it is the one -/// thread the stop still waits on, so the transition it makes next is the -/// stop's last. +/// The thread the kernel's quiesce-last actuators hold, by the name its +/// program gives it: held until it is the one thread the stop still waits +/// on, so the transition it makes next is the stop's last. pub const LAST_THREAD: &str = "quiesce-last"; /// What a line of kernel log carrying a [`Record`] begins with.