From 81c8239e3c1c511ce561aecc3522fc5227e0c32d Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 05:54:34 +0200 Subject: [PATCH 01/54] storage: fsd serves DATA, the log and the boot volume; the kernel's NVMe and FAT go (work in progress) A checkpoint of the file-server branch before main is merged in: the kernel's NVMe driver, page cache, read-write bcachefs adapter and FAT adapter are deleted; fsd serves each role's directories as capabilities init hands out; spawn and dlopen take an image a served file was read into. Tests are still being moved onto the servers. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- kernel/src/actuator.rs | 27 - kernel/src/bcachefs_adapter.rs | 514 +------ kernel/src/block.rs | 56 +- kernel/src/drivers/hda.rs | 2 +- kernel/src/drivers/mod.rs | 1 - kernel/src/drivers/nvme.rs | 908 ------------ kernel/src/drivers/virtio_gpu.rs | 8 +- kernel/src/drivers/virtio_sound.rs | 2 +- kernel/src/drivers/xhci/mod.rs | 14 + kernel/src/drivers/xhci/wait/boot.rs | 7 + kernel/src/fat32_adapter.rs | 1276 ----------------- kernel/src/file_backing.rs | 231 ++- kernel/src/file_cache.rs | 6 - kernel/src/gpt.rs | 235 ++- kernel/src/inventory.rs | 9 +- kernel/src/leak_selftest.rs | 37 +- kernel/src/loader/mod.rs | 40 +- kernel/src/main.rs | 115 +- kernel/src/nvme_gate.rs | 45 - kernel/src/object/ops.rs | 5 - kernel/src/page_cache.rs | 596 -------- kernel/src/pcidev/mod.rs | 2 +- kernel/src/quiesce.rs | 12 - kernel/src/syscall/dispatch.rs | 24 +- kernel/src/syscall/machine.rs | 1 - kernel/src/syscall/proc.rs | 3 +- kernel/src/syscall/vm.rs | 57 +- kernel/src/time.rs | 6 - kernel/src/user_ptr.rs | 4 +- kernel/src/vfs.rs | 11 +- kernel/src/writeback.rs | 5 - src/build.rs | 7 + src/image.rs | 10 +- src/kernelkeys.rs | 15 - src/qemu.rs | 2 +- src/sourcegate.rs | 1 - system.toml | 19 +- tests/blockdcase/system.toml | 33 +- tests/common/iommu.rs | 22 +- tests/common/power.rs | 8 +- tests/common/qemu.rs | 11 +- tests/common/storage.rs | 354 +---- tests/common/swap.rs | 11 +- tests/desktopaudiocase/system.toml | 19 +- tests/desktopcase/system.toml | 19 +- tests/doomcase/system.toml | 19 +- tests/doommusiccase/system.toml | 19 +- tests/e1000case/system.toml | 19 +- tests/e1000leasecase/system.toml | 19 +- tests/e1000talkcase/system.toml | 19 +- tests/flrswapcase/system.toml | 19 +- tests/inspectcase/system.toml | 19 +- tests/jobcase/system.toml | 19 +- tests/jobdeadlinecase/system.toml | 19 +- tests/lancase/system.toml | 19 +- tests/lanicscase/system.toml | 19 +- tests/lanleasecase/system.toml | 19 +- tests/lantalkcase/system.toml | 19 +- tests/latencycase/system.toml | 19 +- tests/layoutcase/system.toml | 19 +- tests/logflushcase/system.toml | 19 +- tests/logkeepcase/system.toml | 19 +- tests/logrotatecase/system.toml | 19 +- tests/logstallcase/system.toml | 19 +- tests/logstreamcase/system.toml | 19 +- tests/logstreame1000case/system.toml | 19 +- tests/metalcase/system.toml | 19 +- tests/metaldevicecase/system.toml | 19 +- tests/netcase/system.toml | 19 +- tests/partclaimcase/system.toml | 19 +- tests/pkgcase/system.toml | 19 +- tests/quiescecase/system.toml | 19 +- tests/quiescelastcase/system.toml | 19 +- tests/quiescetwicecase/system.toml | 19 +- tests/sshdcase/system.toml | 19 +- tests/swapcase/system.toml | 19 +- tests/test-durations | 2 - tests/testcases/system.toml | 19 +- tests/toolkitcase/system.toml | 19 +- .../src/bin/abuse_cwd_growth.rs | 2 +- .../src/bin/abuse_elf_loader.rs | 40 +- .../src/bin/abuse_elf_segments.rs | 20 +- .../src/bin/abuse_handle_table.rs | 2 + .../src/bin/abuse_page_straddle.rs | 2 + .../src/bin/abuse_spawn_argv.rs | 2 + .../src/bin/device_claim_lifetime.rs | 2 + .../src/bin/handle_kill_policy.rs | 2 + .../src/bin/handle_lifetime.rs | 2 +- .../src/bin/handle_transfer.rs | 6 +- .../src/bin/home_fsync_budget.rs | 30 - .../src/bin/pkg_launch_gbae.rs | 6 +- .../src/bin/so_cache_policy.rs | 13 +- tests/toyos-rust-tests/src/bin/spawn_cwd.rs | 24 +- tests/toyos.rs | 15 - tests/updatecase/system.toml | 19 +- toyos-abi/src/inventory.rs | 45 +- toyos-abi/src/syscall.rs | 42 +- toyos-blockring/src/wire.rs | 45 + toyos-inspect/src/dev.rs | 10 +- toyos-manifest/src/lib.rs | 69 + toyos/src/endow.rs | 13 + toyos/src/fs.rs | 581 ++++++++ toyos/src/lib.rs | 1 + userland/Cargo.lock | 19 + userland/Cargo.toml | 1 + userland/blockd/src/lib.rs | 2 +- userland/blockd/src/main.rs | 128 +- userland/blockd/src/session.rs | 17 + userland/fsd/Cargo.toml | 24 + userland/fsd/src/absent.rs | 99 ++ userland/fsd/src/cache.rs | 343 +++++ userland/fsd/src/data.rs | 808 +++++++++++ userland/fsd/src/disk.rs | 240 ++++ userland/fsd/src/fat.rs | 392 +++++ userland/fsd/src/lib.rs | 21 + userland/fsd/src/main.rs | 777 ++++++++++ userland/fsd/src/resolve.rs | 195 +++ userland/fsd/src/volume.rs | 114 ++ userland/init/Cargo.toml | 2 + userland/init/src/main.rs | 465 +++++- 120 files changed, 5641 insertions(+), 4418 deletions(-) delete mode 100644 kernel/src/drivers/nvme.rs delete mode 100644 kernel/src/fat32_adapter.rs delete mode 100644 kernel/src/nvme_gate.rs delete mode 100644 kernel/src/page_cache.rs delete mode 100644 tests/toyos-rust-tests/src/bin/home_fsync_budget.rs create mode 100644 toyos/src/fs.rs create mode 100644 userland/fsd/Cargo.toml create mode 100644 userland/fsd/src/absent.rs create mode 100644 userland/fsd/src/cache.rs create mode 100644 userland/fsd/src/data.rs create mode 100644 userland/fsd/src/disk.rs create mode 100644 userland/fsd/src/fat.rs create mode 100644 userland/fsd/src/lib.rs create mode 100644 userland/fsd/src/main.rs create mode 100644 userland/fsd/src/resolve.rs create mode 100644 userland/fsd/src/volume.rs diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 805501daf4a..f4331ac4380 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -104,17 +104,9 @@ actuators! { /// Read and write every USB disk carrying the gate's stamp in block 0; the stamp, not the parameter, picks the disk, since the boot stick shares the bus. usb_storage_gate = "usb-storage-gate"; - /// Ask the NVMe disk for a block with the caller's operation budget already spent. - nvme_spent_budget = "nvme-spent-budget"; - /// Refuse the first two FAT-1 mirror writes of a write-back drain flush; two, not one, because the retry ladder parks only at attempt 2. - fat_mirror_write_refuse = "fat-mirror-write-refuse"; - /// Refuse the FAT-1 mirror write of a write-back drain flush as a budget expiry on the thread running the shutdown, which parks it in `block::between_attempts`; armed beside `writeback-stall`, which is what leaves a closed file's flush for that drain to find. - quiesce_drain_refuse = "quiesce-drain-refuse"; - /// Refuse the active-FAT write of a `SYS_FSYNC` flush as a budget expiry once the machine is stopping, after its mirror is written, so the stop's second stage reaches a caller parked in `block::between_attempts` over two FATs that disagree. - quiesce_fsync_refuse = "quiesce-fsync-refuse"; /// Hold the thread named `toyos_quiesce::LAST_THREAD` inside `SYS_NANOSLEEP`, and the shutdown until it is held there, until the stop waits on it alone: its park is then the stop's last transition. quiesce_last_park = "quiesce-last-park"; @@ -125,8 +117,6 @@ actuators! { /// Serve a blocked-task dump from the shutdown once its first stage has stopped the machine: the report Ctrl+Alt+D gives on a shutdown stuck in its stop. quiesce_dump = "quiesce-dump"; - /// Refuse the second directory-entry write of the file `writeback_durability` stages for the retry gate — the first is that file's own seed being made durable — as a budget expiry, so a flush fails at its metadata write with its pages already written and settled. - fat_flush_meta_refuse = "fat-flush-meta-refuse"; /// Take the page a shrink has just read off the device, in the window between that read and the lock that spends it — one other CPU's CLOCK sweep, which needs no VFS lock and so runs there. resize_evict_window = "resize-evict-window"; @@ -253,8 +243,6 @@ actuators! { /// expired. fsync_deadman_now = "fsync-deadman-now"; - /// Skip one NVMe completion wait so a submitted command goes unanswered. - nvme_command_silent = "nvme-command-silent"; /// Under-deliver one READ(10) data phase so the byte counts disagree. usb_short_read = "usb-short-read"; @@ -358,8 +346,6 @@ actuators! { /// Deliver the i8042 vector once at arming with no byte behind it — the arming edge, staged. i8042_arm_edge = "i8042-arm-edge"; - /// Blind init's read of CSTS.RDY, staging an NVMe controller that never answers. - nvme_rdy_stuck = "nvme-rdy-stuck"; /// Blind init's read of the reset handshake, staging virtio devices that never answer; the console — the staged boot's capture channel — is spared. virtio_reset_stuck = "virtio-reset-stuck"; @@ -458,11 +444,7 @@ actuators! { /// Stop the boot dead in phase 3, interrupts off, before any log drain. pre_idle_wedge = "pre-idle-wedge"; - /// Fail every re-read of a page of a file through `FatBacking`. - fat_backing_read_fails = "fat-backing-read-fails"; - /// Fail every filesystem read of the mounted boot volume, at the `Fat32` layer rather than `FatBacking`'s page read. - fat_boot_reads_fail = "fat-boot-reads-fail"; /// Leave the NVMe controller out of the IOMMU's root table. iommu_context_absent = "iommu-context-absent"; @@ -514,14 +496,7 @@ actuators! { /// Run the revoked-backing controls (`/tmp` and `/home`) after mount. revoked_backing_selftest = "revoked-backing-selftest"; - /// Wrap the metadata cache's device in a read-fault injector and run the un-index control after mount. - pc_unbind_selftest = "pc-unbind-selftest"; - /// Refuse every read of device block 0 of each NVMe disk — its protective - /// MBR and GPT header — once every mount has been made, so a partition - /// claim meets a disk that does not answer a read of its table. Judged by - /// `partition_claim_gives_up`. - partclaim_table_unanswered = "partclaim-table-unanswered"; /// Reopen init by pid once it is spawned, the way `SYS_PROCESS_OPEN` does. process_reopen_selftest = "process-reopen-selftest"; @@ -529,8 +504,6 @@ actuators! { /// Offer the block layer a second device claiming a registered `DeviceId`, and report what it did with it. block_duplicate_id = "block-duplicate-id"; - /// Write through a page cache over a view that does not start at block 0, and say where on the device the bytes landed. - page_cache_partition_offset = "pc-partition-offset"; /// Arm the watchdog at seconds rather than minutes, so a guest reaches the reset. watchdog_fast = "tco-fast"; diff --git a/kernel/src/bcachefs_adapter.rs b/kernel/src/bcachefs_adapter.rs index 6bad33afc5e..6d492877567 100644 --- a/kernel/src/bcachefs_adapter.rs +++ b/kernel/src/bcachefs_adapter.rs @@ -1,82 +1,21 @@ +//! ROOT as the VFS reads it: the signed image, a read-only bcachefs volume in +//! memory. The only filesystem this kernel mounts; every writable one is a +//! file server's (`/system/bin/fsd`). + use alloc::collections::BTreeMap; use alloc::string::String; -use alloc::sync::{Arc, Weak}; +use alloc::sync::Arc; use alloc::vec::Vec; -use crate::hasher::HashMap; -use bcachefs::{BlockIO, BlockBuf, BlockNum, DeviceError, FsError, Mounted, ReadWrite, ReadOnly, Formatted, Extent}; -use crate::file_backing::{FileBacking, FileBlocks, NvmeBacking, ReadOnlyBacking}; +use bcachefs::{FsError, Mounted, ReadOnly}; +use crate::file_backing::{FileBacking, ReadOnlyBacking}; use crate::mm::PAGE_BYTES; -use crate::file_cache::{self, FileId, Residency}; -use crate::fs_rename::{self, Committed, ReplaceRename}; -use crate::page_cache; +use crate::file_cache::{self, FileId}; use crate::rootfs::MemoryImage; use toyos_abi::syscall::SyscallError; use crate::vfs::FileSystem; -/// Nothing here names a device: the cache it holds does, so a filesystem is -/// mounted on the partition its cache was opened over and on no other. -#[derive(Clone)] -pub struct PageCacheBlockIO(Arc); - -impl PageCacheBlockIO { - pub fn new(cache: Arc) -> Self { - Self(cache) - } -} - -/// The one conversion between the kernel's block verdict and the crate's; -/// keeping the retry discriminant is the point — a `BudgetExpired` collapsed -/// into the device's own word turned into permanent write loss once (F9). -/// `classify` is the crate's only reachable constructor, so this answer *is* -/// the discriminant and no site downstream holds a variant to collapse into. -impl bcachefs::TransferError for crate::block::BlockError { - fn refused_before_attempt(&self) -> bool { - matches!(self, crate::block::BlockError::BudgetExpired) - } -} - -impl From for DeviceError { - fn from(e: crate::block::BlockError) -> Self { - DeviceError::classify(&e) - } -} - -/// Errors propagate unchanged; nothing here invents a value for a refused transfer. -impl BlockIO for PageCacheBlockIO { - fn read_block(&self, block: BlockNum, buf: &mut BlockBuf) -> Result<(), DeviceError> { - let mut guard = self.0.lock(); - let page = guard.read(block.raw()).map_err(DeviceError::from)?; - buf.as_bytes_mut().copy_from_slice(page); - Ok(()) - } - - fn write_block(&self, block: BlockNum, buf: &BlockBuf) -> Result<(), DeviceError> { - let mut guard = self.0.lock(); - let page = guard.write_new(block.raw()).map_err(DeviceError::from)?; - page.copy_from_slice(buf.as_bytes()); - Ok(()) - } - - fn block_count(&self) -> u64 { - self.0.partition().block_count() - } - - fn sync(&self) -> Result<(), DeviceError> { - self.0.lock().sync().map_err(DeviceError::from) - } -} - -/// A budget refusal is `WouldBlock` — the word every retry loop keys on — and -/// only the device's own word is `Io`; `kernel/CLAUDE.md` states the rule. -fn as_device_refusal(e: DeviceError) -> SyscallError { - match e { - DeviceError::Failed(_) => SyscallError::Io, - DeviceError::Refused(_) => SyscallError::WouldBlock, - } -} - /// Exhaustive match: corruption maps to `Io`, never `NotFound` — a btree that won't decode isn't "not there". fn as_syscall_error(err: &FsError) -> SyscallError { match err { @@ -85,10 +24,10 @@ fn as_syscall_error(err: &FsError) -> SyscallError { SyscallError::ResourceExhausted } FsError::NameTooLong { .. } => SyscallError::InvalidArgument, - FsError::DeviceRead(_, e) | FsError::DeviceWrite(_, e) | FsError::DeviceSync(e) => { - as_device_refusal(*e) - } - FsError::BadMagic { .. } + FsError::DeviceRead(..) + | FsError::DeviceWrite(..) + | FsError::DeviceSync(_) + | FsError::BadMagic { .. } | FsError::UnsupportedVersion(_) | FsError::ChecksumMismatch { .. } | FsError::CorruptedKey(_) @@ -121,281 +60,6 @@ fn present(op: &str, name: &str, result: Result, FsError>) -> Resul mapped(op, name, result)?.ok_or(SyscallError::NotFound) } -/// Per-open-file cached resolution state. -struct OpenFileInfo { - name: String, - blocks: Arc, -} - -/// VFS adapter for read-write bcachefs over one cached partition. -pub struct BcacheFsAdapter { - fs: Mounted, - /// The mount's own cache: file data bypasses its slots, not its flush debt. - cache: Arc, - open_files: HashMap, - name_to_id: BTreeMap, - /// The `FileBlocks` every backing for a name shares; keyed by name because - /// `open_backing` hands one out without opening a file at all. `Weak` so the - /// entry costs nothing once the last backing drops. - blocks: BTreeMap>, -} - -impl BcacheFsAdapter { - pub fn new(fs: Mounted, cache: Arc) -> Self { - Self { - fs, - cache, - open_files: HashMap::default(), - name_to_id: BTreeMap::new(), - blocks: BTreeMap::new(), - } - } - - /// The cell every backing for `name` reads through; made from `extents` on first use. - fn blocks_for(&mut self, name: &str, extents: Vec) -> Arc { - // Swept here, not on a timer: this is the only place the map grows. - self.blocks.retain(|_, weak| weak.strong_count() > 0); - - if let Some(live) = self.blocks.get(name).and_then(Weak::upgrade) { - return live; - } - let blocks = FileBlocks::new(extents); - self.blocks.insert(String::from(name), Arc::downgrade(&blocks)); - blocks - } - - /// Give up every backing that reads `name`'s blocks; call before the blocks are - /// reused, or the next file's backing reads its data. - fn revoke(&mut self, name: &str) { - if let Some(blocks) = self.blocks.remove(name).as_ref().and_then(Weak::upgrade) { - blocks.revoke(); - } - } -} - -/// `bcachefs::Mounted::rename` replaces atomically and resolves the source -/// before it touches the tree, so the move commits before the displaced -/// destination's in-memory state is freed: a rename that fails to find its -/// source frees nothing. -impl ReplaceRename for BcacheFsAdapter { - type Displaced = Option; - - fn source_present(&mut self, old: &str) -> Result { - if self.name_to_id.contains_key(old) { - return Ok(true); - } - Ok(mapped("exists", old, self.fs.file_mtime(old))?.is_some()) - } - - fn same_object(&mut self, old: &str, new: &str) -> Result { - // bcachefs hashes the exact name; equal strings are the one entry. - Ok(old == new) - } - - fn commit(&mut self, old: &str, new: &str) -> Result>, SyscallError> { - // Capture the displaced destination's in-memory id but free nothing: the - // backend move replaces the tree entry atomically, and only its success - // licenses freeing what the old destination held. - let displaced = self.name_to_id.get(new).copied(); - mapped("rename", old, self.fs.rename(old, new))?; - Ok(Committed::new(displaced)) - } - - fn release( - &mut self, - old: &str, - new: &str, - committed: Committed>, - ) -> Result<(), SyscallError> { - if let Some(target_id) = committed.into_displaced() { - if file_cache::mark_deleted(target_id) == Residency::Gone { - self.open_files.remove(&target_id); - } - self.name_to_id.remove(new); - } - // The old destination's backings read blocks the move has reassigned. - self.revoke(new); - if let Some(file_id) = self.name_to_id.remove(old) { - self.name_to_id.insert(String::from(new), file_id); - if let Some(info) = self.open_files.get_mut(&file_id) { - info.name = String::from(new); - } - } - if let Some(blocks) = self.blocks.remove(old) { - self.blocks.insert(String::from(new), blocks); - } - Ok(()) - } -} - -impl FileSystem for BcacheFsAdapter { - fn list(&mut self, dir: &str, limit: usize) -> Result, SyscallError> { - // An empty listing on error would be a lie indistinguishable from an empty directory. - mapped("list", dir, self.fs.list(limit, &|name| crate::vfs::under_directory(name, dir))) - } - - fn is_dir(&mut self, dir: &str) -> Result { - mapped("is_dir", dir, self.fs.is_dir(dir)) - } - - fn file_mtime(&mut self, name: &str) -> Result { - present("file_mtime", name, self.fs.file_mtime(name)) - } - - fn read_link(&mut self, name: &str) -> Result, SyscallError> { - mapped("read_link", name, self.fs.read_link(name, MAX_LINK_TARGET)) - } - - fn open_file(&mut self, name: &str) -> Result<(FileId, Option>), SyscallError> { - if let Some(&file_id) = self.name_to_id.get(name) { - let held = file_cache::open(file_id); - let info = self.open_files.get(&file_id).ok_or(SyscallError::NotFound)?; - let backing = Arc::new(NvmeBacking::new( - Arc::clone(&self.cache), - Arc::clone(&info.blocks), - file_cache::size(file_id), - )); - held.commit(); - return Ok((file_id, Some(backing))); - } - - let (extents, size) = present("open", name, self.fs.file_extents(name))?; - let blocks = self.blocks_for(name, extents); - let file_id = file_cache::create_file(true); // evictable - file_cache::set_size(file_id, size); - - self.name_to_id.insert(String::from(name), file_id); - self.open_files.insert(file_id, OpenFileInfo { - name: String::from(name), - blocks: Arc::clone(&blocks), - }); - - Ok((file_id, Some(Arc::new(NvmeBacking::new(Arc::clone(&self.cache), blocks, size))))) - } - - fn create(&mut self, name: &str, mtime: u64) -> Result { - if let Some(&file_id) = self.name_to_id.get(name) { - return Ok(file_id); - } - - // `Mounted::create` frees whatever answered to this name; revoke first. - self.revoke(name); - mapped("create", name, self.fs.create(name, &[], mtime))?; - - let file_id = file_cache::create_file(true); - self.name_to_id.insert(String::from(name), file_id); - let blocks = self.blocks_for(name, Vec::new()); - self.open_files.insert(file_id, OpenFileInfo { - name: String::from(name), - blocks, - }); - Ok(file_id) - } - - fn close_file(&mut self, file_id: FileId) { - if let Some(info) = self.open_files.remove(&file_id) { - // Only while the name is still this file's: `delete` and a rename leave a pinned file holding a name its replacement answers to. - if self.name_to_id.get(&info.name) == Some(&file_id) { - self.name_to_id.remove(&info.name); - } - } - } - - fn delete(&mut self, name: &str) -> Result<(), SyscallError> { - if let Some(&file_id) = self.name_to_id.get(name) { - if file_cache::mark_deleted(file_id) == Residency::Gone { - self.open_files.remove(&file_id); - } - self.name_to_id.remove(name); - } - self.revoke(name); - if mapped("delete", name, self.fs.delete(name))? { - Ok(()) - } else { - Err(SyscallError::NotFound) - } - } - - fn rename(&mut self, old: &str, new: &str) -> Result<(), SyscallError> { - fs_rename::replace_rename(self, old, new) - } - - // `NotSupported`, not a refusal: no directory representation here, so the - // VFS carries created directories itself. - fn create_dir(&mut self, _name: &str) -> Result<(), SyscallError> { - Err(SyscallError::NotSupported) - } - - fn remove_dir(&mut self, _name: &str) -> Result<(), SyscallError> { - Err(SyscallError::NotSupported) - } - - fn write_page(&mut self, file_id: FileId, page_idx: u32, data: &[u8; PAGE_BYTES]) -> Result<(), SyscallError> { - let info = self.open_files.get(&file_id).ok_or(SyscallError::NotFound)?; - let name = info.name.clone(); - let blocks = Arc::clone(&info.blocks); - let block = blocks - .with(|extents| self.fs.resolve_or_alloc_block(extents, page_idx)) - .ok_or(SyscallError::NotFound)?; - let block = mapped("block allocation", &name, block)?; - self.cache.raw_write(block, data).map_err(|e| { - log!("bcachefs: write of block {block} for '{name}' refused: {e:?}"); - as_device_refusal(DeviceError::from(e)) - }) - } - - /// Same order as [`BcacheFsAdapter::update_metadata`]'s own trim, for the - /// same reason: the shortened list is recorded before the blocks are freed. - fn truncate_to(&mut self, file_id: FileId, size: u64, mtime: u64) -> Result<(), SyscallError> { - let info = self.open_files.get(&file_id).ok_or(SyscallError::NotFound)?; - let name = info.name.clone(); - let blocks = Arc::clone(&info.blocks); - let dropped = blocks.truncate_to_blocks(size.div_ceil(crate::mm::PAGE_SIZE)); - if dropped.is_empty() { - return Ok(()); - } - let extents = blocks.with(|extents| extents.clone()).ok_or(SyscallError::NotFound)?; - mapped("truncate_to", &name, self.fs.update_metadata(&name, &extents, size, mtime))?; - mapped("free of a shrunk tail", &name, self.fs.free_extents(&dropped)) - } - - fn update_metadata(&mut self, file_id: FileId, size: u64, mtime: u64) -> Result<(), SyscallError> { - let info = self.open_files.get(&file_id).ok_or(SyscallError::NotFound)?; - let name = info.name.clone(); - let blocks = Arc::clone(&info.blocks); - // A shrink gives the dropped tail up — record the shortened list - // first, free second, so an error between the two leaks blocks rather - // than leaving the entry naming freed ones. Only an Err is ordered by - // this: nothing flushes until sync(), so power loss is not. - let dropped = blocks.truncate_to_blocks(size.div_ceil(crate::mm::PAGE_SIZE)); - let extents = blocks.with(|extents| extents.clone()).ok_or(SyscallError::NotFound)?; - mapped("update_metadata", &name, self.fs.update_metadata(&name, &extents, size, mtime))?; - if !dropped.is_empty() { - mapped("free of a shrunk tail", &name, self.fs.free_extents(&dropped))?; - } - Ok(()) - } - - fn create_symlink(&mut self, name: &str, target: &str) -> Result<(), SyscallError> { - // As `create`: displaces whatever answered to this name; revoke first. - self.revoke(name); - mapped("create_symlink", name, self.fs.create_symlink(name, target)) - } - - fn sync(&mut self) -> Result<(), SyscallError> { - mapped("sync", "/", self.fs.sync()) - } - - fn open_backing(&mut self, name: &str) -> Result, SyscallError> { - let (extents, size) = present("open_backing", name, self.fs.file_extents(name))?; - let blocks = self.blocks_for(name, extents); - Ok(Arc::new(NvmeBacking::new(Arc::clone(&self.cache), blocks, size))) - } - - fn cached_file_id(&mut self, name: &str) -> Option { - self.name_to_id.get(name).copied() - } -} /// VFS adapter for ROOT, a read-only bcachefs volume in memory. /// @@ -508,157 +172,3 @@ impl FileSystem for ReadOnlyBcacheFsAdapter { } } -/// Format a new bcachefs filesystem on the partition `cache` serves. -/// -/// Destroys everything on it; only [`probe`] may call it, and only on [`Storage::Designated`]. -fn format(cache: &Arc) -> Option> { - match Formatted::format(PageCacheBlockIO::new(Arc::clone(cache))) { - Ok(fs) => Some(fs.mount()), - Err(err) => { - // A half-written volume is not one to mount; `open_data` answers `Volatile`. - log!("storage: formatting the designated device failed: {:?}", err); - None - } - } -} - -/// Mount an existing bcachefs filesystem on the partition `cache` serves. -fn mount(cache: &Arc) -> Result, FsError> { - Mounted::::open(PageCacheBlockIO::new(Arc::clone(cache))) -} - -/// What the machine's block device is, as far as we are entitled to care. -/// -/// No default that writes: `Foreign` covers both someone else's disk and a blank one. -pub enum Storage { - /// A ToyOS volume, mounted read-write, identified by its own superblock. - Ours(Mounted), - /// A volume that does not mount and that no refusal says is another's: a - /// superblock of ours that broke, or a device that would not answer. - /// Never written to, and never stood in for. - Unmountable, - /// Carries a designation stamp naming its own size: consent to destroy what is here. - Designated, - /// Anything else. Never written to, under any circumstances. - Foreign, -} - -/// Decide what the device is, from its superblock and block 0. -/// -/// A failed mount is not consent. Only [`FsError::disowns_volume`] lets the -/// stamp be asked for; every other refusal is a volume of ours that did not -/// mount, or a device that did not say. The classification lives on the error -/// type in the bcachefs crate, exhaustive over its variants, so the kernel -/// cannot widen "not ours" by drifting from it. -/// -/// Two residues stay here, unfixed: a volume of ours whose block 0 and backup -/// both lost the magic reads exactly like a disk that was never ours, because -/// nothing on disk still says otherwise; and a designation stamp at block 0 -/// sitting over a backup superblock that carries the magic but fails its own -/// check reads `Unmountable` rather than `Designated`, so the stamp's consent -/// goes unread. Both fail toward no write rather than toward a false format. -pub fn probe(cache: &Arc) -> Storage { - match mount(cache) { - Ok(fs) => { - log!("storage: mounted the ToyOS volume at block 0"); - return Storage::Ours(fs); - } - Err(err) if err.disowns_volume() => {} - Err(err) => { - log!( - "storage: the DATA volume does not mount, and nothing says it is another's: \ - {:?} — nothing will be written to it", - err - ); - return Storage::Unmountable; - } - } - if designated(cache) { - log!("storage: block 0 designates this device for ToyOS — formatting it"); - return Storage::Designated; - } - log!( - "storage: no ToyOS volume and no designation stamp at block 0 — this disk is not \ - ours and nothing will be written to it" - ); - Storage::Foreign -} - -/// Whether block 0 carries a designation stamp for a device of *this* size. -/// -/// The size is checked so a copied image cannot designate a different disk. -fn designated(cache: &Arc) -> bool { - let mut guard = cache.lock(); - let blocks = guard.block_count(); - // A read error is not consent to format. - let Ok(block0) = guard.read(0) else { - log!("storage: block 0 could not be read; this disk is not ours to format"); - return false; - }; - - let magic = bcachefs::DESIGNATION_MAGIC; - if block0.len() < bcachefs::DESIGNATION_BLOCKS_OFFSET + 8 - || block0[..magic.len()] != magic - { - return false; - } - let mut stamped = [0u8; 8]; - stamped.copy_from_slice( - &block0[bcachefs::DESIGNATION_BLOCKS_OFFSET..bcachefs::DESIGNATION_BLOCKS_OFFSET + 8], - ); - let stamped = u64::from_le_bytes(stamped); - if stamped != blocks { - log!( - "storage: a designation stamp at block 0 names {} blocks, but this device has {} — \ - ignoring it", - stamped, blocks - ); - return false; - } - true -} - -/// What `/apps` and `/home` are this boot. -pub enum Data { - /// The DATA volume, mounted read-write. - Mounted(Arc, Mounted), - /// No volume of ours to mount — none, two, one that is not ours, or a - /// format that failed — so a tmpfs stands in, and says so. - Volatile, - /// A volume of ours that did not mount. Nothing stands in: a tmpfs under - /// the paths the owner's data lives at would take their writes into RAM. - Absent, -} - -/// The DATA filesystem `/apps` and `/home` are two paths into, and the only -/// path on which `format` runs. -/// -/// A role names one filesystem, so two TOYOS-DATA partitions are refused rather -/// than guessed between; no arm panics or formats without consent. -pub fn open_data() -> Data { - let candidates = crate::gpt::data_candidates(); - let [candidate] = candidates.as_slice() else { - log!( - "storage: this machine carries {} TOYOS-DATA partitions, and a data volume is one", - candidates.len() - ); - return Data::Volatile; - }; - // The GPT type names this candidate ours, whether or not a view over it - // could be opened — no driver claimed its device, or its geometry is not - // whole blocks. That is the same harm `Absent` exists for: a candidate of - // ours that a tmpfs would silently stand in behind. `over_candidate` - // already logged which. - let Some(cache) = page_cache::over_candidate(candidate) else { return Data::Absent }; - let fs = match probe(&cache) { - Storage::Ours(fs) => fs, - Storage::Designated => match format(&cache) { - Some(fs) => fs, - None => return Data::Volatile, - }, - Storage::Unmountable => return Data::Absent, - Storage::Foreign => return Data::Volatile, - }; - Data::Mounted(cache, fs) -} - diff --git a/kernel/src/block.rs b/kernel/src/block.rs index 41dcdefd388..1805f20af93 100644 --- a/kernel/src/block.rs +++ b/kernel/src/block.rs @@ -179,13 +179,11 @@ pub enum BlockError { BudgetExpired, } -impl BlockError { - /// Combines two failures from one composed operation: `Device` always wins. - pub fn worse(self, other: Self) -> Self { - match (self, other) { - (Self::Device, _) | (_, Self::Device) => Self::Device, - _ => Self::BudgetExpired, - } +/// The one conversion between the kernel's block verdict and the bcachefs +/// crate's: a `BudgetExpired` was never attempted and stays retryable. +impl bcachefs::TransferError for BlockError { + fn refused_before_attempt(&self) -> bool { + matches!(self, BlockError::BudgetExpired) } } @@ -346,26 +344,6 @@ impl BlockDevice for Locked<'_> { } } -/// A block of one partition of one device, minted only by [`Partition::key`]. -/// `partition` is judged (`pc-partition-offset`): it puts a write where the -/// slot was filled from. `device` is not, and no test can see it — one cache -/// serves one partition, so two devices in one map is already unrepresentable. -/// The field order is the sort order, so a run of one view's keys is a run on -/// the device. -#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Debug)] -pub struct BlockKey { - device: DeviceId, - partition: u64, - block: u64, -} - -impl BlockKey { - /// Where the block is on the device, which is what a transfer takes. - pub fn device_block(self) -> u64 { - self.partition + self.block - } -} - /// Who holds a span of a device. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum Holder { @@ -466,20 +444,10 @@ impl Partition { self.handle.device_id() } - pub fn first_block(&self) -> u64 { - self.first_block - } - pub fn block_count(&self) -> u64 { self.blocks } - /// The identity of `block` in this view, or a refusal past its end. - pub fn key(&self, block: u64) -> Result { - self.locate(block, 1)?; - Ok(BlockKey { device: self.device_id(), partition: self.first_block, block }) - } - /// Whether `count` blocks from `block` end inside the view — asked of a /// caller's numbers, so a refusal is the caller's answer and not a line in /// the kernel's log. @@ -590,20 +558,6 @@ pub fn duplicate_id_selftest() { ); } -/// Blocks the metadata cache may hold, per instance: sized from memory and -/// never from the device, so N instances claim N times this and none evicts -/// across them. N is the role partitions a boot caches — one today, from -/// `page_cache::init`'s one call site — and a foreign volume adds none, since a -/// FAT mount reads through `fat32_adapter::FatDevice` and never a `Cached`. -/// Must stay under 14,336 or the hashbrown index crosses the 16,384-bucket bound `nvme_large_device` asserts. -pub fn metadata_cache_blocks() -> usize { - if crate::actuator::test_small_caches() { - return 64; - } - let (total, _) = crate::mm::pmm::stats(); - (((total / 32) / PAGE_SIZE) as usize).clamp(64, 4096) -} - /// Pages the file data cache may hold. pub fn file_cache_pages() -> usize { if crate::actuator::test_small_caches() { diff --git a/kernel/src/drivers/hda.rs b/kernel/src/drivers/hda.rs index daf8f95b3d6..88c96fa28be 100644 --- a/kernel/src/drivers/hda.rs +++ b/kernel/src/drivers/hda.rs @@ -472,7 +472,7 @@ pub fn init(devices: &[PciDevice]) { /// controller's own domain does not map, and start the stream: the list is fetched at `RUN`. #[cfg(feature = "boot-actuators")] fn run_on_a_foreign_bdl(stream: Mmio) { - let foreign = super::nvme::FOREIGN_PROBE.load(Ordering::Relaxed); + let foreign = super::xhci::FOREIGN_PROBE.load(Ordering::Relaxed); assert!(foreign != 0, "hda: this machine staged no foreign pool to aim at"); stream.write_u32(SD_BDPL, foreign as u32); stream.write_u32(SD_BDPU, (foreign >> 32) as u32); diff --git a/kernel/src/drivers/mod.rs b/kernel/src/drivers/mod.rs index 586cf961a4e..d71937ef0b7 100644 --- a/kernel/src/drivers/mod.rs +++ b/kernel/src/drivers/mod.rs @@ -8,7 +8,6 @@ pub mod serial; pub mod serial_lock; pub mod acpi; pub mod pci; -pub mod nvme; pub mod xhci; pub mod usb_storage; pub mod virtio; diff --git a/kernel/src/drivers/nvme.rs b/kernel/src/drivers/nvme.rs deleted file mode 100644 index b8674aa532d..00000000000 --- a/kernel/src/drivers/nvme.rs +++ /dev/null @@ -1,908 +0,0 @@ -//! NVMe, as a [`BlockDevice`]. -//! -//! One command is outstanding at a time; `wait_completion`'s `cid` check is -//! what enforces it. [`COMMAND`] bounds one command; [`crate::block::OPERATION`] -//! bounds the composed operation a caller spends. The operation deadline, -//! established in [`NvmeBlockDevice`]'s trait methods, threads down to -//! `read_sectors`/`write_sectors`; `admin` takes none and is bounded by -//! [`COMMAND`] alone. A refusal is taken between commands, never inside one. - -use crate::arch::barrier; -use toyos_untrusted::{Refused, Untrusted}; -use crate::mm::Mmio; -use super::pci::PciDevice; -use super::DmaPool; -use crate::block::{self, BlockDevice, BlockError, BlockResult, DeviceId}; -use crate::mm::paging::MmioPolicy; -use crate::log; -use crate::mm::{Dma, Unaligned}; -use crate::scheduler::Operation; -use crate::time::{Budget, Deadline, Duration}; - -const REG_CAP: u64 = 0x00; -const REG_CC: u64 = 0x14; -const REG_CSTS: u64 = 0x1C; -const REG_AQA: u64 = 0x24; -const REG_ASQ: u64 = 0x28; -const REG_ACQ: u64 = 0x30; - -const QUEUE_DEPTH: usize = 16; - -/// How long one command may spend before this driver stops waiting for it; -/// must equal `xhci`'s `USB_TIMEOUT_NS` and [`crate::block::OPERATION`]'s own -/// budget, since both derive from the same arithmetic. A command that -/// outlasts it is reclaimed by one controller reset and one post-reset -/// chance (NVMe 2.0 §3.7.2), never by declaring the disk failed outright. -/// -/// A [`Budget`], not a [`crate::time::Bound`]: NVMe defines no per-command -/// timeout — `CAP.TO` bounds the `CSTS.RDY` waits in [`NvmeController::reset`] -/// and [`init`], and nothing else. -const COMMAND: Budget = Budget::of( - Duration::from_secs(2), - "the command is abandoned to a controller reset, and one post-reset silence \ - marks the disk failed", -); - -/// NVMe Identify Namespace (partial). `Copy`: read out of DMA memory by -/// value, never held as a reference into a window the device may write again. -#[repr(C)] -#[derive(Clone, Copy)] -struct IdentifyNamespace { - nsze: u64, // offset 0: namespace size in LBAs - ncap: u64, // offset 8: namespace capacity - nuse: u64, // offset 16: namespace utilization - nsfeat: u8, // offset 24 - nlbaf: u8, // offset 25: number of LBA formats (0-based) - flbas: u8, // offset 26: formatted LBA size - _padding: [u8; 101], // offsets 27..128 - lba_formats: [u32; 64], // offset 128: LBA format descriptors (4 bytes each) -} - -/// Every command this driver has issued, by queue and opcode. -/// -/// **The one machine-checked statement that a boot wrote no disk it was not -/// asked to write.** The metal bench's own Ubuntu install is on the internal -/// NVMe, and the whole safety argument for booting ToyOS on that machine is -/// that nothing here ever issues it a write. A count logged at -/// `SYS_SHUTDOWN`/`SYS_REBOOT` is what turns that argument into a record the -/// boot leaves behind on the log volume, where a reader who was not there can -/// check it. -/// -/// Module-level rather than per controller: the counts answer a question about -/// the machine, and a second controller's writes would be no less a write. -mod census { - use core::sync::atomic::{AtomicU64, Ordering}; - - /// Which queue a command went down, since the opcode numbers overlap: 0x01 - /// is Create I/O Submission Queue on the admin queue and Write on the I/O - /// one. - #[derive(Clone, Copy)] - pub(super) enum Queue { - Admin, - Io, - } - - static IDENTIFY: AtomicU64 = AtomicU64::new(0); - static ADMIN_OTHER: AtomicU64 = AtomicU64::new(0); - static READ: AtomicU64 = AtomicU64::new(0); - static WRITE: AtomicU64 = AtomicU64::new(0); - static IO_OTHER: AtomicU64 = AtomicU64::new(0); - - /// Counted at submission and not at completion: a write the controller - /// never answered still reached the disk. - pub(super) fn issued(queue: Queue, cdw0: u32) { - let opcode = (cdw0 & 0xff) as u8; - let counter = match (queue, opcode) { - (Queue::Admin, super::ADMIN_IDENTIFY) => &IDENTIFY, - (Queue::Admin, _) => &ADMIN_OTHER, - (Queue::Io, super::IO_READ) => &READ, - (Queue::Io, super::IO_WRITE) => &WRITE, - (Queue::Io, _) => &IO_OTHER, - }; - counter.fetch_add(1, Ordering::Relaxed); - crate::block::census::command_issued(); - } - - pub(super) fn line() -> (u64, u64, u64, u64, u64) { - let read = |c: &AtomicU64| c.load(Ordering::Relaxed); - (read(&IDENTIFY), read(&ADMIN_OTHER), read(&READ), read(&WRITE), read(&IO_OTHER)) - } -} - -/// The boot's whole NVMe command census, as one record. -/// -/// Called from `quiesce`, so the count is the boot's total and no process runs -/// after it to add to one. -pub fn log_census() { - let (identify, admin_other, read, write, io_other) = census::line(); - log!( - "nvme: commands identify={identify} admin-other={admin_other} read={read} \ - write={write} io-other={io_other}" - ); -} - -const ADMIN_CREATE_IO_SQ: u8 = 0x01; -const ADMIN_CREATE_IO_CQ: u8 = 0x05; -const ADMIN_IDENTIFY: u8 = 0x06; -const IO_WRITE: u8 = 0x01; -const IO_READ: u8 = 0x02; - -/// EN plus IOSQES/IOCQES for the 64-/16-byte entry sizes above; one constant -/// so [`init`] and [`NvmeController::reset`] enable the controller identically. -const CC_ENABLED: u32 = 1 | (6 << 16) | (4 << 20); - -#[repr(C)] -#[derive(Clone, Copy)] -struct SqEntry { - cdw0: u32, - nsid: u32, - cdw2: u32, - cdw3: u32, - mptr: u64, - prp1: u64, - prp2: u64, - cdw10: u32, - cdw11: u32, - cdw12: u32, - cdw13: u32, - cdw14: u32, - cdw15: u32, -} - -impl SqEntry { - const ZERO: Self = Self { - cdw0: 0, nsid: 0, cdw2: 0, cdw3: 0, - mptr: 0, prp1: 0, prp2: 0, - cdw10: 0, cdw11: 0, cdw12: 0, cdw13: 0, cdw14: 0, cdw15: 0, - }; -} - -#[repr(C)] -#[derive(Clone, Copy)] -struct CqEntry { - dw0: u32, - dw1: u32, - sq_head: u16, - sq_id: u16, - cid: u16, - status: u16, // bit 0 = phase, bits [15:1] = status -} - -/// One submission/completion queue pair, as two [`Dma`] views: the controller -/// accesses both concurrently with this CPU, so every access goes through the -/// volatile discipline and is bounded against the view's own length. -struct NvmeQueue { - sq: Dma<'static>, - cq: Dma<'static>, - sq_tail: u16, - cq_head: u16, - phase: bool, - sq_doorbell: u64, - cq_doorbell: u64, -} - -impl NvmeQueue { - fn new(sq: Dma<'static>, cq: Dma<'static>, qid: u16, stride: u32) -> Self { - let doorbell_stride = 4u64 << stride; - Self { - sq, cq, - sq_tail: 0, cq_head: 0, phase: true, - sq_doorbell: 0x1000 + (2 * qid as u64) * doorbell_stride, - cq_doorbell: 0x1000 + (2 * qid as u64 + 1) * doorbell_stride, - } - } - - /// Resets this queue's software state to freshly-created; the views and - /// doorbell offsets stand, since the reset moves no memory. - fn start_over(&mut self) { - self.sq_tail = 0; - self.cq_head = 0; - self.phase = true; - } - - fn submit(&mut self, bar: &Mmio, cmd: SqEntry) { - // Bounded by `sq_tail % QUEUE_DEPTH` against the page `init` allocated; - // the doorbell below is what tells the device it happened, and as an - // `Mmio` write it is ordered after the entry. - self.sq.write(self.sq_tail as usize * core::mem::size_of::(), cmd); - self.sq_tail = (self.sq_tail + 1) % QUEUE_DEPTH as u16; - bar.write_u32(self.sq_doorbell, self.sq_tail as u32); - } - - /// Wait for the completion at the head of the queue; refuse it unless its - /// `cid` is `expected`. - /// - /// Bounded by [`COMMAND`]: callers reach this holding a page cache's lock - /// and this device's, so an unbounded wait would wedge a CPU holding both. - /// - /// Reads the entry twice: the predicate read inside `settles` is not the - /// read consumed below, sound because one command is outstanding at a - /// time, so nothing rewrites the entry before the second read. - fn wait_completion(&mut self, bar: &Mmio, expected: u16) -> Result { - // Abandons this command without waiting, on the harness's request: no - // QEMU device property makes a real completion not arrive, so this is - // the only way to stage the state a silent controller leaves behind. - #[cfg(feature = "boot-actuators")] - if silent_command::take() { - return Err(Unanswered::Silent); - } - let (cq, head, phase) = (self.cq, self.cq_head, self.phase); - let at = |i: u16| i as usize * core::mem::size_of::(); - let answered = crate::clock::settles(COMMAND.nanos(), || { - // Volatile so the spin observes the phase bit flip rather than - // reading it once (NVMe 2.0 §3.3.3.2); bounded the same way `submit` - // is, by `cq_head % QUEUE_DEPTH`. - let entry: CqEntry = cq.read(at(head)); - ((entry.status & 1) != 0) == phase - }); - if !answered { - return Err(Unanswered::Silent); - } - // The entry, and the data it completes, read after the phase tag that - // says they are there. - barrier::dma_rmb(); - let cq: CqEntry = self.cq.read(at(self.cq_head)); - let status = cq.status >> 1; - let cid = Untrusted::new(cq.cid); - self.cq_head = (self.cq_head + 1) % QUEUE_DEPTH as u16; - if self.cq_head == 0 { - self.phase = !self.phase; - } - bar.write_u32(self.cq_doorbell, self.cq_head as u32); - cid.exactly(expected).map(|_| status).map_err(Unanswered::Wrong) - } - - fn submit_and_wait(&mut self, bar: &Mmio, cmd: SqEntry) -> Result { - // `cmd`'s own cid, not a value trusted back from the caller. - let expected = (cmd.cdw0 >> 16) as u16; - self.submit(bar, cmd); - self.wait_completion(bar, expected) - } -} - -/// The arm behind `nvme-command-silent`: `nvme_gate` arms this at its own -/// read, never at the boot parameter, so `init`'s own Identify wait is never -/// the one skipped. See the take site, [`NvmeQueue::wait_completion`]. -#[cfg(feature = "boot-actuators")] -pub mod silent_command { - use core::sync::atomic::{AtomicBool, Ordering}; - - static ARMED: AtomicBool = AtomicBool::new(false); - - pub fn arm() { - ARMED.store(true, Ordering::Relaxed); - } - - pub(super) fn take() -> bool { - ARMED.swap(false, Ordering::Relaxed) - } -} - -/// Why a submitted command produced no status this driver may use. The two -/// arms differ in what they leave behind: `Wrong` leaves the queue consistent, -/// `Silent` leaves an entry owed, reclaimed only by [`NvmeController::reset`]. -enum Unanswered { - /// The completion queue answered a different command. - Wrong(Refused), - /// The controller did not answer inside [`COMMAND`]. - Silent, -} - -impl core::fmt::Display for Unanswered { - fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::Wrong(refused) => { - write!(f, "the completion queue answered a different command ({refused})") - } - Self::Silent => write!(f, "no completion in {}", COMMAND.duration()), - } - } -} - -const OFF_ADMIN_SQ: usize = 0x0000; -const OFF_ADMIN_CQ: usize = 0x1000; -const OFF_IO_SQ: usize = 0x2000; -const OFF_IO_CQ: usize = 0x3000; -const OFF_IDENTIFY: usize = 0x4000; -const OFF_PRP_LIST: usize = 0x5000; -const OFF_DATA: usize = 0x6000; -const MAX_DATA_PAGES: usize = 32; -const DMA_SIZE: usize = OFF_DATA + MAX_DATA_PAGES * 0x1000; - -/// The upper half of the admin completion queue's page: `init` zeroes it all and -/// the controller writes only the first `QUEUE_DEPTH` entries, so "unchanged" is -/// "still zero" here, and the address is one the gate reads back out of `REG_ACQ`. -#[cfg(feature = "boot-actuators")] -pub(super) const PROBE_OFF: usize = OFF_ADMIN_CQ + 0x800; -#[cfg(feature = "boot-actuators")] -pub(super) const PROBE_LEN: usize = 0x800; -#[cfg(feature = "boot-actuators")] -const _: () = assert!(QUEUE_DEPTH * core::mem::size_of::() <= PROBE_OFF - OFF_ADMIN_CQ); - -/// Physical, not what this controller is programmed with: the actuator has to -/// hand another device an address that device's own domain does not map. -#[cfg(feature = "boot-actuators")] -pub static FOREIGN_PROBE: core::sync::atomic::AtomicU64 = - core::sync::atomic::AtomicU64::new(0); - -/// Fills the PRP list with every data page's address after the first -/// (NVMe 2.0 §4.1.2) and returns the list's own physical address for `prp2`. -/// Unaligned discipline: the list is written before the controller can read it. -fn fill_prp_list(dma: Dma<'static, Unaligned>, pages: usize, data_phys: u64) -> u64 { - // `pages` is bounded by `MAX_DATA_PAGES`, asserted by both callers; - // `subview` turns that bound into a check on the `pages - 1` entries. - let list = dma.subview(OFF_PRP_LIST, (pages - 1) * core::mem::size_of::()); - for i in 1..pages { - list.write::((i - 1) * core::mem::size_of::(), data_phys + i as u64 * 0x1000); - } - dma.device_addr() + OFF_PRP_LIST as u64 -} - -struct NvmeController { - bar: Mmio, - /// This controller's DMA window, leaked at `init` and therefore `'static`. - dma: Dma<'static>, - admin: NvmeQueue, - io: NvmeQueue, - next_cid: u16, - sector_size: u32, - ns_size: u64, - /// Whether this controller has been declared failed; once set, this - /// driver issues nothing more on it. - failed: bool, - /// Whether the last thing on this controller was a reset. One post-reset - /// command is the escalation's whole allowance: silence with this set - /// declares the controller failed; any served command clears it. - fresh_reset: bool, -} - -impl NvmeController { - /// Clear `len` bytes of the DMA window at `off`. Every caller clears a - /// queue, scratch page or descriptor buffer before submitting the command - /// that hands it to the controller. - fn zero_dma(&self, off: usize, len: usize) { - self.dma.subview(off, len).zero(); - } - - fn alloc_cid(&mut self) -> u16 { - let cid = self.next_cid; - self.next_cid = self.next_cid.wrapping_add(1); - cid - } - - /// An admin command's silence ends the controller at once, with no reset - /// between: admin commands run only at bring-up and inside [`Self::reset`] - /// itself, where a reset escalation would recurse into its own failure. - fn admin_command(&mut self, cmd: SqEntry) -> Result { - census::issued(census::Queue::Admin, cmd.cdw0); - let out = self.admin.submit_and_wait(&self.bar, cmd); - if matches!(out, Err(Unanswered::Silent)) && !self.failed { - self.failed = true; - log!("NVMe: this controller is offline: an admin command went unanswered, and \ - the reset escalation is not spent on a controller that cannot be asked to \ - make queues"); - } - out - } - - /// An I/O command, with silence escalated: one controller reset, one - /// post-reset chance, then the disk declared failed. - fn io_command(&mut self, cmd: SqEntry) -> Result { - census::issued(census::Queue::Io, cmd.cdw0); - let out = self.io.submit_and_wait(&self.bar, cmd); - match &out { - Ok(_) => self.fresh_reset = false, - Err(Unanswered::Silent) if !self.failed => { - if self.fresh_reset { - self.failed = true; - log!("NVMe: this controller is offline: the first command after its reset \ - went unanswered too, and one post-reset command is the escalation's \ - whole allowance"); - } else { - log!("NVMe: no completion in {} — resetting the controller: the abandoned \ - command still owns its PRP list and is still owed a completion entry, \ - and a reset is the one way to take both back", COMMAND.duration()); - if self.reset() { - self.fresh_reset = true; - log!("NVMe: controller reset complete; the disk stays online and the \ - caller is told to ask again"); - } else { - self.failed = true; - log!("NVMe: this controller is offline: the reset escalation itself \ - failed"); - } - } - } - Err(_) => {} - } - out - } - - /// What one unanswered I/O command means to [`BlockDevice`]'s caller, - /// decided after [`Self::io_command`]'s escalation has run: a silence the - /// reset reclaimed is [`BlockError::BudgetExpired`], everything else is - /// [`BlockError::Device`]. - fn unanswered(&self, why: Unanswered) -> BlockError { - match why { - Unanswered::Silent if !self.failed => BlockError::BudgetExpired, - Unanswered::Silent | Unanswered::Wrong(_) => BlockError::Device, - } - } - - /// Controller reset: `CC.EN` 0 → 1 plus the I/O queues made afresh - /// (NVMe 2.0 §3.7.2), reclaiming an abandoned command's PRP list and owed - /// completion entry. The two `RDY` waits are bounded by `CAP.TO` - /// (§3.1.4.1, 500 ms units), the controller's own published worst case. - /// The admin queue is not re-created by command, so `AQA`/`ASQ`/`ACQ` are - /// rewritten with the values [`init`] programmed. - fn reset(&mut self) -> bool { - let to = ((self.bar.read_u64(REG_CAP) >> 24) & 0xFF).max(1); - let ready = crate::time::Bound::from_register( - Duration::from_millis(to * 500), - "NVMe CAP.TO, the controller's own worst case for a CSTS.RDY transition", - ); - let cc = self.bar.read_u32(REG_CC); - self.bar.write_u32(REG_CC, cc & !1); - if !crate::clock::settles(ready.nanos(), || self.bar.read_u32(REG_CSTS) & 1 == 0) { - log!("NVMe: reset failed: CSTS.RDY would not clear in {ready}"); - return false; - } - - self.zero_dma(OFF_ADMIN_SQ, 4096); - self.zero_dma(OFF_ADMIN_CQ, 4096); - self.admin.start_over(); - self.io.start_over(); - let aqa = ((QUEUE_DEPTH as u32 - 1) << 16) | (QUEUE_DEPTH as u32 - 1); - self.bar.write_u32(REG_AQA, aqa); - self.bar.write_u64(REG_ASQ, self.dma.device_addr() + OFF_ADMIN_SQ as u64); - self.bar.write_u64(REG_ACQ, self.dma.device_addr() + OFF_ADMIN_CQ as u64); - self.bar.write_u32(REG_CC, CC_ENABLED); - if !crate::clock::settles(ready.nanos(), || self.bar.read_u32(REG_CSTS) & 1 != 0) { - log!("NVMe: reset failed: CSTS.RDY would not set in {ready}"); - return false; - } - // Not re-asked: writing `sector_size` mid-boot would race every - // layout the layers above already derived from it. - self.create_io_cq() && self.create_io_sq() - } - - /// An admin command, with the returned status checked: discarding it would - /// make a controller that refused indistinguishable from one that did not. - /// No deadline argument: bringing a controller up is bounded by - /// [`COMMAND`] alone. - fn admin(&mut self, cmd: SqEntry, what: &str) -> bool { - let status = match self.admin_command(cmd) { - Ok(status) => status, - Err(why) => { - log!("NVMe: {what}: {why}"); - return false; - } - }; - if status != 0 { - log!("NVMe: {what} failed, status={status:#x}"); - return false; - } - true - } - - /// Whether this command may be issued: the controller still has its - /// queues, and the caller's budget has something left. Read between - /// commands and never inside one, so a refusal here costs nothing — no - /// completion is owed and the DMA window is nobody's. An abandoned - /// controller is [`BlockError::Device`]; a spent budget is - /// [`BlockError::BudgetExpired`], never a fact about the controller. - fn may_issue(&self, until: Deadline, op: &str, lba: u64, sector_count: u32) -> BlockResult { - if self.failed { - return Err(BlockError::Device); - } - if until.reached(crate::clock::now()) { - log!("NVMe: {op} of {sector_count} sectors at {lba} not issued: {}", block::OPERATION); - return Err(BlockError::BudgetExpired); - } - Ok(()) - } - - fn identify_controller(&mut self) -> bool { - let dma = self.dma; - let cid = self.alloc_cid(); - let mut cmd = SqEntry::ZERO; - cmd.cdw0 = (cid as u32) << 16 | ADMIN_IDENTIFY as u32; - cmd.prp1 = dma.device_addr() + OFF_IDENTIFY as u64; - cmd.cdw10 = 1; - self.admin(cmd, "Identify Controller") - } - - fn create_io_cq(&mut self) -> bool { - self.zero_dma(OFF_IO_CQ, QUEUE_DEPTH * core::mem::size_of::()); - let dma = self.dma; - let cid = self.alloc_cid(); - let mut cmd = SqEntry::ZERO; - cmd.cdw0 = (cid as u32) << 16 | ADMIN_CREATE_IO_CQ as u32; - cmd.prp1 = dma.device_addr() + OFF_IO_CQ as u64; - cmd.cdw10 = ((QUEUE_DEPTH as u32 - 1) << 16) | 1; - cmd.cdw11 = 1; - self.admin(cmd, "Create I/O Completion Queue") - } - - fn create_io_sq(&mut self) -> bool { - self.zero_dma(OFF_IO_SQ, QUEUE_DEPTH * core::mem::size_of::()); - let dma = self.dma; - let cid = self.alloc_cid(); - let mut cmd = SqEntry::ZERO; - cmd.cdw0 = (cid as u32) << 16 | ADMIN_CREATE_IO_SQ as u32; - cmd.prp1 = dma.device_addr() + OFF_IO_SQ as u64; - cmd.cdw10 = ((QUEUE_DEPTH as u32 - 1) << 16) | 1; - cmd.cdw11 = (1 << 16) | 1; - self.admin(cmd, "Create I/O Submission Queue") - } - - fn identify_namespace(&mut self) -> bool { - let dma = self.dma; - self.zero_dma(OFF_IDENTIFY, 4096); - let cid = self.alloc_cid(); - let mut cmd = SqEntry::ZERO; - cmd.cdw0 = (cid as u32) << 16 | ADMIN_IDENTIFY as u32; - cmd.nsid = 1; - cmd.prp1 = dma.device_addr() + OFF_IDENTIFY as u64; - cmd.cdw10 = 0; - if !self.admin(cmd, "Identify Namespace") { - return false; - } - - // Copy, not a reference into memory the device may write again. - // Unaligned discipline: `admin` returning `true` means the transfer - // has completed, so nothing is writing these bytes (NVMe 2.0 §5.17.2.1). - let ns: IdentifyNamespace = dma.unaligned().read(OFF_IDENTIFY); - let fmt_idx = (ns.flbas & 0x0F) as usize; - let lba_ds = (ns.lba_formats[fmt_idx] >> 16) & 0xFF; - // `1 << lba_ds` overflows above 31, and above 12 `4096 / sector_size` - // is zero, which `NvmeBlockDevice::new` divides `nsze` by — a `#DE` - // before storage is up. 9..=12 is this driver: every path above the - // sector layer needs 4096 to divide the sector size evenly. - assert!( - (9..=12).contains(&lba_ds), - "NVMe: namespace reports 2^{lba_ds}-byte sectors (flbas={:#x}, format {fmt_idx}); \ - this driver serves 4096-byte blocks and needs 512..=4096", - ns.flbas, - ); - self.sector_size = 1 << lba_ds; - self.ns_size = ns.nsze; - log!("NVMe: NS1 size={} sectors, sector_size={}", ns.nsze, self.sector_size); - true - } - - /// Read `sector_count` contiguous sectors starting at `lba` into `buf`. - /// `until` is the whole operation's deadline, not this command's. - fn read_sectors( - &mut self, - lba: u64, - sector_count: u32, - buf: &mut [u8], - until: Deadline, - ) -> BlockResult { - let total_bytes = sector_count as usize * self.sector_size as usize; - assert!(buf.len() >= total_bytes); - assert!(total_bytes <= MAX_DATA_PAGES * 4096); - - self.may_issue(until, "read", lba, sector_count)?; - - let dma = self.dma; - let pages = total_bytes.div_ceil(4096); - let data_phys = dma.device_addr() + OFF_DATA as u64; - - let cid = self.alloc_cid(); - let mut cmd = SqEntry::ZERO; - cmd.cdw0 = (cid as u32) << 16 | IO_READ as u32; - cmd.nsid = 1; - cmd.prp1 = data_phys; - cmd.cdw10 = lba as u32; - cmd.cdw11 = (lba >> 32) as u32; - cmd.cdw12 = sector_count - 1; - - if pages == 2 { - cmd.prp2 = data_phys + 0x1000; - } else if pages > 2 { - cmd.prp2 = fill_prp_list(dma.unaligned(), pages, data_phys); - } - - let status = match self.io_command(cmd) { - Ok(status) => status, - Err(why) => { - log!("NVMe: read of {sector_count} sectors at {lba}: {why}"); - return Err(self.unanswered(why)); - } - }; - if status != 0 { - log!("NVMe: read of {sector_count} sectors at {lba} failed, status={status:#x}"); - return Err(BlockError::Device); - } - - // Copied out rather than referenced, so nothing outlives the instant - // the controller is known done with it; bounds asserted on entry. - dma.copy_to(OFF_DATA, &mut buf[..total_bytes]); - Ok(()) - } - - fn write_sectors( - &mut self, - lba: u64, - sector_count: u32, - buf: &[u8], - until: Deadline, - ) -> BlockResult { - let total_bytes = sector_count as usize * self.sector_size as usize; - assert!(buf.len() >= total_bytes); - assert!(total_bytes <= MAX_DATA_PAGES * 4096); - - self.may_issue(until, "write", lba, sector_count)?; - - let dma = self.dma; - let pages = total_bytes.div_ceil(4096); - let data_phys = dma.device_addr() + OFF_DATA as u64; - - // Bounds asserted on entry; exclusive, since the write command naming - // this window has not been submitted yet. - dma.copy_from(OFF_DATA, &buf[..total_bytes]); - - let cid = self.alloc_cid(); - let mut cmd = SqEntry::ZERO; - cmd.cdw0 = (cid as u32) << 16 | IO_WRITE as u32; - cmd.nsid = 1; - cmd.prp1 = data_phys; - cmd.cdw10 = lba as u32; - cmd.cdw11 = (lba >> 32) as u32; - cmd.cdw12 = sector_count - 1; - - if pages == 2 { - cmd.prp2 = data_phys + 0x1000; - } else if pages > 2 { - cmd.prp2 = fill_prp_list(dma.unaligned(), pages, data_phys); - } - - let status = match self.io_command(cmd) { - Ok(status) => status, - Err(why) => { - log!("NVMe: write of {sector_count} sectors at {lba}: {why}"); - return Err(self.unanswered(why)); - } - }; - if status != 0 { - log!("NVMe: write of {sector_count} sectors at {lba} failed, status={status:#x}"); - return Err(BlockError::Device); - } - Ok(()) - } -} - -/// NVMe block device exposing 4KB block I/O through the BlockDevice trait. -/// `Send` is derived, not asserted: every field, including the queues' -/// [`Dma`] views, is `Send` on its own. -pub struct NvmeBlockDevice { - ctrl: NvmeController, - id: DeviceId, - sectors_per_block: u32, - block_count: u64, -} - -impl NvmeBlockDevice { - fn new(ctrl: NvmeController, id: DeviceId) -> Self { - let sectors_per_block = 4096 / ctrl.sector_size; - let block_count = ctrl.ns_size / sectors_per_block as u64; - log!("NVMe: block device id={} blocks={} ({}MB)", - id, block_count, block_count * 4096 / (1024 * 1024)); - Self { ctrl, id, sectors_per_block, block_count } - } - - /// The namespace's own logical block size, which `BlockDevice` hides; - /// a GPT is laid out in the device's blocks and needs it directly. - pub fn sector_size(&self) -> u32 { - self.ctrl.sector_size - } -} - -impl BlockDevice for NvmeBlockDevice { - fn device_id(&self) -> DeviceId { self.id } - fn block_count(&self) -> u64 { self.block_count } - - /// `let _op`, not `let _`: the latter drops at end of statement, ending - /// the operation before the loop below it bounds. The deadline is read - /// after establishment, since establishing may only narrow it. - fn read_blocks(&mut self, lba: u64, count: u32, buf: &mut [u8]) -> BlockResult { - assert_eq!(buf.len(), count as usize * 4096); - let _op = block::begin_operation(); - let until = Operation::deadline(); - let mut remaining = count; - let mut block = lba; - let mut offset = 0usize; - - while remaining > 0 { - let batch = remaining.min(MAX_DATA_PAGES as u32); - let sector_lba = block * self.sectors_per_block as u64; - let sector_count = batch * self.sectors_per_block; - let bytes = batch as usize * 4096; - - if let Err(e) = self - .ctrl - .read_sectors(sector_lba, sector_count, &mut buf[offset..offset + bytes], until) - { - if e == BlockError::BudgetExpired { - block::census::budget_expired(self.id); - } - return Err(e); - } - - block += batch as u64; - offset += bytes; - remaining -= batch; - } - Ok(()) - } - - fn write_blocks(&mut self, lba: u64, count: u32, buf: &[u8]) -> BlockResult { - assert_eq!(buf.len(), count as usize * 4096); - let _op = block::begin_operation(); - let until = Operation::deadline(); - let mut remaining = count; - let mut block = lba; - let mut offset = 0usize; - - while remaining > 0 { - let batch = remaining.min(MAX_DATA_PAGES as u32); - let sector_lba = block * self.sectors_per_block as u64; - let sector_count = batch * self.sectors_per_block; - let bytes = batch as usize * 4096; - - if let Err(e) = self - .ctrl - .write_sectors(sector_lba, sector_count, &buf[offset..offset + bytes], until) - { - if e == BlockError::BudgetExpired { - block::census::budget_expired(self.id); - } - return Err(e); - } - - block += batch as u64; - offset += bytes; - remaining -= batch; - } - Ok(()) - } - - /// Writes are synchronous, so there is nothing to flush; this establishes - /// no operation because it issues no command. A failed controller is the - /// exception: `Ok` here would tell `page_cache::sync` writes completed - /// that never did. - fn flush(&mut self) -> BlockResult { - if self.ctrl.failed { - return Err(BlockError::Device); - } - Ok(()) - } - - /// A namespace this driver never takes back once it has gone. - fn losses(&self) -> u64 { - 0 - } -} - -/// `CSTS.RDY` as this boot can see it; the actuator blinds the read, so a -/// controller that never answers is stageable on a device that always does. -fn rdy_observed(bar: &crate::mm::Mmio) -> bool { - #[cfg(feature = "boot-actuators")] - if crate::actuator::nvme_rdy_stuck() { - return false; - } - bar.read_u32(REG_CSTS) & 1 != 0 -} - -/// Bring up the machine's first NVMe controller; a second is not served, -/// since the one namespace it returns is hardcoded to `DeviceId` 1 below. -pub fn init(devices: &[PciDevice]) -> Option { - let pci_dev = *devices.iter().find(|d| d.matches_class(0x01, 0x08, None))?; - log!("NVMe: found at PCI {:02x}:{:02x}.{}", pci_dev.bus, pci_dev.dev, pci_dev.func); - - // Refusal, not a panic: NVMe 2.0 §3.1 requires BAR 0 to be memory, so a - // controller publishing otherwise has no disk this driver can drive. - let bar_addr = match pci_dev.memory_bar(0) { - Ok(memory) => memory.address(), - Err(why) => { - log!("NVMe: NOT INITIALISED at PCI {:02x}:{:02x}.{} — its registers are in BAR 0 and \ - {}", pci_dev.bus, pci_dev.dev, pci_dev.func, why); - return None; - } - }; - pci_dev.enable_bus_master(); - log!("NVMe: BAR0={:#x}", bar_addr); - - let bar = crate::mm::paging::map_mmio(bar_addr, 0x4000, MmioPolicy::Uncacheable); - - let cap = bar.read_u64(REG_CAP); - let stride = ((cap >> 32) & 0xF) as u32; - // The same two `CSTS.RDY` transitions `reset` bounds, on the same - // published worst case — unbounded, a controller that never answers - // hangs the boot with nothing on the log to say which one. - let to = ((cap >> 24) & 0xFF).max(1); - let ready = crate::time::Bound::from_register( - Duration::from_millis(to * 500), - "NVMe CAP.TO, the controller's own worst case for a CSTS.RDY transition", - ); - let rdy = || rdy_observed(&bar); - - let cc = bar.read_u32(REG_CC); - if cc & 1 != 0 { - bar.write_u32(REG_CC, cc & !1); - if !crate::clock::settles(ready.nanos(), || !rdy()) { - log!("NVMe: NOT INITIALISED — CSTS.RDY would not clear in {ready}"); - return None; - } - } - - // An address space of this controller's own, attached before an address is - // written to a register. Both isolation actuators mis-program *this* - // function's context entry by hand, and attaching it would write a good one - // over the staging. - let space = if crate::actuator::iommu_context_absent() - || crate::actuator::iommu_empty_domain() - { - crate::iommu::DeviceSpace::Untranslated - } else { - crate::iommu::DeviceSpace::create() - }; - // Leaked, not held in a `static`; allocated after every refusal above so - // a declined NVMe function costs no physical memory. - let dma = DmaPool::alloc_in(DMA_SIZE, space).leak(); - space.attach(pci_dev.bus, pci_dev.dev, pci_dev.func); - const SQ_PAGE: usize = QUEUE_DEPTH * core::mem::size_of::(); - const CQ_PAGE: usize = QUEUE_DEPTH * core::mem::size_of::(); - let admin_sq = dma.subview(OFF_ADMIN_SQ, SQ_PAGE); - let admin_cq = dma.subview(OFF_ADMIN_CQ, CQ_PAGE); - let io_sq = dma.subview(OFF_IO_SQ, SQ_PAGE); - let io_cq = dma.subview(OFF_IO_CQ, CQ_PAGE); - - dma.subview(OFF_ADMIN_SQ, 4096).zero(); - dma.subview(OFF_ADMIN_CQ, 4096).zero(); - #[cfg(feature = "boot-actuators")] - FOREIGN_PROBE.store( - dma.subview(PROBE_OFF, PROBE_LEN).host_phys(), - core::sync::atomic::Ordering::Relaxed, - ); - - let aqa = ((QUEUE_DEPTH as u32 - 1) << 16) | (QUEUE_DEPTH as u32 - 1); - bar.write_u32(REG_AQA, aqa); - bar.write_u64(REG_ASQ, dma.device_addr() + OFF_ADMIN_SQ as u64); - bar.write_u64(REG_ACQ, dma.device_addr() + OFF_ADMIN_CQ as u64); - - bar.write_u32(REG_CC, CC_ENABLED); - - if !crate::clock::settles(ready.nanos(), rdy) { - log!("NVMe: NOT INITIALISED — CSTS.RDY would not set in {ready}"); - return None; - } - log!("NVMe: controller enabled"); - - let mut ctrl = NvmeController { - bar, - dma, - admin: NvmeQueue::new(admin_sq, admin_cq, 0, stride), - io: NvmeQueue::new(io_sq, io_cq, 1, stride), - next_cid: 0, - sector_size: 512, - ns_size: 0, - failed: false, - fresh_reset: false, - }; - - // A controller refusing any of these has given no namespace to serve; - // continuing would derive geometry from a zeroed DMA buffer. - if !ctrl.identify_controller() - || !ctrl.create_io_cq() - || !ctrl.create_io_sq() - || !ctrl.identify_namespace() - { - log!("NVMe: controller did not come up; this machine has no NVMe storage"); - return None; - } - - Some(NvmeBlockDevice::new(ctrl, 1)) -} diff --git a/kernel/src/drivers/virtio_gpu.rs b/kernel/src/drivers/virtio_gpu.rs index d81f73c42cb..46ceab6c645 100644 --- a/kernel/src/drivers/virtio_gpu.rs +++ b/kernel/src/drivers/virtio_gpu.rs @@ -454,15 +454,15 @@ impl GpuController { #[cfg(feature = "boot-actuators")] fn attach_a_foreign_backing(&mut self) { let foreign = - super::nvme::FOREIGN_PROBE.load(core::sync::atomic::Ordering::Relaxed); + super::xhci::FOREIGN_PROBE.load(core::sync::atomic::Ordering::Relaxed); assert!(foreign != 0, "VirtIO GPU: this machine staged no foreign pool to aim at"); // Sized to the probe exactly, so one transfer of the whole resource reads every byte of it. - const ROWS: u32 = super::nvme::PROBE_LEN as u32 / (FOREIGN_COLUMNS * 4); + const ROWS: u32 = super::xhci::PROBE_LEN as u32 / (FOREIGN_COLUMNS * 4); self.create_resource(FOREIGN_RESOURCE_ID, FORMAT_B8G8R8X8_UNORM, FOREIGN_COLUMNS, ROWS); let resp = self.attach_backing_answering( FOREIGN_RESOURCE_ID, foreign, - super::nvme::PROBE_LEN as u32, + super::xhci::PROBE_LEN as u32, ); log!( "VirtIO GPU: a backing at {foreign:#x}, inside another driver's pool, was answered \ @@ -473,7 +473,7 @@ impl GpuController { self.transfer_to_host(FOREIGN_RESOURCE_ID, rect, 0); log!( "VirtIO GPU: the device read all {} bytes of it (actuator)", - super::nvme::PROBE_LEN + super::xhci::PROBE_LEN ); } } diff --git a/kernel/src/drivers/virtio_sound.rs b/kernel/src/drivers/virtio_sound.rs index befdb9b15d8..deb958b9ead 100644 --- a/kernel/src/drivers/virtio_sound.rs +++ b/kernel/src/drivers/virtio_sound.rs @@ -401,7 +401,7 @@ fn answer_into_a_foreign_page( shared: Dma<'static>, ) { const FOREIGN_DESC: usize = 2; - let foreign = super::nvme::FOREIGN_PROBE.load(Ordering::Relaxed); + let foreign = super::xhci::FOREIGN_PROBE.load(Ordering::Relaxed); assert!(foreign != 0, "virtio-sound: this machine staged no foreign pool to aim at"); let slot = controlq.initial_slots().swap_remove(FOREIGN_DESC); controlq.submit( diff --git a/kernel/src/drivers/xhci/mod.rs b/kernel/src/drivers/xhci/mod.rs index f5c2efb8bea..c297239fc93 100644 --- a/kernel/src/drivers/xhci/mod.rs +++ b/kernel/src/drivers/xhci/mod.rs @@ -491,6 +491,20 @@ const PAGE: usize = 0x1000; // The pool's fixed head: one of each, since enumeration is serial — see `device::init_device`. #[allow(clippy::erasing_op)] const OFF_DCBAA: usize = 0 * PAGE; // (max_slots + 1) * 8, 2 KiB at most + +/// The half of the DCBAA's page no slot reaches — its at most 256 entries +/// fill the first — zeroed at bring-up and never written: another device's +/// IOMMU control is aimed here, and it must still read zero afterwards. +#[cfg(feature = "boot-actuators")] +pub(crate) const PROBE_OFF: usize = OFF_DCBAA + 0x800; +#[cfg(feature = "boot-actuators")] +pub const PROBE_LEN: usize = 0x800; + +/// Physical, not what this controller is programmed with: the actuator has to +/// hand another device an address that device's own domain does not map. The +/// first controller's. +#[cfg(feature = "boot-actuators")] +pub static FOREIGN_PROBE: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0); #[allow(clippy::identity_op)] const OFF_CMD_RING: usize = 1 * PAGE; const OFF_ERST: usize = 2 * PAGE; diff --git a/kernel/src/drivers/xhci/wait/boot.rs b/kernel/src/drivers/xhci/wait/boot.rs index e815cff2004..8a94901cc9c 100644 --- a/kernel/src/drivers/xhci/wait/boot.rs +++ b/kernel/src/drivers/xhci/wait/boot.rs @@ -361,6 +361,13 @@ fn init_one(pci_dev: &PciDevice) -> Option { } op_base.write_u64(OP_DCBAAP, dma.device_addr() + OFF_DCBAA as u64); + #[cfg(feature = "boot-actuators")] + let _ = super::super::FOREIGN_PROBE.compare_exchange( + 0, + dma.subview(super::super::PROBE_OFF, super::super::PROBE_LEN).host_phys(), + core::sync::atomic::Ordering::Relaxed, + core::sync::atomic::Ordering::Relaxed, + ); // CRCR bit 0 is RCS; the pointer is 64-byte aligned so `| 1` only sets // that bit (xHCI 1.2 §5.4.5). diff --git a/kernel/src/fat32_adapter.rs b/kernel/src/fat32_adapter.rs deleted file mode 100644 index 6ad2b748d51..00000000000 --- a/kernel/src/fat32_adapter.rs +++ /dev/null @@ -1,1276 +0,0 @@ -//! The two FAT32 partitions this kernel mounts: the boot volume the firmware -//! loaded from, and the log volume beside it. -//! -//! A volume is mounted only when three checks pass: [`gpt::boot_volume`] and -//! [`gpt::log_volume`] name it from the handoff, never by scanning for a FAT -//! signature or a partition type; [`FatDevice`] clamps every read and write to -//! the volume and never the wider partition; and `toyos-fat32` never writes a -//! BPB, so a volume that fails to parse is left untouched. - -use alloc::format; -use alloc::string::String; -use alloc::collections::BTreeMap; -use alloc::sync::{Arc, Weak}; -use alloc::vec; -use alloc::vec::Vec; -use core::sync::atomic::{AtomicBool, Ordering}; -use crate::hasher::HashMap; - -use toyos_abi::syscall::SyscallError; -use toyos_fat32::{BlockAccess, Error, Extent, Fat32, FatTime, IoError, RepairNotice}; - -use crate::block; -use crate::mm::PAGE_BYTES; -use crate::drivers::{usb_storage, xhci}; -use crate::file_backing::FileBacking; -use crate::file_cache::{self, FileId}; -use crate::fs_rename::{self, Committed, ReplaceRename}; -use crate::gpt; -use crate::sync::Lock; -use crate::vfs::FileSystem; - -/// The only transfer unit [`block::BlockDevice`] has, which is `mm::PAGE_SIZE`. -const BLOCK: u64 = crate::mm::PAGE_SIZE; - -/// Which of the two partitions a mount is. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum Role { - /// The partition firmware loaded the bootloader from. - Boot, - /// The log partition, kept off the ESP so macOS mounts it. - Log, -} - -impl Role { - /// The VFS mount name, which is also the top-level directory. - pub fn mount(self) -> &'static str { - match self { - Role::Boot => "boot", - Role::Log => "log", - } - } - - fn slot(self) -> usize { - match self { - Role::Boot => 0, - Role::Log => 1, - } - } - - fn volume(self) -> Option { - match self { - Role::Boot => gpt::boot_volume(), - Role::Log => gpt::log_volume(), - } - } -} - -impl core::fmt::Display for Role { - fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - f.write_str(self.mount()) - } -} - -/// Bounded so the extent-list `Vec` stays under `mm::MAX_HEAP_ALLOC`. -const MAX_EXTENTS: usize = 65_536; - -const _: () = assert!(core::mem::size_of::() == 16); - -/// One volume, as a byte range over a partition view that only does whole -/// 4 KiB blocks; every offset here is the volume's own. -struct FatDevice { - part: block::Partition, - /// How many bytes of the partition this volume uses. - len: u64, - /// On the heap: the deepest caller is the idle loop, whose 16 KiB stack - /// has no guard page. - scratch: Vec, - /// [`RESIDENT_BLOCKS`] blocks, each tagged with the block it holds. - resident: Vec, - tags: [Option; RESIDENT_BLOCKS], - /// Round-robin: the access pattern touches every resident block per - /// operation, so recency ranks nothing. - next_victim: usize, -} - -/// Blocks [`FatDevice`] keeps resident: enough for one append's FAT, mirror -/// FAT, directory, FSInfo and data blocks. -const RESIDENT_BLOCKS: usize = 8; - -/// One function and not `From`: this crosses the crate boundary in the -/// direction that has no orphan. -fn as_io_error(e: crate::block::BlockError) -> IoError { - match e { - crate::block::BlockError::Device => IoError::Device, - crate::block::BlockError::BudgetExpired => IoError::BudgetExpired, - } -} - -impl FatDevice { - /// `offset` itself once it is inside the volume, or [`IoError::Device`] - /// past it. The partition view refuses anything past the partition; this is - /// the tighter of the two bounds, the volume's own. - fn locate(&self, offset: u64, len: usize) -> Result { - let end = offset.checked_add(len as u64).ok_or(IoError::Device)?; - if end > self.len { - return Err(IoError::Device); - } - Ok(offset) - } - - fn slot_of(&self, block: u64) -> Option { - self.tags.iter().position(|&t| t == Some(block)) - } - - /// Leave `block` in `scratch`, reading it only if it is not already here. - fn load(&mut self, block: u64) -> Result<(), IoError> { - if let Some(slot) = self.slot_of(block) { - let at = slot * BLOCK as usize; - self.scratch.copy_from_slice(&self.resident[at..at + BLOCK as usize]); - return Ok(()); - } - let Self { part, scratch, .. } = self; - part.read_blocks(block, 1, scratch).map_err(as_io_error)?; - self.retain(block); - Ok(()) - } - - /// Record `scratch` as this device's `block`; only called where the - /// device already holds those bytes. - fn retain(&mut self, block: u64) { - let slot = self.slot_of(block).unwrap_or_else(|| { - let s = self.next_victim; - self.next_victim = (s + 1) % RESIDENT_BLOCKS; - s - }); - let at = slot * BLOCK as usize; - self.resident[at..at + BLOCK as usize].copy_from_slice(&self.scratch); - self.tags[slot] = Some(block); - } - - fn forget(&mut self, first: u64, count: u64) { - for tag in &mut self.tags { - if tag.is_some_and(|b| b >= first && b < first + count) { - *tag = None; - } - } - } - - fn read_at(&mut self, offset: u64, buf: &mut [u8]) -> Result<(), IoError> { - let base = self.locate(offset, buf.len())?; - let mut done = 0usize; - while done < buf.len() { - let at = base + done as u64; - let block = at / BLOCK; - let within = (at % BLOCK) as usize; - let left = buf.len() - done; - if within == 0 && left >= BLOCK as usize { - let count = left / BLOCK as usize; - let end = done + count * BLOCK as usize; - self.part - .read_blocks(block, count as u32, &mut buf[done..end]) - .map_err(as_io_error)?; - done = end; - } else { - let n = (BLOCK as usize - within).min(left); - self.load(block)?; - buf[done..done + n].copy_from_slice(&self.scratch[within..within + n]); - done += n; - } - } - Ok(()) - } - - fn write_at(&mut self, offset: u64, buf: &[u8]) -> Result<(), IoError> { - let base = self.locate(offset, buf.len())?; - let mut done = 0usize; - while done < buf.len() { - let at = base + done as u64; - let block = at / BLOCK; - let within = (at % BLOCK) as usize; - let left = buf.len() - done; - if within == 0 && left >= BLOCK as usize { - let count = left / BLOCK as usize; - let end = done + count * BLOCK as usize; - self.forget(block, count as u64); - self.part - .write_blocks(block, count as u32, &buf[done..end]) - .map_err(as_io_error)?; - done = end; - } else { - // Bytes this request doesn't cover belong to another file or - // the partition table; preserved by the read-modify-write. - let n = (BLOCK as usize - within).min(left); - self.load(block)?; - self.scratch[within..within + n].copy_from_slice(&buf[done..done + n]); - let Self { part, scratch, .. } = self; - part.write_blocks(block, 1, scratch).map_err(as_io_error)?; - self.retain(block); - done += n; - } - } - Ok(()) - } -} - -/// Lock order: VFS → here → the device handle → `XHCI`; never the other way, -/// and never two of these at once. A page cache is the other holder of a device -/// handle and never takes one of these, so the two orders do not meet. -static VOLUMES: [Lock>; 2] = [Lock::new(None), Lock::new(None)]; - -fn device(role: Role) -> &'static Lock> { - &VOLUMES[role.slot()] -} - -/// One [`VOLUMES`] entry in the shape `toyos-fat32` asks for. -pub struct FatVolume { - role: Role, - bytes: u64, -} - -impl BlockAccess for FatVolume { - fn capacity(&self) -> u64 { - self.bytes - } - - fn read_at(&mut self, offset: u64, buf: &mut [u8]) -> Result<(), IoError> { - let mut guard = device(self.role).lock(); - let served = guard.as_mut().ok_or(IoError::Device)?.read_at(offset, buf); - if injected_read_failure(self.role) { - // Zeroed to match what a real failed read leaves behind. - buf.fill(0); - log!("{}-volume: read of {} B at volume offset {offset} failed", - self.role, buf.len()); - return Err(IoError::Device); - } - served - } - - fn write_at(&mut self, offset: u64, buf: &[u8]) -> Result<(), IoError> { - #[cfg(feature = "boot-actuators")] - if self.role == Role::Log { - match mirror_refuse::should_refuse(offset, buf.len()) { - None => {} - Some(mirror_refuse::Refused::Drain) => { - log!( - "log-volume: fat-mirror-write-refuse: refusing the FAT-1 mirror write \ - of a drain flush at volume offset {offset} as a budget expiry" - ); - return Err(IoError::BudgetExpired); - } - Some(mirror_refuse::Refused::ShutdownDrain { attempt }) => { - log!( - "log-volume: quiesce-drain-refuse: refusing the shutdown drain's FAT-1 \ - mirror write as a budget expiry, {attempt} of {}", - mirror_refuse::SHUTDOWN_REFUSALS - ); - return Err(IoError::BudgetExpired); - } - Some(mirror_refuse::Refused::Fsync { attempt: Some(attempt) }) => { - log!( - "log-volume: quiesce-fsync-refuse: refusing a SYS_FSYNC flush's \ - active-FAT write at volume offset {offset} as a budget expiry, its \ - mirror already written, {attempt} of {}", - mirror_refuse::FSYNC_REFUSALS - ); - return Err(IoError::BudgetExpired); - } - Some(mirror_refuse::Refused::Fsync { attempt: None }) => { - return Err(IoError::BudgetExpired); - } - } - } - let mut guard = device(self.role).lock(); - guard.as_mut().ok_or(IoError::Device)?.write_at(offset, buf) - } - - fn flush(&mut self) -> Result<(), IoError> { - let mut guard = device(self.role).lock(); - guard.as_mut().ok_or(IoError::Device)?.part.flush().map_err(as_io_error) - } -} - -/// Whether a page re-read succeeds; the read is still issued and only its -/// verdict replaced, so a broken transport cannot hide behind it. -fn fat_backing_reads() -> bool { - !crate::actuator::fat_backing_read_fails() -} - -/// Whether the boot volume may still answer a filesystem read once mounted; -/// the metadata-path sibling of [`fat_backing_reads`], which is the -/// page-fault path. -fn boot_volume_reads() -> bool { - !crate::actuator::fat_boot_reads_fail() -} - -/// Set once [`mount`] has installed the boot volume, so the injection cannot -/// fire during mount itself. -static BOOT_MOUNTED: AtomicBool = AtomicBool::new(false); - -// Boot only: the log volume carries the kernel's own log, and failing its -// reads would take the channel the evidence arrives on. -fn injected_read_failure(role: Role) -> bool { - !boot_volume_reads() && role == Role::Boot && BOOT_MOUNTED.load(Ordering::Relaxed) -} - -/// Armed only around the leak-rollback self-test's reopen, to fail [`FatFs::backing`]. -#[cfg(feature = "boot-actuators")] -static SELFTEST_BACKING_FAIL: AtomicBool = AtomicBool::new(false); - -/// Self-test hook: make [`FatFs::backing`] fail like a transient device error. -#[cfg(feature = "boot-actuators")] -pub(crate) fn selftest_fail_backing(on: bool) { - SELFTEST_BACKING_FAIL.store(on, Ordering::Relaxed); -} - -/// Which byte ranges hold one file's data, shared by every [`FatBacking`] for -/// that name so an unlink revokes all of them at once. -struct FatExtents { - /// `None` once the volume has the clusters back. - runs: Lock>>, -} - -/// Where one file offset is, once [`FatExtents`] has been asked. -enum Located { - /// Volume byte offset, and how many contiguous bytes follow. - Run(u64, u64), - /// Past the extent list: the file has no data there; zeros are its own bytes. - Hole, -} - -impl FatExtents { - fn new(runs: Vec) -> Arc { - Arc::new(Self { runs: Lock::new(Some(runs)) }) - } - - /// Give the ranges up; every read through a sharing backing fails from - /// here on. - fn revoke(&self) { - *self.runs.lock() = None; - } - - /// Take a fresh [`Fat32::extents`] list; write-through so every sharer's - /// cell sees it. - fn refresh(&self, runs: Vec) { - *self.runs.lock() = Some(runs); - } - - /// Give up bytes past `len`; runs before `Fat32::set_len` frees the tail - /// so no backing ever names a reissued cluster. - fn truncate_to(&self, len: u64) { - let mut guard = self.runs.lock(); - let Some(runs) = guard.as_mut() else { return }; - let mut kept = 0u64; - runs.retain_mut(|run| { - let room = len.saturating_sub(kept); - run.len = run.len.min(room); - kept += run.len; - run.len != 0 - }); - } - - /// Where `file_offset` is on the volume, or `None` once the file is gone; - /// the lock is a leaf, never held across a device read. - fn locate(&self, file_offset: u64) -> Option { - let guard = self.runs.lock(); - let runs = guard.as_ref()?; - let mut base = 0u64; - for run in runs { - if file_offset < base + run.len { - let within = file_offset - base; - return Some(Located::Run(run.offset + within, run.len - within)); - } - base += run.len; - } - Some(Located::Hole) - } -} - -/// The actuators that refuse a flush's FAT write as a budget expiry. Two -/// refuse a drain flush's FAT-1 mirror write, which leaves the volume -/// untouched: `fat-mirror-write-refuse` the first two, `quiesce-drain-refuse` -/// [`mirror_refuse::SHUTDOWN_REFUSALS`] on the thread running the shutdown. -/// `quiesce-fsync-refuse` refuses the *active* FAT's write in -/// [`mirror_refuse::FSYNC_REFUSALS`] attempts of a `SYS_FSYNC` of the one file -/// it stages, [`FSYNC_STAGED`] — after `set_fat_entry` wrote the mirror, so the -/// two FATs stand split until the caller's next attempt, and a stop asked for -/// meanwhile meets the caller parked between two of them. -#[cfg(feature = "boot-actuators")] -mod mirror_refuse { - use core::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, Ordering}; - - /// The log volume's FAT-1 byte range, volume-relative, captured at mount. - static LO: AtomicU64 = AtomicU64::new(0); - static HI: AtomicU64 = AtomicU64::new(0); - /// Set while `writeback`'s drain holds a `flush_file` open, so refusal - /// targets only the drain path. - static IN_DRAIN: AtomicBool = AtomicBool::new(false); - /// How many drain-flush mirror writes have been refused so far. - static REFUSED: AtomicU32 = AtomicU32::new(0); - /// How many of the shutdown's own drain-flush mirror writes have been. - static SHUTDOWN_REFUSED: AtomicU32 = AtomicU32::new(0); - - /// Two, not one: a single refusal only reaches the retry ladder's attempt - /// 1, which merely yields — attempt 2 is the one that parks. - const REFUSALS: u32 = 2; - - /// Eight, because the ladder parks from attempt 2. - pub const SHUTDOWN_REFUSALS: u32 = 8; - - /// Enough that the stop's sweeps meet the caller parked between two - /// refused attempts, and few enough that the ladder closes inside the - /// stop's budget. - pub const FSYNC_REFUSALS: u32 = 8; - - /// The parks between [`FSYNC_REFUSALS`] refused attempts: none after the - /// first, then `RETRY_SOONEST` doubling, never reaching its ceiling — and - /// all of them inside the stop's budget, so the stop waits the update out - /// rather than resetting over a half-made one. - const _: () = assert!( - crate::block::RETRY_SOONEST.nanos() * ((1 << (FSYNC_REFUSALS - 1)) - 1) - < crate::quiesce::PARK.nanos() - && crate::block::RETRY_SOONEST.nanos() << (FSYNC_REFUSALS - 2) - <= crate::block::RETRY_SLOWEST.nanos(), - "quiesce-fsync-refuse: the refused ladder outlasts the stop's budget", - ); - - /// The active FAT's byte range, captured beside the mirror's. - static ACTIVE_LO: AtomicU64 = AtomicU64::new(0); - static ACTIVE_HI: AtomicU64 = AtomicU64::new(0); - /// Set while one `SYS_FSYNC` attempt holds its flush open, under the VFS - /// lock every FAT write is made under. - static IN_FSYNC: AtomicBool = AtomicBool::new(false); - /// Whether the open attempt has been refused already: every active-FAT - /// write of a refused attempt is, its rollback's included. - static FSYNC_ATTEMPT_REFUSED: AtomicBool = AtomicBool::new(false); - /// How many `SYS_FSYNC` attempts have been refused so far. - static FSYNC_REFUSED: AtomicU32 = AtomicU32::new(0); - - pub enum Refused { - /// `fat-mirror-write-refuse`. - Drain, - /// `quiesce-drain-refuse`, on the thread running the shutdown. - ShutdownDrain { attempt: u32 }, - /// `quiesce-fsync-refuse`; `None` on a refused attempt's later writes, - /// which are refused in silence. - Fsync { attempt: Option }, - } - - pub fn capture(active: (u64, u64), mirror: (u64, u64)) { - ACTIVE_LO.store(active.0, Ordering::Relaxed); - ACTIVE_HI.store(active.1, Ordering::Relaxed); - LO.store(mirror.0, Ordering::Relaxed); - HI.store(mirror.1, Ordering::Relaxed); - } - - /// Whether one `SYS_FSYNC` attempt of the staged file holds its flush open. - pub fn set_in_fsync(on: bool) { - FSYNC_ATTEMPT_REFUSED.store(false, Ordering::Relaxed); - IN_FSYNC.store(on, Ordering::Relaxed); - } - - fn fsync_refuses(offset: u64, end: u64) -> Option { - if !IN_FSYNC.load(Ordering::Relaxed) || !crate::actuator::quiesce_fsync_refuse() { - return None; - } - let (lo, hi) = (ACTIVE_LO.load(Ordering::Relaxed), ACTIVE_HI.load(Ordering::Relaxed)); - if !(hi > lo && offset < hi && end > lo) { - return None; - } - if FSYNC_ATTEMPT_REFUSED.load(Ordering::Relaxed) { - return Some(Refused::Fsync { attempt: None }); - } - if FSYNC_REFUSED.load(Ordering::Relaxed) >= FSYNC_REFUSALS { - return None; - } - FSYNC_ATTEMPT_REFUSED.store(true, Ordering::Relaxed); - let attempt = FSYNC_REFUSED.fetch_add(1, Ordering::Relaxed) + 1; - Some(Refused::Fsync { attempt: Some(attempt) }) - } - - pub fn set_in_drain(on: bool) { - IN_DRAIN.store(on, Ordering::Relaxed); - } - - /// Which actuator refuses this log-volume write, if one does: mid-drain, - /// overlapping the mirror FAT, and under that actuator's own count. - pub fn should_refuse(offset: u64, len: usize) -> Option { - let end = offset + len as u64; - if let Some(refused) = fsync_refuses(offset, end) { - return Some(refused); - } - if !IN_DRAIN.load(Ordering::Relaxed) { - return None; - } - let (lo, hi) = (LO.load(Ordering::Relaxed), HI.load(Ordering::Relaxed)); - if !(hi > lo && offset < hi && end > lo) { - return None; - } - if crate::actuator::fat_mirror_write_refuse() && REFUSED.load(Ordering::Relaxed) < REFUSALS - { - REFUSED.fetch_add(1, Ordering::Relaxed); - return Some(Refused::Drain); - } - if crate::actuator::quiesce_drain_refuse() - && crate::quiesce::runs_the_shutdown() - && SHUTDOWN_REFUSED.load(Ordering::Relaxed) < SHUTDOWN_REFUSALS - { - let attempt = SHUTDOWN_REFUSED.fetch_add(1, Ordering::Relaxed) + 1; - return Some(Refused::ShutdownDrain { attempt }); - } - None - } -} - -/// The `fat-flush-meta-refuse` actuator: refuses one file's second -/// directory-entry write, which is the last step of a flush and the one whose -/// failure leaves the pages before it already written and settled. The second -/// and not the first, because the first is that file's seed being made durable. -#[cfg(feature = "boot-actuators")] -mod meta_refuse { - use core::sync::atomic::{AtomicU32, Ordering}; - - /// Mirrored in `tests/toyos-rust-tests/src/bin/writeback_durability.rs`. - const STAGED: &str = "wb-retry.bin"; - /// Which of this file's metadata writes is refused, counting from zero. - const AT: u32 = 1; - static SEEN: AtomicU32 = AtomicU32::new(0); - - pub fn should_refuse(name: &str) -> bool { - crate::actuator::fat_flush_meta_refuse() - && name.ends_with(STAGED) - && SEEN.fetch_add(1, Ordering::Relaxed) == AT - } -} - -/// Mark that a write-back drain flush is in progress, so the mirror-refuse -/// actuator targets the drain path and not `SYS_FSYNC`. -#[cfg(feature = "boot-actuators")] -pub(crate) fn enter_drain_flush() { - mirror_refuse::set_in_drain(true); -} - -#[cfg(feature = "boot-actuators")] -pub(crate) fn leave_drain_flush() { - mirror_refuse::set_in_drain(false); -} - -/// The one file `quiesce-fsync-refuse` refuses the flush of. Mirrored in -/// `tests/toyos-rust-tests/src/bin/quiesce_fsync.rs`. -#[cfg(feature = "boot-actuators")] -const FSYNC_STAGED: &str = "quiesce-fsync.bin"; - -/// The same for one `SYS_FSYNC` attempt on `path`; called with the VFS lock -/// held. -#[cfg(feature = "boot-actuators")] -pub(crate) fn enter_fsync_flush(path: &str) { - mirror_refuse::set_in_fsync(path.ends_with(FSYNC_STAGED)); -} - -#[cfg(feature = "boot-actuators")] -pub(crate) fn leave_fsync_flush() { - mirror_refuse::set_in_fsync(false); -} - -/// A file's byte ranges, read without going back through the filesystem; -/// `size` is a snapshot, not a view of the live length. -struct FatBacking { - role: Role, - extents: Arc, - size: u64, -} - -impl FileBacking for FatBacking { - fn read_page(&self, file_offset: u64, buf: &mut [u8; PAGE_BYTES]) -> crate::block::BlockResult { - buf.fill(0); - if file_offset >= self.size { - return Ok(()); - } - let valid = (4096u64).min(self.size - file_offset) as usize; - let mut done = 0usize; - // A page can span multiple runs when the cluster size is under 4096. - while done < valid { - // Re-asked every run, so a mid-page revocation stops the rest of - // it here. - let Some(found) = self.extents.locate(file_offset + done as u64) else { - // Clusters already back in the allocator; reading them would - // serve another file's data. - log!("{}-volume: read through a backing whose file was deleted", self.role); - return Err(crate::block::BlockError::Device); - }; - let Located::Run(at, run) = found else { - // Past the extent list: a hole; the zeros already in `buf` - // are correct. - return Ok(()); - }; - let n = (run as usize).min(valid - done); - let mut guard = device(self.role).lock(); - let Some(dev) = guard.as_mut() else { - // Not silent, unlike elsewhere: `serving zeros` is the string - // a triage greps for. - log!("{}-volume: not mounted; serving zeros", self.role); - return Err(crate::block::BlockError::Device); - }; - let served = dev.read_at(at, &mut buf[done..done + n]); - if !fat_backing_reads() { - // Zeroed to match what a real failed read leaves behind. - buf[done..done + n].fill(0); - } - if served.is_err() || !fat_backing_reads() { - log!("{}-volume: read of {n} B at volume offset {at} failed; serving zeros", - self.role); - return Err(crate::block::BlockError::Device); - } - drop(guard); - done += n; - } - Ok(()) - } - - fn file_size(&self) -> u64 { - self.size - } -} - -/// Per-open-file state: the path, and the crate's handle caching directory -/// location and chain position. -struct OpenFile { - name: String, - file: toyos_fat32::File, -} - -/// VFS adapter for one of the two partitions; a file's identity is its path, -/// which `by_name` keys on. -pub struct FatFs { - role: Role, - fs: Fat32, - open: HashMap, - by_name: BTreeMap, - /// The one [`FatExtents`] every backing for a name shares; keyed by name - /// because `open_backing` hands one out without opening a file. - extents: BTreeMap>, - /// Which pending repair a log line last named. - repair_named: RepairNotice, -} - -/// What to stamp on an entry: reads `clock` directly, in local time as FAT -/// requires — the VFS's `mtime` is nanoseconds since boot, not a time of day. -fn now() -> FatTime { - crate::clock::local_secs().map_or(FatTime::EPOCH, FatTime::from_unix_secs) -} - -/// What one of `toyos-fat32`'s errors means to the [`FileSystem`] caller; -/// exhaustive so a new variant fails to compile here. -fn as_syscall_error(e: Error) -> SyscallError { - match e { - Error::NotFound => SyscallError::NotFound, - Error::AlreadyExists => SyscallError::AlreadyExists, - // Structural corruption reads as `Io`, not `NotFound`: the volume - // can't say what's there, not that nothing is. - Error::Io - | Error::NotFat32 - | Error::Truncated - | Error::CorruptChain - | Error::CorruptDirectory => SyscallError::Io, - Error::BudgetExpired | Error::RepairPending => SyscallError::WouldBlock, - // Not `NotFound`: the name resolves; the operation just isn't defined - // for what it names. - Error::NotADirectory | Error::IsADirectory | Error::DirectoryNotEmpty => { - SyscallError::InvalidArgument - } - Error::InvalidName => SyscallError::InvalidArgument, - // `TooLarge` is FAT32's 4 GiB field limit, not a full volume, but both - // mean no room. - Error::NoSpace | Error::TooLarge => SyscallError::ResourceExhausted, - Error::LimitExceeded => SyscallError::ResourceExhausted, - } -} - -/// Leave one handle's file consistent on the volume, or say why it could not -/// be. -/// -/// **The two writes a growing write makes are the chain and then the directory -/// entry**, and everything between them is a volume holding clusters the entry -/// does not reach — which is what `toyos-fat32-check` refuses. Nothing later -/// repaired it, because nothing looked: a closed handle was dropped and a sync -/// only flushed the device. -/// -/// Logged and not returned. A close has no caller to answer, and a sync's -/// answer is about the device; the repair either happened or the volume already -/// held the inconsistency this is trying to remove. -fn reconcile(role: Role, fs: &mut Fat32, named: &mut RepairNotice, info: &mut OpenFile) { - if !info.file.needs_reconcile() { - return; - } - match fs.reconcile(&mut info.file, now()) { - Ok(()) => {} - Err(e) => { - if !name_pending(role, fs, named, || format!("reconcile of {}", info.name), e) { - log!("{role}-volume: {} was left with a chain its entry does not reach: {e}", info.name) - } - } - } -} - -/// Whether `e` is the volume waiting on an unlanded repair, logged once per -/// repair, since otherwise it reads as whatever call met it failing. `what` is -/// only called to build the logged name once there is something to log, so a -/// call that is neither pending nor its repair's first sighting allocates -/// nothing. -fn name_pending( - role: Role, - fs: &Fat32, - named: &mut RepairNotice, - what: impl FnOnce() -> String, - e: Error, -) -> bool { - let episode = fs.repair_episode(); - if !RepairNotice::waits_on(e, episode) { - return false; - } - if named.first_sight(episode) { - log!( - "{role}-volume: {} refused, a repair pending with {} step(s) queued: {e}", - what(), - fs.pending_repair() - ); - } - true -} - -/// Log what the volume said and return its code; `NotFound` is skipped so -/// opening a missing path doesn't write the log it lives on, and a call a -/// pending repair refused is [`name_pending`]'s to name. -fn refused( - role: Role, - fs: &Fat32, - named: &mut RepairNotice, - op: &str, - name: &str, - e: Error, -) -> SyscallError { - if e != Error::NotFound && !name_pending(role, fs, named, || format!("{op} of {name}"), e) { - log!("{role}-volume: {op} of {name}: {e}"); - } - as_syscall_error(e) -} - -impl FatFs { - fn new(role: Role, fs: Fat32) -> Self { - Self { - role, - fs, - open: HashMap::default(), - by_name: BTreeMap::new(), - extents: BTreeMap::new(), - repair_named: RepairNotice::default(), - } - } - - fn backing(&mut self, name: &str) -> Result, SyscallError> { - let role = self.role; - // Self-test: the transient-device-error trigger of the reopen leak. - #[cfg(feature = "boot-actuators")] - if SELFTEST_BACKING_FAIL.load(Ordering::Relaxed) { - return Err(SyscallError::Io); - } - let size = self.fs.metadata(name).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "metadata", name, e))?.len; - let runs = self - .fs - .extents(name, MAX_EXTENTS) - .map_err(|e| refused(role, &self.fs, &mut self.repair_named, "extents", name, e))?; - let extents = self.extents_for(name, runs); - Ok(Arc::new(FatBacking { role, extents, size })) - } - - /// The cell every backing for `name` reads through, carrying the - /// just-read extent list. - fn extents_for(&mut self, name: &str, runs: Vec) -> Arc { - // Swept here, not on a timer: this call is the only place the map - // grows. - self.extents.retain(|_, weak| weak.strong_count() > 0); - - if let Some(live) = self.extents.get(name).and_then(Weak::upgrade) { - live.refresh(runs); - return live; - } - let cell = FatExtents::new(runs); - self.extents.insert(String::from(name), Arc::downgrade(&cell)); - cell - } - - /// The live cell for `name`, if some backing still holds one. - fn live_extents(&self, name: &str) -> Option> { - self.extents.get(name).and_then(Weak::upgrade) - } - - /// Give up every backing reading `name`'s clusters; must be called before - /// the clusters are freed, since [`FatBacking::read_page`] takes no VFS - /// lock. - fn revoke(&mut self, name: &str) { - if let Some(cell) = self.extents.remove(name).as_ref().and_then(Weak::upgrade) { - cell.revoke(); - } - } - - /// Make sure every directory on the way to `name` exists: a create may - /// land under a path no `mkdir` ever touched. - fn ensure_parent(&mut self, name: &str, time: FatTime) -> Result<(), SyscallError> { - let Some((parent, _)) = name.rsplit_once('/') else { return Ok(()) }; - let role = self.role; - self.fs.create_dir_all(parent, time).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "mkdir -p", parent, e)) - } -} - -/// FAT cannot replace an entry in one step. The backend renames the destination -/// aside, moves the source, and returns the still-live displaced entry; only -/// `release` may retire its in-memory state and free its clusters. -impl ReplaceRename for FatFs { - type Displaced = (toyos_fat32::Replaced, Option); - - fn source_present(&mut self, old: &str) -> Result { - let role = self.role; - self.fs.exists(old).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "exists", old, e)) - } - - fn same_object(&mut self, old: &str, new: &str) -> Result { - // Identity is the entry's location: FAT names one entry by two strings. - let role = self.role; - self.fs.same_entry(old, new).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "same_entry", old, e)) - } - - fn commit( - &mut self, - old: &str, - new: &str, - ) -> Result, SyscallError> { - let role = self.role; - let displaced = self.by_name.get(new).copied(); - let replaced = self.fs.replace_rename(old, new).map_err(|e| { - // Nothing on the volume names it, so this line is the only record - // that the destination's data is alive and unreachable. - if let Some(stranded) = &e.stranded { - log!("{role}-volume: {new} could not be put back and is under {stranded}"); - } - refused(role, &self.fs, &mut self.repair_named, "replace rename", old, e.cause) - })?; - Ok(Committed::new((replaced, displaced))) - } - - fn release( - &mut self, - old: &str, - new: &str, - committed: Committed, - ) -> Result<(), SyscallError> { - let (replaced, displaced) = committed.into_displaced(); - let had_destination = replaced.displaced(); - if let Some(file_id) = displaced { - let _ = file_cache::mark_deleted(file_id); - self.open.remove(&file_id); - self.by_name.remove(new); - } - if had_destination { - // Backings under `new` still read the displaced file's clusters. - self.revoke(new); - } - - let role = self.role; - let released = self - .fs - .release_replaced(replaced) - .map_err(|e| refused(role, &self.fs, &mut self.repair_named, "release replaced", new, e)); - - // Re-key, not revoke: the source's data did not move, so backings under - // the old name still read it. - if let Some(file_id) = self.by_name.remove(old) { - self.by_name.insert(String::from(new), file_id); - if let Some(info) = self.open.get_mut(&file_id) { - info.name = String::from(new); - } - } - if let Some(cell) = self.extents.remove(old) { - self.extents.insert(String::from(new), cell); - } - released - } -} - -impl FileSystem for FatFs { - /// The `limit` bound is honoured before each push, not after — unlike the - /// bcachefs adapters. - fn list(&mut self, dir: &str, limit: usize) -> Result, SyscallError> { - let role = self.role; - self.fs.walk(dir, limit).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "list", dir, e)) - } - - fn is_dir(&mut self, dir: &str) -> Result { - let role = self.role; - match self.fs.metadata(dir) { - Ok(meta) => Ok(meta.is_dir), - Err(Error::NotFound | Error::NotADirectory) => Ok(false), - Err(e) => Err(refused(role, &self.fs, &mut self.repair_named, "metadata", dir, e)), - } - } - - fn file_mtime(&mut self, name: &str) -> Result { - let role = self.role; - self.fs - .metadata(name) - .map(|m| m.modified_unix) - .map_err(|e| refused(role, &self.fs, &mut self.repair_named, "metadata", name, e)) - } - - /// Always `Ok(None)`: FAT32 has no symlink representation. - fn read_link(&mut self, _name: &str) -> Result, SyscallError> { - Ok(None) - } - - fn open_file(&mut self, name: &str) -> Result<(FileId, Option>), SyscallError> { - if let Some(&file_id) = self.by_name.get(name) { - let held = file_cache::open(file_id); - let backing = self.backing(name)?; - held.commit(); - return Ok((file_id, Some(backing))); - } - let role = self.role; - let file = self.fs.open(name).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "open", name, e))?; - let size = file.len(); - let backing = self.backing(name)?; - - let file_id = file_cache::create_file(true); - file_cache::set_size(file_id, size); - self.by_name.insert(String::from(name), file_id); - self.open.insert(file_id, OpenFile { name: String::from(name), file }); - Ok((file_id, Some(backing))) - } - - fn create(&mut self, name: &str, _mtime: u64) -> Result { - if let Some(&file_id) = self.by_name.get(name) { - return Ok(file_id); - } - let role = self.role; - let time = now(); - self.ensure_parent(name, time)?; - let file = match self.fs.create(name, time) { - Ok(file) => file, - // Not an error here: `create` also reopens an existing file for - // writing, deliberately — a create that silently opened somebody - // else's file is how a caller comes to believe it owns bytes it - // does not. - Err(Error::AlreadyExists) => { - self.fs.open(name).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "open", name, e))? - } - Err(e) => return Err(refused(role, &self.fs, &mut self.repair_named, "create", name, e)), - }; - let file_id = file_cache::create_file(true); - file_cache::set_size(file_id, file.len()); - self.by_name.insert(String::from(name), file_id); - self.open.insert(file_id, OpenFile { name: String::from(name), file }); - Ok(file_id) - } - - fn close_file(&mut self, file_id: FileId) { - if let Some(mut info) = self.open.remove(&file_id) { - reconcile(self.role, &mut self.fs, &mut self.repair_named, &mut info); - self.by_name.remove(&info.name); - } - } - - /// Unlink, giving up both the write handle and every read backing - /// unconditionally — a held-open handle must not read or write the next - /// file's data. - fn delete(&mut self, name: &str) -> Result<(), SyscallError> { - if let Some(file_id) = self.by_name.remove(name) { - let _ = file_cache::mark_deleted(file_id); - self.open.remove(&file_id); - } - self.revoke(name); - let role = self.role; - self.fs.remove(name).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "delete", name, e)) - } - - fn rename(&mut self, old: &str, new: &str) -> Result<(), SyscallError> { - fs_rename::replace_rename(self, old, new) - } - - fn create_dir(&mut self, name: &str) -> Result<(), SyscallError> { - let role = self.role; - self.fs.create_dir(name, now()).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "mkdir", name, e)) - } - - fn remove_dir(&mut self, name: &str) -> Result<(), SyscallError> { - let role = self.role; - self.fs.remove_dir(name).map_err(|e| refused(role, &self.fs, &mut self.repair_named, "rmdir", name, e)) - } - - fn write_page( - &mut self, - file_id: FileId, - page_idx: u32, - data: &[u8; PAGE_BYTES], - ) -> Result<(), SyscallError> { - let Self { role, fs, open, repair_named, .. } = self; - let role = *role; - let info = open.get_mut(&file_id).ok_or(SyscallError::NotFound)?; - match fs.write(&mut info.file, page_idx as u64 * 4096, data) { - Ok(()) => Ok(()), - Err(e) => Err(refused(role, fs, repair_named, "write", &info.name, e)), - } - } - - /// `set_len` frees the clusters and the cell stops naming them, so a later - /// grow zero-fills what it takes back rather than serving the old tail. - fn truncate_to(&mut self, file_id: FileId, size: u64, _mtime: u64) -> Result<(), SyscallError> { - let role = self.role; - let known = self.open.get(&file_id).ok_or(SyscallError::NotFound)?; - let (name, was) = (known.name.clone(), known.file.len()); - if was <= size { - return Ok(()); - } - if let Some(cell) = self.live_extents(&name) { - cell.truncate_to(size); - } - let Self { fs, open, repair_named, .. } = self; - let info = open.get_mut(&file_id).ok_or(SyscallError::NotFound)?; - fs.set_len(&mut info.file, size).map_err(|e| refused(role, fs, repair_named, "set_len", &name, e)) - } - - /// Record the real length and re-derive the backing; a shrink truncates - /// the extent list before `set_len` frees the tail, so no backing ever - /// names a reissued cluster. - fn update_metadata( - &mut self, - file_id: FileId, - size: u64, - _mtime: u64, - ) -> Result<(), SyscallError> { - let role = self.role; - let time = now(); - let name = { - let known = self.open.get(&file_id).ok_or(SyscallError::NotFound)?; - let (name, was) = (known.name.clone(), known.file.len()); - if was > size { - if let Some(cell) = self.live_extents(&name) { - cell.truncate_to(size); - } - } - let Self { fs, open, repair_named, .. } = self; - let info = open.get_mut(&file_id).ok_or(SyscallError::NotFound)?; - if info.file.len() != size { - fs.set_len(&mut info.file, size) - .map_err(|e| refused(role, fs, repair_named, "set_len", &info.name, e))?; - } - #[cfg(feature = "boot-actuators")] - if meta_refuse::should_refuse(&info.name) { - log!( - "{role}-volume: fat-flush-meta-refuse: refusing the directory-entry write \ - of {} as a budget expiry", - info.name - ); - return Err(SyscallError::WouldBlock); - } - fs.flush_meta(&mut info.file, time) - .map_err(|e| refused(role, fs, repair_named, "flush_meta", &info.name, e))?; - name - }; - // A failure here only costs evictability; the write itself is - // already on the volume. - match self.backing(&name) { - Ok(backing) => file_cache::set_backing(file_id, backing), - Err(_) => log!("{role}-volume: {name} was written but has no re-readable extent list"), - } - Ok(()) - } - - /// Always an error: FAT32 has no symlink representation to write. - fn create_symlink(&mut self, _name: &str, _target: &str) -> Result<(), SyscallError> { - Err(SyscallError::NotSupported) - } - - /// Error returned, not logged via [`refused`]: a log write here would be - /// more pending content for the next sync. The one exception is a volume - /// waiting on an unlanded repair, which [`name_pending`] names. - /// - /// Every open handle is brought level with its chain first: a sync is a - /// caller saying the volume may be left as it stands from here, and a - /// chain that outran its directory entry is not a volume that may be left. - fn sync(&mut self) -> Result<(), SyscallError> { - let Self { role, fs, open, repair_named, .. } = self; - for info in open.values_mut() { - reconcile(*role, fs, repair_named, info); - } - self.fs.sync().map_err(|e| { - name_pending(self.role, &self.fs, &mut self.repair_named, || String::from("sync"), e); - as_syscall_error(e) - }) - } - - fn open_backing(&mut self, name: &str) -> Result, SyscallError> { - self.backing(name) - } - - fn cached_file_id(&mut self, name: &str) -> Option { - self.by_name.get(name).copied() - } -} - -/// Ask every USB disk for the partitions this kernel was given, retrying -/// while the boot volume is still not found. -pub fn probe_boot_disks() { - let deadline = crate::clock::nanos_since_boot() + xhci::PORT_SETTLE_CEILING.nanos(); - let mut probed = 0; - loop { - probed = probe_announced(probed); - // Nothing further can change: no partition named, resolved, or - // ambiguity no device can repair. - if gpt::boot_partition().is_none() || !gpt::boot_volume_still_possible() { - return; - } - if crate::clock::nanos_since_boot() >= deadline { - log!( - "usb-storage: {probed} disk(s) on this machine and none carries the boot \ - partition after {} ms of looking — this boot has no /boot and no /log", - xhci::PORT_SETTLE_CEILING.duration().millis() - ); - return; - } - // Paced, not spun: one MMIO read per port under the controller lock, - // on physical hardware. - let next = crate::clock::nanos_since_boot() + xhci::PORT_POLL.nanos(); - while crate::clock::nanos_since_boot() < next { - core::hint::spin_loop(); - } - xhci::recheck_ports(); - } -} - -/// Probe every disk announced since the last call; indices are stable and -/// dense, so a disk is never probed twice. -fn probe_announced(mut probed: usize) -> usize { - let count = usb_storage::count(); - while probed < count { - let index = probed; - probed += 1; - // One call for both the handle and the block size; the disk carries its - // own geometry, and registering it is what makes it openable by number. - let Some((handle, lba_bytes)) = usb_storage::handle(index) else { - log!( - "usb-storage: disk {index} was announced and was gone again before its partition \ - table could be read — not probed" - ); - continue; - }; - gpt::probe(&handle, lba_bytes); - } - probed -} - -/// Open the partition `role` names, or `None` for any of several ordinary -/// non-matches — never a reason to log. -pub fn mount(role: Role) -> Option { - let volume = role.volume()?; - let device_id = volume.device; - - let Some(handle) = block::open(volume.device) else { - log!( - "{role}-volume: the partition is on device {} and no driver here registered it", - volume.device - ); - return None; - }; - - // Whole device blocks or nothing: a view that began or ended inside one - // would have to read and write the blocks either side of it, which are the - // partition table's and the next partition's. - let (first_block, blocks) = match block::span_blocks(volume.start_lba, volume.blocks, volume.lba_bytes) { - Ok(span) => span, - Err(block::SpanRefused::Overflow) => return None, - Err(block::SpanRefused::NotWhole { start_bytes, len_bytes }) => { - log!( - "{role}-volume: the table puts the partition at {start_bytes}+{len_bytes} bytes, \ - which is not whole {BLOCK}-byte blocks — refusing to mount it" - ); - return None; - } - }; - let len = blocks * BLOCK; - let device_blocks = handle.block_count(); - let part = match block::Partition::of( - handle, - first_block, - blocks, - block::Holder::Kernel(role.mount()), - ) { - Ok(part) => part, - Err(block::ViewRefused::OffDevice) => { - log!( - "{role}-volume: the table puts the partition at {first_block}+{blocks} blocks on \ - a device of {device_blocks} blocks — refusing to mount past the end of it" - ); - return None; - } - Err(block::ViewRefused::Held(by)) => { - log!("{role}-volume: the partition is held by {by} — refusing to mount it"); - return None; - } - }; - - *device(role).lock() = Some(FatDevice { - part, - len, - scratch: vec![0u8; BLOCK as usize], - resident: vec![0u8; RESIDENT_BLOCKS * BLOCK as usize], - tags: [None; RESIDENT_BLOCKS], - next_victim: 0, - }); - - // `probe` only reads, so the bound can tighten from partition to volume - // before any write; this only ever shrinks. - let mut volume = FatVolume { role, bytes: len }; - let geom = match Fat32::probe(&mut volume) { - Ok(geom) => geom, - Err(e) => { - log!("{role}-volume: the partition holds no FAT32 this kernel can mount: {e}"); - *device(role).lock() = None; - return None; - } - }; - let volume_bytes = geom.total_sectors as u64 * geom.bytes_per_sector as u64; - volume.bytes = volume_bytes; - if let Some(mounted) = device(role).lock().as_mut() { - mounted.len = volume_bytes; - } - - // The two FATs' ranges for the FAT-write actuators; empty and inert on a - // volume that does not mirror. - #[cfg(feature = "boot-actuators")] - if role == Role::Log && geom.num_fats >= 2 && geom.active_fat.is_none() { - let one_fat = geom.fat_sectors as u64 * geom.bytes_per_sector as u64; - let (active, mirror) = (geom.fat_base_offset(0), geom.fat_base_offset(1)); - mirror_refuse::capture((active, active + one_fat), (mirror, mirror + one_fat)); - } - - match Fat32::mount(volume) { - Ok(fs) => { - log!( - "{role}-volume: partition mounted from device {device_id}, {volume_bytes} bytes of a \ - {len}-byte partition at device offset {}, {}-byte sectors, {}-byte \ - clusters, {} clusters", - first_block * BLOCK, - geom.bytes_per_sector, - geom.bytes_per_cluster(), - geom.cluster_count - ); - if role == Role::Boot { - BOOT_MOUNTED.store(true, Ordering::Relaxed); - } - Some(FatFs::new(role, fs)) - } - Err(e) => { - log!("{role}-volume: the partition holds no FAT32 this kernel can mount: {e}"); - *device(role).lock() = None; - None - } - } -} diff --git a/kernel/src/file_backing.rs b/kernel/src/file_backing.rs index 61ec61c115b..f420b34a6ed 100644 --- a/kernel/src/file_backing.rs +++ b/kernel/src/file_backing.rs @@ -1,18 +1,15 @@ -use alloc::sync::Arc; use alloc::vec::Vec; use bcachefs::Extent; -use crate::block::{BlockError, BlockResult}; -use crate::page_cache; +use crate::block::BlockResult; use crate::rootfs::MemoryImage; -use crate::sync::Lock; -use crate::time::Deadline; /// `mm::PAGE_SIZE`: `usize` for buffer sizing, `u64` for file offsets. const BLOCK_SIZE: usize = crate::mm::PAGE_SIZE as usize; const BLOCK_SIZE_U64: u64 = crate::mm::PAGE_SIZE; -/// Backing store for a memory-mapped file; callers don't know if it's NVMe, RAM, or something else. +/// Backing store for a memory-mapped file: ROOT's image, an image a caller +/// handed over, or a tmpfs file; callers do not know which. pub trait FileBacking: Send + Sync { /// Reads one page of file data at `file_offset` into `buf`, zero-filling past EOF. #[must_use = "a failed read left the buffer zeroed; it does not hold the file's bytes"] @@ -22,102 +19,6 @@ pub trait FileBacking: Send + Sync { fn file_size(&self) -> u64; } -/// Which blocks a `/home` file's data lives in, and whether they are still that file's. -pub struct FileBlocks { - /// `None` once the filesystem has taken the blocks back. - extents: Lock>>, -} - -impl FileBlocks { - pub fn new(extents: Vec) -> Arc { - Arc::new(Self { extents: Lock::new(Some(extents)) }) - } - - /// Gives the blocks up; every read through a backing that shares this fails from here on. - pub fn revoke(&self) { - // Not refcounted: a read after this must fail, not extend a freed block's life. - *self.extents.lock() = None; - } - - /// Runs `f` over the current extent list, or `None` if the file is gone. - pub fn with(&self, f: impl FnOnce(&mut Vec) -> R) -> Option { - // Lock stays held across `f`: the write path resolves and allocates inside it. - self.extents.lock().as_mut().map(f) - } - - /// Keep the first `keep` blocks and hand back the dropped tail runs, for - /// the caller to free once the shortened record is on the device. Every - /// backing sharing this cell reads the dropped range as a hole from here on. - pub fn truncate_to_blocks(&self, keep: u64) -> Vec { - let mut guard = self.extents.lock(); - let Some(runs) = guard.as_mut() else { return Vec::new() }; - let mut dropped = Vec::new(); - let mut remaining = keep; - let mut kept = Vec::with_capacity(runs.len()); - for run in runs.drain(..) { - let count = run.block_count as u64; - if remaining >= count { - remaining -= count; - kept.push(run); - } else { - if remaining > 0 { - kept.push(Extent { - start_block: run.start_block, - block_count: remaining as u32, - _reserved: 0, - }); - } - dropped.push(Extent { - start_block: run.start_block + remaining, - block_count: (count - remaining) as u32, - _reserved: 0, - }); - remaining = 0; - } - } - *runs = kept; - dropped - } -} - -/// One block of `cache`, retried while the refusal is the budget's. -/// -/// A `BudgetExpired` is a claim about the caller's clock and never a loss, and -/// each attempt here is above `block::Partition`'s device lock, so it queues -/// afresh with a whole `block::OPERATION` to spend. Bounded by -/// `block::DEADMAN`, which is what bounds the run of attempts in both the -/// kernel's other ladders (`writeback::drain_retrying`, `ops::fsync`). -/// -/// It cannot park or yield between attempts, unlike either of those: a -/// demand-paging fill runs under the process-data lock, where a park is the -/// runtime panic `kernel/CLAUDE.md` names. Re-acquiring the device lock is the -/// only wait, so the deadline is checked before each attempt and the caller -/// gets the device's own last word when it is reached. -fn read_block_retrying( - cache: &page_cache::Cached, - block: u64, - raw: &mut [u8; BLOCK_SIZE], -) -> BlockResult { - let began = crate::clock::now(); - let deadman = Deadline::at(began + crate::block::DEADMAN.duration()); - let mut attempts = 0u32; - loop { - attempts += 1; - let answer = cache.raw_read(block, raw); - if answer != Err(BlockError::BudgetExpired) { - return answer; - } - if deadman.reached(crate::clock::now()) { - log!( - "file: block {block} refused on the operation budget {attempts} time(s) in {} — {}", - crate::clock::now() - began, - crate::block::DEADMAN, - ); - return answer; - } - } -} - /// The block holding `file_offset`, if the extents reach that far. fn offset_to_block(extents: &[Extent], file_offset: u64) -> Option { let block_idx = file_offset / BLOCK_SIZE_U64; @@ -132,52 +33,9 @@ fn offset_to_block(extents: &[Extent], file_offset: u64) -> Option { None } -/// File backed by blocks of the partition one page cache serves. -pub struct NvmeBacking { - cache: Arc, - blocks: Arc, - size: u64, -} - -impl NvmeBacking { - pub fn new(cache: Arc, blocks: Arc, size: u64) -> Self { - Self { cache, blocks, size } - } -} - -impl FileBacking for NvmeBacking { - fn read_page(&self, file_offset: u64, buf: &mut [u8; BLOCK_SIZE]) -> BlockResult { - buf.fill(0); - if file_offset >= self.size { - return Ok(()); - } - // Unlinked: blocks may already belong to another file. - let Some(block) = self.blocks.with(|extents| offset_to_block(extents, file_offset)) else { - log!("file: read through a backing whose file was deleted"); - return Err(BlockError::Device); - }; - if let Some(block) = block { - // Bypasses block page cache; file cache is the sole cache for file data. - let mut raw = [0u8; BLOCK_SIZE]; - // `buf` is already zeroed, so a failed read here returns a hole, not stale data. - if let Err(e) = read_block_retrying(&self.cache, block, &mut raw) { - log!("file: read of block {block} failed"); - return Err(e); - } - let valid = BLOCK_SIZE.min((self.size - file_offset) as usize); - buf[..valid].copy_from_slice(&raw[..valid]); - } - Ok(()) - } - - fn file_size(&self) -> u64 { - self.size - } -} - /// File on ROOT, backed by a fixed extent list over the image in memory. /// -/// No revocation cell, unlike [`NvmeBacking`]: nothing can delete or truncate a +/// No revocation cell: nothing can delete or truncate a /// file here, so the blocks a backing was opened over stay that file's for as /// long as it lives. A block outside the image is refused by /// [`MemoryImage::read`], which is where every bound on a number the image @@ -219,3 +77,84 @@ impl FileBacking for ReadOnlyBacking { self.size } } + +/// An executable or library userland read and handed over whole: the bytes are +/// copied into pages of the kernel's own at the call, so nothing its sender +/// writes afterwards reaches a page this serves. What a program in `/apps` is +/// spawned and paged from, since no file server's volume is the kernel's. +pub struct ImageBacking { + pages: Vec, + size: u64, +} + +/// The most bytes one image may carry: the copy is kernel memory for the life +/// of every process paged from it, and nothing charges that to the sender. +pub const MAX_IMAGE_BYTES: u64 = 256 << 20; + +impl ImageBacking { + /// Copy `len` bytes of the caller's memory at `ptr`, a 2 MiB page's run at + /// a time, which is the most one user window maps contiguously. + pub fn copy_in( + ctx: &crate::user_ptr::SyscallContext, + ptr: crate::UserAddr, + len: u64, + ) -> Result { + use toyos_abi::syscall::SyscallError; + const PAGE: u64 = crate::mm::PAGE_2M; + if len == 0 || len > MAX_IMAGE_BYTES { + return Err(SyscallError::InvalidArgument); + } + let mut pages = Vec::new(); + pages.try_reserve_exact(len.div_ceil(PAGE) as usize).map_err(|_| SyscallError::ResourceExhausted)?; + for _ in 0..len.div_ceil(PAGE) { + let page = crate::mm::pmm::alloc_page(crate::mm::pmm::Category::Elf) + .ok_or(SyscallError::ResourceExhausted)?; + pages.push(page); + } + let mut done = 0u64; + while done < len { + let at = ptr.raw().checked_add(done).ok_or(SyscallError::BadAddress)?; + // A run ends at the sender's page boundary or the image's own, whichever is first. + let run = (PAGE - at % PAGE).min(PAGE - done % PAGE).min(len - done); + let window = ctx.user_bytes(crate::UserAddr::new(at), run).ok_or(SyscallError::BadAddress)?; + let page = &pages[(done / PAGE) as usize]; + // SAFETY: the page is this backing's own and 2 MiB long; `done % PAGE + run <= PAGE` by the `min` above. + let dst = unsafe { + core::slice::from_raw_parts_mut( + page.direct_map().as_mut_ptr::().add((done % PAGE) as usize), + run as usize, + ) + }; + window.read_at(0, dst); + done += run; + } + Ok(Self { pages, size: len }) + } +} + +impl FileBacking for ImageBacking { + fn read_page(&self, file_offset: u64, buf: &mut [u8; BLOCK_SIZE]) -> BlockResult { + buf.fill(0); + if file_offset >= self.size { + return Ok(()); + } + const PAGE: u64 = crate::mm::PAGE_2M; + // Every backing is read a page at a time, and a 4 KiB-aligned page never crosses a 2 MiB one. + assert!(file_offset % BLOCK_SIZE_U64 == 0, "an image read at {file_offset:#x} is not page-aligned"); + let valid = BLOCK_SIZE.min((self.size - file_offset) as usize); + let page = &self.pages[(file_offset / PAGE) as usize]; + // SAFETY: the page is this backing's, immutable since `copy_in`, and `file_offset % PAGE + valid <= PAGE`. + let src = unsafe { + core::slice::from_raw_parts( + page.direct_map().as_ptr::().add((file_offset % PAGE) as usize), + valid, + ) + }; + buf[..valid].copy_from_slice(src); + Ok(()) + } + + fn file_size(&self) -> u64 { + self.size + } +} diff --git a/kernel/src/file_cache.rs b/kernel/src/file_cache.rs index c11c67abd65..7883a90dbbb 100644 --- a/kernel/src/file_cache.rs +++ b/kernel/src/file_cache.rs @@ -150,12 +150,6 @@ fn undo_open(file_id: FileId) { } } -/// This file's open-reference count, for the leak-rollback self-test's census. -#[cfg(feature = "boot-actuators")] -pub fn ref_count(file_id: FileId) -> u32 { - FILE_CACHE.lock().files.get(&file_id).map_or(0, |f| f.ref_count) -} - /// The verdict [`release_to_writeback`] hands its caller. #[derive(Clone, Copy, PartialEq, Eq)] pub enum Release { diff --git a/kernel/src/gpt.rs b/kernel/src/gpt.rs index 04984b591bd..9033983c8f8 100644 --- a/kernel/src/gpt.rs +++ b/kernel/src/gpt.rs @@ -1,24 +1,26 @@ -//! Resolves the partitions the bootloader handed off ([`KernelArgs`]) to -//! locations on a probed block device ([`probe`]), and collects every DATA -//! candidate those devices carry. ROOT is none of this module's business: the -//! loader reads it into memory (`rootfs`). +//! The partitions of the disks this kernel drives — the USB disks, until usbd +//! drives the controller — and the ones the bootloader named. //! -//! The boot partition's location must match firmware's account or is -//! refused; the log partition is trusted only on the device already found -//! to carry the boot partition, since its GUID names a file on that volume. -//! A DATA candidate is selected by partition *type* and nothing more — which of -//! them is the role's filesystem is answered against each one's own superblock, -//! by `bcachefs_adapter::probe`. A partition a process claims is found by -//! [`claimable`], on the disks [`probe`] read, and the inventory is answered -//! from the tables [`probe`] listed ([`inventory`]). -//! Nothing here writes. +//! **The loader names three partitions** ([`KernelArgs`]): the ROOT it read, +//! the running slot's volume and the log's. This kernel mounts none of them +//! but ROOT, which is in memory (`rootfs`); it answers their GUIDs in the +//! inventory ([`loaded`]) so the file servers for those roles find them, on +//! whatever disk and whoever drives it. +//! +//! The boot partition's location must match firmware's account on a disk +//! [`probe`] read, and [`probe_usb_disks`] waits for the disk that carries it +//! to appear. A partition a process claims is found by [`claimable`], on the +//! disks [`probe`] read, and the inventory is answered from the tables +//! [`probe`] listed ([`inventory`]). Nothing here writes. use alloc::vec::Vec; use crate::block::{BlockDevice, DeviceId, Handle}; use crate::device::ClaimError; +use crate::drivers::{usb_storage, xhci}; use crate::sync::Lock; use toyos_abi::boot::KernelArgs; +use toyos_abi::inventory::Role; use toyos_abi::part::PartGuid; use toyos_gpt::{GptError, Guid, Partition, Sectors}; @@ -41,27 +43,19 @@ pub struct Volume { pub blocks: u64, } -/// What the kernel knows about where the given partitions live. `Ambiguous` +/// What the kernel knows about where the boot partition lives. `Ambiguous` /// is permanent: two devices carrying one partition GUID means one is a /// clone, and nothing here can tell which one firmware read. enum Resolution { Unknown, - Found { boot: Volume, log: Option }, + Found { boot: Volume }, Ambiguous, } -/// One partition of a ToyOS type, on a device that answered for it. -#[derive(Clone, Copy, Debug)] -pub struct Candidate { - pub volume: Volume, - pub guid: Guid, -} - static FIRMWARE: Lock> = Lock::new(None); -/// The log partition's identity; `None` only before [`init`] runs. -static LOG_GUID: Lock> = Lock::new(None); static RESOLVED: Lock = Lock::new(Resolution::Unknown); -static DATA: Lock> = Lock::new(Vec::new()); +/// The partitions the loader named, by role; empty before [`init`]. +static LOADED: Lock> = Lock::new(Vec::new()); /// Every device [`probe`] read, with its logical block size: the disks a /// partition claim is looked for on. Taken alone. static DISKS: Lock> = Lock::new(Vec::new()); @@ -105,12 +99,16 @@ pub fn inventory() -> Vec<(DeviceId, Partition, Option)> { out } +/// The partitions the loader named, for the inventory. +pub fn loaded() -> Vec<(Role, Guid)> { + LOADED.lock().clone() +} + /// List `handle`'s table into [`LISTED`], once per disk. fn list(sectors: &mut DeviceSectors<'_>, handle: &Handle, lba_bytes: u32) { let id = handle.device_id(); let mut found = alloc::vec![BLANK; MAX_LISTED]; - // A disk with no table this kernel parses carries no partition, and - // `collect` says so, naming the refusal. + // A disk with no table this kernel parses carries no partition. let Ok(scan) = toyos_gpt::list(sectors, &mut found) else { return }; if scan.matched as usize > scan.listed { log!( @@ -123,19 +121,17 @@ fn list(sectors: &mut DeviceSectors<'_>, handle: &Handle, lba_bytes: u32) { LISTED.lock().push(Listed { handle: handle.clone(), lba_bytes, parts: found }); } -/// How many partitions of one ToyOS type one device may offer this kernel. -/// -/// A bound rather than a `Vec` because [`toyos_gpt::locate_type`] fills a -/// caller's slice; a device carrying more says so in the log, and a boot that -/// then finds no match panics naming what it did see. -const MAX_PER_DEVICE: usize = 4; - -/// Take both partitions' identities out of the bootloader's handoff. +/// Take the partitions the loader named out of its handoff. pub fn init(args: &KernelArgs) { - let log_guid = Guid(args.log_partition_guid); - log!("gpt: the boot volume names {log_guid} as the log partition"); - *LOG_GUID.lock() = Some(log_guid); - + let mut loaded = LOADED.lock(); + if args.root_partition_guid != [0; 16] { + loaded.push((Role::Root, Guid(args.root_partition_guid))); + } + if args.log_partition_guid != [0; 16] { + let log_guid = Guid(args.log_partition_guid); + log!("gpt: the boot volume names {log_guid} as the log partition"); + loaded.push((Role::Log, log_guid)); + } if args.boot_partition_present == 0 { log!("gpt: firmware named no boot partition — this machine has none"); return; @@ -149,6 +145,7 @@ pub fn init(args: &KernelArgs) { "gpt: firmware booted us from partition {} at LBA {}+{}", part.guid, part.start_lba, part.blocks ); + loaded.push((Role::Boot, part.guid)); *FIRMWARE.lock() = Some(part); } @@ -156,33 +153,65 @@ pub fn boot_partition() -> Option { *FIRMWARE.lock() } -/// Where the boot partition is, if a device has been found to carry it. -pub fn boot_volume() -> Option { - match *RESOLVED.lock() { - Resolution::Found { boot, .. } => Some(boot), - Resolution::Unknown | Resolution::Ambiguous => None, - } -} - /// True only while resolution is still `Unknown`; `Ambiguous` is permanent. pub fn boot_volume_still_possible() -> bool { matches!(*RESOLVED.lock(), Resolution::Unknown) } -/// Where the log partition is, on the device that carries the boot partition. -pub fn log_volume() -> Option { - match *RESOLVED.lock() { - Resolution::Found { log, .. } => log, - Resolution::Unknown | Resolution::Ambiguous => None, +/// Ask every USB disk for its table, retrying while the boot partition is +/// still not found: a USB-booted disk appears only once the controller binds +/// it, and a claim of one of its partitions is found on the disks read here. +pub fn probe_usb_disks() { + let deadline = crate::clock::nanos_since_boot() + xhci::PORT_SETTLE_CEILING.nanos(); + let mut probed = 0; + loop { + probed = probe_announced(probed); + // Nothing further can change: no partition named, resolved, or + // ambiguity no device can repair. + if boot_partition().is_none() || !boot_volume_still_possible() { + return; + } + if crate::clock::nanos_since_boot() >= deadline { + log!( + "usb-storage: {probed} disk(s) on this machine and none carries the boot \ + partition after {} ms of looking", + xhci::PORT_SETTLE_CEILING.duration().millis() + ); + return; + } + // Paced, not spun: one MMIO read per port under the controller lock, + // on physical hardware. + let next = crate::clock::nanos_since_boot() + xhci::PORT_POLL.nanos(); + while crate::clock::nanos_since_boot() < next { + core::hint::spin_loop(); + } + xhci::recheck_ports(); } } -/// Every DATA candidate seen so far, across every device probed. -pub fn data_candidates() -> Vec { - DATA.lock().clone() +/// Probe every disk announced since the last call; indices are stable and +/// dense, so a disk is never probed twice. +fn probe_announced(mut probed: usize) -> usize { + let count = usb_storage::count(); + while probed < count { + let index = probed; + probed += 1; + // One call for both the handle and the block size; the disk carries its + // own geometry, and registering it is what makes it openable by number. + let Some((handle, lba_bytes)) = usb_storage::handle(index) else { + log!( + "usb-storage: disk {index} was announced and was gone again before its partition \ + table could be read — not probed" + ); + continue; + }; + probe(&handle, lba_bytes); + } + probed } -/// Ask one registered block device what it carries: DATA candidates always, and the boot partition when firmware named one. +/// Ask one registered block device what it carries: its table for the +/// inventory, and the boot partition when firmware named one. pub fn probe(handle: &Handle, lba_bytes: u32) { let id = handle.device_id(); let first = { @@ -197,12 +226,10 @@ pub fn probe(handle: &Handle, lba_bytes: u32) { if first { list(&mut sectors, handle, lba_bytes); } - collect(&mut sectors, id, lba_bytes, "DATA", Guid::TOYOS_DATA, &DATA); let Some(firmware) = boot_partition() else { return; }; - let found = match toyos_gpt::locate(&mut sectors, firmware.guid) { Ok(found) => found, Err(GptError::NotFound { used_entries }) => { @@ -240,8 +267,6 @@ pub fn probe(handle: &Handle, lba_bytes: u32) { let mut resolved = RESOLVED.lock(); match *resolved { Resolution::Unknown => { - // Only here: the log GUID names a file on this volume, not on any other disk. - let log = locate_log(&mut sectors, id, lba_bytes); log!( "gpt: device {id} carries the boot partition at LBA {}+{} ({}-byte blocks), \ entry {} of {} on disk {}{}", @@ -253,7 +278,7 @@ pub fn probe(handle: &Handle, lba_bytes: u32) { found.disk_guid, if part.is_efi_system() { "" } else { " — and its type is not ESP" } ); - *resolved = Resolution::Found { boot: volume, log }; + *resolved = Resolution::Found { boot: volume }; } Resolution::Found { boot: first, .. } => { log!( @@ -270,63 +295,6 @@ pub fn probe(handle: &Handle, lba_bytes: u32) { } } -/// Record every partition on this device whose type is `ty`. -/// -/// Each type match is then located again by its own *unique* GUID, because -/// that road is the one carrying the range and overlap checks: a candidate this -/// records has passed everything `toyos-gpt` refuses a partition for. -fn collect( - sectors: &mut DeviceSectors<'_>, - id: DeviceId, - lba_bytes: u32, - what: &str, - ty: Guid, - into: &Lock>, -) { - let mut found = [BLANK; MAX_PER_DEVICE]; - let scan = match toyos_gpt::locate_type(sectors, ty, &mut found) { - Ok(scan) => scan, - Err(e) => { - log!("gpt: device {id} carries no {what} this kernel can read: {e:?}"); - return; - } - }; - if scan.matched as usize > scan.listed { - log!( - "gpt: device {id} carries {} {what} partitions and this kernel looks at {}", - scan.matched, - scan.listed - ); - } - for candidate in &found[..scan.listed] { - let checked = match toyos_gpt::locate(sectors, candidate.unique_guid) { - Ok(located) => located.partition, - Err(e) => { - log!( - "gpt: device {id} names a {what} {} its own table then refuses: {e:?}", - candidate.unique_guid - ); - continue; - } - }; - log!( - "gpt: device {id} carries the {what} candidate {} at LBA {}+{}", - checked.unique_guid, - checked.first_lba, - checked.lba_count() - ); - into.lock().push(Candidate { - volume: Volume { - device: id, - lba_bytes, - start_lba: checked.first_lba, - blocks: checked.lba_count(), - }, - guid: checked.unique_guid, - }); - } -} - /// One partition a claim may hold: where it is, and its unique GUID. #[derive(Clone, Copy, Debug)] pub struct Claimable { @@ -425,37 +393,6 @@ const BLANK: Partition = Partition { last_lba: 0, }; -/// The log partition on the device already proven to carry the boot partition, or `None`. -fn locate_log(sectors: &mut DeviceSectors<'_>, id: DeviceId, lba_bytes: u32) -> Option { - let target = LOG_GUID.lock().expect("gpt::init runs before any device is probed"); - match toyos_gpt::locate(sectors, target) { - Ok(found) => { - let part = found.partition; - log!( - "gpt: device {id} carries the log partition {target} at LBA {}+{}, entry {} of {}", - part.first_lba, - part.lba_count(), - part.index, - found.used_entries - ); - Some(Volume { - device: id, - lba_bytes, - start_lba: part.first_lba, - blocks: part.lba_count(), - }) - } - Err(e) => { - log!( - "gpt: device {id} carries the boot partition but nothing with the log partition's \ - GUID {target}: {e:?} — this stick has no log partition and the kernel's log \ - stays in memory" - ); - None - } - } -} - /// The kernel's 4 KiB `BlockDevice`, seen in the device's own logical blocks; caches one block. struct DeviceSectors<'a> { dev: &'a Handle, diff --git a/kernel/src/inventory.rs b/kernel/src/inventory.rs index 00d023f72f1..53aff2881fe 100644 --- a/kernel/src/inventory.rs +++ b/kernel/src/inventory.rs @@ -15,13 +15,13 @@ use alloc::sync::Arc; use alloc::vec::Vec; -use toyos_abi::inventory::{Block, Claim, Claimed, Holder, PartState, Partition, Record, NAME_BYTES}; +use toyos_abi::inventory::{Block, Claim, Claimed, Holder, Loaded, PartState, Partition, Record, NAME_BYTES}; use crate::object::KObjectRef; use crate::process; /// Every record, in a fixed order: PCI, USB, block devices, partitions, -/// claims. +/// claims, and the partitions the loader named. pub fn collect() -> Vec { let mut out: Vec = crate::pcidev::inventory().into_iter().map(Record::Pci).collect(); out.extend(crate::drivers::xhci::inventory().into_iter().map(Record::Usb)); @@ -46,6 +46,11 @@ pub fn collect() -> Vec { })); } out.extend(claims().into_iter().map(Record::Claim)); + out.extend( + crate::gpt::loaded() + .into_iter() + .map(|(role, guid)| Record::Loaded(Loaded { role, unique_guid: guid.0 })), + ); out } diff --git a/kernel/src/leak_selftest.rs b/kernel/src/leak_selftest.rs index ffef8cfd6c0..b9db9326c57 100644 --- a/kernel/src/leak_selftest.rs +++ b/kernel/src/leak_selftest.rs @@ -1,42 +1,7 @@ //! Negative controls for the leak-rollback fixes: each reproduces an "acquire //! before a fallible step" site and asserts the in-tree count returns to baseline. Run behind `leak-rollback-selftest`. -/// Runs every leak-rollback control; must run after `/log` is mounted. +/// Runs every leak-rollback control. pub fn run() { crate::object::device::mint_rollback_selftest(); - fat_reopen_census(); -} - -/// The reopen arm takes the file-cache reference before the fallible backing -/// lookup; a transient device error there must leave the reference count unchanged. -fn fat_reopen_census() { - use crate::vfs; - - const PATH: &str = "/log/lrfile"; - - let mtime = crate::clock::nanos_since_boot(); - let id = match vfs::lock().create_file(PATH, mtime) { - Ok(id) => id, - Err(e) => { - crate::log!("leak-selftest: fat-reopen skipped, create failed: {e:?}"); - return; - } - }; - - let before = crate::file_cache::ref_count(id); - crate::fat32_adapter::selftest_fail_backing(true); - // Each `vfs::lock()` is its own statement: the lock is not reentrant, so re-locking a live guard panics. - let target = vfs::lock().resolve_for_open(PATH, vfs::ResolveIntent::KernelOrRead); - let reopened = match target { - Ok(target) => vfs::lock().open_target(&target), - Err(e) => Err(e), - }; - crate::fat32_adapter::selftest_fail_backing(false); - let after = crate::file_cache::ref_count(id); - - let verdict = if reopened.is_err() && after == before { "PASS" } else { "FAIL" }; - crate::log!( - "leak-selftest: fat-reopen {verdict} (refused={} before={before} after={after})", - reopened.is_err() - ); } diff --git a/kernel/src/loader/mod.rs b/kernel/src/loader/mod.rs index 8dc2109c481..85c488214e9 100644 --- a/kernel/src/loader/mod.rs +++ b/kernel/src/loader/mod.rs @@ -355,6 +355,11 @@ fn rela_dyn_from_sections( /// No parent argument: a process has no parent, and the only thing the caller /// contributes beyond its endowment is the working directory it passes in. /// +/// `image` is the program's bytes when the caller read them itself, and then +/// `argv[0]` is only its name: nothing opens it, and its libraries come from +/// `/system/lib` alone, since the directory it names is no volume of this +/// kernel's. +/// /// `Refusal`, not `-> !`, is the error type: every failure below owns a /// partly built process (address space, stack, kernel stack), and nothing /// unwinds, so the error must travel out as a value rather than strand it. @@ -363,6 +368,7 @@ pub fn spawn( pending: PendingHandles, cwd: String, env: Vec, + image: Option>, ) -> Result, crate::object::Refusal> { // An argv of only separators survives sys_spawn's split as an empty slice. let Some(&path) = argv.first() else { @@ -370,13 +376,19 @@ pub fn spawn( }; let t0 = crate::clock::nanos_since_boot(); - // Scoped, not held across the match: dropping `pending` on any `return` here takes the VFS lock. - let opened = vfs::lock().open_backing(path); - let backing: Arc = match opened { - Ok(b) => b, - Err(e) => { - log!("spawn: {}: {e}", path); - return Err(e.into()); + let from_image = image.is_some(); + let backing: Arc = match image { + Some(image) => image, + None => { + // Scoped, not held across the match: dropping `pending` on any `return` here takes the VFS lock. + let opened = vfs::lock().open_backing(path); + match opened { + Ok(b) => b, + Err(e) => { + log!("spawn: {}: {e}", path); + return Err(e.into()); + } + } } }; @@ -433,7 +445,7 @@ pub fn spawn( } let t2 = crate::clock::nanos_since_boot(); - let mut loaded_libs = load_needed_libs(&exe, path)?; + let mut loaded_libs = load_needed_libs(&exe, path, from_image)?; let t_deps = crate::clock::nanos_since_boot(); // ELF segments are demand-faulted; the address space starts with the clock page alone. @@ -679,13 +691,17 @@ struct NeededLibs { /// 2 MiB window, so a `DT_NEEDED` list naming more is refused rather than loaded. const MAX_NEEDED_LIBS: usize = 64; -/// Load each distinct `DT_NEEDED` library, from the executable's own directory first and `/system/lib` second. -fn load_needed_libs(exe: &ExeTables, path: &str) -> Result { +/// Load each distinct `DT_NEEDED` library, from the executable's own directory +/// first and `/system/lib` second; an image's from `/system/lib` alone. +fn load_needed_libs(exe: &ExeTables, path: &str, from_image: bool) -> Result { let mut out = NeededLibs { libs: Vec::new(), paths: Vec::new() }; if exe.needed.is_empty() { return Ok(out); } - let exe_dir = path.rsplit_once('/').map(|(dir, _)| dir).unwrap_or(""); + let exe_dir = match from_image { + true => "/system/lib", + false => path.rsplit_once('/').map(|(dir, _)| dir).unwrap_or(""), + }; // Collapse duplicates to the distinct set and bound it: a repeat resolves to // one library `elf/cache.rs` holds one window for, so it buys no second one. @@ -920,7 +936,7 @@ pub fn spawn_init() -> Pid { }], label.as_bytes().to_vec(), ); - match spawn(&[INIT_PATH], PendingHandles::Ready(handles, endowments), String::from("/"), Vec::new()) { + match spawn(&[INIT_PATH], PendingHandles::Ready(handles, endowments), String::from("/"), Vec::new(), None) { Ok(object) => object.pid(), Err(crate::object::Refusal::Error(e)) => panic!("spawn_init: failed to spawn: {e:?}"), Err(crate::object::Refusal::Handle(e)) => panic!("spawn_init: {e}"), diff --git a/kernel/src/main.rs b/kernel/src/main.rs index 14da5f126ff..ec834ec79d0 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -43,14 +43,11 @@ mod input_merge_test; #[cfg(feature = "boot-actuators")] mod usb_gate; #[cfg(feature = "boot-actuators")] -mod nvme_gate; -#[cfg(feature = "boot-actuators")] mod sched_gate; mod block; mod durability; mod gpt; mod inventory; -mod page_cache; mod rollback; mod rootfs; mod file_cache; @@ -62,7 +59,6 @@ mod writeback; mod tmpfs; mod file_backing; mod bcachefs_adapter; -mod fat32_adapter; mod fs_rename; #[cfg(feature = "boot-actuators")] mod heartbeat; @@ -113,10 +109,9 @@ mod late_panic { use crate::mm::paging::MmioPolicy; use alloc::boxed::Box; -use alloc::sync::Arc; use arch::{cpu, percpu, smp}; pub(crate) use arch::hw; -use drivers::{acpi, gop, nvme, pci, serial, virtio_console, virtio_gpu, virtio_sound, xhci}; +use drivers::{acpi, gop, pci, serial, virtio_console, virtio_gpu, virtio_sound, xhci}; use toyos_abi::boot::{KernelArgs, MemoryMapEntry}; #[panic_handler] @@ -194,10 +189,6 @@ fn register_gpu(driver: Box, info: gpu::GpuInfo) { gpu::register(driver, info); } -/// The names DATA answers to. One filesystem, so `/apps/x` and `/home/x` are -/// two directories of it and never two volumes. -const DATA_PATHS: [&str; 4] = ["apps", "config", "home", "state"]; - /// The boot from power-on, off the loader's TSC readings and `complete`'s, at /// the calibrated rate. The TSC counts from reset, so the first span is /// firmware's unless firmware wrote the counter, which the loader's @@ -222,10 +213,11 @@ fn report_power_on(args: &KernelArgs, complete: u64) { } /// Says where this boot's log can be read, on the last surface still showing it once userland owns the screen. -fn report_log_destination() { +fn report_log_destination(args: &KernelArgs) { // Kernel-side because panic_console owns the panel; logd reports which file it opened separately. - // has_log reflects whether /log mounted, not whether logd could open a file on it. - let has_log = vfs::lock().has_mount(fat32_adapter::Role::Log.mount()); + // Whether the loader named a log partition, not whether its file server mounted it: that + // server says so itself, and this kernel mounts nothing but ROOT. + let has_log = args.log_partition_guid != [0; 16]; // ASCII only: the panel's font renders anything outside 0x20..=0x7E as a dot. match (drivers::serial::has_console(), has_log) { (true, true) => log!("log: this boot is on the console and on /log"), @@ -471,92 +463,22 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { log!("{} {}", rootfs::INIT_WITHOUT_A_DISK, block::census::commands_issued()); // After init's spawn and before it runs: nothing runs a task until - // `smp::set_ready` below, and `/home`, `/apps`, `/boot` and `/log` are - // mounted and every device is up by then. Before the device phase: its - // IOMMU controls aim at the pool NVMe stages. + // `smp::set_ready` below. This kernel mounts no disk: `/apps`, `/config`, + // `/home`, `/state`, `/log` and `/boot` are file servers' (`/system/bin/fsd`), + // and an NVMe controller is `/system/bin/blockd`'s. The USB disks are still + // this kernel's, served to their file servers as partition claims, until + // usbd drives the controller. let t_storage = clock::nanos_since_boot(); - // No controller is a configuration, not a failure — same as a missing xHCI, NIC, or sound device. - match nvme::init(&pci_devices) { - Some(nvme_dev) => { - let sector_size = nvme_dev.sector_size(); - let dev = page_cache::instrumented(Box::new(nvme_dev)); - if let Some(handle) = block::register(dev) { - gpt::probe(&handle, sector_size); - // Before anything mounts the device: the block the gate reads is one nothing else is touching yet. - #[cfg(feature = "boot-actuators")] - if actuator::nvme_spent_budget() { - nvme_gate::run(&handle); - } - #[cfg(feature = "boot-actuators")] - if actuator::nvme_command_silent() { - nvme_gate::silent_command(&handle); - } - // Before DATA is mounted: the device blocks it reads back are ones nothing else has written. - #[cfg(feature = "boot-actuators")] - if actuator::page_cache_partition_offset() { - page_cache::partition_offset_selftest(&handle); - } - } - } - None => log!("NVMe: no controller on this machine, storage unavailable"), - } - xhci::init(&pci_devices); #[cfg(feature = "boot-actuators")] if actuator::usb_storage_gate() { usb_gate::run(); } - // After xhci::init, not beside the NVMe probe: a USB-booted disk doesn't exist until the controller binds it. - fat32_adapter::probe_boot_disks(); + // After xhci::init: a USB-booted disk doesn't exist until the controller binds it. + gpt::probe_usb_disks(); rootfs::hold_source(); - // One filesystem, four paths: each of `DATA_PATHS` is a directory of DATA, - // so one sync settles them all and none can outlive the others. - // tmpfs when there is no DATA volume this kernel may write: persistence is - // the only difference, so the earlier refusal doesn't cascade. Nothing at - // all when the volume is ours and did not mount. - #[cfg_attr(not(feature = "boot-actuators"), allow(unused_variables))] - let data_cache = match bcachefs_adapter::open_data() { - bcachefs_adapter::Data::Mounted(cache, fs) => { - let adapter = bcachefs_adapter::BcacheFsAdapter::new(fs, Arc::clone(&cache)); - vfs::lock().mount(&DATA_PATHS, Box::new(adapter), UserAccess::ReadWrite); - Some(cache) - } - bcachefs_adapter::Data::Volatile => { - log!("storage: /apps, /config, /home and /state are a tmpfs — they will not survive a reboot"); - vfs::lock().mount( - &DATA_PATHS, - Box::new(crate::tmpfs::TmpFs::new()), - UserAccess::ReadWrite, - ); - None - } - bcachefs_adapter::Data::Absent => { - log!("storage: /apps, /config, /home and /state are absent this boot — the DATA volume is ours and did not mount"); - None - } - }; - - // Named by role, not type: both partitions are FAT32 and neither is selected for being FAT32 — a missing one just has no mount. - use fat32_adapter::Role; - // /boot is KernelOnly: a writable /boot lets a process brick the machine — esp_files replayed exactly that attack. - // The filesystem sits outside the capability model by ruling, so no handle is owed for /boot. - // /log is ReadWrite on purpose: it's an ordinary userland file logd owns, and the worst a process can do is cost the diagnostic. - match fat32_adapter::mount(Role::Boot) { - Some(fs) => vfs::lock().mount(&[Role::Boot.mount()], Box::new(fs), UserAccess::KernelOnly), - None => log!("boot-volume: not mounted; the kernel has no /boot this boot"), - } - match fat32_adapter::mount(Role::Log) { - Some(fs) => { - vfs::lock().mount(&[Role::Log.mount()], Box::new(fs), UserAccess::ReadWrite); - } - // No fallback onto /boot: with no log partition the log stays in the in-memory shards, still reachable via screen and console. - None => log!("log-volume: not mounted; this boot's kernel log stays in memory"), - } - - - // After the mounts above: the FAT reopen control drives `/log`. #[cfg(feature = "boot-actuators")] if actuator::leak_rollback_selftest() { leak_selftest::run(); @@ -565,17 +487,6 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { if actuator::revoked_backing_selftest() { revoke_selftest::run(); } - #[cfg(feature = "boot-actuators")] - if actuator::pc_unbind_selftest() { - match &data_cache { - Some(cache) => page_cache::unbind_selftest(cache), - None => log!("pc-unbind-selftest: FAIL (this boot has no metadata page cache)"), - } - } - #[cfg(feature = "boot-actuators")] - if actuator::partclaim_table_unanswered() { - page_cache::refuse_table_reads(); - } // After every driver has registered: the number under test is one a real device holds. #[cfg(feature = "boot-actuators")] if actuator::block_duplicate_id() { @@ -635,7 +546,7 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { pre_idle_wedge(); } - report_log_destination(); + report_log_destination(kernel_args); let complete_tsc = cpu::counter(); boot_phase!("complete", 0); report_power_on(kernel_args, complete_tsc); diff --git a/kernel/src/nvme_gate.rs b/kernel/src/nvme_gate.rs deleted file mode 100644 index 28d36ca768a..00000000000 --- a/kernel/src/nvme_gate.rs +++ /dev/null @@ -1,45 +0,0 @@ -//! In-guest half of the NVMe operation-budget gate: proves a caller with an already-expired deadline is refused before a command reaches the controller. - -use alloc::vec; - -use crate::block::{BlockDevice, Handle}; -use crate::scheduler::Operation; -use crate::time::Deadline; - -/// Any block works: the refusal happens before the LBA reaches a command. -const AT: u64 = 0; - -/// Reads with an expired deadline (refused), then re-reads the same block (must succeed). -pub fn run(dev: &Handle) { - let mut buf = vec![0u8; 4096]; - - // Deadline is set above the trait; `read_blocks` may only narrow it, never widen it. - let refused = { - let _op = Operation::begin(Deadline::passed()); - dev.lock().read_blocks(AT, 1, &mut buf) - }; - // budget= distinguishes a spent budget from an abandoned controller; `may_issue` refuses both. - log!("nvme-gate: read with a spent budget refused={} budget={}", - refused.is_err(), - refused == Err(crate::block::BlockError::BudgetExpired)); - - // Second read proves the controller wasn't abandoned mid-command. - let served = dev.lock().read_blocks(AT, 1, &mut buf).is_ok(); - log!("nvme-gate: the same block read afterwards ok={served}"); -} - -/// Reset-escalation half: an abandoned read forces the driver to reset instead of going permanently offline. -pub fn silent_command(dev: &Handle) { - let mut buf = vec![0u8; 4096]; - let refused = { - // Armed immediately before this read: the skipped wait must be this read's, not init's Identify. - crate::drivers::nvme::silent_command::arm(); - dev.lock().read_blocks(AT, 1, &mut buf) - }; - log!("nvme-gate: the silent command's read refused={} budget={}", - refused.is_err(), - refused == Err(crate::block::BlockError::BudgetExpired)); - - let served = dev.lock().read_blocks(AT, 1, &mut buf).is_ok(); - log!("nvme-gate: the same block read after the reset ok={served}"); -} diff --git a/kernel/src/object/ops.rs b/kernel/src/object/ops.rs index 596ce5965be..2eaee1894d1 100644 --- a/kernel/src/object/ops.rs +++ b/kernel/src/object/ops.rs @@ -629,14 +629,9 @@ pub fn fsync(object: &KObjectRef) -> u64 { // Outside `FileObject`'s lock: this and `OpenFileState::drop` take the VFS lock in the same order. // Flush and sync share one acquisition so this file cannot be unmounted between them. let mut vfs = crate::vfs::lock(); - // Tags the flush as `SYS_FSYNC`'s, for `quiesce-fsync-refuse` to stage on this path. - #[cfg(feature = "boot-actuators")] - crate::fat32_adapter::enter_fsync_flush(&path); let done = vfs .flush_file(&path, file_id, mtime) .and_then(|()| vfs.sync_for_path(&path)); - #[cfg(feature = "boot-actuators")] - crate::fat32_adapter::leave_fsync_flush(); drop(vfs); done }); diff --git a/kernel/src/page_cache.rs b/kernel/src/page_cache.rs deleted file mode 100644 index f8540fe2548..00000000000 --- a/kernel/src/page_cache.rs +++ /dev/null @@ -1,596 +0,0 @@ -//! One page cache per (device, partition) served. Every slot is bound to a -//! [`BlockKey`], so a resident page is written back where it was filled from. -//! -//! Lock order: a cache, then its device (`block::Partition::lock`); never -//! reversed, and no holder of one cache takes another's. - -use alloc::boxed::Box; -use alloc::sync::Arc; -use alloc::vec; -use alloc::vec::Vec; -use crate::hasher::{HashMap, KernelHashState}; - -use crate::block::{self, BlockDevice, BlockError, BlockKey, BlockResult, Partition}; -use crate::mm::PAGE_BYTES; -use crate::sync::{Lock, LockGuard}; - -pub struct Cached { - cache: Lock, - part: Partition, -} - -/// Wraps `dev` in the read-fault injector when `pc-unbind-selftest` or -/// `partclaim-table-unanswered` is armed — at registration, so it sits under -/// the one device object consumers share. -pub fn instrumented(dev: Box) -> Box { - #[cfg(feature = "boot-actuators")] - if crate::actuator::pc_unbind_selftest() || crate::actuator::partclaim_table_unanswered() { - return Box::new(read_fault::FaultDevice(dev)); - } - dev -} - -pub fn init(part: Partition) -> Arc { - let cache = PageCache::new(); - log!( - "page cache: device {} partition +{}, {} partition blocks, index sized for {} cached \ - blocks, cap {} slots, {} index bytes", - part.device_id(), - part.first_block(), - part.block_count(), - cache.index_capacity(), - cache.max_slots, - cache.index_bytes() - ); - Arc::new(Cached { cache: Lock::new(cache), part }) -} - -/// The block every `BlockDevice` transfers. -const PAGE: u64 = PAGE_BYTES as u64; - -/// A cache over the DATA partition `candidate` names, or `None` with the -/// reason logged. Every number below is one the disk chose, so each -/// refusal says which; nothing here decides what the volume holds. -pub fn over_candidate(candidate: &crate::gpt::Candidate) -> Option> { - let volume = candidate.volume; - let guid = candidate.guid; - let Some(handle) = block::open(volume.device) else { - log!("data: candidate {guid} is on device {} and no driver here registered it", volume.device); - return None; - }; - - // Whole device blocks or nothing: a view that began or ended inside one - // would share it with the table or with the next partition. - let (first_block, blocks) = match block::span_blocks(volume.start_lba, volume.blocks, volume.lba_bytes) { - Ok(span) => span, - Err(block::SpanRefused::Overflow) => { - log!( - "data: candidate {guid} claims LBA {}+{} of {} bytes, which is not a byte range \ - — refusing it", - volume.start_lba, - volume.blocks, - volume.lba_bytes - ); - return None; - } - Err(block::SpanRefused::NotWhole { start_bytes, len_bytes }) => { - log!( - "data: candidate {guid} is at {start_bytes}+{len_bytes} bytes, which is not \ - whole {PAGE}-byte blocks — refusing it" - ); - return None; - } - }; - let device_blocks = handle.block_count(); - let part = match Partition::of(handle, first_block, blocks, block::Holder::Kernel("data")) { - Ok(part) => part, - Err(block::ViewRefused::OffDevice) => { - log!( - "data: candidate {guid} is at {first_block}+{blocks} blocks on a device of {} \ - blocks — refusing to read past the end of it", - device_blocks - ); - return None; - } - Err(block::ViewRefused::Held(by)) => { - log!("data: candidate {guid} is held by {by} — refusing to open it a second time"); - return None; - } - }; - Some(init(part)) -} - -impl Cached { - pub fn partition(&self) -> &Partition { - &self.part - } - - /// Locks cache then device, in that order. - pub fn lock(&self) -> PageCacheGuard<'_> { - let cache = self.cache.lock(); - let dev = self.part.lock(); - PageCacheGuard { cache, dev, part: &self.part } - } - - /// Reads a block of this partition past the cache; locks only the device. - #[must_use = "a failed read leaves the buffer holding whatever it held before"] - pub fn raw_read(&self, block: u64, buf: &mut [u8; PAGE_BYTES]) -> BlockResult { - self.part.read_blocks(block, 1, buf) - } - - /// Writes a block past the cache's slots but not its flush debt: the bytes - /// land in the device's write cache, which the next sync owes a flush. - #[must_use = "a failed write did not reach the device"] - pub fn raw_write(&self, block: u64, buf: &[u8; PAGE_BYTES]) -> BlockResult { - let mut guard = self.lock(); - let at = guard.part.locate(block, 1)?; - guard.cache.flush.record_write(); - guard.dev.write_blocks(at, 1, buf) - } -} - -pub struct PageCacheGuard<'a> { - cache: LockGuard<'a, PageCache>, - dev: block::Locked<'a>, - part: &'a Partition, -} - -impl PageCacheGuard<'_> { - pub fn block_count(&self) -> u64 { - self.part.block_count() - } - - pub fn read(&mut self, block: u64) -> Result<&[u8], BlockError> { - let Self { cache, dev, part } = self; - cache.read(part, dev, block) - } - - pub fn write_new(&mut self, block: u64) -> Result<&mut [u8], BlockError> { - let Self { cache, dev, part } = self; - cache.write_new(part, dev, block) - } - - pub fn sync(&mut self) -> BlockResult { - let Self { cache, dev, .. } = self; - cache.sync(dev) - } -} - -/// 256 pages per chunk so each chunk allocation is exactly 1MB. -const PAGES_PER_CHUNK: usize = 256; -const CHUNK_SIZE: usize = PAGES_PER_CHUNK * 4096; - -pub struct PageCache { - /// Maps a block's identity → slot index; sized by the cached set, never by the partition's block count. - block_to_slot: HashMap, - /// Maps slot index → the block it holds, for sync and eviction; `None` is a slot bound to nothing. - slot_to_block: Vec>, - dirty: Vec, - /// CLOCK's second-chance bit: set on every hit, cleared when the hand passes. - /// Without it the cache degenerates to FIFO and can evict the superblock like any other slot. - referenced: Vec, - /// Allocated in fixed 1MB chunks rather than one buffer, to avoid reallocating the whole cache as it grows. - /// `Box<[u8]>`, not `Box<[u8; CHUNK_SIZE]>`: the array type would build a 1 MiB stack temporary before boxing. - chunks: Vec>, - hand: u32, - max_slots: usize, - evictions: u64, - /// The device flush [`Self::sync`] still owes: raised by every write that - /// reached the device's cache, settled only by a `dev.flush()` that - /// returned `Ok` — so an empty dirty set never skips a flush that is owed. - flush: crate::durability::Owed, -} - -impl PageCache { - fn new() -> Self { - let max_slots = block::metadata_cache_blocks(); - Self { - // Reserved up front so `index_capacity` reports the real ceiling immediately. - block_to_slot: HashMap::with_capacity_and_hasher(max_slots, KernelHashState::new()), - slot_to_block: Vec::with_capacity(max_slots), - dirty: Vec::with_capacity(max_slots), - referenced: Vec::with_capacity(max_slots), - chunks: Vec::with_capacity(max_slots.div_ceil(PAGES_PER_CHUNK)), - hand: 0, - max_slots, - evictions: 0, - flush: crate::durability::Owed::new(), - } - } - - /// Blocks the index has room for; must stay far below the partition's block count. - pub fn index_capacity(&self) -> usize { - self.block_to_slot.capacity() - } - - /// What that index costs: a `(key, slot)` pair plus a control byte per - /// bucket, over hashbrown's 7/8 load factor. - pub fn index_bytes(&self) -> usize { - let buckets = self.index_capacity().div_ceil(7) * 8; - buckets * (core::mem::size_of::<(BlockKey, u32)>() + 1) - } - - /// Returns `None` only when every resident slot is dirty and write-back failed. - fn alloc_slot(&mut self, dev: &mut dyn BlockDevice, key: BlockKey) -> Option { - let slot = if self.slot_to_block.len() < self.max_slots { - let slot = self.slot_to_block.len() as u32; - self.slot_to_block.push(Some(key)); - self.dirty.push(false); - self.referenced.push(false); - if slot as usize / PAGES_PER_CHUNK >= self.chunks.len() { - self.chunks.push(vec![0u8; CHUNK_SIZE].into_boxed_slice()); - } - slot - } else { - let slot = self.take_victim(dev)?; - if let Some(evicted) = self.slot_to_block[slot as usize] { - self.block_to_slot.remove(&evicted); - } - self.slot_to_block[slot as usize] = Some(key); - self.dirty[slot as usize] = false; - self.evictions += 1; - // Logs once per full cache turnover, not on a fixed count, so the cadence scales with the bound. - if self.evictions == 1 || self.evictions.is_multiple_of(self.max_slots as u64) { - log!("page cache: {} evictions, {}/{} slots resident", - self.evictions, self.slot_to_block.len(), self.max_slots); - } - slot - }; - self.referenced[slot as usize] = true; - self.block_to_slot.insert(key, slot); - Some(slot) - } - - /// Clears the block's index entry so a later reader misses instead of reading stale data. - fn unbind(&mut self, slot: u32, key: BlockKey) { - self.block_to_slot.remove(&key); - self.slot_to_block[slot as usize] = None; - self.dirty[slot as usize] = false; - self.referenced[slot as usize] = false; - } - - /// Frees a slot for reuse, writing back first when every clean slot is exhausted. - /// Writing back early is safe: there is no journal, so eviction only reorders writes that were never ordered. - fn take_victim(&mut self, dev: &mut dyn BlockDevice) -> Option { - if let Some(slot) = self.clock_pick() { - return Some(slot); - } - // Every resident slot is dirty; one write-back clears them all unless the device refuses. - if self.sync(dev).is_err() { - log!("page cache: write-back failed; no slot could be freed"); - } - self.clock_pick() - } - - /// CLOCK second-chance eviction over two revolutions of the hand. - /// First revolution clears reference bits; second finds a victim unless every slot is dirty. - fn clock_pick(&mut self) -> Option { - let n = self.slot_to_block.len() as u32; - for _ in 0..2 * n { - let slot = self.hand as usize; - self.hand = if self.hand + 1 == n { 0 } else { self.hand + 1 }; - if self.dirty[slot] { - continue; - } - if self.referenced[slot] { - self.referenced[slot] = false; - continue; - } - return Some(slot as u32); - } - None - } - - fn slot_data(&self, slot: u32) -> &[u8] { - let chunk_idx = slot as usize / PAGES_PER_CHUNK; - let page_in_chunk = slot as usize % PAGES_PER_CHUNK; - let off = page_in_chunk * 4096; - &self.chunks[chunk_idx][off..off + 4096] - } - - fn slot_data_mut(&mut self, slot: u32) -> &mut [u8] { - let chunk_idx = slot as usize / PAGES_PER_CHUNK; - let page_in_chunk = slot as usize % PAGES_PER_CHUNK; - let off = page_in_chunk * 4096; - &mut self.chunks[chunk_idx][off..off + 4096] - } - - fn slot_of(&self, key: BlockKey) -> Option { - self.block_to_slot.get(&key).copied() - } - - fn read( - &mut self, - part: &Partition, - dev: &mut dyn BlockDevice, - block: u64, - ) -> Result<&[u8], BlockError> { - let key = part.key(block)?; - if let Some(slot) = self.slot_of(key) { - self.referenced[slot as usize] = true; - return Ok(self.slot_data(slot)); - } - let slot = self.alloc_slot(dev, key).ok_or(BlockError::Device)?; - let page = self.slot_data_mut(slot); - if let Err(e) = dev.read_blocks(key.device_block(), 1, page) { - self.unbind(slot, key); - return Err(e); - } - Ok(self.slot_data(slot)) - } - - fn write_new( - &mut self, - part: &Partition, - dev: &mut dyn BlockDevice, - block: u64, - ) -> Result<&mut [u8], BlockError> { - let key = part.key(block)?; - let slot = match self.slot_of(key) { - Some(slot) => { - self.referenced[slot as usize] = true; - slot - } - None => self.alloc_slot(dev, key).ok_or(BlockError::Device)?, - }; - self.dirty[slot as usize] = true; - let page = self.slot_data_mut(slot); - // A reused slot still holds the evicted block's bytes; zero it before returning. - page.fill(0); - Ok(page) - } - - /// Writes every dirty slot back, coalescing runs of consecutive blocks. - /// Every address comes from the key the slot was filled under, so a page is - /// written where it was read from and nowhere else. - fn sync(&mut self, dev: &mut dyn BlockDevice) -> BlockResult { - let mut pending: Vec = (0..self.slot_to_block.len() as u32) - .filter(|&s| self.dirty[s as usize]) - .collect(); - - // Nothing to write is not nothing to flush: a raw write, or a failed predecessor's runs, is still owed. - if pending.is_empty() && !self.flush.is_owed() { - return Ok(()); - } - pending.sort_unstable_by_key(|&s| self.slot_to_block[s as usize]); - - // A dirty slot is bound by construction: `unbind` clears the dirty bit - // with the key, so nothing here can be resident-and-unbound. - let at = |cache: &Self, slot: u32| { - cache.slot_to_block[slot as usize] - .expect("a dirty slot holds the block it was filled under") - .device_block() - }; - - let mut buf = vec![0u8; 32 * 4096]; - // Errors combine via `worse`, not first-wins, so a caller sees the one that blocks retry. - let mut failed: Option = None; - let mut i = 0; - while i < pending.len() { - let start = at(self, pending[i]); - let mut count = 1usize; - - while i + count < pending.len() - && at(self, pending[i + count]) == start + count as u64 - && count < 32 - { - count += 1; - } - - for j in 0..count { - let page = self.slot_data(pending[i + j]); - buf[j * 4096..(j + 1) * 4096].copy_from_slice(page); - } - - // A failed run stays dirty for retry; the loop continues rather than aborting on it. - match dev.write_blocks(start, count as u32, &buf[..count * 4096]) { - Ok(()) => { - self.flush.record_write(); - for j in 0..count { - self.dirty[pending[i + j] as usize] = false; - } - } - Err(e) => { - log!("page cache: write-back of {count} blocks at {start} failed"); - failed = Some(failed.map_or(e, |had| had.worse(e))); - } - } - - i += count; - } - - // The lock is held throughout, so the snapshot covers every write above. - let upto = self.flush.snapshot(); - match dev.flush() { - Ok(()) => self.flush.settle(upto), - Err(e) => { - log!("page cache: flush failed; the write-back above is not durable"); - failed = Some(failed.map_or(e, |had| had.worse(e))); - } - } - failed.map_or(Ok(()), Err) - } -} - -/// Read-fault injection (`pc-unbind-selftest`): refuse one armed block, count served reads of another. -#[cfg(feature = "boot-actuators")] -mod read_fault { - use core::sync::atomic::{AtomicU64, Ordering}; - - use alloc::boxed::Box; - - use crate::block::{BlockDevice, BlockError, BlockResult, DeviceId}; - - // `u64::MAX` disarms; no device reaches it. - pub(super) static FAIL_BLOCK: AtomicU64 = AtomicU64::new(u64::MAX); - pub(super) static WATCH_BLOCK: AtomicU64 = AtomicU64::new(u64::MAX); - pub(super) static SERVED: AtomicU64 = AtomicU64::new(0); - - pub(super) struct FaultDevice(pub Box); - - fn covers(lba: u64, count: u32, block: u64) -> bool { - lba <= block && block - lba < count as u64 - } - - impl BlockDevice for FaultDevice { - fn device_id(&self) -> DeviceId { - self.0.device_id() - } - - fn block_count(&self) -> u64 { - self.0.block_count() - } - - fn read_blocks(&mut self, lba: u64, count: u32, buf: &mut [u8]) -> BlockResult { - if covers(lba, count, FAIL_BLOCK.load(Ordering::Relaxed)) { - return Err(BlockError::Device); - } - let read = self.0.read_blocks(lba, count, buf); - if read.is_ok() && covers(lba, count, WATCH_BLOCK.load(Ordering::Relaxed)) { - SERVED.fetch_add(1, Ordering::Relaxed); - } - read - } - - fn write_blocks(&mut self, lba: u64, count: u32, buf: &[u8]) -> BlockResult { - self.0.write_blocks(lba, count, buf) - } - - fn flush(&mut self) -> BlockResult { - self.0.flush() - } - - fn losses(&self) -> u64 { - self.0.losses() - } - } -} - -/// `partclaim-table-unanswered`: every instrumented disk refuses reads of its -/// device block 0 from here on. Armed after the mounts, which read their own -/// partitions and nothing there again. -#[cfg(feature = "boot-actuators")] -pub fn refuse_table_reads() { - read_fault::FAIL_BLOCK.store(0, core::sync::atomic::Ordering::Relaxed); - log!("partclaim-table-unanswered: device block 0 of every NVMe disk refuses reads from now on"); -} - -/// The un-index control, behind `pc-unbind-selftest`, for `PageCache::read`'s -/// unbind-on-failed-fill. The count is the assertion: after a refused fill, the -/// next read of the same block reaches the device exactly once — a slot left -/// bound answers from its last tenant and reaches it zero times. The byte -/// comparison against the device (past the cache) is the differential half. -/// One guard held throughout, so nothing touches the armed block mid-sequence. -#[cfg(feature = "boot-actuators")] -pub fn unbind_selftest(cached: &Cached) { - use core::sync::atomic::Ordering; - - let mut guard = cached.lock(); - // The highest non-resident block: read by nothing so far, or long evicted. - let Some((block, key)) = (0..guard.block_count()) - .rev() - .filter_map(|b| guard.part.key(b).ok().map(|k| (b, k))) - .find(|(_, k)| !guard.cache.block_to_slot.contains_key(k)) - else { - log!("pc-unbind-selftest: FAIL (every device block is resident)"); - return; - }; - - read_fault::SERVED.store(0, Ordering::Relaxed); - read_fault::WATCH_BLOCK.store(key.device_block(), Ordering::Relaxed); - read_fault::FAIL_BLOCK.store(key.device_block(), Ordering::Relaxed); - let refused = guard.read(block).is_err(); - read_fault::FAIL_BLOCK.store(u64::MAX, Ordering::Relaxed); - if !refused { - log!("pc-unbind-selftest: FAIL (the injected read fault never fired)"); - return; - } - - let mut reread = vec![0u8; PAGE_BYTES].into_boxed_slice(); - match guard.read(block) { - Ok(data) => reread.copy_from_slice(data), - Err(_) => { - log!("pc-unbind-selftest: FAIL (the re-read after the failed fill was refused)"); - return; - } - } - let served = read_fault::SERVED.load(Ordering::Relaxed); - read_fault::WATCH_BLOCK.store(u64::MAX, Ordering::Relaxed); - - let mut raw = vec![0u8; PAGE_BYTES].into_boxed_slice(); - if guard.dev.read_blocks(key.device_block(), 1, &mut raw).is_err() { - log!("pc-unbind-selftest: FAIL (the ground-truth device read was refused)"); - return; - } - - if served == 1 && reread[..] == raw[..] { - log!("pc-unbind-selftest: PASS (block {block}: 1 device read after the failed fill, bytes match the device)"); - } else { - log!( - "pc-unbind-selftest: FAIL (block {block}: {served} device reads after the failed \ - fill, bytes {}the device's — the slot stayed bound to a block it never read)", - if reread[..] == raw[..] { "match " } else { "differ from " } - ); - } -} - -/// The partition-offset control (`pc-partition-offset`): a cache over a view -/// that does not begin at block 0 writes through the view's block numbers, and -/// the bytes have to land at the device's. Read back past the cache and past -/// the view, so only the key the slot was filled under could have offset it. -#[cfg(feature = "boot-actuators")] -pub fn partition_offset_selftest(handle: &block::Handle) { - let Ok(part) = block::Partition::of( - handle.clone(), - offset_probe::FIRST, - offset_probe::BLOCKS, - block::Holder::Kernel("pc-partition-offset probe"), - ) else { - log!("pc-partition-offset: FAIL (no view over device {})", handle.device_id()); - return; - }; - let cached = init(part); - { - let mut guard = cached.lock(); - let Ok(page) = guard.write_new(offset_probe::AT) else { - log!("pc-partition-offset: FAIL (the view refused a write at its own block)"); - return; - }; - page[..offset_probe::MARK.len()].copy_from_slice(offset_probe::MARK); - if guard.sync().is_err() { - log!("pc-partition-offset: FAIL (the write-back was refused)"); - return; - } - } - - let mut at_offset = vec![0u8; PAGE_BYTES]; - let mut at_zero = vec![0u8; PAGE_BYTES]; - let device_block = offset_probe::FIRST + offset_probe::AT; - if handle.lock().read_blocks(device_block, 1, &mut at_offset).is_err() { - log!("pc-partition-offset: FAIL (device block {device_block} would not read back)"); - return; - } - if handle.lock().read_blocks(offset_probe::AT, 1, &mut at_zero).is_err() { - log!("pc-partition-offset: FAIL (device block {} would not read back)", offset_probe::AT); - return; - } - let marked = |b: &[u8]| b[..offset_probe::MARK.len()] == *offset_probe::MARK; - log!( - "pc-partition-offset: view +{} block {} landed_at_{device_block}={} at_block_{}={}", - offset_probe::FIRST, - offset_probe::AT, - marked(&at_offset), - offset_probe::AT, - marked(&at_zero) - ); -} - -/// Mirrored by `tests/common/storage.rs`, which reads the same two device -/// blocks off the image after the guest has gone. -#[cfg(feature = "boot-actuators")] -mod offset_probe { - pub const FIRST: u64 = 3000; - pub const BLOCKS: u64 = 64; - pub const AT: u64 = 7; - pub const MARK: &[u8] = b"TOYOS-PARTITION-OFFSET"; -} diff --git a/kernel/src/pcidev/mod.rs b/kernel/src/pcidev/mod.rs index 498c99ffa3e..0157506380c 100644 --- a/kernel/src/pcidev/mod.rs +++ b/kernel/src/pcidev/mod.rs @@ -1766,7 +1766,7 @@ fn foreign_if_armed(first: bool, at: u64) -> u64 { #[cfg(feature = "boot-actuators")] if first && crate::actuator::iommu_userdev_foreign_dma() { let foreign = - crate::drivers::nvme::FOREIGN_PROBE.load(core::sync::atomic::Ordering::Relaxed); + crate::drivers::xhci::FOREIGN_PROBE.load(core::sync::atomic::Ordering::Relaxed); if foreign != 0 { return foreign; } diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 5b42565c89b..1d3dcc9cd3f 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -129,18 +129,6 @@ pub fn note_progress() { } } -/// Whether the running thread is the one performing the shutdown: what the -/// `quiesce-drain-refuse` actuator refuses by. -#[cfg(feature = "boot-actuators")] -pub fn runs_the_shutdown() -> bool { - if STAGE.read() != STOPPING { - return false; - } - let (Some(pid), Some(tid)) = (percpu::current_pid(), percpu::current_tid()) else { - return false; - }; - ThreadId { pid: pid.raw(), tid: tid.raw() } == caller() -} /// Stop every userland thread but the caller, and answer with what it took. /// diff --git a/kernel/src/syscall/dispatch.rs b/kernel/src/syscall/dispatch.rs index b70b3206b39..1fe5bdb3bb3 100644 --- a/kernel/src/syscall/dispatch.rs +++ b/kernel/src/syscall/dispatch.rs @@ -230,8 +230,17 @@ pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> } else { alloc::vec::Vec::new() }; + // Last, because it is the one copy that can be large: every refusal + // above costs the caller nothing it must hand over again. + let image = match args.image_len { + 0 => None, + len => match crate::file_backing::ImageBacking::copy_in(&ctx, UserAddr::new(args.image_ptr), len) { + Ok(image) => Some(alloc::sync::Arc::new(image) as alloc::sync::Arc), + Err(e) => return e.to_u64(), + }, + }; let argv: alloc::vec::Vec<&str> = text.split('\0').filter(|s| !s.is_empty()).collect(); - sys_spawn(&argv, pending, cwd, env) + sys_spawn(&argv, pending, cwd, env, image) } SYS_PROCESS_WAIT => sys_process_wait(RawHandle(a1 as u32), a2), SYS_PROCESS_KILL => { @@ -333,8 +342,19 @@ pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> None => return bad_addr, }, }; + let image = match a4 { + 0 => None, + raw => { + let Some(at) = UserAddr::checked(raw) else { return bad_addr }; + let Ok(image) = ctx.copy_in::(at) else { return bad_addr }; + match crate::file_backing::ImageBacking::copy_in(&ctx, UserAddr::new(image.ptr), image.len) { + Ok(image) => Some(image), + Err(e) => return e.to_u64(), + } + } + }; // ctx carries the copy-out: sys_dlopen writes init_out only once the load succeeds. - sys_dlopen(&ctx, &path, init_out) + sys_dlopen(&ctx, &path, init_out, image) } SYS_DLSYM => { let name = match ctx.user_str(UserAddr::new(a2), a3) { Ok(s) => s, Err(e) => return e.to_u64() }; diff --git a/kernel/src/syscall/machine.rs b/kernel/src/syscall/machine.rs index bc91cb9c007..afb4fd33815 100644 --- a/kernel/src/syscall/machine.rs +++ b/kernel/src/syscall/machine.rs @@ -100,7 +100,6 @@ fn quiesce(last: &str) -> Result<(), SyscallError> { crate::vfs::lock().sync_all(); // The final census: no process runs after this to report another. crate::irq_census::log_census(); - crate::drivers::nvme::log_census(); crate::drivers::panic_console::log_census(); // A shortfall is the budget spent, not the reset refused: it is said at // alert level, and the reset lands anyway. diff --git a/kernel/src/syscall/proc.rs b/kernel/src/syscall/proc.rs index 8f3d825ddcf..b7301674582 100644 --- a/kernel/src/syscall/proc.rs +++ b/kernel/src/syscall/proc.rs @@ -37,9 +37,10 @@ pub(super) fn sys_spawn( pending: crate::loader::PendingHandles, cwd: alloc::string::String, env: Vec, + image: Option>, ) -> u64 { // Nothing to clean up yet: spawn's frame owns the child's resources on error. - let object = match process::spawn(args, pending, cwd, env) { + let object = match process::spawn(args, pending, cwd, env, image) { Ok(object) => object, Err(e) => return e.refuse(), }; diff --git a/kernel/src/syscall/vm.rs b/kernel/src/syscall/vm.rs index c714c74c5f5..298165343c0 100644 --- a/kernel/src/syscall/vm.rs +++ b/kernel/src/syscall/vm.rs @@ -193,7 +193,15 @@ pub(super) fn sys_munmap(addr: u64, _size: u64) -> u64 { 0 } -pub(super) fn sys_dlopen(ctx: &crate::user_ptr::SyscallContext, path: &str, init_out: Option) -> u64 { +/// `image` is the library's bytes when the caller read them itself, and `path` +/// then only names it: an image is loaded afresh and kept out of the shared +/// cache, which answers for what a path holds and an image is no path's. +pub(super) fn sys_dlopen( + ctx: &crate::user_ptr::SyscallContext, + path: &str, + init_out: Option, + image: Option, +) -> u64 { let cwd = process::with_process_data(|d| d.cwd.clone()); let resolved = vfs::lock().resolve_absolute(&cwd, path); @@ -211,15 +219,37 @@ pub(super) fn sys_dlopen(ctx: &crate::user_ptr::SyscallContext, path: &str, init return idx as u64; } + let loaded = match image { + Some(image) => match crate::elf::load_shared_lib(&image) { + Ok((lib, _, _)) => Ok(lib), + Err(msg) => { + log!("dlopen: {}: {}", resolved, msg); + return SyscallError::Unknown.to_u64(); + } + }, + None => Err(()), + }; + let lib = match loaded { + Ok(lib) => lib, + Err(()) => match open_cached(&resolved) { + Ok(lib) => lib, + Err(e) => return e, + }, + }; + map_dlopened(ctx, &resolved, init_out, lib) +} + +/// A library at `resolved`, out of the shared cache or loaded into it. +fn open_cached(resolved: &str) -> Result { // Opened before the cache is consulted: the answer depends on what the path holds *now*, and the fast path above is what keeps a `dlopen` loop from paying. - let (backing, id) = match vfs::lock().open_backing_identified(&resolved) { + let (backing, id) = match vfs::lock().open_backing_identified(resolved) { Ok(pair) => pair, Err(e) => { log!("dlopen: {}: {e}", resolved); - return e.to_u64(); + return Err(e.to_u64()); } }; - let mut lib = match crate::elf::try_clone_cached(&resolved, id) { + let lib = match crate::elf::try_clone_cached(resolved, id) { Ok(Some(lib)) => lib, Err(e) => { log!( @@ -227,24 +257,33 @@ pub(super) fn sys_dlopen(ctx: &crate::user_ptr::SyscallContext, path: &str, init it was loaded, and the image cannot be replaced while a process has it mapped", resolved ); - return e.to_u64(); + return Err(e.to_u64()); } Ok(None) => { let (lib, rw_offset, rw_size) = match crate::elf::load_shared_lib(backing.as_ref()) { Ok(result) => result, Err(msg) => { log!("dlopen: {}", msg); - return SyscallError::Unknown.to_u64(); + return Err(SyscallError::Unknown.to_u64()); } }; - match crate::elf::cache_loaded_lib(&resolved, id, lib, rw_offset, rw_size) { + match crate::elf::cache_loaded_lib(resolved, id, lib, rw_offset, rw_size) { Ok(lib) => lib, - Err(e) => return e.to_u64(), + Err(e) => return Err(e.to_u64()), } } }; + Ok(lib) +} +/// Map `lib` into the caller, run its relocations, and register it under `resolved`. +fn map_dlopened( + ctx: &crate::user_ptr::SyscallContext, + resolved: &str, + init_out: Option, + mut lib: crate::elf::LoadedLib, +) -> u64 { let pt = process::current_address_space(); let mapped = process::with_process_data(|_data| { // The module's own program headers decide which pages are writable @@ -347,7 +386,7 @@ pub(super) fn sys_dlopen(ctx: &crate::user_ptr::SyscallContext, path: &str, init is_static: false, }); } - data.elf.lib_paths.push(resolved); + data.elf.lib_paths.push(alloc::string::String::from(resolved)); data.elf.loaded_libs.push(lib); idx as u64 } diff --git a/kernel/src/time.rs b/kernel/src/time.rs index 85cf1c1479a..21e665d7ba6 100644 --- a/kernel/src/time.rs +++ b/kernel/src/time.rs @@ -138,12 +138,6 @@ impl Bound { Self { limit, cite } } - /// A bound a device register publishes; `cite` names the register. Not - /// `const`: the number is read off the hardware at the call site. - pub fn from_register(limit: Duration, cite: &'static str) -> Self { - Self { limit, cite } - } - pub const fn duration(self) -> Duration { self.limit } diff --git a/kernel/src/user_ptr.rs b/kernel/src/user_ptr.rs index 36b4212f7dd..0d1f05f1478 100644 --- a/kernel/src/user_ptr.rs +++ b/kernel/src/user_ptr.rs @@ -41,8 +41,10 @@ unsafe impl UserSafe for [u64; 2] {} // SAFETY: `#[repr(C)] Copy`, three `u64`s, no padding; `file_type` is a `u64`, not the enum it names, so every bit pattern stays valid. unsafe impl UserSafe for crate::object::ops::Stat {} -// SAFETY: `#[repr(C)] Copy`, ten `u64`s, no padding; every field is validated where it is used, not here. +// SAFETY: `#[repr(C)] Copy`, fourteen `u64`s, no padding; every field is validated where it is used, not here. unsafe impl UserSafe for toyos_abi::syscall::SpawnArgs {} +// SAFETY: `#[repr(C)] Copy`, two `u64`s, no padding; both are bounded by the copy that reads them. +unsafe impl UserSafe for toyos_abi::syscall::ImageRef {} // SAFETY: `#[repr(C)] Copy`, `RawHandle`, a `flags: u32`, then six `u64`s — no padding. unsafe impl UserSafe for toyos_abi::syscall::NamespaceBuild {} // SAFETY: `#[repr(C)] Copy`, `u64`, `u64`, `i64` — 24 bytes, no padding. diff --git a/kernel/src/vfs.rs b/kernel/src/vfs.rs index efd94d853a2..1576f8c1c89 100644 --- a/kernel/src/vfs.rs +++ b/kernel/src/vfs.rs @@ -320,8 +320,10 @@ impl Vfs { return Ok(abs); } - // A mount point exists whether or not anything is mounted on it. - if subdir.is_empty() && Self::entry(&mount).is_some() { + // A mount point exists whether or not anything is mounted on it; and + // under one this kernel mounts nothing at, the directory is a file + // server's, which the caller's client asked and this kernel cannot. + if Self::entry(&mount).is_some() && (subdir.is_empty() || self.point(&mount).is_none()) { return Ok(abs); } @@ -673,11 +675,6 @@ impl Vfs { Ok(()) } - /// Is there a filesystem mounted under `name`? - pub fn has_mount(&self, name: &str) -> bool { - self.point(name).is_some() - } - /// `/system/bin/logd` calls a line durable off this call's result, so `sync_for_path` must reach the device's write cache, not stop at the page cache. pub fn sync_for_path(&mut self, path: &str) -> Result<(), SyscallError> { let (mount, _) = self.resolve_path("/", path); diff --git a/kernel/src/writeback.rs b/kernel/src/writeback.rs index 493d3a6ab94..0b465772305 100644 --- a/kernel/src/writeback.rs +++ b/kernel/src/writeback.rs @@ -106,12 +106,7 @@ fn drain_one(vfs: &mut crate::vfs::Vfs, deadman: Deadline) -> Drained { // A deleted file has nothing to flush; its handle is already torn down. let probe = file_cache::writeback_probe(pending.file_id); if probe.flush_owed && !probe.deleted { - // Tags the flush as the drain's, for `fat-mirror-write-refuse` to stage on this path. - #[cfg(feature = "boot-actuators")] - crate::fat32_adapter::enter_drain_flush(); let flushed = vfs.flush_file(&pending.path, pending.file_id, pending.mtime); - #[cfg(feature = "boot-actuators")] - crate::fat32_adapter::leave_drain_flush(); match flushed { Ok(()) => {} // Budget refusal, not a device fact: the flush's debt is unsettled, so the re-enqueued entry redelivers the same bytes. diff --git a/src/build.rs b/src/build.rs index 9e2d138b2f9..55edd4d5e7d 100644 --- a/src/build.rs +++ b/src/build.rs @@ -138,6 +138,11 @@ struct ProgramConfig { /// A system service: init starts it with `HOME` at its own `/state/` /// and makes that directory, where every other row gets the session's. service: bool, + /// The file-server roles this binary serves, one process each + /// (`toyos_manifest::Program::roles`). + roles: Vec, + /// init starts it again when it ends (`toyos_manifest::Program::restart`). + restart: bool, } impl ProgramConfig { @@ -733,6 +738,8 @@ fn render_manifest(config: &SystemConfig) -> Vec { syscap: cfg.syscap.clone(), slots: cfg.slots, service: cfg.service, + roles: cfg.roles.clone(), + restart: cfg.restart, } }) .collect(), diff --git a/src/image.rs b/src/image.rs index 0c416a1fa30..182e7f55cc6 100644 --- a/src/image.rs +++ b/src/image.rs @@ -504,11 +504,11 @@ fn round_up_sectors(n: usize) -> usize { /// Where each partition is made to start. /// -/// A correctness requirement rather than tidiness. The kernel's `BlockDevice` -/// transfers whole 4 KiB blocks and each mounted volume keeps its own resident -/// copies of the blocks it has touched (`fat32_adapter::FatDevice`); two -/// partitions sharing one device block would make each other's copies stale -/// with nothing able to notice. 1 MiB rather than the 4096 the kernel needs, +/// A correctness requirement rather than tidiness. Every block service +/// transfers whole 4 KiB blocks and each file server keeps its own cached +/// copies of the blocks it has touched (`userland/fsd`); two partitions +/// sharing one device block would make each other's copies stale with +/// nothing able to notice. 1 MiB rather than the 4096 the kernel needs, /// because that is what every partitioner uses and what an erase block wants. const PARTITION_ALIGN: usize = 1024 * 1024; diff --git a/src/kernelkeys.rs b/src/kernelkeys.rs index bc54b91b578..6181def2267 100644 --- a/src/kernelkeys.rs +++ b/src/kernelkeys.rs @@ -43,16 +43,6 @@ pub struct Declared { /// Every hashed container in `kernel/src`, with the origin of its keys. pub const DECLARED: &[Declared] = &[ - Declared { - file: "kernel/src/bcachefs_adapter.rs", - ty: "HashMap", - keys: "`file_cache::FileId`, minted by the file cache", - }, - Declared { - file: "kernel/src/fat32_adapter.rs", - ty: "HashMap", - keys: "`file_cache::FileId`, minted by the file cache", - }, Declared { file: "kernel/src/id_map.rs", ty: "HashMap", @@ -63,11 +53,6 @@ pub const DECLARED: &[Declared] = &[ ty: "HashMap", keys: "a physical address the page allocator returned", }, - Declared { - file: "kernel/src/page_cache.rs", - ty: "HashMap", - keys: "`block::BlockKey`, minted only by a `Partition` this kernel opened and bounded by that view", - }, Declared { file: "kernel/src/scheduler.rs", ty: "HashMap>", diff --git a/src/qemu.rs b/src/qemu.rs index 9f77d7545d0..d2773b84adf 100644 --- a/src/qemu.rs +++ b/src/qemu.rs @@ -179,7 +179,7 @@ pub fn launch(opts: &Options) { .arg("-drive") .arg("if=none,id=nvme0,format=raw,file=target/nvme.img") .arg("-device") - .arg("nvme,serial=deadbeef,drive=nvme0"); + .arg("nvme,serial=deadbeef,drive=nvme0,msix-exclusive-bar=on"); if shape.usb_hid { qemu.arg("-device") diff --git a/src/sourcegate.rs b/src/sourcegate.rs index 84f2bd60aa7..7ed5666c2b8 100644 --- a/src/sourcegate.rs +++ b/src/sourcegate.rs @@ -438,7 +438,6 @@ const BUS_MASTER_SITES: &[(&str, usize)] = &[ ("kernel/src/drivers/pci.rs", 1), ("kernel/src/drivers/virtio.rs", 1), ("kernel/src/drivers/hda.rs", 1), - ("kernel/src/drivers/nvme.rs", 1), ("kernel/src/drivers/xhci/wait/boot.rs", 1), ]; diff --git a/system.toml b/system.toml index efb56cebed0..d07a38ed45b 100644 --- a/system.toml +++ b/system.toml @@ -21,7 +21,7 @@ assets = ["assets"] # while the compositor had no way to ask for a program's *declared* authority — # the launcher gives it one, and starting a service from `[boot] start` is not # the thing that needed fixing. -start = ["logd", "compositor", "soundd", "netd", "filepicker"] +start = ["logd", "blockd", "fsd", "compositor", "soundd", "netd", "filepicker"] # What every program launched out of `/apps` holds, whatever its own directory # says. A package is unpacked by `/system/bin/pkg` into a writable directory, so @@ -176,3 +176,20 @@ syscap = ["roster"] "bin/spin" = "/system/bin/toybox" "bin/stats" = "/system/bin/toybox" "bin/tone" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/blockdcase/system.toml b/tests/blockdcase/system.toml index 623c5130aa2..ed219ed161a 100644 --- a/tests/blockdcase/system.toml +++ b/tests/blockdcase/system.toml @@ -1,13 +1,13 @@ -# The blockd boot: `tests/testcases`'s estate, and `/system/bin/blockd` in the -# image started by nothing. `test_rs_blockd_io` is blockd's supervisor as init -# is a service's: it mints the claim on the machine's second NVMe controller -# from its capability, starts blockd holding it and a port it made, and is the -# only thing that can end or restart it. The disks are crafted by +# The blockd boot: `tests/testcases`'s estate, with a second blockd beside the +# one init runs for DATA. `test_rs_blockd_io` is the second's supervisor as +# init is a service's: it mints the claim on the machine's second NVMe +# controller from its capability, starts blockd holding it and a port it made, +# and is the only thing that can end or restart it. The disks are crafted by # `tests/common/blockd.rs`. No soundd: `test_rs_blockd_io dma-pool` claims # virtio-sound itself, to lend the pool its kernel driver keeps. [boot] -start = ["logd", "test-runner"] +start = ["logd", "blockd", "fsd", "test-runner"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -37,10 +37,6 @@ syscap = ["logread"] [programs.test-runner] syscap = ["device", "dup", "logread", "power", "roster"] -# The NVMe driver, from userland. No row starts it and it holds nothing here: -# the test that runs it hands it its claim and its port itself. -[programs.blockd] - [programs.toybox] [symlinks] @@ -62,3 +58,20 @@ syscap = ["device", "dup", "logread", "power", "roster"] # `toyos-rust-tests` drains the same sink perfectly, so a suite that ran only # that one certified a path no user takes. "bin/tone" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/common/iommu.rs b/tests/common/iommu.rs index 3a4d86d5837..a2b92253803 100644 --- a/tests/common/iommu.rs +++ b/tests/common/iommu.rs @@ -2069,10 +2069,9 @@ pub fn userdev_dma_fault( // The fault was handed to the process that drives the stream, and the line // says so: `owner=kernel` here would be a machine that halted, or was about - // to. `slot0` is the first `pcidev` slot, which is netd's — the only claim - // this config mints. + // to. let handled = log.must_say(FAULT)?; - if !handled.contains("owner=slot0") { + if !handled.contains("owner=slot") { return Err(format!( "the unit's fault was recorded against {handled:?}, and the function that faulted \ is one a process drives. A fault the kernel takes as its own is one it halts for" @@ -2147,12 +2146,25 @@ pub fn userdev_residue_is_its_own( if result.exit_code != Some(0) { return Err(format!("userdev_residue: exit {:?}\n{}\n{}", result.exit_code, result.stdout, log.text())); } - let released = log.must_say("[8086:10d3] released from slot 0; reset by")?; + let slot = slot_of(log.text(), "[8086:10d3]")?; + let released = log.must_say(&format!("[8086:10d3] released from slot {slot}; reset by"))?; if !released.contains("reset by nothing") { return Err(format!("the premise: the 82574 was not released by nothing — {released}")); } - let kept = log.must_say("pcidev: slot 0 holds 1 range(s)")?.to_string(); + let kept = log.must_say(&format!("pcidev: slot {slot} holds 1 range(s)"))?.to_string(); log.must_be_clean()?; eprintln!(" [iommu] {}; {}", kept.trim_end(), result.stdout.trim_end()); Ok(()) } + +/// The `pcidev` slot the function with PCI ids `ids` (`[vvvv:dddd]`) was handed +/// over on: a boot's claims are minted in init's order, and the block service's +/// comes first on a machine with an NVMe controller. +pub(crate) fn slot_of(text: &str, ids: &str) -> Result { + text.lines() + .find_map(|l| { + let rest = l.split(&format!("{ids} handed over on slot ")).nth(1)?; + rest.split(|c: char| !c.is_ascii_digit()).next()?.parse().ok() + }) + .ok_or_else(|| format!("no function {ids} was handed over on any slot")) +} diff --git a/tests/common/power.rs b/tests/common/power.rs index 266d7329b15..509f0cf9680 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -170,8 +170,10 @@ pub fn quiesce_stops_the_machine( const WRITERS: u32 = 6; // The threads the stop names besides the writers: the job's own main // thread, parked on init's answer; `test-runner`'s main and deadline - // threads; and `logd`'s. `init` asked for the stop and is its caller. - const OTHERS: u32 = 4; + // threads; `logd`'s; `blockd`'s; one per file server, three roles; and + // `init`'s waiter on each of those four services. `init`'s main thread + // asked for the stop and is its caller. + const OTHERS: u32 = 1 + 2 + 1 + 1 + 3 + 4; let (whole, record) = stopped_boot("tests/quiescecase/system.toml", JOB, &[LATE_WORD], rust_bins)?; if record.in_flight != 0 { @@ -195,7 +197,7 @@ pub fn quiesce_stops_the_machine( if record.sweep.total() != WRITERS + OTHERS { return Err(format!( "this boot's stop named {} userland thread(s); {WRITERS} writers plus the {OTHERS} \ - of the job, test-runner and logd make {}, so this is not the machine the writers \ + of the job, test-runner, logd, the storage services and init's waiters make {}, so this is not the machine the writers \ were on:\n {record}\n{whole}", record.sweep.total(), WRITERS + OTHERS, diff --git a/tests/common/qemu.rs b/tests/common/qemu.rs index 7bc5712fa15..75fe9f9deb7 100644 --- a/tests/common/qemu.rs +++ b/tests/common/qemu.rs @@ -4559,7 +4559,7 @@ fn qemu_command( // volume rides its own NVMe controller and every xHCI carries HID alone. if shape.storage_bus.is_empty() { qemu.arg("-device") - .arg("nvme,serial=bootdisk,id=nvmebootctl,bootindex=0") + .arg("nvme,serial=bootdisk,id=nvmebootctl,bootindex=0,msix-exclusive-bar=on") .arg("-device") .arg("nvme-ns,drive=stick,bus=nvmebootctl,logical_block_size=512,\ physical_block_size=512"); @@ -4614,14 +4614,21 @@ fn qemu_command( // the guest no controller at all, rather than an empty one. A machine // with no NVMe is a shape, and the argv is the only place it is visible: // no console line and no screendump can see a device that is absent. + // + // Its MSI-X table in a BAR of its own, because blockd drives it and a claim + // never maps the BAR holding the table. On a machine that boots off NVMe it + // answers under Intel's ids, so the `pci:1b36:0010` row names the boot + // controller alone and this one is nobody's, as the kernel's first-by-class + // probe left it. if shape.nvme_bytes != 0 { + let ids = if shape.storage_bus.is_empty() { ",use-intel-id=on" } else { "" }; qemu.arg("-drive") .arg(format!( "if=none,id=nvme0,format=raw,file={}", nvme_image.display() )) .arg("-device") - .arg("nvme,serial=deadbeef,id=nvme0ctl") + .arg(format!("nvme,serial=deadbeef,id=nvme0ctl,msix-exclusive-bar=on{ids}")) .arg("-device") .arg(format!( "nvme-ns,drive=nvme0,bus=nvme0ctl,logical_block_size={0},physical_block_size={0}", diff --git a/tests/common/storage.rs b/tests/common/storage.rs index ddd95194979..bc0f65ca82e 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -16,6 +16,26 @@ use toyos_build::fingerprint::{first_difference, whole_device}; use super::qemu::{self, BootOptions, QemuInstance}; +/// fsd's word for a DATA partition that holds neither a volume of ours nor a +/// designation stamp: nothing is written to it. +const FOREIGN: &str = "fsd: no volume of ours and no designation stamp"; +/// fsd's word for a volume of ours that mounted. +const MOUNTED: &str = "fsd: mounted the DATA volume"; +/// fsd's word for a volume of ours that did not, followed by the reason. +const UNMOUNTABLE: &str = "fsd: the DATA volume is ours and does not mount ("; +/// fsd's word for DATA's directories served from memory. +const IN_MEMORY: &str = "are in memory and will not survive a reboot"; + +/// Whether fsd said it serves DATA's directories as absent: every name under +/// them refused, and never a volume in memory under the paths an owner's data +/// lives at. +fn data_absent(log: &str) -> Result<(), String> { + if log.lines().any(|l| l.contains("fsd: Data serving") && l.contains(" — absent: ")) { + return Ok(()); + } + Err(format!("fsd never said it serves DATA's directories absent\n{log}")) +} + /// Boot the guest against a disk that belongs to somebody else, and prove it /// comes back untouched. /// @@ -60,11 +80,10 @@ pub fn foreign_disk_untouched( return Err(format!("{bad:?}: refusing a disk must not be fatal\n{log}")); } } - // The refusal is stated, not inferred. A kernel that never reached the - // storage phase would also leave the image untouched. - const REFUSED: &str = "this disk is not ours"; - if !log.contains(REFUSED) { - return Err(format!("the kernel never said {REFUSED:?} — did it reach storage?\n{log}")); + // The refusal is stated, not inferred. A file server that never reached + // the partition would also leave the image untouched. + if !log.contains(FOREIGN) { + return Err(format!("fsd never said {FOREIGN:?} — did it reach the partition?\n{log}")); } // And the machine still came up, because a refusal that costs the boot is // a refusal nobody will leave switched on. @@ -76,12 +95,12 @@ pub fn foreign_disk_untouched( // format that is still sitting in the page cache has already destroyed the // disk as far as the next sync is concerned. if log.contains("formatting it") { - return Err(format!("the kernel decided to format a disk it was not given\n{log}")); + return Err(format!("fsd decided to format a disk it was not given\n{log}")); } - // Shut down rather than kill: `PageCache::sync` at shutdown is the only - // thing that moves a format from the cache to the device, so a killed QEMU - // fingerprints an image a formatting kernel would also have left untouched. + // Shut down rather than kill: the shutdown's sync of every file server is + // what moves a format from fsd's cache to the device, so a killed QEMU + // fingerprints an image a formatting server would also have left untouched. writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); qemu.flush_stdin(); let tail = qemu.drain_serial(Duration::from_secs(20)); @@ -164,28 +183,26 @@ pub fn volume_from_another_disk( return Err(format!("{bad:?}: refusing a copied volume must not be fatal\n{log}")); } } - const MOUNTED: &str = "mounted the ToyOS volume at block 0"; if log.contains(MOUNTED) { return Err(format!( - "the kernel mounted a volume that did not come from this disk: it said \ - {MOUNTED:?}\n{log}" + "fsd mounted a volume that did not come from this disk: it said {MOUNTED:?}\n{log}" )); } // A superblock of ours that does not describe this device is a volume of // ours that did not mount, not another's disk. - const REFUSED: &str = "does not mount, and nothing says it is another's: BadSuperblock"; - if !log.contains(REFUSED) { - return Err(format!("the kernel never said {REFUSED:?} — did it reach storage?\n{log}")); + let refused = format!("{UNMOUNTABLE}BadSuperblock"); + if !log.contains(&refused) { + return Err(format!("fsd never said {refused:?} — did it reach the partition?\n{log}")); } if log.contains("formatting it") { - return Err(format!("the kernel decided to format a disk it was not given\n{log}")); + return Err(format!("fsd decided to format a disk it was not given\n{log}")); } if !log.contains("Boot: complete") { return Err(format!("the boot did not complete on a volume it refused\n{log}")); } - // Down through `PageCache::sync`, the only thing that moves a write out of - // the cache and onto the device. + // Down through the shutdown's sync of every file server, the only thing + // that moves a write out of fsd's cache and onto the device. writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); qemu.flush_stdin(); let tail = qemu.drain_serial(Duration::from_secs(20)); @@ -275,24 +292,15 @@ pub fn broken_data_volume_is_absent( return Err(format!("{bad:?}: a broken volume must not be fatal\n{log}")); } } - for said in [ - "storage: the DATA volume does not mount, and nothing says it is another's: \ - ChecksumMismatch", - "storage: /apps, /config, /home and /state are absent this boot", - "Boot: complete", - ] { + for said in [&format!("{UNMOUNTABLE}ChecksumMismatch"), "Boot: complete"] { if !log.contains(said) { - return Err(format!("the kernel never said {said:?}\n{log}")); + return Err(format!("the boot never said {said:?}\n{log}")); } } - for unsaid in [ - "are a tmpfs", - "mounted the ToyOS volume", - "formatting it", - "this disk is not ours", - ] { + data_absent(&log)?; + for unsaid in [IN_MEMORY, MOUNTED, "formatting it", FOREIGN] { if log.contains(unsaid) { - return Err(format!("the kernel said {unsaid:?} of a volume of ours that broke\n{log}")); + return Err(format!("fsd said {unsaid:?} of a volume of ours that broke\n{log}")); } } @@ -328,8 +336,8 @@ pub fn broken_data_volume_is_absent( Ok(()) } -/// A TOYOS-DATA partition the GPT type names ours, but whose start `page_cache` -/// refuses before it ever opens a view: the owner's ruling is that this is +/// A TOYOS-DATA partition the GPT type names ours, but whose start blockd +/// refuses to serve: the owner's ruling is that this is /// `Absent`, the same as a volume of ours that did not mount, and never /// `Volatile` — a tmpfs is for a machine that carries no data volume at all, /// not for one whose candidate is ours and unreadable by geometry. @@ -363,24 +371,23 @@ pub fn data_candidate_with_bad_geometry_is_absent( return Err(format!("{bad:?}: a misaligned candidate must not be fatal\n{log}")); } } - for said in ["not whole", "storage: /apps, /config, /home and /state are absent this boot", "Boot: complete"] { + for said in ["is not served: LBA", "not whole", "Boot: complete"] { if !log.contains(said) { - return Err(format!("the kernel never said {said:?}\n{log}")); + return Err(format!("the boot never said {said:?}\n{log}")); } } - for unsaid in ["are a tmpfs", "mounted the ToyOS volume", "formatting it"] { + data_absent(&log)?; + for unsaid in [IN_MEMORY, MOUNTED, "formatting it"] { if log.contains(unsaid) { - return Err(format!( - "the kernel said {unsaid:?} of a candidate its own GPT type names ours\n{log}" - )); + return Err(format!("fsd said {unsaid:?} of a partition its own GPT type names ours\n{log}")); } } let result = qemu.run_test("test_rs_home_absent", Duration::from_secs(20)); if result.exit_code != Some(0) { return Err(format!( - "home_absent guest failed on a candidate `over_candidate` refused:\n{}\nkernel log \ - while it ran:\n{}{}", + "home_absent guest failed on a partition blockd refused:\n{}\nkernel log while it \ + ran:\n{}{}", result.stdout, result.before, result.serial )); } @@ -447,9 +454,10 @@ fn front(path: &Path, at: u64, n: usize) -> Vec { } /// The shared-object cache's two refusals, judged in -/// `tests/toyos-rust-tests/src/bin/so_cache_policy.rs`. The independent oracle is -/// the NVMe image: the replaced library's bytes are read off the device after -/// the shutdown, so the claim rests on nothing the guest says. +/// `tests/toyos-rust-tests/src/bin/so_cache_policy.rs`. The cache answers for a +/// path the kernel opens itself, which is `/system` and `/tmp`, so the files +/// are tmpfs's: the guest holds the rewritten bytes against the library it +/// copied, and the kernel's own lines say it refused. pub fn so_cache_refusals( test_config: &Path, c_bins: &[(String, Vec)], @@ -457,16 +465,6 @@ pub fn so_cache_refusals( ) -> Result<(), String> { /// Without it the budget arm would have to load 256 MiB of libraries. const PARAMS: &[&str] = &["so-cache-tiny"]; - /// Mirrored in the guest binary; `/home` is a directory of DATA, so that is - /// the name the host reader sees on the volume. - const STALE: &str = "home/so-cache-stale.so"; - const SECOND: &str = "libtls_dlopen_lib.so"; - - let want = rust_bins - .iter() - .find(|(name, _)| name == SECOND) - .map(|(_, data)| data.clone()) - .ok_or_else(|| format!("{SECOND} was not built, so there is nothing to compare against"))?; let mut qemu = QemuInstance::boot_with_options( test_config, @@ -479,11 +477,6 @@ pub fn so_cache_refusals( }, ); let boot = qemu.boot_log().to_string(); - if boot.contains("are a tmpfs") { - return Err(format!( - "/apps and /home fell back to tmpfs, so the readback below would judge no device:\n{boot}" - )); - } let result = qemu.run_test("test_rs_so_cache_policy", Duration::from_secs(60)); let log = format!("{boot}\n{}{}{}", result.before, result.stdout, result.serial); @@ -500,106 +493,6 @@ pub fn so_cache_refusals( } } - let image = qemu.nvme_image().to_path_buf(); - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - - let io = FileBlocks::open(&image)?; - let fs = bcachefs::Mounted::<_, bcachefs::ReadOnly>::open(io) - .map_err(|e| format!("the NVMe image does not mount on the host: {e:?}"))?; - let got = fs - .read_file(STALE) - .map_err(|e| format!("reading {STALE} off the image: {e:?}"))?; - if got != want { - let at = got.iter().zip(&want).position(|(a, b)| a != b); - return Err(format!( - "{STALE} on the device is {} bytes against {SECOND}'s {}, first differing at {at:?} \ - — the guest's second write did not reach the device, so the refusal above was \ - about a file that had not changed", - got.len(), - want.len() - )); - } - - eprintln!( - " [so-cache] {} bytes of {SECOND} byte-identical at {STALE} off the NVMe image via the \ - host's own bcachefs reader", - want.len() - ); - Ok(()) -} - -/// F9's negative control: an fsync on `/home` whose first attempt is -/// budget-refused (`fsync-budget-spent`) must be retried to durable — on the -/// erasing adapter the `BudgetExpired` came back as `Io` and the guest's -/// `sync_all` failed on attempt 1. The independent oracle is the NVMe image -/// itself: after the shutdown the file's bytes are read off it on the host, -/// through this crate's own build of the `bcachefs` reader over a plain -/// seek-and-read device — nothing the guest kernel executed. -pub fn home_budget_refusal_retried( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - const PARAMS: &[&str] = &["fsync-budget-spent"]; - /// Mirrored in `tests/toyos-rust-tests/src/bin/home_fsync_budget.rs`, under - /// the `/home` directory of DATA the host reader sees. - const PATH: &str = "home/f9-budget.bin"; - const LEN: usize = 3 * 4096 + 41; - fn pattern() -> Vec { - (0..LEN).map(|i| (i.wrapping_mul(151) ^ 0x3C) as u8).collect() - } - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::MetalDisk, - kernel_params: PARAMS, - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - if boot.contains("are a tmpfs") { - return Err(format!( - "/apps and /home fell back to tmpfs, so nothing below touches the NVMe path:\n{boot}" - )); - } - - let result = qemu.run_test("test_rs_home_fsync_budget", Duration::from_secs(30)); - let log = format!("{boot}\n{}{}{}", result.before, result.stdout, result.serial); - if result.exit_code != Some(0) { - return Err(format!( - "home_fsync_budget guest failed — a budget-refused /home fsync was not retried \ - to durable:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - // Both halves of the staging, or the arm proved nothing: the refusal at the - // shipped NVMe site, and the fsync loop's own retry verdict. - if !log.contains("not issued") { - return Err(format!( - "no `not issued` line, so `fsync-budget-spent` staged no NVMe refusal:\n{log}" - )); - } - let retried = log - .lines() - .find(|l| l.contains("fsync: /home/") && l.contains("durable on attempt")) - .ok_or_else(|| { - format!("no `fsync: /home/... durable on attempt` line — the retry never ran:\n{log}") - })? - .trim() - .to_string(); - - let image = qemu.nvme_image().to_path_buf(); writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); qemu.flush_stdin(); let tail = qemu.drain_serial(Duration::from_secs(20)); @@ -609,26 +502,7 @@ pub fn home_budget_refusal_retried( return Err(format!("{bad:?} on the way down\n{tail}")); } } - - let io = FileBlocks::open(&image)?; - let fs = bcachefs::Mounted::<_, bcachefs::ReadOnly>::open(io) - .map_err(|e| format!("the NVMe image does not mount on the host: {e:?}"))?; - let got = fs - .read_file(PATH) - .map_err(|e| format!("reading {PATH} off the image: {e:?}"))?; - if got != pattern() { - let at = got.iter().zip(pattern()).position(|(a, b)| *a != b); - return Err(format!( - "{PATH} on the device is {} bytes, first differing at {at:?} — the retried fsync \ - reported durable over bytes the device does not hold", - got.len() - )); - } - - eprintln!(" [f9] {retried}"); - eprintln!( - " [f9] {LEN} bytes byte-identical off the NVMe image via the host's own bcachefs reader" - ); + eprintln!(" [so-cache] a changed file and a full budget refused, by the kernel's own word"); Ok(()) } @@ -823,9 +697,10 @@ pub fn apps_and_home_are_one_filesystem( Ok(()) } -/// `/boot` and `/log` off the same NVMe device the page cache serves. +/// `/boot` and `/log` off the same NVMe device the machine booted from, both +/// served by fsd through blockd, which refuses ROOT to every session. /// -/// The oracle is outside the guest and outside the kernel's FAT32: `logd`'s +/// The oracle is outside the guest and outside fsd's FAT32: `logd`'s /// file is read off the image by `fatfs` and the volume judged against /// fatgen103 by `toyos-fat32-check`, with the guest already halted. pub fn internal_disk_boot( @@ -848,7 +723,7 @@ pub fn internal_disk_boot( } let controllers: Vec<&String> = argv.iter().filter(|a| a.starts_with("nvme,serial=")).collect(); - if controllers != ["nvme,serial=bootdisk,id=nvmebootctl,bootindex=0"] { + if controllers != ["nvme,serial=bootdisk,id=nvmebootctl,bootindex=0,msix-exclusive-bar=on"] { return Err(format!( "the machine's NVMe controllers are {controllers:?} — this profile's whole shape is \ one controller, carrying the boot image" @@ -875,26 +750,17 @@ pub fn internal_disk_boot( } } - // `1` is NVMe's fixed `DeviceId` and the USB range starts at 16, so naming - // it is also the assertion that no stick served either mount. - for said in [ - "gpt: device 1 carries the boot partition", - "gpt: device 1 carries the log partition", - "boot-volume: partition mounted", - "log-volume: partition mounted", - ] { + // This machine has no USB, so a volume fsd serves came through blockd. + for said in ["fsd: Boot serving /boot — FAT32 read-only", "fsd: Log serving /log — FAT32,"] { if !boot.contains(said) { return Err(format!( - "the kernel never said {said:?} — a machine booting off its internal disk got \ + "the boot never said {said:?} — a machine booting off its internal disk got \ no /boot and no /log\n{boot}" )); } } - if boot.contains("no driver here can open it") { - return Err(format!( - "the kernel found the partition and had no second handle to the device carrying \ - it\n{boot}" - )); + if !boot.contains("this machine runs from; refusing every session to it") { + return Err(format!("blockd never said it refuses ROOT, which the machine runs from\n{boot}")); } // Down, not killed: the file logd wrote reaches the device on the way out. @@ -934,105 +800,13 @@ pub fn internal_disk_boot( let _ = std::fs::remove_file(&image); eprintln!( - " [internal-disk] /boot and /log both off NVMe device 1, and /log/{name} came back \ + " [internal-disk] /boot and /log both off the boot NVMe through blockd, and /log/{name} came back \ {} bytes through fatfs on a volume fatgen103 has nothing to say about", on_device.len() ); Ok(()) } -/// A write through a page cache whose view does not begin at block 0 lands at -/// the device's block, not the view's. -/// -/// The oracle is the NVMe image after the guest has gone: the mark at -/// `(FIRST + AT) * 4096` and absent at `AT * 4096` is the partition offset in -/// `BlockKey` and nothing else. A foreign disk, so the kernel refuses to format -/// it and the probe's one block is the only byte this boot writes to it. -pub fn page_cache_partition_offset( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - /// Mirrored in `kernel/src/page_cache.rs::offset_probe`. - const FIRST: u64 = 3000; - const AT: u64 = 7; - const MARK: &[u8] = b"TOYOS-PARTITION-OFFSET"; - const BYTES: u64 = 128 * 1024 * 1024; - - let dir = super::lane::dir(); - let image = dir.join("partition-offset.img"); - foreign_disk_image(&image, BYTES); - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - kernel_params: &["pc-partition-offset"], - nvme_image: Some(image.clone()), - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - for bad in ["PANIC:", "panicked at"] { - if boot.contains(bad) { - return Err(format!("{bad:?} with the offset probe armed\n{boot}")); - } - } - let verdict = boot - .lines() - .find(|l| l.contains("pc-partition-offset: ")) - .ok_or_else(|| format!("the kernel never ran the offset probe:\n{boot}"))? - .trim() - .to_string(); - for want in [ - format!("landed_at_{}=true", FIRST + AT), - format!("at_block_{AT}=false"), - ] { - if !verdict.contains(&want) { - return Err(format!( - "a cache over a view at +{FIRST} did not write where the view says — {want:?} is \ - missing from: {verdict}" - )); - } - } - - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - drop(qemu); - - let after = std::fs::read(&image).map_err(|e| format!("read the image back: {e}"))?; - let at = |block: u64| { - let start = (block * 4096) as usize; - after[start..start + MARK.len()].to_vec() - }; - if at(FIRST + AT) != MARK { - return Err(format!( - "device block {} holds {:?} on the image, not the mark — the offset never reached \ - the write", - FIRST + AT, - String::from_utf8_lossy(&at(FIRST + AT)) - )); - } - if at(AT) == MARK { - return Err(format!( - "the mark is at device block {AT} on the image — the view's own block number went to \ - the device unchanged" - )); - } - - let _ = std::fs::remove_file(&image); - eprintln!(" [offset] {verdict}, and the image agrees off the device"); - Ok(()) -} - /// The impostor the actuator offers fills every read with its own mark, so a /// registry that took it is caught serving that mark for a device it is not. pub fn block_duplicate_id( diff --git a/tests/common/swap.rs b/tests/common/swap.rs index 8ca0a32fab2..829aeb172e4 100644 --- a/tests/common/swap.rs +++ b/tests/common/swap.rs @@ -485,7 +485,8 @@ pub fn swap_quiets_the_function( }; let text = rig.console.clone(); let console = serial::Serial::named("the quieting boot", text.as_str()); - let released = console.must_say("released from slot 0; reset by")?.to_string(); + let slot = super::iommu::slot_of(&text, "[8086:10d3]")?; + let released = console.must_say(&format!("released from slot {slot}; reset by"))?.to_string(); let inherited = console.must_say("swap_claim_idle: inherited")?.to_string(); if let Err(why) = console.must_be_clean() { return Err(rig.fail(format!("{why}\n the release said: {}", released.trim_end()))); @@ -530,7 +531,8 @@ pub fn swap_keeps_what_nothing_reset( let text = rig.console.clone(); let judged = (|| { let console = serial::Serial::named("the residue boot", text.as_str()); - let released = console.must_say("[8086:10d3] released from slot 0; reset by")?; + let slot = super::iommu::slot_of(&text, "[8086:10d3]")?; + let released = console.must_say(&format!("[8086:10d3] released from slot {slot}; reset by"))?; if !released.contains("reset by nothing") { return Err(format!("the premise: the 82574 was not released by nothing — {released}")); } @@ -545,7 +547,7 @@ pub fn swap_keeps_what_nothing_reset( return Err(format!("the premise: the part's receive unit was off when it was claimed — {inherited}")); } console.must_be_clean()?; - let taken_over = console.must_say("pcidev: slot 0 holds 1 range(s)")?; + let taken_over = console.must_say(&format!("pcidev: slot {slot} holds 1 range(s)"))?; eprintln!( " [swap] {}; {}; {}; mastered through {taken} SYNs in {took:?}, and the unit saw no fault", released.trim_end(), @@ -608,7 +610,8 @@ pub fn swap_fault_tells_its_holder( let judged = (|| { let console = serial::Serial::named("the astray boot", text.as_str()); let fault = console.must_say(HOLDER_FAULT)?; - if !fault.contains("owner=slot0") || !fault.contains("access=read") { + let slot = super::iommu::slot_of(&text, "[8086:10d3]")?; + if !fault.contains(&format!("owner=slot{slot} ")) || !fault.contains("access=read") { return Err(format!("the premise: the unit refused no descriptor fetch of the claim's part — {fault}")); } let told = console.must_say(ASTRAY_TOLD)?; diff --git a/tests/desktopaudiocase/system.toml b/tests/desktopaudiocase/system.toml index 82d72037c07..162b830f1c7 100644 --- a/tests/desktopaudiocase/system.toml +++ b/tests/desktopaudiocase/system.toml @@ -9,7 +9,7 @@ assets = ["assets"] [boot] -start = ["logd", "compositor", "soundd", "terminal"] +start = ["logd", "blockd", "fsd", "compositor", "soundd", "terminal"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -47,3 +47,20 @@ receives = ["compositor", "soundd", "surface"] [symlinks] "bin/echo" = "/system/bin/toybox" "bin/tone" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/desktopcase/system.toml b/tests/desktopcase/system.toml index 07326adc9c2..b75d6c5aef5 100644 --- a/tests/desktopcase/system.toml +++ b/tests/desktopcase/system.toml @@ -15,7 +15,7 @@ assets = ["assets"] # here first: the port exists before either process runs. No soundd or filepicker # in this config, so nothing receives them. [boot] -start = ["logd", "compositor", "terminal"] +start = ["logd", "blockd", "fsd", "compositor", "terminal"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -59,3 +59,20 @@ receives = ["compositor"] [symlinks] "bin/echo" = "/system/bin/toybox" "bin/locale" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/doomcase/system.toml b/tests/doomcase/system.toml index f9b7b2411c6..ece20a78966 100644 --- a/tests/doomcase/system.toml +++ b/tests/doomcase/system.toml @@ -10,7 +10,7 @@ # or the soundfont, so neither belongs on this ROOT either. [boot] -start = ["logd", "soundd", "test-runner"] +start = ["logd", "blockd", "fsd", "soundd", "test-runner"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -41,3 +41,20 @@ syscap = ["logread"] [programs.doom] receives = ["soundd"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/doommusiccase/system.toml b/tests/doommusiccase/system.toml index 251c91f88d5..02bbf138fd3 100644 --- a/tests/doommusiccase/system.toml +++ b/tests/doommusiccase/system.toml @@ -11,7 +11,7 @@ assets = ["assets"] [boot] -start = ["logd", "soundd", "test-runner"] +start = ["logd", "blockd", "fsd", "soundd", "test-runner"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -41,3 +41,20 @@ syscap = ["logread"] [programs.doom] receives = ["soundd"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/e1000case/system.toml b/tests/e1000case/system.toml index 1742df853ff..855ff03d201 100644 --- a/tests/e1000case/system.toml +++ b/tests/e1000case/system.toml @@ -9,7 +9,7 @@ # have costs an `init:` refusal line on every boot of the config that does. [boot] -start = ["logd", "netd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "test-runner"] [programs.logd] service = true @@ -30,3 +30,20 @@ devices = ["pci:8086:10d3"] [programs.test-runner] receives = ["netd"] syscap = ["logread"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/e1000leasecase/system.toml b/tests/e1000leasecase/system.toml index 51f53e79b2c..85bbdd86bbf 100644 --- a/tests/e1000leasecase/system.toml +++ b/tests/e1000leasecase/system.toml @@ -4,7 +4,7 @@ # netd a client would connect to ends inside the boot. [boot] -start = ["logd", "netd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "test-runner"] [programs.logd] service = true @@ -18,3 +18,20 @@ args = ["--exit-with-lease"] [programs.test-runner] syscap = ["logread"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/e1000talkcase/system.toml b/tests/e1000talkcase/system.toml index 6fa1b3ae18b..a7bf773b28a 100644 --- a/tests/e1000talkcase/system.toml +++ b/tests/e1000talkcase/system.toml @@ -4,7 +4,7 @@ # and its listener on the host's loopback; everything else is that boot's. [boot] -start = ["logd", "netd", "sshd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "sshd", "test-runner"] [programs.logd] service = true @@ -38,3 +38,20 @@ receives = ["power"] [symlinks] "bin/echo" = "/system/bin/toybox" "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/flrswapcase/system.toml b/tests/flrswapcase/system.toml index f38d0488f2a..2999f85ba80 100644 --- a/tests/flrswapcase/system.toml +++ b/tests/flrswapcase/system.toml @@ -4,7 +4,7 @@ # the swap's ssh; the igb is only held. [boot] -start = ["logd", "netd", "sshd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "sshd", "test-runner"] [programs.logd] service = true @@ -36,3 +36,20 @@ receives = ["power"] [symlinks] "bin/echo" = "/system/bin/toybox" "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/inspectcase/system.toml b/tests/inspectcase/system.toml index 1c12310260d..216aa8c64bd 100644 --- a/tests/inspectcase/system.toml +++ b/tests/inspectcase/system.toml @@ -11,7 +11,7 @@ assets = ["assets"] [boot] -start = ["logd", "compositor", "soundd", "netd", "test-runner"] +start = ["logd", "blockd", "fsd", "compositor", "soundd", "netd", "test-runner"] # `every_boot_config_runs_logd` refuses a config without it. `log` is the port # it answers `inspect` on. @@ -59,3 +59,20 @@ syscap = ["inventory"] [symlinks] "bin/grep" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/jobcase/system.toml b/tests/jobcase/system.toml index 09cf48bd6ad..9ec7966e42a 100644 --- a/tests/jobcase/system.toml +++ b/tests/jobcase/system.toml @@ -2,7 +2,7 @@ # job list is the manifest's own and the last thing it does is hand the machine # back to firmware. This is the shape a T14 boot has. [boot] -start = ["logd", "test-runner"] +start = ["logd", "blockd", "fsd", "test-runner"] [programs.logd] service = true @@ -19,3 +19,20 @@ args = ["reboot"] [symlinks] "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/jobdeadlinecase/system.toml b/tests/jobdeadlinecase/system.toml index 7752a504c22..c2eb878b258 100644 --- a/tests/jobdeadlinecase/system.toml +++ b/tests/jobdeadlinecase/system.toml @@ -4,7 +4,7 @@ # `--bound-ms` is the one argument that is not a job. The T14 leaves the bound at # `toyos_tco::JOB_BOUND_MS`; this shortens it for the suite's ceiling. [boot] -start = ["logd", "test-runner"] +start = ["logd", "blockd", "fsd", "test-runner"] [programs.logd] service = true @@ -25,3 +25,20 @@ args = ["--bound-ms=3000", "spin", "echo"] "bin/reboot" = "/system/bin/toybox" "bin/spin" = "/system/bin/toybox" "bin/echo" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/lancase/system.toml b/tests/lancase/system.toml index 380c3e31f92..b469a5032a4 100644 --- a/tests/lancase/system.toml +++ b/tests/lancase/system.toml @@ -8,7 +8,7 @@ # only device any program on this boot claims is the PCI function named below. [boot] -start = ["logd", "netd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "test-runner"] [programs.logd] service = true @@ -21,3 +21,20 @@ devices = ["pci:8086:15fc"] [programs.test-runner] syscap = ["logread"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/lanicscase/system.toml b/tests/lanicscase/system.toml index 90ca68d2e1a..6e17dd2b1a6 100644 --- a/tests/lanicscase/system.toml +++ b/tests/lanicscase/system.toml @@ -9,7 +9,7 @@ # only device any program on this boot claims is the PCI function named below. [boot] -start = ["logd", "netd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "test-runner"] [programs.logd] service = true @@ -23,3 +23,20 @@ args = ["--provoke-message"] [programs.test-runner] syscap = ["logread"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/lanleasecase/system.toml b/tests/lanleasecase/system.toml index f35c29f5979..436ae50ffa3 100644 --- a/tests/lanleasecase/system.toml +++ b/tests/lanleasecase/system.toml @@ -7,7 +7,7 @@ # the PCI function named below. [boot] -start = ["logd", "netd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "test-runner"] [programs.logd] service = true @@ -21,3 +21,20 @@ args = ["--exit-with-lease"] [programs.test-runner] syscap = ["logread"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/lantalkcase/system.toml b/tests/lantalkcase/system.toml index a20ebb36bcc..ef574afd005 100644 --- a/tests/lantalkcase/system.toml +++ b/tests/lantalkcase/system.toml @@ -9,7 +9,7 @@ # row, and the only device any program claims is the PCI function below. [boot] -start = ["logd", "netd", "sshd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "sshd", "test-runner"] [programs.logd] service = true @@ -43,3 +43,20 @@ receives = ["power"] [symlinks] "bin/echo" = "/system/bin/toybox" "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/latencycase/system.toml b/tests/latencycase/system.toml index 4fc002731c9..61c47e53d60 100644 --- a/tests/latencycase/system.toml +++ b/tests/latencycase/system.toml @@ -11,7 +11,7 @@ # `power` because the job list is what ends the boot: there is no host on the # other end of the console on the machine this config exists for. [boot] -start = ["logd", "test-runner"] +start = ["logd", "blockd", "fsd", "test-runner"] [programs.logd] service = true @@ -26,3 +26,20 @@ syscap = ["rt", "dup"] [symlinks] "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/layoutcase/system.toml b/tests/layoutcase/system.toml index 01dcc9f045b..e74af13c381 100644 --- a/tests/layoutcase/system.toml +++ b/tests/layoutcase/system.toml @@ -7,7 +7,7 @@ # `/state/sshd`. [boot] -start = ["logd", "netd", "sshd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "sshd", "test-runner"] [programs.logd] service = true @@ -36,3 +36,20 @@ syscap = ["logread"] [symlinks] "bin/echo" = "/system/bin/toybox" "bin/locale" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/logflushcase/system.toml b/tests/logflushcase/system.toml index 3b78ebf01b8..ebbe0664aa7 100644 --- a/tests/logflushcase/system.toml +++ b/tests/logflushcase/system.toml @@ -4,7 +4,7 @@ # and init's resume reaches `logd` with the flush unrun. The last job stops the # machine for real. `log_resume_meets_its_flush` reads the verdict off `/log`. [boot] -start = ["logd", "test-runner"] +start = ["logd", "blockd", "fsd", "test-runner"] [programs.logd] service = true @@ -20,3 +20,20 @@ args = ["test_rs_log_refused_stop", "reboot"] [symlinks] "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/logkeepcase/system.toml b/tests/logkeepcase/system.toml index 999800deb74..e221fc28dc7 100644 --- a/tests/logkeepcase/system.toml +++ b/tests/logkeepcase/system.toml @@ -4,7 +4,7 @@ # once init says the machine stops. `log_ring_keeps_the_owners_slots` reads # the verdict off `/log`. [boot] -start = ["logd", "test-runner"] +start = ["logd", "blockd", "fsd", "test-runner"] [programs.logd] service = true @@ -23,3 +23,20 @@ args = ["test_rs_log_flood", "reboot"] [symlinks] "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/logrotatecase/system.toml b/tests/logrotatecase/system.toml index 9938b47c3d7..8e7be80da73 100644 --- a/tests/logrotatecase/system.toml +++ b/tests/logrotatecase/system.toml @@ -23,7 +23,7 @@ assets = ["assets"] [boot] -start = ["logd", "compositor", "soundd", "netd", "sshd", "test-runner"] +start = ["logd", "blockd", "fsd", "compositor", "soundd", "netd", "sshd", "test-runner"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -79,3 +79,20 @@ receives = ["soundd"] [symlinks] "bin/shutdown" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/logstallcase/system.toml b/tests/logstallcase/system.toml index 7ad28116336..47ef560880e 100644 --- a/tests/logstallcase/system.toml +++ b/tests/logstallcase/system.toml @@ -5,7 +5,7 @@ # `run shutdown`. [boot] -start = ["logd", "soundd", "test-runner"] +start = ["logd", "blockd", "fsd", "soundd", "test-runner"] [programs.logd] service = true @@ -30,3 +30,20 @@ receives = ["soundd"] [symlinks] "bin/shutdown" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/logstreamcase/system.toml b/tests/logstreamcase/system.toml index a9482fe7f8d..e0ad0c73ecc 100644 --- a/tests/logstreamcase/system.toml +++ b/tests/logstreamcase/system.toml @@ -4,7 +4,7 @@ # volume the guest closed itself. [boot] -start = ["logd", "netd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "test-runner"] # `receives = ["netd"]` is what lets `logd` serve this boot's log on the # network. @@ -26,3 +26,20 @@ receives = ["netd", "power"] [symlinks] "bin/shutdown" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/logstreame1000case/system.toml b/tests/logstreame1000case/system.toml index 8c9821ad5cc..03a622eaa0b 100644 --- a/tests/logstreame1000case/system.toml +++ b/tests/logstreame1000case/system.toml @@ -4,7 +4,7 @@ # `run shutdown`, so every verdict is read off a volume the guest closed itself. [boot] -start = ["logd", "netd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "test-runner"] # `receives = ["netd"]` is what lets `logd` serve this boot's log on the # network. @@ -26,3 +26,20 @@ receives = ["netd", "power"] [symlinks] "bin/shutdown" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/metalcase/system.toml b/tests/metalcase/system.toml index eeca4fe25ed..bd8e132b6d9 100644 --- a/tests/metalcase/system.toml +++ b/tests/metalcase/system.toml @@ -17,7 +17,7 @@ assets = ["assets"] [boot] -start = ["logd", "compositor", "soundd", "netd", "sshd", "test-runner"] +start = ["logd", "blockd", "fsd", "compositor", "soundd", "netd", "sshd", "test-runner"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -60,3 +60,20 @@ receives = ["netd", "launcher"] [programs.test-runner] receives = ["compositor", "soundd", "netd"] syscap = ["logread"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/metaldevicecase/system.toml b/tests/metaldevicecase/system.toml index a0d2b297566..5a38787e9b2 100644 --- a/tests/metaldevicecase/system.toml +++ b/tests/metaldevicecase/system.toml @@ -12,7 +12,7 @@ # and output-path choice are the only exercise of `toyos-hda` that hardware can # give. [boot] -start = ["logd", "soundd", "test-runner"] +start = ["logd", "blockd", "fsd", "soundd", "test-runner"] [programs.logd] service = true @@ -52,3 +52,20 @@ syscap = ["device", "dup"] "bin/usbwrite" = "/system/bin/metalprobe" "bin/usbread" = "/system/bin/metalprobe" "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/netcase/system.toml b/tests/netcase/system.toml index 1ae7dc9ddde..fde773e5b04 100644 --- a/tests/netcase/system.toml +++ b/tests/netcase/system.toml @@ -8,7 +8,7 @@ # ===TEST_START=== protocol. [boot] -start = ["logd", "netd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "test-runner"] # `every_boot_config_runs_logd` is what refuses a boot config without it: the # kernel keeps the record ring and writes no file, so such a boot's `/log` is @@ -60,3 +60,20 @@ receives = ["launcher"] # What `dns_resolve` runs: a name's addresses, asked of netd's resolver. [programs.host] receives = ["netd"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/partclaimcase/system.toml b/tests/partclaimcase/system.toml index 329040673ef..e8a2b4e2dbe 100644 --- a/tests/partclaimcase/system.toml +++ b/tests/partclaimcase/system.toml @@ -7,7 +7,7 @@ [boot] -start = ["logd", "soundd", "test-runner"] +start = ["logd", "blockd", "fsd", "soundd", "test-runner"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -69,3 +69,20 @@ receives = ["soundd"] # `toyos-rust-tests` drains the same sink perfectly, so a suite that ran only # that one certified a path no user takes. "bin/tone" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/pkgcase/system.toml b/tests/pkgcase/system.toml index 7f0bc00ef20..1dd833f7432 100644 --- a/tests/pkgcase/system.toml +++ b/tests/pkgcase/system.toml @@ -10,7 +10,7 @@ assets = ["assets"] [boot] -start = ["logd", "compositor", "soundd", "test-runner"] +start = ["logd", "blockd", "fsd", "compositor", "soundd", "test-runner"] # What every program launched out of `/apps` holds. gbae needs both: it opens # its window through the compositor and its cpal stream through soundd. @@ -61,3 +61,20 @@ syscap = ["logread"] [symlinks] "bin/shutdown" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/quiescecase/system.toml b/tests/quiescecase/system.toml index cff3eccfb5f..a3b9cb4ab94 100644 --- a/tests/quiescecase/system.toml +++ b/tests/quiescecase/system.toml @@ -2,7 +2,7 @@ # them. The job asks for the reset itself, from a thread that is not one of the # writers, so no `reboot` job follows the list. [boot] -start = ["logd", "test-runner"] +start = ["logd", "blockd", "fsd", "test-runner"] [programs.logd] service = true @@ -12,3 +12,20 @@ syscap = ["logread"] [programs.test-runner] receives = ["power"] args = ["test_rs_quiesce_writers"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/quiescelastcase/system.toml b/tests/quiescelastcase/system.toml index 7a4510d7c06..31407b48973 100644 --- a/tests/quiescelastcase/system.toml +++ b/tests/quiescelastcase/system.toml @@ -1,7 +1,7 @@ # The boot `quiesce_wakes_on_the_last_park` and `quiesce_wakes_on_the_last_exit` # judge: its one job starts the two threads the kernel may hold and reboots. [boot] -start = ["logd", "test-runner"] +start = ["logd", "blockd", "fsd", "test-runner"] [programs.logd] service = true @@ -11,3 +11,20 @@ syscap = ["logread"] [programs.test-runner] receives = ["power"] args = ["test_rs_quiesce_last"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/quiescetwicecase/system.toml b/tests/quiescetwicecase/system.toml index 078cc109446..0edffbdcf3d 100644 --- a/tests/quiescetwicecase/system.toml +++ b/tests/quiescetwicecase/system.toml @@ -1,6 +1,6 @@ # The boot `quiesce_refuses_a_second_shutdown` judges. [boot] -start = ["logd", "test-runner"] +start = ["logd", "blockd", "fsd", "test-runner"] [programs.logd] service = true @@ -14,3 +14,20 @@ syscap = ["logread"] receives = ["power"] args = ["test_rs_quiesce_twice"] syscap = ["dup", "logread", "power"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/sshdcase/system.toml b/tests/sshdcase/system.toml index 09320cddb9b..f7788a09876 100644 --- a/tests/sshdcase/system.toml +++ b/tests/sshdcase/system.toml @@ -11,7 +11,7 @@ # would spend one of them. [boot] -start = ["logd", "netd", "sshd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "sshd", "test-runner"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -54,3 +54,20 @@ syscap = ["logread"] "bin/echo" = "/system/bin/toybox" "bin/cat" = "/system/bin/toybox" "bin/spin" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/swapcase/system.toml b/tests/swapcase/system.toml index 2e1e2f06747..bc8283b2a0e 100644 --- a/tests/swapcase/system.toml +++ b/tests/swapcase/system.toml @@ -8,7 +8,7 @@ # file, and `tests/lantalkcase` the T14 itself. [boot] -start = ["logd", "netd", "sshd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "sshd", "test-runner"] [programs.logd] service = true @@ -43,3 +43,20 @@ receives = ["power"] [symlinks] "bin/echo" = "/system/bin/toybox" "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/test-durations b/tests/test-durations index 86acf871b21..54ea13a8f27 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -215,7 +215,6 @@ hda_two_live_refused 4848 heap_ceiling_recovery 10371 hierarchy_paths 87 home_backing_revoked 663 -home_budget_refusal_retried 18661 home_overwrite_reads_back 4835 https_tls13 5166 https_tls13_e1000e 6072 @@ -312,7 +311,6 @@ nvme_image_is_held_by_one_guest 0 nvme_large_device 6052 nvme_wide_sector 3500 operation_nesting 5075 -page_cache_partition_offset 6984 panic_before_peripherals_reboots 3227 panic_key_holds 42033 panic_reboots 9768 diff --git a/tests/testcases/system.toml b/tests/testcases/system.toml index 4936ac577c0..22eb9cd47fb 100644 --- a/tests/testcases/system.toml +++ b/tests/testcases/system.toml @@ -1,6 +1,6 @@ [boot] -start = ["logd", "soundd", "test-runner"] +start = ["logd", "blockd", "fsd", "soundd", "test-runner"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -62,3 +62,20 @@ receives = ["soundd"] # `toyos-rust-tests` drains the same sink perfectly, so a suite that ran only # that one certified a path no user takes. "bin/tone" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/toolkitcase/system.toml b/tests/toolkitcase/system.toml index 79846949354..7f7649f8be1 100644 --- a/tests/toolkitcase/system.toml +++ b/tests/toolkitcase/system.toml @@ -4,7 +4,7 @@ assets = ["assets"] [boot] -start = ["logd", "compositor", "terminal"] +start = ["logd", "blockd", "fsd", "compositor", "terminal"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does — `every_boot_config_runs_logd` is what refuses one without. @@ -31,3 +31,20 @@ receives = ["compositor", "surface"] [symlinks] "bin/echo" = "/system/bin/toybox" "bin/stats" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/tests/toyos-rust-tests/src/bin/abuse_cwd_growth.rs b/tests/toyos-rust-tests/src/bin/abuse_cwd_growth.rs index f71813332d0..5d25ad13776 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_cwd_growth.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_cwd_growth.rs @@ -27,7 +27,7 @@ use std::fs; /// know roughly where the wall is. const MAX_PATH: usize = 4096; -const DIR: &str = "/home/abuse_cwd"; +const DIR: &str = "/tmp/abuse_cwd"; /// One component long enough that a single successful round would blow most of /// the budget, and far longer than `MAX_PATH` on its own. diff --git a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs index 88f16bd3331..d6a289c5f05 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs @@ -16,7 +16,9 @@ use toyos_abi::syscall::{self, SpawnArgs, SyscallError}; /// Where the child starts: `SpawnArgs` names a working directory or the spawn is refused. const CWD: &str = "/"; -const DIR: &str = "/home/abuse_loader"; +/// Kernel-served, so a path spawn reaches the kernel's own open; every refusal +/// is asked again with the same bytes handed over as an image. +const DIR: &str = "/tmp/abuse_loader_exe"; /// The two cases about a table larger than one kernel allocation need a file /// larger than one kernel allocation, because `read_file_range` clamps a @@ -239,6 +241,8 @@ fn spawn_path(path: &str) -> Result { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, + image_ptr: 0, + image_len: 0, }) } .map(|pid| pid.0 as u64) @@ -260,8 +264,35 @@ fn refused(name: &str, outcome: Result) { } } +/// The same bytes handed to the kernel whole, as a spawn from a file server's +/// volume hands them: the image route to the same loader. +fn spawn_image(path: &str, bytes: &[u8]) -> Result { + let argv = format!("{path}\0"); + unsafe { + syscall::spawn(&SpawnArgs { + argv_ptr: argv.as_ptr() as u64, + argv_len: argv.len() as u64, + slot_map_ptr: 0, + slot_map_count: 0, + env_ptr: 0, + env_len: 0, + endow_ptr: 0, + endow_count: 0, + labels_ptr: 0, + labels_len: 0, + cwd_ptr: CWD.as_ptr() as u64, + cwd_len: CWD.len() as u64, + image_ptr: bytes.as_ptr() as u64, + image_len: bytes.len() as u64, + }) + } + .map(|pid| pid.0 as u64) +} + +/// Refused whether the kernel opens the file or is handed its bytes. fn spawn_refused(name: &str, bytes: &[u8]) { refused(name, spawn_result(name, bytes)); + refused(&format!("{name} as an image"), spawn_image(&format!("{DIR}/{name}"), bytes)); } /// Load it and throw it away. These cases are about a *walk* the loader does, @@ -287,6 +318,11 @@ fn dlopen_refused(name: &str, bytes: &[u8]) { } Err(e) => assert!(!format!("{e}").is_empty(), "{name}: dlopen error message"), } + // The same bytes handed over whole, under a name no path holds. + let named = format!("/image/{name}"); + if let Ok(handle) = syscall::dl_open_image(named.as_bytes(), bytes) { + panic!("{name}: dlopen of the image loaded it as handle {handle}, and the loader must refuse it"); + } } /// A minimal, honest exe: one PT_LOAD covering the whole file at vaddr 0. @@ -295,7 +331,7 @@ fn base_exe(size: usize) -> Elf { } fn main() { - fs::create_dir_all(DIR).expect("create /home/abuse_loader"); + fs::create_dir_all(DIR).expect("create /tmp/abuse_loader_exe"); fs::create_dir_all(BIG_DIR).expect("create /tmp/abuse_loader"); // 1. A DT_* vaddr below every PT_LOAD. `vaddr_to_file_offset` searched for diff --git a/tests/toyos-rust-tests/src/bin/abuse_elf_segments.rs b/tests/toyos-rust-tests/src/bin/abuse_elf_segments.rs index 0ee1ebd41b2..717023563f8 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_elf_segments.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_elf_segments.rs @@ -20,7 +20,9 @@ use toyos_abi::syscall::{self, SpawnArgs, SyscallError}; /// Where the child starts: `SpawnArgs` names a working directory or the spawn is refused. const CWD: &str = "/"; -const DIR: &str = "/home/abuse_elf"; +/// Kernel-served, so a path spawn reaches the kernel's own open; every refusal +/// is asked again with the same bytes handed over as an image. +const DIR: &str = "/tmp/abuse_elf"; const ET_DYN: u16 = 3; const EM_X86_64: u16 = 62; @@ -121,11 +123,19 @@ impl Phdr { } } -/// Write `bytes` to `path` and try to spawn it. Returns the spawn error. +/// Write `bytes` to `path` and try to spawn it, then spawn the same bytes as an +/// image. Returns the spawn error, which the two routes must agree on. fn spawn_err(name: &str, bytes: &[u8]) -> SyscallError { let path = format!("{DIR}/{name}"); fs::write(&path, bytes).unwrap_or_else(|e| panic!("write {path}: {e}")); + let by_path = spawn_as(&path, &[]); + let by_image = spawn_as(&path, bytes); + assert_eq!(by_path, by_image, "{name}: the path and the image were refused differently"); + by_path +} +/// A raw spawn of `path`, handing `image` over whole when it is not empty. +fn spawn_as(path: &str, image: &[u8]) -> SyscallError { let argv = format!("{path}\0"); unsafe { syscall::spawn(&SpawnArgs { @@ -141,9 +151,11 @@ fn spawn_err(name: &str, bytes: &[u8]) -> SyscallError { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, + image_ptr: image.as_ptr() as u64, + image_len: image.len() as u64, }) } - .map(|pid| panic!("{name}: spawn succeeded (pid {pid:?}) — the header is malformed")) + .map(|pid| panic!("{path}: spawn succeeded (pid {pid:?}) — the header is malformed")) .unwrap_err() } @@ -158,7 +170,7 @@ fn dlopen_err(name: &str, bytes: &[u8]) -> String { } fn main() { - fs::create_dir_all(DIR).expect("create /home/abuse_elf"); + fs::create_dir_all(DIR).expect("create /tmp/abuse_elf"); // 1. PT_TLS filesz > memsz: 16 KiB copied into a 16-byte kernel heap // allocation. The overflow that made this test necessary. diff --git a/tests/toyos-rust-tests/src/bin/abuse_handle_table.rs b/tests/toyos-rust-tests/src/bin/abuse_handle_table.rs index fd914231430..adb032fbd08 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_handle_table.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_handle_table.rs @@ -71,6 +71,8 @@ fn main() { labels_len: LABELS.len() as u64, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, + image_ptr: 0, + image_len: 0, }; unsafe { syscall::spawn(&args) } }; diff --git a/tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs b/tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs index 1789b86dbe2..dfec2f24623 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs @@ -128,6 +128,8 @@ fn main() { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, + image_ptr: 0, + image_len: 0, }; let placed = (boundary - 8) as *mut SpawnArgs; unsafe { placed.write_volatile(args) }; diff --git a/tests/toyos-rust-tests/src/bin/abuse_spawn_argv.rs b/tests/toyos-rust-tests/src/bin/abuse_spawn_argv.rs index 592bdcf7048..4fcceaf1335 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_spawn_argv.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_spawn_argv.rs @@ -48,6 +48,8 @@ fn main() { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, + image_ptr: 0, + image_len: 0, }; let err = unsafe { diff --git a/tests/toyos-rust-tests/src/bin/device_claim_lifetime.rs b/tests/toyos-rust-tests/src/bin/device_claim_lifetime.rs index 12f3e12d5db..c870d926049 100644 --- a/tests/toyos-rust-tests/src/bin/device_claim_lifetime.rs +++ b/tests/toyos-rust-tests/src/bin/device_claim_lifetime.rs @@ -215,6 +215,8 @@ fn spawn_with_slot_map(handle: toyos_abi::RawHandle) -> Result Result { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, + image_ptr: 0, + image_len: 0, }) } } diff --git a/tests/toyos-rust-tests/src/bin/handle_lifetime.rs b/tests/toyos-rust-tests/src/bin/handle_lifetime.rs index 2ccddcbe2c0..58cb5135adc 100644 --- a/tests/toyos-rust-tests/src/bin/handle_lifetime.rs +++ b/tests/toyos-rust-tests/src/bin/handle_lifetime.rs @@ -36,7 +36,7 @@ const SELF_PATH: &str = "/system/bin/test_rs_handle_lifetime"; /// this process and its children, which is the whole of what a namespace is. const SERVICE: &str = "handle-lifetime-service"; const PATH: &[u8] = b"/tmp/handle-lifetime.txt"; -const KILLED_PATH: &[u8] = b"/home/handle-lifetime-killed.txt"; +const KILLED_PATH: &[u8] = b"/tmp/handle-lifetime-killed.txt"; const PAYLOAD: &[u8] = b"a file outlives the handle that was closed first"; const KILLED_PAYLOAD: &[u8] = b"written by a process that was killed before it could close"; /// `process::HANDLE_FAULT_EXIT_CODE`. diff --git a/tests/toyos-rust-tests/src/bin/handle_transfer.rs b/tests/toyos-rust-tests/src/bin/handle_transfer.rs index ade1cb5b251..dd8a84d286f 100644 --- a/tests/toyos-rust-tests/src/bin/handle_transfer.rs +++ b/tests/toyos-rust-tests/src/bin/handle_transfer.rs @@ -47,9 +47,9 @@ const SELF_PATH: &str = "/system/bin/test_rs_handle_transfer"; /// The name the child's namespace carries the connector under. const SERVICE: &str = "transfer"; -/// A file on the one mount in this config that reaches a real block device, so -/// the flush arm exercises the deep path rather than tmpfs. -const DIRTY_PATH: &str = "/home/handle_transfer_flush.bin"; +/// A kernel file: tmpfs is the one writable mount the kernel still has, and its +/// last handle's release runs the same write-back a device's did. +const DIRTY_PATH: &str = "/tmp/handle_transfer_flush.bin"; const DIRTY_BYTES: &[u8] = b"a file whose last handle went while it was queued"; fn main() { diff --git a/tests/toyos-rust-tests/src/bin/home_fsync_budget.rs b/tests/toyos-rust-tests/src/bin/home_fsync_budget.rs deleted file mode 100644 index 36cdf5854b7..00000000000 --- a/tests/toyos-rust-tests/src/bin/home_fsync_budget.rs +++ /dev/null @@ -1,30 +0,0 @@ -//! An fsync on `/home` whose first attempt is budget-refused (`fsync-budget-spent`) -//! must retry on a fresh budget and succeed — a `BudgetExpired` reaching the -//! bcachefs adapter as `Io` ends the syscall on attempt 1 instead (F9). -//! `tests/common/storage.rs::home_budget_refusal_retried` boots and judges this. - -use std::fs::File; -use std::io::{Read, Write}; - -/// Mirrored in `tests/common/storage.rs::home_budget_refusal_retried`. -const PATH: &str = "/home/f9-budget.bin"; -const LEN: usize = 3 * 4096 + 41; - -fn pattern() -> Vec { - (0..LEN).map(|i| (i.wrapping_mul(151) ^ 0x3C) as u8).collect() -} - -fn main() { - let want = pattern(); - let mut f = File::create(PATH).expect("create on /home"); - f.write_all(&want).expect("write"); - f.sync_all().expect( - "fsync refused: a budget-expired first attempt must be retried on a fresh budget, \ - never returned as the device's word", - ); - - let mut got = Vec::new(); - File::open(PATH).expect("re-open").read_to_end(&mut got).expect("read back"); - assert_eq!(got, want, "the bytes changed across the refused-then-retried fsync"); - println!("budget-refused fsync retried to durable: {LEN} bytes on /home"); -} diff --git a/tests/toyos-rust-tests/src/bin/pkg_launch_gbae.rs b/tests/toyos-rust-tests/src/bin/pkg_launch_gbae.rs index 4e447f2820c..ac154f706a9 100644 --- a/tests/toyos-rust-tests/src/bin/pkg_launch_gbae.rs +++ b/tests/toyos-rust-tests/src/bin/pkg_launch_gbae.rs @@ -53,8 +53,7 @@ const SPELLINGS: [&str; 5] = [ /// nothing gave it. Exit 0 is every spelling refused. fn symlink_row() -> std::process::ExitCode { std::fs::create_dir_all(PLANTED_DIR).expect("/apps is writable"); - toyos_abi::syscall::symlink(DECLARED.as_bytes(), PLANTED.as_bytes()) - .expect("a symlink under /apps is allowed"); + std::os::toyos::fs::symlink(DECLARED, PLANTED).expect("a symlink under /apps is allowed"); for spelling in SPELLINGS { let target = std::fs::read_link(spelling).expect("every spelling reaches the link"); assert_eq!(target.to_str(), Some(DECLARED), "{spelling} does not reach the planted link"); @@ -90,8 +89,7 @@ fn relative_path() -> std::process::ExitCode { std::fs::create_dir_all(DIR).expect("/home is writable"); let link = format!("{DIR}/echo"); let _ = std::fs::remove_file(&link); - toyos_abi::syscall::symlink(DECLARED.as_bytes(), link.as_bytes()) - .expect("a symlink under /home is allowed"); + std::os::toyos::fs::symlink(DECLARED, &link).expect("a symlink under /home is allowed"); let mut ran = 0; for typed in ["./home/toy/reltest/echo", "../home/toy/reltest/echo"] { diff --git a/tests/toyos-rust-tests/src/bin/so_cache_policy.rs b/tests/toyos-rust-tests/src/bin/so_cache_policy.rs index a922011ccff..34979ddd822 100644 --- a/tests/toyos-rust-tests/src/bin/so_cache_policy.rs +++ b/tests/toyos-rust-tests/src/bin/so_cache_policy.rs @@ -10,9 +10,10 @@ use std::process::{Command, Stdio}; use toyos_abi::syscall::{self, SyscallError}; -/// Mirrored in `so_cache_refusals`, which reads its bytes off the device. -const STALE: &str = "/home/so-cache-stale.so"; -const SAME_SIZE: &str = "/home/so-cache-same-size.so"; +/// On tmpfs: the cache answers for a path the kernel opens itself, which +/// DATA's are not. +const STALE: &str = "/tmp/so-cache-stale.so"; +const SAME_SIZE: &str = "/tmp/so-cache-same-size.so"; const FIRST: &str = "/system/lib/libtls_lib.so"; const SECOND: &str = "/system/lib/libtls_dlopen_lib.so"; /// A symbol `FIRST` exports and `SECOND` does not: the verdict is a name. @@ -67,6 +68,10 @@ fn a_changed_file_is_refused() -> Result { } copy(SECOND, STALE); + // The file really changed, or the refusal below would be about one that had not. + if std::fs::read(STALE).ok() != std::fs::read(SECOND).ok() { + return Err(format!("{STALE} does not hold {SECOND}'s bytes after the copy")); + } let second = load_in_child(STALE); if second.contains("SYMBOL-FOUND") { return Err(format!( @@ -108,7 +113,7 @@ fn a_same_size_rewrite_is_refused() -> Result { /// budget as surely as distinct libraries would. fn the_budget_is_refused() -> Result { for attempt in 0..BUDGET_ATTEMPTS { - let path = format!("/home/so-cache-fill-{attempt}.so"); + let path = format!("/tmp/so-cache-fill-{attempt}.so"); copy(FIRST, &path); let said = load_in_child(&path); if said.contains("REFUSED ResourceExhausted") { diff --git a/tests/toyos-rust-tests/src/bin/spawn_cwd.rs b/tests/toyos-rust-tests/src/bin/spawn_cwd.rs index daf96d422bd..38e09279857 100644 --- a/tests/toyos-rust-tests/src/bin/spawn_cwd.rs +++ b/tests/toyos-rust-tests/src/bin/spawn_cwd.rs @@ -2,7 +2,8 @@ //! //! `SpawnArgs` carries the child's working directory, and the kernel starts the //! child there or refuses the spawn by name — it never substitutes the -//! caller's. std's direct spawn states `Command::current_dir` when it is set +//! caller's. A directory on a file server the kernel cannot judge, so std judges +//! it before either road asks. std's direct spawn states `Command::current_dir` when it is set //! and this process's own directory when it is not, and init's launcher states //! the one its client sent. //! @@ -139,6 +140,8 @@ fn spawn_in(cwd: &str) -> Result { labels_len: 0, cwd_ptr: cwd.as_ptr() as u64, cwd_len: cwd.len() as u64, + image_ptr: 0, + image_len: 0, }; // SAFETY: every pointer names a live local for the length beside it. let child = unsafe { syscall::spawn(&args) }?; @@ -162,8 +165,6 @@ fn refusals() { for (cwd, want) in [ (ABSENT, SyscallError::NotFound), (FILE, SyscallError::NotFound), - (LOG_FILE, SyscallError::NotFound), - (HOME_FILE, SyscallError::NotFound), (SELF, SyscallError::NotFound), ("tmp/spawn-cwd/named", SyscallError::InvalidArgument), ("", SyscallError::InvalidArgument), @@ -182,6 +183,23 @@ fn refusals() { } } + // A file server's path the kernel cannot judge, and starts a raw spawn in: + // std judges it before it asks, on either road. + for file in [LOG_FILE, HOME_FILE] { + match Command::new(SELF).arg("pwd").current_dir(file).spawn() { + Err(e) if e.kind() == std::io::ErrorKind::NotFound => {} + Err(e) => wrong.push(format!("std's word for {file}: {:?}, not NotFound", e.kind())), + Ok(mut child) => { + let _ = child.wait(); + wrong.push(format!("std spawned into the file {file}")); + } + } + if let Ok(mut child) = Command::new(TOYBOX).arg("pwd").current_dir(file).spawn() { + let _ = child.wait(); + wrong.push(format!("the launcher started a child in the file {file}")); + } + } + match Command::new(SELF).arg("pwd").current_dir(ABSENT).spawn() { Err(e) if e.kind() == std::io::ErrorKind::NotFound => {} Err(e) => wrong.push(format!("std's word for {ABSENT}: {:?}, not NotFound", e.kind())), diff --git a/tests/toyos.rs b/tests/toyos.rs index 1b0c20dcebc..54d2c9fba86 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -478,9 +478,6 @@ const RUST_SKIP: &[&str] = &[ // succeeds and its two must-refuse assertions red for an honest reason. // `fsync_failed_commit` boots it with the arm. "fsync_flush_failed", - // Needs `fsync-budget-spent` and the NVMe `/home`; unstaged it passes - // vacuously. `home_budget_refusal_retried` boots it with both. - "home_fsync_budget", // Needs `so-cache-tiny` and the NVMe `/home`. `so_cache_refusals` gives both. "so_cache_policy", // Needs the NVMe `/home` and a boot of its own for the readback it is judged against; `home_overwrite_reads_back` runs it. @@ -988,7 +985,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("internal_disk_boot", Sched::Parallel, Tier::Fast), // One boot each, kernel lines and image bytes for verdicts, no clock in either. ("block_duplicate_id", Sched::Parallel, Tier::Fast), - ("page_cache_partition_offset", Sched::Parallel, Tier::Fast), // A partition claimed as a device: one boot, every refusal in the guest, // the neighbours and the target judged off the image. Body in // `tests/common/partclaim.rs`, as are the two below. @@ -1001,10 +997,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // for its own writes, across a close, after another's flush, and at the // shutdown when nobody asked. ("partition_claim_departure", Sched::Parallel, Tier::Fast), - // F9's negative control: a budget-refused /home fsync retried to durable, - // its bytes then read off the NVMe image by the host's own bcachefs - // reader. Body in `tests/common/storage.rs`. - ("home_budget_refusal_retried", Sched::Parallel, Tier::Nightly), // The shared-object cache's two refusals. Body in `tests/common/storage.rs`. ("so_cache_refusals", Sched::Parallel, Tier::Fast), // A same-length overwrite on /home, the guest's read held against the image. Body in `tests/common/storage.rs`. @@ -1737,7 +1729,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("layout_fresh_boot", &["test_rs_layout_paths"]), ("broken_data_volume_is_absent", &["test_rs_home_absent"]), ("data_candidate_with_bad_geometry_is_absent", &["test_rs_home_absent"]), - ("home_budget_refusal_retried", &["test_rs_home_fsync_budget"]), ("home_overwrite_reads_back", &["test_rs_home_overwrite_zero"]), ("so_cache_refusals", &["test_rs_so_cache_policy"]), ("boot_volume_metadata_error", &["test_rs_boot_volume_metadata_error"]), @@ -11435,9 +11426,6 @@ fn run_machine_test( partclaim::partition_claim_departure(test_config, c_bins, rust_bins) } "block_duplicate_id" => storage::block_duplicate_id(test_config, c_bins, rust_bins), - "page_cache_partition_offset" => { - storage::page_cache_partition_offset(test_config, c_bins, rust_bins) - } "volume_from_another_disk" => { storage::volume_from_another_disk(test_config, c_bins, rust_bins) } @@ -11447,9 +11435,6 @@ fn run_machine_test( "data_candidate_with_bad_geometry_is_absent" => { storage::data_candidate_with_bad_geometry_is_absent(test_config, c_bins, rust_bins) } - "home_budget_refusal_retried" => { - storage::home_budget_refusal_retried(test_config, c_bins, rust_bins) - } "so_cache_refusals" => storage::so_cache_refusals(test_config, c_bins, rust_bins), "home_overwrite_reads_back" => { storage::home_overwrite_reads_back(test_config, c_bins, rust_bins) diff --git a/tests/updatecase/system.toml b/tests/updatecase/system.toml index ca95cc23d4f..7bf21b1d3b4 100644 --- a/tests/updatecase/system.toml +++ b/tests/updatecase/system.toml @@ -5,7 +5,7 @@ # image and judges every boot. [boot] -start = ["logd", "netd", "sshd", "test-runner"] +start = ["logd", "blockd", "fsd", "netd", "sshd", "test-runner"] [programs.logd] service = true @@ -34,3 +34,20 @@ syscap = ["power"] [symlinks] "bin/reboot" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/toyos-abi/src/inventory.rs b/toyos-abi/src/inventory.rs index 90167d77057..f4be551442d 100644 --- a/toyos-abi/src/inventory.rs +++ b/toyos-abi/src/inventory.rs @@ -205,6 +205,25 @@ pub struct Claim { pub holder: Holder, } +/// Which of the machine's partitions the loader named for a role. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Role { + /// The ROOT this boot runs, read whole by the loader. + Root, + /// The running slot's volume: the kernel and its parameter. + Boot, + /// The log partition. + Log, +} + +/// A partition the loader named, by its unique GUID: what a file server for +/// that role serves, wherever the disk it is on is driven. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub struct Loaded { + pub role: Role, + pub unique_guid: [u8; 16], +} + /// One record. #[derive(Clone, Copy, PartialEq, Eq, Debug)] pub enum Record { @@ -213,6 +232,7 @@ pub enum Record { Block(Block), Partition(Partition), Claim(Claim), + Loaded(Loaded), } /// Why a record did not decode. @@ -241,6 +261,7 @@ const KIND_USB: u8 = 2; const KIND_BLOCK: u8 = 3; const KIND_PARTITION: u8 = 4; const KIND_CLAIM: u8 = 5; +const KIND_LOADED: u8 = 6; /// Little-endian writes at fixed offsets. struct Out([u8; RECORD_BYTES]); @@ -381,6 +402,18 @@ impl Record { } holder_out(&mut o, 28, &c.holder); } + Self::Loaded(l) => { + o.u8(0, KIND_LOADED); + o.u8( + 1, + match l.role { + Role::Root => 0, + Role::Boot => 1, + Role::Log => 2, + }, + ); + o.bytes(8, &l.unique_guid); + } } RawRecord(o.0) } @@ -444,6 +477,15 @@ impl Record { }, holder: holder_in(&r, 28), }), + KIND_LOADED => Self::Loaded(Loaded { + role: r.variant(1, |b| match b { + 0 => Some(Role::Root), + 1 => Some(Role::Boot), + 2 => Some(Role::Log), + _ => None, + })?, + unique_guid: r.array(8), + }), kind => return Err(Undecodable::Kind(kind)), }) } @@ -459,7 +501,7 @@ mod tests { n } - fn every_kind() -> [Record; 7] { + fn every_kind() -> [Record; 8] { let at = PciAddr { segment: 0x1234, bus: 0, dev: 0x1f, func: 6 }; let holder = Holder { pid: 7, name: name("netd") }; [ @@ -496,6 +538,7 @@ mod tests { on: Claimed::Partition { device: 3, unique_guid: [0x11; 16] }, holder, }), + Record::Loaded(Loaded { role: Role::Log, unique_guid: [0x22; 16] }), ] } diff --git a/toyos-abi/src/syscall.rs b/toyos-abi/src/syscall.rs index 03d9f2d2dff..b16c94e7a17 100644 --- a/toyos-abi/src/syscall.rs +++ b/toyos-abi/src/syscall.rs @@ -385,15 +385,32 @@ pub struct SpawnArgs { /// The label blob every [`EndowEntry`]'s `label_off`/`label_len` indexes. pub labels_ptr: u64, pub labels_len: u64, - /// The child's working directory: an absolute path to a directory that - /// exists, or the spawn is refused — `InvalidArgument` for a path that is - /// not absolute, `NotFound` for one that names no directory. **Always the - /// caller's statement**: the kernel never substitutes the caller's own. + /// The child's working directory, absolute, or the spawn is refused + /// `InvalidArgument`. Under a name the kernel serves (`/system`, `/tmp`) it + /// must be a directory there or the spawn is `NotFound`; under any other + /// name it is a file server's directory, which the caller's client asked + /// that server about and the kernel cannot. **Always the caller's + /// statement**: the kernel never substitutes the caller's own. pub cwd_ptr: u64, pub cwd_len: u64, + /// The program's bytes, read whole by the caller, or `image_len` 0 for a + /// program the kernel opens at `argv[0]` itself. An image is copied at the + /// call and the child is paged from the copy; `argv[0]` names it and is + /// opened by nobody, and its libraries are found in `/system/lib` alone. + pub image_ptr: u64, + pub image_len: u64, } -const _: () = assert!(core::mem::size_of::() == 96); +const _: () = assert!(core::mem::size_of::() == 112); + +/// A library's bytes, read whole by the caller, for `SYS_DLOPEN`'s fourth word: +/// what [`SpawnArgs::image_ptr`] is to a spawn. +#[repr(C)] +#[derive(Clone, Copy)] +pub struct ImageRef { + pub ptr: u64, + pub len: u64, +} /// One `(label, handle)` pair of a process's endowment table. /// @@ -1893,8 +1910,21 @@ pub fn readlink(path: &[u8], buf: &mut [u8]) -> Result { /// Load a shared library (.so) into the current process. /// Runs .init_array constructors after loading. pub fn dl_open(path: &[u8]) -> Result { + dl_open_with(path, 0) +} + +/// Load a shared library whose bytes the caller read itself, under `name`: a +/// library on a file server's volume, which the kernel cannot open. The kernel +/// copies `image` at the call; a second load under the same name in this +/// process answers the first's handle, as a path load does. +pub fn dl_open_image(name: &[u8], image: &[u8]) -> Result { + let image = ImageRef { ptr: image.as_ptr() as u64, len: image.len() as u64 }; + dl_open_with(name, &image as *const ImageRef as u64) +} + +fn dl_open_with(path: &[u8], image: u64) -> Result { let mut init_info: [u64; 2] = [0; 2]; - let handle = check(syscall(SYS_DLOPEN, path.as_ptr() as u64, path.len() as u64, init_info.as_mut_ptr() as u64, 0))?; + let handle = check(syscall(SYS_DLOPEN, path.as_ptr() as u64, path.len() as u64, init_info.as_mut_ptr() as u64, image))?; // Run .init_array constructors (e.g. EH frame finder registration in cdylib std) let init_array_ptr = init_info[0]; let init_count = init_info[1]; diff --git a/toyos-blockring/src/wire.rs b/toyos-blockring/src/wire.rs index 274fe1f36e6..b6c7e04204c 100644 --- a/toyos-blockring/src/wire.rs +++ b/toyos-blockring/src/wire.rs @@ -13,6 +13,11 @@ pub const MSG_OPEN: u32 = 1; pub const MSG_OPENED: u32 = 2; /// Refused; the payload is a [`Refusal`]'s word. pub const MSG_REFUSED: u32 = 3; +/// What partitions the service serves, for a client that finds its own by +/// type: no payload, no handles; answered [`MSG_LISTED`]. +pub const MSG_LIST: u32 = 4; +/// The answer to [`MSG_LIST`]: one [`Listed`] after another. +pub const MSG_LISTED: u32 = 5; /// The bytes of a GUID payload. pub const GUID_BYTES: usize = 16; @@ -45,6 +50,39 @@ impl Opened { } } +/// One partition of the table a service drives, as [`MSG_LISTED`] carries +/// it: its unique and type GUIDs as the table stores them. Every entry is +/// listed, one the service will not open among them, so a client that finds +/// its partition by type learns why from the open's refusal rather than +/// taking the partition for missing. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Listed { + pub unique: [u8; GUID_BYTES], + pub kind: [u8; GUID_BYTES], +} + +impl Listed { + pub const BYTES: usize = 2 * GUID_BYTES; + + pub fn encode(&self) -> [u8; Self::BYTES] { + let mut out = [0u8; Self::BYTES]; + out[..GUID_BYTES].copy_from_slice(&self.unique); + out[GUID_BYTES..].copy_from_slice(&self.kind); + out + } + + /// Every entry of a listing, or `None` for one that is not whole entries. + pub fn decode_all(bytes: &[u8]) -> Option + '_> { + if bytes.len() % Self::BYTES != 0 { + return None; + } + Some(bytes.chunks_exact(Self::BYTES).map(|c| Self { + unique: c[..GUID_BYTES].try_into().expect("sixteen bytes"), + kind: c[GUID_BYTES..].try_into().expect("sixteen bytes"), + })) + } +} + /// Why an open was refused. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum Refusal { @@ -104,6 +142,13 @@ mod tests { #[test] fn opened_and_refusal_survive_their_bytes_and_nothing_else_decodes() { + let listed = [ + Listed { unique: [1; GUID_BYTES], kind: [2; GUID_BYTES] }, + Listed { unique: [4; GUID_BYTES], kind: [5; GUID_BYTES] }, + ]; + let bytes: Vec = listed.iter().flat_map(|l| l.encode()).collect(); + assert_eq!(Listed::decode_all(&bytes).map(|all| all.collect::>()), Some(listed.to_vec())); + assert!(Listed::decode_all(&bytes[1..]).is_none()); let opened = Opened { blocks: u64::MAX - 3, unique: [7; GUID_BYTES] }; assert_eq!(Opened::decode(&opened.encode()), Some(opened)); assert_eq!(Opened::decode(&opened.encode()[1..]), None); diff --git a/toyos-inspect/src/dev.rs b/toyos-inspect/src/dev.rs index 6dbb4ba7302..4e5cd3dc72b 100644 --- a/toyos-inspect/src/dev.rs +++ b/toyos-inspect/src/dev.rs @@ -34,7 +34,7 @@ use alloc::format; use alloc::string::{String, ToString}; use toyos_abi::inventory::{ - Claimed, Driven, Holder, PartState, PciAddr, Record, UsbFunction, UsbSpeed, + Claimed, Driven, Holder, PartState, PciAddr, Record, Role, UsbFunction, UsbSpeed, }; use toyos_abi::part::{PartGuid, GUID_TEXT_LEN}; use toyos_abi::syscall::SYSINFO_HEADER_SIZE; @@ -170,6 +170,14 @@ pub fn render(machine: &Machine, records: &[Record]) -> Result {} + Record::Loaded(l) => { + let role = match l.role { + Role::Root => "root", + Role::Boot => "boot", + Role::Log => "log", + }; + out.put(format!("loaded.{role}"), guid(l.unique_guid).into())?; + } } } diff --git a/toyos-manifest/src/lib.rs b/toyos-manifest/src/lib.rs index 3ac58ee5cbb..a3e9480f9a3 100644 --- a/toyos-manifest/src/lib.rs +++ b/toyos-manifest/src/lib.rs @@ -22,6 +22,8 @@ //! syscap a right on the SysCap dup init endows //! slots the idle slot's partitions and the slot table, claimed by init //! service a system service: its `HOME` is `/state/`, not the session's +//! role a file server for ``: one process of it per role +//! restart init starts it again when it ends //! init-serve a name init serves itself //! start init starts this program at boot //! app-receive a connector every program launched from /apps holds @@ -57,6 +59,25 @@ pub fn session_home() -> String { /// program key. pub const STATE: &str = "/state"; +/// The file-server roles, and the directories each serves: one capability per +/// directory, named `fs:` and the directory. +pub const ROLES: [(&str, &[&str]); 3] = [ + ("data", &["/apps", "/config", "/home", "/state"]), + ("log", &["/log"]), + ("boot", &["/boot"]), +]; + +/// The directories `role` serves, or `None` for a name that is no role. +pub fn role_dirs(role: &str) -> Option<&'static [&'static str]> { + ROLES.iter().find(|(name, _)| *name == role).map(|(_, dirs)| *dirs) +} + +/// How often a `restart` row is started again before init gives up on it: at +/// most this many ends inside [`RESTART_WINDOW_SECS`]. Past it the row's ports +/// close, and a client's next connection is answered `Gone`. +pub const RESTARTS: u32 = 3; +pub const RESTART_WINDOW_SECS: u64 = 10; + pub use toyos_abi::handle::Rights; pub use toyos_abi::syscall::{DeviceRequest, DeviceType}; @@ -147,6 +168,13 @@ pub struct Program { /// than the session user's home, so what it keeps is machine state and no /// user's. pub service: bool, + /// The file-server roles this row serves, one process each: init starts + /// the binary once per role, with the role as its argument and the + /// acceptors of the role's directories ([`role_dirs`]). + pub roles: Vec, + /// init starts it again when it ends, on the same ports, for as long as + /// it does not end faster than [`RESTARTS`] allows. + pub restart: bool, } impl Program { @@ -226,6 +254,8 @@ pub enum RenderError { /// init would start it in the session user's home. A service that serves /// nothing (`sshd`) cannot be told from its row, and is marked by hand. ServesWithoutService(String), + /// A `roles` entry naming no file-server role. + NoSuchRole { program: String, role: String }, } pub fn render(manifest: &Manifest) -> Result, RenderError> { @@ -269,6 +299,15 @@ pub fn render(manifest: &Manifest) -> Result, RenderError> { if program.service { out.push_str("service\n"); } + for role in &program.roles { + if role_dirs(role).is_none() { + return Err(RenderError::NoSuchRole { program: program.name.clone(), role: role.clone() }); + } + out.push_str(&format!("role {role}\n")); + } + if program.restart { + out.push_str("restart\n"); + } } for name in &manifest.init_serves { check("init", "init_serves", name)?; @@ -353,6 +392,8 @@ pub fn parse(text: &str) -> Manifest { "syscap" => program.syscap.push(rest.to_string()), "slots" if rest.is_empty() => program.slots = true, "service" => program.service = true, + "role" => program.roles.push(rest.to_string()), + "restart" if rest.is_empty() => program.restart = true, other => panic!("manifest: unknown record `{other}`"), } } @@ -392,6 +433,14 @@ mod tests { slots: true, ..Program::default() }, + Program { + name: "fsd".into(), + path: "/system/bin/fsd".into(), + receives: vec!["block".into()], + roles: vec!["data".into(), "log".into()], + restart: true, + ..Program::default() + }, Program { name: "terminal".into(), path: "/system/bin/terminal".into(), @@ -540,6 +589,26 @@ mod tests { /// A class name reaches init through this file, so a `devices` entry the /// ABI does not know is a config that renders and cannot boot. + /// A file server's roles reach init as records, and a name that is no + /// role is refused where it is written: init would start a server for + /// directories nobody named. + #[test] + fn a_role_is_one_of_the_three_and_its_directories_are_fixed() { + let m = sample(); + let fsd = m.program("fsd").unwrap(); + assert_eq!(fsd.roles, ["data", "log"]); + assert!(fsd.restart); + assert_eq!(role_dirs("data"), Some(&["/apps", "/config", "/home", "/state"][..])); + assert_eq!(role_dirs("boot"), Some(&["/boot"][..])); + assert_eq!(role_dirs("tmp"), None); + let mut bad = sample(); + bad.programs[3].roles = vec!["tmp".into()]; + assert_eq!( + render(&bad), + Err(RenderError::NoSuchRole { program: "fsd".into(), role: "tmp".into() }) + ); + } + #[test] fn a_device_class_name_is_the_abi_s() { assert_eq!(DeviceType::from_class_name("hda-audio"), Some(DeviceType::HdaAudio)); diff --git a/toyos/src/endow.rs b/toyos/src/endow.rs index e404d5b9304..33d0d51f00c 100644 --- a/toyos/src/endow.rs +++ b/toyos/src/endow.rs @@ -143,6 +143,19 @@ pub fn namespace() -> Option<&'static Namespace> { NAMESPACE.get().as_ref() } +/// Make `ns` this process's namespace, for the one process no parent endows +/// one: init, which builds the machine's ports itself and resolves its own +/// files through them as every program does. Refused when this process already +/// has a namespace or has asked for one. +pub fn adopt_namespace(ns: Namespace) -> Result<(), Namespace> { + let mut offered = Some(ns); + NAMESPACE.0.get_or_init(|| offered.take()); + match offered { + None => Ok(()), + Some(ns) => Err(ns), + } +} + /// Open a connection to `name` in this process's namespace. /// /// The one place a name becomes a connection. It works from the caller's first diff --git a/toyos/src/fs.rs b/toyos/src/fs.rs new file mode 100644 index 00000000000..e347cd5a791 --- /dev/null +++ b/toyos/src/fs.rs @@ -0,0 +1,581 @@ +//! Files a file server holds: the wire between a program and the server behind +//! one of its directory capabilities, and the client end of it. +//! +//! **A directory capability is a connector in the program's namespace**, named +//! [`CAPABILITY_PREFIX`] and the absolute directory it serves (`fs:/home`). +//! init builds each program's set and tells the server which directory and +//! which rights each of its ports serves; a program names a file only under a +//! directory it holds, and the kernel's part is who holds which connector. +//! +//! **The server resolves; the client only asks.** A path on the wire is +//! relative to the capability's directory and canonical — no empty, `.` or +//! `..` component — and the server refuses any other ([`canonical`] is the one +//! rule both ends apply). A symlink whose target stays inside the directory is +//! followed by the server; one that is absolute is answered [`LINK`] with the +//! target and the rest of the path in the window, and resolved again in the +//! client's own table, so it reaches nothing the client does not hold. +//! +//! **One connection per directory per process, a request and then its +//! reply.** The client lends the server a [`WINDOW_BYTES`] region at connect +//! ([`HELLO`]); paths, data and listings travel through it and the frames stay +//! small and fixed. The window is shared memory both sides write, so each side +//! copies what it reads out of it once and acts on the copy. +//! +//! **A server that ends is survivable and not invisible.** [`Dir`] connects +//! again through the same connector — init keeps a file server's ports open +//! across its restart — and counts its connections in a generation. A file id +//! from an earlier generation names nothing on the new connection; the holder +//! of a file opens it again by its path, and a file whose path no longer names +//! it answers [`SyscallError::Gone`]. Nothing here keeps a write the server +//! acknowledged and never made durable: that is what `Fsync` is for. + +use toyos_abi::syscall::{SyscallError, MAX_SERVICE_NAME}; + +use crate::ipc::{Connection, IpcError}; +use crate::ipc_payload; +use crate::namespace::Namespace; +use crate::shm::SharedMemory; +use crate::volatile::Window; +use crate::{AsHandle, Pipe, RawHandle}; + +/// What a directory capability's namespace name begins with. +pub const CAPABILITY_PREFIX: &str = "fs:"; + +/// The data window a client lends its server: one shared-memory page, the size +/// shared memory comes in. +pub const WINDOW_BYTES: usize = 2 << 20; + +/// The longest path either end puts on the wire. +pub const MAX_PATH: usize = 4096; + +/// The largest offset a file has, on every server: a page index that fits a +/// `u32`. A seek past it is refused where the offset is kept, and a write or a +/// truncate past it where the file is. +pub const MAX_FILE_BYTES: u64 = (u32::MAX as u64 + 1) * 4096; + +// Requests, by frame type. +pub const HELLO: u32 = 1; +pub const OPEN: u32 = 2; +pub const CLOSE: u32 = 3; +pub const READ: u32 = 4; +pub const WRITE: u32 = 5; +pub const STAT: u32 = 6; +pub const LSTAT: u32 = 7; +pub const FSTAT: u32 = 8; +pub const TRUNCATE: u32 = 9; +pub const FSYNC: u32 = 10; +pub const READDIR: u32 = 11; +pub const MKDIR: u32 = 12; +pub const RMDIR: u32 = 13; +pub const UNLINK: u32 = 14; +pub const RENAME: u32 = 15; +pub const SYMLINK: u32 = 16; +pub const READLINK: u32 = 17; +pub const STREAM: u32 = 18; +pub const SYNC: u32 = 19; + +// Replies, by frame type. +pub const REPLY: u32 = 0x100; +/// The path met an absolute symlink: the window holds the path to resolve +/// instead, `value` bytes long. +pub const LINK: u32 = 0x101; + +// Open flags. +pub const O_READ: u64 = 1; +pub const O_WRITE: u64 = 2; +pub const O_APPEND: u64 = 4; +pub const O_CREATE: u64 = 8; +pub const O_TRUNCATE: u64 = 16; +pub const O_CREATE_NEW: u64 = 32; +pub const O_KNOWN: u64 = O_READ | O_WRITE | O_APPEND | O_CREATE | O_TRUNCATE | O_CREATE_NEW; + +// Node kinds. +pub const KIND_FILE: u64 = 1; +pub const KIND_DIR: u64 = 2; +pub const KIND_SYMLINK: u64 = 3; + +/// [`HELLO`]'s answer: the capability's rights. +pub const RIGHT_WRITE: u64 = 1; + +ipc_payload! { + /// Every request's words; what each means is its frame type's. Paths are + /// in the window: the first `len` bytes, and for [`RENAME`] and + /// [`SYMLINK`] a second `len2` bytes after it. + pub struct Request { + pub fid: u64, + pub offset: u64, + pub len: u64, + pub len2: u64, + pub flags: u64, + } + + /// Every reply's words. `status` is 0 or a [`SyscallError`]'s wire value. + pub struct Reply { + pub status: u64, + pub kind: u64, + pub value: u64, + pub value2: u64, + pub mtime: u64, + } +} + +impl Request { + pub const fn new() -> Self { + Self { fid: 0, offset: 0, len: 0, len2: 0, flags: 0 } + } +} + +impl Reply { + pub const fn ok() -> Self { + Self { status: 0, kind: 0, value: 0, value2: 0, mtime: 0 } + } + + pub const fn refused(e: SyscallError) -> Self { + Self { status: e.to_u64(), kind: 0, value: 0, value2: 0, mtime: 0 } + } + + fn result(self) -> Result { + match self.status { + 0 => Ok(self), + raw => Err(SyscallError::from_u64(raw).unwrap_or(SyscallError::Unknown)), + } + } +} + +/// Whether `path` is a relative path the wire may carry: components of at +/// least one byte, none of them `.` or `..`, no leading or trailing `/`, and +/// the empty path for the capability's own directory. +pub fn canonical(path: &str) -> bool { + path.len() <= MAX_PATH + && (path.is_empty() || path.split('/').all(|c| !c.is_empty() && c != "." && c != "..")) +} + +/// Copy `data` into `window` at `offset`, word-wise where it can. +pub fn window_put(window: Window, offset: usize, data: &[u8]) { + let target = window.sub(offset, data.len()); + let head = (8 - target.as_ptr() as usize % 8) % 8; + let head = head.min(data.len()); + for (i, byte) in data[..head].iter().enumerate() { + target.write::(i, *byte); + } + let words = (data.len() - head) / 8 * 8; + if words > 0 { + target.sub(head, words).copy_in(0, &data[head..head + words]); + } + for (i, byte) in data[head + words..].iter().enumerate() { + target.write::(head + words + i, *byte); + } +} + +/// Copy `out.len()` bytes out of `window` at `offset`, word-wise where it can. +pub fn window_take(window: Window, offset: usize, out: &mut [u8]) { + let source = window.sub(offset, out.len()); + let head = (8 - source.as_ptr() as usize % 8) % 8; + let head = head.min(out.len()); + for (i, byte) in out[..head].iter_mut().enumerate() { + *byte = source.read::(i); + } + let words = (out.len() - head) / 8 * 8; + if words > 0 { + source.sub(head, words).copy_out(0, &mut out[head..head + words]); + } + let tail = head + words; + for (i, byte) in out[tail..].iter_mut().enumerate() { + *byte = source.read::(tail + i); + } +} + +/// What a file or directory is. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Stat { + pub kind: u64, + pub size: u64, + pub mtime: u64, +} + +/// An open answered. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Opened { + pub fid: u64, + pub stat: Stat, + /// The connection this id belongs to: [`Dir::generation`] when it opened. + pub generation: u64, +} + +/// Why a request on a path was not answered with its result. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Refused { + /// The server's word. + Error(SyscallError), + /// The path met an absolute symlink; [`Dir::link_target`] holds the + /// absolute path, `.0` bytes long, to resolve in the client's own table. + Link(usize), +} + +impl From for Refused { + fn from(e: SyscallError) -> Self { + Refused::Error(e) + } +} + +/// One directory capability, connected. +pub struct Dir { + names: &'static Namespace, + name: [u8; MAX_SERVICE_NAME], + name_len: usize, + conn: Option, + window: SharedMemory, + generation: u64, + writable: bool, +} + +impl Dir { + /// Connect to the directory `names` calls `name`. `NotFound` is a name this + /// process was not given. + pub fn connect(names: &'static Namespace, name: &str) -> Result { + if name.len() > MAX_SERVICE_NAME { + return Err(SyscallError::InvalidArgument); + } + let mut buf = [0u8; MAX_SERVICE_NAME]; + buf[..name.len()].copy_from_slice(name.as_bytes()); + let window = SharedMemory::create(WINDOW_BYTES)?; + let mut dir = + Self { names, name: buf, name_len: name.len(), conn: None, window, generation: 0, writable: false }; + dir.reconnect()?; + Ok(dir) + } + + /// The capability's namespace name. + pub fn name(&self) -> &str { + core::str::from_utf8(&self.name[..self.name_len]).unwrap_or("") + } + + /// Counts this directory's connections; a file id is good only on the one + /// it was answered on. + pub fn generation(&self) -> u64 { + self.generation + } + + /// Whether the capability lets this process change what it names. + pub fn writable(&self) -> bool { + self.writable + } + + fn window(&self) -> Window { + // SAFETY: the region is `WINDOW_BYTES` long and mapped for as long as + // `self.window` lives, which every window taken here does not outlive. + unsafe { Window::new(self.window.as_ptr(), WINDOW_BYTES) } + } + + /// Whether the connection is up: `false` once a call found it ended, which + /// is the one `Gone` a reconnect answers. A `Gone` the server replied is + /// about the file, and a new connection would not change it. + pub fn connected(&self) -> bool { + self.conn.is_some() + } + + /// Open a fresh connection through the same connector, and lend it the + /// window again. The old one's file ids name nothing from here on. + pub fn reconnect(&mut self) -> Result<(), SyscallError> { + self.conn = None; + let name = core::str::from_utf8(&self.name[..self.name_len]).map_err(|_| SyscallError::InvalidArgument)?; + let conn = self.names.open(name)?; + let lent = self.window.share()?; + conn.send_with_handles(&[lent], HELLO, &Request::new()).map_err(transport)?; + let reply = receive(&conn)?.result()?; + self.writable = reply.value & RIGHT_WRITE != 0; + self.conn = Some(conn); + self.generation += 1; + Ok(()) + } + + /// One request and its reply on the current connection. A connection that + /// has ended answers `Gone` and is dropped, so the next call reconnects. + fn call(&mut self, op: u32, request: &Request) -> Result<(u32, Reply), SyscallError> { + if self.conn.is_none() { + self.reconnect()?; + } + let conn = self.conn.as_ref().expect("connected above"); + let answered = conn.send(op, request).map_err(transport).and_then(|()| { + let header = conn.recv_header().map_err(transport)?; + let reply: Reply = conn.recv_payload(&header).map_err(transport)?; + Ok((header.msg_type, reply)) + }); + if answered.is_err() { + self.conn = None; + } + answered + } + + /// A request naming paths: put them in the window first. A server that has + /// gone is asked again once on a fresh connection, since a path means the + /// same thing on both. + fn path_call(&mut self, op: u32, mut request: Request, first: &str, second: &str) -> Result { + if !canonical(first) || !canonical(second) { + return Err(Refused::Error(SyscallError::InvalidArgument)); + } + let mut attempt = 0; + loop { + window_put(self.window(), 0, first.as_bytes()); + window_put(self.window(), first.len(), second.as_bytes()); + request.len = first.len() as u64; + request.len2 = second.len() as u64; + match self.call(op, &request) { + Ok((LINK, reply)) => return Err(Refused::Link(reply.value.min(MAX_PATH as u64) as usize)), + Ok((REPLY, reply)) => return reply.result().map_err(Refused::Error), + Ok(_) => return Err(Refused::Error(SyscallError::Unknown)), + Err(SyscallError::Gone) if attempt == 0 => { + attempt += 1; + self.reconnect()?; + } + Err(e) => return Err(Refused::Error(e)), + } + } + } + + /// The absolute path a [`Refused::Link`] of `len` bytes left in the window. + pub fn link_target<'b>(&self, len: usize, buf: &'b mut [u8; MAX_PATH]) -> &'b [u8] { + window_take(self.window(), 0, &mut buf[..len]); + &buf[..len] + } + + /// A call on a file id: `Gone` when `generation` is not this connection's, + /// and the holder opens the file again by its path. + fn fid_call(&mut self, op: u32, generation: u64, request: &Request) -> Result { + if generation != self.generation || self.conn.is_none() { + return Err(SyscallError::Gone); + } + match self.call(op, request)? { + (REPLY, reply) => reply.result(), + _ => Err(SyscallError::Unknown), + } + } + + pub fn open(&mut self, path: &str, flags: u64) -> Result { + let reply = self.path_call(OPEN, Request { flags, ..Request::new() }, path, "")?; + Ok(Opened { fid: reply.value, stat: stat_of(&reply), generation: self.generation }) + } + + pub fn close(&mut self, fid: u64, generation: u64) { + let _ = self.fid_call(CLOSE, generation, &Request { fid, ..Request::new() }); + } + + /// Read at most `out.len()` bytes, and at most the window, at `offset`. + pub fn read(&mut self, fid: u64, generation: u64, offset: u64, out: &mut [u8]) -> Result { + let len = out.len().min(WINDOW_BYTES); + let reply = self.fid_call(READ, generation, &Request { fid, offset, len: len as u64, ..Request::new() })?; + let n = (reply.value as usize).min(len); + window_take(self.window(), 0, &mut out[..n]); + Ok(n) + } + + /// Write at most the window of `data` at `offset`, or at the end for a file + /// opened to append. Answers how much, and the offset after it. + pub fn write(&mut self, fid: u64, generation: u64, offset: u64, data: &[u8]) -> Result<(usize, u64), SyscallError> { + let len = data.len().min(WINDOW_BYTES); + if generation != self.generation { + return Err(SyscallError::Gone); + } + window_put(self.window(), 0, &data[..len]); + let reply = self.fid_call(WRITE, generation, &Request { fid, offset, len: len as u64, ..Request::new() })?; + Ok(((reply.value as usize).min(len), reply.value2)) + } + + pub fn fstat(&mut self, fid: u64, generation: u64) -> Result { + self.fid_call(FSTAT, generation, &Request { fid, ..Request::new() }).map(|r| stat_of(&r)) + } + + pub fn truncate(&mut self, fid: u64, generation: u64, size: u64) -> Result<(), SyscallError> { + self.fid_call(TRUNCATE, generation, &Request { fid, offset: size, ..Request::new() }).map(drop) + } + + /// Make what this file was written durable on the volume. + pub fn fsync(&mut self, fid: u64, generation: u64) -> Result<(), SyscallError> { + self.fid_call(FSYNC, generation, &Request { fid, ..Request::new() }).map(drop) + } + + /// Make the whole volume durable. + pub fn sync(&mut self) -> Result<(), Refused> { + self.path_call(SYNC, Request::new(), "", "").map(drop) + } + + /// A pipe whose bytes the server appends to this file from `offset` on, + /// for a program's standard output: the write end. + pub fn stream(&mut self, fid: u64, generation: u64, offset: u64) -> Result { + self.fid_call(STREAM, generation, &Request { fid, offset, ..Request::new() })?; + let conn = self.conn.as_ref().ok_or(SyscallError::Gone)?; + let [end] = conn.recv_handles_exact::<1>().ok_or(SyscallError::Unknown)?; + Ok(Pipe(crate::OwnedHandle(end))) + } + + pub fn stat(&mut self, path: &str, follow: bool) -> Result { + let op = if follow { STAT } else { LSTAT }; + self.path_call(op, Request::new(), path, "").map(|r| stat_of(&r)) + } + + /// The listing of `path`, [`for_each_entry`]'s encoding, into `out` when it + /// fits. Answers the listing's length either way, so a caller whose buffer + /// was short asks again with one that long. + pub fn read_dir(&mut self, path: &str, out: &mut [u8]) -> Result { + let reply = self.path_call(READDIR, Request::new(), path, "")?; + let n = (reply.value as usize).min(WINDOW_BYTES); + if n <= out.len() { + window_take(self.window(), 0, &mut out[..n]); + } + Ok(n) + } + + pub fn mkdir(&mut self, path: &str) -> Result<(), Refused> { + self.path_call(MKDIR, Request::new(), path, "").map(drop) + } + + pub fn rmdir(&mut self, path: &str) -> Result<(), Refused> { + self.path_call(RMDIR, Request::new(), path, "").map(drop) + } + + pub fn unlink(&mut self, path: &str) -> Result<(), Refused> { + self.path_call(UNLINK, Request::new(), path, "").map(drop) + } + + pub fn rename(&mut self, from: &str, to: &str) -> Result<(), Refused> { + self.path_call(RENAME, Request::new(), from, to).map(drop) + } + + /// Make `link` a symlink holding `target`, which is stored as written and + /// resolved by whoever follows it. + pub fn symlink(&mut self, target: &str, link: &str) -> Result<(), Refused> { + if target.is_empty() || target.len() > MAX_PATH { + return Err(Refused::Error(SyscallError::InvalidArgument)); + } + // The target is stored, not resolved, so it is not held to `canonical`. + if !canonical(link) { + return Err(Refused::Error(SyscallError::InvalidArgument)); + } + let mut request = Request::new(); + window_put(self.window(), 0, link.as_bytes()); + window_put(self.window(), link.len(), target.as_bytes()); + request.len = link.len() as u64; + request.len2 = target.len() as u64; + match self.call(SYMLINK, &request) { + Ok((REPLY, reply)) => reply.result().map(drop).map_err(Refused::Error), + Ok(_) => Err(Refused::Error(SyscallError::Unknown)), + Err(e) => Err(Refused::Error(e)), + } + } + + /// What the symlink at `path` holds, into `out`. + pub fn read_link(&mut self, path: &str, out: &mut [u8; MAX_PATH]) -> Result { + let reply = self.path_call(READLINK, Request::new(), path, "")?; + let n = (reply.value as usize).min(MAX_PATH); + window_take(self.window(), 0, &mut out[..n]); + Ok(n) + } +} + +/// Each entry of a listing [`Dir::read_dir`] answered: `(kind, size, name)`. +/// An entry is its kind byte, its size, its name's length and the name. +pub fn for_each_entry(listing: &[u8], mut f: impl FnMut(u64, u64, &str)) -> Result<(), SyscallError> { + let mut at = 0; + while at < listing.len() { + let head = listing.get(at..at + 11).ok_or(SyscallError::Unknown)?; + let kind = head[0] as u64; + let size = u64::from_le_bytes(head[1..9].try_into().expect("eight bytes")); + let len = u16::from_le_bytes([head[9], head[10]]) as usize; + let name = listing.get(at + 11..at + 11 + len).ok_or(SyscallError::Unknown)?; + f(kind, size, core::str::from_utf8(name).map_err(|_| SyscallError::Unknown)?); + at += 11 + len; + } + Ok(()) +} + +/// One listing entry's encoding, for the server: `None` once `out` is full. +pub fn encode_entry(out: &mut [u8], kind: u64, size: u64, name: &str) -> Option { + let need = 11 + name.len(); + if need > out.len() || name.len() > u16::MAX as usize { + return None; + } + out[0] = kind as u8; + out[1..9].copy_from_slice(&size.to_le_bytes()); + out[9..11].copy_from_slice(&(name.len() as u16).to_le_bytes()); + out[11..need].copy_from_slice(name.as_bytes()); + Some(need) +} + +fn stat_of(reply: &Reply) -> Stat { + Stat { kind: reply.kind, size: reply.value2, mtime: reply.mtime } +} + +fn receive(conn: &Connection) -> Result { + let header = conn.recv_header().map_err(transport)?; + if header.msg_type != REPLY { + return Err(SyscallError::Unknown); + } + conn.recv_payload(&header).map_err(transport) +} + +/// Every way a connection's transport can fail is the server having gone. +fn transport(e: IpcError) -> SyscallError { + match e { + IpcError::Disconnected => SyscallError::Gone, + IpcError::Syscall(SyscallError::Gone) => SyscallError::Gone, + IpcError::Syscall(e) => e, + IpcError::Malformed | IpcError::TooLarge => SyscallError::Unknown, + } +} + +impl AsHandle for Dir { + fn as_handle(&self) -> RawHandle { + self.conn.as_ref().map_or(toyos_abi::HANDLE_INVALID, |c| c.as_handle()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Every offset and length a path or a short read puts in the window, + /// aligned or not, round-trips byte for byte and touches nothing beside. + #[test] + fn the_window_copies_every_length_at_every_offset() { + let mut backing = vec![0u64; 64]; + let base = backing.as_mut_ptr() as *mut u8; + // SAFETY: 512 bytes of a live, 8-aligned allocation this test owns. + let window = unsafe { Window::new(base, 512) }; + for offset in 0..24 { + for len in 0..40 { + window.zero(); + let data: Vec = (0..len).map(|i| (i as u8).wrapping_mul(7).wrapping_add(1)).collect(); + window_put(window, offset, &data); + let mut back = vec![0u8; len]; + window_take(window, offset, &mut back); + assert_eq!(back, data, "{len} bytes at {offset}"); + let mut whole = vec![0u8; 512]; + window_take(window, 0, &mut whole); + assert!(whole[..offset].iter().all(|&b| b == 0), "before {offset}"); + assert!(whole[offset + len..].iter().all(|&b| b == 0), "after {offset}+{len}"); + } + } + } + + #[test] + fn a_path_is_canonical_only_without_empty_dot_or_dotdot_components() { + for good in ["", "a", "a/b", "toy/Documents/x.txt", "..a", "a..", ".x"] { + assert!(canonical(good), "{good:?}"); + } + for bad in ["/a", "a/", "a//b", ".", "..", "a/./b", "a/../b", "../x"] { + assert!(!canonical(bad), "{bad:?}"); + } + assert!(!canonical(&"a".repeat(MAX_PATH + 1))); + } + + #[test] + fn a_listing_entry_round_trips_and_a_short_one_is_refused() { + let mut buf = [0u8; 64]; + let n = encode_entry(&mut buf, KIND_DIR, 42, "Documents").unwrap(); + let mut seen = Vec::new(); + for_each_entry(&buf[..n], |k, s, name| seen.push((k, s, name.to_string()))).unwrap(); + assert_eq!(seen, [(KIND_DIR, 42, "Documents".to_string())]); + assert!(for_each_entry(&buf[..n - 1], |_, _, _| {}).is_err()); + assert_eq!(encode_entry(&mut buf[..10], KIND_FILE, 0, "x"), None); + } +} diff --git a/toyos/src/lib.rs b/toyos/src/lib.rs index 8907502451c..8ec991b865f 100644 --- a/toyos/src/lib.rs +++ b/toyos/src/lib.rs @@ -15,6 +15,7 @@ pub mod audio; pub mod census; pub mod device; pub mod endow; +pub mod fs; pub mod gpu; pub mod poller; pub mod power; diff --git a/userland/Cargo.lock b/userland/Cargo.lock index a1147e1c2b2..3bab68bc8c8 100644 --- a/userland/Cargo.lock +++ b/userland/Cargo.lock @@ -208,6 +208,10 @@ version = "1.8.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "2af50177e190e07a26ab74f8b1efbfe2ef87da2116221318cb1c2e82baf7de06" +[[package]] +name = "bcachefs" +version = "0.1.0" + [[package]] name = "bcrypt-pbkdf" version = "0.10.0" @@ -1271,6 +1275,20 @@ version = "0.3.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "aa9a19cbb55df58761df49b23516a86d432839add4af60fc256da840f66ed35b" +[[package]] +name = "fsd" +version = "0.0.0" +dependencies = [ + "bcachefs", + "blockd", + "toyos", + "toyos-abi", + "toyos-blockring", + "toyos-fat32", + "toyos-gpt", + "toyos-wallclock", +] + [[package]] name = "futures" version = "0.3.32" @@ -1632,6 +1650,7 @@ version = "0.1.0" dependencies = [ "toyos", "toyos-abi", + "toyos-blockring", "toyos-gpt", "toyos-logstream", "toyos-manifest", diff --git a/userland/Cargo.toml b/userland/Cargo.toml index 810f0398a15..c66e0561980 100644 --- a/userland/Cargo.toml +++ b/userland/Cargo.toml @@ -10,6 +10,7 @@ members = [ "filepicker", "filepicker-api", "files", + "fsd", "host", "init", "input-test", diff --git a/userland/blockd/src/lib.rs b/userland/blockd/src/lib.rs index 33f21bcc1b7..31b787ac9a5 100644 --- a/userland/blockd/src/lib.rs +++ b/userland/blockd/src/lib.rs @@ -8,5 +8,5 @@ pub mod nvme; pub mod region; pub mod session; -pub use session::{Answer, Error, Session, Unsent, Waited}; +pub use session::{list, Answer, Error, Session, Unsent, Waited}; pub use toyos_blockring::client::{Outcome, Ticket}; diff --git a/userland/blockd/src/main.rs b/userland/blockd/src/main.rs index c6a18a93192..215b1468c4e 100644 --- a/userland/blockd/src/main.rs +++ b/userland/blockd/src/main.rs @@ -23,9 +23,12 @@ //! A partition whose range is not whole 4 KiB blocks, or whose GUID the table //! carries twice, is listed and refused by name. //! -//! **What this process does not know** is which of its partitions the machine -//! is running from, so it refuses none of them for that -//! (`issues/filesystem/blockd-serves-the-slot-the-machine-runs-from.md`). +//! **The partition the machine runs from is refused to every session**: its +//! starter names it (`--running `, the ROOT the loader read), because a +//! writer there changes the image under the kernel that booted from it. +//! +//! **A client may ask what is served** ([`wire::MSG_LIST`]) before it opens +//! anything, which is how a file server finds its role's partition by type. //! //! **A server never blocks on a client.** Accept and the open frame are two //! events, the open is buffered until whole, every answer is one `try_send`, @@ -72,6 +75,10 @@ const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(2); /// always answers. const SILENCE_WRITE: &str = "--silence-write"; +/// Argv, followed by a unique GUID: the partition the machine runs from, which +/// no session opens. +const RUNNING: &str = "--running"; + const TOKEN_IRQ: u64 = 0; const TOKEN_ACCEPT: u64 = 1; const TOKEN_PENDING: u64 = 0x1_0000; @@ -80,6 +87,7 @@ const TOKEN_SESSION: u64 = 0x2_0000; /// One partition of the table, in 4 KiB blocks of the device. struct Part { unique: [u8; 16], + kind: [u8; 16], /// Where it starts and how long it is, or why it cannot be served. span: Result<(u64, u64), String>, } @@ -143,6 +151,7 @@ fn read_table(ctrl: &mut Controller) -> Vec { let mut parts = Vec::new(); for listed in &found[..scan.listed] { let unique = listed.unique_guid.0; + let kind = listed.type_guid.0; let span = match toyos_gpt::locate(&mut sectors, listed.unique_guid) { Err(why) => Err(format!("its own table refuses it: {why:?}")), Ok(located) => { @@ -159,7 +168,7 @@ fn read_table(ctrl: &mut Controller) -> Vec { } } }; - parts.push(Part { unique, span }); + parts.push(Part { unique, kind, span }); } parts } @@ -194,6 +203,10 @@ struct Service { losses: u64, sessions: BTreeMap, next_id: u64, + /// The partition the machine is running from, as the loader named it: + /// refused to every session, since a writer there changes the image + /// under the kernel that booted from it. + running: Option<[u8; 16]>, } /// A session decided on and not yet told to its client. @@ -216,6 +229,9 @@ impl Service { Some(Err(_)) => return Err(Refusal::Unusable), Some(Ok(span)) => *span, }; + if self.running == Some(guid) { + return Err(Refusal::Held); + } if self.sessions.len() >= MAX_SESSIONS { return Err(Refusal::Exhausted); } @@ -431,13 +447,75 @@ impl Service { } } -fn claim() -> toyos::PciDev { +/// The PCI function this process was endowed, or `None` on a machine that +/// has none its row names: init says which, and starts it anyway. +fn claim() -> Option { let label = Endowments::get() .labels() .find(|l| l.starts_with(DEV_PREFIX) && l[DEV_PREFIX.len()..].starts_with("pci:")) - .map(str::to_string) - .unwrap_or_else(|| panic!("blockd: started holding no PCI function")); - Endowments::get().take(&label).expect("blockd: the claim its label names") + .map(str::to_string)?; + Some(Endowments::get().take(&label).expect("blockd: the claim its label names")) +} + +/// Serve a machine with no controller: every listing empty, every open +/// `NotFound`, so a file server finds no partition here rather than waiting on +/// one. A connection says one frame and is answered, or is let go. +fn serve_nothing(acceptor: &toyos::port::Acceptor) -> ! { + let poller = Poller::new(2 + MAX_PENDING as u32); + let mut pending: Vec = Vec::new(); + let mut ready: Vec = Vec::new(); + loop { + if pending.len() < MAX_PENDING { + poller.watch(acceptor, READABLE, TOKEN_ACCEPT); + } + for p in &pending { + poller.watch(&p.conn, READABLE, TOKEN_PENDING + p.conn.as_handle().0 as u64); + } + let now = Instant::now(); + let timeout = pending + .iter() + .map(|p| HANDSHAKE_TIMEOUT.saturating_sub(now.duration_since(p.since))) + .min() + .map_or(u64::MAX, |left| left.as_nanos() as u64); + ready.clear(); + poller.wait(1, timeout, |token| ready.push(token)); + let now = Instant::now(); + pending.retain(|p| now.duration_since(p.since) < HANDSHAKE_TIMEOUT); + if ready.contains(&TOKEN_ACCEPT) { + match acceptor.accept() { + Ok(conn) => pending.push(Pending { conn, rx: ipc::FrameRx::new(), since: now }), + Err(why) => panic!("blockd: its own acceptor refused an accept: {why:?}"), + } + } + let mut i = 0; + while i < pending.len() { + let token = TOKEN_PENDING + pending[i].conn.as_handle().0 as u64; + if !ready.contains(&token) { + i += 1; + continue; + } + let step = { + let p = &mut pending[i]; + p.rx.pump(&p.conn) + }; + match step { + RxStep::Idle => i += 1, + RxStep::Eof | RxStep::Malformed => { + pending.remove(i); + } + RxStep::Frame { msg_type, .. } => { + let p = pending.remove(i); + if let Some(handles) = p.conn.recv_handles_exact::<1>() { + toyos_abi::syscall::close(handles[0]); + } + let _ = match msg_type { + wire::MSG_LIST => p.conn.try_send_bytes(wire::MSG_LISTED, &[]), + _ => p.conn.try_send_bytes(wire::MSG_REFUSED, &Refusal::NotFound.encode()), + }; + } + } + } + } } fn main() { @@ -448,8 +526,17 @@ fn main() { .filter(|n| *n > 0) .unwrap_or_else(|| panic!("blockd: {SILENCE_WRITE} takes the write whose answer to withhold, from 1")) }); - let dev = claim(); + let running = args.iter().position(|a| a == RUNNING).map(|at| { + args.get(at + 1) + .and_then(|text| PartGuid::parse(text)) + .map(|guid| guid.0) + .unwrap_or_else(|| panic!("blockd: {RUNNING} takes the running ROOT's unique GUID")) + }); let acceptor = endow::acceptor(PORT).unwrap_or_else(|| panic!("blockd: started serving no `{PORT}` port")); + let Some(dev) = claim() else { + println!("blockd: no NVMe controller this row names is on this machine; serving no partition"); + serve_nothing(&acceptor); + }; let mut ctrl = Controller::open(dev, silence).unwrap_or_else(|why| panic!("blockd: NOT SERVING — {why}")); println!( "blockd: NVMe up: {} I/O queues of {} commands, volatile write cache {}, {}-byte sectors, \ @@ -463,13 +550,18 @@ fn main() { let parts = read_table(&mut ctrl); for part in &parts { match &part.span { + Ok(_) if running == Some(part.unique) => println!( + "blockd: partition {} is the ROOT this machine runs from; refusing every session to it", + guid_text(part.unique) + ), Ok((first, blocks)) => { println!("blockd: partition {} at block {first}, {blocks} blocks", guid_text(part.unique)) } Err(why) => println!("blockd: partition {} is not served: {why}", guid_text(part.unique)), } } - let mut service = Service { ctrl, parts, holds: Holds::new(), losses: 0, sessions: BTreeMap::new(), next_id: 0 }; + let mut service = + Service { ctrl, parts, holds: Holds::new(), losses: 0, sessions: BTreeMap::new(), next_id: 0, running }; serve(&mut service, &acceptor); } @@ -574,11 +666,25 @@ fn doorbells(conn: &Connection) -> bool { } } -/// Answer one connection's first frame: an open, and nothing else. +/// Answer one connection's first frame: a listing, or an open. fn handshake(service: &mut Service, p: Pending, msg_type: u32, payload_len: usize) { let refuse = |conn: &Connection, why: Refusal| { let _ = conn.try_send_bytes(wire::MSG_REFUSED, &why.encode()); }; + if msg_type == wire::MSG_LIST && payload_len == 0 { + let listing: Vec = service + .parts + .iter() + .map(|p| wire::Listed { unique: p.unique, kind: p.kind }) + .flat_map(|l| l.encode()) + .collect(); + // One frame, and the connection is done with: a table larger than a + // frame is one this service lists no part of rather than half of. + if p.conn.try_send_bytes(wire::MSG_LISTED, &listing).is_err() { + refuse(&p.conn, Refusal::Exhausted); + } + return; + } let guid = wire::guid(p.rx.payload(payload_len)); let handles = p.conn.recv_handles_exact::<1>(); let (Some(guid), Some([region]), wire::MSG_OPEN) = (guid, handles, msg_type) else { diff --git a/userland/blockd/src/session.rs b/userland/blockd/src/session.rs index 2a9aea1d4f3..3f5741c8208 100644 --- a/userland/blockd/src/session.rs +++ b/userland/blockd/src/session.rs @@ -447,3 +447,20 @@ fn handshake( _ => Err(Error::Protocol), } } + +/// Every partition the service `names` calls `service` serves: what a client +/// that finds its partition by type asks before it opens one. +pub fn list(names: &Namespace, service: &str) -> Result, Error> { + let conn = names.open(service).map_err(Error::Kernel)?; + conn.signal(wire::MSG_LIST).map_err(|_| Error::Ended)?; + let header = conn.recv_header().map_err(|_| Error::Ended)?; + let mut payload = vec![0u8; header.len() as usize]; + let len = conn.recv_bytes(&header, &mut payload).map_err(|_| Error::Ended)?; + match header.msg_type { + wire::MSG_LISTED => { + wire::Listed::decode_all(&payload[..len]).map(Iterator::collect).ok_or(Error::Protocol) + } + wire::MSG_REFUSED => Err(Refusal::decode(&payload[..len]).map_or(Error::Protocol, Error::Refused)), + _ => Err(Error::Protocol), + } +} diff --git a/userland/fsd/Cargo.toml b/userland/fsd/Cargo.toml new file mode 100644 index 00000000000..6bd12505420 --- /dev/null +++ b/userland/fsd/Cargo.toml @@ -0,0 +1,24 @@ +[package] +name = "fsd" +edition = "2024" +license = "MIT OR Apache-2.0" + +# The library is the server's decisions — the block cache, the volumes and the +# resolver — so the host can test them; the service is `src/main.rs`. +[lib] +path = "src/lib.rs" +doctest = false + +[[bin]] +name = "fsd" +path = "src/main.rs" + +[dependencies] +toyos-abi = { path = "../../toyos-abi" } +toyos = { path = "../../toyos" } +toyos-blockring = { path = "../../toyos-blockring" } +toyos-gpt = { path = "../../toyos-gpt" } +toyos-fat32 = { path = "../../toyos-fat32" } +toyos-wallclock = { path = "../../toyos-wallclock" } +bcachefs = { path = "../../bcachefs" } +blockd = { path = "../blockd" } diff --git a/userland/fsd/src/absent.rs b/userland/fsd/src/absent.rs new file mode 100644 index 00000000000..fc7e52db8e0 --- /dev/null +++ b/userland/fsd/src/absent.rs @@ -0,0 +1,99 @@ +//! A role whose volume is not there this boot, or is ours and did not mount: +//! each directory the role serves exists and is empty, and nothing can be made +//! in it — a write into memory under the paths the owner's data lives at would +//! be taken and then lost, which is the harm a missing volume must not do. + +use toyos_abi::syscall::SyscallError; + +use crate::volume::{Kind, Meta, Node, OpenHow, Volume}; + +pub struct Absent { + roots: Vec, + why: String, +} + +impl Absent { + pub fn new(roots: &[&str], why: String) -> Self { + Self { roots: roots.iter().map(|r| r.to_string()).collect(), why } + } + + fn is_root(&self, path: &str) -> bool { + path.is_empty() || self.roots.iter().any(|r| r == path) + } +} + +impl Volume for Absent { + fn writable(&self) -> bool { + false + } + + fn lstat(&mut self, path: &str) -> Result { + match self.is_root(path) { + true => Ok(Meta { kind: Kind::Dir, size: 0, mtime: 0 }), + false => Err(SyscallError::NotFound), + } + } + + fn read_link(&mut self, _path: &str) -> Result { + Err(SyscallError::NotFound) + } + + fn list(&mut self, dir: &str) -> Result, SyscallError> { + match self.is_root(dir) { + true => Ok(Vec::new()), + false => Err(SyscallError::NotFound), + } + } + + fn open(&mut self, _path: &str, _how: OpenHow) -> Result { + Err(SyscallError::NotFound) + } + + fn close(&mut self, _node: Node) {} + + fn hold(&mut self, _node: Node) {} + + fn node_meta(&mut self, _node: Node) -> Result { + Err(SyscallError::NotFound) + } + + fn read(&mut self, _node: Node, _offset: u64, _out: &mut [u8]) -> Result { + Err(SyscallError::NotFound) + } + + fn write(&mut self, _node: Node, _offset: u64, _data: &[u8]) -> Result<(), SyscallError> { + Err(SyscallError::NotFound) + } + + fn truncate(&mut self, _node: Node, _size: u64) -> Result<(), SyscallError> { + Err(SyscallError::NotFound) + } + + fn mkdir(&mut self, _path: &str) -> Result<(), SyscallError> { + Err(SyscallError::NotFound) + } + + fn rmdir(&mut self, _path: &str) -> Result<(), SyscallError> { + Err(SyscallError::NotFound) + } + + fn unlink(&mut self, _path: &str) -> Result<(), SyscallError> { + Err(SyscallError::NotFound) + } + + fn rename(&mut self, _from: &str, _to: &str) -> Result<(), SyscallError> { + Err(SyscallError::NotFound) + } + + fn symlink(&mut self, _path: &str, _target: &str) -> Result<(), SyscallError> { + Err(SyscallError::NotFound) + } + + fn sync(&mut self) -> Result<(), SyscallError> { + Ok(()) + } + + fn describe(&self) -> String { + format!("absent: {}", self.why) + } +} diff --git a/userland/fsd/src/cache.rs b/userland/fsd/src/cache.rs new file mode 100644 index 00000000000..4697b528162 --- /dev/null +++ b/userland/fsd/src/cache.rs @@ -0,0 +1,343 @@ +//! The one cache every block of a volume passes through: a filesystem's +//! metadata and its files' data alike. +//! +//! **A write lands here and reaches the disk at a flush.** [`Cache::flush`] +//! writes every dirty block, lowest first and in runs, then asks the disk to +//! make them durable; a volume's sync is its own metadata written into the +//! cache and then this. A cache holding more than [`DIRTY_LIMIT`] dirty blocks +//! flushes by itself, so memory bounds the write-back and not the other way +//! round. +//! +//! **Clean blocks are kept up to [`CLEAN_LIMIT`]** and the oldest-read goes +//! first; a dirty block is never dropped. +//! +//! Interior mutability because the bcachefs crate reads through `&self`; the +//! server is one thread, so a `RefCell` and never a lock. + +use std::cell::RefCell; +use std::collections::{BTreeMap, VecDeque}; + +use crate::disk::{Disk, DiskError, BLOCK}; + +/// Clean blocks kept: 64 MiB. +pub const CLEAN_LIMIT: usize = 16 * 1024; + +/// Dirty blocks held before the cache flushes by itself: 32 MiB. +pub const DIRTY_LIMIT: usize = 8 * 1024; + +/// The longest run one disk request carries, in blocks. +const RUN: usize = 32; + +struct Slot { + data: Box<[u8; BLOCK]>, + dirty: bool, +} + +struct Inner { + disk: D, + slots: BTreeMap, + /// Clean blocks in the order they were read or cleaned; stale entries are + /// skipped when popped. + order: VecDeque, + dirty: usize, + reads: u64, + hits: u64, +} + +pub struct Cache { + inner: RefCell>, +} + +/// What a cache has done, for the server's `inspect` line. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] +pub struct Counts { + pub reads: u64, + pub hits: u64, + pub cached: usize, + pub dirty: usize, +} + +impl Cache { + pub fn new(disk: D) -> Self { + Self { + inner: RefCell::new(Inner { + disk, + slots: BTreeMap::new(), + order: VecDeque::new(), + dirty: 0, + reads: 0, + hits: 0, + }), + } + } + + pub fn blocks(&self) -> u64 { + self.inner.borrow().disk.blocks() + } + + /// The disk, once nothing else holds the cache: what it holds is what the + /// last flush left. + pub fn into_disk(self) -> D { + self.inner.into_inner().disk + } + + pub fn counts(&self) -> Counts { + let inner = self.inner.borrow(); + Counts { reads: inner.reads, hits: inner.hits, cached: inner.slots.len(), dirty: inner.dirty } + } + + /// Blocks `first..first + out.len() / BLOCK`, each from the cache where it + /// is there and in runs from the disk where it is not. + pub fn read(&self, first: u64, out: &mut [u8]) -> Result<(), DiskError> { + assert!(out.len() % BLOCK == 0, "fsd: a cache read of {} bytes", out.len()); + let count = out.len() / BLOCK; + let mut inner = self.inner.borrow_mut(); + let mut i = 0; + while i < count { + let block = first + i as u64; + inner.reads += 1; + if let Some(slot) = inner.slots.get(&block) { + out[i * BLOCK..(i + 1) * BLOCK].copy_from_slice(&slot.data[..]); + inner.hits += 1; + i += 1; + continue; + } + // The run of misses from here, which one request fetches. + let mut run = 1; + while i + run < count && run < RUN && !inner.slots.contains_key(&(block + run as u64)) { + run += 1; + } + let span = &mut out[i * BLOCK..(i + run) * BLOCK]; + inner.disk.read(block, span)?; + for (k, chunk) in span.chunks_exact(BLOCK).enumerate() { + let mut data = Box::new([0u8; BLOCK]); + data.copy_from_slice(chunk); + inner.slots.insert(block + k as u64, Slot { data, dirty: false }); + inner.order.push_back(block + k as u64); + } + i += run; + } + inner.evict(); + Ok(()) + } + + /// One block into `out`. + pub fn read_block(&self, block: u64, out: &mut [u8; BLOCK]) -> Result<(), DiskError> { + self.read(block, &mut out[..]) + } + + /// `data`, whole blocks from `first`, into the cache: on the disk at the + /// next flush. + pub fn write(&self, first: u64, data: &[u8]) -> Result<(), DiskError> { + assert!(data.len() % BLOCK == 0, "fsd: a cache write of {} bytes", data.len()); + let over = { + let mut inner = self.inner.borrow_mut(); + if first.checked_add((data.len() / BLOCK) as u64).is_none_or(|end| end > inner.disk.blocks()) { + return Err(DiskError::Range); + } + for (k, chunk) in data.chunks_exact(BLOCK).enumerate() { + let block = first + k as u64; + let newly = match inner.slots.get_mut(&block) { + Some(slot) => { + slot.data.copy_from_slice(chunk); + !std::mem::replace(&mut slot.dirty, true) + } + None => { + let mut buf = Box::new([0u8; BLOCK]); + buf.copy_from_slice(chunk); + inner.slots.insert(block, Slot { data: buf, dirty: true }); + true + } + }; + if newly { + inner.dirty += 1; + } + } + inner.dirty > DIRTY_LIMIT + }; + if over { + self.flush()?; + } + Ok(()) + } + + /// Every dirty block to the disk, lowest first and in runs, then the disk + /// made durable. A block the disk refused stays dirty. + pub fn flush(&self) -> Result<(), DiskError> { + let mut inner = self.inner.borrow_mut(); + let dirty: Vec = inner.slots.iter().filter(|(_, s)| s.dirty).map(|(b, _)| *b).collect(); + let mut i = 0; + let mut buf = vec![0u8; RUN * BLOCK]; + while i < dirty.len() { + let first = dirty[i]; + let mut run = 1; + while i + run < dirty.len() && run < RUN && dirty[i + run] == first + run as u64 { + run += 1; + } + for k in 0..run { + let slot = &inner.slots[&(first + k as u64)]; + buf[k * BLOCK..(k + 1) * BLOCK].copy_from_slice(&slot.data[..]); + } + inner.disk.write(first, &buf[..run * BLOCK])?; + for k in 0..run { + let block = first + k as u64; + inner.slots.get_mut(&block).expect("listed above").dirty = false; + inner.order.push_back(block); + inner.dirty -= 1; + } + i += run; + } + inner.disk.flush()?; + inner.evict(); + Ok(()) + } +} + +impl Inner { + /// Drop the oldest clean blocks past [`CLEAN_LIMIT`]. + fn evict(&mut self) { + while self.slots.len() - self.dirty > CLEAN_LIMIT { + let Some(block) = self.order.pop_front() else { break }; + if self.slots.get(&block).is_some_and(|s| !s.dirty) { + self.slots.remove(&block); + } + } + // A queue of mostly stale entries is compacted, so it stays the size + // of what it orders. + if self.order.len() > 4 * (CLEAN_LIMIT + DIRTY_LIMIT) { + let slots = &self.slots; + self.order.retain(|b| slots.get(b).is_some_and(|s| !s.dirty)); + } + } +} + +/// The cache as the bcachefs crate reads it: shared, since the volume reads +/// files' data through the same cache its btree lives in. +pub struct Shared(pub std::rc::Rc>); + +impl Clone for Shared { + fn clone(&self) -> Self { + Self(std::rc::Rc::clone(&self.0)) + } +} + +impl bcachefs::BlockIO for Shared { + fn read_block(&self, block: bcachefs::BlockNum, buf: &mut bcachefs::BlockBuf) -> Result<(), bcachefs::DeviceError> { + self.0.read(block.raw(), buf.as_bytes_mut()).map_err(|e| bcachefs::DeviceError::classify(&e)) + } + + fn write_block(&self, block: bcachefs::BlockNum, buf: &bcachefs::BlockBuf) -> Result<(), bcachefs::DeviceError> { + self.0.write(block.raw(), buf.as_bytes()).map_err(|e| bcachefs::DeviceError::classify(&e)) + } + + fn block_count(&self) -> u64 { + self.0.blocks() + } + + fn sync(&self) -> Result<(), bcachefs::DeviceError> { + self.0.flush().map_err(|e| bcachefs::DeviceError::classify(&e)) + } +} + +/// Every refusal here was attempted: nothing in this server refuses on a clock. +impl bcachefs::TransferError for DiskError { + fn refused_before_attempt(&self) -> bool { + false + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::disk::Ram; + + /// A disk that counts what reaches it. + struct Counting { + ram: Ram, + reads: u32, + writes: u32, + flushes: u32, + } + + impl Disk for Counting { + fn blocks(&self) -> u64 { + self.ram.blocks() + } + fn read(&mut self, first: u64, out: &mut [u8]) -> Result<(), DiskError> { + self.reads += 1; + self.ram.read(first, out) + } + fn write(&mut self, first: u64, data: &[u8]) -> Result<(), DiskError> { + self.writes += 1; + self.ram.write(first, data) + } + fn flush(&mut self) -> Result<(), DiskError> { + self.flushes += 1; + self.ram.flush() + } + } + + fn cache() -> Cache { + Cache::new(Counting { ram: Ram::new(1024), reads: 0, writes: 0, flushes: 0 }) + } + + #[test] + fn a_write_reaches_the_disk_at_the_flush_and_not_before() { + let c = cache(); + c.write(7, &[0xAB; BLOCK]).unwrap(); + assert_eq!(c.inner.borrow().disk.writes, 0); + let mut out = [0u8; BLOCK]; + c.read_block(7, &mut out).unwrap(); + assert_eq!(out, [0xAB; BLOCK], "a dirty block is served from the cache"); + c.flush().unwrap(); + let inner = c.inner.borrow(); + assert_eq!((inner.disk.writes, inner.disk.flushes, inner.dirty), (1, 1, 0)); + let mut on_disk = [0u8; BLOCK]; + drop(inner); + c.inner.borrow_mut().disk.ram.read(7, &mut on_disk).unwrap(); + assert_eq!(on_disk, [0xAB; BLOCK]); + } + + #[test] + fn contiguous_misses_are_one_request_and_a_second_read_is_none() { + let c = cache(); + let mut out = vec![0u8; 10 * BLOCK]; + c.read(100, &mut out).unwrap(); + assert_eq!(c.inner.borrow().disk.reads, 1); + c.read(100, &mut out).unwrap(); + assert_eq!(c.inner.borrow().disk.reads, 1); + assert_eq!(c.counts().hits, 10); + } + + #[test] + fn dirty_runs_go_out_in_runs() { + let c = cache(); + for b in [3u64, 4, 5, 9, 10] { + c.write(b, &[b as u8; BLOCK]).unwrap(); + } + c.flush().unwrap(); + assert_eq!(c.inner.borrow().disk.writes, 2, "3..=5 and 9..=10"); + } + + #[test] + fn a_write_past_the_disk_is_refused() { + let c = cache(); + assert_eq!(c.write(1024, &[0; BLOCK]), Err(DiskError::Range)); + assert_eq!(c.counts().dirty, 0); + } + + #[test] + fn clean_blocks_are_bounded_and_dirty_ones_are_kept() { + let c = Cache::new(Counting { ram: Ram::new((CLEAN_LIMIT + 64) as u64), reads: 0, writes: 0, flushes: 0 }); + c.write(0, &[1; BLOCK]).unwrap(); + let mut out = vec![0u8; BLOCK]; + for b in 1..(CLEAN_LIMIT as u64 + 32) { + c.read(b, &mut out).unwrap(); + } + let counts = c.counts(); + assert!(counts.cached - counts.dirty <= CLEAN_LIMIT); + assert_eq!(counts.dirty, 1); + c.read(0, &mut out).unwrap(); + assert_eq!(out, vec![1; BLOCK]); + } +} diff --git a/userland/fsd/src/data.rs b/userland/fsd/src/data.rs new file mode 100644 index 00000000000..acb34bcb9d5 --- /dev/null +++ b/userland/fsd/src/data.rs @@ -0,0 +1,808 @@ +//! The DATA role's volume: the bcachefs crate's format, read and written +//! through the server's cache. +//! +//! **The format's namespace is flat**: one entry per file or symlink, keyed by +//! its whole path, and a directory is only a prefix of those. So a directory +//! nothing is under yet is kept as an entry of its own, its path and a +//! trailing `/`, empty — which is what makes it outlive the server and the +//! boot — and a directory with anything under it is one whether or not that +//! entry exists, since volumes written before these entries carry none. +//! +//! **Every name is indexed in memory at mount** ([`DataVolume::names`]), so a +//! lookup, a listing and "is this a directory" are a map query and never the +//! walk of the whole tree the format's own `is_dir` is. +//! +//! **A file's data lives in the blocks its extents name, in the cache.** A +//! write resolves (allocating where it must) the block of each page it +//! touches and writes the page into the cache; its new length and extents +//! reach the file's entry at its close, an fsync, or the volume's sync, and a +//! sync writes every such entry before the cache is flushed. A length that +//! shrinks is recorded before the blocks past it are freed, so a failure +//! between the two leaks blocks rather than leaving an entry naming freed ones. +//! +//! **What a kill costs.** The format updates its btree in place and keeps no +//! journal (`issues/kernel/bcachefs-crate-is-not-bcachefs.md`): what the disk +//! holds is what the last sync wrote, and a server that dies inside a sync can +//! leave a node half of that sync's. Nothing but a sync writes a dirty block, +//! unless the cache is holding more than it keeps. +//! +//! Every name a client chose is bounded by the format ([`FsError::NameTooLong`]) +//! before it reaches the tree. + +use std::collections::BTreeMap; +use std::rc::Rc; + +use bcachefs::{Extent, Formatted, FsError, Mounted, ReadWrite}; +use toyos_abi::syscall::SyscallError; + +use crate::cache::{Cache, Shared}; +use crate::disk::{Disk, DiskError, BLOCK}; +use crate::volume::{join, parent, Kind, Meta, Node, OpenHow, Volume}; + +/// The longest symlink target read back: the wire's path bound. +const MAX_LINK: u64 = toyos::fs::MAX_PATH as u64; + +/// One file open, while any client holds it. +struct Open { + path: String, + extents: Vec, + size: u64, + mtime: u64, + /// Length or extents changed since the entry was last written. + dirty: bool, + /// Unlinked or renamed over: its blocks are no longer its own. + gone: bool, + holders: u32, +} + +pub struct DataVolume { + fs: Mounted, ReadWrite>, + cache: Rc>, + /// Every name on the volume, a directory entry's without its `/`. + names: BTreeMap, + /// Directories that exist whatever the volume holds: each connection's root. + roots: Vec, + open: BTreeMap, + by_path: BTreeMap, + next: Node, + clock: fn() -> u64, +} + +/// What a DATA partition turned out to be, as the kernel's mount decided it. +pub enum Probed { + /// A volume of ours, mounted, or a designated one freshly formatted. + Mounted(DataVolume), + /// A volume of ours that does not mount; nothing stands in and nothing is + /// written to it. + Unmountable(String), + /// No volume of ours and no designation stamp: never written. + Foreign, +} + +fn io(e: &FsError) -> SyscallError { + match e { + FsError::NotFound => SyscallError::NotFound, + FsError::NoSpace { .. } | FsError::EntryTooLarge { .. } | FsError::ListTooLong { .. } => { + SyscallError::ResourceExhausted + } + FsError::NameTooLong { .. } => SyscallError::InvalidArgument, + FsError::DeviceRead(..) + | FsError::DeviceWrite(..) + | FsError::DeviceSync(_) + | FsError::BadMagic { .. } + | FsError::UnsupportedVersion(_) + | FsError::ChecksumMismatch { .. } + | FsError::CorruptedKey(_) + | FsError::CorruptedNode(_) + | FsError::BlockOffDevice { .. } + | FsError::NotEnoughBlocks { .. } + | FsError::TreeTooDeep(_) + | FsError::BadSuperblock { .. } + | FsError::NodeOverfull { .. } + | FsError::TargetTooLong { .. } => SyscallError::Io, + } +} + +/// Logs the format's own account, answers the word a client can act on. +fn mapped(what: &str, path: &str, r: Result) -> Result { + r.map_err(|e| { + println!("fsd: {what} of '{path}' failed: {e:?}"); + io(&e) + }) +} + +fn disk_word(e: DiskError) -> SyscallError { + match e { + DiskError::Device | DiskError::Range => SyscallError::Io, + DiskError::Gone => SyscallError::Gone, + } +} + +/// The block holding page `page`, if the extents reach it. +fn block_for(extents: &[Extent], page: u64) -> Option { + let mut at = 0u64; + for e in extents { + let end = at.checked_add(e.block_count as u64)?; + if page < end { + return e.start_block.checked_add(page - at); + } + at = end; + } + None +} + +/// Keep the first `keep` blocks of `extents`; answer the runs dropped. +fn keep_blocks(extents: &mut Vec, keep: u64) -> Vec { + let mut dropped = Vec::new(); + let mut left = keep; + let mut kept = Vec::with_capacity(extents.len()); + for e in extents.drain(..) { + let n = e.block_count as u64; + if left >= n { + left -= n; + kept.push(e); + } else { + if left > 0 { + kept.push(Extent { start_block: e.start_block, block_count: left as u32, _reserved: 0 }); + } + dropped.push(Extent { start_block: e.start_block + left, block_count: (n - left) as u32, _reserved: 0 }); + left = 0; + } + } + *extents = kept; + dropped +} + +/// Whether block 0 carries a designation stamp naming this disk's size. +fn designated(cache: &Cache) -> bool { + let mut block0 = [0u8; BLOCK]; + if cache.read_block(0, &mut block0).is_err() { + println!("fsd: block 0 would not read; this disk is not ours to format"); + return false; + } + let magic = bcachefs::DESIGNATION_MAGIC; + let at = bcachefs::DESIGNATION_BLOCKS_OFFSET; + if block0[..magic.len()] != magic { + return false; + } + let stamped = u64::from_le_bytes(block0[at..at + 8].try_into().expect("eight bytes")); + if stamped != cache.blocks() { + println!( + "fsd: a designation stamp at block 0 names {stamped} blocks, and this partition has {}; ignoring it", + cache.blocks() + ); + return false; + } + true +} + +impl DataVolume { + /// Mount the volume on `disk`, format it if it is designated, and refuse + /// it otherwise — the kernel's DATA mount's decisions, unchanged. + pub fn probe(disk: D, roots: &[&str], clock: fn() -> u64) -> Probed { + let cache = Rc::new(Cache::new(disk)); + match Mounted::, ReadWrite>::open(Shared(Rc::clone(&cache))) { + Ok(fs) => { + println!("fsd: mounted the DATA volume"); + return Self::over(fs, cache, roots, clock).map_or_else(Probed::Unmountable, Probed::Mounted); + } + Err(e) if e.disowns_volume() => {} + Err(e) => return Probed::Unmountable(format!("{e:?}")), + } + if !designated(&cache) { + println!("fsd: no volume of ours and no designation stamp; nothing will be written to this partition"); + return Probed::Foreign; + } + println!("fsd: block 0 designates this partition for ToyOS; formatting it"); + match Formatted::format(Shared(Rc::clone(&cache))) { + Ok(fs) => Self::over(fs.mount(), cache, roots, clock).map_or_else(Probed::Unmountable, Probed::Mounted), + Err(e) => Probed::Unmountable(format!("the format failed: {e:?}")), + } + } + + /// A fresh volume on `disk`, which holds nothing of anyone's: memory. + pub fn format(disk: D, roots: &[&str], clock: fn() -> u64) -> Result { + let cache = Rc::new(Cache::new(disk)); + let fs = Formatted::format(Shared(Rc::clone(&cache))).map_err(|e| format!("{e:?}"))?; + Self::over(fs.mount(), cache, roots, clock) + } + + fn over( + fs: Mounted, ReadWrite>, + cache: Rc>, + roots: &[&str], + clock: fn() -> u64, + ) -> Result { + let listed = fs.list(usize::MAX, &|_| true).map_err(|e| format!("the volume would not list: {e:?}"))?; + let mut names = BTreeMap::new(); + for (name, _) in listed { + let kind = match name.strip_suffix('/') { + Some(dir) => { + names.insert(dir.to_string(), Kind::Dir); + continue; + } + None if fs.is_symlink(&name).map_err(|e| format!("{e:?}"))? => Kind::Symlink, + None => Kind::File, + }; + names.insert(name, kind); + } + Ok(Self { + fs, + cache, + names, + roots: roots.iter().map(|r| r.to_string()).collect(), + open: BTreeMap::new(), + by_path: BTreeMap::new(), + next: 1, + clock, + }) + } + + /// Whether anything is named beneath `dir`. + fn has_children(&self, dir: &str) -> bool { + let prefix = format!("{dir}/"); + self.names.range(prefix.clone()..).next().is_some_and(|(name, _)| name.starts_with(&prefix)) + } + + fn is_dir(&self, path: &str) -> bool { + path.is_empty() + || self.roots.iter().any(|r| r == path) + || self.names.get(path) == Some(&Kind::Dir) + || (!self.names.contains_key(path) && self.has_children(path)) + } + + fn entry(&mut self, node: Node) -> Result<&mut Open, SyscallError> { + let open = self.open.get_mut(&node).ok_or(SyscallError::NotFound)?; + if open.gone { + return Err(SyscallError::Gone); + } + Ok(open) + } + + /// Write one open file's length and extents to its entry. + fn persist(&mut self, node: Node) -> Result<(), SyscallError> { + let Some(open) = self.open.get_mut(&node) else { return Ok(()) }; + if !open.dirty || open.gone { + return Ok(()); + } + mapped("the entry", &open.path, self.fs.update_metadata(&open.path, &open.extents, open.size, open.mtime))?; + open.dirty = false; + Ok(()) + } + + /// Hand every open file at `path` its end: its blocks go with the name. + fn orphan(&mut self, path: &str) { + if let Some(node) = self.by_path.remove(path) { + if let Some(open) = self.open.get_mut(&node) { + open.gone = true; + } + } + } + + fn new_node(&mut self, path: &str, extents: Vec, size: u64, mtime: u64) -> Node { + let node = self.next; + self.next += 1; + self.open.insert(node, Open { path: path.to_string(), extents, size, mtime, dirty: false, gone: false, holders: 0 }); + self.by_path.insert(path.to_string(), node); + node + } + + /// A name is made under directories that need not exist yet — the format + /// makes every prefix one, as the kernel's mount did — but never under a + /// file or a symlink, which would then be two things at once. + fn require_parent_dir(&self, path: &str) -> Result<(), SyscallError> { + let mut dir = parent(path); + while !dir.is_empty() { + if matches!(self.names.get(dir), Some(Kind::File | Kind::Symlink)) { + return Err(SyscallError::NotFound); + } + dir = parent(dir); + } + Ok(()) + } +} + +impl Volume for DataVolume { + fn writable(&self) -> bool { + true + } + + fn lstat(&mut self, path: &str) -> Result { + match self.names.get(path).copied() { + Some(Kind::Dir) => Ok(Meta { kind: Kind::Dir, size: 0, mtime: 0 }), + Some(kind) => { + if let Some(open) = self.by_path.get(path).and_then(|n| self.open.get(n)) { + return Ok(Meta { kind, size: open.size, mtime: open.mtime }); + } + let (_, size) = mapped("a lookup", path, self.fs.file_extents(path))?.ok_or(SyscallError::NotFound)?; + let mtime = mapped("a lookup", path, self.fs.file_mtime(path))?.unwrap_or(0); + Ok(Meta { kind, size, mtime }) + } + None if self.is_dir(path) => Ok(Meta { kind: Kind::Dir, size: 0, mtime: 0 }), + None => Err(SyscallError::NotFound), + } + } + + fn read_link(&mut self, path: &str) -> Result { + if self.names.get(path) != Some(&Kind::Symlink) { + return Err(if self.names.contains_key(path) || self.is_dir(path) { + SyscallError::InvalidArgument + } else { + SyscallError::NotFound + }); + } + mapped("readlink", path, self.fs.read_link(path, MAX_LINK))?.ok_or(SyscallError::Io) + } + + fn list(&mut self, dir: &str) -> Result, SyscallError> { + if !self.is_dir(dir) { + return Err(if self.names.contains_key(dir) { SyscallError::InvalidArgument } else { SyscallError::NotFound }); + } + let prefix = if dir.is_empty() { String::new() } else { format!("{dir}/") }; + let mut children: BTreeMap = BTreeMap::new(); + for (name, kind) in self.names.range(prefix.clone()..) { + let Some(rest) = name.strip_prefix(&prefix) else { break }; + match rest.split_once('/') { + Some((child, _)) => { + children.insert(child.to_string(), Kind::Dir); + } + None => { + children.entry(rest.to_string()).or_insert(*kind); + } + } + } + let mut out = Vec::with_capacity(children.len()); + for (name, kind) in children { + let meta = match kind { + Kind::Dir => Meta { kind, size: 0, mtime: 0 }, + _ => self.lstat(&join(dir, &name))?, + }; + out.push((name, meta)); + } + Ok(out) + } + + fn open(&mut self, path: &str, how: OpenHow) -> Result { + if self.is_dir(path) { + return Err(SyscallError::InvalidArgument); + } + let node = match self.by_path.get(path).copied() { + Some(_) if how.create_new => return Err(SyscallError::AlreadyExists), + Some(node) => node, + None => match self.names.get(path).copied() { + Some(_) if how.create_new => return Err(SyscallError::AlreadyExists), + Some(Kind::File) => { + let (extents, size) = + mapped("open", path, self.fs.file_extents(path))?.ok_or(SyscallError::NotFound)?; + let mtime = mapped("open", path, self.fs.file_mtime(path))?.unwrap_or(0); + self.new_node(path, extents, size, mtime) + } + // A link the resolver did not follow is one with nothing behind it. + Some(_) => return Err(SyscallError::NotFound), + None if how.create || how.create_new => { + self.require_parent_dir(path)?; + let now = (self.clock)(); + mapped("create", path, self.fs.create(path, &[], now))?; + self.names.insert(path.to_string(), Kind::File); + self.new_node(path, Vec::new(), 0, now) + } + None => return Err(SyscallError::NotFound), + }, + }; + self.open.get_mut(&node).expect("just found or made").holders += 1; + if how.truncate { + if let Err(e) = self.truncate(node, 0) { + self.close(node); + return Err(e); + } + } + Ok(node) + } + + fn close(&mut self, node: Node) { + let Some(open) = self.open.get_mut(&node) else { return }; + open.holders -= 1; + if open.holders > 0 { + return; + } + if let Err(e) = self.persist(node) { + println!("fsd: node {node}'s entry was not written at its last close: {e:?}"); + } + let open = self.open.remove(&node).expect("present above"); + if self.by_path.get(&open.path) == Some(&node) { + self.by_path.remove(&open.path); + } + } + + fn hold(&mut self, node: Node) { + if let Some(open) = self.open.get_mut(&node) { + open.holders += 1; + } + } + + fn node_meta(&mut self, node: Node) -> Result { + let open = self.entry(node)?; + Ok(Meta { kind: Kind::File, size: open.size, mtime: open.mtime }) + } + + fn read(&mut self, node: Node, offset: u64, out: &mut [u8]) -> Result { + let cache = Rc::clone(&self.cache); + let open = self.entry(node)?; + if offset >= open.size { + return Ok(0); + } + let n = out.len().min((open.size - offset) as usize); + let mut done = 0; + let mut page_buf = vec![0u8; BLOCK]; + while done < n { + let at = offset + done as u64; + let page = at / BLOCK as u64; + let within = (at % BLOCK as u64) as usize; + // Whole pages whose blocks are consecutive go as one cache read. + if within == 0 && n - done >= BLOCK { + if let Some(first) = block_for(&open.extents, page) { + let mut run = 1u64; + while ((run + 1) as usize) * BLOCK <= n - done + && run < 256 + && block_for(&open.extents, page + run) == Some(first + run) + { + run += 1; + } + let span = &mut out[done..done + run as usize * BLOCK]; + cache.read(first, span).map_err(disk_word)?; + done += run as usize * BLOCK; + continue; + } + } + let take = (BLOCK - within).min(n - done); + match block_for(&open.extents, page) { + Some(block) => { + cache.read(block, &mut page_buf).map_err(disk_word)?; + out[done..done + take].copy_from_slice(&page_buf[within..within + take]); + } + // Past the extents: a hole, whose bytes are zeros. + None => out[done..done + take].fill(0), + } + done += take; + } + Ok(n) + } + + fn write(&mut self, node: Node, offset: u64, data: &[u8]) -> Result<(), SyscallError> { + let end = offset.checked_add(data.len() as u64).ok_or(SyscallError::InvalidArgument)?; + let cache = Rc::clone(&self.cache); + let now = (self.clock)(); + let open = self.open.get_mut(&node).ok_or(SyscallError::NotFound)?; + if open.gone { + return Err(SyscallError::Gone); + } + let mut done = 0; + let mut page_buf = vec![0u8; BLOCK]; + while done < data.len() { + let at = offset + done as u64; + let page = at / BLOCK as u64; + let within = (at % BLOCK as u64) as usize; + let take = (BLOCK - within).min(data.len() - done); + let existed = block_for(&open.extents, page); + let block = match existed { + Some(block) => block, + None => { + let page_idx = u32::try_from(page).map_err(|_| SyscallError::ResourceExhausted)?; + mapped("an allocation", &open.path, self.fs.resolve_or_alloc_block(&mut open.extents, page_idx))? + } + }; + if take == BLOCK { + cache.write(block, &data[done..done + BLOCK]).map_err(disk_word)?; + } else { + // A page written in part keeps the rest of what it held, which + // for a page the file never reached is zeros. + match existed { + Some(_) if page * (BLOCK as u64) < open.size => cache.read(block, &mut page_buf).map_err(disk_word)?, + _ => page_buf.fill(0), + } + page_buf[within..within + take].copy_from_slice(&data[done..done + take]); + cache.write(block, &page_buf).map_err(disk_word)?; + } + done += take; + } + open.size = open.size.max(end); + open.mtime = now; + open.dirty = true; + Ok(()) + } + + fn truncate(&mut self, node: Node, size: u64) -> Result<(), SyscallError> { + let cache = Rc::clone(&self.cache); + let now = (self.clock)(); + let open = self.open.get_mut(&node).ok_or(SyscallError::NotFound)?; + if open.gone { + return Err(SyscallError::Gone); + } + if size >= open.size { + open.size = size; + open.mtime = now; + open.dirty = true; + return Ok(()); + } + // What the last kept page holds past the new end is zeroed, so a file + // that grows again reads zeros there and not what it held before. + let within = (size % BLOCK as u64) as usize; + if within != 0 { + if let Some(block) = block_for(&open.extents, size / BLOCK as u64) { + let mut page = vec![0u8; BLOCK]; + cache.read(block, &mut page).map_err(disk_word)?; + page[within..].fill(0); + cache.write(block, &page).map_err(disk_word)?; + } + } + let dropped = keep_blocks(&mut open.extents, size.div_ceil(BLOCK as u64)); + open.size = size; + open.mtime = now; + // Recorded first, freed second. + mapped("a truncate", &open.path, self.fs.update_metadata(&open.path, &open.extents, size, now))?; + open.dirty = false; + let path = open.path.clone(); + mapped("a truncate's free", &path, self.fs.free_extents(&dropped)) + } + + fn mkdir(&mut self, path: &str) -> Result<(), SyscallError> { + if self.names.contains_key(path) || self.is_dir(path) { + return Err(SyscallError::AlreadyExists); + } + self.require_parent_dir(path)?; + let now = (self.clock)(); + mapped("mkdir", path, self.fs.create(&format!("{path}/"), &[], now))?; + self.names.insert(path.to_string(), Kind::Dir); + Ok(()) + } + + fn rmdir(&mut self, path: &str) -> Result<(), SyscallError> { + if self.roots.iter().any(|r| r == path) || path.is_empty() { + return Err(SyscallError::PermissionDenied); + } + if !self.is_dir(path) { + return Err(if self.names.contains_key(path) { SyscallError::InvalidArgument } else { SyscallError::NotFound }); + } + if self.has_children(path) { + return Err(SyscallError::InvalidArgument); + } + let marker = format!("{path}/"); + if !mapped("rmdir", path, self.fs.delete(&marker))? { + return Err(SyscallError::NotFound); + } + self.names.remove(path); + Ok(()) + } + + fn unlink(&mut self, path: &str) -> Result<(), SyscallError> { + match self.names.get(path) { + Some(Kind::File) | Some(Kind::Symlink) => {} + Some(Kind::Dir) => return Err(SyscallError::InvalidArgument), + None if self.is_dir(path) => return Err(SyscallError::InvalidArgument), + None => return Err(SyscallError::NotFound), + } + self.orphan(path); + if !mapped("unlink", path, self.fs.delete(path))? { + return Err(SyscallError::NotFound); + } + self.names.remove(path); + Ok(()) + } + + fn rename(&mut self, from: &str, to: &str) -> Result<(), SyscallError> { + if from == to { + return self.lstat(from).map(drop); + } + let kind = self.lstat(from)?.kind; + self.require_parent_dir(to)?; + if self.roots.iter().any(|r| r == from || r == to) { + return Err(SyscallError::PermissionDenied); + } + match kind { + Kind::File | Kind::Symlink => { + if self.is_dir(to) { + return Err(SyscallError::InvalidArgument); + } + // Written first, so a file renamed while open keeps its + // length under its new name. + if let Some(node) = self.by_path.get(from).copied() { + self.persist(node)?; + } + self.orphan(to); + mapped("rename", from, self.fs.rename(from, to))?; + self.names.remove(from); + self.names.insert(to.to_string(), kind); + if let Some(node) = self.by_path.remove(from) { + self.open.get_mut(&node).expect("indexed").path = to.to_string(); + self.by_path.insert(to.to_string(), node); + } + Ok(()) + } + Kind::Dir => { + if self.names.contains_key(to) || self.is_dir(to) { + return Err(SyscallError::AlreadyExists); + } + if to.starts_with(&format!("{from}/")) { + return Err(SyscallError::InvalidArgument); + } + // One entry at a time: the format has no rename of a prefix, + // so a kill in the middle leaves the directory in two halves, + // every entry under exactly one of its names. + let prefix = format!("{from}/"); + let moving: Vec<(String, Kind)> = self + .names + .range(prefix.clone()..) + .take_while(|(n, _)| n.starts_with(&prefix)) + .map(|(n, k)| (n.clone(), *k)) + .collect(); + for (name, kind) in &moving { + let moved = format!("{to}/{}", &name[prefix.len()..]); + if let Some(node) = self.by_path.get(name).copied() { + self.persist(node)?; + } + let (on_disk_from, on_disk_to) = match kind { + Kind::Dir => (format!("{name}/"), format!("{moved}/")), + _ => (name.clone(), moved.clone()), + }; + mapped("rename", name, self.fs.rename(&on_disk_from, &on_disk_to))?; + self.names.remove(name); + self.names.insert(moved.clone(), *kind); + if let Some(node) = self.by_path.remove(name) { + self.open.get_mut(&node).expect("indexed").path = moved.clone(); + self.by_path.insert(moved, node); + } + } + if self.names.get(from) == Some(&Kind::Dir) { + mapped("rename", from, self.fs.rename(&format!("{from}/"), &format!("{to}/")))?; + self.names.remove(from); + self.names.insert(to.to_string(), Kind::Dir); + } + Ok(()) + } + } + } + + fn symlink(&mut self, path: &str, target: &str) -> Result<(), SyscallError> { + if self.names.contains_key(path) || self.is_dir(path) { + return Err(SyscallError::AlreadyExists); + } + self.require_parent_dir(path)?; + mapped("symlink", path, self.fs.create_symlink(path, target))?; + self.names.insert(path.to_string(), Kind::Symlink); + Ok(()) + } + + fn sync(&mut self) -> Result<(), SyscallError> { + let nodes: Vec = self.open.keys().copied().collect(); + for node in nodes { + self.persist(node)?; + } + mapped("sync", "", self.fs.sync()) + } + + fn describe(&self) -> String { + let c = self.cache.counts(); + format!( + "bcachefs, {} names, {} files open; cache {} blocks ({} dirty), {} of {} reads hit", + self.names.len(), + self.open.len(), + c.cached, + c.dirty, + c.hits, + c.reads + ) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::disk::Ram; + + fn clock() -> u64 { + 1_000_000_000 + } + + fn vol() -> DataVolume { + DataVolume::format(Ram::new(4096), &["home", "apps"], clock).unwrap() + } + + const CREATE: OpenHow = OpenHow { create: true, create_new: false, truncate: false }; + const PLAIN: OpenHow = OpenHow { create: false, create_new: false, truncate: false }; + + #[test] + fn a_file_reads_back_what_was_written_across_pages_and_holes() { + let mut v = vol(); + let n = v.open("home/a", CREATE).unwrap(); + let data: Vec = (0..10_000u32).map(|i| i as u8).collect(); + v.write(n, 100, &data).unwrap(); + v.write(n, 20_000, b"tail").unwrap(); + let mut out = vec![0xFFu8; 20_004]; + assert_eq!(v.read(n, 0, &mut out).unwrap(), 20_004); + assert!(out[..100].iter().all(|&b| b == 0)); + assert_eq!(&out[100..10_100], &data[..]); + assert!(out[10_100..20_000].iter().all(|&b| b == 0), "a hole reads zeros"); + assert_eq!(&out[20_000..], b"tail"); + } + + #[test] + fn a_closed_file_keeps_its_length_across_a_remount() { + let mut v = vol(); + v.mkdir("home/toy").unwrap(); + let n = v.open("home/toy/x", CREATE).unwrap(); + v.write(n, 0, &[7; 5000]).unwrap(); + v.close(n); + v.sync().unwrap(); + let DataVolume { fs, cache, .. } = v; + drop(fs); + let disk = Rc::try_unwrap(cache).ok().expect("one owner").into_disk(); + let Probed::Mounted(mut again) = DataVolume::probe(disk, &["home"], clock) else { panic!("remount") }; + assert_eq!(again.lstat("home/toy/x").unwrap().size, 5000); + assert_eq!(again.lstat("home/toy").unwrap().kind, Kind::Dir, "an empty directory outlives the mount"); + let n = again.open("home/toy/x", PLAIN).unwrap(); + let mut out = vec![0u8; 5000]; + again.read(n, 0, &mut out).unwrap(); + assert_eq!(out, vec![7; 5000]); + } + + #[test] + fn a_file_unlinked_while_open_answers_gone() { + let mut v = vol(); + let n = v.open("home/a", CREATE).unwrap(); + v.write(n, 0, b"x").unwrap(); + v.unlink("home/a").unwrap(); + assert_eq!(v.read(n, 0, &mut [0u8; 1]), Err(SyscallError::Gone)); + assert_eq!(v.lstat("home/a"), Err(SyscallError::NotFound)); + } + + #[test] + fn a_shrink_zeroes_what_it_cut_and_regrowth_reads_zeros() { + let mut v = vol(); + let n = v.open("home/a", CREATE).unwrap(); + v.write(n, 0, &[9; 8192]).unwrap(); + v.truncate(n, 10).unwrap(); + v.truncate(n, 8192).unwrap(); + let mut out = vec![0xFFu8; 8192]; + v.read(n, 0, &mut out).unwrap(); + assert_eq!(&out[..10], &[9; 10]); + assert!(out[10..].iter().all(|&b| b == 0)); + } + + #[test] + fn directories_are_listed_and_refuse_what_posix_refuses() { + let mut v = vol(); + let f = v.open("home/nodir/a", CREATE).unwrap(); + v.close(f); + assert_eq!(v.lstat("home/nodir").unwrap().kind, Kind::Dir, "a create makes its directories"); + assert_eq!(v.open("home/nodir/a/b", CREATE), Err(SyscallError::NotFound), "never under a file"); + v.mkdir("home/d").unwrap(); + assert_eq!(v.mkdir("home/d"), Err(SyscallError::AlreadyExists)); + let n = v.open("home/d/f", CREATE).unwrap(); + v.close(n); + assert_eq!(v.rmdir("home/d"), Err(SyscallError::InvalidArgument), "not empty"); + let names: Vec = v.list("home").unwrap().into_iter().map(|(n, _)| n).collect(); + assert_eq!(names, ["d", "nodir"]); + v.unlink("home/d/f").unwrap(); + v.rmdir("home/d").unwrap(); + assert_eq!(v.lstat("home/d"), Err(SyscallError::NotFound)); + assert_eq!(v.rmdir("home"), Err(SyscallError::PermissionDenied), "a root stays"); + } + + #[test] + fn a_rename_moves_an_open_file_and_a_directory_with_its_contents() { + let mut v = vol(); + v.mkdir("home/a").unwrap(); + let n = v.open("home/a/f", CREATE).unwrap(); + v.write(n, 0, b"hello").unwrap(); + v.rename("home/a", "home/b").unwrap(); + assert_eq!(v.lstat("home/b/f").unwrap().size, 5); + let mut out = [0u8; 5]; + v.read(n, 0, &mut out).unwrap(); + assert_eq!(&out, b"hello"); + v.close(n); + let m = v.open("home/b/g", CREATE).unwrap(); + v.close(m); + v.rename("home/b/g", "home/b/f").unwrap(); + assert_eq!(v.lstat("home/b/f").unwrap().size, 0, "replaced"); + } +} diff --git a/userland/fsd/src/disk.rs b/userland/fsd/src/disk.rs new file mode 100644 index 00000000000..ba8c9a6668b --- /dev/null +++ b/userland/fsd/src/disk.rs @@ -0,0 +1,240 @@ +//! Where a volume's blocks come from: a partition blockd serves, a partition +//! the kernel's own disk serves through a claim, or memory. +//! +//! **A block is 4 KiB here, on every source.** A partition that is not whole +//! 4 KiB blocks is refused before it is served (blockd's table read, the +//! kernel's claim), so nothing below has a second block size. +//! +//! **A block service that ends is survived, not hidden**: a [`Served`] disk +//! whose session ended reconnects through the same name and issues again every +//! write no flush had covered (`toyos_blockring::client`), then asks what it +//! was asking. What was on the wire when the session ended is asked again +//! because a block write names its whole destination and means the same thing +//! twice. + +use std::collections::BTreeMap; + +use blockd::{Error as SessionError, Outcome, Session}; +use toyos::PartitionDev; +use toyos_abi::part::{Block, MAX_BLOCKS_PER_CALL}; +use toyos_blockring::MAX_REQUEST_BLOCKS; + +pub const BLOCK: usize = 4096; + +/// Why a disk did not do what it was asked. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum DiskError { + /// The device's own word: the transfer was attempted and did not happen, + /// or a flush left writes the device would not keep. + Device, + /// The block service is gone and would not come back. + Gone, + /// Past the end of the partition. + Range, +} + +/// A partition, a block at a time or several. +pub trait Disk { + fn blocks(&self) -> u64; + /// `out.len()` is whole blocks. + fn read(&mut self, first: u64, out: &mut [u8]) -> Result<(), DiskError>; + /// `data.len()` is whole blocks; durable once [`Disk::flush`] answers `Ok`. + fn write(&mut self, first: u64, data: &[u8]) -> Result<(), DiskError>; + /// Every write answered before this, out of the device's cache. + fn flush(&mut self) -> Result<(), DiskError>; +} + +fn span(first: u64, len: usize, blocks: u64) -> Result { + assert!(len % BLOCK == 0, "fsd: a transfer of {len} bytes is no whole blocks"); + let count = (len / BLOCK) as u64; + match first.checked_add(count) { + Some(end) if end <= blocks => Ok(count), + _ => Err(DiskError::Range), + } +} + +/// A volume in memory: the DATA role on a machine with no DATA partition, as +/// the kernel's tmpfs was. Sparse, so what nothing wrote costs nothing. +pub struct Ram { + blocks: u64, + written: BTreeMap>, +} + +impl Ram { + pub fn new(blocks: u64) -> Self { + Self { blocks, written: BTreeMap::new() } + } +} + +impl Disk for Ram { + fn blocks(&self) -> u64 { + self.blocks + } + + fn read(&mut self, first: u64, out: &mut [u8]) -> Result<(), DiskError> { + span(first, out.len(), self.blocks)?; + for (i, chunk) in out.chunks_exact_mut(BLOCK).enumerate() { + match self.written.get(&(first + i as u64)) { + Some(block) => chunk.copy_from_slice(&block[..]), + None => chunk.fill(0), + } + } + Ok(()) + } + + fn write(&mut self, first: u64, data: &[u8]) -> Result<(), DiskError> { + span(first, data.len(), self.blocks)?; + for (i, chunk) in data.chunks_exact(BLOCK).enumerate() { + let mut block = Box::new([0u8; BLOCK]); + block.copy_from_slice(chunk); + self.written.insert(first + i as u64, block); + } + Ok(()) + } + + fn flush(&mut self) -> Result<(), DiskError> { + Ok(()) + } +} + +/// A partition the kernel's own disk serves: the stick, until usbd serves it. +pub struct Claimed { + claim: PartitionDev, + blocks: u64, +} + +impl Claimed { + pub fn new(claim: PartitionDev) -> Result { + let info = claim.describe().map_err(|_| DiskError::Device)?; + Ok(Self { claim, blocks: info.blocks }) + } +} + +impl Disk for Claimed { + fn blocks(&self) -> u64 { + self.blocks + } + + fn read(&mut self, first: u64, out: &mut [u8]) -> Result<(), DiskError> { + span(first, out.len(), self.blocks)?; + let mut buf: Vec = vec![[0u8; BLOCK]; MAX_BLOCKS_PER_CALL]; + for (i, chunk) in out.chunks_mut(BLOCK * MAX_BLOCKS_PER_CALL).enumerate() { + let n = chunk.len() / BLOCK; + let at = first + (i * MAX_BLOCKS_PER_CALL) as u64; + self.claim.read(at, &mut buf[..n]).map_err(|_| DiskError::Device)?; + for (dst, src) in chunk.chunks_exact_mut(BLOCK).zip(&buf[..n]) { + dst.copy_from_slice(src); + } + } + Ok(()) + } + + fn write(&mut self, first: u64, data: &[u8]) -> Result<(), DiskError> { + span(first, data.len(), self.blocks)?; + let mut buf: Vec = vec![[0u8; BLOCK]; MAX_BLOCKS_PER_CALL]; + for (i, chunk) in data.chunks(BLOCK * MAX_BLOCKS_PER_CALL).enumerate() { + let n = chunk.len() / BLOCK; + for (dst, src) in buf[..n].iter_mut().zip(chunk.chunks_exact(BLOCK)) { + dst.copy_from_slice(src); + } + let at = first + (i * MAX_BLOCKS_PER_CALL) as u64; + self.claim.write(at, &buf[..n]).map_err(|_| DiskError::Device)?; + } + Ok(()) + } + + fn flush(&mut self) -> Result<(), DiskError> { + self.claim.sync().map_err(|_| DiskError::Device) + } +} + +/// A partition blockd serves. +pub struct Served { + session: Session, + blocks: u64, +} + +/// How many times one request is asked again across block service restarts +/// before the disk is called gone: init gives up on a service that ends +/// faster than this. +const MAX_RECONNECTS: u32 = 4; + +impl Served { + pub fn new(session: Session) -> Self { + let blocks = session.blocks(); + Self { session, blocks } + } + + /// Ask, and across a session's end, reconnect and ask again. + fn asked(&mut self, mut ask: impl FnMut(&mut Session) -> Result) -> Result { + let mut reconnects = 0; + loop { + match ask(&mut self.session) { + Ok(answer) => return Ok(answer), + Err(SessionError::Ended) if reconnects < MAX_RECONNECTS => { + reconnects += 1; + println!("fsd: the block service ended; reconnecting ({reconnects} of {MAX_RECONNECTS})"); + if let Err(why) = self.session.reconnect() { + println!("fsd: the block service would not take the session back: {why:?}"); + return Err(DiskError::Gone); + } + } + Err(_) => return Err(DiskError::Gone), + } + } + } +} + +impl Disk for Served { + fn blocks(&self) -> u64 { + self.blocks + } + + fn read(&mut self, first: u64, out: &mut [u8]) -> Result<(), DiskError> { + span(first, out.len(), self.blocks)?; + let per = MAX_REQUEST_BLOCKS as usize; + for (i, chunk) in out.chunks_mut(BLOCK * per).enumerate() { + let n = (chunk.len() / BLOCK) as u32; + let at = first + (i * per) as u64; + let (outcome, data) = self.asked(|s| s.read(at, n))?; + match (outcome, data) { + (Outcome::Done, Some(data)) if data.len() == chunk.len() => chunk.copy_from_slice(&data), + (Outcome::Refused, _) => return Err(DiskError::Gone), + _ => return Err(DiskError::Device), + } + } + Ok(()) + } + + fn write(&mut self, first: u64, data: &[u8]) -> Result<(), DiskError> { + span(first, data.len(), self.blocks)?; + let per = MAX_REQUEST_BLOCKS as usize; + for (i, chunk) in data.chunks(BLOCK * per).enumerate() { + let at = first + (i * per) as u64; + match self.asked(|s| s.write(at, chunk))? { + Outcome::Done => {} + // On the wire when the session ended: the write names its whole + // destination, so it is written again rather than guessed at. + Outcome::Refused => match self.asked(|s| s.write(at, chunk))? { + Outcome::Done => {} + _ => return Err(DiskError::Device), + }, + _ => return Err(DiskError::Device), + } + } + Ok(()) + } + + fn flush(&mut self) -> Result<(), DiskError> { + for _ in 0..=MAX_RECONNECTS { + match self.asked(Session::flush)? { + Outcome::Durable => return Ok(()), + // The session ended under it: the client issues its covered + // writes again first on the new one, then the flush is asked. + Outcome::Refused => continue, + _ => return Err(DiskError::Device), + } + } + Err(DiskError::Gone) + } +} diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs new file mode 100644 index 00000000000..10da11f2a69 --- /dev/null +++ b/userland/fsd/src/fat.rs @@ -0,0 +1,392 @@ +//! The LOG and BOOT roles' volumes: FAT32, over `toyos-fat32`, through the +//! server's cache. +//! +//! **A growing write is two writes: the chain, then the directory entry.** +//! Between them the volume holds clusters its entry does not reach, which +//! fatgen103 refuses, so every open file is brought level — its entry written, +//! its chain reconciled — at its last close, at an fsync and at a sync, as +//! the kernel's adapter did. +//! +//! **A create makes the directories on its way**, as the kernel's did: a log +//! file's path is its own, and FAT has the directories to carry it. A +//! directory made by itself is `mkdir`'s and needs its parent. +//! +//! FAT32 has no symlink representation, so a symlink is refused rather than +//! written as a file its reader would take for one. +//! +//! A volume that is not FAT32 is left untouched: `toyos-fat32` has no format +//! path and its probe writes nothing. + +use std::collections::BTreeMap; +use std::rc::Rc; + +use toyos_abi::syscall::SyscallError; +use toyos_fat32::{BlockAccess, Error, Fat32, FatTime, IoError}; + +use crate::cache::Cache; +use crate::disk::{Disk, BLOCK}; +use crate::volume::{parent, Kind, Meta, Node, OpenHow, Volume}; + +/// The most entries one directory listing materialises. +const MAX_LIST: usize = 16_384; + +/// The volume as `toyos-fat32` reads it: bytes, over the cache's blocks. +pub struct Bytes { + cache: Rc>, + /// The volume's own length, which is at most the partition's. + len: u64, +} + +impl BlockAccess for Bytes { + fn capacity(&self) -> u64 { + self.len + } + + fn read_at(&mut self, offset: u64, buf: &mut [u8]) -> Result<(), IoError> { + let end = offset.checked_add(buf.len() as u64).ok_or(IoError::Device)?; + if end > self.len { + return Err(IoError::Device); + } + let first = offset / BLOCK as u64; + let last = end.div_ceil(BLOCK as u64); + let mut span = vec![0u8; ((last - first) as usize) * BLOCK]; + self.cache.read(first, &mut span).map_err(|_| IoError::Device)?; + let at = (offset % BLOCK as u64) as usize; + buf.copy_from_slice(&span[at..at + buf.len()]); + Ok(()) + } + + fn write_at(&mut self, offset: u64, buf: &[u8]) -> Result<(), IoError> { + let end = offset.checked_add(buf.len() as u64).ok_or(IoError::Device)?; + if end > self.len { + return Err(IoError::Device); + } + let first = offset / BLOCK as u64; + let last = end.div_ceil(BLOCK as u64); + let mut span = vec![0u8; ((last - first) as usize) * BLOCK]; + let at = (offset % BLOCK as u64) as usize; + // What the write does not cover is another file's, or the table's: + // read before it is written back. + if at != 0 || buf.len() % BLOCK != 0 { + self.cache.read(first, &mut span).map_err(|_| IoError::Device)?; + } + span[at..at + buf.len()].copy_from_slice(buf); + self.cache.write(first, &span).map_err(|_| IoError::Device) + } + + fn flush(&mut self) -> Result<(), IoError> { + self.cache.flush().map_err(|_| IoError::Device) + } +} + +struct Open { + path: String, + file: toyos_fat32::File, + holders: u32, + gone: bool, +} + +pub struct FatVolume { + fs: Fat32>, + cache: Rc>, + writable: bool, + open: BTreeMap, + by_path: BTreeMap, + next: Node, + /// Local seconds since the epoch, which is the zone FAT stamps in. + clock: fn() -> u64, +} + +fn word(e: Error) -> SyscallError { + match e { + Error::NotFound => SyscallError::NotFound, + Error::AlreadyExists => SyscallError::AlreadyExists, + Error::Io | Error::NotFat32 | Error::Truncated | Error::CorruptChain | Error::CorruptDirectory => { + SyscallError::Io + } + // Nothing here refuses on a clock, so a repair pending is a volume + // that will not settle: the device's word. + Error::BudgetExpired | Error::RepairPending => SyscallError::Io, + Error::NotADirectory | Error::IsADirectory | Error::DirectoryNotEmpty | Error::InvalidName => { + SyscallError::InvalidArgument + } + Error::NoSpace | Error::TooLarge | Error::LimitExceeded => SyscallError::ResourceExhausted, + } +} + +fn logged(what: &str, path: &str, e: Error) -> SyscallError { + if e != Error::NotFound { + println!("fsd: {what} of '{path}': {e}"); + } + word(e) +} + +impl FatVolume { + /// Mount the FAT32 volume on `disk`, or say why not; nothing is written + /// before the volume is known to be FAT32. + pub fn mount(disk: D, writable: bool, clock: fn() -> u64) -> Result { + let cache = Rc::new(Cache::new(disk)); + let len = cache.blocks() * BLOCK as u64; + let mut bytes = Bytes { cache: Rc::clone(&cache), len }; + let geom = Fat32::probe(&mut bytes).map_err(|e| format!("no FAT32 here: {e}"))?; + // Tightened from the partition to the volume before anything writes. + bytes.len = geom.total_sectors as u64 * geom.bytes_per_sector as u64; + let fs = Fat32::mount(bytes).map_err(|e| format!("no FAT32 here: {e}"))?; + println!( + "fsd: FAT32 mounted, {} bytes, {}-byte sectors, {}-byte clusters", + geom.total_sectors as u64 * geom.bytes_per_sector as u64, + geom.bytes_per_sector, + geom.bytes_per_cluster() + ); + Ok(Self { fs, cache, writable, open: BTreeMap::new(), by_path: BTreeMap::new(), next: 1, clock }) + } + + fn time(&self) -> FatTime { + FatTime::from_unix_secs((self.clock)()) + } + + fn entry(&mut self, node: Node) -> Result<&mut Open, SyscallError> { + let open = self.open.get_mut(&node).ok_or(SyscallError::NotFound)?; + if open.gone { + return Err(SyscallError::Gone); + } + Ok(open) + } + + /// Bring one open file's entry level with its chain. + fn level(&mut self, node: Node) -> Result<(), SyscallError> { + let time = self.time(); + let Some(open) = self.open.get_mut(&node) else { return Ok(()) }; + if open.gone { + return Ok(()); + } + self.fs.flush_meta(&mut open.file, time).map_err(|e| logged("the entry", &open.path, e))?; + if open.file.needs_reconcile() { + self.fs.reconcile(&mut open.file, time).map_err(|e| logged("a reconcile", &open.path, e))?; + } + Ok(()) + } + + fn orphan(&mut self, path: &str) { + if let Some(node) = self.by_path.remove(path) { + if let Some(open) = self.open.get_mut(&node) { + open.gone = true; + } + } + } +} + +impl Volume for FatVolume { + fn writable(&self) -> bool { + self.writable + } + + fn lstat(&mut self, path: &str) -> Result { + if path.is_empty() { + return Ok(Meta { kind: Kind::Dir, size: 0, mtime: 0 }); + } + if let Some(open) = self.by_path.get(path).and_then(|n| self.open.get(n)) { + let len = open.file.len(); + let mtime = self.fs.metadata(path).map(|m| m.modified_unix).unwrap_or(0); + return Ok(Meta { kind: Kind::File, size: len, mtime: mtime * 1_000_000_000 }); + } + let meta = self.fs.metadata(path).map_err(|e| match e { + Error::NotADirectory => SyscallError::NotFound, + e => logged("metadata", path, e), + })?; + let kind = if meta.is_dir { Kind::Dir } else { Kind::File }; + Ok(Meta { kind, size: if meta.is_dir { 0 } else { meta.len }, mtime: meta.modified_unix * 1_000_000_000 }) + } + + fn read_link(&mut self, path: &str) -> Result { + self.lstat(path)?; + Err(SyscallError::InvalidArgument) + } + + fn list(&mut self, dir: &str) -> Result, SyscallError> { + let entries = self.fs.read_dir(dir, MAX_LIST).map_err(|e| logged("list", dir, e))?; + Ok(entries + .into_iter() + .map(|e| { + let kind = if e.is_dir { Kind::Dir } else { Kind::File }; + (e.name, Meta { kind, size: if e.is_dir { 0 } else { e.len }, mtime: e.modified_unix * 1_000_000_000 }) + }) + .collect()) + } + + fn open(&mut self, path: &str, how: OpenHow) -> Result { + if path.is_empty() { + return Err(SyscallError::InvalidArgument); + } + let node = match self.by_path.get(path).copied() { + Some(_) if how.create_new => return Err(SyscallError::AlreadyExists), + Some(node) => node, + None => { + let existing = match self.fs.metadata(path) { + Ok(meta) if meta.is_dir => return Err(SyscallError::InvalidArgument), + Ok(_) => true, + Err(Error::NotFound | Error::NotADirectory) => false, + Err(e) => return Err(logged("metadata", path, e)), + }; + let file = match existing { + true if how.create_new => return Err(SyscallError::AlreadyExists), + true => self.fs.open(path).map_err(|e| logged("open", path, e))?, + false if !(how.create || how.create_new) => return Err(SyscallError::NotFound), + false if !self.writable => return Err(SyscallError::PermissionDenied), + false => { + let time = self.time(); + let dir = parent(path); + if !dir.is_empty() { + self.fs.create_dir_all(dir, time).map_err(|e| logged("mkdir -p", dir, e))?; + } + self.fs.create(path, time).map_err(|e| logged("create", path, e))? + } + }; + let node = self.next; + self.next += 1; + self.open.insert(node, Open { path: path.to_string(), file, holders: 0, gone: false }); + self.by_path.insert(path.to_string(), node); + node + } + }; + self.open.get_mut(&node).expect("just found or made").holders += 1; + if how.truncate { + if let Err(e) = self.truncate(node, 0) { + self.close(node); + return Err(e); + } + } + Ok(node) + } + + fn close(&mut self, node: Node) { + let Some(open) = self.open.get_mut(&node) else { return }; + open.holders -= 1; + if open.holders > 0 { + return; + } + if self.writable { + if let Err(e) = self.level(node) { + println!("fsd: node {node} was not brought level at its last close: {e:?}"); + } + } + let open = self.open.remove(&node).expect("present above"); + if self.by_path.get(&open.path) == Some(&node) { + self.by_path.remove(&open.path); + } + } + + fn hold(&mut self, node: Node) { + if let Some(open) = self.open.get_mut(&node) { + open.holders += 1; + } + } + + fn node_meta(&mut self, node: Node) -> Result { + let open = self.entry(node)?; + let (path, size) = (open.path.clone(), open.file.len()); + let mtime = self.fs.metadata(&path).map(|m| m.modified_unix).unwrap_or(0); + Ok(Meta { kind: Kind::File, size, mtime: mtime * 1_000_000_000 }) + } + + fn read(&mut self, node: Node, offset: u64, out: &mut [u8]) -> Result { + let open = self.open.get_mut(&node).ok_or(SyscallError::NotFound)?; + if open.gone { + return Err(SyscallError::Gone); + } + self.fs.read(&mut open.file, offset, out).map_err(|e| logged("read", &open.path, e)) + } + + fn write(&mut self, node: Node, offset: u64, data: &[u8]) -> Result<(), SyscallError> { + let open = self.open.get_mut(&node).ok_or(SyscallError::NotFound)?; + if open.gone { + return Err(SyscallError::Gone); + } + self.fs.write(&mut open.file, offset, data).map_err(|e| logged("write", &open.path, e)) + } + + fn truncate(&mut self, node: Node, size: u64) -> Result<(), SyscallError> { + let open = self.open.get_mut(&node).ok_or(SyscallError::NotFound)?; + if open.gone { + return Err(SyscallError::Gone); + } + if open.file.len() == size { + return Ok(()); + } + self.fs.set_len(&mut open.file, size).map_err(|e| logged("set_len", &open.path, e)) + } + + fn mkdir(&mut self, path: &str) -> Result<(), SyscallError> { + let time = self.time(); + self.fs.create_dir(path, time).map_err(|e| logged("mkdir", path, e)) + } + + fn rmdir(&mut self, path: &str) -> Result<(), SyscallError> { + if path.is_empty() { + return Err(SyscallError::PermissionDenied); + } + self.fs.remove_dir(path).map_err(|e| logged("rmdir", path, e)) + } + + fn unlink(&mut self, path: &str) -> Result<(), SyscallError> { + if self.fs.metadata(path).map_err(|e| logged("metadata", path, e))?.is_dir { + return Err(SyscallError::InvalidArgument); + } + self.orphan(path); + self.fs.remove(path).map_err(|e| logged("unlink", path, e)) + } + + fn rename(&mut self, from: &str, to: &str) -> Result<(), SyscallError> { + // The source is judged before the destination is disturbed, and a + // rename onto the entry it already is — FAT names one entry by strings + // that differ in case — is POSIX's no-op and never a delete of it. + self.lstat(from)?; + if from == to || self.fs.same_entry(from, to).map_err(|e| logged("same_entry", from, e))? { + return Ok(()); + } + if let Some(node) = self.by_path.get(from).copied() { + self.level(node)?; + } + let replaced = self.fs.replace_rename(from, to).map_err(|e| { + if let Some(stranded) = &e.stranded { + println!("fsd: {to} could not be put back and is under {stranded}"); + } + logged("rename", from, e.cause) + })?; + self.orphan(to); + let released = self.fs.release_replaced(replaced).map_err(|e| logged("release", to, e)); + if let Some(node) = self.by_path.remove(from) { + self.open.get_mut(&node).expect("indexed").path = to.to_string(); + self.by_path.insert(to.to_string(), node); + } + released + } + + fn symlink(&mut self, _path: &str, _target: &str) -> Result<(), SyscallError> { + Err(SyscallError::NotSupported) + } + + fn sync(&mut self) -> Result<(), SyscallError> { + if !self.writable { + return Ok(()); + } + let nodes: Vec = self.open.keys().copied().collect(); + for node in nodes { + self.level(node)?; + } + self.fs.sync().map_err(|e| logged("sync", "", e)) + } + + fn describe(&self) -> String { + let c = self.cache.counts(); + format!( + "FAT32{}, {} files open; cache {} blocks ({} dirty), {} of {} reads hit", + if self.writable { "" } else { " read-only" }, + self.open.len(), + c.cached, + c.dirty, + c.hits, + c.reads + ) + } +} diff --git a/userland/fsd/src/lib.rs b/userland/fsd/src/lib.rs new file mode 100644 index 00000000000..afa56da0b48 --- /dev/null +++ b/userland/fsd/src/lib.rs @@ -0,0 +1,21 @@ +//! A file server's decisions: where its blocks come from ([`disk`]), the one +//! cache every byte of its volume passes through ([`cache`]), the volumes it +//! can serve ([`data`] for the bcachefs DATA role, [`fat`] for FAT32's LOG and +//! BOOT, [`absent`] for a role with no volume this boot), and the resolver that +//! keeps every path inside the directory a connection was given ([`resolve`]). +//! +//! **This is the page cache.** A block of the volume — a btree node, a FAT +//! sector, a file's data — is read into [`cache::Cache`] once and served from +//! there; a write lands there and reaches the disk at a sync, or when the +//! cache has more dirty blocks than it keeps. The kernel holds none of it. +//! +//! The service around it is `src/main.rs`: its port per directory, its clients +//! and their windows, and the wire (`toyos::fs`). + +pub mod absent; +pub mod cache; +pub mod data; +pub mod disk; +pub mod fat; +pub mod resolve; +pub mod volume; diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs new file mode 100644 index 00000000000..0219e8de7e1 --- /dev/null +++ b/userland/fsd/src/main.rs @@ -0,0 +1,777 @@ +//! fsd: one role's file server — DATA, LOG or BOOT — serving the directories +//! of that role as capabilities. +//! +//! **What it holds**: the acceptor of each directory its role serves, endowed +//! by init under `serve:fs:`; for its volume, a claim on a partition of a +//! disk the kernel drives, or blockd's `block` connector in its namespace; and +//! nothing else of the machine. Argv is the role, and for LOG and BOOT the +//! unique GUID of the partition the loader named for it. +//! +//! **A connection is bound to the directory whose port it came in on**, and +//! every path on it is resolved there (`fsd::resolve`); a write on a port +//! whose directory is read-only, or on a volume that is, is refused before the +//! volume sees it. +//! +//! **A server never blocks on a client.** Accept and the first frame are two +//! events; a request is buffered until whole; every reply is one `try_send`, +//! and a client whose pipe will not take one is dropped by name. The window a +//! client lends is read once per request into this process's memory, and +//! what is acted on is that copy. +//! +//! **What is written reaches the disk at a sync**: an fsync, a client's +//! `SYNC`, or [`WRITEBACK`] after the first unsynced write when nothing asked +//! sooner. A kill loses what no sync covered, which is POSIX's promise and no +//! more. + +use std::collections::BTreeMap; +use std::time::{Duration, Instant}; + +use fsd::absent::Absent; +use fsd::data::{DataVolume, Probed}; +use fsd::disk::{Claimed, Disk, Ram, Served}; +use fsd::fat::FatVolume; +use fsd::resolve::{self, Found, Refusal as Escape, Resolved}; +use fsd::volume::{Kind, Meta, Node, OpenHow, Volume}; +use toyos::endow::{self, Endowments}; +use toyos::fs::*; +use toyos::ipc::{self, Connection, RxStep}; +use toyos::poller::{Poller, READABLE}; +use toyos::port::Acceptor; +use toyos::shm::SharedMemory; +use toyos::volatile::Window; +use toyos::Pipe; +use toyos_abi::syscall::{SyscallError, DEV_PREFIX, SERVE_PREFIX}; + +/// How long a write waits for a sync nobody asked for. Policy: the kernel's +/// write-back drained a closed file's pages on its next pass. +const WRITEBACK: Duration = Duration::from_secs(2); + +/// Clients held at once; the next waits in the port's queue. +const MAX_CLIENTS: usize = 128; + +/// Files one client holds open at once. +const MAX_FIDS: usize = 1024; + +/// Streams served at once, machine-wide. +const MAX_STREAMS: usize = 64; + +/// How long an accepted connection may take to lend its window. +const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(2); + +/// The volume memory stands in for when DATA has no partition: 1 GiB of +/// blocks, of which only what is written costs anything. +const RAM_BLOCKS: u64 = 1 << 18; + +const TOKEN_CLIENT: u64 = 1 << 32; +const TOKEN_STREAM: u64 = 2 << 32; + +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +enum Role { + Data, + Log, + Boot, +} + +impl Role { + fn parse(text: &str) -> Option { + match text { + "data" => Some(Self::Data), + "log" => Some(Self::Log), + "boot" => Some(Self::Boot), + _ => None, + } + } +} + +/// One directory this server serves. +struct Capability { + /// Its absolute name, `/home`. + dir: String, + /// Where it is on the volume. + root: String, + writable: bool, + acceptor: Acceptor, +} + +/// One file a client holds open. +struct Fid { + node: Node, + write: bool, + append: bool, +} + +struct Client { + conn: Connection, + rx: ipc::FrameRx<{ core::mem::size_of::() }>, + cap: usize, + window: Option, + fids: BTreeMap, + next_fid: u64, + since: Instant, +} + +/// A pipe a program writes and this server appends to a file. +struct Stream { + pipe: Pipe, + node: Node, + offset: u64, +} + +/// Local seconds since the epoch, the zone FAT stamps in: bracketed so both +/// readings are of one second, as `userland/logd/src/wall.rs` does. +fn local_secs() -> u64 { + for _ in 0..4 { + let (Some(before), Some(local), Some(after)) = ( + toyos_abi::syscall::clock_epoch(), + toyos_abi::syscall::clock_realtime(), + toyos_abi::syscall::clock_epoch(), + ) else { + return 0; + }; + if before != after { + continue; + } + let lsod = local.hours as u64 * 3600 + local.minutes as u64 * 60 + local.seconds as u64; + return match toyos_wallclock::resolve(before, lsod) { + toyos_wallclock::Recovery::Offset(off) => before.saturating_add_signed(off), + // Two real zones a day apart: UTC, which is a time and not a guess. + toyos_wallclock::Recovery::Ambiguous { .. } => before, + }; + } + toyos_abi::syscall::clock_epoch().unwrap_or(0) +} + +fn main() { + let args: Vec = std::env::args().collect(); + let role = args.get(1).and_then(|r| Role::parse(r)).unwrap_or_else(|| { + panic!("fsd: started with {args:?}; the first argument is a role: data, log or boot") + }); + let caps = capabilities(role); + let roots: Vec = caps.iter().map(|c| c.root.clone()).collect(); + let roots: Vec<&str> = roots.iter().map(String::as_str).collect(); + let volume = open_volume(role, args.get(2).map(String::as_str), &roots); + println!( + "fsd: {role:?} serving {} — {}", + caps.iter().map(|c| c.dir.as_str()).collect::>().join(", "), + volume.describe() + ); + Server { + volume, + caps, + clients: BTreeMap::new(), + next_client: 0, + streams: BTreeMap::new(), + next_stream: 0, + dirty_since: None, + } + .serve() +} + +/// Every directory init endowed this process an acceptor for. +fn capabilities(role: Role) -> Vec { + let prefix = format!("{SERVE_PREFIX}{CAPABILITY_PREFIX}"); + let labels: Vec = + Endowments::get().labels().filter(|l| l.starts_with(&prefix)).map(str::to_string).collect(); + let mut caps = Vec::new(); + for label in labels { + let dir = label[prefix.len()..].to_string(); + let acceptor: Acceptor = Endowments::get().take(&label).expect("fsd: an acceptor its label names"); + let root = match role { + Role::Data => dir.trim_start_matches('/').to_string(), + Role::Log | Role::Boot => String::new(), + }; + caps.push(Capability { dir, root, writable: role != Role::Boot, acceptor }); + } + assert!(!caps.is_empty(), "fsd: started serving no directory"); + caps +} + +/// The partition claim init minted, when the role's partition is on a disk +/// the kernel drives. +fn claimed() -> Option { + let prefix = format!("{DEV_PREFIX}part:"); + let label = Endowments::get().labels().find(|l| l.starts_with(&prefix))?.to_string(); + let claim: toyos::PartitionDev = Endowments::get().take(&label)?; + match Claimed::new(claim) { + Ok(disk) => Some(disk), + Err(why) => { + println!("fsd: the partition claim would not describe itself: {why:?}"); + None + } + } +} + +/// A session on the partition blockd serves under `guid`. +fn served(guid: [u8; 16]) -> Option { + let names = endow::namespace()?; + let own = toyos::namespace::build().keep(names, &[toyos_blockring::PORT]).finish().ok()?; + match blockd::Session::open(own, toyos_blockring::PORT, guid) { + Ok(session) => Some(Served::new(session)), + Err(why) => { + println!("fsd: blockd would not open the partition: {why:?}"); + None + } + } +} + +/// The TOYOS-DATA partitions blockd serves. +fn data_partitions() -> Vec<[u8; 16]> { + let Some(names) = endow::namespace() else { return Vec::new() }; + match blockd::list(names, toyos_blockring::PORT) { + Ok(listed) => listed.into_iter().filter(|l| l.kind == toyos_gpt::Guid::TOYOS_DATA.0).map(|l| l.unique).collect(), + // No block service in this image: the namespace has no `block`. + Err(blockd::Error::Kernel(SyscallError::NotFound)) => Vec::new(), + Err(why) => { + println!("fsd: blockd would not list its partitions: {why:?}"); + Vec::new() + } + } +} + +fn open_volume(role: Role, guid: Option<&str>, roots: &[&str]) -> Box { + match role { + Role::Data => { + if let Some(disk) = claimed() { + return data_on(disk, roots); + } + match data_partitions().as_slice() { + [one] => match served(*one) { + Some(disk) => data_on(disk, roots), + None => Box::new(Absent::new(roots, "the DATA partition would not open".into())), + }, + [] => ram(roots, "this machine has no DATA partition"), + many => ram(roots, &format!("this machine has {} DATA partitions, and a volume is one", many.len())), + } + } + Role::Log | Role::Boot => { + let writable = role == Role::Log; + let mounted = if let Some(disk) = claimed() { + Some(fat_on(disk, writable)) + } else { + guid.and_then(toyos_abi::part::PartGuid::parse).and_then(|g| served(g.0)).map(|disk| fat_on(disk, writable)) + }; + match mounted { + Some(Ok(volume)) => volume, + Some(Err(why)) => { + println!("fsd: the {role:?} partition does not mount: {why}"); + Box::new(Absent::new(roots, why)) + } + None => Box::new(Absent::new(roots, format!("no {role:?} partition on a disk this server reaches"))), + } + } + } +} + +fn data_on(disk: D, roots: &[&str]) -> Box { + match DataVolume::probe(disk, roots, fsd::volume::now_nanos) { + Probed::Mounted(volume) => Box::new(volume), + Probed::Unmountable(why) => { + println!("fsd: the DATA volume is ours and does not mount ({why}); it is absent this boot"); + Box::new(Absent::new(roots, why)) + } + Probed::Foreign => ram(roots, "the DATA partition is not ours"), + } +} + +fn ram(roots: &[&str], why: &str) -> Box { + println!("fsd: {why}; /apps, /config, /home and /state are in memory and will not survive a reboot"); + match DataVolume::format(Ram::new(RAM_BLOCKS), roots, fsd::volume::now_nanos) { + Ok(volume) => Box::new(volume), + Err(why) => panic!("fsd: a volume in memory would not format: {why}"), + } +} + +fn fat_on(disk: D, writable: bool) -> Result, String> { + FatVolume::mount(disk, writable, local_secs).map(|v| Box::new(v) as Box) +} + +struct Server { + volume: Box, + caps: Vec, + clients: BTreeMap, + next_client: u64, + streams: BTreeMap, + next_stream: u64, + /// When a write first went unsynced. + dirty_since: Option, +} + +/// What one request is answered. +enum Answer { + Reply(Reply), + /// A path the client resolves again, `len` bytes of the window. + Link(usize), + /// A reply that carries a handle. + WithHandle(Reply, toyos::RawHandle), + /// The client broke the protocol and is let go. + Drop(&'static str), +} + +fn stat_reply(meta: Meta) -> Reply { + Reply { status: 0, kind: meta.kind.wire(), value: 0, value2: meta.size, mtime: meta.mtime } +} + +fn window_of(w: &SharedMemory) -> Window { + // SAFETY: the region is `WINDOW_BYTES` long (it came through `adopt` at + // that size, and shared memory is never less) and lives as long as `w`. + unsafe { Window::new(w.as_ptr(), WINDOW_BYTES) } +} + +impl Server { + fn serve(mut self) -> ! { + let poller = Poller::new(Poller::MAX_HANDLES); + let mut ready = Vec::new(); + loop { + if self.clients.len() < MAX_CLIENTS { + for (i, cap) in self.caps.iter().enumerate() { + poller.watch(&cap.acceptor, READABLE, i as u64); + } + } + for (id, c) in &self.clients { + poller.watch(&c.conn, READABLE, TOKEN_CLIENT + id); + } + for (id, s) in &self.streams { + poller.watch(&s.pipe, READABLE, TOKEN_STREAM + id); + } + let now = Instant::now(); + let timeout = self + .clients + .values() + .filter(|c| c.window.is_none()) + .map(|c| HANDSHAKE_TIMEOUT.saturating_sub(now.duration_since(c.since))) + .chain(self.dirty_since.map(|at| WRITEBACK.saturating_sub(now.duration_since(at)))) + .min() + .map_or(u64::MAX, |left| left.as_nanos() as u64); + ready.clear(); + poller.wait(1, timeout, |t| ready.push(t)); + + for &token in &ready { + if token < TOKEN_CLIENT { + self.accept(token as usize); + } else if token < TOKEN_STREAM { + self.pump(token - TOKEN_CLIENT); + } else { + self.drain(token - TOKEN_STREAM); + } + } + let now = Instant::now(); + let late: Vec = self + .clients + .iter() + .filter(|(_, c)| c.window.is_none() && now.duration_since(c.since) >= HANDSHAKE_TIMEOUT) + .map(|(id, _)| *id) + .collect(); + for id in late { + self.drop_client(id, "it never lent its window"); + } + if self.dirty_since.is_some_and(|at| now.duration_since(at) >= WRITEBACK) { + if let Err(e) = self.volume.sync() { + println!("fsd: the write-back sync failed: {e:?}"); + } + self.dirty_since = None; + } + } + } + + fn accept(&mut self, cap: usize) { + let conn = match self.caps[cap].acceptor.accept() { + Ok(conn) => conn, + Err(why) => panic!("fsd: its own acceptor refused an accept: {why:?}"), + }; + let id = self.next_client; + self.next_client += 1; + let client = Client { + conn, + rx: ipc::FrameRx::new(), + cap, + window: None, + fids: BTreeMap::new(), + next_fid: 1, + since: Instant::now(), + }; + self.clients.insert(id, client); + } + + fn drop_client(&mut self, id: u64, why: &str) { + let Some(client) = self.clients.remove(&id) else { return }; + if !why.is_empty() { + println!("fsd: dropping client {id} of {}: {why}", self.caps[client.cap].dir); + } + for fid in client.fids.values() { + self.volume.close(fid.node); + } + } + + /// Everything one client has sent, a whole request at a time. + fn pump(&mut self, id: u64) { + loop { + let Some(client) = self.clients.get_mut(&id) else { return }; + match client.rx.pump(&client.conn) { + RxStep::Idle => return, + RxStep::Eof => return self.drop_client(id, ""), + RxStep::Malformed => return self.drop_client(id, "it sent a frame this protocol cannot describe"), + RxStep::Frame { msg_type, payload_len } => { + let Ok(request) = ipc::decode_payload::(client.rx.payload(payload_len)) else { + return self.drop_client(id, "its request was short"); + }; + let answer = self.answer(id, msg_type, request); + if !self.reply(id, answer) { + return; + } + } + } + } + } + + /// Send `answer`; `false` once the client is gone. + fn reply(&mut self, id: u64, answer: Answer) -> bool { + let Some(client) = self.clients.get(&id) else { return false }; + let sent = match answer { + Answer::Reply(reply) => client.conn.try_send(REPLY, &reply), + Answer::Link(len) => client.conn.try_send(LINK, &Reply { value: len as u64, ..Reply::ok() }), + Answer::WithHandle(reply, handle) => client.conn.try_send_with_handles(&[handle], REPLY, &reply), + Answer::Drop(why) => { + self.drop_client(id, why); + return false; + } + }; + if let Err(why) = sent { + self.drop_client(id, &format!("its connection would not take a reply ({why:?})")); + return false; + } + true + } + + /// A path from the window: `len` bytes at `at`, UTF-8. + fn path(&self, id: u64, at: u64, len: u64) -> Result { + let window = self.clients[&id].window.as_ref().ok_or(SyscallError::InvalidArgument)?; + let (at, len) = (at as usize, len as usize); + if len > MAX_PATH || at.checked_add(len).is_none_or(|end| end > WINDOW_BYTES) { + return Err(SyscallError::InvalidArgument); + } + let mut bytes = vec![0u8; len]; + window_take(window_of(window), at, &mut bytes); + String::from_utf8(bytes).map_err(|_| SyscallError::InvalidArgument) + } + + /// `rel`, from the client's directory, as a volume path. + fn resolve(&mut self, id: u64, rel: &str, follow_last: bool) -> Result { + if !canonical(rel) { + return Err(SyscallError::InvalidArgument); + } + let root = self.caps[self.clients[&id].cap].root.clone(); + let volume = &mut self.volume; + let mut lookup = |p: &str| match volume.lstat(p) { + Ok(meta) if meta.kind == Kind::Symlink => volume.read_link(p).map(Found::Link).map_err(drop), + Ok(_) | Err(SyscallError::NotFound) => Ok(Found::Other), + Err(_) => Err(()), + }; + resolve::resolve(&root, rel, follow_last, &mut lookup).map_err(|e| match e { + Escape::Escape => SyscallError::PermissionDenied, + Escape::Loop => SyscallError::InvalidArgument, + Escape::NotFound => SyscallError::NotFound, + Escape::Io => SyscallError::Io, + }) + } + + /// Put `bytes` in the client's window at 0. + fn put(&self, id: u64, bytes: &[u8]) { + let window = self.clients[&id].window.as_ref().expect("a request comes after the window"); + window_put(window_of(window), 0, bytes); + } + + fn writable(&self, id: u64) -> bool { + self.caps[self.clients[&id].cap].writable && self.volume.writable() + } + + fn answer(&mut self, id: u64, op: u32, r: Request) -> Answer { + if op == HELLO { + let client = self.clients.get_mut(&id).expect("pumped"); + if client.window.is_some() { + return Answer::Drop("it lent a second window"); + } + let Some([lent]) = client.conn.recv_handles_exact::<1>() else { + return Answer::Drop("its hello carried no window"); + }; + match SharedMemory::adopt(lent, WINDOW_BYTES) { + Ok(window) => client.window = Some(window), + Err(_) => return Answer::Drop("its window would not map"), + } + let rights = if self.writable(id) { RIGHT_WRITE } else { 0 }; + return Answer::Reply(Reply { value: rights, ..Reply::ok() }); + } + if self.clients[&id].window.is_none() { + return Answer::Drop("it asked before it lent a window"); + } + match self.serve_one(id, op, r) { + Ok(answer) => answer, + Err(e) => Answer::Reply(Reply::refused(e)), + } + } + + fn fid(&self, id: u64, fid: u64) -> Result<&Fid, SyscallError> { + self.clients[&id].fids.get(&fid).ok_or(SyscallError::InvalidArgument) + } + + /// Mark the volume written: the write-back sync is due from here. + fn dirtied(&mut self) { + self.dirty_since.get_or_insert_with(Instant::now); + } + + /// `rel` resolved without following its last component, for an operation + /// on the name itself; an absolute link on the way is the client's. + fn name(&mut self, id: u64, rel: &str) -> Result, SyscallError> { + Ok(match self.resolve(id, rel, false)? { + Resolved::Path(p) => Ok(p), + Resolved::Absolute(p) => Err(self.link(id, &p)), + }) + } + + fn serve_one(&mut self, id: u64, op: u32, r: Request) -> Result { + let changes = matches!(op, WRITE | TRUNCATE | MKDIR | RMDIR | UNLINK | RENAME | SYMLINK | STREAM) + || (op == OPEN && r.flags & (O_WRITE | O_APPEND | O_CREATE | O_TRUNCATE | O_CREATE_NEW) != 0); + if changes && !self.writable(id) { + return Err(SyscallError::PermissionDenied); + } + match op { + OPEN => { + if r.flags & !O_KNOWN != 0 { + return Err(SyscallError::InvalidArgument); + } + let rel = self.path(id, 0, r.len)?; + let path = match self.resolve(id, &rel, true)? { + Resolved::Absolute(p) => return Ok(self.link(id, &p)), + Resolved::Path(p) => p, + }; + if self.clients[&id].fids.len() >= MAX_FIDS { + return Err(SyscallError::ResourceExhausted); + } + let how = OpenHow { + create: r.flags & O_CREATE != 0, + create_new: r.flags & O_CREATE_NEW != 0, + truncate: r.flags & O_TRUNCATE != 0, + }; + let node = self.volume.open(&path, how)?; + let meta = match self.volume.node_meta(node) { + Ok(meta) => meta, + Err(e) => { + self.volume.close(node); + return Err(e); + } + }; + if how.create || how.truncate || how.create_new { + self.dirtied(); + } + let client = self.clients.get_mut(&id).expect("pumped"); + let fid = client.next_fid; + client.next_fid += 1; + let write = r.flags & (O_WRITE | O_APPEND) != 0; + client.fids.insert(fid, Fid { node, write, append: r.flags & O_APPEND != 0 }); + Ok(Answer::Reply(Reply { value: fid, ..stat_reply(meta) })) + } + CLOSE => { + let client = self.clients.get_mut(&id).expect("pumped"); + let fid = client.fids.remove(&r.fid).ok_or(SyscallError::InvalidArgument)?; + self.volume.close(fid.node); + Ok(Answer::Reply(Reply::ok())) + } + READ => { + let node = self.fid(id, r.fid)?.node; + let len = (r.len as usize).min(WINDOW_BYTES); + let mut buf = vec![0u8; len]; + let n = self.volume.read(node, r.offset, &mut buf)?; + self.put(id, &buf[..n]); + Ok(Answer::Reply(Reply { value: n as u64, ..Reply::ok() })) + } + WRITE => { + let (node, write, append) = { + let f = self.fid(id, r.fid)?; + (f.node, f.write, f.append) + }; + if !write { + return Err(SyscallError::PermissionDenied); + } + let len = r.len as usize; + if len > WINDOW_BYTES { + return Err(SyscallError::InvalidArgument); + } + let mut data = vec![0u8; len]; + window_take(window_of(self.clients[&id].window.as_ref().expect("lent")), 0, &mut data); + let at = if append { self.volume.node_meta(node)?.size } else { r.offset }; + if at.checked_add(len as u64).is_none_or(|end| end > MAX_FILE_BYTES) { + return Err(SyscallError::InvalidArgument); + } + self.volume.write(node, at, &data)?; + self.dirtied(); + Ok(Answer::Reply(Reply { value: len as u64, value2: at + len as u64, ..Reply::ok() })) + } + FSTAT => { + let node = self.fid(id, r.fid)?.node; + Ok(Answer::Reply(stat_reply(self.volume.node_meta(node)?))) + } + TRUNCATE => { + let f = self.fid(id, r.fid)?; + if !f.write { + return Err(SyscallError::PermissionDenied); + } + let node = f.node; + if r.offset > MAX_FILE_BYTES { + return Err(SyscallError::InvalidArgument); + } + self.volume.truncate(node, r.offset)?; + self.dirtied(); + Ok(Answer::Reply(Reply::ok())) + } + FSYNC => { + self.fid(id, r.fid)?; + self.sync() + } + SYNC => self.sync(), + STREAM => { + let f = self.fid(id, r.fid)?; + if !f.write { + return Err(SyscallError::PermissionDenied); + } + let node = f.node; + if self.streams.len() >= MAX_STREAMS { + return Err(SyscallError::ResourceExhausted); + } + let (read, write) = toyos::pipe_pair()?; + self.volume.hold(node); + let sid = self.next_stream; + self.next_stream += 1; + self.streams.insert(sid, Stream { pipe: read, node, offset: r.offset }); + Ok(Answer::WithHandle(Reply::ok(), write.into_raw())) + } + STAT | LSTAT => { + let rel = self.path(id, 0, r.len)?; + match self.resolve(id, &rel, op == STAT)? { + Resolved::Absolute(p) => Ok(self.link(id, &p)), + Resolved::Path(p) => Ok(Answer::Reply(stat_reply(self.volume.lstat(&p)?))), + } + } + READDIR => { + let rel = self.path(id, 0, r.len)?; + let dir = match self.resolve(id, &rel, true)? { + Resolved::Absolute(p) => return Ok(self.link(id, &p)), + Resolved::Path(p) => p, + }; + let entries = self.volume.list(&dir)?; + let mut out = Vec::new(); + let mut entry = vec![0u8; 11 + MAX_PATH]; + for (name, meta) in &entries { + let n = encode_entry(&mut entry, meta.kind.wire(), meta.size, name) + .ok_or(SyscallError::InvalidArgument)?; + out.extend_from_slice(&entry[..n]); + } + // Refused rather than cut short: a listing that stops early is a + // confidently wrong answer to a caller deleting a tree. + if out.len() > WINDOW_BYTES { + return Err(SyscallError::ResourceExhausted); + } + self.put(id, &out); + Ok(Answer::Reply(Reply { value: out.len() as u64, ..Reply::ok() })) + } + READLINK => { + let rel = self.path(id, 0, r.len)?; + let path = match self.name(id, &rel)? { + Ok(p) => p, + Err(link) => return Ok(link), + }; + let target = self.volume.read_link(&path)?; + self.put(id, target.as_bytes()); + Ok(Answer::Reply(Reply { value: target.len() as u64, ..Reply::ok() })) + } + MKDIR | RMDIR | UNLINK => { + let rel = self.path(id, 0, r.len)?; + if rel.is_empty() { + return Err(if op == MKDIR { SyscallError::AlreadyExists } else { SyscallError::PermissionDenied }); + } + let path = match self.name(id, &rel)? { + Ok(p) => p, + Err(link) => return Ok(link), + }; + match op { + MKDIR => self.volume.mkdir(&path)?, + RMDIR => self.volume.rmdir(&path)?, + _ => self.volume.unlink(&path)?, + } + self.dirtied(); + Ok(Answer::Reply(Reply::ok())) + } + RENAME => { + let from_rel = self.path(id, 0, r.len)?; + let to_rel = self.path(id, r.len, r.len2)?; + if from_rel.is_empty() || to_rel.is_empty() { + return Err(SyscallError::PermissionDenied); + } + let (Resolved::Path(from), Resolved::Path(to)) = + (self.resolve(id, &from_rel, false)?, self.resolve(id, &to_rel, false)?) + else { + // A rename through a link that leaves this directory is a + // rename across two, which no one directory can do. + return Err(SyscallError::NotSupported); + }; + self.volume.rename(&from, &to)?; + self.dirtied(); + Ok(Answer::Reply(Reply::ok())) + } + SYMLINK => { + let link_rel = self.path(id, 0, r.len)?; + let target = self.path(id, r.len, r.len2)?; + if target.is_empty() || link_rel.is_empty() { + return Err(SyscallError::InvalidArgument); + } + let Resolved::Path(link) = self.resolve(id, &link_rel, false)? else { + return Err(SyscallError::NotSupported); + }; + self.volume.symlink(&link, &target)?; + self.dirtied(); + Ok(Answer::Reply(Reply::ok())) + } + _ => Ok(Answer::Drop("it asked for an operation this protocol does not have")), + } + } + + fn sync(&mut self) -> Result { + self.volume.sync()?; + self.dirty_since = None; + Ok(Answer::Reply(Reply::ok())) + } + + fn link(&self, id: u64, path: &str) -> Answer { + if path.len() > MAX_PATH { + return Answer::Reply(Reply::refused(SyscallError::InvalidArgument)); + } + self.put(id, path.as_bytes()); + Answer::Link(path.len()) + } + + /// What a stream's writer has put in the pipe, appended to its file. + fn drain(&mut self, sid: u64) { + let mut buf = vec![0u8; 64 * 1024]; + loop { + let Some(stream) = self.streams.get_mut(&sid) else { return }; + match stream.pipe.read_nonblock(&mut buf) { + Ok(0) | Err(SyscallError::Gone) => break, + Ok(n) => { + let (node, at) = (stream.node, stream.offset); + stream.offset += n as u64; + if let Err(e) = self.volume.write(node, at, &buf[..n]) { + println!("fsd: a stream's write failed ({e:?}); the stream is ended"); + break; + } + self.dirtied(); + } + Err(SyscallError::WouldBlock) => return, + Err(e) => { + println!("fsd: a stream's pipe would not read ({e:?}); the stream is ended"); + break; + } + } + } + if let Some(stream) = self.streams.remove(&sid) { + self.volume.close(stream.node); + } + } +} diff --git a/userland/fsd/src/resolve.rs b/userland/fsd/src/resolve.rs new file mode 100644 index 00000000000..bc3686b024f --- /dev/null +++ b/userland/fsd/src/resolve.rs @@ -0,0 +1,195 @@ +//! Every path a client names, resolved inside the directory its connection was +//! given, and nowhere else. +//! +//! **The root is a floor, not a starting point.** A connection is bound to one +//! directory of the volume (its capability's); a client's path is canonical +//! and relative to it (`toyos::fs::canonical`), and each component is looked up +//! in turn. A symlink met on the way is expanded where it stands: a relative +//! target is read against the link's own directory, and a `..` that would +//! climb above the connection's root is refused — [`Escape`] — rather than +//! clamped, because a clamped path is a different file from the one the link +//! named. An absolute target is not this server's to resolve: it names a path +//! in the *client's* table, which may hold directories this connection does +//! not, and the client resolves it again there ([`Resolved::Absolute`]). +//! +//! **One resolution sees one volume.** The server is one thread and a request +//! is answered before the next is read, so no rename can land between two +//! component lookups of the same path. +//! +//! [`Escape`]: Refusal::Escape + +use crate::volume::join; + +/// The most symlinks one resolution expands, as Linux's `MAXSYMLINKS`. +pub const MAX_LINKS: u32 = 40; + +#[derive(Clone, Debug, PartialEq, Eq)] +pub enum Resolved { + /// The volume path the request acts on. + Path(String), + /// The path met an absolute symlink: resolve this, in the client's table. + Absolute(String), +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Refusal { + /// A link's `..` climbed above the connection's root. + Escape, + /// More than [`MAX_LINKS`] links, or a cycle. + Loop, + /// A name on the way is not a directory, or a link target is empty. + NotFound, + /// The volume would not answer a lookup. + Io, +} + +/// What a lookup of one volume path found: a symlink's target, or anything +/// else (a file, a directory, or nothing, which the operation judges). +pub enum Found { + Link(String), + Other, +} + +/// Resolve `rel` under `root`. `follow_last` follows a symlink in the final +/// component too; an operation on the link itself (`lstat`, `unlink`, +/// `readlink`, a `rename`'s ends) passes `false`. +pub fn resolve( + root: &str, + rel: &str, + follow_last: bool, + lookup: &mut dyn FnMut(&str) -> Result, +) -> Result { + // What is resolved so far, as components under `root`, and what is left. + let mut done: Vec = Vec::new(); + let mut left: Vec = rel.split('/').filter(|c| !c.is_empty()).rev().map(String::from).collect(); + let mut links = 0; + while let Some(component) = left.pop() { + match component.as_str() { + "." => continue, + ".." => { + // Only a link's target carries one: the wire refuses them. + if done.pop().is_none() { + return Err(Refusal::Escape); + } + continue; + } + _ => {} + } + done.push(component); + let last = left.is_empty(); + if last && !follow_last { + break; + } + let path = join(root, &done.join("/")); + let target = match lookup(&path).map_err(|()| Refusal::Io)? { + Found::Other => continue, + Found::Link(target) => target, + }; + links += 1; + if links > MAX_LINKS { + return Err(Refusal::Loop); + } + if target.is_empty() { + return Err(Refusal::NotFound); + } + done.pop(); + if let Some(absolute) = target.strip_prefix('/') { + let mut rest: Vec<&str> = left.iter().rev().map(String::as_str).collect(); + rest.insert(0, absolute.trim_end_matches('/')); + let joined = rest.into_iter().filter(|p| !p.is_empty()).collect::>().join("/"); + return Ok(Resolved::Absolute(format!("/{joined}"))); + } + for part in target.split('/').filter(|c| !c.is_empty()).rev() { + left.push(part.to_string()); + } + } + Ok(Resolved::Path(join(root, &done.join("/")))) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::collections::BTreeMap; + + /// A volume of symlinks: every name not in the map is "something else". + fn vol(links: &[(&str, &str)]) -> BTreeMap { + links.iter().map(|(a, b)| (a.to_string(), b.to_string())).collect() + } + + fn run(links: &BTreeMap, root: &str, rel: &str, follow: bool) -> Result { + resolve(root, rel, follow, &mut |path| { + Ok(links.get(path).map_or(Found::Other, |t| Found::Link(t.clone()))) + }) + } + + fn path(p: &str) -> Result { + Ok(Resolved::Path(p.to_string())) + } + + #[test] + fn a_plain_path_lands_under_the_root() { + let v = vol(&[]); + assert_eq!(run(&v, "home", "toy/notes.txt", true), path("home/toy/notes.txt")); + assert_eq!(run(&v, "", "boot.log", true), path("boot.log")); + assert_eq!(run(&v, "home", "", true), path("home")); + } + + #[test] + fn a_relative_link_is_read_against_its_own_directory() { + let v = vol(&[("home/toy/docs", "Documents"), ("home/toy/up", "../toy/Documents")]); + assert_eq!(run(&v, "home", "toy/docs/a", true), path("home/toy/Documents/a")); + assert_eq!(run(&v, "home", "toy/up/b", true), path("home/toy/Documents/b")); + } + + /// The escape suite's first half: every `..` the literature names that + /// would leave the connection's directory, through a link, is refused. + #[test] + fn no_link_climbs_out_of_the_root() { + let v = vol(&[ + ("home/toy/out", "../../state/sshd"), + ("home/toy/deep", "a/../../../etc"), + ("home/up", ".."), + ("home/dotdot", "../home/../../x"), + ("home/toy/chain1", "chain2"), + ("home/toy/chain2", "../../apps"), + ]); + for rel in ["toy/out", "toy/out/authorized_keys", "toy/deep", "up", "up/x", "dotdot", "toy/chain1/y"] { + assert_eq!(run(&v, "home", rel, true), Err(Refusal::Escape), "{rel}"); + } + // A link to its own root's parent from the root itself. + assert_eq!(run(&vol(&[("x", "..")]), "", "x", true), Err(Refusal::Escape)); + } + + #[test] + fn a_link_that_stays_inside_by_going_up_and_back_resolves() { + let v = vol(&[("home/toy/a/back", "../b")]); + assert_eq!(run(&v, "home", "toy/a/back/f", true), path("home/toy/b/f")); + } + + #[test] + fn an_absolute_link_is_handed_back_with_the_rest_of_the_path() { + let v = vol(&[("home/toy/sys", "/system/share")]); + assert_eq!( + run(&v, "home", "toy/sys/fonts/a.ttf", true), + Ok(Resolved::Absolute("/system/share/fonts/a.ttf".to_string())) + ); + assert_eq!(run(&v, "home", "toy/sys", true), Ok(Resolved::Absolute("/system/share".to_string()))); + } + + #[test] + fn the_last_link_is_left_alone_when_asked() { + let v = vol(&[("home/l", "../../etc")]); + assert_eq!(run(&v, "home", "l", false), path("home/l")); + } + + #[test] + fn a_cycle_is_refused_and_not_followed_for_ever() { + let v = vol(&[("a", "b"), ("b", "a")]); + assert_eq!(run(&v, "", "a", true), Err(Refusal::Loop)); + } + + #[test] + fn an_empty_target_names_nothing() { + assert_eq!(run(&vol(&[("a", "")]), "", "a", true), Err(Refusal::NotFound)); + } +} diff --git a/userland/fsd/src/volume.rs b/userland/fsd/src/volume.rs new file mode 100644 index 00000000000..5cd8f066cbe --- /dev/null +++ b/userland/fsd/src/volume.rs @@ -0,0 +1,114 @@ +//! What a volume answers, whatever its format: every path here is the +//! volume's own — `/`-separated, no leading `/`, `""` its root — and already +//! resolved ([`crate::resolve`]), so a volume never follows a link. +//! +//! **An open file is a [`Node`]**, one per path while any client holds it +//! open, so two opens of one file write one set of pages and see one length. A +//! node whose file was unlinked or renamed over answers `Gone` from then on: +//! its blocks may be another file's. + +use toyos_abi::syscall::SyscallError; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Kind { + File, + Dir, + Symlink, +} + +impl Kind { + pub fn wire(self) -> u64 { + match self { + Kind::File => toyos::fs::KIND_FILE, + Kind::Dir => toyos::fs::KIND_DIR, + Kind::Symlink => toyos::fs::KIND_SYMLINK, + } + } +} + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Meta { + pub kind: Kind, + pub size: u64, + /// Nanoseconds since the Unix epoch. + pub mtime: u64, +} + +/// How an open treats what is, or is not, at its path. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct OpenHow { + pub create: bool, + pub create_new: bool, + pub truncate: bool, +} + +/// An open file, while any client holds it. +pub type Node = u64; + +pub trait Volume { + /// Whether anything here may be changed. + fn writable(&self) -> bool; + + /// What `path` is, without following it. + fn lstat(&mut self, path: &str) -> Result; + + /// What the symlink at `path` holds. + fn read_link(&mut self, path: &str) -> Result; + + /// The entries directly under `dir`: name and what it is. + fn list(&mut self, dir: &str) -> Result, SyscallError>; + + fn open(&mut self, path: &str, how: OpenHow) -> Result; + + /// One holder gone; the node goes with its last. + fn close(&mut self, node: Node); + + /// Hold the node once more, for a second file id or a stream. + fn hold(&mut self, node: Node); + + fn node_meta(&mut self, node: Node) -> Result; + + /// Up to `out.len()` bytes at `offset`; 0 at or past the end. + fn read(&mut self, node: Node, offset: u64, out: &mut [u8]) -> Result; + + fn write(&mut self, node: Node, offset: u64, data: &[u8]) -> Result<(), SyscallError>; + + fn truncate(&mut self, node: Node, size: u64) -> Result<(), SyscallError>; + + fn mkdir(&mut self, path: &str) -> Result<(), SyscallError>; + + fn rmdir(&mut self, path: &str) -> Result<(), SyscallError>; + + /// A file or a symlink; a directory is `rmdir`'s. + fn unlink(&mut self, path: &str) -> Result<(), SyscallError>; + + /// Replaces a file or symlink at `to`; never a directory. + fn rename(&mut self, from: &str, to: &str) -> Result<(), SyscallError>; + + fn symlink(&mut self, path: &str, target: &str) -> Result<(), SyscallError>; + + /// Everything written, and every open file's length, durable. + fn sync(&mut self) -> Result<(), SyscallError>; + + /// One line of what the volume and its cache have done, for a log. + fn describe(&self) -> String; +} + +/// The parent of `path`, `""` for a name at the root. +pub fn parent(path: &str) -> &str { + path.rsplit_once('/').map_or("", |(dir, _)| dir) +} + +/// `dir` and `name` as one path. +pub fn join(dir: &str, name: &str) -> String { + match (dir.is_empty(), name.is_empty()) { + (true, _) => name.to_string(), + (false, true) => dir.to_string(), + (false, false) => format!("{dir}/{name}"), + } +} + +/// Nanoseconds since the Unix epoch, now, to the second the clock reads. +pub fn now_nanos() -> u64 { + toyos_abi::syscall::clock_epoch().map_or(0, |secs| secs.saturating_mul(1_000_000_000)) +} diff --git a/userland/init/Cargo.toml b/userland/init/Cargo.toml index 84b7a38fa87..846d5676975 100644 --- a/userland/init/Cargo.toml +++ b/userland/init/Cargo.toml @@ -14,3 +14,5 @@ toyos-swap = { path = "../../toyos-swap" } toyos-update = { path = "../../toyos-update" } # The partition types that name the running ROOT and the slot table. toyos-gpt = { path = "../../toyos-gpt" } +# The block service's port name, which marks a storage row. +toyos-blockring = { path = "../../toyos-blockring" } diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index f8e66db7df3..0018b79ce11 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -39,7 +39,7 @@ //! the caller's direct spawn carries in place of its own. use std::cell::{Cell, RefCell}; -use std::collections::BTreeMap; +use std::collections::{BTreeMap, VecDeque}; use std::os::toyos::process::{ChildExt, CommandExt}; use std::process::{Child, Command, Stdio}; use std::sync::{Arc, Mutex}; @@ -50,6 +50,7 @@ use toyos_swap::{Refusal, Request as SwapRequest, Word}; use toyos_manifest::package::{self, Package}; use toyos_manifest::{Manifest, Program}; use toyos::endow::Endowments; +use toyos::fs::CAPABILITY_PREFIX; use toyos::ipc::{self, Connection, RxStep}; use toyos::launch::{self, Request}; use toyos::namespace::{self, Namespace}; @@ -136,7 +137,8 @@ const TOKEN_ACCEPTOR: u64 = 0; const TOKEN_SWAP_ACCEPTOR: u64 = 1; const TOKEN_HANGUP: u64 = 2; const TOKEN_POWER_ACCEPTOR: u64 = 3; -const TOKEN_PENDING_BASE: u64 = 4; +const TOKEN_WAKE: u64 = 4; +const TOKEN_PENDING_BASE: u64 = 5; /// How long a stop waits for `logd` to say the log is whole. /// @@ -322,10 +324,8 @@ fn main() { .take(SYSCAP_LABEL) .expect("init: the kernel spawns this program holding the system capability"); - make_session_home(); - - let text = std::fs::read_to_string(toyos_manifest::GUEST_PATH) - .unwrap_or_else(|e| panic!("init: cannot read {}: {e}", toyos_manifest::GUEST_PATH)); + let text = read_manifest() + .unwrap_or_else(|e| panic!("init: cannot read {}: {e:?}", toyos_manifest::GUEST_PATH)); let system = toyos_manifest::parse(&text); // Before anything is spawned, and for every `serves` name in the manifest @@ -345,30 +345,97 @@ fn main() { connectors.insert(name, connector); } + // One port per directory a file server serves, made here for the same + // reason: a program's first open works whether or not its server runs yet. + let mut dirs: BTreeMap = BTreeMap::new(); + let mut dir_acceptors: BTreeMap = BTreeMap::new(); + for role in system.programs.iter().flat_map(|p| p.roles.iter()) { + for dir in toyos_manifest::role_dirs(role).expect("init: the build refuses a role it does not know") { + let name = format!("{CAPABILITY_PREFIX}{dir}"); + let (acceptor, connector) = + port::create().unwrap_or_else(|e| panic!("init: no port for `{name}`: {e:?}")); + dir_acceptors.insert(name.clone(), acceptor); + dirs.insert(name, connector); + } + } + // init's own files are resolved through the same directories as every + // program's, and it is the one process nobody endows a namespace: std + // resolves through this one, and the stop's syncs through the second. + let files: &'static Namespace = { + let build = || { + let mut builder = namespace::build(); + for (name, connector) in &dirs { + builder = builder.add(name, connector); + } + builder.finish().expect("init: no namespace for its own files") + }; + if let Err(e) = std::os::toyos::fs::adopt_namespace(build().into_raw().0) { + panic!("init: std would not take its namespace ({e}), and init is endowed none"); + } + Box::leak(Box::new(build())) + }; + + let (wake_read, wake_write) = toyos::pipe_pair().expect("init: no pipe to hear a service end"); + let wake = wake_write.into_raw(); + // Kept for the machine's life: init is the only thing that can kill a - // daemon, and there is no other way back to a process it started. + // daemon, and there is no other way back to a process it started. The + // storage services go first: every other start may make a directory, and + // a directory is a file server's. let mut services: Vec = Vec::new(); - for name in &system.start { + let storage_first = system + .start + .iter() + .filter(|n| system.program(n).is_some_and(is_storage)) + .chain(system.start.iter().filter(|n| system.program(n).is_none_or(|p| !is_storage(p)))); + let mut homes_made = false; + for name in storage_first { let program = system .program(name) .unwrap_or_else(|| panic!("init: [boot] start names `{name}`, which is not declared")); - let kept = program - .serves - .iter() - .map(|served| { - let acceptor = acceptors.remove(served.as_str()).unwrap_or_else(|| { - panic!("init: `{}` has already been given the `{served}` acceptor", program.name) - }); - (served.clone(), acceptor) - }) - .collect(); - let mut service = Service::new(program, kept); - if let Err(e) = - service.spawn(&program.path, &[], &system, &syscap, &connectors, &mut log) - { - panic!("init: cannot start {}: {e}", program.name); + if !is_storage(program) && !homes_made { + make_session_home(); + homes_made = true; + } + let instances: Vec> = match program.roles.is_empty() { + true => vec![None], + false => program.roles.iter().map(|r| Some(r.as_str())).collect(), + }; + for role in instances { + let kept: Vec<(String, Acceptor)> = match role { + None => program + .serves + .iter() + .map(|served| { + let acceptor = acceptors.remove(served.as_str()).unwrap_or_else(|| { + panic!("init: `{}` has already been given the `{served}` acceptor", program.name) + }); + (served.clone(), acceptor) + }) + .collect(), + Some(role) => toyos_manifest::role_dirs(role) + .expect("known role") + .iter() + .map(|dir| { + let name = format!("{CAPABILITY_PREFIX}{dir}"); + let acceptor = dir_acceptors.remove(&name).unwrap_or_else(|| { + panic!("init: the `{role}` role is served twice") + }); + (name, acceptor) + }) + .collect(), + }; + let mut service = Service::new(program, role, kept, wake); + if let Err(e) = + service.spawn(&program.path, &[], &system, &syscap, &connectors, &dirs, &mut log) + { + panic!("init: cannot start {}: {e}", program.name); + } + services.push(service); } - services.push(service); + } + if !homes_made { + make_session_home(); } // Nothing else holds a `serves` acceptor that has not been launched yet, so @@ -383,10 +450,18 @@ fn main() { let power = acceptors .remove(power::PORT) .expect("init: the manifest declares init serves `power`"); - let mut init = Init { system: &system, syscap: &syscap, acceptors, connectors, services, log }; + let mut init = + Init { system: &system, syscap: &syscap, acceptors, connectors, dirs, files, services, log, wake: wake_read }; init.serve_forever(&launcher, &swap, &power); } +/// A row whose process the machine's files depend on: a block service, or a +/// file server. Started before every other row, and never made a home, which +/// would be a directory on the volume it serves. +fn is_storage(program: &Program) -> bool { + !program.roles.is_empty() || program.serves.iter().any(|s| s == toyos_blockring::PORT) +} + /// Everything init's loop acts on, for the machine's life. struct Init<'a> { system: &'a Manifest, @@ -395,16 +470,25 @@ struct Init<'a> { /// takes by move. acceptors: BTreeMap<&'a str, Acceptor>, connectors: BTreeMap<&'a str, Connector>, - /// What `[boot] start` named, in that order. + /// Every directory capability, by its namespace name: what each program's + /// view is built from. + dirs: BTreeMap, + /// The same directories as a namespace of init's own, for the stop's syncs. + files: &'static Namespace, + /// What `[boot] start` named, a file server once per role. services: Vec>, /// Where a service init (re)starts writes its stdout and stderr, whichever /// process ends up holding its row. log: Log, + /// Readable when a `restart` service has ended and is owed a new process. + wake: Pipe, } /// One program init started at boot, and the ports it serves. struct Service<'a> { program: &'a Program, + /// The file-server role this process serves, for a row with `roles`. + role: Option<&'a str>, /// The binary the running process was started from: the row's path, or a /// swap's installed one. path: String, @@ -414,28 +498,51 @@ struct Service<'a> { /// started in its place is owed. devices: Vec, kept: Arc>, + /// When each recent start after an end happened, for [`toyos_manifest::RESTARTS`]. + restarts: VecDeque, } /// What a service's waiter thread shares with the loop. struct Kept { /// The acceptors of every name the service serves. Empty once the service - /// has ended with no swap owning it: the port is then closed for good. + /// has ended with no swap owning it and no restart owed: the port is then + /// closed for good. acceptors: Vec<(String, Acceptor)>, /// Counts the service's starts, so a waiter answers only for its own. generation: u64, /// A swap owns the service: its process ending is expected and the port /// stays open for the next one. swapping: bool, + /// The row says init starts it again when it ends. + restart: bool, + /// It ended, and the loop owes it a new process on the same ports. + ended: bool, + /// The write end of [`Init::wake`]. + wake: toyos::RawHandle, } impl<'a> Service<'a> { - fn new(program: &'a Program, acceptors: Vec<(String, Acceptor)>) -> Self { + fn new( + program: &'a Program, + role: Option<&'a str>, + acceptors: Vec<(String, Acceptor)>, + wake: toyos::RawHandle, + ) -> Self { Self { program, + role, path: program.path.clone(), child: None, devices: Vec::new(), - kept: Arc::new(Mutex::new(Kept { acceptors, generation: 0, swapping: false })), + kept: Arc::new(Mutex::new(Kept { + acceptors, + generation: 0, + swapping: false, + restart: program.restart, + ended: false, + wake, + })), + restarts: VecDeque::new(), } } @@ -450,6 +557,7 @@ impl<'a> Service<'a> { system: &Manifest, syscap: &SysCap, connectors: &BTreeMap<&str, Connector>, + dirs: &BTreeMap, log: &mut Log, ) -> std::io::Result { // **Checked again at every start, not only on arrival**: the installed @@ -470,6 +578,7 @@ impl<'a> Service<'a> { 0 => Served::Keep(&kept.acceptors), _ => Served::Restart { acceptors: &kept.acceptors, owed }, }; + let storage = storage_endowment(self.program, self.role, syscap); let (child, devices) = start( Command::new(path), self.program, @@ -477,10 +586,13 @@ impl<'a> Service<'a> { syscap, served, connectors, + dirs, &[], + storage, Output::Boot(log), )?; kept.generation += 1; + kept.ended = false; if !kept.acceptors.is_empty() { let process = toyos_abi::syscall::dup(toyos::RawHandle(child.as_raw_handle())) .unwrap_or_else(|e| panic!("init: {}'s process handle: {e:?}", self.program.name)); @@ -500,10 +612,19 @@ impl<'a> Service<'a> { fn pid(&self) -> Option { self.child.as_ref().map(Child::id) } + + /// The name its lines and init's say it under: a file server's role + /// beside its row's. + fn label(&self) -> String { + match self.role { + Some(role) => format!("{} {role}", self.program.name), + None => self.program.name.clone(), + } + } } /// Wait for one start of a service to end, and close its ports if nothing was -/// expecting it to. +/// expecting it to, or wake the loop to start it again if its row says so. /// /// **A thread, because the kernel answers a process's end to a wait and to /// nothing a poll can watch.** It parks in the kernel for the process's life @@ -513,7 +634,13 @@ fn close_when_it_ends(kept: &Mutex, generation: u64, process: toyos::RawHa toyos_abi::syscall::close(process); let mut kept = kept.lock().expect("init: a service's state is poisoned"); if kept.generation == generation && !kept.swapping { - kept.acceptors.clear(); + if kept.restart { + kept.ended = true; + // A full pipe is a wake already pending, which is all this is. + let _ = toyos_abi::syscall::write_nonblock(kept.wake, &[1]); + } else { + kept.acceptors.clear(); + } } } @@ -566,7 +693,7 @@ impl<'a> Init<'a> { /// service it has just killed to finish ending, which is the kernel's /// teardown and no client's. fn serve_forever(&mut self, launcher: &Acceptor, swap: &Acceptor, power: &Acceptor) -> ! { - let poller = Poller::new(4 + MAX_PENDING_LAUNCHES as u32); + let poller = Poller::new(5 + MAX_PENDING_LAUNCHES as u32); let mut pending: Vec = Vec::new(); let mut flight: Option = None; let mut ready: Vec = Vec::new(); @@ -574,6 +701,7 @@ impl<'a> Init<'a> { poller.watch(launcher, READABLE, TOKEN_ACCEPTOR); poller.watch(swap, READABLE, TOKEN_SWAP_ACCEPTOR); poller.watch(power, READABLE, TOKEN_POWER_ACCEPTOR); + poller.watch(&self.wake, READABLE, TOKEN_WAKE); if let Some(Flight { phase: Phase::Answered { conn, .. }, .. }) = &flight { poller.watch(conn, READABLE, TOKEN_HANGUP); } @@ -666,6 +794,7 @@ impl<'a> Init<'a> { self.syscap, &mut self.acceptors, &self.connectors, + &self.dirs, &mut self.log, ), Port::Power => self.stop(&p.conn, msg_type), @@ -695,6 +824,12 @@ impl<'a> Init<'a> { pending.retain(|p| now.duration_since(p.since) < HANDSHAKE_TIMEOUT); flight = flight.and_then(|f| self.advance(f, ready.contains(&TOKEN_HANGUP))); + + if ready.contains(&TOKEN_WAKE) { + let mut sink = [0u8; 64]; + while matches!(self.wake.read_nonblock(&mut sink), Ok(n) if n > 0) {} + self.restart_ended(); + } } } @@ -827,7 +962,7 @@ impl<'a> Init<'a> { let _ = old.wait(); } let started = - service.spawn(&path, &owed, self.system, self.syscap, &self.connectors, &mut self.log); + service.spawn(&path, &owed, self.system, self.syscap, &self.connectors, &self.dirs, &mut self.log); match started { Ok(pid) => { swapped( @@ -908,6 +1043,7 @@ impl<'a> Init<'a> { }; say!("{STOPPING} ({how:?})"); self.log.flush(); + self.sync_files(); let refused = match how { Stop::Reboot => self.syscap.reboot(), Stop::Shutdown => self.syscap.shutdown(), @@ -917,12 +1053,100 @@ impl<'a> Init<'a> { let _ = conn.try_send_bytes(power::MSG_REFUSED, &refused.to_u64().to_le_bytes()); } + /// Have every writable file server make its volume durable, each bounded + /// by [`FLUSH_BOUND`]: the kernel's stop waits for no process, so what a + /// server holds and has not written is lost unless it is asked first. + /// `/log` is `logd`'s flush's, made durable before this. + fn sync_files(&self) { + let roles = self.system.programs.iter().flat_map(|p| p.roles.iter()); + for role in roles.filter(|r| *r != "log" && *r != "boot") { + let Some(dir) = toyos_manifest::role_dirs(role).and_then(|dirs| dirs.first()) else { continue }; + let name = format!("{CAPABILITY_PREFIX}{dir}"); + let (done_tx, done_rx) = std::sync::mpsc::channel(); + let (asked, names) = (name.clone(), self.files); + let spawned = std::thread::Builder::new().name(format!("sync-{role}")).spawn(move || { + let answer = toyos::fs::Dir::connect(names, &asked) + .and_then(|mut dir| { + dir.sync().map_err(|e| match e { + toyos::fs::Refused::Error(e) => e, + toyos::fs::Refused::Link(_) => SyscallError::Unknown, + }) + }); + let _ = done_tx.send(answer); + }); + let Ok(syncer) = spawned else { + say!("init: power: no thread to sync {name}; its unsynced writes are lost with the stop"); + continue; + }; + let answered = done_rx.recv_timeout(FLUSH_BOUND); + // Joined once it has answered, so the stop never counts a thread + // that was on its way out. + if answered.is_ok() { + syncer.join().expect("init: a sync thread panicked"); + } + match answered { + Ok(Ok(())) => {} + Ok(Err(e)) => say!("init: power: {name}'s server would not sync ({e:?})"), + Err(_) => say!( + "init: power: {name}'s server did not sync in {} ms; stopping without it", + FLUSH_BOUND.as_millis() + ), + } + } + } + + /// Start again every `restart` service that ended, on the ports it kept, + /// unless it has ended [`toyos_manifest::RESTARTS`] times inside + /// [`toyos_manifest::RESTART_WINDOW_SECS`]: then its ports close, and a + /// client's next connection is answered `Gone`. + fn restart_ended(&mut self) { + let window = Duration::from_secs(toyos_manifest::RESTART_WINDOW_SECS); + for index in 0..self.services.len() { + let ended = self.services[index].kept.lock().expect("init: a service's state is poisoned").ended; + if !ended { + continue; + } + let service = &mut self.services[index]; + let label = service.label(); + let now = Instant::now(); + while service.restarts.front().is_some_and(|at| now.duration_since(*at) > window) { + service.restarts.pop_front(); + } + let pid = service.pid().map_or(0, |p| p); + if let Some(mut old) = service.child.take() { + let _ = old.wait(); + } + if service.restarts.len() as u32 >= toyos_manifest::RESTARTS { + let mut kept = service.kept.lock().expect("init: a service's state is poisoned"); + kept.ended = false; + kept.acceptors.clear(); + say!( + "init: {label} ended {} times in {} s; its ports are closed and its clients answered Gone", + toyos_manifest::RESTARTS + 1, + toyos_manifest::RESTART_WINDOW_SECS + ); + continue; + } + service.restarts.push_back(now); + let (path, owed) = (service.path.clone(), service.devices.clone()); + match service.spawn(&path, &owed, self.system, self.syscap, &self.connectors, &self.dirs, &mut self.log) { + Ok(new) => say!("init: {label} (pid {pid}) ended; started again as pid {new}"), + Err(e) => { + let mut kept = service.kept.lock().expect("init: a service's state is poisoned"); + kept.ended = false; + kept.acceptors.clear(); + say!("init: {label} ended and would not start again ({e}); its ports are closed"); + } + } + } + } + /// Start the binary a failed swap replaced, or close the service's ports /// when that will not start either. fn restore(&mut self, index: usize, previous: &str, owed: &[String]) { let service = &mut self.services[index]; let name = service.program.name.clone(); - match service.spawn(previous, owed, self.system, self.syscap, &self.connectors, &mut self.log) { + match service.spawn(previous, owed, self.system, self.syscap, &self.connectors, &self.dirs, &mut self.log) { Ok(pid) => { service.kept.lock().expect("init: a service's state is poisoned").swapping = false; swapped(&self.log, &name, Word::Restored, &format!("{previous} as pid {pid}")); @@ -991,6 +1215,7 @@ fn serve_launch<'a>( syscap: &SysCap, acceptors: &mut BTreeMap<&'a str, Acceptor>, connectors: &BTreeMap<&str, Connector>, + dirs: &BTreeMap, log: &mut Log, ) { if msg_type != launch::MSG_LAUNCH { @@ -1106,7 +1331,9 @@ fn serve_launch<'a>( syscap, Served::Move(acceptors), connectors, + dirs, &extras, + Storage::default(), Output::Launch { log, slots: &caller_slots }, ); match started { @@ -1368,16 +1595,19 @@ fn start<'a>( syscap: &SysCap, served: Served<'_, 'a>, connectors: &BTreeMap<&str, Connector>, + dirs: &BTreeMap, extras: &[(&str, Connector)], + storage: Storage, output: Output<'_>, ) -> std::io::Result<(Child, Vec)> { command.args(&program.args); + command.args(&storage.args); let booting = matches!(output, Output::Boot(_)); // **Set here, over whatever a launching caller carried**: the row decides // where a program's home is, and a service's is made before it first runs. let home = program.home(); - if program.service { + if program.service && !is_storage(program) { if let Err(e) = make_dir(&home) { say!("init: {}: {home} could not be made: {e}", program.name); } @@ -1395,7 +1625,12 @@ fn start<'a>( // acceptor is gone can never be served again. let mut taken: Vec<(&'a str, Acceptor)> = Vec::new(); - if let Some(ns) = build_namespace(program, system, connectors, extras)? { + for (label, claim) in storage.claims { + let raw = claim.into_raw(); + command.endow(&label, raw.0); + held.0.push(raw); + } + if let Some(ns) = build_namespace(program, system, connectors, dirs, extras)? { let raw = ns.into_raw(); command.endow(SVC_LABEL, raw.0); held.0.push(raw); @@ -1447,10 +1682,13 @@ fn start<'a>( given_back = Some(acceptors); } Served::Keep(kept) | Served::Restart { acceptors: kept, .. } => { - for name in &program.serves { - let (_, acceptor) = kept.iter().find(|(kept, _)| kept == name).ok_or_else(|| { - std::io::Error::other(format!("the `{name}` port is closed for good")) - })?; + // Every acceptor the service keeps: its row's `serves`, or the + // directories of a file server's role. None left is a service + // whose ports closed for good. + if kept.is_empty() && (!program.serves.is_empty() || !program.roles.is_empty()) { + return Err(std::io::Error::other("its ports are closed for good")); + } + for (name, acceptor) in kept { let handed = toyos_abi::syscall::dup(acceptor.as_handle()).map_err(|e| { std::io::Error::other(format!("the `{name}` acceptor did not duplicate: {e:?}")) })?; @@ -1676,12 +1914,14 @@ fn build_namespace( program: &Program, system: &Manifest, connectors: &BTreeMap<&str, Connector>, + dirs: &BTreeMap, extras: &[(&str, Connector)], ) -> std::io::Result> { // Never the swap port: [`swap_namespace`] says why. let receives: Vec<&String> = program.receives.iter().filter(|name| *name != toyos_swap::PORT).collect(); - if receives.is_empty() && extras.is_empty() { + let view: Vec<(&String, &Connector)> = dirs.iter().filter(|(name, _)| in_view(program, name)).collect(); + if receives.is_empty() && extras.is_empty() && view.is_empty() { return Ok(None); } let mut builder = namespace::build(); @@ -1695,6 +1935,9 @@ fn build_namespace( ), } } + for (name, connector) in view { + builder = builder.add(name, connector); + } // A `provides` name reaches its holder from whoever made the port, never // from init, and this is where it arrives. for (name, connector) in extras { @@ -1709,6 +1952,25 @@ fn build_namespace( } } +/// Whether the directory capability `name` is in `program`'s view. +/// +/// **Every program sees the whole tree the file servers serve**, which is the +/// kernel's old view kept whole until each row declares its own +/// (`issues/isolation/every-program-sees-only-the-files-it-was-given.md`, +/// stage 2), with two exceptions: `/boot` is the updater's alone, and a +/// storage row sees none, since a file server resolving a path of its own +/// through itself waits for ever. +fn in_view(program: &Program, name: &str) -> bool { + if is_storage(program) { + return false; + } + match name.strip_prefix(CAPABILITY_PREFIX) { + Some("/boot") => program.slots, + Some(_) => true, + None => false, + } +} + /// [`toyos_swap::PORT`] in a namespace of its own, for a program whose row /// receives it, endowed under [`toyos_swap::LABEL`]. /// @@ -1744,3 +2006,124 @@ fn swapped(log: &Log, service: &str, word: Word, detail: &str) { } } } + +/// What a storage row's process is started with beside its row: its +/// arguments, and the claims on the partition a file server's role is on +/// where a disk the kernel drives carries it. +#[derive(Default)] +struct Storage { + args: Vec, + claims: Vec<(String, toyos::Device)>, +} + +/// Every record the kernel's inventory answers. +fn inventory(syscap: &SysCap) -> Vec { + use toyos_abi::inventory::{RawRecord, Record}; + let Ok(asked) = syscap.inventory(&mut []) else { return Vec::new() }; + let mut raw = vec![RawRecord::EMPTY; asked]; + let Ok(n) = syscap.inventory(&mut raw) else { return Vec::new() }; + raw[..n].iter().filter_map(|r| Record::decode(r).ok()).collect() +} + +/// The unique GUID the loader named for `role`. +fn loaded(records: &[toyos_abi::inventory::Record], role: toyos_abi::inventory::Role) -> Option<[u8; 16]> { + records.iter().find_map(|r| match r { + toyos_abi::inventory::Record::Loaded(l) if l.role == role => Some(l.unique_guid), + _ => None, + }) +} + +fn guid_text(guid: [u8; 16]) -> String { + let mut buf = [0u8; toyos_abi::part::GUID_TEXT_LEN]; + toyos_abi::part::PartGuid(guid).write_text(&mut buf).to_string() +} + +/// A storage row's arguments and claims for one start. +/// +/// A block service is told the ROOT the machine runs from, which it serves no +/// session on. A file server is told its role, and gets its partition as a +/// claim when a disk the kernel drives carries it — the stick, until usbd +/// serves it — and otherwise the partition's GUID, which it opens through the +/// block service; DATA it finds by type there itself. +fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> Storage { + use toyos_abi::inventory::{Record, Role}; + let mut storage = Storage::default(); + let is_block = program.serves.iter().any(|s| s == toyos_blockring::PORT); + if !is_block && role.is_none() { + return storage; + } + let records = inventory(syscap); + if is_block { + if let Some(root) = loaded(&records, Role::Root) { + storage.args.extend(["--running".to_string(), guid_text(root)]); + } + return storage; + } + let role = role.expect("checked above"); + storage.args.push(role.to_string()); + let on_kernel_disk = |pick: &dyn Fn(&toyos_abi::inventory::Partition) -> bool| -> Vec<[u8; 16]> { + records + .iter() + .filter_map(|r| match r { + Record::Partition(p) if pick(p) => Some(p.unique_guid), + _ => None, + }) + .collect() + }; + let (kernel, named) = match role { + "data" => (on_kernel_disk(&|p| p.type_guid == toyos_gpt::Guid::TOYOS_DATA.0), None), + "log" | "boot" => { + let which = if role == "log" { Role::Log } else { Role::Boot }; + match loaded(&records, which) { + Some(guid) => (on_kernel_disk(&|p| p.unique_guid == guid), Some(guid)), + None => (Vec::new(), None), + } + } + other => panic!("init: `{other}` is no role; the build refuses it"), + }; + match kernel.as_slice() { + [guid] => { + let request = toyos_abi::syscall::DeviceRequest::Partition(toyos_abi::part::PartGuid(*guid)); + let mut buf = [0u8; toyos_abi::syscall::DeviceRequest::MAX_NAME]; + let name = request.write_name(&mut buf).to_string(); + match syscap.claim_partition::(toyos_abi::part::PartGuid(*guid)) { + Ok(claim) => storage.claims.push((format!("{DEV_PREFIX}{name}"), claim)), + Err(e) => say!("init: {}: {}", program.name, refused(&name, e)), + } + } + [] => { + if let Some(guid) = named { + storage.args.push(guid_text(guid)); + } + } + many => say!( + "init: {}: the kernel's disks carry {} partitions for the `{role}` role, and a role is one", + program.name, + many.len() + ), + } + storage +} + +/// The manifest, off ROOT through the kernel's own `open`. +/// +/// **Not through std**: std resolves a path through this process's namespace, +/// and the first path it resolves fixes that namespace for the process's life. +/// init builds its namespace out of the manifest, so the manifest is read +/// before std is asked anything. +fn read_manifest() -> Result { + use toyos_abi::syscall::{self, OpenFlags}; + let handle = syscall::open(toyos_manifest::GUEST_PATH.as_bytes(), OpenFlags::READ)?; + let mut bytes = Vec::new(); + let mut buf = [0u8; 4096]; + let read = loop { + match syscall::read(handle, &mut buf) { + Ok(0) => break Ok(()), + Ok(n) => bytes.extend_from_slice(&buf[..n]), + Err(e) => break Err(e), + } + }; + syscall::close(handle); + read?; + String::from_utf8(bytes).map_err(|_| SyscallError::InvalidArgument) +} From c191577f30e43bbc39e78c92213ce50fcb117ef5 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 07:48:47 +0200 Subject: [PATCH 02/54] storage: the tests move onto the file servers, and fsd answers every client (work in progress) The kernel tests whose subject was deleted go with it; the ones whose claim survives are retargeted onto blockd, fsd and the kernel's USB partition claims. fsd answers one request per client per wait and asks an acceptor before it accepts: a duplicate completion parked the log's server at boot. New: fs_escape, fsd_restart, and fsd's FAT read-error host tests. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- ...sk-blockd-drives-cannot-write-its-slots.md | 22 + ...ed-mid-sync-leaves-a-half-written-btree.md | 25 + .../an-nvme-flush-issues-no-command.md | 32 -- ...ograms-name-no-file-a-file-server-holds.md | 22 + ...refusal-retried-is-red-on-every-nightly.md | 31 -- ...-can-open-every-partition-blockd-serves.md | 20 + ...storage-still-comes-up-before-init-runs.md | 18 +- ...s-can-name-one-arrival-and-accept-parks.md | 33 ++ kernel/src/actuator.rs | 20 +- kernel/src/arch/x86_64/vtd/domain.rs | 4 + kernel/src/arch/x86_64/vtd/mod.rs | 24 +- kernel/src/block.rs | 55 ++ kernel/src/gpt.rs | 28 ++ kernel/src/main.rs | 24 +- kernel/src/quiesce.rs | 1 - kernel/src/revoke_selftest.rs | 8 +- src/metal.rs | 15 - src/metaldevices.rs | 44 +- tests/common/devices.rs | 54 +- tests/common/faults.rs | 14 +- tests/common/gpt.rs | 71 ++- tests/common/inspect.rs | 60 +-- tests/common/iommu.rs | 114 +++-- tests/common/partclaim.rs | 111 ++-- tests/common/power.rs | 22 +- tests/common/storage.rs | 83 ++- tests/common/toybox.rs | 2 +- tests/common/volumes.rs | 474 +----------------- tests/fsdrestartcase/system.toml | 41 ++ tests/inspectcase/system.toml | 3 - tests/test-durations | 3 - .../src/bin/boot_volume_metadata_error.rs | 50 -- .../src/bin/cache_eviction.rs | 193 ------- .../src/bin/fs_dirs_durable.rs | 38 +- tests/toyos-rust-tests/src/bin/fs_escape.rs | 102 ++++ tests/toyos-rust-tests/src/bin/fs_restart.rs | 92 ++++ .../src/bin/log_volume_reread.rs | 60 --- .../src/bin/partition_claimant.rs | 39 +- .../toyos-rust-tests/src/bin/quiesce_fsync.rs | 94 ---- .../toyos-rust-tests/src/bin/quiesce_twice.rs | 24 +- tests/toyos.rs | 461 +++-------------- userland/blockd/src/main.rs | 10 +- userland/fsd/src/fat.rs | 77 +++ userland/fsd/src/main.rs | 115 +++-- userland/init/src/main.rs | 7 +- userland/logd/src/inspect.rs | 12 +- userland/logd/src/main.rs | 2 +- userland/logd/src/serve.rs | 18 +- 48 files changed, 1109 insertions(+), 1763 deletions(-) create mode 100644 issues/boot-media/an-image-on-a-disk-blockd-drives-cannot-write-its-slots.md create mode 100644 issues/filesystem/a-file-server-killed-mid-sync-leaves-a-half-written-btree.md delete mode 100644 issues/filesystem/an-nvme-flush-issues-no-command.md create mode 100644 issues/filesystem/c-programs-name-no-file-a-file-server-holds.md delete mode 100644 issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md create mode 100644 issues/isolation/a-file-server-can-open-every-partition-blockd-serves.md create mode 100644 issues/kernel/two-completions-can-name-one-arrival-and-accept-parks.md create mode 100644 tests/fsdrestartcase/system.toml delete mode 100644 tests/toyos-rust-tests/src/bin/boot_volume_metadata_error.rs delete mode 100644 tests/toyos-rust-tests/src/bin/cache_eviction.rs create mode 100644 tests/toyos-rust-tests/src/bin/fs_escape.rs create mode 100644 tests/toyos-rust-tests/src/bin/fs_restart.rs delete mode 100644 tests/toyos-rust-tests/src/bin/log_volume_reread.rs delete mode 100644 tests/toyos-rust-tests/src/bin/quiesce_fsync.rs diff --git a/issues/boot-media/an-image-on-a-disk-blockd-drives-cannot-write-its-slots.md b/issues/boot-media/an-image-on-a-disk-blockd-drives-cannot-write-its-slots.md new file mode 100644 index 00000000000..1c9622edcb4 --- /dev/null +++ b/issues/boot-media/an-image-on-a-disk-blockd-drives-cannot-write-its-slots.md @@ -0,0 +1,22 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# An image on a disk blockd drives cannot write its slots + +The updater writes the idle slot through a partition claim +(`SYS_DEVICE_CLAIM` with `DeviceType::Partition`, granted by `system.toml`'s +`slots` row), and the kernel answers a claim only on the disks it drives — +the USB disks, until usbd drives the controller. The kernel drives no NVMe +controller: blockd does (`userland/blockd`). So a machine booted from an image +on its NVMe disk — `Profile::InternalDisk`, and the T14 once ToyOS is installed +on its internal disk — finds its slots nowhere a claim reaches, and an update +is refused at the claim. + +Every update test boots off a USB stick, which is why nothing fails. + +**Exit**: the slots of an image on a disk blockd drives are written through a +session init grants the updater on blockd's port, judged by an update test +that boots `Profile::InternalDisk`. diff --git a/issues/filesystem/a-file-server-killed-mid-sync-leaves-a-half-written-btree.md b/issues/filesystem/a-file-server-killed-mid-sync-leaves-a-half-written-btree.md new file mode 100644 index 00000000000..386dd6494e2 --- /dev/null +++ b/issues/filesystem/a-file-server-killed-mid-sync-leaves-a-half-written-btree.md @@ -0,0 +1,25 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A file server killed mid-sync leaves a half-written btree + +`userland/fsd`'s DATA volume is `bcachefs`'s interim format: the btree is +rewritten in place and there is no journal. A sync writes the dirty blocks +fsd's cache holds (`userland/fsd/src/cache.rs`) in block order and then asks +the device to flush; nothing orders a node's children before the node that +names them, and nothing records which of them reached the device. A server +that ends between two of those writes — a crash, a kill, the restart budget +`toyos_manifest::RESTARTS` exists for — or a power cut in the same window +leaves a volume whose next mount reads a node naming blocks that hold another +generation's bytes. + +`fsd_restart` does not see it: it ends the server under a write that no sync +has started, so what is on the device is the last sync's whole state. + +**Exit**: a sync that a cut at any block boundary leaves mountable, as the +last acknowledged sync or the one before it — copy-on-write with a superblock +that is written last, or a journal replayed at mount — with a host test that +cuts a sync after every block it writes and mounts what is left. diff --git a/issues/filesystem/an-nvme-flush-issues-no-command.md b/issues/filesystem/an-nvme-flush-issues-no-command.md deleted file mode 100644 index d3dc9fb24d7..00000000000 --- a/issues/filesystem/an-nvme-flush-issues-no-command.md +++ /dev/null @@ -1,32 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-24 ---- - -# An NVMe flush issues no command, so `fsync` on an NVMe disk answers `Ok` over a volatile write cache - -`NvmeBlockDevice::flush` (`kernel/src/drivers/nvme.rs`) returns `Ok(())` without -sending anything to the controller; its doc says "writes are synchronous, so -there is nothing to flush". A write command's completion says the controller -took the data, not that it is on the medium: a controller that reports a -volatile write cache (the `VWC` field of Identify Controller) may hold -completed writes until a Flush command. So on such a disk `SYS_FSYNC` — a file's and a partition claim's — says durable about writes a power -cut can still lose. - -QEMU's NVMe completes writes against its backing and every guest test passes -either way, which is why nothing here fails today. Whether the T14's -namespace reports a volatile write cache is not measured. - -blockd, the NVMe driver in userland (`userland/blockd/src/nvme.rs`), reads -`VWC` and issues Flush, and `blockd_serves_partitions` reads the Flush commands -off QEMU's own trace. That is a second driver on a second controller: the -kernel's still serves `/apps`, `/home` and partition claims on the first, and -this defect is that driver's. - -**Exit condition.** The kernel's driver reads `VWC` at bring-up and a flush on -a controller that has a volatile write cache issues Flush and waits for its -completion under the caller's operation budget, with a test that stages a -controller whose Flush fails and sees the fsync fail — or the driver is -deleted (`issues/kernel/the-kernel-is-small-interrupts-post-and-threads-wait.md`, -step 9), whichever lands first. diff --git a/issues/filesystem/c-programs-name-no-file-a-file-server-holds.md b/issues/filesystem/c-programs-name-no-file-a-file-server-holds.md new file mode 100644 index 00000000000..156e0c2804e --- /dev/null +++ b/issues/filesystem/c-programs-name-no-file-a-file-server-holds.md @@ -0,0 +1,22 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A C program names no file a file server holds + +`/apps`, `/config`, `/home`, `/state`, `/log` and `/boot` are served by +`/system/bin/fsd` through directory capabilities in each program's namespace, +and std reaches them through `toyos::fs` (`rust/library/std/src/sys/fs/toyos.rs`). +`userland/libc` does not: `open`, `stat`, `opendir` and every other path call +in `userland/libc/src/posix_io.rs` and `userland/libc/src/stdio.rs` go to the +kernel's `SYS_OPEN` family, and the kernel serves ROOT and `/tmp` only. So a C +program — tinycc, doomgeneric, anything built with `toyos-cc` — is refused +every path under those six directories with `ENOENT`, and a file it writes +where a user's files live cannot be written at all. + +**Exit**: libc resolves a path under a directory capability through +`toyos::fs` as std does — open, read, write, seek, stat, readdir, mkdir, +unlink, rename, fsync — with a C test in the corpus that writes a file under +`/home`, reads it back, and lists it. diff --git a/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md b/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md deleted file mode 100644 index 25d447d835b..00000000000 --- a/issues/filesystem/home-budget-refusal-retried-is-red-on-every-nightly.md +++ /dev/null @@ -1,31 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# `home_budget_refusal_retried` is red on every nightly - -`home_budget_refusal_retried` (nightly tier) is red, and red alone, on the -nightlies of three trees, each time with the same two shapes: -- main at e8d7c9c0 (run 36228604597); -- PR #524's branch at 8c5be843 (run 36273557690); -- main at fd62f567 (run 36278449733, guest (12)). - -The two shapes: `no fsync: /home/... durable on attempt line — the retry -never ran`, and `Boot timed out waiting for ===READY===`. In the second, the -console ends at the loader's `Applied 5483 relocations`, with -`fsync-budget-spent` on the boot parameter line. It was green at 3f46a019 -(run 36111884575). - -It is green alone on a dev host at f231c43e (`cargo test --test toyos-build --- --nightly home_budget_refusal_retried` EXIT=0). -`cargo run -- --known-red home_budget_refusal_retried` answers NO. - -`fsync-budget-spent` is the machine-wide actuator that -issues/boot-media/partition-claim-gives-up-reds-beside-other-guests-and-is-green-alone.md -names as racing whichever fsync the boot reaches first. Not shown: whether -that race is this test's cause. - -**Exit**: the cause shown on a red run's log, and the test green on a -nightly. diff --git a/issues/isolation/a-file-server-can-open-every-partition-blockd-serves.md b/issues/isolation/a-file-server-can-open-every-partition-blockd-serves.md new file mode 100644 index 00000000000..d20cf62bab3 --- /dev/null +++ b/issues/isolation/a-file-server-can-open-every-partition-blockd-serves.md @@ -0,0 +1,20 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A file server can open every partition blockd serves + +init gives each `fsd` role the `block` connector (`receives = ["block"]`), and a +session on it is opened by unique GUID (`toyos_blockring::wire::MSG_OPEN`) for +any partition blockd serves but ROOT. So the log's server can open DATA, and +the boot volume's server — whose volume is read-only — can open a partition to +write. Nothing in the protocol ties a session to the role that asked for it; +what keeps each server on its own partition is the argument init passes it, +which is not authority. + +**Exit**: init hands each file server a capability to its own partition's +session and nothing else — a port blockd serves per partition, or a session +init opens and moves — with a negative control: the log's server asking for +DATA is refused. diff --git a/issues/kernel/storage-still-comes-up-before-init-runs.md b/issues/kernel/storage-still-comes-up-before-init-runs.md index c6950ce8f18..f63a192116d 100644 --- a/issues/kernel/storage-still-comes-up-before-init-runs.md +++ b/issues/kernel/storage-still-comes-up-before-init-runs.md @@ -9,25 +9,21 @@ opened: 2026-09-25 ROOT is the loader's image in memory, and the kernel mounts it and spawns init with no storage command issued (`root_from_memory` asserts the count `rootfs::INIT_WITHOUT_A_DISK` logs is zero). But `kernel_main` still brings -up NVMe, xHCI, `fat32_adapter::probe_boot_disks` and the DATA, `/boot` and -`/log` mounts in boot context after that spawn and before `smp::set_ready` -releases the machine, so every one of them runs before init's first -instruction. The track's stage 1 exit +up xHCI and reads every USB disk's table (`gpt::probe_usb_disks`) in boot +context after that spawn and before `smp::set_ready` releases the machine, so +both run before init's first instruction. The track's stage 1 exit (`issues/kernel/the-kernel-is-small-interrupts-post-and-threads-wait.md`) asks for them to be untouched until init runs. -Two things hold it there. `xhci::init` binds a USB stick's mass storage — +What holds it there: `xhci::init` binds a USB stick's mass storage — INQUIRY, READ CAPACITY — inside the one enumeration scan that runs before there is a scheduler, so a USB-boot machine issues storage commands in it -whatever order the rest takes; and nothing in the VFS makes a syscall on -`/home`, `/apps`, `/log` or `/boot` wait for a mount that has not happened -yet, so moving the mounts past the release would hand logd and the shell a -missing directory instead of a late one. +whatever order the rest takes. The volumes are no longer the kernel's: the +file servers mount them once init starts them. Measured on QEMU 11 TCG, USB-stick boot of the `--build-only` image: `Boot: storage ready` is 118 ms of a 222 ms kernel boot, all of it now after init's spawn and before init runs. Exit: a boot whose storage drivers first speak after init has run, with the -USB bind out of the pre-scheduler scan (the track's stage 5) and the -storage mounts awaited rather than absent. +USB bind out of the pre-scheduler scan (the track's stage 5). diff --git a/issues/kernel/two-completions-can-name-one-arrival-and-accept-parks.md b/issues/kernel/two-completions-can-name-one-arrival-and-accept-parks.md new file mode 100644 index 00000000000..0313fffbad5 --- /dev/null +++ b/issues/kernel/two-completions-can-name-one-arrival-and-accept-parks.md @@ -0,0 +1,33 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# Two completions can name one arrival, and an `accept` after the second parks + +`process_watch` (`kernel/src/inbox/mod.rs`) answers a watch on a handle that +is already ready at once, and only a watch it has to register replaces the +armed one on the same handle. So a watch submitted while a connection arrives +can complete immediately beside the older armed watch that the same arrival +fires: two completions for one connection, in one drain or in two. A read +after a spurious completion is harmless where the read does not block, but +`SYS_ACCEPT` (`kernel/src/syscall/ipc.rs`, `sys_accept`) parks until a +connection is queued, so a server that accepts once per completion takes the +connection on the first and parks for good on the second — the port it serves +then answers nobody. + +Measured on `/system/bin/fsd`'s log server under `tests/quiescecase`, whose six +writers connect beside logd at boot: a blocked-task dump showed the server +`ipc parked` (the class only `sys_accept` parks in) while logd and every writer +waited on it, in one boot of four and in one of three in a second series. fsd +now asks the acceptor with a zero-timeout watch on a probe ring before it +accepts (`userland/fsd/src/main.rs`, `Server::accept`), and after the change +the same boot ran six times and never parked. Every other server that accepts after +a completion — blockd, logd's inspect and serve threads, init, +soundd, the compositor, netd, filepicker — does not. + +**Exit**: a spurious completion cannot park a server — an accept that does not +park when nothing is queued, or a watch that withdraws the armed one on its +handle whether or not it answers at once, with a test that submits a watch +while a connection arrives and counts the completions. diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index f4331ac4380..4eebe1a3e17 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -104,10 +104,6 @@ actuators! { /// Read and write every USB disk carrying the gate's stamp in block 0; the stamp, not the parameter, picks the disk, since the boot stick shares the bus. usb_storage_gate = "usb-storage-gate"; - - - - /// Hold the thread named `toyos_quiesce::LAST_THREAD` inside `SYS_NANOSLEEP`, and the shutdown until it is held there, until the stop waits on it alone: its park is then the stop's last transition. quiesce_last_park = "quiesce-last-park"; @@ -117,7 +113,6 @@ actuators! { /// Serve a blocked-task dump from the shutdown once its first stage has stopped the machine: the report Ctrl+Alt+D gives on a shutdown stuck in its stop. quiesce_dump = "quiesce-dump"; - /// Take the page a shrink has just read off the device, in the window between that read and the lock that spends it — one other CPU's CLOCK sweep, which needs no VFS lock and so runs there. resize_evict_window = "resize-evict-window"; @@ -243,7 +238,6 @@ actuators! { /// expired. fsync_deadman_now = "fsync-deadman-now"; - /// Under-deliver one READ(10) data phase so the byte counts disagree. usb_short_read = "usb-short-read"; @@ -346,7 +340,6 @@ actuators! { /// Deliver the i8042 vector once at arming with no byte behind it — the arming edge, staged. i8042_arm_edge = "i8042-arm-edge"; - /// Blind init's read of the reset handshake, staging virtio devices that never answer; the console — the staged boot's capture channel — is spared. virtio_reset_stuck = "virtio-reset-stuck"; @@ -444,9 +437,7 @@ actuators! { /// Stop the boot dead in phase 3, interrupts off, before any log drain. pre_idle_wedge = "pre-idle-wedge"; - - - /// Leave the NVMe controller out of the IOMMU's root table. + /// Leave the xHCI controller out of the IOMMU's root table. iommu_context_absent = "iommu-context-absent"; /// Give it a present context entry naming an empty second-level table, distinct from an absent context: passthrough would fault identically to the row above. @@ -496,15 +487,18 @@ actuators! { /// Run the revoked-backing controls (`/tmp` and `/home`) after mount. revoked_backing_selftest = "revoked-backing-selftest"; - - /// Reopen init by pid once it is spawned, the way `SYS_PROCESS_OPEN` does. process_reopen_selftest = "process-reopen-selftest"; + /// Refuse every read of device block 0 of each disk the kernel drives — its + /// protective MBR and GPT header — once the boot has read its own tables, + /// so a partition claim meets a disk that does not answer a read of its + /// table. Judged by `partition_claim_gives_up`. + partclaim_table_unanswered = "partclaim-table-unanswered"; + /// Offer the block layer a second device claiming a registered `DeviceId`, and report what it did with it. block_duplicate_id = "block-duplicate-id"; - /// Arm the watchdog at seconds rather than minutes, so a guest reaches the reset. watchdog_fast = "tco-fast"; diff --git a/kernel/src/arch/x86_64/vtd/domain.rs b/kernel/src/arch/x86_64/vtd/domain.rs index ab9aaddd951..d068f4df4be 100644 --- a/kernel/src/arch/x86_64/vtd/domain.rs +++ b/kernel/src/arch/x86_64/vtd/domain.rs @@ -164,6 +164,10 @@ pub fn unmap(id: DomainId, at: Iova, bytes: u64) -> Result<(), IommuError> { } pub fn attach(stream: StreamId, id: DomainId) { + if super::staged(stream) { + log!("iommu: {stream} keeps the context an actuator staged, over its move to domain {}", id.raw()); + return; + } let mut domains = DOMAINS.lock(); let domain = *domains.at(id); let mut units = UNITS.lock(); diff --git a/kernel/src/arch/x86_64/vtd/mod.rs b/kernel/src/arch/x86_64/vtd/mod.rs index aa589e0e576..f1ce95c33c3 100644 --- a/kernel/src/arch/x86_64/vtd/mod.rs +++ b/kernel/src/arch/x86_64/vtd/mod.rs @@ -472,19 +472,21 @@ fn enable( // it; both are answered on the device's first *read*, since a // first-write access would cache write permission and never fault. if crate::actuator::iommu_context_absent() - && device.matches_class(NVME_CLASS, NVME_SUBCLASS, None) + && device.matches_class(XHCI_CLASS, XHCI_SUBCLASS, Some(XHCI_PROG_IF)) { log!("iommu: unit{index} leaves {stream} out of the root table (actuator)"); + STAGED.store(u32::from(stream.requester()), core::sync::atomic::Ordering::Relaxed); continue; } // A present context entry naming an empty domain, distinct from a // missing entry: passthrough would fault identically either way. if crate::actuator::iommu_empty_domain() - && device.matches_class(NVME_CLASS, NVME_SUBCLASS, None) + && device.matches_class(XHCI_CLASS, XHCI_SUBCLASS, Some(XHCI_PROG_IF)) { let empty = tables.alloc(); log!("iommu: unit{index} gives {stream} a domain with no mappings (actuator)"); table::bind_identity(&mut tables, root, stream, empty, width); + STAGED.store(u32::from(stream.requester()), core::sync::atomic::Ordering::Relaxed); continue; } table::bind_identity(&mut tables, root, stream, domain, width); @@ -556,11 +558,23 @@ fn enable( ); } -/// Device class the actuators target, not a bus/device/function: QEMU's slot +/// Device class the actuators target — the xHCI controller, which this kernel +/// drives by DMA from boot — not a bus/device/function: QEMU's slot /// choice is not this kernel's business, and the harness reads the same /// class independently out of `pci::enumerate`. -const NVME_CLASS: u8 = 0x01; -const NVME_SUBCLASS: u8 = 0x08; +/// The requester id `iommu-context-absent` or `iommu-empty-domain` staged, so +/// its driver's move to a domain of its own leaves the staging in place; +/// `u32::MAX`, which no requester id is, when neither is armed. +static STAGED: core::sync::atomic::AtomicU32 = core::sync::atomic::AtomicU32::new(u32::MAX); + +/// Whether `stream` is the one an actuator left without a working context. +pub(super) fn staged(stream: StreamId) -> bool { + STAGED.load(core::sync::atomic::Ordering::Relaxed) == u32::from(stream.requester()) +} + +const XHCI_CLASS: u8 = 0x0C; +const XHCI_SUBCLASS: u8 = 0x03; +const XHCI_PROG_IF: u8 = 0x30; /// Slot of the per-width domain cache; exhaustive match so a new `AddressWidth` fails to compile here. fn domain_slot(width: AddressWidth) -> usize { diff --git a/kernel/src/block.rs b/kernel/src/block.rs index 1805f20af93..6c17376bc08 100644 --- a/kernel/src/block.rs +++ b/kernel/src/block.rs @@ -232,6 +232,9 @@ static DEVICES: Lock> = Lock::new(BTreeMap::new()); /// cache, which a plain insert here would arrange in silence. #[must_use = "a refused registration leaves the device unreachable"] pub fn register(dev: Box) -> Option { + #[cfg(feature = "boot-actuators")] + let dev: Box = + if crate::actuator::partclaim_table_unanswered() { Box::new(unanswered::Table(dev)) } else { dev }; let id = dev.device_id(); let blocks = dev.block_count(); let mut devices = DEVICES.lock(); @@ -251,6 +254,58 @@ pub fn register(dev: Box) -> Option { Some(handle) } +/// `partclaim-table-unanswered`: a registered disk that refuses every read of +/// its device block 0 — its protective MBR and GPT header — once +/// [`unanswered::refuse`] has been called. +#[cfg(feature = "boot-actuators")] +pub mod unanswered { + use core::sync::atomic::{AtomicBool, Ordering}; + + use alloc::boxed::Box; + + use super::{BlockDevice, BlockError, BlockResult, DeviceId}; + + static REFUSING: AtomicBool = AtomicBool::new(false); + + pub(super) struct Table(pub(super) Box); + + impl BlockDevice for Table { + fn device_id(&self) -> DeviceId { + self.0.device_id() + } + + fn block_count(&self) -> u64 { + self.0.block_count() + } + + fn read_blocks(&mut self, lba: u64, count: u32, buf: &mut [u8]) -> BlockResult { + if lba == 0 && count > 0 && REFUSING.load(Ordering::Relaxed) { + return Err(BlockError::Device); + } + self.0.read_blocks(lba, count, buf) + } + + fn write_blocks(&mut self, lba: u64, count: u32, buf: &[u8]) -> BlockResult { + self.0.write_blocks(lba, count, buf) + } + + fn flush(&mut self) -> BlockResult { + self.0.flush() + } + + fn losses(&self) -> u64 { + self.0.losses() + } + } + + /// Every disk registered from here on, and before, refuses reads of its + /// block 0. Called once the boot's own table reads are done. + pub fn refuse() { + REFUSING.store(true, Ordering::Relaxed); + log!("partclaim-table-unanswered: device block 0 of every disk refuses reads from now on"); + } +} + pub fn open(id: DeviceId) -> Option { DEVICES.lock().get(&id).cloned() } diff --git a/kernel/src/gpt.rs b/kernel/src/gpt.rs index 9033983c8f8..b1260559aaa 100644 --- a/kernel/src/gpt.rs +++ b/kernel/src/gpt.rs @@ -104,6 +104,34 @@ pub fn loaded() -> Vec<(Role, Guid)> { LOADED.lock().clone() } +/// Where the log partition the loader named is, as far as this kernel sees. +pub enum LogPlace { + /// The loader named none. + Unnamed, + /// On a disk this kernel drives, where its file server claims it. + Driven, + /// Not on the disk this kernel booted from and drives, which is where an + /// image carries it: this machine has no log partition. + Absent, + /// The boot disk is one this kernel does not drive, so whether the log + /// partition is on it is its file server's to say. + Undriven, +} + +/// [`LogPlace`], once [`probe_usb_disks`] has read every disk it will. +pub fn log_place() -> LogPlace { + let Some(guid) = LOADED.lock().iter().find(|(role, _)| *role == Role::Log).map(|(_, g)| *g) else { + return LogPlace::Unnamed; + }; + if LISTED.lock().iter().any(|disk| disk.parts.iter().any(|p| p.unique_guid == guid)) { + return LogPlace::Driven; + } + match *RESOLVED.lock() { + Resolution::Found { .. } => LogPlace::Absent, + Resolution::Unknown | Resolution::Ambiguous => LogPlace::Undriven, + } +} + /// List `handle`'s table into [`LISTED`], once per disk. fn list(sectors: &mut DeviceSectors<'_>, handle: &Handle, lba_bytes: u32) { let id = handle.device_id(); diff --git a/kernel/src/main.rs b/kernel/src/main.rs index ec834ec79d0..4b7bc457c01 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -213,13 +213,21 @@ fn report_power_on(args: &KernelArgs, complete: u64) { } /// Says where this boot's log can be read, on the last surface still showing it once userland owns the screen. -fn report_log_destination(args: &KernelArgs) { +fn report_log_destination() { // Kernel-side because panic_console owns the panel; logd reports which file it opened separately. - // Whether the loader named a log partition, not whether its file server mounted it: that - // server says so itself, and this kernel mounts nothing but ROOT. - let has_log = args.log_partition_guid != [0; 16]; + // Whether the partition is on a disk this kernel reads, not whether its file server mounted it: + // that server says so itself, and this kernel mounts nothing but ROOT. + let console = drivers::serial::has_console(); + let has_log = match gpt::log_place() { + gpt::LogPlace::Driven => true, + gpt::LogPlace::Unnamed | gpt::LogPlace::Absent => false, + gpt::LogPlace::Undriven => { + log!("log: /log is on a disk this kernel does not drive; its file server says whether it mounted"); + return; + } + }; // ASCII only: the panel's font renders anything outside 0x20..=0x7E as a dot. - match (drivers::serial::has_console(), has_log) { + match (console, has_log) { (true, true) => log!("log: this boot is on the console and on /log"), (false, true) => log!("log: no serial console - this boot is on /log and on the screen"), // alert! reddens the panel's Level for exactly the two states that leave no account of this boot anywhere. @@ -478,6 +486,10 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { // After xhci::init: a USB-booted disk doesn't exist until the controller binds it. gpt::probe_usb_disks(); rootfs::hold_source(); + #[cfg(feature = "boot-actuators")] + if actuator::partclaim_table_unanswered() { + block::unanswered::refuse(); + } #[cfg(feature = "boot-actuators")] if actuator::leak_rollback_selftest() { @@ -546,7 +558,7 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { pre_idle_wedge(); } - report_log_destination(kernel_args); + report_log_destination(); let complete_tsc = cpu::counter(); boot_phase!("complete", 0); report_power_on(kernel_args, complete_tsc); diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 1d3dcc9cd3f..9daf9e5e902 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -129,7 +129,6 @@ pub fn note_progress() { } } - /// Stop every userland thread but the caller, and answer with what it took. /// /// Returns when the machine is stopped or when [`PARK`] is spent, never diff --git a/kernel/src/revoke_selftest.rs b/kernel/src/revoke_selftest.rs index d44a28763f7..7de9f2fea14 100644 --- a/kernel/src/revoke_selftest.rs +++ b/kernel/src/revoke_selftest.rs @@ -1,7 +1,6 @@ -//! Revoked-backing controls, behind `revoked-backing-selftest`: on each writable -//! mount, a `FileBacking` read after the file's deletion must fail rather than -//! fault in zeros. `/tmp`'s `TmpfsBacking` and `/home`'s `NvmeBacking` are -//! separate implementations of the one contract, and must answer alike. +//! The revoked-backing control, behind `revoked-backing-selftest`: on the +//! kernel's one writable mount, `/tmp`, a `FileBacking` read after the file's +//! deletion must fail rather than fault in zeros. use crate::file_cache; use crate::mm::PAGE_BYTES; @@ -11,7 +10,6 @@ const FILL: u8 = 0xA7; pub fn run() { probe("/tmp/revoke_probe"); - probe("/home/revoke_probe"); } /// FAIL names the step so the verdict line carries the mechanism, not just the arm. diff --git a/src/metal.rs b/src/metal.rs index b6c5e539df4..25855e1b0aa 100644 --- a/src/metal.rs +++ b/src/metal.rs @@ -757,7 +757,6 @@ pub const FLASHABLE: &[(&str, Flash)] = &[ ("pci-cap-selftest", Flash::Ok), ("process-reopen-selftest", Flash::Ok), ("revoked-backing-selftest", Flash::Ok), - ("pc-unbind-selftest", Flash::Ok), ("leak-rollback-selftest", Flash::Ok), ("lapic-spurious-selftest", Flash::Ok), ("unclaimed-vector-selftest", Flash::Ok), @@ -810,20 +809,6 @@ pub const FLASHABLE: &[(&str, Flash)] = &[ metal verdict is read across — an image armed with it stages its own red", ), ), - ( - "quiesce-drain-refuse", - Flash::Never( - "it refuses the shutdown's own drain of a closed file's flush, so an image armed \ - with it stages a stall inside the one sync a metal verdict rests on", - ), - ), - ( - "quiesce-fsync-refuse", - Flash::Never( - "it refuses `/system/bin/logd`'s own flush for longer than the shutdown waits for \ - it, so an image armed with it never makes its last word durable", - ), - ), ( "xhci-lock-wedged", Flash::Never( diff --git a/src/metaldevices.rs b/src/metaldevices.rs index 4787182045b..bebe745e9ac 100644 --- a/src/metaldevices.rs +++ b/src/metaldevices.rs @@ -117,19 +117,19 @@ pub const RECORDS: &[Record] = &[ Record { about: "i8042", needle: "i8042: ok selftest=0x55", presence: Says }, Record { about: "i8042-quarantine", needle: "i8042: quarantined", presence: NeverSays }, Record { about: "i8042-lost-edge", needle: "i8042: bytes with no IRQ record", presence: NeverSays }, - // The internal disk: identified, and — the whole safety argument for - // running on this machine at all — never written. - Record { about: "nvme", needle: "NVMe: NS1 size=", presence: Says }, - Record { about: "nvme-refused", needle: "NVMe: NOT INITIALISED", presence: NeverSays }, - Record { about: "nvme-offline", needle: "NVMe: this controller is offline", presence: NeverSays }, + // The internal disk — the whole safety argument for running on this + // machine at all — never written: the kernel drives no NVMe, and the one + // process that may drives none here, since its row names only the + // emulated controller. + Record { about: "nvme", needle: NVME_UNDRIVEN, presence: Says }, + Record { about: "nvme-served", needle: "blockd: partition ", presence: NeverSays }, // The boot ended the way the loop's verdict needs it to. Record { about: "boot", needle: "Boot: complete (", presence: Says }, ]; -/// The one census field a boot may not have moved: a write to a disk this -/// project never writes. -pub const NVME_CENSUS: &str = "nvme: commands "; -pub const NVME_NO_WRITES: &[&str] = &["write=0", "io-other=0"]; +/// blockd's word that no NVMe controller its row names is on the machine, so +/// nothing reads or writes the internal disk. +pub const NVME_UNDRIVEN: &str = "blockd: no NVMe controller this row names is on this machine"; /// What `loader.log` must carry, which is a different file and a different /// reader. @@ -140,9 +140,8 @@ pub const NVME_NO_WRITES: &[&str] = &["write=0", "io-other=0"]; /// that pass's `|` prefix. pub const LOADER_RECORDS: &[Record] = &[ Record { about: "chain", needle: "the last boot read DONE", presence: Says }, - // The stop's census and every disk's write cache emptied before anything - // was taken down: records of the stop, sealed in its tail. - Record { about: "nvme-census", needle: NVME_CENSUS, presence: Says }, + // Every disk's write cache emptied before anything was taken down: a + // record of the stop, sealed in its tail. Record { about: "usb-flush", needle: "usb-quiesce: disk ", presence: Says }, Record { about: "usb-quiesce", needle: QUIESCE_HEAD, presence: Says }, // What the stop did about the command a device was inside when it ran. On @@ -312,8 +311,7 @@ pub fn inventory(log: &str) -> Vec { "usb-storage: ", "i8042: ", "hda: ", - "NVMe: ", - "nvme: commands ", + "blockd: ", "GOP: ", "PAT: ", ] { @@ -353,13 +351,6 @@ pub fn unmet(loader: &str, log: &str) -> Vec { _ => {} } } - if let Some(census) = loader.lines().find(|l| l.contains(NVME_CENSUS)) { - for want in NVME_NO_WRITES { - if !census.contains(want) { - out.push(format!("nvme-census: {census:?}, and this boot owed {want}")); - } - } - } match quiesced(loader) { Some(said) if !said.complete() => out.push(format!( "usb-quiesce: the shutdown left a device behind: {said:?}. A reset with a transfer \ @@ -396,7 +387,7 @@ mod tests { line("0.090", "xHCI: max_slots=32 max_ports=16 ctx_size=64 pagesize=0x1"), line("0.140", "usb-storage: 1 device(s)"), line("0.150", "i8042: ok selftest=0x55 cfg=0x45->0x44 port1=ok port2=ok"), - line("0.200", "NVMe: NS1 size=1000215216 sectors, sector_size=512"), + line("0.200", "blockd: no NVMe controller this row names is on this machine; serving no partition"), line("0.319", "GOP: scanout memory type WC (MTRR WB, PAT entry 4)"), line("0.368", "Boot: complete (368ms)"), line("1.100", "exit: usbwrite pid=6 code=402000 cpu=140ms"), @@ -415,8 +406,6 @@ mod tests { | log: this boot's newest records follow, newest first (16)\n\ | log-tail: [kernel 3.960 cpu0] Rebooting.\n\ | log-tail: [kernel 3.955 cpu0] usb-quiesce: disk 0 SYNCHRONIZE CACHE ok\n\ - | log-tail: [kernel 3.950 cpu0] nvme: commands identify=2 admin-other=2 read=1 write=0 \ - io-other=0\n\ | usb-quiesce: no Bulk-Only command was open, so this reset cuts none\n\ | usb-quiesce: xHCI 00:14.0 halted=true USBSTS=0x00000009\n\ | usb-quiesce: 2/2 disk cache(s) flushed, 1 with no cache to flush, \ @@ -501,9 +490,10 @@ mod tests { // The scanout mapped as something other than write-combining. let uncached = a_good_boot().replace("memory type WC ", "memory type UC "); assert_eq!(about(unmet(&a_good_loader(), &uncached)), ["scanout"]); - // One write to the disk this project never writes. - let wrote = a_good_loader().replace("write=0 io-other=0", "write=1 io-other=0"); - assert_eq!(about(unmet(&wrote, &a_good_boot())), ["nvme-census"]); + // A partition of the internal disk served, which is where a write to + // it would come from. + let served = format!("{}{}", a_good_boot(), line("0.3", "blockd: partition 0000 at block 256")); + assert_eq!(about(unmet(&a_good_loader(), &served)), ["nvme-served"]); // A shutdown that never emptied a cache. let unflushed = a_good_loader() .lines() diff --git a/tests/common/devices.rs b/tests/common/devices.rs index 7bafdc6efc5..d6f531af418 100644 --- a/tests/common/devices.rs +++ b/tests/common/devices.rs @@ -83,9 +83,9 @@ fn measurement(job: &str, code: i32) -> Result { /// Every device here is emulated and every one of them is faster or slower than /// the laptop by an amount nobody has measured, so **no span is judged**. What /// is judged is that the boot ran its whole job list, that each job answered -/// with a measurement rather than a refusal, that the NVMe census counted no -/// write, and that the shutdown emptied a cache and halted a controller before -/// it let the reset go — which is a control for the sequence and not for its +/// with a measurement rather than a refusal, that blockd served this guest's +/// NVMe and fsd its DATA, and that the shutdown emptied a cache and halted a +/// controller before it let the reset go — which is a control for the sequence and not for its /// timing. pub fn metal_device_probe( _test_config: &Path, @@ -123,42 +123,30 @@ pub fn metal_device_probe( } // **The positive control for the safety instrument**, and this machine is - // the only one that can give it. On the T14 the internal disk reads - // `Foreign` (`kernel/src/bcachefs_adapter.rs`: "Never written to, under any - // circumstances") and the metal judge asserts `write=0`; a counter that - // was stuck at zero would pass that and say nothing. Here the guest is - // given an NVMe carrying a ToyOS volume of its own, mounts it, and writes - // it — so the census has to *move*, and the mount record is what accounts - // for the writes by name. - // Either arm of `Storage`'s two writable ones, because which the guest - // finds depends on whether an earlier boot in this lane already formatted - // the disk — and both of them write it. + // the only one that can give it. On the T14 the metal judge asserts that + // blockd drives no NVMe there (`metaldevices::NVME_UNDRIVEN`) and serves no + // partition; a judge whose needles could never appear would pass that and + // say nothing. Here blockd's row names this guest's controller, so the + // lines the T14 must never carry have to appear: blockd serves the disk's + // partitions, and fsd mounts or formats the DATA volume on one of them — + // either, because which it finds depends on whether an earlier boot in + // this lane already formatted the disk, and both of them write it. + if text.contains(metaldevices::NVME_UNDRIVEN) || !text.contains("blockd: partition ") { + return Err(format!( + "blockd served no partition of this guest's NVMe, so the needles the T14's judge \ + refuses are ones this machine never showed it can print:\n{text}" + )); + } const OWNED: &[&str] = &[ - "storage: mounted the ToyOS volume at block 0", - "storage: block 0 designates this device for ToyOS", + "fsd: mounted the DATA volume", + "fsd: block 0 designates this partition for ToyOS; formatting it", ]; if !OWNED.iter().any(|said| text.contains(said)) { return Err(format!( - "this guest's NVMe read as neither of {OWNED:?}, so it owns no volume there and the control below would be asserting the wrong thing" + "this guest's DATA read as neither of {OWNED:?}, so it owns no volume there and the \ + control would be asserting the wrong thing:\n{text}" )); } - let census = text - .lines() - .find(|l| l.contains(metaldevices::NVME_CENSUS)) - .ok_or_else(|| format!("no {:?} record", metaldevices::NVME_CENSUS))?; - for field in ["read=", "write="] { - let counted: u64 = census - .split(field) - .nth(1) - .and_then(|rest| rest.split_whitespace().next()) - .and_then(|v| v.parse().ok()) - .ok_or_else(|| format!("the census names no {field}: {census:?}"))?; - if counted == 0 { - return Err(format!( - "this guest owns the NVMe volume and mounted it, so the census owes a non-zero {field} — a counter that cannot move would pass the T14's `write=0` and mean nothing: {census:?}" - )); - } - } // And the shutdown, in the order a device needs it: the cache is emptied // while the volume is still there, and the boot's own last word is still // the last thing in the log. diff --git a/tests/common/faults.rs b/tests/common/faults.rs index a00c2e86b56..04bc78a0d4c 100644 --- a/tests/common/faults.rs +++ b/tests/common/faults.rs @@ -431,14 +431,9 @@ pub fn refused_claim(log: &Serial, claims: &str, why: &str) -> Result<(), String Ok(()) } -/// A machine with no NVMe controller must boot. -/// -/// `.expect("NVMe: no controller found")` killed it at 0.08 s — before -/// storage, before a console on the target laptop, and with the screen still -/// showing whatever the last checkpoint painted. It is the same class M1 -/// closed for xHCI, on a different controller, and the same class the -/// designation stamp closed one layer up: absence of storage is a -/// configuration, not a failure. +/// A machine with no NVMe controller must boot, and its block service and +/// file servers serve what they have: absence of storage is a configuration, +/// not a failure. pub fn diskless_boot( test_config: &Path, c_bins: &[(String, Vec)], @@ -468,7 +463,8 @@ pub fn diskless_boot( // scan is a claim about nothing again. log.must_be_clean()?; log.must_not_say("no controller found")?; - log.must_say("NVMe: no controller on this machine")?; + log.must_say("blockd: no NVMe controller this row names is on this machine; serving no partition")?; + log.must_say("fsd: this machine has no DATA partition;")?; log.must_say("Boot: complete")?; Ok(()) } diff --git a/tests/common/gpt.rs b/tests/common/gpt.rs index 5b470c94a0d..65db32e2f83 100644 --- a/tests/common/gpt.rs +++ b/tests/common/gpt.rs @@ -3,15 +3,15 @@ //! The parser's own reasoning is host-tested inside `toyos-gpt/`, over crafted //! tables and every hostile field, and none of that needs a guest. What only a //! guest can answer is whether the *identity* survives the trip — OVMF's -//! device path, the bootloader, `KernelArgs`, the kernel's NVMe driver — and -//! whether the parser finds that identity on a table it did not author. +//! device path, the bootloader, `KernelArgs`, the kernel's USB storage driver +//! — and whether the parser finds that identity on a table it did not author. //! //! Ground truth is the disk image, read on the host by the `gpt` crate, which //! is a different implementation from the one under test. The guest's own //! account of the partition it booted from is exactly what is in question, so //! it cannot also be the reference. //! -//! The table on the NVMe disk is built to be adversarial in the two ways that +//! The table on a second USB disk is built to be adversarial in the two ways that //! matter. Its *first* entry is an ESP by type — so a matcher keying on the //! type GUID, or taking the first partition, or taking the first ESP, gets a //! different span than the one asserted. And the second boot moves the @@ -50,7 +50,7 @@ pub fn boot_partition_identity( let config = repo.join("tests/metalcase/system.toml"); let dir = super::lane::dir(); - // Built here rather than by `boot_with_options`, because the crafted NVMe + // Built here rather than by `boot_with_options`, because the crafted // table below has to carry this image's partition GUID and the image does // not exist until it is built. `create_gpt_disk` draws a fresh random GUID // every time, so there is no second build that would agree with this one. @@ -74,8 +74,9 @@ pub fn boot_partition_identity( // Positive: the matching entry is third, behind an ESP-typed decoy, and // sits exactly where firmware says it does. - let agreeing = dir.join("gpt-nvme-agree.img"); + let agreeing = dir.join("gpt-decoy-agree.img"); craft_decoy_disk(&agreeing, &esp, 0)?; + let untouched = toyos_build::fingerprint::whole_device(&agreeing); let log = boot(&config, &boot_image, &agreeing)?; let firmware = format!( @@ -94,26 +95,24 @@ pub fn boot_partition_identity( // Entry 2 of 3 is the whole assertion: the decoy at entry 0 is an ESP too, // so an index or a type would both have answered 0. let carries = format!( - "gpt: device 1 carries the boot partition at LBA {}+{} (512-byte blocks), entry 2 of 3", + "carries the boot partition at LBA {}+{} (512-byte blocks), entry 2 of 3", esp.first_lba, esp.blocks() ); - if !log.contains(&carries) { + let Some(decoy) = device_saying(&log, &carries) else { return Err(format!( "the kernel did not find the boot partition where the table put it.\nwanted: \ {carries}\n{}", gpt_lines(&log) )); - } + }; // And then the arm nothing else reaches. The stick this guest booted from // is on the bus and carries the same partition — it is the real one, and - // the crafted NVMe entry above is a clone of it. Two devices claiming one + // the crafted entry above is a clone of it. Two devices claiming one // unique partition GUID is the state `Resolution::Ambiguous` exists for, // and the only safe answer is that this machine has no boot volume at all. - // Nothing tested that until `fat32_adapter::probe_boot_disks` started - // asking the USB bus as well. - if !log.contains("carries the same partition GUID as device 1") { + if !log.contains("carries the same partition GUID as device ") { return Err(format!( "a second device carrying the boot partition GUID did not make the answer \ ambiguous.\n{}", @@ -127,22 +126,11 @@ pub fn boot_partition_identity( )); } - // A boot partition on a disk is not consent to write the disk. This one - // carries no TOYOS-DATA partition, so the kernel takes no volume off it and - // says which count it saw; `foreign_disk_untouched` is where a ToyOS-typed - // partition that is somebody else's is refused at block 0. - if !log.contains("TOYOS-DATA partitions, and a data volume is one") { - return Err(format!( - "the kernel did not say what it found on a disk carrying our boot partition and \ - no data volume:\n{}", - gpt_lines(&log) - )); - } - if log.contains("formatting it") { - return Err(format!( - "finding our boot partition on a disk made the kernel format it:\n{}", - gpt_lines(&log) - )); + // A boot partition on a disk is not consent to write the disk. + if let Some(diff) = + toyos_build::fingerprint::first_difference(&untouched, &toyos_build::fingerprint::whole_device(&agreeing)) + { + return Err(format!("finding our boot partition on a disk wrote the disk: {diff}")); } if !log.contains("Boot: complete") { return Err(format!("the boot did not complete:\n{log}")); @@ -152,12 +140,12 @@ pub fn boot_partition_identity( // it. Two accounts of one partition that disagree means this is not the // disk firmware read, and the next thing anyone does with a boot volume is // write to it. - let disagreeing = dir.join("gpt-nvme-shifted.img"); + let disagreeing = dir.join("gpt-decoy-shifted.img"); craft_decoy_disk(&disagreeing, &esp, 8)?; let log = boot(&config, &boot_image, &disagreeing)?; let refused = format!( - "gpt: device 1 puts {} at LBA {}+{} but firmware said {}+{}", + "gpt: device {decoy} puts {} at LBA {}+{} but firmware said {}+{}", esp.guid, esp.first_lba + 8, esp.blocks() - 8, @@ -171,7 +159,7 @@ pub fn boot_partition_identity( gpt_lines(&log) )); } - if log.contains("gpt: device 1 carries the boot partition") { + if log.contains(&format!("gpt: device {decoy} carries the boot partition")) { return Err(format!( "the kernel claimed a boot volume it had just refused:\n{}", gpt_lines(&log) @@ -181,7 +169,7 @@ pub fn boot_partition_identity( // partition, which is on the USB bus and where firmware said it was — and // with the decoy refused there is no second claimant, so this boot *does* // have a boot volume where the agreeing one above does not. - if !log.contains("gpt: device 16 carries the boot partition") { + if device_saying(&log, "carries the boot partition at LBA ").is_none_or(|stick| stick == decoy) { return Err(format!( "refusing the shifted decoy cost the boot partition on the stick.\n{}", gpt_lines(&log) @@ -201,24 +189,29 @@ pub fn boot_partition_identity( Ok(()) } -fn boot( - config: &Path, - boot_image: &Path, - nvme_image: &Path, -) -> Result { +/// The device a `gpt: device N …` line carrying `what` names. +fn device_saying(log: &str, what: &str) -> Option { + log.lines() + .filter(|l| l.contains(what)) + .find_map(|l| l.split("gpt: device ").nth(1)?.split(' ').next()?.parse().ok()) +} + +/// A boot off the stick with `decoy` on the bus ahead of it, so the decoy's +/// table is the first the kernel reads. +fn boot(config: &Path, boot_image: &Path, decoy: &Path) -> Result { let mut qemu = QemuInstance::boot_with_options( config.parent().expect("system.toml has a directory"), &[], &[], BootOptions { - profile: qemu::Profile::Metal, + profile: qemu::Profile::UsbDiskRefusedFirst, // Pristine, and this test is why that choice exists: it boots one // crafted image twice, and the loader counts an image's attempts // into a file on its own log partition before every handoff — so a // second launch that saw the first one's writes is a retry and // boots no kernel at all. boot_image: Some(qemu::Staged::Pristine(boot_image.to_path_buf())), - nvme_image: Some(nvme_image.to_path_buf()), + usb_images: vec![decoy.to_path_buf()], ..Default::default() }, ); diff --git a/tests/common/inspect.rs b/tests/common/inspect.rs index 56ff8564ce0..78f58823da9 100644 --- a/tests/common/inspect.rs +++ b/tests/common/inspect.rs @@ -6,9 +6,9 @@ //! matches too much is a path in the answer this file did not name, and one //! that matches too little is a named path missing from it. Values are judged //! where the machine fixes them — QEMU's user network leases `10.0.2.15/24`, -//! nothing plays audio until `inspect_plays` does, and the NVMe disk this file -//! crafts has one partition free and one init grants — and read only for shape -//! elsewhere. +//! nothing plays audio until `inspect_plays` does, and the boot stick carries +//! one partition the kernel holds, two file servers hold and one nobody does +//! — and read only for shape elsewhere. use std::collections::BTreeMap; use std::path::Path; @@ -28,11 +28,6 @@ pub const PLAYS: &str = "inspect_plays"; /// The guest binary that sends `SYS_DEVICE_INVENTORY` its edges. pub const BOUNDS: &str = "inventory_bounds"; -/// The crafted disk's partition nobody holds. -const FREE: &str = "9D1E2F30-4A5B-4C6D-8E7F-0A1B2C3D4E5F"; -/// The one init grants test-runner; mirrored in the config. -const GRANTED: &str = "B4C5D6E7-F809-4A1B-8C2D-3E4F5A6B7C8D"; - /// Every path the reader answers for netd on a virtio NIC with a lease. const NET: &[&str] = &[ "net.driver", @@ -57,16 +52,8 @@ pub fn boot(rust_bins: &[(String, Vec)]) -> Result { if bins.len() != 3 { return Err(format!("{DENIED}, {PLAYS} and {BOUNDS} were not all built")); } - let nvme = super::lane::dir().join("inspect-disk.img"); - let mib = 1024 * 1024; - super::partclaim::craft_plain_disk( - &nvme, - &[("free", mib, FREE), ("granted", mib, GRANTED)], - 96 * mib, - )?; let config = Path::new(env!("CARGO_MANIFEST_DIR")).join(CONFIG); - let options = - BootOptions { profile: qemu::Profile::Gop, nvme_image: Some(nvme), ..Default::default() }; + let options = BootOptions { profile: qemu::Profile::Gop, ..Default::default() }; let argv = qemu::profile_argv(&options); if !argv.iter().any(|a| a.contains("virtio-net")) || !argv.iter().any(|a| a.contains("virtio-sound")) { return Err("this test needs a virtio NIC and a virtio sound card".to_string()); @@ -304,26 +291,12 @@ fn holders<'a>(got: &'a BTreeMap, at: &str) -> Vec<&'a str> { got.iter().filter(|(p, _)| p.starts_with(&under)).map(|(_, v)| v.as_str()).collect() } -/// The `dev.disk..part` whose unique GUID is `unique`. -fn partition<'a>(line: &str, got: &'a BTreeMap, unique: &str) -> Result<&'a str, String> { - let unique = unique.to_ascii_lowercase(); - let found: Vec<&str> = got - .iter() - .filter(|(p, v)| p.starts_with("dev.disk.") && p.ends_with(".unique") && **v == unique) - .map(|(p, _)| p.trim_end_matches(".unique")) - .collect(); - match found.as_slice() { - [one] => Ok(one), - _ => Err(format!("`{line}`: {} partitions are {unique}, not one", found.len())), - } -} - /// `inspect dev.*`: the kernel's inventory, judged where QEMU fixes it. The /// virtio NIC is `1af4:1041` and netd holds it; the virtio sound card and the /// framebuffer are classes soundd and the compositor hold; the Gop profile's -/// USB keyboard is on the xHCI; the boot stick carries partitions this kernel -/// mounted; and of the crafted disk's two, one is free and test-runner holds -/// the other. +/// USB keyboard is on the xHCI; and the boot stick carries ROOT, which this +/// kernel holds, the ESP and the log partition, which file servers hold, and +/// the idle slot, which nobody does. fn inventory(qemu: &mut QemuInstance) -> Result<(), String> { let line = "inspect dev.*"; let got = answer(&job(qemu, line, 0)?); @@ -364,15 +337,18 @@ fn inventory(qemu: &mut QemuInstance) -> Result<(), String> { if !got.iter().any(|(p, v)| p.starts_with("dev.disk.") && p.ends_with(".state") && v == "kernel") { return Err(format!("`{line}`: no partition is held by the kernel")); } - let free = partition(line, &got, FREE)?; - expect(line, &got, &format!("{free}.state"), "free")?; - if !holders(&got, free).is_empty() { - return Err(format!("`{line}`: {free} is free and held by {:?}", holders(&got, free))); + let parts: Vec<&str> = got + .keys() + .filter(|p| p.starts_with("dev.disk.") && p.ends_with(".state")) + .map(|p| p.trim_end_matches(".state")) + .collect(); + let state = |part: &str| got.get(&format!("{part}.state")).map(String::as_str); + if !parts.iter().any(|p| state(p) == Some("free") && holders(&got, p).is_empty()) { + return Err(format!("`{line}`: no partition is free and held by nobody")); } - let granted = partition(line, &got, GRANTED)?; - expect(line, &got, &format!("{granted}.state"), "claimed")?; - if holders(&got, granted) != ["test-runner"] { - return Err(format!("`{line}`: {granted} is held by {:?}, not test-runner", holders(&got, granted))); + let by_fsd = parts.iter().filter(|p| state(p) == Some("claimed") && holders(&got, p) == ["fsd"]).count(); + if by_fsd != 2 { + return Err(format!("`{line}`: {by_fsd} partitions are claimed by fsd, not the ESP and the log partition")); } Ok(()) } diff --git a/tests/common/iommu.rs b/tests/common/iommu.rs index a2b92253803..cd17eba6105 100644 --- a/tests/common/iommu.rs +++ b/tests/common/iommu.rs @@ -906,7 +906,8 @@ const SECOND_LEVEL: &[&str] = &["read-permission", "write-permission", "paging-e /// passthrough, the controller below would go on working and this test would /// wait for a fault that never comes. /// -/// [`Profile::Metal`] because it has an NVMe controller and no virtio device. +/// [`Profile::Metal`] because it has an xHCI controller the kernel drives +/// from boot, and no virtio device. /// The distinction matters: QEMU gives a virtio device the bypassing address /// space unless it is created with `iommu_platform=on`, so a virtio-only /// machine could not tell a translating unit from an absent one however the @@ -916,31 +917,31 @@ pub fn iommu_context_absent( c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - let (log, blocked) = fault_boot(test_config, c_bins, rust_bins, &["iommu-context-absent"])?; + let (_qemu, log, blocked) = fault_boot(test_config, c_bins, rust_bins, &["iommu-context-absent"])?; // Which function the actuator left out is decided in the guest by class // code; which function that *is* on this machine is read here from the PCI // walk's own lines. Neither half is told the other's answer, so a fault // naming some other device — or the actuator skipping a device the walk // never saw — is a failure rather than a tautology. - let nvme = class_function(&log, "0108").ok_or_else(|| { - format!("this machine enumerated no NVMe controller to leave out\n{}", log.text()) + let xhci = class_function(&log, "0c03").ok_or_else(|| { + format!("this machine enumerated no xHCI controller to leave out\n{}", log.text()) })?; - if blocked.stream != nvme { + if blocked.stream != xhci { return Err(format!( - "the unit blocked {} but the controller left out of the root table is {nvme}", + "the unit blocked {} but the controller left out of the root table is {xhci}", blocked.stream )); } if blocked.reason != "context-entry-not-present" { return Err(format!( - "the unit blocked {nvme} for {:?}, and a function with no context entry should be \ + "the unit blocked {xhci} for {:?}, and a function with no context entry should be \ blocked for having none", blocked.reason )); } eprintln!( - " [iommu] {nvme} left out of the root table: blocked at {} on a {} for {}", + " [iommu] {xhci} left out of the root table: blocked at {} on a {} for {}", blocked.address, blocked.access, blocked.reason ); Ok(()) @@ -969,53 +970,54 @@ pub fn iommu_empty_domain( c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - let (log, blocked) = fault_boot(test_config, c_bins, rust_bins, &["iommu-empty-domain"])?; + let (qemu, log, blocked) = fault_boot(test_config, c_bins, rust_bins, &["iommu-empty-domain"])?; - let nvme = class_function(&log, "0108").ok_or_else(|| { - format!("this machine enumerated no NVMe controller to strand\n{}", log.text()) + let xhci = class_function(&log, "0c03").ok_or_else(|| { + format!("this machine enumerated no xHCI controller to strand\n{}", log.text()) })?; - if blocked.stream != nvme { + if blocked.stream != xhci { return Err(format!( - "the unit blocked {} but the controller given an empty domain is {nvme}", + "the unit blocked {} but the controller given an empty domain is {xhci}", blocked.stream )); } if !SECOND_LEVEL.contains(&blocked.reason.as_str()) { return Err(format!( - "the unit blocked {nvme} for {:?}, which is not something a second-level page table \ + "the unit blocked {xhci} for {:?}, which is not something a second-level page table \ walk decides. A present context entry over an empty domain has to be refused by the \ walk itself — any other reason means the unit stopped before it, and a context entry \ naming passthrough would not have walked at all", blocked.reason )); } - // Inside the memory the identity domain covers, because that is where the - // driver's descriptors are: a fault somewhere else would be a different - // machine's bug wearing this one's clothes. - let covered = identity_extent(&log)?; + // Inside the pool the driver's own `DCBAAP` begins, read out of the + // controller's registers, because that is where its descriptors are: a + // fault somewhere else would be a different machine's bug wearing this + // one's clothes. + let pool = dcbaa(qemu.qmp_socket(), &log, &xhci)?; let at = u64::from_str_radix(blocked.address.trim_start_matches("0x"), 16) .map_err(|_| format!("unreadable faulting address {:?}", blocked.address))?; - if at == 0 || at >= covered { + if pool == 0 || !(pool..pool + XHCI_POOL).contains(&at) { return Err(format!( - "the unit blocked an access to {}, and the driver's descriptors are inside \ - 0x0..{covered:#x}", + "the unit blocked an access to {}, and the driver's descriptors are in the \ + {XHCI_POOL:#x} bytes from its DCBAAP, {pool:#x}", blocked.address )); } eprintln!( - " [iommu] {nvme} given an empty domain: blocked at {} on a {} for {}", + " [iommu] {xhci} given an empty domain: blocked at {} on a {} for {}", blocked.address, blocked.access, blocked.reason ); Ok(()) } -/// One device aimed at the physical bytes NVMe's admin completion queue page -/// ends with — a page in another driver's pool, which the aimed device's own -/// domain does not map. Three things then hold at once and no two come from -/// the same place: the unit blocks it and names the device and that address; -/// the address is the one NVMe's own `ACQ` holds, resolved through the tables -/// the unit walks; and the function's bus mastering is gone. A write also -/// leaves bytes to check; a read leaves none. +/// One device aimed at the physical bytes the xHCI's device context base +/// address array page ends with — a page in another driver's pool, which the +/// aimed device's own domain does not map. Three things then hold at once and +/// no two come from the same place: the unit blocks it and names the device +/// and that address; the address is the one the xHCI's own `DCBAAP` holds, +/// resolved through the tables the unit walks; and the function's bus +/// mastering is gone. A write also leaves bytes to check; a read leaves none. struct ForeignArm { profile: Profile, params: &'static [&'static str], @@ -1114,7 +1116,7 @@ pub fn iommu_domain_isolation( Ok(()) } -/// A scanout backing in NVMe's pool, by its physical address. The device maps a +/// A scanout backing in the xHCI's pool, by its physical address. The device maps a /// backing when it is attached (`hw/display/virtio-gpu.c:918-931` at v11.1.0: /// `dma_memory_map` answers NULL for a translation the unit refused, and the /// command is answered `VIRTIO_GPU_RESP_ERR_UNSPEC` at `:1010-1014`), so the @@ -1203,8 +1205,8 @@ fn foreign_fault( let aimed = class_function(&log, arm.class).ok_or_else(|| { format!("this machine enumerated no class {} function to aim\n{}", arm.class, log.text()) })?; - let nvme = class_function(&log, "0108") - .ok_or_else(|| format!("this machine enumerated no NVMe controller\n{}", log.text()))?; + let xhci = class_function(&log, "0c03") + .ok_or_else(|| format!("this machine enumerated no xHCI controller\n{}", log.text()))?; if blocked.stream != aimed { return Err(format!( "the unit blocked {} and the device aimed at another driver's pool is {aimed}", @@ -1226,8 +1228,7 @@ fn foreign_fault( } let window = register_window(socket, &log, arm.name)?; - let acq = over_qmp(socket, nvme_bar(socket, &log, &nvme)? + NVME_ACQ, 1, 'g')?[0]; - let victim = translate(socket, window, &nvme, acq)?; + let victim = translate(socket, window, &xhci, dcbaa(socket, &log, &xhci)?)?; let at = u64::from_str_radix(blocked.address.trim_start_matches("0x"), 16) .map_err(|_| format!("unreadable faulting address {:?}", blocked.address))?; // The window and not the page, for the arm that aims a whole grant: the @@ -1238,7 +1239,7 @@ fn foreign_fault( if !(aimed_at..aimed_at + arm.blocked_within).contains(&at) { return Err(format!( "the unit blocked an access to {}, and what the actuator aimed {aimed} at is \ - {:#x}..{:#x} — the page NVMe's ACQ names, through the tables the unit walks. The \ + {:#x}..{:#x} — the page the xHCI's DCBAAP names, through the tables the unit walks. The \ kernel is not reporting the address the device was aimed at", blocked.address, aimed_at, @@ -1253,7 +1254,7 @@ fn foreign_fault( if let Some((i, word)) = words.iter().enumerate().find(|(_, w)| **w != 0) { return Err(format!( "the unit reported blocking {aimed} at {}, and the {} bytes at {probe:#x} inside \ - NVMe's pool hold {word:#018x} at word {i} rather than the zero the NVMe driver \ + the xHCI's pool hold {word:#018x} at word {i} rather than the zero the xHCI driver \ left. The write landed anyway", blocked.address, PROBE_WORDS * 8 @@ -1283,7 +1284,7 @@ fn foreign_fault( } } eprintln!( - " [iommu] {aimed} aimed at {}, inside {nvme}'s pool: blocked on a {} for {}{bytes}, and \ + " [iommu] {aimed} aimed at {}, inside {xhci}'s pool: blocked on a {} for {}{bytes}, and \ its COMMAND reads {command:#06x} — bus mastering gone", blocked.address, blocked.access, blocked.reason, ); @@ -1300,11 +1301,10 @@ fn clean_walk(clean: &QemuInstance, name: &str) -> Result, Stri log.must_say("init: started netd")?; let socket = clean.qmp_socket(); let window = register_window(socket, &log, name)?; - let nvme = class_function(&log, "0108") - .ok_or_else(|| format!("{name}: this machine enumerated no NVMe controller\n{}", log.text()))?; - let acq = over_qmp(socket, nvme_bar(socket, &log, &nvme)? + NVME_ACQ, 1, 'g')?[0]; - let owned = translate(socket, window, &nvme, acq)?; - domains_are_disjoint(socket, &log, window, owned, &nvme)? + let xhci = class_function(&log, "0c03") + .ok_or_else(|| format!("{name}: this machine enumerated no xHCI controller\n{}", log.text()))?; + let owned = translate(socket, window, &xhci, dcbaa(socket, &log, &xhci)?)?; + domains_are_disjoint(socket, &log, window, owned, &xhci)? .keys() .map(|bdf| class_of(&log, bdf)) .collect() @@ -1645,12 +1645,17 @@ fn config_space(log: &Serial, bdf: &str) -> Result { Ok(ecam + (u64::from(bus) << 20) + (u64::from(dev) << 15) + (u64::from(func) << 12)) } -/// The untouched half of NVMe's admin completion queue page: a frame is 1526 +/// The untouched half of the xHCI's DCBAA page: a frame is 1526 /// bytes at most, so this covers the whole of one landing there. const PROBE_WORDS: usize = 256; -/// `REG_ACQ`, NVMe 2.0 Figure 41. -const NVME_ACQ: u64 = 0x30; +/// The xHCI driver's DMA pool, which its DCBAA begins: the `dma 2048 KiB` its +/// bring-up line states. +const XHCI_POOL: u64 = 2 << 20; + +/// `DCBAAP`'s offset in the operational registers, xHCI 1.2 Table 5-18; those +/// begin `CAPLENGTH` bytes into BAR 0, §5.3.1. +const XHCI_DCBAAP: u64 = 0x30; /// Where QEMU puts an `intel-iommu` on q35, which every profile here is: a /// constant on the host side, not a number the guest supplies. @@ -1716,8 +1721,16 @@ fn register_window(socket: &Path, log: &Serial, name: &str) -> Result Result { + let bar = bar0(socket, log, bdf)?; + let caplength = over_qmp(socket, bar, 1, 'w')?[0] & 0xFF; + Ok(over_qmp(socket, bar + caplength + XHCI_DCBAAP, 1, 'g')?[0] & !0x3F) +} + /// A function's memory BAR 0, out of ECAM rather than off a console line. -fn nvme_bar(socket: &Path, log: &Serial, bdf: &str) -> Result { +fn bar0(socket: &Path, log: &Serial, bdf: &str) -> Result { let config = config_space(log, bdf)?; // A window that decodes at all: an ECAM base the kernel invented would read // back all ones here, which is no vendor id. @@ -1792,7 +1805,7 @@ fn fault_boot( c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], params: &'static [&'static str], -) -> Result<(Serial, Blocked), String> { +) -> Result<(QemuInstance, Serial, Blocked), String> { let mut qemu = QemuInstance::boot_with_options( test_config, c_bins, @@ -1801,6 +1814,7 @@ fn fault_boot( profile: Profile::Metal, kernel_params: params, ready_marker: FAULT, + qmp: true, ..Default::default() }, ); @@ -1813,7 +1827,7 @@ fn fault_boot( log.must_not_say(qemu::DEFAULT_READY)?; let blocked = blocked_on(log.must_say(FAULT)?)?; - Ok((log, blocked)) + Ok((qemu, log, blocked)) } /// The fault line's fields; a reason the kernel has no name for is refused. @@ -2042,7 +2056,7 @@ fn argv_check(profile: Profile, argv: &[String]) -> Result<(), String> { /// second, and one that ignored the fault would fail the first. /// /// The stimulus is `iommu-userdev-foreign-dma`: the kernel answers netd's first -/// DMA grant with an address inside NVMe's pool, which the NIC's own domain +/// DMA grant with an address inside the xHCI's pool, which the NIC's own domain /// does not map. netd is unmodified and does with that address exactly what it /// does with a correct one, so what the device is pointed at is a real /// descriptor holding a wrong address rather than a driver written to misbehave. @@ -2159,7 +2173,7 @@ pub fn userdev_residue_is_its_own( /// The `pcidev` slot the function with PCI ids `ids` (`[vvvv:dddd]`) was handed /// over on: a boot's claims are minted in init's order, and the block service's -/// comes first on a machine with an NVMe controller. +/// comes first on a machine with an NVMe controller its row names. pub(crate) fn slot_of(text: &str, ids: &str) -> Result { text.lines() .find_map(|l| { diff --git a/tests/common/partclaim.rs b/tests/common/partclaim.rs index 7828bc1b2c7..d2c648f4d6b 100644 --- a/tests/common/partclaim.rs +++ b/tests/common/partclaim.rs @@ -6,16 +6,18 @@ //! the claim in question — so the verdict is read here, after the guest has //! gone, off the images: //! -//! - every byte of the NVMe disk outside the target and DATA is the byte this +//! - every byte of the crafted disk outside the target and DATA is the byte this //! file wrote: the primary and backup tables, both neighbours, the granted, -//! two misaligned and the twin partitions, and the gaps; +//! two misaligned partitions and the twin of the stick's log partition, and +//! the gaps; //! - both neighbours are FAT32 volumes `toyos-fat32-check` (fatgen103's rules) //! has nothing to say about — the neighbour after the target begins at the //! block after its last, so a write one past the end lands in its boot //! sector; //! - every block of the target is the pattern the guest wrote there; -//! - the `/home` file written between the target's transfers reads back -//! through the host's own bcachefs reader; +//! - the `/home` file fsd wrote between the target's transfers, through its +//! own claim on the same disk, reads back through the host's own bcachefs +//! reader; //! - after a departure, the stick holds what the guest wrote again. //! //! The partition ranges are UEFI 2.10 §5.3.3's, as the `gpt` crate — not the @@ -36,9 +38,10 @@ const GRANTED: &str = "A94F0E6D-3B2C-4E1A-8C7D-6E5F4A3B2C1D"; const MISALIGNED: &str = "3E8A1C5F-7D2B-4F60-9A1E-5C4B3D2E1F07"; /// Mirrored: a partition of whole 4 KiB blocks that begins inside one. const MISSTART: &str = "5A7C9E1B-3D5F-4B71-8C2E-4F6A8B0C2D35"; -/// Mirrored: the unique GUID the NVMe disk and the USB stick both carry. +/// The twin partition's unique GUID where no boot stick's is copied: the +/// crafted disk of the boots that judge no twin. const TWIN: &str = "6D2F9B41-8C3E-4A57-B1D0-2E4F6A8C0B13"; -/// Mirrored: DATA, which the kernel mounts at `/home`. +/// Mirrored: DATA, which fsd serves `/home` from. const DATA: &str = "E3A7C5D9-1B2F-4E6A-8D0C-9F7B5A3E1C24"; /// Mirrored: the partitions of the stick whose device leaves. const DEPARTING: &str = "1F3E5D7C-9B2A-4C6E-8F01-A3B5C7D9E2F4"; @@ -56,14 +59,13 @@ const GRANTED_BLOCKS: u64 = 256; const HOME_FILE: &str = "home/partclaim-interleaved.bin"; const HOME_CHUNK: usize = 32 * 1024; const PAST_END: &[u8; 16] = b"TOYOS-PAST-END\0\0"; -/// The claims the guest expects refused: four mounted partitions, the one init -/// granted test-runner, a twin, one whose length and one whose start is not whole -/// blocks, an absent GUID, the zero GUID, three -/// claims carrying selector words their class does not read, the target a -/// second time, and the target while a child holds it. -const REFUSALS: usize = 15; -/// The ESP, the log partition, ROOT and DATA. -const MOUNTED: usize = 4; +/// The claims the guest expects refused: ROOT, the two partitions init minted +/// for file servers, the one init granted test-runner, the log partition two +/// disks carry, one whose length and one whose start is not whole blocks, an +/// absent GUID, the zero GUID, three claims carrying selector words their +/// class does not read, the target a second time, and the target while a +/// child holds it. +const REFUSALS: usize = 14; const BLOCK: u64 = 4096; const MIB: u64 = 1024 * 1024; @@ -114,8 +116,11 @@ struct Layout { data: Span, } -/// The claims, refusals, idle ROOT slot, releases and neighbours, on the -/// machine this suite flashes plus a second USB stick for the twin GUID. +/// The claims, refusals, idle ROOT slot, releases and neighbours, on a +/// machine booting off its USB stick with the crafted disk beside it on the +/// bus. The crafted disk carries a copy of the stick's log partition's unique +/// GUID, so two disks the kernel drives name one partition: that claim is +/// refused, and no file server is handed the log. pub fn partition_claim( _test_config: &Path, c_bins: &[(String, Vec)], @@ -126,11 +131,8 @@ pub fn partition_claim( std::fs::write(&boot_image, qemu::build_boot_image(&config, c_bins, rust_bins, &[])) .map_err(|e| format!("write the boot image: {e}"))?; let [esp, log, root] = boot_stick_guids(&boot_image)?; - let nvme = super::lane::dir().join("partclaim-disk.img"); - let layout = craft_nvme(&nvme)?; - let stick = super::lane::dir().join("partclaim-stick.img"); - let (stick_bytes, _) = qemu::Profile::UsbDisk.usb_disk().expect("UsbDisk declares a disk"); - let twin_on_stick = craft_stick(&stick, stick_bytes, &[("twin", MIB, TWIN)])?[0]; + let crafted = super::lane::dir().join("partclaim-disk.img"); + let layout = craft_disk(&crafted, &log)?; // The premises, checked rather than assumed: a neighbour that did not // begin where the target ends would let a write past the end land in a @@ -146,7 +148,7 @@ pub fn partition_claim( if layout.target.len != TARGET_BLOCKS * BLOCK { return Err(format!("the target is {} bytes, not {TARGET_BLOCKS} blocks", layout.target.len)); } - let before = std::fs::read(&nvme).map_err(|e| format!("read the crafted disk: {e}"))?; + let before = std::fs::read(&crafted).map_err(|e| format!("read the crafted disk: {e}"))?; for (what, span) in [("first", layout.before), ("second", layout.after)] { let complaints = toyos_fat32_check::check(span.of(&before)); if !complaints.is_empty() { @@ -156,7 +158,6 @@ pub fn partition_claim( )); } } - let twin_before = read_span(&stick, twin_on_stick)?; let mut qemu = QemuInstance::boot_with_options( &config, @@ -165,15 +166,16 @@ pub fn partition_claim( BootOptions { profile: qemu::Profile::UsbDisk, boot_image: Some(Staged::Pristine(boot_image.clone())), - nvme_image: Some(nvme.clone()), - usb_images: vec![stick.clone()], + usb_images: vec![crafted.clone()], ..Default::default() }, ); let boot = qemu.boot_log().to_string(); no_panic("booting the claim disks", &boot)?; - if boot.contains("are a tmpfs") { - return Err(format!("/home fell back to tmpfs, so nothing shares the disk:\n{boot}")); + // Formatted, on a first boot of the crafted disk: its DATA carries the + // designation, and the readback below is what says it was this disk's. + if !boot.contains("fsd: block 0 designates this partition for ToyOS; formatting it") { + return Err(format!("fsd never formatted DATA off the crafted disk, so nothing shares it:\n{boot}")); } // init names what it could not mint; a grant it refused would make the // guest's endowment about nothing. @@ -187,10 +189,10 @@ pub fn partition_claim( let run = format!("test_rs_partition_claimant main {esp} {log} {root}"); let result = qemu.run_test(&run, Duration::from_secs(180)); let tail = shut_down(qemu); - let guest = guest_verdict(&result, &tail, REFUSALS).and_then(|kernel| main_kernel_lines(&kernel)); + let guest = guest_verdict(&result, &tail, REFUSALS).and_then(|kernel| main_kernel_lines(&kernel, &log)); no_panic("on the way down", &tail)?; - let after = std::fs::read(&nvme).map_err(|e| format!("read the disk back: {e}"))?; + let after = std::fs::read(&crafted).map_err(|e| format!("read the disk back: {e}"))?; let neighbours = neighbours_untouched(&layout, &before, &after); if guest.is_err() || !neighbours.is_empty() { return Err(format!( @@ -199,13 +201,10 @@ pub fn partition_claim( if neighbours.is_empty() { "untouched".to_string() } else { neighbours.join("\n") } )); } - if read_span(&stick, twin_on_stick)? != twin_before { - return Err("the twin partition on the stick changed, and nothing may write it".into()); - } target_holds_the_pattern(&layout, &after)?; - home_file_reads_back(&nvme)?; + home_file_reads_back(&crafted)?; - for path in [&nvme, &stick, &boot_image] { + for path in [&crafted, &boot_image] { let _ = std::fs::remove_file(path); } eprintln!( @@ -220,14 +219,15 @@ pub fn partition_claim( /// The two exits of a claim that gets no answer: a disk that does not answer a /// read of its table refuses the claim rather than resolving it on the disks /// that did, and a transfer every attempt of which is refused on its budget -/// ends at the deadman with the device's word. +/// ends at the deadman with the device's word. The crafted disk is a second +/// USB stick beside the boot stick. pub fn partition_claim_gives_up( _test_config: &Path, c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { let config = super::compile::repo_root().join(CONFIG); - let nvme = super::lane::dir().join("partclaim-gives-up.img"); + let crafted = super::lane::dir().join("partclaim-gives-up.img"); let cases: [(&'static [&'static str], &str, usize, &[&str]); 2] = [ ( &["partclaim-table-unanswered"], @@ -246,14 +246,14 @@ pub fn partition_claim_gives_up( ), ]; for (params, role, refusals, wants) in cases { - craft_nvme(&nvme)?; + craft_disk(&crafted, TWIN)?; let mut qemu = QemuInstance::boot_with_options( &config, c_bins, rust_bins, BootOptions { - profile: qemu::Profile::Metal, - nvme_image: Some(nvme.clone()), + profile: qemu::Profile::UsbDisk, + usb_images: vec![crafted.clone()], kernel_params: params, ..Default::default() }, @@ -275,7 +275,7 @@ pub fn partition_claim_gives_up( eprintln!(" [partclaim] {role}: {}", line.trim()); } } - let _ = std::fs::remove_file(&nvme); + let _ = std::fs::remove_file(&crafted); Ok(()) } @@ -428,19 +428,17 @@ fn guest_verdict(result: &qemu::TestResult, tail: &str, refusals: usize) -> Resu Ok(kernel) } -/// The kernel's own account of the main run: its hold named once for each -/// mounted partition refused, the twin and both misaligned partitions refused -/// by name, and not one line for a transfer refused past the end — a caller -/// can ask at syscall rate. -fn main_kernel_lines(kernel: &str) -> Result<(), String> { +/// The kernel's own account of the main run: its hold on ROOT named once, the +/// log partition's twin and both misaligned partitions refused by name, and +/// not one line for a transfer refused past the end — a caller can ask at +/// syscall rate. +fn main_kernel_lines(kernel: &str, twin: &str) -> Result<(), String> { let held = kernel.matches("is held by the kernel").count(); - if held != MOUNTED { - return Err(format!( - "the kernel named its own hold {held} times for {MOUNTED} mounted partitions:\n{kernel}" - )); + if held != 1 { + return Err(format!("the kernel named its own hold {held} times for ROOT alone:\n{kernel}")); } for want in [ - format!("partclaim: {TWIN} is on device "), + format!("partclaim: {twin} is on device "), format!("partclaim: {MISALIGNED} is at "), format!("partclaim: {MISSTART} is at "), ] { @@ -602,7 +600,7 @@ pub(super) fn read_span(path: &Path, span: Span) -> Result, String> { /// One partition a crafted table carries: its name, length in bytes, type, /// unique GUID, and the boundary it begins on in 512-byte LBAs. -pub(super) type Part = (&'static str, u64, &'static str, &'static str, u64); +pub(super) type Part<'a> = (&'static str, u64, &'static str, &'a str, u64); /// Where every partition but one begins. pub(super) const ALIGNED: u64 = MIB / 512; @@ -657,10 +655,11 @@ pub(super) fn table(path: &Path, bytes: u64, parts: &[Part]) -> Result<(Box Result { +/// blocks, one that begins inside a block, a partition carrying `twin` as its +/// unique GUID, and a DATA fsd formats and serves `/home` from. +fn craft_disk(path: &Path, twin: &str) -> Result { const FAT_BYTES: u64 = 34 * MIB; const DATA_BYTES: u64 = 96 * MIB; let parts: [Part; 8] = [ @@ -672,7 +671,7 @@ fn craft_nvme(path: &Path) -> Result { // Right after the one above, at the first LBA past its odd length: whole // blocks long, and beginning 512 bytes into one. ("misaligned start", MIB, PLAIN_TYPE, MISSTART, 1), - ("twin", MIB, PLAIN_TYPE, TWIN, ALIGNED), + ("twin", MIB, PLAIN_TYPE, twin, ALIGNED), ("ToyOS data", DATA_BYTES, toyos_gpt::Guid::TOYOS_DATA_TEXT, DATA, ALIGNED), ]; let total = MIB + parts.iter().map(|p| p.1.next_multiple_of(MIB)).sum::() + 2 * MIB; @@ -694,7 +693,7 @@ fn craft_nvme(path: &Path) -> Result { Ok(layout) } -/// The designation on `data`: the kernel formats a DATA only on this consent. +/// The designation on `data`: fsd formats a DATA only on this consent. fn designate(device: &mut dyn gpt::DiskDevice, data: Span) -> Result<(), String> { let mut stamp = [0u8; BLOCK as usize]; stamp[..bcachefs::DESIGNATION_MAGIC.len()].copy_from_slice(&bcachefs::DESIGNATION_MAGIC); diff --git a/tests/common/power.rs b/tests/common/power.rs index 509f0cf9680..d740f9755b9 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -449,18 +449,12 @@ pub fn quiesce_dump_holds_the_stopped( /// other thread still runs. The job reads the kernel's word that the stop /// waits there and makes the second call — `SYS_SHUTDOWN` against the first's /// `SYS_REBOOT`, so the claim is judged on both syscalls — and starts the -/// thread the stop waits for only once refused by name. The same boot has the -/// stop's own drain refused in its retry ladder over a flush the job left -/// owed, the first call's work after the window. +/// thread the stop waits for only once refused by name. pub fn quiesce_refuses_a_second_shutdown( _test_config: &Path, _c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - // The kernel's `mirror_refuse::SHUTDOWN_REFUSALS`, spelt here because the - // harness cannot link the kernel. - const REFUSALS: usize = 8; - const REFUSED: &str = "quiesce-drain-refuse: refusing the shutdown drain's"; const WAITS: &str = "quiesce-last-park: the stop waits for"; const SECOND_CALLER: &str = "power: this machine is already stopping"; const SYNCING: &str = "Syncing filesystems..."; @@ -468,12 +462,10 @@ pub fn quiesce_refuses_a_second_shutdown( "quiesce-last-park: {} is held until the stop waits on it alone", toyos_quiesce::LAST_THREAD ); - // `writeback-stall` parks `iod`, so the closed file's flush is the stop's - // own drain's to find and no other drainer's to hold. let (whole, _record) = stopped_boot( "tests/quiescetwicecase/system.toml", "quiesce_twice", - &["writeback-stall", "quiesce-drain-refuse", "quiesce-last-park", LATE_WORD], + &["quiesce-last-park", LATE_WORD], rust_bins, )?; let lines: Vec<&str> = whole.lines().collect(); @@ -508,16 +500,6 @@ pub fn quiesce_refuses_a_second_shutdown( in the window the first call held\n{whole}" )); } - // And the first call's own drain met its ladder, after the window. - let refusals = at(REFUSED); - if refusals.len() != REFUSALS || refusals[0] < synced { - return Err(format!( - "the quiesce-drain-refuse actuator refused the stop's drain {} time(s), not the \ - {REFUSALS} the kernel declares after its sync, so the first call's drain was not \ - parked where this boot says it was\n{whole}", - refusals.len(), - )); - } eprintln!( " [power] the second caller was refused while the first held the stop:\n {}\n {}", lines[waits], lines[second], diff --git a/tests/common/storage.rs b/tests/common/storage.rs index bc0f65ca82e..fb061c64fe0 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -605,6 +605,87 @@ pub fn home_overwrite_reads_back( Ok(()) } +/// A file server killed with a write done and unanswered loses nothing a +/// client was told was flushed, and its clients go on; ended past init's +/// budget, its directories answer `Gone`. Judged off the device. +/// +/// `tests/fsdrestartcase` arms every file server to end under a write to +/// `/home/fsd_end` (`--end-on`), and `test_rs_fs_restart` ends DATA's four +/// times: the guest asserts what a client sees, init's and fsd's own lines +/// say who ended and who started again, and with the machine down the DATA +/// partition is read by this crate's own build of the `bcachefs` reader over +/// a plain seek-and-read of the image — nothing the guest executed. The +/// flushed file and the one flushed across the end hold their bytes there. +pub fn fsd_restart( + _test_config: &Path, + c_bins: &[(String, Vec)], + rust_bins: &[(String, Vec)], +) -> Result<(), String> { + /// Mirrored in `tests/toyos-rust-tests/src/bin/fs_restart.rs`, without the + /// mount point. + const KEPT: &str = "home/fs_restart/kept"; + const ACROSS: &str = "home/fs_restart/across"; + const KEPT_LEN: usize = 64 * 1024 + 13; + const ACROSS_BYTES: &[u8] = b"written and flushed before the server ended; written through the same \ + handle after it came back"; + let kept: Vec = (0..KEPT_LEN).map(|i| (i.wrapping_mul(37) ^ 0xC3) as u8).collect(); + + let config = super::compile::repo_root().join("tests/fsdrestartcase"); + let mut qemu = QemuInstance::boot_with_options( + &config, + c_bins, + rust_bins, + BootOptions { profile: qemu::Profile::Metal, ..Default::default() }, + ); + let boot = qemu.boot_log().to_string(); + if !boot.contains(MOUNTED) && !boot.contains("formatting it") { + return Err(format!("fsd served DATA from no partition, so nothing here reaches a device:\n{boot}")); + } + let result = qemu.run_test("test_rs_fs_restart", Duration::from_secs(120)); + let image = qemu.nvme_image().to_path_buf(); + writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); + qemu.flush_stdin(); + let tail = qemu.drain_serial(Duration::from_secs(20)); + drop(qemu); + // The console once: `stdout` is the same lines again, unprefixed. + let log = format!("{boot}\n{}{}{tail}", result.before, result.serial); + if result.exit_code != Some(0) || !result.stdout.contains("fs_restart: PASS") { + return Err(format!("fs_restart guest failed:\n{}\nconsole:\n{log}", result.stdout)); + } + let console = super::serial::Serial::named("fsd_restart", log.as_str()); + let ended = log.matches("fsd: --end-on: ending with a write done and unanswered").count(); + if ended != 4 { + return Err(format!("fsd said it ended {ended} times, not the guest's 4:\n{log}")); + } + let restarted = log.lines().filter(|l| l.contains("init: fsd data (pid ") && l.contains("ended; started again")).count(); + if restarted != 3 { + return Err(format!("init started DATA's server again {restarted} times, not 3:\n{log}")); + } + console.must_say("init: fsd data ended 4 times in 10 s; its ports are closed")?; + console.must_be_clean()?; + + let io = FileBlocks::open(&image)?; + let fs = bcachefs::Mounted::<_, bcachefs::ReadOnly>::open(io) + .map_err(|e| format!("the DATA partition does not mount on the host: {e:?}"))?; + for (name, want) in [(KEPT, &kept[..]), (ACROSS, ACROSS_BYTES)] { + let got = fs.read_file(name).map_err(|e| format!("reading {name} off the image: {e:?}"))?; + if got != want { + let at = got.iter().zip(want).position(|(a, b)| a != b); + return Err(format!( + "{name} on the device is {} bytes, first differing at {at:?}: a write the guest \ + was told was flushed is not what the device holds", + got.len() + )); + } + } + eprintln!( + " [fsd] DATA's server ended four times under an unanswered write; started again three, \ + its clients reopened, the fourth closed /home to Gone; {KEPT} and {ACROSS} read back off \ + the image by the host's own bcachefs reader" + ); + Ok(()) +} + /// `/apps` and `/home` are two paths into one filesystem, judged off the device. /// /// The guest writes one file under each and shuts down; the host then finds @@ -751,7 +832,7 @@ pub fn internal_disk_boot( } // This machine has no USB, so a volume fsd serves came through blockd. - for said in ["fsd: Boot serving /boot — FAT32 read-only", "fsd: Log serving /log — FAT32,"] { + for said in [super::volumes::BOOT_SERVED, super::volumes::LOG_SERVED] { if !boot.contains(said) { return Err(format!( "the boot never said {said:?} — a machine booting off its internal disk got \ diff --git a/tests/common/toybox.rs b/tests/common/toybox.rs index 3cbe3282a7c..8fded239b42 100644 --- a/tests/common/toybox.rs +++ b/tests/common/toybox.rs @@ -176,7 +176,7 @@ pub fn cp_volume( let boot = qemu.boot_log().to_string(); let boot = serial::Serial::named("boot console", boot.as_str()); boot.must_be_clean()?; - boot.must_say("log-volume: partition mounted")?; + boot.must_say(super::volumes::LOG_SERVED)?; let mut log = serial::Serial::named("the three commands and the shutdown", ""); // One: the copy that fits. diff --git a/tests/common/volumes.rs b/tests/common/volumes.rs index 25c0e850573..1836858d2c9 100644 --- a/tests/common/volumes.rs +++ b/tests/common/volumes.rs @@ -42,6 +42,13 @@ use gpt::partition_types; use super::qemu::{self, BootOptions, QemuInstance}; use super::serial; +/// fsd's word for `/log` served off the log partition. +pub(crate) const LOG_SERVED: &str = "fsd: Log serving /log — FAT32,"; +/// fsd's word for `/boot` served off the running slot's volume. +pub(crate) const BOOT_SERVED: &str = "fsd: Boot serving /boot — FAT32 read-only,"; +/// fsd's word for a `/log` it has no volume behind. +pub(crate) const LOG_ABSENT: &str = "fsd: Log serving /log — absent:"; + /// Mirrored in `tests/toyos-rust-tests/src/bin/esp_files.rs`. Two halves of one /// fixture; a change to either alone fails loudly rather than passing quietly. const HOST_NOTE: &str = "host-note.txt"; @@ -296,7 +303,7 @@ pub fn esp_filesystem( ); let boot = qemu.boot_log().to_string(); serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; - if !boot.contains("boot-volume: partition mounted") { + if !boot.contains(BOOT_SERVED) { return Err(format!( "the kernel did not mount the boot partition:\n{}", volume_lines(&boot) @@ -451,7 +458,7 @@ pub fn esp_filesystem( fn volume_lines(log: &str) -> String { let lines: Vec<&str> = log .lines() - .filter(|l| l.contains("-volume:") || l.contains("logd:") || l.contains("gpt:") + .filter(|l| l.contains("fsd:") || l.contains("blockd:") || l.contains("logd:") || l.contains("gpt:") || l.contains("shutdown") || l.contains("Shutting down") || l.contains("Syncing") || l.contains("usb-storage:")) .collect(); @@ -935,7 +942,7 @@ pub fn writeback_durability( ); let boot = qemu.boot_log().to_string(); serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; - if !boot.contains("log-volume: partition mounted") { + if !boot.contains(LOG_SERVED) { return Err(format!( "the log partition did not mount, so the guest had nowhere to write:\n{}", volume_lines(&boot) @@ -1223,7 +1230,7 @@ pub fn fat_backing_revoked( ); let boot = qemu.boot_log().to_string(); serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; - if !boot.contains("log-volume: partition mounted") { + if !boot.contains(LOG_SERVED) { return Err(format!( "the log partition did not mount, so the guest had nowhere to stage the unlink:\n{}", volume_lines(&boot) @@ -1337,7 +1344,7 @@ pub fn fsync_failed_commit( }, ); let mut log = qemu.boot_log().to_string(); - if !log.contains("log-volume: partition mounted") { + if !log.contains(LOG_SERVED) { return Err(format!( "the log partition did not mount, so nothing below asks the device to flush:\n{}", volume_lines(&log) @@ -1420,7 +1427,7 @@ pub fn redirty_mid_flush( ); let boot = qemu.boot_log().to_string(); serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; - if !boot.contains("log-volume: partition mounted") { + if !boot.contains(LOG_SERVED) { return Err(format!( "the log partition did not mount, so the race had nowhere to run:\n{}", volume_lines(&boot) @@ -1514,7 +1521,7 @@ pub fn ftruncate_flush_race( }, ); let boot = qemu.boot_log().to_string(); - if !boot.contains("log-volume: partition mounted") { + if !boot.contains(LOG_SERVED) { return Err(format!( "the log partition did not mount, so the race had nowhere to run:\n{}", volume_lines(&boot) @@ -1634,7 +1641,7 @@ pub fn fs_rename_durable( ); let boot = qemu.boot_log().to_string(); serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; - if !boot.contains("log-volume: partition mounted") { + if !boot.contains(LOG_SERVED) { return Err(format!( "the log partition did not mount, so the guest had nowhere to stage the rename:\n{}", volume_lines(&boot) @@ -1751,7 +1758,7 @@ pub fn fs_dirs_durable( ); let boot = qemu.boot_log().to_string(); serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; - if !boot.contains("log-volume: partition mounted") { + if !boot.contains(LOG_SERVED) { return Err(format!( "the log partition did not mount, so the guest had nowhere to stage directories:\n{}", volume_lines(&boot) @@ -1836,140 +1843,6 @@ pub fn fs_dirs_durable( Ok(()) } -/// **The machine's stop leaves no filesystem update half made**, and -/// `toyos-fat32-check` is who says so. -/// -/// `quiesce-fsync-refuse` refuses one staged file's flush: each attempt writes -/// a new cluster's entry into the mirror FAT and is refused the active one, and -/// the caller parks in `block::between_attempts`. The job asks init for the -/// shutdown once the attempt that parks is refused, so the stop meets a thread -/// parked over two FATs that disagree. A stop that bands the thread where it is -/// parked leaves that volume at the reset; one that lets the update close -/// leaves it whole, and the kernel's own `fsync:` line says which attempt -/// closed it — inside the stop, by the stop record's own clock. -pub fn quiesce_leaves_the_volume_whole( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - // The kernel's `mirror_refuse::FSYNC_REFUSALS`, spelt here because the - // harness cannot link the kernel. - const REFUSALS: usize = 8; - const REFUSED: &str = "quiesce-fsync-refuse: refusing a SYS_FSYNC flush's active-FAT write"; - const CLOSED: &str = "fsync: /log/quiesce-fsync.bin durable on attempt 9"; - const SYNCING: &str = "Syncing filesystems..."; - const PARAMS: &[&str] = &["quiesce-fsync-refuse"]; - - let image_path = test_dir().join("quiesce-volume-whole.img"); - let image = qemu::build_boot_image(test_config, c_bins, rust_bins, PARAMS); - std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - let (start, len) = log_extent(&image, &image_path)?; - - let complaints_before = check(&image[start..start + len]); - if !complaints_before.is_empty() { - return Err(format!( - "the log partition was not born clean, so this gate cannot tell a complaint the \ - stop caused from one it inherited:\n{}", - describe(&complaints_before) - )); - } - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - boot_image: Some(qemu::Staged::Written(image_path.clone())), - kernel_params: PARAMS, - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; - if !boot.contains("log-volume: partition mounted") { - return Err(format!( - "the log partition did not mount, so this boot has no volume to leave whole:\n{}", - volume_lines(&boot) - )); - } - - writeln!(qemu.stdin_mut(), "run test_rs_quiesce_fsync").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - // The drain ends when QEMU exits; a guest still up at its bound is a - // shutdown that never reached its last word, and nothing below is about it. - if !tail.contains("Shutting down.") { - return Err(format!("the guest did not shut down within the drain's 20 s\n{tail}")); - } - - let after = std::fs::read(&image_path).map_err(|e| format!("read the image back: {e}"))?; - if after.len() != image.len() { - return Err(format!("the image is {} bytes, was {}", after.len(), image.len())); - } - - // **The harm, first.** - let complaints_after = check(&after[start..start + len]); - if !complaints_after.is_empty() { - return Err(format!( - "the machine's stop left the log volume breaking the format:\n{}\n{tail}", - describe(&complaints_after) - )); - } - - // The arm fired as many times as the kernel declares, or the silence above - // is about a stop that met nobody parked. - let lines: Vec<&str> = tail.lines().collect(); - let refusals: Vec<&str> = lines.iter().copied().filter(|l| l.contains(REFUSED)).collect(); - if refusals.len() != REFUSALS { - return Err(format!( - "the quiesce-fsync-refuse actuator refused {} attempt(s), not the {REFUSALS} the \ - kernel declares, so no thread was parked where this boot says one was\n{tail}", - refusals.len() - )); - } - // **Where the stop began, by its own record**: it says how long it took, - // and says so after the sync line it writes the moment it is over. - let ms = |line: &str| { - bootlog::record_millis(line).ok_or_else(|| format!("{line:?} carries no kernel time")) - }; - let synced = lines.iter().position(|l| l.contains(SYNCING)); - let closed = lines.iter().position(|l| l.contains(CLOSED)); - let record = lines.iter().find_map(|l| toyos_quiesce::Record::parse(l)); - let (Some(synced), Some(closed), Some(record)) = (synced, closed, record) else { - return Err(format!( - "the volume is whole, but this boot does not say the stop met the parked flush and \ - let it close: the sync is at {synced:?}, the kernel's `{CLOSED}` at {closed:?}, and \ - the stop record {}\n{tail}", - if record.is_some() { "is there" } else { "is missing" }, - )); - }; - let began = ms(lines[synced])?.saturating_sub(record.elapsed_ms); - let first_refusal = ms(refusals[0])?; - let closed_at = ms(lines[closed])?; - if !(first_refusal < began && began <= closed_at && closed < synced) { - return Err(format!( - "the first refusal at {first_refusal} ms, the stop's start at {began} ms and the \ - flush's close at {closed_at} ms (console line {closed}, the sync at {synced}): the \ - stop did not begin while the flush was parked, or did not wait for it to close\n{tail}" - )); - } - - let _ = std::fs::remove_file(&image_path); - eprintln!( - " [fat] the stop began at {began} ms over a flush parked on split FATs and let it close \ - at {closed_at} ms; the checker is silent:\n {}", - lines[closed], - ); - Ok(()) -} - /// The boot disk arrives *after* the port scan, and both mounts still happen. /// @@ -2041,8 +1914,8 @@ pub fn late_storage_connect( } for want in [ - "boot-volume: partition mounted", - "log-volume: partition mounted", + BOOT_SERVED, + LOG_SERVED, "logd: this boot's kernel log is", ] { if !boot.contains(want) { @@ -2098,308 +1971,6 @@ pub fn log_on_device( need(found.pop().flatten(), name) } -/// A page of a `/log` file that the device will not give back, and the partial -/// write that used to merge into the hole and persist it. -/// -/// `file_cache::write_page` re-reads a page it is about to partially overwrite, -/// through the file's backing, and merges the new bytes into what comes back. -/// `FatBacking::read_page` returned `()`, so a failed read was indistinguishable -/// from a page of zeros: the new bytes went into those zeros and `flush_file` -/// wrote the result back over data that was already on the stick. -/// -/// Three separate claims, and none of them is the others: -/// -/// - the failure is **reported** (`serving zeros`, the marker triage greps for, -/// which this path could not emit at all); -/// - the failure **propagates** to the caller — `FatBacking` → -/// `file_cache::write_page` → `ops::try_write` → the process, every one of which -/// returned `()` or swallowed on some link of the chain; -/// - the file on the device is **not corrupted**, checked on the host against -/// the bytes the host itself wrote. This is the claim the other two exist to -/// serve, and the one that stays meaningful if the log lines are reworded. -/// -/// **What stages it is a host-written file, and that is the deterministic form -/// rather than the only one.** The log's own writer reaches the same hazard: -/// `/system/bin/logd` is an ordinary process appending to an ordinary file and -/// `fsync`s every batch, which clears the dirty bit and leaves its tail page an -/// ordinary eviction candidate — so a boot that loses that page re-fetches it -/// on the next append, exactly as this does. Staging it from the guest's own -/// log would make the trigger a matter of which page happened to be evicted; -/// a file the host wrote before the machine existed can have no resident page -/// at all. -pub fn log_backing_read_error( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - // `test-small-caches` is what makes the page actually get evicted: at the - // shipped ceiling the log's few pages stay resident for the whole boot and - // the re-read the injection targets never happens. The eviction code is the - // shipped code; only the bound moves. - const PARAMS: &[&str] = &["fat-backing-read-fails"]; - const SERVING_ZEROS: &str = "failed; serving zeros"; - /// Mirrored in `tests/toyos-rust-tests/src/bin/log_volume_reread.rs`. - const STAGED: &str = "staged-reread.txt"; - /// Printable and longer than the offset the guest writes at, so the page is - /// fetched rather than extended, and so a merge into zeros shows up as a - /// run of NULs in a file that is otherwise entirely text. - const STAGED_TEXT: &[u8] = b"written by the host onto the log volume before this machine \ - started, and not to be changed by it\n"; - - let image_path = test_dir().join("fat-backing-read-fails.img"); - let mut image = qemu::build_boot_image(test_config, c_bins, rust_bins, PARAMS); - // Written before the extent is asked for: `log_extent` parses the GPT off - // the file, not the buffer. - std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - let (start, len) = log_extent(&image, &image_path)?; - // The host's half of the fixture, on the device before there is a guest. - // This is what makes the trigger deterministic rather than a matter of - // whether some page happened to be evicted: none of this file's pages can - // be resident, because the machine has never seen it. - stage_files( - &mut image[start..start + len], - &[(STAGED.to_string(), STAGED_TEXT.to_vec())], - )?; - std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - boot_image: Some(qemu::Staged::Written(image_path.clone())), - kernel_params: PARAMS, - ..Default::default() - }, - ); - let mut log = qemu.boot_log().to_string(); - let attempt = qemu.run_test("test_rs_log_volume_reread", Duration::from_secs(30)); - // Both streams: the kernel's own line about the refused read is on the - // serial console, and the process's account of what it was told is on - // stdout. The claims below need one of each. - log.push_str(&attempt.stdout); - log.push_str(&attempt.serial); - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - log.push_str(&qemu.drain_serial(Duration::from_secs(20))); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if log.contains(bad) { - return Err(format!("{bad:?} on the boot\n{log}")); - } - } - - // 1. The injection reached the code, and the code said so. Before this - // landed there was no string here to find: the two sibling backings - // print `serving zeros` and this one returned in silence. - let reported = log.matches(SERVING_ZEROS).count(); - if reported == 0 { - return Err(format!( - "no {SERVING_ZEROS:?} line — either the injection never reached a page re-read (so \ - this boot proves nothing) or the FAT backing is still failing silently\n{log}" - )); - } - - // 2. It propagated the whole way, to the one caller that can be asked. A - // write reported as succeeding is the defect: the process has no way to - // know its bytes went into a page invented out of a failed read. - if !log.contains("reread: the write failed") { - return Err(format!( - "the process was not told: a refused page has to reach `ops::try_write` as an error \ - instead of being merged into zeros\n{log}" - )); - } - - // 2b. The read of the same page, which is the sharper half and was the - // later fix: `file_cache::read_page` returned `()`, so the process got - // the page zeroed and a success. Nothing above it — not this test, not - // a `cat`, not the ELF loader — can tell that from a file that really - // is zeros there, which is why the count is in the guest's line and - // why this refuses the success rather than checking the bytes. - if !log.contains("reread: the read failed") { - let said = log - .lines() - .map(str::trim) - .find(|l| l.starts_with("reread: the read")) - .unwrap_or("(the guest never reported its read)"); - return Err(format!( - "a page the device would not give back reached the process as data: {said}\n\ - A failed read has to be distinguishable from a hole.\n{log}" - )); - } - - // 3. And the machine is fine. A refusal that costs the boot is not a fix. - if !log.contains("Boot: complete") { - return Err(format!("the boot did not finish\n{log}")); - } - if !log.contains("Shutting down.") { - return Err(format!("the guest did not shut down cleanly\n{log}")); - } - - // 4. Ground truth: the file on the device, against the bytes the host put - // there. A page merged into a failed re-fetch is zeros where the text - // was, so this catches the corruption whether or not anything was said - // about it — the console being exactly what the guest would be wrong - // about. - let after = std::fs::read(&image_path).map_err(|e| format!("read the image back: {e}"))?; - let on_device = need(read_files(&after[start..start + len], &[STAGED])?.pop().flatten(), STAGED)?; - if on_device != STAGED_TEXT { - let at = on_device.iter().zip(STAGED_TEXT).position(|(a, b)| a != b); - return Err(format!( - "the guest changed {STAGED} on the device: {} bytes became {}, first differing at \ - {at:?} — a partial write was merged into a page the device would not give back, and \ - flushed", - STAGED_TEXT.len(), - on_device.len() - )); - } - - let _ = std::fs::remove_file(&image_path); - eprintln!( - " [log] {reported} page re-read(s) refused by the device: reported, propagated to the \ - process that asked, and the {} bytes the host staged are intact", - STAGED_TEXT.len() - ); - Ok(()) -} - -/// A mounted volume that stops answering, and the questions `vfs::FileSystem` -/// used to fold into "no such file". -/// -/// `open` and `read_dir` reached filesystem methods returning an `Option`, a -/// `bool` and a bare `u64`, so a device that refused a transfer was reported to -/// userland as a name that is not there. Nothing downstream can act on that: it -/// creates a file over one that exists, reports a program missing off a stick -/// that is merely unhappy, and unlinks a name it believes is already gone. -/// -/// `fat-boot-reads-fail` is the actuator and its sibling -/// `fat-backing-read-fails` is not: that one injects at -/// `FatBacking::read_page`, which is the page-fault path and reaches no -/// directory entry, so with it armed every question below still succeeds. This -/// one is under `Fat32` itself. Neither can be staged from the host — both -/// partitions live on the disk the guest is running from, so `readonly=on` is -/// writes only and `rerror` takes the whole drive. -/// -/// **The mount line is a load-bearing assertion and not decoration.** A `/boot` -/// that failed to mount is not a mount at all: `Vfs::resolve_fs` falls through -/// to the root filesystem, ROOT has no `boot/` in it, and every question -/// below would then answer `NotFound` for an honest reason — which is precisely -/// the string this test exists to refuse. -pub fn boot_volume_metadata_error( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - const PARAMS: &[&str] = &["fat-boot-reads-fail"]; - /// What the injection prints from under `Fat32`, once per refused read. - const REFUSED: &str = "boot-volume: read of"; - - let image_path = test_dir().join("fat-boot-reads-fail.img"); - let image = qemu::build_boot_image(test_config, c_bins, rust_bins, PARAMS); - std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - boot_image: Some(qemu::Staged::Written(image_path.clone())), - kernel_params: PARAMS, - ..Default::default() - }, - ); - let mut log = qemu.boot_log().to_string(); - if !log.contains("boot-volume: partition mounted") { - return Err(format!( - "the boot partition did not mount, so every question below would answer NotFound \ - for an honest reason and this boot proves nothing:\n{}", - volume_lines(&log) - )); - } - - let attempt = qemu.run_test("test_rs_boot_volume_metadata_error", Duration::from_secs(30)); - log.push_str(&attempt.stdout); - log.push_str(&attempt.serial); - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - log.push_str(&qemu.drain_serial(Duration::from_secs(20))); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if log.contains(bad) { - return Err(format!("{bad:?} on the boot\n{log}")); - } - } - - // 1. The injection reached the device layer under the filesystem, so what - // follows is a volume that was asked and would not answer, rather than a - // code path that never ran. - let refused = log.matches(REFUSED).count(); - if refused == 0 { - return Err(format!( - "no {REFUSED:?} line — nothing read the boot volume after it was mounted, so this \ - boot exercises none of the metadata path\n{log}" - )); - } - - // 2. and 3. The two questions, each judged on the word it came back with. - // `NotFound` is named explicitly because it is the *old* answer and the - // whole defect: a check for "an error" alone would have passed before the - // change, since a missing file is an error too. - for (what, prefix) in [("open", "boot-io: open"), ("read_dir", "boot-io: read_dir")] { - let Some(said) = log.lines().map(str::trim).find(|l| l.starts_with(prefix)) else { - return Err(format!("the guest never reported its {what}\n{log}")); - }; - if said.contains("succeeded") { - return Err(format!( - "{what} of a volume that refused every read succeeded: {said}\n{log}" - )); - } - if said.contains("kind=NotFound") { - return Err(format!( - "{what} reported a device that would not answer as a missing file: {said}\n\ - That is the conflation this gate exists for — `vfs::FileSystem` folding a \ - refused read into NotFound.\n{log}" - )); - } - // The class and not merely "not NotFound": a refused read is - // `IoError::Device`, then `SyscallError::Io`, then `ErrorKind::Other`. - if !said.contains("kind=Other") { - return Err(format!( - "{what} of a volume that refused every read answered {said}\n\ - A device that would not answer has to reach the caller as the I/O class \ - (`SyscallError::Io`, `ErrorKind::Other`). That spelling is owed a change and \ - this site moves with it — the record is\n\ -issues/design-debt/std-maps-a-device-error-to-other-not-uncategorized.md\n{log}" - )); - } - } - - // 4. And it is this volume's refusal and not the machine's. The other FAT - // mount is the same adapter over the same driver, so a break that - // reached both would look identical in the two lines above. - if !log.contains("boot-io: /log still lists") { - return Err(format!( - "/log stopped answering too, so the refusal above is not the boot volume's\n{log}" - )); - } - - if !log.contains("Boot: complete") { - return Err(format!("the boot did not finish\n{log}")); - } - if !log.contains("Shutting down.") { - return Err(format!("the guest did not shut down cleanly\n{log}")); - } - - let _ = std::fs::remove_file(&image_path); - eprintln!( - " [boot] {refused} filesystem read(s) refused by the mounted boot volume: open and \ - read_dir both reported the device rather than a missing file, and /log kept answering" - ); - Ok(()) -} - /// The image side of the whole exercise, with nothing mounted and nothing /// booted: the log partition is what a desktop OS will pick up on plug-in. /// @@ -2761,16 +2332,13 @@ pub fn log_partition_identity( volume_lines(&log) )); } - let refused = format!("nothing with the log partition's GUID {FORGED_TEXT}"); + let refused = format!("{LOG_ABSENT} the Log partition {FORGED_TEXT} is on no disk this server reaches"); if !log.contains(&refused) { return Err(format!( - "the kernel did not refuse the log partition by name.\nwanted: {refused}\n{}", + "fsd did not serve /log absent for the named partition.\nwanted: {refused}\n{}", volume_lines(&log) )); } - if !log.contains("log-volume: not mounted") { - return Err(format!("the kernel mounted a log volume it was never given:\n{}", volume_lines(&log))); - } if log.contains("logd: this boot's kernel log is") { return Err(format!( "logd opened a file with no log partition — a fallback is exactly what this must not \ @@ -2787,7 +2355,7 @@ pub fn log_partition_identity( // And nothing else was lost. The stick is a working stick with one file // changed on it. - if !log.contains("boot-volume: partition mounted") { + if !log.contains(BOOT_SERVED) { return Err(format!("a missing log partition cost the machine /boot:\n{}", volume_lines(&log))); } if !log.contains("Boot: complete") { diff --git a/tests/fsdrestartcase/system.toml b/tests/fsdrestartcase/system.toml new file mode 100644 index 00000000000..acdef491e87 --- /dev/null +++ b/tests/fsdrestartcase/system.toml @@ -0,0 +1,41 @@ +# The boot `fsd_restart` judges: the test estate's shape, with every file +# server armed to end the moment it has taken a write through a file opened to +# append at `/home/fsd_end` and before it answers it — `test_rs_fs_restart` +# ends DATA's four times, and init restarts it three. + +[boot] +start = ["logd", "blockd", "fsd", "test-runner"] + +[programs.logd] +service = true +syscap = ["logread"] + +# `power` because `run shutdown` asks init through the `power` connector, and +# the host reads the DATA partition back once the machine is down. +[programs.test-runner] +receives = ["power"] +syscap = ["dup", "logread", "power"] + +[programs.toybox] + +[symlinks] +"bin/shutdown" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. `--end-on` is the test's actuator: a path on +# the DATA volume, which neither other role's volume carries. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] +args = ["--end-on", "home/fsd_end"] diff --git a/tests/inspectcase/system.toml b/tests/inspectcase/system.toml index 216aa8c64bd..65b967b98da 100644 --- a/tests/inspectcase/system.toml +++ b/tests/inspectcase/system.toml @@ -39,12 +39,9 @@ devices = ["pci:1af4:1041"] # Every owner's connector, and `inventory` for `dev.*`: the runner's namespace # is what its jobs inherit, and `dup` is what hands each job a duplicate of the # capability, which `test_rs_inspect_denied` narrows to prove the refusal. -# The partition is the claimed one on the disk `tests/common/inspect.rs` -# crafts, where the GUID is mirrored; its neighbour there is the free one. [programs.test-runner] receives = ["netd", "soundd", "log", "compositor"] syscap = ["inventory", "dup"] -devices = ["part:B4C5D6E7-F809-4A1B-8C2D-3E4F5A6B7C8D"] # The shipped reader's row. Declared so the image carries it; the runner spawns # it directly, so what it holds here is the runner's namespace. diff --git a/tests/test-durations b/tests/test-durations index 54ea13a8f27..845cbef45a1 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -154,9 +154,7 @@ blocked_dump 3400 blocking_read_stress 41 boot_deadline_ends_a_wedge 19295 boot_partition_identity 5519 -boot_volume_metadata_error 4968 c_capture_ignores_daemon_lines 4578 -cache_eviction 8165 connect_before_serve 23 console_line_atomicity 8925 console_locale_detect 3922 @@ -264,7 +262,6 @@ leak_rollback_selftest 5423 loader_watchdog_arms 10064 locale_detect 9959 locale_detect_unrecognized 160 -log_backing_read_error 4887 log_conservation_smp1 4686 log_conservation_smp4 8248 log_conservation_smp8 5112 diff --git a/tests/toyos-rust-tests/src/bin/boot_volume_metadata_error.rs b/tests/toyos-rust-tests/src/bin/boot_volume_metadata_error.rs deleted file mode 100644 index b20894d5a63..00000000000 --- a/tests/toyos-rust-tests/src/bin/boot_volume_metadata_error.rs +++ /dev/null @@ -1,50 +0,0 @@ -//! What a mount says when its device stops answering the questions that used -//! to have no channel: is this file there, what is in this directory, when was -//! it written. -//! -//! `vfs::FileSystem` answered every one of those with an `Option`, a `bool` or -//! a bare `u64`, so a volume that refused a transfer was reported as a name -//! that is not there. That is the one degradation a caller cannot act on: it -//! creates a file over one that exists, reports a program missing off a stick -//! that is merely unhappy, and unlinks a name it believes is already gone. -//! -//! The kernel is built with `fat-boot-reads-fail`, which fails every read of -//! the boot volume *under `Fat32`* — a directory entry, a FAT chain, an extent -//! list — from the moment the mount finishes. Nothing else in the machine is -//! touched, and the `/log` check at the end is what says so: a refusal that -//! costs the whole machine is not a fix, and one that reads as machine-wide -//! would prove nothing about this volume. -//! -//! The kind is printed rather than asserted here, because the assertion belongs -//! on the host side beside the log it is judged against — see -//! `tests/common/volumes.rs`. - -use std::fs; - -/// Written by the image builder, so it is on every boot volume this project -/// produces. -const NAMED: &str = "/boot/toyos/log.guid"; - -fn main() { - // The defect in one line: this used to be `Option`, and `None` was both - // "no such file" and "the volume would not say". - match fs::File::open(NAMED) { - Ok(_) => println!("boot-io: open succeeded"), - Err(e) => println!("boot-io: open failed: kind={:?}: {e}", e.kind()), - } - - // `FileSystem::list`'s refusal used to be `SyscallError::NotFound` on this - // adapter, which reads to a caller as "there is no such directory". - match fs::read_dir("/boot") { - Ok(entries) => println!("boot-io: read_dir succeeded with {} entries", entries.count()), - Err(e) => println!("boot-io: read_dir failed: kind={:?}: {e}", e.kind()), - } - - // The volume that is *not* injected, from the same adapter over the same - // driver, so a machine-wide breakage cannot be mistaken for the refusal - // this test is about. - match fs::read_dir("/log") { - Ok(entries) => println!("boot-io: /log still lists {} entries", entries.count()), - Err(e) => println!("boot-io: /log answered an error too: {e}"), - } -} diff --git a/tests/toyos-rust-tests/src/bin/cache_eviction.rs b/tests/toyos-rust-tests/src/bin/cache_eviction.rs deleted file mode 100644 index c948ba0de4e..00000000000 --- a/tests/toyos-rust-tests/src/bin/cache_eviction.rs +++ /dev/null @@ -1,193 +0,0 @@ -//! Sustained pressure on both disk caches, and the one question a cache that -//! evicts has to answer: does a page that was thrown away come back as the -//! bytes that were written? -//! -//! Everything here is sized in pages against the `test-small-caches` budget of -//! 64 of them. `BIG_PAGES` is twice that, so the forward pass evicts the head -//! of the file before it reaches the tail and the reverse pass then re-reads -//! every page in the order that guarantees a miss on each one. `SMALL_FILES` -//! exists because the eviction sweep walks (file, page) order, and a single -//! file cannot exercise the step from one file to the next. -//! -//! At the shipped budget the same run fits in cache and proves only the round -//! trip, which is why the harness asserts on the kernel's own eviction series -//! rather than on this exit code alone. -//! -//! `BIG_PAGES` is sized against the cache and nothing else. It used to be -//! capped near 250 by the filesystem — one inline extent per page against a -//! 4040-byte btree value — which is how that panic was found; `fs_large_file` -//! now covers the fixed version at four times the old ceiling. - -use std::fs; -use std::io::{Read, Seek, SeekFrom, Write}; - -const PAGE: usize = 4096; -const BIG: &str = "/home/cache_big.bin"; -const BIG_PAGES: usize = 128; -const SMALL_FILES: usize = 8; -const SMALL_PAGES: usize = 32; - -/// Distinct per (file, page, byte), so a page served from the wrong slot, from -/// the wrong file, or half-written is a mismatch rather than a coincidence. -fn byte_at(tag: usize, page: usize, i: usize) -> u8 { - let mixed = (tag.wrapping_mul(0x9E37_79B9)) - ^ (page.wrapping_mul(0x85EB_CA6B)) - ^ (i.wrapping_mul(0xC2B2_AE35)); - (mixed >> 11) as u8 -} - -fn page_bytes(tag: usize, page: usize) -> Vec { - (0..PAGE).map(|i| byte_at(tag, page, i)).collect() -} - -fn write_file(path: &str, tag: usize, pages: usize) { - let mut f = fs::File::create(path).unwrap_or_else(|e| panic!("create {path}: {e}")); - for page in 0..pages { - f.write_all(&page_bytes(tag, page)) - .unwrap_or_else(|e| panic!("write {path} page {page}: {e}")); - } - f.sync_all().unwrap_or_else(|e| panic!("fsync {path}: {e}")); -} - -fn check_page(f: &mut fs::File, path: &str, tag: usize, page: usize, why: &str) { - f.seek(SeekFrom::Start((page * PAGE) as u64)) - .unwrap_or_else(|e| panic!("seek {path} page {page}: {e}")); - let mut got = vec![0u8; PAGE]; - f.read_exact(&mut got) - .unwrap_or_else(|e| panic!("read {path} page {page}: {e}")); - let want = page_bytes(tag, page); - if let Some(i) = got.iter().zip(&want).position(|(a, b)| a != b) { - panic!( - "{path} page {page} byte {i}: {} != {} ({why})", - got[i], want[i] - ); - } -} - -fn main() { - write_file(BIG, 0, BIG_PAGES); - - // Forward: the pass that fills the cache and then keeps going. - let mut f = fs::File::open(BIG).expect("reopen big file"); - for page in 0..BIG_PAGES { - check_page(&mut f, BIG, 0, page, "forward pass"); - } - // Backward: with a cache smaller than the file, the page the forward pass - // ended on is the only one still resident, so every step from here is a - // miss that has to be satisfied from the backing. - for page in (0..BIG_PAGES).rev() { - check_page(&mut f, BIG, 0, page, "reverse pass"); - } - drop(f); - - // A tmpfs file has no backing: its pages are the file, not a copy of one, - // so the sweep must walk past them however hard the disk cache is being - // pressed. Written before the pressure below and read after it. - const TMP: &str = "/tmp/cache_tmpfs.bin"; - const TMP_PAGES: usize = 48; - write_file(TMP, 100, TMP_PAGES); - - for tag in 0..SMALL_FILES { - write_file(&format!("/home/cache_small_{tag}.bin"), tag + 1, SMALL_PAGES); - } - - // Interleaved across files, so the sweep's file-to-file step is on the - // path every time and no single file's pages are the only candidates. - let mut handles: Vec = (0..SMALL_FILES) - .map(|tag| fs::File::open(format!("/home/cache_small_{tag}.bin")).expect("reopen small")) - .collect(); - for _round in 0..3 { - for page in 0..SMALL_PAGES { - for (tag, f) in handles.iter_mut().enumerate() { - let path = format!("/home/cache_small_{tag}.bin"); - check_page(f, &path, tag + 1, page, "interleaved pass"); - } - } - } - drop(handles); - - // Dirty pages are the other thing eviction may not take: until the flush - // writes them out they are the only copy. Rewriting the head of the big - // file leaves the whole budget dirty, and the read pass underneath it is - // what asks the cache for room it cannot make — the case where the sweep - // has to give up over budget rather than take something it cannot replace. - const DIRTY_PAGES: usize = 64; - // Then past the budget through a second file with every page dirty: the - // sweep must report over-budget rather than hold 64. Eight pages, so a - // stray clean page another process left resident cannot absorb the overage. - const OVERAGE_PAGES: usize = 8; - const OVERAGE_TAG: usize = 77; - let over_path = format!("/home/cache_small_{}.bin", SMALL_FILES - 1); - { - let mut w = fs::OpenOptions::new().write(true).open(BIG).expect("reopen big to rewrite"); - for page in 0..DIRTY_PAGES { - w.seek(SeekFrom::Start((page * PAGE) as u64)).expect("seek to rewrite"); - w.write_all(&page_bytes(7, page)).expect("rewrite page"); - } - let path = format!("/home/cache_small_0.bin"); - let mut r = fs::File::open(&path).expect("reopen pressure file"); - for page in 0..SMALL_PAGES { - check_page(&mut r, &path, 1, page, "pressure while the budget is dirty"); - } - let mut over = - fs::OpenOptions::new().write(true).open(&over_path).expect("reopen for overage"); - for page in 0..OVERAGE_PAGES { - over.seek(SeekFrom::Start((page * PAGE) as u64)).expect("seek overage"); - over.write_all(&page_bytes(OVERAGE_TAG, page)).expect("dirty past the budget"); - } - w.sync_all().expect("fsync the rewrite"); - over.sync_all().expect("fsync the overage"); - } - { - let mut f = fs::File::open(BIG).expect("reopen big after rewrite"); - for page in 0..DIRTY_PAGES { - check_page(&mut f, BIG, 7, page, "dirty page survived eviction pressure"); - } - } - // The budget holds again once the writers flushed: this sweep forces the - // closing samples and round-trips the pages written past the budget. - for tag in 0..SMALL_FILES { - let path = format!("/home/cache_small_{tag}.bin"); - let mut f = fs::File::open(&path).expect("reopen for the flushed sweep"); - for page in 0..SMALL_PAGES { - if tag == SMALL_FILES - 1 && page < OVERAGE_PAGES { - check_page(&mut f, &path, OVERAGE_TAG, page, "flushed sweep, overage pages"); - } else { - check_page(&mut f, &path, tag + 1, page, "flushed sweep"); - } - } - } - - let mut tmp = fs::File::open(TMP).expect("reopen tmpfs file"); - for page in 0..TMP_PAGES { - check_page(&mut tmp, TMP, 100, page, "tmpfs survives disk-cache pressure"); - } - drop(tmp); - - // A file the cache still holds a handle for, truncated and rewritten shorter: - // the pages past the new end must not survive as stale cache entries. - { - let mut f = fs::OpenOptions::new().write(true).open(BIG).expect("reopen for truncate"); - f.set_len((4 * PAGE) as u64).expect("truncate"); - f.seek(SeekFrom::Start(0)).expect("rewind"); - for page in 0..4 { - f.write_all(&page_bytes(9, page)).expect("rewrite"); - } - f.sync_all().expect("fsync truncated"); - } - let back = fs::read(BIG).expect("read truncated file"); - assert_eq!(back.len(), 4 * PAGE, "truncated file is the wrong length"); - for page in 0..4 { - let want = page_bytes(9, page); - let got = &back[page * PAGE..(page + 1) * PAGE]; - let bad = got.iter().zip(&want).position(|(a, b)| a != b); - assert!(bad.is_none(), "rewritten page {page} byte {} differs", bad.unwrap()); - } - - let pages = BIG_PAGES * 2 - + SMALL_FILES * SMALL_PAGES * 4 - + TMP_PAGES - + SMALL_PAGES - + DIRTY_PAGES; - println!("cache eviction ok: {pages} page reads verified"); -} diff --git a/tests/toyos-rust-tests/src/bin/fs_dirs_durable.rs b/tests/toyos-rust-tests/src/bin/fs_dirs_durable.rs index 454392f4346..8a7b3ca28ef 100644 --- a/tests/toyos-rust-tests/src/bin/fs_dirs_durable.rs +++ b/tests/toyos-rust-tests/src/bin/fs_dirs_durable.rs @@ -2,21 +2,23 @@ //! volume keeps, a directory the mount grew for a file's path is visible and //! removable once emptied, and every `rmdir` outcome is the real one. //! `common::volumes::fs_dirs_durable` judges what this leaves off the raw -//! image — a directory the kernel only pretended to make is one `fatfs` +//! image — a directory the server only pretended to make is one `fatfs` //! cannot see. use std::fs::{self, File}; -use std::io::Write; - -use toyos_abi::syscall::{self, SyscallError}; +use std::io::{ErrorKind, Write}; /// Mirrored in `tests/common/volumes.rs`. const KEEP: &str = "/log/fsdir-keep"; const GONE: &str = "/log/fsdir-gone"; -fn readdir_needed(path: &str) -> Result { - let mut buf = [0u8; 1]; - syscall::readdir(path.as_bytes(), &mut buf) +/// How many entries `path` lists, or the kind of its refusal. +fn entries(path: &str) -> Result { + fs::read_dir(path).map(|d| d.count()).map_err(|e| e.kind()) +} + +fn rmdir(path: &str) -> Result<(), ErrorKind> { + fs::remove_dir(path).map_err(|e| e.kind()) } fn main() { @@ -24,7 +26,7 @@ fn main() { fs::create_dir(KEEP).expect("mkdir on the FAT volume"); let err = fs::create_dir(KEEP).expect_err("mkdir of an existing directory must refuse"); assert_eq!(err.kind(), std::io::ErrorKind::AlreadyExists, "mkdir twice reported {err:?}"); - assert_eq!(readdir_needed(KEEP), Ok(0), "a fresh empty directory did not list as empty"); + assert_eq!(entries(KEEP), Ok(0), "a fresh empty directory did not list as empty"); // A directory the mount created for a file's path, then emptied by that // file's unlink: it must stay visible and become removable — on-disk @@ -35,30 +37,30 @@ fn main() { f.sync_all().expect("fsync"); drop(f); assert_eq!( - syscall::rmdir(GONE.as_bytes()), - Err(SyscallError::InvalidArgument), + rmdir(GONE), + Err(ErrorKind::InvalidInput), "rmdir of a non-empty directory must refuse" ); assert_eq!( - syscall::rmdir(file.as_bytes()), - Err(SyscallError::InvalidArgument), + rmdir(&file), + Err(ErrorKind::InvalidInput), "rmdir of a file must refuse" ); fs::remove_file(&file).expect("unlink the file"); assert_eq!( - readdir_needed(GONE), + entries(GONE), Ok(0), "an emptied on-disk directory disappeared from list" ); - syscall::rmdir(GONE.as_bytes()).expect("rmdir of the emptied directory"); + rmdir(GONE).expect("rmdir of the emptied directory"); assert_eq!( - syscall::rmdir(GONE.as_bytes()), - Err(SyscallError::NotFound), + rmdir(GONE), + Err(ErrorKind::NotFound), "rmdir of a removed directory must refuse" ); assert_eq!( - readdir_needed(GONE), - Err(SyscallError::NotFound), + entries(GONE), + Err(ErrorKind::NotFound), "a removed directory still lists" ); diff --git a/tests/toyos-rust-tests/src/bin/fs_escape.rs b/tests/toyos-rust-tests/src/bin/fs_escape.rs new file mode 100644 index 00000000000..88db28edbcc --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/fs_escape.rs @@ -0,0 +1,102 @@ +//! A directory capability is a floor: nothing named through `/home` reaches a +//! file outside it, and a directory this program was not given is refused. +//! +//! - a relative symlink under `/home` whose `..` climbs above it is refused +//! `PermissionDenied` by the server — refused, not clamped to `/home`, which +//! would be a different file from the one the link named (the rule +//! `openat2(2)`'s `RESOLVE_BENEATH` states for Linux); +//! - one that climbs and comes back down inside `/home` is followed; +//! - a client that puts a `..` on the wire itself, past std, is refused +//! `InvalidArgument` before anything is looked up; +//! - `/boot`, which only a `slots` row holds, is ROOT's empty directory for +//! this one, and the loader on the boot volume is not there. + +use std::fs; +use std::io::ErrorKind; +use std::os::toyos::fs::symlink; + +use toyos::fs::{window_put, Reply, Request, HELLO, OPEN, O_READ, REPLY, WINDOW_BYTES}; +use toyos::shm::SharedMemory; +use toyos::volatile::Window; +use toyos_abi::syscall::SyscallError; + +const SECRET: &[u8] = b"/state's own bytes, which nothing under /home may name"; +const PLAIN: &[u8] = b"a file under /home, named by a link that climbs and comes back"; + +fn refused(path: &str, want: ErrorKind) { + match fs::read(path) { + Err(e) if e.kind() == want => println!("fs_escape: {path} refused: {e}"), + Err(e) => panic!("{path}: refused with {e} ({:?}), not {want:?}", e.kind()), + Ok(bytes) if bytes == SECRET => panic!("{path} read /state's secret through /home"), + Ok(bytes) => panic!("{path} read {} bytes; it names nothing under /home", bytes.len()), + } +} + +fn fresh_link(target: &str, link: &str) { + let _ = fs::remove_file(link); + symlink(target, link).unwrap_or_else(|e| panic!("symlink {link} -> {target}: {e}")); +} + +/// A raw client of `fs:/home`, past std's own canonicalisation: `rel` on the +/// wire as it is, and the server's word for it. +fn wire_open(rel: &[u8]) -> Reply { + let names = toyos::endow::namespace().expect("this program was endowed a namespace"); + let conn = names.open("fs:/home").expect("this program holds fs:/home"); + let window = SharedMemory::create(WINDOW_BYTES).expect("a window"); + let lent = window.share().expect("the window, shared"); + conn.send_with_handles(&[lent], HELLO, &Request::new()).expect("hello"); + let answer = |conn: &toyos::ipc::Connection| -> Reply { + let header = conn.recv_header().expect("a reply"); + assert_eq!(header.msg_type, REPLY, "a reply frame"); + conn.recv_payload(&header).expect("a reply's words") + }; + assert_eq!(answer(&conn).status, 0, "fs:/home answers its hello"); + // SAFETY: the region is `WINDOW_BYTES` long and outlives this use. + window_put(unsafe { Window::new(window.as_ptr(), WINDOW_BYTES) }, 0, rel); + conn.send(OPEN, &Request { len: rel.len() as u64, flags: O_READ, ..Request::new() }).expect("open"); + answer(&conn) +} + +fn main() { + fs::create_dir_all("/state/fs_escape").expect("make /state/fs_escape"); + fs::write("/state/fs_escape/secret", SECRET).expect("write the secret"); + fs::create_dir_all("/home/fs_escape/deeper").expect("make /home/fs_escape/deeper"); + fs::write("/home/fs_escape/plain", PLAIN).expect("write the plain file"); + + fresh_link("../state/fs_escape/secret", "/home/fs_escape_out"); + refused("/home/fs_escape_out", ErrorKind::PermissionDenied); + fresh_link("../../../state/fs_escape/secret", "/home/fs_escape/deeper/out"); + refused("/home/fs_escape/deeper/out", ErrorKind::PermissionDenied); + // A directory link climbing out, and a name under it. + fresh_link("../../..", "/home/fs_escape/deeper/up"); + refused("/home/fs_escape/deeper/up/state/fs_escape/secret", ErrorKind::PermissionDenied); + + fresh_link("../plain", "/home/fs_escape/deeper/back"); + let back = fs::read("/home/fs_escape/deeper/back").expect("a link that stays inside /home is followed"); + assert_eq!(back, PLAIN, "the link inside /home names the plain file"); + + for rel in [&b"../state/fs_escape/secret"[..], b"fs_escape/../../state/fs_escape/secret", b"/state"] { + let reply = wire_open(rel); + assert_eq!( + reply.status, + SyscallError::InvalidArgument.to_u64(), + "{:?} on the wire was answered status {}", + core::str::from_utf8(rel), + reply.status + ); + println!("fs_escape: {:?} on the wire refused", core::str::from_utf8(rel).unwrap_or("")); + } + + // ROOT's own `/boot`, the empty directory a view is mounted over, is all + // a program without the capability names there. + let listed: Vec<_> = fs::read_dir("/boot").expect("ROOT's /boot").collect(); + assert!(listed.is_empty(), "/boot listed {} entries for a program whose row holds no slots", listed.len()); + refused("/boot/EFI/BOOT/BOOTX64.EFI", ErrorKind::NotFound); + + for link in ["/home/fs_escape_out", "/home/fs_escape/deeper/out", "/home/fs_escape/deeper/up", "/home/fs_escape/deeper/back"] { + let _ = fs::remove_file(link); + } + let _ = fs::remove_dir_all("/home/fs_escape"); + let _ = fs::remove_dir_all("/state/fs_escape"); + println!("fs_escape: PASS"); +} diff --git a/tests/toyos-rust-tests/src/bin/fs_restart.rs b/tests/toyos-rust-tests/src/bin/fs_restart.rs new file mode 100644 index 00000000000..bc9a7962ba9 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/fs_restart.rs @@ -0,0 +1,92 @@ +//! A file server that ends is survived and not hidden. +//! +//! Booted by `common::storage::fsd_restart` on `tests/fsdrestartcase`, whose +//! file servers end the moment they take a write through a file opened to +//! append at `/home/fsd_end` and before they answer it: +//! +//! - a file written and `fsync`ed before the end — acknowledged and flushed — +//! reads back through a handle held across it, and a handle written across +//! it goes on writing where it was; +//! - the write the server ended under is answered as the server's end, never +//! as done; +//! - DATA's server ended four times inside init's window is not started a +//! fourth: `/home` then answers `Gone` to a new open and to a held handle. +//! +//! What is on the device is the host's to judge, off the image. + +use std::fs::{self, File, OpenOptions}; +use std::io::{ErrorKind, Read, Write}; + +/// Mirrored in `tests/common/storage.rs`: the flushed file, and the one +/// written across the end. +const KEPT: &str = "/home/fs_restart/kept"; +const ACROSS: &str = "/home/fs_restart/across"; +const KEPT_LEN: usize = 64 * 1024 + 13; +const BEFORE: &[u8] = b"written and flushed before the server ended; "; +const AFTER: &[u8] = b"written through the same handle after it came back"; + +/// `--end-on`'s path, in `tests/fsdrestartcase/system.toml`. +const END: &str = "/home/fsd_end"; + +/// Mirrored: what `KEPT` holds. +fn kept() -> Vec { + (0..KEPT_LEN).map(|i| (i.wrapping_mul(37) ^ 0xC3) as u8).collect() +} + +/// End DATA's server: a write it takes and never answers. +fn end_the_server(n: u32) { + let mut f = OpenOptions::new() + .append(true) + .create(true) + .open(END) + .unwrap_or_else(|e| panic!("end {n}: open {END} to append: {e}")); + match f.write(b"the write the server ends under") { + Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => { + println!("fs_restart: end {n}: the write was answered as the server's end ({e})"); + } + Err(e) => panic!("end {n}: the write was refused with {e} ({:?}), not the server's end", e.kind()), + Ok(n) => panic!("end {n}: the write the server ended under was answered done, {n} bytes"), + } +} + +fn main() { + fs::create_dir_all("/home/fs_restart").expect("make /home/fs_restart"); + let mut f = File::create(KEPT).expect("create the kept file"); + f.write_all(&kept()).expect("write the kept file"); + f.sync_all().expect("the kept file is durable"); + drop(f); + let mut held = File::open(KEPT).expect("hold the kept file"); + let mut across = File::create(ACROSS).expect("create the file written across the end"); + across.write_all(BEFORE).expect("write before the end"); + across.sync_all().expect("durable before the end"); + + end_the_server(1); + + let mut back = Vec::new(); + held.read_to_end(&mut back).expect("a handle held across the end reads"); + assert!(back == kept(), "the held handle read {} bytes, not the kept file", back.len()); + across.write_all(AFTER).expect("a handle held across the end writes where it was"); + across.sync_all().expect("durable after the end"); + let mut whole = Vec::new(); + File::open(ACROSS).and_then(|mut f| f.read_to_end(&mut whole)).expect("read the file written across"); + assert_eq!(whole, [BEFORE, AFTER].concat(), "the file written across the end"); + println!("fs_restart: the server came back; a held handle read, another wrote on, both where they were"); + + for n in 2..=4 { + end_the_server(n); + } + match File::open(KEPT) { + Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => { + println!("fs_restart: after four ends a new open is answered Gone ({e})"); + } + Err(e) => panic!("after four ends a new open was refused {e} ({:?}), not Gone", e.kind()), + Ok(_) => panic!("after four ends a new open of {KEPT} was answered"), + } + match held.read(&mut [0u8; 16]) { + Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => { + println!("fs_restart: and a held handle is answered Gone ({e})"); + } + other => panic!("after four ends a held handle read {other:?}, not Gone"), + } + println!("fs_restart: PASS"); +} diff --git a/tests/toyos-rust-tests/src/bin/log_volume_reread.rs b/tests/toyos-rust-tests/src/bin/log_volume_reread.rs deleted file mode 100644 index e6ebbb754e9..00000000000 --- a/tests/toyos-rust-tests/src/bin/log_volume_reread.rs +++ /dev/null @@ -1,60 +0,0 @@ -//! A write into a page of a `/log` file that the device will not give back. -//! -//! The host puts the file on the volume before this machine exists, so its -//! bytes are on the stick and none of its pages are in the file cache. Writing -//! *inside* it is therefore a partial write into a page that has to be fetched -//! first — `file_cache::write_page` re-reads through `FatBacking::read_page` and -//! merges — which is the one code path a failed read can silently turn into a -//! page of zeros written back over real data. -//! -//! The machine's own log reaches this path too — `/system/bin/logd` appends to an -//! ordinary file and `fsync`s every batch, so its tail page is an ordinary -//! eviction candidate and the append after it loses one is this same re-fetch. -//! A host-written file is staged instead because it makes the trigger certain -//! rather than a matter of which page happened to be evicted. -//! -//! The *read* of the same page is the other half and is the sharper of the two: -//! `file_cache::read_page` returned `()`, so a page the device would not give -//! back reached the process as zeros under a success — which nothing above it -//! can tell from a file that really is zeros there. - -use std::fs::OpenOptions; -use std::io::{Seek, SeekFrom, Write}; - -/// Staged by the host in `tests/common/volumes.rs`. Two halves of one fixture. -const PATH: &str = "/log/staged-reread.txt"; -/// Inside the staged bytes, so the page is fetched rather than extended: past -/// the end there is nothing to preserve and `write_page` does not read at all. -const AT: u64 = 50; - -fn main() { - // First, and it costs the write below nothing: a fetch that failed is - // deliberately never made resident, so the write still misses and still - // re-fetches the same page. - match std::fs::read(PATH) { - Ok(bytes) => println!( - "reread: the read succeeded with {} bytes, {} of them zero", - bytes.len(), - bytes.iter().filter(|b| **b == 0).count() - ), - Err(e) => println!("reread: the read failed: {e}"), - } - - let mut file = match OpenOptions::new().write(true).open(PATH) { - Ok(file) => file, - Err(e) => { - println!("reread: {PATH} did not open: {e}"); - return; - } - }; - if let Err(e) = file.seek(SeekFrom::Start(AT)) { - println!("reread: seek failed: {e}"); - return; - } - match file.write_all(b"XXXXXXXX") { - // The defect: the read failed, the cache merged into zeros, and the - // caller was told the write worked. - Ok(()) => println!("reread: the write succeeded"), - Err(e) => println!("reread: the write failed: {e}"), - } -} diff --git a/tests/toyos-rust-tests/src/bin/partition_claimant.rs b/tests/toyos-rust-tests/src/bin/partition_claimant.rs index e9c7d2205c8..f2143e52132 100644 --- a/tests/toyos-rust-tests/src/bin/partition_claimant.rs +++ b/tests/toyos-rust-tests/src/bin/partition_claimant.rs @@ -1,6 +1,7 @@ -//! A GPT partition claimed as a device: its one holder reads and writes its -//! blocks and nobody else's, nothing the kernel holds can be claimed, and a -//! claim's fsync answers for its own writes. +//! A GPT partition on a disk the kernel drives, claimed as a device: its one +//! holder reads and writes its blocks and nobody else's, nothing the kernel +//! or a file server holds can be claimed, and a claim's fsync answers for its +//! own writes. //! //! The disks are crafted and judged by `tests/common/partclaim.rs` on the //! host: this binary's account of what it wrote is exactly what is in @@ -45,9 +46,7 @@ const GRANTED: &str = "A94F0E6D-3B2C-4E1A-8C7D-6E5F4A3B2C1D"; const MISALIGNED: &str = "3E8A1C5F-7D2B-4F60-9A1E-5C4B3D2E1F07"; /// Mirrored: a partition of whole 4 KiB blocks that begins inside one. const MISSTART: &str = "5A7C9E1B-3D5F-4B71-8C2E-4F6A8B0C2D35"; -/// Mirrored: a unique GUID the NVMe disk and the USB stick both carry. -const TWIN: &str = "6D2F9B41-8C3E-4A57-B1D0-2E4F6A8C0B13"; -/// Mirrored: DATA, which the kernel mounts at `/home`. +/// Mirrored: DATA, which fsd serves `/home` from through its claim. const DATA: &str = "E3A7C5D9-1B2F-4E6A-8D0C-9F7B5A3E1C24"; /// Mirrored: the partitions of the stick whose device leaves. const DEPARTING: &str = "1F3E5D7C-9B2A-4C6E-8F01-A3B5C7D9E2F4"; @@ -56,8 +55,8 @@ const EARLIER: &str = "4C6E8A02-3D5F-4179-A0B2-D4F6A8C0E2B4"; /// Mirrored: the target's length in blocks. const TARGET_BLOCKS: u64 = 2048; -/// Mirrored: the `/home` file written between the target's transfers, so the -/// kernel's own writes to the same disk are interleaved with the claim's. +/// Mirrored: the `/home` file written between the target's transfers, so +/// fsd's writes to the same disk are interleaved with the claim's. const HOME_FILE: &str = "/home/partclaim-interleaved.bin"; const HOME_CHUNK: usize = 32 * 1024; @@ -126,25 +125,17 @@ fn test(cap: &SysCap, boot_stick: &[String]) { let [esp, log, root] = boot_stick else { panic!("main takes the boot stick's ESP, log and ROOT GUIDs, got {boot_stick:?}"); }; - for (what, name) in [("the ESP", esp.as_str()), ("the log partition", log), ("ROOT", root), ("DATA", DATA)] - { - refused( - cap, - &format!("{what}, which the kernel has mounted,"), - guid(name), - SyscallError::PermissionDenied, - ); + refused(cap, "ROOT, which the kernel holds,", guid(root), SyscallError::PermissionDenied); + // init minted these for the file servers of their roles. + for (what, name) in [("the ESP", esp.as_str()), ("DATA", DATA)] { + refused(cap, &format!("{what}, which a file server holds,"), guid(name), SyscallError::AlreadyExists); } // init minted this one for test-runner from the manifest's `part:` row. refused(cap, "the partition init granted test-runner", guid(GRANTED), SyscallError::AlreadyExists); - refused( - cap, - "a unique GUID two disks carry", - guid(TWIN), - SyscallError::InvalidArgument, - ); + // The crafted disk carries a copy of the log partition's unique GUID. + refused(cap, "the log partition, whose unique GUID two disks carry,", guid(log), SyscallError::InvalidArgument); refused( cap, "a partition that is not whole 4 KiB blocks", @@ -180,8 +171,8 @@ fn test(cap: &SysCap, boot_stick: &[String]) { past_the_end(&target, info.blocks); - // The whole partition, first block to last, with the kernel's own writes - // to `/home` — the same NVMe disk — between the runs. + // The whole partition, first block to last, with fsd's writes to `/home` + // — the same disk — between the runs. let mut home = std::fs::File::create(HOME_FILE).expect("create the /home file"); let runs = info.blocks.div_ceil(MAX_BLOCKS_PER_CALL as u64); for run in 0..runs { diff --git a/tests/toyos-rust-tests/src/bin/quiesce_fsync.rs b/tests/toyos-rust-tests/src/bin/quiesce_fsync.rs deleted file mode 100644 index 956330352f2..00000000000 --- a/tests/toyos-rust-tests/src/bin/quiesce_fsync.rs +++ /dev/null @@ -1,94 +0,0 @@ -//! A flush parked over split FATs, and the machine's stop asked for above it. -//! -//! `quiesce-fsync-refuse` refuses the active FAT's write of every `SYS_FSYNC` -//! attempt on [`STAGED`] until its ladder closes, and the park between two -//! attempts is where a stop finds the caller. One thread here makes that -//! fsync; this one reads the kernel's log until the attempt that parks is -//! refused, and asks init for the shutdown — so the stop meets the thread -//! parked with the two FATs disagreeing. -//! -//! Nothing here asserts: `common::volumes::quiesce_leaves_the_volume_whole` -//! reads the volume and the kernel's lines. - -use std::fs::File; -use std::io::Write; -use std::time::{Duration, Instant}; - -use toyos::endow::{Endowments, SYSCAP_LABEL}; -use toyos::log::{LogTail, Record, MAX_LOG_SHARDS}; -use toyos::poller::{Poller, READABLE}; -use toyos::power::Stop; -use toyos::syscap::SysCap; - -/// The file the actuator refuses the flush of. Mirrored from the kernel's -/// `fat32_adapter::FSYNC_STAGED`. -const STAGED: &str = "/log/quiesce-fsync.bin"; - -/// The kernel's word for the second refused attempt: the first only yields, -/// and the second is the one whose caller parks. -const PARKS: &str = "quiesce-fsync-refuse: refusing a SYS_FSYNC flush's active-FAT write"; -const SECOND: &str = ", 2 of "; - -/// Enough clusters that the flush allocates, and so writes the FAT. -const CHUNK: usize = 8192; -const CHUNKS: usize = 8; - -/// Records per read; above the shard count, which the call refuses. -const BATCH: usize = 4 * MAX_LOG_SHARDS as usize; -const LOG_TOKEN: u64 = 1; - -/// How long the fsync gets to be refused the second time. -const PARKS_WITHIN: Duration = Duration::from_secs(5); - -fn main() { - let Some(cap) = Endowments::get().take::(SYSCAP_LABEL) else { - eprintln!("quiesce_fsync: this program was endowed no system capability"); - std::process::exit(1); - }; - std::thread::spawn(|| { - let mut f = File::create(STAGED).unwrap_or_else(|e| { - eprintln!("quiesce_fsync: could not create {STAGED}: {e}"); - std::process::exit(1); - }); - for _ in 0..CHUNKS { - f.write_all(&[b'f'; CHUNK]).unwrap_or_else(|e| { - eprintln!("quiesce_fsync: could not write {STAGED}: {e}"); - std::process::exit(1); - }); - } - // Comes back once the ladder closes, which the stop waits for. - if let Err(e) = f.sync_all() { - eprintln!("quiesce_fsync: the fsync failed: {e}"); - std::process::exit(1); - } - }); - - let mut tail = LogTail::new(); - let mut buf = [Record::EMPTY; BATCH]; - let poller = Poller::new(1); - let give_up = Instant::now() + PARKS_WITHIN; - poller.watch(&cap, READABLE, LOG_TOKEN); - poller.wait(0, 0, |_| {}); - loop { - let batch = tail.read(&cap, &mut buf).unwrap_or_else(|e| { - eprintln!("quiesce_fsync: the log would not read ({e:?})"); - std::process::exit(1); - }); - if batch.iter().any(|r| r.message().contains(PARKS) && r.message().contains(SECOND)) { - break; - } - if !batch.is_empty() { - continue; - } - let Some(left) = give_up.checked_duration_since(Instant::now()) else { - eprintln!("quiesce_fsync: the fsync of {STAGED} was not refused twice in {PARKS_WITHIN:?}"); - std::process::exit(1); - }; - poller.wait(1, left.as_nanos() as u64, |_| {}); - poller.watch(&cap, READABLE, LOG_TOKEN); - poller.wait(0, 0, |_| {}); - } - let refused = toyos::power::stop(Stop::Shutdown); - eprintln!("quiesce_fsync: init did not stop the machine ({refused:?})"); - std::process::exit(1); -} diff --git a/tests/toyos-rust-tests/src/bin/quiesce_twice.rs b/tests/toyos-rust-tests/src/bin/quiesce_twice.rs index f178eacdfc4..10f0bc2811c 100644 --- a/tests/toyos-rust-tests/src/bin/quiesce_twice.rs +++ b/tests/toyos-rust-tests/src/bin/quiesce_twice.rs @@ -1,7 +1,7 @@ //! Two callers of the stop, and the one that is refused. //! -//! init makes the first call, asked through its `power` port, after a closed -//! file's flush is left owed. `quiesce-last-park` holds that call once it has +//! init makes the first call, asked through its `power` port. +//! `quiesce-last-park` holds that call once it has //! claimed the stop and before it stops anything, until a thread named //! [`LAST_THREAD`] parks: the window in which every other thread still runs. //! This process reads the kernel's log until the kernel says it waits there, @@ -12,8 +12,6 @@ //! Nothing here asserts: `common::power::quiesce_refuses_a_second_shutdown` is //! the judge, and its doc is the scenario. -use std::fs::File; -use std::io::Write; use std::time::{Duration, Instant}; use toyos::endow::{Endowments, SYSCAP_LABEL}; @@ -28,11 +26,6 @@ use toyos_quiesce::LAST_THREAD; /// for the held thread. const WAITS: &str = "quiesce-last-park: the stop waits for"; -/// Bytes the owed file carries: enough clusters that its flush writes the -/// FAT, whose mirror half `quiesce-drain-refuse` refuses the stop's own drain. -const CHUNK: usize = 8192; -const CHUNKS: usize = 8; - /// Records per read; above the shard count, which the call refuses. const BATCH: usize = 4 * MAX_LOG_SHARDS as usize; @@ -52,19 +45,6 @@ fn main() { eprintln!("quiesce_twice: this program was endowed no system capability"); std::process::exit(1); }; - { - let mut f = File::create("/log/quiesce-owed.bin").unwrap_or_else(|e| { - eprintln!("quiesce_twice: could not create the owed file: {e}"); - std::process::exit(1); - }); - for _ in 0..CHUNKS { - f.write_all(&[b'o'; CHUNK]).unwrap_or_else(|e| { - eprintln!("quiesce_twice: could not write the owed file: {e}"); - std::process::exit(1); - }); - } - } - // The first call, from init. Comes back only refused. std::thread::spawn(|| { let refused = toyos::power::stop(Stop::Reboot); diff --git a/tests/toyos.rs b/tests/toyos.rs index 8d297271a8c..dd4621dc078 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -223,9 +223,6 @@ const RUST_SKIP: &[&str] = &[ // The same, and its verdict is the stop record of a boot staged around it. // `quiesce_wakes_on_the_last_park` and `quiesce_wakes_on_the_last_exit` run it. "quiesce_last", - // The same, and its verdict is the log volume the stop leaves. - // `quiesce_leaves_the_volume_whole` runs it. - "quiesce_fsync", // Its verdict is a count of what reached `/log`, which only a boot of its own // holds, and megabytes of it. `log_program_flood` runs it. "log_flood", @@ -381,12 +378,6 @@ const RUST_SKIP: &[&str] = &[ // Needs a boot image the harness staged a file into before the machine // started, which only `esp_filesystem` builds. "esp_files", - // Every question it asks has the *right* answer on an ordinary kernel, so - // on the shared boot it prints three successes and passes on its exit code - // — a second test of the same name whose verdict is vacuous. - // `boot_volume_metadata_error` runs it on the kernel that refuses the - // reads, which is the only build it says anything about. - "boot_volume_metadata_error", // Two modes, each waiting to be typed at through QMP; on its own nothing // ever answers it. `swiss_german_layout`, `locale_detect` and // `locale_detect_unrecognized` drive it. @@ -438,11 +429,8 @@ const RUST_SKIP: &[&str] = &[ // machine test of one of these that failed wide was re-run *as the shared // binary* and its `ALONE:` line was about a different test. What the shared // copy adds is the binary exiting 0 on a boot that gives it nothing to - // measure — `cache_eviction` in 132 ms against the 22.5 s its own device - // shape costs (run `31247206462`). + // measure. // - // `cache_eviction` needs the small NVMe that makes the cache evict at all. - "cache_eviction", // `writeback_reopen` and `writeback_spawn` each need their own boot with // `writeback-stall` armed; `writeback_durability` writes `/log` and is judged // host-side off the image after a shutdown. All three run as `MACHINE_TESTS`, @@ -469,17 +457,15 @@ const RUST_SKIP: &[&str] = &[ // was written that they are excluded from this boot; now they are. "audio_tone", "audio_tone_load", - // Its whole subject is a page of a file the host wrote onto the volume - // before the machine existed; the shared boot stages nothing, so it prints - // `did not open` and passes on its exit code. `log_backing_read_error` - // stages the file and reads the verdict. - "log_volume_reread", // Needs `usb-flush-fails` armed: on the shared boot the device flush // succeeds and its two must-refuse assertions red for an honest reason. // `fsync_failed_commit` boots it with the arm. "fsync_flush_failed", // Needs `so-cache-tiny` and the NVMe `/home`. `so_cache_refusals` gives both. "so_cache_policy", + // Needs a boot whose file servers are armed to end under its write, and + // ends DATA's for the rest of the boot. `fsd_restart` runs it. + "fs_restart", // Needs the NVMe `/home` and a boot of its own for the readback it is judged against; `home_overwrite_reads_back` runs it. "home_overwrite_zero", // Needs a boot where the DATA volume is ours and absent; on the shared @@ -999,6 +985,10 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("partition_claim_departure", Sched::Parallel, Tier::Fast), // The shared-object cache's two refusals. Body in `tests/common/storage.rs`. ("so_cache_refusals", Sched::Parallel, Tier::Fast), + // DATA's file server killed under an unanswered write, four times: its + // clients reopen, the fourth end closes /home to Gone, and the flushed + // files read back off the image. Body in `tests/common/storage.rs`. + ("fsd_restart", Sched::Parallel, Tier::Fast), // A same-length overwrite on /home, the guest's read held against the image. Body in `tests/common/storage.rs`. ("home_overwrite_reads_back", Sched::Parallel, Tier::Fast), // One filesystem under two paths: the guest writes under each of /apps and @@ -1026,9 +1016,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // Its own boot: it ends the machine, and its verdict is the order of // kernel lines. ("quiesce_refuses_a_second_shutdown", Sched::Parallel, Tier::Fast), - // Its own boot: it ends the machine, and its verdict is the volume that - // boot leaves. - ("quiesce_leaves_the_volume_whole", Sched::Parallel, Tier::Fast), // Its own boot each: it ends the machine, and its verdict is the stop // record that boot writes. ("quiesce_wakes_on_the_last_park", Sched::Parallel, Tier::Fast), @@ -1495,8 +1482,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // so what it stages is an ordering with no wall-clock margin on either // side: nothing here needs the serial tail. ("late_storage_connect", Sched::Parallel, Tier::Nightly), - ("log_backing_read_error", Sched::Parallel, Tier::Fast), - ("boot_volume_metadata_error", Sched::Parallel, Tier::Fast), ("log_partition_layout", Sched::Parallel, Tier::Fast), // What the loader does with a slot's ROOT: bytes its signature does not // cover, a parameter naming another, an overlapping partition and an @@ -1510,7 +1495,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("root_named_twice_on_the_boot_disk", Sched::Serial, Tier::Fast), ("root_named_twice", Sched::Serial, Tier::Nightly), ("log_partition_identity", Sched::Parallel, Tier::Nightly), - ("cache_eviction", Sched::Parallel, Tier::Nightly), // The write-back queue's three negative controls (wall 4 of // `issues/kernel/every-wait-in-this-kernel-is-a-spin.md`). `writeback_reopen` // and `writeback_spawn` arm `writeback-stall`, so each needs its own actuator @@ -1660,7 +1644,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("sched_check_build", &["test_rs_sched_stress"]), ("short_sleep_livelock", &["test_rs_abuse_short_sleep"]), ("heap_ceiling_recovery", &["test_rs_heap_ceiling"]), - ("cache_eviction", &["test_rs_cache_eviction"]), ("irq_census_conservation", &["test_rs_std_mmap"]), ("i8042_health_cadence", &["test_rs_i8042_keyboard"]), ("i8042_health", &["test_rs_i8042_keyboard"]), @@ -1731,7 +1714,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("data_candidate_with_bad_geometry_is_absent", &["test_rs_home_absent"]), ("home_overwrite_reads_back", &["test_rs_home_overwrite_zero"]), ("so_cache_refusals", &["test_rs_so_cache_policy"]), - ("boot_volume_metadata_error", &["test_rs_boot_volume_metadata_error"]), + ("fsd_restart", &["test_rs_fs_restart"]), ("esp_filesystem", &["test_rs_esp_files"]), ("log_flush_retry", &["test_rs_esp_files"]), ("fat_backing_revoked", &["test_rs_fat_backing_revoked"]), @@ -1739,7 +1722,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("fs_rename_durable", &["test_rs_fs_rename_durable", "test_rs_fs_dirs_durable"]), ("fsync_failed_commit", &["test_rs_fsync_flush_failed"]), ("ftruncate_flush_race", &["test_rs_ftruncate_flush_race", "test_rs_fs_rename_durable"]), - ("log_backing_read_error", &["test_rs_log_volume_reread"]), ("redirty_mid_flush", &["test_rs_redirty_mid_flush"]), ("writeback_durability", &["test_rs_writeback_durability", "test_rs_fat_backing_revoked"]), ("kernel_log_file", &["test_rs_writeback_durability"]), @@ -1770,7 +1752,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("quiesce_wakes_on_the_last_park", &["test_rs_quiesce_last"]), ("quiesce_wakes_on_the_last_exit", &["test_rs_quiesce_last"]), ("quiesce_dump_holds_the_stopped", &["test_rs_quiesce_writers"]), - ("quiesce_leaves_the_volume_whole", &["test_rs_quiesce_fsync"]), ("swap_crash_rolls_back", &["test_rs_swap_crash"]), ("swap_quiets_the_function", &["test_rs_swap_claim_idle"]), ("swap_keeps_what_nothing_reset", &["test_rs_swap_claim_running"]), @@ -2137,20 +2118,8 @@ const METAL: &[(&str, metal::Metal)] = &[ metal::Metal::Runs { arms: SELFTESTS, judge: |b| process_reopen(b[0].kernel().text()) }, ), ( - // **Two of its three probes pass here and the third cannot run.** - // Measured on a metal-shaped guest: the two `revoke-selftest` probes - // both PASS, and `pc-unbind-selftest` prints `FAIL (this boot has no - // metadata page cache)` — the slot it needs is one the virtio machine's - // NVMe home has and the T14's USB-only volumes do not. Judging the two - // that do run would be a different test under the same name, so the - // whole registration stays where it can answer for all three. "read_fault_selftests", - metal::Metal::QemuOnly( - "`pc-unbind-selftest` reports `FAIL (this boot has no metadata page cache)` on a \ - machine whose volumes are all on the boot stick; the two `revoke-selftest` probes \ - beside it pass, and splitting them into a name of their own is what would put that \ - half on the machine", - ), + metal::Metal::Runs { arms: SELFTESTS, judge: |b| read_fault_probes(b[0].kernel().text()) }, ), ( "leak_rollback_selftest", @@ -2308,10 +2277,7 @@ const SELFTESTS: &[metal::Arm] = &[metal::once( &[ "pci-cap-selftest", "process-reopen-selftest", - // `revoked-backing-selftest` and `pc-unbind-selftest` are deliberately - // absent: `read_fault_selftests` is the only rider they had and it is - // declared QEMU-only above, so arming them here would put a `FAIL` line - // on the stick that no verdict claims. + "revoked-backing-selftest", "leak-rollback-selftest", "lapic-spurious-selftest", "unclaimed-vector-selftest", @@ -5120,14 +5086,14 @@ fn run_screen_test( let text = dump.text(); print_screen(name, &text); - // Non-vacuity, and it is the half that matters: a boot whose log - // partition mounted would paint the ordinary line, and a screen + // Non-vacuity, and it is the half that matters: a boot whose loader + // named a log partition would paint the ordinary line, and a screen // asserted on without this would pass on a kernel that always says // the alarming thing. - if !console.contains("log-volume: not mounted") { + if !console.contains(common::volumes::NO_LOG_ALERT) { return Err(format!( - "the kernel mounted a log volume it was never given, so nothing here is \ - about a missing /log:\n{console}" + "the kernel was handed a log partition it was never given, so nothing here \ + is about a missing /log:\n{console}" )); } if console.contains("logd: this boot's kernel log is") { @@ -11443,6 +11409,7 @@ fn run_machine_test( storage::data_candidate_with_bad_geometry_is_absent(test_config, c_bins, rust_bins) } "so_cache_refusals" => storage::so_cache_refusals(test_config, c_bins, rust_bins), + "fsd_restart" => storage::fsd_restart(test_config, c_bins, rust_bins), "home_overwrite_reads_back" => { storage::home_overwrite_reads_back(test_config, c_bins, rust_bins) } @@ -11551,7 +11518,6 @@ fn run_machine_test( "ftruncate_flush_race" => common::volumes::ftruncate_flush_race(test_config, c_bins, rust_bins), "fs_rename_durable" => common::volumes::fs_rename_durable(test_config, c_bins, rust_bins), "fs_dirs_durable" => common::volumes::fs_dirs_durable(test_config, c_bins, rust_bins), - "quiesce_leaves_the_volume_whole" => common::volumes::quiesce_leaves_the_volume_whole(test_config, c_bins, rust_bins), // The lost-wake canary with the window it guards held open: every pipe // wait reads its condition, waits for a post to land, then parks, so // the ping-pong's posts land between the two. A commit that ignored the @@ -11977,12 +11943,6 @@ fn run_machine_test( "log_partition_identity" => { common::volumes::log_partition_identity(test_config, c_bins, rust_bins) } - "log_backing_read_error" => { - common::volumes::log_backing_read_error(test_config, c_bins, rust_bins) - } - "boot_volume_metadata_error" => { - common::volumes::boot_volume_metadata_error(test_config, c_bins, rust_bins) - } "usb_storage_write_error" => usb::usb_storage_write_error(test_config, c_bins, rust_bins), "usb_flush_optional" => usb::usb_flush_optional(test_config, c_bins, rust_bins), "xhci_deaf_registers" => usb::xhci_deaf_registers(test_config, c_bins, rust_bins), @@ -12807,11 +12767,10 @@ fn run_machine_test( } "nvme_large_device" => { // Device *size* is a shape dimension, and it is the one nobody had - // varied: every test image was small enough that an index sized - // per device block fit under the object allocator's 2 MiB ceiling, - // so the first boot on the laptop was the first time anything - // asked for a device-sized allocation — and it died in - // page_cache::init before it mounted anything. + // varied: the first boot on the laptop was the first time anything + // asked for a device-sized allocation. blockd drives the device + // and fsd serves DATA off it, and both have to come up on the + // T14's geometry. let options = BootOptions { profile: qemu::Profile::MetalDisk, ..Default::default() @@ -12823,7 +12782,7 @@ fn run_machine_test( // nothing until the guest's own driver says it enumerated a big // namespace. This is the number the T14 printed. let Some(blocks) = parse_nvme_blocks(&log) else { - return Err(format!("the NVMe driver printed no block count:\n{log}")); + return Err(format!("blockd printed no namespace size:\n{log}")); }; if blocks != qemu::NVME_T14_BLOCKS { return Err(format!( @@ -12832,28 +12791,6 @@ fn run_machine_test( )); } - // And the cache did not size its index by that number. - // - // The bound has to sit *below* the allocator's 2 MiB ceiling to be - // able to fire at all, which is a narrower window than it looks: - // a hashbrown index costs 17 B per bucket and its capacities are - // 7/8 of a power of two, so the last one that fits under the - // ceiling is 114,688 and the next is unreachable. 16,384 leaves - // room for a fixed reserve mirroring `slot_to_block`'s 4096 (which - // rounds up to 7168) and rejects every device-proportional reserve - // down to one entry per 4 MiB of disk. Measured red at 57,344, - // which is what `block_count / 1024` asks for and the allocator - // lets through. - let Some(index) = parse_page_cache_index(&log) else { - return Err(format!("the page cache printed no index size:\n{log}")); - }; - if index > 16_384 { - return Err(format!( - "the block index is sized for {index} blocks on a {blocks}-block device — \ - that is proportional to the device again:\n{log}" - )); - } - // The whole storage stack on the real geometry, not just the boot: // format, allocate, write, read back. let result = qemu.run_test("test_rs_nvme_home_roundtrip", Duration::from_secs(20)); @@ -12864,10 +12801,9 @@ fn run_machine_test( )); } - // Then shut down, which is the only thing that runs the page - // cache's write-back over every dirty slot the format left — - // ~1900 of them on a device this size against 8 on the small one, - // so the coalescing loop is only ever exercised at scale here. + // Then shut down, which has init ask fsd to sync every dirty block + // the format left — ~1900 of them on a device this size against 8 + // on the small one, so the write-back runs at scale only here. // // The kernel's own shutdown lines are observable now: the ring // is drained in `acpi::shutdown()` before it cuts the power. @@ -12899,7 +12835,7 @@ fn run_machine_test( // Ground truth at the hardware boundary: the backing file is what // the *device* received, so this is the one place a storage claim // does not rest on the guest's account of itself. The clean flag - // reaches the platter only through `PageCache::sync`, and the + // reaches the platter only through fsd's sync, and the // backup superblock only through a write at the far end of DATA on // a 244 GB device. Where DATA is comes out of the table, never out // of an offset this side computed. @@ -12937,8 +12873,8 @@ fn run_machine_test( )); } eprintln!( - " [nvme] {blocks} blocks, index sized for {index}; both superblocks clean; \ - image {} MiB on disk of {} GB apparent", + " [nvme] {blocks} blocks; both superblocks clean; image {} MiB on disk of {} GB \ + apparent", allocated / (1024 * 1024), apparent / 1_000_000_000 ); @@ -12970,36 +12906,20 @@ fn run_machine_test( "nightly_tier_is_announced" => nightly_tier_is_announced(), "nvme_wide_sector" => { // The other half of "a device's size is a shape dimension": not how - // many sectors, but how big one is. `lba_ds` is an 8-bit - // device-reported shift that reached `1 << lba_ds` and then - // `4096 / sector_size`, so an 8 KiB-format namespace divided by - // zero at 0.068 s — before storage, before a console, and on a - // machine whose only channel out is the one that does not exist - // yet. Every profile in this tree took QEMU's implicit 512-byte - // namespace, so nothing could ask. - // - // The guest is expected to die here, which is what makes - // `ready_marker` the driver's own refusal: anything but - // DEFAULT_READY tells the harness a panic is the outcome under - // test rather than a boot failure. - const REFUSAL: &str = "NVMe: namespace reports"; - let options = BootOptions { - profile: qemu::Profile::NvmeWideSector, - ready_marker: REFUSAL, - ..Default::default() - }; - /// A liveness margin over the work that follows the refusal, never a bound. - const AFTER_REFUSAL: Duration = Duration::from_secs(2); - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - // This profile has no virtio-serial, so stdio *is* the 16550 and - // `boot_log` is the whole record. It ends at the refusal, and the - // drain is the rest of the window the downstream work would be in. - let mut log = serial::Serial::boot(&qemu); - log.push(&qemu.drain_serial(AFTER_REFUSAL)); + // many sectors, but how big one is. The sector size is an 8-bit + // device-reported shift, and a driver that divided a 4 KiB block + // by it divided by zero on an 8 KiB-format namespace. Every profile + // in this tree took QEMU's implicit 512-byte namespace, so nothing + // could ask. blockd refuses the controller by name and serves + // nothing; the machine boots on without the disk. + let options = BootOptions { profile: qemu::Profile::NvmeWideSector, ..Default::default() }; + let qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); + let log = serial::Serial::boot(&qemu); // Named, not just refused: the value the device reported is the // whole diagnostic on a machine that will not boot again without // it. A bare "refused" line would pass with the number wrong. + log.must_say("blockd: NOT SERVING — ")?; log.must_say("2^13-byte sectors")?; // And it refused rather than dividing: the pre-fix failure was // `attempt to divide by zero`, which is also a panic and would @@ -13007,10 +12927,10 @@ fn run_machine_test( // are absence claims, so both go through `must_not_say`, which // fails rather than passing if the capture came back empty. log.must_not_say("divide by zero")?; - // Nothing downstream ran. `block device id=` is the line - // `NvmeBlockDevice::new` logs, and it is the call that divided. - log.must_not_say("NVMe: block device id=")?; - eprintln!(" [nvme] 8 KiB-format namespace refused by name, before storage came up"); + // Nothing downstream ran: no queue was made over the namespace. + log.must_not_say("blockd: NVMe up")?; + log.must_say("Boot: complete")?; + eprintln!(" [nvme] 8 KiB-format namespace refused by name, and the boot went on"); Ok(()) } "va_exhaustion" => { @@ -13824,182 +13744,6 @@ fn run_machine_test( eprintln!(" [heap] {}", line.trim()); Ok(()) } - "cache_eviction" => { - // Both disk caches grew for the life of the boot: nothing ever - // removed a block-cache slot, and the file cache's budget was - // `usize::MAX` because the one function that would have set it had - // no callers. This drives the bounds that replaced that. - // - // `test-small-caches` is the actuator for the same reason - // `xhci-one-slot` is: the shipped bounds are 16 MiB and 64 MiB on - // this guest, and filling them by doing real I/O is minutes of - // NVMe traffic to observe a policy that 256 KiB observes in a - // second. The eviction code is the shipped code — only the number - // moves, and the boot line below is what proves which number is in - // force. - // The T14's namespace, because the two caches are filled by - // different things. File pages come from the guest program below; - // metadata blocks come from the *device*, whose allocator bitmap - // is one bit per block — 1900 blocks of it on a 244 GB namespace - // against 8 on the 128 MiB one, which is the difference between - // overflowing a 64-slot cache during the format and never - // reaching it. Measured: 0 block-cache evictions on Headless. - // - // And it has to be an *unformatted* namespace, which the harness - // gives every boot that names no image: a mount of one an earlier - // boot formatted reads a handful of metadata blocks and evicts - // nothing, and the turnover assertion below goes red on that - // rather than vacuously green. - - // `nvme-spent-budget` and `nvme-command-silent` ride this boot - // rather than buying registered names of their own: each needs the - // test kernel and a real NVMe namespace, which is what this test - // already boots. The first costs one refused read before anything - // mounts the device — no command issued, no cache slot taken; the - // second costs one abandoned read and the controller reset that - // reclaims it, all before the mount, so the eviction series below - // runs on the freshly rebuilt queues — which is itself half the - // point: a reset that left them out of step reds the series. - let options = BootOptions { - profile: qemu::Profile::MetalDisk, - kernel_params: &["test-small-caches", "nvme-spent-budget", "nvme-command-silent"], - ..Default::default() - }; - let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - let boot = qemu.boot_log().to_string(); - - // The NVMe half of `block::OPERATION`. `usb-storage-gate` asserts - // the same refusal on the USB path; this is the one taken with both - // page-cache locks held, which is what made a missing deadline a - // wedged CPU rather than a slow read. - // - // Both lines, and the second is the one easy to leave out: a driver - // that refused by abandoning a command in flight would pass the - // first and fail here, because the queue would still be owed a - // completion and the DMA window still owed a write. - let console = serial::Serial::named("boot console", boot.as_str()); - console.must_say("nvme-gate: read with a spent budget refused=true budget=true")?; - console.must_say("nvme-gate: the same block read afterwards ok=true")?; - - // The reset escalation, NVMe 2.0 §3.7.2: a command whose completion - // wait was skipped is a live controller owing an answer, and until - // 2026-08-23 that single silence was a disk declared dead. Now it - // must be one reset, the silence answered as a budget word rather - // than a device fact, and the same block readable through the - // rebuilt queues. - console.must_say("nvme-gate: the silent command's read refused=true budget=true")?; - console.must_say("NVMe: controller reset complete")?; - console.must_say("nvme-gate: the same block read after the reset ok=true")?; - - let Some(file_budget) = parse_cache_budget(&boot, "file cache: budget ") else { - return Err(format!("the file cache printed no budget:\n{boot}")); - }; - let Some(block_budget) = parse_cache_budget(&boot, "cached blocks, cap ") else { - return Err(format!("the block cache printed no slot cap:\n{boot}")); - }; - if file_budget != 64 || block_budget != 64 { - return Err(format!( - "budgets are {file_budget} file pages and {block_budget} block slots, \ - not the 64 each the feature asks for — the bound under test is not the \ - one the workload was sized against:\n{boot}" - )); - } - - let result = qemu.run_test("test_rs_cache_eviction", Duration::from_secs(180)); - if !check_rust_result(&result) { - return Err(format!( - "a page did not survive being evicted and re-read:\n{}\n{}", - result.stdout, result.serial - )); - } - - // The whole point, and the half a compile cannot fake: residency - // is flat while the eviction count climbs. Boot and test output - // both, since the block cache starts evicting during the format. - let log = format!("{boot}\n{}", result.serial); - let file_series = parse_file_cache_series(&log); - let block_series = parse_cache_series(&log, "page cache: ", "slots resident"); - - // One turnover line means one eviction happened and nothing - // more; the workload is 8x the budget in each cache, so a - // series this short means eviction is not keeping up with the - // pressure — or is not running at all. - if file_series.len() < 4 { - return Err(format!( - "file cache: {} turnover lines, want at least 4 — {file_series:?}\n{log}", - file_series.len() - )); - } - if block_series.len() < 4 { - return Err(format!( - "block cache: {} turnover lines, want at least 4 — {block_series:?}\n{log}", - block_series.len() - )); - } - // The file cache's bound is the derivation, not an absolute: - // eviction never takes a dirty page and gives up only when - // everything resident is dirty, so an over-budget sample is - // lawful exactly when its own line says dirty == resident — a - // clean overage still reds. The guest stages the overage on - // every run, and every episode must close with a sample back - // within the bound (the kernel prints the close unconditionally). - let mut over_samples = 0usize; - for &(evictions, resident, dirty) in &file_series { - if resident > file_budget { - over_samples += 1; - if dirty != resident { - return Err(format!( - "file cache: {resident} entries resident against a {file_budget} \ - bound after {evictions} evictions with only {dirty} dirty — the \ - overage is not the un-flushed working set, so eviction failed to \ - take a clean page it was allowed to:\n{log}" - )); - } - } - } - if over_samples == 0 { - return Err(format!( - "the staged all-dirty overage never printed an over-budget sample, so the \ - budget's one declared escape ran unobserved:\n{log}" - )); - } - let last_over = file_series - .iter() - .rposition(|&(_, resident, _)| resident > file_budget) - .expect("over_samples > 0 was checked above"); - if !file_series[last_over + 1..].iter().any(|&(_, r, _)| r <= file_budget) { - return Err(format!( - "file cache: the last over-budget sample is never followed by one back \ - within the bound, after every writer flushed — the overage outlived its \ - excuse:\n{log}" - )); - } - for &(evictions, resident) in &block_series { - if resident > block_budget { - return Err(format!( - "block cache: {resident} entries resident against a {block_budget} bound \ - after {evictions} evictions — the bound does not hold:\n{log}" - )); - } - } - if file_series[file_series.len() - 1].0 <= file_series[0].0 { - return Err(format!("file cache: eviction count never advanced: {file_series:?}")); - } - if block_series[block_series.len() - 1].0 <= block_series[0].0 { - return Err(format!("block cache: eviction count never advanced: {block_series:?}")); - } - - eprintln!( - " [cache] file {} evictions over {} turnovers ({over_samples} lawful all-dirty \ - over-budget sample(s)), block {} evictions over {}; clean residency never above \ - {file_budget}/{block_budget}", - file_series[file_series.len() - 1].0, - file_series.len(), - block_series[block_series.len() - 1].0, - block_series.len() - ); - Ok(()) - } "xhci_slot_exhaustion" => { // A device count is untrusted input: more devices than the driver // has room for must cost those devices and nothing else. QEMU @@ -14465,10 +14209,10 @@ fn run_machine_test( // Nothing host-side can read it: the type reaches `percpu::cpu_id` // and `driver::current_handle`, and `kernel/` is excluded from the // host workspace, so a `Operation` cannot be constructed off a - // booted machine. The other gates that drive an establishment — - // `cache_eviction`'s `nvme-spent-budget`, `usb_storage_gate`'s — - // prove a narrowing happened by the refusal it produces and read - // none of the values, and all of them establish from a boot phase, + // booted machine. The other gate that drives an establishment — + // `usb_storage_gate`'s — proves a narrowing happened by the refusal + // it produces and reads none of the values, and establishes from a + // boot phase, // which is the *task-less* slot. `kernel/src/sched_gate.rs` runs // three nested establishments with known deadlines in both homes // and prints what every level saw. @@ -14489,9 +14233,9 @@ fn run_machine_test( operation_nesting_log(qemu.boot_log()) } "leak_rollback_selftest" => { - // Two "acquire before a fallible step" controls run in the kernel at - // boot: each prints PASS only when the in-tree count returned to its - // baseline after a refused call; reverting either fix prints FAIL. + // An "acquire before a fallible step" control runs in the kernel at + // boot: it prints PASS only when the in-tree count returned to its + // baseline after a refused call; reverting the fix prints FAIL. let qemu = QemuInstance::boot_with_options( test_config, c_bins, @@ -14519,25 +14263,19 @@ fn run_machine_test( process_reopen(qemu.boot_log()) } "driver_wait_refused" => { - // The actuators blind CSTS.RDY and DEVICE_STATUS, staging a - // controller that never answers; the boot must come up naming the - // refused register — on the unbounded shape it never reaches ready. + // The actuator blinds DEVICE_STATUS, staging a controller that + // never answers; the boot must come up naming the refused register + // — on the unbounded shape it never reaches ready. let qemu = QemuInstance::boot_with_options( test_config, c_bins, rust_bins, BootOptions { - kernel_params: &["nvme-rdy-stuck", "virtio-reset-stuck"], + kernel_params: &["virtio-reset-stuck"], ..Default::default() }, ); let log = qemu.boot_log().to_string(); - let Some(nvme) = log.lines().find(|l| l.contains("NVMe: NOT INITIALISED")) else { - return Err(format!("the stuck NVMe was never refused by name:\n{log}")); - }; - if !nvme.contains("CSTS.RDY would not set in") { - return Err(format!("the NVMe refusal does not name the register: {nvme}")); - } let virtio = log .lines() .filter(|l| l.contains("did not zero DEVICE_STATUS for its reset")) @@ -14545,7 +14283,6 @@ fn run_machine_test( if virtio == 0 { return Err(format!("no stuck virtio device was refused by name:\n{log}")); } - eprintln!(" [waits] {}", nvme.trim()); eprintln!( " [waits] {virtio} virtio device(s) refused on the reset budget; the boot \ came up without them" @@ -14553,15 +14290,14 @@ fn run_machine_test( Ok(()) } "read_fault_selftests" => { - // Three kernel-boot controls: a backing read after deletion is - // refused on both writable mounts, and a page-cache slot whose fill - // the device refused is unbound. Reverting either prints FAIL. + // A kernel-boot control: a backing read after deletion is refused + // on `/tmp`. Reverting it prints FAIL. let qemu = QemuInstance::boot_with_options( test_config, c_bins, rust_bins, BootOptions { - kernel_params: &["revoked-backing-selftest", "pc-unbind-selftest"], + kernel_params: &["revoked-backing-selftest"], ..Default::default() }, ); @@ -16469,76 +16205,11 @@ fn parse_mouse_events(stdout: &str) -> Vec { /// The block count the NVMe driver derived, out of /// `NVMe: block device id=1 blocks=62514774 (244198MB)`. fn parse_nvme_blocks(log: &str) -> Option { - log.lines() - .find_map(|l| l.split("NVMe: block device id=").nth(1))? - .split("blocks=") - .nth(1)? - .split_whitespace() - .next()? - .parse() - .ok() -} - -/// The first number after `marker`, which both caches print their ceiling as -/// exactly once at boot. -fn parse_cache_budget(log: &str, marker: &str) -> Option { - log.lines() - .find_map(|l| l.split(marker).nth(1))? - .split_whitespace() - .next()? - .parse() - .ok() -} - -/// Every `N evictions, R/M ` line, as (evictions, resident). -/// -/// The kernel emits one per full turnover of the cache, so the series is the -/// shape of the answer: a cache that evicts has a climbing first column and a -/// flat second, and a cache that only grows has no lines at all. -fn parse_cache_series(log: &str, prefix: &str, unit: &str) -> Vec<(u64, u64)> { - log.lines() - .filter_map(|l| { - let tail = l.split(prefix).nth(1)?; - if !tail.contains(unit) { - return None; - } - let evictions = tail.split(" evictions,").next()?.trim().parse().ok()?; - let resident = tail.split("evictions, ").nth(1)?.split('/').next()?.parse().ok()?; - Some((evictions, resident)) - }) - .collect() -} - -/// `file cache: E evictions, R/M pages resident, D dirty` as (E, R, D). -/// -/// The dirty count is required, not optional: it is the only lawful reading -/// of a sample over budget, so a kernel line that stops carrying it drops -/// out of the series and fails the length assertion rather than passing as -/// a bound nobody checked. -fn parse_file_cache_series(log: &str) -> Vec<(u64, u64, u64)> { - log.lines() - .filter_map(|l| { - let tail = l.split("file cache: ").nth(1)?; - if !tail.contains("pages resident") { - return None; - } - let evictions = tail.split(" evictions,").next()?.trim().parse().ok()?; - let resident = tail.split("evictions, ").nth(1)?.split('/').next()?.parse().ok()?; - let dirty = tail.split("resident, ").nth(1)?.split(" dirty").next()?.parse().ok()?; - Some((evictions, resident, dirty)) - }) - .collect() -} - -/// How many blocks the page cache's index has room for, out of -/// `page cache: … index sized for C cached blocks, cap S slots, B index bytes`. -fn parse_page_cache_index(log: &str) -> Option { - log.lines() - .find_map(|l| l.split("index sized for ").nth(1))? - .split_whitespace() - .next()? - .parse() - .ok() + let up = log.lines().find_map(|l| l.split("blockd: NVMe up: ").nth(1))?; + let (sector, rest) = up.split_once("-byte sectors, ")?; + let bytes: u64 = sector.rsplit(' ').next()?.parse().ok()?; + let sectors: u64 = rest.split_whitespace().next()?.parse().ok()?; + Some(sectors * bytes / 4096) } /// Decode one bcachefs superblock straight out of a disk image, with the @@ -16814,11 +16485,7 @@ fn process_reopen(log: &str) -> Result<(), String> { /// Text in, a verdict out: every line it reads is a kernel record, so the /// T14's readback and a QEMU boot log are judged by this one predicate. fn read_fault_probes(log: &str) -> Result<(), String> { - for probe in [ - "revoke-selftest: /tmp/revoke_probe", - "revoke-selftest: /home/revoke_probe", - "pc-unbind-selftest:", - ] { + for probe in ["revoke-selftest: /tmp/revoke_probe"] { let Some(verdict) = log.lines().find(|l| l.contains(probe)) else { return Err(format!("{probe} never ran:\n{log}")); }; @@ -16830,12 +16497,12 @@ fn read_fault_probes(log: &str) -> Result<(), String> { Ok(()) } -/// Two "acquire before a fallible step" controls: each count returned to its baseline after a refused call. +/// An "acquire before a fallible step" control: the count returned to its baseline after a refused call. /// /// Text in, a verdict out: every line it reads is a kernel record, so the /// T14's readback and a QEMU boot log are judged by this one predicate. fn leak_rollback(log: &str) -> Result<(), String> { - for probe in ["leak-selftest: device-mint", "leak-selftest: fat-reopen"] { + for probe in ["leak-selftest: device-mint"] { let Some(verdict) = log.lines().find(|l| l.contains(probe)) else { return Err(format!("{probe} never ran:\n{log}")); }; diff --git a/userland/blockd/src/main.rs b/userland/blockd/src/main.rs index 215b1468c4e..91f31ace2ed 100644 --- a/userland/blockd/src/main.rs +++ b/userland/blockd/src/main.rs @@ -537,7 +537,15 @@ fn main() { println!("blockd: no NVMe controller this row names is on this machine; serving no partition"); serve_nothing(&acceptor); }; - let mut ctrl = Controller::open(dev, silence).unwrap_or_else(|why| panic!("blockd: NOT SERVING — {why}")); + // A controller this service cannot use is a machine without one, said by + // name: restarting would meet the same device and the same refusal. + let mut ctrl = match Controller::open(dev, silence) { + Ok(ctrl) => ctrl, + Err(why) => { + println!("blockd: NOT SERVING — {why}; serving no partition"); + serve_nothing(&acceptor); + } + }; println!( "blockd: NVMe up: {} I/O queues of {} commands, volatile write cache {}, {}-byte sectors, \ {} sectors", diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index 10da11f2a69..3be2de02fd5 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -390,3 +390,80 @@ impl Volume for FatVolume { ) } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::disk::{DiskError, Ram}; + + /// A disk that refuses every read of one block, after the fixture's own + /// bytes are on it. + struct Refusing { + ram: Ram, + refused: u64, + } + + impl Disk for Refusing { + fn blocks(&self) -> u64 { + self.ram.blocks() + } + fn read(&mut self, first: u64, out: &mut [u8]) -> Result<(), DiskError> { + let count = (out.len() / BLOCK) as u64; + if (first..first + count).contains(&self.refused) { + return Err(DiskError::Device); + } + self.ram.read(first, out) + } + fn write(&mut self, first: u64, data: &[u8]) -> Result<(), DiskError> { + self.ram.write(first, data) + } + fn flush(&mut self) -> Result<(), DiskError> { + self.ram.flush() + } + } + + /// Four blocks holding `0x5A`, the second of which refuses every read. + fn bytes() -> Bytes { + let mut ram = Ram::new(4); + ram.write(0, &[0x5A; 4 * BLOCK]).unwrap(); + let cache = Rc::new(Cache::new(Refusing { ram, refused: 1 })); + Bytes { cache, len: 4 * BLOCK as u64 } + } + + /// A read the device refused is the device's word to the caller, never a + /// buffer of zeros: a caller that took zeros for data merges its bytes + /// into them and writes the result over what the medium held. + #[test] + fn a_refused_read_is_an_error_and_never_zeros() { + let mut b = bytes(); + let mut buf = [0xEEu8; 100]; + assert_eq!(b.read_at(BLOCK as u64 + 10, &mut buf), Err(IoError::Device)); + let mut fine = [0u8; 100]; + b.read_at(10, &mut fine).unwrap(); + assert!(fine.iter().all(|&x| x == 0x5A), "the readable block reads as written"); + } + + /// A write that covers part of a block reads the rest of it first; when + /// that read is refused the write is refused whole, and nothing reaches + /// the block — the rest of it is another file's, or the table's. + #[test] + fn a_partial_write_over_an_unreadable_block_writes_nothing() { + let mut b = bytes(); + assert_eq!(b.write_at(BLOCK as u64 + 10, &[1, 2, 3]), Err(IoError::Device)); + b.flush().unwrap(); + let cache = Rc::try_unwrap(b.cache).ok().expect("the one holder"); + let mut disk = cache.into_disk(); + disk.refused = u64::MAX; + let mut block = [0u8; BLOCK]; + disk.read(1, &mut block).unwrap(); + assert!(block.iter().all(|&x| x == 0x5A), "the unreadable block was written over"); + } + + /// What the driver says of a volume that stopped answering is `Io`, which + /// no caller takes for a name that is not there. + #[test] + fn a_device_error_is_io_and_not_not_found() { + assert_eq!(word(Error::Io), SyscallError::Io); + assert_eq!(word(Error::NotFound), SyscallError::NotFound); + } +} diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index 0219e8de7e1..0375997a184 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -5,7 +5,9 @@ //! by init under `serve:fs:`; for its volume, a claim on a partition of a //! disk the kernel drives, or blockd's `block` connector in its namespace; and //! nothing else of the machine. Argv is the role, and for LOG and BOOT the -//! unique GUID of the partition the loader named for it. +//! unique GUID of the partition the loader named for it when no claim on it +//! was minted; then the row's own arguments, of which there is one, a test's +//! actuator ([`END_ON`]). //! //! **A connection is bound to the directory whose port it came in on**, and //! every path on it is resolved there (`fsd::resolve`); a write on a port @@ -62,6 +64,12 @@ const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(2); /// blocks, of which only what is written costs anything. const RAM_BLOCKS: u64 = 1 << 18; +/// `--end-on `: a write through a file opened to append at `path` on the +/// volume ends this process once the volume has taken it and before it is +/// answered — a server killed with a write done and unanswered, as a test +/// stages one. Armed by nothing but a boot config's `args`. +const END_ON: &str = "--end-on"; + const TOKEN_CLIENT: u64 = 1 << 32; const TOKEN_STREAM: u64 = 2 << 32; @@ -98,6 +106,8 @@ struct Fid { node: Node, write: bool, append: bool, + /// Opened to append at [`END_ON`]'s path. + ends: bool, } struct Client { @@ -146,15 +156,26 @@ fn main() { let role = args.get(1).and_then(|r| Role::parse(r)).unwrap_or_else(|| { panic!("fsd: started with {args:?}; the first argument is a role: data, log or boot") }); + let (mut guid, mut end_on) = (None, None); + let mut rest = args.iter().skip(2); + while let Some(arg) = rest.next() { + match arg.as_str() { + END_ON => end_on = Some(rest.next().unwrap_or_else(|| panic!("fsd: {END_ON} takes a path")).clone()), + flag if flag.starts_with("--") => panic!("fsd: {flag} is no argument of this server"), + named if guid.is_none() => guid = Some(named), + extra => panic!("fsd: a second partition, {extra}, after {guid:?}"), + } + } let caps = capabilities(role); let roots: Vec = caps.iter().map(|c| c.root.clone()).collect(); let roots: Vec<&str> = roots.iter().map(String::as_str).collect(); - let volume = open_volume(role, args.get(2).map(String::as_str), &roots); + let volume = open_volume(role, guid, &roots); println!( "fsd: {role:?} serving {} — {}", caps.iter().map(|c| c.dir.as_str()).collect::>().join(", "), volume.describe() ); + let caps_len = caps.len() as u32; Server { volume, caps, @@ -163,6 +184,8 @@ fn main() { streams: BTreeMap::new(), next_stream: 0, dirty_since: None, + end_on, + probe: Poller::new(caps_len), } .serve() } @@ -256,7 +279,13 @@ fn open_volume(role: Role, guid: Option<&str>, roots: &[&str]) -> Box Box::new(Absent::new(roots, format!("no {role:?} partition on a disk this server reaches"))), + None => Box::new(Absent::new( + roots, + match guid { + Some(guid) => format!("the {role:?} partition {guid} is on no disk this server reaches"), + None => format!("the loader named no {role:?} partition"), + }, + )), } } } @@ -294,6 +323,11 @@ struct Server { next_stream: u64, /// When a write first went unsynced. dirty_since: Option, + /// [`END_ON`]'s path on the volume. + end_on: Option, + /// Asks an acceptor whether a connection waits, before [`Server::accept`] + /// takes it. + probe: Poller, } /// What one request is answered. @@ -373,7 +407,23 @@ impl Server { } } + /// Take a connection that waits on `cap`'s port, if one does. + /// + /// **Asked first, because `accept` parks and a completion is a hint**: a + /// watch replaced while a connection arrives can answer beside the watch + /// that replaced it, so two completions name one connection, and the + /// second `accept` would park this server for good. This process is the + /// port's one acceptor, so a connection the probe sees is still there to + /// take. The probe's ring is drained whole each time and a completion is + /// read by its port's token, so what counts is an arrival on this port + /// since its last probe, none of them taken. fn accept(&mut self, cap: usize) { + self.probe.watch(&self.caps[cap].acceptor, READABLE, cap as u64); + let mut waiting = false; + self.probe.wait(0, 0, |token| waiting |= token == cap as u64); + if !waiting { + return; + } let conn = match self.caps[cap].acceptor.accept() { Ok(conn) => conn, Err(why) => panic!("fsd: its own acceptor refused an accept: {why:?}"), @@ -402,44 +452,40 @@ impl Server { } } - /// Everything one client has sent, a whole request at a time. + /// One request of one client, and no more: a client whose next request + /// is already queued is answered at the next wait, after every other + /// client that was ready, since a watch on a connection with a frame + /// queued fires at once, and `FrameRx` reads no byte past the frame. A + /// loop until the client went quiet would let one that asks as fast as it + /// is answered hold every other off for as long as it kept asking. fn pump(&mut self, id: u64) { - loop { - let Some(client) = self.clients.get_mut(&id) else { return }; - match client.rx.pump(&client.conn) { - RxStep::Idle => return, - RxStep::Eof => return self.drop_client(id, ""), - RxStep::Malformed => return self.drop_client(id, "it sent a frame this protocol cannot describe"), - RxStep::Frame { msg_type, payload_len } => { - let Ok(request) = ipc::decode_payload::(client.rx.payload(payload_len)) else { - return self.drop_client(id, "its request was short"); - }; - let answer = self.answer(id, msg_type, request); - if !self.reply(id, answer) { - return; - } - } + let Some(client) = self.clients.get_mut(&id) else { return }; + match client.rx.pump(&client.conn) { + RxStep::Idle => {} + RxStep::Eof => self.drop_client(id, ""), + RxStep::Malformed => self.drop_client(id, "it sent a frame this protocol cannot describe"), + RxStep::Frame { msg_type, payload_len } => { + let Ok(request) = ipc::decode_payload::(client.rx.payload(payload_len)) else { + return self.drop_client(id, "its request was short"); + }; + let answer = self.answer(id, msg_type, request); + self.reply(id, answer); } } } - /// Send `answer`; `false` once the client is gone. - fn reply(&mut self, id: u64, answer: Answer) -> bool { - let Some(client) = self.clients.get(&id) else { return false }; + /// Send `answer`, or let the client go when it will not take it. + fn reply(&mut self, id: u64, answer: Answer) { + let Some(client) = self.clients.get(&id) else { return }; let sent = match answer { Answer::Reply(reply) => client.conn.try_send(REPLY, &reply), Answer::Link(len) => client.conn.try_send(LINK, &Reply { value: len as u64, ..Reply::ok() }), Answer::WithHandle(reply, handle) => client.conn.try_send_with_handles(&[handle], REPLY, &reply), - Answer::Drop(why) => { - self.drop_client(id, why); - return false; - } + Answer::Drop(why) => return self.drop_client(id, why), }; if let Err(why) = sent { self.drop_client(id, &format!("its connection would not take a reply ({why:?})")); - return false; } - true } /// A path from the window: `len` bytes at `at`, UTF-8. @@ -566,7 +612,10 @@ impl Server { let fid = client.next_fid; client.next_fid += 1; let write = r.flags & (O_WRITE | O_APPEND) != 0; - client.fids.insert(fid, Fid { node, write, append: r.flags & O_APPEND != 0 }); + let append = r.flags & O_APPEND != 0; + let ends = append && self.end_on.as_deref() == Some(path.as_str()); + let client = self.clients.get_mut(&id).expect("pumped"); + client.fids.insert(fid, Fid { node, write, append, ends }); Ok(Answer::Reply(Reply { value: fid, ..stat_reply(meta) })) } CLOSE => { @@ -584,9 +633,9 @@ impl Server { Ok(Answer::Reply(Reply { value: n as u64, ..Reply::ok() })) } WRITE => { - let (node, write, append) = { + let (node, write, append, ends) = { let f = self.fid(id, r.fid)?; - (f.node, f.write, f.append) + (f.node, f.write, f.append, f.ends) }; if !write { return Err(SyscallError::PermissionDenied); @@ -602,6 +651,10 @@ impl Server { return Err(SyscallError::InvalidArgument); } self.volume.write(node, at, &data)?; + if ends { + println!("fsd: {END_ON}: ending with a write done and unanswered"); + std::process::exit(1); + } self.dirtied(); Ok(Answer::Reply(Reply { value: len as u64, value2: at + len as u64, ..Reply::ok() })) } diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index 0018b79ce11..092ab980372 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -1056,10 +1056,10 @@ impl<'a> Init<'a> { /// Have every writable file server make its volume durable, each bounded /// by [`FLUSH_BOUND`]: the kernel's stop waits for no process, so what a /// server holds and has not written is lost unless it is asked first. - /// `/log` is `logd`'s flush's, made durable before this. + /// After `logd`'s flush, so the log's server writes what logd wrote last. fn sync_files(&self) { let roles = self.system.programs.iter().flat_map(|p| p.roles.iter()); - for role in roles.filter(|r| *r != "log" && *r != "boot") { + for role in roles.filter(|r| *r != "boot") { let Some(dir) = toyos_manifest::role_dirs(role).and_then(|dirs| dirs.first()) else { continue }; let name = format!("{CAPABILITY_PREFIX}{dir}"); let (done_tx, done_rx) = std::sync::mpsc::channel(); @@ -1600,8 +1600,9 @@ fn start<'a>( storage: Storage, output: Output<'_>, ) -> std::io::Result<(Child, Vec)> { - command.args(&program.args); + // A storage row's own arguments first: a file server's role leads its argv. command.args(&storage.args); + command.args(&program.args); let booting = matches!(output, Output::Boot(_)); // **Set here, over whatever a launching caller carried**: the row decides diff --git a/userland/logd/src/inspect.rs b/userland/logd/src/inspect.rs index 1066cd611b7..d4a1a675dbf 100644 --- a/userland/logd/src/inspect.rs +++ b/userland/logd/src/inspect.rs @@ -78,19 +78,16 @@ pub struct Published { file: Mutex>, bytes: AtomicU64, lost: AtomicU64, - stream: bool, } impl Published { - /// `stream` is whether this boot's log is served on the network at all. - pub fn new(stream: bool) -> Self { + pub fn new() -> Self { Self { state: AtomicU8::new(State::ConsoleOnly as u8), part: AtomicU32::new(0), file: Mutex::new(None), bytes: AtomicU64::new(0), lost: AtomicU64::new(0), - stream, } } @@ -108,7 +105,8 @@ impl Published { } } - fn snapshot(&self) -> Vec { + /// `stream` is whether this boot's log is served on the network at all. + fn snapshot(&self, stream: bool) -> Vec { let mut snap = Snapshot::new(toyos_inspect::LOG); snap.put("volume.state", State::word(self.state.load(Ordering::Relaxed))); let file = self.file.lock().expect("the loop does not panic holding this").clone(); @@ -118,7 +116,7 @@ impl Published { snap.put("volume.bytes", self.bytes.load(Ordering::Relaxed)); } snap.put("records.lost", self.lost.load(Ordering::Relaxed)); - snap.put("stream", if self.stream { "on" } else { "off" }); + snap.put("stream", if stream { "on" } else { "off" }); snap.encode().unwrap_or_else(|why| panic!("logd: its snapshot: {why}")) } } @@ -175,7 +173,7 @@ fn run(acceptor: &Acceptor, published: &Published, hub: &Hub) -> ! { match p.rx.pump(&p.conn) { RxStep::Idle => true, RxStep::Frame { msg_type: toyos_inspect::MSG_INSPECT, payload_len: 0 } => { - let _ = p.conn.try_send_bytes(toyos_inspect::MSG_SNAPSHOT, &published.snapshot()); + let _ = p.conn.try_send_bytes(toyos_inspect::MSG_SNAPSHOT, &published.snapshot(hub.network())); false } RxStep::Frame { msg_type: READ, payload_len: 0 } => { diff --git a/userland/logd/src/main.rs b/userland/logd/src/main.rs index 86f4167eee3..05c4b8a1cd8 100644 --- a/userland/logd/src/main.rs +++ b/userland/logd/src/main.rs @@ -179,7 +179,7 @@ fn main() { } let hub = Arc::new(serve::Hub::start(REPLAY_BYTES, boot_local)); - let published = Arc::new(inspect::Published::new(hub.network())); + let published = Arc::new(inspect::Published::new()); if let Some(acceptor) = endow::acceptor(SERVICE) { inspect::serve(acceptor, Arc::clone(&published), Arc::clone(&hub)); } diff --git a/userland/logd/src/serve.rs b/userland/logd/src/serve.rs index 3fdc05151b5..069608d401b 100644 --- a/userland/logd/src/serve.rs +++ b/userland/logd/src/serve.rs @@ -27,7 +27,7 @@ use std::io::Write; use std::os::fd::AsRawFd; -use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; use std::sync::{Arc, Condvar, Mutex}; use std::time::Duration; @@ -84,6 +84,8 @@ struct Shared { grew: Condvar, network: AtomicUsize, local: AtomicUsize, + /// The network server bound [`PORT`]: this program's namespace holds netd. + serving: AtomicBool, /// The wall clock the boot started at, for a line a reader is owed. boot_local: Option, } @@ -98,11 +100,13 @@ impl Hub { grew: Condvar::new(), network: AtomicUsize::new(0), local: AtomicUsize::new(0), + serving: AtomicBool::new(false), boot_local, }); - // A row with no `receives` gives this program no namespace, so no netd, - // and no thread to learn so on: its exit would be a kernel record at a - // time nothing orders, after a shutdown's last word included. + // A row with no namespace has no netd, and no thread to learn so on: + // its exit would be a kernel record at a time nothing orders, after a + // shutdown's last word included. One with a namespace learns it at + // its first bind, at boot. let mut carrier = None; if toyos::endow::namespace().is_some() { let network = Arc::clone(&shared); @@ -116,9 +120,10 @@ impl Hub { Self { shared, carrier } } - /// Whether this boot's log is served on the network at all. + /// Whether this boot's log is served on the network at all: the network + /// server found netd and bound its port. pub fn network(&self) -> bool { - self.carrier.is_some() + self.shared.serving.load(Ordering::Relaxed) } /// A reader on this machine that asked for the log: it gets the read end @@ -171,6 +176,7 @@ fn serve_network(shared: &Arc, told: &Pipe) { const TOLD: u64 = 0; const READY: u64 = 1; let Some(first) = bind_port() else { return }; + shared.serving.store(true, Ordering::Relaxed); let mut listener = Some(first); let poller = Poller::new(2); loop { From dc447c2bca0a884f82489f2863fafc597ae172fa Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 07:56:02 +0200 Subject: [PATCH 03/54] rust: pin the fork's wt-toyos-fsd, std's served files over main's rust-lld pin Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- rust | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/rust b/rust index fbf6ad143d8..20436ec20e9 160000 --- a/rust +++ b/rust @@ -1 +1 @@ -Subproject commit fbf6ad143d8311d61fb61bf18d71d596c385abd2 +Subproject commit 20436ec20e945a8fb66434d3809b39983c0a89c3 From 193600cdf18d502b1e71aa19d60ca69069576f68 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 08:30:56 +0200 Subject: [PATCH 04/54] storage: the nightly tests onto the file servers; the tracker brought up to the tree The nightly storage tests are retargeted onto fsd, blockd and the USB partition claims, or deleted with their kernel subject (writeback's FAT arms, the truncate race and its actuator). The claim path says when a flush was answered on a retry. Every program keeps /boot in its view until the rows declare their own. Issues whose subject left the kernel are closed or re-pointed at fsd, and the weaknesses this change leaves are filed: C programs, a link from a kernel mount, a half-written btree, the slots of an NVMe-installed image, a file server's reach over blockd, duplicate completions, the metal read span, and the cached read's cost. The SDK crates take their minors: toyos-abi 0.17.0, toyos 0.19.0, toyos-window 0.21.0. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- issues/README.md | 11 +- issues/audio/disk-wait-pins-a-cpu.md | 6 + ...eds-intermittently-and-nothing-says-why.md | 45 -- ...r-classify-trusts-a-lying-transfererror.md | 13 +- ...-server-costs-fifteen-times-the-kernels.md | 32 ++ ...ueued-free-is-answered-by-the-next-call.md | 6 +- ...n-repair-is-named-by-which-error-it-was.md | 15 +- .../a-directory-on-data-survives-no-reboot.md | 33 -- ...el-mount-to-a-served-path-leads-nowhere.md | 21 + ...ulted-through-an-old-backing-is-nobodys.md | 42 -- ...t-on-data-keeps-the-deleted-files-bytes.md | 4 +- ...delete-has-already-thrown-the-file-away.md | 64 +-- ...ters-before-the-entry-stops-naming-them.md | 4 +- ...displaces-a-name-and-leaves-its-binding.md | 44 -- ...a-directory-whose-parent-was-never-made.md | 26 + .../first-match-selection-that-remains.md | 7 +- ...es-usb-read-is-answered-from-fsds-cache.md | 19 + .../isolation/probe-mounts-on-a-checksum.md | 10 +- ...es-refusals-are-narrower-than-its-reach.md | 25 +- .../untrusted-sites-not-yet-adopted.md | 75 --- ...king-refused-is-reported-as-a-segfault.md} | 15 +- ...a-read-refused-on-budget-is-not-retried.md | 32 -- .../every-wait-in-this-kernel-is-a-spin.md | 10 +- ...swers-wouldblock-and-nothing-retries-it.md | 62 --- ...back-selftest-create-answers-wouldblock.md | 22 - ...volatile-composites-on-mmio-dma-structs.md | 10 +- kernel/src/actuator.rs | 3 - kernel/src/object/ops.rs | 8 + kernel/src/tmpfs.rs | 3 +- kernel/src/vfs.rs | 25 - tests/blockdcase/system.toml | 31 +- tests/common/blockd.rs | 43 +- tests/common/volumes.rs | 472 ++---------------- tests/test-durations | 2 - .../src/bin/abuse_thread_name.rs | 8 +- tests/toyos-rust-tests/src/bin/blockd_io.rs | 49 +- tests/toyos-rust-tests/src/bin/esp_files.rs | 21 +- tests/toyos-rust-tests/src/bin/fs_escape.rs | 17 +- .../src/bin/ftruncate_flush_race.rs | 59 --- tests/toyos.rs | 23 +- toyos-abi/Cargo.toml | 2 +- toyos/Cargo.toml | 4 +- userland/fsd/src/data.rs | 4 +- userland/fsd/src/fat.rs | 5 +- userland/init/src/main.rs | 14 +- userland/metalprobe/src/usb.rs | 7 +- userland/toyos-window/Cargo.toml | 6 +- 47 files changed, 302 insertions(+), 1157 deletions(-) delete mode 100644 issues/build/ftruncate-flush-race-reds-intermittently-and-nothing-says-why.md create mode 100644 issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md delete mode 100644 issues/filesystem/a-directory-on-data-survives-no-reboot.md create mode 100644 issues/filesystem/a-link-on-a-kernel-mount-to-a-served-path-leads-nowhere.md delete mode 100644 issues/filesystem/a-page-faulted-through-an-old-backing-is-nobodys.md delete mode 100644 issues/filesystem/a-symlink-displaces-a-name-and-leaves-its-binding.md create mode 100644 issues/filesystem/data-takes-a-directory-whose-parent-was-never-made.md create mode 100644 issues/hardware/metalprobes-usb-read-is-answered-from-fsds-cache.md delete mode 100644 issues/isolation/untrusted-sites-not-yet-adopted.md rename issues/kernel/{a-page-in-the-device-refused-is-reported-as-a-segfault.md => a-page-in-its-backing-refused-is-reported-as-a-segfault.md} (72%) delete mode 100644 issues/kernel/a-root-metadata-read-refused-on-budget-is-not-retried.md delete mode 100644 issues/kernel/ftruncate-answers-wouldblock-and-nothing-retries-it.md delete mode 100644 issues/kernel/leak-rollback-selftest-create-answers-wouldblock.md delete mode 100644 tests/toyos-rust-tests/src/bin/ftruncate_flush_race.rs diff --git a/issues/README.md b/issues/README.md index 3954ced5c9b..586302c7ad3 100644 --- a/issues/README.md +++ b/issues/README.md @@ -143,17 +143,16 @@ under `toyos-abi/src`, `toyos/src` or a published crate changes no identity ## Two area notes, carried over from the file this replaced **`filesystem`** — `toyos-fat32/` is new (host tests: `cargo test` inside it) and -its kernel adapter is `kernel/src/fat32_adapter.rs`; `boot-media` carries what -that adapter found. Most of what is filed here is not a defect found later but a +`userland/fsd/src/fat.rs` serves it; `boot-media` carries what the kernel's +adapter it replaced found. Most of what is filed here is not a defect found later but a residual the crate's own gate identified while it was being written, recorded so the adapter's author did not have to rediscover it. -**`boot-media`** — `/boot` and `/log` are both `kernel/src/fat32_adapter.rs` over -`toyos-fat32`, mounted from `gpt::boot_volume()` and `gpt::log_volume()`; +**`boot-media`** — `/boot` and `/log` are both served by `/system/bin/fsd` over +`toyos-fat32`, off the partitions the loader names by unique GUID; the kernel writes no log file — `/system/bin/logd` does, an ordinary user process that owns "every policy about files — where they go, what they are called, how many there are, what happens when the stick stops answering" (`userland/logd/src/main.rs:1-10`). Gated by `esp_filesystem`, -`kernel_log_file`, `log_backing_read_error`, -`boot_volume_metadata_error`, `log_partition_layout`, `log_partition_identity` +`kernel_log_file`, `log_partition_layout`, `log_partition_identity` and `wall_clock_file`, plus `toybox_cp_volume`. diff --git a/issues/audio/disk-wait-pins-a-cpu.md b/issues/audio/disk-wait-pins-a-cpu.md index 12259691990..3f2d013d225 100644 --- a/issues/audio/disk-wait-pins-a-cpu.md +++ b/issues/audio/disk-wait-pins-a-cpu.md @@ -6,6 +6,12 @@ opened: 2026-08-08 # The audio pops are four spinlocks, not one spinning driver +**Measured before the file servers.** `log_file::SINK`, `vfs::VFS` and +`fat32_adapter::VOLUMES` are no longer on the path: logd's writes reach the stick +through fsd and a USB partition claim, whose transfer takes the disk's +block-layer lock and `xhci::XHCI`. +Whether that removes the pops is not measured. + Measured 2026-08-08 on `wt/toyos-asyncusb` at `87835d1`: at the moment a disk transfer is waited for, an ordinary guest is **four ticket spinlocks deep** — `log_file::SINK`, `vfs::VFS`, `fat32_adapter::VOLUMES` and `xhci::XHCI` — and diff --git a/issues/build/ftruncate-flush-race-reds-intermittently-and-nothing-says-why.md b/issues/build/ftruncate-flush-race-reds-intermittently-and-nothing-says-why.md deleted file mode 100644 index a4748478b38..00000000000 --- a/issues/build/ftruncate-flush-race-reds-intermittently-and-nothing-says-why.md +++ /dev/null @@ -1,45 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-09-01 ---- - -# `ftruncate_flush_race` reds intermittently, and no instrument says which of two opposite things happened - -The test races a truncate against a flush the `ftruncate-flush-stall` actuator -holds open for 400 ms, and panics with "the resize does not serialise with the -metadata window" when no attempt waited 150 ms. It reds on the dev host without -that being true, and nothing in the capture separates the two cases. - -Measured on the dev host (14 cores), one session, against a branch carrying the -shrink-mark work and against the same tree with that kernel change reverted: - -| arm | condition | runs | result | -|---|---|---|---| -| branch | alone | 6 | 6 pass; waits 355.1 / 356.2 / 357.5 / 358.2 / 358.6 / 369.4 ms | -| kernel change reverted | alone | 3 | 1 fail, then 2 pass (attempt 0, 358.0 and 356.3 ms) | -| branch | full nightly tier | 3 | 2 fail, 1 pass | -| kernel change reverted | full nightly tier | 1 | pass | - -Both arms fail and both pass, and the only same-configuration comparison with -more than one sample per side is the *alone* one, where the branch is 6 for 6 -and the reverted arm failed once. So the numbers say intermittent; they do not -name a cause and they do not point at a diff. - -**No mechanism is established, and one plausible reading is already refuted.** -`STALLED WINDOW HELD` is not evidence that the truncate arrived after the -window: `SYS_FTRUNCATE` and `SYS_FSYNC` take the same VFS lock and the stall -spins inside it, so a serialised truncate can never land in the window and the -line prints on every attempt, passing or failing. `tests/common/volumes.rs` -requires `HELD` for a *pass* and refuses `BROKEN`, which is the same statement -from the other side. - -What the capture does not hold is when `set_len` entered the kernel relative to -the stall's start. Without it a short `waited` is either a resize that did not -serialise — the defect this gate exists for — or a stimulus that arrived after -the 400 ms window closed, which is a retry; and the guest's own docstring says -the loop is one-sided precisely because a sleep can overshoot. Whoever takes -this decides what to measure; this entry is the rate and the refutation, not a -design. - -`cargo run -- --known-red ftruncate_flush_race` says `NOT ON THE LIST`. diff --git a/issues/design-debt/deviceerror-classify-trusts-a-lying-transfererror.md b/issues/design-debt/deviceerror-classify-trusts-a-lying-transfererror.md index c02e9c1cbb2..71055f2692e 100644 --- a/issues/design-debt/deviceerror-classify-trusts-a-lying-transfererror.md +++ b/issues/design-debt/deviceerror-classify-trusts-a-lying-transfererror.md @@ -52,12 +52,13 @@ error[E0277]: the trait bound `OnBudget: bcachefs::block_io::sealed::Sealed` is --> bcachefs/tests/integration.rs:1009:34 ``` -and by the same rule would refuse `kernel/src/bcachefs_adapter.rs:24` (the -kernel's `block::BlockError`) and `tests/common/storage.rs:267` (the host -harness's file-backed device). Implementing `TransferError` from outside is -what the trait is *for*: a foreign block device is the only authority on -whether its own transfer was attempted. Making `classify` crate-private has -the identical problem, because those three are the callers. +and by the same rule would refuse `kernel/src/block.rs`'s `impl` for the +kernel's `BlockError`, fsd's for its `DiskError` (`userland/fsd/src/cache.rs`) +and the host harness's file-backed device (`tests/common/storage.rs`). +Implementing `TransferError` from outside is what the trait is *for*: a +foreign block device is the only authority on whether its own transfer was +attempted. Making `classify` crate-private has the identical problem, because +those are the callers. ## What would close it diff --git a/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md b/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md new file mode 100644 index 00000000000..823dd3cbd5c --- /dev/null +++ b/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md @@ -0,0 +1,32 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A cached read from a file server costs fifteen to twenty times the kernel's + +Measured on QEMU TCG (`Profile::Metal`, DATA on NVMe), 16 MiB on `/home`, +three interleaved runs per arm, the branch that moved DATA to +`/system/bin/fsd` against main at `16d2e645`: + +| | main (kernel) | fsd | +|---|---|---| +| write 256 KiB at a time, one fsync | 37–64 MiB/s | 31 MiB/s | +| read back at once, 256 KiB at a time | 1164–1642 MiB/s | 80 MiB/s | +| read on the next boot | 84 MiB/s | 41–42 MiB/s | +| 4 KiB create and fsync, p50 | 1.0–1.3 ms | 3.5–3.6 ms | + +The cached read by request size, one run: 47.5 MiB/s at 8 KiB, 91.8 MiB/s at +256 KiB, 182.6 MiB/s at 2 MiB — so about 160 µs a request and about 5 ns a +byte. A byte is copied three times, after a zeroing: out of the block cache +into a buffer the server zeroed (the `READ` arm of `userland/fsd/src/main.rs`, +through `DataVolume::read`), into the client's window a `u64` at a time +with a volatile store each (`toyos::fs::window_put`, +`toyos::volatile::Window::copy_in`), and out of the window a `u64` at a time +again (`window_take`), where the kernel's page cache copies once. + +**Exit**: a cached read within a small factor of the kernel's at the same +request size — the block copied once into the window, and the window's word +loops replaced by a copy the compiler may widen — measured by the same +three-run A/B. diff --git a/issues/filesystem/a-corrupt-chain-met-by-a-queued-free-is-answered-by-the-next-call.md b/issues/filesystem/a-corrupt-chain-met-by-a-queued-free-is-answered-by-the-next-call.md index 1230f2af80c..881d72076df 100644 --- a/issues/filesystem/a-corrupt-chain-met-by-a-queued-free-is-answered-by-the-next-call.md +++ b/issues/filesystem/a-corrupt-chain-met-by-a-queued-free-is-answered-by-the-next-call.md @@ -11,9 +11,9 @@ call starts (`settle_first`, `toyos-fat32/src/repair.rs`). A queued `Repair::Free` walks the rest of a chain whose name is already gone, and a walk that meets a cyclic or out-of-range link answers `Error::CorruptChain`. That error is returned by whichever call came next, for a chain it never -named, and the call did not start. The kernel adapter logs it as that call -failing (`refused` in `kernel/src/fat32_adapter.rs`: "`` of ``: -corrupt cluster chain") and its caller reads `SyscallError::Io`. +named, and the call did not start. fsd logs it as that call failing +(`logged` in `userland/fsd/src/fat.rs`: "`` of '``': corrupt cluster +chain") and its client reads `SyscallError::Io`. Two things are lost: diff --git a/issues/filesystem/a-device-refusal-that-queues-the-calls-own-repair-is-named-by-which-error-it-was.md b/issues/filesystem/a-device-refusal-that-queues-the-calls-own-repair-is-named-by-which-error-it-was.md index dc8436a936e..df1ec1fb567 100644 --- a/issues/filesystem/a-device-refusal-that-queues-the-calls-own-repair-is-named-by-which-error-it-was.md +++ b/issues/filesystem/a-device-refusal-that-queues-the-calls-own-repair-is-named-by-which-error-it-was.md @@ -11,15 +11,12 @@ opened: 2026-09-26 `Error::BudgetExpired` — even when the `BudgetExpired` is the call's own write and that same write is what queued the repair (`toyos-fat32/tests/refused_writes.rs:729`, asserted directly: "its own -refusal, not a wait"). The kernel adapter's `name_pending` -(`kernel/src/fat32_adapter.rs`) uses `waits_on` to choose the log line: an -`Io` in that state logs as "`` of `` refused, a repair pending with -`N` step(s) queued", while a `BudgetExpired` reaching the volume in the same -state — a repair now queued by this call's own write — logs as the call's own -failure, "`` of ``: budget expired". Two device refusals that leave -the volume in the same state (a repair queued, to be re-driven before the -next mutating call) are named two different ways depending only on which of -the two errors the device answered. +refusal, not a wait"). Two device refusals that leave the volume in the same +state (a repair queued, to be re-driven before the next mutating call) are +named two different ways depending only on which of the two errors the device +answered. No caller reads the difference today: the kernel adapter that chose +its log line by `waits_on` is gone, and fsd's disks refuse on no clock, so it +never meets a `BudgetExpired` — the next caller that does inherits it. ## Owner diff --git a/issues/filesystem/a-directory-on-data-survives-no-reboot.md b/issues/filesystem/a-directory-on-data-survives-no-reboot.md deleted file mode 100644 index 4fa2928e343..00000000000 --- a/issues/filesystem/a-directory-on-data-survives-no-reboot.md +++ /dev/null @@ -1,33 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-26 ---- - -# A directory on DATA survives no reboot, and takes a child whose parent was never made - -The DATA adapter answers every `create_dir` with `NotSupported` -(`kernel/src/bcachefs_adapter.rs`), so the VFS keeps the directory itself in -`created_dirs` (`kernel/src/vfs.rs`). Two things follow: - -- **A child is taken whose parent was never made.** `create_dir("/home/toy/Apps")` - succeeds with no `/home/toy`, so `std::fs::create_dir_all` makes the leaf and - nothing above it. init makes every level in turn for this reason (`make_dir` - in `userland/init/src/main.rs`). -- **It is memory.** Nothing reaches the volume, so an empty directory is gone - at the next boot; init remakes the session home and each service's `/state` - every boot. - -Seen building `layout_fresh_boot`: `read_dir /home/toy: entity not found` from -a boot whose init had just made `/home/toy/Apps`. - -## Owner - -The storage track, `issues/filesystem/storage-is-layers-and-a-role-is-a-filesystem.md`: -real bcachefs under DATA carries directories. - -## What would close it - -DATA stores directories, `create_dir` refuses a missing parent, and -`created_dirs` is gone or kept only for a mount that has no directories and -says so. diff --git a/issues/filesystem/a-link-on-a-kernel-mount-to-a-served-path-leads-nowhere.md b/issues/filesystem/a-link-on-a-kernel-mount-to-a-served-path-leads-nowhere.md new file mode 100644 index 00000000000..6b196019f93 --- /dev/null +++ b/issues/filesystem/a-link-on-a-kernel-mount-to-a-served-path-leads-nowhere.md @@ -0,0 +1,21 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A link on a kernel mount to a served path leads nowhere + +The kernel resolves a symlink on its own mounts — ROOT and `/tmp` — in its own +tree (`kernel/src/vfs.rs`), and its tree holds nothing under `/apps`, +`/config`, `/home`, `/state`, `/log` or `/boot` but ROOT's empty directories: +those are file servers', reached through a directory capability +(`rust/library/std/src/sys/fs/toyos.rs`). So a link in `/tmp` whose target is +`/home/toy/notes` opens nothing (`NotFound`), where the same link on a served +directory is followed — a file server hands an absolute target back to the +client, which resolves it in its own table. + +**Exit**: the kernel answers a path that crosses a link into a directory it +does not serve with the target and the rest of the path, as a file server's +`LINK` reply does, and std resolves it again in the caller's table; with a +guest test that opens a `/home` file through a `/tmp` link. diff --git a/issues/filesystem/a-page-faulted-through-an-old-backing-is-nobodys.md b/issues/filesystem/a-page-faulted-through-an-old-backing-is-nobodys.md deleted file mode 100644 index 32d4969376b..00000000000 --- a/issues/filesystem/a-page-faulted-through-an-old-backing-is-nobodys.md +++ /dev/null @@ -1,42 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-01 ---- - -# A page faulted through a backing taken earlier answers from that backing and not from the file cache - -`kernel/src/process.rs`'s demand-paging fault reads `FileBacking::read_page` -directly; `kernel/src/loader/mod.rs` and `kernel/src/elf/mod.rs` do the same -for a spawn's segments and headers. None of them goes through -`file_cache::read_page`, so none of them sees what the file cache knows about -the file: not the pages a writer has dirtied, and not `CachedFile::shrunk_to`, -the mark that makes a shrunk tail read as zeros. - -So an mmap or a spawn whose backing was derived before a write or a shrink -keeps answering from the extents that backing holds. The `read`/`write` -syscalls and the fault path are two readers of one file that can disagree, and -which one a caller gets is decided by how it opened the bytes. - -Two of the three windows that produced this are closed at their own sites: -`Vfs::open_backing` now settles the write-back queue *and* flushes a file the -cache says owes one, so a backing is derived from a file that is on the device. -What is left is the backing that was already handed out. Naming what does cover -part of that, so the gap is the real one: a backing whose blocks are being -reused or freed *is* revoked — `FatFs::revoke` from `delete` and from a -rename's displaced destination (`kernel/src/fat32_adapter.rs:787`, `:684`), and -`BcacheFsAdapter::revoke` from `create`, `delete`, `create_symlink` and rename -(`kernel/src/bcachefs_adapter.rs:269`, `:295`, `:365`, `:209`). A shrink that -reached the device is re-derived too: `FatFs::truncate_to` and -`update_metadata` call `FatExtents::truncate_to`, and the bcachefs adapter's -`truncate_to` shortens its own block cell. Every one of those is keyed to the -mount's blocks changing hands. None of them is keyed to the file cache holding -something the device does not, which is the whole of what is left: a page a -writer dirtied and no flush has carried, and a `shrunk_to` mark before the -flush that trims it. - -The end of it is one authority for a file's bytes: a `FileBacking` that reads -through the file cache from the fault path, and the pinning that makes that -sound. It is not a small change — the fault path takes no VFS lock today and -the cache's miss path drops its own — which is why the window at -`Vfs::open_backing` was closed where it stood instead. diff --git a/issues/filesystem/a-reallocated-extent-on-data-keeps-the-deleted-files-bytes.md b/issues/filesystem/a-reallocated-extent-on-data-keeps-the-deleted-files-bytes.md index 4ee98af6338..13d6b931ed7 100644 --- a/issues/filesystem/a-reallocated-extent-on-data-keeps-the-deleted-files-bytes.md +++ b/issues/filesystem/a-reallocated-extent-on-data-keeps-the-deleted-files-bytes.md @@ -39,9 +39,7 @@ byte. ## What it is not -Not -`issues/filesystem/a-page-faulted-through-an-old-backing-is-nobodys.md` either: -no mapping is taken across the write here. +Not a mapping taken across the write: none is taken here. Which half is wrong is not established. Either the allocator handed out a block whose old contents were never overwritten and the new file's first extent was diff --git a/issues/filesystem/a-refused-delete-has-already-thrown-the-file-away.md b/issues/filesystem/a-refused-delete-has-already-thrown-the-file-away.md index aa13be377fb..a9db8045216 100644 --- a/issues/filesystem/a-refused-delete-has-already-thrown-the-file-away.md +++ b/issues/filesystem/a-refused-delete-has-already-thrown-the-file-away.md @@ -1,59 +1,19 @@ --- status: open -kind: defect +kind: tooling opened: 2026-09-05 --- -# A `/home` delete the device refuses has already discarded the file it did not delete +# No test stages a delete the device refuses -`BcacheFsAdapter::delete` (`kernel/src/bcachefs_adapter.rs:299`) destroys every -piece of in-memory state the file has *before* it asks the device: +A delete that fails on the device has to leave the file as it was: its name, +its blocks, and every holder's view of it. `userland/fsd` asks the volume +first and only then marks the open file gone (`DataVolume::unlink` in +`userland/fsd/src/data.rs`, `FatVolume::unlink` in `userland/fsd/src/fat.rs`), +which is the order that keeps it; the kernel adapter this file was filed +against discarded the cached file, the name and the blocks before it asked. +Nothing stages the refusal, so the order is read off the code and not run. - file_cache::mark_deleted(file_id) // sets `deleted`, kernel/src/file_cache.rs:631 - self.name_to_id.remove(name) - self.revoke(name) // the shared `FileBlocks` cell -> None - mapped("delete", name, self.fs.delete(name))? // :307 — the `?` is the loss - -The device call is the only step that can fail, and it is last. A read the -btree descent makes can be refused on the caller's own budget — -`FsError::DeviceRead` carrying `DeviceError::Refused`, which -`as_device_refusal` (`:72`) maps to `SyscallError::WouldBlock` — so `SYS_UNLINK` -returns an error meaning *nothing happened*, on a call after which four things -have happened: - -- the cached file is marked `deleted`, and `writeback.rs:111` skips a deleted - file's flush, so every page a writer dirtied and no flush has carried is - dropped when `finish_writeback` frees it; -- the blocks cell is revoked, so every backing for that name fails from here on; -- the name is unbound, so the next `open` reads the device; -- and the on-disk entry is still there when the refusal came before - `btree::delete` (`bcachefs/src/fs.rs:854`'s `delete_by_name` removes the entry - and only then frees the extents at `:865`). - -So the re-open serves the file's *pre-delete* contents and the unflushed writes -are gone, out of a syscall that said it did nothing. - -## Reproduction - -Not run. The path is read off the code above; no arm in the tree stages a -refusal on this call. The nearest instrument is the `fsync-budget-spent` -actuator, which stages a spent budget for `SYS_FSYNC` and not for a delete, so -staging this needs an actuator of its own on `BcacheFsAdapter::delete`'s device -call. - -## Not the retry record - -`issues/kernel/ftruncate-answers-wouldblock-and-nothing-retries-it.md` names the -same trigger and a different mechanism: it is about a refusal reaching userland -as `WouldBlock` with no retry ladder behind it, and states of its own subject -that a refused resize "changes nothing", which is what makes it a spurious -failure rather than a loss. This one is a loss, and a retry ladder alone would -not close it — the ordering would still discard the file on the attempt that -gave up. - -## Exit condition - -`delete` asks the device first and touches the cache, the name and the blocks -only on its success — or the refusal path restores all four — with an actuator -staging the refusal and an arm that re-opens the name and finds the file it had -before, its unflushed writes included. +**Exit**: a host test in `userland/fsd` whose disk refuses the reads or writes a +delete makes, and finds the name, the bytes and an open holder's reads as they +were before the refused call. diff --git a/issues/filesystem/a-shrink-frees-clusters-before-the-entry-stops-naming-them.md b/issues/filesystem/a-shrink-frees-clusters-before-the-entry-stops-naming-them.md index 5f1aebb0407..f2d369764ee 100644 --- a/issues/filesystem/a-shrink-frees-clusters-before-the-entry-stops-naming-them.md +++ b/issues/filesystem/a-shrink-frees-clusters-before-the-entry-stops-naming-them.md @@ -16,8 +16,8 @@ during a `set_len` then `flush_meta`: `DIR_FileSize is 2560 bytes, which needs 5 clusters, and the chain holds 2`. It needs no refusal: a machine that stops between a successful shrink and its flush leaves the same volume. -The kernel's `truncate_to` and `update_metadata` (`kernel/src/fat32_adapter.rs`) -are the two callers that shrink. +fsd's `FatVolume::truncate` and `FatVolume::level` (`userland/fsd/src/fat.rs`) +are the two callers: the first shrinks, the second writes the entry. ## Exit condition diff --git a/issues/filesystem/a-symlink-displaces-a-name-and-leaves-its-binding.md b/issues/filesystem/a-symlink-displaces-a-name-and-leaves-its-binding.md deleted file mode 100644 index 493369d3d7b..00000000000 --- a/issues/filesystem/a-symlink-displaces-a-name-and-leaves-its-binding.md +++ /dev/null @@ -1,44 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-05 ---- - -# Creating a symlink over a `/home` file leaves the file bound to the name - -`BcacheFsAdapter::create_symlink` (`kernel/src/bcachefs_adapter.rs:374`) revokes -the name's blocks and then writes the link: - - self.revoke(name); - mapped("create_symlink", name, self.fs.create_symlink(name, target)) - -`Mounted::create_symlink` (`bcachefs/src/fs.rs:765`) is `put`, which the crate -documents at `:769` as "displacing whatever answered to it" — so the file that -answered to `name` is gone from the volume. Its `name_to_id` binding is not: -`create_symlink` is the one path in this adapter that writes over a name on the -volume while a live binding for it stands. `create` (`:271`) returns the bound -id without touching the volume at all, so it cannot displace one; `delete` -(`:299`) unbinds before it asks the device; `ReplaceRename::release` (`:203`) -unbinds the displaced destination. - -Nothing reads it back through that binding today, because `open` of a path whose -last component is now a symlink is resolved by `Vfs::resolve_for_open_depth` -(`kernel/src/vfs.rs:413`) into the link's target before `open_file` is reached. -That is a property of the resolver, not of this adapter: the invariant the map -is supposed to hold — a name is bound to at most one live file, and the file it -is bound to is the one the volume answers with — is false here, and the guard at -`close_file` (`:290`) that keeps a stale teardown from unbinding a live file -does not make it true, because nothing unbinds the displaced file at all. - -## Reproduction - -Not run, and there is no known guest-visible symptom to run for; the record is -the broken invariant and its one asymmetry against the four sibling paths. - -## Exit condition - -`create_symlink` unbinds the displaced name the way `create` and `delete` do, -with an arm that creates a symlink over an open `/home` file and asserts the -adapter answers for the link and not for the file — or a stated reason at the -site why this path alone may leave the binding, which the resolver's behaviour -is not, because it is not this module's to promise. diff --git a/issues/filesystem/data-takes-a-directory-whose-parent-was-never-made.md b/issues/filesystem/data-takes-a-directory-whose-parent-was-never-made.md new file mode 100644 index 00000000000..52763d86e45 --- /dev/null +++ b/issues/filesystem/data-takes-a-directory-whose-parent-was-never-made.md @@ -0,0 +1,26 @@ +--- +status: open +kind: defect +opened: 2026-09-26 +--- + +# DATA takes a directory whose parent was never made + +`userland/fsd/src/data.rs` keeps DATA's directories on the volume as marker +entries, so a directory survives a reboot. What it does not do is refuse one +whose parent is missing: `DataVolume::require_parent_dir` refuses a parent that +is a file or a symlink and nothing else, because the interim format makes +every prefix of a name a directory. So `create_dir("/home/toy/Apps")` succeeds +with no `/home/toy`, and `std::fs::create_dir_all` makes the leaf and nothing +above it — init makes every level in turn for this reason (`make_dir` in +`userland/init/src/main.rs`). + +## Owner + +The storage track, `issues/filesystem/storage-is-layers-and-a-role-is-a-filesystem.md`: +real bcachefs under DATA carries directories. + +## What would close it + +DATA's `mkdir`, a create and a rename refuse a missing parent, with a host test +in `userland/fsd` that asks each for one. diff --git a/issues/hardware/first-match-selection-that-remains.md b/issues/hardware/first-match-selection-that-remains.md index ad62c7e02dd..e6e33da4b1c 100644 --- a/issues/hardware/first-match-selection-that-remains.md +++ b/issues/hardware/first-match-selection-that-remains.md @@ -9,10 +9,9 @@ opened: 2026-08-01 `pci::enumerate` returns every function now, so a driver taking the first match does so visibly. Two do, and both are deliberate: -- **NVMe.** `nvme::init` takes the first class-0108 controller. A machine with - two NVMe drives loses the second, and there is nowhere to put it: - `page_cache::init` takes a single `Box`. Making this an - enumerate-all is a storage-stack change, not a PCI one. +- **NVMe.** blockd serves one controller, the first its row's claim names + (`userland/blockd`); a machine with two NVMe drives loses the second. + Making this an enumerate-all is a block-service change, not a PCI one. - **The four virtio drivers.** Each takes the first device with its (vendor, device) pair. A second NIC or a second GPU would be dropped. These are QEMU-only devices — no virtio function appears on the T14 — so the diff --git a/issues/hardware/metalprobes-usb-read-is-answered-from-fsds-cache.md b/issues/hardware/metalprobes-usb-read-is-answered-from-fsds-cache.md new file mode 100644 index 00000000000..ff88f3818e9 --- /dev/null +++ b/issues/hardware/metalprobes-usb-read-is-answered-from-fsds-cache.md @@ -0,0 +1,19 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# metalprobe's USB read is answered from fsd's cache, not the stick + +`userland/metalprobe/src/usb.rs`'s `read` writes its file to `/log`, closes it, +and times reading it back as "a cache miss the stick has to answer" — true +while the kernel's write-back dropped a closed file from its cache. `/log` is +served by `/system/bin/fsd` now, whose block cache (`userland/fsd/src/cache.rs`) +keeps clean blocks until it holds `CLEAN_LIMIT` of them and drops nothing at a +close. So the timed read is fsd's memory, and the `usbread` span +`tests/metal-profile.toml` prices is no longer the stick's. + +**Exit**: the timed read is one the stick answers — a file fsd has not read +since it started, or a read through the partition claim that bypasses the +file server — with the profile row re-measured on the T14. diff --git a/issues/isolation/probe-mounts-on-a-checksum.md b/issues/isolation/probe-mounts-on-a-checksum.md index 681a3ae3759..1362003b387 100644 --- a/issues/isolation/probe-mounts-on-a-checksum.md +++ b/issues/isolation/probe-mounts-on-a-checksum.md @@ -6,9 +6,11 @@ opened: 2026-08-01 # `probe()` mounts on a checksum, and a stamp over a used volume does not reformat -Two things, from reading `bcachefs_adapter::probe` against the crate: +Two things, from reading DATA's probe (`DataVolume::probe` in +`userland/fsd/src/data.rs`, the kernel's `bcachefs_adapter::probe` before it) +against the crate: -**The threshold does not match the consequence.** `Storage::Ours` is a +**The threshold does not match the consequence.** `Probed::Mounted` is a read-write mount: `sync()` rewrites both superblocks, and any file operation writes the bitmap, btree nodes and data. So mounting a stranger's disk modifies it, which is a weaker form of the wrong the designation stamp exists to prevent. @@ -32,8 +34,8 @@ Recommendation, for the owner to decide: longer has an unchecked extent reaching a block read — or a bitmap write — behind it. The residual bound this left — `read_link` sizing one kernel allocation from the volume rather than the smaller heap ceiling — was closed - by giving `Mounted::read_link` a `max_len` the adapter passes as - `MAX_LINK_TARGET`, refused as `FsError::TargetTooLong` before the allocation. + by giving `Mounted::read_link` a `max_len` its caller passes (fsd's + `MAX_LINK`), refused as `FsError::TargetTooLong` before the allocation. - **The real fix, if the threat model wants one:** read-write requires something the attacker cannot compute — a keyed MAC, or a designation-like stamp — and everything else mounts read-only. ToyOS has no key store and no diff --git a/issues/isolation/the-so-caches-refusals-are-narrower-than-its-reach.md b/issues/isolation/the-so-caches-refusals-are-narrower-than-its-reach.md index b19b428e9e9..4f09c447426 100644 --- a/issues/isolation/the-so-caches-refusals-are-narrower-than-its-reach.md +++ b/issues/isolation/the-so-caches-refusals-are-narrower-than-its-reach.md @@ -8,33 +8,10 @@ opened: 2026-09-03 `kernel/src/elf/cache.rs` now refuses a path whose file changed and a load past its byte budget, which closed the record that tracked the cache's lack of both. -Three things that record carried are not covered by either refusal, and they +Two things that record carried are not covered by either refusal, and they went out of the tracker with it. They are one file because they are one residual: what the refusals do **not** reach. -## The identity cannot see a same-size rewrite on a FAT32 mount - -`vfs::BackingId` is size plus the mount's mtime. On `/home` (bcachefs) the mtime -is `nanos_since_boot` at the flush, so two writes are always apart — -`so_cache_policy`'s `stale-mtime` arm asserts exactly that. - -`/log` is FAT32, mounted `UserAccess::ReadWrite` (`kernel/src/main.rs:461`), and -**FAT stores seconds in units of two**: `toyos-fat32/src/time.rs:5-15` states -the encoding's three lossy properties, `dir.rs:92` passes 0 for the tenths -field, and `kernel/src/fat32_adapter.rs:849` stamps whatever `now()` gives. So -two writes of the same length inside one 2-second bucket carry one mtime, and -the second load is served the first image — the staleness the refusal exists to -prevent, on the one writable FAT mount a process can reach. - -**Mechanism read off the code; not reproduced.** Planting it needs a same-size -rewrite of a library inside 2 s on `/log`, and a 1.9 MB write to the -USB-backed log volume takes about 5.8 s, so the window closes before the second -write lands. - -*Exit condition:* an identity that does not rest on a clock — a content hash, a -per-file generation the mount bumps on every write, or a `FileId` plus a write -counter — or a demonstration that no library can be reached on a FAT mount. - ## The budget is a machine-wide, boot-permanent denial 256 MiB over every cached image, and nothing is ever evicted, so the bytes an diff --git a/issues/isolation/untrusted-sites-not-yet-adopted.md b/issues/isolation/untrusted-sites-not-yet-adopted.md deleted file mode 100644 index b5ff162e4f3..00000000000 --- a/issues/isolation/untrusted-sites-not-yet-adopted.md +++ /dev/null @@ -1,75 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-08-18 ---- - -# The sites that still carry a boundary-crossing number in a plain integer - -`toyos-untrusted`'s `Untrusted` exists and the virtqueue used ring uses it: -`Virtqueue::used_ring_id` and `::used_ring_len` (`kernel/src/drivers/virtio.rs`) -are the only reads of a used ring in the kernel and hand back `Untrusted`, -so `Virtqueue::parse_used` is the only path to an index or a length and it names -its bound at both. The type has no arithmetic, no `Deref`, no `From`, no cast -and no accessor; the four `compile_fail` doctests on `Untrusted` are what that -sentence means, and `src/sourcegate.rs` bans the one shape typing cannot stop -(`at_most(u64::MAX)` and its four siblings). - -**The rest of the class has not adopted it.** Until a site does, its bound is a -thing an author has to remember rather than a thing the compiler asks for. Each -of these is mechanical — find the read, wrap it there, name the bound at the use -— and none needs a design decision. **Each is cited by file and enclosing -symbol**, so whoever takes this does not have to re-derive which value is meant. - -## The sites, hardest consequence first - -- **`kernel/src/drivers/nvme.rs`, the completion queue — converted.** - `wait_completion` now takes the `cid` this driver put in the command - (`(cmd.cdw0 >> 16) as u16`, read back out of the command it just submitted, - not trusted again) and answers `Result`: - `Untrusted::new(cq.cid).exactly(expected)`. `submit_and_wait` and all four - callers (`admin`, `identify_namespace` through it, `read_sectors`, - `write_sectors`) now match on the result and log-and-fail exactly as they - already did for a bad status word. `Untrusted::exactly`'s own behaviour is - covered by `toyos-untrusted`'s host test suite - (`exactly_refuses_by_naming_both_numbers`). - **No dedicated boot-level gate** — unlike the virtqueue used ring, this - queue has no actuator today, and building one (`#[cfg(feature = - "boot-actuators")]` writer + self-test + `actuator.rs` entry + - `MACHINE_TESTS` registration) means a `tests/test-durations` entry, which - this session watched block PR #118's own merge on exactly that mechanism - (`ci/durations`: "committed UNMEASURED profile marker(s) are provisional and - may not land"). Given the issue's own framing — "sound today only because - every submission is synchronous, a property of the caller" — this is - hardening against a future asynchronous queue rather than a live defect, so - the landing risk was judged not worth it this pass. An - actuator + self-test mirroring `drivers/virtio.rs`'s `used_selftest` is the - future work if the owner wants it exercised on every boot. - -## What this type does not answer - -Recorded so nobody tries to make it. One entry in this area was **not** this -class, and wrapping something would have touched it: - -- **netd trusting the ring's closed flags** — a *predicate* a peer writes, not - an index. `RingHeader::flags` lives in the page `SYS_PIPE_MAP` maps writable, - and netd read `is_reader_closed`/`is_writer_closed` as facts about its peer. - There was no bound to compare a bit against, so the answer was not a wrapper - but to stop reading a publication as a channel: netd now asks the kernel's own - `readers`/`writers` counts, which surface as EOF on a read and `BrokenPipe` on - a write. Closed on its own file. - -A second entry used to stand beside it: virtio-net registering its whole -`DmaPool` page, so that the claimant mapped the descriptor tables writable. That -was a *mapping* rather than a value, no read-side type helped while it stood, -and it was closed on 2026-08-23 by splitting the pool — which is why this -section names one entry and not two. - -## What this type is not about - -It carries no *time* and no *rate*, and nothing in it reasons about how long -anything takes. The TCG-versus-KVM p90 divergence PR #119 measured (4–8× in p90, -20× in max for scheduler pass cost on the dev host) does not reach any bound -here: every one is a byte count, a table length or a register encoding, and each -is compared against a number the driver itself published in the same function. -Recorded so that nobody re-checks it. diff --git a/issues/kernel/a-page-in-the-device-refused-is-reported-as-a-segfault.md b/issues/kernel/a-page-in-its-backing-refused-is-reported-as-a-segfault.md similarity index 72% rename from issues/kernel/a-page-in-the-device-refused-is-reported-as-a-segfault.md rename to issues/kernel/a-page-in-its-backing-refused-is-reported-as-a-segfault.md index 7b844ec4079..3090ee942c3 100644 --- a/issues/kernel/a-page-in-the-device-refused-is-reported-as-a-segfault.md +++ b/issues/kernel/a-page-in-its-backing-refused-is-reported-as-a-segfault.md @@ -4,7 +4,7 @@ kind: defect opened: 2026-09-26 --- -# A page-in the device refused is reported as a segfault +# A page-in its backing refused is reported as a segfault A demand fault on a file-backed region whose backing's `read_page` fails logs one line and leaves the fault unhandled (`kernel/src/process.rs`, the @@ -22,12 +22,13 @@ subsystem, and nothing ties the two together. ## Where it is reachable -Not for ROOT: `/system` is served from the image the loader put in memory, -and `ReadOnlyBacking::read_page` (`kernel/src/file_backing.rs`) fails only on -a block outside that image, which is a corrupt image rather than a device. -Every other file a process maps or runs is backed by a device: `NvmeBacking` -on DATA and the FAT backing on `/boot` and `/log`, so a program or library run -from `/home` or `/boot`, and any file mapped from them, reaches it. +No page this kernel faults in is read from a device any more: `/system` is the +image the loader put in memory, a program or library a file server holds is +copied into memory at its spawn or `dlopen` (`ImageBacking` in +`kernel/src/file_backing.rs`), and `/tmp` is memory. What still refuses a +page is a `/tmp` backing whose file was deleted (`revoke_selftest`), so a +program that maps a `/tmp` file another deletes reaches it — reported as its +own wild pointer. ## Owner diff --git a/issues/kernel/a-root-metadata-read-refused-on-budget-is-not-retried.md b/issues/kernel/a-root-metadata-read-refused-on-budget-is-not-retried.md deleted file mode 100644 index 38bb78398ee..00000000000 --- a/issues/kernel/a-root-metadata-read-refused-on-budget-is-not-retried.md +++ /dev/null @@ -1,32 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-04 ---- - -# A ROOT metadata read refused on the caller's budget is not retried - -Every file on ROOT is opened by walking a btree, and every btree block reaches -the kernel through `bcachefs_adapter::PageCacheBlockIO::read_block` — which -locks the cache, then the device, then transfers. A transfer refused on -`block::OPERATION` comes back `BlockError::BudgetExpired`, and that becomes -`DeviceError::Refused`, `FsError::DeviceRead`, `SyscallError::WouldBlock`. -Nothing retries it: a `spawn` whose `file_extents` lookup lands on a refused -btree read reports the program missing on a disk that is merely busy. - -`kernel/CLAUDE.md`'s rule is that a `BudgetExpired` is a claim about the -caller's clock and never a loss, retried on a fresh budget above every lock. -The retry cannot go where the read is: `read_block` runs under the cache lock -and the retry has to be above it. `file_backing::read_block_retrying` is the -same rule applied to the *data* path, which is where it was reached first — -twelve guests sharing one host spent the budget waiting for the xHCI -controller lock and demand-paged executables faulted at `_start+0x0`. - -**Reproduction.** Not reached in a suite yet: the data path was, and this one -shares its mechanism and its device. `usb-slow-device` holds every mass-storage -completion back, and a boot armed with it that also spawns under load is the -shape to try. - -**Exit condition.** A refused metadata read is retried on a fresh budget above -the cache lock, bounded by `block::DEADMAN`, or the mount reports it as the -retryable refusal it is and every caller of `open` acts on that word. diff --git a/issues/kernel/every-wait-in-this-kernel-is-a-spin.md b/issues/kernel/every-wait-in-this-kernel-is-a-spin.md index 511185188d1..34bb004048c 100644 --- a/issues/kernel/every-wait-in-this-kernel-is-a-spin.md +++ b/issues/kernel/every-wait-in-this-kernel-is-a-spin.md @@ -6,6 +6,12 @@ opened: 2026-08-12 # Every wait in this kernel is a spin, and a killed task dies by having its stack discarded +**The storage chains below are older than the file servers.** The kernel holds +no FAT, NVMe or page-cache lock any more: `/apps`, `/config`, `/home`, `/state`, +`/log` and `/boot` are `/system/bin/fsd`'s, over blockd or a USB partition +claim, so what is left of a disk wait in the kernel is a claim's transfer +(`block` → `xhci::XHCI`) and `vfs::VFS` over ROOT and `/tmp`. + **The heading and the paragraph under it are the state at opening, not the state of the tree.** What exists now: the completion core, where every wait in the kernel rechecks one predicate and a waiter lends a watch to the object it @@ -416,8 +422,8 @@ a file's flush. This is the state that unblocks `vfs::VFS`'s conversion (the next chunk): a `Drop` reaching this release site now touches neither a sleep lock nor a device. Negative controls: `writeback_reopen` (an `iod` stalled by `writeback-stall`, a re-open reads the pinned pages) and `writeback_durability` -(a close with no fsync reaches the `/log` volume through the drain, -`toyos-fat32-check` the oracle). Measured, and recorded in `iod.rs`'s header: a +(a close with no fsync reached the `/log` volume through the drain, which the +kernel no longer serves). Measured, and recorded in `iod.rs`'s header: a 360-file close burst on NVMe `/home` drove worst close-to-drained latency to ~72 ms, the single `iod` thread draining the backlog serially. diff --git a/issues/kernel/ftruncate-answers-wouldblock-and-nothing-retries-it.md b/issues/kernel/ftruncate-answers-wouldblock-and-nothing-retries-it.md deleted file mode 100644 index d1cd276db9b..00000000000 --- a/issues/kernel/ftruncate-answers-wouldblock-and-nothing-retries-it.md +++ /dev/null @@ -1,62 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-02 ---- - -# `SYS_FTRUNCATE` can answer `WouldBlock` and no caller in the tree retries it - -A shrink whose new end falls inside a page the cache does not hold reads that -page off the device first (`file_cache::resize`). That read can be refused on the -caller's own time budget — `block::begin_operation` is armed per transfer by -`kernel/src/drivers/nvme.rs` and `kernel/src/drivers/usb_storage.rs`, and a -refusal there is `BlockError::BudgetExpired`. `ops::ftruncate` maps it to -`SyscallError::WouldBlock`, which is the truth: nothing was resized and asking -again on a fresh budget would work. - -Nothing asks again. `userland/libc/src/posix_io.rs`'s `ftruncate` is - - match syscall::ftruncate(fd(raw_fd), length as u64) { - Ok(()) => 0, - Err(e) => set_errno(e), - } - -so the status reaches a C caller as `EAGAIN` from a call POSIX gives no -`EAGAIN`, and Rust's `File::set_len` surfaces it as `ErrorKind::WouldBlock`. The -comparison is `SYS_FSYNC`, which has the retry ladder this does not -(`ops::fsync`'s loop over `block::DEADMAN`): there, a budget expiry is retried -above every lock until the deadman, and only then reported. - -Not a loss either way — a refused resize changes nothing — so this is a spurious -failure, not corruption. It is filed rather than fixed here because the fix is a -retry ladder in a second syscall, which is a decision about where retries live -rather than a patch. - -**`SYS_FTRUNCATE` is not the only one, and one of them has been seen red.** -`kernel/src/fat32_adapter.rs:546` maps `Error::BudgetExpired` to `WouldBlock` -for every FAT operation, so a *delete* carries the same status out with nothing -behind it. #377's adversarial review recorded `fs_transactional` red once in its -runs on the dev host — `ALONE` pass, 0 of 6 further runs, not on -`src/redlist.rs`'s list, all twelve CI shards green: - - usb-storage: ... transport broke on SCSI 0x2a: no answer in the status phase in 2000 ms - log-volume: delete of fstx_keep_dst.bin: the device would not answer in the caller's own budget - -Seen a second time on the dev host, on a `kernel-userland-reach` fast tier of -103 guests with four sibling worktrees running beside it: the same two lines, -`cleanup: Kind(WouldBlock)` at `fs_transactional.rs:48`, `ALONE` pass again. -Two sightings, both beside other guests, and still no rate. - -That is the two-deadlines-in-series producer `src/redlist.rs:2994` retired for -`esp_filesystem` on 2026-08-23 — `USB_TIMEOUT_NS` breached on a WRITE(10) status -phase, then `block::OPERATION` refusing the retry unissued. The retry that -retired it lives in `ops::fsync` and covers `SYS_FSYNC` alone, so the class it -closed still reaches userland on every other path. Whoever fixes this fixes -both, which is the other half of why the answer is where retries live. - -**Exit condition.** Either the budget expiry is retried above the lock the way -`SYS_FSYNC` does — for the truncate and the delete both — and only what survives -the deadman is reported; or the mapping stops claiming retryability and the -reason is written at the site. -Whichever, the `resize-fault-refuse` actuator already stages the refusal, so the -chosen answer has an instrument to be measured with. diff --git a/issues/kernel/leak-rollback-selftest-create-answers-wouldblock.md b/issues/kernel/leak-rollback-selftest-create-answers-wouldblock.md deleted file mode 100644 index 312d51523bc..00000000000 --- a/issues/kernel/leak-rollback-selftest-create-answers-wouldblock.md +++ /dev/null @@ -1,22 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-03 ---- - -# `leak_rollback_selftest`'s `create` answered `WouldBlock` once under a loaded host - -A full fast tier on `w5b13-gop-mode` at 2d9ebb5c, 2026-09-03: 303/304 in -507.6 s beside other worktrees' guests, `ALONE leak_rollback_selftest: GREEN`. -The diff carried no kernel file. - -``` -leak-selftest: fat-reopen skipped, create failed: WouldBlock -``` - -A `create` answering `WouldBlock` is the shape -`issues/kernel/ftruncate-answers-wouldblock-and-nothing-retries-it.md` records -for `ftruncate` — a sibling site, not this one. - -Owed: the site in the create path that lets `WouldBlock` reach the caller, and -whether it retries. diff --git a/issues/kernel/volatile-composites-on-mmio-dma-structs.md b/issues/kernel/volatile-composites-on-mmio-dma-structs.md index 226adca403e..f911dc0bc8d 100644 --- a/issues/kernel/volatile-composites-on-mmio-dma-structs.md +++ b/issues/kernel/volatile-composites-on-mmio-dma-structs.md @@ -8,10 +8,9 @@ opened: 2026-08-20 `clippy::volatile_composites` (nursery) measured **10**, kernel-only, all in driver code that hands a composite type straight to `write_volatile` or -`read_volatile` instead of touching each primitive field on its own: +`read_volatile` instead of touching each primitive field on its own. Two were +the NVMe driver's queue entries, which left the kernel with it; the rest: -- `drivers/nvme.rs:101,119` — `SqEntry`/`CqEntry`, the submission/completion - queue entries. - `drivers/xhci/wait/boot.rs:446` — `ErstEntry`. - `drivers/xhci/mod.rs:560,593,1093` — `Trb`, three sites. - `drivers/virtio.rs:414,487` — the virtio descriptor. @@ -32,9 +31,8 @@ splitting a `write_volatile(ptr, trb)` into per-field stores has to preserve that ordering by hand, on a path this driver depends on for every command and transfer. Getting that wrong silently (works in QEMU, wrong on some real controller's stricter timing, or vice versa) is worse than the current -implementation-defined-but-tested state. NVMe's `SqEntry`/`CqEntry` and -virtio's descriptor likely have their own per-field ordering rules from their -respective specs, not derived here. +implementation-defined-but-tested state. Virtio's descriptor likely has its +own per-field ordering rules from its spec, not derived here. Fixing this is real driver work: read each device's spec for what field ordering its rings actually require, decide per site whether one diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 4eebe1a3e17..86206e4c306 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -334,9 +334,6 @@ actuators! { /// Raise a vector no `idt_vectors!` row claims on this CPU once. unclaimed_vector_selftest = "unclaimed-vector-selftest"; - /// Hold a flush of `truncate-race.bin` inside its metadata window and say whether a truncate got in. - ftruncate_flush_stall = "ftruncate-flush-stall"; - /// Deliver the i8042 vector once at arming with no byte behind it — the arming edge, staged. i8042_arm_edge = "i8042-arm-edge"; diff --git a/kernel/src/object/ops.rs b/kernel/src/object/ops.rs index 2eaee1894d1..8a9fa8468ee 100644 --- a/kernel/src/object/ops.rs +++ b/kernel/src/object/ops.rs @@ -736,6 +736,14 @@ fn partition_fsync(claim: &DeviceClaim) -> u64 { Some(view) => view.flush().map_err(block_word), None => Err(SyscallError::Gone), }); + if let Answered::Answer { answer: Ok(()), attempts, took } = &run { + if *attempts > 1 { + crate::log!( + "partclaim: a flush durable on attempt {attempts} after {took} — a refused \ + attempt was asked again on a fresh budget", + ); + } + } partition_word("a flush", run) } diff --git a/kernel/src/tmpfs.rs b/kernel/src/tmpfs.rs index bede5f38f89..bfa415616c8 100644 --- a/kernel/src/tmpfs.rs +++ b/kernel/src/tmpfs.rs @@ -14,8 +14,7 @@ use crate::vfs::FileSystem; struct TmpfsBacking { file_id: FileId, - /// Shared by every backing for the entry; cleared by [`retire`], as - /// `FileBlocks::revoke` clears `/home`'s. + /// Shared by every backing for the entry; cleared by [`retire`]. alive: Arc, } diff --git a/kernel/src/vfs.rs b/kernel/src/vfs.rs index 1576f8c1c89..29763ff0643 100644 --- a/kernel/src/vfs.rs +++ b/kernel/src/vfs.rs @@ -32,29 +32,6 @@ pub fn lock() -> VfsGuard { VfsGuard(VFS.lock()) } -/// Holds one flush of the staged file inside its size-read/`update_metadata` -/// pair and says which way the race went; 400ms, under the ticket lock's tripwire. -#[cfg(feature = "boot-actuators")] -fn stalled_metadata_window(path: &str, file_id: FileId, size_read: u64) { - if !crate::actuator::ftruncate_flush_stall() || !path.ends_with("truncate-race.bin") { - return; - } - const STALL_NS: u64 = 400_000_000; - let until = crate::clock::nanos_since_boot().saturating_add(STALL_NS); - while crate::clock::nanos_since_boot() < until { - core::hint::spin_loop(); - } - let size_after = crate::file_cache::size(file_id); - if size_after == size_read { - crate::log!("vfs: STALLED WINDOW HELD — {path} still {size_read} bytes after 400ms"); - } else { - crate::log!( - "vfs: STALLED WINDOW BROKEN — a truncate landed inside {path}'s metadata window \ - ({size_read} -> {size_after})" - ); - } -} - /// A device that would not answer is `Io`; `NotFound` means only that the name is absent. pub trait FileSystem: Send { /// Every name [`under_directory`] puts at or under `dir` (`""` is the mount @@ -513,8 +490,6 @@ impl Vfs { crate::file_cache::settle_pages(file_id, &flushed); let size = crate::file_cache::size(file_id); - #[cfg(feature = "boot-actuators")] - stalled_metadata_window(path, file_id, size); fs.update_metadata(file_id, size, mtime)?; crate::file_cache::settle_file(file_id, plan.file); diff --git a/tests/blockdcase/system.toml b/tests/blockdcase/system.toml index ed219ed161a..8dd8ceaf9c0 100644 --- a/tests/blockdcase/system.toml +++ b/tests/blockdcase/system.toml @@ -1,13 +1,14 @@ -# The blockd boot: `tests/testcases`'s estate, with a second blockd beside the -# one init runs for DATA. `test_rs_blockd_io` is the second's supervisor as -# init is a service's: it mints the claim on the machine's second NVMe -# controller from its capability, starts blockd holding it and a port it made, -# and is the only thing that can end or restart it. The disks are crafted by -# `tests/common/blockd.rs`. No soundd: `test_rs_blockd_io dma-pool` claims +# The blockd boot: `tests/testcases`'s estate, and `/system/bin/blockd` in the +# image started by nothing. `test_rs_blockd_io` is blockd's supervisor as init +# is a service's: it mints the claim on the machine's second NVMe controller +# from its capability, starts blockd holding it and a port it made, and is the +# only thing that can end or restart it. Nothing drives the first controller, +# so every command in QEMU's NVMe trace is that blockd's. The disks are crafted +# by `tests/common/blockd.rs`. No soundd: `test_rs_blockd_io dma-pool` claims # virtio-sound itself, to lend the pool its kernel driver keeps. [boot] -start = ["logd", "blockd", "fsd", "test-runner"] +start = ["logd", "fsd", "test-runner"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and writes no file at all, so a @@ -59,19 +60,13 @@ syscap = ["device", "dup", "logread", "power", "roster"] # that one certified a path no user takes. "bin/tone" = "/system/bin/toybox" -# The block service: the NVMe controller this machine's DATA is on, driven -# from userland through its claim. Started again when it ends; the claim goes -# back with the process and is minted again for the next. +# The NVMe driver, from userland. No row starts it and it holds nothing here: +# the test that runs it hands it its claim and its port itself. [programs.blockd] -service = true -restart = true -serves = ["block"] -devices = ["pci:1b36:0010"] -# The file servers: DATA, the log and the running slot's volume, one process -# each, serving every program the directories of its role. Started again when -# one ends, on the same ports. +# The file servers: the log and the running slot's volume off the stick, and +# DATA in memory, since no block service runs. Started again when one ends, on +# the same ports. [programs.fsd] restart = true roles = ["data", "log", "boot"] -receives = ["block"] diff --git a/tests/common/blockd.rs b/tests/common/blockd.rs index 10390e75b8f..1806ade99d3 100644 --- a/tests/common/blockd.rs +++ b/tests/common/blockd.rs @@ -10,9 +10,9 @@ //! command, completion and flush, which no driver can print on the device's //! behalf. //! -//! The machine has two NVMe controllers: the kernel's (QEMU's own ids, first -//! by enumeration, so the kernel's first-by-class probe takes it) with DATA and -//! a bench partition, and blockd's (Intel's ids) with the partitions below. +//! The machine has two NVMe controllers: the first (QEMU's own ids) with a DATA +//! and a partition nothing names, which no driver runs on this boot, and +//! blockd's (Intel's ids) with the partitions below. use std::collections::{BTreeMap, BTreeSet}; use std::io::{Read, Seek, SeekFrom, Write}; @@ -29,7 +29,8 @@ const FS: &str = "B2D4F6A8-1C3E-4A57-9B0D-E2F4A6C8E0A1"; const BENCH: &str = "C3E5A7B9-2D4F-4B68-8C1E-F3A5B7D9F1B2"; const MISALIGNED: &str = "E5A7C9DB-4F6B-4D8A-8E30-B5C7D9FB13D4"; const MISSTART: &str = "F6B8DAEC-5A7C-4E9B-9F41-C6D8EA0C24E5"; -/// Mirrored: the kernel's disk. +/// The first controller's disk: the system's own DATA, which init's blockd +/// serves, beside a partition nothing names. const KBENCH: &str = "D4F6B8CA-3E5A-4C79-9D2F-A4B6C8EA02C3"; const TARGET_BLOCKS: u64 = 2048; const BENCH_BLOCKS: u64 = 8192; @@ -78,8 +79,8 @@ struct Layout { /// and two partitions that are not whole 4 KiB blocks. /// /// The neighbours are long enough that the volume starts past every byte of -/// the kernel's disk: QEMU's trace names no controller, so a write is blockd's -/// by the sector it lands on ([`reissued_after`]). +/// the first controller's disk: QEMU's trace names no controller, so a write +/// is blockd's by the sector it lands on ([`reissued_after`]). fn craft_blockd_disk(path: &Path) -> Result { const NEIGHBOUR_BYTES: u64 = 64 * MIB; const FS_BYTES: u64 = 64 * MIB; @@ -115,14 +116,14 @@ fn boot( params: &'static [&'static str], ) -> Result<(QemuInstance, Layout, PathBuf, PathBuf, Vec), String> { let config = super::compile::repo_root().join(CONFIG); - let kernel_disk = super::lane::dir().join(format!("{name}-kernel.img")); - partclaim::craft_plain_disk(&kernel_disk, &[("kernel bench", BENCH_BLOCKS * BLOCK, KBENCH)], 96 * MIB)?; + let first_disk = super::lane::dir().join(format!("{name}-first.img")); + partclaim::craft_plain_disk(&first_disk, &[("unnamed", BENCH_BLOCKS * BLOCK, KBENCH)], 96 * MIB)?; let blockd_disk = super::lane::dir().join(format!("{name}-blockd.img")); let layout = craft_blockd_disk(&blockd_disk)?; - let kernel_bytes = std::fs::metadata(&kernel_disk).map_err(|e| format!("the kernel's disk: {e}"))?.len(); - if kernel_bytes > layout.fs.start { + let first_bytes = std::fs::metadata(&first_disk).map_err(|e| format!("the first disk: {e}"))?.len(); + if first_bytes > layout.fs.start { return Err(format!( - "the kernel's disk is {kernel_bytes} bytes and blockd's volume starts at {}: a traced write \ + "the first controller's disk is {first_bytes} bytes and blockd's volume starts at {}: a traced write \ there could be either controller's", layout.fs.start )); @@ -135,7 +136,7 @@ fn boot( c_bins, rust_bins, BootOptions { - nvme_image: Some(kernel_disk), + nvme_image: Some(first_disk), userland_nvme: Some(blockd_disk.clone()), nvme_trace: Some(trace.clone()), kernel_params: params, @@ -172,8 +173,8 @@ struct Traced { flushes: usize, /// Submission queues reads and writes arrived on. queues: BTreeSet, - /// The most commands outstanding at once on queues 2 and up — the - /// kernel's driver has one I/O queue, so those are blockd's alone. + /// The most commands outstanding at once on queues 2 and up — nothing but + /// blockd drives a controller on this boot. peak: usize, } @@ -256,9 +257,10 @@ fn trace_events(trace: &Path) -> Result, String> { /// again first thing after — read off QEMU's trace alone. /// /// **A write is blockd's by where it lands**: the trace names no controller, -/// and every partition blockd writes starts past the last byte of the kernel's -/// disk ([`boot`] refuses a layout where the volume does not, and the bench -/// partition lies past it). A lifetime is what follows one controller start — +/// and every partition blockd writes starts past the last byte of the first +/// controller's disk ([`boot`] refuses a layout where the volume does not, and +/// the bench partition lies past it). A lifetime is what follows one +/// controller start — /// blockd's bring-up, or its reset. The one the loss ended is the lifetime /// whose writes to `span` did not end in a flush and after which another /// lifetime wrote to it; its last write is the one blockd withheld the answer @@ -268,7 +270,7 @@ fn trace_events(trace: &Path) -> Result, String> { /// which is what reissue means, and what a client that forgot them cannot do /// by accident. Two writes that do not overlap may be in flight together, so /// the device's order between them is neither lifetime's to keep. Every Flush -/// in the trace is blockd's: the kernel's driver issues none. +/// in the trace is blockd's: nothing else drives a controller. fn reissued_after(trace: &Path, span: Span) -> Result, String> { let mut lives: Vec> = Vec::new(); for did in trace_events(trace)? { @@ -358,14 +360,13 @@ fn neighbours_untouched(layout: &Layout, before: &[u8], after: &[u8]) -> Result< Ok(()) } -/// The partition claims served by blockd, and blockd against the kernel's -/// driver. +/// The partition claims served by blockd, and blockd's rate. /// /// - Every refusal by name, the idle ROOT slot's 2048 blocks written whole /// through a session and read back, and one holder at a time across two /// processes — the slot a second client is refused while it is held and /// opened once it is not. -/// - The same bytes through the kernel's driver and through blockd, timed. +/// - The same bytes through blockd, one request at a time and many, timed. /// - Off the image: the slot holds every block the guest wrote, and nothing /// outside the sessions' partitions moved. /// - Off QEMU's trace, which no driver writes: blockd's Flush reached the diff --git a/tests/common/volumes.rs b/tests/common/volumes.rs index 1836858d2c9..e97286de694 100644 --- a/tests/common/volumes.rs +++ b/tests/common/volumes.rs @@ -1,14 +1,15 @@ -//! The boot stick's two partitions, mounted and written from inside ToyOS. +//! The boot stick's two partitions, served and written from inside ToyOS. //! //! The ESP holds what firmware and the bootloader read. The log partition //! beside it holds the kernel's own log and exists for one reason: it is typed //! so that a desktop OS mounts it on plug-in, which an EFI-typed partition is -//! not. Both are FAT32 and neither is found by being FAT32 — the kernel is -//! handed both by unique GUID, and `log_partition_identity` is the gate that -//! says so by moving the name and watching the mount disappear. +//! not. Both are FAT32 and neither is found by being FAT32 — the loader names +//! both by unique GUID, fsd serves each off the partition that name finds, +//! and `log_partition_identity` is the gate that says so by moving the name +//! and watching `/log` go absent. //! //! Ground truth is the disk image the *device* received, read on the host by -//! implementations that are not the kernel's: the `fatfs` crate and +//! implementations that are not fsd's: the `fatfs` crate and //! `toyos-fat32-check`. The guest's account of a write it made is exactly what //! is in question, so it cannot also be the evidence; `esp_files` asserts what //! only a process inside the machine can see, and everything it claims about @@ -460,7 +461,7 @@ fn volume_lines(log: &str) -> String { .lines() .filter(|l| l.contains("fsd:") || l.contains("blockd:") || l.contains("logd:") || l.contains("gpt:") || l.contains("shutdown") || l.contains("Shutting down") || l.contains("Syncing") - || l.contains("usb-storage:")) + || l.contains("usb-storage:") || l.contains("partclaim:")) .collect(); if lines.is_empty() { return format!("the guest said nothing about its volumes at all\n{log}"); @@ -852,321 +853,6 @@ fn rotation( Ok(()) } -/// A file closed **without** an fsync still reaches the volume — by the -/// write-back queue and the shutdown drain — and leaves the volume parsable. -/// -/// `OpenFileState::drop` no longer flushes on the closing thread; it pins the -/// file and hands it to `iod`, and `SYS_SHUTDOWN` drains that queue before it -/// commits the devices' write caches (`kernel::writeback`). The guest -/// (`test_rs_writeback_durability`) writes a blob to `/log`, drops the handle -/// with no sync, and reads it back after `iod` drains — the in-guest half of the -/// no-loss claim. This is the **independent oracle**: after `run shutdown` the -/// `/log` volume is read off the image by the `fatfs` crate and checked by -/// `toyos-fat32-check`, neither the kernel's own cache logic, so both the bytes -/// on the device and the structure around them are judged by something that is -/// not the code under test. -pub fn writeback_durability( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - /// Mirrored in `tests/toyos-rust-tests/src/bin/writeback_durability.rs`. - const BLOB_NAME: &str = "wb-durable.bin"; - const BLOB_LEN: usize = 5 * 4096 + 91; - fn blob() -> Vec { - (0..BLOB_LEN).map(|i| (i.wrapping_mul(97) ^ 0x5A) as u8).collect() - } - /// The same mirroring, for the file the guest shrank and regrew. - const SHRUNK_NAME: &str = "wb-shrunk.bin"; - const SHRUNK_LEN: usize = 3 * 4096; - const CUT: usize = 100; - fn seed() -> Vec { - (0..SHRUNK_LEN).map(|i| (i.wrapping_mul(31).wrapping_add(7)) as u8 | 1).collect() - } - /// And for the bytes a read got back from a file shrunk only once cold. - const SERVED_NAME: &str = "wb-served.bin"; - /// And for the file whose straddled read is refused, which leaves it whole. - const REFUSED_NAME: &str = "wb-refused.bin"; - /// And for the file whose flush is refused at its directory-entry write. - const RETRY_NAME: &str = "wb-retry.bin"; - const RETRY_AT: usize = 2 * 4096; - fn kept() -> Vec { - (0..4096usize).map(|i| (i.wrapping_mul(53) ^ 0xC3) as u8).collect() - } - - // Stage the one failure QEMU will not produce: a budget expiry on the FAT-1 - // mirror write of the blob's cluster allocation, on the write-back drain path - // (`fat-mirror-write-refuse`). Without the drain's retry and - // `set_fat_entry`'s active-last order the volume is left with the FATs split; - // with them the drain re-drives the flush and heals it, and the checker below - // is the independent judge either way. The image is built with the parameter - // so the guest boots the test kernel that carries the actuator. - // The second refuses one file's second directory-entry write, so a flush - // fails at its metadata write with its pages already written and settled. - // Different file, different site: neither stands in for the other. - // The third is the sweep any other CPU can run in the window between that - // read and the lock that spends it, holding no VFS lock of its own. - const PARAMS: &[&str] = &[ - "fat-mirror-write-refuse", - "fat-flush-meta-refuse", - "resize-evict-window", - "resize-fault-refuse", - ]; - - let image_path = test_dir().join("writeback-durability.img"); - let image = qemu::build_boot_image(test_config, c_bins, rust_bins, PARAMS); - std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - let (start, len) = log_extent(&image, &image_path)?; - - // Born clean, asserted rather than assumed, so a complaint after the run is - // the guest's and not one it inherited. - let complaints_before = check(&image[start..start + len]); - if !complaints_before.is_empty() { - return Err(format!( - "the log partition was not born clean, so this gate cannot tell a complaint the \ - guest caused from one it inherited:\n{}", - describe(&complaints_before) - )); - } - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - boot_image: Some(qemu::Staged::Written(image_path.clone())), - kernel_params: PARAMS, - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; - if !boot.contains(LOG_SERVED) { - return Err(format!( - "the log partition did not mount, so the guest had nowhere to write:\n{}", - volume_lines(&boot) - )); - } - - let result = qemu.run_test("test_rs_writeback_durability", Duration::from_secs(60)); - if let Some(err) = &result.error { - return Err(format!("the guest stopped answering: {err}\nserial:\n{}", result.serial)); - } - if result.exit_code != Some(0) { - // The kernel's own lines too: a refused write on the log path reaches - // userland as one `Io`, and which layer refused is only in a `log!`. - return Err(format!( - "writeback_durability guest failed:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - serial::Serial::named("test serial", result.serial.as_str()).must_be_clean()?; - - // The shutdown drains the write-back queue and then commits the device's own - // write cache: it is what makes the host's view of the backing file the - // device's view of it. - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - - // The actuator must have fired **twice**, or this gate proves nothing: a - // green run with an inert arm is the worst harness defect - // (`tests/common/qemu.rs`'s `refuse_a_staged_image_this_boot_did_not_ask_for`). - // Two refusals are what drive the drain's retry ladder to attempt 2 — the - // first depth that *parks* between attempts, where `iod`'s standing `WORK` - // arm made `between_attempts` a double-arm and panicked the machine. One - // refusal reaches only attempt 1 (a yield), which is why the machine passed - // the single-refusal control while broken. So this gate proves both halves of - // the durability fix: `set_fat_entry`'s active-last order heals the mirror - // (checker silent, below), and `drain_all_iod`'s arm-reusing backoff survives - // the attempt-2 park (`must_be_clean` above and the `PANIC` scan of the - // shutdown tail catch the double-arm). It fires on `iod`'s drain during the - // run, or on the shutdown drain if `iod` was late, so the whole capture is - // searched. - const FIRED: &str = "fat-mirror-write-refuse: refusing the FAT-1 mirror write"; - let fired = [result.before.as_str(), result.serial.as_str(), tail.as_str()] - .iter() - .map(|cap| cap.matches(FIRED).count()) - .sum::(); - if fired < 2 { - return Err(format!( - "the fat-mirror-write-refuse actuator fired {fired} time(s), not the 2 that drive the \ - drain's retry ladder to the attempt that parks — the gate never reached the path \ - that panicked, so it proves nothing\nkernel log:\n{}{}\nshutdown:\n{}", - result.before, result.serial, tail - )); - } - - let after = std::fs::read(&image_path).map_err(|e| format!("read the image back: {e}"))?; - if after.len() != image.len() { - return Err(format!("the image is {} bytes, was {}", after.len(), image.len())); - } - let volume = &after[start..start + len]; - - // The strongest claim first: the volume is still a volume. A driver that - // wrote the right bytes into a broken FAT would pass the byte comparison - // below and leave a stick that cannot boot. - let complaints_after = check(volume); - if !complaints_after.is_empty() { - return Err(format!( - "the write-back left the log volume breaking the format:\n{}", - describe(&complaints_after) - )); - } - - // Ground truth: the bytes on the device, against the bytes the guest wrote, - // read by the host's own FAT implementation and never by the kernel. - let mut files = - read_files(volume, &[BLOB_NAME, SHRUNK_NAME, SERVED_NAME, RETRY_NAME, REFUSED_NAME])?; - let refused = need(files.pop().flatten(), REFUSED_NAME)?; - let retried = need(files.pop().flatten(), RETRY_NAME)?; - let served = need(files.pop().flatten(), SERVED_NAME)?; - let shrunk = need(files.pop().flatten(), SHRUNK_NAME)?; - let got = need(files.pop().flatten(), BLOB_NAME)?; - if got.len() != BLOB_LEN { - return Err(format!( - "{BLOB_NAME} is {} bytes on the volume; the guest wrote {BLOB_LEN} and never fsynced — \ - a write-back closed but not drained is a lost write", - got.len() - )); - } - if let Some(at) = got.iter().zip(blob()).position(|(a, b)| *a != b) { - return Err(format!("{BLOB_NAME} differs on the volume from what the guest wrote at byte {at}")); - } - - // The shrink the drain's one `update_metadata` could not see. The volume is - // valid either way — a file naming its own old clusters breaks no rule the - // checker knows — so the bytes are the only place this shows. - if shrunk.len() != SHRUNK_LEN { - return Err(format!( - "{SHRUNK_NAME} is {} bytes on the volume; the guest regrew it to {SHRUNK_LEN}", - shrunk.len() - )); - } - if shrunk[..CUT] != seed()[..CUT] { - return Err(format!("{SHRUNK_NAME}'s surviving head changed across the shrink and regrow")); - } - if let Some(at) = shrunk[CUT..].iter().position(|&b| b != 0) { - return Err(format!( - "{SHRUNK_NAME} byte {} on the volume is {:#04x}, not zero — the shrink gave no clusters \ - back and the regrow served the discarded tail", - CUT + at, - shrunk[CUT + at], - )); - } - - // The same shrink over a page the cache did not hold, judged through the - // bytes a read got back: `Fat32::set_len` zero-fills on every grow, so the - // volume's own bytes are right whatever the cache answered. - // The refused fault. The actuator must have fired, or the truncate was - // never refused; then the file has to be untouched on the volume. - const FAULT_REFUSED: &str = "resize-fault-refuse: refusing the straddled read"; - let fault_refused = [result.before.as_str(), result.serial.as_str(), tail.as_str()] - .iter() - .map(|cap| cap.matches(FAULT_REFUSED).count()) - .sum::(); - if fault_refused != 1 { - return Err(format!( - "the resize-fault-refuse actuator fired {fault_refused} time(s), not the 1 the staged \ - shrink asks for — the refused error path this gate covers never ran\nkernel log:\n{}{}\nshutdown:\n{}", - result.before, result.serial, tail - )); - } - if refused.len() != SHRUNK_LEN || refused[..] != seed()[..] { - return Err(format!( - "{REFUSED_NAME} is {} bytes on the volume and not the {SHRUNK_LEN} bytes the guest \ - seeded it with — a truncate whose straddled read was refused changed the file", - refused.len() - )); - } - - // The sweep must have run in the window, or the eviction this arm is about - // never happened and the bytes below prove only the ordinary cold shrink. - const SWEPT: &str = "resize-evict-window swept page"; - let swept = [result.before.as_str(), result.serial.as_str(), tail.as_str()] - .iter() - .map(|cap| cap.matches(SWEPT).count()) - .sum::(); - if swept != 1 { - return Err(format!( - "the resize-evict-window actuator swept {swept} time(s), not the 1 the cold shrink \ - stages — nothing ran in the window this arm exists for\nkernel log:\n{}{}\nshutdown:\n{}", - result.before, result.serial, tail - )); - } - if served.len() != SHRUNK_LEN { - return Err(format!( - "{SERVED_NAME} is {} bytes; the guest read back a file it had regrown to {SHRUNK_LEN}", - served.len() - )); - } - if served[..CUT] != seed()[..CUT] { - return Err(format!("{SERVED_NAME}'s surviving head changed across the shrink and regrow")); - } - if let Some(at) = served[CUT..].iter().position(|&b| b != 0) { - return Err(format!( - "{SERVED_NAME} byte {} is {:#04x}, not zero — a read of a file shrunk after it left \ - the cache was served the discarded tail", - CUT + at, - served[CUT + at], - )); - } - - // The retry gate. The actuator must have fired, or the flush never failed - // and this proves nothing; then the page written above the mark has to be - // on the volume, which a second trim would have freed. - const META_REFUSED: &str = "fat-flush-meta-refuse: refusing the directory-entry write"; - let meta_fired = [result.before.as_str(), result.serial.as_str(), tail.as_str()] - .iter() - .map(|cap| cap.matches(META_REFUSED).count()) - .sum::(); - if meta_fired != 1 { - return Err(format!( - "the fat-flush-meta-refuse actuator fired {meta_fired} time(s), not the 1 that makes \ - the flush fail at its metadata write — the retry this gate is about never \ - happened\nkernel log:\n{}{}\nshutdown:\n{}", - result.before, result.serial, tail - )); - } - if retried.len() != SHRUNK_LEN { - return Err(format!( - "{RETRY_NAME} is {} bytes on the volume; the guest regrew it to {SHRUNK_LEN}", - retried.len() - )); - } - if retried[RETRY_AT..] != kept()[..] { - let at = retried[RETRY_AT..].iter().zip(kept()).position(|(a, b)| *a != b); - return Err(format!( - "{RETRY_NAME} byte {:?} of the page written above the shrink is not what the guest \ - wrote — the flush's retry trimmed a second time and freed it", - at.map(|i| RETRY_AT + i), - )); - } - if let Some(at) = retried[CUT..RETRY_AT].iter().position(|&b| b != 0) { - return Err(format!( - "{RETRY_NAME} byte {} between the shrink and the rewritten page is {:#04x}, not zero", - CUT + at, - retried[CUT + at], - )); - } - - let _ = std::fs::remove_file(&image_path); - eprintln!( - " [wb] {BLOB_LEN} bytes closed without fsync reached the log volume through the \ - write-back queue and the shutdown drain; {SHRUNK_NAME}'s regrown tail is zeros and so \ - is what a cold shrink served into {SERVED_NAME}; {RETRY_NAME} kept the page it wrote \ - through a refused metadata write; the checker is silent" - ); - Ok(()) -} - /// A `FatBacking` handed out before an unlink reads nothing after it, and the /// delete-and-reallocate cycle leaves the volume a volume. /// @@ -1489,109 +1175,6 @@ pub fn redirty_mid_flush( Ok(()) } -/// A truncate staged inside a flush's `update_metadata` window -/// (`ftruncate-flush-stall`), which `SYS_FTRUNCATE`'s lockless resize once -/// landed in. The guest makes the race and asserts its truncate serialised; -/// the shut-down volume is re-judged by the FAT reader and the fatgen103 -/// checker — `fs_rename_durable`'s oracle. -pub fn ftruncate_flush_race( - test_config: &Path, - c_bins: &[(String, Vec)], - rust_bins: &[(String, Vec)], -) -> Result<(), String> { - const PARAMS: &[&str] = &["ftruncate-flush-stall"]; - /// Mirrored in `tests/toyos-rust-tests/src/bin/ftruncate_flush_race.rs`. - const TARGET: &str = "truncate-race.bin"; - const SHORT: usize = 5000; - - let image_path = test_dir().join("ftruncate-flush-race.img"); - let image = qemu::build_boot_image(test_config, c_bins, rust_bins, PARAMS); - std::fs::write(&image_path, &image).map_err(|e| format!("write the boot image: {e}"))?; - let (start, len) = log_extent(&image, &image_path)?; - - let mut qemu = QemuInstance::boot_with_options( - test_config, - c_bins, - rust_bins, - BootOptions { - profile: qemu::Profile::Metal, - boot_image: Some(qemu::Staged::Written(image_path.clone())), - kernel_params: PARAMS, - ..Default::default() - }, - ); - let boot = qemu.boot_log().to_string(); - if !boot.contains(LOG_SERVED) { - return Err(format!( - "the log partition did not mount, so the race had nowhere to run:\n{}", - volume_lines(&boot) - )); - } - - let result = qemu.run_test("test_rs_ftruncate_flush_race", Duration::from_secs(120)); - if let Some(err) = &result.error { - return Err(format!("the guest stopped answering: {err}\nserial:\n{}", result.serial)); - } - if result.exit_code != Some(0) { - return Err(format!( - "ftruncate_flush_race guest failed — the truncate did not serialise with the \ - stalled flush:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - let kernel_log = format!("{}{}", result.before, result.serial); - if kernel_log.contains("STALLED WINDOW BROKEN") { - return Err(format!( - "a truncate landed inside the flush's metadata pair — the resize is not under \ - the VFS lock:\n{kernel_log}" - )); - } - if !kernel_log.contains("STALLED WINDOW HELD") { - return Err(format!( - "the stalled window never reported, so nothing was staged and this gate judged \ - nothing:\n{kernel_log}" - )); - } - - writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); - qemu.flush_stdin(); - let tail = qemu.drain_serial(Duration::from_secs(20)); - drop(qemu); - for bad in ["PANIC:", "panicked at"] { - if tail.contains(bad) { - return Err(format!("{bad:?} on the way down\n{tail}")); - } - } - - let after = std::fs::read(&image_path).map_err(|e| format!("read the image back: {e}"))?; - let volume = &after[start..start + len]; - let complaints = check(volume); - if !complaints.is_empty() { - return Err(format!( - "the staged race left the log volume breaking the format:\n{}", - describe(&complaints) - )); - } - let got = need(read_files(volume, &[TARGET])?.pop().flatten(), TARGET)?; - if got.len() != SHORT { - return Err(format!( - "{TARGET} is {} bytes on the device, not the {SHORT} the truncate settled — \ - the flush's metadata pair recorded a size the file no longer has", - got.len() - )); - } - if let Some(at) = got.iter().position(|&b| b != 0xB6) { - return Err(format!("{TARGET} byte {at} is {:#04x}, not the 0xB6 the guest wrote", got[at])); - } - - let _ = std::fs::remove_file(&image_path); - eprintln!( - " [ftruncate] the truncate waited out the stalled window in-guest; {SHORT} bytes on \ - the device by the host's own reader, checker silent" - ); - Ok(()) -} - /// A rename whose source is absent leaves the destination on the FAT `/log` /// volume, and `rename(p, p)` leaves the file — the **independent oracle** for /// the same reason `fat_backing_revoked` is one. The guest @@ -1913,11 +1496,14 @@ pub fn late_storage_connect( )); } - for want in [ - BOOT_SERVED, - LOG_SERVED, - "logd: this boot's kernel log is", - ] { + // logd opens its file once its file server answers, which need not be + // before the runner's ready line: read on until it says so. + const LOG_OPENED: &str = "logd: this boot's kernel log is"; + let mut boot = boot; + if !boot.contains(LOG_OPENED) { + boot.push_str(&qemu.drain_until(Duration::from_secs(20), |l| l.contains(LOG_OPENED))); + } + for want in [BOOT_SERVED, LOG_SERVED, LOG_OPENED] { if !boot.contains(want) { return Err(format!( "the disk arrived after the port scan and {want:?} never happened — the probe \ @@ -2411,10 +1997,11 @@ pub fn log_partition_identity( /// `toyos-fat32-check`'s silence over the volume is the outside judge that /// the retry also left a consistent filesystem. /// 2. **The deadman is the third failure evidence.** `fsync-deadman-now` -/// expires it at once, so the first refused attempt is the last: the kernel -/// declares the volume failed with `SyscallError::Io`, and logd — which -/// keeps a volume across any number of `WouldBlock`s — ends it on what is -/// now a device word, exactly as it would for an error status. +/// expires it at once, so every claim transfer's first refused attempt is +/// the last: the kernel answers fsd's reads of the log partition +/// `SyscallError::Io`, fsd serves `/log` absent, and logd — which keeps a +/// volume across any number of `WouldBlock`s — has none, on what is a +/// device word, exactly as it would be for an error status. /// 3. **A failed reset escalation is the second.** `usb-transport-break` /// abandons the first WRITE(10)'s data phase and `usb-reset-break` makes /// the recovery ladder meet a device that answers nothing on EP0 — a truly @@ -2483,13 +2070,14 @@ pub fn log_flush_retry( volume_lines(&log) )); } + // `/log` is flushed through fsd's partition claim, whose retry says so. let retried = log .lines() - .find(|l| l.contains("fsync: ") && l.contains("durable on attempt")) + .find(|l| l.contains("partclaim: a flush durable on attempt")) .ok_or_else(|| { format!( - "no `fsync: … durable on attempt` line, so the operation-level retry never \ - ran:\n{}", + "no `partclaim: a flush durable on attempt` line, so the operation-level retry \ + never ran:\n{}", volume_lines(&log) ) })? @@ -2571,14 +2159,22 @@ pub fn log_flush_retry( } drop(qemu); serial::Serial::named("deadman boot console", log.as_str()).must_be_clean()?; + // Every claim transfer's first attempt is the last here, fsd's reads of + // the log partition among them, so the death is declared at the mount. let declared = log .lines() - .find(|l| l.contains("fsync: ") && l.contains("is not durable after")) + .find(|l| l.contains("partclaim: ") && l.contains("still refused after 1 attempt(s)")) .ok_or_else(|| { - format!("the deadman never declared the volume failed:\n{}", volume_lines(&log)) + format!("the deadman never declared a claim's transfer failed:\n{}", volume_lines(&log)) })? .trim() .to_string(); + if !log.contains(&format!("{LOG_ABSENT} no FAT32 here: device I/O failed")) { + return Err(format!( + "fsd served /log off a partition the deadman had declared failed:\n{}", + volume_lines(&log) + )); + } let gave_up = log .lines() .find(|l| l.contains("logd:") && l.contains("on the console only")) diff --git a/tests/test-durations b/tests/test-durations index 845cbef45a1..c1d6cab69f1 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -196,7 +196,6 @@ fs_rename_durable 9346 fs_transactional 85 fs_truncate_persist 17 fsync_failed_commit 8386 -ftruncate_flush_race 9452 futex_wake_counts 1007 gpu_set_resolution 8610 gsbase_locked 13091 @@ -427,7 +426,6 @@ wall_clock_zone 9347 watchdog_fed 23405 watchdog_resets 9393 window_refusal 16 -writeback_durability 8888 writeback_reopen 4886 writeback_spawn 8820 xhci_deaf_registers 21244 diff --git a/tests/toyos-rust-tests/src/bin/abuse_thread_name.rs b/tests/toyos-rust-tests/src/bin/abuse_thread_name.rs index 4bdc43a3b37..3059d3ec74f 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_thread_name.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_thread_name.rs @@ -1,10 +1,8 @@ //! `SYS_SET_THREAD_NAME` used to clamp an oversized name with `(a2 as //! usize).min(THREAD_NAME_LEN)` and set the truncated prefix — a silent -//! clamp, and the shape -//! `issues/isolation/untrusted-sites-not-yet-adopted.md` named for the -//! whole of `kernel/src/syscall/`. `Untrusted::at_most` replaced it -//! with a refusal, which is a behaviour change worth its own gate: this -//! proves the refusal actually fires, rather than the clamp it replaced. +//! clamp. `Untrusted::at_most` replaced it with a refusal, which is a +//! behaviour change worth its own gate: this proves the refusal actually +//! fires, rather than the clamp it replaced. //! //! `toyos_abi::syscall::set_thread_name` throws its own return value away — //! it always has, and that is not this test's to fix — so the return code diff --git a/tests/toyos-rust-tests/src/bin/blockd_io.rs b/tests/toyos-rust-tests/src/bin/blockd_io.rs index c80803d460c..aabbffd08f3 100644 --- a/tests/toyos-rust-tests/src/bin/blockd_io.rs +++ b/tests/toyos-rust-tests/src/bin/blockd_io.rs @@ -13,8 +13,8 @@ //! two processes; //! - `holder ` — the second process: opens the slot and says what it //! was answered; -//! - `bench` — the same bytes through the kernel's driver (a partition claim on -//! the first controller) and through blockd, timed; +//! - `bench` — the same bytes through blockd, one request at a time and many, +//! timed; //! - `reset` — blockd started withholding its second answer: the silence ends //! in a controller reset, the withheld write is answered not done, and the //! write acknowledged before it is on the medium after the next flush; @@ -60,8 +60,6 @@ const BENCH: &str = "C3E5A7B9-2D4F-4B68-8C1E-F3A5B7D9F1B2"; const MISALIGNED: &str = "E5A7C9DB-4F6B-4D8A-8E30-B5C7D9FB13D4"; const MISSTART: &str = "F6B8DAEC-5A7C-4E9B-9F41-C6D8EA0C24E5"; const ABSENT: &str = "0A1B2C3D-4E5F-4A6B-8C7D-9E0F1A2B3C4D"; -/// Mirrored: the kernel's disk, on the first controller. -const KBENCH: &str = "D4F6B8CA-3E5A-4C79-9D2F-A4B6C8EA02C3"; /// Mirrored: the idle slot's length in blocks, and what each block holds. const TARGET_BLOCKS: u64 = 2048; /// Mirrored: what the bench moves each way, each side. @@ -401,41 +399,11 @@ fn mb_per_s(blocks: u64, took: Duration) -> f64 { (blocks * BLOCK_BYTES as u64) as f64 / (1024.0 * 1024.0) / took.as_secs_f64() } -/// The same bytes through the kernel's driver and through blockd, the data -/// built before and checked after what is timed, so each number is the -/// driver's path and nothing of this binary's. +/// The same bytes through blockd one request at a time and then as many as +/// the arena holds, the data built before and checked after what is timed, +/// so each number is the driver's path and nothing of this binary's. fn bench() { - // The kernel's driver, through a partition claim on the first controller: - // one request at a time, as it moves them. - let syscap = capability(); - let part: toyos::PartitionDev = syscap - .claim_partition(PartGuid(guid(KBENCH))) - .unwrap_or_else(|e| fail(format!("the kernel's bench partition was refused: {e:?}"))); - let written = chunks(BENCH_BLOCKS, 0x3C); - let blocks: Vec> = written - .iter() - .map(|c| c.chunks(BLOCK_BYTES).map(|b| b.try_into().expect("a block")).collect()) - .collect(); - let started = Instant::now(); - let mut lba = 0u64; - for chunk in &blocks { - part.write(lba, chunk).unwrap_or_else(|e| fail(format!("a kernel write: {e:?}"))); - lba += chunk.len() as u64; - } - part.sync().unwrap_or_else(|e| fail(format!("the kernel's fsync: {e:?}"))); - let kernel_write = started.elapsed(); - let mut read = vec![[0u8; BLOCK_BYTES]; BENCH_BLOCKS as usize]; - let per = toyos_abi::part::MAX_BLOCKS_PER_CALL; - let started = Instant::now(); - for (i, chunk) in read.chunks_mut(per).enumerate() { - part.read((i * per) as u64, chunk).unwrap_or_else(|e| fail(format!("a kernel read: {e:?}"))); - } - let kernel_read = started.elapsed(); - holds(&read.concat(), &written, "the kernel's bench partition"); - drop(part); - - // blockd, one request at a time and then as many as the arena holds. - let blockd = Blockd::with(syscap, &[]); + let blockd = Blockd::with(capability(), &[]); let mut s = open(blockd.names(), BENCH); let mut runs = Vec::new(); for (salt, in_flight) in [(0x3D, 1usize), (0x3C, 15)] { @@ -455,11 +423,8 @@ fn bench() { )); } println!( - "blockd_io: bench {} MiB each way: kernel driver write {:.1} MiB/s (its one fsync issues no \ - Flush) read {:.1} MiB/s; blockd {}; at most {} requests on the wire", + "blockd_io: bench {} MiB each way: blockd {}; at most {} requests on the wire", BENCH_BLOCKS * BLOCK_BYTES as u64 / (1024 * 1024), - mb_per_s(BENCH_BLOCKS, kernel_write), - mb_per_s(BENCH_BLOCKS, kernel_read), runs.join("; "), s.peak_on_the_wire() ); diff --git a/tests/toyos-rust-tests/src/bin/esp_files.rs b/tests/toyos-rust-tests/src/bin/esp_files.rs index 1811f95f287..97d843bb93f 100644 --- a/tests/toyos-rust-tests/src/bin/esp_files.rs +++ b/tests/toyos-rust-tests/src/bin/esp_files.rs @@ -18,7 +18,8 @@ //! is the other half of that; this half is the attack. use std::fs; -use std::io::Write; +use std::io::{Read, Write}; +use std::os::toyos::fs::symlink; use toyos_abi::syscall::{self, OpenFlags}; @@ -54,11 +55,10 @@ fn names(dir: &str) -> Vec { /// The first 64 bytes of the loader; a plain WRITE at offset 0 shows in content, not length. fn loader_prefix() -> [u8; 64] { - let h = syscall::open(LOADER.as_bytes(), OpenFlags::READ).expect("open the loader for read"); let mut buf = [0u8; 64]; - let n = syscall::read(h, &mut buf).expect("read the loader's prefix"); - syscall::close(h); - assert_eq!(n, buf.len(), "short read of the loader's prefix"); + fs::File::open(LOADER) + .and_then(|mut f| f.read_exact(&mut buf)) + .expect("read the loader's prefix"); buf } @@ -111,10 +111,7 @@ fn boot_refuses_every_way_of_changing_it() { fs::remove_file(HOST_NOTE).expect_err("deleting a file on /boot was permitted"); fs::create_dir("/boot/toyos/newdir").expect_err("mkdir on /boot was permitted"); fs::rename(HOST_NOTE, "/boot/toyos/moved.txt").expect_err("rename on /boot was permitted"); - assert!( - toyos_abi::syscall::symlink(b"/boot/EFI/BOOT/BOOTx64.EFI", b"/boot/toyos/link").is_err(), - "symlink on /boot was permitted" - ); + assert!(symlink("/boot/EFI/BOOT/BOOTx64.EFI", "/boot/toyos/link").is_err(), "symlink on /boot was permitted"); // A read-only open still works, and reads. The refusal is of changes, not // of the mount. @@ -136,12 +133,12 @@ fn boot_refuses_every_way_of_changing_it() { "a /tmp symlink opened {LOADER} for writing", ); - let reader = syscall::open(LOADER.as_bytes(), OpenFlags::READ).expect("read /boot is allowed"); + let reader = fs::File::open(LOADER).expect("read /boot is allowed"); assert!( syscall::open(b"/tmp/evil", OpenFlags::WRITE).is_err(), "a /tmp symlink opened {LOADER} for writing while a /boot read handle was held", ); - syscall::close(reader); + drop(reader); assert_eq!(loader_prefix(), before, "a refused symlink write still changed the loader"); println!(" PASS a /tmp symlink to {LOADER} is refused for writing"); } @@ -171,7 +168,7 @@ fn log_takes_writes() { // leaving a regular file the caller believes is a link. On a mount that // permits writes, so what is being refused is the format and not the // policy. - let err = toyos_abi::syscall::symlink(b"/log/guest-note.txt", b"/log/link"); + let err = symlink("/log/guest-note.txt", "/log/link"); assert!(err.is_err(), "creating a symlink on FAT32 reported success"); assert!(!names("/log").iter().any(|n| n == "link"), "a refused symlink left a file"); println!(" PASS a symlink on /log is refused, and leaves nothing behind"); diff --git a/tests/toyos-rust-tests/src/bin/fs_escape.rs b/tests/toyos-rust-tests/src/bin/fs_escape.rs index 88db28edbcc..c8ee7e55598 100644 --- a/tests/toyos-rust-tests/src/bin/fs_escape.rs +++ b/tests/toyos-rust-tests/src/bin/fs_escape.rs @@ -8,8 +8,8 @@ //! - one that climbs and comes back down inside `/home` is followed; //! - a client that puts a `..` on the wire itself, past std, is refused //! `InvalidArgument` before anything is looked up; -//! - `/boot`, which only a `slots` row holds, is ROOT's empty directory for -//! this one, and the loader on the boot volume is not there. +//! - a directory no capability names is nobody's, and nothing under it is +//! there. use std::fs; use std::io::ErrorKind; @@ -87,11 +87,14 @@ fn main() { println!("fs_escape: {:?} on the wire refused", core::str::from_utf8(rel).unwrap_or("")); } - // ROOT's own `/boot`, the empty directory a view is mounted over, is all - // a program without the capability names there. - let listed: Vec<_> = fs::read_dir("/boot").expect("ROOT's /boot").collect(); - assert!(listed.is_empty(), "/boot listed {} entries for a program whose row holds no slots", listed.len()); - refused("/boot/EFI/BOOT/BOOTX64.EFI", ErrorKind::NotFound); + // A directory this program holds no capability for is not a file + // server's: the name is refused by the namespace, and the path goes to + // ROOT, which has no such directory. + assert!( + toyos::endow::namespace().expect("a namespace").open("fs:/fs_escape_nobodys").is_err(), + "a capability this program was never given answered" + ); + refused("/fs_escape_nobodys/secret", ErrorKind::NotFound); for link in ["/home/fs_escape_out", "/home/fs_escape/deeper/out", "/home/fs_escape/deeper/up", "/home/fs_escape/deeper/back"] { let _ = fs::remove_file(link); diff --git a/tests/toyos-rust-tests/src/bin/ftruncate_flush_race.rs b/tests/toyos-rust-tests/src/bin/ftruncate_flush_race.rs deleted file mode 100644 index 707daa26078..00000000000 --- a/tests/toyos-rust-tests/src/bin/ftruncate_flush_race.rs +++ /dev/null @@ -1,59 +0,0 @@ -//! A truncate raced against a flush's stalled `update_metadata` window. -//! -//! `SYS_FTRUNCATE`'s resize once took no VFS lock, so a truncate could land -//! between a flush's two steps and record the older size. `ftruncate-flush-stall` -//! holds every flush of this file open for 400ms; this binary races a truncate -//! against it. Looped, since the stall spins preemption-off: one sleep can -//! overshoot into a lock-free gap, and a lockless resize can never make a -//! contended attempt — the loop is one-sided. - -use std::fs::OpenOptions; -use std::io::{Seek, SeekFrom, Write}; -use std::thread; -use std::time::{Duration, Instant}; - -/// Mirrored in `tests/common/volumes.rs::ftruncate_flush_race`, and in the -/// actuator's own path filter (`kernel/src/vfs.rs::stalled_metadata_window`). -const PATH: &str = "/log/truncate-race.bin"; -const FULL: usize = 3 * 4096; -const SHORT: u64 = 5000; - -const INTO_WINDOW: Duration = Duration::from_millis(50); -/// Only a serialised truncate waits this long against the 400ms stall; a lockless one returns in microseconds. -const CONTENDED: Duration = Duration::from_millis(150); -const ATTEMPTS: u32 = 10; - -fn main() { - let mut f = OpenOptions::new().create(true).write(true).open(PATH).expect("create"); - - let mut contended = None; - for attempt in 0..ATTEMPTS { - f.seek(SeekFrom::Start(0)).expect("rewind"); - f.write_all(&vec![0xB6u8; FULL]).expect("fill"); - - let flusher = { - let f = f.try_clone().expect("clone handle"); - thread::spawn(move || f.sync_all().expect("the stalled fsync")) - }; - thread::sleep(INTO_WINDOW); - let began = Instant::now(); - f.set_len(SHORT).expect("truncate"); - let waited = began.elapsed(); - flusher.join().expect("flusher panicked"); - - if waited >= CONTENDED { - contended = Some((attempt, waited)); - break; - } - } - let Some((attempt, waited)) = contended else { - panic!( - "in {ATTEMPTS} attempts the truncate never once waited for the stalled flush — \ - the resize does not serialise with the metadata window", - ); - }; - - // The truncated size durable before the host reads the shut-down volume. - f.sync_all().expect("the settling fsync"); - println!("attempt {attempt}: the truncate waited {waited:?} for the stalled flush; {SHORT} bytes settled"); -} diff --git a/tests/toyos.rs b/tests/toyos.rs index dd4621dc078..8012d598c44 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -432,9 +432,9 @@ const RUST_SKIP: &[&str] = &[ // measure. // // `writeback_reopen` and `writeback_spawn` each need their own boot with - // `writeback-stall` armed; `writeback_durability` writes `/log` and is judged - // host-side off the image after a shutdown. All three run as `MACHINE_TESTS`, - // not on the shared boot. + // `writeback-stall` armed; `writeback_durability` writes `/log` and + // `kernel_log_file` judges it host-side off the image after a shutdown. + // All three run in `MACHINE_TESTS`, not on the shared boot. "writeback_reopen", "writeback_spawn", "writeback_durability", @@ -480,9 +480,6 @@ const RUST_SKIP: &[&str] = &[ // Needs `test-small-caches` for the eviction its read-back rests on, and a // boot of its own for the host-side re-read. `redirty_mid_flush` runs it. "redirty_mid_flush", - // Needs `ftruncate-flush-stall` and a boot of its own for the host-side - // re-read. `ftruncate_flush_race` runs it. - "ftruncate_flush_race", // Needs the `smp-skip-ap` boot; `smp_failed_ap_leaves_no_hole` runs it there. "smp_hole_shootdown", // Its listings are exact against `tests/layoutcase`, and it takes what that @@ -1495,13 +1492,12 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("root_named_twice_on_the_boot_disk", Sched::Serial, Tier::Fast), ("root_named_twice", Sched::Serial, Tier::Nightly), ("log_partition_identity", Sched::Parallel, Tier::Nightly), - // The write-back queue's three negative controls (wall 4 of + // The write-back queue's two negative controls (wall 4 of // `issues/kernel/every-wait-in-this-kernel-is-a-spin.md`). `writeback_reopen` // and `writeback_spawn` arm `writeback-stall`, so each needs its own actuator // boot: one holds the queue open across a *handle* re-open, which the file // cache answers, and the other across a *spawn*, which is a device view and - // does not. `writeback_durability` is a host-side volume oracle that shuts the - // guest down and reads `/log` back with `toyos-fat32-check`. + // does not. // The watch's lost-wake window, staged: `watch-window` holds every pipe // waiter between reading its condition and parking, so the peer's post lands where // only the notified bit carries it to the commit. @@ -1512,13 +1508,11 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("user_copy_races_munmap", Sched::Parallel, Tier::Fast), ("writeback_reopen", Sched::Parallel, Tier::Fast), ("writeback_spawn", Sched::Parallel, Tier::Nightly), - ("writeback_durability", Sched::Parallel, Tier::Nightly), // `KernelHw::switch`'s SS reload (AMD `X86_BUG_SYSRET_SS_ATTRS`) observed the // one way a guest can, since its `SYSRET` does not reproduce the erratum. Reds // the day that `mov ss` leaves the switch. ("sysret_ss_reload", Sched::Parallel, Tier::Fast), - // The FAT32 read side's revocation gate, and a host-side volume oracle for - // the same reason `writeback_durability` is one: whether the clusters the + // The FAT32 read side's revocation gate, and a host-side volume oracle: whether the clusters the // unlink freed were really reissued, and whether the cycle left a volume, are // both questions the guest that staged them cannot answer about itself. ("fat_backing_revoked", Sched::Parallel, Tier::Nightly), @@ -1528,7 +1522,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("fsync_failed_commit", Sched::Parallel, Tier::Nightly), ("redirty_mid_flush", Sched::Parallel, Tier::Nightly), // A truncate staged inside a flush's metadata window, re-read off the image. - ("ftruncate_flush_race", Sched::Parallel, Tier::Nightly), // The rename gate's FAT arm, a host-side volume oracle like `fat_backing_revoked`. ("fs_rename_durable", Sched::Parallel, Tier::Nightly), // The directory work's FAT arm, `fs_rename_durable`'s oracle shape. @@ -1721,9 +1714,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("fs_dirs_durable", &["test_rs_fs_dirs_durable"]), ("fs_rename_durable", &["test_rs_fs_rename_durable", "test_rs_fs_dirs_durable"]), ("fsync_failed_commit", &["test_rs_fsync_flush_failed"]), - ("ftruncate_flush_race", &["test_rs_ftruncate_flush_race", "test_rs_fs_rename_durable"]), ("redirty_mid_flush", &["test_rs_redirty_mid_flush"]), - ("writeback_durability", &["test_rs_writeback_durability", "test_rs_fat_backing_revoked"]), ("kernel_log_file", &["test_rs_writeback_durability"]), ("double_fault_stack", &["test_rs_test_panic_child"]), ("idle_stack_guard", &["test_rs_test_panic_child"]), @@ -11490,7 +11481,6 @@ fn run_machine_test( "kernel_log_file" => common::volumes::kernel_log_file(test_config, c_bins, rust_bins), // Body in `tests/common/volumes.rs`, same reason: the host-side oracle // shuts the guest down and reads `/log` back with `toyos-fat32-check`. - "writeback_durability" => common::volumes::writeback_durability(test_config, c_bins, rust_bins), // Same again: the FAT32 read side's revocation, judged off the volume the // guest's unlink-and-reallocate cycle left behind. "fat_backing_revoked" => common::volumes::fat_backing_revoked(test_config, c_bins, rust_bins), @@ -11515,7 +11505,6 @@ fn run_machine_test( } "fsync_failed_commit" => common::volumes::fsync_failed_commit(test_config, c_bins, rust_bins), "redirty_mid_flush" => common::volumes::redirty_mid_flush(test_config, c_bins, rust_bins), - "ftruncate_flush_race" => common::volumes::ftruncate_flush_race(test_config, c_bins, rust_bins), "fs_rename_durable" => common::volumes::fs_rename_durable(test_config, c_bins, rust_bins), "fs_dirs_durable" => common::volumes::fs_dirs_durable(test_config, c_bins, rust_bins), // The lost-wake canary with the window it guards held open: every pipe diff --git a/toyos-abi/Cargo.toml b/toyos-abi/Cargo.toml index 153716fec8a..09ed8733c04 100644 --- a/toyos-abi/Cargo.toml +++ b/toyos-abi/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" edition = "2021" license = "MIT OR Apache-2.0" description = "The ToyOS kernel ABI: syscall numbers, struct layouts and typed syscall wrappers." diff --git a/toyos/Cargo.toml b/toyos/Cargo.toml index ffd3ccfb7e4..0faf294ae93 100644 --- a/toyos/Cargo.toml +++ b/toyos/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "toyos" -version = "0.18.0" +version = "0.19.0" edition = "2021" license = "MIT OR Apache-2.0" description = "The ToyOS userland SDK: typed handles, IPC, ports, namespaces, surfaces and net." @@ -28,4 +28,4 @@ unexpected_cfgs = { level = "warn", check-cfg = ['cfg(target_os, values("toyos") [dependencies] core = { version = "1.0.0", optional = true, package = "rustc-std-workspace-core" } -toyos-abi = { path = "../toyos-abi", version = "0.16.0", default-features = false } +toyos-abi = { path = "../toyos-abi", version = "0.17.0", default-features = false } diff --git a/userland/fsd/src/data.rs b/userland/fsd/src/data.rs index acb34bcb9d5..80a9488b9af 100644 --- a/userland/fsd/src/data.rs +++ b/userland/fsd/src/data.rs @@ -581,10 +581,12 @@ impl Volume for DataVolume { None if self.is_dir(path) => return Err(SyscallError::InvalidArgument), None => return Err(SyscallError::NotFound), } - self.orphan(path); + // Asked of the volume first: a delete the device refused leaves every + // holder of the file its file. if !mapped("unlink", path, self.fs.delete(path))? { return Err(SyscallError::NotFound); } + self.orphan(path); self.names.remove(path); Ok(()) } diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index 3be2de02fd5..ec242432455 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -332,8 +332,11 @@ impl Volume for FatVolume { if self.fs.metadata(path).map_err(|e| logged("metadata", path, e))?.is_dir { return Err(SyscallError::InvalidArgument); } + // Asked of the volume first: a delete the device refused leaves every + // holder of the file its file. + self.fs.remove(path).map_err(|e| logged("unlink", path, e))?; self.orphan(path); - self.fs.remove(path).map_err(|e| logged("unlink", path, e)) + Ok(()) } fn rename(&mut self, from: &str, to: &str) -> Result<(), SyscallError> { diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index 092ab980372..eb060f7eae1 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -1958,18 +1958,10 @@ fn build_namespace( /// **Every program sees the whole tree the file servers serve**, which is the /// kernel's old view kept whole until each row declares its own /// (`issues/isolation/every-program-sees-only-the-files-it-was-given.md`, -/// stage 2), with two exceptions: `/boot` is the updater's alone, and a -/// storage row sees none, since a file server resolving a path of its own -/// through itself waits for ever. +/// stage 2), with one exception: a storage row sees none, since a file server +/// resolving a path of its own through itself waits for ever. fn in_view(program: &Program, name: &str) -> bool { - if is_storage(program) { - return false; - } - match name.strip_prefix(CAPABILITY_PREFIX) { - Some("/boot") => program.slots, - Some(_) => true, - None => false, - } + !is_storage(program) && name.starts_with(CAPABILITY_PREFIX) } /// [`toyos_swap::PORT`] in a namespace of its own, for a program whose row diff --git a/userland/metalprobe/src/usb.rs b/userland/metalprobe/src/usb.rs index fe0c515671e..7e802294c9c 100644 --- a/userland/metalprobe/src/usb.rs +++ b/userland/metalprobe/src/usb.rs @@ -82,10 +82,9 @@ pub fn write() -> Measured { /// Read [`BYTES`] back off the device and check them, timing only the read. /// -/// **The staged file is closed and left to drain before the clock starts.** -/// `iod` flushes what the close pinned and drops the file from the cache, so -/// the timed read is a cache miss the stick has to answer — the same reason -/// `writeback_durability` waits here. +/// **The staged file is closed and left to drain before the clock starts**, +/// though the read is answered from the file server's cache +/// (`issues/hardware/metalprobes-usb-read-is-answered-from-fsds-cache.md`). pub fn read() -> Measured { let blob = payload(); { diff --git a/userland/toyos-window/Cargo.toml b/userland/toyos-window/Cargo.toml index 050be016156..b0da9a17a8f 100644 --- a/userland/toyos-window/Cargo.toml +++ b/userland/toyos-window/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "toyos-window" -version = "0.20.0" +version = "0.21.0" edition = "2021" license = "MIT OR Apache-2.0" description = "Client library for ToyOS windows: create one, draw into it, read its events." @@ -8,9 +8,9 @@ repository = "https://github.com/ToyOSOrg/ToyOS" [dependencies] toyos-font = { path = "../toyos-font", version = "0.2.0" } -toyos-abi = { path = "../../toyos-abi", version = "0.16.0" } +toyos-abi = { path = "../../toyos-abi", version = "0.17.0" } toyos-keymap = { path = "../../toyos-keymap", version = "0.1.0" } -toyos = { path = "../../toyos", version = "0.18.0" } +toyos = { path = "../../toyos", version = "0.19.0" } # `window` is taken on crates.io, so the package is `toyos-window`; the lib # target keeps the name every caller spells, `toyos/src`'s citations included. From 2aa73f12f2cb057a3ff78bc16ead14431917504d Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 12:39:05 +0200 Subject: [PATCH 05/54] storage: the console and diag boots get their file servers; the locale wizard its /config Both boots had no fsd row, so /config was nobody's and the locale applet was refused its write; locale_gate's wizard namespace now keeps fs:/config beside the surface. The lockfiles take the SDK's new minors. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- Cargo.lock | 2 +- bootloader/Cargo.lock | 2 +- console/system.toml | 19 ++++++++++++++++++- diag/system.toml | 19 ++++++++++++++++++- kernel/Cargo.lock | 2 +- tests/ssh-client-host/Cargo.lock | 2 +- tests/toyos-rust-tests/Cargo.lock | 6 +++--- tests/toyos-rust-tests/src/bin/locale_gate.rs | 5 ++++- userland/Cargo.lock | 6 +++--- userland/libc/Cargo.lock | 4 ++-- 10 files changed, 52 insertions(+), 15 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 802affd915b..a1faeb3a021 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1164,7 +1164,7 @@ checksum = "5d99f8c9a7727884afe522e9bd5edbfc91a3312b36a77b5fb8926e4c31a41801" [[package]] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" dependencies = [ "rustc-std-workspace-core", ] diff --git a/bootloader/Cargo.lock b/bootloader/Cargo.lock index f7dfd89b21e..13676ce84e4 100644 --- a/bootloader/Cargo.lock +++ b/bootloader/Cargo.lock @@ -257,7 +257,7 @@ dependencies = [ [[package]] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" [[package]] name = "toyos-acpi" diff --git a/console/system.toml b/console/system.toml index b1c697fb9f0..a0b96aa1f17 100644 --- a/console/system.toml +++ b/console/system.toml @@ -16,7 +16,7 @@ assets = ["assets"] [boot] -start = ["logd", "console"] +start = ["logd", "blockd", "fsd", "console"] # `/system/bin/console` owns the screen and provides the surface its shell and the # shell's children inherit. No compositor here, so nothing receives one. @@ -64,3 +64,20 @@ syscap = ["roster"] "bin/rm" = "/system/bin/toybox" "bin/shutdown" = "/system/bin/toybox" "bin/stats" = "/system/bin/toybox" + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/diag/system.toml b/diag/system.toml index 7892237af94..579bfd1b170 100644 --- a/diag/system.toml +++ b/diag/system.toml @@ -26,7 +26,7 @@ # # The other program it starts prints the cwd and exits. [boot] -start = ["logd", "toybox"] +start = ["logd", "blockd", "fsd", "toybox"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and the console and writes no @@ -41,3 +41,20 @@ syscap = ["logread"] [programs.toybox] args = ["pwd"] + +# The block service: the NVMe controller this machine's DATA is on, driven +# from userland through its claim. Started again when it ends; the claim goes +# back with the process and is minted again for the next. +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +# The file servers: DATA, the log and the running slot's volume, one process +# each, serving every program the directories of its role. Started again when +# one ends, on the same ports. +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] diff --git a/kernel/Cargo.lock b/kernel/Cargo.lock index e6d0370abcf..ea172bd3889 100644 --- a/kernel/Cargo.lock +++ b/kernel/Cargo.lock @@ -77,7 +77,7 @@ checksum = "b50b8869d9fc858ce7266cce0194bd74df58b9d0e3f6df3a9fc8eb470d95c09d" [[package]] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" [[package]] name = "toyos-acpi" diff --git a/tests/ssh-client-host/Cargo.lock b/tests/ssh-client-host/Cargo.lock index d8eb51ea815..dd269501945 100644 --- a/tests/ssh-client-host/Cargo.lock +++ b/tests/ssh-client-host/Cargo.lock @@ -1848,7 +1848,7 @@ dependencies = [ [[package]] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" [[package]] name = "toyos-manifest" diff --git a/tests/toyos-rust-tests/Cargo.lock b/tests/toyos-rust-tests/Cargo.lock index bc2b2c2b2b1..78a9ecbdb60 100644 --- a/tests/toyos-rust-tests/Cargo.lock +++ b/tests/toyos-rust-tests/Cargo.lock @@ -2084,14 +2084,14 @@ dependencies = [ [[package]] name = "toyos" -version = "0.18.0" +version = "0.19.0" dependencies = [ "toyos-abi", ] [[package]] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" [[package]] name = "toyos-blockhold" @@ -2180,7 +2180,7 @@ version = "0.1.0" [[package]] name = "toyos-window" -version = "0.20.0" +version = "0.21.0" dependencies = [ "toyos", "toyos-abi", diff --git a/tests/toyos-rust-tests/src/bin/locale_gate.rs b/tests/toyos-rust-tests/src/bin/locale_gate.rs index 3f0b3a9b207..673d752624c 100644 --- a/tests/toyos-rust-tests/src/bin/locale_gate.rs +++ b/tests/toyos-rust-tests/src/bin/locale_gate.rs @@ -128,8 +128,11 @@ impl Surface { } fn spawn_locale(args: &[&str], surface: &Connector) -> Child { - // The whole of what the wizard is given: one connector, to this surface. + // The whole of what the wizard is given: one connector, to this surface, + // and the directory its layout is written to. + let names = toyos::endow::namespace().expect("locale_gate: this program was endowed a namespace"); let child_ns = namespace::build() + .keep(names, &["fs:/config"]) .add(surface::SERVICE, surface) .finish() .expect("locale_gate: the kernel refused a namespace for the wizard"); diff --git a/userland/Cargo.lock b/userland/Cargo.lock index 3bab68bc8c8..cead428683f 100644 --- a/userland/Cargo.lock +++ b/userland/Cargo.lock @@ -4059,14 +4059,14 @@ dependencies = [ [[package]] name = "toyos" -version = "0.18.0" +version = "0.19.0" dependencies = [ "toyos-abi", ] [[package]] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" [[package]] name = "toyos-blockhold" @@ -4183,7 +4183,7 @@ version = "0.1.0" [[package]] name = "toyos-window" -version = "0.20.0" +version = "0.21.0" dependencies = [ "toyos", "toyos-abi", diff --git a/userland/libc/Cargo.lock b/userland/libc/Cargo.lock index df18e043698..376e2955eaf 100644 --- a/userland/libc/Cargo.lock +++ b/userland/libc/Cargo.lock @@ -27,14 +27,14 @@ checksum = "b5b646652bf6661599e1da8901b3b9522896f01e736bad5f723fe7a3a27f899d" [[package]] name = "toyos" -version = "0.18.0" +version = "0.19.0" dependencies = [ "toyos-abi", ] [[package]] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" [[package]] name = "toyos-libc" From 334fbd8bdad81db9714ba7a396cfa43e747dbeb6 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 12:45:23 +0200 Subject: [PATCH 06/54] sdk: move toyos/ and iced-counter's lockfiles onto the bumped versions The ABI bump declared toyos 0.19.0, toyos-abi 0.17.0 and toyos-window 0.21.0, but toyos/Cargo.lock and tests/iced-counter/Cargo.lock still locked 0.18.0, 0.16.0 and 0.20.0, and `abi-split` refused them (run 36313219131). Moved with `cargo update --offline -p ...` against each lockfile's own manifest; every other tracked lockfile already held the new versions. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- tests/iced-counter/Cargo.lock | 6 +++--- toyos/Cargo.lock | 4 ++-- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/tests/iced-counter/Cargo.lock b/tests/iced-counter/Cargo.lock index 3c8ef1a1859..838c2389d5d 100644 --- a/tests/iced-counter/Cargo.lock +++ b/tests/iced-counter/Cargo.lock @@ -2957,14 +2957,14 @@ dependencies = [ [[package]] name = "toyos" -version = "0.18.0" +version = "0.19.0" dependencies = [ "toyos-abi", ] [[package]] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" [[package]] name = "toyos-font" @@ -2976,7 +2976,7 @@ version = "0.1.0" [[package]] name = "toyos-window" -version = "0.20.0" +version = "0.21.0" dependencies = [ "toyos", "toyos-abi", diff --git a/toyos/Cargo.lock b/toyos/Cargo.lock index c38f92db341..705625672b4 100644 --- a/toyos/Cargo.lock +++ b/toyos/Cargo.lock @@ -10,7 +10,7 @@ checksum = "aa9c45b374136f52f2d6311062c7146bff20fec063c3f5d46a410bd937746955" [[package]] name = "toyos" -version = "0.18.0" +version = "0.19.0" dependencies = [ "rustc-std-workspace-core", "toyos-abi", @@ -18,7 +18,7 @@ dependencies = [ [[package]] name = "toyos-abi" -version = "0.16.0" +version = "0.17.0" dependencies = [ "rustc-std-workspace-core", ] From 3c2db98f39cb5cdfb4e201c934e0a0ea42998738 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 12:59:04 +0200 Subject: [PATCH 07/54] storage: toyos-build's own gates on the file servers' configs `cargo test --lib` in the root had 20 reds at 334fbd8b, all this branch's, and no gate run on the branch had reached them: - `no_diag_program_claims_the_screen`: 2aa73f12 gave the diag boot blockd, which claims the NVMe controller, and the diag image's whole guarantee is a config that declares no `devices`. The diag boot now runs fsd for the log role alone: the log partition is the kernel's partition claim on the boot stick, so logd keeps its /log and nothing in the image claims a device. screen_diag_boot and screen_log_absent EXIT=0 after it. - `every_shipped_boot_config_is_covered`: tests/fsdrestartcase joins ALL_CONFIGS, and with it every per-config gate that list drives. - the 18 `heartbeat::tests`: tests/metalcase now starts blockd and fsd, and DONE had no done line for either. DONE carries one line per process a row starts: blockd's `NVMe up`, and fsd's ` serving` for each of its three roles. The recorded nightly captures were booted without either program, so the tests replay them against that start list's lines, and the table is held against metalcase's own start list in `a_program_the_done_table_does_not_know_is_refused`. kernel_heartbeat EXIT=0 on a metalcase boot after it. `cargo test --lib`: 396 passed, EXIT=0. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- diag/system.toml | 20 +++++------------ src/build.rs | 1 + src/heartbeat.rs | 56 +++++++++++++++++++++++++++++++----------------- 3 files changed, 42 insertions(+), 35 deletions(-) diff --git a/diag/system.toml b/diag/system.toml index 579bfd1b170..3888d960e60 100644 --- a/diag/system.toml +++ b/diag/system.toml @@ -26,7 +26,7 @@ # # The other program it starts prints the cwd and exits. [boot] -start = ["logd", "blockd", "fsd", "toybox"] +start = ["logd", "fsd", "toybox"] # **Every image that carries a `TOYOS-LOG` partition runs this**, and every # image does. The kernel keeps the record ring and the console and writes no @@ -42,19 +42,9 @@ syscap = ["logread"] [programs.toybox] args = ["pwd"] -# The block service: the NVMe controller this machine's DATA is on, driven -# from userland through its claim. Started again when it ends; the claim goes -# back with the process and is minted again for the next. -[programs.blockd] -service = true -restart = true -serves = ["block"] -devices = ["pci:1b36:0010"] - -# The file servers: DATA, the log and the running slot's volume, one process -# each, serving every program the directories of its role. Started again when -# one ends, on the same ports. +# `/log`'s file server, and only it: the log partition is the kernel's +# partition claim on the boot stick, so this row declares no `devices`, and +# DATA's block service would be one. [programs.fsd] restart = true -roles = ["data", "log", "boot"] -receives = ["block"] +roles = ["log"] diff --git a/src/build.rs b/src/build.rs index 71ba8f21192..1c6fbaf7a34 100644 --- a/src/build.rs +++ b/src/build.rs @@ -3079,6 +3079,7 @@ mod tests { "tests/e1000leasecase/system.toml", "tests/e1000talkcase/system.toml", "tests/flrswapcase/system.toml", + "tests/fsdrestartcase/system.toml", "tests/inspectcase/system.toml", "tests/jobcase/system.toml", "tests/jobdeadlinecase/system.toml", diff --git a/src/heartbeat.rs b/src/heartbeat.rs index eddfe9d099d..d0af64fb462 100644 --- a/src/heartbeat.rs +++ b/src/heartbeat.rs @@ -120,19 +120,25 @@ fn millis(field: &str) -> Option { Some(s.parse::().ok()? * 1000 + ms.parse::().ok()?) } -/// `tests/metalcase`'s `[boot] start` programs and the line each says it has -/// finished starting with. The one table: [`done_lines`] holds it against the -/// config, and a caller reads it through that rather than declaring its own. -const DONE: &[(&str, &str)] = &[ - ("logd", "logd: this boot's kernel log is"), - ("compositor", "compositor: ready"), - ("soundd", "soundd: null sink idle"), - ("netd", "exit: netd pid="), - ("sshd", "exit: sshd pid="), - ("test-runner", "===READY==="), +/// `tests/metalcase`'s `[boot] start` programs and the lines each says it has +/// finished starting with — one per process a row starts. The one table: +/// [`done_lines`] holds it against the config, and a caller reads it through +/// that rather than declaring its own. +const DONE: &[(&str, &[&str])] = &[ + ("logd", &["logd: this boot's kernel log is"]), + // Said before the partition table is read, which is done before the DATA + // server's own line can be: that server opens its partition through blockd. + ("blockd", &["blockd: NVMe up"]), + // One process per role in the row. + ("fsd", &["fsd: Data serving", "fsd: Log serving", "fsd: Boot serving"]), + ("compositor", &["compositor: ready"]), + ("soundd", &["soundd: null sink idle"]), + ("netd", &["exit: netd pid="]), + ("sshd", &["exit: sshd pid="]), + ("test-runner", &["===READY==="]), ]; -/// The done line of each program in `start`, or the disagreement between the +/// The done lines of the programs in `start`, or the disagreement between the /// config and [`DONE`] — a `[boot] start` program with no done line here leaves /// its own start-up inside the window, which is the one thing the window exists /// to exclude. @@ -141,11 +147,11 @@ pub fn done_lines(start: &[String]) -> Result, String> { let known: Vec<&str> = DONE.iter().map(|(program, _)| *program).collect(); if start != known { return Err(format!( - "`tests/metalcase` starts {start:?} and `src/heartbeat.rs` knows the done line of \ + "`tests/metalcase` starts {start:?} and `src/heartbeat.rs` knows the done lines of \ {known:?} — a program without one leaves its start-up inside the window" )); } - Ok(DONE.iter().map(|(_, line)| *line).collect()) + Ok(DONE.iter().flat_map(|(_, lines)| lines.iter().copied()).collect()) } /// Why a capture is not a claim about a settled, running machine's CPUs. @@ -249,12 +255,21 @@ mod tests { use super::*; - /// `tests/metalcase`'s done lines, as the test passes them. - fn started() -> Vec<&'static str> { - done_lines(&crate::build::boot_start( + /// `tests/metalcase`'s `[boot] start`, as the test reads it. + fn metalcase() -> Vec { + crate::build::boot_start( &Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/metalcase/system.toml"), - )) - .unwrap() + ) + } + + /// The done lines of the `[boot] start` the recorded captures below were + /// booted with, which started no block service and no file server. + fn started() -> Vec<&'static str> { + const RECORDED: &[&str] = &["logd", "compositor", "soundd", "netd", "sshd", "test-runner"]; + RECORDED + .iter() + .flat_map(|p| DONE.iter().find(|(q, _)| q == p).expect("a program DONE knows").1.iter().copied()) + .collect() } /// Nightly `35072262489`, guest shard 8, suite run: every heartbeat line and @@ -742,8 +757,9 @@ compositor: ready /// program it does not know is refused by name, here and not at the guest. #[test] fn a_program_the_done_table_does_not_know_is_refused() { - let known: Vec = DONE.iter().map(|(program, _)| (*program).to_string()).collect(); - assert_eq!(done_lines(&known).unwrap(), started()); + let known = metalcase(); + let every: Vec<&str> = DONE.iter().flat_map(|(_, lines)| lines.iter().copied()).collect(); + assert_eq!(done_lines(&known).unwrap(), every); let mut added = known.clone(); added.push("sniffer".to_string()); From 7e2e1043463467c09af15f1b88159b181e35e0f3 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 13:09:39 +0200 Subject: [PATCH 08/54] redlist: lan_mdns_answer's SUN_LEN shape, and where a served read goes lan_mdns_answer's two fast-tier failures are one defect of the dev host's scratch path, measured by an A/B at 16d2e645 against this branch, two runs per arm per $TMPDIR, one at a time: the default $TMPDIR reds on both with `path must be shorter than SUN_LEN` (a 104-byte socket path, which QEMU binds and the host's connect refuses); a $TMPDIR three bytes longer reds on both with QEMU refusing its own -chardev and exiting 1 before READY, the shape the fast tier showed on lane 11 (105 bytes); a $TMPDIR under target/ passes on both, with no IOMMU line from either guest. The IOMMU translation faults beside the fast-tier failure were other tests' deliberate foreign-DMA actuators interleaved into the shared log. The first shape is quarantined at the issue that owns it; the second cannot be, since its text is only the harness's boot-death framing. With the row, the default $TMPDIR run is XFAIL, EXIT=0. The cached-read issue now carries the measurement of where a request goes, fsd's READ arm timed from inside the server beside the kernel's /tmp in the same guest: a 104 us round trip against 2.6 us, and at 256 KiB the two volatile word copies through the window at ~87%. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- ...t-path-outgrows-sun-len-on-the-dev-host.md | 29 +++++++++-- ...-server-costs-fifteen-times-the-kernels.md | 49 ++++++++++++++----- src/redlist.rs | 5 ++ 3 files changed, 67 insertions(+), 16 deletions(-) diff --git a/issues/build/a-lane-s-tap-socket-path-outgrows-sun-len-on-the-dev-host.md b/issues/build/a-lane-s-tap-socket-path-outgrows-sun-len-on-the-dev-host.md index 230a3d13bbe..2958bb39462 100644 --- a/issues/build/a-lane-s-tap-socket-path-outgrows-sun-len-on-the-dev-host.md +++ b/issues/build/a-lane-s-tap-socket-path-outgrows-sun-len-on-the-dev-host.md @@ -16,8 +16,31 @@ directory sits under `$TMPDIR`, so the path is 104 bytes, past the 103 a Seen in the fast tier twice in one session: at `origin/main` checked out in the `toyos-guiplat` worktree (alone with this message, wide as `QEMU died before ===READY===`), and at PR #528's head after it merged `e48604c0` (this -message wide and alone). `cargo run --- --known-red lan_mdns_answer` answers NO. +message wide and alone). + +## Two shapes, one length + +`$TMPDIR/toyos-tmp--/tests-0/lane-/tap-{in,out}-.sock` is +104 bytes with a five-digit pid and a one-digit lane, and 105 with a +two-digit lane. At 104 QEMU binds the socket and the host's +`UnixStream::connect` refuses it (`path must be shorter than SUN_LEN`); at +105 QEMU refuses its own `-chardev` (`UNIX socket path … is too long / Path +must be less than 104 bytes`) and exits 1 with nothing on the console or +the UART, which the harness reports as `QEMU died before ===READY=== +(status: … 256)`. Those two numbers are the fast-tier lines of PR #536's +logs, lane-5 and lane-11. + +An A/B at `16d2e645` against PR #536, two runs per arm per `$TMPDIR`: +the default reds with the first shape on both, a `$TMPDIR` three bytes +longer reds with the second on both, and a `$TMPDIR` under the worktree's +`target/` passes on both. + +The first shape is quarantined in `src/redlist.rs`. The second is not +and cannot be: its failure text is the harness's boot-death framing and +nothing of this defect, so a row quoting it would excuse every guest +death of this test. A wide run that lands the test on lane 10 or 11 on +this host reds the suite. **Exit**: the socket paths fit a `sockaddr_un` wherever the scratch -directory is, with `lan_mdns_answer` green on this host. +directory is, with `lan_mdns_answer` green on this host and its row gone +from `src/redlist.rs`. diff --git a/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md b/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md index 823dd3cbd5c..2e755a85524 100644 --- a/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md +++ b/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md @@ -17,16 +17,39 @@ three interleaved runs per arm, the branch that moved DATA to | read on the next boot | 84 MiB/s | 41–42 MiB/s | | 4 KiB create and fsync, p50 | 1.0–1.3 ms | 3.5–3.6 ms | -The cached read by request size, one run: 47.5 MiB/s at 8 KiB, 91.8 MiB/s at -256 KiB, 182.6 MiB/s at 2 MiB — so about 160 µs a request and about 5 ns a -byte. A byte is copied three times, after a zeroing: out of the block cache -into a buffer the server zeroed (the `READ` arm of `userland/fsd/src/main.rs`, -through `DataVolume::read`), into the client's window a `u64` at a time -with a volatile store each (`toyos::fs::window_put`, -`toyos::volatile::Window::copy_in`), and out of the window a `u64` at a time -again (`window_take`), where the kernel's page cache copies once. - -**Exit**: a cached read within a small factor of the kernel's at the same -request size — the block copied once into the window, and the window's word -loops replaced by a copy the compiler may widen — measured by the same -three-run A/B. +## Where a request goes + +One guest, the file in fsd's block cache (64 MiB of clean blocks, so all of +it) and the same bytes in the kernel's `/tmp`, three interleaved passes; fsd's +`READ` arm timed from inside the server: + +| per request | 4 KiB | 256 KiB | 2 MiB | +|---|---|---|---| +| kernel `/tmp` | 4.2–4.5 µs | 98–100 µs | 751–771 µs | +| fsd, at the client | 115–117 µs | 1906–1922 µs | 4603–4727 µs | +| of which the server's `vec![0; len]` | 3.4 µs | 47–49 µs | 363–490 µs | +| the block cache into it (`DataVolume::read`) | 6.2–6.4 µs | 108–115 µs | 663–665 µs | +| `window_put` into the client's window | 3.7 µs | 790–804 µs | 1794 µs | +| the gap to the next request: reply, the client's `window_take`, the next send | 102–103 µs | 973–989 µs | 1986–1988 µs | + +A request that moves no bytes (a one-byte read at the end of the file) costs +104 µs against the kernel's 2.6 µs: that is the round trip, and it is the +whole of a small read. At 256 KiB the two word-at-a-time volatile copies +through the window — `toyos::fs::window_put` in the server and `window_take` +in the client, 2.9–3.5 ns a byte each at 64 KiB to 1 MiB — are about 87% of +the request; the zeroing and the cache copy are 8%, and the round trip 5%. +The same two loops ran at 0.86–0.90 ns a byte on 2 MiB requests and at +0.69–3.17 ns a byte over local memory: under TCG their price per byte depends +on the addresses, a mechanism not isolated here. The kernel's copy is one, at +0.37 ns a byte. + +Not the cause: the client holds no cache, but the server's cache answers +every block (the `DataVolume::read` row); and nothing contends for a lock — +the reader is one thread and fsd is one thread over a `RefCell`. + +The round trip is what a served read is and does not go; the copies are a +defect. Timings are TCG's: a verdict on their size belongs on metal. + +**Exit**: a 256 KiB cached read within a small factor of the kernel's in the +same guest, the block copied once into the window and the window's copies no +longer a scalar volatile loop, measured by the same interleaved runs. diff --git a/src/redlist.rs b/src/redlist.rs index edc5ca2554a..27f970c681f 100644 --- a/src/redlist.rs +++ b/src/redlist.rs @@ -84,6 +84,11 @@ pub const QUARANTINE: &[Quarantined] = &[ ], issue: "issues/kernel/deferred-release-outlives-its-syscall.md", }, + Quarantined { + test: "lan_mdns_answer", + says: &["path must be shorter than SUN_LEN"], + issue: "issues/build/a-lane-s-tap-socket-path-outgrows-sun-len-on-the-dev-host.md", + }, Quarantined { test: "latency_wake", says: &["the p99 landed in the histogram's last bucket"], From b534a32d278da08f172f1f0ac97274771c1b61ba Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 13:50:12 +0200 Subject: [PATCH 09/54] fsd: a refused rename-over leaves the destination's holders their file `DataVolume::rename` marked every holder of `to` gone before asking the format for the rename, so a rename the format refused (EntryTooLarge, NoSpace, a device error) had already dropped `to`'s unsynced length and extents: `persist` skips a gone node, and `to` still named the file. The orphan now follows the format's answer, as fat.rs's rename and data.rs's own unlink already order it. Host test `a_refused_rename_over_an_open_file_keeps_its_unsynced_writes`: a rename refused EntryTooLarge (240 extents fit `from`'s short name and not `to`'s 300-byte one) leaves `to` readable and its length on the volume after the next sync. With the orphan moved back before the rename (a checked patch): EXIT=101, the read answers Gone. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- userland/fsd/src/data.rs | 33 ++++++++++++++++++++++++++++++++- 1 file changed, 32 insertions(+), 1 deletion(-) diff --git a/userland/fsd/src/data.rs b/userland/fsd/src/data.rs index 80a9488b9af..a3f5d209fe8 100644 --- a/userland/fsd/src/data.rs +++ b/userland/fsd/src/data.rs @@ -610,8 +610,10 @@ impl Volume for DataVolume { if let Some(node) = self.by_path.get(from).copied() { self.persist(node)?; } - self.orphan(to); + // Orphaned only once the format has taken the rename: a refused + // one leaves `to` naming its file, holders and unsynced length whole. mapped("rename", from, self.fs.rename(from, to))?; + self.orphan(to); self.names.remove(from); self.names.insert(to.to_string(), kind); if let Some(node) = self.by_path.remove(from) { @@ -807,4 +809,33 @@ mod tests { v.rename("home/b/g", "home/b/f").unwrap(); assert_eq!(v.lstat("home/b/f").unwrap().size, 0, "replaced"); } + + /// A rename over an open, unsynced file that the format refuses leaves + /// that file whole: readable through its holder, and its length reaching + /// the volume at the next sync. The refusal is `EntryTooLarge`: `from`'s + /// extents fit its own short name and not `to`'s long one. + #[test] + fn a_refused_rename_over_an_open_file_keeps_its_unsynced_writes() { + let mut v = vol(); + let from = v.open("home/f", CREATE).unwrap(); + let spacer = v.open("home/spacer", CREATE).unwrap(); + // Alternating allocations, so no two of `from`'s blocks are adjacent + // and each is an extent of its own. + for page in 0..240u64 { + v.write(from, page * BLOCK as u64, &[1; BLOCK]).unwrap(); + v.write(spacer, page * BLOCK as u64, &[2; BLOCK]).unwrap(); + } + let to_path = format!("home/{}", "t".repeat(300)); + let to = v.open(&to_path, CREATE).unwrap(); + v.write(to, 0, b"written and not synced").unwrap(); + + assert_eq!(v.rename("home/f", &to_path), Err(SyscallError::ResourceExhausted)); + + let mut back = [0u8; 22]; + assert_eq!(v.read(to, 0, &mut back), Ok(22), "the holder of `to` still reads it"); + assert_eq!(&back, b"written and not synced"); + v.sync().unwrap(); + v.close(to); + assert_eq!(v.lstat(&to_path).unwrap().size, 22, "`to`'s length reached the volume"); + } } From 10892fd1defe90ea80601063a5a4ea4c22400a61 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 14:40:57 +0200 Subject: [PATCH 10/54] storage: init restarts a file server while it waits on one, and fsd bounds what a client names The review of #536 at 7e2e1043, blockers 1, 2, 3, 5 and 7: - init made its file calls (the session home, a service's home, a launch's `/apps` package) from its one loop, which is also where a file server that ended is started again: a call a server's end left in the port's queue waited for ever. They run on one worker thread now (`Worker`), and `Init::files` waits on the call and on a service's end together, restarting while it waits; a server alive and silent costs `FILES_BOUND` (30 s, policy). fsd's `--end-at-hello ` actuator ends the server at the first hello on `dir` this boot; `tests/fsdrestartcase` arms it on `/apps`, and `fs_restart` launches an `/apps` path first: init's resolution is that hello, and the launch is answered. - `STREAM` refuses an offset past `MAX_FILE_BYTES` (`fs_stream_offset`), a drain ends its stream by name at that bound, and a drain takes one read per wake, `pump`'s fairness. Every other client number fsd keeps was already bounded where it is kept. - A handle held across a restart reopens by path and keeps the file only when the server answers the identity it last saw (`Stat::ident`, a hash of what the entry records: DATA's first block, length and mtime; FAT's short name, creation stamp, first cluster and length). `fs_restart` renames a file over a held one, ends the server, and the held handle's write is Gone; the host reads the replacement's bytes back untouched. - The window: one bounds-checked `copy_nonoverlapping` each way, never a reference over it, with who owns it when at the site; fsd's `READ` has the cache copy each block into the window once (`Out`, `Cache::visit`), and a `WRITE` lands in a kept scratch. - A client past `MAX_SERVED` is answered `ResourceExhausted` at its hello and let go by name; acceptors wait only on `MAX_HANDSHAKES`, each answered or let go within `HANDSHAKE_TIMEOUT` (`fs_client_bound`). Also: `Capability.writable` and `Dir::write`'s second generation check are gone; a blockd whose controller would not open answers `Unusable`, and fsd says DATA is absent rather than putting it in memory; `fs_cache_eviction` reads a file longer than the cache keeps back off the disk. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- rust | 2 +- tests/common/storage.rs | 33 +- tests/fsdrestartcase/system.toml | 19 +- .../src/bin/fs_cache_eviction.rs | 56 ++ .../src/bin/fs_client_bound.rs | 48 ++ tests/toyos-rust-tests/src/bin/fs_restart.rs | 50 +- .../src/bin/fs_stream_offset.rs | 74 ++ .../src/bin/writeback_durability.rs | 192 ------ toyos-blockring/src/wire.rs | 3 +- toyos-fat32/src/fs.rs | 7 + toyos/src/fs.rs | 90 +-- userland/blockd/src/main.rs | 21 +- userland/fsd/src/absent.rs | 8 +- userland/fsd/src/cache.rs | 15 +- userland/fsd/src/data.rs | 77 ++- userland/fsd/src/fat.rs | 37 +- userland/fsd/src/main.rs | 217 ++++-- userland/fsd/src/volume.rs | 53 +- userland/init/src/main.rs | 644 +++++++++++------- 19 files changed, 1044 insertions(+), 602 deletions(-) create mode 100644 tests/toyos-rust-tests/src/bin/fs_cache_eviction.rs create mode 100644 tests/toyos-rust-tests/src/bin/fs_client_bound.rs create mode 100644 tests/toyos-rust-tests/src/bin/fs_stream_offset.rs delete mode 100644 tests/toyos-rust-tests/src/bin/writeback_durability.rs diff --git a/rust b/rust index 20436ec20e9..3428b0096a3 160000 --- a/rust +++ b/rust @@ -1 +1 @@ -Subproject commit 20436ec20e945a8fb66434d3809b39983c0a89c3 +Subproject commit 3428b0096a32ef7320442220f17165b2f6f6405e diff --git a/tests/common/storage.rs b/tests/common/storage.rs index fb061c64fe0..28ae4ff1336 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -610,12 +610,15 @@ pub fn home_overwrite_reads_back( /// budget, its directories answer `Gone`. Judged off the device. /// /// `tests/fsdrestartcase` arms every file server to end under a write to -/// `/home/fsd_end` (`--end-on`), and `test_rs_fs_restart` ends DATA's four -/// times: the guest asserts what a client sees, init's and fsd's own lines -/// say who ended and who started again, and with the machine down the DATA -/// partition is read by this crate's own build of the `bcachefs` reader over -/// a plain seek-and-read of the image — nothing the guest executed. The -/// flushed file and the one flushed across the end hold their bytes there. +/// `/home/fsd_end` (`--end-on`) and at the first hello on `/apps` +/// (`--end-at-hello`), and `test_rs_fs_restart` ends DATA's four times, the +/// first under init's own resolution of a launch: the guest asserts what a +/// client sees, init's and fsd's own lines say who ended and who started +/// again, and with the machine down the DATA partition is read by this +/// crate's own build of the `bcachefs` reader over a plain seek-and-read of +/// the image — nothing the guest executed. The flushed file holds its bytes +/// there, and the file a held handle wrote across an end holds the bytes of +/// the file renamed over it and nothing that handle wrote after the next. pub fn fsd_restart( _test_config: &Path, c_bins: &[(String, Vec)], @@ -626,8 +629,7 @@ pub fn fsd_restart( const KEPT: &str = "home/fs_restart/kept"; const ACROSS: &str = "home/fs_restart/across"; const KEPT_LEN: usize = 64 * 1024 + 13; - const ACROSS_BYTES: &[u8] = b"written and flushed before the server ended; written through the same \ - handle after it came back"; + const ACROSS_BYTES: &[u8] = b"renamed over the file a handle held across the end"; let kept: Vec = (0..KEPT_LEN).map(|i| (i.wrapping_mul(37) ^ 0xC3) as u8).collect(); let config = super::compile::repo_root().join("tests/fsdrestartcase"); @@ -654,8 +656,12 @@ pub fn fsd_restart( } let console = super::serial::Serial::named("fsd_restart", log.as_str()); let ended = log.matches("fsd: --end-on: ending with a write done and unanswered").count(); - if ended != 4 { - return Err(format!("fsd said it ended {ended} times, not the guest's 4:\n{log}")); + if ended != 3 { + return Err(format!("fsd said it ended under a write {ended} times, not the guest's 3:\n{log}")); + } + let at_hello = log.matches("fsd: --end-at-hello: ending before /apps's first hello is answered").count(); + if at_hello != 1 { + return Err(format!("fsd said it ended at a hello {at_hello} times, not the launch's 1:\n{log}")); } let restarted = log.lines().filter(|l| l.contains("init: fsd data (pid ") && l.contains("ended; started again")).count(); if restarted != 3 { @@ -679,9 +685,10 @@ pub fn fsd_restart( } } eprintln!( - " [fsd] DATA's server ended four times under an unanswered write; started again three, \ - its clients reopened, the fourth closed /home to Gone; {KEPT} and {ACROSS} read back off \ - the image by the host's own bcachefs reader" + " [fsd] DATA's server ended four times, the first under init's resolution of a launch that \ + was answered; started again three, its clients reopened, a handle on a file renamed over \ + answered Gone, the fourth closed /home to Gone; {KEPT} and {ACROSS} read back off the \ + image by the host's own bcachefs reader" ); Ok(()) } diff --git a/tests/fsdrestartcase/system.toml b/tests/fsdrestartcase/system.toml index acdef491e87..643c7b41944 100644 --- a/tests/fsdrestartcase/system.toml +++ b/tests/fsdrestartcase/system.toml @@ -1,7 +1,8 @@ # The boot `fsd_restart` judges: the test estate's shape, with every file # server armed to end the moment it has taken a write through a file opened to -# append at `/home/fsd_end` and before it answers it — `test_rs_fs_restart` -# ends DATA's four times, and init restarts it three. +# append at `/home/fsd_end` and before it answers it, and at the first hello +# on `/apps` this boot — `test_rs_fs_restart` ends DATA's four times, the +# first through a launch init resolves, and init restarts it three. [boot] start = ["logd", "blockd", "fsd", "test-runner"] @@ -11,12 +12,15 @@ service = true syscap = ["logread"] # `power` because `run shutdown` asks init through the `power` connector, and -# the host reads the DATA partition back once the machine is down. +# the host reads the DATA partition back once the machine is down; `launcher` +# for the launch of an `/apps` path the test ends DATA's server under. [programs.test-runner] -receives = ["power"] +receives = ["power", "launcher"] syscap = ["dup", "logread", "power"] +# `power` for `run shutdown`, which the test-runner launches through init. [programs.toybox] +receives = ["power"] [symlinks] "bin/shutdown" = "/system/bin/toybox" @@ -32,10 +36,11 @@ devices = ["pci:1b36:0010"] # The file servers: DATA, the log and the running slot's volume, one process # each, serving every program the directories of its role. Started again when -# one ends, on the same ports. `--end-on` is the test's actuator: a path on -# the DATA volume, which neither other role's volume carries. +# one ends, on the same ports. `--end-on` and `--end-at-hello` are the test's +# actuators: a path on the DATA volume and a directory of DATA's, which +# neither other role's volume carries. [programs.fsd] restart = true roles = ["data", "log", "boot"] receives = ["block"] -args = ["--end-on", "home/fsd_end"] +args = ["--end-on", "home/fsd_end", "--end-at-hello", "/apps"] diff --git a/tests/toyos-rust-tests/src/bin/fs_cache_eviction.rs b/tests/toyos-rust-tests/src/bin/fs_cache_eviction.rs new file mode 100644 index 00000000000..4ed66ed8933 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/fs_cache_eviction.rs @@ -0,0 +1,56 @@ +//! What DATA's server evicted from its cache it reads back off its disk, byte +//! for byte. +//! +//! The file is longer than the clean blocks the server keeps: once it is +//! durable every block of it is clean, the oldest are let go, and reading it +//! back from the start is a round trip through the disk for every block the +//! cache no longer holds, each checked against what was written. + +use std::fs::{self, File}; +use std::io::{Read, Write}; + +/// Mirrored from `userland/fsd/src/cache.rs`, in 4 KiB blocks. +const CLEAN_LIMIT: usize = 16 * 1024; + +const PATH: &str = "/home/fs_cache_eviction.bin"; +const CHUNK: usize = 256 * 1024; +/// 8 MiB past what the cache keeps. +const LEN: usize = (CLEAN_LIMIT + 2048) * 4096; + +/// The byte at `at`: each block opens with its own index, so a block read back +/// from anywhere else is seen, and the rest varies with the offset in it. +fn byte(at: usize) -> u8 { + let (block, within) = (at / 4096, at % 4096); + match within { + 0..8 => (block as u64).to_le_bytes()[within], + _ => (block ^ within.wrapping_mul(7)) as u8, + } +} + +fn main() { + let mut chunk = vec![0u8; CHUNK]; + { + let mut f = File::create(PATH).unwrap_or_else(|e| panic!("create {PATH}: {e}")); + for start in (0..LEN).step_by(CHUNK) { + for (i, b) in chunk.iter_mut().enumerate() { + *b = byte(start + i); + } + f.write_all(&chunk).unwrap_or_else(|e| panic!("write at {start}: {e}")); + } + f.sync_all().expect("the file is durable"); + } + println!("fs_cache_eviction: wrote {} MiB, past the {} MiB the cache keeps", LEN >> 20, (CLEAN_LIMIT * 4096) >> 20); + + let mut f = File::open(PATH).expect("open it again"); + let mut done = 0; + while done < LEN { + f.read_exact(&mut chunk).unwrap_or_else(|e| panic!("read at {done}: {e}")); + if let Some(i) = (0..CHUNK).find(|&i| chunk[i] != byte(done + i)) { + panic!("byte {} read back {:#04x}, written {:#04x}", done + i, chunk[i], byte(done + i)); + } + done += CHUNK; + } + drop(f); + fs::remove_file(PATH).expect("remove the file"); + println!("fs_cache_eviction: PASS"); +} diff --git a/tests/toyos-rust-tests/src/bin/fs_client_bound.rs b/tests/toyos-rust-tests/src/bin/fs_client_bound.rs new file mode 100644 index 00000000000..06b7ade1bb0 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/fs_client_bound.rs @@ -0,0 +1,48 @@ +//! A file server serves a bounded number of clients, and a client past the +//! bound is answered `ResourceExhausted` at its hello — refused by name, never +//! left waiting in the port's queue. +//! +//! Raw connections to `fs:/home`, each lending the same window, which costs +//! one region however many there are. Every other program on this boot holds +//! connections of its own, so the refusal comes at or before this process's +//! `MAX_SERVED + 1`th. + +use toyos::fs::{Reply, Request, HELLO, REPLY, WINDOW_BYTES}; +use toyos::ipc::Connection; +use toyos::shm::SharedMemory; +use toyos_abi::syscall::SyscallError; + +/// Mirrored from `userland/fsd/src/main.rs`. +const MAX_SERVED: usize = 128; + +fn answer(conn: &Connection) -> Reply { + let header = conn.recv_header().expect("a reply to the hello"); + assert_eq!(header.msg_type, REPLY, "a reply frame"); + conn.recv_payload(&header).expect("a reply's words") +} + +fn main() { + let names = toyos::endow::namespace().expect("this program was endowed a namespace"); + let window = SharedMemory::create(WINDOW_BYTES).expect("a window"); + let mut held = Vec::new(); + for n in 1..=MAX_SERVED + 1 { + let conn = names.open("fs:/home").expect("this program holds fs:/home"); + conn.send_with_handles(&[window.share().expect("the window, shared")], HELLO, &Request::new()) + .expect("hello"); + let reply = answer(&conn); + if reply.status == 0 { + held.push(conn); + continue; + } + assert_eq!( + reply.status, + SyscallError::ResourceExhausted.to_u64(), + "connection {n} was refused status {}, not ResourceExhausted", + reply.status + ); + println!("fs_client_bound: connection {n} of this process refused ResourceExhausted at its hello"); + println!("fs_client_bound: PASS"); + return; + } + panic!("{} connections of this process were all served, past the bound of {MAX_SERVED}", MAX_SERVED + 1); +} diff --git a/tests/toyos-rust-tests/src/bin/fs_restart.rs b/tests/toyos-rust-tests/src/bin/fs_restart.rs index bc9a7962ba9..cbefc62cb8b 100644 --- a/tests/toyos-rust-tests/src/bin/fs_restart.rs +++ b/tests/toyos-rust-tests/src/bin/fs_restart.rs @@ -2,13 +2,20 @@ //! //! Booted by `common::storage::fsd_restart` on `tests/fsdrestartcase`, whose //! file servers end the moment they take a write through a file opened to -//! append at `/home/fsd_end` and before they answer it: +//! append at `/home/fsd_end` and before they answer it, and at the first hello +//! on `/apps` this boot: //! -//! - a file written and `fsync`ed before the end — acknowledged and flushed — +//! - a launch of an `/apps` path is answered though DATA's server ends while +//! init resolves it: init makes that call on its file worker and starts the +//! server again while it waits, where a call from its loop would wait for +//! ever on a server only its loop could start; +//! - a file written and `fsync`ed before an end — acknowledged and flushed — //! reads back through a handle held across it, and a handle written across //! it goes on writing where it was; //! - the write the server ended under is answered as the server's end, never //! as done; +//! - a handle held across an end, on a file another was renamed over, is +//! answered `Gone` and writes nothing into the file now at its path; //! - DATA's server ended four times inside init's window is not started a //! fourth: `/home` then answers `Gone` to a new open and to a held handle. //! @@ -16,18 +23,24 @@ use std::fs::{self, File, OpenOptions}; use std::io::{ErrorKind, Read, Write}; +use std::process::Command; -/// Mirrored in `tests/common/storage.rs`: the flushed file, and the one -/// written across the end. +/// Mirrored in `tests/common/storage.rs`: the flushed file, the one written +/// across the first end, and what is renamed over it. const KEPT: &str = "/home/fs_restart/kept"; const ACROSS: &str = "/home/fs_restart/across"; +const REPLACEMENT: &str = "/home/fs_restart/replacement"; const KEPT_LEN: usize = 64 * 1024 + 13; const BEFORE: &[u8] = b"written and flushed before the server ended; "; const AFTER: &[u8] = b"written through the same handle after it came back"; +const REPLACING: &[u8] = b"renamed over the file a handle held across the end"; /// `--end-on`'s path, in `tests/fsdrestartcase/system.toml`. const END: &str = "/home/fsd_end"; +/// A path under `--end-at-hello`'s directory: no package answers for it. +const LAUNCHED: &str = "/apps/fs_restart/nothing"; + /// Mirrored: what `KEPT` holds. fn kept() -> Vec { (0..KEPT_LEN).map(|i| (i.wrapping_mul(37) ^ 0xC3) as u8).collect() @@ -50,6 +63,12 @@ fn end_the_server(n: u32) { } fn main() { + // End 1: the first hello on `/apps` is init's, resolving this launch. + match Command::new(LAUNCHED).status() { + Err(e) => println!("fs_restart: end 1: the launch of {LAUNCHED} was answered, refused ({e})"), + Ok(status) => panic!("{LAUNCHED}, which no package answers for, ran and exited {status}"), + } + fs::create_dir_all("/home/fs_restart").expect("make /home/fs_restart"); let mut f = File::create(KEPT).expect("create the kept file"); f.write_all(&kept()).expect("write the kept file"); @@ -60,7 +79,7 @@ fn main() { across.write_all(BEFORE).expect("write before the end"); across.sync_all().expect("durable before the end"); - end_the_server(1); + end_the_server(2); let mut back = Vec::new(); held.read_to_end(&mut back).expect("a handle held across the end reads"); @@ -72,9 +91,26 @@ fn main() { assert_eq!(whole, [BEFORE, AFTER].concat(), "the file written across the end"); println!("fs_restart: the server came back; a held handle read, another wrote on, both where they were"); - for n in 2..=4 { - end_the_server(n); + // Another file renamed over the one `across` holds, durably, and then an end. + let mut replacement = File::create(REPLACEMENT).expect("create the replacement"); + replacement.write_all(REPLACING).expect("write the replacement"); + replacement.sync_all().expect("the replacement is durable"); + drop(replacement); + fs::rename(REPLACEMENT, ACROSS).expect("rename the replacement over the held file"); + File::open(ACROSS).and_then(|f| f.sync_all()).expect("the rename is durable"); + + end_the_server(3); + + match across.write_all(b"written into whatever is at the path now") { + Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => { + println!("fs_restart: a handle on a file renamed over is answered Gone ({e})"); + } + Err(e) => panic!("a handle on a file renamed over was refused {e} ({:?}), not Gone", e.kind()), + Ok(()) => panic!("a handle on a file renamed over wrote into the file now at its path"), } + + end_the_server(4); + match File::open(KEPT) { Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => { println!("fs_restart: after four ends a new open is answered Gone ({e})"); diff --git a/tests/toyos-rust-tests/src/bin/fs_stream_offset.rs b/tests/toyos-rust-tests/src/bin/fs_stream_offset.rs new file mode 100644 index 00000000000..c95322ea6c5 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/fs_stream_offset.rs @@ -0,0 +1,74 @@ +//! A client's number is bounded where the server keeps it: a stream asked to +//! start past the largest offset a file has is refused, and the server that +//! refused it goes on answering. +//! +//! A raw client of `fs:/home`, past std: std never asks for such a stream, and +//! any program holding `/home` can. Accepted, the offset would grow by what +//! the pipe carries until it overflowed, which panics the server that every +//! program's `/apps`, `/config`, `/home` and `/state` are on. + +use toyos::fs::{ + window_put, Reply, Request, HELLO, MAX_FILE_BYTES, OPEN, O_CREATE, O_WRITE, REPLY, STREAM, WINDOW_BYTES, +}; +use toyos::ipc::Connection; +use toyos::shm::SharedMemory; +use toyos::volatile::Window; +use toyos_abi::syscall::SyscallError; + +/// Under `/home`, relative to it. +const NAME: &[u8] = b"fs_stream_offset"; + +fn answer(conn: &Connection) -> Reply { + let header = conn.recv_header().expect("a reply: the server is still there"); + assert_eq!(header.msg_type, REPLY, "a reply frame"); + conn.recv_payload(&header).expect("a reply's words") +} + +fn open(conn: &Connection, window: &SharedMemory) -> Reply { + // SAFETY: the region is `WINDOW_BYTES` long and outlives this use. + window_put(unsafe { Window::new(window.as_ptr(), WINDOW_BYTES) }, 0, NAME); + let request = Request { len: NAME.len() as u64, flags: O_WRITE | O_CREATE, ..Request::new() }; + conn.send(OPEN, &request).expect("open"); + answer(conn) +} + +fn main() { + let names = toyos::endow::namespace().expect("this program was endowed a namespace"); + let conn = names.open("fs:/home").expect("this program holds fs:/home"); + let window = SharedMemory::create(WINDOW_BYTES).expect("a window"); + conn.send_with_handles(&[window.share().expect("the window, shared")], HELLO, &Request::new()) + .expect("hello"); + assert_eq!(answer(&conn).status, 0, "fs:/home answers its hello"); + + let opened = open(&conn, &window); + assert_eq!(opened.status, 0, "{} opened to write", core::str::from_utf8(NAME).unwrap_or("")); + let fid = opened.value; + + for offset in [u64::MAX - 1, MAX_FILE_BYTES + 1] { + conn.send(STREAM, &Request { fid, offset, ..Request::new() }).expect("stream"); + let reply = answer(&conn); + if reply.status == 0 { + // What the server's end would have been: the pipe's bytes drained + // at an offset that overflows. + if let Some([end]) = conn.recv_handles_exact::<1>() { + let _ = toyos_abi::syscall::write(end, &[0x5A; 16]); + toyos_abi::syscall::close(end); + } + panic!("a stream at offset {offset:#x} was accepted"); + } + assert_eq!( + reply.status, + SyscallError::InvalidArgument.to_u64(), + "a stream at offset {offset:#x} was answered status {}", + reply.status + ); + println!("fs_stream_offset: a stream at {offset:#x} refused InvalidArgument"); + } + + // The same connection, so an end of the server between the two would be + // this read failing and not a reconnect hiding it. + assert_eq!(open(&conn, &window).status, 0, "the server answers after the refusals"); + drop(conn); + let _ = std::fs::remove_file("/home/fs_stream_offset"); + println!("fs_stream_offset: PASS"); +} diff --git a/tests/toyos-rust-tests/src/bin/writeback_durability.rs b/tests/toyos-rust-tests/src/bin/writeback_durability.rs deleted file mode 100644 index 952d2c5a2a6..00000000000 --- a/tests/toyos-rust-tests/src/bin/writeback_durability.rs +++ /dev/null @@ -1,192 +0,0 @@ -//! Dropping the last handle of a modified file loses no bytes: the write-back -//! queue and the shutdown drain make them durable with no fsync. -//! -//! The write-back queue defers a closed file's flush onto `iod`; this proves the -//! flush still happens. It writes distinctive bytes to a `/log` file, drops the -//! handle **without** fsync, lets `iod` drain — which flushes the pages to the -//! device and drops the file from the cache — and re-reads. The bytes now come -//! from the device, because the cache no longer holds the file; if `iod` had -//! skipped the flush, the device would answer a short/zero file. -//! -//! The host side (`tests/common/volumes.rs::writeback_durability`) then shuts the -//! machine down and reads the `/log` volume off the image with an independent FAT -//! implementation and the fatgen103 checker — neither the kernel's own cache -//! logic — so the bytes on the device and the structure around them are judged -//! by something that is not the code under test. - -use std::fs::{self, OpenOptions}; -use std::io::{Read, Seek, SeekFrom, Write}; -use std::thread; -use std::time::Duration; - -/// Mirrored in `tests/common/volumes.rs::writeback_durability`. -const PATH: &str = "/log/wb-durable.bin"; -const LEN: usize = 5 * 4096 + 91; -/// The second file, and the same mirroring. -const SHRUNK: &str = "/log/wb-shrunk.bin"; -const SHRUNK_LEN: usize = 3 * 4096; -pub const CUT: u64 = 100; - -fn blob() -> Vec { - (0..LEN).map(|i| (i.wrapping_mul(97) ^ 0x5A) as u8).collect() -} - -fn seed() -> Vec { - (0..SHRUNK_LEN).map(|i| (i.wrapping_mul(31).wrapping_add(7)) as u8 | 1).collect() -} - -/// The file shrunk only once cold, and where its read lands; same mirroring. -const REOPENED: &str = "/log/wb-reopened.bin"; -const WITNESS: &str = "/log/wb-served.bin"; - -/// The file `fat-flush-meta-refuse` refuses; mirrored in `kernel/src/fat32_adapter.rs`. -const RETRY: &str = "/log/wb-retry.bin"; -const RETRY_AT: u64 = 2 * 4096; - -/// The file whose straddled read `resize-fault-refuse` refuses; the length is -/// what keys the actuator, and is mirrored in `kernel/src/file_cache.rs`. -const REFUSED: &str = "/log/wb-refused.bin"; -const REFUSED_CUT: u64 = 133; - -fn kept() -> Vec { - (0..4096usize).map(|i| (i.wrapping_mul(53) ^ 0xC3) as u8).collect() -} - -fn main() { - let want = blob(); - - // Write and drop the handle. No `sync_all`: durability must come from the - // write-back queue and the shutdown drain, not from the caller asking. - { - let mut f = fs::File::create(PATH).unwrap_or_else(|e| panic!("create {PATH}: {e}")); - f.write_all(&want).expect("write the blob"); - } - - // Give `iod` a turn to drain the write-back the close pinned. The sleep - // yields the CPU, so on one core `iod` runs and on more it has long - // finished — either way the file is flushed to the device and dropped from - // the cache before the read below, so the read is a cache miss served by the - // device. The margin is enormous: the drain is microseconds of work. - thread::sleep(Duration::from_millis(200)); - - let mut got = Vec::new(); - { - let mut f = fs::File::open(PATH).expect("re-open after the write-back drained"); - f.read_to_end(&mut got).expect("read the file back from the device"); - } - assert_eq!( - got.len(), - want.len(), - "after close + drain, {PATH} is {} bytes on re-read, wrote {} — a deferred flush was lost", - got.len(), - want.len() - ); - if let Some(at) = got.iter().zip(&want).position(|(a, b)| a != b) { - panic!("{PATH} differs from what was written at byte {at}: the write-back did not reach the device"); - } - - shrink_unflushed_then_regrow(); - shrink_a_file_the_cache_no_longer_holds(); - a_refused_metadata_write_keeps_the_pages_it_wrote(); - a_refused_fault_resizes_nothing(); - - println!("wrote {LEN} bytes, closed without fsync, and read them back after the write-back drained"); -} - -/// A flush that trims a shrink, writes a page above the mark, and is then -/// refused at its metadata write. The retry has no dirty pages left, so a mark -/// that outlived the trim would trim a second time and free the page the first -/// attempt wrote. The volume is where that shows, and the host reads it. -fn a_refused_metadata_write_keeps_the_pages_it_wrote() { - let mut f = OpenOptions::new() - .read(true).write(true).create(true).truncate(true) - .open(RETRY) - .unwrap_or_else(|e| panic!("create {RETRY}: {e}")); - f.write_all(&seed()).expect("write the seed"); - f.sync_all().expect("fsync the seed"); - f.set_len(CUT).expect("shrink into the first page"); - f.set_len(SHRUNK_LEN as u64).expect("regrow"); - f.seek(SeekFrom::Start(RETRY_AT)).expect("seek above the mark"); - f.write_all(&kept()).expect("write a page above the mark"); - - // Refused once by the actuator and retried by `SYS_FSYNC` itself, so this - // returning is the retry having succeeded, not the first attempt. - f.sync_all().expect("the fsync whose second metadata write is refused"); - drop(f); - println!("shrank {RETRY} to {CUT}, regrew it, wrote a page at {RETRY_AT}, fsynced through one refusal"); -} - -/// The same shrink and regrow over a page the cache does not hold, with what -/// the kernel *served* carried to the volume for the host to judge — the device -/// is right either way, because `Fat32::set_len` zero-fills on every grow. -fn shrink_a_file_the_cache_no_longer_holds() { - { - let mut f = fs::File::create(REOPENED).unwrap_or_else(|e| panic!("create {REOPENED}: {e}")); - f.write_all(&seed()).expect("write the seed"); - f.sync_all().expect("fsync the seed"); - } - thread::sleep(Duration::from_millis(200)); - - let mut f = OpenOptions::new() - .read(true).write(true) - .open(REOPENED) - .unwrap_or_else(|e| panic!("reopen {REOPENED}: {e}")); - f.set_len(CUT).expect("shrink into a page the cache does not hold"); - f.set_len(SHRUNK_LEN as u64).expect("regrow"); - let mut served = Vec::new(); - f.read_to_end(&mut served).expect("read the regrown file back"); - drop(f); - - let mut w = fs::File::create(WITNESS).unwrap_or_else(|e| panic!("create {WITNESS}: {e}")); - w.write_all(&served).expect("write what the read served"); - w.sync_all().expect("fsync the witness"); - drop(w); - - thread::sleep(Duration::from_millis(200)); - println!("shrank a cold {REOPENED} to {CUT}, regrew it, and wrote the {} bytes it served to {WITNESS}", served.len()); -} - - -/// A device that will not give the straddled page back leaves the file alone: -/// zeroing half a page is a read-modify-write, with nothing to modify. -fn a_refused_fault_resizes_nothing() { - { - let mut f = fs::File::create(REFUSED).unwrap_or_else(|e| panic!("create {REFUSED}: {e}")); - f.write_all(&seed()).expect("write the seed"); - f.sync_all().expect("fsync the seed"); - } - thread::sleep(Duration::from_millis(200)); - - let f = OpenOptions::new() - .read(true).write(true) - .open(REFUSED) - .unwrap_or_else(|e| panic!("reopen {REFUSED}: {e}")); - let err = f.set_len(REFUSED_CUT).expect_err("a refused fault must refuse the truncate"); - let len = f.metadata().expect("stat").len(); - assert_eq!(len, SHRUNK_LEN as u64, "{REFUSED} is {len} bytes after a truncate that was refused"); - drop(f); - - thread::sleep(Duration::from_millis(200)); - println!("the refused straddled read left {REFUSED} whole: {err}"); -} - -/// A durable file shrunk and regrown with nothing flushed between the two, and -/// then only closed. The drain's single `update_metadata` carries the final -/// size alone, so the volume the host judges is the only place the dropped -/// clusters can still be named — and the checker cannot see the harm, because -/// a file naming its own old clusters is a structurally valid file. -fn shrink_unflushed_then_regrow() { - let want = seed(); - let mut f = OpenOptions::new() - .read(true).write(true).create(true).truncate(true) - .open(SHRUNK) - .unwrap_or_else(|e| panic!("create {SHRUNK}: {e}")); - f.write_all(&want).expect("write the seed"); - f.sync_all().expect("fsync the seed"); - f.set_len(CUT).expect("shrink into the first page"); - f.set_len(SHRUNK_LEN as u64).expect("regrow"); - drop(f); - - thread::sleep(Duration::from_millis(200)); - println!("shrank {SHRUNK} to {CUT} and regrew it to {SHRUNK_LEN}, closed without fsync"); -} diff --git a/toyos-blockring/src/wire.rs b/toyos-blockring/src/wire.rs index b6c7e04204c..d92241c4031 100644 --- a/toyos-blockring/src/wire.rs +++ b/toyos-blockring/src/wire.rs @@ -92,7 +92,8 @@ pub enum Refusal { /// A session holds it already. Held, /// The partition is there and cannot be served: its range is not whole - /// blocks, its GUID is on two entries, or its table did not read. + /// blocks, its GUID is on two entries, its table did not read, or its + /// controller would not open — the answer to a listing then too. Unusable, /// The frame, its handles or its region are not what this protocol sends. Malformed, diff --git a/toyos-fat32/src/fs.rs b/toyos-fat32/src/fs.rs index 4b9a3e6bd4f..071917c75e5 100644 --- a/toyos-fat32/src/fs.rs +++ b/toyos-fat32/src/fs.rs @@ -135,6 +135,13 @@ impl File { self.size == 0 } + /// What tells this file from another at the same path: its entry's short + /// name and creation stamp, and its first cluster — 0 for a file that has + /// none yet. + pub fn identity(&self) -> ([u8; 16], u32) { + (self.identity.0, self.first_cluster.map_or(0, Cluster::raw)) + } + /// Whether stopping now would leave the volume holding clusters this /// handle's directory entry does not reach. /// diff --git a/toyos/src/fs.rs b/toyos/src/fs.rs index e347cd5a791..4f9ff186f28 100644 --- a/toyos/src/fs.rs +++ b/toyos/src/fs.rs @@ -25,9 +25,11 @@ //! again through the same connector — init keeps a file server's ports open //! across its restart — and counts its connections in a generation. A file id //! from an earlier generation names nothing on the new connection; the holder -//! of a file opens it again by its path, and a file whose path no longer names -//! it answers [`SyscallError::Gone`]. Nothing here keeps a write the server -//! acknowledged and never made durable: that is what `Fsync` is for. +//! of a file opens it again by its path and keeps it only when the server +//! answers the [`Stat::ident`] it last saw. Any other answer — another file at +//! the path, this one changed by a write the restart lost, or a volume that +//! cannot tell — is [`SyscallError::Gone`]. Nothing here keeps a write the +//! server acknowledged and never made durable: that is what `Fsync` is for. use toyos_abi::syscall::{SyscallError, MAX_SERVICE_NAME}; @@ -109,13 +111,15 @@ ipc_payload! { pub flags: u64, } - /// Every reply's words. `status` is 0 or a [`SyscallError`]'s wire value. + /// Every reply's words. `status` is 0 or a [`SyscallError`]'s wire value; + /// `ident` is a file's [`Stat::ident`], on every reply about an open file. pub struct Reply { pub status: u64, pub kind: u64, pub value: u64, pub value2: u64, pub mtime: u64, + pub ident: u64, } } @@ -127,11 +131,11 @@ impl Request { impl Reply { pub const fn ok() -> Self { - Self { status: 0, kind: 0, value: 0, value2: 0, mtime: 0 } + Self { status: 0, kind: 0, value: 0, value2: 0, mtime: 0, ident: 0 } } pub const fn refused(e: SyscallError) -> Self { - Self { status: e.to_u64(), kind: 0, value: 0, value2: 0, mtime: 0 } + Self { status: e.to_u64(), kind: 0, value: 0, value2: 0, mtime: 0, ident: 0 } } fn result(self) -> Result { @@ -150,39 +154,29 @@ pub fn canonical(path: &str) -> bool { && (path.is_empty() || path.split('/').all(|c| !c.is_empty() && c != "." && c != "..")) } -/// Copy `data` into `window` at `offset`, word-wise where it can. +// **Who owns the window when.** The server, from the moment a request is sent +// until its reply is received; the client, from then until it sends the next. +// The send and the receive are syscalls, which order the two sides, so honest +// peers never touch the window at once. A dishonest one races only bytes it +// could have written anyway: each side copies what it reads out of the window +// once, into memory of its own, and acts on that copy alone. Nothing here polls +// the window or waits on it, so it is copied as plain bytes — one fetch of each, +// as a volatile loop would be, and no reference is ever formed over it. + +/// Copy `data` into `window` at `offset`. pub fn window_put(window: Window, offset: usize, data: &[u8]) { let target = window.sub(offset, data.len()); - let head = (8 - target.as_ptr() as usize % 8) % 8; - let head = head.min(data.len()); - for (i, byte) in data[..head].iter().enumerate() { - target.write::(i, *byte); - } - let words = (data.len() - head) / 8 * 8; - if words > 0 { - target.sub(head, words).copy_in(0, &data[head..head + words]); - } - for (i, byte) in data[head + words..].iter().enumerate() { - target.write::(head + words + i, *byte); - } + // SAFETY: `sub` bounded `offset + data.len()` inside the window, which its + // constructor's contract says is mapped; `data` is this process's own + // memory and never the window, so the two do not overlap. + unsafe { core::ptr::copy_nonoverlapping(data.as_ptr(), target.as_ptr(), data.len()) } } -/// Copy `out.len()` bytes out of `window` at `offset`, word-wise where it can. +/// Copy `out.len()` bytes out of `window` at `offset`. pub fn window_take(window: Window, offset: usize, out: &mut [u8]) { let source = window.sub(offset, out.len()); - let head = (8 - source.as_ptr() as usize % 8) % 8; - let head = head.min(out.len()); - for (i, byte) in out[..head].iter_mut().enumerate() { - *byte = source.read::(i); - } - let words = (out.len() - head) / 8 * 8; - if words > 0 { - source.sub(head, words).copy_out(0, &mut out[head..head + words]); - } - let tail = head + words; - for (i, byte) in out[tail..].iter_mut().enumerate() { - *byte = source.read::(tail + i); - } + // SAFETY: as in `window_put`, with the two sides swapped. + unsafe { core::ptr::copy_nonoverlapping(source.as_ptr(), out.as_mut_ptr(), out.len()) } } /// What a file or directory is. @@ -191,6 +185,22 @@ pub struct Stat { pub kind: u64, pub size: u64, pub mtime: u64, + /// For an open file: the server's token for this file as it stands, the + /// same across a restart of the server for the same file unchanged and + /// different for any other file at the path or any change to this one; + /// 0 where the volume cannot tell one file from another, and for a path. + pub ident: u64, +} + +/// A write answered. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct Written { + /// How many bytes it took. + pub len: usize, + /// The offset after them: for a file opened to append, the file's end. + pub offset: u64, + /// The file's [`Stat::ident`] after it. + pub ident: u64, } /// An open answered. @@ -371,22 +381,20 @@ impl Dir { /// Write at most the window of `data` at `offset`, or at the end for a file /// opened to append. Answers how much, and the offset after it. - pub fn write(&mut self, fid: u64, generation: u64, offset: u64, data: &[u8]) -> Result<(usize, u64), SyscallError> { + pub fn write(&mut self, fid: u64, generation: u64, offset: u64, data: &[u8]) -> Result { let len = data.len().min(WINDOW_BYTES); - if generation != self.generation { - return Err(SyscallError::Gone); - } window_put(self.window(), 0, &data[..len]); let reply = self.fid_call(WRITE, generation, &Request { fid, offset, len: len as u64, ..Request::new() })?; - Ok(((reply.value as usize).min(len), reply.value2)) + Ok(Written { len: (reply.value as usize).min(len), offset: reply.value2, ident: reply.ident }) } pub fn fstat(&mut self, fid: u64, generation: u64) -> Result { self.fid_call(FSTAT, generation, &Request { fid, ..Request::new() }).map(|r| stat_of(&r)) } - pub fn truncate(&mut self, fid: u64, generation: u64, size: u64) -> Result<(), SyscallError> { - self.fid_call(TRUNCATE, generation, &Request { fid, offset: size, ..Request::new() }).map(drop) + /// Answers the file's [`Stat::ident`] after it. + pub fn truncate(&mut self, fid: u64, generation: u64, size: u64) -> Result { + self.fid_call(TRUNCATE, generation, &Request { fid, offset: size, ..Request::new() }).map(|r| r.ident) } /// Make what this file was written durable on the volume. @@ -502,7 +510,7 @@ pub fn encode_entry(out: &mut [u8], kind: u64, size: u64, name: &str) -> Option< } fn stat_of(reply: &Reply) -> Stat { - Stat { kind: reply.kind, size: reply.value2, mtime: reply.mtime } + Stat { kind: reply.kind, size: reply.value2, mtime: reply.mtime, ident: reply.ident } } fn receive(conn: &Connection) -> Result { diff --git a/userland/blockd/src/main.rs b/userland/blockd/src/main.rs index 91f31ace2ed..287b45b3e1c 100644 --- a/userland/blockd/src/main.rs +++ b/userland/blockd/src/main.rs @@ -457,10 +457,12 @@ fn claim() -> Option { Some(Endowments::get().take(&label).expect("blockd: the claim its label names")) } -/// Serve a machine with no controller: every listing empty, every open -/// `NotFound`, so a file server finds no partition here rather than waiting on -/// one. A connection says one frame and is answered, or is let go. -fn serve_nothing(acceptor: &toyos::port::Acceptor) -> ! { +/// Serve no partition: a machine with no controller answers every listing +/// empty and every open `NotFound`, and one whose controller is `unusable` +/// answers both `Unusable`, so a file server says the disk failed rather than +/// that there is none, and waits on neither. A connection says one frame and +/// is answered, or is let go. +fn serve_nothing(acceptor: &toyos::port::Acceptor, unusable: bool) -> ! { let poller = Poller::new(2 + MAX_PENDING as u32); let mut pending: Vec = Vec::new(); let mut ready: Vec = Vec::new(); @@ -508,9 +510,10 @@ fn serve_nothing(acceptor: &toyos::port::Acceptor) -> ! { if let Some(handles) = p.conn.recv_handles_exact::<1>() { toyos_abi::syscall::close(handles[0]); } - let _ = match msg_type { - wire::MSG_LIST => p.conn.try_send_bytes(wire::MSG_LISTED, &[]), - _ => p.conn.try_send_bytes(wire::MSG_REFUSED, &Refusal::NotFound.encode()), + let _ = match (msg_type, unusable) { + (_, true) => p.conn.try_send_bytes(wire::MSG_REFUSED, &Refusal::Unusable.encode()), + (wire::MSG_LIST, false) => p.conn.try_send_bytes(wire::MSG_LISTED, &[]), + (_, false) => p.conn.try_send_bytes(wire::MSG_REFUSED, &Refusal::NotFound.encode()), }; } } @@ -535,7 +538,7 @@ fn main() { let acceptor = endow::acceptor(PORT).unwrap_or_else(|| panic!("blockd: started serving no `{PORT}` port")); let Some(dev) = claim() else { println!("blockd: no NVMe controller this row names is on this machine; serving no partition"); - serve_nothing(&acceptor); + serve_nothing(&acceptor, false); }; // A controller this service cannot use is a machine without one, said by // name: restarting would meet the same device and the same refusal. @@ -543,7 +546,7 @@ fn main() { Ok(ctrl) => ctrl, Err(why) => { println!("blockd: NOT SERVING — {why}; serving no partition"); - serve_nothing(&acceptor); + serve_nothing(&acceptor, true); } }; println!( diff --git a/userland/fsd/src/absent.rs b/userland/fsd/src/absent.rs index fc7e52db8e0..a3fc95549f7 100644 --- a/userland/fsd/src/absent.rs +++ b/userland/fsd/src/absent.rs @@ -5,7 +5,7 @@ use toyos_abi::syscall::SyscallError; -use crate::volume::{Kind, Meta, Node, OpenHow, Volume}; +use crate::volume::{Kind, Meta, Node, OpenHow, Out, Volume}; pub struct Absent { roots: Vec, @@ -57,7 +57,11 @@ impl Volume for Absent { Err(SyscallError::NotFound) } - fn read(&mut self, _node: Node, _offset: u64, _out: &mut [u8]) -> Result { + fn ident(&mut self, _node: Node) -> Result { + Err(SyscallError::NotFound) + } + + fn read(&mut self, _node: Node, _offset: u64, _out: &mut dyn Out) -> Result { Err(SyscallError::NotFound) } diff --git a/userland/fsd/src/cache.rs b/userland/fsd/src/cache.rs index 4697b528162..a8b60c65a91 100644 --- a/userland/fsd/src/cache.rs +++ b/userland/fsd/src/cache.rs @@ -90,14 +90,20 @@ impl Cache { /// is there and in runs from the disk where it is not. pub fn read(&self, first: u64, out: &mut [u8]) -> Result<(), DiskError> { assert!(out.len() % BLOCK == 0, "fsd: a cache read of {} bytes", out.len()); - let count = out.len() / BLOCK; + self.visit(first, out.len() / BLOCK, |k, block| out[k * BLOCK..(k + 1) * BLOCK].copy_from_slice(block)) + } + + /// Blocks `first..first + count`, each handed to `put` with its index from + /// `first`, out of the cache itself: a block that is not there is fetched + /// into it first, in runs, so a read of cached blocks copies each once. + pub fn visit(&self, first: u64, count: usize, mut put: impl FnMut(usize, &[u8; BLOCK])) -> Result<(), DiskError> { let mut inner = self.inner.borrow_mut(); let mut i = 0; while i < count { let block = first + i as u64; inner.reads += 1; if let Some(slot) = inner.slots.get(&block) { - out[i * BLOCK..(i + 1) * BLOCK].copy_from_slice(&slot.data[..]); + put(i, &slot.data); inner.hits += 1; i += 1; continue; @@ -107,11 +113,12 @@ impl Cache { while i + run < count && run < RUN && !inner.slots.contains_key(&(block + run as u64)) { run += 1; } - let span = &mut out[i * BLOCK..(i + run) * BLOCK]; - inner.disk.read(block, span)?; + let mut span = vec![0u8; run * BLOCK]; + inner.disk.read(block, &mut span)?; for (k, chunk) in span.chunks_exact(BLOCK).enumerate() { let mut data = Box::new([0u8; BLOCK]); data.copy_from_slice(chunk); + put(i + k, &data); inner.slots.insert(block + k as u64, Slot { data, dirty: false }); inner.order.push_back(block + k as u64); } diff --git a/userland/fsd/src/data.rs b/userland/fsd/src/data.rs index a3f5d209fe8..03d095c3913 100644 --- a/userland/fsd/src/data.rs +++ b/userland/fsd/src/data.rs @@ -37,7 +37,7 @@ use toyos_abi::syscall::SyscallError; use crate::cache::{Cache, Shared}; use crate::disk::{Disk, DiskError, BLOCK}; -use crate::volume::{join, parent, Kind, Meta, Node, OpenHow, Volume}; +use crate::volume::{identity, join, parent, Kind, Meta, Node, OpenHow, Out, Volume}; /// The longest symlink target read back: the wire's path bound. const MAX_LINK: u64 = toyos::fs::MAX_PATH as u64; @@ -425,7 +425,18 @@ impl Volume for DataVolume { Ok(Meta { kind: Kind::File, size: open.size, mtime: open.mtime }) } - fn read(&mut self, node: Node, offset: u64, out: &mut [u8]) -> Result { + fn ident(&mut self, node: Node) -> Result { + let open = self.entry(node)?; + // The first block is this file's alone among the live ones, and the + // length and mtime move with every write; a file with no block yet is + // told from another by nothing the format records. + Ok(match open.extents.first() { + Some(first) => identity(&[first.start_block, open.size, open.mtime]), + None => 0, + }) + } + + fn read(&mut self, node: Node, offset: u64, out: &mut dyn Out) -> Result { let cache = Rc::clone(&self.cache); let open = self.entry(node)?; if offset >= open.size { @@ -433,7 +444,6 @@ impl Volume for DataVolume { } let n = out.len().min((open.size - offset) as usize); let mut done = 0; - let mut page_buf = vec![0u8; BLOCK]; while done < n { let at = offset + done as u64; let page = at / BLOCK as u64; @@ -448,8 +458,8 @@ impl Volume for DataVolume { { run += 1; } - let span = &mut out[done..done + run as usize * BLOCK]; - cache.read(first, span).map_err(disk_word)?; + let base = done; + cache.visit(first, run as usize, |k, block| out.put(base + k * BLOCK, block)).map_err(disk_word)?; done += run as usize * BLOCK; continue; } @@ -457,11 +467,11 @@ impl Volume for DataVolume { let take = (BLOCK - within).min(n - done); match block_for(&open.extents, page) { Some(block) => { - cache.read(block, &mut page_buf).map_err(disk_word)?; - out[done..done + take].copy_from_slice(&page_buf[within..within + take]); + let base = done; + cache.visit(block, 1, |_, b| out.put(base, &b[within..within + take])).map_err(disk_word)?; } // Past the extents: a hole, whose bytes are zeros. - None => out[done..done + take].fill(0), + None => out.zero(done, take), } done += take; } @@ -702,6 +712,7 @@ impl Volume for DataVolume { mod tests { use super::*; use crate::disk::Ram; + use crate::volume::Buf; fn clock() -> u64 { 1_000_000_000 @@ -722,7 +733,7 @@ mod tests { v.write(n, 100, &data).unwrap(); v.write(n, 20_000, b"tail").unwrap(); let mut out = vec![0xFFu8; 20_004]; - assert_eq!(v.read(n, 0, &mut out).unwrap(), 20_004); + assert_eq!(v.read(n, 0, &mut Buf(&mut out)).unwrap(), 20_004); assert!(out[..100].iter().all(|&b| b == 0)); assert_eq!(&out[100..10_100], &data[..]); assert!(out[10_100..20_000].iter().all(|&b| b == 0), "a hole reads zeros"); @@ -745,17 +756,55 @@ mod tests { assert_eq!(again.lstat("home/toy").unwrap().kind, Kind::Dir, "an empty directory outlives the mount"); let n = again.open("home/toy/x", PLAIN).unwrap(); let mut out = vec![0u8; 5000]; - again.read(n, 0, &mut out).unwrap(); + again.read(n, 0, &mut Buf(&mut out)).unwrap(); assert_eq!(out, vec![7; 5000]); } + /// What a holder re-opening by path after a restart compares: the same + /// file unchanged states the same identity off the device, and the file + /// renamed over it, or a write to it, does not. A file with no block is + /// told from another by nothing, and says so. + #[test] + fn an_identity_survives_a_remount_and_tells_a_replacement_apart() { + let remount = |v: DataVolume| { + let DataVolume { fs, cache, .. } = v; + drop(fs); + let disk = Rc::try_unwrap(cache).ok().expect("one owner").into_disk(); + let Probed::Mounted(again) = DataVolume::probe(disk, &["home"], clock) else { panic!("remount") }; + again + }; + let mut v = vol(); + let n = v.open("home/x", CREATE).unwrap(); + assert_eq!(v.ident(n), Ok(0), "no block yet: nothing tells it from another"); + v.write(n, 0, &[7; 5000]).unwrap(); + let held = v.ident(n).unwrap(); + assert_ne!(held, 0); + v.sync().unwrap(); + + let mut v = remount(v); + let n = v.open("home/x", PLAIN).unwrap(); + assert_eq!(v.ident(n), Ok(held), "the same file, unchanged, off the device"); + v.write(n, 5000, b"more").unwrap(); + assert_ne!(v.ident(n).unwrap(), held, "a write changes it"); + v.close(n); + + let y = v.open("home/y", CREATE).unwrap(); + v.write(y, 0, &[9; 5000]).unwrap(); + v.close(y); + v.rename("home/y", "home/x").unwrap(); + v.sync().unwrap(); + let mut v = remount(v); + let n = v.open("home/x", PLAIN).unwrap(); + assert_ne!(v.ident(n).unwrap(), held, "the file renamed over it is another"); + } + #[test] fn a_file_unlinked_while_open_answers_gone() { let mut v = vol(); let n = v.open("home/a", CREATE).unwrap(); v.write(n, 0, b"x").unwrap(); v.unlink("home/a").unwrap(); - assert_eq!(v.read(n, 0, &mut [0u8; 1]), Err(SyscallError::Gone)); + assert_eq!(v.read(n, 0, &mut Buf(&mut [0u8; 1])), Err(SyscallError::Gone)); assert_eq!(v.lstat("home/a"), Err(SyscallError::NotFound)); } @@ -767,7 +816,7 @@ mod tests { v.truncate(n, 10).unwrap(); v.truncate(n, 8192).unwrap(); let mut out = vec![0xFFu8; 8192]; - v.read(n, 0, &mut out).unwrap(); + v.read(n, 0, &mut Buf(&mut out)).unwrap(); assert_eq!(&out[..10], &[9; 10]); assert!(out[10..].iter().all(|&b| b == 0)); } @@ -801,7 +850,7 @@ mod tests { v.rename("home/a", "home/b").unwrap(); assert_eq!(v.lstat("home/b/f").unwrap().size, 5); let mut out = [0u8; 5]; - v.read(n, 0, &mut out).unwrap(); + v.read(n, 0, &mut Buf(&mut out)).unwrap(); assert_eq!(&out, b"hello"); v.close(n); let m = v.open("home/b/g", CREATE).unwrap(); @@ -832,7 +881,7 @@ mod tests { assert_eq!(v.rename("home/f", &to_path), Err(SyscallError::ResourceExhausted)); let mut back = [0u8; 22]; - assert_eq!(v.read(to, 0, &mut back), Ok(22), "the holder of `to` still reads it"); + assert_eq!(v.read(to, 0, &mut Buf(&mut back)), Ok(22), "the holder of `to` still reads it"); assert_eq!(&back, b"written and not synced"); v.sync().unwrap(); v.close(to); diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index ec242432455..9477083d461 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -25,7 +25,7 @@ use toyos_fat32::{BlockAccess, Error, Fat32, FatTime, IoError}; use crate::cache::Cache; use crate::disk::{Disk, BLOCK}; -use crate::volume::{parent, Kind, Meta, Node, OpenHow, Volume}; +use crate::volume::{identity, parent, Kind, Meta, Node, OpenHow, Out, Volume}; /// The most entries one directory listing materialises. const MAX_LIST: usize = 16_384; @@ -95,6 +95,10 @@ pub struct FatVolume { next: Node, /// Local seconds since the epoch, which is the zone FAT stamps in. clock: fn() -> u64, + /// Where a read lands before it goes out: `toyos-fat32` reads into a + /// slice, and a client's window is never one. Kept, so a read allocates + /// nothing. + scratch: Vec, } fn word(e: Error) -> SyscallError { @@ -138,7 +142,16 @@ impl FatVolume { geom.bytes_per_sector, geom.bytes_per_cluster() ); - Ok(Self { fs, cache, writable, open: BTreeMap::new(), by_path: BTreeMap::new(), next: 1, clock }) + Ok(Self { + fs, + cache, + writable, + open: BTreeMap::new(), + by_path: BTreeMap::new(), + next: 1, + clock, + scratch: Vec::new(), + }) } fn time(&self) -> FatTime { @@ -289,12 +302,28 @@ impl Volume for FatVolume { Ok(Meta { kind: Kind::File, size, mtime: mtime * 1_000_000_000 }) } - fn read(&mut self, node: Node, offset: u64, out: &mut [u8]) -> Result { + fn ident(&mut self, node: Node) -> Result { + let open = self.entry(node)?; + let (entry, cluster) = open.file.identity(); + if cluster == 0 { + return Ok(0); + } + let half = |at: usize| u64::from_le_bytes(entry[at..at + 8].try_into().expect("eight bytes")); + Ok(identity(&[half(0), half(8), cluster as u64, open.file.len()])) + } + + fn read(&mut self, node: Node, offset: u64, out: &mut dyn Out) -> Result { let open = self.open.get_mut(&node).ok_or(SyscallError::NotFound)?; if open.gone { return Err(SyscallError::Gone); } - self.fs.read(&mut open.file, offset, out).map_err(|e| logged("read", &open.path, e)) + if self.scratch.len() < out.len() { + self.scratch.resize(out.len(), 0); + } + let buf = &mut self.scratch[..out.len()]; + let n = self.fs.read(&mut open.file, offset, buf).map_err(|e| logged("read", &open.path, e))?; + out.put(0, &buf[..n]); + Ok(n) } fn write(&mut self, node: Node, offset: u64, data: &[u8]) -> Result<(), SyscallError> { diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index 0375997a184..fde4e2cda98 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -6,13 +6,12 @@ //! disk the kernel drives, or blockd's `block` connector in its namespace; and //! nothing else of the machine. Argv is the role, and for LOG and BOOT the //! unique GUID of the partition the loader named for it when no claim on it -//! was minted; then the row's own arguments, of which there is one, a test's -//! actuator ([`END_ON`]). +//! was minted; then the row's own arguments, which are tests' actuators +//! ([`END_ON`], [`END_AT_HELLO`]). //! //! **A connection is bound to the directory whose port it came in on**, and -//! every path on it is resolved there (`fsd::resolve`); a write on a port -//! whose directory is read-only, or on a volume that is, is refused before the -//! volume sees it. +//! every path on it is resolved there (`fsd::resolve`); a write on a read-only +//! volume is refused before the volume sees it. //! //! **A server never blocks on a client.** Accept and the first frame are two //! events; a request is buffered until whole; every reply is one `try_send`, @@ -33,7 +32,7 @@ use fsd::data::{DataVolume, Probed}; use fsd::disk::{Claimed, Disk, Ram, Served}; use fsd::fat::FatVolume; use fsd::resolve::{self, Found, Refusal as Escape, Resolved}; -use fsd::volume::{Kind, Meta, Node, OpenHow, Volume}; +use fsd::volume::{Kind, Meta, Node, OpenHow, Out, Volume}; use toyos::endow::{self, Endowments}; use toyos::fs::*; use toyos::ipc::{self, Connection, RxStep}; @@ -48,8 +47,15 @@ use toyos_abi::syscall::{SyscallError, DEV_PREFIX, SERVE_PREFIX}; /// write-back drained a closed file's pages on its next pass. const WRITEBACK: Duration = Duration::from_secs(2); -/// Clients held at once; the next waits in the port's queue. -const MAX_CLIENTS: usize = 128; +/// Clients served at once, machine-wide: one connection per directory per +/// process. The next is answered `ResourceExhausted` at its hello and let go, +/// by name. +const MAX_SERVED: usize = 128; + +/// Connections taken and not yet answered at their hello. While this many +/// wait, the next waits in its port's queue — for at most +/// [`HANDSHAKE_TIMEOUT`], by which each of these is answered or let go. +const MAX_HANDSHAKES: usize = 32; /// Files one client holds open at once. const MAX_FIDS: usize = 1024; @@ -57,6 +63,12 @@ const MAX_FIDS: usize = 1024; /// Streams served at once, machine-wide. const MAX_STREAMS: usize = 64; +/// Directories one server serves: DATA's four, with room. +const MAX_DIRS: usize = 32; + +// One wait watches every acceptor, connection and stream at once. +const _: () = assert!(MAX_SERVED + MAX_HANDSHAKES + MAX_STREAMS + MAX_DIRS <= Poller::MAX_HANDLES as usize); + /// How long an accepted connection may take to lend its window. const HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(2); @@ -70,6 +82,12 @@ const RAM_BLOCKS: u64 = 1 << 18; /// stages one. Armed by nothing but a boot config's `args`. const END_ON: &str = "--end-on"; +/// `--end-at-hello `: the first hello on `dir`'s port this boot ends this +/// process before it is answered — a server killed while its first client +/// waits, as a test stages one. Once a boot, across restarts: the kernel's +/// `/tmp` keeps the mark. Armed by nothing but a boot config's `args`. +const END_AT_HELLO: &str = "--end-at-hello"; + const TOKEN_CLIENT: u64 = 1 << 32; const TOKEN_STREAM: u64 = 2 << 32; @@ -97,7 +115,6 @@ struct Capability { dir: String, /// Where it is on the volume. root: String, - writable: bool, acceptor: Acceptor, } @@ -156,11 +173,15 @@ fn main() { let role = args.get(1).and_then(|r| Role::parse(r)).unwrap_or_else(|| { panic!("fsd: started with {args:?}; the first argument is a role: data, log or boot") }); - let (mut guid, mut end_on) = (None, None); + let (mut guid, mut end_on, mut end_at_hello) = (None, None, None); let mut rest = args.iter().skip(2); while let Some(arg) = rest.next() { match arg.as_str() { END_ON => end_on = Some(rest.next().unwrap_or_else(|| panic!("fsd: {END_ON} takes a path")).clone()), + END_AT_HELLO => { + end_at_hello = + Some(rest.next().unwrap_or_else(|| panic!("fsd: {END_AT_HELLO} takes a directory")).clone()) + } flag if flag.starts_with("--") => panic!("fsd: {flag} is no argument of this server"), named if guid.is_none() => guid = Some(named), extra => panic!("fsd: a second partition, {extra}, after {guid:?}"), @@ -185,7 +206,9 @@ fn main() { next_stream: 0, dirty_since: None, end_on, + end_at_hello, probe: Poller::new(caps_len), + scratch: Vec::new(), } .serve() } @@ -203,9 +226,10 @@ fn capabilities(role: Role) -> Vec { Role::Data => dir.trim_start_matches('/').to_string(), Role::Log | Role::Boot => String::new(), }; - caps.push(Capability { dir, root, writable: role != Role::Boot, acceptor }); + caps.push(Capability { dir, root, acceptor }); } assert!(!caps.is_empty(), "fsd: started serving no directory"); + assert!(caps.len() <= MAX_DIRS, "fsd: started serving {} directories, past {MAX_DIRS}", caps.len()); caps } @@ -237,17 +261,17 @@ fn served(guid: [u8; 16]) -> Option { } } -/// The TOYOS-DATA partitions blockd serves. -fn data_partitions() -> Vec<[u8; 16]> { - let Some(names) = endow::namespace() else { return Vec::new() }; +/// The TOYOS-DATA partitions blockd serves, or why it would not say: a disk +/// that failed is not a machine without one, and is never answered with memory. +fn data_partitions() -> Result, String> { + let Some(names) = endow::namespace() else { return Ok(Vec::new()) }; match blockd::list(names, toyos_blockring::PORT) { - Ok(listed) => listed.into_iter().filter(|l| l.kind == toyos_gpt::Guid::TOYOS_DATA.0).map(|l| l.unique).collect(), - // No block service in this image: the namespace has no `block`. - Err(blockd::Error::Kernel(SyscallError::NotFound)) => Vec::new(), - Err(why) => { - println!("fsd: blockd would not list its partitions: {why:?}"); - Vec::new() + Ok(listed) => { + Ok(listed.into_iter().filter(|l| l.kind == toyos_gpt::Guid::TOYOS_DATA.0).map(|l| l.unique).collect()) } + // No block service in this image: the namespace has no `block`. + Err(blockd::Error::Kernel(SyscallError::NotFound)) => Ok(Vec::new()), + Err(why) => Err(format!("the block service would not list its partitions ({why:?})")), } } @@ -257,13 +281,17 @@ fn open_volume(role: Role, guid: Option<&str>, roots: &[&str]) -> Box match served(*one) { + match data_partitions().as_deref() { + Ok([one]) => match served(*one) { Some(disk) => data_on(disk, roots), None => Box::new(Absent::new(roots, "the DATA partition would not open".into())), }, - [] => ram(roots, "this machine has no DATA partition"), - many => ram(roots, &format!("this machine has {} DATA partitions, and a volume is one", many.len())), + Ok([]) => ram(roots, "this machine has no DATA partition"), + Ok(many) => ram(roots, &format!("this machine has {} DATA partitions, and a volume is one", many.len())), + Err(why) => { + println!("fsd: {why}; DATA is absent this boot"); + Box::new(Absent::new(roots, why.clone())) + } } } Role::Log | Role::Boot => { @@ -325,9 +353,15 @@ struct Server { dirty_since: Option, /// [`END_ON`]'s path on the volume. end_on: Option, + /// [`END_AT_HELLO`]'s directory. + end_at_hello: Option, /// Asks an acceptor whether a connection waits, before [`Server::accept`] /// takes it. probe: Poller, + /// A write's bytes, copied out of the client's window into this process's + /// own memory before the volume sees them. Kept, so a write allocates + /// nothing. + scratch: Vec, } /// What one request is answered. @@ -339,10 +373,25 @@ enum Answer { WithHandle(Reply, toyos::RawHandle), /// The client broke the protocol and is let go. Drop(&'static str), + /// The client is answered this refusal and let go, for this reason. + Refuse(SyscallError, String), } fn stat_reply(meta: Meta) -> Reply { - Reply { status: 0, kind: meta.kind.wire(), value: 0, value2: meta.size, mtime: meta.mtime } + Reply { kind: meta.kind.wire(), value2: meta.size, mtime: meta.mtime, ..Reply::ok() } +} + +/// Whether this is [`END_AT_HELLO`]'s first firing this boot for `dir`: its +/// mark is made once in the kernel's `/tmp`, which outlives this process and +/// not the boot. A mark that can be neither made nor found is an actuator +/// that cannot do its one job. +fn first_hello_this_boot(dir: &str) -> bool { + let mark = format!("/tmp/fsd-end-at-hello{}", dir.replace('/', "-")); + match std::fs::OpenOptions::new().write(true).create_new(true).open(&mark) { + Ok(_) => true, + Err(e) if e.kind() == std::io::ErrorKind::AlreadyExists => false, + Err(e) => panic!("fsd: {END_AT_HELLO}: its mark {mark} would not be made: {e}"), + } } fn window_of(w: &SharedMemory) -> Window { @@ -351,12 +400,30 @@ fn window_of(w: &SharedMemory) -> Window { unsafe { Window::new(w.as_ptr(), WINDOW_BYTES) } } +/// A read's destination: the first bytes of a client's window, which the +/// volume copies its cache's blocks into once and never holds a reference to. +struct WindowOut(Window); + +impl Out for WindowOut { + fn len(&self) -> usize { + self.0.bytes() + } + + fn put(&mut self, at: usize, bytes: &[u8]) { + window_put(self.0, at, bytes); + } + + fn zero(&mut self, at: usize, len: usize) { + self.0.sub(at, len).zero(); + } +} + impl Server { fn serve(mut self) -> ! { let poller = Poller::new(Poller::MAX_HANDLES); let mut ready = Vec::new(); loop { - if self.clients.len() < MAX_CLIENTS { + if self.clients.values().filter(|c| c.window.is_none()).count() < MAX_HANDSHAKES { for (i, cap) in self.caps.iter().enumerate() { poller.watch(&cap.acceptor, READABLE, i as u64); } @@ -482,6 +549,11 @@ impl Server { Answer::Link(len) => client.conn.try_send(LINK, &Reply { value: len as u64, ..Reply::ok() }), Answer::WithHandle(reply, handle) => client.conn.try_send_with_handles(&[handle], REPLY, &reply), Answer::Drop(why) => return self.drop_client(id, why), + Answer::Refuse(e, why) => { + // Let go whether or not the refusal went: either way it is said. + let _ = client.conn.try_send(REPLY, &Reply::refused(e)); + return self.drop_client(id, &why); + } }; if let Err(why) = sent { self.drop_client(id, &format!("its connection would not take a reply ({why:?})")); @@ -526,16 +598,17 @@ impl Server { window_put(window_of(window), 0, bytes); } - fn writable(&self, id: u64) -> bool { - self.caps[self.clients[&id].cap].writable && self.volume.writable() - } - fn answer(&mut self, id: u64, op: u32, r: Request) -> Answer { if op == HELLO { + let served = self.clients.values().filter(|c| c.window.is_some()).count(); let client = self.clients.get_mut(&id).expect("pumped"); if client.window.is_some() { return Answer::Drop("it lent a second window"); } + if served >= MAX_SERVED { + let why = format!("it is refused, since {MAX_SERVED} clients are served already"); + return Answer::Refuse(SyscallError::ResourceExhausted, why); + } let Some([lent]) = client.conn.recv_handles_exact::<1>() else { return Answer::Drop("its hello carried no window"); }; @@ -543,7 +616,12 @@ impl Server { Ok(window) => client.window = Some(window), Err(_) => return Answer::Drop("its window would not map"), } - let rights = if self.writable(id) { RIGHT_WRITE } else { 0 }; + let dir = &self.caps[self.clients[&id].cap].dir; + if self.end_at_hello.as_ref() == Some(dir) && first_hello_this_boot(dir) { + println!("fsd: {END_AT_HELLO}: ending before {dir}'s first hello is answered"); + std::process::exit(1); + } + let rights = if self.volume.writable() { RIGHT_WRITE } else { 0 }; return Answer::Reply(Reply { value: rights, ..Reply::ok() }); } if self.clients[&id].window.is_none() { @@ -576,7 +654,7 @@ impl Server { fn serve_one(&mut self, id: u64, op: u32, r: Request) -> Result { let changes = matches!(op, WRITE | TRUNCATE | MKDIR | RMDIR | UNLINK | RENAME | SYMLINK | STREAM) || (op == OPEN && r.flags & (O_WRITE | O_APPEND | O_CREATE | O_TRUNCATE | O_CREATE_NEW) != 0); - if changes && !self.writable(id) { + if changes && !self.volume.writable() { return Err(SyscallError::PermissionDenied); } match op { @@ -598,8 +676,9 @@ impl Server { truncate: r.flags & O_TRUNCATE != 0, }; let node = self.volume.open(&path, how)?; - let meta = match self.volume.node_meta(node) { - Ok(meta) => meta, + let stated = self.volume.node_meta(node).and_then(|meta| self.volume.ident(node).map(|ident| (meta, ident))); + let (meta, ident) = match stated { + Ok(stated) => stated, Err(e) => { self.volume.close(node); return Err(e); @@ -616,7 +695,7 @@ impl Server { let ends = append && self.end_on.as_deref() == Some(path.as_str()); let client = self.clients.get_mut(&id).expect("pumped"); client.fids.insert(fid, Fid { node, write, append, ends }); - Ok(Answer::Reply(Reply { value: fid, ..stat_reply(meta) })) + Ok(Answer::Reply(Reply { value: fid, ident, ..stat_reply(meta) })) } CLOSE => { let client = self.clients.get_mut(&id).expect("pumped"); @@ -627,9 +706,8 @@ impl Server { READ => { let node = self.fid(id, r.fid)?.node; let len = (r.len as usize).min(WINDOW_BYTES); - let mut buf = vec![0u8; len]; - let n = self.volume.read(node, r.offset, &mut buf)?; - self.put(id, &buf[..n]); + let window = window_of(self.clients[&id].window.as_ref().expect("lent")).sub(0, len); + let n = self.volume.read(node, r.offset, &mut WindowOut(window))?; Ok(Answer::Reply(Reply { value: n as u64, ..Reply::ok() })) } WRITE => { @@ -644,23 +722,28 @@ impl Server { if len > WINDOW_BYTES { return Err(SyscallError::InvalidArgument); } - let mut data = vec![0u8; len]; - window_take(window_of(self.clients[&id].window.as_ref().expect("lent")), 0, &mut data); + if self.scratch.len() < len { + self.scratch.resize(len, 0); + } + let window = window_of(self.clients[&id].window.as_ref().expect("lent")); + window_take(window, 0, &mut self.scratch[..len]); let at = if append { self.volume.node_meta(node)?.size } else { r.offset }; if at.checked_add(len as u64).is_none_or(|end| end > MAX_FILE_BYTES) { return Err(SyscallError::InvalidArgument); } - self.volume.write(node, at, &data)?; + self.volume.write(node, at, &self.scratch[..len])?; if ends { println!("fsd: {END_ON}: ending with a write done and unanswered"); std::process::exit(1); } self.dirtied(); - Ok(Answer::Reply(Reply { value: len as u64, value2: at + len as u64, ..Reply::ok() })) + let ident = self.volume.ident(node)?; + Ok(Answer::Reply(Reply { value: len as u64, value2: at + len as u64, ident, ..Reply::ok() })) } FSTAT => { let node = self.fid(id, r.fid)?.node; - Ok(Answer::Reply(stat_reply(self.volume.node_meta(node)?))) + let meta = self.volume.node_meta(node)?; + Ok(Answer::Reply(Reply { ident: self.volume.ident(node)?, ..stat_reply(meta) })) } TRUNCATE => { let f = self.fid(id, r.fid)?; @@ -673,7 +756,7 @@ impl Server { } self.volume.truncate(node, r.offset)?; self.dirtied(); - Ok(Answer::Reply(Reply::ok())) + Ok(Answer::Reply(Reply { ident: self.volume.ident(node)?, ..Reply::ok() })) } FSYNC => { self.fid(id, r.fid)?; @@ -685,6 +768,11 @@ impl Server { if !f.write { return Err(SyscallError::PermissionDenied); } + // Every client number this server keeps is bounded where it is + // kept: a stream's offset grows by what it drains. + if r.offset > MAX_FILE_BYTES { + return Err(SyscallError::InvalidArgument); + } let node = f.node; if self.streams.len() >= MAX_STREAMS { return Err(SyscallError::ResourceExhausted); @@ -800,28 +888,35 @@ impl Server { Answer::Link(path.len()) } - /// What a stream's writer has put in the pipe, appended to its file. + /// One read of what a stream's writer has put in the pipe, appended to + /// its file, and no more: a pipe with more waiting fires at the next + /// wait, after every other client that was ready — [`Self::pump`]'s rule. fn drain(&mut self, sid: u64) { let mut buf = vec![0u8; 64 * 1024]; - loop { - let Some(stream) = self.streams.get_mut(&sid) else { return }; - match stream.pipe.read_nonblock(&mut buf) { - Ok(0) | Err(SyscallError::Gone) => break, - Ok(n) => { - let (node, at) = (stream.node, stream.offset); - stream.offset += n as u64; - if let Err(e) = self.volume.write(node, at, &buf[..n]) { - println!("fsd: a stream's write failed ({e:?}); the stream is ended"); - break; + let Some(stream) = self.streams.get_mut(&sid) else { return }; + let ended = match stream.pipe.read_nonblock(&mut buf) { + Err(SyscallError::WouldBlock) => return, + Ok(0) | Err(SyscallError::Gone) => None, + Err(e) => Some(format!("its pipe would not read ({e:?})")), + Ok(n) => { + let (node, at) = (stream.node, stream.offset); + match at.checked_add(n as u64).filter(|&end| end <= MAX_FILE_BYTES) { + None => Some("it reached the largest offset a file has".to_string()), + Some(end) => { + stream.offset = end; + match self.volume.write(node, at, &buf[..n]) { + Ok(()) => { + self.dirtied(); + return; + } + Err(e) => Some(format!("its write failed ({e:?})")), + } } - self.dirtied(); - } - Err(SyscallError::WouldBlock) => return, - Err(e) => { - println!("fsd: a stream's pipe would not read ({e:?}); the stream is ended"); - break; } } + }; + if let Some(why) = ended { + println!("fsd: a stream is ended: {why}"); } if let Some(stream) = self.streams.remove(&sid) { self.volume.close(stream.node); diff --git a/userland/fsd/src/volume.rs b/userland/fsd/src/volume.rs index 5cd8f066cbe..d66ca0be23a 100644 --- a/userland/fsd/src/volume.rs +++ b/userland/fsd/src/volume.rs @@ -45,6 +45,34 @@ pub struct OpenHow { /// An open file, while any client holds it. pub type Node = u64; +/// Where a read's bytes go, `len()` of them at most: the client's window, which +/// is shared with the client and so is written through a copy and never +/// through a reference, or a buffer of this process's own. +pub trait Out { + fn len(&self) -> usize; + /// `bytes` at `at`, inside `len()`. + fn put(&mut self, at: usize, bytes: &[u8]); + /// `len` zeros at `at`, inside `len()`: a hole's bytes. + fn zero(&mut self, at: usize, len: usize); +} + +/// A buffer of this process's own as a read's destination. +pub struct Buf<'a>(pub &'a mut [u8]); + +impl Out for Buf<'_> { + fn len(&self) -> usize { + self.0.len() + } + + fn put(&mut self, at: usize, bytes: &[u8]) { + self.0[at..at + bytes.len()].copy_from_slice(bytes); + } + + fn zero(&mut self, at: usize, len: usize) { + self.0[at..at + len].fill(0); + } +} + pub trait Volume { /// Whether anything here may be changed. fn writable(&self) -> bool; @@ -68,8 +96,15 @@ pub trait Volume { fn node_meta(&mut self, node: Node) -> Result; - /// Up to `out.len()` bytes at `offset`; 0 at or past the end. - fn read(&mut self, node: Node, offset: u64, out: &mut [u8]) -> Result; + /// The file as it stands, as `toyos::fs::Stat::ident` states it: what + /// the volume itself records of it, so the same after a restart for the + /// same file unchanged; 0 where the volume records nothing that tells + /// this file from another. + fn ident(&mut self, node: Node) -> Result; + + /// Up to `out.len()` bytes at `offset`, from the start of `out`; 0 at or + /// past the end. + fn read(&mut self, node: Node, offset: u64, out: &mut dyn Out) -> Result; fn write(&mut self, node: Node, offset: u64, data: &[u8]) -> Result<(), SyscallError>; @@ -94,6 +129,20 @@ pub trait Volume { fn describe(&self) -> String; } +/// One token for `words`, never 0: 0 is a volume saying it cannot tell. Two +/// different lists meet on one token as two 64-bit hashes do. +pub fn identity(words: &[u64]) -> u64 { + // splitmix64's finaliser, chained over the words. + let mut h = 0x243F_6A88_85A3_08D3u64; + for &w in words { + let mut z = (h ^ w).wrapping_add(0x9E37_79B9_7F4A_7C15); + z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); + z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); + h = z ^ (z >> 31); + } + h.max(1) +} + /// The parent of `path`, `""` for a name at the root. pub fn parent(path: &str) -> &str { path.rsplit_once('/').map_or("", |(dir, _)| dir) diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index eb060f7eae1..bf197008edb 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -32,6 +32,13 @@ //! server nothing keeps. A program started through `launcher` still gets its //! acceptor by move and is started once per boot. //! +//! **init never waits on a file server it supervises from its loop alone**: +//! the loop is also where a server that ended is started again. Its calls into +//! the file servers run on one worker thread ([`Worker`]), and init waits on +//! the call and on a service's end together ([`Init::files`]), so a call a +//! server's end left in the port's queue goes on to the server started in its +//! place, and one alive and silent costs a bounded wait ([`FILES_BOUND`]). +//! //! **Every program it starts gets `HOME` from its row** (`Program::home`), over //! anything a launching caller carried: a service its own `/state/`, made //! before it runs, and everything else the session user's home, made at boot. @@ -290,8 +297,8 @@ impl Log { } /// The session user's home and [`HOME_FOLDERS`]. A boot whose DATA volume did -/// not mount has no `/home`, which the kernel has already said; this says what -/// it cost and starts the machine without it. +/// not mount has no `/home`, which its file server has already said; this says +/// what it cost and starts the machine without it. fn make_session_home() { let home = toyos_manifest::session_home(); if let Err(e) = make_dir(&home) { @@ -305,9 +312,7 @@ fn make_session_home() { } } -/// One directory, whose parent is there already. Not `create_dir_all`: DATA -/// keeps no directories, and the VFS's own record of one takes a child whose -/// parent it never made, so every level is made in turn. +/// One directory, whose parent is there already. fn make_dir(path: &str) -> std::io::Result<()> { match std::fs::create_dir(path) { Err(e) if e.kind() != std::io::ErrorKind::AlreadyExists => Err(e), @@ -318,7 +323,7 @@ fn make_dir(path: &str) -> std::io::Result<()> { fn main() { // Before anything is started, and before init says anything: from here // init's own lines are records in its ring. - let mut log = Log::open(); + let log = Log::open(); let syscap: SysCap = Endowments::get() .take(SYSCAP_LABEL) @@ -326,7 +331,9 @@ fn main() { let text = read_manifest() .unwrap_or_else(|e| panic!("init: cannot read {}: {e:?}", toyos_manifest::GUEST_PATH)); - let system = toyos_manifest::parse(&text); + // For the machine's life, and `'static` so a launch's path can be resolved + // against it on init's file worker. + let system: &'static Manifest = Box::leak(Box::new(toyos_manifest::parse(&text))); // Before anything is spawned, and for every `serves` name in the manifest // rather than only the ones `[boot] start` names: the filepicker is @@ -369,92 +376,109 @@ fn main() { } builder.finish().expect("init: no namespace for its own files") }; - if let Err(e) = std::os::toyos::fs::adopt_namespace(build().into_raw().0) { + // SAFETY: a namespace handle built here and used nowhere else. + if let Err(e) = unsafe { std::os::toyos::fs::adopt_namespace(build().into_raw().0) } { panic!("init: std would not take its namespace ({e}), and init is endowed none"); } Box::leak(Box::new(build())) }; let (wake_read, wake_write) = toyos::pipe_pair().expect("init: no pipe to hear a service end"); - let wake = wake_write.into_raw(); - - // Kept for the machine's life: init is the only thing that can kill a - // daemon, and there is no other way back to a process it started. The - // storage services go first: every other start may make a directory, and - // a directory is a file server's. - let mut services: Vec = Vec::new(); - let storage_first = system - .start - .iter() - .filter(|n| system.program(n).is_some_and(is_storage)) - .chain(system.start.iter().filter(|n| system.program(n).is_none_or(|p| !is_storage(p)))); - let mut homes_made = false; - for name in storage_first { - let program = system - .program(name) - .unwrap_or_else(|| panic!("init: [boot] start names `{name}`, which is not declared")); - if !is_storage(program) && !homes_made { - make_session_home(); - homes_made = true; - } - let instances: Vec> = match program.roles.is_empty() { - true => vec![None], - false => program.roles.iter().map(|r| Some(r.as_str())).collect(), - }; - for role in instances { - let kept: Vec<(String, Acceptor)> = match role { - None => program - .serves - .iter() - .map(|served| { - let acceptor = acceptors.remove(served.as_str()).unwrap_or_else(|| { - panic!("init: `{}` has already been given the `{served}` acceptor", program.name) - }); - (served.clone(), acceptor) - }) - .collect(), - Some(role) => toyos_manifest::role_dirs(role) - .expect("known role") - .iter() - .map(|dir| { - let name = format!("{CAPABILITY_PREFIX}{dir}"); - let acceptor = dir_acceptors.remove(&name).unwrap_or_else(|| { - panic!("init: the `{role}` role is served twice") - }); - (name, acceptor) - }) - .collect(), - }; - let mut service = Service::new(program, role, kept, wake); - if let Err(e) = - service.spawn(&program.path, &[], &system, &syscap, &connectors, &dirs, &mut log) - { - panic!("init: cannot start {}: {e}", program.name); - } - services.push(service); - } - } - if !homes_made { - make_session_home(); - } + let mut init = Init { + system, + syscap: &syscap, + acceptors, + connectors, + dirs, + files, + services: Vec::new(), + log, + wake: wake_read, + worker: Worker::start(), + restarting: false, + }; + init.boot(&mut dir_acceptors, wake_write.into_raw()); // Nothing else holds a `serves` acceptor that has not been launched yet, so // init outliving its children is what keeps those ports open. It parks // here. - let launcher = acceptors + let launcher = init + .acceptors .remove(LAUNCHER) .expect("init: the manifest declares init serves `launcher`"); - let swap = acceptors + let swap = init + .acceptors .remove(toyos_swap::PORT) .expect("init: the manifest declares init serves `swap`"); - let power = acceptors + let power = init + .acceptors .remove(power::PORT) .expect("init: the manifest declares init serves `power`"); - let mut init = - Init { system: &system, syscap: &syscap, acceptors, connectors, dirs, files, services, log, wake: wake_read }; init.serve_forever(&launcher, &swap, &power); } +/// How long init waits on one call into a file server that is alive and has +/// not answered, before it answers without it. +/// +/// **A policy number, and not the bound on a server that ended**: one that +/// ends is started again while init waits (`Init::files`), and the call then +/// goes on to the new process. What this bounds is a server alive and silent, +/// which would otherwise hold the machine's only way to start a process. A +/// first boot's DATA server formats and mounts before its first answer, so the +/// bound is generous. +const FILES_BOUND: Duration = Duration::from_secs(30); + +/// A job on init's file worker. +type Job = Box; + +/// The one thread init makes its calls into the file servers on. +/// +/// **A call into a file server is a wait on a process init supervises**, and +/// init's loop is also where that process is started again when it ends: a +/// call made from the loop, to a server that ended under it, waits in the +/// port's queue for a server only the loop can start. So the call runs here +/// and the loop waits on it and on the end together ([`Init::files`]). +struct Worker { + jobs: std::sync::mpsc::Sender, + /// Readable once per job finished. + done: Pipe, + /// Jobs handed over and not seen finished: one that outlived its bound is + /// still here, and the worker takes no other until it finishes. + owed: u32, + /// What the job still owed is, for the refusal a second one gets. + owed_what: String, + /// Waits on `done` and on a service's end together. + poller: Poller, +} + +impl Worker { + fn start() -> Self { + let (jobs, queue) = std::sync::mpsc::channel::(); + let (done, finished) = toyos::pipe_pair().expect("init: no pipe for its file worker"); + std::thread::Builder::new() + .name("files".into()) + .spawn(move || { + for job in queue { + job(); + finished.write(&[1]).expect("init: its file worker could not say a job finished"); + } + }) + .expect("init: its file worker could not be started"); + Self { jobs, done, owed: 0, owed_what: String::new(), poller: Poller::new(2) } + } + + /// Count the jobs that have finished since last asked. + fn settle(&mut self) { + let mut seen = [0u8; 8]; + while self.owed > 0 { + match self.done.read_nonblock(&mut seen) { + Ok(n) if n > 0 => self.owed -= n as u32, + _ => return, + } + } + } +} + /// A row whose process the machine's files depend on: a block service, or a /// file server. Started before every other row, and never made a home, which /// would be a directory on the volume it serves. @@ -464,7 +488,7 @@ fn is_storage(program: &Program) -> bool { /// Everything init's loop acts on, for the machine's life. struct Init<'a> { - system: &'a Manifest, + system: &'static Manifest, syscap: &'a SysCap, /// The `serves` acceptors nobody has been started with yet, which a launch /// takes by move. @@ -482,6 +506,11 @@ struct Init<'a> { log: Log, /// Readable when a `restart` service has ended and is owed a new process. wake: Pipe, + /// Where init's calls into the file servers are made. + worker: Worker, + /// Inside [`Init::restart_ended`]: a file call made there leaves the wake + /// for the loop, which is the one caller that acts on it. + restarting: bool, } /// One program init started at boot, and the ports it serves. @@ -675,6 +704,135 @@ impl Flight { } impl<'a> Init<'a> { + /// Start `[boot] start`, kept for the machine's life: init is the only + /// thing that can kill a daemon, and there is no other way back to a + /// process it started. The storage rows go first — every other start may + /// make a directory, and a directory is a file server's — and the + /// session's home is made before the first row that is not one. + fn boot(&mut self, dir_acceptors: &mut BTreeMap, wake: toyos::RawHandle) { + let system = self.system; + let storage_first = system + .start + .iter() + .filter(|n| system.program(n).is_some_and(is_storage)) + .chain(system.start.iter().filter(|n| system.program(n).is_none_or(|p| !is_storage(p)))); + let mut homes_made = false; + for name in storage_first { + let program = system + .program(name) + .unwrap_or_else(|| panic!("init: [boot] start names `{name}`, which is not declared")); + if !is_storage(program) && !homes_made { + self.make_session_home(); + homes_made = true; + } + let instances: Vec> = match program.roles.is_empty() { + true => vec![None], + false => program.roles.iter().map(|r| Some(r.as_str())).collect(), + }; + for role in instances { + let kept: Vec<(String, Acceptor)> = match role { + None => program + .serves + .iter() + .map(|served| { + let acceptor = self.acceptors.remove(served.as_str()).unwrap_or_else(|| { + panic!("init: `{}` has already been given the `{served}` acceptor", program.name) + }); + (served.clone(), acceptor) + }) + .collect(), + Some(role) => toyos_manifest::role_dirs(role) + .expect("known role") + .iter() + .map(|dir| { + let name = format!("{CAPABILITY_PREFIX}{dir}"); + let acceptor = dir_acceptors + .remove(&name) + .unwrap_or_else(|| panic!("init: the `{role}` role is served twice")); + (name, acceptor) + }) + .collect(), + }; + self.make_home(program); + let mut service = Service::new(program, role, kept, wake); + let started = + service.spawn(&program.path, &[], system, self.syscap, &self.connectors, &self.dirs, &mut self.log); + if let Err(e) = started { + panic!("init: cannot start {}: {e}", program.name); + } + self.services.push(service); + } + } + if !homes_made { + self.make_session_home(); + } + } + + fn make_session_home(&mut self) { + if let Err(why) = self.files("the session home", make_session_home) { + say!("init: the session home was not made: {why}"); + } + } + + /// A service's own `HOME`, made before it first runs. + fn make_home(&mut self, program: &Program) { + if !program.service || is_storage(program) { + return; + } + let home = program.home(); + let asked = home.clone(); + match self.files("a service's home", move || make_dir(&asked)) { + Ok(Ok(())) => {} + Ok(Err(e)) => say!("init: {}: {home} could not be made: {e}", program.name), + Err(why) => say!("init: {}: {home} was not made: {why}", program.name), + } + } + + /// `work`, a call into the file servers, made on [`Worker`], and its + /// answer — with every server that ends meanwhile started again, so a call + /// its end left waiting in the port's queue goes on to the new process. + /// `Err` is the call still unanswered at [`FILES_BOUND`], which the worker + /// goes on waiting for alone, or the worker still waiting on an earlier + /// one. + fn files(&mut self, what: &str, work: impl FnOnce() -> T + Send + 'static) -> Result { + const DONE: u64 = 0; + const WAKE: u64 = 1; + self.worker.settle(); + if self.worker.owed > 0 { + return Err(format!("init's file worker still waits on {}", self.worker.owed_what)); + } + let (answer, answered) = std::sync::mpsc::channel(); + let job: Job = Box::new(move || { + // The receiver is gone once the bound has passed. + let _ = answer.send(work()); + }); + self.worker.jobs.send(job).expect("init: its file worker has ended"); + self.worker.owed = 1; + self.worker.owed_what = what.to_string(); + let deadline = Instant::now() + FILES_BOUND; + loop { + self.worker.poller.watch(&self.worker.done, READABLE, DONE); + if !self.restarting { + self.worker.poller.watch(&self.wake, READABLE, WAKE); + } + let left = deadline.saturating_duration_since(Instant::now()); + let mut woken = false; + self.worker.poller.wait(1, left.as_nanos() as u64, |token| woken |= token == WAKE); + if woken { + let mut sink = [0u8; 64]; + while matches!(self.wake.read_nonblock(&mut sink), Ok(n) if n > 0) {} + self.restart_ended(); + } + self.worker.settle(); + if self.worker.owed == 0 { + return Ok(answered.recv().expect("init: a finished file job answered")); + } + if Instant::now() >= deadline { + return Err(format!("{what} was not answered in {} s", FILES_BOUND.as_secs())); + } + } + } + /// Serve `launcher` and `swap` for the rest of the machine's life. /// /// **An event loop and not an accept loop, because the rule every other @@ -786,17 +944,7 @@ impl<'a> Init<'a> { RxStep::Frame { msg_type, payload_len } => { let p = pending.remove(i); match p.port { - Port::Launcher => serve_launch( - &p.conn, - msg_type, - p.rx.payload(payload_len), - self.system, - self.syscap, - &mut self.acceptors, - &self.connectors, - &self.dirs, - &mut self.log, - ), + Port::Launcher => self.serve_launch(&p.conn, msg_type, p.rx.payload(payload_len)), Port::Power => self.stop(&p.conn, msg_type), Port::Swap => { let payload = p.rx.payload(payload_len).to_vec(); @@ -945,6 +1093,8 @@ impl<'a> Init<'a> { // What the process being stopped holds, which its replacement and the // binary a failed swap starts again are each owed. let owed = self.services[index].devices.clone(); + let program = self.services[index].program; + self.make_home(program); let service = &mut self.services[index]; swapped( &self.log, @@ -1100,6 +1250,7 @@ impl<'a> Init<'a> { /// [`toyos_manifest::RESTART_WINDOW_SECS`]: then its ports close, and a /// client's next connection is answered `Gone`. fn restart_ended(&mut self) { + self.restarting = true; let window = Duration::from_secs(toyos_manifest::RESTART_WINDOW_SECS); for index in 0..self.services.len() { let ended = self.services[index].kept.lock().expect("init: a service's state is poisoned").ended; @@ -1128,7 +1279,9 @@ impl<'a> Init<'a> { continue; } service.restarts.push_back(now); - let (path, owed) = (service.path.clone(), service.devices.clone()); + let (path, owed, program) = (service.path.clone(), service.devices.clone(), service.program); + self.make_home(program); + let service = &mut self.services[index]; match service.spawn(&path, &owed, self.system, self.syscap, &self.connectors, &self.dirs, &mut self.log) { Ok(new) => say!("init: {label} (pid {pid}) ended; started again as pid {new}"), Err(e) => { @@ -1139,11 +1292,14 @@ impl<'a> Init<'a> { } } } + self.restarting = false; } /// Start the binary a failed swap replaced, or close the service's ports /// when that will not start either. fn restore(&mut self, index: usize, previous: &str, owed: &[String]) { + let program = self.services[index].program; + self.make_home(program); let service = &mut self.services[index]; let name = service.program.name.clone(); match service.spawn(previous, owed, self.system, self.syscap, &self.connectors, &self.dirs, &mut self.log) { @@ -1167,7 +1323,8 @@ impl<'a> Init<'a> { } /// A staged or installed binary's bytes, refused past [`toyos_swap::MAX_BINARY_BYTES`] -/// before any of it is read. +/// before any of it is read. Made from the loop: a swap's files are under +/// [`toyos_swap::STAGING`], which the kernel serves. fn read_binary(path: &str) -> Result, Refusal> { let unreadable = |e: std::io::Error| Refusal::Unreadable { path: path.to_string(), why: e.to_string() }; let len = std::fs::metadata(path).map_err(unreadable)?.len(); @@ -1199,165 +1356,169 @@ fn forget(path: &str) { } } -/// One `MSG_LAUNCH`, from the frame to the `Process` handle that answers it. -/// -/// **Everything in the request is a client's claim about itself.** A program -/// nothing declares is refused by name; a frame that does not decode is a -/// dropped connection and nothing else. What the child ends up holding is the -/// manifest's row for it plus whatever connectors the caller transferred, and -/// the caller could only transfer what it already had — so a launch confers -/// exactly the manifest row and nothing beyond it. -fn serve_launch<'a>( - conn: &Connection, - msg_type: u32, - payload: &[u8], - system: &'a Manifest, - syscap: &SysCap, - acceptors: &mut BTreeMap<&'a str, Acceptor>, - connectors: &BTreeMap<&str, Connector>, - dirs: &BTreeMap, - log: &mut Log, -) { - if msg_type != launch::MSG_LAUNCH { - return; - } - let mut batch = [toyos::RawHandle(0); toyos_abi::syscall::MAX_TRANSFER_HANDLES]; - let received = conn.recv_handles(&mut batch).unwrap_or(0); - - // **Owned on the statement after they arrive, and before anything can - // refuse.** The send moved them into init's table, so every path out of - // here releases them — a launcher that leaked a handle per refused launch - // would exhaust the one table the machine cannot do without, and a client - // picks which refusal it takes. - let mut held = Moved(batch[..received].to_vec()); - - let Some(request) = Request::decode(payload) else { return }; - // `extra_names` drops an empty or non-UTF-8 name, so its count is what - // will actually be paired with a handle. A frame whose two counts - // disagree would otherwise leave the unpaired handles behind. - let names: Vec<&str> = request.extra_names().collect(); - if received != request.slot_count() + request.extra_count - || names.len() != request.extra_count - { - say!( - "init: launcher: a frame promising {} handles under {} names carried {received}", - request.slot_count() + request.extra_count, - names.len(), - ); - return; - } - - // Past every refusal that does not know which handle is which, so ownership - // can be split. Both halves still release on every path below. - let all = held.take(); - let (slot_handles, extra_handles) = all.split_at(request.slot_count()); - let slots = Moved(slot_handles.to_vec()); - // Owned, so they close when this call returns: `SYS_NAMESPACE_BUILD` copies - // a connector into the namespace and leaves the caller's handle, and init's - // copy of a client's connector has no life beyond this launch. - let extras: Vec<(&str, Connector)> = names - .into_iter() - .zip(extra_handles.iter().copied()) - // SAFETY: the kernel moved these into init's table with the frame, and - // nothing else answers for them. **Not a claim about the type** — a - // client sends what it likes, and everything below treats a wrong one - // as a refused launch rather than as init's own bug. - .map(|(name, handle)| (name, unsafe { Connector::from_raw(handle) })) - .collect(); - - let installed; - let program = match resolve(system, request.program) { - Resolved::Row(row) => row, - Resolved::Package(row) => { - installed = row; - &installed - } - Resolved::NotDeclared => { - // **`try_send_bytes` and not `send`.** A blocking write is the - // other half of the rule that made the read side an event loop: a - // client that never drains its end decides when init runs again. - // The `HOME` is the session's: a program no row names is no service. - let home = toyos_manifest::session_home(); - let _ = conn.try_send_bytes(launch::MSG_NOT_DECLARED, home.as_bytes()); +impl Init<'_> { + /// One `MSG_LAUNCH`, from the frame to the `Process` handle that answers it. + /// + /// **Everything in the request is a client's claim about itself.** A program + /// nothing declares is refused by name; a frame that does not decode is a + /// dropped connection and nothing else. What the child ends up holding is the + /// manifest's row for it plus whatever connectors the caller transferred, and + /// the caller could only transfer what it already had — so a launch confers + /// exactly the manifest row and nothing beyond it. + fn serve_launch(&mut self, conn: &Connection, msg_type: u32, payload: &[u8]) { + if msg_type != launch::MSG_LAUNCH { return; } - Resolved::Refused(why) => { - say!("init: launcher: {why}"); - let _ = conn.try_signal(launch::MSG_REFUSED); + let mut batch = [toyos::RawHandle(0); toyos_abi::syscall::MAX_TRANSFER_HANDLES]; + let received = conn.recv_handles(&mut batch).unwrap_or(0); + + // **Owned on the statement after they arrive, and before anything can + // refuse.** The send moved them into init's table, so every path out of + // here releases them — a launcher that leaked a handle per refused launch + // would exhaust the one table the machine cannot do without, and a client + // picks which refusal it takes. + let mut held = Moved(batch[..received].to_vec()); + + let Some(request) = Request::decode(payload) else { return }; + // `extra_names` drops an empty or non-UTF-8 name, so its count is what + // will actually be paired with a handle. A frame whose two counts + // disagree would otherwise leave the unpaired handles behind. + let names: Vec<&str> = request.extra_names().collect(); + if received != request.slot_count() + request.extra_count + || names.len() != request.extra_count + { + say!( + "init: launcher: a frame promising {} handles under {} names carried {received}", + request.slot_count() + request.extra_count, + names.len(), + ); return; } - }; - // std joins a relative `current_dir` onto init's own cwd, so passing one - // on would start the child in init's `/` — the default this field removes. - if !request.cwd.starts_with('/') { - say!("init: launcher: refused a working directory that is not absolute"); - let _ = conn.try_signal(launch::MSG_REFUSED); - return; - } + // Past every refusal that does not know which handle is which, so ownership + // can be split. Both halves still release on every path below. + let all = held.take(); + let (slot_handles, extra_handles) = all.split_at(request.slot_count()); + let slots = Moved(slot_handles.to_vec()); + // Owned, so they close when this call returns: `SYS_NAMESPACE_BUILD` copies + // a connector into the namespace and leaves the caller's handle, and init's + // copy of a client's connector has no life beyond this launch. + let extras: Vec<(&str, Connector)> = names + .into_iter() + .zip(extra_handles.iter().copied()) + // SAFETY: the kernel moved these into init's table with the frame, and + // nothing else answers for them. **Not a claim about the type** — a + // client sends what it likes, and everything below treats a wrong one + // as a refused launch rather than as init's own bug. + .map(|(name, handle)| (name, unsafe { Connector::from_raw(handle) })) + .collect(); + + let installed; + // On the worker: a path under `/apps` is read off its package, and that is + // a call into a file server init supervises. + let (system, path) = (self.system, request.program.to_string()); + let resolved = match self.files("a launch's path", move || resolve(system, &path)) { + Ok(resolved) => resolved, + Err(why) => { + say!("init: launcher: {} was not resolved: {why}", request.program); + let _ = conn.try_signal(launch::MSG_REFUSED); + return; + } + }; + let program = match resolved { + Resolved::Row(row) => row, + Resolved::Package(row) => { + installed = row; + &installed + } + Resolved::NotDeclared => { + // **`try_send_bytes` and not `send`.** A blocking write is the + // other half of the rule that made the read side an event loop: a + // client that never drains its end decides when init runs again. + // The `HOME` is the session's: a program no row names is no service. + let home = toyos_manifest::session_home(); + let _ = conn.try_send_bytes(launch::MSG_NOT_DECLARED, home.as_bytes()); + return; + } + Resolved::Refused(why) => { + say!("init: launcher: {why}"); + let _ = conn.try_signal(launch::MSG_REFUSED); + return; + } + }; - // **The caller's path, not the row's, and `argv[0]` is why.** `declared` - // has already established that the two name one binary, so this grants - // nothing extra — and `/system/bin/echo` spawned as `/system/bin/toybox` is a toybox that - // was never told which applet it is. - let mut command = Command::new(request.program); - let caller_slots: Vec<(u32, toyos::RawHandle)> = - request.slot_numbers().zip(slots.0.iter().copied()).collect(); - // **Carried, not inherited.** A child of the launcher would otherwise get - // init's environment and init's working directory, so `cd /tmp && ls` would - // list `/`. The launcher is a spawn service, not a session. - command.env_clear(); - for entry in request.env.split(|&b| b == 0).filter(|e| !e.is_empty()) { - let Some(eq) = entry.iter().position(|&b| b == b'=') else { continue }; - if let (Ok(key), Ok(value)) = - (std::str::from_utf8(&entry[..eq]), std::str::from_utf8(&entry[eq + 1..])) - { - command.env(key, value); + // std joins a relative `current_dir` onto init's own cwd, so passing one + // on would start the child in init's `/` — the default this field removes. + if !request.cwd.starts_with('/') { + say!("init: launcher: refused a working directory that is not absolute"); + let _ = conn.try_signal(launch::MSG_REFUSED); + return; } - } - command.current_dir(request.cwd); - for arg in request.argv.split(|&b| b == 0).skip(1).filter(|a| !a.is_empty()) { - if let Ok(arg) = std::str::from_utf8(arg) { - command.arg(arg); + + // **The caller's path, not the row's, and `argv[0]` is why.** `declared` + // has already established that the two name one binary, so this grants + // nothing extra — and `/system/bin/echo` spawned as `/system/bin/toybox` is a toybox that + // was never told which applet it is. + let mut command = Command::new(request.program); + let caller_slots: Vec<(u32, toyos::RawHandle)> = + request.slot_numbers().zip(slots.0.iter().copied()).collect(); + // **Carried, not inherited.** A child of the launcher would otherwise get + // init's environment and init's working directory, so `cd /tmp && ls` would + // list `/`. The launcher is a spawn service, not a session. + command.env_clear(); + for entry in request.env.split(|&b| b == 0).filter(|e| !e.is_empty()) { + let Some(eq) = entry.iter().position(|&b| b == b'=') else { continue }; + if let (Ok(key), Ok(value)) = + (std::str::from_utf8(&entry[..eq]), std::str::from_utf8(&entry[eq + 1..])) + { + command.env(key, value); + } + } + command.current_dir(request.cwd); + for arg in request.argv.split(|&b| b == 0).skip(1).filter(|a| !a.is_empty()) { + if let Ok(arg) = std::str::from_utf8(arg) { + command.arg(arg); + } } - } - // `inherit_handle` duplicates into the child, so init's own copies go with - // `slots` when this returns. - let started = start( - command, - program, - system, - syscap, - Served::Move(acceptors), - connectors, - dirs, - &extras, - Storage::default(), - Output::Launch { log, slots: &caller_slots }, - ); - match started { - Ok((child, _)) => { - let handle = toyos::RawHandle(child.into_raw_handle()); - // **Which side owns the handle is the whole of what the two arms - // differ by.** A refused `handle_send` leaves it in init's table - // and init must close it — init keeps no `Process` handle from a - // launch, or a client launching `/system/bin/true` in a loop exhausts the - // one table the machine cannot do without. A send that *took* it - // and a frame that then did not go leaves it queued on a connection - // this call is about to drop, which releases it — and closing it - // here would be closing a handle init no longer holds, which under - // the bad-handle policy is init exiting. - match toyos_abi::syscall::handle_send(conn.as_handle(), &[handle]) { - Ok(()) => { - let _ = conn.try_signal(launch::MSG_LAUNCHED); + // `inherit_handle` duplicates into the child, so init's own copies go with + // `slots` when this returns. + self.make_home(program); + let started = start( + command, + program, + self.system, + self.syscap, + Served::Move(&mut self.acceptors), + &self.connectors, + &self.dirs, + &extras, + Storage::default(), + Output::Launch { log: &mut self.log, slots: &caller_slots }, + ); + match started { + Ok((child, _)) => { + let handle = toyos::RawHandle(child.into_raw_handle()); + // **Which side owns the handle is the whole of what the two arms + // differ by.** A refused `handle_send` leaves it in init's table + // and init must close it — init keeps no `Process` handle from a + // launch, or a client launching `/system/bin/true` in a loop exhausts the + // one table the machine cannot do without. A send that *took* it + // and a frame that then did not go leaves it queued on a connection + // this call is about to drop, which releases it — and closing it + // here would be closing a handle init no longer holds, which under + // the bad-handle policy is init exiting. + match toyos_abi::syscall::handle_send(conn.as_handle(), &[handle]) { + Ok(()) => { + let _ = conn.try_signal(launch::MSG_LAUNCHED); + } + Err(_) => toyos_abi::syscall::close(handle), } - Err(_) => toyos_abi::syscall::close(handle), } - } - Err(e) => { - say!("init: launcher: cannot start {}: {e}", program.name); - let _ = conn.try_signal(launch::MSG_REFUSED); + Err(e) => { + say!("init: launcher: cannot start {}: {e}", program.name); + let _ = conn.try_signal(launch::MSG_REFUSED); + } } } } @@ -1606,14 +1767,9 @@ fn start<'a>( let booting = matches!(output, Output::Boot(_)); // **Set here, over whatever a launching caller carried**: the row decides - // where a program's home is, and a service's is made before it first runs. - let home = program.home(); - if program.service && !is_storage(program) { - if let Err(e) = make_dir(&home) { - say!("init: {}: {home} could not be made: {e}", program.name); - } - } - command.env("HOME", &home); + // where a program's home is, and a service's is made before this + // (`Init::make_home`), on init's file worker. + command.env("HOME", program.home()); // **Everything endowed stays owned until the spawn that moves it // succeeds.** `endow` records a number; a refused spawn moves nothing From 44ff53d97b0844458160018bc2be7930772995b0 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 14:41:09 +0200 Subject: [PATCH 11/54] abi: a spawned or loaded image is the caller's memory object, not bytes the kernel copies Owner-approved revision of the image ABI this branch added. `SpawnArgs` carries `image` (a handle to a shared memory object carrying `MAP`) and `image_len` in place of a pointer and a length, and `SYS_DLOPEN`'s fourth word points at an `ImageRef { handle, len }`. The caller reads the program into an object it owns and is charged for; the kernel pages the child from that object (`SharedImage`), copying each page once as it is read, and refuses an object that is no memory it allocated or a length the object does not hold. `SYS_DLOPEN` answers a name the process already holds before it looks at the object. `ImageBacking`, which copied up to 256 MiB into kernel pages charged to nobody, and `MAX_IMAGE_BYTES` go. `spawn_image_object`: a program in an object runs; an object without `MAP` is refused `PermissionDenied` and a length past it `InvalidArgument`; an object overwritten with `hlt` right after the spawn ends its child and the next spawn runs; and a held name's `dlopen` is answered through a handle it could not read. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- ...cking-refused-is-reported-as-a-segfault.md | 2 +- kernel/src/file_backing.rs | 93 ++++++---------- kernel/src/syscall/dispatch.rs | 27 +++-- kernel/src/syscall/vm.rs | 34 ++++-- .../src/bin/abuse_elf_loader.rs | 24 +++-- .../src/bin/abuse_elf_segments.rs | 9 +- .../src/bin/abuse_handle_table.rs | 2 +- .../src/bin/abuse_page_straddle.rs | 2 +- .../src/bin/abuse_spawn_argv.rs | 2 +- .../src/bin/device_claim_lifetime.rs | 2 +- .../src/bin/handle_kill_policy.rs | 2 +- tests/toyos-rust-tests/src/bin/spawn_cwd.rs | 2 +- .../src/bin/spawn_image_object.rs | 102 ++++++++++++++++++ toyos-abi/src/syscall.rs | 36 ++++--- toyos-userbound/src/span.rs | 3 +- 15 files changed, 232 insertions(+), 110 deletions(-) create mode 100644 tests/toyos-rust-tests/src/bin/spawn_image_object.rs diff --git a/issues/kernel/a-page-in-its-backing-refused-is-reported-as-a-segfault.md b/issues/kernel/a-page-in-its-backing-refused-is-reported-as-a-segfault.md index 3090ee942c3..1bd7a24d04c 100644 --- a/issues/kernel/a-page-in-its-backing-refused-is-reported-as-a-segfault.md +++ b/issues/kernel/a-page-in-its-backing-refused-is-reported-as-a-segfault.md @@ -24,7 +24,7 @@ subsystem, and nothing ties the two together. No page this kernel faults in is read from a device any more: `/system` is the image the loader put in memory, a program or library a file server holds is -copied into memory at its spawn or `dlopen` (`ImageBacking` in +paged from the memory object its caller read it into (`SharedImage` in `kernel/src/file_backing.rs`), and `/tmp` is memory. What still refuses a page is a `/tmp` backing whose file was deleted (`revoke_selftest`), so a program that maps a `/tmp` file another deletes reaches it — reported as its diff --git a/kernel/src/file_backing.rs b/kernel/src/file_backing.rs index f420b34a6ed..b1d3a757594 100644 --- a/kernel/src/file_backing.rs +++ b/kernel/src/file_backing.rs @@ -1,8 +1,11 @@ +use alloc::sync::Arc; use alloc::vec::Vec; use bcachefs::Extent; use crate::block::BlockResult; +use crate::object::shm::SharedMemObject; use crate::rootfs::MemoryImage; +use toyos_abi::syscall::SyscallError; /// `mm::PAGE_SIZE`: `usize` for buffer sizing, `u64` for file offsets. const BLOCK_SIZE: usize = crate::mm::PAGE_SIZE as usize; @@ -78,79 +81,51 @@ impl FileBacking for ReadOnlyBacking { } } -/// An executable or library userland read and handed over whole: the bytes are -/// copied into pages of the kernel's own at the call, so nothing its sender -/// writes afterwards reaches a page this serves. What a program in `/apps` is -/// spawned and paged from, since no file server's volume is the kernel's. -pub struct ImageBacking { - pages: Vec, +/// An executable or library a caller read into a shared memory object of its +/// own and handed over by handle: what a program in `/apps` is spawned and +/// paged from, since no file server's volume is the kernel's. +/// +/// **Nothing is copied at the call, and every page is copied once when it is +/// read.** The object is the caller's memory, charged to it and alive while +/// any process pages from it; its bytes stay the caller's to change. So each +/// read takes one copy of the page into the reader's own buffer and what the +/// loader keeps is that copy: a change after the hand-off reaches only pages +/// not yet read, which is the caller changing its own child's program, and +/// never a value the kernel checked and then read again. +pub struct SharedImage { + object: Arc, size: u64, } -/// The most bytes one image may carry: the copy is kernel memory for the life -/// of every process paged from it, and nothing charges that to the sender. -pub const MAX_IMAGE_BYTES: u64 = 256 << 20; - -impl ImageBacking { - /// Copy `len` bytes of the caller's memory at `ptr`, a 2 MiB page's run at - /// a time, which is the most one user window maps contiguously. - pub fn copy_in( - ctx: &crate::user_ptr::SyscallContext, - ptr: crate::UserAddr, - len: u64, - ) -> Result { - use toyos_abi::syscall::SyscallError; - const PAGE: u64 = crate::mm::PAGE_2M; - if len == 0 || len > MAX_IMAGE_BYTES { +impl SharedImage { + /// The first `len` bytes of `object`. Refused unless they are ordinary + /// memory the kernel allocated — a device aperture is no program, and a + /// read of one is a device access — and the object holds them all. + pub fn over(object: Arc, len: u64) -> Result { + if len == 0 || len > object.size() || object.ram().is_none() { return Err(SyscallError::InvalidArgument); } - let mut pages = Vec::new(); - pages.try_reserve_exact(len.div_ceil(PAGE) as usize).map_err(|_| SyscallError::ResourceExhausted)?; - for _ in 0..len.div_ceil(PAGE) { - let page = crate::mm::pmm::alloc_page(crate::mm::pmm::Category::Elf) - .ok_or(SyscallError::ResourceExhausted)?; - pages.push(page); - } - let mut done = 0u64; - while done < len { - let at = ptr.raw().checked_add(done).ok_or(SyscallError::BadAddress)?; - // A run ends at the sender's page boundary or the image's own, whichever is first. - let run = (PAGE - at % PAGE).min(PAGE - done % PAGE).min(len - done); - let window = ctx.user_bytes(crate::UserAddr::new(at), run).ok_or(SyscallError::BadAddress)?; - let page = &pages[(done / PAGE) as usize]; - // SAFETY: the page is this backing's own and 2 MiB long; `done % PAGE + run <= PAGE` by the `min` above. - let dst = unsafe { - core::slice::from_raw_parts_mut( - page.direct_map().as_mut_ptr::().add((done % PAGE) as usize), - run as usize, - ) - }; - window.read_at(0, dst); - done += run; - } - Ok(Self { pages, size: len }) + Ok(Self { object, size: len }) } } -impl FileBacking for ImageBacking { +impl FileBacking for SharedImage { fn read_page(&self, file_offset: u64, buf: &mut [u8; BLOCK_SIZE]) -> BlockResult { buf.fill(0); if file_offset >= self.size { return Ok(()); } - const PAGE: u64 = crate::mm::PAGE_2M; - // Every backing is read a page at a time, and a 4 KiB-aligned page never crosses a 2 MiB one. - assert!(file_offset % BLOCK_SIZE_U64 == 0, "an image read at {file_offset:#x} is not page-aligned"); let valid = BLOCK_SIZE.min((self.size - file_offset) as usize); - let page = &self.pages[(file_offset / PAGE) as usize]; - // SAFETY: the page is this backing's, immutable since `copy_in`, and `file_offset % PAGE + valid <= PAGE`. - let src = unsafe { - core::slice::from_raw_parts( - page.direct_map().as_ptr::().add((file_offset % PAGE) as usize), - valid, - ) - }; - buf[..valid].copy_from_slice(src); + // SAFETY: `over` held `size` inside the object, which is one physically + // contiguous run the kernel allocated (`ram`) and keeps while `object` + // lives, so `file_offset + valid <= size` bytes from its direct-map + // address are mapped; `buf` is the reader's own and never the object. + // No reference is formed over the object's bytes, which its holders + // may be writing: this is the one fetch of each. + unsafe { + let from = self.object.phys().as_ptr::().add(file_offset as usize); + core::ptr::copy_nonoverlapping(from, buf.as_mut_ptr(), valid); + } Ok(()) } diff --git a/kernel/src/syscall/dispatch.rs b/kernel/src/syscall/dispatch.rs index 1fe5bdb3bb3..0567d9d8d8b 100644 --- a/kernel/src/syscall/dispatch.rs +++ b/kernel/src/syscall/dispatch.rs @@ -50,7 +50,7 @@ use super::proc::{ sys_endowments, sys_exit, sys_nanosleep, sys_process_open, sys_process_stats, sys_process_wait, sys_rt_enter, sys_spawn, sys_thread_exit, sys_thread_join, sys_thread_spawn, }; -use super::vm::{sys_dlopen, sys_dlsym, sys_mmap, sys_munmap, sys_query_modules, sys_tls_alloc_block}; +use super::vm::{shared_image, sys_dlopen, sys_dlsym, sys_mmap, sys_munmap, sys_query_modules, sys_tls_alloc_block}; /// A number a deleted syscall used is retired, never reused. macro_rules! retired_syscalls { @@ -185,6 +185,15 @@ pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> SYS_PIPE => sys_pipe(), SYS_SPAWN => { let Ok(args) = ctx.copy_in::(UserAddr::new(a1)) else { return bad_addr }; + // First, and nothing copied: the image is the caller's object, and + // an endowment below may move the caller's own handle to it. + let image = match args.image_len { + 0 => None, + len => match shared_image(args.image, len) { + Ok(image) => Some(alloc::sync::Arc::new(image) as alloc::sync::Arc), + Err(refused) => return refused, + }, + }; let text = match ctx.user_str(UserAddr::new(args.argv_ptr), args.argv_len) { Ok(s) => s, Err(e) => return e.to_u64() }; let cwd = match ctx.user_str(UserAddr::new(args.cwd_ptr), args.cwd_len).and_then(|p| spawn_cwd(&p)) { Ok(cwd) => cwd, @@ -230,15 +239,6 @@ pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> } else { alloc::vec::Vec::new() }; - // Last, because it is the one copy that can be large: every refusal - // above costs the caller nothing it must hand over again. - let image = match args.image_len { - 0 => None, - len => match crate::file_backing::ImageBacking::copy_in(&ctx, UserAddr::new(args.image_ptr), len) { - Ok(image) => Some(alloc::sync::Arc::new(image) as alloc::sync::Arc), - Err(e) => return e.to_u64(), - }, - }; let argv: alloc::vec::Vec<&str> = text.split('\0').filter(|s| !s.is_empty()).collect(); sys_spawn(&argv, pending, cwd, env, image) } @@ -342,15 +342,14 @@ pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> None => return bad_addr, }, }; + // Only named here: `sys_dlopen` answers a name this process holds + // before it looks at the object at all. let image = match a4 { 0 => None, raw => { let Some(at) = UserAddr::checked(raw) else { return bad_addr }; let Ok(image) = ctx.copy_in::(at) else { return bad_addr }; - match crate::file_backing::ImageBacking::copy_in(&ctx, UserAddr::new(image.ptr), image.len) { - Ok(image) => Some(image), - Err(e) => return e.to_u64(), - } + Some(image) } }; // ctx carries the copy-out: sys_dlopen writes init_out only once the load succeeds. diff --git a/kernel/src/syscall/vm.rs b/kernel/src/syscall/vm.rs index 298165343c0..ad320809137 100644 --- a/kernel/src/syscall/vm.rs +++ b/kernel/src/syscall/vm.rs @@ -193,6 +193,20 @@ pub(super) fn sys_munmap(addr: u64, _size: u64) -> u64 { 0 } +/// The first `len` bytes of the shared memory object `handle` names, as a +/// program or library image; `Err` is the syscall's answer. +pub(super) fn shared_image(handle: u64, len: u64) -> Result { + let Ok(raw) = u32::try_from(handle) else { return Err(SyscallError::InvalidArgument.to_u64()) }; + let object = process::with_process_data(|data| { + data.handles.get::( + toyos_abi::handle::RawHandle(raw), + toyos_abi::handle::Rights::MAP, + ) + }) + .map_err(|e| e.refuse())?; + crate::file_backing::SharedImage::over(object, len).map_err(|e| e.to_u64()) +} + /// `image` is the library's bytes when the caller read them itself, and `path` /// then only names it: an image is loaded afresh and kept out of the shared /// cache, which answers for what a path holds and an image is no path's. @@ -200,7 +214,7 @@ pub(super) fn sys_dlopen( ctx: &crate::user_ptr::SyscallContext, path: &str, init_out: Option, - image: Option, + image: Option, ) -> u64 { let cwd = process::with_process_data(|d| d.cwd.clone()); let resolved = vfs::lock().resolve_absolute(&cwd, path); @@ -220,13 +234,19 @@ pub(super) fn sys_dlopen( } let loaded = match image { - Some(image) => match crate::elf::load_shared_lib(&image) { - Ok((lib, _, _)) => Ok(lib), - Err(msg) => { - log!("dlopen: {}: {}", resolved, msg); - return SyscallError::Unknown.to_u64(); + Some(image) => { + let image = match shared_image(image.handle, image.len) { + Ok(image) => image, + Err(refused) => return refused, + }; + match crate::elf::load_shared_lib(&image) { + Ok((lib, _, _)) => Ok(lib), + Err(msg) => { + log!("dlopen: {}: {}", resolved, msg); + return SyscallError::Unknown.to_u64(); + } } - }, + } None => Err(()), }; let lib = match loaded { diff --git a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs index d6a289c5f05..6ef014a45c0 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs @@ -241,7 +241,7 @@ fn spawn_path(path: &str) -> Result { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, - image_ptr: 0, + image: 0, image_len: 0, }) } @@ -264,10 +264,21 @@ fn refused(name: &str, outcome: Result) { } } -/// The same bytes handed to the kernel whole, as a spawn from a file server's -/// volume hands them: the image route to the same loader. +/// `bytes` in a memory object of this process's own, as a spawn from a file +/// server's volume reads a program into one. +fn image_object(bytes: &[u8]) -> toyos::shm::SharedMemory { + let object = toyos::shm::SharedMemory::create(bytes.len().max(1)).expect("a memory object for the image"); + // SAFETY: the region is at least `bytes.len()` long, mapped here, and this + // process's alone; `bytes` is not in it. + unsafe { core::ptr::copy_nonoverlapping(bytes.as_ptr(), object.as_ptr(), bytes.len()) }; + object +} + +/// The same bytes handed to the kernel by handle, as a spawn from a file +/// server's volume hands them: the image route to the same loader. fn spawn_image(path: &str, bytes: &[u8]) -> Result { let argv = format!("{path}\0"); + let object = image_object(bytes); unsafe { syscall::spawn(&SpawnArgs { argv_ptr: argv.as_ptr() as u64, @@ -282,7 +293,7 @@ fn spawn_image(path: &str, bytes: &[u8]) -> Result { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, - image_ptr: bytes.as_ptr() as u64, + image: toyos::AsHandle::as_handle(&object).0 as u64, image_len: bytes.len() as u64, }) } @@ -318,9 +329,10 @@ fn dlopen_refused(name: &str, bytes: &[u8]) { } Err(e) => assert!(!format!("{e}").is_empty(), "{name}: dlopen error message"), } - // The same bytes handed over whole, under a name no path holds. + // The same bytes handed over by handle, under a name no path holds. let named = format!("/image/{name}"); - if let Ok(handle) = syscall::dl_open_image(named.as_bytes(), bytes) { + let object = image_object(bytes); + if let Ok(handle) = syscall::dl_open_image(named.as_bytes(), toyos::AsHandle::as_handle(&object), bytes.len() as u64) { panic!("{name}: dlopen of the image loaded it as handle {handle}, and the loader must refuse it"); } } diff --git a/tests/toyos-rust-tests/src/bin/abuse_elf_segments.rs b/tests/toyos-rust-tests/src/bin/abuse_elf_segments.rs index 717023563f8..35c44595173 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_elf_segments.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_elf_segments.rs @@ -134,9 +134,14 @@ fn spawn_err(name: &str, bytes: &[u8]) -> SyscallError { by_path } -/// A raw spawn of `path`, handing `image` over whole when it is not empty. +/// A raw spawn of `path`, handing `image` over in a memory object of this +/// process's own when it is not empty. fn spawn_as(path: &str, image: &[u8]) -> SyscallError { let argv = format!("{path}\0"); + let object = toyos::shm::SharedMemory::create(image.len().max(1)).expect("a memory object for the image"); + // SAFETY: the region is at least `image.len()` long, mapped here, and this + // process's alone; `image` is not in it. + unsafe { core::ptr::copy_nonoverlapping(image.as_ptr(), object.as_ptr(), image.len()) }; unsafe { syscall::spawn(&SpawnArgs { argv_ptr: argv.as_ptr() as u64, @@ -151,7 +156,7 @@ fn spawn_as(path: &str, image: &[u8]) -> SyscallError { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, - image_ptr: image.as_ptr() as u64, + image: toyos::AsHandle::as_handle(&object).0 as u64, image_len: image.len() as u64, }) } diff --git a/tests/toyos-rust-tests/src/bin/abuse_handle_table.rs b/tests/toyos-rust-tests/src/bin/abuse_handle_table.rs index adb032fbd08..d251bc39fff 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_handle_table.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_handle_table.rs @@ -71,7 +71,7 @@ fn main() { labels_len: LABELS.len() as u64, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, - image_ptr: 0, + image: 0, image_len: 0, }; unsafe { syscall::spawn(&args) } diff --git a/tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs b/tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs index dfec2f24623..d35a56bd2aa 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs @@ -128,7 +128,7 @@ fn main() { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, - image_ptr: 0, + image: 0, image_len: 0, }; let placed = (boundary - 8) as *mut SpawnArgs; diff --git a/tests/toyos-rust-tests/src/bin/abuse_spawn_argv.rs b/tests/toyos-rust-tests/src/bin/abuse_spawn_argv.rs index 4fcceaf1335..34b14f0c892 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_spawn_argv.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_spawn_argv.rs @@ -48,7 +48,7 @@ fn main() { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, - image_ptr: 0, + image: 0, image_len: 0, }; diff --git a/tests/toyos-rust-tests/src/bin/device_claim_lifetime.rs b/tests/toyos-rust-tests/src/bin/device_claim_lifetime.rs index c870d926049..b85157f5ca8 100644 --- a/tests/toyos-rust-tests/src/bin/device_claim_lifetime.rs +++ b/tests/toyos-rust-tests/src/bin/device_claim_lifetime.rs @@ -215,7 +215,7 @@ fn spawn_with_slot_map(handle: toyos_abi::RawHandle) -> Result Result { labels_len: 0, cwd_ptr: CWD.as_ptr() as u64, cwd_len: CWD.len() as u64, - image_ptr: 0, + image: 0, image_len: 0, }) } diff --git a/tests/toyos-rust-tests/src/bin/spawn_cwd.rs b/tests/toyos-rust-tests/src/bin/spawn_cwd.rs index 38e09279857..340910cd2f7 100644 --- a/tests/toyos-rust-tests/src/bin/spawn_cwd.rs +++ b/tests/toyos-rust-tests/src/bin/spawn_cwd.rs @@ -140,7 +140,7 @@ fn spawn_in(cwd: &str) -> Result { labels_len: 0, cwd_ptr: cwd.as_ptr() as u64, cwd_len: cwd.len() as u64, - image_ptr: 0, + image: 0, image_len: 0, }; // SAFETY: every pointer names a live local for the length beside it. diff --git a/tests/toyos-rust-tests/src/bin/spawn_image_object.rs b/tests/toyos-rust-tests/src/bin/spawn_image_object.rs new file mode 100644 index 00000000000..42c67159713 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/spawn_image_object.rs @@ -0,0 +1,102 @@ +//! A program handed to the kernel as a memory object is paged from that +//! object, and the object is the caller's to the end: what the kernel takes +//! from it is refused or safe, whatever the caller does with it. +//! +//! - A program read into an object of this process's own runs. +//! - A handle to the object without `MAP` is no right to its bytes: refused +//! `PermissionDenied`. A length the object does not hold: `InvalidArgument`. +//! - The object overwritten with `hlt` the moment the spawn returns: the child +//! runs what it had copied and faults on the rest, and the machine goes on — +//! the next spawn from a fresh object runs. +//! - `dlopen` of a name this process already holds answers that library and +//! never looks at the object: a handle it could not read is not refused. + +use toyos::shm::SharedMemory; +use toyos::AsHandle; +use toyos_abi::handle::{RawHandle, Rights}; +use toyos_abi::syscall::{self, SpawnArgs, SyscallError}; + +const SELF: &str = "/system/bin/test_rs_spawn_image_object"; +const LIB: &str = "/system/lib/libtls_lib.so"; +const CHILD: &str = "spawned-from-an-object"; +const CWD: &str = "/"; + +/// `path` read whole into a memory object of this process's own. +fn object_of(path: &str) -> (SharedMemory, u64) { + let bytes = std::fs::read(path).unwrap_or_else(|e| panic!("read {path}: {e}")); + let object = SharedMemory::create(bytes.len()).expect("a memory object"); + // SAFETY: the region is at least `bytes.len()` long, mapped here, and not + // yet handed to anyone; `bytes` is not in it. + unsafe { core::ptr::copy_nonoverlapping(bytes.as_ptr(), object.as_ptr(), bytes.len()) }; + (object, bytes.len() as u64) +} + +fn spawn(image: RawHandle, len: u64) -> Result { + let argv = format!("{SELF}\0{CHILD}\0"); + // SAFETY: every pointer names a live local for the whole call. + unsafe { + syscall::spawn(&SpawnArgs { + argv_ptr: argv.as_ptr() as u64, + argv_len: argv.len() as u64, + slot_map_ptr: 0, + slot_map_count: 0, + env_ptr: 0, + env_len: 0, + endow_ptr: 0, + endow_count: 0, + labels_ptr: 0, + labels_len: 0, + cwd_ptr: CWD.as_ptr() as u64, + cwd_len: CWD.len() as u64, + image: image.0 as u64, + image_len: len, + }) + } +} + +fn runs(what: &str) { + let (object, len) = object_of(SELF); + let child = spawn(object.as_handle(), len).unwrap_or_else(|e| panic!("{what}: spawn from an object: {e:?}")); + drop(object); + let code = syscall::process_wait(child).unwrap_or_else(|e| panic!("{what}: wait: {e:?}")); + assert_eq!(code, 0, "{what}: the child spawned from an object exited {code}"); + println!("spawn_image_object: {what}: a program in a memory object runs"); +} + +fn main() { + if std::env::args().nth(1).as_deref() == Some(CHILD) { + return; + } + + runs("first"); + + let (object, len) = object_of(SELF); + let unmappable = syscall::dup_narrowed(object.as_handle(), Rights::DUP.union(Rights::TRANSFER)) + .expect("a duplicate without MAP"); + assert_eq!(spawn(unmappable, len).err(), Some(SyscallError::PermissionDenied), "an object without MAP"); + syscall::close(unmappable); + // Past the whole object, which the kernel rounds up to 2 MiB pages. + let past = (len | ((2 << 20) - 1)) + 2; + assert_eq!(spawn(object.as_handle(), past).err(), Some(SyscallError::InvalidArgument), "a length past the object"); + println!("spawn_image_object: an object without MAP and a length past one are refused"); + + let child = spawn(object.as_handle(), len).expect("spawn from the object that is then overwritten"); + // SAFETY: the region is `object.len()` long and mapped here; the kernel + // reads it only by copying, which is what this races. + unsafe { core::ptr::write_bytes(object.as_ptr(), 0xF4, object.len()) }; + let code = syscall::process_wait(child).expect("the overwritten child is waited for"); + println!("spawn_image_object: a child whose object was overwritten after the spawn ended ({code})"); + drop(object); + runs("after the overwrite"); + + let (lib, lib_len) = object_of(LIB); + let name = b"/image/spawn_image_object/libtls_lib.so"; + let first = syscall::dl_open_image(name, lib.as_handle(), lib_len).expect("dlopen from an object"); + let unreadable = syscall::dup_narrowed(lib.as_handle(), Rights::DUP).expect("a duplicate without MAP"); + let again = syscall::dl_open_image(name, unreadable, lib_len); + assert_eq!(again, Ok(first), "a name this process holds was answered from the object, not the name"); + syscall::close(unreadable); + println!("spawn_image_object: a held name's dlopen answered its library without reading the object"); + + println!("spawn_image_object: PASS"); +} diff --git a/toyos-abi/src/syscall.rs b/toyos-abi/src/syscall.rs index b16c94e7a17..446109c7290 100644 --- a/toyos-abi/src/syscall.rs +++ b/toyos-abi/src/syscall.rs @@ -393,22 +393,29 @@ pub struct SpawnArgs { /// statement**: the kernel never substitutes the caller's own. pub cwd_ptr: u64, pub cwd_len: u64, - /// The program's bytes, read whole by the caller, or `image_len` 0 for a - /// program the kernel opens at `argv[0]` itself. An image is copied at the - /// call and the child is paged from the copy; `argv[0]` names it and is - /// opened by nobody, and its libraries are found in `/system/lib` alone. - pub image_ptr: u64, + /// The program's bytes, in a shared memory object the caller made and + /// read the program into: `image` is a handle to it carrying `MAP`, and + /// `image_len` how many of its first bytes the program is — or 0, for a + /// program the kernel opens at `argv[0]` itself. `PermissionDenied` for a + /// handle without `MAP`; `InvalidArgument` for a length the object does not + /// hold, or an object that is no memory the kernel allocated. The object + /// stays the caller's: what the kernel reads of it, and when, is + /// `kernel/src/file_backing.rs`'s (`SharedImage`). `argv[0]` names the + /// program and is opened by nobody, and its libraries are found in + /// `/system/lib` alone. + pub image: u64, pub image_len: u64, } const _: () = assert!(core::mem::size_of::() == 112); -/// A library's bytes, read whole by the caller, for `SYS_DLOPEN`'s fourth word: -/// what [`SpawnArgs::image_ptr`] is to a spawn. +/// A library's bytes in a shared memory object, for `SYS_DLOPEN`'s fourth +/// word: what [`SpawnArgs::image`] and [`SpawnArgs::image_len`] are to a spawn. #[repr(C)] #[derive(Clone, Copy)] pub struct ImageRef { - pub ptr: u64, + /// A handle to the object, carrying `MAP`. + pub handle: u64, pub len: u64, } @@ -1913,12 +1920,13 @@ pub fn dl_open(path: &[u8]) -> Result { dl_open_with(path, 0) } -/// Load a shared library whose bytes the caller read itself, under `name`: a -/// library on a file server's volume, which the kernel cannot open. The kernel -/// copies `image` at the call; a second load under the same name in this -/// process answers the first's handle, as a path load does. -pub fn dl_open_image(name: &[u8], image: &[u8]) -> Result { - let image = ImageRef { ptr: image.as_ptr() as u64, len: image.len() as u64 }; +/// Load a shared library whose bytes the caller read itself into the first +/// `len` bytes of the shared memory object `image`, under `name`: a library on +/// a file server's volume, which the kernel cannot open. The object is taken +/// as [`SpawnArgs::image`] says; a load under a name this process already +/// holds answers that library's handle, and the object is not asked about. +pub fn dl_open_image(name: &[u8], image: RawHandle, len: u64) -> Result { + let image = ImageRef { handle: image.0 as u64, len }; dl_open_with(name, &image as *const ImageRef as u64) } diff --git a/toyos-userbound/src/span.rs b/toyos-userbound/src/span.rs index 2a49bf7a669..083bacf96e9 100644 --- a/toyos-userbound/src/span.rs +++ b/toyos-userbound/src/span.rs @@ -132,7 +132,8 @@ mod tests { ("Stat", 24, 8), ("SchedInfo", 24, 8), ("FramebufferInfo", 32, 4), - ("SpawnArgs", 96, 8), + ("SpawnArgs", 112, 8), + ("ImageRef", 16, 8), ("NamespaceBuild", 56, 8), ("InboxSetup", 16, 8), ("ProcessStats", 128, 8), From ce6046ebe77b622fe8478be7cb8b5db18a6a6a4c Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 14:41:28 +0200 Subject: [PATCH 12/54] tests: the write-back guards stand on /tmp, and what no test arms goes Blocker 6 of the review. The kernel's write-back queue is still the teardown path of every closed kernel file, and `/tmp` is the one writable directory the kernel serves; it cannot move to a file server while C programs name no served file (`issues/filesystem/c-programs-name-no-file-a-file-server-holds.md`). So the tests guard it there: - `writeback_reopen` and `writeback_spawn` stage `/tmp` files with `iod` parked, and the harness holds `iod`'s one line saying `writeback-stall` parked it. - `resize-fault-refuse` and `resize-evict-window` are deleted with their hooks: the device read a shrink made has no device under it any more. - `writeback_durability`, which nothing ran, is deleted with its sleeps. - `esp_files` stages the loader attack as a link on `/home`, absolute and climbing, where it resolves to something: both are refused for writing. The IOMMU actuators' staging (`STAGED`, `staged`, the xHCI class) is compiled only with `boot-actuators`. The review's REMOVE lines are deleted: NVMe in the foreign-DMA actuator's doc, the panel census and the USB id base, and a test doc left above the wrong test. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- kernel/src/actuator.rs | 8 +- kernel/src/arch/x86_64/vtd/domain.rs | 1 + kernel/src/arch/x86_64/vtd/mod.rs | 8 ++ kernel/src/drivers/panic_console/mod.rs | 2 +- kernel/src/drivers/usb_storage.rs | 2 +- kernel/src/file_cache.rs | 44 +-------- kernel/src/iod.rs | 11 ++- tests/toyos-rust-tests/src/bin/esp_files.rs | 36 ++++--- .../src/bin/writeback_reopen.rs | 43 ++++----- .../src/bin/writeback_spawn.rs | 94 +++++++------------ tests/toyos.rs | 34 ++++--- toyos-manifest/src/lib.rs | 2 - 12 files changed, 114 insertions(+), 171 deletions(-) diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 86206e4c306..2c927c6a96c 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -113,12 +113,6 @@ actuators! { /// Serve a blocked-task dump from the shutdown once its first stage has stopped the machine: the report Ctrl+Alt+D gives on a shutdown stuck in its stop. quiesce_dump = "quiesce-dump"; - /// Take the page a shrink has just read off the device, in the window between that read and the lock that spends it — one other CPU's CLOCK sweep, which needs no VFS lock and so runs there. - resize_evict_window = "resize-evict-window"; - - /// Refuse the device read a shrink makes for the page its new end falls inside, for one staged length only; the one failure on that path QEMU will not produce. - resize_fault_refuse = "resize-fault-refuse"; - /// Establish three nested `scheduler::Operation`s and report what each observed and restored; it stages nothing, touching no device. sched_operation_nesting = "sched-operation-nesting"; @@ -440,7 +434,7 @@ actuators! { /// Give it a present context entry naming an empty second-level table, distinct from an absent context: passthrough would fault identically to the row above. iommu_empty_domain = "iommu-empty-domain"; - /// Answer a claimed function's first DMA grant with the physical bytes NVMe's admin completion queue page ends with — an address in another driver's pool, which the claimed function's own domain does not map. + /// Answer a claimed function's first DMA grant with an address in another driver's pool, which the claimed function's own domain does not map. iommu_userdev_foreign_dma = "iommu-userdev-foreign-dma"; /// Point a scanout backing at that same page, which the display's own domain does not map. diff --git a/kernel/src/arch/x86_64/vtd/domain.rs b/kernel/src/arch/x86_64/vtd/domain.rs index d068f4df4be..afe5caede28 100644 --- a/kernel/src/arch/x86_64/vtd/domain.rs +++ b/kernel/src/arch/x86_64/vtd/domain.rs @@ -164,6 +164,7 @@ pub fn unmap(id: DomainId, at: Iova, bytes: u64) -> Result<(), IommuError> { } pub fn attach(stream: StreamId, id: DomainId) { + #[cfg(feature = "boot-actuators")] if super::staged(stream) { log!("iommu: {stream} keeps the context an actuator staged, over its move to domain {}", id.raw()); return; diff --git a/kernel/src/arch/x86_64/vtd/mod.rs b/kernel/src/arch/x86_64/vtd/mod.rs index f1ce95c33c3..8762dcd9e2b 100644 --- a/kernel/src/arch/x86_64/vtd/mod.rs +++ b/kernel/src/arch/x86_64/vtd/mod.rs @@ -471,6 +471,7 @@ fn enable( // Unreachable from the host side, so these actuators substitute for // it; both are answered on the device's first *read*, since a // first-write access would cache write permission and never fault. + #[cfg(feature = "boot-actuators")] if crate::actuator::iommu_context_absent() && device.matches_class(XHCI_CLASS, XHCI_SUBCLASS, Some(XHCI_PROG_IF)) { @@ -480,6 +481,7 @@ fn enable( } // A present context entry naming an empty domain, distinct from a // missing entry: passthrough would fault identically either way. + #[cfg(feature = "boot-actuators")] if crate::actuator::iommu_empty_domain() && device.matches_class(XHCI_CLASS, XHCI_SUBCLASS, Some(XHCI_PROG_IF)) { @@ -565,15 +567,21 @@ fn enable( /// The requester id `iommu-context-absent` or `iommu-empty-domain` staged, so /// its driver's move to a domain of its own leaves the staging in place; /// `u32::MAX`, which no requester id is, when neither is armed. +#[cfg(feature = "boot-actuators")] static STAGED: core::sync::atomic::AtomicU32 = core::sync::atomic::AtomicU32::new(u32::MAX); /// Whether `stream` is the one an actuator left without a working context. +#[cfg(feature = "boot-actuators")] pub(super) fn staged(stream: StreamId) -> bool { STAGED.load(core::sync::atomic::Ordering::Relaxed) == u32::from(stream.requester()) } +/// The class the two IOMMU actuators stage on. +#[cfg(feature = "boot-actuators")] const XHCI_CLASS: u8 = 0x0C; +#[cfg(feature = "boot-actuators")] const XHCI_SUBCLASS: u8 = 0x03; +#[cfg(feature = "boot-actuators")] const XHCI_PROG_IF: u8 = 0x30; /// Slot of the per-width domain cache; exhaustive match so a new `AddressWidth` fails to compile here. diff --git a/kernel/src/drivers/panic_console/mod.rs b/kernel/src/drivers/panic_console/mod.rs index 71ed2178f64..1feaad11cba 100644 --- a/kernel/src/drivers/panic_console/mod.rs +++ b/kernel/src/drivers/panic_console/mod.rs @@ -1137,7 +1137,7 @@ impl core::fmt::Display for Census { } } -/// The panel's own row in the shutdown census, beside `irq:` and `nvme:`. +/// The panel's own row in the shutdown census, beside `irq:`. pub fn log_census() { log!("{Census}"); } diff --git a/kernel/src/drivers/usb_storage.rs b/kernel/src/drivers/usb_storage.rs index 017556dd63f..980ab593552 100644 --- a/kernel/src/drivers/usb_storage.rs +++ b/kernel/src/drivers/usb_storage.rs @@ -13,7 +13,7 @@ use crate::block::{self, BlockDevice, BlockError, BlockResult, DeviceId, Handle} use crate::log; use super::xhci; -/// Where USB disks start in the [`DeviceId`] space; must stay clear of NVMe's range, or `block::register` refuses the second driver's disk. +/// Where USB disks start in the [`DeviceId`] space. const USB_DEVICE_ID_BASE: DeviceId = 16; /// Disk numbers issued this boot; `0..count()` names every bound disk, and a number never moves or is reissued. diff --git a/kernel/src/file_cache.rs b/kernel/src/file_cache.rs index 7883a90dbbb..351f4023a83 100644 --- a/kernel/src/file_cache.rs +++ b/kernel/src/file_cache.rs @@ -460,17 +460,10 @@ pub fn resize( // under the same hold that spends it: a page admitted and released again is // clean and unreferenced, which is what a sweep takes at or after its hand — // and `sys_read`/`sys_write` sweep holding no VFS lock, so they run here. - let straddled = straddled_by_shrink(file_id, new_size); - #[cfg(feature = "boot-actuators")] - if straddled.is_some() && refuse_fault(new_size) { - return Err(block::BlockError::Device); - } - let fetched = match straddled { + let fetched = match straddled_by_shrink(file_id, new_size) { Some(page_idx) => fetch_page(file_id, page_idx)?.map(|page| (page_idx, page)), None => None, }; - #[cfg(feature = "boot-actuators")] - swept_fault_window(file_id, fetched.as_ref().map(|&(idx, _)| idx)); let mut cache = FILE_CACHE.lock(); if let Some((page_idx, page)) = fetched { @@ -490,41 +483,6 @@ pub fn resize( Ok(()) } -/// The `resize-fault-refuse` actuator: the device read a shrink makes for its -/// straddled page, refused. Keyed on the staged length, so no other shrink in -/// the boot can spend it. -#[cfg(feature = "boot-actuators")] -fn refuse_fault(new_size: u64) -> bool { - /// Mirrored in `tests/toyos-rust-tests/src/bin/writeback_durability.rs`. - const STAGED: u64 = 133; - let refuse = crate::actuator::resize_fault_refuse() && new_size == STAGED; - if refuse { - log!("file cache: resize-fault-refuse: refusing the straddled read of a shrink to {STAGED}"); - } - refuse -} - -/// The `resize-evict-window` actuator: one other CPU's CLOCK sweep, landing in -/// the window between the fault's device read and the lock that spends it. -#[cfg(feature = "boot-actuators")] -fn swept_fault_window(file_id: FileId, page_idx: Option) { - let Some(page_idx) = page_idx.filter(|_| crate::actuator::resize_evict_window()) else { - return; - }; - let mut cache = FILE_CACHE.lock(); - let mut taken = false; - if let Some(file) = cache.files.get_mut(&file_id) { - let is_cache = file.is_cache(); - if file.pages.get(&page_idx).is_some_and(|p| !p.is_dirty()) { - file.pages.remove(&page_idx); - taken = is_cache; - } - } - cache.cached_pages -= usize::from(taken); - drop(cache); - log!("file cache: resize-evict-window swept page {page_idx} of file {file_id}: took it = {taken}"); -} - /// The page a shrink to `new_size` would cut in half and does not hold. /// /// [`set_size_locked`] can only zero and dirty a *resident* page, and diff --git a/kernel/src/iod.rs b/kernel/src/iod.rs index 600d9e32cc5..01d4fc246a6 100644 --- a/kernel/src/iod.rs +++ b/kernel/src/iod.rs @@ -35,12 +35,15 @@ extern "C" fn body(_arg: u64) -> ! { WaitClass::Io, ) .expect("a kernel thread is a task and can arm"); - loop { - #[cfg(feature = "boot-actuators")] - if crate::actuator::writeback_stall() { + #[cfg(feature = "boot-actuators")] + if crate::actuator::writeback_stall() { + // Said once, so a test can hold that the queue it stages was held. + crate::log!("iod: writeback-stall: parked for the boot; the write-back queue is not drained"); + loop { let _ = watch::wait(&parkable, &armed, Deadline::never()); - continue; } + } + loop { // drain_all_iod, not drain_all: this thread already holds the WORK arm; drain's backoff parks on it instead of arming a second. crate::writeback::drain_all_iod(&parkable, &armed); // No deadline: only a push should end this wait; a periodic wake would need audio sign-off. diff --git a/tests/toyos-rust-tests/src/bin/esp_files.rs b/tests/toyos-rust-tests/src/bin/esp_files.rs index 97d843bb93f..a27549290c5 100644 --- a/tests/toyos-rust-tests/src/bin/esp_files.rs +++ b/tests/toyos-rust-tests/src/bin/esp_files.rs @@ -21,7 +21,6 @@ use std::fs; use std::io::{Read, Write}; use std::os::toyos::fs::symlink; -use toyos_abi::syscall::{self, OpenFlags}; /// Mirrored in `tests/common/volumes.rs`. Two halves of one fixture; a change /// to either without the other shows up as a mismatch here, not as a silent @@ -124,23 +123,30 @@ fn boot_refuses_every_way_of_changing_it() { } println!(" PASS create, delete, mkdir, rename and symlink are all refused on /boot"); - // The path checked must be the path opened, and plain WRITE is the hole: CREATE/TRUNCATE unlink the link. + // The path checked must be the path opened, and plain WRITE is the hole: + // CREATE/TRUNCATE unlink the link. A link on a writable served directory + // to the loader: an absolute one is handed back and resolved again in this + // process's own table, where `/boot`'s server refuses the write; one that + // climbs out is refused by the server it lies on. let before = loader_prefix(); assert_eq!(&before[..2], b"MZ", "the loader is not a PE image before the symlink attack"); - syscall::symlink(b"../boot/EFI/BOOT/BOOTx64.EFI", b"/tmp/evil").expect("a /tmp symlink is allowed"); - assert!( - syscall::open(b"/tmp/evil", OpenFlags::WRITE).is_err(), - "a /tmp symlink opened {LOADER} for writing", - ); - - let reader = fs::File::open(LOADER).expect("read /boot is allowed"); - assert!( - syscall::open(b"/tmp/evil", OpenFlags::WRITE).is_err(), - "a /tmp symlink opened {LOADER} for writing while a /boot read handle was held", - ); - drop(reader); + for (link, target) in + [("/home/esp_evil", "/boot/EFI/BOOT/BOOTx64.EFI"), ("/home/esp_evil_up", "../boot/EFI/BOOT/BOOTx64.EFI")] + { + let _ = fs::remove_file(link); + symlink(target, link).unwrap_or_else(|e| panic!("a link on /home is allowed: {link}: {e}")); + let write = || fs::OpenOptions::new().write(true).open(link); + match write() { + Err(e) => println!(" {link} -> {target} refused for writing: {e}"), + Ok(_) => panic!("{link} -> {target} opened {LOADER} for writing"), + } + let reader = fs::File::open(LOADER).expect("read /boot is allowed"); + assert!(write().is_err(), "{link} -> {target} opened {LOADER} for writing while a /boot read handle was held"); + drop(reader); + fs::remove_file(link).unwrap_or_else(|e| panic!("remove {link}: {e}")); + } assert_eq!(loader_prefix(), before, "a refused symlink write still changed the loader"); - println!(" PASS a /tmp symlink to {LOADER} is refused for writing"); + println!(" PASS a link on /home to {LOADER}, absolute or climbing, is refused for writing"); } /// The write direction, on the volume userland is allowed to have. diff --git a/tests/toyos-rust-tests/src/bin/writeback_reopen.rs b/tests/toyos-rust-tests/src/bin/writeback_reopen.rs index 32b3b659f1e..cc88bbb6574 100644 --- a/tests/toyos-rust-tests/src/bin/writeback_reopen.rs +++ b/tests/toyos-rust-tests/src/bin/writeback_reopen.rs @@ -1,23 +1,21 @@ -//! A re-open racing a pending write-back reads the buffered pages, not the -//! device. +//! A re-open racing a pending write-back reads what was written. //! -//! When the last handle of a modified file drops, its dirty pages no longer -//! flush on the closing thread — they are pinned in the cache and `iod` flushes -//! them later (`kernel::writeback`). This is the invariant the whole write-back -//! queue rests on: a file's dirty pages outlive the handle that dirtied them, -//! and a re-open before the drain sees them rather than the device. +//! When the last handle of a modified file drops, the kernel does not tear it +//! down on the closing thread: the file is pinned in the cache and `iod` tears +//! it down later (`kernel::writeback`). This is the invariant the write-back +//! queue rests on: a file's pages outlive the handle that dirtied them, and a +//! re-open before the drain is handed the pinned file. //! -//! **`writeback-stall` parks `iod` before any teardown**, so the write-back is -//! provably still pending when this re-opens the file: the re-open precedes the -//! flush and must still read what was written. This is on `/home`, which is -//! NVMe-backed, so if the pin were broken — the last close discarding the pages, -//! as it did before this change — the re-open would read the device, which has -//! not been written, and get zeros or a short file. +//! **`writeback-stall` parks `iod` before any teardown**, so the teardown is +//! provably still owed when this re-opens the file. `/tmp` is the one writable +//! directory the kernel serves, so it is where the queue has files: if the +//! last close released the file instead of pinning it, the re-open would find +//! its name and none of its pages, and read an empty file. use std::fs; use std::io::{Read, Write}; -const PATH: &str = "/home/wb_reopen.bin"; +const PATH: &str = "/tmp/wb_reopen.bin"; /// Three pages and a bit, so the file is several dirty pages rather than one. const LEN: usize = 3 * 4096 + 137; @@ -28,33 +26,30 @@ fn distinctive() -> Vec { fn main() { let want = distinctive(); - // Write and drop the handle. No `sync_all`: the point is the un-flushed, - // buffered write. The bytes now live only in the file cache, pinned by the - // write-back queue because the last handle just went. + // Write and drop the handle: the last close pins the file and queues its + // teardown, which `iod` never runs on this boot. { let mut f = fs::File::create(PATH).unwrap_or_else(|e| panic!("create {PATH}: {e}")); f.write_all(&want).expect("write the distinctive bytes"); } - // `iod` is parked before it can flush (writeback-stall), so this re-open - // provably precedes the flush. It must read the pinned pages, not the - // device. let mut got = Vec::new(); { let mut f = fs::File::open(PATH).expect("re-open before the write-back drains"); - f.read_to_end(&mut got).expect("read back the buffered file"); + f.read_to_end(&mut got).expect("read back the pinned file"); } assert_eq!( got.len(), want.len(), - "re-open read {} bytes, wrote {} — a pending write-back was not read from the cache", + "re-open read {} bytes, wrote {} — the last close released a file its teardown still owed", got.len(), want.len() ); if let Some(at) = got.iter().zip(&want).position(|(a, b)| a != b) { - panic!("re-open differs from what was written at byte {at}: the buffered pages were not read"); + panic!("re-open differs from what was written at byte {at}: the pinned pages were not read"); } + let _ = fs::remove_file(PATH); - println!("re-open before the write-back drains read all {LEN} buffered bytes"); + println!("re-open before the write-back drains read all {LEN} pinned bytes"); } diff --git a/tests/toyos-rust-tests/src/bin/writeback_spawn.rs b/tests/toyos-rust-tests/src/bin/writeback_spawn.rs index 68925630235..9aa018df986 100644 --- a/tests/toyos-rust-tests/src/bin/writeback_spawn.rs +++ b/tests/toyos-rust-tests/src/bin/writeback_spawn.rs @@ -1,49 +1,38 @@ -//! A program written, closed and spawned runs, with the write-back still owed. +//! A program written, closed and spawned runs, with its write-back still owed. //! //! Write, close, exec is what a compiler does with the binary it just produced, -//! and it is the project's self-hosting north star. Since the write-back queue -//! landed the last close no longer flushes: the dirty pages are pinned in the -//! file cache and `iod` writes them out later (`kernel::writeback`). But -//! `loader::spawn` does not read the file through the cache at all — it takes a -//! *device* view (`Vfs::open_backing`), which on `/home` is the extent list and -//! the length bcachefs has recorded. Before the drain those still say what -//! `create` wrote: no extents, length 0. The kernel answered +//! and it is the project's self-hosting north star. The last close does not +//! tear the file down: it is pinned in the file cache and `iod` tears it down +//! later (`kernel::writeback`). `loader::spawn` does not read the file through +//! a handle — it takes a view of the file (`Vfs::open_backing`), which drains +//! the queue inline first, so the spawn runs the teardown `iod` owes. //! -//! ```text -//! spawn: /home/disk_backtrace/child: ELF: fewer bytes than a file header -//! ``` -//! -//! **`writeback-stall` parks `iod` before any teardown**, so the flush is -//! provably still owed when the spawn below happens — the whole race, held open -//! for as long as the test needs it, instead of the few milliseconds a loaded -//! host happened to give CI. `disk_backtrace` is the same sequence unstaged, and -//! reproduced this once in eleven local runs and four times out of four on a -//! starved hosted shard. -//! -//! `/home` and not `/tmp`: tmpfs pages *are* the file, so nothing there can be -//! behind. The claim is about a device. +//! **`writeback-stall` parks `iod` before any teardown**, so the teardown is +//! provably the spawn's own. `/tmp` is the one writable directory the kernel +//! serves: a teardown that released a live file's pages there would leave the +//! spawn's view reading zeros where the binary was, and the load refused. //! //! The payload is this binary itself, re-run with an argument, so the thing -//! spawned off the disk is a megabyte-scale PIE whose text, relocations and -//! symbols all demand-page through the backing — a short file would prove the -//! header and nothing after it. +//! spawned is a megabyte-scale PIE whose text, relocations and symbols all +//! demand-page through the view — a short file would prove the header and +//! nothing after it. use std::fs; use std::io::Write; use std::process::Command; -const DIR: &str = "/home/writeback_spawn"; +const DIR: &str = "/tmp/writeback_spawn"; const IN_ROOT: &str = "/system/bin/test_rs_writeback_spawn"; -const ON_DISK: &str = "/home/writeback_spawn/child"; -const STILL_OPEN: &str = "/home/writeback_spawn/held"; +const WRITTEN: &str = "/tmp/writeback_spawn/child"; +const STILL_OPEN: &str = "/tmp/writeback_spawn/held"; /// What tells this binary it is the copy being run rather than the test. -const CHILD: &str = "spawned-from-disk"; +const CHILD: &str = "spawned-from-tmp"; fn main() { if std::env::args().nth(1).as_deref() == Some(CHILD) { // The marker is the proof the child reached its own code, rather than // an exit status a refused spawn could also produce. - println!(" child: running from {ON_DISK}"); + println!(" child: running from {WRITTEN}"); return; } @@ -51,52 +40,39 @@ fn main() { let image = fs::read(IN_ROOT).unwrap_or_else(|e| panic!("read {IN_ROOT}: {e}")); // `fs::write` opens, writes and drops the handle. With `iod` parked that - // drop pins the pages and enqueues the file; nothing has reached the device. - fs::write(ON_DISK, &image).unwrap_or_else(|e| panic!("write {ON_DISK}: {e}")); - println!(" copied {} bytes to {ON_DISK} with the write-back still owed", image.len()); + // drop pins the file and queues its teardown. + fs::write(WRITTEN, &image).unwrap_or_else(|e| panic!("write {WRITTEN}: {e}")); + println!(" copied {} bytes to {WRITTEN} with its teardown still owed", image.len()); - let status = Command::new(ON_DISK) + let status = Command::new(WRITTEN) .arg(CHILD) .status() - .unwrap_or_else(|e| panic!("spawn {ON_DISK}: {e}")); + .unwrap_or_else(|e| panic!("spawn {WRITTEN}: {e}")); assert!( status.success(), - "the child spawned off a file whose write-back is still pending exited {:?}", + "the child spawned off a file whose teardown was still owed exited {:?}", status.code() ); - // The differential: the same bytes read back a second way, off the device. - // The spawn drained the queue, so the file left the cache and this re-open - // resolves fresh extents and reads NVMe blocks — nothing of what was written - // is still buffered. Compared against ROOT's copy, which is a different - // mount and a different filesystem, so a length that matched by construction - // could not hide a wrong byte. - let back = fs::read(ON_DISK).unwrap_or_else(|e| panic!("read back {ON_DISK}: {e}")); - assert_eq!( - back.len(), - image.len(), - "read back {} bytes off the device, wrote {}", - back.len(), - image.len() - ); + // The same bytes read back a second way, through a handle, after the + // spawn ran the teardown: compared against ROOT's copy, a different mount, + // so a length that matched by construction could not hide a wrong byte. + let back = fs::read(WRITTEN).unwrap_or_else(|e| panic!("read back {WRITTEN}: {e}")); + assert_eq!(back.len(), image.len(), "read back {} bytes, wrote {}", back.len(), image.len()); if let Some(at) = back.iter().zip(&image).position(|(a, b)| a != b) { - panic!("what the device holds differs from what was written at byte {at}"); + panic!("what the file holds after its teardown differs from what was written at byte {at}"); } - println!(" PASS: spawned {ON_DISK} before its write-back drained, {} bytes verified off the device", back.len()); + println!(" PASS: spawned {WRITTEN} before its teardown, {} bytes verified after it", back.len()); still_open_and_dirty(&image); + let _ = fs::remove_dir_all(DIR); println!("writeback spawn test passed"); } -/// The same race with the handle still held, which no queue knows about. -/// -/// A closed file is on the write-back queue and `Vfs::open_backing` drains it; -/// a file still open is on no queue, so its bytes are cache pages and the -/// mount's record is what `create` wrote. Handle reads answered from the cache -/// and a spawn's device view did not, so which of two readers of one file a -/// caller got was decided by which syscall it used. +/// The same with the handle still held, which no queue knows about: the view +/// a spawn takes flushes a file still open and dirty before it reads it. fn still_open_and_dirty(image: &[u8]) { let mut held = fs::File::create(STILL_OPEN).unwrap_or_else(|e| panic!("create {STILL_OPEN}: {e}")); held.write_all(image).unwrap_or_else(|e| panic!("write {STILL_OPEN}: {e}")); @@ -115,7 +91,7 @@ fn still_open_and_dirty(image: &[u8]) { let back = fs::read(STILL_OPEN).unwrap_or_else(|e| panic!("read back {STILL_OPEN}: {e}")); assert_eq!(back.len(), image.len(), "read back {} bytes, wrote {}", back.len(), image.len()); if let Some(at) = back.iter().zip(image).position(|(a, b)| a != b) { - panic!("what the device holds differs from what was written at byte {at}"); + panic!("what the file holds differs from what was written at byte {at}"); } println!(" PASS: spawned {STILL_OPEN} while its writer still held it, {} bytes verified", back.len()); } diff --git a/tests/toyos.rs b/tests/toyos.rs index 8012d598c44..92fa750c1ca 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -432,13 +432,11 @@ const RUST_SKIP: &[&str] = &[ // measure. // // `writeback_reopen` and `writeback_spawn` each need their own boot with - // `writeback-stall` armed; `writeback_durability` writes `/log` and - // `kernel_log_file` judges it host-side off the image after a shutdown. - // All three run in `MACHINE_TESTS`, not on the shared boot. + // `writeback-stall` armed, and run in `MACHINE_TESTS`, not on the shared + // boot. "writeback_reopen", "writeback_spawn", - "writeback_durability", - // Same shape as `writeback_durability`: what it stages on `/log` — a file + // What it stages on `/log` — a file // unlinked out from under a held descriptor, its clusters handed to the next // writer — is only half the claim, and the other half is the volume read // back off the image after a shutdown by a FAT implementation that is not @@ -1496,8 +1494,8 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // `issues/kernel/every-wait-in-this-kernel-is-a-spin.md`). `writeback_reopen` // and `writeback_spawn` arm `writeback-stall`, so each needs its own actuator // boot: one holds the queue open across a *handle* re-open, which the file - // cache answers, and the other across a *spawn*, which is a device view and - // does not. + // cache answers, and the other across a *spawn*, whose view of the file + // drains the queue itself. // The watch's lost-wake window, staged: `watch-window` holds every pipe // waiter between reading its condition and parking, so the peer's post lands where // only the notified bit carries it to the commit. @@ -1715,7 +1713,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("fs_rename_durable", &["test_rs_fs_rename_durable", "test_rs_fs_dirs_durable"]), ("fsync_failed_commit", &["test_rs_fsync_flush_failed"]), ("redirty_mid_flush", &["test_rs_redirty_mid_flush"]), - ("kernel_log_file", &["test_rs_writeback_durability"]), ("double_fault_stack", &["test_rs_test_panic_child"]), ("idle_stack_guard", &["test_rs_test_panic_child"]), ("dump_left_pending_is_owed", &["test_rs_dump_stage_load"]), @@ -11357,6 +11354,10 @@ fn metal_sim_client_death(boot: &mut Boot) -> Result<(), String> { Ok(()) } +/// What `iod` says when `writeback-stall` parks it: a write-back test's proof +/// that the queue it stages was held (`kernel/src/iod.rs`). +const WRITEBACK_STALLED: &str = "iod: writeback-stall: parked for the boot"; + /// Run one machine-shape test. Like `run_screen_test`, each of these owns its /// QEMU — the machine shape *is* the test — except for the runs of adjacent /// names that share one through `held` (see [`group_boot`]). @@ -11550,7 +11551,7 @@ fn run_machine_test( } // The write-back queue's re-open control: `writeback-stall` parks `iod` // before it drains, so the guest can prove a re-open before the flush - // reads the pinned pages and not the NVMe `/home` device. + // reads the pinned pages. "writeback_reopen" => { let options = BootOptions { kernel_params: &["writeback-stall"], @@ -11559,7 +11560,9 @@ fn run_machine_test( let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); let boot = qemu.boot_log().to_string(); - serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; + let console = serial::Serial::named("boot console", boot.as_str()); + console.must_be_clean()?; + console.must_say(WRITEBACK_STALLED)?; let result = qemu.run_test("test_rs_writeback_reopen", Duration::from_secs(30)); if !check_rust_result(&result) { return Err(format!( @@ -11570,10 +11573,9 @@ fn run_machine_test( Ok(()) } // The other half of the same stall, on the path the file cache does not - // answer: a spawn reads a *device* view (`Vfs::open_backing`), so a - // binary written and closed with the write-back still owed used to load - // as `ELF: fewer bytes than a file header`. Same actuator, and the same - // reason it needs its own boot. + // answer: a spawn takes a view of the file (`Vfs::open_backing`), which + // runs the teardown `iod` owes before it reads. Same actuator, and the + // same reason it needs its own boot. "writeback_spawn" => { let options = BootOptions { kernel_params: &["writeback-stall"], @@ -11582,7 +11584,9 @@ fn run_machine_test( let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); let boot = qemu.boot_log().to_string(); - serial::Serial::named("boot console", boot.as_str()).must_be_clean()?; + let console = serial::Serial::named("boot console", boot.as_str()); + console.must_be_clean()?; + console.must_say(WRITEBACK_STALLED)?; let result = qemu.run_test("test_rs_writeback_spawn", Duration::from_secs(30)); if !check_rust_result(&result) { return Err(format!( diff --git a/toyos-manifest/src/lib.rs b/toyos-manifest/src/lib.rs index a3e9480f9a3..646965bdb17 100644 --- a/toyos-manifest/src/lib.rs +++ b/toyos-manifest/src/lib.rs @@ -587,8 +587,6 @@ mod tests { assert!(syscap_rights(&["sysinfo".into()]).is_err()); } - /// A class name reaches init through this file, so a `devices` entry the - /// ABI does not know is a config that renders and cannot boot. /// A file server's roles reach init as records, and a name that is no /// role is refused where it is written: init would start a server for /// directories nobody named. From ad12805d1cae7ec97e22ade2fd10a7541cbf5572 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 14:41:36 +0200 Subject: [PATCH 13/54] issues: what the review's notes leave true, filed - a held file is known across a restart by what its entry records, which a same-length replacement in the freed first block within the second passes; - a file server maps a lent window at the size it expects; - every flush of a served file syncs its whole volume; - one write far past a file's end holds DATA's server; - the log's server ending under logd's append is unmeasured; - a served directory costs each process a 2 MiB window, unmeasured; - the boot volume's server is endowed a claim that writes (extends the isolation issue on blockd sessions). `lan_mdns_answer`'s issue is `expected-red`: `src/redlist.rs` quarantines it. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- ...t-path-outgrows-sun-len-on-the-dev-host.md | 2 +- ...ps-a-lent-window-at-the-size-it-expects.md | 23 ++++++++++++++ ...oss-a-restart-by-what-its-entry-records.md | 31 +++++++++++++++++++ ...osts-each-process-a-two-mebibyte-window.md | 19 ++++++++++++ ...of-a-served-file-syncs-its-whole-volume.md | 20 ++++++++++++ ...ar-past-a-files-end-holds-data-s-server.md | 21 +++++++++++++ ...ending-under-logds-append-is-unmeasured.md | 19 ++++++++++++ ...-can-open-every-partition-blockd-serves.md | 11 +++++-- 8 files changed, 143 insertions(+), 3 deletions(-) create mode 100644 issues/filesystem/a-file-server-maps-a-lent-window-at-the-size-it-expects.md create mode 100644 issues/filesystem/a-held-file-is-known-across-a-restart-by-what-its-entry-records.md create mode 100644 issues/filesystem/a-served-directory-costs-each-process-a-two-mebibyte-window.md create mode 100644 issues/filesystem/every-flush-of-a-served-file-syncs-its-whole-volume.md create mode 100644 issues/filesystem/one-write-far-past-a-files-end-holds-data-s-server.md create mode 100644 issues/filesystem/the-logs-server-ending-under-logds-append-is-unmeasured.md diff --git a/issues/build/a-lane-s-tap-socket-path-outgrows-sun-len-on-the-dev-host.md b/issues/build/a-lane-s-tap-socket-path-outgrows-sun-len-on-the-dev-host.md index 2958bb39462..b64a9bb6e27 100644 --- a/issues/build/a-lane-s-tap-socket-path-outgrows-sun-len-on-the-dev-host.md +++ b/issues/build/a-lane-s-tap-socket-path-outgrows-sun-len-on-the-dev-host.md @@ -1,5 +1,5 @@ --- -status: open +status: expected-red kind: defect opened: 2026-09-26 --- diff --git a/issues/filesystem/a-file-server-maps-a-lent-window-at-the-size-it-expects.md b/issues/filesystem/a-file-server-maps-a-lent-window-at-the-size-it-expects.md new file mode 100644 index 00000000000..b04c0e22f84 --- /dev/null +++ b/issues/filesystem/a-file-server-maps-a-lent-window-at-the-size-it-expects.md @@ -0,0 +1,23 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A file server maps a lent window at the size it expects, not the size it is + +fsd adopts the region a client lends at its hello with +`SharedMemory::adopt(lent, WINDOW_BYTES)` (`userland/fsd/src/main.rs`), and +`adopt` checks nothing about the region: its doc says a peer that promises a +size and sends a smaller region is the reader's to bound. A region +`SYS_SHM_CREATE` made is whole 2 MiB pages, so it is never short of the window. +A region a device claim answered is not made that way — a BAR aperture or a +scanout is whatever the device is — and a program holding one can lend it. The +server then copies a read's bytes into device memory, or past the end of a +short mapping into whatever follows, and a fault there ends DATA's server, +whose restarts are budgeted for the whole machine. + +**Exit**: the kernel answers a region's size and whether it is memory it +allocated to the holder of a handle to it, fsd refuses a hello whose window is +not `WINDOW_BYTES` of that memory by name, and a guest test lends a region a +device claim answered and is refused while the server answers the next client. diff --git a/issues/filesystem/a-held-file-is-known-across-a-restart-by-what-its-entry-records.md b/issues/filesystem/a-held-file-is-known-across-a-restart-by-what-its-entry-records.md new file mode 100644 index 00000000000..65550738888 --- /dev/null +++ b/issues/filesystem/a-held-file-is-known-across-a-restart-by-what-its-entry-records.md @@ -0,0 +1,31 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A held file is known across a file server's restart by what its entry records + +A handle held across a restart of its file server reopens the file by path +and keeps it only when the server answers the identity the handle last saw +(`toyos::fs::Stat::ident`). The volumes have no object id to answer with: the +interim DATA format's entry is a name, a length, an mtime and extents +(`bcachefs/src/fs.rs`), and a FAT entry a short name, a creation stamp, a +first cluster and a length. So the identity is a hash of what the entry +records — on DATA the first block, the length and the mtime, which the format +keeps to the second; on FAT the short name, the creation stamp, the first +cluster and the length (`userland/fsd/src/volume.rs`, `Volume::ident`). + +That tells a file renamed over the held one, a file made again at its path, +and a write the restart lost from the file the handle held. It does not tell +a replacement that takes the held file's freed first block — the allocator +hands back the most recently freed first — and has the same length, written in +the same second: the handle then writes into the replacement. A file with no +block yet is refused outright (`ident` 0), since nothing tells it from +another. + +**Exit**: DATA's entry carries an object id and a generation that nothing +reuses, and `Volume::ident` answers them — the format the track moves DATA to +has inode numbers — with a host test that deletes a held file, makes a +same-length file in its freed block within the second, and has the reopen +refused. diff --git a/issues/filesystem/a-served-directory-costs-each-process-a-two-mebibyte-window.md b/issues/filesystem/a-served-directory-costs-each-process-a-two-mebibyte-window.md new file mode 100644 index 00000000000..7e320876046 --- /dev/null +++ b/issues/filesystem/a-served-directory-costs-each-process-a-two-mebibyte-window.md @@ -0,0 +1,19 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# A served directory costs each process a 2 MiB window, and nobody measured it + +Every process that names a file under a served directory connects to that +directory's server once and lends it a `toyos::fs::WINDOW_BYTES` region +(`toyos/src/fs.rs`): 2 MiB of the process's memory per directory it has +touched, held for the process's life, and mapped into the server too. A +desktop process that touches `/home`, `/config`, `/state` and `/apps` holds +8 MiB of windows; fsd maps up to `MAX_SERVED` of them. Whether the window's +size is paid for by the reads that fill it, and what the desktop's processes +hold in windows together, has not been measured. + +**Exit**: the resident memory of a desktop boot's processes with and without +their windows, measured, and the window's size justified from it or cut. diff --git a/issues/filesystem/every-flush-of-a-served-file-syncs-its-whole-volume.md b/issues/filesystem/every-flush-of-a-served-file-syncs-its-whole-volume.md new file mode 100644 index 00000000000..0ba1081bd40 --- /dev/null +++ b/issues/filesystem/every-flush-of-a-served-file-syncs-its-whole-volume.md @@ -0,0 +1,20 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# Every flush of a served file syncs its whole volume + +The std fork's `File::flush` is `File::fsync` +(`rust/library/std/src/sys/fs/toyos.rs`), where every other platform's is a +no-op, and fsd answers `FSYNC` by syncing the volume, every open file's entry +and every dirty block of the cache (`userland/fsd/src/main.rs`, `FSYNC`). So a +`BufWriter` over a file on `/home` syncs all of DATA each time it flushes, and a +4 KiB create and fsync cost 3.5–3.6 ms p50 on QEMU TCG against the kernel's +1.0–1.3 ms, measured on the branch that moved DATA to fsd. + +**Exit**: `File::flush` hands the server what is buffered and makes nothing +durable, as std's contract for `Write::flush` is, and `FSYNC` makes durable the +file it names and what that file's entry depends on, with a guest measurement +of a `BufWriter` flush beside an fsync. diff --git a/issues/filesystem/one-write-far-past-a-files-end-holds-data-s-server.md b/issues/filesystem/one-write-far-past-a-files-end-holds-data-s-server.md new file mode 100644 index 00000000000..eed0f5be5d9 --- /dev/null +++ b/issues/filesystem/one-write-far-past-a-files-end-holds-data-s-server.md @@ -0,0 +1,21 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# One write far past a file's end holds DATA's server for as long as the gap + +fsd takes a `WRITE` at any offset up to `toyos::fs::MAX_FILE_BYTES` (16 TiB), +and `Mounted::resolve_or_alloc_block` bridges the gap from the file's end to it +by allocating and zeroing every block between +(`issues/filesystem/a-bcachefs-hole-costs-a-block-write-per-page.md` is why). +fsd is one thread, so the one request runs until the volume is full or the gap +is bridged, and every other client of `/apps`, `/config`, `/home` and `/state` +waits the whole of it: any program holding `/home` can stop every program's +files with one `seek` and one byte. + +**Exit**: a write whose gap is past what a request may cost is refused by name +before anything is allocated — or the format expresses a hole — with a guest +test that writes one byte a gigabyte past a file's end and has another client +answered meanwhile. diff --git a/issues/filesystem/the-logs-server-ending-under-logds-append-is-unmeasured.md b/issues/filesystem/the-logs-server-ending-under-logds-append-is-unmeasured.md new file mode 100644 index 00000000000..cd5ed0e177a --- /dev/null +++ b/issues/filesystem/the-logs-server-ending-under-logds-append-is-unmeasured.md @@ -0,0 +1,19 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# The log's server ending under logd's append is unmeasured + +`fsd_restart` ends only DATA's server. The log's is restarted by the same row, +and logd holds its file open across it: a handle reopened after an end is +checked against the identity it last saw, and an append the server ended +under is answered `StaleNetworkFileHandle` and not retried. What logd does +then — whether the lines of that append are lost, repeated, or written after a +reopen — and whether the log file the host reads back after the stop is whole, +has not been run. + +**Exit**: a boot that ends the log's server under logd's append (`--end-on` on +a path of the log's volume), and the log file read back off the image by a FAT +implementation that is not fsd's, with what logd said about the lost append. diff --git a/issues/isolation/a-file-server-can-open-every-partition-blockd-serves.md b/issues/isolation/a-file-server-can-open-every-partition-blockd-serves.md index d20cf62bab3..16be5d4c25e 100644 --- a/issues/isolation/a-file-server-can-open-every-partition-blockd-serves.md +++ b/issues/isolation/a-file-server-can-open-every-partition-blockd-serves.md @@ -14,7 +14,14 @@ write. Nothing in the protocol ties a session to the role that asked for it; what keeps each server on its own partition is the argument init passes it, which is not authority. +The partition claim init mints for a role on a disk the kernel drives +(`storage_endowment` in `userland/init/src/main.rs`) is the same: the boot +volume's server is endowed a claim that writes, on the ESP the firmware loads +the loader from. Before the file servers the kernel refused a `/boot` write +itself; now only that server's promise to mount its volume read-only does. + **Exit**: init hands each file server a capability to its own partition's session and nothing else — a port blockd serves per partition, or a session -init opens and moves — with a negative control: the log's server asking for -DATA is refused. +init opens and moves — and the boot volume's server a claim that cannot +write, with negative controls: the log's server asking for DATA is refused, +and a write through the boot server's claim is refused by the kernel. From 28094fd3d99b46b3ed5df1377fabcbb93f94be25 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 15:18:41 +0200 Subject: [PATCH 14/54] fsd: the restart test ends DATA's server under init's request, where the deadlock was `--end-at-hello` ended the server at init's first hello on `/apps`, whose connect then failed and was answered: it never reached the retry that reconnects and waits in the port's queue, so init resolving from its loop passed it too. `--end-at-request` ends the server at the first request after a hello, on a connection init holds; its retry reconnects into the queue. With init's resolution moved back onto its loop (a checked patch), `fsd_restart` wedges the machine: fsd's end is the last line, and init starts nothing again (EXIT=1, 322 s, twice). With the worker: EXIT=0. The mark that makes the actuator once a boot is looked for and then made: the std fork's `create_new` on a kernel path is not exclusive, which two servers took as two first times; filed. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_014iqcj4jDKpaiDX8B7CMvmK --- ...e-new-on-a-kernel-path-is-not-exclusive.md | 20 +++++++ tests/common/storage.rs | 10 ++-- tests/fsdrestartcase/system.toml | 8 +-- tests/toyos-rust-tests/src/bin/fs_restart.rs | 15 ++--- userland/fsd/src/main.rs | 60 ++++++++++--------- 5 files changed, 70 insertions(+), 43 deletions(-) create mode 100644 issues/filesystem/create-new-on-a-kernel-path-is-not-exclusive.md diff --git a/issues/filesystem/create-new-on-a-kernel-path-is-not-exclusive.md b/issues/filesystem/create-new-on-a-kernel-path-is-not-exclusive.md new file mode 100644 index 00000000000..ca2d3a76b13 --- /dev/null +++ b/issues/filesystem/create-new-on-a-kernel-path-is-not-exclusive.md @@ -0,0 +1,20 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# `create_new` on a kernel path is not exclusive + +std's `OpenOptions::create_new(true)` promises an open that fails with +`AlreadyExists` when the file is there. The std fork's `to_flags` +(`rust/library/std/src/sys/fs/toyos.rs`) turns `create_new` into the kernel's +plain `OpenFlags::CREATE`, and the kernel has no exclusive flag to turn it +into, so on `/tmp` a second `create_new` of one path opens the file the first +made and says nothing. A served path does not share it: the file protocol +carries `O_CREATE_NEW` and fsd refuses. Found when fsd's test actuator took two +`create_new`s of one `/tmp` mark from two processes as two first times. + +**Exit**: an exclusive create reaches the kernel's `/tmp` and is refused +`AlreadyExists` there, with a guest test that makes one path twice with +`create_new` and has the second refused. diff --git a/tests/common/storage.rs b/tests/common/storage.rs index 28ae4ff1336..e8bd8883e4a 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -610,8 +610,8 @@ pub fn home_overwrite_reads_back( /// budget, its directories answer `Gone`. Judged off the device. /// /// `tests/fsdrestartcase` arms every file server to end under a write to -/// `/home/fsd_end` (`--end-on`) and at the first hello on `/apps` -/// (`--end-at-hello`), and `test_rs_fs_restart` ends DATA's four times, the +/// `/home/fsd_end` (`--end-on`) and at the first request on `/apps` +/// (`--end-at-request`), and `test_rs_fs_restart` ends DATA's four times, the /// first under init's own resolution of a launch: the guest asserts what a /// client sees, init's and fsd's own lines say who ended and who started /// again, and with the machine down the DATA partition is read by this @@ -659,9 +659,9 @@ pub fn fsd_restart( if ended != 3 { return Err(format!("fsd said it ended under a write {ended} times, not the guest's 3:\n{log}")); } - let at_hello = log.matches("fsd: --end-at-hello: ending before /apps's first hello is answered").count(); - if at_hello != 1 { - return Err(format!("fsd said it ended at a hello {at_hello} times, not the launch's 1:\n{log}")); + let at_request = log.matches("fsd: --end-at-request: ending before /apps's first request is answered").count(); + if at_request != 1 { + return Err(format!("fsd said it ended under a request {at_request} times, not the launch's 1:\n{log}")); } let restarted = log.lines().filter(|l| l.contains("init: fsd data (pid ") && l.contains("ended; started again")).count(); if restarted != 3 { diff --git a/tests/fsdrestartcase/system.toml b/tests/fsdrestartcase/system.toml index 643c7b41944..e64595ac5df 100644 --- a/tests/fsdrestartcase/system.toml +++ b/tests/fsdrestartcase/system.toml @@ -1,8 +1,8 @@ # The boot `fsd_restart` judges: the test estate's shape, with every file # server armed to end the moment it has taken a write through a file opened to -# append at `/home/fsd_end` and before it answers it, and at the first hello +# append at `/home/fsd_end` and before it answers it, and at the first request # on `/apps` this boot — `test_rs_fs_restart` ends DATA's four times, the -# first through a launch init resolves, and init restarts it three. +# first under a launch init resolves, and init restarts it three. [boot] start = ["logd", "blockd", "fsd", "test-runner"] @@ -36,11 +36,11 @@ devices = ["pci:1b36:0010"] # The file servers: DATA, the log and the running slot's volume, one process # each, serving every program the directories of its role. Started again when -# one ends, on the same ports. `--end-on` and `--end-at-hello` are the test's +# one ends, on the same ports. `--end-on` and `--end-at-request` are the test's # actuators: a path on the DATA volume and a directory of DATA's, which # neither other role's volume carries. [programs.fsd] restart = true roles = ["data", "log", "boot"] receives = ["block"] -args = ["--end-on", "home/fsd_end", "--end-at-hello", "/apps"] +args = ["--end-on", "home/fsd_end", "--end-at-request", "/apps"] diff --git a/tests/toyos-rust-tests/src/bin/fs_restart.rs b/tests/toyos-rust-tests/src/bin/fs_restart.rs index cbefc62cb8b..57136fedde1 100644 --- a/tests/toyos-rust-tests/src/bin/fs_restart.rs +++ b/tests/toyos-rust-tests/src/bin/fs_restart.rs @@ -2,13 +2,14 @@ //! //! Booted by `common::storage::fsd_restart` on `tests/fsdrestartcase`, whose //! file servers end the moment they take a write through a file opened to -//! append at `/home/fsd_end` and before they answer it, and at the first hello +//! append at `/home/fsd_end` and before they answer it, and at the first request //! on `/apps` this boot: //! -//! - a launch of an `/apps` path is answered though DATA's server ends while -//! init resolves it: init makes that call on its file worker and starts the -//! server again while it waits, where a call from its loop would wait for -//! ever on a server only its loop could start; +//! - a launch of an `/apps` path is answered though DATA's server ends under +//! init's request resolving it: init's retry connects again and waits in the +//! port's queue, on its file worker, and init starts the server again while +//! it waits — a call from its loop would wait for ever on a server only its +//! loop could start; //! - a file written and `fsync`ed before an end — acknowledged and flushed — //! reads back through a handle held across it, and a handle written across //! it goes on writing where it was; @@ -38,7 +39,7 @@ const REPLACING: &[u8] = b"renamed over the file a handle held across the end"; /// `--end-on`'s path, in `tests/fsdrestartcase/system.toml`. const END: &str = "/home/fsd_end"; -/// A path under `--end-at-hello`'s directory: no package answers for it. +/// A path under `--end-at-request`'s directory: no package answers for it. const LAUNCHED: &str = "/apps/fs_restart/nothing"; /// Mirrored: what `KEPT` holds. @@ -63,7 +64,7 @@ fn end_the_server(n: u32) { } fn main() { - // End 1: the first hello on `/apps` is init's, resolving this launch. + // End 1: the first request on `/apps` is init's, resolving this launch. match Command::new(LAUNCHED).status() { Err(e) => println!("fs_restart: end 1: the launch of {LAUNCHED} was answered, refused ({e})"), Ok(status) => panic!("{LAUNCHED}, which no package answers for, ran and exited {status}"), diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index fde4e2cda98..141b8231e88 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -7,7 +7,7 @@ //! nothing else of the machine. Argv is the role, and for LOG and BOOT the //! unique GUID of the partition the loader named for it when no claim on it //! was minted; then the row's own arguments, which are tests' actuators -//! ([`END_ON`], [`END_AT_HELLO`]). +//! ([`END_ON`], [`END_AT_REQUEST`]). //! //! **A connection is bound to the directory whose port it came in on**, and //! every path on it is resolved there (`fsd::resolve`); a write on a read-only @@ -82,11 +82,12 @@ const RAM_BLOCKS: u64 = 1 << 18; /// stages one. Armed by nothing but a boot config's `args`. const END_ON: &str = "--end-on"; -/// `--end-at-hello `: the first hello on `dir`'s port this boot ends this -/// process before it is answered — a server killed while its first client -/// waits, as a test stages one. Once a boot, across restarts: the kernel's -/// `/tmp` keeps the mark. Armed by nothing but a boot config's `args`. -const END_AT_HELLO: &str = "--end-at-hello"; +/// `--end-at-request `: the first request on `dir`'s port this boot after +/// a hello ends this process before it is answered — a server killed under a +/// client that holds a connection, whose retry connects again and waits in the +/// port's queue, as a test stages one. Once a boot, across restarts: the +/// kernel's `/tmp` keeps the mark. Armed by nothing but a boot config's `args`. +const END_AT_REQUEST: &str = "--end-at-request"; const TOKEN_CLIENT: u64 = 1 << 32; const TOKEN_STREAM: u64 = 2 << 32; @@ -173,14 +174,14 @@ fn main() { let role = args.get(1).and_then(|r| Role::parse(r)).unwrap_or_else(|| { panic!("fsd: started with {args:?}; the first argument is a role: data, log or boot") }); - let (mut guid, mut end_on, mut end_at_hello) = (None, None, None); + let (mut guid, mut end_on, mut end_at_request) = (None, None, None); let mut rest = args.iter().skip(2); while let Some(arg) = rest.next() { match arg.as_str() { END_ON => end_on = Some(rest.next().unwrap_or_else(|| panic!("fsd: {END_ON} takes a path")).clone()), - END_AT_HELLO => { - end_at_hello = - Some(rest.next().unwrap_or_else(|| panic!("fsd: {END_AT_HELLO} takes a directory")).clone()) + END_AT_REQUEST => { + end_at_request = + Some(rest.next().unwrap_or_else(|| panic!("fsd: {END_AT_REQUEST} takes a directory")).clone()) } flag if flag.starts_with("--") => panic!("fsd: {flag} is no argument of this server"), named if guid.is_none() => guid = Some(named), @@ -206,7 +207,7 @@ fn main() { next_stream: 0, dirty_since: None, end_on, - end_at_hello, + end_at_request, probe: Poller::new(caps_len), scratch: Vec::new(), } @@ -353,8 +354,8 @@ struct Server { dirty_since: Option, /// [`END_ON`]'s path on the volume. end_on: Option, - /// [`END_AT_HELLO`]'s directory. - end_at_hello: Option, + /// [`END_AT_REQUEST`]'s directory. + end_at_request: Option, /// Asks an acceptor whether a connection waits, before [`Server::accept`] /// takes it. probe: Poller, @@ -381,16 +382,21 @@ fn stat_reply(meta: Meta) -> Reply { Reply { kind: meta.kind.wire(), value2: meta.size, mtime: meta.mtime, ..Reply::ok() } } -/// Whether this is [`END_AT_HELLO`]'s first firing this boot for `dir`: its +/// Whether this is [`END_AT_REQUEST`]'s first firing this boot for `dir`: its /// mark is made once in the kernel's `/tmp`, which outlives this process and -/// not the boot. A mark that can be neither made nor found is an actuator -/// that cannot do its one job. -fn first_hello_this_boot(dir: &str) -> bool { - let mark = format!("/tmp/fsd-end-at-hello{}", dir.replace('/', "-")); - match std::fs::OpenOptions::new().write(true).create_new(true).open(&mark) { - Ok(_) => true, - Err(e) if e.kind() == std::io::ErrorKind::AlreadyExists => false, - Err(e) => panic!("fsd: {END_AT_HELLO}: its mark {mark} would not be made: {e}"), +/// not the boot. Looked for and then made, not made exclusively — the std +/// fork's `create_new` on a kernel path is not exclusive — which one server +/// of a role at a time makes safe. A mark that can be neither found nor made +/// is an actuator that cannot do its one job. +fn first_request_this_boot(dir: &str) -> bool { + let mark = format!("/tmp/fsd-end-at-request{}", dir.replace('/', "-")); + match std::fs::metadata(&mark) { + Ok(_) => false, + Err(e) if e.kind() == std::io::ErrorKind::NotFound => { + std::fs::write(&mark, b"").unwrap_or_else(|e| panic!("fsd: {END_AT_REQUEST}: its mark {mark}: {e}")); + true + } + Err(e) => panic!("fsd: {END_AT_REQUEST}: its mark {mark} would not be looked up: {e}"), } } @@ -616,17 +622,17 @@ impl Server { Ok(window) => client.window = Some(window), Err(_) => return Answer::Drop("its window would not map"), } - let dir = &self.caps[self.clients[&id].cap].dir; - if self.end_at_hello.as_ref() == Some(dir) && first_hello_this_boot(dir) { - println!("fsd: {END_AT_HELLO}: ending before {dir}'s first hello is answered"); - std::process::exit(1); - } let rights = if self.volume.writable() { RIGHT_WRITE } else { 0 }; return Answer::Reply(Reply { value: rights, ..Reply::ok() }); } if self.clients[&id].window.is_none() { return Answer::Drop("it asked before it lent a window"); } + let dir = &self.caps[self.clients[&id].cap].dir; + if self.end_at_request.as_ref() == Some(dir) && first_request_this_boot(dir) { + println!("fsd: {END_AT_REQUEST}: ending before {dir}'s first request is answered"); + std::process::exit(1); + } match self.serve_one(id, op, r) { Ok(answer) => answer, Err(e) => Answer::Reply(Reply::refused(e)), From 73ddd5294ec9481f71a70539cdce09a559ce2b69 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 15:45:39 +0200 Subject: [PATCH 15/54] storage: the stop counts init's file worker; the cached read measured, and what load showed, filed `quiesce_stops_the_machine` counts every userland thread the stop names; init's file worker is one more. The cached read, measured by the same interleaved guest runs as before (an uncommitted bench and a probe in fsd's `READ`, applied and restored), one guest, 256 KiB requests on a 16 MiB file in DATA's cache against the kernel's `/tmp`: before (the volatile loops, a zeroed buffer per request): fsd 1952-1975 us a request (127-128 MiB/s), 945-954 us of it in the server after: fsd 988-997 us (251-253 MiB/s), 107 us in the server the kernel: 99.5-99.8 us 2 MiB requests went from 367-402 MiB/s to 1159-1392 MiB/s. What is left at 256 KiB is the client's one copy out of the window: a 256 KiB `copy_nonoverlapping` out of a shared mapping costs 2.86 ns a byte under TCG in the same guest, against 0.23 between two heap buffers. The issue that recorded the copies as a compromise goes; one asking for the measurement on metal takes its place. Filed from load-coincident reds on a loaded dev host, each green alone: threads of one process starving on one directory's connection (`quiesce_stops_the_machine`), the stop's flush and syncs outlasting `quiesce-last`'s 10 s hold (`quiesce_wakes_on_the_last_exit`), and `swap_crash_rolls_back`'s log stream redial spending its ceiling inside the swap's probation. Co-Authored-By: Claude Opus 5.5 --- ...-up-its-log-stream-inside-the-probation.md | 21 +++++++ ...-server-costs-fifteen-times-the-kernels.md | 55 ------------------- ...-file-server-is-measured-only-under-tcg.md | 30 ++++++++++ ...-connection-and-one-can-starve-the-rest.md | 23 ++++++++ ...le-syncs-can-outlast-quiesce-lasts-hold.md | 25 +++++++++ tests/common/power.rs | 8 +-- 6 files changed, 103 insertions(+), 59 deletions(-) create mode 100644 issues/build/swap-crash-rolls-back-gives-up-its-log-stream-inside-the-probation.md delete mode 100644 issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md create mode 100644 issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md create mode 100644 issues/filesystem/threads-of-one-process-take-turns-on-one-connection-and-one-can-starve-the-rest.md create mode 100644 issues/kernel/the-stops-flush-and-file-syncs-can-outlast-quiesce-lasts-hold.md diff --git a/issues/build/swap-crash-rolls-back-gives-up-its-log-stream-inside-the-probation.md b/issues/build/swap-crash-rolls-back-gives-up-its-log-stream-inside-the-probation.md new file mode 100644 index 00000000000..9438c029284 --- /dev/null +++ b/issues/build/swap-crash-rolls-back-gives-up-its-log-stream-inside-the-probation.md @@ -0,0 +1,21 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# `swap_crash_rolls_back` gives up its log stream inside the swap's probation + +The test swaps netd for a binary that panics at once, and judges init's words +on the swap off logd's network stream, which netd carries. The stream is down +for the whole of the failed binary's probation (`toyos_swap::PROBATION_MS`, +5 s) until init restores netd, and the harness redials it with a ceiling of +64 refusals. On a loaded dev host, run as one named test, the ceiling was +spent before the restore twice: "the stream's redial was turned away 64 +time(s), its ceiling of 64, and gave up", with init's words read as +`["accepted"]`, while the console of the same run carries every word on time +— stopping, started, failed at 5.953 s, restored at 5.985 s. Run alone right +after, both times green in 8–9 s. + +**Exit**: the redial's bound is the probation plus the restore, stated in time +and not in refusals, and the wide run green on a loaded host. diff --git a/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md b/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md deleted file mode 100644 index 2e755a85524..00000000000 --- a/issues/filesystem/a-cached-read-from-a-file-server-costs-fifteen-times-the-kernels.md +++ /dev/null @@ -1,55 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# A cached read from a file server costs fifteen to twenty times the kernel's - -Measured on QEMU TCG (`Profile::Metal`, DATA on NVMe), 16 MiB on `/home`, -three interleaved runs per arm, the branch that moved DATA to -`/system/bin/fsd` against main at `16d2e645`: - -| | main (kernel) | fsd | -|---|---|---| -| write 256 KiB at a time, one fsync | 37–64 MiB/s | 31 MiB/s | -| read back at once, 256 KiB at a time | 1164–1642 MiB/s | 80 MiB/s | -| read on the next boot | 84 MiB/s | 41–42 MiB/s | -| 4 KiB create and fsync, p50 | 1.0–1.3 ms | 3.5–3.6 ms | - -## Where a request goes - -One guest, the file in fsd's block cache (64 MiB of clean blocks, so all of -it) and the same bytes in the kernel's `/tmp`, three interleaved passes; fsd's -`READ` arm timed from inside the server: - -| per request | 4 KiB | 256 KiB | 2 MiB | -|---|---|---|---| -| kernel `/tmp` | 4.2–4.5 µs | 98–100 µs | 751–771 µs | -| fsd, at the client | 115–117 µs | 1906–1922 µs | 4603–4727 µs | -| of which the server's `vec![0; len]` | 3.4 µs | 47–49 µs | 363–490 µs | -| the block cache into it (`DataVolume::read`) | 6.2–6.4 µs | 108–115 µs | 663–665 µs | -| `window_put` into the client's window | 3.7 µs | 790–804 µs | 1794 µs | -| the gap to the next request: reply, the client's `window_take`, the next send | 102–103 µs | 973–989 µs | 1986–1988 µs | - -A request that moves no bytes (a one-byte read at the end of the file) costs -104 µs against the kernel's 2.6 µs: that is the round trip, and it is the -whole of a small read. At 256 KiB the two word-at-a-time volatile copies -through the window — `toyos::fs::window_put` in the server and `window_take` -in the client, 2.9–3.5 ns a byte each at 64 KiB to 1 MiB — are about 87% of -the request; the zeroing and the cache copy are 8%, and the round trip 5%. -The same two loops ran at 0.86–0.90 ns a byte on 2 MiB requests and at -0.69–3.17 ns a byte over local memory: under TCG their price per byte depends -on the addresses, a mechanism not isolated here. The kernel's copy is one, at -0.37 ns a byte. - -Not the cause: the client holds no cache, but the server's cache answers -every block (the `DataVolume::read` row); and nothing contends for a lock — -the reader is one thread and fsd is one thread over a `RefCell`. - -The round trip is what a served read is and does not go; the copies are a -defect. Timings are TCG's: a verdict on their size belongs on metal. - -**Exit**: a 256 KiB cached read within a small factor of the kernel's in the -same guest, the block copied once into the window and the window's copies no -longer a scalar volatile loop, measured by the same interleaved runs. diff --git a/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md b/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md new file mode 100644 index 00000000000..88e66e03828 --- /dev/null +++ b/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md @@ -0,0 +1,30 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# A cached read from a file server is measured only under TCG, where a copy out of a shared mapping is slow + +One guest (QEMU TCG, the test estate's boot), a 16 MiB file in DATA's block +cache and the same bytes in the kernel's `/tmp`, three interleaved passes of +256 KiB reads: + +| per request | fsd | kernel `/tmp` | +|---|---|---| +| at the client | 988–997 µs (251–253 MiB/s) | 99.5–99.8 µs | +| of which the server's `READ` | 107 µs | — | +| a 4 KiB read, the round trip | 153 µs | 4.1 µs | + +The server copies each cached block into the client's window once, and the +client copies the window into its buffer once (`toyos::fs::window_take`, one +`copy_nonoverlapping`). In the same guest a 256 KiB `copy_nonoverlapping` out +of a shared-memory mapping into a heap buffer costs 2.86 ns a byte — about +750 µs, the rest of the request — where the same copy between two heap +buffers costs 0.23 ns a byte and a 2 MiB one out of the mapping 0.45 ns. What +under TCG makes a copy out of the mapping slow at 256 KiB is not isolated, and +nothing says whether metal pays it. + +**Exit**: the same interleaved runs on the T14, with the copy out of a +shared mapping timed beside a heap-to-heap one; a cost metal also pays is a +defect filed against the mapping, one it does not is closed here. diff --git a/issues/filesystem/threads-of-one-process-take-turns-on-one-connection-and-one-can-starve-the-rest.md b/issues/filesystem/threads-of-one-process-take-turns-on-one-connection-and-one-can-starve-the-rest.md new file mode 100644 index 00000000000..df3f2a315a2 --- /dev/null +++ b/issues/filesystem/threads-of-one-process-take-turns-on-one-connection-and-one-can-starve-the-rest.md @@ -0,0 +1,23 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# Threads of one process take turns on one connection, and one can starve the rest + +The std fork holds one connection per directory per process, behind a +`Mutex` (`rust/library/std/src/sys/fs/toyos.rs`, +`Capability`), and a call holds it across the whole request and reply. std's +mutex on ToyOS is the futex one, which lets a thread that lets go take the +lock again before a waiter it woke has run. A thread whose calls come back to +back can so keep every other thread of its process off the directory. + +Seen on `quiesce_stops_the_machine` on a loaded dev host, one named test run +wide: `quiesce_writers`' six threads each create, write and fsync a file on +`/log` in a loop, and one of the six reached its loop in the test's 5 s; run +alone, all six did. + +**Exit**: a thread waiting for a directory's connection is served in its turn +— a lock that hands over, or a connection per thread — with a guest test +whose threads each count their passes on one directory and none is starved. diff --git a/issues/kernel/the-stops-flush-and-file-syncs-can-outlast-quiesce-lasts-hold.md b/issues/kernel/the-stops-flush-and-file-syncs-can-outlast-quiesce-lasts-hold.md new file mode 100644 index 00000000000..2d6216339c7 --- /dev/null +++ b/issues/kernel/the-stops-flush-and-file-syncs-can-outlast-quiesce-lasts-hold.md @@ -0,0 +1,25 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# The stop's flush and file syncs can outlast `quiesce-last`'s hold + +init sequences a stop as logd's flush, bounded by `FLUSH_BOUND` (5 s), then a +sync of every writable file server's volume, each bounded by the same 5 s +(`userland/init/src/main.rs`, `Init::stop`, `sync_files`), and only then asks +the kernel. `quiesce-last-exit` and `quiesce-last-park` hold their thread for +`STAGED` (10 s, `kernel/src/quiesce.rs`) waiting for the stop to come down to +it alone, and panic past it. The two bounds are not held to each other: a +stop that spends its whole flush bound and part of the syncs' reaches the +kernel after the hold has given up. + +Seen on `quiesce_wakes_on_the_last_exit`, red wide and green alone, one +named test on a loaded dev host: logd's write to `/log` took 5.016 s, init +said logd did not answer the flush in 5000 ms, and the kernel panicked at +10.499 s naming the stop that never came down to the held thread. + +**Exit**: the staging's bound is derived from the stop's own worst case — the +flush bound plus every sync's — or the stop's bounds are one budget the +staging reads, with the wide run green on a loaded host. diff --git a/tests/common/power.rs b/tests/common/power.rs index d740f9755b9..ec4bc7b5c0a 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -171,9 +171,9 @@ pub fn quiesce_stops_the_machine( // The threads the stop names besides the writers: the job's own main // thread, parked on init's answer; `test-runner`'s main and deadline // threads; `logd`'s; `blockd`'s; one per file server, three roles; and - // `init`'s waiter on each of those four services. `init`'s main thread - // asked for the stop and is its caller. - const OTHERS: u32 = 1 + 2 + 1 + 1 + 3 + 4; + // `init`'s waiter on each of those four services, and its file worker. + // `init`'s main thread asked for the stop and is its caller. + const OTHERS: u32 = 1 + 2 + 1 + 1 + 3 + 4 + 1; let (whole, record) = stopped_boot("tests/quiescecase/system.toml", JOB, &[LATE_WORD], rust_bins)?; if record.in_flight != 0 { @@ -197,7 +197,7 @@ pub fn quiesce_stops_the_machine( if record.sweep.total() != WRITERS + OTHERS { return Err(format!( "this boot's stop named {} userland thread(s); {WRITERS} writers plus the {OTHERS} \ - of the job, test-runner, logd, the storage services and init's waiters make {}, so this is not the machine the writers \ + of the job, test-runner, logd, the storage services, init's waiters and its file worker make {}, so this is not the machine the writers \ were on:\n {record}\n{whole}", record.sweep.total(), WRITERS + OTHERS, From 06a60776f9f423e6bfc5b9b3e478b340d33761aa Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 15:51:21 +0200 Subject: [PATCH 16/54] metal: the shared block's third boot is priced like its second The four guest binaries this round adds to the shared block (`fs_stream_offset`, `fs_client_bound`, `fs_cache_eviction`, `spawn_image_object`) cut its T14 list in three, and the metal profile refuses a list nobody priced. `shared-3`'s rows carry `shared-2`'s ceilings, from the same constants, and no `measured`: no run has taken that boot yet. With them, `--metal --metal-readback` stages all 25 images (EXIT=2, staged and not judged, the machine untouched). Co-Authored-By: Claude Opus 5.5 --- tests/metal-profile.toml | 42 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 42 insertions(+) diff --git a/tests/metal-profile.toml b/tests/metal-profile.toml index fa0ece05879..3300d08fd14 100644 --- a/tests/metal-profile.toml +++ b/tests/metal-profile.toml @@ -544,6 +544,12 @@ unit = "ms" ceiling = 1400 ceiling_from = "as list.shared.job_ms — the same list, cut in two" +[[number]] +name = "list.shared-3.job_ms" +unit = "ms" +ceiling = 1400 +ceiling_from = "as list.shared.job_ms — the same list, cut in three" + [[number]] name = "list.shared-debug.job_ms" unit = "ms" @@ -656,18 +662,36 @@ unit = "ms" ceiling = 60000 ceiling_from = "toyos_tco::JOB_BOUND_MS — as boot.testcases.complete_ms" +[[number]] +name = "boot.shared-3.complete_ms" +unit = "ms" +ceiling = 60000 +ceiling_from = "toyos_tco::JOB_BOUND_MS — as boot.testcases.complete_ms" + [[number]] name = "boot.shared-2.back_secs" unit = "s" ceiling = 420 ceiling_from = "toyos_build::metal::return_secs" +[[number]] +name = "boot.shared-3.back_secs" +unit = "s" +ceiling = 420 +ceiling_from = "toyos_build::metal::return_secs" + [[number]] name = "boot.shared-2.stick_secs" unit = "s" ceiling = 30 ceiling_from = "as boot.testcases.stick_secs" +[[number]] +name = "boot.shared-3.stick_secs" +unit = "s" +ceiling = 30 +ceiling_from = "as boot.testcases.stick_secs" + [[number]] name = "boot.ccorpus-2.complete_ms" unit = "ms" @@ -838,6 +862,12 @@ unit = "us" ceiling = 16667 ceiling_from = "as boot.testcases.panel_max_us" +[[number]] +name = "boot.shared-3.panel_max_us" +unit = "us" +ceiling = 16667 +ceiling_from = "as boot.testcases.panel_max_us" + [[number]] name = "boot.shared-debug.panel_max_us" unit = "us" @@ -961,6 +991,12 @@ unit = "us" ceiling = 100000 ceiling_from = "as boot.testcases.panel_us" +[[number]] +name = "boot.shared-3.panel_us" +unit = "us" +ceiling = 100000 +ceiling_from = "as boot.testcases.panel_us" + [[number]] name = "boot.shared-debug.panel_us" unit = "us" @@ -1076,6 +1112,12 @@ unit = "operations" ceiling = 0 ceiling_from = "as boot.testcases.park_open_operations" +[[number]] +name = "boot.shared-3.park_open_operations" +unit = "operations" +ceiling = 0 +ceiling_from = "as boot.testcases.park_open_operations" + [[number]] name = "boot.shared-debug.park_open_operations" unit = "operations" From e00e669f4a38b550d2482d58c3a841335b7c2b6b Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 15:52:23 +0200 Subject: [PATCH 17/54] issues: the cached read's numbers over both boots of the after arm Co-Authored-By: Claude Opus 5.5 --- ...-from-a-file-server-is-measured-only-under-tcg.md | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md b/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md index 88e66e03828..1185b610d34 100644 --- a/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md +++ b/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md @@ -6,15 +6,15 @@ opened: 2026-09-27 # A cached read from a file server is measured only under TCG, where a copy out of a shared mapping is slow -One guest (QEMU TCG, the test estate's boot), a 16 MiB file in DATA's block -cache and the same bytes in the kernel's `/tmp`, three interleaved passes of -256 KiB reads: +QEMU TCG, the test estate's boot, a 16 MiB file in DATA's block cache and the +same bytes in the kernel's `/tmp`, three interleaved passes a boot over two +boots: | per request | fsd | kernel `/tmp` | |---|---|---| -| at the client | 988–997 µs (251–253 MiB/s) | 99.5–99.8 µs | -| of which the server's `READ` | 107 µs | — | -| a 4 KiB read, the round trip | 153 µs | 4.1 µs | +| 256 KiB at the client | 988–1069 µs (234–253 MiB/s) | 99.5–102.3 µs | +| of which the server's `READ` | 107–110 µs | — | +| a 4 KiB read, the round trip | 153–188 µs | 3.9–5.1 µs | The server copies each cached block into the client's window once, and the client copies the window into its buffer once (`toyos::fs::window_take`, one From 506bbdc60480cbc59c25df8667ae3654b8f46e2d Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 17:01:05 +0200 Subject: [PATCH 18/54] merge: the prose the resolution rewrote is deleted The table-read injector's module doc, `root_withheld`'s doc and the storm check's comment were rewritten in the merge; they go, and the check keeps one clause for its fixed window. Co-Authored-By: Claude Opus 5.5 --- ...-up-its-log-stream-inside-the-probation.md | 21 ------------- ...oss-a-restart-by-what-its-entry-records.md | 31 ------------------- kernel/src/block.rs | 8 ++--- tests/common/partclaim.rs | 4 --- tests/common/volumes.rs | 7 +---- 5 files changed, 3 insertions(+), 68 deletions(-) delete mode 100644 issues/build/swap-crash-rolls-back-gives-up-its-log-stream-inside-the-probation.md delete mode 100644 issues/filesystem/a-held-file-is-known-across-a-restart-by-what-its-entry-records.md diff --git a/issues/build/swap-crash-rolls-back-gives-up-its-log-stream-inside-the-probation.md b/issues/build/swap-crash-rolls-back-gives-up-its-log-stream-inside-the-probation.md deleted file mode 100644 index 9438c029284..00000000000 --- a/issues/build/swap-crash-rolls-back-gives-up-its-log-stream-inside-the-probation.md +++ /dev/null @@ -1,21 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-09-27 ---- - -# `swap_crash_rolls_back` gives up its log stream inside the swap's probation - -The test swaps netd for a binary that panics at once, and judges init's words -on the swap off logd's network stream, which netd carries. The stream is down -for the whole of the failed binary's probation (`toyos_swap::PROBATION_MS`, -5 s) until init restores netd, and the harness redials it with a ceiling of -64 refusals. On a loaded dev host, run as one named test, the ceiling was -spent before the restore twice: "the stream's redial was turned away 64 -time(s), its ceiling of 64, and gave up", with init's words read as -`["accepted"]`, while the console of the same run carries every word on time -— stopping, started, failed at 5.953 s, restored at 5.985 s. Run alone right -after, both times green in 8–9 s. - -**Exit**: the redial's bound is the probation plus the restore, stated in time -and not in refusals, and the wide run green on a loaded host. diff --git a/issues/filesystem/a-held-file-is-known-across-a-restart-by-what-its-entry-records.md b/issues/filesystem/a-held-file-is-known-across-a-restart-by-what-its-entry-records.md deleted file mode 100644 index 65550738888..00000000000 --- a/issues/filesystem/a-held-file-is-known-across-a-restart-by-what-its-entry-records.md +++ /dev/null @@ -1,31 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# A held file is known across a file server's restart by what its entry records - -A handle held across a restart of its file server reopens the file by path -and keeps it only when the server answers the identity the handle last saw -(`toyos::fs::Stat::ident`). The volumes have no object id to answer with: the -interim DATA format's entry is a name, a length, an mtime and extents -(`bcachefs/src/fs.rs`), and a FAT entry a short name, a creation stamp, a -first cluster and a length. So the identity is a hash of what the entry -records — on DATA the first block, the length and the mtime, which the format -keeps to the second; on FAT the short name, the creation stamp, the first -cluster and the length (`userland/fsd/src/volume.rs`, `Volume::ident`). - -That tells a file renamed over the held one, a file made again at its path, -and a write the restart lost from the file the handle held. It does not tell -a replacement that takes the held file's freed first block — the allocator -hands back the most recently freed first — and has the same length, written in -the same second: the handle then writes into the replacement. A file with no -block yet is refused outright (`ident` 0), since nothing tells it from -another. - -**Exit**: DATA's entry carries an object id and a generation that nothing -reuses, and `Volume::ident` answers them — the format the track moves DATA to -has inode numbers — with a host test that deletes a held file, makes a -same-length file in its freed block within the second, and has the reopen -refused. diff --git a/kernel/src/block.rs b/kernel/src/block.rs index e854f2d5de5..978033307b1 100644 --- a/kernel/src/block.rs +++ b/kernel/src/block.rs @@ -258,9 +258,6 @@ pub fn register(dev: Box) -> Option { Some(handle) } -/// `partclaim-table-unanswered` and `partclaim-root-withheld`: a registered disk -/// that refuses every read of its device block 0 — its protective MBR and GPT -/// header — from [`unanswered::refuse`] until [`unanswered::answer`]. #[cfg(feature = "boot-actuators")] pub mod unanswered { use core::sync::atomic::{AtomicBool, Ordering}; @@ -306,13 +303,12 @@ pub mod unanswered { /// block 0. pub fn refuse() { REFUSING.store(true, Ordering::Relaxed); - log!("unanswered: device block 0 of every disk refuses reads from now on"); + log!("block: device block 0 of every disk refuses reads from now on"); } - /// Every disk answers reads of its block 0 again. pub fn answer() { REFUSING.store(false, Ordering::Relaxed); - log!("unanswered: device block 0 of every disk answers reads again"); + log!("block: device block 0 of every disk answers reads again"); } } diff --git a/tests/common/partclaim.rs b/tests/common/partclaim.rs index 95b5e70cea1..2e96b47d586 100644 --- a/tests/common/partclaim.rs +++ b/tests/common/partclaim.rs @@ -281,10 +281,6 @@ pub fn partition_claim_gives_up( Ok(()) } -/// ROOT's source on the boot stick, which did not answer ROOT's hold and -/// answers every read after it: its GUID is withheld, so the claim that now -/// finds its span on a disk that answers, and unheld, is refused as the -/// kernel's. fn root_withheld( config: &Path, crafted: &Path, diff --git a/tests/common/volumes.rs b/tests/common/volumes.rs index 0c58ffbca87..527f56919ef 100644 --- a/tests/common/volumes.rs +++ b/tests/common/volumes.rs @@ -2089,12 +2089,7 @@ pub fn log_flush_retry( volume_lines(&log) )); } - // And refused once per partition and kind, not on every flush: `logd` - // flushes every round it wrote a line in, and a refused flush is records of - // its own, so a refusal on every flush keeps `/log` retrying for as long as - // the machine runs. An absence has no event to wait on, so it is judged over - // a fixed window: the storm retries there without pause, the once-per-kind - // refusal never. + // An absence has no event to wait on, so a fixed window judges it. let after = qemu.drain_serial(Duration::from_secs(2)); let again: Vec<&str> = after.lines().filter(|l| l.contains("partclaim: a flush durable on attempt")).collect(); From 9777ac345b9634114a7706f71d61e80f3f802db0 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 17:01:19 +0200 Subject: [PATCH 19/54] redlist: swap_crash_rolls_back's turned-away redial is main's, quarantined The signature main's nightly at 1ce71831 left, "the stream's redial was turned away", is quarantined against main's issue, and what the branch's duplicate measured moves there. The duplicate's deletion went in with 506bbdc6, as did the deletion of the hash identity's issue. Co-Authored-By: Claude Opus 5.5 --- ...ls-back-redial-turned-away-once-on-mains-nightly.md | 10 ++++++++-- src/redlist.rs | 5 +++++ 2 files changed, 13 insertions(+), 2 deletions(-) diff --git a/issues/build/swap-crash-rolls-back-redial-turned-away-once-on-mains-nightly.md b/issues/build/swap-crash-rolls-back-redial-turned-away-once-on-mains-nightly.md index f9a3abde491..edd3a1cf308 100644 --- a/issues/build/swap-crash-rolls-back-redial-turned-away-once-on-mains-nightly.md +++ b/issues/build/swap-crash-rolls-back-redial-turned-away-once-on-mains-nightly.md @@ -1,5 +1,5 @@ --- -status: open +status: expected-red kind: finding opened: 2026-09-27 --- @@ -16,7 +16,13 @@ FAIL swap_crash_rolls_back: 2 finding(s): `ALONE swap_crash_rolls_back: GREEN` twice. It was not red on the nightly before #527 (run 36285169430). -`cargo run -- --known-red swap_crash_rolls_back` answers NO. + +Twice on #536, one named test on a dev host loaded by another agent's LLVM +build, green alone right after: the stream is down for the failed binary's +whole probation (`toyos_swap::PROBATION_MS`, 5 s) until init restores netd, +the harness's redial ceiling is 64 refusals and not a time, and the console +of the same runs carries every word on time (failed at 5.953 s, restored at +5.985 s). Not shown: what turned the redial away 64 times, and why. diff --git a/src/redlist.rs b/src/redlist.rs index 27f970c681f..f4fa5599a09 100644 --- a/src/redlist.rs +++ b/src/redlist.rs @@ -114,6 +114,11 @@ pub const QUARANTINE: &[Quarantined] = &[ says: &["no \"byte budget; refused\" line — the kernel refused nothing"], issue: "issues/kernel/so-cache-refusals-saw-the-kernel-refuse-nothing-once.md", }, + Quarantined { + test: "swap_crash_rolls_back", + says: &["the stream's redial was turned away"], + issue: "issues/build/swap-crash-rolls-back-redial-turned-away-once-on-mains-nightly.md", + }, Quarantined { test: "usb_disk_index_stable", says: &["nothing enumerated on the first controller; there is no renumbering to survive"], From 48428b649ded5cf384180ee0cc203aa70fc37f15 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 17:01:30 +0200 Subject: [PATCH 20/54] fs: a file held across a file server's restart answers Gone; a directory's connection is handed over in turn B3, the orchestrator's ruling: a hash of what an entry records is no identity, since the allocator hands a freed block back first and a same-length file written within the second passed it. `Stat::ident` goes from the protocol, fsd's two implementations and `toyos-fat32`'s accessor, and std (fork 2b2fb74a303) no longer reopens a held file by path: a handle whose server ended answers StaleNetworkFileHandle. `fs_restart` now expects Gone from a handle held on an unchanged file across an end, reads the flushed file through a new open, and still reds if a handle on a file renamed over is reopened into it. The exit is issues/filesystem/a-file-held-across-a-file-servers-restart-answers-gone.md. std's futex mutex let the thread that let go of a directory's connection take it back before the waiter it woke ran; `quiesce_stops_the_machine`'s six writers on `/log` starved five on a loaded host. Callers take a ticket and are served in order. `fs_turns` has six threads write one directory until one made 32 passes, and every one must have made 16. Co-Authored-By: Claude Opus 5.5 --- ...oss-a-file-servers-restart-answers-gone.md | 17 +++++++ rust | 2 +- tests/common/storage.rs | 7 ++- tests/toyos-rust-tests/src/bin/fs_restart.rs | 32 +++++++----- tests/toyos-rust-tests/src/bin/fs_turns.rs | 38 ++++++++++++++ toyos-fat32/src/fs.rs | 7 --- toyos/src/fs.rs | 45 +++++----------- userland/fsd/src/absent.rs | 4 -- userland/fsd/src/data.rs | 51 +------------------ userland/fsd/src/fat.rs | 12 +---- userland/fsd/src/main.rs | 14 +++-- userland/fsd/src/volume.rs | 20 -------- 12 files changed, 98 insertions(+), 151 deletions(-) create mode 100644 issues/filesystem/a-file-held-across-a-file-servers-restart-answers-gone.md create mode 100644 tests/toyos-rust-tests/src/bin/fs_turns.rs diff --git a/issues/filesystem/a-file-held-across-a-file-servers-restart-answers-gone.md b/issues/filesystem/a-file-held-across-a-file-servers-restart-answers-gone.md new file mode 100644 index 00000000000..3a281aff943 --- /dev/null +++ b/issues/filesystem/a-file-held-across-a-file-servers-restart-answers-gone.md @@ -0,0 +1,17 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A file held across a file server's restart answers Gone + +No volume records an object id, so nothing tells the held file from another +put at its path since, and a handle held across a restart answers `Gone` +(`std`'s `sys/fs/toyos.rs`); `logd` loses `/log` for the boot if LOG's server +restarts. + +**Exit**: DATA's entry carries an object id and a generation that nothing +reuses, a held handle reopens only when both match, and a host test that +makes a same-length file in a held file's freed block within the second has +the reopen refused. diff --git a/rust b/rust index 3428b0096a3..2b2fb74a303 160000 --- a/rust +++ b/rust @@ -1 +1 @@ -Subproject commit 3428b0096a32ef7320442220f17165b2f6f6405e +Subproject commit 2b2fb74a3034c4398bc5332b44edf3aeccb04c23 diff --git a/tests/common/storage.rs b/tests/common/storage.rs index e8bd8883e4a..b1d67e2d2af 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -617,8 +617,7 @@ pub fn home_overwrite_reads_back( /// again, and with the machine down the DATA partition is read by this /// crate's own build of the `bcachefs` reader over a plain seek-and-read of /// the image — nothing the guest executed. The flushed file holds its bytes -/// there, and the file a held handle wrote across an end holds the bytes of -/// the file renamed over it and nothing that handle wrote after the next. +/// there. pub fn fsd_restart( _test_config: &Path, c_bins: &[(String, Vec)], @@ -686,8 +685,8 @@ pub fn fsd_restart( } eprintln!( " [fsd] DATA's server ended four times, the first under init's resolution of a launch that \ - was answered; started again three, its clients reopened, a handle on a file renamed over \ - answered Gone, the fourth closed /home to Gone; {KEPT} and {ACROSS} read back off the \ + was answered; started again three, every handle held across an end answered Gone, new \ + opens answered, the fourth closed /home to Gone; {KEPT} and {ACROSS} read back off the \ image by the host's own bcachefs reader" ); Ok(()) diff --git a/tests/toyos-rust-tests/src/bin/fs_restart.rs b/tests/toyos-rust-tests/src/bin/fs_restart.rs index 57136fedde1..7b0772cc085 100644 --- a/tests/toyos-rust-tests/src/bin/fs_restart.rs +++ b/tests/toyos-rust-tests/src/bin/fs_restart.rs @@ -10,11 +10,9 @@ //! port's queue, on its file worker, and init starts the server again while //! it waits — a call from its loop would wait for ever on a server only its //! loop could start; -//! - a file written and `fsync`ed before an end — acknowledged and flushed — -//! reads back through a handle held across it, and a handle written across -//! it goes on writing where it was; //! - the write the server ended under is answered as the server's end, never //! as done; +//! - a handle held across an end is answered `Gone`; //! - a handle held across an end, on a file another was renamed over, is //! answered `Gone` and writes nothing into the file now at its path; //! - DATA's server ended four times inside init's window is not started a @@ -26,14 +24,12 @@ use std::fs::{self, File, OpenOptions}; use std::io::{ErrorKind, Read, Write}; use std::process::Command; -/// Mirrored in `tests/common/storage.rs`: the flushed file, the one written -/// across the first end, and what is renamed over it. +/// Mirrored in `tests/common/storage.rs`. const KEPT: &str = "/home/fs_restart/kept"; const ACROSS: &str = "/home/fs_restart/across"; const REPLACEMENT: &str = "/home/fs_restart/replacement"; const KEPT_LEN: usize = 64 * 1024 + 13; const BEFORE: &[u8] = b"written and flushed before the server ended; "; -const AFTER: &[u8] = b"written through the same handle after it came back"; const REPLACING: &[u8] = b"renamed over the file a handle held across the end"; /// `--end-on`'s path, in `tests/fsdrestartcase/system.toml`. @@ -82,17 +78,26 @@ fn main() { end_the_server(2); + // The same files, unchanged at their paths, and still `Gone`. + match held.read(&mut [0u8; 16]) { + Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => {} + other => panic!("a handle held across the end read {other:?}, not Gone"), + } + match across.write_all(b"written through a handle held across the end") { + Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => {} + other => panic!("a handle held across the end wrote {other:?}, not Gone"), + } + drop((held, across)); let mut back = Vec::new(); - held.read_to_end(&mut back).expect("a handle held across the end reads"); - assert!(back == kept(), "the held handle read {} bytes, not the kept file", back.len()); - across.write_all(AFTER).expect("a handle held across the end writes where it was"); - across.sync_all().expect("durable after the end"); + File::open(KEPT).and_then(|mut f| f.read_to_end(&mut back)).expect("a new open reads the kept file"); + assert!(back == kept(), "a new open read {} bytes, not the kept file", back.len()); let mut whole = Vec::new(); - File::open(ACROSS).and_then(|mut f| f.read_to_end(&mut whole)).expect("read the file written across"); - assert_eq!(whole, [BEFORE, AFTER].concat(), "the file written across the end"); - println!("fs_restart: the server came back; a held handle read, another wrote on, both where they were"); + File::open(ACROSS).and_then(|mut f| f.read_to_end(&mut whole)).expect("read the file held across"); + assert_eq!(whole, BEFORE, "the file held across the end"); + println!("fs_restart: the server came back; held handles answered Gone, a new open read the flushed file"); // Another file renamed over the one `across` holds, durably, and then an end. + let mut across = OpenOptions::new().write(true).open(ACROSS).expect("hold the file again"); let mut replacement = File::create(REPLACEMENT).expect("create the replacement"); replacement.write_all(REPLACING).expect("write the replacement"); replacement.sync_all().expect("the replacement is durable"); @@ -109,6 +114,7 @@ fn main() { Err(e) => panic!("a handle on a file renamed over was refused {e} ({:?}), not Gone", e.kind()), Ok(()) => panic!("a handle on a file renamed over wrote into the file now at its path"), } + let mut held = File::open(KEPT).expect("hold the kept file for the last end"); end_the_server(4); diff --git a/tests/toyos-rust-tests/src/bin/fs_turns.rs b/tests/toyos-rust-tests/src/bin/fs_turns.rs new file mode 100644 index 00000000000..d547c95ec66 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/fs_turns.rs @@ -0,0 +1,38 @@ +//! Threads of one process sharing one directory's connection are served in turn. + +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Barrier}; + +const THREADS: usize = 6; +/// The passes the first thread to finish makes. +const PASSES: u32 = 32; + +fn main() { + std::fs::create_dir_all("/home/fs_turns").expect("make /home/fs_turns"); + let start = Arc::new(Barrier::new(THREADS)); + let done = Arc::new(AtomicBool::new(false)); + let workers: Vec<_> = (0..THREADS) + .map(|n| { + let (start, done) = (Arc::clone(&start), Arc::clone(&done)); + std::thread::spawn(move || { + let path = format!("/home/fs_turns/{n}"); + start.wait(); + let mut passes: u32 = 0; + while !done.load(Ordering::Acquire) { + std::fs::write(&path, passes.to_le_bytes()).unwrap_or_else(|e| panic!("thread {n}: write {path}: {e}")); + passes += 1; + if passes == PASSES { + done.store(true, Ordering::Release); + } + } + passes + }) + }) + .collect(); + let passes: Vec = workers.into_iter().map(|w| w.join().expect("a writer panicked")).collect(); + println!("fs_turns: passes per thread {passes:?}"); + let _ = std::fs::remove_dir_all("/home/fs_turns"); + let least = *passes.iter().min().expect("threads"); + assert!(least >= PASSES / 2, "a thread made {least} passes while another made {PASSES}: {passes:?}"); + println!("fs_turns: PASS"); +} diff --git a/toyos-fat32/src/fs.rs b/toyos-fat32/src/fs.rs index 071917c75e5..4b9a3e6bd4f 100644 --- a/toyos-fat32/src/fs.rs +++ b/toyos-fat32/src/fs.rs @@ -135,13 +135,6 @@ impl File { self.size == 0 } - /// What tells this file from another at the same path: its entry's short - /// name and creation stamp, and its first cluster — 0 for a file that has - /// none yet. - pub fn identity(&self) -> ([u8; 16], u32) { - (self.identity.0, self.first_cluster.map_or(0, Cluster::raw)) - } - /// Whether stopping now would leave the volume holding clusters this /// handle's directory entry does not reach. /// diff --git a/toyos/src/fs.rs b/toyos/src/fs.rs index 4f9ff186f28..09fa7a83a51 100644 --- a/toyos/src/fs.rs +++ b/toyos/src/fs.rs @@ -24,12 +24,9 @@ //! **A server that ends is survivable and not invisible.** [`Dir`] connects //! again through the same connector — init keeps a file server's ports open //! across its restart — and counts its connections in a generation. A file id -//! from an earlier generation names nothing on the new connection; the holder -//! of a file opens it again by its path and keeps it only when the server -//! answers the [`Stat::ident`] it last saw. Any other answer — another file at -//! the path, this one changed by a write the restart lost, or a volume that -//! cannot tell — is [`SyscallError::Gone`]. Nothing here keeps a write the -//! server acknowledged and never made durable: that is what `Fsync` is for. +//! from an earlier generation names nothing on the new connection. Nothing +//! here keeps a write the server acknowledged and never made durable: that is +//! what `Fsync` is for. use toyos_abi::syscall::{SyscallError, MAX_SERVICE_NAME}; @@ -111,15 +108,13 @@ ipc_payload! { pub flags: u64, } - /// Every reply's words. `status` is 0 or a [`SyscallError`]'s wire value; - /// `ident` is a file's [`Stat::ident`], on every reply about an open file. + /// Every reply's words. `status` is 0 or a [`SyscallError`]'s wire value. pub struct Reply { pub status: u64, pub kind: u64, pub value: u64, pub value2: u64, pub mtime: u64, - pub ident: u64, } } @@ -131,11 +126,11 @@ impl Request { impl Reply { pub const fn ok() -> Self { - Self { status: 0, kind: 0, value: 0, value2: 0, mtime: 0, ident: 0 } + Self { status: 0, kind: 0, value: 0, value2: 0, mtime: 0 } } pub const fn refused(e: SyscallError) -> Self { - Self { status: e.to_u64(), kind: 0, value: 0, value2: 0, mtime: 0, ident: 0 } + Self { status: e.to_u64(), kind: 0, value: 0, value2: 0, mtime: 0 } } fn result(self) -> Result { @@ -185,11 +180,6 @@ pub struct Stat { pub kind: u64, pub size: u64, pub mtime: u64, - /// For an open file: the server's token for this file as it stands, the - /// same across a restart of the server for the same file unchanged and - /// different for any other file at the path or any change to this one; - /// 0 where the volume cannot tell one file from another, and for a path. - pub ident: u64, } /// A write answered. @@ -199,8 +189,6 @@ pub struct Written { pub len: usize, /// The offset after them: for a file opened to append, the file's end. pub offset: u64, - /// The file's [`Stat::ident`] after it. - pub ident: u64, } /// An open answered. @@ -277,16 +265,9 @@ impl Dir { unsafe { Window::new(self.window.as_ptr(), WINDOW_BYTES) } } - /// Whether the connection is up: `false` once a call found it ended, which - /// is the one `Gone` a reconnect answers. A `Gone` the server replied is - /// about the file, and a new connection would not change it. - pub fn connected(&self) -> bool { - self.conn.is_some() - } - /// Open a fresh connection through the same connector, and lend it the /// window again. The old one's file ids name nothing from here on. - pub fn reconnect(&mut self) -> Result<(), SyscallError> { + fn reconnect(&mut self) -> Result<(), SyscallError> { self.conn = None; let name = core::str::from_utf8(&self.name[..self.name_len]).map_err(|_| SyscallError::InvalidArgument)?; let conn = self.names.open(name)?; @@ -349,8 +330,7 @@ impl Dir { &buf[..len] } - /// A call on a file id: `Gone` when `generation` is not this connection's, - /// and the holder opens the file again by its path. + /// A call on a file id: `Gone` when `generation` is not this connection's. fn fid_call(&mut self, op: u32, generation: u64, request: &Request) -> Result { if generation != self.generation || self.conn.is_none() { return Err(SyscallError::Gone); @@ -385,16 +365,15 @@ impl Dir { let len = data.len().min(WINDOW_BYTES); window_put(self.window(), 0, &data[..len]); let reply = self.fid_call(WRITE, generation, &Request { fid, offset, len: len as u64, ..Request::new() })?; - Ok(Written { len: (reply.value as usize).min(len), offset: reply.value2, ident: reply.ident }) + Ok(Written { len: (reply.value as usize).min(len), offset: reply.value2 }) } pub fn fstat(&mut self, fid: u64, generation: u64) -> Result { self.fid_call(FSTAT, generation, &Request { fid, ..Request::new() }).map(|r| stat_of(&r)) } - /// Answers the file's [`Stat::ident`] after it. - pub fn truncate(&mut self, fid: u64, generation: u64, size: u64) -> Result { - self.fid_call(TRUNCATE, generation, &Request { fid, offset: size, ..Request::new() }).map(|r| r.ident) + pub fn truncate(&mut self, fid: u64, generation: u64, size: u64) -> Result<(), SyscallError> { + self.fid_call(TRUNCATE, generation, &Request { fid, offset: size, ..Request::new() }).map(drop) } /// Make what this file was written durable on the volume. @@ -510,7 +489,7 @@ pub fn encode_entry(out: &mut [u8], kind: u64, size: u64, name: &str) -> Option< } fn stat_of(reply: &Reply) -> Stat { - Stat { kind: reply.kind, size: reply.value2, mtime: reply.mtime, ident: reply.ident } + Stat { kind: reply.kind, size: reply.value2, mtime: reply.mtime } } fn receive(conn: &Connection) -> Result { diff --git a/userland/fsd/src/absent.rs b/userland/fsd/src/absent.rs index a3fc95549f7..474083653d8 100644 --- a/userland/fsd/src/absent.rs +++ b/userland/fsd/src/absent.rs @@ -57,10 +57,6 @@ impl Volume for Absent { Err(SyscallError::NotFound) } - fn ident(&mut self, _node: Node) -> Result { - Err(SyscallError::NotFound) - } - fn read(&mut self, _node: Node, _offset: u64, _out: &mut dyn Out) -> Result { Err(SyscallError::NotFound) } diff --git a/userland/fsd/src/data.rs b/userland/fsd/src/data.rs index 03d095c3913..dcd41032ad3 100644 --- a/userland/fsd/src/data.rs +++ b/userland/fsd/src/data.rs @@ -37,7 +37,7 @@ use toyos_abi::syscall::SyscallError; use crate::cache::{Cache, Shared}; use crate::disk::{Disk, DiskError, BLOCK}; -use crate::volume::{identity, join, parent, Kind, Meta, Node, OpenHow, Out, Volume}; +use crate::volume::{join, parent, Kind, Meta, Node, OpenHow, Out, Volume}; /// The longest symlink target read back: the wire's path bound. const MAX_LINK: u64 = toyos::fs::MAX_PATH as u64; @@ -425,17 +425,6 @@ impl Volume for DataVolume { Ok(Meta { kind: Kind::File, size: open.size, mtime: open.mtime }) } - fn ident(&mut self, node: Node) -> Result { - let open = self.entry(node)?; - // The first block is this file's alone among the live ones, and the - // length and mtime move with every write; a file with no block yet is - // told from another by nothing the format records. - Ok(match open.extents.first() { - Some(first) => identity(&[first.start_block, open.size, open.mtime]), - None => 0, - }) - } - fn read(&mut self, node: Node, offset: u64, out: &mut dyn Out) -> Result { let cache = Rc::clone(&self.cache); let open = self.entry(node)?; @@ -760,44 +749,6 @@ mod tests { assert_eq!(out, vec![7; 5000]); } - /// What a holder re-opening by path after a restart compares: the same - /// file unchanged states the same identity off the device, and the file - /// renamed over it, or a write to it, does not. A file with no block is - /// told from another by nothing, and says so. - #[test] - fn an_identity_survives_a_remount_and_tells_a_replacement_apart() { - let remount = |v: DataVolume| { - let DataVolume { fs, cache, .. } = v; - drop(fs); - let disk = Rc::try_unwrap(cache).ok().expect("one owner").into_disk(); - let Probed::Mounted(again) = DataVolume::probe(disk, &["home"], clock) else { panic!("remount") }; - again - }; - let mut v = vol(); - let n = v.open("home/x", CREATE).unwrap(); - assert_eq!(v.ident(n), Ok(0), "no block yet: nothing tells it from another"); - v.write(n, 0, &[7; 5000]).unwrap(); - let held = v.ident(n).unwrap(); - assert_ne!(held, 0); - v.sync().unwrap(); - - let mut v = remount(v); - let n = v.open("home/x", PLAIN).unwrap(); - assert_eq!(v.ident(n), Ok(held), "the same file, unchanged, off the device"); - v.write(n, 5000, b"more").unwrap(); - assert_ne!(v.ident(n).unwrap(), held, "a write changes it"); - v.close(n); - - let y = v.open("home/y", CREATE).unwrap(); - v.write(y, 0, &[9; 5000]).unwrap(); - v.close(y); - v.rename("home/y", "home/x").unwrap(); - v.sync().unwrap(); - let mut v = remount(v); - let n = v.open("home/x", PLAIN).unwrap(); - assert_ne!(v.ident(n).unwrap(), held, "the file renamed over it is another"); - } - #[test] fn a_file_unlinked_while_open_answers_gone() { let mut v = vol(); diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index 9477083d461..c26caac3ec0 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -25,7 +25,7 @@ use toyos_fat32::{BlockAccess, Error, Fat32, FatTime, IoError}; use crate::cache::Cache; use crate::disk::{Disk, BLOCK}; -use crate::volume::{identity, parent, Kind, Meta, Node, OpenHow, Out, Volume}; +use crate::volume::{parent, Kind, Meta, Node, OpenHow, Out, Volume}; /// The most entries one directory listing materialises. const MAX_LIST: usize = 16_384; @@ -302,16 +302,6 @@ impl Volume for FatVolume { Ok(Meta { kind: Kind::File, size, mtime: mtime * 1_000_000_000 }) } - fn ident(&mut self, node: Node) -> Result { - let open = self.entry(node)?; - let (entry, cluster) = open.file.identity(); - if cluster == 0 { - return Ok(0); - } - let half = |at: usize| u64::from_le_bytes(entry[at..at + 8].try_into().expect("eight bytes")); - Ok(identity(&[half(0), half(8), cluster as u64, open.file.len()])) - } - fn read(&mut self, node: Node, offset: u64, out: &mut dyn Out) -> Result { let open = self.open.get_mut(&node).ok_or(SyscallError::NotFound)?; if open.gone { diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index 141b8231e88..0e78b83dd2c 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -682,9 +682,8 @@ impl Server { truncate: r.flags & O_TRUNCATE != 0, }; let node = self.volume.open(&path, how)?; - let stated = self.volume.node_meta(node).and_then(|meta| self.volume.ident(node).map(|ident| (meta, ident))); - let (meta, ident) = match stated { - Ok(stated) => stated, + let meta = match self.volume.node_meta(node) { + Ok(meta) => meta, Err(e) => { self.volume.close(node); return Err(e); @@ -701,7 +700,7 @@ impl Server { let ends = append && self.end_on.as_deref() == Some(path.as_str()); let client = self.clients.get_mut(&id).expect("pumped"); client.fids.insert(fid, Fid { node, write, append, ends }); - Ok(Answer::Reply(Reply { value: fid, ident, ..stat_reply(meta) })) + Ok(Answer::Reply(Reply { value: fid, ..stat_reply(meta) })) } CLOSE => { let client = self.clients.get_mut(&id).expect("pumped"); @@ -743,13 +742,12 @@ impl Server { std::process::exit(1); } self.dirtied(); - let ident = self.volume.ident(node)?; - Ok(Answer::Reply(Reply { value: len as u64, value2: at + len as u64, ident, ..Reply::ok() })) + Ok(Answer::Reply(Reply { value: len as u64, value2: at + len as u64, ..Reply::ok() })) } FSTAT => { let node = self.fid(id, r.fid)?.node; let meta = self.volume.node_meta(node)?; - Ok(Answer::Reply(Reply { ident: self.volume.ident(node)?, ..stat_reply(meta) })) + Ok(Answer::Reply(stat_reply(meta))) } TRUNCATE => { let f = self.fid(id, r.fid)?; @@ -762,7 +760,7 @@ impl Server { } self.volume.truncate(node, r.offset)?; self.dirtied(); - Ok(Answer::Reply(Reply { ident: self.volume.ident(node)?, ..Reply::ok() })) + Ok(Answer::Reply(Reply::ok())) } FSYNC => { self.fid(id, r.fid)?; diff --git a/userland/fsd/src/volume.rs b/userland/fsd/src/volume.rs index d66ca0be23a..ae7e1df765a 100644 --- a/userland/fsd/src/volume.rs +++ b/userland/fsd/src/volume.rs @@ -96,12 +96,6 @@ pub trait Volume { fn node_meta(&mut self, node: Node) -> Result; - /// The file as it stands, as `toyos::fs::Stat::ident` states it: what - /// the volume itself records of it, so the same after a restart for the - /// same file unchanged; 0 where the volume records nothing that tells - /// this file from another. - fn ident(&mut self, node: Node) -> Result; - /// Up to `out.len()` bytes at `offset`, from the start of `out`; 0 at or /// past the end. fn read(&mut self, node: Node, offset: u64, out: &mut dyn Out) -> Result; @@ -129,20 +123,6 @@ pub trait Volume { fn describe(&self) -> String; } -/// One token for `words`, never 0: 0 is a volume saying it cannot tell. Two -/// different lists meet on one token as two 64-bit hashes do. -pub fn identity(words: &[u64]) -> u64 { - // splitmix64's finaliser, chained over the words. - let mut h = 0x243F_6A88_85A3_08D3u64; - for &w in words { - let mut z = (h ^ w).wrapping_add(0x9E37_79B9_7F4A_7C15); - z = (z ^ (z >> 30)).wrapping_mul(0xBF58_476D_1CE4_E5B9); - z = (z ^ (z >> 27)).wrapping_mul(0x94D0_49BB_1331_11EB); - h = z ^ (z >> 31); - } - h.max(1) -} - /// The parent of `path`, `""` for a name at the root. pub fn parent(path: &str) -> &str { path.rsplit_once('/').map_or("", |(dir, _)| dir) From 0b7a20993dca6304f5f2eb8e42c4d9e944f05d5b Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 17:01:59 +0200 Subject: [PATCH 21/54] quiesce: the staged hold is derived from the stop's own bounds The thread `quiesce-last-*` holds is held before init prepares the stop, so its 10 s hold was spent by logd's flush and the file servers' syncs on a loaded host (`quiesce_wakes_on_the_last_exit`, logd's write 5.016 s). Both bounds move into toyos-quiesce, `FLUSH_MS` and `SYNC_MS`; init's syncs run together under the one `SYNC_MS`, so their worst case no longer grows with the number of roles, and the kernel's `STAGED` is their sum plus its own 10 s. Also closes the starvation issue the previous commit fixed. Co-Authored-By: Claude Opus 5.5 --- ...-connection-and-one-can-starve-the-rest.md | 23 ---------------- ...le-syncs-can-outlast-quiesce-lasts-hold.md | 25 ------------------ kernel/src/quiesce.rs | 3 ++- toyos-quiesce/src/lib.rs | 7 +++++ userland/Cargo.lock | 5 ++++ userland/init/Cargo.toml | 2 ++ userland/init/src/main.rs | 26 +++++++++++-------- 7 files changed, 31 insertions(+), 60 deletions(-) delete mode 100644 issues/filesystem/threads-of-one-process-take-turns-on-one-connection-and-one-can-starve-the-rest.md delete mode 100644 issues/kernel/the-stops-flush-and-file-syncs-can-outlast-quiesce-lasts-hold.md diff --git a/issues/filesystem/threads-of-one-process-take-turns-on-one-connection-and-one-can-starve-the-rest.md b/issues/filesystem/threads-of-one-process-take-turns-on-one-connection-and-one-can-starve-the-rest.md deleted file mode 100644 index df3f2a315a2..00000000000 --- a/issues/filesystem/threads-of-one-process-take-turns-on-one-connection-and-one-can-starve-the-rest.md +++ /dev/null @@ -1,23 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# Threads of one process take turns on one connection, and one can starve the rest - -The std fork holds one connection per directory per process, behind a -`Mutex` (`rust/library/std/src/sys/fs/toyos.rs`, -`Capability`), and a call holds it across the whole request and reply. std's -mutex on ToyOS is the futex one, which lets a thread that lets go take the -lock again before a waiter it woke has run. A thread whose calls come back to -back can so keep every other thread of its process off the directory. - -Seen on `quiesce_stops_the_machine` on a loaded dev host, one named test run -wide: `quiesce_writers`' six threads each create, write and fsync a file on -`/log` in a loop, and one of the six reached its loop in the test's 5 s; run -alone, all six did. - -**Exit**: a thread waiting for a directory's connection is served in its turn -— a lock that hands over, or a connection per thread — with a guest test -whose threads each count their passes on one directory and none is starved. diff --git a/issues/kernel/the-stops-flush-and-file-syncs-can-outlast-quiesce-lasts-hold.md b/issues/kernel/the-stops-flush-and-file-syncs-can-outlast-quiesce-lasts-hold.md deleted file mode 100644 index 2d6216339c7..00000000000 --- a/issues/kernel/the-stops-flush-and-file-syncs-can-outlast-quiesce-lasts-hold.md +++ /dev/null @@ -1,25 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# The stop's flush and file syncs can outlast `quiesce-last`'s hold - -init sequences a stop as logd's flush, bounded by `FLUSH_BOUND` (5 s), then a -sync of every writable file server's volume, each bounded by the same 5 s -(`userland/init/src/main.rs`, `Init::stop`, `sync_files`), and only then asks -the kernel. `quiesce-last-exit` and `quiesce-last-park` hold their thread for -`STAGED` (10 s, `kernel/src/quiesce.rs`) waiting for the stop to come down to -it alone, and panic past it. The two bounds are not held to each other: a -stop that spends its whole flush bound and part of the syncs' reaches the -kernel after the hold has given up. - -Seen on `quiesce_wakes_on_the_last_exit`, red wide and green alone, one -named test on a loaded dev host: logd's write to `/log` took 5.016 s, init -said logd did not answer the flush in 5000 ms, and the kernel panicked at -10.499 s naming the stop that never came down to the held thread. - -**Exit**: the staging's bound is derived from the stop's own worst case — the -flush bound plus every sync's — or the stop's bounds are one budget the -staging reads, with the wide run green on a loaded host. diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 10edad91a2c..1fab1917539 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -274,8 +274,9 @@ pub mod last { } /// How long either side waits for the other before the boot dies by name. + // The thread is held before init prepares the stop, so its wait covers that too. const STAGED: Budget = Budget::of( - Duration::from_secs(10), + Duration::from_millis(toyos_quiesce::FLUSH_MS + toyos_quiesce::SYNC_MS + 10_000), "the boot panics naming the side of the staging that never arrived", ); diff --git a/toyos-quiesce/src/lib.rs b/toyos-quiesce/src/lib.rs index 6397d07b861..bb7d73044eb 100644 --- a/toyos-quiesce/src/lib.rs +++ b/toyos-quiesce/src/lib.rs @@ -67,6 +67,13 @@ impl Sweep { } } +/// A policy number: how long init waits for `logd`'s flush before a stop. +pub const FLUSH_MS: u64 = 5_000; + +/// A policy number: how long init waits for every file server's sync, run +/// together after the flush, before a stop. +pub const SYNC_MS: u64 = 5_000; + /// The thread the kernel's `quiesce-last-park` and `quiesce-last-exit` /// actuators hold, by the name its program gives it: held in its syscall /// until it is the one thread the stop still waits on, so the transition it diff --git a/userland/Cargo.lock b/userland/Cargo.lock index cead428683f..92793503913 100644 --- a/userland/Cargo.lock +++ b/userland/Cargo.lock @@ -1654,6 +1654,7 @@ dependencies = [ "toyos-gpt", "toyos-logstream", "toyos-manifest", + "toyos-quiesce", "toyos-swap", "toyos-update", ] @@ -4153,6 +4154,10 @@ dependencies = [ name = "toyos-mixer" version = "0.1.0" +[[package]] +name = "toyos-quiesce" +version = "0.1.0" + [[package]] name = "toyos-swap" version = "0.1.0" diff --git a/userland/init/Cargo.toml b/userland/init/Cargo.toml index 846d5676975..89b5d470a2e 100644 --- a/userland/init/Cargo.toml +++ b/userland/init/Cargo.toml @@ -16,3 +16,5 @@ toyos-update = { path = "../../toyos-update" } toyos-gpt = { path = "../../toyos-gpt" } # The block service's port name, which marks a storage row. toyos-blockring = { path = "../../toyos-blockring" } +# How long a stop is prepared before the kernel is asked for it. +toyos-quiesce = { path = "../../toyos-quiesce" } diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index bf197008edb..5dd3a9285f7 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -153,7 +153,7 @@ const TOKEN_PENDING_BASE: u64 = 5; /// device flush, and a stick that takes longer than this is one whose /// last lines this boot gives up on rather than the stop it was asked for. /// Past it the stop goes ahead, and the lines are on the console. -const FLUSH_BOUND: Duration = Duration::from_secs(5); +const FLUSH_BOUND: Duration = Duration::from_millis(toyos_quiesce::FLUSH_MS); /// The connection init hands `logd` every program's ring on, and the rest of /// what init owns of the log. @@ -1203,12 +1203,13 @@ impl<'a> Init<'a> { let _ = conn.try_send_bytes(power::MSG_REFUSED, &refused.to_u64().to_le_bytes()); } - /// Have every writable file server make its volume durable, each bounded - /// by [`FLUSH_BOUND`]: the kernel's stop waits for no process, so what a - /// server holds and has not written is lost unless it is asked first. - /// After `logd`'s flush, so the log's server writes what logd wrote last. + /// Have every writable file server make its volume durable: the kernel's + /// stop waits for no process, so what a server holds and has not written + /// is lost unless it is asked first. After `logd`'s flush, so the log's + /// server writes what logd wrote last. fn sync_files(&self) { let roles = self.system.programs.iter().flat_map(|p| p.roles.iter()); + let mut syncing = Vec::new(); for role in roles.filter(|r| *r != "boot") { let Some(dir) = toyos_manifest::role_dirs(role).and_then(|dirs| dirs.first()) else { continue }; let name = format!("{CAPABILITY_PREFIX}{dir}"); @@ -1224,11 +1225,14 @@ impl<'a> Init<'a> { }); let _ = done_tx.send(answer); }); - let Ok(syncer) = spawned else { - say!("init: power: no thread to sync {name}; its unsynced writes are lost with the stop"); - continue; - }; - let answered = done_rx.recv_timeout(FLUSH_BOUND); + match spawned { + Ok(syncer) => syncing.push((name, done_rx, syncer)), + Err(_) => say!("init: power: no thread to sync {name}; its unsynced writes are lost with the stop"), + } + } + let until = Instant::now() + Duration::from_millis(toyos_quiesce::SYNC_MS); + for (name, done_rx, syncer) in syncing { + let answered = done_rx.recv_timeout(until.saturating_duration_since(Instant::now())); // Joined once it has answered, so the stop never counts a thread // that was on its way out. if answered.is_ok() { @@ -1239,7 +1243,7 @@ impl<'a> Init<'a> { Ok(Err(e)) => say!("init: power: {name}'s server would not sync ({e:?})"), Err(_) => say!( "init: power: {name}'s server did not sync in {} ms; stopping without it", - FLUSH_BOUND.as_millis() + toyos_quiesce::SYNC_MS ), } } From 94dd327bda3cc548c10c0d6f55113a00ac3cb51e Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 17:36:10 +0200 Subject: [PATCH 22/54] usb: the owed flush is told to /log's writer, which is fsd's partition claim `usb_transport_break`'s OwedFlush arm still expected the kernel to be `/log`'s writer, and the kernel said "writes a process's partition claim made before its disk came back owing a flush" (EXIT=1 wide and alone). With the expectation moved, EXIT=0. Co-Authored-By: Claude Opus 5.5 --- tests/common/usb.rs | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/common/usb.rs b/tests/common/usb.rs index 333a93fd168..5217d13619a 100644 --- a/tests/common/usb.rs +++ b/tests/common/usb.rs @@ -1929,7 +1929,8 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { "made before its disk came back owing a flush may not have survived, and its flush says so"; // The stick's one writer at the break is `/log`: logd's first batch is // the first WRITE(10) that goes out owing a flush. - const LOG_TOLD: &str = "writes the kernel (log) made before its disk came back owing a flush"; + const LOG_TOLD: &str = + "writes a process's partition claim made before its disk came back owing a flush"; const STILL_HELD: &str = "is still held when this call may wait no longer"; const AFTER_A_FLUSH: &str = "breaks next (usb-transport-break-flushed)"; const SILENT: &str = "answers nothing on the operation sent again on it (usb-return-silent)"; From 64c59c9666b55e8d47c534e2d71cb80a4a2a34b4 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 18:31:13 +0200 Subject: [PATCH 23/54] review #536 r3: the NOTEs and REMOVEs, as deletions where they can be - toyos-fat32: `IoError::BudgetExpired` had no producer once the kernel's fat32_adapter went, so it goes with `Error::BudgetExpired`, `Error::RepairPending` (produced only by a budget-refused re-drive), `settle_first`, `RepairNotice`, `repair_episode` and the episode counter, none of which had a production caller. The tests that staged a budget refusal are deleted; refused_writes stages every refusal as the device's. The issue about `RepairNotice::waits_on` closes with its subject. - fsd: a stream's pipe is read into the kept scratch, 64 KiB a turn as before. - iommu: userdev_dma_fault names netd's slot (`slot_of`), as its siblings do. - rootfs: ROOT on no disk this kernel drives is said so, apart from a disk that did not answer. - quiesce: STAGED is init's file call, flush and sync, then PARK; init's FILES_BOUND is toyos_quiesce::FILES_MS. - block: `Holder::Claim` names its partition, and usb_transport_break's told line must name the log partition's GUID. - partclaim's retried flush names its partition, and log_flush_retry judges the refused-once flush against the stop's record, after init's last logd flush, instead of a fixed 2 s window. - Filed: the machine-wide client bound (isolation), fs_turns' floor, a package's own libraries, logd's would-block arm with no producer. The C issue's exit is the owner's order. - REMOVEs deleted. Co-Authored-By: Claude Opus 5.5 --- ...n-repair-is-named-by-which-error-it-was.md | 31 --------- ...a-package-cannot-ship-its-own-libraries.md | 20 ++++++ ...ograms-name-no-file-a-file-server-holds.md | 5 +- ...urns-judges-the-scheduler-with-the-lock.md | 21 ++++++ ...get-refusal-leaves-the-device-untouched.md | 2 +- ...ould-block-flush-policy-has-no-producer.md | 20 ++++++ ...can-take-every-file-servers-client-slot.md | 20 ++++++ kernel/src/arch/x86_64/vtd/mod.rs | 4 -- kernel/src/block.rs | 6 +- kernel/src/device.rs | 4 +- kernel/src/inventory.rs | 2 +- kernel/src/object/ops.rs | 5 +- kernel/src/quiesce.rs | 10 ++- kernel/src/rootfs.rs | 3 + kernel/src/syscall/machine.rs | 2 +- tests/common/iommu.rs | 3 +- tests/common/usb.rs | 19 +++-- tests/common/volumes.rs | 53 +++++++------- .../src/bin/home_backing_revoked.rs | 6 -- tests/toyos.rs | 4 +- toyos-fat32/src/device.rs | 20 +----- toyos-fat32/src/error.rs | 20 ------ toyos-fat32/src/fs.rs | 8 +-- toyos-fat32/src/lib.rs | 2 +- toyos-fat32/src/repair.rs | 62 ++--------------- toyos-fat32/tests/common/mod.rs | 69 +------------------ toyos-fat32/tests/host_write.rs | 54 +-------------- toyos-fat32/tests/hostile.rs | 51 ++------------ toyos-fat32/tests/refused_writes.rs | 64 ++++++----------- toyos-quiesce/src/lib.rs | 5 ++ userland/fsd/src/fat.rs | 3 - userland/fsd/src/main.rs | 17 +++-- userland/init/src/main.rs | 2 +- userland/logd/src/policy.rs | 3 +- 34 files changed, 204 insertions(+), 416 deletions(-) delete mode 100644 issues/filesystem/a-device-refusal-that-queues-the-calls-own-repair-is-named-by-which-error-it-was.md create mode 100644 issues/filesystem/a-package-cannot-ship-its-own-libraries.md create mode 100644 issues/filesystem/fs-turns-judges-the-scheduler-with-the-lock.md create mode 100644 issues/filesystem/logds-would-block-flush-policy-has-no-producer.md create mode 100644 issues/isolation/one-program-can-take-every-file-servers-client-slot.md diff --git a/issues/filesystem/a-device-refusal-that-queues-the-calls-own-repair-is-named-by-which-error-it-was.md b/issues/filesystem/a-device-refusal-that-queues-the-calls-own-repair-is-named-by-which-error-it-was.md deleted file mode 100644 index df1ec1fb567..00000000000 --- a/issues/filesystem/a-device-refusal-that-queues-the-calls-own-repair-is-named-by-which-error-it-was.md +++ /dev/null @@ -1,31 +0,0 @@ ---- -status: open -kind: finding -opened: 2026-09-26 ---- - -# A device refusal that queues the call's own repair is named by which error it was - -`RepairNotice::waits_on` (`toyos-fat32/src/repair.rs`) answers `true` for -`Error::Io` and `Error::RepairPending`, and `false` for -`Error::BudgetExpired` — even when the `BudgetExpired` is the call's own write -and that same write is what queued the repair -(`toyos-fat32/tests/refused_writes.rs:729`, asserted directly: "its own -refusal, not a wait"). Two device refusals that leave the volume in the same -state (a repair queued, to be re-driven before the next mutating call) are -named two different ways depending only on which of the two errors the device -answered. No caller reads the difference today: the kernel adapter that chose -its log line by `waits_on` is gone, and fsd's disks refuse on no clock, so it -never meets a `BudgetExpired` — the next caller that does inherits it. - -## Owner - -`toyos-fat32::RepairNotice::waits_on`, whose contract is what a caller reads -after a refusal; the adapter's log line follows from what it answers. - -## Exit condition - -Either the two device refusals are named the same way when they leave the -volume in the same state, or the difference is stated as intentional at -`waits_on`'s doc comment with the reason a call's own `BudgetExpired` is -never read as "waiting" the way an `Io` is. diff --git a/issues/filesystem/a-package-cannot-ship-its-own-libraries.md b/issues/filesystem/a-package-cannot-ship-its-own-libraries.md new file mode 100644 index 00000000000..00fa381e15d --- /dev/null +++ b/issues/filesystem/a-package-cannot-ship-its-own-libraries.md @@ -0,0 +1,20 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A package cannot ship its own libraries + +A program on a file server is spawned from a memory image its parent read +(`SpawnArgs::image`), and `kernel/src/loader/mod.rs`'s `load_needed_libs` then +finds its `DT_NEEDED` libraries in `/system/lib` alone: the kernel opens no path +a file server holds. A package under `/apps//` whose binary needs a +library it carries beside it fails to spawn, where before the file servers +moved out of the kernel the executable's own directory was searched first. + +## Exit condition + +A spawn from an image loads a `DT_NEEDED` library from the package's own +directory, shown by a guest test that installs a package carrying one and +launches it. diff --git a/issues/filesystem/c-programs-name-no-file-a-file-server-holds.md b/issues/filesystem/c-programs-name-no-file-a-file-server-holds.md index 156e0c2804e..51a849f9a32 100644 --- a/issues/filesystem/c-programs-name-no-file-a-file-server-holds.md +++ b/issues/filesystem/c-programs-name-no-file-a-file-server-holds.md @@ -16,7 +16,4 @@ program — tinycc, doomgeneric, anything built with `toyos-cc` — is refused every path under those six directories with `ENOENT`, and a file it writes where a user's files live cannot be written at all. -**Exit**: libc resolves a path under a directory capability through -`toyos::fs` as std does — open, read, write, seek, stat, readdir, mkdir, -unlink, rename, fsync — with a C test in the corpus that writes a file under -`/home`, reads it back, and lists it. +**Exit**: one file-server client crate in the SDK, the protocol's only implementation, which std uses; libc on top of it, with a C test that writes a file under `/home`, reads it back and lists it; then `/system` over the same protocol and the kernel's file syscalls deleted — the owner accepted the gap until then. diff --git a/issues/filesystem/fs-turns-judges-the-scheduler-with-the-lock.md b/issues/filesystem/fs-turns-judges-the-scheduler-with-the-lock.md new file mode 100644 index 00000000000..52da5d33983 --- /dev/null +++ b/issues/filesystem/fs-turns-judges-the-scheduler-with-the-lock.md @@ -0,0 +1,21 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# `fs_turns` judges the scheduler with the lock + +`tests/toyos-rust-tests/src/bin/fs_turns.rs` asserts that every writer sharing +one directory's connection made at least 16 passes while the first made 32. A +writer the scheduler — or the host, under TCG — keeps off-CPU between its reply +and its next request holds no ticket in that time, and the others pass it with +no lock at fault, so the floor reds on a slow schedule as well as on a lock that +lets its releaser take it straight back. + +## Exit condition + +The verdict separates ticket-order handover from a lock that barges whatever +the schedule: a check on the order requests are served in against the order +they were queued in, or the handover decided in a host-tested crate and +checked there. diff --git a/issues/filesystem/logd-still-says-a-budget-refusal-leaves-the-device-untouched.md b/issues/filesystem/logd-still-says-a-budget-refusal-leaves-the-device-untouched.md index 97beeff3048..ce84b6b5de1 100644 --- a/issues/filesystem/logd-still-says-a-budget-refusal-leaves-the-device-untouched.md +++ b/issues/filesystem/logd-still-says-a-budget-refusal-leaves-the-device-untouched.md @@ -7,7 +7,7 @@ opened: 2026-09-26 # logd still says a budget refusal leaves the device untouched A write refused on `block::OPERATION` may already be on the medium -(`IoError::BudgetExpired`'s doc, `toyos-fat32/src/repair.rs`). +(`toyos-fat32/src/repair.rs`). `userland/logd/src/policy.rs` states the opposite twice: its module doc says a budget that expired means "nothing was issued, the device is untouched", and the comment on `fate`'s `(Step::Flush, WouldBlock)` arm says the same. diff --git a/issues/filesystem/logds-would-block-flush-policy-has-no-producer.md b/issues/filesystem/logds-would-block-flush-policy-has-no-producer.md new file mode 100644 index 00000000000..ae2e9780453 --- /dev/null +++ b/issues/filesystem/logds-would-block-flush-policy-has-no-producer.md @@ -0,0 +1,20 @@ +--- +status: open +kind: finding +opened: 2026-09-27 +--- + +# logd's would-block flush policy has no producer + +`userland/logd/src/policy.rs`'s `fate` retries a flush answered +`io::ErrorKind::WouldBlock` for up to `LOG_WRITE_BUDGET`, and its module doc and +`userland/logd/src/main.rs`'s header argue why. logd's `/log` is served by fsd, +whose `word` (`userland/fsd/src/fat.rs`) answers no refusal `WouldBlock`, and +`toyos-fat32` has no budget refusal left to carry one, so the arm, its tests and +both docs describe an answer nothing sends. + +## Exit condition + +The `(Step::Flush, WouldBlock)` arm, its tests and the prose arguing for it are +deleted, or fsd answers a refusal `WouldBlock` and a guest test shows logd +retrying it. diff --git a/issues/isolation/one-program-can-take-every-file-servers-client-slot.md b/issues/isolation/one-program-can-take-every-file-servers-client-slot.md new file mode 100644 index 00000000000..2e09e915c61 --- /dev/null +++ b/issues/isolation/one-program-can-take-every-file-servers-client-slot.md @@ -0,0 +1,20 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# One program can take every file server's client slot + +`userland/fsd/src/main.rs`'s `MAX_SERVED` bounds the clients one file server +serves at once, machine-wide: the 129th hello is refused `ResourceExhausted`. +Any program holding `fs:/home` can open 128 connections and hold them, and +from then on every other program's first connection to DATA's directories is +refused, by name instead of by a hang. `test_rs_fs_client_bound` does exactly +this for the length of its run. + +## Exit condition + +A program's connections count against a bound of its own, so one program +holding all it may hold leaves another's first connection answered, shown by a +guest test that holds one program at its bound while another connects. diff --git a/kernel/src/arch/x86_64/vtd/mod.rs b/kernel/src/arch/x86_64/vtd/mod.rs index 8762dcd9e2b..5acfba2df57 100644 --- a/kernel/src/arch/x86_64/vtd/mod.rs +++ b/kernel/src/arch/x86_64/vtd/mod.rs @@ -560,10 +560,6 @@ fn enable( ); } -/// Device class the actuators target — the xHCI controller, which this kernel -/// drives by DMA from boot — not a bus/device/function: QEMU's slot -/// choice is not this kernel's business, and the harness reads the same -/// class independently out of `pci::enumerate`. /// The requester id `iommu-context-absent` or `iommu-empty-domain` staged, so /// its driver's move to a domain of its own leaves the staging in place; /// `u32::MAX`, which no requester id is, when neither is armed. diff --git a/kernel/src/block.rs b/kernel/src/block.rs index 978033307b1..ac8e5db1c4f 100644 --- a/kernel/src/block.rs +++ b/kernel/src/block.rs @@ -410,15 +410,15 @@ impl BlockDevice for Locked<'_> { pub enum Holder { /// Something in this kernel — a mount, a probe — named for the log. Kernel(&'static str), - /// A process, through a partition claim. - Claim, + /// A process, through its claim of the partition this unique GUID names. + Claim(toyos_gpt::Guid), } impl core::fmt::Display for Holder { fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { match self { Self::Kernel(what) => write!(f, "the kernel ({what})"), - Self::Claim => f.write_str("a process's partition claim"), + Self::Claim(guid) => write!(f, "a process's claim of partition {guid}"), } } } diff --git a/kernel/src/device.rs b/kernel/src/device.rs index 8fe19f0fea8..17c7f60d5a0 100644 --- a/kernel/src/device.rs +++ b/kernel/src/device.rs @@ -218,13 +218,13 @@ fn partition_view(found: &crate::gpt::Claimable) -> Result Ok(view), Err(ViewRefused::Held(Holder::Kernel(what))) => { log!("partclaim: {guid} is held by the kernel ({what}) and cannot be claimed"); Err(ClaimError::KernelDriven) } - Err(ViewRefused::Held(Holder::Claim)) => Err(ClaimError::Owned), + Err(ViewRefused::Held(Holder::Claim(_))) => Err(ClaimError::Owned), Err(ViewRefused::OffDevice) => { log!( "partclaim: {guid} is at {first_block}+{blocks} blocks, off device {}", diff --git a/kernel/src/inventory.rs b/kernel/src/inventory.rs index 53aff2881fe..7fde6c0880e 100644 --- a/kernel/src/inventory.rs +++ b/kernel/src/inventory.rs @@ -41,7 +41,7 @@ pub fn collect() -> Vec { state: match holder { None => PartState::Free, Some(crate::block::Holder::Kernel(_)) => PartState::Kernel, - Some(crate::block::Holder::Claim) => PartState::Claimed, + Some(crate::block::Holder::Claim(_)) => PartState::Claimed, }, })); } diff --git a/kernel/src/object/ops.rs b/kernel/src/object/ops.rs index 66e42a1ef8a..e48bda71f1f 100644 --- a/kernel/src/object/ops.rs +++ b/kernel/src/object/ops.rs @@ -771,11 +771,12 @@ fn partition_fsync(claim: &DeviceClaim) -> u64 { Some(view) => view.flush().map_err(block_word), None => Err(SyscallError::Gone), }); - if let Answered::Answer { answer: Ok(()), attempts, took } = &run { + if let (Answered::Answer { answer: Ok(()), attempts, took }, Some((_, guid))) = (&run, claim.partition_on()) { if *attempts > 1 { crate::log!( - "partclaim: a flush durable on attempt {attempts} after {took} — a refused \ + "partclaim: a flush of {} durable on attempt {attempts} after {took} — a refused \ attempt was asked again on a fresh budget", + toyos_gpt::Guid(guid), ); } } diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 1fab1917539..06e76564df0 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -273,10 +273,14 @@ pub mod last { } } - /// How long either side waits for the other before the boot dies by name. - // The thread is held before init prepares the stop, so its wait covers that too. + /// How long either side waits for the other before the boot dies by name: + /// the thread is held before init takes the stop request, so its wait is + /// init's file call, flush and sync, then the stop's own sweeps. const STAGED: Budget = Budget::of( - Duration::from_millis(toyos_quiesce::FLUSH_MS + toyos_quiesce::SYNC_MS + 10_000), + Duration::from_nanos( + (toyos_quiesce::FILES_MS + toyos_quiesce::FLUSH_MS + toyos_quiesce::SYNC_MS) * 1_000_000 + + super::PARK.nanos(), + ), "the boot panics naming the side of the staging that never arrived", ); diff --git a/kernel/src/rootfs.rs b/kernel/src/rootfs.rs index 1c82fcee304..971ee5c511a 100644 --- a/kernel/src/rootfs.rs +++ b/kernel/src/rootfs.rs @@ -185,6 +185,9 @@ pub fn hold_source() { .map(|view| (found, view)) .map_err(|()| "it is no span a view can hold") } + // Every disk this kernel drives answered and lacks it: a disk a process + // drives, NVMe's under blockd, is never asked. + Ok(None) if sought.silent.is_empty() => Err("it is on no disk this kernel drives"), Ok(None) => Err("it is on no disk that answered"), Err(Unnamed::Ambiguous) => Err("it is carried twice"), Err(Unnamed::Unusable) => Err("its table refuses it"), diff --git a/kernel/src/syscall/machine.rs b/kernel/src/syscall/machine.rs index 6fb0bf6ef22..03228906b4c 100644 --- a/kernel/src/syscall/machine.rs +++ b/kernel/src/syscall/machine.rs @@ -90,7 +90,7 @@ fn quiesce(last: &str) -> Result<(), SyscallError> { // process that issues a `write` after `sync_all` returns has dirty pages // nothing will flush. Every userland thread stops here, the log's writer // with the rest: `/system/bin/init` had it flush before it asked for this - // stop, and what it wrote since is in the page cache the sync below takes. + // stop. #[cfg(feature = "boot-actuators")] crate::quiesce::last::await_the_held_thread(); let stopped = crate::quiesce::stop(); diff --git a/tests/common/iommu.rs b/tests/common/iommu.rs index 269665ac7b7..1e0db6d861e 100644 --- a/tests/common/iommu.rs +++ b/tests/common/iommu.rs @@ -2085,7 +2085,8 @@ pub fn userdev_dma_fault( // says so: `owner=kernel` here would be a machine that halted, or was about // to. let handled = log.must_say(FAULT)?; - if !handled.contains("owner=slot") { + let slot = slot_of(log.text(), "[8086:10d3]")?; + if !handled.contains(&format!("owner=slot{slot} ")) { return Err(format!( "the unit's fault was recorded against {handled:?}, and the function that faulted \ is one a process drives. A fault the kernel takes as its own is one it halts for" diff --git a/tests/common/usb.rs b/tests/common/usb.rs index 5217d13619a..4b5c440185c 100644 --- a/tests/common/usb.rs +++ b/tests/common/usb.rs @@ -1927,10 +1927,6 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { flush of each writer whose writes they were fails"; const FLUSH_LOST: &str = "made before its disk came back owing a flush may not have survived, and its flush says so"; - // The stick's one writer at the break is `/log`: logd's first batch is - // the first WRITE(10) that goes out owing a flush. - const LOG_TOLD: &str = - "writes a process's partition claim made before its disk came back owing a flush"; const STILL_HELD: &str = "is still held when this call may wait no longer"; const AFTER_A_FLUSH: &str = "breaks next (usb-transport-break-flushed)"; const SILENT: &str = "answers nothing on the operation sent again on it (usb-return-silent)"; @@ -1956,6 +1952,16 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { let image = test_dir().join(name); std::fs::write(&image, qemu::build_boot_image(&case, &[], &[], params)) .map_err(|e| format!("write {}: {e}", image.display()))?; + // The stick's one writer at the break is `/log`: logd's first batch is + // the first WRITE(10) that goes out owing a flush. + let log_told = { + let mut file = std::fs::File::open(&image).map_err(|e| format!("{}: {e}", image.display()))?; + let guid = toyos_build::image::unique_guid_of(&mut file, toyos_gpt::Guid::MICROSOFT_BASIC)?; + format!( + "writes a process's claim of partition {} made before its disk came back owing a flush", + toyos_gpt::Guid(guid) + ) + }; let mut qemu = QemuInstance::boot_with_options( &case, &[], @@ -2143,7 +2149,7 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { let mut want = left.to_vec(); want.extend(back.iter().cloned()); want.push(OWED.to_string()); - want.push(LOG_TOLD.to_string()); + want.push(log_told.clone()); in_order(&want)?; for never in [" did not come back within ", "disk 1 ready", " is not disk 0 come back"] { if let Some(line) = log.lines().find(|l| l.contains(never)) { @@ -2151,8 +2157,7 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { } } eprintln!( - " [usb] a stick that left owing a flush was taken back as disk 0, and the flush of \ - /log, whose write it lost, failed by name" + " [usb] a stick that left owing a flush was taken back as disk 0" ); } Moved::SilentReturn => { diff --git a/tests/common/volumes.rs b/tests/common/volumes.rs index 527f56919ef..23a439c1d77 100644 --- a/tests/common/volumes.rs +++ b/tests/common/volumes.rs @@ -321,7 +321,7 @@ pub fn esp_filesystem( // `flush_meta`, the FSInfo write, the device cache flush — reaches // userland as one `SyscallError::Io`, which `std` flattens to // `Kind(Other)`; *which* layer refused is in a `log!` line and nowhere - // else (`fat32_adapter::refused`, `usb_storage`'s three trait methods, + // else (`usb_storage`'s three trait methods, // `xhci::wait::msc`'s `log_refusal` and its budget line). Reporting // `stdout` alone left `fsync the blob: Kind(Other)` as the whole of the // evidence for a real 2026-08-21 sighting, and the next author with no @@ -1984,11 +1984,7 @@ pub fn log_partition_identity( /// one volume: /// /// 1. **Retry keeps the volume, and a refused attempt discarded nothing.** -/// `fsync-budget-spent` runs each file's first `SYS_FSYNC` attempt under an -/// already-spent operation — the state a loaded dev host reproduced 1 in -/// 73 full 12-wide suites (2026-08-22, this test's own blob fsync on -/// `/log`) — so each file's first flush is refused once at the shipped site -/// and retried on a fresh budget. The guest's own +/// The guest's own /// fsync must succeed, logd must never give its volume up, and the blob is /// then read off the *image* by the host: the safety invariant is that the /// refused attempt left every un-flushed page dirty, so the retry delivered @@ -2070,14 +2066,20 @@ pub fn log_flush_retry( volume_lines(&log) )); } - // `/log` is flushed through fsd's partition claim, whose retry says so. + // `/log` is flushed through fsd's claim of the log partition, whose retry + // says so by that partition's name. + let log_guid = { + let mut file = + std::fs::File::open(&image_path).map_err(|e| format!("{}: {e}", image_path.display()))?; + toyos_gpt::Guid(toyos_build::image::unique_guid_of(&mut file, toyos_gpt::Guid::MICROSOFT_BASIC)?) + }; + let log_retry = format!("partclaim: a flush of {log_guid} durable on attempt"); let retried = log .lines() - .find(|l| l.contains("partclaim: a flush durable on attempt")) + .find(|l| l.contains(&log_retry)) .ok_or_else(|| { format!( - "no `partclaim: a flush durable on attempt` line, so the operation-level retry \ - never ran:\n{}", + "no `{log_retry}` line, so the operation-level retry never ran:\n{}", volume_lines(&log) ) })? @@ -2089,19 +2091,6 @@ pub fn log_flush_retry( volume_lines(&log) )); } - // An absence has no event to wait on, so a fixed window judges it. - let after = qemu.drain_serial(Duration::from_secs(2)); - let again: Vec<&str> = - after.lines().filter(|l| l.contains("partclaim: a flush durable on attempt")).collect(); - if !again.is_empty() { - return Err(format!( - "{} flush(es) retried in the 2 s after the guest's, first {:?}: every flush is being \ - refused, and each refusal's records are the next flush\n{after}", - again.len(), - again[0] - )); - } - // The device's view, after a clean shutdown: what the retried flushes left. writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); qemu.flush_stdin(); @@ -2112,6 +2101,24 @@ pub fn log_flush_retry( return Err(format!("{bad:?} on the way down\n{tail}")); } } + // init has logd flush before it asks for the stop, so a flush of the log + // partition follows the guest's; every record before the stop's own is on + // the wire ahead of it, so a retry of that flush would be here too. + let whole = format!("{log}\n{tail}"); + let stop = whole + .lines() + .position(|l| toyos_quiesce::Record::parse(l).is_some()) + .ok_or_else(|| format!("the shutdown wrote no stop record:\n{tail}"))?; + let retries: Vec<&str> = whole.lines().take(stop).filter(|l| l.contains(&log_retry)).collect(); + if retries.len() != 1 { + return Err(format!( + "{} flush(es) of the log partition retried before the stop, and the refused one is \ + the first alone: every flush is being refused, and each refusal's records are the \ + next flush\n{}", + retries.len(), + retries.join("\n") + )); + } let after = std::fs::read(&image_path).map_err(|e| format!("read the image back: {e}"))?; let (log_start, log_len) = log_extent(&after, &image_path)?; let volume = &after[log_start..log_start + log_len]; diff --git a/tests/toyos-rust-tests/src/bin/home_backing_revoked.rs b/tests/toyos-rust-tests/src/bin/home_backing_revoked.rs index 62315bf3c54..73041c13e61 100644 --- a/tests/toyos-rust-tests/src/bin/home_backing_revoked.rs +++ b/tests/toyos-rust-tests/src/bin/home_backing_revoked.rs @@ -1,11 +1,5 @@ //! A file backing must not outlive the file it reads. //! -//! `/home` hands every open file an `NvmeBacking` holding the blocks its data -//! lives in. Unlink the file and those blocks go back to bcachefs's allocator; -//! the next file takes them. A backing that still names them reads that file's -//! contents — an information disclosure through `open`, `rm` and `cp`, with -//! nothing crafted about it and no privilege needed. -//! //! Staged here rather than reasoned about: the victim's blocks are freed and //! then deliberately handed to a file whose bytes are nothing like the //! victim's, and the still-open descriptor is read afterwards. diff --git a/tests/toyos.rs b/tests/toyos.rs index 11c43a259c8..ac913dbc55c 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -459,12 +459,12 @@ const RUST_SKIP: &[&str] = &[ // succeeds and its two must-refuse assertions red for an honest reason. // `fsync_failed_commit` boots it with the arm. "fsync_flush_failed", - // Needs `so-cache-tiny` and the NVMe `/home`. `so_cache_refusals` gives both. + // Needs `so-cache-tiny`. `so_cache_refusals` gives it. "so_cache_policy", // Needs a boot whose file servers are armed to end under its write, and // ends DATA's for the rest of the boot. `fsd_restart` runs it. "fs_restart", - // Needs the NVMe `/home` and a boot of its own for the readback it is judged against; `home_overwrite_reads_back` runs it. + // Needs a boot of its own for the readback it is judged against; `home_overwrite_reads_back` runs it. "home_overwrite_zero", // Needs a boot where the DATA volume is ours and absent; on the shared // boot `/apps` and `/home` are ordinarily mounted, so every refusal it diff --git a/toyos-fat32/src/device.rs b/toyos-fat32/src/device.rs index 0b386382a06..324adeccf58 100644 --- a/toyos-fat32/src/device.rs +++ b/toyos-fat32/src/device.rs @@ -32,30 +32,12 @@ pub trait BlockAccess { fn flush(&mut self) -> Result<(), IoError>; } -/// The volume did not do it, and which of two kinds of "did not" it was. +/// The volume did not do it. /// -/// **Two variants, because one of them is not about the device.** Everything a -/// device can say about itself — a transfer it refused, a controller that gave -/// up, a request past the end — is one fact this crate can act on none of; that -/// is [`IoError::Device`], and it carries no detail for the reason the single -/// unit struct this replaces carried none. But an implementor of -/// [`BlockAccess`] may also have a *bound of its own* on how long one call may -/// take, and reaching it is a statement about the caller's clock rather than -/// about the volume, and asking again later is the honest response. Flattening -/// the two costs the caller the only decision it could make. -/// -/// Neither variant says whether a refused write reached the medium: an -/// implementor may have issued it before its bound or the device gave up. /// This crate treats every refused write as of unknown outcome — see /// `crate::repair`. -/// -/// The kernel's implementor is `kernel/src/fat32_adapter.rs` over -/// `block::BlockDevice`, whose `block::OPERATION` budget is that bound. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum IoError { /// The device refused, failed, or would not answer. Device, - /// The implementor's own bound on the operation expired. A write may - /// have landed before it did; the caller may ask again. - BudgetExpired, } diff --git a/toyos-fat32/src/error.rs b/toyos-fat32/src/error.rs index 2d17ccc3631..ef36e24172c 100644 --- a/toyos-fat32/src/error.rs +++ b/toyos-fat32/src/error.rs @@ -45,28 +45,12 @@ pub enum Error { /// listing, because a caller checking that a name is absent gets a /// confident wrong answer. LimitExceeded, - /// The device's implementor refused on *its own* bound - /// ([`IoError::BudgetExpired`]). - /// - /// **One of the two variants here that are not a fact about the volume**, - /// with [`Error::RepairPending`], and so one a caller may honestly answer - /// by asking again: the refused write may have landed, but the call has - /// undone or carried through what it did, or queued that for the next call - /// to finish first. Every other variant describes something true of the - /// medium or of the request, which will be just as true next time. - BudgetExpired, - /// The call was not started: an earlier call's refused writes are still - /// queued ([`crate::Fat32::pending_repair`]), their re-drive was refused - /// again on the device's own bound, and nothing changes this volume until - /// they land. Asking again re-drives them first. - RepairPending, } impl From for Error { fn from(e: IoError) -> Self { match e { IoError::Device => Error::Io, - IoError::BudgetExpired => Error::BudgetExpired, } } } @@ -88,10 +72,6 @@ impl Error { Error::NoSpace => "no space left on volume", Error::TooLarge => "file too large for FAT32", Error::LimitExceeded => "limit exceeded", - Error::BudgetExpired => "the device would not answer in the caller's own budget", - Error::RepairPending => { - "an earlier call's refused writes are still queued and the device refused them again" - } } } } diff --git a/toyos-fat32/src/fs.rs b/toyos-fat32/src/fs.rs index 4b9a3e6bd4f..2e709a2adc0 100644 --- a/toyos-fat32/src/fs.rs +++ b/toyos-fat32/src/fs.rs @@ -44,9 +44,6 @@ pub struct Fat32 { /// mutating call starts until it lands. Never past /// [`MAX_REPAIR_STEPS`], and allocated at that capacity once. pub(crate) repair: Vec, - /// Calls that left a repair queued, counted from mount: which one the - /// queue belongs to, see [`Fat32::repair_episode`]. - pub(crate) episodes: u64, /// A mutating call is running; a second one inside it panics. pub(crate) in_call: bool, /// The running call has passed its commit, so what the repair holds is @@ -253,7 +250,6 @@ impl Fat32 { scratch, scratch_at: None, repair: Vec::with_capacity(MAX_REPAIR_STEPS), - episodes: 0, in_call: false, committed: false, }) @@ -1147,7 +1143,7 @@ impl Fat32 { /// mounted with no count has it taken from the FAT rather than written as /// unknown. FSInfo is written before the flush so the flush covers it. pub fn sync(&mut self) -> Result<(), Error> { - self.settle_first()?; + self.settle()?; if self.fsinfo.dirty { if self.fsinfo.free_count.is_none() { self.fsinfo.free_count = Some(self.count_free()?); @@ -1170,7 +1166,7 @@ impl Fat32 { let free = match self.fsinfo.free_count { Some(n) => n, None => { - self.settle_first()?; + self.settle()?; let n = self.count_free()?; self.fsinfo.free_count = Some(n); self.fsinfo.dirty = true; diff --git a/toyos-fat32/src/lib.rs b/toyos-fat32/src/lib.rs index c76975e92f9..9eae7629358 100644 --- a/toyos-fat32/src/lib.rs +++ b/toyos-fat32/src/lib.rs @@ -87,5 +87,5 @@ pub use dir::MAX_DIR_ENTRIES; pub use error::Error; pub use fs::{DirEntry, Extent, Fat32, File, Metadata, ReplaceFailed, Replaced}; pub use name::MAX_LFN_CHARS; -pub use repair::{RepairNotice, MAX_REPAIR_STEPS}; +pub use repair::MAX_REPAIR_STEPS; pub use time::FatTime; diff --git a/toyos-fat32/src/repair.rs b/toyos-fat32/src/repair.rs index 11aebabbdd9..5001b9c291e 100644 --- a/toyos-fat32/src/repair.rs +++ b/toyos-fat32/src/repair.rs @@ -1,8 +1,7 @@ //! What a mutating call does when one of its device reads or writes is //! refused, and a refused write's outcome is unknown. //! -//! A refused write may already be on the medium: a block layer can issue a -//! write, lose the answer, and report its own budget expired. So a refusal says +//! A refused write may already be on the medium. So a refusal says //! nothing about which of the two states an entry is in, and nothing here reads //! to find out — a cache under [`BlockAccess`] may still serve the bytes the //! refusal did not update. What is known is the state the call wanted and the @@ -14,8 +13,7 @@ //! call that returns `Err` re-drives them; one the device refuses stays, and //! [`Fat32::atomic`] re-drives it before the next mutating call touches //! anything, so nothing is allocated, freed or linked over an entry whose -//! value on the medium is unknown. A call refused because that re-drive was -//! refused again answers [`Error::RepairPending`]. +//! value on the medium is unknown. //! //! Before a call's commit the repair is its rollback, so a refused call is a //! call that did not happen. The one commit this crate has is a free: once @@ -84,31 +82,6 @@ pub(crate) enum Repair { Entry { offset: u64, raw: RawEntry }, } -/// The [`Fat32::repair_episode`] a caller last announced, so a log names each -/// pending repair once rather than at every call that meets it. -#[derive(Debug, Default)] -pub struct RepairNotice(Option); - -impl RepairNotice { - /// Whether a call answering `e` while `episode` ([`Fat32::repair_episode`]) - /// is queued leaves the volume waiting on that repair: `RepairPending` is a - /// budget refusing its re-drive, and `Io` a device failure refusing either - /// the re-drive or the call's own write, which queues it the same way. - pub fn waits_on(e: Error, episode: Option) -> bool { - episode.is_some() && matches!(e, Error::RepairPending | Error::Io) - } - - /// Whether `episode` is a pending repair not yet announced; from here it - /// is. `None`, nothing pending, is never one. - pub fn first_sight(&mut self, episode: Option) -> bool { - if episode.is_none() || episode == self.0 { - return false; - } - self.0 = episode; - true - } -} - impl Fat32 { /// Run one mutating call so that an `Err` leaves the volume where the /// repair takes it. @@ -121,7 +94,7 @@ impl Fat32 { op: impl FnOnce(&mut Self) -> Result, ) -> Result { assert!(!self.in_call, "toyos-fat32: a mutating call nested inside another"); - self.settle_first()?; + self.settle()?; self.in_call = true; let result = op(self); self.in_call = false; @@ -133,11 +106,11 @@ impl Fat32 { // and report. Ok(v) if committed => { let mut driven = self.settle(); - if matches!(driven, Err(Error::BudgetExpired | Error::Io)) { + if driven == Err(Error::Io) { driven = self.settle(); } match driven { - Ok(()) | Err(Error::BudgetExpired | Error::Io) => Ok(v), + Ok(()) | Err(Error::Io) => Ok(v), Err(e) => Err(e), } } @@ -150,10 +123,6 @@ impl Fat32 { // the answer rather than the error that started the rollback. Err(e) => self.settle().map_or_else(Err, |()| Err(e)), }; - // The queue was empty when `settle_first` let this call start. - if !self.repair.is_empty() { - self.episodes += 1; - } answer } @@ -179,27 +148,6 @@ impl Fat32 { self.repair.len() } - /// Which refused call's repair is queued, numbered from mount, or `None` - /// when nothing is. Two answers that differ are two refused calls, whatever - /// landed between them. - pub fn repair_episode(&self) -> Option { - (!self.repair.is_empty()).then_some(self.episodes) - } - - /// [`Self::settle`] before a call of its own, naming a refusal as the - /// earlier call's repair rather than this call's write. - /// - /// Every other error it answers is the earlier call's too, answered by a - /// call that did not start: a device failure, or a - /// [`Error::CorruptChain`] met walking a queued free, whose chain past the - /// corrupt link leaks and is recorded nowhere else. - pub(crate) fn settle_first(&mut self) -> Result<(), Error> { - self.settle().map_err(|e| match e { - Error::BudgetExpired => Error::RepairPending, - e => e, - }) - } - /// Re-drive every queued repair, the most recent write's first. pub(crate) fn settle(&mut self) -> Result<(), Error> { while let Some(&step) = self.repair.last() { diff --git a/toyos-fat32/tests/common/mod.rs b/toyos-fat32/tests/common/mod.rs index dfcaa569c0d..6a0a03bb04c 100644 --- a/toyos-fat32/tests/common/mod.rs +++ b/toyos-fat32/tests/common/mod.rs @@ -148,13 +148,6 @@ pub struct SparseDevice { sectors: HashMap, capacity: u64, pub fail_reads_past: Option, - /// Which refusal [`fail_reads_past`](Self::fail_reads_past) and - /// [`flush_refuses`](Self::flush_refuses) answer with. - /// - /// `IoError` is two variants and only one of them is a fact about the - /// device, so a fake that could only ever say `Device` could not exercise - /// the other half of the mapping at all. - pub refusal: IoError, /// Whether [`BlockAccess::flush`] refuses, which is the one call /// `Fat32::sync` makes into the device after the FSInfo write. pub flush_refuses: bool, @@ -176,7 +169,6 @@ impl SparseDevice { sectors: HashMap::new(), capacity, fail_reads_past: None, - refusal: IoError::Device, flush_refuses: false, volume_bytes: None, out_of_volume: 0, @@ -242,7 +234,7 @@ impl BlockAccess for SparseDevice { } if let Some(limit) = self.fail_reads_past { if end > limit { - return Err(self.refusal); + return Err(IoError::Device); } } self.note_range(end); @@ -262,71 +254,14 @@ impl BlockAccess for SparseDevice { fn flush(&mut self) -> Result<(), IoError> { if self.flush_refuses { - return Err(self.refusal); + return Err(IoError::Device); } Ok(()) } } -/// A device that refuses the next write touching a chosen byte range, once, and -/// passes everything else straight through to `inner`. -/// -/// It stages the one failure QEMU will not produce and no host option can inject -/// (`kernel/src/block.rs`'s `OPERATION` budget note): a mirror write taken and -/// the active-FAT write of the *same* `set_fat_entry` refused on the device's -/// own budget after the first is durable. `set_fat_entry` writes the active FAT -/// last for exactly this, so armed on the active FAT's region this refuses that -/// second write and leaves the split a re-drive must heal. -pub struct RefuseOnceInRange { - inner: D, - /// `Some(lo, hi)` while armed. A write overlapping `[lo, hi)` is refused and - /// disarms this, so only one write is lost — the way one expired budget - /// refuses one operation and the next runs on a fresh one. - armed: Option<(u64, u64)>, - refusal: IoError, -} - -impl RefuseOnceInRange { - /// Disarmed: wrap now, arm with [`Self::arm`] once the setup writes that - /// must succeed (the create, any directory growth) are done. - pub fn new(inner: D, refusal: IoError) -> RefuseOnceInRange { - RefuseOnceInRange { inner, armed: None, refusal } - } - - pub fn arm(&mut self, range: (u64, u64)) { - self.armed = Some(range); - } -} - -impl BlockAccess for RefuseOnceInRange { - fn capacity(&self) -> u64 { - self.inner.capacity() - } - - fn read_at(&mut self, offset: u64, buf: &mut [u8]) -> Result<(), IoError> { - self.inner.read_at(offset, buf) - } - - fn write_at(&mut self, offset: u64, buf: &[u8]) -> Result<(), IoError> { - if let Some((lo, hi)) = self.armed { - let end = offset + buf.len() as u64; - if offset < hi && end > lo { - self.armed = None; - return Err(self.refusal); - } - } - self.inner.write_at(offset, buf) - } - - fn flush(&mut self) -> Result<(), IoError> { - self.inner.flush() - } -} - -/// `kernel/src/fat32_adapter.rs`'s `BLOCK`. pub const ADAPTER_BLOCK: usize = 4096; -/// `kernel/src/fat32_adapter.rs`'s `RESIDENT_BLOCKS`. pub const ADAPTER_RESIDENT_BLOCKS: usize = 8; /// Where a refused block write got to. diff --git a/toyos-fat32/tests/host_write.rs b/toyos-fat32/tests/host_write.rs index d48577a43d2..ddbb61c2a71 100644 --- a/toyos-fat32/tests/host_write.rs +++ b/toyos-fat32/tests/host_write.rs @@ -13,8 +13,8 @@ use std::collections::BTreeMap; use std::fs; use std::path::Path; -use common::{pattern, read_all, sorted_walk, walk_expectation, write_new, Image, RefuseOnceInRange}; -use toyos_fat32::{Error, Fat32, FatTime, IoError}; +use common::{pattern, read_all, sorted_walk, walk_expectation, write_new, Image}; +use toyos_fat32::{Error, Fat32, FatTime}; /// 2024-06-01 12:34:56, so every entry this crate stamps is checkable rather /// than whatever the clock said. @@ -223,56 +223,6 @@ fn every_fat_copy_stays_in_step() { image.fsck(); } -/// A device budget that expires partway through a cluster allocation's -/// two-copy FAT write must not leave the copies split at rest: the write's own -/// repair and the retry a budget refusal invites leave both copies agreeing. -/// -/// `RefuseOnceInRange` is armed on the **mirror** (FAT 1) region and refuses the -/// one write landing there with the retryable [`IoError::BudgetExpired`] — the -/// mid-mirror refusal a starved host produces and QEMU will not — and the -/// re-drive is what `kernel::writeback`'s drain does on a `WouldBlock` flush. -/// Every write of every structural call, refused either way, is -/// `tests/refused_writes.rs`. -#[test] -fn a_refused_mirror_write_heals_on_the_retry() { - let image = image("mirror-refusal"); - let geom = *Fat32::mount(image.device()).expect("probe mount").geometry(); - assert_eq!(geom.num_fats, 2, "the corpus has one FAT, so a mirror split is unreachable"); - assert!(geom.active_fat.is_none(), "the corpus disabled mirroring, so FAT 0 is the active copy"); - // The mirror (FAT 1) region: refusing a write here catches the active FAT - // having *already* taken the update, which is the split only the broken - // ordering can produce. - let mirror_lo = geom.fat_base_offset(1); - let mirror_hi = geom.fat_base_offset(2); - - let data = pattern(200_000, 91); // multi-cluster, so the first write allocates - { - let mut fs = Fat32::mount(RefuseOnceInRange::new(image.device(), IoError::BudgetExpired)) - .expect("mount"); - let mut f = fs.create("log.bin", stamp()).expect("create"); - // Arm only now: the create (and any directory growth it needed) must - // reach the device, so the one refusal falls on the file's own - // allocation and not on its entry. - fs.device().arm((mirror_lo, mirror_hi)); - - let refused = fs.write(&mut f, 0, &data).expect_err("the mirror write was armed to refuse"); - assert_eq!( - refused, - Error::BudgetExpired, - "a budget refusal must stay the retryable kind, or the drain would give up instead of retrying", - ); - - // The re-drive, on a fresh budget (the fault is spent) — the same handle, - // the same bytes, as the write-back drain re-flushes a still-dirty file. - fs.write(&mut f, 0, &data).expect("the retry after a spent budget"); - fs.flush_meta(&mut f, stamp()).expect("flush"); - fs.sync().expect("sync"); - image.assert_fats_agree(fs.geometry()); - assert_eq!(read_all(&mut fs, "log.bin"), data, "the file must read back what was written"); - } - image.fsck(); -} - #[test] fn appending_extends_the_chain() { let image = image("append"); diff --git a/toyos-fat32/tests/hostile.rs b/toyos-fat32/tests/hostile.rs index 22075fcc044..ae681255239 100644 --- a/toyos-fat32/tests/hostile.rs +++ b/toyos-fat32/tests/hostile.rs @@ -29,7 +29,7 @@ mod common; use std::sync::OnceLock; use common::{pattern, Image, SparseDevice}; -use toyos_fat32::{BlockAccess, Error, Fat32, FatTime, IoError, MAX_DIR_ENTRIES}; +use toyos_fat32::{BlockAccess, Error, Fat32, FatTime, MAX_DIR_ENTRIES}; /// Enough of the volume to hold the reserved sectors, both FATs, and the first /// thousand data clusters. Everything past it reads as zeroes, which is what @@ -263,58 +263,15 @@ fn a_device_that_fails_mid_read_reports_it() { assert_eq!(fs.walk("", 1024).unwrap_err(), Error::Io); } -/// **A budget that expired is not a device that failed, and this crate must not -/// flatten the two.** -/// -/// `IoError` grew a second variant on 2026-08-22 for one reason: the kernel's -/// implementor bounds an operation with `block::OPERATION`, and reaching that -/// bound is a statement about the caller's clock. Flattening it into -/// `Error::Io` is what made `/system/bin/logd` end a boot's log for a stick that was -/// answering — 1 red in 73 full 12-wide suites (2026-08-22), one `SYS_FSYNC` -/// held for 2.1 s while the guest's peers booted in 1.4 s. -/// The two `assert_ne!`s are the point of the test: `Error::Io` is exactly the -/// answer the collapsed version gives. -#[test] -fn a_budget_that_expired_is_not_a_device_that_failed() { - let (prefix, cap) = corpus(); - let mut dev = SparseDevice::from_prefix(prefix, *cap); - dev.fail_reads_past = Some(256); - dev.refusal = IoError::BudgetExpired; - let refused = common::mount_err(dev); - assert_eq!(refused, Error::BudgetExpired); - assert_ne!(refused, Error::Io); - - let l = layout(); - let mut dev = pristine(); - dev.fail_reads_past = Some(l.first_data * l.bps); - dev.refusal = IoError::BudgetExpired; - let mut fs = Fat32::mount(dev).expect("mount"); - assert_eq!(fs.walk("", 1024).unwrap_err(), Error::BudgetExpired); -} - -/// The flush is the call `/system/bin/logd`'s durability claim rests on, so it is the -/// one that has to keep the distinction all the way up. -/// -/// Both arms against the same device, so what separates them is the refusal -/// and nothing else. +/// The flush is the call `/system/bin/logd`'s durability claim rests on: a +/// device that refuses it is `Io`, and one that flushes is `Ok`. #[test] -fn a_flush_says_which_of_the_two_refusals_it_was() { +fn a_refused_flush_is_io() { let mut dev = pristine(); dev.flush_refuses = true; - dev.refusal = IoError::Device; let mut fs = Fat32::mount(dev).expect("mount"); assert_eq!(fs.sync().unwrap_err(), Error::Io); - let mut dev = pristine(); - dev.flush_refuses = true; - dev.refusal = IoError::BudgetExpired; - let mut fs = Fat32::mount(dev).expect("mount"); - let refused = fs.sync().unwrap_err(); - assert_eq!(refused, Error::BudgetExpired); - assert_ne!(refused, Error::Io); - - // And a device that flushes is still `Ok`, so the two arms above are about - // the refusal rather than about `sync` never succeeding on this fake. let mut fs = Fat32::mount(pristine()).expect("mount"); assert_eq!(fs.sync(), Ok(())); } diff --git a/toyos-fat32/tests/refused_writes.rs b/toyos-fat32/tests/refused_writes.rs index a0ea65560ec..4e097be8797 100644 --- a/toyos-fat32/tests/refused_writes.rs +++ b/toyos-fat32/tests/refused_writes.rs @@ -1,8 +1,7 @@ //! A device write whose outcome is unknown, at every write of every call that //! changes the volume's structure. //! -//! A refused write may already be on the medium: the kernel's USB path can -//! issue a write, lose the answer, and report its own budget expired. So each +//! A refused write may already be on the medium. So each //! write is refused twice over — once before it reaches the bytes, once after — //! and then in bursts that also refuse the call's own repair and the next //! attempts, as a stopping machine refuses every retry for a while. Reads and @@ -26,7 +25,7 @@ mod spec_volume; use spec_volume::{fat_offset, Volume, BYTES_PER_SECTOR, CLUSTERS, FAT_SECTORS, NUM_FATS}; use common::{AdapterCache, Landing}; use toyos_fat32::{ - BlockAccess, Error, Fat32, FatTime, File, IoError, RepairNotice, MAX_LFN_CHARS, MAX_REPAIR_STEPS, + BlockAccess, Error, Fat32, FatTime, File, IoError, MAX_LFN_CHARS, MAX_REPAIR_STEPS, }; use toyos_fat32_check::Complaint; @@ -36,24 +35,15 @@ enum Outcome { Refused, /// The write reached the bytes and was answered as refused anyway. Landed, - /// The device failed the write, which never reached the bytes. - Failed, } impl Outcome { fn landing(self) -> Landing { match self { - Outcome::Refused | Outcome::Failed => Landing::NotReached, + Outcome::Refused => Landing::NotReached, Outcome::Landed => Landing::Reached, } } - - fn error(self) -> IoError { - match self { - Outcome::Refused | Outcome::Landed => IoError::BudgetExpired, - Outcome::Failed => IoError::Device, - } - } } type Plan = Box Option>; @@ -122,7 +112,7 @@ impl BlockAccess for Faulty { self.reads += 1; if self.read_plan.as_mut().is_some_and(|p| p(index)) { self.refused += 1; - return Err(IoError::BudgetExpired); + return Err(IoError::Device); } self.cache.read(offset, buf); Ok(()) @@ -136,9 +126,9 @@ impl BlockAccess for Faulty { self.cache.write(offset, buf, verdict.map(Outcome::landing)); match verdict { None => Ok(()), - Some(outcome) => { + Some(_) => { self.refused += 1; - Err(outcome.error()) + Err(IoError::Device) } } } @@ -263,7 +253,7 @@ fn refused_case( return false; } if !(s.commits && first.is_ok()) { - assert_eq!(first, Err(Error::BudgetExpired), "{context}"); + assert_eq!(first, Err(Error::Io), "{context}"); } // A handle whose entry lags its chain is `File::needs_reconcile`'s // window, which a stop leaves with no refusal at all. @@ -724,30 +714,22 @@ fn a_stop_that_refuses_every_retry_for_a_while_leaks_nothing() { seen += 1; (seen > 1).then_some(Outcome::Refused) })); - assert_eq!(append_two_clusters(&mut fs, &mut h, 0), Err(Error::BudgetExpired)); + assert_eq!(append_two_clusters(&mut fs, &mut h, 0), Err(Error::Io)); assert!(fs.device().refused >= 2, "the link and the rollback's re-drive were both refused"); - assert!(!RepairNotice::waits_on(Error::BudgetExpired, fs.repair_episode()), "its own refusal, not a wait"); - let episode = fs.repair_episode(); - assert!(episode.is_some(), "the refused rollback is queued"); - let mut notice = RepairNotice::default(); - assert!(notice.first_sight(episode), "a repair not yet announced"); + assert!(fs.pending_repair() > 0, "the refused rollback is queued"); // Each later attempt, and a sync, meets the first attempt's repair still - // unlanded, and says so by name rather than as a refusal of its own. + // unlanded. for attempt in 2..=9 { fs.device().arm(Box::new(move |_, offset, len| active(offset, len).then_some(Outcome::Refused))); - assert_eq!(append_two_clusters(&mut fs, &mut h, attempt), Err(Error::RepairPending)); - assert_eq!(fs.sync(), Err(Error::RepairPending), "sync on attempt {attempt}"); - assert_eq!(fs.repair_episode(), episode, "the same repair, still pending"); - assert!(!notice.first_sight(fs.repair_episode()), "announced again on attempt {attempt}"); - assert!(RepairNotice::waits_on(Error::RepairPending, fs.repair_episode())); + assert_eq!(append_two_clusters(&mut fs, &mut h, attempt), Err(Error::Io)); + assert_eq!(fs.sync(), Err(Error::Io), "sync on attempt {attempt}"); + assert!(fs.pending_repair() > 0, "the repair, still pending on attempt {attempt}"); } fs.device().plan = None; append_two_clusters(&mut fs, &mut h, 10).expect("attempt 10 on an answering device"); - assert_eq!(fs.repair_episode(), None); + assert_eq!(fs.pending_repair(), 0); - // A repair a later call leaves is another one, though no sync answered - // between the two. let mut seen = 0u32; fs.device().arm(Box::new(move |_, offset, len| { if !active(offset, len) { @@ -757,14 +739,11 @@ fn a_stop_that_refuses_every_retry_for_a_while_leaks_nothing() { (seen > 1).then_some(Outcome::Refused) })); let f = h.as_mut().expect("a handle"); - assert_eq!(fs.write(f, 1536, &common::pattern(1024, 7)), Err(Error::BudgetExpired)); - let later = fs.repair_episode(); - assert!(later.is_some() && later != episode, "{later:?} after {episode:?}"); - assert!(notice.first_sight(later), "a later repair is announced too"); + assert_eq!(fs.write(f, 1536, &common::pattern(1024, 7)), Err(Error::Io)); + assert!(fs.pending_repair() > 0, "the later refusal's repair is queued"); fs.device().plan = None; fs.sync().expect("sync lands the later repair"); - assert_eq!(fs.repair_episode(), None); - assert!(!notice.first_sight(fs.repair_episode()), "nothing pending is announced"); + assert_eq!(fs.pending_repair(), 0); assert_clean(&mut fs, true, "after the tenth attempt and the refused write after it"); (s.verify)(&mut fs, &mut h); } @@ -867,7 +846,7 @@ fn an_unlanded_rollback_is_the_answer() { (*n >= 3).then_some(Outcome::Refused) })); let too_big = vec![0u8; (before + 512) as usize]; - assert_eq!(fs.write(&mut f, 0, &too_big), Err(Error::BudgetExpired)); + assert_eq!(fs.write(&mut f, 0, &too_big), Err(Error::Io)); assert!(fs.pending_repair() > 0); fs.device().plan = None; @@ -901,8 +880,7 @@ fn a_remove_reports_a_corrupt_chain_under_the_name_it_erased() { assert_eq!(fs.pending_repair(), 0); } -/// A device that fails the free's writes outright, rather than on its own -/// budget: the name is gone, so the remove still answers `Ok` with the free +/// A device that fails the free's writes: the name is gone, so the remove still answers `Ok` with the free /// queued, and the next call answers the device's failure, having started /// nothing, until the device answers and the free lands first. #[test] @@ -913,7 +891,7 @@ fn a_remove_whose_free_the_device_fails_answers_ok_and_the_next_call_the_failure let lo = fat_offset(0, 0) as u64; let hi = fat_offset(NUM_FATS, 0) as u64; let fat = move |offset: u64, len: usize| offset < hi && offset + len as u64 > lo; - fs.device().arm(Box::new(move |_, offset, len| fat(offset, len).then_some(Outcome::Failed))); + fs.device().arm(Box::new(move |_, offset, len| fat(offset, len).then_some(Outcome::Refused))); assert_eq!(fs.remove(DOOMED), Ok(())); assert!(fs.device().refused >= 2, "the free was driven twice and failed both times"); @@ -925,12 +903,10 @@ fn a_remove_whose_free_the_device_fails_answers_ok_and_the_next_call_the_failure assert!(fs.device().refused > refused, "the next call drove the free first"); assert!(!fs.exists("after.txt").expect("exists"), "a call behind an unlanded free starts nothing"); assert_eq!(fs.sync(), Err(Error::Io)); - assert!(RepairNotice::waits_on(Error::Io, fs.repair_episode()), "an `Io` behind the queued free is that free"); fs.device().plan = None; fs.create("after.txt", stamp()).expect("create once the device answers"); assert_eq!(fs.pending_repair(), 0); - assert!(!RepairNotice::waits_on(Error::Io, fs.repair_episode())); fs.sync().expect("sync"); assert_clean(&mut fs, true, "after the failed free landed"); (s.verify)(&mut fs, &mut None); diff --git a/toyos-quiesce/src/lib.rs b/toyos-quiesce/src/lib.rs index bb7d73044eb..be679a6afc4 100644 --- a/toyos-quiesce/src/lib.rs +++ b/toyos-quiesce/src/lib.rs @@ -67,6 +67,11 @@ impl Sweep { } } +/// A policy number: how long init waits on one call into a file server that +/// is alive and has not answered — and so how long a stop request can wait +/// behind one, since init serves `power` between its loop's passes. +pub const FILES_MS: u64 = 30_000; + /// A policy number: how long init waits for `logd`'s flush before a stop. pub const FLUSH_MS: u64 = 5_000; diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index c26caac3ec0..3b8647784d4 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -108,9 +108,6 @@ fn word(e: Error) -> SyscallError { Error::Io | Error::NotFat32 | Error::Truncated | Error::CorruptChain | Error::CorruptDirectory => { SyscallError::Io } - // Nothing here refuses on a clock, so a repair pending is a volume - // that will not settle: the device's word. - Error::BudgetExpired | Error::RepairPending => SyscallError::Io, Error::NotADirectory | Error::IsADirectory | Error::DirectoryNotEmpty | Error::InvalidName => { SyscallError::InvalidArgument } diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index 0e78b83dd2c..2a93d0431e3 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -63,6 +63,9 @@ const MAX_FIDS: usize = 1024; /// Streams served at once, machine-wide. const MAX_STREAMS: usize = 64; +/// What one turn of a stream appends at most. +const STREAM_READ: usize = 64 * 1024; + /// Directories one server serves: DATA's four, with room. const MAX_DIRS: usize = 32; @@ -359,9 +362,9 @@ struct Server { /// Asks an acceptor whether a connection waits, before [`Server::accept`] /// takes it. probe: Poller, - /// A write's bytes, copied out of the client's window into this process's - /// own memory before the volume sees them. Kept, so a write allocates - /// nothing. + /// A write's bytes, copied out of the client's window or a stream's pipe + /// into this process's own memory before the volume sees them. Kept, so a + /// write allocates nothing. scratch: Vec, } @@ -896,9 +899,11 @@ impl Server { /// its file, and no more: a pipe with more waiting fires at the next /// wait, after every other client that was ready — [`Self::pump`]'s rule. fn drain(&mut self, sid: u64) { - let mut buf = vec![0u8; 64 * 1024]; + if self.scratch.len() < STREAM_READ { + self.scratch.resize(STREAM_READ, 0); + } let Some(stream) = self.streams.get_mut(&sid) else { return }; - let ended = match stream.pipe.read_nonblock(&mut buf) { + let ended = match stream.pipe.read_nonblock(&mut self.scratch[..STREAM_READ]) { Err(SyscallError::WouldBlock) => return, Ok(0) | Err(SyscallError::Gone) => None, Err(e) => Some(format!("its pipe would not read ({e:?})")), @@ -908,7 +913,7 @@ impl Server { None => Some("it reached the largest offset a file has".to_string()), Some(end) => { stream.offset = end; - match self.volume.write(node, at, &buf[..n]) { + match self.volume.write(node, at, &self.scratch[..n]) { Ok(()) => { self.dirtied(); return; diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index 5dd3a9285f7..01d9cc7e3f9 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -426,7 +426,7 @@ fn main() { /// which would otherwise hold the machine's only way to start a process. A /// first boot's DATA server formats and mounts before its first answer, so the /// bound is generous. -const FILES_BOUND: Duration = Duration::from_secs(30); +const FILES_BOUND: Duration = Duration::from_millis(toyos_quiesce::FILES_MS); /// A job on init's file worker. type Job = Box; diff --git a/userland/logd/src/policy.rs b/userland/logd/src/policy.rs index 3417c1cb110..118835e28f3 100644 --- a/userland/logd/src/policy.rs +++ b/userland/logd/src/policy.rs @@ -13,8 +13,7 @@ //! which is what `kernel/src/block.rs`'s `BlockError::Device` becomes — and //! ending the boot's log on one is right: the writes are not durable and no //! number of retries will make them so. A *budget* that expired is -//! `io::ErrorKind::WouldBlock` — `BlockError::BudgetExpired` through -//! `toyos_fat32::Error::BudgetExpired` and `SyscallError::WouldBlock` — and +//! `io::ErrorKind::WouldBlock` — and //! ending the log on one is wrong: nothing was issued, the device is untouched, //! and the next operation gets a whole fresh `block::OPERATION`. //! From 2b2563ca4f50fdfd90f7bda43f20b9fd45c91474 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 18:46:31 +0200 Subject: [PATCH 24/54] fsd_restart: DATA ends under the launch's image read, which init's loop makes fsd's `--end-at-request ` becomes `--end-at-read `, repeatable: the first open of a path to read only, this boot, ends the server before it is answered. fs_restart installs itself as a package, synced, and launches it: DATA ends under init's read of the manifest (resolution, on the worker since round 2) and again under std's read of the image, which `start` still makes on init's loop. The two write ends that followed are one end now, holding a file and a file renamed over together, so the boot still ends DATA four times inside init's window. fs_client_bound moves off the shared boot onto this one, which nothing else needs DATA on while it holds every client slot. Red at this commit: `cargo test --test toyos-build -- fsd_restart` EXIT=1, both runs: `fsd: --end-at-read: ending before the first read of apps/fs_restart/fs_restart is answered`, then no `started again` and a guest idle until the run's timeout. Co-Authored-By: Claude Opus 5.5 --- tests/common/storage.rs | 48 ++++++---- tests/fsdrestartcase/system.toml | 21 +++-- tests/toyos-rust-tests/src/bin/fs_restart.rs | 93 ++++++++++++-------- tests/toyos.rs | 5 +- userland/fsd/src/main.rs | 48 +++++----- 5 files changed, 127 insertions(+), 88 deletions(-) diff --git a/tests/common/storage.rs b/tests/common/storage.rs index b1d67e2d2af..aa4cbd16d32 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -610,14 +610,17 @@ pub fn home_overwrite_reads_back( /// budget, its directories answer `Gone`. Judged off the device. /// /// `tests/fsdrestartcase` arms every file server to end under a write to -/// `/home/fsd_end` (`--end-on`) and at the first request on `/apps` -/// (`--end-at-request`), and `test_rs_fs_restart` ends DATA's four times, the -/// first under init's own resolution of a launch: the guest asserts what a -/// client sees, init's and fsd's own lines say who ended and who started -/// again, and with the machine down the DATA partition is read by this -/// crate's own build of the `bcachefs` reader over a plain seek-and-read of -/// the image — nothing the guest executed. The flushed file holds its bytes -/// there. +/// `/home/fsd_end` (`--end-on`) and at the first read of an installed +/// package's manifest and binary (`--end-at-read`), and `test_rs_fs_restart` +/// ends DATA's four times, the first two under init's own resolution of a +/// launch and its read of the image: the guest asserts what a client sees, +/// init's and fsd's own lines say who ended and who started again, and with +/// the machine down the DATA partition is read by this crate's own build of +/// the `bcachefs` reader over a plain seek-and-read of the image — nothing the +/// guest executed. The flushed file holds its bytes there. +/// +/// `test_rs_fs_client_bound` runs first on the same boot, which nothing else +/// needs DATA on while it holds every client slot. pub fn fsd_restart( _test_config: &Path, c_bins: &[(String, Vec)], @@ -642,6 +645,10 @@ pub fn fsd_restart( if !boot.contains(MOUNTED) && !boot.contains("formatting it") { return Err(format!("fsd served DATA from no partition, so nothing here reaches a device:\n{boot}")); } + let bound = qemu.run_test("test_rs_fs_client_bound", Duration::from_secs(60)); + if bound.exit_code != Some(0) || !bound.stdout.contains("fs_client_bound: PASS") { + return Err(format!("fs_client_bound guest failed:\n{}\nconsole:\n{}{}", bound.stdout, bound.before, bound.serial)); + } let result = qemu.run_test("test_rs_fs_restart", Duration::from_secs(120)); let image = qemu.nvme_image().to_path_buf(); writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); @@ -649,18 +656,21 @@ pub fn fsd_restart( let tail = qemu.drain_serial(Duration::from_secs(20)); drop(qemu); // The console once: `stdout` is the same lines again, unprefixed. - let log = format!("{boot}\n{}{}{tail}", result.before, result.serial); + let log = format!("{boot}\n{}{}{}{}{tail}", bound.before, bound.serial, result.before, result.serial); if result.exit_code != Some(0) || !result.stdout.contains("fs_restart: PASS") { return Err(format!("fs_restart guest failed:\n{}\nconsole:\n{log}", result.stdout)); } let console = super::serial::Serial::named("fsd_restart", log.as_str()); let ended = log.matches("fsd: --end-on: ending with a write done and unanswered").count(); - if ended != 3 { - return Err(format!("fsd said it ended under a write {ended} times, not the guest's 3:\n{log}")); - } - let at_request = log.matches("fsd: --end-at-request: ending before /apps's first request is answered").count(); - if at_request != 1 { - return Err(format!("fsd said it ended under a request {at_request} times, not the launch's 1:\n{log}")); + if ended != 2 { + return Err(format!("fsd said it ended under a write {ended} times, not the guest's 2:\n{log}")); + } + for read in ["apps/fs_restart/manifest.toml", "apps/fs_restart/fs_restart"] { + let said = format!("fsd: --end-at-read: ending before the first read of {read} is answered"); + let at_read = log.matches(&said).count(); + if at_read != 1 { + return Err(format!("fsd said {said:?} {at_read} times, not the launch's 1:\n{log}")); + } } let restarted = log.lines().filter(|l| l.contains("init: fsd data (pid ") && l.contains("ended; started again")).count(); if restarted != 3 { @@ -684,10 +694,10 @@ pub fn fsd_restart( } } eprintln!( - " [fsd] DATA's server ended four times, the first under init's resolution of a launch that \ - was answered; started again three, every handle held across an end answered Gone, new \ - opens answered, the fourth closed /home to Gone; {KEPT} and {ACROSS} read back off the \ - image by the host's own bcachefs reader" + " [fsd] DATA's server ended four times, the first two under init's resolution and image \ + read of a launch that was answered; started again three, every handle held across an \ + end answered Gone, new opens answered, the fourth closed /home to Gone; {KEPT} and \ + {ACROSS} read back off the image by the host's own bcachefs reader" ); Ok(()) } diff --git a/tests/fsdrestartcase/system.toml b/tests/fsdrestartcase/system.toml index e64595ac5df..e8352c71c39 100644 --- a/tests/fsdrestartcase/system.toml +++ b/tests/fsdrestartcase/system.toml @@ -1,8 +1,9 @@ # The boot `fsd_restart` judges: the test estate's shape, with every file # server armed to end the moment it has taken a write through a file opened to -# append at `/home/fsd_end` and before it answers it, and at the first request -# on `/apps` this boot — `test_rs_fs_restart` ends DATA's four times, the -# first under a launch init resolves, and init restarts it three. +# append at `/home/fsd_end` and before it answers it, and at the first read of +# an installed package's manifest and of its binary this boot — +# `test_rs_fs_restart` ends DATA's four times, the first two under a launch +# init resolves and reads the image of, and init restarts it three. [boot] start = ["logd", "blockd", "fsd", "test-runner"] @@ -13,7 +14,7 @@ syscap = ["logread"] # `power` because `run shutdown` asks init through the `power` connector, and # the host reads the DATA partition back once the machine is down; `launcher` -# for the launch of an `/apps` path the test ends DATA's server under. +# for the launch of the `/apps` package the test ends DATA's server under. [programs.test-runner] receives = ["power", "launcher"] syscap = ["dup", "logread", "power"] @@ -36,11 +37,15 @@ devices = ["pci:1b36:0010"] # The file servers: DATA, the log and the running slot's volume, one process # each, serving every program the directories of its role. Started again when -# one ends, on the same ports. `--end-on` and `--end-at-request` are the test's -# actuators: a path on the DATA volume and a directory of DATA's, which -# neither other role's volume carries. +# one ends, on the same ports. `--end-on` and `--end-at-read` are the test's +# actuators: paths on the DATA volume, which neither other role's volume +# carries. [programs.fsd] restart = true roles = ["data", "log", "boot"] receives = ["block"] -args = ["--end-on", "home/fsd_end", "--end-at-request", "/apps"] +args = [ + "--end-on", "home/fsd_end", + "--end-at-read", "apps/fs_restart/manifest.toml", + "--end-at-read", "apps/fs_restart/fs_restart", +] diff --git a/tests/toyos-rust-tests/src/bin/fs_restart.rs b/tests/toyos-rust-tests/src/bin/fs_restart.rs index 7b0772cc085..be78c607489 100644 --- a/tests/toyos-rust-tests/src/bin/fs_restart.rs +++ b/tests/toyos-rust-tests/src/bin/fs_restart.rs @@ -2,14 +2,14 @@ //! //! Booted by `common::storage::fsd_restart` on `tests/fsdrestartcase`, whose //! file servers end the moment they take a write through a file opened to -//! append at `/home/fsd_end` and before they answer it, and at the first request -//! on `/apps` this boot: +//! append at `/home/fsd_end` and before they answer it, and at the first read +//! of an installed package's manifest and of its binary this boot: //! -//! - a launch of an `/apps` path is answered though DATA's server ends under -//! init's request resolving it: init's retry connects again and waits in the -//! port's queue, on its file worker, and init starts the server again while -//! it waits — a call from its loop would wait for ever on a server only its -//! loop could start; +//! - a launch of an installed package is answered though DATA's server ends +//! under init's read of its manifest and again under the read of its image: +//! each call is init's file worker's, whose retry connects again and waits in +//! the port's queue while init's loop starts the server again — a call from +//! the loop would wait for ever on a server only the loop could start; //! - the write the server ended under is answered as the server's end, never //! as done; //! - a handle held across an end is answered `Gone`; @@ -35,8 +35,13 @@ const REPLACING: &[u8] = b"renamed over the file a handle held across the end"; /// `--end-on`'s path, in `tests/fsdrestartcase/system.toml`. const END: &str = "/home/fsd_end"; -/// A path under `--end-at-request`'s directory: no package answers for it. -const LAUNCHED: &str = "/apps/fs_restart/nothing"; +/// The package, whose manifest and binary are `--end-at-read`'s paths. +const PACKAGE: &str = "fs_restart"; +const MANIFEST: &str = "/apps/fs_restart/manifest.toml"; +const PROGRAM: &str = "/apps/fs_restart/fs_restart"; +const SELF: &str = "/system/bin/test_rs_fs_restart"; +/// What tells this binary it is the package's copy, launched. +const LAUNCHED: &str = "launched"; /// Mirrored: what `KEPT` holds. fn kept() -> Vec { @@ -59,13 +64,38 @@ fn end_the_server(n: u32) { } } +/// This binary installed as a package, by writes alone — no read of either +/// file before init's — and durable, since an end loses what no sync covered. +fn install() { + fs::create_dir_all(format!("/apps/{PACKAGE}")).expect("make the package's directory"); + fs::copy(SELF, PROGRAM).unwrap_or_else(|e| panic!("copy {SELF} to {PROGRAM}: {e}")); + OpenOptions::new() + .write(true) + .open(PROGRAM) + .and_then(|f| f.sync_all()) + .unwrap_or_else(|e| panic!("sync {PROGRAM}: {e}")); + let manifest = format!( + "name = \"{PACKAGE}\"\nversion = \"1\"\ndigest = \"{}\"\nprogram = \"{PROGRAM}\"\n", + "0".repeat(64) + ); + let mut f = File::create(MANIFEST).unwrap_or_else(|e| panic!("create {MANIFEST}: {e}")); + f.write_all(manifest.as_bytes()).unwrap_or_else(|e| panic!("write {MANIFEST}: {e}")); + f.sync_all().unwrap_or_else(|e| panic!("sync {MANIFEST}: {e}")); +} + fn main() { - // End 1: the first request on `/apps` is init's, resolving this launch. - match Command::new(LAUNCHED).status() { - Err(e) => println!("fs_restart: end 1: the launch of {LAUNCHED} was answered, refused ({e})"), - Ok(status) => panic!("{LAUNCHED}, which no package answers for, ran and exited {status}"), + if std::env::args().nth(1).as_deref() == Some(LAUNCHED) { + println!("fs_restart: running from {PROGRAM}"); + return; } + // Ends 1 and 2: init's reads of the manifest and of the image. + install(); + let status = + Command::new(PROGRAM).arg(LAUNCHED).status().unwrap_or_else(|e| panic!("launch {PROGRAM}: {e}")); + assert!(status.success(), "{PROGRAM}, launched across two ends, exited {status}"); + println!("fs_restart: ends 1 and 2: the launch of {PROGRAM} was answered and it ran"); + fs::create_dir_all("/home/fs_restart").expect("make /home/fs_restart"); let mut f = File::create(KEPT).expect("create the kept file"); f.write_all(&kept()).expect("write the kept file"); @@ -75,29 +105,7 @@ fn main() { let mut across = File::create(ACROSS).expect("create the file written across the end"); across.write_all(BEFORE).expect("write before the end"); across.sync_all().expect("durable before the end"); - - end_the_server(2); - - // The same files, unchanged at their paths, and still `Gone`. - match held.read(&mut [0u8; 16]) { - Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => {} - other => panic!("a handle held across the end read {other:?}, not Gone"), - } - match across.write_all(b"written through a handle held across the end") { - Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => {} - other => panic!("a handle held across the end wrote {other:?}, not Gone"), - } - drop((held, across)); - let mut back = Vec::new(); - File::open(KEPT).and_then(|mut f| f.read_to_end(&mut back)).expect("a new open reads the kept file"); - assert!(back == kept(), "a new open read {} bytes, not the kept file", back.len()); - let mut whole = Vec::new(); - File::open(ACROSS).and_then(|mut f| f.read_to_end(&mut whole)).expect("read the file held across"); - assert_eq!(whole, BEFORE, "the file held across the end"); - println!("fs_restart: the server came back; held handles answered Gone, a new open read the flushed file"); - - // Another file renamed over the one `across` holds, durably, and then an end. - let mut across = OpenOptions::new().write(true).open(ACROSS).expect("hold the file again"); + // Another file renamed over the one `across` holds, durably. let mut replacement = File::create(REPLACEMENT).expect("create the replacement"); replacement.write_all(REPLACING).expect("write the replacement"); replacement.sync_all().expect("the replacement is durable"); @@ -107,6 +115,10 @@ fn main() { end_the_server(3); + match held.read(&mut [0u8; 16]) { + Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => {} + other => panic!("a handle held across the end read {other:?}, not Gone"), + } match across.write_all(b"written into whatever is at the path now") { Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => { println!("fs_restart: a handle on a file renamed over is answered Gone ({e})"); @@ -114,6 +126,15 @@ fn main() { Err(e) => panic!("a handle on a file renamed over was refused {e} ({:?}), not Gone", e.kind()), Ok(()) => panic!("a handle on a file renamed over wrote into the file now at its path"), } + drop((held, across)); + let mut back = Vec::new(); + File::open(KEPT).and_then(|mut f| f.read_to_end(&mut back)).expect("a new open reads the kept file"); + assert!(back == kept(), "a new open read {} bytes, not the kept file", back.len()); + let mut whole = Vec::new(); + File::open(ACROSS).and_then(|mut f| f.read_to_end(&mut whole)).expect("read the file at the held path"); + assert_eq!(whole, REPLACING, "the file at the path a handle held across the end"); + println!("fs_restart: the server came back; held handles answered Gone, a new open read the flushed file"); + let mut held = File::open(KEPT).expect("hold the kept file for the last end"); end_the_server(4); diff --git a/tests/toyos.rs b/tests/toyos.rs index ac913dbc55c..8bf53a44219 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -464,6 +464,9 @@ const RUST_SKIP: &[&str] = &[ // Needs a boot whose file servers are armed to end under its write, and // ends DATA's for the rest of the boot. `fsd_restart` runs it. "fs_restart", + // Holds every DATA client slot while it runs, so no other program may + // need DATA on its boot. `fsd_restart` runs it. + "fs_client_bound", // Needs a boot of its own for the readback it is judged against; `home_overwrite_reads_back` runs it. "home_overwrite_zero", // Needs a boot where the DATA volume is ours and absent; on the shared @@ -1706,7 +1709,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("data_candidate_with_bad_geometry_is_absent", &["test_rs_home_absent"]), ("home_overwrite_reads_back", &["test_rs_home_overwrite_zero"]), ("so_cache_refusals", &["test_rs_so_cache_policy"]), - ("fsd_restart", &["test_rs_fs_restart"]), + ("fsd_restart", &["test_rs_fs_client_bound", "test_rs_fs_restart"]), ("esp_filesystem", &["test_rs_esp_files"]), ("log_flush_retry", &["test_rs_esp_files"]), ("fat_backing_revoked", &["test_rs_fat_backing_revoked"]), diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index 2a93d0431e3..75dfdaedd91 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -7,7 +7,7 @@ //! nothing else of the machine. Argv is the role, and for LOG and BOOT the //! unique GUID of the partition the loader named for it when no claim on it //! was minted; then the row's own arguments, which are tests' actuators -//! ([`END_ON`], [`END_AT_REQUEST`]). +//! ([`END_ON`], [`END_AT_READ`]). //! //! **A connection is bound to the directory whose port it came in on**, and //! every path on it is resolved there (`fsd::resolve`); a write on a read-only @@ -85,12 +85,13 @@ const RAM_BLOCKS: u64 = 1 << 18; /// stages one. Armed by nothing but a boot config's `args`. const END_ON: &str = "--end-on"; -/// `--end-at-request `: the first request on `dir`'s port this boot after -/// a hello ends this process before it is answered — a server killed under a -/// client that holds a connection, whose retry connects again and waits in the -/// port's queue, as a test stages one. Once a boot, across restarts: the -/// kernel's `/tmp` keeps the mark. Armed by nothing but a boot config's `args`. -const END_AT_REQUEST: &str = "--end-at-request"; +/// `--end-at-read `, repeatable: the first open of `path` on the volume +/// to read only, this boot, ends this process before it is answered — a +/// server killed under a client that holds a connection, whose retry connects +/// again and waits in the port's queue, as a test stages one. Once a boot per +/// path, across restarts: the kernel's `/tmp` keeps the mark. Armed by nothing +/// but a boot config's `args`. +const END_AT_READ: &str = "--end-at-read"; const TOKEN_CLIENT: u64 = 1 << 32; const TOKEN_STREAM: u64 = 2 << 32; @@ -177,14 +178,13 @@ fn main() { let role = args.get(1).and_then(|r| Role::parse(r)).unwrap_or_else(|| { panic!("fsd: started with {args:?}; the first argument is a role: data, log or boot") }); - let (mut guid, mut end_on, mut end_at_request) = (None, None, None); + let (mut guid, mut end_on, mut end_at_read) = (None, None, Vec::new()); let mut rest = args.iter().skip(2); while let Some(arg) = rest.next() { match arg.as_str() { END_ON => end_on = Some(rest.next().unwrap_or_else(|| panic!("fsd: {END_ON} takes a path")).clone()), - END_AT_REQUEST => { - end_at_request = - Some(rest.next().unwrap_or_else(|| panic!("fsd: {END_AT_REQUEST} takes a directory")).clone()) + END_AT_READ => { + end_at_read.push(rest.next().unwrap_or_else(|| panic!("fsd: {END_AT_READ} takes a path")).clone()) } flag if flag.starts_with("--") => panic!("fsd: {flag} is no argument of this server"), named if guid.is_none() => guid = Some(named), @@ -210,7 +210,7 @@ fn main() { next_stream: 0, dirty_since: None, end_on, - end_at_request, + end_at_read, probe: Poller::new(caps_len), scratch: Vec::new(), } @@ -357,8 +357,8 @@ struct Server { dirty_since: Option, /// [`END_ON`]'s path on the volume. end_on: Option, - /// [`END_AT_REQUEST`]'s directory. - end_at_request: Option, + /// [`END_AT_READ`]'s paths on the volume. + end_at_read: Vec, /// Asks an acceptor whether a connection waits, before [`Server::accept`] /// takes it. probe: Poller, @@ -385,21 +385,21 @@ fn stat_reply(meta: Meta) -> Reply { Reply { kind: meta.kind.wire(), value2: meta.size, mtime: meta.mtime, ..Reply::ok() } } -/// Whether this is [`END_AT_REQUEST`]'s first firing this boot for `dir`: its +/// Whether this is [`END_AT_READ`]'s first firing this boot for `path`: its /// mark is made once in the kernel's `/tmp`, which outlives this process and /// not the boot. Looked for and then made, not made exclusively — the std /// fork's `create_new` on a kernel path is not exclusive — which one server /// of a role at a time makes safe. A mark that can be neither found nor made /// is an actuator that cannot do its one job. -fn first_request_this_boot(dir: &str) -> bool { - let mark = format!("/tmp/fsd-end-at-request{}", dir.replace('/', "-")); +fn first_read_this_boot(path: &str) -> bool { + let mark = format!("/tmp/fsd-end-at-read-{}", path.replace('/', "-")); match std::fs::metadata(&mark) { Ok(_) => false, Err(e) if e.kind() == std::io::ErrorKind::NotFound => { - std::fs::write(&mark, b"").unwrap_or_else(|e| panic!("fsd: {END_AT_REQUEST}: its mark {mark}: {e}")); + std::fs::write(&mark, b"").unwrap_or_else(|e| panic!("fsd: {END_AT_READ}: its mark {mark}: {e}")); true } - Err(e) => panic!("fsd: {END_AT_REQUEST}: its mark {mark} would not be looked up: {e}"), + Err(e) => panic!("fsd: {END_AT_READ}: its mark {mark} would not be looked up: {e}"), } } @@ -631,11 +631,6 @@ impl Server { if self.clients[&id].window.is_none() { return Answer::Drop("it asked before it lent a window"); } - let dir = &self.caps[self.clients[&id].cap].dir; - if self.end_at_request.as_ref() == Some(dir) && first_request_this_boot(dir) { - println!("fsd: {END_AT_REQUEST}: ending before {dir}'s first request is answered"); - std::process::exit(1); - } match self.serve_one(id, op, r) { Ok(answer) => answer, Err(e) => Answer::Reply(Reply::refused(e)), @@ -676,6 +671,11 @@ impl Server { Resolved::Absolute(p) => return Ok(self.link(id, &p)), Resolved::Path(p) => p, }; + let reads_only = r.flags & (O_WRITE | O_APPEND | O_CREATE | O_TRUNCATE | O_CREATE_NEW) == 0; + if reads_only && self.end_at_read.contains(&path) && first_read_this_boot(&path) { + println!("fsd: {END_AT_READ}: ending before the first read of {path} is answered"); + std::process::exit(1); + } if self.clients[&id].fids.len() >= MAX_FIDS { return Err(SyscallError::ResourceExhausted); } From 3acc23fa2fd2a1396599f3aa919ec42daf056da4 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 18:56:51 +0200 Subject: [PATCH 25/54] init: a launch's files are read on the file worker, its spawn after MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `serve_launch` builds the command first and hands it to the worker with the path: the package's resolution and std's new `CommandExt::prepare` — the program found, its working directory judged, a program on a file server read into a `toyos::process::Image` — run there, so `start`'s spawn on init's loop calls no file server. A relative working directory is refused before any of it rather than after resolution; std's client never sends one. `toyos::process::Image` is the SDK's half of `SpawnArgs::image`: the bytes in a memory object of the caller's own, read to its whole length, and the two words the spawn carries. std's `read_image` is it plus the error words. Fork: wt-toyos-fsd 0bf12938d13. Green: `cargo test --test toyos-build -- fsd_restart` EXIT=0 (red EXIT=1 at 2b2563ca). Co-Authored-By: Claude Opus 5.5 --- rust | 2 +- toyos/src/process.rs | 51 ++++++++++++++++++++ userland/init/src/main.rs | 99 +++++++++++++++++++++++---------------- 3 files changed, 111 insertions(+), 41 deletions(-) diff --git a/rust b/rust index 2b2fb74a303..0bf12938d13 160000 --- a/rust +++ b/rust @@ -1 +1 @@ -Subproject commit 2b2fb74a3034c4398bc5332b44edf3aeccb04c23 +Subproject commit 0bf12938d135ffbe2f35dda168571aee472d4cbe diff --git a/toyos/src/process.rs b/toyos/src/process.rs index bd807e146ca..b5b9eb1ea19 100644 --- a/toyos/src/process.rs +++ b/toyos/src/process.rs @@ -10,6 +10,7 @@ use toyos_abi::handle::Rights; use toyos_abi::syscall::{self, ProcessStats, SyscallError}; use crate::endow::FromHandle; +use crate::shm::SharedMemory; use crate::{AsHandle, OwnedHandle, RawHandle}; pub struct Process(pub(crate) OwnedHandle); @@ -70,3 +71,53 @@ impl FromHandle for Process { Self(OwnedHandle(raw)) } } + +/// A program's bytes in a memory object of this process's own, which the +/// kernel pages a child from: how a program the kernel does not serve is +/// spawned (`SpawnArgs::image`). +pub struct Image { + object: SharedMemory, + len: u64, +} + +/// Why [`Image::read`] made no image. +#[derive(Debug)] +pub enum ImageRefused { + /// The program is an empty file. + Empty, + /// No memory object would hold it. + Memory(SyscallError), + /// The reader ran out before the length it was read at. + Shrank, + /// The reader refused. + Read(E), +} + +impl Image { + /// `len` bytes, from `read` asked until it has given them all. + pub fn read( + len: u64, + mut read: impl FnMut(&mut [u8]) -> Result, + ) -> Result> { + let size = usize::try_from(len).map_err(|_| ImageRefused::Memory(SyscallError::InvalidArgument))?; + if size == 0 { + return Err(ImageRefused::Empty); + } + let mut object = SharedMemory::create(size).map_err(ImageRefused::Memory)?; + let bytes = object.as_mut_slice(); + let mut done = 0; + while done < size { + match read(&mut bytes[done..]).map_err(ImageRefused::Read)? { + 0 => return Err(ImageRefused::Shrank), + n => done += n, + } + } + Ok(Self { object, len }) + } + + /// What `SpawnArgs::image` and `SpawnArgs::image_len` carry. The object is + /// this process's until the spawn returns; the child keeps its own. + pub fn spawn_words(&self) -> (u64, u64) { + (u64::from(self.object.as_handle().0), self.len) + } +} diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index 01d9cc7e3f9..e3471acb4e2 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -34,10 +34,12 @@ //! //! **init never waits on a file server it supervises from its loop alone**: //! the loop is also where a server that ended is started again. Its calls into -//! the file servers run on one worker thread ([`Worker`]), and init waits on -//! the call and on a service's end together ([`Init::files`]), so a call a -//! server's end left in the port's queue goes on to the server started in its -//! place, and one alive and silent costs a bounded wait ([`FILES_BOUND`]). +//! the file servers run on one worker thread ([`Worker`]) — a launch's files +//! included, read there before its spawn (`CommandExt::prepare`) — and init +//! waits on the call and on a service's end together ([`Init::files`]), so a +//! call a server's end left in the port's queue goes on to the server started +//! in its place, and one alive and silent costs a bounded wait +//! ([`FILES_BOUND`]). //! //! **Every program it starts gets `HOME` from its row** (`Program::home`), over //! anything a launching caller carried: a service its own `/state/`, made @@ -1417,40 +1419,6 @@ impl Init<'_> { .map(|(name, handle)| (name, unsafe { Connector::from_raw(handle) })) .collect(); - let installed; - // On the worker: a path under `/apps` is read off its package, and that is - // a call into a file server init supervises. - let (system, path) = (self.system, request.program.to_string()); - let resolved = match self.files("a launch's path", move || resolve(system, &path)) { - Ok(resolved) => resolved, - Err(why) => { - say!("init: launcher: {} was not resolved: {why}", request.program); - let _ = conn.try_signal(launch::MSG_REFUSED); - return; - } - }; - let program = match resolved { - Resolved::Row(row) => row, - Resolved::Package(row) => { - installed = row; - &installed - } - Resolved::NotDeclared => { - // **`try_send_bytes` and not `send`.** A blocking write is the - // other half of the rule that made the read side an event loop: a - // client that never drains its end decides when init runs again. - // The `HOME` is the session's: a program no row names is no service. - let home = toyos_manifest::session_home(); - let _ = conn.try_send_bytes(launch::MSG_NOT_DECLARED, home.as_bytes()); - return; - } - Resolved::Refused(why) => { - say!("init: launcher: {why}"); - let _ = conn.try_signal(launch::MSG_REFUSED); - return; - } - }; - // std joins a relative `current_dir` onto init's own cwd, so passing one // on would start the child in init's `/` — the default this field removes. if !request.cwd.starts_with('/') { @@ -1464,8 +1432,6 @@ impl Init<'_> { // nothing extra — and `/system/bin/echo` spawned as `/system/bin/toybox` is a toybox that // was never told which applet it is. let mut command = Command::new(request.program); - let caller_slots: Vec<(u32, toyos::RawHandle)> = - request.slot_numbers().zip(slots.0.iter().copied()).collect(); // **Carried, not inherited.** A child of the launcher would otherwise get // init's environment and init's working directory, so `cd /tmp && ls` would // list `/`. The launcher is a spawn service, not a session. @@ -1485,6 +1451,59 @@ impl Init<'_> { } } + let installed; + // On the worker, every file the launch reads: a path under `/apps` is + // read off its package, and the program and its working directory may + // be a file server's, which init supervises. + let (system, path) = (self.system, request.program.to_string()); + let found = self.files("a launch's files", move || { + let resolved = resolve(system, &path); + let prepared = match resolved { + Resolved::Row(_) | Resolved::Package(_) => command.prepare().map(drop), + Resolved::NotDeclared | Resolved::Refused(_) => Ok(()), + }; + (resolved, prepared.map(|()| command)) + }); + let (resolved, prepared) = match found { + Ok(found) => found, + Err(why) => { + say!("init: launcher: {} was not resolved: {why}", request.program); + let _ = conn.try_signal(launch::MSG_REFUSED); + return; + } + }; + let program = match resolved { + Resolved::Row(row) => row, + Resolved::Package(row) => { + installed = row; + &installed + } + Resolved::NotDeclared => { + // **`try_send_bytes` and not `send`.** A blocking write is the + // other half of the rule that made the read side an event loop: a + // client that never drains its end decides when init runs again. + // The `HOME` is the session's: a program no row names is no service. + let home = toyos_manifest::session_home(); + let _ = conn.try_send_bytes(launch::MSG_NOT_DECLARED, home.as_bytes()); + return; + } + Resolved::Refused(why) => { + say!("init: launcher: {why}"); + let _ = conn.try_signal(launch::MSG_REFUSED); + return; + } + }; + let command = match prepared { + Ok(command) => command, + Err(e) => { + say!("init: launcher: cannot start {}: {e}", program.name); + let _ = conn.try_signal(launch::MSG_REFUSED); + return; + } + }; + let caller_slots: Vec<(u32, toyos::RawHandle)> = + request.slot_numbers().zip(slots.0.iter().copied()).collect(); + // `inherit_handle` duplicates into the child, so init's own copies go with // `slots` when this returns. self.make_home(program); From 7a6567403d49bdde9bec56ac91cb8944ed4919f9 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 18:56:52 +0200 Subject: [PATCH 26/54] abi: dlopen from a memory object goes; only spawn takes one `dl_open_image`, `ImageRef`, SYS_DLOPEN's fourth word and the kernel's image arm of `sys_dlopen` had no caller outside tests: std has no dlopen, and libc's reaches no served path until it is a file-server client. `sys_dlopen` is main's again; `shared_image` stays for spawn. The tests' dlopen-of-an-image arms go with it. Co-Authored-By: Claude Opus 5.5 --- kernel/src/file_backing.rs | 2 +- kernel/src/syscall/dispatch.rs | 12 +--- kernel/src/syscall/vm.rs | 65 +++---------------- kernel/src/user_ptr.rs | 2 - .../src/bin/abuse_elf_loader.rs | 6 -- .../src/bin/spawn_image_object.rs | 12 ---- toyos-abi/src/syscall.rs | 26 +------- toyos-userbound/src/span.rs | 1 - 8 files changed, 13 insertions(+), 113 deletions(-) diff --git a/kernel/src/file_backing.rs b/kernel/src/file_backing.rs index b1d3a757594..39a3f1bb285 100644 --- a/kernel/src/file_backing.rs +++ b/kernel/src/file_backing.rs @@ -81,7 +81,7 @@ impl FileBacking for ReadOnlyBacking { } } -/// An executable or library a caller read into a shared memory object of its +/// An executable a caller read into a shared memory object of its /// own and handed over by handle: what a program in `/apps` is spawned and /// paged from, since no file server's volume is the kernel's. /// diff --git a/kernel/src/syscall/dispatch.rs b/kernel/src/syscall/dispatch.rs index 0567d9d8d8b..c6bef4b2c0f 100644 --- a/kernel/src/syscall/dispatch.rs +++ b/kernel/src/syscall/dispatch.rs @@ -342,18 +342,8 @@ pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> None => return bad_addr, }, }; - // Only named here: `sys_dlopen` answers a name this process holds - // before it looks at the object at all. - let image = match a4 { - 0 => None, - raw => { - let Some(at) = UserAddr::checked(raw) else { return bad_addr }; - let Ok(image) = ctx.copy_in::(at) else { return bad_addr }; - Some(image) - } - }; // ctx carries the copy-out: sys_dlopen writes init_out only once the load succeeds. - sys_dlopen(&ctx, &path, init_out, image) + sys_dlopen(&ctx, &path, init_out) } SYS_DLSYM => { let name = match ctx.user_str(UserAddr::new(a2), a3) { Ok(s) => s, Err(e) => return e.to_u64() }; diff --git a/kernel/src/syscall/vm.rs b/kernel/src/syscall/vm.rs index adcf3bd7bed..03223e27bc4 100644 --- a/kernel/src/syscall/vm.rs +++ b/kernel/src/syscall/vm.rs @@ -199,7 +199,7 @@ pub(super) fn sys_munmap(addr: u64, _size: u64) -> u64 { } /// The first `len` bytes of the shared memory object `handle` names, as a -/// program or library image; `Err` is the syscall's answer. +/// program image; `Err` is the syscall's answer. pub(super) fn shared_image(handle: u64, len: u64) -> Result { let Ok(raw) = u32::try_from(handle) else { return Err(SyscallError::InvalidArgument.to_u64()) }; let object = process::with_process_data(|data| { @@ -212,15 +212,7 @@ pub(super) fn shared_image(handle: u64, len: u64) -> Result, - image: Option, -) -> u64 { +pub(super) fn sys_dlopen(ctx: &crate::user_ptr::SyscallContext, path: &str, init_out: Option) -> u64 { let cwd = process::with_process_data(|d| d.cwd.clone()); let resolved = vfs::lock().resolve_absolute(&cwd, path); @@ -238,43 +230,15 @@ pub(super) fn sys_dlopen( return idx as u64; } - let loaded = match image { - Some(image) => { - let image = match shared_image(image.handle, image.len) { - Ok(image) => image, - Err(refused) => return refused, - }; - match crate::elf::load_shared_lib(&image) { - Ok((lib, _, _)) => Ok(lib), - Err(msg) => { - log!("dlopen: {}: {}", resolved, msg); - return SyscallError::Unknown.to_u64(); - } - } - } - None => Err(()), - }; - let lib = match loaded { - Ok(lib) => lib, - Err(()) => match open_cached(&resolved) { - Ok(lib) => lib, - Err(e) => return e, - }, - }; - map_dlopened(ctx, &resolved, init_out, lib) -} - -/// A library at `resolved`, out of the shared cache or loaded into it. -fn open_cached(resolved: &str) -> Result { // Opened before the cache is consulted: the answer depends on what the path holds *now*, and the fast path above is what keeps a `dlopen` loop from paying. - let (backing, id) = match vfs::lock().open_backing_identified(resolved) { + let (backing, id) = match vfs::lock().open_backing_identified(&resolved) { Ok(pair) => pair, Err(e) => { log!("dlopen: {}: {e}", resolved); - return Err(e.to_u64()); + return e.to_u64(); } }; - let lib = match crate::elf::try_clone_cached(resolved, id) { + let mut lib = match crate::elf::try_clone_cached(&resolved, id) { Ok(Some(lib)) => lib, Err(e) => { log!( @@ -282,33 +246,24 @@ fn open_cached(resolved: &str) -> Result { it was loaded, and the image cannot be replaced while a process has it mapped", resolved ); - return Err(e.to_u64()); + return e.to_u64(); } Ok(None) => { let (lib, rw_offset, rw_size) = match crate::elf::load_shared_lib(backing.as_ref()) { Ok(result) => result, Err(msg) => { log!("dlopen: {}", msg); - return Err(SyscallError::Unknown.to_u64()); + return SyscallError::Unknown.to_u64(); } }; - match crate::elf::cache_loaded_lib(resolved, id, lib, rw_offset, rw_size) { + match crate::elf::cache_loaded_lib(&resolved, id, lib, rw_offset, rw_size) { Ok(lib) => lib, - Err(e) => return Err(e.to_u64()), + Err(e) => return e.to_u64(), } } }; - Ok(lib) -} -/// Map `lib` into the caller, run its relocations, and register it under `resolved`. -fn map_dlopened( - ctx: &crate::user_ptr::SyscallContext, - resolved: &str, - init_out: Option, - mut lib: crate::elf::LoadedLib, -) -> u64 { let pt = process::current_address_space(); let mapped = process::with_process_data(|_data| { // The module's own program headers decide which pages are writable @@ -411,7 +366,7 @@ fn map_dlopened( is_static: false, }); } - data.elf.lib_paths.push(alloc::string::String::from(resolved)); + data.elf.lib_paths.push(resolved); data.elf.loaded_libs.push(lib); idx as u64 } diff --git a/kernel/src/user_ptr.rs b/kernel/src/user_ptr.rs index 0d1f05f1478..137f912a7d8 100644 --- a/kernel/src/user_ptr.rs +++ b/kernel/src/user_ptr.rs @@ -43,8 +43,6 @@ unsafe impl UserSafe for crate::object::ops::Stat {} // SAFETY: `#[repr(C)] Copy`, fourteen `u64`s, no padding; every field is validated where it is used, not here. unsafe impl UserSafe for toyos_abi::syscall::SpawnArgs {} -// SAFETY: `#[repr(C)] Copy`, two `u64`s, no padding; both are bounded by the copy that reads them. -unsafe impl UserSafe for toyos_abi::syscall::ImageRef {} // SAFETY: `#[repr(C)] Copy`, `RawHandle`, a `flags: u32`, then six `u64`s — no padding. unsafe impl UserSafe for toyos_abi::syscall::NamespaceBuild {} // SAFETY: `#[repr(C)] Copy`, `u64`, `u64`, `i64` — 24 bytes, no padding. diff --git a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs index 6ef014a45c0..e7d594f4a99 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs @@ -329,12 +329,6 @@ fn dlopen_refused(name: &str, bytes: &[u8]) { } Err(e) => assert!(!format!("{e}").is_empty(), "{name}: dlopen error message"), } - // The same bytes handed over by handle, under a name no path holds. - let named = format!("/image/{name}"); - let object = image_object(bytes); - if let Ok(handle) = syscall::dl_open_image(named.as_bytes(), toyos::AsHandle::as_handle(&object), bytes.len() as u64) { - panic!("{name}: dlopen of the image loaded it as handle {handle}, and the loader must refuse it"); - } } /// A minimal, honest exe: one PT_LOAD covering the whole file at vaddr 0. diff --git a/tests/toyos-rust-tests/src/bin/spawn_image_object.rs b/tests/toyos-rust-tests/src/bin/spawn_image_object.rs index 42c67159713..28b728fd5a4 100644 --- a/tests/toyos-rust-tests/src/bin/spawn_image_object.rs +++ b/tests/toyos-rust-tests/src/bin/spawn_image_object.rs @@ -8,8 +8,6 @@ //! - The object overwritten with `hlt` the moment the spawn returns: the child //! runs what it had copied and faults on the rest, and the machine goes on — //! the next spawn from a fresh object runs. -//! - `dlopen` of a name this process already holds answers that library and -//! never looks at the object: a handle it could not read is not refused. use toyos::shm::SharedMemory; use toyos::AsHandle; @@ -17,7 +15,6 @@ use toyos_abi::handle::{RawHandle, Rights}; use toyos_abi::syscall::{self, SpawnArgs, SyscallError}; const SELF: &str = "/system/bin/test_rs_spawn_image_object"; -const LIB: &str = "/system/lib/libtls_lib.so"; const CHILD: &str = "spawned-from-an-object"; const CWD: &str = "/"; @@ -89,14 +86,5 @@ fn main() { drop(object); runs("after the overwrite"); - let (lib, lib_len) = object_of(LIB); - let name = b"/image/spawn_image_object/libtls_lib.so"; - let first = syscall::dl_open_image(name, lib.as_handle(), lib_len).expect("dlopen from an object"); - let unreadable = syscall::dup_narrowed(lib.as_handle(), Rights::DUP).expect("a duplicate without MAP"); - let again = syscall::dl_open_image(name, unreadable, lib_len); - assert_eq!(again, Ok(first), "a name this process holds was answered from the object, not the name"); - syscall::close(unreadable); - println!("spawn_image_object: a held name's dlopen answered its library without reading the object"); - println!("spawn_image_object: PASS"); } diff --git a/toyos-abi/src/syscall.rs b/toyos-abi/src/syscall.rs index 446109c7290..21544f35a8e 100644 --- a/toyos-abi/src/syscall.rs +++ b/toyos-abi/src/syscall.rs @@ -409,16 +409,6 @@ pub struct SpawnArgs { const _: () = assert!(core::mem::size_of::() == 112); -/// A library's bytes in a shared memory object, for `SYS_DLOPEN`'s fourth -/// word: what [`SpawnArgs::image`] and [`SpawnArgs::image_len`] are to a spawn. -#[repr(C)] -#[derive(Clone, Copy)] -pub struct ImageRef { - /// A handle to the object, carrying `MAP`. - pub handle: u64, - pub len: u64, -} - /// One `(label, handle)` pair of a process's endowment table. /// /// `label_off`/`label_len` index the label blob that travels beside the @@ -1917,22 +1907,8 @@ pub fn readlink(path: &[u8], buf: &mut [u8]) -> Result { /// Load a shared library (.so) into the current process. /// Runs .init_array constructors after loading. pub fn dl_open(path: &[u8]) -> Result { - dl_open_with(path, 0) -} - -/// Load a shared library whose bytes the caller read itself into the first -/// `len` bytes of the shared memory object `image`, under `name`: a library on -/// a file server's volume, which the kernel cannot open. The object is taken -/// as [`SpawnArgs::image`] says; a load under a name this process already -/// holds answers that library's handle, and the object is not asked about. -pub fn dl_open_image(name: &[u8], image: RawHandle, len: u64) -> Result { - let image = ImageRef { handle: image.0 as u64, len }; - dl_open_with(name, &image as *const ImageRef as u64) -} - -fn dl_open_with(path: &[u8], image: u64) -> Result { let mut init_info: [u64; 2] = [0; 2]; - let handle = check(syscall(SYS_DLOPEN, path.as_ptr() as u64, path.len() as u64, init_info.as_mut_ptr() as u64, image))?; + let handle = check(syscall(SYS_DLOPEN, path.as_ptr() as u64, path.len() as u64, init_info.as_mut_ptr() as u64, 0))?; // Run .init_array constructors (e.g. EH frame finder registration in cdylib std) let init_array_ptr = init_info[0]; let init_count = init_info[1]; diff --git a/toyos-userbound/src/span.rs b/toyos-userbound/src/span.rs index b4209becc0b..9678395432a 100644 --- a/toyos-userbound/src/span.rs +++ b/toyos-userbound/src/span.rs @@ -143,7 +143,6 @@ mod tests { ("SchedInfo", 24, 8), ("FramebufferInfo", 32, 4), ("SpawnArgs", 112, 8), - ("ImageRef", 16, 8), ("NamespaceBuild", 56, 8), ("InboxSetup", 16, 8), ("ProcessStats", 128, 8), From 80a1f1ceb242d73e47a9a935fc31f90d5cced033 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 19:23:35 +0200 Subject: [PATCH 27/54] kernel: the write-back queue, iod and the durability ledger go; /tmp stays in memory MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The kernel's one writable mount is tmpfs, whose pages are the file, and ROOT is read-only from memory: nothing the kernel flushes reaches a device any more. So the machinery that protected such flushes protects nothing, and writeback_reopen and writeback_spawn — whose `/tmp` files kept their pages pinned or not — could not fail on their claim. Both tests go with it. Deleted: `writeback.rs`, `iod.rs`, `durability.rs` and kernel-loom's model of it with its `durability-settle-blind` control; the file cache's dirt, pin, shrink mark and flush plan; the VFS's flush, sync, durability and close surface and the trait methods only they called; `writeback-stall`; the stop's "Syncing filesystems..." sync; `block::OpenUpdate` and the scheduler's `MID_UPDATE` bit, whose one user was the kernel-file fsync. What stays and how: a handle's drop releases its reference under the file cache's lock alone; the last release of a deleted file takes its pages, and any other file's clean pages are CLOCK's to evict, since a governed page is a read-only mount's and never dirty. `SYS_FSYNC` on a kernel file answers 0; on a partition claim it is unchanged. A tmpfs file's mtime lives in the file cache, set at create and by every write and truncate (`touch`), where the flush used to carry the handle's to the entry. `ftruncate` is `set_size`. `sysret-ss-probe` and `sched-operation-nesting` rode iod; they run on a `probe` kernel thread spawned only when one is armed. The stop's two readers of "Syncing filesystems..." read the stop record instead. Flat drain waits in two backing tests and metalprobe, and fs_transactional's drain-by-spawn, waited on iod and go. `iommu-userdev-foreign-dma` now stages a network function's first grant alone: it also hit blockd's NVMe, which failed and gave slot 0 back before netd claimed, so netd held slot 0 and a kernel blaming the first claimed slot passed `userdev_dma_fault`. With blockd alive netd holds slot 1. Co-Authored-By: Claude Opus 5.5 --- kernel-loom/Cargo.toml | 9 - kernel-loom/src/lib.rs | 5 - kernel-loom/tests/durability.rs | 115 ------ kernel/Cargo.toml | 3 - kernel/src/actuator.rs | 7 +- kernel/src/bcachefs_adapter.rs | 34 +- kernel/src/block.rs | 23 -- kernel/src/durability.rs | 59 --- kernel/src/file_cache.rs | 380 +++--------------- kernel/src/iod.rs | 53 --- kernel/src/main.rs | 22 +- kernel/src/object/file.rs | 11 +- kernel/src/object/ops.rs | 81 +--- kernel/src/pcidev/mod.rs | 14 +- kernel/src/quiesce.rs | 23 +- kernel/src/revoke_selftest.rs | 7 +- kernel/src/sched/kthread.rs | 4 +- kernel/src/sched_gate.rs | 2 +- kernel/src/scheduler.rs | 4 +- kernel/src/syscall/machine.rs | 11 +- kernel/src/tmpfs.rs | 47 +-- kernel/src/vfs.rs | 148 +------ kernel/src/writeback.rs | 134 ------ src/build.rs | 3 - src/ci.rs | 4 - src/metal.rs | 2 +- tests/common/iommu.rs | 2 +- tests/common/origin.rs | 14 +- tests/common/power.rs | 3 +- tests/test-durations | 2 - .../src/bin/fat_backing_revoked.rs | 16 +- .../src/bin/fs_transactional.rs | 13 - .../src/bin/handle_transfer.rs | 6 +- .../src/bin/home_backing_revoked.rs | 18 +- .../src/bin/writeback_reopen.rs | 55 --- .../src/bin/writeback_spawn.rs | 97 ----- tests/toyos.rs | 115 ++---- toyos-sched/src/task.rs | 42 +- userland/metalprobe/src/usb.rs | 19 +- userland/toybox/src/cp.rs | 5 +- 40 files changed, 170 insertions(+), 1442 deletions(-) delete mode 100644 kernel-loom/tests/durability.rs delete mode 100644 kernel/src/durability.rs delete mode 100644 kernel/src/iod.rs delete mode 100644 kernel/src/writeback.rs delete mode 100644 tests/toyos-rust-tests/src/bin/writeback_reopen.rs delete mode 100644 tests/toyos-rust-tests/src/bin/writeback_spawn.rs diff --git a/kernel-loom/Cargo.toml b/kernel-loom/Cargo.toml index a389a4896db..f8939f4f580 100644 --- a/kernel-loom/Cargo.toml +++ b/kernel-loom/Cargo.toml @@ -147,15 +147,6 @@ shard-publish-relaxed = [] # Never on by default and never reachable from a kernel build, which declares # the same name only so `cfg` checking knows it. sleeplock-acquire-off = [] -# The negative control for the durability generations. It turns `settle` back -# into the pre-generation blind clear, and `durability.rs` must red under it — -# a page marked clean whose bytes never reached the device: -# -# cargo test --manifest-path kernel-loom/Cargo.toml --features durability-settle-blind \ -# --test durability -# -# Never on by default, and no kernel build can reach it. -durability-settle-blind = [] # The negative control for a claimed PCI function's interrupt record. Its two # read-modify-writes — the reader's `swap` of the count and the scheduler pass's # `swap` of the wake flag — become a load and a store, which is the whole of diff --git a/kernel-loom/src/lib.rs b/kernel-loom/src/lib.rs index 2f2b84aa8b0..a4bde19bb6a 100644 --- a/kernel-loom/src/lib.rs +++ b/kernel-loom/src/lib.rs @@ -199,11 +199,6 @@ pub mod device_irq; #[path = "../../kernel/src/sched/poison.rs"] pub mod poison; -/// Durability debt as generations. Pure `core`, so it compiles here unshimmed; -/// `tests/durability.rs` drives the kernel's flush protocol over it. -#[path = "../../kernel/src/durability.rs"] -pub mod durability; - /// A poll ring's one-shot answer. It names atomics and nothing else, and that /// narrowness is load-bearing: a file that named a ring's page or a watch could /// not be compiled here at all, and the race would stop being checked by diff --git a/kernel-loom/tests/durability.rs b/kernel-loom/tests/durability.rs deleted file mode 100644 index 709143e3cf4..00000000000 --- a/kernel-loom/tests/durability.rs +++ /dev/null @@ -1,115 +0,0 @@ -//! Loom: the durability generations under a racing writer. -//! -//! `kernel/src/durability.rs` compiles into this crate unshimmed and is driven -//! in the exact call order `Vfs::flush_file` and `Vfs::sync_mount` use: -//! snapshot-with-copy under the lock, device work with the lock dropped, -//! settle with the snapshot. The models prove the accounting — no interleaving -//! discharges debt over bytes the device does not hold; that the kernel -//! settles only on success paths is the type's half (no other discharge -//! compiles), held end-to-end by the guest controls. -//! -//! The negative control is a cargo feature rather than a comment: -//! -//! ```text -//! cargo test --manifest-path kernel-loom/Cargo.toml --features durability-settle-blind \ -//! --test durability -//! ``` -//! -//! turns `settle` back into the blind clear, and both two-threaded models red. - -#![cfg(feature = "loom")] - -use kernel_loom::durability::Owed; -use loom::sync::{Arc, Mutex}; - -/// One cached page and its debt; the u32 stands for the page's bytes. -struct Page { - bytes: u32, - dirt: Owed, -} - -/// `Vfs::flush_file`'s per-page protocol. -fn flush(page: &Mutex, device: &Mutex) { - let (copied, upto) = { - let page = page.lock().unwrap(); - (page.bytes, page.dirt.snapshot()) - }; - *device.lock().unwrap() = copied; - page.lock().unwrap().dirt.settle(upto); -} - -/// A write racing one flush either reaches the device or stays owed — and the -/// retry the kernel believes in (`fsync`'s next attempt) then delivers it. -#[test] -fn a_page_marked_clean_is_on_the_device() { - loom::model(|| { - let page = Arc::new(Mutex::new(Page { bytes: 0, dirt: Owed::new() })); - let device = Arc::new(Mutex::new(0u32)); - - let writer = { - let page = page.clone(); - loom::thread::spawn(move || { - let mut page = page.lock().unwrap(); - page.bytes = 7; - page.dirt.record_write(); - }) - }; - flush(&page, &device); - writer.join().unwrap(); - - if page.lock().unwrap().dirt.is_owed() { - flush(&page, &device); - } - let page = page.lock().unwrap(); - assert!( - !page.dirt.is_owed() && *device.lock().unwrap() == page.bytes, - "a page marked clean is not on the device: the settle covered a write the flush \ - never copied", - ); - }); -} - -/// The mount's commit debt: a settle covers only writes the device flush could -/// have committed, so F5's second-call lie is unwritable. -#[test] -fn a_settled_commit_covers_only_flushed_writes() { - loom::model(|| { - // (writes in the device's cache, the mount's debt, writes made durable). - let mount = Arc::new(Mutex::new((0u32, Owed::new(), 0u32))); - - let writer = { - let mount = mount.clone(); - loom::thread::spawn(move || { - let mut m = mount.lock().unwrap(); - m.0 += 1; - m.1.record_write(); - }) - }; - let upto = mount.lock().unwrap().1.snapshot(); - { - // The device flush commits whatever its cache holds at this instant. - let mut m = mount.lock().unwrap(); - m.2 = m.0; - } - mount.lock().unwrap().1.settle(upto); - writer.join().unwrap(); - - let m = mount.lock().unwrap(); - assert!( - m.1.is_owed() || m.2 == m.0, - "commit debt settled while a write had not been made durable", - ); - }); -} - -/// A flush that failed presents nothing, so the debt stands and the next -/// caller still owes the device — the discharge is the settlement, not the attempt. -#[test] -fn a_failed_commit_settles_nothing() { - loom::model(|| { - let mut debt = Owed::new(); - debt.record_write(); - let _upto = debt.snapshot(); - assert!(debt.is_owed(), "an unpresented settlement discharged the debt"); - }); -} diff --git a/kernel/Cargo.toml b/kernel/Cargo.toml index 70caaf26652..d4cfb123188 100644 --- a/kernel/Cargo.toml +++ b/kernel/Cargo.toml @@ -55,9 +55,6 @@ log-commit-release-off = [] # `src/log/registry.rs`'s pointer store and load go `Relaxed`, and # `log_publish` reds. shard-publish-relaxed = [] -# `src/durability.rs`'s `settle` becomes the pre-generation blind clear, and -# `durability` reds. -durability-settle-blind = [] # `src/pcidev/record.rs`'s two read-modify-writes become a load and a store, so # a message the ISR records between a reader's two halves is lost and a wake can # be owed twice, and `device_irq` reds. That pair *is* the record's design, so diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 2cbb14541fb..b2a52c0ce31 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -61,7 +61,7 @@ actuators! { /// record before it and `screen_early_panel` reads a paint it can attribute. test_early_halt = "test-early-halt"; - /// Have `iod` null SS, force a switch, and report whether it reloaded — the + /// Have the `probe` thread null SS, force a switch, and report whether it reloaded — the /// AMD `SYSRET` SS-attributes workaround's only guest-observable proof. sysret_ss_probe = "sysret-ss-probe"; @@ -250,9 +250,6 @@ actuators! { /// Report the preempt depth and backtrace at the deepest point of a disk transfer; it stages nothing, only measures. io_depth_probe = "io-depth-probe"; - /// Park `iod` before it drains so a closed file's write-back stays pending. - writeback_stall = "writeback-stall"; - /// Hold every thread that waits on a watch between reading its condition and /// parking, so a post lands in the window its commit must refuse the park over. watch_window = "watch-window"; @@ -440,7 +437,7 @@ actuators! { /// Give it a present context entry naming an empty second-level table, distinct from an absent context: passthrough would fault identically to the row above. iommu_empty_domain = "iommu-empty-domain"; - /// Answer a claimed function's first DMA grant with an address in another driver's pool, which the claimed function's own domain does not map. + /// Answer a claimed network function's first DMA grant with an address in another driver's pool, which its own domain does not map. iommu_userdev_foreign_dma = "iommu-userdev-foreign-dma"; /// Point a scanout backing at that same page, which the display's own domain does not map. diff --git a/kernel/src/bcachefs_adapter.rs b/kernel/src/bcachefs_adapter.rs index 6d492877567..ae40cf98b62 100644 --- a/kernel/src/bcachefs_adapter.rs +++ b/kernel/src/bcachefs_adapter.rs @@ -9,7 +9,6 @@ use alloc::vec::Vec; use bcachefs::{FsError, Mounted, ReadOnly}; use crate::file_backing::{FileBacking, ReadOnlyBacking}; -use crate::mm::PAGE_BYTES; use crate::file_cache::{self, FileId}; use crate::rootfs::MemoryImage; use toyos_abi::syscall::SyscallError; @@ -101,7 +100,8 @@ impl FileSystem for ReadOnlyBcacheFsAdapter { return Ok((file_id, Some(backing))); } - let file_id = file_cache::create_file(true); + // Its mtime is the volume's (`file_mtime`), never the cache's. + let file_id = file_cache::create_file(true, 0); file_cache::set_size(file_id, size); self.name_to_id.insert(String::from(name), file_id); @@ -114,16 +114,6 @@ impl FileSystem for ReadOnlyBcacheFsAdapter { Err(SyscallError::PermissionDenied) } - fn close_file(&mut self, file_id: FileId) { - let name = self.name_to_id.iter() - .find(|(_, &v)| v == file_id) - .map(|(k, _)| k.clone()); - if let Some(name) = name { - self.name_to_id.remove(&name); - } - } - - /// `PermissionDenied`, not `Io`: retrying this write is never right, unlike a device retry. fn delete(&mut self, _name: &str) -> Result<(), SyscallError> { Err(SyscallError::PermissionDenied) } @@ -142,33 +132,13 @@ impl FileSystem for ReadOnlyBcacheFsAdapter { Err(SyscallError::NotSupported) } - fn write_page(&mut self, _file_id: FileId, _page_idx: u32, _data: &[u8; PAGE_BYTES]) -> Result<(), SyscallError> { - Err(SyscallError::PermissionDenied) - } - - fn update_metadata(&mut self, _file_id: FileId, _size: u64, _mtime: u64) -> Result<(), SyscallError> { - Err(SyscallError::PermissionDenied) - } - - fn truncate_to(&mut self, _file_id: FileId, _size: u64, _mtime: u64) -> Result<(), SyscallError> { - Err(SyscallError::PermissionDenied) - } - fn create_symlink(&mut self, _name: &str, _target: &str) -> Result<(), SyscallError> { Err(SyscallError::PermissionDenied) } - fn sync(&mut self) -> Result<(), SyscallError> { - Ok(()) - } - fn open_backing(&mut self, name: &str) -> Result, SyscallError> { let (extents, size) = present("open_backing", name, self.fs.file_extents(name))?; Ok(Arc::new(ReadOnlyBacking::new(*self.fs.io(), extents, size))) } - - fn cached_file_id(&mut self, name: &str) -> Option { - self.name_to_id.get(name).copied() - } } diff --git a/kernel/src/block.rs b/kernel/src/block.rs index ac8e5db1c4f..f7d0a03a4af 100644 --- a/kernel/src/block.rs +++ b/kernel/src/block.rs @@ -95,29 +95,6 @@ pub(crate) fn between_attempts(attempt: u32) { ); } -/// The running thread is inside a filesystem update whose attempts park in -/// [`between_attempts`]: a refused attempt leaves the volume half written and -/// only this thread's next one completes it, so the machine's stop leaves it -/// running until the guard drops instead of banding it where it parks. -#[must_use = "the update lasts exactly as long as this guard"] -pub struct OpenUpdate(Option>); - -pub fn begin_update() -> OpenUpdate { - let shared = crate::sched::driver::current_shared(); - if let Some(shared) = &shared { - shared.begin_update(); - } - OpenUpdate(shared) -} - -impl Drop for OpenUpdate { - fn drop(&mut self) { - if let Some(shared) = &self.0 { - shared.end_update(); - } - } -} - /// Operations open right now on a thread the machine's stop stops, and how /// many such operations this boot began. static OPEN_OPERATIONS: core::sync::atomic::AtomicU32 = core::sync::atomic::AtomicU32::new(0); diff --git a/kernel/src/durability.rs b/kernel/src/durability.rs deleted file mode 100644 index fcc6f75d9a0..00000000000 --- a/kernel/src/durability.rs +++ /dev/null @@ -1,59 +0,0 @@ -//! Durability debt, counted in write generations — one rule at three sites -//! (a page's dirty state, a file's flush, a mount's device commit): debt is -//! discharged only by presenting a [`Settlement`] minted before the work that -//! discharged it, so "success recorded before the device committed" and -//! "a clear that erases a later write" are both unwritable, not just absent. -//! `kernel-loom/tests/durability.rs` drives this file over every interleaving. - -/// What a durability site owes: writes recorded against writes settled. -pub struct Owed { - written: u64, - settled: u64, -} - -/// The only value [`Owed::settle`] accepts; minted by [`Owed::snapshot`], so a -/// discharge is always bounded by a state observed before the work began. -#[derive(Clone, Copy)] -pub struct Settlement(u64); - -impl Owed { - pub const fn new() -> Self { - Self { written: 0, settled: 0 } - } - - /// Record one write; the debt stands until a settlement minted at or after - /// this call is presented. - pub fn record_write(&mut self) { - self.written += 1; - } - - pub fn is_owed(&self) -> bool { - self.written != self.settled - } - - #[must_use = "an unpresented settlement discharges nothing"] - pub fn snapshot(&self) -> Settlement { - Settlement(self.written) - } - - /// Discharge every write the settlement covers; a write recorded after its - /// mint stays owed. `settled` never passes `written`: the settlement was - /// minted from a `written` that only grows. - #[cfg(not(feature = "durability-settle-blind"))] - pub fn settle(&mut self, upto: Settlement) { - self.settled = self.settled.max(upto.0); - } - - /// The pre-generation kernel's shape — a blind clear — compiled only by - /// `kernel-loom`'s negative control, where `tests/durability.rs` must red. - #[cfg(feature = "durability-settle-blind")] - pub fn settle(&mut self, _upto: Settlement) { - self.settled = self.written; - } -} - -impl Default for Owed { - fn default() -> Self { - Self::new() - } -} diff --git a/kernel/src/file_cache.rs b/kernel/src/file_cache.rs index 351f4023a83..3e30efea39a 100644 --- a/kernel/src/file_cache.rs +++ b/kernel/src/file_cache.rs @@ -1,11 +1,9 @@ use alloc::boxed::Box; use alloc::collections::btree_map::Entry; use alloc::collections::BTreeMap; -use alloc::collections::BTreeSet; use alloc::sync::Arc; use crate::block; -use crate::durability::{Owed, Settlement}; use crate::file_backing::FileBacking; use crate::sync::Lock; use crate::user_ptr::{ByteSource, UserBytesMut}; @@ -21,18 +19,10 @@ pub const MAX_FILE_SIZE: u64 = (u32::MAX as u64 + 1) * PAGE_SIZE as u64; struct CachedPage { data: Box<[u8; PAGE_SIZE]>, - /// Dirty state as generations, settled only against what a flush copied. - dirt: Owed, /// CLOCK's second-chance bit: set on every hit, cleared when the sweep passes it over. referenced: bool, } -impl CachedPage { - fn is_dirty(&self) -> bool { - self.dirt.is_owed() - } -} - struct CachedFile { pages: BTreeMap, size: u64, @@ -41,14 +31,8 @@ struct CachedFile { backing: Option>, ref_count: u32, deleted: bool, - /// Pins the file alive at `ref_count == 0` for the write-back queue; cleared only by [`finish_writeback`]. - teardown_owed: bool, - /// The file, not any one handle, owes a flush; settled only by [`settle_file`] on a flush that succeeded. - dirt: Owed, - /// The smallest size a shrink has taken this file to since the last settled - /// metadata write; `None` when none has. Everything at or above it was - /// discarded, so a page missing from here is zeros and never the backing's. - shrunk_to: Option, + /// When it was last written, for a mount whose pages are the file. + mtime: u64, } impl CachedFile { @@ -67,8 +51,6 @@ struct FileCache { evictions: u64, /// CLOCK hand, in (file, page) key order; kept across calls so eviction costs one step, not a full scan. hand: (FileId, u32), - /// The over-budget state has been said; cleared when residency returns within budget, so an episode costs one line. - over_said: bool, } static FILE_CACHE: Lock = Lock::new(FileCache { @@ -79,7 +61,6 @@ static FILE_CACHE: Lock = Lock::new(FileCache { max_pages: 0, evictions: 0, hand: (0, 0), - over_said: false, }); /// Install the memory budget; must run after the PMM sizes RAM and before any file is opened. @@ -90,7 +71,7 @@ pub fn init() { } /// Allocate a new FileId. The file cache is the sole allocator. -pub fn create_file(evictable: bool) -> FileId { +pub fn create_file(evictable: bool, mtime: u64) -> FileId { let mut cache = FILE_CACHE.lock(); let id = cache.next_id; cache.next_id += 1; @@ -101,9 +82,7 @@ pub fn create_file(evictable: bool) -> FileId { backing: None, ref_count: 1, deleted: false, - teardown_owed: false, - dirt: Owed::new(), - shrunk_to: None, + mtime, }); id } @@ -122,15 +101,9 @@ pub fn set_backing(file_id: FileId, backing: Arc) { evict_if_needed(&mut cache); } -/// Whether an evicted page of this file could be read back; false for tmpfs and for a disk file with no blocks yet. -pub fn has_backing(file_id: FileId) -> bool { - FILE_CACHE.lock().files.get(&file_id).is_some_and(|f| f.backing.is_some()) -} - /// Increment ref_count for one more open, returning a guard that undoes the /// increment on drop unless committed: a re-open whose backing lookup fails -/// after this must not pin the file. Caller holds the VFS lock, which -/// [`finish_writeback`] also reads `ref_count` under to serialise against teardown. +/// after this must not pin the file. #[must_use = "commit() once the re-open cannot fail, or the reference is released"] pub fn open(file_id: FileId) -> crate::rollback::Rollback { { @@ -142,7 +115,7 @@ pub fn open(file_id: FileId) -> crate::rollback::Rollback { crate::rollback::Rollback::new(move || undo_open(file_id)) } -/// Undo one [`open`]: a re-open decrements a count another handle or a pending teardown keeps, never orphaning the file. +/// Undo one [`open`]: a re-open decrements a count another handle keeps, never orphaning the file. fn undo_open(file_id: FileId) { let mut cache = FILE_CACHE.lock(); if let Some(file) = cache.files.get_mut(&file_id) { @@ -150,83 +123,28 @@ fn undo_open(file_id: FileId) { } } -/// The verdict [`release_to_writeback`] hands its caller. -#[derive(Clone, Copy, PartialEq, Eq)] -pub enum Release { - /// Other handles still hold the file. Nothing is owed. - StillHeld, - /// This was the last handle: the file is pinned for write-back and the caller must [`crate::writeback::enqueue`] it. - TeardownOwed, - /// This was the last handle, but a teardown was already owed and enqueued; nothing to enqueue. - AlreadyOwed, -} - -/// Drop one open reference; if it was the last, pins the file for write-back instead of dropping it here — eviction never takes a dirty page, so a re-open before the drain reads the pinned data, not the device. -pub fn release_to_writeback(file_id: FileId) -> Release { +/// Drop one open reference. The last of a deleted file takes its pages with +/// it; any other file stays, its pages clean and CLOCK's to evict, so a +/// re-open finds the id its mount keeps for the name. +pub fn release(file_id: FileId) { let mut cache = FILE_CACHE.lock(); - let Some(file) = cache.files.get_mut(&file_id) else { return Release::StillHeld }; + let Some(file) = cache.files.get_mut(&file_id) else { return }; file.ref_count = file.ref_count.saturating_sub(1); - if file.ref_count != 0 { - return Release::StillHeld; - } - if file.teardown_owed { - return Release::AlreadyOwed; + if file.ref_count == 0 && file.deleted { + drop_file(&mut cache, file_id); } - file.teardown_owed = true; - Release::TeardownOwed } -/// What [`finish_writeback`] found, and what the drainer does with the filesystem-side handle. -#[derive(Clone, Copy, PartialEq, Eq)] -pub enum Teardown { - /// The file left the cache (or is a live tmpfs file whose pages stay): release the handle with `Vfs::close_file`. - Released, - /// A re-open adopted the file between the enqueue and the drain, so it is left alive with its handle. - Adopted, - /// The file is already gone from the cache. Nothing to close. - Vanished, +/// When the file was last written, as [`create_file`] and [`touch`] said. +pub fn mtime(file_id: FileId) -> u64 { + FILE_CACHE.lock().files.get(&file_id).map_or(0, |f| f.mtime) } -/// What `iod`/shutdown needs to know before flushing a queued file, read in one lock. -#[derive(Clone, Copy)] -pub struct WritebackProbe { - pub flush_owed: bool, - /// A deleted file's drain skips the flush — its data is going away. - pub deleted: bool, -} - -/// Read a queued file's flush state; a file the queue pinned is always present, so absence reads as "nothing to flush" rather than a panic. -pub fn writeback_probe(file_id: FileId) -> WritebackProbe { - let cache = FILE_CACHE.lock(); - match cache.files.get(&file_id) { - Some(file) => WritebackProbe { flush_owed: file.dirt.is_owed(), deleted: file.deleted }, - None => WritebackProbe { flush_owed: false, deleted: false }, - } -} - -/// The FILE_CACHE half of a write-back teardown; must run under the VFS lock, which serialises the drain against a re-open. -pub fn finish_writeback(file_id: FileId) -> Teardown { - let mut cache = FILE_CACHE.lock(); - let Some(file) = cache.files.get_mut(&file_id) else { return Teardown::Vanished }; - if !file.teardown_owed { - // A re-open's own release already cleared it; nothing owed here. - return Teardown::Adopted; - } - file.teardown_owed = false; - if file.ref_count != 0 { - // A re-open adopted the file between the enqueue and now. - return Teardown::Adopted; - } - if file.deleted || file.evictable { - drop_file(&mut cache, file_id); +/// Record a write at `mtime`. +pub fn touch(file_id: FileId, mtime: u64) { + if let Some(file) = FILE_CACHE.lock().files.get_mut(&file_id) { + file.mtime = mtime; } - // else: a live tmpfs file keeps its (non-evictable) pages. - Teardown::Released -} - -/// Whether a shrink discarded this page, so the backing's copy is no longer the file's. -fn discarded(file: &CachedFile, page_idx: u32) -> bool { - file.shrunk_to.is_some_and(|mark| page_idx as u64 * PAGE_SIZE as u64 >= mark) } /// Read a file page into `buf`; on `Err` the fetch failed and `buf` holds zeros, not the file's bytes. @@ -254,7 +172,7 @@ pub fn read_page( copy_page_region_to_buf(&page.data[..], offset, buf, avail); return Ok(()); } - backing = if discarded(file, page_idx) { None } else { file.backing.clone() }; + backing = file.backing.clone(); } // Cache miss: unlock, fetch from backing, re-lock, insert if still absent. @@ -305,8 +223,6 @@ pub fn write_page( if file.pages.contains_key(&page_idx) { apply_write(file, page_idx, offset, data); backing = None; - } else if discarded(file, page_idx) { - backing = Some(None); } else { backing = Some(file.backing.clone()); } @@ -327,7 +243,7 @@ pub fn write_page( } } - // Re-fetching after a sibling's eviction is always correct: only clean pages are ever evicted. + // Re-fetching after a sibling's eviction is always correct: a page written is never evicted. let mut cache = FILE_CACHE.lock(); let mut added = 0; { @@ -353,10 +269,7 @@ fn apply_write( let page = file.pages.get_mut(&page_idx).expect("write_page: page not resident"); let end = (offset + data.len()).min(PAGE_SIZE); data.read_at(0, &mut page.data[offset..end]); - page.dirt.record_write(); page.referenced = true; - // Both recorded under the one FILE_CACHE lock, so a flush never observes one without the other. - file.dirt.record_write(); let write_end = page_idx as u64 * PAGE_SIZE as u64 + end as u64; if write_end > file.size { @@ -364,73 +277,14 @@ fn apply_write( } } -/// Copy a resident page out, with the settlement its flusher must present to -/// mark it clean; `None` leaves `buf` untouched — an absent page is not zeros. -#[must_use] -pub fn copy_page_out(file_id: FileId, page_idx: u32, buf: &mut [u8; PAGE_SIZE]) -> Option { +/// Copy a resident page out; `false` leaves `buf` untouched — an absent page is not zeros. +pub fn copy_page_out(file_id: FileId, page_idx: u32, buf: &mut [u8; PAGE_SIZE]) -> bool { let cache = FILE_CACHE.lock(); - let file = cache.files.get(&file_id)?; - let page = file.pages.get(&page_idx)?; + let Some(page) = cache.files.get(&file_id).and_then(|file| file.pages.get(&page_idx)) else { + return false; + }; *buf = *page.data; - Some(page.dirt.snapshot()) -} - -/// What one flush attempt owes, snapshotted in one lock; nothing is cleared -/// here — a write landing mid-flush outruns the settlement and stays owed. -pub struct FlushPlan { - pub file: Settlement, - pub pages: BTreeSet, - /// What the mount has to give back before this flush's pages land on top of it. - pub shrunk_to: Option, -} - -pub fn begin_flush(file_id: FileId) -> FlushPlan { - let cache = FILE_CACHE.lock(); - match cache.files.get(&file_id) { - Some(file) => FlushPlan { - file: file.dirt.snapshot(), - pages: file.pages.iter().filter(|(_, p)| p.is_dirty()).map(|(&i, _)| i).collect(), - shrunk_to: file.shrunk_to, - }, - None => FlushPlan { - file: Owed::new().snapshot(), - pages: BTreeSet::new(), - shrunk_to: None, - }, - } -} - -/// Whether a file owes a write-back; `fsync` reads this so a handle that did not itself write still flushes a file another handle dirtied. -pub fn flush_owed(file_id: FileId) -> bool { - FILE_CACHE.lock().files.get(&file_id).is_some_and(|f| f.dirt.is_owed()) -} - -/// Settle each page up to what its flush copied; a page written since keeps its debt and the next flush delivers it. -pub fn settle_pages(file_id: FileId, flushed: &[(u32, Settlement)]) { - let mut cache = FILE_CACHE.lock(); - if let Some(file) = cache.files.get_mut(&file_id) { - for (page_idx, copied) in flushed { - if let Some(page) = file.pages.get_mut(page_idx) { - page.dirt.settle(*copied); - } - } - } -} - -/// Settle the file's flush debt; only a flush that wrote its pages and its metadata calls this. -pub fn settle_file(file_id: FileId, upto: Settlement) { - if let Some(file) = FILE_CACHE.lock().files.get_mut(&file_id) { - file.dirt.settle(upto); - } -} - -/// Clear the shrink mark: the mount has given the tail back, so nothing above it -/// is named any more and a second trim would free what the flush then writes. -/// Sound without a generation because every shrink runs under the VFS lock a flush holds. -pub fn settle_shrink(file_id: FileId) { - if let Some(file) = FILE_CACHE.lock().files.get_mut(&file_id) { - file.shrunk_to = None; - } + true } /// Get the authoritative file size. @@ -438,101 +292,10 @@ pub fn size(file_id: FileId) -> u64 { FILE_CACHE.lock().files.get(&file_id).map_or(0, |f| f.size) } -/// Set file size and drop pages past it; the establishing form (mount-time), does not mark the file dirty — see [`resize`] for a user truncate. +/// Set file size and drop pages past it. pub fn set_size(file_id: FileId, new_size: u64) { let mut cache = FILE_CACHE.lock(); set_size_locked(&mut cache, file_id, new_size); - if let Some(file) = cache.files.get_mut(&file_id) { - file.shrunk_to = None; - } -} - -/// A user truncate: [`set_size`], plus marks the file dirty even when no page changed. -/// The `&mut Vfs` is the witness: every flusher's size-read/`update_metadata` -/// pair runs under the VFS lock, so a resize outside it could record a stale size. -/// `Err` is the straddled page unreadable, with nothing resized. -pub fn resize( - _vfs: &mut crate::vfs::Vfs, - file_id: FileId, - new_size: u64, -) -> Result<(), block::BlockError> { - // Fetched outside the lock because it is a device read, and admitted below - // under the same hold that spends it: a page admitted and released again is - // clean and unreferenced, which is what a sweep takes at or after its hand — - // and `sys_read`/`sys_write` sweep holding no VFS lock, so they run here. - let fetched = match straddled_by_shrink(file_id, new_size) { - Some(page_idx) => fetch_page(file_id, page_idx)?.map(|page| (page_idx, page)), - None => None, - }; - - let mut cache = FILE_CACHE.lock(); - if let Some((page_idx, page)) = fetched { - admit_locked(&mut cache, file_id, page_idx, page); - } - let shrank = cache.files.get(&file_id).is_some_and(|f| new_size < f.size); - set_size_locked(&mut cache, file_id, new_size); - if let Some(file) = cache.files.get_mut(&file_id) { - file.dirt.record_write(); - if shrank { - file.shrunk_to = Some(file.shrunk_to.map_or(new_size, |mark| mark.min(new_size))); - } - } - // Safe on the page just admitted: `set_size_locked` dirtied it, and - // `evict_one` never takes a dirty page. - evict_if_needed(&mut cache); - Ok(()) -} - -/// The page a shrink to `new_size` would cut in half and does not hold. -/// -/// [`set_size_locked`] can only zero and dirty a *resident* page, and -/// [`discarded`] cannot cover this one: it is partly still the file's. -fn straddled_by_shrink(file_id: FileId, new_size: u64) -> Option { - if new_size.is_multiple_of(PAGE_SIZE as u64) { - return None; - } - let idx = (new_size / PAGE_SIZE as u64) as u32; - let cache = FILE_CACHE.lock(); - let file = cache.files.get(&file_id)?; - (new_size < file.size && !file.pages.contains_key(&idx) && file.backing.is_some()) - .then_some(idx) -} - -/// Read `page_idx` through the backing, holding no lock across the device. -/// `None` is nothing to admit: the page arrived, or the mark now covers it. -fn fetch_page( - file_id: FileId, - page_idx: u32, -) -> Result>, block::BlockError> { - let backing = { - let cache = FILE_CACHE.lock(); - let Some(file) = cache.files.get(&file_id) else { return Ok(None) }; - if file.pages.contains_key(&page_idx) || discarded(file, page_idx) { - return Ok(None); - } - file.backing.clone() - }; - let mut fetched = blank_page(); - if let Some(backing) = &backing { - backing.read_page(page_idx as u64 * PAGE_SIZE as u64, &mut fetched)?; - } - Ok(Some(fetched)) -} - -/// Insert a fetched page if the file still wants it, under the caller's lock. -/// Re-checked here because the fetch above released the lock across a device read. -fn admit_locked(cache: &mut FileCache, file_id: FileId, page_idx: u32, page: Box<[u8; PAGE_SIZE]>) { - let mut added = 0; - if let Some(file) = cache.files.get_mut(&file_id) { - let is_cache = file.is_cache(); - if !discarded(file, page_idx) { - if let Entry::Vacant(slot) = file.pages.entry(page_idx) { - slot.insert(CachedPage::new(page)); - added = usize::from(is_cache); - } - } - } - cache.cached_pages += added; } fn set_size_locked(cache: &mut FileCache, file_id: FileId, new_size: u64) { @@ -548,16 +311,13 @@ fn set_size_locked(cache: &mut FileCache, file_id: FileId, new_size: u64) { file.pages.remove(k); } // The page the new end falls inside is kept; zero its bytes past the - // new end and dirty it, in the one step that sets the size, so a - // later grow reads the hole as zeros rather than the discarded tail - // and the flush carries those zeros to the device. + // new end in the one step that sets the size, so a later grow reads + // the hole as zeros rather than the discarded tail. let tail = (new_size % PAGE_SIZE as u64) as usize; if tail != 0 { let straddled = (new_size / PAGE_SIZE as u64) as u32; if let Some(page) = file.pages.get_mut(&straddled) { page.data[tail..].fill(0); - page.dirt.record_write(); - file.dirt.record_write(); } } if is_cache { removed.len() } else { 0 } @@ -569,32 +329,20 @@ fn set_size_locked(cache: &mut FileCache, file_id: FileId, new_size: u64) { cache.cached_pages -= dropped; } -/// What the cache holds for a file after an operation that may have freed it. -#[derive(Clone, Copy, PartialEq, Eq)] -pub enum Residency { - /// Something still holds it — an open handle, or the write-back queue — so its filesystem handle must survive until that holder's teardown. - Held, - /// The cache holds nothing for this id; a filesystem may drop whatever it keeps alongside. - Gone, -} - -/// Mark a file as deleted (unlink). If no handles hold it, free immediately. -#[must_use] -pub fn mark_deleted(file_id: FileId) -> Residency { +/// Mark a file as deleted (unlink). If no handles hold it, free immediately; +/// otherwise the last [`release`] does. +pub fn mark_deleted(file_id: FileId) { let mut cache = FILE_CACHE.lock(); - let Some(file) = cache.files.get_mut(&file_id) else { return Residency::Gone }; + let Some(file) = cache.files.get_mut(&file_id) else { return }; file.deleted = true; - // A pinned file (teardown_owed) is left marked deleted for `finish_writeback` to drop; `Held` because the cache still holds it. - if file.ref_count > 0 || file.teardown_owed { - return Residency::Held; + if file.ref_count == 0 { + drop_file(&mut cache, file_id); } - drop_file(&mut cache, file_id); - Residency::Gone } impl CachedPage { fn new(data: Box<[u8; PAGE_SIZE]>) -> Self { - Self { data, dirt: Owed::new(), referenced: false } + Self { data, referenced: false } } } @@ -637,47 +385,22 @@ fn copy_page_region_to_buf(page: &[u8], offset: usize, buf: &mut UserBytesMut, v fn evict_if_needed(cache: &mut FileCache) { assert!(cache.max_pages != 0, "file cache used before init installed a budget"); - if cache.cached_pages <= cache.max_pages { - // A teardown can end an over-budget episode between admissions; the closing line still prints. - if cache.over_said { - cache.over_said = false; - turnover_line(cache); - } - return; - } let before = cache.evictions; while cache.cached_pages > cache.max_pages { - if !evict_one(cache) { - // Everything resident is dirty: write-back is the handle layer's job, so nothing here bounds dirty pages further. - break; - } - } - // Once per full turnover so the rate scales with the budget, plus once at - // each over-budget episode's start and end. + // A governed page is a read-only mount's and never written, so one + // revolution always finds one to take. + assert!( + evict_one(cache), + "file cache: {}/{} pages resident and none evictable", + cache.cached_pages, + cache.max_pages + ); + } + // Once per full turnover, so the rate scales with the budget. let turnover = cache.max_pages as u64; - let over = cache.cached_pages > cache.max_pages; - let crossed = - cache.evictions != before && (before == 0 || before / turnover != cache.evictions / turnover); - if crossed || over != cache.over_said { - turnover_line(cache); + if cache.evictions != before && (before == 0 || before / turnover != cache.evictions / turnover) { + log!("file cache: {} evictions, {}/{} pages resident", cache.evictions, cache.cached_pages, cache.max_pages); } - cache.over_said = over; -} - -/// Dirty is on the line because over-budget is lawful exactly when every resident page is dirty; the harness holds that shape. -fn turnover_line(cache: &FileCache) { - log!("file cache: {} evictions, {}/{} pages resident, {} dirty", - cache.evictions, cache.cached_pages, cache.max_pages, dirty_pages(cache)); -} - -/// Governed dirty pages, counted under the same lock hold as the residency they explain. -fn dirty_pages(cache: &FileCache) -> usize { - cache - .files - .values() - .filter(|f| f.is_cache()) - .map(|f| f.pages.values().filter(|p| p.is_dirty()).count()) - .sum() } /// One CLOCK step-and-evict; returns false when a full revolution found no page it was allowed to take. @@ -694,9 +417,6 @@ fn evict_one(cache: &mut FileCache) -> bool { { let Some(file) = cache.files.get_mut(&fid) else { continue }; let Some(page) = file.pages.get_mut(&idx) else { continue }; - if page.is_dirty() { - continue; - } if page.referenced { page.referenced = false; continue; diff --git a/kernel/src/iod.rs b/kernel/src/iod.rs deleted file mode 100644 index 01d4fc246a6..00000000000 --- a/kernel/src/iod.rs +++ /dev/null @@ -1,53 +0,0 @@ -//! Drains the write-back queue and evicts the page cache; `SYS_CLOSE` does not wait on it, `SYS_FSYNC` flushes inline instead. - -use toyos_sched::task::WaitClass; - -use crate::watch; -use crate::sched::kthread::{self, OnPanic}; -use crate::scheduler; -use crate::time::Deadline; - -// Relied on by name in `sched::dump`, `ps`, and crash reports. -const NAME: &str = "iod"; - -/// Spawns the `iod` kthread; call once, from `kernel_main`. -pub fn start() { - // Recoverable: unlike klogd's silent loss, a killed iod's stalled write-back is visible to SYS_FSYNC and logd. - let _ = kthread::spawn(NAME, body, 0, OnPanic::Recover); -} - -extern "C" fn body(_arg: u64) -> ! { - let parkable = scheduler::Parkable::at_entry(); - // Probes the nesting gate here: iod is the one task every boot has, idle at entry. - #[cfg(feature = "boot-actuators")] - if crate::actuator::sched_operation_nesting() { - crate::sched_gate::run("iod"); - } - // The SS-reload probe rides iod rather than spawning a task of its own. - #[cfg(feature = "boot-actuators")] - if crate::actuator::sysret_ss_probe() { - crate::hw::sysret_ss_probe(&parkable); - } - // Held across the loop: a push during a drain must still find this watch armed. - let armed = watch::arm( - &crate::writeback::WORK, - 0, - WaitClass::Io, - ) - .expect("a kernel thread is a task and can arm"); - #[cfg(feature = "boot-actuators")] - if crate::actuator::writeback_stall() { - // Said once, so a test can hold that the queue it stages was held. - crate::log!("iod: writeback-stall: parked for the boot; the write-back queue is not drained"); - loop { - let _ = watch::wait(&parkable, &armed, Deadline::never()); - } - } - loop { - // drain_all_iod, not drain_all: this thread already holds the WORK arm; drain's backoff parks on it instead of arming a second. - crate::writeback::drain_all_iod(&parkable, &armed); - // No deadline: only a push should end this wait; a periodic wake would need audio sign-off. - // Discarded: nothing retires a kernel thread, so this wait never reports Cancelled. - let _ = watch::wait(&parkable, &armed, Deadline::never()); - } -} diff --git a/kernel/src/main.rs b/kernel/src/main.rs index 7a330a9b6d3..a90d46b6618 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -45,7 +45,6 @@ mod usb_gate; #[cfg(feature = "boot-actuators")] mod sched_gate; mod block; -mod durability; mod gpt; mod inventory; mod rollback; @@ -55,7 +54,6 @@ mod file_cache; mod leak_selftest; #[cfg(feature = "boot-actuators")] mod revoke_selftest; -mod writeback; mod tmpfs; mod file_backing; mod bcachefs_adapter; @@ -78,7 +76,6 @@ mod time; mod clock; mod watch; -mod iod; mod object; mod inbox; mod pipe; @@ -591,7 +588,10 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { log::console::start(); // After klogd so their own spawn logs have a drainer. drivers::xhci::usbd::start(); - iod::start(); + #[cfg(feature = "boot-actuators")] + if actuator::sched_operation_nesting() || actuator::sysret_ss_probe() { + let _ = sched::kthread::spawn("probe", probe, 0, sched::kthread::OnPanic::Recover); + } smp::set_ready(); @@ -615,3 +615,17 @@ fn pre_idle_wedge() -> ! { core::hint::spin_loop(); } } + +/// The actuators judged on a kernel thread of its own, idle at its entry: its +/// task word, and a switch away from it. Spawned only when one is armed. +#[cfg(feature = "boot-actuators")] +extern "C" fn probe(_arg: u64) -> ! { + let parkable = scheduler::Parkable::at_entry(); + if actuator::sched_operation_nesting() { + sched_gate::run("probe"); + } + if actuator::sysret_ss_probe() { + hw::sysret_ss_probe(&parkable); + } + watch::park_forever() +} diff --git a/kernel/src/object/file.rs b/kernel/src/object/file.rs index 79f6042628d..df7b35a788a 100644 --- a/kernel/src/object/file.rs +++ b/kernel/src/object/file.rs @@ -1,27 +1,22 @@ //! An open file: two handles to one `FileObject` share the cursor; an independent cursor opens the path again. -//! `SYS_CLOSE` does not report write-back errors — a durability claim must go through `SYS_FSYNC`. -use alloc::string::String; use alloc::sync::Arc; -use crate::file_cache::{self, FileId, Release}; +use crate::file_cache::{self, FileId}; use crate::sync::Lock; use super::{KObjectVariant, ObjectCore}; pub struct OpenFileState { - pub path: String, pub file_id: FileId, pub position: usize, pub mtime: u64, } -// Drop runs under `Lock` and cannot take a sleep lock or wait on a device, so it enqueues to writeback instead of flushing. +// Drop runs under `Lock`, so it takes the file cache's lock and no other. impl Drop for OpenFileState { fn drop(&mut self) { - if let Release::TeardownOwed = file_cache::release_to_writeback(self.file_id) { - crate::writeback::enqueue(self.file_id, core::mem::take(&mut self.path), self.mtime); - } + file_cache::release(self.file_id); } } diff --git a/kernel/src/object/ops.rs b/kernel/src/object/ops.rs index e48bda71f1f..6786ec80753 100644 --- a/kernel/src/object/ops.rs +++ b/kernel/src/object/ops.rs @@ -135,15 +135,14 @@ pub fn open(table: &mut HandleTable, path: &str, flags: OpenFlags) -> u64 { Err(e) => Err(e), } }; - built.map(|(file_id, mtime, position)| (target, file_id, mtime, position)) + built }; - let (target, file_id, mtime, position) = match opened { + let (file_id, mtime, position) = match opened { Ok(v) => v, Err(e) => return e.to_u64(), }; let object = KObjectRef::File(FileObject::new(OpenFileState { - path: target.into_string(), file_id, position, mtime, @@ -522,8 +521,8 @@ pub fn try_write(object: &KObjectRef, buf: &UserBytes) -> Option { return Some(SyscallError::Io.to_u64()); } state.position += written; - // Dirty state lives in the cache now, set by `write_page`; the handle keeps only the mtime. state.mtime = crate::clock::nanos_since_boot(); + file_cache::touch(state.file_id, state.mtime); Some(written as u64) }), KObjectRef::PipeWrite(w) => write_pipe(w.id(), buf), @@ -606,56 +605,14 @@ pub fn fstat(object: &KObjectRef) -> Stat { } } -/// `SYS_FSYNC`: the file's bytes on the device, and the device told to commit them. -/// -/// The device-commit step is not optional: `/system/bin/logd` calls a line durable off `fsync`'s result, so a flush that stopped at the page cache would make that a claim about nothing. +/// `SYS_FSYNC`: a partition claim's writes on its device and the device told to +/// commit them; a kernel file's pages are the file (`/tmp`) or never written +/// (ROOT), so it owes nothing. pub fn fsync(object: &KObjectRef) -> u64 { - let file = match object { - KObjectRef::File(file) => file, - KObjectRef::Device(claim) => return partition_fsync(claim), - _ => return SyscallError::PermissionDenied.to_u64(), - }; - let (path, file_id, mtime) = - file.with(|state| (state.path.clone(), state.file_id, state.mtime)); - // The file's debt or its mount's, not the handle's: another handle's write, and a - // device commit an earlier attempt failed to deliver, are both still owed here. - if !crate::vfs::lock().durability_owed(&path, file_id) { - return 0; - } - // A refused attempt can leave the two FATs split, and the park between two attempts is where the machine's stop would find this thread. - let _update = crate::block::begin_update(); - // A refused attempt discards nothing — an unsettled debt needs no restoring. - let run = until_answered(|| Run::Fsync(file_id), || { - // Outside `FileObject`'s lock: this and `OpenFileState::drop` take the VFS lock in the same order. - // Flush and sync share one acquisition so this file cannot be unmounted between them. - let mut vfs = crate::vfs::lock(); - let done = vfs - .flush_file(&path, file_id, mtime) - .and_then(|()| vfs.sync_for_path(&path)); - drop(vfs); - done - }); - match run { - Answered::Answer { answer: Ok(()), attempts, took } => { - if attempts > 1 { - crate::log!( - "fsync: {path} durable on attempt {attempts} after {took} — a refused \ - attempt kept every page dirty and a later one delivered them", - ); - } - // `flush_file` settled the file's debt and `sync_for_path` the mount's; there is no per-handle flag to clear. - 0 - } - // The device's own word (an error status, or a recovery that gave up) is passed through unchanged. - Answered::Answer { answer: Err(e), .. } => e.to_u64(), - Answered::Killed => SyscallError::WouldBlock.to_u64(), - Answered::Deadman { attempts, took } => { - crate::log!( - "fsync: {path} is not durable after {attempts} attempt(s) in {took} — {}", - crate::block::DEADMAN, - ); - SyscallError::Io.to_u64() - } + match object { + KObjectRef::File(_) => 0, + KObjectRef::Device(claim) => partition_fsync(claim), + _ => SyscallError::PermissionDenied.to_u64(), } } @@ -676,8 +633,6 @@ pub(crate) enum Answered { #[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] #[cfg_attr(not(feature = "boot-actuators"), allow(dead_code))] pub(crate) enum Run { - /// `SYS_FSYNC` on one file. - Fsync(file_cache::FileId), /// One kind of transfer on one claimed partition; `None` for a claim /// whose partition is already let go, whose every attempt answers `Gone`. Claim(Option<(crate::block::DeviceId, [u8; 16])>, ClaimOp), @@ -806,23 +761,11 @@ pub fn ftruncate(object: &KObjectRef, size: u64) -> u64 { if size > file_cache::MAX_FILE_SIZE { return SyscallError::InvalidArgument.to_u64(); } - let file_id = file.with(|state| state.file_id); - { - // The VFS lock outside `FileObject`'s (fsync's order) is `resize`'s witness. - let mut vfs = crate::vfs::lock(); - // A refused resize changed nothing, so the size stays as it was. - // A budget expiry is the caller's own bound and not a fact about the device: retryable. - if let Err(e) = file_cache::resize(&mut vfs, file_id, size) { - return match e { - crate::block::BlockError::BudgetExpired => SyscallError::WouldBlock, - crate::block::BlockError::Device => SyscallError::Io, - } - .to_u64(); - } - } // The seek pointer is not touched (POSIX ftruncate): a shrink leaves it past EOF. file.with(|state| { + file_cache::set_size(state.file_id, size); state.mtime = crate::clock::nanos_since_boot(); + file_cache::touch(state.file_id, state.mtime); 0 }) } diff --git a/kernel/src/pcidev/mod.rs b/kernel/src/pcidev/mod.rs index 0157506380c..124d37fba16 100644 --- a/kernel/src/pcidev/mod.rs +++ b/kernel/src/pcidev/mod.rs @@ -1660,7 +1660,7 @@ pub fn dma_alloc( // After the mapping and never before: the first thing this function may // reach has to exist before it may reach anything. bound.start_mastering(); - Ok((memory, foreign_if_armed(first, at), span)) + Ok((memory, foreign_if_armed(first, &bound.pci, at), span)) }) } @@ -1756,15 +1756,17 @@ pub fn dma_unmap(slot: usize, at: u64) -> Result<(), SyscallError> { Ok(()) } -/// The address a grant answers with, or — for a claim's first grant, with the -/// actuator armed — another driver's pool. +/// The address a grant answers with, or — for a network function's first +/// grant, with the actuator armed — another driver's pool. /// /// The grant is real and mapped; only the address the driver is *told* is one /// this function's domain does not have, so what the device is pointed at is a /// wrong descriptor rather than a driver written to misbehave. -fn foreign_if_armed(first: bool, at: u64) -> u64 { +fn foreign_if_armed(first: bool, pci: &PciDevice, at: u64) -> u64 { + // A network function's alone: the staging is netd's, and any other claim + // it met would fail beside it and give its slot up. #[cfg(feature = "boot-actuators")] - if first && crate::actuator::iommu_userdev_foreign_dma() { + if first && crate::actuator::iommu_userdev_foreign_dma() && pci.matches_class(0x02, 0x00, None) { let foreign = crate::drivers::xhci::FOREIGN_PROBE.load(core::sync::atomic::Ordering::Relaxed); if foreign != 0 { @@ -1772,7 +1774,7 @@ fn foreign_if_armed(first: bool, at: u64) -> u64 { } } #[cfg(not(feature = "boot-actuators"))] - let _ = first; + let _ = (first, pci); at } diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 06e76564df0..e4b2dcfd9e2 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -20,12 +20,10 @@ //! **Tasks stop; CPUs do not.** Every CPU keeps `IF` set, keeps taking its //! LAPIC timer and every device interrupt, and keeps taking scheduler passes — //! it simply has no userland left to dispatch. That is what the USB stop below -//! the boot's last word needs, and what the kernel threads that carry the -//! sync to its volumes need. Freezing CPUs inside a pass instead would strand -//! whatever lock the thread on that CPU was holding, and `sync_all` is the -//! first thing that would wait on it. +//! the boot's last word needs. Freezing CPUs inside a pass instead would strand +//! whatever lock the thread on that CPU was holding. //! -//! Kernel threads are exempt by identity, not by accident: `klogd`, `iod` and +//! Kernel threads are exempt by identity, not by accident: `klogd` and //! `usbd` are in the process table like anything else, and //! [`crate::sched::kthread::is_kernel_task`] is what tells them apart. //! @@ -56,10 +54,8 @@ use crate::time::{Budget, Deadline, Duration}; /// **A budget, and not a bound the kernel can prove.** One `QUANTUM_NS` is /// what a thread running in Ring 3 needs to reach the boundary, and one /// `block::OPERATION` is the longest a thread lies inside the block layer -/// without parking — but one syscall may open several operations in a row, -/// parking between them inside a `block::OpenUpdate` this stop waits out, and -/// `block::DEADMAN` is what bounds that sequence. A thread can therefore -/// outlast this, which is why its expiry is a clause in the record. +/// without parking. A thread can therefore outlast this, which is why its +/// expiry is a clause in the record. pub(crate) const PARK: Budget = Budget::of( Duration::from_nanos(toyos_sched::fair::QUANTUM_NS + crate::block::OPERATION.nanos()), "the reset lands wherever the threads that never reached a safe point are, and \ @@ -99,8 +95,7 @@ fn stops(stage: u32) -> bool { return false; }; // A kernel thread reaches this boundary on its first dispatch and has no - // Ring 3 to be stopped from; `iod` is also what carries the sync to its - // volume while userland is being stopped around it. + // Ring 3 to be stopped from. if crate::sched::kthread::is_kernel_task(TaskId(pid, tid)) { return false; } @@ -220,9 +215,9 @@ fn sweep(caller: ThreadId) -> Sweep { let sched = thread .sched() .expect("quiesce::sweep: a live thread in the table with no task"); - // `stop_if_blocked` refuses a running thread and one parked inside - // a `block::OpenUpdate`: each has to reach its own safe point, and - // until it does it is what this sweep is waiting for. + // `stop_if_blocked` refuses a running thread: it has to reach its + // own safe point, and until it does it is what this sweep is + // waiting for. if sched.shared.stop_pending() || sched.shared.stop_if_blocked() { out.stopped += 1; } else { diff --git a/kernel/src/revoke_selftest.rs b/kernel/src/revoke_selftest.rs index 7de9f2fea14..f4336df9680 100644 --- a/kernel/src/revoke_selftest.rs +++ b/kernel/src/revoke_selftest.rs @@ -26,11 +26,8 @@ fn probe(path: &str) { if file_cache::write_page(id, 0, 0, &[FILL; 64][..]).is_err() { return fail(path, "write"); } - // Released the way `OpenFileState::drop` does, then drained. - if let file_cache::Release::TeardownOwed = file_cache::release_to_writeback(id) { - crate::writeback::enqueue(id, alloc::string::String::from(path), mtime); - } - crate::writeback::drain_all(); + // Released the way `OpenFileState::drop` does. + file_cache::release(id); let backing = match vfs::lock().open_backing(path) { Ok(b) => b, Err(e) => return fail(path, &alloc::format!("open_backing: {e:?}")), diff --git a/kernel/src/sched/kthread.rs b/kernel/src/sched/kthread.rs index 2faf66b2ca9..f6cd1a87fb5 100644 --- a/kernel/src/sched/kthread.rs +++ b/kernel/src/sched/kthread.rs @@ -19,9 +19,9 @@ use crate::sync::Lock; use super::payload::ThreadSched; -/// `klogd`, `usbd` and `iod`, plus one `log-storm` thread per shard in the actuator build. +/// `klogd` and `usbd`, plus `probe` and one `log-storm` thread per shard in the actuator build. #[cfg(not(feature = "boot-actuators"))] -const MAX_KERNEL_TASKS: usize = 3; +const MAX_KERNEL_TASKS: usize = 2; #[cfg(feature = "boot-actuators")] const MAX_KERNEL_TASKS: usize = 3 + toyos_abi::log::MAX_LOG_SHARDS; diff --git a/kernel/src/sched_gate.rs b/kernel/src/sched_gate.rs index 1f4c2157889..0f4460ea971 100644 --- a/kernel/src/sched_gate.rs +++ b/kernel/src/sched_gate.rs @@ -1,6 +1,6 @@ //! In-guest gate on [`Operation`] nesting: `begin` narrows the current //! deadline and never widens it, and `Drop` restores what it displaced. -//! [`run`] is called from a boot-phase CPU context and from the `iod` task, +//! [`run`] is called from a boot-phase CPU context and from the `probe` task, //! the two places a deadline slot can be established. use crate::clock; diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index ec0f4c03c26..47b6b39ce4d 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -336,8 +336,8 @@ pub fn block_on(ticket: Ticket, deadline: Deadline) { /// Give the CPU up voluntarily, keeping the claim on it: the pass decides /// whether anything else deserves the quantum. Asserts the calling context's -/// own baseline, not a flat trap level, since a kernel thread (`iod`'s -/// write-back retry) yields at zero and a flat assert would panic it. +/// own baseline, not a flat trap level, since a kernel thread yields at zero +/// and a flat assert would panic it. #[track_caller] pub fn yield_now() { assert_baseline(blocking_baseline()); diff --git a/kernel/src/syscall/machine.rs b/kernel/src/syscall/machine.rs index 03228906b4c..1afe563d3f9 100644 --- a/kernel/src/syscall/machine.rs +++ b/kernel/src/syscall/machine.rs @@ -86,11 +86,8 @@ fn quiesce(last: &str) -> Result<(), SyscallError> { } // First: what follows outlasts a feed cadence, and no pass runs to feed again. crate::arch::watchdog::disarm(); - // **Before the sync, because the sync is a claim about a machine.** A - // process that issues a `write` after `sync_all` returns has dirty pages - // nothing will flush. Every userland thread stops here, the log's writer - // with the rest: `/system/bin/init` had it flush before it asked for this - // stop. + // Every userland thread stops here, the log's writer with the rest: + // `/system/bin/init` had it flush before it asked for this stop. #[cfg(feature = "boot-actuators")] crate::quiesce::last::await_the_held_thread(); let stopped = crate::quiesce::stop(); @@ -105,10 +102,6 @@ fn quiesce(last: &str) -> Result<(), SyscallError> { if crate::actuator::quiesce_dump() { crate::sched::dump::serve_for_the_stop(); } - log!("Syncing filesystems..."); - // drain_all before sync_all: a closed-but-undrained file's dirty pages are only in the cache, which sync_all would miss. - crate::writeback::drain_all(); - crate::vfs::lock().sync_all(); // The final census: no process runs after this to report another. crate::irq_census::log_census(); crate::drivers::panic_console::log_census(); diff --git a/kernel/src/tmpfs.rs b/kernel/src/tmpfs.rs index bfa415616c8..0b06d83d89d 100644 --- a/kernel/src/tmpfs.rs +++ b/kernel/src/tmpfs.rs @@ -24,7 +24,7 @@ impl FileBacking for TmpfsBacking { // copy_page_out, not file_cache::read_page: reading through the miss path here would recurse. // A hole below the file size, left by a seek-and-write, reads as zero. if file_offset >= file_cache::size(self.file_id) - || file_cache::copy_page_out(self.file_id, (file_offset / 4096) as u32, buf).is_none() + || !file_cache::copy_page_out(self.file_id, (file_offset / 4096) as u32, buf) { // After the miss: `retire` clears the flag before the pages drop. if !self.alive.load(Ordering::Acquire) { @@ -44,14 +44,14 @@ impl FileBacking for TmpfsBacking { /// Ends a file entry's life: cleared first (so every backing fails), pages after. fn retire(id: FileId, alive: &AtomicBool) { alive.store(false, Ordering::Release); - let _ = file_cache::mark_deleted(id); + file_cache::mark_deleted(id); } /// One name's one entry — a file or a symlink, never both. A single map keys /// every name once, so `create`, `create_symlink`, `rename` and `delete` cannot /// each see a different namespace. pub(crate) enum Entry { - File { id: FileId, mtime: u64, alive: Arc }, + File { id: FileId, alive: Arc }, Symlink { target: String }, } @@ -130,7 +130,7 @@ impl FileSystem for TmpFs { fn file_mtime(&mut self, name: &str) -> Result { match self.entries.get(name) { - Some(Entry::File { mtime, .. }) => Ok(*mtime), + Some(Entry::File { id, .. }) => Ok(file_cache::mtime(*id)), _ => Err(SyscallError::NotFound), } } @@ -157,16 +157,11 @@ impl FileSystem for TmpFs { return Ok(*id); } // A dangling symlink of this name is displaced: one name, one entry. - let id = file_cache::create_file(false); // non-evictable - self.entries - .insert(String::from(name), Entry::File { id, mtime, alive: Arc::new(AtomicBool::new(true)) }); + let id = file_cache::create_file(false, mtime); // non-evictable + self.entries.insert(String::from(name), Entry::File { id, alive: Arc::new(AtomicBool::new(true)) }); Ok(id) } - fn close_file(&mut self, _file_id: FileId) { - // No-op: tmpfs pages already persist in the non-evictable file cache. - } - fn delete(&mut self, name: &str) -> Result<(), SyscallError> { match self.entries.remove(name) { Some(Entry::File { id, alive, .. }) => { @@ -192,27 +187,6 @@ impl FileSystem for TmpFs { Err(SyscallError::NotSupported) } - fn write_page(&mut self, _file_id: FileId, _page_idx: u32, _data: &[u8; PAGE_BYTES]) -> Result<(), SyscallError> { - Ok(()) // tmpfs: data is already in the file cache (canonical storage) - } - - fn update_metadata(&mut self, file_id: FileId, _size: u64, mtime: u64) -> Result<(), SyscallError> { - for entry in self.entries.values_mut() { - if let Entry::File { id, mtime: mt, .. } = entry { - if *id == file_id { - *mt = mtime; - break; - } - } - } - Ok(()) - } - - /// Nothing to give back: the pages are the file, and `set_size` dropped them. - fn truncate_to(&mut self, _file_id: FileId, _size: u64, _mtime: u64) -> Result<(), SyscallError> { - Ok(()) - } - fn create_symlink(&mut self, name: &str, target: &str) -> Result<(), SyscallError> { // Displaces whatever answered to this name: one name, one entry. if let Some(Entry::File { id, alive, .. }) = @@ -223,10 +197,6 @@ impl FileSystem for TmpFs { Ok(()) } - fn sync(&mut self) -> Result<(), SyscallError> { - Ok(()) - } - fn open_backing(&mut self, name: &str) -> Result, SyscallError> { match self.entries.get(name) { Some(Entry::File { id, alive, .. }) => { @@ -235,9 +205,4 @@ impl FileSystem for TmpFs { _ => Err(SyscallError::NotFound), } } - - /// `TmpfsBacking` reads the file cache, so it is never behind it. - fn cached_file_id(&mut self, _name: &str) -> Option { - None - } } diff --git a/kernel/src/vfs.rs b/kernel/src/vfs.rs index 29763ff0643..024350c1fa7 100644 --- a/kernel/src/vfs.rs +++ b/kernel/src/vfs.rs @@ -6,9 +6,7 @@ use alloc::vec::Vec; use core::ops::{Deref, DerefMut}; use toyos_abi::syscall::SyscallError; -use crate::durability::Owed; use crate::file_cache::FileId; -use crate::mm::PAGE_BYTES; use crate::sync::{Lock, LockGuard}; static VFS: Lock> = Lock::new(None); @@ -54,8 +52,6 @@ pub trait FileSystem: Send { fn open_file(&mut self, name: &str) -> Result<(FileId, Option>), SyscallError>; /// Create an empty file, registered under `name`. fn create(&mut self, name: &str, mtime: u64) -> Result; - /// Release filesystem state for `file_id`, after the file cache dropped its last reference under the VFS lock. - fn close_file(&mut self, file_id: FileId); /// Unlink `name`, or `NotFound` if there was nothing by that name. fn delete(&mut self, name: &str) -> Result<(), SyscallError>; @@ -68,28 +64,10 @@ pub trait FileSystem: Send { /// a non-empty directory each by its own error. fn remove_dir(&mut self, name: &str) -> Result<(), SyscallError>; - /// Write one dirty page to the device, allocating its block if needed. - fn write_page(&mut self, file_id: FileId, page_idx: u32, data: &[u8; PAGE_BYTES]) -> Result<(), SyscallError>; - /// Update file metadata (size, mtime) after flushing dirty pages. - fn update_metadata(&mut self, file_id: FileId, size: u64, mtime: u64) -> Result<(), SyscallError>; - - /// Give everything above `size` back before a flush writes its pages: - /// [`FileSystem::update_metadata`] sees only the final size, which cannot - /// tell a file that shrank and regrew from one that was always that long. - fn truncate_to(&mut self, file_id: FileId, size: u64, mtime: u64) -> Result<(), SyscallError>; - fn create_symlink(&mut self, name: &str, target: &str) -> Result<(), SyscallError>; - /// An implementation of `sync` must not swallow a lower-level failure and report success: the log depends on this call telling the truth about durability. - fn sync(&mut self) -> Result<(), SyscallError>; - /// `open_backing` has no default body: an unimplemented one would silently report every file on that mount as missing (the sentinel this trait exists to remove). fn open_backing(&mut self, name: &str) -> Result, SyscallError>; - - /// The `FileId` this mount already minted for `name`, when a device view of - /// it could be behind the file cache; `None` from a mount whose backing - /// reads the cache itself, and from a name it holds no id for. - fn cached_file_id(&mut self, name: &str) -> Option; } @@ -109,10 +87,6 @@ pub const ROOT_ENTRIES: [&str; 9] = struct Mount { fs: Box, access: UserAccess, - names: Vec<&'static str>, - /// The device commit this mount still owes: raised by every flush that may - /// have reached the device, settled only by a [`FileSystem::sync`] that returned `Ok`. - commit: Owed, } /// A name at `/`, and where in its filesystem that name begins. @@ -140,7 +114,6 @@ pub struct OpenTarget(String); impl OpenTarget { pub fn as_str(&self) -> &str { &self.0 } - pub fn into_string(self) -> String { self.0 } } #[derive(Clone, Copy, PartialEq, Eq)] @@ -214,10 +187,10 @@ impl Vfs { } /// Mount one filesystem under `names`, each a [`ROOT_ENTRIES`] name. Several - /// names put each under its own directory of it: one filesystem, one sync. + /// names put each under its own directory of it. pub fn mount(&mut self, names: &[&'static str], fs: Box, access: UserAccess) { let index = self.mounts.len(); - self.mounts.push(Mount { fs, access, names: names.to_vec(), commit: Owed::new() }); + self.mounts.push(Mount { fs, access }); for name in names { let prefix = if names.len() == 1 { "" } else { *name }; let slot = Self::entry(name).expect("a mount name is one of the root's entries"); @@ -453,65 +426,6 @@ impl Vfs { fs.create(&fs_path, mtime) } - /// No early return on an empty dirty set: `ftruncate` changes the size without dirtying a page. - /// A refused attempt restores nothing, because nothing was cleared: debt is - /// settled per page against what was copied, and for the file only past `update_metadata`. - pub fn flush_file(&mut self, path: &str, file_id: FileId, mtime: u64) -> Result<(), SyscallError> { - let plan = crate::file_cache::begin_flush(file_id); - let (mount, file) = self.resolve_path("/", path); - if file.is_empty() { return Err(SyscallError::InvalidArgument); } - let point = self.point(&mount).ok_or(SyscallError::NotFound)?; - // Raised before the first write, not after the last: a flush that failed - // half-way may already have reached the device's cache. - self.mounts[point.fs].commit.record_write(); - let fs_path = point.path(&file); - let fs = self.mounts[point.fs].fs.as_mut(); - - // Before the pages: a page rewritten above the mark must outlive the trim. - // Settled by the trim itself and not by the metadata write below, which - // may refuse: past this point the device names nothing above the mark, - // and a retry that trimmed again would free the pages it just wrote. - if let Some(mark) = plan.shrunk_to { - fs.truncate_to(file_id, mark, mtime)?; - crate::file_cache::settle_shrink(file_id); - } - - // On the heap: the idle loop's 16 KiB stack has no guard page. - let mut heap = alloc::vec![0u8; PAGE_BYTES].into_boxed_slice(); - let buf: &mut [u8; PAGE_BYTES] = (&mut heap[..]).try_into().expect("PAGE_BYTES bytes"); - let mut flushed: Vec<(u32, crate::durability::Settlement)> = - Vec::with_capacity(plan.pages.len()); - for &page_idx in &plan.pages { - if let Some(copied) = crate::file_cache::copy_page_out(file_id, page_idx, buf) { - fs.write_page(file_id, page_idx, buf)?; - flushed.push((page_idx, copied)); - } - } - crate::file_cache::settle_pages(file_id, &flushed); - - let size = crate::file_cache::size(file_id); - fs.update_metadata(file_id, size, mtime)?; - crate::file_cache::settle_file(file_id, plan.file); - - // A refusal is logged, not returned: the bytes are on the device; only evictability is lost. - if !crate::file_cache::has_backing(file_id) { - match fs.open_backing(&fs_path) { - Ok(backing) => crate::file_cache::set_backing(file_id, backing), - Err(e) => log!("vfs: {path} was flushed but has no backing to evict through: {e}"), - } - } - Ok(()) - } - - /// Close a file (release filesystem state when last ref drops). - pub fn close_file(&mut self, path: &str, file_id: FileId) { - let (mount, file) = self.resolve_path("/", path); - if file.is_empty() { return; } - if let Some((fs, _fs_path)) = self.resolve_fs(&mount, &file) { - fs.close_file(file_id); - } - } - /// Unlink `path` on its mount. pub fn delete_file(&mut self, path: &str) -> Result<(), SyscallError> { let (mount, file) = self.resolve_path("/", path); @@ -628,50 +542,6 @@ impl Vfs { self.delete_file(path) } - /// Whether `SYS_FSYNC` still owes this file work: its own flush, or the - /// device commit its mount has raised and not settled — which is how a - /// failed `sync` keeps the next fsync honest with every page flushed. - pub fn durability_owed(&self, path: &str, file_id: FileId) -> bool { - if crate::file_cache::flush_owed(file_id) { - return true; - } - let (mount, _) = self.resolve_path("/", path); - self.point(&mount).is_some_and(|p| self.mounts[p.fs].commit.is_owed()) - } - - /// Make one mount point's filesystem durable — and with it every other name - /// that reaches it, because one filesystem has one device commit. - pub fn sync_mount(&mut self, name: &str) -> Result<(), SyscallError> { - let point = self.point(name).ok_or(SyscallError::NotFound)?; - let mount = &mut self.mounts[point.fs]; - let upto = mount.commit.snapshot(); - mount.fs.sync()?; - mount.commit.settle(upto); - Ok(()) - } - - /// `/system/bin/logd` calls a line durable off this call's result, so `sync_for_path` must reach the device's write cache, not stop at the page cache. - pub fn sync_for_path(&mut self, path: &str) -> Result<(), SyscallError> { - let (mount, _) = self.resolve_path("/", path); - // Nothing mounted is not an error: the write being made durable cannot have happened. - if self.point(&mount).is_none() { - return Ok(()); - } - self.sync_mount(&mount) - } - - /// A refusal is logged, not returned, so one mount failing does not stop the rest. - pub fn sync_all(&mut self) { - for mount in self.mounts.iter_mut() { - let upto = mount.commit.snapshot(); - match mount.fs.sync() { - Ok(()) => mount.commit.settle(upto), - Err(e) => log!("vfs: {:?} would not sync: {e}", mount.names), - } - } - } - - /// The write-back queue is drained whole, not filtered by path, because it is keyed by the name a handle was opened under, and a symlink, rename, or relative open names the same file differently. pub fn open_backing(&mut self, path: &str) -> Result, SyscallError> { self.open_backing_identified(path).map(|(backing, _)| backing) } @@ -682,22 +552,8 @@ impl Vfs { &mut self, path: &str, ) -> Result<(alloc::sync::Arc, BackingId), SyscallError> { - crate::writeback::drain_held(self); let target = self.resolve_for_open(path, ResolveIntent::KernelOrRead)?; - // A file still open is on no write-back queue, so the drain above cannot - // see it and a device view taken now would read round its cache pages. - let dirty = { - let (fs, fs_path) = self.fs_for_target(&target)?; - fs.cached_file_id(&fs_path).filter(|&id| crate::file_cache::flush_owed(id)) - }; - if let Some(file_id) = dirty { - // The flush's own instant: no one handle's last write is the file's. - let mtime = crate::clock::nanos_since_boot(); - let owner = String::from(target.as_str()); - self.flush_file(&owner, file_id, mtime)?; - } let (fs, fs_path) = self.fs_for_target(&target)?; - // After the flush, so the identity is the file as this call leaves it. let mtime = fs.file_mtime(&fs_path)?; let backing = fs.open_backing(&fs_path)?; let id = BackingId { size: backing.file_size(), mtime }; diff --git a/kernel/src/writeback.rs b/kernel/src/writeback.rs deleted file mode 100644 index 0b465772305..00000000000 --- a/kernel/src/writeback.rs +++ /dev/null @@ -1,134 +0,0 @@ -//! The write-back queue: dirty-page teardown for a closed file, deferred -//! off the closing thread onto `iod`. -//! -//! A queued file stays pinned (`teardown_owed`) until drained; eviction -//! never takes a dirty page. Entries pop one at a time under the VFS lock. -//! A flush refused on budget re-enqueues rather than tearing down — no -//! flush is lost to a timeout. [`drain_held`] cannot park: a refusal there -//! re-enqueues and leaves the retry to `iod`. - -use alloc::collections::VecDeque; -use alloc::string::String; - -use toyos_abi::syscall::SyscallError; - -use crate::watch::{self, Armed, Watch}; -use crate::file_cache::{self, FileId, Teardown}; -use crate::scheduler::Parkable; -use crate::sync::Lock; -use crate::time::Deadline; - -struct Pending { - file_id: FileId, - path: String, - mtime: u64, -} - -static QUEUE: Lock> = Lock::new(VecDeque::new()); - -/// The subject `iod` parks on: closes post here, and `iod` holds the sole arm across its loop. -pub static WORK: Watch = Watch::new(); - -/// Enqueues a dropped file's teardown and wakes `iod`; takes only the queue lock, so callable from `Drop`. -pub fn enqueue(file_id: FileId, path: String, mtime: u64) { - QUEUE.lock().push_back(Pending { file_id, path, mtime }); - // Edge-triggered: `iod` rechecks the queue itself, never reads state off the post. - WORK.post(); -} - -/// Runs every pending teardown to durability; for callers holding no watch registration (`iod` uses [`drain_all_iod`]). -pub fn drain_all() { - drain_retrying(crate::block::between_attempts); -} - -/// Same as [`drain_all`], for `iod`: reuses its standing [`WORK`] arm since a second arm panics the machine. -pub fn drain_all_iod(parkable: &Parkable, armed: &Armed<'_>) { - drain_retrying(|attempt| { - if attempt <= 1 { - crate::scheduler::yield_now(); - } else { - // Reuses `armed`; arming again panics. - let deadline = Deadline::at(crate::clock::now() + crate::block::backoff_step(attempt)); - let _ = watch::wait(parkable, armed, deadline); - } - }); -} - -/// Shared retry loop for [`drain_all`] and [`drain_all_iod`]: passes over the queue, backing off between passes, until drained or `block::DEADMAN` gives up. -fn drain_retrying(mut backoff: impl FnMut(u32)) { - let deadman = Deadline::at(crate::clock::now() + crate::block::DEADMAN.duration()); - let mut attempt = 0u32; - loop { - let mut owed = false; - // Snapshot: a re-enqueued entry waits for the next pass, not this one. - let n = QUEUE.lock().len(); - for _ in 0..n { - // Fresh VFS lock per entry: backoff below holds no lock. - match drain_one(&mut crate::vfs::lock(), deadman) { - Drained::Empty => break, - Drained::Done => {} - Drained::Owed => owed = true, - } - } - if !owed { - return; - } - attempt += 1; - backoff(attempt); - } -} - -/// Drains the queue under the caller's held VFS lock (`Vfs::open_backing`) so no device view is taken while an entry is owed; cannot park, so a refusal re-enqueues for [`drain_all`] to retry. -pub fn drain_held(vfs: &mut crate::vfs::Vfs) { - // Bounds the pass: a re-enqueued entry is not spun on within this call. - let n = QUEUE.lock().len(); - for _ in 0..n { - if let Drained::Empty = drain_one(vfs, Deadline::never()) { - break; - } - } -} - -enum Drained { - Empty, - /// Torn down, including a give-up that left pages unflushed. - Done, - /// Never torn down: still pinned and owed, for a caller's retry loop to re-drive. - Owed, -} - -/// Pops one entry and flushes/tears it down under the caller's VFS lock; a budget refusal within `deadman` returns [`Drained::Owed`] instead. -fn drain_one(vfs: &mut crate::vfs::Vfs, deadman: Deadline) -> Drained { - let Some(pending) = QUEUE.lock().pop_front() else { - return Drained::Empty; - }; - - // A deleted file has nothing to flush; its handle is already torn down. - let probe = file_cache::writeback_probe(pending.file_id); - if probe.flush_owed && !probe.deleted { - let flushed = vfs.flush_file(&pending.path, pending.file_id, pending.mtime); - match flushed { - Ok(()) => {} - // Budget refusal, not a device fact: the flush's debt is unsettled, so the re-enqueued entry redelivers the same bytes. - Err(SyscallError::WouldBlock) if !deadman.reached(crate::clock::now()) => { - // Re-enqueued under the same held VFS lock the pop used: never absent from both at once. - QUEUE.lock().push_back(pending); - return Drained::Owed; - } - // Device error or deadman: undurable; logged loudly since no caller can see this error. - Err(e) => { - crate::log!( - "writeback: {} is not durable ({e:?}); its unflushed pages are lost", - pending.path, - ); - } - } - } - - // Re-checked under the VFS lock a re-open needs: either re-open adopted it or this removes the name. - match file_cache::finish_writeback(pending.file_id) { - Teardown::Released => vfs.close_file(&pending.path, pending.file_id), - Teardown::Adopted | Teardown::Vanished => {} - } - Drained::Done -} diff --git a/src/build.rs b/src/build.rs index 1c6fbaf7a34..47265990b77 100644 --- a/src/build.rs +++ b/src/build.rs @@ -2605,9 +2605,6 @@ mod tests { // Costs no kernel build, for `wake-fence-off`'s reason: only // `kernel-loom` turns it on, and `dump_request` must red under it. "dump-report-relaxed", - // Costs no kernel build, for `wake-fence-off`'s reason: only - // `kernel-loom` turns it on, and `durability` must red under it. - "durability-settle-blind", // The kernel this tree had before `arch::entry`'s `cld`: the // instruction gone and `DF` back out of the `SYSCALL` mask, so a // build carrying it inherits a set direction flag from whatever diff --git a/src/ci.rs b/src/ci.rs index 69f6b80181b..d458c497911 100644 --- a/src/ci.rs +++ b/src/ci.rs @@ -300,10 +300,6 @@ pub(crate) const CONTROLS: &[Control] = &[ "a_parking_contender_observes_the_holders_writes ... FAILED", "two_holders_never_overlap ... FAILED", ]), - red(KERNEL_LOOM, "durability-settle-blind", Some("durability"), &[ - "a_page_marked_clean_is_on_the_device ... FAILED", - "a_settled_commit_covers_only_flushed_writes ... FAILED", - ]), red(KERNEL_LOOM, "device-irq-lossy", Some("device_irq"), &[ "every_message_is_counted_once ... FAILED", "one_message_is_one_wake ... FAILED", diff --git a/src/metal.rs b/src/metal.rs index 25855e1b0aa..9d7b1051852 100644 --- a/src/metal.rs +++ b/src/metal.rs @@ -763,7 +763,7 @@ pub const FLASHABLE: &[(&str, Flash)] = &[ ("xhci-xecp-selftest", Flash::Ok), ("xhci-descriptor-selftest", Flash::Ok), // Two probes rather than staged inputs, and both are reads: the SS-reload - // one runs inside `iod`'s own context switch, and the input-core one merges + // one runs inside the `probe` thread's own context switch, and the input-core one merges // events it made up itself. ("sysret-ss-probe", Flash::Ok), ("test-input-merge", Flash::Ok), diff --git a/tests/common/iommu.rs b/tests/common/iommu.rs index 1e0db6d861e..c94203e3b7a 100644 --- a/tests/common/iommu.rs +++ b/tests/common/iommu.rs @@ -2085,7 +2085,7 @@ pub fn userdev_dma_fault( // says so: `owner=kernel` here would be a machine that halted, or was about // to. let handled = log.must_say(FAULT)?; - let slot = slot_of(log.text(), "[8086:10d3]")?; + let slot = slot_of(log.text(), "[1af4:1041]")?; if !handled.contains(&format!("owner=slot{slot} ")) { return Err(format!( "the unit's fault was recorded against {handled:?}, and the function that faulted \ diff --git a/tests/common/origin.rs b/tests/common/origin.rs index 2a3cd88b136..85052c32ce4 100644 --- a/tests/common/origin.rs +++ b/tests/common/origin.rs @@ -353,8 +353,6 @@ pub fn refused_stop(c_bins: &[(String, Vec)], rust_bins: &[(String, Vec) /// long it waits first (`userland/init`'s `FLUSH_BOUND`). const FLUSH_WAITED_OUT: &str = "init: logd did not answer the flush in"; const FLUSH_BOUND_MS: u64 = 5_000; -/// The kernel's record as a stop begins its sync, after every thread stopped. -const SYNCING: &str = "Syncing filesystems..."; /// The milliseconds since boot a program's line in `/log` carries: the one /// field of its head with a decimal point. @@ -471,20 +469,20 @@ pub fn keeps_the_owners_slots(rust_bins: &[(String, Vec)]) -> Result<(), Str // whatever the slots did, so that is its own verdict and not this one's. // init's word is in its ring when the machine stops, so the console // rarely carries it. An answered flush wrote init's stop line, which is - // stamped before the flush was asked; and the kernel's sync starting a - // flush bound or more after that line is the wait on two records of one - // clock. + // stamped before the flush was asked; and the kernel's stop record, after + // every thread stopped, a flush bound or more after that line is the wait + // on two records of one clock. let sync = tail .lines() - .find(|l| l.contains(SYNCING)) + .find(|l| toyos_quiesce::Record::parse(l).is_some()) .and_then(bootlog::record_millis) - .ok_or_else(|| format!("the console carries no {SYNCING:?} with a time\n{tail}"))?; + .ok_or_else(|| format!("the console carries no stop record with a time\n{tail}"))?; let stop = bootlog::stopping_line(&log).and_then(program_millis); let unanswered = match stop { _ if tail.contains(FLUSH_WAITED_OUT) => Some("init said so".to_string()), None => Some("init's stop line never reached /log".to_string()), Some(stop) => (sync.saturating_sub(stop) >= FLUSH_BOUND_MS).then(|| { - format!("the kernel synced {} ms after init's stop line", sync.saturating_sub(stop)) + format!("the kernel stopped {} ms after init's stop line", sync.saturating_sub(stop)) }), }; if let Some(why) = unanswered { diff --git a/tests/common/power.rs b/tests/common/power.rs index 606f12dc1ea..e99fb640c6c 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -369,7 +369,8 @@ fn woken_by_the_held_thread( toyos_quiesce::LAST_THREAD, ); let at = |needle: &str| whole.lines().position(|line| line.contains(needle)); - let (Some(held_at), Some(synced_at)) = (at(&held), at("Syncing filesystems...")) else { + let stopped_at = whole.lines().position(|line| toyos_quiesce::Record::parse(line).is_some()); + let (Some(held_at), Some(synced_at)) = (at(&held), stopped_at) else { return Err(format!("the kernel never held the thread it names ({held:?})\n{whole}")); }; if held_at > synced_at { diff --git a/tests/test-durations b/tests/test-durations index c1d6cab69f1..4c005b64c89 100644 --- a/tests/test-durations +++ b/tests/test-durations @@ -426,8 +426,6 @@ wall_clock_zone 9347 watchdog_fed 23405 watchdog_resets 9393 window_refusal 16 -writeback_reopen 4886 -writeback_spawn 8820 xhci_deaf_registers 21244 xhci_descriptor_walk 5186 xhci_flap 8220 diff --git a/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs b/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs index ee0e124b5b8..3b46cd93de6 100644 --- a/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs +++ b/tests/toyos-rust-tests/src/bin/fat_backing_revoked.rs @@ -71,13 +71,6 @@ fn write_file(path: &str, byte: u8) { f.write_all(&vec![byte; LEN]).unwrap_or_else(|e| panic!("write {path}: {e}")); patient(&format!("fsync {path}"), || f.sync_all()); } // close: the last handle drops here. - // The last close no longer drops the file from the cache on this thread — - // it pins it and hands the teardown to `iod` (`kernel::writeback`). Let that - // drain, so a later open of this name is served by the backing rather than - // adopting the pages this write left cached; the revocation this test checks - // lives in the backing, and a cache-served read would never reach it. The - // margin is enormous: the drain is microseconds of work. - thread::sleep(Duration::from_millis(200)); } fn read_all(f: &mut fs::File) -> std::io::Result> { @@ -87,10 +80,7 @@ fn read_all(f: &mut fs::File) -> std::io::Result> { } fn main() { - // The control. `write_file` drains the write-back so the file leaves the - // cache, so this open is served by the backing and not by pages the write - // left cached — if it were not, the attack below would prove nothing about - // that path. + // The control. write_file(CONTROL, VICTIM_BYTE); let control = read_all(&mut fs::File::open(CONTROL).expect("open the control")) .expect("read the control"); @@ -102,9 +92,7 @@ fn main() { write_file(VICTIM, VICTIM_BYTE); - // Held open, and deliberately not read: `write_file` drained the victim out - // of the cache, so every page is absent and each one is a fault the backing - // has to answer. + // Held open, and deliberately not read. let mut held = fs::File::open(VICTIM).expect("open the victim"); fs::remove_file(VICTIM).expect("unlink the victim"); diff --git a/tests/toyos-rust-tests/src/bin/fs_transactional.rs b/tests/toyos-rust-tests/src/bin/fs_transactional.rs index 7e5d2d9d8b6..670ed066bb4 100644 --- a/tests/toyos-rust-tests/src/bin/fs_transactional.rs +++ b/tests/toyos-rust-tests/src/bin/fs_transactional.rs @@ -10,7 +10,6 @@ use std::fs::{self, File, OpenOptions}; use std::io::{Read, Seek, SeekFrom, Write}; -use std::process::Command; const PAGE: usize = 4096; @@ -192,8 +191,6 @@ fn shrink_unflushed_then_regrow_reads_zeros(dir: &str, seed_len: usize) { f.sync_all().expect("fsync the pair"); } - // The read below is the device's answer rather than the pages that were just in it. - drained(); let back = fs::read(&path).unwrap_or_else(|e| panic!("reread {path}: {e}")); assert_eq!(back.len(), seed_len, "{dir}: the regrown length did not survive the close"); @@ -214,7 +211,6 @@ fn shrink_a_reopened_file_reads_zeros(dir: &str, seed_len: usize) { f.write_all(&seed).expect("write the seed"); f.sync_all().expect("fsync the seed"); } - drained(); let mut f = OpenOptions::new() .read(true).write(true) @@ -230,7 +226,6 @@ fn shrink_a_reopened_file_reads_zeros(dir: &str, seed_len: usize) { f.sync_all().expect("fsync the pair"); drop(f); - drained(); let back = fs::read(&path).unwrap_or_else(|e| panic!("reread {path}: {e}")); assert_eq!(back.len(), seed_len, "{dir}: the regrown length did not survive the close"); @@ -253,7 +248,6 @@ fn shrink_then_write_above_the_mark(dir: &str, seed_len: usize) { f.write_all(&seed).expect("write the seed"); f.sync_all().expect("fsync the seed"); } - drained(); let mut f = OpenOptions::new() .read(true).write(true) @@ -265,7 +259,6 @@ fn shrink_then_write_above_the_mark(dir: &str, seed_len: usize) { f.write_all(&payload).expect("write a page above the mark"); f.sync_all().expect("fsync the shrink and the page above it"); drop(f); - drained(); let back = fs::read(&path).unwrap_or_else(|e| panic!("reread {path}: {e}")); assert_eq!(back.len(), seed_len, "{dir}: the regrown length did not survive the close"); @@ -280,12 +273,6 @@ fn shrink_then_write_above_the_mark(dir: &str, seed_len: usize) { println!("{dir}: the hole under a page written above a shrink is zeros on the device"); } -/// A spawn settles the write-back queue, so a just-closed file has left the cache. -fn drained() { - let echo = Command::new("/system/bin/echo").arg("drained").output().expect("run echo"); - assert!(echo.status.success()); -} - fn check_hole(dir: &str, got: &[u8], seed: &[u8], cut: usize) { assert_eq!(&got[..cut], &seed[..cut], "{dir}: the surviving head changed across shrink and regrow"); if let Some(at) = got[cut..].iter().position(|&b| b != 0) { diff --git a/tests/toyos-rust-tests/src/bin/handle_transfer.rs b/tests/toyos-rust-tests/src/bin/handle_transfer.rs index dd8a84d286f..66bd318eb60 100644 --- a/tests/toyos-rust-tests/src/bin/handle_transfer.rs +++ b/tests/toyos-rust-tests/src/bin/handle_transfer.rs @@ -47,8 +47,7 @@ const SELF_PATH: &str = "/system/bin/test_rs_handle_transfer"; /// The name the child's namespace carries the connector under. const SERVICE: &str = "transfer"; -/// A kernel file: tmpfs is the one writable mount the kernel still has, and its -/// last handle's release runs the same write-back a device's did. +/// A kernel file: tmpfs is the one writable mount the kernel still has. const DIRTY_PATH: &str = "/tmp/handle_transfer_flush.bin"; const DIRTY_BYTES: &[u8] = b"a file whose last handle went while it was queued"; @@ -243,8 +242,7 @@ fn a_senders_exit_does_not_retract_what_it_sent() { /// /// `HandleQueue` holds arbitrary `HandleEntry`s and `ConnectionEnd` is a /// `deferred` row, so its zero-handle hook drops whatever is queued. A `File` -/// is an `immediate` row, so its destructor runs *there*: `vfs::lock()`, -/// `flush_file`, the FAT32 adapter and a device round trip, wherever the drain +/// is an `immediate` row, so its destructor runs *there*, wherever the drain /// is running. /// /// **Which stack that is, is not this test's to choose**, and that is the whole diff --git a/tests/toyos-rust-tests/src/bin/home_backing_revoked.rs b/tests/toyos-rust-tests/src/bin/home_backing_revoked.rs index 73041c13e61..b6a1b1fb974 100644 --- a/tests/toyos-rust-tests/src/bin/home_backing_revoked.rs +++ b/tests/toyos-rust-tests/src/bin/home_backing_revoked.rs @@ -6,8 +6,6 @@ use std::fs; use std::io::{Read, Write}; -use std::thread; -use std::time::Duration; const VICTIM: &str = "/home/revoke_victim.bin"; const ATTACKER: &str = "/home/revoke_attacker.bin"; @@ -26,13 +24,6 @@ fn write_file(path: &str, byte: u8) { f.write_all(&vec![byte; LEN]).unwrap_or_else(|e| panic!("write {path}: {e}")); f.sync_all().unwrap_or_else(|e| panic!("fsync {path}: {e}")); } // close: the last handle drops here. - // The last close no longer drops the file from the cache on this thread — - // it pins it and hands the teardown to `iod` (`kernel::writeback`). Let that - // drain, so a later open of this name is served by the backing rather than - // adopting the pages this write left cached; the revocation this test checks - // lives in the backing, and a cache-served read would never reach it. The - // margin is enormous: the drain is microseconds of work. - thread::sleep(Duration::from_millis(200)); } fn read_all(f: &mut fs::File) -> std::io::Result> { @@ -42,10 +33,7 @@ fn read_all(f: &mut fs::File) -> std::io::Result> { } fn main() { - // The control. `write_file` drains the write-back so the file leaves the - // cache, so this open is served by the backing and not by pages the write - // left cached — if it were not, the attack below would prove nothing about - // that path. + // The control. write_file(CONTROL, VICTIM_BYTE); let control = read_all(&mut fs::File::open(CONTROL).expect("open the control")) .expect("read the control"); @@ -57,9 +45,7 @@ fn main() { write_file(VICTIM, VICTIM_BYTE); - // Held open, and deliberately not read: `write_file` drained the victim out - // of the cache, so every page is absent and each one is a fault the backing - // has to answer. + // Held open, and deliberately not read. let mut held = fs::File::open(VICTIM).expect("open the victim"); fs::remove_file(VICTIM).expect("unlink the victim"); diff --git a/tests/toyos-rust-tests/src/bin/writeback_reopen.rs b/tests/toyos-rust-tests/src/bin/writeback_reopen.rs deleted file mode 100644 index cc88bbb6574..00000000000 --- a/tests/toyos-rust-tests/src/bin/writeback_reopen.rs +++ /dev/null @@ -1,55 +0,0 @@ -//! A re-open racing a pending write-back reads what was written. -//! -//! When the last handle of a modified file drops, the kernel does not tear it -//! down on the closing thread: the file is pinned in the cache and `iod` tears -//! it down later (`kernel::writeback`). This is the invariant the write-back -//! queue rests on: a file's pages outlive the handle that dirtied them, and a -//! re-open before the drain is handed the pinned file. -//! -//! **`writeback-stall` parks `iod` before any teardown**, so the teardown is -//! provably still owed when this re-opens the file. `/tmp` is the one writable -//! directory the kernel serves, so it is where the queue has files: if the -//! last close released the file instead of pinning it, the re-open would find -//! its name and none of its pages, and read an empty file. - -use std::fs; -use std::io::{Read, Write}; - -const PATH: &str = "/tmp/wb_reopen.bin"; -/// Three pages and a bit, so the file is several dirty pages rather than one. -const LEN: usize = 3 * 4096 + 137; - -fn distinctive() -> Vec { - (0..LEN).map(|i| (i.wrapping_mul(31) ^ 0xA5) as u8).collect() -} - -fn main() { - let want = distinctive(); - - // Write and drop the handle: the last close pins the file and queues its - // teardown, which `iod` never runs on this boot. - { - let mut f = fs::File::create(PATH).unwrap_or_else(|e| panic!("create {PATH}: {e}")); - f.write_all(&want).expect("write the distinctive bytes"); - } - - let mut got = Vec::new(); - { - let mut f = fs::File::open(PATH).expect("re-open before the write-back drains"); - f.read_to_end(&mut got).expect("read back the pinned file"); - } - - assert_eq!( - got.len(), - want.len(), - "re-open read {} bytes, wrote {} — the last close released a file its teardown still owed", - got.len(), - want.len() - ); - if let Some(at) = got.iter().zip(&want).position(|(a, b)| a != b) { - panic!("re-open differs from what was written at byte {at}: the pinned pages were not read"); - } - let _ = fs::remove_file(PATH); - - println!("re-open before the write-back drains read all {LEN} pinned bytes"); -} diff --git a/tests/toyos-rust-tests/src/bin/writeback_spawn.rs b/tests/toyos-rust-tests/src/bin/writeback_spawn.rs deleted file mode 100644 index 9aa018df986..00000000000 --- a/tests/toyos-rust-tests/src/bin/writeback_spawn.rs +++ /dev/null @@ -1,97 +0,0 @@ -//! A program written, closed and spawned runs, with its write-back still owed. -//! -//! Write, close, exec is what a compiler does with the binary it just produced, -//! and it is the project's self-hosting north star. The last close does not -//! tear the file down: it is pinned in the file cache and `iod` tears it down -//! later (`kernel::writeback`). `loader::spawn` does not read the file through -//! a handle — it takes a view of the file (`Vfs::open_backing`), which drains -//! the queue inline first, so the spawn runs the teardown `iod` owes. -//! -//! **`writeback-stall` parks `iod` before any teardown**, so the teardown is -//! provably the spawn's own. `/tmp` is the one writable directory the kernel -//! serves: a teardown that released a live file's pages there would leave the -//! spawn's view reading zeros where the binary was, and the load refused. -//! -//! The payload is this binary itself, re-run with an argument, so the thing -//! spawned is a megabyte-scale PIE whose text, relocations and symbols all -//! demand-page through the view — a short file would prove the header and -//! nothing after it. - -use std::fs; -use std::io::Write; -use std::process::Command; - -const DIR: &str = "/tmp/writeback_spawn"; -const IN_ROOT: &str = "/system/bin/test_rs_writeback_spawn"; -const WRITTEN: &str = "/tmp/writeback_spawn/child"; -const STILL_OPEN: &str = "/tmp/writeback_spawn/held"; -/// What tells this binary it is the copy being run rather than the test. -const CHILD: &str = "spawned-from-tmp"; - -fn main() { - if std::env::args().nth(1).as_deref() == Some(CHILD) { - // The marker is the proof the child reached its own code, rather than - // an exit status a refused spawn could also produce. - println!(" child: running from {WRITTEN}"); - return; - } - - let _ = fs::create_dir(DIR); - - let image = fs::read(IN_ROOT).unwrap_or_else(|e| panic!("read {IN_ROOT}: {e}")); - // `fs::write` opens, writes and drops the handle. With `iod` parked that - // drop pins the file and queues its teardown. - fs::write(WRITTEN, &image).unwrap_or_else(|e| panic!("write {WRITTEN}: {e}")); - println!(" copied {} bytes to {WRITTEN} with its teardown still owed", image.len()); - - let status = Command::new(WRITTEN) - .arg(CHILD) - .status() - .unwrap_or_else(|e| panic!("spawn {WRITTEN}: {e}")); - assert!( - status.success(), - "the child spawned off a file whose teardown was still owed exited {:?}", - status.code() - ); - - // The same bytes read back a second way, through a handle, after the - // spawn ran the teardown: compared against ROOT's copy, a different mount, - // so a length that matched by construction could not hide a wrong byte. - let back = fs::read(WRITTEN).unwrap_or_else(|e| panic!("read back {WRITTEN}: {e}")); - assert_eq!(back.len(), image.len(), "read back {} bytes, wrote {}", back.len(), image.len()); - if let Some(at) = back.iter().zip(&image).position(|(a, b)| a != b) { - panic!("what the file holds after its teardown differs from what was written at byte {at}"); - } - - println!(" PASS: spawned {WRITTEN} before its teardown, {} bytes verified after it", back.len()); - - still_open_and_dirty(&image); - - let _ = fs::remove_dir_all(DIR); - println!("writeback spawn test passed"); -} - -/// The same with the handle still held, which no queue knows about: the view -/// a spawn takes flushes a file still open and dirty before it reads it. -fn still_open_and_dirty(image: &[u8]) { - let mut held = fs::File::create(STILL_OPEN).unwrap_or_else(|e| panic!("create {STILL_OPEN}: {e}")); - held.write_all(image).unwrap_or_else(|e| panic!("write {STILL_OPEN}: {e}")); - - let status = Command::new(STILL_OPEN) - .arg(CHILD) - .status() - .unwrap_or_else(|e| panic!("spawn {STILL_OPEN}: {e}")); - assert!( - status.success(), - "the child spawned off a still-open dirty file exited {:?}", - status.code() - ); - - drop(held); - let back = fs::read(STILL_OPEN).unwrap_or_else(|e| panic!("read back {STILL_OPEN}: {e}")); - assert_eq!(back.len(), image.len(), "read back {} bytes, wrote {}", back.len(), image.len()); - if let Some(at) = back.iter().zip(image).position(|(a, b)| a != b) { - panic!("what the file holds differs from what was written at byte {at}"); - } - println!(" PASS: spawned {STILL_OPEN} while its writer still held it, {} bytes verified", back.len()); -} diff --git a/tests/toyos.rs b/tests/toyos.rs index 8bf53a44219..24755dba09f 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -430,12 +430,7 @@ const RUST_SKIP: &[&str] = &[ // binary* and its `ALONE:` line was about a different test. What the shared // copy adds is the binary exiting 0 on a boot that gives it nothing to // measure. - // - // `writeback_reopen` and `writeback_spawn` each need their own boot with - // `writeback-stall` armed, and run in `MACHINE_TESTS`, not on the shared - // boot. - "writeback_reopen", - "writeback_spawn", + // What it stages on `/log` — a file // unlinked out from under a held descriptor, its clusters handed to the next // writer — is only half the claim, and the other half is the volume read @@ -1494,12 +1489,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ ("root_named_twice_on_the_boot_disk", Sched::Serial, Tier::Fast), ("root_named_twice", Sched::Serial, Tier::Nightly), ("log_partition_identity", Sched::Parallel, Tier::Nightly), - // The write-back queue's two negative controls (wall 4 of - // `issues/kernel/every-wait-in-this-kernel-is-a-spin.md`). `writeback_reopen` - // and `writeback_spawn` arm `writeback-stall`, so each needs its own actuator - // boot: one holds the queue open across a *handle* re-open, which the file - // cache answers, and the other across a *spawn*, whose view of the file - // drains the queue itself. // The watch's lost-wake window, staged: `watch-window` holds every pipe // waiter between reading its condition and parking, so the peer's post lands where // only the notified bit carries it to the commit. @@ -1508,8 +1497,6 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // and its store (`copy-meets-a-remap`): the store never reaches the region // mapped after it. ("user_copy_races_munmap", Sched::Parallel, Tier::Fast), - ("writeback_reopen", Sched::Parallel, Tier::Fast), - ("writeback_spawn", Sched::Parallel, Tier::Nightly), // `KernelHw::switch`'s SS reload (AMD `X86_BUG_SYSRET_SS_ATTRS`) observed the // one way a guest can, since its `SYSRET` does not reproduce the erratum. Reds // the day that `mov ss` leaves the switch. @@ -1622,8 +1609,6 @@ const CARRIES: &[(&str, &[&str])] = &[ ("update_refused_pass_credits_no_image", &[]), ("blocking_read_window", &["test_rs_blocking_read_stress"]), ("user_copy_races_munmap", &["test_rs_copy_out_races_munmap"]), - ("writeback_reopen", &["test_rs_writeback_reopen"]), - ("writeback_spawn", &["test_rs_writeback_spawn"]), ("xhci_second_controller", &["test_rs_input_events"]), ("xhci_msi_only", &["test_rs_input_events"]), ("metal_sim_input", &["test_rs_input_events"]), @@ -10542,20 +10527,19 @@ fn blocked_dump() -> Result<(), String> { return Err(format!("no parked task was named by pid and tid:\n{report}")); } - // **All three kernel threads, by name.** They are almost always blocked, so + // **Both kernel threads, by name.** They are almost always blocked, so // the parked lines above carry them as a pid and a tid and nothing else — - // and on a machine that has gone quiet the question is *which* of the three - // is stuck. `sched::dump`'s census tags a kernel thread whatever it is - // doing, which is C6's gate: three kernel threads split the work — - // `klogd` the console drain, `usbd` the xHCI port machine, `iod` the - // write-back queue — precisely so that one of them wedging does not stop - // the other two. A report that cannot tell them apart cannot say which did. + // and on a machine that has gone quiet the question is *which* is stuck. + // `sched::dump`'s census tags a kernel thread whatever it is doing, which + // is C6's gate: `klogd` the console drain and `usbd` the xHCI port machine + // split the work precisely so that one wedging does not stop the other. A + // report that cannot tell them apart cannot say which did. // // Matched with the ` cpu=` that follows the name on the census line, because // a bare name appears in every one of these programs' own log lines and // `/system/bin/init` speaks in a program's name before that program runs // (`tests/CLAUDE.md`). - let unnamed: Vec<&str> = ["klogd", "usbd", "iod"] + let unnamed: Vec<&str> = ["klogd", "usbd"] .into_iter() .filter(|name| !report.contains(&format!(" {name} cpu="))) .collect(); @@ -11358,9 +11342,6 @@ fn metal_sim_client_death(boot: &mut Boot) -> Result<(), String> { Ok(()) } -/// What `iod` says when `writeback-stall` parks it: a write-back test's proof -/// that the queue it stages was held (`kernel/src/iod.rs`). -const WRITEBACK_STALLED: &str = "iod: writeback-stall: parked for the boot"; /// Run one machine-shape test. Like `run_screen_test`, each of these owns its /// QEMU — the machine shape *is* the test — except for the runs of adjacent @@ -11489,7 +11470,7 @@ fn run_machine_test( // Same again: the FAT32 read side's revocation, judged off the volume the // guest's unlink-and-reallocate cycle left behind. "fat_backing_revoked" => common::volumes::fat_backing_revoked(test_config, c_bins, rust_bins), - // `sysret-ss-probe` has iod null SS, force a switch, and log whether the + // `sysret-ss-probe` has the `probe` thread null SS, force a switch, and log whether the // switch reloaded it; a missing `mov ss` turns `reloaded` into `NOT`. "sysret_ss_reload" => { let options = BootOptions { @@ -11499,7 +11480,7 @@ fn run_machine_test( let mut qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); // A liveness ceiling, not a pace: a loaded shard once took past a - // fixed 500 ms drain to run iod's probe (run 33246638742, alone-green). + // fixed 500 ms drain to run the probe (run 33246638742, alone-green). // The T14's readback needs no drain at all — the whole boot's records // are on the stick — so the wait is here and the predicate is shared. let log = qemu.boot_log().to_string() @@ -11553,53 +11534,6 @@ fn run_machine_test( } Ok(()) } - // The write-back queue's re-open control: `writeback-stall` parks `iod` - // before it drains, so the guest can prove a re-open before the flush - // reads the pinned pages. - "writeback_reopen" => { - let options = BootOptions { - kernel_params: &["writeback-stall"], - ..Default::default() - }; - let mut qemu = - QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - let boot = qemu.boot_log().to_string(); - let console = serial::Serial::named("boot console", boot.as_str()); - console.must_be_clean()?; - console.must_say(WRITEBACK_STALLED)?; - let result = qemu.run_test("test_rs_writeback_reopen", Duration::from_secs(30)); - if !check_rust_result(&result) { - return Err(format!( - "writeback_reopen failed:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - Ok(()) - } - // The other half of the same stall, on the path the file cache does not - // answer: a spawn takes a view of the file (`Vfs::open_backing`), which - // runs the teardown `iod` owes before it reads. Same actuator, and the - // same reason it needs its own boot. - "writeback_spawn" => { - let options = BootOptions { - kernel_params: &["writeback-stall"], - ..Default::default() - }; - let mut qemu = - QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - let boot = qemu.boot_log().to_string(); - let console = serial::Serial::named("boot console", boot.as_str()); - console.must_be_clean()?; - console.must_say(WRITEBACK_STALLED)?; - let result = qemu.run_test("test_rs_writeback_spawn", Duration::from_secs(30)); - if !check_rust_result(&result) { - return Err(format!( - "writeback_spawn failed:\n{}\nkernel log while it ran:\n{}{}", - result.stdout, result.before, result.serial - )); - } - Ok(()) - } "kernel_heartbeat" => { // The instrument for a machine whose log cannot say whether it was // alive: ten of the owner's boots are byte-identical between the @@ -13353,7 +13287,7 @@ fn run_machine_test( // names goes through `poison_tid`, the idle loop's `reap_poisoned` // and `zombify_poisoned`, none of which had ever seen a task with // no user address space. A row that quietly halted the machine - // would make `usbd` and `iod` worse than the thread they were + // would make `usbd` worse than the thread it was // split off from. // // The verdict is content in the same window and never a timeout: @@ -16666,7 +16600,7 @@ fn sysret_ss(log: &str) -> Result<(), String> { } if !log.contains("sysret-ss: reloaded") { return Err(format!( - "the SS-reload probe never reported — iod may not have run it:\n{log}" + "the SS-reload probe never reported — the probe thread may not have run it:\n{log}" )); } eprintln!(" [sysret-ss] the switch reloads SS from null before a sysretq can see it"); @@ -16707,7 +16641,7 @@ fn operation_nesting_log(log: &str) -> Result<(), String> { // with no task uses one slot per CPU, and the two are reached by // different arms of `operation_slot` — so a gate that ran in one // place would leave the other arm unexecuted by any test at all. - for site in ["boot", "iod"] { + for site in ["boot", "probe"] { let say = |what: &str| -> Result { let needle = format!("sched-op: {site} {what}"); log.lines() @@ -16798,7 +16732,7 @@ fn operation_nesting_log(log: &str) -> Result<(), String> { /// The machine's three kernel threads are hosted, and each claims the panic row /// its own loss demands. /// -/// Text in, a verdict out: all three lines are `log!` records, so the T14's +/// Text in, a verdict out: both lines are `log!` records, so the T14's /// readback and a QEMU boot log are judged by this one predicate. fn klogd_hosted(boot: &serial::Serial) -> Result<(), String> { boot.must_be_clean()?; @@ -16808,19 +16742,16 @@ fn klogd_hosted(boot: &serial::Serial) -> Result<(), String> { } eprintln!(" [klogd] {}", line.trim()); - // **The other two threads, and the opposite row.** `usbd` owns the xHCI - // port machine and `iod` the write-back queue, so a stuck USB enumeration - // cannot stop the log. Their panics are *recoverable* and `klogd`'s - // deliberately is not — a killed drainer is the one loss nothing left alive - // can report — and this is the one boot in the suite where all three rows - // are on the wire together. - for name in ["usbd", "iod"] { - let line = boot.must_say(&format!("kthread: {name}"))?; - if !line.contains("kills the thread") { - return Err(format!("{name} is hosted but claims the wrong panic row: {line:?}")); - } - eprintln!(" [kthread] {}", line.trim()); + // **The other thread, and the opposite row.** `usbd` owns the xHCI port + // machine, so a stuck USB enumeration cannot stop the log. Its panic is + // *recoverable* and `klogd`'s deliberately is not — a killed drainer is the + // one loss nothing left alive can report — and this is the one boot in the + // suite where both rows are on the wire together. + let line = boot.must_say("kthread: usbd")?; + if !line.contains("kills the thread") { + return Err(format!("usbd is hosted but claims the wrong panic row: {line:?}")); } + eprintln!(" [kthread] {}", line.trim()); Ok(()) } diff --git a/toyos-sched/src/task.rs b/toyos-sched/src/task.rs index 642f5af5439..cca751f4aff 100644 --- a/toyos-sched/src/task.rs +++ b/toyos-sched/src/task.rs @@ -174,13 +174,6 @@ const RETIRE_QUEUED: u64 = 1 << 63; /// task carries both, [`SafePoint`] is where that is decided. const STOP: u64 = 1 << 61; const STICKY: u64 = KILL | RETIRE_QUEUED | STOP; -/// The task is inside an update it may park in and must finish: where it parks -/// it has left something half made that only its own next attempt completes. -/// [`TaskShared::stop_if_blocked`] refuses it in the exchange that reads -/// `Blocked`, so it takes [`STOP`] at its own safe point, after the update. -/// In the word so that read and that mark cannot come apart; set and cleared -/// only by the task itself, while it runs. -const MID_UPDATE: u64 = 1 << 60; /// A post reached this task while it was neither parked nor committing. The /// next [`TaskShared::begin_commit`] consumes it and refuses the park, so a post /// landing between a waiter's registration on a watch and its commit is a @@ -193,7 +186,7 @@ const NOTIFIED: u64 = 1 << 59; /// commit, and cleared only by the task's own next registration. const REVOKED: u64 = 1 << 58; /// What every transition carries over. -const KEPT: u64 = STICKY | MID_UPDATE | NOTIFIED | REVOKED; +const KEPT: u64 = STICKY | NOTIFIED | REVOKED; /// What a thread standing at a Ring 3 boundary does instead of returning to /// userland. @@ -439,8 +432,7 @@ impl TaskShared { /// own safe point instead, through `SchedPass::dispose_stop`. `false` here /// therefore means "not parked, and not this caller's to stop" — including /// the task a waker claimed between the read and the exchange, which is on - /// its way to a CPU that will dispatch it to that safe point, and the task - /// parked [`Self::begin_update`]d, which has an update to finish first. + /// its way to a CPU that will dispatch it to that safe point. /// /// Idempotent: a task already carrying the bit answers `true` without /// writing. @@ -450,7 +442,7 @@ impl TaskShared { if cur & STOP != 0 { return true; } - if cur & MID_UPDATE != 0 || !matches!(unpack(cur), TaskState::Blocked(_)) { + if !matches!(unpack(cur), TaskState::Blocked(_)) { return false; } match self.state.compare_exchange_weak( @@ -689,19 +681,6 @@ impl TaskShared { pub(crate) fn mark_stop(&self) { self.state.fetch_or(STOP, Ordering::AcqRel); } - - /// The running task enters an update it may park in and must finish; until - /// [`Self::end_update`] no sweep stops it where it parks. Called by the - /// task itself, and not nested. - pub fn begin_update(&self) { - let was = self.state.fetch_or(MID_UPDATE, Ordering::AcqRel); - assert!(was & MID_UPDATE == 0, "a task began an update inside an update"); - } - - pub fn end_update(&self) { - let was = self.state.fetch_and(!MID_UPDATE, Ordering::AcqRel); - assert!(was & MID_UPDATE != 0, "a task ended an update it never began"); - } } // The linear task value and its five lifecycle types @@ -1304,21 +1283,6 @@ mod tests { ); } - /// The sweep that finds a task parked inside an update leaves it for its - /// own safe point: banded there, what it left half made stays half made. - #[test] - fn a_task_parked_mid_update_refuses_the_parked_mark() { - let s = running(C0); - s.begin_update(); - let generation = s.begin_commit(C0).expect("nothing notified this task"); - assert_eq!(s.commit_park(C0, generation), ParkOutcome::Parked); - assert_eq!(s.state(), TaskState::Blocked(C0)); - assert!(!s.stop_if_blocked(), "the update is open across this park"); - assert!(!s.stop_pending(), "and nothing was marked"); - s.end_update(); - assert!(s.stop_if_blocked(), "the same park with the update closed"); - } - #[test] fn park_then_wake_is_the_ordinary_path() { let s = running(C0); diff --git a/userland/metalprobe/src/usb.rs b/userland/metalprobe/src/usb.rs index 7e802294c9c..e43467b2f96 100644 --- a/userland/metalprobe/src/usb.rs +++ b/userland/metalprobe/src/usb.rs @@ -1,5 +1,5 @@ //! What the boot stick answers about itself, through the whole stack that -//! reaches it: the FAT32 driver, the page cache, `iod` and the xHCI mass-storage +//! reaches it: the FAT32 driver, the page cache and the xHCI mass-storage //! transport. //! //! **`/log` is the one writable place on a metal boot.** A flashed image carries @@ -9,8 +9,7 @@ use std::fs; use std::io::{Read, Write}; -use std::thread::sleep; -use std::time::{Duration, Instant}; +use std::time::Instant; use crate::{span, Measured, Refusal}; @@ -18,7 +17,7 @@ use crate::{span, Measured, Refusal}; /// /// **Not a device figure — the whole path's**, and an order of magnitude under /// what the device itself can do: what sizes [`BYTES`] is the cost of a write -/// through this kernel's page cache, `iod`, the FAT32 driver and the +/// through this kernel's page cache, the FAT32 driver and the /// mass-storage transport together, measured end to end at about 80 KiB/s and /// rounded down here to a power of two. const SLOWEST_KIB_S: u64 = 64; @@ -82,8 +81,8 @@ pub fn write() -> Measured { /// Read [`BYTES`] back off the device and check them, timing only the read. /// -/// **The staged file is closed and left to drain before the clock starts**, -/// though the read is answered from the file server's cache +/// **The staged file is closed before the clock starts**, though the read is +/// answered from the file server's cache /// (`issues/hardware/metalprobes-usb-read-is-answered-from-fsds-cache.md`). pub fn read() -> Measured { let blob = payload(); @@ -94,7 +93,6 @@ pub fn read() -> Measured { } f.sync_all().map_err(|_| Refusal::IoFailed)?; } - sleep(DRAIN); let began = Instant::now(); let mut got = Vec::with_capacity(BYTES); @@ -109,10 +107,3 @@ pub fn read() -> Measured { span(took.as_nanos()) } -/// What the close above is given to reach the device before the clock starts. -/// -/// **`sync_all` returns when the bytes are durable and not when the cache has -/// dropped them**, and nothing here can wait on `iod`'s own pass; generous -/// rather than tight, because a wait too short reports the page cache as the -/// stick. -const DRAIN: Duration = Duration::from_millis(200); diff --git a/userland/toybox/src/cp.rs b/userland/toybox/src/cp.rs index c3cd0158003..296faa70860 100644 --- a/userland/toybox/src/cp.rs +++ b/userland/toybox/src/cp.rs @@ -106,9 +106,8 @@ fn stream(reader: &mut File, source: &Path, partial: &Path) -> Result<(), String sync(&writer, partial) } -/// Closing a file reports nothing: the last handle's drop hands its dirty pages -/// to the kernel's write-back queue and returns, so a copy that never asks is a -/// copy that cannot be told the volume was full. `fsync` is the channel that +/// Closing a file reports nothing, so a copy that never asks is a copy that +/// cannot be told the volume was full. `fsync` is the channel that /// answers, which is why every write here goes through this function. fn sync(writer: &File, partial: &Path) -> Result<(), String> { writer.sync_all().map_err(|e| format!("flushing {}: {e}", partial.display())) From 58ad95d6d7c7407d07444b62d93e5f626432e5cb Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 19:41:53 +0200 Subject: [PATCH 28/54] partclaim: the departure's told line names the departing partition MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `Holder::Claim` now names its partition, so the old "a process's partition claim" line no longer exists; the judge counts the told lines and requires each to be DEPARTING's. `partition_claim` EXIT=0, all three. Filed: a child can flood its parent's log slots before logd names the ring's owner — `log_ring_keeps_the_owners_slots` EXIT=1 wide and alone at b58c22c3, green before the merge on the same code. Co-Authored-By: Claude Opus 5.5 --- ...s-log-slots-before-logd-names-the-owner.md | 26 +++++++++++++++++++ tests/common/partclaim.rs | 12 ++++++--- 2 files changed, 34 insertions(+), 4 deletions(-) create mode 100644 issues/filesystem/a-child-can-flood-its-parents-log-slots-before-logd-names-the-owner.md diff --git a/issues/filesystem/a-child-can-flood-its-parents-log-slots-before-logd-names-the-owner.md b/issues/filesystem/a-child-can-flood-its-parents-log-slots-before-logd-names-the-owner.md new file mode 100644 index 00000000000..91dda62f89d --- /dev/null +++ b/issues/filesystem/a-child-can-flood-its-parents-log-slots-before-logd-names-the-owner.md @@ -0,0 +1,26 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A child can flood its parent's log slots before logd names the ring's owner + +A ring keeps `CHILD_KEEP` slots for its owner only once `ring.own(pid)` has +run (`toyos/src/log/region.rs`), and logd runs it when it reads init's +registration of the ring (`userland/logd/src/origin.rs`). Until then the owner +word is 0 and a child's records take every slot. With `/log` served by fsd, +logd makes file-server round trips before it reads registrations, so a child +spawned in the first tens of milliseconds can fill the ring first. + +`log_ring_keeps_the_owners_slots` measured it at b58c22c3: EXIT=1 wide and +alone, 1917 flood lines in `/log` and no `===TEST_END test_rs_log_flood +exit=0===`, logd's first line at 0.509 s against the flood's at 0.498 s. The +same test was green on the same code before the merge of 6c9e2cb2, with 1853 +flood lines: which side of the race it lands on is the boot's timing. + +## Exit condition + +A ring's owner is named before its owner can run — by init at the spawn that +creates it, or by a registration logd reads before anything else — shown by +`log_ring_keeps_the_owners_slots` green across repeated boots. diff --git a/tests/common/partclaim.rs b/tests/common/partclaim.rs index a3dbc244245..044e0988499 100644 --- a/tests/common/partclaim.rs +++ b/tests/common/partclaim.rs @@ -347,15 +347,19 @@ pub fn partition_claim_departure( c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - const TOLD: &str = "writes a process's partition claim made before its disk came back owing \ - a flush may not have survived"; + const TOLD: &str = "writes a process's claim of partition "; const FLUSHED: &str = "usb-quiesce: disk 0 SYNCHRONIZE CACHE ok"; const UNTOLD: &str = "no writer's flush has said so; the disk is not counted flushed"; + let departing = format!( + "{TOLD}{DEPARTING} made before its disk came back owing a flush may not have survived" + ); for (role, told) in [("departure", 1), ("silent", 1), ("untold", 0)] { let (kernel, tail, spans) = departed(test_config, c_bins, rust_bins, role, told)?; let count = kernel.matches(TOLD).count(); - if count != told { - return Err(format!("{role}: {count} flushes were told of the loss, not {told}:\n{kernel}")); + if count != told || kernel.matches(departing.as_str()).count() != told { + return Err(format!( + "{role}: {count} flushes were told of the loss, not {told}, each {DEPARTING}'s:\n{kernel}" + )); } // The untold line begins as the flushed one does, so a flushed disk is // a flushed line without it. From c86dd840ac9a9f24b9f9ec36ac3d822de6f1458e Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 20:12:02 +0200 Subject: [PATCH 29/54] init: a log ring's owner is named at the spawn, and no writer is it before A child could take the slots a program's ring keeps for the program (`CHILD_KEEP`) because the owner word stayed 0 until logd read init's registration, and 0 keeps nothing from anyone. logd reads registrations only in its loop, after `Volume::open`'s calls to the log's file server, so at 58ad95d6 logd was spawned at 0.842 s and said its first line at 0.936 s while test-runner's job began flooding at 0.922 s: 1917 flood records and test-runner's two lines took all 1919 shared slots, and `log_ring_keeps_the_owners_slots` was red wide and alone. Before the merge of 6c9e2cb2 the same test was green at 80a1f1ce with the flood's first record and logd's first line both stamped 1.329 s: the merge touched neither logd nor the ring, and the race was already inside a millisecond. init now lays each ring out owned by `Pid::MAX`, which no writer has, so every writer leaves the kept slots until the owner is named, and names the program itself once the spawn returns its pid, before the ring goes to logd. logd no longer writes the ring. logd's `--stall` now holds the stalled program's registration until `--stall-until` instead of skipping its reads, so `log_ring_keeps_the_owners_slots` holds whatever logd has done: logd takes test-runner's ring only at the stop. Arms, `cargo test --test toyos-build -- log_ring_keeps_the_owners_slots`: 58ad95d6 EXIT=1 wide and alone (1917 flood lines); this change EXIT=0 (1853); the naming moved back to logd's open, with the placeholder gone, EXIT=1 wide and alone (1917). Co-Authored-By: Claude Opus 5.5 --- ...s-log-slots-before-logd-names-the-owner.md | 26 -------- userland/init/src/main.rs | 44 +++++++------ userland/logd/src/main.rs | 63 ++++++++++--------- userland/logd/src/origin.rs | 1 - 4 files changed, 60 insertions(+), 74 deletions(-) delete mode 100644 issues/filesystem/a-child-can-flood-its-parents-log-slots-before-logd-names-the-owner.md diff --git a/issues/filesystem/a-child-can-flood-its-parents-log-slots-before-logd-names-the-owner.md b/issues/filesystem/a-child-can-flood-its-parents-log-slots-before-logd-names-the-owner.md deleted file mode 100644 index 91dda62f89d..00000000000 --- a/issues/filesystem/a-child-can-flood-its-parents-log-slots-before-logd-names-the-owner.md +++ /dev/null @@ -1,26 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# A child can flood its parent's log slots before logd names the ring's owner - -A ring keeps `CHILD_KEEP` slots for its owner only once `ring.own(pid)` has -run (`toyos/src/log/region.rs`), and logd runs it when it reads init's -registration of the ring (`userland/logd/src/origin.rs`). Until then the owner -word is 0 and a child's records take every slot. With `/log` served by fsd, -logd makes file-server round trips before it reads registrations, so a child -spawned in the first tens of milliseconds can fill the ring first. - -`log_ring_keeps_the_owners_slots` measured it at b58c22c3: EXIT=1 wide and -alone, 1917 flood lines in `/log` and no `===TEST_END test_rs_log_flood -exit=0===`, logd's first line at 0.509 s against the flood's at 0.498 s. The -same test was green on the same code before the merge of 6c9e2cb2, with 1853 -flood lines: which side of the race it lands on is the boot's timing. - -## Exit condition - -A ring's owner is named before its owner can run — by init at the spawn that -creates it, or by a registration logd reads before anything else — shown by -`log_ring_keeps_the_owners_slots` green across repeated boots. diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index e3471acb4e2..56bdf2b28b4 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -184,16 +184,23 @@ struct Log { const STDIN_RIGHTS: Rights = Rights::READ.union(Rights::WAIT).union(Rights::DUP).union(Rights::TRANSFER); -/// A fresh log ring, laid out before any other process can map it: init's -/// handle, which a spawn duplicates into the program and [`Log::register`] -/// moves to `logd`. -fn new_ring() -> toyos::RawHandle { +/// A fresh log ring, laid out before any other process can map it, which a +/// spawn duplicates into the program and [`Log::register`] names it the owner +/// of. **Owned by no process until then**: the program's pid exists only once +/// it runs, and a child it spawns may write before init has it. +fn new_ring() -> SharedMemory { let region = SharedMemory::create(RING_BYTES).expect("init: no memory for a log ring"); + let ring = view(®ion); + ring.lay_out(); + ring.own(toyos_abi::Pid::MAX.0); + region +} + +fn view(region: &SharedMemory) -> Ring { let base = core::ptr::NonNull::new(region.as_ptr()).expect("a mapped region is not null"); - // SAFETY: `region` maps `RING_BYTES` here and stays mapped until the - // handle below is the only one left, which this function returns. - unsafe { Ring::at(base) }.lay_out(); - region.share().expect("init: a log ring would not duplicate") + // SAFETY: `region` maps `RING_BYTES` for as long as it is held, and the + // view is used only while it is. + unsafe { Ring::at(base) } } impl Log { @@ -217,7 +224,7 @@ impl Log { // kernel filled with its console. let ring = new_ring(); for slot in [1, 2] { - toyos_abi::syscall::dup2(ring, slot).expect("init: its ring would not take its own slot"); + toyos_abi::syscall::dup2(ring.as_handle(), slot).expect("init: its ring would not take its own slot"); } let (alive, keep) = toyos::pipe_pair().expect("init: no pipe to say it lives"); // Held for the machine's life: init's end is the machine's. @@ -226,12 +233,15 @@ impl Log { log } - /// Move `ring`, program `pid`'s log, to `logd` under `name`, with the read - /// end of the pipe whose only writer is that program. - fn register(&self, name: &str, pid: u32, ring: toyos::RawHandle, alive: Pipe) { + /// Name program `pid` the owner of `ring`, its log, and move the ring to + /// `logd` under `name`, with the read end of the pipe whose only writer is + /// that program. + fn register(&self, name: &str, pid: u32, ring: SharedMemory, alive: Pipe) { let Some(tag) = Tag::new(name) else { panic!("init: `{name}` is not a name a line of the log can carry"); }; + view(&ring).own(pid); + let ring = ring.share().expect("init: a log ring would not duplicate"); let mut payload = [0u8; 4 + MAX_TAG]; let len = Registration { pid, tag }.encode(&mut payload); let handles = [ring, alive.into_raw()]; @@ -1985,16 +1995,15 @@ fn start<'a>( // the rings reach it on and the one console handle that may write. A // launch keeps the slots its caller sent, but for a ring among them — // the caller's own log — which is replaced by the program's. - let mut ring: Option = None; + let mut ring: Option = None; let mut origins: Option = None; let log: &mut Log = match output { Output::Boot(log) => { - let own = new_ring(); + let own = ring.insert(new_ring()).as_handle(); command.stdin(Stdio::null()).stdout(Stdio::null()).stderr(Stdio::null()); command.inherit_handle(0, log.stdin.0); command.inherit_handle(1, own.0); command.inherit_handle(2, own.0); - ring = Some(own); if program.name == LOGD { let Some(acceptor) = log.acceptor.take() else { panic!("init: `{LOGD}` is started twice, and the log's origins are the first's"); @@ -2014,7 +2023,7 @@ fn start<'a>( && toyos_abi::syscall::fstat(handle) .is_ok_and(|stat| stat.file_type == FileType::SharedMemory); if caller_ring { - let own = *ring.get_or_insert_with(new_ring); + let own = ring.get_or_insert_with(new_ring).as_handle(); command.inherit_handle(slot, own.0); } else { command.inherit_handle(slot, handle.0); @@ -2066,9 +2075,6 @@ fn start<'a>( if let Some(acceptor) = origins { log.acceptor = Some(acceptor); } - if let Some(ring) = ring { - toyos_abi::syscall::close(ring); - } Err(e) } } diff --git a/userland/logd/src/main.rs b/userland/logd/src/main.rs index 05c4b8a1cd8..7e96ee59bce 100644 --- a/userland/logd/src/main.rs +++ b/userland/logd/src/main.rs @@ -356,17 +356,9 @@ impl Log { // SAFETY: the kernel moved this handle into this table with // the frame that names it, and nothing else answers for it. let alive = unsafe { Pipe::from_raw(alive) }; - if self.origins.len() >= MAX_ORIGINS { - toyos::warn!( - "logd: refusing {name}'s ring: {MAX_ORIGINS} programs' rings are \ - already read" - ); - toyos_abi::syscall::close(ring); - continue; - } - match Origin::open(registration.tag, registration.pid, ring, alive) { - Ok(origin) => self.origins.push(origin), - Err(why) => toyos::error!("logd: refusing a ring init sent: {why}"), + match &mut self.stall { + Some(stall) if stall.origin == name => stall.held.push((registration.pid, ring, alive)), + _ => self.take(registration.tag, registration.pid, ring, alive), } } RxStep::Frame { msg_type: SWAP, payload_len } => { @@ -394,6 +386,22 @@ impl Log { } } + /// A ring init registered, read from the next round on. + fn take(&mut self, tag: Tag<'_>, pid: u32, ring: toyos::RawHandle, alive: Pipe) { + if self.origins.len() >= MAX_ORIGINS { + toyos::warn!( + "logd: refusing {}'s ring: {MAX_ORIGINS} programs' rings are already read", + tag.as_str() + ); + toyos_abi::syscall::close(ring); + return; + } + match Origin::open(tag, pid, ring, alive) { + Ok(origin) => self.origins.push(origin), + Err(why) => toyos::error!("logd: refusing a ring init sent: {why}"), + } + } + /// One round: every ring, then the kernel's records, then everything /// stamped before the round began written in stamp order — every line /// held, whatever its stamp, where `all` asks it for a flush. A program @@ -404,9 +412,6 @@ impl Log { let mut read: Vec = Vec::new(); let mut notes: Vec = Vec::new(); for (i, origin) in self.origins.iter_mut().enumerate() { - if self.stall.as_ref().is_some_and(|s| s.origin == origin.tag) { - continue; - } let counted = origin.read(i, cut, &mut read); // The machine stops after a flush: a line begun is said as far as it got. if all { @@ -775,20 +780,19 @@ impl Log { /// End a `--stall` once any program has said its `--stall-until` line. fn release_stall(&mut self, read: &[Said]) { - let Some(stall) = &self.stall else { return }; - if read.iter().any(|r| r.text == stall.until.as_bytes()) { - let origin = stall.origin.clone(); - self.stall = None; - let (waiting, slots) = self - .origins - .iter() - .find(|o| o.tag == origin) - .map_or((0, 0), |o| o.waiting()); - say!( - "logd: reading {origin} again, as `--stall-until` asked, with {waiting} of its \ - ring's {slots} records waiting" - ); + let Some(stall) = self.stall.take_if(|s| read.iter().any(|r| r.text == s.until.as_bytes())) else { + return; + }; + let origin = stall.origin; + for (pid, ring, alive) in stall.held { + self.take(Tag::new(&origin).expect("a name init registered is a tag"), pid, ring, alive); } + let (waiting, slots) = + self.origins.iter().find(|o| o.tag == origin).map_or((0, 0), |o| o.waiting()); + say!( + "logd: reading {origin} again, as `--stall-until` asked, with {waiting} of its \ + ring's {slots} records waiting" + ); } } @@ -807,13 +811,16 @@ fn ahead_note(ahead: u64, tag: &str) -> String { struct Stall { origin: String, until: String, + /// Its rings as init registered them, taken only once the stall ends, so + /// nothing a ring's writers are kept to rests on this program's reading. + held: Vec<(u32, toyos::RawHandle, Pipe)>, } impl Stall { fn from_args() -> Option { let arg = |key: &str| std::env::args().find_map(|a| a.strip_prefix(key).map(str::to_string)); match (arg("--stall="), arg("--stall-until=")) { - (Some(origin), Some(until)) => Some(Stall { origin, until }), + (Some(origin), Some(until)) => Some(Stall { origin, until, held: Vec::new() }), (None, None) => None, (origin, until) => panic!( "logd: `--stall` and `--stall-until` are armed together or not at all, and this \ diff --git a/userland/logd/src/origin.rs b/userland/logd/src/origin.rs index 9ab20b26874..c4e9466d012 100644 --- a/userland/logd/src/origin.rs +++ b/userland/logd/src/origin.rs @@ -173,7 +173,6 @@ impl Origin { if !ring.is_laid_out() { return Err(format!("{}'s region is not a log ring", tag.as_str())); } - ring.own(pid); Ok(Self { tag: tag.as_str().to_string(), pid, From d565ff4189b0ed9fa1e58aac13effd39503a328d Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 20:33:16 +0200 Subject: [PATCH 30/54] review #536: the three clippy lints, region.rs's two stale comments, and a host test for the placeholder owner MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `--clippy` was red at c86dd840: `let_and_return` in fat32's repair drive, `manual_is_multiple_of` and `chunks_exact_to_as_chunks` in blockring's `Listed::decode_all`. Fixing them let clippy reach two targets its earlier compile errors had hidden entirely — `tests/toyos.rs`'s two `single_element_loop` probes and `tests/common/usb.rs`'s one `redundant_clone` — both already on this branch, from 64c59c96; `--clippy` now exits 0 at ten of ten invocations. `region.rs`'s `OWNER_AT` and `Ring::own` docs described logd naming the owner; c86dd840 moved that to init at the spawn, and the clauses about logd and about a reader's word to say are false. Deleted rather than rewritten. init's placeholder owner (`Pid::MAX`, c86dd840) had no host test: the guest harness cannot hold init between the spawn and the naming. `new_ring`'s two calls (`lay_out`, `own(Pid::MAX.0)`) are now `Ring::lay_out_with_placeholder_owner`, one function both init and the new test call, so the placeholder rule lives in one place. The test fills a freshly laid-out ring's kept slots and asserts a next write from any pid is refused until `own` names the real one, which then takes them. Arms, `cd toyos && cargo test --lib log::proof::a_placeholder_owned_ring_keeps_its_slots_from_every_pid_until_named`: placeholder owner (`Pid::MAX`) EXIT=0; mutated to the old unowned state (`own(0)`) EXIT=101, red on the kept-slot assertion — confirming the test depends on the placeholder rather than passing regardless. Gates: `cargo test --lib` EXIT=0 (389 passed); `cargo test --workspace --exclude toyos-build` EXIT=0 (174 suites ok); `cargo run -- --clippy` EXIT=0 (10 of 10 invocations clean). Co-Authored-By: Claude Opus 5.5 --- tests/common/usb.rs | 2 +- tests/toyos.rs | 30 ++++++++++++++---------------- toyos-blockring/src/wire.rs | 5 +++-- toyos-fat32/src/repair.rs | 5 ++--- toyos/src/log/proof.rs | 26 +++++++++++++++++++++++++- toyos/src/log/region.rs | 14 +++++++++++--- userland/init/src/main.rs | 4 +--- 7 files changed, 57 insertions(+), 29 deletions(-) diff --git a/tests/common/usb.rs b/tests/common/usb.rs index 4b5c440185c..9dc0daaaed9 100644 --- a/tests/common/usb.rs +++ b/tests/common/usb.rs @@ -2149,7 +2149,7 @@ fn a_stick_its_reset_moved_carries_on(moved: Moved) -> Result<(), String> { let mut want = left.to_vec(); want.extend(back.iter().cloned()); want.push(OWED.to_string()); - want.push(log_told.clone()); + want.push(log_told); in_order(&want)?; for never in [" did not come back within ", "disk 1 ready", " is not disk 0 come back"] { if let Some(line) = log.lines().find(|l| l.contains(never)) { diff --git a/tests/toyos.rs b/tests/toyos.rs index 24755dba09f..c2e63faa2a0 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -16416,15 +16416,14 @@ fn process_reopen(log: &str) -> Result<(), String> { /// Text in, a verdict out: every line it reads is a kernel record, so the /// T14's readback and a QEMU boot log are judged by this one predicate. fn read_fault_probes(log: &str) -> Result<(), String> { - for probe in ["revoke-selftest: /tmp/revoke_probe"] { - let Some(verdict) = log.lines().find(|l| l.contains(probe)) else { - return Err(format!("{probe} never ran:\n{log}")); - }; - if !verdict.contains("PASS") { - return Err(format!("{}\n{log}", verdict.trim())); - } - eprintln!(" [read-fault] {}", verdict.trim()); + let probe = "revoke-selftest: /tmp/revoke_probe"; + let Some(verdict) = log.lines().find(|l| l.contains(probe)) else { + return Err(format!("{probe} never ran:\n{log}")); + }; + if !verdict.contains("PASS") { + return Err(format!("{}\n{log}", verdict.trim())); } + eprintln!(" [read-fault] {}", verdict.trim()); Ok(()) } @@ -16433,15 +16432,14 @@ fn read_fault_probes(log: &str) -> Result<(), String> { /// Text in, a verdict out: every line it reads is a kernel record, so the /// T14's readback and a QEMU boot log are judged by this one predicate. fn leak_rollback(log: &str) -> Result<(), String> { - for probe in ["leak-selftest: device-mint"] { - let Some(verdict) = log.lines().find(|l| l.contains(probe)) else { - return Err(format!("{probe} never ran:\n{log}")); - }; - if !verdict.contains("PASS") { - return Err(format!("{}\n{log}", verdict.trim())); - } - eprintln!(" [leak] {}", verdict.trim()); + let probe = "leak-selftest: device-mint"; + let Some(verdict) = log.lines().find(|l| l.contains(probe)) else { + return Err(format!("{probe} never ran:\n{log}")); + }; + if !verdict.contains("PASS") { + return Err(format!("{}\n{log}", verdict.trim())); } + eprintln!(" [leak] {}", verdict.trim()); Ok(()) } diff --git a/toyos-blockring/src/wire.rs b/toyos-blockring/src/wire.rs index d92241c4031..bc3e53c4624 100644 --- a/toyos-blockring/src/wire.rs +++ b/toyos-blockring/src/wire.rs @@ -73,10 +73,11 @@ impl Listed { /// Every entry of a listing, or `None` for one that is not whole entries. pub fn decode_all(bytes: &[u8]) -> Option + '_> { - if bytes.len() % Self::BYTES != 0 { + if !bytes.len().is_multiple_of(Self::BYTES) { return None; } - Some(bytes.chunks_exact(Self::BYTES).map(|c| Self { + let (chunks, _) = bytes.as_chunks::<{ Self::BYTES }>(); + Some(chunks.iter().map(|c| Self { unique: c[..GUID_BYTES].try_into().expect("sixteen bytes"), kind: c[GUID_BYTES..].try_into().expect("sixteen bytes"), })) diff --git a/toyos-fat32/src/repair.rs b/toyos-fat32/src/repair.rs index 5001b9c291e..a2af2fb2e04 100644 --- a/toyos-fat32/src/repair.rs +++ b/toyos-fat32/src/repair.rs @@ -99,7 +99,7 @@ impl Fat32 { let result = op(self); self.in_call = false; let committed = core::mem::take(&mut self.committed); - let answer = match result { + match result { // What the repair holds past a commit is the call's own remaining // work, driven twice like a call and its re-drive; a step the // device still refuses stays queued for the next call to finish @@ -122,8 +122,7 @@ impl Fat32 { // An unlanded re-drive is what the volume is waiting on, so it is // the answer rather than the error that started the rollback. Err(e) => self.settle().map_or_else(Err, |()| Err(e)), - }; - answer + } } /// Mark the call committed: its rollback is discarded, and what it queues diff --git a/toyos/src/log/proof.rs b/toyos/src/log/proof.rs index 071a04fb624..23811def935 100644 --- a/toyos/src/log/proof.rs +++ b/toyos/src/log/proof.rs @@ -13,7 +13,7 @@ use std::sync::Arc; use std::time::{Duration, Instant}; use std::vec::Vec; -use super::region::{Body, Ring, LANE_SLOTS, RING_BYTES, SHARED_SLOTS}; +use super::region::{Body, Ring, CHILD_KEEP, LANE_SLOTS, RING_BYTES, SHARED_SLOTS}; use super::ring::{push, push_lane, push_leaving, Pushed, Reader, Shared, Slots}; use super::stdio::compose; use toyos_abi::log::Severity; @@ -362,3 +362,27 @@ fn a_ring_has_four_lanes_to_claim() { } assert!(ring.claim_lane(1, 4).is_none()); } + +/// A ring laid out exactly as `init`'s `new_ring` lays each program's out — +/// through [`Ring::lay_out_with_placeholder_owner`], the same call — keeps +/// its owner's slots from every pid, because none of them is the placeholder, +/// until the real owner is named. +#[test] +fn a_placeholder_owned_ring_keeps_its_slots_from_every_pid_until_named() { + let mut words = std::vec![0u64; RING_BYTES / 8]; + let base = core::ptr::NonNull::new(words.as_mut_ptr() as *mut u8).unwrap(); + // SAFETY: as for `region`. + let ring = unsafe { Ring::at(base) }; + ring.lay_out_with_placeholder_owner(); + + let body = |pid| Body { pid, ..Body::EMPTY }; + let limit = (SHARED_SLOTS - CHILD_KEEP) as usize; + for _ in 0..limit { + assert_eq!(ring.push(&body(7)), Pushed::Written); + } + assert_eq!(ring.push(&body(7)), Pushed::Refused, "the placeholder still keeps its owner's slots"); + assert_eq!(ring.push(&body(9)), Pushed::Refused, "no pid is the placeholder's owner"); + + ring.own(7); + assert_eq!(ring.push(&body(7)), Pushed::Written, "the named owner may take its kept slots"); +} diff --git a/toyos/src/log/region.rs b/toyos/src/log/region.rs index f0a5aea11a5..fceb5943d1e 100644 --- a/toyos/src/log/region.rs +++ b/toyos/src/log/region.rs @@ -49,8 +49,7 @@ pub const MAGIC: u64 = 0x544F_594F_534C_4F47; // "TOYOSLOG" /// The header's words, each group a cache line of its own: writers hammer /// `head` and `refused`, the reader `tail`. const MAGIC_AT: usize = 0; -/// The process init started with the ring, as `logd` registered it: zero -/// until then. +/// The process init started with the ring. const OWNER_AT: usize = 8; const HEAD_AT: usize = 64; const TAIL_AT: usize = 128; @@ -189,8 +188,17 @@ impl Ring { unsafe { core::ptr::read_volatile(self.at_offset(MAGIC_AT) as *const u64) == MAGIC } } + /// [`Self::lay_out`], owned by a pid no process holds: until the real + /// owner is named, every writer is a stranger to it and [`Self::push`] + /// leaves it the kept slots rather than the zero it would leave an + /// owner it could confuse for none. + pub fn lay_out_with_placeholder_owner(&self) { + self.lay_out(); + self.own(toyos_abi::Pid::MAX.0); + } + /// Name the process the ring is its log of, which [`Self::push`] keeps - /// the ring's last slots for. The reader's to say, once it knows. + /// the ring's last slots for. pub fn own(&self, pid: u32) { self.word(OWNER_AT).store(u64::from(pid), Ordering::Relaxed); } diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index 56bdf2b28b4..79a7879c9e8 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -190,9 +190,7 @@ const STDIN_RIGHTS: Rights = /// it runs, and a child it spawns may write before init has it. fn new_ring() -> SharedMemory { let region = SharedMemory::create(RING_BYTES).expect("init: no memory for a log ring"); - let ring = view(®ion); - ring.lay_out(); - ring.own(toyos_abi::Pid::MAX.0); + view(®ion).lay_out_with_placeholder_owner(); region } From e318afc6aa2d3647c139964712a9260c399ba408 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 20:50:47 +0200 Subject: [PATCH 31/54] kernel: the two task probes run on the boot's first syscall, not a thread The kernel creates no thread except the per-CPU idle loop, and K5 deletes iod without a replacement: the `probe` kernel thread this branch spawned to host `sched-operation-nesting` and `sysret-ss-probe` goes. Both need a task, and a user thread in a syscall is one. The first syscall of the boot, init's, runs them once: `sched_gate::run("syscall")` in the task's own deadline slot, and the SS probe's park, which switches away from that task and back before its `sysretq`. The boot half of the nesting gate stays where it was. `sysret_ss_reload` judges the boot log alone: init's first syscall precedes every program, so the probe's line precedes the ready marker, and the 10 s drain it waited on after the line had already arrived is gone. Co-Authored-By: Claude Opus 5.5 --- kernel/src/actuator.rs | 2 +- kernel/src/main.rs | 17 ----------------- kernel/src/sched/kthread.rs | 2 +- kernel/src/sched_gate.rs | 4 ++-- kernel/src/syscall/dispatch.rs | 22 ++++++++++++++++++++++ src/metal.rs | 2 +- tests/toyos.rs | 20 ++++++-------------- 7 files changed, 33 insertions(+), 36 deletions(-) diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index b2a52c0ce31..c1d6342039c 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -61,7 +61,7 @@ actuators! { /// record before it and `screen_early_panel` reads a paint it can attribute. test_early_halt = "test-early-halt"; - /// Have the `probe` thread null SS, force a switch, and report whether it reloaded — the + /// Have the first syscall null SS, force a switch, and report whether it reloaded — the /// AMD `SYSRET` SS-attributes workaround's only guest-observable proof. sysret_ss_probe = "sysret-ss-probe"; diff --git a/kernel/src/main.rs b/kernel/src/main.rs index a90d46b6618..dc2b30a8dc8 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -588,10 +588,6 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { log::console::start(); // After klogd so their own spawn logs have a drainer. drivers::xhci::usbd::start(); - #[cfg(feature = "boot-actuators")] - if actuator::sched_operation_nesting() || actuator::sysret_ss_probe() { - let _ = sched::kthread::spawn("probe", probe, 0, sched::kthread::OnPanic::Recover); - } smp::set_ready(); @@ -616,16 +612,3 @@ fn pre_idle_wedge() -> ! { } } -/// The actuators judged on a kernel thread of its own, idle at its entry: its -/// task word, and a switch away from it. Spawned only when one is armed. -#[cfg(feature = "boot-actuators")] -extern "C" fn probe(_arg: u64) -> ! { - let parkable = scheduler::Parkable::at_entry(); - if actuator::sched_operation_nesting() { - sched_gate::run("probe"); - } - if actuator::sysret_ss_probe() { - hw::sysret_ss_probe(&parkable); - } - watch::park_forever() -} diff --git a/kernel/src/sched/kthread.rs b/kernel/src/sched/kthread.rs index f6cd1a87fb5..05b98e72b77 100644 --- a/kernel/src/sched/kthread.rs +++ b/kernel/src/sched/kthread.rs @@ -19,7 +19,7 @@ use crate::sync::Lock; use super::payload::ThreadSched; -/// `klogd` and `usbd`, plus `probe` and one `log-storm` thread per shard in the actuator build. +/// `klogd` and `usbd`, plus `lognest` and one `log-storm` thread per shard in the actuator build. #[cfg(not(feature = "boot-actuators"))] const MAX_KERNEL_TASKS: usize = 2; #[cfg(feature = "boot-actuators")] diff --git a/kernel/src/sched_gate.rs b/kernel/src/sched_gate.rs index 0f4460ea971..b8b4e96cba3 100644 --- a/kernel/src/sched_gate.rs +++ b/kernel/src/sched_gate.rs @@ -1,7 +1,7 @@ //! In-guest gate on [`Operation`] nesting: `begin` narrows the current //! deadline and never widens it, and `Drop` restores what it displaced. -//! [`run`] is called from a boot-phase CPU context and from the `probe` task, -//! the two places a deadline slot can be established. +//! [`run`] is called from a boot-phase CPU context and from a syscall, the two +//! places a deadline slot can be established. use crate::clock; use crate::scheduler::Operation; diff --git a/kernel/src/syscall/dispatch.rs b/kernel/src/syscall/dispatch.rs index c6bef4b2c0f..6ba753f0292 100644 --- a/kernel/src/syscall/dispatch.rs +++ b/kernel/src/syscall/dispatch.rs @@ -97,10 +97,32 @@ retired_syscalls! { 96 => "SYS_SET_RT_PRIORITY", } +/// `sched-operation-nesting`'s task half and `sysret-ss-probe`, run once a boot +/// on the first syscall: a task's own deadline slot, and a park that switches +/// away from it and back. +#[cfg(feature = "boot-actuators")] +fn task_probes() { + use core::sync::atomic::{AtomicBool, Ordering::Relaxed}; + static RAN: AtomicBool = AtomicBool::new(false); + let nesting = crate::actuator::sched_operation_nesting(); + let ss = crate::actuator::sysret_ss_probe(); + if !(nesting || ss) || RAN.swap(true, Relaxed) { + return; + } + if nesting { + crate::sched_gate::run("syscall"); + } + if ss { + crate::hw::sysret_ss_probe(&crate::scheduler::Parkable::at_entry()); + } +} + pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> u64 { // Placed first so the architecture counts the call whatever it turns out to be. #[cfg(feature = "boot-actuators")] crate::arch::syscall::note_entry(); + #[cfg(feature = "boot-actuators")] + task_probes(); let t0 = crate::clock::nanos_since_boot(); process::with_current_data(|data| { diff --git a/src/metal.rs b/src/metal.rs index 5b0e2092e66..c41d87ad209 100644 --- a/src/metal.rs +++ b/src/metal.rs @@ -763,7 +763,7 @@ pub const FLASHABLE: &[(&str, Flash)] = &[ ("xhci-xecp-selftest", Flash::Ok), ("xhci-descriptor-selftest", Flash::Ok), // Two probes rather than staged inputs, and both are reads: the SS-reload - // one runs inside the `probe` thread's own context switch, and the input-core one merges + // one runs inside the first syscall's own context switch, and the input-core one merges // events it made up itself. ("sysret-ss-probe", Flash::Ok), ("test-input-merge", Flash::Ok), diff --git a/tests/toyos.rs b/tests/toyos.rs index c2e63faa2a0..2ef14ccc4d3 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -11470,24 +11470,16 @@ fn run_machine_test( // Same again: the FAT32 read side's revocation, judged off the volume the // guest's unlink-and-reallocate cycle left behind. "fat_backing_revoked" => common::volumes::fat_backing_revoked(test_config, c_bins, rust_bins), - // `sysret-ss-probe` has the `probe` thread null SS, force a switch, and log whether the + // `sysret-ss-probe` has the first syscall null SS, force a switch, and log whether the // switch reloaded it; a missing `mov ss` turns `reloaded` into `NOT`. "sysret_ss_reload" => { let options = BootOptions { kernel_params: &["sysret-ss-probe"], ..Default::default() }; - let mut qemu = - QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); - // A liveness ceiling, not a pace: a loaded shard once took past a - // fixed 500 ms drain to run the probe (run 33246638742, alone-green). - // The T14's readback needs no drain at all — the whole boot's records - // are on the stick — so the wait is here and the predicate is shared. - let log = qemu.boot_log().to_string() - + &qemu.drain_until(Duration::from_secs(10), |l| { - l.contains("sysret-ss: reloaded") || l.contains("sysret-ss: NOT reloaded") - }); - sysret_ss(&log) + // The first syscall is init's, so the probe's line precedes the ready marker. + let qemu = QemuInstance::boot_with_options(test_config, c_bins, rust_bins, options); + sysret_ss(qemu.boot_log()) } "fsync_failed_commit" => common::volumes::fsync_failed_commit(test_config, c_bins, rust_bins), "redirty_mid_flush" => common::volumes::redirty_mid_flush(test_config, c_bins, rust_bins), @@ -16598,7 +16590,7 @@ fn sysret_ss(log: &str) -> Result<(), String> { } if !log.contains("sysret-ss: reloaded") { return Err(format!( - "the SS-reload probe never reported — the probe thread may not have run it:\n{log}" + "the SS-reload probe never reported — the first syscall may not have run it:\n{log}" )); } eprintln!(" [sysret-ss] the switch reloads SS from null before a sysretq can see it"); @@ -16639,7 +16631,7 @@ fn operation_nesting_log(log: &str) -> Result<(), String> { // with no task uses one slot per CPU, and the two are reached by // different arms of `operation_slot` — so a gate that ran in one // place would leave the other arm unexecuted by any test at all. - for site in ["boot", "probe"] { + for site in ["boot", "syscall"] { let say = |what: &str| -> Result { let needle = format!("sched-op: {site} {what}"); log.lines() From 9b281cef947a2fa0ffa68fd7a33db55cc19f3dfa Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 20:56:19 +0200 Subject: [PATCH 32/54] tests/CLAUDE.md: the iod drain caveat goes with iod The kernel no longer caches file data or writes it back, so a closed file has no queue for a test to wait out. Co-Authored-By: Claude Opus 5.5 --- tests/CLAUDE.md | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/CLAUDE.md b/tests/CLAUDE.md index db0d9d8f609..a47ecc6ca7b 100644 --- a/tests/CLAUDE.md +++ b/tests/CLAUDE.md @@ -14,7 +14,6 @@ The mechanics live where the work is: profiles and shapes in `tests/common/`, re - **`/system/bin/init` speaks in every program's name before that program runs** — a predicate keyed on a `: ` prefix is satisfied by the wrong speaker; wait for the whole line, in the constant the assertion also reads. - **A guest binary cannot ask what a handle it does not hold does** — the probe ends its caller with exit 139, so it runs in a child, one fault per child; `handle_kill_policy` is the pattern. - **A boot's capture has two pieces** — `boot_log()` ends at the ready marker, `run_test`'s capture begins at `===TEST_START===`, and the lines between land in `TestResult::before`; a test reading a daemon's boot line appends it. -- **A file's last close does not flush it synchronously** — a test that reads the backing device behind the kernel's back first lets `iod` drain; a spawn or `dlopen` of a just-closed file needs no drain (`Vfs::open_backing` settles the queue itself). - **Every guest this host boots is TCG** — anything vendor-dependent is gated only by CI's KVM shards, and TCG prices an uncontended atomic read-modify-write unlike hardware. - **CI's `guest` lane is GitHub-hosted shards, never the T14** — the T14 is the orchestrator's own metal loop (`src/metal.rs`), reached by nothing in `.github/workflows/`; `tests/test-durations` is hosted, and a duration measured on the T14 does not transfer. - **The dev host's guests boot `-cpu qemu64`, which has no PCID** — every `INVPCID` path is dead locally, so a change gated on a CPUID feature is unverified by a green local suite. From 1bd2ef43229d6d81357028987173c4ac17fb8c59 Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 22:17:46 +0200 Subject: [PATCH 33/54] review #536 r7: init's restart, DATA's lost writes, one blockd loop, the BAR spawn arm init can no longer wedge the machine on a file server it restarts. std's `capability()` held the process-wide capability table's lock across the connect, which waits for the server's hello; a DATA server that ended before it accepted left init's worker holding that lock, and the restart's `is_served` waited on it for good. The fork connects outside the lock (`wt-toyos-fsd` 9abdd61d95a). `fsd_end_at_mount` stages it: fsd's `--end-at-mount data` ends DATA's first server, once a boot, when a connection waits on it and before it accepts one. DATA keeps every write it accepts. A file's extents live in its one btree entry, at most 4064 bytes: 250 runs for `home/a`. Two causes filled it in normal use: - the allocator handed each page the block after the last one handed out, so two files growing in turn got one run per page and passed 250 runs at about 1 MiB; a file now extends in place when the block after its last is free, and looks 256 blocks further on when another file took it; - a two-way leaf split had no point that fits a large entry landing between small ones (`NodeOverfull`, which the code named as the shape a three-way split fixes); leaves now split into as many nodes as their entries fill, in order. What remains is refused whole at the write: a write whose runs the entry could not name is `ResourceExhausted`, said by name, and its allocations go back. `sync` writes every file it can and names the ones it could not, so an fsync fails only for its own file; a close whose entry would not write answers the error and keeps the node for the next sync, which the write-back retries. FAT's volume follows the same contract. blockd has one accept and handshake loop. `serve_nothing` answered a listing with a payload and a malformed open as no refusal and `NotFound`; the controller's absence is now `Drive::Absent` or `Drive::Unusable` in the one `serve`, and `blockd_serves_nothing` asks it each first frame. `blockd_io dma-inside` spawns from the claimed function's BAR object, with an argv no process can read: `InvalidArgument` is the image refused before the argv is read, and the same spawn from a RAM region reaches the argv (`BadAddress`). The three Fast-tier tests that waited for `Syncing filesystems...` wait for the stop's record, which `toyos_quiesce::Record::parse` reads. logd's `WouldBlock` flush arm had no producer: it goes, with `Fate::Retry`, the retry run, `State::Retrying`, their tests, the prose for them and the issue. `fs_turns`' floor is one pass, the barging lock's own signature. The cached-read issue is a defect; the client-slot issue covers streams. Co-Authored-By: Claude Opus 5.5 --- bcachefs/src/alloc_bitmap.rs | 26 ++- bcachefs/src/btree.rs | 168 +++++++------- bcachefs/src/fs.rs | 34 ++- bcachefs/src/lib.rs | 2 +- ...olchains-rebuild-and-its-cargo-has-none.md | 27 +++ ...-file-server-is-measured-only-under-tcg.md | 2 +- ...urns-judges-the-scheduler-with-the-lock.md | 6 +- ...ould-block-flush-policy-has-no-producer.md | 20 -- ...can-take-every-file-servers-client-slot.md | 15 +- kernel/src/actuator.rs | 2 +- kernel/src/log/mod.rs | 4 +- rust | 2 +- src/build.rs | 1 + tests/common/blockd.rs | 19 ++ tests/common/power.rs | 36 +-- tests/common/storage.rs | 49 ++++ tests/common/usb.rs | 3 +- tests/fsdmountcase/system.toml | 34 +++ tests/toyos-rust-tests/src/bin/blockd_io.rs | 80 ++++++- tests/toyos-rust-tests/src/bin/fs_turns.rs | 6 +- tests/toyos.rs | 22 +- userland/blockd/src/main.rs | 154 ++++++------- userland/fsd/src/absent.rs | 8 +- userland/fsd/src/data.rs | 212 ++++++++++++++---- userland/fsd/src/fat.rs | 40 ++-- userland/fsd/src/main.rs | 94 +++++--- userland/fsd/src/volume.rs | 13 +- userland/logd/src/inspect.rs | 7 +- userland/logd/src/main.rs | 67 ++---- userland/logd/src/policy.rs | 179 ++------------- userland/netd/src/report.rs | 5 +- 31 files changed, 783 insertions(+), 554 deletions(-) create mode 100644 issues/build/a-sysroot-made-between-the-toolchains-rebuild-and-its-cargo-has-none.md delete mode 100644 issues/filesystem/logds-would-block-flush-policy-has-no-producer.md create mode 100644 tests/fsdmountcase/system.toml diff --git a/bcachefs/src/alloc_bitmap.rs b/bcachefs/src/alloc_bitmap.rs index 7fb7941ff48..c6a5db186f9 100644 --- a/bcachefs/src/alloc_bitmap.rs +++ b/bcachefs/src/alloc_bitmap.rs @@ -82,21 +82,30 @@ impl BitmapAllocator { Ok(self.alloc_exact(io, 1)?.start) } - /// Reserve as much of `wanted` as one contiguous run can cover. + /// Whether `block` is free. + pub fn is_free(&self, io: &dyn BlockIO, block: BlockNum) -> Result { + let (bitmap_block, byte_off, bit) = self.bit_of(block); + let mut buf = BlockBuf::zeroed(); + io.read(bitmap_block, &mut buf)?; + Ok(buf.0[byte_off] & (1 << bit) == 0) + } + + /// Reserve as much of `wanted` as one contiguous run can cover, scanning + /// from block `from`. /// /// The run is never empty and may be shorter than asked for, so every /// caller has to loop or has to be wrong. - pub fn alloc_up_to(&mut self, io: &dyn BlockIO, wanted: u32) -> Result { + pub fn alloc_up_to(&mut self, io: &dyn BlockIO, from: u64, wanted: u32) -> Result { // A zero-length run would let a caller's loop spin without progress. let wanted = wanted.max(1); - let (start, len) = self.longest_free_run(io, wanted)?; + let (start, len) = self.longest_free_run(io, from % self.total_blocks, wanted)?; self.reserve(io, start, len.min(wanted)) } /// Reserve all of `count` or nothing, for callers that cannot place a /// short run. Nothing is marked used unless the whole run is there. pub fn alloc_exact(&mut self, io: &dyn BlockIO, count: u32) -> Result { - let (start, len) = self.longest_free_run(io, count)?; + let (start, len) = self.longest_free_run(io, self.next_alloc, count)?; if len < count { return Err(FsError::NoSpace { requested: count, @@ -122,9 +131,9 @@ impl BitmapAllocator { Ok(Run { start: start_block, len }) } - /// The longest free run found scanning from the `next_alloc` cursor, - /// wrapping once, stopping early once `wanted` blocks are in hand. - fn longest_free_run(&self, io: &dyn BlockIO, wanted: u32) -> Result<(u64, u32), FsError> { + /// The longest free run found scanning from block `start_pos`, wrapping + /// once, stopping early once `wanted` blocks are in hand. + fn longest_free_run(&self, io: &dyn BlockIO, start_pos: u64, wanted: u32) -> Result<(u64, u32), FsError> { if self.free_blocks == 0 { return Err(FsError::NoSpace { requested: wanted, @@ -133,11 +142,10 @@ impl BitmapAllocator { } let total = self.total_blocks; - let start_pos = self.next_alloc; let mut best_start = None; let mut best_count = 0u32; - // Scan from cursor, wrap once + // Scan from start_pos, wrap once let mut pos = start_pos; let mut wrapped = false; let mut run_start = None; diff --git a/bcachefs/src/btree.rs b/bcachefs/src/btree.rs index fda512d334e..1050772da8e 100644 --- a/bcachefs/src/btree.rs +++ b/bcachefs/src/btree.rs @@ -129,11 +129,20 @@ pub struct Entry { impl Entry { /// Total size on disk: key header + value, padded to 8-byte alignment. pub fn disk_size(&self) -> usize { - let raw = KEY_HEADER_SIZE + self.value.len(); - (raw + 7) & !7 + disk_size(self.value.len()) } } +fn disk_size(value_len: usize) -> usize { + (KEY_HEADER_SIZE + value_len + 7) & !7 +} + +/// Whether an entry whose value is `value_len` bytes passes +/// [`check_entry_fits`], asked before the value is built. +pub fn value_fits(value_len: usize) -> bool { + disk_size(value_len) <= MAX_ENTRY_SIZE +} + /// A child pointer: the minimum key of the subtree, and the block it lives in. #[derive(Debug, Clone, Copy)] pub struct Child { @@ -501,7 +510,7 @@ pub fn insert( match insert_recursive(io, alloc, root, Depth::ROOT, entry)? { InsertResult::Done => Ok(root), - InsertResult::Split { new_block, split_key } => { + InsertResult::Split(siblings) => { let level = Node::read(io, root)? .level() .checked_add(1) @@ -509,14 +518,9 @@ pub fn insert( let old_min_key = min_key(io, root, Depth::ROOT)?; let new_root_block = alloc.alloc_block(io)?; - let new_root = Node::Interior { - level, - children: alloc::vec![ - Child { key: old_min_key, block: root }, - Child { key: split_key, block: new_block }, - ], - }; - new_root.write(io, new_root_block)?; + let mut children = alloc::vec![Child { key: old_min_key, block: root }]; + children.extend(siblings); + Node::Interior { level, children }.write(io, new_root_block)?; Ok(new_root_block) } @@ -525,10 +529,8 @@ pub fn insert( enum InsertResult { Done, - Split { - new_block: BlockNum, - split_key: Key, - }, + /// The node split: these follow it, in key order. + Split(Vec), } fn insert_recursive( @@ -560,13 +562,15 @@ fn insert_recursive( match insert_recursive(io, alloc, child_block, deeper, entry)? { InsertResult::Done => Ok(InsertResult::Done), - InsertResult::Split { new_block, split_key } => { + InsertResult::Split(siblings) => { let mut children = children; - let pos = match children.binary_search_by(|c| c.key.cmp(&split_key)) { - Ok(i) => i + 1, - Err(i) => i, - }; - children.insert(pos, Child { key: split_key, block: new_block }); + for sibling in siblings { + let pos = match children.binary_search_by(|c| c.key.cmp(&sibling.key)) { + Ok(i) => i + 1, + Err(i) => i, + }; + children.insert(pos, sibling); + } write_or_split(io, alloc, block, Node::Interior { level, children }) } } @@ -594,7 +598,7 @@ fn split_node( node: Node, ) -> Result { match node { - Node::Leaf(mut entries) => { + Node::Leaf(entries) => { // One entry is not a split problem. Halving by *count* used to // produce `mid == 0` here, which drained every entry into the right // node and left an empty one behind — and the right node was still @@ -604,37 +608,30 @@ fn split_node( return Err(FsError::EntryTooLarge { size, max: MAX_ENTRY_SIZE }); } - // By size, not by count: leaf entries are variable-length (a file's - // extent list lives inline), so half the entries can be far more - // than half the bytes. - let mid = split_point(&entries); - - // Both halves are checked before either is written. A split that - // has already replaced the left node on disk and then fails is a - // corrupt tree; a split that fails before writing is an error the - // caller can return. - if leaf_size(&entries[..mid]) > BLOCK_SIZE || leaf_size(&entries[mid..]) > BLOCK_SIZE { - // Unreachable while every entry is <= MAX_ENTRY_SIZE and the - // node was legal before this insert, except for one shape: a - // node of large entries where the new one lands in the middle. - // Splitting three ways is what would fix it; extent merging is - // what stops values getting near that size in the first place. - return Err(FsError::NodeOverfull { - used: leaf_size(&entries) - NODE_HEADER_SIZE, - max: MAX_PAYLOAD, - }); + let mut nodes = pack(entries); + let first = nodes.remove(0); + let mut blocks = Vec::with_capacity(nodes.len()); + for _ in &nodes { + match alloc.alloc_block(io) { + Ok(sibling) => blocks.push(sibling), + Err(e) => { + for taken in blocks { + alloc.free_range(io, taken, 1)?; + } + return Err(e); + } + } } + // The siblings first: a failure before `block` is replaced leaves + // the tree as it was and blocks nothing names. + let mut children = Vec::with_capacity(nodes.len()); + for (node, sibling) in nodes.into_iter().zip(blocks) { + children.push(Child { key: node[0].key, block: sibling }); + Node::Leaf(node).write(io, sibling)?; + } + Node::Leaf(first).write(io, block)?; - let right: Vec = entries.drain(mid..).collect(); - let Some(split_key) = right.first().map(|e| e.key) else { - return Err(FsError::CorruptedNode(block)); - }; - - let right_block = alloc.alloc_block(io)?; - Node::Leaf(entries).write(io, block)?; - Node::Leaf(right).write(io, right_block)?; - - Ok(InsertResult::Split { new_block: right_block, split_key }) + Ok(InsertResult::Split(children)) } Node::Interior { level, mut children } => { if children.len() < 2 { @@ -652,25 +649,27 @@ fn split_node( Node::Interior { level, children }.write(io, block)?; Node::Interior { level, children: right }.write(io, right_block)?; - Ok(InsertResult::Split { new_block: right_block, split_key }) + Ok(InsertResult::Split(alloc::vec![Child { key: split_key, block: right_block }])) } } } -/// The largest prefix of `entries` that still fits in a node, clamped so both -/// sides of the split get at least one entry. Caller guarantees `len >= 2`. -fn split_point(entries: &[Entry]) -> usize { - let mut used = NODE_HEADER_SIZE; - let mut n = 0; +/// `entries` in order, each node filled before the next is begun. Every entry +/// fits a node alone ([`check_entry_fits`]), so a large entry that lands +/// between small ones takes a node of its own: a split in two has no point +/// that leaves both sides within a block for that shape. +fn pack(entries: Vec) -> Vec> { + let mut nodes: Vec> = Vec::new(); + let mut used = BLOCK_SIZE; for entry in entries { - let next = used + entry.disk_size(); - if next > BLOCK_SIZE { - break; + if used + entry.disk_size() > BLOCK_SIZE { + nodes.push(Vec::new()); + used = NODE_HEADER_SIZE; } - used = next; - n += 1; + used += entry.disk_size(); + nodes.last_mut().expect("a node was begun").push(entry); } - n.clamp(1, entries.len() - 1) + nodes } /// Find the minimum key in a subtree. @@ -715,37 +714,24 @@ mod tests { buf } - #[test] - fn split_point_is_the_largest_prefix_that_fits() { - // Two entries that only fit apart: the rule has to put one on each - // side, which halving by count also gets right. - let two = [entry(3000), entry(3000)]; - assert_eq!(split_point(&two), 1); - - // And the shape it does not: a small entry ahead of two large ones. - // Halving by count gives mid=1, leaving 6048 bytes of entries in the - // right node and a block that cannot hold them. - let skewed = [entry(1000), entry(3000), entry(3000)]; - let mid = split_point(&skewed); - assert_eq!(mid, 2); - assert!(leaf_size(&skewed[..mid]) <= BLOCK_SIZE, "left half does not fit"); - assert!(leaf_size(&skewed[mid..]) <= BLOCK_SIZE, "right half does not fit"); - assert!(leaf_size(&skewed[..2]) > BLOCK_SIZE / 2, "the shape under test is not skewed"); + fn counts(nodes: &[Vec]) -> Vec { + nodes.iter().map(Vec::len).collect() } #[test] - fn split_point_always_leaves_both_sides_a_entry() { - // The clamp matters at both ends. One entry so large that no prefix - // fits must still yield 1, not 0 — a 0 drains every entry into the - // right node and writes an empty one back. - let huge_first = [entry(MAX_ENTRY_SIZE), entry(16)]; - assert_eq!(split_point(&huge_first), 1); - - // And a node of entries that all fit must still give the right side - // something, or the split makes no progress. - let tiny = [entry(8), entry(8), entry(8)]; - let mid = split_point(&tiny); - assert!((1..=2).contains(&mid), "mid={mid} leaves a side empty"); + fn pack_fills_each_node_in_order_and_none_past_a_block() { + // Two entries that only fit apart. + assert_eq!(counts(&pack(vec![entry(3000), entry(3000)])), [1, 1]); + // A small entry ahead of two large ones, which halving by count + // leaves 6048 bytes in one node. + assert_eq!(counts(&pack(vec![entry(1000), entry(3000), entry(3000)])), [2, 1]); + // The largest entry between small ones, which no split in two fits: + // a leaf of names where one file's entry has grown. + let mut middle: Vec = (0..40).map(|_| entry(72)).collect(); + middle.insert(20, entry(MAX_ENTRY_SIZE - KEY_HEADER_SIZE)); + let nodes = pack(middle); + assert_eq!(counts(&nodes), [20, 1, 20]); + assert!(nodes.iter().all(|n| leaf_size(n) <= BLOCK_SIZE)); } #[test] diff --git a/bcachefs/src/fs.rs b/bcachefs/src/fs.rs index 776e0fdcff8..e83a49d17e3 100644 --- a/bcachefs/src/fs.rs +++ b/bcachefs/src/fs.rs @@ -248,6 +248,19 @@ const _: () = assert!( "decode_leaf_value reads 1 as a file and 2 as a symlink", ); +/// The length of a leaf value naming `name` and `extents` runs. +fn leaf_value_len(name: &str, extents: usize) -> usize { + // 1 (entry_type) + 2 (name_len) + 8 (size) + 8 (mtime) + name + extents + 1 + 2 + 8 + 8 + name.len() + extents * EXTENT_SIZE +} + +/// Whether the one entry that holds a file named `name` can name `extents`: +/// a writer that grows the list refuses the write this answers `false` for, +/// before anything the entry would have to record is accepted. +pub fn file_entry_fits(name: &str, extents: &[Extent]) -> bool { + btree::value_fits(leaf_value_len(name, extents.len())) +} + /// Encode a file/symlink leaf value. fn encode_leaf_value( entry_type: KeyType, @@ -258,10 +271,7 @@ fn encode_leaf_value( ) -> Vec { let name_bytes = name.as_bytes(); let name_len = name_bytes.len(); - // 1 (entry_type) + 2 (name_len) + 8 (size) + 8 (mtime) + name + extents - let extent_bytes = extents.len() * EXTENT_SIZE; - let total = 1 + 2 + 8 + 8 + name_len + extent_bytes; - let mut val = vec![0u8; total]; + let mut val = vec![0u8; leaf_value_len(name, extents.len())]; val[0] = entry_type as u8; val[1..3].copy_from_slice(&(name_len as u16).to_le_bytes()); @@ -420,7 +430,7 @@ fn write_data( let mut data_offset = 0usize; while remaining > 0 { - let run = match alloc.alloc_up_to(io, remaining) { + let run = match alloc.alloc_up_to(io, alloc.next_alloc, remaining) { Ok(run) => run, Err(err) => return Err(give_back(io, alloc, &extents, err)), }; @@ -1041,7 +1051,13 @@ impl Mounted { let hole = covered; while covered <= target { let want = (target - covered + 1).min(u32::MAX as u64) as u32; - let run = self.alloc.alloc_up_to(&self.io, want)?; + let total = self.alloc.total_blocks; + let from = match extents.last().and_then(|e| e.start_block.checked_add(e.block_count as u64)) { + None => self.alloc.next_alloc, + Some(next) if next < total && self.alloc.is_free(&self.io, BlockNum::new(next))? => next, + Some(next) => next % total + SPREAD, + }; + let run = self.alloc.alloc_up_to(&self.io, from, want)?; push_extent(extents, run.start.raw(), run.len); covered += run.len as u64; } @@ -1059,6 +1075,12 @@ impl Mounted { } } +/// How far past a file's last block its next run is looked for when another +/// file has taken that block. Two files that grow in turn then each extend +/// runs this long rather than alternate blocks, so the runs one entry can name +/// ([`file_entry_fits`]) last this many times longer. +const SPREAD: u64 = 256; + /// The block holding `page_idx`, if the extents already reach that far. /// /// The one definition of where a page lives, used both to answer a resolve and diff --git a/bcachefs/src/lib.rs b/bcachefs/src/lib.rs index b22c8fbf48a..64afa13218a 100644 --- a/bcachefs/src/lib.rs +++ b/bcachefs/src/lib.rs @@ -26,7 +26,7 @@ pub mod upstream; pub use block_io::{BlockIO, BlockBuf, BlockNum, DeviceError, TransferError}; #[cfg(feature = "std")] pub use block_io::VecBlockIO; -pub use fs::{Formatted, Mounted, ReadOnly, ReadWrite, FsError, Extent}; +pub use fs::{file_entry_fits, Formatted, Mounted, ReadOnly, ReadWrite, FsError, Extent}; pub use superblock::{DESIGNATION_BLOCKS_OFFSET, DESIGNATION_MAGIC, FsUuid, Superblock}; /// Records the largest single allocation each test thread makes, so a test can diff --git a/issues/build/a-sysroot-made-between-the-toolchains-rebuild-and-its-cargo-has-none.md b/issues/build/a-sysroot-made-between-the-toolchains-rebuild-and-its-cargo-has-none.md new file mode 100644 index 00000000000..d14ea64eaf5 --- /dev/null +++ b/issues/build/a-sysroot-made-between-the-toolchains-rebuild-and-its-cargo-has-none.md @@ -0,0 +1,27 @@ +--- +status: open +kind: tooling +opened: 2026-09-27 +--- + +# A sysroot made between the toolchain's rebuild and its cargo has none, for good + +`src/toolchain.rs` rebuilds the primary's toolchain in steps that each take +the global lock and let it go: "build the ToyOS-hosted rustc" recreates +`stage2/bin/`, and "give the toyos toolchain its own cargo" puts `bin/cargo` +back afterwards, under a lock of its own. A sysroot build in another worktree +(`src/sysroot.rs`'s `build`) takes the compiler lock shared between the two, +clones `stage2/` with no `cargo` in it, and writes the sysroot finished. A +sysroot is never written again, so every later build of that key refuses at +`assert_toolchain_is_honest`: "the toyos toolchain at …/sysroots//bin is +missing cargo", and a re-run refuses the same way. + +Measured on 2026-09-27: `sysroots/5dc157f7fac727be/bin` (22:03) and +`sysroots/e8d0e10f5498e275/bin` (22:07) hold `rustc` and `rustdoc` and no +`cargo`; the primary's `stage2/bin/cargo` link is dated 22:08. + +## Exit condition + +No sysroot can clone `stage2/` while its `bin/` is being remade: the rebuild +and the cargo that goes with it are one step under one hold of the lock, or +the clone refuses a `bin/` without `cargo` before it writes the sysroot. diff --git a/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md b/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md index 1185b610d34..e8c4e781a39 100644 --- a/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md +++ b/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md @@ -1,6 +1,6 @@ --- status: open -kind: tooling +kind: defect opened: 2026-09-27 --- diff --git a/issues/filesystem/fs-turns-judges-the-scheduler-with-the-lock.md b/issues/filesystem/fs-turns-judges-the-scheduler-with-the-lock.md index 52da5d33983..43309d83e07 100644 --- a/issues/filesystem/fs-turns-judges-the-scheduler-with-the-lock.md +++ b/issues/filesystem/fs-turns-judges-the-scheduler-with-the-lock.md @@ -7,11 +7,11 @@ opened: 2026-09-27 # `fs_turns` judges the scheduler with the lock `tests/toyos-rust-tests/src/bin/fs_turns.rs` asserts that every writer sharing -one directory's connection made at least 16 passes while the first made 32. A +one directory's connection made more than one pass while the first made 32. A writer the scheduler — or the host, under TCG — keeps off-CPU between its reply and its next request holds no ticket in that time, and the others pass it with -no lock at fault, so the floor reds on a slow schedule as well as on a lock that -lets its releaser take it straight back. +no lock at fault, so the floor reds on a thread kept off for 31 of the first's +turns as well as on a lock that lets its releaser take it straight back. ## Exit condition diff --git a/issues/filesystem/logds-would-block-flush-policy-has-no-producer.md b/issues/filesystem/logds-would-block-flush-policy-has-no-producer.md deleted file mode 100644 index ae2e9780453..00000000000 --- a/issues/filesystem/logds-would-block-flush-policy-has-no-producer.md +++ /dev/null @@ -1,20 +0,0 @@ ---- -status: open -kind: finding -opened: 2026-09-27 ---- - -# logd's would-block flush policy has no producer - -`userland/logd/src/policy.rs`'s `fate` retries a flush answered -`io::ErrorKind::WouldBlock` for up to `LOG_WRITE_BUDGET`, and its module doc and -`userland/logd/src/main.rs`'s header argue why. logd's `/log` is served by fsd, -whose `word` (`userland/fsd/src/fat.rs`) answers no refusal `WouldBlock`, and -`toyos-fat32` has no budget refusal left to carry one, so the arm, its tests and -both docs describe an answer nothing sends. - -## Exit condition - -The `(Step::Flush, WouldBlock)` arm, its tests and the prose arguing for it are -deleted, or fsd answers a refusal `WouldBlock` and a guest test shows logd -retrying it. diff --git a/issues/isolation/one-program-can-take-every-file-servers-client-slot.md b/issues/isolation/one-program-can-take-every-file-servers-client-slot.md index 2e09e915c61..e46ffdd4c99 100644 --- a/issues/isolation/one-program-can-take-every-file-servers-client-slot.md +++ b/issues/isolation/one-program-can-take-every-file-servers-client-slot.md @@ -1,11 +1,13 @@ --- -status: open +status: assigned kind: defect opened: 2026-09-27 --- # One program can take every file server's client slot +Held by the orchestrator. + `userland/fsd/src/main.rs`'s `MAX_SERVED` bounds the clients one file server serves at once, machine-wide: the 129th hello is refused `ResourceExhausted`. Any program holding `fs:/home` can open 128 connections and hold them, and @@ -13,8 +15,13 @@ from then on every other program's first connection to DATA's directories is refused, by name instead of by a hang. `test_rs_fs_client_bound` does exactly this for the length of its run. +`MAX_STREAMS` is the same bound on streams: 64 a server, machine-wide, which +one program's clients can take whole, and every other program's `STREAM` — a +child's output into a served file — is then refused. + ## Exit condition -A program's connections count against a bound of its own, so one program -holding all it may hold leaves another's first connection answered, shown by a -guest test that holds one program at its bound while another connects. +A program's connections and streams each count against a share of their own, +so one program holding all it may hold leaves another's first connection and +first stream answered, shown by a guest test that holds one program at both +bounds while another connects and streams. diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index c1d6342039c..707de190ad2 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -478,7 +478,7 @@ actuators! { /// Run the leak-rollback controls (device mint, FAT reopen) after mount. leak_rollback_selftest = "leak-rollback-selftest"; - /// Run the revoked-backing controls (`/tmp` and `/home`) after mount. + /// Run the revoked-backing controls after mount. revoked_backing_selftest = "revoked-backing-selftest"; /// Reopen init by pid once it is spawned, the way `SYS_PROCESS_OPEN` does. diff --git a/kernel/src/log/mod.rs b/kernel/src/log/mod.rs index 5d4208f7258..b4b083de0a5 100644 --- a/kernel/src/log/mod.rs +++ b/kernel/src/log/mod.rs @@ -54,8 +54,8 @@ const TAIL_HEAD: &str = "log: this boot's newest records follow, newest first"; /// /// **The kernel does not wait for `/system/bin/logd`, so it does not know what /// reached `/log`.** `/system/bin/init` has `logd` flush before it asks for the -/// stop; everything committed after that — the stop's own record, `Syncing -/// filesystems...`, the last word — is on the console, and here, where the next +/// stop; everything committed after that — the stop's own record, the last +/// word — is on the console, and here, where the next /// loader pass prints it into `loader.log`. /// /// Called from the quiesce path under [`crate::blackbox::record_done`], where diff --git a/rust b/rust index 0bf12938d13..9abdd61d95a 160000 --- a/rust +++ b/rust @@ -1 +1 @@ -Subproject commit 0bf12938d135ffbe2f35dda168571aee472d4cbe +Subproject commit 9abdd61d95a8c9f8f060437e452dcff9ab172030 diff --git a/src/build.rs b/src/build.rs index 47265990b77..f6b0fed2581 100644 --- a/src/build.rs +++ b/src/build.rs @@ -3076,6 +3076,7 @@ mod tests { "tests/e1000leasecase/system.toml", "tests/e1000talkcase/system.toml", "tests/flrswapcase/system.toml", + "tests/fsdmountcase/system.toml", "tests/fsdrestartcase/system.toml", "tests/inspectcase/system.toml", "tests/jobcase/system.toml", diff --git a/tests/common/blockd.rs b/tests/common/blockd.rs index 1806ade99d3..75a6ee46e17 100644 --- a/tests/common/blockd.rs +++ b/tests/common/blockd.rs @@ -575,6 +575,7 @@ pub fn blockd_dma_outside_the_lent( let result = role(&mut qemu, name, Duration::from_secs(120))?; if name == "dma-inside" { said(&result, "a register window, and a region already lent, are refused with InvalidArgument")?; + said(&result, "a spawn from a register window is refused with InvalidArgument, and one from a region reaches its argv")?; } if let Some(line) = result.stdout.lines().find(|l| l.contains("aiming the device at ")) { let at = line @@ -617,6 +618,24 @@ pub fn blockd_dma_outside_the_lent( Ok(()) } +/// blockd started holding no claim answers each first frame on the loop and +/// the handshake it serves a controller on: a listing with a payload and an +/// open with no region or a short GUID `Malformed`, a listing empty, an open +/// `NotFound`. The guest's own account; no disk is crafted, since this blockd +/// drives none. +pub fn blockd_serves_nothing( + _test_config: &Path, + c_bins: &[(String, Vec)], + rust_bins: &[(String, Vec)], +) -> Result<(), String> { + let config = super::compile::repo_root().join(CONFIG); + let mut qemu = QemuInstance::boot_with_options(&config, c_bins, rust_bins, BootOptions::default()); + partclaim::no_panic("booting", qemu.boot_log())?; + role(&mut qemu, "nothing", Duration::from_secs(60))?; + eprintln!(" [blockd] with no controller, every first frame answered as the controller's loop answers it"); + Ok(()) +} + /// What a claim may lend its function, and what it may not. /// /// On a boot whose device domains have [`NARROW`] of addresses diff --git a/tests/common/power.rs b/tests/common/power.rs index e99fb640c6c..d22bce5fefe 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -407,17 +407,17 @@ pub fn quiesce_dump_holds_the_stopped( )?; let lines: Vec<&str> = whole.lines().collect(); let at = |needle: &str| lines.iter().position(|line| line.contains(needle)); - let (Some(began), Some(ended), Some(synced)) = ( + let (Some(began), Some(ended), Some(recorded)) = ( at("=== blocked-task dump:"), at("=== end of dump ==="), - at("Syncing filesystems..."), + lines.iter().position(|line| toyos_quiesce::Record::parse(line).is_some()), ) else { - return Err(format!("no whole dump and sync in this boot\n{whole}")); + return Err(format!("no whole dump and stop record in this boot\n{whole}")); }; - if !(began < ended && ended < synced) { + if !(began < ended && ended < recorded) { return Err(format!( - "the dump (lines {began} to {ended}) did not finish inside the stop, which ends at \ - line {synced}\n{whole}" + "the dump (lines {began} to {ended}) did not finish inside the stop, whose record is \ + line {recorded}\n{whole}" )); } let report = &lines[began..=ended]; @@ -479,7 +479,6 @@ pub fn quiesce_refuses_a_second_shutdown( ) -> Result<(), String> { const WAITS: &str = "quiesce-last-park: the stop waits for"; const SECOND_CALLER: &str = "power: this machine is already stopping"; - const SYNCING: &str = "Syncing filesystems..."; let held = format!( "quiesce-last-park: {} is held until the stop waits on it alone", toyos_quiesce::LAST_THREAD @@ -496,12 +495,17 @@ pub fn quiesce_refuses_a_second_shutdown( }; // **The harm, first**: a second caller let in runs a second stop over the - // first, and either way this line is written twice. - let syncs = at(SYNCING); - if syncs.len() != 1 { + // first, and either way the stop's record is written twice. + let records: Vec = lines + .iter() + .enumerate() + .filter(|(_, l)| toyos_quiesce::Record::parse(l).is_some()) + .map(|(i, _)| i) + .collect(); + if records.len() != 1 { return Err(format!( "this boot ran {} shutdowns, not one: the second caller was let in\n{whole}", - syncs.len() + records.len() )); } let once = |needle: &str| -> Result { @@ -514,12 +518,12 @@ pub fn quiesce_refuses_a_second_shutdown( // it waits, before the thread it waits for was held — which the job starts // only on reading `AlreadyExists`, so the held line is the refusal having // reached Ring 3 as that word. - let (waits, second, held, synced) = (once(WAITS)?, once(SECOND_CALLER)?, once(&held)?, syncs[0]); - if !(waits < second && second < held && held < synced) { + let (waits, second, held, recorded) = (once(WAITS)?, once(SECOND_CALLER)?, once(&held)?, records[0]); + if !(waits < second && second < held && held < recorded) { return Err(format!( - "the first call's wait, the second call's refusal, the held thread and the sync are \ - at console lines {waits}, {second}, {held} and {synced}: the refusal was not made \ - in the window the first call held\n{whole}" + "the first call's wait, the second call's refusal, the held thread and the stop's \ + record are at console lines {waits}, {second}, {held} and {recorded}: the refusal was \ + not made in the window the first call held\n{whole}" )); } eprintln!( diff --git a/tests/common/storage.rs b/tests/common/storage.rs index aa4cbd16d32..ab26336e536 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -702,6 +702,55 @@ pub fn fsd_restart( Ok(()) } +/// DATA's first file server ending before it accepts a connection that init's +/// own file worker is waiting on costs init nothing: it starts the server +/// again, the waiting call goes on to it, and the boot reaches the ready +/// marker with the session home made. `tests/fsdmountcase` arms the end +/// (`--end-at-mount data`); the home is judged off the image by the host's own +/// bcachefs reader once the machine is down. +pub fn fsd_end_at_mount( + _test_config: &Path, + c_bins: &[(String, Vec)], + rust_bins: &[(String, Vec)], +) -> Result<(), String> { + const ENDED: &str = "fsd: --end-at-mount: ending with a connection waiting and unaccepted"; + let config = super::compile::repo_root().join("tests/fsdmountcase"); + let mut qemu = QemuInstance::boot_with_options( + &config, + c_bins, + rust_bins, + BootOptions { profile: qemu::Profile::Metal, ..Default::default() }, + ); + let boot = qemu.boot_log().to_string(); + let image = qemu.nvme_image().to_path_buf(); + writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); + qemu.flush_stdin(); + let tail = qemu.drain_serial(Duration::from_secs(20)); + drop(qemu); + let log = format!("{boot}\n{tail}"); + let ended = log.matches(ENDED).count(); + if ended != 1 { + return Err(format!("fsd said {ENDED:?} {ended} times, not once:\n{log}")); + } + let restarted = log.lines().filter(|l| l.contains("init: fsd data (pid ") && l.contains("ended; started again")).count(); + if restarted != 1 { + return Err(format!("init started DATA's server again {restarted} times, not once:\n{log}")); + } + let console = super::serial::Serial::named("fsd_end_at_mount", log.as_str()); + console.must_not_say("session home")?; + console.must_be_clean()?; + + let home = toyos_manifest::session_home(); + let home = home.trim_start_matches('/'); + let fs = bcachefs::Mounted::<_, bcachefs::ReadOnly>::open(FileBlocks::open(&image)?) + .map_err(|e| format!("the DATA partition does not mount on the host: {e:?}"))?; + if !fs.is_dir(home).map_err(|e| format!("asking the image for {home}: {e:?}"))? { + return Err(format!("{home} is not on the DATA volume: init made no session home\n{log}")); + } + eprintln!(" [fsd] DATA's first server ended under init's waiting call; init started it again and {home} is on the image"); + Ok(()) +} + /// `/apps` and `/home` are two paths into one filesystem, judged off the device. /// /// The guest writes one file under each and shuts down; the host then finds diff --git a/tests/common/usb.rs b/tests/common/usb.rs index 9dc0daaaed9..c67b125e017 100644 --- a/tests/common/usb.rs +++ b/tests/common/usb.rs @@ -1050,8 +1050,7 @@ fn failed_flush_stops_once( const PROBES: usize = 12; /// A per-failure line, and the thing that has to stay bounded. Before the /// fix it is emitted by every pass of the idle loop for the life of the - /// boot. After it: one by the write that gives up, and one per mount by the - /// shutdown's `sync_all`, which is the last caller left. + /// boot. After it: one by the write that gives up. /// /// **This is the number that caught the retry**, and it is worth saying what /// it caught. A logd that retried inside `LOG_WRITE_BUDGET` measured diff --git a/tests/fsdmountcase/system.toml b/tests/fsdmountcase/system.toml new file mode 100644 index 00000000000..58ae0a5fac2 --- /dev/null +++ b/tests/fsdmountcase/system.toml @@ -0,0 +1,34 @@ +# The boot `fsd_end_at_mount` judges: DATA's first file server ends once its +# volume is mounted and a connection waits on it, before it accepts one — the +# connection init's own file worker makes for the session home. + +[boot] +start = ["logd", "blockd", "fsd", "test-runner"] + +[programs.logd] +service = true +syscap = ["logread"] + +# `power` because `run shutdown` asks init through the `power` connector, and +# the host reads the DATA partition back once the machine is down. +[programs.test-runner] +receives = ["power"] +syscap = ["dup", "logread", "power"] + +[programs.toybox] +receives = ["power"] + +[symlinks] +"bin/shutdown" = "/system/bin/toybox" + +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] +args = ["--end-at-mount", "data"] diff --git a/tests/toyos-rust-tests/src/bin/blockd_io.rs b/tests/toyos-rust-tests/src/bin/blockd_io.rs index aabbffd08f3..29fb6b05d9b 100644 --- a/tests/toyos-rust-tests/src/bin/blockd_io.rs +++ b/tests/toyos-rust-tests/src/bin/blockd_io.rs @@ -28,6 +28,8 @@ //! back until ten domains' worth of addresses went by; //! - `dma-residue` — on a boot where no release resets the function, three //! claims in turn, and none lends where the first one did. +//! - `nothing` — blockd started holding no claim: each first frame, malformed +//! and well-formed, answered as `serve` answers it. use std::io::{BufRead, BufReader, Write}; use std::os::toyos::process::{ChildExt, CommandExt}; @@ -44,7 +46,7 @@ use toyos::shm::SharedMemory; use toyos::syscap::SysCap; use toyos::AsHandle; use toyos_abi::part::PartGuid; -use toyos_abi::syscall::{DeviceType, PciId, SyscallError, DEV_PREFIX, SERVE_PREFIX, SYSCAP_LABEL}; +use toyos_abi::syscall::{self, DeviceType, PciId, SpawnArgs, SyscallError, DEV_PREFIX, SERVE_PREFIX, SYSCAP_LABEL}; use toyos_blockring::wire::{self, Refusal}; use toyos_blockring::{BLOCK_BYTES, MAX_REQUEST_BLOCKS, PORT}; @@ -637,6 +639,30 @@ fn aim() -> (Controller, SharedMemory) { (ctrl, region) } +/// A spawn from `image`'s first `len` bytes, with an argv no process can read. +fn spawn_unreadable_argv(image: toyos::RawHandle, len: u64) -> Result { + // SAFETY: argv names the null page, which the kernel refuses to read, and + // every other pointer is null with a zero length. + unsafe { + syscall::spawn(&SpawnArgs { + argv_ptr: 8, + argv_len: 8, + slot_map_ptr: 0, + slot_map_count: 0, + env_ptr: 0, + env_len: 0, + endow_ptr: 0, + endow_count: 0, + labels_ptr: 0, + labels_len: 0, + cwd_ptr: 0, + cwd_len: 0, + image: image.0 as u64, + image_len: len, + }) + } +} + /// Device block 0, the disk's protective MBR, ends 0x55 0xAA. fn is_block_zero(bytes: &[u8]) -> bool { bytes[510] == 0x55 && bytes[511] == 0xAA @@ -669,6 +695,18 @@ fn dma(role: &str) { other => fail(format!("lending the region a second time was answered {other:?}")), } println!("blockd_io: a register window, and a region already lent, are refused with InvalidArgument"); + // Nor is a register window a program: the spawn refuses it before + // it reads anything else, where a region's is taken and the spawn + // goes on to refuse the argv. + match spawn_unreadable_argv(bar.as_handle(), 4096) { + Err(SyscallError::InvalidArgument) => {} + other => fail(format!("a spawn from the register window was answered {other:?}")), + } + match spawn_unreadable_argv(region.as_handle(), 4096) { + Err(SyscallError::BadAddress) => {} + other => fail(format!("a spawn from the region with an unreadable argv was answered {other:?}")), + } + println!("blockd_io: a spawn from a register window is refused with InvalidArgument, and one from a region reaches its argv"); } "dma-outside" => { let past = mapping.device_addr + mapping.bytes; @@ -923,9 +961,49 @@ fn dma_residue() { println!("blockd_io: PASS dma-residue"); } +/// blockd started holding no controller answers a connection's first frame as +/// it answers every other: the malformed refused as such, a listing empty and +/// an open `NotFound`. +fn nothing() { + let (acceptor, connector) = port::create().unwrap_or_else(|e| fail(format!("no port: {e:?}"))); + let mut command = Command::new("/system/bin/blockd"); + command.endow(&format!("{SERVE_PREFIX}{PORT}"), acceptor.into_raw().0); + let mut child = command.spawn().unwrap_or_else(|e| fail(format!("blockd did not start: {e}"))); + let names = + namespace::build().add(PORT, &connector).finish().unwrap_or_else(|e| fail(format!("no namespace: {e:?}"))); + let region = || { + let region = SharedMemory::create(REGION).unwrap_or_else(|e| fail(format!("a region: {e:?}"))); + vec![region.share().unwrap_or_else(|e| fail(format!("a second handle: {e:?}")))] + }; + let absent = guid(ABSENT); + let malformed = Some(Refusal::Malformed); + for (what, msg_type, payload, handles, answered, refusal) in [ + ("a listing that carries a payload", wire::MSG_LIST, &[0u8; 4][..], vec![], wire::MSG_REFUSED, malformed), + ("an open with no region", wire::MSG_OPEN, &absent[..], vec![], wire::MSG_REFUSED, malformed), + ("an open whose GUID is short", wire::MSG_OPEN, &absent[..8], region(), wire::MSG_REFUSED, malformed), + ("a listing", wire::MSG_LIST, &[][..], vec![], wire::MSG_LISTED, None), + ("an open", wire::MSG_OPEN, &absent[..], region(), wire::MSG_REFUSED, Some(Refusal::NotFound)), + ] { + let conn = names.open(PORT).unwrap_or_else(|e| fail(format!("the port: {e:?}"))); + conn.send_bytes_with_handles(&handles, msg_type, payload).unwrap_or_else(|e| fail(format!("{what}: {e:?}"))); + let header = conn.recv_header().unwrap_or_else(|e| fail(format!("{what}'s answer: {e:?}"))); + let mut answer = [0u8; 64]; + let len = conn.recv_bytes(&header, &mut answer).unwrap_or_else(|e| fail(format!("{what}'s answer: {e:?}"))); + let got = Refusal::decode(&answer[..len]); + if header.msg_type != answered || got != refusal || (refusal.is_none() && len != 0) { + fail(format!("{what} was answered {} {got:?} in {len} bytes, not {answered} {refusal:?}", header.msg_type)); + } + println!("blockd_io: with no controller, {what} was answered {answered} {refusal:?}"); + } + let _ = child.kill(); + let _ = child.wait(); + println!("blockd_io: PASS nothing"); +} + fn main() { let args: Vec = std::env::args().collect(); match args.get(1).map(String::as_str) { + Some("nothing") => nothing(), Some("claims") => claims(), Some("holder") => holder_role(args.get(2).map_or("", String::as_str)), Some("bench") => bench(), diff --git a/tests/toyos-rust-tests/src/bin/fs_turns.rs b/tests/toyos-rust-tests/src/bin/fs_turns.rs index d547c95ec66..f1be3aa3e0e 100644 --- a/tests/toyos-rust-tests/src/bin/fs_turns.rs +++ b/tests/toyos-rust-tests/src/bin/fs_turns.rs @@ -33,6 +33,10 @@ fn main() { println!("fs_turns: passes per thread {passes:?}"); let _ = std::fs::remove_dir_all("/home/fs_turns"); let least = *passes.iter().min().expect("threads"); - assert!(least >= PASSES / 2, "a thread made {least} passes while another made {PASSES}: {passes:?}"); + // One pass is a lock that lets its releaser take it straight back: every + // other thread gets in once, at the start. A floor any higher also + // measures how long a thread is kept off-CPU between its reply and its + // next request, when it holds no ticket. + assert!(least > 1, "a thread made {least} passes while another made {PASSES}: {passes:?}"); println!("fs_turns: PASS"); } diff --git a/tests/toyos.rs b/tests/toyos.rs index 2ef14ccc4d3..8b833fb24cc 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -983,6 +983,10 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // clients reopen, the fourth end closes /home to Gone, and the flushed // files read back off the image. Body in `tests/common/storage.rs`. ("fsd_restart", Sched::Parallel, Tier::Fast), + // DATA's first file server ended before it accepts init's own waiting call: + // init starts it again and the boot reaches ready with the session home. + // Body in `tests/common/storage.rs`. + ("fsd_end_at_mount", Sched::Parallel, Tier::Fast), // A same-length overwrite on /home, the guest's read held against the image. Body in `tests/common/storage.rs`. ("home_overwrite_reads_back", Sched::Parallel, Tier::Fast), // One filesystem under two paths: the guest writes under each of /apps and @@ -1548,6 +1552,9 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // standing; then, on a second boot, a function no release resets is never // lent where it was left aimed. ("blockd_lends_within_its_bound", Sched::Parallel, Tier::Nightly), + // blockd holding no controller: each first frame answered on the loop and + // the handshake a controller is served on, the malformed refused as such. + ("blockd_serves_nothing", Sched::Parallel, Tier::Fast), // H4: soundd driving an Intel HDA controller itself, read back off the // device. Serial — its verdict is a wav capture, and one taken while eleven // other guests contend for the host measures the host. @@ -1658,6 +1665,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("blockd_survives_its_death", &["test_rs_blockd_io"]), ("blockd_dma_outside_the_lent", &["test_rs_blockd_io"]), ("blockd_lends_within_its_bound", &["test_rs_blockd_io"]), + ("blockd_serves_nothing", &["test_rs_blockd_io"]), ( "inspect_reads_its_owners", &["test_rs_inspect_denied", "test_rs_inspect_plays", "test_rs_inventory_bounds"], @@ -11387,6 +11395,7 @@ fn run_machine_test( } "so_cache_refusals" => storage::so_cache_refusals(test_config, c_bins, rust_bins), "fsd_restart" => storage::fsd_restart(test_config, c_bins, rust_bins), + "fsd_end_at_mount" => storage::fsd_end_at_mount(test_config, c_bins, rust_bins), "home_overwrite_reads_back" => { storage::home_overwrite_reads_back(test_config, c_bins, rust_bins) } @@ -11954,6 +11963,7 @@ fn run_machine_test( common::blockd::blockd_lends_within_its_bound(test_config, c_bins, rust_bins) } // Body in `tests/common/hda.rs`, same reason. + "blockd_serves_nothing" => common::blockd::blockd_serves_nothing(test_config, c_bins, rust_bins), "hda_tone" => common::hda::hda_tone(test_config, c_bins, rust_bins), "hda_client_stall" => common::hda::hda_client_stall(test_config, c_bins, rust_bins), "hda_two_live_refused" => { @@ -12726,7 +12736,7 @@ fn run_machine_test( // Then shut down, which has init ask fsd to sync every dirty block // the format left — ~1900 of them on a device this size against 8 - // on the small one, so the write-back runs at scale only here. + // on the small one. // // The kernel's own shutdown lines are observable now: the ring // is drained in `acpi::shutdown()` before it cuts the power. @@ -12738,8 +12748,9 @@ fn run_machine_test( qemu.flush_stdin(); let tail = qemu.drain_serial(Duration::from_secs(20)); - for line in ["Syncing filesystems...", "Shutting down."] { - if !tail.contains(line) { + let recorded = tail.lines().any(|line| toyos_quiesce::Record::parse(line).is_some()); + for (line, said) in [("the stop's record", recorded), ("Shutting down.", tail.contains("Shutting down."))] { + if !said { return Err(format!( "{line:?} never reached the host — the ring was still \ holding it when the power was cut:\n{tail}" @@ -12776,10 +12787,7 @@ fn run_machine_test( )); } if !sb.is_clean() { - return Err(format!( - "the {name} superblock is not marked clean — the write-back at \ - shutdown did not reach the device" - )); + return Err(format!("the {name} superblock is not marked clean")); } } diff --git a/userland/blockd/src/main.rs b/userland/blockd/src/main.rs index 85e926efda1..096ff706d3b 100644 --- a/userland/blockd/src/main.rs +++ b/userland/blockd/src/main.rs @@ -192,8 +192,39 @@ struct Served { posted: bool, } +/// The controller this service drives, or why it drives none. Without one no +/// partition is listed and no session opened, and a connection says one frame +/// and is answered, or is let go. +enum Drive { + Up(Controller), + /// No controller this row names is on this machine: a listing is empty + /// and an open `NotFound`. + Absent, + /// A controller this service cannot use: a listing and an open are + /// refused `Unusable`, so a file server says the disk failed rather than + /// that there is none. + Unusable, +} + +impl Drive { + /// The controller a session is on: a service without one opens none. + fn up(&mut self) -> &mut Controller { + match self { + Drive::Up(ctrl) => ctrl, + Drive::Absent | Drive::Unusable => unreachable!("blockd: a session with no controller"), + } + } + + fn oldest(&self) -> Option { + match self { + Drive::Up(ctrl) => ctrl.oldest(), + Drive::Absent | Drive::Unusable => None, + } + } +} + struct Service { - ctrl: Controller, + ctrl: Drive, parts: Vec, holds: Holds, /// The device's loss count: every reset may have dropped its cache. @@ -216,10 +247,17 @@ struct Opening { } impl Service { + fn new(ctrl: Drive, parts: Vec, running: Option<[u8; 16]>) -> Self { + Self { ctrl, parts, holds: Holds::new(), losses: 0, sessions: BTreeMap::new(), next_id: 0, running } + } + /// What an open answers: a session to admit, or its refusal. `region` is /// the client's and is consumed either way. fn open(&mut self, guid: [u8; 16], region: toyos::RawHandle) -> Result { let region = Region::adopt(region).map_err(|_| Refusal::Malformed)?; + if let Drive::Unusable = self.ctrl { + return Err(Refusal::Unusable); + } let part = self.parts.iter().find(|p| p.unique == guid && guid != [0; 16]); let (first, blocks) = match part.map(|p| &p.span) { None => return Err(Refusal::NotFound), @@ -235,14 +273,14 @@ impl Service { if self.holds.hold(first, first + blocks, self.next_id).is_err() { return Err(Refusal::Held); } - match self.ctrl.claim().dma_map(region.handle()) { + match self.ctrl.up().claim().dma_map(region.handle()) { Ok(mapping) if mapping.bytes == SESSION_BYTES as u64 => { Ok(Opening { region, device_addr: mapping.device_addr, first, blocks, unique: guid }) } // A region longer than a session would spend the claim's bound on // the kernel's side for every other client: refused whole. Ok(mapping) => { - if let Err(why) = self.ctrl.claim().dma_unmap(mapping.device_addr) { + if let Err(why) = self.ctrl.up().claim().dma_unmap(mapping.device_addr) { panic!("blockd: the kernel would not take an oversized region back: {why:?}"); } self.holds.release(first); @@ -283,7 +321,7 @@ impl Service { /// The client went before it heard: nothing of the session reached the /// device, and what was taken for it goes back. fn abandon(&mut self, opening: Opening) { - if let Err(why) = self.ctrl.claim().dma_unmap(opening.device_addr) { + if let Err(why) = self.ctrl.up().claim().dma_unmap(opening.device_addr) { panic!("blockd: the kernel would not take an abandoned session's region back: {why:?}"); } self.holds.release(opening.first); @@ -332,7 +370,7 @@ impl Service { break; }; let unread = (DEPTH - space) as usize; - if s.state.inflight() + unread >= DEPTH as usize || !self.ctrl.has_room() { + if s.state.inflight() + unread >= DEPTH as usize || !self.ctrl.up().has_room() { break; } let words = match s.rings.0.pop(page) { @@ -355,9 +393,9 @@ impl Service { Op::Read | Op::Write => { let at = s.device_addr + arena_byte(req.arena) as u64; let block = s.state.first() + req.lba; - self.ctrl.submit_io(req.op == Op::Write, block, req.blocks, at, owner); + self.ctrl.up().submit_io(req.op == Op::Write, block, req.blocks, at, owner); } - Op::Flush if self.ctrl.vwc => self.ctrl.submit_flush(owner), + Op::Flush if self.ctrl.up().vwc => self.ctrl.up().submit_flush(owner), // No volatile cache: every write answered is on the // medium already. Op::Flush => { @@ -399,7 +437,8 @@ impl Service { .collect(); for id in done { let s = self.sessions.remove(&id).expect("just listed"); - if let Err(why) = self.ctrl.claim().dma_unmap(s.device_addr) { + let ctrl = self.ctrl.up(); + if let Err(why) = ctrl.claim().dma_unmap(s.device_addr) { panic!("blockd: the kernel would not take session {id}'s region back: {why:?}"); } self.holds.release(s.state.first()); @@ -408,8 +447,8 @@ impl Service { commands at once, issued per queue {}", guid_text(s.unique), s.requests, - self.ctrl.peak, - self.ctrl.spread() + ctrl.peak, + ctrl.spread() ); } } @@ -422,6 +461,7 @@ impl Service { let mut done = Vec::new(); let aborted = self .ctrl + .up() .reset(&mut done) .unwrap_or_else(|why| panic!("blockd: the controller did not come back from its reset: {why}")); for d in done { @@ -454,70 +494,6 @@ fn claim() -> Option { Some(Endowments::get().take(&label).expect("blockd: the claim its label names")) } -/// Serve no partition: a machine with no controller answers every listing -/// empty and every open `NotFound`, and one whose controller is `unusable` -/// answers both `Unusable`, so a file server says the disk failed rather than -/// that there is none, and waits on neither. A connection says one frame and -/// is answered, or is let go. -fn serve_nothing(acceptor: &toyos::port::Acceptor, unusable: bool) -> ! { - let poller = Poller::new(2 + MAX_PENDING as u32); - let mut pending: Vec = Vec::new(); - let mut ready: Vec = Vec::new(); - loop { - if pending.len() < MAX_PENDING { - poller.watch(acceptor, READABLE, TOKEN_ACCEPT); - } - for p in &pending { - poller.watch(&p.conn, READABLE, TOKEN_PENDING + p.conn.as_handle().0 as u64); - } - let now = Instant::now(); - let timeout = pending - .iter() - .map(|p| HANDSHAKE_TIMEOUT.saturating_sub(now.duration_since(p.since))) - .min() - .map_or(u64::MAX, |left| left.as_nanos() as u64); - ready.clear(); - poller.wait(1, timeout, |token| ready.push(token)); - let now = Instant::now(); - pending.retain(|p| now.duration_since(p.since) < HANDSHAKE_TIMEOUT); - if ready.contains(&TOKEN_ACCEPT) { - match acceptor.accept() { - Ok(conn) => pending.push(Pending { conn, rx: ipc::FrameRx::new(), since: now }), - Err(why) => panic!("blockd: its own acceptor refused an accept: {why:?}"), - } - } - let mut i = 0; - while i < pending.len() { - let token = TOKEN_PENDING + pending[i].conn.as_handle().0 as u64; - if !ready.contains(&token) { - i += 1; - continue; - } - let step = { - let p = &mut pending[i]; - p.rx.pump(&p.conn) - }; - match step { - RxStep::Idle => i += 1, - RxStep::Eof | RxStep::Malformed => { - pending.remove(i); - } - RxStep::Frame { msg_type, .. } => { - let p = pending.remove(i); - if let Some(handles) = p.conn.recv_handles_exact::<1>() { - toyos_abi::syscall::close(handles[0]); - } - let _ = match (msg_type, unusable) { - (_, true) => p.conn.try_send_bytes(wire::MSG_REFUSED, &Refusal::Unusable.encode()), - (wire::MSG_LIST, false) => p.conn.try_send_bytes(wire::MSG_LISTED, &[]), - (_, false) => p.conn.try_send_bytes(wire::MSG_REFUSED, &Refusal::NotFound.encode()), - }; - } - } - } - } -} - fn main() { let args: Vec = std::env::args().collect(); let silence = args.iter().position(|a| a == SILENCE_WRITE).map(|at| { @@ -535,7 +511,7 @@ fn main() { let acceptor = endow::acceptor(PORT).unwrap_or_else(|| panic!("blockd: started serving no `{PORT}` port")); let Some(dev) = claim() else { println!("blockd: no NVMe controller this row names is on this machine; serving no partition"); - serve_nothing(&acceptor, false); + serve(&mut Service::new(Drive::Absent, Vec::new(), running), &acceptor); }; // A controller this service cannot use is a machine without one, said by // name: restarting would meet the same device and the same refusal. @@ -543,7 +519,7 @@ fn main() { Ok(ctrl) => ctrl, Err(why) => { println!("blockd: NOT SERVING — {why}; serving no partition"); - serve_nothing(&acceptor, true); + serve(&mut Service::new(Drive::Unusable, Vec::new(), running), &acceptor); } }; println!( @@ -568,9 +544,7 @@ fn main() { Err(why) => println!("blockd: partition {} is not served: {why}", guid_text(part.unique)), } } - let mut service = - Service { ctrl, parts, holds: Holds::new(), losses: 0, sessions: BTreeMap::new(), next_id: 0, running }; - serve(&mut service, &acceptor); + serve(&mut Service::new(Drive::Up(ctrl), parts, running), &acceptor); } fn serve(service: &mut Service, acceptor: &toyos::port::Acceptor) -> ! { @@ -579,7 +553,9 @@ fn serve(service: &mut Service, acceptor: &toyos::port::Acceptor) -> ! { let mut ready: Vec = Vec::new(); let mut done: Vec = Vec::new(); loop { - poller.watch(service.ctrl.claim(), READABLE, TOKEN_IRQ); + if let Drive::Up(ctrl) = &service.ctrl { + poller.watch(ctrl.claim(), READABLE, TOKEN_IRQ); + } if pending.len() < MAX_PENDING { poller.watch(acceptor, READABLE, TOKEN_ACCEPT); } @@ -601,13 +577,15 @@ fn serve(service: &mut Service, acceptor: &toyos::port::Acceptor) -> ! { ready.clear(); poller.wait(1, timeout, |token| ready.push(token)); - if let Err(why) = service.ctrl.take_interrupt() { - panic!( - "blockd: the claim refused its interrupt read ({why:?}): the unit refused this \ - controller an access, and the claim answers nothing now" - ); + if let Drive::Up(ctrl) = &mut service.ctrl { + if let Err(why) = ctrl.take_interrupt() { + panic!( + "blockd: the claim refused its interrupt read ({why:?}): the unit refused this \ + controller an access, and the claim answers nothing now" + ); + } + ctrl.reap(&mut done); } - service.ctrl.reap(&mut done); for d in done.drain(..) { service.deliver(d); } @@ -680,6 +658,10 @@ fn handshake(service: &mut Service, p: Pending, msg_type: u32, payload_len: usiz let _ = conn.try_send_bytes(wire::MSG_REFUSED, &why.encode()); }; if msg_type == wire::MSG_LIST && payload_len == 0 { + if let Drive::Unusable = service.ctrl { + refuse(&p.conn, Refusal::Unusable); + return; + } let listing: Vec = service .parts .iter() diff --git a/userland/fsd/src/absent.rs b/userland/fsd/src/absent.rs index 474083653d8..4a53b77281e 100644 --- a/userland/fsd/src/absent.rs +++ b/userland/fsd/src/absent.rs @@ -49,7 +49,9 @@ impl Volume for Absent { Err(SyscallError::NotFound) } - fn close(&mut self, _node: Node) {} + fn close(&mut self, _node: Node) -> Result<(), SyscallError> { + Ok(()) + } fn hold(&mut self, _node: Node) {} @@ -89,8 +91,8 @@ impl Volume for Absent { Err(SyscallError::NotFound) } - fn sync(&mut self) -> Result<(), SyscallError> { - Ok(()) + fn sync(&mut self) -> Result, SyscallError> { + Ok(Vec::new()) } fn describe(&self) -> String { diff --git a/userland/fsd/src/data.rs b/userland/fsd/src/data.rs index dcd41032ad3..7b61d4a880d 100644 --- a/userland/fsd/src/data.rs +++ b/userland/fsd/src/data.rs @@ -270,6 +270,14 @@ impl DataVolume { Ok(()) } + /// Forget a node nobody holds and whose entry is written. + fn release(&mut self, node: Node) { + let open = self.open.remove(&node).expect("an open node"); + if self.by_path.get(&open.path) == Some(&node) { + self.by_path.remove(&open.path); + } + } + /// Hand every open file at `path` its end: its blocks go with the name. fn orphan(&mut self, path: &str) { if let Some(node) = self.by_path.remove(path) { @@ -392,26 +400,26 @@ impl Volume for DataVolume { self.open.get_mut(&node).expect("just found or made").holders += 1; if how.truncate { if let Err(e) = self.truncate(node, 0) { - self.close(node); + // A refused close is said, and its node kept for the next sync. + let _ = self.close(node); return Err(e); } } Ok(node) } - fn close(&mut self, node: Node) { - let Some(open) = self.open.get_mut(&node) else { return }; + fn close(&mut self, node: Node) -> Result<(), SyscallError> { + let Some(open) = self.open.get_mut(&node) else { return Ok(()) }; open.holders -= 1; if open.holders > 0 { - return; + return Ok(()); } if let Err(e) = self.persist(node) { - println!("fsd: node {node}'s entry was not written at its last close: {e:?}"); - } - let open = self.open.remove(&node).expect("present above"); - if self.by_path.get(&open.path) == Some(&node) { - self.by_path.remove(&open.path); + println!("fsd: node {node}'s entry was not written at its last close ({e:?}); the next sync writes it"); + return Err(e); } + self.release(node); + Ok(()) } fn hold(&mut self, node: Node) { @@ -475,6 +483,30 @@ impl Volume for DataVolume { if open.gone { return Err(SyscallError::Gone); } + // Every page past the extents is allocated before a byte is written, + // so a write the file's one entry could not name is refused whole. + let covered: u64 = open.extents.iter().map(|e| e.block_count as u64).sum(); + for page in (offset / BLOCK as u64).max(covered)..end.div_ceil(BLOCK as u64) { + let grown = u32::try_from(page) + .map_err(|_| SyscallError::ResourceExhausted) + .and_then(|p| mapped("an allocation", &open.path, self.fs.resolve_or_alloc_block(&mut open.extents, p))) + .and_then(|_| match bcachefs::file_entry_fits(&open.path, &open.extents) { + true => Ok(()), + false => { + println!( + "fsd: a write to '{}' is refused: its entry would name {} runs of blocks, more than it holds", + open.path, + open.extents.len() + ); + Err(SyscallError::ResourceExhausted) + } + }); + if let Err(e) = grown { + let dropped = keep_blocks(&mut open.extents, covered); + mapped("a refused write's free", &open.path, self.fs.free_extents(&dropped))?; + return Err(e); + } + } let mut done = 0; let mut page_buf = vec![0u8; BLOCK]; while done < data.len() { @@ -482,22 +514,16 @@ impl Volume for DataVolume { let page = at / BLOCK as u64; let within = (at % BLOCK as u64) as usize; let take = (BLOCK - within).min(data.len() - done); - let existed = block_for(&open.extents, page); - let block = match existed { - Some(block) => block, - None => { - let page_idx = u32::try_from(page).map_err(|_| SyscallError::ResourceExhausted)?; - mapped("an allocation", &open.path, self.fs.resolve_or_alloc_block(&mut open.extents, page_idx))? - } - }; + let block = block_for(&open.extents, page).expect("allocated above"); if take == BLOCK { cache.write(block, &data[done..done + BLOCK]).map_err(disk_word)?; } else { // A page written in part keeps the rest of what it held, which // for a page the file never reached is zeros. - match existed { - Some(_) if page * (BLOCK as u64) < open.size => cache.read(block, &mut page_buf).map_err(disk_word)?, - _ => page_buf.fill(0), + if page < covered && page * (BLOCK as u64) < open.size { + cache.read(block, &mut page_buf).map_err(disk_word)?; + } else { + page_buf.fill(0); } page_buf[within..within + take].copy_from_slice(&data[done..done + take]); cache.write(block, &page_buf).map_err(disk_word)?; @@ -675,12 +701,18 @@ impl Volume for DataVolume { Ok(()) } - fn sync(&mut self) -> Result<(), SyscallError> { + fn sync(&mut self) -> Result, SyscallError> { + let mut unwritten = Vec::new(); let nodes: Vec = self.open.keys().copied().collect(); for node in nodes { - self.persist(node)?; + match self.persist(node) { + Err(e) => unwritten.push((node, e)), + Ok(()) if self.open[&node].holders == 0 => self.release(node), + Ok(()) => {} + } } - mapped("sync", "", self.fs.sync()) + mapped("sync", "", self.fs.sync())?; + Ok(unwritten) } fn describe(&self) -> String { @@ -735,12 +767,9 @@ mod tests { v.mkdir("home/toy").unwrap(); let n = v.open("home/toy/x", CREATE).unwrap(); v.write(n, 0, &[7; 5000]).unwrap(); - v.close(n); + v.close(n).unwrap(); v.sync().unwrap(); - let DataVolume { fs, cache, .. } = v; - drop(fs); - let disk = Rc::try_unwrap(cache).ok().expect("one owner").into_disk(); - let Probed::Mounted(mut again) = DataVolume::probe(disk, &["home"], clock) else { panic!("remount") }; + let mut again = remount(v); assert_eq!(again.lstat("home/toy/x").unwrap().size, 5000); assert_eq!(again.lstat("home/toy").unwrap().kind, Kind::Dir, "an empty directory outlives the mount"); let n = again.open("home/toy/x", PLAIN).unwrap(); @@ -776,13 +805,13 @@ mod tests { fn directories_are_listed_and_refuse_what_posix_refuses() { let mut v = vol(); let f = v.open("home/nodir/a", CREATE).unwrap(); - v.close(f); + v.close(f).unwrap(); assert_eq!(v.lstat("home/nodir").unwrap().kind, Kind::Dir, "a create makes its directories"); assert_eq!(v.open("home/nodir/a/b", CREATE), Err(SyscallError::NotFound), "never under a file"); v.mkdir("home/d").unwrap(); assert_eq!(v.mkdir("home/d"), Err(SyscallError::AlreadyExists)); let n = v.open("home/d/f", CREATE).unwrap(); - v.close(n); + v.close(n).unwrap(); assert_eq!(v.rmdir("home/d"), Err(SyscallError::InvalidArgument), "not empty"); let names: Vec = v.list("home").unwrap().into_iter().map(|(n, _)| n).collect(); assert_eq!(names, ["d", "nodir"]); @@ -803,27 +832,132 @@ mod tests { let mut out = [0u8; 5]; v.read(n, 0, &mut Buf(&mut out)).unwrap(); assert_eq!(&out, b"hello"); - v.close(n); + v.close(n).unwrap(); let m = v.open("home/b/g", CREATE).unwrap(); - v.close(m); + v.close(m).unwrap(); v.rename("home/b/g", "home/b/f").unwrap(); assert_eq!(v.lstat("home/b/f").unwrap().size, 0, "replaced"); } + fn remount(v: DataVolume) -> DataVolume { + let DataVolume { fs, cache, .. } = v; + drop(fs); + let disk = Rc::try_unwrap(cache).ok().expect("one owner").into_disk(); + let Probed::Mounted(again) = DataVolume::probe(disk, &["home"], clock) else { panic!("remount") }; + again + } + + /// A full volume of one-page files, every other one then deleted: no two + /// free blocks are adjacent, so every run a file is given is one block. + fn fragmented() -> DataVolume { + let mut v = DataVolume::format(Ram::new(1024), &["home"], clock).unwrap(); + let mut made = 0; + while let Ok(n) = v.open(&format!("home/fill/{made}"), CREATE) { + let written = v.write(n, 0, &[9; BLOCK]); + v.close(n).unwrap(); + if written.is_err() { + break; + } + made += 1; + } + for i in (0..made).step_by(2) { + v.unlink(&format!("home/fill/{i}")).unwrap(); + } + v + } + + fn page_count(v: &mut DataVolume, path: &str) -> u64 { + v.lstat(path).unwrap().size / BLOCK as u64 + } + + /// Two files that grow a page at a time in turn each keep their runs, so + /// neither's entry fills with one extent per page. + #[test] + fn two_files_written_in_turn_keep_every_accepted_write() { + let mut v = vol(); + let a = v.open("home/a", CREATE).unwrap(); + let b = v.open("home/b", CREATE).unwrap(); + for page in 0..300u64 { + v.write(a, page * BLOCK as u64, &[1; BLOCK]).unwrap(); + v.write(b, page * BLOCK as u64, &[2; BLOCK]).unwrap(); + } + let c = v.open("home/c", CREATE).unwrap(); + v.write(c, 0, b"c").unwrap(); + assert_eq!(v.sync(), Ok(Vec::new())); + for n in [a, b, c] { + v.close(n).unwrap(); + } + let mut v = remount(v); + assert_eq!(page_count(&mut v, "home/a"), 300); + assert_eq!(page_count(&mut v, "home/b"), 300); + } + + /// On a volume whose free blocks are all apart, a file's entry fills: the + /// write that would take it past what one entry names is refused whole, + /// and every write before it reaches the volume. + #[test] + fn a_write_its_entry_could_not_name_is_refused_and_every_accepted_one_kept() { + let mut v = fragmented(); + let a = v.open("home/a", CREATE).unwrap(); + let mut pages = 0u64; + let refused = loop { + match v.write(a, pages * BLOCK as u64, &[3; BLOCK]) { + Ok(()) => pages += 1, + Err(e) => break e, + } + }; + assert_eq!(refused, SyscallError::ResourceExhausted); + // One block a run: (4064 − 24 − 19 − "home/a".len()) / 16 runs is + // what the one entry holding `home/a` names. + assert_eq!(pages, 250, "the entry filled, not the volume"); + assert_eq!(page_count(&mut v, "home/a"), pages, "the refused write changed nothing"); + assert_eq!(v.sync(), Ok(Vec::new())); + v.close(a).unwrap(); + let mut v = remount(v); + assert_eq!(page_count(&mut v, "home/a"), pages); + let a = v.open("home/a", PLAIN).unwrap(); + let mut out = vec![0u8; pages as usize * BLOCK]; + v.read(a, 0, &mut Buf(&mut out)).unwrap(); + assert!(out.iter().all(|&b| b == 3)); + } + + /// One file whose entry is refused leaves every other file's sync whole, + /// its own close refused, and its writes kept until a sync writes them. + #[test] + fn an_entry_refused_costs_only_its_own_file() { + let mut v = vol(); + let a = v.open("home/a", CREATE).unwrap(); + let b = v.open("home/b", CREATE).unwrap(); + v.write(a, 0, &[1; BLOCK]).unwrap(); + v.write(b, 0, &[2; 3 * BLOCK]).unwrap(); + // More runs than the entry names, as no write can give it. + let real = v.open[&a].extents.clone(); + let too_many = vec![real[0]; 300]; + v.open.get_mut(&a).unwrap().extents = too_many.clone(); + + assert_eq!(v.sync(), Ok(vec![(a, SyscallError::ResourceExhausted)])); + v.close(b).unwrap(); + assert_eq!(v.close(a), Err(SyscallError::ResourceExhausted), "a close that lost the writes is refused"); + assert_eq!(page_count(&mut v, "home/a"), 1, "the refused file is still what its writes made it"); + + v.open.get_mut(&a).unwrap().extents = real; + assert_eq!(v.sync(), Ok(Vec::new())); + assert!(v.open.is_empty(), "written, the unheld node goes"); + let mut v = remount(v); + assert_eq!(page_count(&mut v, "home/a"), 1); + assert_eq!(page_count(&mut v, "home/b"), 3); + } + /// A rename over an open, unsynced file that the format refuses leaves /// that file whole: readable through its holder, and its length reaching /// the volume at the next sync. The refusal is `EntryTooLarge`: `from`'s /// extents fit its own short name and not `to`'s long one. #[test] fn a_refused_rename_over_an_open_file_keeps_its_unsynced_writes() { - let mut v = vol(); + let mut v = fragmented(); let from = v.open("home/f", CREATE).unwrap(); - let spacer = v.open("home/spacer", CREATE).unwrap(); - // Alternating allocations, so no two of `from`'s blocks are adjacent - // and each is an extent of its own. for page in 0..240u64 { v.write(from, page * BLOCK as u64, &[1; BLOCK]).unwrap(); - v.write(spacer, page * BLOCK as u64, &[2; BLOCK]).unwrap(); } let to_path = format!("home/{}", "t".repeat(300)); let to = v.open(&to_path, CREATE).unwrap(); @@ -834,8 +968,8 @@ mod tests { let mut back = [0u8; 22]; assert_eq!(v.read(to, 0, &mut Buf(&mut back)), Ok(22), "the holder of `to` still reads it"); assert_eq!(&back, b"written and not synced"); - v.sync().unwrap(); - v.close(to); + assert_eq!(v.sync(), Ok(Vec::new())); + v.close(to).unwrap(); assert_eq!(v.lstat(&to_path).unwrap().size, 22, "`to`'s length reached the volume"); } } diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index 3b8647784d4..a557a37c1c6 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -177,6 +177,14 @@ impl FatVolume { Ok(()) } + /// Forget a node nobody holds and whose entry is level. + fn release(&mut self, node: Node) { + let open = self.open.remove(&node).expect("an open node"); + if self.by_path.get(&open.path) == Some(&node) { + self.by_path.remove(&open.path); + } + } + fn orphan(&mut self, path: &str) { if let Some(node) = self.by_path.remove(path) { if let Some(open) = self.open.get_mut(&node) { @@ -262,28 +270,28 @@ impl Volume for FatVolume { self.open.get_mut(&node).expect("just found or made").holders += 1; if how.truncate { if let Err(e) = self.truncate(node, 0) { - self.close(node); + // A refused close is said, and its node kept for the next sync. + let _ = self.close(node); return Err(e); } } Ok(node) } - fn close(&mut self, node: Node) { - let Some(open) = self.open.get_mut(&node) else { return }; + fn close(&mut self, node: Node) -> Result<(), SyscallError> { + let Some(open) = self.open.get_mut(&node) else { return Ok(()) }; open.holders -= 1; if open.holders > 0 { - return; + return Ok(()); } if self.writable { if let Err(e) = self.level(node) { - println!("fsd: node {node} was not brought level at its last close: {e:?}"); + println!("fsd: node {node} was not brought level at its last close ({e:?}); the next sync does it"); + return Err(e); } } - let open = self.open.remove(&node).expect("present above"); - if self.by_path.get(&open.path) == Some(&node) { - self.by_path.remove(&open.path); - } + self.release(node); + Ok(()) } fn hold(&mut self, node: Node) { @@ -385,15 +393,21 @@ impl Volume for FatVolume { Err(SyscallError::NotSupported) } - fn sync(&mut self) -> Result<(), SyscallError> { + fn sync(&mut self) -> Result, SyscallError> { + let mut unlevel = Vec::new(); if !self.writable { - return Ok(()); + return Ok(unlevel); } let nodes: Vec = self.open.keys().copied().collect(); for node in nodes { - self.level(node)?; + match self.level(node) { + Err(e) => unlevel.push((node, e)), + Ok(()) if self.open[&node].holders == 0 => self.release(node), + Ok(()) => {} + } } - self.fs.sync().map_err(|e| logged("sync", "", e)) + self.fs.sync().map_err(|e| logged("sync", "", e))?; + Ok(unlevel) } fn describe(&self) -> String { diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index 75dfdaedd91..b732b389e8e 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -7,7 +7,7 @@ //! nothing else of the machine. Argv is the role, and for LOG and BOOT the //! unique GUID of the partition the loader named for it when no claim on it //! was minted; then the row's own arguments, which are tests' actuators -//! ([`END_ON`], [`END_AT_READ`]). +//! ([`END_ON`], [`END_AT_READ`], [`END_AT_MOUNT`]). //! //! **A connection is bound to the directory whose port it came in on**, and //! every path on it is resolved there (`fsd::resolve`); a write on a read-only @@ -93,6 +93,16 @@ const END_ON: &str = "--end-on"; /// but a boot config's `args`. const END_AT_READ: &str = "--end-at-read"; +/// `--end-at-mount `: that role's first server this boot, its volume +/// mounted, waits for a connection and ends without accepting it — a server +/// gone before its first answer, under a client already waiting on it, as a +/// test stages one. Once a boot, as [`END_AT_READ`] is. Armed by nothing but a +/// boot config's `args`. +const END_AT_MOUNT: &str = "--end-at-mount"; + +/// How long [`END_AT_MOUNT`] waits for the connection it ends under. +const END_AT_MOUNT_WAIT: Duration = Duration::from_secs(60); + const TOKEN_CLIENT: u64 = 1 << 32; const TOKEN_STREAM: u64 = 2 << 32; @@ -178,7 +188,7 @@ fn main() { let role = args.get(1).and_then(|r| Role::parse(r)).unwrap_or_else(|| { panic!("fsd: started with {args:?}; the first argument is a role: data, log or boot") }); - let (mut guid, mut end_on, mut end_at_read) = (None, None, Vec::new()); + let (mut guid, mut end_on, mut end_at_read, mut end_at_mount) = (None, None, Vec::new(), None); let mut rest = args.iter().skip(2); while let Some(arg) = rest.next() { match arg.as_str() { @@ -186,6 +196,11 @@ fn main() { END_AT_READ => { end_at_read.push(rest.next().unwrap_or_else(|| panic!("fsd: {END_AT_READ} takes a path")).clone()) } + END_AT_MOUNT => { + end_at_mount = Some( + rest.next().and_then(|r| Role::parse(r)).unwrap_or_else(|| panic!("fsd: {END_AT_MOUNT} takes a role")), + ) + } flag if flag.starts_with("--") => panic!("fsd: {flag} is no argument of this server"), named if guid.is_none() => guid = Some(named), extra => panic!("fsd: a second partition, {extra}, after {guid:?}"), @@ -200,6 +215,9 @@ fn main() { caps.iter().map(|c| c.dir.as_str()).collect::>().join(", "), volume.describe() ); + if end_at_mount == Some(role) && first_this_boot(&format!("end-at-mount-{role:?}")) { + end_under_a_waiting_connection(&caps); + } let caps_len = caps.len() as u32; Server { volume, @@ -385,24 +403,38 @@ fn stat_reply(meta: Meta) -> Reply { Reply { kind: meta.kind.wire(), value2: meta.size, mtime: meta.mtime, ..Reply::ok() } } -/// Whether this is [`END_AT_READ`]'s first firing this boot for `path`: its -/// mark is made once in the kernel's `/tmp`, which outlives this process and -/// not the boot. Looked for and then made, not made exclusively — the std -/// fork's `create_new` on a kernel path is not exclusive — which one server -/// of a role at a time makes safe. A mark that can be neither found nor made -/// is an actuator that cannot do its one job. -fn first_read_this_boot(path: &str) -> bool { - let mark = format!("/tmp/fsd-end-at-read-{}", path.replace('/', "-")); +/// Whether this is an actuator's first firing this boot for `what`: its mark +/// is made once in the kernel's `/tmp`, which outlives this process and not +/// the boot. Looked for and then made, not made exclusively — the std fork's +/// `create_new` on a kernel path is not exclusive — which one server of a role +/// at a time makes safe. A mark that can be neither found nor made is an +/// actuator that cannot do its one job. +fn first_this_boot(what: &str) -> bool { + let mark = format!("/tmp/fsd-{}", what.replace('/', "-")); match std::fs::metadata(&mark) { Ok(_) => false, Err(e) if e.kind() == std::io::ErrorKind::NotFound => { - std::fs::write(&mark, b"").unwrap_or_else(|e| panic!("fsd: {END_AT_READ}: its mark {mark}: {e}")); + std::fs::write(&mark, b"").unwrap_or_else(|e| panic!("fsd: an actuator's mark {mark}: {e}")); true } - Err(e) => panic!("fsd: {END_AT_READ}: its mark {mark} would not be looked up: {e}"), + Err(e) => panic!("fsd: an actuator's mark {mark} would not be looked up: {e}"), } } +/// [`END_AT_MOUNT`]'s end: once a connection waits on one of `caps`' ports, +/// and before it is taken. +fn end_under_a_waiting_connection(caps: &[Capability]) -> ! { + let poller = Poller::new(caps.len() as u32); + for (i, cap) in caps.iter().enumerate() { + poller.watch(&cap.acceptor, READABLE, i as u64); + } + let mut waiting = false; + poller.wait(1, END_AT_MOUNT_WAIT.as_nanos() as u64, |_| waiting = true); + assert!(waiting, "fsd: {END_AT_MOUNT}: no connection came in {} s", END_AT_MOUNT_WAIT.as_secs()); + println!("fsd: {END_AT_MOUNT}: ending with a connection waiting and unaccepted"); + std::process::exit(1); +} + fn window_of(w: &SharedMemory) -> Window { // SAFETY: the region is `WINDOW_BYTES` long (it came through `adopt` at // that size, and shared memory is never less) and lives as long as `w`. @@ -475,10 +507,10 @@ impl Server { self.drop_client(id, "it never lent its window"); } if self.dirty_since.is_some_and(|at| now.duration_since(at) >= WRITEBACK) { - if let Err(e) = self.volume.sync() { + self.dirty_since = None; + if let Err(e) = self.sync(None) { println!("fsd: the write-back sync failed: {e:?}"); } - self.dirty_since = None; } } } @@ -524,7 +556,9 @@ impl Server { println!("fsd: dropping client {id} of {}: {why}", self.caps[client.cap].dir); } for fid in client.fids.values() { - self.volume.close(fid.node); + // Nobody is left to answer: a refused close is said, and its node + // kept for the next sync. + let _ = self.volume.close(fid.node); } } @@ -672,7 +706,7 @@ impl Server { Resolved::Path(p) => p, }; let reads_only = r.flags & (O_WRITE | O_APPEND | O_CREATE | O_TRUNCATE | O_CREATE_NEW) == 0; - if reads_only && self.end_at_read.contains(&path) && first_read_this_boot(&path) { + if reads_only && self.end_at_read.contains(&path) && first_this_boot(&format!("end-at-read-{path}")) { println!("fsd: {END_AT_READ}: ending before the first read of {path} is answered"); std::process::exit(1); } @@ -688,7 +722,7 @@ impl Server { let meta = match self.volume.node_meta(node) { Ok(meta) => meta, Err(e) => { - self.volume.close(node); + let _ = self.volume.close(node); return Err(e); } }; @@ -708,7 +742,7 @@ impl Server { CLOSE => { let client = self.clients.get_mut(&id).expect("pumped"); let fid = client.fids.remove(&r.fid).ok_or(SyscallError::InvalidArgument)?; - self.volume.close(fid.node); + self.volume.close(fid.node)?; Ok(Answer::Reply(Reply::ok())) } READ => { @@ -766,10 +800,10 @@ impl Server { Ok(Answer::Reply(Reply::ok())) } FSYNC => { - self.fid(id, r.fid)?; - self.sync() + let node = self.fid(id, r.fid)?.node; + self.sync(Some(node)) } - SYNC => self.sync(), + SYNC => self.sync(None), STREAM => { let f = self.fid(id, r.fid)?; if !f.write { @@ -881,10 +915,16 @@ impl Server { } } - fn sync(&mut self) -> Result { - self.volume.sync()?; - self.dirty_since = None; - Ok(Answer::Reply(Reply::ok())) + /// The volume synced, answered as `node`'s file fared, or with none as + /// every file did. A file whose entry was refused keeps the write-back + /// due, so the next one tries it again. + fn sync(&mut self, node: Option) -> Result { + let unwritten = self.volume.sync()?; + self.dirty_since = (!unwritten.is_empty()).then(Instant::now); + match unwritten.into_iter().find(|&(n, _)| node.is_none_or(|node| n == node)) { + Some((_, e)) => Err(e), + None => Ok(Answer::Reply(Reply::ok())), + } } fn link(&self, id: u64, path: &str) -> Answer { @@ -928,7 +968,9 @@ impl Server { println!("fsd: a stream is ended: {why}"); } if let Some(stream) = self.streams.remove(&sid) { - self.volume.close(stream.node); + // Nobody is left to answer: a refused close is said, and its node + // kept for the next sync. + let _ = self.volume.close(stream.node); } } } diff --git a/userland/fsd/src/volume.rs b/userland/fsd/src/volume.rs index ae7e1df765a..d084a4123a3 100644 --- a/userland/fsd/src/volume.rs +++ b/userland/fsd/src/volume.rs @@ -88,8 +88,10 @@ pub trait Volume { fn open(&mut self, path: &str, how: OpenHow) -> Result; - /// One holder gone; the node goes with its last. - fn close(&mut self, node: Node); + /// One holder gone; the node goes with its last, once its entry is + /// written. `Err` is that entry refused: the node stays, unheld, for the + /// next sync to write, and the refusal is the last holder's answer. + fn close(&mut self, node: Node) -> Result<(), SyscallError>; /// Hold the node once more, for a second file id or a stream. fn hold(&mut self, node: Node); @@ -116,8 +118,11 @@ pub trait Volume { fn symlink(&mut self, path: &str, target: &str) -> Result<(), SyscallError>; - /// Everything written, and every open file's length, durable. - fn sync(&mut self) -> Result<(), SyscallError>; + /// Everything written, and every open file's length, durable, but for the + /// files the answer names: each is one whose entry was refused, with why, + /// and no refusal keeps any other file from the device. `Err` is the + /// volume's own flush refused. + fn sync(&mut self) -> Result, SyscallError>; /// One line of what the volume and its cache have done, for a log. fn describe(&self) -> String; diff --git a/userland/logd/src/inspect.rs b/userland/logd/src/inspect.rs index d4a1a675dbf..d1551f9babe 100644 --- a/userland/logd/src/inspect.rs +++ b/userland/logd/src/inspect.rs @@ -48,10 +48,8 @@ pub enum State { Writing = 0, /// Every round answered, and slower than a log is worth. Degraded = 1, - /// A round the volume would not start, carried into the next. - Retrying = 2, /// No volume, or one given up on: the log is on the console only. - ConsoleOnly = 3, + ConsoleOnly = 2, } impl State { @@ -59,8 +57,7 @@ impl State { match raw { 0 => "writing", 1 => "degraded", - 2 => "retrying", - 3 => "console-only", + 2 => "console-only", other => unreachable!("logd: volume state {other} was never published"), } } diff --git a/userland/logd/src/main.rs b/userland/logd/src/main.rs index 7e96ee59bce..91c89c7ad15 100644 --- a/userland/logd/src/main.rs +++ b/userland/logd/src/main.rs @@ -63,9 +63,7 @@ //! //! **A round's lines are written and the volume made durable before the next //! round**, and again when init asks before the machine stops -//! ([`toyos_logstream::FLUSH`]). A sync the kernel declined to start is owed -//! and asked again each round until it is made. The kernel waits on none of -//! it: a panicking kernel's report is in its black box, and so are the stop's +//! ([`toyos_logstream::FLUSH`]). The kernel waits on none of it: a panicking kernel's report is in its black box, and so are the stop's //! own last records. //! **A line after that flush reaches the console and is held back from the //! file**, to a bound ([`STOPPING_BYTES`]): the stop syncs no file still open, @@ -73,14 +71,6 @@ //! served log is handed what the file holds and no more. A stop the kernel //! refused is init's [`toyos_logstream::RESUME`], and what was held is written //! then. -//! -//! `SYS_FSYNC` reaches the device's own cache flush. **A flush that would block -//! is not a flush that failed**: `io::ErrorKind::WouldBlock` from `sync_all` is -//! `kernel/src/block.rs`'s `BlockError::BudgetExpired`, which means the kernel -//! declined to *start* the operation on the caller's own clock — nothing was -//! issued, the device is untouched, and the bytes are still in the file waiting -//! for the next flush. `policy::fate` is the whole decision, and `policy`'s own -//! header is the argument. mod inspect; mod origin; @@ -198,7 +188,6 @@ fn main() { waiting: Vec::new(), volume, owed: false, - retrying_since: None, degraded: false, boot_local, hub, @@ -226,10 +215,8 @@ struct Log { /// round merges. waiting: Vec, volume: Option, - /// Lines the volume took and has not made durable: its sync was declined. + /// Lines the volume took and has not made durable. owed: bool, - /// When the current run of consecutive refused rounds began. - retrying_since: Option, /// Whether the volume answers, slower than `LOG_WRITE_BUDGET` a round. degraded: bool, boot_local: Option, @@ -272,11 +259,10 @@ impl Log { // Programs whose end a watch reported, until a round has swept them. let mut ended: Vec = Vec::new(); loop { - let state = match (&self.volume, self.retrying_since, self.degraded) { - (None, _, _) => inspect::State::ConsoleOnly, - (Some(_), Some(_), _) => inspect::State::Retrying, - (Some(_), None, true) => inspect::State::Degraded, - (Some(_), None, false) => inspect::State::Writing, + let state = match (&self.volume, self.degraded) { + (None, _) => inspect::State::ConsoleOnly, + (Some(_), true) => inspect::State::Degraded, + (Some(_), false) => inspect::State::Writing, }; published.publish(self.volume.as_ref(), state, self.tail.lost()); @@ -637,7 +623,7 @@ impl Log { fn to_volume(&mut self, text: &[u8], sync: bool) { let Some(v) = self.volume.as_mut() else { return }; let began = Instant::now(); - let mut refused = v.write(text).err().map(|e| (Step::Append, e.kind(), e.to_string())); + let mut refused = v.write(text).err().map(|e| (Step::Append, e.to_string())); if refused.is_none() { self.owed = true; if sync { @@ -646,36 +632,31 @@ impl Log { } // A volume that answered, and took longer than a log is worth doing it. if refused.is_none() && began.elapsed() > LOG_WRITE_BUDGET { - refused = - Some((Step::TooSlow, std::io::ErrorKind::Other, format!("it took {:?}", began.elapsed()))); + refused = Some((Step::TooSlow, format!("it took {:?}", began.elapsed()))); } - self.answered(began, refused); + self.answered(refused); } - /// Ask again for a sync the volume declined, whether or not this round - /// wrote anything. fn sync_owed(&mut self) { if self.owed { - let began = Instant::now(); let refused = self.sync().err(); - self.answered(began, refused); + self.answered(refused); } } /// Make the volume durable now. - fn sync(&mut self) -> Result<(), (Step, std::io::ErrorKind, String)> { + fn sync(&mut self) -> Result<(), (Step, String)> { let Some(v) = self.volume.as_mut() else { return Ok(()) }; - v.sync().map_err(|e| (Step::Flush, e.kind(), e.to_string()))?; + v.sync().map_err(|e| (Step::Flush, e.to_string()))?; self.owed = false; Ok(()) } /// What a round's write came to: rotation after a clean one, and the /// give-up policy after a refused one. - fn answered(&mut self, began: Instant, refused: Option<(Step, std::io::ErrorKind, String)>) { + fn answered(&mut self, refused: Option<(Step, String)>) { let Some(path) = self.volume.as_ref().map(Volume::path) else { return }; - let Some((step, kind, why)) = refused else { - self.retrying_since = None; + let Some((step, why)) = refused else { if self.degraded { self.degraded = false; say!("logd: {DIR} answers at pace again - {path}"); @@ -683,11 +664,7 @@ impl Log { self.rotate_if_full(); return; }; - // The run of consecutive retries, which is what `LOG_WRITE_BUDGET` - // bounds, from when its first round began. - let first = self.retrying_since.is_none(); - let since = *self.retrying_since.get_or_insert(began); - match fate(step, kind, since.elapsed()) { + match fate(step) { // Stop feeding the volume, say so once, and keep running. Fate::GiveUp => { toyos::error!( @@ -697,20 +674,8 @@ impl Log { ); self.volume = None; } - // Nothing is durable, so the next round's flush covers these bytes - // as well as its own; one line per run. - Fate::Retry => { - if first { - toyos::warn!( - "logd: {DIR} would not start ({}: {why}) - nothing was lost, so {path} is \ - still this boot's log and the next round is a retry", - step.as_str() - ); - } - } // Every call answered, slowly: the round is durable. Fate::Degraded => { - self.retrying_since = None; self.owed = false; if !self.degraded { self.degraded = true; @@ -740,7 +705,7 @@ impl Log { /// written by now, so the volume is made durable, and init is told. fn flushed(&mut self) { let refused = self.sync().err(); - self.answered(Instant::now(), refused); + self.answered(refused); self.stopping = Some(Stopping { text: String::new(), unwritten: 0 }); self.feed_console(); if let Err(e) = self.from_init.signal(FLUSHED) { diff --git a/userland/logd/src/policy.rs b/userland/logd/src/policy.rs index 118835e28f3..d99c37ca53a 100644 --- a/userland/logd/src/policy.rs +++ b/userland/logd/src/policy.rs @@ -3,98 +3,19 @@ //! **One pure function, because this is the whole refused-batch policy and it used to be //! `if let Err(_) = … { volume = None }`.** Everything above [`fate`] is I/O //! and everything below it is what the console says; the decision itself is a -//! function of three things a host test can hand it, which is why this file -//! exists rather than the two lines it replaces living in the loop. -//! -//! # The distinction this is built on -//! -//! `SYS_FSYNC` can fail for two reasons that used to arrive as one word. A -//! stick that cannot flush is `io::ErrorKind::Other` — `SyscallError::Io`, -//! which is what `kernel/src/block.rs`'s `BlockError::Device` becomes — and -//! ending the boot's log on one is right: the writes are not durable and no -//! number of retries will make them so. A *budget* that expired is -//! `io::ErrorKind::WouldBlock` — and -//! ending the log on one is wrong: nothing was issued, the device is untouched, -//! and the next operation gets a whole fresh `block::OPERATION`. -//! -//! The measurement that made the difference worth threading (2026-08-22): one -//! red in 73 full 12-wide suites, a guest that spent `syscall_wall=2108ms` -//! inside one `SYS_FSYNC` while its peers booted in 1,385 ms, and a boot's log -//! ended for a stick that answered every transfer. -//! -//! # Who bounds slowness now (owner ruling 2026-08-23) -//! -//! The kernel. `SYS_FSYNC` retries a budget-refused flush itself, between -//! attempts and off every lock, bounded by `kernel/src/block.rs`'s `DEADMAN` — -//! so a syscall this program makes either comes back durable inside that -//! deadman or comes back with a device error, and the three ways a volume dies -//! are all device facts by the time they arrive here: an error status, a reset -//! escalation that failed, or the deadman expired. What is left to this -//! program is the *policy over answers*: an error ends the volume, a refusal -//! that would block retries inside its own bounded run, and a round that -//! succeeded slowly is **degraded, not dead** — the records are durable, the -//! volume is kept, and the console says so once. While a slow round is in -//! flight nothing is dropped: the lines it covers are in the file and not yet -//! durable — buffered, said in one word. -//! -//! # Why only the flush retries -//! -//! A refused *append* is bytes this boot's file will not have — the records -//! have already left the cursor, so there is nothing to re-write — and a file -//! with a hole in it is exactly what this program's give-up policy exists to -//! stop pretending about. A refused *flush* loses nothing: the bytes are in the -//! file, and the next batch's flush covers them as well as its own. So the -//! asymmetry is deliberate and is what [`Step`] is for. +//! function of what a host test can hand it, which is why this file exists +//! rather than the two lines it replaces living in the loop. use std::time::Duration; -/// The longest a run of refused batches may keep being retried, and the point -/// past which a slow round is *announced* — no longer the point past -/// which a volume is declared dead for slowness, which since the 2026-08-23 -/// ruling nothing in this program is: only a device fact ends a volume, and -/// the kernel's `DEADMAN` is what turns a hung device into one. +/// The round time past which a volume that answered every call is announced +/// slow. Nothing in this program ends a volume for slowness: only the device's +/// own refusal does. /// /// **A policy number, and it says so**: nothing about the device supplies one. /// Five seconds: long enough that a slow stick under a boot's worth of other /// I/O is not called degraded, short enough that a person watching the console /// learns about it while they are still watching. -/// -/// **It is measured around a syscall, so it is reachable only if the syscall -/// returns** — every bound below it is what decides whether this policy runs at -/// all. There are two of them and they answer different failures. The transport -/// bounds one device round trip (`USB_TIMEOUT_NS`, 2 s in -/// `kernel/src/drivers/xhci`), which is what turns a stick that *stopped -/// answering* into an `Err` here rather than an unbounded wait; that bound is -/// never reached by a device that answers, so on its own it says nothing about -/// how long a call may take. `kernel/src/block.rs`'s `OPERATION` is the other, -/// 2 s over one whole block-device operation — the batching, the retries and -/// the recoveries a single `read_blocks` composes — and it is what bounds a -/// device that answers every transfer and takes too long over the work. Two -/// plus one command's overshoot is what leaves this constant a second to notice -/// with. -/// -/// **What it bounds is slowness and not errors, and that split is measured -/// rather than chosen.** The "repeated errors" half of a slowness-and-errors -/// policy does not survive this tree: a failing write is *itself* logged by the driver -/// (`usb-storage: cache flush failed on disk 0`), which commits a kernel record, -/// which is a record this program then tries to write, which fails. Retrying -/// inside a budget therefore does not sample a device that might recover — it -/// runs a feedback loop, measured at **1,737 failing flushes over six seconds** -/// under `usb-flush-fails` before this constant was given the narrower job. -/// -/// So an **error about the device** ends it at once, which is what -/// `kernel/src/log_file.rs` did and for a reason that turns out to be this one -/// rather than an idle loop's convenience. What this bounds is a run of budget -/// refusals — a flush the kernel keeps answering `WouldBlock` — so a -/// permanently loaded host does not keep a volume nobody is writing to; and a -/// *successful* round past it is the degradation threshold, where the console -/// learns the volume is slow without the log being thrown away over it. A -/// slow-but-answering device stopped being this constant's to end on -/// 2026-08-23: the kernel's `SYS_FSYNC` now bounds a slow device itself -/// (retry inside `block::DEADMAN`, then a device error), so a round here can -/// outlast this number and still be a durability the machine got. -/// `usb_flush_optional` is the gate for the error half and `--slow-usb` the -/// instrument for the slow one. pub const LOG_WRITE_BUDGET: Duration = Duration::from_secs(5); /// Which call refused. @@ -127,9 +48,6 @@ impl Step { /// What happens to this boot's volume. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Fate { - /// Keep it, publish nothing for this batch, and try again on the next one. - /// The bytes are in the file and the next flush covers them. - Retry, /// Keep it, publish this batch — every call answered, so the records are /// durable — and say once that the volume is slow. The state the whole /// slow-vs-failed split exists for: degraded is not dead, and a log that @@ -140,94 +58,39 @@ pub enum Fate { GiveUp, } -/// The decision, from the three things it is a function of. -/// -/// `retried_for` is how long the *run* of consecutive retries has lasted, and -/// [`Duration::ZERO`] when this is the first refusal after an answered batch. -/// It is the run and not the round because that is what -/// [`LOG_WRITE_BUDGET`] is about: a round that is refused early costs almost no -/// time, so a per-round check on a permanently loaded host would keep a volume -/// forever. -pub fn fate(step: Step, kind: std::io::ErrorKind, retried_for: Duration) -> Fate { - match (step, kind) { - // Nothing was issued, the device is untouched, and the next operation - // gets a whole fresh budget — as long as the run of them is still - // inside what a log is worth waiting for. - (Step::Flush, std::io::ErrorKind::WouldBlock) if retried_for <= LOG_WRITE_BUDGET => { - Fate::Retry - } +/// The decision, from the call that refused. +pub fn fate(step: Step) -> Fate { + match step { // Every call answered: the records are durable, however long the round - // took, and a volume is never ended on elapsed time — only on the - // device's own word, which for a hung device the kernel's deadman - // supplies as an error the arm below sees. - (Step::TooSlow, _) => Fate::Degraded, - _ => Fate::GiveUp, + // took, and a volume is never ended on elapsed time. + Step::TooSlow => Fate::Degraded, + Step::Append | Step::Flush => Fate::GiveUp, } } #[cfg(test)] mod tests { use super::*; - use std::io::ErrorKind; - - /// The sighting this exists for: `esp_filesystem`'s `fsync` refused under a - /// loaded host, on a stick that answered every transfer. - #[test] - fn a_flush_that_ran_out_of_budget_keeps_the_volume() { - assert_eq!(fate(Step::Flush, ErrorKind::WouldBlock, Duration::ZERO), Fate::Retry); - } - - /// **The one the whole change is measured against.** A stick that cannot - /// flush is `Other` (`SyscallError::Io`), and ending the boot's log on one - /// is the shipped behaviour this must not change: the writes are not - /// durable and no retry makes them so. - #[test] - fn a_device_that_cannot_flush_still_ends_the_volume() { - assert_eq!(fate(Step::Flush, ErrorKind::Other, Duration::ZERO), Fate::GiveUp); - assert_eq!(fate(Step::Flush, ErrorKind::PermissionDenied, Duration::ZERO), Fate::GiveUp); - assert_eq!(fate(Step::Flush, ErrorKind::NotFound, Duration::ZERO), Fate::GiveUp); - assert_eq!(fate(Step::Flush, ErrorKind::OutOfMemory, Duration::ZERO), Fate::GiveUp); - } - - /// A refused append is a hole in the file, whatever refused it: the records - /// have left the cursor and there is nothing to re-write. - #[test] - fn a_refused_append_ends_the_volume_whichever_word_it_used() { - assert_eq!(fate(Step::Append, ErrorKind::WouldBlock, Duration::ZERO), Fate::GiveUp); - assert_eq!(fate(Step::Append, ErrorKind::Other, Duration::ZERO), Fate::GiveUp); - } - /// A run of retries is itself the slow volume `LOG_WRITE_BUDGET` bounds, so - /// a host that stays loaded does not keep a volume nobody is writing to. + /// A stick that cannot flush ends the boot's log: the writes are not + /// durable and no retry makes them so. A refused append is a hole in the + /// file: the records have left the cursor and there is nothing to + /// re-write. #[test] - fn a_run_of_retries_longer_than_the_budget_ends_the_volume() { - let kind = ErrorKind::WouldBlock; - assert_eq!(fate(Step::Flush, kind, LOG_WRITE_BUDGET), Fate::Retry); - assert_eq!( - fate(Step::Flush, kind, LOG_WRITE_BUDGET + Duration::from_millis(1)), - Fate::GiveUp - ); + fn a_refused_append_or_flush_ends_the_volume() { + assert_eq!(fate(Step::Flush), Fate::GiveUp); + assert_eq!(fate(Step::Append), Fate::GiveUp); } /// A volume that answered every call and took too long over it is /// degraded, never dead: the records are durable and elapsed time is not - /// one of the three evidences a volume may be declared failed on. Until - /// 2026-08-23 this arm was `GiveUp`, which was declaring death on the - /// clock — the exact policy the slow-vs-failed split deletes. + /// one of the evidences a volume may be declared failed on. #[test] fn a_slow_round_degrades_and_keeps_the_volume() { - assert_eq!(fate(Step::TooSlow, ErrorKind::Other, Duration::ZERO), Fate::Degraded); - // However long the run has gone on: slowness never hardens into death - // here, because for a device that stops answering entirely the - // kernel's deadman supplies a real error and the GiveUp arm sees it. - assert_eq!( - fate(Step::TooSlow, ErrorKind::Other, LOG_WRITE_BUDGET + Duration::from_secs(60)), - Fate::Degraded - ); + assert_eq!(fate(Step::TooSlow), Fate::Degraded); } - /// The console words, which the four-way table in `main` is written - /// against. + /// The console words, which the table in `main` is written against. #[test] fn every_step_names_itself() { assert_eq!(Step::Append.as_str(), "the append"); diff --git a/userland/netd/src/report.rs b/userland/netd/src/report.rs index 76a23aab6dd..e11d58a26c2 100644 --- a/userland/netd/src/report.rs +++ b/userland/netd/src/report.rs @@ -2,9 +2,8 @@ //! file on the log volume, each flushed to the device before the process goes //! on. //! -//! **`sync_all` is the whole claim**: `SYS_FSYNC` reaches the device's own cache -//! flush, which is what `/system/bin/logd` rests its durability word on, so a -//! line this returned from is on the stick whatever the machine does next. A +//! **`sync_all` is the whole claim**: a line this returned from is on the +//! stick whatever the machine does next. A //! line that cannot be made durable ends the process: a report with a hole in //! it says the wrong thing. From 004d4559684a78b78b59c0fe17a77e025710ae1f Mon Sep 17 00:00:00 2001 From: japabu Date: Sun, 27 Sep 2026 23:38:48 +0200 Subject: [PATCH 34/54] review #536 r8: the refused-entry, give-back, FAT and Unusable arms tested; a map-model oracle for DATA DATA (userland/fsd/src/data.rs): - a sync after a refused close, with the entry still too large, names the file again and keeps its node (review patch at :709 goes red); - a refused write leaves the synced superblock's free count where it was (`let _ = dropped;` goes red); - `every_file_reads_back_as_a_plain_map_of_its_accepted_writes`: three files grow in turn past 2 x 250 pages through overwrites, holes, a shrink and writes refused for want of blocks, against a HashMap of what each accepted write made, compared byte for byte after a remount. SPREAD = 2 and the cursor-only allocation are each refused a write past 250 pages. It replaces the length-only two-file test. FAT (userland/fsd/src/fat.rs): on the specification's fixture volume, from toyos-fat32-check's tests, one file's entry block evicted and refused: the sync names only it and levels the other, its close is refused and its node kept, and the next sync levels it. blockd: the partitions live inside `Drive::Up`, so no listing names one without a controller. `listing` and `place` are the handshake's and the open's decisions, host-tested: `Unusable` is refused `Unusable`, `Absent` lists nothing and opens `NotFound`. `nvme_wide_sector` now also requires fsd's "DATA is absent this boot" and forbids "no DATA partition". Also: fsd's write-back decision is `WriteBack`, host-tested; the split's give-back of a sibling block is tested in bcachefs; logd's `Fate` map and the dead `owed`/`sync_owed` go; blockd_io's `nothing` starts blockd through `Blockd`; the dead "are a tmpfs" guards look for fsd's in-memory line; the fork expects getcwd's UTF-8 and loses "connecting once". Filed: the 250-run extent ceiling, a served file's `as_raw_fd` panic, `Dir::symlink`'s missing reconnect, and a refused write-back that waits for the next write. The sysroot issue goes to the toolchain PR that owns it. Co-Authored-By: Claude Opus 5.5 --- bcachefs/src/btree.rs | 22 ++++ ...olchains-rebuild-and-its-cargo-has-none.md | 27 ----- ...fused-at-250-runs-while-blocks-are-free.md | 25 ++++ ...erved-file-panics-when-asked-its-raw-fd.md | 20 ++++ ...device-refused-waits-for-the-next-write.md | 20 ++++ ...e-reconnect-every-other-path-call-makes.md | 18 +++ kernel/src/actuator.rs | 2 +- rust | 2 +- tests/common/pkg.rs | 4 +- tests/common/storage.rs | 8 +- tests/toyos-rust-tests/src/bin/blockd_io.rs | 50 ++++---- tests/toyos.rs | 21 ++-- userland/blockd/src/main.rs | 107 +++++++++++------- userland/fsd/src/data.rs | 94 ++++++++++++--- userland/fsd/src/fat.rs | 75 +++++++++++- userland/fsd/src/lib.rs | 1 + userland/fsd/src/main.rs | 22 ++-- userland/fsd/src/writeback.rs | 64 +++++++++++ userland/logd/src/main.rs | 34 ++---- userland/logd/src/policy.rs | 53 --------- 20 files changed, 448 insertions(+), 221 deletions(-) delete mode 100644 issues/build/a-sysroot-made-between-the-toolchains-rebuild-and-its-cargo-has-none.md create mode 100644 issues/filesystem/a-file-on-data-is-refused-at-250-runs-while-blocks-are-free.md create mode 100644 issues/filesystem/a-served-file-panics-when-asked-its-raw-fd.md create mode 100644 issues/filesystem/a-write-back-the-device-refused-waits-for-the-next-write.md create mode 100644 issues/filesystem/dir-symlink-skips-the-reconnect-every-other-path-call-makes.md create mode 100644 userland/fsd/src/writeback.rs diff --git a/bcachefs/src/btree.rs b/bcachefs/src/btree.rs index 1050772da8e..59e6ba3d8cc 100644 --- a/bcachefs/src/btree.rs +++ b/bcachefs/src/btree.rs @@ -806,6 +806,28 @@ mod tests { ); } + /// A split that finds no block for a later sibling gives back the ones it + /// took for the earlier: nothing names them yet, so nothing else frees + /// them. + #[test] + fn a_split_short_of_a_sibling_gives_back_the_ones_it_took() { + let io = crate::block_io::VecBlockIO::new(16); + let mut alloc = BitmapAllocator::format(&io, BlockNum::new(0), 1, 16, 1).unwrap(); + let taken = alloc.free_blocks as u32 - 2; + let _ = alloc.alloc_exact(&io, taken).unwrap(); + let leaf = alloc.alloc_block(&io).unwrap(); + assert_eq!(alloc.free_blocks, 1); + + let mut middle: Vec = (0..40).map(|_| entry(72)).collect(); + middle.insert(20, entry(MAX_ENTRY_SIZE - KEY_HEADER_SIZE)); + assert_eq!(pack(middle.clone()).len(), 3, "two siblings, and a block for one"); + assert!(matches!( + split_node(&io, &mut alloc, leaf, Node::Leaf(middle)), + Err(FsError::NoSpace { .. }), + )); + assert_eq!(alloc.free_blocks, 1, "the first sibling's block went back"); + } + #[test] fn a_descent_gives_up_before_it_runs_out_of_stack() { let mut depth = Depth::ROOT; diff --git a/issues/build/a-sysroot-made-between-the-toolchains-rebuild-and-its-cargo-has-none.md b/issues/build/a-sysroot-made-between-the-toolchains-rebuild-and-its-cargo-has-none.md deleted file mode 100644 index d14ea64eaf5..00000000000 --- a/issues/build/a-sysroot-made-between-the-toolchains-rebuild-and-its-cargo-has-none.md +++ /dev/null @@ -1,27 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-09-27 ---- - -# A sysroot made between the toolchain's rebuild and its cargo has none, for good - -`src/toolchain.rs` rebuilds the primary's toolchain in steps that each take -the global lock and let it go: "build the ToyOS-hosted rustc" recreates -`stage2/bin/`, and "give the toyos toolchain its own cargo" puts `bin/cargo` -back afterwards, under a lock of its own. A sysroot build in another worktree -(`src/sysroot.rs`'s `build`) takes the compiler lock shared between the two, -clones `stage2/` with no `cargo` in it, and writes the sysroot finished. A -sysroot is never written again, so every later build of that key refuses at -`assert_toolchain_is_honest`: "the toyos toolchain at …/sysroots//bin is -missing cargo", and a re-run refuses the same way. - -Measured on 2026-09-27: `sysroots/5dc157f7fac727be/bin` (22:03) and -`sysroots/e8d0e10f5498e275/bin` (22:07) hold `rustc` and `rustdoc` and no -`cargo`; the primary's `stage2/bin/cargo` link is dated 22:08. - -## Exit condition - -No sysroot can clone `stage2/` while its `bin/` is being remade: the rebuild -and the cargo that goes with it are one step under one hold of the lock, or -the clone refuses a `bin/` without `cargo` before it writes the sysroot. diff --git a/issues/filesystem/a-file-on-data-is-refused-at-250-runs-while-blocks-are-free.md b/issues/filesystem/a-file-on-data-is-refused-at-250-runs-while-blocks-are-free.md new file mode 100644 index 00000000000..39bf56e6813 --- /dev/null +++ b/issues/filesystem/a-file-on-data-is-refused-at-250-runs-while-blocks-are-free.md @@ -0,0 +1,25 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A file on DATA is refused at 250 runs while the volume has free blocks + +The DATA volume (`userland/fsd/src/data.rs`, over the `bcachefs` crate) keeps a +file's extents in its one btree entry, and one entry is at most +`MAX_ENTRY_SIZE` bytes (`bcachefs/src/btree.rs`): 250 runs for `home/a` +(`file_entry_fits`, `bcachefs/src/fs.rs`). On a volume whose free blocks are +all apart every run is one block, so a write that takes a file past 250 blocks +(1 MiB) is refused `ResourceExhausted` while the volume still has free blocks. +`SPREAD` lengthens the runs of files that grow in turn; it does nothing for +free space that is already fragmented. + +Owner: the fsd DATA volume. + +Evidence: `a_write_its_entry_could_not_name_is_refused_and_every_accepted_one_kept` +(`userland/fsd/src/data.rs`) is refused at `pages == 250` on a volume with +free blocks. + +**Exit**: a file's runs held outside its one entry, so that test writes past +250 runs. diff --git a/issues/filesystem/a-served-file-panics-when-asked-its-raw-fd.md b/issues/filesystem/a-served-file-panics-when-asked-its-raw-fd.md new file mode 100644 index 00000000000..65fb247e0b9 --- /dev/null +++ b/issues/filesystem/a-served-file-panics-when-asked-its-raw-fd.md @@ -0,0 +1,20 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A served file panics when asked its raw fd + +`std::os::toyos::io::AsRawFd` for `std::fs::File` calls the fork's +`File::as_raw_fd` (`rust/library/std/src/sys/fs/toyos.rs`), which panics with +"a file on a file server has no kernel handle" for every file under `/apps`, +`/config`, `/home`, `/state`, `/log` and `/boot`. A crate that takes a file's +fd — to lock it, map it or hand it to a C library — builds for ToyOS and +panics there, which is what "existing Rust just works" rules out. + +Owner: the std fork's ToyOS file layer. + +**Exit**: `as_raw_fd` on a served file answers without a panic — a kernel +handle that reaches the file's server, as `as_child_stdio`'s pipe does for a +child's writes — with a guest test that asks it of a file under `/home`. diff --git a/issues/filesystem/a-write-back-the-device-refused-waits-for-the-next-write.md b/issues/filesystem/a-write-back-the-device-refused-waits-for-the-next-write.md new file mode 100644 index 00000000000..e1b1d58aab4 --- /dev/null +++ b/issues/filesystem/a-write-back-the-device-refused-waits-for-the-next-write.md @@ -0,0 +1,20 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# A write-back the device refused waits for the next write + +fsd's write-back (`userland/fsd/src/writeback.rs`) is spent when it comes due, +and only a sync that answers is recorded: one that left a file unwritten stays +due and is tried again. A sync the volume refuses whole — the cache's flush or +the device — returns before that, so the write-back is spent and nothing +syncs the dirty blocks the cache still holds until a client writes or syncs +again. + +Owner: the fsd server loop (`userland/fsd/src/main.rs`). + +**Exit**: a refused sync keeps the write-back due at a bound that does not +say the refusal once every two seconds for the rest of the boot, with a host +test of `WriteBack` over a refused sync. diff --git a/issues/filesystem/dir-symlink-skips-the-reconnect-every-other-path-call-makes.md b/issues/filesystem/dir-symlink-skips-the-reconnect-every-other-path-call-makes.md new file mode 100644 index 00000000000..2fad0a24c4c --- /dev/null +++ b/issues/filesystem/dir-symlink-skips-the-reconnect-every-other-path-call-makes.md @@ -0,0 +1,18 @@ +--- +status: open +kind: defect +opened: 2026-09-27 +--- + +# `Dir::symlink` skips the reconnect every other path call makes + +`toyos/src/fs.rs`'s `Dir::symlink` writes its own request instead of going +through `Dir::path_call`, so it answers a server's restart `Gone` where every +other path call reconnects once and asks again. Its only difference of +substance is that the target is stored, not resolved, and so is not held to +`canonical`. + +Owner: the SDK's file-server client (`toyos/src/fs.rs`). + +**Exit**: `Dir::symlink` is one `path_call`, with the target exempt from +`canonical`, and a test that makes a symlink across a server's restart. diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 0c9e3afd646..5ad2c738515 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -482,7 +482,7 @@ actuators! { /// Make firmware name its own timezone. rtc_zone_east = "rtc-zone-east"; - /// Run the leak-rollback controls (device mint, FAT reopen) after mount. + /// Run the leak-rollback controls (device mint) after mount. leak_rollback_selftest = "leak-rollback-selftest"; /// Run the revoked-backing controls after mount. diff --git a/rust b/rust index 9abdd61d95a..fee9fa6f7bb 160000 --- a/rust +++ b/rust @@ -1 +1 @@ -Subproject commit 9abdd61d95a8c9f8f060437e452dcff9ab172030 +Subproject commit fee9fa6f7bb3c93a14ed7e0ab3b2505441f7392b diff --git a/tests/common/pkg.rs b/tests/common/pkg.rs index d737e2863dd..6428eed0304 100644 --- a/tests/common/pkg.rs +++ b/tests/common/pkg.rs @@ -78,9 +78,9 @@ pub fn pkg_install_gbae( }; let mut qemu = QemuInstance::boot_with_options(&config, &[], &bins, options); let boot = qemu.boot_log().to_string(); - if boot.contains("are a tmpfs") { + if boot.contains("are in memory and will not survive a reboot") { return Err(format!( - "/apps and /home fell back to tmpfs, so the readback below would judge no device:\n\ + "/apps and /home fell back to memory, so the readback below would judge no device:\n\ {boot}" )); } diff --git a/tests/common/storage.rs b/tests/common/storage.rs index ab26336e536..e02d3aec9df 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -532,9 +532,9 @@ pub fn home_overwrite_reads_back( BootOptions { profile: qemu::Profile::MetalDisk, ..Default::default() }, ); let boot = qemu.boot_log().to_string(); - if boot.contains("are a tmpfs") { + if boot.contains("are in memory and will not survive a reboot") { return Err(format!( - "/apps and /home fell back to tmpfs, so nothing below touches the NVMe path:\n{boot}" + "/apps and /home fell back to memory, so nothing below touches the NVMe path:\n{boot}" )); } @@ -778,9 +778,9 @@ pub fn apps_and_home_are_one_filesystem( BootOptions { profile: qemu::Profile::MetalDisk, ..Default::default() }, ); let boot = qemu.boot_log().to_string(); - if boot.contains("are a tmpfs") { + if boot.contains("are in memory and will not survive a reboot") { return Err(format!( - "/apps and /home fell back to tmpfs, so the readback below would judge no device:\n\ + "/apps and /home fell back to memory, so the readback below would judge no device:\n\ {boot}" )); } diff --git a/tests/toyos-rust-tests/src/bin/blockd_io.rs b/tests/toyos-rust-tests/src/bin/blockd_io.rs index 29fb6b05d9b..2e5b3c135c1 100644 --- a/tests/toyos-rust-tests/src/bin/blockd_io.rs +++ b/tests/toyos-rust-tests/src/bin/blockd_io.rs @@ -113,11 +113,12 @@ fn fail(what: String) -> ! { std::process::exit(1) } -/// blockd, held by this process: its claim minted here, the port's acceptor -/// kept here, so a blockd can end and another take its place on the same -/// name. What blockd says goes to this process's stdout, a line at a time. +/// blockd, held by this process: its claim minted here, from `syscap` when +/// there is one, the port's acceptor kept here, so a blockd can end and +/// another take its place on the same name. What blockd says goes to this +/// process's stdout, a line at a time. struct Blockd { - syscap: SysCap, + syscap: Option, acceptor: Acceptor, connector: Connector, child: Option, @@ -125,10 +126,10 @@ struct Blockd { impl Blockd { fn start(args: &[&str]) -> Self { - Self::with(capability(), args) + Self::with(Some(capability()), args) } - fn with(syscap: SysCap, args: &[&str]) -> Self { + fn with(syscap: Option, args: &[&str]) -> Self { let (acceptor, connector) = port::create().unwrap_or_else(|e| fail(format!("no port: {e:?}"))); let mut blockd = Self { syscap, acceptor, connector, child: None }; blockd.spawn(args, false); @@ -145,22 +146,24 @@ impl Blockd { /// write's answer: the write is done on the device, and its session never /// hears. fn spawn(&mut self, args: &[&str], kill_on_withheld: bool) { - let asked = Instant::now(); - let claim: toyos::Device = loop { - match self.syscap.claim_pci(BLOCKD) { - Err(SyscallError::AlreadyExists) if asked.elapsed() < CLAIM_RETURN => { - std::thread::sleep(Duration::from_millis(1)); + let mut command = Command::new("/system/bin/blockd"); + if let Some(syscap) = &self.syscap { + let asked = Instant::now(); + let claim: toyos::Device = loop { + match syscap.claim_pci(BLOCKD) { + Err(SyscallError::AlreadyExists) if asked.elapsed() < CLAIM_RETURN => { + std::thread::sleep(Duration::from_millis(1)); + } + Ok(claim) => break claim, + Err(e) => fail(format!("the controller's claim was refused: {e:?}")), } - Ok(claim) => break claim, - Err(e) => fail(format!("the controller's claim was refused: {e:?}")), - } - }; + }; + command.endow(&format!("{DEV_PREFIX}pci:8086:5845"), claim.into_raw().0); + } let acceptor = toyos_abi::syscall::dup(self.acceptor.as_handle()) .unwrap_or_else(|e| fail(format!("the acceptor would not duplicate: {e:?}"))); - let mut command = Command::new("/system/bin/blockd"); command.args(args); command.stdout(Stdio::piped()); - command.endow(&format!("{DEV_PREFIX}pci:8086:5845"), claim.into_raw().0); command.endow(&format!("{SERVE_PREFIX}{PORT}"), acceptor.0); let mut child = command.spawn().unwrap_or_else(|e| fail(format!("blockd did not start: {e}"))); let out = child.stdout.take().expect("piped"); @@ -405,7 +408,7 @@ fn mb_per_s(blocks: u64, took: Duration) -> f64 { /// the arena holds, the data built before and checked after what is timed, /// so each number is the driver's path and nothing of this binary's. fn bench() { - let blockd = Blockd::with(capability(), &[]); + let blockd = Blockd::start(&[]); let mut s = open(blockd.names(), BENCH); let mut runs = Vec::new(); for (salt, in_flight) in [(0x3D, 1usize), (0x3C, 15)] { @@ -965,12 +968,8 @@ fn dma_residue() { /// it answers every other: the malformed refused as such, a listing empty and /// an open `NotFound`. fn nothing() { - let (acceptor, connector) = port::create().unwrap_or_else(|e| fail(format!("no port: {e:?}"))); - let mut command = Command::new("/system/bin/blockd"); - command.endow(&format!("{SERVE_PREFIX}{PORT}"), acceptor.into_raw().0); - let mut child = command.spawn().unwrap_or_else(|e| fail(format!("blockd did not start: {e}"))); - let names = - namespace::build().add(PORT, &connector).finish().unwrap_or_else(|e| fail(format!("no namespace: {e:?}"))); + let blockd = Blockd::with(None, &[]); + let names = blockd.names(); let region = || { let region = SharedMemory::create(REGION).unwrap_or_else(|e| fail(format!("a region: {e:?}"))); vec![region.share().unwrap_or_else(|e| fail(format!("a second handle: {e:?}")))] @@ -995,8 +994,7 @@ fn nothing() { } println!("blockd_io: with no controller, {what} was answered {answered} {refusal:?}"); } - let _ = child.kill(); - let _ = child.wait(); + drop(blockd); println!("blockd_io: PASS nothing"); } diff --git a/tests/toyos.rs b/tests/toyos.rs index 3b2a99da2d9..8607076bbbf 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -421,20 +421,11 @@ const RUST_SKIP: &[&str] = &[ // shared boot it would be reported against whichever test came next — and // every one after that. `short_sleep_livelock` gives it a boot of its own. "abuse_short_sleep", - // The four below were **running twice under one name**, once here on the - // plain boot and once as the test that owns the name, and the collision was - // invisible: `check_registration` compared the three declared lists against - // each other and never against the binaries the registry discovers. - // `check_no_collisions` closes that, and this is what it found. - // - // What the shared copy adds is the binary exiting 0 on a boot that gives it - // nothing to measure. - // What it stages on `/log` — a file // unlinked out from under a held descriptor, its clusters handed to the next // writer — is only half the claim, and the other half is the volume read - // back off the image after a shutdown by a FAT implementation that is not - // the kernel's. `fat_backing_revoked` runs it. + // back off the image after a shutdown by a FAT implementation. + // `fat_backing_revoked` runs it. "fat_backing_revoked", // Stages a rename with an absent source on `/log` and leaves the // destination for `fs_rename_durable` to read back off the image. @@ -12019,8 +12010,8 @@ fn run_machine_test( "blockd_lends_within_its_bound" => { common::blockd::blockd_lends_within_its_bound(test_config, c_bins, rust_bins) } - // Body in `tests/common/hda.rs`, same reason. "blockd_serves_nothing" => common::blockd::blockd_serves_nothing(test_config, c_bins, rust_bins), + // Body in `tests/common/hda.rs`, same reason. "hda_tone" => common::hda::hda_tone(test_config, c_bins, rust_bins), "hda_client_stall" => common::hda::hda_client_stall(test_config, c_bins, rust_bins), "hda_two_live_refused" => { @@ -12914,6 +12905,12 @@ fn run_machine_test( log.must_not_say("divide by zero")?; // Nothing downstream ran: no queue was made over the namespace. log.must_not_say("blockd: NVMe up")?; + // A refused controller is a disk that failed: DATA is absent, and + // never served from memory as on a machine with no disk. + log.must_say( + "fsd: the block service would not list its partitions (Refused(Unusable)); DATA is absent this boot", + )?; + log.must_not_say("this machine has no DATA partition")?; log.must_say("Boot: complete")?; eprintln!(" [nvme] 8 KiB-format namespace refused by name, and the boot went on"); Ok(()) diff --git a/userland/blockd/src/main.rs b/userland/blockd/src/main.rs index 096ff706d3b..d4236619a28 100644 --- a/userland/blockd/src/main.rs +++ b/userland/blockd/src/main.rs @@ -192,11 +192,11 @@ struct Served { posted: bool, } -/// The controller this service drives, or why it drives none. Without one no -/// partition is listed and no session opened, and a connection says one frame -/// and is answered, or is let go. +/// The controller this service drives and the partitions its table names, or +/// why it drives none. Without one no partition is listed and no session +/// opened, and a connection says one frame and is answered, or is let go. enum Drive { - Up(Controller), + Up(Controller, Vec), /// No controller this row names is on this machine: a listing is empty /// and an open `NotFound`. Absent, @@ -207,17 +207,18 @@ enum Drive { } impl Drive { - /// The controller a session is on: a service without one opens none. + /// The controller a session is on: a service without one has no partition + /// to open. fn up(&mut self) -> &mut Controller { match self { - Drive::Up(ctrl) => ctrl, + Drive::Up(ctrl, _) => ctrl, Drive::Absent | Drive::Unusable => unreachable!("blockd: a session with no controller"), } } fn oldest(&self) -> Option { match self { - Drive::Up(ctrl) => ctrl.oldest(), + Drive::Up(ctrl, _) => ctrl.oldest(), Drive::Absent | Drive::Unusable => None, } } @@ -225,7 +226,6 @@ impl Drive { struct Service { ctrl: Drive, - parts: Vec, holds: Holds, /// The device's loss count: every reset may have dropped its cache. losses: u64, @@ -247,18 +247,29 @@ struct Opening { } impl Service { - fn new(ctrl: Drive, parts: Vec, running: Option<[u8; 16]>) -> Self { - Self { ctrl, parts, holds: Holds::new(), losses: 0, sessions: BTreeMap::new(), next_id: 0, running } + fn new(ctrl: Drive, running: Option<[u8; 16]>) -> Self { + Self { ctrl, holds: Holds::new(), losses: 0, sessions: BTreeMap::new(), next_id: 0, running } } - /// What an open answers: a session to admit, or its refusal. `region` is - /// the client's and is consumed either way. - fn open(&mut self, guid: [u8; 16], region: toyos::RawHandle) -> Result { - let region = Region::adopt(region).map_err(|_| Refusal::Malformed)?; - if let Drive::Unusable = self.ctrl { - return Err(Refusal::Unusable); + /// What a listing answers: every partition the table names. + fn listing(&self) -> Result, Refusal> { + match &self.ctrl { + Drive::Up(_, parts) => { + Ok(parts.iter().flat_map(|p| wire::Listed { unique: p.unique, kind: p.kind }.encode()).collect()) + } + Drive::Absent => Ok(Vec::new()), + Drive::Unusable => Err(Refusal::Unusable), } - let part = self.parts.iter().find(|p| p.unique == guid && guid != [0; 16]); + } + + /// The span an open of `guid` is served on, held for it; or its refusal. + fn place(&mut self, guid: [u8; 16]) -> Result<(u64, u64), Refusal> { + let parts = match &self.ctrl { + Drive::Up(_, parts) => parts, + Drive::Absent => return Err(Refusal::NotFound), + Drive::Unusable => return Err(Refusal::Unusable), + }; + let part = parts.iter().find(|p| p.unique == guid && guid != [0; 16]); let (first, blocks) = match part.map(|p| &p.span) { None => return Err(Refusal::NotFound), Some(Err(_)) => return Err(Refusal::Unusable), @@ -273,6 +284,14 @@ impl Service { if self.holds.hold(first, first + blocks, self.next_id).is_err() { return Err(Refusal::Held); } + Ok((first, blocks)) + } + + /// What an open answers: a session to admit, or its refusal. `region` is + /// the client's and is consumed either way. + fn open(&mut self, guid: [u8; 16], region: toyos::RawHandle) -> Result { + let region = Region::adopt(region).map_err(|_| Refusal::Malformed)?; + let (first, blocks) = self.place(guid)?; match self.ctrl.up().claim().dma_map(region.handle()) { Ok(mapping) if mapping.bytes == SESSION_BYTES as u64 => { Ok(Opening { region, device_addr: mapping.device_addr, first, blocks, unique: guid }) @@ -511,15 +530,13 @@ fn main() { let acceptor = endow::acceptor(PORT).unwrap_or_else(|| panic!("blockd: started serving no `{PORT}` port")); let Some(dev) = claim() else { println!("blockd: no NVMe controller this row names is on this machine; serving no partition"); - serve(&mut Service::new(Drive::Absent, Vec::new(), running), &acceptor); + serve(&mut Service::new(Drive::Absent, running), &acceptor); }; - // A controller this service cannot use is a machine without one, said by - // name: restarting would meet the same device and the same refusal. let mut ctrl = match Controller::open(dev, silence) { Ok(ctrl) => ctrl, Err(why) => { println!("blockd: NOT SERVING — {why}; serving no partition"); - serve(&mut Service::new(Drive::Unusable, Vec::new(), running), &acceptor); + serve(&mut Service::new(Drive::Unusable, running), &acceptor); } }; println!( @@ -544,7 +561,7 @@ fn main() { Err(why) => println!("blockd: partition {} is not served: {why}", guid_text(part.unique)), } } - serve(&mut Service::new(Drive::Up(ctrl), parts, running), &acceptor); + serve(&mut Service::new(Drive::Up(ctrl, parts), running), &acceptor); } fn serve(service: &mut Service, acceptor: &toyos::port::Acceptor) -> ! { @@ -553,7 +570,7 @@ fn serve(service: &mut Service, acceptor: &toyos::port::Acceptor) -> ! { let mut ready: Vec = Vec::new(); let mut done: Vec = Vec::new(); loop { - if let Drive::Up(ctrl) = &service.ctrl { + if let Drive::Up(ctrl, _) = &service.ctrl { poller.watch(ctrl.claim(), READABLE, TOKEN_IRQ); } if pending.len() < MAX_PENDING { @@ -577,7 +594,7 @@ fn serve(service: &mut Service, acceptor: &toyos::port::Acceptor) -> ! { ready.clear(); poller.wait(1, timeout, |token| ready.push(token)); - if let Drive::Up(ctrl) = &mut service.ctrl { + if let Drive::Up(ctrl, _) = &mut service.ctrl { if let Err(why) = ctrl.take_interrupt() { panic!( "blockd: the claim refused its interrupt read ({why:?}): the unit refused this \ @@ -658,20 +675,15 @@ fn handshake(service: &mut Service, p: Pending, msg_type: u32, payload_len: usiz let _ = conn.try_send_bytes(wire::MSG_REFUSED, &why.encode()); }; if msg_type == wire::MSG_LIST && payload_len == 0 { - if let Drive::Unusable = service.ctrl { - refuse(&p.conn, Refusal::Unusable); - return; - } - let listing: Vec = service - .parts - .iter() - .map(|p| wire::Listed { unique: p.unique, kind: p.kind }) - .flat_map(|l| l.encode()) - .collect(); - // One frame, and the connection is done with: a table larger than a - // frame is one this service lists no part of rather than half of. - if p.conn.try_send_bytes(wire::MSG_LISTED, &listing).is_err() { - refuse(&p.conn, Refusal::Exhausted); + match service.listing() { + Err(why) => refuse(&p.conn, why), + // One frame, and the connection is done with: a table larger than + // a frame is one this service lists no part of rather than half of. + Ok(listing) => { + if p.conn.try_send_bytes(wire::MSG_LISTED, &listing).is_err() { + refuse(&p.conn, Refusal::Exhausted); + } + } } return; } @@ -700,3 +712,22 @@ fn handshake(service: &mut Service, p: Pending, msg_type: u32, payload_len: usiz } } } + +#[cfg(test)] +mod tests { + use super::*; + + /// A controller this service cannot use is a disk that failed and never a + /// machine without one: its listing and its open are refused `Unusable`, + /// where an absent controller's listing is empty and its open `NotFound`. + #[test] + fn an_unusable_controller_is_refused_and_an_absent_one_lists_nothing() { + let guid = [7; 16]; + let mut unusable = Service::new(Drive::Unusable, None); + assert_eq!(unusable.listing(), Err(Refusal::Unusable)); + assert_eq!(unusable.place(guid), Err(Refusal::Unusable)); + let mut absent = Service::new(Drive::Absent, None); + assert_eq!(absent.listing(), Ok(Vec::new())); + assert_eq!(absent.place(guid), Err(Refusal::NotFound)); + } +} diff --git a/userland/fsd/src/data.rs b/userland/fsd/src/data.rs index 7b61d4a880d..ee8d410bdab 100644 --- a/userland/fsd/src/data.rs +++ b/userland/fsd/src/data.rs @@ -731,6 +731,8 @@ impl Volume for DataVolume { #[cfg(test)] mod tests { + use std::collections::HashMap; + use super::*; use crate::disk::Ram; use crate::volume::Buf; @@ -870,26 +872,86 @@ mod tests { v.lstat(path).unwrap().size / BLOCK as u64 } - /// Two files that grow a page at a time in turn each keep their runs, so - /// neither's entry fills with one extent per page. + /// The free count a sync leaves in the superblock. + fn free_blocks(v: &mut DataVolume) -> u64 { + assert_eq!(v.sync(), Ok(Vec::new())); + bcachefs::Superblock::read(v.fs.io()).unwrap().free_blocks + } + + fn write_into(file: &mut Vec, offset: u64, data: &[u8]) { + let end = offset as usize + data.len(); + if file.len() < end { + file.resize(end, 0); + } + file[offset as usize..end].copy_from_slice(data); + } + + /// Differential against a plain map of what each accepted write made + /// each file: three files grow in turn past twice the runs one entry + /// names at a block a run, through overwrites, holes, a shrink and writes + /// the volume refuses, and every byte the map holds is what the remounted + /// volume reads. #[test] - fn two_files_written_in_turn_keep_every_accepted_write() { + fn every_file_reads_back_as_a_plain_map_of_its_accepted_writes() { + const PAST: usize = (2 * 250 + 20) * BLOCK; + let paths = ["home/a", "home/b", "home/c"]; let mut v = vol(); - let a = v.open("home/a", CREATE).unwrap(); - let b = v.open("home/b", CREATE).unwrap(); - for page in 0..300u64 { - v.write(a, page * BLOCK as u64, &[1; BLOCK]).unwrap(); - v.write(b, page * BLOCK as u64, &[2; BLOCK]).unwrap(); + let nodes: Vec = paths.iter().map(|p| v.open(p, CREATE).unwrap()).collect(); + let mut model: HashMap<&str, Vec> = paths.iter().map(|p| (*p, Vec::new())).collect(); + let mut state = 0x2545_F491_4F6C_DD1Du64; + let mut next = move || { + state ^= state << 13; + state ^= state >> 7; + state ^= state << 17; + state + }; + let mut round = 0u64; + while paths.iter().any(|p| model[p].len() < PAST) { + for (i, (&path, &node)) in paths.iter().zip(&nodes).enumerate() { + let held = model.get_mut(path).unwrap(); + let len = held.len() as u64; + let (offset, n) = match (round + i as u64) % 11 { + 3 if len > 0 => (next() % len, (next() % (3 * BLOCK as u64)) as usize + 1), + 7 => (len + next() % (2 * BLOCK as u64), BLOCK), + _ => (len, BLOCK + (next() % 64) as usize), + }; + let data: Vec = (0..n).map(|_| next() as u8).collect(); + assert_eq!(v.write(node, offset, &data), Ok(()), "{path}: {n} bytes at {offset}, of {len} held"); + write_into(held, offset, &data); + } + if round % 97 == 50 { + let at = round as usize % paths.len(); + assert_eq!( + v.write(nodes[at], 64 << 20, b"past every block"), + Err(SyscallError::ResourceExhausted), + "{}: a write the volume has no blocks for", + paths[at] + ); + } + if round == 300 { + let b = model.get_mut("home/b").unwrap(); + let to = b.len() / 3 + 17; + v.truncate(nodes[1], to as u64).unwrap(); + b.truncate(to); + } + round += 1; } - let c = v.open("home/c", CREATE).unwrap(); - v.write(c, 0, b"c").unwrap(); assert_eq!(v.sync(), Ok(Vec::new())); - for n in [a, b, c] { + for n in nodes { v.close(n).unwrap(); } let mut v = remount(v); - assert_eq!(page_count(&mut v, "home/a"), 300); - assert_eq!(page_count(&mut v, "home/b"), 300); + let listed: Vec = v.list("home").unwrap().into_iter().map(|(n, _)| format!("home/{n}")).collect(); + assert_eq!(listed, paths); + for path in paths { + let held = &model[path]; + assert_eq!(v.lstat(path).unwrap().size, held.len() as u64, "{path}'s length"); + let n = v.open(path, PLAIN).unwrap(); + let mut out = vec![0xEEu8; held.len()]; + assert_eq!(v.read(n, 0, &mut Buf(&mut out)), Ok(held.len())); + let first = out.iter().zip(held).position(|(a, b)| a != b); + assert_eq!(first, None, "{path}: the first byte the volume and the map disagree on"); + } } /// On a volume whose free blocks are all apart, a file's entry fills: the @@ -911,7 +973,9 @@ mod tests { // what the one entry holding `home/a` names. assert_eq!(pages, 250, "the entry filled, not the volume"); assert_eq!(page_count(&mut v, "home/a"), pages, "the refused write changed nothing"); - assert_eq!(v.sync(), Ok(Vec::new())); + let before = free_blocks(&mut v); + assert_eq!(v.write(a, pages * BLOCK as u64, &[4; 4 * BLOCK]), Err(SyscallError::ResourceExhausted)); + assert_eq!(free_blocks(&mut v), before, "a refused write's blocks went back"); v.close(a).unwrap(); let mut v = remount(v); assert_eq!(page_count(&mut v, "home/a"), pages); @@ -938,6 +1002,8 @@ mod tests { assert_eq!(v.sync(), Ok(vec![(a, SyscallError::ResourceExhausted)])); v.close(b).unwrap(); assert_eq!(v.close(a), Err(SyscallError::ResourceExhausted), "a close that lost the writes is refused"); + assert_eq!(v.sync(), Ok(vec![(a, SyscallError::ResourceExhausted)]), "unheld, it is named again"); + assert!(v.open.contains_key(&a), "and kept again"); assert_eq!(page_count(&mut v, "home/a"), 1, "the refused file is still what its writes made it"); v.open.get_mut(&a).unwrap().extents = real; diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index a557a37c1c6..f2d3bd28030 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -424,16 +424,25 @@ impl Volume for FatVolume { } } +/// A FAT32 volume built from fatgen103 by the checker's tests, and not by the +/// driver under test. +#[cfg(test)] +#[path = "../../../toyos-fat32-check/tests/common/mod.rs"] +mod spec_volume; + #[cfg(test)] mod tests { + use std::cell::Cell; + use super::*; + use crate::cache::CLEAN_LIMIT; use crate::disk::{DiskError, Ram}; /// A disk that refuses every read of one block, after the fixture's own /// bytes are on it. struct Refusing { ram: Ram, - refused: u64, + refused: Rc>, } impl Disk for Refusing { @@ -442,7 +451,7 @@ mod tests { } fn read(&mut self, first: u64, out: &mut [u8]) -> Result<(), DiskError> { let count = (out.len() / BLOCK) as u64; - if (first..first + count).contains(&self.refused) { + if (first..first + count).contains(&self.refused.get()) { return Err(DiskError::Device); } self.ram.read(first, out) @@ -459,7 +468,7 @@ mod tests { fn bytes() -> Bytes { let mut ram = Ram::new(4); ram.write(0, &[0x5A; 4 * BLOCK]).unwrap(); - let cache = Rc::new(Cache::new(Refusing { ram, refused: 1 })); + let cache = Rc::new(Cache::new(Refusing { ram, refused: Rc::new(Cell::new(1)) })); Bytes { cache, len: 4 * BLOCK as u64 } } @@ -486,12 +495,70 @@ mod tests { b.flush().unwrap(); let cache = Rc::try_unwrap(b.cache).ok().expect("the one holder"); let mut disk = cache.into_disk(); - disk.refused = u64::MAX; + disk.refused.set(u64::MAX); let mut block = [0u8; BLOCK]; disk.read(1, &mut block).unwrap(); assert!(block.iter().all(|&x| x == 0x5A), "the unreadable block was written over"); } + /// The specification's fixture volume, writable, on a disk with room past + /// it for [`evict`] to read, and the handle that moves its refused block. + fn spec_fat() -> (FatVolume, Rc>, spec_volume::Volume) { + let spec = spec_volume::fixture(); + let blocks = spec.bytes.len().div_ceil(BLOCK); + let mut image = spec.bytes.clone(); + image.resize(blocks * BLOCK, 0); + let mut ram = Ram::new((blocks + 2 * CLEAN_LIMIT) as u64); + ram.write(0, &image).unwrap(); + let refused = Rc::new(Cell::new(u64::MAX)); + let v = FatVolume::mount(Refusing { ram, refused: Rc::clone(&refused) }, true, || 1_717_245_296).unwrap(); + (v, refused, spec) + } + + /// Every clean block out of the cache, so the next read of one asks the disk. + fn evict(v: &FatVolume) { + let mut block = [0u8; BLOCK]; + for b in v.cache.blocks() - 2 * CLEAN_LIMIT as u64..v.cache.blocks() { + v.cache.read_block(b, &mut block).unwrap(); + } + } + + /// One file whose entry will not read back leaves every other file's sync + /// whole, its own close refused and its node kept, and the next sync that + /// reaches it brings it level. + #[test] + fn a_file_that_will_not_level_costs_only_its_own_file() { + const CREATE: OpenHow = OpenHow { create: true, create_new: false, truncate: false }; + let (mut v, refused, spec) = spec_fat(); + let a = v.open("a.txt", CREATE).unwrap(); + let b = v.open("sub/b.txt", CREATE).unwrap(); + for n in [a, b] { + v.write(n, 0, &[1; 3000]).unwrap(); + } + assert_eq!(v.sync(), Ok(Vec::new())); + for n in [a, b] { + v.write(n, 3000, &[2; 3000]).unwrap(); + } + let root = (spec_volume::cluster_offset(spec_volume::ROOT_CLUSTER) / BLOCK) as u64; + let sub = (spec_volume::cluster_offset(spec.at("sub").first) / BLOCK) as u64; + assert_ne!(root, sub, "`a.txt`'s entry and `sub/b.txt`'s are in different blocks"); + evict(&v); + refused.set(root); + + assert_eq!(v.sync(), Ok(vec![(a, SyscallError::Io)]), "only the file that did not level is named"); + assert!(!v.open[&b].file.needs_reconcile(), "the other file was brought level"); + assert!(v.open[&a].file.needs_reconcile()); + assert_eq!(v.close(a), Err(SyscallError::Io), "a close that left the entry behind is refused"); + assert!(v.open.contains_key(&a), "and its node kept"); + + refused.set(u64::MAX); + assert_eq!(v.sync(), Ok(Vec::new())); + assert!(!v.open.contains_key(&a), "levelled, the unheld node goes"); + assert_eq!(v.lstat("a.txt").unwrap().size, 6000); + v.close(b).unwrap(); + assert_eq!(v.lstat("sub/b.txt").unwrap().size, 6000); + } + /// What the driver says of a volume that stopped answering is `Io`, which /// no caller takes for a name that is not there. #[test] diff --git a/userland/fsd/src/lib.rs b/userland/fsd/src/lib.rs index afa56da0b48..ee7767b426d 100644 --- a/userland/fsd/src/lib.rs +++ b/userland/fsd/src/lib.rs @@ -19,3 +19,4 @@ pub mod disk; pub mod fat; pub mod resolve; pub mod volume; +pub mod writeback; diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index b732b389e8e..a168a6885ea 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -20,7 +20,7 @@ //! what is acted on is that copy. //! //! **What is written reaches the disk at a sync**: an fsync, a client's -//! `SYNC`, or [`WRITEBACK`] after the first unsynced write when nothing asked +//! `SYNC`, or [`fsd::writeback::WRITEBACK`] after the first unsynced write when nothing asked //! sooner. A kill loses what no sync covered, which is POSIX's promise and no //! more. @@ -33,6 +33,7 @@ use fsd::disk::{Claimed, Disk, Ram, Served}; use fsd::fat::FatVolume; use fsd::resolve::{self, Found, Refusal as Escape, Resolved}; use fsd::volume::{Kind, Meta, Node, OpenHow, Out, Volume}; +use fsd::writeback::WriteBack; use toyos::endow::{self, Endowments}; use toyos::fs::*; use toyos::ipc::{self, Connection, RxStep}; @@ -43,10 +44,6 @@ use toyos::volatile::Window; use toyos::Pipe; use toyos_abi::syscall::{SyscallError, DEV_PREFIX, SERVE_PREFIX}; -/// How long a write waits for a sync nobody asked for. Policy: the kernel's -/// write-back drained a closed file's pages on its next pass. -const WRITEBACK: Duration = Duration::from_secs(2); - /// Clients served at once, machine-wide: one connection per directory per /// process. The next is answered `ResourceExhausted` at its hello and let go, /// by name. @@ -226,7 +223,7 @@ fn main() { next_client: 0, streams: BTreeMap::new(), next_stream: 0, - dirty_since: None, + writeback: WriteBack::default(), end_on, end_at_read, probe: Poller::new(caps_len), @@ -371,8 +368,8 @@ struct Server { next_client: u64, streams: BTreeMap, next_stream: u64, - /// When a write first went unsynced. - dirty_since: Option, + /// When the sync nobody asked for is due. + writeback: WriteBack, /// [`END_ON`]'s path on the volume. end_on: Option, /// [`END_AT_READ`]'s paths on the volume. @@ -481,7 +478,7 @@ impl Server { .values() .filter(|c| c.window.is_none()) .map(|c| HANDSHAKE_TIMEOUT.saturating_sub(now.duration_since(c.since))) - .chain(self.dirty_since.map(|at| WRITEBACK.saturating_sub(now.duration_since(at)))) + .chain(self.writeback.left(now)) .min() .map_or(u64::MAX, |left| left.as_nanos() as u64); ready.clear(); @@ -506,8 +503,7 @@ impl Server { for id in late { self.drop_client(id, "it never lent its window"); } - if self.dirty_since.is_some_and(|at| now.duration_since(at) >= WRITEBACK) { - self.dirty_since = None; + if self.writeback.take_due(now) { if let Err(e) = self.sync(None) { println!("fsd: the write-back sync failed: {e:?}"); } @@ -677,7 +673,7 @@ impl Server { /// Mark the volume written: the write-back sync is due from here. fn dirtied(&mut self) { - self.dirty_since.get_or_insert_with(Instant::now); + self.writeback.dirtied(Instant::now()); } /// `rel` resolved without following its last component, for an operation @@ -920,7 +916,7 @@ impl Server { /// due, so the next one tries it again. fn sync(&mut self, node: Option) -> Result { let unwritten = self.volume.sync()?; - self.dirty_since = (!unwritten.is_empty()).then(Instant::now); + self.writeback.synced(!unwritten.is_empty(), Instant::now()); match unwritten.into_iter().find(|&(n, _)| node.is_none_or(|node| n == node)) { Some((_, e)) => Err(e), None => Ok(Answer::Reply(Reply::ok())), diff --git a/userland/fsd/src/writeback.rs b/userland/fsd/src/writeback.rs new file mode 100644 index 00000000000..b15d691973a --- /dev/null +++ b/userland/fsd/src/writeback.rs @@ -0,0 +1,64 @@ +//! When the sync nobody asked for is due: [`WRITEBACK`] after the first write +//! no sync has answered for, and again after a sync that left a file unwritten. + +use std::time::{Duration, Instant}; + +/// How long a write waits for a sync nobody asked for. Policy: the kernel's +/// write-back drained a closed file's pages on its next pass. +pub const WRITEBACK: Duration = Duration::from_secs(2); + +#[derive(Default)] +pub struct WriteBack { + since: Option, +} + +impl WriteBack { + /// The volume was written at `now`. + pub fn dirtied(&mut self, now: Instant) { + self.since.get_or_insert(now); + } + + /// How long until it is due, while one is owed. + pub fn left(&self, now: Instant) -> Option { + self.since.map(|at| WRITEBACK.saturating_sub(now.duration_since(at))) + } + + /// Whether it is due at `now`; one that is, is spent. + pub fn take_due(&mut self, now: Instant) -> bool { + let due = self.since.is_some_and(|at| now.duration_since(at) >= WRITEBACK); + if due { + self.since = None; + } + due + } + + /// A sync answered at `now`: one that left a file unwritten keeps the + /// write-back due, so the next one tries it again. + pub fn synced(&mut self, unwritten: bool, now: Instant) { + self.since = unwritten.then_some(now); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_sync_that_left_a_file_unwritten_is_due_again() { + let t0 = Instant::now(); + let mut wb = WriteBack::default(); + assert_eq!(wb.left(t0), None, "nothing written, nothing owed"); + wb.dirtied(t0); + wb.dirtied(t0 + WRITEBACK / 2); + assert!(!wb.take_due(t0 + WRITEBACK / 2), "due from the first write, not the last"); + assert!(wb.take_due(t0 + WRITEBACK)); + assert_eq!(wb.left(t0 + WRITEBACK), None, "spent"); + + let t1 = t0 + 2 * WRITEBACK; + wb.synced(true, t1); + assert_eq!(wb.left(t1), Some(WRITEBACK)); + assert!(wb.take_due(t1 + WRITEBACK), "tried again"); + wb.synced(false, t1 + WRITEBACK); + assert_eq!(wb.left(t1 + WRITEBACK), None, "every file written"); + } +} diff --git a/userland/logd/src/main.rs b/userland/logd/src/main.rs index 91c89c7ad15..d5eda3ef45f 100644 --- a/userland/logd/src/main.rs +++ b/userland/logd/src/main.rs @@ -98,7 +98,7 @@ use toyos_logstream::{ use toyos_wallclock::Civil; use origin::{Origin, Said}; -use policy::{fate, Fate, Step, LOG_WRITE_BUDGET}; +use policy::{Step, LOG_WRITE_BUDGET}; use store::{Volume, DIR, MAX_LOG_BYTES, MAX_LOG_FILES, ROTATE_FAST_BYTES}; /// Records asked of `SYS_LOG_READ` at once: above `MAX_LOG_SHARDS`, which the @@ -187,7 +187,6 @@ fn main() { lost: 0, waiting: Vec::new(), volume, - owed: false, degraded: false, boot_local, hub, @@ -215,8 +214,6 @@ struct Log { /// round merges. waiting: Vec, volume: Option, - /// Lines the volume took and has not made durable. - owed: bool, /// Whether the volume answers, slower than `LOG_WRITE_BUDGET` a round. degraded: bool, boot_local: Option, @@ -298,7 +295,6 @@ impl Log { self.flushed(); read_any = true; } - self.sync_owed(); self.feed_console(); cadence = if read_any { POLL_QUICK } else { (cadence * 2).min(POLL_SLOW) }; @@ -306,8 +302,7 @@ impl Log { continue; } // Nothing new: park until the kernel posts, init speaks, a program - // ends, the console has room, or the cadence comes round, which is - // also when an owed sync is asked again. + // ends, the console has room, or the cadence comes round. poller.wait(1, cadence.as_nanos() as u64, |token| { if token >= ORIGIN_BASE { ended.push((token - ORIGIN_BASE) as usize); @@ -624,11 +619,8 @@ impl Log { let Some(v) = self.volume.as_mut() else { return }; let began = Instant::now(); let mut refused = v.write(text).err().map(|e| (Step::Append, e.to_string())); - if refused.is_none() { - self.owed = true; - if sync { - refused = self.sync().err(); - } + if refused.is_none() && sync { + refused = self.sync().err(); } // A volume that answered, and took longer than a log is worth doing it. if refused.is_none() && began.elapsed() > LOG_WRITE_BUDGET { @@ -637,19 +629,10 @@ impl Log { self.answered(refused); } - fn sync_owed(&mut self) { - if self.owed { - let refused = self.sync().err(); - self.answered(refused); - } - } - /// Make the volume durable now. fn sync(&mut self) -> Result<(), (Step, String)> { let Some(v) = self.volume.as_mut() else { return Ok(()) }; - v.sync().map_err(|e| (Step::Flush, e.to_string()))?; - self.owed = false; - Ok(()) + v.sync().map_err(|e| (Step::Flush, e.to_string())) } /// What a round's write came to: rotation after a clean one, and the @@ -664,9 +647,9 @@ impl Log { self.rotate_if_full(); return; }; - match fate(step) { + match step { // Stop feeding the volume, say so once, and keep running. - Fate::GiveUp => { + Step::Append | Step::Flush => { toyos::error!( "logd: {DIR} has not answered ({}: {why}) - this boot's log is on the console \ only from {path}", @@ -675,8 +658,7 @@ impl Log { self.volume = None; } // Every call answered, slowly: the round is durable. - Fate::Degraded => { - self.owed = false; + Step::TooSlow => { if !self.degraded { self.degraded = true; toyos::warn!( diff --git a/userland/logd/src/policy.rs b/userland/logd/src/policy.rs index d99c37ca53a..b957fc2167f 100644 --- a/userland/logd/src/policy.rs +++ b/userland/logd/src/policy.rs @@ -1,11 +1,3 @@ -//! What a refused batch does to this boot's log volume. -//! -//! **One pure function, because this is the whole refused-batch policy and it used to be -//! `if let Err(_) = … { volume = None }`.** Everything above [`fate`] is I/O -//! and everything below it is what the console says; the decision itself is a -//! function of what a host test can hand it, which is why this file exists -//! rather than the two lines it replaces living in the loop. - use std::time::Duration; /// The round time past which a volume that answered every call is announced @@ -19,10 +11,6 @@ use std::time::Duration; pub const LOG_WRITE_BUDGET: Duration = Duration::from_secs(5); /// Which call refused. -/// -/// Named rather than a `&str`, because [`fate`] matches on it: the append and -/// the flush have different answers and a string cannot be matched -/// exhaustively. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Step { /// One of the batch's lines did not reach the file. @@ -45,51 +33,10 @@ impl Step { } } -/// What happens to this boot's volume. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum Fate { - /// Keep it, publish this batch — every call answered, so the records are - /// durable — and say once that the volume is slow. The state the whole - /// slow-vs-failed split exists for: degraded is not dead, and a log that - /// arrives late beats a log that was thrown away. - Degraded, - /// Stop feeding the volume for the rest of the boot. The log is on the - /// console only from here. - GiveUp, -} - -/// The decision, from the call that refused. -pub fn fate(step: Step) -> Fate { - match step { - // Every call answered: the records are durable, however long the round - // took, and a volume is never ended on elapsed time. - Step::TooSlow => Fate::Degraded, - Step::Append | Step::Flush => Fate::GiveUp, - } -} - #[cfg(test)] mod tests { use super::*; - /// A stick that cannot flush ends the boot's log: the writes are not - /// durable and no retry makes them so. A refused append is a hole in the - /// file: the records have left the cursor and there is nothing to - /// re-write. - #[test] - fn a_refused_append_or_flush_ends_the_volume() { - assert_eq!(fate(Step::Flush), Fate::GiveUp); - assert_eq!(fate(Step::Append), Fate::GiveUp); - } - - /// A volume that answered every call and took too long over it is - /// degraded, never dead: the records are durable and elapsed time is not - /// one of the evidences a volume may be declared failed on. - #[test] - fn a_slow_round_degrades_and_keeps_the_volume() { - assert_eq!(fate(Step::TooSlow), Fate::Degraded); - } - /// The console words, which the table in `main` is written against. #[test] fn every_step_names_itself() { From 904bd5c61b4c2274645122a719f506bb7c89d610 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 00:36:04 +0200 Subject: [PATCH 35/54] storage: one inventory reader, a bounded dirty cache, and an image's libraries asked of /system/lib - toyos-inventory: the kernel's device inventory read whole. A refused count, a refused read and a record that does not decode are each the answer, never a shorter list. init's storage start and slot grant both read through it, so blockd is never started without `--running` because the inventory was refused: the start fails by name. - fsd's cache: a write that would take it past DIRTY_LIMIT flushes first and is refused whole when that flush is, so a disk that refuses leaves the cache no larger. - FAT: the next failing sync keeps a node whose close was refused. - abuse_elf_loader: the three cases whose DT_NEEDED library is written beside the file are refused on the path route; on the image route they answer NotFound, because an image's libraries come from /system/lib alone and the kernel never reaches the check the path route refuses. - SharedImage::over: the len == 0 test goes; its one caller maps 0 to None. - The harness: the userland NVMe no longer needs a kernel one beside it, so the assert, blockd's first-controller disk, KBENCH and craft_plain_disk go; IN_MEMORY is used where its literal was repeated; prose about the kernel's NVMe driver and format is deleted. - The cached-read issue names its owner. Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 7 ++ Cargo.toml | 1 + ...-file-server-is-measured-only-under-tcg.md | 2 + kernel/src/file_backing.rs | 2 +- src/image.rs | 2 +- tests/common/blockd.rs | 32 +----- tests/common/partclaim.rs | 18 ---- tests/common/pkg.rs | 4 +- tests/common/qemu.rs | 10 +- tests/common/storage.rs | 6 +- .../src/bin/abuse_elf_loader.rs | 15 ++- tests/toyos.rs | 5 +- toyos-inventory/Cargo.toml | 15 +++ toyos-inventory/src/lib.rs | 93 ++++++++++++++++ userland/Cargo.lock | 8 ++ userland/fsd/src/cache.rs | 101 +++++++++++++----- userland/fsd/src/fat.rs | 2 + userland/init/Cargo.toml | 2 + userland/init/src/main.rs | 33 +++--- 19 files changed, 245 insertions(+), 113 deletions(-) create mode 100644 toyos-inventory/Cargo.toml create mode 100644 toyos-inventory/src/lib.rs diff --git a/Cargo.lock b/Cargo.lock index 3d51a3aa6a3..c00dce40b84 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1309,6 +1309,13 @@ dependencies = [ "toyos-abi", ] +[[package]] +name = "toyos-inventory" +version = "0.1.0" +dependencies = [ + "toyos-abi", +] + [[package]] name = "toyos-keymap" version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index e7de9c08ab5..b15d9c4fbef 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -35,6 +35,7 @@ members = [ "toyos-hda", "toyos-i219", "toyos-inspect", + "toyos-inventory", "toyos-keymap", "toyos-ld", "toyos-libc-copies", diff --git a/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md b/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md index e8c4e781a39..6ce7166feea 100644 --- a/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md +++ b/issues/filesystem/a-cached-read-from-a-file-server-is-measured-only-under-tcg.md @@ -28,3 +28,5 @@ nothing says whether metal pays it. **Exit**: the same interleaved runs on the T14, with the copy out of a shared mapping timed beside a heap-to-heap one; a cost metal also pays is a defect filed against the mapping, one it does not is closed here. + +Owner: the orchestrator, which holds the T14 the exit runs on. diff --git a/kernel/src/file_backing.rs b/kernel/src/file_backing.rs index 39a3f1bb285..ce46f3b8d05 100644 --- a/kernel/src/file_backing.rs +++ b/kernel/src/file_backing.rs @@ -102,7 +102,7 @@ impl SharedImage { /// memory the kernel allocated — a device aperture is no program, and a /// read of one is a device access — and the object holds them all. pub fn over(object: Arc, len: u64) -> Result { - if len == 0 || len > object.size() || object.ram().is_none() { + if len > object.size() || object.ram().is_none() { return Err(SyscallError::InvalidArgument); } Ok(Self { object, size: len }) diff --git a/src/image.rs b/src/image.rs index 17eef5eb021..b17abed3d9a 100644 --- a/src/image.rs +++ b/src/image.rs @@ -918,7 +918,7 @@ pub fn misaligned_data_disk(path: &Path, len: u64) -> u64 { start } -/// Block 0 of a volume the kernel may format: the magic and its block count. +/// Block 0 of a volume: the magic and its block count. fn designation(blocks: u64) -> [u8; SECTOR] { let mut block = [0u8; SECTOR]; block[..bcachefs::DESIGNATION_MAGIC.len()].copy_from_slice(&bcachefs::DESIGNATION_MAGIC); diff --git a/tests/common/blockd.rs b/tests/common/blockd.rs index a019d341567..bf60bb386d8 100644 --- a/tests/common/blockd.rs +++ b/tests/common/blockd.rs @@ -10,9 +10,9 @@ //! command, completion and flush, which no driver can print on the device's //! behalf. //! -//! The machine has two NVMe controllers: the first (QEMU's own ids) with a DATA -//! and a partition nothing names, which no driver runs on this boot, and -//! blockd's (Intel's ids) with the partitions below. +//! The machine has two NVMe controllers: the first (QEMU's own ids), which no +//! driver runs on this boot, and blockd's (Intel's ids) with the partitions +//! below. use std::collections::{BTreeMap, BTreeSet}; use std::io::{Read, Seek, SeekFrom, Write}; @@ -29,9 +29,6 @@ const FS: &str = "B2D4F6A8-1C3E-4A57-9B0D-E2F4A6C8E0A1"; const BENCH: &str = "C3E5A7B9-2D4F-4B68-8C1E-F3A5B7D9F1B2"; const MISALIGNED: &str = "E5A7C9DB-4F6B-4D8A-8E30-B5C7D9FB13D4"; const MISSTART: &str = "F6B8DAEC-5A7C-4E9B-9F41-C6D8EA0C24E5"; -/// The first controller's disk: the system's own DATA, which init's blockd -/// serves, beside a partition nothing names. -const KBENCH: &str = "D4F6B8CA-3E5A-4C79-9D2F-A4B6C8EA02C3"; const TARGET_BLOCKS: u64 = 2048; const BENCH_BLOCKS: u64 = 8192; const FILES: usize = 6; @@ -77,10 +74,6 @@ struct Layout { /// blockd's disk: a FAT32 neighbour, the idle ROOT slot, a FAT32 neighbour /// touching it, the FAT32 volume the crash role writes, the bench partition, /// and two partitions that are not whole 4 KiB blocks. -/// -/// The neighbours are long enough that the volume starts past every byte of -/// the first controller's disk: QEMU's trace names no controller, so a write -/// is blockd's by the sector it lands on ([`reissued_after`]). fn craft_blockd_disk(path: &Path) -> Result { const NEIGHBOUR_BYTES: u64 = 64 * MIB; const FS_BYTES: u64 = 64 * MIB; @@ -116,19 +109,9 @@ fn boot( params: &'static [&'static str], ) -> Result<(QemuInstance, Layout, PathBuf, PathBuf, Vec), String> { let config = super::compile::repo_root().join(CONFIG); - let first_disk = super::lane::dir().join(format!("{name}-first.img")); - partclaim::craft_plain_disk(&first_disk, &[("unnamed", BENCH_BLOCKS * BLOCK, KBENCH)], 96 * MIB)?; let blockd_disk = super::lane::dir().join(format!("{name}-blockd.img")); let layout = craft_blockd_disk(&blockd_disk)?; - let first_bytes = std::fs::metadata(&first_disk).map_err(|e| format!("the first disk: {e}"))?.len(); - if first_bytes > layout.fs.start { - return Err(format!( - "the first controller's disk is {first_bytes} bytes and blockd's volume starts at {}: a traced write \ - there could be either controller's", - layout.fs.start - )); - } - let before = std::fs::read(&blockd_disk).map_err(|e| format!("read the crafted disk: {e}"))?; + let before =std::fs::read(&blockd_disk).map_err(|e| format!("read the crafted disk: {e}"))?; let trace = super::lane::dir().join(format!("{name}-nvme.trace")); let _ = std::fs::remove_file(&trace); let qemu = QemuInstance::boot_with_options( @@ -136,7 +119,6 @@ fn boot( c_bins, rust_bins, BootOptions { - nvme_image: Some(first_disk), userland_nvme: Some(blockd_disk.clone()), nvme_trace: Some(trace.clone()), kernel_params: params, @@ -256,11 +238,7 @@ fn trace_events(trace: &Path) -> Result, String> { /// controller was reset or it was killed, as the device saw them, each written /// again first thing after — read off QEMU's trace alone. /// -/// **A write is blockd's by where it lands**: the trace names no controller, -/// and every partition blockd writes starts past the last byte of the first -/// controller's disk ([`boot`] refuses a layout where the volume does not, and -/// the bench partition lies past it). A lifetime is what follows one -/// controller start — +/// A lifetime is what follows one controller start — /// blockd's bring-up, or its reset. The one the loss ended is the lifetime /// whose writes to `span` did not end in a flush and after which another /// lifetime wrote to it; its last write is the one blockd withheld the answer diff --git a/tests/common/partclaim.rs b/tests/common/partclaim.rs index 044e0988499..6f3f2eb8d18 100644 --- a/tests/common/partclaim.rs +++ b/tests/common/partclaim.rs @@ -744,24 +744,6 @@ fn designate(device: &mut dyn gpt::DiskDevice, data: Span) -> Result<(), String> device.write_all(&stamp).map_err(|e| format!("stamp DATA: {e}")) } -/// A disk at `path` carrying `parts` — each a name, a length in bytes and its -/// unique GUID, on its own MiB boundary — and after them a DATA of -/// `data_bytes` the kernel may format, so `/home` mounts from this disk. -pub(super) fn craft_plain_disk( - path: &Path, - parts: &[(&'static str, u64, &'static str)], - data_bytes: u64, -) -> Result<(), String> { - const DATA_GUID: &str = "C1A2B3D4-E5F6-4718-9A0B-1C2D3E4F5A6B"; - let mut all: Vec = - parts.iter().map(|&(name, len, unique)| (name, len, PLAIN_TYPE, unique, ALIGNED)).collect(); - all.push(("ToyOS data", data_bytes, toyos_gpt::Guid::TOYOS_DATA_TEXT, DATA_GUID, ALIGNED)); - let total = MIB + all.iter().map(|p| p.1.next_multiple_of(MIB)).sum::() + 2 * MIB; - let (mut device, spans) = table(path, total, &all)?; - designate(&mut *device, *spans.last().expect("DATA was just added"))?; - device.flush().map_err(|e| format!("flush the disk: {e}")) -} - /// A USB stick of `bytes` carrying `parts`, each a name, a length and its /// unique GUID; their spans. pub(super) fn craft_stick( diff --git a/tests/common/pkg.rs b/tests/common/pkg.rs index 6428eed0304..f49c9b2697a 100644 --- a/tests/common/pkg.rs +++ b/tests/common/pkg.rs @@ -19,7 +19,7 @@ use std::path::Path; use std::time::Duration; use super::qemu::{self, BootOptions, QemuInstance}; -use super::storage::{superblock_at, FileBlocks}; +use super::storage::{superblock_at, FileBlocks, IN_MEMORY}; /// gbae v0.2.0's release archive and the sums file published beside it, both /// committed under `tests/fixtures` and named in `NOTICE`. @@ -78,7 +78,7 @@ pub fn pkg_install_gbae( }; let mut qemu = QemuInstance::boot_with_options(&config, &[], &bins, options); let boot = qemu.boot_log().to_string(); - if boot.contains("are in memory and will not survive a reboot") { + if boot.contains(IN_MEMORY) { return Err(format!( "/apps and /home fell back to memory, so the readback below would judge no device:\n\ {boot}" diff --git a/tests/common/qemu.rs b/tests/common/qemu.rs index 5b84a956fd9..a9b87706e9d 100644 --- a/tests/common/qemu.rs +++ b/tests/common/qemu.rs @@ -2447,13 +2447,11 @@ pub struct BootOptions { /// A second NVMe controller, for a driver in userland, backed by this file. /// /// QEMU's NVMe under Intel's ids (`use-intel-id`, `8086:5845`), so a claim - /// names it apart from the one the kernel drives; its MSI-X table in a BAR + /// names it; its MSI-X table in a BAR /// of its own (`msix-exclusive-bar`), because a claim never maps the BAR /// holding the table and NVMe keeps its registers in BAR 0; and its /// namespace's write cache on, so the controller has a volatile cache a - /// flush has to issue Flush for. Emitted after the kernel's controller, so - /// the kernel's first-by-class probe takes that one, and refused on a - /// profile with none — the kernel would take this one. + /// flush has to issue Flush for. pub userland_nvme: Option, /// Have QEMU record every NVMe command it is sent, every completion it /// posts, every write with its sectors, every flush it runs and every @@ -4613,10 +4611,6 @@ fn qemu_command( )); } if let Some(image) = &options.userland_nvme { - assert!( - shape.nvme_bytes != 0, - "a userland NVMe on a machine whose kernel drives none is the one the kernel takes" - ); qemu.arg("-drive") .arg(format!("if=none,id=nvme1,format=raw,file={}", image.display())) .arg("-device") diff --git a/tests/common/storage.rs b/tests/common/storage.rs index e02d3aec9df..a2ef3ff1469 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -24,7 +24,7 @@ const MOUNTED: &str = "fsd: mounted the DATA volume"; /// fsd's word for a volume of ours that did not, followed by the reason. const UNMOUNTABLE: &str = "fsd: the DATA volume is ours and does not mount ("; /// fsd's word for DATA's directories served from memory. -const IN_MEMORY: &str = "are in memory and will not survive a reboot"; +pub(super) const IN_MEMORY: &str = "are in memory and will not survive a reboot"; /// Whether fsd said it serves DATA's directories as absent: every name under /// them refused, and never a volume in memory under the paths an owner's data @@ -532,7 +532,7 @@ pub fn home_overwrite_reads_back( BootOptions { profile: qemu::Profile::MetalDisk, ..Default::default() }, ); let boot = qemu.boot_log().to_string(); - if boot.contains("are in memory and will not survive a reboot") { + if boot.contains(IN_MEMORY) { return Err(format!( "/apps and /home fell back to memory, so nothing below touches the NVMe path:\n{boot}" )); @@ -778,7 +778,7 @@ pub fn apps_and_home_are_one_filesystem( BootOptions { profile: qemu::Profile::MetalDisk, ..Default::default() }, ); let boot = qemu.boot_log().to_string(); - if boot.contains("are in memory and will not survive a reboot") { + if boot.contains(IN_MEMORY) { return Err(format!( "/apps and /home fell back to memory, so the readback below would judge no device:\n\ {boot}" diff --git a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs index d1a9fc78a36..32ef907b096 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs @@ -307,6 +307,15 @@ fn spawn_refused(name: &str, bytes: &[u8]) { refused(&format!("{name} as an image"), spawn_image(&format!("{DIR}/{name}"), bytes)); } +/// Refused where the kernel opens the file, whose `DT_NEEDED` library is +/// written beside it; handed its bytes, it is `NotFound`, since an image's +/// libraries come from `/system/lib` alone. +fn spawn_refused_beside_its_library(name: &str, bytes: &[u8]) { + refused(name, spawn_result(name, bytes)); + let image = spawn_image(&format!("{DIR}/{name}"), bytes); + assert_eq!(image, Err(SyscallError::NotFound), "{name} as an image: its library is not in /system/lib"); +} + /// Load it and throw it away. These cases are about a *walk* the loader does, /// not about a header it can reject: a library whose `.gnu.hash` never /// terminates or whose TLS symbol resolves nowhere is still a library, and it @@ -751,7 +760,7 @@ fn values_are_bounded_by_the_image() { .sym(0x1418, 1, (STB_GLOBAL << 4) | STT_FUNC, 1, FAR_VALUE) .poke(0x1801, FAR.as_bytes()) .poke(0x1800 + name_at as usize, dep.as_bytes()); - spawn_refused("export_past_image", &exe.build()); + spawn_refused_beside_its_library("export_past_image", &exe.build()); } dlopen_refused("so_relative_addend_past_image.so", &so_with(&[], &[(0x1400, 0, R_X86_64_RELATIVE, i64::MAX)], None)); @@ -916,7 +925,7 @@ fn tls_apply_time_refusals_are_reached() { .sym(0x1418, 1, (STB_GLOBAL << 4) | STT_TLS, 0, 0) .poke(0x1801, b"wtls\0") .poke(0x1806, dep.as_bytes()); - spawn_refused("tpoff_overflow_spawn", &exe.build()); + spawn_refused_beside_its_library("tpoff_overflow_spawn", &exe.build()); let defs = write_file("tpoff_overflow_defs.so", &tls_defs_so(b"vtls", PAST_I64)); let lib_defs = unsafe { libloading::Library::new(&defs) }.expect("dlopen tpoff_overflow_defs.so"); @@ -943,7 +952,7 @@ fn globdat_past_short_dynsym() { ) .poke(0x1801, dep.as_bytes()) .poke(0x3000, &gnu_hash); - spawn_refused("globdat_past_dynsym", &exe.build()); + spawn_refused_beside_its_library("globdat_past_dynsym", &exe.build()); } /// A shared object defining its own `xtls` (`STT_TLS`, offset 8) in a 0x20-byte diff --git a/tests/toyos.rs b/tests/toyos.rs index 8607076bbbf..517ae3e0d41 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -1530,9 +1530,8 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // still mapped: the verdict is the first holder's own grant, read in the // guest after the second holder wrote its own. ("userdev_residue_is_its_own", Sched::Parallel, Tier::Fast), - // blockd, the NVMe driver in userland, on a second controller beside the - // kernel's: partitions served and timed against the kernel's driver; a - // controller reset and its own death, each survived by the client and + // blockd, the NVMe driver in userland, on a second controller: partitions + // served; a controller reset and its own death, each survived by the client and // judged off the image by the host's readers; and a transfer outside what // its function was lent, which is a fault record. Each boot runs several // blockd lifetimes and one waits out a ten-second silence. diff --git a/toyos-inventory/Cargo.toml b/toyos-inventory/Cargo.toml new file mode 100644 index 00000000000..7f2232e94ab --- /dev/null +++ b/toyos-inventory/Cargo.toml @@ -0,0 +1,15 @@ +# A member of the host workspace (root `Cargo.toml`): the kernel's device +# inventory read whole, which init reads and the host tests. + +[package] +name = "toyos-inventory" +version = "0.1.0" +edition = "2021" +license = "MIT OR Apache-2.0" +publish = false + +[dependencies] +toyos-abi = { path = "../toyos-abi" } + +[lints.rust] +warnings = "deny" diff --git a/toyos-inventory/src/lib.rs b/toyos-inventory/src/lib.rs new file mode 100644 index 00000000000..41064eb242f --- /dev/null +++ b/toyos-inventory/src/lib.rs @@ -0,0 +1,93 @@ +//! The kernel's device inventory, read whole: a count, then that many records, +//! each of which decodes. A refused count, a refused read and a record that +//! does not decode are each the answer, never a shorter list. + +#![cfg_attr(not(test), no_std)] +#![forbid(unsafe_code)] + +extern crate alloc; + +use alloc::vec; +use alloc::vec::Vec; + +use toyos_abi::inventory::{RawRecord, Record, Undecodable}; +use toyos_abi::syscall::SyscallError; + +/// Why the inventory was not read. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Unread { + Count(SyscallError), + Read(SyscallError), + Record { index: usize, why: Undecodable }, +} + +impl core::fmt::Display for Unread { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Count(e) => write!(f, "the inventory would not count: {e:?}"), + Self::Read(e) => write!(f, "the inventory would not read: {e:?}"), + Self::Record { index, why } => write!(f, "the inventory's record {index} does not decode: {why}"), + } + } +} + +/// Every record `ask` answers, where `ask` is the inventory call: an empty +/// buffer asks how many, and a buffer that long is filled. +pub fn read(mut ask: impl FnMut(&mut [RawRecord]) -> Result) -> Result, Unread> { + let asked = ask(&mut []).map_err(Unread::Count)?; + let mut raw = vec![RawRecord::EMPTY; asked]; + let n = ask(&mut raw).map_err(Unread::Read)?; + raw[..n] + .iter() + .enumerate() + .map(|(index, r)| Record::decode(r).map_err(|why| Unread::Record { index, why })) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + use toyos_abi::inventory::{Loaded, Role}; + + fn loaded(role: Role) -> Record { + Record::Loaded(Loaded { role, unique_guid: [7; 16] }) + } + + /// An inventory of `records`, as the kernel answers one. + fn kernel(records: Vec) -> impl FnMut(&mut [RawRecord]) -> Result { + move |buf| match buf.len() { + 0 => Ok(records.len()), + n if n < records.len() => Err(SyscallError::ResourceExhausted), + _ => { + buf[..records.len()].copy_from_slice(&records); + Ok(records.len()) + } + } + } + + #[test] + fn every_record_is_read() { + let records = [loaded(Role::Root), loaded(Role::Log)]; + assert_eq!(read(kernel(records.iter().map(Record::encode).collect())), Ok(records.to_vec())); + } + + #[test] + fn a_refused_count_is_refused_and_not_an_empty_inventory() { + assert_eq!(read(|_| Err(SyscallError::PermissionDenied)), Err(Unread::Count(SyscallError::PermissionDenied))); + } + + #[test] + fn a_refused_read_is_refused_and_not_an_empty_inventory() { + let refused = |buf: &mut [RawRecord]| match buf.len() { + 0 => Ok(1), + _ => Err(SyscallError::ResourceExhausted), + }; + assert_eq!(read(refused), Err(Unread::Read(SyscallError::ResourceExhausted))); + } + + #[test] + fn a_record_that_does_not_decode_is_refused_and_not_dropped() { + let records = vec![loaded(Role::Root).encode(), RawRecord::EMPTY, loaded(Role::Boot).encode()]; + assert_eq!(read(kernel(records)), Err(Unread::Record { index: 1, why: Undecodable::Kind(0) })); + } +} diff --git a/userland/Cargo.lock b/userland/Cargo.lock index f77dcdf6886..5d24b653269 100644 --- a/userland/Cargo.lock +++ b/userland/Cargo.lock @@ -1652,6 +1652,7 @@ dependencies = [ "toyos-abi", "toyos-blockring", "toyos-gpt", + "toyos-inventory", "toyos-logstream", "toyos-manifest", "toyos-quiesce", @@ -4126,6 +4127,13 @@ dependencies = [ "toyos-abi", ] +[[package]] +name = "toyos-inventory" +version = "0.1.0" +dependencies = [ + "toyos-abi", +] + [[package]] name = "toyos-keymap" version = "0.1.0" diff --git a/userland/fsd/src/cache.rs b/userland/fsd/src/cache.rs index a8b60c65a91..6280939a9f4 100644 --- a/userland/fsd/src/cache.rs +++ b/userland/fsd/src/cache.rs @@ -4,9 +4,9 @@ //! **A write lands here and reaches the disk at a flush.** [`Cache::flush`] //! writes every dirty block, lowest first and in runs, then asks the disk to //! make them durable; a volume's sync is its own metadata written into the -//! cache and then this. A cache holding more than [`DIRTY_LIMIT`] dirty blocks -//! flushes by itself, so memory bounds the write-back and not the other way -//! round. +//! cache and then this. A write that would take the cache past [`DIRTY_LIMIT`] +//! dirty blocks flushes first, and is refused whole when that flush is, so +//! memory bounds the write-back and not the other way round. //! //! **Clean blocks are kept up to [`CLEAN_LIMIT`]** and the oldest-read goes //! first; a dirty block is never dropped. @@ -137,34 +137,36 @@ impl Cache { /// next flush. pub fn write(&self, first: u64, data: &[u8]) -> Result<(), DiskError> { assert!(data.len() % BLOCK == 0, "fsd: a cache write of {} bytes", data.len()); - let over = { - let mut inner = self.inner.borrow_mut(); - if first.checked_add((data.len() / BLOCK) as u64).is_none_or(|end| end > inner.disk.blocks()) { - return Err(DiskError::Range); - } - for (k, chunk) in data.chunks_exact(BLOCK).enumerate() { - let block = first + k as u64; - let newly = match inner.slots.get_mut(&block) { - Some(slot) => { - slot.data.copy_from_slice(chunk); - !std::mem::replace(&mut slot.dirty, true) - } - None => { - let mut buf = Box::new([0u8; BLOCK]); - buf.copy_from_slice(chunk); - inner.slots.insert(block, Slot { data: buf, dirty: true }); - true - } - }; - if newly { - inner.dirty += 1; - } - } - inner.dirty > DIRTY_LIMIT + let (end, over) = { + let inner = self.inner.borrow(); + let end = match first.checked_add((data.len() / BLOCK) as u64) { + Some(end) if end <= inner.disk.blocks() => end, + _ => return Err(DiskError::Range), + }; + let newly = (first..end).filter(|b| !inner.slots.get(b).is_some_and(|s| s.dirty)).count(); + (end, inner.dirty + newly > DIRTY_LIMIT) }; if over { self.flush()?; } + let mut inner = self.inner.borrow_mut(); + for (block, chunk) in (first..end).zip(data.chunks_exact(BLOCK)) { + let newly = match inner.slots.get_mut(&block) { + Some(slot) => { + slot.data.copy_from_slice(chunk); + !std::mem::replace(&mut slot.dirty, true) + } + None => { + let mut buf = Box::new([0u8; BLOCK]); + buf.copy_from_slice(chunk); + inner.slots.insert(block, Slot { data: buf, dirty: true }); + true + } + }; + if newly { + inner.dirty += 1; + } + } Ok(()) } @@ -347,4 +349,49 @@ mod tests { c.read(0, &mut out).unwrap(); assert_eq!(out, vec![1; BLOCK]); } + + /// A disk that takes writes until it is told to refuse them. + struct Refusing { + ram: Ram, + refusing: bool, + } + + impl Disk for Refusing { + fn blocks(&self) -> u64 { + self.ram.blocks() + } + fn read(&mut self, first: u64, out: &mut [u8]) -> Result<(), DiskError> { + self.ram.read(first, out) + } + fn write(&mut self, first: u64, data: &[u8]) -> Result<(), DiskError> { + match self.refusing { + true => Err(DiskError::Device), + false => self.ram.write(first, data), + } + } + fn flush(&mut self) -> Result<(), DiskError> { + match self.refusing { + true => Err(DiskError::Device), + false => self.ram.flush(), + } + } + } + + #[test] + fn at_the_dirty_limit_a_refused_flush_refuses_the_write_and_the_cache_grows_no_larger() { + let limit = DIRTY_LIMIT as u64; + let c = Cache::new(Refusing { ram: Ram::new(limit + 16), refusing: false }); + for b in 0..limit { + c.write(b, &[1; BLOCK]).unwrap(); + } + c.inner.borrow_mut().disk.refusing = true; + let held = c.counts(); + assert_eq!(held.dirty, DIRTY_LIMIT); + for b in limit..limit + 3 { + assert_eq!(c.write(b, &[2; BLOCK]), Err(DiskError::Device), "block {b}"); + } + assert_eq!(c.counts(), held, "a refused write left the cache no larger"); + c.write(0, &[3; BLOCK]).expect("a block already dirty is rewritten in place"); + assert_eq!(c.counts(), held); + } } diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index f2d3bd28030..5c517ab8201 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -550,6 +550,8 @@ mod tests { assert!(v.open[&a].file.needs_reconcile()); assert_eq!(v.close(a), Err(SyscallError::Io), "a close that left the entry behind is refused"); assert!(v.open.contains_key(&a), "and its node kept"); + assert_eq!(v.sync(), Ok(vec![(a, SyscallError::Io)]), "a sync the entry still refuses names it again"); + assert!(v.open.contains_key(&a), "and keeps its node"); refused.set(u64::MAX); assert_eq!(v.sync(), Ok(Vec::new())); diff --git a/userland/init/Cargo.toml b/userland/init/Cargo.toml index 89b5d470a2e..469de96aa17 100644 --- a/userland/init/Cargo.toml +++ b/userland/init/Cargo.toml @@ -18,3 +18,5 @@ toyos-gpt = { path = "../../toyos-gpt" } toyos-blockring = { path = "../../toyos-blockring" } # How long a stop is prepared before the kernel is asked for it. toyos-quiesce = { path = "../../toyos-quiesce" } +# The kernel's device inventory, read whole or refused. +toyos-inventory = { path = "../../toyos-inventory" } diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index 79a7879c9e8..f8b752367d3 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -617,7 +617,7 @@ impl<'a> Service<'a> { 0 => Served::Keep(&kept.acceptors), _ => Served::Restart { acceptors: &kept.acceptors, owed }, }; - let storage = storage_endowment(self.program, self.role, syscap); + let storage = storage_endowment(self.program, self.role, syscap).map_err(std::io::Error::other)?; let (child, devices) = start( Command::new(path), self.program, @@ -1660,14 +1660,11 @@ fn resolve<'a>(system: &'a Manifest, path: &str) -> Resolved<'a> { /// its two partitions are claimed only as `toyos_update::slots::grant` admits /// them. fn slot_grant(syscap: &SysCap) -> Result<[(&'static str, toyos::Device); 3], String> { - use toyos_abi::inventory::{PartState, Partition, RawRecord, Record}; - let asked = syscap.inventory(&mut []).map_err(|e| format!("the inventory would not count: {e:?}"))?; - let mut raw = vec![RawRecord::EMPTY; asked]; - let n = syscap.inventory(&mut raw).map_err(|e| format!("the inventory would not read: {e:?}"))?; - let parts: Vec = raw[..n] - .iter() - .filter_map(|r| match Record::decode(r) { - Ok(Record::Partition(p)) => Some(p), + use toyos_abi::inventory::{PartState, Partition, Record}; + let parts: Vec = inventory(syscap)? + .into_iter() + .filter_map(|r| match r { + Record::Partition(p) => Some(p), _ => None, }) .collect(); @@ -2193,12 +2190,8 @@ struct Storage { } /// Every record the kernel's inventory answers. -fn inventory(syscap: &SysCap) -> Vec { - use toyos_abi::inventory::{RawRecord, Record}; - let Ok(asked) = syscap.inventory(&mut []) else { return Vec::new() }; - let mut raw = vec![RawRecord::EMPTY; asked]; - let Ok(n) = syscap.inventory(&mut raw) else { return Vec::new() }; - raw[..n].iter().filter_map(|r| Record::decode(r).ok()).collect() +fn inventory(syscap: &SysCap) -> Result, String> { + toyos_inventory::read(|buf| syscap.inventory(buf)).map_err(|why| why.to_string()) } /// The unique GUID the loader named for `role`. @@ -2221,19 +2214,19 @@ fn guid_text(guid: [u8; 16]) -> String { /// claim when a disk the kernel drives carries it — the stick, until usbd /// serves it — and otherwise the partition's GUID, which it opens through the /// block service; DATA it finds by type there itself. -fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> Storage { +fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> Result { use toyos_abi::inventory::{Record, Role}; let mut storage = Storage::default(); let is_block = program.serves.iter().any(|s| s == toyos_blockring::PORT); if !is_block && role.is_none() { - return storage; + return Ok(storage); } - let records = inventory(syscap); + let records = inventory(syscap)?; if is_block { if let Some(root) = loaded(&records, Role::Root) { storage.args.extend(["--running".to_string(), guid_text(root)]); } - return storage; + return Ok(storage); } let role = role.expect("checked above"); storage.args.push(role.to_string()); @@ -2278,7 +2271,7 @@ fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> many.len() ), } - storage + Ok(storage) } /// The manifest, off ROOT through the kernel's own `open`. From 2015fdd2a466b4aeda0e5e25400a2615851980f2 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 01:43:25 +0200 Subject: [PATCH 36/54] abuse_elf_loader: the image route's refusal is the kernel's first, not always NotFound spawn_refused_beside_its_library assumed the image route always answers NotFound, because an image's libraries come from /system/lib alone. That holds for export_past_image and tpoff_overflow_spawn, whose refusals (the export map in spawn(), apply_tls_relocs) run after load_needed_libs. globdat_past_dynsym's GLOB_DAT symbol index is bounds-checked by rela::parse inside read_exe_tables, before load_needed_libs ever runs, so the image route never reaches the /system/lib walk and answers InvalidArgument instead. The helper now takes the expected image-route answer per case. Co-Authored-By: Claude Opus 5.5 --- .../src/bin/abuse_elf_loader.rs | 23 +++++++++++++------ 1 file changed, 16 insertions(+), 7 deletions(-) diff --git a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs index 32ef907b096..fc00fc530f0 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs @@ -308,12 +308,15 @@ fn spawn_refused(name: &str, bytes: &[u8]) { } /// Refused where the kernel opens the file, whose `DT_NEEDED` library is -/// written beside it; handed its bytes, it is `NotFound`, since an image's -/// libraries come from `/system/lib` alone. -fn spawn_refused_beside_its_library(name: &str, bytes: &[u8]) { +/// written beside it. The image route's answer is the kernel's *first* +/// refusal on that route: `image_expected` is `NotFound` when nothing rejects +/// the file before `DT_NEEDED` is walked from `/system/lib` alone and finds +/// nothing there, `InvalidArgument` when a table check upstream of that walk +/// — read while the file is still opened directly — refuses first. +fn spawn_refused_beside_its_library(name: &str, bytes: &[u8], image_expected: SyscallError) { refused(name, spawn_result(name, bytes)); let image = spawn_image(&format!("{DIR}/{name}"), bytes); - assert_eq!(image, Err(SyscallError::NotFound), "{name} as an image: its library is not in /system/lib"); + assert_eq!(image, Err(image_expected), "{name} as an image"); } /// Load it and throw it away. These cases are about a *walk* the loader does, @@ -760,7 +763,9 @@ fn values_are_bounded_by_the_image() { .sym(0x1418, 1, (STB_GLOBAL << 4) | STT_FUNC, 1, FAR_VALUE) .poke(0x1801, FAR.as_bytes()) .poke(0x1800 + name_at as usize, dep.as_bytes()); - spawn_refused_beside_its_library("export_past_image", &exe.build()); + // The export map is built after `load_needed_libs` (`kernel/src/loader/mod.rs`), so the + // image route never reaches it: the dependency missing from `/system/lib` answers first. + spawn_refused_beside_its_library("export_past_image", &exe.build(), SyscallError::NotFound); } dlopen_refused("so_relative_addend_past_image.so", &so_with(&[], &[(0x1400, 0, R_X86_64_RELATIVE, i64::MAX)], None)); @@ -925,7 +930,9 @@ fn tls_apply_time_refusals_are_reached() { .sym(0x1418, 1, (STB_GLOBAL << 4) | STT_TLS, 0, 0) .poke(0x1801, b"wtls\0") .poke(0x1806, dep.as_bytes()); - spawn_refused_beside_its_library("tpoff_overflow_spawn", &exe.build()); + // `apply_tls_relocs` (`kernel/src/loader/mod.rs`) runs after `load_needed_libs`, so the image + // route never reaches it: the dependency missing from `/system/lib` answers first. + spawn_refused_beside_its_library("tpoff_overflow_spawn", &exe.build(), SyscallError::NotFound); let defs = write_file("tpoff_overflow_defs.so", &tls_defs_so(b"vtls", PAST_I64)); let lib_defs = unsafe { libloading::Library::new(&defs) }.expect("dlopen tpoff_overflow_defs.so"); @@ -952,7 +959,9 @@ fn globdat_past_short_dynsym() { ) .poke(0x1801, dep.as_bytes()) .poke(0x3000, &gnu_hash); - spawn_refused_beside_its_library("globdat_past_dynsym", &exe.build()); + // `rela::parse` (`toyos-elf/src/rela.rs`) bounds `r_sym` against the exe's own `.dynsym` while + // `read_exe_tables` still holds the file open directly, before `load_needed_libs` ever runs. + spawn_refused_beside_its_library("globdat_past_dynsym", &exe.build(), SyscallError::InvalidArgument); } /// A shared object defining its own `xtls` (`STT_TLS`, offset 8) in a 0x20-byte From 4aef9a0d0915e306583577b6dff10bae1289b8ea Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 10:49:49 +0200 Subject: [PATCH 37/54] heap_ceiling_bounds: the lowered sysinfo bound is the machine's own live threads plus 16 `LOWER_SYSINFO_BOUND` put a constant 16 in `MAX_SYSINFO_THREADS`'s place. This branch's storage stack adds resting threads to every test boot (blockd, three fsd processes, init's `files` worker and its `wait-*` threads), so the actuator boot was past 16 before the test spawned anything and `sys_sysinfo` refused at the first extra thread and never recovered. The kernel now counts the live threads at arming, with the one function `sys_sysinfo` counts its roster with, and stores that count plus 16. The bound moves with whatever daemons a boot runs. The test reads the header's count before arming and prints it on its PASS line, so the orchestrator's run records the machine's count. Negative control: `entry_count as usize > sysinfo_thread_bound()` made `false &&` it must fail at the test's "64 extra threads and sysinfo never refused" panic. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- kernel/src/syscall/dispatch.rs | 4 +- kernel/src/syscall/machine.rs | 29 ++++-- .../toyos-rust-tests/src/bin/heap_ceiling.rs | 32 ++++--- toyos-inventory/Cargo.toml | 16 ---- toyos-inventory/src/lib.rs | 93 ------------------- 5 files changed, 42 insertions(+), 132 deletions(-) delete mode 100644 toyos-inventory/Cargo.toml delete mode 100644 toyos-inventory/src/lib.rs diff --git a/kernel/src/syscall/dispatch.rs b/kernel/src/syscall/dispatch.rs index d8d328e71db..8140181badc 100644 --- a/kernel/src/syscall/dispatch.rs +++ b/kernel/src/syscall/dispatch.rs @@ -45,7 +45,7 @@ use super::machine::{ MAX_INVENTORY_RECORDS, }; #[cfg(feature = "test-actuators")] -use super::machine::SYSINFO_BOUND_LOWERED; +use super::machine::lower_sysinfo_bound; use super::proc::{ sys_endowments, sys_exit, sys_nanosleep, sys_process_open, sys_process_stats, sys_process_wait, sys_rt_enter, sys_spawn, sys_thread_exit, sys_thread_join, sys_thread_spawn, @@ -629,7 +629,7 @@ pub(crate) fn syscall_dispatch(num: u64, a1: u64, a2: u64, a3: u64, a4: u64) -> // Armed, not #[cfg]'d, so it doesn't ship in every kernel this suite boots: // the real bound is unreachable (no guest makes 65,536 threads). DA::LOWER_SYSINFO_BOUND => { - SYSINFO_BOUND_LOWERED.store(true, core::sync::atomic::Ordering::Relaxed); + lower_sysinfo_bound(); 0 } // Puts one free slot one lifecycle from the end so retirement is reachable without diff --git a/kernel/src/syscall/machine.rs b/kernel/src/syscall/machine.rs index 1afe563d3f9..478937724e3 100644 --- a/kernel/src/syscall/machine.rs +++ b/kernel/src/syscall/machine.rs @@ -199,23 +199,36 @@ pub(super) fn sys_reboot(syscap: RawHandle) -> u64 { /// The most live threads `SYS_SYSINFO` will describe; kept under `mm::MAX_HEAP_ALLOC` so an unbounded thread count cannot trip the allocator's fail-fast assert. const MAX_SYSINFO_THREADS: usize = 65_536; -/// Test-only override for `MAX_SYSINFO_THREADS`, armed at runtime by `DA::LOWER_SYSINFO_BOUND` so the shipped bound stays exercised. +/// How far past the machine's live threads at arming `DA::LOWER_SYSINFO_BOUND` puts the bound. #[cfg(feature = "test-actuators")] -const GATED_SYSINFO_THREADS: usize = 16; +const LOWERED_SYSINFO_HEADROOM: usize = 16; +/// `MAX_SYSINFO_THREADS` until `DA::LOWER_SYSINFO_BOUND` lowers it for the rest of the boot. #[cfg(feature = "test-actuators")] -pub(super) static SYSINFO_BOUND_LOWERED: core::sync::atomic::AtomicBool = - core::sync::atomic::AtomicBool::new(false); +static SYSINFO_BOUND: core::sync::atomic::AtomicUsize = + core::sync::atomic::AtomicUsize::new(MAX_SYSINFO_THREADS); + +/// Lowers [`sys_sysinfo`]'s bound to the machine's own live threads plus a fixed headroom, counted as `sys_sysinfo` counts them. +#[cfg(feature = "test-actuators")] +pub(super) fn lower_sysinfo_bound() { + let guard = process::PROCESS_TABLE.lock(); + let live = live_threads(guard.as_ref().unwrap()); + SYSINFO_BOUND.store(live + LOWERED_SYSINFO_HEADROOM, core::sync::atomic::Ordering::Relaxed); +} /// What [`sys_sysinfo`] compares against on this boot. fn sysinfo_thread_bound() -> usize { #[cfg(feature = "test-actuators")] - if SYSINFO_BOUND_LOWERED.load(core::sync::atomic::Ordering::Relaxed) { - return GATED_SYSINFO_THREADS; - } + return SYSINFO_BOUND.load(core::sync::atomic::Ordering::Relaxed); + #[cfg(not(feature = "test-actuators"))] MAX_SYSINFO_THREADS } +/// Every thread in the process table, zombies included: the roster's entries. +fn live_threads(table: &process::ProcessTable) -> usize { + table.iter().map(|(_, proc)| proc.threads().iter().count()).sum() +} + /// The machine's header, then the live-thread roster for as much of `out` as fits; the roster requires a `SysCap` carrying `Rights::ROSTER`, demanded only when `out` has room for an entry. pub(super) fn sys_sysinfo(syscap: RawHandle, out: &mut UserBytesMut) -> u64 { const HEADER_SIZE: usize = toyos_abi::syscall::SYSINFO_HEADER_SIZE; @@ -240,7 +253,7 @@ pub(super) fn sys_sysinfo(syscap: RawHandle, out: &mut UserBytesMut) -> u64 { let guard = process::PROCESS_TABLE.lock(); let table = guard.as_ref().unwrap(); - let entry_count: u32 = table.iter().flat_map(|(_, proc)| proc.threads().iter().map(move |(tid, thread)| (tid, proc, thread))).count() as u32; + let entry_count = live_threads(table) as u32; if entry_count as usize > sysinfo_thread_bound() { return SyscallError::ResourceExhausted.to_u64(); } diff --git a/tests/toyos-rust-tests/src/bin/heap_ceiling.rs b/tests/toyos-rust-tests/src/bin/heap_ceiling.rs index d524e957657..4069d7bca73 100644 --- a/tests/toyos-rust-tests/src/bin/heap_ceiling.rs +++ b/tests/toyos-rust-tests/src/bin/heap_ceiling.rs @@ -7,8 +7,9 @@ // `SYS_DEBUG` actions a `test-actuators` kernel provides. The first two take // one kernel heap allocation each and release it again — at -// `mm::MAX_HEAP_ALLOC`, and at `MAX_HEAP_ALLOC` with 4096-byte alignment; the last lowers `SYS_SYSINFO`'s thread bound to a count -// this guest can reach. +// `mm::MAX_HEAP_ALLOC`, and at `MAX_HEAP_ALLOC` with 4096-byte alignment; the +// last lowers `SYS_SYSINFO`'s thread bound to the machine's live threads plus +// 16. use toyos_abi::syscall::debug_action::{ HEAP_AT_CEILING, HEAP_AT_CEILING_PAGE_ALIGNED, LOWER_SYSINFO_BOUND, }; @@ -31,9 +32,10 @@ fn main() { /// ask the heap for more than `MAX_HEAP_ALLOC` and trip the assert three /// functions above — from any process, with no privilege. /// -/// [`LOWER_SYSINFO_BOUND`] puts 16 in `MAX_SYSINFO_THREADS`'s place, because -/// 65,536 threads is 8 GiB of kernel stacks and no guest can make them. The -/// count, the comparison and the refusal are the shipped ones. +/// [`LOWER_SYSINFO_BOUND`] puts the machine's live threads plus 16 in +/// `MAX_SYSINFO_THREADS`'s place, because 65,536 threads is 8 GiB of kernel +/// stacks and no guest can make them. The count, the comparison and the +/// refusal are the shipped ones. /// /// **Armed here rather than compiled in, and the arming is itself an /// assertion**: the bound is the shipped 65,536 until this call, so a kernel @@ -44,7 +46,7 @@ fn sysinfo_refuses_rather_than_allocating_past_the_ceiling() { use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::Arc; - assert!(sysinfo_answers(), "sysinfo already refuses with no threads of ours"); + let live = sysinfo_live().expect("sysinfo already refuses with no threads of ours"); let rc = toyos_abi::syscall::debug(LOWER_SYSINFO_BOUND); assert_eq!(rc, 0, "SYS_DEBUG {LOWER_SYSINFO_BOUND} did not lower the bound (rc={rc:#x})"); @@ -60,7 +62,7 @@ fn sysinfo_refuses_rather_than_allocating_past_the_ceiling() { std::thread::sleep(std::time::Duration::from_millis(5)); } })); - if !sysinfo_answers() { + if sysinfo_live().is_none() { refused_at = Some(i + 1); break; } @@ -76,15 +78,19 @@ fn sysinfo_refuses_rather_than_allocating_past_the_ceiling() { t.join().expect("join a parked thread"); } // A bound, not a one-way door: with the threads gone it answers again. - assert!(sysinfo_answers(), "sysinfo stayed refused after the threads exited"); - println!(" PASS: sysinfo refused past its bound at {at} extra threads, and recovered"); + assert!(sysinfo_live().is_some(), "sysinfo stayed refused after the threads exited"); + println!( + " PASS: sysinfo refused past its bound at {at} extra threads over {live} live before arming, and recovered" + ); } -/// Whether `SYS_SYSINFO` filled its header. The ABI wrapper reports an error -/// as `0`, and the header is the smallest buffer it accepts. -fn sysinfo_answers() -> bool { +/// The live threads `SYS_SYSINFO`'s header counts, or `None` when it refused. +/// The ABI wrapper reports an error as `0`, and the header is the smallest +/// buffer it accepts. +fn sysinfo_live() -> Option { let mut buf = [0u8; toyos::system::SYSINFO_HEADER_SIZE]; - toyos::system::sysinfo(&mut buf) == buf.len() + (toyos::system::sysinfo(&mut buf) == buf.len()) + .then(|| toyos_abi::syscall::SysinfoHeader::decode(&buf).entries) } /// The documented ceiling is a size the heap actually serves. diff --git a/toyos-inventory/Cargo.toml b/toyos-inventory/Cargo.toml deleted file mode 100644 index f2923579630..00000000000 --- a/toyos-inventory/Cargo.toml +++ /dev/null @@ -1,16 +0,0 @@ -# A member of the host workspace (root `Cargo.toml`): the kernel's device -# inventory read whole, which init reads and the host tests. - -[package] -name = "toyos-inventory" -description = "The kernel's device inventory read whole, or refused whole, never a shorter list." -version = "0.1.0" -edition = "2021" -license = "MIT OR Apache-2.0" -publish = false - -[dependencies] -toyos-abi = { path = "../toyos-abi" } - -[lints.rust] -warnings = "deny" diff --git a/toyos-inventory/src/lib.rs b/toyos-inventory/src/lib.rs deleted file mode 100644 index 41064eb242f..00000000000 --- a/toyos-inventory/src/lib.rs +++ /dev/null @@ -1,93 +0,0 @@ -//! The kernel's device inventory, read whole: a count, then that many records, -//! each of which decodes. A refused count, a refused read and a record that -//! does not decode are each the answer, never a shorter list. - -#![cfg_attr(not(test), no_std)] -#![forbid(unsafe_code)] - -extern crate alloc; - -use alloc::vec; -use alloc::vec::Vec; - -use toyos_abi::inventory::{RawRecord, Record, Undecodable}; -use toyos_abi::syscall::SyscallError; - -/// Why the inventory was not read. -#[derive(Clone, Copy, PartialEq, Eq, Debug)] -pub enum Unread { - Count(SyscallError), - Read(SyscallError), - Record { index: usize, why: Undecodable }, -} - -impl core::fmt::Display for Unread { - fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { - match self { - Self::Count(e) => write!(f, "the inventory would not count: {e:?}"), - Self::Read(e) => write!(f, "the inventory would not read: {e:?}"), - Self::Record { index, why } => write!(f, "the inventory's record {index} does not decode: {why}"), - } - } -} - -/// Every record `ask` answers, where `ask` is the inventory call: an empty -/// buffer asks how many, and a buffer that long is filled. -pub fn read(mut ask: impl FnMut(&mut [RawRecord]) -> Result) -> Result, Unread> { - let asked = ask(&mut []).map_err(Unread::Count)?; - let mut raw = vec![RawRecord::EMPTY; asked]; - let n = ask(&mut raw).map_err(Unread::Read)?; - raw[..n] - .iter() - .enumerate() - .map(|(index, r)| Record::decode(r).map_err(|why| Unread::Record { index, why })) - .collect() -} - -#[cfg(test)] -mod tests { - use super::*; - use toyos_abi::inventory::{Loaded, Role}; - - fn loaded(role: Role) -> Record { - Record::Loaded(Loaded { role, unique_guid: [7; 16] }) - } - - /// An inventory of `records`, as the kernel answers one. - fn kernel(records: Vec) -> impl FnMut(&mut [RawRecord]) -> Result { - move |buf| match buf.len() { - 0 => Ok(records.len()), - n if n < records.len() => Err(SyscallError::ResourceExhausted), - _ => { - buf[..records.len()].copy_from_slice(&records); - Ok(records.len()) - } - } - } - - #[test] - fn every_record_is_read() { - let records = [loaded(Role::Root), loaded(Role::Log)]; - assert_eq!(read(kernel(records.iter().map(Record::encode).collect())), Ok(records.to_vec())); - } - - #[test] - fn a_refused_count_is_refused_and_not_an_empty_inventory() { - assert_eq!(read(|_| Err(SyscallError::PermissionDenied)), Err(Unread::Count(SyscallError::PermissionDenied))); - } - - #[test] - fn a_refused_read_is_refused_and_not_an_empty_inventory() { - let refused = |buf: &mut [RawRecord]| match buf.len() { - 0 => Ok(1), - _ => Err(SyscallError::ResourceExhausted), - }; - assert_eq!(read(refused), Err(Unread::Read(SyscallError::ResourceExhausted))); - } - - #[test] - fn a_record_that_does_not_decode_is_refused_and_not_dropped() { - let records = vec![loaded(Role::Root).encode(), RawRecord::EMPTY, loaded(Role::Boot).encode()]; - assert_eq!(read(kernel(records)), Err(Unread::Record { index: 1, why: Undecodable::Kind(0) })); - } -} From 14976229010cf38c8723163e37779c1804227566 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 10:50:05 +0200 Subject: [PATCH 38/54] One inventory reader, in the SDK, with the bounded retry; toyos-inventory goes init read the inventory through `toyos-inventory`, which turned a machine that grew between the count and the read (a USB device enumerating, any claim minted) into a refusal, failing blockd's or fsd's start. inspect had its own reader with a bounded retry. Both now call `SysCap::records`: count, read into a buffer of that count, ask again while the kernel answers `ResourceExhausted`, at most four rounds, and refuse by name (`Unread::Grew`) when the machine grew on every one. A count of zero is the whole answer, since an empty buffer would ask the count again. The SDK takes no allocator, so the caller hands the buffer and the collection type. Host tests in `toyos/src/syscap.rs`: a machine that grows once is read grown, one that grows every round is refused by name after four counts and four reads, plus the four refusal tests the deleted crate carried and an empty inventory. Deleting the `ResourceExhausted => continue` arm turns the grown-once test red, EXIT=101, `Err(Read(ResourceExhausted))` against the grown list. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- Cargo.lock | 7 -- Cargo.toml | 1 - toyos/src/syscap.rs | 172 +++++++++++++++++++++++++++++++++++ userland/Cargo.lock | 8 -- userland/init/Cargo.toml | 2 - userland/init/src/main.rs | 4 +- userland/inspect/src/main.rs | 29 +----- 7 files changed, 176 insertions(+), 47 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index df88c07292f..ddf4a4324a1 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1309,13 +1309,6 @@ dependencies = [ "toyos-abi", ] -[[package]] -name = "toyos-inventory" -version = "0.1.0" -dependencies = [ - "toyos-abi", -] - [[package]] name = "toyos-keymap" version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index 07acce874d8..ccc0d35460f 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -35,7 +35,6 @@ members = [ "toyos-hda", "toyos-i219", "toyos-inspect", - "toyos-inventory", "toyos-keymap", "toyos-ld", "toyos-libc-copies", diff --git a/toyos/src/syscap.rs b/toyos/src/syscap.rs index 2bc9aec4b45..099c715847d 100644 --- a/toyos/src/syscap.rs +++ b/toyos/src/syscap.rs @@ -8,6 +8,7 @@ //! processes that can ever do any of them is exactly what init endowed. use toyos_abi::handle::Rights; +use toyos_abi::inventory::{RawRecord, Record, Undecodable}; use toyos_abi::syscall::{self, DeviceRequest, DeviceType, SyscallError}; use crate::endow::FromHandle; @@ -118,6 +119,18 @@ impl SysCap { syscall::device_inventory(self.0.raw(), buf) } + /// Every inventory record, read whole or refused whole, never a shorter + /// list; `buffer(n)` is `n` records to read into. + /// + /// Needs [`Rights::INVENTORY`]. + pub fn records(&self, buffer: impl FnMut(usize) -> B) -> Result + where + B: AsMut<[RawRecord]>, + C: FromIterator, + { + read_whole(|buf| self.inventory(buf), buffer) + } + /// A second handle to this capability carrying **less**. /// /// How init gives a program the RT band and nothing else: rights only @@ -150,3 +163,162 @@ impl AsHandle for SysCap { self.0.raw() } } + +/// How many times [`SysCap::records`] asks again after the machine grew +/// between counting its inventory and reading it. Policy: a device arriving on +/// every round is a machine the reader names rather than chases. +const INVENTORY_ROUNDS: usize = 4; + +/// Why [`SysCap::records`] did not read the inventory. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Unread { + Count(SyscallError), + Read(SyscallError), + /// The machine grew between the count and the read on every round. + Grew, + Record { index: usize, why: Undecodable }, +} + +impl core::fmt::Display for Unread { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self { + Self::Count(SyscallError::PermissionDenied) | Self::Read(SyscallError::PermissionDenied) => { + write!(f, "the kernel refused: this program's capability does not carry `inventory`") + } + Self::Count(e) => write!(f, "the inventory would not count: {e:?}"), + Self::Read(e) => write!(f, "the inventory would not read: {e:?}"), + Self::Grew => write!(f, "the machine changed on each of {INVENTORY_ROUNDS} reads of its inventory"), + Self::Record { index, why } => write!(f, "the inventory's record {index} does not decode: {why}"), + } + } +} + +/// Every record `ask` answers, where `ask` is the inventory call: an empty +/// buffer asks how many, and a buffer that long is filled or refused with +/// `ResourceExhausted` because the machine grew since. +fn read_whole( + mut ask: impl FnMut(&mut [RawRecord]) -> Result, + mut buffer: impl FnMut(usize) -> B, +) -> Result +where + B: AsMut<[RawRecord]>, + C: FromIterator, +{ + for _ in 0..INVENTORY_ROUNDS { + let count = ask(&mut []).map_err(Unread::Count)?; + // An empty buffer would ask the count again rather than read. + if count == 0 { + return Ok(core::iter::empty().collect()); + } + let mut raw = buffer(count); + let raw = raw.as_mut(); + assert_eq!(raw.len(), count, "`buffer({count})` answered {} records", raw.len()); + match ask(raw) { + Ok(n) => { + return raw[..n] + .iter() + .enumerate() + .map(|(index, r)| Record::decode(r).map_err(|why| Unread::Record { index, why })) + .collect(); + } + Err(SyscallError::ResourceExhausted) => continue, + Err(e) => return Err(Unread::Read(e)), + } + } + Err(Unread::Grew) +} + +#[cfg(test)] +mod tests { + use super::*; + use toyos_abi::inventory::{Loaded, Role}; + + fn loaded(role: Role) -> Record { + Record::Loaded(Loaded { role, unique_guid: [7; 16] }) + } + + fn read(ask: impl FnMut(&mut [RawRecord]) -> Result) -> Result, Unread> { + read_whole(ask, |n| vec![RawRecord::EMPTY; n]) + } + + /// The kernel's answer to `buf` from a machine of `records`. + fn answer(records: &[RawRecord], buf: &mut [RawRecord]) -> Result { + match buf.len() { + 0 => Ok(records.len()), + n if n < records.len() => Err(SyscallError::ResourceExhausted), + _ => { + buf[..records.len()].copy_from_slice(records); + Ok(records.len()) + } + } + } + + #[test] + fn every_record_is_read() { + let records = [loaded(Role::Root), loaded(Role::Log)]; + let raw: Vec = records.iter().map(Record::encode).collect(); + assert_eq!(read(|buf| answer(&raw, buf)), Ok(records.to_vec())); + } + + #[test] + fn a_machine_that_grew_once_is_read_grown() { + let before = [loaded(Role::Root)]; + let after = [loaded(Role::Root), loaded(Role::Log)]; + let (before_raw, after_raw): (Vec, Vec) = + (before.iter().map(Record::encode).collect(), after.iter().map(Record::encode).collect()); + let mut counted = false; + let grows_after_the_first_count = |buf: &mut [RawRecord]| { + let machine = if counted { &after_raw } else { &before_raw }; + counted |= buf.is_empty(); + answer(machine, buf) + }; + assert_eq!(read(grows_after_the_first_count), Ok(after.to_vec())); + } + + #[test] + fn a_machine_that_grows_every_round_is_refused_by_name() { + let mut machine = vec![loaded(Role::Root).encode()]; + let mut asks = 0; + let grows_after_every_count = |buf: &mut [RawRecord]| { + let answered = answer(&machine, buf); + if buf.is_empty() { + machine.push(loaded(Role::Boot).encode()); + } + asks += 1; + answered + }; + assert_eq!(read(grows_after_every_count), Err(Unread::Grew)); + assert_eq!(asks, 2 * INVENTORY_ROUNDS, "every round counted and read once"); + } + + #[test] + fn an_empty_inventory_is_its_count() { + let mut asks = 0; + let empty_then_grown = |buf: &mut [RawRecord]| { + asks += 1; + answer(&vec![loaded(Role::Root).encode(); asks - 1], buf) + }; + assert_eq!(read(empty_then_grown), Ok(Vec::new())); + assert_eq!(asks, 1, "the count was the whole answer"); + } + + #[test] + fn a_refused_count_is_refused_and_not_an_empty_inventory() { + assert_eq!(read(|_| Err(SyscallError::PermissionDenied)), Err(Unread::Count(SyscallError::PermissionDenied))); + } + + #[test] + fn a_refused_read_is_refused_and_not_an_empty_inventory() { + let refused = |buf: &mut [RawRecord]| match buf.len() { + 0 => Ok(1), + _ => Err(SyscallError::PermissionDenied), + }; + assert_eq!(read(refused), Err(Unread::Read(SyscallError::PermissionDenied))); + } + + #[test] + fn a_record_that_does_not_decode_is_refused_and_not_dropped() { + let raw = [loaded(Role::Root).encode(), RawRecord::EMPTY, loaded(Role::Boot).encode()]; + assert_eq!(read(|buf| answer(&raw, buf)), Err(Unread::Record { index: 1, why: Undecodable::Kind(0) })); + } +} diff --git a/userland/Cargo.lock b/userland/Cargo.lock index 5d24b653269..f77dcdf6886 100644 --- a/userland/Cargo.lock +++ b/userland/Cargo.lock @@ -1652,7 +1652,6 @@ dependencies = [ "toyos-abi", "toyos-blockring", "toyos-gpt", - "toyos-inventory", "toyos-logstream", "toyos-manifest", "toyos-quiesce", @@ -4127,13 +4126,6 @@ dependencies = [ "toyos-abi", ] -[[package]] -name = "toyos-inventory" -version = "0.1.0" -dependencies = [ - "toyos-abi", -] - [[package]] name = "toyos-keymap" version = "0.1.0" diff --git a/userland/init/Cargo.toml b/userland/init/Cargo.toml index 469de96aa17..89b5d470a2e 100644 --- a/userland/init/Cargo.toml +++ b/userland/init/Cargo.toml @@ -18,5 +18,3 @@ toyos-gpt = { path = "../../toyos-gpt" } toyos-blockring = { path = "../../toyos-blockring" } # How long a stop is prepared before the kernel is asked for it. toyos-quiesce = { path = "../../toyos-quiesce" } -# The kernel's device inventory, read whole or refused. -toyos-inventory = { path = "../../toyos-inventory" } diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index f8b752367d3..2b24bb235de 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -2191,7 +2191,9 @@ struct Storage { /// Every record the kernel's inventory answers. fn inventory(syscap: &SysCap) -> Result, String> { - toyos_inventory::read(|buf| syscap.inventory(buf)).map_err(|why| why.to_string()) + syscap + .records(|n| vec![toyos_abi::inventory::RawRecord::EMPTY; n]) + .map_err(|why| why.to_string()) } /// The unique GUID the loader named for `role`. diff --git a/userland/inspect/src/main.rs b/userland/inspect/src/main.rs index 7e98270b7da..e0b5a07682b 100644 --- a/userland/inspect/src/main.rs +++ b/userland/inspect/src/main.rs @@ -21,7 +21,6 @@ use std::time::{Duration, Instant}; use toyos::endow::{self, EndowError, Endowments, SYSCAP_LABEL}; use toyos::syscap::SysCap; use toyos_abi::inventory::{RawRecord, Record}; -use toyos_abi::syscall::SyscallError; use toyos::ipc::{self, FrameRx, RxStep}; use toyos::poller::{Poller, READABLE}; use toyos_inspect::{Invocation, Owner, Value, MAX_SNAPSHOT_BYTES, MSG_INSPECT, MSG_SNAPSHOT}; @@ -139,11 +138,6 @@ fn ask(owner: Owner) -> Result, String> { } } -/// How many times the inventory is asked for again after the machine changed -/// between counting it and reading it. Policy: a device arriving on every -/// round is a machine the reader names rather than chases. -const INVENTORY_ROUNDS: usize = 4; - /// The kernel's inventory, asked with this process's `SysCap`, and the /// machine `SYS_SYSINFO`'s ambient header describes, as `dev.*` paths. fn inventory() -> Result, String> { @@ -162,26 +156,5 @@ fn records() -> Result, String> { return Err("this program holds no system capability, so the inventory is not its to read" .to_string()); }; - let refused = |e: SyscallError| match e { - SyscallError::PermissionDenied => { - "the kernel refused: this program's capability does not carry `inventory`".to_string() - } - other => format!("the kernel refused the inventory ({other:?})"), - }; - for _ in 0..INVENTORY_ROUNDS { - let count = cap.inventory(&mut []).map_err(refused)?; - let mut raw = vec![RawRecord::EMPTY; count]; - match cap.inventory(&mut raw) { - Ok(n) => { - return raw[..n] - .iter() - .map(|r| Record::decode(r).map_err(|why| format!("a record did not decode: {why}"))) - .collect(); - } - // The machine grew between the two calls. - Err(SyscallError::ResourceExhausted) => continue, - Err(e) => return Err(refused(e)), - } - } - Err(format!("the machine changed on each of {INVENTORY_ROUNDS} reads of its inventory")) + cap.records(|n| vec![RawRecord::EMPTY; n]).map_err(|why| why.to_string()) } From a20936636f384787d1547cdee002f01bc6832207 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 10:50:26 +0200 Subject: [PATCH 39/54] Cut the prose the branch made false; a missing space; file a stale actuator list - `usb_gate`'s header and `toyos-gpt`'s cited `bcachefs_adapter::probe`, which this branch deleted, and `toyos-gpt`'s a commit hash. - `block.rs`'s command counter named NVMe, which the kernel no longer drives. - `blockd.rs`'s `boot` crafts one disk, not both. - `abuse_elf_loader`: on the image route nothing opens the file directly. - Filed: `needs_actuators` lists `LOWER_SYSINFO_BOUND` as an action that carries a payload; it takes none (on main too). 4aef9a0d also carries `toyos-inventory`'s deletion, which was staged before it; 14976229 moves the crate's users, so 4aef9a0d alone does not build. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...s-lower-sysinfo-bound-as-a-payload-action.md | 17 +++++++++++++++++ kernel/src/block.rs | 2 +- kernel/src/usb_gate.rs | 3 --- tests/common/blockd.rs | 6 +++--- .../src/bin/abuse_elf_loader.rs | 6 +++--- toyos-gpt/src/lib.rs | 3 +-- 6 files changed, 25 insertions(+), 12 deletions(-) create mode 100644 issues/build/needs-actuators-names-lower-sysinfo-bound-as-a-payload-action.md diff --git a/issues/build/needs-actuators-names-lower-sysinfo-bound-as-a-payload-action.md b/issues/build/needs-actuators-names-lower-sysinfo-bound-as-a-payload-action.md new file mode 100644 index 00000000000..18c9d25900f --- /dev/null +++ b/issues/build/needs-actuators-names-lower-sysinfo-bound-as-a-payload-action.md @@ -0,0 +1,17 @@ +--- +status: open +kind: finding +opened: 2026-09-28 +--- + +# `needs_actuators` names `LOWER_SYSINFO_BOUND` as a payload-carrying action + +The comment in `tests/toyos.rs`'s `needs_actuators` lists `LOWER_SYSINFO_BOUND` +among the `SYS_DEBUG` actions that carry a payload and are reached through +`debug_with`. It takes no argument: `heap_ceiling.rs` reaches it through +`syscall::debug(LOWER_SYSINFO_BOUND)`, and the kernel's arm in +`kernel/src/syscall/dispatch.rs` reads no `a2`. + +**Evidence:** `git grep -n "CENSUS_KIND, LOWER_SYSINFO_BOUND" origin/main -- tests/toyos.rs`. + +**Exit condition:** the list names only actions that read an argument. diff --git a/kernel/src/block.rs b/kernel/src/block.rs index f7d0a03a4af..886ee141ed4 100644 --- a/kernel/src/block.rs +++ b/kernel/src/block.rs @@ -644,7 +644,7 @@ pub mod census { &SLOTS[DEVICES - 1] } - /// Every command a storage driver has put to a disk, NVMe or USB, counted + /// Every command a storage driver has put to a disk, counted /// where each driver hands one to its transport: the one number that says a /// stretch of the boot needed no disk. static COMMANDS: AtomicU64 = AtomicU64::new(0); diff --git a/kernel/src/usb_gate.rs b/kernel/src/usb_gate.rs index 159da5dc04a..a6b187918c3 100644 --- a/kernel/src/usb_gate.rs +++ b/kernel/src/usb_gate.rs @@ -2,9 +2,6 @@ //! //! Verifies blocks the host wrote and writes blocks the host can check, so //! neither half of the driver certifies itself. -//! -//! Never writes to a disk without the block-0 stamp `bcachefs_adapter::probe` -//! requires before formatting `/home`. use alloc::vec; diff --git a/tests/common/blockd.rs b/tests/common/blockd.rs index bf60bb386d8..678770aef5c 100644 --- a/tests/common/blockd.rs +++ b/tests/common/blockd.rs @@ -100,8 +100,8 @@ fn craft_blockd_disk(path: &Path) -> Result { Ok(layout) } -/// A boot with both disks crafted fresh, the actuators `params` armed, and -/// QEMU tracing NVMe to `trace`. +/// A boot with the actuators `params` armed, and QEMU tracing NVMe to +/// `trace`. fn boot( c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], @@ -111,7 +111,7 @@ fn boot( let config = super::compile::repo_root().join(CONFIG); let blockd_disk = super::lane::dir().join(format!("{name}-blockd.img")); let layout = craft_blockd_disk(&blockd_disk)?; - let before =std::fs::read(&blockd_disk).map_err(|e| format!("read the crafted disk: {e}"))?; + let before = std::fs::read(&blockd_disk).map_err(|e| format!("read the crafted disk: {e}"))?; let trace = super::lane::dir().join(format!("{name}-nvme.trace")); let _ = std::fs::remove_file(&trace); let qemu = QemuInstance::boot_with_options( diff --git a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs index fc00fc530f0..8c53bc870c6 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_elf_loader.rs @@ -312,7 +312,7 @@ fn spawn_refused(name: &str, bytes: &[u8]) { /// refusal on that route: `image_expected` is `NotFound` when nothing rejects /// the file before `DT_NEEDED` is walked from `/system/lib` alone and finds /// nothing there, `InvalidArgument` when a table check upstream of that walk -/// — read while the file is still opened directly — refuses first. +/// refuses first. fn spawn_refused_beside_its_library(name: &str, bytes: &[u8], image_expected: SyscallError) { refused(name, spawn_result(name, bytes)); let image = spawn_image(&format!("{DIR}/{name}"), bytes); @@ -959,8 +959,8 @@ fn globdat_past_short_dynsym() { ) .poke(0x1801, dep.as_bytes()) .poke(0x3000, &gnu_hash); - // `rela::parse` (`toyos-elf/src/rela.rs`) bounds `r_sym` against the exe's own `.dynsym` while - // `read_exe_tables` still holds the file open directly, before `load_needed_libs` ever runs. + // `rela::parse` (`toyos-elf/src/rela.rs`) bounds `r_sym` against the exe's own `.dynsym` + // before `load_needed_libs` ever runs. spawn_refused_beside_its_library("globdat_past_dynsym", &exe.build(), SyscallError::InvalidArgument); } diff --git a/toyos-gpt/src/lib.rs b/toyos-gpt/src/lib.rs index 73e5862dd54..bef84611d09 100644 --- a/toyos-gpt/src/lib.rs +++ b/toyos-gpt/src/lib.rs @@ -6,8 +6,7 @@ //! are still alive and hands it to the kernel; this crate finds *that* GUID in //! *that* device's table, or refuses. Searching for a type GUID, or for the //! first FAT-looking thing, is how an operating system reformats a disk that -//! belongs to somebody else — the same class of defect `bcachefs_adapter::probe` -//! exists to prevent, and the reason `5dff9aa` exists. +//! belongs to somebody else. //! //! Everything here treats the disk as hostile. A GPT is bytes an attacker (or //! a dying flash controller) may have written: every length, count and LBA in From 01e254157caa11ffb606f11ab903bf64dd3578b7 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 11:21:20 +0200 Subject: [PATCH 40/54] A refused partition claim refuses its file server's start; the role is absent, never in memory MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit init's storage start for a file-server role said a refused partition claim, or a role two partitions on the kernel's disks carry, and started the server anyway: DATA then served /apps, /config, /home and /state from RAM over a disk that exists, and LOG blamed a loader that had named its partition. Each is now a `StartError::Partition` refusal of the start: - on a restart, init closes the role's ports, as for any start refused; - at boot, init says the role did not start and closes its ports, and the boot goes on without it; any other first start refused still panics init; - a log or boot role the loader named no partition for is refused the same way, so fsd is never started with neither its claim nor its GUID and panics by name if it is; its "the loader named no partition" answer goes. The guest test `fsd_claim_held` (Fast) holds DATA's claim across its server's restart. fsd's new actuator `--let-go-at-read ` syncs and lets its partition go at the first read of `path` and ends at that client's next request, so the guest takes the claim in between with no race against init. The guest's last open waits in the role's port until init answers: Gone when init refused the restart, an answer when a server runs without the partition. The host asserts init's "would not start again (… is already claimed)" line, fsd's two actuator lines, and no line saying DATA is in memory. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- src/build.rs | 1 + tests/common/partclaim.rs | 2 +- tests/common/storage.rs | 61 +++++++++++++++++ tests/fsdclaimcase/system.toml | 36 ++++++++++ .../toyos-rust-tests/src/bin/fs_claim_held.rs | 55 ++++++++++++++++ tests/toyos.rs | 6 ++ userland/fsd/src/main.rs | 63 ++++++++++++++---- userland/init/src/main.rs | 66 ++++++++++++++----- 8 files changed, 260 insertions(+), 30 deletions(-) create mode 100644 tests/fsdclaimcase/system.toml create mode 100644 tests/toyos-rust-tests/src/bin/fs_claim_held.rs diff --git a/src/build.rs b/src/build.rs index 6113107dff7..913ba575d73 100644 --- a/src/build.rs +++ b/src/build.rs @@ -3070,6 +3070,7 @@ mod tests { "tests/e1000leasecase/system.toml", "tests/e1000talkcase/system.toml", "tests/flrswapcase/system.toml", + "tests/fsdclaimcase/system.toml", "tests/fsdmountcase/system.toml", "tests/fsdrestartcase/system.toml", "tests/inspectcase/system.toml", diff --git a/tests/common/partclaim.rs b/tests/common/partclaim.rs index 6f3f2eb8d18..2b37a934e31 100644 --- a/tests/common/partclaim.rs +++ b/tests/common/partclaim.rs @@ -735,7 +735,7 @@ fn craft_disk(path: &Path, twin: &str) -> Result { } /// The designation on `data`: fsd formats a DATA only on this consent. -fn designate(device: &mut dyn gpt::DiskDevice, data: Span) -> Result<(), String> { +pub(super) fn designate(device: &mut dyn gpt::DiskDevice, data: Span) -> Result<(), String> { let mut stamp = [0u8; BLOCK as usize]; stamp[..bcachefs::DESIGNATION_MAGIC.len()].copy_from_slice(&bcachefs::DESIGNATION_MAGIC); let at = bcachefs::DESIGNATION_BLOCKS_OFFSET; diff --git a/tests/common/storage.rs b/tests/common/storage.rs index a2ef3ff1469..d97e84e93b0 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -751,6 +751,67 @@ pub fn fsd_end_at_mount( Ok(()) } +/// A partition claim held elsewhere refuses its file server's restart by name, +/// and the role's paths are then Gone, never served from memory. +/// +/// DATA is on a USB stick the kernel drives, so its server holds the +/// partition's claim. `tests/fsdclaimcase` arms `--let-go-at-read`, and +/// `test_rs_fs_claim_held` takes the claim the server let go, ends the server, +/// and holds the claim until init has answered the role's restart. +pub fn fsd_claim_held( + _test_config: &Path, + c_bins: &[(String, Vec)], + rust_bins: &[(String, Vec)], +) -> Result<(), String> { + /// Mirrored in `tests/toyos-rust-tests/src/bin/fs_claim_held.rs`. + const DATA: &str = "7E2B4C6D-8F1A-4B3C-9D5E-6F7A8B9C0D1E"; + const MIB: u64 = 1024 * 1024; + const LET_GO: &str = "fsd: --let-go-at-read: home/fsd_let_go: the partition is let go"; + const ENDED: &str = "fsd: --let-go-at-read: ending at its client's next request"; + const REFUSED: &str = "init: fsd data ended and would not start again ("; + const HELD: &str = "is already claimed); its ports are closed"; + + let stick = super::lane::dir().join("fsd-claim-held.img"); + let data = ("ToyOS data", 96 * MIB, toyos_gpt::Guid::TOYOS_DATA_TEXT, DATA, super::partclaim::ALIGNED); + let (mut device, spans) = super::partclaim::table(&stick, 100 * MIB, &[data])?; + super::partclaim::designate(&mut *device, spans[0])?; + device.flush().map_err(|e| format!("flush the stick: {e}"))?; + drop(device); + + let config = super::compile::repo_root().join("tests/fsdclaimcase"); + let mut qemu = QemuInstance::boot_with_options( + &config, + c_bins, + rust_bins, + BootOptions { profile: qemu::Profile::UsbDisk, usb_images: vec![stick.clone()], ..Default::default() }, + ); + let boot = qemu.boot_log().to_string(); + // The premise: DATA's first server holds the stick's partition. + if !boot.contains("fsd: block 0 designates this partition for ToyOS; formatting it") { + return Err(format!("fsd never formatted DATA off the stick, so no server held its claim:\n{boot}")); + } + let result = qemu.run_test("test_rs_fs_claim_held", Duration::from_secs(60)); + writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); + qemu.flush_stdin(); + let tail = qemu.drain_serial(Duration::from_secs(20)); + drop(qemu); + let log = format!("{boot}\n{}{}{tail}", result.before, result.serial); + if result.exit_code != Some(0) || !result.stdout.contains("fs_claim_held: PASS") { + return Err(format!("fs_claim_held guest failed:\n{}\nconsole:\n{log}", result.stdout)); + } + let console = super::serial::Serial::named("fsd_claim_held", log.as_str()); + console.must_say(LET_GO)?; + console.must_say(ENDED)?; + let Some(refused) = log.lines().find(|l| l.contains(REFUSED) && l.contains(HELD)) else { + return Err(format!("init never said DATA's restart was refused for the held claim:\n{log}")); + }; + console.must_not_say(IN_MEMORY)?; + console.must_be_clean()?; + let _ = std::fs::remove_file(&stick); + eprintln!(" [fsd] {}", refused.trim()); + Ok(()) +} + /// `/apps` and `/home` are two paths into one filesystem, judged off the device. /// /// The guest writes one file under each and shuts down; the host then finds diff --git a/tests/fsdclaimcase/system.toml b/tests/fsdclaimcase/system.toml new file mode 100644 index 00000000000..80edd6de4df --- /dev/null +++ b/tests/fsdclaimcase/system.toml @@ -0,0 +1,36 @@ +# The boot `fsd_claim_held` judges: DATA is on a USB stick the kernel drives, +# so its file server holds the partition's claim; `--let-go-at-read` lets it +# go at the guest's first read of `home/fsd_let_go`, the guest takes the +# claim, and the server's end then finds it held when init starts the role +# again. + +[boot] +start = ["logd", "blockd", "fsd", "test-runner"] + +[programs.logd] +service = true +syscap = ["logread"] + +# `device` because the guest mints DATA's claim, and `power` because +# `run shutdown` asks init through the `power` connector. +[programs.test-runner] +receives = ["power"] +syscap = ["device", "dup", "logread", "power"] + +[programs.toybox] +receives = ["power"] + +[symlinks] +"bin/shutdown" = "/system/bin/toybox" + +[programs.blockd] +service = true +restart = true +serves = ["block"] +devices = ["pci:1b36:0010"] + +[programs.fsd] +restart = true +roles = ["data", "log", "boot"] +receives = ["block"] +args = ["--let-go-at-read", "home/fsd_let_go"] diff --git a/tests/toyos-rust-tests/src/bin/fs_claim_held.rs b/tests/toyos-rust-tests/src/bin/fs_claim_held.rs new file mode 100644 index 00000000000..1638c15ed73 --- /dev/null +++ b/tests/toyos-rust-tests/src/bin/fs_claim_held.rs @@ -0,0 +1,55 @@ +//! DATA's partition claim held across its file server's restart. +//! +//! `tests/fsdclaimcase` arms DATA's server with `--let-go-at-read`: the first +//! read-only open of [`LET_GO`] lets its partition go and is refused, and this +//! client's next request ends the server. This binary takes the claim in +//! between, so init's restart of the role finds it held, and holds it until +//! init has answered: a new open waits in the role's port until init either +//! closes it, which answers Gone, or starts a server that answers it. The host +//! judges init's and fsd's lines (`tests/common/storage.rs`'s `fsd_claim_held`). + +use std::fs::File; +use std::io::ErrorKind; + +use toyos::endow::Endowments; +use toyos::syscap::SysCap; +use toyos::PartitionDev; +use toyos_abi::part::PartGuid; +use toyos_abi::syscall::SYSCAP_LABEL; + +/// Mirrored in `tests/common/storage.rs`: the DATA partition on the crafted +/// stick. +const DATA: &str = "7E2B4C6D-8F1A-4B3C-9D5E-6F7A8B9C0D1E"; +/// Mirrored in `tests/fsdclaimcase/system.toml`. +const LET_GO: &str = "/home/fsd_let_go"; +/// A name on DATA nothing makes: asked only to reach its server. +const AFTER: &str = "/home/fsd_claim_held"; + +fn main() { + let cap: SysCap = + Endowments::get().take(SYSCAP_LABEL).expect("the test estate is endowed a device-minting capability"); + let data = PartGuid::parse(DATA).expect("DATA is a GUID"); + + match File::open(LET_GO) { + Err(e) => println!("fs_claim_held: the server let its partition go, and the open was refused ({e})"), + Ok(_) => panic!("the open of {LET_GO} was answered: the server did not let its partition go"), + } + let held = cap + .claim_partition::(data) + .unwrap_or_else(|e| panic!("DATA's claim, which its server let go, was refused: {e:?}")); + println!("fs_claim_held: holding DATA's claim"); + + match File::open(AFTER) { + Err(e) => println!("fs_claim_held: the request the server ends under was refused ({e})"), + Ok(_) => panic!("the server answered the request it was armed to end under"), + } + match File::open(AFTER) { + Err(e) if e.kind() == ErrorKind::StaleNetworkFileHandle => { + println!("fs_claim_held: init did not start DATA's server again, and {AFTER} answers Gone ({e})") + } + Err(e) => panic!("{AFTER} was refused {e} ({:?}), not Gone", e.kind()), + Ok(_) => panic!("{AFTER} was answered while this process held DATA's claim: a server runs without it"), + } + drop(held); + println!("fs_claim_held: PASS"); +} diff --git a/tests/toyos.rs b/tests/toyos.rs index a1ee1497d61..fdb76d26556 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -945,6 +945,10 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // init starts it again and the boot reaches ready with the session home. // Body in `tests/common/storage.rs`. ("fsd_end_at_mount", Sched::Parallel, Tier::Fast), + // DATA's partition claim held by the guest across its server's restart: + // init refuses the restart by name and /home answers Gone. Body in + // `tests/common/storage.rs`. + ("fsd_claim_held", Sched::Parallel, Tier::Fast), // A same-length overwrite on /home, the guest's read held against the image. Body in `tests/common/storage.rs`. ("home_overwrite_reads_back", Sched::Parallel, Tier::Fast), // One filesystem under two paths: the guest writes under each of /apps and @@ -1661,6 +1665,7 @@ const CARRIES: &[(&str, &[&str])] = &[ ("home_overwrite_reads_back", &["test_rs_home_overwrite_zero"]), ("so_cache_refusals", &["test_rs_so_cache_policy"]), ("fsd_restart", &["test_rs_fs_client_bound", "test_rs_fs_restart"]), + ("fsd_claim_held", &["test_rs_fs_claim_held"]), ("esp_filesystem", &["test_rs_esp_files"]), ("log_flush_retry", &["test_rs_esp_files"]), ("fat_backing_revoked", &["test_rs_fat_backing_revoked"]), @@ -11297,6 +11302,7 @@ fn run_machine_test( "so_cache_refusals" => storage::so_cache_refusals(test_config, c_bins, rust_bins), "fsd_restart" => storage::fsd_restart(test_config, c_bins, rust_bins), "fsd_end_at_mount" => storage::fsd_end_at_mount(test_config, c_bins, rust_bins), + "fsd_claim_held" => storage::fsd_claim_held(test_config, c_bins, rust_bins), "home_overwrite_reads_back" => { storage::home_overwrite_reads_back(test_config, c_bins, rust_bins) } diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index a168a6885ea..d4e1683e067 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -97,6 +97,14 @@ const END_AT_READ: &str = "--end-at-read"; /// boot config's `args`. const END_AT_MOUNT: &str = "--end-at-mount"; +/// `--let-go-at-read `: the first open of `path` on the volume to read +/// only, this boot, syncs the volume, lets its partition go (its claim closed, +/// the volume absent from then on) and is refused; that client's next request +/// ends this process before it is answered. So a test can take the partition's +/// claim before init starts this role again. Once a boot, as [`END_AT_READ`] +/// is. Armed by nothing but a boot config's `args`. +const LET_GO_AT_READ: &str = "--let-go-at-read"; + /// How long [`END_AT_MOUNT`] waits for the connection it ends under. const END_AT_MOUNT_WAIT: Duration = Duration::from_secs(60); @@ -185,7 +193,8 @@ fn main() { let role = args.get(1).and_then(|r| Role::parse(r)).unwrap_or_else(|| { panic!("fsd: started with {args:?}; the first argument is a role: data, log or boot") }); - let (mut guid, mut end_on, mut end_at_read, mut end_at_mount) = (None, None, Vec::new(), None); + let (mut guid, mut end_on, mut end_at_read, mut end_at_mount, mut let_go_at_read) = + (None, None, Vec::new(), None, None); let mut rest = args.iter().skip(2); while let Some(arg) = rest.next() { match arg.as_str() { @@ -193,6 +202,10 @@ fn main() { END_AT_READ => { end_at_read.push(rest.next().unwrap_or_else(|| panic!("fsd: {END_AT_READ} takes a path")).clone()) } + LET_GO_AT_READ => { + let_go_at_read = + Some(rest.next().unwrap_or_else(|| panic!("fsd: {LET_GO_AT_READ} takes a path")).clone()) + } END_AT_MOUNT => { end_at_mount = Some( rest.next().and_then(|r| Role::parse(r)).unwrap_or_else(|| panic!("fsd: {END_AT_MOUNT} takes a role")), @@ -226,6 +239,8 @@ fn main() { writeback: WriteBack::default(), end_on, end_at_read, + let_go_at_read, + let_go_client: None, probe: Poller::new(caps_len), scratch: Vec::new(), } @@ -315,23 +330,27 @@ fn open_volume(role: Role, guid: Option<&str>, roots: &[&str]) -> Box { let writable = role == Role::Log; - let mounted = if let Some(disk) = claimed() { - Some(fat_on(disk, writable)) - } else { - guid.and_then(toyos_abi::part::PartGuid::parse).and_then(|g| served(g.0)).map(|disk| fat_on(disk, writable)) + let mounted = match claimed() { + Some(disk) => Ok(fat_on(disk, writable)), + // init starts this role with its partition's claim or its GUID, never neither. + None => { + let guid = guid.unwrap_or_else(|| { + panic!("fsd: the {role:?} server was started with neither its partition's claim nor its GUID") + }); + let parsed = toyos_abi::part::PartGuid::parse(guid) + .unwrap_or_else(|| panic!("fsd: {guid} is no partition GUID")); + served(parsed.0).map(|disk| fat_on(disk, writable)).ok_or(guid) + } }; match mounted { - Some(Ok(volume)) => volume, - Some(Err(why)) => { + Ok(Ok(volume)) => volume, + Ok(Err(why)) => { println!("fsd: the {role:?} partition does not mount: {why}"); Box::new(Absent::new(roots, why)) } - None => Box::new(Absent::new( + Err(guid) => Box::new(Absent::new( roots, - match guid { - Some(guid) => format!("the {role:?} partition {guid} is on no disk this server reaches"), - None => format!("the loader named no {role:?} partition"), - }, + format!("the {role:?} partition {guid} is on no disk this server reaches"), )), } } @@ -374,6 +393,10 @@ struct Server { end_on: Option, /// [`END_AT_READ`]'s paths on the volume. end_at_read: Vec, + /// [`LET_GO_AT_READ`]'s path on the volume. + let_go_at_read: Option, + /// The client whose next request ends this server, once [`LET_GO_AT_READ`] fired. + let_go_client: Option, /// Asks an acceptor whether a connection waits, before [`Server::accept`] /// takes it. probe: Poller, @@ -686,6 +709,10 @@ impl Server { } fn serve_one(&mut self, id: u64, op: u32, r: Request) -> Result { + if self.let_go_client == Some(id) { + println!("fsd: {LET_GO_AT_READ}: ending at its client's next request"); + std::process::exit(1); + } let changes = matches!(op, WRITE | TRUNCATE | MKDIR | RMDIR | UNLINK | RENAME | SYMLINK | STREAM) || (op == OPEN && r.flags & (O_WRITE | O_APPEND | O_CREATE | O_TRUNCATE | O_CREATE_NEW) != 0); if changes && !self.volume.writable() { @@ -706,6 +733,18 @@ impl Server { println!("fsd: {END_AT_READ}: ending before the first read of {path} is answered"); std::process::exit(1); } + if reads_only + && self.let_go_at_read.as_deref() == Some(path.as_str()) + && first_this_boot(&format!("let-go-at-read-{path}")) + { + let unwritten = self.volume.sync()?; + assert!(unwritten.is_empty(), "fsd: {LET_GO_AT_READ}: {unwritten:?} did not sync"); + let roots: Vec<&str> = self.caps.iter().map(|c| c.root.as_str()).collect(); + self.volume = Box::new(Absent::new(&roots, format!("{LET_GO_AT_READ} let the partition go"))); + self.let_go_client = Some(id); + println!("fsd: {LET_GO_AT_READ}: {path}: the partition is let go"); + return Err(SyscallError::NotFound); + } if self.clients[&id].fids.len() >= MAX_FIDS { return Err(SyscallError::ResourceExhausted); } diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index 2b24bb235de..035be2c9fa4 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -598,7 +598,7 @@ impl<'a> Service<'a> { connectors: &BTreeMap<&str, Connector>, dirs: &BTreeMap, log: &mut Log, - ) -> std::io::Result { + ) -> Result { // **Checked again at every start, not only on arrival**: the installed // file lives in an ambient directory, so what init verified when it // wrote it is not a claim about what is there now. This narrows the @@ -617,7 +617,7 @@ impl<'a> Service<'a> { 0 => Served::Keep(&kept.acceptors), _ => Served::Restart { acceptors: &kept.acceptors, owed }, }; - let storage = storage_endowment(self.program, self.role, syscap).map_err(std::io::Error::other)?; + let storage = storage_endowment(self.program, self.role, syscap)?; let (child, devices) = start( Command::new(path), self.program, @@ -662,6 +662,29 @@ impl<'a> Service<'a> { } } +/// Why a service did not start. +enum StartError { + /// The partition its file-server role is on was refused, so the role is + /// absent: its ports close and nothing serves its paths from memory. + Partition(String), + Other(std::io::Error), +} + +impl From for StartError { + fn from(e: std::io::Error) -> Self { + Self::Other(e) + } +} + +impl std::fmt::Display for StartError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Partition(why) => f.write_str(why), + Self::Other(e) => e.fmt(f), + } + } +} + /// Wait for one start of a service to end, and close its ports if nothing was /// expecting it to, or wake the loop to start it again if its row says so. /// @@ -767,8 +790,13 @@ impl<'a> Init<'a> { let mut service = Service::new(program, role, kept, wake); let started = service.spawn(&program.path, &[], system, self.syscap, &self.connectors, &self.dirs, &mut self.log); - if let Err(e) = started { - panic!("init: cannot start {}: {e}", program.name); + match started { + Ok(_) => {} + Err(StartError::Partition(why)) => { + service.kept.lock().expect("init: a service's state is poisoned").acceptors.clear(); + say!("init: {} did not start ({why}); its ports are closed this boot", service.label()); + } + Err(e) => panic!("init: cannot start {}: {e}", program.name), } self.services.push(service); } @@ -2215,15 +2243,18 @@ fn guid_text(guid: [u8; 16]) -> String { /// session on. A file server is told its role, and gets its partition as a /// claim when a disk the kernel drives carries it — the stick, until usbd /// serves it — and otherwise the partition's GUID, which it opens through the -/// block service; DATA it finds by type there itself. -fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> Result { +/// block service; DATA it finds by type there itself. A log or boot role the +/// loader named no partition for, a claim the kernel refuses, and a role two +/// partitions on the kernel's disks carry each refuse the start: the role is +/// then absent, never served from memory. +fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> Result { use toyos_abi::inventory::{Record, Role}; let mut storage = Storage::default(); let is_block = program.serves.iter().any(|s| s == toyos_blockring::PORT); if !is_block && role.is_none() { return Ok(storage); } - let records = inventory(syscap)?; + let records = inventory(syscap).map_err(|why| StartError::Other(std::io::Error::other(why)))?; if is_block { if let Some(root) = loaded(&records, Role::Root) { storage.args.extend(["--running".to_string(), guid_text(root)]); @@ -2245,10 +2276,10 @@ fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> "data" => (on_kernel_disk(&|p| p.type_guid == toyos_gpt::Guid::TOYOS_DATA.0), None), "log" | "boot" => { let which = if role == "log" { Role::Log } else { Role::Boot }; - match loaded(&records, which) { - Some(guid) => (on_kernel_disk(&|p| p.unique_guid == guid), Some(guid)), - None => (Vec::new(), None), - } + let Some(guid) = loaded(&records, which) else { + return Err(StartError::Partition(format!("the loader named no `{role}` partition"))); + }; + (on_kernel_disk(&|p| p.unique_guid == guid), Some(guid)) } other => panic!("init: `{other}` is no role; the build refuses it"), }; @@ -2259,7 +2290,7 @@ fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> let name = request.write_name(&mut buf).to_string(); match syscap.claim_partition::(toyos_abi::part::PartGuid(*guid)) { Ok(claim) => storage.claims.push((format!("{DEV_PREFIX}{name}"), claim)), - Err(e) => say!("init: {}: {}", program.name, refused(&name, e)), + Err(e) => return Err(StartError::Partition(refused(&name, e))), } } [] => { @@ -2267,11 +2298,12 @@ fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> storage.args.push(guid_text(guid)); } } - many => say!( - "init: {}: the kernel's disks carry {} partitions for the `{role}` role, and a role is one", - program.name, - many.len() - ), + many => { + return Err(StartError::Partition(format!( + "the kernel's disks carry {} partitions for the `{role}` role, and a role is one", + many.len() + ))) + } } Ok(storage) } From fae0a5d3e89d727e650f431e26f2066d56a45207 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 11:29:40 +0200 Subject: [PATCH 41/54] issues: a worktree's bootstrap cache outlives a compiler rebuilt at the same version Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-a-compiler-rebuilt-at-the-same-version.md | 30 +++++++++++++++++++ 1 file changed, 30 insertions(+) create mode 100644 issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md diff --git a/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md b/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md new file mode 100644 index 00000000000..2f1015a7eef --- /dev/null +++ b/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md @@ -0,0 +1,30 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# A worktree's bootstrap cache outlives a compiler rebuilt at the same version + +A linked worktree builds its std sysroot through its own fork checkout's +bootstrap, whose cargo target directory is +`rust/build/toyos-std/bootstrap`, compiled by the primary's compiler named in +`bootstrap.toml`'s `rustc`. Every build of that compiler answers `rustc +1.99.0-dev`, `commit-hash: unknown`. Cargo keys freshness on that answer, so +once the primary's compiler is rebuilt, +the cached dependencies are reused, and the new +compiler cannot load them. Bootstrap then fails to compile with `error[E0463]: +can't find crate for serde` (and `clap`, `xz2`), with nothing compiled before +it, and the sysroot build panics at `src/toolchain.rs` with "std did not +compile". + +**Evidence:** PR #536's worktree at 1f969973, sysroot key 8ec6e31551091ba8: +`cargo run -- --build-only` failed the same way twice. Its +`libserde-*.rmeta` dated from 2026-09-27; the compiler directory +`compilers/a04a68b92a50e478` from 2026-09-28 09:07. The same sources built +cleanly into a fresh target directory. With the stale directory moved aside, +the next `--build-only` succeeded. + +**Exit condition:** a worktree's bootstrap cache is keyed on, or cleared with, +the compiler that builds it, so a rebuilt compiler never meets artifacts from +the one it replaced. From 5235eda7bebf117d7f05e5fde4c4514752d6112a Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 12:44:52 +0200 Subject: [PATCH 42/54] fs_claim_held is driven, not shared; its judge asserts Gone, not a console line The binary needs DATA on a stick the kernel drives and a server armed with --let-go-at-read, which only tests/fsdclaimcase boots, so it goes on RUST_SKIP: suite_split named it driven and shared, and the Fast tier ran it on the shared boot, where it exited 101. fsd_claim_held's host check that the console never said DATA was in memory could not red first: the guest exits non-zero unless the role's path answers Gone, and any server for the role, in memory or not, answers otherwise. Under the say-and-continue control the restarted server took the NVMe's DATA through blockd, so the line was never printed either. The check is deleted and the doc says what is judged. The issue on a worktree's bootstrap cache is deleted: #573 fixed and closed the same defect under another name. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-a-compiler-rebuilt-at-the-same-version.md | 30 ------------------- tests/common/storage.rs | 3 +- tests/toyos.rs | 4 +++ 3 files changed, 5 insertions(+), 32 deletions(-) delete mode 100644 issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md diff --git a/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md b/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md deleted file mode 100644 index 2f1015a7eef..00000000000 --- a/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md +++ /dev/null @@ -1,30 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-09-28 ---- - -# A worktree's bootstrap cache outlives a compiler rebuilt at the same version - -A linked worktree builds its std sysroot through its own fork checkout's -bootstrap, whose cargo target directory is -`rust/build/toyos-std/bootstrap`, compiled by the primary's compiler named in -`bootstrap.toml`'s `rustc`. Every build of that compiler answers `rustc -1.99.0-dev`, `commit-hash: unknown`. Cargo keys freshness on that answer, so -once the primary's compiler is rebuilt, -the cached dependencies are reused, and the new -compiler cannot load them. Bootstrap then fails to compile with `error[E0463]: -can't find crate for serde` (and `clap`, `xz2`), with nothing compiled before -it, and the sysroot build panics at `src/toolchain.rs` with "std did not -compile". - -**Evidence:** PR #536's worktree at 1f969973, sysroot key 8ec6e31551091ba8: -`cargo run -- --build-only` failed the same way twice. Its -`libserde-*.rmeta` dated from 2026-09-27; the compiler directory -`compilers/a04a68b92a50e478` from 2026-09-28 09:07. The same sources built -cleanly into a fresh target directory. With the stale directory moved aside, -the next `--build-only` succeeded. - -**Exit condition:** a worktree's bootstrap cache is keyed on, or cleared with, -the compiler that builds it, so a rebuilt compiler never meets artifacts from -the one it replaced. diff --git a/tests/common/storage.rs b/tests/common/storage.rs index d97e84e93b0..105e2020b8b 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -752,7 +752,7 @@ pub fn fsd_end_at_mount( } /// A partition claim held elsewhere refuses its file server's restart by name, -/// and the role's paths are then Gone, never served from memory. +/// and the role's paths are then Gone: no server answers them. /// /// DATA is on a USB stick the kernel drives, so its server holds the /// partition's claim. `tests/fsdclaimcase` arms `--let-go-at-read`, and @@ -805,7 +805,6 @@ pub fn fsd_claim_held( let Some(refused) = log.lines().find(|l| l.contains(REFUSED) && l.contains(HELD)) else { return Err(format!("init never said DATA's restart was refused for the held claim:\n{log}")); }; - console.must_not_say(IN_MEMORY)?; console.must_be_clean()?; let _ = std::fs::remove_file(&stick); eprintln!(" [fsd] {}", refused.trim()); diff --git a/tests/toyos.rs b/tests/toyos.rs index fdb76d26556..b06c08f3026 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -414,6 +414,10 @@ const RUST_SKIP: &[&str] = &[ // Holds every DATA client slot while it runs, so no other program may // need DATA on its boot. `fsd_restart` runs it. "fs_client_bound", + // Needs DATA on a stick the kernel drives and a server armed with + // `--let-go-at-read`, which only `tests/fsdclaimcase` boots. + // `fsd_claim_held` runs it there. + "fs_claim_held", // Needs a boot of its own for the readback it is judged against; `home_overwrite_reads_back` runs it. "home_overwrite_zero", // Needs a boot where the DATA volume is ours and absent; on the shared From 06c6195f1c1b8d8e33032e6283e551ead3eebb12 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 14:25:44 +0200 Subject: [PATCH 43/54] A refused NVMe claim and two DATA partitions are refused by name, never memory A controller blockd's row names that is on the machine and whose claim the kernel refused was served as no controller at all: blockd said none was on the machine, fsd said the machine had no DATA partition, and /apps, /config, /home and /state went into memory on every machine with no DMAR. init now tells a block service which of its row's claims were refused with anything but NotFound (`--claim-refused `); blockd serves Drive::ClaimRefused, whose listing and opens are refused with the new wire word ClaimRefused, and fsd serves DATA absent by name. DATA's partition was decided three ways: two on the kernel's disks refused the start, two on blockd's went to memory, and one on each took the kernel's without asking blockd. init now claims every partition of a role the kernel's disks carry, and fsd::data::find counts those claims and blockd's TOYOS-DATA listing together once: none is memory, one is served, two or more are refused by name and DATA is absent, and a listing refused leaves the count unknown, so DATA is absent then too. A log or boot claim is by the loader's GUID, which the kernel refuses when two partitions carry it, so that arm is unchanged. iommu_virtio_platform's no-unit arm, which boots tests/netcase with its blockd row, accepts 00:02.0's refusal by name, and asserts init's, blockd's and fsd's lines and that the in-memory line is never said. fsd_two_data boots DATA on a stick and on NVMe and asserts the refusal, the absent volume, no format and the stick unchanged byte for byte. fsd_claim_held and the three partclaim boots, which stage DATA on a stick, get an NVMe disk with no table so the machine has one DATA. The two "plus 16" restatements of the kernel's headroom in heap_ceiling.rs are deleted. The issue on a worktree's bootstrap cache is restored: the pull request that fixes it is not merged. A foreign DATA partition still answered with memory is filed. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-a-compiler-rebuilt-at-the-same-version.md | 30 +++++++++ ...-data-partition-is-answered-with-memory.md | 23 +++++++ tests/common/faults.rs | 15 +++-- tests/common/iommu.rs | 22 ++++-- tests/common/partclaim.rs | 16 ++++- tests/common/storage.rs | 61 ++++++++++++++++- .../toyos-rust-tests/src/bin/heap_ceiling.rs | 5 +- tests/toyos.rs | 5 ++ toyos-blockring/src/wire.rs | 15 ++++- userland/blockd/src/main.rs | 45 ++++++++++--- userland/fsd/src/data.rs | 54 +++++++++++++++ userland/fsd/src/main.rs | 67 +++++++++++-------- userland/init/src/main.rs | 53 +++++++-------- 13 files changed, 328 insertions(+), 83 deletions(-) create mode 100644 issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md create mode 100644 issues/filesystem/a-foreign-data-partition-is-answered-with-memory.md diff --git a/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md b/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md new file mode 100644 index 00000000000..2f1015a7eef --- /dev/null +++ b/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md @@ -0,0 +1,30 @@ +--- +status: open +kind: tooling +opened: 2026-09-28 +--- + +# A worktree's bootstrap cache outlives a compiler rebuilt at the same version + +A linked worktree builds its std sysroot through its own fork checkout's +bootstrap, whose cargo target directory is +`rust/build/toyos-std/bootstrap`, compiled by the primary's compiler named in +`bootstrap.toml`'s `rustc`. Every build of that compiler answers `rustc +1.99.0-dev`, `commit-hash: unknown`. Cargo keys freshness on that answer, so +once the primary's compiler is rebuilt, +the cached dependencies are reused, and the new +compiler cannot load them. Bootstrap then fails to compile with `error[E0463]: +can't find crate for serde` (and `clap`, `xz2`), with nothing compiled before +it, and the sysroot build panics at `src/toolchain.rs` with "std did not +compile". + +**Evidence:** PR #536's worktree at 1f969973, sysroot key 8ec6e31551091ba8: +`cargo run -- --build-only` failed the same way twice. Its +`libserde-*.rmeta` dated from 2026-09-27; the compiler directory +`compilers/a04a68b92a50e478` from 2026-09-28 09:07. The same sources built +cleanly into a fresh target directory. With the stale directory moved aside, +the next `--build-only` succeeded. + +**Exit condition:** a worktree's bootstrap cache is keyed on, or cleared with, +the compiler that builds it, so a rebuilt compiler never meets artifacts from +the one it replaced. diff --git a/issues/filesystem/a-foreign-data-partition-is-answered-with-memory.md b/issues/filesystem/a-foreign-data-partition-is-answered-with-memory.md new file mode 100644 index 00000000000..327ddb4a9a0 --- /dev/null +++ b/issues/filesystem/a-foreign-data-partition-is-answered-with-memory.md @@ -0,0 +1,23 @@ +--- +status: open +kind: defect +opened: 2026-09-28 +--- + +# A foreign DATA partition is answered with memory + +`userland/fsd/src/main.rs`'s `data_on` answers `Probed::Foreign` — a +TOYOS-DATA partition that holds no volume of ours and no designation stamp — +with `ram()`: `/apps`, `/config`, `/home` and `/state` are served from memory, +and everything written under them is lost at the next reboot. That is a disk +that is there and cannot be used, answered as though the machine had none, +which is the harm `fsd::absent::Absent` exists to refuse; a refused NVMe +claim and two DATA partitions are refused by name for the same reason. + +`foreign_disk_untouched` (`tests/common/storage.rs`) boots this arm and +asserts the partition is not written and the `no volume of ours` line, and +nothing about what stands in for it. + +**Exit condition:** a foreign DATA partition serves DATA absent by name, and +`foreign_disk_untouched` asserts fsd's absent line and that the in-memory line +is never said. diff --git a/tests/common/faults.rs b/tests/common/faults.rs index 04bc78a0d4c..74b1d1e780c 100644 --- a/tests/common/faults.rs +++ b/tests/common/faults.rs @@ -265,6 +265,7 @@ pub fn virtio_net_no_msix() -> Result<(), String> { &log, super::https::VIRTIO.claims, "neither its MSI-X nor its MSI could be armed", + &[], )?; // And it reached userland rather than stopping at a log line. exited?; @@ -296,7 +297,7 @@ pub fn claim_caps_truncated() -> Result<(), String> { // Refused by the reason that is true of it: what the list holds past that // link was never read — not "it has no table". - refused_claim(&log, bench.claims, "its capability list ends at a link the PCI spec forbids")?; + refused_claim(&log, bench.claims, "its capability list ends at a link the PCI spec forbids", &[])?; // And it reached userland rather than stopping at a log line. exited?; // And the machine is otherwise whole: one claim refused costs networking @@ -398,18 +399,20 @@ fn netd_answered(mut qemu: QemuInstance) -> (Serial, Result<(), String>) { /// **The claim on [`CLAIMED_AT`] was refused for `why`, and the refusal spent /// nothing**: no BAR of that function moved, neither of its two message /// mechanisms is armed, `claims` reached no holder, and init said so in the -/// boot config's own spelling. +/// boot config's own spelling. `beside` is every other function this machine +/// refuses, each judged by its own caller. /// /// The three arms that refuse a claim read this one judge, so a kernel that /// answered a refusal by logging it and handing the function over anyway is red /// wherever the refusal is reached. `slot_space` put back below `place_bars` /// reds on the two unspent lines. -pub fn refused_claim(log: &Serial, claims: &str, why: &str) -> Result<(), String> { +pub fn refused_claim(log: &Serial, claims: &str, why: &str, beside: &[&str]) -> Result<(), String> { let refused = functions_named(log, "NOT HANDED OVER")?; - if refused.is_empty() || refused.iter().any(|at| *at != CLAIMED_AT) { + let others: std::collections::BTreeSet<&str> = refused.iter().copied().filter(|at| *at != CLAIMED_AT).collect(); + if !refused.contains(&CLAIMED_AT) || others != beside.iter().copied().collect() { return Err(format!( - "the claim this judges is the one on {CLAIMED_AT}; this console refused \ - {refused:?}:\n{}", + "the claim this judges is the one on {CLAIMED_AT}, beside {beside:?}; this console \ + refused {refused:?}:\n{}", log.text() )); } diff --git a/tests/common/iommu.rs b/tests/common/iommu.rs index f8a5e6eaded..89d1f5e4b30 100644 --- a/tests/common/iommu.rs +++ b/tests/common/iommu.rs @@ -778,6 +778,10 @@ pub fn iommu_virtio_platform( declining_is_not_free(test_config, c_bins, rust_bins) } +/// The slot QEMU's `-device` order puts `tests/netcase`'s NVMe controller on, +/// the one its blockd row claims. +const NVME_AT: &str = "00:02.0"; + /// **A machine with no unit hands no function to a process**, and says so /// three times over. /// @@ -787,17 +791,25 @@ pub fn iommu_virtio_platform( /// physical address is an arbitrary read and write over all of memory. So the /// kernel refuses the claim by name, init says which device it could not mint, /// and netd exits rather than driving anything — and the machine finishes -/// booting, which is the half a refusal that panicked would fail. +/// booting, which is the half a refusal that panicked would fail. The NVMe +/// controller is refused the same, and DATA with it by name: a disk that is +/// there and cannot be used is never answered with memory. fn no_unit_is_no_claim(log: &Serial) -> Result<(), String> { + const NO_DOMAIN: &str = "it would have no address space of its own"; // The same judge the two arms in `faults` read, so a refusal that spent // something is red wherever it is reached. netd's own exit is the third // saying, and is not read here: it speaks after the ready marker this // capture ends at. - super::faults::refused_claim( - log, - super::https::VIRTIO.claims, - "it would have no address space of its own", + super::faults::refused_claim(log, super::https::VIRTIO.claims, NO_DOMAIN, &[NVME_AT])?; + log.must_say(&format!("pcidev: PCI {NVME_AT} NOT HANDED OVER — {NO_DOMAIN}"))?; + log.must_say("init: blockd: pci:1b36:0010 is on this machine and could not be handed over")?; + log.must_say( + "blockd: NOT SERVING — pci:1b36:0010 is on this machine and the kernel refused this service its claim", + )?; + log.must_say( + "fsd: the block service would not list its partitions (Refused(ClaimRefused)); DATA is absent this boot", )?; + log.must_not_say(super::storage::IN_MEMORY)?; // And this machine handed *nothing* over, which is more than the claim's // own refusal says: with no unit there is no function any process could be // given an address space for. diff --git a/tests/common/partclaim.rs b/tests/common/partclaim.rs index 2b37a934e31..ab143314eab 100644 --- a/tests/common/partclaim.rs +++ b/tests/common/partclaim.rs @@ -25,7 +25,7 @@ //! here, where the table and the `system.toml` naming it are both written. use std::io::{Read, Seek, SeekFrom, Write}; -use std::path::Path; +use std::path::{Path, PathBuf}; use std::time::Duration; use super::qemu::{self, BootOptions, QemuInstance, Staged}; @@ -167,6 +167,7 @@ pub fn partition_claim( profile: qemu::Profile::UsbDisk, boot_image: Some(Staged::Pristine(boot_image.clone())), usb_images: vec![crafted.clone()], + nvme_image: Some(tableless_nvme("partclaim-nvme.img")?), ..Default::default() }, ); @@ -255,6 +256,7 @@ pub fn partition_claim_gives_up( BootOptions { profile: qemu::Profile::UsbDisk, usb_images: vec![crafted.clone()], + nvme_image: Some(tableless_nvme("partclaim-nvme.img")?), kernel_params: params, ..Default::default() }, @@ -301,6 +303,7 @@ fn root_withheld( profile: qemu::Profile::UsbDisk, boot_image: Some(Staged::Pristine(image.clone())), usb_images: vec![crafted.to_path_buf()], + nvme_image: Some(tableless_nvme("partclaim-nvme.img")?), kernel_params: PARAMS, ..Default::default() }, @@ -734,6 +737,17 @@ fn craft_disk(path: &Path, twin: &str) -> Result { Ok(layout) } +/// An NVMe disk with no partition table, named `name` in the lane: beside a +/// stick carrying DATA, the machine's one DATA partition is the stick's, where +/// the lane's blank NVMe image would carry a second and DATA be refused. +pub(super) fn tableless_nvme(name: &str) -> Result { + let path = super::lane::dir().join(name); + std::fs::File::create(&path) + .and_then(|file| file.set_len(qemu::NVME_SMALL)) + .map_err(|e| format!("make {}: {e}", path.display()))?; + Ok(path) +} + /// The designation on `data`: fsd formats a DATA only on this consent. pub(super) fn designate(device: &mut dyn gpt::DiskDevice, data: Span) -> Result<(), String> { let mut stamp = [0u8; BLOCK as usize]; diff --git a/tests/common/storage.rs b/tests/common/storage.rs index 105e2020b8b..85c9e39d798 100644 --- a/tests/common/storage.rs +++ b/tests/common/storage.rs @@ -755,7 +755,8 @@ pub fn fsd_end_at_mount( /// and the role's paths are then Gone: no server answers them. /// /// DATA is on a USB stick the kernel drives, so its server holds the -/// partition's claim. `tests/fsdclaimcase` arms `--let-go-at-read`, and +/// partition's claim, and the NVMe disk carries no table, so the stick's is +/// the machine's one DATA. `tests/fsdclaimcase` arms `--let-go-at-read`, and /// `test_rs_fs_claim_held` takes the claim the server let go, ends the server, /// and holds the claim until init has answered the role's restart. pub fn fsd_claim_held( @@ -783,7 +784,12 @@ pub fn fsd_claim_held( &config, c_bins, rust_bins, - BootOptions { profile: qemu::Profile::UsbDisk, usb_images: vec![stick.clone()], ..Default::default() }, + BootOptions { + profile: qemu::Profile::UsbDisk, + usb_images: vec![stick.clone()], + nvme_image: Some(super::partclaim::tableless_nvme("fsd-claim-held-nvme.img")?), + ..Default::default() + }, ); let boot = qemu.boot_log().to_string(); // The premise: DATA's first server holds the stick's partition. @@ -811,6 +817,57 @@ pub fn fsd_claim_held( Ok(()) } +/// Two DATA partitions, one on a stick the kernel drives and one on the NVMe +/// disk blockd serves, are refused by name and never guessed between: DATA is +/// absent, nothing stands in from memory, and neither is formatted — the +/// stick is held byte for byte against what it carried before the boot. +pub fn fsd_two_data( + test_config: &Path, + c_bins: &[(String, Vec)], + rust_bins: &[(String, Vec)], +) -> Result<(), String> { + const DATA: &str = "5A1C0E2B-3D4F-4A6B-8C9D-0E1F2A3B4C5D"; + const MIB: u64 = 1024 * 1024; + const REFUSED: &str = "fsd: this machine has 2 DATA partitions, 1 on the kernel's disks and 1 the block \ + service serves, and a volume is one; DATA is absent this boot"; + + let stick = super::lane::dir().join("fsd-two-data.img"); + let data = ("ToyOS data", 96 * MIB, toyos_gpt::Guid::TOYOS_DATA_TEXT, DATA, super::partclaim::ALIGNED); + let (mut device, spans) = super::partclaim::table(&stick, 100 * MIB, &[data])?; + super::partclaim::designate(&mut *device, spans[0])?; + device.flush().map_err(|e| format!("flush the stick: {e}"))?; + drop(device); + let before = whole_device(&stick); + + // The lane's blank NVMe image is the second: a designated DATA partition. + let mut qemu = QemuInstance::boot_with_options( + test_config, + c_bins, + rust_bins, + BootOptions { profile: qemu::Profile::UsbDisk, usb_images: vec![stick.clone()], ..Default::default() }, + ); + let boot = qemu.boot_log().to_string(); + // Shut down rather than kill: a format sitting in a server's cache reaches + // the device at the stop's sync, and the stick is judged after it. + writeln!(qemu.stdin_mut(), "run shutdown").expect("write to QEMU stdin"); + qemu.flush_stdin(); + let tail = qemu.drain_serial(Duration::from_secs(20)); + drop(qemu); + let log = format!("{boot}\n{tail}"); + let console = super::serial::Serial::named("fsd_two_data", log.as_str()); + console.must_say(REFUSED)?; + data_absent(&log)?; + console.must_not_say(IN_MEMORY)?; + console.must_not_say("formatting it")?; + console.must_be_clean()?; + if let Some(diff) = first_difference(&before, &whole_device(&stick)) { + return Err(format!("a DATA partition of two was written: {diff}\n{log}")); + } + let _ = std::fs::remove_file(&stick); + eprintln!(" [fsd] two DATA partitions, one per source, refused by name; the stick untouched"); + Ok(()) +} + /// `/apps` and `/home` are two paths into one filesystem, judged off the device. /// /// The guest writes one file under each and shuts down; the host then finds diff --git a/tests/toyos-rust-tests/src/bin/heap_ceiling.rs b/tests/toyos-rust-tests/src/bin/heap_ceiling.rs index 4069d7bca73..b35b6a44fc9 100644 --- a/tests/toyos-rust-tests/src/bin/heap_ceiling.rs +++ b/tests/toyos-rust-tests/src/bin/heap_ceiling.rs @@ -8,8 +8,7 @@ // `SYS_DEBUG` actions a `test-actuators` kernel provides. The first two take // one kernel heap allocation each and release it again — at // `mm::MAX_HEAP_ALLOC`, and at `MAX_HEAP_ALLOC` with 4096-byte alignment; the -// last lowers `SYS_SYSINFO`'s thread bound to the machine's live threads plus -// 16. +// last lowers `SYS_SYSINFO`'s thread bound to the machine's live threads. use toyos_abi::syscall::debug_action::{ HEAP_AT_CEILING, HEAP_AT_CEILING_PAGE_ALIGNED, LOWER_SYSINFO_BOUND, }; @@ -32,7 +31,7 @@ fn main() { /// ask the heap for more than `MAX_HEAP_ALLOC` and trip the assert three /// functions above — from any process, with no privilege. /// -/// [`LOWER_SYSINFO_BOUND`] puts the machine's live threads plus 16 in +/// [`LOWER_SYSINFO_BOUND`] puts the machine's live threads in /// `MAX_SYSINFO_THREADS`'s place, because 65,536 threads is 8 GiB of kernel /// stacks and no guest can make them. The count, the comparison and the /// refusal are the shipped ones. diff --git a/tests/toyos.rs b/tests/toyos.rs index 7f9101cc372..8d86316ee09 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -947,6 +947,10 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // init refuses the restart by name and /home answers Gone. Body in // `tests/common/storage.rs`. ("fsd_claim_held", Sched::Parallel, Tier::Fast), + // DATA on a stick and on NVMe, one partition per source: fsd refuses both + // by name, serves DATA absent, and the stick is untouched. Body in + // `tests/common/storage.rs`. + ("fsd_two_data", Sched::Parallel, Tier::Fast), // A same-length overwrite on /home, the guest's read held against the image. Body in `tests/common/storage.rs`. ("home_overwrite_reads_back", Sched::Parallel, Tier::Fast), // One filesystem under two paths: the guest writes under each of /apps and @@ -11301,6 +11305,7 @@ fn run_machine_test( "fsd_restart" => storage::fsd_restart(test_config, c_bins, rust_bins), "fsd_end_at_mount" => storage::fsd_end_at_mount(test_config, c_bins, rust_bins), "fsd_claim_held" => storage::fsd_claim_held(test_config, c_bins, rust_bins), + "fsd_two_data" => storage::fsd_two_data(test_config, c_bins, rust_bins), "home_overwrite_reads_back" => { storage::home_overwrite_reads_back(test_config, c_bins, rust_bins) } diff --git a/toyos-blockring/src/wire.rs b/toyos-blockring/src/wire.rs index bc3e53c4624..489444b9c06 100644 --- a/toyos-blockring/src/wire.rs +++ b/toyos-blockring/src/wire.rs @@ -100,6 +100,10 @@ pub enum Refusal { Malformed, /// The service is holding as many sessions as it serves. Exhausted, + /// The controller is on this machine and the kernel would not hand the + /// service its claim: nothing on it is served, and a listing is refused + /// the same. + ClaimRefused, } impl Refusal { @@ -110,6 +114,7 @@ impl Refusal { Self::Unusable => 3, Self::Malformed => 4, Self::Exhausted => 5, + Self::ClaimRefused => 6, } } @@ -120,6 +125,7 @@ impl Refusal { 3 => Some(Self::Unusable), 4 => Some(Self::Malformed), 5 => Some(Self::Exhausted), + 6 => Some(Self::ClaimRefused), _ => None, } } @@ -154,7 +160,14 @@ mod tests { let opened = Opened { blocks: u64::MAX - 3, unique: [7; GUID_BYTES] }; assert_eq!(Opened::decode(&opened.encode()), Some(opened)); assert_eq!(Opened::decode(&opened.encode()[1..]), None); - for r in [Refusal::NotFound, Refusal::Held, Refusal::Unusable, Refusal::Malformed, Refusal::Exhausted] { + for r in [ + Refusal::NotFound, + Refusal::Held, + Refusal::Unusable, + Refusal::Malformed, + Refusal::Exhausted, + Refusal::ClaimRefused, + ] { assert_eq!(Refusal::decode(&r.encode()), Some(r)); } assert_eq!(Refusal::decode(&0u32.to_le_bytes()), None); diff --git a/userland/blockd/src/main.rs b/userland/blockd/src/main.rs index 660efaf0fb1..dc53ba86c29 100644 --- a/userland/blockd/src/main.rs +++ b/userland/blockd/src/main.rs @@ -23,6 +23,11 @@ //! A partition whose range is not whole 4 KiB blocks, or whose GUID the table //! carries twice, is listed and refused by name. //! +//! **A controller its row names that is on the machine and whose claim the +//! kernel refused is not an absent one**: its starter names it +//! (`--claim-refused `), and a listing and every open are refused +//! `ClaimRefused`, so a file server says so rather than finding no disk. +//! //! **The partition the machine runs from is refused to every session**: its //! starter names it (`--running `, the ROOT the loader read), because a //! writer there changes the image under the kernel that booted from it. @@ -78,6 +83,10 @@ const SILENCE_WRITE: &str = "--silence-write"; /// no session opens. const RUNNING: &str = "--running"; +/// Argv, followed by a device name of this service's row: on this machine, and +/// its claim refused. +const CLAIM_REFUSED: &str = "--claim-refused"; + const TOKEN_IRQ: u64 = 0; const TOKEN_ACCEPT: u64 = 1; const TOKEN_PENDING: u64 = 0x1_0000; @@ -203,6 +212,9 @@ enum Drive { /// refused `Unusable`, so a file server says the disk failed rather than /// that there is none. Unusable, + /// A controller this row names is on this machine and its claim was + /// refused: a listing and an open are refused `ClaimRefused`. + ClaimRefused, } impl Drive { @@ -211,14 +223,16 @@ impl Drive { fn up(&mut self) -> &mut Controller { match self { Drive::Up(ctrl, _) => ctrl, - Drive::Absent | Drive::Unusable => unreachable!("blockd: a session with no controller"), + Drive::Absent | Drive::Unusable | Drive::ClaimRefused => { + unreachable!("blockd: a session with no controller") + } } } fn oldest(&self) -> Option { match self { Drive::Up(ctrl, _) => ctrl.oldest(), - Drive::Absent | Drive::Unusable => None, + Drive::Absent | Drive::Unusable | Drive::ClaimRefused => None, } } } @@ -258,6 +272,7 @@ impl Service { } Drive::Absent => Ok(Vec::new()), Drive::Unusable => Err(Refusal::Unusable), + Drive::ClaimRefused => Err(Refusal::ClaimRefused), } } @@ -267,6 +282,7 @@ impl Service { Drive::Up(_, parts) => parts, Drive::Absent => return Err(Refusal::NotFound), Drive::Unusable => return Err(Refusal::Unusable), + Drive::ClaimRefused => return Err(Refusal::ClaimRefused), }; let part = parts.iter().find(|p| p.unique == guid && guid != [0; 16]); let (first, blocks) = match part.map(|p| &p.span) { @@ -502,8 +518,7 @@ impl Service { } } -/// The PCI function this process was endowed, or `None` on a machine that -/// has none its row names: init says which, and starts it anyway. +/// The PCI function this process was endowed. fn claim() -> Option { let label = Endowments::get() .labels() @@ -526,7 +541,17 @@ fn main() { .map(|guid| guid.0) .unwrap_or_else(|| panic!("blockd: {RUNNING} takes the running ROOT's unique GUID")) }); + let refused = args.iter().position(|a| a == CLAIM_REFUSED).map(|at| { + args.get(at + 1).unwrap_or_else(|| panic!("blockd: {CLAIM_REFUSED} takes the device its claim was refused on")) + }); let acceptor = endow::acceptor(PORT).unwrap_or_else(|| panic!("blockd: started serving no `{PORT}` port")); + if let Some(name) = refused { + println!( + "blockd: NOT SERVING — {name} is on this machine and the kernel refused this service its \ + claim; every partition on it is refused" + ); + serve(&mut Service::new(Drive::ClaimRefused, running), &acceptor); + } let Some(dev) = claim() else { println!("blockd: no NVMe controller this row names is on this machine; serving no partition"); serve(&mut Service::new(Drive::Absent, running), &acceptor); @@ -716,15 +741,19 @@ fn handshake(service: &mut Service, p: Pending, msg_type: u32, payload_len: usiz mod tests { use super::*; - /// A controller this service cannot use is a disk that failed and never a - /// machine without one: its listing and its open are refused `Unusable`, - /// where an absent controller's listing is empty and its open `NotFound`. + /// A controller this service cannot use, or one whose claim was refused, + /// is a disk there and never a machine without one: its listing and its + /// open are refused by that word, where an absent controller's listing is + /// empty and its open `NotFound`. #[test] - fn an_unusable_controller_is_refused_and_an_absent_one_lists_nothing() { + fn an_unusable_or_unclaimed_controller_is_refused_and_an_absent_one_lists_nothing() { let guid = [7; 16]; let mut unusable = Service::new(Drive::Unusable, None); assert_eq!(unusable.listing(), Err(Refusal::Unusable)); assert_eq!(unusable.place(guid), Err(Refusal::Unusable)); + let mut unclaimed = Service::new(Drive::ClaimRefused, None); + assert_eq!(unclaimed.listing(), Err(Refusal::ClaimRefused)); + assert_eq!(unclaimed.place(guid), Err(Refusal::ClaimRefused)); let mut absent = Service::new(Drive::Absent, None); assert_eq!(absent.listing(), Ok(Vec::new())); assert_eq!(absent.place(guid), Err(Refusal::NotFound)); diff --git a/userland/fsd/src/data.rs b/userland/fsd/src/data.rs index ee8d410bdab..dc8d3b07664 100644 --- a/userland/fsd/src/data.rs +++ b/userland/fsd/src/data.rs @@ -79,6 +79,39 @@ pub enum Probed { Foreign, } +/// Where DATA's partition is, counted over both sources at once: the claims +/// init minted on the kernel's disks and the TOYOS-DATA partitions the block +/// service lists, or why it would not list them. +#[derive(Debug, PartialEq, Eq)] +pub enum Located { + /// This machine has none: memory stands in. + Nowhere, + /// The one claim init minted. + Claimed, + /// The one partition the block service serves, by its unique GUID. + Served([u8; 16]), + /// Two or more, never guessed between, or none countable: DATA is absent. + Refused(String), +} + +pub fn find(claims: usize, served: Result<&[[u8; 16]], String>) -> Located { + let served = match served { + Ok(served) => served, + Err(why) => return Located::Refused(why), + }; + match (claims, served) { + (0, []) => Located::Nowhere, + (1, []) => Located::Claimed, + (0, [one]) => Located::Served(*one), + (claims, served) => Located::Refused(format!( + "this machine has {} DATA partitions, {claims} on the kernel's disks and {} the block \ + service serves, and a volume is one", + claims + served.len(), + served.len() + )), + } +} + fn io(e: &FsError) -> SyscallError { match e { FsError::NotFound => SyscallError::NotFound, @@ -748,6 +781,27 @@ mod tests { const CREATE: OpenHow = OpenHow { create: true, create_new: false, truncate: false }; const PLAIN: OpenHow = OpenHow { create: false, create_new: false, truncate: false }; + /// Two DATA partitions are refused wherever each is, and a block service + /// that would not list what it has leaves DATA absent even beside a claim: + /// memory stands in only where both sources say there is none. + #[test] + fn data_is_one_partition_counted_over_both_sources() { + let (a, b) = ([1; 16], [2; 16]); + assert_eq!(find(0, Ok(&[])), Located::Nowhere); + assert_eq!(find(1, Ok(&[])), Located::Claimed); + assert_eq!(find(0, Ok(&[a])), Located::Served(a)); + let two = |claims, served: &[[u8; 16]]| match find(claims, Ok(served)) { + Located::Refused(why) => why, + other => panic!("{claims} claims and {} served located {other:?}", served.len()), + }; + assert!(two(1, &[a]).starts_with("this machine has 2 DATA partitions, 1 on the kernel's disks and 1")); + assert!(two(2, &[]).starts_with("this machine has 2 DATA partitions, 2 on the kernel's disks and 0")); + assert!(two(0, &[a, b]).starts_with("this machine has 2 DATA partitions, 0 on the kernel's disks and 2")); + for claims in [0, 1] { + assert_eq!(find(claims, Err("unlisted".into())), Located::Refused("unlisted".into())); + } + } + #[test] fn a_file_reads_back_what_was_written_across_pages_and_holes() { let mut v = vol(); diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index d4e1683e067..529d11b8b6d 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -2,9 +2,11 @@ //! of that role as capabilities. //! //! **What it holds**: the acceptor of each directory its role serves, endowed -//! by init under `serve:fs:`; for its volume, a claim on a partition of a -//! disk the kernel drives, or blockd's `block` connector in its namespace; and -//! nothing else of the machine. Argv is the role, and for LOG and BOOT the +//! by init under `serve:fs:`; for its volume, a claim on each partition +//! of its role a disk the kernel drives carries, and blockd's `block` +//! connector in its namespace; and nothing else of the machine. DATA is one +//! partition counted over both, and two are refused by name, never guessed +//! between (`fsd::data::find`). Argv is the role, and for LOG and BOOT the //! unique GUID of the partition the loader named for it when no claim on it //! was minted; then the row's own arguments, which are tests' actuators //! ([`END_ON`], [`END_AT_READ`], [`END_AT_MOUNT`]). @@ -28,7 +30,7 @@ use std::collections::BTreeMap; use std::time::{Duration, Instant}; use fsd::absent::Absent; -use fsd::data::{DataVolume, Probed}; +use fsd::data::{DataVolume, Located, Probed}; use fsd::disk::{Claimed, Disk, Ram, Served}; use fsd::fat::FatVolume; use fsd::resolve::{self, Found, Refusal as Escape, Resolved}; @@ -267,19 +269,17 @@ fn capabilities(role: Role) -> Vec { caps } -/// The partition claim init minted, when the role's partition is on a disk +/// The partition claims init minted: one per partition of the role on a disk /// the kernel drives. -fn claimed() -> Option { +fn claims() -> Vec { let prefix = format!("{DEV_PREFIX}part:"); - let label = Endowments::get().labels().find(|l| l.starts_with(&prefix))?.to_string(); - let claim: toyos::PartitionDev = Endowments::get().take(&label)?; - match Claimed::new(claim) { - Ok(disk) => Some(disk), - Err(why) => { - println!("fsd: the partition claim would not describe itself: {why:?}"); - None - } - } + let labels: Vec = + Endowments::get().labels().filter(|l| l.starts_with(&prefix)).map(str::to_string).collect(); + labels.iter().map(|l| Endowments::get().take(l).expect("fsd: the claim its label names")).collect() +} + +fn describe(claim: toyos::PartitionDev) -> Result { + Claimed::new(claim).map_err(|why| format!("the partition claim would not describe itself ({why:?})")) } /// A session on the partition blockd serves under `guid`. @@ -312,26 +312,35 @@ fn data_partitions() -> Result, String> { fn open_volume(role: Role, guid: Option<&str>, roots: &[&str]) -> Box { match role { Role::Data => { - if let Some(disk) = claimed() { - return data_on(disk, roots); - } - match data_partitions().as_deref() { - Ok([one]) => match served(*one) { + let mut claims = claims(); + let found = fsd::data::find(claims.len(), data_partitions().as_deref().map_err(String::clone)); + let absent = |why: String| -> Box { + println!("fsd: {why}; DATA is absent this boot"); + Box::new(Absent::new(roots, why)) + }; + match found { + Located::Nowhere => ram(roots, "this machine has no DATA partition"), + Located::Claimed => match describe(claims.pop().expect("one claim")) { + Ok(disk) => data_on(disk, roots), + Err(why) => absent(why), + }, + Located::Served(guid) => match served(guid) { Some(disk) => data_on(disk, roots), - None => Box::new(Absent::new(roots, "the DATA partition would not open".into())), + None => absent("the DATA partition would not open".into()), }, - Ok([]) => ram(roots, "this machine has no DATA partition"), - Ok(many) => ram(roots, &format!("this machine has {} DATA partitions, and a volume is one", many.len())), - Err(why) => { - println!("fsd: {why}; DATA is absent this boot"); - Box::new(Absent::new(roots, why.clone())) - } + Located::Refused(why) => absent(why), } } Role::Log | Role::Boot => { let writable = role == Role::Log; - let mounted = match claimed() { - Some(disk) => Ok(fat_on(disk, writable)), + let mut claims = claims(); + let mounted = match claims.pop() { + Some(claim) => { + // init claims by the loader's GUID, which the kernel + // refuses when two partitions carry it. + assert!(claims.is_empty(), "fsd: init handed the {role:?} server two claims"); + Ok(describe(claim).and_then(|disk| fat_on(disk, writable))) + } // init starts this role with its partition's claim or its GUID, never neither. None => { let guid = guid.unwrap_or_else(|| { diff --git a/userland/init/src/main.rs b/userland/init/src/main.rs index 035be2c9fa4..7fa2469fc46 100644 --- a/userland/init/src/main.rs +++ b/userland/init/src/main.rs @@ -1972,7 +1972,14 @@ fn start<'a>( // has none" is a configuration and every other answer is a fault, // and one sentence for all six sends whoever reads the line looking // in the wrong place. - Err(e) => say!("init: {}: {}", program.name, refused(name, e)), + Err(e) => { + say!("init: {}: {}", program.name, refused(name, e)); + // A block service is told, so the partitions on a controller + // the machine has are refused and never taken for none. + if e != SyscallError::NotFound && program.serves.iter().any(|s| s == toyos_blockring::PORT) { + command.args(["--claim-refused", name.as_str()]); + } + } } } // The idle slot, for the one program whose row asks for it: minted here, @@ -2240,13 +2247,13 @@ fn guid_text(guid: [u8; 16]) -> String { /// A storage row's arguments and claims for one start. /// /// A block service is told the ROOT the machine runs from, which it serves no -/// session on. A file server is told its role, and gets its partition as a -/// claim when a disk the kernel drives carries it — the stick, until usbd -/// serves it — and otherwise the partition's GUID, which it opens through the -/// block service; DATA it finds by type there itself. A log or boot role the -/// loader named no partition for, a claim the kernel refuses, and a role two -/// partitions on the kernel's disks carry each refuse the start: the role is -/// then absent, never served from memory. +/// session on. A file server is told its role, and gets a claim on every +/// partition of its role a disk the kernel drives carries — the stick, until +/// usbd serves it — and otherwise the partition's GUID, which it opens through +/// the block service; DATA it finds by type there itself, and counts with its +/// claims. A log or boot role the loader named no partition for, and a claim +/// the kernel refuses, each refuse the start: the role is then absent, never +/// served from memory. fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> Result { use toyos_abi::inventory::{Record, Role}; let mut storage = Storage::default(); @@ -2283,28 +2290,18 @@ fn storage_endowment(program: &Program, role: Option<&str>, syscap: &SysCap) -> } other => panic!("init: `{other}` is no role; the build refuses it"), }; - match kernel.as_slice() { - [guid] => { - let request = toyos_abi::syscall::DeviceRequest::Partition(toyos_abi::part::PartGuid(*guid)); - let mut buf = [0u8; toyos_abi::syscall::DeviceRequest::MAX_NAME]; - let name = request.write_name(&mut buf).to_string(); - match syscap.claim_partition::(toyos_abi::part::PartGuid(*guid)) { - Ok(claim) => storage.claims.push((format!("{DEV_PREFIX}{name}"), claim)), - Err(e) => return Err(StartError::Partition(refused(&name, e))), - } - } - [] => { - if let Some(guid) = named { - storage.args.push(guid_text(guid)); - } - } - many => { - return Err(StartError::Partition(format!( - "the kernel's disks carry {} partitions for the `{role}` role, and a role is one", - many.len() - ))) + for guid in &kernel { + let request = toyos_abi::syscall::DeviceRequest::Partition(toyos_abi::part::PartGuid(*guid)); + let mut buf = [0u8; toyos_abi::syscall::DeviceRequest::MAX_NAME]; + let name = request.write_name(&mut buf).to_string(); + match syscap.claim_partition::(toyos_abi::part::PartGuid(*guid)) { + Ok(claim) => storage.claims.push((format!("{DEV_PREFIX}{name}"), claim)), + Err(e) => return Err(StartError::Partition(refused(&name, e))), } } + if let (Some(guid), []) = (named, kernel.as_slice()) { + storage.args.push(guid_text(guid)); + } Ok(storage) } From 25ae71a5867bcca4a98a0aae572066ef4c1e1661 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 16:58:57 +0200 Subject: [PATCH 44/54] A wait on a connection is IPC, so watch-window holds pipes and not file calls Every file call on this branch is a request on a connection to a file server and a blocking read of its reply, and sys_read/sys_write charged a connection's wait to WaitClass::Pipe. watch-window holds every pipe-class waiter in a Ring 0 spin of up to 50 ms with no preemption point, so under the actuator each of logd's writes through fsd, and fsd's through blockd, became such a spin beside blocking_read_window's canary. A connection's wait is now WaitClass::Ipc, which is also what the blocked-time breakdown should have said of it; a bare pipe's stays Pipe and stays held. process_stats gains the arm that names it: a child parked reading a connection, answered once the roster says it is blocked, charges its park to blocked_ipc_ns. home_overwrite_reads_back's guest fsyncs a file before the pinned overwrite, so the volume's format is on the device and the stop's sync is the only thing that carries the pinned file there: without that sync the host's read names the lost file rather than an unformatted volume. The comment naming SYS_SHUTDOWN's drain, which no longer exists, is deleted. Filed: watch-window spins out a hold whose poster is queued behind it on the same CPU, the mechanism the canary's 50 ms-a-half-trip reds fit. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-out-a-hold-its-poster-is-queued-behind.md | 34 ++++++++ kernel/src/syscall/io.rs | 26 ++++-- .../src/bin/home_overwrite_zero.rs | 6 +- .../toyos-rust-tests/src/bin/process_stats.rs | 86 ++++++++++++++++++- 4 files changed, 142 insertions(+), 10 deletions(-) create mode 100644 issues/kernel/watch-window-spins-out-a-hold-its-poster-is-queued-behind.md diff --git a/issues/kernel/watch-window-spins-out-a-hold-its-poster-is-queued-behind.md b/issues/kernel/watch-window-spins-out-a-hold-its-poster-is-queued-behind.md new file mode 100644 index 00000000000..cc592270741 --- /dev/null +++ b/issues/kernel/watch-window-spins-out-a-hold-its-poster-is-queued-behind.md @@ -0,0 +1,34 @@ +--- +status: open +kind: defect +opened: 2026-09-28 +--- + +# `watch-window` spins out a hold whose poster is queued behind it on the same CPU + +`kernel/src/watch.rs`'s `window::hold` spins in Ring 0 until the waiter's +notified bit is set or its 50 ms `WINDOW` lapses, and nothing in the loop is a +preemption point: a Ring 0 timer fire only sets `need_resched`, and an ordinary +wake to a CPU that is not asleep rings no IPI (`Urgency::Normal`) and is +drained at that CPU's next pass. So when the task that would post is queued on +the holding CPU, or its wake waits in that CPU's mailbox, the post cannot land +and the hold always lapses. + +`blocking_read_stress` is a strict ping-pong, so this is self-sustaining once +its two processes share a CPU: each half-trip's reader holds 50 ms with its +writer queued behind it, then parks, and the writer runs. The idle CPU's steal +probe cannot break it, because `answer_steal_requests` hands over nothing at +`fair_len() <= 1` and at most one of the pair is runnable at any pass. The +recorded reds fit it: `only 27 of 500 round trips completed inside 3s` is +about 111 ms a round trip, two lapsed holds. In that run (#536 at `06c6195f`, +`orch-runs/ab-brw-536-7.log`) the echo child made 28 reads and spent +`cpu=1471ms`, about 50 ms each; the parent spent `cpu=1665ms`. + +Neither that the pair shared a CPU nor how it came to was observed; the guest +prints no placement for it. This is read from the code and one run's numbers, +not reproduced. + +**Exit**: a hold gives its CPU up while that CPU owes a pass or has work queued +(`CpuHandle::doorbell().kick_pending()`, `CpuHandle::load()`), so a post from a +task on the same CPU lands in it; and `blocking_read_window` shown green at a +rate on a guest whose canary pair shares a CPU. diff --git a/kernel/src/syscall/io.rs b/kernel/src/syscall/io.rs index 20560cc0cdc..97b36b9c189 100644 --- a/kernel/src/syscall/io.rs +++ b/kernel/src/syscall/io.rs @@ -21,7 +21,7 @@ use super::handles::with_object_ref; /// What `sys_write` does when the object took nothing. enum WriteBlock { - Pipe(pipe::PipeId), + Pipe(pipe::PipeId, WaitClass), Refused(u64), /// Carried out of the process's lock: `HandleError::refuse` may take the /// process down and cannot run under a guard. @@ -30,7 +30,7 @@ enum WriteBlock { /// What `sys_read` parks on when the handle has nothing to give. enum ReadBlock { - Pipe(alloc::sync::Arc, pipe::PipeId), + Pipe(alloc::sync::Arc, pipe::PipeId, WaitClass), VirtioSound, Hda, /// A claimed keyboard, woken by its own IRQ. @@ -56,7 +56,7 @@ pub(super) fn sys_write(h: RawHandle, buf: &UserBytes) -> u64 { match ops::try_write(object, buf) { Some(n) => Ok((n, ops::pipe_id_write(object))), None => Err(match ops::pipe_id_write(object) { - Some(id) => WriteBlock::Pipe(id), + Some(id) => WriteBlock::Pipe(id, pipe_wait(object)), None => WriteBlock::Refused(SyscallError::NotFound.to_u64()), }), } @@ -66,14 +66,14 @@ pub(super) fn sys_write(h: RawHandle, buf: &UserBytes) -> u64 { if let Some(id) = pipe_id { process::wake_pipe_readers(id); } return n; } - Err(WriteBlock::Pipe(id)) => match pipe::write_watch(id) { + Err(WriteBlock::Pipe(id, class)) => match pipe::write_watch(id) { Some(end) => { let parkable = crate::scheduler::Parkable::at_entry(); if watch::wait_until( &parkable, &end, 0, - WaitClass::Pipe, + class, Deadline::never(), || pipe::has_space(id), ) @@ -90,6 +90,16 @@ pub(super) fn sys_write(h: RawHandle, buf: &UserBytes) -> u64 { } } +/// A connection's pipe is waited on for its peer's answer, which is IPC; a +/// bare pipe's is a pipe wait, the one class `watch-window` holds. +fn pipe_wait(object: &KObjectRef) -> WaitClass { + if matches!(object, KObjectRef::Connection(_)) { + WaitClass::Ipc + } else { + WaitClass::Pipe + } +} + /// Only these four device classes block; the rest answer `NotFound` on an /// empty blocking read. fn read_block_device(claim: &crate::object::device::DeviceClaim) -> ReadBlock { @@ -114,7 +124,7 @@ fn read_block(object: &KObjectRef) -> ReadBlock { ReadBlock::Console(Deadline::at(crate::clock::now() + CONSOLE_REPOLL.duration())) } _ => match ops::pipe_id_read(object).and_then(|id| { - pipe::read_watch(id).map(|end| ReadBlock::Pipe(end, id)) + pipe::read_watch(id).map(|end| ReadBlock::Pipe(end, id, pipe_wait(object))) }) { Some(block) => block, None => ReadBlock::Refused(SyscallError::NotFound.to_u64()), @@ -153,13 +163,13 @@ pub(super) fn sys_read(h: RawHandle, buf: &mut UserBytesMut) -> u64 { if let Some(id) = pipe_id { process::wake_pipe_writers(id); } return n; } - Err(ReadBlock::Pipe(end, id)) => { + Err(ReadBlock::Pipe(end, id, class)) => { let parkable = crate::scheduler::Parkable::at_entry(); if watch::wait_until( &parkable, &end, 0, - WaitClass::Pipe, + class, Deadline::never(), || pipe::has_data(id), ) diff --git a/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs b/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs index 48df97c4791..5b0c9a57c48 100644 --- a/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs +++ b/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs @@ -31,6 +31,11 @@ fn main() { let first = payload(0x11); let second = payload(0x22); recorded_shape(&first, &second); + // The volume and every write so far on the device, so the pinned file's + // writes are the ones only the stop's sync carries there. + File::open(LOOPED) + .and_then(|f| f.sync_all()) + .unwrap_or_else(|e| panic!("fsync {LOOPED}: {e}")); pinned_overwrite(&first, &second); println!("all home overwrite tests passed"); } @@ -76,7 +81,6 @@ fn pinned_overwrite(first: &[u8], second: &[u8]) { // Before any verdict: the host holds this count against the device, and needs it on the failing arm. println!("HOME-OVERWRITE {PINNED} read back {lowest} bytes"); - // Dropped first, so `SYS_SHUTDOWN`'s drain carries the overwrite to the device whatever the name answered. drop(writer); assert_eq!(lowest, LEN as u64, "the same-length overwrite read back short"); diff --git a/tests/toyos-rust-tests/src/bin/process_stats.rs b/tests/toyos-rust-tests/src/bin/process_stats.rs index 5a84025450e..f8d8d89440a 100644 --- a/tests/toyos-rust-tests/src/bin/process_stats.rs +++ b/tests/toyos-rust-tests/src/bin/process_stats.rs @@ -16,23 +16,32 @@ //! were permanently zero, and nothing here noticed. use std::io::{BufRead, BufReader, Read, Write}; -use std::os::toyos::process::ChildExt; +use std::os::toyos::process::{ChildExt, CommandExt}; use std::process::{Command, Stdio}; +use std::time::{Duration, Instant}; +use toyos::endow::{Endowments, SVC_LABEL, SYSCAP_LABEL}; use toyos::process::Process; +use toyos::syscap::SysCap; +use toyos::{namespace, port, AsHandle}; use toyos_abi::handle::Rights; use toyos_abi::syscall::{self, ProcessStats, SyscallError}; const SELF_PATH: &str = "/system/bin/test_rs_process_stats"; +/// The name the `held-ipc` child finds its parent's port under. +const HELD_SERVICE: &str = "held"; + fn main() { match std::env::args().nth(1).as_deref() { Some("held") => return held(), + Some("held-ipc") => return held_ipc(), Some("refused") => return refused_child(), _ => {} } exited_child(); live_process(); blocked_time_names_what_it_waited_on(); + a_wait_on_a_connection_is_ipc(); repeatable(); refused_without_read(); refused_calls_are_timed(); @@ -247,6 +256,81 @@ fn blocked_time_names_what_it_waited_on() { ); } +/// A park on a connection is charged to IPC: it waits for the peer's answer, +/// which is what every file call's wait on its server is. +/// +/// The child parks reading a connection whose other end this process +/// accepted, and is answered only once the kernel's roster says it is blocked, +/// so the park is there to charge. Nothing else it does waits on IPC, so +/// `blocked_ipc_ns` moves only if the connection's wait is classed as IPC. +fn a_wait_on_a_connection_is_ipc() { + let (acceptor, connector) = port::create().expect("a port"); + let names = namespace::build().add(HELD_SERVICE, &connector).finish().expect("a namespace"); + let mut child = Command::new(SELF_PATH) + .arg("held-ipc") + .endow(SVC_LABEL, names.into_raw().0) + .stdout(Stdio::piped()) + .spawn() + .expect("spawn the held-ipc child"); + let mut out = BufReader::new(child.stdout.take().expect("held-ipc stdout")); + let mut line = String::new(); + out.read_line(&mut line).expect("the held-ipc child's marker"); + assert_eq!(line.trim(), "running", "the held-ipc child said {line:?}"); + let conn = acceptor.accept().expect("accept the held-ipc child's connection"); + + let pid = stats_of(&child).expect("the held-ipc child answers").pid; + let give_up = Instant::now() + Duration::from_secs(5); + while !main_thread_blocked(pid) { + assert!(Instant::now() < give_up, "the held-ipc child never parked on its connection"); + } + conn.write_nonblock(b"g").expect("release the held-ipc child"); + let status = child.wait().expect("wait the held-ipc child"); + assert!(status.success(), "the held-ipc child exited with {status}"); + + let s = stats_of(&child).expect("an exited child still answers"); + assert!( + s.blocked_ipc_ns > 0, + "a child that parked reading a connection charged {} ns to ipc and {} ns to pipe — a \ + wait on a connection's answer is IPC", + s.blocked_ipc_ns, + s.blocked_pipe_ns, + ); + println!(" a connection's wait: ok (ipc={}ns pipe={}ns)", s.blocked_ipc_ns, s.blocked_pipe_ns); +} + +/// Says it is running, then blocks reading the connection its namespace names +/// until the parent answers it. +fn held_ipc() { + let conn = toyos::endow::service(HELD_SERVICE).expect("held-ipc: the parent's port"); + println!("running"); + std::io::stdout().flush().expect("held-ipc: flush the marker"); + let mut buf = [0u8; 1]; + let n = syscall::read(conn.as_handle(), &mut buf).expect("held-ipc: read the connection"); + assert_eq!(n, 1, "held-ipc: the parent's answer"); +} + +/// `sched::payload::SCHED_BLOCKED` — the state column `ps` prints. +const BLOCKED: u8 = 2; + +/// Whether the kernel's roster has `pid`'s main thread blocked: `Rights::ROSTER` +/// on the system capability test-runner endows every binary. +fn main_thread_blocked(pid: u32) -> bool { + const HEADER: usize = toyos::system::SYSINFO_HEADER_SIZE; + const ENTRY: usize = toyos::system::SYSINFO_ENTRY_SIZE; + static CAP: std::sync::OnceLock = std::sync::OnceLock::new(); + let cap = CAP.get_or_init(|| { + Endowments::get().take(SYSCAP_LABEL).expect("test-runner endows a system capability") + }); + let mut buf = vec![0u8; HEADER + ENTRY * 256]; + let n = cap.roster(&mut buf); + assert!((HEADER..=buf.len()).contains(&n), "sysinfo answered {n}"); + (HEADER..).step_by(ENTRY).take_while(|pos| pos + ENTRY <= n).any(|pos| { + u32::from_le_bytes(buf[pos..pos + 4].try_into().unwrap()) == pid + && buf[pos + 9] == 0 + && buf[pos + 8] == BLOCKED + }) +} + /// Reading does not spend it. This asserted the opposite before the handle: /// the snapshot lived on the parent and the read deleted it. fn repeatable() { From 069722c3476b7963ffd0fa5a8c6e850b6478c679 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 17:10:41 +0200 Subject: [PATCH 45/54] Close the bootstrap-cache issue: main's #573 clears the cache on a compiler switch The issue's exit was a worktree's bootstrap cache keyed on, or cleared with, the compiler that builds it. The merge of ec06384e brought src/sysroot.rs's forget_another_compiler: build_std records the compiler's identity in /build/toyos-std/compiled-by, and on any other identity empties that directory of everything but bootstrap's downloads (cache, /ci-llvm, /rustfmt), so rust/build/toyos-std/bootstrap, where the stale serde rlibs lived, goes with it. The rule is stated at that function; nothing else cites the issue. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-a-compiler-rebuilt-at-the-same-version.md | 30 ------------------- 1 file changed, 30 deletions(-) delete mode 100644 issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md diff --git a/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md b/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md deleted file mode 100644 index 2f1015a7eef..00000000000 --- a/issues/build/a-worktrees-bootstrap-cache-outlives-a-compiler-rebuilt-at-the-same-version.md +++ /dev/null @@ -1,30 +0,0 @@ ---- -status: open -kind: tooling -opened: 2026-09-28 ---- - -# A worktree's bootstrap cache outlives a compiler rebuilt at the same version - -A linked worktree builds its std sysroot through its own fork checkout's -bootstrap, whose cargo target directory is -`rust/build/toyos-std/bootstrap`, compiled by the primary's compiler named in -`bootstrap.toml`'s `rustc`. Every build of that compiler answers `rustc -1.99.0-dev`, `commit-hash: unknown`. Cargo keys freshness on that answer, so -once the primary's compiler is rebuilt, -the cached dependencies are reused, and the new -compiler cannot load them. Bootstrap then fails to compile with `error[E0463]: -can't find crate for serde` (and `clap`, `xz2`), with nothing compiled before -it, and the sysroot build panics at `src/toolchain.rs` with "std did not -compile". - -**Evidence:** PR #536's worktree at 1f969973, sysroot key 8ec6e31551091ba8: -`cargo run -- --build-only` failed the same way twice. Its -`libserde-*.rmeta` dated from 2026-09-27; the compiler directory -`compilers/a04a68b92a50e478` from 2026-09-28 09:07. The same sources built -cleanly into a fresh target directory. With the stale directory moved aside, -the next `--build-only` succeeded. - -**Exit condition:** a worktree's bootstrap cache is keyed on, or cleared with, -the compiler that builds it, so a rebuilt compiler never meets artifacts from -the one it replaced. From 0baa8b3410188c2d20bf59f3a2f094cbb421dddf Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 18:35:02 +0200 Subject: [PATCH 46/54] Round 15 review: a full connection's write is measured, and its class is chosen where the kind is - `ops::pipe_read` and `ops::pipe_write` answer a blocking read's or write's pipe together with its wait class, in the exhaustive matches that already name every pipe-bearing kind, so a new one cannot compile without choosing. `io.rs`'s `pipe_wait`, a `matches!` that fell back to `Pipe`, is deleted; `pipe_id_read` and `pipe_id_write` are those two with the class dropped. - `process_stats` gains `a_wait_to_write_a_full_connection_is_ipc`: a child fills a connection, parks writing one byte more, and this process reads to make room once the roster has it blocked. Both connection arms share `parked_on_a_connection`. - `tests/toyos-rust-tests/src/roster.rs` is the one roster decoder, its `BLOCKED`, the capability it reads with and the bounded poll, included by `process_stats` and `process_lifecycle` in place of their two copies. - Deleted: `watch_window`'s doc in `actuator.rs`, which said it holds every watch waiter; the narration in `a_wait_on_a_connection_is_ipc`'s doc; and a scratch log path cited in the watch-window issue. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-out-a-hold-its-poster-is-queued-behind.md | 6 +- kernel/src/actuator.rs | 2 - kernel/src/object/ops.rs | 22 +++- kernel/src/syscall/io.rs | 18 +-- .../src/bin/process_lifecycle.rs | 62 ++------- .../toyos-rust-tests/src/bin/process_stats.rs | 124 +++++++++++------- tests/toyos-rust-tests/src/roster.rs | 65 +++++++++ 7 files changed, 171 insertions(+), 128 deletions(-) create mode 100644 tests/toyos-rust-tests/src/roster.rs diff --git a/issues/kernel/watch-window-spins-out-a-hold-its-poster-is-queued-behind.md b/issues/kernel/watch-window-spins-out-a-hold-its-poster-is-queued-behind.md index cc592270741..6bb4518c7ef 100644 --- a/issues/kernel/watch-window-spins-out-a-hold-its-poster-is-queued-behind.md +++ b/issues/kernel/watch-window-spins-out-a-hold-its-poster-is-queued-behind.md @@ -20,9 +20,9 @@ writer queued behind it, then parks, and the writer runs. The idle CPU's steal probe cannot break it, because `answer_steal_requests` hands over nothing at `fair_len() <= 1` and at most one of the pair is runnable at any pass. The recorded reds fit it: `only 27 of 500 round trips completed inside 3s` is -about 111 ms a round trip, two lapsed holds. In that run (#536 at `06c6195f`, -`orch-runs/ab-brw-536-7.log`) the echo child made 28 reads and spent -`cpu=1471ms`, about 50 ms each; the parent spent `cpu=1665ms`. +about 111 ms a round trip, two lapsed holds. In that run (#536 at +`06c6195f`) the echo child made 28 reads and spent `cpu=1471ms`, about +50 ms each; the parent spent `cpu=1665ms`. Neither that the pair shared a CPU nor how it came to was observed; the guest prints no placement for it. This is read from the code and one run's numbers, diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 1db798afb06..a0bd856a414 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -257,8 +257,6 @@ actuators! { /// Report the preempt depth and backtrace at the deepest point of a disk transfer; it stages nothing, only measures. io_depth_probe = "io-depth-probe"; - /// Hold every thread that waits on a watch between reading its condition and - /// parking, so a post lands in the window its commit must refuse the park over. watch_window = "watch-window"; /// Starve the four xHCI bring-up register waits in `init_one`. diff --git a/kernel/src/object/ops.rs b/kernel/src/object/ops.rs index 6786ec80753..68d8d6bca4c 100644 --- a/kernel/src/object/ops.rs +++ b/kernel/src/object/ops.rs @@ -9,6 +9,7 @@ use alloc::vec::Vec; use toyos_abi::handle::{RawHandle, Rights}; use toyos_abi::syscall::{FileType, OpenFlags, SeekFrom, SyscallError}; +use toyos_sched::task::WaitClass; use crate::drivers::serial; use crate::file_cache; @@ -202,9 +203,19 @@ pub fn close_all(table: &mut HandleTable) { } pub fn pipe_id_read(object: &KObjectRef) -> Option { + pipe_read(object).map(|(id, _)| id) +} + +pub fn pipe_id_write(object: &KObjectRef) -> Option { + pipe_write(object).map(|(id, _)| id) +} + +/// The pipe a blocking read of `object` parks on, and the class its wait is +/// charged to: a connection's is its peer's answer, which is IPC. +pub fn pipe_read(object: &KObjectRef) -> Option<(PipeId, WaitClass)> { match object { - KObjectRef::PipeRead(r) => Some(r.id()), - KObjectRef::Connection(c) => Some(c.rx()), + KObjectRef::PipeRead(r) => Some((r.id(), WaitClass::Pipe)), + KObjectRef::Connection(c) => Some((c.rx(), WaitClass::Ipc)), KObjectRef::PipeWrite(_) | KObjectRef::File(_) | KObjectRef::Device(_) | KObjectRef::Console(_) | KObjectRef::Acceptor(_) | KObjectRef::Inbox(_) | KObjectRef::SysCap(_) @@ -213,10 +224,11 @@ pub fn pipe_id_read(object: &KObjectRef) -> Option { } } -pub fn pipe_id_write(object: &KObjectRef) -> Option { +/// [`pipe_read`]'s answer for a blocking write. +pub fn pipe_write(object: &KObjectRef) -> Option<(PipeId, WaitClass)> { match object { - KObjectRef::PipeWrite(w) => Some(w.id()), - KObjectRef::Connection(c) => Some(c.tx()), + KObjectRef::PipeWrite(w) => Some((w.id(), WaitClass::Pipe)), + KObjectRef::Connection(c) => Some((c.tx(), WaitClass::Ipc)), KObjectRef::PipeRead(_) | KObjectRef::File(_) | KObjectRef::Device(_) | KObjectRef::Console(_) | KObjectRef::Acceptor(_) | KObjectRef::Inbox(_) | KObjectRef::SysCap(_) diff --git a/kernel/src/syscall/io.rs b/kernel/src/syscall/io.rs index 97b36b9c189..ad5c6d25f0e 100644 --- a/kernel/src/syscall/io.rs +++ b/kernel/src/syscall/io.rs @@ -55,8 +55,8 @@ pub(super) fn sys_write(h: RawHandle, buf: &UserBytes) -> u64 { }; match ops::try_write(object, buf) { Some(n) => Ok((n, ops::pipe_id_write(object))), - None => Err(match ops::pipe_id_write(object) { - Some(id) => WriteBlock::Pipe(id, pipe_wait(object)), + None => Err(match ops::pipe_write(object) { + Some((id, class)) => WriteBlock::Pipe(id, class), None => WriteBlock::Refused(SyscallError::NotFound.to_u64()), }), } @@ -90,16 +90,6 @@ pub(super) fn sys_write(h: RawHandle, buf: &UserBytes) -> u64 { } } -/// A connection's pipe is waited on for its peer's answer, which is IPC; a -/// bare pipe's is a pipe wait, the one class `watch-window` holds. -fn pipe_wait(object: &KObjectRef) -> WaitClass { - if matches!(object, KObjectRef::Connection(_)) { - WaitClass::Ipc - } else { - WaitClass::Pipe - } -} - /// Only these four device classes block; the rest answer `NotFound` on an /// empty blocking read. fn read_block_device(claim: &crate::object::device::DeviceClaim) -> ReadBlock { @@ -123,8 +113,8 @@ fn read_block(object: &KObjectRef) -> ReadBlock { ); ReadBlock::Console(Deadline::at(crate::clock::now() + CONSOLE_REPOLL.duration())) } - _ => match ops::pipe_id_read(object).and_then(|id| { - pipe::read_watch(id).map(|end| ReadBlock::Pipe(end, id, pipe_wait(object))) + _ => match ops::pipe_read(object).and_then(|(id, class)| { + pipe::read_watch(id).map(|end| ReadBlock::Pipe(end, id, class)) }) { Some(block) => block, None => ReadBlock::Refused(SyscallError::NotFound.to_u64()), diff --git a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs index aa98fa60e9a..98eef95253b 100644 --- a/tests/toyos-rust-tests/src/bin/process_lifecycle.rs +++ b/tests/toyos-rust-tests/src/bin/process_lifecycle.rs @@ -28,16 +28,18 @@ use std::io::Read; use std::os::toyos::process::{ChildExt, CommandExt}; use std::process::{Child, ChildStdin, Command, Stdio}; use std::sync::atomic::{AtomicBool, Ordering}; -use std::sync::OnceLock; -use std::time::{Duration, Instant}; -use toyos::endow::{Endowments, SYSCAP_LABEL}; +use toyos::endow::Endowments; use toyos::AsHandle; use toyos::process::Process; -use toyos::syscap::SysCap; use toyos_abi::syscall::{self, SyscallError}; use toyos_abi::RawHandle; +#[path = "../roster.rs"] +mod roster; + +use roster::{await_true, cap}; + const SELF_PATH: &str = "/system/bin/test_rs_process_lifecycle"; /// The label the `waiter` role finds its subject under. A local name in one @@ -196,20 +198,6 @@ fn an_unrelated_wake_does_not_end_the_wait() { println!(" a wake meant for something else does not end a wait"); } -/// Poll until `cond` holds. The bound is a hang guard and not a timing -/// assumption: both callers wait for a state the kernel has already decided and -/// reaches in microseconds, and the `sysinfo` call inside `cond` is the loop's -/// preemption point (`thread::yield_now` is a spin hint on this platform). -fn await_true(what: &str, cond: fn() -> bool) { - let give_up = Instant::now() + Duration::from_secs(5); - while !cond() { - assert!(Instant::now() < give_up, "{what}"); - } -} - -/// `sched::payload::SCHED_BLOCKED` — the state column `ps` prints. -const BLOCKED: u8 = 2; - /// `SCHED_UNKNOWN`, which `sys_sysinfo` also answers for a thread whose entry /// is a zombie. A live thread's scheduler record is installed under the same /// table lock that inserts its entry, so a thread of ours reading this has @@ -217,46 +205,12 @@ const BLOCKED: u8 = 2; const ZOMBIE: u8 = 3; fn main_thread_is_parked() -> bool { - my_threads().iter().any(|&(is_thread, state)| !is_thread && state == BLOCKED) + roster::main_thread_blocked(syscall::getpid().raw()) } fn a_thread_of_mine_has_exited() -> bool { - my_threads().iter().any(|&(is_thread, state)| is_thread && state == ZOMBIE) -} - -/// The estate's system capability, taken once. -/// -/// **Once, because taking is a swap**: a second `take` of the same label finds -/// `HANDLE_INVALID` and answers `None`, and two arms here want the same cap — -/// one for the `MANAGE` refusal, one for the roster below. -fn cap() -> &'static SysCap { - static CAP: OnceLock = OnceLock::new(); - CAP.get_or_init(|| { - Endowments::get() - .take(SYSCAP_LABEL) - .expect("test-runner endows every binary it spawns a system capability") - }) -} - -/// This process's threads as the kernel publishes them: `(is a child thread, -/// scheduler state)`. -/// -/// Even one's own threads arrive in the machine-wide roster, which is -/// `Rights::ROSTER` on a `SysCap` — there is no narrower question in the ABI, -/// and `tests/testcases` names `roster` on the test-runner row for this. -fn my_threads() -> Vec<(bool, u8)> { - const HEADER: usize = toyos::system::SYSINFO_HEADER_SIZE; - const ENTRY: usize = toyos::system::SYSINFO_ENTRY_SIZE; - let mut buf = vec![0u8; HEADER + ENTRY * 256]; - let n = cap().roster(&mut buf); - assert!((HEADER..=buf.len()).contains(&n), "sysinfo answered {n}"); let me = syscall::getpid().raw(); - (HEADER..) - .step_by(ENTRY) - .take_while(|pos| pos + ENTRY <= n) - .filter(|&pos| u32::from_le_bytes(buf[pos..pos + 4].try_into().unwrap()) == me) - .map(|pos| (buf[pos + 9] != 0, buf[pos + 8])) - .collect() + roster::roster().iter().any(|e| e.pid == me && e.is_thread && e.state == ZOMBIE) } /// A second handle is a second name for one object, and the object is where the diff --git a/tests/toyos-rust-tests/src/bin/process_stats.rs b/tests/toyos-rust-tests/src/bin/process_stats.rs index f8d8d89440a..29855474ee1 100644 --- a/tests/toyos-rust-tests/src/bin/process_stats.rs +++ b/tests/toyos-rust-tests/src/bin/process_stats.rs @@ -18,14 +18,15 @@ use std::io::{BufRead, BufReader, Read, Write}; use std::os::toyos::process::{ChildExt, CommandExt}; use std::process::{Command, Stdio}; -use std::time::{Duration, Instant}; -use toyos::endow::{Endowments, SVC_LABEL, SYSCAP_LABEL}; +use toyos::endow::SVC_LABEL; use toyos::process::Process; -use toyos::syscap::SysCap; -use toyos::{namespace, port, AsHandle}; +use toyos::{namespace, port, AsHandle, Connection}; use toyos_abi::handle::Rights; use toyos_abi::syscall::{self, ProcessStats, SyscallError}; +#[path = "../roster.rs"] +mod roster; + const SELF_PATH: &str = "/system/bin/test_rs_process_stats"; /// The name the `held-ipc` child finds its parent's port under. @@ -35,6 +36,7 @@ fn main() { match std::env::args().nth(1).as_deref() { Some("held") => return held(), Some("held-ipc") => return held_ipc(), + Some("held-ipc-write") => return held_ipc_write(), Some("refused") => return refused_child(), _ => {} } @@ -42,6 +44,7 @@ fn main() { live_process(); blocked_time_names_what_it_waited_on(); a_wait_on_a_connection_is_ipc(); + a_wait_to_write_a_full_connection_is_ipc(); repeatable(); refused_without_read(); refused_calls_are_timed(); @@ -256,38 +259,16 @@ fn blocked_time_names_what_it_waited_on() { ); } -/// A park on a connection is charged to IPC: it waits for the peer's answer, -/// which is what every file call's wait on its server is. +/// A park on a connection is charged to IPC: it waits for the peer's answer. /// /// The child parks reading a connection whose other end this process /// accepted, and is answered only once the kernel's roster says it is blocked, /// so the park is there to charge. Nothing else it does waits on IPC, so /// `blocked_ipc_ns` moves only if the connection's wait is classed as IPC. fn a_wait_on_a_connection_is_ipc() { - let (acceptor, connector) = port::create().expect("a port"); - let names = namespace::build().add(HELD_SERVICE, &connector).finish().expect("a namespace"); - let mut child = Command::new(SELF_PATH) - .arg("held-ipc") - .endow(SVC_LABEL, names.into_raw().0) - .stdout(Stdio::piped()) - .spawn() - .expect("spawn the held-ipc child"); - let mut out = BufReader::new(child.stdout.take().expect("held-ipc stdout")); - let mut line = String::new(); - out.read_line(&mut line).expect("the held-ipc child's marker"); - assert_eq!(line.trim(), "running", "the held-ipc child said {line:?}"); - let conn = acceptor.accept().expect("accept the held-ipc child's connection"); - - let pid = stats_of(&child).expect("the held-ipc child answers").pid; - let give_up = Instant::now() + Duration::from_secs(5); - while !main_thread_blocked(pid) { - assert!(Instant::now() < give_up, "the held-ipc child never parked on its connection"); - } - conn.write_nonblock(b"g").expect("release the held-ipc child"); - let status = child.wait().expect("wait the held-ipc child"); - assert!(status.success(), "the held-ipc child exited with {status}"); - - let s = stats_of(&child).expect("an exited child still answers"); + let s = parked_on_a_connection("held-ipc", |conn| { + conn.write_nonblock(b"g").expect("release the held-ipc child"); + }); assert!( s.blocked_ipc_ns > 0, "a child that parked reading a connection charged {} ns to ipc and {} ns to pipe — a \ @@ -298,6 +279,53 @@ fn a_wait_on_a_connection_is_ipc() { println!(" a connection's wait: ok (ipc={}ns pipe={}ns)", s.blocked_ipc_ns, s.blocked_pipe_ns); } +/// The write side of the same wait: the child parks writing a connection it +/// filled, and this process's read is what makes room. +fn a_wait_to_write_a_full_connection_is_ipc() { + let s = parked_on_a_connection("held-ipc-write", |conn| { + let mut room = [0u8; 4096]; + conn.read_nonblock(&mut room).expect("make room for the held-ipc-write child"); + }); + assert!( + s.blocked_ipc_ns > 0, + "a child that parked writing a full connection charged {} ns to ipc and {} ns to pipe \ + — a wait for room on a connection is IPC", + s.blocked_ipc_ns, + s.blocked_pipe_ns, + ); + println!( + " a full connection's wait: ok (ipc={}ns pipe={}ns)", + s.blocked_ipc_ns, s.blocked_pipe_ns, + ); +} + +/// Runs `role` on a connection to this process, `release`s it once the +/// roster has its main thread blocked, and answers the exited child's numbers. +fn parked_on_a_connection(role: &str, release: impl FnOnce(&Connection)) -> ProcessStats { + let (acceptor, connector) = port::create().expect("a port"); + let names = namespace::build().add(HELD_SERVICE, &connector).finish().expect("a namespace"); + let mut child = Command::new(SELF_PATH) + .arg(role) + .endow(SVC_LABEL, names.into_raw().0) + .stdout(Stdio::piped()) + .spawn() + .expect("spawn the connection child"); + let mut out = BufReader::new(child.stdout.take().expect("the connection child's stdout")); + let mut line = String::new(); + out.read_line(&mut line).expect("the connection child's marker"); + assert_eq!(line.trim(), "running", "the {role} child said {line:?}"); + let conn = acceptor.accept().expect("accept the connection child's connection"); + + let pid = stats_of(&child).expect("the connection child answers").pid; + roster::await_true(&format!("the {role} child never parked on its connection"), || { + roster::main_thread_blocked(pid) + }); + release(&conn); + let status = child.wait().expect("wait the connection child"); + assert!(status.success(), "the {role} child exited with {status}"); + stats_of(&child).expect("an exited child still answers") +} + /// Says it is running, then blocks reading the connection its namespace names /// until the parent answers it. fn held_ipc() { @@ -309,26 +337,22 @@ fn held_ipc() { assert_eq!(n, 1, "held-ipc: the parent's answer"); } -/// `sched::payload::SCHED_BLOCKED` — the state column `ps` prints. -const BLOCKED: u8 = 2; - -/// Whether the kernel's roster has `pid`'s main thread blocked: `Rights::ROSTER` -/// on the system capability test-runner endows every binary. -fn main_thread_blocked(pid: u32) -> bool { - const HEADER: usize = toyos::system::SYSINFO_HEADER_SIZE; - const ENTRY: usize = toyos::system::SYSINFO_ENTRY_SIZE; - static CAP: std::sync::OnceLock = std::sync::OnceLock::new(); - let cap = CAP.get_or_init(|| { - Endowments::get().take(SYSCAP_LABEL).expect("test-runner endows a system capability") - }); - let mut buf = vec![0u8; HEADER + ENTRY * 256]; - let n = cap.roster(&mut buf); - assert!((HEADER..=buf.len()).contains(&n), "sysinfo answered {n}"); - (HEADER..).step_by(ENTRY).take_while(|pos| pos + ENTRY <= n).any(|pos| { - u32::from_le_bytes(buf[pos..pos + 4].try_into().unwrap()) == pid - && buf[pos + 9] == 0 - && buf[pos + 8] == BLOCKED - }) +/// Fills the connection its namespace names, says it is running, then blocks +/// writing one more byte until the parent reads. +fn held_ipc_write() { + let conn = toyos::endow::service(HELD_SERVICE).expect("held-ipc-write: the parent's port"); + let chunk = [0u8; 65536]; + loop { + match conn.write_nonblock(&chunk) { + Ok(_) => {} + Err(SyscallError::WouldBlock) => break, + Err(e) => panic!("held-ipc-write: filling the connection: {e:?}"), + } + } + println!("running"); + std::io::stdout().flush().expect("held-ipc-write: flush the marker"); + let n = syscall::write(conn.as_handle(), &[1]).expect("held-ipc-write: write the connection"); + assert_eq!(n, 1, "held-ipc-write: the byte the parent made room for"); } /// Reading does not spend it. This asserted the opposite before the handle: diff --git a/tests/toyos-rust-tests/src/roster.rs b/tests/toyos-rust-tests/src/roster.rs new file mode 100644 index 00000000000..d1d865496e4 --- /dev/null +++ b/tests/toyos-rust-tests/src/roster.rs @@ -0,0 +1,65 @@ +//! The kernel's thread roster as the tests that read it decode it: +//! `Rights::ROSTER` on the system capability test-runner endows every binary, +//! the one question in the ABI that says whether a thread is parked. + +use std::sync::OnceLock; +use std::time::{Duration, Instant}; + +use toyos::endow::{Endowments, SYSCAP_LABEL}; +use toyos::syscap::SysCap; + +/// `sched::payload::SCHED_BLOCKED` — the state column `ps` prints. +pub const BLOCKED: u8 = 2; + +/// One thread of the roster. +pub struct Entry { + pub pid: u32, + /// A child thread, not its process's main one. + pub is_thread: bool, + pub state: u8, +} + +/// The estate's system capability, taken once: a second `take` of the label +/// finds `HANDLE_INVALID`. +pub fn cap() -> &'static SysCap { + static CAP: OnceLock = OnceLock::new(); + CAP.get_or_init(|| { + Endowments::get() + .take(SYSCAP_LABEL) + .expect("test-runner endows every binary it spawns a system capability") + }) +} + +/// Every thread on the machine. +pub fn roster() -> Vec { + const HEADER: usize = toyos::system::SYSINFO_HEADER_SIZE; + const ENTRY: usize = toyos::system::SYSINFO_ENTRY_SIZE; + let mut buf = vec![0u8; HEADER + ENTRY * 256]; + let n = cap().roster(&mut buf); + assert!((HEADER..=buf.len()).contains(&n), "sysinfo answered {n}"); + (HEADER..) + .step_by(ENTRY) + .take_while(|pos| pos + ENTRY <= n) + .map(|pos| Entry { + pid: u32::from_le_bytes(buf[pos..pos + 4].try_into().unwrap()), + is_thread: buf[pos + 9] != 0, + state: buf[pos + 8], + }) + .collect() +} + +/// Whether `pid`'s main thread is parked. +pub fn main_thread_blocked(pid: u32) -> bool { + roster().iter().any(|e| e.pid == pid && !e.is_thread && e.state == BLOCKED) +} + +/// Poll until `cond` holds. The bound is a hang guard and not a timing +/// assumption: every caller waits for a state the kernel has already decided, +/// and the `sysinfo` call inside `cond` is the loop's preemption point +/// (`thread::yield_now` is a spin hint on this platform). +pub fn await_true(what: &str, cond: impl Fn() -> bool) { + let give_up = Instant::now() + Duration::from_secs(5); + while !cond() { + assert!(Instant::now() < give_up, "{what}"); + } +} From 2e158df30b801593734db0922bb15cc062f55971 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 20:00:41 +0200 Subject: [PATCH 47/54] Round 17 review fixes: merge main, drop the redlist row and issue for a test this branch deletes, move the stop-sync's only guest control to Nightly, and rename a misnamed binding Merging origin/main (c089fce6, #580) brought back a redlist row and issue for `quiesce_leaves_the_volume_whole`, a test this branch already deletes; the harness refuses a disabled row nothing registers, so both go. `home_overwrite_reads_back` is the only guest check on init's stop sync and was priced at Weekly for the old kernel-`/home` test it replaced, so a lost sync goes unseen by every PR and nightly; it moves to Nightly. `tests/common/power.rs`'s `synced_at` bound the stop record's line, not anything synced there. Also deletes two stale doc comments in `process_stats.rs` that only narrated the code below them, and a roster.rs clause the reviewer found false for `process_stats`'s own child. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...-flush-to-close-inside-the-stops-budget.md | 49 ------------------- src/redlist.rs | 4 -- tests/common/power.rs | 4 +- .../toyos-rust-tests/src/bin/process_stats.rs | 4 -- tests/toyos-rust-tests/src/roster.rs | 3 +- tests/toyos.rs | 4 +- 6 files changed, 5 insertions(+), 63 deletions(-) delete mode 100644 issues/build/quiesce-leaves-the-volume-whole-needs-its-flush-to-close-inside-the-stops-budget.md diff --git a/issues/build/quiesce-leaves-the-volume-whole-needs-its-flush-to-close-inside-the-stops-budget.md b/issues/build/quiesce-leaves-the-volume-whole-needs-its-flush-to-close-inside-the-stops-budget.md deleted file mode 100644 index a1a8d73fced..00000000000 --- a/issues/build/quiesce-leaves-the-volume-whole-needs-its-flush-to-close-inside-the-stops-budget.md +++ /dev/null @@ -1,49 +0,0 @@ ---- -status: expected-red -kind: tooling -opened: 2026-09-28 ---- - -# `quiesce_leaves_the_volume_whole` passes only when the refused flush closes inside the stop's `PARK`, which a host stall takes away - -The verdict needs `fsync: … durable on attempt 9` on the console before -`Syncing filesystems...`. In other words, the `quiesce-fsync-refuse` ladder has -to close before `quiesce::stop` spends `PARK`. The -kernel's own `const` assert in `fat32_adapter::mirror_refuse` only covers the -parks: 1270 ms of `RETRY_SOONEST` doubling against `PARK`. The I/O of nine -attempts, and any time the guest is not running, are not covered by anything. -`PARK`'s own doc says a thread can outlast it and that the record then names -the shortfall. So the test asserts an outcome the kernel does not promise, and -under TCG the guest clock runs with the host's. - -The one red, the Fast tier for PR #562 at `2a9c77ee`: - -``` -refusals 1..8 at 630 643 658 681 723 809 974 1304 ms (the nominal ladder) -{0.649 init} init: power: the machine stops … -[kernel 2.703 cpu0] Syncing filesystems... -[kernel 2.705 cpu1 tid=1] fsync: /log/quiesce-fsync.bin durable on attempt 9 after 2073ms -stop: 6 of 7 userland thread(s) stopped … in 2037 ms of a 2010 ms budget -``` - -Attempt 9 was due at about 1944 ms (a 640 ms park after 1304). It closed at -2705 ms, two milliseconds after the stop's own deadline wake, which fired -27 ms late. A single stall of the whole guest from before 1944 ms to past -2676 ms explains both late wakes firing together. Nothing in the guest was -waiting on the other. Nothing here is PR #562's doing either: on that branch -the kernel paths this boot runs (`quiesce.rs`, `block.rs`, `fat32_adapter.rs`) -are the same as on `main`. The one change to `quiesce_fsync.rs` removes a -deadline the guest only reached on a hang. - -## Exit condition - -The verdict no longer rests on guest time. One way: a stop whose budget ran out -over the parked update is read as its own outcome, with the volume judged whole -or not by the checker. Another: the actuator holds the ladder open until the -stop has swept, rather than for a fixed ladder of parks. Then this file and its -`src/redlist.rs` row are deleted. - -## Owner - -`tests/common/volumes.rs` `quiesce_leaves_the_volume_whole`, the -`quiesce-fsync-refuse` actuator. Nobody holds it. diff --git a/src/redlist.rs b/src/redlist.rs index 87d2f4104f8..ba8865bd48a 100644 --- a/src/redlist.rs +++ b/src/redlist.rs @@ -60,10 +60,6 @@ pub const DISABLED: &[Disabled] = &[ test: "partition_claim_departure", issue: "issues/boot-media/partition-claim-departure-exits-clean-with-none-of-its-refusals-said.md", }, - Disabled { - test: "quiesce_leaves_the_volume_whole", - issue: "issues/build/quiesce-leaves-the-volume-whole-needs-its-flush-to-close-inside-the-stops-budget.md", - }, Disabled { test: "quiesce_stops_the_machine", issue: "issues/kernel/a-quiesce-writers-first-pass-outlasts-the-jobs-five-second-spin-up.md", diff --git a/tests/common/power.rs b/tests/common/power.rs index 9ff91cd6849..dbb1cea226a 100644 --- a/tests/common/power.rs +++ b/tests/common/power.rs @@ -352,10 +352,10 @@ pub fn quiesce_wakes_on_the_last_park( ); let at = |needle: &str| whole.lines().position(|line| line.contains(needle)); let stopped_at = whole.lines().position(|line| toyos_quiesce::Record::parse(line).is_some()); - let (Some(held_at), Some(synced_at)) = (at(&held), stopped_at) else { + let (Some(held_at), Some(stopped_at)) = (at(&held), stopped_at) else { return Err(format!("the kernel never held the thread it names ({held:?})\n{whole}")); }; - if held_at > synced_at { + if held_at > stopped_at { return Err(format!("the thread was held after the stop was over\n{whole}")); } // More than one sweep: the stop found the held thread running and had diff --git a/tests/toyos-rust-tests/src/bin/process_stats.rs b/tests/toyos-rust-tests/src/bin/process_stats.rs index 29855474ee1..f1ecdc123a0 100644 --- a/tests/toyos-rust-tests/src/bin/process_stats.rs +++ b/tests/toyos-rust-tests/src/bin/process_stats.rs @@ -299,8 +299,6 @@ fn a_wait_to_write_a_full_connection_is_ipc() { ); } -/// Runs `role` on a connection to this process, `release`s it once the -/// roster has its main thread blocked, and answers the exited child's numbers. fn parked_on_a_connection(role: &str, release: impl FnOnce(&Connection)) -> ProcessStats { let (acceptor, connector) = port::create().expect("a port"); let names = namespace::build().add(HELD_SERVICE, &connector).finish().expect("a namespace"); @@ -337,8 +335,6 @@ fn held_ipc() { assert_eq!(n, 1, "held-ipc: the parent's answer"); } -/// Fills the connection its namespace names, says it is running, then blocks -/// writing one more byte until the parent reads. fn held_ipc_write() { let conn = toyos::endow::service(HELD_SERVICE).expect("held-ipc-write: the parent's port"); let chunk = [0u8; 65536]; diff --git a/tests/toyos-rust-tests/src/roster.rs b/tests/toyos-rust-tests/src/roster.rs index d1d865496e4..8f4c55c0cbe 100644 --- a/tests/toyos-rust-tests/src/roster.rs +++ b/tests/toyos-rust-tests/src/roster.rs @@ -54,8 +54,7 @@ pub fn main_thread_blocked(pid: u32) -> bool { } /// Poll until `cond` holds. The bound is a hang guard and not a timing -/// assumption: every caller waits for a state the kernel has already decided, -/// and the `sysinfo` call inside `cond` is the loop's preemption point +/// assumption: the `sysinfo` call inside `cond` is the loop's preemption point /// (`thread::yield_now` is a spin hint on this platform). pub fn await_true(what: &str, cond: impl Fn() -> bool) { let give_up = Instant::now() + Duration::from_secs(5); diff --git a/tests/toyos.rs b/tests/toyos.rs index 21be367e6ad..c3368e425a3 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -937,8 +937,8 @@ const MACHINE_TESTS: &[(&str, Sched, Tier)] = &[ // by name, serves DATA absent, and the stick is untouched. Body in // `tests/common/storage.rs`. ("fsd_two_data", Sched::Parallel, Tier::Fast), - // A same-length overwrite on /home, the guest's read held against the image. Body in `tests/common/storage.rs`. - ("home_overwrite_reads_back", Sched::Parallel, Tier::Weekly), + // A same-length overwrite on /home, the guest's read held against the image, and the only guest control on init's stop sync. Body in `tests/common/storage.rs`. + ("home_overwrite_reads_back", Sched::Parallel, Tier::Nightly), // One filesystem under two paths: the guest writes under each of /apps and // /home, the host finds both in one volume on the image. Body in // `tests/common/storage.rs`. From b607ce7431f519af452a4786703f8b49684e1a79 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 08:46:01 +0200 Subject: [PATCH 48/54] Delete a parallel-red sighting of a test this branch deletes `issues/build/parallel-tests-red-under-other-suites.md` recorded `quiesce_leaves_the_volume_whole` red under `quiesce-fsync-refuse` with the kernel's log volume. The test, the actuator and that volume are all gone on this branch, so the sighting is deleted and not rewritten. Co-Authored-By: Claude Opus 5.5 --- .../build/parallel-tests-red-under-other-suites.md | 14 -------------- 1 file changed, 14 deletions(-) diff --git a/issues/build/parallel-tests-red-under-other-suites.md b/issues/build/parallel-tests-red-under-other-suites.md index 23580b7a723..021f82406b6 100644 --- a/issues/build/parallel-tests-red-under-other-suites.md +++ b/issues/build/parallel-tests-red-under-other-suites.md @@ -350,20 +350,6 @@ changes. twelve. `screen_loader_lines` grows 12 or 13 rows there for 14 or 15 lines; the lines between its two markers are the same on both trees, so what moves its growth is how far the loader has got when the GOP-query dump lands. - `quiesce_leaves_the_volume_whole` was red in four of the six branch runs and - none of `main`'s, each `ALONE … GREEN`, all under `quiesce-fsync-refuse`: twice - `FAT 1 differs from FAT 0 at entry 45` in the volume the stop left, once with - a cluster no entry reaches, and twice `log-volume: … was left with a chain its - entry does not reach: corrupt cluster chain` — the stop's second stage, since - removed, on a boot that now carries 13 MiB of ROOT where it carried 619. - Not the host: `main`'s own FAT refusal defects at `b0adc600`, which the - branch reaches by timing: a refused link write leaked the cluster - `append_cluster` had just claimed, and a refused free split the FATs; a - forced interleaving reddened both trees. #510 closed both with - `toyos-fat32/src/repair.rs`: the same forcing patch, alone, is red 2 of 2 on - this branch before the merge (`5b3cf8ae`, each of its four boots leaving one - cluster no directory entry reaches) and green 3 of 3 with `main` `5e36908c` - merged (`d5c2d9c9`). `metal_job_reboot` was red in one of the six, `ALONE` red too, and in one further branch run (`the job drain carried no kernel output at all (24 bytes)`), and green alone three times on each tree, alternating; From 806da129cd49e67bab5602e7e1f8729b35f47daa Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 12:52:44 +0200 Subject: [PATCH 49/54] metal_device_probe: read blockd's records from blockd's own lines On the T14 at b607ce74 metal_device_probe failed with nvme: no record carries "blockd: no NVMe controller this row names is on this machine" although the readback's log carries that line: {2026-09-29 08:43:12 1.162 blockd} blockd: no NVMe controller this row names is on this machine; serving no partition The judge handed `unmet` `Readback::kernel()`, which is the log with every program's line taken out (`bootlog::kernel_records`). On main the NVMe records were the kernel's; on this branch they are blockd's, a program's line, so the filter removed the one record the row owed, and `nvme-served` (NeverSays "blockd: partition ") could never have seen a served partition either. The unit fixture hid it by writing blockd's line as a kernel record. blockd's two records move to their own table, BLOCKD_RECORDS, judged against `bootlog::lines_of(log, "blockd")`; RECORDS stays judged against the kernel's records alone, now filtered inside `unmet`, which takes the whole log. A kernel line or another program saying blockd's words does not count, nor a program saying "Boot: complete (". A new test holds BLOCKD_RECORDS to blockd's source. The inventory reads the whole log so its "blockd: " row reports something. Oracle: the recorded T14 readback (target/metal/metaldevicecase), run through both readings as a temporary test: the old one reproduces the T14's single finding exactly, the new one reports none. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01U6SVYFkdvV2t38KzNrESxs --- src/metaldevices.rs | 73 +++++++++++++++++++++++++++++++++-------- tests/common/devices.rs | 4 +-- 2 files changed, 61 insertions(+), 16 deletions(-) diff --git a/src/metaldevices.rs b/src/metaldevices.rs index bebe745e9ac..83bbd06278f 100644 --- a/src/metaldevices.rs +++ b/src/metaldevices.rs @@ -99,7 +99,8 @@ use Presence::{NeverSays, Says}; /// The device inventory, asserted on every boot. /// -/// **Every needle is a prefix of a record the shipping kernel writes**, so +/// **Every needle is a prefix of a record the shipping kernel writes**, read +/// out of the kernel's records and never a program's line, so /// nothing here needs an actuator and nothing here is about a number. The /// counts and identities that *are* about this machine — how many controllers, /// which silicon, which mode — are read out and reported by [`inventory`], and @@ -117,16 +118,25 @@ pub const RECORDS: &[Record] = &[ Record { about: "i8042", needle: "i8042: ok selftest=0x55", presence: Says }, Record { about: "i8042-quarantine", needle: "i8042: quarantined", presence: NeverSays }, Record { about: "i8042-lost-edge", needle: "i8042: bytes with no IRQ record", presence: NeverSays }, - // The internal disk — the whole safety argument for running on this - // machine at all — never written: the kernel drives no NVMe, and the one - // process that may drives none here, since its row names only the - // emulated controller. - Record { about: "nvme", needle: NVME_UNDRIVEN, presence: Says }, - Record { about: "nvme-served", needle: "blockd: partition ", presence: NeverSays }, // The boot ended the way the loop's verdict needs it to. Record { about: "boot", needle: "Boot: complete (", presence: Says }, ]; +/// The name init starts the block service under, which is the tag `logd` +/// heads each of its lines with. +pub const BLOCKD: &str = "blockd"; + +/// What [`BLOCKD`]'s own lines must carry, read out of those and nothing else. +/// +/// The internal disk — the whole safety argument for running on this machine +/// at all — is never written: the kernel drives no NVMe, and the one process +/// that may drives none here, since its row names only the emulated +/// controller. +pub const BLOCKD_RECORDS: &[Record] = &[ + Record { about: "nvme", needle: NVME_UNDRIVEN, presence: Says }, + Record { about: "nvme-served", needle: "blockd: partition ", presence: NeverSays }, +]; + /// blockd's word that no NVMe controller its row names is on the machine, so /// nothing reads or writes the internal disk. pub const NVME_UNDRIVEN: &str = "blockd: no NVMe controller this row names is on this machine"; @@ -326,17 +336,23 @@ pub fn inventory(log: &str) -> Vec { } /// Every record that had to be there and was not, and every one that had to be -/// absent and was not — the loader's file and `logd`'s judged by their own -/// tables, plus the two whose *content* is the assertion. Each line opens with -/// the `about` it failed, so a caller can name them without reading the prose. +/// absent and was not — the loader's file, and `logd`'s whole, programs' lines +/// included, each table judged against its own writer's part of it. Each line +/// opens with the `about` it failed, so a caller can name them without reading +/// the prose. /// /// A `Vec` of failures and not a `Vec`: the device suite's numbers are /// priced in `tests/metal-profile.toml` and judged there, so what is left here /// is the half that is about records. pub fn unmet(loader: &str, log: &str) -> Vec { let mut out = Vec::new(); - for (record, text) in - RECORDS.iter().map(|r| (r, log)).chain(LOADER_RECORDS.iter().map(|r| (r, loader))) + let kernel = crate::bootlog::kernel_records(log); + let blockd = crate::bootlog::lines_of(log, BLOCKD); + for (record, text) in RECORDS + .iter() + .map(|r| (r, kernel.as_str())) + .chain(BLOCKD_RECORDS.iter().map(|r| (r, blockd.as_str()))) + .chain(LOADER_RECORDS.iter().map(|r| (r, loader))) { match (record.presence, text.contains(record.needle)) { (Says, false) => { @@ -387,7 +403,9 @@ mod tests { line("0.090", "xHCI: max_slots=32 max_ports=16 ctx_size=64 pagesize=0x1"), line("0.140", "usb-storage: 1 device(s)"), line("0.150", "i8042: ok selftest=0x55 cfg=0x45->0x44 port1=ok port2=ok"), - line("0.200", "blockd: no NVMe controller this row names is on this machine; serving no partition"), + "{2026-09-29 08:43:12 1.162 blockd} blockd: no NVMe controller this row names is on \ + this machine; serving no partition\n" + .to_string(), line("0.319", "GOP: scanout memory type WC (MTRR WB, PAT entry 4)"), line("0.368", "Boot: complete (368ms)"), line("1.100", "exit: usbwrite pid=6 code=402000 cpu=140ms"), @@ -492,8 +510,22 @@ mod tests { assert_eq!(about(unmet(&a_good_loader(), &uncached)), ["scanout"]); // A partition of the internal disk served, which is where a write to // it would come from. - let served = format!("{}{}", a_good_boot(), line("0.3", "blockd: partition 0000 at block 256")); + let served = format!( + "{}{{2026-09-29 08:43:12 1.300 blockd}} blockd: partition 0000 at block 256\n", + a_good_boot() + ); assert_eq!(about(unmet(&a_good_loader(), &served)), ["nvme-served"]); + // blockd's word in any other writer's mouth is not blockd's, and a + // kernel record in a program's is not the kernel's. + for writer in ["[2026-09-29 08:43:12 1.162 cpu0]", "{2026-09-29 08:43:12 1.162 init}"] { + let forged = a_good_boot().replace("{2026-09-29 08:43:12 1.162 blockd}", writer); + assert_eq!(about(unmet(&a_good_loader(), &forged)), ["nvme"], "{writer}"); + } + let unbooted = a_good_boot().replace( + &line("0.368", "Boot: complete (368ms)"), + "{2026-09-29 08:43:12 0.368 init} Boot: complete (368ms)\n", + ); + assert_eq!(about(unmet(&a_good_loader(), &unbooted)), ["boot"]); // A shutdown that never emptied a cache. let unflushed = a_good_loader() .lines() @@ -581,6 +613,19 @@ mod tests { } } + /// blockd's table is held to blockd's own source, since a needle it never + /// writes passes a `NeverSays` and fails every `Says`. + #[test] + fn blockd_writes_the_lines_its_table_reads() { + let at = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("userland/blockd/src/main.rs"); + let written = code_of(&std::fs::read_to_string(&at).expect("blockd's own source")); + for record in BLOCKD_RECORDS { + let needle = record.needle; + assert!(written.contains(needle), "{} does not write {needle:?}", at.display()); + } + } + /// A line the kernel only *talks* about is not a line the kernel writes: /// every needle above appears in that file's own prose as well, so a scan /// over the raw source passes on a kernel that emits none of them. diff --git a/tests/common/devices.rs b/tests/common/devices.rs index f87e86581a8..7e5dff4412d 100644 --- a/tests/common/devices.rs +++ b/tests/common/devices.rs @@ -34,7 +34,7 @@ const WAIT: std::time::Duration = std::time::Duration::from_secs(120); pub fn on_metal(back: &metal::Readback) -> Result<(), String> { let root = super::compile::repo_root(); let profile = Profile::load(&root).map_err(|why| why.to_string())?; - let mut bad = metaldevices::unmet(back.loader().text(), back.kernel().text()); + let mut bad = metaldevices::unmet(back.loader().text(), back.log().text()); for job in JOBS { let code = match back.exit_code(job) { @@ -59,7 +59,7 @@ pub fn on_metal(back: &metal::Readback) -> Result<(), String> { // ceiling or writes an inventory row next has to read, and they exist only // in a log that came off the machine. eprintln!(" [devices] what {} answered:", back.label); - for line in metaldevices::inventory(back.kernel().text()) { + for line in metaldevices::inventory(back.log().text()) { eprintln!(" {line}"); } From 5337e7914e928a6b1152c3573d557c0581379cc9 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 15:20:14 +0200 Subject: [PATCH 50/54] fsd stamps UTC nanoseconds off one wall-clock reader #587's contract: a file's mtime is nanoseconds since the Unix epoch, UTC, DATA keeps the nanosecond, and 0 is undated. fsd stamped DATA with `clock_epoch()` times 10^9, whole seconds, so two writes to `/home` inside one second carried one mtime. It also read the clock a second time for FAT (`utc_secs`), with the same undated rule in another unit. - `fsd::volume::now_nanos` is the one reader. It derives the kernel's anchor exactly: `SYS_CLOCK_EPOCH` is `BOOT_SECS` plus the whole seconds of the counter at the call, and the process's clock page is the same counter on the kernel's formula, so a call bracketed by two readings inside one second names `BOOT_SECS`. A bracket that straddles a second is asked again, and three that straddle panic by name. The anchor is taken once: the kernel sets it once, before the first process. A refused clock is undated, 0. - FAT's volume takes the same clock and divides it, as main's `stamp(mtime_now())` did. `utc_secs` is deleted. - `file_mtime` judges `/home` too: two writes, the second strictly later, and not both whole seconds. `file_mtime undated` writes `/home` beside `/tmp` and requires both undated. - The host test `the_anchor_is_the_kernels` holds the derivation to the kernel's arithmetic, a straddled bracket included. - userlands-wall-clock-...md: the sentence that a file server can stamp only whole seconds is deleted; fsd stamps nanoseconds now. - home_overwrite_zero.rs: the comment that `fs::metadata` is the kernel's `file_cache::size` is deleted; `/home` is fsd's. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01U6SVYFkdvV2t38KzNrESxs --- ...le-seconds-and-an-mtime-has-nanoseconds.md | 5 +- tests/common/wallclock.rs | 12 ++-- tests/toyos-rust-tests/src/bin/file_mtime.rs | 48 ++++++++++---- .../src/bin/home_overwrite_zero.rs | 1 - userland/fsd/src/fat.rs | 10 +-- userland/fsd/src/main.rs | 8 +-- userland/fsd/src/volume.rs | 64 ++++++++++++++++++- 7 files changed, 112 insertions(+), 36 deletions(-) diff --git a/issues/filesystem/userlands-wall-clock-has-whole-seconds-and-an-mtime-has-nanoseconds.md b/issues/filesystem/userlands-wall-clock-has-whole-seconds-and-an-mtime-has-nanoseconds.md index 58930bd1c52..81c046a3a2e 100644 --- a/issues/filesystem/userlands-wall-clock-has-whole-seconds-and-an-mtime-has-nanoseconds.md +++ b/issues/filesystem/userlands-wall-clock-has-whole-seconds-and-an-mtime-has-nanoseconds.md @@ -15,10 +15,7 @@ and `gettimeofday` (`userland/libc/src/time.rs`) all round down to the second. So a file written a moment ago reads as up to a second in the future against the program's own "now", which a tool comparing at nanoseconds — GNU make's -"modification time in the future" check — reports as clock skew. And a file -server in userland can stamp only what it can read — a stamp it takes off -`SYS_CLOCK_EPOCH` is a whole second, so two writes inside one second carry one -mtime, which a build tool reads as "not newer". +"modification time in the future" check — reports as clock skew. **Exit condition.** Userland reads the wall clock at the resolution the kernel stamps at — the boot's UTC anchor published where a process reads the diff --git a/tests/common/wallclock.rs b/tests/common/wallclock.rs index 5f8cb5996f3..4ef66c03271 100644 --- a/tests/common/wallclock.rs +++ b/tests/common/wallclock.rs @@ -552,14 +552,16 @@ pub fn file_mtime_survives_a_reboot( Ok(()) } -/// On a machine whose RTC never answered, a file's mtime is undated — 0, which -/// std reports as an error — and never 1970 plus the boot's uptime. +/// On a machine whose RTC never answered, a file's mtime on `/tmp` or on fsd's +/// `/home` is undated — 0, which std reports as an error — and never 1970 plus +/// the boot's uptime. pub fn file_mtime_undated( test_config: &Path, c_bins: &[(String, Vec)], rust_bins: &[(String, Vec)], ) -> Result<(), String> { - const SAID: &str = "file-mtime: /tmp/file-mtime-undated is undated"; + const SAID: [&str; 2] = + ["file-mtime: /tmp/file-mtime-undated is undated", "file-mtime: /home/file-mtime-undated is undated"]; let mut qemu = QemuInstance::boot_with_options( test_config, c_bins, @@ -583,12 +585,12 @@ pub fn file_mtime_undated( return Err(format!("{bad:?} on the way down\n{tail}")); } } - if result.exit_code != Some(0) || !result.stdout.contains(SAID) { + if result.exit_code != Some(0) || !SAID.iter().all(|said| result.stdout.contains(said)) { return Err(format!( "`file_mtime undated` exited {:?}:\n{}\nkernel log while it ran:\n{}{}", result.exit_code, result.stdout, result.before, result.serial )); } - eprintln!(" [clock] rtc-dead: a file written in /tmp is undated"); + eprintln!(" [clock] rtc-dead: a file written in /tmp or /home is undated"); Ok(()) } diff --git a/tests/toyos-rust-tests/src/bin/file_mtime.rs b/tests/toyos-rust-tests/src/bin/file_mtime.rs index d371d818018..3ec6dd97244 100644 --- a/tests/toyos-rust-tests/src/bin/file_mtime.rs +++ b/tests/toyos-rust-tests/src/bin/file_mtime.rs @@ -5,12 +5,13 @@ //! between two `SYS_CLOCK_EPOCH` readings taken around what made it — a write, //! a create with truncation, a create of a missing file, a truncation — and a //! later write's stamp is later, which a clock of whole seconds cannot say. +//! Then `/home`, fsd's DATA, which keeps the nanosecond too. //! Then `/log`: FAT keeps the whole seconds of its flush, read back after a //! reopen. //! `write ` makes the first judgement on `path` and `read ` only //! prints what it finds there. `undated` runs on a machine whose RTC never -//! answered and never asks the time: a file written there is undated, which -//! std reports as an error and not as 1970. +//! answered and never asks the time: a file written there, on `/tmp` or on +//! `/home`, is undated, which std reports as an error and not as 1970. use std::fs::{self, OpenOptions}; use std::io::{ErrorKind, Write}; @@ -100,6 +101,23 @@ fn main() { {resized} ns" ); + let home_first = write_judged("/home/file-mtime-first", b"first"); + let home_second = write_judged("/home/file-mtime-second", b"second"); + assert!( + home_second > home_first, + "on /home a write after another is stamped {home_second} ns and the one before \ + it {home_first} ns" + ); + // A whole second is one stamp in 10⁹; two of them are a clock of seconds. + assert!( + home_first % NANOS_PER_SEC != 0 || home_second % NANOS_PER_SEC != 0, + "/home stamps {home_first} ns and {home_second} ns: whole seconds, not the \ + nanosecond DATA keeps" + ); + for path in ["/home/file-mtime-first", "/home/file-mtime-second"] { + fs::remove_file(path).unwrap_or_else(|e| panic!("remove {path}: {e}")); + } + // The flush stamps FAT, which keeps whole seconds and drops an odd one. const FAT: &str = "/log/file-mtime"; let before = epoch(); @@ -114,20 +132,24 @@ fn main() { before its write and {after} s after", ); fs::remove_file(FAT).unwrap_or_else(|e| panic!("remove {FAT}: {e}")); - println!("file-mtime: /tmp stamps {first} then {second}, /log {fat}"); + println!( + "file-mtime: /tmp stamps {first} then {second}, /home {home_first} then \ + {home_second}, /log {fat}" + ); } [_, mode] if mode == "undated" => { - const UNDATED: &str = "/tmp/file-mtime-undated"; - write(UNDATED, b"no clock answered"); - let meta = fs::metadata(UNDATED).unwrap_or_else(|e| panic!("stat {UNDATED}: {e}")); - match meta.modified() { - Err(e) if e.kind() == ErrorKind::Unsupported => {} - other => panic!( - "{UNDATED} was written on a machine whose RTC never answered, and its mtime \ - reads {other:?}" - ), + for undated in ["/tmp/file-mtime-undated", "/home/file-mtime-undated"] { + write(undated, b"no clock answered"); + let meta = fs::metadata(undated).unwrap_or_else(|e| panic!("stat {undated}: {e}")); + match meta.modified() { + Err(e) if e.kind() == ErrorKind::Unsupported => {} + other => panic!( + "{undated} was written on a machine whose RTC never answered, and its \ + mtime reads {other:?}" + ), + } + println!("file-mtime: {undated} is undated"); } - println!("file-mtime: {UNDATED} is undated"); } [_, mode, path] if mode == "write" => { let stamp = write_judged(path, b"stamped at its write"); diff --git a/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs b/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs index 17c818746d3..cd4682d4c14 100644 --- a/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs +++ b/tests/toyos-rust-tests/src/bin/home_overwrite_zero.rs @@ -56,7 +56,6 @@ fn pinned_overwrite(first: &[u8], second: &[u8]) { // One reading and no window: the device the host reads after the shutdown's // drain is the time-free judge, and this length is what it is held against. - // `fs::metadata` is the `file_cache::size` a read bounds itself by. let mut lowest = fs::metadata(PINNED).unwrap_or_else(|e| panic!("stat {PINNED}: {e}")).len(); let mut got = Vec::new(); File::open(PINNED) diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index eeff485a98a..1e39a13a414 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -25,7 +25,7 @@ use toyos_fat32::{BlockAccess, Error, Fat32, FatTime, IoError}; use crate::cache::Cache; use crate::disk::{Disk, BLOCK}; -use crate::volume::{parent, Kind, Meta, Node, OpenHow, Out, Volume}; +use crate::volume::{parent, Kind, Meta, Node, OpenHow, Out, Volume, NANOS_PER_SEC}; /// The most entries one directory listing materialises. const MAX_LIST: usize = 16_384; @@ -93,7 +93,9 @@ pub struct FatVolume { open: BTreeMap, by_path: BTreeMap, next: Node, - /// Seconds since the epoch this volume stamps its entries with. + /// What this volume stamps its entries with, [`crate::volume::now_nanos`]'s + /// unit. FAT specifies local time; this stamps UTC because the owner ruled + /// the hardware clock is UTC. clock: fn() -> u64, /// Where a read lands before it goes out: `toyos-fat32` reads into a /// slice, and a client's window is never one. Kept, so a read allocates @@ -152,7 +154,7 @@ impl FatVolume { } fn time(&self) -> FatTime { - FatTime::from_unix_secs((self.clock)()) + FatTime::from_unix_secs((self.clock)() / NANOS_PER_SEC) } fn entry(&mut self, node: Node) -> Result<&mut Open, SyscallError> { @@ -511,7 +513,7 @@ mod tests { let mut ram = Ram::new((blocks + 2 * CLEAN_LIMIT) as u64); ram.write(0, &image).unwrap(); let refused = Rc::new(Cell::new(u64::MAX)); - let v = FatVolume::mount(Refusing { ram, refused: Rc::clone(&refused) }, true, || 1_717_245_296).unwrap(); + let v = FatVolume::mount(Refusing { ram, refused: Rc::clone(&refused) }, true, || 1_717_245_296 * NANOS_PER_SEC).unwrap(); (v, refused, spec) } diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index 1cd9cb009cc..703c5121585 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -166,12 +166,6 @@ struct Stream { offset: u64, } -/// Seconds since the epoch, UTC. FAT specifies local time; this stamps UTC -/// because the owner ruled the hardware clock is UTC. -fn utc_secs() -> u64 { - toyos_abi::syscall::clock_epoch().unwrap_or(0) -} - fn main() { let args: Vec = std::env::args().collect(); let role = args.get(1).and_then(|r| Role::parse(r)).unwrap_or_else(|| { @@ -368,7 +362,7 @@ fn ram(roots: &[&str], why: &str) -> Box { } fn fat_on(disk: D, writable: bool) -> Result, String> { - FatVolume::mount(disk, writable, utc_secs).map(|v| Box::new(v) as Box) + FatVolume::mount(disk, writable, fsd::volume::now_nanos).map(|v| Box::new(v) as Box) } struct Server { diff --git a/userland/fsd/src/volume.rs b/userland/fsd/src/volume.rs index d084a4123a3..abd8a5215dd 100644 --- a/userland/fsd/src/volume.rs +++ b/userland/fsd/src/volume.rs @@ -7,6 +7,9 @@ //! node whose file was unlinked or renamed over answers `Gone` from then on: //! its blocks may be another file's. +use std::sync::OnceLock; + +use toyos_abi::clock::nanos_since_boot; use toyos_abi::syscall::SyscallError; #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -142,7 +145,64 @@ pub fn join(dir: &str, name: &str) -> String { } } -/// Nanoseconds since the Unix epoch, now, to the second the clock reads. +pub const NANOS_PER_SEC: u64 = 1_000_000_000; + +/// What a file written now is stamped with (`toyos_abi::syscall::Stat::mtime`): +/// nanoseconds since the Unix epoch, UTC, as the kernel's `clock::mtime_now` +/// reckons them, and 0 — undated — on a machine whose RTC never answered. pub fn now_nanos() -> u64 { - toyos_abi::syscall::clock_epoch().map_or(0, |secs| secs.saturating_mul(1_000_000_000)) + static ANCHOR: OnceLock> = OnceLock::new(); + let anchor = ANCHOR.get_or_init(|| { + anchor(|| (nanos_since_boot(), toyos_abi::syscall::clock_epoch(), nanos_since_boot())) + }); + anchor.map_or(0, |secs| secs.saturating_mul(NANOS_PER_SEC).saturating_add(nanos_since_boot())) +} + +/// The Unix second at the counter's nanosecond zero, the kernel's `BOOT_SECS`, +/// out of `read`'s counter reading, `SYS_CLOCK_EPOCH` and counter reading. +/// The epoch is the anchor plus the whole seconds of the counter at the call, +/// so readings on either side within one second name it exactly. `None` is the +/// clock refused. +fn anchor(mut read: impl FnMut() -> (u64, Option, u64)) -> Option { + for _ in 0..3 { + let (before, epoch, after) = read(); + let epoch = epoch?; + if before / NANOS_PER_SEC == after / NANOS_PER_SEC { + return Some(epoch - before / NANOS_PER_SEC); + } + } + panic!("fsd: three calls to the wall clock each straddled a second of the counter"); +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The kernel's arithmetic, `BOOT_SECS + ⌊n/10⁹⌋` at the counter reading + /// `n` its call took: the anchor comes back exactly, and a call whose + /// readings straddle a second is asked again rather than guessed at. + #[test] + fn the_anchor_is_the_kernels() { + const BOOT_SECS: u64 = 2_000_000_000; + let kernel = |n: u64| Some(BOOT_SECS + n / NANOS_PER_SEC); + + assert_eq!(anchor(|| (4_200_000_000, kernel(4_500_000_000), 4_700_000_000)), Some(BOOT_SECS)); + + let mut calls = [(4_900_000_000, 5_000_000_100, 5_100_000_000), (5_200_000_000, 5_250_000_000, 5_300_000_000)] + .into_iter(); + let straddled = anchor(|| { + let (before, at, after) = calls.next().expect("asked at most twice"); + (before, kernel(at), after) + }); + assert_eq!(straddled, Some(BOOT_SECS)); + assert_eq!(calls.next(), None, "the straddled call was asked again"); + + assert_eq!(anchor(|| (1, None, 2)), None); + } + + #[test] + #[should_panic(expected = "each straddled a second")] + fn a_clock_that_always_straddles_is_refused_by_name() { + anchor(|| (999_999_999, Some(1), 1_000_000_000)); + } } From 607af439326976f4a9ff00e55e88d32e076bec0c Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 16:34:13 +0200 Subject: [PATCH 51/54] fsd refuses an unreadable FAT entry's mtime, and /home is judged whole A stat of an open FAT file whose directory entry the device refused answered mtime 0, which is "undated", so a device error on /log reached std as Unsupported and not as Io. lstat and node_meta now refuse it as every other metadata read in fat.rs does; the host test an_unreadable_entry_is_io_and_not_undated refuses the root directory's block past the cache and wants Io from both. file_mtime judged only writes on /home. The /tmp block's acts - two writes, a create with truncation, a create of a missing file and a truncation - are one function now, run on /tmp and on /home, so an fsd that leaves a grown truncation or a create unstamped is red. fsd takes the wall clock's anchor at its start, before it holds any client's data, rather than at the first stamp; now_nanos before it is a panic. DATA's second mtime lookup after the entry's extents were found refuses a missing leaf as NotFound instead of answering undated. FAT's seconds become nanoseconds through NANOS_PER_SEC. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01U6SVYFkdvV2t38KzNrESxs --- tests/toyos-rust-tests/src/bin/file_mtime.rs | 104 +++++++++---------- userland/fsd/src/data.rs | 4 +- userland/fsd/src/fat.rs | 28 +++-- userland/fsd/src/main.rs | 1 + userland/fsd/src/volume.rs | 14 ++- 5 files changed, 87 insertions(+), 64 deletions(-) diff --git a/tests/toyos-rust-tests/src/bin/file_mtime.rs b/tests/toyos-rust-tests/src/bin/file_mtime.rs index 3ec6dd97244..9eeb75419d2 100644 --- a/tests/toyos-rust-tests/src/bin/file_mtime.rs +++ b/tests/toyos-rust-tests/src/bin/file_mtime.rs @@ -5,7 +5,7 @@ //! between two `SYS_CLOCK_EPOCH` readings taken around what made it — a write, //! a create with truncation, a create of a missing file, a truncation — and a //! later write's stamp is later, which a clock of whole seconds cannot say. -//! Then `/home`, fsd's DATA, which keeps the nanosecond too. +//! Then the same on `/home`, fsd's DATA, which keeps the nanosecond too. //! Then `/log`: FAT keeps the whole seconds of its flush, read back after a //! reopen. //! `write ` makes the first judgement on `path` and `read ` only @@ -19,11 +19,11 @@ use std::time::UNIX_EPOCH; const NANOS_PER_SEC: u64 = 1_000_000_000; -fn mtime(path: &str) -> u64 { +fn mtime(path: &str, after: &str) -> u64 { let modified = fs::metadata(path) - .unwrap_or_else(|e| panic!("stat {path}: {e}")) + .unwrap_or_else(|e| panic!("stat {path} after {after}: {e}")) .modified() - .unwrap_or_else(|e| panic!("{path} has no mtime: {e}")); + .unwrap_or_else(|e| panic!("{path} has no mtime after {after}: {e}")); let since = modified .duration_since(UNIX_EPOCH) .unwrap_or_else(|_| panic!("{path}'s mtime is before the epoch")); @@ -46,7 +46,7 @@ fn judged(path: &str, what: &str, act: impl FnOnce()) -> u64 { let before = epoch(); act(); let after = epoch(); - let stamp = mtime(path); + let stamp = mtime(path, what); // `SYS_CLOCK_EPOCH` is the whole seconds of the clock that stamps, so the // stamp is at or past `before` and short of the second after `after`. assert!( @@ -61,68 +61,68 @@ fn write_judged(path: &str, bytes: &[u8]) -> u64 { judged(path, "a write", || write(path, bytes)) } -fn main() { - let args: Vec = std::env::args().collect(); - match args.as_slice() { - [_] => { - let first = write_judged("/tmp/file-mtime-first", b"first"); - let second = write_judged("/tmp/file-mtime-second", b"second"); - assert!( - second > first, - "a write after another is stamped {second} ns and the one before it {first} ns" - ); +/// Judges in `dir` a write, a later write, a create with truncation, a create +/// of a missing file and a truncation; returns the two writes' stamps. +fn dated(dir: &str) -> (u64, u64) { + let first = write_judged(&format!("{dir}/file-mtime-first"), b"first"); + let second = write_judged(&format!("{dir}/file-mtime-second"), b"second"); + assert!( + second > first, + "on {dir} a write after another is stamped {second} ns and the one before it {first} ns" + ); - const TRUNCATED: &str = "/tmp/file-mtime-truncated"; - let created = judged(TRUNCATED, "a create with truncation", || { - fs::File::create(TRUNCATED).unwrap_or_else(|e| panic!("create {TRUNCATED}: {e}")); - }); + let truncated = format!("{dir}/file-mtime-truncated"); + let created = judged(&truncated, "a create with truncation", || { + fs::File::create(&truncated).unwrap_or_else(|e| panic!("create {truncated}: {e}")); + }); - const MISSING: &str = "/tmp/file-mtime-missing"; - assert!(fs::metadata(MISSING).is_err(), "{MISSING} exists before this creates it"); - judged(MISSING, "a create of a missing file", || { - OpenOptions::new() - .write(true) - .create(true) - .open(MISSING) - .unwrap_or_else(|e| panic!("create {MISSING}: {e}")); - }); + let missing = format!("{dir}/file-mtime-missing"); + assert!(fs::metadata(&missing).is_err(), "{missing} exists before this creates it"); + judged(&missing, "a create of a missing file", || { + OpenOptions::new() + .write(true) + .create(true) + .open(&missing) + .unwrap_or_else(|e| panic!("create {missing}: {e}")); + }); - let resized = judged(TRUNCATED, "a truncation", || { - let f = OpenOptions::new() - .write(true) - .open(TRUNCATED) - .unwrap_or_else(|e| panic!("reopen {TRUNCATED}: {e}")); - f.set_len(1).unwrap_or_else(|e| panic!("ftruncate {TRUNCATED}: {e}")); - f.sync_all().unwrap_or_else(|e| panic!("fsync {TRUNCATED}: {e}")); - }); - assert!( - resized > created, - "{TRUNCATED} was created at {created} ns and a later truncation stamped it \ - {resized} ns" - ); + let resized = judged(&truncated, "a truncation", || { + let f = OpenOptions::new() + .write(true) + .open(&truncated) + .unwrap_or_else(|e| panic!("reopen {truncated}: {e}")); + f.set_len(1).unwrap_or_else(|e| panic!("ftruncate {truncated}: {e}")); + f.sync_all().unwrap_or_else(|e| panic!("fsync {truncated}: {e}")); + }); + assert!( + resized > created, + "{truncated} was created at {created} ns and a later truncation stamped it {resized} ns" + ); + (first, second) +} - let home_first = write_judged("/home/file-mtime-first", b"first"); - let home_second = write_judged("/home/file-mtime-second", b"second"); - assert!( - home_second > home_first, - "on /home a write after another is stamped {home_second} ns and the one before \ - it {home_first} ns" - ); +fn main() { + let args: Vec = std::env::args().collect(); + match args.as_slice() { + [_] => { + let (first, second) = dated("/tmp"); + let (home_first, home_second) = dated("/home"); // A whole second is one stamp in 10⁹; two of them are a clock of seconds. assert!( home_first % NANOS_PER_SEC != 0 || home_second % NANOS_PER_SEC != 0, "/home stamps {home_first} ns and {home_second} ns: whole seconds, not the \ nanosecond DATA keeps" ); - for path in ["/home/file-mtime-first", "/home/file-mtime-second"] { - fs::remove_file(path).unwrap_or_else(|e| panic!("remove {path}: {e}")); + for name in ["first", "second", "truncated", "missing"] { + let path = format!("/home/file-mtime-{name}"); + fs::remove_file(&path).unwrap_or_else(|e| panic!("remove {path}: {e}")); } // The flush stamps FAT, which keeps whole seconds and drops an odd one. const FAT: &str = "/log/file-mtime"; let before = epoch(); write(FAT, b"on FAT"); - let fat = mtime(FAT); + let fat = mtime(FAT, "its write"); let after = epoch(); assert!( (before - 1) * NANOS_PER_SEC <= fat @@ -156,7 +156,7 @@ fn main() { println!("file-mtime: {path} mtime={stamp}"); } [_, mode, path] if mode == "read" => { - println!("file-mtime: {path} mtime={}", mtime(path)); + println!("file-mtime: {path} mtime={}", mtime(path, "the boot that wrote it")); } _ => panic!("usage: file_mtime [undated | write | read ], got {args:?}"), } diff --git a/userland/fsd/src/data.rs b/userland/fsd/src/data.rs index dc8d3b07664..7739991518c 100644 --- a/userland/fsd/src/data.rs +++ b/userland/fsd/src/data.rs @@ -356,7 +356,7 @@ impl Volume for DataVolume { return Ok(Meta { kind, size: open.size, mtime: open.mtime }); } let (_, size) = mapped("a lookup", path, self.fs.file_extents(path))?.ok_or(SyscallError::NotFound)?; - let mtime = mapped("a lookup", path, self.fs.file_mtime(path))?.unwrap_or(0); + let mtime = mapped("a lookup", path, self.fs.file_mtime(path))?.ok_or(SyscallError::NotFound)?; Ok(Meta { kind, size, mtime }) } None if self.is_dir(path) => Ok(Meta { kind: Kind::Dir, size: 0, mtime: 0 }), @@ -415,7 +415,7 @@ impl Volume for DataVolume { Some(Kind::File) => { let (extents, size) = mapped("open", path, self.fs.file_extents(path))?.ok_or(SyscallError::NotFound)?; - let mtime = mapped("open", path, self.fs.file_mtime(path))?.unwrap_or(0); + let mtime = mapped("open", path, self.fs.file_mtime(path))?.ok_or(SyscallError::NotFound)?; self.new_node(path, extents, size, mtime) } // A link the resolver did not follow is one with nothing behind it. diff --git a/userland/fsd/src/fat.rs b/userland/fsd/src/fat.rs index 1e39a13a414..5eb5bbbf496 100644 --- a/userland/fsd/src/fat.rs +++ b/userland/fsd/src/fat.rs @@ -207,15 +207,15 @@ impl Volume for FatVolume { } if let Some(open) = self.by_path.get(path).and_then(|n| self.open.get(n)) { let len = open.file.len(); - let mtime = self.fs.metadata(path).map(|m| m.modified_unix).unwrap_or(0); - return Ok(Meta { kind: Kind::File, size: len, mtime: mtime * 1_000_000_000 }); + let mtime = self.fs.metadata(path).map_err(|e| logged("metadata", path, e))?.modified_unix; + return Ok(Meta { kind: Kind::File, size: len, mtime: mtime * NANOS_PER_SEC }); } let meta = self.fs.metadata(path).map_err(|e| match e { Error::NotADirectory => SyscallError::NotFound, e => logged("metadata", path, e), })?; let kind = if meta.is_dir { Kind::Dir } else { Kind::File }; - Ok(Meta { kind, size: if meta.is_dir { 0 } else { meta.len }, mtime: meta.modified_unix * 1_000_000_000 }) + Ok(Meta { kind, size: if meta.is_dir { 0 } else { meta.len }, mtime: meta.modified_unix * NANOS_PER_SEC }) } fn read_link(&mut self, path: &str) -> Result { @@ -229,7 +229,7 @@ impl Volume for FatVolume { .into_iter() .map(|e| { let kind = if e.is_dir { Kind::Dir } else { Kind::File }; - (e.name, Meta { kind, size: if e.is_dir { 0 } else { e.len }, mtime: e.modified_unix * 1_000_000_000 }) + (e.name, Meta { kind, size: if e.is_dir { 0 } else { e.len }, mtime: e.modified_unix * NANOS_PER_SEC }) }) .collect()) } @@ -305,8 +305,8 @@ impl Volume for FatVolume { fn node_meta(&mut self, node: Node) -> Result { let open = self.entry(node)?; let (path, size) = (open.path.clone(), open.file.len()); - let mtime = self.fs.metadata(&path).map(|m| m.modified_unix).unwrap_or(0); - Ok(Meta { kind: Kind::File, size, mtime: mtime * 1_000_000_000 }) + let mtime = self.fs.metadata(&path).map_err(|e| logged("metadata", &path, e))?.modified_unix; + Ok(Meta { kind: Kind::File, size, mtime: mtime * NANOS_PER_SEC }) } fn read(&mut self, node: Node, offset: u64, out: &mut dyn Out) -> Result { @@ -563,6 +563,22 @@ mod tests { assert_eq!(v.lstat("sub/b.txt").unwrap().size, 6000); } + /// An open file's entry the device will not read is `Io` to a stat, + /// never mtime 0, which is "undated". + #[test] + fn an_unreadable_entry_is_io_and_not_undated() { + const CREATE: OpenHow = OpenHow { create: true, create_new: false, truncate: false }; + let (mut v, refused, _) = spec_fat(); + let a = v.open("a.txt", CREATE).unwrap(); + v.write(a, 0, &[1; 3000]).unwrap(); + assert_eq!(v.sync(), Ok(Vec::new())); + evict(&v); + refused.set((spec_volume::cluster_offset(spec_volume::ROOT_CLUSTER) / BLOCK) as u64); + + assert_eq!(v.lstat("a.txt"), Err(SyscallError::Io)); + assert_eq!(v.node_meta(a), Err(SyscallError::Io)); + } + /// What the driver says of a volume that stopped answering is `Io`, which /// no caller takes for a name that is not there. #[test] diff --git a/userland/fsd/src/main.rs b/userland/fsd/src/main.rs index 703c5121585..44a3b3e910c 100644 --- a/userland/fsd/src/main.rs +++ b/userland/fsd/src/main.rs @@ -194,6 +194,7 @@ fn main() { extra => panic!("fsd: a second partition, {extra}, after {guid:?}"), } } + fsd::volume::take_anchor(); let caps = capabilities(role); let roots: Vec = caps.iter().map(|c| c.root.clone()).collect(); let roots: Vec<&str> = roots.iter().map(String::as_str).collect(); diff --git a/userland/fsd/src/volume.rs b/userland/fsd/src/volume.rs index abd8a5215dd..11b1bde4240 100644 --- a/userland/fsd/src/volume.rs +++ b/userland/fsd/src/volume.rs @@ -147,14 +147,20 @@ pub fn join(dir: &str, name: &str) -> String { pub const NANOS_PER_SEC: u64 = 1_000_000_000; +static ANCHOR: OnceLock> = OnceLock::new(); + +/// Anchors [`now_nanos`]. fsd does it once, at its start: a clock that will +/// not anchor is a panic, and there it holds no client's data. +pub fn take_anchor() { + let taken = anchor(|| (nanos_since_boot(), toyos_abi::syscall::clock_epoch(), nanos_since_boot())); + ANCHOR.set(taken).expect("the anchor is taken once"); +} + /// What a file written now is stamped with (`toyos_abi::syscall::Stat::mtime`): /// nanoseconds since the Unix epoch, UTC, as the kernel's `clock::mtime_now` /// reckons them, and 0 — undated — on a machine whose RTC never answered. pub fn now_nanos() -> u64 { - static ANCHOR: OnceLock> = OnceLock::new(); - let anchor = ANCHOR.get_or_init(|| { - anchor(|| (nanos_since_boot(), toyos_abi::syscall::clock_epoch(), nanos_since_boot())) - }); + let anchor = ANCHOR.get().expect("fsd takes the anchor at its start"); anchor.map_or(0, |secs| secs.saturating_mul(NANOS_PER_SEC).saturating_add(nanos_since_boot())) } From 1be7b136d6392155e8cec4edd0b1ecaa40db03c4 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 16:37:15 +0200 Subject: [PATCH 52/54] issues: the NVMe widening goes with the kernel's NVMe driver "A widening ... which keeps NVMe" is false once this branch deletes the kernel's NVMe driver, and with it the line counting two constraints. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01U6SVYFkdvV2t38KzNrESxs --- issues/kernel/every-driver-is-still-in-the-kernel.md | 6 ------ 1 file changed, 6 deletions(-) diff --git a/issues/kernel/every-driver-is-still-in-the-kernel.md b/issues/kernel/every-driver-is-still-in-the-kernel.md index e6bf7cd67f1..ac4f86ca87a 100644 --- a/issues/kernel/every-driver-is-still-in-the-kernel.md +++ b/issues/kernel/every-driver-is-still-in-the-kernel.md @@ -44,12 +44,6 @@ What is left of the staged work: `SYS_DEVICE_BAR_MAP` and `SYS_DEVICE_DMA_ALLOC`, with config space readable and unwritable and the interrupt delivered as a record on the claim. -Two constraints that were not obvious before the code was read: - - **USB HID cannot move to userspace without moving the boot block device or splitting the controller.** It shares the controller, the event ring and the lock with the boot disk. -- The exception criterion is "a driver stays in the kernel only if the kernel - needs it while userspace is dead". A widening to "if a *service the kernel - itself provides* needs it while userspace is dead" — which keeps NVMe and - changes nothing else — is waiting on the owner. From faab4874a6a4d74a7f0e2f964dcfdaf1d98056e4 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 21:56:32 +0200 Subject: [PATCH 53/54] tests: identity_extent goes; nothing on this branch reads it iommu_empty_domain here judges the fault against the pool the xHCI's own DCBAAP names, read over QMP, and no longer against the extent of the identity domain. identity_extent lost its one caller then, which tests/common/mod.rs's allow(dead_code) hid until #621 removed it, so the merge of ace064f9 made it a warning in both test targets. Co-Authored-By: Claude Opus 5.5 --- tests/common/iommu.rs | 11 ----------- 1 file changed, 11 deletions(-) diff --git a/tests/common/iommu.rs b/tests/common/iommu.rs index eaf77669f92..66293f205e5 100644 --- a/tests/common/iommu.rs +++ b/tests/common/iommu.rs @@ -1874,17 +1874,6 @@ fn blocked_on(line: &str) -> Result { Ok(Blocked { stream: field("stream")?, address: field("addr")?, access: field("access")?, reason }) } -/// How far up the identity domain reaches, off the line that built it. -fn identity_extent(log: &Serial) -> Result { - let line = log.must_say("iommu: identity domain")?; - let range = line - .split_whitespace() - .find(|w| w.starts_with("0x0..")) - .ok_or_else(|| format!("no extent on {line:?}"))?; - let top = range.trim_start_matches("0x0..0x"); - u64::from_str_radix(top, 16).map_err(|_| format!("unreadable extent on {line:?}")) -} - /// The one function `pci::enumerate` printed with this class, or none. /// /// `None` rather than a first match over several: two controllers of one class From 10dc46c934996e234e8599c2b64681d2ec7437d5 Mon Sep 17 00:00:00 2001 From: japabu Date: Wed, 30 Sep 2026 13:46:38 +0200 Subject: [PATCH 54/54] Answer the round-21 review of #536: quiesce_twice keeps the completion it drained, DATA's shrink is judged quiesce_twice armed its log watch with `wait(0, 0, |_| {})`, which submits the watch and then drains every completion and discards it. The SysCap log readiness is an edge that klogd posts after each pass, so a post landing between the registration and that drain was thrown away with the watch it spent, and the next empty read parked in `wait(1, u64::MAX)` with nothing armed. That is the `quiesce_refuses_a_second_shutdown` red in the whole suite at faab4874 (536r20-whole.log:1195): init's two `sync_files` threads put kernel records into that window just before WAITS. The loop now records whether the arming drain took a completion and does not wait when it did: the review's patch, as given. file_mtime's `dated()` judges a shrinking truncation (`set_len(0)`, then `sync_all`) after the grow and requires its stamp later than the grow's. The grow reaches only `DataVolume::truncate`'s grow arm; the shrink arm's stamp had no judge. The kernel's `ftruncate` stamps either direction, so `/tmp` is unchanged. fsd's manifest says what fsd is. tmpfs's page index divides by `PAGE_BYTES` rather than a private 4096. `watch_window`'s doc is main's again. Issues: - deleted, each describing only code this branch deletes and cited nowhere: the tmpfs fallback's missing judge (`open_data`, `TmpFs` for DATA), the shutdown's drain count while `iod` holds an entry, a killed shutdown caller's second drain backoff (`writeback::drain_all`), and a mount's loss report (the kernel's `/log` mount and page-cache flush); - quiesce-wakes-on-the-last-park: Hypothesis B, which rests on `begin_update` and `MID_UPDATE`; - the quiesce-last staging issue: its heading and first sentence, false now that `STAGED` is 42010 ms; - disk-wait-pins-a-cpu and every-wait-in-this-kernel-is-a-spin: the prefaces this branch wrote over them, and every-wait's two sentences naming `writeback_reopen`, `writeback_durability` and `iod.rs`'s header. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_016t9wjdQkB8SH7bmfUoiy6L --- issues/audio/disk-wait-pins-a-cpu.md | 6 --- ...ging-dies-on-ten-seconds-of-guest-clock.md | 4 +- ...port-goes-to-whoever-fsyncs-first-on-it.md | 27 ----------- ...fallback-for-apps-and-home-has-no-judge.md | 43 ----------------- ...ller-panics-on-its-second-drain-backoff.md | 22 --------- .../every-wait-in-this-kernel-is-a-spin.md | 13 +---- ...ve-up-on-one-thread-beside-the-held-one.md | 10 ---- ...unts-the-queue-while-iod-holds-an-entry.md | 48 ------------------- kernel/src/actuator.rs | 2 + kernel/src/tmpfs.rs | 2 +- tests/toyos-rust-tests/src/bin/file_mtime.rs | 14 ++++++ .../toyos-rust-tests/src/bin/quiesce_twice.rs | 12 +++-- userland/fsd/Cargo.toml | 1 + 13 files changed, 29 insertions(+), 175 deletions(-) delete mode 100644 issues/filesystem/a-mounts-loss-report-goes-to-whoever-fsyncs-first-on-it.md delete mode 100644 issues/filesystem/the-tmpfs-fallback-for-apps-and-home-has-no-judge.md delete mode 100644 issues/kernel/a-killed-shutdown-caller-panics-on-its-second-drain-backoff.md delete mode 100644 issues/kernel/the-shutdowns-drain-counts-the-queue-while-iod-holds-an-entry.md diff --git a/issues/audio/disk-wait-pins-a-cpu.md b/issues/audio/disk-wait-pins-a-cpu.md index d9f550d2025..a02bb9b617f 100644 --- a/issues/audio/disk-wait-pins-a-cpu.md +++ b/issues/audio/disk-wait-pins-a-cpu.md @@ -6,12 +6,6 @@ opened: 2026-08-08 # The audio pops are four spinlocks, not one spinning driver -**Measured before the file servers.** `log_file::SINK`, `vfs::VFS` and -`fat32_adapter::VOLUMES` are no longer on the path: logd's writes reach the stick -through fsd and a USB partition claim, whose transfer takes the disk's -block-layer lock and `xhci::XHCI`. -Whether that removes the pops is not measured. - Measured 2026-08-08 on `wt/toyos-asyncusb` at `87835d1`: at the moment a disk transfer is waited for, an ordinary guest is **four ticket spinlocks deep** — `log_file::SINK`, `vfs::VFS`, `fat32_adapter::VOLUMES` and `xhci::XHCI` — and diff --git a/issues/diagnostics/the-quiesce-last-staging-dies-on-ten-seconds-of-guest-clock.md b/issues/diagnostics/the-quiesce-last-staging-dies-on-ten-seconds-of-guest-clock.md index 9d85bdfc8df..1ba5032e22e 100644 --- a/issues/diagnostics/the-quiesce-last-staging-dies-on-ten-seconds-of-guest-clock.md +++ b/issues/diagnostics/the-quiesce-last-staging-dies-on-ten-seconds-of-guest-clock.md @@ -4,9 +4,7 @@ kind: tooling opened: 2026-09-28 --- -# The `quiesce-last-*` staging dies on ten seconds of guest clock - -`kernel/src/quiesce.rs`'s `last::STAGED` is a 10 s in-guest deadline. `hold` +`hold` panics the boot when the stop has not counted the held thread alone within it, and `await_the_held_thread` panics when no thread named `quiesce-last` has reached its syscall within it. `quiesce-last-park` and diff --git a/issues/filesystem/a-mounts-loss-report-goes-to-whoever-fsyncs-first-on-it.md b/issues/filesystem/a-mounts-loss-report-goes-to-whoever-fsyncs-first-on-it.md deleted file mode 100644 index 0c89f559b2d..00000000000 --- a/issues/filesystem/a-mounts-loss-report-goes-to-whoever-fsyncs-first-on-it.md +++ /dev/null @@ -1,27 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-24 ---- - -# A mount's loss report goes to whoever fsyncs first on it, not to the writer whose file lost the writes - -The block layer answers a flush for the writer that made the writes -(`toyos-blockhold`), and a mount is one writer: its partition's view is one -account for every file on it (`kernel/src/block.rs`, the page cache's flush in -`kernel/src/vfs.rs`). So when a USB disk comes back having lost writes the -kernel wrote for `/log`, the first `fsync` of *any* file on that mount — or -`sync_all` at shutdown — takes the one `Io` the loss is owed, and logd's own -`fsync` of `/log` after it answers `Ok`. - -The paths are ambient by the owner's ruling (root `CLAUDE.md`, *Capabilities*), -so a second process that writes under `/log` and fsyncs can take logd's -report. Nothing does today — logd is the only writer there — so nothing fails, -but the claim "a flush answers for its writer's own writes" is true of a -partition claim and of a mount, and not of a file. - -**Exit condition.** The capability end-state track -(`issues/kernel/the-capability-end-state-is-twelve-answers.md`) commits the -ambient set, and `/log` either leaves it — logd is its one writer by -construction — or the page cache keeps the account per file, and a test with -two writers on one mount sees each told only of its own file's loss. diff --git a/issues/filesystem/the-tmpfs-fallback-for-apps-and-home-has-no-judge.md b/issues/filesystem/the-tmpfs-fallback-for-apps-and-home-has-no-judge.md deleted file mode 100644 index f82b03fe2bf..00000000000 --- a/issues/filesystem/the-tmpfs-fallback-for-apps-and-home-has-no-judge.md +++ /dev/null @@ -1,43 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-04 ---- - -# Nothing asserts a machine with no DATA volume gets a working `/apps` and `/home` - -## What is unjudged - -`kernel/src/main.rs`'s `open_data()` answers `None` three ways — no TOYOS-DATA -partition, two of them, or a volume that is not ours — and every one lands on -one `TmpFs` mounted under both names: - - storage: /apps and /home are a tmpfs — they will not survive a reboot - -Three tests read that line, and all three read it as a *negative premise*: -`tests/common/storage.rs:277`, `:366` and `:458` each refuse to continue if it -appears, because their readback would then judge no device. No test boots a -machine without a DATA volume and asserts the line is there, that both `/apps` -and `/home` are writable, and that the two are still one filesystem — which is -the whole of what the fallback promises. So the arm a first boot on unprepared -hardware takes is the arm nothing runs. - -The shape it hides: the fallback mounts one `TmpFs` under `&DATA_PATHS`, so -`/home/x` is `home/x` and `/apps/y` is `apps/y` inside it. A fallback that -mounted two `TmpFs` instances, or one under a single name, would pass every -test in the suite. - -## Reproduction - -`Profile::Diskless` and `Profile::Metal8042Absent` both declare `nvme_bytes: 0` -(`tests/common/qemu.rs:1695`, `:1742`), which emits no controller, no namespace -and no backing file. Boot either and the line is printed; nothing then asks the -guest anything about the two paths. - -## Exit condition - -A machine test boots a `nvme_bytes: 0` profile, asserts the line, and runs a -guest binary that writes under `/apps` and under `/home`, reads both back, and -finds neither name answering for the other's file — the same four claims -`hierarchy_paths` makes against DATA, against the fallback instead. Closes when -that test is registered and green. diff --git a/issues/kernel/a-killed-shutdown-caller-panics-on-its-second-drain-backoff.md b/issues/kernel/a-killed-shutdown-caller-panics-on-its-second-drain-backoff.md deleted file mode 100644 index 26e9490cc79..00000000000 --- a/issues/kernel/a-killed-shutdown-caller-panics-on-its-second-drain-backoff.md +++ /dev/null @@ -1,22 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-27 ---- - -# A killed shutdown caller panics on its second drain backoff - -`writeback::drain_all` (`kernel/src/writeback.rs`) retries a refused drain -through `block::between_attempts`, whose park is cancellable, and discards what -it answers. A thread whose kill bit is set gets `Cancelled` from that park at -once, retries, and parks again; `TaskHandle::take_cancel` asserts on the second -cancel reported to one thread, so the kernel panics. `ops::until_answered` -reads the kill bit before it parks and does not have this; `drain_all` does not. - -The caller is the shutdown syscall (`syscall/machine.rs`), which a kill posted -before `quiesce::stop` can still reach, and it takes a second backoff only when -the device refuses two drain attempts on budget. Not reproduced. - -Exit condition: `drain_all`'s retry either stops parking once its caller is -killed or parks uncancellably, with a test that kills a caller parked in a -refused drain. diff --git a/issues/kernel/every-wait-in-this-kernel-is-a-spin.md b/issues/kernel/every-wait-in-this-kernel-is-a-spin.md index 302e6cf2062..e86fa18ee1b 100644 --- a/issues/kernel/every-wait-in-this-kernel-is-a-spin.md +++ b/issues/kernel/every-wait-in-this-kernel-is-a-spin.md @@ -6,12 +6,6 @@ opened: 2026-08-12 # Every wait in this kernel is a spin, and a killed task dies by having its stack discarded -**The storage chains below are older than the file servers.** The kernel holds -no FAT, NVMe or page-cache lock any more: `/apps`, `/config`, `/home`, `/state`, -`/log` and `/boot` are `/system/bin/fsd`'s, over blockd or a USB partition -claim, so what is left of a disk wait in the kernel is a claim's transfer -(`block` → `xhci::XHCI`) and `vfs::VFS` over ROOT and `/tmp`. - **The heading and the paragraph under it are the state at opening, not the state of the tree.** What exists now: the completion core, where every wait in the kernel rechecks one predicate and a waiter lends a watch to the object it @@ -411,12 +405,7 @@ lock converted** — `vfs::VFS` stays a spinlock, `iod` spins in the driver, and the drain pops under the VFS lock so `sync_all` cannot commit a device ahead of a file's flush. This is the state that unblocks `vfs::VFS`'s conversion (the next chunk): a `Drop` reaching this release site now touches neither a sleep -lock nor a device. Negative controls: `writeback_reopen` (an `iod` stalled by -`writeback-stall`, a re-open reads the pinned pages) and `writeback_durability` -(a close with no fsync reached the `/log` volume through the drain, which the -kernel no longer serves). Measured, and recorded in `iod.rs`'s header: a -360-file close burst on NVMe `/home` drove worst close-to-drained latency to -~72 ms, the single `iod` thread draining the backlog serially. +lock nor a device. ### Wall 5: demand paging holds `ProcessData` across the device, and a nested trap is a level above the baseline diff --git a/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md b/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md index db35aca27b8..323466776ad 100644 --- a/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md +++ b/issues/kernel/quiesce-wakes-on-the-last-park-gave-up-on-one-thread-beside-the-held-one.md @@ -36,16 +36,6 @@ the `stop:` record carries only counts. thread queued on that CPU then runs only if `dispose_yield` (`toyos-sched/src/cpu.rs`) re-inserts the spinner behind it. -**Hypothesis B, untested.** `tests/quiescelastcase/system.toml` starts `logd`, -which fsyncs `/log`. `begin_update`'s only caller is `SYS_FSYNC` -(`kernel/src/object/ops.rs:626`), and that update spans every retry, including -the park in `between_attempts`. A thread parked there is `Blocked` with -`MID_UPDATE`; `stop_if_blocked` refuses it (`toyos-sched/src/task.rs:450`), the -sweep counts it running, and it adds 0 to `in_flight` — so `logd` parked -between refused fsync attempts past the 2010 ms budget is a candidate the -records cannot exclude, and it sits on the same `/log` fsync path the writers -issue owns. - **What no enabled guest test checks while this is disabled.** - that a band, a park or an exit wakes the stop, rather than its deadline; diff --git a/issues/kernel/the-shutdowns-drain-counts-the-queue-while-iod-holds-an-entry.md b/issues/kernel/the-shutdowns-drain-counts-the-queue-while-iod-holds-an-entry.md deleted file mode 100644 index 0c31cee8310..00000000000 --- a/issues/kernel/the-shutdowns-drain-counts-the-queue-while-iod-holds-an-entry.md +++ /dev/null @@ -1,48 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-18 ---- - -# The shutdown's drain counts the write-back queue while `iod` holds an entry - -`kernel/src/writeback.rs`'s `drain_retrying` decides how many entries a pass -owes with `QUEUE.lock().len()`, taken under the queue lock alone. -`drain_one` pops under the VFS lock its caller already holds, and re-enqueues -a budget-refused flush under that same lock — so an entry `iod` has popped is -on no queue for as long as `iod`'s attempt lasts, and a second drainer that -counts in that window counts zero, sees nothing owed, and returns. - -The second drainer is the shutdown. `quiesce` calls `writeback::drain_all` -and then `sync_all`, `Rebooting.` and the reset. If -`iod`'s attempt in that window is refused on budget, the file is still owed -after the shutdown's drain has said nothing is, `iod` parks in its backoff, -and the reset lands over dirty pages nothing flushed. An attempt that -succeeds is harmless: `sync_all` takes the VFS lock and so waits for it. - -## What has been seen - -The count coming back zero with an entry in `iod`'s hands, staged: a boot of -`tests/quiescetwicecase` on a kernel whose `quiesce-drain-refuse` actuator -refused `iod`'s flush of `/log/quiesce-owed.bin` every time. The shutdown -wrote `Syncing filesystems...` at 0.719 s on cpu0 and `Rebooting.` at 0.722 s -with no attempt of its own on that file, while cpu1 wrote `log-volume: write -of quiesce-owed.bin: the device would not answer in the caller's own budget` -at 0.719, 0.721, 0.734, 0.755, 0.798, 0.881, 1.044, 1.367 and 2.010 s — seven -of them under the boot's last word. One boot in the five that actuator's -first form was booted for; that config now arms `writeback-stall` instead, -which parks `iod` before it drains, so it no longer stages this. - -Not seen: the loss itself on a kernel with no actuator, which needs a real -budget expiry on `iod`'s flush inside the window. A USB stick slow enough to -expire `block::OPERATION` is the machine that would produce one. - -## What would show it - -An actuator that refuses `iod`'s drain flush once, on budget, while a -shutdown is in its first stage, judged by `toyos-fat32-check` and a byte -comparison of the file on the image after the reset. - -**Exit condition**: the shutdown's drain cannot return while a flush is owed -to anybody — the count and the pop under one lock, or the drain waiting on -what `iod` holds — and the boot above reads the file back whole. diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index d758b0d983b..b7e6fd53a79 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -250,6 +250,8 @@ actuators! { /// Report the preempt depth and backtrace at the deepest point of a disk transfer; it stages nothing, only measures. io_depth_probe = "io-depth-probe"; + /// Hold every thread that waits on a watch between reading its condition and + /// parking, so a post lands in the window its commit must refuse the park over. watch_window = "watch-window"; /// Starve the four xHCI bring-up register waits in `init_one`. diff --git a/kernel/src/tmpfs.rs b/kernel/src/tmpfs.rs index 0b06d83d89d..1a6fc879c85 100644 --- a/kernel/src/tmpfs.rs +++ b/kernel/src/tmpfs.rs @@ -24,7 +24,7 @@ impl FileBacking for TmpfsBacking { // copy_page_out, not file_cache::read_page: reading through the miss path here would recurse. // A hole below the file size, left by a seek-and-write, reads as zero. if file_offset >= file_cache::size(self.file_id) - || !file_cache::copy_page_out(self.file_id, (file_offset / 4096) as u32, buf) + || !file_cache::copy_page_out(self.file_id, (file_offset / PAGE_BYTES as u64) as u32, buf) { // After the miss: `retire` clears the flag before the pages drop. if !self.alive.load(Ordering::Acquire) { diff --git a/tests/toyos-rust-tests/src/bin/file_mtime.rs b/tests/toyos-rust-tests/src/bin/file_mtime.rs index 9eeb75419d2..b7496af83cd 100644 --- a/tests/toyos-rust-tests/src/bin/file_mtime.rs +++ b/tests/toyos-rust-tests/src/bin/file_mtime.rs @@ -98,6 +98,20 @@ fn dated(dir: &str) -> (u64, u64) { resized > created, "{truncated} was created at {created} ns and a later truncation stamped it {resized} ns" ); + + let shrunk = judged(&truncated, "a shrinking truncation", || { + let f = OpenOptions::new() + .write(true) + .open(&truncated) + .unwrap_or_else(|e| panic!("reopen {truncated}: {e}")); + f.set_len(0).unwrap_or_else(|e| panic!("ftruncate {truncated} to 0: {e}")); + f.sync_all().unwrap_or_else(|e| panic!("fsync {truncated}: {e}")); + }); + assert!( + shrunk > resized, + "{truncated} was grown at {resized} ns and a later shrinking truncation stamped it \ + {shrunk} ns" + ); (first, second) } diff --git a/tests/toyos-rust-tests/src/bin/quiesce_twice.rs b/tests/toyos-rust-tests/src/bin/quiesce_twice.rs index 34903cfb2b7..3d0b1f2a278 100644 --- a/tests/toyos-rust-tests/src/bin/quiesce_twice.rs +++ b/tests/toyos-rust-tests/src/bin/quiesce_twice.rs @@ -51,7 +51,10 @@ fn main() { let mut buf = [Record::EMPTY; BATCH]; let poller = Poller::new(1); poller.watch(&cap, READABLE, LOG_TOKEN); - poller.wait(0, 0, |_| {}); + // A completion drained while arming is the watch spent: waiting on it + // afterwards parks on nothing. + let mut spent = false; + poller.wait(0, 0, |_| spent = true); loop { let batch = tail.read(&cap, &mut buf).unwrap_or_else(|e| { eprintln!("quiesce_twice: the log would not read ({e:?})"); @@ -65,9 +68,12 @@ fn main() { } // No deadline: a first call that never waits is a hang the harness // ceiling reds. - poller.wait(1, u64::MAX, |_| {}); + if !spent { + poller.wait(1, u64::MAX, |_| {}); + } poller.watch(&cap, READABLE, LOG_TOKEN); - poller.wait(0, 0, |_| {}); + spent = false; + poller.wait(0, 0, |_| spent = true); } // The other power syscall, so the one boot judges the claim on both. diff --git a/userland/fsd/Cargo.toml b/userland/fsd/Cargo.toml index 8f90fb272b1..8c743a4d975 100644 --- a/userland/fsd/Cargo.toml +++ b/userland/fsd/Cargo.toml @@ -2,6 +2,7 @@ name = "fsd" edition = "2024" license = "MIT OR Apache-2.0" +description = "ToyOS's file server: one process per storage role — DATA, the log or the boot volume — serves that role's paths." # The library is the server's decisions — the block cache, the volumes and the # resolver — so the host can test them; the service is `src/main.rs`.