From e14376e62b012ca6084a09245b698b8cf1309e85 Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 21:42:46 +0200 Subject: [PATCH 01/11] AArch64 stage 4 on one CPU: the kernel's own tables, the GICv3 and the timer, and user mode The port's stage 4 (issues/kernel/toyos-runs-on-arm64.md), ahead of small-kernel stage 6 on the owner's word, with the drivers that track moves out of the kernel refused at the narrowest seam rather than ported. AArch64: - paging: TTBR1_EL1 holds the kernel's direct map of every 4 KiB page firmware's map calls memory, and nothing else; each user space is a TTBR0_EL1 root under a 16-bit ASID from toyos-pcid. Live entries are replaced break-before-make, a page is never retyped, and executable pages are made coherent with the instruction stream before they map. - control registers: CPACR lets EL1 and EL0 use FP/SIMD, CNTKCTL gives EL0 the virtual count, TCR asks for 16-bit ASIDs (checked), ICC_SRE puts the GICv3 CPU interface in system registers (ICC_SRE_EL2 first at EL2), and a CPU without one halts in a named refusal. - per-CPU state through TPIDR_EL1; the GICv3 distributor and this CPU's redistributor, SGIs and the virtual timer's PPI; the vectors dispatch an SVC to the syscall dispatcher, a translation fault to the demand pager, a user fault to the end of its process, and an interrupt to the tick or kick that preempts at EL0; the context switch carries FP/SIMD, since the soft-float kernel never touches it; trampolines to EL0 zero every register but the argument. - irq-storm: a boot actuator that ticks the timer under an SGI flood and says whether any tick was lost. Shared: - KernelCtx and the spawn paths name the stack pointer and thread pointer by role, which closes issues/kernel/the-saved-kernel-context-names-x86-registers.md. - The exit-to-user epilogue moves from x86's idt into scheduler.rs; the region bookkeeping from x86's AddressSpace into vma::Regions; the idle stack arena from x86's percpu into sched::idle_stack. - arch::msi_message may refuse, and on AArch64 does: the GICv3 ITS moves to stage 6, where a claimed function behind the SMMUv3 is its only consumer the small-kernel track leaves. - gop::init refuses a scanout that is not whole 2 MiB pages of its own. - paging::init is handed the scanout, which AArch64 maps before its switch. Tests (Tier::Local, VirtEl2): virt_user_mode, virt_timer_preempts (a new tests/virtpreemptcase), virt_irq_storm. Filed: issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md, issues/kernel/the-x86-address-space-keeps-a-page-map-nothing-fills.md. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01W6rME2DoqwjcYFStYHHY4j --- ...s-ring-3-holding-kernel-register-values.md | 22 + ...-routing-take-an-x86-vector-and-apic-id.md | 4 +- ...aved-kernel-context-names-x86-registers.md | 20 - ...ss-space-keeps-a-page-map-nothing-fills.md | 16 + issues/kernel/toyos-runs-on-arm64.md | 23 +- kernel/src/actuator.rs | 6 +- kernel/src/arch/aarch64/boot.rs | 83 +- kernel/src/arch/aarch64/cache.rs | 45 +- kernel/src/arch/aarch64/console_uart.rs | 7 + kernel/src/arch/aarch64/control_regs.rs | 65 +- kernel/src/arch/aarch64/cpu.rs | 8 + kernel/src/arch/aarch64/entry.rs | 98 ++- kernel/src/arch/aarch64/fpu.rs | 11 +- kernel/src/arch/aarch64/hw.rs | 183 +++- kernel/src/arch/aarch64/irqchip.rs | 364 +++++++- kernel/src/arch/aarch64/mod.rs | 27 +- kernel/src/arch/aarch64/paging.rs | 789 ++++++++++++++++-- kernel/src/arch/aarch64/percpu.rs | 284 ++++++- kernel/src/arch/aarch64/pmu.rs | 13 +- kernel/src/arch/aarch64/smp.rs | 16 +- kernel/src/arch/aarch64/switch.rs | 106 +++ kernel/src/arch/aarch64/syscall.rs | 13 +- kernel/src/arch/aarch64/tlb.rs | 100 ++- kernel/src/arch/aarch64/trap.rs | 459 +++++++++- kernel/src/arch/aarch64/watchdog.rs | 17 +- kernel/src/arch/x86_64/apic.rs | 4 +- kernel/src/arch/x86_64/boot.rs | 1 + kernel/src/arch/x86_64/hw.rs | 28 +- kernel/src/arch/x86_64/idt/mod.rs | 30 +- kernel/src/arch/x86_64/paging.rs | 89 +- kernel/src/arch/x86_64/percpu.rs | 87 +- kernel/src/drivers/gop.rs | 10 +- kernel/src/drivers/panic_console/mod.rs | 11 +- kernel/src/drivers/pci.rs | 8 +- kernel/src/drivers/xhci/wait/msc.rs | 2 +- kernel/src/hardlockup/mod.rs | 42 +- kernel/src/hardlockup/probe.rs | 2 +- kernel/src/loader/mod.rs | 8 +- kernel/src/loader/tls.rs | 4 +- kernel/src/main.rs | 8 +- kernel/src/mm/mod.rs | 2 +- kernel/src/process.rs | 23 +- kernel/src/sched/driver.rs | 36 +- kernel/src/sched/idle_stack.rs | 88 ++ kernel/src/sched/kthread.rs | 4 +- kernel/src/sched/mod.rs | 1 + kernel/src/sched/payload.rs | 7 +- kernel/src/scheduler.rs | 47 +- kernel/src/vma.rs | 80 +- src/build.rs | 1 + src/metal.rs | 2 +- src/sourcegate.rs | 6 +- tests/toyos.rs | 109 +++ tests/virtpreemptcase/system.toml | 19 + toyos-bootmap/src/aarch64.rs | 75 +- toyos-bootmap/src/lib.rs | 7 +- toyos-bootmap/tests/aarch64_direct_map.rs | 84 ++ 57 files changed, 3097 insertions(+), 607 deletions(-) create mode 100644 issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md delete mode 100644 issues/kernel/the-saved-kernel-context-names-x86-registers.md create mode 100644 issues/kernel/the-x86-address-space-keeps-a-page-map-nothing-fills.md create mode 100644 kernel/src/arch/aarch64/switch.rs create mode 100644 kernel/src/sched/idle_stack.rs create mode 100644 tests/virtpreemptcase/system.toml create mode 100644 toyos-bootmap/tests/aarch64_direct_map.rs diff --git a/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md b/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md new file mode 100644 index 0000000000..1df4fd59d2 --- /dev/null +++ b/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md @@ -0,0 +1,22 @@ +--- +status: open +kind: defect +opened: 2026-09-28 +--- + +# A new x86-64 thread enters Ring 3 holding the kernel's register values + +`kernel/src/arch/x86_64/entry.rs`'s `process_start` and `thread_start` call +`sched::driver::trampoline_entry`, restore only `r12`–`r14` and the FP state, +and `iretq`: every other general register — `rax`–`rdx`, `rsi`, `rbp`, +`r8`–`r11`, `r15`, and `rdi` in a process's case — reaches the thread's first +instruction holding whatever the kernel left there, kernel stack and heap +addresses among them. A thread reads the kernel's layout off its own +registers. + +The AArch64 trampoline zeroes every register but the argument before its +`ERET`, and is the shape to copy. + +**Exit condition**: a fresh thread's first instruction sees zero in every +general register but its stack pointer and its argument, and a guest test that +reads them at `_start` says so. diff --git a/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md b/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md index 6c9e88011c..256298fad6 100644 --- a/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md +++ b/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md @@ -13,8 +13,8 @@ vector delivered to an xAPIC id. A GICv3 message names an LPI through the ITS (a 32-bit event id, translated to an INTID above 8191) and a redistributor, and neither fits a `u8`. -Owned by stage 4 of `issues/kernel/toyos-runs-on-arm64.md`, which brings up -the GIC and its ITS. +Owned by stage 6 of `issues/kernel/toyos-runs-on-arm64.md`, which brings up +the GICv3 ITS for the claimed functions its SMMUv3 translates. **Exit condition**: a PCI function's interrupt is programmed from an arch-provided message (address and data, as `arch::msi` already provides the diff --git a/issues/kernel/the-saved-kernel-context-names-x86-registers.md b/issues/kernel/the-saved-kernel-context-names-x86-registers.md deleted file mode 100644 index 7984c7af40..0000000000 --- a/issues/kernel/the-saved-kernel-context-names-x86-registers.md +++ /dev/null @@ -1,20 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-26 ---- - -# The saved kernel context names x86 registers in generic code - -`kernel/src/sched/payload.rs`'s `KernelCtx`, which every CPU's switch loads, -carries `rsp` and `fs_base`: the stack pointer and the user thread pointer by -their x86 names, in code outside `arch/`. AArch64 saves `sp` and -`TPIDR_EL0` in the same two roles, so the port either writes AArch64 state -into fields named for x86 or grows a second copy of the struct. - -Owned by stage 4 of `issues/kernel/toyos-runs-on-arm64.md`, which writes the -AArch64 context switch. - -**Exit condition**: the fields are named by role (the saved kernel stack -pointer, the user thread pointer), the x86 and AArch64 switches both load -them, and nothing outside `kernel/src/arch/` names `rsp` or `fs_base`. diff --git a/issues/kernel/the-x86-address-space-keeps-a-page-map-nothing-fills.md b/issues/kernel/the-x86-address-space-keeps-a-page-map-nothing-fills.md new file mode 100644 index 0000000000..b12ea86615 --- /dev/null +++ b/issues/kernel/the-x86-address-space-keeps-a-page-map-nothing-fills.md @@ -0,0 +1,16 @@ +--- +status: open +kind: defect +opened: 2026-09-28 +--- + +# The x86-64 address space keeps a page map nothing fills + +`kernel/src/arch/x86_64/paging.rs`'s `AddressSpace` carries +`pages: HashMap`, documented as the user pages it frees on +drop, and `unmap` removes from it; nothing anywhere inserts into it, so it is +always empty and the removal does nothing. Dead code the compiler cannot see, +because a field that is read is not dead to it. + +**Exit condition**: the field and its removal are gone, or what it claims to +own is put in it by the path that maps the page. diff --git a/issues/kernel/toyos-runs-on-arm64.md b/issues/kernel/toyos-runs-on-arm64.md index aa73177814..dfa468a740 100644 --- a/issues/kernel/toyos-runs-on-arm64.md +++ b/issues/kernel/toyos-runs-on-arm64.md @@ -243,8 +243,7 @@ before any aarch64 file exists, with x86 as its only user: Each is its own issue, owned by the stage that removes it: -- `issues/kernel/the-saved-kernel-context-names-x86-registers.md` (stage 4) -- `issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md` (stage 4) +- `issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md` (stage 6) - `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md` (stage 4) - `issues/kernel/the-crash-evidence-records-x86-fault-registers.md` (stage 5) - `issues/kernel/the-aarch64-kernel-builds-with-dead-code-allowed.md` (stage 7) @@ -298,6 +297,26 @@ Each stage names its exit; "measured" means a number from a run. one stays green in `virt_el2_drop`, because stage 3 reads no counter and runs no FP. This stage's timer and FP tests run under that EL2 profile too, and each of the three deletions is shown red. + **Built on one CPU, ahead of small-kernel stage 6 by the owner's word:** + the kernel's own tables (`TTBR1_EL1` holding memory and nothing else, each + user space on `TTBR0_EL1` under a 16-bit ASID from `toyos-pcid`), the + GICv3's SGIs and the virtual timer's PPI, the EL0 entry, and the context + switch carrying FP/SIMD; `virt_user_mode`, `virt_timer_preempts` and + `virt_irq_storm` judge it under the EL2 profile, emulated, because HVF + exposes no RNDR and the kernel's hash seed refuses there until stage 6's + virtio-rng. Owed before the exit holds: the interrupts-off window against + x86's (the storm reports the latest tick it took; x86 has no counterpart + instrument); `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md`, + whose `KernelArgs` rename waits on the loader's change to that struct; and + the three deletions shown red. The ITS moves to stage 6: a claimed function + is its only consumer the small-kernel track leaves, and it needs that + stage's SMMUv3 first. Stubbed on AArch64, each owned by the small-kernel + track, which moves the driver out of the kernel: + - `arch::msi_message` refuses, so the kernel's xHCI (`virt`'s boot stick), + NVMe, HDA, virtio-sound, virtio-console and virtio-gpu drivers each + refuse their function by name. + - `drivers::gop` refuses a scanout that is not whole 2 MiB pages of its + own, which a `ramfb` scanout carved out of RAM need not be. 5. **SMP through PSCI.** `CPU_ON` from MADT GICC entries, SGIs as the IPI, broadcast TLBI behind the machine-wide invalidation contract. **Exit**: diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 1c79d5b12f..69bf3e896f 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -305,7 +305,7 @@ actuators! { /// Storm the CPU spinning on `syscall` from Ring 3 with NMIs. syscall_window_nmi = "syscall-window-nmi"; - /// Take the IST index off vector 2's gate — the negative control on the row above: the CPU builds the NMI frame at whatever `rsp` holds and takes a `#DF`. + /// Take the IST index off vector 2's gate — the negative control on the row above: the CPU builds the NMI frame at whatever the stack pointer holds and takes a `#DF`. nmi_without_ist = "nmi-without-ist"; /// Return from the NMI handler via `iretq` with a second NMI already pending. @@ -347,6 +347,10 @@ actuators! { /// Raise a vector no `idt_vectors!` row claims on this CPU once. unclaimed_vector_selftest = "unclaimed-vector-selftest"; + /// Tick the timer at a fixed period while this CPU floods itself with + /// interrupts, and say whether any tick went a whole period untaken. + irq_storm = "irq-storm"; + /// Hold a flush of `truncate-race.bin` inside its metadata window and say whether a truncate got in. ftruncate_flush_stall = "ftruncate-flush-stall"; diff --git a/kernel/src/arch/aarch64/boot.rs b/kernel/src/arch/aarch64/boot.rs index c3fd3144c8..c2aa4b188d 100644 --- a/kernel/src/arch/aarch64/boot.rs +++ b/kernel/src/arch/aarch64/boot.rs @@ -33,6 +33,11 @@ use crate::mm::Region; pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { core::arch::naked_asm!( "mov x19, x0", + // `ID_AA64PFR0_EL1.GIC`: no GICv3 system-register interface, and the + // `ICC_SRE` writes below are undefined instructions. + "mrs x1, id_aa64pfr0_el1", + "ubfx x1, x1, #24, #4", + "cbz x1, {refuse_gic}", // x20 = the stack's top, physical: kernel image + stack offset + stack size. "ldr x20, [x19, #{kernel_memory}]", "ldr x1, [x19, #{stack_offset}]", @@ -61,7 +66,12 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { "msr tcr_el1, x2", "msr ttbr0_el1, x3", "msr ttbr1_el1, x3", - "msr cpacr_el1, xzr", + "ldr x1, ={cpacr}", + "msr cpacr_el1, x1", + "mov x1, #{cntkctl}", + "msr cntkctl_el1, x1", + "mov x1, #{icc_sre}", + "msr S3_0_C12_C12_5, x1", "tlbi vmalle1", "dsb nsh", "isb", @@ -79,11 +89,20 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { "mrs x6, hcr_el2", "cmp x6, x1", "b.ne {refuse_hcr}", + // `ICC_SRE_EL2` before `ICC_SRE_EL1`: its `Enable` is what lets EL1's be written at all. + "mov x1, #{icc_sre_el2}", + "msr S3_4_C12_C9_5, x1", + "isb", "msr mair_el1, x4", "msr tcr_el1, x2", "msr ttbr0_el1, x3", "msr ttbr1_el1, x3", - "msr cpacr_el1, xzr", + "ldr x1, ={cpacr}", + "msr cpacr_el1, x1", + "mov x1, #{cntkctl}", + "msr cntkctl_el1, x1", + "mov x1, #{icc_sre}", + "msr S3_0_C12_C12_5, x1", "msr sctlr_el1, x5", "mov x1, #{cnthctl}", "msr cnthctl_el2, x1", @@ -128,6 +147,11 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { refuse_hcr = sym refused_hcr_el2_readback, cnthctl = const regs::CNTHCTL_EL2, cptr = const regs::CPTR_EL2, + cpacr = const regs::CPACR, + cntkctl = const regs::CNTKCTL, + icc_sre = const regs::ICC_SRE, + icc_sre_el2 = const regs::ICC_SRE_EL2, + refuse_gic = sym refused_no_gicv3, spsr = const regs::SPSR_EL2_TO_EL1, phys_offset = const crate::PHYS_OFFSET, entry_el = sym regs::ENTRY_EL, @@ -148,6 +172,16 @@ unsafe extern "C" fn refused_hcr_el2_readback() -> ! { core::arch::naked_asm!("1:", "wfe", "b 1b") } +/// Where a CPU whose `ID_AA64PFR0_EL1.GIC` names no GICv3 system-register +/// interface halts, before the first `ICC_` write would be an undefined +/// instruction under firmware's vectors: this kernel's interrupt controller is +/// a GICv3. Silent for [`refused_hcr_el2_readback`]'s reason. +#[unsafe(naked)] +#[no_mangle] +unsafe extern "C" fn refused_no_gicv3() -> ! { + core::arch::naked_asm!("1:", "wfe", "b 1b") +} + /// The ACPI tables this architecture decodes: the MADT for its GIC and CPUs, /// the FADT for PSCI and reset, the GTDT for the timer, the SPCR for the /// console, and the MCFG for ECAM. @@ -238,37 +272,52 @@ pub fn reserved() -> Region { Region { start: 0, end: 0 } } -/// What the boot learns bringing interrupts up and hands later steps. -pub struct Platform { - never: core::convert::Infallible, -} +/// What the boot learns bringing interrupts up and hands later steps: nothing +/// yet, since the other CPUs the MADT names are the port's stage 5's to read. +pub struct Platform; -/// Interrupt delivery, this CPU's per-CPU block and the syscall gate. -pub fn interrupts(_rsdp_addr: u64) -> Platform { - owed!("interrupt delivery", "stage 4") +/// Interrupt delivery: this CPU's per-CPU block, the GIC and the timer's +/// interrupt, and interrupts unmasked. The syscall gate is the vectors' own. +pub fn interrupts(rsdp_addr: u64) -> Platform { + super::percpu::init_bsp(); + super::irqchip::init(rsdp_addr); + super::cpu::enable_interrupts(); + Platform } -/// The clock: the generic timer's counter at `CNTFRQ_EL0`. +/// The clock: the generic timer's count, at the rate firmware states in +/// `CNTFRQ_EL0`, which the Arm ARM makes firmware's to program and which is +/// what the counter counts at. No wall clock: `super::rtc` says why. pub fn clock(_args: &KernelArgs) { - owed!("the clock", "stage 4") + let hz = super::cpu::stated_counter_hz().expect("clock: CNTFRQ_EL0 states no rate for the generic timer"); + crate::clock::set_counter(super::cpu::counter(), 1_000_000_000_000_000 / hz); + log!("clock: the generic timer counts at {hz} Hz; no wall clock is read on this architecture"); } -/// The per-CPU timer. +/// The per-CPU timer counts the clock's own ticks, so there is nothing to +/// calibrate: it stays stopped until the scheduler first arms it. pub fn timer() { - owed!("the timer", "stage 4") + log!("timer: the EL1 virtual timer, PPI {}, stopped until the scheduler arms it", super::irqchip::timer_intid()); } /// The platform's own devices that are not PCI functions: none this kernel /// drives on an ACPI Arm machine. pub fn platform_devices(_rsdp_addr: u64) {} -/// Every other CPU, running. -pub fn start_other_cpus(platform: &Platform, _args: &KernelArgs) { - match platform.never {} +/// Every other CPU, running: the port's stage 5, which starts each with PSCI +/// `CPU_ON`. Until then the boot CPU runs alone. +pub fn start_other_cpus(_platform: &Platform, _args: &KernelArgs) { + log!("smp: the boot CPU runs alone; the other CPUs are the port's stage 5 (PSCI CPU_ON)"); } /// The interrupt-controller selftests an actuator asks for. #[cfg(feature = "boot-actuators")] pub fn interrupt_selftests() { - owed!("the interrupt controller", "stage 4") + if crate::actuator::irq_storm() { + super::trap::storm::run(); + } + assert!( + !crate::actuator::lapic_spurious_selftest() && !crate::actuator::unclaimed_vector_selftest(), + "the local APIC's selftests are x86-64's, and this machine has a GIC" + ); } diff --git a/kernel/src/arch/aarch64/cache.rs b/kernel/src/arch/aarch64/cache.rs index 0b0e9df4d6..45503d3c8a 100644 --- a/kernel/src/arch/aarch64/cache.rs +++ b/kernel/src/arch/aarch64/cache.rs @@ -1,13 +1,18 @@ -//! Writing memory back out of the caches, for the one reader that is not a -//! CPU: DRAM across a reset. +//! Cache maintenance, for the two readers that are not this CPU's data side: +//! DRAM across a reset, and the instruction stream. -/// The smallest data cache line on this machine, from `CTR_EL0.DminLine` -/// (log2 of the word count), so a walk by it misses no line. -fn line() -> u64 { +/// `CTR_EL0`, which EL1 may always read. +fn ctr() -> u64 { let ctr: u64; // SAFETY: reads `CTR_EL0`, which EL1 may always read. unsafe { core::arch::asm!("mrs {}, ctr_el0", out(reg) ctr, options(nomem, nostack, preserves_flags)) }; - 4 << ((ctr >> 16) & 0xF) + ctr +} + +/// The smallest data cache line on this machine, from `CTR_EL0.DminLine` +/// (log2 of the word count), so a walk by it misses no line. +fn line() -> u64 { + 4 << ((ctr() >> 16) & 0xF) } /// Every line of `[at, at + len)` cleaned to the point of coherency and @@ -28,3 +33,31 @@ pub fn write_back(at: u64, len: usize) { // SAFETY: a barrier; waits for the maintenance above to complete. unsafe { core::arch::asm!("dsb sy", options(nostack, preserves_flags)) }; } + +/// Instructions this CPU wrote through the data side at `[at, at + len)` made +/// the ones every CPU fetches: the data cleaned to the point of unification +/// unless `CTR_EL0.IDC` says it need not be, then every instruction cache +/// invalidated unless `CTR_EL0.DIC` says the same (Arm ARM K.a, D7.5.9.2). +/// Before a mapping that executes them is written. +pub fn make_executable(at: u64, len: usize) { + let ctr = ctr(); + if ctr >> 28 & 1 == 0 { + let step = line(); + let mut addr = at & !(step - 1); + while addr < at + len as u64 { + // SAFETY: `DC CVAU` cleans the line holding a mapped address the + // caller owns; it changes no memory's contents. + unsafe { core::arch::asm!("dc cvau, {}", in(reg) addr, options(nostack, preserves_flags)) }; + addr += step; + } + // SAFETY: a barrier; waits for the cleaning above to complete. + unsafe { core::arch::asm!("dsb ish", options(nostack, preserves_flags)) }; + } + if ctr >> 29 & 1 == 0 { + // SAFETY: invalidating instruction caches changes no memory; the + // barrier waits for it on every CPU. + unsafe { core::arch::asm!("ic ialluis", "dsb ish", options(nostack, preserves_flags)) }; + } + // SAFETY: a context synchronization; touches nothing. + unsafe { core::arch::asm!("isb", options(nostack, preserves_flags)) }; +} diff --git a/kernel/src/arch/aarch64/console_uart.rs b/kernel/src/arch/aarch64/console_uart.rs index e316a04030..a0b235f076 100644 --- a/kernel/src/arch/aarch64/console_uart.rs +++ b/kernel/src/arch/aarch64/console_uart.rs @@ -27,6 +27,13 @@ const FRAME: u64 = 0x1000; /// The register frame's physical address; zero until [`init`] found one. static BASE: AtomicU64 = AtomicU64::new(0); +/// The register frame's physical address, once [`init`] found one: what the +/// kernel's own tables map before they replace the loader's. +pub fn frame() -> Option { + let base = BASE.load(Ordering::Relaxed); + (base != 0).then_some(base) +} + fn regs() -> Mmio { let base = BASE.load(Ordering::Relaxed); assert!(base != 0, "console UART: a byte moved before `init` found the UART"); diff --git a/kernel/src/arch/aarch64/control_regs.rs b/kernel/src/arch/aarch64/control_regs.rs index 9c82addbb2..275ab97d5d 100644 --- a/kernel/src/arch/aarch64/control_regs.rs +++ b/kernel/src/arch/aarch64/control_regs.rs @@ -5,7 +5,9 @@ //! refuses a CPU whose registers say anything else. Nothing else writes any //! of them. //! -//! Field positions are Arm ARM K.a, chapter D24 (the register descriptions). +//! Field positions are Arm ARM K.a, chapter D24 (the register descriptions), +//! and the GIC architecture specification (IHI 0069H), chapter 12, for the +//! `ICC_` registers. use core::sync::atomic::{AtomicU64, Ordering}; @@ -19,10 +21,9 @@ pub use toyos_bootmap::aarch64::MAIR; /// stack alignment checked at EL1 and EL0 (`SA`, `SA0`), AArch32 EL0's `IT` /// and `SETEND` disabled (`ITD`, `SED` — RES1 on a CPU with no AArch32 EL0, /// and nothing this kernel runs is AArch32), over the bits Armv8.0 makes -/// RES1 (29, 28, 23, 22, 20, 11). `WXN` stays clear: the -/// loader's blocks are writable and executable both until the kernel owns -/// its tables. Little-endian at both levels; EL0's cache maintenance and -/// `WFI`/`WFE` trap. +/// RES1 (29, 28, 23, 22, 20, 11). `WXN` stays clear: the direct map the +/// kernel runs from is writable and executable both. Little-endian at both +/// levels; EL0's cache maintenance and `WFI`/`WFE` trap. pub const SCTLR: u64 = SCTLR_RES1 | 1 << 0 | 1 << 2 | 1 << 3 | 1 << 4 | 1 << 7 | 1 << 8 | 1 << 12; /// `SCTLR_EL1` with the MMU and caches off — [`SCTLR`] less `M`, `C`, `I`, @@ -34,30 +35,49 @@ const SCTLR_RES1: u64 = 1 << 29 | 1 << 28 | 1 << 23 | 1 << 22 | 1 << 20 | 1 << 1 /// `TCR_EL1` but for `IPS`: 48-bit regions from both tables (`T0SZ` = `T1SZ` = /// 16), 4 KiB granules (`TG0` = 0, `TG1` = 2), walks inner-shareable and -/// write-back write-allocate cacheable, 8-bit ASIDs. -pub const TCR: u64 = 16 | 1 << 8 | 1 << 10 | 3 << 12 | 16 << 16 | 1 << 24 | 1 << 26 | 3 << 28 | 2 << 30; +/// write-back write-allocate cacheable, `TTBR0_EL1` naming the ASID (`A1` +/// clear), and 16-bit ASIDs (`AS`), which [`check`] requires the CPU to have. +pub const TCR: u64 = + 16 | 1 << 8 | 1 << 10 | 3 << 12 | 16 << 16 | 1 << 24 | 1 << 26 | 3 << 28 | 2 << 30 | 1 << 36; /// `TCR_EL1.IPS`'s position: the output address size, taken from /// `ID_AA64MMFR0_EL1.PARange` (the same encoding), because a larger `IPS` /// than the CPU implements is a reserved value. pub const TCR_IPS_SHIFT: u64 = 32; -/// `CPACR_EL1`: `FPEN` = 0, so FP and SIMD trap at EL1 and EL0. The kernel is -/// built soft-float and uses neither; user mode's are stage 7's. -pub const CPACR: u64 = 0; +/// `CPACR_EL1`: `FPEN` = 0b11, so FP and SIMD trap at neither EL1 nor EL0. The +/// kernel is built soft-float and touches them only to save and restore a +/// thread's registers across every entry from EL0 (`super::trap`); `ZEN` and +/// `SMEN` stay clear, so SVE and SME trap everywhere. +pub const CPACR: u64 = 0b11 << 20; + +/// `CNTKCTL_EL1`: `EL0VCTEN` alone, so EL0 reads the virtual count and its +/// frequency — the clock page's counter (`toyos_abi::arch::counter`) — and +/// nothing else of the generic timer. +pub const CNTKCTL: u64 = 1 << 1; + +/// `ICC_SRE_EL1`: the GICv3 CPU interface through system registers (`SRE`), +/// with FIQ and IRQ bypass disabled (`DFB`, `DIB`); [`check`] reads back `SRE`. +pub const ICC_SRE: u64 = 0b111; /// `HCR_EL2` when entered at EL2: `RW`, so EL1 is AArch64, and nothing else — -/// no stage-2 translation, no trap, `E2H` clear. +/// no stage-2 translation, no trap, `E2H` clear, and `IMO`/`FMO` clear, so a +/// physical interrupt is taken at EL1. pub const HCR_EL2: u64 = 1 << 31; -/// `CNTHCTL_EL2` when entered at EL2: `EL1PCTEN` and `EL1PCEN`, so EL1 reads the -/// physical counter and programs its timer without trapping. +/// `CNTHCTL_EL2` when entered at EL2: `EL1PCTEN` and `EL1PCEN`, and every +/// other field clear — among them FEAT_ECV's `EL1TVT` and `EL1TVCT`, which +/// reset UNKNOWN and would trap EL1's virtual timer and count to EL2. pub const CNTHCTL_EL2: u64 = 1 << 1 | 1 << 0; /// `CPTR_EL2` when entered at EL2 (`E2H` clear): its RES1 bits (13, 12, 9:0) /// and `TFP` clear, so FP is `CPACR_EL1`'s decision alone. pub const CPTR_EL2: u64 = 0x33FF; +/// `ICC_SRE_EL2` when entered at EL2: [`ICC_SRE`]'s three bits at EL2, and +/// `Enable`, without which EL1's own `ICC_SRE_EL1` traps to EL2. +pub const ICC_SRE_EL2: u64 = 0b1111; + /// `SPSR_EL2` for the drop: EL1 on `SP_EL1` (`M` = 0b0101), `D`, `A`, `I`, `F` masked. pub const SPSR_EL2_TO_EL1: u64 = 0x3C5; @@ -82,24 +102,37 @@ pub fn tcr() -> u64 { /// Read every EL1 register the declaration names back, and refuse a CPU that /// holds anything else; then say what it holds. pub fn check() { + // `ID_AA64MMFR0_EL1.ASIDBits` = 0b0010: without 16-bit ASIDs `TCR_EL1.AS` + // is RES0, and the read-back below would name the register, not the reason. + let asid_bits = read!("id_aa64mmfr0_el1") >> 4 & 0xF; + assert_eq!(asid_bits, 0b0010, "control registers: this CPU has 8-bit ASIDs, and the declaration's TCR_EL1.AS needs 16"); let declared = [ ("SCTLR_EL1", read!("sctlr_el1"), SCTLR), ("TCR_EL1", read!("tcr_el1"), tcr()), ("MAIR_EL1", read!("mair_el1"), MAIR), ("CPACR_EL1", read!("cpacr_el1"), CPACR), + ("CNTKCTL_EL1", read!("cntkctl_el1"), CNTKCTL), ]; for (name, live, value) in declared { assert_eq!(live, value, "control registers: {name} holds {live:#x}, and the declaration says {value:#x}"); } + // `SRE` alone: `DFB` and `DIB` are RAO/WI or RAZ/WI as the implementation, + // or a hypervisor under it, chooses, and this kernel uses neither bypass. + let sre = read!("S3_0_C12_C12_5"); + assert!(sre & 1 != 0, "control registers: ICC_SRE_EL1 reads {sre:#x}, so the GICv3 CPU interface is not in system registers"); // What the drop from EL2 left, or the EL1 entry kept: EL1, on `SP_EL1`. let (el, spsel) = (read!("CurrentEL") >> 2 & 3, read!("SPSel") & 1); assert_eq!((el, spsel), (1, 1), "control registers: running at EL{el} on SP_EL{spsel}, not EL1 on SP_EL1"); let el = ENTRY_EL.load(Ordering::Relaxed); log!( - "control registers: SCTLR_EL1={SCTLR:#x} TCR_EL1={:#x} MAIR_EL1={:#x} CPACR_EL1={CPACR:#x}, \ - as declared; entered at EL{el}{}", + "control registers: SCTLR_EL1={SCTLR:#x} TCR_EL1={:#x} MAIR_EL1={:#x} CPACR_EL1={CPACR:#x} \ + CNTKCTL_EL1={CNTKCTL:#x} ICC_SRE_EL1.SRE=1, as declared; entered at EL{el}{}", tcr(), MAIR, - if el == 2 { ", HCR_EL2 read back as declared, CNTHCTL_EL2/CPTR_EL2 written, and dropped to EL1" } else { "" }, + if el == 2 { + ", HCR_EL2 read back as declared, CNTHCTL_EL2/CNTVOFF_EL2/CPTR_EL2/ICC_SRE_EL2 written, and dropped to EL1" + } else { + "" + }, ); } diff --git a/kernel/src/arch/aarch64/cpu.rs b/kernel/src/arch/aarch64/cpu.rs index cfe942ddd0..b32f79aa38 100644 --- a/kernel/src/arch/aarch64/cpu.rs +++ b/kernel/src/arch/aarch64/cpu.rs @@ -57,6 +57,14 @@ pub fn thread_pointer() -> u64 { tp } +/// Install the thread pointer the next return to EL0 runs with. +/// # Safety +/// `tp` is the thread's own, which the kernel computed for it. +pub unsafe fn write_thread_pointer(tp: u64) { + // SAFETY: `TPIDR_EL0` is EL0's register; the kernel never reads through it. + unsafe { asm!("msr tpidr_el0, {}", in(reg) tp, options(nomem, nostack, preserves_flags)) }; +} + /// Unmask interrupts on this CPU: `DAIF.I` and `DAIF.F`. pub fn enable_interrupts() { // SAFETY: writes two `DAIF` bits; a compiler barrier, so no access moves across it. diff --git a/kernel/src/arch/aarch64/entry.rs b/kernel/src/arch/aarch64/entry.rs index 8207fc9d14..31a0cdf09b 100644 --- a/kernel/src/arch/aarch64/entry.rs +++ b/kernel/src/arch/aarch64/entry.rs @@ -1,27 +1,97 @@ -//! Where a new context starts: the frame `switch` restores and the -//! trampolines to user mode and to a kernel thread, the port's stages 4 and 7. +//! Where a new context starts: the frame [`super::switch::context_switch`] +//! restores it from, and the trampolines its return lands in — to EL0 for the +//! first time, or into a kernel thread's body. +//! +//! A trampoline to EL0 leaves nothing of the kernel in a register: every +//! general register but the argument is zeroed, and the FP/SIMD state is the +//! zero the initial frame carried. +use super::switch::{DAIF_AT, FRAME_BYTES, RETURN_AT}; -pub(crate) extern "C" fn process_start() { - owed!("user mode", "stage 7") +/// `DAIF` a new context starts with: every exception masked, as an entry from +/// EL0 is, for `trampoline_entry`'s contract. +const DAIF_MASKED: u64 = 0b1111 << 6; + +/// Zero x1–x30, the registers a first entry to EL0 must not carry. +macro_rules! zero_registers { + () => { + concat!( + "mov x1, xzr\n", "mov x2, xzr\n", "mov x3, xzr\n", "mov x4, xzr\n", "mov x5, xzr\n", + "mov x6, xzr\n", "mov x7, xzr\n", "mov x8, xzr\n", "mov x9, xzr\n", "mov x10, xzr\n", + "mov x11, xzr\n", "mov x12, xzr\n", "mov x13, xzr\n", "mov x14, xzr\n", "mov x15, xzr\n", + "mov x16, xzr\n", "mov x17, xzr\n", "mov x18, xzr\n", "mov x19, xzr\n", "mov x20, xzr\n", + "mov x21, xzr\n", "mov x22, xzr\n", "mov x23, xzr\n", "mov x24, xzr\n", "mov x25, xzr\n", + "mov x26, xzr\n", "mov x27, xzr\n", "mov x28, xzr\n", "mov x29, xzr\n", "mov x30, xzr\n", + ) + }; } -pub(crate) extern "C" fn thread_start() { - owed!("user mode", "stage 7") +/// A thread's first entry to EL0 at `x19` on the stack `x20`, with `x0` = +/// `x21`: a process's argument is zero, a thread's is its own. `SPSR_EL1` +/// zero is EL0 with every exception unmasked, and `SP_EL1` is left at the +/// top of this thread's kernel stack, which is where its entries land. +#[unsafe(naked)] +pub(crate) extern "C" fn process_start() { + core::arch::naked_asm!( + "bl {unlock}", + "msr elr_el1, x19", + "msr sp_el0, x20", + "msr spsr_el1, xzr", + "mov x0, x21", + zero_registers!(), + "eret", + unlock = sym crate::sched::driver::trampoline_entry, + ); } +/// A thread's first entry is a process's, with its argument in `x0`. +pub(crate) use process_start as thread_start; + +/// Entry point for a kernel thread: `x19` = body, `x21` = argument. Never +/// reaches EL0; unmasks interrupts once `trampoline_entry`, which requires +/// them masked, is done. +#[unsafe(naked)] pub(crate) extern "C" fn kernel_start() { - owed!("kernel threads", "stage 4") + core::arch::naked_asm!( + "bl {unlock}", + "msr daifclr, #3", + "mov x0, x21", + "blr x19", + "bl {returned}", + unlock = sym crate::sched::driver::trampoline_entry, + returned = sym kernel_thread_returned, + ); +} + +/// What [`kernel_start`] calls when a kernel thread's body returns: panics rather than halting silently. +extern "C" fn kernel_thread_returned() -> ! { + panic!("a kernel thread's body returned; nothing runs on this stack now"); } +/// Lay out, just below `top`, the frame `context_switch` restores a new +/// context from, and answer the stack pointer that names it: `trampoline` is +/// where its return lands, with the entry, the stack and the argument where +/// the trampolines read them (`x19`, `x20`, `x21`), and FP/SIMD state zero. /// # Safety -/// `top` is the end of a fresh kernel stack nothing else references. +/// `top` is the 16-byte-aligned end of a fresh kernel stack nothing else +/// references, at least [`FRAME_BYTES`] deep. pub unsafe fn initial_frame( - _top: u64, - _trampoline: unsafe extern "C" fn(), - _user_entry: u64, - _user_sp: u64, - _arg: u64, + top: u64, + trampoline: unsafe extern "C" fn(), + user_entry: u64, + user_sp: u64, + arg: u64, ) -> u64 { - owed!("kernel threads", "stage 4") + let frame = top - FRAME_BYTES as u64; + // SAFETY: `[frame, top)` is the top `FRAME_BYTES` of the stack the caller owns. + unsafe { + core::ptr::write_bytes(frame as *mut u8, 0, FRAME_BYTES); + let word = |at: usize| (frame + at as u64) as *mut u64; + *word(0) = user_entry; + *word(8) = user_sp; + *word(16) = arg; + *word(DAIF_AT) = DAIF_MASKED; + *word(RETURN_AT) = trampoline as usize as u64; + } + frame } diff --git a/kernel/src/arch/aarch64/fpu.rs b/kernel/src/arch/aarch64/fpu.rs index 1613f2f8d5..94a674d4ba 100644 --- a/kernel/src/arch/aarch64/fpu.rs +++ b/kernel/src/arch/aarch64/fpu.rs @@ -1,2 +1,9 @@ -//! User FP/SIMD state: `CPACR_EL1.FPEN` traps it at EL1 and EL0 until the -//! port's stage 7 saves and restores it. The kernel itself never uses it. +//! User FP/SIMD state: v0–v31, `FPCR` and `FPSR`. The kernel is built +//! soft-float and never touches them, so a thread's own stay in the registers +//! across every entry from EL0; they are saved and restored only where a +//! thread stops running, in [`super::switch`]'s frame, and a thread that has +//! never run starts from zero — `FPCR` zero is round-to-nearest with every +//! trap disabled (Arm ARM K.a, D24.2.62). + +/// The state's bytes in a switch frame: 32 128-bit registers, then `FPCR` and `FPSR`. +pub const STATE_BYTES: usize = 32 * 16 + 16; diff --git a/kernel/src/arch/aarch64/hw.rs b/kernel/src/arch/aarch64/hw.rs index c7f2adf025..5c68a49f74 100644 --- a/kernel/src/arch/aarch64/hw.rs +++ b/kernel/src/arch/aarch64/hw.rs @@ -1,11 +1,15 @@ //! `KernelHw` — the kernel's side of the scheduler-core hardware boundary: -//! the generic timer, SGIs, `WFI` and the context switch, the port's stage 4. +//! the generic timer, SGIs, `WFI` and the context switch. Nothing here makes +//! a scheduling decision. -use toyos_sched::cpu::RunToken; +use toyos_sched::cpu::{RunToken, SleepToken}; +use toyos_sched::fair::QUANTUM_NS; use toyos_sched::hw::{CpuId, Hw, Kicker, Machine, Nanos, TraceEvent}; use toyos_sched::task::{TaskAccounting, TaskKey}; -use crate::sched::payload::KernelPayload; +use super::switch::{context_switch, RETURN_AT}; +use super::{cpu, irqchip, percpu}; +use crate::sched::payload::{KernelCtx, KernelPayload}; /// The one instance; zero-sized, holds no per-CPU state. pub static HW: KernelHw = KernelHw; @@ -18,8 +22,8 @@ pub fn now_ns() -> u64 { } impl Kicker for KernelHw { - fn kick(&self, _target: CpuId) { - owed!("the interrupt controller", "stage 4") + fn kick(&self, target: CpuId) { + irqchip::kick_cpu(target.0); } } @@ -28,47 +32,180 @@ impl Machine for KernelHw { Nanos(crate::clock::nanos_since_boot()) } - fn set_timer(&self, _deadline: Nanos) { - owed!("the timer", "stage 4") + /// The absolute deadline as the one-shot's span from now; one already past + /// becomes the floor rather than an interrupt at once. + fn set_timer(&self, deadline: Nanos) { + irqchip::arm_one_shot(deadline.0.saturating_sub(self.now().0)); } fn stop_timer(&self) { - owed!("the timer", "stage 4") + irqchip::stop_timer(); } + /// `WFI` with interrupts masked, then unmasked: a pending interrupt wakes + /// `WFI` whatever `DAIF` says (Arm ARM K.a, D1.6.2), so a wake that lands + /// between the decision and the wait is taken right after it, not slept through. fn halt(&self) { - owed!("the interrupt controller", "stage 4") + // SAFETY: waits for an interrupt and unmasks `I` and `F`; touches no memory. + unsafe { core::arch::asm!("wfi", "msr daifclr, #3", "isb", options(nomem, nostack)) }; } - fn need_resched(&self, _cpu: CpuId) { - owed!("the interrupt controller", "stage 4") + /// An SGI is how a remote CPU's `need_resched` gets set — there is no way to write it directly. + fn need_resched(&self, cpu: CpuId) { + if cpu.0 == percpu::cpu_id() { + crate::preempt::set_need_resched(); + } else { + self.kick(cpu); + } } fn trace(&self, ev: TraceEvent) { crate::trace::record(ev); } + + /// Diagnostic builds arm a periodic wake before halting so a quiescent CPU + /// still reports; and a boot under a deadline re-arms after the wake, as + /// x86-64's does, because the poll rests on some CPU taking a timer interrupt. + fn idle_wait(&self, token: SleepToken) { + let _consumed = token; + #[cfg(feature = "boot-actuators")] + if crate::actuator::diag_tick() { + irqchip::arm_within(DIAG_TICK_NS); + } + self.halt(); + if crate::deadline::armed() { + irqchip::arm_within(QUANTUM_NS); + } + } +} + +/// Longest sleep on a `diag-tick` build; kept under `heartbeat`'s reporting period so a healthy CPU reports on every line. +#[cfg(feature = "boot-actuators")] +const DIAG_TICK_NS: u64 = 100_000_000; + +/// Which context each CPU last switched onto; read by [`report_contexts`] on crash. +static RUNNING_CTX: [core::sync::atomic::AtomicU64; crate::sched::MAX_CPUS] = + [const { core::sync::atomic::AtomicU64::new(0) }; crate::sched::MAX_CPUS]; + +/// Prints which CPU is standing on which context and stack, on every kernel crash. +/// +/// Allocates, locks or formats nothing but integers, since a crash may already +/// hold any lock this could try to take. +pub fn report_contexts(sp: u64, subject: Option) { + let me = percpu::cpu_id() as usize; + let count = (super::smp::cpu_count() as usize).min(crate::sched::MAX_CPUS); + let mine = RUNNING_CTX.get(me).map_or(0, |slot| slot.load(core::sync::atomic::Ordering::Relaxed)); + let subject = subject.unwrap_or(mine); + crate::log!(" Contexts: cpu{me} crashed at sp={sp:#018x}, asking about ctx {subject:#x}"); + for (cpu, slot) in RUNNING_CTX.iter().enumerate().take(count) { + let held = slot.load(core::sync::atomic::Ordering::Relaxed); + if !crate::mm::is_kernel_addr(held) || !held.is_multiple_of(8) { + crate::log!(" cpu{cpu} is on ctx {held:#x} (never switched, or not a context)"); + continue; + } + // SAFETY: a pointer this kernel's own `Hw::switch` stored, into the boxed, always-mapped direct map. + let ctx = unsafe { &*(held as *const KernelCtx) }; + let top = ctx.kernel_stack_top; + match ctx.id { + None => crate::log!(" cpu{cpu} is on ctx {held:#x} (its idle context) saved_sp={:#018x}", ctx.sp), + Some(id) => crate::log!( + " cpu{cpu} is on ctx {held:#x} pid={} tid={} stack_top={top:#018x} saved_sp={:#018x}{}", + id.0.raw(), + id.1.raw(), + ctx.sp, + if cpu != me && sp <= top && sp > top.wrapping_sub(crate::process::KERNEL_STACK_SIZE as u64) { + " <== AND THIS CRASH IS ON THAT STACK" + } else { + "" + }, + ), + } + } + crate::mm::report_on_crash(); +} + +/// Panics before the switch's `ret` would land somewhere that makes the failure unnameable. +#[cold] +#[inline(never)] +fn switch_frame_is_wrong(ctx: &KernelCtx, sp: u64) -> ! { + report_contexts(sp, Some(ctx as *const KernelCtx as u64)); + panic!( + "context_switch: the frame about to be restored is not one — its sp {sp:#018x} is not a \ + 16-byte-aligned kernel address, or its return slot is not kernel text (stack top {:#018x})", + ctx.kernel_stack_top, + ); +} + +/// The incoming context's saved stack pointer, checked; the only load of it. +#[inline] +#[must_use] +fn check_switch_frame(ctx: &KernelCtx) -> u64 { + let sp = ctx.sp; + if !crate::mm::is_kernel_addr(sp) || !sp.is_multiple_of(16) { + switch_frame_is_wrong(ctx, sp); + } + #[cfg(feature = "stack-witness")] + { + let top = match ctx.id { + Some(_) => ctx.kernel_stack_top, + None => percpu::idle_stack_top(), + }; + if sp > top || sp <= top - crate::process::KERNEL_STACK_SIZE as u64 { + switch_frame_is_wrong(ctx, sp); + } + } + // SAFETY: `sp` is aligned inside the incoming stack, so its frame's return slot is mapped. + let ret = unsafe { core::ptr::read_volatile((sp + RETURN_AT as u64) as *const u64) }; + if !crate::mm::is_kernel_addr(ret) { + switch_frame_is_wrong(ctx, sp); + } + sp } impl Hw for KernelHw { type Payload = KernelPayload; - unsafe fn switch(&self, _token: RunToken) { - owed!("the context switch", "stage 4") + /// Outgoing per-CPU state is captured, and the incoming root and thread + /// pointer installed, before the stack pointer moves — after that this + /// frame no longer exists. + unsafe fn switch(&self, token: RunToken) { + let save = token.save_ptr(); + let restore = token.restore_ptr(); + // SAFETY: `save`/`restore` are live Box-backed contexts from + // `SchedPass::finish`, freed only by a later pass. + unsafe { + (*save).thread_pointer = cpu::thread_pointer(); + (*save).preempt = crate::preempt::count(); + let incoming: &KernelCtx = &*restore; + let sp = check_switch_frame(incoming); + crate::preempt::set_count(incoming.preempt); + percpu::set_current_tid(incoming.id.map(|id| id.1)); + percpu::set_current_pid(incoming.id.map(|id| id.0)); + match incoming.id { + Some(_) => { + #[cfg(feature = "boot-actuators")] + crate::heartbeat::note_dispatch(); + percpu::set_kernel_stack(incoming.kernel_stack_top); + incoming.root.activate(); + cpu::write_thread_pointer(incoming.thread_pointer); + } + None => { + percpu::set_kernel_stack(percpu::idle_stack_top()); + incoming.root.activate(); + } + } + RUNNING_CTX[percpu::cpu_id() as usize].store(restore as u64, core::sync::atomic::Ordering::Relaxed); + context_switch(&raw mut (*save).sp, sp); + } } - fn release(&self, _key: TaskKey, _payload: KernelPayload, _acct: TaskAccounting) { - owed!("the context switch", "stage 4") + /// Reached once per task, from a later pass running on another stack, so + /// dropping `payload` here never frees the stack this call stands on. + fn release(&self, _key: TaskKey, payload: KernelPayload, acct: TaskAccounting) { + payload.handle.finalize(acct); } } -/// What every kernel crash says about the machine's contexts: until stage 4 -/// switches any, only the memory facts. Never owed: a crash report that -/// panicked would bury the crash. -pub fn report_contexts(sp: u64, _subject: Option) { - crate::log!(" Contexts: the boot CPU crashed at sp={sp:#018x}; no context has been switched"); - crate::mm::report_on_crash(); -} - /// AMD's `SYSRET` erratum has no AArch64 counterpart; the probe is x86-64's. #[cfg(feature = "boot-actuators")] pub fn sysret_ss_probe(_parkable: &crate::scheduler::Parkable) { diff --git a/kernel/src/arch/aarch64/irqchip.rs b/kernel/src/arch/aarch64/irqchip.rs index 3edc03a567..d3b9a42b08 100644 --- a/kernel/src/arch/aarch64/irqchip.rs +++ b/kernel/src/arch/aarch64/irqchip.rs @@ -1,31 +1,361 @@ -//! The interrupt controller: a GICv3 — distributor, one redistributor per CPU -//! and an ITS for MSIs — the port's stage 4. Until then no interrupt is -//! delivered, and every way of raising one is owed. +//! The interrupt controller and the timer: a GICv3 — the distributor the +//! MADT names, this CPU's redistributor, and the system-register CPU +//! interface (GIC architecture specification IHI 0069H) — and the generic +//! timer's EL1 virtual timer (Arm ARM K.a, chapter D12), whose PPI the GTDT +//! names. +//! +//! **What this kernel takes is SGIs and the timer's PPI, and nothing else.** +//! No SPI is routed and no LPI exists: a device's interrupt is a message the +//! GICv3 ITS translates, and every device that would take one is either a +//! driver the small-kernel track moves out of the kernel or a claimed function, +//! which the SMMUv3 of the port's stage 6 must translate first +//! (`super::msi_message`). +//! +//! **The virtual timer, not the physical**: it is the one EL1 owns outright — +//! under a hypervisor the physical one traps — and with `CNTVOFF_EL2` written +//! zero by the entry from EL2 the two count alike. It is level-triggered, so a +//! handler that neither re-arms nor stops it takes it again at once. +use core::sync::atomic::{AtomicU32, AtomicU64, Ordering::Relaxed}; -/// Wake `cpu` so it runs a scheduler pass: an SGI. -pub fn kick_cpu(_cpu: u32) { - owed!("the interrupt controller", "stage 4") +use toyos_acpi::MadtEntry; + +use super::{cpu, percpu}; +use crate::drivers::acpi::direct_phys; +use crate::log; +use crate::mm::policy::MmioPolicy; +use crate::mm::{DirectMap, Mmio}; +use crate::time::{Duration, Floor}; + +/// The SGI that asks a CPU for a scheduler pass: x86-64's kick, which rides +/// the timer's vector there. +pub(super) const SGI_KICK: u32 = 0; +/// The SGI `irq-storm` floods this CPU with. +#[cfg(feature = "boot-actuators")] +pub(super) const SGI_STORM: u32 = 2; + +/// `GICD_CTLR`, and its `ARE` (affinity routing — `ARE_NS` as a non-secure +/// access sees it) and Group 1 enable (`EnableGrp1` or `EnableGrp1A`, the +/// same bit either way), and the write-pending bit. +const GICD_CTLR: u64 = 0x0000; +const CTLR_ARE: u32 = 1 << 4; +const CTLR_GRP1: u32 = 1 << 1; +const RWP: u32 = 1 << 31; +/// `GICD_PIDR2.ArchRev`, bits 7:4: 3 is GICv3, 4 is GICv4. +const GICD_PIDR2: u64 = 0xFFE8; + +/// A redistributor's frames: `RD_base`, then `SGI_base` 64 KiB above it, and +/// two more for virtual LPIs where `GICR_TYPER.VLPIS` says so. +const GICR_CTLR: u64 = 0x0000; +const GICR_TYPER: u64 = 0x0008; +const GICR_WAKER: u64 = 0x0014; +const WAKER_PROCESSOR_SLEEP: u32 = 1 << 1; +const WAKER_CHILDREN_ASLEEP: u32 = 1 << 2; +const TYPER_VLPIS: u64 = 1 << 1; +const TYPER_LAST: u64 = 1 << 4; +const FRAME: u64 = 0x1_0000; +const SGI_BASE: u64 = FRAME; +const GICR_IGROUPR0: u64 = SGI_BASE + 0x0080; +const GICR_ISENABLER0: u64 = SGI_BASE + 0x0100; +const GICR_ICENABLER0: u64 = SGI_BASE + 0x0180; +const GICR_ICPENDR0: u64 = SGI_BASE + 0x0280; +const GICR_ICACTIVER0: u64 = SGI_BASE + 0x0380; +const GICR_IPRIORITYR: u64 = SGI_BASE + 0x0400; +const GICR_ICFGR1: u64 = SGI_BASE + 0x0C04; +const GICR_IGRPMODR0: u64 = SGI_BASE + 0x0D00; + +/// Every SGI and PPI this kernel takes runs at one priority, below the mask. +const PRIORITY: u8 = 0x80; +/// `ICC_PMR_EL1`: priorities numerically below this are signalled. +const PRIORITY_MASK: u64 = 0xF0; + +/// What the GIC answers `ICC_IAR1_EL1` with when nothing is pending for it. +const SPURIOUS: u32 = 1023; + +/// The timer's PPI, as the GTDT names it; zero until [`init`]. +static TIMER_INTID: AtomicU32 = AtomicU32::new(0); +/// The distributor's frame, for [`log_state`]; zero until [`init`]. +static GICD: AtomicU64 = AtomicU64::new(0); + +/// Wait until `done`, for at most `1 / per_second` of a second counted at the +/// rate firmware states: the boot has no calibrated clock yet. +fn settles(per_second: u64, what: &str, done: impl Fn() -> bool) { + let hz = cpu::stated_counter_hz().expect("GIC: CNTFRQ_EL0 states no rate to bound a wait with"); + let until = cpu::counter() + hz / per_second; + while !done() { + assert!(cpu::counter() < until, "GIC: {what} did not settle within 1/{per_second} s"); + core::hint::spin_loop(); + } +} + +fn read_sysreg_pmr() -> u64 { + let v: u64; + // SAFETY: reads `ICC_PMR_EL1`, which `ICC_SRE_EL1.SRE` (the declaration) makes accessible. + unsafe { core::arch::asm!("mrs {}, S3_0_C4_C6_0", out(reg) v, options(nomem, nostack, preserves_flags)) }; + v +} + +/// Bring the distributor, this CPU's redistributor, its CPU interface and its +/// timer up: every SGI and the timer's PPI enabled at [`PRIORITY`], the timer +/// stopped. Interrupts stay masked at `DAIF`; the caller unmasks them. +pub fn init(rsdp_addr: u64) { + let madt = toyos_acpi::find_table(direct_phys(), rsdp_addr, b"APIC", toyos_acpi::MADT_ENTRIES) + .unwrap_or_else(|e| panic!("GIC: the MADT is unusable: {e:?}")); + let me = cpu::hardware_id(); + let (mut gicd, mut own_frame, mut ranges) = (None, None, [(0u64, 0u32); 4]); + let mut range_count = 0; + for entry in toyos_acpi::madt_entries(&madt) { + match entry { + Ok(MadtEntry::Gicd { base, .. }) => gicd = Some(base), + Ok(MadtEntry::Gicr { base, length }) => { + assert!(range_count < ranges.len(), "GIC: the MADT names more redistributor ranges than {}", ranges.len()); + ranges[range_count] = (base, length); + range_count += 1; + } + Ok(MadtEntry::Gicc(gicc)) if packed_affinity(gicc.mpidr) == me && gicc.gicr_base != 0 => { + own_frame = Some(gicc.gicr_base) + } + Ok(_) => {} + Err(halt) => panic!("GIC: a MADT entry at +{} declares {} bytes of a {}-byte list", halt.at, halt.declared, halt.list_len), + } + } + let gicd = gicd.expect("GIC: the MADT names no distributor"); + let distributor = crate::mm::paging::map_mmio(gicd, FRAME, MmioPolicy::Uncacheable); + GICD.store(gicd, Relaxed); + let revision = distributor.read_u32(GICD_PIDR2) >> 4 & 0xF; + assert!(revision >= 3, "GIC: GICD_PIDR2.ArchRev is {revision}, and this kernel drives a GICv3 or later"); + + // The frame is found in the ranges when the GICC entry does not name it: + // each redistributor says whose it is in `GICR_TYPER`'s affinity. + let redistributor = match own_frame { + Some(frame) => crate::mm::paging::map_mmio(frame, 2 * FRAME, MmioPolicy::Uncacheable), + None => ranges[..range_count] + .iter() + .find_map(|&(base, length)| find_redistributor(base, u64::from(length), me)) + .unwrap_or_else(|| panic!("GIC: no redistributor answers for MPIDR affinity {me:#x}")), + }; + + // The distributor: affinity routing on, and Group 1 on. + distributor.write_u32(GICD_CTLR, CTLR_ARE | CTLR_GRP1); + settles(100, "GICD_CTLR", || distributor.read_u32(GICD_CTLR) & RWP == 0); + assert!( + distributor.read_u32(GICD_CTLR) & CTLR_ARE != 0, + "GIC: GICD_CTLR.ARE reads clear, so the distributor stays in legacy mode and no redistributor is used" + ); + + // Awake, then every SGI and PPI off and cleared of what firmware left. + let waker = redistributor.read_u32(GICR_WAKER); + redistributor.write_u32(GICR_WAKER, waker & !WAKER_PROCESSOR_SLEEP); + settles(100, "GICR_WAKER.ChildrenAsleep", || redistributor.read_u32(GICR_WAKER) & WAKER_CHILDREN_ASLEEP == 0); + redistributor.write_u32(GICR_ICENABLER0, u32::MAX); + settles(100, "GICR_CTLR after disabling every SGI and PPI", || redistributor.read_u32(GICR_CTLR) & (1 << 3) == 0); + redistributor.write_u32(GICR_ICPENDR0, u32::MAX); + redistributor.write_u32(GICR_ICACTIVER0, u32::MAX); + redistributor.write_u32(GICR_IGROUPR0, u32::MAX); + redistributor.write_u32(GICR_IGRPMODR0, 0); + let priorities = u32::from_ne_bytes([PRIORITY; 4]); + for word in 0..8 { + redistributor.write_u32(GICR_IPRIORITYR + word * 4, priorities); + } + + let timer = toyos_acpi::gtdt(direct_phys(), rsdp_addr) + .unwrap_or_else(|e| panic!("GIC: the GTDT is unusable: {e:?}")) + .virtual_el1; + assert!((16..32).contains(&timer.gsiv), "GIC: the GTDT puts the virtual timer at INTID {}, which is not a PPI", timer.gsiv); + TIMER_INTID.store(timer.gsiv, Relaxed); + // Each PPI's two bits in `GICR_ICFGR1`: 0b10 edge, 0b00 level. + let edge = u32::from(timer.edge()) << (2 * (timer.gsiv - 16) + 1); + redistributor.write_u32(GICR_ICFGR1, edge); + + stop_timer_hardware(); + let enabled = 1 << SGI_KICK | 1 << timer.gsiv; + #[cfg(feature = "boot-actuators")] + let enabled = enabled | 1 << u32::from(super::trap::LOG_NEST_VECTOR) | 1 << SGI_STORM; + redistributor.write_u32(GICR_ISENABLER0, enabled); + + // The CPU interface: the priority mask open above `PRIORITY`, one + // priority drop and deactivation per EOI, Group 1 on. + // SAFETY: GICv3 CPU interface registers, which the declaration's + // `ICC_SRE_EL1.SRE` makes system registers; none touches memory. + unsafe { + core::arch::asm!( + "msr S3_0_C4_C6_0, {pmr}", + "msr S3_0_C12_C12_4, xzr", + "msr S3_0_C12_C12_7, {one}", + "isb", + pmr = in(reg) PRIORITY_MASK, + one = in(reg) 1u64, + options(nomem, nostack, preserves_flags), + ); + } + assert_eq!(read_sysreg_pmr(), PRIORITY_MASK, "GIC: ICC_PMR_EL1 did not take {PRIORITY_MASK:#x}"); + log!( + "GIC: v{revision} distributor at {gicd:#x}, this CPU's redistributor at {:#x}; SGIs and the virtual \ + timer's PPI {} ({}) enabled at priority {PRIORITY:#x}", + redistributor.addr() - crate::mm::PHYS_OFFSET, + timer.gsiv, + if timer.edge() { "edge" } else { "level" }, + ); +} + +/// `MPIDR_EL1`'s four affinity fields packed as `cpu::hardware_id` packs +/// them, and as `GICR_TYPER` carries them in its top word. +fn packed_affinity(mpidr: u64) -> u32 { + ((mpidr & 0xFF_FFFF) | ((mpidr >> 8) & 0xFF00_0000)) as u32 +} + +/// The redistributor frame in `[base, base + length)` whose affinity is `me`. +fn find_redistributor(base: u64, length: u64, me: u32) -> Option { + let range = crate::mm::paging::map_mmio(base, length, MmioPolicy::Uncacheable); + let mut offset = 0; + while offset + 2 * FRAME <= length { + let typer = range.read_u64(offset + GICR_TYPER); + if (typer >> 32) as u32 == me { + return Some(Mmio::new(DirectMap::from_phys(base + offset), 2 * FRAME)); + } + if typer & TYPER_LAST != 0 { + return None; + } + offset += if typer & TYPER_VLPIS != 0 { 4 * FRAME } else { 2 * FRAME }; + } + None +} + +/// The INTID the GIC hands this CPU, or `None` when it answered spurious. +pub(super) fn acknowledge() -> Option { + let intid: u64; + // SAFETY: reads `ICC_IAR1_EL1`, which acknowledges the highest pending + // Group 1 interrupt; the caller ends every one it is given. + unsafe { core::arch::asm!("mrs {}, S3_0_C12_C12_0", out(reg) intid, options(nomem, nostack, preserves_flags)) }; + let intid = (intid & 0xFF_FFFF) as u32; + (intid != SPURIOUS).then_some(intid) +} + +/// Drop the running priority and deactivate `intid`. +pub(super) fn end(intid: u32) { + // SAFETY: writes `ICC_EOIR1_EL1` with an INTID [`acknowledge`] handed out. + unsafe { core::arch::asm!("msr S3_0_C12_C12_1, {}", in(reg) u64::from(intid), options(nomem, nostack, preserves_flags)) }; +} + +/// The timer's INTID, which [`init`] took from the GTDT. +pub(super) fn timer_intid() -> u32 { + TIMER_INTID.load(Relaxed) +} + +/// Raise SGI `intid` on the CPU whose packed affinity is `target`. +fn sgi(intid: u32, target: u32) { + let (aff0, aff1, aff2, aff3) = + (u64::from(target & 0xFF), u64::from(target >> 8 & 0xFF), u64::from(target >> 16 & 0xFF), u64::from(target >> 24)); + let value = aff3 << 48 | (aff0 >> 4) << 44 | aff2 << 32 | u64::from(intid) << 24 | aff1 << 16 | 1 << (aff0 & 0xF); + // SAFETY: writes `ICC_SGI1R_EL1`, which raises an SGI and touches no + // memory; the `ISB` sends it before whatever follows. + unsafe { core::arch::asm!("msr S3_0_C12_C11_5, {}", "isb", in(reg) value, options(nomem, nostack, preserves_flags)) }; +} + +/// Wake `cpu` so it runs a scheduler pass. Before the port's stage 5 the boot +/// CPU is the only one, so the only `cpu` there is is this one. +pub fn kick_cpu(cpu: u32) { + assert_eq!(cpu, percpu::cpu_id(), "irqchip: a kick for cpu{cpu}, and other CPUs are the port's stage 5"); + sgi(SGI_KICK, cpu::hardware_id()); } pub fn kick_all_but_self() { - owed!("the interrupt controller", "stage 4") + let me = percpu::cpu_id(); + for cpu in (0..super::smp::cpu_count()).filter(|&cpu| cpu != me) { + kick_cpu(cpu); + } } -/// Raise `vector` on this CPU. -pub fn send_self(_vector: u8) { - owed!("the interrupt controller", "stage 4") +/// Raise `vector` — an SGI's INTID — on this CPU. +pub fn send_self(vector: u8) { + sgi(u32::from(vector), cpu::hardware_id()); } -/// A pseudo-NMI to `cpu`. +/// A pseudo-NMI: an interrupt at a priority `DAIF.I` does not mask, which +/// needs `ICC_PMR_EL1` priority masking in place of `DAIF` everywhere. pub fn send_nmi(_cpu: u32) { - owed!("the interrupt controller", "stage 4") + owed!("a pseudo-NMI", "no stage yet") } -/// This CPU's timer, armed to fire within `nanos`. -pub fn arm_within(_nanos: u64) { - owed!("the timer", "stage 4") +/// Nothing to stop: before the port's stage 5 the boot CPU is the only one running. +pub fn stop_other_cpus() {} + +/// The shortest one-shot this kernel arms: above an interrupt's entry and +/// return, as x86-64's floor is. +const MIN_ONE_SHOT: Floor = Floor::policy(Duration::from_micros(10), "above an interrupt entry and eret, a thousandth of QUANTUM_NS"); + +fn counter_ticks(nanos: u64) -> u64 { + crate::clock::tsc_ticks(nanos).max(crate::clock::tsc_ticks(MIN_ONE_SHOT.nanos())).max(1) } -/// Nothing to stop: before stage 5 the boot CPU is the only one running. -pub fn stop_other_cpus() {} +/// `CNTV_CTL_EL0.ENABLE`; `IMASK` stays clear. +const TIMER_ENABLE: u64 = 1; + +/// Fire `ticks` counter ticks from now, and remember it as what an EL1 fire re-arms with. +fn arm_ticks(ticks: u64) { + percpu::set_armed_ticks(ticks); + // SAFETY: the EL1 virtual timer's comparator and control; CPACR has + // nothing to say about them and `CNTKCTL_EL1` keeps EL0 out. + unsafe { + core::arch::asm!( + "msr cntv_cval_el0, {cval}", + "msr cntv_ctl_el0, {enable}", + "isb", + cval = in(reg) cpu::counter() + ticks, + enable = in(reg) TIMER_ENABLE, + options(nomem, nostack, preserves_flags), + ); + } +} + +fn stop_timer_hardware() { + // SAFETY: the EL1 virtual timer's control, cleared: it asserts nothing. + unsafe { core::arch::asm!("msr cntv_ctl_el0, xzr", "isb", options(nomem, nostack, preserves_flags)) }; +} + +/// This CPU's timer, armed to fire `nanos` from now, or after [`MIN_ONE_SHOT`] if that is longer. +pub fn arm_one_shot(nanos: u64) { + arm_ticks(counter_ticks(nanos)); + crate::trace::trace(crate::trace::Kind::TimerArm, nanos as u32); +} + +/// This CPU's timer, armed to fire within `nanos`: sooner than it is armed +/// for, or armed if it is stopped. +pub fn arm_within(nanos: u64) { + let want = counter_ticks(nanos); + let remaining = match percpu::armed_ticks() { + 0 => want, + _ => { + let cval: u64; + // SAFETY: reads the EL1 virtual timer's comparator. + unsafe { core::arch::asm!("mrs {}, cntv_cval_el0", out(reg) cval, options(nomem, nostack, preserves_flags)) }; + cval.saturating_sub(cpu::counter()) + } + }; + arm_ticks(want.min(remaining).max(1)); +} + +/// Stop the timer: no interrupt until it is armed again. +pub fn stop_timer() { + percpu::set_armed_ticks(0); + stop_timer_hardware(); + crate::trace::trace(crate::trace::Kind::TimerStop, 0); +} + +/// A timer interrupt taken: armed again for what it was last armed for, or +/// stopped if it was stopped — the one thing that deasserts it. +pub(super) fn rearm() { + match percpu::armed_ticks() { + 0 => stop_timer_hardware(), + ticks => arm_ticks(ticks), + } +} + +/// How many counter ticks past its comparator the timer is being taken. +#[cfg(feature = "boot-actuators")] +pub(super) fn lateness() -> u64 { + let cval: u64; + // SAFETY: reads the EL1 virtual timer's comparator. + unsafe { core::arch::asm!("mrs {}, cntv_cval_el0", out(reg) cval, options(nomem, nostack, preserves_flags)) }; + cpu::counter().saturating_sub(cval) +} diff --git a/kernel/src/arch/aarch64/mod.rs b/kernel/src/arch/aarch64/mod.rs index 1ac0e061e9..95404aa52f 100644 --- a/kernel/src/arch/aarch64/mod.rs +++ b/kernel/src/arch/aarch64/mod.rs @@ -3,14 +3,13 @@ //! //! Every `unsafe` block here carries a one-line `SAFETY:` comment, enforced by the lint above. //! -//! **What exists and what is owed.** The boot reaches its console: the entry -//! drops from EL2, applies the control-register declaration, turns on the -//! loader's tables and installs the exception vectors; the PL011 is found -//! through SPCR. Everything the kernel does after the console — the interrupt -//! controller, the timer, its own page tables, other CPUs, user mode — is -//! owed by a stage of the port (`issues/kernel/toyos-runs-on-arm64.md`), and -//! each item that stands for it here is an [`owed!`] that panics naming it. A kernel that reaches -//! one stops loudly on its panel; none of them returns a guess. +//! **What exists and what is owed.** One CPU runs the kernel and user mode: +//! its own page tables, the GICv3 and the generic timer, preemption, and the +//! entry from EL0. Other CPUs, the IOMMU, an MSI and the platform's own +//! devices are owed by a stage of the port (`issues/kernel/toyos-runs-on-arm64.md`) +//! or by none yet, and each item that stands for one here is an [`owed!`] +//! that panics naming it, or a refusal a caller reports by name. None of them +//! returns a guess. /// Stands for work the port owes: panics naming what and which stage of the /// track owns it (`issues/kernel/toyos-runs-on-arm64.md`), or that none does yet. @@ -39,6 +38,7 @@ pub mod pio; pub mod pmu; pub mod rtc; pub mod smp; +pub mod switch; pub mod syscall; pub mod tlb; pub mod trap; @@ -48,10 +48,13 @@ pub mod watchdog; pub const ELF_MACHINE: toyos_elf::Machine = toyos_elf::Machine::Aarch64; /// A message-signalled interrupt's address and data for `vector` on CPU -/// `dest`. On AArch64 the doorbell is an ITS's `GITS_TRANSLATER`, one per ITS -/// the MADT names, and the data is an event the ITS maps: stage 4 builds both. -pub fn msi_message(_dest: u32, _vector: u8) -> (u32, u32) { - owed!("an MSI doorbell (the GICv3 ITS)", "stage 4") +/// `dest`, which this machine does not give: the doorbell is an ITS's +/// `GITS_TRANSLATER` and the data an event the ITS maps, and nothing here +/// drives an ITS. Every function that would take one is a driver the +/// small-kernel track moves out of the kernel, or a claimed function the +/// SMMUv3 of the port's stage 6 must translate first; each is refused by name. +pub fn msi_message(_dest: u32, _vector: u8) -> Result<(u32, u32), &'static str> { + Err("AArch64 delivers no message-signalled interrupt to this kernel: the GICv3 ITS is unported") } /// Interrupts masked on this CPU for as long as the guard lives, and then put diff --git a/kernel/src/arch/aarch64/paging.rs b/kernel/src/arch/aarch64/paging.rs index 5f532fa241..db065d75b2 100644 --- a/kernel/src/arch/aarch64/paging.rs +++ b/kernel/src/arch/aarch64/paging.rs @@ -1,154 +1,687 @@ -//! Page tables. The boot runs on the loader's: one L0 table that `TTBR0_EL1` -//! walks for the identity view and `TTBR1_EL1` for the view at `PHYS_OFFSET`, -//! 2 MiB blocks whose `AttrIndx` names [`super::control_regs`]'s `MAIR_EL1`. -//! The kernel's own tables — the direct map, MMIO windows, and a user -//! address space per process with an ASID — are the port's stage 4, so an -//! [`AddressSpace`] cannot exist yet: the type is uninhabited, and every -//! method on it is a match on nothing. - -use core::convert::Infallible; - -use crate::mm::UserAddr; +//! Page tables and address spaces: VMSAv8-64 stage 1 with a 4 KiB granule +//! (Arm ARM K.a, chapter D8), and the only code that writes a translation +//! table entry. +//! +//! **Two tables, not one root with two halves.** `TTBR1_EL1` walks the +//! kernel's tables — the direct map at `PHYS_OFFSET`, which every address +//! space shares because every one runs under the same `TTBR1_EL1` — and +//! `TTBR0_EL1` walks one [`AddressSpace`]'s user half, tagged with the ASID it +//! owns. A user space copies nothing of the kernel's, and switching to one is +//! one register write. +//! +//! **The direct map holds memory and nothing else**: every 4 KiB page +//! firmware's map calls memory the kernel reads (`toyos_bootmap::aarch64`), +//! Normal write-back; a device's registers only as [`map_mmio`] maps them; and +//! the scanout. A page is never retyped: a mapping that disagrees with the one +//! already there is refused by name, because one page under two memory types +//! loses coherency (D8.2.12). +//! +//! **A live entry is replaced break-before-make**: written invalid, its +//! translation dropped on every CPU (`super::tlb`), and only then written +//! again (D8.14.1). An entry that was invalid owes nothing, since no TLB holds +//! a translation that faulted. + +use alloc::boxed::Box; +use alloc::vec::Vec; + +use toyos_bootmap::aarch64::{coverage, direct_map_end, Coverage, ATTR_DEVICE, ATTR_NORMAL, ATTR_NORMAL_NC}; +use toyos_pcid::{Alloc, Pcid, PcidPool}; + +use super::tlb; use crate::mm::policy::{CachePolicy, MmioPolicy, Prot, WindowProt}; -use crate::sync::Lock; -use crate::vma::{Occupancy, Region, RegionKind}; +use crate::mm::{DirectMap, UserAddr, PAGE_2M, PHYS_OFFSET}; +use crate::sync::{Lock, LockGuard}; +use crate::vma::{self, Occupancy, Region, RegionKind}; use crate::MemoryMapEntry; -/// A translation table base: what `TTBR0_EL1` is loaded with for a space. +const VALID: u64 = 1 << 0; +/// At levels 0 to 2 a descriptor naming the next table, at level 3 a page; +/// clear at level 2 is a 2 MiB block. +const TABLE: u64 = 1 << 1; +const ATTR_INDEX: u64 = 0b111 << 2; +/// `AP[1]`: EL0 may access. +const AP_EL0: u64 = 1 << 6; +/// `AP[2]`: read-only at every level that may access. +const AP_READ_ONLY: u64 = 1 << 7; +const INNER_SHAREABLE: u64 = 0b11 << 8; +const OUTER_SHAREABLE: u64 = 0b10 << 8; +/// The access flag: set, so no first access faults. +const AF: u64 = 1 << 10; +/// Not global: the translation is the owning ASID's alone. +const NOT_GLOBAL: u64 = 1 << 11; +const PXN: u64 = 1 << 53; +const UXN: u64 = 1 << 54; +const ADDR: u64 = 0x0000_FFFF_FFFF_F000; +const ADDR_2M: u64 = 0x0000_FFFF_FFE0_0000; +const PAGE_4K: u64 = crate::mm::PAGE_SIZE; + +/// A leaf's memory type and shareability. A device is never executable, +/// since a speculative fetch from one is a read of its registers. +fn typed(cache: CachePolicy) -> u64 { + match cache { + CachePolicy::Normal => ATTR_NORMAL << 2 | INNER_SHAREABLE, + CachePolicy::Uncacheable => ATTR_DEVICE << 2 | PXN | UXN, + CachePolicy::WriteCombining => ATTR_NORMAL_NC << 2 | OUTER_SHAREABLE, + } +} + +/// The type a leaf this file wrote names; any other index is one it never wrote. +fn policy_of(leaf: u64) -> CachePolicy { + match (leaf & ATTR_INDEX) >> 2 { + ATTR_NORMAL => CachePolicy::Normal, + ATTR_DEVICE => CachePolicy::Uncacheable, + ATTR_NORMAL_NC => CachePolicy::WriteCombining, + index => panic!("paging: the leaf {leaf:#x} names MAIR_EL1 index {index}, which this kernel never writes"), + } +} + +impl Prot { + /// A user leaf's permissions: EL0 reads every one, and EL1 executes none. + fn user_bits(self) -> u64 { + match self { + Self::Read => AP_EL0 | AP_READ_ONLY | UXN | PXN, + Self::ReadWrite => AP_EL0 | UXN | PXN, + Self::ReadExec => AP_EL0 | AP_READ_ONLY | PXN, + } + } +} + +/// A user leaf, bar its address and whether it is a block or a page. +fn user_leaf(prot: Prot, cache: CachePolicy) -> u64 { + VALID | AF | NOT_GLOBAL | typed(cache) | prot.user_bits() +} + +/// A kernel leaf: EL1 read-write, EL0 nothing, global, and executable at EL1 +/// only where it is memory — the kernel runs from the direct map. +fn kernel_leaf(cache: CachePolicy) -> u64 { + VALID | AF | typed(cache) | UXN +} + +/// A leaf with its address and its block-or-page bit taken out: what two +/// mappings of one page must agree on. +fn attributes(leaf: u64) -> u64 { + leaf & !ADDR & !TABLE +} + +/// The index `va` takes at `level` (0 to 3) of a 4 KiB-granule walk. +fn index(va: u64, level: u32) -> usize { + ((va >> (39 - 9 * level)) & 0x1FF) as usize +} + +/// One translation table: 512 descriptors in one aligned page. +#[repr(C, align(4096))] +struct Table([u64; 512]); + +impl Table { + fn new() -> Box { + Box::new(Self([0; 512])) + } + + fn phys(&self) -> u64 { + DirectMap::phys_of(self) + } + + /// # Safety + /// `phys` is a table this file built and linked, alive while the reference is. + unsafe fn at<'a>(phys: u64) -> &'a Table { + // SAFETY: the caller's contract. + unsafe { &*DirectMap::from_phys(phys).as_ptr::() } + } + + /// # Safety + /// As [`Table::at`], and the reference is the only live one. + unsafe fn at_mut<'a>(phys: u64) -> &'a mut Table { + // SAFETY: the caller's contract. + unsafe { &mut *DirectMap::from_phys(phys).as_mut_ptr::
() } + } + + /// The next table down from a valid table descriptor at `i`. + fn child(&self, i: usize) -> Option<&Table> { + let entry = self.0[i]; + // SAFETY: valid and a table, so it names one this file linked. + (entry & (VALID | TABLE) == VALID | TABLE).then(|| unsafe { Table::at(entry & ADDR) }) + } + + /// Write the entry at `i`, which the hardware may be walking: one aligned + /// store, so no walker sees half a descriptor. + fn set(&mut self, i: usize, value: u64) { + // SAFETY: `i` indexes this table, a live `&mut`. + unsafe { core::ptr::write_volatile(&raw mut self.0[i], value) }; + } +} + +/// Every table written reached every walker, and this CPU walks afresh. +fn published() { + // SAFETY: barriers; they touch no memory. + unsafe { core::arch::asm!("dsb ishst", "isb", options(nostack, preserves_flags)) }; +} + +/// A root and every table under it, owned together and freed together. +struct Tables { + root: Box
, + children: Vec>, +} + +impl Tables { + fn new() -> Self { + Self { root: Table::new(), children: Vec::new() } + } + + fn adopt(&mut self, table: Box
) -> u64 { + let phys = table.phys(); + self.children.push(table); + phys + } + + /// The level-2 table over `va`, built down to it where absent. Only an + /// invalid entry is written, which owes no invalidation. + fn directory(&mut self, va: u64) -> &mut Table { + let mut table: *mut Table = &mut *self.root; + for level in 0..2 { + // SAFETY: `table` is this root or a table it linked, and + // `&mut self` makes this the only reference. + let current = unsafe { &mut *table }; + let i = index(va, level); + if current.0[i] & VALID == 0 { + let phys = self.adopt(Table::new()); + current.set(i, phys | TABLE | VALID); + } + let entry = current.0[i]; + assert!(entry & TABLE != 0, "paging: a block at level {level} over {va:#x}, which this kernel never writes"); + // SAFETY: valid and a table: one this root linked. + table = unsafe { Table::at_mut(entry & ADDR) }; + } + // SAFETY: as in the loop. + unsafe { &mut *table } + } + + /// The level-2 table over `va`, if the walk reaches one. + fn find_directory(&self, va: u64) -> Option<&Table> { + self.root.child(index(va, 0))?.child(index(va, 1)) + } + + /// The leaf that maps `va` and the size it maps, if one does. + fn leaf(&self, va: u64) -> Option<(u64, u64)> { + let directory = self.find_directory(va)?; + let entry = directory.0[index(va, 2)]; + if entry & VALID == 0 { + return None; + } + if entry & TABLE == 0 { + return Some((entry, PAGE_2M)); + } + let page = directory.child(index(va, 2))?.0[index(va, 3)]; + (page & VALID != 0).then_some((page, PAGE_4K)) + } + + /// Map `[phys, phys + size)` at `PHYS_OFFSET + phys` with `leaf`'s + /// attributes: a whole 2 MiB page as a block where nothing maps it yet, + /// the rest page by page. A page already mapped the same way is left as + /// it is; one mapped any other way is refused. + fn map_direct(&mut self, phys: u64, size: u64, leaf: u64) { + let (start, end) = (phys & !(PAGE_4K - 1), (phys + size).next_multiple_of(PAGE_4K)); + let mut at = start; + while at < end { + let va = PHYS_OFFSET + at; + let block = at & !(PAGE_2M - 1); + let i = index(va, 2); + let entry = self.directory(va).0[i]; + if entry & (VALID | TABLE) == VALID { + refuse_unless_same(at, entry, leaf); + at = block + PAGE_2M; + continue; + } + if entry & VALID == 0 && at == block && block + PAGE_2M <= end { + self.directory(va).set(i, block | leaf); + at += PAGE_2M; + continue; + } + let pages = if entry & VALID == 0 { + let phys = self.adopt(Table::new()); + self.directory(va).set(i, phys | TABLE | VALID); + phys + } else { + entry & ADDR + }; + // SAFETY: a table this root linked, reached under `&mut self`. + let pages = unsafe { Table::at_mut(pages) }; + let j = index(va, 3); + if pages.0[j] & VALID != 0 { + refuse_unless_same(at, pages.0[j], leaf); + } else { + pages.set(j, at | leaf | TABLE); + } + at += PAGE_4K; + } + published(); + } +} + +/// A second mapping of the page at `phys` agrees with the first, or the +/// kernel stops: [`Tables::map_direct`]'s refusal. +fn refuse_unless_same(phys: u64, existing: u64, wanted: u64) { + assert!( + attributes(existing) == attributes(wanted), + "paging: {phys:#x} is mapped {:?} ({existing:#x}) and cannot also be {:?}", + policy_of(existing), + policy_of(wanted), + ); +} + +/// What `TTBR0_EL1` is loaded with for a space: its ASID in bits 63:48 and +/// its root table's address. #[derive(Clone, Copy)] -pub struct Root(Infallible); +pub struct Root(u64); impl Root { pub fn phys(self) -> u64 { - match self.0 {} + self.0 & ADDR } + /// No invalidation: the ASID is this space's alone, and a returned one was + /// dropped from every CPU before it was issued again (`toyos_pcid`). /// # Safety /// The underlying page tables must be valid and live. pub unsafe fn activate(self) { - match self.0 {} + // SAFETY: the caller's contract; the `ISB` makes the next walk use it. + unsafe { core::arch::asm!("msr ttbr0_el1, {}", "isb", in(reg) self.0, options(nostack, preserves_flags)) }; + } +} + +/// The ASID allocator: `toyos_pcid`'s tags, which 16-bit ASIDs hold whole, +/// and its reclaim given the shootdown it asks for. +static ASIDS: Lock = Lock::new(PcidPool::new()); + +/// A user ASID owned for one space's life; its drop returns it. +struct AsidGuard(Pcid); + +impl Drop for AsidGuard { + fn drop(&mut self) { + ASIDS.lock().free(self.0); + } +} + +enum Asid { + /// ASID 0, the kernel space's, whose user half is empty. + Kernel, + User(AsidGuard), +} + +impl Asid { + fn value(&self) -> u16 { + match self { + Self::Kernel => toyos_pcid::KERNEL_PCID, + Self::User(guard) => guard.0.get(), + } } } -/// A process's address space: its regions and the tables that map them. +/// `None` when every user ASID is held by a live space. +fn alloc_asid() -> Option { + let mut pool = ASIDS.lock(); + loop { + match pool.alloc() { + Alloc::Ready(tag) => return Some(AsidGuard(tag)), + Alloc::NeedsFlush => { + tlb::shootdown(crate::invalidation::Origin::Pcid); + pool.reclaim(); + } + Alloc::Exhausted => return None, + } + } +} + +/// A process's user half, walked through `TTBR0_EL1`: its tables, its +/// regions and its ASID. pub struct AddressSpace { - never: Infallible, + tables: Tables, + regions: vma::Regions, + asid: Asid, } impl AddressSpace { + /// An empty user half, or `None` when every user ASID is held. pub fn new_user() -> Option { - owed!("a user address space", "stage 4") + let asid = alloc_asid()?; + Some(Self { tables: Tables::new(), regions: vma::Regions::default(), asid: Asid::User(asid) }) } pub fn root(&self) -> Root { - match self.never {} + Root(u64::from(self.asid.value()) << 48 | self.tables.root.phys()) + } + + /// Replace the level-2 entry over `va`, break-before-make where it was + /// valid; a table it named stays owned by this space until it drops. + fn replace(&mut self, va: u64, value: u64) { + let asid = self.asid.value(); + let directory = self.tables.directory(va); + let i = index(va, 2); + let prior = directory.0[i]; + if prior & VALID != 0 { + directory.set(i, 0); + if prior & TABLE != 0 { + tlb::asid(asid); + } else { + tlb::page(asid, va); + } + } + directory.set(i, value); + published(); + } + + /// Empty level-2 entries (aligned, asserted) mean nothing can be stale. + pub fn map_range(&mut self, vaddr: UserAddr, phys: u64, size: u64, prot: Prot, cache: CachePolicy) { + assert!(vaddr.raw() & (PAGE_2M - 1) == 0, "map_range: vaddr not 2MB-aligned"); + assert!(phys & (PAGE_2M - 1) == 0, "map_range: phys {phys:#x} not 2MB-aligned"); + if prot == Prot::ReadExec { + super::cache::make_executable(DirectMap::from_phys(phys).as_ptr::() as u64, size as usize); + } + let mut offset = 0u64; + while offset < size { + let va = vaddr.raw() + offset; + let directory = self.tables.directory(va); + let i = index(va, 2); + assert!(directory.0[i] & VALID == 0, "map_range: an install at {va:#x} found the present entry {:#x}", directory.0[i]); + directory.set(i, (phys + offset) | user_leaf(prot, cache)); + offset += PAGE_2M; + } + published(); } - pub fn map_range(&mut self, _vaddr: UserAddr, _phys: u64, _size: u64, _prot: Prot, _cache: CachePolicy) { - match self.never {} + fn unmap_range(&mut self, vaddr: UserAddr, size: u64) { + let mut offset = 0u64; + while offset < size { + self.unmap(UserAddr::new(vaddr.raw() + offset)); + offset += PAGE_2M; + } } - pub fn remap(&mut self, _vaddr: UserAddr, _phys: u64, _prot: Prot) { - match self.never {} + /// Replaces whatever maps `vaddr`, in this space only. + pub fn remap(&mut self, vaddr: UserAddr, phys: u64, prot: Prot) { + let va = vaddr.raw(); + assert!(va & (PAGE_2M - 1) == 0, "remap: vaddr {va:#x} not 2MB-aligned"); + assert!(phys & (PAGE_2M - 1) == 0, "remap: phys {phys:#x} not 2MB-aligned"); + if prot == Prot::ReadExec { + super::cache::make_executable(DirectMap::from_phys(phys).as_ptr::() as u64, PAGE_2M as usize); + } + self.replace(va, phys | user_leaf(prot, CachePolicy::Normal)); } - pub fn map_window(&mut self, _vaddr: UserAddr, _phys: u64, _prot: &WindowProt) { - match self.never {} + /// A mixed window's table is filled before it is linked, so nothing walks + /// it half-written. Must not be called twice on one address. + pub fn map_window(&mut self, vaddr: UserAddr, phys: u64, prot: &WindowProt) { + if let Some(uniform) = prot.agreed() { + self.remap(vaddr, phys, uniform); + return; + } + let va = vaddr.raw(); + assert!(va & (PAGE_2M - 1) == 0, "map_window: vaddr {va:#x} not 2MB-aligned"); + assert!(phys & (PAGE_2M - 1) == 0, "map_window: phys {phys:#x} not 2MB-aligned"); + let mut table = Table::new(); + for (i, page_prot) in prot.pages().enumerate() { + let at = phys + i as u64 * PAGE_4K; + if page_prot == Prot::ReadExec { + super::cache::make_executable(DirectMap::from_phys(at).as_ptr::() as u64, PAGE_4K as usize); + } + table.0[i] = at | user_leaf(page_prot, CachePolicy::Normal) | TABLE; + } + let table = self.tables.adopt(table); + self.replace(va, table | TABLE | VALID); } - pub fn map_window_if_absent(&mut self, _vaddr: UserAddr, _phys: u64, _prot: &WindowProt) -> bool { - match self.never {} + /// `false` leaves the mapping as found and `phys` the caller's to free: + /// the check and the write are one critical section under the caller's lock. + pub fn map_window_if_absent(&mut self, vaddr: UserAddr, phys: u64, prot: &WindowProt) -> bool { + if self.translate(vaddr).is_some() { + return false; + } + self.map_window(vaddr, phys, prot); + true } - pub fn unmap(&mut self, _vaddr: UserAddr) { - match self.never {} + /// Ends any futex wait on the frame (its token is a physical address) + /// before the frame can reach the PMM and be reissued under a waiter. + pub fn unmap(&mut self, vaddr: UserAddr) { + let va = vaddr.raw(); + assert!(va & (PAGE_2M - 1) == 0, "unmap: vaddr {va:#x} not 2MB-aligned"); + let asid = self.asid.value(); + let Some(directory) = self.tables.find_directory(va) else { return }; + let i = index(va, 2); + let entry = directory.0[i]; + if entry & VALID == 0 { + return; + } + // A split window's entries all address one 2 MiB frame, so its first names it. + let phys = match directory.child(i) { + Some(pages) => pages.0[0] & ADDR_2M, + None => entry & ADDR_2M, + }; + self.tables.directory(va).set(i, 0); + if entry & TABLE != 0 { + tlb::asid(asid); + } else { + tlb::page(asid, va); + } + crate::sched::futex::revoke_range(phys, PAGE_2M); } - pub fn translate(&self, _vaddr: UserAddr) -> Option { - match self.never {} + /// Checked here, not at the callers: only a user address names user memory. + pub fn translate(&self, vaddr: UserAddr) -> Option { + self.walk(vaddr).map(|(at, _)| at) } - pub fn translate_writable(&self, _vaddr: UserAddr) -> Option { - match self.never {} + /// As [`translate`](Self::translate), but only where an EL0 store would + /// land: a leaf EL0 may write. A kernel copy into user memory goes + /// through this, so a syscall cannot write a page the process itself may + /// not — the clock page, a shared library's `.text`. + pub fn translate_writable(&self, vaddr: UserAddr) -> Option { + self.walk(vaddr).and_then(|(at, leaf)| (leaf & (AP_EL0 | AP_READ_ONLY) == AP_EL0).then_some(at)) } - pub fn alloc_region(&mut self, _size: u64, _kind: RegionKind) -> Option { - match self.never {} + fn walk(&self, vaddr: UserAddr) -> Option<(DirectMap, u64)> { + let va = vaddr.raw(); + if !toyos_userbound::is_user_addr(va) { + return None; + } + let (leaf, size) = self.tables.leaf(va)?; + let base = leaf & if size == PAGE_2M { ADDR_2M } else { ADDR }; + Some((DirectMap::from_phys(base + (va & (size - 1))), leaf)) } - pub fn alloc_and_map( - &mut self, - _phys: u64, - _size: u64, - _prot: Prot, - _cache: CachePolicy, - ) -> Option<(UserAddr, u64)> { - match self.never {} + pub fn alloc_region(&mut self, size: u64, kind: RegionKind) -> Option { + self.regions.alloc(size, kind) } - pub fn free_and_unmap(&mut self, _addr: UserAddr) -> Option { - match self.never {} + pub fn alloc_and_map(&mut self, phys: u64, size: u64, prot: Prot, cache: CachePolicy) -> Option<(UserAddr, u64)> { + assert!(phys & (PAGE_2M - 1) == 0, "alloc_and_map: phys {phys:#x} not 2MB-aligned"); + let (addr, aligned) = self.regions.alloc_mapped(size)?; + self.map_range(addr, phys, aligned, prot, cache); + Some((addr, aligned)) } - pub fn insert_region(&mut self, _addr: UserAddr, _region: Region) { - match self.never {} + pub fn free_and_unmap(&mut self, addr: UserAddr) -> Option { + let size = self.regions.remove(addr)?; + self.unmap_range(addr, size); + Some(size) } - pub fn find_region(&self, _addr: UserAddr) -> Option<(UserAddr, &Region)> { - match self.never {} + pub fn insert_region(&mut self, addr: UserAddr, region: Region) { + self.regions.insert(addr, region); } - pub fn occupancy(&self, _addr: UserAddr, _size: u64) -> Occupancy { - match self.never {} + pub fn find_region(&self, addr: UserAddr) -> Option<(UserAddr, &Region)> { + self.regions.find(addr) } - pub fn overlapping_regions( - &self, - _start: UserAddr, - _end: UserAddr, - ) -> impl Iterator { - match self.never {} - #[allow(unreachable_code)] - core::iter::empty() + pub fn occupancy(&self, addr: UserAddr, size: u64) -> Occupancy { + self.regions.occupancy(addr, size) } - pub fn direct_map_policy(&self, _phys: u64) -> Option { - match self.never {} + pub fn overlapping_regions(&self, start: UserAddr, end: UserAddr) -> impl Iterator { + self.regions.overlapping(start, end) } - pub fn user_policy(&self, _addr: UserAddr) -> Option { - match self.never {} + /// The direct map's type for `phys`, read off the kernel's tables, which + /// every space shares. + pub fn direct_map_policy(&self, phys: u64) -> Option { + high().as_ref()?.leaf(DirectMap::from_phys(phys).as_ptr::() as u64).map(|(leaf, _)| policy_of(leaf)) } - pub fn guard_4k(&mut self, _phys: u64) { - match self.never {} + pub fn user_policy(&self, addr: UserAddr) -> Option { + self.tables.leaf(addr.raw()).map(|(leaf, _)| policy_of(leaf)) } + + /// Take the 4 KiB page at `phys` out of the direct map, for good: the + /// caller owns it for the machine's life. The 2 MiB page around it is + /// split break-before-make, so it must be one nothing is running from. + pub fn guard_4k(&mut self, phys: u64) { + assert!(phys & (PAGE_4K - 1) == 0, "guard_4k: phys {phys:#x} not 4 KiB-aligned"); + let va = DirectMap::from_phys(phys).as_ptr::() as u64; + let mut high = high(); + let high = high.as_mut().expect("guard_4k: before the kernel's tables exist"); + let i = index(va, 2); + let entry = high.directory(va).0[i]; + assert!(entry & VALID != 0, "guard_4k: {phys:#x} is not in the direct map"); + if entry & TABLE == 0 { + let base = entry & ADDR_2M; + let mut pages = Table::new(); + for (j, page) in pages.0.iter_mut().enumerate() { + *page = (base + j as u64 * PAGE_4K) | attributes(entry) | TABLE; + } + let pages = high.adopt(pages); + let directory = high.directory(va); + directory.set(i, 0); + tlb::kernel_page(va); + directory.set(i, pages | TABLE | VALID); + } + let pages = high.directory(va).0[i] & ADDR; + // SAFETY: the table the kernel's root linked, just above or before. + let pages = unsafe { Table::at_mut(pages) }; + let j = index(va, 3); + assert!(pages.0[j] & VALID != 0, "guard_4k: {phys:#x} is already unmapped"); + pages.set(j, 0); + tlb::kernel_page(va); + } +} + +/// The kernel's own tables, `TTBR1_EL1`'s. +static HIGH: Lock> = Lock::new(None); + +fn high() -> LockGuard<'static, Option> { + HIGH.lock() } +/// The kernel's address space: the empty user half a kernel thread and an +/// idle CPU run under, ASID 0. Leaked, since it outlives every task. +static KERNEL: core::sync::atomic::AtomicPtr>> = + core::sync::atomic::AtomicPtr::new(core::ptr::null_mut()); + +/// Its root, cached for lock-free access from panic and crash paths. +static KERNEL_TTBR0: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0); + pub fn kernel() -> &'static alloc::sync::Arc> { - owed!("the kernel's page tables", "stage 4") + let ptr = KERNEL.load(core::sync::atomic::Ordering::Acquire); + assert!(!ptr.is_null(), "paging not initialized"); + // SAFETY: written once in `init` with the `Release` this pairs with, never + // cleared, so the pointer is live for the machine's life. + unsafe { &*ptr } } pub fn kernel_root() -> Root { - owed!("the kernel's page tables", "stage 4") + Root(KERNEL_TTBR0.load(core::sync::atomic::Ordering::Relaxed)) } +/// Leave whatever user half is current for the kernel's empty one. pub fn activate_kernel() { - owed!("the kernel's page tables", "stage 4") + // SAFETY: the kernel space's root is an empty table that lives forever. + unsafe { kernel_root().activate() }; } -pub fn map_mmio(_phys: u64, _size: u64, _policy: MmioPolicy) -> crate::mm::Mmio { - owed!("the kernel's page tables", "stage 4") +/// Map a device's registers, or the scanout, into the direct map: 4 KiB pages +/// exactly over `[phys, phys + size)`, a block where a whole 2 MiB page is +/// asked for and free. No invalidation is owed, since only an invalid entry is +/// ever written. +pub fn map_mmio(phys: u64, size: u64, policy: MmioPolicy) -> crate::mm::Mmio { + high().as_mut().expect("map_mmio: before the kernel's tables exist").map_direct(phys, size, kernel_leaf(policy.cache())); + let installed = kernel().lock().direct_map_policy(phys).expect("map_mmio: the window was just mapped"); + assert!(installed == policy.cache(), "map_mmio: {phys:#x} installed {installed:?}"); + crate::log!("mmio: {phys:#x}+{size:#x} {installed:?}"); + crate::mm::Mmio::new(DirectMap::from_phys(phys), size) } -pub(crate) fn init(_memory_map: &[MemoryMapEntry]) -> toyos_bootmap::DirectMapEnd { - owed!("the kernel's page tables", "stage 4") -} +/// Build the kernel's tables — every page of memory firmware's map names, +/// and the console UART and the scanout the boot's own records reach — +/// switch `TTBR1_EL1` to them and `TTBR0_EL1` to the kernel's empty user +/// half, and answer where the direct map ends. +pub(crate) fn init(memory_map: &[MemoryMapEntry], scanout: Option<(u64, u64)>) -> toyos_bootmap::DirectMapEnd { + let extent = direct_map_end(memory_map).unwrap_or_else(|refusal| panic!("paging: firmware's memory map: {refusal}")); + let end = extent.get(); + let memory = kernel_leaf(CachePolicy::Normal); + let mut tables = Tables::new(); + let (mut blocks, mut pages) = (0u64, 0u64); + let mut at = 0; + while at < end { + let va = PHYS_OFFSET + at; + match coverage(memory_map, at) { + Coverage::Nothing => {} + Coverage::Whole => { + tables.directory(va).set(index(va, 2), at | memory); + blocks += 1; + } + held @ Coverage::Pages(_) => { + let mut table = Table::new(); + for j in (0..512).filter(|&j| held.holds(j)) { + table.0[j as usize] = (at + j * PAGE_4K) | memory | TABLE; + pages += 1; + } + let table = tables.adopt(table); + tables.directory(va).set(index(va, 2), table | TABLE | VALID); + } + } + at += PAGE_2M; + } + if let Some(uart) = super::console_uart::frame() { + tables.map_direct(uart, PAGE_4K, kernel_leaf(CachePolicy::Uncacheable)); + } + if let Some((scanout, size)) = scanout { + tables.map_direct(scanout, size, kernel_leaf(CachePolicy::WriteCombining)); + } -pub(crate) fn seal_kernel_half() { - owed!("the kernel's page tables", "stage 4") + let ttbr1 = tables.root.phys(); + *high() = Some(tables); + let space = AddressSpace { tables: Tables::new(), regions: vma::Regions::default(), asid: Asid::Kernel }; + let ttbr0 = space.root(); + KERNEL_TTBR0.store(ttbr0.0, core::sync::atomic::Ordering::Release); + let published: &'static alloc::sync::Arc> = + Box::leak(Box::new(alloc::sync::Arc::new(Lock::new(space)))); + KERNEL.store(published as *const _ as *mut _, core::sync::atomic::Ordering::Release); + + // SAFETY: both tables are built and published above, and the new + // `TTBR1_EL1` maps the code, stack and data this runs on exactly as the + // loader's did; the loader's identity view goes with the old `TTBR0_EL1`, + // and the local `TLBI` drops every translation either left behind. + unsafe { + core::arch::asm!( + "dsb ishst", + "msr ttbr1_el1, {ttbr1}", + "msr ttbr0_el1, {ttbr0}", + "isb", + "tlbi vmalle1", + "dsb nsh", + "isb", + ttbr1 = in(reg) ttbr1, + ttbr0 = in(reg) ttbr0.0, + options(nostack, preserves_flags), + ); + } + crate::log!("paging: the direct map holds memory below {end:#x} in {blocks} 2 MiB blocks and {pages} 4 KiB pages"); + extent } +/// Nothing to seal: every space runs under the one `TTBR1_EL1`, so no space +/// holds a copy of the kernel's root slots to fall behind. +pub(crate) fn seal_kernel_half() {} + /// `PAR_EL1` after the MMU translated `addr` for an EL1 read in the tables /// this CPU runs on now (`AT S1E1R`; Arm ARM K.a, C6.2.x and D24.2.131): /// bit 0 set is a fault, and otherwise bits 63:56 are the memory type. @@ -181,7 +714,7 @@ pub fn present_in_current_tables(addr: u64) -> bool { /// loader maps it so. Asked of the MMU page by page rather than assumed, and /// nothing is retyped: the type the loader wrote is the final one. pub fn boot_map_write_combining(phys: u64, size: u64) -> bool { - [phys, crate::mm::PHYS_OFFSET + phys].iter().all(|&base| { + [phys, PHYS_OFFSET + phys].iter().all(|&base| { (0..size.div_ceil(4096)).all(|page| { let par = translate_read(base + page * 4096); // Outer Normal non-cacheable, and an inner nibble that says the same: @@ -194,9 +727,99 @@ pub fn boot_map_write_combining(phys: u64, size: u64) -> bool { }) } -/// What memory type the scanout is mapped with, for the GPU's report. -pub fn scanout_memory_type(_addr: u64, _size: u64) -> impl core::fmt::Display { - owed!("the kernel's page tables", "stage 4"); - #[allow(unreachable_code)] - "" +/// What memory type the scanout is mapped with, for the GPU's report: the +/// MMU's own answer for its first byte in the direct map. +pub fn scanout_memory_type(addr: u64, _size: u64) -> impl core::fmt::Display { + struct Report(u64); + impl core::fmt::Display for Report { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self.0 & 1 { + 0 => write!(f, "PAR_EL1.ATTR {:#04x}", self.0 >> 56), + _ => write!(f, "untranslated (PAR_EL1 {:#x})", self.0), + } + } + } + Report(translate_read(PHYS_OFFSET + addr)) +} + +/// Each level's descriptor for `addr` in the tables this CPU runs under now, +/// read without a lock for a crash report. +pub fn debug_page_walk(addr: u64) { + let (name, ttbr) = if addr >> 63 == 0 { + let ttbr: u64; + // SAFETY: reads `TTBR0_EL1`. + unsafe { core::arch::asm!("mrs {}, ttbr0_el1", out(reg) ttbr, options(nomem, nostack, preserves_flags)) }; + ("TTBR0_EL1", ttbr) + } else { + let ttbr: u64; + // SAFETY: reads `TTBR1_EL1`. + unsafe { core::arch::asm!("mrs {}, ttbr1_el1", out(reg) ttbr, options(nomem, nostack, preserves_flags)) }; + ("TTBR1_EL1", ttbr) + }; + crate::log!(" Page walk for {addr:#x} through {name}={ttbr:#x}:"); + let mut table = ttbr & ADDR; + for level in 0..4 { + // SAFETY: a table the live translation register or a valid table + // descriptor above it names, which the MMU is walking now. + let entry = unsafe { Table::at(table) }.0[index(addr, level)]; + crate::log!(" L{level}[{}] = {entry:#018x}", index(addr, level)); + if entry & VALID == 0 || level == 3 || entry & TABLE == 0 { + return; + } + table = entry & ADDR; + } +} + +/// The word at user address `addr` in the space this CPU runs under now, +/// read through the direct map by a walk that takes no lock and faults +/// nowhere: for a crash report, which may hold any lock already. +pub(super) fn read_user_word(addr: u64) -> Option { + if !addr.is_multiple_of(8) || !toyos_userbound::is_user_addr(addr) { + return None; + } + let ttbr0: u64; + // SAFETY: reads `TTBR0_EL1`. + unsafe { core::arch::asm!("mrs {}, ttbr0_el1", out(reg) ttbr0, options(nomem, nostack, preserves_flags)) }; + let mut table = ttbr0 & ADDR; + for level in 0..4 { + // SAFETY: the live `TTBR0_EL1`'s table or one a valid table + // descriptor above names, which the MMU walks now. + let entry = unsafe { Table::at(table) }.0[index(addr, level)]; + if entry & VALID == 0 { + return None; + } + let size = match (level, entry & TABLE != 0) { + (3, _) => PAGE_4K, + (2, false) => PAGE_2M, + (_, true) => { + table = entry & ADDR; + continue; + } + (_, false) => return None, + }; + let base = entry & if size == PAGE_2M { ADDR_2M } else { ADDR }; + // SAFETY: a byte of a frame a valid leaf maps, reached through the direct map. + return Some(unsafe { core::ptr::read_volatile(DirectMap::from_phys(base + (addr & (size - 1))).as_ptr::()) }); + } + None +} + +/// Whether a process can be given the scanout at `[phys, phys + size)`, +/// which it is in 2 MiB pages: its base on one, and nothing the last page +/// covers past it memory the direct map holds — firmware carves a scanout +/// like `ramfb`'s out of RAM, and the rest of that page is somebody else's. +pub fn scanout_whole_pages(phys: u64, size: u64) -> Result<(), &'static str> { + if !phys.is_multiple_of(PAGE_2M) { + return Err("its base is not on a 2 MiB page, and a process is given a scanout in 2 MiB pages"); + } + let high = high(); + let tables = high.as_ref().expect("scanout_whole_pages: before the kernel's tables exist"); + let mut at = (phys + size).next_multiple_of(PAGE_4K); + while at < (phys + size).next_multiple_of(PAGE_2M) { + if tables.leaf(PHYS_OFFSET + at).is_some_and(|(leaf, _)| policy_of(leaf) == CachePolicy::Normal) { + return Err("its last 2 MiB page holds memory past it, which a process given the page would reach"); + } + at += PAGE_4K; + } + Ok(()) } diff --git a/kernel/src/arch/aarch64/percpu.rs b/kernel/src/arch/aarch64/percpu.rs index 065865a077..c1471f4c6d 100644 --- a/kernel/src/arch/aarch64/percpu.rs +++ b/kernel/src/arch/aarch64/percpu.rs @@ -1,8 +1,18 @@ -//! Per-CPU state. On AArch64 it is reached through `TPIDR_EL1`, and the block -//! it names is built by the port's stage 5 (other CPUs) on top of stage 4's -//! exception entry; until then every accessor is owed. The boot never reaches -//! one: the log and the panic path read [`crate::log::PERCPU_READY`] first. +//! Per-CPU state, reached through `TPIDR_EL1`, which holds this CPU's +//! [`PerCpu`] from [`init_bsp`] on and which nothing else writes. EL0 cannot +//! read it, so a thread learns nothing of the kernel's layout from it. +//! +//! Every field another context on the same CPU can write — an interrupt, a +//! nested exception — is an atomic, so the whole block is only ever reached +//! through a shared reference, and an increment is one exclusive-monitor loop +//! an exception on this CPU cannot split (Arm ARM K.a, B2.17.5: taking an +//! exception clears the monitor). +use core::sync::atomic::{AtomicU32, AtomicU64, AtomicU8, Ordering::Relaxed}; + +use alloc::boxed::Box; + +use crate::log; use crate::process::{Pid, Tid}; /// Per-CPU fault state machine for the escalation policy on nested faults. @@ -15,100 +25,300 @@ pub enum CpuFaultState { Panic = 3, } +/// One CPU's block. Only this CPU writes it, bar [`irq_counts`](Self::irq_counts), +/// which `irq_census` reads from any. +pub struct PerCpu { + cpu_id: u32, + /// `u32::MAX` when no thread runs here. + current_tid: AtomicU32, + current_pid: AtomicU32, + preempt_count: AtomicU32, + need_resched: AtomicU8, + fault_state: AtomicU8, + /// Timer interrupts taken at EL1, which only count and ask for a pass. + ring0_timer_fires: AtomicU32, + last_seen_ring0_fires: AtomicU32, + /// Counter ticks the timer was last armed for: what a timer interrupt + /// taken at EL1 re-arms it with, and zero when it is stopped. + armed_ticks: AtomicU64, + /// The kernel stack a switch last installed, for the stack witness: on + /// AArch64 an entry from EL0 lands on `SP_EL1` as the last `ERET` left it, + /// and nothing here points it anywhere. + kernel_stack: AtomicU64, + idle_stack_top: u64, + /// What the last entry by `SVC` said, for a crash report: its return + /// address, number, frame pointer and stack pointer. + syscall_pc: AtomicU64, + syscall_num: AtomicU64, + syscall_fp: AtomicU64, + syscall_sp: AtomicU64, + /// The task whose syscall this CPU is inside: packed pid:tid, or [`NO_SYSCALL`]. + syscall_task: AtomicU64, + /// This CPU's [`log::Shard`]: the boot shard on CPU 0. + log_shard: &'static log::Shard, + /// One counter per `irq_census::Source`, and the total. + irq_counts: [AtomicU64; crate::irq_census::SLOTS], +} + +/// This CPU's block. +#[inline] +fn this() -> &'static PerCpu { + let block: u64; + // SAFETY: reads `TPIDR_EL1`, which [`init_bsp`] points at a leaked + // `PerCpu` before any accessor here runs (`log::PERCPU_READY` gates them). + unsafe { + core::arch::asm!("mrs {}, tpidr_el1", out(reg) block, options(nomem, nostack, preserves_flags)); + &*(block as *const PerCpu) + } +} + +/// Build the boot CPU's block and publish it through `TPIDR_EL1`; after it, +/// every accessor here answers. The thread registers EL0 can read are cleared, +/// so a thread's first read of one finds nothing firmware left. +pub fn init_bsp() { + let block: &'static PerCpu = Box::leak(Box::new(PerCpu { + cpu_id: 0, + current_tid: AtomicU32::new(u32::MAX), + current_pid: AtomicU32::new(u32::MAX), + preempt_count: AtomicU32::new(0), + need_resched: AtomicU8::new(0), + fault_state: AtomicU8::new(CpuFaultState::Normal as u8), + ring0_timer_fires: AtomicU32::new(0), + last_seen_ring0_fires: AtomicU32::new(0), + armed_ticks: AtomicU64::new(0), + kernel_stack: AtomicU64::new(0), + idle_stack_top: crate::sched::idle_stack::alloc(), + syscall_pc: AtomicU64::new(0), + syscall_num: AtomicU64::new(0), + syscall_fp: AtomicU64::new(0), + syscall_sp: AtomicU64::new(0), + syscall_task: AtomicU64::new(NO_SYSCALL), + log_shard: &log::BOOT_SHARD, + irq_counts: [const { AtomicU64::new(0) }; crate::irq_census::SLOTS], + })); + crate::irq_census::publish(0, block.irq_counts.as_ptr()); + // SAFETY: the block lives for the machine's life; `TPIDRRO_EL0` and + // `TPIDR_EL0` are EL0's to read and hold nothing of the kernel's. + unsafe { + core::arch::asm!( + "msr tpidr_el1, {}", + "msr tpidrro_el0, xzr", + "msr tpidr_el0, xzr", + in(reg) block as *const PerCpu as u64, + options(nostack, preserves_flags), + ); + } + crate::log::PERCPU_READY.store(true, core::sync::atomic::Ordering::Release); + log!("percpu: BSP cpu_id=0 mpidr={:#x}", super::cpu::hardware_id()); +} + pub fn cpu_id() -> u32 { - owed!("per-CPU state", "stage 4") + this().cpu_id } +/// `None` means idle. pub fn current_tid() -> Option { - owed!("per-CPU state", "stage 4") + match this().current_tid.load(Relaxed) { + u32::MAX => None, + raw => Some(Tid::from_raw(raw)), + } } -pub fn set_current_tid(_tid: Option) { - owed!("per-CPU state", "stage 4") +pub fn set_current_tid(tid: Option) { + this().current_tid.store(tid.map_or(u32::MAX, |t| t.raw()), Relaxed); } +/// `None` means idle. pub fn current_pid() -> Option { - owed!("per-CPU state", "stage 4") + match this().current_pid.load(Relaxed) { + u32::MAX => None, + raw => Some(Pid::from_raw(raw)), + } } -pub fn set_current_pid(_pid: Option) { - owed!("per-CPU state", "stage 4") +pub fn set_current_pid(pid: Option) { + this().current_pid.store(pid.map_or(u32::MAX, |p| p.raw()), Relaxed); } +/// Record the stack the next entry from EL0 lands on: the one a switch just +/// installed, which `SP_EL1` already is. /// # Safety -/// `top` is the top of the kernel stack the next entry from user mode lands on. -pub unsafe fn set_kernel_stack(_top: u64) { - owed!("per-CPU state", "stage 4") +/// Called on the CPU whose block it is. +pub unsafe fn set_kernel_stack(top: u64) { + this().kernel_stack.store(top, Relaxed); } +/// The stack an entry from EL0 lands on, twice: AArch64 has one where x86-64 +/// has `kernel_rsp` and `tss.rsp0`. /// # Safety /// Read on the CPU whose entry stacks they are. +#[cfg(feature = "stack-witness")] pub unsafe fn entry_stacks() -> (u64, u64) { - owed!("per-CPU state", "stage 4") + let top = this().kernel_stack.load(Relaxed); + (top, top) } pub fn idle_stack_top() -> u64 { - owed!("per-CPU state", "stage 4") + this().idle_stack_top } +/// The last byte of this CPU's idle guard page — the first byte an overflow reaches. #[cfg(feature = "test-actuators")] pub fn idle_guard_byte() -> u64 { - owed!("per-CPU state", "stage 4") + idle_stack_top() - crate::sched::idle_stack::SIZE as u64 - 1 } +/// How big one idle stack is; read by `SYS_DEBUG` for scale. #[cfg(feature = "test-actuators")] pub fn idle_stack_size() -> usize { - owed!("per-CPU state", "stage 4") + crate::sched::idle_stack::SIZE } #[cfg(feature = "test-actuators")] pub fn idle_stack_high_water() -> usize { - owed!("per-CPU state", "stage 4") + crate::sched::idle_stack::high_water() +} + +/// No task on this CPU is inside a syscall; [`pack_task`] never produces this value. +const NO_SYSCALL: u64 = u64::MAX; + +fn pack_task(pid: u32, tid: u32) -> u64 { + (u64::from(pid) << 32) | u64::from(tid) +} + +/// Enter this CPU's syscall bracket, with what the entry saw. `super::trap` is the only caller. +pub(super) fn enter_syscall(pc: u64, num: u64, fp: u64, sp: u64) { + let block = this(); + block.syscall_pc.store(pc, Relaxed); + block.syscall_num.store(num, Relaxed); + block.syscall_fp.store(fp, Relaxed); + block.syscall_sp.store(sp, Relaxed); + block.syscall_task.store(pack_task(block.current_pid.load(Relaxed), block.current_tid.load(Relaxed)), Relaxed); +} + +/// …and leave it. +pub(super) fn leave_syscall() { + this().syscall_task.store(NO_SYSCALL, Relaxed); +} + +/// Whether the task this CPU runs is inside a syscall now. +pub(super) fn in_syscall() -> bool { + let block = this(); + let recorded = block.syscall_task.load(Relaxed); + recorded != NO_SYSCALL + && recorded == pack_task(block.current_pid.load(Relaxed), block.current_tid.load(Relaxed)) +} + +/// The last syscall's return address, frame pointer and stack pointer; +/// meaningful only while [`in_syscall`] holds. +pub(super) fn syscall_context() -> (u64, u64, u64) { + let block = this(); + (block.syscall_pc.load(Relaxed), block.syscall_fp.load(Relaxed), block.syscall_sp.load(Relaxed)) } pub fn syscall_num() -> u64 { - owed!("per-CPU state", "stage 4") + this().syscall_num.load(Relaxed) +} + +/// Not atomic as a pair, and needs not be: only exception and panic entry +/// touch it, with interrupts masked. +pub fn swap_fault_state(new: CpuFaultState) -> CpuFaultState { + match this().fault_state.swap(new as u8, Relaxed) { + 0 => CpuFaultState::Normal, + 1 => CpuFaultState::PageFault, + 2 => CpuFaultState::Fatal, + _ => CpuFaultState::Panic, + } } -pub fn swap_fault_state(_new: CpuFaultState) -> CpuFaultState { - owed!("per-CPU state", "stage 4") +pub fn set_fault_state(new: CpuFaultState) { + this().fault_state.store(new as u8, Relaxed); } /// This CPU's shard, its identity, and one sequence number out of that shard. -pub fn reserve_log_slot( - _guard: &crate::arch::IrqGuard, -) -> (*const crate::log::Shard, u64, u32, u32, u32) { - owed!("per-CPU state", "stage 4") +pub fn reserve_log_slot(guard: &crate::arch::IrqGuard) -> (*const log::Shard, u64, u32, u32, u32) { + let block = this(); + let (shard, cpu, tid, pid) = + (block.log_shard, block.cpu_id, block.current_tid.load(Relaxed), block.current_pid.load(Relaxed)); + // `log-nested-reserve`'s injection point: between the shard read and the reservation. + crate::log::nested::reserve_window(); + // SAFETY: `guard` masks this CPU, the only one that reserves in its shard. + let seq = unsafe { shard.reserve(guard) }; + (shard, seq, cpu, tid, pid) +} + +/// One delivery of `source`, counted in this CPU's block. +pub(super) fn irq_took(source: crate::irq_census::Source) { + let counts = &this().irq_counts; + counts[crate::irq_census::TOTAL].fetch_add(1, Relaxed); + counts[1 + source as usize].fetch_add(1, Relaxed); } -pub fn irq_counts_here(_first: usize, _second: usize) -> (u64, u64) { - owed!("per-CPU state", "stage 4") +/// Two of this CPU's interrupt counters. +pub fn irq_counts_here(first: usize, second: usize) -> (u64, u64) { + let counts = &this().irq_counts; + (counts[first].load(Relaxed), counts[second].load(Relaxed)) } +#[inline] pub fn preempt_count() -> u32 { - owed!("per-CPU state", "stage 4") + this().preempt_count.load(Relaxed) } -pub fn set_preempt_count(_value: u32) { - owed!("per-CPU state", "stage 4") +#[inline] +pub fn set_preempt_count(value: u32) { + this().preempt_count.store(value, Relaxed); } +/// One increment, atomic against an interrupt on this CPU. +#[inline] pub fn preempt_count_up() { - owed!("per-CPU state", "stage 4") + this().preempt_count.fetch_add(1, Relaxed); } +#[inline] pub fn preempt_count_down() { - owed!("per-CPU state", "stage 4") + this().preempt_count.fetch_sub(1, Relaxed); } +#[inline] pub fn resched_owed() -> bool { - owed!("per-CPU state", "stage 4") + this().need_resched.load(Relaxed) != 0 } -pub fn set_resched_owed(_owed: bool) { - owed!("per-CPU state", "stage 4") +#[inline] +pub fn set_resched_owed(owed: bool) { + this().need_resched.store(u8::from(owed), Relaxed); } +/// Whether this CPU is inside a fault or panic. +#[inline] pub fn faulting() -> bool { - owed!("per-CPU state", "stage 4") + this().fault_state.load(Relaxed) != 0 +} + +/// Timer interrupts taken at EL1. +pub fn ring0_timer_fires() -> u32 { + this().ring0_timer_fires.load(Relaxed) +} + +pub fn last_seen_ring0_fires() -> u32 { + this().last_seen_ring0_fires.load(Relaxed) +} + +pub fn set_last_seen_ring0_fires(v: u32) { + this().last_seen_ring0_fires.store(v, Relaxed); +} + +pub(super) fn note_ring0_timer_fire() { + this().ring0_timer_fires.fetch_add(1, Relaxed); +} + +/// What the timer was last armed for, in counter ticks; zero is stopped. +pub(super) fn armed_ticks() -> u64 { + this().armed_ticks.load(Relaxed) +} + +pub(super) fn set_armed_ticks(ticks: u64) { + this().armed_ticks.store(ticks, Relaxed); } diff --git a/kernel/src/arch/aarch64/pmu.rs b/kernel/src/arch/aarch64/pmu.rs index f9697ad93a..c42ac5435f 100644 --- a/kernel/src/arch/aarch64/pmu.rs +++ b/kernel/src/arch/aarch64/pmu.rs @@ -1,20 +1,21 @@ //! The performance monitor's overflow, as the hard-lockup detector's sample //! source. On AArch64 that is `PMCCNTR_EL0` overflowing into a pseudo-NMI — -//! an interrupt the GICv3 delivers at a priority `DAIF.I` does not mask — -//! which needs the interrupt controller of the port's stage 4. - +//! an interrupt at a priority `DAIF.I` does not mask, which needs every +//! interrupt mask in this kernel moved from `DAIF` to `ICC_PMR_EL1` — and no +//! stage of the port has taken that on. Only a boot under a deadline arms it, +//! and such a boot is refused here by name. /// Start this CPU's counter overflowing into an NMI every `period` counter ticks. pub fn arm(_period: u64) -> bool { - owed!("the PMU overflow NMI", "stage 4") + owed!("the PMU overflow NMI", "no stage yet") } /// Whether this CPU's counter is the reason this NMI arrived. pub fn overflowed() -> bool { - owed!("the PMU overflow NMI", "stage 4") + owed!("the PMU overflow NMI", "no stage yet") } /// After an overflow's sample: clear it and reload. pub fn rearm(_period: u64) { - owed!("the PMU overflow NMI", "stage 4") + owed!("the PMU overflow NMI", "no stage yet") } diff --git a/kernel/src/arch/aarch64/smp.rs b/kernel/src/arch/aarch64/smp.rs index f52708238e..a46829cd40 100644 --- a/kernel/src/arch/aarch64/smp.rs +++ b/kernel/src/arch/aarch64/smp.rs @@ -1,18 +1,20 @@ //! Other CPUs: PSCI `CPU_ON` for each GICC the MADT names, the port's stage 5. -//! Until then the boot CPU is the only one running, and the machine is never -//! released to the scheduler. +//! Until then the boot CPU is the only one the roster holds. +use crate::smp_roster::Roster; + +static ROSTER: Roster = Roster::new(); -/// CPUs running: the boot CPU alone. pub fn cpu_count() -> u32 { - 1 + ROSTER.count() } +/// Release the machine to the scheduler: with one CPU, nothing waits on it. pub fn set_ready() { - owed!("other CPUs", "stage 5") + ROSTER.release(); } -/// Never: the boot ends before the machine is released. +/// Whether [`set_ready`] has run. pub fn is_ready() -> bool { - false + ROSTER.released() } diff --git a/kernel/src/arch/aarch64/switch.rs b/kernel/src/arch/aarch64/switch.rs new file mode 100644 index 0000000000..c7f98cac72 --- /dev/null +++ b/kernel/src/arch/aarch64/switch.rs @@ -0,0 +1,106 @@ +//! `context_switch`: what a context stands on, saved onto the outgoing stack +//! and restored off the incoming one — the callee-saved registers, `DAIF`, +//! and the thread's FP/SIMD state ([`super::fpu`]), which travels with the +//! stack of the thread it belongs to. +//! +//! The frame, from the saved stack pointer up: x19–x28, x29 and `DAIF`, x30 +//! and a pad word, then [`super::fpu::STATE_BYTES`] of FP/SIMD state — the +//! general registers first, because a pair's store reaches only 504 bytes. + +use core::arch::naked_asm; + +use super::fpu::STATE_BYTES; + +/// Where `DAIF` is saved, beside x29. +pub const DAIF_AT: usize = 88; +/// Where x30 is saved: the address the frame returns to. +pub const RETURN_AT: usize = 96; +/// Where the FP/SIMD state starts. +pub const FP_AT: usize = 112; +/// One saved context's bytes. +pub const FRAME_BYTES: usize = FP_AT + STATE_BYTES; + +const _: () = assert!(FRAME_BYTES.is_multiple_of(16)); + +/// Save this context's frame below the stack pointer, store that pointer at +/// `old_sp`, and return into the frame at `new_sp`. +/// # Safety +/// `old_sp` is the outgoing context's save slot and `new_sp` a frame this +/// function or `super::entry::initial_frame` wrote, on a stack that is live. +#[unsafe(naked)] +pub(crate) unsafe extern "C" fn context_switch(old_sp: *mut u64, new_sp: u64) { + naked_asm!( + // The kernel is soft-float: the assembler is told these are here for this block alone. + ".arch_extension fp", + ".arch_extension simd", + "sub sp, sp, #{frame}", + "stp x19, x20, [sp, #0]", + "stp x21, x22, [sp, #16]", + "stp x23, x24, [sp, #32]", + "stp x25, x26, [sp, #48]", + "stp x27, x28, [sp, #64]", + "mrs x9, daif", + "stp x29, x9, [sp, #80]", + "str x30, [sp, #{ret}]", + "add x9, sp, #{fp}", + "stp q0, q1, [x9, #0]", + "stp q2, q3, [x9, #32]", + "stp q4, q5, [x9, #64]", + "stp q6, q7, [x9, #96]", + "stp q8, q9, [x9, #128]", + "stp q10, q11, [x9, #160]", + "stp q12, q13, [x9, #192]", + "stp q14, q15, [x9, #224]", + "stp q16, q17, [x9, #256]", + "stp q18, q19, [x9, #288]", + "stp q20, q21, [x9, #320]", + "stp q22, q23, [x9, #352]", + "stp q24, q25, [x9, #384]", + "stp q26, q27, [x9, #416]", + "stp q28, q29, [x9, #448]", + "stp q30, q31, [x9, #480]", + "mrs x10, fpcr", + "str x10, [x9, #512]", + "mrs x10, fpsr", + "str x10, [x9, #520]", + "mov x9, sp", + "str x9, [x0]", + "mov sp, x1", + "add x9, sp, #{fp}", + "ldp q0, q1, [x9, #0]", + "ldp q2, q3, [x9, #32]", + "ldp q4, q5, [x9, #64]", + "ldp q6, q7, [x9, #96]", + "ldp q8, q9, [x9, #128]", + "ldp q10, q11, [x9, #160]", + "ldp q12, q13, [x9, #192]", + "ldp q14, q15, [x9, #224]", + "ldp q16, q17, [x9, #256]", + "ldp q18, q19, [x9, #288]", + "ldp q20, q21, [x9, #320]", + "ldp q22, q23, [x9, #352]", + "ldp q24, q25, [x9, #384]", + "ldp q26, q27, [x9, #416]", + "ldp q28, q29, [x9, #448]", + "ldp q30, q31, [x9, #480]", + "ldr x10, [x9, #512]", + "msr fpcr, x10", + "ldr x10, [x9, #520]", + "msr fpsr, x10", + "ldp x19, x20, [sp, #0]", + "ldp x21, x22, [sp, #16]", + "ldp x23, x24, [sp, #32]", + "ldp x25, x26, [sp, #48]", + "ldp x27, x28, [sp, #64]", + "ldp x29, x9, [sp, #80]", + "ldr x30, [sp, #{ret}]", + "add sp, sp, #{frame}", + "msr daif, x9", + "ret", + ".arch_extension nosimd", + ".arch_extension nofp", + frame = const FRAME_BYTES, + fp = const FP_AT, + ret = const RETURN_AT, + ); +} diff --git a/kernel/src/arch/aarch64/syscall.rs b/kernel/src/arch/aarch64/syscall.rs index 667d1f82fc..a8c05ecb96 100644 --- a/kernel/src/arch/aarch64/syscall.rs +++ b/kernel/src/arch/aarch64/syscall.rs @@ -1,12 +1,11 @@ -//! The system-call gate: `SVC` from EL0 through the vectors' lower-EL entry, -//! the port's stage 7. +//! The system-call gate is an `SVC` from EL0, taken through the vectors' +//! lower-EL synchronous entry (`super::trap`), which switches to `SP_EL1` in +//! hardware: there is no window with a user stack under the kernel. - -/// A syscall entered this CPU; x86-64's NMI gate counts it, and AArch64 has no -/// such gate. +/// A syscall entered this CPU; x86-64's NMI gate counts it, and AArch64 has no such gate. pub fn note_entry() {} -/// The actuator that storms NMIs at the syscall window. +/// x86-64's storms NMIs at its `SYSCALL` window, which AArch64 has none of. pub fn window_storm() { - owed!("the syscall gate", "stage 7") + panic!("syscall-window-nmi: AArch64's SVC switches stacks in hardware, so there is no window to storm"); } diff --git a/kernel/src/arch/aarch64/tlb.rs b/kernel/src/arch/aarch64/tlb.rs index df6dec4b13..0e0792bc6d 100644 --- a/kernel/src/arch/aarch64/tlb.rs +++ b/kernel/src/arch/aarch64/tlb.rs @@ -1,33 +1,109 @@ -//! TLB invalidation across CPUs. AArch64 broadcasts it in hardware (`TLBI -//! …IS`), so the shootdown this interface stands for is a local instruction -//! plus `DSB ISH` once the kernel owns its page tables, the port's stage 4. +//! TLB invalidation. AArch64 broadcasts it in hardware: a `TLBI …IS` reaches +//! every CPU in the inner-shareable domain, and the `DSB ISH` after it +//! returns once every one of them has dropped the entry (Arm ARM K.a, +//! D8.13.4). So nothing here sends an interrupt or waits for an +//! acknowledgement, and [`shootdown`] is one instruction. +//! +//! Every operation below is bracketed the same way: `DSB ISHST` so the table +//! write it answers for is visible to every walker first, then the `TLBI`, +//! then `DSB ISH` for its completion and `ISB` so this CPU's next fetch walks +//! afresh. +use core::sync::atomic::{AtomicU64, Ordering}; use crate::invalidation::Origin; -pub fn log_census() { - owed!("TLB invalidation", "stage 4") +/// Issuer-side census, as x86-64's counts it; there is no receiver side. +static ISSUED: [AtomicU64; Origin::COUNT] = [const { AtomicU64::new(0) }; Origin::COUNT]; +/// Total at the last print; process exit logs once per batch. +static REPORTED: AtomicU64 = AtomicU64::new(0); + +macro_rules! tlbi { + ($op:literal, $operand:expr) => { + // SAFETY: TLB maintenance and barriers change no memory; dropping a + // translation only makes the next access walk the tables again. + unsafe { + core::arch::asm!( + "dsb ishst", + concat!("tlbi ", $op, ", {}"), + "dsb ish", + "isb", + in(reg) $operand, + options(nostack, preserves_flags), + ) + } + }; +} + +/// Every translation `asid` holds for the 4 KiB page at `va`, on every CPU. +pub(super) fn page(asid: u16, va: u64) { + tlbi!("vae1is", u64::from(asid) << 48 | (va >> 12) & 0xFFF_FFFF_FFFF); +} + +/// Every translation `asid` holds, on every CPU. +pub(super) fn asid(asid: u16) { + tlbi!("aside1is", u64::from(asid) << 48); } -pub fn shootdown(_origin: Origin) { - owed!("TLB invalidation", "stage 4") +/// A global (kernel) translation of the 4 KiB page at `va`, on every CPU. +pub(super) fn kernel_page(va: u64) { + tlbi!("vaae1is", (va >> 12) & 0xFFF_FFFF_FFFF); } -pub fn poll() { - owed!("TLB invalidation", "stage 4") +/// Every EL1&0 translation, on every CPU. +fn all() { + // SAFETY: as `tlbi!`'s. + unsafe { + core::arch::asm!("dsb ishst", "tlbi vmalle1is", "dsb ish", "isb", options(nostack, preserves_flags)); + } +} + +/// Every translation every CPU holds, dropped before this returns — what a +/// returned ASID needs before it is issued again. +pub fn shootdown(origin: Origin) { + ISSUED[origin as usize].fetch_add(1, Ordering::Relaxed); + all(); +} + +/// Nothing to answer: no CPU waits on another's acknowledgement here. +pub fn poll() {} + +/// One `tlb:` line when the counts moved, at process exit. +pub fn log_census() { + let mut counts = [0u64; Origin::COUNT]; + for (slot, count) in ISSUED.iter().zip(counts.iter_mut()) { + *count = slot.load(Ordering::Relaxed); + } + let total: u64 = counts.iter().sum(); + if total == 0 || REPORTED.swap(total, Ordering::Relaxed) == total { + return; + } + struct Fields([u64; Origin::COUNT]); + impl core::fmt::Display for Fields { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + for (name, count) in Origin::NAMES.iter().zip(self.0) { + write!(f, " {name}={count}")?; + } + Ok(()) + } + } + crate::log!("tlb: broadcast invalidations={total}{}", Fields(counts)); } +/// x86-64's measures an IPI round trip, and there is none here. #[cfg(feature = "boot-actuators")] pub fn bench() { - owed!("TLB invalidation", "stage 4") + panic!("tlb-shootdown-bench: AArch64 invalidates by broadcast, so there is no IPI round trip to measure"); } +/// x86-64's delays an acknowledgement, and there is none here. #[cfg(feature = "test-actuators")] pub fn debug_arm_ack_delay(_nanos: u64) -> u64 { - owed!("TLB invalidation", "stage 4") + panic!("SYS_DEBUG: AArch64 invalidates by broadcast, so there is no acknowledgement to delay"); } +/// x86-64's delays an acknowledgement, and there is none here. #[cfg(feature = "test-actuators")] pub fn debug_disarm_ack_delay() -> u64 { - owed!("TLB invalidation", "stage 4") + panic!("SYS_DEBUG: AArch64 invalidates by broadcast, so there is no acknowledgement to delay"); } diff --git a/kernel/src/arch/aarch64/trap.rs b/kernel/src/arch/aarch64/trap.rs index 4747673475..cf4731c65a 100644 --- a/kernel/src/arch/aarch64/trap.rs +++ b/kernel/src/arch/aarch64/trap.rs @@ -1,16 +1,29 @@ -//! Exceptions: the vector table `VBAR_EL1` names, and what the kernel says -//! when one is taken. +//! Exceptions: the vector table `VBAR_EL1` names, and what each one taken +//! comes to. //! -//! Every exception is fatal until the port's stage 4 gives interrupts and -//! user faults somewhere to go: the entry saves the interrupted context into a -//! [`Frame`] on the stack it was taken on, and [`exception`] reports it and -//! panics. A table that does not reach [`exception`] — misaligned, or never +//! Every entry saves the interrupted context into a [`Frame`] on the stack +//! it was taken on — for an entry from EL0, `SP_EL1` as the last `ERET` left +//! it, the top of the running thread's kernel stack — and [`dispatch`] routes +//! it: an interrupt to [`irq`], an `SVC` to the syscall dispatcher, a +//! translation fault to the demand pager, anything else from EL0 to the end +//! of its process, and anything else from EL1 to a panic. Every return to EL0 +//! runs `scheduler::exit_to_user` last. The entry saves no FP/SIMD register: +//! the kernel never touches one, and a switch saves the thread's +//! (`super::switch`). +//! +//! A table that does not reach [`dispatch`] — misaligned, or never //! installed — is what the aarch64 boot test's fault arm catches, because //! then nothing reports at all. use core::fmt; +use core::sync::atomic::{AtomicU32, AtomicU64, Ordering::Relaxed}; + +use toyos_sched::hw::{CpuId, Machine, TraceEvent, TraceKind}; -use crate::log; +use super::percpu::{self, CpuFaultState}; +use super::{cpu, irqchip}; +use crate::irq_census::Source; +use crate::{alert, log}; /// The interrupted context, as the entry stores it: `x0`–`x30`, the stack /// pointer before the exception, then the four system registers that say @@ -28,9 +41,19 @@ pub struct Frame { const FRAME_BYTES: usize = core::mem::size_of::(); const _: () = assert!(FRAME_BYTES == 288 && FRAME_BYTES.is_multiple_of(16)); -/// Which of the table's sixteen entries was taken: Arm ARM K.a, D1.3.1, -/// Table D1-7 — four groups by where the exception came from, and in each the -/// synchronous, IRQ, FIQ and SError entries. +/// The table's entries this kernel acts on: Arm ARM K.a, D1.3.1, Table D1-7. +const EL1_SYNC: u64 = 4; +const EL1_IRQ: u64 = 5; +const EL0_SYNC: u64 = 8; +const EL0_IRQ: u64 = 9; + +/// `ESR_EL1.EC` of the classes an EL0 entry acts on. +const EC_SVC64: u64 = 0x15; +const EC_IABT_LOWER: u64 = 0x20; +const EC_DABT_LOWER: u64 = 0x24; + +/// Which of the table's sixteen entries was taken: four groups by where the +/// exception came from, and in each the synchronous, IRQ, FIQ and SError entries. #[derive(Clone, Copy)] struct Entry(u64); @@ -52,6 +75,7 @@ fn class_name(esr: u64) -> &'static str { 0x0E => "illegal execution state", 0x15 => "SVC from AArch64", 0x18 => "system register access trapped", + 0x19 => "SVE trapped", 0x20 => "instruction abort from a lower EL", 0x21 => "instruction abort", 0x22 => "PC alignment fault", @@ -64,15 +88,29 @@ fn class_name(esr: u64) -> &'static str { } } -/// The Rust half of every vector entry: report the context and panic. The -/// panic handler's early branch puts the report on the console and the panel. -extern "C" fn exception(frame: &Frame, entry: u64) -> ! { +/// The Rust half of every vector entry. Returns to the entry, which restores +/// the frame and returns from the exception; everything fatal diverges. +extern "C" fn dispatch(frame: &mut Frame, entry: u64) { + match entry { + EL1_IRQ => irq(false), + EL0_SYNC => { + el0_sync(frame); + crate::scheduler::exit_to_user(); + } + EL0_IRQ => { + irq(true); + crate::scheduler::exit_to_user(); + } + _ => exception(frame, entry), + } +} + +/// The report of an exception this kernel does not return from: its own, at +/// EL1, or an FIQ or SError from anywhere. The panic handler's branch puts it +/// on the console and the panel. +fn exception(frame: &Frame, entry: u64) -> ! { let entry = Entry(entry); - log!( - "KERNEL PANIC: {entry}: {} (ESR={:#010x})", - class_name(frame.esr), - frame.esr - ); + log!("KERNEL PANIC: {entry}: {} (ESR={:#010x})", class_name(frame.esr), frame.esr); log!(" elr={:#018x} far={:#018x} spsr={:#010x}", frame.elr, frame.far, frame.spsr); for pair in 0..15 { log!( @@ -84,14 +122,250 @@ extern "C" fn exception(frame: &Frame, entry: u64) -> ! { ); } log!(" x30={:#018x} sp ={:#018x}", frame.x[30], frame.sp); + if entry.0 == EL1_SYNC && matches!(frame.esr >> 26, 0x21 | 0x25) && crate::log::PERCPU_READY.load(Relaxed) { + super::paging::debug_page_walk(frame.far); + } crate::symbols::kernel_backtrace(frame.x[29], 20); panic!("{entry}: {} at {:#x}", class_name(frame.esr), frame.elr); } +/// One interrupt: acknowledged, handled, ended. From EL0 a tick or a kick +/// preempts here, where the interrupted context holds nothing; from EL1 it +/// only asks for the pass the context will run when it may. +fn irq(from_el0: bool) { + let Some(intid) = irqchip::acknowledge() else { + percpu::irq_took(Source::Spurious); + return; + }; + if intid == irqchip::timer_intid() { + #[cfg(feature = "boot-actuators")] + let late = irqchip::lateness(); + // Before anything that can take a lock or panic: a timer left + // asserted re-fires forever, and one left stopped never fires again. + irqchip::rearm(); + percpu::irq_took(Source::Timer); + // Before anything that can take a lock, in both levels: a CPU spinning + // on one still takes this interrupt, which is why the poll is here. + crate::deadline::poll(); + #[cfg(feature = "boot-actuators")] + storm::tick(late); + if from_el0 { + // Only an EL0 tick reaches here, so the interrupted context is user + // code and holds no `Lock`. + assert_eq!(crate::preempt::count(), 0, "a timer interrupt from EL0 found a preempt depth"); + let hw = &super::hw::HW; + hw.trace(TraceEvent { ts: hw.now(), cpu: CpuId(percpu::cpu_id()), kind: TraceKind::TimerFire }); + irqchip::end(intid); + crate::scheduler::do_preempt(); + } else { + crate::preempt::set_need_resched(); + percpu::note_ring0_timer_fire(); + irqchip::end(intid); + } + return; + } + match intid { + irqchip::SGI_KICK => { + percpu::irq_took(Source::Timer); + irqchip::end(intid); + if from_el0 { + crate::scheduler::do_preempt(); + } else { + crate::preempt::set_need_resched(); + } + } + #[cfg(feature = "boot-actuators")] + intid if intid == u32::from(LOG_NEST_VECTOR) => { + percpu::preempt_count_up(); + crate::log::nested::deliver(); + percpu::preempt_count_down(); + irqchip::end(intid); + } + #[cfg(feature = "boot-actuators")] + irqchip::SGI_STORM => { + storm::sgi(); + irqchip::end(intid); + } + _ => { + percpu::irq_took(Source::Unclaimed); + UNCLAIMED.fetch_add(1, Relaxed); + LAST_UNCLAIMED.store(intid, Relaxed); + irqchip::end(intid); + } + } +} + +/// Interrupts no handler here claims, and the last one's INTID. +static UNCLAIMED: AtomicU64 = AtomicU64::new(0); +static LAST_UNCLAIMED: AtomicU32 = AtomicU32::new(0); +/// The count at the last report; process exit logs once per batch. +static UNCLAIMED_REPORTED: AtomicU64 = AtomicU64::new(0); + +pub(crate) fn log_unclaimed() { + let count = UNCLAIMED.load(Relaxed); + if count == 0 || UNCLAIMED_REPORTED.swap(count, Relaxed) == count { + return; + } + log!("irq: unclaimed interrupts={count}, the last INTID {}", LAST_UNCLAIMED.load(Relaxed)); +} + +/// A synchronous exception from EL0: a syscall, a fault the demand pager may +/// serve, or the end of the process. +fn el0_sync(frame: &mut Frame) { + match frame.esr >> 26 { + EC_SVC64 => syscall(frame), + EC_IABT_LOWER | EC_DABT_LOWER => user_abort(frame), + _ => { + percpu::preempt_count_up(); + user_fatal(frame); + } + } +} + +/// `SVC #0`: the number in `x0`, four arguments in `x1`–`x4`, the answer +/// back in `x0`, and every other register the thread's as it left it. +fn syscall(frame: &mut Frame) { + percpu::enter_syscall(frame.elr, frame.x[0], frame.x[29], frame.sp); + percpu::preempt_count_up(); + let answer = crate::syscall::dispatch::syscall_dispatch(frame.x[0], frame.x[1], frame.x[2], frame.x[3], frame.x[4]); + percpu::preempt_count_down(); + percpu::leave_syscall(); + frame.x[0] = answer; +} + +/// `DFSC`/`IFSC` levels 0 to 3 of a translation fault: nothing mapped there. +fn is_translation_fault(esr: u64) -> bool { + esr & 0b11_1100 == 0b00_0100 +} + +/// An abort from EL0: a translation fault in a region the demand pager +/// fills, or the end of the process. +fn user_abort(frame: &mut Frame) { + percpu::preempt_count_up(); + // `FnV`: a data abort whose `FAR_EL1` is not valid names no address to fill. + let far_valid = frame.esr >> 26 == EC_IABT_LOWER || frame.esr & (1 << 10) == 0; + if is_translation_fault(frame.esr) && far_valid { + let prev = percpu::swap_fault_state(CpuFaultState::PageFault); + if prev != CpuFaultState::Normal { + // Put back: `user_fatal` classifies a recursive fault by what it finds. + percpu::set_fault_state(prev); + user_fatal(frame); + } + cpu::enable_interrupts(); + let served = crate::process::handle_page_fault(frame.far, frame.esr); + cpu::disable_interrupts(); + if served { + percpu::set_fault_state(CpuFaultState::Normal); + percpu::preempt_count_down(); + return; + } + log!( + "#PF UNHANDLED: far={:#x} pc={:#x} esr={:#x} tid={:?}", + frame.far, + frame.elr, + frame.esr, + percpu::current_tid() + ); + } + user_fatal(frame); +} + +/// A fault from EL0 nothing serves: report it, and end the process it came +/// from; a second fault while this one reports halts the machine. +fn user_fatal(frame: &Frame) -> ! { + let prev = percpu::swap_fault_state(CpuFaultState::Fatal); + let recursive = matches!(prev, CpuFaultState::Fatal | CpuFaultState::Panic); + // First: a panic anywhere below reaches the panic handler as DOUBLE + // PANIC, which can only report what was captured here. + crate::panic::record_fault(class_name(frame.esr), frame.elr, frame.far, frame.esr); + if crate::actuator::panic_in_report() { + panic!("panic-in-report: the crash report panicked before it said anything"); + } + let tid = percpu::current_tid().map_or(u32::MAX, |t| t.raw()); + alert!( + "FAULT pc={:#018x} far={:#018x} esr={:#010x} sp={:#018x} tid={tid}{}", + frame.elr, + frame.far, + frame.esr, + frame.sp, + if recursive { " RECURSIVE" } else { "" } + ); + if recursive { + crate::panic::halt_all_cpus(); + } + user_report(frame); + percpu::set_fault_state(CpuFaultState::Normal); + crate::panic::forget(); + crate::syscall::kill_process(-1); +} + +/// What a fault from EL0 says about the thread: the signal it would be +/// elsewhere, its registers, and its frames, read without a lock. +fn user_report(frame: &Frame) { + let tid = percpu::current_tid().map_or(u32::MAX, |t| t.raw()); + let class = frame.esr >> 26; + match class { + EC_IABT_LOWER | EC_DABT_LOWER => { + let action = if class == EC_IABT_LOWER { + "execute" + } else if frame.esr & (1 << 6) != 0 { + "write" + } else { + "read" + }; + let cause = match frame.esr & 0b11_1100 { + 0b00_0100 => "unmapped address", + 0b00_1100 => "protection violation", + _ => "fault", + }; + log!("SEGFAULT tid={tid}: {action} {cause} at {:#x} (FSC {:#x})", frame.far, frame.esr & 0x3F); + } + 0x00 => log!("SIGILL tid={tid}: illegal instruction"), + 0x22 | 0x26 => log!("SIGBUS tid={tid}: {}", class_name(frame.esr)), + 0x3C => log!("SIGTRAP tid={tid}: BRK #{:#x}", frame.esr & 0xFFFF), + _ => log!("FATAL tid={tid}: {} (ESR={:#010x})", class_name(frame.esr), frame.esr), + } + log!(" pc:"); + if percpu::current_pid().is_some() { + crate::process::resolve_user_symbol(frame.elr).log_bare(frame.elr); + } else { + log!(" {:#x}", frame.elr); + } + if matches!(class, EC_IABT_LOWER | EC_DABT_LOWER) { + super::paging::debug_page_walk(frame.far); + } + log!(" Registers:"); + for pair in 0..15 { + log!(" x{:<2}={:#018x} x{:<2}={:#018x}", pair * 2, frame.x[pair * 2], pair * 2 + 1, frame.x[pair * 2 + 1]); + } + log!(" x30={:#018x} sp={:#018x} spsr={:#010x}", frame.x[30], frame.sp, frame.spsr); + log!(" Backtrace:"); + user_backtrace(frame.x[29], 32); + let crash_addr = if matches!(class, EC_IABT_LOWER | EC_DABT_LOWER) { frame.far } else { 0 }; + crate::process::dump_crash_diagnostics(crash_addr, frame.elr); +} + +/// The user frame-pointer chain from `fp`: each frame's saved `x29` at `fp` +/// and its return address at `fp + 8`, read through the running space's tables. +fn user_backtrace(start_fp: u64, max_frames: usize) { + let mut fp = start_fp; + for _ in 0..max_frames { + let Some(saved) = super::paging::read_user_word(fp) else { break }; + let Some(ret) = super::paging::read_user_word(fp + 8) else { break }; + if ret == 0 { + break; + } + crate::process::resolve_user_symbol_return(ret).log_bare(ret); + fp = saved; + } +} + // The table: sixteen entries of 0x80 bytes, 2 KiB aligned (`VBAR_EL1` bits // 10:0 are RES0). Each entry makes room for a `Frame`, saves x0/x1, names -// itself in x1 and branches to the common save, which fills in the rest and -// calls `exception` with the frame in x0. +// itself in x1 and branches to the common save, which fills in the rest, calls +// `dispatch` with the frame in x0, and restores whatever it returns to. The +// stack an entry from EL0 names is `SP_EL0`, and a return to EL0 (`SPSR_EL1.M` +// zero) puts the frame's back. core::arch::global_asm!( ".macro toyos_vector n", ".balign 0x80", @@ -123,7 +397,12 @@ core::arch::global_asm!( "stp x24, x25, [sp, #192]", "stp x26, x27, [sp, #208]", "stp x28, x29, [sp, #224]", + "tbnz x1, #3, 2f", "add x2, sp, #{frame}", + "b 3f", + "2:", + "mrs x2, sp_el0", + "3:", "stp x30, x2, [sp, #240]", "mrs x2, elr_el1", "mrs x3, spsr_el1", @@ -132,18 +411,43 @@ core::arch::global_asm!( "mrs x3, far_el1", "stp x2, x3, [sp, #272]", "mov x0, sp", - "bl {exception}", - "brk #0", + "bl {dispatch}", + "ldp x2, x3, [sp, #256]", + "msr elr_el1, x2", + "msr spsr_el1, x3", + "ldr x2, [sp, #248]", + "tst x3, #0xF", + "b.ne 4f", + "msr sp_el0, x2", + "4:", + "ldp x2, x3, [sp, #16]", + "ldp x4, x5, [sp, #32]", + "ldp x6, x7, [sp, #48]", + "ldp x8, x9, [sp, #64]", + "ldp x10, x11, [sp, #80]", + "ldp x12, x13, [sp, #96]", + "ldp x14, x15, [sp, #112]", + "ldp x16, x17, [sp, #128]", + "ldp x18, x19, [sp, #144]", + "ldp x20, x21, [sp, #160]", + "ldp x22, x23, [sp, #176]", + "ldp x24, x25, [sp, #192]", + "ldp x26, x27, [sp, #208]", + "ldp x28, x29, [sp, #224]", + "ldr x30, [sp, #240]", + "ldp x0, x1, [sp, #0]", + "add sp, sp, #{frame}", + "eret", ".popsection", frame = const FRAME_BYTES, - exception = sym exception, + dispatch = sym dispatch, ); /// Point `VBAR_EL1` at the table. Called by the entry before anything that can fault. pub fn install() { // SAFETY: the table is 2 KiB aligned, lives in the kernel image for the - // life of the machine, and every entry ends in `exception`, which never - // returns. The `ISB` makes the new base the one the next exception uses. + // life of the machine, and every entry ends in `dispatch`. The `ISB` makes + // the new base the one the next exception uses. unsafe { core::arch::asm!( "adrp {t}, toyos_vectors", @@ -156,9 +460,10 @@ pub fn install() { } } -/// Interrupt identifiers the generic drivers program. Inert until stage 4 -/// assigns them to GIC interrupts: every path that would deliver one -/// ([`super::msi_message`], [`super::irqchip::send_self`]) is owed. +/// The numbers the generic drivers program as an interrupt's identity. On +/// AArch64 [`LOG_NEST_VECTOR`] is the SGI `irqchip::send_self` raises; the +/// other two name a message-signalled interrupt, which `super::msi_message` +/// refuses on this machine. #[repr(u8)] enum Vector { LogNest = 1, @@ -170,11 +475,23 @@ pub const HDA_VECTOR: u8 = Vector::Hda as u8; pub const VIRTIO_SOUND_VECTOR: u8 = Vector::VirtioSound as u8; pub const LOG_NEST_VECTOR: u8 = Vector::LogNest as u8; -/// The crash report for a panic, from the frame pointer the panic handler stood on. +/// The crash report for a panic, from the frame pointer the panic handler +/// stood on: the backtrace, which CPU is on which stack, and what the +/// running thread was doing. pub(crate) fn report_panic(message: &core::panic::PanicInfo, frame: u64) { - crate::alert!("PANIC: {}", message); + alert!("PANIC: {}", message); log!(" Backtrace:"); crate::symbols::kernel_backtrace(frame, 20); + super::hw::report_contexts(cpu::stack_pointer(), None); + let Some(pid) = percpu::current_pid() else { return }; + log!(" Running: pid={} tid={:?}", pid, percpu::current_tid()); + if percpu::in_syscall() { + let (pc, fp, sp) = percpu::syscall_context(); + log!(" Syscall: num={} user_pc={pc:#x} user_sp={sp:#x}", percpu::syscall_num()); + log!(" User backtrace:"); + crate::process::resolve_user_symbol(pc).log_bare(pc); + user_backtrace(fp, 20); + } } /// Whether the interrupted context an `SPSR` describes could have taken an @@ -186,15 +503,85 @@ pub(crate) const fn frame_interrupts_enabled(spsr: u64) -> bool { /// Nothing to report: the vectors run on the stack they interrupted. pub(crate) fn report_fault_stack() {} +/// Every return to EL0's last word, and a new thread's first. pub fn kernel_exit_to_user_check() { - owed!("the return to user mode", "stage 7") -} - -pub(crate) fn log_unclaimed() { - owed!("the interrupt controller", "stage 4") + crate::scheduler::exit_to_user(); } +/// x86-64's lands a `#DF` on its IST stack; AArch64 has no double fault, and +/// an exception on a broken stack takes the same vector again. #[cfg(feature = "test-actuators")] pub(crate) fn provoke_double_fault() -> ! { - owed!("a fault on the exception stack", "stage 4") + panic!("SYS_DEBUG: AArch64 has no double fault to provoke"); +} + +/// `irq-storm`: the timer ticking at a known period while this CPU floods +/// itself with SGIs. A tick is lost when its expiry goes a whole period +/// untaken — the next one would have been due before it — or when it never +/// comes, so the verdict is the latest any tick was taken and how many came, +/// beside every SGI sent being taken. Each tick is armed from when the last +/// was taken, so the count falls short of the span's periods by the latency +/// they add up to, and nine in ten is the floor that leaves room for it. +#[cfg(feature = "boot-actuators")] +pub(crate) mod storm { + use core::sync::atomic::{AtomicBool, AtomicU64, Ordering::Relaxed}; + + use super::irqchip; + use crate::log; + + static RUNNING: AtomicBool = AtomicBool::new(false); + static TICKS: AtomicU64 = AtomicU64::new(0); + static LATEST: AtomicU64 = AtomicU64::new(0); + static SGIS: AtomicU64 = AtomicU64::new(0); + + /// One tick, `late` counter ticks past its comparator. + pub(super) fn tick(late: u64) { + if RUNNING.load(Relaxed) { + TICKS.fetch_add(1, Relaxed); + LATEST.fetch_max(late, Relaxed); + } + } + + pub(super) fn sgi() { + SGIS.fetch_add(1, Relaxed); + } + + /// The timer's period. + const PERIOD_NS: u64 = 1_000_000; + /// How long the storm lasts: two thousand periods. + const SPAN_NS: u64 = 2_000 * PERIOD_NS; + + /// Run the storm, with interrupts open on this CPU for its length, and + /// say what it counted. + pub fn run() { + let _guard = crate::arch::IrqGuard::close(); + TICKS.store(0, Relaxed); + LATEST.store(0, Relaxed); + SGIS.store(0, Relaxed); + irqchip::arm_one_shot(PERIOD_NS); + RUNNING.store(true, Relaxed); + let start = crate::clock::nanos_since_boot(); + let mut sent = 0u64; + crate::arch::cpu::enable_interrupts(); + while crate::clock::nanos_since_boot() - start < SPAN_NS { + irqchip::send_self(irqchip::SGI_STORM as u8); + sent += 1; + } + crate::arch::cpu::disable_interrupts(); + RUNNING.store(false, Relaxed); + irqchip::stop_timer(); + let (ticks, taken) = (TICKS.load(Relaxed), SGIS.load(Relaxed)); + let latest_ns = crate::clock::nanos_of_ticks(LATEST.load(Relaxed)); + // The last SGI sent may still be in flight when the mask closes on it. + let sgis_whole = taken + 1 >= sent && taken <= sent; + let owed = SPAN_NS / PERIOD_NS; + let verdict = if sgis_whole && ticks >= owed * 9 / 10 && latest_ns < PERIOD_NS { "PASS" } else { "FAIL" }; + log!( + "irq-storm: {verdict} sgis={taken}/{sent} ticks={ticks} of {owed} over {} ms at a {} us period, \ + the latest taken {} us past its expiry", + SPAN_NS / 1_000_000, + PERIOD_NS / 1_000, + latest_ns / 1_000, + ); + } } diff --git a/kernel/src/arch/aarch64/watchdog.rs b/kernel/src/arch/aarch64/watchdog.rs index 628608537f..7e88727e37 100644 --- a/kernel/src/arch/aarch64/watchdog.rs +++ b/kernel/src/arch/aarch64/watchdog.rs @@ -1,16 +1,17 @@ //! The platform watchdog: on an ACPI Arm machine, an SBSA generic watchdog -//! the GTDT names, the port's stage 6. +//! the GTDT names, which no stage of the port has taken on yet. So one is +//! never armed, and a boot that asks for one is refused by name. use crate::drivers::pci::PciDevice; pub fn init(_devices: &[PciDevice]) { - owed!("the platform watchdog", "no stage yet") + if crate::params::watchdog() { + owed!("the platform watchdog (the SBSA generic watchdog the GTDT names)", "no stage yet"); + } } -pub fn feed(_now: u64) { - owed!("the platform watchdog", "no stage yet") -} +/// Nothing is armed to feed. +pub fn feed(_now: u64) {} -pub fn disarm() { - owed!("the platform watchdog", "no stage yet") -} +/// Nothing is armed to disarm. +pub fn disarm() {} diff --git a/kernel/src/arch/x86_64/apic.rs b/kernel/src/arch/x86_64/apic.rs index d6b2b703d9..11ab84c42b 100644 --- a/kernel/src/arch/x86_64/apic.rs +++ b/kernel/src/arch/x86_64/apic.rs @@ -48,8 +48,8 @@ pub const MSI_DOORBELL: u32 = 0xFEE0_0000; /// The compatibility-format message that raises `vector` on the CPU whose APIC /// ID is `dest`: the destination in address bits 19:12, the vector in the data. -pub fn msi_message(dest: u32, vector: u8) -> (u32, u32) { - (MSI_DOORBELL | (dest << 12), vector as u32) +pub fn msi_message(dest: u32, vector: u8) -> Result<(u32, u32), &'static str> { + Ok((MSI_DOORBELL | (dest << 12), vector as u32)) } /// Calibrated LAPIC timer ticks per 10ms (computed on BSP, reused by APs). diff --git a/kernel/src/arch/x86_64/boot.rs b/kernel/src/arch/x86_64/boot.rs index b2b3cf9840..b299f725df 100644 --- a/kernel/src/arch/x86_64/boot.rs +++ b/kernel/src/arch/x86_64/boot.rs @@ -138,4 +138,5 @@ pub fn interrupt_selftests() { if crate::actuator::unclaimed_vector_selftest() { idt::unclaimed::selftest(); } + assert!(!crate::actuator::irq_storm(), "irq-storm is the GIC's selftest, and this machine has a local APIC"); } diff --git a/kernel/src/arch/x86_64/hw.rs b/kernel/src/arch/x86_64/hw.rs index 78a458f669..a677adcba2 100644 --- a/kernel/src/arch/x86_64/hw.rs +++ b/kernel/src/arch/x86_64/hw.rs @@ -137,7 +137,7 @@ pub fn report_contexts(rsp: u64, subject: Option) { None => crate::log!( " cpu{cpu} is on ctx {held:#x} (its idle context) stack_top={top:#018x} \ saved_rsp={:#018x}{}{}", - ctx.rsp, + ctx.sp, if same { " <== THE SAME CONTEXT" } else { "" }, if top == 0 { "" } else { " <== AN IDLE CONTEXT'S STACK TOP IS ZERO BY CONSTRUCTION" }, ), @@ -146,7 +146,7 @@ pub fn report_contexts(rsp: u64, subject: Option) { saved_rsp={:#018x}{}{}", id.0.raw(), id.1.raw(), - ctx.rsp, + ctx.sp, if same { " <== THE SAME CONTEXT" } else { "" }, if on_its_stack { " <== AND THIS CRASH IS ON THAT STACK" } else { "" }, ), @@ -159,7 +159,7 @@ pub fn report_contexts(rsp: u64, subject: Option) { #[cold] #[inline(never)] fn switch_frame_is_wrong(ctx: &KernelCtx, token: &RunToken) -> ! { - let rsp = ctx.rsp; + let rsp = ctx.sp; let (pid, tid) = ctx.id.map_or((u32::MAX, u32::MAX), |id| (id.0.raw(), id.1.raw())); crate::log!( "CONTEXT SWITCH ONTO A FRAME THAT IS NOT ONE: cpu={} pid={pid} tid={tid} \ @@ -171,7 +171,7 @@ fn switch_frame_is_wrong(ctx: &KernelCtx, token: &RunToken) -> ! ctx.kernel_stack_top, ctx.kernel_stack_top.wrapping_sub(rsp) as i64, ctx.preempt, - ctx.fs_base, + ctx.thread_pointer, token.incoming().map(|k| k.0), token.outgoing().map(|k| k.0), ); @@ -193,11 +193,11 @@ fn switch_frame_is_wrong(ctx: &KernelCtx, token: &RunToken) -> ! ); } -/// Returns the validated `rsp` — callers must use this value; re-reading `ctx.rsp` here would be a second, unguarded load after `cr3.activate()`'s memory clobber. +/// Returns the validated `rsp` — callers must use this value; re-reading `ctx.sp` here would be a second, unguarded load after `cr3.activate()`'s memory clobber. #[inline] #[must_use] fn check_switch_frame(ctx: &KernelCtx, token: &RunToken) -> u64 { - let rsp = ctx.rsp; + let rsp = ctx.sp; if !crate::mm::is_kernel_addr(rsp) || !rsp.is_multiple_of(8) { switch_frame_is_wrong(ctx, token); } @@ -288,7 +288,7 @@ pub(crate) unsafe extern "C" fn switch_witness_verify(rsp: u64) { *word = unsafe { core::ptr::read_volatile((rsp + (i as u64) * 8) as *const u64) }; } // SAFETY: `shadow.ctx` is a live `KernelCtx` from this kernel's own pass. - let field = unsafe { core::ptr::read_volatile(&raw const (*shadow.ctx).rsp) }; + let field = unsafe { core::ptr::read_volatile(&raw const (*shadow.ctx).sp) }; if rsp == shadow.rsp && field == shadow.rsp && now == shadow.words { return; } @@ -351,7 +351,7 @@ unsafe fn switch_witness_mutate(restore: *const KernelCtx) { return; } // SAFETY: `restore` is a live `KernelCtx` from this kernel's own pass. - let rsp = unsafe { (*restore).rsp }; + let rsp = unsafe { (*restore).sp }; #[cfg(feature = "switch-witness-mutate-frame")] // SAFETY: the `rbx` slot of the frame `check_switch_frame` has just validated. unsafe { @@ -362,7 +362,7 @@ unsafe fn switch_witness_mutate(restore: *const KernelCtx) { // the report and not by a different crash. // SAFETY: as above, and the field is this context's own. unsafe { - core::ptr::write_volatile(&raw const (*restore).rsp as *mut u64, rsp + 8) + core::ptr::write_volatile(&raw const (*restore).sp as *mut u64, rsp + 8) }; } @@ -418,12 +418,12 @@ impl Hw for KernelHw { unsafe fn switch(&self, token: RunToken) { let save = token.save_ptr(); let restore = token.restore_ptr(); - // SAFETY: `save`/`restore` are live Box-backed contexts from `SchedPass::finish`, freed only by a later pass; `incoming.fs_base` is this kernel's own canonical value for the thread being installed. + // SAFETY: `save`/`restore` are live Box-backed contexts from `SchedPass::finish`, freed only by a later pass; `incoming.thread_pointer` is this kernel's own canonical value for the thread being installed. unsafe { - (*save).fs_base = cpu::read_fs_base(); + (*save).thread_pointer = cpu::read_fs_base(); (*save).preempt = crate::preempt::count(); let incoming: &KernelCtx = &*restore; - // The only load of `incoming.rsp`: reading it again after `cr3.activate()`'s clobber would be a second, unguarded load. + // The only load of `incoming.sp`: reading it again after `cr3.activate()`'s clobber would be a second, unguarded load. let rsp = check_switch_frame(incoming, &token); #[cfg(any( feature = "switch-witness-mutate-frame", @@ -442,7 +442,7 @@ impl Hw for KernelHw { crate::heartbeat::note_dispatch(); percpu::set_kernel_stack(incoming.kernel_stack_top); incoming.root.activate(); - cpu::write_fs_base(incoming.fs_base); + cpu::write_fs_base(incoming.thread_pointer); } // idle's stack top is per-CPU, unknowable at boot-time init, so it is read here instead. None => { @@ -461,7 +461,7 @@ impl Hw for KernelHw { ds = in(reg) percpu::KERNEL_DS as u64, options(nomem, nostack, preserves_flags), ); - context_switch(&raw mut (*save).rsp, rsp); + context_switch(&raw mut (*save).sp, rsp); } } diff --git a/kernel/src/arch/x86_64/idt/mod.rs b/kernel/src/arch/x86_64/idt/mod.rs index 3c7d56d78a..be72d3d92c 100644 --- a/kernel/src/arch/x86_64/idt/mod.rs +++ b/kernel/src/arch/x86_64/idt/mod.rs @@ -347,35 +347,7 @@ extern "sysv64" fn common_entry() { /// Deferred-preempt epilogue; caller must have IF=0 on entry and it returns with IF=0. pub(crate) extern "sysv64" fn kernel_exit_to_user_check() { - flush_ring0_timer_fires_to_trace(); - loop { - // A killed or stopped thread returns to Ring 3 exactly once more: never. - crate::scheduler::leave_ring3_if_due(); - // `do_preempt` owns clearing `need_resched`; this function never clears it itself. - if !crate::preempt::need_resched() { - #[cfg(feature = "boot-actuators")] - if crate::actuator::dump_in_blocking_pass() { - crate::sched::dump::staged::note_return_to_ring3(); - } - return; - } - assert!(!crate::scheduler::in_schedule_self(), - "exit-to-user inside a scheduler pass"); - // Not an IrqGuard: both loop exits must set IF, not restore a saved value. - cpu::enable_interrupts(); - crate::scheduler::do_preempt(); - cpu::disable_interrupts(); - flush_ring0_timer_fires_to_trace(); - } -} - -fn flush_ring0_timer_fires_to_trace() { - let cur = percpu::ring0_timer_fires(); - let missed = cur.wrapping_sub(percpu::last_seen_ring0_fires()); - if missed > 0 { - crate::trace::trace(crate::trace::Kind::TimerFireBurst, missed); - percpu::set_last_seen_ring0_fires(cur); - } + crate::scheduler::exit_to_user(); } /// Routes by vector to the appropriate handler; #DF and #MC get dedicated arms because they are aborts with no instruction to return to. diff --git a/kernel/src/arch/x86_64/paging.rs b/kernel/src/arch/x86_64/paging.rs index 43991575b4..dd20b205ea 100644 --- a/kernel/src/arch/x86_64/paging.rs +++ b/kernel/src/arch/x86_64/paging.rs @@ -8,7 +8,6 @@ use crate::log; use alloc::boxed::Box; -use alloc::collections::BTreeMap; use alloc::vec::Vec; use crate::hasher::HashMap; @@ -21,7 +20,6 @@ use crate::arch::cpu::Invpcid; use crate::sync::Lock; use crate::vma::{self, Occupancy, Region, RegionKind}; use toyos_bootmap::ROOT_HIGH_HALF; -use toyos_userbound::PageSpan; use crate::MemoryMapEntry; const PAGE_PRESENT: u64 = 1 << 0; @@ -368,7 +366,7 @@ pub struct AddressSpace { /// Physical data pages mapped into user space, keyed by physical address. Freed on drop. pages: HashMap, /// All virtual memory regions, keyed by start address. - regions: BTreeMap, + regions: vma::Regions, /// Owned for this space's life: dropping the space returns a user tag, so two /// live spaces can never share one. pcid: PcidHandle, @@ -396,7 +394,7 @@ impl AddressSpace { root: pml4, children: Vec::new(), pages: HashMap::default(), - regions: BTreeMap::new(), + regions: vma::Regions::default(), pcid: PcidHandle::User(pcid), }) } @@ -600,20 +598,9 @@ impl AddressSpace { Some((crate::mm::DirectMap::from_phys(page_phys + offset), rights)) } - /// Where `span` goes, top-down and never below the floor: a region the - /// kernel placed under it, the clock page, bounds no gap. - fn find_gap(&self, span: PageSpan) -> Option { - let taken = self.regions.iter().rev().map(|(start, region)| (start.raw(), region.size)); - vma::window().gap(span, taken).map(UserAddr::new) - } - - /// Allocate a virtual address range and register the region. `size` is - /// made a [`PageSpan`] before anything is summed on it. + /// Allocate a virtual address range and register the region. pub fn alloc_region(&mut self, size: u64, kind: RegionKind) -> Option { - let span = vma::window().span(size)?; - let addr = self.find_gap(span)?; - self.regions.insert(addr, Region { size: span.bytes(), kind }); - Some(addr) + self.regions.alloc(size, kind) } /// A mixed-`Prot` image uses [`alloc_region`](Self::alloc_region) plus [`map_window`](Self::map_window) per 2 MiB instead. @@ -628,59 +615,30 @@ impl AddressSpace { phys & (PAGE_2M - 1) == 0, "alloc_and_map: phys {phys:#x} not 2MB-aligned" ); - let span = vma::window().span(size)?; - let addr = self.find_gap(span)?; - let aligned = span.bytes(); - self.regions.insert( - addr, - Region { - size: aligned, - kind: RegionKind::Mapped, - }, - ); + let (addr, aligned) = self.regions.alloc_mapped(size)?; self.map_range(addr, phys, aligned, prot, cache); Some((addr, aligned)) } /// Free a previously allocated region and unmap it. pub fn free_and_unmap(&mut self, addr: UserAddr) -> Option { - let size = self.regions.remove(&addr)?.size; + let size = self.regions.remove(addr)?; self.unmap_range(addr, size); Some(size) } /// Insert a region at a specific address (for ELF segments, stack, etc.) pub fn insert_region(&mut self, addr: UserAddr, region: Region) { - assert!( - self.find_region(addr).is_none(), - "insert_region: address {:#x} already occupied", - addr.raw() - ); self.regions.insert(addr, region); } /// Find the region containing `addr`. Returns (start_addr, region). pub fn find_region(&self, addr: UserAddr) -> Option<(UserAddr, &Region)> { - let (&start, region) = self.regions.range(..=addr).next_back()?; - if addr.raw() < start.raw() + region.size { - Some((start, region)) - } else { - None - } + self.regions.find(addr) } - /// The end is saturating so a caller's arithmetic cannot wrap into a smaller range. pub fn occupancy(&self, addr: UserAddr, size: u64) -> Occupancy { - let end = UserAddr::new(addr.raw().saturating_add(size)); - let mut over = self.overlapping_regions(addr, end); - let Some((&start, region)) = over.next() else { - return Occupancy::Free; - }; - if over.next().is_none() && start == addr && region.size == size { - Occupancy::Whole - } else { - Occupancy::Partial - } + self.regions.occupancy(addr, size) } /// Iterate all regions that overlap the range [start, end). @@ -689,10 +647,7 @@ impl AddressSpace { start: UserAddr, end: UserAddr, ) -> impl Iterator { - // Overlaps [start, end) iff s < end && s+n > start; `range(..end)` prunes the first half. - self.regions - .range(..end) - .filter(move |(&s, r)| s.raw() + r.size > start.raw()) + self.regions.overlapping(start, end) } /// Private: not safe to use until every CPU is told — the free fn [`map_mmio`] is the whole operation. @@ -914,18 +869,11 @@ pub fn map_mmio(phys: u64, size: u64, policy: MmioPolicy) -> crate::mm::Mmio { mmio } - -/// Take the 4 KiB page holding `addr` out of the kernel direct map; `addr`'s -/// page must be owned by the caller forever (see [`AddressSpace::guard_4k`]). -pub fn guard_kernel_page(addr: u64) { - assert!(crate::mm::is_kernel_addr(addr), "guard_kernel_page: {addr:#x} is not a kernel address"); - kernel().lock().guard_4k(crate::mm::DirectMap::phys_of(addr as *const u8)); -} - /// Build kernel page tables: the direct map in the high half, in 2 MiB pages, /// as far as [`toyos_bootmap::x86_64::direct_map_end`] reaches, and answer -/// that end. -pub(crate) fn init(memory_map: &[MemoryMapEntry]) -> toyos_bootmap::DirectMapEnd { +/// that end. The scanout is not mapped for itself: in the low map the MTRRs +/// type it until `panic_console::remap` gives it a type of its own. +pub(crate) fn init(memory_map: &[MemoryMapEntry], _scanout: Option<(u64, u64)>) -> toyos_bootmap::DirectMapEnd { let extent = toyos_bootmap::x86_64::direct_map_end(memory_map) .unwrap_or_else(|refusal| panic!("paging: firmware's memory map: {refusal}")); let end = extent.get(); @@ -934,7 +882,7 @@ pub(crate) fn init(memory_map: &[MemoryMapEntry]) -> toyos_bootmap::DirectMapEnd root: Box::new(PageTablePage([0; 512])), children: Vec::new(), pages: HashMap::default(), - regions: BTreeMap::new(), + regions: vma::Regions::default(), pcid: PcidHandle::Kernel, }; @@ -1186,3 +1134,14 @@ pub fn scanout_memory_type(addr: u64, size: u64) -> impl core::fmt::Display { } Report(crate::arch::mtrr::range_type(addr, size)) } + +/// Whether a process can be given the scanout at `phys`, which it is in +/// 2 MiB pages: its base on one, as the loader already requires, since what +/// the last page covers past it is the aperture firmware put it in. +pub fn scanout_whole_pages(phys: u64, _size: u64) -> Result<(), &'static str> { + if phys.is_multiple_of(PAGE_2M) { + Ok(()) + } else { + Err("its base is not on a 2 MiB page, and a process is given a scanout in 2 MiB pages") + } +} diff --git a/kernel/src/arch/x86_64/percpu.rs b/kernel/src/arch/x86_64/percpu.rs index f70b189345..9c92b71899 100644 --- a/kernel/src/arch/x86_64/percpu.rs +++ b/kernel/src/arch/x86_64/percpu.rs @@ -5,6 +5,7 @@ use alloc::alloc::alloc_zeroed; use core::alloc::Layout; use super::cpu; +use crate::sched::idle_stack::{words, FILL as STACK_FILL, FILL_WORD as STACK_FILL_WORD}; use crate::log; const MSR_GS_BASE: u32 = 0xC000_0101; @@ -310,27 +311,16 @@ pub(crate) mod gs { } } -/// Same size as a task's kernel stack: a `deferred` [`kobject!`] object may -/// own an `immediate` one, whose destructor then runs here instead. -const IDLE_STACK_SIZE: usize = crate::process::KERNEL_STACK_SIZE; - -/// One unmapped 4 KiB page below every idle stack: unmapped, not filled like -/// [`IST_GUARD_SIZE`], so a fault here escalates to IST1's `#DF` rather than silently corrupting memory. -const IDLE_GUARD_SIZE: usize = 4096; - /// The IST stacks this machine has: IST1 `#DF`, IST2 NMI, IST3 `#MC` — vectors /// that can arrive with `rsp` not a kernel stack (SDM Vol. 3A §6.14.5); `ist[n-1]` is IST*n*. pub(crate) const IST_STACKS: usize = 3; -/// One size for every IST stack, for [`IDLE_STACK_SIZE`]'s reason; must leave room to double the measured high water, which `double_fault_stack` asserts. +/// One size for every IST stack, for [`crate::sched::idle_stack::SIZE`]'s reason; must leave room to double the measured high water, which `double_fault_stack` asserts. const IST_STACK_SIZE: usize = 16384; /// Filled with [`STACK_FILL`], not unmapped: a fault already on IST1 is a triple fault, so detecting after the fact beats trapping it. const IST_GUARD_SIZE: usize = 4096; -/// Chosen so a zeroed or ASCII byte cannot be mistaken for untouched stack. -const STACK_FILL: u8 = 0xA5; -const STACK_FILL_WORD: u64 = u64::from_ne_bytes([STACK_FILL; 8]); /// Allocate and initialize `PerCpu` for a CPU; the pointer lives forever, one `write` of the whole struct so a new field must be given a value here. fn alloc_percpu(cpu_id: u32) -> *mut PerCpu { @@ -439,78 +429,20 @@ pub fn reserve_log_slot( (shard as *const log::Shard, seq, cpu, tid, pid) } -/// One idle stack and the guard page under it. -const IDLE_SLOT: usize = IDLE_GUARD_SIZE + IDLE_STACK_SIZE; - -/// Idle stacks come from their own 2 MiB pages, not the kernel heap: the guard's hole in the direct map would split a heap-shared leaf's TLB entry into 512. -/// Never freed — a leaf returned to the PMM would keep the hole. -static IDLE_STACKS: crate::sync::Lock = crate::sync::Lock::new(IdleArena { - pages: alloc::vec::Vec::new(), - stacks: alloc::vec::Vec::new(), - next: 0, - left: 0, -}); - -struct IdleArena { - pages: alloc::vec::Vec, - /// The bottom of every idle stack, so the deepest any CPU has gone reads from one. - stacks: alloc::vec::Vec, - /// Direct-map address of the next free slot. - next: u64, - left: usize, -} - -/// A 4 KiB-aligned `IDLE_SLOT` from the arena. -fn alloc_idle_slot() -> u64 { - let mut arena = IDLE_STACKS.lock(); - if arena.left < IDLE_SLOT { - let page = crate::mm::pmm::alloc_page(crate::mm::pmm::Category::KernelHeap) - .expect("percpu: no physical page for an idle stack"); - arena.next = page.direct_map().as_mut_ptr::() as u64; - arena.left = crate::mm::PAGE_2M as usize; - arena.pages.push(page); - } - let base = arena.next; - arena.next += IDLE_SLOT as u64; - arena.left -= IDLE_SLOT; - arena.stacks.push(base + IDLE_GUARD_SIZE as u64); - base -} - fn alloc_idle_stack(percpu: &mut PerCpu) { - let base = alloc_idle_slot(); - crate::mm::paging::guard_kernel_page(base); - // SAFETY: exactly `IDLE_STACK_SIZE` bytes above the unmapped guard, within the returned `IDLE_SLOT` — filled, not zeroed, so zero can't mark "untouched" for [`idle_stack_high_water`]. - unsafe { - core::ptr::write_bytes( - (base + IDLE_GUARD_SIZE as u64) as *mut u8, - STACK_FILL, - IDLE_STACK_SIZE, - ) - }; - percpu.idle_stack_top = base + IDLE_SLOT as u64; + percpu.idle_stack_top = crate::sched::idle_stack::alloc(); } /// How big one idle stack is; read by `SYS_DEBUG` for scale. #[cfg(feature = "test-actuators")] pub fn idle_stack_size() -> usize { - IDLE_STACK_SIZE + crate::sched::idle_stack::SIZE } -/// The deepest any CPU's idle stack has ever been, in bytes, read from the bottom up: nothing legitimate writes [`STACK_FILL`], so a touched byte stays changed. +/// The deepest any CPU's idle stack has ever been, in bytes. #[cfg(feature = "test-actuators")] pub fn idle_stack_high_water() -> usize { - let arena = IDLE_STACKS.lock(); - arena - .stacks - .iter() - .map(|&bottom| { - let untouched = - words(bottom, IDLE_STACK_SIZE).take_while(|&w| w == STACK_FILL_WORD).count() * 8; - IDLE_STACK_SIZE - untouched - }) - .max() - .unwrap_or(0) + crate::sched::idle_stack::high_water() } /// One stack per [`IST_STACKS`] row; an `ist[n-1]` left zero faults to address 0 unchecked. @@ -572,11 +504,6 @@ pub fn ist1_report() { }); } -/// Sequential u64s from `base`; every address is inside the caller's already-bounds-checked allocation. -fn words(base: u64, len: usize) -> impl Iterator { - // SAFETY: `i < len/8` bounds each address inside the caller's checked allocation; `read_volatile` keeps the fill-pattern read. - (0..len / 8).map(move |i| unsafe { core::ptr::read_volatile((base as *const u64).add(i)) }) -} /// Initialize per-CPU data for the BSP, and bring this CPU's exception /// handlers up. Call after paging + allocator, before `syscall::init`. @@ -729,7 +656,7 @@ pub fn set_last_armed_ticks(ticks: u32) { /// The last byte of this CPU's idle guard page — the first byte an overflow reaches. #[cfg(feature = "test-actuators")] pub fn idle_guard_byte() -> u64 { - idle_stack_top() - IDLE_STACK_SIZE as u64 - 1 + idle_stack_top() - crate::sched::idle_stack::SIZE as u64 - 1 } /// Top of this CPU's idle stack. diff --git a/kernel/src/drivers/gop.rs b/kernel/src/drivers/gop.rs index 1eb49d2831..1846ba1fda 100644 --- a/kernel/src/drivers/gop.rs +++ b/kernel/src/drivers/gop.rs @@ -25,6 +25,8 @@ impl Gpu for GopGpu { } /// `addr` is the physical address of the framebuffer supplied by firmware. +/// `None` is a scanout no process can be given: one is handed over in 2 MiB +/// pages, and the architecture says whether this one's are its own. pub fn init( addr: u64, size: u64, @@ -32,7 +34,11 @@ pub fn init( height: u32, stride: u32, pixel_format: u32, -) -> (Box, GpuInfo) { +) -> Option<(Box, GpuInfo)> { + if let Err(why) = crate::mm::paging::scanout_whole_pages(addr, size) { + log!("GOP: the scanout at {addr:#x}+{size:#x} is refused: {why}"); + return None; + } // A size smaller than stride*height*4 maps less than the compositor writes. // Boot-time firmware data has no actionable error path, so this panics rather than returning one. let needed = stride as u64 * height as u64 * 4; @@ -92,5 +98,5 @@ pub fn init( flags: 0, }; - (Box::new(GopGpu), info) + Some((Box::new(GopGpu), info)) } diff --git a/kernel/src/drivers/panic_console/mod.rs b/kernel/src/drivers/panic_console/mod.rs index bd46003575..2a9e2cfa29 100644 --- a/kernel/src/drivers/panic_console/mod.rs +++ b/kernel/src/drivers/panic_console/mod.rs @@ -24,7 +24,7 @@ use crate::log; use crate::panic_reboot::Bound; use crate::time::{Budget, Cadence, Duration}; use crate::mm::policy::MmioPolicy; -use crate::mm::{self, DirectMap, align_2m}; +use crate::mm::{self, DirectMap}; /// 1 bpp 8x16, codepoints 0x20..=0x7E, one byte per row, bit 7 leftmost. /// `tests/common/screen.rs` decodes this same file, so decoder and renderer cannot drift. @@ -495,6 +495,13 @@ pub fn arm(args: &KernelArgs, maps: &[MemoryMapEntry]) { } } +/// The scanout the panel paints, `(phys, bytes)`, while it is armed: what a +/// direct map that holds only memory maps before the boot's next record paints. +pub fn scanout() -> Option<(u64, u64)> { + let phys = RAW_PHYS.load(Ordering::Relaxed); + (phys != 0).then(|| (phys, RAW_SIZE.load(Ordering::Relaxed))) +} + /// Re-establish the mapping after `mm::init` replaces the bootloader's page /// tables. pub fn remap() { @@ -503,7 +510,7 @@ pub fn remap() { return; } let size = RAW_SIZE.load(Ordering::Relaxed); - mm::paging::map_mmio(phys, align_2m(size as usize) as u64, MmioPolicy::WriteCombining); + mm::paging::map_mmio(phys, size, MmioPolicy::WriteCombining); rearm(); } diff --git a/kernel/src/drivers/pci.rs b/kernel/src/drivers/pci.rs index 5e2737de80..14b18a5410 100644 --- a/kernel/src/drivers/pci.rs +++ b/kernel/src/drivers/pci.rs @@ -375,7 +375,13 @@ impl PciDevice { let dest = if crate::actuator::iommu_dest_apic1() { 1 } else { MSG_DEST }; match crate::iommu::remap_msi(self.bus, self.dev, self.func, vector, dest) { // The same CPU the remappable one would name. - crate::iommu::Delivery::Direct => Some(crate::arch::msi_message(dest, vector)), + crate::iommu::Delivery::Direct => match crate::arch::msi_message(dest, vector) { + Ok(message) => Some(message), + Err(why) => { + log!("PCI {:02x}:{:02x}.{}: not armed — {why}", self.bus, self.dev, self.func); + None + } + }, crate::iommu::Delivery::Remapped(m) => Some((m.address, m.data)), crate::iommu::Delivery::Refused(why) => { log!("PCI {:02x}:{:02x}.{}: not armed — {why}", self.bus, self.dev, self.func); diff --git a/kernel/src/drivers/xhci/wait/msc.rs b/kernel/src/drivers/xhci/wait/msc.rs index 8f1fe5421b..dfa8afb908 100644 --- a/kernel/src/drivers/xhci/wait/msc.rs +++ b/kernel/src/drivers/xhci/wait/msc.rs @@ -889,7 +889,7 @@ fn block_witness_holds(dev: &MscDevice, entered: BlockWitness) { "USB BOT WITNESS: MscDevice::block changed inside one round trip — the field at \ {at:#018x} held {:#018x} and now holds {:#018x} (the frame moved by {}). This CPU's \ Ring 3 entry stack is {top:#018x}, so the field stands {} bytes below it and the \ - running rsp is {:#018x}. `with_storage` copies the device onto this stack, so a \ + running stack pointer is {:#018x}. `with_storage` copies the device onto this stack, so a \ kernel text value here is a return address something else pushed.", entered.was, dev.block, diff --git a/kernel/src/hardlockup/mod.rs b/kernel/src/hardlockup/mod.rs index 97cfca1e71..33b88039a1 100644 --- a/kernel/src/hardlockup/mod.rs +++ b/kernel/src/hardlockup/mod.rs @@ -25,7 +25,7 @@ //! that has not moved is a CPU that has taken nothing at all. A CPU whose count //! is stale for [`toyos_tco::hard_lockup_bound_ms`] of //! the bound this boot named, *and* whose sampled frame has `IF` clear, is -//! stuck: it seals a `WEDGED` record naming itself, its `rip` and `rsp` from the +//! stuck: it seals a `WEDGED` record naming itself, its `pc` and `sp` from the //! NMI frame, the lock it is spinning on if `Lock::lock` recorded one, a line //! for every other CPU, and the tail of the log ring — then writes the reset //! register through `acpi::reset_now`. @@ -107,9 +107,9 @@ static STOOD_DOWN: AtomicBool = AtomicBool::new(false); static PROGRESS: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; static STILL_SINCE: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; static AT_TSC: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; -static AT_RIP: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; -static AT_RSP: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; -static AT_RFLAGS: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; +static AT_PC: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; +static AT_SP: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; +static AT_FLAGS: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; static ARMED_PMU: [AtomicBool; MAX_CPUS] = [const { AtomicBool::new(false) }; MAX_CPUS]; /// The lock a CPU is spinning on and the `#[track_caller]` site that asked for @@ -225,7 +225,7 @@ pub fn bound_ms() -> u64 { /// Called from `arch::trap::nmi`'s `note` and nowhere else. Returns on every NMI /// that is not this CPU's own overflow, so the diagnostic senders — the blocked /// task dump's probe, the syscall-window storm — cost one load and one compare. -pub fn sample(rip: u64, rsp: u64, rflags: u64) { +pub fn sample(pc: u64, sp: u64, flags: u64) { if BOUND_TSC.load(Relaxed) == 0 || STOOD_DOWN.load(Relaxed) { return; } @@ -241,9 +241,9 @@ pub fn sample(rip: u64, rsp: u64, rflags: u64) { return; } let now = cpu::counter(); - AT_RIP[me].store(rip, Relaxed); - AT_RSP[me].store(rsp, Relaxed); - AT_RFLAGS[me].store(rflags, Relaxed); + AT_PC[me].store(pc, Relaxed); + AT_SP[me].store(sp, Relaxed); + AT_FLAGS[me].store(flags, Relaxed); AT_TSC[me].store(now, Relaxed); let taken = crate::irq_census::taken_here(); @@ -251,10 +251,10 @@ pub fn sample(rip: u64, rsp: u64, rflags: u64) { // `IF` set is the whole difference between this bound and the deadline's: a // CPU that can still take an interrupt is one the timer entry's poll // reaches, and this mechanism is not about it. - if moved || trap::frame_interrupts_enabled(rflags) { + if moved || trap::frame_interrupts_enabled(flags) { STILL_SINCE[me].store(now, Relaxed); } else if now.wrapping_sub(STILL_SINCE[me].load(Relaxed)) >= BOUND_TSC.load(Relaxed) { - locked_up(me, rip, rsp, now) + locked_up(me, pc, sp, now) } // **After the decision and never before it.** Re-arming clears the mask // hardware set on delivery; leaving it set is what stops a second NMI @@ -269,11 +269,11 @@ pub fn sample(rip: u64, rsp: u64, rflags: u64) { /// Seal where every CPU is and hand the machine back. /// /// Runs on the stuck CPU itself, from its own NMI frame, which is the only -/// context that has its `rip`. No CPU is asked anything from here: an NMI sent +/// context that has its `pc`. No CPU is asked anything from here: an NMI sent /// to a sibling already inside its own would enter `nested_nmi` and stop the /// machine instead of resetting it, so the record is built out of what each CPU /// last wrote about itself. -fn locked_up(me: usize, rip: u64, rsp: u64, now: u64) -> ! { +fn locked_up(me: usize, pc: u64, sp: u64, now: u64) -> ! { if !crate::deadline::claim_the_reset() { // Another CPU is already sealing and resetting. This one has nothing to // add and must not race it into the page. @@ -281,7 +281,7 @@ fn locked_up(me: usize, rip: u64, rsp: u64, now: u64) -> ! { } crate::drivers::panic_console::seal_wedge(format_args!( "{}", - Report { me, rip, rsp, now, cpus: (smp::cpu_count() as usize).min(MAX_CPUS) } + Report { me, pc, sp, now, cpus: (smp::cpu_count() as usize).min(MAX_CPUS) } )); // The seal first, because the USB stop `reset_now` makes before it writes // the register is bounded but not instant, and this record is the @@ -370,7 +370,7 @@ impl fmt::Display for Waiting { } } -/// Where a `rip` is, spelled without saying a word: `symbols::resolve_kernel` +/// Where a `pc` is, spelled without saying a word: `symbols::resolve_kernel` /// writes a log record, which is the one thing this path may not do. struct At(u64); @@ -391,15 +391,15 @@ impl fmt::Display for At { /// What the machine looked like from the CPU that ended it. struct Report { me: usize, - rip: u64, - rsp: u64, + pc: u64, + sp: u64, now: u64, cpus: usize, } impl fmt::Display for Report { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - let Self { me, rip, rsp, now, cpus } = *self; + let Self { me, pc, sp, now, cpus } = *self; writeln!( f, "{LOCKED_UP}: cpu{me} has taken no interrupt for {}, with `IF` clear at every sample \ @@ -407,8 +407,8 @@ impl fmt::Display for Report { Ms(now.wrapping_sub(STILL_SINCE[me].load(Relaxed))), BOUND_MS.load(Relaxed), )?; - writeln!(f, " rip={}", At(rip))?; - writeln!(f, " rsp={rsp:#018x}{}", Waiting(me))?; + writeln!(f, " pc={}", At(pc))?; + writeln!(f, " sp={sp:#018x}{}", Waiting(me))?; // Every CPU, and each line is what that CPU last wrote about itself: // the holder of whatever the stuck one is waiting for is somewhere in // this list, and nothing else in a wedged machine can point at it. @@ -432,8 +432,8 @@ impl fmt::Display for Report { " cpu{cpu} irqs={irqs} (={} when sampled {} ago) if={} at {}{}", PROGRESS[cpu].load(Relaxed), Ms(now.wrapping_sub(at)), - u8::from(trap::frame_interrupts_enabled(AT_RFLAGS[cpu].load(Relaxed))), - At(AT_RIP[cpu].load(Relaxed)), + u8::from(trap::frame_interrupts_enabled(AT_FLAGS[cpu].load(Relaxed))), + At(AT_PC[cpu].load(Relaxed)), Waiting(cpu), )?; } diff --git a/kernel/src/hardlockup/probe.rs b/kernel/src/hardlockup/probe.rs index 87e6b23721..c68a6204df 100644 --- a/kernel/src/hardlockup/probe.rs +++ b/kernel/src/hardlockup/probe.rs @@ -139,7 +139,7 @@ fn go_deaf(me: usize, bound_ms: u64) -> ! { while cpu::counter() < until { core::hint::spin_loop(); } - // Never acquires. The detector's NMI is what ends this CPU, and its `rip` + // Never acquires. The detector's NMI is what ends this CPU, and its `pc` // is inside `Lock::lock`'s spin when it does. let _never = PROBE_LOCK.lock(); loop { diff --git a/kernel/src/loader/mod.rs b/kernel/src/loader/mod.rs index ddd821de37..3203ad6200 100644 --- a/kernel/src/loader/mod.rs +++ b/kernel/src/loader/mod.rs @@ -547,7 +547,7 @@ pub fn spawn( }; log!("spawn: TLS {} modules, total_memsz={}", tls_modules.len(), tls.total_memsz()); - let Some((tls_pages, fs_base, _)) = + let Some((tls_pages, thread_pointer, _)) = tls::TlsBlock::build(&tls_modules, tls).and_then(|b| b.publish(&child_pt)) else { log!("spawn: {}: failed to allocate TLS ({} bytes)", path, tls.total_memsz()); @@ -566,7 +566,7 @@ pub fn spawn( ); let sym_bytes = syms.resident_bytes(); - let (ks_alloc, ks_rsp) = match alloc_kernel_stack(process_start, entry, sp, 0) { + let (ks_alloc, ks_sp) = match alloc_kernel_stack(process_start, entry, sp, 0) { Some(ks) => ks, None => { log!("spawn: {}: failed to allocate kernel stack", path); @@ -642,9 +642,9 @@ pub fn spawn( let (sched, dst) = scheduler::enqueue_new( scheduler::TaskId(pid, tid), ks_alloc, - ks_rsp, + ks_sp, child_pt.clone(), - fs_base, + thread_pointer, syms, ); table.get_mut(pid).unwrap().threads_mut().get_mut(tid).unwrap().set_sched(sched); diff --git a/kernel/src/loader/tls.rs b/kernel/src/loader/tls.rs index 466738d9af..75fb1a07cf 100644 --- a/kernel/src/loader/tls.rs +++ b/kernel/src/loader/tls.rs @@ -57,8 +57,8 @@ impl TlsBlock { // by no mapping until `publish` maps it after this returns. unsafe { rebase(frames, tp_offset, at) } })?; - let fs_base = (pages.vaddr() + tp_offset as u64).raw(); - Some((pages, fs_base, tp_offset)) + let thread_pointer = (pages.vaddr() + tp_offset as u64).raw(); + Some((pages, thread_pointer, tp_offset)) } } diff --git a/kernel/src/main.rs b/kernel/src/main.rs index ef64ca60a9..237682042b 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -609,15 +609,17 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { register_gpu(gpu_driver, gpu_info); } else if kernel_args.gop_framebuffer != 0 { log!("GPU: using UEFI GOP"); - let (gpu_driver, gpu_info) = gop::init( + match gop::init( kernel_args.gop_framebuffer, kernel_args.gop_framebuffer_size, kernel_args.gop_width, kernel_args.gop_height, kernel_args.gop_stride, kernel_args.gop_pixel_format, - ); - register_gpu(gpu_driver, gpu_info); + ) { + Some((gpu_driver, gpu_info)) => register_gpu(gpu_driver, gpu_info), + None => log!("GPU: none this boot, running headless"), + } } else { log!("GPU: none found, running headless"); } diff --git a/kernel/src/mm/mod.rs b/kernel/src/mm/mod.rs index 304e0dd18b..609b9fd3ae 100644 --- a/kernel/src/mm/mod.rs +++ b/kernel/src/mm/mod.rs @@ -159,7 +159,7 @@ impl core::fmt::Debug for DirectMap { pub fn init(memory_map: &[MemoryMapEntry], reserved: &[Region]) { alloc::init_early(); pmm::init(memory_map, reserved); - DIRECT_MAP_END.set(paging::init(memory_map)); + DIRECT_MAP_END.set(paging::init(memory_map, crate::drivers::panic_console::scanout())); alloc::init(); paging::seal_kernel_half(); } diff --git a/kernel/src/process.rs b/kernel/src/process.rs index f8680f7be2..36a84b042e 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -842,7 +842,7 @@ pub fn spawn_thread(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> Op // block's address is chosen from, so every pointer in it is final before // the mapping exists. let block = TlsBlock::build(&tls_modules, tls)?; - let (tls_alloc, fs_base, tp_offset) = { + let (tls_alloc, thread_pointer, tp_offset) = { let parent_data = process_data_arc.lock(); if crate::actuator::tls_rebase_window() { crate::loader::rebase_window::spawning(arg); @@ -854,7 +854,7 @@ pub fn spawn_thread(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> Op }; let tls_alloc_tcb = tls_alloc.ptr().wrapping_add(tp_offset); - let (ks_alloc, ks_rsp) = match alloc_kernel_stack(thread_start, entry, stack_ptr, arg) { + let (ks_alloc, ks_sp) = match alloc_kernel_stack(thread_start, entry, stack_ptr, arg) { Some(ks) => ks, None => { tls_alloc.release(&parent_addr_space); @@ -905,9 +905,9 @@ pub fn spawn_thread(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> Op let (sched, _dst) = scheduler::enqueue_new( TaskId(parent_process, tid), ks_alloc, - ks_rsp, + ks_sp, parent_addr_space, - fs_base, + thread_pointer, symbols, ); proc.threads.get_mut(tid).unwrap().set_sched(sched); @@ -1474,24 +1474,23 @@ pub fn dump_crash_diagnostics(fault_addr: u64, rip: u64) { } dump_region("rip", rip); - let fs_base = crate::arch::cpu::thread_pointer(); - if fs_base != 0 { - log!(" FS base: {:#x}", fs_base); - if let Some(self_ptr) = read_user(fs_base) { - log!(" fs:[0] = {:#x} (expected {:#x})", self_ptr, fs_base); + let tp = crate::arch::cpu::thread_pointer(); + if tp != 0 { + log!(" Thread pointer: {:#x}", tp); + if read_user(tp).is_some() { for i in 0..8u64 { - let addr = fs_base + i * 8; + let addr = tp + i * 8; let Some(val) = read_user(addr) else { break }; log!(" TP+{:#x} = {:#018x}", i * 8, val); } log!(" TLS data before TP:"); for i in 1..=4u64 { - let addr = fs_base - i * 8; + let addr = tp - i * 8; let Some(val) = read_user(addr) else { break }; log!(" TP-{:#x} = {:#018x}", i * 8, val); } } else { - log!(" FS base {:#x} NOT MAPPED!", fs_base); + log!(" Thread pointer {:#x} NOT MAPPED!", tp); } } } diff --git a/kernel/src/sched/driver.rs b/kernel/src/sched/driver.rs index 43233c1261..9cf0038b5e 100644 --- a/kernel/src/sched/driver.rs +++ b/kernel/src/sched/driver.rs @@ -374,9 +374,9 @@ pub fn init() { /// The context a CPU runs on when idle — never a dead task's stack, so a pass can free the previous zombie. fn idle_ctx() -> KernelCtx { KernelCtx { - rsp: 0, + sp: 0, root: crate::mm::paging::kernel_root(), - fs_base: 0, + thread_pointer: 0, kernel_stack_top: 0, id: None, // Never read: the idle loop is entered by jump, not switch, and @@ -394,15 +394,15 @@ fn placement(now: Nanos) -> CpuId { cpus().place(start, now) } -/// Everything a new thread needs. `entry_rsp` points at the trampoline frame `alloc_kernel_stack` built; +/// Everything a new thread needs. `entry_sp` points at the trampoline frame `alloc_kernel_stack` built; /// `address_space` is not `Option` — every kernel thread uses the kernel address space, so one declaration /// decides `cr3`. pub struct NewTask { pub id: TaskId, pub kernel_stack: OwnedAlloc, - pub entry_rsp: u64, + pub entry_sp: u64, pub address_space: PageTables, - pub fs_base: u64, + pub thread_pointer: u64, pub share: Arc, /// The process's symbol table; a kernel thread names an empty one. pub symbols: Arc, @@ -416,9 +416,9 @@ pub fn spawn(new: NewTask) -> (ThreadSched, CpuId) { let root = new.address_space.lock().root(); let kernel_stack_top = new.kernel_stack.ptr() as u64 + KERNEL_STACK_SIZE as u64; let ctx = KernelCtx { - rsp: new.entry_rsp, + sp: new.entry_sp, root, - fs_base: new.fs_base, + thread_pointer: new.thread_pointer, kernel_stack_top, id: Some(new.id), // The one level `trampoline_entry` discharges before the first `iretq`. @@ -939,30 +939,30 @@ fn check_stack_canary(payload: &KernelPayload) { /// Does every Ring 3 → Ring 0 entry land on the stack of the task this CPU is running, and is this CPU standing on it? /// -/// A stray `kernel_rsp` or `tss.rsp0` aims a future entry at a stack it did not grow; this catches it before -/// that entry, not after. +/// A stray entry stack aims a future entry at a stack it did not grow; this catches it before that entry, not +/// after. #[cfg(feature = "stack-witness")] fn check_stack_ownership(payload: &KernelPayload) { let bottom = payload.kernel_stack.ptr() as u64; let top = bottom + KERNEL_STACK_SIZE as u64; // SAFETY: a pass runs on the CPU whose GS base is its own `PerCpu`. - let (kernel_rsp, rsp0) = unsafe { percpu::entry_stacks() }; - let rsp = crate::arch::cpu::stack_pointer(); - if kernel_rsp == top && rsp0 == top && rsp <= top && rsp > bottom { + let (entry, interrupt) = unsafe { percpu::entry_stacks() }; + let sp = crate::arch::cpu::stack_pointer(); + if entry == top && interrupt == top && sp <= top && sp > bottom { return; } panic!( "STACK WITNESS: cpu{} is passing on tid={} whose stack is \ - [{bottom:#018x}, {top:#018x}) — kernel_rsp={kernel_rsp:#018x} \ - (off by {}), tss.rsp0={rsp0:#018x} (off by {}), rsp={rsp:#018x} \ - ({} bytes below the top). A Ring 3 entry takes its stack from one of \ + [{bottom:#018x}, {top:#018x}) — the syscall entry's stack {entry:#018x} \ + (off by {}), the interrupt entry's {interrupt:#018x} (off by {}), sp={sp:#018x} \ + ({} bytes below the top). An entry from user mode takes its stack from one of \ those two words, so one that is not this task's top aims the next \ entry's return addresses into memory another execution owns.", percpu::cpu_id(), payload.id.1, - kernel_rsp.wrapping_sub(top) as i64, - rsp0.wrapping_sub(top) as i64, - top.wrapping_sub(rsp) as i64, + entry.wrapping_sub(top) as i64, + interrupt.wrapping_sub(top) as i64, + top.wrapping_sub(sp) as i64, ); } diff --git a/kernel/src/sched/idle_stack.rs b/kernel/src/sched/idle_stack.rs new file mode 100644 index 0000000000..b108143034 --- /dev/null +++ b/kernel/src/sched/idle_stack.rs @@ -0,0 +1,88 @@ +//! Every CPU's idle stack: one per CPU, the size of a task's kernel stack, above +//! a guard page the direct map does not hold, so running off its bottom is a +//! fault rather than a write into memory another execution owns. +//! +//! **Taken from 2 MiB pages of their own, not the kernel heap**: the guard's +//! hole in the direct map splits the page that holds it, and a heap page split +//! so would cost every other allocation in it its 2 MiB translation. Never +//! freed — a page returned to the PMM would keep the hole. + +use alloc::vec::Vec; + +use crate::mm::DirectMap; +use crate::sync::Lock; + +/// Same size as a task's kernel stack: a `deferred` [`kobject!`] object may +/// own an `immediate` one, whose destructor then runs here instead. +pub const SIZE: usize = crate::process::KERNEL_STACK_SIZE; + +/// One unmapped 4 KiB page below every idle stack. +const GUARD: usize = crate::mm::PAGE_BYTES; + +/// What an untouched byte of a filled stack holds: chosen so a zeroed or ASCII +/// byte cannot be mistaken for one. +pub const FILL: u8 = 0xA5; +pub const FILL_WORD: u64 = u64::from_ne_bytes([FILL; 8]); + +/// One idle stack and the guard page under it. +const SLOT: usize = GUARD + SIZE; + +static ARENA: Lock = Lock::new(Arena { pages: Vec::new(), stacks: Vec::new(), next: 0, left: 0 }); + +struct Arena { + pages: Vec, + /// The bottom of every idle stack, so the deepest any CPU has gone reads from one. + stacks: Vec, + /// Direct-map address of the next free slot. + next: u64, + left: usize, +} + +/// A 4 KiB-aligned [`SLOT`] from the arena. +fn alloc_slot() -> u64 { + let mut arena = ARENA.lock(); + if arena.left < SLOT { + let page = crate::mm::pmm::alloc_page(crate::mm::pmm::Category::KernelHeap) + .expect("idle stack: no physical page for one"); + arena.next = page.direct_map().as_mut_ptr::() as u64; + arena.left = crate::mm::PAGE_2M as usize; + arena.pages.push(page); + } + let base = arena.next; + arena.next += SLOT as u64; + arena.left -= SLOT; + arena.stacks.push(base + GUARD as u64); + base +} + +/// A fresh idle stack, filled, over its unmapped guard page: its top. +pub fn alloc() -> u64 { + let base = alloc_slot(); + crate::mm::paging::kernel().lock().guard_4k(DirectMap::phys_of(base as *const u8)); + // SAFETY: exactly `SIZE` bytes above the unmapped guard, within the slot + // `alloc_slot` returned — filled, not zeroed, so zero can't mark + // "untouched" for [`high_water`]. + unsafe { core::ptr::write_bytes((base + GUARD as u64) as *mut u8, FILL, SIZE) }; + base + SLOT as u64 +} + +/// The deepest any CPU's idle stack has ever been, in bytes, read from the +/// bottom up: nothing legitimate writes [`FILL`], so a touched byte stays changed. +#[cfg(feature = "test-actuators")] +pub fn high_water() -> usize { + let arena = ARENA.lock(); + arena + .stacks + .iter() + .map(|&bottom| SIZE - words(bottom, SIZE).take_while(|&w| w == FILL_WORD).count() * 8) + .max() + .unwrap_or(0) +} + +/// Sequential u64s from `base`; every address is inside the caller's +/// already-bounds-checked allocation. +pub fn words(base: u64, len: usize) -> impl Iterator { + // SAFETY: `i < len/8` bounds each address inside the caller's checked + // allocation; `read_volatile` keeps the fill-pattern read. + (0..len / 8).map(move |i| unsafe { core::ptr::read_volatile((base as *const u64).add(i)) }) +} diff --git a/kernel/src/sched/kthread.rs b/kernel/src/sched/kthread.rs index decb64376a..2e3ee9543a 100644 --- a/kernel/src/sched/kthread.rs +++ b/kernel/src/sched/kthread.rs @@ -94,7 +94,7 @@ pub fn open_selftest() { /// Start a kernel thread running `body(arg)` on its own kernel stack and return its scheduler faces. pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64) -> ThreadSched { - let (stack, entry_rsp) = crate::loader::alloc_kernel_stack( + let (stack, entry_sp) = crate::loader::alloc_kernel_stack( crate::loader::kernel_start, body as usize as u64, 0, @@ -130,7 +130,7 @@ pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64) -> ThreadSched let (sched, _dst) = scheduler::enqueue_new( TaskId(pid, tid), stack, - entry_rsp, + entry_sp, crate::mm::paging::kernel().clone(), 0, syms, diff --git a/kernel/src/sched/mod.rs b/kernel/src/sched/mod.rs index 0a7a33898b..71df61c919 100644 --- a/kernel/src/sched/mod.rs +++ b/kernel/src/sched/mod.rs @@ -9,6 +9,7 @@ pub mod kthread; pub mod payload; pub mod reap_gate; pub mod futex; +pub mod idle_stack; /// Ceiling on CPUs the percpu arrays are sized for. pub const MAX_CPUS: usize = 8; diff --git a/kernel/src/sched/payload.rs b/kernel/src/sched/payload.rs index 0e2aa7dfbc..523e2a6ede 100644 --- a/kernel/src/sched/payload.rs +++ b/kernel/src/sched/payload.rs @@ -40,12 +40,13 @@ pub type KShare = FairShare>; /// The core's wait ticket; blocking sites use `driver::Ticket`, which wraps it in the needed preempt guard. pub type RawTicket = WaitTicket; -/// The saved callee context; everything `Hw::switch` must load without dereferencing anything else. +/// The saved callee context; everything `Hw::switch` must load without dereferencing anything else, named +/// by role because every architecture's switch loads it. pub struct KernelCtx { /// Saved kernel stack pointer, written by the `context_switch` asm. - pub rsp: u64, + pub sp: u64, pub root: Root, - pub fs_base: u64, + pub thread_pointer: u64, pub kernel_stack_top: u64, /// `None` is this CPU's idle context. pub id: Option, diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index 14b41910db..a7fb3c3e4c 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -293,17 +293,17 @@ pub fn global_min_vruntime() -> u64 { pub fn enqueue_new( id: TaskId, kernel_stack: crate::process::OwnedAlloc, - entry_rsp: u64, + entry_sp: u64, address_space: crate::process::PageTables, - fs_base: u64, + thread_pointer: u64, symbols: alloc::sync::Arc, ) -> (ThreadSched, CpuId) { driver::spawn(NewTask { id, kernel_stack, - entry_rsp, + entry_sp, address_space, - fs_base, + thread_pointer, share: share_for(id.0), symbols, }) @@ -348,7 +348,7 @@ pub fn may_yield() -> bool { crate::preempt::count() == blocking_baseline() } -/// Unified preempt entry: the Ring 3 timer path, `kernel_exit_to_user_check` +/// Unified preempt entry: the Ring 3 timer path, [`exit_to_user`] /// and the `preempt::enable` slow path all funnel through here. #[track_caller] pub fn do_preempt() { @@ -368,7 +368,7 @@ pub fn do_preempt() { } /// The last thing a thread does before returning to Ring 3, if either mark it -/// can carry says it never does. `kernel_exit_to_user_check` is the one caller; +/// can carry says it never does. [`exit_to_user`] is the one caller; /// `kernel/src/quiesce.rs`'s header says why that boundary is the safe point. /// /// **One call and one match, so the two marks have no order to disagree @@ -398,6 +398,41 @@ pub fn leave_ring3_if_due() { } } + +/// The deferred-preempt epilogue every return to user mode runs last, with +/// interrupts masked on entry and on return: a killed or stopped thread leaves +/// here, and a reschedule owed since the entry is served before the thread +/// sees user mode again. +pub fn exit_to_user() { + flush_ring0_timer_fires_to_trace(); + loop { + // A killed or stopped thread returns to Ring 3 exactly once more: never. + leave_ring3_if_due(); + // `do_preempt` owns clearing `need_resched`; this function never clears it itself. + if !crate::preempt::need_resched() { + #[cfg(feature = "boot-actuators")] + if crate::actuator::dump_in_blocking_pass() { + crate::sched::dump::staged::note_return_to_ring3(); + } + return; + } + assert!(!in_schedule_self(), "exit-to-user inside a scheduler pass"); + // Not an IrqGuard: both loop exits must set IF, not restore a saved value. + crate::arch::cpu::enable_interrupts(); + do_preempt(); + crate::arch::cpu::disable_interrupts(); + flush_ring0_timer_fires_to_trace(); + } +} + +fn flush_ring0_timer_fires_to_trace() { + let cur = percpu::ring0_timer_fires(); + let missed = cur.wrapping_sub(percpu::last_seen_ring0_fires()); + if missed > 0 { + crate::trace::trace(crate::trace::Kind::TimerFireBurst, missed); + percpu::set_last_seen_ring0_fires(cur); + } +} /// The exit pass of a thread that has left its process (`process::leave`). #[track_caller] pub fn exit_current() -> ! { diff --git a/kernel/src/vma.rs b/kernel/src/vma.rs index 38903b6b58..f83f2c7319 100644 --- a/kernel/src/vma.rs +++ b/kernel/src/vma.rs @@ -1,9 +1,10 @@ +use alloc::collections::BTreeMap; use alloc::sync::Arc; use crate::file_backing::FileBacking; use crate::mm::policy::Prot; -use crate::mm::PAGE_2M; -use toyos_userbound::Window; +use crate::mm::{UserAddr, PAGE_2M}; +use toyos_userbound::{PageSpan, Window}; /// The stack extends upward to the PIE base, so no usable VA space exists above it. pub const ALLOC_CEILING: u64 = STACK_BASE; @@ -61,3 +62,78 @@ pub enum Occupancy { /// Part of a region, several regions, or one that merely starts here. Partial, } + +/// Every region an address space registers, keyed by start: where a new one +/// goes and what an address falls in. The page tables are the architecture's +/// (`mm::paging::AddressSpace`), and this is the half both share. +#[derive(Default)] +pub struct Regions(BTreeMap); + +impl Regions { + /// Where `span` goes, top-down and never below the floor: a region the + /// kernel placed under it, the clock page, bounds no gap. + fn find_gap(&self, span: PageSpan) -> Option { + let taken = self.0.iter().rev().map(|(start, region)| (start.raw(), region.size)); + window().gap(span, taken).map(UserAddr::new) + } + + /// Allocate a virtual address range and register the region. `size` is + /// made a [`PageSpan`] before anything is summed on it. + pub fn alloc(&mut self, size: u64, kind: RegionKind) -> Option { + let span = window().span(size)?; + let addr = self.find_gap(span)?; + self.0.insert(addr, Region { size: span.bytes(), kind }); + Some(addr) + } + + /// A [`RegionKind::Mapped`] region for `size` bytes: its address and its + /// size in whole pages, which the caller maps. + pub fn alloc_mapped(&mut self, size: u64) -> Option<(UserAddr, u64)> { + let span = window().span(size)?; + let addr = self.find_gap(span)?; + let aligned = span.bytes(); + self.0.insert(addr, Region { size: aligned, kind: RegionKind::Mapped }); + Some((addr, aligned)) + } + + /// Unregister the region at `addr`, answering its size. + pub fn remove(&mut self, addr: UserAddr) -> Option { + Some(self.0.remove(&addr)?.size) + } + + /// Insert a region at a specific address (for ELF segments, stack, etc.) + pub fn insert(&mut self, addr: UserAddr, region: Region) { + assert!(self.find(addr).is_none(), "insert_region: address {:#x} already occupied", addr.raw()); + self.0.insert(addr, region); + } + + /// Find the region containing `addr`. Returns (start_addr, region). + pub fn find(&self, addr: UserAddr) -> Option<(UserAddr, &Region)> { + let (&start, region) = self.0.range(..=addr).next_back()?; + if addr.raw() < start.raw() + region.size { + Some((start, region)) + } else { + None + } + } + + /// The end is saturating so a caller's arithmetic cannot wrap into a smaller range. + pub fn occupancy(&self, addr: UserAddr, size: u64) -> Occupancy { + let end = UserAddr::new(addr.raw().saturating_add(size)); + let mut over = self.overlapping(addr, end); + let Some((&start, region)) = over.next() else { + return Occupancy::Free; + }; + if over.next().is_none() && start == addr && region.size == size { + Occupancy::Whole + } else { + Occupancy::Partial + } + } + + /// Iterate all regions that overlap the range [start, end). + pub fn overlapping(&self, start: UserAddr, end: UserAddr) -> impl Iterator { + // Overlaps [start, end) iff s < end && s+n > start; `range(..end)` prunes the first half. + self.0.range(..end).filter(move |(&s, r)| s.raw() + r.size > start.raw()) + } +} diff --git a/src/build.rs b/src/build.rs index c5b98186a8..6bff632b6b 100644 --- a/src/build.rs +++ b/src/build.rs @@ -3142,6 +3142,7 @@ mod tests { "tests/testcases/system.toml", "tests/toolkitcase/system.toml", "tests/updatecase/system.toml", + "tests/virtpreemptcase/system.toml", ]; fn load(cfg: &str) -> SystemConfig { diff --git a/src/metal.rs b/src/metal.rs index e1f7d6d4bd..277180fc59 100644 --- a/src/metal.rs +++ b/src/metal.rs @@ -3350,7 +3350,7 @@ mod tests { let locked = format!( "Previous boot's panic: the last boot read WEDGED\n| {}: cpu7 has taken no interrupt \ for 60004 ms, with `IF` clear at every sample in that span. Its bound is 60000 ms.\n\ - | rip=0xffff800060a5903f >::lock+0x12f\n", + | pc=0xffff800060a5903f >::lock+0x12f\n", bootlog::LOCKED_UP ); assert_eq!(wedged_boot(&locked, booted), Ok(1200)); diff --git a/src/sourcegate.rs b/src/sourcegate.rs index 21e9f5e583..3ae72eb2d3 100644 --- a/src/sourcegate.rs +++ b/src/sourcegate.rs @@ -205,9 +205,9 @@ const BANS: &[Ban] = &[ allowed: &[ // Cache lines to walk, not bytes. ("kernel/src/arch/x86_64/control_regs.rs", 1), - // Two guard-page sizes: the mapping is 4 KiB because a guard is one - // hardware page, and `PAGE_SIZE` is 2 MiB territory here. - ("kernel/src/arch/x86_64/percpu.rs", 2), + // A guard page's size: a guard is one hardware page, and + // `PAGE_SIZE` is 2 MiB territory here. + ("kernel/src/arch/x86_64/percpu.rs", 1), // A device's TX buffer. ("kernel/src/drivers/virtio_console.rs", 1), // A VT-d table is 4 KiB by the specification, not by this kernel. diff --git a/tests/toyos.rs b/tests/toyos.rs index ff90ef5aae..c789724e94 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -554,6 +554,9 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ ("virt_early_panic", Sched::Parallel, Tier::Local), ("virt_early_fault", Sched::Parallel, Tier::Local), ("virt_el2_drop", Sched::Parallel, Tier::Local), + ("virt_user_mode", Sched::Parallel, Tier::Local), + ("virt_timer_preempts", Sched::Parallel, Tier::Local), + ("virt_irq_storm", Sched::Parallel, Tier::Local), ]; /// What `screen_console_shell` types, and what it then looks for on its own. @@ -5926,6 +5929,112 @@ fn run_screen_test( } Ok(()) } + "virt_user_mode" => { + // The port's stage 4 on one CPU, under the EL2 profile whose + // entry also writes what the drop leaves EL2 holding: the kernel's + // own tables, the GIC and the timer, and a process at EL0 — init, + // whose every page arrives by a demand fault and whose spawn of + // `logd` is a syscall the kernel answered. Emulated, and not under + // HVF, which exposes no RNDR for the kernel's hash seed. + let mut qemu = QemuInstance::boot_with_options( + test_config, + &[], + &[], + BootOptions { + profile: qemu::Profile::VirtEl2, + ready_marker: "control registers: SCTLR_EL1=", + ..Default::default() + }, + ); + const SPAWNED: &str = "spawn: /system/bin/logd pid="; + let rest = qemu.drain_until(Duration::from_secs(180), |l| l.contains(SPAWNED)); + let serial = format!("{}\n{rest}", qemu.boot_log()); + for want in [ + "paging: the direct map holds memory below", + "percpu: BSP cpu_id=0", + "GIC: v", + "clock: the generic timer counts at", + "spawned /system/bin/init pid=", + SPAWNED, + ] { + if !serial.contains(want) { + return Err(format!("{want:?} not on the PL011\nserial:\n{serial}")); + } + } + Ok(()) + } + "virt_timer_preempts" => { + // After its first line `spin` makes no syscall, yet on this + // machine's one CPU the runner's deadline thread wakes — at a + // tick, the only thing that can take the CPU from `spin` — and + // ends it. The CPU time `spin` had by then says it held the CPU. + let config = compile::repo_root().join("tests/virtpreemptcase/system.toml"); + let case = config.parent().expect("system.toml has a directory"); + let mut qemu = QemuInstance::boot_with_options( + case, + &[], + &[], + BootOptions { + profile: qemu::Profile::VirtEl2, + ready_marker: "control registers: SCTLR_EL1=", + ..Default::default() + }, + ); + let exited = format!("{}spin pid=", bootlog::EXIT); + let rest = qemu.drain_until(Duration::from_secs(300), |l| l.contains(&exited)); + let serial = format!("{}\n{rest}", qemu.boot_log()); + let said = format!("{} spin", bootlog::JOB_DEADLINE_SAID); + for want in ["===TEST_START spin===", said.as_str(), exited.as_str()] { + if !serial.contains(want) { + return Err(format!("{want:?} not on the PL011\nserial:\n{serial}")); + } + } + let line = serial.lines().find(|l| l.contains(&exited)).expect("the exit line was just found"); + let cpu_ms: u64 = line + .split("cpu=") + .nth(1) + .and_then(|rest| rest.strip_suffix("ms")) + .and_then(|ms| ms.trim().parse().ok()) + .ok_or_else(|| format!("no cpu=ms on the exit line: {line:?}"))?; + // Seconds, where a preemption that never came would leave `spin` + // no CPU time taken from the deadline thread, and a deadline thread + // that never ran would leave no exit line at all. + const HELD_MS: u64 = 5_000; + if cpu_ms < HELD_MS { + return Err(format!( + "spin ran {cpu_ms} ms before the deadline thread ended it, under the {HELD_MS} ms \ + that shows it held the CPU\nserial:\n{serial}" + )); + } + eprintln!(" [virt] spin held the one CPU for {cpu_ms} ms until a tick handed it to the deadline thread"); + Ok(()) + } + "virt_irq_storm" => { + // The timer ticking at a fixed period while the CPU floods itself + // with SGIs: every SGI sent is taken, and no tick goes a whole + // period untaken, which is what losing one would be. + let mut qemu = QemuInstance::boot_with_options( + test_config, + &[], + &[], + BootOptions { + profile: qemu::Profile::VirtEl2, + kernel_params: &["irq-storm"], + ready_marker: "control registers: SCTLR_EL1=", + ..Default::default() + }, + ); + let rest = qemu.drain_until(Duration::from_secs(180), |l| l.contains("irq-storm: ")); + let serial = format!("{}\n{rest}", qemu.boot_log()); + let Some(verdict) = serial.lines().find(|l| l.contains("irq-storm: ")) else { + return Err(format!("the storm never reported\nserial:\n{serial}")); + }; + eprintln!(" [virt] {verdict}"); + if !verdict.contains("irq-storm: PASS") { + return Err(format!("{verdict}\nserial:\n{serial}")); + } + Ok(()) + } "screen_late_panic" => { // The ordinary fatal panic, which no userland process can produce: // crash_report, capture, panic_flush, halt_all_cpus, render. The diff --git a/tests/virtpreemptcase/system.toml b/tests/virtpreemptcase/system.toml new file mode 100644 index 0000000000..6bf6f4bcc8 --- /dev/null +++ b/tests/virtpreemptcase/system.toml @@ -0,0 +1,19 @@ +# One CPU, one job that never yields, and the runner's deadline thread: the +# thread wakes on a tick or not at all, so a job list that ends is a timer +# that preempts. `--bound-ms` is from boot, and long enough for `spin` to hold +# the CPU for seconds before the deadline falls. +[boot] +start = ["logd", "test-runner"] + +[programs.logd] +service = true +syscap = ["logread"] + +[programs.test-runner] +receives = ["power"] +args = ["--bound-ms=45000", "spin"] + +[programs.toybox] + +[symlinks] +"bin/spin" = "/system/bin/toybox" diff --git a/toyos-bootmap/src/aarch64.rs b/toyos-bootmap/src/aarch64.rs index 9d833667a0..08b33f7542 100644 --- a/toyos-bootmap/src/aarch64.rs +++ b/toyos-bootmap/src/aarch64.rs @@ -1,9 +1,12 @@ //! AArch64's encoding of a [`Plan`](crate::Plan): the VMSAv8-64 stage 1 //! descriptors of a 4 KiB granule (Arm ARM K.a, D8.3), and the one `MAIR_EL1` //! whose indices they name — the loader writes the one, the kernel's entry -//! loads the other, and both read them here. +//! loads the other, and both read them here. And which pages the kernel's own +//! direct map holds, since on AArch64 that is decided page by page. -use crate::Cache; +use toyos_abi::boot::MemoryMapEntry; + +use crate::{is_read_as_memory, Cache, DirectMapEnd, Refusal, DIRECT_MAP_WINDOW, PAGE_2M, PAGE_4K}; /// `MAIR_EL1` index 0: Device-nGnRE, registers. pub const ATTR_DEVICE: u64 = 0; @@ -63,3 +66,71 @@ const fn attributes(cache: Cache) -> u64 { pub const fn page(phys: u64, cache: Cache) -> u64 { phys | PAGE | AF | attributes(cache) } + +/// 4 KiB pages in one 2 MiB page. +const PAGES: u64 = PAGE_2M / PAGE_4K; + +/// How the kernel's own direct map holds one 2 MiB page of physical memory. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Coverage { + /// No byte of it is memory the kernel reads: it is not mapped. + Nothing, + /// Every byte is: one block. + Whole, + /// Some of its 4 KiB pages are: a table whose page `i` is mapped where bit + /// `i % 64` of word `i / 64` is set. + Pages([u64; 8]), +} + +impl Coverage { + /// Whether the 4 KiB page `index` of the 2 MiB page is mapped. + pub const fn holds(&self, index: u64) -> bool { + match self { + Self::Nothing => false, + Self::Whole => true, + Self::Pages(bits) => bits[(index / 64) as usize] & 1 << (index % 64) != 0, + } + } +} + +/// One past the kernel direct map's last byte on AArch64: the end of the +/// highest range the kernel reads as memory, in whole 2 MiB pages. Nothing +/// below it is mapped for being low, as x86-64's low pages are: a page is +/// mapped only for being memory ([`coverage`]), because a register or a hole +/// mapped Normal is one a speculative access can reach. A range off a 4 KiB +/// page, or ending past [`DIRECT_MAP_WINDOW`], is refused. +pub fn direct_map_end(map: &[MemoryMapEntry]) -> Result { + map.iter().filter(|entry| is_read_as_memory(entry.uefi_type)).try_fold(0, |end: u64, entry| { + if !entry.start.is_multiple_of(PAGE_4K) { + return Err(Refusal::OffPage(entry.start)); + } + if !entry.end.is_multiple_of(PAGE_4K) { + return Err(Refusal::OffPage(entry.end)); + } + if entry.end < entry.start { + return Err(Refusal::Extent { base: entry.start, len: entry.end.wrapping_sub(entry.start) }); + } + if entry.end > DIRECT_MAP_WINDOW { + return Err(Refusal::PastWindow(entry.end)); + } + Ok(end.max(entry.end.next_multiple_of(PAGE_2M))) + }) + .map(DirectMapEnd) +} + +/// Which of the 4 KiB pages of the 2 MiB page at `page` firmware's map calls +/// memory the kernel reads, over a map [`direct_map_end`] accepted. +pub fn coverage(map: &[MemoryMapEntry], page: u64) -> Coverage { + let mut bits = [0u64; 8]; + for entry in map.iter().filter(|entry| is_read_as_memory(entry.uefi_type)) { + let (low, high) = (entry.start.max(page), entry.end.min(page + PAGE_2M)); + for index in (low.saturating_sub(page) / PAGE_4K)..(high.saturating_sub(page) / PAGE_4K) { + bits[(index / 64) as usize] |= 1 << (index % 64); + } + } + match bits.iter().map(|word| u64::from(word.count_ones())).sum::() { + 0 => Coverage::Nothing, + PAGES => Coverage::Whole, + _ => Coverage::Pages(bits), + } +} diff --git a/toyos-bootmap/src/lib.rs b/toyos-bootmap/src/lib.rs index ce22f5526a..92f7237bfe 100644 --- a/toyos-bootmap/src/lib.rs +++ b/toyos-bootmap/src/lib.rs @@ -59,8 +59,7 @@ pub const BOOT_MAP_BYTES: u64 = 4 * GIB; /// hold: every slot from there to the root's last. pub const DIRECT_MAP_WINDOW: u64 = (512 - ROOT_HIGH_HALF as u64) * GIB_PER_PDPT * GIB; -/// One past the kernel direct map's last byte. Made only by -/// [`x86_64::direct_map_end`]. +/// One past the kernel direct map's last byte. /// /// ```compile_fail,E0603 /// let _ = toyos_bootmap::DirectMapEnd(1 << 52); @@ -163,6 +162,9 @@ pub enum Refusal { Mixed(u64), /// Memory that ends here, past [`DIRECT_MAP_WINDOW`]. PastWindow(u64), + /// A range of firmware's map that begins or ends here, off the 4 KiB page + /// UEFI describes memory in. + OffPage(u64), } impl fmt::Display for Refusal { @@ -190,6 +192,7 @@ impl fmt::Display for Refusal { f, "memory ends at {end:#x}, past the {DIRECT_MAP_WINDOW:#x} bytes a direct map can hold" ), + Self::OffPage(at) => write!(f, "a range of firmware's map is bounded at {at:#x}, off a {PAGE_4K:#x}-byte page"), } } } diff --git a/toyos-bootmap/tests/aarch64_direct_map.rs b/toyos-bootmap/tests/aarch64_direct_map.rs new file mode 100644 index 0000000000..73c8b15187 --- /dev/null +++ b/toyos-bootmap/tests/aarch64_direct_map.rs @@ -0,0 +1,84 @@ +//! Which pages the AArch64 kernel's direct map holds: every 4 KiB page +//! firmware's map calls memory the kernel reads, and no other. + +use toyos_abi::boot::MemoryMapEntry; +use toyos_bootmap::aarch64::{coverage, direct_map_end, Coverage}; +use toyos_bootmap::{DirectMapEnd, Refusal, DIRECT_MAP_WINDOW, PAGE_2M}; + +const RESERVED: u32 = 0; +const LOADER_DATA: u32 = 2; +const RUNTIME_DATA: u32 = 6; +const CONVENTIONAL: u32 = 7; +const ACPI_RECLAIM: u32 = 9; +const MMIO: u32 = 11; + +const GIB: u64 = 1 << 30; + +const fn e(uefi_type: u32, start: u64, end: u64) -> MemoryMapEntry { + MemoryMapEntry { uefi_type, start, end } +} + +fn end(map: &[MemoryMapEntry]) -> Result { + direct_map_end(map).map(DirectMapEnd::get) +} + +/// RAM from 1 GiB, as `virt` has it, with firmware's carve-outs inside it. +const VIRT: [MemoryMapEntry; 7] = [ + e(MMIO, 0x0900_0000, 0x0900_1000), + e(CONVENTIONAL, GIB, GIB + 0x1F_F000), + e(RESERVED, GIB + 0x1F_F000, GIB + 0x20_0000), + e(CONVENTIONAL, GIB + 0x20_0000, GIB + 0x60_0000), + e(RUNTIME_DATA, GIB + 0x60_0000, GIB + 0x60_3000), + e(ACPI_RECLAIM, GIB + 0x60_3000, GIB + 0x60_4000), + e(LOADER_DATA, GIB + 0x60_4000, 3 * GIB), +]; + +#[test] +fn memory_is_mapped_and_a_register_is_not() { + assert_eq!(end(&VIRT), Ok(3 * GIB)); + assert_eq!(coverage(&VIRT, 0x0900_0000 & !(PAGE_2M - 1)), Coverage::Nothing); + assert_eq!(coverage(&VIRT, 0), Coverage::Nothing); +} + +#[test] +fn a_page_wholly_memory_is_one_block() { + assert_eq!(coverage(&VIRT, GIB + PAGE_2M), Coverage::Whole); + assert_eq!(coverage(&VIRT, 2 * GIB), Coverage::Whole); +} + +#[test] +fn a_reserved_page_inside_ram_is_left_out_of_its_block() { + let first = coverage(&VIRT, GIB); + assert!(matches!(first, Coverage::Pages(_))); + assert!(first.holds(0)); + assert!(first.holds(510)); + assert!(!first.holds(511), "EfiReservedMemoryType is not memory the kernel reads"); +} + +#[test] +fn runtime_services_data_is_left_out_and_acpi_tables_are_not() { + let page = coverage(&VIRT, GIB + 3 * PAGE_2M); + for runtime in 0..3 { + assert!(!page.holds(runtime), "runtime services data, page {runtime}"); + } + assert!(page.holds(3), "the ACPI table page"); + assert!(page.holds(4), "loader data"); + assert!(page.holds(511), "loader data"); +} + +#[test] +fn off_a_page_is_refused() { + let map = [e(CONVENTIONAL, GIB, GIB + 0x800)]; + assert_eq!(end(&map), Err(Refusal::OffPage(GIB + 0x800))); +} + +#[test] +fn past_the_window_is_refused() { + let map = [e(CONVENTIONAL, DIRECT_MAP_WINDOW, DIRECT_MAP_WINDOW + PAGE_2M)]; + assert_eq!(end(&map), Err(Refusal::PastWindow(DIRECT_MAP_WINDOW + PAGE_2M))); +} + +#[test] +fn a_map_with_no_memory_ends_at_zero() { + assert_eq!(end(&[e(MMIO, 0x0900_0000, 0x0900_1000)]), Ok(0)); +} From 9b7ed0cec4fa1e1829fa4138fb2c60e7159d442c Mon Sep 17 00:00:00 2001 From: japabu Date: Mon, 28 Sep 2026 23:34:23 +0200 Subject: [PATCH 02/11] AArch64 stage 4: the GIC leaves the entry, and the guest tests judge events, not time The orchestrator's guest run at 57382f78 had four reds. virt_early_panic and virt_early_fault (HVF, entered at EL1) went silent after the loader's handoff, where main is green: this branch put two GIC instructions ahead of the first record on the EL1 path, the ID_AA64PFR0_EL1.GIC gate that halts silently in refused_no_gicv3 and an ICC_SRE_EL1 write under firmware's vectors. QEMU 11.1.1's HVF sets that register's GIC field from env->gicv3state at vCPU creation (target/arm/hvf/hvf.c:1481-1483), which the GIC's realize fills in only after virt.c has realized every CPU (hw/arm/virt.c:3140 before 3158), so under HVF the field is whatever Hypervisor.framework reports for an Apple core; which of the two ended the boot is not measured. Both leave the entry: ICC_SRE_EL1 is written and its SRE read back in irqchip::init, under this kernel's vectors, where a CPU interface that is not there is an undefined instruction the kernel reports. The EL2 path still writes ICC_SRE_EL2, skipped where the ID register names no interface, so that case too fails under this kernel's vectors rather than firmware's. virt_irq_storm and virt_timer_preempts judged rates under TCG: ticks per 2000 ms and a tick's lateness, and a job's CPU time against a 45 s bound. No QEMU test measures time. The storm now floods SGIs until the timer has fired a thousand times and then waits for every SGI it sent, so a tick lost or never re-armed, or an SGI lost, leaves it unsaid; irqchip::lateness goes with the lateness verdict. virt_timer_preempts runs toybox's new `preempt`: a thread counting with no syscall, and a thread that yields until it has seen the count move twice, which on one CPU happens only when an interrupt took the CPU from the counter. The old test's red was its own ordering: it drained to spin's exit line, and the runner's line reaches the console after it. The track records that no QEMU test can show the EL2 deletions red: QEMU resets CNTHCTL_EL2, CNTVOFF_EL2 and CPTR_EL2 to values the declaration agrees with, and ICC_SRE_EL2 is a constant; each deletion stayed green in virt_user_mode at 57382f78. The interrupts-off window is metal's to measure. Co-Authored-By: Claude Opus 5.5 --- issues/kernel/toyos-runs-on-arm64.md | 20 ++++++--- kernel/src/arch/aarch64/boot.rs | 31 ++++---------- kernel/src/arch/aarch64/control_regs.rs | 13 +++--- kernel/src/arch/aarch64/irqchip.rs | 32 ++++++++------ kernel/src/arch/aarch64/trap.rs | 55 +++++++++---------------- tests/toyos.rs | 48 +++++++-------------- tests/virtpreemptcase/system.toml | 10 ++--- userland/toybox/src/main.rs | 3 +- userland/toybox/src/preempt.rs | 30 ++++++++++++++ 9 files changed, 121 insertions(+), 121 deletions(-) create mode 100644 userland/toybox/src/preempt.rs diff --git a/issues/kernel/toyos-runs-on-arm64.md b/issues/kernel/toyos-runs-on-arm64.md index dfa468a740..11b60d920a 100644 --- a/issues/kernel/toyos-runs-on-arm64.md +++ b/issues/kernel/toyos-runs-on-arm64.md @@ -304,12 +304,20 @@ Each stage names its exit; "measured" means a number from a run. switch carrying FP/SIMD; `virt_user_mode`, `virt_timer_preempts` and `virt_irq_storm` judge it under the EL2 profile, emulated, because HVF exposes no RNDR and the kernel's hash seed refuses there until stage 6's - virtio-rng. Owed before the exit holds: the interrupts-off window against - x86's (the storm reports the latest tick it took; x86 has no counterpart - instrument); `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md`, - whose `KernelArgs` rename waits on the loader's change to that struct; and - the three deletions shown red. The ITS moves to stage 6: a claimed function - is its only consumer the small-kernel track leaves, and it needs that + virtio-rng. Each judges an event, never a rate: no QEMU test measures time. + Owed before the exit holds: the interrupts-off window against x86's, a + measurement only metal can make, with no instrument on either arch yet; + `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md`, whose + `KernelArgs` rename waits on the loader's change to that struct; and the + three deletions shown red, which **no QEMU test can do**: QEMU 11.1.1 + resets `CNTHCTL_EL2` to 3 and `CNTVOFF_EL2` to 0, the declared values, and + `CPTR_EL2` to 0, whose `TFP` is as clear as the declaration's + (`target/arm/helper.c`); each deletion, and the fourth, of `ICC_SRE_EL2` + (constant `0xf` in QEMU, `hw/intc/arm_gicv3_cpuif.c`), stayed green in + `virt_user_mode` at `57382f78`. They are shown red on a machine whose + firmware leaves the registers otherwise, or by a loader that writes the + opposite values before the handoff. The ITS moves to stage 6: a claimed + function is its only consumer the small-kernel track leaves, and it needs that stage's SMMUv3 first. Stubbed on AArch64, each owned by the small-kernel track, which moves the driver out of the kernel: - `arch::msi_message` refuses, so the kernel's xHCI (`virt`'s boot stick), diff --git a/kernel/src/arch/aarch64/boot.rs b/kernel/src/arch/aarch64/boot.rs index c2aa4b188d..64e874caf0 100644 --- a/kernel/src/arch/aarch64/boot.rs +++ b/kernel/src/arch/aarch64/boot.rs @@ -33,11 +33,6 @@ use crate::mm::Region; pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { core::arch::naked_asm!( "mov x19, x0", - // `ID_AA64PFR0_EL1.GIC`: no GICv3 system-register interface, and the - // `ICC_SRE` writes below are undefined instructions. - "mrs x1, id_aa64pfr0_el1", - "ubfx x1, x1, #24, #4", - "cbz x1, {refuse_gic}", // x20 = the stack's top, physical: kernel image + stack offset + stack size. "ldr x20, [x19, #{kernel_memory}]", "ldr x1, [x19, #{stack_offset}]", @@ -70,8 +65,6 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { "msr cpacr_el1, x1", "mov x1, #{cntkctl}", "msr cntkctl_el1, x1", - "mov x1, #{icc_sre}", - "msr S3_0_C12_C12_5, x1", "tlbi vmalle1", "dsb nsh", "isb", @@ -89,10 +82,18 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { "mrs x6, hcr_el2", "cmp x6, x1", "b.ne {refuse_hcr}", - // `ICC_SRE_EL2` before `ICC_SRE_EL1`: its `Enable` is what lets EL1's be written at all. + // `ICC_SRE_EL2`, whose `Enable` lets EL1 write its own `ICC_SRE_EL1` + // (`super::irqchip::init`). Skipped where `ID_AA64PFR0_EL1.GIC` names + // no system-register interface: there the write is an undefined + // instruction under firmware's vectors and ends the boot silently, + // while EL1's own access fails under this kernel's, which report it. + "mrs x1, id_aa64pfr0_el1", + "ubfx x1, x1, #24, #4", + "cbz x1, 4f", "mov x1, #{icc_sre_el2}", "msr S3_4_C12_C9_5, x1", "isb", + "4:", "msr mair_el1, x4", "msr tcr_el1, x2", "msr ttbr0_el1, x3", @@ -101,8 +102,6 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { "msr cpacr_el1, x1", "mov x1, #{cntkctl}", "msr cntkctl_el1, x1", - "mov x1, #{icc_sre}", - "msr S3_0_C12_C12_5, x1", "msr sctlr_el1, x5", "mov x1, #{cnthctl}", "msr cnthctl_el2, x1", @@ -149,9 +148,7 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { cptr = const regs::CPTR_EL2, cpacr = const regs::CPACR, cntkctl = const regs::CNTKCTL, - icc_sre = const regs::ICC_SRE, icc_sre_el2 = const regs::ICC_SRE_EL2, - refuse_gic = sym refused_no_gicv3, spsr = const regs::SPSR_EL2_TO_EL1, phys_offset = const crate::PHYS_OFFSET, entry_el = sym regs::ENTRY_EL, @@ -172,16 +169,6 @@ unsafe extern "C" fn refused_hcr_el2_readback() -> ! { core::arch::naked_asm!("1:", "wfe", "b 1b") } -/// Where a CPU whose `ID_AA64PFR0_EL1.GIC` names no GICv3 system-register -/// interface halts, before the first `ICC_` write would be an undefined -/// instruction under firmware's vectors: this kernel's interrupt controller is -/// a GICv3. Silent for [`refused_hcr_el2_readback`]'s reason. -#[unsafe(naked)] -#[no_mangle] -unsafe extern "C" fn refused_no_gicv3() -> ! { - core::arch::naked_asm!("1:", "wfe", "b 1b") -} - /// The ACPI tables this architecture decodes: the MADT for its GIC and CPUs, /// the FADT for PSCI and reset, the GTDT for the timer, the SPCR for the /// console, and the MCFG for ECAM. diff --git a/kernel/src/arch/aarch64/control_regs.rs b/kernel/src/arch/aarch64/control_regs.rs index 275ab97d5d..51b133d87e 100644 --- a/kernel/src/arch/aarch64/control_regs.rs +++ b/kernel/src/arch/aarch64/control_regs.rs @@ -3,7 +3,7 @@ //! [`super::boot`]'s entry writes every register here whole, from these //! constants, before the MMU is on; [`check`] reads the EL1 ones back and //! refuses a CPU whose registers say anything else. Nothing else writes any -//! of them. +//! of them, bar [`ICC_SRE`]. //! //! Field positions are Arm ARM K.a, chapter D24 (the register descriptions), //! and the GIC architecture specification (IHI 0069H), chapter 12, for the @@ -57,7 +57,10 @@ pub const CPACR: u64 = 0b11 << 20; pub const CNTKCTL: u64 = 1 << 1; /// `ICC_SRE_EL1`: the GICv3 CPU interface through system registers (`SRE`), -/// with FIQ and IRQ bypass disabled (`DFB`, `DIB`); [`check`] reads back `SRE`. +/// with FIQ and IRQ bypass disabled (`DFB`, `DIB`). Written and read back by +/// `super::irqchip::init`, not the entry: on a CPU with no such interface the +/// access is an undefined instruction, which this kernel's vectors report and +/// firmware's, still installed at the entry, do not. pub const ICC_SRE: u64 = 0b111; /// `HCR_EL2` when entered at EL2: `RW`, so EL1 is AArch64, and nothing else — @@ -116,17 +119,13 @@ pub fn check() { for (name, live, value) in declared { assert_eq!(live, value, "control registers: {name} holds {live:#x}, and the declaration says {value:#x}"); } - // `SRE` alone: `DFB` and `DIB` are RAO/WI or RAZ/WI as the implementation, - // or a hypervisor under it, chooses, and this kernel uses neither bypass. - let sre = read!("S3_0_C12_C12_5"); - assert!(sre & 1 != 0, "control registers: ICC_SRE_EL1 reads {sre:#x}, so the GICv3 CPU interface is not in system registers"); // What the drop from EL2 left, or the EL1 entry kept: EL1, on `SP_EL1`. let (el, spsel) = (read!("CurrentEL") >> 2 & 3, read!("SPSel") & 1); assert_eq!((el, spsel), (1, 1), "control registers: running at EL{el} on SP_EL{spsel}, not EL1 on SP_EL1"); let el = ENTRY_EL.load(Ordering::Relaxed); log!( "control registers: SCTLR_EL1={SCTLR:#x} TCR_EL1={:#x} MAIR_EL1={:#x} CPACR_EL1={CPACR:#x} \ - CNTKCTL_EL1={CNTKCTL:#x} ICC_SRE_EL1.SRE=1, as declared; entered at EL{el}{}", + CNTKCTL_EL1={CNTKCTL:#x}, as declared; entered at EL{el}{}", tcr(), MAIR, if el == 2 { diff --git a/kernel/src/arch/aarch64/irqchip.rs b/kernel/src/arch/aarch64/irqchip.rs index d3b9a42b08..355d46f1c3 100644 --- a/kernel/src/arch/aarch64/irqchip.rs +++ b/kernel/src/arch/aarch64/irqchip.rs @@ -90,7 +90,7 @@ fn settles(per_second: u64, what: &str, done: impl Fn() -> bool) { fn read_sysreg_pmr() -> u64 { let v: u64; - // SAFETY: reads `ICC_PMR_EL1`, which `ICC_SRE_EL1.SRE` (the declaration) makes accessible. + // SAFETY: reads `ICC_PMR_EL1`, which `ICC_SRE_EL1.SRE` ([`init`]) makes accessible. unsafe { core::arch::asm!("mrs {}, S3_0_C4_C6_0", out(reg) v, options(nomem, nostack, preserves_flags)) }; v } @@ -173,10 +173,27 @@ pub fn init(rsdp_addr: u64) { let enabled = enabled | 1 << u32::from(super::trap::LOG_NEST_VECTOR) | 1 << SGI_STORM; redistributor.write_u32(GICR_ISENABLER0, enabled); + // The CPU interface in system registers, as the declaration says. + // SAFETY: writes `ICC_SRE_EL1`, which touches no memory. + unsafe { + core::arch::asm!( + "msr S3_0_C12_C12_5, {}", + "isb", + in(reg) super::control_regs::ICC_SRE, + options(nomem, nostack, preserves_flags), + ); + } + let sre: u64; + // SAFETY: reads `ICC_SRE_EL1`. + unsafe { core::arch::asm!("mrs {}, S3_0_C12_C12_5", out(reg) sre, options(nomem, nostack, preserves_flags)) }; + // `SRE` alone: `DFB` and `DIB` are RAO/WI or RAZ/WI as the implementation, + // or a hypervisor under it, chooses, and this kernel uses neither bypass. + assert!(sre & 1 != 0, "GIC: ICC_SRE_EL1 reads {sre:#x}, so the CPU interface is not in system registers"); + // The CPU interface: the priority mask open above `PRIORITY`, one // priority drop and deactivation per EOI, Group 1 on. - // SAFETY: GICv3 CPU interface registers, which the declaration's - // `ICC_SRE_EL1.SRE` makes system registers; none touches memory. + // SAFETY: GICv3 CPU interface registers, which `ICC_SRE_EL1.SRE` makes + // system registers; none touches memory. unsafe { core::arch::asm!( "msr S3_0_C4_C6_0, {pmr}", @@ -350,12 +367,3 @@ pub(super) fn rearm() { ticks => arm_ticks(ticks), } } - -/// How many counter ticks past its comparator the timer is being taken. -#[cfg(feature = "boot-actuators")] -pub(super) fn lateness() -> u64 { - let cval: u64; - // SAFETY: reads the EL1 virtual timer's comparator. - unsafe { core::arch::asm!("mrs {}, cntv_cval_el0", out(reg) cval, options(nomem, nostack, preserves_flags)) }; - cpu::counter().saturating_sub(cval) -} diff --git a/kernel/src/arch/aarch64/trap.rs b/kernel/src/arch/aarch64/trap.rs index cf4731c65a..5468c6811f 100644 --- a/kernel/src/arch/aarch64/trap.rs +++ b/kernel/src/arch/aarch64/trap.rs @@ -138,8 +138,6 @@ fn irq(from_el0: bool) { return; }; if intid == irqchip::timer_intid() { - #[cfg(feature = "boot-actuators")] - let late = irqchip::lateness(); // Before anything that can take a lock or panic: a timer left // asserted re-fires forever, and one left stopped never fires again. irqchip::rearm(); @@ -148,7 +146,7 @@ fn irq(from_el0: bool) { // on one still takes this interrupt, which is why the poll is here. crate::deadline::poll(); #[cfg(feature = "boot-actuators")] - storm::tick(late); + storm::tick(); if from_el0 { // Only an EL0 tick reaches here, so the interrupted context is user // code and holds no `Lock`. @@ -515,13 +513,12 @@ pub(crate) fn provoke_double_fault() -> ! { panic!("SYS_DEBUG: AArch64 has no double fault to provoke"); } -/// `irq-storm`: the timer ticking at a known period while this CPU floods -/// itself with SGIs. A tick is lost when its expiry goes a whole period -/// untaken — the next one would have been due before it — or when it never -/// comes, so the verdict is the latest any tick was taken and how many came, -/// beside every SGI sent being taken. Each tick is armed from when the last -/// was taken, so the count falls short of the span's periods by the latency -/// they add up to, and nine in ten is the floor that leaves room for it. +/// `irq-storm`: this CPU floods itself with SGIs until the timer has fired +/// `TICKS_OWED` times through the flood, then waits for every SGI it +/// sent to be taken. A tick lost, or never re-armed, leaves the flood running +/// and an SGI lost leaves the wait running, so neither says anything: the +/// harness's ceiling is the only clock this judges by, and the one verdict +/// the storm can say is `FAIL` for an SGI taken that was never sent. #[cfg(feature = "boot-actuators")] pub(crate) mod storm { use core::sync::atomic::{AtomicBool, AtomicU64, Ordering::Relaxed}; @@ -531,14 +528,11 @@ pub(crate) mod storm { static RUNNING: AtomicBool = AtomicBool::new(false); static TICKS: AtomicU64 = AtomicU64::new(0); - static LATEST: AtomicU64 = AtomicU64::new(0); static SGIS: AtomicU64 = AtomicU64::new(0); - /// One tick, `late` counter ticks past its comparator. - pub(super) fn tick(late: u64) { + pub(super) fn tick() { if RUNNING.load(Relaxed) { TICKS.fetch_add(1, Relaxed); - LATEST.fetch_max(late, Relaxed); } } @@ -546,42 +540,33 @@ pub(crate) mod storm { SGIS.fetch_add(1, Relaxed); } - /// The timer's period. + /// The timer's period, which each tick re-arms. const PERIOD_NS: u64 = 1_000_000; - /// How long the storm lasts: two thousand periods. - const SPAN_NS: u64 = 2_000 * PERIOD_NS; + /// The ticks the flood lasts for. + const TICKS_OWED: u64 = 1_000; - /// Run the storm, with interrupts open on this CPU for its length, and - /// say what it counted. + /// Run the storm, with interrupts open on this CPU until it ends, and say + /// what it counted. pub fn run() { let _guard = crate::arch::IrqGuard::close(); TICKS.store(0, Relaxed); - LATEST.store(0, Relaxed); SGIS.store(0, Relaxed); irqchip::arm_one_shot(PERIOD_NS); RUNNING.store(true, Relaxed); - let start = crate::clock::nanos_since_boot(); let mut sent = 0u64; crate::arch::cpu::enable_interrupts(); - while crate::clock::nanos_since_boot() - start < SPAN_NS { + while TICKS.load(Relaxed) < TICKS_OWED { irqchip::send_self(irqchip::SGI_STORM as u8); sent += 1; } - crate::arch::cpu::disable_interrupts(); RUNNING.store(false, Relaxed); irqchip::stop_timer(); + while SGIS.load(Relaxed) < sent { + core::hint::spin_loop(); + } + crate::arch::cpu::disable_interrupts(); let (ticks, taken) = (TICKS.load(Relaxed), SGIS.load(Relaxed)); - let latest_ns = crate::clock::nanos_of_ticks(LATEST.load(Relaxed)); - // The last SGI sent may still be in flight when the mask closes on it. - let sgis_whole = taken + 1 >= sent && taken <= sent; - let owed = SPAN_NS / PERIOD_NS; - let verdict = if sgis_whole && ticks >= owed * 9 / 10 && latest_ns < PERIOD_NS { "PASS" } else { "FAIL" }; - log!( - "irq-storm: {verdict} sgis={taken}/{sent} ticks={ticks} of {owed} over {} ms at a {} us period, \ - the latest taken {} us past its expiry", - SPAN_NS / 1_000_000, - PERIOD_NS / 1_000, - latest_ns / 1_000, - ); + let verdict = if taken == sent { "PASS" } else { "FAIL" }; + log!("irq-storm: {verdict} sgis={taken}/{sent} ticks={ticks}: the timer fired through the flood, and every SGI sent was taken"); } } diff --git a/tests/toyos.rs b/tests/toyos.rs index c789724e94..c363c12dc5 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -5964,10 +5964,12 @@ fn run_screen_test( Ok(()) } "virt_timer_preempts" => { - // After its first line `spin` makes no syscall, yet on this - // machine's one CPU the runner's deadline thread wakes — at a - // tick, the only thing that can take the CPU from `spin` — and - // ends it. The CPU time `spin` had by then says it held the CPU. + // `preempt` on this machine's one CPU: one thread counts in a loop + // that enters the kernel only when an interrupt takes it there, and + // the other yields until it has seen the count move twice. It runs + // between those reads only when a tick took the CPU from the + // counting thread, so a timer that never preempts leaves the line + // unsaid, and the ceiling here is the only clock. let config = compile::repo_root().join("tests/virtpreemptcase/system.toml"); let case = config.parent().expect("system.toml has a directory"); let mut qemu = QemuInstance::boot_with_options( @@ -5980,39 +5982,21 @@ fn run_screen_test( ..Default::default() }, ); - let exited = format!("{}spin pid=", bootlog::EXIT); - let rest = qemu.drain_until(Duration::from_secs(300), |l| l.contains(&exited)); + // Spelled in `userland/toybox/src/preempt.rs`. + const PREEMPTED: &str = "preempt: the counting thread was preempted twice"; + let rest = qemu.drain_until(Duration::from_secs(300), |l| l.contains(PREEMPTED)); let serial = format!("{}\n{rest}", qemu.boot_log()); - let said = format!("{} spin", bootlog::JOB_DEADLINE_SAID); - for want in ["===TEST_START spin===", said.as_str(), exited.as_str()] { - if !serial.contains(want) { - return Err(format!("{want:?} not on the PL011\nserial:\n{serial}")); - } - } - let line = serial.lines().find(|l| l.contains(&exited)).expect("the exit line was just found"); - let cpu_ms: u64 = line - .split("cpu=") - .nth(1) - .and_then(|rest| rest.strip_suffix("ms")) - .and_then(|ms| ms.trim().parse().ok()) - .ok_or_else(|| format!("no cpu=ms on the exit line: {line:?}"))?; - // Seconds, where a preemption that never came would leave `spin` - // no CPU time taken from the deadline thread, and a deadline thread - // that never ran would leave no exit line at all. - const HELD_MS: u64 = 5_000; - if cpu_ms < HELD_MS { - return Err(format!( - "spin ran {cpu_ms} ms before the deadline thread ended it, under the {HELD_MS} ms \ - that shows it held the CPU\nserial:\n{serial}" - )); + match serial.lines().find(|l| l.contains(PREEMPTED)) { + Some(line) => eprintln!(" [virt] {line}"), + None => return Err(format!("{PREEMPTED:?} not on the PL011\nserial:\n{serial}")), } - eprintln!(" [virt] spin held the one CPU for {cpu_ms} ms until a tick handed it to the deadline thread"); Ok(()) } "virt_irq_storm" => { - // The timer ticking at a fixed period while the CPU floods itself - // with SGIs: every SGI sent is taken, and no tick goes a whole - // period untaken, which is what losing one would be. + // The CPU floods itself with SGIs until the timer has fired a + // thousand times through the flood, then waits for every SGI it + // sent. A tick lost or never re-armed, or an SGI lost, leaves the + // storm running and the verdict unsaid. let mut qemu = QemuInstance::boot_with_options( test_config, &[], diff --git a/tests/virtpreemptcase/system.toml b/tests/virtpreemptcase/system.toml index 6bf6f4bcc8..5f9a06bd81 100644 --- a/tests/virtpreemptcase/system.toml +++ b/tests/virtpreemptcase/system.toml @@ -1,7 +1,5 @@ -# One CPU, one job that never yields, and the runner's deadline thread: the -# thread wakes on a tick or not at all, so a job list that ends is a timer -# that preempts. `--bound-ms` is from boot, and long enough for `spin` to hold -# the CPU for seconds before the deadline falls. +# One CPU and one job, `preempt`, whose line comes only after the timer has +# taken the CPU from a thread that never enters the kernel itself. [boot] start = ["logd", "test-runner"] @@ -11,9 +9,9 @@ syscap = ["logread"] [programs.test-runner] receives = ["power"] -args = ["--bound-ms=45000", "spin"] +args = ["preempt"] [programs.toybox] [symlinks] -"bin/spin" = "/system/bin/toybox" +"bin/preempt" = "/system/bin/toybox" diff --git a/userland/toybox/src/main.rs b/userland/toybox/src/main.rs index 9e3df68770..5e703d041b 100644 --- a/userland/toybox/src/main.rs +++ b/userland/toybox/src/main.rs @@ -9,6 +9,7 @@ mod ls; mod mkdir; mod mv; mod net; +mod preempt; mod ps; mod pwd; mod reboot; @@ -30,7 +31,7 @@ macro_rules! commands { }; } -commands!(cat, cp, echo, free, grep, hexdump, locale, ls, mkdir, mv, net, ps, pwd, reboot, rm, screen, shutdown, spin, stats, tone); +commands!(cat, cp, echo, free, grep, hexdump, locale, ls, mkdir, mv, net, preempt, ps, pwd, reboot, rm, screen, shutdown, spin, stats, tone); fn main() { let args: Vec = std::env::args().collect(); diff --git a/userland/toybox/src/preempt.rs b/userland/toybox/src/preempt.rs new file mode 100644 index 0000000000..60e147a793 --- /dev/null +++ b/userland/toybox/src/preempt.rs @@ -0,0 +1,30 @@ +//! `preempt`: whether an interrupt takes the CPU from a thread that never +//! enters the kernel itself. A second thread counts in a loop with no syscall, +//! and once it has gone round twice no fault either; this one yields until the +//! count has moved past what it last read, twice. On one CPU it runs between +//! those reads only if the counting thread lost the CPU to an interrupt, so on +//! a kernel whose timer never preempts the line below never comes. + +use std::sync::atomic::{AtomicU64, Ordering::Relaxed}; + +static COUNT: AtomicU64 = AtomicU64::new(0); + +pub fn main(_args: Vec) { + std::thread::spawn(|| loop { + COUNT.fetch_add(1, Relaxed); + }); + let first = moved_past(1); + let second = moved_past(first); + println!("preempt: the counting thread was preempted twice, at counts {first} and {second}"); +} + +/// Yield until the count is past `seen`, and say where it is. +fn moved_past(seen: u64) -> u64 { + loop { + std::thread::yield_now(); + let now = COUNT.load(Relaxed); + if now > seen { + return now; + } + } +} From d2e14390a923e75fa7b3de8062285a3cdf8b4f5f Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 03:22:02 +0200 Subject: [PATCH 03/11] AArch64 stage 4: one KernelHw, the timer floor, and SYS_DEBUG refusals The review of #589 at 9b7ed0ce, B1, B2, N2, N3, N8, N9, N10, N12: - B1: `irqchip::arm_within` armed `want.min(remaining).max(1)`, as little as one counter tick, and stored it as what `rearm` repeats on every EL1 fire. The floor is now enforced once, in `arm_ticks`, the only write of the comparator, as x86's `OneShot::ticks` enforces it. `timer-floor` is the boot actuator that makes the timer due, calls `arm_within`, and takes a hundred EL1 fires with interrupts open. - B2: `KernelHw`, its `Kicker` and `Machine` impls (with `need_resched`, `idle_wait` and `DIAG_TICK_NS`), `RUNNING_CTX`, `report_contexts` and `MIN_ONE_SHOT` move to the portable `kernel/src/hw.rs`; each architecture's `hw.rs` keeps its halt and its `Hw::switch`. The unified `report_contexts` is x86's, which said more. - B2/N8: `toyos-gicv3`, a pure crate with host tests, holds the MPIDR affinity packing (`cpu::hardware_id` and the MADT match both use it), the `ICC_SGI1R_EL1` encoding with its `RS` field, and the redistributor walk with its `VLPIS` stride. - N2: `clock::tsc_ticks` is `counter_ticks`; the percpu timer-fire accessors are `kernel_timer_fires` and kin; `leave_ring3_if_due` is `leave_user_if_due`, and `exit_to_user`'s comments speak of user mode and interrupts rather than Ring 3 and `IF`. - N3: `SYS_DEBUG`'s double fault and TLB acknowledgement delay answer `NotSupported` on AArch64 instead of panicking the kernel. - N9: one `irqchip::Intid` enum declares the kick, log-nest and storm SGIs and the two MSI identities, in one INTID space. - N10: the storm sends each SGI once the last is taken, since an SGI sent while one is pending merges with it on hardware. - N12: `trampoline_entry` calls `scheduler::exit_to_user` itself, and the AArch64 forwarder is gone. - REMOVE: the storm actuator's false "whole period untaken" clause, and "a thousandth of QUANTUM_NS" from the floor's reason. Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 4 + Cargo.toml | 1 + kernel/Cargo.lock | 5 + kernel/Cargo.toml | 1 + kernel/src/actuator.rs | 6 +- kernel/src/arch/aarch64/boot.rs | 3 + kernel/src/arch/aarch64/cpu.rs | 2 +- kernel/src/arch/aarch64/hw.rs | 131 +++-------------------- kernel/src/arch/aarch64/irqchip.rs | 116 ++++++++++++-------- kernel/src/arch/aarch64/percpu.rs | 24 ++--- kernel/src/arch/aarch64/tlb.rs | 8 +- kernel/src/arch/aarch64/trap.rs | 55 ++++------ kernel/src/arch/x86_64/apic.rs | 9 +- kernel/src/arch/x86_64/boot.rs | 5 +- kernel/src/arch/x86_64/hw.rs | 157 ++------------------------- kernel/src/arch/x86_64/percpu.rs | 6 +- kernel/src/clock.rs | 10 +- kernel/src/hardlockup/mod.rs | 6 +- kernel/src/hw.rs | 164 +++++++++++++++++++++++++++++ kernel/src/iod.rs | 2 +- kernel/src/main.rs | 2 +- kernel/src/quiesce.rs | 2 +- kernel/src/sched/driver.rs | 2 +- kernel/src/sched/dump.rs | 4 +- kernel/src/scheduler.rs | 32 +++--- src/sourcegate.rs | 1 + toyos-gicv3/Cargo.toml | 7 ++ toyos-gicv3/src/lib.rs | 54 ++++++++++ toyos-gicv3/tests/gicv3.rs | 67 ++++++++++++ 29 files changed, 482 insertions(+), 404 deletions(-) create mode 100644 kernel/src/hw.rs create mode 100644 toyos-gicv3/Cargo.toml create mode 100644 toyos-gicv3/src/lib.rs create mode 100644 toyos-gicv3/tests/gicv3.rs diff --git a/Cargo.lock b/Cargo.lock index 1e18b1d023..60171beec4 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1055,6 +1055,10 @@ dependencies = [ name = "toyos-fat32-check" version = "0.1.0" +[[package]] +name = "toyos-gicv3" +version = "0.1.0" + [[package]] name = "toyos-gpt" version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index 661c2ad735..70a2ae4db9 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -30,6 +30,7 @@ members = [ "toyos-elide", "toyos-fat32", "toyos-fat32-check", + "toyos-gicv3", "toyos-gpt", "toyos-hda", "toyos-i219", diff --git a/kernel/Cargo.lock b/kernel/Cargo.lock index be62dcc5ee..90c5ca15e1 100644 --- a/kernel/Cargo.lock +++ b/kernel/Cargo.lock @@ -46,6 +46,7 @@ dependencies = [ "toyos-elf", "toyos-elide", "toyos-fat32", + "toyos-gicv3", "toyos-gpt", "toyos-hda", "toyos-pci", @@ -121,6 +122,10 @@ dependencies = [ "toyos-wallclock", ] +[[package]] +name = "toyos-gicv3" +version = "0.1.0" + [[package]] name = "toyos-gpt" version = "0.1.0" diff --git a/kernel/Cargo.toml b/kernel/Cargo.toml index f8d2ea7f32..81e0a9b315 100644 --- a/kernel/Cargo.toml +++ b/kernel/Cargo.toml @@ -400,6 +400,7 @@ toyos-blockhold = { path = "../toyos-blockhold" } toyos-bootmap = { path = "../toyos-bootmap" } toyos-fat32 = { path = "../toyos-fat32" } toyos-elf = { path = "../toyos-elf" } +toyos-gicv3 = { path = "../toyos-gicv3" } toyos-gpt = { path = "../toyos-gpt" } toyos-hda = { path = "../toyos-hda" } toyos-pci = { path = "../toyos-pci" } diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 5dc44d2747..8584a0f8a0 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -342,9 +342,13 @@ actuators! { unclaimed_vector_selftest = "unclaimed-vector-selftest"; /// Tick the timer at a fixed period while this CPU floods itself with - /// interrupts, and say whether any tick went a whole period untaken. + /// interrupts. irq_storm = "irq-storm"; + /// Make this CPU's timer due with interrupts masked, ask it to fire within + /// a quantum, and take its interrupts with them open. + timer_floor = "timer-floor"; + /// Hold a flush of `truncate-race.bin` inside its metadata window and say whether a truncate got in. ftruncate_flush_stall = "ftruncate-flush-stall"; diff --git a/kernel/src/arch/aarch64/boot.rs b/kernel/src/arch/aarch64/boot.rs index 64e874caf0..3923fd5315 100644 --- a/kernel/src/arch/aarch64/boot.rs +++ b/kernel/src/arch/aarch64/boot.rs @@ -300,6 +300,9 @@ pub fn start_other_cpus(_platform: &Platform, _args: &KernelArgs) { /// The interrupt-controller selftests an actuator asks for. #[cfg(feature = "boot-actuators")] pub fn interrupt_selftests() { + if crate::actuator::timer_floor() { + super::irqchip::floor_selftest(); + } if crate::actuator::irq_storm() { super::trap::storm::run(); } diff --git a/kernel/src/arch/aarch64/cpu.rs b/kernel/src/arch/aarch64/cpu.rs index b32f79aa38..31e5fceb94 100644 --- a/kernel/src/arch/aarch64/cpu.rs +++ b/kernel/src/arch/aarch64/cpu.rs @@ -122,5 +122,5 @@ pub fn hardware_id() -> u32 { let mpidr: u64; // SAFETY: reads an ID register. unsafe { asm!("mrs {}, mpidr_el1", out(reg) mpidr, options(nomem, nostack, preserves_flags)) }; - ((mpidr & 0xFF_FFFF) | ((mpidr >> 8) & 0xFF00_0000)) as u32 + toyos_gicv3::packed_affinity(mpidr) } diff --git a/kernel/src/arch/aarch64/hw.rs b/kernel/src/arch/aarch64/hw.rs index 5c68a49f74..cb74189d88 100644 --- a/kernel/src/arch/aarch64/hw.rs +++ b/kernel/src/arch/aarch64/hw.rs @@ -1,127 +1,20 @@ -//! `KernelHw` — the kernel's side of the scheduler-core hardware boundary: -//! the generic timer, SGIs, `WFI` and the context switch. Nothing here makes -//! a scheduling decision. +//! AArch64's half of `crate::hw::KernelHw`: the halt and the context switch. -use toyos_sched::cpu::{RunToken, SleepToken}; -use toyos_sched::fair::QUANTUM_NS; -use toyos_sched::hw::{CpuId, Hw, Kicker, Machine, Nanos, TraceEvent}; +use toyos_sched::cpu::RunToken; +use toyos_sched::hw::Hw; use toyos_sched::task::{TaskAccounting, TaskKey}; use super::switch::{context_switch, RETURN_AT}; -use super::{cpu, irqchip, percpu}; +use super::{cpu, percpu}; +use crate::hw::{report_contexts, KernelHw}; use crate::sched::payload::{KernelCtx, KernelPayload}; -/// The one instance; zero-sized, holds no per-CPU state. -pub static HW: KernelHw = KernelHw; - -pub struct KernelHw; - -/// The scheduler clock, in raw nanoseconds. -pub fn now_ns() -> u64 { - HW.now().0 -} - -impl Kicker for KernelHw { - fn kick(&self, target: CpuId) { - irqchip::kick_cpu(target.0); - } -} - -impl Machine for KernelHw { - fn now(&self) -> Nanos { - Nanos(crate::clock::nanos_since_boot()) - } - - /// The absolute deadline as the one-shot's span from now; one already past - /// becomes the floor rather than an interrupt at once. - fn set_timer(&self, deadline: Nanos) { - irqchip::arm_one_shot(deadline.0.saturating_sub(self.now().0)); - } - - fn stop_timer(&self) { - irqchip::stop_timer(); - } - - /// `WFI` with interrupts masked, then unmasked: a pending interrupt wakes - /// `WFI` whatever `DAIF` says (Arm ARM K.a, D1.6.2), so a wake that lands - /// between the decision and the wait is taken right after it, not slept through. - fn halt(&self) { - // SAFETY: waits for an interrupt and unmasks `I` and `F`; touches no memory. - unsafe { core::arch::asm!("wfi", "msr daifclr, #3", "isb", options(nomem, nostack)) }; - } - - /// An SGI is how a remote CPU's `need_resched` gets set — there is no way to write it directly. - fn need_resched(&self, cpu: CpuId) { - if cpu.0 == percpu::cpu_id() { - crate::preempt::set_need_resched(); - } else { - self.kick(cpu); - } - } - - fn trace(&self, ev: TraceEvent) { - crate::trace::record(ev); - } - - /// Diagnostic builds arm a periodic wake before halting so a quiescent CPU - /// still reports; and a boot under a deadline re-arms after the wake, as - /// x86-64's does, because the poll rests on some CPU taking a timer interrupt. - fn idle_wait(&self, token: SleepToken) { - let _consumed = token; - #[cfg(feature = "boot-actuators")] - if crate::actuator::diag_tick() { - irqchip::arm_within(DIAG_TICK_NS); - } - self.halt(); - if crate::deadline::armed() { - irqchip::arm_within(QUANTUM_NS); - } - } -} - -/// Longest sleep on a `diag-tick` build; kept under `heartbeat`'s reporting period so a healthy CPU reports on every line. -#[cfg(feature = "boot-actuators")] -const DIAG_TICK_NS: u64 = 100_000_000; - -/// Which context each CPU last switched onto; read by [`report_contexts`] on crash. -static RUNNING_CTX: [core::sync::atomic::AtomicU64; crate::sched::MAX_CPUS] = - [const { core::sync::atomic::AtomicU64::new(0) }; crate::sched::MAX_CPUS]; - -/// Prints which CPU is standing on which context and stack, on every kernel crash. -/// -/// Allocates, locks or formats nothing but integers, since a crash may already -/// hold any lock this could try to take. -pub fn report_contexts(sp: u64, subject: Option) { - let me = percpu::cpu_id() as usize; - let count = (super::smp::cpu_count() as usize).min(crate::sched::MAX_CPUS); - let mine = RUNNING_CTX.get(me).map_or(0, |slot| slot.load(core::sync::atomic::Ordering::Relaxed)); - let subject = subject.unwrap_or(mine); - crate::log!(" Contexts: cpu{me} crashed at sp={sp:#018x}, asking about ctx {subject:#x}"); - for (cpu, slot) in RUNNING_CTX.iter().enumerate().take(count) { - let held = slot.load(core::sync::atomic::Ordering::Relaxed); - if !crate::mm::is_kernel_addr(held) || !held.is_multiple_of(8) { - crate::log!(" cpu{cpu} is on ctx {held:#x} (never switched, or not a context)"); - continue; - } - // SAFETY: a pointer this kernel's own `Hw::switch` stored, into the boxed, always-mapped direct map. - let ctx = unsafe { &*(held as *const KernelCtx) }; - let top = ctx.kernel_stack_top; - match ctx.id { - None => crate::log!(" cpu{cpu} is on ctx {held:#x} (its idle context) saved_sp={:#018x}", ctx.sp), - Some(id) => crate::log!( - " cpu{cpu} is on ctx {held:#x} pid={} tid={} stack_top={top:#018x} saved_sp={:#018x}{}", - id.0.raw(), - id.1.raw(), - ctx.sp, - if cpu != me && sp <= top && sp > top.wrapping_sub(crate::process::KERNEL_STACK_SIZE as u64) { - " <== AND THIS CRASH IS ON THAT STACK" - } else { - "" - }, - ), - } - } - crate::mm::report_on_crash(); +/// `WFI` with interrupts masked, then unmasked: a pending interrupt wakes +/// `WFI` whatever `DAIF` says (Arm ARM K.a, D1.6.2), so a wake that lands +/// between the decision and the wait is taken right after it, not slept through. +pub fn halt() { + // SAFETY: waits for an interrupt and unmasks `I` and `F`; touches no memory. + unsafe { core::arch::asm!("wfi", "msr daifclr, #3", "isb", options(nomem, nostack)) }; } /// Panics before the switch's `ret` would land somewhere that makes the failure unnameable. @@ -194,7 +87,7 @@ impl Hw for KernelHw { incoming.root.activate(); } } - RUNNING_CTX[percpu::cpu_id() as usize].store(restore as u64, core::sync::atomic::Ordering::Relaxed); + crate::hw::note_running(restore); context_switch(&raw mut (*save).sp, sp); } } diff --git a/kernel/src/arch/aarch64/irqchip.rs b/kernel/src/arch/aarch64/irqchip.rs index 355d46f1c3..5211649956 100644 --- a/kernel/src/arch/aarch64/irqchip.rs +++ b/kernel/src/arch/aarch64/irqchip.rs @@ -19,20 +19,34 @@ use core::sync::atomic::{AtomicU32, AtomicU64, Ordering::Relaxed}; use toyos_acpi::MadtEntry; +use toyos_gicv3::{packed_affinity, FRAME}; use super::{cpu, percpu}; use crate::drivers::acpi::direct_phys; use crate::log; use crate::mm::policy::MmioPolicy; use crate::mm::{DirectMap, Mmio}; -use crate::time::{Duration, Floor}; +use crate::hw::MIN_ONE_SHOT; + +/// Every INTID this kernel names, in one space: the SGIs it raises, then the +/// identities the generic drivers program for a message-signalled interrupt, +/// which `super::msi_message` refuses on this machine. +#[repr(u8)] +pub(super) enum Intid { + /// Asks a CPU for a scheduler pass: x86-64's kick, which rides the + /// timer's vector there. + Kick = 0, + /// `crate::log::nested`'s delivery, which [`send_self`] raises. + LogNest, + /// What `irq-storm` floods this CPU with. + Storm, + Hda, + VirtioSound, +} -/// The SGI that asks a CPU for a scheduler pass: x86-64's kick, which rides -/// the timer's vector there. -pub(super) const SGI_KICK: u32 = 0; -/// The SGI `irq-storm` floods this CPU with. +pub(super) const SGI_KICK: u32 = Intid::Kick as u32; #[cfg(feature = "boot-actuators")] -pub(super) const SGI_STORM: u32 = 2; +pub(super) const SGI_STORM: u32 = Intid::Storm as u32; /// `GICD_CTLR`, and its `ARE` (affinity routing — `ARE_NS` as a non-secure /// access sees it) and Group 1 enable (`EnableGrp1` or `EnableGrp1A`, the @@ -44,16 +58,12 @@ const RWP: u32 = 1 << 31; /// `GICD_PIDR2.ArchRev`, bits 7:4: 3 is GICv3, 4 is GICv4. const GICD_PIDR2: u64 = 0xFFE8; -/// A redistributor's frames: `RD_base`, then `SGI_base` 64 KiB above it, and -/// two more for virtual LPIs where `GICR_TYPER.VLPIS` says so. +/// A redistributor's frames: `RD_base`, then `SGI_base` 64 KiB above it. const GICR_CTLR: u64 = 0x0000; const GICR_TYPER: u64 = 0x0008; const GICR_WAKER: u64 = 0x0014; const WAKER_PROCESSOR_SLEEP: u32 = 1 << 1; const WAKER_CHILDREN_ASLEEP: u32 = 1 << 2; -const TYPER_VLPIS: u64 = 1 << 1; -const TYPER_LAST: u64 = 1 << 4; -const FRAME: u64 = 0x1_0000; const SGI_BASE: u64 = FRAME; const GICR_IGROUPR0: u64 = SGI_BASE + 0x0080; const GICR_ISENABLER0: u64 = SGI_BASE + 0x0100; @@ -170,7 +180,7 @@ pub fn init(rsdp_addr: u64) { stop_timer_hardware(); let enabled = 1 << SGI_KICK | 1 << timer.gsiv; #[cfg(feature = "boot-actuators")] - let enabled = enabled | 1 << u32::from(super::trap::LOG_NEST_VECTOR) | 1 << SGI_STORM; + let enabled = enabled | 1 << Intid::LogNest as u32 | 1 << SGI_STORM; redistributor.write_u32(GICR_ISENABLER0, enabled); // The CPU interface in system registers, as the declaration says. @@ -215,27 +225,11 @@ pub fn init(rsdp_addr: u64) { ); } -/// `MPIDR_EL1`'s four affinity fields packed as `cpu::hardware_id` packs -/// them, and as `GICR_TYPER` carries them in its top word. -fn packed_affinity(mpidr: u64) -> u32 { - ((mpidr & 0xFF_FFFF) | ((mpidr >> 8) & 0xFF00_0000)) as u32 -} - /// The redistributor frame in `[base, base + length)` whose affinity is `me`. fn find_redistributor(base: u64, length: u64, me: u32) -> Option { let range = crate::mm::paging::map_mmio(base, length, MmioPolicy::Uncacheable); - let mut offset = 0; - while offset + 2 * FRAME <= length { - let typer = range.read_u64(offset + GICR_TYPER); - if (typer >> 32) as u32 == me { - return Some(Mmio::new(DirectMap::from_phys(base + offset), 2 * FRAME)); - } - if typer & TYPER_LAST != 0 { - return None; - } - offset += if typer & TYPER_VLPIS != 0 { 4 * FRAME } else { 2 * FRAME }; - } - None + let offset = toyos_gicv3::find_redistributor(length, me, |at| range.read_u64(at + GICR_TYPER))?; + Some(Mmio::new(DirectMap::from_phys(base + offset), 2 * FRAME)) } /// The INTID the GIC hands this CPU, or `None` when it answered spurious. @@ -261,9 +255,7 @@ pub(super) fn timer_intid() -> u32 { /// Raise SGI `intid` on the CPU whose packed affinity is `target`. fn sgi(intid: u32, target: u32) { - let (aff0, aff1, aff2, aff3) = - (u64::from(target & 0xFF), u64::from(target >> 8 & 0xFF), u64::from(target >> 16 & 0xFF), u64::from(target >> 24)); - let value = aff3 << 48 | (aff0 >> 4) << 44 | aff2 << 32 | u64::from(intid) << 24 | aff1 << 16 | 1 << (aff0 & 0xF); + let value = toyos_gicv3::sgi1r(intid, target); // SAFETY: writes `ICC_SGI1R_EL1`, which raises an SGI and touches no // memory; the `ISB` sends it before whatever follows. unsafe { core::arch::asm!("msr S3_0_C12_C11_5, {}", "isb", in(reg) value, options(nomem, nostack, preserves_flags)) }; @@ -297,19 +289,22 @@ pub fn send_nmi(_cpu: u32) { /// Nothing to stop: before the port's stage 5 the boot CPU is the only one running. pub fn stop_other_cpus() {} -/// The shortest one-shot this kernel arms: above an interrupt's entry and -/// return, as x86-64's floor is. -const MIN_ONE_SHOT: Floor = Floor::policy(Duration::from_micros(10), "above an interrupt entry and eret, a thousandth of QUANTUM_NS"); - -fn counter_ticks(nanos: u64) -> u64 { - crate::clock::tsc_ticks(nanos).max(crate::clock::tsc_ticks(MIN_ONE_SHOT.nanos())).max(1) -} - /// `CNTV_CTL_EL0.ENABLE`; `IMASK` stays clear. const TIMER_ENABLE: u64 = 1; +/// `CNTV_CTL_EL0.ISTATUS`: the timer's condition is met. +#[cfg(feature = "boot-actuators")] +const TIMER_ISTATUS: u64 = 1 << 2; + +/// [`MIN_ONE_SHOT`] in counter ticks, and never none. +fn floor_ticks() -> u64 { + crate::clock::counter_ticks(MIN_ONE_SHOT.nanos()).max(1) +} -/// Fire `ticks` counter ticks from now, and remember it as what an EL1 fire re-arms with. +/// The only write of the comparator: fire `ticks` counter ticks from now, or +/// after [`MIN_ONE_SHOT`] if that is longer, and remember the span as what an +/// EL1 fire re-arms with. fn arm_ticks(ticks: u64) { + let ticks = ticks.max(floor_ticks()); percpu::set_armed_ticks(ticks); // SAFETY: the EL1 virtual timer's comparator and control; CPACR has // nothing to say about them and `CNTKCTL_EL1` keeps EL0 out. @@ -332,14 +327,14 @@ fn stop_timer_hardware() { /// This CPU's timer, armed to fire `nanos` from now, or after [`MIN_ONE_SHOT`] if that is longer. pub fn arm_one_shot(nanos: u64) { - arm_ticks(counter_ticks(nanos)); + arm_ticks(crate::clock::counter_ticks(nanos)); crate::trace::trace(crate::trace::Kind::TimerArm, nanos as u32); } /// This CPU's timer, armed to fire within `nanos`: sooner than it is armed /// for, or armed if it is stopped. pub fn arm_within(nanos: u64) { - let want = counter_ticks(nanos); + let want = crate::clock::counter_ticks(nanos); let remaining = match percpu::armed_ticks() { 0 => want, _ => { @@ -349,7 +344,7 @@ pub fn arm_within(nanos: u64) { cval.saturating_sub(cpu::counter()) } }; - arm_ticks(want.min(remaining).max(1)); + arm_ticks(want.min(remaining)); } /// Stop the timer: no interrupt until it is armed again. @@ -367,3 +362,34 @@ pub(super) fn rearm() { ticks => arm_ticks(ticks), } } + +/// `timer-floor`: this CPU's timer made due with interrupts masked, then +/// asked to fire within a quantum, which leaves it nothing to fire within; +/// then [`FLOOR_FIRES`] of its interrupts taken at EL1 with them open. A +/// re-arm shorter than [`MIN_ONE_SHOT`] is what an EL1 fire repeats, and one +/// that re-fires on its own return leaves this CPU no progress to say +/// anything with. +#[cfg(feature = "boot-actuators")] +pub fn floor_selftest() { + const FLOOR_FIRES: u32 = 100; + let _guard = crate::arch::IrqGuard::close(); + arm_one_shot(0); + let due = || { + let ctl: u64; + // SAFETY: reads the EL1 virtual timer's control. + unsafe { core::arch::asm!("mrs {}, cntv_ctl_el0", out(reg) ctl, options(nomem, nostack, preserves_flags)) }; + ctl & TIMER_ISTATUS != 0 + }; + settles(100, "the timer armed for its floor", due); + arm_within(toyos_sched::fair::QUANTUM_NS); + let (armed, floor) = (percpu::armed_ticks(), floor_ticks()); + let before = percpu::kernel_timer_fires(); + cpu::enable_interrupts(); + while percpu::kernel_timer_fires().wrapping_sub(before) < FLOOR_FIRES { + core::hint::spin_loop(); + } + cpu::disable_interrupts(); + stop_timer(); + let verdict = if armed >= floor { "PASS" } else { "FAIL" }; + log!("timer-floor: {verdict} armed={armed} floor={floor} ticks: {FLOOR_FIRES} fires taken at EL1 re-armed for the floor"); +} diff --git a/kernel/src/arch/aarch64/percpu.rs b/kernel/src/arch/aarch64/percpu.rs index c1471f4c6d..5d7a8b8f30 100644 --- a/kernel/src/arch/aarch64/percpu.rs +++ b/kernel/src/arch/aarch64/percpu.rs @@ -36,8 +36,8 @@ pub struct PerCpu { need_resched: AtomicU8, fault_state: AtomicU8, /// Timer interrupts taken at EL1, which only count and ask for a pass. - ring0_timer_fires: AtomicU32, - last_seen_ring0_fires: AtomicU32, + kernel_timer_fires: AtomicU32, + last_seen_kernel_timer_fires: AtomicU32, /// Counter ticks the timer was last armed for: what a timer interrupt /// taken at EL1 re-arms it with, and zero when it is stopped. armed_ticks: AtomicU64, @@ -83,8 +83,8 @@ pub fn init_bsp() { preempt_count: AtomicU32::new(0), need_resched: AtomicU8::new(0), fault_state: AtomicU8::new(CpuFaultState::Normal as u8), - ring0_timer_fires: AtomicU32::new(0), - last_seen_ring0_fires: AtomicU32::new(0), + kernel_timer_fires: AtomicU32::new(0), + last_seen_kernel_timer_fires: AtomicU32::new(0), armed_ticks: AtomicU64::new(0), kernel_stack: AtomicU64::new(0), idle_stack_top: crate::sched::idle_stack::alloc(), @@ -298,20 +298,20 @@ pub fn faulting() -> bool { } /// Timer interrupts taken at EL1. -pub fn ring0_timer_fires() -> u32 { - this().ring0_timer_fires.load(Relaxed) +pub fn kernel_timer_fires() -> u32 { + this().kernel_timer_fires.load(Relaxed) } -pub fn last_seen_ring0_fires() -> u32 { - this().last_seen_ring0_fires.load(Relaxed) +pub fn last_seen_kernel_timer_fires() -> u32 { + this().last_seen_kernel_timer_fires.load(Relaxed) } -pub fn set_last_seen_ring0_fires(v: u32) { - this().last_seen_ring0_fires.store(v, Relaxed); +pub fn set_last_seen_kernel_timer_fires(v: u32) { + this().last_seen_kernel_timer_fires.store(v, Relaxed); } -pub(super) fn note_ring0_timer_fire() { - this().ring0_timer_fires.fetch_add(1, Relaxed); +pub(super) fn note_kernel_timer_fire() { + this().kernel_timer_fires.fetch_add(1, Relaxed); } /// What the timer was last armed for, in counter ticks; zero is stopped. diff --git a/kernel/src/arch/aarch64/tlb.rs b/kernel/src/arch/aarch64/tlb.rs index 0e0792bc6d..51e836cf66 100644 --- a/kernel/src/arch/aarch64/tlb.rs +++ b/kernel/src/arch/aarch64/tlb.rs @@ -96,14 +96,14 @@ pub fn bench() { panic!("tlb-shootdown-bench: AArch64 invalidates by broadcast, so there is no IPI round trip to measure"); } -/// x86-64's delays an acknowledgement, and there is none here. +/// x86-64's delays an acknowledgement, and there is none here: refused. #[cfg(feature = "test-actuators")] pub fn debug_arm_ack_delay(_nanos: u64) -> u64 { - panic!("SYS_DEBUG: AArch64 invalidates by broadcast, so there is no acknowledgement to delay"); + toyos_abi::syscall::SyscallError::NotSupported.to_u64() } -/// x86-64's delays an acknowledgement, and there is none here. +/// x86-64's delays an acknowledgement, and there is none here: refused. #[cfg(feature = "test-actuators")] pub fn debug_disarm_ack_delay() -> u64 { - panic!("SYS_DEBUG: AArch64 invalidates by broadcast, so there is no acknowledgement to delay"); + toyos_abi::syscall::SyscallError::NotSupported.to_u64() } diff --git a/kernel/src/arch/aarch64/trap.rs b/kernel/src/arch/aarch64/trap.rs index 5468c6811f..a8c433f5aa 100644 --- a/kernel/src/arch/aarch64/trap.rs +++ b/kernel/src/arch/aarch64/trap.rs @@ -151,13 +151,13 @@ fn irq(from_el0: bool) { // Only an EL0 tick reaches here, so the interrupted context is user // code and holds no `Lock`. assert_eq!(crate::preempt::count(), 0, "a timer interrupt from EL0 found a preempt depth"); - let hw = &super::hw::HW; + let hw = &crate::hw::HW; hw.trace(TraceEvent { ts: hw.now(), cpu: CpuId(percpu::cpu_id()), kind: TraceKind::TimerFire }); irqchip::end(intid); crate::scheduler::do_preempt(); } else { crate::preempt::set_need_resched(); - percpu::note_ring0_timer_fire(); + percpu::note_kernel_timer_fire(); irqchip::end(intid); } return; @@ -458,20 +458,9 @@ pub fn install() { } } -/// The numbers the generic drivers program as an interrupt's identity. On -/// AArch64 [`LOG_NEST_VECTOR`] is the SGI `irqchip::send_self` raises; the -/// other two name a message-signalled interrupt, which `super::msi_message` -/// refuses on this machine. -#[repr(u8)] -enum Vector { - LogNest = 1, - Hda, - VirtioSound, -} - -pub const HDA_VECTOR: u8 = Vector::Hda as u8; -pub const VIRTIO_SOUND_VECTOR: u8 = Vector::VirtioSound as u8; -pub const LOG_NEST_VECTOR: u8 = Vector::LogNest as u8; +pub const HDA_VECTOR: u8 = irqchip::Intid::Hda as u8; +pub const VIRTIO_SOUND_VECTOR: u8 = irqchip::Intid::VirtioSound as u8; +pub const LOG_NEST_VECTOR: u8 = irqchip::Intid::LogNest as u8; /// The crash report for a panic, from the frame pointer the panic handler /// stood on: the backtrace, which CPU is on which stack, and what the @@ -480,7 +469,7 @@ pub(crate) fn report_panic(message: &core::panic::PanicInfo, frame: u64) { alert!("PANIC: {}", message); log!(" Backtrace:"); crate::symbols::kernel_backtrace(frame, 20); - super::hw::report_contexts(cpu::stack_pointer(), None); + crate::hw::report_contexts(cpu::stack_pointer(), None); let Some(pid) = percpu::current_pid() else { return }; log!(" Running: pid={} tid={:?}", pid, percpu::current_tid()); if percpu::in_syscall() { @@ -501,24 +490,22 @@ pub(crate) const fn frame_interrupts_enabled(spsr: u64) -> bool { /// Nothing to report: the vectors run on the stack they interrupted. pub(crate) fn report_fault_stack() {} -/// Every return to EL0's last word, and a new thread's first. -pub fn kernel_exit_to_user_check() { - crate::scheduler::exit_to_user(); -} - /// x86-64's lands a `#DF` on its IST stack; AArch64 has no double fault, and -/// an exception on a broken stack takes the same vector again. +/// an exception on a broken stack takes the same vector again, so the call is +/// refused. #[cfg(feature = "test-actuators")] -pub(crate) fn provoke_double_fault() -> ! { - panic!("SYS_DEBUG: AArch64 has no double fault to provoke"); +pub(crate) fn provoke_double_fault() -> u64 { + toyos_abi::syscall::SyscallError::NotSupported.to_u64() } -/// `irq-storm`: this CPU floods itself with SGIs until the timer has fired -/// `TICKS_OWED` times through the flood, then waits for every SGI it -/// sent to be taken. A tick lost, or never re-armed, leaves the flood running -/// and an SGI lost leaves the wait running, so neither says anything: the -/// harness's ceiling is the only clock this judges by, and the one verdict -/// the storm can say is `FAIL` for an SGI taken that was never sent. +/// `irq-storm`: this CPU floods itself with SGIs, sending each as soon as the +/// last is taken, until the timer has fired `TICKS_OWED` times through the +/// flood, then waits for the last SGI to be taken. One at a time, because an +/// SGI sent while the last is still pending merges with it. A tick lost, or +/// never re-armed, leaves the flood running and an SGI lost leaves the wait +/// running, so neither says anything: the harness's ceiling is the only clock +/// this judges by, and the one verdict the storm can say is `FAIL` for an SGI +/// taken that was never sent. #[cfg(feature = "boot-actuators")] pub(crate) mod storm { use core::sync::atomic::{AtomicBool, AtomicU64, Ordering::Relaxed}; @@ -556,8 +543,10 @@ pub(crate) mod storm { let mut sent = 0u64; crate::arch::cpu::enable_interrupts(); while TICKS.load(Relaxed) < TICKS_OWED { - irqchip::send_self(irqchip::SGI_STORM as u8); - sent += 1; + if SGIS.load(Relaxed) == sent { + irqchip::send_self(irqchip::SGI_STORM as u8); + sent += 1; + } } RUNNING.store(false, Relaxed); irqchip::stop_timer(); diff --git a/kernel/src/arch/x86_64/apic.rs b/kernel/src/arch/x86_64/apic.rs index 11ab84c42b..31b7902968 100644 --- a/kernel/src/arch/x86_64/apic.rs +++ b/kernel/src/arch/x86_64/apic.rs @@ -1,8 +1,9 @@ use core::sync::atomic::{AtomicBool, AtomicU32, Ordering}; use super::{cpu, percpu}; +use crate::hw::MIN_ONE_SHOT; use crate::log; -use crate::time::{Delay, Duration, Floor}; +use crate::time::{Delay, Duration}; /// The local APIC registers and MSRs this file may name. // Every variant is an architectural local-APIC register touching no memory or control transfer, so none can make `Reg::write`'s unsafe wrmsr unsound. @@ -226,12 +227,6 @@ pub fn init_timer() { log!("LAPIC timer: {} ticks/10ms, so {}Hz", ticks_10ms, ticks_10ms as u64 * 100); } -// Floor on every arm: a count that expires before the interrupt it schedules retires cannot outlast itself and livelocks the CPU forever. -const MIN_ONE_SHOT: Floor = Floor::policy( - Duration::from_micros(10), - "above an interrupt entry and iretq, a thousandth of QUANTUM_NS", -); - // The only path to Reg::TimerInit / last_armed_ticks — the floor is enforced once here, not at each of the three call sites. struct OneShot(u32); diff --git a/kernel/src/arch/x86_64/boot.rs b/kernel/src/arch/x86_64/boot.rs index b299f725df..494450576d 100644 --- a/kernel/src/arch/x86_64/boot.rs +++ b/kernel/src/arch/x86_64/boot.rs @@ -138,5 +138,8 @@ pub fn interrupt_selftests() { if crate::actuator::unclaimed_vector_selftest() { idt::unclaimed::selftest(); } - assert!(!crate::actuator::irq_storm(), "irq-storm is the GIC's selftest, and this machine has a local APIC"); + assert!( + !crate::actuator::irq_storm() && !crate::actuator::timer_floor(), + "irq-storm and timer-floor are the GIC and generic timer's selftests, and this machine has a local APIC" + ); } diff --git a/kernel/src/arch/x86_64/hw.rs b/kernel/src/arch/x86_64/hw.rs index a677adcba2..7f224e72f1 100644 --- a/kernel/src/arch/x86_64/hw.rs +++ b/kernel/src/arch/x86_64/hw.rs @@ -1,158 +1,20 @@ -//! `KernelHw` — the kernel's side of the scheduler-core hardware boundary. -//! -//! Everything here is x2APIC, TSC or a single instruction; nothing here -//! makes scheduling decisions. The simulator that exercises the scheduler -//! core replaces this file and nothing else. +//! x86-64's half of `crate::hw::KernelHw`: the halt and the context switch. use core::arch::asm; -use toyos_sched::cpu::SleepToken; use toyos_sched::cpu::RunToken; -use toyos_sched::fair::QUANTUM_NS; -use toyos_sched::hw::{CpuId, Hw, Kicker, Machine, Nanos, TraceEvent}; +use toyos_sched::hw::Hw; use toyos_sched::task::{TaskAccounting, TaskKey}; -use crate::arch::{apic, cpu, percpu}; +use crate::arch::{cpu, percpu}; +use crate::hw::{report_contexts, KernelHw}; use super::switch::context_switch; use crate::sched::payload::{KernelCtx, KernelPayload}; -/// The one instance; zero-sized, holds no per-CPU state. -pub static HW: KernelHw = KernelHw; - -pub struct KernelHw; - -/// The scheduler clock, in raw nanoseconds. -pub fn now_ns() -> u64 { - HW.now().0 -} - -impl Kicker for KernelHw { - fn kick(&self, target: CpuId) { - apic::kick_cpu(target.0); - } -} - -impl Machine for KernelHw { - fn now(&self) -> Nanos { - Nanos(crate::clock::nanos_since_boot()) - } - - /// Converts the trait's absolute deadline to the one-shot timer's relative count; a deadline - /// already past saturates to zero rather than firing immediately, which would spin the Ring 0 - /// stub in a reload loop. - fn set_timer(&self, deadline: Nanos) { - apic::arm_one_shot(deadline.0.saturating_sub(self.now().0)); - } - - fn stop_timer(&self) { - apic::stop_timer(); - } - - fn halt(&self) { - // SAFETY: `sti; hlt` is the atomic enable-and-wait pair — a wake landing between the two is not lost. - unsafe { asm!("sti; hlt", options(nomem, nostack)); } - } - - /// A kick IPI is how a remote CPU's `need_resched` gets set — there is no way to write it directly. - fn need_resched(&self, cpu: CpuId) { - if cpu.0 == percpu::cpu_id() { - crate::preempt::set_need_resched(); - } else { - self.kick(cpu); - } - } - - fn trace(&self, ev: TraceEvent) { - crate::trace::record(ev); - } - - /// Diagnostic builds arm a periodic wake before halting so a quiescent CPU still reports. - fn idle_wait(&self, token: SleepToken) { - let _consumed = token; - #[cfg(feature = "boot-actuators")] - if crate::actuator::diag_tick() { - apic::arm_within(DIAG_TICK_NS); - } - self.halt(); - // **A CPU that is executing has a one-shot armed, and this is where - // that becomes true again.** `TimerPlan::Stop` left this one at zero - // before the halt above and only a pass reaching `apply_timer` arms - // another, so a CPU woken by an IPI or a device — never by its own - // timer, which is stopped — and then held in Ring 0 takes no timer - // interrupt at all. `crate::deadline`'s poll and - // `crate::hardlockup`'s sample both rest on some CPU taking one. - // Arming earlier than the scheduler planned is a spurious pass and - // never a missed deadline (`toyos_sched::timer::TimerPlan`), and the - // next pass replaces it either way. - // - // Asked, because it is only those two that need it: a boot under no - // bound pays an x2APIC read and two writes per wake for nothing. - if crate::deadline::armed() { - apic::arm_within(QUANTUM_NS); - } - } -} - -/// Longest sleep on a `diag-tick` build; kept under `heartbeat`'s reporting period so a healthy CPU reports on every line. -#[cfg(feature = "boot-actuators")] -const DIAG_TICK_NS: u64 = 100_000_000; - -/// Which context each CPU last switched onto; read by [`report_contexts`] on crash, since a -/// sibling's real `CpuSched` is `!Sync` and unreadable directly. -static RUNNING_CTX: [core::sync::atomic::AtomicU64; crate::sched::MAX_CPUS] = - [const { core::sync::atomic::AtomicU64::new(0) }; crate::sched::MAX_CPUS]; - -/// Prints which CPU is standing on which context and stack, on every kernel crash. -/// -/// `subject` (`None` for this CPU's own) is the context flagged as "the same". -/// -/// Allocates, locks or formats nothing but integers, since a crash may already hold any lock this -/// could try to take. -pub fn report_contexts(rsp: u64, subject: Option) { - let me = percpu::cpu_id() as usize; - let count = (crate::arch::smp::cpu_count() as usize).min(crate::sched::MAX_CPUS); - let mine = RUNNING_CTX - .get(me) - .map_or(0, |slot| slot.load(core::sync::atomic::Ordering::Relaxed)); - let subject = subject.unwrap_or(mine); - crate::log!(" Contexts: cpu{me} crashed at rsp={rsp:#018x}, asking about ctx {subject:#x}"); - for (cpu, slot) in RUNNING_CTX.iter().enumerate().take(count) { - let held = slot.load(core::sync::atomic::Ordering::Relaxed); - if !crate::mm::is_kernel_addr(held) || !held.is_multiple_of(8) { - crate::log!(" cpu{cpu} is on ctx {held:#x} (never switched, or not a context)"); - continue; - } - // SAFETY: `held` is a pointer this kernel's own `Hw::switch` stored, into the boxed, always-mapped direct map. - let ctx = unsafe { &*(held as *const KernelCtx) }; - let top = ctx.kernel_stack_top; - let same = held == subject && cpu != me; - // `top != 0` excludes idle contexts, whose stack top is zero by construction — the - // containment test below never fires for one; that is a gap in this report, not a bug. - let on_its_stack = cpu != me - && top != 0 - && rsp <= top - && rsp > top.wrapping_sub(crate::process::KERNEL_STACK_SIZE as u64); - // idle's `kernel_stack_top` is zero by construction; rendering it as a task would misread as corruption. - match ctx.id { - None => crate::log!( - " cpu{cpu} is on ctx {held:#x} (its idle context) stack_top={top:#018x} \ - saved_rsp={:#018x}{}{}", - ctx.sp, - if same { " <== THE SAME CONTEXT" } else { "" }, - if top == 0 { "" } else { " <== AN IDLE CONTEXT'S STACK TOP IS ZERO BY CONSTRUCTION" }, - ), - Some(id) => crate::log!( - " cpu{cpu} is on ctx {held:#x} pid={} tid={} stack_top={top:#018x} \ - saved_rsp={:#018x}{}{}", - id.0.raw(), - id.1.raw(), - ctx.sp, - if same { " <== THE SAME CONTEXT" } else { "" }, - if on_its_stack { " <== AND THIS CRASH IS ON THAT STACK" } else { "" }, - ), - } - } - crate::mm::report_on_crash(); +/// `sti; hlt`, the atomic enable-and-wait pair: a wake landing between the two is not lost. +pub fn halt() { + // SAFETY: enables interrupts and waits for one; touches no memory. + unsafe { asm!("sti; hlt", options(nomem, nostack)); } } /// Panics before the wild `ret` would restore register state that makes the failure unnameable. @@ -450,8 +312,7 @@ impl Hw for KernelHw { incoming.root.activate(); } } - RUNNING_CTX[percpu::cpu_id() as usize] - .store(restore as u64, core::sync::atomic::Ordering::Relaxed); + crate::hw::note_running(restore); // AMD's `SYSRET` reloads SS's selector but not its cached descriptor, so // a `sysretq` onto a task an IDT entry left with SS NULL hands userland an // unusable SS; reloading it here keeps it valid (`X86_BUG_SYSRET_SS_ATTRS`). diff --git a/kernel/src/arch/x86_64/percpu.rs b/kernel/src/arch/x86_64/percpu.rs index 9c92b71899..c9050d340c 100644 --- a/kernel/src/arch/x86_64/percpu.rs +++ b/kernel/src/arch/x86_64/percpu.rs @@ -636,15 +636,15 @@ pub fn percpu_ptr() -> *mut PerCpu { } /// Ring 0 timer fires the assembly stub has taken; written with a plain `inc` (IF clear there). -pub fn ring0_timer_fires() -> u32 { +pub fn kernel_timer_fires() -> u32 { gs::read_u32::() } -pub fn last_seen_ring0_fires() -> u32 { +pub fn last_seen_kernel_timer_fires() -> u32 { gs::read_u32::() } -pub fn set_last_seen_ring0_fires(v: u32) { +pub fn set_last_seen_kernel_timer_fires(v: u32) { gs::write_u32::(v); } diff --git a/kernel/src/clock.rs b/kernel/src/clock.rs index 439b8c83a4..f2337c3ac9 100644 --- a/kernel/src/clock.rs +++ b/kernel/src/clock.rs @@ -88,14 +88,14 @@ pub fn now() -> Instant { /// The [`cpu::rdtsc`] value `nanos` in the future, for a wait loop that must /// not call the nanosecond clock. pub fn tsc_deadline(nanos: u64) -> u64 { - cpu::counter().saturating_add(tsc_ticks(nanos)) + cpu::counter().saturating_add(counter_ticks(nanos)) } -/// `nanos` as a count of TSC ticks: a span converted once and then compared -/// against `rdtsc` differences, which is what a sampler that may not divide +/// `nanos` as a count of [`cpu::counter`] ticks: a span converted once and then compared +/// against counter differences, which is what a sampler that may not divide /// needs. Before [`init`] the period is unknown and this is zero, so a bound /// derived from it is one its arm has to refuse. -pub fn tsc_ticks(nanos: u64) -> u64 { +pub fn counter_ticks(nanos: u64) -> u64 { let period_fs = TSC_PERIOD_FS.load(Relaxed); if period_fs == 0 { return 0; @@ -103,7 +103,7 @@ pub fn tsc_ticks(nanos: u64) -> u64 { ((nanos as u128 * 1_000_000) / period_fs as u128) as u64 } -/// The span a count of [`tsc_ticks`] stands for, for a caller that measured +/// The span a count of [`counter_ticks`] stands for, for a caller that measured /// before there was a period to measure with and converts once, afterwards. /// Zero while the period is unknown, so a span taken on a machine that never /// calibrated reads as no time rather than as an invented one. diff --git a/kernel/src/hardlockup/mod.rs b/kernel/src/hardlockup/mod.rs index 33b88039a1..b91a0022b2 100644 --- a/kernel/src/hardlockup/mod.rs +++ b/kernel/src/hardlockup/mod.rs @@ -132,15 +132,15 @@ static SPIN_AT: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; #[must_use] pub fn start(deadline_ms: u64) -> Armed { let ms = toyos_tco::hard_lockup_bound_ms(deadline_ms); - let ticks = crate::clock::tsc_ticks(ms.saturating_mul(1_000_000)); - let per_ms = crate::clock::tsc_ticks(1_000_000); + let ticks = crate::clock::counter_ticks(ms.saturating_mul(1_000_000)); + let per_ms = crate::clock::counter_ticks(1_000_000); if ms == 0 || ticks == 0 || per_ms == 0 { return Armed { ms: 0, sampled: false }; } BOUND_MS.store(ms, Relaxed); BOUND_TSC.store(ticks, Relaxed); TICKS_PER_MS.store(per_ms, Relaxed); - PERIOD.store(crate::clock::tsc_ticks(SAMPLE_NS), Relaxed); + PERIOD.store(crate::clock::counter_ticks(SAMPLE_NS), Relaxed); arm_this_cpu(); Armed { ms, sampled: has_a_counter(percpu::cpu_id() as usize) } } diff --git a/kernel/src/hw.rs b/kernel/src/hw.rs new file mode 100644 index 0000000000..c3268132ba --- /dev/null +++ b/kernel/src/hw.rs @@ -0,0 +1,164 @@ +//! `KernelHw` — the kernel's side of the scheduler-core hardware boundary: the +//! one-shot timer, the kick, the halt and the context switch. Nothing here +//! makes a scheduling decision; the simulator that exercises the scheduler +//! core replaces this and nothing else. +//! +//! The halt and the switch are the architecture's (`crate::arch::hw`); +//! everything else is here, once, reaching the machine through +//! `crate::arch::irqchip`. + +use core::sync::atomic::{AtomicU64, Ordering::Relaxed}; + +use toyos_sched::cpu::SleepToken; +use toyos_sched::fair::QUANTUM_NS; +use toyos_sched::hw::{CpuId, Kicker, Machine, Nanos, TraceEvent}; + +use crate::arch::{irqchip, percpu}; +use crate::sched::payload::KernelCtx; +use crate::time::{Duration, Floor}; + +/// The one instance; zero-sized, holds no per-CPU state. +pub static HW: KernelHw = KernelHw; + +pub struct KernelHw; + +/// The scheduler clock, in raw nanoseconds. +pub fn now_ns() -> u64 { + HW.now().0 +} + +/// The shortest one-shot either timer is armed for, whoever asks: a count +/// that expires before the interrupt it schedules retires re-fires on the +/// return from it, and a context held with interrupts open makes no progress. +pub(crate) const MIN_ONE_SHOT: Floor = + Floor::policy(Duration::from_micros(10), "above an interrupt's entry and return"); + +impl Kicker for KernelHw { + fn kick(&self, target: CpuId) { + irqchip::kick_cpu(target.0); + } +} + +impl Machine for KernelHw { + fn now(&self) -> Nanos { + Nanos(crate::clock::nanos_since_boot()) + } + + /// The absolute deadline as the one-shot's span from now; one already + /// past becomes [`MIN_ONE_SHOT`] rather than an interrupt at once. + fn set_timer(&self, deadline: Nanos) { + irqchip::arm_one_shot(deadline.0.saturating_sub(self.now().0)); + } + + fn stop_timer(&self) { + irqchip::stop_timer(); + } + + fn halt(&self) { + crate::arch::hw::halt(); + } + + /// A kick is how a remote CPU's `need_resched` gets set — there is no way to write it directly. + fn need_resched(&self, cpu: CpuId) { + if cpu.0 == percpu::cpu_id() { + crate::preempt::set_need_resched(); + } else { + self.kick(cpu); + } + } + + fn trace(&self, ev: TraceEvent) { + crate::trace::record(ev); + } + + /// Diagnostic builds arm a periodic wake before halting so a quiescent CPU still reports. + fn idle_wait(&self, token: SleepToken) { + let _consumed = token; + #[cfg(feature = "boot-actuators")] + if crate::actuator::diag_tick() { + irqchip::arm_within(DIAG_TICK_NS); + } + self.halt(); + // **A CPU that is executing has a one-shot armed, and this is where + // that becomes true again.** `TimerPlan::Stop` left this one at zero + // before the halt above and only a pass reaching `apply_timer` arms + // another, so a CPU woken by a kick or a device — never by its own + // timer, which is stopped — and then held in the kernel takes no timer + // interrupt at all. `crate::deadline`'s poll and `crate::hardlockup`'s + // sample both rest on some CPU taking one. Arming earlier than the + // scheduler planned is a spurious pass and never a missed deadline + // (`toyos_sched::timer::TimerPlan`), and the next pass replaces it + // either way. + // + // Asked, because it is only those two that need it: a boot under no + // bound pays a timer read and two writes per wake for nothing. + if crate::deadline::armed() { + irqchip::arm_within(QUANTUM_NS); + } + } +} + +/// Longest sleep on a `diag-tick` build; kept under `heartbeat`'s reporting period so a healthy CPU reports on every line. +#[cfg(feature = "boot-actuators")] +const DIAG_TICK_NS: u64 = 100_000_000; + +/// Which context each CPU last switched onto; read by [`report_contexts`] on crash, since a +/// sibling's real `CpuSched` is `!Sync` and unreadable directly. +static RUNNING_CTX: [AtomicU64; crate::sched::MAX_CPUS] = + [const { AtomicU64::new(0) }; crate::sched::MAX_CPUS]; + +/// This CPU is about to stand on `ctx`: `Hw::switch`'s last word before the stack moves. +pub(crate) fn note_running(ctx: *const KernelCtx) { + RUNNING_CTX[percpu::cpu_id() as usize].store(ctx as u64, Relaxed); +} + +/// Prints which CPU is standing on which context and stack, on every kernel crash. +/// +/// `subject` (`None` for this CPU's own) is the context flagged as "the same". +/// +/// Allocates, locks or formats nothing but integers, since a crash may already hold any lock this +/// could try to take. +pub fn report_contexts(sp: u64, subject: Option) { + let me = percpu::cpu_id() as usize; + let count = (crate::arch::smp::cpu_count() as usize).min(crate::sched::MAX_CPUS); + let mine = RUNNING_CTX.get(me).map_or(0, |slot| slot.load(Relaxed)); + let subject = subject.unwrap_or(mine); + crate::log!(" Contexts: cpu{me} crashed at sp={sp:#018x}, asking about ctx {subject:#x}"); + for (cpu, slot) in RUNNING_CTX.iter().enumerate().take(count) { + let held = slot.load(Relaxed); + if !crate::mm::is_kernel_addr(held) || !held.is_multiple_of(8) { + crate::log!(" cpu{cpu} is on ctx {held:#x} (never switched, or not a context)"); + continue; + } + // SAFETY: `held` is a pointer this kernel's own `Hw::switch` stored, into the boxed, always-mapped direct map. + let ctx = unsafe { &*(held as *const KernelCtx) }; + let top = ctx.kernel_stack_top; + let same = held == subject && cpu != me; + // `top != 0` excludes idle contexts, whose stack top is zero by construction — the + // containment test below never fires for one; that is a gap in this report, not a bug. + let on_its_stack = cpu != me + && top != 0 + && sp <= top + && sp > top.wrapping_sub(crate::process::KERNEL_STACK_SIZE as u64); + // idle's `kernel_stack_top` is zero by construction; rendering it as a task would misread as corruption. + match ctx.id { + None => crate::log!( + " cpu{cpu} is on ctx {held:#x} (its idle context) stack_top={top:#018x} \ + saved_sp={:#018x}{}{}", + ctx.sp, + if same { " <== THE SAME CONTEXT" } else { "" }, + if top == 0 { "" } else { " <== AN IDLE CONTEXT'S STACK TOP IS ZERO BY CONSTRUCTION" }, + ), + Some(id) => crate::log!( + " cpu{cpu} is on ctx {held:#x} pid={} tid={} stack_top={top:#018x} \ + saved_sp={:#018x}{}{}", + id.0.raw(), + id.1.raw(), + ctx.sp, + if same { " <== THE SAME CONTEXT" } else { "" }, + if on_its_stack { " <== AND THIS CRASH IS ON THAT STACK" } else { "" }, + ), + } + } + crate::mm::report_on_crash(); +} diff --git a/kernel/src/iod.rs b/kernel/src/iod.rs index 263612ff59..7fddc9a0a4 100644 --- a/kernel/src/iod.rs +++ b/kernel/src/iod.rs @@ -25,7 +25,7 @@ extern "C" fn body(_arg: u64) -> ! { // The SS-reload probe rides iod rather than spawning a task of its own. #[cfg(feature = "boot-actuators")] if crate::actuator::sysret_ss_probe() { - crate::hw::sysret_ss_probe(&parkable); + crate::arch::hw::sysret_ss_probe(&parkable); } // Held across the loop: a push during a drain must still find this watch armed. let armed = watch::arm( diff --git a/kernel/src/main.rs b/kernel/src/main.rs index 237682042b..5f098f0a0e 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -73,6 +73,7 @@ mod process; mod loader; mod scheduler; mod sched; +mod hw; mod iommu; mod preempt; mod irq_census; @@ -115,7 +116,6 @@ use crate::mm::policy::MmioPolicy; use alloc::boxed::Box; use alloc::sync::Arc; use arch::{cpu, percpu, smp}; -pub(crate) use arch::hw; use drivers::{acpi, gop, nvme, pci, serial, virtio_console, virtio_gpu, virtio_sound, xhci}; use toyos_abi::boot::{KernelArgs, MemoryMapEntry}; diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 216e94ceb9..b343638287 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -6,7 +6,7 @@ //! //! # Where the stop is taken, and why there //! -//! [`stops_this_thread`] is read by `scheduler::leave_ring3_if_due`, called +//! [`stops_this_thread`] is read by `scheduler::leave_user_if_due`, called //! from `kernel_exit_to_user_check` — the one function every return to Ring 3 //! in this kernel passes through: the syscall gate, every device interrupt, the //! timer, the TLB shootdown IPI, the general trap epilogue and a task's first diff --git a/kernel/src/sched/driver.rs b/kernel/src/sched/driver.rs index 9cf0038b5e..7b9797324a 100644 --- a/kernel/src/sched/driver.rs +++ b/kernel/src/sched/driver.rs @@ -864,7 +864,7 @@ pub fn for_each_parked(mut f: impl FnMut(ParkedInfo)) -> bool { /// preempt-count bracket's other half is owed. pub extern "C" fn trampoline_entry() { crate::preempt::enable_no_resched(); - crate::arch::trap::kernel_exit_to_user_check(); + crate::scheduler::exit_to_user(); } const STACK_CANARY: u64 = 0xDEAD_BEEF_CAFE_BABE; diff --git a/kernel/src/sched/dump.rs b/kernel/src/sched/dump.rs index aed0ffd54d..3a5a8ccd2f 100644 --- a/kernel/src/sched/dump.rs +++ b/kernel/src/sched/dump.rs @@ -506,8 +506,8 @@ pub mod staged { } } - /// From the Ring 3 exit check, once it has nothing more to run. - pub fn note_return_to_ring3() { + /// From the exit to user mode, once it has nothing more to run. + pub fn note_return_to_user() { if STAGED.load(Ordering::Acquire) != NOTHING && REQUEST.pending() { RETURNS.fetch_add(1, Ordering::AcqRel); } diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index 7ef96d81b0..16ee7a3f99 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -348,7 +348,7 @@ pub fn may_yield() -> bool { crate::preempt::count() == blocking_baseline() } -/// Unified preempt entry: the Ring 3 timer path, [`exit_to_user`] +/// Unified preempt entry: the user-mode timer path, [`exit_to_user`] /// and the `preempt::enable` slow path all funnel through here. #[track_caller] pub fn do_preempt() { @@ -367,7 +367,7 @@ pub fn do_preempt() { driver::pass(Dispose::None); } -/// The last thing a thread does before returning to Ring 3, if either mark it +/// The last thing a thread does before returning to user mode, if either mark it /// can carry says it never does. [`exit_to_user`] is the one caller; /// `kernel/src/quiesce.rs`'s header says why that boundary is the safe point. /// @@ -375,7 +375,7 @@ pub fn do_preempt() { /// about**: `toyos_sched::task::SafePoint` ranks them, here and in /// `CpuSched::place` alike. #[track_caller] -pub fn leave_ring3_if_due() { +pub fn leave_user_if_due() { let Some(due) = driver::current_safe_point(crate::quiesce::stops_this_thread()) else { return; }; @@ -383,17 +383,17 @@ pub fn leave_ring3_if_due() { match due { SafePoint::Stop => { driver::pass(Dispose::Stop); - unreachable!("leave_ring3_if_due: a stopped task was dispatched again"); + unreachable!("leave_user_if_due: a stopped task was dispatched again"); } SafePoint::Exit => { - // `IF` set across the teardown, as a syscall's exit runs it: its + // Interrupts open across the teardown, as a syscall's exit runs it: its // closes and address-space drop are no interrupt latency. The // depth stays this boundary's, which is `do_preempt`'s own. crate::arch::cpu::enable_interrupts(); process::leave(None); crate::arch::cpu::disable_interrupts(); driver::pass(Dispose::Exit); - unreachable!("leave_ring3_if_due: returned from the exit pass"); + unreachable!("leave_user_if_due: returned from the exit pass"); } } } @@ -404,33 +404,33 @@ pub fn leave_ring3_if_due() { /// here, and a reschedule owed since the entry is served before the thread /// sees user mode again. pub fn exit_to_user() { - flush_ring0_timer_fires_to_trace(); + flush_kernel_timer_fires_to_trace(); loop { - // A killed or stopped thread returns to Ring 3 exactly once more: never. - leave_ring3_if_due(); + // A killed or stopped thread returns to user mode exactly once more: never. + leave_user_if_due(); // `do_preempt` owns clearing `need_resched`; this function never clears it itself. if !crate::preempt::need_resched() { #[cfg(feature = "boot-actuators")] if crate::actuator::dump_in_blocking_pass() { - crate::sched::dump::staged::note_return_to_ring3(); + crate::sched::dump::staged::note_return_to_user(); } return; } assert!(!in_schedule_self(), "exit-to-user inside a scheduler pass"); - // Not an IrqGuard: both loop exits must set IF, not restore a saved value. + // Not an IrqGuard: both loop exits must open interrupts, not restore a saved value. crate::arch::cpu::enable_interrupts(); do_preempt(); crate::arch::cpu::disable_interrupts(); - flush_ring0_timer_fires_to_trace(); + flush_kernel_timer_fires_to_trace(); } } -fn flush_ring0_timer_fires_to_trace() { - let cur = percpu::ring0_timer_fires(); - let missed = cur.wrapping_sub(percpu::last_seen_ring0_fires()); +fn flush_kernel_timer_fires_to_trace() { + let cur = percpu::kernel_timer_fires(); + let missed = cur.wrapping_sub(percpu::last_seen_kernel_timer_fires()); if missed > 0 { crate::trace::trace(crate::trace::Kind::TimerFireBurst, missed); - percpu::set_last_seen_ring0_fires(cur); + percpu::set_last_seen_kernel_timer_fires(cur); } } /// The exit pass of a thread that has left its process (`process::leave`). diff --git a/src/sourcegate.rs b/src/sourcegate.rs index 9166b9b6db..3b4fc9bd7f 100644 --- a/src/sourcegate.rs +++ b/src/sourcegate.rs @@ -1435,6 +1435,7 @@ const PURE_CRATES: &[&str] = &[ "toyos-desktop/", "toyos-dma/", "toyos-elide/", + "toyos-gicv3/", "toyos-hda/", "toyos-mixer/", "toyos-pci/", diff --git a/toyos-gicv3/Cargo.toml b/toyos-gicv3/Cargo.toml new file mode 100644 index 0000000000..689de4fab4 --- /dev/null +++ b/toyos-gicv3/Cargo.toml @@ -0,0 +1,7 @@ +[package] +name = "toyos-gicv3" +description = "The GICv3's pure decisions: affinity packing, SGI addressing and the redistributor walk." +version = "0.1.0" +edition = "2021" +license = "MIT OR Apache-2.0" +publish = false diff --git a/toyos-gicv3/src/lib.rs b/toyos-gicv3/src/lib.rs new file mode 100644 index 0000000000..40e73e90d2 --- /dev/null +++ b/toyos-gicv3/src/lib.rs @@ -0,0 +1,54 @@ +//! The GICv3's pure decisions (GIC architecture specification IHI 0069H): +//! how a CPU's affinity is packed, how an SGI names the CPU it is raised on, +//! and how a redistributor region is walked to the frame of one CPU. The +//! kernel's `arch::aarch64::irqchip` reads the registers; everything here is +//! arithmetic on what it read, so the host can ask it about CPUs the boot CPU +//! alone never exercises. + +#![no_std] + +/// One 64 KiB register frame. A redistributor is two — `RD_base`, then +/// `SGI_base` — and two more for virtual LPIs where `GICR_TYPER.VLPIS` says so. +pub const FRAME: u64 = 0x1_0000; + +/// `GICR_TYPER.VLPIS`: this redistributor has the two virtual-LPI frames. +const TYPER_VLPIS: u64 = 1 << 1; +/// `GICR_TYPER.Last`: the last redistributor in its region. +const TYPER_LAST: u64 = 1 << 4; + +/// `MPIDR_EL1`'s four affinity fields packed into 32 bits, +/// `Aff3:Aff2:Aff1:Aff0`, as `GICR_TYPER` carries them in its top word. +pub const fn packed_affinity(mpidr: u64) -> u32 { + ((mpidr & 0xFF_FFFF) | ((mpidr >> 8) & 0xFF00_0000)) as u32 +} + +/// `ICC_SGI1R_EL1` raising SGI `intid` on the one CPU whose packed affinity is +/// `target`: `Aff3`, `Aff2` and `Aff1` name its cluster, `RS` the range of +/// sixteen `Aff0` values it falls in, and the target list its one bit there. +pub const fn sgi1r(intid: u32, target: u32) -> u64 { + assert!(intid < 16, "an SGI's INTID is below 16"); + let aff0 = (target & 0xFF) as u64; + let aff1 = (target >> 8 & 0xFF) as u64; + let aff2 = (target >> 16 & 0xFF) as u64; + let aff3 = (target >> 24) as u64; + aff3 << 48 | (aff0 >> 4) << 44 | aff2 << 32 | (intid as u64) << 24 | aff1 << 16 | 1 << (aff0 & 0xF) +} + +/// The offset, in a redistributor region `length` bytes long, of the +/// redistributor whose affinity is `me`; `typer` reads `GICR_TYPER` of the +/// redistributor at an offset. The walk steps by each one's own frames and +/// stops at the one marked last. +pub fn find_redistributor(length: u64, me: u32, typer: impl Fn(u64) -> u64) -> Option { + let mut offset = 0; + while offset + 2 * FRAME <= length { + let word = typer(offset); + if (word >> 32) as u32 == me { + return Some(offset); + } + if word & TYPER_LAST != 0 { + return None; + } + offset += if word & TYPER_VLPIS != 0 { 4 * FRAME } else { 2 * FRAME }; + } + None +} diff --git a/toyos-gicv3/tests/gicv3.rs b/toyos-gicv3/tests/gicv3.rs new file mode 100644 index 0000000000..73a55f8b75 --- /dev/null +++ b/toyos-gicv3/tests/gicv3.rs @@ -0,0 +1,67 @@ +use toyos_gicv3::{find_redistributor, packed_affinity, sgi1r, FRAME}; + +#[test] +fn affinity_drops_mpidr_flags_between_aff2_and_aff3() { + // Aff3 in 39:32; `U` (30), `MT` (24) and the RES1 bit 31 in between. + let mpidr = 0x12 << 32 | 1 << 31 | 1 << 30 | 1 << 24 | 0x34_5678; + assert_eq!(packed_affinity(mpidr), 0x1234_5678); +} + +#[test] +fn sgi_names_aff0_by_range_and_bit() { + // Aff0 0x13 is range 1, bit 3. + let value = sgi1r(2, 0x0A0B_0C13); + assert_eq!(value >> 44 & 0xF, 1, "RS"); + assert_eq!(value & 0xFFFF, 1 << 3, "target list"); + assert_eq!(value >> 16 & 0xFF, 0x0C, "Aff1"); + assert_eq!(value >> 24 & 0xF, 2, "INTID"); + assert_eq!(value >> 32 & 0xFF, 0x0B, "Aff2"); + assert_eq!(value >> 48 & 0xFF, 0x0A, "Aff3"); + assert_eq!(value & !(0xFF << 48 | 0xF << 44 | 0xFF << 32 | 0xF << 24 | 0xFF << 16 | 0xFFFF), 0, "IRM and RES0"); +} + +#[test] +fn sgi_to_cpu_zero_is_bit_zero_of_range_zero() { + assert_eq!(sgi1r(0, 0), 1); +} + +/// A region of redistributors, each `(affinity, vlpis, last)`, as `GICR_TYPER` reads at each one's offset. +fn region(cpus: &[(u32, bool, bool)]) -> (u64, impl Fn(u64) -> u64 + '_) { + let stride = |vlpis: bool| if vlpis { 4 * FRAME } else { 2 * FRAME }; + let length = cpus.iter().map(|&(_, vlpis, _)| stride(vlpis)).sum(); + let typer = move |at: u64| { + let mut offset = 0; + for &(affinity, vlpis, last) in cpus { + if offset == at { + return u64::from(affinity) << 32 | u64::from(vlpis) << 1 | u64::from(last) << 4; + } + offset += stride(vlpis); + } + panic!("a GICR_TYPER read at {at:#x}, which is no redistributor's first frame"); + }; + (length, typer) +} + +#[test] +fn walk_steps_over_virtual_lpi_frames() { + let (length, typer) = region(&[(0, true, false), (1, true, false), (2, true, true)]); + assert_eq!(find_redistributor(length, 2, typer), Some(8 * FRAME)); +} + +#[test] +fn walk_steps_two_frames_without_them() { + let (length, typer) = region(&[(0, false, false), (1, false, true)]); + assert_eq!(find_redistributor(length, 1, typer), Some(2 * FRAME)); +} + +#[test] +fn walk_stops_at_the_last() { + let (_, typer) = region(&[(0, false, false), (1, false, true)]); + assert_eq!(find_redistributor(16 * FRAME, 7, typer), None); +} + +#[test] +fn walk_stops_at_the_region_end() { + let (length, typer) = region(&[(0, false, false), (1, false, false)]); + assert_eq!(find_redistributor(length, 7, typer), None); +} From b4a705147cb7fe7b055ecdbba50cfe96246bd498 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 03:32:30 +0200 Subject: [PATCH 04/11] AArch64 stage 4: guest tests for the FP switch, the first entry, the unmap and the refusals The review of #589 at 9b7ed0ce, B1, B3, B4, B5, N3, N7, N11 and the REMOVEs: - `tests/virtpreemptcase` becomes `tests/virtjobcase`, whose test-runner runs five toybox applets on one CPU under the EL2 profile, each judged by its own `virt_` test through `virt_job`: `preempt` (`virt_timer_preempts`), `fp_isolation` (B3), `first_entry` (B4), `unmap_touch` (B5) and `debug_refused` (N3). The boots carry `TEST_KERNEL`, for `SYS_DEBUG`, and a job bound of four minutes. - `fp_isolation`: pins v0-v31, FPCR and FPSR and holds them until a sibling that loads another state has been seen to run three times, which on one CPU means three switches away and back. - `first_entry`: a raw thread whose first instruction stores x1-x30. - `unmap_touch`: four children each write a page, unmap it and read it; each must die on the read. - `debug_refused`: `SYS_DEBUG`'s TLB acknowledgement delay answers `NotSupported`. - `virt_timer_floor` (B1) boots the `timer-floor` actuator; `virt_selftest` judges it and `virt_irq_storm` alike. - The assembly lives in `userland/toybox/src/arch/`, placed in the source gate as metalprobe's is; x86-64's module names where its own probes are. - N7: two host tests for `direct_map_end`'s start-off-page and end-below-start refusals. - N11: the crash dump says whether TP+0 holds variant II's self-pointer again, where the TLS variant is II. - Filed: `issues/isolation/aarch64-el1-runs-without-pan.md` (N1), `issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md` (N4), `issues/kernel/portable-kernel-code-names-the-tsc.md` (N2's remainder). - N6: the track's stage 4 records the instruction-cache, break-before-make and ASID-reclaim claims as owed, with the first HVF run as their exit. - REMOVE: the track's "no QEMU test can do" sentence and its provenance, the isolation issue's "shape to copy", `aarch64/mod.rs`'s status paragraph, `CPACR`'s "across every entry from EL0", and `virt_timer_preempts`' restated comment. Co-Authored-By: Claude Opus 5.5 --- ...s-ring-3-holding-kernel-register-values.md | 3 - .../isolation/aarch64-el1-runs-without-pan.md | 23 ++ ...rash-report-reads-through-any-user-leaf.md | 23 ++ .../portable-kernel-code-names-the-tsc.md | 19 ++ issues/kernel/toyos-runs-on-arm64.md | 18 +- kernel/src/arch/aarch64/control_regs.rs | 4 +- kernel/src/arch/aarch64/mod.rs | 8 - kernel/src/process.rs | 5 +- src/build.rs | 2 +- src/sourcegate.rs | 2 + tests/toyos.rs | 125 +++++---- tests/virtjobcase/system.toml | 23 ++ tests/virtpreemptcase/system.toml | 17 -- toyos-bootmap/tests/aarch64_direct_map.rs | 12 + userland/toybox/src/arch/aarch64.rs | 239 ++++++++++++++++++ userland/toybox/src/arch/mod.rs | 12 + userland/toybox/src/arch/x86_64.rs | 16 ++ userland/toybox/src/debug_refused.rs | 18 ++ userland/toybox/src/main.rs | 7 +- userland/toybox/src/unmap_touch.rs | 55 ++++ 20 files changed, 542 insertions(+), 89 deletions(-) create mode 100644 issues/isolation/aarch64-el1-runs-without-pan.md create mode 100644 issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md create mode 100644 issues/kernel/portable-kernel-code-names-the-tsc.md create mode 100644 tests/virtjobcase/system.toml delete mode 100644 tests/virtpreemptcase/system.toml create mode 100644 userland/toybox/src/arch/aarch64.rs create mode 100644 userland/toybox/src/arch/mod.rs create mode 100644 userland/toybox/src/arch/x86_64.rs create mode 100644 userland/toybox/src/debug_refused.rs create mode 100644 userland/toybox/src/unmap_touch.rs diff --git a/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md b/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md index 1df4fd59d2..03fe5f90cc 100644 --- a/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md +++ b/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md @@ -14,9 +14,6 @@ instruction holding whatever the kernel left there, kernel stack and heap addresses among them. A thread reads the kernel's layout off its own registers. -The AArch64 trampoline zeroes every register but the argument before its -`ERET`, and is the shape to copy. - **Exit condition**: a fresh thread's first instruction sees zero in every general register but its stack pointer and its argument, and a guest test that reads them at `_start` says so. diff --git a/issues/isolation/aarch64-el1-runs-without-pan.md b/issues/isolation/aarch64-el1-runs-without-pan.md new file mode 100644 index 0000000000..8724eb0954 --- /dev/null +++ b/issues/isolation/aarch64-el1-runs-without-pan.md @@ -0,0 +1,23 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# AArch64's EL1 runs without PAN + +`kernel/src/arch/aarch64/control_regs.rs`'s `SCTLR` leaves `SPAN` set, so an +exception taken to EL1 leaves `PSTATE.PAN` as it was, and nothing writes +`PSTATE.PAN`: EL1 can load and store through any EL0 mapping. x86-64 runs +with SMAP (`kernel/src/arch/x86_64/control_regs.rs`), so a kernel bug that +dereferences a user pointer directly faults there and not here. + +The kernel reaches user memory through the direct map (`kernel/src/user_ptr.rs`) +and never through a user address, so PAN costs no +unprivileged-access instruction; what it needs is FEAT_PAN in the +declaration's check and `SPAN` clear. + +**Exit condition**: `SCTLR_EL1.SPAN` is clear and `PSTATE.PAN` set at EL1 on +every CPU, `control_regs::check` refuses a CPU without FEAT_PAN, and a guest +test whose kernel reads a user address directly under `test-actuators` +takes a permission fault. diff --git a/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md b/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md new file mode 100644 index 0000000000..40e669d059 --- /dev/null +++ b/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md @@ -0,0 +1,23 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# An AArch64 crash report reads through any user leaf + +`kernel/src/arch/aarch64/paging.rs`'s `read_user_word`, which the report +of an EL0 fault calls with the faulting thread's own `x29` to walk its +frames (`trap.rs`'s `user_backtrace`), reads the frame any valid user leaf +names through the direct map. The direct map holds memory and nothing +else, so a leaf naming a device's registers — a claimed function's BAR, +once the port's stage 6 maps one into a process — names an address the +direct map does not hold, and the report's read of it is an EL1 data abort: +a user fault with `x29` pointed into its own BAR ends the machine. + +Nothing maps a BAR into an AArch64 process yet, so this is latent until +stage 6 of `issues/kernel/toyos-runs-on-arm64.md`. + +**Exit condition**: `read_user_word` reads only a leaf of the memory type +the direct map holds, and a guest test whose process faults with `x29` +inside a mapped BAR sees its process end and the kernel live. diff --git a/issues/kernel/portable-kernel-code-names-the-tsc.md b/issues/kernel/portable-kernel-code-names-the-tsc.md new file mode 100644 index 0000000000..d2c1ea18d5 --- /dev/null +++ b/issues/kernel/portable-kernel-code-names-the-tsc.md @@ -0,0 +1,19 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# Portable kernel code names the TSC + +On AArch64 the CPU's counter is the generic timer's `CNTVCT_EL0`, and the +portable kernel still calls it the TSC: `kernel/src/clock.rs`'s +`tsc_deadline`, `TSC_BOOT` and `TSC_PERIOD_FS`, `kernel/src/deadline.rs`'s +`AT_TSC`, and the xHCI driver's, `hardlockup`'s and `panic_reboot`'s waits +and comments read it by that name. `clock::counter_ticks`, which the AArch64 +timer reads, is renamed; the rest reads as x86-64's on both machines. +`issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md` is the ABI's +half of the same name. + +**Exit condition**: no item or comment outside `kernel/src/arch/x86_64/` +names the TSC for the counter `crate::arch::cpu::counter` reads. diff --git a/issues/kernel/toyos-runs-on-arm64.md b/issues/kernel/toyos-runs-on-arm64.md index 11b60d920a..4eedfda4ea 100644 --- a/issues/kernel/toyos-runs-on-arm64.md +++ b/issues/kernel/toyos-runs-on-arm64.md @@ -301,20 +301,20 @@ Each stage names its exit; "measured" means a number from a run. the kernel's own tables (`TTBR1_EL1` holding memory and nothing else, each user space on `TTBR0_EL1` under a 16-bit ASID from `toyos-pcid`), the GICv3's SGIs and the virtual timer's PPI, the EL0 entry, and the context - switch carrying FP/SIMD; `virt_user_mode`, `virt_timer_preempts` and - `virt_irq_storm` judge it under the EL2 profile, emulated, because HVF + switch carrying FP/SIMD; the `virt_` tests other than `virt_early_panic`, + `virt_early_fault` and `virt_el2_drop` judge it under the EL2 profile, emulated, because HVF exposes no RNDR and the kernel's hash seed refuses there until stage 6's virtio-rng. Each judges an event, never a rate: no QEMU test measures time. Owed before the exit holds: the interrupts-off window against x86's, a measurement only metal can make, with no instrument on either arch yet; `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md`, whose - `KernelArgs` rename waits on the loader's change to that struct; and the - three deletions shown red, which **no QEMU test can do**: QEMU 11.1.1 - resets `CNTHCTL_EL2` to 3 and `CNTVOFF_EL2` to 0, the declared values, and - `CPTR_EL2` to 0, whose `TFP` is as clear as the declaration's - (`target/arm/helper.c`); each deletion, and the fourth, of `ICC_SRE_EL2` - (constant `0xf` in QEMU, `hw/intc/arm_gicv3_cpuif.c`), stayed green in - `virt_user_mode` at `57382f78`. They are shown red on a machine whose + `KernelArgs` rename waits on the loader's change to that struct; the + instruction-cache maintenance before a mapping is executable + (`cache::make_executable`), the break-before-make ordering of a live + entry's replacement, and the TLB flush before a reclaimed ASID is issued + again, which QEMU's TCG, the only oracle this stage has, cannot fail on: + the first HVF run, once stage 6 gives HVF its RNDR, is their exit; and the + three deletions shown red. They are shown red on a machine whose firmware leaves the registers otherwise, or by a loader that writes the opposite values before the handoff. The ITS moves to stage 6: a claimed function is its only consumer the small-kernel track leaves, and it needs that diff --git a/kernel/src/arch/aarch64/control_regs.rs b/kernel/src/arch/aarch64/control_regs.rs index 51b133d87e..6251ba1bcd 100644 --- a/kernel/src/arch/aarch64/control_regs.rs +++ b/kernel/src/arch/aarch64/control_regs.rs @@ -47,8 +47,8 @@ pub const TCR_IPS_SHIFT: u64 = 32; /// `CPACR_EL1`: `FPEN` = 0b11, so FP and SIMD trap at neither EL1 nor EL0. The /// kernel is built soft-float and touches them only to save and restore a -/// thread's registers across every entry from EL0 (`super::trap`); `ZEN` and -/// `SMEN` stay clear, so SVE and SME trap everywhere. +/// thread's registers; `ZEN` and `SMEN` stay clear, so SVE and SME trap +/// everywhere. pub const CPACR: u64 = 0b11 << 20; /// `CNTKCTL_EL1`: `EL0VCTEN` alone, so EL0 reads the virtual count and its diff --git a/kernel/src/arch/aarch64/mod.rs b/kernel/src/arch/aarch64/mod.rs index 95404aa52f..b1aee4ca6e 100644 --- a/kernel/src/arch/aarch64/mod.rs +++ b/kernel/src/arch/aarch64/mod.rs @@ -2,14 +2,6 @@ //! AArch64: the Arm A-profile at EL1, found and described through ACPI. //! //! Every `unsafe` block here carries a one-line `SAFETY:` comment, enforced by the lint above. -//! -//! **What exists and what is owed.** One CPU runs the kernel and user mode: -//! its own page tables, the GICv3 and the generic timer, preemption, and the -//! entry from EL0. Other CPUs, the IOMMU, an MSI and the platform's own -//! devices are owed by a stage of the port (`issues/kernel/toyos-runs-on-arm64.md`) -//! or by none yet, and each item that stands for one here is an [`owed!`] -//! that panics naming it, or a refusal a caller reports by name. None of them -//! returns a guess. /// Stands for work the port owes: panics naming what and which stage of the /// track owns it (`issues/kernel/toyos-runs-on-arm64.md`), or that none does yet. diff --git a/kernel/src/process.rs b/kernel/src/process.rs index 36a84b042e..6dc27af382 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -1477,7 +1477,10 @@ pub fn dump_crash_diagnostics(fault_addr: u64, rip: u64) { let tp = crate::arch::cpu::thread_pointer(); if tp != 0 { log!(" Thread pointer: {:#x}", tp); - if read_user(tp).is_some() { + if let Some(first) = read_user(tp) { + if matches!(crate::loader::TLS_VARIANT, toyos_elf::tls::Variant::II) { + log!(" [TP] = {:#x} (expected {:#x}, variant II's self-pointer)", first, tp); + } for i in 0..8u64 { let addr = tp + i * 8; let Some(val) = read_user(addr) else { break }; diff --git a/src/build.rs b/src/build.rs index 04129e9850..8ac742457c 100644 --- a/src/build.rs +++ b/src/build.rs @@ -3140,7 +3140,7 @@ mod tests { "tests/testcases/system.toml", "tests/toolkitcase/system.toml", "tests/updatecase/system.toml", - "tests/virtpreemptcase/system.toml", + "tests/virtjobcase/system.toml", ]; fn load(cfg: &str) -> SystemConfig { diff --git a/src/sourcegate.rs b/src/sourcegate.rs index 3b4fc9bd7f..afed1a7ec3 100644 --- a/src/sourcegate.rs +++ b/src/sourcegate.rs @@ -1466,6 +1466,7 @@ const ARCH_RULES: &[PlaceRule] = &[ ("toyos-abi/src/arch/", "toyos-abi's per-architecture reads: the counter and the thread's id"), ("userland/libc/src/arch/", "libc's architecture modules"), ("userland/metalprobe/src/arch/", "metalprobe's architecture modules"), + ("userland/toybox/src/arch/", "toybox's architecture modules"), ("userland/toyos-window/src/arch/", "toyos-window's architecture modules"), ("tests/toyos-rust-tests/src/bin/abuse_kernel_addr.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs", USERLAND_ASM), @@ -1509,6 +1510,7 @@ const ARCH_RULES: &[PlaceRule] = &[ ("userland/libc/src/arch/", "libc's architecture modules and their selector"), ("userland/toyos-window/src/arch/", "toyos-window's architecture modules and their selector"), ("userland/metalprobe/src/arch/", "metalprobe's architecture modules and their selector"), + ("userland/toybox/src/arch/", "toybox's architecture modules and their selector"), ], }, PlaceRule { diff --git a/tests/toyos.rs b/tests/toyos.rs index 051196b8a4..c09bc9301d 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -530,6 +530,11 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ ("virt_user_mode", Sched::Parallel, Tier::Local), ("virt_timer_preempts", Sched::Parallel, Tier::Local), ("virt_irq_storm", Sched::Parallel, Tier::Local), + ("virt_timer_floor", Sched::Parallel, Tier::Local), + ("virt_fp_isolation", Sched::Parallel, Tier::Local), + ("virt_first_entry", Sched::Parallel, Tier::Local), + ("virt_unmap_touch", Sched::Parallel, Tier::Local), + ("virt_debug_refused", Sched::Parallel, Tier::Local), ]; /// What `screen_console_shell` types, and what it then looks for on its own. @@ -3555,6 +3560,67 @@ fn check_no_stale_cells(dump: &screen::Ppm, console: &str) -> Result<(), String> /// Run one screen test. `Err` carries the decoded screen, because a failure /// here is almost always "the text is not what I expected" and the decoded /// grid is the only readable form of that. +/// Boot `tests/virtjobcase` under the EL2 profile and judge its job `job`: +/// it ends with exit 0, having said `said`. The kernel carries `SYS_DEBUG` +/// for `debug_refused`, and every job runs in every boot of the case. +fn virt_job(job: &str, said: &str) -> Result<(), String> { + let config = compile::repo_root().join("tests/virtjobcase/system.toml"); + let case = config.parent().expect("system.toml has a directory"); + let mut qemu = QemuInstance::boot_with_options( + case, + &[], + &[], + BootOptions { + profile: qemu::Profile::VirtEl2, + kernel_features: toyos_build::build::TEST_KERNEL, + ready_marker: "control registers: SCTLR_EL1=", + ..Default::default() + }, + ); + let end = format!("===TEST_END {job} "); + let rest = qemu.drain_until(Duration::from_secs(300), |l| l.contains(&end)); + let serial = format!("{}\n{rest}", qemu.boot_log()); + let Some(ended) = serial.lines().find(|l| l.contains(&end)) else { + return Err(format!("the job {job} never ended\nserial:\n{serial}")); + }; + let Some(line) = serial.lines().find(|l| l.contains(said)) else { + return Err(format!("{said:?} not on the PL011 ({ended})\nserial:\n{serial}")); + }; + eprintln!(" [virt] {line}"); + if !ended.contains(&format!("===TEST_END {job} exit=0===")) { + return Err(format!("{ended}\nserial:\n{serial}")); + } + Ok(()) +} + +/// Boot `test_config` under the EL2 profile with the kernel selftest `armed` +/// names, and judge its one line: `: PASS`. +fn virt_selftest(test_config: &Path, armed: &'static [&'static str; 1]) -> Result<(), String> { + let [param] = armed; + let mut qemu = QemuInstance::boot_with_options( + test_config, + &[], + &[], + BootOptions { + profile: qemu::Profile::VirtEl2, + kernel_params: armed, + ready_marker: "control registers: SCTLR_EL1=", + ..Default::default() + }, + ); + let said = format!("{param}: "); + let rest = qemu.drain_until(Duration::from_secs(180), |l| l.contains(&said)); + let serial = format!("{}\n{rest}", qemu.boot_log()); + let Some(verdict) = serial.lines().find(|l| l.contains(&said)) else { + return Err(format!("{param} never reported\nserial:\n{serial}")); + }; + eprintln!(" [virt] {verdict}"); + if !verdict.contains(&format!("{param}: PASS")) { + return Err(format!("{verdict}\nserial:\n{serial}")); + } + Ok(()) +} + fn run_screen_test( name: &str, test_config: &Path, @@ -5046,60 +5112,25 @@ fn run_screen_test( Ok(()) } "virt_timer_preempts" => { - // `preempt` on this machine's one CPU: one thread counts in a loop - // that enters the kernel only when an interrupt takes it there, and - // the other yields until it has seen the count move twice. It runs - // between those reads only when a tick took the CPU from the - // counting thread, so a timer that never preempts leaves the line - // unsaid, and the ceiling here is the only clock. - let config = compile::repo_root().join("tests/virtpreemptcase/system.toml"); - let case = config.parent().expect("system.toml has a directory"); - let mut qemu = QemuInstance::boot_with_options( - case, - &[], - &[], - BootOptions { - profile: qemu::Profile::VirtEl2, - ready_marker: "control registers: SCTLR_EL1=", - ..Default::default() - }, - ); // Spelled in `userland/toybox/src/preempt.rs`. - const PREEMPTED: &str = "preempt: the counting thread was preempted twice"; - let rest = qemu.drain_until(Duration::from_secs(300), |l| l.contains(PREEMPTED)); - let serial = format!("{}\n{rest}", qemu.boot_log()); - match serial.lines().find(|l| l.contains(PREEMPTED)) { - Some(line) => eprintln!(" [virt] {line}"), - None => return Err(format!("{PREEMPTED:?} not on the PL011\nserial:\n{serial}")), - } - Ok(()) + virt_job("preempt", "preempt: the counting thread was preempted twice") } + "virt_fp_isolation" => virt_job("fp_isolation", "fp_isolation: v0-v31, FPCR and FPSR survived"), + "virt_first_entry" => virt_job("first_entry", "first_entry: x1-x30 were zero"), + "virt_unmap_touch" => virt_job("unmap_touch", "unmap_touch: 4 reads of a page just unmapped"), + "virt_debug_refused" => virt_job("debug_refused", "debug_refused: SYS_DEBUG's TLB acknowledgement delay was refused"), "virt_irq_storm" => { // The CPU floods itself with SGIs until the timer has fired a // thousand times through the flood, then waits for every SGI it // sent. A tick lost or never re-armed, or an SGI lost, leaves the // storm running and the verdict unsaid. - let mut qemu = QemuInstance::boot_with_options( - test_config, - &[], - &[], - BootOptions { - profile: qemu::Profile::VirtEl2, - kernel_params: &["irq-storm"], - ready_marker: "control registers: SCTLR_EL1=", - ..Default::default() - }, - ); - let rest = qemu.drain_until(Duration::from_secs(180), |l| l.contains("irq-storm: ")); - let serial = format!("{}\n{rest}", qemu.boot_log()); - let Some(verdict) = serial.lines().find(|l| l.contains("irq-storm: ")) else { - return Err(format!("the storm never reported\nserial:\n{serial}")); - }; - eprintln!(" [virt] {verdict}"); - if !verdict.contains("irq-storm: PASS") { - return Err(format!("{verdict}\nserial:\n{serial}")); - } - Ok(()) + virt_selftest(test_config, &["irq-storm"]) + } + "virt_timer_floor" => { + // The timer made due and then asked to fire within a quantum: a + // re-arm shorter than the floor re-fires on its own return, and + // the CPU taking it at EL1 never gets back to say anything. + virt_selftest(test_config, &["timer-floor"]) } "screen_late_panic" => { // The ordinary fatal panic, which no userland process can produce: diff --git a/tests/virtjobcase/system.toml b/tests/virtjobcase/system.toml new file mode 100644 index 0000000000..f08e05c649 --- /dev/null +++ b/tests/virtjobcase/system.toml @@ -0,0 +1,23 @@ +# One CPU and the AArch64 guest's jobs, each a toybox applet whose line comes +# only if the kernel kept what that applet asks about. +[boot] +start = ["logd", "test-runner"] + +[programs.logd] +service = true +syscap = ["logread"] + +[programs.test-runner] +receives = ["power"] +# Four minutes from boot rather than one: every job here runs on an emulated +# CPU, and a job past the bound reboots the guest with the job named. +args = ["--bound-ms=240000", "preempt", "fp_isolation", "first_entry", "unmap_touch", "debug_refused"] + +[programs.toybox] + +[symlinks] +"bin/preempt" = "/system/bin/toybox" +"bin/fp_isolation" = "/system/bin/toybox" +"bin/first_entry" = "/system/bin/toybox" +"bin/unmap_touch" = "/system/bin/toybox" +"bin/debug_refused" = "/system/bin/toybox" diff --git a/tests/virtpreemptcase/system.toml b/tests/virtpreemptcase/system.toml deleted file mode 100644 index 5f9a06bd81..0000000000 --- a/tests/virtpreemptcase/system.toml +++ /dev/null @@ -1,17 +0,0 @@ -# One CPU and one job, `preempt`, whose line comes only after the timer has -# taken the CPU from a thread that never enters the kernel itself. -[boot] -start = ["logd", "test-runner"] - -[programs.logd] -service = true -syscap = ["logread"] - -[programs.test-runner] -receives = ["power"] -args = ["preempt"] - -[programs.toybox] - -[symlinks] -"bin/preempt" = "/system/bin/toybox" diff --git a/toyos-bootmap/tests/aarch64_direct_map.rs b/toyos-bootmap/tests/aarch64_direct_map.rs index 73c8b15187..3adcb659bf 100644 --- a/toyos-bootmap/tests/aarch64_direct_map.rs +++ b/toyos-bootmap/tests/aarch64_direct_map.rs @@ -82,3 +82,15 @@ fn past_the_window_is_refused() { fn a_map_with_no_memory_ends_at_zero() { assert_eq!(end(&[e(MMIO, 0x0900_0000, 0x0900_1000)]), Ok(0)); } + +#[test] +fn a_start_off_a_page_is_refused() { + let map = [e(CONVENTIONAL, GIB + 0x800, GIB + PAGE_2M)]; + assert_eq!(end(&map), Err(Refusal::OffPage(GIB + 0x800))); +} + +#[test] +fn an_end_below_its_start_is_refused() { + let map = [e(CONVENTIONAL, GIB + PAGE_2M, GIB)]; + assert_eq!(end(&map), Err(Refusal::Extent { base: GIB + PAGE_2M, len: GIB.wrapping_sub(GIB + PAGE_2M) })); +} diff --git a/userland/toybox/src/arch/aarch64.rs b/userland/toybox/src/arch/aarch64.rs new file mode 100644 index 0000000000..144fd23e91 --- /dev/null +++ b/userland/toybox/src/arch/aarch64.rs @@ -0,0 +1,239 @@ +//! What an AArch64 thread finds in its registers after the kernel has had the +//! CPU: its own FP/SIMD state across the switches that took the CPU from it, +//! and nothing of the kernel's at its first instruction. + +/// `fp_isolation`: this thread pins a distinctive FP/SIMD state — v0–v31, +/// `FPCR` and `FPSR` — and holds it while a sibling that loads another state +/// takes the CPU from it; the state it reads back must be the one it pinned. +/// The kernel saves and restores that state only where a thread stops +/// running, so on one CPU every switch is a chance to hand this thread the +/// sibling's. +pub mod fp_isolation { + use std::sync::atomic::{AtomicBool, AtomicU64, Ordering::Relaxed}; + use std::sync::Arc; + + /// The saved state: v0–v31, then `FPCR` and `FPSR`. + #[repr(C, align(16))] + #[derive(Clone, Copy, PartialEq, Eq, Debug)] + struct FpState { + v: [u128; 32], + fpcr: u64, + fpsr: u64, + } + + const fn state(tag: u128, fpcr: u64, fpsr: u64) -> FpState { + let mut v = [0; 32]; + let mut i = 0; + while i < 32 { + v[i] = tag | (i as u128 + 1); + i += 1; + } + FpState { v, fpcr, fpsr } + } + + /// Default-NaN, flush-to-zero, round toward zero; every cumulative flag + /// and `QC` raised. + static PINNED: FpState = state(0xF9A3 << 112, 1 << 25 | 1 << 24 | 0b11 << 22, 1 << 27 | 0x1F); + /// Round toward plus infinity, no flag. + static NOISE: FpState = state(0x5A5A << 112, 0b01 << 22, 0); + const EMPTY: FpState = state(0, 0, 0); + + /// Times the sibling must be seen to have run while this thread held its state. + const SWITCHES: u64 = 3; + + /// Loads the state at `[x9]` into the FP/SIMD registers. + macro_rules! load_state { + () => { + concat!( + "ldp q0, q1, [x9, #0]\n", "ldp q2, q3, [x9, #32]\n", "ldp q4, q5, [x9, #64]\n", + "ldp q6, q7, [x9, #96]\n", "ldp q8, q9, [x9, #128]\n", "ldp q10, q11, [x9, #160]\n", + "ldp q12, q13, [x9, #192]\n", "ldp q14, q15, [x9, #224]\n", "ldp q16, q17, [x9, #256]\n", + "ldp q18, q19, [x9, #288]\n", "ldp q20, q21, [x9, #320]\n", "ldp q22, q23, [x9, #352]\n", + "ldp q24, q25, [x9, #384]\n", "ldp q26, q27, [x9, #416]\n", "ldp q28, q29, [x9, #448]\n", + "ldp q30, q31, [x9, #480]\n", + "ldr x12, [x9, #512]\n", "msr fpcr, x12\n", + "ldr x12, [x9, #520]\n", "msr fpsr, x12\n", + ) + }; + } + + /// Stores the FP/SIMD registers at `[$reg]`. + macro_rules! store_state { + ($reg:literal) => { + concat!( + "stp q0, q1, [", $reg, ", #0]\n", "stp q2, q3, [", $reg, ", #32]\n", + "stp q4, q5, [", $reg, ", #64]\n", "stp q6, q7, [", $reg, ", #96]\n", + "stp q8, q9, [", $reg, ", #128]\n", "stp q10, q11, [", $reg, ", #160]\n", + "stp q12, q13, [", $reg, ", #192]\n", "stp q14, q15, [", $reg, ", #224]\n", + "stp q16, q17, [", $reg, ", #256]\n", "stp q18, q19, [", $reg, ", #288]\n", + "stp q20, q21, [", $reg, ", #320]\n", "stp q22, q23, [", $reg, ", #352]\n", + "stp q24, q25, [", $reg, ", #384]\n", "stp q26, q27, [", $reg, ", #416]\n", + "stp q28, q29, [", $reg, ", #448]\n", "stp q30, q31, [", $reg, ", #480]\n", + "mrs x12, fpcr\n", "str x12, [", $reg, ", #512]\n", + "mrs x12, fpsr\n", "str x12, [", $reg, ", #520]\n", + ) + }; + } + + pub fn main(_args: Vec) { + let ran = Arc::new(AtomicU64::new(0)); + let stop = Arc::new(AtomicBool::new(false)); + let sibling = { + let (ran, stop) = (Arc::clone(&ran), Arc::clone(&stop)); + std::thread::spawn(move || noise_until(&ran, &stop)) + }; + let (mut before, mut after) = (EMPTY, EMPTY); + pin_and_watch(&ran, &mut before, &mut after); + stop.store(true, Relaxed); + sibling.join().expect("the noise thread died"); + + assert_eq!(before.v, PINNED.v, "the pinned registers did not read back as pinned"); + assert_ne!(before.fpcr, NOISE.fpcr, "FPCR read back as the noise's, so the arm cannot tell them apart"); + let lost: Vec = (0..32).filter(|&i| after.v[i] != before.v[i]).collect(); + assert!( + lost.is_empty() && after.fpcr == before.fpcr && after.fpsr == before.fpsr, + "the FP/SIMD state did not survive {SWITCHES} switches to a thread that loads another: \ + v{lost:?} changed, FPCR {:#x} became {:#x}, FPSR {:#x} became {:#x}", + before.fpcr, + after.fpcr, + before.fpsr, + after.fpsr, + ); + println!("fp_isolation: v0-v31, FPCR and FPSR survived {SWITCHES} switches to a thread that loads another state"); + } + + /// Pin [`PINNED`], read it back into `before`, hold it until `ran` has + /// been seen to move [`SWITCHES`] times, and read it into `after`. On one + /// CPU `ran` moves only while this thread is off it. Nothing between the + /// pin and the second read touches an FP/SIMD register. + fn pin_and_watch(ran: &AtomicU64, before: &mut FpState, after: &mut FpState) { + // SAFETY: reads `PINNED` and `ran`, writes `before` and `after`, each + // named by its register and live across the block; every FP/SIMD + // register it writes is declared clobbered, and `FPCR`/`FPSR` are put + // back as found. + unsafe { + core::arch::asm!( + "mrs x16, fpcr", + "mrs x17, fpsr", + load_state!(), + store_state!("x10"), + "ldr x14, [x13]", + "mov x15, #{switches}", + "2:", + "ldr x12, [x13]", + "cmp x12, x14", + "b.eq 2b", + "mov x14, x12", + "subs x15, x15, #1", + "b.ne 2b", + store_state!("x11"), + "msr fpcr, x16", + "msr fpsr, x17", + switches = const SWITCHES, + in("x9") &raw const PINNED, + in("x10") core::ptr::from_mut(before), + in("x11") core::ptr::from_mut(after), + in("x13") core::ptr::from_ref(ran), + out("x12") _, out("x14") _, out("x15") _, out("x16") _, out("x17") _, + out("v0") _, out("v1") _, out("v2") _, out("v3") _, out("v4") _, out("v5") _, + out("v6") _, out("v7") _, out("v8") _, out("v9") _, out("v10") _, out("v11") _, + out("v12") _, out("v13") _, out("v14") _, out("v15") _, out("v16") _, out("v17") _, + out("v18") _, out("v19") _, out("v20") _, out("v21") _, out("v22") _, out("v23") _, + out("v24") _, out("v25") _, out("v26") _, out("v27") _, out("v28") _, out("v29") _, + out("v30") _, out("v31") _, + options(nostack), + ); + } + } + + /// Load [`NOISE`] and count one round in `ran`, over and over, until `stop`. + fn noise_until(ran: &AtomicU64, stop: &AtomicBool) { + // SAFETY: reads `NOISE` and `stop`, writes `ran`; every FP/SIMD + // register it writes is declared clobbered, and `FPCR`/`FPSR` are put + // back as found. + unsafe { + core::arch::asm!( + "mrs x16, fpcr", + "mrs x17, fpsr", + "2:", + load_state!(), + "ldr x12, [x13]", + "add x12, x12, #1", + "str x12, [x13]", + "ldrb w12, [x14]", + "cbz w12, 2b", + "msr fpcr, x16", + "msr fpsr, x17", + in("x9") &raw const NOISE, + in("x13") core::ptr::from_ref(ran), + in("x14") core::ptr::from_ref(stop), + out("x12") _, out("x16") _, out("x17") _, + out("v0") _, out("v1") _, out("v2") _, out("v3") _, out("v4") _, out("v5") _, + out("v6") _, out("v7") _, out("v8") _, out("v9") _, out("v10") _, out("v11") _, + out("v12") _, out("v13") _, out("v14") _, out("v15") _, out("v16") _, out("v17") _, + out("v18") _, out("v19") _, out("v20") _, out("v21") _, out("v22") _, out("v23") _, + out("v24") _, out("v25") _, out("v26") _, out("v27") _, out("v28") _, out("v29") _, + out("v30") _, out("v31") _, + options(nostack), + ); + } + } +} + +/// `first_entry`: a raw thread whose first instruction stores x1–x30 and +/// exits. The kernel hands a new thread its argument in x0 and its stack in +/// `SP_EL0`, and every other general register must reach that instruction +/// zero: whatever else is there is the kernel's. +pub mod first_entry { + use toyos_abi::syscall::{thread_join, thread_spawn, SYS_THREAD_EXIT}; + + /// x1 to x30, as the probe found them. + #[repr(C, align(16))] + struct Registers([u64; 30]); + + pub fn main(_args: Vec) { + let mut found = Registers([u64::MAX; 30]); + let mut stack = vec![0u128; 1024]; + let base = stack.as_mut_ptr() as u64; + let top = base + (stack.len() * 16) as u64; + // SAFETY: `probe` touches only the `Registers` its argument names, + // which outlives the join below, and never its stack. + let tid = unsafe { thread_spawn(probe as *const () as u64, top, (&raw mut found) as u64, base) }; + assert!(tid < 1_000_000, "thread_spawn refused: {tid:#x}"); + assert_eq!(thread_join(tid), 0, "thread_join failed"); + drop(stack); + let held: Vec = (0..30) + .filter(|&i| found.0[i] != 0) + .map(|i| format!("x{}={:#x}", i + 1, found.0[i])) + .collect(); + assert!(held.is_empty(), "a new thread's first instruction found {}", held.join(" ")); + println!("first_entry: x1-x30 were zero at a new thread's first instruction"); + } + + /// Store x1–x30 at `[x0]` before any of them is written, and exit the thread. + #[unsafe(naked)] + extern "C" fn probe() { + core::arch::naked_asm!( + "stp x1, x2, [x0, #0]", + "stp x3, x4, [x0, #16]", + "stp x5, x6, [x0, #32]", + "stp x7, x8, [x0, #48]", + "stp x9, x10, [x0, #64]", + "stp x11, x12, [x0, #80]", + "stp x13, x14, [x0, #96]", + "stp x15, x16, [x0, #112]", + "stp x17, x18, [x0, #128]", + "stp x19, x20, [x0, #144]", + "stp x21, x22, [x0, #160]", + "stp x23, x24, [x0, #176]", + "stp x25, x26, [x0, #192]", + "stp x27, x28, [x0, #208]", + "stp x29, x30, [x0, #224]", + "mov x0, #{exit}", + "mov x1, xzr", + "svc #0", + "brk #0", + exit = const SYS_THREAD_EXIT, + ); + } +} diff --git a/userland/toybox/src/arch/mod.rs b/userland/toybox/src/arch/mod.rs new file mode 100644 index 0000000000..157ce79a88 --- /dev/null +++ b/userland/toybox/src/arch/mod.rs @@ -0,0 +1,12 @@ +//! The applets that ask a CPU what the kernel left in its registers, one +//! module per architecture. + +#[cfg(target_arch = "aarch64")] +mod aarch64; +#[cfg(target_arch = "aarch64")] +pub use aarch64::*; + +#[cfg(target_arch = "x86_64")] +mod x86_64; +#[cfg(target_arch = "x86_64")] +pub use x86_64::*; diff --git a/userland/toybox/src/arch/x86_64.rs b/userland/toybox/src/arch/x86_64.rs new file mode 100644 index 0000000000..9e54bfec8c --- /dev/null +++ b/userland/toybox/src/arch/x86_64.rs @@ -0,0 +1,16 @@ +//! x86-64 has these probes elsewhere, or owes them. + +pub mod fp_isolation { + pub fn main(_args: Vec) { + panic!("fp_isolation probes AArch64's FP/SIMD switch; x86-64's is test_rs_fpu_isolation"); + } +} + +pub mod first_entry { + pub fn main(_args: Vec) { + panic!( + "first_entry probes AArch64's first entry to EL0; x86-64's is owed by \ + issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md" + ); + } +} diff --git a/userland/toybox/src/debug_refused.rs b/userland/toybox/src/debug_refused.rs new file mode 100644 index 0000000000..de062fda28 --- /dev/null +++ b/userland/toybox/src/debug_refused.rs @@ -0,0 +1,18 @@ +//! `debug_refused`: a `SYS_DEBUG` action this machine has no counterpart for +//! is refused, and the kernel lives: the TLB acknowledgement delay, which +//! AArch64's broadcast invalidation has no acknowledgement for. Needs a +//! kernel built with `test-actuators`. + +use toyos_abi::syscall::debug_action::{TLB_ACK_DELAY_ARM, TLB_ACK_DELAY_DISARM}; +use toyos_abi::syscall::{debug, debug_with, SyscallError}; + +pub fn main(_args: Vec) { + let refused = SyscallError::NotSupported.to_u64(); + for (name, answer) in [ + ("TLB_ACK_DELAY_ARM", debug_with(TLB_ACK_DELAY_ARM, 1_000_000)), + ("TLB_ACK_DELAY_DISARM", debug(TLB_ACK_DELAY_DISARM)), + ] { + assert_eq!(answer, refused, "SYS_DEBUG {name} answered {answer:#x}, not NotSupported"); + } + println!("debug_refused: SYS_DEBUG's TLB acknowledgement delay was refused, twice"); +} diff --git a/userland/toybox/src/main.rs b/userland/toybox/src/main.rs index 5e703d041b..449ce6e89d 100644 --- a/userland/toybox/src/main.rs +++ b/userland/toybox/src/main.rs @@ -1,5 +1,7 @@ +mod arch; mod cat; mod cp; +mod debug_refused; mod echo; mod free; mod grep; @@ -19,6 +21,7 @@ mod shutdown; mod spin; mod stats; mod tone; +mod unmap_touch; macro_rules! commands { ($($name:ident),*) => { @@ -31,7 +34,9 @@ macro_rules! commands { }; } -commands!(cat, cp, echo, free, grep, hexdump, locale, ls, mkdir, mv, net, preempt, ps, pwd, reboot, rm, screen, shutdown, spin, stats, tone); +use arch::{first_entry, fp_isolation}; + +commands!(cat, cp, debug_refused, echo, first_entry, fp_isolation, free, grep, hexdump, locale, ls, mkdir, mv, net, preempt, ps, pwd, reboot, rm, screen, shutdown, spin, stats, tone, unmap_touch); fn main() { let args: Vec = std::env::args().collect(); diff --git a/userland/toybox/src/unmap_touch.rs b/userland/toybox/src/unmap_touch.rs new file mode 100644 index 0000000000..f0ff799684 --- /dev/null +++ b/userland/toybox/src/unmap_touch.rs @@ -0,0 +1,55 @@ +//! `unmap_touch`: whether a page's unmapping reaches the TLB. A child writes +//! a page, unmaps it and reads it at once; the read must end the child, +//! because a translation left cached past the unmap still reaches the frame. +//! Several children, so the one read that a switch between the unmap and the +//! read would have made fault anyway cannot decide the verdict alone. + +use std::process::Command; + +use toyos_abi::syscall::{mmap, munmap, MmapFlags, MmapProt}; + +const TRIALS: usize = 4; +const PAGE: usize = 2 * 1024 * 1024; +/// The child's last line before the unmap and the read. +const UNMAPPING: &str = "unmap_touch: written; unmapping and reading"; +/// The child's line if the read came back. +const READ_BACK: &str = "unmap_touch: read"; + +pub fn main(args: Vec) { + match args.first().map(String::as_str) { + None => judge(), + Some("touch") => touch(), + Some(other) => panic!("unmap_touch: unknown mode {other:?}"), + } +} + +fn judge() { + for trial in 0..TRIALS { + let out = Command::new("/system/bin/unmap_touch") + .arg("touch") + .output() + .unwrap_or_else(|e| panic!("unmap_touch: the child would not spawn: {e}")); + let said = String::from_utf8_lossy(&out.stdout); + assert!(said.contains(UNMAPPING), "trial {trial}: the child never reached the unmap: {said:?}"); + assert!( + !out.status.success() && !said.contains(READ_BACK), + "trial {trial}: a read of an unmapped page came back, and the child exited {:?}: {said:?}", + out.status.code(), + ); + } + println!("unmap_touch: {TRIALS} reads of a page just unmapped each ended their process"); +} + +fn touch() { + // SAFETY: a fresh anonymous mapping this function alone names. + let page = unsafe { mmap(core::ptr::null_mut(), PAGE, MmapProt::READ | MmapProt::WRITE, MmapFlags::ANONYMOUS | MmapFlags::PRIVATE) }; + assert!(!page.is_null(), "unmap_touch: mmap refused"); + // SAFETY: inside the mapping just made. + unsafe { page.write_volatile(0x5A) }; + println!("{UNMAPPING}"); + // SAFETY: the region `mmap` returned, whole; nothing else holds it. + unsafe { munmap(page, PAGE) }.expect("unmap_touch: munmap refused"); + // SAFETY: none — the read of an unmapped page is what this child exists to make. + let value = unsafe { page.read_volatile() }; + println!("{READ_BACK} {value:#x} after the unmap"); +} From 12861153993e5b3c721f6fac614a1279e13f95d7 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 08:05:34 +0200 Subject: [PATCH 05/11] AArch64 stage 4, round 4: one invalidation per unmap, the floor read off the comparator B5. Every AArch64 munmap was invalidated twice: `AddressSpace::unmap` issues `TLBI VAE1IS`/`ASIDE1IS` + `DSB ISH`, which already reaches every CPU, and then `Unmapped`'s drop called `tlb::shootdown`, a `TLBI VMALLE1IS` that also dropped every other ASID's and every global entry. So deleting the unmap's TLBI (b5) left `virt_unmap_touch` green on any oracle. The page-table edit is the single place now: it is precise, it covers every caller of `unmap` and `replace`, and it is what the module header already promised. `arch::tlb::shootdown` does nothing on AArch64, since every portable caller (`Unmapped`, the pipe window revoke, dlopen) follows an `AddressSpace` edit that has already broadcast; the ASID pool's reclaim calls `tlb::all` directly, and the census counts only that. `unmap_touch`'s child now reads the page, calls munmap and reads again in one assembly block after its last print (`arch::read_unmap_read`), so the only switch that could drop the cached translation is the unmap itself. The parent requires exit status -1, the status `kill_process(-1)` gives a fault; a refused munmap or a panic no longer passes. B7. `timer-floor` judged `percpu::armed_ticks()`, a software copy, and its 100-fire loop could not fail: the b1 log shows all 100 fires done at armed=1. It now reads the counter, calls `arm_within(QUANTUM_NS)` on a due timer, reads `CNTV_CVAL_EL0` back and requires the comparator to be at least `floor_ticks()` past that counter reading. The reading tells floored from unfloored only while the ask took less than the floor, so a wider window is a FAIL as well. The fire loop is deleted. NOTEs. `debug_refused` also asks for SYS_DEBUG's double fault. `virt_job` waits with `await_marker`, whose kernel-death reading (`serial::died`, `kernel_died_here`, `WaitVerdict`) is the x86 harness's: a panicked guest ends the wait once it goes quiet, with the panic report in the verdict, instead of after the whole drain ceiling (633 s for a panic at 2.58 s under n3). The two x86 toybox applets that only panic are filed as issues/design-debt/the-x86-toybox-ships-two-applets-that-only-panic.md. The screen-test doc comment `virt_job` had split from `run_screen_test` goes back to it. REMOVE. The "never gets back to say anything" comment on `virt_timer_floor`, the selftest's matching "no progress to say anything with", and `unmap_touch`'s "Several children" line are deleted. Co-Authored-By: Claude Opus 5.5 --- ...oybox-ships-two-applets-that-only-panic.md | 18 +++++++++ kernel/src/arch/aarch64/irqchip.rs | 28 +++++++------- kernel/src/arch/aarch64/paging.rs | 2 +- kernel/src/arch/aarch64/tlb.rs | 18 ++++----- tests/toyos.rs | 31 +++++++++------- userland/toybox/src/arch/aarch64.rs | 37 +++++++++++++++++-- userland/toybox/src/arch/mod.rs | 3 +- userland/toybox/src/arch/x86_64.rs | 37 ++++++++++++++++++- userland/toybox/src/debug_refused.rs | 12 +++--- userland/toybox/src/unmap_touch.rs | 22 ++++++----- 10 files changed, 148 insertions(+), 60 deletions(-) create mode 100644 issues/design-debt/the-x86-toybox-ships-two-applets-that-only-panic.md diff --git a/issues/design-debt/the-x86-toybox-ships-two-applets-that-only-panic.md b/issues/design-debt/the-x86-toybox-ships-two-applets-that-only-panic.md new file mode 100644 index 0000000000..7dae8903a8 --- /dev/null +++ b/issues/design-debt/the-x86-toybox-ships-two-applets-that-only-panic.md @@ -0,0 +1,18 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# The x86-64 toybox ships two applets that only panic + +`userland/toybox/src/main.rs` puts `fp_isolation` and `first_entry` in one +`commands!` list for every architecture, so the x86-64 toybox that ships in +every image answers both names, and `userland/toybox/src/arch/x86_64.rs`'s +body for each is a `panic!` naming where x86-64's probe lives or is owed +(`test_rs_fpu_isolation`, +`issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md`). +A shipping binary carries two commands whose only behaviour is to die. + +**Exit condition**: the x86-64 toybox answers neither name with a panic — +each is left out of its `commands!`, or runs a probe of x86-64's own. diff --git a/kernel/src/arch/aarch64/irqchip.rs b/kernel/src/arch/aarch64/irqchip.rs index 5211649956..c314dd518d 100644 --- a/kernel/src/arch/aarch64/irqchip.rs +++ b/kernel/src/arch/aarch64/irqchip.rs @@ -364,14 +364,13 @@ pub(super) fn rearm() { } /// `timer-floor`: this CPU's timer made due with interrupts masked, then -/// asked to fire within a quantum, which leaves it nothing to fire within; -/// then [`FLOOR_FIRES`] of its interrupts taken at EL1 with them open. A -/// re-arm shorter than [`MIN_ONE_SHOT`] is what an EL1 fire repeats, and one -/// that re-fires on its own return leaves this CPU no progress to say -/// anything with. +/// asked to fire within a quantum, which leaves it nothing to fire within. +/// The comparator it is left holding must be at least [`MIN_ONE_SHOT`] past +/// the counter read just before the ask. That reading tells a floored +/// comparator from an unfloored one only while the ask took less than the +/// floor, so a wider window is a `FAIL` too. #[cfg(feature = "boot-actuators")] pub fn floor_selftest() { - const FLOOR_FIRES: u32 = 100; let _guard = crate::arch::IrqGuard::close(); arm_one_shot(0); let due = || { @@ -381,15 +380,14 @@ pub fn floor_selftest() { ctl & TIMER_ISTATUS != 0 }; settles(100, "the timer armed for its floor", due); + let before = cpu::counter(); arm_within(toyos_sched::fair::QUANTUM_NS); - let (armed, floor) = (percpu::armed_ticks(), floor_ticks()); - let before = percpu::kernel_timer_fires(); - cpu::enable_interrupts(); - while percpu::kernel_timer_fires().wrapping_sub(before) < FLOOR_FIRES { - core::hint::spin_loop(); - } - cpu::disable_interrupts(); + let cval: u64; + // SAFETY: reads the EL1 virtual timer's comparator. + unsafe { core::arch::asm!("mrs {}, cntv_cval_el0", out(reg) cval, options(nomem, nostack, preserves_flags)) }; + let window = cpu::counter() - before; stop_timer(); - let verdict = if armed >= floor { "PASS" } else { "FAIL" }; - log!("timer-floor: {verdict} armed={armed} floor={floor} ticks: {FLOOR_FIRES} fires taken at EL1 re-armed for the floor"); + let (span, floor) = (cval.saturating_sub(before), floor_ticks()); + let verdict = if span >= floor && window < floor { "PASS" } else { "FAIL" }; + log!("timer-floor: {verdict} span={span} floor={floor} window={window} ticks: the comparator past the counter before the ask, and the ask's own width"); } diff --git a/kernel/src/arch/aarch64/paging.rs b/kernel/src/arch/aarch64/paging.rs index db065d75b2..80a48d9dba 100644 --- a/kernel/src/arch/aarch64/paging.rs +++ b/kernel/src/arch/aarch64/paging.rs @@ -323,7 +323,7 @@ fn alloc_asid() -> Option { match pool.alloc() { Alloc::Ready(tag) => return Some(AsidGuard(tag)), Alloc::NeedsFlush => { - tlb::shootdown(crate::invalidation::Origin::Pcid); + tlb::all(crate::invalidation::Origin::Pcid); pool.reclaim(); } Alloc::Exhausted => return None, diff --git a/kernel/src/arch/aarch64/tlb.rs b/kernel/src/arch/aarch64/tlb.rs index 51e836cf66..d46f19b02c 100644 --- a/kernel/src/arch/aarch64/tlb.rs +++ b/kernel/src/arch/aarch64/tlb.rs @@ -2,7 +2,8 @@ //! every CPU in the inner-shareable domain, and the `DSB ISH` after it //! returns once every one of them has dropped the entry (Arm ARM K.a, //! D8.13.4). So nothing here sends an interrupt or waits for an -//! acknowledgement, and [`shootdown`] is one instruction. +//! acknowledgement, and the page-table edit that clears an entry is the one +//! place its translation is dropped: [`shootdown`] has nothing left to do. //! //! Every operation below is bracketed the same way: `DSB ISHST` so the table //! write it answers for is visible to every walker first, then the `TLBI`, @@ -50,20 +51,19 @@ pub(super) fn kernel_page(va: u64) { tlbi!("vaae1is", (va >> 12) & 0xFFF_FFFF_FFFF); } -/// Every EL1&0 translation, on every CPU. -fn all() { +/// Every EL1&0 translation every CPU holds, dropped before this returns — +/// what a returned ASID needs before it is issued again. +pub(super) fn all(origin: Origin) { + ISSUED[origin as usize].fetch_add(1, Ordering::Relaxed); // SAFETY: as `tlbi!`'s. unsafe { core::arch::asm!("dsb ishst", "tlbi vmalle1is", "dsb ish", "isb", options(nostack, preserves_flags)); } } -/// Every translation every CPU holds, dropped before this returns — what a -/// returned ASID needs before it is issued again. -pub fn shootdown(origin: Origin) { - ISSUED[origin as usize].fetch_add(1, Ordering::Relaxed); - all(); -} +/// Nothing to drop: every caller follows an `AddressSpace` edit whose own +/// `page` or `asid` already reached every CPU before it returned. +pub fn shootdown(_origin: Origin) {} /// Nothing to answer: no CPU waits on another's acknowledgement here. pub fn poll() {} diff --git a/tests/toyos.rs b/tests/toyos.rs index 34942b86a6..f886b68c02 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -3552,9 +3552,6 @@ fn check_no_stale_cells(dump: &screen::Ppm, console: &str) -> Result<(), String> Ok(()) } -/// Run one screen test. `Err` carries the decoded screen, because a failure -/// here is almost always "the text is not what I expected" and the decoded -/// grid is the only readable form of that. /// Boot `tests/virtjobcase` under the EL2 profile and judge its job `job`: /// it ends with exit 0, having said `said`. The kernel carries `SYS_DEBUG` /// for `debug_refused`, and every job runs in every boot of the case. @@ -3573,11 +3570,16 @@ fn virt_job(job: &str, said: &str) -> Result<(), String> { }, ); let end = format!("===TEST_END {job} "); - let rest = qemu.drain_until(Duration::from_secs(300), |l| l.contains(&end)); + let mut rest = String::new(); + let waited = await_marker(&mut qemu, &mut rest, &end, &format!("the job {job} to end")); let serial = format!("{}\n{rest}", qemu.boot_log()); - let Some(ended) = serial.lines().find(|l| l.contains(&end)) else { - return Err(format!("the job {job} never ended\nserial:\n{serial}")); - }; + if let Err(why) = waited { + return Err(format!("{why}\nserial:\n{serial}")); + } + let ended = serial + .lines() + .find(|l| l.contains(&end)) + .expect("await_marker answered Ok, so the marker is in what it drained"); let Some(line) = serial.lines().find(|l| l.contains(said)) else { return Err(format!("{said:?} not on the PL011 ({ended})\nserial:\n{serial}")); }; @@ -3616,6 +3618,9 @@ fn virt_selftest(test_config: &Path, armed: &'static [&'static str; 1]) -> Resul Ok(()) } +/// Run one screen test. `Err` carries the decoded screen, because a failure +/// here is almost always "the text is not what I expected" and the decoded +/// grid is the only readable form of that. fn run_screen_test( name: &str, test_config: &Path, @@ -5113,7 +5118,10 @@ fn run_screen_test( "virt_fp_isolation" => virt_job("fp_isolation", "fp_isolation: v0-v31, FPCR and FPSR survived"), "virt_first_entry" => virt_job("first_entry", "first_entry: x1-x30 were zero"), "virt_unmap_touch" => virt_job("unmap_touch", "unmap_touch: 4 reads of a page just unmapped"), - "virt_debug_refused" => virt_job("debug_refused", "debug_refused: SYS_DEBUG's TLB acknowledgement delay was refused"), + "virt_debug_refused" => virt_job( + "debug_refused", + "debug_refused: SYS_DEBUG's double fault and TLB acknowledgement delay were refused", + ), "virt_irq_storm" => { // The CPU floods itself with SGIs until the timer has fired a // thousand times through the flood, then waits for every SGI it @@ -5121,12 +5129,7 @@ fn run_screen_test( // storm running and the verdict unsaid. virt_selftest(test_config, &["irq-storm"]) } - "virt_timer_floor" => { - // The timer made due and then asked to fire within a quantum: a - // re-arm shorter than the floor re-fires on its own return, and - // the CPU taking it at EL1 never gets back to say anything. - virt_selftest(test_config, &["timer-floor"]) - } + "virt_timer_floor" => virt_selftest(test_config, &["timer-floor"]), "screen_late_panic" => { // The ordinary fatal panic, which no userland process can produce: // crash_report, capture, panic_flush, halt_all_cpus, render. The diff --git a/userland/toybox/src/arch/aarch64.rs b/userland/toybox/src/arch/aarch64.rs index 144fd23e91..a07a1cc54b 100644 --- a/userland/toybox/src/arch/aarch64.rs +++ b/userland/toybox/src/arch/aarch64.rs @@ -1,6 +1,6 @@ -//! What an AArch64 thread finds in its registers after the kernel has had the -//! CPU: its own FP/SIMD state across the switches that took the CPU from it, -//! and nothing of the kernel's at its first instruction. +//! What an AArch64 thread finds after the kernel has had the CPU: its own +//! FP/SIMD state across the switches that took the CPU from it, nothing of the +//! kernel's at its first instruction, and no translation of a page it unmapped. /// `fp_isolation`: this thread pins a distinctive FP/SIMD state — v0–v31, /// `FPCR` and `FPSR` — and holds it while a sibling that loads another state @@ -237,3 +237,34 @@ pub mod first_entry { ); } } + +/// Reads the word at `page`, unmaps the `len` bytes there with `SYS_MUNMAP` +/// and reads the word again, in one block, so nothing but the unmap itself +/// can switch the CPU between the reads. Answers the first read, the unmap's +/// answer and the second read. +/// +/// # Safety +/// `page` starts a mapping of `len` bytes that nothing else names. The second +/// read ends the process unless the unmap was refused or left its +/// translation standing. +pub unsafe fn read_unmap_read(page: *const u64, len: usize) -> (u64, u64, u64) { + let (first, answer, second): (u64, u64, u64); + // SAFETY: the caller's; the two loads name `page` and the `svc` is the + // ABI's (`toyos_abi::syscall`), which preserves every register but x0. + unsafe { + core::arch::asm!( + "ldr {first}, [{page}]", + "svc #0", + "ldr {second}, [{page}]", + page = in(reg) page, + first = out(reg) first, + second = lateout(reg) second, + inlateout("x0") toyos_abi::syscall::SYS_MUNMAP => answer, + in("x1") page, + in("x2") len, + in("x3") 0u64, + in("x4") 0u64, + ); + } + (first, answer, second) +} diff --git a/userland/toybox/src/arch/mod.rs b/userland/toybox/src/arch/mod.rs index 157ce79a88..95e0e9d154 100644 --- a/userland/toybox/src/arch/mod.rs +++ b/userland/toybox/src/arch/mod.rs @@ -1,5 +1,4 @@ -//! The applets that ask a CPU what the kernel left in its registers, one -//! module per architecture. +//! What toybox's applets must say in assembly, one module per architecture. #[cfg(target_arch = "aarch64")] mod aarch64; diff --git a/userland/toybox/src/arch/x86_64.rs b/userland/toybox/src/arch/x86_64.rs index 9e54bfec8c..c08523a994 100644 --- a/userland/toybox/src/arch/x86_64.rs +++ b/userland/toybox/src/arch/x86_64.rs @@ -1,4 +1,5 @@ -//! x86-64 has these probes elsewhere, or owes them. +//! x86-64's half of `unmap_touch`; its FP and first-entry probes are +//! elsewhere, or owed. pub mod fp_isolation { pub fn main(_args: Vec) { @@ -14,3 +15,37 @@ pub mod first_entry { ); } } + +/// Reads the word at `page`, unmaps the `len` bytes there with `SYS_MUNMAP` +/// and reads the word again, in one block, so nothing but the unmap itself +/// can switch the CPU between the reads. Answers the first read, the unmap's +/// answer and the second read. +/// +/// # Safety +/// `page` starts a mapping of `len` bytes that nothing else names. The second +/// read ends the process unless the unmap was refused or left its +/// translation standing. +pub unsafe fn read_unmap_read(page: *const u64, len: usize) -> (u64, u64, u64) { + let (first, answer, second): (u64, u64, u64); + // SAFETY: the caller's; the two loads name `page` and the `syscall` is + // the ABI's (`toyos_abi::syscall`), which clobbers rax, rcx and r11. + unsafe { + core::arch::asm!( + "mov {first}, qword ptr [{page}]", + "syscall", + "mov {second}, qword ptr [{page}]", + page = in(reg) page, + first = out(reg) first, + second = lateout(reg) second, + in("rdi") toyos_abi::syscall::SYS_MUNMAP, + in("rsi") page, + in("rdx") len, + in("r8") 0u64, + in("r9") 0u64, + out("rax") answer, + out("rcx") _, + out("r11") _, + ); + } + (first, answer, second) +} diff --git a/userland/toybox/src/debug_refused.rs b/userland/toybox/src/debug_refused.rs index de062fda28..00cf3350f0 100644 --- a/userland/toybox/src/debug_refused.rs +++ b/userland/toybox/src/debug_refused.rs @@ -1,18 +1,20 @@ //! `debug_refused`: a `SYS_DEBUG` action this machine has no counterpart for -//! is refused, and the kernel lives: the TLB acknowledgement delay, which -//! AArch64's broadcast invalidation has no acknowledgement for. Needs a -//! kernel built with `test-actuators`. +//! is refused, and the kernel lives: the double fault, which AArch64 has no +//! exception for, and the TLB acknowledgement delay, which its broadcast +//! invalidation has no acknowledgement for. Needs a kernel built with +//! `test-actuators`. -use toyos_abi::syscall::debug_action::{TLB_ACK_DELAY_ARM, TLB_ACK_DELAY_DISARM}; +use toyos_abi::syscall::debug_action::{DOUBLE_FAULT, TLB_ACK_DELAY_ARM, TLB_ACK_DELAY_DISARM}; use toyos_abi::syscall::{debug, debug_with, SyscallError}; pub fn main(_args: Vec) { let refused = SyscallError::NotSupported.to_u64(); for (name, answer) in [ + ("DOUBLE_FAULT", debug(DOUBLE_FAULT)), ("TLB_ACK_DELAY_ARM", debug_with(TLB_ACK_DELAY_ARM, 1_000_000)), ("TLB_ACK_DELAY_DISARM", debug(TLB_ACK_DELAY_DISARM)), ] { assert_eq!(answer, refused, "SYS_DEBUG {name} answered {answer:#x}, not NotSupported"); } - println!("debug_refused: SYS_DEBUG's TLB acknowledgement delay was refused, twice"); + println!("debug_refused: SYS_DEBUG's double fault and TLB acknowledgement delay were refused"); } diff --git a/userland/toybox/src/unmap_touch.rs b/userland/toybox/src/unmap_touch.rs index f0ff799684..7acc4b9697 100644 --- a/userland/toybox/src/unmap_touch.rs +++ b/userland/toybox/src/unmap_touch.rs @@ -1,15 +1,16 @@ //! `unmap_touch`: whether a page's unmapping reaches the TLB. A child writes //! a page, unmaps it and reads it at once; the read must end the child, //! because a translation left cached past the unmap still reaches the frame. -//! Several children, so the one read that a switch between the unmap and the -//! read would have made fault anyway cannot decide the verdict alone. use std::process::Command; -use toyos_abi::syscall::{mmap, munmap, MmapFlags, MmapProt}; +use toyos_abi::syscall::{mmap, MmapFlags, MmapProt}; const TRIALS: usize = 4; const PAGE: usize = 2 * 1024 * 1024; +/// The status `kill_process(-1)` gives a process whose fault nothing serves; +/// a refused unmap or a panic ends the child with another. +const FAULTED: i32 = -1; /// The child's last line before the unmap and the read. const UNMAPPING: &str = "unmap_touch: written; unmapping and reading"; /// The child's line if the read came back. @@ -32,8 +33,8 @@ fn judge() { let said = String::from_utf8_lossy(&out.stdout); assert!(said.contains(UNMAPPING), "trial {trial}: the child never reached the unmap: {said:?}"); assert!( - !out.status.success() && !said.contains(READ_BACK), - "trial {trial}: a read of an unmapped page came back, and the child exited {:?}: {said:?}", + out.status.code() == Some(FAULTED) && !said.contains(READ_BACK), + "trial {trial}: the child exited {:?}, not faulted ({FAULTED}), on the read of a page it had just unmapped: {said:?}", out.status.code(), ); } @@ -47,9 +48,10 @@ fn touch() { // SAFETY: inside the mapping just made. unsafe { page.write_volatile(0x5A) }; println!("{UNMAPPING}"); - // SAFETY: the region `mmap` returned, whole; nothing else holds it. - unsafe { munmap(page, PAGE) }.expect("unmap_touch: munmap refused"); - // SAFETY: none — the read of an unmapped page is what this child exists to make. - let value = unsafe { page.read_volatile() }; - println!("{READ_BACK} {value:#x} after the unmap"); + // After the last print: a print can switch the CPU, and a switch can drop + // the translation the first read caches. + // SAFETY: the mapping just made, whole, which nothing else names; the + // second read ending this process is what the child exists to make. + let (first, answer, second) = unsafe { crate::arch::read_unmap_read(page.cast(), PAGE) }; + println!("{READ_BACK} {second:#x} after the unmap answered {answer:#x}, {first:#x} before it"); } From 4256e5af737c744f88c33c958709919cd44c66e9 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 08:17:30 +0200 Subject: [PATCH 06/11] Drop timer-floor's timing verdict, keep only the value relation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit PR #562 forbids a time verdict in a QEMU selftest: under TCG load a slow call must not fail a test that measures a value, not a duration. The timer-floor selftest's `window < floor` clause was exactly that — it failed whenever the arm_within call itself ran long, for no defect. The verdict is now only CVAL >= counter_before + floor_ticks(), which holds however long the call took. Co-Authored-By: Claude Opus 5.5 --- kernel/src/arch/aarch64/irqchip.rs | 9 +++------ 1 file changed, 3 insertions(+), 6 deletions(-) diff --git a/kernel/src/arch/aarch64/irqchip.rs b/kernel/src/arch/aarch64/irqchip.rs index c314dd518d..3916fd65f0 100644 --- a/kernel/src/arch/aarch64/irqchip.rs +++ b/kernel/src/arch/aarch64/irqchip.rs @@ -366,9 +366,7 @@ pub(super) fn rearm() { /// `timer-floor`: this CPU's timer made due with interrupts masked, then /// asked to fire within a quantum, which leaves it nothing to fire within. /// The comparator it is left holding must be at least [`MIN_ONE_SHOT`] past -/// the counter read just before the ask. That reading tells a floored -/// comparator from an unfloored one only while the ask took less than the -/// floor, so a wider window is a `FAIL` too. +/// the counter read just before the ask. #[cfg(feature = "boot-actuators")] pub fn floor_selftest() { let _guard = crate::arch::IrqGuard::close(); @@ -385,9 +383,8 @@ pub fn floor_selftest() { let cval: u64; // SAFETY: reads the EL1 virtual timer's comparator. unsafe { core::arch::asm!("mrs {}, cntv_cval_el0", out(reg) cval, options(nomem, nostack, preserves_flags)) }; - let window = cpu::counter() - before; stop_timer(); let (span, floor) = (cval.saturating_sub(before), floor_ticks()); - let verdict = if span >= floor && window < floor { "PASS" } else { "FAIL" }; - log!("timer-floor: {verdict} span={span} floor={floor} window={window} ticks: the comparator past the counter before the ask, and the ask's own width"); + let verdict = if span >= floor { "PASS" } else { "FAIL" }; + log!("timer-floor: {verdict} span={span} floor={floor} ticks: the comparator past the counter before the ask"); } From 8cda9b22e5a5750942f072847c6fca13edbd7e41 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 09:09:39 +0200 Subject: [PATCH 07/11] timer-floor: verdict off the counter arm_within set CVAL from, not a framing read MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit PR #562's rule against a timing verdict in a QEMU selftest made `floor_selftest` compare CVAL to a counter read taken before the call to `arm_within`. Under TCG that call itself can run tens of thousands of ticks long — far past `floor_ticks()` — so `span >= floor` held whether or not the floor clamp fired: a QEMU host under load hid a missing or bypassed clamp rather than catching it. `arm_ticks` (and `arm_within`, which ends in it) now return the counter value they read to compute CVAL, so the selftest relates CVAL to the exact `now` the arm used — a value relation, not a second, independent read framed around however long the call took. Co-Authored-By: Claude Opus 5.5 --- kernel/src/arch/aarch64/irqchip.rs | 26 +++++++++++++++----------- 1 file changed, 15 insertions(+), 11 deletions(-) diff --git a/kernel/src/arch/aarch64/irqchip.rs b/kernel/src/arch/aarch64/irqchip.rs index 3916fd65f0..940d7a53f1 100644 --- a/kernel/src/arch/aarch64/irqchip.rs +++ b/kernel/src/arch/aarch64/irqchip.rs @@ -302,10 +302,12 @@ fn floor_ticks() -> u64 { /// The only write of the comparator: fire `ticks` counter ticks from now, or /// after [`MIN_ONE_SHOT`] if that is longer, and remember the span as what an -/// EL1 fire re-arms with. -fn arm_ticks(ticks: u64) { +/// EL1 fire re-arms with. Returns the counter value the comparator was set +/// from, for a caller that must relate CVAL back to it without a second read. +fn arm_ticks(ticks: u64) -> u64 { let ticks = ticks.max(floor_ticks()); percpu::set_armed_ticks(ticks); + let now = cpu::counter(); // SAFETY: the EL1 virtual timer's comparator and control; CPACR has // nothing to say about them and `CNTKCTL_EL1` keeps EL0 out. unsafe { @@ -313,11 +315,12 @@ fn arm_ticks(ticks: u64) { "msr cntv_cval_el0, {cval}", "msr cntv_ctl_el0, {enable}", "isb", - cval = in(reg) cpu::counter() + ticks, + cval = in(reg) now + ticks, enable = in(reg) TIMER_ENABLE, options(nomem, nostack, preserves_flags), ); } + now } fn stop_timer_hardware() { @@ -332,8 +335,9 @@ pub fn arm_one_shot(nanos: u64) { } /// This CPU's timer, armed to fire within `nanos`: sooner than it is armed -/// for, or armed if it is stopped. -pub fn arm_within(nanos: u64) { +/// for, or armed if it is stopped. Returns the counter value the comparator +/// was set from ([`arm_ticks`]). +pub fn arm_within(nanos: u64) -> u64 { let want = crate::clock::counter_ticks(nanos); let remaining = match percpu::armed_ticks() { 0 => want, @@ -344,7 +348,7 @@ pub fn arm_within(nanos: u64) { cval.saturating_sub(cpu::counter()) } }; - arm_ticks(want.min(remaining)); + arm_ticks(want.min(remaining)) } /// Stop the timer: no interrupt until it is armed again. @@ -366,7 +370,8 @@ pub(super) fn rearm() { /// `timer-floor`: this CPU's timer made due with interrupts masked, then /// asked to fire within a quantum, which leaves it nothing to fire within. /// The comparator it is left holding must be at least [`MIN_ONE_SHOT`] past -/// the counter read just before the ask. +/// the counter value [`arm_within`] set it from — not a counter read framing +/// the call, which a slow call (a QEMU host under load) widens for no defect. #[cfg(feature = "boot-actuators")] pub fn floor_selftest() { let _guard = crate::arch::IrqGuard::close(); @@ -378,13 +383,12 @@ pub fn floor_selftest() { ctl & TIMER_ISTATUS != 0 }; settles(100, "the timer armed for its floor", due); - let before = cpu::counter(); - arm_within(toyos_sched::fair::QUANTUM_NS); + let now = arm_within(toyos_sched::fair::QUANTUM_NS); let cval: u64; // SAFETY: reads the EL1 virtual timer's comparator. unsafe { core::arch::asm!("mrs {}, cntv_cval_el0", out(reg) cval, options(nomem, nostack, preserves_flags)) }; stop_timer(); - let (span, floor) = (cval.saturating_sub(before), floor_ticks()); + let (span, floor) = (cval.saturating_sub(now), floor_ticks()); let verdict = if span >= floor { "PASS" } else { "FAIL" }; - log!("timer-floor: {verdict} span={span} floor={floor} ticks: the comparator past the counter before the ask"); + log!("timer-floor: {verdict} span={span} floor={floor} ticks: the comparator past the counter it was set from"); } From 07515372e9539c6c9b03ac5eef5d1016e0140778 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 09:12:02 +0200 Subject: [PATCH 08/11] Discard rearm's arm_ticks return: the match arm never needed it arm_ticks now returns the counter value it armed the comparator from, for floor_selftest. rearm's match still had a bare `arm_ticks(ticks)` arm against `stop_timer_hardware()`'s `()`, which no longer type-checks. Co-Authored-By: Claude Opus 5.5 --- kernel/src/arch/aarch64/irqchip.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/kernel/src/arch/aarch64/irqchip.rs b/kernel/src/arch/aarch64/irqchip.rs index 940d7a53f1..676f0e8050 100644 --- a/kernel/src/arch/aarch64/irqchip.rs +++ b/kernel/src/arch/aarch64/irqchip.rs @@ -363,7 +363,9 @@ pub fn stop_timer() { pub(super) fn rearm() { match percpu::armed_ticks() { 0 => stop_timer_hardware(), - ticks => arm_ticks(ticks), + ticks => { + arm_ticks(ticks); + } } } From 57923550e0addcd8860c4c91bf665ec489ab99a6 Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 15:12:16 +0200 Subject: [PATCH 09/11] PR #589 round 7: abuse_readonly_copyout runs on the AArch64 job case The review's blocker: no AArch64 test could fail on `leaf()`'s Write check. `test_rs_abuse_readonly_copyout` issued its calls with the x86 `syscall` instruction; it now goes through `toyos_abi::syscall`'s typed wrappers, with the target named by a slice over the read-only address (as `tls_dtv_race` does), so it builds for both architectures and its row in the source gate's assembly exemptions goes. - The typed-value arm was `fstat`, whose wrapper returns its `Stat` rather than taking an address. It is now `process_stats` of a child the test spawns first: the same `copy_out` path, reachable through a wrapper that takes the caller's `&mut`. - Every arm runs; the exit status carries one bit per arm that let a write through (1 read-only mmap, 2 own text, 4 straddle, 8 clock page), which the kernel's own exit record prints even once a rewritten clock page has taken logd down. The straddle arm's search for the first unwritable page reports rather than panics when no page refuses, so a kernel that grants every write still reaches the clock arm. - `tests/virtjobcase` runs it as its last job; `virt_job` builds that one binary for AArch64 (`build::build_toyos_bin`, the crate's other binaries do not all build there) and puts it on ROOT for every boot of the case. `virt_readonly_copyout` judges it. The x86 shared run stays and is declared in DRIVEN_AND_SHARED. Also: the device-memory issue now names `user_ptr`'s direct-map copies and `dump_crash_diagnostics` beside `read_user_word`; the arm64 track drops the KernelArgs clause #583 made false; the assembly issue drops its probe count. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01U6SVYFkdvV2t38KzNrESxs --- ...rch-module-in-userland-and-guest-probes.md | 2 +- ...rash-report-reads-through-any-user-leaf.md | 27 ++- issues/kernel/toyos-runs-on-arm64.md | 3 +- src/build.rs | 15 ++ src/sourcegate.rs | 1 - tests/common/qemu.rs | 7 + .../src/bin/abuse_readonly_copyout.rs | 216 +++++++++++------- tests/toyos.rs | 21 +- tests/virtjobcase/system.toml | 8 +- 9 files changed, 203 insertions(+), 97 deletions(-) diff --git a/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md b/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md index 6035bb6f8a..e0fa421979 100644 --- a/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md +++ b/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md @@ -18,7 +18,7 @@ pointing here: `_mm_sfence` after writing a write-combining framebuffer. Userland has no portable way to say "drain my stores to the scanout"; the SDK (`toyos/src`) owes one, and it is also only changed under an ABI brief. -- Twenty guest probes in `tests/toyos-rust-tests/src/bin/`, whose subject +- Guest probes in `tests/toyos-rust-tests/src/bin/`, whose subject is an x86 instruction (`rdgsbase`, `fxsave64`, `int1`, x87 control words) or the raw `syscall` gate with arguments no SDK call will pass. They run in the x86-64 suite, which is the only suite until the harness gains its arch diff --git a/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md b/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md index 40e669d059..d727b3fb36 100644 --- a/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md +++ b/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md @@ -15,9 +15,28 @@ once the port's stage 6 maps one into a process — names an address the direct map does not hold, and the report's read of it is an EL1 data abort: a user fault with `x29` pointed into its own BAR ends the machine. -Nothing maps a BAR into an AArch64 process yet, so this is latent until +Two more readers take the same leaf to the direct map: + +- **Every syscall's user copy.** `AddressSpace::leaf` answers a Device + leaf as it answers a Normal one, for a `Write` too where the leaf is + EL0 read-write, and `kernel/src/user_ptr.rs` copies through + `DirectMap::from_phys` of what it answered: `translate_now`, `object_run` + (`copy_in`, `copy_out`) and `View::pieces` (`UserBytes::read_at`, + `UserBytesMut::write_at` and their kin). A `read(fd, bar, n)` is an EL1 + abort on an address the direct map does not hold, or, where `map_mmio` + holds it Device-nGnRE, an alignment fault on `memcpy`'s unaligned access: + a kernel panic a process chose. +- **The fault dump.** `kernel/src/process.rs`'s `dump_crash_diagnostics`, + which AArch64's `trap.rs` calls on an EL0 fault, reads the words around + the fault address and the faulting PC through `translate` and the direct + map, the same way. + +Nothing maps a BAR into an AArch64 process yet — `arch::msi_message` +refuses there, so pcidev refuses every hand-over — so this is latent until stage 6 of `issues/kernel/toyos-runs-on-arm64.md`. -**Exit condition**: `read_user_word` reads only a leaf of the memory type -the direct map holds, and a guest test whose process faults with `x29` -inside a mapped BAR sees its process end and the kernel live. +**Exit condition**: `read_user_word`, `leaf` and the fault dump's +`translate` answer only a leaf of the memory type the direct map holds, and +guest tests see a process end and the kernel live when it faults with `x29` +inside a mapped BAR, faults at an address inside one, and names one as a +`read`'s buffer — the last refused with `BadAddress`. diff --git a/issues/kernel/toyos-runs-on-arm64.md b/issues/kernel/toyos-runs-on-arm64.md index 6f16cce62e..6ca1f8f319 100644 --- a/issues/kernel/toyos-runs-on-arm64.md +++ b/issues/kernel/toyos-runs-on-arm64.md @@ -268,8 +268,7 @@ Each stage names its exit; "measured" means a number from a run. virtio-rng. Each judges an event, never a rate: no QEMU test measures time. Owed before the exit holds: the interrupts-off window against x86's, a measurement only metal can make, with no instrument on either arch yet; - `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md`, whose - `KernelArgs` rename waits on the loader's change to that struct; the + `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md`; the instruction-cache maintenance before a mapping is executable (`cache::make_executable`), the break-before-make ordering of a live entry's replacement, and the TLB flush before a reclaimed ASID is issued diff --git a/src/build.rs b/src/build.rs index 5796e26921..a5bf170c47 100644 --- a/src/build.rs +++ b/src/build.rs @@ -2286,6 +2286,21 @@ pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) results } +/// The one binary `name` of the crate at `crate_path`, built for `arch`: for an +/// architecture the crate's other binaries do not all build for. +pub fn build_toyos_bin(root: &Path, arch: Arch, crate_path: &Path, name: &str, quiet: bool) -> Vec { + let target = arch.userland(); + let mut lock = buildlock::shared(root, "a test binary"); + let sysroot = crate::toolchain::ensure(root, false, &mut lock); + let env = GuestEnv::new(&sysroot); + invalidate_stale(root, &mut lock, &env.toolchain, &[(crate_path.to_path_buf(), Clean::All)]); + // Built and read under one hold, as `build_toyos_bins`' are. + let _artifact = buildlock::artifact(root); + cargo_build(crate_path, target, &["--bin", name], &env, &[], quiet); + let binary = crate_path.join(format!("target/{target}/{PROFILE}/{name}")); + fs::read(&binary).unwrap_or_else(|e| panic!("read the test binary {}: {e}", binary.display())) +} + // --- Internal helpers --- /// The ToyOS-hosted rustc, and the target libraries it compiles against: this diff --git a/src/sourcegate.rs b/src/sourcegate.rs index b13d17a5b9..591c9103cb 100644 --- a/src/sourcegate.rs +++ b/src/sourcegate.rs @@ -1490,7 +1490,6 @@ const ARCH_RULES: &[PlaceRule] = &[ ("userland/toyos-window/src/arch/", "toyos-window's architecture modules"), ("tests/toyos-rust-tests/src/bin/abuse_kernel_addr.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs", USERLAND_ASM), - ("tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/abuse_tls_alloc.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/copy_out_races_munmap.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/debug_trap.rs", USERLAND_ASM), diff --git a/tests/common/qemu.rs b/tests/common/qemu.rs index fb4b98710d..f9c06f7508 100644 --- a/tests/common/qemu.rs +++ b/tests/common/qemu.rs @@ -3004,6 +3004,13 @@ pub fn build_toyos_bins(crate_path: &Path) -> Vec<(String, Vec)> { toyos_build::build::build_toyos_bins(&repo, SUITE_ARCH, crate_path, quiet) } +/// One binary of a test crate, built for `arch`: for a guest the crate's other +/// binaries do not all build for. +pub fn build_toyos_bin(arch: Arch, crate_path: &Path, name: &str) -> Vec { + let quiet = !VERBOSE.load(Ordering::Relaxed); + toyos_build::build::build_toyos_bin(&compile::repo_root(), arch, crate_path, name, quiet) +} + /// A kernel record's console line: `klogd` renders every one with this head, /// and nothing else writes it — a program's line reaches the console only /// through `logd`, under the program's own head. diff --git a/tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs b/tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs index cf87aa185d..718940dd6c 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs @@ -7,22 +7,27 @@ //! //! **The memory is the verdict, before the return value**: each arm snapshots //! the target, makes the call, and compares, so a kernel that wrote and then -//! answered an error still fails here. +//! answered an error still fails here. Each arm asks through both copies a +//! syscall writes with: `read`'s bulk window and `process_stats`'s typed value. +//! +//! **Every arm runs, and the exit status carries one bit per arm that let a +//! write through**, which the kernel's own exit record prints: a rewritten +//! clock page asserts in every stamp any process takes, `logd`'s among them, so +//! the clock arm's verdict may reach no other line. +use std::os::toyos::process::ChildExt; use std::process::Command; use toyos_abi::clock::{ClockPage, CLOCK_MAGIC, CLOCK_PAGE}; -use toyos_abi::syscall::{ - self, MmapFlags, MmapProt, OpenFlags, SeekFrom, SyscallError, SYS_FSTAT, SYS_READ, SYS_WRITE, -}; +use toyos_abi::syscall::{self, MmapFlags, MmapProt, OpenFlags, ProcessStats, SeekFrom, SyscallError}; use toyos_abi::RawHandle; const SELF_PATH: &str = "/system/bin/test_rs_abuse_readonly_copyout"; const CHILD_ARG: &str = "reads-the-clock"; const PAGE_2M: usize = 2 * 1024 * 1024; const PAGE_4K: u64 = 4096; -/// Longer than `Stat`, and ends inside the file this reads. -const LEN: usize = 64; +/// Covers a `ProcessStats`, and ends inside the file this reads. +const LEN: usize = core::mem::size_of::(); /// Bytes `0..=255`, so a probe can offer any page its own byte back, then the /// bytes the straddling `read` offers, at [`STRADDLE_AT`]. const PROBE_PATH: &[u8] = b"/tmp/abuse_readonly_copyout.bin"; @@ -30,29 +35,16 @@ const STRADDLE_AT: u64 = 256; /// The straddling `read`: half on the last writable page, half on the next. const STRADDLE: usize = 16; +/// Each arm's bit in the exit status. +const MMAP_BIT: i32 = 1; +const TEXT_BIT: i32 = 2; +const STRADDLE_BIT: i32 = 4; +const CLOCK_BIT: i32 = 8; + /// In `.bss`, so its 2 MiB window also holds the pages past the image's end, /// which the pager maps read-only. static mut BSS: u8 = 0; -/// The typed wrappers take a `&mut [u8]`, which a read-only page cannot be. -fn raw(num: u64, a1: u64, a2: u64, a3: u64) -> u64 { - let ret: u64; - unsafe { - core::arch::asm!( - "syscall", - in("rdi") num, - in("rsi") a1, - in("rdx") a2, - in("r8") a3, - in("r9") 0u64, - lateout("rax") ret, - out("rcx") _, - out("r11") _, - ); - } - ret -} - fn snapshot(addr: u64) -> [u8; LEN] { let mut out = [0u8; LEN]; for (i, b) in out.iter_mut().enumerate() { @@ -66,66 +58,122 @@ fn open_self() -> RawHandle { syscall::open(SELF_PATH.as_bytes(), OpenFlags::READ).expect("open self") } -/// `read` and `fstat` into `addr`: both refused, and not one byte moved. -fn refused(what: &str, addr: u64) { +/// `len` bytes at `addr`, as the buffer a `read` is asked to fill. +/// +/// # Safety +/// Nothing reads or writes through the slice while it lives: it only names the +/// address the kernel is asked to write, which may be a page nothing may write. +unsafe fn target(addr: u64, len: usize) -> &'static mut [u8] { + unsafe { core::slice::from_raw_parts_mut(addr as *mut u8, len) } +} + +/// `read` and `process_stats` (of `stats`) into `addr`, which is 8-aligned: +/// both refused, and not one byte moved. +fn refused(what: &str, addr: u64, stats: RawHandle) -> Result<(), String> { let before = snapshot(addr); let fd = open_self(); - let ret = raw(SYS_READ, fd.0 as u64, addr, LEN as u64); - assert_eq!(before, snapshot(addr), "read wrote into {what}"); - assert_eq!(SyscallError::from_u64(ret), Some(SyscallError::BadAddress), "read into {what}: {ret:#x}"); - let ret = raw(SYS_FSTAT, fd.0 as u64, addr, 0); - assert_eq!(before, snapshot(addr), "fstat wrote into {what}"); - assert_eq!(SyscallError::from_u64(ret), Some(SyscallError::BadAddress), "fstat into {what}: {ret:#x}"); + let ret = syscall::read(fd, unsafe { target(addr, LEN) }); syscall::close(fd); + if before != snapshot(addr) { + return Err(format!("read wrote into {what}")); + } + if ret != Err(SyscallError::BadAddress) { + return Err(format!("read into {what}: {ret:?}")); + } + // SAFETY: as `target`'s; `ProcessStats` is 8-aligned and every bit pattern is one. + let ret = syscall::process_stats(stats, unsafe { &mut *(addr as *mut ProcessStats) }); + if before != snapshot(addr) { + return Err(format!("process_stats wrote into {what}")); + } + if ret != Err(SyscallError::BadAddress) { + return Err(format!("process_stats into {what}: {ret:?}")); + } + Ok(()) +} + +fn refused_mmap(stats: RawHandle) -> Result<(), String> { + let ro = unsafe { + syscall::mmap( + core::ptr::null_mut(), + PAGE_2M, + MmapProt::READ, + MmapFlags::ANONYMOUS | MmapFlags::PRIVATE, + ) + }; + assert!(!ro.is_null(), "mmap a read-only page"); + let verdict = refused("a read-only mmap", ro as u64, stats); + unsafe { syscall::munmap(ro, PAGE_2M) }.expect("munmap"); + verdict } /// The first page at or above [`BSS`], inside its 2 MiB window, a `read` may /// not write; each page below it took a 1-byte `read` of its own first byte. -fn first_unwritable_page(probe: RawHandle) -> u64 { +fn first_unwritable_page(probe: RawHandle) -> Option { let pipe = syscall::pipe().expect("pipe"); let bss = &raw const BSS as u64; let window_end = (bss & !(PAGE_2M as u64 - 1)) + PAGE_2M as u64; let mut page = bss & !(PAGE_4K - 1); + let mut found = None; while page < window_end { - let ret = raw(SYS_WRITE, pipe.write.0 as u64, page, 1); - assert_eq!(ret, 1, "write from {page:#x}, in .bss's window: {ret:#x}"); + let from = unsafe { core::slice::from_raw_parts(page as *const u8, 1) }; + assert_eq!(syscall::write(pipe.write, from), Ok(1), "write from {page:#x}, in .bss's window"); syscall::read(pipe.read, &mut [0u8; 1]).expect("drain the pipe"); let own = unsafe { (page as *const u8).read_volatile() }; syscall::seek(probe, SeekFrom::Start(own as u64)).expect("seek the probe"); - let ret = raw(SYS_READ, probe.0 as u64, page, 1); - if SyscallError::from_u64(ret) == Some(SyscallError::BadAddress) { - syscall::close(pipe.read); - syscall::close(pipe.write); - return page; + let ret = syscall::read(probe, unsafe { target(page, 1) }); + if ret == Err(SyscallError::BadAddress) { + found = Some(page); + break; } - assert_eq!(ret, 1, "a 1-byte read into {page:#x}: {ret:#x}"); + assert_eq!(ret, Ok(1), "a 1-byte read into {page:#x}"); page += PAGE_4K; } - panic!("no page above .bss at {bss:#x} refuses a write below {window_end:#x}: the image ends on its window's edge"); + syscall::close(pipe.read); + syscall::close(pipe.write); + found } /// A `read` that starts on a writable page and runs into the read-only page /// after it, in one 2 MiB window, is refused whole. -fn refused_across_pages() { +fn refused_across_pages() -> Result<(), String> { let flags = OpenFlags::READ | OpenFlags::WRITE | OpenFlags::CREATE | OpenFlags::TRUNCATE; let probe = syscall::open(PROBE_PATH, flags).expect("create the probe file"); let every: Vec = (0..=255).collect(); syscall::write(probe, &every).expect("fill the probe file"); - let page = first_unwritable_page(probe); - assert!(page > &raw const BSS as u64, "BSS's own page refuses a write"); - let at = page - (STRADDLE / 2) as u64; - let before = snapshot(at); - // The writable half is offered its own bytes and the read-only half their - // complement, so a write to the read-only page shows and one below harms nothing. - let offered: [u8; STRADDLE] = core::array::from_fn(|i| if i < STRADDLE / 2 { before[i] } else { !before[i] }); - syscall::seek(probe, SeekFrom::Start(STRADDLE_AT)).expect("seek the probe"); - syscall::write(probe, &offered).expect("write the straddle's bytes"); - syscall::seek(probe, SeekFrom::Start(STRADDLE_AT)).expect("seek the probe"); - let ret = raw(SYS_READ, probe.0 as u64, at, STRADDLE as u64); - assert_eq!(before, snapshot(at), "read wrote across a writable page into the read-only one at {page:#x}"); - assert_eq!(SyscallError::from_u64(ret), Some(SyscallError::BadAddress), "read across into {page:#x}: {ret:#x}"); + let verdict = match first_unwritable_page(probe) { + None => Err(format!( + "no page above .bss at {:#x} refuses a write in its 2 MiB window", + &raw const BSS as u64 + )), + Some(page) => { + assert!(page > &raw const BSS as u64, "BSS's own page refuses a write"); + let at = page - (STRADDLE / 2) as u64; + let before = snapshot(at); + // The writable half is offered its own bytes and the read-only half their + // complement, so a write to the read-only page shows and one below harms nothing. + let offered: [u8; STRADDLE] = + core::array::from_fn(|i| if i < STRADDLE / 2 { before[i] } else { !before[i] }); + syscall::seek(probe, SeekFrom::Start(STRADDLE_AT)).expect("seek the probe"); + syscall::write(probe, &offered).expect("write the straddle's bytes"); + syscall::seek(probe, SeekFrom::Start(STRADDLE_AT)).expect("seek the probe"); + let ret = syscall::read(probe, unsafe { target(at, STRADDLE) }); + if before != snapshot(at) { + Err(format!("read wrote across a writable page into the read-only one at {page:#x}")) + } else if ret != Err(SyscallError::BadAddress) { + Err(format!("read across into {page:#x}: {ret:?}")) + } else { + Ok(()) + } + } + }; syscall::close(probe); syscall::delete(PROBE_PATH).expect("delete the probe file"); + verdict +} + +/// A second process still reads the clock. +fn second_reader() -> bool { + Command::new(SELF_PATH).arg(CHILD_ARG).status().expect("spawn the second reader").success() } fn main() { @@ -136,38 +184,36 @@ fn main() { } // The control: the same calls into memory this process may write succeed, - // so every refusal below is about the page and nothing else. + // so every refusal below is about the page and nothing else. The first + // reader is the process `process_stats` answers for. + let mut first = Command::new(SELF_PATH).arg(CHILD_ARG).spawn().expect("spawn the first reader"); + assert!(first.wait().expect("wait for the first reader").success(), "the first reader failed"); + let stats = RawHandle(first.as_raw_handle()); let mut ok = [0u8; LEN]; let fd = open_self(); - let ret = raw(SYS_READ, fd.0 as u64, ok.as_mut_ptr() as u64, LEN as u64); - assert_eq!(ret, LEN as u64, "read into a writable buffer: {ret:#x}"); + assert_eq!(syscall::read(fd, &mut ok), Ok(LEN), "read into a writable buffer"); assert_eq!(&ok[..4], b"\x7fELF", "read put something other than this file in the buffer"); syscall::close(fd); + syscall::process_stats(stats, &mut ProcessStats::default()).expect("process_stats into a writable value"); + + let mut wrote = 0; + for (bit, verdict) in [ + (MMAP_BIT, refused_mmap(stats)), + (TEXT_BIT, refused("this program's own text", main as *const () as u64 & !7, stats)), + (STRADDLE_BIT, refused_across_pages()), + ] { + if let Err(why) = verdict { + eprintln!("{why}"); + wrote |= bit; + } + } - let ro = unsafe { - syscall::mmap( - core::ptr::null_mut(), - PAGE_2M, - MmapProt::READ, - MmapFlags::ANONYMOUS | MmapFlags::PRIVATE, - ) - }; - assert!(!ro.is_null(), "mmap a read-only page"); - refused("a read-only mmap", ro as u64); - unsafe { syscall::munmap(ro, PAGE_2M) }.expect("munmap"); - - refused("this program's own text", main as *const () as u64); - refused_across_pages(); - - // Last: a written clock page asserts in every stamp this process and every - // other one takes, the verdict's own path included, so the arms whose harm - // is local answer first. - let clock = snapshot(CLOCK_PAGE); - refused("the clock page", CLOCK_PAGE); - assert_eq!(clock, snapshot(CLOCK_PAGE)); - - let status = Command::new(SELF_PATH).arg(CHILD_ARG).status().expect("spawn the second reader"); - assert!(status.success(), "the second process could not read the clock: {status:?}"); - + // Last, and said by its bit alone: its harm reaches every stamping reader. + if refused("the clock page", CLOCK_PAGE, stats).is_err() || !second_reader() { + wrote |= CLOCK_BIT; + } + if wrote != 0 { + std::process::exit(wrote); + } println!("a syscall writes only where its caller could store"); } diff --git a/tests/toyos.rs b/tests/toyos.rs index dad457f07d..29098e3f30 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -454,6 +454,9 @@ const RUST_SKIP: &[&str] = &[ /// assert-carrying build there. #[allow(dead_code, reason = "`suite_split` reads it, in `toyos-checks` alone")] const DRIVEN_AND_SHARED: &[&str] = &[ + // Its shared run is the x86-64 verdict; `virt_readonly_copyout` builds it + // for AArch64 and runs it on that architecture's job case. + "abuse_readonly_copyout", // The lost-wake canary: its shared run is the count on the shipping // kernel with nothing staged, and `blocking_read_window` drives it again // with the watch's window held open. @@ -537,6 +540,7 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ ("virt_first_entry", Sched::Parallel, Tier::Local), ("virt_unmap_touch", Sched::Parallel, Tier::Local), ("virt_debug_refused", Sched::Parallel, Tier::Local), + ("virt_readonly_copyout", Sched::Parallel, Tier::Local), ]; /// What `screen_console_shell` types, and what it then looks for on its own. @@ -3562,20 +3566,30 @@ fn check_no_stale_cells(dump: &screen::Ppm, console: &str) -> Result<(), String> Ok(()) } +/// `tests/toyos-rust-tests`' binary that `tests/virtjobcase` runs as its job +/// `test_rs_abuse_readonly_copyout`. +const VIRT_COPYOUT: &str = "abuse_readonly_copyout"; + /// Boot `tests/virtjobcase` under the EL2 profile and judge its job `job`: /// it ends with exit 0, having said `said`. The kernel carries `SYS_DEBUG` /// for `debug_refused`, and every job runs in every boot of the case. fn virt_job(job: &str, said: &str) -> Result<(), String> { let config = compile::repo_root().join("tests/virtjobcase/system.toml"); let case = config.parent().expect("system.toml has a directory"); + let profile = qemu::Profile::VirtEl2; + static COPYOUT: std::sync::OnceLock> = std::sync::OnceLock::new(); + let copyout = COPYOUT.get_or_init(|| { + qemu::build_toyos_bin(profile.arch(), &compile::repo_root().join("tests/toyos-rust-tests"), VIRT_COPYOUT) + }); let mut qemu = QemuInstance::boot_with_options( case, &[], &[], BootOptions { - profile: qemu::Profile::VirtEl2, + profile, kernel_features: toyos_build::build::TEST_KERNEL, ready_marker: "control registers: SCTLR_EL1=", + extra_root_files: vec![(format!("bin/test_rs_{VIRT_COPYOUT}"), copyout.clone())], ..Default::default() }, ); @@ -5132,6 +5146,11 @@ fn run_screen_test( "debug_refused", "debug_refused: SYS_DEBUG's double fault and TLB acknowledgement delay were refused", ), + // `tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs`, which + // exits with one bit per arm that let a write through. + "virt_readonly_copyout" => { + virt_job(&format!("test_rs_{VIRT_COPYOUT}"), "a syscall writes only where its caller could store") + } "virt_irq_storm" => { // The CPU floods itself with SGIs until the timer has fired a // thousand times through the flood, then waits for every SGI it diff --git a/tests/virtjobcase/system.toml b/tests/virtjobcase/system.toml index f08e05c649..190d1b33ac 100644 --- a/tests/virtjobcase/system.toml +++ b/tests/virtjobcase/system.toml @@ -1,5 +1,7 @@ -# One CPU and the AArch64 guest's jobs, each a toybox applet whose line comes -# only if the kernel kept what that applet asks about. +# One CPU and the AArch64 guest's jobs, each a toybox applet or, last because +# a kernel it fails on rewrites the clock page every reader asserts on, the +# suite's `test_rs_abuse_readonly_copyout`, which the harness puts on ROOT; +# each job's line comes only if the kernel kept what that job asks about. [boot] start = ["logd", "test-runner"] @@ -11,7 +13,7 @@ syscap = ["logread"] receives = ["power"] # Four minutes from boot rather than one: every job here runs on an emulated # CPU, and a job past the bound reboots the guest with the job named. -args = ["--bound-ms=240000", "preempt", "fp_isolation", "first_entry", "unmap_touch", "debug_refused"] +args = ["--bound-ms=240000", "preempt", "fp_isolation", "first_entry", "unmap_touch", "debug_refused", "test_rs_abuse_readonly_copyout"] [programs.toybox] From 6475015417a33e7081d768d21bcb56dc6ebb5f1f Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 16:00:54 +0200 Subject: [PATCH 10/11] PR #589 round 8: virt_job names a dying job's code; one test-build setup virt_job also ends its wait on the kernel's exit: record of a non-zero code for the job, so a job that dies before its TEST_END (W1 under logd's death) is judged by the code, which names the arms, not by a stall. build_toyos_bin and build_toyos_bins share their setup in TestBuild. The comment restating the exit protocol at virt_readonly_copyout is deleted. The &mut over pages nothing may write is filed. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01U6SVYFkdvV2t38KzNrESxs --- ...-forms-mut-over-pages-nothing-may-write.md | 24 ++++++++ src/build.rs | 58 +++++++++++-------- tests/toyos.rs | 16 ++--- 3 files changed, 67 insertions(+), 31 deletions(-) create mode 100644 issues/design-debt/the-readonly-copyout-test-forms-mut-over-pages-nothing-may-write.md diff --git a/issues/design-debt/the-readonly-copyout-test-forms-mut-over-pages-nothing-may-write.md b/issues/design-debt/the-readonly-copyout-test-forms-mut-over-pages-nothing-may-write.md new file mode 100644 index 0000000000..312fa64600 --- /dev/null +++ b/issues/design-debt/the-readonly-copyout-test-forms-mut-over-pages-nothing-may-write.md @@ -0,0 +1,24 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# The read-only copy-out test forms `&mut` over pages nothing may write + +`tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs` names its target to +`syscall::read` and `syscall::process_stats` through `target()` and a cast +`&mut *(addr as *mut ProcessStats)`, because `toyos_abi::syscall`'s typed +wrappers take `&mut [u8]` and `&mut ProcessStats` and no wrapper takes a raw +address. The `&mut` covers a read-only anonymous map, the binary's own `.text` +and the clock page, and is never written through, so the test rests on the +compiler inventing no store through it, not on a language guarantee. +`tls_dtv_race` forms the same reference. + +Owner: `toyos-abi`'s syscall wrappers, which would need a raw-address entry +for a caller that names memory it may not hold a `&mut` to; that is an ABI +change and is not this test's to make. + +**Exit condition**: the test issues its calls through a wrapper that takes an +address, and `rg 'from_raw_parts_mut|&mut \*\(' tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs` +finds nothing. diff --git a/src/build.rs b/src/build.rs index a5bf170c47..d78ce57a10 100644 --- a/src/build.rs +++ b/src/build.rs @@ -2188,11 +2188,6 @@ fn host_judge(root: &Path, (dir, bin): Judge) -> PathBuf { /// nothing in the tree can produce any more — into the ROOT image, into the test list, /// and over the name of whatever gets it next. pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) -> Vec<(String, Vec)> { - let target = arch.userland(); - let mut lock = buildlock::shared(root, "test binaries"); - let sysroot = crate::toolchain::ensure(root, false, &mut lock); - let env = GuestEnv::new(&sysroot); - let mut targets = vec![(crate_path.to_path_buf(), Clean::All)]; for entry in fs::read_dir(crate_path).into_iter().flatten().flatten() { let sub_path = entry.path(); @@ -2200,17 +2195,11 @@ pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) targets.push((sub_path, Clean::All)); } } - invalidate_stale(root, &mut lock, &env.toolchain, &targets); + let build = TestBuild::begin(root, arch, "test binaries", &targets); - let mut results = Vec::new(); + let (target, env) = (build.target, &build.env); - // Every build→read pair below is under one hold, for the reason the - // "Artifact staging" section above gives: cargo keys an artifact path on - // (crate, target, profile), so a second `cargo test` in this tree writes the - // very `.so` and test binaries this one reads back. Between the `read_dir` - // and the `read` that was enough to kill a run outright — four concurrent - // suites, one dead on `Result::unwrap()` on a `NotFound` naming no file. - let _artifact = buildlock::artifact(root); + let mut results = Vec::new(); // Build cdylib subcrates first let mut lib_search_dirs = Vec::new(); @@ -2233,7 +2222,7 @@ pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) if !quiet { eprintln!("[build] Building cdylib subcrate: {lib_name}"); } - cargo_build(&sub_path, target, &[], &env, &[], quiet); + cargo_build(&sub_path, target, &[], env, &[], quiet); let lib_out = sub_path.join(format!("target/{target}/{PROFILE}")); lib_search_dirs.push(lib_out.clone()); @@ -2260,7 +2249,7 @@ pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) } else { vec![("RUSTFLAGS", link_flags.trim_end())] }; - cargo_build(crate_path, target, &["--bins"], &env, &extra_env, quiet); + cargo_build(crate_path, target, &["--bins"], env, &extra_env, quiet); let bin_dir = crate_path.join(format!("target/{target}/{PROFILE}")); let bin_src = crate_path.join("src/bin"); @@ -2286,17 +2275,38 @@ pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) results } +/// What building test binaries starts from, held until the read of what was built is done. +/// +/// Every build→read pair is under one hold, for the reason the "Artifact +/// staging" section above gives: cargo keys an artifact path on (crate, target, +/// profile), so a second `cargo test` in this tree writes the very `.so` and +/// test binaries this one reads back. Between the `read_dir` and the `read` +/// that was enough to kill a run outright — four concurrent suites, one dead on +/// `Result::unwrap()` on a `NotFound` naming no file. +struct TestBuild { + target: &'static str, + env: GuestEnv, + _lock: buildlock::Held, + _artifact: buildlock::Guard, +} + +impl TestBuild { + fn begin(root: &Path, arch: Arch, what: &str, stale_targets: &[(PathBuf, Clean)]) -> Self { + let mut lock = buildlock::shared(root, what); + let sysroot = crate::toolchain::ensure(root, false, &mut lock); + let env = GuestEnv::new(&sysroot); + invalidate_stale(root, &mut lock, &env.toolchain, stale_targets); + let artifact = buildlock::artifact(root); + TestBuild { target: arch.userland(), env, _lock: lock, _artifact: artifact } + } +} + /// The one binary `name` of the crate at `crate_path`, built for `arch`: for an /// architecture the crate's other binaries do not all build for. pub fn build_toyos_bin(root: &Path, arch: Arch, crate_path: &Path, name: &str, quiet: bool) -> Vec { - let target = arch.userland(); - let mut lock = buildlock::shared(root, "a test binary"); - let sysroot = crate::toolchain::ensure(root, false, &mut lock); - let env = GuestEnv::new(&sysroot); - invalidate_stale(root, &mut lock, &env.toolchain, &[(crate_path.to_path_buf(), Clean::All)]); - // Built and read under one hold, as `build_toyos_bins`' are. - let _artifact = buildlock::artifact(root); - cargo_build(crate_path, target, &["--bin", name], &env, &[], quiet); + let build = TestBuild::begin(root, arch, "a test binary", &[(crate_path.to_path_buf(), Clean::All)]); + let (target, env) = (build.target, &build.env); + cargo_build(crate_path, target, &["--bin", name], env, &[], quiet); let binary = crate_path.join(format!("target/{target}/{PROFILE}/{name}")); fs::read(&binary).unwrap_or_else(|e| panic!("read the test binary {}: {e}", binary.display())) } diff --git a/tests/toyos.rs b/tests/toyos.rs index feea8d25bc..a6eec01632 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -3595,15 +3595,19 @@ fn virt_job(job: &str, said: &str) -> Result<(), String> { ); let end = format!("===TEST_END {job} "); let mut rest = String::new(); - let waited = await_marker(&mut qemu, &mut rest, &end, &format!("the job {job} to end")); + // A job that dies never reaches its `TEST_END`, so the kernel's own record of a non-zero exit ends the wait too: the verdict then names the code. + let died = format!("exit: {job} "); + let died_nonzero = |l: &str| l.contains(&died) && !l.contains(" code=0 "); + let waited = await_guest(&mut qemu, &mut rest, &format!("the job {job} to end"), |log| { + log.contains(&end) || log.lines().any(died_nonzero) + }); let serial = format!("{}\n{rest}", qemu.boot_log()); if let Err(why) = waited { return Err(format!("{why}\nserial:\n{serial}")); } - let ended = serial - .lines() - .find(|l| l.contains(&end)) - .expect("await_marker answered Ok, so the marker is in what it drained"); + let Some(ended) = serial.lines().find(|l| l.contains(&end) || died_nonzero(l)) else { + unreachable!("await_guest answered Ok, so a line ends the job"); + }; let Some(line) = serial.lines().find(|l| l.contains(said)) else { return Err(format!("{said:?} not on the PL011 ({ended})\nserial:\n{serial}")); }; @@ -5146,8 +5150,6 @@ fn run_screen_test( "debug_refused", "debug_refused: SYS_DEBUG's double fault and TLB acknowledgement delay were refused", ), - // `tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs`, which - // exits with one bit per arm that let a write through. "virt_readonly_copyout" => { virt_job(&format!("test_rs_{VIRT_COPYOUT}"), "a syscall writes only where its caller could store") } From 20ded4dbbdc073a4cbc594e883ea2b39dfc3827d Mon Sep 17 00:00:00 2001 From: japabu Date: Tue, 29 Sep 2026 16:15:22 +0200 Subject: [PATCH 11/11] PR #589 round 8: revert virt_job's early exit on a non-zero job code virt_unmap_touch's job is killed with code=-1 by design and is judged by its line after that, so ending the wait on any non-zero exit record failed it. The optional NOTE 5 change is reverted whole. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01U6SVYFkdvV2t38KzNrESxs --- tests/toyos.rs | 14 +++++--------- 1 file changed, 5 insertions(+), 9 deletions(-) diff --git a/tests/toyos.rs b/tests/toyos.rs index a6eec01632..cbb7b0f466 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -3595,19 +3595,15 @@ fn virt_job(job: &str, said: &str) -> Result<(), String> { ); let end = format!("===TEST_END {job} "); let mut rest = String::new(); - // A job that dies never reaches its `TEST_END`, so the kernel's own record of a non-zero exit ends the wait too: the verdict then names the code. - let died = format!("exit: {job} "); - let died_nonzero = |l: &str| l.contains(&died) && !l.contains(" code=0 "); - let waited = await_guest(&mut qemu, &mut rest, &format!("the job {job} to end"), |log| { - log.contains(&end) || log.lines().any(died_nonzero) - }); + let waited = await_marker(&mut qemu, &mut rest, &end, &format!("the job {job} to end")); let serial = format!("{}\n{rest}", qemu.boot_log()); if let Err(why) = waited { return Err(format!("{why}\nserial:\n{serial}")); } - let Some(ended) = serial.lines().find(|l| l.contains(&end) || died_nonzero(l)) else { - unreachable!("await_guest answered Ok, so a line ends the job"); - }; + let ended = serial + .lines() + .find(|l| l.contains(&end)) + .expect("await_marker answered Ok, so the marker is in what it drained"); let Some(line) = serial.lines().find(|l| l.contains(said)) else { return Err(format!("{said:?} not on the PL011 ({ended})\nserial:\n{serial}")); };