diff --git a/Cargo.lock b/Cargo.lock index 1e18b1d023..60171beec4 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1055,6 +1055,10 @@ dependencies = [ name = "toyos-fat32-check" version = "0.1.0" +[[package]] +name = "toyos-gicv3" +version = "0.1.0" + [[package]] name = "toyos-gpt" version = "0.1.0" diff --git a/Cargo.toml b/Cargo.toml index 661c2ad735..70a2ae4db9 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -30,6 +30,7 @@ members = [ "toyos-elide", "toyos-fat32", "toyos-fat32-check", + "toyos-gicv3", "toyos-gpt", "toyos-hda", "toyos-i219", diff --git a/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md b/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md index 6035bb6f8a..e0fa421979 100644 --- a/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md +++ b/issues/build/assembly-outside-an-arch-module-in-userland-and-guest-probes.md @@ -18,7 +18,7 @@ pointing here: `_mm_sfence` after writing a write-combining framebuffer. Userland has no portable way to say "drain my stores to the scanout"; the SDK (`toyos/src`) owes one, and it is also only changed under an ABI brief. -- Twenty guest probes in `tests/toyos-rust-tests/src/bin/`, whose subject +- Guest probes in `tests/toyos-rust-tests/src/bin/`, whose subject is an x86 instruction (`rdgsbase`, `fxsave64`, `int1`, x87 control words) or the raw `syscall` gate with arguments no SDK call will pass. They run in the x86-64 suite, which is the only suite until the harness gains its arch diff --git a/issues/design-debt/the-readonly-copyout-test-forms-mut-over-pages-nothing-may-write.md b/issues/design-debt/the-readonly-copyout-test-forms-mut-over-pages-nothing-may-write.md new file mode 100644 index 0000000000..312fa64600 --- /dev/null +++ b/issues/design-debt/the-readonly-copyout-test-forms-mut-over-pages-nothing-may-write.md @@ -0,0 +1,24 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# The read-only copy-out test forms `&mut` over pages nothing may write + +`tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs` names its target to +`syscall::read` and `syscall::process_stats` through `target()` and a cast +`&mut *(addr as *mut ProcessStats)`, because `toyos_abi::syscall`'s typed +wrappers take `&mut [u8]` and `&mut ProcessStats` and no wrapper takes a raw +address. The `&mut` covers a read-only anonymous map, the binary's own `.text` +and the clock page, and is never written through, so the test rests on the +compiler inventing no store through it, not on a language guarantee. +`tls_dtv_race` forms the same reference. + +Owner: `toyos-abi`'s syscall wrappers, which would need a raw-address entry +for a caller that names memory it may not hold a `&mut` to; that is an ABI +change and is not this test's to make. + +**Exit condition**: the test issues its calls through a wrapper that takes an +address, and `rg 'from_raw_parts_mut|&mut \*\(' tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs` +finds nothing. diff --git a/issues/design-debt/the-x86-toybox-ships-two-applets-that-only-panic.md b/issues/design-debt/the-x86-toybox-ships-two-applets-that-only-panic.md new file mode 100644 index 0000000000..7dae8903a8 --- /dev/null +++ b/issues/design-debt/the-x86-toybox-ships-two-applets-that-only-panic.md @@ -0,0 +1,18 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# The x86-64 toybox ships two applets that only panic + +`userland/toybox/src/main.rs` puts `fp_isolation` and `first_entry` in one +`commands!` list for every architecture, so the x86-64 toybox that ships in +every image answers both names, and `userland/toybox/src/arch/x86_64.rs`'s +body for each is a `panic!` naming where x86-64's probe lives or is owed +(`test_rs_fpu_isolation`, +`issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md`). +A shipping binary carries two commands whose only behaviour is to die. + +**Exit condition**: the x86-64 toybox answers neither name with a panic — +each is left out of its `commands!`, or runs a probe of x86-64's own. diff --git a/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md b/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md new file mode 100644 index 0000000000..03fe5f90cc --- /dev/null +++ b/issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md @@ -0,0 +1,19 @@ +--- +status: open +kind: defect +opened: 2026-09-28 +--- + +# A new x86-64 thread enters Ring 3 holding the kernel's register values + +`kernel/src/arch/x86_64/entry.rs`'s `process_start` and `thread_start` call +`sched::driver::trampoline_entry`, restore only `r12`–`r14` and the FP state, +and `iretq`: every other general register — `rax`–`rdx`, `rsi`, `rbp`, +`r8`–`r11`, `r15`, and `rdi` in a process's case — reaches the thread's first +instruction holding whatever the kernel left there, kernel stack and heap +addresses among them. A thread reads the kernel's layout off its own +registers. + +**Exit condition**: a fresh thread's first instruction sees zero in every +general register but its stack pointer and its argument, and a guest test that +reads them at `_start` says so. diff --git a/issues/isolation/aarch64-el1-runs-without-pan.md b/issues/isolation/aarch64-el1-runs-without-pan.md new file mode 100644 index 0000000000..8724eb0954 --- /dev/null +++ b/issues/isolation/aarch64-el1-runs-without-pan.md @@ -0,0 +1,23 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# AArch64's EL1 runs without PAN + +`kernel/src/arch/aarch64/control_regs.rs`'s `SCTLR` leaves `SPAN` set, so an +exception taken to EL1 leaves `PSTATE.PAN` as it was, and nothing writes +`PSTATE.PAN`: EL1 can load and store through any EL0 mapping. x86-64 runs +with SMAP (`kernel/src/arch/x86_64/control_regs.rs`), so a kernel bug that +dereferences a user pointer directly faults there and not here. + +The kernel reaches user memory through the direct map (`kernel/src/user_ptr.rs`) +and never through a user address, so PAN costs no +unprivileged-access instruction; what it needs is FEAT_PAN in the +declaration's check and `SPAN` clear. + +**Exit condition**: `SCTLR_EL1.SPAN` is clear and `PSTATE.PAN` set at EL1 on +every CPU, `control_regs::check` refuses a CPU without FEAT_PAN, and a guest +test whose kernel reads a user address directly under `test-actuators` +takes a permission fault. diff --git a/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md b/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md new file mode 100644 index 0000000000..d727b3fb36 --- /dev/null +++ b/issues/kernel/an-aarch64-crash-report-reads-through-any-user-leaf.md @@ -0,0 +1,42 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# An AArch64 crash report reads through any user leaf + +`kernel/src/arch/aarch64/paging.rs`'s `read_user_word`, which the report +of an EL0 fault calls with the faulting thread's own `x29` to walk its +frames (`trap.rs`'s `user_backtrace`), reads the frame any valid user leaf +names through the direct map. The direct map holds memory and nothing +else, so a leaf naming a device's registers — a claimed function's BAR, +once the port's stage 6 maps one into a process — names an address the +direct map does not hold, and the report's read of it is an EL1 data abort: +a user fault with `x29` pointed into its own BAR ends the machine. + +Two more readers take the same leaf to the direct map: + +- **Every syscall's user copy.** `AddressSpace::leaf` answers a Device + leaf as it answers a Normal one, for a `Write` too where the leaf is + EL0 read-write, and `kernel/src/user_ptr.rs` copies through + `DirectMap::from_phys` of what it answered: `translate_now`, `object_run` + (`copy_in`, `copy_out`) and `View::pieces` (`UserBytes::read_at`, + `UserBytesMut::write_at` and their kin). A `read(fd, bar, n)` is an EL1 + abort on an address the direct map does not hold, or, where `map_mmio` + holds it Device-nGnRE, an alignment fault on `memcpy`'s unaligned access: + a kernel panic a process chose. +- **The fault dump.** `kernel/src/process.rs`'s `dump_crash_diagnostics`, + which AArch64's `trap.rs` calls on an EL0 fault, reads the words around + the fault address and the faulting PC through `translate` and the direct + map, the same way. + +Nothing maps a BAR into an AArch64 process yet — `arch::msi_message` +refuses there, so pcidev refuses every hand-over — so this is latent until +stage 6 of `issues/kernel/toyos-runs-on-arm64.md`. + +**Exit condition**: `read_user_word`, `leaf` and the fault dump's +`translate` answer only a leaf of the memory type the direct map holds, and +guest tests see a process end and the kernel live when it faults with `x29` +inside a mapped BAR, faults at an address inside one, and names one as a +`read`'s buffer — the last refused with `BadAddress`. diff --git a/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md b/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md index 6c9e88011c..256298fad6 100644 --- a/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md +++ b/issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md @@ -13,8 +13,8 @@ vector delivered to an xAPIC id. A GICv3 message names an LPI through the ITS (a 32-bit event id, translated to an INTID above 8191) and a redistributor, and neither fits a `u8`. -Owned by stage 4 of `issues/kernel/toyos-runs-on-arm64.md`, which brings up -the GIC and its ITS. +Owned by stage 6 of `issues/kernel/toyos-runs-on-arm64.md`, which brings up +the GICv3 ITS for the claimed functions its SMMUv3 translates. **Exit condition**: a PCI function's interrupt is programmed from an arch-provided message (address and data, as `arch::msi` already provides the diff --git a/issues/kernel/portable-kernel-code-names-the-tsc.md b/issues/kernel/portable-kernel-code-names-the-tsc.md new file mode 100644 index 0000000000..d2c1ea18d5 --- /dev/null +++ b/issues/kernel/portable-kernel-code-names-the-tsc.md @@ -0,0 +1,19 @@ +--- +status: open +kind: defect +opened: 2026-09-29 +--- + +# Portable kernel code names the TSC + +On AArch64 the CPU's counter is the generic timer's `CNTVCT_EL0`, and the +portable kernel still calls it the TSC: `kernel/src/clock.rs`'s +`tsc_deadline`, `TSC_BOOT` and `TSC_PERIOD_FS`, `kernel/src/deadline.rs`'s +`AT_TSC`, and the xHCI driver's, `hardlockup`'s and `panic_reboot`'s waits +and comments read it by that name. `clock::counter_ticks`, which the AArch64 +timer reads, is renamed; the rest reads as x86-64's on both machines. +`issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md` is the ABI's +half of the same name. + +**Exit condition**: no item or comment outside `kernel/src/arch/x86_64/` +names the TSC for the counter `crate::arch::cpu::counter` reads. diff --git a/issues/kernel/the-saved-kernel-context-names-x86-registers.md b/issues/kernel/the-saved-kernel-context-names-x86-registers.md deleted file mode 100644 index 7984c7af40..0000000000 --- a/issues/kernel/the-saved-kernel-context-names-x86-registers.md +++ /dev/null @@ -1,20 +0,0 @@ ---- -status: open -kind: defect -opened: 2026-09-26 ---- - -# The saved kernel context names x86 registers in generic code - -`kernel/src/sched/payload.rs`'s `KernelCtx`, which every CPU's switch loads, -carries `rsp` and `fs_base`: the stack pointer and the user thread pointer by -their x86 names, in code outside `arch/`. AArch64 saves `sp` and -`TPIDR_EL0` in the same two roles, so the port either writes AArch64 state -into fields named for x86 or grows a second copy of the struct. - -Owned by stage 4 of `issues/kernel/toyos-runs-on-arm64.md`, which writes the -AArch64 context switch. - -**Exit condition**: the fields are named by role (the saved kernel stack -pointer, the user thread pointer), the x86 and AArch64 switches both load -them, and nothing outside `kernel/src/arch/` names `rsp` or `fs_base`. diff --git a/issues/kernel/the-x86-address-space-keeps-a-page-map-nothing-fills.md b/issues/kernel/the-x86-address-space-keeps-a-page-map-nothing-fills.md new file mode 100644 index 0000000000..b12ea86615 --- /dev/null +++ b/issues/kernel/the-x86-address-space-keeps-a-page-map-nothing-fills.md @@ -0,0 +1,16 @@ +--- +status: open +kind: defect +opened: 2026-09-28 +--- + +# The x86-64 address space keeps a page map nothing fills + +`kernel/src/arch/x86_64/paging.rs`'s `AddressSpace` carries +`pages: HashMap`, documented as the user pages it frees on +drop, and `unmap` removes from it; nothing anywhere inserts into it, so it is +always empty and the removal does nothing. Dead code the compiler cannot see, +because a field that is read is not dead to it. + +**Exit condition**: the field and its removal are gone, or what it claims to +own is put in it by the path that maps the page. diff --git a/issues/kernel/toyos-runs-on-arm64.md b/issues/kernel/toyos-runs-on-arm64.md index 6a44c5f117..6ca1f8f319 100644 --- a/issues/kernel/toyos-runs-on-arm64.md +++ b/issues/kernel/toyos-runs-on-arm64.md @@ -205,8 +205,7 @@ before any aarch64 file exists, with x86 as its only user: Each is its own issue, owned by the stage that removes it: -- `issues/kernel/the-saved-kernel-context-names-x86-registers.md` (stage 4) -- `issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md` (stage 4) +- `issues/kernel/msi-and-pin-routing-take-an-x86-vector-and-apic-id.md` (stage 6) - `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md` (stage 4) - `issues/kernel/the-crash-evidence-records-x86-fault-registers.md` (stage 5) - `issues/kernel/the-aarch64-kernel-builds-with-dead-code-allowed.md` (stage 7) @@ -259,6 +258,33 @@ Each stage names its exit; "measured" means a number from a run. one stays green in `virt_el2_drop`, because stage 3 reads no counter and runs no FP. This stage's timer and FP tests run under that EL2 profile too, and each of the three deletions is shown red. + **Built on one CPU, ahead of small-kernel stage 6 by the owner's word:** + the kernel's own tables (`TTBR1_EL1` holding memory and nothing else, each + user space on `TTBR0_EL1` under a 16-bit ASID from `toyos-pcid`), the + GICv3's SGIs and the virtual timer's PPI, the EL0 entry, and the context + switch carrying FP/SIMD; the `virt_` tests other than `virt_early_panic`, + `virt_early_fault` and `virt_el2_drop` judge it under the EL2 profile, emulated, because HVF + exposes no RNDR and the kernel's hash seed refuses there until stage 6's + virtio-rng. Each judges an event, never a rate: no QEMU test measures time. + Owed before the exit holds: the interrupts-off window against x86's, a + measurement only metal can make, with no instrument on either arch yet; + `issues/kernel/the-boot-timing-handoff-is-named-for-the-tsc.md`; the + instruction-cache maintenance before a mapping is executable + (`cache::make_executable`), the break-before-make ordering of a live + entry's replacement, and the TLB flush before a reclaimed ASID is issued + again, which QEMU's TCG, the only oracle this stage has, cannot fail on: + the first HVF run, once stage 6 gives HVF its RNDR, is their exit; and the + three deletions shown red. They are shown red on a machine whose + firmware leaves the registers otherwise, or by a loader that writes the + opposite values before the handoff. The ITS moves to stage 6: a claimed + function is its only consumer the small-kernel track leaves, and it needs that + stage's SMMUv3 first. Stubbed on AArch64, each owned by the small-kernel + track, which moves the driver out of the kernel: + - `arch::msi_message` refuses, so the kernel's xHCI (`virt`'s boot stick), + NVMe, HDA, virtio-sound, virtio-console and virtio-gpu drivers each + refuse their function by name. + - `drivers::gop` refuses a scanout that is not whole 2 MiB pages of its + own, which a `ramfb` scanout carved out of RAM need not be. 5. **SMP through PSCI.** `CPU_ON` from MADT GICC entries, SGIs as the IPI, broadcast TLBI behind the machine-wide invalidation contract. **Exit**: diff --git a/kernel/Cargo.lock b/kernel/Cargo.lock index be62dcc5ee..90c5ca15e1 100644 --- a/kernel/Cargo.lock +++ b/kernel/Cargo.lock @@ -46,6 +46,7 @@ dependencies = [ "toyos-elf", "toyos-elide", "toyos-fat32", + "toyos-gicv3", "toyos-gpt", "toyos-hda", "toyos-pci", @@ -121,6 +122,10 @@ dependencies = [ "toyos-wallclock", ] +[[package]] +name = "toyos-gicv3" +version = "0.1.0" + [[package]] name = "toyos-gpt" version = "0.1.0" diff --git a/kernel/Cargo.toml b/kernel/Cargo.toml index f8d2ea7f32..81e0a9b315 100644 --- a/kernel/Cargo.toml +++ b/kernel/Cargo.toml @@ -400,6 +400,7 @@ toyos-blockhold = { path = "../toyos-blockhold" } toyos-bootmap = { path = "../toyos-bootmap" } toyos-fat32 = { path = "../toyos-fat32" } toyos-elf = { path = "../toyos-elf" } +toyos-gicv3 = { path = "../toyos-gicv3" } toyos-gpt = { path = "../toyos-gpt" } toyos-hda = { path = "../toyos-hda" } toyos-pci = { path = "../toyos-pci" } diff --git a/kernel/src/actuator.rs b/kernel/src/actuator.rs index 8cb364027c..7e65a02bbf 100644 --- a/kernel/src/actuator.rs +++ b/kernel/src/actuator.rs @@ -304,7 +304,7 @@ actuators! { /// Storm the CPU spinning on `syscall` from Ring 3 with NMIs. syscall_window_nmi = "syscall-window-nmi"; - /// Take the IST index off vector 2's gate — the negative control on the row above: the CPU builds the NMI frame at whatever `rsp` holds and takes a `#DF`. + /// Take the IST index off vector 2's gate — the negative control on the row above: the CPU builds the NMI frame at whatever the stack pointer holds and takes a `#DF`. nmi_without_ist = "nmi-without-ist"; /// Return from the NMI handler via `iretq` with a second NMI already pending. @@ -346,6 +346,14 @@ actuators! { /// Raise a vector no `idt_vectors!` row claims on this CPU once. unclaimed_vector_selftest = "unclaimed-vector-selftest"; + /// Tick the timer at a fixed period while this CPU floods itself with + /// interrupts. + irq_storm = "irq-storm"; + + /// Make this CPU's timer due with interrupts masked, ask it to fire within + /// a quantum, and take its interrupts with them open. + timer_floor = "timer-floor"; + /// Hold a flush of `truncate-race.bin` inside its metadata window and say whether a truncate got in. ftruncate_flush_stall = "ftruncate-flush-stall"; diff --git a/kernel/src/arch/aarch64/boot.rs b/kernel/src/arch/aarch64/boot.rs index 4e974a930f..cf6dd35e31 100644 --- a/kernel/src/arch/aarch64/boot.rs +++ b/kernel/src/arch/aarch64/boot.rs @@ -61,7 +61,10 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { "msr tcr_el1, x2", "msr ttbr0_el1, x3", "msr ttbr1_el1, x3", - "msr cpacr_el1, xzr", + "ldr x1, ={cpacr}", + "msr cpacr_el1, x1", + "mov x1, #{cntkctl}", + "msr cntkctl_el1, x1", "tlbi vmalle1", "dsb nsh", "isb", @@ -79,11 +82,26 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { "mrs x6, hcr_el2", "cmp x6, x1", "b.ne {refuse_hcr}", + // `ICC_SRE_EL2`, whose `Enable` lets EL1 write its own `ICC_SRE_EL1` + // (`super::irqchip::init`). Skipped where `ID_AA64PFR0_EL1.GIC` names + // no system-register interface: there the write is an undefined + // instruction under firmware's vectors and ends the boot silently, + // while EL1's own access fails under this kernel's, which report it. + "mrs x1, id_aa64pfr0_el1", + "ubfx x1, x1, #24, #4", + "cbz x1, 4f", + "mov x1, #{icc_sre_el2}", + "msr S3_4_C12_C9_5, x1", + "isb", + "4:", "msr mair_el1, x4", "msr tcr_el1, x2", "msr ttbr0_el1, x3", "msr ttbr1_el1, x3", - "msr cpacr_el1, xzr", + "ldr x1, ={cpacr}", + "msr cpacr_el1, x1", + "mov x1, #{cntkctl}", + "msr cntkctl_el1, x1", "msr sctlr_el1, x5", "mov x1, #{cnthctl}", "msr cnthctl_el2, x1", @@ -128,6 +146,9 @@ pub unsafe extern "C" fn _start(_kernel_args: &KernelArgs) -> ! { refuse_hcr = sym refused_hcr_el2_readback, cnthctl = const regs::CNTHCTL_EL2, cptr = const regs::CPTR_EL2, + cpacr = const regs::CPACR, + cntkctl = const regs::CNTKCTL, + icc_sre_el2 = const regs::ICC_SRE_EL2, spsr = const regs::SPSR_EL2_TO_EL1, phys_offset = const crate::PHYS_OFFSET, entry_el = sym regs::ENTRY_EL, @@ -238,41 +259,59 @@ pub fn reserved() -> Region { Region { start: 0, end: 0 } } -/// What the boot learns bringing interrupts up and hands later steps. -pub struct Platform { - never: core::convert::Infallible, -} +/// What the boot learns bringing interrupts up and hands later steps: nothing +/// yet, since the other CPUs the MADT names are the port's stage 5's to read. +pub struct Platform; -/// Interrupt delivery, this CPU's per-CPU block and the syscall gate. -pub fn interrupts(_rsdp_addr: u64) -> Platform { - owed!("interrupt delivery", "stage 4") +/// Interrupt delivery: this CPU's per-CPU block, the GIC and the timer's +/// interrupt, and interrupts unmasked. The syscall gate is the vectors' own. +pub fn interrupts(rsdp_addr: u64) -> Platform { + super::percpu::init_bsp(); + super::irqchip::init(rsdp_addr); + super::cpu::enable_interrupts(); + Platform } -/// The clock: the generic timer's counter at `CNTFRQ_EL0`. +/// The clock: the generic timer's count, at the rate firmware states in +/// `CNTFRQ_EL0`, which the Arm ARM makes firmware's to program and which is +/// what the counter counts at. No wall clock: `super::rtc` says why. pub fn clock(_args: &KernelArgs) { - owed!("the clock", "stage 4") + let hz = super::cpu::stated_counter_hz().expect("clock: CNTFRQ_EL0 states no rate for the generic timer"); + crate::clock::set_counter(super::cpu::counter(), 1_000_000_000_000_000 / hz); + log!("clock: the generic timer counts at {hz} Hz; no wall clock is read on this architecture"); } /// Nothing: where the generic timer counts from is firmware's, and no /// register says it. pub fn report_counter_origin() {} -/// The per-CPU timer. +/// The per-CPU timer counts the clock's own ticks, so there is nothing to +/// calibrate: it stays stopped until the scheduler first arms it. pub fn timer() { - owed!("the timer", "stage 4") + log!("timer: the EL1 virtual timer, PPI {}, stopped until the scheduler arms it", super::irqchip::timer_intid()); } /// The platform's own devices that are not PCI functions: none this kernel /// drives on an ACPI Arm machine. pub fn platform_devices(_rsdp_addr: u64) {} -/// Every other CPU, running. -pub fn start_other_cpus(platform: &Platform, _args: &KernelArgs) { - match platform.never {} +/// Every other CPU, running: the port's stage 5, which starts each with PSCI +/// `CPU_ON`. Until then the boot CPU runs alone. +pub fn start_other_cpus(_platform: &Platform, _args: &KernelArgs) { + log!("smp: the boot CPU runs alone; the other CPUs are the port's stage 5 (PSCI CPU_ON)"); } /// The interrupt-controller selftests an actuator asks for. #[cfg(feature = "boot-actuators")] pub fn interrupt_selftests() { - owed!("the interrupt controller", "stage 4") + if crate::actuator::timer_floor() { + super::irqchip::floor_selftest(); + } + if crate::actuator::irq_storm() { + super::trap::storm::run(); + } + assert!( + !crate::actuator::lapic_spurious_selftest() && !crate::actuator::unclaimed_vector_selftest(), + "the local APIC's selftests are x86-64's, and this machine has a GIC" + ); } diff --git a/kernel/src/arch/aarch64/cache.rs b/kernel/src/arch/aarch64/cache.rs index 0b0e9df4d6..45503d3c8a 100644 --- a/kernel/src/arch/aarch64/cache.rs +++ b/kernel/src/arch/aarch64/cache.rs @@ -1,13 +1,18 @@ -//! Writing memory back out of the caches, for the one reader that is not a -//! CPU: DRAM across a reset. +//! Cache maintenance, for the two readers that are not this CPU's data side: +//! DRAM across a reset, and the instruction stream. -/// The smallest data cache line on this machine, from `CTR_EL0.DminLine` -/// (log2 of the word count), so a walk by it misses no line. -fn line() -> u64 { +/// `CTR_EL0`, which EL1 may always read. +fn ctr() -> u64 { let ctr: u64; // SAFETY: reads `CTR_EL0`, which EL1 may always read. unsafe { core::arch::asm!("mrs {}, ctr_el0", out(reg) ctr, options(nomem, nostack, preserves_flags)) }; - 4 << ((ctr >> 16) & 0xF) + ctr +} + +/// The smallest data cache line on this machine, from `CTR_EL0.DminLine` +/// (log2 of the word count), so a walk by it misses no line. +fn line() -> u64 { + 4 << ((ctr() >> 16) & 0xF) } /// Every line of `[at, at + len)` cleaned to the point of coherency and @@ -28,3 +33,31 @@ pub fn write_back(at: u64, len: usize) { // SAFETY: a barrier; waits for the maintenance above to complete. unsafe { core::arch::asm!("dsb sy", options(nostack, preserves_flags)) }; } + +/// Instructions this CPU wrote through the data side at `[at, at + len)` made +/// the ones every CPU fetches: the data cleaned to the point of unification +/// unless `CTR_EL0.IDC` says it need not be, then every instruction cache +/// invalidated unless `CTR_EL0.DIC` says the same (Arm ARM K.a, D7.5.9.2). +/// Before a mapping that executes them is written. +pub fn make_executable(at: u64, len: usize) { + let ctr = ctr(); + if ctr >> 28 & 1 == 0 { + let step = line(); + let mut addr = at & !(step - 1); + while addr < at + len as u64 { + // SAFETY: `DC CVAU` cleans the line holding a mapped address the + // caller owns; it changes no memory's contents. + unsafe { core::arch::asm!("dc cvau, {}", in(reg) addr, options(nostack, preserves_flags)) }; + addr += step; + } + // SAFETY: a barrier; waits for the cleaning above to complete. + unsafe { core::arch::asm!("dsb ish", options(nostack, preserves_flags)) }; + } + if ctr >> 29 & 1 == 0 { + // SAFETY: invalidating instruction caches changes no memory; the + // barrier waits for it on every CPU. + unsafe { core::arch::asm!("ic ialluis", "dsb ish", options(nostack, preserves_flags)) }; + } + // SAFETY: a context synchronization; touches nothing. + unsafe { core::arch::asm!("isb", options(nostack, preserves_flags)) }; +} diff --git a/kernel/src/arch/aarch64/console_uart.rs b/kernel/src/arch/aarch64/console_uart.rs index e316a04030..a0b235f076 100644 --- a/kernel/src/arch/aarch64/console_uart.rs +++ b/kernel/src/arch/aarch64/console_uart.rs @@ -27,6 +27,13 @@ const FRAME: u64 = 0x1000; /// The register frame's physical address; zero until [`init`] found one. static BASE: AtomicU64 = AtomicU64::new(0); +/// The register frame's physical address, once [`init`] found one: what the +/// kernel's own tables map before they replace the loader's. +pub fn frame() -> Option { + let base = BASE.load(Ordering::Relaxed); + (base != 0).then_some(base) +} + fn regs() -> Mmio { let base = BASE.load(Ordering::Relaxed); assert!(base != 0, "console UART: a byte moved before `init` found the UART"); diff --git a/kernel/src/arch/aarch64/control_regs.rs b/kernel/src/arch/aarch64/control_regs.rs index 9c82addbb2..6251ba1bcd 100644 --- a/kernel/src/arch/aarch64/control_regs.rs +++ b/kernel/src/arch/aarch64/control_regs.rs @@ -3,9 +3,11 @@ //! [`super::boot`]'s entry writes every register here whole, from these //! constants, before the MMU is on; [`check`] reads the EL1 ones back and //! refuses a CPU whose registers say anything else. Nothing else writes any -//! of them. +//! of them, bar [`ICC_SRE`]. //! -//! Field positions are Arm ARM K.a, chapter D24 (the register descriptions). +//! Field positions are Arm ARM K.a, chapter D24 (the register descriptions), +//! and the GIC architecture specification (IHI 0069H), chapter 12, for the +//! `ICC_` registers. use core::sync::atomic::{AtomicU64, Ordering}; @@ -19,10 +21,9 @@ pub use toyos_bootmap::aarch64::MAIR; /// stack alignment checked at EL1 and EL0 (`SA`, `SA0`), AArch32 EL0's `IT` /// and `SETEND` disabled (`ITD`, `SED` — RES1 on a CPU with no AArch32 EL0, /// and nothing this kernel runs is AArch32), over the bits Armv8.0 makes -/// RES1 (29, 28, 23, 22, 20, 11). `WXN` stays clear: the -/// loader's blocks are writable and executable both until the kernel owns -/// its tables. Little-endian at both levels; EL0's cache maintenance and -/// `WFI`/`WFE` trap. +/// RES1 (29, 28, 23, 22, 20, 11). `WXN` stays clear: the direct map the +/// kernel runs from is writable and executable both. Little-endian at both +/// levels; EL0's cache maintenance and `WFI`/`WFE` trap. pub const SCTLR: u64 = SCTLR_RES1 | 1 << 0 | 1 << 2 | 1 << 3 | 1 << 4 | 1 << 7 | 1 << 8 | 1 << 12; /// `SCTLR_EL1` with the MMU and caches off — [`SCTLR`] less `M`, `C`, `I`, @@ -34,30 +35,52 @@ const SCTLR_RES1: u64 = 1 << 29 | 1 << 28 | 1 << 23 | 1 << 22 | 1 << 20 | 1 << 1 /// `TCR_EL1` but for `IPS`: 48-bit regions from both tables (`T0SZ` = `T1SZ` = /// 16), 4 KiB granules (`TG0` = 0, `TG1` = 2), walks inner-shareable and -/// write-back write-allocate cacheable, 8-bit ASIDs. -pub const TCR: u64 = 16 | 1 << 8 | 1 << 10 | 3 << 12 | 16 << 16 | 1 << 24 | 1 << 26 | 3 << 28 | 2 << 30; +/// write-back write-allocate cacheable, `TTBR0_EL1` naming the ASID (`A1` +/// clear), and 16-bit ASIDs (`AS`), which [`check`] requires the CPU to have. +pub const TCR: u64 = + 16 | 1 << 8 | 1 << 10 | 3 << 12 | 16 << 16 | 1 << 24 | 1 << 26 | 3 << 28 | 2 << 30 | 1 << 36; /// `TCR_EL1.IPS`'s position: the output address size, taken from /// `ID_AA64MMFR0_EL1.PARange` (the same encoding), because a larger `IPS` /// than the CPU implements is a reserved value. pub const TCR_IPS_SHIFT: u64 = 32; -/// `CPACR_EL1`: `FPEN` = 0, so FP and SIMD trap at EL1 and EL0. The kernel is -/// built soft-float and uses neither; user mode's are stage 7's. -pub const CPACR: u64 = 0; +/// `CPACR_EL1`: `FPEN` = 0b11, so FP and SIMD trap at neither EL1 nor EL0. The +/// kernel is built soft-float and touches them only to save and restore a +/// thread's registers; `ZEN` and `SMEN` stay clear, so SVE and SME trap +/// everywhere. +pub const CPACR: u64 = 0b11 << 20; + +/// `CNTKCTL_EL1`: `EL0VCTEN` alone, so EL0 reads the virtual count and its +/// frequency — the clock page's counter (`toyos_abi::arch::counter`) — and +/// nothing else of the generic timer. +pub const CNTKCTL: u64 = 1 << 1; + +/// `ICC_SRE_EL1`: the GICv3 CPU interface through system registers (`SRE`), +/// with FIQ and IRQ bypass disabled (`DFB`, `DIB`). Written and read back by +/// `super::irqchip::init`, not the entry: on a CPU with no such interface the +/// access is an undefined instruction, which this kernel's vectors report and +/// firmware's, still installed at the entry, do not. +pub const ICC_SRE: u64 = 0b111; /// `HCR_EL2` when entered at EL2: `RW`, so EL1 is AArch64, and nothing else — -/// no stage-2 translation, no trap, `E2H` clear. +/// no stage-2 translation, no trap, `E2H` clear, and `IMO`/`FMO` clear, so a +/// physical interrupt is taken at EL1. pub const HCR_EL2: u64 = 1 << 31; -/// `CNTHCTL_EL2` when entered at EL2: `EL1PCTEN` and `EL1PCEN`, so EL1 reads the -/// physical counter and programs its timer without trapping. +/// `CNTHCTL_EL2` when entered at EL2: `EL1PCTEN` and `EL1PCEN`, and every +/// other field clear — among them FEAT_ECV's `EL1TVT` and `EL1TVCT`, which +/// reset UNKNOWN and would trap EL1's virtual timer and count to EL2. pub const CNTHCTL_EL2: u64 = 1 << 1 | 1 << 0; /// `CPTR_EL2` when entered at EL2 (`E2H` clear): its RES1 bits (13, 12, 9:0) /// and `TFP` clear, so FP is `CPACR_EL1`'s decision alone. pub const CPTR_EL2: u64 = 0x33FF; +/// `ICC_SRE_EL2` when entered at EL2: [`ICC_SRE`]'s three bits at EL2, and +/// `Enable`, without which EL1's own `ICC_SRE_EL1` traps to EL2. +pub const ICC_SRE_EL2: u64 = 0b1111; + /// `SPSR_EL2` for the drop: EL1 on `SP_EL1` (`M` = 0b0101), `D`, `A`, `I`, `F` masked. pub const SPSR_EL2_TO_EL1: u64 = 0x3C5; @@ -82,11 +105,16 @@ pub fn tcr() -> u64 { /// Read every EL1 register the declaration names back, and refuse a CPU that /// holds anything else; then say what it holds. pub fn check() { + // `ID_AA64MMFR0_EL1.ASIDBits` = 0b0010: without 16-bit ASIDs `TCR_EL1.AS` + // is RES0, and the read-back below would name the register, not the reason. + let asid_bits = read!("id_aa64mmfr0_el1") >> 4 & 0xF; + assert_eq!(asid_bits, 0b0010, "control registers: this CPU has 8-bit ASIDs, and the declaration's TCR_EL1.AS needs 16"); let declared = [ ("SCTLR_EL1", read!("sctlr_el1"), SCTLR), ("TCR_EL1", read!("tcr_el1"), tcr()), ("MAIR_EL1", read!("mair_el1"), MAIR), ("CPACR_EL1", read!("cpacr_el1"), CPACR), + ("CNTKCTL_EL1", read!("cntkctl_el1"), CNTKCTL), ]; for (name, live, value) in declared { assert_eq!(live, value, "control registers: {name} holds {live:#x}, and the declaration says {value:#x}"); @@ -96,10 +124,14 @@ pub fn check() { assert_eq!((el, spsel), (1, 1), "control registers: running at EL{el} on SP_EL{spsel}, not EL1 on SP_EL1"); let el = ENTRY_EL.load(Ordering::Relaxed); log!( - "control registers: SCTLR_EL1={SCTLR:#x} TCR_EL1={:#x} MAIR_EL1={:#x} CPACR_EL1={CPACR:#x}, \ - as declared; entered at EL{el}{}", + "control registers: SCTLR_EL1={SCTLR:#x} TCR_EL1={:#x} MAIR_EL1={:#x} CPACR_EL1={CPACR:#x} \ + CNTKCTL_EL1={CNTKCTL:#x}, as declared; entered at EL{el}{}", tcr(), MAIR, - if el == 2 { ", HCR_EL2 read back as declared, CNTHCTL_EL2/CPTR_EL2 written, and dropped to EL1" } else { "" }, + if el == 2 { + ", HCR_EL2 read back as declared, CNTHCTL_EL2/CNTVOFF_EL2/CPTR_EL2/ICC_SRE_EL2 written, and dropped to EL1" + } else { + "" + }, ); } diff --git a/kernel/src/arch/aarch64/cpu.rs b/kernel/src/arch/aarch64/cpu.rs index cfe942ddd0..31e5fceb94 100644 --- a/kernel/src/arch/aarch64/cpu.rs +++ b/kernel/src/arch/aarch64/cpu.rs @@ -57,6 +57,14 @@ pub fn thread_pointer() -> u64 { tp } +/// Install the thread pointer the next return to EL0 runs with. +/// # Safety +/// `tp` is the thread's own, which the kernel computed for it. +pub unsafe fn write_thread_pointer(tp: u64) { + // SAFETY: `TPIDR_EL0` is EL0's register; the kernel never reads through it. + unsafe { asm!("msr tpidr_el0, {}", in(reg) tp, options(nomem, nostack, preserves_flags)) }; +} + /// Unmask interrupts on this CPU: `DAIF.I` and `DAIF.F`. pub fn enable_interrupts() { // SAFETY: writes two `DAIF` bits; a compiler barrier, so no access moves across it. @@ -114,5 +122,5 @@ pub fn hardware_id() -> u32 { let mpidr: u64; // SAFETY: reads an ID register. unsafe { asm!("mrs {}, mpidr_el1", out(reg) mpidr, options(nomem, nostack, preserves_flags)) }; - ((mpidr & 0xFF_FFFF) | ((mpidr >> 8) & 0xFF00_0000)) as u32 + toyos_gicv3::packed_affinity(mpidr) } diff --git a/kernel/src/arch/aarch64/entry.rs b/kernel/src/arch/aarch64/entry.rs index 8207fc9d14..31a0cdf09b 100644 --- a/kernel/src/arch/aarch64/entry.rs +++ b/kernel/src/arch/aarch64/entry.rs @@ -1,27 +1,97 @@ -//! Where a new context starts: the frame `switch` restores and the -//! trampolines to user mode and to a kernel thread, the port's stages 4 and 7. +//! Where a new context starts: the frame [`super::switch::context_switch`] +//! restores it from, and the trampolines its return lands in — to EL0 for the +//! first time, or into a kernel thread's body. +//! +//! A trampoline to EL0 leaves nothing of the kernel in a register: every +//! general register but the argument is zeroed, and the FP/SIMD state is the +//! zero the initial frame carried. +use super::switch::{DAIF_AT, FRAME_BYTES, RETURN_AT}; -pub(crate) extern "C" fn process_start() { - owed!("user mode", "stage 7") +/// `DAIF` a new context starts with: every exception masked, as an entry from +/// EL0 is, for `trampoline_entry`'s contract. +const DAIF_MASKED: u64 = 0b1111 << 6; + +/// Zero x1–x30, the registers a first entry to EL0 must not carry. +macro_rules! zero_registers { + () => { + concat!( + "mov x1, xzr\n", "mov x2, xzr\n", "mov x3, xzr\n", "mov x4, xzr\n", "mov x5, xzr\n", + "mov x6, xzr\n", "mov x7, xzr\n", "mov x8, xzr\n", "mov x9, xzr\n", "mov x10, xzr\n", + "mov x11, xzr\n", "mov x12, xzr\n", "mov x13, xzr\n", "mov x14, xzr\n", "mov x15, xzr\n", + "mov x16, xzr\n", "mov x17, xzr\n", "mov x18, xzr\n", "mov x19, xzr\n", "mov x20, xzr\n", + "mov x21, xzr\n", "mov x22, xzr\n", "mov x23, xzr\n", "mov x24, xzr\n", "mov x25, xzr\n", + "mov x26, xzr\n", "mov x27, xzr\n", "mov x28, xzr\n", "mov x29, xzr\n", "mov x30, xzr\n", + ) + }; } -pub(crate) extern "C" fn thread_start() { - owed!("user mode", "stage 7") +/// A thread's first entry to EL0 at `x19` on the stack `x20`, with `x0` = +/// `x21`: a process's argument is zero, a thread's is its own. `SPSR_EL1` +/// zero is EL0 with every exception unmasked, and `SP_EL1` is left at the +/// top of this thread's kernel stack, which is where its entries land. +#[unsafe(naked)] +pub(crate) extern "C" fn process_start() { + core::arch::naked_asm!( + "bl {unlock}", + "msr elr_el1, x19", + "msr sp_el0, x20", + "msr spsr_el1, xzr", + "mov x0, x21", + zero_registers!(), + "eret", + unlock = sym crate::sched::driver::trampoline_entry, + ); } +/// A thread's first entry is a process's, with its argument in `x0`. +pub(crate) use process_start as thread_start; + +/// Entry point for a kernel thread: `x19` = body, `x21` = argument. Never +/// reaches EL0; unmasks interrupts once `trampoline_entry`, which requires +/// them masked, is done. +#[unsafe(naked)] pub(crate) extern "C" fn kernel_start() { - owed!("kernel threads", "stage 4") + core::arch::naked_asm!( + "bl {unlock}", + "msr daifclr, #3", + "mov x0, x21", + "blr x19", + "bl {returned}", + unlock = sym crate::sched::driver::trampoline_entry, + returned = sym kernel_thread_returned, + ); +} + +/// What [`kernel_start`] calls when a kernel thread's body returns: panics rather than halting silently. +extern "C" fn kernel_thread_returned() -> ! { + panic!("a kernel thread's body returned; nothing runs on this stack now"); } +/// Lay out, just below `top`, the frame `context_switch` restores a new +/// context from, and answer the stack pointer that names it: `trampoline` is +/// where its return lands, with the entry, the stack and the argument where +/// the trampolines read them (`x19`, `x20`, `x21`), and FP/SIMD state zero. /// # Safety -/// `top` is the end of a fresh kernel stack nothing else references. +/// `top` is the 16-byte-aligned end of a fresh kernel stack nothing else +/// references, at least [`FRAME_BYTES`] deep. pub unsafe fn initial_frame( - _top: u64, - _trampoline: unsafe extern "C" fn(), - _user_entry: u64, - _user_sp: u64, - _arg: u64, + top: u64, + trampoline: unsafe extern "C" fn(), + user_entry: u64, + user_sp: u64, + arg: u64, ) -> u64 { - owed!("kernel threads", "stage 4") + let frame = top - FRAME_BYTES as u64; + // SAFETY: `[frame, top)` is the top `FRAME_BYTES` of the stack the caller owns. + unsafe { + core::ptr::write_bytes(frame as *mut u8, 0, FRAME_BYTES); + let word = |at: usize| (frame + at as u64) as *mut u64; + *word(0) = user_entry; + *word(8) = user_sp; + *word(16) = arg; + *word(DAIF_AT) = DAIF_MASKED; + *word(RETURN_AT) = trampoline as usize as u64; + } + frame } diff --git a/kernel/src/arch/aarch64/fpu.rs b/kernel/src/arch/aarch64/fpu.rs index 1613f2f8d5..94a674d4ba 100644 --- a/kernel/src/arch/aarch64/fpu.rs +++ b/kernel/src/arch/aarch64/fpu.rs @@ -1,2 +1,9 @@ -//! User FP/SIMD state: `CPACR_EL1.FPEN` traps it at EL1 and EL0 until the -//! port's stage 7 saves and restores it. The kernel itself never uses it. +//! User FP/SIMD state: v0–v31, `FPCR` and `FPSR`. The kernel is built +//! soft-float and never touches them, so a thread's own stay in the registers +//! across every entry from EL0; they are saved and restored only where a +//! thread stops running, in [`super::switch`]'s frame, and a thread that has +//! never run starts from zero — `FPCR` zero is round-to-nearest with every +//! trap disabled (Arm ARM K.a, D24.2.62). + +/// The state's bytes in a switch frame: 32 128-bit registers, then `FPCR` and `FPSR`. +pub const STATE_BYTES: usize = 32 * 16 + 16; diff --git a/kernel/src/arch/aarch64/hw.rs b/kernel/src/arch/aarch64/hw.rs index c7f2adf025..cb74189d88 100644 --- a/kernel/src/arch/aarch64/hw.rs +++ b/kernel/src/arch/aarch64/hw.rs @@ -1,74 +1,104 @@ -//! `KernelHw` — the kernel's side of the scheduler-core hardware boundary: -//! the generic timer, SGIs, `WFI` and the context switch, the port's stage 4. +//! AArch64's half of `crate::hw::KernelHw`: the halt and the context switch. use toyos_sched::cpu::RunToken; -use toyos_sched::hw::{CpuId, Hw, Kicker, Machine, Nanos, TraceEvent}; +use toyos_sched::hw::Hw; use toyos_sched::task::{TaskAccounting, TaskKey}; -use crate::sched::payload::KernelPayload; - -/// The one instance; zero-sized, holds no per-CPU state. -pub static HW: KernelHw = KernelHw; - -pub struct KernelHw; - -/// The scheduler clock, in raw nanoseconds. -pub fn now_ns() -> u64 { - HW.now().0 +use super::switch::{context_switch, RETURN_AT}; +use super::{cpu, percpu}; +use crate::hw::{report_contexts, KernelHw}; +use crate::sched::payload::{KernelCtx, KernelPayload}; + +/// `WFI` with interrupts masked, then unmasked: a pending interrupt wakes +/// `WFI` whatever `DAIF` says (Arm ARM K.a, D1.6.2), so a wake that lands +/// between the decision and the wait is taken right after it, not slept through. +pub fn halt() { + // SAFETY: waits for an interrupt and unmasks `I` and `F`; touches no memory. + unsafe { core::arch::asm!("wfi", "msr daifclr, #3", "isb", options(nomem, nostack)) }; } -impl Kicker for KernelHw { - fn kick(&self, _target: CpuId) { - owed!("the interrupt controller", "stage 4") - } +/// Panics before the switch's `ret` would land somewhere that makes the failure unnameable. +#[cold] +#[inline(never)] +fn switch_frame_is_wrong(ctx: &KernelCtx, sp: u64) -> ! { + report_contexts(sp, Some(ctx as *const KernelCtx as u64)); + panic!( + "context_switch: the frame about to be restored is not one — its sp {sp:#018x} is not a \ + 16-byte-aligned kernel address, or its return slot is not kernel text (stack top {:#018x})", + ctx.kernel_stack_top, + ); } -impl Machine for KernelHw { - fn now(&self) -> Nanos { - Nanos(crate::clock::nanos_since_boot()) - } - - fn set_timer(&self, _deadline: Nanos) { - owed!("the timer", "stage 4") - } - - fn stop_timer(&self) { - owed!("the timer", "stage 4") +/// The incoming context's saved stack pointer, checked; the only load of it. +#[inline] +#[must_use] +fn check_switch_frame(ctx: &KernelCtx) -> u64 { + let sp = ctx.sp; + if !crate::mm::is_kernel_addr(sp) || !sp.is_multiple_of(16) { + switch_frame_is_wrong(ctx, sp); } - - fn halt(&self) { - owed!("the interrupt controller", "stage 4") + #[cfg(feature = "stack-witness")] + { + let top = match ctx.id { + Some(_) => ctx.kernel_stack_top, + None => percpu::idle_stack_top(), + }; + if sp > top || sp <= top - crate::process::KERNEL_STACK_SIZE as u64 { + switch_frame_is_wrong(ctx, sp); + } } - - fn need_resched(&self, _cpu: CpuId) { - owed!("the interrupt controller", "stage 4") - } - - fn trace(&self, ev: TraceEvent) { - crate::trace::record(ev); + // SAFETY: `sp` is aligned inside the incoming stack, so its frame's return slot is mapped. + let ret = unsafe { core::ptr::read_volatile((sp + RETURN_AT as u64) as *const u64) }; + if !crate::mm::is_kernel_addr(ret) { + switch_frame_is_wrong(ctx, sp); } + sp } impl Hw for KernelHw { type Payload = KernelPayload; - unsafe fn switch(&self, _token: RunToken) { - owed!("the context switch", "stage 4") + /// Outgoing per-CPU state is captured, and the incoming root and thread + /// pointer installed, before the stack pointer moves — after that this + /// frame no longer exists. + unsafe fn switch(&self, token: RunToken) { + let save = token.save_ptr(); + let restore = token.restore_ptr(); + // SAFETY: `save`/`restore` are live Box-backed contexts from + // `SchedPass::finish`, freed only by a later pass. + unsafe { + (*save).thread_pointer = cpu::thread_pointer(); + (*save).preempt = crate::preempt::count(); + let incoming: &KernelCtx = &*restore; + let sp = check_switch_frame(incoming); + crate::preempt::set_count(incoming.preempt); + percpu::set_current_tid(incoming.id.map(|id| id.1)); + percpu::set_current_pid(incoming.id.map(|id| id.0)); + match incoming.id { + Some(_) => { + #[cfg(feature = "boot-actuators")] + crate::heartbeat::note_dispatch(); + percpu::set_kernel_stack(incoming.kernel_stack_top); + incoming.root.activate(); + cpu::write_thread_pointer(incoming.thread_pointer); + } + None => { + percpu::set_kernel_stack(percpu::idle_stack_top()); + incoming.root.activate(); + } + } + crate::hw::note_running(restore); + context_switch(&raw mut (*save).sp, sp); + } } - fn release(&self, _key: TaskKey, _payload: KernelPayload, _acct: TaskAccounting) { - owed!("the context switch", "stage 4") + /// Reached once per task, from a later pass running on another stack, so + /// dropping `payload` here never frees the stack this call stands on. + fn release(&self, _key: TaskKey, payload: KernelPayload, acct: TaskAccounting) { + payload.handle.finalize(acct); } } -/// What every kernel crash says about the machine's contexts: until stage 4 -/// switches any, only the memory facts. Never owed: a crash report that -/// panicked would bury the crash. -pub fn report_contexts(sp: u64, _subject: Option) { - crate::log!(" Contexts: the boot CPU crashed at sp={sp:#018x}; no context has been switched"); - crate::mm::report_on_crash(); -} - /// AMD's `SYSRET` erratum has no AArch64 counterpart; the probe is x86-64's. #[cfg(feature = "boot-actuators")] pub fn sysret_ss_probe(_parkable: &crate::scheduler::Parkable) { diff --git a/kernel/src/arch/aarch64/irqchip.rs b/kernel/src/arch/aarch64/irqchip.rs index 3edc03a567..676f0e8050 100644 --- a/kernel/src/arch/aarch64/irqchip.rs +++ b/kernel/src/arch/aarch64/irqchip.rs @@ -1,31 +1,396 @@ -//! The interrupt controller: a GICv3 — distributor, one redistributor per CPU -//! and an ITS for MSIs — the port's stage 4. Until then no interrupt is -//! delivered, and every way of raising one is owed. +//! The interrupt controller and the timer: a GICv3 — the distributor the +//! MADT names, this CPU's redistributor, and the system-register CPU +//! interface (GIC architecture specification IHI 0069H) — and the generic +//! timer's EL1 virtual timer (Arm ARM K.a, chapter D12), whose PPI the GTDT +//! names. +//! +//! **What this kernel takes is SGIs and the timer's PPI, and nothing else.** +//! No SPI is routed and no LPI exists: a device's interrupt is a message the +//! GICv3 ITS translates, and every device that would take one is either a +//! driver the small-kernel track moves out of the kernel or a claimed function, +//! which the SMMUv3 of the port's stage 6 must translate first +//! (`super::msi_message`). +//! +//! **The virtual timer, not the physical**: it is the one EL1 owns outright — +//! under a hypervisor the physical one traps — and with `CNTVOFF_EL2` written +//! zero by the entry from EL2 the two count alike. It is level-triggered, so a +//! handler that neither re-arms nor stops it takes it again at once. +use core::sync::atomic::{AtomicU32, AtomicU64, Ordering::Relaxed}; -/// Wake `cpu` so it runs a scheduler pass: an SGI. -pub fn kick_cpu(_cpu: u32) { - owed!("the interrupt controller", "stage 4") +use toyos_acpi::MadtEntry; +use toyos_gicv3::{packed_affinity, FRAME}; + +use super::{cpu, percpu}; +use crate::drivers::acpi::direct_phys; +use crate::log; +use crate::mm::policy::MmioPolicy; +use crate::mm::{DirectMap, Mmio}; +use crate::hw::MIN_ONE_SHOT; + +/// Every INTID this kernel names, in one space: the SGIs it raises, then the +/// identities the generic drivers program for a message-signalled interrupt, +/// which `super::msi_message` refuses on this machine. +#[repr(u8)] +pub(super) enum Intid { + /// Asks a CPU for a scheduler pass: x86-64's kick, which rides the + /// timer's vector there. + Kick = 0, + /// `crate::log::nested`'s delivery, which [`send_self`] raises. + LogNest, + /// What `irq-storm` floods this CPU with. + Storm, + Hda, + VirtioSound, +} + +pub(super) const SGI_KICK: u32 = Intid::Kick as u32; +#[cfg(feature = "boot-actuators")] +pub(super) const SGI_STORM: u32 = Intid::Storm as u32; + +/// `GICD_CTLR`, and its `ARE` (affinity routing — `ARE_NS` as a non-secure +/// access sees it) and Group 1 enable (`EnableGrp1` or `EnableGrp1A`, the +/// same bit either way), and the write-pending bit. +const GICD_CTLR: u64 = 0x0000; +const CTLR_ARE: u32 = 1 << 4; +const CTLR_GRP1: u32 = 1 << 1; +const RWP: u32 = 1 << 31; +/// `GICD_PIDR2.ArchRev`, bits 7:4: 3 is GICv3, 4 is GICv4. +const GICD_PIDR2: u64 = 0xFFE8; + +/// A redistributor's frames: `RD_base`, then `SGI_base` 64 KiB above it. +const GICR_CTLR: u64 = 0x0000; +const GICR_TYPER: u64 = 0x0008; +const GICR_WAKER: u64 = 0x0014; +const WAKER_PROCESSOR_SLEEP: u32 = 1 << 1; +const WAKER_CHILDREN_ASLEEP: u32 = 1 << 2; +const SGI_BASE: u64 = FRAME; +const GICR_IGROUPR0: u64 = SGI_BASE + 0x0080; +const GICR_ISENABLER0: u64 = SGI_BASE + 0x0100; +const GICR_ICENABLER0: u64 = SGI_BASE + 0x0180; +const GICR_ICPENDR0: u64 = SGI_BASE + 0x0280; +const GICR_ICACTIVER0: u64 = SGI_BASE + 0x0380; +const GICR_IPRIORITYR: u64 = SGI_BASE + 0x0400; +const GICR_ICFGR1: u64 = SGI_BASE + 0x0C04; +const GICR_IGRPMODR0: u64 = SGI_BASE + 0x0D00; + +/// Every SGI and PPI this kernel takes runs at one priority, below the mask. +const PRIORITY: u8 = 0x80; +/// `ICC_PMR_EL1`: priorities numerically below this are signalled. +const PRIORITY_MASK: u64 = 0xF0; + +/// What the GIC answers `ICC_IAR1_EL1` with when nothing is pending for it. +const SPURIOUS: u32 = 1023; + +/// The timer's PPI, as the GTDT names it; zero until [`init`]. +static TIMER_INTID: AtomicU32 = AtomicU32::new(0); +/// The distributor's frame, for [`log_state`]; zero until [`init`]. +static GICD: AtomicU64 = AtomicU64::new(0); + +/// Wait until `done`, for at most `1 / per_second` of a second counted at the +/// rate firmware states: the boot has no calibrated clock yet. +fn settles(per_second: u64, what: &str, done: impl Fn() -> bool) { + let hz = cpu::stated_counter_hz().expect("GIC: CNTFRQ_EL0 states no rate to bound a wait with"); + let until = cpu::counter() + hz / per_second; + while !done() { + assert!(cpu::counter() < until, "GIC: {what} did not settle within 1/{per_second} s"); + core::hint::spin_loop(); + } +} + +fn read_sysreg_pmr() -> u64 { + let v: u64; + // SAFETY: reads `ICC_PMR_EL1`, which `ICC_SRE_EL1.SRE` ([`init`]) makes accessible. + unsafe { core::arch::asm!("mrs {}, S3_0_C4_C6_0", out(reg) v, options(nomem, nostack, preserves_flags)) }; + v +} + +/// Bring the distributor, this CPU's redistributor, its CPU interface and its +/// timer up: every SGI and the timer's PPI enabled at [`PRIORITY`], the timer +/// stopped. Interrupts stay masked at `DAIF`; the caller unmasks them. +pub fn init(rsdp_addr: u64) { + let madt = toyos_acpi::find_table(direct_phys(), rsdp_addr, b"APIC", toyos_acpi::MADT_ENTRIES) + .unwrap_or_else(|e| panic!("GIC: the MADT is unusable: {e:?}")); + let me = cpu::hardware_id(); + let (mut gicd, mut own_frame, mut ranges) = (None, None, [(0u64, 0u32); 4]); + let mut range_count = 0; + for entry in toyos_acpi::madt_entries(&madt) { + match entry { + Ok(MadtEntry::Gicd { base, .. }) => gicd = Some(base), + Ok(MadtEntry::Gicr { base, length }) => { + assert!(range_count < ranges.len(), "GIC: the MADT names more redistributor ranges than {}", ranges.len()); + ranges[range_count] = (base, length); + range_count += 1; + } + Ok(MadtEntry::Gicc(gicc)) if packed_affinity(gicc.mpidr) == me && gicc.gicr_base != 0 => { + own_frame = Some(gicc.gicr_base) + } + Ok(_) => {} + Err(halt) => panic!("GIC: a MADT entry at +{} declares {} bytes of a {}-byte list", halt.at, halt.declared, halt.list_len), + } + } + let gicd = gicd.expect("GIC: the MADT names no distributor"); + let distributor = crate::mm::paging::map_mmio(gicd, FRAME, MmioPolicy::Uncacheable); + GICD.store(gicd, Relaxed); + let revision = distributor.read_u32(GICD_PIDR2) >> 4 & 0xF; + assert!(revision >= 3, "GIC: GICD_PIDR2.ArchRev is {revision}, and this kernel drives a GICv3 or later"); + + // The frame is found in the ranges when the GICC entry does not name it: + // each redistributor says whose it is in `GICR_TYPER`'s affinity. + let redistributor = match own_frame { + Some(frame) => crate::mm::paging::map_mmio(frame, 2 * FRAME, MmioPolicy::Uncacheable), + None => ranges[..range_count] + .iter() + .find_map(|&(base, length)| find_redistributor(base, u64::from(length), me)) + .unwrap_or_else(|| panic!("GIC: no redistributor answers for MPIDR affinity {me:#x}")), + }; + + // The distributor: affinity routing on, and Group 1 on. + distributor.write_u32(GICD_CTLR, CTLR_ARE | CTLR_GRP1); + settles(100, "GICD_CTLR", || distributor.read_u32(GICD_CTLR) & RWP == 0); + assert!( + distributor.read_u32(GICD_CTLR) & CTLR_ARE != 0, + "GIC: GICD_CTLR.ARE reads clear, so the distributor stays in legacy mode and no redistributor is used" + ); + + // Awake, then every SGI and PPI off and cleared of what firmware left. + let waker = redistributor.read_u32(GICR_WAKER); + redistributor.write_u32(GICR_WAKER, waker & !WAKER_PROCESSOR_SLEEP); + settles(100, "GICR_WAKER.ChildrenAsleep", || redistributor.read_u32(GICR_WAKER) & WAKER_CHILDREN_ASLEEP == 0); + redistributor.write_u32(GICR_ICENABLER0, u32::MAX); + settles(100, "GICR_CTLR after disabling every SGI and PPI", || redistributor.read_u32(GICR_CTLR) & (1 << 3) == 0); + redistributor.write_u32(GICR_ICPENDR0, u32::MAX); + redistributor.write_u32(GICR_ICACTIVER0, u32::MAX); + redistributor.write_u32(GICR_IGROUPR0, u32::MAX); + redistributor.write_u32(GICR_IGRPMODR0, 0); + let priorities = u32::from_ne_bytes([PRIORITY; 4]); + for word in 0..8 { + redistributor.write_u32(GICR_IPRIORITYR + word * 4, priorities); + } + + let timer = toyos_acpi::gtdt(direct_phys(), rsdp_addr) + .unwrap_or_else(|e| panic!("GIC: the GTDT is unusable: {e:?}")) + .virtual_el1; + assert!((16..32).contains(&timer.gsiv), "GIC: the GTDT puts the virtual timer at INTID {}, which is not a PPI", timer.gsiv); + TIMER_INTID.store(timer.gsiv, Relaxed); + // Each PPI's two bits in `GICR_ICFGR1`: 0b10 edge, 0b00 level. + let edge = u32::from(timer.edge()) << (2 * (timer.gsiv - 16) + 1); + redistributor.write_u32(GICR_ICFGR1, edge); + + stop_timer_hardware(); + let enabled = 1 << SGI_KICK | 1 << timer.gsiv; + #[cfg(feature = "boot-actuators")] + let enabled = enabled | 1 << Intid::LogNest as u32 | 1 << SGI_STORM; + redistributor.write_u32(GICR_ISENABLER0, enabled); + + // The CPU interface in system registers, as the declaration says. + // SAFETY: writes `ICC_SRE_EL1`, which touches no memory. + unsafe { + core::arch::asm!( + "msr S3_0_C12_C12_5, {}", + "isb", + in(reg) super::control_regs::ICC_SRE, + options(nomem, nostack, preserves_flags), + ); + } + let sre: u64; + // SAFETY: reads `ICC_SRE_EL1`. + unsafe { core::arch::asm!("mrs {}, S3_0_C12_C12_5", out(reg) sre, options(nomem, nostack, preserves_flags)) }; + // `SRE` alone: `DFB` and `DIB` are RAO/WI or RAZ/WI as the implementation, + // or a hypervisor under it, chooses, and this kernel uses neither bypass. + assert!(sre & 1 != 0, "GIC: ICC_SRE_EL1 reads {sre:#x}, so the CPU interface is not in system registers"); + + // The CPU interface: the priority mask open above `PRIORITY`, one + // priority drop and deactivation per EOI, Group 1 on. + // SAFETY: GICv3 CPU interface registers, which `ICC_SRE_EL1.SRE` makes + // system registers; none touches memory. + unsafe { + core::arch::asm!( + "msr S3_0_C4_C6_0, {pmr}", + "msr S3_0_C12_C12_4, xzr", + "msr S3_0_C12_C12_7, {one}", + "isb", + pmr = in(reg) PRIORITY_MASK, + one = in(reg) 1u64, + options(nomem, nostack, preserves_flags), + ); + } + assert_eq!(read_sysreg_pmr(), PRIORITY_MASK, "GIC: ICC_PMR_EL1 did not take {PRIORITY_MASK:#x}"); + log!( + "GIC: v{revision} distributor at {gicd:#x}, this CPU's redistributor at {:#x}; SGIs and the virtual \ + timer's PPI {} ({}) enabled at priority {PRIORITY:#x}", + redistributor.addr() - crate::mm::PHYS_OFFSET, + timer.gsiv, + if timer.edge() { "edge" } else { "level" }, + ); +} + +/// The redistributor frame in `[base, base + length)` whose affinity is `me`. +fn find_redistributor(base: u64, length: u64, me: u32) -> Option { + let range = crate::mm::paging::map_mmio(base, length, MmioPolicy::Uncacheable); + let offset = toyos_gicv3::find_redistributor(length, me, |at| range.read_u64(at + GICR_TYPER))?; + Some(Mmio::new(DirectMap::from_phys(base + offset), 2 * FRAME)) +} + +/// The INTID the GIC hands this CPU, or `None` when it answered spurious. +pub(super) fn acknowledge() -> Option { + let intid: u64; + // SAFETY: reads `ICC_IAR1_EL1`, which acknowledges the highest pending + // Group 1 interrupt; the caller ends every one it is given. + unsafe { core::arch::asm!("mrs {}, S3_0_C12_C12_0", out(reg) intid, options(nomem, nostack, preserves_flags)) }; + let intid = (intid & 0xFF_FFFF) as u32; + (intid != SPURIOUS).then_some(intid) +} + +/// Drop the running priority and deactivate `intid`. +pub(super) fn end(intid: u32) { + // SAFETY: writes `ICC_EOIR1_EL1` with an INTID [`acknowledge`] handed out. + unsafe { core::arch::asm!("msr S3_0_C12_C12_1, {}", in(reg) u64::from(intid), options(nomem, nostack, preserves_flags)) }; +} + +/// The timer's INTID, which [`init`] took from the GTDT. +pub(super) fn timer_intid() -> u32 { + TIMER_INTID.load(Relaxed) +} + +/// Raise SGI `intid` on the CPU whose packed affinity is `target`. +fn sgi(intid: u32, target: u32) { + let value = toyos_gicv3::sgi1r(intid, target); + // SAFETY: writes `ICC_SGI1R_EL1`, which raises an SGI and touches no + // memory; the `ISB` sends it before whatever follows. + unsafe { core::arch::asm!("msr S3_0_C12_C11_5, {}", "isb", in(reg) value, options(nomem, nostack, preserves_flags)) }; +} + +/// Wake `cpu` so it runs a scheduler pass. Before the port's stage 5 the boot +/// CPU is the only one, so the only `cpu` there is is this one. +pub fn kick_cpu(cpu: u32) { + assert_eq!(cpu, percpu::cpu_id(), "irqchip: a kick for cpu{cpu}, and other CPUs are the port's stage 5"); + sgi(SGI_KICK, cpu::hardware_id()); } pub fn kick_all_but_self() { - owed!("the interrupt controller", "stage 4") + let me = percpu::cpu_id(); + for cpu in (0..super::smp::cpu_count()).filter(|&cpu| cpu != me) { + kick_cpu(cpu); + } } -/// Raise `vector` on this CPU. -pub fn send_self(_vector: u8) { - owed!("the interrupt controller", "stage 4") +/// Raise `vector` — an SGI's INTID — on this CPU. +pub fn send_self(vector: u8) { + sgi(u32::from(vector), cpu::hardware_id()); } -/// A pseudo-NMI to `cpu`. +/// A pseudo-NMI: an interrupt at a priority `DAIF.I` does not mask, which +/// needs `ICC_PMR_EL1` priority masking in place of `DAIF` everywhere. pub fn send_nmi(_cpu: u32) { - owed!("the interrupt controller", "stage 4") + owed!("a pseudo-NMI", "no stage yet") } -/// This CPU's timer, armed to fire within `nanos`. -pub fn arm_within(_nanos: u64) { - owed!("the timer", "stage 4") +/// Nothing to stop: before the port's stage 5 the boot CPU is the only one running. +pub fn stop_other_cpus() {} + +/// `CNTV_CTL_EL0.ENABLE`; `IMASK` stays clear. +const TIMER_ENABLE: u64 = 1; +/// `CNTV_CTL_EL0.ISTATUS`: the timer's condition is met. +#[cfg(feature = "boot-actuators")] +const TIMER_ISTATUS: u64 = 1 << 2; + +/// [`MIN_ONE_SHOT`] in counter ticks, and never none. +fn floor_ticks() -> u64 { + crate::clock::counter_ticks(MIN_ONE_SHOT.nanos()).max(1) } -/// Nothing to stop: before stage 5 the boot CPU is the only one running. -pub fn stop_other_cpus() {} +/// The only write of the comparator: fire `ticks` counter ticks from now, or +/// after [`MIN_ONE_SHOT`] if that is longer, and remember the span as what an +/// EL1 fire re-arms with. Returns the counter value the comparator was set +/// from, for a caller that must relate CVAL back to it without a second read. +fn arm_ticks(ticks: u64) -> u64 { + let ticks = ticks.max(floor_ticks()); + percpu::set_armed_ticks(ticks); + let now = cpu::counter(); + // SAFETY: the EL1 virtual timer's comparator and control; CPACR has + // nothing to say about them and `CNTKCTL_EL1` keeps EL0 out. + unsafe { + core::arch::asm!( + "msr cntv_cval_el0, {cval}", + "msr cntv_ctl_el0, {enable}", + "isb", + cval = in(reg) now + ticks, + enable = in(reg) TIMER_ENABLE, + options(nomem, nostack, preserves_flags), + ); + } + now +} + +fn stop_timer_hardware() { + // SAFETY: the EL1 virtual timer's control, cleared: it asserts nothing. + unsafe { core::arch::asm!("msr cntv_ctl_el0, xzr", "isb", options(nomem, nostack, preserves_flags)) }; +} + +/// This CPU's timer, armed to fire `nanos` from now, or after [`MIN_ONE_SHOT`] if that is longer. +pub fn arm_one_shot(nanos: u64) { + arm_ticks(crate::clock::counter_ticks(nanos)); + crate::trace::trace(crate::trace::Kind::TimerArm, nanos as u32); +} + +/// This CPU's timer, armed to fire within `nanos`: sooner than it is armed +/// for, or armed if it is stopped. Returns the counter value the comparator +/// was set from ([`arm_ticks`]). +pub fn arm_within(nanos: u64) -> u64 { + let want = crate::clock::counter_ticks(nanos); + let remaining = match percpu::armed_ticks() { + 0 => want, + _ => { + let cval: u64; + // SAFETY: reads the EL1 virtual timer's comparator. + unsafe { core::arch::asm!("mrs {}, cntv_cval_el0", out(reg) cval, options(nomem, nostack, preserves_flags)) }; + cval.saturating_sub(cpu::counter()) + } + }; + arm_ticks(want.min(remaining)) +} + +/// Stop the timer: no interrupt until it is armed again. +pub fn stop_timer() { + percpu::set_armed_ticks(0); + stop_timer_hardware(); + crate::trace::trace(crate::trace::Kind::TimerStop, 0); +} + +/// A timer interrupt taken: armed again for what it was last armed for, or +/// stopped if it was stopped — the one thing that deasserts it. +pub(super) fn rearm() { + match percpu::armed_ticks() { + 0 => stop_timer_hardware(), + ticks => { + arm_ticks(ticks); + } + } +} + +/// `timer-floor`: this CPU's timer made due with interrupts masked, then +/// asked to fire within a quantum, which leaves it nothing to fire within. +/// The comparator it is left holding must be at least [`MIN_ONE_SHOT`] past +/// the counter value [`arm_within`] set it from — not a counter read framing +/// the call, which a slow call (a QEMU host under load) widens for no defect. +#[cfg(feature = "boot-actuators")] +pub fn floor_selftest() { + let _guard = crate::arch::IrqGuard::close(); + arm_one_shot(0); + let due = || { + let ctl: u64; + // SAFETY: reads the EL1 virtual timer's control. + unsafe { core::arch::asm!("mrs {}, cntv_ctl_el0", out(reg) ctl, options(nomem, nostack, preserves_flags)) }; + ctl & TIMER_ISTATUS != 0 + }; + settles(100, "the timer armed for its floor", due); + let now = arm_within(toyos_sched::fair::QUANTUM_NS); + let cval: u64; + // SAFETY: reads the EL1 virtual timer's comparator. + unsafe { core::arch::asm!("mrs {}, cntv_cval_el0", out(reg) cval, options(nomem, nostack, preserves_flags)) }; + stop_timer(); + let (span, floor) = (cval.saturating_sub(now), floor_ticks()); + let verdict = if span >= floor { "PASS" } else { "FAIL" }; + log!("timer-floor: {verdict} span={span} floor={floor} ticks: the comparator past the counter it was set from"); +} diff --git a/kernel/src/arch/aarch64/mod.rs b/kernel/src/arch/aarch64/mod.rs index 5c9e00e144..0b8952c634 100644 --- a/kernel/src/arch/aarch64/mod.rs +++ b/kernel/src/arch/aarch64/mod.rs @@ -2,15 +2,6 @@ //! AArch64: the Arm A-profile at EL1, found and described through ACPI. //! //! Every `unsafe` block here carries a one-line `SAFETY:` comment, enforced by the lint above. -//! -//! **What exists and what is owed.** The boot reaches its console: the entry -//! drops from EL2, applies the control-register declaration, turns on the -//! loader's tables and installs the exception vectors; the PL011 is found -//! through SPCR. Everything the kernel does after the console — the interrupt -//! controller, the timer, its own page tables, other CPUs, user mode — is -//! owed by a stage of the port (`issues/kernel/toyos-runs-on-arm64.md`), and -//! each item that stands for it here is an [`owed!`] that panics naming it. A kernel that reaches -//! one stops loudly on its panel; none of them returns a guess. /// Stands for work the port owes: panics naming what and which stage of the /// track owns it (`issues/kernel/toyos-runs-on-arm64.md`), or that none does yet. @@ -39,6 +30,7 @@ pub mod pio; pub mod pmu; pub mod rtc; pub mod smp; +pub mod switch; pub mod syscall; pub mod tlb; pub mod trap; @@ -48,10 +40,13 @@ pub mod watchdog; pub const ELF_MACHINE: toyos_elf::Machine = toyos_elf::Machine::Aarch64; /// A message-signalled interrupt's address and data for `vector` on CPU -/// `dest`. On AArch64 the doorbell is an ITS's `GITS_TRANSLATER`, one per ITS -/// the MADT names, and the data is an event the ITS maps: stage 4 builds both. -pub fn msi_message(_dest: u32, _vector: u8) -> (u32, u32) { - owed!("an MSI doorbell (the GICv3 ITS)", "stage 4") +/// `dest`, which this machine does not give: the doorbell is an ITS's +/// `GITS_TRANSLATER` and the data an event the ITS maps, and nothing here +/// drives an ITS. Every function that would take one is a driver the +/// small-kernel track moves out of the kernel, or a claimed function the +/// SMMUv3 of the port's stage 6 must translate first; each is refused by name. +pub fn msi_message(_dest: u32, _vector: u8) -> Result<(u32, u32), &'static str> { + Err("AArch64 delivers no message-signalled interrupt to this kernel: the GICv3 ITS is unported") } /// Interrupts masked on this CPU for as long as the guard lives, and then put diff --git a/kernel/src/arch/aarch64/paging.rs b/kernel/src/arch/aarch64/paging.rs index 0fd3db1943..df990ec4c3 100644 --- a/kernel/src/arch/aarch64/paging.rs +++ b/kernel/src/arch/aarch64/paging.rs @@ -1,154 +1,695 @@ -//! Page tables. The boot runs on the loader's: one L0 table that `TTBR0_EL1` -//! walks for the identity view and `TTBR1_EL1` for the view at `PHYS_OFFSET`, -//! 2 MiB blocks whose `AttrIndx` names [`super::control_regs`]'s `MAIR_EL1`. -//! The kernel's own tables — the direct map, MMIO windows, and a user -//! address space per process with an ASID — are the port's stage 4, so an -//! [`AddressSpace`] cannot exist yet: the type is uninhabited, and every -//! method on it is a match on nothing. - -use core::convert::Infallible; - -use crate::mm::UserAddr; +//! Page tables and address spaces: VMSAv8-64 stage 1 with a 4 KiB granule +//! (Arm ARM K.a, chapter D8), and the only code that writes a translation +//! table entry. +//! +//! **Two tables, not one root with two halves.** `TTBR1_EL1` walks the +//! kernel's tables — the direct map at `PHYS_OFFSET`, which every address +//! space shares because every one runs under the same `TTBR1_EL1` — and +//! `TTBR0_EL1` walks one [`AddressSpace`]'s user half, tagged with the ASID it +//! owns. A user space copies nothing of the kernel's, and switching to one is +//! one register write. +//! +//! **The direct map holds memory and nothing else**: every 4 KiB page +//! firmware's map calls memory the kernel reads (`toyos_bootmap::aarch64`), +//! Normal write-back; a device's registers only as [`map_mmio`] maps them; and +//! the scanout. A page is never retyped: a mapping that disagrees with the one +//! already there is refused by name, because one page under two memory types +//! loses coherency (D8.2.12). +//! +//! **A live entry is replaced break-before-make**: written invalid, its +//! translation dropped on every CPU (`super::tlb`), and only then written +//! again (D8.14.1). An entry that was invalid owes nothing, since no TLB holds +//! a translation that faulted. + +use alloc::boxed::Box; +use alloc::vec::Vec; + +use toyos_bootmap::aarch64::{coverage, direct_map_end, Coverage, ATTR_DEVICE, ATTR_NORMAL, ATTR_NORMAL_NC}; +use toyos_pcid::{Alloc, Pcid, PcidPool}; + +use super::tlb; use crate::mm::policy::{CachePolicy, MmioPolicy, Prot, WindowProt}; -use crate::sync::Lock; -use crate::vma::{Occupancy, Region, RegionKind}; +use crate::mm::{DirectMap, UserAddr, PAGE_2M, PHYS_OFFSET}; +use crate::sync::{Lock, LockGuard}; +use crate::vma::{self, Occupancy, Region, RegionKind}; use crate::MemoryMapEntry; -/// A translation table base: what `TTBR0_EL1` is loaded with for a space. +const VALID: u64 = 1 << 0; +/// At levels 0 to 2 a descriptor naming the next table, at level 3 a page; +/// clear at level 2 is a 2 MiB block. +const TABLE: u64 = 1 << 1; +const ATTR_INDEX: u64 = 0b111 << 2; +/// `AP[1]`: EL0 may access. +const AP_EL0: u64 = 1 << 6; +/// `AP[2]`: read-only at every level that may access. +const AP_READ_ONLY: u64 = 1 << 7; +const INNER_SHAREABLE: u64 = 0b11 << 8; +const OUTER_SHAREABLE: u64 = 0b10 << 8; +/// The access flag: set, so no first access faults. +const AF: u64 = 1 << 10; +/// Not global: the translation is the owning ASID's alone. +const NOT_GLOBAL: u64 = 1 << 11; +const PXN: u64 = 1 << 53; +const UXN: u64 = 1 << 54; +const ADDR: u64 = 0x0000_FFFF_FFFF_F000; +const ADDR_2M: u64 = 0x0000_FFFF_FFE0_0000; +const PAGE_4K: u64 = crate::mm::PAGE_SIZE; + +/// A leaf's memory type and shareability. A device is never executable, +/// since a speculative fetch from one is a read of its registers. +fn typed(cache: CachePolicy) -> u64 { + match cache { + CachePolicy::Normal => ATTR_NORMAL << 2 | INNER_SHAREABLE, + CachePolicy::Uncacheable => ATTR_DEVICE << 2 | PXN | UXN, + CachePolicy::WriteCombining => ATTR_NORMAL_NC << 2 | OUTER_SHAREABLE, + } +} + +/// The type a leaf this file wrote names; any other index is one it never wrote. +fn policy_of(leaf: u64) -> CachePolicy { + match (leaf & ATTR_INDEX) >> 2 { + ATTR_NORMAL => CachePolicy::Normal, + ATTR_DEVICE => CachePolicy::Uncacheable, + ATTR_NORMAL_NC => CachePolicy::WriteCombining, + index => panic!("paging: the leaf {leaf:#x} names MAIR_EL1 index {index}, which this kernel never writes"), + } +} + +impl Prot { + /// A user leaf's permissions: EL0 reads every one, and EL1 executes none. + fn user_bits(self) -> u64 { + match self { + Self::Read => AP_EL0 | AP_READ_ONLY | UXN | PXN, + Self::ReadWrite => AP_EL0 | UXN | PXN, + Self::ReadExec => AP_EL0 | AP_READ_ONLY | PXN, + } + } +} + +/// A user leaf, bar its address and whether it is a block or a page. +fn user_leaf(prot: Prot, cache: CachePolicy) -> u64 { + VALID | AF | NOT_GLOBAL | typed(cache) | prot.user_bits() +} + +/// A kernel leaf: EL1 read-write, EL0 nothing, global, and executable at EL1 +/// only where it is memory — the kernel runs from the direct map. +fn kernel_leaf(cache: CachePolicy) -> u64 { + VALID | AF | typed(cache) | UXN +} + +/// A leaf with its address and its block-or-page bit taken out: what two +/// mappings of one page must agree on. +fn attributes(leaf: u64) -> u64 { + leaf & !ADDR & !TABLE +} + +/// The index `va` takes at `level` (0 to 3) of a 4 KiB-granule walk. +fn index(va: u64, level: u32) -> usize { + ((va >> (39 - 9 * level)) & 0x1FF) as usize +} + +/// One translation table: 512 descriptors in one aligned page. +#[repr(C, align(4096))] +struct Table([u64; 512]); + +impl Table { + fn new() -> Box { + Box::new(Self([0; 512])) + } + + fn phys(&self) -> u64 { + DirectMap::phys_of(self) + } + + /// # Safety + /// `phys` is a table this file built and linked, alive while the reference is. + unsafe fn at<'a>(phys: u64) -> &'a Table { + // SAFETY: the caller's contract. + unsafe { &*DirectMap::from_phys(phys).as_ptr::() } + } + + /// # Safety + /// As [`Table::at`], and the reference is the only live one. + unsafe fn at_mut<'a>(phys: u64) -> &'a mut Table { + // SAFETY: the caller's contract. + unsafe { &mut *DirectMap::from_phys(phys).as_mut_ptr::
() } + } + + /// The next table down from a valid table descriptor at `i`. + fn child(&self, i: usize) -> Option<&Table> { + let entry = self.0[i]; + // SAFETY: valid and a table, so it names one this file linked. + (entry & (VALID | TABLE) == VALID | TABLE).then(|| unsafe { Table::at(entry & ADDR) }) + } + + /// Write the entry at `i`, which the hardware may be walking: one aligned + /// store, so no walker sees half a descriptor. + fn set(&mut self, i: usize, value: u64) { + // SAFETY: `i` indexes this table, a live `&mut`. + unsafe { core::ptr::write_volatile(&raw mut self.0[i], value) }; + } +} + +/// Every table written reached every walker, and this CPU walks afresh. +fn published() { + // SAFETY: barriers; they touch no memory. + unsafe { core::arch::asm!("dsb ishst", "isb", options(nostack, preserves_flags)) }; +} + +/// A root and every table under it, owned together and freed together. +struct Tables { + root: Box
, + children: Vec>, +} + +impl Tables { + fn new() -> Self { + Self { root: Table::new(), children: Vec::new() } + } + + fn adopt(&mut self, table: Box
) -> u64 { + let phys = table.phys(); + self.children.push(table); + phys + } + + /// The level-2 table over `va`, built down to it where absent. Only an + /// invalid entry is written, which owes no invalidation. + fn directory(&mut self, va: u64) -> &mut Table { + let mut table: *mut Table = &mut *self.root; + for level in 0..2 { + // SAFETY: `table` is this root or a table it linked, and + // `&mut self` makes this the only reference. + let current = unsafe { &mut *table }; + let i = index(va, level); + if current.0[i] & VALID == 0 { + let phys = self.adopt(Table::new()); + current.set(i, phys | TABLE | VALID); + } + let entry = current.0[i]; + assert!(entry & TABLE != 0, "paging: a block at level {level} over {va:#x}, which this kernel never writes"); + // SAFETY: valid and a table: one this root linked. + table = unsafe { Table::at_mut(entry & ADDR) }; + } + // SAFETY: as in the loop. + unsafe { &mut *table } + } + + /// The level-2 table over `va`, if the walk reaches one. + fn find_directory(&self, va: u64) -> Option<&Table> { + self.root.child(index(va, 0))?.child(index(va, 1)) + } + + /// The leaf that maps `va` and the size it maps, if one does. + fn leaf(&self, va: u64) -> Option<(u64, u64)> { + let directory = self.find_directory(va)?; + let entry = directory.0[index(va, 2)]; + if entry & VALID == 0 { + return None; + } + if entry & TABLE == 0 { + return Some((entry, PAGE_2M)); + } + let page = directory.child(index(va, 2))?.0[index(va, 3)]; + (page & VALID != 0).then_some((page, PAGE_4K)) + } + + /// Map `[phys, phys + size)` at `PHYS_OFFSET + phys` with `leaf`'s + /// attributes: a whole 2 MiB page as a block where nothing maps it yet, + /// the rest page by page. A page already mapped the same way is left as + /// it is; one mapped any other way is refused. + fn map_direct(&mut self, phys: u64, size: u64, leaf: u64) { + let (start, end) = (phys & !(PAGE_4K - 1), (phys + size).next_multiple_of(PAGE_4K)); + let mut at = start; + while at < end { + let va = PHYS_OFFSET + at; + let block = at & !(PAGE_2M - 1); + let i = index(va, 2); + let entry = self.directory(va).0[i]; + if entry & (VALID | TABLE) == VALID { + refuse_unless_same(at, entry, leaf); + at = block + PAGE_2M; + continue; + } + if entry & VALID == 0 && at == block && block + PAGE_2M <= end { + self.directory(va).set(i, block | leaf); + at += PAGE_2M; + continue; + } + let pages = if entry & VALID == 0 { + let phys = self.adopt(Table::new()); + self.directory(va).set(i, phys | TABLE | VALID); + phys + } else { + entry & ADDR + }; + // SAFETY: a table this root linked, reached under `&mut self`. + let pages = unsafe { Table::at_mut(pages) }; + let j = index(va, 3); + if pages.0[j] & VALID != 0 { + refuse_unless_same(at, pages.0[j], leaf); + } else { + pages.set(j, at | leaf | TABLE); + } + at += PAGE_4K; + } + published(); + } +} + +/// A second mapping of the page at `phys` agrees with the first, or the +/// kernel stops: [`Tables::map_direct`]'s refusal. +fn refuse_unless_same(phys: u64, existing: u64, wanted: u64) { + assert!( + attributes(existing) == attributes(wanted), + "paging: {phys:#x} is mapped {:?} ({existing:#x}) and cannot also be {:?}", + policy_of(existing), + policy_of(wanted), + ); +} + +/// What `TTBR0_EL1` is loaded with for a space: its ASID in bits 63:48 and +/// its root table's address. #[derive(Clone, Copy)] -pub struct Root(Infallible); +pub struct Root(u64); impl Root { pub fn phys(self) -> u64 { - match self.0 {} + self.0 & ADDR } + /// No invalidation: the ASID is this space's alone, and a returned one was + /// dropped from every CPU before it was issued again (`toyos_pcid`). /// # Safety /// The underlying page tables must be valid and live. pub unsafe fn activate(self) { - match self.0 {} + // SAFETY: the caller's contract; the `ISB` makes the next walk use it. + unsafe { core::arch::asm!("msr ttbr0_el1, {}", "isb", in(reg) self.0, options(nostack, preserves_flags)) }; + } +} + +/// The ASID allocator: `toyos_pcid`'s tags, which 16-bit ASIDs hold whole, +/// and its reclaim given the shootdown it asks for. +static ASIDS: Lock = Lock::new(PcidPool::new()); + +/// A user ASID owned for one space's life; its drop returns it. +struct AsidGuard(Pcid); + +impl Drop for AsidGuard { + fn drop(&mut self) { + ASIDS.lock().free(self.0); + } +} + +enum Asid { + /// ASID 0, the kernel space's, whose user half is empty. + Kernel, + User(AsidGuard), +} + +impl Asid { + fn value(&self) -> u16 { + match self { + Self::Kernel => toyos_pcid::KERNEL_PCID, + Self::User(guard) => guard.0.get(), + } } } -/// A process's address space: its regions and the tables that map them. +/// `None` when every user ASID is held by a live space. +fn alloc_asid() -> Option { + let mut pool = ASIDS.lock(); + loop { + match pool.alloc() { + Alloc::Ready(tag) => return Some(AsidGuard(tag)), + Alloc::NeedsFlush => { + tlb::all(crate::invalidation::Origin::Pcid); + pool.reclaim(); + } + Alloc::Exhausted => return None, + } + } +} + +/// A process's user half, walked through `TTBR0_EL1`: its tables, its +/// regions and its ASID. pub struct AddressSpace { - never: Infallible, + tables: Tables, + regions: vma::Regions, + asid: Asid, } impl AddressSpace { + /// An empty user half, or `None` when every user ASID is held. pub fn new_user() -> Option { - owed!("a user address space", "stage 4") + let asid = alloc_asid()?; + Some(Self { tables: Tables::new(), regions: vma::Regions::default(), asid: Asid::User(asid) }) } pub fn root(&self) -> Root { - match self.never {} + Root(u64::from(self.asid.value()) << 48 | self.tables.root.phys()) + } + + /// Replace the level-2 entry over `va`, break-before-make where it was + /// valid; a table it named stays owned by this space until it drops. + fn replace(&mut self, va: u64, value: u64) { + let asid = self.asid.value(); + let directory = self.tables.directory(va); + let i = index(va, 2); + let prior = directory.0[i]; + if prior & VALID != 0 { + directory.set(i, 0); + if prior & TABLE != 0 { + tlb::asid(asid); + } else { + tlb::page(asid, va); + } + } + directory.set(i, value); + published(); + } + + /// Empty level-2 entries (aligned, asserted) mean nothing can be stale. + pub fn map_range(&mut self, vaddr: UserAddr, phys: u64, size: u64, prot: Prot, cache: CachePolicy) { + assert!(vaddr.raw() & (PAGE_2M - 1) == 0, "map_range: vaddr not 2MB-aligned"); + assert!(phys & (PAGE_2M - 1) == 0, "map_range: phys {phys:#x} not 2MB-aligned"); + if prot == Prot::ReadExec { + super::cache::make_executable(DirectMap::from_phys(phys).as_ptr::() as u64, size as usize); + } + let mut offset = 0u64; + while offset < size { + let va = vaddr.raw() + offset; + let directory = self.tables.directory(va); + let i = index(va, 2); + assert!(directory.0[i] & VALID == 0, "map_range: an install at {va:#x} found the present entry {:#x}", directory.0[i]); + directory.set(i, (phys + offset) | user_leaf(prot, cache)); + offset += PAGE_2M; + } + published(); } - pub fn map_range(&mut self, _vaddr: UserAddr, _phys: u64, _size: u64, _prot: Prot, _cache: CachePolicy) { - match self.never {} + fn unmap_range(&mut self, vaddr: UserAddr, size: u64) { + let mut offset = 0u64; + while offset < size { + self.unmap(UserAddr::new(vaddr.raw() + offset)); + offset += PAGE_2M; + } } - pub fn remap(&mut self, _vaddr: UserAddr, _phys: u64, _prot: Prot) { - match self.never {} + /// Replaces whatever maps `vaddr`, in this space only. + pub fn remap(&mut self, vaddr: UserAddr, phys: u64, prot: Prot) { + let va = vaddr.raw(); + assert!(va & (PAGE_2M - 1) == 0, "remap: vaddr {va:#x} not 2MB-aligned"); + assert!(phys & (PAGE_2M - 1) == 0, "remap: phys {phys:#x} not 2MB-aligned"); + if prot == Prot::ReadExec { + super::cache::make_executable(DirectMap::from_phys(phys).as_ptr::() as u64, PAGE_2M as usize); + } + self.replace(va, phys | user_leaf(prot, CachePolicy::Normal)); } - pub fn map_window(&mut self, _vaddr: UserAddr, _phys: u64, _prot: &WindowProt) { - match self.never {} + /// A mixed window's table is filled before it is linked, so nothing walks + /// it half-written. Must not be called twice on one address. + pub fn map_window(&mut self, vaddr: UserAddr, phys: u64, prot: &WindowProt) { + if let Some(uniform) = prot.agreed() { + self.remap(vaddr, phys, uniform); + return; + } + let va = vaddr.raw(); + assert!(va & (PAGE_2M - 1) == 0, "map_window: vaddr {va:#x} not 2MB-aligned"); + assert!(phys & (PAGE_2M - 1) == 0, "map_window: phys {phys:#x} not 2MB-aligned"); + let mut table = Table::new(); + for (i, page_prot) in prot.pages().enumerate() { + let at = phys + i as u64 * PAGE_4K; + if page_prot == Prot::ReadExec { + super::cache::make_executable(DirectMap::from_phys(at).as_ptr::() as u64, PAGE_4K as usize); + } + table.0[i] = at | user_leaf(page_prot, CachePolicy::Normal) | TABLE; + } + let table = self.tables.adopt(table); + self.replace(va, table | TABLE | VALID); } - pub fn map_window_if_absent(&mut self, _vaddr: UserAddr, _phys: u64, _prot: &WindowProt) -> bool { - match self.never {} + /// `false` leaves the mapping as found and `phys` the caller's to free: + /// the check and the write are one critical section under the caller's lock. + pub fn map_window_if_absent(&mut self, vaddr: UserAddr, phys: u64, prot: &WindowProt) -> bool { + if self.translate(vaddr).is_some() { + return false; + } + self.map_window(vaddr, phys, prot); + true } - pub fn unmap(&mut self, _vaddr: UserAddr) { - match self.never {} + /// Ends any futex wait on the frame (its token is a physical address) + /// before the frame can reach the PMM and be reissued under a waiter. + pub fn unmap(&mut self, vaddr: UserAddr) { + let va = vaddr.raw(); + assert!(va & (PAGE_2M - 1) == 0, "unmap: vaddr {va:#x} not 2MB-aligned"); + let asid = self.asid.value(); + let Some(directory) = self.tables.find_directory(va) else { return }; + let i = index(va, 2); + let entry = directory.0[i]; + if entry & VALID == 0 { + return; + } + // A split window's entries all address one 2 MiB frame, so its first names it. + let phys = match directory.child(i) { + Some(pages) => pages.0[0] & ADDR_2M, + None => entry & ADDR_2M, + }; + self.tables.directory(va).set(i, 0); + if entry & TABLE != 0 { + tlb::asid(asid); + } else { + tlb::page(asid, va); + } + crate::sched::futex::revoke_range(phys, PAGE_2M); } - pub fn translate(&self, _vaddr: UserAddr) -> Option { - match self.never {} + /// Checked here, not at the callers: only a user address names user memory. + pub fn translate(&self, vaddr: UserAddr) -> Option { + self.walk(vaddr).map(|(at, _, _)| at) } - pub fn leaf(&self, _vaddr: UserAddr, _access: toyos_userbound::Access) -> Option<(u64, u64)> { - match self.never {} + /// `vaddr`'s physical address and the bytes from it to the end of the leaf + /// that maps it, which that one walk answers for. A `Write` is answered + /// only where an EL0 store would land: a leaf EL0 may write, and no table + /// descriptor this file writes sets `APTable`. A kernel copy into user + /// memory goes through this, so a syscall cannot write a page the process + /// itself may not — the clock page, a shared library's `.text`. + pub fn leaf(&self, vaddr: UserAddr, access: toyos_userbound::Access) -> Option<(u64, u64)> { + let (at, leaf, size) = self.walk(vaddr)?; + let granted = match access { + toyos_userbound::Access::Read => true, + toyos_userbound::Access::Write => leaf & (AP_EL0 | AP_READ_ONLY) == AP_EL0, + }; + granted.then(|| (at.phys(), size - (vaddr.raw() & (size - 1)))) } - pub fn alloc_region(&mut self, _size: u64, _kind: RegionKind) -> Option { - match self.never {} + /// The direct-map address of `vaddr`, the leaf descriptor that maps it, and the size that leaf maps. + fn walk(&self, vaddr: UserAddr) -> Option<(DirectMap, u64, u64)> { + let va = vaddr.raw(); + if !toyos_userbound::is_user_addr(va) { + return None; + } + let (leaf, size) = self.tables.leaf(va)?; + let base = leaf & if size == PAGE_2M { ADDR_2M } else { ADDR }; + Some((DirectMap::from_phys(base + (va & (size - 1))), leaf, size)) } - pub fn alloc_and_map( - &mut self, - _phys: u64, - _size: u64, - _prot: Prot, - _cache: CachePolicy, - ) -> Option<(UserAddr, u64)> { - match self.never {} + pub fn alloc_region(&mut self, size: u64, kind: RegionKind) -> Option { + self.regions.alloc(size, kind) } - pub fn free_and_unmap(&mut self, _addr: UserAddr) -> Option { - match self.never {} + pub fn alloc_and_map(&mut self, phys: u64, size: u64, prot: Prot, cache: CachePolicy) -> Option<(UserAddr, u64)> { + assert!(phys & (PAGE_2M - 1) == 0, "alloc_and_map: phys {phys:#x} not 2MB-aligned"); + let (addr, aligned) = self.regions.alloc_mapped(size)?; + self.map_range(addr, phys, aligned, prot, cache); + Some((addr, aligned)) } - pub fn insert_region(&mut self, _addr: UserAddr, _region: Region) { - match self.never {} + pub fn free_and_unmap(&mut self, addr: UserAddr) -> Option { + let size = self.regions.remove(addr)?; + self.unmap_range(addr, size); + Some(size) } - pub fn find_region(&self, _addr: UserAddr) -> Option<(UserAddr, &Region)> { - match self.never {} + pub fn insert_region(&mut self, addr: UserAddr, region: Region) { + self.regions.insert(addr, region); } - pub fn occupancy(&self, _addr: UserAddr, _size: u64) -> Occupancy { - match self.never {} + pub fn find_region(&self, addr: UserAddr) -> Option<(UserAddr, &Region)> { + self.regions.find(addr) } - pub fn overlapping_regions( - &self, - _start: UserAddr, - _end: UserAddr, - ) -> impl Iterator { - match self.never {} - #[allow(unreachable_code)] - core::iter::empty() + pub fn occupancy(&self, addr: UserAddr, size: u64) -> Occupancy { + self.regions.occupancy(addr, size) } - pub fn direct_map_policy(&self, _phys: u64) -> Option { - match self.never {} + pub fn overlapping_regions(&self, start: UserAddr, end: UserAddr) -> impl Iterator { + self.regions.overlapping(start, end) } - pub fn user_policy(&self, _addr: UserAddr) -> Option { - match self.never {} + /// The direct map's type for `phys`, read off the kernel's tables, which + /// every space shares. + pub fn direct_map_policy(&self, phys: u64) -> Option { + high().as_ref()?.leaf(DirectMap::from_phys(phys).as_ptr::() as u64).map(|(leaf, _)| policy_of(leaf)) } - pub fn guard_4k(&mut self, _phys: u64) { - match self.never {} + pub fn user_policy(&self, addr: UserAddr) -> Option { + self.tables.leaf(addr.raw()).map(|(leaf, _)| policy_of(leaf)) } + + /// Take the 4 KiB page at `phys` out of the direct map, for good: the + /// caller owns it for the machine's life. The 2 MiB page around it is + /// split break-before-make, so it must be one nothing is running from. + pub fn guard_4k(&mut self, phys: u64) { + assert!(phys & (PAGE_4K - 1) == 0, "guard_4k: phys {phys:#x} not 4 KiB-aligned"); + let va = DirectMap::from_phys(phys).as_ptr::() as u64; + let mut high = high(); + let high = high.as_mut().expect("guard_4k: before the kernel's tables exist"); + let i = index(va, 2); + let entry = high.directory(va).0[i]; + assert!(entry & VALID != 0, "guard_4k: {phys:#x} is not in the direct map"); + if entry & TABLE == 0 { + let base = entry & ADDR_2M; + let mut pages = Table::new(); + for (j, page) in pages.0.iter_mut().enumerate() { + *page = (base + j as u64 * PAGE_4K) | attributes(entry) | TABLE; + } + let pages = high.adopt(pages); + let directory = high.directory(va); + directory.set(i, 0); + tlb::kernel_page(va); + directory.set(i, pages | TABLE | VALID); + } + let pages = high.directory(va).0[i] & ADDR; + // SAFETY: the table the kernel's root linked, just above or before. + let pages = unsafe { Table::at_mut(pages) }; + let j = index(va, 3); + assert!(pages.0[j] & VALID != 0, "guard_4k: {phys:#x} is already unmapped"); + pages.set(j, 0); + tlb::kernel_page(va); + } +} + +/// The kernel's own tables, `TTBR1_EL1`'s. +static HIGH: Lock> = Lock::new(None); + +fn high() -> LockGuard<'static, Option> { + HIGH.lock() } +/// The kernel's address space: the empty user half a kernel thread and an +/// idle CPU run under, ASID 0. Leaked, since it outlives every task. +static KERNEL: core::sync::atomic::AtomicPtr>> = + core::sync::atomic::AtomicPtr::new(core::ptr::null_mut()); + +/// Its root, cached for lock-free access from panic and crash paths. +static KERNEL_TTBR0: core::sync::atomic::AtomicU64 = core::sync::atomic::AtomicU64::new(0); + pub fn kernel() -> &'static alloc::sync::Arc> { - owed!("the kernel's page tables", "stage 4") + let ptr = KERNEL.load(core::sync::atomic::Ordering::Acquire); + assert!(!ptr.is_null(), "paging not initialized"); + // SAFETY: written once in `init` with the `Release` this pairs with, never + // cleared, so the pointer is live for the machine's life. + unsafe { &*ptr } } pub fn kernel_root() -> Root { - owed!("the kernel's page tables", "stage 4") + Root(KERNEL_TTBR0.load(core::sync::atomic::Ordering::Relaxed)) } +/// Leave whatever user half is current for the kernel's empty one. pub fn activate_kernel() { - owed!("the kernel's page tables", "stage 4") + // SAFETY: the kernel space's root is an empty table that lives forever. + unsafe { kernel_root().activate() }; } -pub fn map_mmio(_phys: u64, _size: u64, _policy: MmioPolicy) -> crate::mm::Mmio { - owed!("the kernel's page tables", "stage 4") +/// Map a device's registers, or the scanout, into the direct map: 4 KiB pages +/// exactly over `[phys, phys + size)`, a block where a whole 2 MiB page is +/// asked for and free. No invalidation is owed, since only an invalid entry is +/// ever written. +pub fn map_mmio(phys: u64, size: u64, policy: MmioPolicy) -> crate::mm::Mmio { + high().as_mut().expect("map_mmio: before the kernel's tables exist").map_direct(phys, size, kernel_leaf(policy.cache())); + let installed = kernel().lock().direct_map_policy(phys).expect("map_mmio: the window was just mapped"); + assert!(installed == policy.cache(), "map_mmio: {phys:#x} installed {installed:?}"); + crate::log!("mmio: {phys:#x}+{size:#x} {installed:?}"); + crate::mm::Mmio::new(DirectMap::from_phys(phys), size) } -pub(crate) fn init(_memory_map: &[MemoryMapEntry]) -> toyos_bootmap::DirectMapEnd { - owed!("the kernel's page tables", "stage 4") -} +/// Build the kernel's tables — every page of memory firmware's map names, +/// and the console UART and the scanout the boot's own records reach — +/// switch `TTBR1_EL1` to them and `TTBR0_EL1` to the kernel's empty user +/// half, and answer where the direct map ends. +pub(crate) fn init(memory_map: &[MemoryMapEntry], scanout: Option<(u64, u64)>) -> toyos_bootmap::DirectMapEnd { + let extent = direct_map_end(memory_map).unwrap_or_else(|refusal| panic!("paging: firmware's memory map: {refusal}")); + let end = extent.get(); + let memory = kernel_leaf(CachePolicy::Normal); + let mut tables = Tables::new(); + let (mut blocks, mut pages) = (0u64, 0u64); + let mut at = 0; + while at < end { + let va = PHYS_OFFSET + at; + match coverage(memory_map, at) { + Coverage::Nothing => {} + Coverage::Whole => { + tables.directory(va).set(index(va, 2), at | memory); + blocks += 1; + } + held @ Coverage::Pages(_) => { + let mut table = Table::new(); + for j in (0..512).filter(|&j| held.holds(j)) { + table.0[j as usize] = (at + j * PAGE_4K) | memory | TABLE; + pages += 1; + } + let table = tables.adopt(table); + tables.directory(va).set(index(va, 2), table | TABLE | VALID); + } + } + at += PAGE_2M; + } + if let Some(uart) = super::console_uart::frame() { + tables.map_direct(uart, PAGE_4K, kernel_leaf(CachePolicy::Uncacheable)); + } + if let Some((scanout, size)) = scanout { + tables.map_direct(scanout, size, kernel_leaf(CachePolicy::WriteCombining)); + } -pub(crate) fn seal_kernel_half() { - owed!("the kernel's page tables", "stage 4") + let ttbr1 = tables.root.phys(); + *high() = Some(tables); + let space = AddressSpace { tables: Tables::new(), regions: vma::Regions::default(), asid: Asid::Kernel }; + let ttbr0 = space.root(); + KERNEL_TTBR0.store(ttbr0.0, core::sync::atomic::Ordering::Release); + let published: &'static alloc::sync::Arc> = + Box::leak(Box::new(alloc::sync::Arc::new(Lock::new(space)))); + KERNEL.store(published as *const _ as *mut _, core::sync::atomic::Ordering::Release); + + // SAFETY: both tables are built and published above, and the new + // `TTBR1_EL1` maps the code, stack and data this runs on exactly as the + // loader's did; the loader's identity view goes with the old `TTBR0_EL1`, + // and the local `TLBI` drops every translation either left behind. + unsafe { + core::arch::asm!( + "dsb ishst", + "msr ttbr1_el1, {ttbr1}", + "msr ttbr0_el1, {ttbr0}", + "isb", + "tlbi vmalle1", + "dsb nsh", + "isb", + ttbr1 = in(reg) ttbr1, + ttbr0 = in(reg) ttbr0.0, + options(nostack, preserves_flags), + ); + } + crate::log!("paging: the direct map holds memory below {end:#x} in {blocks} 2 MiB blocks and {pages} 4 KiB pages"); + extent } +/// Nothing to seal: every space runs under the one `TTBR1_EL1`, so no space +/// holds a copy of the kernel's root slots to fall behind. +pub(crate) fn seal_kernel_half() {} + /// `PAR_EL1` after the MMU translated `addr` for an EL1 read in the tables /// this CPU runs on now (`AT S1E1R`; Arm ARM K.a, C6.2.x and D24.2.131): /// bit 0 set is a fault, and otherwise bits 63:56 are the memory type. @@ -181,7 +722,7 @@ pub fn present_in_current_tables(addr: u64) -> bool { /// loader maps it so. Asked of the MMU page by page rather than assumed, and /// nothing is retyped: the type the loader wrote is the final one. pub fn boot_map_write_combining(phys: u64, size: u64) -> bool { - [phys, crate::mm::PHYS_OFFSET + phys].iter().all(|&base| { + [phys, PHYS_OFFSET + phys].iter().all(|&base| { (0..size.div_ceil(4096)).all(|page| { let par = translate_read(base + page * 4096); // Outer Normal non-cacheable, and an inner nibble that says the same: @@ -194,9 +735,99 @@ pub fn boot_map_write_combining(phys: u64, size: u64) -> bool { }) } -/// What memory type the scanout is mapped with, for the GPU's report. -pub fn scanout_memory_type(_addr: u64, _size: u64) -> impl core::fmt::Display { - owed!("the kernel's page tables", "stage 4"); - #[allow(unreachable_code)] - "" +/// What memory type the scanout is mapped with, for the GPU's report: the +/// MMU's own answer for its first byte in the direct map. +pub fn scanout_memory_type(addr: u64, _size: u64) -> impl core::fmt::Display { + struct Report(u64); + impl core::fmt::Display for Report { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + match self.0 & 1 { + 0 => write!(f, "PAR_EL1.ATTR {:#04x}", self.0 >> 56), + _ => write!(f, "untranslated (PAR_EL1 {:#x})", self.0), + } + } + } + Report(translate_read(PHYS_OFFSET + addr)) +} + +/// Each level's descriptor for `addr` in the tables this CPU runs under now, +/// read without a lock for a crash report. +pub fn debug_page_walk(addr: u64) { + let (name, ttbr) = if addr >> 63 == 0 { + let ttbr: u64; + // SAFETY: reads `TTBR0_EL1`. + unsafe { core::arch::asm!("mrs {}, ttbr0_el1", out(reg) ttbr, options(nomem, nostack, preserves_flags)) }; + ("TTBR0_EL1", ttbr) + } else { + let ttbr: u64; + // SAFETY: reads `TTBR1_EL1`. + unsafe { core::arch::asm!("mrs {}, ttbr1_el1", out(reg) ttbr, options(nomem, nostack, preserves_flags)) }; + ("TTBR1_EL1", ttbr) + }; + crate::log!(" Page walk for {addr:#x} through {name}={ttbr:#x}:"); + let mut table = ttbr & ADDR; + for level in 0..4 { + // SAFETY: a table the live translation register or a valid table + // descriptor above it names, which the MMU is walking now. + let entry = unsafe { Table::at(table) }.0[index(addr, level)]; + crate::log!(" L{level}[{}] = {entry:#018x}", index(addr, level)); + if entry & VALID == 0 || level == 3 || entry & TABLE == 0 { + return; + } + table = entry & ADDR; + } +} + +/// The word at user address `addr` in the space this CPU runs under now, +/// read through the direct map by a walk that takes no lock and faults +/// nowhere: for a crash report, which may hold any lock already. +pub(super) fn read_user_word(addr: u64) -> Option { + if !addr.is_multiple_of(8) || !toyos_userbound::is_user_addr(addr) { + return None; + } + let ttbr0: u64; + // SAFETY: reads `TTBR0_EL1`. + unsafe { core::arch::asm!("mrs {}, ttbr0_el1", out(reg) ttbr0, options(nomem, nostack, preserves_flags)) }; + let mut table = ttbr0 & ADDR; + for level in 0..4 { + // SAFETY: the live `TTBR0_EL1`'s table or one a valid table + // descriptor above names, which the MMU walks now. + let entry = unsafe { Table::at(table) }.0[index(addr, level)]; + if entry & VALID == 0 { + return None; + } + let size = match (level, entry & TABLE != 0) { + (3, _) => PAGE_4K, + (2, false) => PAGE_2M, + (_, true) => { + table = entry & ADDR; + continue; + } + (_, false) => return None, + }; + let base = entry & if size == PAGE_2M { ADDR_2M } else { ADDR }; + // SAFETY: a byte of a frame a valid leaf maps, reached through the direct map. + return Some(unsafe { core::ptr::read_volatile(DirectMap::from_phys(base + (addr & (size - 1))).as_ptr::()) }); + } + None +} + +/// Whether a process can be given the scanout at `[phys, phys + size)`, +/// which it is in 2 MiB pages: its base on one, and nothing the last page +/// covers past it memory the direct map holds — firmware carves a scanout +/// like `ramfb`'s out of RAM, and the rest of that page is somebody else's. +pub fn scanout_whole_pages(phys: u64, size: u64) -> Result<(), &'static str> { + if !phys.is_multiple_of(PAGE_2M) { + return Err("its base is not on a 2 MiB page, and a process is given a scanout in 2 MiB pages"); + } + let high = high(); + let tables = high.as_ref().expect("scanout_whole_pages: before the kernel's tables exist"); + let mut at = (phys + size).next_multiple_of(PAGE_4K); + while at < (phys + size).next_multiple_of(PAGE_2M) { + if tables.leaf(PHYS_OFFSET + at).is_some_and(|(leaf, _)| policy_of(leaf) == CachePolicy::Normal) { + return Err("its last 2 MiB page holds memory past it, which a process given the page would reach"); + } + at += PAGE_4K; + } + Ok(()) } diff --git a/kernel/src/arch/aarch64/percpu.rs b/kernel/src/arch/aarch64/percpu.rs index 065865a077..5d7a8b8f30 100644 --- a/kernel/src/arch/aarch64/percpu.rs +++ b/kernel/src/arch/aarch64/percpu.rs @@ -1,8 +1,18 @@ -//! Per-CPU state. On AArch64 it is reached through `TPIDR_EL1`, and the block -//! it names is built by the port's stage 5 (other CPUs) on top of stage 4's -//! exception entry; until then every accessor is owed. The boot never reaches -//! one: the log and the panic path read [`crate::log::PERCPU_READY`] first. +//! Per-CPU state, reached through `TPIDR_EL1`, which holds this CPU's +//! [`PerCpu`] from [`init_bsp`] on and which nothing else writes. EL0 cannot +//! read it, so a thread learns nothing of the kernel's layout from it. +//! +//! Every field another context on the same CPU can write — an interrupt, a +//! nested exception — is an atomic, so the whole block is only ever reached +//! through a shared reference, and an increment is one exclusive-monitor loop +//! an exception on this CPU cannot split (Arm ARM K.a, B2.17.5: taking an +//! exception clears the monitor). +use core::sync::atomic::{AtomicU32, AtomicU64, AtomicU8, Ordering::Relaxed}; + +use alloc::boxed::Box; + +use crate::log; use crate::process::{Pid, Tid}; /// Per-CPU fault state machine for the escalation policy on nested faults. @@ -15,100 +25,300 @@ pub enum CpuFaultState { Panic = 3, } +/// One CPU's block. Only this CPU writes it, bar [`irq_counts`](Self::irq_counts), +/// which `irq_census` reads from any. +pub struct PerCpu { + cpu_id: u32, + /// `u32::MAX` when no thread runs here. + current_tid: AtomicU32, + current_pid: AtomicU32, + preempt_count: AtomicU32, + need_resched: AtomicU8, + fault_state: AtomicU8, + /// Timer interrupts taken at EL1, which only count and ask for a pass. + kernel_timer_fires: AtomicU32, + last_seen_kernel_timer_fires: AtomicU32, + /// Counter ticks the timer was last armed for: what a timer interrupt + /// taken at EL1 re-arms it with, and zero when it is stopped. + armed_ticks: AtomicU64, + /// The kernel stack a switch last installed, for the stack witness: on + /// AArch64 an entry from EL0 lands on `SP_EL1` as the last `ERET` left it, + /// and nothing here points it anywhere. + kernel_stack: AtomicU64, + idle_stack_top: u64, + /// What the last entry by `SVC` said, for a crash report: its return + /// address, number, frame pointer and stack pointer. + syscall_pc: AtomicU64, + syscall_num: AtomicU64, + syscall_fp: AtomicU64, + syscall_sp: AtomicU64, + /// The task whose syscall this CPU is inside: packed pid:tid, or [`NO_SYSCALL`]. + syscall_task: AtomicU64, + /// This CPU's [`log::Shard`]: the boot shard on CPU 0. + log_shard: &'static log::Shard, + /// One counter per `irq_census::Source`, and the total. + irq_counts: [AtomicU64; crate::irq_census::SLOTS], +} + +/// This CPU's block. +#[inline] +fn this() -> &'static PerCpu { + let block: u64; + // SAFETY: reads `TPIDR_EL1`, which [`init_bsp`] points at a leaked + // `PerCpu` before any accessor here runs (`log::PERCPU_READY` gates them). + unsafe { + core::arch::asm!("mrs {}, tpidr_el1", out(reg) block, options(nomem, nostack, preserves_flags)); + &*(block as *const PerCpu) + } +} + +/// Build the boot CPU's block and publish it through `TPIDR_EL1`; after it, +/// every accessor here answers. The thread registers EL0 can read are cleared, +/// so a thread's first read of one finds nothing firmware left. +pub fn init_bsp() { + let block: &'static PerCpu = Box::leak(Box::new(PerCpu { + cpu_id: 0, + current_tid: AtomicU32::new(u32::MAX), + current_pid: AtomicU32::new(u32::MAX), + preempt_count: AtomicU32::new(0), + need_resched: AtomicU8::new(0), + fault_state: AtomicU8::new(CpuFaultState::Normal as u8), + kernel_timer_fires: AtomicU32::new(0), + last_seen_kernel_timer_fires: AtomicU32::new(0), + armed_ticks: AtomicU64::new(0), + kernel_stack: AtomicU64::new(0), + idle_stack_top: crate::sched::idle_stack::alloc(), + syscall_pc: AtomicU64::new(0), + syscall_num: AtomicU64::new(0), + syscall_fp: AtomicU64::new(0), + syscall_sp: AtomicU64::new(0), + syscall_task: AtomicU64::new(NO_SYSCALL), + log_shard: &log::BOOT_SHARD, + irq_counts: [const { AtomicU64::new(0) }; crate::irq_census::SLOTS], + })); + crate::irq_census::publish(0, block.irq_counts.as_ptr()); + // SAFETY: the block lives for the machine's life; `TPIDRRO_EL0` and + // `TPIDR_EL0` are EL0's to read and hold nothing of the kernel's. + unsafe { + core::arch::asm!( + "msr tpidr_el1, {}", + "msr tpidrro_el0, xzr", + "msr tpidr_el0, xzr", + in(reg) block as *const PerCpu as u64, + options(nostack, preserves_flags), + ); + } + crate::log::PERCPU_READY.store(true, core::sync::atomic::Ordering::Release); + log!("percpu: BSP cpu_id=0 mpidr={:#x}", super::cpu::hardware_id()); +} + pub fn cpu_id() -> u32 { - owed!("per-CPU state", "stage 4") + this().cpu_id } +/// `None` means idle. pub fn current_tid() -> Option { - owed!("per-CPU state", "stage 4") + match this().current_tid.load(Relaxed) { + u32::MAX => None, + raw => Some(Tid::from_raw(raw)), + } } -pub fn set_current_tid(_tid: Option) { - owed!("per-CPU state", "stage 4") +pub fn set_current_tid(tid: Option) { + this().current_tid.store(tid.map_or(u32::MAX, |t| t.raw()), Relaxed); } +/// `None` means idle. pub fn current_pid() -> Option { - owed!("per-CPU state", "stage 4") + match this().current_pid.load(Relaxed) { + u32::MAX => None, + raw => Some(Pid::from_raw(raw)), + } } -pub fn set_current_pid(_pid: Option) { - owed!("per-CPU state", "stage 4") +pub fn set_current_pid(pid: Option) { + this().current_pid.store(pid.map_or(u32::MAX, |p| p.raw()), Relaxed); } +/// Record the stack the next entry from EL0 lands on: the one a switch just +/// installed, which `SP_EL1` already is. /// # Safety -/// `top` is the top of the kernel stack the next entry from user mode lands on. -pub unsafe fn set_kernel_stack(_top: u64) { - owed!("per-CPU state", "stage 4") +/// Called on the CPU whose block it is. +pub unsafe fn set_kernel_stack(top: u64) { + this().kernel_stack.store(top, Relaxed); } +/// The stack an entry from EL0 lands on, twice: AArch64 has one where x86-64 +/// has `kernel_rsp` and `tss.rsp0`. /// # Safety /// Read on the CPU whose entry stacks they are. +#[cfg(feature = "stack-witness")] pub unsafe fn entry_stacks() -> (u64, u64) { - owed!("per-CPU state", "stage 4") + let top = this().kernel_stack.load(Relaxed); + (top, top) } pub fn idle_stack_top() -> u64 { - owed!("per-CPU state", "stage 4") + this().idle_stack_top } +/// The last byte of this CPU's idle guard page — the first byte an overflow reaches. #[cfg(feature = "test-actuators")] pub fn idle_guard_byte() -> u64 { - owed!("per-CPU state", "stage 4") + idle_stack_top() - crate::sched::idle_stack::SIZE as u64 - 1 } +/// How big one idle stack is; read by `SYS_DEBUG` for scale. #[cfg(feature = "test-actuators")] pub fn idle_stack_size() -> usize { - owed!("per-CPU state", "stage 4") + crate::sched::idle_stack::SIZE } #[cfg(feature = "test-actuators")] pub fn idle_stack_high_water() -> usize { - owed!("per-CPU state", "stage 4") + crate::sched::idle_stack::high_water() +} + +/// No task on this CPU is inside a syscall; [`pack_task`] never produces this value. +const NO_SYSCALL: u64 = u64::MAX; + +fn pack_task(pid: u32, tid: u32) -> u64 { + (u64::from(pid) << 32) | u64::from(tid) +} + +/// Enter this CPU's syscall bracket, with what the entry saw. `super::trap` is the only caller. +pub(super) fn enter_syscall(pc: u64, num: u64, fp: u64, sp: u64) { + let block = this(); + block.syscall_pc.store(pc, Relaxed); + block.syscall_num.store(num, Relaxed); + block.syscall_fp.store(fp, Relaxed); + block.syscall_sp.store(sp, Relaxed); + block.syscall_task.store(pack_task(block.current_pid.load(Relaxed), block.current_tid.load(Relaxed)), Relaxed); +} + +/// …and leave it. +pub(super) fn leave_syscall() { + this().syscall_task.store(NO_SYSCALL, Relaxed); +} + +/// Whether the task this CPU runs is inside a syscall now. +pub(super) fn in_syscall() -> bool { + let block = this(); + let recorded = block.syscall_task.load(Relaxed); + recorded != NO_SYSCALL + && recorded == pack_task(block.current_pid.load(Relaxed), block.current_tid.load(Relaxed)) +} + +/// The last syscall's return address, frame pointer and stack pointer; +/// meaningful only while [`in_syscall`] holds. +pub(super) fn syscall_context() -> (u64, u64, u64) { + let block = this(); + (block.syscall_pc.load(Relaxed), block.syscall_fp.load(Relaxed), block.syscall_sp.load(Relaxed)) } pub fn syscall_num() -> u64 { - owed!("per-CPU state", "stage 4") + this().syscall_num.load(Relaxed) +} + +/// Not atomic as a pair, and needs not be: only exception and panic entry +/// touch it, with interrupts masked. +pub fn swap_fault_state(new: CpuFaultState) -> CpuFaultState { + match this().fault_state.swap(new as u8, Relaxed) { + 0 => CpuFaultState::Normal, + 1 => CpuFaultState::PageFault, + 2 => CpuFaultState::Fatal, + _ => CpuFaultState::Panic, + } } -pub fn swap_fault_state(_new: CpuFaultState) -> CpuFaultState { - owed!("per-CPU state", "stage 4") +pub fn set_fault_state(new: CpuFaultState) { + this().fault_state.store(new as u8, Relaxed); } /// This CPU's shard, its identity, and one sequence number out of that shard. -pub fn reserve_log_slot( - _guard: &crate::arch::IrqGuard, -) -> (*const crate::log::Shard, u64, u32, u32, u32) { - owed!("per-CPU state", "stage 4") +pub fn reserve_log_slot(guard: &crate::arch::IrqGuard) -> (*const log::Shard, u64, u32, u32, u32) { + let block = this(); + let (shard, cpu, tid, pid) = + (block.log_shard, block.cpu_id, block.current_tid.load(Relaxed), block.current_pid.load(Relaxed)); + // `log-nested-reserve`'s injection point: between the shard read and the reservation. + crate::log::nested::reserve_window(); + // SAFETY: `guard` masks this CPU, the only one that reserves in its shard. + let seq = unsafe { shard.reserve(guard) }; + (shard, seq, cpu, tid, pid) +} + +/// One delivery of `source`, counted in this CPU's block. +pub(super) fn irq_took(source: crate::irq_census::Source) { + let counts = &this().irq_counts; + counts[crate::irq_census::TOTAL].fetch_add(1, Relaxed); + counts[1 + source as usize].fetch_add(1, Relaxed); } -pub fn irq_counts_here(_first: usize, _second: usize) -> (u64, u64) { - owed!("per-CPU state", "stage 4") +/// Two of this CPU's interrupt counters. +pub fn irq_counts_here(first: usize, second: usize) -> (u64, u64) { + let counts = &this().irq_counts; + (counts[first].load(Relaxed), counts[second].load(Relaxed)) } +#[inline] pub fn preempt_count() -> u32 { - owed!("per-CPU state", "stage 4") + this().preempt_count.load(Relaxed) } -pub fn set_preempt_count(_value: u32) { - owed!("per-CPU state", "stage 4") +#[inline] +pub fn set_preempt_count(value: u32) { + this().preempt_count.store(value, Relaxed); } +/// One increment, atomic against an interrupt on this CPU. +#[inline] pub fn preempt_count_up() { - owed!("per-CPU state", "stage 4") + this().preempt_count.fetch_add(1, Relaxed); } +#[inline] pub fn preempt_count_down() { - owed!("per-CPU state", "stage 4") + this().preempt_count.fetch_sub(1, Relaxed); } +#[inline] pub fn resched_owed() -> bool { - owed!("per-CPU state", "stage 4") + this().need_resched.load(Relaxed) != 0 } -pub fn set_resched_owed(_owed: bool) { - owed!("per-CPU state", "stage 4") +#[inline] +pub fn set_resched_owed(owed: bool) { + this().need_resched.store(u8::from(owed), Relaxed); } +/// Whether this CPU is inside a fault or panic. +#[inline] pub fn faulting() -> bool { - owed!("per-CPU state", "stage 4") + this().fault_state.load(Relaxed) != 0 +} + +/// Timer interrupts taken at EL1. +pub fn kernel_timer_fires() -> u32 { + this().kernel_timer_fires.load(Relaxed) +} + +pub fn last_seen_kernel_timer_fires() -> u32 { + this().last_seen_kernel_timer_fires.load(Relaxed) +} + +pub fn set_last_seen_kernel_timer_fires(v: u32) { + this().last_seen_kernel_timer_fires.store(v, Relaxed); +} + +pub(super) fn note_kernel_timer_fire() { + this().kernel_timer_fires.fetch_add(1, Relaxed); +} + +/// What the timer was last armed for, in counter ticks; zero is stopped. +pub(super) fn armed_ticks() -> u64 { + this().armed_ticks.load(Relaxed) +} + +pub(super) fn set_armed_ticks(ticks: u64) { + this().armed_ticks.store(ticks, Relaxed); } diff --git a/kernel/src/arch/aarch64/pmu.rs b/kernel/src/arch/aarch64/pmu.rs index f9697ad93a..c42ac5435f 100644 --- a/kernel/src/arch/aarch64/pmu.rs +++ b/kernel/src/arch/aarch64/pmu.rs @@ -1,20 +1,21 @@ //! The performance monitor's overflow, as the hard-lockup detector's sample //! source. On AArch64 that is `PMCCNTR_EL0` overflowing into a pseudo-NMI — -//! an interrupt the GICv3 delivers at a priority `DAIF.I` does not mask — -//! which needs the interrupt controller of the port's stage 4. - +//! an interrupt at a priority `DAIF.I` does not mask, which needs every +//! interrupt mask in this kernel moved from `DAIF` to `ICC_PMR_EL1` — and no +//! stage of the port has taken that on. Only a boot under a deadline arms it, +//! and such a boot is refused here by name. /// Start this CPU's counter overflowing into an NMI every `period` counter ticks. pub fn arm(_period: u64) -> bool { - owed!("the PMU overflow NMI", "stage 4") + owed!("the PMU overflow NMI", "no stage yet") } /// Whether this CPU's counter is the reason this NMI arrived. pub fn overflowed() -> bool { - owed!("the PMU overflow NMI", "stage 4") + owed!("the PMU overflow NMI", "no stage yet") } /// After an overflow's sample: clear it and reload. pub fn rearm(_period: u64) { - owed!("the PMU overflow NMI", "stage 4") + owed!("the PMU overflow NMI", "no stage yet") } diff --git a/kernel/src/arch/aarch64/smp.rs b/kernel/src/arch/aarch64/smp.rs index f52708238e..a46829cd40 100644 --- a/kernel/src/arch/aarch64/smp.rs +++ b/kernel/src/arch/aarch64/smp.rs @@ -1,18 +1,20 @@ //! Other CPUs: PSCI `CPU_ON` for each GICC the MADT names, the port's stage 5. -//! Until then the boot CPU is the only one running, and the machine is never -//! released to the scheduler. +//! Until then the boot CPU is the only one the roster holds. +use crate::smp_roster::Roster; + +static ROSTER: Roster = Roster::new(); -/// CPUs running: the boot CPU alone. pub fn cpu_count() -> u32 { - 1 + ROSTER.count() } +/// Release the machine to the scheduler: with one CPU, nothing waits on it. pub fn set_ready() { - owed!("other CPUs", "stage 5") + ROSTER.release(); } -/// Never: the boot ends before the machine is released. +/// Whether [`set_ready`] has run. pub fn is_ready() -> bool { - false + ROSTER.released() } diff --git a/kernel/src/arch/aarch64/switch.rs b/kernel/src/arch/aarch64/switch.rs new file mode 100644 index 0000000000..c7f98cac72 --- /dev/null +++ b/kernel/src/arch/aarch64/switch.rs @@ -0,0 +1,106 @@ +//! `context_switch`: what a context stands on, saved onto the outgoing stack +//! and restored off the incoming one — the callee-saved registers, `DAIF`, +//! and the thread's FP/SIMD state ([`super::fpu`]), which travels with the +//! stack of the thread it belongs to. +//! +//! The frame, from the saved stack pointer up: x19–x28, x29 and `DAIF`, x30 +//! and a pad word, then [`super::fpu::STATE_BYTES`] of FP/SIMD state — the +//! general registers first, because a pair's store reaches only 504 bytes. + +use core::arch::naked_asm; + +use super::fpu::STATE_BYTES; + +/// Where `DAIF` is saved, beside x29. +pub const DAIF_AT: usize = 88; +/// Where x30 is saved: the address the frame returns to. +pub const RETURN_AT: usize = 96; +/// Where the FP/SIMD state starts. +pub const FP_AT: usize = 112; +/// One saved context's bytes. +pub const FRAME_BYTES: usize = FP_AT + STATE_BYTES; + +const _: () = assert!(FRAME_BYTES.is_multiple_of(16)); + +/// Save this context's frame below the stack pointer, store that pointer at +/// `old_sp`, and return into the frame at `new_sp`. +/// # Safety +/// `old_sp` is the outgoing context's save slot and `new_sp` a frame this +/// function or `super::entry::initial_frame` wrote, on a stack that is live. +#[unsafe(naked)] +pub(crate) unsafe extern "C" fn context_switch(old_sp: *mut u64, new_sp: u64) { + naked_asm!( + // The kernel is soft-float: the assembler is told these are here for this block alone. + ".arch_extension fp", + ".arch_extension simd", + "sub sp, sp, #{frame}", + "stp x19, x20, [sp, #0]", + "stp x21, x22, [sp, #16]", + "stp x23, x24, [sp, #32]", + "stp x25, x26, [sp, #48]", + "stp x27, x28, [sp, #64]", + "mrs x9, daif", + "stp x29, x9, [sp, #80]", + "str x30, [sp, #{ret}]", + "add x9, sp, #{fp}", + "stp q0, q1, [x9, #0]", + "stp q2, q3, [x9, #32]", + "stp q4, q5, [x9, #64]", + "stp q6, q7, [x9, #96]", + "stp q8, q9, [x9, #128]", + "stp q10, q11, [x9, #160]", + "stp q12, q13, [x9, #192]", + "stp q14, q15, [x9, #224]", + "stp q16, q17, [x9, #256]", + "stp q18, q19, [x9, #288]", + "stp q20, q21, [x9, #320]", + "stp q22, q23, [x9, #352]", + "stp q24, q25, [x9, #384]", + "stp q26, q27, [x9, #416]", + "stp q28, q29, [x9, #448]", + "stp q30, q31, [x9, #480]", + "mrs x10, fpcr", + "str x10, [x9, #512]", + "mrs x10, fpsr", + "str x10, [x9, #520]", + "mov x9, sp", + "str x9, [x0]", + "mov sp, x1", + "add x9, sp, #{fp}", + "ldp q0, q1, [x9, #0]", + "ldp q2, q3, [x9, #32]", + "ldp q4, q5, [x9, #64]", + "ldp q6, q7, [x9, #96]", + "ldp q8, q9, [x9, #128]", + "ldp q10, q11, [x9, #160]", + "ldp q12, q13, [x9, #192]", + "ldp q14, q15, [x9, #224]", + "ldp q16, q17, [x9, #256]", + "ldp q18, q19, [x9, #288]", + "ldp q20, q21, [x9, #320]", + "ldp q22, q23, [x9, #352]", + "ldp q24, q25, [x9, #384]", + "ldp q26, q27, [x9, #416]", + "ldp q28, q29, [x9, #448]", + "ldp q30, q31, [x9, #480]", + "ldr x10, [x9, #512]", + "msr fpcr, x10", + "ldr x10, [x9, #520]", + "msr fpsr, x10", + "ldp x19, x20, [sp, #0]", + "ldp x21, x22, [sp, #16]", + "ldp x23, x24, [sp, #32]", + "ldp x25, x26, [sp, #48]", + "ldp x27, x28, [sp, #64]", + "ldp x29, x9, [sp, #80]", + "ldr x30, [sp, #{ret}]", + "add sp, sp, #{frame}", + "msr daif, x9", + "ret", + ".arch_extension nosimd", + ".arch_extension nofp", + frame = const FRAME_BYTES, + fp = const FP_AT, + ret = const RETURN_AT, + ); +} diff --git a/kernel/src/arch/aarch64/syscall.rs b/kernel/src/arch/aarch64/syscall.rs index 667d1f82fc..a8c05ecb96 100644 --- a/kernel/src/arch/aarch64/syscall.rs +++ b/kernel/src/arch/aarch64/syscall.rs @@ -1,12 +1,11 @@ -//! The system-call gate: `SVC` from EL0 through the vectors' lower-EL entry, -//! the port's stage 7. +//! The system-call gate is an `SVC` from EL0, taken through the vectors' +//! lower-EL synchronous entry (`super::trap`), which switches to `SP_EL1` in +//! hardware: there is no window with a user stack under the kernel. - -/// A syscall entered this CPU; x86-64's NMI gate counts it, and AArch64 has no -/// such gate. +/// A syscall entered this CPU; x86-64's NMI gate counts it, and AArch64 has no such gate. pub fn note_entry() {} -/// The actuator that storms NMIs at the syscall window. +/// x86-64's storms NMIs at its `SYSCALL` window, which AArch64 has none of. pub fn window_storm() { - owed!("the syscall gate", "stage 7") + panic!("syscall-window-nmi: AArch64's SVC switches stacks in hardware, so there is no window to storm"); } diff --git a/kernel/src/arch/aarch64/tlb.rs b/kernel/src/arch/aarch64/tlb.rs index df6dec4b13..d46f19b02c 100644 --- a/kernel/src/arch/aarch64/tlb.rs +++ b/kernel/src/arch/aarch64/tlb.rs @@ -1,33 +1,109 @@ -//! TLB invalidation across CPUs. AArch64 broadcasts it in hardware (`TLBI -//! …IS`), so the shootdown this interface stands for is a local instruction -//! plus `DSB ISH` once the kernel owns its page tables, the port's stage 4. +//! TLB invalidation. AArch64 broadcasts it in hardware: a `TLBI …IS` reaches +//! every CPU in the inner-shareable domain, and the `DSB ISH` after it +//! returns once every one of them has dropped the entry (Arm ARM K.a, +//! D8.13.4). So nothing here sends an interrupt or waits for an +//! acknowledgement, and the page-table edit that clears an entry is the one +//! place its translation is dropped: [`shootdown`] has nothing left to do. +//! +//! Every operation below is bracketed the same way: `DSB ISHST` so the table +//! write it answers for is visible to every walker first, then the `TLBI`, +//! then `DSB ISH` for its completion and `ISB` so this CPU's next fetch walks +//! afresh. +use core::sync::atomic::{AtomicU64, Ordering}; use crate::invalidation::Origin; -pub fn log_census() { - owed!("TLB invalidation", "stage 4") +/// Issuer-side census, as x86-64's counts it; there is no receiver side. +static ISSUED: [AtomicU64; Origin::COUNT] = [const { AtomicU64::new(0) }; Origin::COUNT]; +/// Total at the last print; process exit logs once per batch. +static REPORTED: AtomicU64 = AtomicU64::new(0); + +macro_rules! tlbi { + ($op:literal, $operand:expr) => { + // SAFETY: TLB maintenance and barriers change no memory; dropping a + // translation only makes the next access walk the tables again. + unsafe { + core::arch::asm!( + "dsb ishst", + concat!("tlbi ", $op, ", {}"), + "dsb ish", + "isb", + in(reg) $operand, + options(nostack, preserves_flags), + ) + } + }; +} + +/// Every translation `asid` holds for the 4 KiB page at `va`, on every CPU. +pub(super) fn page(asid: u16, va: u64) { + tlbi!("vae1is", u64::from(asid) << 48 | (va >> 12) & 0xFFF_FFFF_FFFF); +} + +/// Every translation `asid` holds, on every CPU. +pub(super) fn asid(asid: u16) { + tlbi!("aside1is", u64::from(asid) << 48); } -pub fn shootdown(_origin: Origin) { - owed!("TLB invalidation", "stage 4") +/// A global (kernel) translation of the 4 KiB page at `va`, on every CPU. +pub(super) fn kernel_page(va: u64) { + tlbi!("vaae1is", (va >> 12) & 0xFFF_FFFF_FFFF); } -pub fn poll() { - owed!("TLB invalidation", "stage 4") +/// Every EL1&0 translation every CPU holds, dropped before this returns — +/// what a returned ASID needs before it is issued again. +pub(super) fn all(origin: Origin) { + ISSUED[origin as usize].fetch_add(1, Ordering::Relaxed); + // SAFETY: as `tlbi!`'s. + unsafe { + core::arch::asm!("dsb ishst", "tlbi vmalle1is", "dsb ish", "isb", options(nostack, preserves_flags)); + } +} + +/// Nothing to drop: every caller follows an `AddressSpace` edit whose own +/// `page` or `asid` already reached every CPU before it returned. +pub fn shootdown(_origin: Origin) {} + +/// Nothing to answer: no CPU waits on another's acknowledgement here. +pub fn poll() {} + +/// One `tlb:` line when the counts moved, at process exit. +pub fn log_census() { + let mut counts = [0u64; Origin::COUNT]; + for (slot, count) in ISSUED.iter().zip(counts.iter_mut()) { + *count = slot.load(Ordering::Relaxed); + } + let total: u64 = counts.iter().sum(); + if total == 0 || REPORTED.swap(total, Ordering::Relaxed) == total { + return; + } + struct Fields([u64; Origin::COUNT]); + impl core::fmt::Display for Fields { + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + for (name, count) in Origin::NAMES.iter().zip(self.0) { + write!(f, " {name}={count}")?; + } + Ok(()) + } + } + crate::log!("tlb: broadcast invalidations={total}{}", Fields(counts)); } +/// x86-64's measures an IPI round trip, and there is none here. #[cfg(feature = "boot-actuators")] pub fn bench() { - owed!("TLB invalidation", "stage 4") + panic!("tlb-shootdown-bench: AArch64 invalidates by broadcast, so there is no IPI round trip to measure"); } +/// x86-64's delays an acknowledgement, and there is none here: refused. #[cfg(feature = "test-actuators")] pub fn debug_arm_ack_delay(_nanos: u64) -> u64 { - owed!("TLB invalidation", "stage 4") + toyos_abi::syscall::SyscallError::NotSupported.to_u64() } +/// x86-64's delays an acknowledgement, and there is none here: refused. #[cfg(feature = "test-actuators")] pub fn debug_disarm_ack_delay() -> u64 { - owed!("TLB invalidation", "stage 4") + toyos_abi::syscall::SyscallError::NotSupported.to_u64() } diff --git a/kernel/src/arch/aarch64/trap.rs b/kernel/src/arch/aarch64/trap.rs index 4747673475..a8c433f5aa 100644 --- a/kernel/src/arch/aarch64/trap.rs +++ b/kernel/src/arch/aarch64/trap.rs @@ -1,16 +1,29 @@ -//! Exceptions: the vector table `VBAR_EL1` names, and what the kernel says -//! when one is taken. +//! Exceptions: the vector table `VBAR_EL1` names, and what each one taken +//! comes to. //! -//! Every exception is fatal until the port's stage 4 gives interrupts and -//! user faults somewhere to go: the entry saves the interrupted context into a -//! [`Frame`] on the stack it was taken on, and [`exception`] reports it and -//! panics. A table that does not reach [`exception`] — misaligned, or never +//! Every entry saves the interrupted context into a [`Frame`] on the stack +//! it was taken on — for an entry from EL0, `SP_EL1` as the last `ERET` left +//! it, the top of the running thread's kernel stack — and [`dispatch`] routes +//! it: an interrupt to [`irq`], an `SVC` to the syscall dispatcher, a +//! translation fault to the demand pager, anything else from EL0 to the end +//! of its process, and anything else from EL1 to a panic. Every return to EL0 +//! runs `scheduler::exit_to_user` last. The entry saves no FP/SIMD register: +//! the kernel never touches one, and a switch saves the thread's +//! (`super::switch`). +//! +//! A table that does not reach [`dispatch`] — misaligned, or never //! installed — is what the aarch64 boot test's fault arm catches, because //! then nothing reports at all. use core::fmt; +use core::sync::atomic::{AtomicU32, AtomicU64, Ordering::Relaxed}; + +use toyos_sched::hw::{CpuId, Machine, TraceEvent, TraceKind}; -use crate::log; +use super::percpu::{self, CpuFaultState}; +use super::{cpu, irqchip}; +use crate::irq_census::Source; +use crate::{alert, log}; /// The interrupted context, as the entry stores it: `x0`–`x30`, the stack /// pointer before the exception, then the four system registers that say @@ -28,9 +41,19 @@ pub struct Frame { const FRAME_BYTES: usize = core::mem::size_of::(); const _: () = assert!(FRAME_BYTES == 288 && FRAME_BYTES.is_multiple_of(16)); -/// Which of the table's sixteen entries was taken: Arm ARM K.a, D1.3.1, -/// Table D1-7 — four groups by where the exception came from, and in each the -/// synchronous, IRQ, FIQ and SError entries. +/// The table's entries this kernel acts on: Arm ARM K.a, D1.3.1, Table D1-7. +const EL1_SYNC: u64 = 4; +const EL1_IRQ: u64 = 5; +const EL0_SYNC: u64 = 8; +const EL0_IRQ: u64 = 9; + +/// `ESR_EL1.EC` of the classes an EL0 entry acts on. +const EC_SVC64: u64 = 0x15; +const EC_IABT_LOWER: u64 = 0x20; +const EC_DABT_LOWER: u64 = 0x24; + +/// Which of the table's sixteen entries was taken: four groups by where the +/// exception came from, and in each the synchronous, IRQ, FIQ and SError entries. #[derive(Clone, Copy)] struct Entry(u64); @@ -52,6 +75,7 @@ fn class_name(esr: u64) -> &'static str { 0x0E => "illegal execution state", 0x15 => "SVC from AArch64", 0x18 => "system register access trapped", + 0x19 => "SVE trapped", 0x20 => "instruction abort from a lower EL", 0x21 => "instruction abort", 0x22 => "PC alignment fault", @@ -64,15 +88,29 @@ fn class_name(esr: u64) -> &'static str { } } -/// The Rust half of every vector entry: report the context and panic. The -/// panic handler's early branch puts the report on the console and the panel. -extern "C" fn exception(frame: &Frame, entry: u64) -> ! { +/// The Rust half of every vector entry. Returns to the entry, which restores +/// the frame and returns from the exception; everything fatal diverges. +extern "C" fn dispatch(frame: &mut Frame, entry: u64) { + match entry { + EL1_IRQ => irq(false), + EL0_SYNC => { + el0_sync(frame); + crate::scheduler::exit_to_user(); + } + EL0_IRQ => { + irq(true); + crate::scheduler::exit_to_user(); + } + _ => exception(frame, entry), + } +} + +/// The report of an exception this kernel does not return from: its own, at +/// EL1, or an FIQ or SError from anywhere. The panic handler's branch puts it +/// on the console and the panel. +fn exception(frame: &Frame, entry: u64) -> ! { let entry = Entry(entry); - log!( - "KERNEL PANIC: {entry}: {} (ESR={:#010x})", - class_name(frame.esr), - frame.esr - ); + log!("KERNEL PANIC: {entry}: {} (ESR={:#010x})", class_name(frame.esr), frame.esr); log!(" elr={:#018x} far={:#018x} spsr={:#010x}", frame.elr, frame.far, frame.spsr); for pair in 0..15 { log!( @@ -84,14 +122,248 @@ extern "C" fn exception(frame: &Frame, entry: u64) -> ! { ); } log!(" x30={:#018x} sp ={:#018x}", frame.x[30], frame.sp); + if entry.0 == EL1_SYNC && matches!(frame.esr >> 26, 0x21 | 0x25) && crate::log::PERCPU_READY.load(Relaxed) { + super::paging::debug_page_walk(frame.far); + } crate::symbols::kernel_backtrace(frame.x[29], 20); panic!("{entry}: {} at {:#x}", class_name(frame.esr), frame.elr); } +/// One interrupt: acknowledged, handled, ended. From EL0 a tick or a kick +/// preempts here, where the interrupted context holds nothing; from EL1 it +/// only asks for the pass the context will run when it may. +fn irq(from_el0: bool) { + let Some(intid) = irqchip::acknowledge() else { + percpu::irq_took(Source::Spurious); + return; + }; + if intid == irqchip::timer_intid() { + // Before anything that can take a lock or panic: a timer left + // asserted re-fires forever, and one left stopped never fires again. + irqchip::rearm(); + percpu::irq_took(Source::Timer); + // Before anything that can take a lock, in both levels: a CPU spinning + // on one still takes this interrupt, which is why the poll is here. + crate::deadline::poll(); + #[cfg(feature = "boot-actuators")] + storm::tick(); + if from_el0 { + // Only an EL0 tick reaches here, so the interrupted context is user + // code and holds no `Lock`. + assert_eq!(crate::preempt::count(), 0, "a timer interrupt from EL0 found a preempt depth"); + let hw = &crate::hw::HW; + hw.trace(TraceEvent { ts: hw.now(), cpu: CpuId(percpu::cpu_id()), kind: TraceKind::TimerFire }); + irqchip::end(intid); + crate::scheduler::do_preempt(); + } else { + crate::preempt::set_need_resched(); + percpu::note_kernel_timer_fire(); + irqchip::end(intid); + } + return; + } + match intid { + irqchip::SGI_KICK => { + percpu::irq_took(Source::Timer); + irqchip::end(intid); + if from_el0 { + crate::scheduler::do_preempt(); + } else { + crate::preempt::set_need_resched(); + } + } + #[cfg(feature = "boot-actuators")] + intid if intid == u32::from(LOG_NEST_VECTOR) => { + percpu::preempt_count_up(); + crate::log::nested::deliver(); + percpu::preempt_count_down(); + irqchip::end(intid); + } + #[cfg(feature = "boot-actuators")] + irqchip::SGI_STORM => { + storm::sgi(); + irqchip::end(intid); + } + _ => { + percpu::irq_took(Source::Unclaimed); + UNCLAIMED.fetch_add(1, Relaxed); + LAST_UNCLAIMED.store(intid, Relaxed); + irqchip::end(intid); + } + } +} + +/// Interrupts no handler here claims, and the last one's INTID. +static UNCLAIMED: AtomicU64 = AtomicU64::new(0); +static LAST_UNCLAIMED: AtomicU32 = AtomicU32::new(0); +/// The count at the last report; process exit logs once per batch. +static UNCLAIMED_REPORTED: AtomicU64 = AtomicU64::new(0); + +pub(crate) fn log_unclaimed() { + let count = UNCLAIMED.load(Relaxed); + if count == 0 || UNCLAIMED_REPORTED.swap(count, Relaxed) == count { + return; + } + log!("irq: unclaimed interrupts={count}, the last INTID {}", LAST_UNCLAIMED.load(Relaxed)); +} + +/// A synchronous exception from EL0: a syscall, a fault the demand pager may +/// serve, or the end of the process. +fn el0_sync(frame: &mut Frame) { + match frame.esr >> 26 { + EC_SVC64 => syscall(frame), + EC_IABT_LOWER | EC_DABT_LOWER => user_abort(frame), + _ => { + percpu::preempt_count_up(); + user_fatal(frame); + } + } +} + +/// `SVC #0`: the number in `x0`, four arguments in `x1`–`x4`, the answer +/// back in `x0`, and every other register the thread's as it left it. +fn syscall(frame: &mut Frame) { + percpu::enter_syscall(frame.elr, frame.x[0], frame.x[29], frame.sp); + percpu::preempt_count_up(); + let answer = crate::syscall::dispatch::syscall_dispatch(frame.x[0], frame.x[1], frame.x[2], frame.x[3], frame.x[4]); + percpu::preempt_count_down(); + percpu::leave_syscall(); + frame.x[0] = answer; +} + +/// `DFSC`/`IFSC` levels 0 to 3 of a translation fault: nothing mapped there. +fn is_translation_fault(esr: u64) -> bool { + esr & 0b11_1100 == 0b00_0100 +} + +/// An abort from EL0: a translation fault in a region the demand pager +/// fills, or the end of the process. +fn user_abort(frame: &mut Frame) { + percpu::preempt_count_up(); + // `FnV`: a data abort whose `FAR_EL1` is not valid names no address to fill. + let far_valid = frame.esr >> 26 == EC_IABT_LOWER || frame.esr & (1 << 10) == 0; + if is_translation_fault(frame.esr) && far_valid { + let prev = percpu::swap_fault_state(CpuFaultState::PageFault); + if prev != CpuFaultState::Normal { + // Put back: `user_fatal` classifies a recursive fault by what it finds. + percpu::set_fault_state(prev); + user_fatal(frame); + } + cpu::enable_interrupts(); + let served = crate::process::handle_page_fault(frame.far, frame.esr); + cpu::disable_interrupts(); + if served { + percpu::set_fault_state(CpuFaultState::Normal); + percpu::preempt_count_down(); + return; + } + log!( + "#PF UNHANDLED: far={:#x} pc={:#x} esr={:#x} tid={:?}", + frame.far, + frame.elr, + frame.esr, + percpu::current_tid() + ); + } + user_fatal(frame); +} + +/// A fault from EL0 nothing serves: report it, and end the process it came +/// from; a second fault while this one reports halts the machine. +fn user_fatal(frame: &Frame) -> ! { + let prev = percpu::swap_fault_state(CpuFaultState::Fatal); + let recursive = matches!(prev, CpuFaultState::Fatal | CpuFaultState::Panic); + // First: a panic anywhere below reaches the panic handler as DOUBLE + // PANIC, which can only report what was captured here. + crate::panic::record_fault(class_name(frame.esr), frame.elr, frame.far, frame.esr); + if crate::actuator::panic_in_report() { + panic!("panic-in-report: the crash report panicked before it said anything"); + } + let tid = percpu::current_tid().map_or(u32::MAX, |t| t.raw()); + alert!( + "FAULT pc={:#018x} far={:#018x} esr={:#010x} sp={:#018x} tid={tid}{}", + frame.elr, + frame.far, + frame.esr, + frame.sp, + if recursive { " RECURSIVE" } else { "" } + ); + if recursive { + crate::panic::halt_all_cpus(); + } + user_report(frame); + percpu::set_fault_state(CpuFaultState::Normal); + crate::panic::forget(); + crate::syscall::kill_process(-1); +} + +/// What a fault from EL0 says about the thread: the signal it would be +/// elsewhere, its registers, and its frames, read without a lock. +fn user_report(frame: &Frame) { + let tid = percpu::current_tid().map_or(u32::MAX, |t| t.raw()); + let class = frame.esr >> 26; + match class { + EC_IABT_LOWER | EC_DABT_LOWER => { + let action = if class == EC_IABT_LOWER { + "execute" + } else if frame.esr & (1 << 6) != 0 { + "write" + } else { + "read" + }; + let cause = match frame.esr & 0b11_1100 { + 0b00_0100 => "unmapped address", + 0b00_1100 => "protection violation", + _ => "fault", + }; + log!("SEGFAULT tid={tid}: {action} {cause} at {:#x} (FSC {:#x})", frame.far, frame.esr & 0x3F); + } + 0x00 => log!("SIGILL tid={tid}: illegal instruction"), + 0x22 | 0x26 => log!("SIGBUS tid={tid}: {}", class_name(frame.esr)), + 0x3C => log!("SIGTRAP tid={tid}: BRK #{:#x}", frame.esr & 0xFFFF), + _ => log!("FATAL tid={tid}: {} (ESR={:#010x})", class_name(frame.esr), frame.esr), + } + log!(" pc:"); + if percpu::current_pid().is_some() { + crate::process::resolve_user_symbol(frame.elr).log_bare(frame.elr); + } else { + log!(" {:#x}", frame.elr); + } + if matches!(class, EC_IABT_LOWER | EC_DABT_LOWER) { + super::paging::debug_page_walk(frame.far); + } + log!(" Registers:"); + for pair in 0..15 { + log!(" x{:<2}={:#018x} x{:<2}={:#018x}", pair * 2, frame.x[pair * 2], pair * 2 + 1, frame.x[pair * 2 + 1]); + } + log!(" x30={:#018x} sp={:#018x} spsr={:#010x}", frame.x[30], frame.sp, frame.spsr); + log!(" Backtrace:"); + user_backtrace(frame.x[29], 32); + let crash_addr = if matches!(class, EC_IABT_LOWER | EC_DABT_LOWER) { frame.far } else { 0 }; + crate::process::dump_crash_diagnostics(crash_addr, frame.elr); +} + +/// The user frame-pointer chain from `fp`: each frame's saved `x29` at `fp` +/// and its return address at `fp + 8`, read through the running space's tables. +fn user_backtrace(start_fp: u64, max_frames: usize) { + let mut fp = start_fp; + for _ in 0..max_frames { + let Some(saved) = super::paging::read_user_word(fp) else { break }; + let Some(ret) = super::paging::read_user_word(fp + 8) else { break }; + if ret == 0 { + break; + } + crate::process::resolve_user_symbol_return(ret).log_bare(ret); + fp = saved; + } +} + // The table: sixteen entries of 0x80 bytes, 2 KiB aligned (`VBAR_EL1` bits // 10:0 are RES0). Each entry makes room for a `Frame`, saves x0/x1, names -// itself in x1 and branches to the common save, which fills in the rest and -// calls `exception` with the frame in x0. +// itself in x1 and branches to the common save, which fills in the rest, calls +// `dispatch` with the frame in x0, and restores whatever it returns to. The +// stack an entry from EL0 names is `SP_EL0`, and a return to EL0 (`SPSR_EL1.M` +// zero) puts the frame's back. core::arch::global_asm!( ".macro toyos_vector n", ".balign 0x80", @@ -123,7 +395,12 @@ core::arch::global_asm!( "stp x24, x25, [sp, #192]", "stp x26, x27, [sp, #208]", "stp x28, x29, [sp, #224]", + "tbnz x1, #3, 2f", "add x2, sp, #{frame}", + "b 3f", + "2:", + "mrs x2, sp_el0", + "3:", "stp x30, x2, [sp, #240]", "mrs x2, elr_el1", "mrs x3, spsr_el1", @@ -132,18 +409,43 @@ core::arch::global_asm!( "mrs x3, far_el1", "stp x2, x3, [sp, #272]", "mov x0, sp", - "bl {exception}", - "brk #0", + "bl {dispatch}", + "ldp x2, x3, [sp, #256]", + "msr elr_el1, x2", + "msr spsr_el1, x3", + "ldr x2, [sp, #248]", + "tst x3, #0xF", + "b.ne 4f", + "msr sp_el0, x2", + "4:", + "ldp x2, x3, [sp, #16]", + "ldp x4, x5, [sp, #32]", + "ldp x6, x7, [sp, #48]", + "ldp x8, x9, [sp, #64]", + "ldp x10, x11, [sp, #80]", + "ldp x12, x13, [sp, #96]", + "ldp x14, x15, [sp, #112]", + "ldp x16, x17, [sp, #128]", + "ldp x18, x19, [sp, #144]", + "ldp x20, x21, [sp, #160]", + "ldp x22, x23, [sp, #176]", + "ldp x24, x25, [sp, #192]", + "ldp x26, x27, [sp, #208]", + "ldp x28, x29, [sp, #224]", + "ldr x30, [sp, #240]", + "ldp x0, x1, [sp, #0]", + "add sp, sp, #{frame}", + "eret", ".popsection", frame = const FRAME_BYTES, - exception = sym exception, + dispatch = sym dispatch, ); /// Point `VBAR_EL1` at the table. Called by the entry before anything that can fault. pub fn install() { // SAFETY: the table is 2 KiB aligned, lives in the kernel image for the - // life of the machine, and every entry ends in `exception`, which never - // returns. The `ISB` makes the new base the one the next exception uses. + // life of the machine, and every entry ends in `dispatch`. The `ISB` makes + // the new base the one the next exception uses. unsafe { core::arch::asm!( "adrp {t}, toyos_vectors", @@ -156,25 +458,27 @@ pub fn install() { } } -/// Interrupt identifiers the generic drivers program. Inert until stage 4 -/// assigns them to GIC interrupts: every path that would deliver one -/// ([`super::msi_message`], [`super::irqchip::send_self`]) is owed. -#[repr(u8)] -enum Vector { - LogNest = 1, - Hda, - VirtioSound, -} - -pub const HDA_VECTOR: u8 = Vector::Hda as u8; -pub const VIRTIO_SOUND_VECTOR: u8 = Vector::VirtioSound as u8; -pub const LOG_NEST_VECTOR: u8 = Vector::LogNest as u8; +pub const HDA_VECTOR: u8 = irqchip::Intid::Hda as u8; +pub const VIRTIO_SOUND_VECTOR: u8 = irqchip::Intid::VirtioSound as u8; +pub const LOG_NEST_VECTOR: u8 = irqchip::Intid::LogNest as u8; -/// The crash report for a panic, from the frame pointer the panic handler stood on. +/// The crash report for a panic, from the frame pointer the panic handler +/// stood on: the backtrace, which CPU is on which stack, and what the +/// running thread was doing. pub(crate) fn report_panic(message: &core::panic::PanicInfo, frame: u64) { - crate::alert!("PANIC: {}", message); + alert!("PANIC: {}", message); log!(" Backtrace:"); crate::symbols::kernel_backtrace(frame, 20); + crate::hw::report_contexts(cpu::stack_pointer(), None); + let Some(pid) = percpu::current_pid() else { return }; + log!(" Running: pid={} tid={:?}", pid, percpu::current_tid()); + if percpu::in_syscall() { + let (pc, fp, sp) = percpu::syscall_context(); + log!(" Syscall: num={} user_pc={pc:#x} user_sp={sp:#x}", percpu::syscall_num()); + log!(" User backtrace:"); + crate::process::resolve_user_symbol(pc).log_bare(pc); + user_backtrace(fp, 20); + } } /// Whether the interrupted context an `SPSR` describes could have taken an @@ -186,15 +490,72 @@ pub(crate) const fn frame_interrupts_enabled(spsr: u64) -> bool { /// Nothing to report: the vectors run on the stack they interrupted. pub(crate) fn report_fault_stack() {} -pub fn kernel_exit_to_user_check() { - owed!("the return to user mode", "stage 7") +/// x86-64's lands a `#DF` on its IST stack; AArch64 has no double fault, and +/// an exception on a broken stack takes the same vector again, so the call is +/// refused. +#[cfg(feature = "test-actuators")] +pub(crate) fn provoke_double_fault() -> u64 { + toyos_abi::syscall::SyscallError::NotSupported.to_u64() } -pub(crate) fn log_unclaimed() { - owed!("the interrupt controller", "stage 4") -} +/// `irq-storm`: this CPU floods itself with SGIs, sending each as soon as the +/// last is taken, until the timer has fired `TICKS_OWED` times through the +/// flood, then waits for the last SGI to be taken. One at a time, because an +/// SGI sent while the last is still pending merges with it. A tick lost, or +/// never re-armed, leaves the flood running and an SGI lost leaves the wait +/// running, so neither says anything: the harness's ceiling is the only clock +/// this judges by, and the one verdict the storm can say is `FAIL` for an SGI +/// taken that was never sent. +#[cfg(feature = "boot-actuators")] +pub(crate) mod storm { + use core::sync::atomic::{AtomicBool, AtomicU64, Ordering::Relaxed}; -#[cfg(feature = "test-actuators")] -pub(crate) fn provoke_double_fault() -> ! { - owed!("a fault on the exception stack", "stage 4") + use super::irqchip; + use crate::log; + + static RUNNING: AtomicBool = AtomicBool::new(false); + static TICKS: AtomicU64 = AtomicU64::new(0); + static SGIS: AtomicU64 = AtomicU64::new(0); + + pub(super) fn tick() { + if RUNNING.load(Relaxed) { + TICKS.fetch_add(1, Relaxed); + } + } + + pub(super) fn sgi() { + SGIS.fetch_add(1, Relaxed); + } + + /// The timer's period, which each tick re-arms. + const PERIOD_NS: u64 = 1_000_000; + /// The ticks the flood lasts for. + const TICKS_OWED: u64 = 1_000; + + /// Run the storm, with interrupts open on this CPU until it ends, and say + /// what it counted. + pub fn run() { + let _guard = crate::arch::IrqGuard::close(); + TICKS.store(0, Relaxed); + SGIS.store(0, Relaxed); + irqchip::arm_one_shot(PERIOD_NS); + RUNNING.store(true, Relaxed); + let mut sent = 0u64; + crate::arch::cpu::enable_interrupts(); + while TICKS.load(Relaxed) < TICKS_OWED { + if SGIS.load(Relaxed) == sent { + irqchip::send_self(irqchip::SGI_STORM as u8); + sent += 1; + } + } + RUNNING.store(false, Relaxed); + irqchip::stop_timer(); + while SGIS.load(Relaxed) < sent { + core::hint::spin_loop(); + } + crate::arch::cpu::disable_interrupts(); + let (ticks, taken) = (TICKS.load(Relaxed), SGIS.load(Relaxed)); + let verdict = if taken == sent { "PASS" } else { "FAIL" }; + log!("irq-storm: {verdict} sgis={taken}/{sent} ticks={ticks}: the timer fired through the flood, and every SGI sent was taken"); + } } diff --git a/kernel/src/arch/aarch64/watchdog.rs b/kernel/src/arch/aarch64/watchdog.rs index 628608537f..7e88727e37 100644 --- a/kernel/src/arch/aarch64/watchdog.rs +++ b/kernel/src/arch/aarch64/watchdog.rs @@ -1,16 +1,17 @@ //! The platform watchdog: on an ACPI Arm machine, an SBSA generic watchdog -//! the GTDT names, the port's stage 6. +//! the GTDT names, which no stage of the port has taken on yet. So one is +//! never armed, and a boot that asks for one is refused by name. use crate::drivers::pci::PciDevice; pub fn init(_devices: &[PciDevice]) { - owed!("the platform watchdog", "no stage yet") + if crate::params::watchdog() { + owed!("the platform watchdog (the SBSA generic watchdog the GTDT names)", "no stage yet"); + } } -pub fn feed(_now: u64) { - owed!("the platform watchdog", "no stage yet") -} +/// Nothing is armed to feed. +pub fn feed(_now: u64) {} -pub fn disarm() { - owed!("the platform watchdog", "no stage yet") -} +/// Nothing is armed to disarm. +pub fn disarm() {} diff --git a/kernel/src/arch/x86_64/apic.rs b/kernel/src/arch/x86_64/apic.rs index d6b2b703d9..31b7902968 100644 --- a/kernel/src/arch/x86_64/apic.rs +++ b/kernel/src/arch/x86_64/apic.rs @@ -1,8 +1,9 @@ use core::sync::atomic::{AtomicBool, AtomicU32, Ordering}; use super::{cpu, percpu}; +use crate::hw::MIN_ONE_SHOT; use crate::log; -use crate::time::{Delay, Duration, Floor}; +use crate::time::{Delay, Duration}; /// The local APIC registers and MSRs this file may name. // Every variant is an architectural local-APIC register touching no memory or control transfer, so none can make `Reg::write`'s unsafe wrmsr unsound. @@ -48,8 +49,8 @@ pub const MSI_DOORBELL: u32 = 0xFEE0_0000; /// The compatibility-format message that raises `vector` on the CPU whose APIC /// ID is `dest`: the destination in address bits 19:12, the vector in the data. -pub fn msi_message(dest: u32, vector: u8) -> (u32, u32) { - (MSI_DOORBELL | (dest << 12), vector as u32) +pub fn msi_message(dest: u32, vector: u8) -> Result<(u32, u32), &'static str> { + Ok((MSI_DOORBELL | (dest << 12), vector as u32)) } /// Calibrated LAPIC timer ticks per 10ms (computed on BSP, reused by APs). @@ -226,12 +227,6 @@ pub fn init_timer() { log!("LAPIC timer: {} ticks/10ms, so {}Hz", ticks_10ms, ticks_10ms as u64 * 100); } -// Floor on every arm: a count that expires before the interrupt it schedules retires cannot outlast itself and livelocks the CPU forever. -const MIN_ONE_SHOT: Floor = Floor::policy( - Duration::from_micros(10), - "above an interrupt entry and iretq, a thousandth of QUANTUM_NS", -); - // The only path to Reg::TimerInit / last_armed_ticks — the floor is enforced once here, not at each of the three call sites. struct OneShot(u32); diff --git a/kernel/src/arch/x86_64/boot.rs b/kernel/src/arch/x86_64/boot.rs index bbffc71d8a..341a7edd20 100644 --- a/kernel/src/arch/x86_64/boot.rs +++ b/kernel/src/arch/x86_64/boot.rs @@ -151,4 +151,8 @@ pub fn interrupt_selftests() { if crate::actuator::unclaimed_vector_selftest() { idt::unclaimed::selftest(); } + assert!( + !crate::actuator::irq_storm() && !crate::actuator::timer_floor(), + "irq-storm and timer-floor are the GIC and generic timer's selftests, and this machine has a local APIC" + ); } diff --git a/kernel/src/arch/x86_64/hw.rs b/kernel/src/arch/x86_64/hw.rs index 78a458f669..7f224e72f1 100644 --- a/kernel/src/arch/x86_64/hw.rs +++ b/kernel/src/arch/x86_64/hw.rs @@ -1,165 +1,27 @@ -//! `KernelHw` — the kernel's side of the scheduler-core hardware boundary. -//! -//! Everything here is x2APIC, TSC or a single instruction; nothing here -//! makes scheduling decisions. The simulator that exercises the scheduler -//! core replaces this file and nothing else. +//! x86-64's half of `crate::hw::KernelHw`: the halt and the context switch. use core::arch::asm; -use toyos_sched::cpu::SleepToken; use toyos_sched::cpu::RunToken; -use toyos_sched::fair::QUANTUM_NS; -use toyos_sched::hw::{CpuId, Hw, Kicker, Machine, Nanos, TraceEvent}; +use toyos_sched::hw::Hw; use toyos_sched::task::{TaskAccounting, TaskKey}; -use crate::arch::{apic, cpu, percpu}; +use crate::arch::{cpu, percpu}; +use crate::hw::{report_contexts, KernelHw}; use super::switch::context_switch; use crate::sched::payload::{KernelCtx, KernelPayload}; -/// The one instance; zero-sized, holds no per-CPU state. -pub static HW: KernelHw = KernelHw; - -pub struct KernelHw; - -/// The scheduler clock, in raw nanoseconds. -pub fn now_ns() -> u64 { - HW.now().0 -} - -impl Kicker for KernelHw { - fn kick(&self, target: CpuId) { - apic::kick_cpu(target.0); - } -} - -impl Machine for KernelHw { - fn now(&self) -> Nanos { - Nanos(crate::clock::nanos_since_boot()) - } - - /// Converts the trait's absolute deadline to the one-shot timer's relative count; a deadline - /// already past saturates to zero rather than firing immediately, which would spin the Ring 0 - /// stub in a reload loop. - fn set_timer(&self, deadline: Nanos) { - apic::arm_one_shot(deadline.0.saturating_sub(self.now().0)); - } - - fn stop_timer(&self) { - apic::stop_timer(); - } - - fn halt(&self) { - // SAFETY: `sti; hlt` is the atomic enable-and-wait pair — a wake landing between the two is not lost. - unsafe { asm!("sti; hlt", options(nomem, nostack)); } - } - - /// A kick IPI is how a remote CPU's `need_resched` gets set — there is no way to write it directly. - fn need_resched(&self, cpu: CpuId) { - if cpu.0 == percpu::cpu_id() { - crate::preempt::set_need_resched(); - } else { - self.kick(cpu); - } - } - - fn trace(&self, ev: TraceEvent) { - crate::trace::record(ev); - } - - /// Diagnostic builds arm a periodic wake before halting so a quiescent CPU still reports. - fn idle_wait(&self, token: SleepToken) { - let _consumed = token; - #[cfg(feature = "boot-actuators")] - if crate::actuator::diag_tick() { - apic::arm_within(DIAG_TICK_NS); - } - self.halt(); - // **A CPU that is executing has a one-shot armed, and this is where - // that becomes true again.** `TimerPlan::Stop` left this one at zero - // before the halt above and only a pass reaching `apply_timer` arms - // another, so a CPU woken by an IPI or a device — never by its own - // timer, which is stopped — and then held in Ring 0 takes no timer - // interrupt at all. `crate::deadline`'s poll and - // `crate::hardlockup`'s sample both rest on some CPU taking one. - // Arming earlier than the scheduler planned is a spurious pass and - // never a missed deadline (`toyos_sched::timer::TimerPlan`), and the - // next pass replaces it either way. - // - // Asked, because it is only those two that need it: a boot under no - // bound pays an x2APIC read and two writes per wake for nothing. - if crate::deadline::armed() { - apic::arm_within(QUANTUM_NS); - } - } -} - -/// Longest sleep on a `diag-tick` build; kept under `heartbeat`'s reporting period so a healthy CPU reports on every line. -#[cfg(feature = "boot-actuators")] -const DIAG_TICK_NS: u64 = 100_000_000; - -/// Which context each CPU last switched onto; read by [`report_contexts`] on crash, since a -/// sibling's real `CpuSched` is `!Sync` and unreadable directly. -static RUNNING_CTX: [core::sync::atomic::AtomicU64; crate::sched::MAX_CPUS] = - [const { core::sync::atomic::AtomicU64::new(0) }; crate::sched::MAX_CPUS]; - -/// Prints which CPU is standing on which context and stack, on every kernel crash. -/// -/// `subject` (`None` for this CPU's own) is the context flagged as "the same". -/// -/// Allocates, locks or formats nothing but integers, since a crash may already hold any lock this -/// could try to take. -pub fn report_contexts(rsp: u64, subject: Option) { - let me = percpu::cpu_id() as usize; - let count = (crate::arch::smp::cpu_count() as usize).min(crate::sched::MAX_CPUS); - let mine = RUNNING_CTX - .get(me) - .map_or(0, |slot| slot.load(core::sync::atomic::Ordering::Relaxed)); - let subject = subject.unwrap_or(mine); - crate::log!(" Contexts: cpu{me} crashed at rsp={rsp:#018x}, asking about ctx {subject:#x}"); - for (cpu, slot) in RUNNING_CTX.iter().enumerate().take(count) { - let held = slot.load(core::sync::atomic::Ordering::Relaxed); - if !crate::mm::is_kernel_addr(held) || !held.is_multiple_of(8) { - crate::log!(" cpu{cpu} is on ctx {held:#x} (never switched, or not a context)"); - continue; - } - // SAFETY: `held` is a pointer this kernel's own `Hw::switch` stored, into the boxed, always-mapped direct map. - let ctx = unsafe { &*(held as *const KernelCtx) }; - let top = ctx.kernel_stack_top; - let same = held == subject && cpu != me; - // `top != 0` excludes idle contexts, whose stack top is zero by construction — the - // containment test below never fires for one; that is a gap in this report, not a bug. - let on_its_stack = cpu != me - && top != 0 - && rsp <= top - && rsp > top.wrapping_sub(crate::process::KERNEL_STACK_SIZE as u64); - // idle's `kernel_stack_top` is zero by construction; rendering it as a task would misread as corruption. - match ctx.id { - None => crate::log!( - " cpu{cpu} is on ctx {held:#x} (its idle context) stack_top={top:#018x} \ - saved_rsp={:#018x}{}{}", - ctx.rsp, - if same { " <== THE SAME CONTEXT" } else { "" }, - if top == 0 { "" } else { " <== AN IDLE CONTEXT'S STACK TOP IS ZERO BY CONSTRUCTION" }, - ), - Some(id) => crate::log!( - " cpu{cpu} is on ctx {held:#x} pid={} tid={} stack_top={top:#018x} \ - saved_rsp={:#018x}{}{}", - id.0.raw(), - id.1.raw(), - ctx.rsp, - if same { " <== THE SAME CONTEXT" } else { "" }, - if on_its_stack { " <== AND THIS CRASH IS ON THAT STACK" } else { "" }, - ), - } - } - crate::mm::report_on_crash(); +/// `sti; hlt`, the atomic enable-and-wait pair: a wake landing between the two is not lost. +pub fn halt() { + // SAFETY: enables interrupts and waits for one; touches no memory. + unsafe { asm!("sti; hlt", options(nomem, nostack)); } } /// Panics before the wild `ret` would restore register state that makes the failure unnameable. #[cold] #[inline(never)] fn switch_frame_is_wrong(ctx: &KernelCtx, token: &RunToken) -> ! { - let rsp = ctx.rsp; + let rsp = ctx.sp; let (pid, tid) = ctx.id.map_or((u32::MAX, u32::MAX), |id| (id.0.raw(), id.1.raw())); crate::log!( "CONTEXT SWITCH ONTO A FRAME THAT IS NOT ONE: cpu={} pid={pid} tid={tid} \ @@ -171,7 +33,7 @@ fn switch_frame_is_wrong(ctx: &KernelCtx, token: &RunToken) -> ! ctx.kernel_stack_top, ctx.kernel_stack_top.wrapping_sub(rsp) as i64, ctx.preempt, - ctx.fs_base, + ctx.thread_pointer, token.incoming().map(|k| k.0), token.outgoing().map(|k| k.0), ); @@ -193,11 +55,11 @@ fn switch_frame_is_wrong(ctx: &KernelCtx, token: &RunToken) -> ! ); } -/// Returns the validated `rsp` — callers must use this value; re-reading `ctx.rsp` here would be a second, unguarded load after `cr3.activate()`'s memory clobber. +/// Returns the validated `rsp` — callers must use this value; re-reading `ctx.sp` here would be a second, unguarded load after `cr3.activate()`'s memory clobber. #[inline] #[must_use] fn check_switch_frame(ctx: &KernelCtx, token: &RunToken) -> u64 { - let rsp = ctx.rsp; + let rsp = ctx.sp; if !crate::mm::is_kernel_addr(rsp) || !rsp.is_multiple_of(8) { switch_frame_is_wrong(ctx, token); } @@ -288,7 +150,7 @@ pub(crate) unsafe extern "C" fn switch_witness_verify(rsp: u64) { *word = unsafe { core::ptr::read_volatile((rsp + (i as u64) * 8) as *const u64) }; } // SAFETY: `shadow.ctx` is a live `KernelCtx` from this kernel's own pass. - let field = unsafe { core::ptr::read_volatile(&raw const (*shadow.ctx).rsp) }; + let field = unsafe { core::ptr::read_volatile(&raw const (*shadow.ctx).sp) }; if rsp == shadow.rsp && field == shadow.rsp && now == shadow.words { return; } @@ -351,7 +213,7 @@ unsafe fn switch_witness_mutate(restore: *const KernelCtx) { return; } // SAFETY: `restore` is a live `KernelCtx` from this kernel's own pass. - let rsp = unsafe { (*restore).rsp }; + let rsp = unsafe { (*restore).sp }; #[cfg(feature = "switch-witness-mutate-frame")] // SAFETY: the `rbx` slot of the frame `check_switch_frame` has just validated. unsafe { @@ -362,7 +224,7 @@ unsafe fn switch_witness_mutate(restore: *const KernelCtx) { // the report and not by a different crash. // SAFETY: as above, and the field is this context's own. unsafe { - core::ptr::write_volatile(&raw const (*restore).rsp as *mut u64, rsp + 8) + core::ptr::write_volatile(&raw const (*restore).sp as *mut u64, rsp + 8) }; } @@ -418,12 +280,12 @@ impl Hw for KernelHw { unsafe fn switch(&self, token: RunToken) { let save = token.save_ptr(); let restore = token.restore_ptr(); - // SAFETY: `save`/`restore` are live Box-backed contexts from `SchedPass::finish`, freed only by a later pass; `incoming.fs_base` is this kernel's own canonical value for the thread being installed. + // SAFETY: `save`/`restore` are live Box-backed contexts from `SchedPass::finish`, freed only by a later pass; `incoming.thread_pointer` is this kernel's own canonical value for the thread being installed. unsafe { - (*save).fs_base = cpu::read_fs_base(); + (*save).thread_pointer = cpu::read_fs_base(); (*save).preempt = crate::preempt::count(); let incoming: &KernelCtx = &*restore; - // The only load of `incoming.rsp`: reading it again after `cr3.activate()`'s clobber would be a second, unguarded load. + // The only load of `incoming.sp`: reading it again after `cr3.activate()`'s clobber would be a second, unguarded load. let rsp = check_switch_frame(incoming, &token); #[cfg(any( feature = "switch-witness-mutate-frame", @@ -442,7 +304,7 @@ impl Hw for KernelHw { crate::heartbeat::note_dispatch(); percpu::set_kernel_stack(incoming.kernel_stack_top); incoming.root.activate(); - cpu::write_fs_base(incoming.fs_base); + cpu::write_fs_base(incoming.thread_pointer); } // idle's stack top is per-CPU, unknowable at boot-time init, so it is read here instead. None => { @@ -450,8 +312,7 @@ impl Hw for KernelHw { incoming.root.activate(); } } - RUNNING_CTX[percpu::cpu_id() as usize] - .store(restore as u64, core::sync::atomic::Ordering::Relaxed); + crate::hw::note_running(restore); // AMD's `SYSRET` reloads SS's selector but not its cached descriptor, so // a `sysretq` onto a task an IDT entry left with SS NULL hands userland an // unusable SS; reloading it here keeps it valid (`X86_BUG_SYSRET_SS_ATTRS`). @@ -461,7 +322,7 @@ impl Hw for KernelHw { ds = in(reg) percpu::KERNEL_DS as u64, options(nomem, nostack, preserves_flags), ); - context_switch(&raw mut (*save).rsp, rsp); + context_switch(&raw mut (*save).sp, rsp); } } diff --git a/kernel/src/arch/x86_64/idt/mod.rs b/kernel/src/arch/x86_64/idt/mod.rs index 3c7d56d78a..be72d3d92c 100644 --- a/kernel/src/arch/x86_64/idt/mod.rs +++ b/kernel/src/arch/x86_64/idt/mod.rs @@ -347,35 +347,7 @@ extern "sysv64" fn common_entry() { /// Deferred-preempt epilogue; caller must have IF=0 on entry and it returns with IF=0. pub(crate) extern "sysv64" fn kernel_exit_to_user_check() { - flush_ring0_timer_fires_to_trace(); - loop { - // A killed or stopped thread returns to Ring 3 exactly once more: never. - crate::scheduler::leave_ring3_if_due(); - // `do_preempt` owns clearing `need_resched`; this function never clears it itself. - if !crate::preempt::need_resched() { - #[cfg(feature = "boot-actuators")] - if crate::actuator::dump_in_blocking_pass() { - crate::sched::dump::staged::note_return_to_ring3(); - } - return; - } - assert!(!crate::scheduler::in_schedule_self(), - "exit-to-user inside a scheduler pass"); - // Not an IrqGuard: both loop exits must set IF, not restore a saved value. - cpu::enable_interrupts(); - crate::scheduler::do_preempt(); - cpu::disable_interrupts(); - flush_ring0_timer_fires_to_trace(); - } -} - -fn flush_ring0_timer_fires_to_trace() { - let cur = percpu::ring0_timer_fires(); - let missed = cur.wrapping_sub(percpu::last_seen_ring0_fires()); - if missed > 0 { - crate::trace::trace(crate::trace::Kind::TimerFireBurst, missed); - percpu::set_last_seen_ring0_fires(cur); - } + crate::scheduler::exit_to_user(); } /// Routes by vector to the appropriate handler; #DF and #MC get dedicated arms because they are aborts with no instruction to return to. diff --git a/kernel/src/arch/x86_64/paging.rs b/kernel/src/arch/x86_64/paging.rs index 466d1ecf4b..1ad2ef7bad 100644 --- a/kernel/src/arch/x86_64/paging.rs +++ b/kernel/src/arch/x86_64/paging.rs @@ -8,7 +8,6 @@ use crate::log; use alloc::boxed::Box; -use alloc::collections::BTreeMap; use alloc::vec::Vec; use crate::hasher::HashMap; @@ -21,7 +20,6 @@ use crate::arch::cpu::Invpcid; use crate::sync::Lock; use crate::vma::{self, Occupancy, Region, RegionKind}; use toyos_bootmap::ROOT_HIGH_HALF; -use toyos_userbound::PageSpan; use crate::MemoryMapEntry; const PAGE_PRESENT: u64 = 1 << 0; @@ -368,7 +366,7 @@ pub struct AddressSpace { /// Physical data pages mapped into user space, keyed by physical address. Freed on drop. pages: HashMap, /// All virtual memory regions, keyed by start address. - regions: BTreeMap, + regions: vma::Regions, /// Owned for this space's life: dropping the space returns a user tag, so two /// live spaces can never share one. pcid: PcidHandle, @@ -396,7 +394,7 @@ impl AddressSpace { root: pml4, children: Vec::new(), pages: HashMap::default(), - regions: BTreeMap::new(), + regions: vma::Regions::default(), pcid: PcidHandle::User(pcid), }) } @@ -607,20 +605,9 @@ impl AddressSpace { Some((crate::mm::DirectMap::from_phys(page_phys + offset), rights, PAGE_2M)) } - /// Where `span` goes, top-down and never below the floor: a region the - /// kernel placed under it, the clock page, bounds no gap. - fn find_gap(&self, span: PageSpan) -> Option { - let taken = self.regions.iter().rev().map(|(start, region)| (start.raw(), region.size)); - vma::window().gap(span, taken).map(UserAddr::new) - } - - /// Allocate a virtual address range and register the region. `size` is - /// made a [`PageSpan`] before anything is summed on it. + /// Allocate a virtual address range and register the region. pub fn alloc_region(&mut self, size: u64, kind: RegionKind) -> Option { - let span = vma::window().span(size)?; - let addr = self.find_gap(span)?; - self.regions.insert(addr, Region { size: span.bytes(), kind }); - Some(addr) + self.regions.alloc(size, kind) } /// A mixed-`Prot` image uses [`alloc_region`](Self::alloc_region) plus [`map_window`](Self::map_window) per 2 MiB instead. @@ -635,59 +622,30 @@ impl AddressSpace { phys & (PAGE_2M - 1) == 0, "alloc_and_map: phys {phys:#x} not 2MB-aligned" ); - let span = vma::window().span(size)?; - let addr = self.find_gap(span)?; - let aligned = span.bytes(); - self.regions.insert( - addr, - Region { - size: aligned, - kind: RegionKind::Mapped, - }, - ); + let (addr, aligned) = self.regions.alloc_mapped(size)?; self.map_range(addr, phys, aligned, prot, cache); Some((addr, aligned)) } /// Free a previously allocated region and unmap it. pub fn free_and_unmap(&mut self, addr: UserAddr) -> Option { - let size = self.regions.remove(&addr)?.size; + let size = self.regions.remove(addr)?; self.unmap_range(addr, size); Some(size) } /// Insert a region at a specific address (for ELF segments, stack, etc.) pub fn insert_region(&mut self, addr: UserAddr, region: Region) { - assert!( - self.find_region(addr).is_none(), - "insert_region: address {:#x} already occupied", - addr.raw() - ); self.regions.insert(addr, region); } /// Find the region containing `addr`. Returns (start_addr, region). pub fn find_region(&self, addr: UserAddr) -> Option<(UserAddr, &Region)> { - let (&start, region) = self.regions.range(..=addr).next_back()?; - if addr.raw() < start.raw() + region.size { - Some((start, region)) - } else { - None - } + self.regions.find(addr) } - /// The end is saturating so a caller's arithmetic cannot wrap into a smaller range. pub fn occupancy(&self, addr: UserAddr, size: u64) -> Occupancy { - let end = UserAddr::new(addr.raw().saturating_add(size)); - let mut over = self.overlapping_regions(addr, end); - let Some((&start, region)) = over.next() else { - return Occupancy::Free; - }; - if over.next().is_none() && start == addr && region.size == size { - Occupancy::Whole - } else { - Occupancy::Partial - } + self.regions.occupancy(addr, size) } /// Iterate all regions that overlap the range [start, end). @@ -696,10 +654,7 @@ impl AddressSpace { start: UserAddr, end: UserAddr, ) -> impl Iterator { - // Overlaps [start, end) iff s < end && s+n > start; `range(..end)` prunes the first half. - self.regions - .range(..end) - .filter(move |(&s, r)| s.raw() + r.size > start.raw()) + self.regions.overlapping(start, end) } /// Private: not safe to use until every CPU is told — the free fn [`map_mmio`] is the whole operation. @@ -921,18 +876,11 @@ pub fn map_mmio(phys: u64, size: u64, policy: MmioPolicy) -> crate::mm::Mmio { mmio } - -/// Take the 4 KiB page holding `addr` out of the kernel direct map; `addr`'s -/// page must be owned by the caller forever (see [`AddressSpace::guard_4k`]). -pub fn guard_kernel_page(addr: u64) { - assert!(crate::mm::is_kernel_addr(addr), "guard_kernel_page: {addr:#x} is not a kernel address"); - kernel().lock().guard_4k(crate::mm::DirectMap::phys_of(addr as *const u8)); -} - /// Build kernel page tables: the direct map in the high half, in 2 MiB pages, /// as far as [`toyos_bootmap::x86_64::direct_map_end`] reaches, and answer -/// that end. -pub(crate) fn init(memory_map: &[MemoryMapEntry]) -> toyos_bootmap::DirectMapEnd { +/// that end. The scanout is not mapped for itself: in the low map the MTRRs +/// type it until `panic_console::remap` gives it a type of its own. +pub(crate) fn init(memory_map: &[MemoryMapEntry], _scanout: Option<(u64, u64)>) -> toyos_bootmap::DirectMapEnd { let extent = toyos_bootmap::x86_64::direct_map_end(memory_map) .unwrap_or_else(|refusal| panic!("paging: firmware's memory map: {refusal}")); let end = extent.get(); @@ -941,7 +889,7 @@ pub(crate) fn init(memory_map: &[MemoryMapEntry]) -> toyos_bootmap::DirectMapEnd root: Box::new(PageTablePage([0; 512])), children: Vec::new(), pages: HashMap::default(), - regions: BTreeMap::new(), + regions: vma::Regions::default(), pcid: PcidHandle::Kernel, }; @@ -1193,3 +1141,14 @@ pub fn scanout_memory_type(addr: u64, size: u64) -> impl core::fmt::Display { } Report(crate::arch::mtrr::range_type(addr, size)) } + +/// Whether a process can be given the scanout at `phys`, which it is in +/// 2 MiB pages: its base on one, as the loader already requires, since what +/// the last page covers past it is the aperture firmware put it in. +pub fn scanout_whole_pages(phys: u64, _size: u64) -> Result<(), &'static str> { + if phys.is_multiple_of(PAGE_2M) { + Ok(()) + } else { + Err("its base is not on a 2 MiB page, and a process is given a scanout in 2 MiB pages") + } +} diff --git a/kernel/src/arch/x86_64/percpu.rs b/kernel/src/arch/x86_64/percpu.rs index f70b189345..c9050d340c 100644 --- a/kernel/src/arch/x86_64/percpu.rs +++ b/kernel/src/arch/x86_64/percpu.rs @@ -5,6 +5,7 @@ use alloc::alloc::alloc_zeroed; use core::alloc::Layout; use super::cpu; +use crate::sched::idle_stack::{words, FILL as STACK_FILL, FILL_WORD as STACK_FILL_WORD}; use crate::log; const MSR_GS_BASE: u32 = 0xC000_0101; @@ -310,27 +311,16 @@ pub(crate) mod gs { } } -/// Same size as a task's kernel stack: a `deferred` [`kobject!`] object may -/// own an `immediate` one, whose destructor then runs here instead. -const IDLE_STACK_SIZE: usize = crate::process::KERNEL_STACK_SIZE; - -/// One unmapped 4 KiB page below every idle stack: unmapped, not filled like -/// [`IST_GUARD_SIZE`], so a fault here escalates to IST1's `#DF` rather than silently corrupting memory. -const IDLE_GUARD_SIZE: usize = 4096; - /// The IST stacks this machine has: IST1 `#DF`, IST2 NMI, IST3 `#MC` — vectors /// that can arrive with `rsp` not a kernel stack (SDM Vol. 3A §6.14.5); `ist[n-1]` is IST*n*. pub(crate) const IST_STACKS: usize = 3; -/// One size for every IST stack, for [`IDLE_STACK_SIZE`]'s reason; must leave room to double the measured high water, which `double_fault_stack` asserts. +/// One size for every IST stack, for [`crate::sched::idle_stack::SIZE`]'s reason; must leave room to double the measured high water, which `double_fault_stack` asserts. const IST_STACK_SIZE: usize = 16384; /// Filled with [`STACK_FILL`], not unmapped: a fault already on IST1 is a triple fault, so detecting after the fact beats trapping it. const IST_GUARD_SIZE: usize = 4096; -/// Chosen so a zeroed or ASCII byte cannot be mistaken for untouched stack. -const STACK_FILL: u8 = 0xA5; -const STACK_FILL_WORD: u64 = u64::from_ne_bytes([STACK_FILL; 8]); /// Allocate and initialize `PerCpu` for a CPU; the pointer lives forever, one `write` of the whole struct so a new field must be given a value here. fn alloc_percpu(cpu_id: u32) -> *mut PerCpu { @@ -439,78 +429,20 @@ pub fn reserve_log_slot( (shard as *const log::Shard, seq, cpu, tid, pid) } -/// One idle stack and the guard page under it. -const IDLE_SLOT: usize = IDLE_GUARD_SIZE + IDLE_STACK_SIZE; - -/// Idle stacks come from their own 2 MiB pages, not the kernel heap: the guard's hole in the direct map would split a heap-shared leaf's TLB entry into 512. -/// Never freed — a leaf returned to the PMM would keep the hole. -static IDLE_STACKS: crate::sync::Lock = crate::sync::Lock::new(IdleArena { - pages: alloc::vec::Vec::new(), - stacks: alloc::vec::Vec::new(), - next: 0, - left: 0, -}); - -struct IdleArena { - pages: alloc::vec::Vec, - /// The bottom of every idle stack, so the deepest any CPU has gone reads from one. - stacks: alloc::vec::Vec, - /// Direct-map address of the next free slot. - next: u64, - left: usize, -} - -/// A 4 KiB-aligned `IDLE_SLOT` from the arena. -fn alloc_idle_slot() -> u64 { - let mut arena = IDLE_STACKS.lock(); - if arena.left < IDLE_SLOT { - let page = crate::mm::pmm::alloc_page(crate::mm::pmm::Category::KernelHeap) - .expect("percpu: no physical page for an idle stack"); - arena.next = page.direct_map().as_mut_ptr::() as u64; - arena.left = crate::mm::PAGE_2M as usize; - arena.pages.push(page); - } - let base = arena.next; - arena.next += IDLE_SLOT as u64; - arena.left -= IDLE_SLOT; - arena.stacks.push(base + IDLE_GUARD_SIZE as u64); - base -} - fn alloc_idle_stack(percpu: &mut PerCpu) { - let base = alloc_idle_slot(); - crate::mm::paging::guard_kernel_page(base); - // SAFETY: exactly `IDLE_STACK_SIZE` bytes above the unmapped guard, within the returned `IDLE_SLOT` — filled, not zeroed, so zero can't mark "untouched" for [`idle_stack_high_water`]. - unsafe { - core::ptr::write_bytes( - (base + IDLE_GUARD_SIZE as u64) as *mut u8, - STACK_FILL, - IDLE_STACK_SIZE, - ) - }; - percpu.idle_stack_top = base + IDLE_SLOT as u64; + percpu.idle_stack_top = crate::sched::idle_stack::alloc(); } /// How big one idle stack is; read by `SYS_DEBUG` for scale. #[cfg(feature = "test-actuators")] pub fn idle_stack_size() -> usize { - IDLE_STACK_SIZE + crate::sched::idle_stack::SIZE } -/// The deepest any CPU's idle stack has ever been, in bytes, read from the bottom up: nothing legitimate writes [`STACK_FILL`], so a touched byte stays changed. +/// The deepest any CPU's idle stack has ever been, in bytes. #[cfg(feature = "test-actuators")] pub fn idle_stack_high_water() -> usize { - let arena = IDLE_STACKS.lock(); - arena - .stacks - .iter() - .map(|&bottom| { - let untouched = - words(bottom, IDLE_STACK_SIZE).take_while(|&w| w == STACK_FILL_WORD).count() * 8; - IDLE_STACK_SIZE - untouched - }) - .max() - .unwrap_or(0) + crate::sched::idle_stack::high_water() } /// One stack per [`IST_STACKS`] row; an `ist[n-1]` left zero faults to address 0 unchecked. @@ -572,11 +504,6 @@ pub fn ist1_report() { }); } -/// Sequential u64s from `base`; every address is inside the caller's already-bounds-checked allocation. -fn words(base: u64, len: usize) -> impl Iterator { - // SAFETY: `i < len/8` bounds each address inside the caller's checked allocation; `read_volatile` keeps the fill-pattern read. - (0..len / 8).map(move |i| unsafe { core::ptr::read_volatile((base as *const u64).add(i)) }) -} /// Initialize per-CPU data for the BSP, and bring this CPU's exception /// handlers up. Call after paging + allocator, before `syscall::init`. @@ -709,15 +636,15 @@ pub fn percpu_ptr() -> *mut PerCpu { } /// Ring 0 timer fires the assembly stub has taken; written with a plain `inc` (IF clear there). -pub fn ring0_timer_fires() -> u32 { +pub fn kernel_timer_fires() -> u32 { gs::read_u32::() } -pub fn last_seen_ring0_fires() -> u32 { +pub fn last_seen_kernel_timer_fires() -> u32 { gs::read_u32::() } -pub fn set_last_seen_ring0_fires(v: u32) { +pub fn set_last_seen_kernel_timer_fires(v: u32) { gs::write_u32::(v); } @@ -729,7 +656,7 @@ pub fn set_last_armed_ticks(ticks: u32) { /// The last byte of this CPU's idle guard page — the first byte an overflow reaches. #[cfg(feature = "test-actuators")] pub fn idle_guard_byte() -> u64 { - idle_stack_top() - IDLE_STACK_SIZE as u64 - 1 + idle_stack_top() - crate::sched::idle_stack::SIZE as u64 - 1 } /// Top of this CPU's idle stack. diff --git a/kernel/src/clock.rs b/kernel/src/clock.rs index 019dfeb65c..bc9aa56d70 100644 --- a/kernel/src/clock.rs +++ b/kernel/src/clock.rs @@ -88,14 +88,14 @@ pub fn now() -> Instant { /// The [`cpu::rdtsc`] value `nanos` in the future, for a wait loop that must /// not call the nanosecond clock. pub fn tsc_deadline(nanos: u64) -> u64 { - cpu::counter().saturating_add(tsc_ticks(nanos)) + cpu::counter().saturating_add(counter_ticks(nanos)) } -/// `nanos` as a count of TSC ticks: a span converted once and then compared -/// against `rdtsc` differences, which is what a sampler that may not divide +/// `nanos` as a count of [`cpu::counter`] ticks: a span converted once and then compared +/// against counter differences, which is what a sampler that may not divide /// needs. Before [`init`] the period is unknown and this is zero, so a bound /// derived from it is one its arm has to refuse. -pub fn tsc_ticks(nanos: u64) -> u64 { +pub fn counter_ticks(nanos: u64) -> u64 { let period_fs = TSC_PERIOD_FS.load(Relaxed); if period_fs == 0 { return 0; @@ -103,7 +103,7 @@ pub fn tsc_ticks(nanos: u64) -> u64 { ((nanos as u128 * 1_000_000) / period_fs as u128) as u64 } -/// The span a count of [`tsc_ticks`] stands for, for a caller that measured +/// The span a count of [`counter_ticks`] stands for, for a caller that measured /// before there was a period to measure with and converts once, afterwards. /// Zero while the period is unknown, so a span taken on a machine that never /// calibrated reads as no time rather than as an invented one. diff --git a/kernel/src/drivers/gop.rs b/kernel/src/drivers/gop.rs index 1eb49d2831..1846ba1fda 100644 --- a/kernel/src/drivers/gop.rs +++ b/kernel/src/drivers/gop.rs @@ -25,6 +25,8 @@ impl Gpu for GopGpu { } /// `addr` is the physical address of the framebuffer supplied by firmware. +/// `None` is a scanout no process can be given: one is handed over in 2 MiB +/// pages, and the architecture says whether this one's are its own. pub fn init( addr: u64, size: u64, @@ -32,7 +34,11 @@ pub fn init( height: u32, stride: u32, pixel_format: u32, -) -> (Box, GpuInfo) { +) -> Option<(Box, GpuInfo)> { + if let Err(why) = crate::mm::paging::scanout_whole_pages(addr, size) { + log!("GOP: the scanout at {addr:#x}+{size:#x} is refused: {why}"); + return None; + } // A size smaller than stride*height*4 maps less than the compositor writes. // Boot-time firmware data has no actionable error path, so this panics rather than returning one. let needed = stride as u64 * height as u64 * 4; @@ -92,5 +98,5 @@ pub fn init( flags: 0, }; - (Box::new(GopGpu), info) + Some((Box::new(GopGpu), info)) } diff --git a/kernel/src/drivers/panic_console/mod.rs b/kernel/src/drivers/panic_console/mod.rs index bd46003575..2a9e2cfa29 100644 --- a/kernel/src/drivers/panic_console/mod.rs +++ b/kernel/src/drivers/panic_console/mod.rs @@ -24,7 +24,7 @@ use crate::log; use crate::panic_reboot::Bound; use crate::time::{Budget, Cadence, Duration}; use crate::mm::policy::MmioPolicy; -use crate::mm::{self, DirectMap, align_2m}; +use crate::mm::{self, DirectMap}; /// 1 bpp 8x16, codepoints 0x20..=0x7E, one byte per row, bit 7 leftmost. /// `tests/common/screen.rs` decodes this same file, so decoder and renderer cannot drift. @@ -495,6 +495,13 @@ pub fn arm(args: &KernelArgs, maps: &[MemoryMapEntry]) { } } +/// The scanout the panel paints, `(phys, bytes)`, while it is armed: what a +/// direct map that holds only memory maps before the boot's next record paints. +pub fn scanout() -> Option<(u64, u64)> { + let phys = RAW_PHYS.load(Ordering::Relaxed); + (phys != 0).then(|| (phys, RAW_SIZE.load(Ordering::Relaxed))) +} + /// Re-establish the mapping after `mm::init` replaces the bootloader's page /// tables. pub fn remap() { @@ -503,7 +510,7 @@ pub fn remap() { return; } let size = RAW_SIZE.load(Ordering::Relaxed); - mm::paging::map_mmio(phys, align_2m(size as usize) as u64, MmioPolicy::WriteCombining); + mm::paging::map_mmio(phys, size, MmioPolicy::WriteCombining); rearm(); } diff --git a/kernel/src/drivers/pci.rs b/kernel/src/drivers/pci.rs index 5e2737de80..14b18a5410 100644 --- a/kernel/src/drivers/pci.rs +++ b/kernel/src/drivers/pci.rs @@ -375,7 +375,13 @@ impl PciDevice { let dest = if crate::actuator::iommu_dest_apic1() { 1 } else { MSG_DEST }; match crate::iommu::remap_msi(self.bus, self.dev, self.func, vector, dest) { // The same CPU the remappable one would name. - crate::iommu::Delivery::Direct => Some(crate::arch::msi_message(dest, vector)), + crate::iommu::Delivery::Direct => match crate::arch::msi_message(dest, vector) { + Ok(message) => Some(message), + Err(why) => { + log!("PCI {:02x}:{:02x}.{}: not armed — {why}", self.bus, self.dev, self.func); + None + } + }, crate::iommu::Delivery::Remapped(m) => Some((m.address, m.data)), crate::iommu::Delivery::Refused(why) => { log!("PCI {:02x}:{:02x}.{}: not armed — {why}", self.bus, self.dev, self.func); diff --git a/kernel/src/drivers/xhci/wait/msc.rs b/kernel/src/drivers/xhci/wait/msc.rs index 01cae26755..001358849e 100644 --- a/kernel/src/drivers/xhci/wait/msc.rs +++ b/kernel/src/drivers/xhci/wait/msc.rs @@ -884,7 +884,7 @@ fn block_witness_holds(dev: &MscDevice, entered: BlockWitness) { "USB BOT WITNESS: MscDevice::block changed inside one round trip — the field at \ {at:#018x} held {:#018x} and now holds {:#018x} (the frame moved by {}). This CPU's \ Ring 3 entry stack is {top:#018x}, so the field stands {} bytes below it and the \ - running rsp is {:#018x}. `with_storage` copies the device onto this stack, so a \ + running stack pointer is {:#018x}. `with_storage` copies the device onto this stack, so a \ kernel text value here is a return address something else pushed.", entered.was, dev.block, diff --git a/kernel/src/hardlockup/mod.rs b/kernel/src/hardlockup/mod.rs index 97cfca1e71..b91a0022b2 100644 --- a/kernel/src/hardlockup/mod.rs +++ b/kernel/src/hardlockup/mod.rs @@ -25,7 +25,7 @@ //! that has not moved is a CPU that has taken nothing at all. A CPU whose count //! is stale for [`toyos_tco::hard_lockup_bound_ms`] of //! the bound this boot named, *and* whose sampled frame has `IF` clear, is -//! stuck: it seals a `WEDGED` record naming itself, its `rip` and `rsp` from the +//! stuck: it seals a `WEDGED` record naming itself, its `pc` and `sp` from the //! NMI frame, the lock it is spinning on if `Lock::lock` recorded one, a line //! for every other CPU, and the tail of the log ring — then writes the reset //! register through `acpi::reset_now`. @@ -107,9 +107,9 @@ static STOOD_DOWN: AtomicBool = AtomicBool::new(false); static PROGRESS: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; static STILL_SINCE: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; static AT_TSC: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; -static AT_RIP: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; -static AT_RSP: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; -static AT_RFLAGS: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; +static AT_PC: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; +static AT_SP: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; +static AT_FLAGS: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; static ARMED_PMU: [AtomicBool; MAX_CPUS] = [const { AtomicBool::new(false) }; MAX_CPUS]; /// The lock a CPU is spinning on and the `#[track_caller]` site that asked for @@ -132,15 +132,15 @@ static SPIN_AT: [AtomicU64; MAX_CPUS] = [const { AtomicU64::new(0) }; MAX_CPUS]; #[must_use] pub fn start(deadline_ms: u64) -> Armed { let ms = toyos_tco::hard_lockup_bound_ms(deadline_ms); - let ticks = crate::clock::tsc_ticks(ms.saturating_mul(1_000_000)); - let per_ms = crate::clock::tsc_ticks(1_000_000); + let ticks = crate::clock::counter_ticks(ms.saturating_mul(1_000_000)); + let per_ms = crate::clock::counter_ticks(1_000_000); if ms == 0 || ticks == 0 || per_ms == 0 { return Armed { ms: 0, sampled: false }; } BOUND_MS.store(ms, Relaxed); BOUND_TSC.store(ticks, Relaxed); TICKS_PER_MS.store(per_ms, Relaxed); - PERIOD.store(crate::clock::tsc_ticks(SAMPLE_NS), Relaxed); + PERIOD.store(crate::clock::counter_ticks(SAMPLE_NS), Relaxed); arm_this_cpu(); Armed { ms, sampled: has_a_counter(percpu::cpu_id() as usize) } } @@ -225,7 +225,7 @@ pub fn bound_ms() -> u64 { /// Called from `arch::trap::nmi`'s `note` and nowhere else. Returns on every NMI /// that is not this CPU's own overflow, so the diagnostic senders — the blocked /// task dump's probe, the syscall-window storm — cost one load and one compare. -pub fn sample(rip: u64, rsp: u64, rflags: u64) { +pub fn sample(pc: u64, sp: u64, flags: u64) { if BOUND_TSC.load(Relaxed) == 0 || STOOD_DOWN.load(Relaxed) { return; } @@ -241,9 +241,9 @@ pub fn sample(rip: u64, rsp: u64, rflags: u64) { return; } let now = cpu::counter(); - AT_RIP[me].store(rip, Relaxed); - AT_RSP[me].store(rsp, Relaxed); - AT_RFLAGS[me].store(rflags, Relaxed); + AT_PC[me].store(pc, Relaxed); + AT_SP[me].store(sp, Relaxed); + AT_FLAGS[me].store(flags, Relaxed); AT_TSC[me].store(now, Relaxed); let taken = crate::irq_census::taken_here(); @@ -251,10 +251,10 @@ pub fn sample(rip: u64, rsp: u64, rflags: u64) { // `IF` set is the whole difference between this bound and the deadline's: a // CPU that can still take an interrupt is one the timer entry's poll // reaches, and this mechanism is not about it. - if moved || trap::frame_interrupts_enabled(rflags) { + if moved || trap::frame_interrupts_enabled(flags) { STILL_SINCE[me].store(now, Relaxed); } else if now.wrapping_sub(STILL_SINCE[me].load(Relaxed)) >= BOUND_TSC.load(Relaxed) { - locked_up(me, rip, rsp, now) + locked_up(me, pc, sp, now) } // **After the decision and never before it.** Re-arming clears the mask // hardware set on delivery; leaving it set is what stops a second NMI @@ -269,11 +269,11 @@ pub fn sample(rip: u64, rsp: u64, rflags: u64) { /// Seal where every CPU is and hand the machine back. /// /// Runs on the stuck CPU itself, from its own NMI frame, which is the only -/// context that has its `rip`. No CPU is asked anything from here: an NMI sent +/// context that has its `pc`. No CPU is asked anything from here: an NMI sent /// to a sibling already inside its own would enter `nested_nmi` and stop the /// machine instead of resetting it, so the record is built out of what each CPU /// last wrote about itself. -fn locked_up(me: usize, rip: u64, rsp: u64, now: u64) -> ! { +fn locked_up(me: usize, pc: u64, sp: u64, now: u64) -> ! { if !crate::deadline::claim_the_reset() { // Another CPU is already sealing and resetting. This one has nothing to // add and must not race it into the page. @@ -281,7 +281,7 @@ fn locked_up(me: usize, rip: u64, rsp: u64, now: u64) -> ! { } crate::drivers::panic_console::seal_wedge(format_args!( "{}", - Report { me, rip, rsp, now, cpus: (smp::cpu_count() as usize).min(MAX_CPUS) } + Report { me, pc, sp, now, cpus: (smp::cpu_count() as usize).min(MAX_CPUS) } )); // The seal first, because the USB stop `reset_now` makes before it writes // the register is bounded but not instant, and this record is the @@ -370,7 +370,7 @@ impl fmt::Display for Waiting { } } -/// Where a `rip` is, spelled without saying a word: `symbols::resolve_kernel` +/// Where a `pc` is, spelled without saying a word: `symbols::resolve_kernel` /// writes a log record, which is the one thing this path may not do. struct At(u64); @@ -391,15 +391,15 @@ impl fmt::Display for At { /// What the machine looked like from the CPU that ended it. struct Report { me: usize, - rip: u64, - rsp: u64, + pc: u64, + sp: u64, now: u64, cpus: usize, } impl fmt::Display for Report { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - let Self { me, rip, rsp, now, cpus } = *self; + let Self { me, pc, sp, now, cpus } = *self; writeln!( f, "{LOCKED_UP}: cpu{me} has taken no interrupt for {}, with `IF` clear at every sample \ @@ -407,8 +407,8 @@ impl fmt::Display for Report { Ms(now.wrapping_sub(STILL_SINCE[me].load(Relaxed))), BOUND_MS.load(Relaxed), )?; - writeln!(f, " rip={}", At(rip))?; - writeln!(f, " rsp={rsp:#018x}{}", Waiting(me))?; + writeln!(f, " pc={}", At(pc))?; + writeln!(f, " sp={sp:#018x}{}", Waiting(me))?; // Every CPU, and each line is what that CPU last wrote about itself: // the holder of whatever the stuck one is waiting for is somewhere in // this list, and nothing else in a wedged machine can point at it. @@ -432,8 +432,8 @@ impl fmt::Display for Report { " cpu{cpu} irqs={irqs} (={} when sampled {} ago) if={} at {}{}", PROGRESS[cpu].load(Relaxed), Ms(now.wrapping_sub(at)), - u8::from(trap::frame_interrupts_enabled(AT_RFLAGS[cpu].load(Relaxed))), - At(AT_RIP[cpu].load(Relaxed)), + u8::from(trap::frame_interrupts_enabled(AT_FLAGS[cpu].load(Relaxed))), + At(AT_PC[cpu].load(Relaxed)), Waiting(cpu), )?; } diff --git a/kernel/src/hardlockup/probe.rs b/kernel/src/hardlockup/probe.rs index 87e6b23721..c68a6204df 100644 --- a/kernel/src/hardlockup/probe.rs +++ b/kernel/src/hardlockup/probe.rs @@ -139,7 +139,7 @@ fn go_deaf(me: usize, bound_ms: u64) -> ! { while cpu::counter() < until { core::hint::spin_loop(); } - // Never acquires. The detector's NMI is what ends this CPU, and its `rip` + // Never acquires. The detector's NMI is what ends this CPU, and its `pc` // is inside `Lock::lock`'s spin when it does. let _never = PROBE_LOCK.lock(); loop { diff --git a/kernel/src/hw.rs b/kernel/src/hw.rs new file mode 100644 index 0000000000..c3268132ba --- /dev/null +++ b/kernel/src/hw.rs @@ -0,0 +1,164 @@ +//! `KernelHw` — the kernel's side of the scheduler-core hardware boundary: the +//! one-shot timer, the kick, the halt and the context switch. Nothing here +//! makes a scheduling decision; the simulator that exercises the scheduler +//! core replaces this and nothing else. +//! +//! The halt and the switch are the architecture's (`crate::arch::hw`); +//! everything else is here, once, reaching the machine through +//! `crate::arch::irqchip`. + +use core::sync::atomic::{AtomicU64, Ordering::Relaxed}; + +use toyos_sched::cpu::SleepToken; +use toyos_sched::fair::QUANTUM_NS; +use toyos_sched::hw::{CpuId, Kicker, Machine, Nanos, TraceEvent}; + +use crate::arch::{irqchip, percpu}; +use crate::sched::payload::KernelCtx; +use crate::time::{Duration, Floor}; + +/// The one instance; zero-sized, holds no per-CPU state. +pub static HW: KernelHw = KernelHw; + +pub struct KernelHw; + +/// The scheduler clock, in raw nanoseconds. +pub fn now_ns() -> u64 { + HW.now().0 +} + +/// The shortest one-shot either timer is armed for, whoever asks: a count +/// that expires before the interrupt it schedules retires re-fires on the +/// return from it, and a context held with interrupts open makes no progress. +pub(crate) const MIN_ONE_SHOT: Floor = + Floor::policy(Duration::from_micros(10), "above an interrupt's entry and return"); + +impl Kicker for KernelHw { + fn kick(&self, target: CpuId) { + irqchip::kick_cpu(target.0); + } +} + +impl Machine for KernelHw { + fn now(&self) -> Nanos { + Nanos(crate::clock::nanos_since_boot()) + } + + /// The absolute deadline as the one-shot's span from now; one already + /// past becomes [`MIN_ONE_SHOT`] rather than an interrupt at once. + fn set_timer(&self, deadline: Nanos) { + irqchip::arm_one_shot(deadline.0.saturating_sub(self.now().0)); + } + + fn stop_timer(&self) { + irqchip::stop_timer(); + } + + fn halt(&self) { + crate::arch::hw::halt(); + } + + /// A kick is how a remote CPU's `need_resched` gets set — there is no way to write it directly. + fn need_resched(&self, cpu: CpuId) { + if cpu.0 == percpu::cpu_id() { + crate::preempt::set_need_resched(); + } else { + self.kick(cpu); + } + } + + fn trace(&self, ev: TraceEvent) { + crate::trace::record(ev); + } + + /// Diagnostic builds arm a periodic wake before halting so a quiescent CPU still reports. + fn idle_wait(&self, token: SleepToken) { + let _consumed = token; + #[cfg(feature = "boot-actuators")] + if crate::actuator::diag_tick() { + irqchip::arm_within(DIAG_TICK_NS); + } + self.halt(); + // **A CPU that is executing has a one-shot armed, and this is where + // that becomes true again.** `TimerPlan::Stop` left this one at zero + // before the halt above and only a pass reaching `apply_timer` arms + // another, so a CPU woken by a kick or a device — never by its own + // timer, which is stopped — and then held in the kernel takes no timer + // interrupt at all. `crate::deadline`'s poll and `crate::hardlockup`'s + // sample both rest on some CPU taking one. Arming earlier than the + // scheduler planned is a spurious pass and never a missed deadline + // (`toyos_sched::timer::TimerPlan`), and the next pass replaces it + // either way. + // + // Asked, because it is only those two that need it: a boot under no + // bound pays a timer read and two writes per wake for nothing. + if crate::deadline::armed() { + irqchip::arm_within(QUANTUM_NS); + } + } +} + +/// Longest sleep on a `diag-tick` build; kept under `heartbeat`'s reporting period so a healthy CPU reports on every line. +#[cfg(feature = "boot-actuators")] +const DIAG_TICK_NS: u64 = 100_000_000; + +/// Which context each CPU last switched onto; read by [`report_contexts`] on crash, since a +/// sibling's real `CpuSched` is `!Sync` and unreadable directly. +static RUNNING_CTX: [AtomicU64; crate::sched::MAX_CPUS] = + [const { AtomicU64::new(0) }; crate::sched::MAX_CPUS]; + +/// This CPU is about to stand on `ctx`: `Hw::switch`'s last word before the stack moves. +pub(crate) fn note_running(ctx: *const KernelCtx) { + RUNNING_CTX[percpu::cpu_id() as usize].store(ctx as u64, Relaxed); +} + +/// Prints which CPU is standing on which context and stack, on every kernel crash. +/// +/// `subject` (`None` for this CPU's own) is the context flagged as "the same". +/// +/// Allocates, locks or formats nothing but integers, since a crash may already hold any lock this +/// could try to take. +pub fn report_contexts(sp: u64, subject: Option) { + let me = percpu::cpu_id() as usize; + let count = (crate::arch::smp::cpu_count() as usize).min(crate::sched::MAX_CPUS); + let mine = RUNNING_CTX.get(me).map_or(0, |slot| slot.load(Relaxed)); + let subject = subject.unwrap_or(mine); + crate::log!(" Contexts: cpu{me} crashed at sp={sp:#018x}, asking about ctx {subject:#x}"); + for (cpu, slot) in RUNNING_CTX.iter().enumerate().take(count) { + let held = slot.load(Relaxed); + if !crate::mm::is_kernel_addr(held) || !held.is_multiple_of(8) { + crate::log!(" cpu{cpu} is on ctx {held:#x} (never switched, or not a context)"); + continue; + } + // SAFETY: `held` is a pointer this kernel's own `Hw::switch` stored, into the boxed, always-mapped direct map. + let ctx = unsafe { &*(held as *const KernelCtx) }; + let top = ctx.kernel_stack_top; + let same = held == subject && cpu != me; + // `top != 0` excludes idle contexts, whose stack top is zero by construction — the + // containment test below never fires for one; that is a gap in this report, not a bug. + let on_its_stack = cpu != me + && top != 0 + && sp <= top + && sp > top.wrapping_sub(crate::process::KERNEL_STACK_SIZE as u64); + // idle's `kernel_stack_top` is zero by construction; rendering it as a task would misread as corruption. + match ctx.id { + None => crate::log!( + " cpu{cpu} is on ctx {held:#x} (its idle context) stack_top={top:#018x} \ + saved_sp={:#018x}{}{}", + ctx.sp, + if same { " <== THE SAME CONTEXT" } else { "" }, + if top == 0 { "" } else { " <== AN IDLE CONTEXT'S STACK TOP IS ZERO BY CONSTRUCTION" }, + ), + Some(id) => crate::log!( + " cpu{cpu} is on ctx {held:#x} pid={} tid={} stack_top={top:#018x} \ + saved_sp={:#018x}{}{}", + id.0.raw(), + id.1.raw(), + ctx.sp, + if same { " <== THE SAME CONTEXT" } else { "" }, + if on_its_stack { " <== AND THIS CRASH IS ON THAT STACK" } else { "" }, + ), + } + } + crate::mm::report_on_crash(); +} diff --git a/kernel/src/iod.rs b/kernel/src/iod.rs index 263612ff59..7fddc9a0a4 100644 --- a/kernel/src/iod.rs +++ b/kernel/src/iod.rs @@ -25,7 +25,7 @@ extern "C" fn body(_arg: u64) -> ! { // The SS-reload probe rides iod rather than spawning a task of its own. #[cfg(feature = "boot-actuators")] if crate::actuator::sysret_ss_probe() { - crate::hw::sysret_ss_probe(&parkable); + crate::arch::hw::sysret_ss_probe(&parkable); } // Held across the loop: a push during a drain must still find this watch armed. let armed = watch::arm( diff --git a/kernel/src/loader/mod.rs b/kernel/src/loader/mod.rs index ddd821de37..3203ad6200 100644 --- a/kernel/src/loader/mod.rs +++ b/kernel/src/loader/mod.rs @@ -547,7 +547,7 @@ pub fn spawn( }; log!("spawn: TLS {} modules, total_memsz={}", tls_modules.len(), tls.total_memsz()); - let Some((tls_pages, fs_base, _)) = + let Some((tls_pages, thread_pointer, _)) = tls::TlsBlock::build(&tls_modules, tls).and_then(|b| b.publish(&child_pt)) else { log!("spawn: {}: failed to allocate TLS ({} bytes)", path, tls.total_memsz()); @@ -566,7 +566,7 @@ pub fn spawn( ); let sym_bytes = syms.resident_bytes(); - let (ks_alloc, ks_rsp) = match alloc_kernel_stack(process_start, entry, sp, 0) { + let (ks_alloc, ks_sp) = match alloc_kernel_stack(process_start, entry, sp, 0) { Some(ks) => ks, None => { log!("spawn: {}: failed to allocate kernel stack", path); @@ -642,9 +642,9 @@ pub fn spawn( let (sched, dst) = scheduler::enqueue_new( scheduler::TaskId(pid, tid), ks_alloc, - ks_rsp, + ks_sp, child_pt.clone(), - fs_base, + thread_pointer, syms, ); table.get_mut(pid).unwrap().threads_mut().get_mut(tid).unwrap().set_sched(sched); diff --git a/kernel/src/loader/tls.rs b/kernel/src/loader/tls.rs index 466738d9af..75fb1a07cf 100644 --- a/kernel/src/loader/tls.rs +++ b/kernel/src/loader/tls.rs @@ -57,8 +57,8 @@ impl TlsBlock { // by no mapping until `publish` maps it after this returns. unsafe { rebase(frames, tp_offset, at) } })?; - let fs_base = (pages.vaddr() + tp_offset as u64).raw(); - Some((pages, fs_base, tp_offset)) + let thread_pointer = (pages.vaddr() + tp_offset as u64).raw(); + Some((pages, thread_pointer, tp_offset)) } } diff --git a/kernel/src/main.rs b/kernel/src/main.rs index f328ff347e..e77c5dd828 100644 --- a/kernel/src/main.rs +++ b/kernel/src/main.rs @@ -73,6 +73,7 @@ mod process; mod loader; mod scheduler; mod sched; +mod hw; mod iommu; mod preempt; mod irq_census; @@ -115,7 +116,6 @@ use crate::mm::policy::MmioPolicy; use alloc::boxed::Box; use alloc::sync::Arc; use arch::{cpu, percpu, smp}; -pub(crate) use arch::hw; use drivers::{acpi, gop, nvme, pci, serial, virtio_console, virtio_gpu, virtio_sound, xhci}; use toyos_abi::boot::{KernelArgs, MemoryMapEntry}; use toyos_rootimage::handoff::{held, Descriptor}; @@ -654,15 +654,17 @@ pub(crate) unsafe extern "C" fn kernel_main(kernel_args: &KernelArgs) -> ! { register_gpu(gpu_driver, gpu_info); } else if kernel_args.gop_framebuffer != 0 { log!("GPU: using UEFI GOP"); - let (gpu_driver, gpu_info) = gop::init( + match gop::init( kernel_args.gop_framebuffer, kernel_args.gop_framebuffer_size, kernel_args.gop_width, kernel_args.gop_height, kernel_args.gop_stride, kernel_args.gop_pixel_format, - ); - register_gpu(gpu_driver, gpu_info); + ) { + Some((gpu_driver, gpu_info)) => register_gpu(gpu_driver, gpu_info), + None => log!("GPU: none this boot, running headless"), + } } else { log!("GPU: none found, running headless"); } diff --git a/kernel/src/mm/mod.rs b/kernel/src/mm/mod.rs index 304e0dd18b..609b9fd3ae 100644 --- a/kernel/src/mm/mod.rs +++ b/kernel/src/mm/mod.rs @@ -159,7 +159,7 @@ impl core::fmt::Debug for DirectMap { pub fn init(memory_map: &[MemoryMapEntry], reserved: &[Region]) { alloc::init_early(); pmm::init(memory_map, reserved); - DIRECT_MAP_END.set(paging::init(memory_map)); + DIRECT_MAP_END.set(paging::init(memory_map, crate::drivers::panic_console::scanout())); alloc::init(); paging::seal_kernel_half(); } diff --git a/kernel/src/process.rs b/kernel/src/process.rs index f8680f7be2..6dc27af382 100644 --- a/kernel/src/process.rs +++ b/kernel/src/process.rs @@ -842,7 +842,7 @@ pub fn spawn_thread(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> Op // block's address is chosen from, so every pointer in it is final before // the mapping exists. let block = TlsBlock::build(&tls_modules, tls)?; - let (tls_alloc, fs_base, tp_offset) = { + let (tls_alloc, thread_pointer, tp_offset) = { let parent_data = process_data_arc.lock(); if crate::actuator::tls_rebase_window() { crate::loader::rebase_window::spawning(arg); @@ -854,7 +854,7 @@ pub fn spawn_thread(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> Op }; let tls_alloc_tcb = tls_alloc.ptr().wrapping_add(tp_offset); - let (ks_alloc, ks_rsp) = match alloc_kernel_stack(thread_start, entry, stack_ptr, arg) { + let (ks_alloc, ks_sp) = match alloc_kernel_stack(thread_start, entry, stack_ptr, arg) { Some(ks) => ks, None => { tls_alloc.release(&parent_addr_space); @@ -905,9 +905,9 @@ pub fn spawn_thread(entry: u64, stack_ptr: u64, arg: u64, stack_base: u64) -> Op let (sched, _dst) = scheduler::enqueue_new( TaskId(parent_process, tid), ks_alloc, - ks_rsp, + ks_sp, parent_addr_space, - fs_base, + thread_pointer, symbols, ); proc.threads.get_mut(tid).unwrap().set_sched(sched); @@ -1474,24 +1474,26 @@ pub fn dump_crash_diagnostics(fault_addr: u64, rip: u64) { } dump_region("rip", rip); - let fs_base = crate::arch::cpu::thread_pointer(); - if fs_base != 0 { - log!(" FS base: {:#x}", fs_base); - if let Some(self_ptr) = read_user(fs_base) { - log!(" fs:[0] = {:#x} (expected {:#x})", self_ptr, fs_base); + let tp = crate::arch::cpu::thread_pointer(); + if tp != 0 { + log!(" Thread pointer: {:#x}", tp); + if let Some(first) = read_user(tp) { + if matches!(crate::loader::TLS_VARIANT, toyos_elf::tls::Variant::II) { + log!(" [TP] = {:#x} (expected {:#x}, variant II's self-pointer)", first, tp); + } for i in 0..8u64 { - let addr = fs_base + i * 8; + let addr = tp + i * 8; let Some(val) = read_user(addr) else { break }; log!(" TP+{:#x} = {:#018x}", i * 8, val); } log!(" TLS data before TP:"); for i in 1..=4u64 { - let addr = fs_base - i * 8; + let addr = tp - i * 8; let Some(val) = read_user(addr) else { break }; log!(" TP-{:#x} = {:#018x}", i * 8, val); } } else { - log!(" FS base {:#x} NOT MAPPED!", fs_base); + log!(" Thread pointer {:#x} NOT MAPPED!", tp); } } } diff --git a/kernel/src/quiesce.rs b/kernel/src/quiesce.rs index 216e94ceb9..b343638287 100644 --- a/kernel/src/quiesce.rs +++ b/kernel/src/quiesce.rs @@ -6,7 +6,7 @@ //! //! # Where the stop is taken, and why there //! -//! [`stops_this_thread`] is read by `scheduler::leave_ring3_if_due`, called +//! [`stops_this_thread`] is read by `scheduler::leave_user_if_due`, called //! from `kernel_exit_to_user_check` — the one function every return to Ring 3 //! in this kernel passes through: the syscall gate, every device interrupt, the //! timer, the TLB shootdown IPI, the general trap epilogue and a task's first diff --git a/kernel/src/sched/driver.rs b/kernel/src/sched/driver.rs index 43233c1261..7b9797324a 100644 --- a/kernel/src/sched/driver.rs +++ b/kernel/src/sched/driver.rs @@ -374,9 +374,9 @@ pub fn init() { /// The context a CPU runs on when idle — never a dead task's stack, so a pass can free the previous zombie. fn idle_ctx() -> KernelCtx { KernelCtx { - rsp: 0, + sp: 0, root: crate::mm::paging::kernel_root(), - fs_base: 0, + thread_pointer: 0, kernel_stack_top: 0, id: None, // Never read: the idle loop is entered by jump, not switch, and @@ -394,15 +394,15 @@ fn placement(now: Nanos) -> CpuId { cpus().place(start, now) } -/// Everything a new thread needs. `entry_rsp` points at the trampoline frame `alloc_kernel_stack` built; +/// Everything a new thread needs. `entry_sp` points at the trampoline frame `alloc_kernel_stack` built; /// `address_space` is not `Option` — every kernel thread uses the kernel address space, so one declaration /// decides `cr3`. pub struct NewTask { pub id: TaskId, pub kernel_stack: OwnedAlloc, - pub entry_rsp: u64, + pub entry_sp: u64, pub address_space: PageTables, - pub fs_base: u64, + pub thread_pointer: u64, pub share: Arc, /// The process's symbol table; a kernel thread names an empty one. pub symbols: Arc, @@ -416,9 +416,9 @@ pub fn spawn(new: NewTask) -> (ThreadSched, CpuId) { let root = new.address_space.lock().root(); let kernel_stack_top = new.kernel_stack.ptr() as u64 + KERNEL_STACK_SIZE as u64; let ctx = KernelCtx { - rsp: new.entry_rsp, + sp: new.entry_sp, root, - fs_base: new.fs_base, + thread_pointer: new.thread_pointer, kernel_stack_top, id: Some(new.id), // The one level `trampoline_entry` discharges before the first `iretq`. @@ -864,7 +864,7 @@ pub fn for_each_parked(mut f: impl FnMut(ParkedInfo)) -> bool { /// preempt-count bracket's other half is owed. pub extern "C" fn trampoline_entry() { crate::preempt::enable_no_resched(); - crate::arch::trap::kernel_exit_to_user_check(); + crate::scheduler::exit_to_user(); } const STACK_CANARY: u64 = 0xDEAD_BEEF_CAFE_BABE; @@ -939,30 +939,30 @@ fn check_stack_canary(payload: &KernelPayload) { /// Does every Ring 3 → Ring 0 entry land on the stack of the task this CPU is running, and is this CPU standing on it? /// -/// A stray `kernel_rsp` or `tss.rsp0` aims a future entry at a stack it did not grow; this catches it before -/// that entry, not after. +/// A stray entry stack aims a future entry at a stack it did not grow; this catches it before that entry, not +/// after. #[cfg(feature = "stack-witness")] fn check_stack_ownership(payload: &KernelPayload) { let bottom = payload.kernel_stack.ptr() as u64; let top = bottom + KERNEL_STACK_SIZE as u64; // SAFETY: a pass runs on the CPU whose GS base is its own `PerCpu`. - let (kernel_rsp, rsp0) = unsafe { percpu::entry_stacks() }; - let rsp = crate::arch::cpu::stack_pointer(); - if kernel_rsp == top && rsp0 == top && rsp <= top && rsp > bottom { + let (entry, interrupt) = unsafe { percpu::entry_stacks() }; + let sp = crate::arch::cpu::stack_pointer(); + if entry == top && interrupt == top && sp <= top && sp > bottom { return; } panic!( "STACK WITNESS: cpu{} is passing on tid={} whose stack is \ - [{bottom:#018x}, {top:#018x}) — kernel_rsp={kernel_rsp:#018x} \ - (off by {}), tss.rsp0={rsp0:#018x} (off by {}), rsp={rsp:#018x} \ - ({} bytes below the top). A Ring 3 entry takes its stack from one of \ + [{bottom:#018x}, {top:#018x}) — the syscall entry's stack {entry:#018x} \ + (off by {}), the interrupt entry's {interrupt:#018x} (off by {}), sp={sp:#018x} \ + ({} bytes below the top). An entry from user mode takes its stack from one of \ those two words, so one that is not this task's top aims the next \ entry's return addresses into memory another execution owns.", percpu::cpu_id(), payload.id.1, - kernel_rsp.wrapping_sub(top) as i64, - rsp0.wrapping_sub(top) as i64, - top.wrapping_sub(rsp) as i64, + entry.wrapping_sub(top) as i64, + interrupt.wrapping_sub(top) as i64, + top.wrapping_sub(sp) as i64, ); } diff --git a/kernel/src/sched/dump.rs b/kernel/src/sched/dump.rs index aed0ffd54d..3a5a8ccd2f 100644 --- a/kernel/src/sched/dump.rs +++ b/kernel/src/sched/dump.rs @@ -506,8 +506,8 @@ pub mod staged { } } - /// From the Ring 3 exit check, once it has nothing more to run. - pub fn note_return_to_ring3() { + /// From the exit to user mode, once it has nothing more to run. + pub fn note_return_to_user() { if STAGED.load(Ordering::Acquire) != NOTHING && REQUEST.pending() { RETURNS.fetch_add(1, Ordering::AcqRel); } diff --git a/kernel/src/sched/idle_stack.rs b/kernel/src/sched/idle_stack.rs new file mode 100644 index 0000000000..b108143034 --- /dev/null +++ b/kernel/src/sched/idle_stack.rs @@ -0,0 +1,88 @@ +//! Every CPU's idle stack: one per CPU, the size of a task's kernel stack, above +//! a guard page the direct map does not hold, so running off its bottom is a +//! fault rather than a write into memory another execution owns. +//! +//! **Taken from 2 MiB pages of their own, not the kernel heap**: the guard's +//! hole in the direct map splits the page that holds it, and a heap page split +//! so would cost every other allocation in it its 2 MiB translation. Never +//! freed — a page returned to the PMM would keep the hole. + +use alloc::vec::Vec; + +use crate::mm::DirectMap; +use crate::sync::Lock; + +/// Same size as a task's kernel stack: a `deferred` [`kobject!`] object may +/// own an `immediate` one, whose destructor then runs here instead. +pub const SIZE: usize = crate::process::KERNEL_STACK_SIZE; + +/// One unmapped 4 KiB page below every idle stack. +const GUARD: usize = crate::mm::PAGE_BYTES; + +/// What an untouched byte of a filled stack holds: chosen so a zeroed or ASCII +/// byte cannot be mistaken for one. +pub const FILL: u8 = 0xA5; +pub const FILL_WORD: u64 = u64::from_ne_bytes([FILL; 8]); + +/// One idle stack and the guard page under it. +const SLOT: usize = GUARD + SIZE; + +static ARENA: Lock = Lock::new(Arena { pages: Vec::new(), stacks: Vec::new(), next: 0, left: 0 }); + +struct Arena { + pages: Vec, + /// The bottom of every idle stack, so the deepest any CPU has gone reads from one. + stacks: Vec, + /// Direct-map address of the next free slot. + next: u64, + left: usize, +} + +/// A 4 KiB-aligned [`SLOT`] from the arena. +fn alloc_slot() -> u64 { + let mut arena = ARENA.lock(); + if arena.left < SLOT { + let page = crate::mm::pmm::alloc_page(crate::mm::pmm::Category::KernelHeap) + .expect("idle stack: no physical page for one"); + arena.next = page.direct_map().as_mut_ptr::() as u64; + arena.left = crate::mm::PAGE_2M as usize; + arena.pages.push(page); + } + let base = arena.next; + arena.next += SLOT as u64; + arena.left -= SLOT; + arena.stacks.push(base + GUARD as u64); + base +} + +/// A fresh idle stack, filled, over its unmapped guard page: its top. +pub fn alloc() -> u64 { + let base = alloc_slot(); + crate::mm::paging::kernel().lock().guard_4k(DirectMap::phys_of(base as *const u8)); + // SAFETY: exactly `SIZE` bytes above the unmapped guard, within the slot + // `alloc_slot` returned — filled, not zeroed, so zero can't mark + // "untouched" for [`high_water`]. + unsafe { core::ptr::write_bytes((base + GUARD as u64) as *mut u8, FILL, SIZE) }; + base + SLOT as u64 +} + +/// The deepest any CPU's idle stack has ever been, in bytes, read from the +/// bottom up: nothing legitimate writes [`FILL`], so a touched byte stays changed. +#[cfg(feature = "test-actuators")] +pub fn high_water() -> usize { + let arena = ARENA.lock(); + arena + .stacks + .iter() + .map(|&bottom| SIZE - words(bottom, SIZE).take_while(|&w| w == FILL_WORD).count() * 8) + .max() + .unwrap_or(0) +} + +/// Sequential u64s from `base`; every address is inside the caller's +/// already-bounds-checked allocation. +pub fn words(base: u64, len: usize) -> impl Iterator { + // SAFETY: `i < len/8` bounds each address inside the caller's checked + // allocation; `read_volatile` keeps the fill-pattern read. + (0..len / 8).map(move |i| unsafe { core::ptr::read_volatile((base as *const u64).add(i)) }) +} diff --git a/kernel/src/sched/kthread.rs b/kernel/src/sched/kthread.rs index baa4c07fd6..03b65a2a71 100644 --- a/kernel/src/sched/kthread.rs +++ b/kernel/src/sched/kthread.rs @@ -91,7 +91,7 @@ pub fn open_selftest() { /// Start a kernel thread running `body(arg)` on its own kernel stack and return its scheduler faces. pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64) -> ThreadSched { - let (stack, entry_rsp) = crate::loader::alloc_kernel_stack( + let (stack, entry_sp) = crate::loader::alloc_kernel_stack( crate::loader::kernel_start, body as usize as u64, 0, @@ -127,7 +127,7 @@ pub fn spawn(name: &str, body: extern "C" fn(u64) -> !, arg: u64) -> ThreadSched let (sched, _dst) = scheduler::enqueue_new( TaskId(pid, tid), stack, - entry_rsp, + entry_sp, crate::mm::paging::kernel().clone(), 0, syms, diff --git a/kernel/src/sched/mod.rs b/kernel/src/sched/mod.rs index 0a7a33898b..71df61c919 100644 --- a/kernel/src/sched/mod.rs +++ b/kernel/src/sched/mod.rs @@ -9,6 +9,7 @@ pub mod kthread; pub mod payload; pub mod reap_gate; pub mod futex; +pub mod idle_stack; /// Ceiling on CPUs the percpu arrays are sized for. pub const MAX_CPUS: usize = 8; diff --git a/kernel/src/sched/payload.rs b/kernel/src/sched/payload.rs index 0e2aa7dfbc..523e2a6ede 100644 --- a/kernel/src/sched/payload.rs +++ b/kernel/src/sched/payload.rs @@ -40,12 +40,13 @@ pub type KShare = FairShare>; /// The core's wait ticket; blocking sites use `driver::Ticket`, which wraps it in the needed preempt guard. pub type RawTicket = WaitTicket; -/// The saved callee context; everything `Hw::switch` must load without dereferencing anything else. +/// The saved callee context; everything `Hw::switch` must load without dereferencing anything else, named +/// by role because every architecture's switch loads it. pub struct KernelCtx { /// Saved kernel stack pointer, written by the `context_switch` asm. - pub rsp: u64, + pub sp: u64, pub root: Root, - pub fs_base: u64, + pub thread_pointer: u64, pub kernel_stack_top: u64, /// `None` is this CPU's idle context. pub id: Option, diff --git a/kernel/src/scheduler.rs b/kernel/src/scheduler.rs index e35e2301e9..16ee7a3f99 100644 --- a/kernel/src/scheduler.rs +++ b/kernel/src/scheduler.rs @@ -293,17 +293,17 @@ pub fn global_min_vruntime() -> u64 { pub fn enqueue_new( id: TaskId, kernel_stack: crate::process::OwnedAlloc, - entry_rsp: u64, + entry_sp: u64, address_space: crate::process::PageTables, - fs_base: u64, + thread_pointer: u64, symbols: alloc::sync::Arc, ) -> (ThreadSched, CpuId) { driver::spawn(NewTask { id, kernel_stack, - entry_rsp, + entry_sp, address_space, - fs_base, + thread_pointer, share: share_for(id.0), symbols, }) @@ -348,7 +348,7 @@ pub fn may_yield() -> bool { crate::preempt::count() == blocking_baseline() } -/// Unified preempt entry: the Ring 3 timer path, `kernel_exit_to_user_check` +/// Unified preempt entry: the user-mode timer path, [`exit_to_user`] /// and the `preempt::enable` slow path all funnel through here. #[track_caller] pub fn do_preempt() { @@ -367,15 +367,15 @@ pub fn do_preempt() { driver::pass(Dispose::None); } -/// The last thing a thread does before returning to Ring 3, if either mark it -/// can carry says it never does. `kernel_exit_to_user_check` is the one caller; +/// The last thing a thread does before returning to user mode, if either mark it +/// can carry says it never does. [`exit_to_user`] is the one caller; /// `kernel/src/quiesce.rs`'s header says why that boundary is the safe point. /// /// **One call and one match, so the two marks have no order to disagree /// about**: `toyos_sched::task::SafePoint` ranks them, here and in /// `CpuSched::place` alike. #[track_caller] -pub fn leave_ring3_if_due() { +pub fn leave_user_if_due() { let Some(due) = driver::current_safe_point(crate::quiesce::stops_this_thread()) else { return; }; @@ -383,21 +383,56 @@ pub fn leave_ring3_if_due() { match due { SafePoint::Stop => { driver::pass(Dispose::Stop); - unreachable!("leave_ring3_if_due: a stopped task was dispatched again"); + unreachable!("leave_user_if_due: a stopped task was dispatched again"); } SafePoint::Exit => { - // `IF` set across the teardown, as a syscall's exit runs it: its + // Interrupts open across the teardown, as a syscall's exit runs it: its // closes and address-space drop are no interrupt latency. The // depth stays this boundary's, which is `do_preempt`'s own. crate::arch::cpu::enable_interrupts(); process::leave(None); crate::arch::cpu::disable_interrupts(); driver::pass(Dispose::Exit); - unreachable!("leave_ring3_if_due: returned from the exit pass"); + unreachable!("leave_user_if_due: returned from the exit pass"); } } } + +/// The deferred-preempt epilogue every return to user mode runs last, with +/// interrupts masked on entry and on return: a killed or stopped thread leaves +/// here, and a reschedule owed since the entry is served before the thread +/// sees user mode again. +pub fn exit_to_user() { + flush_kernel_timer_fires_to_trace(); + loop { + // A killed or stopped thread returns to user mode exactly once more: never. + leave_user_if_due(); + // `do_preempt` owns clearing `need_resched`; this function never clears it itself. + if !crate::preempt::need_resched() { + #[cfg(feature = "boot-actuators")] + if crate::actuator::dump_in_blocking_pass() { + crate::sched::dump::staged::note_return_to_user(); + } + return; + } + assert!(!in_schedule_self(), "exit-to-user inside a scheduler pass"); + // Not an IrqGuard: both loop exits must open interrupts, not restore a saved value. + crate::arch::cpu::enable_interrupts(); + do_preempt(); + crate::arch::cpu::disable_interrupts(); + flush_kernel_timer_fires_to_trace(); + } +} + +fn flush_kernel_timer_fires_to_trace() { + let cur = percpu::kernel_timer_fires(); + let missed = cur.wrapping_sub(percpu::last_seen_kernel_timer_fires()); + if missed > 0 { + crate::trace::trace(crate::trace::Kind::TimerFireBurst, missed); + percpu::set_last_seen_kernel_timer_fires(cur); + } +} /// The exit pass of a thread that has left its process (`process::leave`). #[track_caller] pub fn exit_current() -> ! { diff --git a/kernel/src/vma.rs b/kernel/src/vma.rs index 38903b6b58..f83f2c7319 100644 --- a/kernel/src/vma.rs +++ b/kernel/src/vma.rs @@ -1,9 +1,10 @@ +use alloc::collections::BTreeMap; use alloc::sync::Arc; use crate::file_backing::FileBacking; use crate::mm::policy::Prot; -use crate::mm::PAGE_2M; -use toyos_userbound::Window; +use crate::mm::{UserAddr, PAGE_2M}; +use toyos_userbound::{PageSpan, Window}; /// The stack extends upward to the PIE base, so no usable VA space exists above it. pub const ALLOC_CEILING: u64 = STACK_BASE; @@ -61,3 +62,78 @@ pub enum Occupancy { /// Part of a region, several regions, or one that merely starts here. Partial, } + +/// Every region an address space registers, keyed by start: where a new one +/// goes and what an address falls in. The page tables are the architecture's +/// (`mm::paging::AddressSpace`), and this is the half both share. +#[derive(Default)] +pub struct Regions(BTreeMap); + +impl Regions { + /// Where `span` goes, top-down and never below the floor: a region the + /// kernel placed under it, the clock page, bounds no gap. + fn find_gap(&self, span: PageSpan) -> Option { + let taken = self.0.iter().rev().map(|(start, region)| (start.raw(), region.size)); + window().gap(span, taken).map(UserAddr::new) + } + + /// Allocate a virtual address range and register the region. `size` is + /// made a [`PageSpan`] before anything is summed on it. + pub fn alloc(&mut self, size: u64, kind: RegionKind) -> Option { + let span = window().span(size)?; + let addr = self.find_gap(span)?; + self.0.insert(addr, Region { size: span.bytes(), kind }); + Some(addr) + } + + /// A [`RegionKind::Mapped`] region for `size` bytes: its address and its + /// size in whole pages, which the caller maps. + pub fn alloc_mapped(&mut self, size: u64) -> Option<(UserAddr, u64)> { + let span = window().span(size)?; + let addr = self.find_gap(span)?; + let aligned = span.bytes(); + self.0.insert(addr, Region { size: aligned, kind: RegionKind::Mapped }); + Some((addr, aligned)) + } + + /// Unregister the region at `addr`, answering its size. + pub fn remove(&mut self, addr: UserAddr) -> Option { + Some(self.0.remove(&addr)?.size) + } + + /// Insert a region at a specific address (for ELF segments, stack, etc.) + pub fn insert(&mut self, addr: UserAddr, region: Region) { + assert!(self.find(addr).is_none(), "insert_region: address {:#x} already occupied", addr.raw()); + self.0.insert(addr, region); + } + + /// Find the region containing `addr`. Returns (start_addr, region). + pub fn find(&self, addr: UserAddr) -> Option<(UserAddr, &Region)> { + let (&start, region) = self.0.range(..=addr).next_back()?; + if addr.raw() < start.raw() + region.size { + Some((start, region)) + } else { + None + } + } + + /// The end is saturating so a caller's arithmetic cannot wrap into a smaller range. + pub fn occupancy(&self, addr: UserAddr, size: u64) -> Occupancy { + let end = UserAddr::new(addr.raw().saturating_add(size)); + let mut over = self.overlapping(addr, end); + let Some((&start, region)) = over.next() else { + return Occupancy::Free; + }; + if over.next().is_none() && start == addr && region.size == size { + Occupancy::Whole + } else { + Occupancy::Partial + } + } + + /// Iterate all regions that overlap the range [start, end). + pub fn overlapping(&self, start: UserAddr, end: UserAddr) -> impl Iterator { + // Overlaps [start, end) iff s < end && s+n > start; `range(..end)` prunes the first half. + self.0.range(..end).filter(move |(&s, r)| s.raw() + r.size > start.raw()) + } +} diff --git a/src/build.rs b/src/build.rs index b52cf50094..d78ce57a10 100644 --- a/src/build.rs +++ b/src/build.rs @@ -2188,11 +2188,6 @@ fn host_judge(root: &Path, (dir, bin): Judge) -> PathBuf { /// nothing in the tree can produce any more — into the ROOT image, into the test list, /// and over the name of whatever gets it next. pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) -> Vec<(String, Vec)> { - let target = arch.userland(); - let mut lock = buildlock::shared(root, "test binaries"); - let sysroot = crate::toolchain::ensure(root, false, &mut lock); - let env = GuestEnv::new(&sysroot); - let mut targets = vec![(crate_path.to_path_buf(), Clean::All)]; for entry in fs::read_dir(crate_path).into_iter().flatten().flatten() { let sub_path = entry.path(); @@ -2200,17 +2195,11 @@ pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) targets.push((sub_path, Clean::All)); } } - invalidate_stale(root, &mut lock, &env.toolchain, &targets); + let build = TestBuild::begin(root, arch, "test binaries", &targets); - let mut results = Vec::new(); + let (target, env) = (build.target, &build.env); - // Every build→read pair below is under one hold, for the reason the - // "Artifact staging" section above gives: cargo keys an artifact path on - // (crate, target, profile), so a second `cargo test` in this tree writes the - // very `.so` and test binaries this one reads back. Between the `read_dir` - // and the `read` that was enough to kill a run outright — four concurrent - // suites, one dead on `Result::unwrap()` on a `NotFound` naming no file. - let _artifact = buildlock::artifact(root); + let mut results = Vec::new(); // Build cdylib subcrates first let mut lib_search_dirs = Vec::new(); @@ -2233,7 +2222,7 @@ pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) if !quiet { eprintln!("[build] Building cdylib subcrate: {lib_name}"); } - cargo_build(&sub_path, target, &[], &env, &[], quiet); + cargo_build(&sub_path, target, &[], env, &[], quiet); let lib_out = sub_path.join(format!("target/{target}/{PROFILE}")); lib_search_dirs.push(lib_out.clone()); @@ -2260,7 +2249,7 @@ pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) } else { vec![("RUSTFLAGS", link_flags.trim_end())] }; - cargo_build(crate_path, target, &["--bins"], &env, &extra_env, quiet); + cargo_build(crate_path, target, &["--bins"], env, &extra_env, quiet); let bin_dir = crate_path.join(format!("target/{target}/{PROFILE}")); let bin_src = crate_path.join("src/bin"); @@ -2286,6 +2275,42 @@ pub fn build_toyos_bins(root: &Path, arch: Arch, crate_path: &Path, quiet: bool) results } +/// What building test binaries starts from, held until the read of what was built is done. +/// +/// Every build→read pair is under one hold, for the reason the "Artifact +/// staging" section above gives: cargo keys an artifact path on (crate, target, +/// profile), so a second `cargo test` in this tree writes the very `.so` and +/// test binaries this one reads back. Between the `read_dir` and the `read` +/// that was enough to kill a run outright — four concurrent suites, one dead on +/// `Result::unwrap()` on a `NotFound` naming no file. +struct TestBuild { + target: &'static str, + env: GuestEnv, + _lock: buildlock::Held, + _artifact: buildlock::Guard, +} + +impl TestBuild { + fn begin(root: &Path, arch: Arch, what: &str, stale_targets: &[(PathBuf, Clean)]) -> Self { + let mut lock = buildlock::shared(root, what); + let sysroot = crate::toolchain::ensure(root, false, &mut lock); + let env = GuestEnv::new(&sysroot); + invalidate_stale(root, &mut lock, &env.toolchain, stale_targets); + let artifact = buildlock::artifact(root); + TestBuild { target: arch.userland(), env, _lock: lock, _artifact: artifact } + } +} + +/// The one binary `name` of the crate at `crate_path`, built for `arch`: for an +/// architecture the crate's other binaries do not all build for. +pub fn build_toyos_bin(root: &Path, arch: Arch, crate_path: &Path, name: &str, quiet: bool) -> Vec { + let build = TestBuild::begin(root, arch, "a test binary", &[(crate_path.to_path_buf(), Clean::All)]); + let (target, env) = (build.target, &build.env); + cargo_build(crate_path, target, &["--bin", name], env, &[], quiet); + let binary = crate_path.join(format!("target/{target}/{PROFILE}/{name}")); + fs::read(&binary).unwrap_or_else(|e| panic!("read the test binary {}: {e}", binary.display())) +} + // --- Internal helpers --- /// The ToyOS-hosted rustc, and the target libraries it compiles against: this @@ -3140,6 +3165,7 @@ mod tests { "tests/testcases/system.toml", "tests/toolkitcase/system.toml", "tests/updatecase/system.toml", + "tests/virtjobcase/system.toml", ]; fn load(cfg: &str) -> SystemConfig { diff --git a/src/metal.rs b/src/metal.rs index b71edc5cd5..516f2b9a86 100644 --- a/src/metal.rs +++ b/src/metal.rs @@ -3359,7 +3359,7 @@ mod tests { let locked = format!( "Previous boot's panic: the last boot read WEDGED\n| {}: cpu7 has taken no interrupt \ for 60004 ms, with `IF` clear at every sample in that span. Its bound is 60000 ms.\n\ - | rip=0xffff800060a5903f >::lock+0x12f\n", + | pc=0xffff800060a5903f >::lock+0x12f\n", bootlog::LOCKED_UP ); assert_eq!(wedged_boot(&locked, booted), Ok(1200)); diff --git a/src/sourcegate.rs b/src/sourcegate.rs index 4dd9dbcecb..591c9103cb 100644 --- a/src/sourcegate.rs +++ b/src/sourcegate.rs @@ -205,9 +205,9 @@ const BANS: &[Ban] = &[ allowed: &[ // Cache lines to walk, not bytes. ("kernel/src/arch/x86_64/control_regs.rs", 1), - // Two guard-page sizes: the mapping is 4 KiB because a guard is one - // hardware page, and `PAGE_SIZE` is 2 MiB territory here. - ("kernel/src/arch/x86_64/percpu.rs", 2), + // A guard page's size: a guard is one hardware page, and + // `PAGE_SIZE` is 2 MiB territory here. + ("kernel/src/arch/x86_64/percpu.rs", 1), // A device's TX buffer. ("kernel/src/drivers/virtio_console.rs", 1), // A VT-d table is 4 KiB by the specification, not by this kernel. @@ -1455,6 +1455,7 @@ const PURE_CRATES: &[&str] = &[ "toyos-desktop/", "toyos-dma/", "toyos-elide/", + "toyos-gicv3/", "toyos-hda/", "toyos-mixer/", "toyos-pci/", @@ -1485,10 +1486,10 @@ const ARCH_RULES: &[PlaceRule] = &[ ("toyos-abi/src/arch/", "toyos-abi's per-architecture reads: the counter and the thread's id"), ("userland/libc/src/arch/", "libc's architecture modules"), ("userland/metalprobe/src/arch/", "metalprobe's architecture modules"), + ("userland/toybox/src/arch/", "toybox's architecture modules"), ("userland/toyos-window/src/arch/", "toyos-window's architecture modules"), ("tests/toyos-rust-tests/src/bin/abuse_kernel_addr.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/abuse_page_straddle.rs", USERLAND_ASM), - ("tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/abuse_tls_alloc.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/copy_out_races_munmap.rs", USERLAND_ASM), ("tests/toyos-rust-tests/src/bin/debug_trap.rs", USERLAND_ASM), @@ -1528,6 +1529,7 @@ const ARCH_RULES: &[PlaceRule] = &[ ("userland/libc/src/arch/", "libc's architecture modules and their selector"), ("userland/toyos-window/src/arch/", "toyos-window's architecture modules and their selector"), ("userland/metalprobe/src/arch/", "metalprobe's architecture modules and their selector"), + ("userland/toybox/src/arch/", "toybox's architecture modules and their selector"), ], }, PlaceRule { diff --git a/tests/common/qemu.rs b/tests/common/qemu.rs index fb4b98710d..f9c06f7508 100644 --- a/tests/common/qemu.rs +++ b/tests/common/qemu.rs @@ -3004,6 +3004,13 @@ pub fn build_toyos_bins(crate_path: &Path) -> Vec<(String, Vec)> { toyos_build::build::build_toyos_bins(&repo, SUITE_ARCH, crate_path, quiet) } +/// One binary of a test crate, built for `arch`: for a guest the crate's other +/// binaries do not all build for. +pub fn build_toyos_bin(arch: Arch, crate_path: &Path, name: &str) -> Vec { + let quiet = !VERBOSE.load(Ordering::Relaxed); + toyos_build::build::build_toyos_bin(&compile::repo_root(), arch, crate_path, name, quiet) +} + /// A kernel record's console line: `klogd` renders every one with this head, /// and nothing else writes it — a program's line reaches the console only /// through `logd`, under the program's own head. diff --git a/tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs b/tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs index cf87aa185d..718940dd6c 100644 --- a/tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs +++ b/tests/toyos-rust-tests/src/bin/abuse_readonly_copyout.rs @@ -7,22 +7,27 @@ //! //! **The memory is the verdict, before the return value**: each arm snapshots //! the target, makes the call, and compares, so a kernel that wrote and then -//! answered an error still fails here. +//! answered an error still fails here. Each arm asks through both copies a +//! syscall writes with: `read`'s bulk window and `process_stats`'s typed value. +//! +//! **Every arm runs, and the exit status carries one bit per arm that let a +//! write through**, which the kernel's own exit record prints: a rewritten +//! clock page asserts in every stamp any process takes, `logd`'s among them, so +//! the clock arm's verdict may reach no other line. +use std::os::toyos::process::ChildExt; use std::process::Command; use toyos_abi::clock::{ClockPage, CLOCK_MAGIC, CLOCK_PAGE}; -use toyos_abi::syscall::{ - self, MmapFlags, MmapProt, OpenFlags, SeekFrom, SyscallError, SYS_FSTAT, SYS_READ, SYS_WRITE, -}; +use toyos_abi::syscall::{self, MmapFlags, MmapProt, OpenFlags, ProcessStats, SeekFrom, SyscallError}; use toyos_abi::RawHandle; const SELF_PATH: &str = "/system/bin/test_rs_abuse_readonly_copyout"; const CHILD_ARG: &str = "reads-the-clock"; const PAGE_2M: usize = 2 * 1024 * 1024; const PAGE_4K: u64 = 4096; -/// Longer than `Stat`, and ends inside the file this reads. -const LEN: usize = 64; +/// Covers a `ProcessStats`, and ends inside the file this reads. +const LEN: usize = core::mem::size_of::(); /// Bytes `0..=255`, so a probe can offer any page its own byte back, then the /// bytes the straddling `read` offers, at [`STRADDLE_AT`]. const PROBE_PATH: &[u8] = b"/tmp/abuse_readonly_copyout.bin"; @@ -30,29 +35,16 @@ const STRADDLE_AT: u64 = 256; /// The straddling `read`: half on the last writable page, half on the next. const STRADDLE: usize = 16; +/// Each arm's bit in the exit status. +const MMAP_BIT: i32 = 1; +const TEXT_BIT: i32 = 2; +const STRADDLE_BIT: i32 = 4; +const CLOCK_BIT: i32 = 8; + /// In `.bss`, so its 2 MiB window also holds the pages past the image's end, /// which the pager maps read-only. static mut BSS: u8 = 0; -/// The typed wrappers take a `&mut [u8]`, which a read-only page cannot be. -fn raw(num: u64, a1: u64, a2: u64, a3: u64) -> u64 { - let ret: u64; - unsafe { - core::arch::asm!( - "syscall", - in("rdi") num, - in("rsi") a1, - in("rdx") a2, - in("r8") a3, - in("r9") 0u64, - lateout("rax") ret, - out("rcx") _, - out("r11") _, - ); - } - ret -} - fn snapshot(addr: u64) -> [u8; LEN] { let mut out = [0u8; LEN]; for (i, b) in out.iter_mut().enumerate() { @@ -66,66 +58,122 @@ fn open_self() -> RawHandle { syscall::open(SELF_PATH.as_bytes(), OpenFlags::READ).expect("open self") } -/// `read` and `fstat` into `addr`: both refused, and not one byte moved. -fn refused(what: &str, addr: u64) { +/// `len` bytes at `addr`, as the buffer a `read` is asked to fill. +/// +/// # Safety +/// Nothing reads or writes through the slice while it lives: it only names the +/// address the kernel is asked to write, which may be a page nothing may write. +unsafe fn target(addr: u64, len: usize) -> &'static mut [u8] { + unsafe { core::slice::from_raw_parts_mut(addr as *mut u8, len) } +} + +/// `read` and `process_stats` (of `stats`) into `addr`, which is 8-aligned: +/// both refused, and not one byte moved. +fn refused(what: &str, addr: u64, stats: RawHandle) -> Result<(), String> { let before = snapshot(addr); let fd = open_self(); - let ret = raw(SYS_READ, fd.0 as u64, addr, LEN as u64); - assert_eq!(before, snapshot(addr), "read wrote into {what}"); - assert_eq!(SyscallError::from_u64(ret), Some(SyscallError::BadAddress), "read into {what}: {ret:#x}"); - let ret = raw(SYS_FSTAT, fd.0 as u64, addr, 0); - assert_eq!(before, snapshot(addr), "fstat wrote into {what}"); - assert_eq!(SyscallError::from_u64(ret), Some(SyscallError::BadAddress), "fstat into {what}: {ret:#x}"); + let ret = syscall::read(fd, unsafe { target(addr, LEN) }); syscall::close(fd); + if before != snapshot(addr) { + return Err(format!("read wrote into {what}")); + } + if ret != Err(SyscallError::BadAddress) { + return Err(format!("read into {what}: {ret:?}")); + } + // SAFETY: as `target`'s; `ProcessStats` is 8-aligned and every bit pattern is one. + let ret = syscall::process_stats(stats, unsafe { &mut *(addr as *mut ProcessStats) }); + if before != snapshot(addr) { + return Err(format!("process_stats wrote into {what}")); + } + if ret != Err(SyscallError::BadAddress) { + return Err(format!("process_stats into {what}: {ret:?}")); + } + Ok(()) +} + +fn refused_mmap(stats: RawHandle) -> Result<(), String> { + let ro = unsafe { + syscall::mmap( + core::ptr::null_mut(), + PAGE_2M, + MmapProt::READ, + MmapFlags::ANONYMOUS | MmapFlags::PRIVATE, + ) + }; + assert!(!ro.is_null(), "mmap a read-only page"); + let verdict = refused("a read-only mmap", ro as u64, stats); + unsafe { syscall::munmap(ro, PAGE_2M) }.expect("munmap"); + verdict } /// The first page at or above [`BSS`], inside its 2 MiB window, a `read` may /// not write; each page below it took a 1-byte `read` of its own first byte. -fn first_unwritable_page(probe: RawHandle) -> u64 { +fn first_unwritable_page(probe: RawHandle) -> Option { let pipe = syscall::pipe().expect("pipe"); let bss = &raw const BSS as u64; let window_end = (bss & !(PAGE_2M as u64 - 1)) + PAGE_2M as u64; let mut page = bss & !(PAGE_4K - 1); + let mut found = None; while page < window_end { - let ret = raw(SYS_WRITE, pipe.write.0 as u64, page, 1); - assert_eq!(ret, 1, "write from {page:#x}, in .bss's window: {ret:#x}"); + let from = unsafe { core::slice::from_raw_parts(page as *const u8, 1) }; + assert_eq!(syscall::write(pipe.write, from), Ok(1), "write from {page:#x}, in .bss's window"); syscall::read(pipe.read, &mut [0u8; 1]).expect("drain the pipe"); let own = unsafe { (page as *const u8).read_volatile() }; syscall::seek(probe, SeekFrom::Start(own as u64)).expect("seek the probe"); - let ret = raw(SYS_READ, probe.0 as u64, page, 1); - if SyscallError::from_u64(ret) == Some(SyscallError::BadAddress) { - syscall::close(pipe.read); - syscall::close(pipe.write); - return page; + let ret = syscall::read(probe, unsafe { target(page, 1) }); + if ret == Err(SyscallError::BadAddress) { + found = Some(page); + break; } - assert_eq!(ret, 1, "a 1-byte read into {page:#x}: {ret:#x}"); + assert_eq!(ret, Ok(1), "a 1-byte read into {page:#x}"); page += PAGE_4K; } - panic!("no page above .bss at {bss:#x} refuses a write below {window_end:#x}: the image ends on its window's edge"); + syscall::close(pipe.read); + syscall::close(pipe.write); + found } /// A `read` that starts on a writable page and runs into the read-only page /// after it, in one 2 MiB window, is refused whole. -fn refused_across_pages() { +fn refused_across_pages() -> Result<(), String> { let flags = OpenFlags::READ | OpenFlags::WRITE | OpenFlags::CREATE | OpenFlags::TRUNCATE; let probe = syscall::open(PROBE_PATH, flags).expect("create the probe file"); let every: Vec = (0..=255).collect(); syscall::write(probe, &every).expect("fill the probe file"); - let page = first_unwritable_page(probe); - assert!(page > &raw const BSS as u64, "BSS's own page refuses a write"); - let at = page - (STRADDLE / 2) as u64; - let before = snapshot(at); - // The writable half is offered its own bytes and the read-only half their - // complement, so a write to the read-only page shows and one below harms nothing. - let offered: [u8; STRADDLE] = core::array::from_fn(|i| if i < STRADDLE / 2 { before[i] } else { !before[i] }); - syscall::seek(probe, SeekFrom::Start(STRADDLE_AT)).expect("seek the probe"); - syscall::write(probe, &offered).expect("write the straddle's bytes"); - syscall::seek(probe, SeekFrom::Start(STRADDLE_AT)).expect("seek the probe"); - let ret = raw(SYS_READ, probe.0 as u64, at, STRADDLE as u64); - assert_eq!(before, snapshot(at), "read wrote across a writable page into the read-only one at {page:#x}"); - assert_eq!(SyscallError::from_u64(ret), Some(SyscallError::BadAddress), "read across into {page:#x}: {ret:#x}"); + let verdict = match first_unwritable_page(probe) { + None => Err(format!( + "no page above .bss at {:#x} refuses a write in its 2 MiB window", + &raw const BSS as u64 + )), + Some(page) => { + assert!(page > &raw const BSS as u64, "BSS's own page refuses a write"); + let at = page - (STRADDLE / 2) as u64; + let before = snapshot(at); + // The writable half is offered its own bytes and the read-only half their + // complement, so a write to the read-only page shows and one below harms nothing. + let offered: [u8; STRADDLE] = + core::array::from_fn(|i| if i < STRADDLE / 2 { before[i] } else { !before[i] }); + syscall::seek(probe, SeekFrom::Start(STRADDLE_AT)).expect("seek the probe"); + syscall::write(probe, &offered).expect("write the straddle's bytes"); + syscall::seek(probe, SeekFrom::Start(STRADDLE_AT)).expect("seek the probe"); + let ret = syscall::read(probe, unsafe { target(at, STRADDLE) }); + if before != snapshot(at) { + Err(format!("read wrote across a writable page into the read-only one at {page:#x}")) + } else if ret != Err(SyscallError::BadAddress) { + Err(format!("read across into {page:#x}: {ret:?}")) + } else { + Ok(()) + } + } + }; syscall::close(probe); syscall::delete(PROBE_PATH).expect("delete the probe file"); + verdict +} + +/// A second process still reads the clock. +fn second_reader() -> bool { + Command::new(SELF_PATH).arg(CHILD_ARG).status().expect("spawn the second reader").success() } fn main() { @@ -136,38 +184,36 @@ fn main() { } // The control: the same calls into memory this process may write succeed, - // so every refusal below is about the page and nothing else. + // so every refusal below is about the page and nothing else. The first + // reader is the process `process_stats` answers for. + let mut first = Command::new(SELF_PATH).arg(CHILD_ARG).spawn().expect("spawn the first reader"); + assert!(first.wait().expect("wait for the first reader").success(), "the first reader failed"); + let stats = RawHandle(first.as_raw_handle()); let mut ok = [0u8; LEN]; let fd = open_self(); - let ret = raw(SYS_READ, fd.0 as u64, ok.as_mut_ptr() as u64, LEN as u64); - assert_eq!(ret, LEN as u64, "read into a writable buffer: {ret:#x}"); + assert_eq!(syscall::read(fd, &mut ok), Ok(LEN), "read into a writable buffer"); assert_eq!(&ok[..4], b"\x7fELF", "read put something other than this file in the buffer"); syscall::close(fd); + syscall::process_stats(stats, &mut ProcessStats::default()).expect("process_stats into a writable value"); + + let mut wrote = 0; + for (bit, verdict) in [ + (MMAP_BIT, refused_mmap(stats)), + (TEXT_BIT, refused("this program's own text", main as *const () as u64 & !7, stats)), + (STRADDLE_BIT, refused_across_pages()), + ] { + if let Err(why) = verdict { + eprintln!("{why}"); + wrote |= bit; + } + } - let ro = unsafe { - syscall::mmap( - core::ptr::null_mut(), - PAGE_2M, - MmapProt::READ, - MmapFlags::ANONYMOUS | MmapFlags::PRIVATE, - ) - }; - assert!(!ro.is_null(), "mmap a read-only page"); - refused("a read-only mmap", ro as u64); - unsafe { syscall::munmap(ro, PAGE_2M) }.expect("munmap"); - - refused("this program's own text", main as *const () as u64); - refused_across_pages(); - - // Last: a written clock page asserts in every stamp this process and every - // other one takes, the verdict's own path included, so the arms whose harm - // is local answer first. - let clock = snapshot(CLOCK_PAGE); - refused("the clock page", CLOCK_PAGE); - assert_eq!(clock, snapshot(CLOCK_PAGE)); - - let status = Command::new(SELF_PATH).arg(CHILD_ARG).status().expect("spawn the second reader"); - assert!(status.success(), "the second process could not read the clock: {status:?}"); - + // Last, and said by its bit alone: its harm reaches every stamping reader. + if refused("the clock page", CLOCK_PAGE, stats).is_err() || !second_reader() { + wrote |= CLOCK_BIT; + } + if wrote != 0 { + std::process::exit(wrote); + } println!("a syscall writes only where its caller could store"); } diff --git a/tests/toyos.rs b/tests/toyos.rs index 19e05cd1c2..cbb7b0f466 100644 --- a/tests/toyos.rs +++ b/tests/toyos.rs @@ -454,6 +454,9 @@ const RUST_SKIP: &[&str] = &[ /// assert-carrying build there. #[allow(dead_code, reason = "`suite_split` reads it, in `toyos-checks` alone")] const DRIVEN_AND_SHARED: &[&str] = &[ + // Its shared run is the x86-64 verdict; `virt_readonly_copyout` builds it + // for AArch64 and runs it on that architecture's job case. + "abuse_readonly_copyout", // The lost-wake canary: its shared run is the count on the shipping // kernel with nothing staged, and `blocking_read_window` drives it again // with the watch's window held open. @@ -529,6 +532,15 @@ const SCREEN_TESTS: &[(&str, Sched, Tier)] = &[ ("virt_early_panic", Sched::Parallel, Tier::Local), ("virt_early_fault", Sched::Parallel, Tier::Local), ("virt_el2_drop", Sched::Parallel, Tier::Local), + ("virt_user_mode", Sched::Parallel, Tier::Local), + ("virt_timer_preempts", Sched::Parallel, Tier::Local), + ("virt_irq_storm", Sched::Parallel, Tier::Local), + ("virt_timer_floor", Sched::Parallel, Tier::Local), + ("virt_fp_isolation", Sched::Parallel, Tier::Local), + ("virt_first_entry", Sched::Parallel, Tier::Local), + ("virt_unmap_touch", Sched::Parallel, Tier::Local), + ("virt_debug_refused", Sched::Parallel, Tier::Local), + ("virt_readonly_copyout", Sched::Parallel, Tier::Local), ]; /// What `screen_console_shell` types, and what it then looks for on its own. @@ -3554,6 +3566,82 @@ fn check_no_stale_cells(dump: &screen::Ppm, console: &str) -> Result<(), String> Ok(()) } +/// `tests/toyos-rust-tests`' binary that `tests/virtjobcase` runs as its job +/// `test_rs_abuse_readonly_copyout`. +const VIRT_COPYOUT: &str = "abuse_readonly_copyout"; + +/// Boot `tests/virtjobcase` under the EL2 profile and judge its job `job`: +/// it ends with exit 0, having said `said`. The kernel carries `SYS_DEBUG` +/// for `debug_refused`, and every job runs in every boot of the case. +fn virt_job(job: &str, said: &str) -> Result<(), String> { + let config = compile::repo_root().join("tests/virtjobcase/system.toml"); + let case = config.parent().expect("system.toml has a directory"); + let profile = qemu::Profile::VirtEl2; + static COPYOUT: std::sync::OnceLock> = std::sync::OnceLock::new(); + let copyout = COPYOUT.get_or_init(|| { + qemu::build_toyos_bin(profile.arch(), &compile::repo_root().join("tests/toyos-rust-tests"), VIRT_COPYOUT) + }); + let mut qemu = QemuInstance::boot_with_options( + case, + &[], + &[], + BootOptions { + profile, + kernel_features: toyos_build::build::TEST_KERNEL, + ready_marker: "control registers: SCTLR_EL1=", + extra_root_files: vec![(format!("bin/test_rs_{VIRT_COPYOUT}"), copyout.clone())], + ..Default::default() + }, + ); + let end = format!("===TEST_END {job} "); + let mut rest = String::new(); + let waited = await_marker(&mut qemu, &mut rest, &end, &format!("the job {job} to end")); + let serial = format!("{}\n{rest}", qemu.boot_log()); + if let Err(why) = waited { + return Err(format!("{why}\nserial:\n{serial}")); + } + let ended = serial + .lines() + .find(|l| l.contains(&end)) + .expect("await_marker answered Ok, so the marker is in what it drained"); + let Some(line) = serial.lines().find(|l| l.contains(said)) else { + return Err(format!("{said:?} not on the PL011 ({ended})\nserial:\n{serial}")); + }; + eprintln!(" [virt] {line}"); + if !ended.contains(&format!("===TEST_END {job} exit=0===")) { + return Err(format!("{ended}\nserial:\n{serial}")); + } + Ok(()) +} + +/// Boot `test_config` under the EL2 profile with the kernel selftest `armed` +/// names, and judge its one line: `: PASS`. +fn virt_selftest(test_config: &Path, armed: &'static [&'static str; 1]) -> Result<(), String> { + let [param] = armed; + let mut qemu = QemuInstance::boot_with_options( + test_config, + &[], + &[], + BootOptions { + profile: qemu::Profile::VirtEl2, + kernel_params: armed, + ready_marker: "control registers: SCTLR_EL1=", + ..Default::default() + }, + ); + let said = format!("{param}: "); + let rest = qemu.drain_until(Duration::from_secs(180), |l| l.contains(&said)); + let serial = format!("{}\n{rest}", qemu.boot_log()); + let Some(verdict) = serial.lines().find(|l| l.contains(&said)) else { + return Err(format!("{param} never reported\nserial:\n{serial}")); + }; + eprintln!(" [virt] {verdict}"); + if !verdict.contains(&format!("{param}: PASS")) { + return Err(format!("{verdict}\nserial:\n{serial}")); + } + Ok(()) +} + /// Run one screen test. `Err` carries the decoded screen, because a failure /// here is almost always "the text is not what I expected" and the decoded /// grid is the only readable form of that. @@ -5013,6 +5101,62 @@ fn run_screen_test( } Ok(()) } + "virt_user_mode" => { + // The port's stage 4 on one CPU, under the EL2 profile whose + // entry also writes what the drop leaves EL2 holding: the kernel's + // own tables, the GIC and the timer, and a process at EL0 — init, + // whose every page arrives by a demand fault and whose spawn of + // `logd` is a syscall the kernel answered. Emulated, and not under + // HVF, which exposes no RNDR for the kernel's hash seed. + let mut qemu = QemuInstance::boot_with_options( + test_config, + &[], + &[], + BootOptions { + profile: qemu::Profile::VirtEl2, + ready_marker: "control registers: SCTLR_EL1=", + ..Default::default() + }, + ); + const SPAWNED: &str = "spawn: /system/bin/logd pid="; + let rest = qemu.drain_until(Duration::from_secs(180), |l| l.contains(SPAWNED)); + let serial = format!("{}\n{rest}", qemu.boot_log()); + for want in [ + "paging: the direct map holds memory below", + "percpu: BSP cpu_id=0", + "GIC: v", + "clock: the generic timer counts at", + "spawned /system/bin/init pid=", + SPAWNED, + ] { + if !serial.contains(want) { + return Err(format!("{want:?} not on the PL011\nserial:\n{serial}")); + } + } + Ok(()) + } + "virt_timer_preempts" => { + // Spelled in `userland/toybox/src/preempt.rs`. + virt_job("preempt", "preempt: the counting thread was preempted twice") + } + "virt_fp_isolation" => virt_job("fp_isolation", "fp_isolation: v0-v31, FPCR and FPSR survived"), + "virt_first_entry" => virt_job("first_entry", "first_entry: x1-x30 were zero"), + "virt_unmap_touch" => virt_job("unmap_touch", "unmap_touch: 4 reads of a page just unmapped"), + "virt_debug_refused" => virt_job( + "debug_refused", + "debug_refused: SYS_DEBUG's double fault and TLB acknowledgement delay were refused", + ), + "virt_readonly_copyout" => { + virt_job(&format!("test_rs_{VIRT_COPYOUT}"), "a syscall writes only where its caller could store") + } + "virt_irq_storm" => { + // The CPU floods itself with SGIs until the timer has fired a + // thousand times through the flood, then waits for every SGI it + // sent. A tick lost or never re-armed, or an SGI lost, leaves the + // storm running and the verdict unsaid. + virt_selftest(test_config, &["irq-storm"]) + } + "virt_timer_floor" => virt_selftest(test_config, &["timer-floor"]), "screen_late_panic" => { // The ordinary fatal panic, which no userland process can produce: // crash_report, capture, panic_flush, halt_all_cpus, render. The diff --git a/tests/virtjobcase/system.toml b/tests/virtjobcase/system.toml new file mode 100644 index 0000000000..190d1b33ac --- /dev/null +++ b/tests/virtjobcase/system.toml @@ -0,0 +1,25 @@ +# One CPU and the AArch64 guest's jobs, each a toybox applet or, last because +# a kernel it fails on rewrites the clock page every reader asserts on, the +# suite's `test_rs_abuse_readonly_copyout`, which the harness puts on ROOT; +# each job's line comes only if the kernel kept what that job asks about. +[boot] +start = ["logd", "test-runner"] + +[programs.logd] +service = true +syscap = ["logread"] + +[programs.test-runner] +receives = ["power"] +# Four minutes from boot rather than one: every job here runs on an emulated +# CPU, and a job past the bound reboots the guest with the job named. +args = ["--bound-ms=240000", "preempt", "fp_isolation", "first_entry", "unmap_touch", "debug_refused", "test_rs_abuse_readonly_copyout"] + +[programs.toybox] + +[symlinks] +"bin/preempt" = "/system/bin/toybox" +"bin/fp_isolation" = "/system/bin/toybox" +"bin/first_entry" = "/system/bin/toybox" +"bin/unmap_touch" = "/system/bin/toybox" +"bin/debug_refused" = "/system/bin/toybox" diff --git a/toyos-bootmap/src/aarch64.rs b/toyos-bootmap/src/aarch64.rs index 9d833667a0..08b33f7542 100644 --- a/toyos-bootmap/src/aarch64.rs +++ b/toyos-bootmap/src/aarch64.rs @@ -1,9 +1,12 @@ //! AArch64's encoding of a [`Plan`](crate::Plan): the VMSAv8-64 stage 1 //! descriptors of a 4 KiB granule (Arm ARM K.a, D8.3), and the one `MAIR_EL1` //! whose indices they name — the loader writes the one, the kernel's entry -//! loads the other, and both read them here. +//! loads the other, and both read them here. And which pages the kernel's own +//! direct map holds, since on AArch64 that is decided page by page. -use crate::Cache; +use toyos_abi::boot::MemoryMapEntry; + +use crate::{is_read_as_memory, Cache, DirectMapEnd, Refusal, DIRECT_MAP_WINDOW, PAGE_2M, PAGE_4K}; /// `MAIR_EL1` index 0: Device-nGnRE, registers. pub const ATTR_DEVICE: u64 = 0; @@ -63,3 +66,71 @@ const fn attributes(cache: Cache) -> u64 { pub const fn page(phys: u64, cache: Cache) -> u64 { phys | PAGE | AF | attributes(cache) } + +/// 4 KiB pages in one 2 MiB page. +const PAGES: u64 = PAGE_2M / PAGE_4K; + +/// How the kernel's own direct map holds one 2 MiB page of physical memory. +#[derive(Clone, Copy, PartialEq, Eq, Debug)] +pub enum Coverage { + /// No byte of it is memory the kernel reads: it is not mapped. + Nothing, + /// Every byte is: one block. + Whole, + /// Some of its 4 KiB pages are: a table whose page `i` is mapped where bit + /// `i % 64` of word `i / 64` is set. + Pages([u64; 8]), +} + +impl Coverage { + /// Whether the 4 KiB page `index` of the 2 MiB page is mapped. + pub const fn holds(&self, index: u64) -> bool { + match self { + Self::Nothing => false, + Self::Whole => true, + Self::Pages(bits) => bits[(index / 64) as usize] & 1 << (index % 64) != 0, + } + } +} + +/// One past the kernel direct map's last byte on AArch64: the end of the +/// highest range the kernel reads as memory, in whole 2 MiB pages. Nothing +/// below it is mapped for being low, as x86-64's low pages are: a page is +/// mapped only for being memory ([`coverage`]), because a register or a hole +/// mapped Normal is one a speculative access can reach. A range off a 4 KiB +/// page, or ending past [`DIRECT_MAP_WINDOW`], is refused. +pub fn direct_map_end(map: &[MemoryMapEntry]) -> Result { + map.iter().filter(|entry| is_read_as_memory(entry.uefi_type)).try_fold(0, |end: u64, entry| { + if !entry.start.is_multiple_of(PAGE_4K) { + return Err(Refusal::OffPage(entry.start)); + } + if !entry.end.is_multiple_of(PAGE_4K) { + return Err(Refusal::OffPage(entry.end)); + } + if entry.end < entry.start { + return Err(Refusal::Extent { base: entry.start, len: entry.end.wrapping_sub(entry.start) }); + } + if entry.end > DIRECT_MAP_WINDOW { + return Err(Refusal::PastWindow(entry.end)); + } + Ok(end.max(entry.end.next_multiple_of(PAGE_2M))) + }) + .map(DirectMapEnd) +} + +/// Which of the 4 KiB pages of the 2 MiB page at `page` firmware's map calls +/// memory the kernel reads, over a map [`direct_map_end`] accepted. +pub fn coverage(map: &[MemoryMapEntry], page: u64) -> Coverage { + let mut bits = [0u64; 8]; + for entry in map.iter().filter(|entry| is_read_as_memory(entry.uefi_type)) { + let (low, high) = (entry.start.max(page), entry.end.min(page + PAGE_2M)); + for index in (low.saturating_sub(page) / PAGE_4K)..(high.saturating_sub(page) / PAGE_4K) { + bits[(index / 64) as usize] |= 1 << (index % 64); + } + } + match bits.iter().map(|word| u64::from(word.count_ones())).sum::() { + 0 => Coverage::Nothing, + PAGES => Coverage::Whole, + _ => Coverage::Pages(bits), + } +} diff --git a/toyos-bootmap/src/lib.rs b/toyos-bootmap/src/lib.rs index ce22f5526a..92f7237bfe 100644 --- a/toyos-bootmap/src/lib.rs +++ b/toyos-bootmap/src/lib.rs @@ -59,8 +59,7 @@ pub const BOOT_MAP_BYTES: u64 = 4 * GIB; /// hold: every slot from there to the root's last. pub const DIRECT_MAP_WINDOW: u64 = (512 - ROOT_HIGH_HALF as u64) * GIB_PER_PDPT * GIB; -/// One past the kernel direct map's last byte. Made only by -/// [`x86_64::direct_map_end`]. +/// One past the kernel direct map's last byte. /// /// ```compile_fail,E0603 /// let _ = toyos_bootmap::DirectMapEnd(1 << 52); @@ -163,6 +162,9 @@ pub enum Refusal { Mixed(u64), /// Memory that ends here, past [`DIRECT_MAP_WINDOW`]. PastWindow(u64), + /// A range of firmware's map that begins or ends here, off the 4 KiB page + /// UEFI describes memory in. + OffPage(u64), } impl fmt::Display for Refusal { @@ -190,6 +192,7 @@ impl fmt::Display for Refusal { f, "memory ends at {end:#x}, past the {DIRECT_MAP_WINDOW:#x} bytes a direct map can hold" ), + Self::OffPage(at) => write!(f, "a range of firmware's map is bounded at {at:#x}, off a {PAGE_4K:#x}-byte page"), } } } diff --git a/toyos-bootmap/tests/aarch64_direct_map.rs b/toyos-bootmap/tests/aarch64_direct_map.rs new file mode 100644 index 0000000000..3adcb659bf --- /dev/null +++ b/toyos-bootmap/tests/aarch64_direct_map.rs @@ -0,0 +1,96 @@ +//! Which pages the AArch64 kernel's direct map holds: every 4 KiB page +//! firmware's map calls memory the kernel reads, and no other. + +use toyos_abi::boot::MemoryMapEntry; +use toyos_bootmap::aarch64::{coverage, direct_map_end, Coverage}; +use toyos_bootmap::{DirectMapEnd, Refusal, DIRECT_MAP_WINDOW, PAGE_2M}; + +const RESERVED: u32 = 0; +const LOADER_DATA: u32 = 2; +const RUNTIME_DATA: u32 = 6; +const CONVENTIONAL: u32 = 7; +const ACPI_RECLAIM: u32 = 9; +const MMIO: u32 = 11; + +const GIB: u64 = 1 << 30; + +const fn e(uefi_type: u32, start: u64, end: u64) -> MemoryMapEntry { + MemoryMapEntry { uefi_type, start, end } +} + +fn end(map: &[MemoryMapEntry]) -> Result { + direct_map_end(map).map(DirectMapEnd::get) +} + +/// RAM from 1 GiB, as `virt` has it, with firmware's carve-outs inside it. +const VIRT: [MemoryMapEntry; 7] = [ + e(MMIO, 0x0900_0000, 0x0900_1000), + e(CONVENTIONAL, GIB, GIB + 0x1F_F000), + e(RESERVED, GIB + 0x1F_F000, GIB + 0x20_0000), + e(CONVENTIONAL, GIB + 0x20_0000, GIB + 0x60_0000), + e(RUNTIME_DATA, GIB + 0x60_0000, GIB + 0x60_3000), + e(ACPI_RECLAIM, GIB + 0x60_3000, GIB + 0x60_4000), + e(LOADER_DATA, GIB + 0x60_4000, 3 * GIB), +]; + +#[test] +fn memory_is_mapped_and_a_register_is_not() { + assert_eq!(end(&VIRT), Ok(3 * GIB)); + assert_eq!(coverage(&VIRT, 0x0900_0000 & !(PAGE_2M - 1)), Coverage::Nothing); + assert_eq!(coverage(&VIRT, 0), Coverage::Nothing); +} + +#[test] +fn a_page_wholly_memory_is_one_block() { + assert_eq!(coverage(&VIRT, GIB + PAGE_2M), Coverage::Whole); + assert_eq!(coverage(&VIRT, 2 * GIB), Coverage::Whole); +} + +#[test] +fn a_reserved_page_inside_ram_is_left_out_of_its_block() { + let first = coverage(&VIRT, GIB); + assert!(matches!(first, Coverage::Pages(_))); + assert!(first.holds(0)); + assert!(first.holds(510)); + assert!(!first.holds(511), "EfiReservedMemoryType is not memory the kernel reads"); +} + +#[test] +fn runtime_services_data_is_left_out_and_acpi_tables_are_not() { + let page = coverage(&VIRT, GIB + 3 * PAGE_2M); + for runtime in 0..3 { + assert!(!page.holds(runtime), "runtime services data, page {runtime}"); + } + assert!(page.holds(3), "the ACPI table page"); + assert!(page.holds(4), "loader data"); + assert!(page.holds(511), "loader data"); +} + +#[test] +fn off_a_page_is_refused() { + let map = [e(CONVENTIONAL, GIB, GIB + 0x800)]; + assert_eq!(end(&map), Err(Refusal::OffPage(GIB + 0x800))); +} + +#[test] +fn past_the_window_is_refused() { + let map = [e(CONVENTIONAL, DIRECT_MAP_WINDOW, DIRECT_MAP_WINDOW + PAGE_2M)]; + assert_eq!(end(&map), Err(Refusal::PastWindow(DIRECT_MAP_WINDOW + PAGE_2M))); +} + +#[test] +fn a_map_with_no_memory_ends_at_zero() { + assert_eq!(end(&[e(MMIO, 0x0900_0000, 0x0900_1000)]), Ok(0)); +} + +#[test] +fn a_start_off_a_page_is_refused() { + let map = [e(CONVENTIONAL, GIB + 0x800, GIB + PAGE_2M)]; + assert_eq!(end(&map), Err(Refusal::OffPage(GIB + 0x800))); +} + +#[test] +fn an_end_below_its_start_is_refused() { + let map = [e(CONVENTIONAL, GIB + PAGE_2M, GIB)]; + assert_eq!(end(&map), Err(Refusal::Extent { base: GIB + PAGE_2M, len: GIB.wrapping_sub(GIB + PAGE_2M) })); +} diff --git a/toyos-gicv3/Cargo.toml b/toyos-gicv3/Cargo.toml new file mode 100644 index 0000000000..689de4fab4 --- /dev/null +++ b/toyos-gicv3/Cargo.toml @@ -0,0 +1,7 @@ +[package] +name = "toyos-gicv3" +description = "The GICv3's pure decisions: affinity packing, SGI addressing and the redistributor walk." +version = "0.1.0" +edition = "2021" +license = "MIT OR Apache-2.0" +publish = false diff --git a/toyos-gicv3/src/lib.rs b/toyos-gicv3/src/lib.rs new file mode 100644 index 0000000000..40e73e90d2 --- /dev/null +++ b/toyos-gicv3/src/lib.rs @@ -0,0 +1,54 @@ +//! The GICv3's pure decisions (GIC architecture specification IHI 0069H): +//! how a CPU's affinity is packed, how an SGI names the CPU it is raised on, +//! and how a redistributor region is walked to the frame of one CPU. The +//! kernel's `arch::aarch64::irqchip` reads the registers; everything here is +//! arithmetic on what it read, so the host can ask it about CPUs the boot CPU +//! alone never exercises. + +#![no_std] + +/// One 64 KiB register frame. A redistributor is two — `RD_base`, then +/// `SGI_base` — and two more for virtual LPIs where `GICR_TYPER.VLPIS` says so. +pub const FRAME: u64 = 0x1_0000; + +/// `GICR_TYPER.VLPIS`: this redistributor has the two virtual-LPI frames. +const TYPER_VLPIS: u64 = 1 << 1; +/// `GICR_TYPER.Last`: the last redistributor in its region. +const TYPER_LAST: u64 = 1 << 4; + +/// `MPIDR_EL1`'s four affinity fields packed into 32 bits, +/// `Aff3:Aff2:Aff1:Aff0`, as `GICR_TYPER` carries them in its top word. +pub const fn packed_affinity(mpidr: u64) -> u32 { + ((mpidr & 0xFF_FFFF) | ((mpidr >> 8) & 0xFF00_0000)) as u32 +} + +/// `ICC_SGI1R_EL1` raising SGI `intid` on the one CPU whose packed affinity is +/// `target`: `Aff3`, `Aff2` and `Aff1` name its cluster, `RS` the range of +/// sixteen `Aff0` values it falls in, and the target list its one bit there. +pub const fn sgi1r(intid: u32, target: u32) -> u64 { + assert!(intid < 16, "an SGI's INTID is below 16"); + let aff0 = (target & 0xFF) as u64; + let aff1 = (target >> 8 & 0xFF) as u64; + let aff2 = (target >> 16 & 0xFF) as u64; + let aff3 = (target >> 24) as u64; + aff3 << 48 | (aff0 >> 4) << 44 | aff2 << 32 | (intid as u64) << 24 | aff1 << 16 | 1 << (aff0 & 0xF) +} + +/// The offset, in a redistributor region `length` bytes long, of the +/// redistributor whose affinity is `me`; `typer` reads `GICR_TYPER` of the +/// redistributor at an offset. The walk steps by each one's own frames and +/// stops at the one marked last. +pub fn find_redistributor(length: u64, me: u32, typer: impl Fn(u64) -> u64) -> Option { + let mut offset = 0; + while offset + 2 * FRAME <= length { + let word = typer(offset); + if (word >> 32) as u32 == me { + return Some(offset); + } + if word & TYPER_LAST != 0 { + return None; + } + offset += if word & TYPER_VLPIS != 0 { 4 * FRAME } else { 2 * FRAME }; + } + None +} diff --git a/toyos-gicv3/tests/gicv3.rs b/toyos-gicv3/tests/gicv3.rs new file mode 100644 index 0000000000..73a55f8b75 --- /dev/null +++ b/toyos-gicv3/tests/gicv3.rs @@ -0,0 +1,67 @@ +use toyos_gicv3::{find_redistributor, packed_affinity, sgi1r, FRAME}; + +#[test] +fn affinity_drops_mpidr_flags_between_aff2_and_aff3() { + // Aff3 in 39:32; `U` (30), `MT` (24) and the RES1 bit 31 in between. + let mpidr = 0x12 << 32 | 1 << 31 | 1 << 30 | 1 << 24 | 0x34_5678; + assert_eq!(packed_affinity(mpidr), 0x1234_5678); +} + +#[test] +fn sgi_names_aff0_by_range_and_bit() { + // Aff0 0x13 is range 1, bit 3. + let value = sgi1r(2, 0x0A0B_0C13); + assert_eq!(value >> 44 & 0xF, 1, "RS"); + assert_eq!(value & 0xFFFF, 1 << 3, "target list"); + assert_eq!(value >> 16 & 0xFF, 0x0C, "Aff1"); + assert_eq!(value >> 24 & 0xF, 2, "INTID"); + assert_eq!(value >> 32 & 0xFF, 0x0B, "Aff2"); + assert_eq!(value >> 48 & 0xFF, 0x0A, "Aff3"); + assert_eq!(value & !(0xFF << 48 | 0xF << 44 | 0xFF << 32 | 0xF << 24 | 0xFF << 16 | 0xFFFF), 0, "IRM and RES0"); +} + +#[test] +fn sgi_to_cpu_zero_is_bit_zero_of_range_zero() { + assert_eq!(sgi1r(0, 0), 1); +} + +/// A region of redistributors, each `(affinity, vlpis, last)`, as `GICR_TYPER` reads at each one's offset. +fn region(cpus: &[(u32, bool, bool)]) -> (u64, impl Fn(u64) -> u64 + '_) { + let stride = |vlpis: bool| if vlpis { 4 * FRAME } else { 2 * FRAME }; + let length = cpus.iter().map(|&(_, vlpis, _)| stride(vlpis)).sum(); + let typer = move |at: u64| { + let mut offset = 0; + for &(affinity, vlpis, last) in cpus { + if offset == at { + return u64::from(affinity) << 32 | u64::from(vlpis) << 1 | u64::from(last) << 4; + } + offset += stride(vlpis); + } + panic!("a GICR_TYPER read at {at:#x}, which is no redistributor's first frame"); + }; + (length, typer) +} + +#[test] +fn walk_steps_over_virtual_lpi_frames() { + let (length, typer) = region(&[(0, true, false), (1, true, false), (2, true, true)]); + assert_eq!(find_redistributor(length, 2, typer), Some(8 * FRAME)); +} + +#[test] +fn walk_steps_two_frames_without_them() { + let (length, typer) = region(&[(0, false, false), (1, false, true)]); + assert_eq!(find_redistributor(length, 1, typer), Some(2 * FRAME)); +} + +#[test] +fn walk_stops_at_the_last() { + let (_, typer) = region(&[(0, false, false), (1, false, true)]); + assert_eq!(find_redistributor(16 * FRAME, 7, typer), None); +} + +#[test] +fn walk_stops_at_the_region_end() { + let (length, typer) = region(&[(0, false, false), (1, false, false)]); + assert_eq!(find_redistributor(length, 7, typer), None); +} diff --git a/userland/toybox/src/arch/aarch64.rs b/userland/toybox/src/arch/aarch64.rs new file mode 100644 index 0000000000..a07a1cc54b --- /dev/null +++ b/userland/toybox/src/arch/aarch64.rs @@ -0,0 +1,270 @@ +//! What an AArch64 thread finds after the kernel has had the CPU: its own +//! FP/SIMD state across the switches that took the CPU from it, nothing of the +//! kernel's at its first instruction, and no translation of a page it unmapped. + +/// `fp_isolation`: this thread pins a distinctive FP/SIMD state — v0–v31, +/// `FPCR` and `FPSR` — and holds it while a sibling that loads another state +/// takes the CPU from it; the state it reads back must be the one it pinned. +/// The kernel saves and restores that state only where a thread stops +/// running, so on one CPU every switch is a chance to hand this thread the +/// sibling's. +pub mod fp_isolation { + use std::sync::atomic::{AtomicBool, AtomicU64, Ordering::Relaxed}; + use std::sync::Arc; + + /// The saved state: v0–v31, then `FPCR` and `FPSR`. + #[repr(C, align(16))] + #[derive(Clone, Copy, PartialEq, Eq, Debug)] + struct FpState { + v: [u128; 32], + fpcr: u64, + fpsr: u64, + } + + const fn state(tag: u128, fpcr: u64, fpsr: u64) -> FpState { + let mut v = [0; 32]; + let mut i = 0; + while i < 32 { + v[i] = tag | (i as u128 + 1); + i += 1; + } + FpState { v, fpcr, fpsr } + } + + /// Default-NaN, flush-to-zero, round toward zero; every cumulative flag + /// and `QC` raised. + static PINNED: FpState = state(0xF9A3 << 112, 1 << 25 | 1 << 24 | 0b11 << 22, 1 << 27 | 0x1F); + /// Round toward plus infinity, no flag. + static NOISE: FpState = state(0x5A5A << 112, 0b01 << 22, 0); + const EMPTY: FpState = state(0, 0, 0); + + /// Times the sibling must be seen to have run while this thread held its state. + const SWITCHES: u64 = 3; + + /// Loads the state at `[x9]` into the FP/SIMD registers. + macro_rules! load_state { + () => { + concat!( + "ldp q0, q1, [x9, #0]\n", "ldp q2, q3, [x9, #32]\n", "ldp q4, q5, [x9, #64]\n", + "ldp q6, q7, [x9, #96]\n", "ldp q8, q9, [x9, #128]\n", "ldp q10, q11, [x9, #160]\n", + "ldp q12, q13, [x9, #192]\n", "ldp q14, q15, [x9, #224]\n", "ldp q16, q17, [x9, #256]\n", + "ldp q18, q19, [x9, #288]\n", "ldp q20, q21, [x9, #320]\n", "ldp q22, q23, [x9, #352]\n", + "ldp q24, q25, [x9, #384]\n", "ldp q26, q27, [x9, #416]\n", "ldp q28, q29, [x9, #448]\n", + "ldp q30, q31, [x9, #480]\n", + "ldr x12, [x9, #512]\n", "msr fpcr, x12\n", + "ldr x12, [x9, #520]\n", "msr fpsr, x12\n", + ) + }; + } + + /// Stores the FP/SIMD registers at `[$reg]`. + macro_rules! store_state { + ($reg:literal) => { + concat!( + "stp q0, q1, [", $reg, ", #0]\n", "stp q2, q3, [", $reg, ", #32]\n", + "stp q4, q5, [", $reg, ", #64]\n", "stp q6, q7, [", $reg, ", #96]\n", + "stp q8, q9, [", $reg, ", #128]\n", "stp q10, q11, [", $reg, ", #160]\n", + "stp q12, q13, [", $reg, ", #192]\n", "stp q14, q15, [", $reg, ", #224]\n", + "stp q16, q17, [", $reg, ", #256]\n", "stp q18, q19, [", $reg, ", #288]\n", + "stp q20, q21, [", $reg, ", #320]\n", "stp q22, q23, [", $reg, ", #352]\n", + "stp q24, q25, [", $reg, ", #384]\n", "stp q26, q27, [", $reg, ", #416]\n", + "stp q28, q29, [", $reg, ", #448]\n", "stp q30, q31, [", $reg, ", #480]\n", + "mrs x12, fpcr\n", "str x12, [", $reg, ", #512]\n", + "mrs x12, fpsr\n", "str x12, [", $reg, ", #520]\n", + ) + }; + } + + pub fn main(_args: Vec) { + let ran = Arc::new(AtomicU64::new(0)); + let stop = Arc::new(AtomicBool::new(false)); + let sibling = { + let (ran, stop) = (Arc::clone(&ran), Arc::clone(&stop)); + std::thread::spawn(move || noise_until(&ran, &stop)) + }; + let (mut before, mut after) = (EMPTY, EMPTY); + pin_and_watch(&ran, &mut before, &mut after); + stop.store(true, Relaxed); + sibling.join().expect("the noise thread died"); + + assert_eq!(before.v, PINNED.v, "the pinned registers did not read back as pinned"); + assert_ne!(before.fpcr, NOISE.fpcr, "FPCR read back as the noise's, so the arm cannot tell them apart"); + let lost: Vec = (0..32).filter(|&i| after.v[i] != before.v[i]).collect(); + assert!( + lost.is_empty() && after.fpcr == before.fpcr && after.fpsr == before.fpsr, + "the FP/SIMD state did not survive {SWITCHES} switches to a thread that loads another: \ + v{lost:?} changed, FPCR {:#x} became {:#x}, FPSR {:#x} became {:#x}", + before.fpcr, + after.fpcr, + before.fpsr, + after.fpsr, + ); + println!("fp_isolation: v0-v31, FPCR and FPSR survived {SWITCHES} switches to a thread that loads another state"); + } + + /// Pin [`PINNED`], read it back into `before`, hold it until `ran` has + /// been seen to move [`SWITCHES`] times, and read it into `after`. On one + /// CPU `ran` moves only while this thread is off it. Nothing between the + /// pin and the second read touches an FP/SIMD register. + fn pin_and_watch(ran: &AtomicU64, before: &mut FpState, after: &mut FpState) { + // SAFETY: reads `PINNED` and `ran`, writes `before` and `after`, each + // named by its register and live across the block; every FP/SIMD + // register it writes is declared clobbered, and `FPCR`/`FPSR` are put + // back as found. + unsafe { + core::arch::asm!( + "mrs x16, fpcr", + "mrs x17, fpsr", + load_state!(), + store_state!("x10"), + "ldr x14, [x13]", + "mov x15, #{switches}", + "2:", + "ldr x12, [x13]", + "cmp x12, x14", + "b.eq 2b", + "mov x14, x12", + "subs x15, x15, #1", + "b.ne 2b", + store_state!("x11"), + "msr fpcr, x16", + "msr fpsr, x17", + switches = const SWITCHES, + in("x9") &raw const PINNED, + in("x10") core::ptr::from_mut(before), + in("x11") core::ptr::from_mut(after), + in("x13") core::ptr::from_ref(ran), + out("x12") _, out("x14") _, out("x15") _, out("x16") _, out("x17") _, + out("v0") _, out("v1") _, out("v2") _, out("v3") _, out("v4") _, out("v5") _, + out("v6") _, out("v7") _, out("v8") _, out("v9") _, out("v10") _, out("v11") _, + out("v12") _, out("v13") _, out("v14") _, out("v15") _, out("v16") _, out("v17") _, + out("v18") _, out("v19") _, out("v20") _, out("v21") _, out("v22") _, out("v23") _, + out("v24") _, out("v25") _, out("v26") _, out("v27") _, out("v28") _, out("v29") _, + out("v30") _, out("v31") _, + options(nostack), + ); + } + } + + /// Load [`NOISE`] and count one round in `ran`, over and over, until `stop`. + fn noise_until(ran: &AtomicU64, stop: &AtomicBool) { + // SAFETY: reads `NOISE` and `stop`, writes `ran`; every FP/SIMD + // register it writes is declared clobbered, and `FPCR`/`FPSR` are put + // back as found. + unsafe { + core::arch::asm!( + "mrs x16, fpcr", + "mrs x17, fpsr", + "2:", + load_state!(), + "ldr x12, [x13]", + "add x12, x12, #1", + "str x12, [x13]", + "ldrb w12, [x14]", + "cbz w12, 2b", + "msr fpcr, x16", + "msr fpsr, x17", + in("x9") &raw const NOISE, + in("x13") core::ptr::from_ref(ran), + in("x14") core::ptr::from_ref(stop), + out("x12") _, out("x16") _, out("x17") _, + out("v0") _, out("v1") _, out("v2") _, out("v3") _, out("v4") _, out("v5") _, + out("v6") _, out("v7") _, out("v8") _, out("v9") _, out("v10") _, out("v11") _, + out("v12") _, out("v13") _, out("v14") _, out("v15") _, out("v16") _, out("v17") _, + out("v18") _, out("v19") _, out("v20") _, out("v21") _, out("v22") _, out("v23") _, + out("v24") _, out("v25") _, out("v26") _, out("v27") _, out("v28") _, out("v29") _, + out("v30") _, out("v31") _, + options(nostack), + ); + } + } +} + +/// `first_entry`: a raw thread whose first instruction stores x1–x30 and +/// exits. The kernel hands a new thread its argument in x0 and its stack in +/// `SP_EL0`, and every other general register must reach that instruction +/// zero: whatever else is there is the kernel's. +pub mod first_entry { + use toyos_abi::syscall::{thread_join, thread_spawn, SYS_THREAD_EXIT}; + + /// x1 to x30, as the probe found them. + #[repr(C, align(16))] + struct Registers([u64; 30]); + + pub fn main(_args: Vec) { + let mut found = Registers([u64::MAX; 30]); + let mut stack = vec![0u128; 1024]; + let base = stack.as_mut_ptr() as u64; + let top = base + (stack.len() * 16) as u64; + // SAFETY: `probe` touches only the `Registers` its argument names, + // which outlives the join below, and never its stack. + let tid = unsafe { thread_spawn(probe as *const () as u64, top, (&raw mut found) as u64, base) }; + assert!(tid < 1_000_000, "thread_spawn refused: {tid:#x}"); + assert_eq!(thread_join(tid), 0, "thread_join failed"); + drop(stack); + let held: Vec = (0..30) + .filter(|&i| found.0[i] != 0) + .map(|i| format!("x{}={:#x}", i + 1, found.0[i])) + .collect(); + assert!(held.is_empty(), "a new thread's first instruction found {}", held.join(" ")); + println!("first_entry: x1-x30 were zero at a new thread's first instruction"); + } + + /// Store x1–x30 at `[x0]` before any of them is written, and exit the thread. + #[unsafe(naked)] + extern "C" fn probe() { + core::arch::naked_asm!( + "stp x1, x2, [x0, #0]", + "stp x3, x4, [x0, #16]", + "stp x5, x6, [x0, #32]", + "stp x7, x8, [x0, #48]", + "stp x9, x10, [x0, #64]", + "stp x11, x12, [x0, #80]", + "stp x13, x14, [x0, #96]", + "stp x15, x16, [x0, #112]", + "stp x17, x18, [x0, #128]", + "stp x19, x20, [x0, #144]", + "stp x21, x22, [x0, #160]", + "stp x23, x24, [x0, #176]", + "stp x25, x26, [x0, #192]", + "stp x27, x28, [x0, #208]", + "stp x29, x30, [x0, #224]", + "mov x0, #{exit}", + "mov x1, xzr", + "svc #0", + "brk #0", + exit = const SYS_THREAD_EXIT, + ); + } +} + +/// Reads the word at `page`, unmaps the `len` bytes there with `SYS_MUNMAP` +/// and reads the word again, in one block, so nothing but the unmap itself +/// can switch the CPU between the reads. Answers the first read, the unmap's +/// answer and the second read. +/// +/// # Safety +/// `page` starts a mapping of `len` bytes that nothing else names. The second +/// read ends the process unless the unmap was refused or left its +/// translation standing. +pub unsafe fn read_unmap_read(page: *const u64, len: usize) -> (u64, u64, u64) { + let (first, answer, second): (u64, u64, u64); + // SAFETY: the caller's; the two loads name `page` and the `svc` is the + // ABI's (`toyos_abi::syscall`), which preserves every register but x0. + unsafe { + core::arch::asm!( + "ldr {first}, [{page}]", + "svc #0", + "ldr {second}, [{page}]", + page = in(reg) page, + first = out(reg) first, + second = lateout(reg) second, + inlateout("x0") toyos_abi::syscall::SYS_MUNMAP => answer, + in("x1") page, + in("x2") len, + in("x3") 0u64, + in("x4") 0u64, + ); + } + (first, answer, second) +} diff --git a/userland/toybox/src/arch/mod.rs b/userland/toybox/src/arch/mod.rs new file mode 100644 index 0000000000..95e0e9d154 --- /dev/null +++ b/userland/toybox/src/arch/mod.rs @@ -0,0 +1,11 @@ +//! What toybox's applets must say in assembly, one module per architecture. + +#[cfg(target_arch = "aarch64")] +mod aarch64; +#[cfg(target_arch = "aarch64")] +pub use aarch64::*; + +#[cfg(target_arch = "x86_64")] +mod x86_64; +#[cfg(target_arch = "x86_64")] +pub use x86_64::*; diff --git a/userland/toybox/src/arch/x86_64.rs b/userland/toybox/src/arch/x86_64.rs new file mode 100644 index 0000000000..c08523a994 --- /dev/null +++ b/userland/toybox/src/arch/x86_64.rs @@ -0,0 +1,51 @@ +//! x86-64's half of `unmap_touch`; its FP and first-entry probes are +//! elsewhere, or owed. + +pub mod fp_isolation { + pub fn main(_args: Vec) { + panic!("fp_isolation probes AArch64's FP/SIMD switch; x86-64's is test_rs_fpu_isolation"); + } +} + +pub mod first_entry { + pub fn main(_args: Vec) { + panic!( + "first_entry probes AArch64's first entry to EL0; x86-64's is owed by \ + issues/isolation/a-new-x86-thread-enters-ring-3-holding-kernel-register-values.md" + ); + } +} + +/// Reads the word at `page`, unmaps the `len` bytes there with `SYS_MUNMAP` +/// and reads the word again, in one block, so nothing but the unmap itself +/// can switch the CPU between the reads. Answers the first read, the unmap's +/// answer and the second read. +/// +/// # Safety +/// `page` starts a mapping of `len` bytes that nothing else names. The second +/// read ends the process unless the unmap was refused or left its +/// translation standing. +pub unsafe fn read_unmap_read(page: *const u64, len: usize) -> (u64, u64, u64) { + let (first, answer, second): (u64, u64, u64); + // SAFETY: the caller's; the two loads name `page` and the `syscall` is + // the ABI's (`toyos_abi::syscall`), which clobbers rax, rcx and r11. + unsafe { + core::arch::asm!( + "mov {first}, qword ptr [{page}]", + "syscall", + "mov {second}, qword ptr [{page}]", + page = in(reg) page, + first = out(reg) first, + second = lateout(reg) second, + in("rdi") toyos_abi::syscall::SYS_MUNMAP, + in("rsi") page, + in("rdx") len, + in("r8") 0u64, + in("r9") 0u64, + out("rax") answer, + out("rcx") _, + out("r11") _, + ); + } + (first, answer, second) +} diff --git a/userland/toybox/src/debug_refused.rs b/userland/toybox/src/debug_refused.rs new file mode 100644 index 0000000000..00cf3350f0 --- /dev/null +++ b/userland/toybox/src/debug_refused.rs @@ -0,0 +1,20 @@ +//! `debug_refused`: a `SYS_DEBUG` action this machine has no counterpart for +//! is refused, and the kernel lives: the double fault, which AArch64 has no +//! exception for, and the TLB acknowledgement delay, which its broadcast +//! invalidation has no acknowledgement for. Needs a kernel built with +//! `test-actuators`. + +use toyos_abi::syscall::debug_action::{DOUBLE_FAULT, TLB_ACK_DELAY_ARM, TLB_ACK_DELAY_DISARM}; +use toyos_abi::syscall::{debug, debug_with, SyscallError}; + +pub fn main(_args: Vec) { + let refused = SyscallError::NotSupported.to_u64(); + for (name, answer) in [ + ("DOUBLE_FAULT", debug(DOUBLE_FAULT)), + ("TLB_ACK_DELAY_ARM", debug_with(TLB_ACK_DELAY_ARM, 1_000_000)), + ("TLB_ACK_DELAY_DISARM", debug(TLB_ACK_DELAY_DISARM)), + ] { + assert_eq!(answer, refused, "SYS_DEBUG {name} answered {answer:#x}, not NotSupported"); + } + println!("debug_refused: SYS_DEBUG's double fault and TLB acknowledgement delay were refused"); +} diff --git a/userland/toybox/src/main.rs b/userland/toybox/src/main.rs index 9e3df68770..449ce6e89d 100644 --- a/userland/toybox/src/main.rs +++ b/userland/toybox/src/main.rs @@ -1,5 +1,7 @@ +mod arch; mod cat; mod cp; +mod debug_refused; mod echo; mod free; mod grep; @@ -9,6 +11,7 @@ mod ls; mod mkdir; mod mv; mod net; +mod preempt; mod ps; mod pwd; mod reboot; @@ -18,6 +21,7 @@ mod shutdown; mod spin; mod stats; mod tone; +mod unmap_touch; macro_rules! commands { ($($name:ident),*) => { @@ -30,7 +34,9 @@ macro_rules! commands { }; } -commands!(cat, cp, echo, free, grep, hexdump, locale, ls, mkdir, mv, net, ps, pwd, reboot, rm, screen, shutdown, spin, stats, tone); +use arch::{first_entry, fp_isolation}; + +commands!(cat, cp, debug_refused, echo, first_entry, fp_isolation, free, grep, hexdump, locale, ls, mkdir, mv, net, preempt, ps, pwd, reboot, rm, screen, shutdown, spin, stats, tone, unmap_touch); fn main() { let args: Vec = std::env::args().collect(); diff --git a/userland/toybox/src/preempt.rs b/userland/toybox/src/preempt.rs new file mode 100644 index 0000000000..60e147a793 --- /dev/null +++ b/userland/toybox/src/preempt.rs @@ -0,0 +1,30 @@ +//! `preempt`: whether an interrupt takes the CPU from a thread that never +//! enters the kernel itself. A second thread counts in a loop with no syscall, +//! and once it has gone round twice no fault either; this one yields until the +//! count has moved past what it last read, twice. On one CPU it runs between +//! those reads only if the counting thread lost the CPU to an interrupt, so on +//! a kernel whose timer never preempts the line below never comes. + +use std::sync::atomic::{AtomicU64, Ordering::Relaxed}; + +static COUNT: AtomicU64 = AtomicU64::new(0); + +pub fn main(_args: Vec) { + std::thread::spawn(|| loop { + COUNT.fetch_add(1, Relaxed); + }); + let first = moved_past(1); + let second = moved_past(first); + println!("preempt: the counting thread was preempted twice, at counts {first} and {second}"); +} + +/// Yield until the count is past `seen`, and say where it is. +fn moved_past(seen: u64) -> u64 { + loop { + std::thread::yield_now(); + let now = COUNT.load(Relaxed); + if now > seen { + return now; + } + } +} diff --git a/userland/toybox/src/unmap_touch.rs b/userland/toybox/src/unmap_touch.rs new file mode 100644 index 0000000000..7acc4b9697 --- /dev/null +++ b/userland/toybox/src/unmap_touch.rs @@ -0,0 +1,57 @@ +//! `unmap_touch`: whether a page's unmapping reaches the TLB. A child writes +//! a page, unmaps it and reads it at once; the read must end the child, +//! because a translation left cached past the unmap still reaches the frame. + +use std::process::Command; + +use toyos_abi::syscall::{mmap, MmapFlags, MmapProt}; + +const TRIALS: usize = 4; +const PAGE: usize = 2 * 1024 * 1024; +/// The status `kill_process(-1)` gives a process whose fault nothing serves; +/// a refused unmap or a panic ends the child with another. +const FAULTED: i32 = -1; +/// The child's last line before the unmap and the read. +const UNMAPPING: &str = "unmap_touch: written; unmapping and reading"; +/// The child's line if the read came back. +const READ_BACK: &str = "unmap_touch: read"; + +pub fn main(args: Vec) { + match args.first().map(String::as_str) { + None => judge(), + Some("touch") => touch(), + Some(other) => panic!("unmap_touch: unknown mode {other:?}"), + } +} + +fn judge() { + for trial in 0..TRIALS { + let out = Command::new("/system/bin/unmap_touch") + .arg("touch") + .output() + .unwrap_or_else(|e| panic!("unmap_touch: the child would not spawn: {e}")); + let said = String::from_utf8_lossy(&out.stdout); + assert!(said.contains(UNMAPPING), "trial {trial}: the child never reached the unmap: {said:?}"); + assert!( + out.status.code() == Some(FAULTED) && !said.contains(READ_BACK), + "trial {trial}: the child exited {:?}, not faulted ({FAULTED}), on the read of a page it had just unmapped: {said:?}", + out.status.code(), + ); + } + println!("unmap_touch: {TRIALS} reads of a page just unmapped each ended their process"); +} + +fn touch() { + // SAFETY: a fresh anonymous mapping this function alone names. + let page = unsafe { mmap(core::ptr::null_mut(), PAGE, MmapProt::READ | MmapProt::WRITE, MmapFlags::ANONYMOUS | MmapFlags::PRIVATE) }; + assert!(!page.is_null(), "unmap_touch: mmap refused"); + // SAFETY: inside the mapping just made. + unsafe { page.write_volatile(0x5A) }; + println!("{UNMAPPING}"); + // After the last print: a print can switch the CPU, and a switch can drop + // the translation the first read caches. + // SAFETY: the mapping just made, whole, which nothing else names; the + // second read ending this process is what the child exists to make. + let (first, answer, second) = unsafe { crate::arch::read_unmap_read(page.cast(), PAGE) }; + println!("{READ_BACK} {second:#x} after the unmap answered {answer:#x}, {first:#x} before it"); +}