Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
15 commits
Select commit Hold shift + click to select a range
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
93 changes: 46 additions & 47 deletions bootloader/src/main.rs
Original file line number Diff line number Diff line change
Expand Up @@ -8,7 +8,7 @@ use core::mem;
use alloc::vec;
use alloc::alloc::Layout;
use toyos_elf::section::SectionTable;
use toyos_elf::{RelaTable, RelocKind};
use toyos_elf::{rela, ImageOffset, Op, RelaTable, SymTab};
use uefi::{
prelude::*,
CStr16,
Expand Down Expand Up @@ -344,20 +344,24 @@ fn load_kernel_elf(kernel_elf_bytes: &[u8]) -> LoadedKernel {
// make the image runnable, so a file with no readable table is refused
// rather than started unrelocated.
let sections = layout
.section_headers
.section_headers()
.and_then(|table| file_range(kernel_elf_bytes, table.file_offset, table.byte_len() as u64))
.map(SectionTable::new)
.expect("kernel.elf: no section header table inside the file");

let stack_size: usize = 8 * 1024 * 1024; // 8MB

println!("Kernel stack size: {}", stack_size);
// `vaddr_max` is the largest `p_vaddr + p_memsz` over the `PT_LOAD`
// The extent's end is the largest `p_vaddr + p_memsz` over the `PT_LOAD`
// segments, and the image is laid out at its own vaddrs — so it is what the
// kernel's memory has to cover before the stack is added after it.
let placed = toyos_elf::StackedImage::place(layout.vaddr_max, stack_size as u64)
let extent = layout.extent();
let placed = toyos_elf::StackedImage::place(extent.max(), stack_size as u64)
.expect("kernel.elf: image plus stack does not fit an allocation");
let mem_size = usize::try_from(placed.size).expect("kernel.elf: image plus stack does not fit an allocation");
// Where an image offset lies in `process_mem`: the image sits at its own
// vaddrs, which begin at the extent's start.
let at = |offset: ImageOffset| (extent.min() + offset.get()) as usize;

println!("Kernel memory size: {}", mem_size);

Expand All @@ -366,9 +370,9 @@ fn load_kernel_elf(kernel_elf_bytes: &[u8]) -> LoadedKernel {

for segment in layout.segments() {
println!("Loading segment: {:?}", segment);
let src = file_range(kernel_elf_bytes, segment.file_offset, segment.filesz)
let src = file_range(kernel_elf_bytes, segment.file_offset(), segment.filesz())
.expect("kernel.elf: PT_LOAD file extent is past the end of the file");
let vstart = segment.vaddr as usize;
let vstart = at(segment.image().start());
// In bounds by construction: `mem_size` is at least
// `p_vaddr + p_memsz` for this segment and `p_filesz <= p_memsz`.
process_mem[vstart..vstart + src.len()].copy_from_slice(src);
Expand All @@ -377,58 +381,53 @@ fn load_kernel_elf(kernel_elf_bytes: &[u8]) -> LoadedKernel {
let rela_sections =
sections.rela_sections().unwrap_or_else(|form| panic!("kernel.elf: {form} is not supported"));

// Both fields of a relocation index the image and both come out of the
// file: `r_offset` is the destination of an 8-byte store and `r_addend` the
// address stored. Unchecked, the store is an arbitrary write anywhere in the
// machine, made before ExitBootServices with firmware still live — so every
// entry goes through the parse the kernel's own loader uses.
let rules = rela::Rules {
extent,
window: (0, mem_size as u64),
fill: None,
tls: None,
};
let mut reloc_count = 0u64;
for section in rela_sections {
let table = file_range(kernel_elf_bytes, section.offset, section.size)
.expect("kernel.elf: SHT_RELA section is past the end of the file");
for rela in RelaTable::new(table, arch::ELF_MACHINE).iter() {
match rela.kind {
RelocKind::Relative => {
// Both fields index the image and both come out of the
// file: `r_offset` is the destination of an 8-byte store
// and `r_addend` is the address stored. Unchecked, the
// store is an arbitrary write anywhere in the machine, made
// before ExitBootServices with firmware still live.
let offset = rela.offset;
let addend = rela.addend;
assert!(
offset.checked_add(8).is_some_and(|end| end <= mem_size as u64),
"kernel.elf: relocation stores 8 bytes at {offset:#x}, outside the {mem_size:#x}-byte image"
);
assert!(
(0..=mem_size as i64).contains(&addend),
"kernel.elf: relocation addend {addend:#x} is outside the {mem_size:#x}-byte image"
);
// SAFETY: `addend` is asserted above to be in `0..=mem_size`,
// so this is at most one byte past the end of `process_mem`'s
// allocation — in bounds for pointer arithmetic, and never
// dereferenced: only the resulting address is used.
let value = PHYS_OFFSET + unsafe { process_mem.as_ptr().add(addend as usize) } as u64;
unsafe {
// SAFETY: `offset + 8 <= mem_size` is asserted above, so
// the 8-byte write lands fully inside `process_mem`'s
// allocation. `write_unaligned`, not `write`: an
// `r_offset` from the file is not guaranteed 8-byte
// aligned by anything checked here, only by the
// linker emitting `R_X86_64_RELATIVE` against aligned
// slots — a fact this reader has no way to verify.
process_mem
.as_mut_ptr()
.add(offset as usize)
.cast::<u64>()
.write_unaligned(value);
}
reloc_count += 1;
}
kind => panic!("kernel.elf: unsupported relocation type {kind:?}"),
for raw in RelaTable::new(table, arch::ELF_MACHINE).iter() {
let parsed = rela::parse(raw, &rules, SymTab::empty()).unwrap_or_else(|e| panic!("kernel.elf: {e}"));
let Some((offset, Op::Relative(target))) = parsed.map(|r| (r.offset(), r.op())) else {
panic!("kernel.elf: unsupported relocation type {:?}", raw.kind());
};
// SAFETY: `target` is inside the image, so this is at most one byte
// past the end of `process_mem`'s allocation — in bounds for
// pointer arithmetic, and never dereferenced: only the resulting
// address is used.
let value = PHYS_OFFSET + unsafe { process_mem.as_ptr().add(at(target)) } as u64;
unsafe {
// SAFETY: the parse put `offset + 8` inside `[0, mem_size)`, so
// the 8-byte write lands fully inside `process_mem`'s
// allocation. `write_unaligned`, not `write`: an `r_offset`
// from the file is not guaranteed 8-byte aligned by anything
// checked here, only by the linker emitting
// `R_X86_64_RELATIVE` against aligned slots — a fact this
// reader has no way to verify.
process_mem
.as_mut_ptr()
.add(offset as usize)
.cast::<u64>()
.write_unaligned(value);
}
reloc_count += 1;
}
}
println!("Applied {} relocations", reloc_count);

LoadedKernel {
memory: process_mem,
entry_offset: layout.entry as usize,
entry_offset: at(layout.entry()),
stack_offset: placed.stack as usize,
stack_size,
}
Expand Down
20 changes: 20 additions & 0 deletions issues/design-debt/elf-domain-lint-line-not-yet-added.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
---
status: assigned
kind: track
opened: 2026-09-27
---

# `toyos-elf` does not `forbid(clippy::arithmetic_side_effects)` yet

The domain-lint line —
`#![forbid(clippy::arithmetic_side_effects, clippy::indexing_slicing, …)]` on
`toyos-elf/src/lib.rs` — is what would make an unchecked `+`/`-`/`[]` on a
file-chosen value a compile error, so the "every number is checked" property
this crate rests on is enforced rather than reviewed.

It is not added.

Exit condition: the line is on `lib.rs`, every site inside it is checked or
carries a one-clause `#[allow]`, and CI runs clippy on the crate with it.

Held by the orchestrator.
Original file line number Diff line number Diff line change
@@ -0,0 +1,28 @@
---
status: assigned
kind: finding
opened: 2026-09-27
---

# `LoadedLib::write_at`'s "sole writer" `# Safety` is false in the `dlopen` path

`LoadedLib::write_at`'s `# Safety` says the caller must be the sole writer of
the module's image for the call's duration. In `load_shared_lib` that holds: the
image is exclusively owned before it is mapped. In `sys_dlopen` it does not.
`map_into` maps the module into the running process first, and only then does
`resolve_dlopen_relocs` / `apply_tpoff_relocs` / `apply_dtpoff_relocs` call
`write_at` — so a peer thread of the same process can be reading (or, for a
`Shared` module's private window, touching) those pages while the relocation
writes land.

For an `Owned` module the writes go to freshly allocated pages the peer has no
handle to yet, so the race is benign in practice; for a `Shared` module the
writes go to the private `rw_alloc`, likewise not yet handed out. But the
`# Safety` contract as written is not the one the `dlopen` caller meets, so it
cannot be the thing that makes those `unsafe` blocks sound.

Exit condition: either the writes move before `map_into`, or the
`# Safety` is restated to the invariant the `dlopen` path actually upholds and
every call site's `SAFETY:` cites it.

Held by the orchestrator.
7 changes: 7 additions & 0 deletions kernel/src/actuator.rs
Original file line number Diff line number Diff line change
Expand Up @@ -187,6 +187,13 @@ actuators! {
/// and `mmap` staged inside the copy. Judged by `user_copy_races_munmap`.
copy_meets_a_remap = "copy-meets-a-remap";

/// Hold a thread spawn whose argument carries `loader::rebase_window`'s
/// mark between its TLS block being given an address and the block's
/// pointers being rebased to it: where the process can already reach the
/// block, until a sibling has stored into its DTV; where it cannot, it
/// says so. Judged by `tls_rebase_window`.
tls_rebase_window = "tls-rebase-window";

/// Stall the bind of a disk that arrives while another is held for its
/// device, for less than `usb-slow-return` does, and leave every transfer
/// of the operation the held call sends again on it unanswered, once, each
Expand Down
66 changes: 30 additions & 36 deletions kernel/src/elf/cache.rs
Original file line number Diff line number Diff line change
Expand Up @@ -20,25 +20,28 @@ use crate::process::PageAlloc;
use crate::sync::Lock;
use crate::vfs::BackingId;
use crate::UserAddr;
use toyos_elf::{RelaCounts, RelocKind};
use toyos_elf::dynamic::InitArray;
use toyos_elf::rela::Rules;
use toyos_elf::{ImageRange, Op, RelaCounts, RelocKind, SymIndex, TlsRef};

/// A module's non-`RELATIVE` relocations, extracted once at cache time.
/// A module's non-`RELATIVE` relocations, parsed and extracted once at cache
/// time, each as `(r_offset, what it names)`.
#[derive(Clone)]
pub struct CachedRelocs {
/// `GLOB_DAT` and `JUMP_SLOT`: (offset, symbol).
pub bind: Vec<(u64, u32)>,
pub tpoff64: Vec<(u64, u32, i64)>,
pub tpoff32: Vec<(u64, u32, i64)>,
/// The kernel writes a module id here.
pub dtpmod64: Vec<(u64, u32, i64)>,
/// `GLOB_DAT` and `JUMP_SLOT`.
pub bind: Vec<(u64, SymIndex)>,
pub tpoff64: Vec<(u64, TlsRef)>,
pub tpoff32: Vec<(u64, TlsRef)>,
/// The kernel writes a module id here; `None` names the module itself.
pub dtpmod64: Vec<(u64, Option<SymIndex>)>,
/// The kernel writes a TLS offset within the module here.
pub dtpoff64: Vec<(u64, u32, i64)>,
pub dtpoff64: Vec<(u64, TlsRef)>,
}

// Extracts every non-`RELATIVE` entry, or `None` if it would not fit one kernel allocation.
fn prescan_relocs(lib: &LoadedLib) -> Option<CachedRelocs> {
let counts = RelaCounts::of(lib.relocations());
let widest = core::mem::size_of::<(u64, u32, i64)>();
let counts = RelaCounts::of(lib.raw_relocations());
let widest = core::mem::size_of::<(u64, TlsRef)>().max(core::mem::size_of::<(u64, Option<SymIndex>)>());
// Excludes `Relative`: bounding on it would refuse to cache nearly every library.
let kept = [RelocKind::GlobDat, RelocKind::Tpoff64, RelocKind::Tpoff32,
RelocKind::DtpMod64, RelocKind::DtpOff64];
Expand All @@ -55,13 +58,13 @@ fn prescan_relocs(lib: &LoadedLib) -> Option<CachedRelocs> {
dtpoff64: Vec::with_capacity(counts.dtpoff64),
};
for r in lib.relocations() {
match r.kind {
RelocKind::GlobDat | RelocKind::JumpSlot => relocs.bind.push((r.offset, r.sym)),
RelocKind::Tpoff64 => relocs.tpoff64.push((r.offset, r.sym, r.addend)),
RelocKind::Tpoff32 => relocs.tpoff32.push((r.offset, r.sym, r.addend)),
RelocKind::DtpMod64 => relocs.dtpmod64.push((r.offset, r.sym, r.addend)),
RelocKind::DtpOff64 => relocs.dtpoff64.push((r.offset, r.sym, r.addend)),
_ => {}
match r.op() {
Op::Bind(sym) => relocs.bind.push((r.offset(), sym)),
Op::Tpoff64(t) => relocs.tpoff64.push((r.offset(), t)),
Op::Tpoff32(t) => relocs.tpoff32.push((r.offset(), t)),
Op::DtpMod64(sym) => relocs.dtpmod64.push((r.offset(), sym)),
Op::DtpOff64(t) => relocs.dtpoff64.push((r.offset(), t)),
Op::Relative(_) => {}
}
}
Some(relocs)
Expand All @@ -74,15 +77,12 @@ struct Snapshot {
dynsym: Option<KernelSlice>,
dynstr: Option<KernelSlice>,
tls_template: Option<KernelSlice>,
tls_memsz: usize,
tls_align: usize,
rela: Option<KernelSlice>,
jmprel: Option<KernelSlice>,
gnu_hash: Option<KernelSlice>,
eh_frame_hdr_vaddr: u64,
eh_frame_hdr_size: u64,
init_array_vaddr: u64,
init_array_size: u64,
rules: Rules,
eh_frame_hdr: Option<ImageRange>,
init_array: Option<InitArray>,
span: u64,
rw_lo: u64,
rw_hi: u64,
Expand All @@ -95,15 +95,12 @@ impl Snapshot {
dynsym: lib.dynsym,
dynstr: lib.dynstr,
tls_template: lib.tls_template,
tls_memsz: lib.tls_memsz,
tls_align: lib.tls_align,
rela: lib.rela,
jmprel: lib.jmprel,
gnu_hash: lib.gnu_hash,
eh_frame_hdr_vaddr: lib.eh_frame_hdr_vaddr,
eh_frame_hdr_size: lib.eh_frame_hdr_size,
init_array_vaddr: lib.init_array_vaddr,
init_array_size: lib.init_array_size,
rules: lib.rules,
eh_frame_hdr: lib.eh_frame_hdr,
init_array: lib.init_array,
span: lib.span,
rw_lo: lib.rw_lo,
rw_hi: lib.rw_hi,
Expand All @@ -124,16 +121,13 @@ impl Snapshot {
dynsym: self.dynsym,
dynstr: self.dynstr,
tls_template: self.tls_template,
tls_memsz: self.tls_memsz,
tls_align: self.tls_align,
rela: self.rela,
jmprel: self.jmprel,
gnu_hash: self.gnu_hash,
cached_relocs,
eh_frame_hdr_vaddr: self.eh_frame_hdr_vaddr,
eh_frame_hdr_size: self.eh_frame_hdr_size,
init_array_vaddr: self.init_array_vaddr,
init_array_size: self.init_array_size,
rules: self.rules,
eh_frame_hdr: self.eh_frame_hdr,
init_array: self.init_array,
span: self.span,
rw_lo: self.rw_lo,
rw_hi: self.rw_hi,
Expand Down
Loading
Loading