diff --git a/.gitignore b/.gitignore index 40daf08..4d27d65 100644 --- a/.gitignore +++ b/.gitignore @@ -8,3 +8,6 @@ .DS_Store /benchmark-*.txt /.claude/ + +# Standalone bench crates (their own workspaces) build next to themselves. +bench/*/target/ diff --git a/Cargo.toml b/Cargo.toml index 3467311..7396d6e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -155,7 +155,7 @@ unsafe_op_in_unsafe_fn = "deny" # unify across the dependency graph, so two consumers wanting different # geometries would silently get one of them. A cfg is set by the DELIVERABLE, # the same way a Janus firmware picks its chip. -unexpected_cfgs = { level = "warn", check-cfg = ["cfg(loom)", "cfg(kani)", "cfg(ra_small_profile)", "cfg(ra_single_threaded)", "cfg(ra_max_extents, values(\"8\", \"16\", \"64\"))", "cfg(ra_aligned_region)", "cfg(ra_segment_size, values(\"256k\"))", "cfg(ra_generic_collect, values(\"64\", \"4096\", \"65536\"))"] } +unexpected_cfgs = { level = "warn", check-cfg = ["cfg(loom)", "cfg(kani)", "cfg(ra_small_profile)", "cfg(ra_single_threaded)", "cfg(ra_max_extents, values(\"8\", \"16\", \"64\"))", "cfg(ra_aligned_region)", "cfg(ra_segment_size, values(\"256k\"))", "cfg(ra_generic_collect, values(\"64\", \"4096\", \"65536\"))", "cfg(ra_small_wsize, values(\"256\", \"512\"))"] } # Lint policy (hardening gate H-15). `pedantic` and `nursery` are ENABLED at # workspace level and the build is clean under them, because every group diff --git a/README.md b/README.md index cc4ac82..99be7e4 100644 --- a/README.md +++ b/README.md @@ -23,7 +23,11 @@ does not offer. ## The headline - Counting only the allocator's own instructions, where the comparison actually lives: - **0.66× / 0.83× / 0.85×**. + **0.52× / 0.81× / 0.83× mimalloc** on lua / perl / sqlite. +- **Rust programs got their own fast path**: `#[global_allocator]` users see + **up to −27 % whole-program instructions** against 2.2.0, with outputs + checked byte-for-byte on real consumers — see + [Rust programs](#rust-programs-globalalloc). - **A double free aborts instead of corrupting.** Upstream mimalloc accepts it silently in release builds; we detect it on both the local and the cross-thread path and abort. @@ -80,7 +84,7 @@ WHOLE-PROGRAM instructions: | workload | vs mimalloc | vs jemalloc | vs glibc | |---|---:|---:|---:| -| lua | **0.97** | **0.84** | 0.82 | +| lua | **0.96** | **0.84** | 0.81 | | perl | **0.99** | **0.89** | 0.81 | | sqlite | **1.00** | **0.98** | 0.99 | @@ -90,9 +94,61 @@ runs, decomposed: | workload | allocator share | **allocator-only ra/mi** | whole program | floor if our allocator cost ZERO | |---|---:|---:|---:|---:| -| lua | 4.9% | **0.66** | 0.97 | 0.93 | -| perl | 3.3% | **0.83** | 0.99 | 0.96 | -| sqlite | 1.5% | **0.85** | 1.00 | 0.98 | +| lua | 3.9% | **0.52** | 0.96 | 0.93 | +| perl | 3.2% | **0.81** | 0.99 | 0.96 | +| sqlite | 1.5% | **0.83** | 1.00 | 0.98 | + +Allocator-only Ir is summed by OBJECT from the raw callgrind file +(`bench/icount-arms.sh`), so code inlined from any source file counts. Measured +2026-09-25 against the mimalloc v2.4.5 oracle build. + +**What moved since 2.2.0** — four rounds of exact-instruction work, every +change gated on the counts and on the full suites (`docs/LEDGER.md`, +CURIOSITY rounds one to four). Per-op, `bench/opscan.c` (lower is better): + +| op | 2.2.0 (`c9631f4`) | now | Δ | +|---|---:|---:|---:| +| cross-thread free (`xthread`) | 113.71 | **77.26** | −32 % | +| huge (2 MiB) alloc + free | 689.68 | **293.68** | −57 % | +| `big` (4 KiB) alloc + free | 121.00 | **92.00** | −24 % | +| `mixed` sizes | 110.93 | **92.82** | −16 % | +| `calloc` | 107.57 | **93.19** | −13 % | +| `posix_memalign` | 97.88 | **83.69** | −14 % | +| `realloc` (grow) | 259.86 | **231.20** | −11 % | +| `small` (32 B) alloc + free | 54.39 | **52.25** | −4 % | + +Process start-up also got cheaper: reading the allocator's options used to +walk the environment 76 times; on Linux it now walks it once, about 25,000 +instructions saved on every process launch. + +### Rust programs (`GlobalAlloc`) + +Rust deliverables reach the allocator through `GlobalAlloc`, which the C +instruments above never exercise. Three deterministic Rust workloads +(`bench/rust-globalalloc.sh`, whole-program Ir per 20,000 steps, identical +checksums between arms): + +| workload | 2.2.0 (`c9631f4`) | now | Δ | +|---|---:|---:|---:| +| boxed trees, buffers, `Rc` | 23,357,670 | 18,436,430 | **−21.1 %** | +| hash maps, B-trees, strings | 28,220,088 | 26,577,921 | **−5.8 %** | +| cache-line-aligned buffers (`realloc`) | 1,856,092 | 1,349,905 | **−27.3 %** | +| short-lived threads | 12,201,584 | 12,105,054 | −0.8 % | + +What changed: the `GlobalAlloc` methods are `#[inline]` (as the `mimalloc` +crate's are), so rustc's `__rust_alloc` shims carry the fast path instead of +jumping to it; `dealloc` is the free fast path with its null test folded +away; layouts aligned to 16 bytes — every hashbrown table — are served from +the ordinary size classes, which are already 16-aligned, instead of the +aligned path; and `realloc` at any alignment keeps a block in place when it +fits instead of always copying. + +**Checked on real consumers, not just counted.** On 2026-09-25 SpaceDB (whole +workspace, 314 tests) and rusty_zstd (194 tests plus C-zstd cross tests) ran +on this allocator on Windows and Linux, and the Silesia corpus went through +rusty_zstd's CLI: every file at levels 1/3/9/19 and 4 threads compressed +**byte-identically** to the same CLI on 2.2.0 and round-tripped through C zstd +1.5.7 both ways (`tools/corpus/`, `docs/LEDGER.md` DOWNSTREAM CORPUS). **Where this came from: `free`.** It is the function a real program spends its allocator time in, so it is the one worth counting instruction by instruction. @@ -318,6 +374,16 @@ suite asserts the allocator *aborts* on a poisoned free list. **Portability** — x86-64 and aarch64, Linux, macOS and Windows, plus `wasm32-unknown-unknown` via `memory.grow`. +**Memory** — freed memory stays mapped and committed by default, as it does +in upstream mimalloc: `purge_delay` ships at `-1`, so a process that once held +a large peak keeps reading near that peak in `RSS` / working set afterwards +(a consumer measured 461 MB retained after freeing 456 MB of huge blocks, +where the Windows heap fell back to 4 MB — and mimalloc read the same 461 MB). +That is the trade that makes re-use fast. To hand freed spans back, set +`RUSTY_ALLOC_PURGE_DELAY` (or `MIMALLOC_PURGE_DELAY`) to a delay in +milliseconds, `0` for as soon as a span is free, or call +`options::set(15, delay)` before the first allocation. + ## Install ```toml diff --git a/bench/alignednew.cpp b/bench/alignednew.cpp new file mode 100644 index 0000000..1e134a9 --- /dev/null +++ b/bench/alignednew.cpp @@ -0,0 +1,26 @@ +// Deterministic C++17 over-aligned new/delete churn: the `align_val_t` +// operators, which a type declared `alignas(64)` (a cache-line-padded +// counter, a SIMD block) reaches on every `new`. Nothing else in the +// repository exercises them. Same two-point shape as sizedchurn.cpp: +// Ir/op = (Ir(2n) - Ir(n)) / n, argv[1] = n. +#include + +struct alignas(64) Line { unsigned char b[64]; }; +struct alignas(32) Pair { double a[4]; }; + +int main(int argc, char** argv) { + const long iters = (argc > 1) ? atol(argv[1]) : 200000; + Line* keep[64] = {}; + for (long k = 0; k < iters; k++) { + Line* l = new Line; + l->b[0] = (unsigned char)k; + Pair* p = new Pair[3]; + p[0].a[0] = (double)k; + delete[] p; + int s = (int)(k & 63); + delete keep[s]; + keep[s] = l; + } + for (auto* l : keep) delete l; + return 0; +} diff --git a/bench/fastpath-ir/Cargo.lock b/bench/fastpath-ir/Cargo.lock new file mode 100644 index 0000000..930e2af --- /dev/null +++ b/bench/fastpath-ir/Cargo.lock @@ -0,0 +1,118 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "fastpath-ir" +version = "0.0.0" +dependencies = [ + "rusty_alloc-api", +] + +[[package]] +name = "libc" +version = "0.2.189" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" + +[[package]] +name = "portable-atomic" +version = "1.15.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" + +[[package]] +name = "rusty_alloc" +version = "2.2.0" +dependencies = [ + "libc", + "portable-atomic", + "windows-sys", +] + +[[package]] +name = "rusty_alloc-api" +version = "2.2.0" +dependencies = [ + "rusty_alloc", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-sys" +version = "0.60.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2f500e4d28234f72040990ec9d39e3a6b950f9f22d3dba18416c35882612bcb" +dependencies = [ + "windows-targets", +] + +[[package]] +name = "windows-targets" +version = "0.53.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4945f9f551b88e0d65f3db0bc25c33b8acea4d9e41163edf90dcd0b19f9069f3" +dependencies = [ + "windows-link", + "windows_aarch64_gnullvm", + "windows_aarch64_msvc", + "windows_i686_gnu", + "windows_i686_gnullvm", + "windows_i686_msvc", + "windows_x86_64_gnu", + "windows_x86_64_gnullvm", + "windows_x86_64_msvc", +] + +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a9d8416fa8b42f5c947f8482c43e7d89e73a173cead56d044f6a56104a6d1b53" + +[[package]] +name = "windows_aarch64_msvc" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9d782e804c2f632e395708e99a94275910eb9100b2114651e04744e9b125006" + +[[package]] +name = "windows_i686_gnu" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "960e6da069d81e09becb0ca57a65220ddff016ff2d6af6a223cf372a506593a3" + +[[package]] +name = "windows_i686_gnullvm" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fa7359d10048f68ab8b09fa71c3daccfb0e9b559aed648a8f95469c27057180c" + +[[package]] +name = "windows_i686_msvc" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e7ac75179f18232fe9c285163565a57ef8d3c89254a30685b57d83a38d326c2" + +[[package]] +name = "windows_x86_64_gnu" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9c3842cdd74a865a8066ab39c8a7a473c0778a3f29370b5fd6b4b9aa7df4a499" + +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ffa179e2d07eee8ad8f57493436566c7cc30ac536a3379fdf008f47f6bb7ae1" + +[[package]] +name = "windows_x86_64_msvc" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6bbff5f0aada427a1e5a6da5f1f98158182f26556f345ac9e04d36d0ebed650" diff --git a/bench/fastpath-ir/Cargo.toml b/bench/fastpath-ir/Cargo.toml new file mode 100644 index 0000000..36fcdb1 --- /dev/null +++ b/bench/fastpath-ir/Cargo.toml @@ -0,0 +1,15 @@ +[package] +name = "fastpath-ir" +version = "0.0.0" +edition = "2021" +publish = false + +[workspace] + +[dependencies] +rusty_alloc-api = { path = "../../crates/rusty_alloc_api" } + +[profile.release] +opt-level = 3 +lto = true +codegen-units = 1 diff --git a/bench/fastpath-ir/src/main.rs b/bench/fastpath-ir/src/main.rs new file mode 100644 index 0000000..46064dd --- /dev/null +++ b/bench/fastpath-ir/src/main.rs @@ -0,0 +1,128 @@ +//! How many INSTRUCTIONS an alloc/free pair costs, so the Xtensa CYCLE +//! figure can be read for what it is — and what a fixed-size POOL costs +//! instead. +//! +//! The A/B cell on an ESP32-S3 measures **88 cycles** for an alloc/free pair +//! at 16 bytes and **101 at 256**. Reading the fast path says it is seven +//! operations, with `free` as its mirror. Seven operations cannot cost 88 +//! cycles unless something other than instruction count sets the price, so +//! the count is worth having exactly. +//! +//! Callgrind is deterministic, so these numbers have no error bars. +//! +//! # The two arms +//! +//! * `general` — `RustyAlloc` through `GlobalAlloc`, which is what the A/B +//! cell measures and what a `Box` or a `Vec` reaches. +//! * `pool` — a fixed-size free list. **A different strategy, not a faster +//! allocator**, and it is only applicable where the size is known at +//! compile time and the capacity can be pre-sized. An RTOS is full of +//! exactly that: TCBs, queue items, timer records. It answers a narrower +//! question, which is why it can answer it in fewer instructions. +//! +//! Both do the identical surrounding work — the same write, read, `black_box` +//! and checksum — so the difference is the allocation strategy and nothing +//! else. +//! +//! # Method +//! +//! Counted at several lengths; the cost of one pair is the **slope**, so +//! process start-up, `ld.so`, allocator init and first-touch warm-up cancel +//! exactly instead of being estimated. Same shape as `bench/opscan.sh` here. +//! +//! ```sh +//! cargo build --release +//! for n in 1000000 2000000 3000000; do +//! valgrind --tool=callgrind --callgrind-out-file=/dev/null \ +//! ./target/release/fastpath-ir "$n" 256 general +//! done +//! ``` +//! +//! # What it does NOT measure +//! +//! Cycles, and so not the quantity the A/B cell reports. This is also a +//! 64-bit host, where `SMALL_SIZE_MAX` is 1 KiB rather than the 512 bytes a +//! 32-bit chip gets, so the size BOUNDARY sits elsewhere here. Only the +//! shape carries over, which is all this is asked for. + +use std::alloc::{GlobalAlloc, Layout}; + +#[global_allocator] +static ALLOC: rusty_alloc_api::RustyAlloc = rusty_alloc_api::RustyAlloc; + +/// A fixed-size free list over one contiguous arena. +/// +/// No size classes, no bins, no page lookup: the block size is a constant of +/// the pool, so `alloc` is a pop and `free` is a push. The bounds checks are +/// left in — this is what a SAFE pool costs, not what an unchecked one +/// could. +struct Pool { + arena: Vec, + block: usize, + free: Vec, +} + +impl Pool { + fn new(block: usize, blocks: usize) -> Self { + Self { + arena: vec![0u8; block * blocks], + block, + // Highest offset first, so the first pop hands out offset 0. + free: (0..blocks).rev().map(|i| i * block).collect(), + } + } + + #[inline] + fn alloc(&mut self) -> Option { + self.free.pop() + } + + #[inline] + fn dealloc(&mut self, at: usize) { + self.free.push(at); + } +} + +fn main() { + let mut args = std::env::args().skip(1); + let rounds: usize = args + .next() + .and_then(|a| a.parse().ok()) + .unwrap_or(1_000_000); + let size: usize = args.next().and_then(|a| a.parse().ok()).unwrap_or(256); + let arm = args.next().unwrap_or_else(|| "general".to_owned()); + + let mut sum: u64 = 0; + + match arm.as_str() { + "pool" => { + let mut pool = Pool::new(size, 64); + for i in 0..rounds { + let at = pool.alloc().expect("the pool is not exhausted"); + if let Some(slot) = pool.arena.get_mut(at) { + *slot = (i & 0xff) as u8; + sum = sum.wrapping_add(u64::from(*slot)); + } + std::hint::black_box(at); + pool.dealloc(at); + } + } + _ => { + let layout = Layout::from_size_align(size, 8).expect("a valid layout"); + for i in 0..rounds { + // SAFETY: a non-zero layout, and the pointer is freed exactly + // once below with the layout it was allocated with. + unsafe { + let p = ALLOC.alloc(layout); + assert!(!p.is_null(), "out of memory"); + p.write((i & 0xff) as u8); + sum = sum.wrapping_add(u64::from(p.read())); + std::hint::black_box(p); + ALLOC.dealloc(p, layout); + } + } + } + } + + println!("rounds={rounds} size={size} arm={arm} checksum={sum}"); +} diff --git a/bench/rust-globalalloc.sh b/bench/rust-globalalloc.sh new file mode 100644 index 0000000..3bdaf11 --- /dev/null +++ b/bench/rust-globalalloc.sh @@ -0,0 +1,21 @@ +#!/usr/bin/env bash +# Whole-program Ir per `n` steps of the two Rust GlobalAlloc workloads +# (bench/rust-globalalloc), by the two-point estimator Ir(2n) - Ir(n), plus +# .text size. Linux/WSL with valgrind. usage: bench/rust-globalalloc.sh [n] +# +# Read WHOLE-program Ir here, not an allocator-only filter: with the +# GlobalAlloc methods inline, rustc places the allocator fast paths inside +# the program's own functions, where a per-file filter cannot see them. +# The workloads print a checksum; it must match between the arms. +set -euo pipefail +root=$(cd "$(dirname "$0")/.." && pwd) +n=${1:-20000} +tgt=${CARGO_TARGET_DIR:-$root/target/rust-globalalloc} +CARGO_TARGET_DIR=$tgt cargo build --release -q --manifest-path "$root/bench/rust-globalalloc/Cargo.toml" +for b in maps trees threads overaligned; do + bin=$tgt/release/$b + ir() { local o; o=$(mktemp); valgrind --tool=callgrind --callgrind-out-file="$o" --cache-sim=no --branch-sim=no "$bin" "$1" >/dev/null 2>&1; grep -m1 '^summary:' "$o" | awk '{print $2}'; rm -f "$o"; } + a=$(ir "$n"); c=$(ir $((n * 2))) + printf '%-6s Ir/%d steps %12d text %8d checksum %s\n' "$b" "$n" $((c - a)) \ + "$(size -A "$bin" | awk '$1==".text"{print $2}')" "$("$bin" "$n")" +done diff --git a/bench/rust-globalalloc/Cargo.lock b/bench/rust-globalalloc/Cargo.lock new file mode 100644 index 0000000..5e0516c --- /dev/null +++ b/bench/rust-globalalloc/Cargo.lock @@ -0,0 +1,118 @@ +# This file is automatically @generated by Cargo. +# It is not intended for manual editing. +version = 4 + +[[package]] +name = "libc" +version = "0.2.189" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" + +[[package]] +name = "portable-atomic" +version = "1.15.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" + +[[package]] +name = "rust-globalalloc" +version = "0.0.0" +dependencies = [ + "rusty_alloc-api", +] + +[[package]] +name = "rusty_alloc" +version = "2.2.0" +dependencies = [ + "libc", + "portable-atomic", + "windows-sys", +] + +[[package]] +name = "rusty_alloc-api" +version = "2.2.0" +dependencies = [ + "rusty_alloc", +] + +[[package]] +name = "windows-link" +version = "0.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f0805222e57f7521d6a62e36fa9163bc891acd422f971defe97d64e70d0a4fe5" + +[[package]] +name = "windows-sys" +version = "0.60.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f2f500e4d28234f72040990ec9d39e3a6b950f9f22d3dba18416c35882612bcb" +dependencies = [ + "windows-targets", +] + +[[package]] +name = "windows-targets" +version = "0.53.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4945f9f551b88e0d65f3db0bc25c33b8acea4d9e41163edf90dcd0b19f9069f3" +dependencies = [ + "windows-link", + "windows_aarch64_gnullvm", + "windows_aarch64_msvc", + "windows_i686_gnu", + "windows_i686_gnullvm", + "windows_i686_msvc", + "windows_x86_64_gnu", + "windows_x86_64_gnullvm", + "windows_x86_64_msvc", +] + +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a9d8416fa8b42f5c947f8482c43e7d89e73a173cead56d044f6a56104a6d1b53" + +[[package]] +name = "windows_aarch64_msvc" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9d782e804c2f632e395708e99a94275910eb9100b2114651e04744e9b125006" + +[[package]] +name = "windows_i686_gnu" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "960e6da069d81e09becb0ca57a65220ddff016ff2d6af6a223cf372a506593a3" + +[[package]] +name = "windows_i686_gnullvm" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "fa7359d10048f68ab8b09fa71c3daccfb0e9b559aed648a8f95469c27057180c" + +[[package]] +name = "windows_i686_msvc" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1e7ac75179f18232fe9c285163565a57ef8d3c89254a30685b57d83a38d326c2" + +[[package]] +name = "windows_x86_64_gnu" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9c3842cdd74a865a8066ab39c8a7a473c0778a3f29370b5fd6b4b9aa7df4a499" + +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ffa179e2d07eee8ad8f57493436566c7cc30ac536a3379fdf008f47f6bb7ae1" + +[[package]] +name = "windows_x86_64_msvc" +version = "0.53.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d6bbff5f0aada427a1e5a6da5f1f98158182f26556f345ac9e04d36d0ebed650" diff --git a/bench/rust-globalalloc/Cargo.toml b/bench/rust-globalalloc/Cargo.toml new file mode 100644 index 0000000..3e78da4 --- /dev/null +++ b/bench/rust-globalalloc/Cargo.toml @@ -0,0 +1,18 @@ +# Deterministic Rust workloads for the `GlobalAlloc` entry points +# (`rusty_alloc_api::RustyAlloc`), which no C instrument in this repository +# reaches: `__rust_alloc` and friends, hashbrown's 16-byte-aligned tables, +# over-aligned realloc, and the drop path. Standalone (its own workspace) so the +# main build never compiles it. Run `bench/rust-globalalloc.sh` under WSL/Linux. +[package] +name = "rust-globalalloc" +version = "0.0.0" +edition = "2024" +publish = false + +[workspace] + +[dependencies] +rusty_alloc-api = { path = "../../crates/rusty_alloc_api" } + +[profile.release] +debug = true diff --git a/bench/rust-globalalloc/src/bin/maps.rs b/bench/rust-globalalloc/src/bin/maps.rs new file mode 100644 index 0000000..0858213 --- /dev/null +++ b/bench/rust-globalalloc/src/bin/maps.rs @@ -0,0 +1,29 @@ +//! `maps`: HashMap/HashSet churn (hashbrown allocates its tables 16-byte +//! aligned), formatted Strings, a BTreeMap and a growing `Vec`. +//! Argument: step count; the harness takes Ir(2n) - Ir(n). +use std::collections::{BTreeMap, HashMap, HashSet}; + +#[global_allocator] +static A: rusty_alloc_api::RustyAlloc = rusty_alloc_api::RustyAlloc; + +fn main() { + let n: u64 = std::env::args().nth(1).and_then(|s| s.parse().ok()).unwrap_or(20000); + let mut sink = 0u64; + for r in 0..n / 1000 { + // many small hash maps: hashbrown allocates with 16-byte alignment + for k in 0..200u64 { + let mut m: HashMap = HashMap::new(); + for i in 0..(k % 13) { m.insert(i * 7 + r, i); } + let s: HashSet = (0..(k % 7) as u32).collect(); + sink = sink.wrapping_add(m.len() as u64 + s.len() as u64); + } + // strings and vectors that grow (realloc) + let mut v: Vec = Vec::new(); + for i in 0..400u64 { v.push(format!("item-{i}-{r}")); } + let mut b: BTreeMap = BTreeMap::new(); + for (i, s) in v.iter().enumerate() { b.insert(s.clone(), i); } + let w: Vec = (0..(r % 50) as u128).collect(); + sink = sink.wrapping_add(b.len() as u64 + w.len() as u64); + } + println!("{sink}"); +} diff --git a/bench/rust-globalalloc/src/bin/overaligned.rs b/bench/rust-globalalloc/src/bin/overaligned.rs new file mode 100644 index 0000000..8cee5ad --- /dev/null +++ b/bench/rust-globalalloc/src/bin/overaligned.rs @@ -0,0 +1,30 @@ +//! `overaligned`: buffers of cache-line-aligned elements (align 64, above +//! what the size classes guarantee) that grow, shrink to fit and are +//! re-reserved — the `GlobalAlloc::realloc` arm for alignments above two +//! words. Argument: step count. +#[global_allocator] +static A: rusty_alloc_api::RustyAlloc = rusty_alloc_api::RustyAlloc; + +#[derive(Clone, Copy)] +#[repr(align(64))] +struct Line([u64; 8]); + +fn main() { + let n: u64 = std::env::args().nth(1).and_then(|s| s.parse().ok()).unwrap_or(20000); + let mut sink = 0u64; + for r in 0..n / 20 { + let mut v: Vec = Vec::with_capacity(4); + for i in 0..(r % 29 + 3) { + v.push(Line([i; 8])); + } + // Shrink to what is used (in place when at least half stays), then + // grow back by a little (in place within the class's slack). + v.shrink_to_fit(); + v.truncate(v.len() * 3 / 4); + v.shrink_to_fit(); + v.reserve_exact(1); + v.push(Line([r; 8])); + sink = sink.wrapping_add(v.iter().map(|l| l.0[0]).sum::() + v.capacity() as u64); + } + println!("{sink}"); +} diff --git a/bench/rust-globalalloc/src/bin/threads.rs b/bench/rust-globalalloc/src/bin/threads.rs new file mode 100644 index 0000000..794ea7f --- /dev/null +++ b/bench/rust-globalalloc/src/bin/threads.rs @@ -0,0 +1,18 @@ +//! `threads`: short-lived threads, each allocating a little and exiting — +//! a blocking pool's shape. Prices per-thread heap setup and teardown, which +//! no other workload here reaches. Argument: step count (threads = n / 100). +#[global_allocator] +static A: rusty_alloc_api::RustyAlloc = rusty_alloc_api::RustyAlloc; + +fn main() { + let n: u64 = std::env::args().nth(1).and_then(|s| s.parse().ok()).unwrap_or(20000); + let mut sink = 0u64; + for t in 0..n / 100 { + let h = std::thread::spawn(move || { + let v: Vec> = (0..16).map(|i| Box::new(i + t)).collect(); + v.iter().map(|b| **b).sum::() + }); + sink = sink.wrapping_add(h.join().unwrap()); + } + println!("{sink}"); +} diff --git a/bench/rust-globalalloc/src/bin/trees.rs b/bench/rust-globalalloc/src/bin/trees.rs new file mode 100644 index 0000000..6707d36 --- /dev/null +++ b/bench/rust-globalalloc/src/bin/trees.rs @@ -0,0 +1,29 @@ +//! `trees`: no hashing. Boxed trees, growing +//! byte buffers, `Rc` graphs and short-lived `Vec>`, the allocation +//! pattern of a parser or an AST pass. +use std::rc::Rc; + +#[global_allocator] +static A: rusty_alloc_api::RustyAlloc = rusty_alloc_api::RustyAlloc; + +enum Node { Leaf(u64), Pair(Box, Box) } + +fn build(d: u32, s: u64) -> Box { + if d == 0 { Box::new(Node::Leaf(s)) } else { Box::new(Node::Pair(build(d - 1, s * 3), build(d - 1, s + 1))) } +} +fn sum(n: &Node) -> u64 { match n { Node::Leaf(v) => *v, Node::Pair(a, b) => sum(a).wrapping_add(sum(b)) } } + +fn main() { + let n: u64 = std::env::args().nth(1).and_then(|s| s.parse().ok()).unwrap_or(20000); + let mut sink = 0u64; + for r in 0..n / 1000 { + for k in 0..20 { sink = sink.wrapping_add(sum(&build(8, r + k))); } + let mut buf: Vec = Vec::new(); + for i in 0..3000u64 { buf.extend_from_slice(&i.to_le_bytes()[..(i % 8) as usize]); } + let shared: Vec>> = (0..300).map(|i| Rc::new(vec![i; (i % 17) as usize])).collect(); + let copies: Vec>> = shared.iter().cloned().collect(); + let rows: Vec> = (0..200u32).map(|i| (0..i % 23).collect()).collect(); + sink = sink.wrapping_add(buf.len() as u64 + copies.len() as u64 + rows.iter().map(|v| v.len() as u64).sum::()); + } + println!("{sink}"); +} diff --git a/bench/wasm-speed/target/.rustc_info.json b/bench/wasm-speed/target/.rustc_info.json deleted file mode 100644 index 6f6f186..0000000 --- a/bench/wasm-speed/target/.rustc_info.json +++ /dev/null @@ -1 +0,0 @@ -{"rustc_fingerprint":10247008043229276228,"outputs":{"7971740275564407648":{"success":true,"status":"","code":0,"stdout":"___.exe\nlib___.rlib\n___.dll\n___.dll\n___.lib\n___.dll\nC:\\Users\\talmo\\.rustup\\toolchains\\1.97.1-x86_64-pc-windows-msvc\npacked\n___\ndebug_assertions\npanic=\"unwind\"\nproc_macro\ntarget_abi=\"\"\ntarget_arch=\"x86_64\"\ntarget_endian=\"little\"\ntarget_env=\"msvc\"\ntarget_family=\"windows\"\ntarget_feature=\"cmpxchg16b\"\ntarget_feature=\"fxsr\"\ntarget_feature=\"sse\"\ntarget_feature=\"sse2\"\ntarget_feature=\"sse3\"\ntarget_has_atomic=\"128\"\ntarget_has_atomic=\"16\"\ntarget_has_atomic=\"32\"\ntarget_has_atomic=\"64\"\ntarget_has_atomic=\"8\"\ntarget_has_atomic=\"ptr\"\ntarget_has_atomic_primitive_alignment=\"128\"\ntarget_has_atomic_primitive_alignment=\"16\"\ntarget_has_atomic_primitive_alignment=\"32\"\ntarget_has_atomic_primitive_alignment=\"64\"\ntarget_has_atomic_primitive_alignment=\"8\"\ntarget_has_atomic_primitive_alignment=\"ptr\"\ntarget_os=\"windows\"\ntarget_pointer_width=\"64\"\ntarget_vendor=\"pc\"\nwindows\n","stderr":""},"11652014622397750202":{"success":true,"status":"","code":0,"stdout":"___.wasm\nlib___.rlib\n___.wasm\nlib___.a\nC:\\Users\\talmo\\.rustup\\toolchains\\1.97.1-x86_64-pc-windows-msvc\noff\n___\ndebug_assertions\npanic=\"abort\"\nproc_macro\ntarget_abi=\"\"\ntarget_arch=\"wasm32\"\ntarget_endian=\"little\"\ntarget_env=\"\"\ntarget_family=\"wasm\"\ntarget_feature=\"bulk-memory\"\ntarget_feature=\"multivalue\"\ntarget_feature=\"mutable-globals\"\ntarget_feature=\"nontrapping-fptoint\"\ntarget_feature=\"reference-types\"\ntarget_feature=\"sign-ext\"\ntarget_has_atomic=\"16\"\ntarget_has_atomic=\"32\"\ntarget_has_atomic=\"64\"\ntarget_has_atomic=\"8\"\ntarget_has_atomic=\"ptr\"\ntarget_has_atomic_primitive_alignment=\"16\"\ntarget_has_atomic_primitive_alignment=\"32\"\ntarget_has_atomic_primitive_alignment=\"64\"\ntarget_has_atomic_primitive_alignment=\"8\"\ntarget_has_atomic_primitive_alignment=\"ptr\"\ntarget_os=\"unknown\"\ntarget_pointer_width=\"32\"\ntarget_vendor=\"unknown\"\n","stderr":"warning: dropping unsupported crate type `dylib` for target `wasm32-unknown-unknown`\n\nwarning: dropping unsupported crate type `proc-macro` for target `wasm32-unknown-unknown`\n\nwarning: 2 warnings emitted\n\n"},"8480363167076041836":{"success":true,"status":"","code":0,"stdout":"rustc 1.97.1 (8bab26f4f 2026-07-14)\nbinary: rustc\ncommit-hash: 8bab26f4f68e0e26f0bb7960be334d5b520ea452\ncommit-date: 2026-07-14\nhost: x86_64-pc-windows-msvc\nrelease: 1.97.1\nLLVM version: 22.1.6\n","stderr":""}},"successes":{}} \ No newline at end of file diff --git a/bench/wasm-speed/target/CACHEDIR.TAG b/bench/wasm-speed/target/CACHEDIR.TAG deleted file mode 100644 index 20d7c31..0000000 --- a/bench/wasm-speed/target/CACHEDIR.TAG +++ /dev/null @@ -1,3 +0,0 @@ -Signature: 8a477f597d28d172789f06886806bc55 -# This file is a cache directory tag created by cargo. -# For information about cache directory tags see https://bford.info/cachedir/ diff --git a/bench/wasm-speed/target/release/.cargo-artifact-lock b/bench/wasm-speed/target/release/.cargo-artifact-lock deleted file mode 100644 index e69de29..0000000 diff --git a/bench/wasm-speed/target/release/.cargo-build-lock b/bench/wasm-speed/target/release/.cargo-build-lock deleted file mode 100644 index e69de29..0000000 diff --git a/bench/wasm-speed/target/release/.cargo-lock b/bench/wasm-speed/target/release/.cargo-lock deleted file mode 100644 index e69de29..0000000 diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/CACHEDIR.TAG b/bench/wasm-speed/target/wasm32-unknown-unknown/CACHEDIR.TAG deleted file mode 100644 index 20d7c31..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/CACHEDIR.TAG +++ /dev/null @@ -1,3 +0,0 @@ -Signature: 8a477f597d28d172789f06886806bc55 -# This file is a cache directory tag created by cargo. -# For information about cache directory tags see https://bford.info/cachedir/ diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.cargo-artifact-lock b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.cargo-artifact-lock deleted file mode 100644 index e69de29..0000000 diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.cargo-build-lock b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.cargo-build-lock deleted file mode 100644 index e69de29..0000000 diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.cargo-lock b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.cargo-lock deleted file mode 100644 index e69de29..0000000 diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/dep-lib-rusty_alloc b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/dep-lib-rusty_alloc deleted file mode 100644 index c35a78f..0000000 Binary files a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/dep-lib-rusty_alloc and /dev/null differ diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/invoked.timestamp b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/invoked.timestamp deleted file mode 100644 index e00328d..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/invoked.timestamp +++ /dev/null @@ -1 +0,0 @@ -This file has an mtime of when this was started. \ No newline at end of file diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/lib-rusty_alloc b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/lib-rusty_alloc deleted file mode 100644 index 628d8de..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/lib-rusty_alloc +++ /dev/null @@ -1 +0,0 @@ -9fde96e39c204a3d \ No newline at end of file diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/lib-rusty_alloc.json b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/lib-rusty_alloc.json deleted file mode 100644 index cf67230..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/lib-rusty_alloc.json +++ /dev/null @@ -1 +0,0 @@ -{"rustc":3720210673988096810,"features":"[\"default\", \"std\"]","declared_features":"[\"blockmap\", \"debug_checks\", \"default\", \"linkcheck\", \"profile\", \"secure\", \"std\"]","target":10085266625313816817,"profile":12927086336690686898,"path":2916297298608968487,"deps":[],"local":[{"CheckDepInfo":{"dep_info":"wasm32-unknown-unknown\\release\\.fingerprint\\rusty_alloc-30203c3fd9d3b00b\\dep-lib-rusty_alloc","checksum":false}}],"rustflags":[],"config":9396254390672932401,"compile_kind":14682669768258224367} \ No newline at end of file diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/output-lib-rusty_alloc b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/output-lib-rusty_alloc deleted file mode 100644 index ef30ac7..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-30203c3fd9d3b00b/output-lib-rusty_alloc +++ /dev/null @@ -1,2 +0,0 @@ -{"$message_type":"diagnostic","message":"function `parse_value` is never used","code":{"code":"dead_code","explanation":null},"level":"warning","spans":[{"file_name":"F:\\coding\\rusty_alloc\\crates\\rusty_alloc\\src\\options.rs","byte_start":13654,"byte_end":13665,"line_start":333,"line_end":333,"column_start":4,"column_end":15,"is_primary":true,"text":[{"text":"fn parse_value(s: &str) -> Option {","highlight_start":4,"highlight_end":15}],"label":null,"suggested_replacement":null,"suggestion_applicability":null,"expansion":null}],"children":[{"message":"`#[warn(dead_code)]` (part of `#[warn(unused)]`) on by default","code":null,"level":"note","spans":[],"children":[],"rendered":null}],"rendered":"\u001b[1m\u001b[93mwarning\u001b[0m\u001b[1m\u001b[97m: function `parse_value` is never used\u001b[0m\n \u001b[1m\u001b[96m--> \u001b[0mF:\\coding\\rusty_alloc\\crates\\rusty_alloc\\src\\options.rs:333:4\n \u001b[1m\u001b[96m|\u001b[0m\n\u001b[1m\u001b[96m333\u001b[0m \u001b[1m\u001b[96m|\u001b[0m fn parse_value(s: &str) -> Option {\n \u001b[1m\u001b[96m|\u001b[0m \u001b[1m\u001b[93m^^^^^^^^^^^\u001b[0m\n \u001b[1m\u001b[96m|\u001b[0m\n \u001b[1m\u001b[96m= \u001b[0m\u001b[1m\u001b[97mnote\u001b[0m: `#[warn(dead_code)]` (part of `#[warn(unused)]`) on by default\n\n"} -{"$message_type":"diagnostic","message":"1 warning emitted","code":null,"level":"warning","spans":[],"children":[],"rendered":"\u001b[1m\u001b[93mwarning\u001b[0m\u001b[1m\u001b[97m: 1 warning emitted\u001b[0m\n\n"} diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/dep-lib-rusty_alloc_api b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/dep-lib-rusty_alloc_api deleted file mode 100644 index 02bca30..0000000 Binary files a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/dep-lib-rusty_alloc_api and /dev/null differ diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/invoked.timestamp b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/invoked.timestamp deleted file mode 100644 index e00328d..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/invoked.timestamp +++ /dev/null @@ -1 +0,0 @@ -This file has an mtime of when this was started. \ No newline at end of file diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/lib-rusty_alloc_api b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/lib-rusty_alloc_api deleted file mode 100644 index eccdbee..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/lib-rusty_alloc_api +++ /dev/null @@ -1 +0,0 @@ -1e99924ea3cbaf62 \ No newline at end of file diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/lib-rusty_alloc_api.json b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/lib-rusty_alloc_api.json deleted file mode 100644 index b5c8202..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/rusty_alloc-api-65fdac25e68ac577/lib-rusty_alloc_api.json +++ /dev/null @@ -1 +0,0 @@ -{"rustc":3720210673988096810,"features":"[\"default\", \"std\"]","declared_features":"[\"debug_checks\", \"default\", \"profile\", \"secure\", \"std\"]","target":10444181462946702350,"profile":12927086336690686898,"path":17481420428904646363,"deps":[[2387993517738183206,"rusty_alloc",false,4416378242795495071]],"local":[{"CheckDepInfo":{"dep_info":"wasm32-unknown-unknown\\release\\.fingerprint\\rusty_alloc-api-65fdac25e68ac577\\dep-lib-rusty_alloc_api","checksum":false}}],"rustflags":[],"config":9396254390672932401,"compile_kind":14682669768258224367} \ No newline at end of file diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/dep-lib-speedprobe b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/dep-lib-speedprobe deleted file mode 100644 index 02bca30..0000000 Binary files a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/dep-lib-speedprobe and /dev/null differ diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/invoked.timestamp b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/invoked.timestamp deleted file mode 100644 index e00328d..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/invoked.timestamp +++ /dev/null @@ -1 +0,0 @@ -This file has an mtime of when this was started. \ No newline at end of file diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/lib-speedprobe b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/lib-speedprobe deleted file mode 100644 index dfcf9b3..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/lib-speedprobe +++ /dev/null @@ -1 +0,0 @@ -3fa7fc3be998846a \ No newline at end of file diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/lib-speedprobe.json b/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/lib-speedprobe.json deleted file mode 100644 index 3570d03..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/.fingerprint/speedprobe-f92ee3d991fb435d/lib-speedprobe.json +++ /dev/null @@ -1 +0,0 @@ -{"rustc":3720210673988096810,"features":"[\"ra\"]","declared_features":"[\"ra\"]","target":16610879620610110527,"profile":5652530863572545030,"path":10763286916239946207,"deps":[[2387993517738183206,"rusty_alloc",false,4416378242795495071],[17777520122732431131,"rusty_alloc_api",false,7111126238899640606]],"local":[{"CheckDepInfo":{"dep_info":"wasm32-unknown-unknown\\release\\.fingerprint\\speedprobe-f92ee3d991fb435d\\dep-lib-speedprobe","checksum":false}}],"rustflags":[],"config":9396254390672932401,"compile_kind":14682669768258224367} \ No newline at end of file diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc-30203c3fd9d3b00b.rlib b/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc-30203c3fd9d3b00b.rlib deleted file mode 100644 index 6064302..0000000 Binary files a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc-30203c3fd9d3b00b.rlib and /dev/null differ diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc-30203c3fd9d3b00b.rmeta b/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc-30203c3fd9d3b00b.rmeta deleted file mode 100644 index 2e62cbf..0000000 Binary files a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc-30203c3fd9d3b00b.rmeta and /dev/null differ diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc_api-65fdac25e68ac577.rlib b/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc_api-65fdac25e68ac577.rlib deleted file mode 100644 index 73c71fd..0000000 Binary files a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc_api-65fdac25e68ac577.rlib and /dev/null differ diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc_api-65fdac25e68ac577.rmeta b/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc_api-65fdac25e68ac577.rmeta deleted file mode 100644 index fa7a7d0..0000000 Binary files a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/librusty_alloc_api-65fdac25e68ac577.rmeta and /dev/null differ diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/rusty_alloc-30203c3fd9d3b00b.d b/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/rusty_alloc-30203c3fd9d3b00b.d deleted file mode 100644 index 0633985..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/rusty_alloc-30203c3fd9d3b00b.d +++ /dev/null @@ -1,26 +0,0 @@ -F:\coding\rusty_alloc\bench\wasm-speed\target\wasm32-unknown-unknown\release\deps\rusty_alloc-30203c3fd9d3b00b.d: F:\coding\rusty_alloc\crates\rusty_alloc\src\lib.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\alloc.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\arena.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\bins.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\heap.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\init.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\options.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\os.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\page.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\mod.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\wasm.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\fixed.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\random.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\segment.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\segment_map.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\slice_pool.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\stats.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\types.rs - -F:\coding\rusty_alloc\bench\wasm-speed\target\wasm32-unknown-unknown\release\deps\librusty_alloc-30203c3fd9d3b00b.rlib: F:\coding\rusty_alloc\crates\rusty_alloc\src\lib.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\alloc.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\arena.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\bins.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\heap.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\init.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\options.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\os.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\page.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\mod.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\wasm.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\fixed.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\random.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\segment.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\segment_map.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\slice_pool.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\stats.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\types.rs - -F:\coding\rusty_alloc\bench\wasm-speed\target\wasm32-unknown-unknown\release\deps\librusty_alloc-30203c3fd9d3b00b.rmeta: F:\coding\rusty_alloc\crates\rusty_alloc\src\lib.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\alloc.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\arena.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\bins.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\heap.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\init.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\options.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\os.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\page.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\mod.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\wasm.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\fixed.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\random.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\segment.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\segment_map.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\slice_pool.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\stats.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\types.rs - -F:\coding\rusty_alloc\crates\rusty_alloc\src\lib.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\alloc.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\arena.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\bins.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\heap.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\init.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\options.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\os.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\page.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\mod.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\wasm.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\fixed.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\random.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\segment.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\segment_map.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\slice_pool.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\stats.rs: -F:\coding\rusty_alloc\crates\rusty_alloc\src\types.rs: - -# env-dep:CARGO_PKG_VERSION=2.0.0 diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/rusty_alloc_api-65fdac25e68ac577.d b/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/rusty_alloc_api-65fdac25e68ac577.d deleted file mode 100644 index dca9d55..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/rusty_alloc_api-65fdac25e68ac577.d +++ /dev/null @@ -1,7 +0,0 @@ -F:\coding\rusty_alloc\bench\wasm-speed\target\wasm32-unknown-unknown\release\deps\rusty_alloc_api-65fdac25e68ac577.d: F:\coding\rusty_alloc\crates\rusty_alloc_api\src\lib.rs - -F:\coding\rusty_alloc\bench\wasm-speed\target\wasm32-unknown-unknown\release\deps\librusty_alloc_api-65fdac25e68ac577.rlib: F:\coding\rusty_alloc\crates\rusty_alloc_api\src\lib.rs - -F:\coding\rusty_alloc\bench\wasm-speed\target\wasm32-unknown-unknown\release\deps\librusty_alloc_api-65fdac25e68ac577.rmeta: F:\coding\rusty_alloc\crates\rusty_alloc_api\src\lib.rs - -F:\coding\rusty_alloc\crates\rusty_alloc_api\src\lib.rs: diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/speedprobe.d b/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/speedprobe.d deleted file mode 100644 index eaf0fbf..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/speedprobe.d +++ /dev/null @@ -1,5 +0,0 @@ -F:\coding\rusty_alloc\bench\wasm-speed\target\wasm32-unknown-unknown\release\deps\speedprobe.d: src\lib.rs - -F:\coding\rusty_alloc\bench\wasm-speed\target\wasm32-unknown-unknown\release\deps\speedprobe.wasm: src\lib.rs - -src\lib.rs: diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/speedprobe.wasm b/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/speedprobe.wasm deleted file mode 100644 index 5630084..0000000 Binary files a/bench/wasm-speed/target/wasm32-unknown-unknown/release/deps/speedprobe.wasm and /dev/null differ diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/speedprobe.d b/bench/wasm-speed/target/wasm32-unknown-unknown/release/speedprobe.d deleted file mode 100644 index 635fd0c..0000000 --- a/bench/wasm-speed/target/wasm32-unknown-unknown/release/speedprobe.d +++ /dev/null @@ -1 +0,0 @@ -F:\coding\rusty_alloc\bench\wasm-speed\target\wasm32-unknown-unknown\release\speedprobe.wasm: F:\coding\rusty_alloc\bench\wasm-speed\src\lib.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\alloc.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\arena.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\bins.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\heap.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\init.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\lib.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\options.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\os.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\page.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\fixed.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\mod.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\prim\wasm.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\random.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\segment.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\segment_map.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\slice_pool.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\stats.rs F:\coding\rusty_alloc\crates\rusty_alloc\src\types.rs F:\coding\rusty_alloc\crates\rusty_alloc_api\src\lib.rs diff --git a/bench/wasm-speed/target/wasm32-unknown-unknown/release/speedprobe.wasm b/bench/wasm-speed/target/wasm32-unknown-unknown/release/speedprobe.wasm deleted file mode 100644 index 5630084..0000000 Binary files a/bench/wasm-speed/target/wasm32-unknown-unknown/release/speedprobe.wasm and /dev/null differ diff --git a/crates/rusty_alloc/CHANGELOG.md b/crates/rusty_alloc/CHANGELOG.md index 5da82e7..1b7ec6e 100644 --- a/crates/rusty_alloc/CHANGELOG.md +++ b/crates/rusty_alloc/CHANGELOG.md @@ -9,6 +9,36 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed +- **Growing a huge block with `realloc` copied its whole reservation.** A + huge segment's page reported its usable size as the chunk-rounded + reservation minus the header — 64 MiB for a 33 MB request, 96 MiB for a + 64 MB one — and `realloc` copies `usable_size` bytes when it moves a block, + so growing a 33 MB block copied, and first-touched on both sides, twice what + the caller had written; `zalloc` on a recycled chunk zeroed the same extent. + A consumer measured the two `realloc` steps that cross the segment size at + **1.41x and 1.23x mimalloc, 0/6 pairs**, with every step below it faster + than mimalloc, and a `Vec` pushed to 64 MB at 1.23x + (`docs/plans/youslowbro.md` §3). The page now reports the request rounded + up to a slice — upstream's `psize`, and what `mi_usable_size` returns there + — while the reservation itself is unchanged. On the consumer's own + harness, pinned and ABBA-paired against the tree before the fix: see the + LEDGER entry for the numbers. `tests/alloc_core.rs::huge_usable_size_is_the_request_not_the_reservation` + pins the reported size and the bytes a move preserves. +- **The options pass made 251 allocations through the global allocator on + the first allocation of every process.** `options::ensure_init` built its + 38 x 2 environment keys with `to_uppercase` and `format!` and read them with + `std::env::var`, every one an owned `String` — inside the heap that was + still being set up, and on every short CLI run. mimalloc makes none. The + keys are now built in a stack buffer sized by the table and read through + `prim::getenv` (a raw `getenv` / `GetEnvironmentVariableA`, the calls + upstream's prim makes) into a second stack buffer, and the value grammar is + parsed in place. On the consumer's counting probe the process's start-up + went from **263 allocations to 12, which is exactly what mimalloc reads + there**. `tests/options_env.rs` proves the variables still arrive (both + prefixes, precedence, the boolean grammar, KiB scaling) from a child + process, and `rusty_alloc_api/tests/reentrancy.rs` fails the build if the + allocator ever allocates through the global allocator again while serving + a request (`docs/plans/youslowbro.md` §4). - **A cross-thread double free could hang the next collect instead of aborting.** The README says a double free aborts "on both the local and the cross-thread path". The cross-thread arm of that check lived in @@ -78,8 +108,75 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 refuses such a handle instead of installing it (OH-rusty_alloc-16, 26, 54, 58, 65, 66, 97–102). +### Performance + +- **Process start-up: the options pass walks the environment once** (Linux) + instead of calling `getenv` 76 times, each of which walked all of it — + about 25,000 fewer instructions on the first allocation of every process + (a one-line `sort` −1.6 % whole-process), more with a larger environment. + Same semantics, pinned by `tests/options_env.rs`. +- **Four more instruction-count reductions**, CURIOSITY ROUND FOUR in + `docs/LEDGER.md`: `posix_memalign` and Rust's over-aligned `alloc` carry the + aligned fast path inline (opscan `aligned` −9.2 %); a medium allocation pops + its page before collecting it (`big`/`large` −3.7 %, `mixed` −3.0 %); C++17 + aligned `new` takes the power-of-two fast path; and Rust `realloc` above + two words of alignment keeps a block in place when it fits instead of + always moving it (−27.3 % against 2.2.0 on an over-aligned buffer workload). +- **Rust programs: the `GlobalAlloc` entry points were paying for a shim + and a detour on every call** (`rusty_alloc-api`). The trait methods are now + `#[inline]`, as the `mimalloc` crate's are, so `__rust_alloc` and friends + carry the fast paths instead of jumping through the GOT to them; `dealloc` + carries the free body with its null test folded away; and layouts aligned + up to two words (16 bytes on x86-64, what every hashbrown table asks for) + are served from the ordinary size classes, which are already aligned that + far, instead of the aligned path, and can now `realloc` in place instead of + always allocating, copying and freeing. Two deterministic Rust workloads + (`bench/rust-globalalloc.sh`), measured against 2.2.0: **−21.1 %** and **−5.8 %** whole-program + instructions, for about 2–3 KB of text. `tests/natural_align.rs` pins the + alignment for every size class through alloc, alloc_zeroed and realloc. +- **Two more on the allocator core**, with the three above making up + CURIOSITY ROUND THREE in `docs/LEDGER.md`: the page-extend floor now matches upstream's + `MI_MIN_EXTEND` of four blocks (perl −454,165 whole-program, opscan + `big`/`large` −4.5 %), and span marking no longer rewrites interior slices + that already point to their span (opscan `huge` −53 %). +- **Ten deterministic instruction-count reductions** on the allocation and + free paths, each measured alone under callgrind on opscan and on real + programs; `docs/LEDGER.md` (CURIOSITY) has every number and the seven + refutations. The largest: `realloc(NULL, n)` no longer pays the moving + path's five-register frame — Lua routes every allocation through + `realloc`, and its allocator instruction count fell **39.9 %** — and + `malloc_slow` is a tail call again, which with a leaf `calloc` fast path, a + single-branch keep-one-warm test, a load-first `options::get` and a + thread-local that LLVM had hoisted above its guard takes opscan + `big`/`large` −16.5 %, `mixed` −12.9 %, `calloc` −11.9 % and `huge` −7.4 %. + No op regressed. `stats.delayed_frees` is now counted in debug builds + only, like `allocs` and `frees`. +- **Ten more, measured the same way plus Python, jq, gawk and sort** + (`docs/LEDGER.md`, CURIOSITY, ROUND TWO). Cross-thread frees are **31 % + cheaper** on opscan `xthread`: a remote free to a parked page now releases + the page to NORMAL after its one delayed-list push, as upstream mimalloc + does, so later remote frees are a single CAS onto the page's own list + instead of three CASes and a per-block drain by the owner (single-block + pages keep the delayed route; the loom model gained a case for it). + Also: `realloc` decides a move in one compare, a medium allocation whose + front page is dry grows it without re-running the heartbeat, the generic + path tail-calls page growth again, `free_general` needs no stack frame, + and `posix_memalign` / `GlobalAlloc` skip a power-of-two re-test through + the new `alloc::malloc_aligned_pow2`. perl −456,939 and Python −389,886 + whole-program instructions; opscan `big`/`large`/`mixed` read +0.7…+1.0 + on the medium-grow change against round one's intermediate state and are + still −1 % over the round. + ### Changed +- **`usable_size` (`mi_usable_size`) of a huge block is the request rounded + up to a slice, no longer the reservation.** Two behaviours follow from that + and both match upstream: a `realloc` that grows a huge block into what used + to be reported as slack now moves it instead of returning it in place, and + `expand` refuses such a growth. A shrink to at least half the request stays + in place, where before it could MOVE (a 33 MB block shrunk to 20 MB was + below half of the 64 MiB it reported). The address space reserved for a + huge block is exactly what it was. - **What was NOT taken from the Openheimer campaign, and why.** The campaign's log carries 202 findings; 139 of them are one probe — an internal `unsafe fn` handed null, `0x1`, or an address just past a diff --git a/crates/rusty_alloc/Cargo.toml b/crates/rusty_alloc/Cargo.toml index b07f5b2..9e76ed6 100644 --- a/crates/rusty_alloc/Cargo.toml +++ b/crates/rusty_alloc/Cargo.toml @@ -111,6 +111,7 @@ secure = [] [target.'cfg(windows)'.dependencies] windows-sys = { version = "0.60", features = [ "Win32_Foundation", + "Win32_System_Environment", "Win32_System_Memory", "Win32_System_SystemInformation", "Win32_System_Threading", diff --git a/crates/rusty_alloc/UNSAFE.md b/crates/rusty_alloc/UNSAFE.md index 8518a7e..b4a50f7 100644 --- a/crates/rusty_alloc/UNSAFE.md +++ b/crates/rusty_alloc/UNSAFE.md @@ -10,7 +10,7 @@ Census basis: `unsafe` occurrences in CODE per file (comments stripped; `unsafe fn` signatures and the four `unsafe impl`s included). **The baseline is now MACHINE-ENFORCED** (H-11): `tools/unsafe-census.sh` counts `unsafe` in code (comments stripped) per file against -`tools/unsafe-baseline.txt` — **928 occurrences across 23 files, 2026-09-16** +`tools/unsafe-baseline.txt` — **963 occurrences across 23 files, 2026-09-25** — and FAILS if the total grows. Growth is not forbidden, it is required to be deliberate: add the new sites here with their purpose and audit date, then re-run with `--update` in the same commit. `cargo geiger`, the registry's @@ -20,21 +20,21 @@ unsafe in DEPENDENCIES, and ours are `libc` plus bindings-only `windows-sys`. | Module | Count | What the `unsafe` is for | Last audit | |---|---:|---|---| -| `alloc.rs` | 94 | The public entry points: raw-pointer reads on the malloc fast path (must never form `&mut` on the shared empty-heap sentinel), pointer-derived metadata on the free path (`segment_of`/`page_of`), block-content copies in the realloc family. **+15 on 2026-08-22**: the free path's two `asm!` sites — the memory-destination `used--` whose flags drive the retire branch (its `label` block is a separate item and carries its own `unsafe`), and the fused `cmp {tid}, fs:0` in both `free_inline` and `free_general` — plus `malloc_or`/`malloc_or_slow`, which give `operator new` a fast path whose miss is a tail call. Each asm reads or writes exactly one field it already had a valid pointer to; none widens what the surrounding code could already touch | 2026-08-22 (free campaign; every new block reviewed at the site) | -| `heap.rs` | 71 | Owner-thread page/queue manipulation under raw pointers (no two `&mut Page` may coexist), the aligned-allocation peek, span carving. **+3 on 2026-08-19**: `try_unlink_huge_segment` split out of `remove_huge_segment`. **+15 on 2026-08-22**: `malloc_generic` split into a small entry plus `malloc_generic_walk`, with `grow_front`, `try_guarded` and `drain_delayed` as cold arms — each split adds an `unsafe fn` signature and its block while dereferencing nothing the single function did not — and the immortal `EMPTY_DELAYED` sentinel that lets the heartbeat read its list without a null test. **+2 on 2026-09-07 (P4e, §2.15 of `docs/plans/small-metal.md`)**: the reclamation fixes — one `unsafe` for the periodic `generic_collect` sweep and one for the reclaim-and-retry that runs before `malloc_generic` returns null. Both call `collect_inner`, which allocates nothing and touches only this heap’s own pages on the owner thread; each carries its SAFETY line. `malloc_generic` itself became a safe wrapper over the renamed `malloc_generic_once`, so the split added no signature. **+2 on 2026-09-08:** the medium collect-and-retry ahead of the heartbeat -- one `unsafe` around `page_collect` + `page_pop` on the bin queue front, one reading `free_is_zero` off the page just popped. Both are the SAME operations `malloc_generic_walk` performs on the same page a few lines later, on the owner thread under the heap lock: the block moves earlier, nothing new is dereferenced. Each carries its SAFETY line. **+2 on 2026-09-16 (Openheimer):** the two refuse arms in `malloc_aligned_at_slow` when `bins::aligned_at_from` reports that `block + offset + align` would wrap for a caller-chosen offset — each is one `free_local` of the block just allocated on this thread, which has not escaped | 2026-09-16 | +| `alloc.rs` | 94 | The public entry points: raw-pointer reads on the malloc fast path (must never form `&mut` on the shared empty-heap sentinel), pointer-derived metadata on the free path (`segment_of`/`page_of`), block-content copies in the realloc family. **+15 on 2026-08-22**: the free path's two `asm!` sites — the memory-destination `used--` whose flags drive the retire branch (its `label` block is a separate item and carries its own `unsafe`), and the fused `cmp {tid}, fs:0` in both `free_inline` and `free_general` — plus `malloc_or`/`malloc_or_slow`, which give `operator new` a fast path whose miss is a tail call. Each asm reads or writes exactly one field it already had a valid pointer to; none widens what the surrounding code could already touch. **+2 on 2026-09-24 (instruction-count campaign, LEDGER):** `realloc` split into an inline null test and `unsafe fn realloc_live(p: NonNull, ..)` — one signature, and the `NonNull::new_unchecked` sits inside the block that already called it, on a pointer the line above proved non-null. The body is the old one unchanged; nothing new is dereferenced. **+10 on 2026-09-24 (round two, LEDGER):** (a) `realloc_move`, the moving arm shared by the growth and the shrink — one `unsafe fn`, its body block, and the two call-site blocks: the old inline move code, now reached with the copy length each arm already holds; (b) `zalloc` in `malloc`'s raw-read shape — its fast-path block reads through `hb`, which may be the sentinel, exactly as `malloc` does (no `&mut`, no write unless a block was popped, which proves the heap is real), and the cold `unsafe fn zalloc_slow` that performs the sentinel test the fast path dropped (signature + block); (c) `free_local_owned` / `free_local_no_xheap`, `owner_heap` split so its heap-creating fallback is a tail call — two signatures and their two blocks, the same loads and the same `free_local_at` as before; (d) `malloc_aligned_pow2`, an `unsafe fn` only because its caller must have proven the alignment a power of two (a violation is not memory-unsafe — the bound test still refuses a mask at or above half a segment). Nothing new is dereferenced anywhere; each carries its SAFETY line | 2026-09-24 | +| `heap.rs` | 77 | Owner-thread page/queue manipulation under raw pointers (no two `&mut Page` may coexist), the aligned-allocation peek, span carving. **+3 on 2026-08-19**: `try_unlink_huge_segment` split out of `remove_huge_segment`. **+15 on 2026-08-22**: `malloc_generic` split into a small entry plus `malloc_generic_walk`, with `grow_front`, `try_guarded` and `drain_delayed` as cold arms — each split adds an `unsafe fn` signature and its block while dereferencing nothing the single function did not — and the immortal `EMPTY_DELAYED` sentinel that lets the heartbeat read its list without a null test. **+2 on 2026-09-07 (P4e, §2.15 of `docs/plans/small-metal.md`)**: the reclamation fixes — one `unsafe` for the periodic `generic_collect` sweep and one for the reclaim-and-retry that runs before `malloc_generic` returns null. Both call `collect_inner`, which allocates nothing and touches only this heap’s own pages on the owner thread; each carries its SAFETY line. `malloc_generic` itself became a safe wrapper over the renamed `malloc_generic_once`, so the split added no signature. **+2 on 2026-09-08:** the medium collect-and-retry ahead of the heartbeat -- one `unsafe` around `page_collect` + `page_pop` on the bin queue front, one reading `free_is_zero` off the page just popped. Both are the SAME operations `malloc_generic_walk` performs on the same page a few lines later, on the owner thread under the heap lock: the block moves earlier, nothing new is dereferenced. Each carries its SAFETY line. **+2 on 2026-09-16 (Openheimer):** the two refuse arms in `malloc_aligned_at_slow` when `bins::aligned_at_from` reports that `block + offset + align` would wrap for a caller-chosen offset — each is one `free_local` of the block just allocated on this thread, which has not escaped. **+1 on 2026-09-24 (instruction-count campaign, LEDGER):** the medium collect-and-retry's `page_collect` + `page_pop` block became two blocks so the `saw_remote_free` latch is stored between them — same two calls on the same page, owner thread, in the same order; nothing new is dereferenced. **+1 on 2026-09-24 (round two):** a medium miss grows the front page directly — one block around the `capacity < reserved` test and the `grow_front` call, on the same page the block above just collected, which is `grow_front`'s whole contract **+2 on 2026-09-25 (CURIOSITY ROUND FOUR):** the medium arm pops the front page BEFORE collecting it — one `page_pop` block and the `free_is_zero` read beside it, the same two operations the collect-then-pop below already performs on the same owner-thread page | 2026-09-25 | | `init.rs` | 42 | Thread/heap lifecycle: the initial-exec TLS slot (`global_asm!` + fs-relative asm reads), thread-pointer register reads (`fs:0`/`gs:0x30`/`tpidrro_el0`), heap-box creation/teardown, the abandonment path run inside platform TLS destructors. **+1 on 2026-09-09 (`firmware-what-is-left.md` §7.3): an ATTRIBUTE, not an operation** — `unsafe(link_section = ".rodata.…")` on `EMPTY_HEAP_BOX`, bare metal only, so the never-written sentinel lives in flash instead of costing 1,752 bytes of RAM. The token is counted because the census is a text search; the contract it rests on (raw reads only, no page ever stores its address) is the one the `Sync` impl below already carries, and a write would now fault against the flash cache instead of silently landing. `create_heap`'s template copy from it is a `ptr::read` inside the block that already existed. **+5 on 2026-09-16 (Openheimer, OH-rusty_alloc-13):** thread exit now abandons EVERY heap the dying thread still owns, not only the one in `done_slot` — a first-class heap that was never `heap_delete`d, or one installed with `set_default_heap`, used to stay DELAYED under a dead owner forever. `take_heap_owned_by` unlinks one such box under `HEAPS_LOCK` (one block: the registry walk); `thread_done` reads `owner_tid` off its own box and calls the split-out `unsafe fn thread_done_one` for the primary box and for each extra one (three sites plus the signature). Nothing new is dereferenced that the single function did not; each carries its SAFETY line | 2026-09-16 | | `segment.rs` | 35 | Segment/page metadata addressing: the mask trick (`segment_of`), `page_of`'s contract-based indexing — **the bound is now PROVED for every in-segment offset by `proofs.rs` (Kani), not merely asserted** — span tiling, purge/recommit | 2026-08-19 | -| `prim/windows.rs` | 33 | OS FFI: VirtualAlloc family, FLS destructors, QPC, BCryptGenRandom. **+3 on 2026-09-16 (Openheimer):** `range_is_reserved` (`zeroed` `MEMORY_BASIC_INFORMATION` + `VirtualQuery`, the same OS answer `unix.rs` gets from `mincore`), and `alloc_aligned` now refuses a garbage alignment and a `size + align` that wraps BEFORE it reserves anything, and releases a re-reservation that the OS placed anywhere but the aligned address it asked for instead of returning it misaligned (`VirtualFree`, one block) | 2026-09-16 | +| `prim/windows.rs` | 34 | OS FFI: VirtualAlloc family, FLS destructors, QPC, BCryptGenRandom. **+3 on 2026-09-16 (Openheimer):** `range_is_reserved` (`zeroed` `MEMORY_BASIC_INFORMATION` + `VirtualQuery`, the same OS answer `unix.rs` gets from `mincore`), and `alloc_aligned` now refuses a garbage alignment and a `size + align` that wraps BEFORE it reserves anything, and releases a re-reservation that the OS placed anywhere but the aligned address it asked for instead of returning it misaligned (`VirtualFree`, one block). **+1 on 2026-09-24 (`docs/plans/youslowbro.md` §4):** `getenv` — one `GetEnvironmentVariableA` into a caller buffer of exactly the length passed, the allocation-free environment read that replaced `std::env::var`'s owned strings in the options pass (251 allocations through the global allocator on every process's first allocation, now none) | 2026-09-24 | | `page.rs` | 38 | Free-list links written into dead blocks, the lock-free `xthread_free` four-state protocol (loom-modeled in `tests/loom_xthread.rs`), the immortal `EMPTY_PAGE` sentinel — **+1 on 2026-09-09: the same `unsafe(link_section)` attribute as `init.rs`, placing it in flash on bare metal; its free list is permanently null, so nothing writes it**. **+8 on 2026-08-22**: `page_link_local` split out of `page_push_local` so the caller can decrement `used` in one memory-destination RMW, `page_collect_impl` const-generic over whether it also writes the protocol flag, and `USED_OFFSET` — whose value is asserted against `offset_of!(Page, used)` by a unit test, because an asm operand is not type-checked and a field reordering would silently decrement the wrong bytes | 2026-08-22 (free campaign) | -| `prim/unix.rs` | 28 | OS FFI: mmap family, madvise/decommit, pthread keys, /dev/urandom. **+1 on 2026-09-16 (Openheimer, OH-201):** `range_is_reserved` asks `mincore` whether a caller-supplied `manage_os_memory` range is mapped at all before it becomes an arena — one FFI call on a page-aligned probe | 2026-09-16 | +| `prim/unix.rs` | 31 | OS FFI: mmap family, madvise/decommit, pthread keys, /dev/urandom. **+1 on 2026-09-16 (Openheimer, OH-201):** `range_is_reserved` asks `mincore` whether a caller-supplied `manage_os_memory` range is mapped at all before it becomes an arena — one FFI call on a page-aligned probe. **+1 on 2026-09-24 (`docs/plans/youslowbro.md` §4):** `getenv` — `libc::getenv` and a byte-by-byte read of the returned C string up to its NUL or the caller buffer's end, nothing written through it; the same call upstream's prim makes, with the same standing caveat about a concurrent `setenv` **+2 on 2026-09-25:** `env_for_each` — an `unsafe extern "C"` declaration of `environ` and one block that walks it to its NULL terminator, handing each NUL-terminated entry on as a pointer; read only, with the same concurrent-`setenv` caveat as `getenv` | 2026-09-25 | | `prim/mod.rs` | 18 | The `TlsSlot` abstraction (`unsafe impl Send/Sync`, justified at the impls), dispatch to the platform backends | 2026-08-08 | -| `rusty_alloc_api/src/lib.rs` | 14 | The `GlobalAlloc`/`Allocator` impls forwarding layouts to the core; `unsafe` is inherent to those traits' contracts | 2026-08-19 | +| `rusty_alloc_api/src/lib.rs` | 16 | The `GlobalAlloc`/`Allocator` impls forwarding layouts to the core; `unsafe` is inherent to those traits' contracts . **+1 on 2026-09-24:** `GlobalAlloc::alloc`'s over-aligned arm calls `alloc::malloc_aligned_pow2`, an `unsafe fn` whose one added precondition — a power-of-two alignment — `Layout` guarantees **+1 on 2026-09-25:** the over-aligned `realloc` arm reads `usable_size(ptr)` to keep a block in place — `ptr` is a live non-null block of ours by the `GlobalAlloc` contract | 2026-09-25 | | `os.rs` | 12 | The prim-layer wrapper: commit/decommit/protect plumbing | 2026-08-08 | | `prim/mock.rs` | 8 | Miri-only mock OS backend (never shipped; `cfg(miri)`) | 2026-08-06 | | `arena.rs` | 15 | Lock-free chunk bitmap claim/verify, recycled-chunk scrubbing (the 0.1.0-alpha.2 UAF fix lives here: `wait_no_remote_in_flight` on every recycle path). The baseline had drifted to 13 without a row edit; recorded now. **+2 on 2026-09-16 (Openheimer, `docs/plans/openheimer-run.md`):** `arena_register` claims its slot with a CAS instead of a load-then-store, and the two new blocks are the REFUSE arm — `os::free` of the descriptor and of the OS mapping when the table is full or the slot was lost — so a refused arena leaks nothing. Both free memory this function allocated a few lines earlier and nothing else has seen; each carries its SAFETY line | 2026-09-16 | | `prim/fixed.rs` | 37 | **New 2026-09-07 (P1 of `docs/plans/small-metal.md`).** The fixed-region backend for a target with no OS: memory is a `&'static mut [u8]` handed over once. **6 of the 18 are `unsafe fn` signatures the prim seam requires** (`alloc`/`free`/`commit`/`decommit`/`reset`/`protect`) whose *bodies contain no unsafe operation at all* — the free list is `AtomicUsize` arrays under a spin lock and the pointers are built with the safe `with_exposed_provenance_mut`, so this backend adds **zero** unsafe dereferences to the shipped crate. The other 12 are in `#[cfg(test)]`: two `&raw mut` static-region handoffs and ten calls through the `unsafe fn` seam, each with its SAFETY line. **+1 on 2026-09-07 (P2):** the region test became two-sided — where the shipped geometry refuses a segment-sized request from a 512 KiB region, the small profile SERVES one, so the test now frees it too. Audited at the site; the module is `allow(dead_code)` and unreachable on every platform that has an arm above it. **+7 on 2026-09-07 (P4b, §2.9/§2.10 of the same plan):** the two-ended `place` rule added ZERO unsafe to shipped code — `place` is a pure arithmetic `fn` and the scan around it is unchanged — and all seven are in `#[cfg(test)]`: one `slice::from_raw_parts_mut` carving the `REGION_ALIGN`-aligned window out of `BACKING` (replacing a `&mut *ptr` that a `repr(align(65536))` static would have needed, which rustc 1.97.1 on MSVC cannot compile), one `ptr::add` to reach that window, and five calls through the `unsafe fn` seam in the placement assertions and in `greedy_segments`, which allocates segments until refusal and frees every one before returning. Each carries its SAFETY line **+1 on 2026-09-09 (morning, `#21`):** the two-sided alignment test frees the SEGMENT_SIZE-aligned page it is served when the region straddles a boundary — `#[cfg(test)]`, banked without a row here at the time; recorded now. **+5 on 2026-09-09 (`docs/plans/finished/region-alignment-bug.md` §7): the module ships its first unsafe OPERATIONS.** (1) `unsafe impl Sync for Region` and (2) the `&mut *self.bytes.get()` in `Region::give` — the once-only handoff the Janus seam has carried since 2.0.0, moved here so the alignment travels with it; a module-wide `REGION_GIVEN` swap precedes the `&mut`, so a second `give` on any instance is refused before it could alias. (3) `unsafe impl Sync for FirstHeapBox`, the bare-metal static holding the first heap's descriptor (handed out once by `take_first_heap_box`, on the one thread such a build has). The other two are `#[cfg(test)]`: a `from_raw_parts_mut` carving the report's misaligned base out of a static for the refusal probe, and a `Box::new_zeroed().assume_init()` allocating a `Region` in place, because materialising a 64 KiB-aligned value on the stack first faults on Windows. Each carries its SAFETY line. **+2 on 2026-09-09 (`docs/plans/finished/region-alignment-dissolve.md`): zero new unsafe in shipped code** — segments now stride from the region's base, and every piece of that is safe: `stride_base` is a relaxed load, `place` takes an `origin` and does arithmetic, `install_region` aligns the base up. The two are `#[cfg(test)]`, the `--cfg ra_aligned_region` arm of the two-sided alignment test (an `alloc` and a `free` through the `unsafe fn` seam: the 2.0.4 straddle case, kept under the knob that restores the mask), and the `Box::new_zeroed().assume_init()` moved under that same cfg — the default arm is a plain `static Region` now, which a 16-byte-aligned type can be on every host toolchain. Each carries its SAFETY line. **+4 on 2026-09-09 (`docs/plans/finished/esp32-large-alloc-ceiling.md`): all four are `#[cfg(test)]`, and shipped code gained none.** The large-allocation ceiling reported from `rusty_zstd` is reproduced against the real extent allocator by `greedy_dedicated`, which serves `huge_alloc`-shaped reservations until the region refuses and frees every one before returning (two calls through the `unsafe fn` seam), plus one `alloc`/`free` pair modelling the whole segment a first small allocation claims. The fix itself -- the `ra_segment_size` geometry knob, `dedicated_segments`, `region_for_allocs` and `region_capacity` -- is arithmetic over the existing atomics and adds no unsafe operation at all. Each carries its SAFETY line | 2026-09-09 | | `prim/wasm.rs` | 6 | `memory.grow` linear-memory backend | 2026-08-06 | -| `options.rs` | 6 | Env parsing at init, registered-hook invocation | 2026-08-06 | +| `options.rs` | 17 | Env parsing at init, registered-hook invocation. **+1 on 2026-09-24: `#[cfg(test)]` only** — `std::env::set_var` (an `unsafe fn` in edition 2024) seeding two probe variables for the in-place lookup test. Shipped code gained none: the pass itself went from 251 allocations to zero by building its keys in a stack buffer and reading through `prim::getenv` **+10 on 2026-09-25:** the one-walk environment pass (Linux) — three `unsafe fn` helpers (`strip`, `match_name`, `copy_value`) that read a NUL-terminated entry byte by byte and stop at the first mismatch or its NUL, and the blocks that call them; nothing is written through an entry | 2026-09-25 | | `stats.rs` | 3 | Volatile whole-struct snapshot of racy-by-design counters | 2026-08-06 | | `random.rs` | 2 | OS entropy seeding via the prim layer | 2026-08-08 | | `lib.rs` | 2 | **New 2026-09-07 (P3).** One `unsafe impl Sync for SingleThreadCell` — the `no_std` half of `ra_thread_local!`. With `std` the macro is `std::thread_local!` verbatim and this type does not exist; without it, a "thread-local" is a plain `static`, sound because the crate serves `no_std` only on single-threaded targets (the same standing assumption as `prim::fixed`: constant thread id, TLS destructors that never fire, a spin lock that never contends). A `no_std` build on a target that grows threads must revisit this type first — the SAFETY comment says so at the impl. **+1 on 2026-09-07 (P5): PROSE, not code.** The census is a text search, and the `compile_error!` that now forces `--cfg ra_single_threaded` on a `no_std` build names `unsafe impl Sync` in its message so the person who hits it knows what they are opting into. Counted, and left counted rather than reworded: the ratchet is allowed to be conservative, and a message that names the thing is worth one line of baseline | 2026-09-07 (P5) | @@ -50,7 +50,7 @@ comment, and all four are in the H-11 ratchet's baseline. | Module | Count | What the `unsafe` is for | Last audit | |---|---:|---|---| -| `rusty_alloc_ffi/src/lib.rs` | 350 | **The workspace's untrusted boundary**: 157 `extern "C"` entry points receiving caller pointers and sizes. Every out-parameter writer null-guards; `mi_posix_memalign` validates alignment (power of two, ≥ `sizeof(void*)`) and returns EINVAL/ENOMEM; all 8 `count × size` sites use `checked_mul`. Panics cannot unwind into C (edition 2024 aborts on `extern "C"` unwind; release is `panic = "abort"`). **+2 on 2026-08-22**: `new_impl` routes through `alloc::malloc_or` so the OOM arm is a tail call rather than a null test that keeps `size` live. **+6 on 2026-09-16 (Openheimer, `docs/plans/openheimer-run.md`):** `heap_ptr` — one `unsafe fn` plus its deref — is now the ONE place a C heap handle is turned into a `Heap`, after a null / misaligned / immortal-empty-box check (shape only: what a well-formed handle points at is the caller's, as for `free`); and `mi_dupenv_s` / `mi_wdupenv_s` write null and 0 through the caller's out-pointers on the EINVAL path (four writes) instead of leaving them untouched, as the CRT contract says. Every out-parameter write in the crate goes through `out_mut`, which refuses null and misaligned pointers; the 64 KiB address floor it briefly carried is gone, because it would have refused legitimate out-pointers on wasm32, where static data starts at 1 KiB | 2026-09-16 | +| `rusty_alloc_ffi/src/lib.rs` | 358 | **The workspace's untrusted boundary**: 157 `extern "C"` entry points receiving caller pointers and sizes. Every out-parameter writer null-guards; `mi_posix_memalign` validates alignment (power of two, ≥ `sizeof(void*)`) and returns EINVAL/ENOMEM; all 8 `count × size` sites use `checked_mul`. Panics cannot unwind into C (edition 2024 aborts on `extern "C"` unwind; release is `panic = "abort"`). **+2 on 2026-08-22**: `new_impl` routes through `alloc::malloc_or` so the OOM arm is a tail call rather than a null test that keeps `size` live. **+6 on 2026-09-16 (Openheimer, `docs/plans/openheimer-run.md`):** `heap_ptr` — one `unsafe fn` plus its deref — is now the ONE place a C heap handle is turned into a `Heap`, after a null / misaligned / immortal-empty-box check (shape only: what a well-formed handle points at is the caller's, as for `free`); and `mi_dupenv_s` / `mi_wdupenv_s` write null and 0 through the caller's out-pointers on the EINVAL path (four writes) instead of leaving them untouched, as the CRT contract says. Every out-parameter write in the crate goes through `out_mut`, which refuses null and misaligned pointers; the 64 KiB address floor it briefly carried is gone, because it would have refused legitimate out-pointers on wasm32, where static data starts at 1 KiB . **+1 on 2026-09-24:** `posix_memalign` calls `alloc::malloc_aligned_pow2` after its own EINVAL test has proven the alignment a power of two, so the core skips re-testing it **+1 on 2026-09-25:** C++ aligned `new` calls `malloc_aligned_pow2` after testing `align & (align - 1) == 0` itself; an invalid alignment keeps the old checked route | 2026-09-25 | | `rusty_alloc_override/src/lib.rs` | 51 | `malloc`/`free`/`operator new` interposition for `LD_PRELOAD` — thin forwarding to `alloc::*`, plus the `free_inline` export that carries the fast-path body. No state of its own. **+2 on 2026-08-22**: the sized-delete exports follow the same `free_inline` shape as the unsized ones | 2026-08-22 | | `rusty_alloc_bench/src/*.rs` | 29 | Tier-B kernels and the `.ratrace` replayer. The trace parser's `unwrap`s are infallible by TYPE (`Record::decode` takes `&[u8; RECORD_SIZE]`), and the one genuinely-invalid field returns `InvalidData` | 2026-08-20 (first audit) | | `rusty_alloc_wasm/src/lib.rs` | 21 | The `ra_selftest` cdylib fixture: raw block writes and read-back checks that prove the wasm build actually allocates | 2026-08-20 (first audit) | diff --git a/crates/rusty_alloc/src/alloc.rs b/crates/rusty_alloc/src/alloc.rs index 106a871..65d42d7 100644 --- a/crates/rusty_alloc/src/alloc.rs +++ b/crates/rusty_alloc/src/alloc.rs @@ -68,6 +68,41 @@ unsafe fn unalign(pg: *mut Page, p: *mut u8) -> *mut u8 { } } +/// `free_local_at` on the owning heap, for [`free_general`]'s local arm. +/// +/// [`owner_heap`] inlined there carried its fallback — `my_heap()`, which can +/// CREATE the heap — as a non-tail call, and that one call gave +/// `free_general` a three-register frame paid by every general free, +/// including the cross-thread frees it hands to `remote_free` and never +/// reach this arm (callgrind, opscan `xthread`). Here the common case is a +/// tail call and the fallback is its own cold function. +/// +/// # Safety +/// As [`owner_heap`] and `Heap::free_local_at`. +#[inline(always)] +unsafe fn free_local_owned(seg: *mut Segment, pg: *mut Page, block: *mut u8) { + // SAFETY: forwarded contract. + unsafe { + let xh = (*pg).xheap.load(core::sync::atomic::Ordering::Acquire); + if xh == 0 { + return free_local_no_xheap(seg, pg, block); + } + (*(*init::box_of_xheap(xh)).heap.get()).free_local_at(seg, pg, block); + } +} + +/// [`owner_heap`]'s fallback arm, out of line so its call stays out of +/// `free_general`'s frame. +/// +/// # Safety +/// As [`free_local_owned`]. +#[cold] +#[inline(never)] +unsafe fn free_local_no_xheap(seg: *mut Segment, pg: *mut Page, block: *mut u8) { + // SAFETY: forwarded contract. + unsafe { (*owner_heap(pg)).free_local_at(seg, pg, block) }; +} + /// The heap that OWNS `pg` — recovered from the page's `xheap` back-pointer /// (container-of over the `HeapBox`'s offset-0 delayed list), so it is right /// even when the thread holds several first-class heaps. @@ -355,13 +390,52 @@ pub fn zalloc(size: usize) -> *mut u8 { // sentinel, and so carries the once-per-thread initialisation and a null // check for its failure on a path taken once per ALLOCATION. `rptest` // calls calloc 43,449 times and paid it on every one. + // + // And no compare against the sentinel either, now: the fast path is + // `malloc`'s — RAW reads only, so `hb` may be the sentinel, whose direct + // table of empty pages always misses — and the sentinel test moved into + // the cold miss. It was a `cmp; je` on every `calloc` (Python: 60,336 of + // them in the real-program gate) for a case that a miss already routes. + // The zeroing keeps the popped page in hand, as `Heap::zalloc` does + // (opps.md #5). let hb = init::heap_box_fast(); + // SAFETY: raw reads only, exactly as `malloc`: `direct` entries point at a + // live page of this thread's heap or at the immortal empty page, and + // `page_pop` on that returns null before its first store. A non-null + // block means `hb` is this thread's real heap. + unsafe { + if size <= SMALL_SIZE_MAX { + let w = crate::types::wsize_from_size(size); + let h = (*hb).heap.get(); + let p = (*h).direct[w]; + let b = crate::page::page_pop(p); + if !b.is_null() { + #[cfg(debug_assertions)] + { + (*h).stats.allocs += 1; + } + if (*p).free_is_zero { + b.cast::().write(0); + } else { + ptr::write_bytes(b, 0, (*p).block_size); + } + return b; + } + } + zalloc_slow(hb, size) + } +} + +/// [`zalloc`]'s miss: the sentinel test, then the heap's own zalloc. +/// +/// # Safety +/// `hb` is `heap_box_fast()` of the calling thread. +#[cold] +#[inline(never)] +unsafe fn zalloc_slow(hb: *mut init::HeapBox, size: usize) -> *mut u8 { if hb == init::empty_heap_box_ptr() { return zalloc_first(size); } - // `Heap::zalloc` zeroes with the popped page in hand, avoiding the - // `usable_size` re-resolution the old `malloc` + `zero_block` pair paid on - // every recycled block (opps.md #5). // SAFETY: a non-sentinel box is this thread's live, initialised box. unsafe { (*(*hb).heap.get()).zalloc(size) } } @@ -436,6 +510,29 @@ pub fn malloc_aligned(size: usize, align: usize) -> *mut u8 { // reverted. The export gains a frame worth more than the check it removes, // the same result the `realloc_inline` twin produced. pub fn malloc_aligned_at(size: usize, align: usize, offset: usize) -> *mut u8 { + aligned_at_impl::(size, align, offset) +} + +/// [`malloc_aligned`] for a caller that has ALREADY established that `align` +/// is a power of two: the C shims that validate it for EINVAL +/// (`posix_memalign`), and Rust's `GlobalAlloc`, whose `Layout` guarantees +/// it. The public entry re-tested it on every call — a `test; jne` after the +/// mask it needs anyway (callgrind, opscan `aligned`). +/// +/// # Safety +/// `align` must be a power of two. (A violation is not memory-unsafe here — +/// the bound test still refuses a mask at or above half a segment — but the +/// block would not be aligned as asked.) +#[inline] +pub unsafe fn malloc_aligned_pow2(size: usize, align: usize) -> *mut u8 { + debug_assert!(align.is_power_of_two(), "malloc_aligned_pow2: {align}"); + aligned_at_impl::(size, align, 0) +} + +/// The aligned fast path; `POW2` skips the power-of-two test for callers that +/// have proven it. +#[inline(always)] +fn aligned_at_impl(size: usize, align: usize, offset: usize) -> *mut u8 { let hb = init::heap_box_fast(); // SAFETY: raw reads only, exactly as `malloc` does — `hb` may be the // shared immortal sentinel, which must never see a `&mut` or a write. Its @@ -454,7 +551,7 @@ pub fn malloc_aligned_at(size: usize, align: usize, offset: usize) -> *mut u8 { if offset == 0 && size <= SMALL_SIZE_MAX && mask < crate::types::SEGMENT_SIZE / 2 - && align & mask == 0 + && (POW2 || align & mask == 0) { let h = (*hb).heap.get(); let w = crate::types::wsize_from_size(size); @@ -990,10 +1087,17 @@ pub unsafe fn free_inline(p: *mut u8) { // `unsafe`; it is a separate item. // SAFETY: `pg` is the live page this free resolved, // owned by this thread. + // + // ONE test, not three: `used`, `next` and `prev` + // OR-ed are zero exactly when the page emptied + // (not a wrapped double free) and is its queue's + // only member. Three compare-and-branch pairs + // were 6 Ir on every free that empties a page — + // every free of an alloc/free loop cycling one + // block. `retire_or_abort` re-tests each case. unsafe { - if (*pg).used == 0 - && (*pg).next.is_null() - && (*pg).prev.is_null() + if ((*pg).used as usize | (*pg).next.addr() | (*pg).prev.addr()) + == 0 { return; } @@ -1008,7 +1112,8 @@ pub unsafe fn free_inline(p: *mut u8) { let u = (*pg).used.wrapping_sub(1); (*pg).used = u; if (u as i32) <= 0 { - if u == 0 && (*pg).next.is_null() && (*pg).prev.is_null() { + // One test, as in the x86-64 arm above. + if (u as usize | (*pg).next.addr() | (*pg).prev.addr()) == 0 { return; } return retire_or_abort(pg); @@ -1169,12 +1274,12 @@ unsafe fn free_general(p: *mut u8, seg: *mut Segment, pg: *mut Page, owner_tid: options(nostack, readonly), ); // Fell through: this thread owns the page. - (*owner_heap(pg)).free_local_at(seg, pg, block); + free_local_owned(seg, pg, block); } #[cfg(not(all(target_arch = "x86_64", target_os = "linux", not(miri))))] if crate::ONE_THREAD || owner_tid == init::thread_id() { // Hand the already-resolved segment through (M9 brick #2). - (*owner_heap(pg)).free_local_at(seg, pg, block); + free_local_owned(seg, pg, block); } else { // Remote: the loom-modeled protocol (huge pages sit DELAYED, so // this lands on the owner's delayed list and the owner's @@ -1258,16 +1363,57 @@ unsafe fn usable_size_slow(pg: *mut Page, p: *const u8, flags: u8) -> usize { /// /// # Safety /// `p` must be null or a live pointer from this allocator; invalidated on move. +/// +/// **The null test is in front of the frame, and that is the whole reason +/// this is two functions.** In one body, LLVM placed the move arm's five +/// callee-saved pushes BEFORE `test %rdi,%rdi`, so `realloc(NULL, n)` paid ten +/// frame instructions to reach a tail call into `malloc`. That is not a corner +/// case: Lua's allocator routes every allocation through `realloc`, and in the +/// `lua` real-program instrument **480,342 of 540,393 calls had a null +/// pointer** (callgrind, per instruction). Here the null arm inlines `malloc`'s +/// fast path with no frame, and a live pointer costs one `jmp` into +/// [`realloc_live`] with its arguments already in place. +#[inline] pub unsafe fn realloc(p: *mut u8, newsize: usize) -> *mut u8 { if p.is_null() { return malloc(newsize); } + // SAFETY: forwarded contract; `p` is non-null here. + unsafe { realloc_live(ptr::NonNull::new_unchecked(p), newsize) } +} + +/// [`realloc`] of a non-null pointer. +/// +/// `NonNull`, not `*mut u8`, and it is measured: split out with a raw pointer, +/// LLVM no longer knew `p` was non-null in here, and the null tests inside the +/// inlined `usable_size` and `free_inline` came back — two `test; je` pairs +/// per moving realloc, opscan `realloc` +11.84 Ir/op. The `nonnull` parameter +/// attribute folds both away again. +/// +/// # Safety +/// `p` must be a live pointer from this allocator; invalidated on move. +#[inline(never)] +unsafe fn realloc_live(p: ptr::NonNull, newsize: usize) -> *mut u8 { + let p = p.as_ptr(); // SAFETY: p live per contract. let usable = unsafe { usable_size(p) }; // NOTE: rewriting this as the one-compare unsigned range check // `newsize.wrapping_sub(usable >> 1) <= usable - (usable >> 1)` measured // FLAT — LLVM already emits that shape from the readable form. - if newsize <= usable && newsize >= usable / 2 { + // + // GROWTH first, on its own move path. A live realloc from a real program + // almost always MOVES — lua's grow, and Python's shrink below half + // (60,100 of its 60,669 took the second arm below) — and the two-sided + // test compiled branch-free — `setbe`, `shr`, `setae`, `test`, `je` — so + // every move paid all of it (callgrind, per instruction). As two + // branches each move arm decides in one compare, and each knows its copy + // length outright (`usable` for a growth, `newsize` for a shrink), so the + // `min` goes too. + if newsize > usable { + // SAFETY: forwarded contract; a growth copies the whole old block. + return unsafe { realloc_move(p, newsize, usable) }; + } + if newsize >= usable / 2 { // Keep in place. NOTE: this counter costs a TLS heap lookup (a call // into ld.so's __tls_get_addr in a cdylib) on the most common realloc // outcome, so removing it was tried as a brick -- and measured FLAT on @@ -1283,13 +1429,26 @@ pub unsafe fn realloc(p: *mut u8, newsize: usize) -> *mut u8 { // **+12.00 Ir/op** and was reverted. The argument setup plus the call and // return are paid on every MOVING realloc, and the scan is all moves; the // frame it saved was smaller than the call it added. + // A shrink below half: the copy is the new, smaller size. + // SAFETY: forwarded contract. + unsafe { realloc_move(p, newsize, newsize) } +} + +/// The moving arm of [`realloc_live`]: allocate `newsize`, copy `copy` bytes, +/// free `p`. Inlined into both of its call sites, so each copy length is a +/// value the site already holds. +/// +/// # Safety +/// `p` live from this allocator; `copy <= min(usable_size(p), newsize)`. +#[inline(always)] +unsafe fn realloc_move(p: *mut u8, newsize: usize, copy: usize) -> *mut u8 { let np = malloc(newsize); if np.is_null() { return ptr::null_mut(); } // SAFETY: both live and disjoint; prefix preserved then p consumed. unsafe { - core::ptr::copy_nonoverlapping(p, np, usable.min(newsize)); + core::ptr::copy_nonoverlapping(p, np, copy); // `free_inline`, not the outlined `free`: this function has ALREADY // masked `p` to its segment for `usable_size`, and inlining the free // here lets LLVM common-subexpression that away instead of masking @@ -1297,6 +1456,15 @@ pub unsafe fn realloc(p: *mut u8, newsize: usize) -> *mut u8 { // pass the segment explicitly was tried first and cost +1 Ir on EVERY // free (batch_lifo 60.00 -> 61.00); letting the inliner find it costs // other callers nothing because only this one opts in. + // + // REFUTED (2026-09-24): resolving this free's page, flags and owner + // BEFORE the copy, so the opaque `memcpy` would not force LLVM to + // re-derive them after it (five instructions per move). `free` stayed + // byte-identical, but LLVM re-loaded the page index before the copy + // anyway (the `malloc` above may write memory), and holding four + // values across the call took a sixth callee-saved register plus a + // stack adjustment: opscan `realloc` +8.00 (+4 per move), lua + // +240,221 and python +302,857 allocator Ir. free_inline(p); stat_realloc(false); } diff --git a/crates/rusty_alloc/src/bins.rs b/crates/rusty_alloc/src/bins.rs index 8b360f2..2ac7faa 100644 --- a/crates/rusty_alloc/src/bins.rs +++ b/crates/rusty_alloc/src/bins.rs @@ -32,7 +32,16 @@ pub fn bin(size: usize) -> usize { } else { // Four bins per power of two: index by the top bit and the next two. let w = wsize - 1; - let b = (usize::BITS - 1 - w.leading_zeros()) as usize; // bsr(w) + // bsr(w). + // + // NOTE (2026-09-24, REFUTED): `(w | 1).leading_zeros()`, to let LLVM + // see a non-zero input and drop the zero-input fallback (`mov $0x7f` + // before the `bsr`) on the generic path's medium trip, removed the + // `mov` and added a register copy and the `or`; LLVM keeps the + // `leading_zeros` form either way (`bsr; xor $0x3f`) because the shift + // count and the bin index are both derived from it. Opscan `big` + // +1.00, `aligned` +0.69, `mixed` +0.72. Leave it as written. + let b = (usize::BITS - 1 - w.leading_zeros()) as usize; ((b << 2) + ((w >> (b - 2)) & 0x03)) - 3 } } diff --git a/crates/rusty_alloc/src/heap.rs b/crates/rusty_alloc/src/heap.rs index d735615..a6927f5 100644 --- a/crates/rusty_alloc/src/heap.rs +++ b/crates/rusty_alloc/src/heap.rs @@ -4,9 +4,17 @@ //! own heap, `free` routes by the segment's owner id, and cross-thread frees go //! through the loom-modeled 4-state protocol in `page.rs`. +use core::cell::Cell; use core::ptr; use core::sync::atomic::Ordering; +ra_thread_local! { + /// Set while `Heap::malloc_generic_retry` runs its one reclaim-and-retry, + /// so a second null from the nested pass ends the attempt. Read only on + /// the OOM path; see `Heap::malloc_generic` for why it is not a parameter. + static IN_OOM_RETRY: Cell = const { Cell::new(false) }; +} + use crate::bins::{self, BIN_COUNT, PAGES_DIRECT}; use crate::page::{ Block, DelayedList, Page, XFLAG_DELAYED, XFLAG_NORMAL, block_next, page_all_free, page_collect, @@ -393,6 +401,19 @@ impl Heap { return b; } } + self.zalloc_generic(size) + } + + /// `zalloc`'s miss, out of line and in TAIL position. + /// + /// Inline, this arm's call to `malloc_generic` had to return here to zero + /// the block, so `size` was live across it and every `calloc` — hit or + /// miss — paid a frame for it: `push`/`push` on entry and `add`/`pop`/ + /// `pop` before the fast path's `jmp memset` (callgrind, opscan `calloc`, + /// hit rate 94 %). Out here the fast path is a leaf. + #[cold] + #[inline(never)] + fn zalloc_generic(&mut self, size: usize) -> *mut u8 { // Slow/large path: rare, and the generic allocator has resolved the // page anyway. Recover the usable size the general way. let (b, is_zero) = self.malloc_generic(size); @@ -434,20 +455,84 @@ impl Heap { /// only ~649 generic trips in total, far under the 10,000 default, so the /// timer never fires. This trigger is failure, not a clock. /// - /// **Costs nothing on the happy path**: it runs only when the allocation - /// was about to fail. `collect_inner(true, true)` is `mi_collect(true)` — - /// reclaim orphans too, since this is the last resort before null. + /// `collect_inner(true, true)` is `mi_collect(true)` — reclaim orphans + /// too, since this is the last resort before null. + /// + /// **Where the retry test lives is a cost decision.** It first sat here as + /// `let r = once(size); if r.0.is_null() { collect; once(size) }`, which + /// read as free on the happy path and was not: a test AFTER the call means + /// the call is no longer in tail position, so `alloc::malloc_slow` — whose + /// own doc says "this arm must stay a TAIL call" — grew a frame, a spill of + /// `size` and a `test; je` again, 16 Ir around a 1-instruction jump on + /// every slow-path allocation (callgrind, opscan `big`). Now the test is in + /// `malloc_generic_once`'s epilogue, where the result is already in a + /// register, and this wrapper is a tail call. + /// + /// And "only once" is a THREAD-LOCAL, not a parameter. As a `retry: bool` + /// argument it cost every generic trip a `mov $1` at the call, a copy into + /// a callee-saved register, and a flag test ahead of the null test in the + /// epilogue — about four instructions to decide something only a null + /// result ever asks. Now the epilogue tests the result alone, and the flag + /// is read on the OOM path only. + /// + /// Inside the chain the zero flag is a `u8`, and it becomes a `bool` only + /// here. A `(*mut u8, bool)` pair return makes the caller re-truncate the + /// flag (`and $0x1,%dl`) after every call that returns one, which is what + /// kept `malloc_generic_once` from TAIL-calling `grow_front` — the + /// commonest generic outcome on a real program (jq: 10,307 of 11,229 + /// trips) — and the walk. Callers that want only the pointer, like + /// `alloc::malloc_slow`, drop the conversion entirely. + #[inline] pub(crate) fn malloc_generic(&mut self, size: usize) -> (*mut u8, bool) { - let r = self.malloc_generic_once(size); - if !r.0.is_null() { - return r; + let (p, z) = self.malloc_generic_once(size); + (p, z != 0) + } + + /// One pass of the generic path; on null, one reclaim-and-retry through + /// [`Heap::malloc_generic_retry`]. + /// + /// NOTE (2026-09-24, REFUTED on real programs): splitting the medium + /// collect-and-retry into a frameless entry that tail-calls the rest was + /// worth **-11.00 Ir/op** on opscan `big`/`large` and -6.83 on `mixed` — + /// the arm was paying this function's four-register frame. It cost every + /// OTHER generic trip ~8 Ir (an alignment `push` pinned by the collect + /// walk's diverging `double_free_abort`, a `pop`, a bool zero-extension + /// and a jump), and real programs make more of those than medium hits: + /// allocator Ir **lua +31,945, perl +23,549, sqlite +15,997**. Reverted. + /// Making the abort a may-return callee to drop the `push` was far worse + /// (see `page_collect_impl`). + /// + /// The OOM test is NOT here. In this epilogue it sat after every arm, + /// including the one that serves most real-program trips — jq: 10,307 of + /// 11,229 generic trips end in `grow_front` — which it turned from a tail + /// call into `call`, `jmp`, a copy and a `test; jne` (callgrind, per + /// instruction), for an arm that cannot fail. It lives at the three exits + /// that CAN return null: the fresh-page carve at the end of the walk, and + /// the large and huge arms. + #[inline(never)] + fn malloc_generic_once(&mut self, size: usize) -> (*mut u8, u8) { + self.malloc_generic_body(size) + } + + /// The reclaim-and-retry. The nested pass it makes lands back here on a + /// second null, and `IN_OOM_RETRY` is what turns that into the final + /// answer instead of another round. + #[cold] + #[inline(never)] + fn malloc_generic_retry(&mut self, size: usize) -> (*mut u8, u8) { + if IN_OOM_RETRY.with(Cell::get) { + return (ptr::null_mut(), 0); } + IN_OOM_RETRY.with(|c| c.set(true)); // SAFETY: owner thread; `collect_inner` allocates nothing. unsafe { self.collect_inner(true, true) }; - self.malloc_generic_once(size) + let r = self.malloc_generic_once(size); + IN_OOM_RETRY.with(|c| c.set(false)); + r } - fn malloc_generic_once(&mut self, size: usize) -> (*mut u8, bool) { + #[inline(always)] + fn malloc_generic_body(&mut self, size: usize) -> (*mut u8, u8) { self.stats.generic += 1; // Guarded objects (secure/guarded builds): sampled allocations get a // dedicated segment whose trailing page is PROT_NONE, so an overflow @@ -460,7 +545,7 @@ impl Heap { && self.guarded_rate != 0 && let Some(r) = self.try_guarded(size) { - return r; + return (r.0, u8::from(r.1)); } // COLLECT-AND-RETRY for the medium band, and it TURNS ITSELF OFF. // @@ -487,24 +572,74 @@ impl Heap { // allocations keeps the win however many threads the process has; a // producer whose pages a consumer frees pays one detection and then // behaves exactly as before. + // NOTE (2026-09-24, REFUTED): the arm usually MISSES on a real program + // (perl: 7,708 of 11,384 generic trips entered it, 7,695 found the + // front page dry after the collect), so those trips derive this bin + // twice — here and after the heartbeat. Computing it ONCE above both + // uses made it live across the heartbeat's calls: opscan `big`/`large` + // +17.00, `mixed` +11.86, perl allocator +101,962. Recomputing a pure + // value is cheaper than holding it in a callee-saved register. if !self.saw_remote_free && size > SMALL_SIZE_MAX && size <= MEDIUM_OBJ_SIZE_MAX { let bin = bins::bin(size); let p = self.pages[bin].first; if !p.is_null() { + // POP FIRST, as `malloc`'s own fast path does: when the front + // page's free list is non-empty the collect below changes + // nothing the pop needs — it would test the list, read the + // cross-thread word and re-load the list, eight instructions + // ahead of the pop on every medium hit (opscan `mixed`, + // callgrind per instruction). Remote frees still get + // collected, and the latch below still set, the first time + // the list runs dry. + // SAFETY: as for the collect-and-pop below. + let b = unsafe { page_pop(p) }; + if !b.is_null() { + self.stat_alloc(); + // SAFETY: p live per above. + return (b, u8::from(unsafe { (*p).free_is_zero })); + } // SAFETY: queue members are live pages of this heap, we are the // owner thread, and `page_collect` is the same operation the // walk below performs on this page. - let (stole, b) = unsafe { - let stole = crate::page::page_collect(p); - (stole, page_pop(p)) - }; - if stole { + // + // The latch is stored BEFORE the pop, inside the branch that + // knows it. Written as `(stole, pop)` and tested after, LLVM + // merged the two collect outcomes first and then materialised + // and re-tested a flag that is constant on the path that + // matters — `xor %eax,%eax` … `test %al,%al; je`, two dead + // instructions on every medium hit (opscan `big`/`large`). + if unsafe { crate::page::page_collect(p) } { self.saw_remote_free = true; } + // SAFETY: as above. + let b = unsafe { page_pop(p) }; if !b.is_null() { self.stat_alloc(); // SAFETY: p live per above. - return (b, unsafe { (*p).free_is_zero }); + return (b, u8::from(unsafe { (*p).free_is_zero })); + } + // The front page is dry but not fully carved: grow it HERE. + // Falling through, the trip ran the heartbeat, derived this + // same bin again, collected this same page again and only + // then reached `grow_front` — and on a real program that is + // the common medium outcome, not the hit (perl: 7,693 of + // 7,707 medium trips missed here; callgrind, per + // instruction). Skipping the heartbeat is what the hit above + // already does; it still runs when the page is exhausted and + // the walk is needed. + // SAFETY: `p` is the live front page of this bin's queue, + // its free list just came back empty, and it has room to + // carve — `grow_front`'s contract. + // + // `w` is passed as "above the direct table" rather than + // computed: `grow_front` reads it only for that test, a medium + // size is always above it, and computing `wsize_from_size` + // here made LLVM rewrite the bin arithmetic on the HIT path + // above (+5.00 Ir/op on opscan `big`/`large`). + unsafe { + if (*p).capacity < (*p).reserved { + return self.grow_front(bin, crate::types::SMALL_WSIZE_MAX + 1, p); + } } } } @@ -527,20 +662,30 @@ impl Heap { // // `reclaim = false`: this is a routine sweep of our own pages, not the // orphan adoption a forced `mi_collect(true)` performs. - if self.generic_countdown == 0 { + // + // One decrement whose wrap means "it was zero", rather than a test + // and then a decrement: the same schedule (N trips between sweeps, + // reset on the trip that finds zero), in a form that is a + // memory-destination `sub` and one branch on every generic trip + // instead of load, test, branch, decrement and store. + let left = self.generic_countdown.wrapping_sub(1); + self.generic_countdown = left; + if left == usize::MAX { self.generic_countdown = crate::options::get_clamp(crate::options::GENERIC_COLLECT, 1, 1_000_000) as usize; // SAFETY: owner thread, and `collect_inner` allocates nothing. unsafe { self.collect_inner(false, false) }; - } else { - self.generic_countdown -= 1; } if size > MEDIUM_OBJ_SIZE_MAX { - return if size <= LARGE_OBJ_SIZE_MAX { + let r = if size <= LARGE_OBJ_SIZE_MAX { self.large_alloc(size) } else { self.huge_alloc(size, 8, 0) }; + if r.0.is_null() { + return self.malloc_generic_retry(size); + } + return (r.0, u8::from(r.1)); } let bin = bins::bin(size); let w = wsize_from_size(size); @@ -568,7 +713,7 @@ impl Heap { } let b = page_pop(p); self.stat_alloc(); - return (b, (*p).free_is_zero); + return (b, u8::from((*p).free_is_zero)); } if (*p).capacity < (*p).reserved { return self.grow_front(bin, w, p); @@ -613,7 +758,7 @@ impl Heap { /// `capacity < reserved`. #[cold] #[inline(never)] - unsafe fn grow_front(&mut self, bin: usize, w: usize, p: *mut Page) -> (*mut u8, bool) { + unsafe fn grow_front(&mut self, bin: usize, w: usize, p: *mut Page) -> (*mut u8, u8) { // SAFETY: forwarded contract. unsafe { page_extend(p, (*p).area); @@ -626,7 +771,7 @@ impl Heap { } let b = page_pop(p); self.stat_alloc(); - (b, (*p).free_is_zero) + (b, u8::from((*p).free_is_zero)) } } @@ -646,7 +791,7 @@ impl Heap { /// # Safety /// `bin` indexes `self.pages`. #[inline(never)] - unsafe fn malloc_generic_walk(&mut self, bin: usize) -> (*mut u8, bool) { + unsafe fn malloc_generic_walk(&mut self, bin: usize) -> (*mut u8, u8) { // SAFETY: all page/queue manipulation below happens under the heap // lock on pages owned by this heap; raw pointers are used so no two // Rust references to the same Page coexist. @@ -692,7 +837,7 @@ impl Heap { self.update_direct(bin); let b = page_pop(p); self.stat_alloc(); - return (b, (*p).free_is_zero); + return (b, u8::from((*p).free_is_zero)); } // Truly full: park it so the queue front stays useful. let next = (*p).next; @@ -721,14 +866,18 @@ impl Heap { let p = self.fresh_page(bin, bsize); if p.is_null() { self.update_direct(bin); - return (ptr::null_mut(), false); // OOM + // OOM: the one reclaim-and-retry, with this bin's block size — + // the same class and the same block the caller would get. + // Here and not in `malloc_generic_once`'s epilogue: see + // `Heap::malloc_generic`. + return self.malloc_generic_retry(bsize); } page_extend(p, (*p).area); self.stats.extends += 1; self.update_direct(bin); let b = page_pop(p); self.stat_alloc(); - (b, (*p).free_is_zero) + (b, u8::from((*p).free_is_zero)) } } @@ -964,6 +1113,11 @@ impl Heap { } } + /// Out of line: `segment::huge_alloc` returns its `Result` through a stack + /// slot, and inlined here that slot became part of `malloc_generic_once`'s + /// frame — a `sub`/`add` of the stack pointer on EVERY generic trip for + /// an arm taken on requests above 32 MiB (callgrind, per instruction). + #[inline(never)] fn huge_alloc(&mut self, size: usize, align: usize, offset: usize) -> (*mut u8, bool) { match segment::huge_alloc(size, align, offset, self.arena_id) { Ok((seg, block)) => { @@ -1284,7 +1438,16 @@ impl Heap { // EXACTLY flat: LLVM already inlines `free_local` and proves // the block non-null from this loop's own condition.) self.free_local(b.cast()); - self.stats.delayed_frees += 1; + // Debug-only, like `allocs`/`frees` (`stat_alloc`) and + // upstream's `MI_STAT`: a read-modify-write per drained block + // on the cross-thread path, and nothing reads it but the stats + // printer — no test, bench or probe (unlike `generic` and + // `extends`, which the release bench prints as work-parity + // counters and so stay live). + #[cfg(debug_assertions)] + { + self.stats.delayed_frees += 1; + } b = next; } } @@ -1363,6 +1526,15 @@ impl Heap { // Running queue pointer: `self.pages[bin]` re-indexed the array // (and re-borrowed `self`) on every one of the MAX_NORMAL_BIN // steps, on every collect. + // + // REFUTED (2026-09-25): moving the per-queue body out of line so + // this scan would keep its pointer in a register (inlined, the + // body's calls spill it, and every EMPTY queue pays a reload and a + // store: nine instructions per bin). The thread-exit workload + // gained 272 instructions per thread, but the split changed + // inlining around the generic path's periodic collect and every + // opscan op got worse — `big`/`large` +4.00, `mixed` +2.89 — with + // perl +45,287 and python +79,035 whole-program. let qbase: *mut PageQueue = (&raw mut self.pages).cast::(); let mut q: *mut PageQueue = qbase.add(1); while bin <= MAX_NORMAL_BIN { diff --git a/crates/rusty_alloc/src/options.rs b/crates/rusty_alloc/src/options.rs index 7b38cc7..00cd5aa 100644 --- a/crates/rusty_alloc/src/options.rs +++ b/crates/rusty_alloc/src/options.rs @@ -322,7 +322,25 @@ const DEFAULTS: [i64; OPTION_COUNT] = [ static VALUES: [AtomicI64; OPTION_COUNT] = [const { AtomicI64::new(i64::MIN) }; OPTION_COUNT]; static ENV_PARSED: AtomicBool = AtomicBool::new(false); +/// Parse the environment once. The common case — every call after the first — +/// is one plain load and a branch, inlined into the caller. +/// +/// This used to be the `swap` below on EVERY call: a locked read-modify-write +/// behind an out-of-line call, on every `options::get`. `span_free` reads +/// `purge_delay` on every span it frees, so a 2 MiB alloc/free pair paid +/// **19 Ir in `ensure_init`** per operation (callgrind, opscan `huge`) to +/// learn that the environment had already been read. The swap stays, in the +/// cold arm, as what decides WHICH thread runs the pass. +#[inline(always)] fn ensure_init() { + if !ENV_PARSED.load(Ordering::Acquire) { + ensure_init_slow(); + } +} + +#[cold] +#[inline(never)] +fn ensure_init_slow() { if ENV_PARSED.swap(true, Ordering::AcqRel) { return; } @@ -335,13 +353,15 @@ fn ensure_init() { VALUES[i].store(DEFAULTS[i], Ordering::Release); } // ...and neither has `wasm32-unknown-unknown`. `std::env::var` there is a - // stub that always fails, so this loop formatted 76 strings, allocated 76 - // `String`s and read an environment that cannot exist — on every startup, - // to find nothing. It also dragged `core::fmt`, `alloc::fmt::format` and - // `str::to_uppercase` into a module that otherwise needs none of them: - // `options::get` was the LARGEST function in a wasm build at 3,708 bytes, - // ahead of anything in the allocator proper. Same deletion as the `no_std` - // arm above, for the same reason — there is nothing to read. + // stub that always fails, so the pass used to format 76 strings, allocate + // 76 `String`s and read an environment that cannot exist — on every + // startup, to find nothing. It also dragged `core::fmt`, + // `alloc::fmt::format` and `str::to_uppercase` into a module that + // otherwise needs none of them: `options::get` was the LARGEST function in + // a wasm build at 3,708 bytes, ahead of anything in the allocator proper. + // Same deletion as the `no_std` arm above, for the same reason — there is + // nothing to read. (The pass owns no memory any more, see `env`, but the + // wasm size ratchet is kept exactly where it was by leaving it out.) // // `target_os = "unknown"` and not `target_arch` alone: wasm32-wasip1 does // have an environment and keeps the pass. @@ -349,24 +369,286 @@ fn ensure_init() { feature = "std", not(all(target_arch = "wasm32", target_os = "unknown")) ))] - for i in 0..OPTION_COUNT { - let name = OPTION_NAMES[i].to_uppercase(); - let val = std::env::var(std::format!("RUSTY_ALLOC_{name}")) - .or_else(|_| std::env::var(std::format!("MIMALLOC_{name}"))) - .ok() - .and_then(|s| parse_value(&s)); - if let Some(v) = val { - VALUES[i].store(v, Ordering::Release); + env::pass(); +} + +/// The environment pass, and why it owns no memory. +/// +/// It used to be `OPTION_NAMES[i].to_uppercase()` and two `format!`ed keys +/// handed to `std::env::var`, every one of which returns an owned `String`: +/// 38 options x (one uppercase name, two keys, up to two values) made **251 +/// allocations through the global allocator on the first allocation of every +/// process**, measured from a consumer behind a counting `GlobalAlloc` +/// (`docs/plans/youslowbro.md` §4), where mimalloc makes none. Each one landed +/// in the heap that was still being set up — allocator re-entrancy during +/// initialisation — and every short CLI run paid all of them. +/// +/// Now the key is built by hand in a stack buffer sized by the table, the +/// value lands in a second stack buffer through `prim::getenv` — the raw +/// `getenv` / `GetEnvironmentVariableA` that upstream's prim calls — and the +/// parse reads the bytes in place. Zero allocations, and +/// `crates/rusty_alloc_api/tests/reentrancy.rs` fails if one comes back. +#[cfg(all( + feature = "std", + not(all(target_arch = "wasm32", target_os = "unknown")) +))] +mod env { + use core::sync::atomic::Ordering; + + use super::{OPTION_COUNT, OPTION_NAMES, VALUES}; + + /// Longest option name: the key buffer is sized by the table, not by hand. + const MAX_NAME_LEN: usize = { + let mut m = 0; + let mut i = 0; + while i < OPTION_COUNT { + if OPTION_NAMES[i].len() > m { + m = OPTION_NAMES[i].len(); + } + i += 1; + } + m + }; + /// `RUSTY_ALLOC_` is the longer prefix; `+ 1` for the NUL a C `getenv` needs. + const KEY_CAP: usize = "RUSTY_ALLOC_".len() + MAX_NAME_LEN + 1; + /// Values are integers and booleans; upstream's buffer is 64 bytes as well. + const VALUE_CAP: usize = 64; + + /// `RUSTY_ALLOC_` first, `MIMALLOC_` second, as before. + pub(super) fn pass() { + #[cfg(all(target_os = "linux", not(miri)))] + { + scan_pass(); + } + #[cfg(not(all(target_os = "linux", not(miri))))] + { + keyed_pass(); } } -} -#[cfg(feature = "std")] -fn parse_value(s: &str) -> Option { - match s.trim().to_ascii_lowercase().as_str() { - "" | "1" | "true" | "yes" | "on" => Some(1), - "0" | "false" | "no" | "off" => Some(0), - t => t.parse::().ok(), + /// The same pass as ONE walk of the environment (`prim::env_for_each`) + /// instead of 76 `getenv` calls, each of which walks all of it: 27,011 + /// instructions on the first allocation of every process, 2 % of a + /// one-line `sort` (callgrind, inclusive). Keeps [`keyed_pass`]'s + /// semantics exactly: the FIRST occurrence of a key is the one `getenv` + /// returns, a value too long for `VALUE_CAP` reads as unset (so it falls + /// back to `MIMALLOC_`), and a present-but-unparsable `RUSTY_ALLOC_` value + /// keeps the default without consulting `MIMALLOC_`. + #[cfg(all(target_os = "linux", not(miri)))] + fn scan_pass() { + let mut rusty = [core::ptr::null::(); OPTION_COUNT]; + let mut mi = [core::ptr::null::(); OPTION_COUNT]; + crate::prim::env_for_each(|e| { + // SAFETY: `e` is a NUL-terminated environment entry; `strip` + // and `match_name` stop at the first mismatch, so no read passes + // its NUL. + unsafe { + let (table, rest) = if let Some(r) = strip(e, b"RUSTY_ALLOC_") { + (&mut rusty, r) + } else if let Some(r) = strip(e, b"MIMALLOC_") { + (&mut mi, r) + } else { + return; + }; + for i in 0..OPTION_COUNT { + if let Some(v) = match_name(rest, OPTION_NAMES[i].as_bytes()) { + if table[i].is_null() { + table[i] = v; + } + break; + } + } + } + }); + let mut val = [0u8; VALUE_CAP]; + for i in 0..OPTION_COUNT { + // SAFETY: each non-null entry points at a NUL-terminated value. + let got = + unsafe { copy_value(rusty[i], &mut val).or_else(|| copy_value(mi[i], &mut val)) }; + if let Some(v) = got.and_then(|len| parse_value(&val[..len])) { + VALUES[i].store(v, Ordering::Release); + } + } + } + + /// `e` past `prefix`, if it starts with it. + /// + /// # Safety + /// `e` NUL-terminated. + #[cfg(all(target_os = "linux", not(miri)))] + unsafe fn strip(e: *const u8, prefix: &[u8]) -> Option<*const u8> { + for (k, &b) in prefix.iter().enumerate() { + // SAFETY: a mismatch (the NUL included) returns before reading on. + if unsafe { *e.add(k) } != b { + return None; + } + } + // SAFETY: every byte up to here matched a non-NUL prefix byte. + Some(unsafe { e.add(prefix.len()) }) + } + + /// The value after `NAME=` when `rest` is exactly the uppercased `name` + /// followed by `=` — `getenv`'s case-sensitive match. + /// + /// # Safety + /// `rest` NUL-terminated. + #[cfg(all(target_os = "linux", not(miri)))] + unsafe fn match_name(rest: *const u8, name: &[u8]) -> Option<*const u8> { + for (k, &b) in name.iter().enumerate() { + // SAFETY: as in `strip`. + if unsafe { *rest.add(k) } != b.to_ascii_uppercase() { + return None; + } + } + // SAFETY: as in `strip`. + unsafe { (*rest.add(name.len()) == b'=').then(|| rest.add(name.len() + 1)) } + } + + /// `prim::getenv`'s copy: the value into `out`, `None` when `v` is null or + /// the value does not fit. + /// + /// # Safety + /// `v` null or NUL-terminated. + #[cfg(all(target_os = "linux", not(miri)))] + unsafe fn copy_value(v: *const u8, out: &mut [u8]) -> Option { + if v.is_null() { + return None; + } + for (n, slot) in out.iter_mut().enumerate() { + // SAFETY: stops at the value's NUL. + let b = unsafe { *v.add(n) }; + if b == 0 { + return Some(n); + } + *slot = b; + } + None // longer than the buffer: not an option value + } + + /// One `getenv` per key: every host without `prim::env_for_each`. + #[cfg_attr(all(target_os = "linux", not(miri)), allow(dead_code))] + fn keyed_pass() { + let mut key = [0u8; KEY_CAP]; + let mut val = [0u8; VALUE_CAP]; + for i in 0..OPTION_COUNT { + let name = OPTION_NAMES[i].as_bytes(); + let n = build_key(&mut key, b"RUSTY_ALLOC_", name); + let got = lookup(&key[..n], &mut val).or_else(|| { + let n = build_key(&mut key, b"MIMALLOC_", name); + lookup(&key[..n], &mut val) + }); + if let Some(v) = got.and_then(|len| parse_value(&val[..len])) { + VALUES[i].store(v, Ordering::Release); + } + } + } + + /// `\0`, `name` ASCII-uppercased, into `key`; returns the + /// length written, NUL included. `KEY_CAP` bounds every combination. + fn build_key(key: &mut [u8; KEY_CAP], prefix: &[u8], name: &[u8]) -> usize { + let mut n = 0; + for &b in prefix { + key[n] = b; + n += 1; + } + for &b in name { + key[n] = b.to_ascii_uppercase(); + n += 1; + } + key[n] = 0; + n + 1 + } + + /// One variable, no allocation: `prim::getenv` on a hosted OS. + #[cfg(all(any(windows, unix), not(miri)))] + fn lookup(key: &[u8], out: &mut [u8]) -> Option { + crate::prim::getenv(key, out) + } + + /// The same through `std::env` where there is no raw `getenv` to call: + /// Miri, which interprets `std::env` itself, and wasm32-wasip1. This arm + /// allocates, and that is accepted — neither is a shipping host. + #[cfg(not(all(any(windows, unix), not(miri))))] + fn lookup(key: &[u8], out: &mut [u8]) -> Option { + let name = core::str::from_utf8(key.split_last()?.1).ok()?; + let v = std::env::var_os(name)?; + let v = v.to_str()?.as_bytes(); + if v.len() >= out.len() { + return None; + } + out[..v.len()].copy_from_slice(v); + Some(v.len()) + } + + /// mimalloc's value grammar, on the bytes in place: 1/0/true/false/yes/ + /// no/on/off in any case (an empty value is 1), else a decimal integer. + fn parse_value(v: &[u8]) -> Option { + let t = core::str::from_utf8(v).ok()?.trim(); + if ["", "1", "true", "yes", "on"] + .iter() + .any(|w| t.eq_ignore_ascii_case(w)) + { + return Some(1); + } + if ["0", "false", "no", "off"] + .iter() + .any(|w| t.eq_ignore_ascii_case(w)) + { + return Some(0); + } + t.parse::().ok() + } + + #[cfg(test)] + mod tests { + use super::*; + + #[test] + fn keys_are_the_uppercased_names_with_a_nul_and_always_fit() { + let mut key = [0u8; KEY_CAP]; + let n = build_key(&mut key, b"RUSTY_ALLOC_", b"purge_delay"); + assert_eq!(&key[..n], b"RUSTY_ALLOC_PURGE_DELAY\0"); + let n = build_key(&mut key, b"MIMALLOC_", b"show_stats"); + assert_eq!(&key[..n], b"MIMALLOC_SHOW_STATS\0"); + // The longest name with the longer prefix is exactly what the + // buffer was sized for; `build_key` would panic past it. + for name in OPTION_NAMES { + let n = build_key(&mut key, b"RUSTY_ALLOC_", name.as_bytes()); + assert!(n <= KEY_CAP, "{name}"); + assert_eq!(key[n - 1], 0); + } + } + + #[test] + fn values_follow_mimalloc_grammar() { + assert_eq!(parse_value(b""), Some(1)); + assert_eq!(parse_value(b" On "), Some(1)); + assert_eq!(parse_value(b"TRUE"), Some(1)); + assert_eq!(parse_value(b"off"), Some(0)); + assert_eq!(parse_value(b"No"), Some(0)); + assert_eq!(parse_value(b" -1 "), Some(-1)); + assert_eq!(parse_value(b"1048576"), Some(1_048_576)); + assert_eq!(parse_value(b"maybe"), None); + assert_eq!(parse_value(&[0xff, 0xfe]), None); + } + + #[test] + fn lookup_reads_the_process_environment_in_place() { + // SAFETY: this test is the only writer of these names, and the + // one reader of the environment in this process — the one-shot + // option pass — ran when the harness made its first allocation. + unsafe { + std::env::set_var("RUSTY_ALLOC_TEST_PROBE", " 42 "); + std::env::set_var("RUSTY_ALLOC_TEST_LONG_PROBE", "x".repeat(VALUE_CAP)); + } + let mut out = [0u8; VALUE_CAP]; + let n = lookup(b"RUSTY_ALLOC_TEST_PROBE\0", &mut out).expect("set"); + assert_eq!(&out[..n], b" 42 "); + assert_eq!(parse_value(&out[..n]), Some(42)); + assert_eq!(lookup(b"RUSTY_ALLOC_TEST_PROBE_UNSET\0", &mut out), None); + // A value that does not fit is not an option value. + assert_eq!(lookup(b"RUSTY_ALLOC_TEST_LONG_PROBE\0", &mut out), None); + } } } @@ -385,6 +667,13 @@ pub fn get(option: usize) -> i64 { if crate::ONE_REGION { return DEFAULTS[option]; } + // NOTE (2026-09-24, REFUTED): a "table complete" flag, so the steady + // state could skip the `i64::MIN` sentinel test below, measured flat on + // every opscan op and slightly worse on real programs (allocator Ir lua + // +527, perl +255, sqlite +162). The sentinel is already free where it is + // hot: the callers test the value's SIGN, `i64::MIN` is negative, and + // LLVM folds both questions into one `js` (see `span_free`'s + // `purge_delay` read). ensure_init(); let v = VALUES[option].load(Ordering::Acquire); if v == i64::MIN { DEFAULTS[option] } else { v } @@ -633,12 +922,6 @@ pub fn deferred_free(force: bool) { if DEFERRED_FUN.load_fun().is_null() { return; } - // A hook that allocates re-enters `malloc_generic` → `deferred_free`. - // Without a per-thread guard that recurses until the stack dies - // (OH-rusty_alloc-17). - if IN_DEFERRED.with(|c| c.get()) { - return; - } fire_deferred(force); } @@ -649,9 +932,25 @@ pub fn deferred_free(force: bool) { /// forces the whole heartbeat's caller to preserve callee-saved registers, on /// every slow-path allocation, for a hook that is unregistered in nearly every /// process. The peek above is all the common path executes. +/// +/// The re-entry guard lives HERE, not in `deferred_free`, and that is the +/// point of it. In a cdylib `IN_DEFERRED` is a general-dynamic thread-local, +/// so reading it is a call to `__tls_get_addr` — and LLVM treats that address +/// computation as free to hoist, so with the guard inline it ran ABOVE the +/// null test of the hook pointer: **12 Ir in `__tls_get_addr` on every +/// slow-path allocation**, for a hook no process had registered (callgrind, +/// opscan `huge`/`big`; the disassembly loads `DEFERRED_FUN` and calls +/// `__tls_get_addr` before testing it). Out here it runs only when a hook +/// exists. #[cold] #[inline(never)] fn fire_deferred(force: bool) { + // A hook that allocates re-enters `malloc_generic` → `deferred_free`. + // Without a per-thread guard that recurses until the stack dies + // (OH-rusty_alloc-17). + if IN_DEFERRED.with(|c| c.get()) { + return; + } IN_DEFERRED.with(|c| c.set(true)); let _clear = HookExit(|| IN_DEFERRED.with(|c| c.set(false))); let (f, a) = DEFERRED_FUN.load(); diff --git a/crates/rusty_alloc/src/page.rs b/crates/rusty_alloc/src/page.rs index e6c5cdd..4ffa09d 100644 --- a/crates/rusty_alloc/src/page.rs +++ b/crates/rusty_alloc/src/page.rs @@ -831,15 +831,37 @@ pub unsafe fn remote_free(page: *mut Page, block: *mut Block) { break; } } - // Restore DELAYED, preserving whatever the owner did - // to the pointer bits meanwhile. + // Release FREEING, preserving whatever the owner did + // to the pointer bits meanwhile — to NORMAL, as + // upstream does (`mi_tf_set_delayed(.., MI_NO_DELAYED_FREE)` + // in `_mi_free_block_mt`), not back to DELAYED. + // + // One block on the owner's delayed list is all the + // owner needs: draining it un-parks the page + // (`free_local_at`), and from then on the page is + // scanned. Restoring DELAYED sent EVERY later remote + // free to a parked page through this three-CAS route + // and the owner through `drain_delayed` + + // `free_local_at` per block — 82 % of the frees on the + // `xthread` op. After NORMAL they are one CAS onto the + // page's own list, collected in bulk. + // + // SINGLE-BLOCK pages (large spans, huge segments) keep + // DELAYED: they are never scanned, so their only route + // to the owner is this list. + let restore = + if (*page).flags.load(Ordering::Relaxed) & pflags::SINGLE_BLOCK != 0 { + XFLAG_DELAYED + } else { + XFLAG_NORMAL + }; loop { let y = (*page).xthread_free.load(Ordering::Acquire); if (*page) .xthread_free .compare_exchange_weak( y, - (y & !XMASK) | XFLAG_DELAYED, + (y & !XMASK) | restore, Ordering::AcqRel, Ordering::Relaxed, ) @@ -1056,6 +1078,14 @@ unsafe fn page_collect_impl(page: *mut Page, flag: usize) let mut tail = head; let mut n = 1u32; loop { + // NOTE (2026-09-24, REFUTED): calling the may-return + // `remote_double_free` here instead — to stop this diverging + // call pinning an alignment `push` to the top of + // `malloc_generic_once`'s frameless medium arm — made it far + // WORSE: a call that may return, inside the walk loop, keeps + // `tail`/`n`/`page` live across it, and the entry grew a + // six-register frame (opscan `huge` +20.00, every op worse). + // The may-return trick works only in TAIL position. if n > used { double_free_abort(); } @@ -1091,6 +1121,17 @@ const EXTEND_SHIFT_BASE: u32 = { crate::types::SEGMENT_SLICE_SIZE.trailing_zeros() - 12 }; +/// The fewest blocks one [`page_extend`] links, whatever the block size — +/// upstream's `MI_MIN_EXTEND` (4 outside its secure mode). +/// +/// Ours was an implicit 1 (`.max(1)`), a drift from upstream: every class +/// whose 4 KiB byte bound rounds below four blocks — everything above 1 KiB — +/// was carved one to three blocks per extend, so on perl 8,990 of 10,639 +/// extends linked one block and each was a full generic trip (callgrind, per +/// instruction). The floor only moves list-linking earlier inside a span +/// that is already committed; it touches at most four block headers. +const MIN_EXTEND: usize = 4; + /// Lazily extend the free list into never-used capacity (`mi_page_extend_free`). /// /// # Safety @@ -1210,7 +1251,11 @@ pub unsafe fn page_extend(page: *mut Page, area: *mut u8) { // Derived, so it is correct at every geometry and byte-identical at the // default (65536 >> 12 == 16, whose log2 is the old 4). let span_shift = EXTEND_SHIFT_BASE + (*page).slice_count.trailing_zeros(); - let take = ((reserved >> span_shift).max(1)).min(reserved - capacity); + // At least `MIN_EXTEND` blocks, as upstream (`MI_MIN_EXTEND`, 4): the + // byte bound alone gives ONE block to any class above 4 KiB and one + // to three to classes above 1 KiB, so a fresh medium page served one + // allocation per slow-path trip. See `MIN_EXTEND`. + let take = ((reserved >> span_shift).max(MIN_EXTEND)).min(reserved - capacity); let start = area.add(capacity * bsize); // Link the fresh blocks in address order. // diff --git a/crates/rusty_alloc/src/prim/fixed.rs b/crates/rusty_alloc/src/prim/fixed.rs index 29deaec..3451e58 100644 --- a/crates/rusty_alloc/src/prim/fixed.rs +++ b/crates/rusty_alloc/src/prim/fixed.rs @@ -2085,6 +2085,10 @@ mod tests { assert!(!shape_of(SMALL_SIZE_MAX + 1).direct_route); // The route top is 1024 on 64-bit and 512 on 32-bit, whatever the // geometry -- it is a pointer-width fact, not a profile one. + // At the DEFAULT `SMALL_WSIZE_MAX`. The relationship asserted above + // holds at every setting of `ra_small_wsize`; these literals are the + // default arm's, and are pinned as such rather than unconditionally. + #[cfg(not(any(ra_small_wsize = "256", ra_small_wsize = "512")))] if core::mem::size_of::() == 8 { assert_eq!(SMALL_SIZE_MAX, 1024); } else if core::mem::size_of::() == 4 { diff --git a/crates/rusty_alloc/src/prim/mod.rs b/crates/rusty_alloc/src/prim/mod.rs index 8945d42..ecfe55c 100644 --- a/crates/rusty_alloc/src/prim/mod.rs +++ b/crates/rusty_alloc/src/prim/mod.rs @@ -202,6 +202,43 @@ pub fn numa_node_count() -> usize { sys::numa_node_count().max(1) } +/// Copy the value of the environment variable `name` — NUL-terminated ASCII — +/// into `out`, **without allocating** (`_mi_prim_getenv`). `Some(len)` with the +/// value's byte length when the variable is set and fits; `None` when it is +/// unset, empty on Windows (the API reports both as 0 and upstream's prim +/// treats both as unset), or longer than `out`. +/// +/// This exists because `options.rs` read the environment through +/// `std::env::var`, whose key and value are both owned `String`s: 38 options +/// x 2 prefixes made **251 allocations through the global allocator on the +/// first allocation of every process** — inside the heap that was being set +/// up (`docs/plans/youslowbro.md` §4). Hosted platforms only: a firmware has +/// no environment, wasm has no raw `getenv`, and Miri interprets `std::env` +/// itself, so `options.rs` keeps a `std::env` fallback for those. +#[cfg(all(any(windows, unix), not(miri)))] +pub fn getenv(name: &[u8], out: &mut [u8]) -> Option { + debug_assert_eq!(name.last(), Some(&0), "getenv: name must be NUL-terminated"); + if name.last() != Some(&0) || out.is_empty() { + return None; + } + sys::getenv(name, out) +} + +/// Call `f` with a pointer to every `NAME=VALUE` entry of the process +/// environment, in order, each NUL-terminated — one walk of `environ`, without +/// allocating. Linux only (glibc and musl export `environ`; a macOS dylib has +/// to go through `_NSGetEnviron`, and Windows has no such block of C strings). +/// +/// For `options.rs`, which needs 38 options under two prefixes: as 76 +/// [`getenv`] calls that is 76 walks of the whole environment, 17,372 +/// instructions inside libc and 27,011 inclusive on the first allocation of +/// every process — 2 % of a one-line `sort` (callgrind). One walk that looks +/// only at entries starting with either prefix does the same job. +#[cfg(all(target_os = "linux", not(miri)))] +pub fn env_for_each(f: impl FnMut(*const u8)) { + sys::env_for_each(f); +} + /// Cheap unique id of the calling thread (the heap-ownership key from M4). #[inline] pub fn thread_id() -> usize { diff --git a/crates/rusty_alloc/src/prim/unix.rs b/crates/rusty_alloc/src/prim/unix.rs index 5a64278..d7540e7 100644 --- a/crates/rusty_alloc/src/prim/unix.rs +++ b/crates/rusty_alloc/src/prim/unix.rs @@ -230,6 +230,54 @@ pub(super) fn numa_node_count() -> usize { 1 // sysfs/getcpu wiring lands with arenas (M6) } +/// `_mi_prim_getenv`: the value of `name` into `out`, no allocation of ours. +pub(super) fn getenv(name: &[u8], out: &mut [u8]) -> Option { + // SAFETY: `name` is NUL-terminated (checked by `prim::getenv`). `getenv` + // returns null or a pointer into the process environment block, which + // lives for the process and is read here once, byte by byte, stopping at + // its NUL or at `out.len()`; nothing is written through it. This is not + // safe against a concurrent `setenv`, which is the standing caveat on + // `std::env::set_var` as well, and the same call upstream's prim makes. + unsafe { + let v: *const u8 = libc::getenv(name.as_ptr().cast()).cast(); + if v.is_null() { + return None; + } + let mut n = 0; + while n < out.len() { + let b = *v.add(n); + if b == 0 { + return Some(n); + } + out[n] = b; + n += 1; + } + } + None // longer than the buffer: not an option value +} + +/// Every entry of `environ`, in order; see `prim::env_for_each`. +#[cfg(target_os = "linux")] +pub(super) fn env_for_each(mut f: impl FnMut(*const u8)) { + unsafe extern "C" { + static environ: *const *const u8; + } + // SAFETY: `environ` is the C runtime's NULL-terminated array of + // NUL-terminated `NAME=VALUE` strings, live for the process. It is only + // read here, and each entry is handed on as a pointer, not copied. As with + // `getenv` above, this is not safe against a concurrent `setenv`. + unsafe { + let mut p = environ; + if p.is_null() { + return; + } + while !(*p).is_null() { + f(*p); + p = p.add(1); + } + } +} + #[inline] pub(super) fn thread_id() -> usize { // SAFETY: no preconditions; pthread_self is async-signal-safe. diff --git a/crates/rusty_alloc/src/prim/windows.rs b/crates/rusty_alloc/src/prim/windows.rs index 8a80683..dd4524a 100644 --- a/crates/rusty_alloc/src/prim/windows.rs +++ b/crates/rusty_alloc/src/prim/windows.rs @@ -13,6 +13,7 @@ use core::ptr; use core::sync::atomic::{AtomicU64, Ordering}; use windows_sys::Win32::Foundation::GetLastError; +use windows_sys::Win32::System::Environment::GetEnvironmentVariableA; use windows_sys::Win32::System::Memory::{ GetLargePageMinimum, MEM_COMMIT, MEM_DECOMMIT, MEM_FREE, MEM_LARGE_PAGES, MEM_RELEASE, MEM_RESERVE, MEM_RESET, MEMORY_BASIC_INFORMATION, PAGE_NOACCESS, PAGE_READWRITE, VirtualAlloc, @@ -234,6 +235,28 @@ pub(super) fn numa_node_count() -> usize { if ok != 0 { highest as usize + 1 } else { 1 } } +/// `_mi_prim_getenv`: the value of `name` into `out`, no allocation of ours. +/// +/// The ANSI form, as upstream's prim uses: option values are integers and +/// booleans, so there is nothing to lose to the code page. The API converts +/// the name to UTF-16 on the process heap — never through this allocator. +pub(super) fn getenv(name: &[u8], out: &mut [u8]) -> Option { + let cap = u32::try_from(out.len()).ok()?; + // SAFETY: `name` is NUL-terminated (checked by `prim::getenv`); `out` is + // a live buffer of exactly `cap` bytes, and the call writes at most `cap` + // bytes into it — when the value does not fit it writes nothing and + // returns the size it would need. + let n = unsafe { GetEnvironmentVariableA(name.as_ptr(), out.as_mut_ptr(), cap) }; + // 0 is "unset" or "empty", which the API cannot distinguish without a + // `GetLastError` round-trip; upstream's prim treats both as unset, and so + // does this. `n >= cap` means it did not fit: not an option value. + if n == 0 || n >= cap { + None + } else { + Some(n as usize) + } +} + #[inline] pub(super) fn thread_id() -> usize { // SAFETY: no preconditions. diff --git a/crates/rusty_alloc/src/segment.rs b/crates/rusty_alloc/src/segment.rs index e13518d..c5cbd5c 100644 --- a/crates/rusty_alloc/src/segment.rs +++ b/crates/rusty_alloc/src/segment.rs @@ -455,9 +455,22 @@ pub unsafe fn segment_free(seg: *mut Segment) -> Result<(), PrimError> { /// Write the span markers for a span at `idx` of `len` slices: first slot /// offset 0, interior+last offsets pointing back. /// +/// Interior slices below `from` are NOT rewritten: the caller knows they +/// already point back to `idx` in both representations. That holds for every +/// slice of every span in a normal segment (this is their only writer, and +/// `debug_validate_segment` checks it), so a span taken from the FRONT of a +/// free span, or a freed span that did not merge into its left neighbour, +/// starts where its interiors already point. Rewriting them was the largest +/// loop in both `span_alloc` and `span_free`, and it stored values that were +/// already there: 31 of 31 interior slices, twice, per 2 MiB allocate-and-free +/// (opscan `huge`, callgrind per instruction). `from == 1` marks everything. +/// /// # Safety -/// `[idx, idx+len)` must lie in the carved region of `seg` under the heap lock. -unsafe fn span_mark(seg: *mut Segment, idx: usize, len: usize) { +/// `[idx, idx+len)` must lie in the carved region of `seg` under the heap lock; +/// `from >= 1`, and slices `idx+1 .. idx+min(from, len)` must already point +/// back to `idx`. +unsafe fn span_mark(seg: *mut Segment, idx: usize, len: usize, from: usize) { + debug_assert!(from >= 1, "span_mark: slot 0 is the head, not an interior"); // SAFETY: caller contract keeps every index in bounds. unsafe { (*seg).pages[idx].slice_offset = 0; @@ -478,10 +491,17 @@ unsafe fn span_mark(seg: *mut Segment, idx: usize, len: usize) { let owner = page_off_for(idx); let tab: *mut u32 = (&raw mut (*seg).page_off).cast(); *tab.wrapping_add(idx) = owner; + // NOTE (2026-09-24, REFUTED): splitting the owner-table stores into + // their own `slice::fill` pass, away from the stride-88 `slice_offset` + // stores, made both loops cheaper (151 -> 92 + 43 instructions for a + // 32-slice span) and the function DEARER: each loop brings its own + // unroll setup and scalar remainder. Opscan `huge` +28.00, perl +340, + // lua -8,327, sqlite -862 allocator Ir — a sign flip by span length, + // not a win. Same verdict as the fill `huge_alloc` measured flat. let base: *mut Page = (&raw mut (*seg).pages).cast(); - let mut slot = base.wrapping_add(idx + 1); - let mut ent = tab.wrapping_add(idx + 1); - let mut j = 1; + let mut slot = base.wrapping_add(idx + from); + let mut ent = tab.wrapping_add(idx + from); + let mut j = from; while j < len { // SLICES back to the span start, not bytes — see Page::slice_offset. (*slot).slice_offset = j as u16; @@ -516,10 +536,10 @@ unsafe fn span_mark(seg: *mut Segment, idx: usize, len: usize) { /// /// # Safety /// As [`span_mark`]; the span must not be in the free list already. -unsafe fn span_mark_free(seg: *mut Segment, idx: usize, len: usize) { +unsafe fn span_mark_free(seg: *mut Segment, idx: usize, len: usize, from: usize) { // SAFETY: caller contract. unsafe { - span_mark(seg, idx, len); + span_mark(seg, idx, len, from); let slot: *mut Page = &raw mut (*seg).pages[idx]; (*slot).block_size = 0; // free marker (*slot).prev = ptr::null_mut(); @@ -642,9 +662,11 @@ pub unsafe fn span_alloc(seg: *mut Segment, slices: usize) -> (*mut Page, bool) // remainder inherits the (now cleared) purged state. span_recommit(seg, idx, len); if len > slices { - span_mark_free(seg, idx + slices, len - slices); + // The remainder starts at a new slice: re-mark all of it. + span_mark_free(seg, idx + slices, len - slices, 1); } - span_mark(seg, idx, slices); + // The FRONT of a free span: its interiors already point here. + span_mark(seg, idx, slices, slices); (*seg).used_pages += 1; return (s, false); } @@ -659,7 +681,8 @@ pub unsafe fn span_alloc(seg: *mut Segment, slices: usize) -> (*mut Page, bool) (*seg).next_free_slice = (idx + slices) as u32; (*seg).used_pages += 1; let start: *mut Page = &raw mut (*seg).pages[idx]; - span_mark(seg, idx, slices); + // Never-carved slices: mark all of them. + span_mark(seg, idx, slices, 1); (start, (*seg).mem_is_zero) } } @@ -683,6 +706,9 @@ pub unsafe fn span_free(seg: *mut Segment, page: *mut Page) -> bool { unsafe { let mut idx = page_index(seg, page); let mut len = (*page).slice_count as usize; + // Where the freed span sat before any merge: its interiors already + // point back to `freed_idx` (see `span_mark`). + let (freed_idx, freed_len) = (idx, len); (*seg).used_pages -= 1; // Scrub page state so a stale slot can't masquerade as live. (*page).block_size = 0; @@ -718,15 +744,26 @@ pub unsafe fn span_free(seg: *mut Segment, page: *mut Page) -> bool { idx = lstart_idx; } } - span_mark_free(seg, idx, len); + // Re-mark only what does not already point to `idx`: with no left + // merge the freed span itself starts here and only an absorbed right + // neighbour moves; after a left merge the left span is already right + // and everything from the freed span on moves. + let from = if idx == freed_idx { + freed_len + } else { + freed_idx - idx + }; + span_mark_free(seg, idx, len, from); // PURGE the coalesced free span (the RSS lever): return its pages to // the OS. Only worthwhile for multi-slice spans — a syscall costs // more than the pages a single 64 KiB slice returns. The span is // marked `purged` so reuse re-commits it; skipping that recommit is // an access violation on Windows (MEM_DECOMMIT), which is how this // was caught (2026-08-05). - let purge_delay = crate::options::get(15); - if purge_delay >= 0 && len >= crate::types::MEDIUM_PAGE_SLICES { + // Length FIRST: it is in a register, the option is a table load, a + // bounds test and a sentinel test — and most freed spans are short, so + // most frees never need the option at all. + if len >= crate::types::MEDIUM_PAGE_SLICES && crate::options::get(15) >= 0 { let area = page_area(seg, idx); let bytes = len * SEGMENT_SLICE_SIZE; let decommits = crate::options::is_enabled(5); // purge_decommits @@ -917,7 +954,22 @@ pub fn huge_alloc( // offsets back to it (only slices 1..512 are addressable via the mask // trick, and aligned offsets stay < SEGMENT_SIZE/2 by the contract). let page: *mut Page = &raw mut (*seg).pages[1]; - (*page).block_size = b.size - (block.addr() - seg.addr()); + // The block's usable size is what was ASKED for, rounded up to a + // slice — upstream's `psize` (`mi_segment_huge_page_alloc`, oracle + // segment.c:1599) — and NOT the reservation. The reservation is + // chunk-rounded (`chunks * SEGMENT_SIZE` through an arena), so a 33 MB + // request holds a 64 MiB chunk pair and a 64 MB one holds 96 MiB. + // This field used to report that whole extent, and `usable_size` is + // the copy length of every `realloc` that MOVES: growing a 33 MB block + // copied — and first-touched, on both sides — 64 MiB, and a 64 MB + // block 96 MiB. Measured from a consumer at **1.41x / 1.23x mimalloc, + // 0/6 pairs** for exactly those two steps, at parity for every step + // below the segment size (`docs/plans/youslowbro.md` §3). `zalloc` on + // a recycled chunk zeroes `usable_size` bytes and paid the same tax. + // The capacity is still `total_size`, and nothing reads past + // `block_size`; `usable_size` now agrees with `mi_usable_size` here. + let capacity = b.size - (block.addr() - seg.addr()); + (*page).block_size = crate::prim::align_up(size.max(1), SEGMENT_SLICE_SIZE).min(capacity); (*page).used = 1; (*page).capacity = 1; (*page).reserved = 1; diff --git a/crates/rusty_alloc/src/types.rs b/crates/rusty_alloc/src/types.rs index 30d42aa..345f645 100644 --- a/crates/rusty_alloc/src/types.rs +++ b/crates/rusty_alloc/src/types.rs @@ -11,7 +11,33 @@ pub const INTPTR_SIZE: usize = core::mem::size_of::(); /// Maximum "small" allocation in machine words (`MI_SMALL_WSIZE_MAX` = 128). /// /// Source: `mimalloc.h` v2.4.5 — `#define MI_SMALL_WSIZE_MAX (128)`. -pub const SMALL_WSIZE_MAX: usize = 128; +/// +/// `--cfg ra_small_wsize="256"` / `"512"` raises it, and the default is +/// unchanged. The cfg exists for the same reason [`GENERIC_COLLECT_DEFAULT`] +/// grew one: a bare-metal consumer could measure that this bound was costing +/// it and had no way to move it. +/// +/// **Why a 32-bit target wants it.** [`SMALL_SIZE_MAX`] is this times +/// `INTPTR_SIZE`, so the `direct[]` fast path covers 1 KiB on a 64-bit host +/// and only **512 bytes on a 32-bit chip** — the same pointer-width trap that +/// made the `direct`-route step invisible on a host sweep. Measured on an +/// ESP32-S3 (`rusty_rtos_core/firmware/esp32s3-devkit-alloc-ab`), an +/// alloc/free pair costs **114 cycles at 512 bytes and 249 at 513**: one byte +/// over the bound is +135 cycles, because the request falls off the end of +/// the table and takes the generic path. +/// +/// The price is the table: `direct` is `SMALL_WSIZE_MAX + 1` pointers per +/// heap, so `"512"` costs 2,052 bytes on a 32-bit target where 128 costs 516. +/// Nothing else moves — the bin geometry, `good_size` and every ABI-visible +/// answer are computed elsewhere and are unchanged, which is what makes this +/// safe to offer as a knob rather than a fork. +pub const SMALL_WSIZE_MAX: usize = if cfg!(ra_small_wsize = "256") { + 256 +} else if cfg!(ra_small_wsize = "512") { + 512 +} else { + 128 +}; /// Maximum "small" allocation in bytes (`MI_SMALL_SIZE_MAX` = 1 KiB on 64-bit). /// `mi_malloc_small` / `mi_zalloc_small` require `size <= SMALL_SIZE_MAX`. @@ -202,7 +228,16 @@ mod tests { fn constants_match_oracle_64bit() { // Pinned to mimalloc v2.4.5 on x86_64 / aarch64 (64-bit words). assert_eq!(INTPTR_SIZE, 8); + // `SMALL_WSIZE_MAX` is a knob (`ra_small_wsize`), so pin EACH ARM + // rather than the default's number -- the same treatment + // `ra_segment_size` gets below, and for the same reason: pinning only + // the default passes vacuously at every other setting. + #[cfg(not(any(ra_small_wsize = "256", ra_small_wsize = "512")))] assert_eq!(SMALL_SIZE_MAX, 1024); + #[cfg(ra_small_wsize = "256")] + assert_eq!(SMALL_SIZE_MAX, 2048); + #[cfg(ra_small_wsize = "512")] + assert_eq!(SMALL_SIZE_MAX, 4096); #[cfg(not(ra_small_profile))] assert_eq!(SEGMENT_SIZE, 32 * 1024 * 1024); // The small profile's geometry is a DECISION, pinned here so moving it diff --git a/crates/rusty_alloc/tests/alloc_core.rs b/crates/rusty_alloc/tests/alloc_core.rs index 13386dc..0d5fb28 100644 --- a/crates/rusty_alloc/tests/alloc_core.rs +++ b/crates/rusty_alloc/tests/alloc_core.rs @@ -342,6 +342,49 @@ fn align_storm() { } } +/// A huge block's usable size is what was asked for, rounded up to a slice — +/// not the chunk-rounded reservation it lives in. +/// +/// This is the deterministic half of `docs/plans/youslowbro.md` §3: `realloc` +/// copies `usable_size` bytes when it moves a block, and a 33 MB block used +/// to report the whole 64 MiB chunk pair as usable — so growing it copied +/// (and first-touched, on both sides) twice what the caller had written, and +/// a consumer measured the step at 1.41x mimalloc. The reservation itself is +/// unchanged; only what is REPORTED, and therefore what is copied and what +/// `zalloc` zeroes on a recycled chunk. +#[test] +fn huge_usable_size_is_the_request_not_the_reservation() { + use rusty_alloc::types::{SEGMENT_SIZE, SEGMENT_SLICE_SIZE}; + for mb in [33usize, 64] { + let size = mb << 20; + assert!(size > SEGMENT_SIZE, "{mb} MB must be a huge allocation"); + let p = malloc(size); + assert!(!p.is_null(), "malloc({size})"); + // SAFETY: fresh live block of ≥ size bytes; freed once below. + unsafe { + let us = usable_size(p); + assert!(us >= size, "usable {us} < requested {size}"); + assert!( + us < size + SEGMENT_SLICE_SIZE, + "usable {us} for a {mb} MB block reports the reservation \ + ({} would be the chunk boundary), not the request", + size.div_ceil(SEGMENT_SIZE) * SEGMENT_SIZE + ); + // The bytes a move preserves are exactly the ones the caller + // could have written: fill the whole usable extent, grow past + // the reservation, and check the prefix survived. + p.write(0xA5); + p.add(us - 1).write(0x5A); + let np = realloc(p, 2 * size); + assert!(!np.is_null(), "realloc({size} -> {})", 2 * size); + assert_eq!(np.read(), 0xA5); + assert_eq!(np.add(us - 1).read(), 0x5A); + assert!(usable_size(np) >= 2 * size); + free(np); + } + } +} + #[test] fn realloc_storm() { // Seeded randomized realloc churn with content verification — the M3 diff --git a/crates/rusty_alloc/tests/loom_xthread.rs b/crates/rusty_alloc/tests/loom_xthread.rs index fcfc140..e2ad0fa 100644 --- a/crates/rusty_alloc/tests/loom_xthread.rs +++ b/crates/rusty_alloc/tests/loom_xthread.rs @@ -90,16 +90,19 @@ impl Model { break; } } - // Restore DELAYED (list bits may have changed? no — - // only WE can push while FREEING; owner may collect - // though, so re-read and preserve the pointer bits). + // Release FREEING to NORMAL, as `page::remote_free` + // does for a multi-block page (upstream's + // MI_NO_DELAYED_FREE): the delayed entry just pushed + // is what makes the owner un-park the page, and later + // remotes then push onto the page list. Re-read and + // preserve the pointer bits — the owner may collect. loop { let y = self.xthread.load(Ordering::Acquire); if self .xthread .compare_exchange( y, - (y & !XMASK) | DELAYED, + (y & !XMASK) | NORMAL, Ordering::AcqRel, Ordering::Relaxed, ) @@ -216,8 +219,9 @@ fn delayed_push_vs_abandon() { }); } -/// Extended soak (set LOOM_EXTENDED=1; CI nightly): two remotes vs the -/// abandoner under a preemption bound — the wide-space variant. +/// Extended soak (set LOOM_EXTENDED=1): two remotes vs the abandoner under a +/// preemption bound — the wide-space variant. No CI workflow runs it (none of +/// `.github/workflows` mentions loom); it is run by hand, for ~45 minutes. #[test] fn delayed_push_vs_abandon_two_remotes_extended() { if std::env::var_os("LOOM_EXTENDED").is_none() { @@ -248,6 +252,34 @@ fn delayed_push_vs_abandon_two_remotes_extended() { }); } +/// A parked (DELAYED) page receives TWO remote frees while the owner drains +/// its delayed list and collects: the first goes to the delayed list and +/// releases the page to NORMAL, the second lands on the page list. Every +/// block must end in exactly one place, and the heap is never touched after +/// FREEING is released. +#[test] +fn delayed_then_normal_vs_owner() { + loom::model(|| { + let m = Arc::new(Model::new(DELAYED)); + let m1 = m.clone(); + let t1 = thread::spawn(move || { + m1.remote_free(1); + m1.remote_free(2); + }); + let mo = m.clone(); + let to = thread::spawn(move || mo.owner_drain_delayed() + mo.owner_collect()); + t1.join().unwrap(); + let during = to.join().unwrap(); + let after = m.owner_drain_delayed() + m.owner_collect(); + assert_eq!(during + after, 2, "a block was lost or double-counted"); + assert_eq!( + m.xthread.load(Ordering::Relaxed) & XMASK, + NORMAL, + "the page must be left scannable" + ); + }); +} + /// Remote NORMAL pushes race the owner's collect; nothing may be lost. #[test] fn normal_push_vs_collect() { diff --git a/crates/rusty_alloc/tests/options_env.rs b/crates/rusty_alloc/tests/options_env.rs new file mode 100644 index 0000000..bc01edc --- /dev/null +++ b/crates/rusty_alloc/tests/options_env.rs @@ -0,0 +1,97 @@ +//! The environment pass still READS the environment now that it owns no +//! memory (`options::env`, `docs/plans/youslowbro.md` §4). +//! +//! The pass runs once per process, on the first option read, so the only way +//! to test it end to end is a CHILD with the variables set before it starts — +//! the same shape `foreign_free.rs` and `double_free.rs` use. The parent sets +//! both prefixes, both value grammars and a precedence conflict; the child +//! reads them back through the public `options::get`. + +use std::process::Command; + +use rusty_alloc::options::{get, get_size, is_enabled}; + +const MARKER: &str = "RUSTY_ALLOC_OPTIONS_ENV_CHILD"; + +// ABI indices (see `options::OPTION_NAMES`). +const MAX_ERRORS: usize = 19; +const MAX_WARNINGS: usize = 20; +const PURGE_DELAY: usize = 15; +const SHOW_STATS: usize = 1; +const ARENA_RESERVE: usize = 23; +const OS_TAG: usize = 18; +const MAX_SEGMENT_RECLAIM: usize = 21; + +fn child() -> ! { + // Touch the allocator first, as any real process does: the pass has run + // by the time `get` is called, whichever order the harness chose. + let v = rusty_alloc::alloc::malloc(64); + assert!(!v.is_null()); + // SAFETY: live block, freed once. + unsafe { rusty_alloc::alloc::free(v) }; + + assert_eq!(get(MAX_ERRORS), 7, "RUSTY_ALLOC_MAX_ERRORS"); + assert_eq!( + get(MAX_WARNINGS), + 11, + "MIMALLOC_MAX_WARNINGS (compat prefix)" + ); + assert_eq!(get(PURGE_DELAY), 5, "RUSTY_ALLOC_ wins over MIMALLOC_"); + assert!(is_enabled(SHOW_STATS), "MIMALLOC_SHOW_STATS=YES (any case)"); + assert_eq!( + get_size(ARENA_RESERVE), + 4096 * 1024, + "RUSTY_ALLOC_ARENA_RESERVE in KiB" + ); + // The one-walk pass (Linux) must keep `getenv`'s semantics exactly. + assert_eq!( + get(OS_TAG), + 3, + "a RUSTY_ALLOC_ value too long to be an option reads as unset: MIMALLOC_ applies" + ); + assert_eq!( + get(MAX_SEGMENT_RECLAIM), + 10, + "a present but unparsable RUSTY_ALLOC_ value keeps the default; MIMALLOC_ is not read" + ); + assert_eq!(get(MAX_ERRORS), 7, "a longer look-alike key does not match"); + // The exit CODE is the signal: libtest captures a test's stdout and never + // flushes it through `process::exit`, and a plain 0 is also what a + // harness that never reached this function would return. + std::process::exit(CHILD_OK); +} + +const CHILD_OK: i32 = 42; + +#[test] +#[cfg_attr(miri, ignore)] // spawns a child process +fn environment_variables_reach_the_options_table() { + if std::env::var(MARKER).is_ok() { + child(); + } + let exe = std::env::current_exe().expect("current_exe"); + let out = Command::new(exe) + .env(MARKER, "1") + .env("RUSTY_ALLOC_MAX_ERRORS", "7") + .env("MIMALLOC_MAX_WARNINGS", " 11 ") + .env("RUSTY_ALLOC_PURGE_DELAY", "5") + .env("MIMALLOC_PURGE_DELAY", "9") + .env("MIMALLOC_SHOW_STATS", "YES") + .env("RUSTY_ALLOC_ARENA_RESERVE", "4096") + .env("RUSTY_ALLOC_OS_TAG", "1".repeat(70)) + .env("MIMALLOC_OS_TAG", "3") + .env("RUSTY_ALLOC_MAX_SEGMENT_RECLAIM", "abc") + .env("MIMALLOC_MAX_SEGMENT_RECLAIM", "4") + .env("RUSTY_ALLOC_MAX_ERRORSX", "99") + .arg("--test-threads=1") + .output() + .expect("spawn child"); + let stdout = String::from_utf8_lossy(&out.stdout); + let stderr = String::from_utf8_lossy(&out.stderr); + assert_eq!( + out.status.code(), + Some(CHILD_OK), + "child did not read its environment back\nstatus: {:?}\nstdout:\n{stdout}\nstderr:\n{stderr}", + out.status + ); +} diff --git a/crates/rusty_alloc_api/src/lib.rs b/crates/rusty_alloc_api/src/lib.rs index 1de5dfb..9633b10 100644 --- a/crates/rusty_alloc_api/src/lib.rs +++ b/crates/rusty_alloc_api/src/lib.rs @@ -21,6 +21,21 @@ pub use rusty_alloc::{MI_COMPAT_VERSION, VERSION, version}; /// The global allocator handle (zero-sized). pub struct RustyAlloc; +/// One machine word: every block the allocator hands out is aligned this far. +const WORD: usize = core::mem::size_of::(); + +/// The alignment the size classes already give every block of MORE than one +/// word: `bins::bin` rounds word counts up to even ones (upstream's +/// `MI_ALIGN2W`), and every page area starts on a slice boundary, so a block +/// of two words or more sits on a two-word boundary — 16 bytes on a 64-bit +/// target, `MAX_ALIGN_SIZE`. Only the one-word class is not, which is why a +/// request aligned between one and two words is raised to two words. +/// +/// This is the alignment hashbrown asks for on every table (its SSE2 control +/// group is 16 bytes), so a Rust `HashMap` used to go through the aligned +/// path on every allocation and through allocate-copy-free on every realloc. +const NATURAL_ALIGN: usize = 2 * WORD; + /// A first-class heap (plan §5.14): `Drop` runs `mi_heap_delete` semantics /// (blocks migrate to the thread's backing heap and stay valid) unless built /// with [`Heap::new_destroyable`], where `Drop` releases every block at once. @@ -136,40 +151,97 @@ impl Drop for Heap { // SAFETY: GlobalAlloc contract — Layout-described allocation/free delegated to // the rusty_alloc core, which returns blocks satisfying the layout's size and -// alignment (natural bins for align ≤ 8; the aligned path otherwise) and -// accepts any such block back in `free` regardless of which thread frees it -// (M4: per-thread heaps, no lock — `free` routes by the segment's owner and +// alignment (natural bins up to `NATURAL_ALIGN`; the aligned path above it) +// and accepts any such block back in `free` regardless of which thread frees +// it (M4: per-thread heaps, no lock — `free` routes by the segment's owner and // hands cross-thread blocks to the loom-modeled remote protocol). +// +// Every method is `#[inline]`, as the `mimalloc` crate's are. rustc generates +// `__rust_alloc` and friends in the crate that declares `#[global_allocator]`, +// and without the hint each one was a shim that loaded `&self`, shuffled the +// arguments and jumped through the GOT into an out-of-line method — on every +// Rust allocation and every drop. With it the fast paths land in the shims and +// in their callers. Measured whole-program (callgrind, same output): a boxed- +// tree/buffer/`Rc` workload −17.9 %, a HashMap/BTreeMap/String one −4.8 %, for +// +1,888 and +2,784 bytes of text. unsafe impl GlobalAlloc for RustyAlloc { + #[inline] unsafe fn alloc(&self, layout: Layout) -> *mut u8 { - if layout.align() <= 8 { + if layout.align() <= WORD { rusty_alloc::alloc::malloc(layout.size()) + } else if layout.align() <= NATURAL_ALIGN { + // See `NATURAL_ALIGN`: a class of two words or more is already + // aligned this far, so only a one-word request needs raising. + rusty_alloc::alloc::malloc(layout.size().max(NATURAL_ALIGN)) } else { - rusty_alloc::alloc::malloc_aligned(layout.size(), layout.align()) + // SAFETY: `Layout` guarantees a power-of-two alignment, the one + // precondition `malloc_aligned_pow2` adds over `malloc_aligned`. + unsafe { rusty_alloc::alloc::malloc_aligned_pow2(layout.size(), layout.align()) } } } + #[inline] unsafe fn dealloc(&self, ptr: *mut u8, _layout: Layout) { - // SAFETY: GlobalAlloc contract — ptr came from `alloc` and is freed once. - unsafe { rusty_alloc::alloc::free(ptr) } + // `free_inline`, not `free`: `dealloc` IS a free and does nothing + // else, the case `free_inline` exists for (the LD_PRELOAD export is + // the other). Through `free` every Rust deallocation paid a `jmp` + // into it and its null test; `GlobalAlloc` never passes null, and the + // hint lets that test fold away. + // SAFETY: GlobalAlloc contract — ptr came from `alloc`, is non-null + // and is freed once. + unsafe { + core::hint::assert_unchecked(!ptr.is_null()); + rusty_alloc::alloc::free_inline(ptr) + } } + #[inline] unsafe fn alloc_zeroed(&self, layout: Layout) -> *mut u8 { - if layout.align() <= 8 { + if layout.align() <= WORD { rusty_alloc::alloc::zalloc(layout.size()) + } else if layout.align() <= NATURAL_ALIGN { + rusty_alloc::alloc::zalloc(layout.size().max(NATURAL_ALIGN)) } else { rusty_alloc::alloc::zalloc_aligned(layout.size(), layout.align()) } } + #[inline] unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { - if layout.align() <= 8 { - // SAFETY: GlobalAlloc contract — ptr live, invalidated on move; - // our realloc preserves min(old, new) bytes. - unsafe { rusty_alloc::alloc::realloc(ptr, new_size) } + if layout.align() <= NATURAL_ALIGN { + // Up to `NATURAL_ALIGN` the size classes carry the alignment, so + // the in-place arm is open to these layouts too: a block that + // stays keeps its address, and a move lands in a class of at + // least two words. Before this, every realloc aligned above one + // word (a growing `Vec`, say) was an allocate-copy-free. + let new_size = if layout.align() <= WORD { + new_size + } else { + new_size.max(NATURAL_ALIGN) + }; + // SAFETY: GlobalAlloc contract — ptr live and non-null, + // invalidated on move; our realloc preserves min(old, new) bytes. + // The hint lets `realloc`'s null arm fold away. + unsafe { + core::hint::assert_unchecked(!ptr.is_null()); + rusty_alloc::alloc::realloc(ptr, new_size) + } } else { - // Aligned realloc lands in M5; the default alloc-copy-dealloc is - // correct through our aligned paths meanwhile. + // Above `NATURAL_ALIGN`. The block already satisfies + // `layout.align()` and keeps it if it stays, so when the new size + // fits and at least half the block stays in use — `realloc`'s own + // in-place rule, which `mi_realloc_aligned` applies too — it + // stays. This arm used to allocate, copy and free every time, + // shrinks and fits included (bench/rust-globalalloc + // `overaligned`). + // SAFETY: GlobalAlloc contract — ptr is a live, non-null block + // of ours. + let usable = unsafe { rusty_alloc::alloc::usable_size(ptr) }; + if new_size <= usable && new_size >= usable / 2 { + return ptr; + } + // Otherwise allocate through the aligned path, copy the bytes the + // layout says are live, free. // SAFETY: forwarded GlobalAlloc contract. unsafe { let new_layout = Layout::from_size_align_unchecked(new_size, layout.align()); diff --git a/crates/rusty_alloc_api/tests/natural_align.rs b/crates/rusty_alloc_api/tests/natural_align.rs new file mode 100644 index 0000000..f8f8ae8 --- /dev/null +++ b/crates/rusty_alloc_api/tests/natural_align.rs @@ -0,0 +1,142 @@ +//! `GlobalAlloc` serves layouts aligned up to two words from the natural size +//! classes (`NATURAL_ALIGN` in `lib.rs`), trusting that every class of two +//! words or more is two-word aligned. This pins that trust for every small +//! size, across the medium and large bands and into huge, through `alloc`, +//! `alloc_zeroed` and every `realloc` step between them — including the +//! one-word requests that must be raised to reach an aligned class. + +use rusty_alloc_api::RustyAlloc; +use std::alloc::{GlobalAlloc, Layout}; + +/// Two words: 16 on a 64-bit target, the alignment hashbrown asks for. +const NATURAL: usize = 2 * core::mem::size_of::(); + +fn sizes() -> impl Iterator { + (0..=4200usize).chain([ + 8191, + 16_385, + 65_535, + 131_072, + 300_001, + 1 << 20, + (4 << 20) + 3, + ]) +} + +#[test] +fn natural_alignment_holds_for_alloc_zeroed_and_realloc() { + let a = RustyAlloc; + let align = NATURAL; + { + let mut prev: Option<(*mut u8, Layout)> = None; + for size in sizes() { + let l = Layout::from_size_align(size, align).unwrap(); + // SAFETY: valid layout; each block is freed exactly once below. + unsafe { + let p = a.alloc(l); + assert!(!p.is_null()); + assert_eq!(p.addr() % align, 0, "alloc({size}, {align}) misaligned"); + p.write_bytes(0xA5, size); + a.dealloc(p, l); + + let z = a.alloc_zeroed(l); + assert!(!z.is_null()); + assert_eq!( + z.addr() % align, + 0, + "alloc_zeroed({size}, {align}) misaligned" + ); + assert!(std::slice::from_raw_parts(z, size).iter().all(|&b| b == 0)); + a.dealloc(z, l); + + // Walk one block through every size: grows, and (at the + // jumps back down between bands) shrinks, in place or moved. + let (q, ql) = match prev.take() { + None => { + let q = a.alloc(l); + q.write_bytes(0x5A, size); + (q, l) + } + Some((old, ol)) => { + let q = a.realloc(old, ol, size); + assert!(!q.is_null()); + let keep = ol.size().min(size); + assert!( + std::slice::from_raw_parts(q, keep) + .iter() + .all(|&b| b == 0x5A), + "realloc {} -> {size} lost bytes", + ol.size() + ); + q.add(keep).write_bytes(0x5A, size - keep); + (q, l) + } + }; + assert_eq!( + q.addr() % align, + 0, + "realloc(.., {size}) at {align} misaligned" + ); + prev = Some((q, ql)); + } + } + if let Some((q, ql)) = prev { + // SAFETY: last live block of the walk. + unsafe { a.dealloc(q, ql) }; + } + } +} + +/// Above two words `realloc` may now keep a block in place. Walk one block +/// up and down through sizes at alignments the classes do not give for free: +/// every result must keep the alignment and the live prefix. +#[test] +fn over_aligned_realloc_keeps_alignment_and_bytes() { + let a = RustyAlloc; + for align in [64usize, 4096] { + let steps = [ + 64usize, 128, 100, 64, 40, 700, 350, 5000, 2600, 70_000, 36_000, 64, + ]; + let mut l = Layout::from_size_align(steps[0], align).unwrap(); + // SAFETY: valid layout; the block is realloc'd through the walk and + // freed once at the end. + unsafe { + let mut p = a.alloc(l); + assert!(!p.is_null()); + for (i, b) in std::slice::from_raw_parts_mut(p, l.size()) + .iter_mut() + .enumerate() + { + *b = i as u8; + } + for &next in &steps[1..] { + let q = a.realloc(p, l, next); + assert!(!q.is_null()); + assert_eq!( + q.addr() % align, + 0, + "realloc {} -> {next} at {align}", + l.size() + ); + let keep = l.size().min(next); + assert!( + std::slice::from_raw_parts(q, keep) + .iter() + .enumerate() + .all(|(i, &b)| b == i as u8), + "realloc {} -> {next} at {align} lost bytes", + l.size() + ); + for (i, b) in std::slice::from_raw_parts_mut(q, next) + .iter_mut() + .enumerate() + { + *b = i as u8; + } + p = q; + l = Layout::from_size_align(next, align).unwrap(); + } + a.dealloc(p, l); + } + } +} diff --git a/crates/rusty_alloc_api/tests/reentrancy.rs b/crates/rusty_alloc_api/tests/reentrancy.rs new file mode 100644 index 0000000..08ebcd2 --- /dev/null +++ b/crates/rusty_alloc_api/tests/reentrancy.rs @@ -0,0 +1,129 @@ +//! The allocator must never allocate THROUGH the global allocator while it is +//! serving a request. +//! +//! It did: the options environment pass in `options.rs` built its keys with +//! `to_uppercase` and `format!` and read them with `std::env::var`, all of +//! which return owned `String`s — **251 allocations re-entered the global +//! allocator on the first allocation of every process** on Windows (fewer on +//! Linux, where `std::env::var` allocates less per call), each landing in the +//! heap that was still being set up. mimalloc makes none. Reported from a +//! consumer behind a counting `GlobalAlloc` (`docs/plans/youslowbro.md` §4); +//! this test is that instrument, made permanent. +//! +//! The wrapper counts a call as re-entrant when it arrives while the same +//! thread is already inside one — a per-thread depth, so concurrent threads +//! cannot produce a false positive. The counter is cumulative from process +//! start, so the harness's own first allocation (which runs the pass) is +//! covered without any ordering assumption about which test runs first. + +use std::alloc::{GlobalAlloc, Layout}; +use std::cell::Cell; +use std::hint::black_box; +use std::sync::atomic::{AtomicUsize, Ordering::Relaxed}; + +use rusty_alloc_api::RustyAlloc; + +static CALLS: AtomicUsize = AtomicUsize::new(0); +static REENTRANT: AtomicUsize = AtomicUsize::new(0); + +thread_local! { + // `const` and `Copy`: no lazy initialisation and no destructor, so reading + // it inside the allocator allocates nothing and cannot fail at thread exit. + static DEPTH: Cell = const { Cell::new(0) }; +} + +struct Counting; + +impl Counting { + /// Enter one allocator call; `true` when this thread was already inside one. + fn enter() -> bool { + CALLS.fetch_add(1, Relaxed); + let nested = DEPTH.with(|d| { + let n = d.get(); + d.set(n + 1); + n > 0 + }); + if nested { + REENTRANT.fetch_add(1, Relaxed); + } + nested + } + + fn leave() { + DEPTH.with(|d| d.set(d.get() - 1)); + } +} + +// SAFETY: every method forwards to `RustyAlloc` with the same arguments and +// returns its result unchanged; the bookkeeping around the call touches only +// atomics and a `Copy` thread-local. +unsafe impl GlobalAlloc for Counting { + unsafe fn alloc(&self, l: Layout) -> *mut u8 { + Counting::enter(); + // SAFETY: forwarded GlobalAlloc contract. + let p = unsafe { RustyAlloc.alloc(l) }; + Counting::leave(); + p + } + + unsafe fn dealloc(&self, p: *mut u8, l: Layout) { + Counting::enter(); + // SAFETY: forwarded GlobalAlloc contract. + unsafe { RustyAlloc.dealloc(p, l) }; + Counting::leave(); + } + + unsafe fn alloc_zeroed(&self, l: Layout) -> *mut u8 { + Counting::enter(); + // SAFETY: forwarded GlobalAlloc contract. + let p = unsafe { RustyAlloc.alloc_zeroed(l) }; + Counting::leave(); + p + } + + unsafe fn realloc(&self, p: *mut u8, l: Layout, n: usize) -> *mut u8 { + Counting::enter(); + // SAFETY: forwarded GlobalAlloc contract. + let q = unsafe { RustyAlloc.realloc(p, l, n) }; + Counting::leave(); + q + } +} + +#[global_allocator] +static A: Counting = Counting; + +#[test] +fn the_allocator_never_re_enters_the_global_allocator() { + // Walk every regime once so the paths that set state up have all run on + // this thread: a fresh heap's first small block, a medium page, a large + // span, a dedicated huge segment, a moving realloc, and a cross-thread + // free (a worker's heap is created, used and abandoned). + let small = black_box(vec![1u8; 64]); + let medium = black_box(vec![2u8; 8 << 10]); + let large = black_box(vec![3u8; 300 << 10]); + #[cfg(not(miri))] + let huge = black_box(vec![4u8; 33 << 20]); + let mut grown: Vec = Vec::with_capacity(8); + grown.extend(0..(1 << 16)); + let from_worker = std::thread::spawn(|| black_box(vec![5u8; 4096])) + .join() + .expect("worker"); + drop((small, medium, large, black_box(grown), from_worker)); + #[cfg(not(miri))] + drop(huge); + + let calls = CALLS.load(Relaxed); + let reentrant = REENTRANT.load(Relaxed); + println!("global allocator calls so far: {calls}, re-entrant: {reentrant}"); + assert!( + calls > 0, + "the counting wrapper is not the global allocator" + ); + assert_eq!( + reentrant, 0, + "rusty_alloc allocated through the global allocator {reentrant} times \ + while serving a request — the start-up options pass used to do this \ + 251 times (docs/plans/youslowbro.md §4)" + ); +} diff --git a/crates/rusty_alloc_ffi/src/lib.rs b/crates/rusty_alloc_ffi/src/lib.rs index 64240ed..c18e49e 100644 --- a/crates/rusty_alloc_ffi/src/lib.rs +++ b/crates/rusty_alloc_ffi/src/lib.rs @@ -678,7 +678,9 @@ pub unsafe fn posix_memalign_impl(out: *mut *mut c_void, alignment: usize, size: if alignment & (alignment - 1) != 0 { return einval(); // not a power of two } - let p = alloc::malloc_aligned(size, alignment); + // SAFETY: `alignment` was proven a power of two just above, which is the + // one precondition `malloc_aligned_pow2` adds — it skips re-testing it. + let p = unsafe { alloc::malloc_aligned_pow2(size, alignment) }; if p.is_null() { return enomem(); } @@ -2082,7 +2084,21 @@ pub extern "C" fn mi_new_aligned(size: usize, alignment: usize) -> *mut c_void { /// Sibling of [`new_impl`]. #[inline] pub fn new_aligned_impl(size: usize, alignment: usize) -> *mut c_void { - let p = alloc::malloc_aligned(size, alignment); + // C++ requires `align_val_t` to be a power of two; test it the cheap way + // (`x & (x - 1)`: `is_power_of_two()` is a SWAR popcount without + // `popcnt`) and take the inline `malloc_aligned_pow2` fast path, as + // `posix_memalign` does. Anything else keeps the old route and its error. + // Measured on `bench/alignednew.cpp`: see the LEDGER, CURIOSITY ROUND FOUR. + #[allow( + clippy::manual_is_power_of_two, + reason = "is_power_of_two() is a SWAR popcount here; see above" + )] + let p = if alignment != 0 && alignment & (alignment - 1) == 0 { + // SAFETY: `alignment` was just shown to be a power of two. + unsafe { alloc::malloc_aligned_pow2(size, alignment) } + } else { + alloc::malloc_aligned(size, alignment) + }; if p.is_null() { new_aligned_oom(); } diff --git a/docs/LEDGER.md b/docs/LEDGER.md index a4eb507..6ca1413 100644 --- a/docs/LEDGER.md +++ b/docs/LEDGER.md @@ -4,6 +4,567 @@ One entry per milestone/brick: what landed, the numbers with their method lines, what was reverted and **which kind** of revert (measured-worse vs within-noise). Newest first. +## DOWNSTREAM CORPUS — real consumers on this tree after four curiosity rounds (2026-09-25) + +Every consumer below built against THIS working tree, verified per row from +the candidate's `Cargo.lock` (`rusty_alloc` 2.2.0 by path, no registry +source) — a check the harness did not have, and whose absence had let two +rows pass while testing the crates.io allocator (`tools/corpus/README.md`, +lessons 5–7). + +| check | platform | result | +|---|---|---| +| `tools/corpus/run.sh --test`: spacedb-sdk, spacedb-sdk `secure`, rusty_alloc_default, rusty_zstd, spacedb published mirror | Windows | **5 PASS** | +| same: rusty_maplibre | Windows | FAIL on the known 2.0.0 `no_std` break; with the documented migration, **1,199 passed, 0 failed** | +| spacedb-sdk and `secure`, rusty_zstd (C cross tests vs zstd 1.5.7) | Linux | 12 + 12 + **174 passed, 0 failed** | +| SpaceDB whole workspace, this tree `LD_PRELOAD`ed into every test binary | Linux | **314 passed, 0 failed** | +| rusty_zstd whole workspace (lib/bins/tests), preloaded | Linux | **194 passed, 0 failed** | +| Silesia, 12 files × levels 1/3/9/19 plus `-T4`, through `rzstd` | Windows and Linux | **60 + 60 cases, 0 failures** | + +The Silesia check, per file and level: this tree's `rzstd` compresses +**byte-identically** to the same CLI on the published allocator (2.0.5), its +frame decodes to the original through itself and through C zstd, and it +decodes C zstd's own frame. The `-T4` runs put cross-thread frees under a +real multi-threaded workload. + +Not ours, reported for the owners: rusty_zstd's ignored +`encode::tests::size_table_silesia` panics at `encode/tables.rs:850`, a +`debug_assert!` on the 24-bit position field of a packed chain head, when a +debug build reaches the 51 MB `mozilla`. It fails identically on the system +allocator (control run), and release builds, which skip the assertion, +round-trip `mozilla` at every level above. SpaceDB's ignored set is five +README snippets marked `ignore`, not runtime tests. + +## CURIOSITY, ROUND FOUR — five more, and the allocator column that saw a quarter of the allocator (2026-09-25) + +**The instrument first, because it changes how every earlier entry reads.** +The "allocator-only Ir" column that rounds one to three quote summed the +lines of `callgrind_annotate` that carry the library's `[object]` tag — and +annotate prints that tag only on a function's FIRST line. Code inlined from +another source file (`page.rs` into `malloc`, `init.rs`'s TLS read, the +`page_extend` loop inside `grow_front`) prints untagged and was never +counted. On perl the column read 6.62 M; the allocator's real self cost is +**24.66 M**. The column is now an exact sum over the raw callgrind file by +each function's object (`objir.awk`: object names are compressed and may +first appear on a `cob=` line; the inclusive line after each `calls=` is +skipped), and it agreed to the instruction with whole-program Ir on the +brick that exposed it (perl −26,655 both ways, where the old column read ++0). Consequences for the earlier entries: every allocator-only DELTA quoted +there was a real saving but a subset of the true one, and every percentage +was taken against a base about 3–4x too small — lua's "40 % lighter" in +round one, lua −3.8 % and jq −2.4 % in round two. The whole-program columns, +which decided perl, sqlite and python, are unaffected. + +| instrument | before | after | Δ | +|---|---:|---:|---:| +| opscan `aligned` | 92.13 | 83.69 | **−9.2 %** | +| opscan `big` / `large` | 95.50 | 92.00 | −3.7 % | +| opscan `mixed` | 95.58 | 92.70 | −3.0 % | +| C++ `alignednew` (new, `bench/alignednew.cpp`) | 149.04 | 145.16 | −2.6 % | +| Rust `overaligned` (new), per 20,000 steps | 1,784,927 | 1,349,905 | **−24.4 %** | +| one-line `sort`, whole process | 1,371,706 | 1,349,222 | −1.6 % | + +| program | exact allocator Δ | whole-program Δ | +|---|---:|---:| +| perl | −35,612 | −53,352 | +| sqlite | −12,128 | −29,628 | +| python | −9,858 | −25,651 | +| gawk | −9,390 | −27,548 | +| lua | −9,002 | (moves run to run) | +| jq | −8,992 | (moves run to run) | + +**The five:** + +1. **The options pass walks the environment once** (`options.rs`, + `prim::env_for_each`, Linux). 38 options under two prefixes were 76 + `getenv` calls, each walking the whole environment: 27,011 instructions + on the first allocation of every process, 17,372 of them inside libc — 2 % + of a one-line `sort`, and 77 % of all the allocator work that process did. + One walk that looks only at entries starting with either prefix keeps the + old semantics exactly (first occurrence wins, a value too long for the + buffer reads as unset and falls back to `MIMALLOC_`, an unparsable + `RUSTY_ALLOC_` value keeps the default); `tests/options_env.rs` now pins + those three, on the old path (Windows) and the new one (Linux). The saving + grows with the environment: this box's shell has 21 variables. +2. **`malloc_aligned_pow2` is `#[inline]`**, so `posix_memalign` and + `GlobalAlloc` carry its fast path instead of calling it: `aligned` −8.44. +3. **The medium arm pops before it collects** (`heap.rs`). When the front + page's free list is non-empty the collect changed nothing the pop needed, + but it tested the list, read the cross-thread word and re-loaded the list + first — eight instructions on every medium hit. `malloc`'s own fast path + already pops without collecting; remote frees are still collected, and the + latch still set, when the list runs dry. `big`/`large` −3.50, `mixed` + −2.88, perl −26,655. +4. **C++ aligned `new` takes the pow2 fast path** (`rusty_alloc_ffi`): + `align_val_t` must be a power of two, tested the cheap way (`x & (x-1)`; + `is_power_of_two()` is a SWAR popcount without `popcnt`), with anything + else keeping the old checked route and its error. `alignednew` −3.88. +5. **`GlobalAlloc::realloc` above two words keeps a block in place** when + the new size fits and at least half stays — `realloc`'s own rule, which + `mi_realloc_aligned` applies too. It used to allocate, copy and free every + time; when it must move it still copies only the layout's live bytes. + `overaligned` −24.4 %; `tests/natural_align.rs` walks a block through + twelve sizes at 64- and 4096-byte alignment checking alignment and bytes. + +**Refuted:** moving `collect_inner`'s per-queue body out of line so the bin +scan keeps its pointer in a register (inlined, the body's calls spill it and +every EMPTY queue pays a reload and a store — nine instructions per bin, twice +per thread exit). The new `threads` workload gained 272 per thread, but the +split changed inlining around the generic path's periodic collect and every +opscan op got worse (`big`/`large` +4.00, `mixed` +2.89, perl +45,287, +python +79,035). Noted at the site. + +**Looked at and left:** a recycled segment's 47 KB header `memset` is 76 % +of a thread-churn workload's Ir, but glibc zeroes that size with `rep +stosb`, which callgrind counts once per byte — an instrument inflation, not +a target, and upstream zeroes the same header. jq retires ~700 emptied pages +and re-carves them; upstream retires non-sole empty pages immediately too, so +that is a memory policy, not a finding. + +**Gates:** Windows core suites at default, `debug_checks`, `secure` (144/0 +each), `blockmap` (146/0) and the small profile (143/0); api 8/0 on x86-64 +and on `i686-pc-windows-msvc`; ffi 8/0; Linux suites at default, +`debug_checks` and `secure` (141/0 each) and the thread-heavy tests ten times +each in release (0 failures); fmt clean; clippy clean on Windows and Linux +apart from the known `ra_small_wsize` cfg warnings; unsafe census 963 (+16, +each in `UNSAFE.md`). + +## CURIOSITY, ROUND THREE — five more, and the entry point no instrument could see (2026-09-24) + +Same discipline, one new instrument. Every instrument in this repository +drives the allocator through its C ABI, but every Rust deliverable reaches it +through `GlobalAlloc` — `__rust_alloc` and friends, generated in the crate that +declares `#[global_allocator]` — and nothing measured that path. Two +deterministic Rust workloads now do (`bench/rust-globalalloc.sh`: `maps`, hash +maps plus B-trees plus strings; `trees`, boxed trees plus buffers plus `Rc`), +read as whole-program Ir by the two-point estimator, with a checksum that must +match between arms. The mimalloc oracle build was also scanned: it is slower +than ours on every opscan op, so every target below came from our own code. + +| opscan | before | after | Δ | +|---|---:|---:|---:| +| **huge** | 629.00 | **293.00** | **−53.4 %** | +| big / large | 100.00 | 95.50 | −4.5 % | +| every other op | | | ±0.01 | + +| program | measure | before | after | Δ | +|---|---|---:|---:|---:| +| perl | whole | 772,938,946 | 772,484,781 | **−454,165** | +| sqlite | whole | 316,086,544 | 316,030,456 | −56,088 | +| python | whole | 512,666,763 | 512,625,101 | −41,662 | +| jq | allocator | 5,563,583 | 5,523,229 | −40,354 | +| Rust `trees` | whole, per 20,000 steps | 23,340,202 | 18,436,920 | **−21.0 %** (+2,080 B text) | +| Rust `maps` | whole, per 20,000 steps | 28,164,304 | 26,820,518 | **−4.8 %** (+2,976 B text) | + +**The five:** + +1. **`page_extend` links at least four blocks** (`page.rs`, `MIN_EXTEND`), as + upstream's `MI_MIN_EXTEND` does. Ours had an implicit floor of one, so every + class above 1 KiB was carved one to three blocks per extend — on perl 8,990 + of 10,639 extends linked ONE block, each a full generic trip (the remainder + loop ran 8,996 times over 8,990 calls). perl −453,584, sqlite −53,052, + `big`/`large` −4.50. The raised byte bound this site refuted in August cost + last-level misses (+2.2 % to +6.4 %); this floor, measured with cachegrind: + perl +4,487 D1 misses and **+9** LL, sqlite +55 D1 and +94 LL. +2. **`GlobalAlloc` serves alignment up to two words from the natural classes** + (`rusty_alloc_api`, `NATURAL_ALIGN`). `bins::bin` rounds to even word + counts, so every block of two words or more is two-word aligned — 16 bytes + on x86-64, which is what hashbrown asks for on every table. Those layouts + took the aligned path on every allocation and allocate-copy-free on every + realloc; now they are `malloc`/`realloc` with a one-word request raised to + two words. `tests/natural_align.rs` pins it for every size to 4,200 plus + medium, large and huge, through alloc, alloc_zeroed and a realloc walk; with + the raise removed it fails on the first size-0 request. On a 32-bit target + an align-8 layout of four bytes or less was previously given a one-word + (4-byte) block; the two-word rule covers it. +3. **`dealloc` carries the free body inline** with the non-null fact stated, + instead of a `jmp` to `free` and its null test: `maps` −230,000 at 40,000 + steps, about four instructions per drop. +4. **The `GlobalAlloc` methods are `#[inline]`**, as the `mimalloc` crate's + are. Without it each `__rust_*` shim loaded `&self`, shuffled arguments and + jumped through the GOT into an out-of-line method. With it the fast paths + land in the shims and in their callers: most of the Rust columns above. + Measured whole-program because an allocator-only filter undercounts once + the paths inline into the program's own functions. +5. **`span_mark` skips interior slices that already point home** + (`segment.rs`). Every slice of every span in a normal segment points back to + its span start (`span_mark` is the only writer; `debug_validate_segment` + checks it), so a span taken from the FRONT of a free span, or a freed span + that did not merge left, already carries every interior marker. The loop + rewrote them anyway — 31 of 31 slices, twice, per 2 MiB allocate-and-free. + Now only slices whose span start moved are written. This keeps the + invariant the August assessment declined to weaken (it only stops + re-storing values that already satisfy it). **`huge` −336.00 (−53 %).** + +**Refuted:** resolving the old block's free in `realloc`'s move arm before +the `memcpy` so the opaque call would not force a re-derivation (five +instructions per move). `free` stayed byte-identical, but LLVM re-loaded the +page index anyway and holding four values across the call cost a sixth +callee-saved register: `realloc` +8.00, lua +240,221, python +302,857. Noted +at the site. **Not built:** `mi_strndup` measures its length with a byte loop +where `strnlen` would do; no workload calls it, so no instrument could see a +change. + +**Gates:** Windows core suites at default, `debug_checks`, `secure` (144/0 +each), `blockmap` (146/0) and the small profile (143/0); api 7/0, and 7/0 +again on `i686-pc-windows-msvc`, where two words is 8 bytes; ffi 8/0; wasm32 +builds clean; +Linux suites at default, `debug_checks` and `secure` (141/0 each) and the +thread-heavy tests ten times each in release (0 failures); fmt clean; clippy +clean apart from the known `ra_small_wsize` cfg warnings; unsafe census 947, +unchanged. + +## CURIOSITY, ROUND TWO — ten more, a protocol change back to upstream's, and the instrument that lied about a tail jump (2026-09-24) + +Same method as the round below, with the real-program gate widened: four +more programs (Python with every object through malloc, jq, gawk, sort; +fixtures seeded, in a persistent WSL directory — `/tmp` is wiped when the VM +stops, which once handed jq a missing file and a 25x-too-small profile), two +more opscan ops (`shbench`, the sized C++ churn driver), and — after brick 7 +— a whole-program column beside the allocator-only one. + +**Cumulative, round one's final tree → this one** (opscan Ir/op; real +programs whole-program Ir, exact for perl/sqlite/python; allocator-only Ir +for lua and jq, whose whole count moves run to run): + +| opscan | before | after | Δ | +|---|---:|---:|---:| +| **xthread** | 112.59 | **77.29** | **−31.4 %** | +| realloc | 248.37 | 231.20 | −6.9 % | +| aligned | 94.50 | 92.13 | −2.5 % | +| calloc | 94.82 | 93.19 | −1.7 % | +| huge | 638.00 | 629.00 | −1.4 % | +| big / large / mixed | 101 / 101 / 96.54 | 100 / 100 / 95.58 | −1.0 % | +| every other op | | | −0.0 … −0.7 %, `usable` flat | + +| program | measure | before | after | Δ | +|---|---|---:|---:|---:| +| perl | whole | 773,394,562 | 772,937,623 | **−456,939** | +| python | whole | 513,051,186 | 512,661,300 | **−389,886** | +| sqlite | whole | 316,142,142 | 316,086,572 | −55,570 | +| lua | allocator | 9,485,488 | 9,120,942 | −364,546 (−3.8 %) | +| jq | allocator | 5,700,261 | 5,562,891 | −137,370 (−2.4 %) | + +**The ten:** + +1. **The OOM retry test moved to the exits that can fail** (`heap.rs`): the + walk's fresh-page carve and the large/huge arms, instead of the epilogue + of `malloc_generic_once`, where it sat after arms that cannot return + null. All seven programs improved; `huge` +1 (its arm tests itself). +2. **`realloc` decides a move in one compare, and each move knows its copy + length** (`alloc.rs`): the two-sided in-place test had compiled + branch-free (`setbe`/`shr`/`setae`), so every move paid all of it. Lua's + moves are growths, Python's are shrinks below half (60,100 of 60,669). + opscan `realloc` −16.31; lua −301,818, python −125,709 allocator Ir. +3. **The generic-collect countdown is one decrement** whose wrap means "was + zero" — a memory-destination `sub` and one branch instead of load, test, + branch, decrement, store: every program −708 … −59,630. +4. **A remote free to a parked page releases FREEING to NORMAL, as upstream + does** (`page.rs::remote_free`; upstream `_mi_free_block_mt` sets + `MI_NO_DELAYED_FREE`). We restored DELAYED, so every later remote free to + that page took the three-CAS delayed route and the owner freed each block + through `drain_delayed` + `free_local_at`; after NORMAL they are one CAS + onto the page's list, collected in bulk. Single-block pages (large, huge) + keep DELAYED — never scanned, the delayed list is their only route. + **opscan `xthread` −26.6 %**; single-threaded code byte-identical. Gates: + the loom protocol model updated to mirror it plus a new + `delayed_then_normal_vs_owner` model (a parked page takes a delayed push + and then a list push while the owner drains and collects — no block lost, + page left scannable), all five models green; Linux suites at default, + `debug_checks` and `secure`, and the thread-heavy tests (`double_free`, + `stress_mt`, `abandon_rss`, `heaps`, `openheimer`, `teardown_reclaim`) + **ten times each in release, 0 failures**; Windows suites green. +5. **`free_general` is frameless** (`alloc.rs`): `owner_heap`'s fallback — + `my_heap()`, which can create a heap — was a non-tail call inlined into + it, and that one call pinned a three-register frame on every general + free, remote ones included. The fallback is its own cold function + reached by a tail call: `xthread` −5.37, `huge` −6. +6. **`Heap::huge_alloc` is out of line**: `segment::huge_alloc` returns its + `Result` through a stack slot, and inlined, that slot put `sub`/`add + %rsp` on EVERY generic trip. All seven programs improved (−104,976 + allocator Ir). +7. **`calloc`'s sentinel test moved into its miss** (`alloc.rs::zalloc`): + the fast path is `malloc`'s raw-read shape, where the sentinel's empty + direct table already routes to the slow path. `calloc` −1.31; python + **−110,119 whole-program**. See the instrument note below — the + allocator-only column read +300,948 for this brick. +8. **The zero flag is a `u8` inside the generic chain** and a `bool` only at + `malloc_generic`: a `(*mut u8, bool)` pair return makes every caller + re-truncate the flag (`and $0x1,%dl`), which blocked the tail calls to + `grow_front` — the commonest generic outcome on a real program (jq: + 10,307 of 11,229 trips) — and to the walk. perl −67,053, jq −65,005, + python −41,375, lua −21,826 whole-program; `aligned` +0.06 (its slow + path converts at the `Heap` boundary). +9. **A medium miss grows the front page directly** (`heap.rs`): perl sent + 7,707 generic trips into the medium arm and 7,693 missed — the front + page dry but not fully carved — and then ran the heartbeat, derived the + same bin again and collected the same page again before reaching + `grow_front`. **perl −317,649, jq −593,325, sqlite −35,351 whole-program; + opscan `big`/`large` +1.00, `mixed` +0.66** — a recorded trade, real + programs being the verdict when the two instruments disagree in sign + (round one reverted a split on the same rule). The first version computed + `wsize_from_size` for `grow_front` and cost the hit path +5; `w` is now + passed as "above the direct table", the only thing `grow_front` reads it + for. +10. **`malloc_aligned_pow2`** for callers that have proven the alignment a + power of two — `posix_memalign` (which validates it for EINVAL) and + `GlobalAlloc` (whose `Layout` guarantees it): no second test. `aligned` + −2.00. + +**Refuted or not built this round:** `grow_front` as a tail call was first +blocked by the pair-return truncation (#8 fixed that); a running-pointer +`page_extend` is already a recorded refutation at the site; the span-marking +split and the `bins::bin` rewrite stay refuted from round one; `stats.extends` +stays live (release work-parity counter). + +**Instrument lesson — allocator-only Ir mis-attributes around a tail jump +into another object.** Brick 7 moved `calloc`'s `jmp *memset` from +`alloc::zalloc` into `alloc::calloc`; the per-object column then read python +**+300,948** while per-instruction totals (1,401,231 → 1,290,107 on the +calloc paths) and the whole-program count (−110,119) both said win. +callgrind's call-stack bookkeeping charges instructions differently when a +tail jump leaves the object. The gate now prints whole-program Ir beside the +allocator column, and a brick whose two columns disagree is decided by the +exact whole count. + +**Gates:** see the entry's commit. Also: the extended loom soak +(`LOOM_EXTENDED=1`, two remotes vs the abandoner) does NOT pass — on this +change or before it. It stops on loom's "Model exceeded maximum number of +branches" (`max_branches = 100_000`) after 46 minutes with the new model and +after 60 minutes with the committed one, so the overflow predates this +change; no assertion (lost block, use-after-free) fired in either. A run of +the new model with a tenfold budget was killed by the OS +(`STATUS_IN_PAGE_ERROR`, a full system drive) before a verdict. No CI +workflow runs loom at all; the test's comment claiming a nightly run was +wrong and has been corrected. Open item: find why one execution runs past +the budget (the two yield-spins on FREEING under a preemption bound are the +suspect) and make the soak pass. + +## CURIOSITY — ten instruction wins, seven refutations, and lua's allocator 40 % lighter (2026-09-24) + +Hunted with the `rusty-curiosity` discipline: profile, descend one layer +into the surprising number, read the siblings at the site, gate everything +on an exact instrument. Built on top of the YOU SLOW, BRO entry below. + +**Instruments.** Both exact, both in WSL on this box (callgrind 3.26, +rustc 1.97.1): the repo's `bench/opscan.c` two-point estimator, ra only, run +three times identical to the hundredth before anything was measured; and the +`icount-arms` real programs (perl and sqlite with the hash seed pinned, and +lua) reduced to allocator-only self Ir by object — identical to the +instruction across repeat runs, lua included. Per-instruction attribution +came from `callgrind --dump-instr` joined to `objdump`. Every brick was +built, scanned and put through the real-program gate before the next. + +**Cumulative, the tree at the start of this entry → the end:** + +| opscan op | before | after | Δ | +|---|---:|---:|---:| +| big / large | 121.00 | **101.00** | −16.5 % | +| mixed | 110.81 | **96.54** | −12.9 % | +| calloc | 107.57 | **94.82** | −11.9 % | +| huge | 689.00 | **638.00** | −7.4 % | +| med | 60.75 | 56.82 | −6.5 % | +| realloc | 259.86 | 248.37 | −4.4 % | +| small / small_touch | 54.39 / 60.39 | 52.27 / 58.27 | −3.9 % | +| aligned | 97.88 | 94.50 | −3.5 % | +| xthread, batch, liveset | | | −0.3 … −1.0 % | +| usable | 22.00 | 22.00 | 0 | + +| real program, allocator Ir | before | after | Δ | +|---|---:|---:|---:| +| **lua** | 15,781,865 | **9,485,488** | **−39.9 %** | +| perl | 7,124,383 | 7,002,677 | −1.7 % | +| sqlite | 1,418,197 | 1,389,797 | −2.0 % | + +**The ten, in order kept** (each measured alone against the one before): + +1. **`options::get` swapped an atomic on every read** (`options.rs`). The + `huge` op showed 19 Ir/op in `ensure_init` — `span_free` reads + `purge_delay` per span, and every read was a locked `swap` behind a call. + Load-first, swap in the cold arm: `huge` −17.00. +2. **A thread-local lookup hoisted above the test that guarded it** + (`options.rs`). `__tls_get_addr` on every slow-path allocation; the + disassembly loaded `DEFERRED_FUN` and called `__tls_get_addr` before + testing it. The re-entry guard moved into the cold hook caller: every op + improved, `huge` −20, `realloc` −3.44, `mixed` −2.50. +3. **`malloc_slow` had lost its tail call** (`heap.rs`). Its own doc says + "this arm must stay a TAIL call"; the later reclaim-and-retry wrapper + tested the result, which put the call back in non-tail position (16 Ir + around a jump). The test moved into `malloc_generic_once`'s epilogue: + `big`/`large` −7, `mixed` −5.06, `calloc` −4.69. +4. **`zalloc`'s miss pinned a frame on its hit path** (`heap.rs`). Out of + line and in tail position, the calloc fast path is a leaf ending in + `jmp memset`: `calloc` −4.56, perl −1,383. +5. **The keep-one-warm test was three branches** (`alloc.rs`, the `free` + asm label block). `used | next | prev == 0` is one test: −2.00 on every + alloc/free pair op, `realloc` −6.00. The first attempt edited + `retire_or_abort` — the function the comment names — and was flat; the + profile showed the hot copy inline in the label block. +6. **`realloc(NULL, n)` paid the move path's frame** (`alloc.rs`). Lua's + allocator routes EVERY allocation through `realloc`: 480,342 of 540,393 + calls had a null pointer, and each pushed and popped five registers to + reach `malloc`. The null test now sits in front of the frame, with the + live-pointer body in `realloc_live(NonNull, ..)` — `NonNull` because + the raw-pointer split let two null tests back in (+11.84 on opscan + `realloc`). **lua allocator −6,233,185 (−39.6 %)**, opscan `realloc` + −0.16. +7. **`span_free` read the purge option before testing the span length** + (`segment.rs`): `huge` −5, perl −924, lua −932, sqlite −296. +8. **A constant flag materialised and re-tested** (`heap.rs`, the medium + collect-and-retry). `(stole, pop)` as a tuple made LLVM merge the collect + outcomes, then `xor`/`test` a flag that is constant on the hit path. The + latch is now stored inside the branch that knows it: `big`/`large` −5, + `mixed` −3.31, perl −23,085, sqlite −2,727. +9. **The retry flag from #3 was a parameter** (`heap.rs`): a `mov $1` at the + call, a copy into a callee-saved register and a flag test before the null + test, on every generic trip. Now a thread-local read only on the OOM path: + every op improved (`huge` −6, `big`/`large` −4, `mixed` −2.89) and it + paid back #3's `aligned` +0.12; perl −56,937, lua −27,933, sqlite −10,020. +10. **`stats.delayed_frees` was a release counter** (`heap.rs`): a + read-modify-write per drained cross-thread block that only the stats + printer reads. Debug-only, like `allocs`/`frees`: `xthread` −0.77. + +**Refuted, each recorded at its site with its number:** + +- **Frameless medium entry** (split the generic path so the medium + collect-and-retry skips the four-register frame): opscan `big`/`large` + −11, `mixed` −6.83 — and every other generic trip +8, so real programs + went WORSE: lua +31,945, perl +23,549, sqlite +15,997. The opscan table + alone would have shipped it. +- **May-return abort in the collect walk**, to drop that entry's alignment + push: a call inside the loop keeps three values live; six-register frame, + `huge` +20. The trick only works in TAIL position. +- **`(w | 1).leading_zeros()`** in `bins::bin` to drop the zero-input + fallback: LLVM keeps the `leading_zeros` form; `big` +1, `aligned` +0.69. +- **Bin computed once** above both its uses in the generic path: live across + the heartbeat's calls, `big` +17, perl +101,962. +- **Owner-table `fill`** split out of the span-marking loop: both loops got + cheaper and the function dearer (two unroll setups and remainders); + `huge` +28 with a sign flip by span length on real programs. +- **A "table complete" flag** in `options::get` to skip the `i64::MIN` + sentinel: flat; the callers test the SIGN and LLVM already folds both into + one `js`. +- **Debug-only `stats.extends`** — not built: the release bench prints it + as a work-parity counter (`rusty_alloc_bench` `COUNTERS`, the wasm speed + probe), the same reason `stats.generic` stays live. Likewise inlining + `malloc_aligned` into the C shims, already refuted at +4.56 (2026-08-21). + +**Instrument lessons.** Twice a library that was never rebuilt read "+0.00 +on every op": first the 9p mtimes from a Windows edit, then `cargo` not on +`PATH` in a non-login shell, swallowed by `|| true`. The build helper now +refuses to report unless the library file is newer, and prints a code +checksum. And the per-instruction join is what found #2, #5, #6 and #8 — +each was invisible at function granularity. + +**Gates.** See the commit; `cargo test -p rusty_alloc` at default, +`debug_checks`, `secure`, `blockmap` and `--cfg ra_small_profile`; +`rusty_alloc-api` (incl. the re-entrancy gate) and `rusty_alloc-ffi`; +no_std `--cfg ra_single_threaded` check; clippy clean on the new code; +fmt; unsafe census 931 → 934 (`realloc_live`'s signature and block, and the +medium arm's collect/pop split into two blocks), each rowed in `UNSAFE.md`. +**Not measured:** wall clock (a Windows box at 60–75 % foreign load), and +mimalloc's arm — its opscan column is from earlier sessions. + +## YOU SLOW, BRO — a huge `realloc` copied its reservation, and 251 start-up allocations re-entered the allocator (2026-09-24) + +`docs/plans/youslowbro.md`, from `rusty_esp_sense` (a candle + rayon host +pipeline) measuring `rusty_alloc-api` 2.2.0 against mimalloc 0.1.52 and the +Windows heap on the patterns that pipeline produces: parity or better on 13 of +16 workloads, one real loss, one start-up cost, two memory notes. + +**Finding 1 — `realloc` that MOVES a huge block: 1.41x / 1.23x mimalloc, +0/6 pairs.** The report ruled out fresh huge allocation, reuse across sizes and +several live huge blocks with its own probes, and left four candidates. It was +B. `segment::huge_alloc` set the huge page's `block_size` to the whole +chunk-rounded reservation minus the header — a 33 MB request lives in a 64 MiB +chunk pair, a 64 MB one in 96 MiB — and `usable_size` is the length `realloc` +copies when it moves. Growing 33 MB therefore copied, and first-touched on both +sides, 64 MiB; growing 64 MB copied 96 MiB; `zalloc` on a recycled chunk +zeroed the same. Pinned by a test before anything was timed: +`tests/alloc_core.rs::huge_usable_size_is_the_request_not_the_reservation` +(the usable size of a 33 MiB and a 64 MiB block is within one slice of the +request, and a move preserves the whole of it). The fix is one line — +`block_size = align_up(size, SEGMENT_SLICE_SIZE).min(capacity)`, upstream's +`psize` — and the reservation is untouched, so nothing about placement, +arena recycling or `huge_free` moves. Not A: the copy was `memcpy` on both +sides. Not D: the paths were at parity in the report's own probe 3. + +**Static counts** (Windows x86-64 release, `cargo rustc --lib -- --emit asm`, +HEAD worktree vs tree, instructions per symbol): `alloc::realloc` 163 → 163, +`usable_size` 17 → 17, `free` 66 → 66, `page_extend` 42 → 42, `options::get` +20 → 20; `segment::huge_alloc` 199 → 203 (cold; the `min` and the round-up); +`options::ensure_init` 538 → 817 (runs once per process; the key-building loop +is now inline where `format!` and `to_uppercase` were calls into `std`). + +**Timing, on the consumer's own probe** (`F:/janus-data/raprobe`, copied to +scratch twice and patched through `[patch.crates-io]` to link the HEAD worktree +and the tree; a third copy links mimalloc). Method: whole processes pinned to +one core at High priority, ABBA with the leading arm swapped each round, +best-of-N inside each run, median and minimum of the per-run bests, paired +wins and z; the box sat at 62–76 % load from another process throughout, so +the **null arm** (one binary in both arms, 6 rounds) is the floor: median +ratios 0.89–1.03, no |z| ≥ 2. A verdict below needs |z| ≥ 2 AND a ratio +outside ±10 %. + +| workload | HEAD | tree | ratio | tree wins | z | +|---|---:|---:|---:|---:|---:| +| realloc step 33 → 66 MB | 7.707 ms | **5.243 ms** | **0.680** | **8/8** | +2.83 | +| realloc step 64 → 128 MB | 13.237 ms | **10.152 ms** | **0.767** | **8/8** | +2.83 | +| `Vec` growth by push to 64 MB × 3 | 45.201 ms | **36.673 ms** | **0.811** | **6/6** | +2.45 | +| realloc step 8 → 16 / 16 → 32 MB | | | 0.975 / 0.923 | 5/8, 6/8 | inside the floor | +| fresh 16–128 MB buffers, per buffer | | | 0.97–1.06 | 1–4/6 | inside the floor (path untouched) | + +Against mimalloc afterwards, same method, 8 and 6 rounds: **33 → 66 MB 1.013 +(5/8, z +0.71) — parity, was 1.407 at 0/6; 64 → 128 MB 0.969 (7/8, z +2.12) — +faster, was 1.233 at 0/6; `Vec` to 64 MB 1.015 (3/6) — parity, was 1.231 at +0/6.** Fresh 30–64 MB buffers stay at 1.02–1.06 (the report's 1.01–1.04): the +fix does not touch that path, and HEAD vs tree reads 3/6 there. + +**Finding 2 — 251 allocations through the global allocator on the first +allocation of every process.** `options::ensure_init` built 38 x 2 environment +keys with `to_uppercase` and `format!` and read them with `std::env::var`, +each an owned `String`, inside the heap being set up. Now: the key is built in +a stack buffer sized by the table (`KEY_CAP` is a `const` over +`OPTION_NAMES`), read through the new `prim::getenv` — `libc::getenv` on unix, +`GetEnvironmentVariableA` on Windows, the calls upstream's prim makes — into a +64-byte stack buffer, and parsed in place; Miri and wasm32-wasip1 keep a +`std::env` fallback, and wasm32-unknown-unknown keeps the pass compiled out so +the size ratchet is unmoved by construction. **On the consumer's counting +probe: 263 allocations at start-up → 12, and mimalloc reads 12 on the same +probe.** Two tests: `tests/options_env.rs` (a child process with both +prefixes, the precedence rule, the boolean grammar and KiB scaling set, read +back through `options::get`) and `rusty_alloc_api/tests/reentrancy.rs` (a +depth-counting `GlobalAlloc` around `RustyAlloc`: 395 calls, **0 re-entrant**; +it would have read 251). Unsafe census 928 → 931: one FFI block per prim +backend and one `set_var` in a test, each rowed in `UNSAFE.md`. + +**Not built, and why — `docs/opps.md` #10.** The old fat `usable_size` made a +`realloc` that fit inside the reservation's slack free, and that went with +the fix (upstream moves there too). Keeping it deliberately is a `HUGE_SEGMENT` +test in the move arm: a load and a branch on EVERY moving `realloc`, which is +the whole of the opscan `realloc` op, for a sub-2x grow of a block already +above 32 MiB that no report has named — the doublings consumers hit never fit. +Recorded with the hook, not built. + +**Memory notes (report §5).** README gained a *Memory* paragraph (freed memory +stays committed by default, `purge_delay` = -1, the retained working set and +how to purge). The consumer's `retain` probe reads identically before and +after — 461 MB retained after freeing 456 MB of huge blocks, 492 MB with 45 MB +of 224-byte blocks live, mimalloc 461 / 491 — and the mechanism is not a +missing recycler: the chunks come back through the arena bitmap and a fresh +segment takes one; what rises is first-touch of the pages a 38 MB block never +wrote in its 64 MiB pair. Backlog as `docs/opps.md` #11. + +**Gates.** `cargo test -p rusty_alloc` 144 passed / 0 failed (default), plus +`--features debug_checks` and `--features secure`; `rusty_alloc-api` tests +including the new re-entrancy gate; clippy clean on the new code (the tree +also carries `unexpected_cfgs` warnings for an unregistered `ra_small_wsize` +knob from uncommitted work outside this campaign, in `types.rs` and +`prim/fixed.rs`); `cargo fmt --check`; `tools/unsafe-census.sh --update`. +**Not measured:** callgrind — this is a Windows box; the icount farm re-reads +the opscan `realloc` and `huge` ops, where the static counts above predict +0 and +4. + ## OPENHEIMER, SPLIT — 202 findings, 139 withdrawn as hot-path tax; the one fix the log claimed and never wrote (2026-09-16) `docs/plans/openheimer-run.md`, from the campaign's red log diff --git a/docs/opps.md b/docs/opps.md index 7598722..54e1045 100644 --- a/docs/opps.md +++ b/docs/opps.md @@ -32,6 +32,8 @@ are to try. | 7 | `wait_no_remote_in_flight` 512-slot scan | 3 | latency | **LANDED — bounded to the carved region** | | 8 | Collect-loop double `block_next` | — | batch loss | **banked (landed)** | | 9 | `process_delayed` swaps an empty list every slow-path alloc | 3 | slow path | **LANDED — load-before-swap** | +| 10 | Huge `realloc` grows in place inside the reservation's slack | 3 | huge realloc | **NOT BUILT (2026-09-24) — no measured workload reaches it; a load and a branch on every moving realloc** | +| 11 | Recycle retained huge chunks' untouched tail for small objects | 3 | footprint | **OPEN — backlog from `docs/plans/youslowbro.md` §5** | --- @@ -309,6 +311,47 @@ so it would read as `aligned` regressing in every opscan. Not shipped; `page_collect`'s xthread steal was already load-guarded before it CASes, so #9 was the one place the pattern actually paid. +## From a consumer report — `docs/plans/youslowbro.md` (2026-09-24) + +### 10. Huge `realloc` in place inside the reservation — NOT BUILT + +**Where:** `alloc.rs::realloc`'s move arm; `segment.rs::huge_alloc`. + +A huge block's reservation is chunk-rounded — 64 MiB for a 33 MB request, 96 +MiB for 64 MB — so up to a chunk of committed, never-touched slack sits after +the block. Until 2026-09-24 the page REPORTED that slack as its usable size, +which made a `realloc` that fit inside it free and made every `realloc` past +it copy the whole slack: the 1.41x / 1.23x loss `youslowbro.md` §3 measured. +The fix reports the request (upstream's `psize`), and the in-place growth +went with it. + +It could be kept deliberately: in the move arm, if the page is `HUGE_SEGMENT` +and `newsize <= total_size - (block - seg)`, bump `block_size` and return +`p`. The arithmetic against building it now: the growth that consumers hit +is a `Vec` doubling (33 -> 66 MB, 64 -> 128 MB), which never fits the slack +(capacity 64 MiB - 64 KiB and 96 MiB - 64 KiB respectively), so the only +beneficiary is a sub-2x grow of a block already above 32 MiB — a pattern no +report has named — while the test is a flags load and a branch on EVERY +moving `realloc`, which is the whole of the `realloc` opscan op (277 Ir/op, +all moves). Not built. If a workload with that shape appears, price it on +opscan `realloc` first; the hook is a `#[cold]` arm after the in-place test, +and `total_size` already holds the capacity. + +### 11. Recycle a retained huge chunk's untouched tail for small objects — OPEN + +**Where:** `arena.rs` chunk bitmap; `segment.rs::segment_alloc`. + +`youslowbro.md` §5: after 456 MB of huge blocks are freed, 45 MB of 224-byte +blocks raise the working set from 461 to 494 MB — mimalloc reads 491, so it +is not a regression against upstream. The chunks DO come back through the +arena bitmap and a fresh segment takes one, so the recycling exists; what +rises is first-touch of the pages inside those chunks that the huge block +never wrote (a 38 MB block leaves 26 MB of its 64 MiB pair untouched). A +segment carved from a recycled chunk could prefer the touched prefix, or the +arena could hand a 64 MiB pair's touched half out first. Backlog: needs the +`retain` probe from the consumer's harness as its instrument, and a working +set number, not an instruction count, as its verdict. + ## Banked ### 8. Collect-loop double `block_next` — LANDED 2026-08-20 diff --git a/docs/plans/fast-trans.md b/docs/plans/finished/fast-trans.md similarity index 100% rename from docs/plans/fast-trans.md rename to docs/plans/finished/fast-trans.md diff --git a/docs/plans/finished/youslowbro.md b/docs/plans/finished/youslowbro.md new file mode 100644 index 0000000..d4b4216 --- /dev/null +++ b/docs/plans/finished/youslowbro.md @@ -0,0 +1,372 @@ +# You slow, bro? Where rusty_alloc loses, measured from a real consumer + +**Status: RESOLVED 2026-09-24 — see §8.** Both findings fixed and measured on +this report's own probe: the huge `realloc` steps went from 1.41x / 1.23x +mimalloc at 0/6 to 1.01x (parity) and 0.97x (7/8), and start-up allocations +from 263 to 12, which is mimalloc's own count on the same probe. Candidate B +was the mechanism. §5 got its README line and a backlog entry. + +Originally: measured, nothing implemented. · **Date:** 2026-09-24 · +**Build:** `rusty_alloc-api` **2.2.0 from crates.io** (not this repo's HEAD, +which has uncommitted changes that were left untouched) · **Box:** Windows 11 +Home, Intel i7-14650HX, 16 cores / 24 threads, 32 GB, **native Windows, not +WSL** · **Reference:** `mimalloc` 0.1.52 (`libmimalloc-sys` 0.1.49, built +from its C source) and the Windows system heap + +This note comes from outside the team. `rusty_esp_sense`, a Janus host +pipeline built on candle and rayon, adopted rusty_alloc as its binary's +global allocator. Its benchmark got **7.5 % faster** (1,945 → 1,759 ms, 15/16 +alternating pairs, z 3.50). While pricing that change we measured rusty_alloc +directly against mimalloc and the system heap on the allocation patterns +that pipeline produces. The short version: **you are not slow in general.** +There is one real loss, one start-up cost, and two memory notes worth +knowing. + +--- + +## 0. The answer in one paragraph + +Against mimalloc, rusty_alloc is at parity or better on 13 of the 16 +workloads we ran, and clearly faster on three of them (cross-thread free of +224 B blocks 0.73×, 24-thread churn of 4 KB blocks 0.77×, repeated fresh +blocks below 32 MiB 0.81-0.85×). **The one real loss is `realloc` that +moves a HUGE block.** Growing a block that is already larger than a segment +costs **1.41×** mimalloc's time for 33 → 66 MB and **1.23×** for +64 → 128 MB, and loses every pair (0/6). The same step below the segment +size is *faster* than mimalloc (0.87×, 0.93×). A realistic `Vec` that grows +to 64 MB by `push` pays it: **1.23×, 0/6**. Three mechanisms are already +ruled out (§3). Separately, **the options pass makes 251 allocations through +the global allocator** on the first allocation of every process, where +mimalloc makes none, and its source line is identified (§4). + +--- + +## 1. Method, and what makes a number admissible here + +- **One probe source, three binaries.** A small Rust program runs each + workload best-of-N and prints milliseconds. It is compiled three times: as + `#[global_allocator]` it uses `rusty_alloc_api::RustyAlloc`, + `mimalloc::MiMalloc` or `std::alloc::System`. All three sit behind the same + thin counting wrapper, so the counts in §4 are like for like. Source: + `F:/janus-data/raprobe/`, `cargo build --release --features rusty|mi`. +- **Alternating runs.** The binaries run in turn, 5 or 6 rounds each. The + verdict is the median of each arm's per-round best and the **paired win + count**. A cell below is a verdict only if one side won at least 5 of 6 + (or 5 of 5). +- **Wall clock only, no instruction counts.** This is a Windows host with no + callgrind. Every ranking here should be re-read on your deterministic + instrument before anything is built. +- **The box was loaded.** Another process held the CPU at up to 100 % + during these runs. Pairing cancels drift that both arms share; it does not + make small effects admissible. Read anything within ±5 % that lacks a + 5/6 sweep as parity. + +--- + +## 2. The numbers + +### Speed against mimalloc (ratio = rusty_alloc / mimalloc; below 1 means rusty_alloc is faster) + +| workload | mimalloc | rusty_alloc | ratio | rusty faster | +|---|---:|---:|---:|---:| +| 5,900 × 224 B kept, then freed | 0.245 ms | 0.228 ms | 0.931 | 4/5 | +| 5,900 × 11.2 KB kept, then freed | 10.6 ms | 10.8 ms | 1.017 | 2/5 | +| fresh 1 MB × 50 | 2.47 ms | 2.51 ms | 1.016 | 2/5 | +| fresh 8 MB × 20 | 8.51 ms | 8.37 ms | 0.983 | 4/5 | +| Vec growth by push to 8 MB × 10 | 10.3 ms | 10.5 ms | 1.020 | 1/5 | +| cross-thread free, 24 × 20k × 224 B | 17.4 ms | 12.6 ms | **0.725** | 5/5 | +| cross-thread free, 24 × 500 × 64 KB | 18.6 ms | 18.7 ms | 1.004 | 3/5 | +| 24-thread churn, 64 B | 70.0 ms | 70.9 ms | 1.012 | 3/5 | +| 24-thread churn, 4 KB | 45.1 ms | 34.9 ms | **0.774** | 5/5 | +| fresh 16/30 MB alternating (below a segment) | 12.8 ms | 10.9 ms | **0.846** | 6/6 | +| **realloc step 8 → 16 MB** | 0.726 ms | 0.628 ms | 0.865 | 6/6 | +| **realloc step 16 → 32 MB** | 1.578 ms | 1.466 ms | 0.929 | 5/6 | +| **realloc step 33 → 66 MB** | 4.552 ms | 6.405 ms | **1.407** | **0/6** | +| **realloc step 64 → 128 MB** | 8.489 ms | 10.466 ms | **1.233** | **0/6** | +| **Vec growth by push to 64 MB × 3** | 30.2 ms | 37.2 ms | **1.231** | **0/6** | +| fresh huge 33-128 MB, per buffer | 2.06-8.83 ms | 2.14-9.08 ms | 1.01-1.04 | 1-2/6 | + +Against the Windows system heap, rusty_alloc wins every workload above, from +0.05× to 0.67×. For a Windows consumer that is the headline, and it is why +`rusty_esp_sense` ships it. + +### Memory (working set from `K32GetProcessMemoryInfo`) + +| probe | system | mimalloc | rusty_alloc | +|---|---:|---:|---:| +| 12 × 38 MB live | 460 MB | 461 MB | 461 MB | +| …all freed | **4 MB** | 461 MB | 461 MB | +| …500 ms later | 4 MB | 461 MB | 462 MB | +| then 45 MB of 224 B blocks live | 54 MB | 491 MB | 494 MB | + +--- + +## 3. Finding 1: realloc that MOVES a huge block is 1.23-1.41× mimalloc + +**What is established:** + +- It is specific to growing a block that is **already huge**, larger than a + 32 MiB segment. The same step entirely below the segment size is faster + than mimalloc (8 → 16 MB 0.865×, 16 → 32 MB 0.929×, 6/6 and 5/6). +- It is not the policy. `alloc.rs::realloc` keeps a block in place only + when `newsize <= usable && newsize >= usable / 2`, and otherwise mallocs, + copies and frees. That is upstream's generic shape too. + +**Ruled out**, each with its own single-variable probe at 6 pairs: + +1. **Fresh huge allocation is not the cost.** Fresh 33, 38, 64 and 128 MB + blocks, written and freed, run 1.01-1.04× (1-2/6): near parity, at most a + few percent. +2. **Reuse of freed huge space across sizes is not the cost.** Fresh blocks + alternating 33/66 MB run 1.013× (2/6), against 0.996× for one size + repeated. +3. **Several live huge blocks freed together is not the cost.** Two live + (33 + 66 MB) run 0.964× (4/6), and four live 0.979× (4/6). That is a + realloc's footprint without the realloc, and it runs at parity. + +So the gap lives inside `realloc`'s own moving path for a huge block. + +**Candidates, in the order we would test them:** + +- **A. The copy itself.** The move is `copy_nonoverlapping(p, np, usable.min(newsize))` + over 33-64 MB, which on Windows is the MSVC runtime's `memcpy`. Upstream + copies through `_mi_memcpy`. Time the two copies alone at 33 MB and 64 MB + into a freshly committed destination. If rusty's copy is slower there, the + fix is a copy strategy chosen by size: `rep movsb` where the CPU has ERMS + or FSRM, and no non-temporal stores into a destination that is about to be + written again. +- **B. How much is copied.** `usable.min(newsize)` copies the whole + *usable* size of the old block. If a huge block's usable size is rounded + well past what was requested (to a slice or segment granularity), the + copy moves more bytes than the caller ever wrote. Compare + `usable_size(p)` with the requested size for 33 MB and 64 MB blocks, and + compare with upstream's `mi_usable_size`. +- **C. Growing a huge block where it stands.** If the huge segment's + reservation, or the address space right after it, has room, a huge block + can grow by committing more pages instead of moving. That removes the copy + and the fresh destination's page faults together. Check whether upstream + 2.2.x does anything of the kind for huge pages on Windows before deciding + it is a feature gap rather than a defect. +- **D. The huge-block paths realloc takes on the way.** `usable_size` takes + `usable_size_slow` for huge pages, and `free_inline` frees a huge segment. + Both are cheap in isolation (probe 3 frees huge blocks at parity), but + count them on your instrument inside `realloc` for completeness. + +**Why it matters to consumers:** a `Vec` that grows past 32 MiB by `push` +(buffered file reads, collecting a large result) crosses exactly this path +once per doubling above the segment size. That is 1.23×, 0/6, measured. + +--- + +## 4. Finding 2: 251 start-up allocations re-enter the global allocator + +On the first allocation of a process, rusty_alloc makes **251 allocations +through the global allocator** before returning. mimalloc and the system +heap make **none**. The same probe, behind the same counting wrapper, reads +261 against 10. In `rusty_esp_sense` this showed up as a constant +251 +allocations (+9 KB) in every command's census, `run` included. + +**Mechanism** (`crates/rusty_alloc/src/options.rs`, the `std` environment +pass, around lines 352-356): + +```rust +for i in 0..OPTION_COUNT { // 38 options + let name = OPTION_NAMES[i].to_uppercase(); // a String + let val = std::env::var(std::format!("RUSTY_ALLOC_{name}")) // a String + env lookup + .or_else(|_| std::env::var(std::format!("MIMALLOC_{name}"))) // another pair + ... +``` + +That is 38 options × about 6.6 allocations each: the uppercase name, two +formatted keys, and two `std::env::var` calls. On Windows each call converts +the key to UTF-16 and the value back, which allocates too. The total is 251. + +**Why it is worth fixing, beyond the count:** + +- It is allocator re-entrancy during initialisation. Every one of those + allocations lands in the heap that is still being set up. +- It is a start-up cost paid by every process, including the short CLI runs + that are most of a command-line tool's life. +- You already deleted this same pass on `no_std` and on + `wasm32-unknown-unknown`, for size. This is the same problem on hosted + targets, for time and re-entrancy. + +**Fixes, cheapest first:** + +1. **Precompute the uppercase names** as a `const` table beside + `OPTION_NAMES`, and build each key in a stack buffer with no `format!`. + That removes about three of the six or seven allocations per option. +2. **Read the environment once, not 76 times.** Walk the environment block + a single time (on Windows `GetEnvironmentStringsW`, which the `windows` + primitives can already reach) and match each entry against the two + prefixes. Parse only the entries that match, usually none. That takes + the pass from 76 lookups to one walk, and from about 251 allocations to + zero. +3. **Or make the pass lazy.** Only the options actually read on the hot + path need the environment at first allocation; the rest can resolve on + their first `get`. + +The deterministic check: the counting-wrapper probe should read 10 at start-up, +the same as mimalloc. + +--- + +## 5. Two memory notes (shared with upstream, not defects against it) + +- **Freed memory is not returned, by design.** `purge_delay` defaults to -1 + (purging is opt-in). After 456 MB of huge blocks are freed, the working + set stays at 461 MB, and at 462 MB 500 ms later. mimalloc behaves the same + on this box. The system heap returns to 4 MB. In `rusty_esp_sense` the + peak working set rose 25-40 %: 115 → 144 MB single-threaded, 509 → 717 MB + for the raw-window run at 24 threads. That is a fair trade, and the + consumer ships it behind a feature flag, but a short "memory will look + larger" line in the README would save the next adopter a surprise. +- **Freed huge segments are not recycled for small objects.** After those + 456 MB are freed, 45 MB of 224 B blocks raise the working set to 494 MB, + where the freed space could have held them. mimalloc does the same (491 MB) + so this is not a regression against upstream. It is the obvious next step + if the team ever wants to beat upstream on footprint as well as speed: + carve small-object segments out of retained huge ones. + +--- + +## 6. Where you win, for the record + +- **Cross-thread free of small blocks: 0.725× mimalloc, 5/5.** This is + rayon's collect-then-drop pattern: workers allocate and the main thread + frees. It is the pattern a data-parallel pipeline hits most. +- **24-thread churn of 4 KB blocks: 0.774×, 5/5.** +- **Repeated fresh blocks below a segment: 0.81-0.85×, 6/6.** +- **Against the Windows system heap: 0.05-0.67× on every workload**, and + 7.5 % off a whole candle + rayon benchmark end to end. + +--- + +## 7. What we would do first + +1. **§4, the options pass.** It is small, deterministic to verify (start-up + allocations 261 → 10) and removes allocator re-entrancy during + initialisation. +2. **§3, candidates A and B**, on the deterministic instrument: price the + huge copy and the bytes it moves. Only if both come back at parity, look + at C, growing huge blocks in place. +3. **§5**, a README line, and a backlog entry for recycling retained huge + segments. + +--- + +## 8. Executed (2026-09-24) + +Both fixes were built against the tree at `c9631f4` and measured on this +report's own probe (`F:/janus-data/raprobe`, copied to scratch twice and +patched with `[patch.crates-io]` to link a worktree of that commit and the +fixed tree; a third copy links mimalloc 0.1.52). The full record, with the +method line and every raw per-round value, is the LEDGER entry of the same +date; this section is the report's own questions, answered in its own order. + +### §3 — it was candidate B + +- **A, the copy itself: no.** Both sides call the platform `memcpy`; nothing + about the copy changed and nothing needed to. +- **B, how much is copied: yes, and it was twice what the caller wrote.** + `segment::huge_alloc` reserves in whole 32 MiB chunks through the arena — + a 33 MB request holds a 64 MiB chunk pair, a 64 MB one holds 96 MiB — and + then set the page's `block_size` to that whole reservation minus the + header. `usable_size(p)` is the length `realloc` copies when it moves, so + growing 33 MB copied and first-touched 64 MiB on both sides, and growing + 64 MB copied 96 MiB. The report's arithmetic already fit that shape: the + two steps below the segment size have no such slack and were faster than + mimalloc. The page now reports `align_up(size, SEGMENT_SLICE_SIZE)` + capped at the reservation — upstream's `psize`, and what `mi_usable_size` + returns there. The reservation itself is unchanged. `zalloc` on a + recycled chunk zeroed the same extent and is fixed by the same line. +- **C, growing in place: not built, recorded.** The fat usable size *was* an + in-place grow for anything that fit inside the slack; the fix removes it, + as upstream never had it. Keeping it deliberately costs a flags load and a + branch on every moving `realloc` (the whole of the deterministic `realloc` + op) for a sub-2x grow of a block already above 32 MiB, which the doublings + a `Vec` performs never are. `docs/opps.md` #10 has the hook and the + arithmetic. +- **D, the paths on the way: no.** `usable_size_slow` and `free_inline` on + a huge block are unchanged; the report's probe 3 had already priced them + at parity. + +Deterministic check first (`tests/alloc_core.rs::huge_usable_size_is_the_request_not_the_reservation`): +usable of a 33 MiB block was 67,043,328 bytes (64 MiB − 64 KiB) and is now +within one slice of 34,603,008; a 64 MiB block reported 96 MiB − 64 KiB and +now reports 64 MiB. Static instruction counts from the emitted x86-64 assembly: +`realloc`, `usable_size`, `free` and `page_extend` byte-for-byte the same +length (163 / 17 / 66 / 42); `huge_alloc` 199 → 203, cold. + +Then the clock, on this probe, pinned to one core at High priority, whole +processes ABBA-alternated with the leading arm swapped each round, median of +the per-run bests, paired wins, z. The box was at 62–76 % load from another +process, so a null arm (one binary in both arms, 6 rounds) set the floor: +median ratios 0.89–1.03, no |z| ≥ 2. + +| workload | 2.2.0 tree | fixed | ratio | wins | z | +|---|---:|---:|---:|---:|---:| +| realloc step 33 → 66 MB | 7.707 ms | **5.243 ms** | **0.680** | **8/8** | +2.83 | +| realloc step 64 → 128 MB | 13.237 ms | **10.152 ms** | **0.767** | **8/8** | +2.83 | +| Vec growth by push to 64 MB × 3 | 45.201 ms | **36.673 ms** | **0.811** | **6/6** | +2.45 | +| realloc step 8 → 16 / 16 → 32 MB | | | 0.975 / 0.923 | 5/8, 6/8 | floor | +| fresh 16–128 MB, per buffer | | | 0.97–1.06 | 1–4/6 | floor | + +And against mimalloc, which is what §2 asked: + +| workload | §2 (before) | now | wins now | z | +|---|---:|---:|---:|---:| +| realloc step 33 → 66 MB | 1.407, 0/6 | **1.013** | 5/8 | +0.71 (parity) | +| realloc step 64 → 128 MB | 1.233, 0/6 | **0.969** | 7/8 | +2.12 (faster) | +| Vec growth by push to 64 MB × 3 | 1.231, 0/6 | **1.015** | 3/6 | 0.00 (parity) | +| fresh 30–64 MB, per buffer | 1.01–1.04 | 1.02–1.06 | 0–3/6 | unchanged: the fix does not touch this path (tree vs tree 3/6) | + +### §4 — zero allocations, the same 76 lookups + +Fixes 1 and 2 from the list, combined and without the single environment +walk: the key is built by hand in a stack buffer whose size is a `const` over +the table (`RUSTY_ALLOC_` + the longest name + NUL), the value lands in a +64-byte stack buffer through a new `prim::getenv` — `libc::getenv` on unix and +`GetEnvironmentVariableA` on Windows, the two calls upstream's own prim makes +— and the value grammar is parsed on those bytes in place. One lookup per +prefix per option as before, 76 in all, and none of them owns memory. The +Windows API converts the name to UTF-16 on the process heap, which a +counting wrapper on the global allocator does not and should not see. +wasm32-unknown-unknown keeps the pass compiled out (its size ratchet is +unmoved by construction); Miri and wasm32-wasip1 keep a `std::env` fallback. + +The deterministic check this section asked for, on this probe's counting +wrapper: **`startup` read 263 on 2.2.0 and reads 12 now; mimalloc reads 12.** +Two tests keep it: `tests/options_env.rs` sets both prefixes, a precedence +conflict, `YES` and a KiB size in a child process and reads them back through +`options::get`; `rusty_alloc_api/tests/reentrancy.rs` wraps `RustyAlloc` in a +depth-counting `GlobalAlloc` and fails if the allocator ever allocates through +the global allocator while serving a request (it read 0 of 395 calls; it +would have read 251). + +### §5 — the README line, and what the backlog entry says + +The README's Features section has a *Memory* paragraph: freed memory stays +committed by default (`purge_delay` = -1, as upstream ships), the working set +reads near its past peak, and the two environment variables or +`options::set(15, …)` that turn purging on. This probe's `retain` reads the +same before and after — 461 MB retained, 492 MB with the 224 B blocks live — +because nothing here touched retention. + +The second note is backlog `docs/opps.md` #11, with one correction to the +report's framing: the freed huge chunks DO come back through the arena bitmap +and a fresh small-object segment takes one, so a recycler is not what is +missing. What raises the working set is first-touch of the pages the huge +block never wrote (a 38 MB block leaves 26 MB of its 64 MiB pair untouched). +Preferring the touched prefix when carving is the lever, and its instrument +is this probe's working-set number, not an instruction count. + +### Gates + +`cargo test -p rusty_alloc` 144/0 at default features, plus `debug_checks` +and `secure`; the `rusty_alloc-api` suite with the new re-entrancy test; +clippy clean on the new code; `cargo fmt --check`; `tools/unsafe-census.sh +--update` (928 → 931: one FFI block per prim backend, one `set_var` in a +test, each rowed in `UNSAFE.md`). Not run here: callgrind (no Linux box) — +the icount farm re-reads the `realloc` and `huge` opscan ops, where the +static counts predict 0 and +4. diff --git a/tools/corpus/README.md b/tools/corpus/README.md index ea8cd57..f4889bd 100644 --- a/tools/corpus/README.md +++ b/tools/corpus/README.md @@ -52,7 +52,26 @@ rusty_alloc-api = { version = "2", default-features = false, features = ["std"] With that applied to the copy, the `rusty_alloc` error is gone. -## Four things this harness got wrong before it got anything right +## Run of 2026-09-25 (`--test`, after CURIOSITY rounds three and four) + +The first run to verify, per row, that the candidate resolved THIS tree. + +| consumer | result | +|---|---| +| `spacedb-sdk` | PASS | +| `spacedb-sdk` (`secure`) | PASS | +| `rusty_alloc_default` | PASS | +| `rusty_zstd` (through the patched shim) | PASS | +| `spacedb (published mirror)` (`F:/coding/spacedb`, new) | PASS | +| `rusty_maplibre` | FAIL, the known 2.0.0 `no_std` break. With the documented `features = ["std"]` migration applied to the copy: **1,199 tests passed, 0 failed** | + +Beyond the harness, the same day (see `docs/LEDGER.md`, DOWNSTREAM CORPUS; +the Silesia check is `tools/corpus/silesia.sh`): +both SpaceDB and `rusty_zstd` workspaces ran their whole test suites on Linux +with this tree's allocator `LD_PRELOAD`ed into every test binary, and Silesia +went through `rusty_zstd`'s CLI against C zstd 1.5.7 on Windows and Linux. + +## Seven things this harness got wrong before it got anything right Recorded because each one made it lie, and a corpus that lies is worse than none — it gets muted. @@ -77,6 +96,24 @@ Recorded because each one made it lie, and a corpus that lies is worse than none true FAIL. **If a run shows fork or cygheap errors, re-run the affected consumer in isolation before believing its row.** +5. **A PASS can test the wrong allocator.** Two ways, both found on + 2026-09-25, and the first run's `rusty_zstd` PASS was one of them. + `rusty_zstd` names no `rusty_alloc*` crate; it reaches us through the + published `rusty_alloc_default` shim, which pins the crates.io allocator, + so its candidate never built this tree. And with `TMPDIR` inside this + repository, `repoint` skipped every manifest (its `/target/` exclusion + matched the copy's own path), so SpaceDB's candidates built crates.io + 1.1.6 and PASSED. The shim is now patched to a repointed local copy, the + script refuses a work dir inside the repo, and **every candidate's + `Cargo.lock` is checked: a registry `rusty_alloc` in it is reported + NOT-REPOINTED, never PASS.** +6. **A work dir inside this repository breaks copies that have no workspace + root:** cargo walks up into this repo's `[workspace]` and refuses to build + (`rusty_alloc_default` read FAIL for that reason alone). +7. **`{ workspace = true }` entries must be left alone.** `repoint` added a + `path` to `rmap-alloc`'s inherited dependencies; the root is rewritten on + its own. + ## What it still does not do Compile gates only, unless `--test` is passed. It does not measure downstream diff --git a/tools/corpus/corpus.toml b/tools/corpus/corpus.toml index 599ac4e..785ed11 100644 --- a/tools/corpus/corpus.toml +++ b/tools/corpus/corpus.toml @@ -49,3 +49,13 @@ name = "rusty_maplibre" path = "F:/coding/rusty_maplibre" features = "" note = "KNOWN RISK: takes rusty_alloc AND rusty_alloc-api with default-features = false, which is exactly what 2.0.0 redefines" + +[[consumer]] +name = "spacedb (published mirror)" +path = "F:/coding/spacedb" +pkg = "spacedb-sdk" +features = "" +# The standalone checkout `release/release.sh` generates from mata-master's +# packages — the tree crates.io receives. Registered separately because it is +# what ships, and it carries its own workspace root (no synth needed). +note = "generated mirror of packages/spacedb; rusty_alloc-api =1.1.6, default-on" diff --git a/tools/corpus/run.sh b/tools/corpus/run.sh index 9a5a4c0..109ac98 100644 --- a/tools/corpus/run.sh +++ b/tools/corpus/run.sh @@ -23,6 +23,18 @@ set -uo pipefail root="$(cd "$(dirname "$0")/../.." && pwd)" work="${TMPDIR:-/tmp}/ra-corpus" +# The copies must live OUTSIDE this repository, for two reasons that each made +# a run lie (2026-09-25, with TMPDIR under this repo's `target/`): `repoint` +# skips every path containing `/target/`, so no manifest was rewritten and the +# candidates quietly built the crates.io allocator and PASSED; and a copy with +# no workspace root of its own walked up into this repo's `[workspace]` and +# failed to build at all. +mkdir -p "$work" +case "$(cd "$(dirname "$work")" 2>/dev/null && pwd -P)/" in + "$(cd "$root" && pwd -P)"/*) + echo "refusing: work dir $work is inside $root -- set TMPDIR elsewhere" >&2 + exit 2 ;; +esac mode="check" [ "${1:-}" = "--test" ] && mode="test" @@ -47,6 +59,11 @@ orig = s # `name = { version = "...", ... }` -> keep the rest, swap in a path def patch(m): name, body = m.group(1), m.group(2) + # `{ workspace = true }` inherits from the root, which is rewritten on its + # own; adding a `path` here makes an entry cargo should reject (seen on + # rusty_maplibre's rmap-alloc, 2026-09-25). + if re.search(r'\bworkspace\s*=\s*true', body): + return m.group(0) tgt = api if 'api' in name else alloc body = re.sub(r'version\s*=\s*"[^"]*"\s*,?\s*', '', body) body = body.strip().strip(',').strip() @@ -62,6 +79,44 @@ PY done } +# Consumers that reach us THROUGH the `rusty_alloc_default` shim (rusty_zstd) +# name no `rusty_alloc*` crate themselves, so `repoint` leaves them on the +# published shim — and the published shim pins the crates.io allocator. Until +# 2026-09-25 their candidate therefore tested the release, not this tree. Patch +# the shim to a copy of the local checkout that IS repointed. A +# `[patch.crates-io]` is legal here: the local shim carries the same version. +SHIM_SRC="${RA_SHIM_SRC:-$root/../rusty_alloc_default}" +patch_shim() { + local tree="$1" + grep -rqsE --include=Cargo.toml --exclude-dir=target '^rusty_alloc_default\s*=' "$tree" || return 0 + # The shim is a consumer in its own right; it is repointed directly. + grep -qE '^name\s*=\s*"rusty_alloc_default"' "$tree/Cargo.toml" 2>/dev/null && return 0 + local shim="$work/_shim" + if [ ! -d "$shim" ]; then + [ -d "$SHIM_SRC" ] || { echo " (no $SHIM_SRC to patch the shim with)" >&2; return 0; } + mkdir -p "$shim" + (cd "$SHIM_SRC" && tar -cf - --exclude=./target --exclude=./.git .) | (cd "$shim" && tar -xf -) + repoint "$shim" + fi + local p; p="$(cygpath -m "$shim" 2>/dev/null || echo "$shim")" + if grep -q '^\[patch.crates-io\]' "$tree/Cargo.toml"; then + sed -i "/^\[patch.crates-io\]/a rusty_alloc_default = { path = \"$p\" }" "$tree/Cargo.toml" + else + printf '\n[patch.crates-io]\nrusty_alloc_default = { path = "%s" }\n' "$p" >> "$tree/Cargo.toml" + fi +} + +# Did the candidate actually build THIS tree? A registry `source` on any +# rusty_alloc crate in its lock means no — and such a row used to PASS. +resolved_from_registry() { + local lock="$1/Cargo.lock" + [ -f "$lock" ] || return 1 + awk '/^name = "rusty_alloc(-api)?"$/ { n = 1; next } + n && /^source = "registry/ { bad = 1 } + /^$/ { n = 0 } + END { exit !bad }' "$lock" +} + # Some consumers live in a workspace whose ROOT is not in this checkout -- # `packages/spacedb/*` inherit their sibling dependencies with # `workspace = true`, and mata-master carries no manifest above them. Without a @@ -141,10 +196,18 @@ run_one() { copy_tree "$dst" [ "${synth:-}" = "yes" ] && synth_workspace "$dst" repoint "$dst" + patch_shim "$dst" local cand_out cand_rc cand_out="$(cd "$dst" && cargo "$mode" --quiet "${fargs[@]}" 2>&1)" cand_rc=$? + if resolved_from_registry "$dst"; then + echo " NOT-REPOINTED the candidate resolved rusty_alloc from crates.io -- this row tests nothing" + ROWS+=("NOT-REPOINTED|$name|candidate still built the crates.io allocator") + fail=$((fail + 1)) + return + fi + if grep -q "failed to find a workspace root" <<<"$base_out"; then echo " SKIP workspace root absent from this checkout -- cannot answer here" ROWS+=("SKIP|$name|workspace root not in this checkout") diff --git a/tools/corpus/silesia.sh b/tools/corpus/silesia.sh new file mode 100644 index 0000000..59d5a49 --- /dev/null +++ b/tools/corpus/silesia.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash +# Silesia under rusty_zstd's CLI, published allocator (BASE) vs this tree (CAND), +# with C zstd 1.5.7 as the independent oracle. Per file and level: +# 1. CAND compresses byte-identically to BASE (an allocator must not change output) +# 2. CAND decodes its own frame to the original +# 3. C zstd decodes CAND's frame to the original +# 4. CAND decodes C zstd's frame to the original +# plus the same with 4 worker threads (-T4), which exercises cross-thread frees. +# usage: silesia.sh +set -u +B="$1"; C="$2"; O="$3"; S="$4"; W="$5"; mkdir -p "$W" +pass=0; fail=0 +bad() { echo "FAIL $*"; fail=$((fail + 1)); } +for f in "$S"/*; do + n=$(basename "$f") + for lv in 1 3 9 19; do + "$B" -q -$lv -c "$f" > "$W/b.zst" || { bad "$n -$lv base compress"; continue; } + "$C" -q -$lv -c "$f" > "$W/c.zst" || { bad "$n -$lv cand compress"; continue; } + cmp -s "$W/b.zst" "$W/c.zst" || bad "$n -$lv compressed bytes differ from base" + "$C" -q -d -c "$W/c.zst" > "$W/c.out" && cmp -s "$f" "$W/c.out" || bad "$n -$lv cand->cand roundtrip" + "$O" -q -d -c "$W/c.zst" > "$W/o.out" && cmp -s "$f" "$W/o.out" || bad "$n -$lv cand->C decode" + "$O" -q -$lv -c "$f" > "$W/o.zst" && "$C" -q -d -c "$W/o.zst" > "$W/co.out" && cmp -s "$f" "$W/co.out" || bad "$n -$lv C->cand decode" + pass=$((pass + 1)) + done + "$B" -q -3 -T4 -c "$f" > "$W/bt.zst" && "$C" -q -3 -T4 -c "$f" > "$W/ct.zst" || { bad "$n -T4 compress"; continue; } + cmp -s "$W/bt.zst" "$W/ct.zst" || bad "$n -T4 compressed bytes differ from base" + "$O" -q -d -c "$W/ct.zst" > "$W/ot.out" && cmp -s "$f" "$W/ot.out" || bad "$n -T4 cand->C decode" + "$C" -q -d -c "$W/ct.zst" > "$W/ct.out" && cmp -s "$f" "$W/ct.out" || bad "$n -T4 cand->cand roundtrip" + pass=$((pass + 1)) + echo "done $n" +done +rm -f "$W"/*.zst "$W"/*.out +echo "silesia: $pass file-level cases, $fail failures" diff --git a/tools/unsafe-baseline.txt b/tools/unsafe-baseline.txt index c57726a..0aa1e91 100644 --- a/tools/unsafe-baseline.txt +++ b/tools/unsafe-baseline.txt @@ -1,23 +1,23 @@ - 94 crates/rusty_alloc/src/alloc.rs + 106 crates/rusty_alloc/src/alloc.rs 15 crates/rusty_alloc/src/arena.rs - 73 crates/rusty_alloc/src/heap.rs + 77 crates/rusty_alloc/src/heap.rs 42 crates/rusty_alloc/src/init.rs 2 crates/rusty_alloc/src/lib.rs - 6 crates/rusty_alloc/src/options.rs + 17 crates/rusty_alloc/src/options.rs 12 crates/rusty_alloc/src/os.rs 38 crates/rusty_alloc/src/page.rs 37 crates/rusty_alloc/src/prim/fixed.rs 8 crates/rusty_alloc/src/prim/mock.rs 17 crates/rusty_alloc/src/prim/mod.rs - 28 crates/rusty_alloc/src/prim/unix.rs + 31 crates/rusty_alloc/src/prim/unix.rs 6 crates/rusty_alloc/src/prim/wasm.rs - 33 crates/rusty_alloc/src/prim/windows.rs + 34 crates/rusty_alloc/src/prim/windows.rs 2 crates/rusty_alloc/src/random.rs 36 crates/rusty_alloc/src/segment.rs 3 crates/rusty_alloc/src/stats.rs - 14 crates/rusty_alloc_api/src/lib.rs + 16 crates/rusty_alloc_api/src/lib.rs 15 crates/rusty_alloc_bench/src/kernels.rs 14 crates/rusty_alloc_bench/src/replay.rs - 356 crates/rusty_alloc_ffi/src/lib.rs + 358 crates/rusty_alloc_ffi/src/lib.rs 51 crates/rusty_alloc_override/src/lib.rs 26 crates/rusty_alloc_wasm/src/lib.rs