From b190c147be34148b67c25f6a990cc6a739f40883 Mon Sep 17 00:00:00 2001 From: rldyourmnd Date: Sun, 27 Sep 2026 14:40:56 +0500 Subject: [PATCH 1/3] feat(agent): gate connection admission on process fd/RSS ceilings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit AgentLimits gains with_process_budget(max_fds, max_rss_mb); when either ceiling is set the agent samples the kernel view (/proc/self/fd + VmRSS on Linux, /dev/fd + proc_pidinfo on macOS, re-statting at most every 200ms) and refuses new connections while usage sits at or above the ceiling — before a connection slot or stream task is consumed. Refusals reuse ConnectionBudgetExhausted; unobservable platforms keep serving. Settings expose limits.max_fds/max_rss_mb (positive, flag-overridable via --max-fds/--max-rss-mb) and the metrics snapshot reports rds_agent_process_fds/rds_agent_process_rss_bytes where observed. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- Cargo.lock | 1 + crates/rds-agent/Cargo.toml | 3 + crates/rds-agent/src/lib.rs | 25 ++++++ crates/rds-agent/src/limits.rs | 113 +++++++++++++++++++++++++++- crates/rds-agent/src/main.rs | 28 +++++-- crates/rds-agent/src/settings.rs | 53 ++++++++++++- crates/rds-agent/src/sys.rs | 80 ++++++++++++++++++++ crates/rds-agent/tests/lifecycle.rs | 84 +++++++++++++++++++++ 8 files changed, 378 insertions(+), 9 deletions(-) create mode 100644 crates/rds-agent/src/sys.rs diff --git a/Cargo.lock b/Cargo.lock index 6bd4a10..42cb477 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3375,6 +3375,7 @@ dependencies = [ "data-encoding", "ed25519-dalek", "iroh-relay", + "libc", "rand 0.10.3", "rds-cli", "rds-client", diff --git a/crates/rds-agent/Cargo.toml b/crates/rds-agent/Cargo.toml index c8da597..decf9f0 100644 --- a/crates/rds-agent/Cargo.toml +++ b/crates/rds-agent/Cargo.toml @@ -32,6 +32,9 @@ thiserror.workspace = true tokio.workspace = true tracing.workspace = true +[target.'cfg(target_os = "macos")'.dependencies] +libc = "0.2" + [dev-dependencies] rustix = { workspace = true, features = ["process"] } iroh-relay = { workspace = true, features = ["server"] } diff --git a/crates/rds-agent/src/lib.rs b/crates/rds-agent/src/lib.rs index 3493674..6770bbb 100644 --- a/crates/rds-agent/src/lib.rs +++ b/crates/rds-agent/src/lib.rs @@ -39,6 +39,7 @@ mod authz; mod limits; mod revocations; pub mod settings; +mod sys; use authz::{ConnAuthz, ConnectionLifetime, authorize}; pub use limits::AgentLimits; pub use revocations::{RevocationFeed, RevocationPolicy, watch_revocations}; @@ -317,6 +318,7 @@ pub struct Agent { limits: AgentLimits, admission: Arc, stream_counter: limits::StreamCounter, + gate: Option, } /// Weak, in-memory observation: keeping an exporter alive never owns agent I/O. @@ -349,6 +351,14 @@ impl AgentMetrics { u64::from(agent.policy.grants_required()), ), ]); + // Process footprint where the kernel reports it; an unobservable + // platform simply omits the key rather than inventing a number. + if let Some(fds) = sys::open_fds() { + values.insert("rds_agent_process_fds", fds as u64); + } + if let Some(rss) = sys::rss_bytes() { + values.insert("rds_agent_process_rss_bytes", rss); + } let grants = agent.policy.active_grants.try_lock().ok(); values.insert("rds_agent_active_grants_known", u64::from(grants.is_some())); if let Some(grants) = grants { @@ -379,6 +389,7 @@ impl Agent { limits, admission: Arc::new(Semaphore::new(limits.connections())), stream_counter: Default::default(), + gate: limits::ResourceGate::new(limits), } } @@ -387,6 +398,7 @@ impl Agent { pub fn with_limits(mut self, limits: AgentLimits) -> Self { self.limits = limits; self.admission = Arc::new(Semaphore::new(limits.connections())); + self.gate = limits::ResourceGate::new(limits); self } @@ -428,6 +440,14 @@ impl Agent { } incoming = self.endpoint.accept() => { let Some(incoming) = incoming else { break; }; + if self.gate.as_ref().is_some_and(|gate| !gate.allows()) { + rds_observe::emit(rds_observe::Event::ConnectionBudgetExhausted); + // Dropping Incoming refuses the handshake without a + // parked application task or a new connection slot. + drop(incoming); + debug!("connection refused: process resource budget exceeded"); + continue; + } let Ok(permit) = self.admission.clone().try_acquire_owned() else { rds_observe::emit(rds_observe::Event::ConnectionBudgetExhausted); // Dropping Incoming refuses the handshake without a @@ -490,6 +510,11 @@ impl Agent { conn.close(5u32.into(), b"invalid grant stream budget"); anyhow::bail!("grant mode requires at least two stream slots"); } + if self.gate.as_ref().is_some_and(|gate| !gate.allows()) { + rds_observe::emit(rds_observe::Event::ConnectionBudgetExhausted); + conn.close(5u32.into(), b"agent process resource budget exceeded"); + anyhow::bail!("agent process resource budget exceeded"); + } let Ok(_permit) = self.admission.clone().try_acquire_owned() else { rds_observe::emit(rds_observe::Event::ConnectionBudgetExhausted); conn.close(5u32.into(), b"agent connection budget exhausted"); diff --git a/crates/rds-agent/src/limits.rs b/crates/rds-agent/src/limits.rs index ca789e9..e388ae6 100644 --- a/crates/rds-agent/src/limits.rs +++ b/crates/rds-agent/src/limits.rs @@ -4,12 +4,15 @@ use std::num::NonZeroU16; use std::sync::Arc; use std::sync::atomic::{AtomicUsize, Ordering}; -/// Per-agent connection slots and per-connection service tasks. The constructor -/// requires positive values; callers may choose smaller budgets for their host. +/// Per-agent connection slots and per-connection service tasks, plus an +/// optional process resource ceiling. The constructor requires positive +/// values; callers may choose smaller budgets for their host. #[derive(Clone, Copy, Debug)] pub struct AgentLimits { connections: usize, streams: usize, + max_fds: Option, + max_rss_bytes: Option, } impl Default for AgentLimits { @@ -17,6 +20,8 @@ impl Default for AgentLimits { Self { connections: 32, streams: 64, + max_fds: None, + max_rss_bytes: None, } } } @@ -26,9 +31,29 @@ impl AgentLimits { Self { connections: usize::from(connections.get()), streams: usize::from(streams.get()), + ..Default::default() } } + /// Process-level ceiling: refuse admissions once the process holds + /// `max_fds` descriptors or `max_rss_mb` resident MiB. `None` leaves + /// that quantity ungated. + pub fn with_process_budget(mut self, max_fds: Option, max_rss_mb: Option) -> Self { + self.max_fds = max_fds; + self.max_rss_bytes = max_rss_mb.and_then(|mb| mb.checked_mul(1024 * 1024)); + self + } + + /// Configured fd ceiling, if any. + pub fn max_fds(self) -> Option { + self.max_fds + } + + /// Configured resident-set ceiling in bytes, if any. + pub fn max_rss_bytes(self) -> Option { + self.max_rss_bytes + } + /// Pending handshakes plus admitted connections, shared by run and serve. pub fn connections(self) -> usize { self.connections @@ -62,3 +87,87 @@ impl Drop for StreamTask { self.0.fetch_sub(1, Ordering::Relaxed); } } + +/// How often the process budget re-observes the kernel's view. Re-statting +/// per accepted connection would put procfs/`dev` scans on the hot path. +const SAMPLE_INTERVAL: std::time::Duration = std::time::Duration::from_millis(200); + +/// Process-level admission ceiling. Ungated (`None` limits) or +/// unobservable (platform reports neither fds nor RSS) configurations +/// never refuse — a bound only exists where the kernel actually reports it. +pub(super) struct ResourceGate { + max_fds: Option, + max_rss_bytes: Option, + cache: std::sync::Mutex<(std::time::Instant, bool)>, +} + +impl ResourceGate { + pub fn new(limits: AgentLimits) -> Option { + if limits.max_fds.is_none() && limits.max_rss_bytes.is_none() { + return None; + } + Some(Self { + max_fds: limits.max_fds, + max_rss_bytes: limits.max_rss_bytes, + cache: std::sync::Mutex::new((std::time::Instant::now() - 2 * SAMPLE_INTERVAL, true)), + }) + } + + /// True while the observed process usage fits the configured budget. + /// An unobservable quantity contributes nothing — the gate reports + /// only what the kernel actually showed it. + pub fn allows(&self) -> bool { + let mut cache = crate::lock(&self.cache); + let (at, verdict) = *cache; + if at.elapsed() < SAMPLE_INTERVAL { + return verdict; + } + let mut ok = true; + let mut observed = false; + if let Some(max) = self.max_fds + && let Some(fds) = crate::sys::open_fds() + { + observed = true; + ok &= (fds as u64) < max; + } + if let Some(max) = self.max_rss_bytes + && let Some(rss) = crate::sys::rss_bytes() + { + observed = true; + ok &= rss < max; + } + // Nothing observable: keep serving rather than gate on a guess. + let verdict = ok || !observed; + *cache = (std::time::Instant::now(), verdict); + verdict + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn no_budget_means_no_gate() { + assert!(ResourceGate::new(AgentLimits::default()).is_none()); + assert!( + ResourceGate::new(AgentLimits::default().with_process_budget(None, None)).is_none() + ); + } + + #[cfg(any(target_os = "linux", target_os = "macos"))] + #[test] + fn impossible_fd_ceiling_refuses() { + let limits = AgentLimits::default().with_process_budget(Some(1), None); + let gate = ResourceGate::new(limits).expect("budget configured"); + assert!(!gate.allows()); + } + + #[cfg(any(target_os = "linux", target_os = "macos"))] + #[test] + fn generous_ceiling_allows() { + let limits = AgentLimits::default().with_process_budget(Some(u64::MAX), Some(u64::MAX)); + let gate = ResourceGate::new(limits).expect("budget configured"); + assert!(gate.allows()); + } +} diff --git a/crates/rds-agent/src/main.rs b/crates/rds-agent/src/main.rs index b6e9d9e..4c7b93e 100644 --- a/crates/rds-agent/src/main.rs +++ b/crates/rds-agent/src/main.rs @@ -78,6 +78,14 @@ struct Cli { /// one slot is reserved from service bodies for authorization/renewal. #[arg(long)] max_streams: Option, + /// Process-wide open-descriptor ceiling; new connections are refused + /// while the process holds this many or more (Linux/macOS). + #[arg(long)] + max_fds: Option, + /// Process-wide resident-set ceiling in MiB; new connections are + /// refused while resident memory meets or exceeds it. + #[arg(long)] + max_rss_mb: Option, /// Inbound connection handshake deadline in seconds (1..=3600). #[arg(long)] handshake_timeout: Option, @@ -212,6 +220,8 @@ async fn run(cli: Cli) -> anyhow::Result<()> { revocations_interval: cli.revocations_interval, max_connections: cli.max_connections, max_streams: cli.max_streams, + max_fds: cli.max_fds, + max_rss_mb: cli.max_rss_mb, handshake_timeout: cli.handshake_timeout, hello_timeout: cli.hello_timeout, authz_timeout: cli.authz_timeout, @@ -399,12 +409,18 @@ async fn run(cli: Cli) -> anyhow::Result<()> { }; let agent = std::sync::Arc::new( - Agent::new(endpoint, policy).with_limits(AgentLimits::new( - resolved - .max_connections - .unwrap_or(std::num::NonZeroU16::new(32).expect("positive limit")), - max_streams, - )), + Agent::new(endpoint, policy).with_limits( + AgentLimits::new( + resolved + .max_connections + .unwrap_or(std::num::NonZeroU16::new(32).expect("positive limit")), + max_streams, + ) + .with_process_budget( + resolved.max_fds.map(std::num::NonZeroU64::get), + resolved.max_rss_mb.map(std::num::NonZeroU64::get), + ), + ), ); let metrics = agent.metrics(); let mut control = diff --git a/crates/rds-agent/src/settings.rs b/crates/rds-agent/src/settings.rs index 1745831..a0a1ab6 100644 --- a/crates/rds-agent/src/settings.rs +++ b/crates/rds-agent/src/settings.rs @@ -9,7 +9,7 @@ use std::collections::BTreeSet; use std::io::Read; -use std::num::NonZeroU16; +use std::num::{NonZeroU16, NonZeroU64}; use std::path::{Path, PathBuf}; use std::str::FromStr; use std::time::Duration; @@ -190,6 +190,13 @@ pub struct AuthoritySettings { pub struct LimitSettings { pub max_connections: Option, pub max_streams: Option, + /// Process-wide open-descriptor ceiling; once the kernel reports the + /// process at or above it, new connections are refused until usage + /// falls. Absent = ungated. + pub max_fds: Option, + /// Process-wide resident-set ceiling in MiB, same admission rule as + /// `max_fds`. Absent = ungated. + pub max_rss_mb: Option, } /// Connection-admission and stream-greeting deadlines in seconds, each @@ -263,6 +270,8 @@ pub struct AgentOverrides { pub revocations_interval: Option, pub max_connections: Option, pub max_streams: Option, + pub max_fds: Option, + pub max_rss_mb: Option, pub handshake_timeout: Option, pub hello_timeout: Option, pub authz_timeout: Option, @@ -299,6 +308,10 @@ pub struct ResolvedAgent { pub revocations_interval: Option, pub max_connections: Option, pub max_streams: Option, + /// Process fd ceiling (see `LimitSettings::max_fds`). + pub max_fds: Option, + /// Process resident-set ceiling in MiB. + pub max_rss_mb: Option, pub timeouts: Option, } @@ -467,6 +480,12 @@ impl AgentSettings { if flags.max_streams.is_some() { self.limits.max_streams = flags.max_streams; } + if flags.max_fds.is_some() { + self.limits.max_fds = flags.max_fds; + } + if flags.max_rss_mb.is_some() { + self.limits.max_rss_mb = flags.max_rss_mb; + } if flags.handshake_timeout.is_some() { self.timeouts.handshake_secs = flags.handshake_timeout; } @@ -712,6 +731,8 @@ impl AgentSettings { revocations_interval: revocations.and_then(|r| r.interval_secs), max_connections: self.limits.max_connections, max_streams: self.limits.max_streams, + max_fds: self.limits.max_fds, + max_rss_mb: self.limits.max_rss_mb, timeouts, }) } @@ -870,6 +891,36 @@ mod tests { AgentSettings::from_json(br#"{"schema_version":1,"limits":{"max_streams":0}}"#) .is_err() ); + assert!( + AgentSettings::from_json(br#"{"schema_version":1,"limits":{"max_fds":0}}"#).is_err() + ); + assert!( + AgentSettings::from_json(br#"{"schema_version":1,"limits":{"max_rss_mb":0}}"#).is_err() + ); + } + + #[test] + fn resource_budgets_merge_and_resolve() { + let resolved = + resolve_ok(r#"{"schema_version":1,"limits":{"max_fds":512,"max_rss_mb":256}}"#); + assert_eq!(resolved.max_fds.unwrap().get(), 512); + assert_eq!(resolved.max_rss_mb.unwrap().get(), 256); + + // Flag values replace file values per-field. + let settings = + parse(r#"{"schema_version":1,"limits":{"max_fds":128}}"#).apply(AgentOverrides { + max_fds: NonZeroU64::new(256), + max_rss_mb: NonZeroU64::new(64), + ..Default::default() + }); + let resolved = settings.resolve().unwrap(); + assert_eq!(resolved.max_fds.unwrap().get(), 256); + assert_eq!(resolved.max_rss_mb.unwrap().get(), 64); + + // Absent everywhere resolves to no process gate. + let resolved = resolve_ok(r#"{"schema_version":1}"#); + assert!(resolved.max_fds.is_none()); + assert!(resolved.max_rss_mb.is_none()); } #[test] diff --git a/crates/rds-agent/src/sys.rs b/crates/rds-agent/src/sys.rs new file mode 100644 index 0000000..c9a2aef --- /dev/null +++ b/crates/rds-agent/src/sys.rs @@ -0,0 +1,80 @@ +//! Process resource observation for admission budgets (W2.5). +//! +//! Platform coverage is honest: an unobservable quantity reports `None` +//! and the corresponding gate stays open rather than pretending a bound. + +/// Open file descriptors owned by this process, where the kernel exposes +/// them (`/proc/self/fd` on Linux, `/dev/fd` on macOS). +#[cfg(any(target_os = "linux", target_os = "macos"))] +pub(crate) fn open_fds() -> Option { + #[cfg(target_os = "linux")] + const FD_DIR: &str = "/proc/self/fd"; + #[cfg(target_os = "macos")] + const FD_DIR: &str = "/dev/fd"; + Some(std::fs::read_dir(FD_DIR).ok()?.count()) +} + +#[cfg(not(any(target_os = "linux", target_os = "macos")))] +pub(crate) fn open_fds() -> Option { + None +} + +/// Resident set size in bytes. +#[cfg(target_os = "linux")] +pub(crate) fn rss_bytes() -> Option { + let status = std::fs::read_to_string("/proc/self/status").ok()?; + let line = status.lines().find(|l| l.starts_with("VmRSS:"))?; + let kb: u64 = line + .trim_start_matches("VmRSS:") + .trim_end_matches("kB") + .trim() + .parse() + .ok()?; + kb.checked_mul(1024) +} + +/// Resident set size in bytes via `proc_pidinfo` (`PROC_PIDTASKINFO`). +#[cfg(target_os = "macos")] +#[allow(unsafe_code)] +pub(crate) fn rss_bytes() -> Option { + use std::mem::{MaybeUninit, size_of}; + let mut info = MaybeUninit::::uninit(); + // SAFETY: `info` is live, correctly aligned taskinfo storage for the + // PROC_PIDTASKINFO request; the call fills it on a full-size return. + let size = unsafe { + libc::proc_pidinfo( + libc::getpid(), + libc::PROC_PIDTASKINFO, + 0, + info.as_mut_ptr().cast(), + size_of::() as i32, + ) + }; + if size == size_of::() as i32 { + // SAFETY: proc_pidinfo wrote exactly one complete taskinfo record. + Some(unsafe { info.assume_init() }.pti_resident_size) + } else { + None + } +} + +#[cfg(not(any(target_os = "linux", target_os = "macos")))] +pub(crate) fn rss_bytes() -> Option { + None +} + +#[cfg(all(test, any(target_os = "linux", target_os = "macos")))] +mod tests { + use super::*; + + #[test] + fn open_fds_reports_a_running_process() { + // stdin/stdout/stderr at minimum; a test process holds more. + assert!(open_fds().expect("kernel reports fds") >= 3); + } + + #[test] + fn rss_reports_a_running_process() { + assert!(rss_bytes().expect("kernel reports RSS") > 0); + } +} diff --git a/crates/rds-agent/tests/lifecycle.rs b/crates/rds-agent/tests/lifecycle.rs index 37d1552..caf7175 100644 --- a/crates/rds-agent/tests/lifecycle.rs +++ b/crates/rds-agent/tests/lifecycle.rs @@ -296,6 +296,90 @@ async fn direct_serve_enforces_budget_and_cancellation_releases_it() { } } +/// Fixture variant carrying a process fd ceiling (MiB ceiling unused here). +#[cfg(any(target_os = "linux", target_os = "macos"))] +async fn fixture_with_fd_budget( + backend: Backend, + connections: u16, + streams: u16, + max_fds: u64, +) -> (Arc, Vec) { + let config = || EndpointConfig { + backend, + discovery: false, + bind_addrs: vec!["127.0.0.1:0".parse().unwrap()], + ..Default::default() + }; + let clients = vec![ + bind_endpoint(config()).await.unwrap(), + bind_endpoint(config()).await.unwrap(), + ]; + let mut policy = AgentPolicy::ssh_only(("127.0.0.1".into(), 9)); + policy.allow.extend(clients.iter().map(Endpoint::id)); + let agent = Agent::new(bind_endpoint(config()).await.unwrap(), policy).with_limits( + AgentLimits::new( + std::num::NonZeroU16::new(connections).unwrap(), + std::num::NonZeroU16::new(streams).unwrap(), + ) + .with_process_budget(Some(max_fds), None), + ); + (Arc::new(agent), clients) +} + +/// A ceiling of one descriptor is already exceeded, so the runner refuses +/// the handshake and `serve` refuses an established conn — neither consumes +/// a connection slot. +#[cfg(any(target_os = "linux", target_os = "macos"))] +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn process_fd_budget_refuses_runner_and_serve() { + for backend in backends() { + let (agent, clients) = fixture_with_fd_budget(backend, 4, 2, 1).await; + let runner = start(&agent); + let refused = tokio::time::timeout(Duration::from_secs(2), async { + rds_cli::connect(&clients[0], agent.endpoint.addr()) + .await + .is_err() + }) + .await + .expect("over-budget agent left a connection attempt parked"); + assert!( + refused, + "{backend:?} over-budget agent admitted a connection" + ); + state(&agent, 0, 0).await; + runner.abort(); + assert!(runner.await.unwrap_err().is_cancelled()); + // serve() applies the same gate to an already-established conn. + let (conn, server) = manual_pair(&agent, &clients[1]).await; + let error = agent.serve(server).await.unwrap_err(); + assert!(error.to_string().contains("resource budget")); + tokio::time::timeout(Duration::from_secs(2), conn.wait_closed()) + .await + .unwrap(); + assert_eq!(agent.active_connections(), 0); + agent.endpoint.close().await; + for client in clients { + client.close().await; + } + } +} + +/// A ceiling above real usage admits normally — the gate must not disturb +/// the configured-but-untriggered path. +#[cfg(any(target_os = "linux", target_os = "macos"))] +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn process_fd_budget_above_usage_still_serves() { + for backend in backends() { + let (agent, clients) = fixture_with_fd_budget(backend, 1, 2, u64::MAX).await; + let runner = start(&agent); + let conn = rds_cli::connect(&clients[0], agent.endpoint.addr()) + .await + .unwrap(); + rds_cli::ping(&conn, rand::random::()).await.unwrap(); + stop(&agent, &clients, runner).await; + } +} + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn metrics_track_real_admission_streams_and_do_not_own_agent_io() { for backend in backends() { From 6a22c83db8a379da21be8b4505f02f9d094cded9 Mon Sep 17 00:00:00 2001 From: rldyourmnd Date: Sun, 27 Sep 2026 14:40:56 +0500 Subject: [PATCH 2/3] docs(agent): document process resource ceilings (W2.5 partial) Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- docs/agent-configuration.md | 18 +++++++++++++++++- docs/remediation-progress.md | 31 ++++++++++++++++++++++++++++++- 2 files changed, 47 insertions(+), 2 deletions(-) diff --git a/docs/agent-configuration.md b/docs/agent-configuration.md index 6babf1d..3f3341f 100644 --- a/docs/agent-configuration.md +++ b/docs/agent-configuration.md @@ -34,7 +34,7 @@ identity creation or socket binding, alongside the endpoint preflight. | `service` | `ssh_target`, `tcp_targets`, `allow_any_tcp`, `sync_dir` | | `peers` | `allow`: endpoint-id strings, at most 256 | | `authority` | `issuers`, `grant_ttl_secs` (1–86400), `tenant`, `policy_min_revision`, `directory`, `directory_ca`, `record_ttl_secs`, `record_state`, `registry`, `revocations` | -| `limits` | `max_connections`, `max_streams`; positive 16-bit | +| `limits` | `max_connections`, `max_streams` (positive 16-bit); `max_fds`, `max_rss_mb` (positive, optional process ceilings) | | `timeouts` | `handshake_secs`, `hello_secs`, `authz_secs`, `shutdown_secs`; each 1–3600 | `registry` holds `key` (required when present), `epoch`, `state` and @@ -93,6 +93,22 @@ values. Per-service budgets (frame streams, transfer deadlines, renewal windows) remain service-internal and are not set from this file; client dial, idle and media/progress classes are still W2.6 open items. +## Process resource ceilings + +`limits.max_fds` and `limits.max_rss_mb` bound the whole process, not a +single connection: while the kernel reports the agent holding that many +open descriptors (Linux `/proc/self/fd`, macOS `/dev/fd`) or that much +resident memory (Linux `VmRSS`, macOS `proc_pidinfo`), new connections are +refused at admission — the pending handshake is dropped before a slot or +task is consumed — until usage falls below the ceiling again. Refusals +emit `ConnectionBudgetExhausted` like the connection semaphore; the +observed `rds_agent_process_fds` and `rds_agent_process_rss_bytes` keys +appear in the metrics snapshot wherever the kernel reports them. +Platforms without an observable quantity keep serving rather than +gating on a guess. Flags `--max-fds` and `--max-rss-mb` override file +values. Per-service fairness and disk/media job budgets are still open +under W2.5. + ## Example [Access role](../examples/agent-access.json) keeps the historical default: diff --git a/docs/remediation-progress.md b/docs/remediation-progress.md index 11db746..914178c 100644 --- a/docs/remediation-progress.md +++ b/docs/remediation-progress.md @@ -49,7 +49,7 @@ Neither increment closes these product gaps or any wave. | W2.2 | Partial; exact ALPN selection + sync/desktop session routing | Immutable per-protocol TLS offers prevent silent fallback and concurrent request interference. Managed single-file transfers use fresh control/uni routing IDs and negotiate version/limits in the `SyncTransferV2` session envelope before any filesystem operation. Desktop sessions mint random per-session IDs in `StreamHello::DesktopV2` and route frames through `UniHello::DesktopFrames { id }`, isolating stale streams and allowing concurrent sessions; the same display grant scope check covers both greetings. Service-wide capability negotiation remains open. | | W2.3 | Partial; destination-bound renewable grants, directional scopes, tenant/policy binding and per-path sync scopes | Grant v2 adds a strict signature domain, audience and stable session ID across positive lease revisions. Same-scope renewal preserves streams/revocation, retains one replay slot/watchdog and enforces wall/continuous expiry. Explicit managed renewal uses IPC v3 and a control-completion barrier. `SyncRead`/`SyncWrite` and `DesktopView`/`DesktopControl` are enforced before filesystem/input operations. Grant v3 adds the `tenant`/`policy_revision` claims and the `constraints.sync_paths` subtree scope, checked after `rel_path` normalization before any filesystem work; v2 payloads still verify with all claims absent, and agents pin the binding via `authority.tenant`/`policy_min_revision` (`--tenant`/`--policy-min-revision`), refusing unscoped or stale grants. Account-level scopes and automatic GDS issuer integration remain open. See [contract](grant-leases.md). | | W2.4 | Partial; default connectivity manager + managed desktop channel | Agent local control is enabled by default; ordinary ticket/ping/info/SSH/forward/send/recv/desktop commands and keyless `rds session` reuse its endpoint (local wire v5; `desktop --direct` bypasses). Same-UID IPC, pinned streams, cancellation and aggregate metrics are implemented. Agent/direct CLI/owned relay acquire exclusive ownership of a validated seed inode. Coordinated installed-binary migration, native macOS and real multi-user/relay qualification remain open. See [contract](local-sessions.md) and [migration receipt](reports/rds-identity-migration-20260925.md). | -| W2.5 | Partial; transport and agent task ownership | Owned policy tasks terminate, including explicit shutdown after stopped protocol I/O; uni routing is bounded and acyclic. Agent and client forwarding groups own cancellation, normal joins and positive admission budgets. Client relay queues/peer leases and server admission/owned shutdown are bounded. Metric samplers use weak backend observations, release their gauges on drop and wake on closure independently of the sampling interval. Global RSS/FD bounds, per-service fairness and broader disk/media cancellation remain open. | +| W2.5 | Partial; transport and agent task ownership | Owned policy tasks terminate, including explicit shutdown after stopped protocol I/O; uni routing is bounded and acyclic. Agent and client forwarding groups own cancellation, normal joins and positive admission budgets. Client relay queues/peer leases and server admission/owned shutdown are bounded. Metric samplers use weak backend observations, release their gauges on drop and wake on closure independently of the sampling interval. Process-wide fd/RSS ceilings now gate connection admission (`limits.max_fds`/`max_rss_mb`, `--max-fds`/`--max-rss-mb`) with kernel-reported observability on Linux/macOS and honest ungated behavior elsewhere; per-service fairness, broader disk/media cancellation and storm-grade RSS/FD proof remain open. | | W2.6 | Partial; agent timeout classes complete, publish retry bounded | One request deadline covers stream credit, writes, replies and Ping echo; canceled Authz closes its connection. Agent policy now owns all four server-side classes — handshake, hello, authz reply and shutdown join — as `TimeoutPolicy` tunables (`timeouts.*_secs`, `--*-timeout`, 1..=3600). Directory publish retries use bounded exponential backoff with equal jitter (`RetryPolicy`, 1s→30s default) instead of the fixed ~1s poll cadence; fatal 4xx still fails closed. Agent and owned relay handshake/shutdown budgets exist; agent local startup no longer waits indefinitely for an iroh relay. Client dial/idle classes, transport-level retry policy reuse, desktop/media deadlines and broader startup recovery remain open. | | W3.1 | Partial; fair bounded candidate race | Canonical direct candidates alternate supported families under one eight-address cap, plus attached relay; attempts share a deadline and one authenticated winner. Independent relay bootstrap, progressive probing, remote scope/interface discovery and real topology qualification remain open. | | W3.2 | Partial; owned binary runtime checked on Linux | Both server binaries share strict backend/allow/key/limit/TLS config, persistent relay identity, local readiness and checked joined shutdown. Real processes forward inner authenticated traffic and retain identity/catalog across restart. Malformed datagrams are charged before parsing, and routing uses authenticated key-table lookup. Unexpected service-runner completion now initiates joined host shutdown with retained failure. Hung-task/recovery policy, global/reconnect/control budgets and platform/network qualification remain open. | @@ -2078,3 +2078,32 @@ alive — the fixed-cadence storm the criterion rules out. Remaining W2.6: client dial/idle classes, transport-level retry reuse beyond announce, desktop/media deadlines, broader startup recovery. + +## 2026-09-27 — process resource ceilings gate admission (W2.5 partial) + +`AgentLimits` gains an optional process budget +(`with_process_budget(max_fds, max_rss_mb)`) surfaced through +`limits.max_fds`/`limits.max_rss_mb` and `--max-fds`/`--max-rss-mb`. When +either ceiling is configured the agent samples the kernel's view — +`/proc/self/fd` + `VmRSS` on Linux, `/dev/fd` + `proc_pidinfo` on macOS, +re-statting at most every 200ms so procfs scans stay off the accept hot +path — and refuses new connections while usage sits at or above the +ceiling: the pending `Incoming` is dropped (runner path) or the +established connection is closed with a distinct error (`serve` path), +in both cases before a connection slot or stream task is consumed. +Refusals reuse `ConnectionBudgetExhausted`; unobservable platforms keep +serving rather than gate on a guess, and the metrics snapshot exposes +`rds_agent_process_fds`/`rds_agent_process_rss_bytes` wherever the +kernel reports them. + +Tests: kernel sanity (`open_fds`, `rss_bytes` report a live process), +gate construction (`None` budgets install no gate, an impossible +ceiling refuses, a generous one admits), settings merge/resolve/zero +rejection, and e2e coverage of both refusal paths — a runner under a +one-descriptor ceiling refuses the handshake and `serve` refuses an +established conn with `resource budget` in the error, while a ceiling +above real usage admits and pings normally. + +Remaining W2.5: per-service fairness inside the stream budget, sync +disk-job and desktop/media cancellation breadth, and storm-grade +RSS/FD proof under adversarial slow peers. From 495416479bf1bae72bedd998730c8038927d06c1 Mon Sep 17 00:00:00 2001 From: rldyourmnd Date: Sun, 27 Sep 2026 14:40:56 +0500 Subject: [PATCH 3/3] fix(bench): randomize ping payload flagged as hard-coded crypto literal scenario.rs already generates ping payloads via rand::random elsewhere; the literal `1` tripped the CodeQL rust/hard-coded-cryptographic-value rule (critical alert on main). Same treatment. Generated with [Devin](https://devin.ai) Co-Authored-By: Devin <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- crates/rds-bench/src/scenario.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/rds-bench/src/scenario.rs b/crates/rds-bench/src/scenario.rs index c7e3dd3..4a91bc9 100644 --- a/crates/rds-bench/src/scenario.rs +++ b/crates/rds-bench/src/scenario.rs @@ -611,7 +611,7 @@ async fn resolve_connect(p: &Params) -> anyhow::Result { let conn = rds_cli::connect(&client_ep, addr).await?; let d_connect = t_connect.elapsed(); let t_first = Instant::now(); - rds_cli::ping(&conn, 1).await?; + rds_cli::ping(&conn, rand::random::()).await?; let d_first = t_first.elapsed(); client_ep.metrics().sampler(conn.clone()).sample(); Ok::<_, anyhow::Error>((d_resolve, d_connect, d_first, t_resolve.elapsed()))