diff --git a/README.md b/README.md index b507ee9..b045004 100644 --- a/README.md +++ b/README.md @@ -112,6 +112,12 @@ Headless `run`/`resume`에서 도구 인자 검증이 거절되면 기존 verdic 값·실제 경로·원문 오류는 출력하지 않으며, 이 정보는 자동 재시도 권한이나 durable 증빙이 아니다. 호스트 연동 계약과 제한은 [ADR-0038](docs/adr/0038-safe-invocation-diagnostics.md)을 따른다. +불확정 모델 호출은 `xgeny recover RUN_ID`로 오프라인 조회할 수 있습니다. 응답 수락을 명시적으로 +포기하려면 조회한 정확한 ID로 `xgeny recover RUN_ID --discard-model-call CALL_ID`를 실행합니다. +소비한 호출 예산은 복구되지 않으며 이 명령은 모델·도구를 호출하거나 Run을 재개하지 않습니다. +이후 재개에도 원래 workspace·권한·남은 예산이 필요합니다. +[복구 절차와 호스트 연동 경계](docs/development/local-model-call-recovery.md)를 먼저 확인하세요. + ## 제품 원칙 - 사용자는 `xgeny` 하나만 설치합니다. diff --git a/crates/xgeny-cli/src/composition.rs b/crates/xgeny-cli/src/composition.rs index 9af10c9..09737ca 100644 --- a/crates/xgeny-cli/src/composition.rs +++ b/crates/xgeny-cli/src/composition.rs @@ -549,6 +549,128 @@ pub enum PublicRunError { Internal, } +/// Bounded offline recovery view, never a provider response or a completion claim. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct ModelCallRecoveryReport { + pub format_version: u32, + pub run_id: String, + pub journal_sequence: u64, + pub journal_head_digest: String, + pub active_call: Option, + pub lifecycle_configured: bool, + pub reserved_calls: u32, + pub max_model_calls: u32, + pub settled_calls: u32, + pub unknown_calls: u32, + pub effect_recovery_required: bool, + pub discarded_call_id: Option, +} + +/// Only a Core-generated identity and the durable closed status taxonomy. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct RecoveryModelCall { + pub call_id: String, + pub status: ModelCallStatus, +} + +/// Inspect model-call recovery under the Run lease without changing journal state. +/// Does not resolve credentials, open the workspace, or invoke providers/tools. +/// +/// # Errors +/// Returns a closed public error for an invalid layout, held lease, or corrupt state. +pub fn inspect_local_model_call(run_id: &str) -> Result { + local_model_call_recovery(run_id, None) +} + +/// Explicitly stop accepting the exact active call's response. Never refunds a slot, +/// asserts non-delivery, retries a request, or resumes the Run. +/// +/// # Errors +/// Also fails before mutation for a missing/stale call ID or a completed Run. +pub fn discard_local_model_call( + run_id: &str, + call_id: &str, +) -> Result { + local_model_call_recovery(run_id, Some(call_id)) +} + +fn local_model_call_recovery( + run_id: &str, + discard_call_id: Option<&str>, +) -> Result { + let state_root = discover_state_root().map_err(|_| PublicRunError::Configuration)?; + let layout = + RunLayout::existing(&state_root, run_id).map_err(|_| PublicRunError::Configuration)?; + let lease = acquire_lease(&layout, run_id)?; + let manifest = layout + .read_manifest() + .map_err(|_| PublicRunError::Integrity)?; + let store = SqliteRunStore::open_existing_read_only(layout.database_path()) + .map_err(|_| PublicRunError::Integrity)?; + let mut state = store + .load_current() + .map_err(|_| PublicRunError::Integrity)? + .ok_or(PublicRunError::Integrity)?; + verify_manifest_state(&manifest, &state)?; + let completed = load_offline_completion(&store, &state)?.is_some(); + if let Some(call_id) = discard_call_id { + let active = state + .agent_loop + .as_ref() + .and_then(|agent| agent.model_calls.as_ref()) + .and_then(|calls| calls.active_call.as_ref()); + if completed || active.is_none_or(|call| call.reservation.call_id() != call_id) { + return Err(PublicRunError::Configuration); + } + drop(store); + let mut store = reopen_writable_verified(&layout, &manifest, &state)?; + let runtime = AgentLoop::with_model_call_budget( + manifest + .budget() + .agent_loop() + .map_err(|_| PublicRunError::Integrity)?, + manifest + .budget() + .model_calls() + .map_err(|_| PublicRunError::Integrity)?, + ); + let tick = runtime + .abandon_model_call(&mut store, &mut HostEventFactory, &lease, call_id) + .map_err(|_| PublicRunError::Integrity)?; + if !matches!(tick, AgentLoopTick::ModelCallAbandoned { .. }) { + return Err(PublicRunError::Integrity); + } + state = store + .load_current() + .map_err(|_| PublicRunError::Integrity)? + .ok_or(PublicRunError::Integrity)?; + } + let calls = state + .agent_loop + .as_ref() + .and_then(|agent| agent.model_calls.as_ref()); + Ok(ModelCallRecoveryReport { + format_version: 1, + run_id: state.run_id.clone(), + journal_sequence: state.journal_sequence, + journal_head_digest: state.journal_head_digest.clone(), + active_call: calls + .and_then(|calls| calls.active_call.as_ref()) + .map(|call| RecoveryModelCall { + call_id: call.reservation.call_id().to_owned(), + status: call.status, + }), + lifecycle_configured: calls.is_some(), + reserved_calls: calls.map_or(0, |calls| calls.reserved_calls), + max_model_calls: manifest.budget().max_model_calls, + settled_calls: calls.map_or(0, |calls| calls.settled_calls), + unknown_calls: calls.map_or(0, |calls| calls.unknown_calls), + effect_recovery_required: executing_step_id(&state).is_some() + || has_effect_uncertainty(&state), + discarded_call_id: discard_call_id.map(str::to_owned), + }) +} + impl PublicRunError { #[must_use] pub const fn code(self) -> &'static str { diff --git a/crates/xgeny-cli/src/main.rs b/crates/xgeny-cli/src/main.rs index 51d7a7b..e7d4e86 100644 --- a/crates/xgeny-cli/src/main.rs +++ b/crates/xgeny-cli/src/main.rs @@ -11,7 +11,8 @@ use xgeny_cli::{ LocalProcessSession, LocalResumeRequest, LocalRunRequest, ModelCheckError, ModelCheckRequest, ModelCredentialStore, ModelProfile, ModelProfileError, ModelProfileStore, OsModelCredentialStore, PublicRunError, check_openai_compatibility, check_openai_model, - list_openai_models, new_credential_reference, prepare_local_process_session, resume_local, + discard_local_model_call, inspect_local_model_call, list_openai_models, + new_credential_reference, prepare_local_process_session, resume_local, resume_local_with_model_resolver, resume_local_with_model_resolver_and_progress, resume_local_with_process_session_and_model_resolver_progress, run_local_with_process_session_progress, run_local_with_started, @@ -57,6 +58,20 @@ enum Command { Run(RunArgs), /// Continue an existing Run, or replay its durable completion without model access. Resume(ResumeArgs), + /// Inspect or explicitly discard an unresolved model call offline; never resumes the Run. + Recover(RecoverArgs), +} + +#[derive(Debug, Args)] +#[command( + after_long_help = "Without --discard-model-call this command only inspects verified journal state. Discard stops accepting the exact call's response; it does NOT prove non-delivery or refund consumed budget. Neither form calls a model or tool. Continue separately with resume and the original workspace, catalogs, profile, permissions and remaining budget. Output is a bounded JSON report, not a completion result." +)] +struct RecoverArgs { + /// Durable Run identifier printed by `xgeny run`. + run_id: String, + /// Explicitly discard this exact active call ID obtained from a prior inspection. + #[arg(long, value_name = "CALL_ID")] + discard_model_call: Option, } #[derive(Debug, Subcommand)] @@ -284,6 +299,7 @@ fn main() -> ExitCode { Some(Command::Model { command }) => run_model_command(command), Some(Command::Run(args)) => run_command(args), Some(Command::Resume(args)) => resume_command(args), + Some(Command::Recover(args)) => recover_command(&args), } } @@ -620,6 +636,27 @@ fn resume_command(args: ResumeArgs) -> ExitCode { } } +fn recover_command(args: &RecoverArgs) -> ExitCode { + let result = match &args.discard_model_call { + Some(call_id) => discard_local_model_call(&args.run_id, call_id), + None => inspect_local_model_call(&args.run_id), + }; + match result { + Ok(report) => { + // A failed stdout write can follow a committed discard. Inspect before retrying. + let mut stdout = std::io::stdout().lock(); + if serde_json::to_writer(&mut stdout, &report).is_err() + || stdout.write_all(b"\n").is_err() + || stdout.flush().is_err() + { + return present(Err(PublicRunError::Internal)); + } + ExitCode::SUCCESS + } + Err(error) => present(Err(error)), + } +} + const MAX_TOKEN_INPUT_BYTES: u64 = 16 * 1024; struct ResolvedModel { diff --git a/crates/xgeny-cli/tests/public_run_resume.rs b/crates/xgeny-cli/tests/public_run_resume.rs index 72a31df..218a1dd 100644 --- a/crates/xgeny-cli/tests/public_run_resume.rs +++ b/crates/xgeny-cli/tests/public_run_resume.rs @@ -143,6 +143,10 @@ impl BlockingServer { impl OneTurnServer { fn spawn(response: Vec) -> Self { + Self::spawn_status(200, response) + } + + fn spawn_status(status: u16, response: Vec) -> Self { let listener = TcpListener::bind("127.0.0.1:0").expect("test listener should bind"); let address = listener .local_addr() @@ -153,7 +157,7 @@ impl OneTurnServer { let observed = read_http_request(&mut stream); let _ = request_sender.send(observed); let headers = format!( - "HTTP/1.1 200 OK\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n", + "HTTP/1.1 {status} Fixture\r\nContent-Type: application/json\r\nContent-Length: {}\r\nConnection: close\r\n\r\n", response.len() ); stream @@ -570,6 +574,10 @@ fn interrupted_reserved_model_call_becomes_unknown_without_egress_or_retry() { server.release.send(()).expect("server should release"); server.handle.join().expect("server should finish"); + let inspected = recovery_report(&state_root, &run_id); + assert_eq!(inspected["active_call"]["status"]["status"], "reserved"); + assert_eq!(inspected["reserved_calls"], 1); + let first_recovery = xgeny(&state_root) .args(["resume", &run_id]) .bounded_output() @@ -610,6 +618,412 @@ fn interrupted_reserved_model_call_becomes_unknown_without_egress_or_retry() { assert_eq!(after_repeat.records, after_mark.records); } +fn recovery_report(state_root: &Path, run_id: &str) -> Value { + let output = xgeny(state_root) + // Offline recovery must not parse or resolve model configuration. + .env("XGENY_OPENAI_INFERENCE_TIMEOUT", "invalid-unused-value") + .env("XGENY_MODEL_PROFILE", "missing-unused-profile") + .args(["recover", run_id]) + .bounded_output() + .expect("recovery inspection should return"); + assert_exit(&output, 0); + serde_json::from_slice(&output.stdout).expect("bounded recovery report should be JSON") +} + +#[test] +#[allow(clippy::too_many_lines)] +fn explicit_discard_of_reserved_call_is_offline_and_lease_guarded() { + let fixture = tempdir().expect("test directory should exist"); + let state_root = fixture.path().join("state"); + let workspace = fixture.path().join("workspace"); + fs::create_dir(&workspace).expect("workspace should create"); + fs::write(workspace.join("README.md"), FILE_MARKER).expect("fixture should write"); + let server = BlockingServer::spawn(); + let mut child = ChildGuard::new( + xgeny(&state_root) + .args([ + "run", + "goal", + "--workspace", + path_text(&workspace), + "--base-url", + &server.base_url, + "--model", + MODEL, + "--tokenizer", + TOKENIZER, + "--allow-file", + "README.md", + "--allow-remote-model-egress", + ]) + .stdout(Stdio::null()) + .stderr(Stdio::piped()) + .spawn() + .expect("blocked process starts"), + ); + server + .request + .recv_timeout(TEST_TIMEOUT) + .expect("request arrives"); + child.terminate(); + let run_id = extract_run_id(&child.stderr_text()); + server.release.send(()).expect("server releases"); + server.handle.join().expect("server stops"); + let database = run_database(&state_root, &run_id); + let before = SqliteRunStore::open_existing_read_only(&database) + .expect("store opens") + .load() + .expect("snapshot loads") + .expect("run exists"); + let report = recovery_report(&state_root, &run_id); + assert_eq!(report["active_call"]["status"]["status"], "reserved"); + let call_id = report["active_call"]["call_id"].as_str().expect("call ID"); + let lease = xgeny_runtime::LocalRunLease::try_acquire( + &run_id, + state_root.join("runs").join(&run_id).join("run.lock"), + ) + .expect("test holds lease"); + for options in [ + vec!["recover", &run_id], + vec!["recover", &run_id, "--discard-model-call", call_id], + ] { + let busy = xgeny(&state_root) + .args(options) + .bounded_output() + .expect("busy returns"); + assert_exit(&busy, 75); + } + drop(lease); + + // Integrity preflight cannot mutate the journal or overwrite a damaged manifest. + let manifest_path = state_root.join("runs").join(&run_id).join("manifest.json"); + let manifest = fs::read(&manifest_path).expect("manifest readable"); + fs::write(&manifest_path, b"{}").expect("corrupt fixture manifest"); + let corrupt = xgeny(&state_root) + .args(["recover", &run_id, "--discard-model-call", call_id]) + .bounded_output() + .expect("corrupt returns"); + assert_exit(&corrupt, 70); + assert_eq!(fs::read(&manifest_path).expect("manifest readable"), b"{}"); + fs::write(&manifest_path, manifest).expect("restore fixture manifest"); + let unchanged = SqliteRunStore::open_existing_read_only(&database) + .expect("store opens") + .load() + .expect("snapshot loads") + .expect("run exists"); + assert_eq!(unchanged, before); + + // Recovery does not require or recreate a missing physical workspace. + fs::rename(&workspace, fixture.path().join("moved-workspace")).expect("fixture moves"); + let result = xgeny(&state_root) + .args(["recover", &run_id, "--discard-model-call", call_id]) + .bounded_output() + .expect("discard returns"); + assert_exit(&result, 0); + let after = SqliteRunStore::open_existing_read_only(&database) + .expect("store opens") + .load() + .expect("snapshot loads") + .expect("run exists"); + assert_eq!(after.records.len(), before.records.len() + 1); + assert!(after.state.steps.is_empty()); + let closed = recovery_report(&state_root, &run_id); + assert_eq!(closed["reserved_calls"], 1); + assert_eq!(closed["settled_calls"], 1); + assert_eq!(closed["unknown_calls"], 0); + assert!(closed["active_call"].is_null()); + assert!(!workspace.exists()); +} + +#[test] +fn recovery_rejects_missing_runs_without_creating_state_or_echoing_arguments() { + let fixture = tempdir().expect("test directory should exist"); + let state_root = fixture.path().join("missing-state"); + for run_id in ["../invalid-run", "run-00000000000000000000000000000000"] { + let result = xgeny(&state_root) + .args(["recover", run_id, "--discard-model-call", "invalid-call"]) + .bounded_output() + .expect("invalid recovery returns"); + assert_exit(&result, 64); + assert!(result.stdout.is_empty()); + assert_eq!( + stderr(&result).trim(), + "XGENY_ERROR code=configuration_mismatch" + ); + } + assert!(!state_root.exists()); +} + +#[test] +#[allow(clippy::too_many_lines)] +fn explicit_recovery_never_refunds_exhausted_budget_or_discards_a_newer_call() { + let fixture = tempdir().expect("test directory should exist"); + let state_root = fixture.path().join("state"); + let workspace = fixture.path().join("workspace"); + fs::create_dir(&workspace).expect("workspace creates"); + fs::write(workspace.join("README.md"), FILE_MARKER).expect("source writes"); + let unavailable = OneTurnServer::spawn_status(503, Vec::new()); + let base_url = unavailable.base_url.clone(); + let first = xgeny(&state_root) + .args([ + "run", + "goal", + "--workspace", + path_text(&workspace), + "--base-url", + &base_url, + "--model", + MODEL, + "--tokenizer", + TOKENIZER, + "--allow-file", + "README.md", + "--allow-remote-model-egress", + ]) + .bounded_output() + .expect("unavailable call returns"); + assert_exit(&first, 30); + unavailable + .request + .recv_timeout(TEST_TIMEOUT) + .expect("first request arrives"); + unavailable.handle.join().expect("first server finishes"); + let run_id = extract_run_id(&stderr(&first)); + let initial = recovery_report(&state_root, &run_id); + assert_eq!(initial["max_model_calls"], 4); + let mut previous_id = String::new(); + for count in 1..=4 { + let active = recovery_report(&state_root, &run_id); + assert_eq!(active["reserved_calls"], count); + let id = active["active_call"]["call_id"] + .as_str() + .expect("active call"); + assert_ne!(id, previous_id); + if !previous_id.is_empty() { + let stale = xgeny(&state_root) + .args(["recover", &run_id, "--discard-model-call", &previous_id]) + .bounded_output() + .expect("stale request returns"); + assert_exit(&stale, 64); + assert_eq!(recovery_report(&state_root, &run_id), active); + } + let discard = xgeny(&state_root) + .args(["recover", &run_id, "--discard-model-call", id]) + .bounded_output() + .expect("explicit discard returns"); + assert_exit(&discard, 0); + let closed = recovery_report(&state_root, &run_id); + assert_eq!(closed["reserved_calls"], count); + assert_eq!(closed["settled_calls"], count); + previous_id = id.to_owned(); + if count < 4 { + let next_server = OneTurnServer::spawn_status(503, Vec::new()); + let next = resume_process( + &state_root, + &run_id, + &workspace, + &next_server.base_url, + "README.md", + ); + assert_exit(&next, 30); + next_server + .request + .recv_timeout(TEST_TIMEOUT) + .expect("one fresh request arrives"); + next_server.handle.join().expect("next server finishes"); + } + } + let exhausted = recovery_report(&state_root, &run_id); + let listener = TcpListener::bind("127.0.0.1:0").expect("new endpoint listens"); + listener + .set_nonblocking(true) + .expect("nonblocking listener"); + let fresh_endpoint = format!( + "http://{}/v1", + listener.local_addr().expect("local address") + ); + let stopped = resume_process( + &state_root, + &run_id, + &workspace, + &fresh_endpoint, + "README.md", + ); + assert_exit(&stopped, 10); + assert_eq!( + listener + .accept() + .expect_err("exhausted budget sends no request") + .kind(), + io::ErrorKind::WouldBlock + ); + assert_eq!(recovery_report(&state_root, &run_id), exhausted); +} + +#[test] +#[allow(clippy::too_many_lines)] +fn explicit_recovery_preserves_budget_and_verified_effect_before_separate_resume() { + let fixture = tempdir().expect("test directory should exist"); + let state_root = fixture.path().join("state"); + let workspace = fixture.path().join("workspace"); + fs::create_dir(&workspace).expect("workspace should create"); + fs::write(workspace.join("README.md"), FILE_MARKER).expect("fixture should write"); + let server = OneTurnServer::spawn(plan_response()); + let output = xgeny(&state_root) + .args([ + "run", + "goal", + "--workspace", + path_text(&workspace), + "--base-url", + &server.base_url, + "--model", + MODEL, + "--tokenizer", + TOKENIZER, + "--allow-file", + "README.md", + "--allow-read", + "--allow-remote-model-egress", + ]) + .bounded_output() + .expect("one verified read then unavailable provider should return"); + assert_exit(&output, 30); + server + .request + .recv_timeout(TEST_TIMEOUT) + .expect("first request should arrive"); + server.handle.join().expect("server should finish"); + let run_id = extract_run_id(&stderr(&output)); + let database = run_database(&state_root, &run_id); + let before = SqliteRunStore::open_existing_read_only(&database) + .expect("store should open") + .load() + .expect("snapshot should load") + .expect("run exists"); + assert_eq!(before.state.steps.len(), 1); + assert!( + before + .state + .steps + .values() + .all(|step| step.status == xgeny_workgraph::StepStatus::Completed) + ); + let report = recovery_report(&state_root, &run_id); + assert_eq!(report["reserved_calls"], 2); + assert_eq!(report["settled_calls"], 1); + assert_eq!(report["active_call"]["status"]["status"], "unknown"); + let call_id = report["active_call"]["call_id"] + .as_str() + .expect("active ID exists"); + for wrong_id in [ + "", + "not-a-call", + "control\nvalue", + &"x".repeat(4096), + &format!("model-call-{}", "0".repeat(64)), + ] { + let wrong = xgeny(&state_root) + .args(["recover", &run_id, "--discard-model-call", wrong_id]) + .bounded_output() + .expect("wrong ID returns"); + assert_exit(&wrong, 64); + assert_eq!( + stderr(&wrong).trim(), + "XGENY_ERROR code=configuration_mismatch" + ); + assert!(wrong.stdout.is_empty()); + } + let unchanged = SqliteRunStore::open_existing_read_only(&database) + .expect("store should open") + .load() + .expect("snapshot loads") + .expect("run exists"); + assert_eq!(unchanged, before); + + let discarded = xgeny(&state_root) + .env("XGENY_OPENAI_INFERENCE_TIMEOUT", "unused-invalid") + .args(["recover", &run_id, "--discard-model-call", call_id]) + .bounded_output() + .expect("discard returns"); + assert_exit(&discarded, 0); + let closed: Value = serde_json::from_slice(&discarded.stdout).expect("JSON report"); + assert_eq!(closed["discarded_call_id"], call_id); + assert_eq!(closed["reserved_calls"], 2); + assert_eq!(closed["settled_calls"], 2); + assert_eq!(closed["max_model_calls"], report["max_model_calls"]); + assert!(closed["active_call"].is_null()); + let after = SqliteRunStore::open_existing_read_only(&database) + .expect("store should open") + .load() + .expect("snapshot loads") + .expect("run exists"); + assert_eq!(after.records.len(), before.records.len() + 1); + assert_eq!(after.state.steps, before.state.steps); + assert!(matches!( + after.records.last().expect("settlement exists").event.body, + RunEventBody::ModelCallSettled { + settlement: xgeny_workgraph::ModelCallSettlement::Abandoned { + reason: xgeny_workgraph::ModelCallAbandonmentReason::RecoveryDiscarded + }, + .. + } + )); + let repeated = xgeny(&state_root) + .args(["recover", &run_id, "--discard-model-call", call_id]) + .bounded_output() + .expect("repeat returns"); + assert_exit(&repeated, 64); + let no_egress = xgeny(&state_root) + .args(["resume", &run_id]) + .bounded_output() + .expect("resume requires separate consent"); + assert_exit(&no_egress, 10); + let unchanged = SqliteRunStore::open_existing_read_only(&database) + .expect("store should open") + .load() + .expect("snapshot loads") + .expect("run exists"); + assert_eq!(unchanged, after); + + // Remove the source to prove the already verified effect is not re-executed. + fs::remove_file(workspace.join("README.md")).expect("fixture source can be removed"); + let completion = OneTurnServer::spawn(completion_response()); + let resumed = resume_process( + &state_root, + &run_id, + &workspace, + &completion.base_url, + "README.md", + ); + assert_exit(&resumed, 0); + let request = completion + .request + .recv_timeout(TEST_TIMEOUT) + .expect("fresh reservation calls provider"); + assert!(String::from_utf8_lossy(&request).contains(FILE_MARKER)); + completion + .handle + .join() + .expect("completion server should finish"); + let final_report = recovery_report(&state_root, &run_id); + assert_eq!(final_report["reserved_calls"], 3); + assert_eq!(final_report["settled_calls"], 3); + let store = SqliteRunStore::open_existing_read_only(&database).expect("final store opens"); + assert_eq!( + store + .load_execution_receipts() + .expect("receipts load") + .len(), + 1 + ); + let stale = xgeny(&state_root) + .args(["recover", &run_id, "--discard-model-call", call_id]) + .bounded_output() + .expect("completed run cannot discard"); + assert_exit(&stale, 64); +} + #[test] #[allow(clippy::too_many_lines)] fn outcome_commit_failure_is_immediately_uncertain_and_offline_resume_never_reexecutes() { @@ -715,6 +1129,10 @@ fn outcome_commit_failure_is_immediately_uncertain_and_offline_resume_never_reex .expect("offline effect recovery should run"); assert_exit(&recovered, 30); assert!(stderr(&recovered).contains("reason=effect_outcome_unknown")); + assert_eq!( + recovery_report(&state_root, &run_id)["effect_recovery_required"], + true + ); let after_mark_store = SqliteRunStore::open_existing(&database).expect("unknown store should reopen"); let after_mark = after_mark_store diff --git a/docs/adr/0040-explicit-local-model-call-recovery.md b/docs/adr/0040-explicit-local-model-call-recovery.md new file mode 100644 index 0000000..827a23a --- /dev/null +++ b/docs/adr/0040-explicit-local-model-call-recovery.md @@ -0,0 +1,59 @@ +# ADR-0040: Explicit local model-call recovery + +- Date: 2026-09-15 +- Status: Proposed +- Extends: ADR-0016, ADR-0023 +- Protocol / journal / SQLite schema changes: none + +## Problem + +A killed CLI can leave a Reserved model call; ordinary resume classifies it as +Unknown/Interrupted and refuses implicit retry. Core already provides +`AgentLoop::abandon_model_call`, but an external harness cannot invoke that safe +operation through the public binary. Editing SQLite, declaring success from a +partial output, or starting a replacement Run would evade the durable boundary. + +## Decision + +Expose a separate offline command: + +```text +xgeny recover RUN_ID +xgeny recover RUN_ID --discard-model-call EXACT_CALL_ID +``` + +The first form verifies the manifest and journal under the exclusive Run lease +and prints a bounded JSON report without changing journal state. It includes +the active call's Core-generated ID and Reserved/Unknown status, consumed and +maximum possible-send counters, settled/unknown counters, and whether effect +recovery is still required. It is not a transcript or provider diagnostic dump. + +The second form is explicit operator/host authority to stop accepting the named +unresolved call's response. Validate exact active identity before opening the +store writable, recheck the verified state after reopening, and use the existing +Core API/CAS to append one RecoveryDiscarded settlement. Reserved and Unknown +are both eligible. Do not invent an Unknown reason for a directly discarded +reservation. A wrong/stale ID, a completed Run, or no active call fails closed. +Repeating a successful command must not settle a later call or append again. + +Discard is NOT evidence that no request was sent or billed. No reserved slot, +accepted-turn budget, or effect record is refunded/reset. A late old response +cannot be applied. Neither form resolves credentials, reads the workspace, +contacts a provider, invokes tools, completes the Run, nor resumes it. A separate +ordinary resume requires the original workspace identity, catalogs, model +profile, approvals and remaining original budget. Unknown effects remain blocked. + +JSON output acknowledges the inspected/committed state, not provider settlement. +If output delivery fails after commit, inspect again; do not blindly retry with a +different ID. There is no wildcard, automatic discard, budget override, or Run +replacement command. Default resume behavior and existing exit codes stay intact. + +## Verification + +Exercise the public binary against local fixture providers: inspection leaves the +journal unchanged, discard spends no new reservation, wrong/stale IDs and held +leases cannot mutate, a verified tool result is not replayed after discard and +explicit resume, consumed reservations remain consumed, and no credential/profile +configuration is needed for recovery. Retain Core's stale-response and exhausted +budget regression tests. Real-provider recovery and missing-workspace restoration +are separate harness acceptance work, not proven by this CLI change. diff --git a/docs/development/durable-model-call-lifecycle.md b/docs/development/durable-model-call-lifecycle.md index cb90c0e..b372846 100644 --- a/docs/development/durable-model-call-lifecycle.md +++ b/docs/development/durable-model-call-lifecycle.md @@ -204,6 +204,11 @@ Explicit discard는 previous request가 전송되지 않았다는 증명이 아 현재 public recovery entry point는 `AgentLoop::abandon_model_call(..., call_id)`다. Exact active call만 `ModelCallSettlement::Abandoned { reason: ModelCallAbandonmentReason::RecoveryDiscarded }`를 담은 `ModelCallSettled` event로 닫고 `ModelCallAbandoned` tick을 반환한다. 자동 timeout worker나 implicit retry가 이 API를 대신 호출해서는 안 된다. +CLI에서는 `xgeny recover RUN_ID`로 오프라인 조회하고, 정확한 active call ID를 명시한 +`xgeny recover RUN_ID --discard-model-call CALL_ID`로 같은 Core API를 호출한다. +모델·도구 실행과 별도이며 [복구 절차](local-model-call-recovery.md)의 기존 예산·권한·workspace +보존 조건을 따른다. Default `resume`의 자동 재시도 금지는 변하지 않는다. + ### Timeout과 unavailable 기존 port failure 이름만 보고 전송 여부를 추정하지 않는다. Timeout과 delivery가 불명확한 unavailable은 Unknown이다. Provider adapter가 request가 전송되기 전에 실패했음을 future contract로 증명하더라도 이번 기본형의 보수적 상한을 약화하지 않는다. diff --git a/docs/development/local-model-call-recovery.md b/docs/development/local-model-call-recovery.md new file mode 100644 index 0000000..325b856 --- /dev/null +++ b/docs/development/local-model-call-recovery.md @@ -0,0 +1,90 @@ +# Offline model-call recovery + +This source change exposes the existing Core discard operation (ADR-0016) through +the CLI. It does not add a provider retry policy. See [ADR-0040](../adr/0040-explicit-local-model-call-recovery.md). + +## Inspect, decide, then act on one exact call + +Use the same private `XGENY_STATE_HOME` as the original Run: + +```text +xgeny recover RUN_ID +xgeny recover RUN_ID --discard-model-call CALL_ID_FROM_INSPECTION +``` + +The first command never appends journal events, including for Reserved calls. +Both forms hold the exclusive Run lease; a live owner returns exit 75 rather than +being interrupted. They need neither the model endpoint/credentials nor workspace. +The output is one bounded JSON object with `format_version: 1`: + +- `run_id`, `journal_sequence`, `journal_head_digest`: inspected/committed identity. +- `active_call`: null, or `{call_id, status}` where status is + `{status: "reserved"}` or `{status: "unknown", reason: "interrupted"}` (also + `timeout` / `transport_unavailable`). The ID is generated by Core, not a provider ID. +- `lifecycle_configured`, `reserved_calls`, `max_model_calls`, `settled_calls`, + `unknown_calls`: original possible-send budget and its durable counters. +- `effect_recovery_required`: an executing/unknown/reconciling/manual effect is + still unresolved; model-call discard does not resolve it. +- `discarded_call_id`: null for inspection; exact ID for a successful discard. + +Without a configured lifecycle the counters are zero, not a claim of zero +historical calls. Legacy accounting limitations still apply (ADR-0016). +There are no prompts, responses, credentials, file paths, or raw provider errors +in this report. Exit 0 acknowledges this offline command only, **not Run completion**. + +Discard explicitly stops accepting an uncertain response. It does not prove that +the provider never processed or billed it. Core appends one `ModelCallSettled` +with `Abandoned/RecoveryDiscarded`; it never erases the reservation. A directly +discarded Reserved call is not falsely reclassified as a known provider outcome. +Wrong/stale/missing IDs return exit 64 without appending; damaged state returns +70. Repeating the same discard never settles a newer call. If stdout failed after +commit, inspect again instead of assuming rollback. + +## Separate, bounded continuation + +Ordinary `resume` is unchanged: an unresolved call still returns recovery-required +(30) without a new model request. After an explicit discard, a separate `resume` +may progress only under the original physical workspace identity, exact catalogs, +model/request profile, fresh action/egress consent, and remaining original budget. +Earlier passed-Receipt effects and durable tool outputs are preserved, not replayed. +An exhausted budget cannot be replenished by discarding. Unknown effects stay blocked. + +An external harness must record who authorized discarding which call, bind the +returned head to its own failure/budget record, and retain the original workspace +and scope for continuation. Do not put this command into an unconditional timeout +retry handler. Do not manufacture completion from a partially written artifact, +change inference limits bound into the Run's request profile, or start a replacement +Run to hide the failure. A deleted workspace is a separate recovery limitation; +this command neither restores it nor relaxes physical-identity checks. + +## Local verification + +`crates/xgeny-cli/tests/public_run_resume.rs` exercises separate binary processes +with loopback fixture providers, including an actual process kill. It checks +read-only inspection, exact discard, lease/integrity errors, consumed reservations, +and a verified read followed by failed inference, discard, and fresh completion +without repeating the read. These are not real-provider/ML recovery claims. + +Run format, clippy, workspace tests, protocol check and release build using the +[engineering method](engineering-method.md). Platform integration and production +rollout require their own tests; macOS/Windows are checked by repository CI. + +### 2026-09-15 verification record (Linux x86_64 / Rust 1.98.0) + +- Red: the new recovery E2E failed on the baseline with `unrecognized subcommand + 'recover'` (exit 2). +- Green: all 10 public run/resume tests passed, including four new recovery tests. +- Complete workspace test command passed with an isolated executable tmpfs test + directory. An earlier overlapping run failed an executable-material check while + another test build replaced the cataloged debug binary; the final gate used a + fixed binary, no overlapping rebuild, and passed that test unchanged. Do not + weaken executable identity checks to accommodate concurrent test rebuilds. +- `cargo fmt --all -- --check`, workspace clippy with `-D warnings`, bundled + protocol check, release build, RC3 public-docs and release-workflow checks passed. +- An optimized binary inspected and discarded one Interrupted call in an isolated + copy of an existing failed Run: journal 11→12, reserved calls 2→2, settled calls + 1→2, original maximum 16 unchanged. Repeated discard returned 64; subsequent + no-egress resume returned 10. Original DB/WAL/project hashes stayed unchanged. +- No real provider call, ML training, production binary replacement, or actual + failed-project continuation was performed. macOS/Windows and remote PR CI are + separate checks, not included in these local results.