diff --git a/Cargo.lock b/Cargo.lock index 1105a7d3..63c72d83 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -425,6 +425,12 @@ dependencies = [ "r-efi 6.0.0", ] +[[package]] +name = "glob" +version = "0.3.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e4eba85ea1d0a966a983acd07deee566e67395d2d96b6fb39e62b5a833f1eb0b" + [[package]] name = "h2" version = "0.4.18" @@ -471,6 +477,12 @@ dependencies = [ "hashbrown 0.17.1", ] +[[package]] +name = "hex" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7f24254aa9a54b5c858eaee2f5bccdb46aaf0e486a595ed5fd8f86ba55232a70" + [[package]] name = "http" version = "1.4.2" @@ -1614,11 +1626,16 @@ dependencies = [ "anyhow", "async-trait", "chrono", + "glob", + "hex", + "parking_lot", "rusqlite", "serde", "serde_json", + "sha2", "tempfile", "tinyagents-harness", + "tinytools-agent 0.4.1", "tokio", "tracing", ] diff --git a/crates/tinyagents-session/Cargo.toml b/crates/tinyagents-session/Cargo.toml index 99e9ab4b..654c2ae8 100644 --- a/crates/tinyagents-session/Cargo.toml +++ b/crates/tinyagents-session/Cargo.toml @@ -16,7 +16,14 @@ rusqlite = { workspace = true } serde = { workspace = true } serde_json = { workspace = true } tinyagents-harness = { path = "../tinyagents-harness", version = "2.1.2", default-features = false, features = ["sqlite"] } +tinytools-agent = { path = "../../vendor/tinytools/crates/tinytools-agent", version = "0.4.1", default-features = false } +glob = "0.3" +hex = "0.4" +sha2 = "0.11" +parking_lot = "0.12" +tempfile = { workspace = true } tracing = { workspace = true } +tokio = { workspace = true, default-features = true, features = ["sync"] } [features] default = [] @@ -26,7 +33,6 @@ default = [] tracing = ["tinyagents-harness/tracing"] [dev-dependencies] -tempfile = { workspace = true } tokio = { workspace = true, default-features = true, features = ["macros", "rt-multi-thread"] } [lints] diff --git a/crates/tinyagents-session/src/lib.rs b/crates/tinyagents-session/src/lib.rs index da80fb6b..9f494269 100644 --- a/crates/tinyagents-session/src/lib.rs +++ b/crates/tinyagents-session/src/lib.rs @@ -83,6 +83,7 @@ pub mod run_ledger; mod store; pub mod testkit; pub mod transcript; +pub mod turn_state; pub mod types; pub use tinyagents_harness::error::{Result, TinyAgentsError}; diff --git a/crates/tinyagents-session/src/transcript.rs b/crates/tinyagents-session/src/transcript.rs index 20fbf5ca..2f23f52f 100644 --- a/crates/tinyagents-session/src/transcript.rs +++ b/crates/tinyagents-session/src/transcript.rs @@ -110,6 +110,7 @@ mod adoption; mod history; +pub mod import; mod jsonl; mod legacy_md; mod markdown; @@ -119,6 +120,7 @@ mod reader; mod session; mod thread_lookup; mod types; +pub mod view; mod writer; pub use adoption::{SessionAdoption, adopt_legacy_session_transcripts}; diff --git a/crates/tinyagents-session/src/transcript/import/README.md b/crates/tinyagents-session/src/transcript/import/README.md new file mode 100644 index 00000000..4042b398 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/README.md @@ -0,0 +1,90 @@ +# transcript::import + +One-time migration of legacy session transcripts into TinyAgents +`Store`/`AppendStore` records ([`ops::run_import`]). Hosts expose it as an +explicit command, never a boot hook. The module also owns the live +dual-write and shadow-read paths that keep new turns landing in the same +store layout (`live.rs`), and [`ops::open_session_stores`], which every other +TinyAgents-store consumer reuses. The host supplies a +[`convert::JournalProjector`] so journal records carry its message metadata; +whether the live paths run is the host's decision. + +> The migration was specified in a design doc for issue #4249 that is not +> checked into this repository; `types.rs` still refers to it when explaining +> why stream names are dot-separated. + +## Sources and destination + +- Transcript JSONL under `session_raw/`: the current flat layout + (`session_raw/{stem}.jsonl`) and the legacy date-folder layout + (`session_raw/{DDMMYYYY}/{stem}.jsonl`). +- Markdown-only sessions under `sessions/{dir}/{stem}.md` that have no JSONL + twin. A Markdown file next to a JSONL source is recorded as a companion + pointer only, never read. +- Precedence per session stem: flat JSONL beats legacy-dir JSONL beats + Markdown-only (`scan.rs::discover_sources`); a duplicate stem is a warning, + not a second item. +- `session_db/sessions.db` is opened read-only, when present, to join + `agent_runs` ids onto the descriptor and cross-check parent lineage. The + stem chain (`a__b` maps to parent `a`) wins over a disagreeing ledger. + +All writes land under `{workspace}/tinyagents_store/`: + +- `kv/sessions/{sanitized stem}.json`: `SessionDescriptor` compatibility + record mapping the OpenHuman session key to TinyAgents identifiers. +- `kv/migration_items/{sha256(relative source path)}.json`: per-item + idempotency ledger (`ItemLedgerRecord`). +- `kv/migrations/session_import_v1.json`: global run marker. +- `journal/session.{stem}.messages.jsonl`: message journal, one + `StoreRecord` per line via TinyAgents `JsonlAppendStore`. + +Source files are never mutated or deleted. + +## Idempotency + +- A full run (no `only`) that is not a dry run writes the global marker + (`MARKER_KEY`). A later full, non-forced, non-dry run sees the marker and + returns `already_done` without scanning. +- Independently, every written source gets a ledger entry keyed by the sha256 + of its workspace-relative path holding `IMPORT_VERSION`, size, and mtime. + A source whose fingerprint still matches is `skipped_unchanged`, even under + `only` or after the global marker is gone. +- `force` ignores both the marker and the ledger. +- `dry_run` bypasses the marker fast path, reports `would_import` per item, + and writes nothing. +- Re-import is a full rewrite: the journal stream file is deleted and + re-appended because `JsonlAppendStore` has no truncate. + +## Key files + +| File | Role | +| --- | --- | +| `mod.rs` | Module docs and re-exports (`ImportOptions`, `ImportSummary`, controller registration). | +| `types.rs` | Serde types: `ImportOptions`, `ImportSummary`, `ItemReport`, `SessionDescriptor`, `JournalMessage`, `ItemLedgerRecord`, and the store-layout constants (`KV_SUBDIR`, `JOURNAL_SUBDIR`, `NS_*`, `MARKER_KEY`, `IMPORT_VERSION`). | +| `scan.rs` | `discover_sources` walks `session_raw/` and `sessions/`, deduping by stem per the precedence order above. | +| `convert.rs` | Pure helpers: `parent_session_key` stem lineage, `sanitize_store_name`, `stream_name`, `effective_thread_id` (synthesizes `imported-{stem}` when `_meta` has none), `build_descriptor`, `journal_messages`. | +| `ops.rs` | `run_import` scans, reads all, plans/writes per item, then writes the marker; `open_session_stores` opens the shared KV/journal handles over `{workspace}/tinyagents_store/{kv,journal}`. | +| `live.rs` | Dual-write of each persisted turn into the same layout (`write_live_turn`), gated by `AgentConfig::session_dual_write` (default on) with the `OPENHUMAN_SESSION_DUAL_WRITE` env kill switch; store-backed shadow read of the same session (`shadow_read_compare`), gated by `AgentConfig::session_shadow_reads` (default on) with `OPENHUMAN_SESSION_SHADOW_READS`; `session_kv_store` exposes the KV store on `RunContext.stores` as `TINYAGENTS_SESSION_KV_STORE`. Env vars can only force off, never on. Errors are logged and swallowed by callers; the legacy transcript stays authoritative for both reads and writes. | +| `schemas.rs` | `session_import.run` `ControllerSchema` and handler; resolves the workspace from config unless `workspace` is passed. | +| `*_tests.rs` | Sibling test suites for `convert`, `live`, `ops`, `schemas`. | + +## RPC + +`session_import.run` accepts `dry_run`, `only` (glob over session stems), +`force`, `verbose`, and an optional `workspace` override, and returns an +`ImportSummary`. Registered via `all_session_import_registered_controllers` +in `crates/openhuman-core/src/core/all.rs`. + +## Used by + +- `crates/openhuman-core/src/core/all.rs`: registers the controller. +- `tinyagents_session::transcript`: the legacy transcript readers + (`read_transcript`, `read_transcript_legacy_md`) the importer converts from. +- `crates/openhuman-core/src/agent/session_host/turn/session_io/transcript_persist.rs`: + calls `live::write_live_turn` after each transcript write and + `live::shadow_read_compare` after each transcript load, both on background tasks. +- `crates/openhuman-core/src/agent/tinyagents/turn_runner.rs`: registers + `live::session_kv_store` on the per-turn `RunContext`. +- `open_session_stores` is reused by `agent/tinyagents/{journal,reaper,todos,replay/ops}.rs` + and `threads/goals/migration.rs` so every TinyAgents-store consumer shares + one layout. diff --git a/crates/tinyagents-session/src/transcript/import/convert.rs b/crates/tinyagents-session/src/transcript/import/convert.rs new file mode 100644 index 00000000..b3248215 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/convert.rs @@ -0,0 +1,139 @@ +//! Pure conversion helpers: stem lineage, stream naming, descriptor +//! assembly, and message-record projection. + +use crate::transcript::{SessionTranscript, TranscriptMessage}; +use sha2::{Digest, Sha256}; + +use super::types::{ + DescriptorImport, DescriptorSource, DescriptorUsage, IMPORT_VERSION, JournalMessage, + SessionDescriptor, +}; + +/// Parent session key from the `__` stem chain. +/// +/// Stems are `{parent_chain}__{unix_ts}_{agent_id}`; the parent key is +/// everything before the **last** `__`. Roots (no `__`) have no parent. +pub fn parent_session_key(stem: &str) -> Option { + stem.rfind("__").map(|idx| stem[..idx].to_string()) +} + +/// Sanitize a string into the TinyAgents store-name alphabet (ASCII +/// alphanumerics, `-`, `_`, `.`); anything else becomes `_`. Empty input +/// becomes `"session"` to match the transcript layer's fallback. +pub fn sanitize_store_name(name: &str) -> String { + let cleaned: String = name + .chars() + .map(|c| { + if c.is_ascii_alphanumeric() || c == '-' || c == '_' || c == '.' { + c + } else { + '_' + } + }) + .collect(); + if cleaned.is_empty() { + "session".to_string() + } else if cleaned.bytes().all(|b| b == b'.') { + let digest = Sha256::digest(name.as_bytes()); + format!("session_{}", hex::encode(&digest[..6])) + } else if cleaned != name { + let digest = Sha256::digest(name.as_bytes()); + format!("{cleaned}_{}", hex::encode(&digest[..6])) + } else { + cleaned + } +} + +/// Journal stream name for a session. +/// +/// Per-session streams (`session.{stem}.messages`) rather than per-thread: +/// multiple transcript files can share one `_meta.thread_id`, and appending +/// them into a shared stream would interleave sessions. The descriptor +/// carries `thread_id` so thread-level views can still be projected. +pub fn stream_name(session_key: &str) -> String { + format!("session.{}.messages", sanitize_store_name(session_key)) +} + +/// Effective thread id for a transcript: `_meta.thread_id` when present, +/// otherwise a synthesized stable id. Returns `(thread_id, synthesized)`. +pub fn effective_thread_id(session_key: &str, meta_thread_id: Option<&str>) -> (String, bool) { + match meta_thread_id { + Some(t) if !t.is_empty() => (t.to_string(), false), + _ => ( + format!("imported-{}", sanitize_store_name(session_key)), + true, + ), + } +} + +/// Build the `sessions/{session_key}` descriptor from a parsed transcript. +#[allow(clippy::too_many_arguments)] +pub fn build_descriptor( + session_key: &str, + transcript: &SessionTranscript, + thread_id: String, + thread_id_synthesized: bool, + run_ids: Vec, + source: DescriptorSource, + imported_at: String, + warnings: usize, +) -> SessionDescriptor { + let meta = &transcript.meta; + SessionDescriptor { + session_key: session_key.to_string(), + parent_session_key: parent_session_key(session_key), + thread_id, + thread_id_synthesized, + task_id: meta.task_id.clone(), + run_ids, + stream: stream_name(session_key), + dispatcher: meta.dispatcher.clone(), + agent_name: meta.agent_name.clone(), + agent_id: meta.agent_id.clone(), + agent_type: meta.agent_type.clone(), + provider: meta.provider.clone(), + model: meta.model.clone(), + created: meta.created.clone(), + updated: meta.updated.clone(), + turn_count: meta.turn_count, + usage: DescriptorUsage { + input: meta.input_tokens, + output: meta.output_tokens, + cached_input: meta.cached_input_tokens, + cost_usd: meta.charged_amount_usd, + }, + source, + import: DescriptorImport { + version: IMPORT_VERSION, + imported_at, + warnings, + }, + } +} + +/// Host-supplied projection of one durable transcript row into its journal +/// record. The journal must carry whatever host metadata the host folds into a +/// row when it reads a transcript back (OpenHuman reconstructs +/// `openhuman_turn_usage` and friends), so the importer takes the projection as +/// a parameter instead of assuming a message type. +pub type JournalProjector = fn(TranscriptMessage) -> JournalMessage; + +/// Neutral projector: copies `id`, `role`, `content` and `extra_metadata` +/// verbatim and adds nothing. For hosts with no message-shape sidecars, and for +/// tests. +pub fn plain_journal_message(message: TranscriptMessage) -> JournalMessage { + JournalMessage { + id: message.id, + role: message.role, + content: message.content, + extra_metadata: message.extra_metadata, + } +} + +/// Project a transcript's messages into journal records. +pub fn journal_messages( + transcript: &SessionTranscript, + project: JournalProjector, +) -> Vec { + transcript.messages.iter().cloned().map(project).collect() +} diff --git a/crates/tinyagents-session/src/transcript/import/convert_test.rs b/crates/tinyagents-session/src/transcript/import/convert_test.rs new file mode 100644 index 00000000..03b76f65 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/convert_test.rs @@ -0,0 +1,47 @@ +use super::convert::*; + +#[test] +fn parent_key_follows_last_double_underscore() { + assert_eq!(parent_session_key("1719_orchestrator"), None); + assert_eq!( + parent_session_key("1719_orchestrator__1720_researcher"), + Some("1719_orchestrator".to_string()) + ); + assert_eq!( + parent_session_key("a__b__c"), + Some("a__b".to_string()), + "two-level chains keep the full parent chain as the parent key" + ); +} + +#[test] +fn sanitize_maps_unsafe_bytes_and_guards_dots() { + assert_eq!(sanitize_store_name("1719_agent"), "1719_agent"); + assert!(sanitize_store_name("a/b:c d").starts_with("a_b_c_d_")); + assert_eq!(sanitize_store_name(""), "session"); + assert!(sanitize_store_name("..").starts_with("session_")); +} + +#[test] +fn thread_id_synthesized_only_when_absent() { + assert_eq!( + effective_thread_id("s1", Some("t-1")), + ("t-1".to_string(), false) + ); + assert_eq!( + effective_thread_id("s1", None), + ("imported-s1".to_string(), true) + ); + assert_eq!( + effective_thread_id("s1", Some("")), + ("imported-s1".to_string(), true) + ); +} + +#[test] +fn stream_name_is_store_safe() { + assert_eq!( + stream_name("1719_a__1720_b"), + "session.1719_a__1720_b.messages" + ); +} diff --git a/crates/tinyagents-session/src/transcript/import/live.rs b/crates/tinyagents-session/src/transcript/import/live.rs new file mode 100644 index 00000000..403f5965 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/live.rs @@ -0,0 +1,241 @@ +//! Live dual-write of new session turns into the TinyAgents store, and the +//! store-backed shadow read that checks it against the legacy transcript. +//! +//! Additive and best-effort: the legacy `session_raw/*.jsonl` transcript stays +//! the primary and authoritative writer; [`write_live_turn`] mirrors each +//! *already-persisted* turn into the same store layout the importer produces +//! (`{workspace}/tinyagents_store/{kv,journal}`), reusing [`super::convert`] +//! normalization so live and imported records are shape-identical. Whether the +//! mirror or the shadow read runs at all is the host's decision (config flag, +//! kill-switch env var); this module only does the work. A store-write failure +//! must never fail or alter a chat turn: the caller treats every error as +//! non-fatal (log + swallow). + +use std::path::Path; +use std::sync::OnceLock; + +use anyhow::{Context, Result}; +use tinyagents_harness::store::{AppendStore, Store}; + +use crate::transcript::SessionTranscript; + +use super::convert::{ + JournalProjector, build_descriptor, effective_thread_id, journal_messages, sanitize_store_name, + stream_name, +}; +use super::ops::{SessionStores, open_session_stores}; +use super::types::{DescriptorSource, JournalMessage, NS_SESSIONS}; + +static LIVE_REWRITE_LOCK: OnceLock> = OnceLock::new(); + +/// Mirror one completed turn's transcript into the TinyAgents store. +/// +/// Best-effort: the caller must treat any returned error as non-fatal (log and +/// swallow) — it never fails or alters the legacy chat turn. +/// +/// Mirrors the importer's **full-rewrite** semantics: the legacy JSONL +/// transcript is rewritten in full on every turn (not appended), and the +/// `JsonlAppendStore` has no truncate, so the journal stream file is dropped and +/// re-appended each turn. This keeps the store stream shape-identical to an +/// import of the final JSONL. The descriptor is upserted in `NS_SESSIONS` +/// exactly as the importer would, reusing [`build_descriptor`]. +pub async fn write_live_turn( + workspace: &Path, + session_key: &str, + transcript: &SessionTranscript, + project: JournalProjector, +) -> Result<()> { + let _rewrite_guard = LIVE_REWRITE_LOCK + .get_or_init(|| tokio::sync::Mutex::new(())) + .lock() + .await; + tracing::debug!( + "[session-store] dual-write enter stem={session_key} workspace={} messages={}", + workspace.display(), + transcript.messages.len() + ); + + let SessionStores { + kv, + journal, + journal_root, + } = open_session_stores(workspace); + + let stream = stream_name(session_key); + + // Full-rewrite parity: drop the stream file, then re-append every message so + // the journal reflects the current transcript exactly (the importer resets + // the same way on re-import). Layout: `{journal_root}/{stream}.jsonl`. + let stream_file = journal_root.join(format!("{stream}.jsonl")); + if stream_file.exists() { + std::fs::remove_file(&stream_file) + .with_context(|| format!("reset journal stream {stream}"))?; + } + + let records = journal_messages(transcript, project); + let message_count = records.len(); + for (idx, record) in records.iter().enumerate() { + let value = serde_json::to_value(record) + .with_context(|| format!("serialize live message {idx}"))?; + journal + .append(&stream, value) + .await + .with_context(|| format!("journal append failed at message {idx}"))?; + } + + // Descriptor: same projection the importer uses. No run-ledger join here + // (live turns have no `agent_runs` link yet) and zero warnings; the source + // pointer records the workspace-relative JSONL twin. + let (thread_id, synthesized) = + effective_thread_id(session_key, transcript.meta.thread_id.as_deref()); + let descriptor = build_descriptor( + session_key, + transcript, + thread_id, + synthesized, + Vec::new(), + DescriptorSource { + jsonl: Some(format!("session_raw/{session_key}.jsonl")), + md: None, + }, + chrono::Utc::now().to_rfc3339(), + 0, + ); + let descriptor_key = sanitize_store_name(session_key); + let descriptor_value = + serde_json::to_value(&descriptor).context("serialize live session descriptor")?; + kv.put(NS_SESSIONS, &descriptor_key, descriptor_value) + .await + .context("descriptor write failed")?; + + tracing::debug!( + "[session-store] dual-write exit stem={session_key} stream={stream} messages={message_count}" + ); + Ok(()) +} + +// ───────────────────────────────────────────────────────────────────────────── +// Store-backed SHADOW READ (issue #4249, sessions 04.2 phase 2) +// +// Beside the legacy authoritative transcript reader +// (`session/turn/session_io.rs` → `try_load_session_transcript`), read the +// same session's messages back from the crate journal store, normalize both +// sides through the same `convert` machinery the dual-write uses, compare, and +// log divergence. Legacy stays authoritative: this observes + logs only and +// never affects, fails, or slows the authoritative read. +// ───────────────────────────────────────────────────────────────────────────── + +/// Outcome of one shadow-read comparison. Returned for tests/observability; +/// the compact divergence summary is also logged (`[session_shadow_read]`). +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum ShadowReadOutcome { + /// The store stream rendered exactly the legacy transcript's messages. + Match { messages: usize }, + /// The store stream diverged from the legacy render. Carries only compact + /// counts + the first differing index — never message bodies (PII). + Divergence { + legacy: usize, + shadow: usize, + first_diff: Option, + }, + /// No shadow available: the store read errored, or the stream is + /// empty/absent for a non-empty legacy transcript (e.g. dual-write was off + /// when this session was written). Treated as non-divergent — the legacy + /// read is authoritative regardless. + Unavailable, +} + +/// Read a session's messages back from the crate journal store +/// (`{workspace}/tinyagents_store/journal`, stream `session.{stem}.messages`) +/// as normalized [`JournalMessage`]s — the same shape the importer and live +/// dual-write write. A missing stream yields an empty vec (not an error). +async fn read_shadow_messages(workspace: &Path, session_key: &str) -> Result> { + let SessionStores { journal, .. } = open_session_stores(workspace); + let stream = stream_name(session_key); + let records = journal + .read_from(&stream, 0) + .await + .with_context(|| format!("shadow read of stream {stream}"))?; + let mut out = Vec::with_capacity(records.len()); + for (offset, value) in records { + let msg: JournalMessage = serde_json::from_value(value) + .with_context(|| format!("shadow record shape at offset {offset}"))?; + out.push(msg); + } + Ok(out) +} + +/// Shadow-read the given session back from the store and compare it against the +/// legacy transcript, logging divergence. Legacy stays authoritative — the +/// caller ignores the returned outcome for control flow (it exists for tests / +/// observability). Best-effort: any store-read error is logged at debug and +/// reported as [`ShadowReadOutcome::Unavailable`]; it never breaks or slows the +/// authoritative read. +/// +/// Both sides are normalized through the importer's `convert` machinery +/// ([`journal_messages`]) so live/legacy renders are directly comparable, then +/// compared by message count and normalized content. On mismatch a **compact** +/// summary (counts + first differing index) is warn-logged; message bodies are +/// never emitted (PII). +pub async fn shadow_read_compare( + workspace: &Path, + session_key: &str, + legacy: &SessionTranscript, + project: JournalProjector, +) -> ShadowReadOutcome { + let expected = journal_messages(legacy, project); + tracing::debug!( + "[session_shadow_read] enter stem={session_key} workspace={} legacy_messages={}", + workspace.display(), + expected.len() + ); + + let shadow = match read_shadow_messages(workspace, session_key).await { + Ok(v) => v, + Err(err) => { + tracing::debug!( + "[session_shadow_read] store read error stem={session_key}: {err:#} — no shadow available" + ); + return ShadowReadOutcome::Unavailable; + } + }; + + // Empty/absent store stream against a non-empty legacy transcript: the + // session simply was not mirrored (dual-write off when it was written). + // Treat as "no shadow" rather than a spurious divergence. + if shadow.is_empty() && !expected.is_empty() { + tracing::debug!( + "[session_shadow_read] no store stream stem={session_key} legacy_messages={} — no shadow available", + expected.len() + ); + return ShadowReadOutcome::Unavailable; + } + + if shadow == expected { + tracing::debug!( + "[session_shadow_read] parity OK stem={session_key} messages={}", + expected.len() + ); + return ShadowReadOutcome::Match { + messages: expected.len(), + }; + } + + // Divergence: first index where the two normalized renders differ, or the + // shorter length when one is a strict prefix of the other. Compact only. + let first_diff = expected + .iter() + .zip(shadow.iter()) + .position(|(a, b)| a != b) + .or_else(|| (expected.len() != shadow.len()).then(|| expected.len().min(shadow.len()))); + tracing::warn!( + "[session_shadow_read] DIVERGENCE stem={session_key} legacy_count={} shadow_count={} first_diff={first_diff:?}", + expected.len(), + shadow.len() + ); + ShadowReadOutcome::Divergence { + legacy: expected.len(), + shadow: shadow.len(), + first_diff, + } +} diff --git a/crates/tinyagents-session/src/transcript/import/live_test.rs b/crates/tinyagents-session/src/transcript/import/live_test.rs new file mode 100644 index 00000000..799a3497 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/live_test.rs @@ -0,0 +1,151 @@ +//! Live dual-write, shadow read and persisted-shape tests (neutral projector). + +use std::path::Path; + +use serde_json::json; +use tempfile::TempDir; +use tinyagents_harness::store::{AppendStore, JsonlAppendStore}; + +use super::convert::{JournalProjector, journal_messages, plain_journal_message, stream_name}; +use super::live::{ShadowReadOutcome, shadow_read_compare, write_live_turn}; +use super::ops::store_root; +use super::types::{ItemLedgerRecord, JournalMessage}; +use crate::transcript::{SessionTranscript, read_transcript}; + +const STEM: &str = "1719000000_orchestrator"; + +fn seed(ws: &Path, assistant: &str) -> SessionTranscript { + let dir = ws.join("session_raw"); + std::fs::create_dir_all(&dir).unwrap(); + let path = dir.join(format!("{STEM}.jsonl")); + std::fs::write( + &path, + format!( + "{{\"_meta\":{{\"agent\":\"orchestrator\",\"dispatcher\":\"native\",\ + \"created\":\"2024-01-01T00:00:00Z\",\"updated\":\"2024-01-01T00:05:00Z\",\ + \"turn_count\":1,\"input_tokens\":1,\"output_tokens\":1,\"cached_input_tokens\":0,\ + \"charged_amount_usd\":0.0,\"thread_id\":\"t-1\"}}}}\n\ + {{\"role\":\"user\",\"content\":\"hi\"}}\n\ + {{\"role\":\"assistant\",\"content\":\"{assistant}\"}}\n" + ), + ) + .unwrap(); + read_transcript(&path).unwrap() +} + +async fn readback(ws: &Path) -> Vec { + JsonlAppendStore::new(store_root(ws).join("journal")) + .read_from(&stream_name(STEM), 0) + .await + .unwrap() + .into_iter() + .map(|(_, v)| serde_json::from_value(v).unwrap()) + .collect() +} + +#[tokio::test] +async fn live_write_matches_projection_and_is_rewritten_each_turn() { + let ws = TempDir::new().unwrap(); + let t = seed(ws.path(), "done"); + write_live_turn(ws.path(), STEM, &t, plain_journal_message) + .await + .unwrap(); + write_live_turn(ws.path(), STEM, &t, plain_journal_message) + .await + .unwrap(); + let got = readback(ws.path()).await; + assert_eq!(got, journal_messages(&t, plain_journal_message)); + assert_eq!(got.len(), 2, "second write replaces, not appends"); +} + +#[tokio::test] +async fn shadow_read_match_unavailable_and_divergence() { + let ws = TempDir::new().unwrap(); + let t = seed(ws.path(), "done"); + assert_eq!( + shadow_read_compare(ws.path(), STEM, &t, plain_journal_message).await, + ShadowReadOutcome::Unavailable + ); + write_live_turn(ws.path(), STEM, &t, plain_journal_message) + .await + .unwrap(); + assert_eq!( + shadow_read_compare(ws.path(), STEM, &t, plain_journal_message).await, + ShadowReadOutcome::Match { messages: 2 } + ); + let changed = seed(ws.path(), "different"); + assert_eq!( + shadow_read_compare(ws.path(), STEM, &changed, plain_journal_message).await, + ShadowReadOutcome::Divergence { + legacy: 2, + shadow: 2, + first_diff: Some(1) + } + ); +} + +#[test] +fn host_projector_seam_controls_journal_content() { + let ws = TempDir::new().unwrap(); + let t = seed(ws.path(), "done"); + let project: JournalProjector = |m| { + let mut rec = plain_journal_message(m); + rec.extra_metadata = Some(json!({"host": true})); + rec + }; + let recs = journal_messages(&t, project); + assert!( + recs.iter() + .all(|r| r.extra_metadata == Some(json!({"host": true}))) + ); +} + +/// Persisted shapes are wire contracts: pin them as literal JSON. +#[test] +fn persisted_record_shapes_are_unchanged() { + let msg = JournalMessage { + id: Some("c1".into()), + role: "tool".into(), + content: "x".into(), + extra_metadata: Some(json!({"k": 1})), + }; + assert_eq!( + serde_json::to_value(&msg).unwrap(), + json!({"id":"c1","role":"tool","content":"x","extra_metadata":{"k":1}}) + ); + let bare = JournalMessage { + id: None, + role: "user".into(), + content: "y".into(), + extra_metadata: None, + }; + assert_eq!( + serde_json::to_value(&bare).unwrap(), + json!({"role":"user","content":"y"}) + ); + + let ledger = ItemLedgerRecord { + version: 1, + session_key: "s".into(), + source: "session_raw/s.jsonl".into(), + size: 3, + mtime_ms: 4, + stream: "session.s.messages".into(), + messages: 2, + imported_at: "2024-01-01T00:00:00Z".into(), + }; + assert_eq!( + serde_json::to_value(&ledger).unwrap(), + json!({"version":1,"session_key":"s","source":"session_raw/s.jsonl","size":3,"mtime_ms":4, + "stream":"session.s.messages","messages":2,"imported_at":"2024-01-01T00:00:00Z"}) + ); +} + +#[test] +fn ledger_key_is_lowercase_hex_sha256() { + // sha256("abc") + assert_eq!( + super::ops::ledger_key("abc"), + "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad" + ); +} diff --git a/crates/tinyagents-session/src/transcript/import/mod.rs b/crates/tinyagents-session/src/transcript/import/mod.rs new file mode 100644 index 00000000..1e752052 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/mod.rs @@ -0,0 +1,28 @@ +//! One-time import of legacy sessions into TinyAgents stores, plus the live +//! dual-write and shadow read that keep new turns in the same layout. +//! +//! Legacy transcript JSONL (`session_raw/`, flat and `DDMMYYYY` date folders) +//! and legacy Markdown sessions are normalized into TinyAgents +//! `Store`/`AppendStore` records under `{workspace}/tinyagents_store/`. Sources +//! are never mutated; [`ops::run_import`] is idempotent (global marker + +//! per-item fingerprint ledger). See [`README.md`](README.md) for the +//! source/destination layout. +//! +//! The host supplies one seam: a [`convert::JournalProjector`] that turns a +//! durable [`TranscriptMessage`](crate::transcript::TranscriptMessage) into the +//! journal record, so a host that folds sidecar metadata into its message type +//! keeps that metadata in the journal. [`convert::plain_journal_message`] is the +//! neutral default. Whether the live mirror runs at all is the host's decision. + +pub mod convert; +pub mod live; +pub mod ops; +pub mod scan; +pub mod types; + +#[cfg(test)] +mod convert_test; +#[cfg(test)] +mod live_test; +#[cfg(test)] +mod ops_test; diff --git a/crates/tinyagents-session/src/transcript/import/ops.rs b/crates/tinyagents-session/src/transcript/import/ops.rs new file mode 100644 index 00000000..7a356c6b --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/ops.rs @@ -0,0 +1,511 @@ +//! The importer: scan → plan → (optionally) write TinyAgents store records. +//! +//! Sources are never mutated or deleted. All writes land under +//! `{workspace}/tinyagents_store/`: +//! +//! - `kv/sessions/{session_key}.json` — compatibility descriptor; +//! - `kv/migration_items/{sha256(source)}.json` — per-item idempotency ledger; +//! - `kv/migrations/session_import_v1.json` — global run marker; +//! - `journal/session.{stem}.messages.jsonl` — message journal +//! (`StoreRecord` per line via TinyAgents `JsonlAppendStore`). + +use std::collections::HashMap; +use std::path::{Path, PathBuf}; + +use anyhow::{Context, Result}; +use serde_json::json; +use sha2::{Digest, Sha256}; +use tinyagents_harness::store::{AppendStore, FileStore, JsonlAppendStore, Store}; + +use crate::transcript::{SessionTranscript, read_transcript, read_transcript_legacy_md}; + +use super::convert::{ + JournalProjector, build_descriptor, effective_thread_id, journal_messages, parent_session_key, + stream_name, +}; +use super::scan::{SourceItem, discover_sources}; +use super::types::{ + DescriptorSource, IMPORT_VERSION, ImportOptions, ImportSummary, ItemAction, ItemLedgerRecord, + ItemReport, JOURNAL_SUBDIR, KV_SUBDIR, MARKER_KEY, NS_MIGRATION_ITEMS, NS_MIGRATIONS, + NS_SESSIONS, SourceKind, +}; + +/// Root of the TinyAgents store tree inside a workspace. +pub fn store_root(workspace: &Path) -> PathBuf { + workspace.join("tinyagents_store") +} + +/// The KV + journal store handles opened over a workspace's TinyAgents store +/// tree (`{workspace}/tinyagents_store/{kv,journal}`), plus the journal root +/// for stream-file layout math. +pub struct SessionStores { + pub kv: FileStore, + pub journal: JsonlAppendStore, + pub journal_root: PathBuf, +} + +/// Open the KV + journal stores under `{workspace}/tinyagents_store/{kv,journal}`. +/// +/// Shared by the one-time importer ([`run_import`]) and the live dual-write +/// ([`super::live::write_live_turn`]) so both use the exact same store layout and +/// records land in the same place regardless of who wrote them. +pub fn open_session_stores(workspace: &Path) -> SessionStores { + let root = store_root(workspace); + let journal_root = root.join(JOURNAL_SUBDIR); + SessionStores { + kv: FileStore::new(root.join(KV_SUBDIR)), + journal: JsonlAppendStore::new(&journal_root), + journal_root, + } +} + +/// Best-effort run-ledger links read from `session_db/sessions.db` +/// (read-only; missing DB or table is not an error). +#[derive(Debug, Default)] +struct RunLedgerLinks { + /// thread id → sorted, deduped `agent_runs.id`s touching it. + runs_by_thread: HashMap>, + /// worker thread id → `agent_runs.parent_thread_id` (for lineage checks). + parent_thread_by_worker: HashMap, +} + +/// Run the import. Never mutates sources; per-item failures become warnings. +pub async fn run_import( + workspace: &Path, + opts: &ImportOptions, + project: JournalProjector, +) -> Result { + let SessionStores { + kv, + journal, + journal_root, + } = open_session_stores(workspace); + + tracing::info!( + "[session-import] start workspace={} dry_run={} only={:?} force={}", + workspace.display(), + opts.dry_run, + opts.only, + opts.force + ); + + // Global marker fast path: a completed full import skips the scan + // entirely, unless targeted (--only), forced, or a dry-run plan. + let full_scan = opts.only.is_none(); + if full_scan + && !opts.force + && !opts.dry_run + && let Ok(Some(marker)) = kv.get(NS_MIGRATIONS, MARKER_KEY).await + && marker.get("version").and_then(serde_json::Value::as_u64) + == Some(u64::from(IMPORT_VERSION)) + { + tracing::info!("[session-import] marker present, nothing to do: {marker}"); + return Ok(ImportSummary { + already_done: true, + ..Default::default() + }); + } + + let only_pattern = match opts.only.as_deref() { + Some(raw) => { + Some(glob::Pattern::new(raw).with_context(|| format!("invalid --only glob: {raw:?}"))?) + } + None => None, + }; + + let (sources, scan_warnings) = discover_sources(workspace); + let mut summary = ImportSummary { + dry_run: opts.dry_run, + warnings: scan_warnings, + ..Default::default() + }; + for w in &summary.warnings { + tracing::warn!("[session-import] scan: {w}"); + } + + let sources: Vec = sources + .into_iter() + .filter(|s| only_pattern.as_ref().is_none_or(|p| p.matches(&s.stem))) + .collect(); + summary.scanned = sources.len(); + tracing::info!("[session-import] scanned {} source(s)", summary.scanned); + + // Pass 1: read every transcript so lineage cross-checks can see sibling + // metadata. Read failures are recorded and retried as Markdown when a + // companion exists. + let mut parsed: HashMap = HashMap::new(); + let mut read_failures: HashMap = HashMap::new(); + for item in &sources { + match read_source(item) { + Ok(t) => { + parsed.insert(item.stem.clone(), t); + } + Err(err) => { + read_failures.insert(item.stem.clone(), format!("{err:#}")); + } + } + } + + let links = read_run_ledger_links(workspace, &mut summary.warnings); + let imported_at = chrono::Utc::now().to_rfc3339(); + + // Pass 2: plan + write per item. + for item in &sources { + let report = process_item( + item, + &parsed, + &read_failures, + &links, + &kv, + &journal, + &journal_root, + &imported_at, + opts, + project, + ) + .await; + + match report.action { + ItemAction::Imported => { + summary.imported += 1; + summary.messages_written += report.messages; + } + ItemAction::WouldImport => summary.imported += 1, + ItemAction::SkippedUnchanged => summary.skipped += 1, + ItemAction::Failed => summary.failed += 1, + } + if opts.verbose { + tracing::info!( + "[session-import] {:?} stem={} source={} messages={} warnings={}", + report.action, + report.session_key, + report.source, + report.messages, + report.warnings.len() + ); + } else { + tracing::debug!( + "[session-import] {:?} stem={} source={}", + report.action, + report.session_key, + report.source + ); + } + summary.items.push(report); + } + + // Global marker only after a full, non-dry scan. + if full_scan && !opts.dry_run { + let marker = json!({ + "version": IMPORT_VERSION, + "imported_at": imported_at, + "scanned": summary.scanned, + "imported": summary.imported, + "skipped": summary.skipped, + "failed": summary.failed, + "messages_written": summary.messages_written, + "warnings": summary.warnings.len(), + }); + if let Err(err) = kv.put(NS_MIGRATIONS, MARKER_KEY, marker).await { + summary + .warnings + .push(format!("failed to write global marker: {err}")); + } + } + + tracing::info!( + "[session-import] done scanned={} imported={} skipped={} failed={} messages={} dry_run={}", + summary.scanned, + summary.imported, + summary.skipped, + summary.failed, + summary.messages_written, + summary.dry_run + ); + Ok(summary) +} + +/// Read one source with the appropriate reader. +fn read_source(item: &SourceItem) -> Result { + match item.kind { + SourceKind::Jsonl | SourceKind::JsonlLegacyDir => read_transcript(&item.path) + .with_context(|| format!("read_transcript({})", item.relative)), + SourceKind::Markdown => read_transcript_legacy_md(&item.path) + .with_context(|| format!("read_transcript_legacy_md({})", item.relative)), + } +} + +#[allow(clippy::too_many_arguments)] +async fn process_item( + item: &SourceItem, + parsed: &HashMap, + read_failures: &HashMap, + links: &RunLedgerLinks, + kv: &FileStore, + journal: &JsonlAppendStore, + journal_root: &Path, + imported_at: &str, + opts: &ImportOptions, + project: JournalProjector, +) -> ItemReport { + let mut report = ItemReport { + session_key: item.stem.clone(), + source: item.relative.clone(), + kind: item.kind, + action: ItemAction::Failed, + stream: None, + thread_id: None, + messages: 0, + warnings: Vec::new(), + }; + + if let Some(err) = read_failures.get(&item.stem) { + report.warnings.push(format!("unreadable source: {err}")); + tracing::warn!( + "[session-import] failed stem={} source={}: {err}", + item.stem, + item.relative + ); + return report; + } + let Some(transcript) = parsed.get(&item.stem) else { + report + .warnings + .push("internal: parsed transcript missing".into()); + return report; + }; + + let stream = stream_name(&item.stem); + let (thread_id, synthesized) = + effective_thread_id(&item.stem, transcript.meta.thread_id.as_deref()); + if synthesized { + report + .warnings + .push(format!("no thread_id in _meta; synthesized {thread_id}")); + } + report.stream = Some(stream.clone()); + report.thread_id = Some(thread_id.clone()); + report.messages = transcript.messages.len(); + + // Lineage cross-check: stem chain is the write-time truth; a + // disagreeing run-ledger parent is only a warning. + if let Some(parent_stem) = parent_session_key(&item.stem) + && let (Some(ledger_parent), Some(parent_transcript)) = ( + links.parent_thread_by_worker.get(&thread_id), + parsed.get(&parent_stem), + ) + && let Some(parent_thread) = parent_transcript.meta.thread_id.as_deref() + && ledger_parent != parent_thread + { + report.warnings.push(format!( + "run-ledger parent thread {ledger_parent} disagrees with stem-chain \ + parent thread {parent_thread}; keeping the stem chain" + )); + } + + // Idempotency: skip unchanged sources unless forced. + let item_key = ledger_key(&item.relative); + let (size, mtime_ms) = file_fingerprint(&item.path); + if !opts.force + && let Ok(Some(prior)) = kv.get(NS_MIGRATION_ITEMS, &item_key).await + && let Ok(prior) = serde_json::from_value::(prior) + && prior.version == IMPORT_VERSION + && prior.size == size + && prior.mtime_ms == mtime_ms + { + report.action = ItemAction::SkippedUnchanged; + return report; + } + + if opts.dry_run { + report.action = ItemAction::WouldImport; + return report; + } + + // Re-import overwrites: the append store has no truncate, so drop the + // stream file (layout: `{journal_root}/{stream}.jsonl`) before writing. + let stream_file = journal_root.join(format!("{stream}.jsonl")); + if stream_file.exists() + && let Err(err) = std::fs::remove_file(&stream_file) + { + report + .warnings + .push(format!("cannot reset journal stream {stream}: {err}")); + return report; + } + + for (idx, record) in journal_messages(transcript, project).iter().enumerate() { + let value = match serde_json::to_value(record) { + Ok(v) => v, + Err(err) => { + report + .warnings + .push(format!("message {idx} not serializable: {err}")); + return report; + } + }; + if let Err(err) = journal.append(&stream, value).await { + report + .warnings + .push(format!("journal append failed at message {idx}: {err}")); + return report; + } + } + + let run_ids = links + .runs_by_thread + .get(&thread_id) + .cloned() + .unwrap_or_default(); + let descriptor = build_descriptor( + &item.stem, + transcript, + thread_id, + synthesized, + run_ids, + DescriptorSource { + jsonl: matches!(item.kind, SourceKind::Jsonl | SourceKind::JsonlLegacyDir) + .then(|| item.relative.clone()), + md: match item.kind { + SourceKind::Markdown => Some(item.relative.clone()), + _ => item.md_companion.clone(), + }, + }, + imported_at.to_string(), + report.warnings.len(), + ); + let descriptor_key = super::convert::sanitize_store_name(&item.stem); + let descriptor_value = match serde_json::to_value(&descriptor) { + Ok(v) => v, + Err(err) => { + report + .warnings + .push(format!("descriptor not serializable: {err}")); + return report; + } + }; + if let Err(err) = kv.put(NS_SESSIONS, &descriptor_key, descriptor_value).await { + report + .warnings + .push(format!("descriptor write failed: {err}")); + return report; + } + + let ledger = ItemLedgerRecord { + version: IMPORT_VERSION, + session_key: item.stem.clone(), + source: item.relative.clone(), + size, + mtime_ms, + stream, + messages: report.messages, + imported_at: imported_at.to_string(), + }; + match serde_json::to_value(&ledger) { + Ok(v) => { + if let Err(err) = kv.put(NS_MIGRATION_ITEMS, &item_key, v).await { + report + .warnings + .push(format!("item ledger write failed: {err}")); + return report; + } + } + Err(err) => { + report + .warnings + .push(format!("item ledger not serializable: {err}")); + return report; + } + } + + report.action = ItemAction::Imported; + report +} + +/// Item-ledger key: hex sha256 of the workspace-relative source path. +pub(super) fn ledger_key(relative: &str) -> String { + let mut hasher = Sha256::new(); + hasher.update(relative.as_bytes()); + hasher + .finalize() + .iter() + .map(|byte| format!("{byte:02x}")) + .collect() +} + +/// `(size, mtime_ms)` fingerprint; `(0, 0)` when unreadable. +fn file_fingerprint(path: &Path) -> (u64, u64) { + let Ok(meta) = std::fs::metadata(path) else { + return (0, 0); + }; + let mtime_ms = meta + .modified() + .ok() + .and_then(|t| t.duration_since(std::time::UNIX_EPOCH).ok()) + .map(|d| d.as_millis() as u64) + .unwrap_or(0); + (meta.len(), mtime_ms) +} + +/// Read run-ledger links from `session_db/sessions.db`, read-only. +/// +/// Missing file/table → empty links, no warning (fresh workspaces are +/// normal). Query errors are warnings, never failures. +fn read_run_ledger_links(workspace: &Path, warnings: &mut Vec) -> RunLedgerLinks { + let db_path = workspace.join("session_db").join("sessions.db"); + if !db_path.exists() { + return RunLedgerLinks::default(); + } + let conn = match rusqlite::Connection::open_with_flags( + &db_path, + rusqlite::OpenFlags::SQLITE_OPEN_READ_ONLY, + ) { + Ok(c) => c, + Err(err) => { + warnings.push(format!("run ledger unreadable ({err}); skipping run links")); + return RunLedgerLinks::default(); + } + }; + + let mut links = RunLedgerLinks::default(); + let mut stmt = + match conn.prepare("SELECT id, parent_thread_id, worker_thread_id FROM agent_runs") { + Ok(s) => s, + Err(_) => return RunLedgerLinks::default(), // table absent: nothing to link + }; + let rows = stmt.query_map([], |row| { + Ok(( + row.get::<_, String>(0)?, + row.get::<_, Option>(1)?, + row.get::<_, Option>(2)?, + )) + }); + let rows = match rows { + Ok(r) => r, + Err(err) => { + warnings.push(format!( + "run ledger query failed ({err}); skipping run links" + )); + return RunLedgerLinks::default(); + } + }; + for row in rows.flatten() { + let (id, parent_thread, worker_thread) = row; + for thread in [parent_thread.as_deref(), worker_thread.as_deref()] + .into_iter() + .flatten() + { + let entry = links.runs_by_thread.entry(thread.to_string()).or_default(); + if !entry.contains(&id) { + entry.push(id.clone()); + } + } + if let (Some(worker), Some(parent)) = (worker_thread, parent_thread) { + links.parent_thread_by_worker.insert(worker, parent); + } + } + for ids in links.runs_by_thread.values_mut() { + ids.sort(); + } + links +} diff --git a/crates/tinyagents-session/src/transcript/import/ops_test.rs b/crates/tinyagents-session/src/transcript/import/ops_test.rs new file mode 100644 index 00000000..050d6a1d --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/ops_test.rs @@ -0,0 +1,471 @@ +//! Fixture-matrix tests for the session importer — one test per row of the +//! matrix in `docs/tinyagents-session-migration-design.md`. + +use std::fs; +use std::path::{Path, PathBuf}; + +use tempfile::TempDir; +use tinyagents_harness::store::{AppendStore, FileStore, JsonlAppendStore, Store}; + +use super::convert::{journal_messages, plain_journal_message, sanitize_store_name}; +use super::ops::{run_import as run_import_with, store_root}; +use super::types::{ + ImportOptions, ImportSummary, ItemAction, JournalMessage, MARKER_KEY, NS_MIGRATIONS, + NS_SESSIONS, SessionDescriptor, +}; +use crate::transcript::{read_transcript, read_transcript_legacy_md}; + +async fn run_import(ws: &Path, opts: &ImportOptions) -> anyhow::Result { + run_import_with(ws, opts, plain_journal_message).await +} + +fn ws() -> TempDir { + TempDir::new().expect("tempdir") +} + +fn write_file(path: &Path, contents: &str) { + fs::create_dir_all(path.parent().unwrap()).unwrap(); + fs::write(path, contents).unwrap(); +} + +fn flat_jsonl(ws: &Path, stem: &str) -> PathBuf { + ws.join("session_raw").join(format!("{stem}.jsonl")) +} + +fn meta_line(thread_id: Option<&str>, dispatcher: &str) -> String { + let thread = thread_id + .map(|t| format!(",\"thread_id\":\"{t}\"")) + .unwrap_or_default(); + format!( + "{{\"_meta\":{{\"agent\":\"orchestrator\",\"agent_id\":\"orchestrator\",\ + \"dispatcher\":\"{dispatcher}\",\"provider\":\"anthropic\",\"model\":\"claude\",\ + \"created\":\"2024-01-01T00:00:00Z\",\"updated\":\"2024-01-01T00:05:00Z\",\ + \"turn_count\":1,\"input_tokens\":100,\"output_tokens\":50,\ + \"cached_input_tokens\":20,\"charged_amount_usd\":0.05{thread}}}}}\n" + ) +} + +/// A native transcript: user turn + assistant turn carrying usage and a +/// tool call with Gemini-style `extra_content` passthrough. +fn native_body() -> &'static str { + concat!( + "{\"role\":\"user\",\"content\":\"hi\"}\n", + "{\"role\":\"assistant\",\"content\":\"done\",\"provider\":\"anthropic\",", + "\"model\":\"claude\",\"usage\":{\"input\":100,\"output\":50,\"cached_input\":20,", + "\"context_window\":200000,\"cost_usd\":0.05},\"ts\":\"2024-01-01T00:00:01Z\",", + "\"iteration\":1,\"tool_calls\":[{\"id\":\"tc1\",\"name\":\"read_file\",", + "\"arguments\":\"{\\\"path\\\":\\\"x\\\"}\",", + "\"extra_content\":{\"google\":{\"thought_signature\":\"sig\"}}}]}\n", + ) +} + +async fn run(ws: &Path, opts: ImportOptions) -> ImportSummary { + run_import(ws, &opts).await.expect("run_import") +} + +async fn descriptor(ws: &Path, stem: &str) -> SessionDescriptor { + let kv = FileStore::new(store_root(ws).join("kv")); + let value = kv + .get(NS_SESSIONS, &sanitize_store_name(stem)) + .await + .expect("kv get") + .unwrap_or_else(|| panic!("descriptor missing for {stem}")); + serde_json::from_value(value).expect("descriptor shape") +} + +async fn journal_readback(ws: &Path, stream: &str) -> Vec { + let journal = JsonlAppendStore::new(store_root(ws).join("journal")); + journal + .read_from(stream, 0) + .await + .expect("journal read") + .into_iter() + .map(|(_, v)| serde_json::from_value(v).expect("journal record shape")) + .collect() +} + +/// Parity: the journal read-back must equal what `read_transcript` returns +/// for the source, field for field (including reattached turn-usage +/// metadata). +async fn assert_parity_jsonl(ws: &Path, stem: &str, source: &Path) { + let expected = journal_messages( + &read_transcript(source).expect("read_transcript"), + plain_journal_message, + ); + let actual = journal_readback(ws, &format!("session.{stem}.messages")).await; + assert_eq!(actual, expected, "journal read-back diverges for {stem}"); +} + +// Fixture 1 + 7: current flat layout, native dispatcher, tool_calls with +// extra_content — full parity + descriptor mapping. +#[tokio::test] +async fn imports_flat_native_jsonl_with_parity() { + let ws = ws(); + let stem = "1719000000_orchestrator"; + let source = flat_jsonl(ws.path(), stem); + write_file( + &source, + &(meta_line(Some("t-root"), "native") + native_body()), + ); + + let summary = run(ws.path(), ImportOptions::default()).await; + assert_eq!(summary.scanned, 1); + assert_eq!(summary.imported, 1); + assert_eq!(summary.failed, 0); + assert_eq!(summary.messages_written, 2); + + let desc = descriptor(ws.path(), stem).await; + assert_eq!(desc.session_key, stem); + assert_eq!(desc.thread_id, "t-root"); + assert!(!desc.thread_id_synthesized); + assert_eq!(desc.parent_session_key, None); + assert_eq!(desc.dispatcher, "native"); + assert_eq!(desc.usage.input, 100); + assert_eq!(desc.usage.cost_usd, 0.05); + assert_eq!( + desc.source.jsonl.as_deref(), + Some("session_raw/1719000000_orchestrator.jsonl") + ); + + assert_parity_jsonl(ws.path(), stem, &source).await; + + // Global marker written after a full non-dry run. + let kv = FileStore::new(store_root(ws.path()).join("kv")); + assert!(kv.get(NS_MIGRATIONS, MARKER_KEY).await.unwrap().is_some()); +} + +// Fixture 2: legacy date-folder layout. +#[tokio::test] +async fn imports_legacy_date_folder_jsonl() { + let ws = ws(); + let stem = "1718000000_researcher"; + let source = ws + .path() + .join("session_raw") + .join("01062024") + .join(format!("{stem}.jsonl")); + write_file( + &source, + &(meta_line(Some("t-legacy"), "native") + native_body()), + ); + + let summary = run(ws.path(), ImportOptions::default()).await; + assert_eq!(summary.imported, 1, "warnings: {:?}", summary.warnings); + let desc = descriptor(ws.path(), stem).await; + assert_eq!( + desc.source.jsonl.as_deref(), + Some("session_raw/01062024/1718000000_researcher.jsonl") + ); + assert_parity_jsonl(ws.path(), stem, &source).await; +} + +// Fixture 3: Markdown-only session via the legacy `` reader. +#[tokio::test] +async fn imports_markdown_only_session() { + let ws = ws(); + let stem = "1717000000_helper"; + let md = ws + .path() + .join("sessions") + .join("2024_06_01") + .join(format!("{stem}.md")); + write_file( + &md, + "\n\ + \nhello\n\n\ + \nhi there\n\n", + ); + + let summary = run(ws.path(), ImportOptions::default()).await; + assert_eq!(summary.imported, 1, "warnings: {:?}", summary.warnings); + + let desc = descriptor(ws.path(), stem).await; + assert_eq!(desc.thread_id, "t-md"); + assert_eq!(desc.source.jsonl, None); + assert!(desc.source.md.as_deref().unwrap().ends_with(".md")); + + let expected = journal_messages( + &read_transcript_legacy_md(&md).unwrap(), + plain_journal_message, + ); + let actual = journal_readback(ws.path(), &format!("session.{stem}.messages")).await; + assert_eq!(actual, expected); +} + +// Fixture 4: sub-agent stems, two-level chain → parent keys from the stem. +#[tokio::test] +async fn subagent_stem_chain_sets_parent_lineage() { + let ws = ws(); + let root = "100_a"; + let child = "100_a__200_b"; + let grandchild = "100_a__200_b__300_c"; + for (stem, thread) in [(root, "t-a"), (child, "t-b"), (grandchild, "t-c")] { + write_file( + &flat_jsonl(ws.path(), stem), + &(meta_line(Some(thread), "native") + native_body()), + ); + } + + let summary = run(ws.path(), ImportOptions::default()).await; + assert_eq!(summary.imported, 3); + assert_eq!(descriptor(ws.path(), root).await.parent_session_key, None); + assert_eq!( + descriptor(ws.path(), child) + .await + .parent_session_key + .as_deref(), + Some("100_a") + ); + assert_eq!( + descriptor(ws.path(), grandchild) + .await + .parent_session_key + .as_deref(), + Some("100_a__200_b") + ); +} + +// Fixtures 5 + 6: XML and P-format tool markup stays verbatim in content. +#[tokio::test] +async fn xml_and_pformat_markup_preserved_verbatim() { + let ws = ws(); + let xml_stem = "1716000000_xmlagent"; + let xml_markup = + "calling now {\"name\":\"shell\",\"arguments\":{\"cmd\":\"ls\"}}"; + write_file( + &flat_jsonl(ws.path(), xml_stem), + &format!( + "{}{{\"role\":\"assistant\",\"content\":\"{}\"}}\n", + meta_line(Some("t-xml"), "xml"), + xml_markup.replace('"', "\\\"") + ), + ); + let p_stem = "1716000001_pfagent"; + let p_markup = "read_file[notes.txt|10]"; + write_file( + &flat_jsonl(ws.path(), p_stem), + &format!( + "{}{{\"role\":\"assistant\",\"content\":\"{}\"}}\n", + meta_line(Some("t-pf"), "pformat"), + p_markup.replace('"', "\\\"") + ), + ); + + let summary = run(ws.path(), ImportOptions::default()).await; + assert_eq!(summary.imported, 2); + + let xml_msgs = journal_readback(ws.path(), &format!("session.{xml_stem}.messages")).await; + assert_eq!(xml_msgs[0].content, xml_markup); + assert_eq!(descriptor(ws.path(), xml_stem).await.dispatcher, "xml"); + + let p_msgs = journal_readback(ws.path(), &format!("session.{p_stem}.messages")).await; + assert_eq!(p_msgs[0].content, p_markup); + assert_eq!(descriptor(ws.path(), p_stem).await.dispatcher, "pformat"); +} + +// Fixture 8: malformed sources fail per-item without aborting the batch; +// a truncated trailing line is tolerated by the reader (skip + warning). +#[tokio::test] +async fn malformed_sources_fail_without_aborting_batch() { + let ws = ws(); + write_file( + &flat_jsonl(ws.path(), "1715000000_good"), + &(meta_line(Some("t-good"), "native") + native_body()), + ); + // Missing `_meta` first line. + write_file( + &flat_jsonl(ws.path(), "1715000001_nometa"), + "{\"role\":\"user\",\"content\":\"hi\"}\n", + ); + // Empty file. + write_file(&flat_jsonl(ws.path(), "1715000002_empty"), ""); + // Truncated last line after one valid message. + write_file( + &flat_jsonl(ws.path(), "1715000003_truncated"), + &format!( + "{}{{\"role\":\"user\",\"content\":\"ok\"}}\n{{\"role\":\"assistant\",\"conte", + meta_line(Some("t-trunc"), "native") + ), + ); + + let summary = run(ws.path(), ImportOptions::default()).await; + assert_eq!(summary.scanned, 4); + assert_eq!(summary.imported, 2, "good + truncated-but-tolerated"); + assert_eq!(summary.failed, 2, "no-meta + empty fail per-item"); + + let truncated = journal_readback(ws.path(), "session.1715000003_truncated.messages").await; + assert_eq!(truncated.len(), 1, "malformed trailing line skipped"); + + // The batch still completed and the marker landed. + let kv = FileStore::new(store_root(ws.path()).join("kv")); + assert!(kv.get(NS_MIGRATIONS, MARKER_KEY).await.unwrap().is_some()); +} + +// Fixture 9: `_meta` without thread_id → synthesized stable id + warning. +#[tokio::test] +async fn synthesizes_thread_id_when_meta_lacks_one() { + let ws = ws(); + let stem = "1714000000_nothread"; + write_file( + &flat_jsonl(ws.path(), stem), + &(meta_line(None, "native") + native_body()), + ); + + let summary = run(ws.path(), ImportOptions::default()).await; + let desc = descriptor(ws.path(), stem).await; + assert_eq!(desc.thread_id, format!("imported-{stem}")); + assert!(desc.thread_id_synthesized); + let item = &summary.items[0]; + assert!( + item.warnings.iter().any(|w| w.contains("synthesized")), + "warnings: {:?}", + item.warnings + ); +} + +// Fixture 10: run-ledger rows join run_ids; a disagreeing ledger parent +// thread is a warning, and the stem chain wins. +#[tokio::test] +async fn run_ledger_links_join_and_disagreement_warns() { + let ws = ws(); + let root = "100_root"; + let child = "100_root__200_child"; + write_file( + &flat_jsonl(ws.path(), root), + &(meta_line(Some("t-root"), "native") + native_body()), + ); + write_file( + &flat_jsonl(ws.path(), child), + &(meta_line(Some("t-child"), "native") + native_body()), + ); + + let db_dir = ws.path().join("session_db"); + fs::create_dir_all(&db_dir).unwrap(); + let conn = rusqlite::Connection::open(db_dir.join("sessions.db")).unwrap(); + conn.execute_batch( + "CREATE TABLE agent_runs (id TEXT PRIMARY KEY, parent_thread_id TEXT, worker_thread_id TEXT); + INSERT INTO agent_runs VALUES ('run-1', 't-DIFFERENT', 't-child');", + ) + .unwrap(); + drop(conn); + + let summary = run(ws.path(), ImportOptions::default()).await; + assert_eq!(summary.imported, 2, "warnings: {:?}", summary.warnings); + + let child_desc = descriptor(ws.path(), child).await; + assert_eq!(child_desc.run_ids, vec!["run-1".to_string()]); + assert_eq!( + child_desc.parent_session_key.as_deref(), + Some("100_root"), + "stem chain wins" + ); + let child_item = summary + .items + .iter() + .find(|i| i.session_key == child) + .unwrap(); + assert!( + child_item.warnings.iter().any(|w| w.contains("disagrees")), + "warnings: {:?}", + child_item.warnings + ); +} + +// Fixture 11: full rerun short-circuits on the marker; a targeted rerun +// skips unchanged items and re-imports only the touched one. +#[tokio::test] +async fn idempotent_rerun_skips_then_touch_reimports() { + let ws = ws(); + let a = "1713000000_a"; + let b = "1713000001_b"; + for stem in [a, b] { + write_file( + &flat_jsonl(ws.path(), stem), + &(meta_line(Some(&format!("t-{stem}")), "native") + native_body()), + ); + } + + let first = run(ws.path(), ImportOptions::default()).await; + assert_eq!(first.imported, 2); + + // Full rerun: global marker short-circuits. + let second = run(ws.path(), ImportOptions::default()).await; + assert!(second.already_done); + assert_eq!(second.scanned, 0); + + // Targeted rerun: everything unchanged → skipped. + let targeted = run( + ws.path(), + ImportOptions { + only: Some("*".into()), + ..Default::default() + }, + ) + .await; + assert_eq!(targeted.skipped, 2); + assert_eq!(targeted.imported, 0); + + // Touch one source (size changes) → only it re-imports. + let path = flat_jsonl(ws.path(), a); + let mut contents = fs::read_to_string(&path).unwrap(); + contents.push_str("{\"role\":\"user\",\"content\":\"follow-up\"}\n"); + fs::write(&path, contents).unwrap(); + + let after_touch = run( + ws.path(), + ImportOptions { + only: Some("*".into()), + ..Default::default() + }, + ) + .await; + assert_eq!(after_touch.imported, 1); + assert_eq!(after_touch.skipped, 1); + + // The re-imported stream was reset, not appended twice. + let msgs = journal_readback(ws.path(), &format!("session.{a}.messages")).await; + assert_eq!(msgs.len(), 3); +} + +// Dry-run: reports the plan, writes nothing. +#[tokio::test] +async fn dry_run_writes_nothing() { + let ws = ws(); + write_file( + &flat_jsonl(ws.path(), "1712000000_plan"), + &(meta_line(Some("t-plan"), "native") + native_body()), + ); + + let summary = run( + ws.path(), + ImportOptions { + dry_run: true, + ..Default::default() + }, + ) + .await; + assert!(summary.dry_run); + assert_eq!(summary.imported, 1); + assert_eq!(summary.messages_written, 0); + assert_eq!(summary.items[0].action, ItemAction::WouldImport); + assert!( + !store_root(ws.path()).join("kv").exists() + && !store_root(ws.path()).join("journal").exists(), + "dry run must not create store dirs" + ); +} + +// Sources are never mutated: byte-identical after import. +#[tokio::test] +async fn sources_untouched_after_import() { + let ws = ws(); + let source = flat_jsonl(ws.path(), "1711000000_untouched"); + let contents = meta_line(Some("t-u"), "native") + native_body(); + write_file(&source, &contents); + + run(ws.path(), ImportOptions::default()).await; + assert_eq!(fs::read_to_string(&source).unwrap(), contents); +} diff --git a/crates/tinyagents-session/src/transcript/import/scan.rs b/crates/tinyagents-session/src/transcript/import/scan.rs new file mode 100644 index 00000000..67a14553 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/scan.rs @@ -0,0 +1,178 @@ +//! Source discovery: which legacy session files exist in a workspace. + +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; + +use super::types::SourceKind; + +/// One discovered source, keyed by session stem. +#[derive(Debug, Clone)] +pub struct SourceItem { + pub stem: String, + pub kind: SourceKind, + /// Absolute path of the file the messages will be read from. + pub path: PathBuf, + /// Workspace-relative path of `path`, for descriptors and the ledger. + pub relative: String, + /// Workspace-relative Markdown companion, when one exists next to a + /// JSONL source (informational only — never read when JSONL exists). + pub md_companion: Option, +} + +/// Scan `session_raw/` (flat + legacy `DDMMYYYY` folders) and `sessions/` +/// Markdown directories. Returns items sorted by stem, plus scan warnings. +/// +/// Precedence per stem: flat JSONL > legacy-dir JSONL > Markdown-only. The +/// same stem never yields two items. +pub fn discover_sources(workspace: &Path) -> (Vec, Vec) { + let mut warnings = Vec::new(); + // stem → item, first writer wins per the precedence order below. + let mut by_stem: BTreeMap = BTreeMap::new(); + + let raw_dir = workspace.join("session_raw"); + + // 1. Flat JSONL (current layout). + for path in list_files(&raw_dir, "jsonl", &mut warnings) { + insert_stem( + &mut by_stem, + workspace, + path, + SourceKind::Jsonl, + &mut warnings, + ); + } + + // 2. Legacy date-folder JSONL. + for sub in list_dirs(&raw_dir, &mut warnings) { + let name = sub + .file_name() + .unwrap_or_default() + .to_string_lossy() + .to_string(); + if !is_ddmmyyyy(&name) { + continue; + } + for path in list_files(&sub, "jsonl", &mut warnings) { + insert_stem( + &mut by_stem, + workspace, + path, + SourceKind::JsonlLegacyDir, + &mut warnings, + ); + } + } + + // 3. Markdown sessions: only stems with no JSONL anywhere. Also record + // companions for stems that do have JSONL. + let sessions_dir = workspace.join("sessions"); + for sub in list_dirs(&sessions_dir, &mut warnings) { + for path in list_files(&sub, "md", &mut warnings) { + let Some(stem) = file_stem(&path) else { + continue; + }; + let relative = relative_to(workspace, &path); + match by_stem.get_mut(&stem) { + Some(item) => { + if item.md_companion.is_none() { + item.md_companion = Some(relative); + } + } + None => { + by_stem.insert( + stem.clone(), + SourceItem { + stem, + kind: SourceKind::Markdown, + relative, + path, + md_companion: None, + }, + ); + } + } + } + } + + (by_stem.into_values().collect(), warnings) +} + +fn insert_stem( + by_stem: &mut BTreeMap, + workspace: &Path, + path: PathBuf, + kind: SourceKind, + warnings: &mut Vec, +) { + let Some(stem) = file_stem(&path) else { + warnings.push(format!("unreadable file name, skipped: {}", path.display())); + return; + }; + let relative = relative_to(workspace, &path); + if let Some(existing) = by_stem.get(&stem) { + warnings.push(format!( + "duplicate stem '{stem}': keeping {}, ignoring {relative}", + existing.relative + )); + return; + } + by_stem.insert( + stem.clone(), + SourceItem { + stem, + kind, + relative, + path, + md_companion: None, + }, + ); +} + +fn file_stem(path: &Path) -> Option { + path.file_stem().map(|s| s.to_string_lossy().to_string()) +} + +fn relative_to(workspace: &Path, path: &Path) -> String { + path.strip_prefix(workspace) + .unwrap_or(path) + .to_string_lossy() + .to_string() +} + +fn list_files(dir: &Path, ext: &str, warnings: &mut Vec) -> Vec { + list_entries(dir, warnings) + .into_iter() + .filter(|p| p.is_file() && p.extension().is_some_and(|e| e == ext)) + .collect() +} + +fn list_dirs(dir: &Path, warnings: &mut Vec) -> Vec { + list_entries(dir, warnings) + .into_iter() + .filter(|p| p.is_dir()) + .collect() +} + +fn list_entries(dir: &Path, warnings: &mut Vec) -> Vec { + if !dir.exists() { + return Vec::new(); + } + match std::fs::read_dir(dir) { + Ok(entries) => { + let mut paths: Vec = + entries.filter_map(|e| e.ok()).map(|e| e.path()).collect(); + paths.sort(); + paths + } + Err(err) => { + warnings.push(format!("cannot read {}: {err}", dir.display())); + Vec::new() + } + } +} + +/// Exactly 8 ASCII digits — the legacy `DDMMYYYY` folder convention (matches +/// `session::migration::is_ddmmyyyy`). +fn is_ddmmyyyy(name: &str) -> bool { + name.len() == 8 && name.bytes().all(|b| b.is_ascii_digit()) +} diff --git a/crates/tinyagents-session/src/transcript/import/types.rs b/crates/tinyagents-session/src/transcript/import/types.rs new file mode 100644 index 00000000..744d37b3 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/import/types.rs @@ -0,0 +1,193 @@ +//! Serde types for the one-time session import into TinyAgents stores. + +use serde::{Deserialize, Serialize}; +use serde_json::Value; + +/// Schema version of the importer. Bump when the record shapes change and a +/// re-import should be forced. +pub const IMPORT_VERSION: u32 = 1; + +/// Store namespaces / stream layout under `{workspace}/tinyagents_store/`. +/// +/// TinyAgents store names are slash-free (ASCII alphanumerics plus `-_.` +/// only — the crate's path-traversal guard), so streams use dot-separated +/// names rather than the `thread/{id}/messages` shape sketched in the design +/// doc. +pub const KV_SUBDIR: &str = "kv"; +pub const JOURNAL_SUBDIR: &str = "journal"; +pub const NS_SESSIONS: &str = "sessions"; +pub const NS_MIGRATIONS: &str = "migrations"; +pub const NS_MIGRATION_ITEMS: &str = "migration_items"; +pub const MARKER_KEY: &str = "session_import_v1"; + +/// Options accepted by `session_import.run`. +#[derive(Debug, Clone, Default, Deserialize)] +pub struct ImportOptions { + /// Plan only: read and report, write nothing. + #[serde(default)] + pub dry_run: bool, + /// Optional glob over session stems (e.g. `1719*_orchestrator`); when + /// set, only matching sources are considered and the global marker is + /// neither honoured as a skip nor written. + #[serde(default)] + pub only: Option, + /// Re-import even when the global marker or an unchanged item ledger + /// entry would normally skip the work. + #[serde(default)] + pub force: bool, + /// Per-item info logging instead of debug. + #[serde(default)] + pub verbose: bool, +} + +/// What kind of source a scanned item came from. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum SourceKind { + /// Current flat `session_raw/{stem}.jsonl`. + Jsonl, + /// Legacy date-folder `session_raw/{DDMMYYYY}/{stem}.jsonl`. + JsonlLegacyDir, + /// Markdown-only session (no JSONL twin anywhere). + Markdown, +} + +/// Per-item action taken (or planned, in dry-run). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ItemAction { + Imported, + WouldImport, + SkippedUnchanged, + Failed, +} + +/// Per-source report line in the summary. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ItemReport { + pub session_key: String, + /// Source path relative to the workspace root. + pub source: String, + pub kind: SourceKind, + pub action: ItemAction, + /// Journal stream the messages went to (absent for failures). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub stream: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub thread_id: Option, + pub messages: usize, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub warnings: Vec, +} + +/// Whole-run summary returned by `session_import.run`. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct ImportSummary { + pub dry_run: bool, + /// True when the global marker short-circuited the scan. + pub already_done: bool, + pub scanned: usize, + pub imported: usize, + pub skipped: usize, + pub failed: usize, + pub messages_written: usize, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub warnings: Vec, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub items: Vec, +} + +/// Usage roll-up carried on the session descriptor. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct DescriptorUsage { + pub input: u64, + pub output: u64, + pub cached_input: u64, + pub cost_usd: f64, +} + +/// Source pointers preserved on the descriptor (workspace-relative). +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct DescriptorSource { + #[serde(default, skip_serializing_if = "Option::is_none")] + pub jsonl: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub md: Option, +} + +/// Import provenance block on the descriptor. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct DescriptorImport { + pub version: u32, + pub imported_at: String, + pub warnings: usize, +} + +/// The `sessions/{session_key}` compatibility descriptor: maps the OpenHuman +/// session key to TinyAgents-side identifiers and back. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct SessionDescriptor { + pub session_key: String, + /// Parent session key derived from the `__` stem chain (`None` = root). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub parent_session_key: Option, + /// From `_meta.thread_id`, or synthesized `imported-{session_key}`. + pub thread_id: String, + /// True when `thread_id` was synthesized because the source had none. + #[serde(default)] + pub thread_id_synthesized: bool, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub task_id: Option, + /// Run-ledger `agent_runs.id`s joined via `thread_id` (best-effort). + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub run_ids: Vec, + /// Journal stream holding this session's messages. + pub stream: String, + pub dispatcher: String, + pub agent_name: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub agent_id: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub agent_type: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub provider: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub model: Option, + pub created: String, + pub updated: String, + pub turn_count: usize, + pub usage: DescriptorUsage, + pub source: DescriptorSource, + pub import: DescriptorImport, +} + +/// One message journal entry. +/// +/// A host's provider message type may mark `id`/`extra_metadata` +/// `skip_serializing` (they must not reach providers), so the journal carries +/// its own full-fidelity record of what `read_transcript()` returns — including +/// any host metadata (for OpenHuman, the reconstructed `openhuman_turn_usage`) +/// the host's [`JournalProjector`](super::convert::JournalProjector) folds in. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct JournalMessage { + #[serde(default, skip_serializing_if = "Option::is_none")] + pub id: Option, + pub role: String, + pub content: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub extra_metadata: Option, +} + +/// Item-ledger record under `migration_items/{sha256(relative source path)}`. +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ItemLedgerRecord { + pub version: u32, + pub session_key: String, + /// Workspace-relative source path. + pub source: String, + pub size: u64, + pub mtime_ms: u64, + pub stream: String, + pub messages: usize, + pub imported_at: String, +} diff --git a/crates/tinyagents-session/src/transcript/view/cache.rs b/crates/tinyagents-session/src/transcript/view/cache.rs new file mode 100644 index 00000000..7eeef690 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/cache.rs @@ -0,0 +1,157 @@ +//! In-memory, mtime-keyed projection cache for the transcript view. +//! +//! The settled transcript is derived from `session_raw/*.jsonl` on every +//! request. Re-projecting a long thread on each page fetch is wasteful, so we +//! memoize per-thread projections keyed on the backing files' `(path, mtime, +//! len)` signature. An append (new turn, interrupted partial, sub-agent file) +//! changes the signature and transparently invalidates the entry — there are +//! **no disk writes** and no explicit invalidation call. The cache is bounded +//! (LRU over a few dozen threads) so a long-lived core can't grow it without +//! limit. + +use std::collections::HashMap; +use std::collections::VecDeque; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, Mutex, OnceLock}; +use std::time::SystemTime; + +use super::project; +use super::types::ProjectedTranscript; + +const LOG_PREFIX: &str = "[threads][transcript][cache]"; + +/// Number of distinct threads whose projections are retained. Beyond this the +/// least-recently-used entry is evicted. +const CACHE_CAPACITY: usize = 32; + +/// Signature of one backing file — changes on any append/rewrite. +#[derive(Debug, Clone, PartialEq, Eq)] +struct FileSig { + path: PathBuf, + mtime: Option, + len: u64, +} + +fn file_sig(path: &Path) -> FileSig { + let (mtime, len) = match std::fs::metadata(path) { + Ok(meta) => (meta.modified().ok(), meta.len()), + Err(_) => (None, 0), + }; + FileSig { + path: path.to_path_buf(), + mtime, + len, + } +} + +struct CacheEntry { + signature: Vec, + projected: Arc, +} + +#[derive(Default)] +struct CacheInner { + entries: HashMap, + /// LRU order — front is least-recently-used. + order: VecDeque, +} + +/// Bounded per-thread projection cache. +#[derive(Default)] +pub struct TranscriptViewCache { + inner: Mutex, +} + +impl TranscriptViewCache { + /// Project `thread_id`'s transcript, serving a cached result when the + /// backing files are unchanged. `None` when the thread has no transcript. + pub fn get_or_project( + &self, + workspace_dir: &Path, + thread_id: &str, + ) -> Option> { + let (root_paths, sub_paths) = project::resolve_files(workspace_dir, thread_id)?; + let signature: Vec = root_paths + .iter() + .map(|path| file_sig(path)) + .chain(sub_paths.iter().map(|p| file_sig(p))) + .collect(); + + { + let mut inner = self.inner.lock().ok()?; + if let Some(entry) = inner.entries.get(thread_id) { + if entry.signature == signature { + let projected = entry.projected.clone(); + touch(&mut inner, thread_id); + tracing::debug!( + "{LOG_PREFIX} hit thread={thread_id} items={}", + projected.items.len() + ); + return Some(projected); + } + tracing::debug!("{LOG_PREFIX} miss (signature changed) thread={thread_id}"); + } else { + tracing::debug!("{LOG_PREFIX} miss (cold) thread={thread_id}"); + } + } + + let projected = Arc::new(project::project_from_files( + thread_id, + &root_paths, + &sub_paths, + Some(workspace_dir), + )); + + let mut inner = self.inner.lock().ok()?; + inner.entries.insert( + thread_id.to_string(), + CacheEntry { + signature, + projected: projected.clone(), + }, + ); + touch(&mut inner, thread_id); + evict_if_needed(&mut inner); + tracing::debug!( + "{LOG_PREFIX} stored thread={thread_id} items={} cache_size={}", + projected.items.len(), + inner.entries.len() + ); + Some(projected) + } + + #[cfg(test)] + fn len(&self) -> usize { + self.inner.lock().unwrap().entries.len() + } +} + +/// Move `thread_id` to the most-recently-used end of the LRU order. +fn touch(inner: &mut CacheInner, thread_id: &str) { + if let Some(pos) = inner.order.iter().position(|t| t == thread_id) { + inner.order.remove(pos); + } + inner.order.push_back(thread_id.to_string()); +} + +/// Evict least-recently-used entries until within [`CACHE_CAPACITY`]. +fn evict_if_needed(inner: &mut CacheInner) { + while inner.entries.len() > CACHE_CAPACITY { + let Some(victim) = inner.order.pop_front() else { + break; + }; + inner.entries.remove(&victim); + tracing::debug!("{LOG_PREFIX} evicted thread={victim}"); + } +} + +/// Process-wide cache singleton. The transcript view is read-only and derived, +/// so one shared cache across RPC calls is correct. +pub fn global() -> &'static TranscriptViewCache { + static CACHE: OnceLock = OnceLock::new(); + CACHE.get_or_init(TranscriptViewCache::default) +} + +#[cfg(test)] +#[path = "cache_tests.rs"] +mod tests; diff --git a/crates/tinyagents-session/src/transcript/view/cache_tests.rs b/crates/tinyagents-session/src/transcript/view/cache_tests.rs new file mode 100644 index 00000000..014483c8 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/cache_tests.rs @@ -0,0 +1,74 @@ +//! Cache hit/miss + invalidation tests for the transcript view cache. + +use super::TranscriptViewCache; +use crate::transcript; +use std::path::Path; +use tempfile::TempDir; + +fn meta_line(thread_id: &str) -> String { + format!( + r#"{{"_meta":{{"version":1,"agent":"orchestrator","dispatcher":"native","created":"2026-07-21T00:00:00Z","updated":"2026-07-21T00:00:00Z","turn_count":1,"input_tokens":0,"output_tokens":0,"cached_input_tokens":0,"charged_amount_usd":0.0,"thread_id":"{thread_id}"}}}}"# + ) +} + +fn write_raw(workspace: &Path, stem: &str, thread_id: &str, body: &[&str]) -> std::path::PathBuf { + let path = transcript::resolve_keyed_transcript_path(workspace, stem).unwrap(); + let mut buf = meta_line(thread_id); + buf.push('\n'); + for line in body { + buf.push_str(line); + buf.push('\n'); + } + std::fs::write(&path, buf).unwrap(); + path +} + +#[test] +fn recomputes_when_file_grows() { + let dir = TempDir::new().unwrap(); + let path = write_raw( + dir.path(), + "100_orchestrator", + "thr_cache", + &[r#"{"role":"user","content":"first"}"#], + ); + let cache = TranscriptViewCache::default(); + + let a = cache + .get_or_project(dir.path(), "thr_cache") + .expect("first"); + assert_eq!(a.items.len(), 1); + + // Second call, unchanged file → same cached Arc (hit). + let b = cache + .get_or_project(dir.path(), "thr_cache") + .expect("second"); + assert!( + std::sync::Arc::ptr_eq(&a, &b), + "unchanged file must serve cached Arc" + ); + + // Append a line → file length changes → signature invalidates → recompute. + let mut appended = std::fs::read_to_string(&path).unwrap(); + appended.push_str(r#"{"role":"assistant","content":"second"}"#); + appended.push('\n'); + std::fs::write(&path, appended).unwrap(); + + let c = cache + .get_or_project(dir.path(), "thr_cache") + .expect("third"); + assert!(!std::sync::Arc::ptr_eq(&a, &c), "grown file must recompute"); + assert_eq!( + c.items.len(), + 2, + "recomputed projection reflects the append" + ); +} + +#[test] +fn missing_thread_returns_none() { + let dir = TempDir::new().unwrap(); + let cache = TranscriptViewCache::default(); + assert!(cache.get_or_project(dir.path(), "ghost").is_none()); + assert_eq!(cache.len(), 0); +} diff --git a/crates/tinyagents-session/src/transcript/view/mod.rs b/crates/tinyagents-session/src/transcript/view/mod.rs new file mode 100644 index 00000000..75063a8b --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/mod.rs @@ -0,0 +1,145 @@ +//! Transcript-derived view: project the append-only `session_raw/*.jsonl` +//! source of truth into typed display items for the chat renderer, with +//! newest-first pagination over a bounded in-memory cache. +//! +//! Entry point: [`get_page`] (used by the `threads.transcript_get` RPC). + +mod cache; +mod project; +mod prompt_tools; +mod resolve; +mod subagents; +pub mod types; + +use std::path::Path; + +use serde::Serialize; + +pub use project::{project_records, project_thread, project_thread_scoped, resolve_files_scoped}; +pub use types::{DisplayItem, ProjectedTranscript, SubagentStatus, ToolCallStatus}; + +/// Key under which the writer stamps per-result tool failures into a +/// transcript message's extra metadata (`{call_id: {detail}}`); the +/// projection reads it back to mark a tool row as failed. The writer side +/// lives in the host's transcript codec and imports this constant, so the two +/// halves cannot drift. +pub const TOOL_RESULT_FAILURES_METADATA_KEY: &str = "openhuman_tool_failures"; + +const LOG_PREFIX: &str = "[threads][transcript]"; + +/// Default page size — roughly one screen of chat items. +pub const DEFAULT_LIMIT: usize = 50; +/// Hard upper bound on a single page so a client can't request an unbounded +/// projection slice. +pub const MAX_LIMIT: usize = 500; + +/// One newest-first page of a thread's projected transcript. +#[derive(Debug, Clone, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct TranscriptPage { + pub thread_id: String, + /// Display items for this page, **newest-first**. + pub items: Vec, + /// Total top-level items available for the thread. + pub total: usize, + /// Opaque cursor to pass back for the next (older) page; `null` at the end. + #[serde(skip_serializing_if = "Option::is_none")] + pub next_cursor: Option, + /// `true` when more (older) items remain beyond this page. + pub has_more: bool, + /// `false` when the thread has no persisted transcript yet (empty page). + pub has_transcript: bool, +} + +/// Project `thread_id`'s transcript and return one newest-first page. +/// +/// `cursor` is the opaque token returned as `next_cursor` by a previous call +/// (an offset from the newest item); `None`/empty starts at the newest item. +/// `limit` defaults to [`DEFAULT_LIMIT`] and is clamped to [`MAX_LIMIT`]. +pub fn get_page( + workspace_dir: &Path, + thread_id: &str, + cursor: Option<&str>, + limit: Option, +) -> TranscriptPage { + get_page_scoped(workspace_dir, thread_id, None, cursor, limit) +} + +/// Fetch a page scoped to an owning agent. Use this when thread IDs can be +/// supplied by callers and shared by multiple agents. +pub fn get_page_scoped( + workspace_dir: &Path, + thread_id: &str, + agent_id: Option<&str>, + cursor: Option<&str>, + limit: Option, +) -> TranscriptPage { + let limit = limit.unwrap_or(DEFAULT_LIMIT).clamp(1, MAX_LIMIT); + let offset = parse_cursor(cursor); + + let projected = if let Some(agent_id) = agent_id { + project::project_thread_scoped(workspace_dir, thread_id, Some(agent_id)) + .map(std::sync::Arc::new) + } else { + cache::global().get_or_project(workspace_dir, thread_id) + }; + let Some(projected) = projected else { + tracing::debug!("{LOG_PREFIX} get_page thread={thread_id}: no transcript"); + return TranscriptPage { + thread_id: thread_id.to_string(), + items: Vec::new(), + total: 0, + next_cursor: None, + has_more: false, + has_transcript: false, + }; + }; + + let total = projected.items.len(); + let start = offset.min(total); + let end = offset.saturating_add(limit).min(total); + // Newest-first: item `offset` is the newest, walking backwards from the end. + let items: Vec = (start..end) + .map(|i| projected.items[total - 1 - i].clone()) + .collect(); + let has_more = end < total; + let next_cursor = has_more.then(|| end.to_string()); + + tracing::debug!( + "{LOG_PREFIX} get_page thread={thread_id} total={total} offset={offset} returned={} has_more={has_more}", + items.len() + ); + + TranscriptPage { + thread_id: thread_id.to_string(), + items, + total, + next_cursor, + has_more, + has_transcript: true, + } +} + +/// Parse the opaque cursor into a numeric offset (0 on absent/invalid). +fn parse_cursor(cursor: Option<&str>) -> usize { + cursor + .map(str::trim) + .filter(|c| !c.is_empty()) + .and_then(|c| c.parse::().ok()) + .unwrap_or(0) +} + +#[cfg(test)] +#[path = "transcript_subagent_anchor_tests.rs"] +mod subagent_anchor_tests; +#[cfg(test)] +#[path = "transcript_view_tests.rs"] +mod tests; + +#[cfg(test)] +#[path = "prompt_tools_tests.rs"] +mod prompt_tools_tests; + +#[cfg(test)] +#[path = "transcript_ordering_tests.rs"] +mod ordering_tests; diff --git a/crates/tinyagents-session/src/transcript/view/project.rs b/crates/tinyagents-session/src/transcript/view/project.rs new file mode 100644 index 00000000..08f550de --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/project.rs @@ -0,0 +1,699 @@ +//! Project raw session-transcript records into typed display items. +//! +//! Turns the append-only log's [`DisplayRecord`]s (message lines, compaction +//! markers, interrupted partials) into the frontend's chat vocabulary +//! ([`DisplayItem`]), sanitizing injected scaffolding as it goes. File +//! resolution lives in [`super::resolve`]; sub-agent trails are placed by +//! [`super::subagents`]. + +use std::collections::{HashSet, VecDeque}; +use std::path::{Path, PathBuf}; + +use crate::transcript::{self, CompactionMarker, DisplayMessage, DisplayRecord}; + +use super::TOOL_RESULT_FAILURES_METADATA_KEY; + +use super::prompt_tools::{self, CallRegistry, PromptToolResult}; +use super::resolve; +use super::subagents; +use super::types::{DisplayItem, ProjectedTranscript, ToolCallFailure, ToolCallStatus}; + +const LOG_PREFIX: &str = "[threads][transcript]"; + +/// The scaffolding line injected onto every user message (see +/// `agent::prompts::current_datetime_line`). Stripped at projection so the UI +/// shows the user's actual words, not the per-turn time stamp. +const DATETIME_PREFIX: &str = "Current Date & Time:"; + +/// A legacy/alternate channel-context prefix. Kept for defensiveness; the +/// live injector currently only prepends [`DATETIME_PREFIX`]. +const CHANNEL_CONTEXT_PREFIX: &str = "[Channel context]"; + +type NativeToolCall = (String, String, String); +type NativeToolEnvelope = (String, Vec); + +/// Resolve a thread's root transcript, discover its sub-agent siblings, and +/// project everything into display items. Returns `None` when the thread has +/// no root transcript yet (brand-new thread / first turn not persisted). +pub fn project_thread(workspace_dir: &Path, thread_id: &str) -> Option { + project_thread_scoped(workspace_dir, thread_id, None) +} + +/// Project only the transcript roots owned by `agent_id` when supplied. +pub fn project_thread_scoped( + workspace_dir: &Path, + thread_id: &str, + agent_id: Option<&str>, +) -> Option { + let (root_paths, sub_paths) = + resolve::resolve_files_scoped(workspace_dir, thread_id, agent_id)?; + Some(project_from_files( + thread_id, + &root_paths, + &sub_paths, + Some(workspace_dir), + )) +} + +/// Resolve the on-disk file set backing a thread's transcript view: the root +/// generations in chain order plus every sub-agent sibling file. `None` when +/// the thread has no root transcript yet. Exposed so the cache can key on +/// these paths (and their mtimes/lengths) without re-projecting. +pub fn resolve_files( + workspace_dir: &Path, + thread_id: &str, +) -> Option<(Vec, Vec)> { + resolve::resolve_files(workspace_dir, thread_id) +} + +/// Resolve the files for one agent's transcript when `agent_id` is supplied. +pub fn resolve_files_scoped( + workspace_dir: &Path, + thread_id: &str, + agent_id: Option<&str>, +) -> Option<(Vec, Vec)> { + resolve::resolve_files_scoped(workspace_dir, thread_id, agent_id) +} + +/// Project a thread from an already-resolved file set (root generations + +/// sub-agent siblings). Missing/unreadable files degrade to empty rather than +/// failing. +/// +/// A root whose `_meta.parent_session_id` names the previous root is that +/// session's next compaction generation: it opens with the retained set +/// rewritten, so those rows are dropped (see [`resolve::drop_retained_rows`]) +/// and a [`DisplayItem::Compaction`] marks the seam instead. +pub fn project_from_files( + thread_id: &str, + root_paths: &[PathBuf], + sub_paths: &[PathBuf], + workspace_dir: Option<&Path>, +) -> ProjectedTranscript { + tracing::debug!( + "{LOG_PREFIX} projecting thread={thread_id} roots={} subagent_files={}", + root_paths.len(), + sub_paths.len() + ); + + let mut records: Vec = Vec::new(); + let mut previous: Option<(Option, Vec)> = None; + for root_path in root_paths { + let display = match transcript::read_transcript_display(root_path) { + Ok(display) => display, + Err(err) => { + tracing::warn!( + "{LOG_PREFIX} failed to read root transcript {}: {err}", + root_path.display() + ); + continue; + } + }; + let successor_of_previous = matches!( + (&previous, display.meta.parent_session_id.as_deref()), + (Some((Some(prev_id), _)), Some(parent)) if prev_id == parent + ); + if successor_of_previous { + let predecessor = previous + .as_ref() + .map(|(_, prev_records)| resolve::generation_rows(prev_records)) + .unwrap_or_default(); + let (kept, retained) = resolve::drop_retained_rows(&display.records, predecessor); + tracing::debug!( + "{LOG_PREFIX} generation {} retained={} new_records={}", + root_path.display(), + retained.len(), + kept.len() + ); + let first_kept = kept.iter().find_map(|record| match record { + DisplayRecord::Message(msg) => Some(msg), + DisplayRecord::Compaction(_) => None, + }); + let request_id = first_kept.and_then(|message| message.request_id.clone()); + records.push(DisplayRecord::Compaction(CompactionMarker { + replacement: retained, + ts: first_kept + .and_then(|message| message.ts.clone()) + .or_else(|| Some(display.meta.created.clone())) + .filter(|ts| !ts.is_empty()), + request_id, + })); + records.extend(kept); + } else { + records.extend(display.records.iter().cloned()); + } + previous = Some((display.meta.session_id.clone(), display.records)); + } + + let mut items = project_records(&records); + let top_level = items.len(); + subagents::attach( + &mut items, + sub_paths, + &subagents::turn_segments(&records), + workspace_dir, + ); + tracing::debug!( + "{LOG_PREFIX} projected thread={thread_id} top_level_items={top_level} subagents={}", + items.len() - top_level + ); + + ProjectedTranscript { + thread_id: thread_id.to_string(), + items, + } +} + +/// Project one file's display records into display items, in file order. +/// +/// - System lines are dropped (they carry the tool-policy preamble and other +/// scaffolding that must never render as a chat item). +/// - `reasoning_content` on an assistant line becomes a [`DisplayItem::Reasoning`] +/// preceding its message, carrying the same `iteration`. +/// - An assistant line's tool calls come from its native provider envelope. +/// Only a line that is *not* an envelope falls back to the calls recorded on +/// its usage (legacy text-dialect rows) — and never to a call id already +/// projected, which is how an aggregate list copied onto a final answer +/// would otherwise duplicate every call of the turn. +/// - Each call registers a pending [`DisplayItem::ToolCall`]; a later +/// `role:"tool"` line pairs to one by id, falling back to FIFO order. +/// - A prompt-guided round (calls in the visible text, results folded into one +/// `[Tool results]` user line) projects to the same shape: see +/// [`prompt_tools`]. +/// - Interrupted partials and compaction markers pass through as their items. +/// - A [`DisplayItem::TurnBoundary`] is emitted whenever `request_id` changes. +pub fn project_records(records: &[DisplayRecord]) -> Vec { + let mut projector = Projector { + calls: CallRegistry::from_records(records), + ..Projector::default() + }; + for (index, record) in records.iter().enumerate() { + match record { + DisplayRecord::Message(msg) => { + projector.turn_boundary(msg); + // A prompt-guided round's results are on the NEXT line; they are + // the only record of which calls this assistant line issued. + projector.next_results = match records.get(index + 1) { + Some(DisplayRecord::Message(next)) => prompt_tools::parse_tool_results(next), + _ => None, + }; + projector.message(msg); + } + DisplayRecord::Compaction(marker) => { + project_compaction(marker, &mut projector.items); + // A compaction supersedes prior context; drop stale pending + // pairings so a post-compaction result never binds to them. + projector.pending.clear(); + } + } + } + projector.items +} + +#[derive(Default)] +struct Projector { + items: Vec, + /// Pending tool calls awaiting a result line: (call_id, index into `items`). + pending: VecDeque<(String, usize)>, + last_request_id: Option, + /// Every tool-call id already projected, so a row repeating calls issued + /// earlier (the aggregate usage list) does not render them twice. + seen_call_ids: HashSet, + /// The model-call ordinal of the last assistant row in the current turn — + /// the fallback `iteration` for rows written without one. + step: u32, + /// Every call the file records, for resolving a prompt-guided round. + calls: CallRegistry, + /// Result blocks of the line after the one being projected, when that line + /// is a prompt-guided `[Tool results]` turn. + next_results: Option>, +} + +impl Projector { + fn turn_boundary(&mut self, msg: &DisplayMessage) { + let Some(rid) = msg.request_id.as_deref() else { + return; + }; + if self.last_request_id.as_deref() != Some(rid) { + self.items.push(DisplayItem::TurnBoundary { + request_id: rid.to_string(), + }); + self.last_request_id = Some(rid.to_string()); + self.step = 0; + self.seen_call_ids.clear(); + self.pending.clear(); + } + } + + fn message(&mut self, msg: &DisplayMessage) { + // Interrupted partial: display-only, carries its own thinking. + if msg.interrupted { + self.items.push(DisplayItem::InterruptedPartial { + text: msg.message.content.clone(), + thinking: msg.reasoning_content.clone(), + }); + return; + } + + match msg.message.role.as_str() { + "system" => { + // Scaffolding (tool-policy preamble, etc.) — never a display item. + tracing::debug!("{LOG_PREFIX} sanitize: dropped system line from projection"); + } + "user" => { + // The strict dialect replay shape also starts with + // `[Tool results]`. Handle it first so its persisted failure + // IDs are applied rather than treating every block as success. + if let Some(results) = + tinytools_agent::dialect::parse_replayed_results(&msg.message.content) + { + self.text_tool_results(msg, results); + return; + } + if let Some(blocks) = prompt_tools::parse_tool_results(msg) { + // Tool plumbing, not something the user said. + for block in blocks { + self.pair_result( + block.id, + block.body, + ToolCallStatus::Success, + None, + msg.ts.clone(), + ); + } + return; + } + // A legacy turn without request ids still restarts the step + // count at its prompt. + self.step = 0; + self.seen_call_ids.clear(); + let raw = msg.message.content.clone(); + let sanitized = sanitize_user_content(&raw); + if sanitized.is_some() { + tracing::debug!( + "{LOG_PREFIX} sanitize: stripped injected prefix from user message" + ); + } + self.items.push(DisplayItem::UserMessage { + content: raw, + display_content: sanitized, + request_id: msg.request_id.clone(), + ts: msg.ts.clone(), + }); + } + "assistant" => self.assistant(msg), + "tool" => self.tool_result(msg), + other => { + tracing::debug!( + "{LOG_PREFIX} projecting unknown role {other:?} as assistant message" + ); + self.items.push(DisplayItem::AssistantMessage { + content: msg.message.content.clone(), + interim: false, + request_id: msg.request_id.clone(), + model: msg.turn_usage.as_ref().map(|tu| tu.model.clone()), + iteration: msg.iteration, + ts: msg.ts.clone(), + }); + } + } + } + + fn assistant(&mut self, msg: &DisplayMessage) { + // Rows the writer stamped keep their own number; earlier writers only + // stamped the final row, so unstamped rows count up within the turn. + let iteration = msg.iteration.unwrap_or(self.step + 1); + self.step = iteration; + + // Reasoning precedes the message it belongs to. + if let Some(reasoning) = msg.reasoning_content.as_deref() + && !reasoning.trim().is_empty() + { + self.items.push(DisplayItem::Reasoning { + text: reasoning.to_string(), + iteration: Some(iteration), + }); + } + + let native_envelope = parse_native_tool_envelope(&msg.message.content); + let tool_calls: Vec = match &native_envelope { + Some((_, calls)) => calls.clone(), + None => { + let recorded: Vec = msg + .turn_usage + .as_ref() + .map(|tu| { + tu.tool_calls + .iter() + .map(|call| { + (call.id.clone(), call.name.clone(), call.arguments.clone()) + }) + .collect() + }) + .unwrap_or_default(); + let total = recorded.len(); + let fresh: Vec = recorded + .into_iter() + .filter(|(id, _, _)| id.is_empty() || !self.seen_call_ids.contains(id)) + .collect(); + if fresh.len() < total { + tracing::debug!( + "{LOG_PREFIX} dropped {} already-projected tool call(s) repeated on an assistant row iteration={iteration}", + total - fresh.len() + ); + } + fresh + } + }; + // A prompt-guided line records no calls of its own; the `[Tool results]` + // line after it names the calls it made. + let tool_calls = match self.next_results.take() { + Some(blocks) if tool_calls.is_empty() => { + let inferred = self + .calls + .calls_for(&msg.request_id, &blocks, &self.seen_call_ids); + tracing::debug!( + "{LOG_PREFIX} prompt-guided round: {} call(s) inferred from the next [Tool results] line iteration={iteration}", + inferred.len() + ); + inferred + } + _ => tool_calls, + }; + let interim = !tool_calls.is_empty(); + + // Native tool-call turns are persisted as their provider envelope so they + // can be replayed byte-faithfully. The display projection needs only the + // envelope's visible `content`; rendering/sanitizing the whole JSON object + // makes the narration disappear (and risks showing raw tool JSON). + let visible_content = native_envelope + .map(|(content, _)| content) + .unwrap_or_else(|| msg.message.content.clone()); + + // The assistant's prose (if any) shows before its tool calls. + if !visible_content.trim().is_empty() { + self.items.push(DisplayItem::AssistantMessage { + content: visible_content, + interim, + request_id: msg.request_id.clone(), + model: msg.turn_usage.as_ref().map(|tu| tu.model.clone()), + iteration: Some(iteration), + ts: msg.ts.clone(), + }); + } + + for (call_id, name, arguments) in tool_calls { + let args = parse_tool_args(&arguments); + if !call_id.is_empty() { + self.seen_call_ids.insert(call_id.clone()); + } + // Transcripts written while the codec filed a turn's calls on its + // final row record a text-dialect call *after* its own result, + // which already projected as an orphan. Name that settled row + // rather than adding a second one that never settles. + if let Some(DisplayItem::ToolCall { + name: settled_name, + args: settled_args, + iteration: settled_iteration, + .. + }) = settled_orphan_mut(&mut self.items, &call_id) + { + tracing::debug!("{LOG_PREFIX} call {call_id} recorded after its result — merged"); + *settled_name = name; + *settled_args = args; + *settled_iteration = Some(iteration); + continue; + } + self.items.push(DisplayItem::ToolCall { + call_id: call_id.clone(), + name, + iteration: Some(iteration), + args, + result: None, + status: ToolCallStatus::Running, + failure: None, + ts: msg.ts.clone(), + }); + self.pending.push_back((call_id, self.items.len() - 1)); + } + } + + fn tool_result(&mut self, msg: &DisplayMessage) { + let (result, wrapped_id) = unwrap_tool_result(&msg.message.content); + // A failed tool line (`ToolResult::is_error`, stamped at persistence) + // pairs to an error row with a failure payload instead of a false + // success. + let (status, failure) = if msg.failure { + ( + ToolCallStatus::Error, + Some(ToolCallFailure { + detail: msg.failure_detail.clone(), + }), + ) + } else { + (ToolCallStatus::Success, None) + }; + let call_id = msg.message.id.clone().or(wrapped_id); + self.pair_result(call_id, result, status, failure, msg.ts.clone()); + } + + /// Settle the pending call a result belongs to — by id first, else FIFO — or + /// emit an orphan row so the output is not lost. + fn pair_result( + &mut self, + call_id: Option, + result: String, + status: ToolCallStatus, + failure: Option, + ts: Option, + ) { + // Pair by explicit call id first, else FIFO. + let idx = call_id + .as_deref() + .and_then(|id| take_pending_by_id(&mut self.pending, id)) + .or_else(|| self.pending.pop_front().map(|(_, idx)| idx)); + + if let Some(idx) = idx + && let Some(DisplayItem::ToolCall { + result: slot, + status: status_slot, + failure: failure_slot, + .. + }) = self.items.get_mut(idx) + { + *slot = Some(result); + *status_slot = status; + *failure_slot = failure; + return; + } + + // Orphan result (no matching assistant tool_call recorded) — surface it + // as a best-effort completed tool row so the output is not lost. + tracing::debug!("{LOG_PREFIX} tool result with no pending call — emitting orphan tool row"); + self.items.push(DisplayItem::ToolCall { + call_id: call_id.unwrap_or_default(), + name: "tool".to_string(), + iteration: None, + args: None, + result: Some(result), + status, + failure, + ts, + }); + } +} + +/// Decode the native provider replay envelope embedded in `ChatMessage.content`. +/// Returns visible assistant prose plus `(id, name, arguments)` calls. +pub(super) fn parse_native_tool_envelope(raw: &str) -> Option { + let value = serde_json::from_str::(raw).ok()?; + let object = value.as_object()?; + let calls = object.get("tool_calls")?.as_array()?; + let content = match object.get("content") { + Some(serde_json::Value::String(content)) => content.clone(), + Some(serde_json::Value::Null) | None => String::new(), + _ => return None, + }; + let calls = calls + .iter() + .filter_map(|call| { + let call = call.as_object()?; + let id = call.get("id")?.as_str()?.to_string(); + let name = call.get("name")?.as_str()?.to_string(); + let arguments = match call.get("arguments") { + Some(serde_json::Value::String(arguments)) => arguments.clone(), + Some(arguments) => arguments.to_string(), + None => String::new(), + }; + Some((id, name, arguments)) + }) + .collect(); + Some((content, calls)) +} + +/// Keys a native tool-result replay envelope may carry besides `content`. +const TOOL_RESULT_ENVELOPE_KEYS: &[&str] = &["content", "tool_call_id", "tool_name", "name"]; + +/// Unwrap a native tool-result replay envelope +/// (`{"tool_call_id":…,"content":…}`) to the tool's own output plus the call +/// id it names. Anything else — including a tool whose genuine output happens +/// to be JSON with other keys — is returned verbatim. +fn unwrap_tool_result(raw: &str) -> (String, Option) { + let Ok(serde_json::Value::Object(object)) = serde_json::from_str::(raw) + else { + return (raw.to_string(), None); + }; + let (Some(content), Some(call_id)) = ( + object.get("content").and_then(serde_json::Value::as_str), + object + .get("tool_call_id") + .and_then(serde_json::Value::as_str), + ) else { + return (raw.to_string(), None); + }; + if !object + .keys() + .all(|key| TOOL_RESULT_ENVELOPE_KEYS.contains(&key.as_str())) + { + return (raw.to_string(), None); + } + ( + content.to_string(), + Some(call_id.to_string()).filter(|id| !id.is_empty()), + ) +} + +/// Pair each result of a text-dialect `[Tool results]` row with its pending +/// call. Failure status comes from the ids the session codec recorded on the +/// row ([`TOOL_RESULT_FAILURES_METADATA_KEY`]); a result with no pending call +/// surfaces as an orphan row, as for a native `tool` line. +impl Projector { + fn text_tool_results( + &mut self, + msg: &DisplayMessage, + results: Vec, + ) { + project_text_tool_results(msg, results, &mut self.items, &mut self.pending); + } +} + +fn project_text_tool_results( + msg: &DisplayMessage, + results: Vec, + items: &mut Vec, + pending: &mut VecDeque<(String, usize)>, +) { + let failed: Vec<&str> = msg + .message + .extra_metadata + .as_ref() + .and_then(|meta| meta.get(TOOL_RESULT_FAILURES_METADATA_KEY)) + .and_then(serde_json::Value::as_array) + .map(|ids| ids.iter().filter_map(serde_json::Value::as_str).collect()) + .unwrap_or_default(); + tracing::debug!( + "{LOG_PREFIX} text-dialect results row results={} failed={} pending={}", + results.len(), + failed.len(), + pending.len() + ); + for result in results { + let (status, failure) = if failed.contains(&result.tool_call_id.as_str()) { + ( + ToolCallStatus::Error, + Some(ToolCallFailure { detail: None }), + ) + } else { + (ToolCallStatus::Success, None) + }; + if let Some(idx) = take_pending_by_id(pending, &result.tool_call_id) + && let Some(DisplayItem::ToolCall { + result: slot, + status: status_slot, + failure: failure_slot, + .. + }) = items.get_mut(idx) + { + *slot = Some(result.content); + *status_slot = status; + *failure_slot = failure; + continue; + } + items.push(DisplayItem::ToolCall { + call_id: result.tool_call_id, + name: "tool".to_string(), + iteration: None, + args: None, + result: Some(result.content), + status, + failure, + ts: msg.ts.clone(), + }); + } +} + +/// The already-settled orphan row for `call_id` in the current turn — a result +/// that projected before any call named it. +fn settled_orphan_mut<'a>( + items: &'a mut [DisplayItem], + call_id: &str, +) -> Option<&'a mut DisplayItem> { + let turn_start = items + .iter() + .rposition(|item| matches!(item, DisplayItem::TurnBoundary { .. })) + .map_or(0, |idx| idx + 1); + items[turn_start..].iter_mut().find(|item| { + matches!( + item, + DisplayItem::ToolCall { call_id: id, name, result: Some(_), .. } + if id == call_id && name == "tool" + ) + }) +} + +/// Remove and return the pending entry whose call id matches `id`, if any. +fn take_pending_by_id(pending: &mut VecDeque<(String, usize)>, id: &str) -> Option { + let pos = pending.iter().position(|(cid, _)| cid == id)?; + pending.remove(pos).map(|(_, idx)| idx) +} + +fn project_compaction(marker: &CompactionMarker, items: &mut Vec) { + items.push(DisplayItem::Compaction { + replaced_count: 0, + kept_count: marker.replacement.len(), + ts: marker.ts.clone(), + request_id: marker.request_id.clone(), + }); +} + +/// Parse a tool call's raw argument string into JSON when possible; a +/// non-JSON string is wrapped so the frontend still receives structured args. +fn parse_tool_args(raw: &str) -> Option { + let trimmed = raw.trim(); + if trimmed.is_empty() { + return None; + } + match serde_json::from_str::(trimmed) { + Ok(value) => Some(value), + Err(_) => Some(serde_json::Value::String(trimmed.to_string())), + } +} + +/// Strip the injected scaffolding prefix from a user message, returning the +/// sanitized body only when a prefix was actually present (so the caller can +/// tag rather than mutate — the raw `content` is preserved alongside). +fn sanitize_user_content(content: &str) -> Option { + let trimmed_start = content.trim_start(); + if trimmed_start.starts_with(DATETIME_PREFIX) + || trimmed_start.starts_with(CHANNEL_CONTEXT_PREFIX) + { + // The injector prepends the scaffolding line followed by a blank line, + // then the user's actual text. Strip the first paragraph. + if let Some(idx) = content.find("\n\n") { + let body = content[idx + 2..].to_string(); + return Some(body); + } + // No body after the prefix — the whole message was scaffolding. + return Some(String::new()); + } + None +} diff --git a/crates/tinyagents-session/src/transcript/view/prompt_tools.rs b/crates/tinyagents-session/src/transcript/view/prompt_tools.rs new file mode 100644 index 00000000..051a88a4 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/prompt_tools.rs @@ -0,0 +1,165 @@ +//! Prompt-guided (text-mode) tool turns in the display projection. +//! +//! A provider without native tool calling persists a tool round differently +//! from a native one, and the projection used to read only the native shape: +//! +//! - the assistant line that issued the calls carries **no** `tool_calls` +//! (the calls rode its visible text and were parsed out of it); +//! - the results come back as ONE `user` line, `[Tool results]` followed by a +//! `…` block per call, not as `tool` lines; +//! - the turn's calls are recorded once, on the FINAL assistant line's +//! `tool_calls`, which is the answer rather than a line that called anything. +//! +//! Read naively that inverts the turn: every narration projected as a final +//! answer (non-interim, so the UI dropped it), the real answer projected as an +//! interim tool-calling step with every call of the turn hung after it, and +//! the `[Tool results]` line projected as something the user said. A reopened +//! thread then showed the answer twice and its tools after it — nothing like +//! the turn the user watched stream. +//! +//! This module recognises that shape so `project` can put each call right +//! after the narration that issued it, pair each result block to its call, and +//! stop re-attaching already-shown calls to the answer. + +use std::collections::HashMap; + +use crate::transcript::{DisplayMessage, DisplayRecord}; + +/// The prefix `messages_to_text_mode_chat` / `coalesce_prompt_tool_results` +/// write on a folded tool-results user turn. +const TOOL_RESULTS_MARKER: &str = "[Tool results]"; +const RESULT_OPEN: &str = "` block of a `[Tool results]` user line. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(super) struct PromptToolResult { + /// The call id, when the block carries one (``). + pub id: Option, + pub body: String, +} + +/// The result blocks of a `[Tool results]` user line, or `None` when the line +/// is not one. A marker with no parseable block still returns `Some(vec![])`: +/// it is tool plumbing either way and must never render as a user message. +pub(super) fn parse_tool_results(msg: &DisplayMessage) -> Option> { + if msg.message.role != "user" { + return None; + } + let rest = msg + .message + .content + .trim_start() + .strip_prefix(TOOL_RESULTS_MARKER)?; + let mut blocks = Vec::new(); + let mut cursor = rest; + while let Some(start) = cursor.find(RESULT_OPEN) { + let after_open = &cursor[start + RESULT_OPEN.len()..]; + let Some(tag_end) = after_open.find('>') else { + break; + }; + let id = attribute(&after_open[..tag_end], "id"); + let body_and_rest = &after_open[tag_end + 1..]; + let (body, next) = match body_and_rest.find(RESULT_CLOSE) { + Some(end) => ( + &body_and_rest[..end], + &body_and_rest[end + RESULT_CLOSE.len()..], + ), + None => (body_and_rest, ""), + }; + blocks.push(PromptToolResult { + id, + body: body.trim_matches('\n').to_string(), + }); + cursor = next; + } + Some(blocks) +} + +/// `name="value"` out of a tag's attribute text. +fn attribute(attrs: &str, name: &str) -> Option { + let needle = format!("{name}=\""); + let start = attrs.find(&needle)? + needle.len(); + let value = &attrs[start..]; + let value = &value[..value.find('"')?]; + (!value.is_empty()).then(|| value.to_string()) +} + +/// `(name, arguments)` of every call a file records, by call id, plus each +/// turn's call ids in issue order (for id-less result blocks). +#[derive(Debug, Default)] +pub(super) struct CallRegistry { + by_id: HashMap, + by_turn: HashMap, Vec>, +} + +impl CallRegistry { + /// Index the calls every assistant line records (native per-line calls, + /// and the turn-level list a prompt-guided turn stamps on its answer). + pub(super) fn from_records(records: &[DisplayRecord]) -> Self { + let mut registry = Self::default(); + for record in records { + let DisplayRecord::Message(msg) = record else { + continue; + }; + if msg.message.role != "assistant" { + continue; + } + let Some(usage) = msg.turn_usage.as_ref() else { + continue; + }; + for call in &usage.tool_calls { + if call.id.is_empty() || registry.by_id.contains_key(&call.id) { + continue; + } + registry + .by_id + .insert(call.id.clone(), (call.name.clone(), call.arguments.clone())); + registry + .by_turn + .entry(msg.request_id.clone()) + .or_default() + .push(call.id.clone()); + } + } + registry + } + + /// The calls a `[Tool results]` line answers, as `(id, name, arguments)`, + /// in block order. A block with an id resolves by it; an id-less block + /// takes the turn's next call not yet `emitted`. Unknown ids keep their id + /// with the generic name `tool`, so the result still lands on a row. + pub(super) fn calls_for( + &self, + request_id: &Option, + blocks: &[PromptToolResult], + emitted: &std::collections::HashSet, + ) -> Vec<(String, String, String)> { + let mut taken: Vec = Vec::new(); + let mut turn_order = self + .by_turn + .get(request_id) + .into_iter() + .flatten() + .filter(|id| !emitted.contains(*id)); + blocks + .iter() + .map(|block| { + let id = match &block.id { + Some(id) => id.clone(), + None => turn_order + .find(|id| !taken.contains(id)) + .cloned() + .unwrap_or_default(), + }; + taken.push(id.clone()); + let (name, arguments) = self + .by_id + .get(&id) + .cloned() + .unwrap_or_else(|| ("tool".to_string(), String::new())); + (id, name, arguments) + }) + .collect() + } +} diff --git a/crates/tinyagents-session/src/transcript/view/prompt_tools_tests.rs b/crates/tinyagents-session/src/transcript/view/prompt_tools_tests.rs new file mode 100644 index 00000000..0bd225dd --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/prompt_tools_tests.rs @@ -0,0 +1,147 @@ +//! Prompt-guided (text-mode) tool rounds in the display projection — see +//! `prompt_tools.rs`. + +use super::project::project_records; +use super::types::DisplayItem; +use crate::transcript::{self, read_transcript_display}; +use std::path::{Path, PathBuf}; +use tempfile::TempDir; + +fn write_raw(workspace: &Path, stem: &str, thread_id: &str, body: &[&str]) -> PathBuf { + let path = transcript::resolve_keyed_transcript_path(workspace, stem).expect("resolve"); + let mut buf = format!( + r#"{{"_meta":{{"version":1,"agent":"orchestrator","dispatcher":"xml","created":"2026-09-24T00:00:00Z","updated":"2026-09-24T00:00:10Z","turn_count":1,"input_tokens":1,"output_tokens":1,"cached_input_tokens":0,"charged_amount_usd":0.0,"thread_id":"{thread_id}"}}}}"# + ); + buf.push('\n'); + for line in body { + buf.push_str(line); + buf.push('\n'); + } + std::fs::write(&path, buf).expect("write raw transcript"); + path +} + +/// A native turn: the calling line carries its own `tool_calls`. +fn native_turn_body() -> Vec<&'static str> { + vec![ + r#"{"role":"user","content":"What's the weather in NYC?","request_id":"req-1"}"#, + r#"{"role":"assistant","content":"Let me check.","provider":"anthropic","model":"claude-x","usage":{"input":10,"output":5,"cached_input":0,"cost_usd":0.001},"ts":"2026-07-21T09:00:01Z","tool_calls":[{"id":"call-1","name":"get_weather","arguments":"{\"city\":\"NYC\"}"}],"iteration":1,"request_id":"req-1"}"#, + r#"{"role":"tool","content":"72F and sunny","id":"call-1","request_id":"req-1"}"#, + r#"{"role":"assistant","content":"It's 72F and sunny in NYC.","provider":"anthropic","model":"claude-x","usage":{"input":20,"output":8,"cached_input":0,"cost_usd":0.002},"ts":"2026-07-21T09:00:02Z","iteration":2,"request_id":"req-1"}"#, + ] +} + +/// A prompt-guided (text-mode) turn, as `session_raw` actually records it: the +/// calling lines carry no `tool_calls`, each round's results are ONE +/// `[Tool results]` user line, and the turn's calls are stamped once on the +/// final answer line. +fn prompt_guided_turn_body(result_ids: bool) -> Vec { + let block = |id: &str, body: &str| { + if result_ids { + format!("\\n{body}\\n") + } else { + format!("\\n{body}\\n") + } + }; + vec![ + r#"{"role":"user","content":"Check the config, then search.","request_id":"req-p"}"#.to_string(), + r#"{"role":"assistant","content":"Let me read the config first.","request_id":"req-p"}"#.to_string(), + format!( + r#"{{"role":"user","content":"[Tool results]\n{}","request_id":"req-p"}}"#, + block("call-read", "port = 8080") + ), + r#"{"role":"assistant","content":"Now I will search for the setting.","request_id":"req-p"}"#.to_string(), + format!( + r#"{{"role":"user","content":"[Tool results]\n{}","request_id":"req-p"}}"#, + block("call-grep", "config.toml:3: setting = on") + ), + r#"{"role":"assistant","content":"The setting is on.","provider":"openhuman","model":"m","usage":{"input":1,"output":1,"cached_input":0,"cost_usd":0.0},"tool_calls":[{"id":"call-read","name":"file_read","arguments":"{\"path\":\"/etc/config.toml\"}"},{"id":"call-grep","name":"grep","arguments":"{\"pattern\":\"setting\"}"}],"iteration":3,"ts":"2026-09-24T05:20:52Z","request_id":"req-p"}"#.to_string(), + ] +} + +/// What a reader sees of a projected turn, in order. +fn outline(items: &[DisplayItem]) -> Vec { + items + .iter() + .filter_map(|item| match item { + DisplayItem::UserMessage { content, .. } => Some(format!("user:{content}")), + DisplayItem::AssistantMessage { + content, interim, .. + } => Some(format!( + "{}:{content}", + if *interim { "narration" } else { "answer" } + )), + DisplayItem::ToolCall { + call_id, + name, + status, + result, + .. + } => Some(format!( + "tool:{call_id}:{name}:{status:?}:{}", + result.as_deref().unwrap_or("-") + )), + _ => None, + }) + .collect() +} + +/// The prompt-guided turn used to project inverted: both narrations as final +/// answers, the answer as an interim step with every call after it, and each +/// `[Tool results]` line as something the user said. A reopened thread showed +/// the answer twice with its tools below it. +#[test] +fn prompt_guided_turn_projects_like_a_native_one() { + let dir = TempDir::new().unwrap(); + let body = prompt_guided_turn_body(true); + let body: Vec<&str> = body.iter().map(String::as_str).collect(); + let path = write_raw(dir.path(), "300_orchestrator", "thr_p", &body); + let display = read_transcript_display(&path).unwrap(); + + assert_eq!( + outline(&project_records(&display.records)), + vec![ + "user:Check the config, then search.".to_string(), + "narration:Let me read the config first.".to_string(), + "tool:call-read:file_read:Success:port = 8080".to_string(), + "narration:Now I will search for the setting.".to_string(), + "tool:call-grep:grep:Success:config.toml:3: setting = on".to_string(), + "answer:The setting is on.".to_string(), + ] + ); +} + +#[test] +fn prompt_guided_results_without_ids_pair_in_issue_order() { + let dir = TempDir::new().unwrap(); + let body = prompt_guided_turn_body(false); + let body: Vec<&str> = body.iter().map(String::as_str).collect(); + let path = write_raw(dir.path(), "301_orchestrator", "thr_q", &body); + let display = read_transcript_display(&path).unwrap(); + + let outline = outline(&project_records(&display.records)); + assert_eq!(outline[2], "tool:call-read:file_read:Success:port = 8080"); + assert_eq!( + outline[4], + "tool:call-grep:grep:Success:config.toml:3: setting = on" + ); + assert_eq!(outline[5], "answer:The setting is on."); +} + +/// A native turn records its calls on the line that made them; the answer +/// line's calls are only skipped when they were already placed. +#[test] +fn native_turn_projection_is_unchanged() { + let dir = TempDir::new().unwrap(); + let path = write_raw(dir.path(), "302_orchestrator", "thr_n", &native_turn_body()); + let display = read_transcript_display(&path).unwrap(); + assert_eq!( + outline(&project_records(&display.records)), + vec![ + "user:What's the weather in NYC?".to_string(), + "narration:Let me check.".to_string(), + "tool:call-1:get_weather:Success:72F and sunny".to_string(), + "answer:It's 72F and sunny in NYC.".to_string(), + ] + ); +} diff --git a/crates/tinyagents-session/src/transcript/view/resolve.rs b/crates/tinyagents-session/src/transcript/view/resolve.rs new file mode 100644 index 00000000..dd0b5818 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/resolve.rs @@ -0,0 +1,269 @@ +//! Which files back a thread's transcript view, and in what order. +//! +//! Two things make this more than "every root whose `_meta.thread_id` +//! matches": +//! +//! - **Session generations.** A compaction seals generation `n` and opens +//! `{stem}.g{n+1}`, which inherits `_meta.created` and opens with the +//! retained message set rewritten. Ordering by `created` then path put +//! `X.g1` before `X` (and `.g10` before `.g2`), and concatenating every +//! generation rendered the retained rows twice. Generations are ordered by +//! the tinyagents `session_chain` instead, and [`drop_retained_rows`] removes +//! the rewritten prefix when a successor is projected. +//! - **Adopted legacy roots.** A pre-identity conversation's timestamped roots +//! are folded into the session file on first resume and left untouched on +//! disk, so once a session file exists they are duplicates of its head. +//! +//! Sub-agent files are discovered by `_meta.thread_id` as well as by the +//! legacy `{root_stem}__` prefix: a sub-agent's stem chains the parent's +//! *session key* (`{unix}_{agent}`), which for a session-identity thread is not +//! the root file's stem (`{thread}.{agent}` with digests), so the prefix alone +//! found none of them. + +use std::collections::{HashMap, HashSet}; +use std::fs; +use std::io::{BufRead, BufReader}; +use std::path::{Path, PathBuf}; + +use crate::transcript::{ + self, DisplayMessage, DisplayRecord, FileTranscriptLocator, SessionRef, TranscriptLocator, +}; + +const LOG_PREFIX: &str = "[threads][transcript][resolve]"; + +/// The `_meta` header fields resolution needs, read from a file's first line +/// only (the full reader parses the whole file). +#[derive(Debug, Default, Clone)] +pub(super) struct HeadMeta { + pub(super) thread_id: Option, + pub(super) agent_id: Option, + pub(super) session_id: Option, +} + +/// Read the first-line `_meta` header of a transcript. `None` when the file +/// cannot be read or does not start with a meta line. +pub(super) fn read_head_meta(path: &Path) -> Option { + let file = fs::File::open(path).ok()?; + let mut first = String::new(); + BufReader::new(file).read_line(&mut first).ok()?; + let value: serde_json::Value = serde_json::from_str(first.trim()).ok()?; + let meta = value.get("_meta")?; + let field = |key: &str| { + meta.get(key) + .and_then(serde_json::Value::as_str) + .map(str::to_string) + }; + Some(HeadMeta { + thread_id: field("thread_id"), + agent_id: field("agent_id"), + session_id: field("session_id"), + }) +} + +/// Resolve the root generations and sub-agent siblings backing `thread_id`. +/// `None` when the thread has no root transcript yet. +pub fn resolve_files( + workspace_dir: &Path, + thread_id: &str, +) -> Option<(Vec, Vec)> { + resolve_files_scoped(workspace_dir, thread_id, None) +} + +pub(super) fn resolve_files_scoped( + workspace_dir: &Path, + thread_id: &str, + agent_id: Option<&str>, +) -> Option<(Vec, Vec)> { + let mut found = transcript::find_root_transcripts_for_thread(workspace_dir, thread_id); + if let Some(agent_id) = agent_id { + found.retain(|path| { + read_head_meta(path) + .and_then(|meta| meta.agent_id) + .as_deref() + == Some(agent_id) + }); + } + if found.is_empty() { + return None; + } + let raw_dir = found[0].parent()?.to_path_buf(); + let roots = order_root_files(workspace_dir, thread_id, found); + let subs = discover_subagent_files(&raw_dir, thread_id, &roots); + tracing::debug!( + "{LOG_PREFIX} thread={thread_id} roots={} subagent_files={}", + roots.len(), + subs.len() + ); + Some((roots, subs)) +} + +fn file_name(path: &Path) -> Option<&std::ffi::OsStr> { + path.file_name() +} + +/// Order the thread's roots: each session's generations oldest-first (from +/// `session_chain`), legacy (pre-identity) roots dropped once a session file +/// exists. A thread with no session file keeps the scan's order unchanged. +fn order_root_files(workspace_dir: &Path, thread_id: &str, roots: Vec) -> Vec { + let metas: Vec<(PathBuf, HeadMeta)> = roots + .into_iter() + .map(|path| { + let meta = read_head_meta(&path).unwrap_or_default(); + (path, meta) + }) + .collect(); + if !metas.iter().any(|(_, meta)| meta.session_id.is_some()) { + return metas.into_iter().map(|(path, _)| path).collect(); + } + + let locator = FileTranscriptLocator::new(workspace_dir); + let mut seen: HashSet = HashSet::new(); + let mut ordered = Vec::new(); + let mut dropped_legacy = 0usize; + for (path, meta) in &metas { + if meta.session_id.is_none() { + dropped_legacy += 1; + continue; + } + let Some(name) = file_name(path) else { + continue; + }; + if seen.contains(name) { + continue; + } + let chain: Vec = meta + .agent_id + .as_deref() + .map(|agent_id| { + let session = SessionRef::scoped(thread_id, agent_id); + locator + .session_chain(&session) + .iter() + .filter_map(|generation| { + transcript::resolve_keyed_transcript_path( + workspace_dir, + &transcript::session_stem(generation), + ) + .ok() + }) + .collect() + }) + .unwrap_or_default(); + if chain.iter().any(|link| file_name(link) == Some(name)) { + for link in chain { + if let Some(link_name) = file_name(&link) + && seen.insert(link_name.to_os_string()) + { + ordered.push(link); + } + } + } else { + // A session file this thread's own identity does not derive (a + // different key scheme): keep it where the scan put it. + seen.insert(name.to_os_string()); + ordered.push(path.clone()); + } + } + if dropped_legacy > 0 { + tracing::debug!( + "{LOG_PREFIX} thread={thread_id} dropped {dropped_legacy} adopted legacy root(s) \ + in favour of the session chain" + ); + } + ordered +} + +/// Every sub-agent (`__`) transcript of this thread in `raw_dir`: those whose +/// stem extends a root stem (legacy layout), plus those whose `_meta.thread_id` +/// names the thread (session-identity layout). Sorted by path. +fn discover_subagent_files(raw_dir: &Path, thread_id: &str, roots: &[PathBuf]) -> Vec { + let prefixes: Vec = roots + .iter() + .filter_map(|root| root.file_stem().and_then(|stem| stem.to_str())) + .map(|stem| format!("{stem}__")) + .collect(); + let entries = match fs::read_dir(raw_dir) { + Ok(entries) => entries, + Err(error) => { + tracing::debug!( + "{LOG_PREFIX} subagent discovery read_dir failed dir={} error={error}", + raw_dir.display() + ); + return Vec::new(); + } + }; + let mut paths: Vec = entries + .flatten() + .map(|entry| entry.path()) + .filter(|path| path.extension().and_then(|ext| ext.to_str()) == Some("jsonl")) + .filter(|path| { + let Some(stem) = path.file_stem().and_then(|stem| stem.to_str()) else { + return false; + }; + if !stem.contains("__") { + return false; + } + prefixes.iter().any(|prefix| stem.starts_with(prefix)) + || read_head_meta(path) + .and_then(|meta| meta.thread_id) + .is_some_and(|id| id == thread_id) + }) + .collect(); + paths.sort(); + paths +} + +/// Identity of one message row for cross-generation de-duplication. +type RowKey = (String, String, Option, Option); + +fn row_key(msg: &DisplayMessage) -> RowKey { + ( + msg.message.role.clone(), + msg.message.content.clone(), + msg.message.id.clone(), + msg.request_id.clone(), + ) +} + +/// Multiset of the message rows of one generation, for [`drop_retained_rows`]. +pub(super) fn generation_rows(records: &[DisplayRecord]) -> HashMap { + let mut rows = HashMap::new(); + for record in records { + if let DisplayRecord::Message(msg) = record { + *rows.entry(row_key(msg)).or_insert(0) += 1; + } + } + rows +} + +/// Split a successor generation's records into `(new records, retained rows)`. +/// +/// A successor opens with the retained set rewritten verbatim (same role, +/// content, id and — because retained rows keep their own correlation id — +/// the same `request_id`). Each such row consumes one occurrence from the +/// predecessor's multiset, so a genuinely repeated row beyond what the +/// predecessor held still renders. +pub(super) fn drop_retained_rows( + records: &[DisplayRecord], + mut predecessor: HashMap, +) -> (Vec, Vec) { + let mut kept = Vec::with_capacity(records.len()); + let mut retained = Vec::new(); + let mut matching_prefix = true; + for record in records { + if let DisplayRecord::Message(msg) = record + && matching_prefix + && let Some(count) = predecessor.get_mut(&row_key(msg)) + && *count > 0 + { + *count -= 1; + retained.push((**msg).clone()); + continue; + } + if matches!(record, DisplayRecord::Message(_)) { + matching_prefix = false; + } + kept.push(record.clone()); + } + (kept, retained) +} diff --git a/crates/tinyagents-session/src/transcript/view/subagents.rs b/crates/tinyagents-session/src/transcript/view/subagents.rs new file mode 100644 index 00000000..786ff083 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/subagents.rs @@ -0,0 +1,448 @@ +//! Sub-agent trails: project each delegated run's sibling transcript and +//! place it next to the tool call that spawned it. +//! +//! The transcript records no explicit delegation-call → file link: a child's +//! `_meta` carries its own `task_id`, the parent's rows carry only tool-call +//! ids, and neither names the other. So correlation is by evidence, in order: +//! +//! 1. **Turn** — the child's spawn time (the leading unix seconds of its stem +//! suffix) against the parent turns' commit timestamps +//! ([`anchor_request_id`]). +//! 2. **Call** — within that turn, the first unclaimed tool call that targets +//! the child's agent (`delegate_{agent}`, an `agent_id` argument, …), else +//! the first unclaimed delegation-shaped call. +//! +//! An uncorrelated child lands at the end of its turn (or of the list when +//! there are no turns) instead of after every root item, which is where all +//! sub-agents used to go. + +use std::path::{Path, PathBuf}; + +use crate::transcript::{self, DisplayRecord}; + +use super::project::{parse_native_tool_envelope, project_records}; +use super::types::{DisplayItem, SubagentStatus, ToolCallStatus}; + +const LOG_PREFIX: &str = "[threads][transcript][subagents]"; + +/// Max sub-agent nesting depth the projection descends; bounded so a worker +/// that itself delegates still surfaces, without unbounded fan-out. +const MAX_SUBAGENT_DEPTH: usize = 3; + +/// Prefix the delegation runner puts on a result it gave up on. +const INCOMPLETE_MARKER: &str = "[SUBAGENT_INCOMPLETE]"; + +/// Prefix of an async spawn's acknowledgement — success of the *spawn*, not +/// of the run, so it says nothing about the child's terminal state. +const ASYNC_ACCEPTED_PREFIX: &str = "Accepted async sub-agent"; + +/// Argument keys a spawn/delegate tool uses to name its target agent. +const TARGET_ARG_KEYS: &[&str] = &["agent_id", "agent", "subagent", "subagent_type", "target"]; + +/// A projected child run awaiting placement. +struct ChildRun { + /// Unix seconds the child was spawned at, from its stem. + spawn_unix: Option, + agent_id: Option, + /// Spawn task id, when the transcript recorded one — the key the run + /// ledger's `AgentRunUpsert.id` uses, so it's also the key for the exact + /// `parentCallId` correlation in [`find_exact_spawning_call`]. + task_id: Option, + item: DisplayItem, + /// The child's own terminal evidence, before the spawning call is known. + own_state: OwnState, +} + +#[derive(Clone, Copy, PartialEq, Eq)] +enum OwnState { + Completed, + Interrupted, + Unknown, +} + +/// Place every direct child of the root (`__`-once stems) into `items`. +/// `segments` are the root turns' `(request_id, commit unix)` pairs. +pub(super) fn attach( + items: &mut Vec, + sub_paths: &[PathBuf], + segments: &[(String, i64)], + workspace_dir: Option<&Path>, +) { + let children = build_children(sub_paths, None, 0, workspace_dir); + place(items, children, segments, workspace_dir); +} + +/// Project the direct children of `parent_stem` (or of the roots, when +/// `None`), recursing into their own children. +fn build_children( + sub_paths: &[PathBuf], + parent_stem: Option<&str>, + depth: usize, + workspace_dir: Option<&Path>, +) -> Vec { + if depth >= MAX_SUBAGENT_DEPTH { + return Vec::new(); + } + let mut children = Vec::new(); + for path in sub_paths { + let Some(stem) = path.file_stem().and_then(|s| s.to_str()) else { + continue; + }; + let suffix = match parent_stem { + Some(parent) => match stem.strip_prefix(parent).and_then(|r| r.strip_prefix("__")) { + Some(rest) if !rest.contains("__") => rest, + _ => continue, + }, + // A root's direct child has exactly one `__` separator. + None => match stem.split_once("__") { + Some((_, rest)) if !rest.contains("__") => rest, + _ => continue, + }, + }; + if let Some(child) = build_child(path, stem, suffix, sub_paths, depth, workspace_dir) { + children.push(child); + } + } + children.sort_by_key(|child| child.spawn_unix); + children +} + +fn build_child( + path: &Path, + stem: &str, + suffix: &str, + sub_paths: &[PathBuf], + depth: usize, + workspace_dir: Option<&Path>, +) -> Option { + let display = match transcript::read_transcript_display(path) { + Ok(display) => display, + Err(err) => { + tracing::warn!( + "{LOG_PREFIX} failed to read sub-agent transcript {}: {err}", + path.display() + ); + return None; + } + }; + let own_state = own_state(&display.records); + let mut items = project_records(&display.records); + let grandchildren = build_children(sub_paths, Some(stem), depth + 1, workspace_dir); + place( + &mut items, + grandchildren, + &turn_segments(&display.records), + workspace_dir, + ); + + let task_id = display.meta.task_id.clone().filter(|id| !id.is_empty()); + let agent_id = display + .meta + .agent_id + .clone() + .or_else(|| Some(display.meta.agent_name.clone())) + .filter(|id| !id.is_empty()); + let id = task_id.clone().unwrap_or_else(|| suffix.to_string()); + let spawn_unix = child_spawn_unix(suffix); + // The spawn timestamp encoded in the sub-agent's own file stem (used + // above to anchor it to a parent turn) doubles as this item's `ts` — + // sub-agent transcripts carry no back-link to a delegating request, so + // there is no per-message `ts` to inherit the way the root projector + // pulls one from `DisplayMessage.ts`. + let ts = spawn_unix + .and_then(|unix| chrono::DateTime::from_timestamp(unix, 0).map(|dt| dt.to_rfc3339())); + Some(ChildRun { + spawn_unix, + agent_id: agent_id.clone(), + task_id: task_id.clone(), + item: DisplayItem::Subagent { + id, + agent_id, + task_id, + call_id: None, + status: SubagentStatus::Running, + request_id: None, + ts, + items, + }, + own_state, + }) +} + +/// Exact correlation: the run ledger's `AgentRunUpsert.metadata.parentCallId` +/// for this task (stamped by `progress_bridge`'s `SubagentSpawned` handling), +/// resolved to the unclaimed [`DisplayItem::ToolCall`] with that `call_id`. +/// +/// Preferred over [`find_spawning_call`]'s timestamp/target-argument +/// heuristic whenever it resolves — the ledger has the actual call id, no +/// guessing required. `None` on any miss (no workspace, no task id, no +/// ledger row, no matching/unclaimed call), so callers fall back to the +/// heuristic unconditionally. +fn find_exact_spawning_call( + items: &[DisplayItem], + claimed: &[bool], + task_id: Option<&str>, + workspace_dir: Option<&Path>, +) -> Option { + let workspace_dir = workspace_dir?; + let task_id = task_id?; + let run = crate::run_ledger::get_agent_run(workspace_dir, task_id) + .ok() + .flatten()?; + let parent_call_id = run.metadata.get("parentCallId")?.as_str()?; + items + .iter() + .enumerate() + .find_map(|(index, item)| match item { + DisplayItem::ToolCall { call_id, .. } + if !claimed[index] && call_id == parent_call_id => + { + Some(index) + } + _ => None, + }) +} + +/// What the child's own transcript says about how it ended. +fn own_state(records: &[DisplayRecord]) -> OwnState { + let last = records.iter().rev().find_map(|record| match record { + DisplayRecord::Message(msg) if msg.message.role != "system" => Some(msg), + _ => None, + }); + match last { + Some(msg) if msg.interrupted => OwnState::Interrupted, + Some(msg) + if msg.message.role == "assistant" + && parse_native_tool_envelope(&msg.message.content) + .is_none_or(|(_, calls)| calls.is_empty()) => + { + OwnState::Completed + } + _ => OwnState::Unknown, + } +} + +/// Insert `children` into `items`, each after its correlated spawning call +/// (claimed at most once), else at the end of its anchored turn. +fn place( + items: &mut Vec, + children: Vec, + segments: &[(String, i64)], + workspace_dir: Option<&Path>, +) { + if children.is_empty() { + return; + } + let mut claimed = vec![false; items.len()]; + // (insert position, order) — applied back-to-front afterwards. + let mut inserts: Vec<(usize, usize, DisplayItem)> = Vec::new(); + for (order, mut child) in children.into_iter().enumerate() { + let request_id = anchor_request_id(child.spawn_unix, segments); + let (start, end) = turn_range(items, request_id.as_deref()); + let pick = + find_exact_spawning_call(items, &claimed, child.task_id.as_deref(), workspace_dir) + .or_else(|| { + find_spawning_call(items, &claimed, start, end, child.agent_id.as_deref()) + }); + let (position, call) = match pick { + Some(index) => { + claimed[index] = true; + (index + 1, Some(index)) + } + None => (end, None), + }; + let (call_id, call_status, call_result) = match call.and_then(|i| items.get(i)) { + Some(DisplayItem::ToolCall { + call_id, + status, + result, + .. + }) => (Some(call_id.clone()), Some(*status), result.clone()), + _ => (None, None, None), + }; + let status = derive_status(child.own_state, call_status, call_result.as_deref()); + if let DisplayItem::Subagent { + id, + call_id: call_slot, + status: status_slot, + request_id: request_slot, + .. + } = &mut child.item + { + tracing::debug!( + "{LOG_PREFIX} subagent id={id} agent={:?} request_id={request_id:?} call_id={call_id:?} status={status:?}", + child.agent_id + ); + *call_slot = call_id; + *status_slot = status; + *request_slot = request_id; + } + inserts.push((position, order, child.item)); + } + // Back-to-front keeps earlier positions valid; for one position, the + // later child is inserted first so spawn order is preserved. + inserts.sort_by(|a, b| b.0.cmp(&a.0).then(b.1.cmp(&a.1))); + for (position, _, item) in inserts { + items.insert(position.min(items.len()), item); + } +} + +/// Terminal state from the spawning call's outcome and the child's own +/// transcript. A failed or incomplete delegation wins; then the child's own +/// ending; then a settled synchronous call. +fn derive_status( + own: OwnState, + call_status: Option, + call_result: Option<&str>, +) -> SubagentStatus { + let result = call_result.map(str::trim_start).unwrap_or_default(); + if call_status == Some(ToolCallStatus::Error) || result.starts_with(INCOMPLETE_MARKER) { + return SubagentStatus::Failed; + } + match own { + OwnState::Interrupted => SubagentStatus::Interrupted, + OwnState::Completed => SubagentStatus::Completed, + OwnState::Unknown + if call_status == Some(ToolCallStatus::Success) + && !result.starts_with(ASYNC_ACCEPTED_PREFIX) => + { + SubagentStatus::Completed + } + OwnState::Unknown => SubagentStatus::Running, + } +} + +/// `[start, end)` of `request_id`'s items (after its boundary, up to the next +/// one); the whole list when the turn is unknown. +fn turn_range(items: &[DisplayItem], request_id: Option<&str>) -> (usize, usize) { + let Some(request_id) = request_id else { + return (0, items.len()); + }; + let Some(boundary) = items.iter().position( + |item| matches!(item, DisplayItem::TurnBoundary { request_id: rid } if rid == request_id), + ) else { + return (0, items.len()); + }; + let end = items[boundary + 1..] + .iter() + .position(|item| matches!(item, DisplayItem::TurnBoundary { .. })) + .map_or(items.len(), |offset| boundary + 1 + offset); + (boundary + 1, end) +} + +fn find_spawning_call( + items: &[DisplayItem], + claimed: &[bool], + start: usize, + end: usize, + agent_id: Option<&str>, +) -> Option { + let candidates = || { + (start..end).filter_map(|index| match &items[index] { + DisplayItem::ToolCall { name, args, .. } if !claimed[index] => { + Some((index, name.as_str(), args.as_ref())) + } + _ => None, + }) + }; + if let Some(agent_id) = agent_id + && let Some((index, ..)) = + candidates().find(|(_, name, args)| call_targets_agent(name, *args, agent_id)) + { + return Some(index); + } + candidates() + .find(|(_, name, _)| is_delegation_tool(name)) + .map(|(index, ..)| index) +} + +/// Whether a tool call names `agent_id` as its delegation target. +fn call_targets_agent(name: &str, args: Option<&serde_json::Value>, agent_id: &str) -> bool { + let agent = agent_id.to_ascii_lowercase(); + let name = name.to_ascii_lowercase(); + if name == format!("delegate_{agent}") || name == format!("delegate_to_{agent}") { + return true; + } + let named_in_args = args + .and_then(serde_json::Value::as_object) + .is_some_and(|args| { + TARGET_ARG_KEYS.iter().any(|key| { + args.get(*key) + .and_then(serde_json::Value::as_str) + .is_some_and(|value| value.eq_ignore_ascii_case(&agent)) + }) + }); + if named_in_args { + return true; + } + // Alias tools such as `plan` → `planner`. The length floor keeps a + // short generic tool name from matching an agent by accident. + let stripped = name + .strip_prefix("delegate_to_") + .or_else(|| name.strip_prefix("delegate_")) + .unwrap_or(&name); + stripped.len() >= 5 && agent.starts_with(stripped) +} + +fn is_delegation_tool(name: &str) -> bool { + name.starts_with("delegate") || name.starts_with("spawn_") +} + +/// The turns' `(request_id, commit unix)` pairs, in file order: the last +/// parseable timestamp of each `request_id` run. +/// +/// Every stamped row of a turn carries the turn's *commit* time (the writer +/// stamps it when the turn is appended), so this is when the turn ended, not +/// when it began. +pub(super) fn turn_segments(records: &[DisplayRecord]) -> Vec<(String, i64)> { + let mut segments: Vec<(String, i64)> = Vec::new(); + for record in records { + let DisplayRecord::Message(msg) = record else { + continue; + }; + let (Some(rid), Some(ts)) = (msg.request_id.as_deref(), msg.ts.as_deref()) else { + continue; + }; + let Some(unix) = parse_rfc3339_unix(ts) else { + continue; + }; + match segments.last_mut() { + Some((last, end)) if last == rid => *end = (*end).max(unix), + _ => segments.push((rid.to_string(), unix)), + } + } + segments +} + +/// Extract a sub-agent's spawn unix timestamp (seconds) from its stem suffix +/// (`{unix}_{nanos}_{agent}…`). `None` for non-numeric legacy stems. +fn child_spawn_unix(stem_suffix: &str) -> Option { + stem_suffix + .split('_') + .next() + .and_then(|s| s.parse::().ok()) +} + +fn parse_rfc3339_unix(ts: &str) -> Option { + chrono::DateTime::parse_from_rfc3339(ts) + .ok() + .map(|dt| dt.timestamp()) +} + +/// Anchor a sub-agent to the turn that was running at `child_unix`: the first +/// turn whose commit time is at or after the spawn. +/// +/// Fallbacks: no segments → `None` (unanchored); unknown spawn time, or a +/// spawn after every recorded commit (a turn still in flight) → the newest +/// turn. +fn anchor_request_id(child_unix: Option, segments: &[(String, i64)]) -> Option { + let last = segments.last()?; + let Some(child_unix) = child_unix else { + return Some(last.0.clone()); + }; + segments + .iter() + .find(|(_, end)| *end >= child_unix) + .or(Some(last)) + .map(|(rid, _)| rid.clone()) +} diff --git a/crates/tinyagents-session/src/transcript/view/transcript_ordering_tests.rs b/crates/tinyagents-session/src/transcript/view/transcript_ordering_tests.rs new file mode 100644 index 00000000..82d91274 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/transcript_ordering_tests.rs @@ -0,0 +1,566 @@ +//! Turn-ordering tests for the transcript view that go through the **real +//! writer** (`append_transcript_turn`) rather than hand-built JSONL: tool-call +//! de-duplication, tool-result unwrapping, per-step iterations, sub-agent +//! placement, and compaction-generation chains. + +use super::project::{project_records, project_thread, resolve_files}; +use super::types::{DisplayItem, SubagentStatus, ToolCallStatus}; +use crate::transcript::{ + self, SessionRef, TranscriptMessage, TranscriptMeta, TranscriptToolCall, TurnUsage, + read_transcript, read_transcript_display, +}; +use std::path::{Path, PathBuf}; +use tempfile::TempDir; + +fn meta(thread_id: &str, session_id: Option, parent: Option) -> TranscriptMeta { + TranscriptMeta { + session_id, + parent_session_id: parent, + agent_name: "orchestrator".into(), + agent_id: Some("orchestrator".into()), + agent_type: Some("root".into()), + dispatcher: "native".into(), + provider: Some("anthropic".into()), + model: Some("claude-x".into()), + created: "2023-11-14T22:00:00+00:00".into(), + updated: "2023-11-14T22:00:00+00:00".into(), + turn_count: 1, + prefix_message_count: None, + input_tokens: 0, + output_tokens: 0, + cached_input_tokens: 0, + charged_amount_usd: 0.0, + thread_id: Some(thread_id.into()), + task_id: None, + } +} + +fn rfc3339(unix: i64) -> String { + chrono::DateTime::from_timestamp(unix, 0) + .unwrap() + .to_rfc3339() +} + +fn usage(iteration: u32, unix: i64, tool_calls: Vec) -> TurnUsage { + TurnUsage { + provider: "anthropic".into(), + model: "claude-x".into(), + usage: transcript::MessageUsage { + input: 10, + output: 5, + cached_input: 0, + context_window: 0, + cost_usd: 0.001, + }, + ts: rfc3339(unix), + reasoning_content: None, + tool_calls, + iteration, + } +} + +fn call(id: &str, name: &str, arguments: &str) -> TranscriptToolCall { + TranscriptToolCall { + id: id.into(), + name: name.into(), + arguments: arguments.into(), + extra_content: None, + } +} + +/// An assistant row as the native dialect persists it: the provider replay +/// envelope, with its reasoning in `extra_metadata`. +fn envelope(content: &str, calls: &[(&str, &str, &str)], reasoning: &str) -> TranscriptMessage { + let calls: Vec = calls + .iter() + .map(|(id, name, args)| serde_json::json!({"id": id, "name": name, "arguments": args})) + .collect(); + let mut row = TranscriptMessage::assistant( + serde_json::json!({"content": content, "tool_calls": calls}).to_string(), + ); + row.extra_metadata = Some(serde_json::json!({ "reasoning_content": reasoning })); + row +} + +/// A tool-result row as the native dialect persists it: wrapped. +fn tool_result(id: &str, output: &str) -> TranscriptMessage { + let mut row = TranscriptMessage::new( + "tool", + serde_json::json!({"tool_call_id": id, "content": output}).to_string(), + ); + row.id = Some(id.into()); + row +} + +fn final_answer(text: &str, reasoning: &str) -> TranscriptMessage { + let mut row = TranscriptMessage::assistant(text); + row.extra_metadata = Some(serde_json::json!({ "reasoning_content": reasoning })); + row +} + +/// One two-step tool turn: step 1 narrates and calls `get_weather`, step 2 +/// calls two tools in parallel with no narration, step 3 answers. +fn weather_turn() -> Vec { + vec![ + TranscriptMessage::new("system", "policy"), + TranscriptMessage::new("user", "Weather in NYC and SF?"), + envelope( + "Let me check.", + &[("c1", "get_weather", r#"{"city":"NYC"}"#)], + "think one", + ), + tool_result("c1", "72F"), + envelope( + "", + &[ + ("c2", "get_weather", r#"{"city":"SF"}"#), + ("c3", "web_search", r#"{"q":"sf fog"}"#), + ], + "think two", + ), + tool_result("c2", "60F"), + tool_result("c3", "foggy"), + final_answer("NYC 72F, SF 60F and foggy.", "think three"), + ] +} + +fn write_turn(path: &Path, rows: &[TranscriptMessage], turn_usage: &TurnUsage) { + transcript::append_transcript_turn( + path, + &[], + rows, + &meta("thr_w", None, None), + Some(turn_usage), + Some("req-1"), + ) + .unwrap(); +} + +fn assert_weather_projection(items: &[DisplayItem]) { + let tools: Vec<(&str, u32, &str, ToolCallStatus)> = items + .iter() + .filter_map(|item| match item { + DisplayItem::ToolCall { + call_id, + iteration, + result, + status, + .. + } => Some(( + call_id.as_str(), + iteration.unwrap_or_default(), + result.as_deref().unwrap_or_default(), + *status, + )), + _ => None, + }) + .collect(); + assert_eq!( + tools, + vec![ + ("c1", 1, "72F", ToolCallStatus::Success), + ("c2", 2, "60F", ToolCallStatus::Success), + ("c3", 2, "foggy", ToolCallStatus::Success), + ], + "each call once, settled, with its unwrapped output: {items:#?}" + ); + + let answers: Vec<(&str, bool, Option)> = items + .iter() + .filter_map(|item| match item { + DisplayItem::AssistantMessage { + content, + interim, + iteration, + .. + } => Some((content.as_str(), *interim, *iteration)), + _ => None, + }) + .collect(); + assert_eq!( + answers, + vec![ + ("Let me check.", true, Some(1)), + ("NYC 72F, SF 60F and foggy.", false, Some(3)), + ] + ); + + // Each reasoning block carries the iteration of the step it precedes. + let reasoning: Vec<(&str, Option)> = items + .iter() + .filter_map(|item| match item { + DisplayItem::Reasoning { text, iteration } => Some((text.as_str(), *iteration)), + _ => None, + }) + .collect(); + assert_eq!( + reasoning, + vec![ + ("think one", Some(1)), + ("think two", Some(2)), + ("think three", Some(3)), + ] + ); +} + +/// The shape every turn written before this fix has on disk: the turn's +/// aggregate tool outcomes were copied onto the usage of the final answer, +/// which the writer then stamped as that row's `tool_calls`. +#[test] +fn real_writer_turn_with_aggregate_usage_calls_projects_each_call_once() { + let dir = TempDir::new().unwrap(); + let path = transcript::resolve_keyed_transcript_path(dir.path(), "1_orchestrator").unwrap(); + let aggregate = vec![ + call("c1", "get_weather", r#"{"city":"NYC"}"#), + call("c2", "get_weather", r#"{"city":"SF"}"#), + call("c3", "web_search", r#"{"q":"sf fog"}"#), + ]; + write_turn(&path, &weather_turn(), &usage(3, 1_700_000_000, aggregate)); + + let raw = std::fs::read_to_string(&path).unwrap(); + assert!( + raw.lines() + .any(|line| line.contains("NYC 72F") && line.contains("\"tool_calls\"")), + "precondition: the legacy final answer row carries the duplicated calls" + ); + let display = read_transcript_display(&path).unwrap(); + assert_weather_projection(&project_records(&display.records)); +} + +/// The shape the codec writes now: usage without tool calls. Every step is +/// also stamped with its own iteration by the writer. +#[test] +fn real_writer_turn_projects_steps_in_order() { + let dir = TempDir::new().unwrap(); + let path = transcript::resolve_keyed_transcript_path(dir.path(), "2_orchestrator").unwrap(); + write_turn(&path, &weather_turn(), &usage(3, 1_700_000_000, Vec::new())); + + let raw = std::fs::read_to_string(&path).unwrap(); + assert!( + !raw.lines() + .any(|line| line.contains("NYC 72F") && line.contains("\"tool_calls\"")), + "the final answer row carries no tool calls" + ); + let display = read_transcript_display(&path).unwrap(); + assert_weather_projection(&project_records(&display.records)); +} + +/// Only the native `{tool_call_id, content}` wrapper is unwrapped; a tool +/// whose real output is JSON with other keys is shown verbatim. +#[test] +fn tool_output_that_is_not_the_replay_wrapper_is_kept_verbatim() { + let dir = TempDir::new().unwrap(); + let path = transcript::resolve_keyed_transcript_path(dir.path(), "3_orchestrator").unwrap(); + let mut rows = vec![ + TranscriptMessage::new("user", "go"), + envelope("", &[("c1", "fetch_json", "{}")], ""), + ]; + let mut own_json = TranscriptMessage::new("tool", r#"{"content":"x","status":200}"#); + own_json.id = Some("c1".into()); + rows.push(own_json); + rows.push(final_answer("done", "")); + write_turn(&path, &rows, &usage(2, 1_700_000_000, Vec::new())); + + let display = read_transcript_display(&path).unwrap(); + let result = project_records(&display.records) + .into_iter() + .find_map(|item| match item { + DisplayItem::ToolCall { result, .. } => result, + _ => None, + }) + .expect("tool result projected"); + assert_eq!(result, r#"{"content":"x","status":200}"#); +} + +/// Write a session-identity root for `thread_id` (the stem the host binds via +/// `SessionRef::scoped`) and return its path. +fn session_root(workspace: &Path, thread_id: &str) -> (SessionRef, PathBuf) { + let session = SessionRef::scoped(thread_id, "orchestrator"); + let path = + transcript::resolve_keyed_transcript_path(workspace, &transcript::session_stem(&session)) + .unwrap(); + (session, path) +} + +/// A session-identity root is named `{thread}.{agent}` (with digests) while a +/// sub-agent's file chains the parent's *session key* (`{unix}_{agent}…`), so +/// prefix discovery never found it. It is discovered by `_meta.thread_id`, +/// placed right after the call that spawned it, and keyed by its task id. +#[test] +fn subagent_of_a_session_root_is_discovered_and_placed_after_its_spawning_call() { + let dir = TempDir::new().unwrap(); + let thread_id = "thr_sub_place"; + let (session, root) = session_root(dir.path(), thread_id); + let root_meta = meta(thread_id, Some(session.session_id()), None); + + let turn1 = vec![ + TranscriptMessage::new("user", "hi"), + final_answer("hello", ""), + ]; + transcript::append_transcript_turn( + &root, + &[], + &turn1, + &root_meta, + Some(&usage(1, 1_700_000_000, Vec::new())), + Some("req-1"), + ) + .unwrap(); + let persisted = read_transcript(&root).unwrap().messages; + let mut turn2 = persisted.clone(); + turn2.extend([ + TranscriptMessage::new("user", "research bali"), + envelope( + "On it.", + &[ + ("c-fetch", "web_fetch", r#"{"url":"https://x"}"#), + ("c-research", "research", r#"{"prompt":"bali"}"#), + ], + "", + ), + tool_result("c-fetch", "page"), + tool_result("c-research", "Bali is great."), + final_answer("Here's the plan.", ""), + ]); + transcript::append_transcript_turn( + &root, + &persisted, + &turn2, + &root_meta, + Some(&usage(2, 1_700_000_200, Vec::new())), + Some("req-2"), + ) + .unwrap(); + + // Spawned during turn 2 (after turn 1 committed at …000, before …200). + let child_stem = "1699999990_orchestrator_thread-x__1700000100_000000001_researcher_sub-abc"; + let child = transcript::resolve_keyed_transcript_path(dir.path(), child_stem).unwrap(); + let mut child_meta = meta(thread_id, None, None); + child_meta.agent_name = "researcher".into(); + child_meta.agent_id = Some("researcher".into()); + child_meta.agent_type = Some("subagent".into()); + child_meta.task_id = Some("sub-abc-123".into()); + transcript::write_transcript( + &child, + &[ + TranscriptMessage::new("user", "bali"), + TranscriptMessage::assistant("Bali is great."), + ], + &child_meta, + None, + ) + .unwrap(); + + let (_, subs) = resolve_files(dir.path(), thread_id).expect("thread resolves"); + assert_eq!(subs, vec![child.clone()], "discovered by _meta.thread_id"); + + let projected = project_thread(dir.path(), thread_id).expect("project"); + let items = &projected.items; + let research_call = items + .iter() + .position( + |item| matches!(item, DisplayItem::ToolCall { call_id, .. } if call_id == "c-research"), + ) + .expect("research call projected"); + match &items[research_call + 1] { + DisplayItem::Subagent { + id, + agent_id, + task_id, + call_id, + status, + request_id, + items, + .. + } => { + assert_eq!(id, "sub-abc-123"); + assert_eq!(agent_id.as_deref(), Some("researcher")); + assert_eq!(task_id.as_deref(), Some("sub-abc-123")); + assert_eq!(call_id.as_deref(), Some("c-research")); + assert_eq!(*status, SubagentStatus::Completed); + assert_eq!(request_id.as_deref(), Some("req-2")); + assert!(items.iter().any(|inner| matches!( + inner, + DisplayItem::AssistantMessage { content, .. } if content == "Bali is great." + ))); + } + other => panic!("expected the sub-agent right after its call, got {other:?}"), + } + // Not also appended after everything else. + assert_eq!( + items + .iter() + .filter(|item| matches!(item, DisplayItem::Subagent { .. })) + .count(), + 1 + ); + assert!( + matches!(items.last(), Some(DisplayItem::AssistantMessage { content, .. }) if content == "Here's the plan."), + "the turn's final answer still closes the list" + ); +} + +/// A delegation reported incomplete by the runner is a failed run, whatever +/// the child's last row says. +#[test] +fn subagent_status_follows_an_incomplete_delegation_result() { + let dir = TempDir::new().unwrap(); + let root_stem = "900_orchestrator"; + let thread_id = "thr_sub_fail"; + let root = transcript::resolve_keyed_transcript_path(dir.path(), root_stem).unwrap(); + transcript::append_transcript_turn( + &root, + &[], + &[ + TranscriptMessage::new("user", "go"), + envelope("", &[("c1", "delegate_coder", "{}")], ""), + tool_result("c1", "[SUBAGENT_INCOMPLETE] gave up"), + final_answer("Sorry.", ""), + ], + &meta(thread_id, None, None), + Some(&usage(2, 1_700_000_000, Vec::new())), + Some("req-1"), + ) + .unwrap(); + let child = transcript::resolve_keyed_transcript_path( + dir.path(), + &format!("{root_stem}__1699999999_000000001_coder_sub-1"), + ) + .unwrap(); + let mut child_meta = meta(thread_id, None, None); + child_meta.agent_id = Some("coder".into()); + transcript::write_transcript( + &child, + &[ + TranscriptMessage::new("user", "task"), + TranscriptMessage::assistant("partial thoughts"), + ], + &child_meta, + None, + ) + .unwrap(); + + let projected = project_thread(dir.path(), thread_id).expect("project"); + let status = projected + .items + .iter() + .find_map(|item| match item { + DisplayItem::Subagent { + status, call_id, .. + } => Some((*status, call_id.clone())), + _ => None, + }) + .expect("sub-agent projected"); + assert_eq!(status, (SubagentStatus::Failed, Some("c1".to_string()))); +} + +/// A compaction opens `{stem}.g1`, which inherits `created` and starts with +/// the retained rows rewritten. The chain is projected oldest-first, the +/// retained rows once, with a compaction marker at the seam — and an adopted +/// legacy root of the same thread is not projected a second time. +#[test] +fn generation_chain_projects_in_order_without_duplicating_retained_rows() { + let dir = TempDir::new().unwrap(); + let thread_id = "thr_generations"; + let (session, g0) = session_root(dir.path(), thread_id); + let successor = session.next_generation(); + let g1 = transcript::resolve_keyed_transcript_path( + dir.path(), + &transcript::session_stem(&successor), + ) + .unwrap(); + + // A pre-identity root of the same thread, already adopted into g0. + let legacy = + transcript::resolve_keyed_transcript_path(dir.path(), "1600000000_orchestrator").unwrap(); + transcript::append_transcript_turn( + &legacy, + &[], + &[ + TranscriptMessage::new("user", "one"), + final_answer("answer one", ""), + ], + &meta(thread_id, None, None), + None, + Some("req-1"), + ) + .unwrap(); + + let g0_meta = meta(thread_id, Some(session.session_id()), None); + transcript::append_transcript_turn( + &g0, + &[], + &[ + TranscriptMessage::new("user", "one"), + final_answer("answer one", ""), + ], + &g0_meta, + None, + Some("req-1"), + ) + .unwrap(); + let after_one = read_transcript(&g0).unwrap().messages; + let mut two = after_one.clone(); + two.extend([ + TranscriptMessage::new("user", "two"), + final_answer("answer two", ""), + ]); + transcript::append_transcript_turn(&g0, &after_one, &two, &g0_meta, None, Some("req-2")) + .unwrap(); + + // Compaction during turn 3: turn 2 is retained, turn 1 summarised away. + let persisted = read_transcript(&g0).unwrap().messages; + let mut retained: Vec = persisted[2..].to_vec(); + retained.extend([ + TranscriptMessage::new("user", "three"), + final_answer("answer three", ""), + ]); + let g1_meta = meta( + thread_id, + Some(successor.session_id()), + Some(session.session_id()), + ); + transcript::append_transcript_turn(&g1, &[], &retained, &g1_meta, None, Some("req-3")).unwrap(); + + let (roots, _) = resolve_files(dir.path(), thread_id).expect("thread resolves"); + assert_eq!( + roots, + vec![g0.clone(), g1.clone()], + "chain order, legacy dropped" + ); + + let items = project_thread(dir.path(), thread_id) + .expect("project") + .items; + let users: Vec<&str> = items + .iter() + .filter_map(|item| match item { + DisplayItem::UserMessage { content, .. } => Some(content.as_str()), + _ => None, + }) + .collect(); + assert_eq!(users, vec!["one", "two", "three"], "{items:#?}"); + let answers = items + .iter() + .filter(|item| matches!(item, DisplayItem::AssistantMessage { content, .. } if content == "answer two")) + .count(); + assert_eq!(answers, 1, "the retained answer renders once"); + assert!(items.iter().any(|item| matches!( + item, + DisplayItem::Compaction { kept_count, .. } if *kept_count == 2 + ))); +} + +#[test] +fn get_page_missing_thread_is_empty_not_error() { + let dir = TempDir::new().unwrap(); + let page = super::get_page( + dir.path(), + "no_such_thread", + None, + Some(super::DEFAULT_LIMIT), + ); + assert!(!page.has_transcript); + assert_eq!(page.total, 0); + assert!(page.items.is_empty()); +} diff --git a/crates/tinyagents-session/src/transcript/view/transcript_subagent_anchor_tests.rs b/crates/tinyagents-session/src/transcript/view/transcript_subagent_anchor_tests.rs new file mode 100644 index 00000000..5fc07f58 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/transcript_subagent_anchor_tests.rs @@ -0,0 +1,67 @@ +use super::project::project_thread; +use super::tests::write_raw; +use super::types::DisplayItem; +use tempfile::TempDir; + +#[test] +fn subagent_anchors_to_parent_turn_by_spawn_timestamp() { + let dir = TempDir::new().unwrap(); + let root_stem = "800_orchestrator"; + let thread_id = "thr_anchor"; + let t1 = chrono::DateTime::from_timestamp(1_000_000, 0) + .unwrap() + .to_rfc3339(); + let t2 = chrono::DateTime::from_timestamp(2_000_000, 0) + .unwrap() + .to_rfc3339(); + let root_body = [ + r#"{"role":"user","content":"one","request_id":"req-1"}"#.to_string(), + format!( + r#"{{"role":"assistant","content":"a1","provider":"anthropic","model":"m","usage":{{"input":1,"output":1,"cached_input":0,"cost_usd":0.0}},"ts":"{t1}","iteration":1,"request_id":"req-1"}}"# + ), + r#"{"role":"user","content":"two","request_id":"req-2"}"#.to_string(), + format!( + r#"{{"role":"assistant","content":"a2","provider":"anthropic","model":"m","usage":{{"input":1,"output":1,"cached_input":0,"cost_usd":0.0}},"ts":"{t2}","iteration":1,"request_id":"req-2"}}"# + ), + ]; + let root_refs: Vec<&str> = root_body.iter().map(String::as_str).collect(); + write_raw(dir.path(), root_stem, thread_id, &root_refs); + write_raw( + dir.path(), + &format!("{root_stem}__999950_coder"), + thread_id, + &[r#"{"role":"assistant","content":"coder work"}"#], + ); + write_raw( + dir.path(), + &format!("{root_stem}__1999950_planner"), + thread_id, + &[r#"{"role":"assistant","content":"planner work"}"#], + ); + + let mut anchors: Vec<(String, Option)> = project_thread(dir.path(), thread_id) + .expect("project thread") + .items + .iter() + .filter_map(|item| match item { + DisplayItem::Subagent { + request_id, items, .. + } => items.iter().find_map(|inner| match inner { + DisplayItem::AssistantMessage { content, .. } => { + Some((content.clone(), request_id.clone())) + } + _ => None, + }), + _ => None, + }) + .collect(); + anchors.sort(); + + assert_eq!( + anchors, + vec![ + ("coder work".to_string(), Some("req-1".to_string())), + ("planner work".to_string(), Some("req-2".to_string())), + ] + ); +} diff --git a/crates/tinyagents-session/src/transcript/view/transcript_view_subagent_tests.rs b/crates/tinyagents-session/src/transcript/view/transcript_view_subagent_tests.rs new file mode 100644 index 00000000..ca4f44fc --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/transcript_view_subagent_tests.rs @@ -0,0 +1,139 @@ +use super::*; + +#[test] +fn subagent_anchors_to_parent_turn_by_spawn_timestamp() { + let dir = TempDir::new().unwrap(); + let root_stem = "800_orchestrator"; + let thread_id = "thr_anchor"; + let t1 = chrono::DateTime::from_timestamp(1_000_000, 0) + .unwrap() + .to_rfc3339(); + let t2 = chrono::DateTime::from_timestamp(2_000_000, 0) + .unwrap() + .to_rfc3339(); + let root_body = [ + r#"{"role":"user","content":"one","request_id":"req-1"}"#.to_string(), + format!( + r#"{{"role":"assistant","content":"a1","provider":"anthropic","model":"m","usage":{{"input":1,"output":1,"cached_input":0,"cost_usd":0.0}},"ts":"{t1}","iteration":1,"request_id":"req-1"}}"# + ), + r#"{"role":"user","content":"two","request_id":"req-2"}"#.to_string(), + format!( + r#"{{"role":"assistant","content":"a2","provider":"anthropic","model":"m","usage":{{"input":1,"output":1,"cached_input":0,"cost_usd":0.0}},"ts":"{t2}","iteration":1,"request_id":"req-2"}}"# + ), + ]; + let root_refs: Vec<&str> = root_body.iter().map(String::as_str).collect(); + write_raw(dir.path(), root_stem, thread_id, &root_refs); + write_raw( + dir.path(), + &format!("{root_stem}__999950_coder"), + thread_id, + &[r#"{"role":"assistant","content":"coder work"}"#], + ); + write_raw( + dir.path(), + &format!("{root_stem}__1000050_planner"), + thread_id, + &[r#"{"role":"assistant","content":"planner work"}"#], + ); + + let projected = project_thread(dir.path(), thread_id).expect("project thread"); + let mut anchors: Vec<(String, Option)> = projected + .items + .iter() + .filter_map(|item| match item { + DisplayItem::Subagent { + request_id, items, .. + } => { + let marker = items.iter().find_map(|inner| match inner { + DisplayItem::AssistantMessage { content, .. } => Some(content.clone()), + _ => None, + })?; + Some((marker, request_id.clone())) + } + _ => None, + }) + .collect(); + anchors.sort(); + assert_eq!( + anchors, + vec![ + ("coder work".to_string(), Some("req-1".to_string())), + ("planner work".to_string(), Some("req-2".to_string())), + ] + ); +} + +/// Exact correlation (#C1): when the run ledger records the spawning +/// `parentCallId` for this task, it wins over the timestamp/target-argument +/// heuristic — which would otherwise pick the first unclaimed +/// delegation-shaped call, regardless of which one actually spawned this +/// child. +#[test] +fn subagent_correlates_by_ledger_parent_call_id_over_the_heuristic() { + let dir = TempDir::new().unwrap(); + let root_stem = "800_orch_exact"; + let thread_id = "thr_exact"; + let commit_ts = chrono::DateTime::from_timestamp(1_900_000, 0) + .unwrap() + .to_rfc3339(); + let root_body = [ + r#"{"role":"user","content":"do research","request_id":"req-1"}"#.to_string(), + format!( + r#"{{"role":"assistant","content":"","provider":"test","model":"test","usage":{{"input":1,"output":1,"cached_input":0,"cost_usd":0.0}},"tool_calls":[{{"id":"call-decoy","name":"spawn_async_subagent","arguments":"{{}}"}},{{"id":"call-real","name":"spawn_async_subagent","arguments":"{{}}"}}],"iteration":1,"request_id":"req-1","ts":"{commit_ts}"}}"# + ), + ]; + let root_refs: Vec<&str> = root_body.iter().map(String::as_str).collect(); + write_raw(dir.path(), root_stem, thread_id, &root_refs); + + // Spawned at unix 2_000_000 — after the only turn's commit, so it + // anchors to that turn either way; the heuristic would still pick the + // first unclaimed `spawn_*`-shaped call (`call-decoy`) since neither + // call names an agent. Only the exact ledger lookup can tell them apart. + let child_stem = format!("{root_stem}__2000000_000000001_researcher"); + let child = transcript::resolve_keyed_transcript_path(dir.path(), &child_stem).unwrap(); + // Written by hand (not `write_raw_at`'s default `meta_line`) so the + // `_meta` header carries `task_id`/`agent_id` — the ledger correlation + // key `build_child` reads. + let child_meta_line = format!( + r#"{{"_meta":{{"version":1,"agent":"researcher","agent_id":"researcher","agent_type":"subagent","dispatcher":"native","created":"2026-07-21T00:00:00Z","updated":"2026-07-21T00:00:10Z","turn_count":1,"input_tokens":1,"output_tokens":1,"cached_input_tokens":0,"charged_amount_usd":0.0,"thread_id":"{thread_id}","task_id":"sub-exact-1"}}}}"# + ); + std::fs::write( + &child, + format!("{child_meta_line}\n{{\"role\":\"assistant\",\"content\":\"Bali is great.\"}}\n"), + ) + .unwrap(); + + crate::run_ledger::upsert_agent_run( + dir.path(), + crate::run_ledger::AgentRunUpsert { + id: "sub-exact-1".to_string(), + kind: crate::run_ledger::AgentRunKind::Subagent, + parent_run_id: None, + parent_thread_id: Some(thread_id.to_string()), + agent_id: Some("researcher".to_string()), + status: crate::run_ledger::AgentRunStatus::Completed, + prompt_ref: None, + worker_thread_id: None, + checkpoint_path: None, + checkpoint: None, + summary: None, + error: None, + metadata: serde_json::json!({ "parentCallId": "call-real" }), + started_at: None, + completed_at: None, + }, + ) + .expect("seed run ledger row"); + + let projected = project_thread(dir.path(), thread_id).expect("project thread"); + let subagent_call_id = projected.items.iter().find_map(|item| match item { + DisplayItem::Subagent { call_id, .. } => Some(call_id.clone()), + _ => None, + }); + assert_eq!( + subagent_call_id, + Some(Some("call-real".to_string())), + "exact ledger correlation must win over the first-unclaimed heuristic; items={:#?}", + projected.items + ); +} diff --git a/crates/tinyagents-session/src/transcript/view/transcript_view_tests.rs b/crates/tinyagents-session/src/transcript/view/transcript_view_tests.rs new file mode 100644 index 00000000..aa520c4d --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/transcript_view_tests.rs @@ -0,0 +1,659 @@ +//! Projection + pagination + sanitization tests for the transcript view. + +use super::get_page; +use super::project::{project_records, project_thread}; +use super::types::{DisplayItem, ToolCallStatus}; +use crate::transcript::{self, read_transcript_display}; +use std::path::{Path, PathBuf}; +use tempfile::TempDir; + +fn meta_line(thread_id: &str) -> String { + format!( + r#"{{"_meta":{{"version":1,"agent":"orchestrator","dispatcher":"native","created":"2026-07-21T00:00:00Z","updated":"2026-07-21T00:00:10Z","turn_count":1,"input_tokens":30,"output_tokens":13,"cached_input_tokens":0,"charged_amount_usd":0.003,"thread_id":"{thread_id}"}}}}"# + ) +} + +/// Write a raw JSONL transcript (meta header + given body lines) into +/// `session_raw/{stem}.jsonl` and return the path. +pub(super) fn write_raw(workspace: &Path, stem: &str, thread_id: &str, body: &[&str]) -> PathBuf { + let path = transcript::resolve_keyed_transcript_path(workspace, stem).expect("resolve"); + write_raw_at(&path, thread_id, body); + path +} + +fn write_raw_at(path: &Path, thread_id: &str, body: &[&str]) { + let mut buf = meta_line(thread_id); + buf.push('\n'); + for line in body { + buf.push_str(line); + buf.push('\n'); + } + std::fs::write(path, buf).expect("write raw transcript"); +} + +/// A full turn: system scaffolding, a user prompt with the injected datetime +/// prefix, an assistant tool-calling step (reasoning + tool_calls), a tool +/// result, then the final assistant answer. +fn full_turn_body() -> Vec<&'static str> { + vec![ + r#"{"role":"system","content":"[tool-policy preamble] you may use tools ..."}"#, + r#"{"role":"user","content":"Current Date & Time: 2026-07-21 09:00:00 UTC\n\nWhat's the weather in NYC?","request_id":"req-1"}"#, + r#"{"role":"assistant","content":"Let me check.","provider":"anthropic","model":"claude-x","usage":{"input":10,"output":5,"cached_input":0,"cost_usd":0.001},"ts":"2026-07-21T09:00:01Z","reasoning_content":"I should call the weather tool.","tool_calls":[{"id":"call-1","name":"get_weather","arguments":"{\"city\":\"NYC\"}"}],"iteration":1,"request_id":"req-1"}"#, + r#"{"role":"tool","content":"72F and sunny","id":"call-1","request_id":"req-1"}"#, + r#"{"role":"assistant","content":"It's 72F and sunny in NYC.","provider":"anthropic","model":"claude-x","usage":{"input":20,"output":8,"cached_input":0,"cost_usd":0.002},"ts":"2026-07-21T09:00:02Z","iteration":2,"request_id":"req-1"}"#, + ] +} + +#[test] +fn projects_turn_with_tools_reasoning_and_sanitization() { + let dir = TempDir::new().unwrap(); + let path = write_raw(dir.path(), "100_orchestrator", "thr_w", &full_turn_body()); + let display = read_transcript_display(&path).unwrap(); + let items = project_records(&display.records); + + // Expected order: turnBoundary, userMessage, reasoning, assistant(interim), + // toolCall(paired), assistant(final). System line dropped. + assert_eq!(items.len(), 6, "unexpected items: {items:#?}"); + + match &items[0] { + DisplayItem::TurnBoundary { request_id } => assert_eq!(request_id, "req-1"), + other => panic!("expected turnBoundary, got {other:?}"), + } + match &items[1] { + DisplayItem::UserMessage { + content, + display_content, + request_id, + .. + } => { + assert!(content.starts_with("Current Date & Time:"), "raw kept"); + assert_eq!( + display_content.as_deref(), + Some("What's the weather in NYC?"), + "datetime prefix stripped into displayContent" + ); + assert_eq!(request_id.as_deref(), Some("req-1")); + } + other => panic!("expected userMessage, got {other:?}"), + } + match &items[2] { + DisplayItem::Reasoning { text, iteration } => { + assert_eq!(text, "I should call the weather tool."); + assert_eq!( + *iteration, + Some(1), + "reasoning carries its step's iteration" + ); + } + other => panic!("expected reasoning, got {other:?}"), + } + match &items[3] { + DisplayItem::AssistantMessage { + content, + interim, + ts, + .. + } => { + assert_eq!(content, "Let me check."); + assert!(*interim, "tool-calling assistant step is interim"); + assert_eq!( + ts.as_deref(), + Some("2026-07-21T09:00:01Z"), + "assistantMessage carries the underlying record's ts" + ); + } + other => panic!("expected interim assistantMessage, got {other:?}"), + } + match &items[4] { + DisplayItem::ToolCall { + call_id, + name, + args, + result, + status, + failure, + .. + } => { + assert_eq!(call_id, "call-1"); + assert_eq!(name, "get_weather"); + assert_eq!( + args.as_ref() + .and_then(|v| v.get("city")) + .and_then(|v| v.as_str()), + Some("NYC") + ); + assert_eq!(result.as_deref(), Some("72F and sunny"), "paired by id"); + assert_eq!(*status, ToolCallStatus::Success); + assert!(failure.is_none(), "successful tool carries no failure"); + } + other => panic!("expected toolCall, got {other:?}"), + } + match &items[5] { + DisplayItem::AssistantMessage { + content, + interim, + ts, + .. + } => { + assert_eq!(content, "It's 72F and sunny in NYC."); + assert!(!*interim, "final answer is not interim"); + assert_eq!( + ts.as_deref(), + Some("2026-07-21T09:00:02Z"), + "final assistantMessage carries its own record's ts, not the interim step's" + ); + } + other => panic!("expected final assistantMessage, got {other:?}"), + } +} + +#[test] +fn reuses_synthetic_tool_call_ids_in_a_later_turn() { + let dir = TempDir::new().unwrap(); + let path = write_raw( + dir.path(), + "synthetic_ids", + "thr_synthetic", + &[ + r#"{"role":"user","content":"one","request_id":"req-1"}"#, + r#"{"role":"assistant","content":"","provider":"test","model":"test","usage":{"input":1,"output":1,"cached_input":0,"cost_usd":0.0},"ts":"2026-07-21T00:00:01Z","tool_calls":[{"id":"call_0","name":"first","arguments":"{}"}],"request_id":"req-1"}"#, + r#"{"role":"tool","content":"first result","id":"call_0","request_id":"req-1"}"#, + r#"{"role":"user","content":"two","request_id":"req-2"}"#, + r#"{"role":"assistant","content":"","provider":"test","model":"test","usage":{"input":1,"output":1,"cached_input":0,"cost_usd":0.0},"ts":"2026-07-21T00:00:02Z","tool_calls":[{"id":"call_0","name":"second","arguments":"{}"}],"request_id":"req-2"}"#, + r#"{"role":"tool","content":"second result","id":"call_0","request_id":"req-2"}"#, + ], + ); + let display = read_transcript_display(&path).unwrap(); + let items = project_records(&display.records); + + let calls: Vec<_> = items + .iter() + .filter_map(|item| match item { + DisplayItem::ToolCall { name, result, .. } => Some((name.as_str(), result.as_deref())), + _ => None, + }) + .collect(); + assert_eq!( + calls, + vec![ + ("first", Some("first result")), + ("second", Some("second result")) + ] + ); +} + +#[test] +fn recovers_tool_name_from_native_envelope_without_turn_usage() { + let dir = TempDir::new().unwrap(); + let envelope = serde_json::json!({ + "content": null, + "tool_calls": [{ + "id": "call-web-1", + "name": "web_fetch", + "arguments": { "url": "https://example.com" } + }] + }) + .to_string(); + let assistant = serde_json::json!({ + "role": "assistant", + "content": envelope, + "request_id": "req-native" + }) + .to_string(); + let result = + r#"{"role":"tool","content":"Example Domain","id":"call-web-1","request_id":"req-native"}"#; + let path = write_raw( + dir.path(), + "101_orchestrator", + "thr_native", + &[assistant.as_str(), result], + ); + let display = read_transcript_display(&path).unwrap(); + let items = project_records(&display.records); + + let tool = items + .iter() + .find_map(|item| match item { + DisplayItem::ToolCall { + name, + args, + result, + status, + .. + } => Some((name, args, result, status)), + _ => None, + }) + .expect("native envelope tool call projected"); + assert_eq!(tool.0, "web_fetch"); + assert_eq!( + tool.1 + .as_ref() + .and_then(|args| args.get("url")) + .and_then(serde_json::Value::as_str), + Some("https://example.com") + ); + assert_eq!(tool.2.as_deref(), Some("Example Domain")); + assert_eq!(*tool.3, ToolCallStatus::Success); +} + +#[test] +fn projects_compaction_and_interrupted_partial() { + let dir = TempDir::new().unwrap(); + let body = vec![ + r#"{"role":"user","content":"hi","request_id":"req-1"}"#, + r#"{"kind":"compaction","replacement":[{"role":"user","content":"summary so far"}],"ts":"2026-07-21T09:05:00Z","request_id":"req-2"}"#, + r#"{"role":"assistant","content":"partial ans","interrupted":true,"reasoning_content":"mid-thought","iteration":3,"request_id":"req-2"}"#, + ]; + let path = write_raw(dir.path(), "200_orchestrator", "thr_c", &body); + let display = read_transcript_display(&path).unwrap(); + let items = project_records(&display.records); + + let has_compaction = items.iter().any(|i| { + matches!( + i, + DisplayItem::Compaction { kept_count, .. } if *kept_count == 1 + ) + }); + assert!(has_compaction, "compaction projected: {items:#?}"); + + let partial = items + .iter() + .find_map(|i| match i { + DisplayItem::InterruptedPartial { text, thinking } => Some((text, thinking)), + _ => None, + }) + .expect("interrupted partial projected"); + assert_eq!(partial.0, "partial ans"); + assert_eq!(partial.1.as_deref(), Some("mid-thought")); +} + +#[test] +fn legacy_file_without_version_or_request_id_projects() { + let dir = TempDir::new().unwrap(); + // Legacy meta: no `version`, messages carry no `request_id`. + let path = transcript::resolve_keyed_transcript_path(dir.path(), "300_orchestrator").unwrap(); + let raw = concat!( + r#"{"_meta":{"agent":"orchestrator","dispatcher":"native","created":"2026-01-01T00:00:00Z","updated":"2026-01-01T00:00:00Z","turn_count":1,"input_tokens":0,"output_tokens":0,"cached_input_tokens":0,"charged_amount_usd":0.0,"thread_id":"thr_legacy"}}"#, + "\n", + r#"{"role":"user","content":"plain question"}"#, + "\n", + r#"{"role":"assistant","content":"plain answer"}"#, + "\n", + ); + std::fs::write(&path, raw).unwrap(); + let display = read_transcript_display(&path).unwrap(); + let items = project_records(&display.records); + + // No request_id → no turn boundary; user + assistant still project, and an + // un-prefixed user message keeps no displayContent (nothing to strip). + assert!( + !items + .iter() + .any(|i| matches!(i, DisplayItem::TurnBoundary { .. })), + "legacy lines have no request_id, so no boundary" + ); + match items.first() { + Some(DisplayItem::UserMessage { + content, + display_content, + .. + }) => { + assert_eq!(content, "plain question"); + assert_eq!( + display_content.as_deref(), + None, + "no prefix, no displayContent" + ); + } + other => panic!("expected userMessage first, got {other:?}"), + } + assert!(items.iter().any( + |i| matches!(i, DisplayItem::AssistantMessage { content, .. } if content == "plain answer") + )); +} + +#[test] +fn subagent_file_projects_as_nested_item() { + let dir = TempDir::new().unwrap(); + let root_stem = "400_orchestrator"; + write_raw( + dir.path(), + root_stem, + "thr_s", + &[r#"{"role":"user","content":"delegate please","request_id":"req-1"}"#], + ); + // Sub-agent sibling shares the root stem with a `__` suffix. + write_raw( + dir.path(), + &format!("{root_stem}__100_coder"), + "thr_s", + &[ + r#"{"role":"assistant","content":"sub work done","provider":"anthropic","model":"claude-x","usage":{"input":5,"output":3,"cached_input":0,"cost_usd":0.0},"ts":"2026-07-21T09:10:00Z","iteration":1}"#, + ], + ); + + let projected = project_thread(dir.path(), "thr_s").expect("project thread"); + let subagent = projected + .items + .iter() + .find_map(|i| match i { + DisplayItem::Subagent { id, items, .. } => Some((id, items)), + _ => None, + }) + .expect("subagent item present"); + assert_eq!(subagent.0, "100_coder", "unique run id, not the agent name"); + assert!(subagent.1.iter().any( + |i| matches!(i, DisplayItem::AssistantMessage { content, .. } if content == "sub work done") + )); +} + +#[test] +fn canonical_root_and_subagent_project_together() { + let dir = TempDir::new().unwrap(); + let raw_dir = dir.path().join("session_raw"); + std::fs::create_dir_all(&raw_dir).unwrap(); + let root_stem = "450_orchestrator"; + let thread_id = "thr_canonical"; + + write_raw_at( + &raw_dir.join(format!("{root_stem}.jsonl")), + thread_id, + &[r#"{"role":"user","content":"delegate","request_id":"req-1"}"#], + ); + write_raw_at( + &raw_dir.join(format!("{root_stem}__451_coder.jsonl")), + thread_id, + &[r#"{"role":"assistant","content":"scoped sub work"}"#], + ); + + let projected = project_thread(dir.path(), thread_id).expect("project scoped thread"); + assert!(projected.items.iter().any(|item| matches!( + item, + DisplayItem::Subagent { items, .. } + if items.iter().any(|inner| matches!( + inner, + DisplayItem::AssistantMessage { content, .. } + if content == "scoped sub work" + )) + ))); +} + +#[test] +fn get_page_paginates_newest_first_with_cursor() { + let dir = TempDir::new().unwrap(); + // Five plain user messages → five top-level items. + let body: Vec = (0..5) + .map(|i| format!(r#"{{"role":"user","content":"msg-{i}"}}"#)) + .collect(); + let body_refs: Vec<&str> = body.iter().map(String::as_str).collect(); + write_raw(dir.path(), "500_orchestrator", "thr_p", &body_refs); + + // First page: newest first, limit 2 → msg-4, msg-3. + let page1 = get_page(dir.path(), "thr_p", None, Some(2)); + assert_eq!(page1.total, 5); + assert!(page1.has_more); + assert_eq!(page1.items.len(), 2); + assert!( + matches!(&page1.items[0], DisplayItem::UserMessage { content, .. } if content == "msg-4") + ); + assert!( + matches!(&page1.items[1], DisplayItem::UserMessage { content, .. } if content == "msg-3") + ); + + let cursor = page1.next_cursor.clone().expect("next cursor"); + let page2 = get_page(dir.path(), "thr_p", Some(&cursor), Some(2)); + assert!( + matches!(&page2.items[0], DisplayItem::UserMessage { content, .. } if content == "msg-2") + ); + + // Walk to the end. + let last = get_page(dir.path(), "thr_p", page2.next_cursor.as_deref(), Some(2)); + assert!(!last.has_more, "final page exhausts the thread"); + assert!(last.next_cursor.is_none()); +} + +#[test] +fn failed_tool_line_projects_error_status_with_failure_payload() { + let dir = TempDir::new().unwrap(); + // Assistant issues a tool call; the paired tool result line carries the + // additive `failure` flag (stamped at persistence from `is_error`). + let body = vec![ + r#"{"role":"assistant","content":"trying","provider":"anthropic","model":"m","usage":{"input":1,"output":1,"cached_input":0,"cost_usd":0.0},"ts":"2026-07-21T09:00:01Z","tool_calls":[{"id":"call-9","name":"shell","arguments":"{\"cmd\":\"boom\"}"}],"iteration":1,"request_id":"req-1"}"#, + r#"{"role":"tool","content":"error: command not found","id":"call-9","request_id":"req-1","failure":true,"failure_detail":"error: command not found"}"#, + ]; + let path = write_raw(dir.path(), "600_orchestrator", "thr_f", &body); + let display = read_transcript_display(&path).unwrap(); + let items = project_records(&display.records); + + let tool = items + .iter() + .find_map(|i| match i { + DisplayItem::ToolCall { + status, failure, .. + } => Some((status, failure)), + _ => None, + }) + .expect("toolCall projected"); + assert_eq!(*tool.0, ToolCallStatus::Error, "failed tool → error status"); + let failure = tool.1.as_ref().expect("failure payload present"); + assert_eq!(failure.detail.as_deref(), Some("error: command not found")); +} + +/// The **golden test for the turn path's write call site**. +/// +/// Every other test in this file hand-writes JSONL string literals, so they pin +/// the reader/projector but say nothing about the writer the live turn loop +/// actually calls. `tool_failure_metadata_round_trips_write_to_display_line` +/// goes through `write_transcript`, not `append_transcript_turn`. That left the +/// seam this test covers — `append_transcript_turn` → `read_transcript_display` +/// → `project_thread` — with no coverage at all, which is exactly the seam the +/// tinyagents `ChatHistory` migration touches. +/// +/// It fails loudly if a write ever drops `request_id` or `turn_usage`: without +/// `request_id` there is no `DisplayItem::TurnBoundary` (project.rs +/// `maybe_emit_turn_boundary`) and `turn_segments` goes empty, unanchoring every +/// sub-agent; without `turn_usage` every `DisplayItem::ToolCall` disappears +/// (tool calls are read off `turn_usage.tool_calls`), `Reasoning` vanishes, and +/// `AssistantMessage.{model,iteration,interim}` collapse to `None`/`false`. +/// +/// All timestamps are fixed literals so nothing here is clock-dependent. +#[test] +fn append_transcript_turn_projects_full_display_shape() { + let dir = TempDir::new().unwrap(); + let now = "2026-07-21T09:00:00Z".to_string(); + let meta = transcript::TranscriptMeta { + session_id: None, + parent_session_id: None, + agent_name: "orchestrator".into(), + agent_id: Some("orchestrator".into()), + agent_type: Some("root".into()), + dispatcher: "native".into(), + provider: Some("anthropic".into()), + model: Some("claude-x".into()), + created: now.clone(), + updated: now, + turn_count: 1, + prefix_message_count: None, + input_tokens: 30, + output_tokens: 13, + cached_input_tokens: 0, + charged_amount_usd: 0.003, + thread_id: Some("thr_golden".into()), + task_id: None, + }; + + let usage = |cost: f64| transcript::MessageUsage { + input: 20, + output: 8, + cached_input: 0, + context_window: 200_000, + cost_usd: cost, + }; + + // Iteration 1 of the turn: the model reasons and emits a native tool call. + // `turn_usage` attaches to the last assistant row of the written slice, so + // this one lands on the interim assistant. + let interim_usage = transcript::TurnUsage { + provider: "anthropic".into(), + model: "claude-x".into(), + usage: usage(0.001), + ts: "2026-07-21T09:00:01Z".into(), + reasoning_content: Some("I should call the weather tool.".into()), + tool_calls: vec![transcript::TranscriptToolCall { + id: "call-1".into(), + name: "get_weather".into(), + arguments: r#"{"city":"NYC"}"#.into(), + extra_content: None, + }], + iteration: 1, + }; + // Iteration 2: the final answer, no further tool calls. + let final_usage = transcript::TurnUsage { + provider: "anthropic".into(), + model: "claude-x".into(), + usage: usage(0.002), + ts: "2026-07-21T09:00:02Z".into(), + reasoning_content: None, + tool_calls: vec![], + iteration: 2, + }; + + let msg = |id: Option<&str>, role: &str, content: &str| transcript::TranscriptMessage { + id: id.map(str::to_string), + role: role.into(), + content: content.into(), + extra_metadata: None, + cache_breakpoints: Vec::new(), + turn_usage: None, + request_id: None, + preserve_request_id: false, + interrupted: false, + tool_failure: None, + }; + + let first = vec![ + msg(None, "user", "What's the weather in NYC?"), + msg(None, "assistant", "Let me check."), + ]; + let mut second = first.clone(); + second.push(msg(Some("call-1"), "tool", "72F and sunny")); + second.push(msg(None, "assistant", "It's 72F and sunny in NYC.")); + + let path = transcript::resolve_keyed_transcript_path(dir.path(), "900_orchestrator").unwrap(); + // First write creates the file (meta + all lines); the second is a pure + // extension appending only the new tail — both are the real turn-path shape. + transcript::append_transcript_turn( + &path, + &[], + &first, + &meta, + Some(&interim_usage), + Some("req-1"), + ) + .unwrap(); + transcript::append_transcript_turn( + &path, + &first, + &second, + &meta, + Some(&final_usage), + Some("req-1"), + ) + .unwrap(); + + let display = read_transcript_display(&path).unwrap(); + let items = project_records(&display.records); + + // A turn boundary must be emitted from the stamped request_id. + let boundary = items + .iter() + .find_map(|i| match i { + DisplayItem::TurnBoundary { request_id } => Some(request_id.clone()), + _ => None, + }) + .expect("turnBoundary projected from request_id"); + assert_eq!(boundary, "req-1"); + + // Reasoning comes off turn_usage.reasoning_content. + let reasoning = items + .iter() + .find_map(|i| match i { + DisplayItem::Reasoning { text, .. } => Some(text.clone()), + _ => None, + }) + .expect("reasoning projected from turn_usage"); + assert_eq!(reasoning, "I should call the weather tool."); + + // The tool call itself is read off turn_usage.tool_calls, and pairs with the + // role:"tool" line by id. + let tool = items + .iter() + .find_map(|i| match i { + DisplayItem::ToolCall { + call_id, + name, + args, + result, + status, + .. + } => Some(( + call_id.clone(), + name.clone(), + args.clone(), + result.clone(), + *status, + )), + _ => None, + }) + .expect("toolCall projected from turn_usage.tool_calls"); + assert_eq!(tool.0, "call-1"); + assert_eq!(tool.1, "get_weather"); + assert_eq!( + tool.2 + .as_ref() + .and_then(|v| v.get("city")) + .and_then(|v| v.as_str()), + Some("NYC") + ); + assert_eq!(tool.3.as_deref(), Some("72F and sunny")); + assert_eq!(tool.4, ToolCallStatus::Success); + + // Model / iteration / request_id / interim all come off the persisted + // `turn_usage` + `request_id`; each is `None`/`false` if either is dropped. + let assistants: Vec<_> = items + .iter() + .filter_map(|i| match i { + DisplayItem::AssistantMessage { + content, + model, + iteration, + request_id, + interim, + .. + } => Some(( + content.clone(), + model.clone(), + *iteration, + request_id.clone(), + *interim, + )), + _ => None, + }) + .collect(); + assert_eq!(assistants.len(), 2, "unexpected items: {items:#?}"); + + assert_eq!(assistants[0].0, "Let me check."); + assert_eq!(assistants[0].1.as_deref(), Some("claude-x")); + assert_eq!(assistants[0].2, Some(1)); + assert_eq!(assistants[0].3.as_deref(), Some("req-1")); + assert!(assistants[0].4, "tool-calling step is interim"); + + assert_eq!(assistants[1].0, "It's 72F and sunny in NYC."); + assert_eq!(assistants[1].1.as_deref(), Some("claude-x")); + assert_eq!(assistants[1].2, Some(2)); + assert_eq!(assistants[1].3.as_deref(), Some("req-1")); + assert!(!assistants[1].4, "final answer is not interim"); +} + +#[path = "transcript_view_subagent_tests.rs"] +mod subagent_tests; diff --git a/crates/tinyagents-session/src/transcript/view/types.rs b/crates/tinyagents-session/src/transcript/view/types.rs new file mode 100644 index 00000000..fd1322f7 --- /dev/null +++ b/crates/tinyagents-session/src/transcript/view/types.rs @@ -0,0 +1,189 @@ +//! Typed display items for the transcript projection RPC +//! (`threads.transcript_get`). +//! +//! These mirror the frontend's existing chat vocabulary (user/assistant +//! bubbles, reasoning drawer, tool timeline rows, sub-agent activity) so the +//! Phase C renderer can map them onto the same components. Serde is camelCase +//! on the wire — the frontend reads `displayContent`, `callId`, `requestId`, +//! etc. + +use serde::Serialize; + +/// Terminal state of a projected tool call. Mirrors the live timeline's +/// `ToolTimelineStatus` vocabulary (`running` / `success` / `error`) so the +/// settled projection and the live stream render identically. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum ToolCallStatus { + /// Call issued but no result line has been paired yet. + Running, + /// A result line was paired to the call. + Success, + /// A result line the projection identified as a **failure**: the persisted + /// tool line carried the additive `failure` flag (stamped at turn-loop + /// persistence from the tool's `ToolResult::is_error` outcome). Paired with + /// a [`ToolCallFailure`] payload on the item. + Error, +} + +/// Terminal state of a projected sub-agent run. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum SubagentStatus { + /// The run ended with a final answer. + Completed, + /// The spawning call failed, or reported the run incomplete. + Failed, + /// The run's last record is an interrupted partial answer. + Interrupted, + /// No terminal record yet (still running, or never settled). + Running, +} + +/// Failure payload attached to an errored [`DisplayItem::ToolCall`]. Minimal by +/// design: the persisted transcript only records that the call failed plus an +/// optional short reason. The frontend mapper expands this into its richer +/// `ToolFailureExplanation` shape (`class` / `category` / `causePlain` / +/// `nextAction`) for the `ToolFailureLines` renderer. +#[derive(Debug, Clone, PartialEq, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct ToolCallFailure { + /// Short, single-line reason for the failure, when the writer captured one. + #[serde(skip_serializing_if = "Option::is_none")] + pub detail: Option, +} + +/// One item in a projected transcript, in the frontend's display vocabulary. +/// +/// `#[serde(tag = "kind")]` gives each variant a camelCase discriminator +/// (`userMessage`, `assistantMessage`, …) and every field is camelCase. +#[derive(Debug, Clone, PartialEq, Serialize)] +#[serde( + tag = "kind", + rename_all = "camelCase", + rename_all_fields = "camelCase" +)] +pub enum DisplayItem { + /// A user prompt. `content` is the raw persisted content (may carry the + /// injected `Current Date & Time:` scaffolding line); `displayContent` is + /// the sanitized version to show, present only when it differs from raw. + UserMessage { + content: String, + #[serde(skip_serializing_if = "Option::is_none")] + display_content: Option, + #[serde(skip_serializing_if = "Option::is_none")] + request_id: Option, + /// RFC3339 timestamp of the underlying `DisplayMessage`, when the + /// transcript record carried one. `None` for older records written + /// before timestamps were persisted — never backfilled. + #[serde(skip_serializing_if = "Option::is_none")] + ts: Option, + }, + /// An assistant answer. `interim: true` marks a non-terminal tool-calling + /// step within a multi-iteration turn (not the final answer bubble). + AssistantMessage { + content: String, + #[serde(default, skip_serializing_if = "is_false")] + interim: bool, + #[serde(skip_serializing_if = "Option::is_none")] + request_id: Option, + #[serde(skip_serializing_if = "Option::is_none")] + model: Option, + #[serde(skip_serializing_if = "Option::is_none")] + iteration: Option, + #[serde(skip_serializing_if = "Option::is_none")] + ts: Option, + }, + /// The model's reasoning/thinking that preceded an assistant message. + /// `iteration` is the model call it belongs to — the same value as the + /// message/tool calls that follow it — so a renderer groups it with the + /// step it explains rather than the step before. + Reasoning { + text: String, + #[serde(skip_serializing_if = "Option::is_none")] + iteration: Option, + }, + /// A tool invocation with its paired result, when available. + ToolCall { + call_id: String, + name: String, + /// The model call (1-based, within the turn) that issued this call. + #[serde(skip_serializing_if = "Option::is_none")] + iteration: Option, + #[serde(skip_serializing_if = "Option::is_none")] + args: Option, + #[serde(skip_serializing_if = "Option::is_none")] + result: Option, + status: ToolCallStatus, + /// Present only when `status` is `Error` — the failure payload the + /// frontend expands for the `ToolFailureLines` renderer. + #[serde(skip_serializing_if = "Option::is_none")] + failure: Option, + #[serde(skip_serializing_if = "Option::is_none")] + ts: Option, + }, + /// A delegated sub-agent run, with its own nested projected items. + /// + /// Placed in the item list directly after the tool call that spawned it + /// when that call can be correlated (`call_id`), else at the end of the + /// turn it was spawned in. Sub-agent transcripts are sibling files with no + /// explicit back-link to the delegating tool call, so the turn is derived + /// by matching the sub-agent's spawn timestamp (encoded in its file stem) + /// against the parent turns' commit timestamps, and the call within that + /// turn by the delegation target (see `subagents::attach`). + /// + /// `id` is unique per run: the spawn `task_id` when recorded, else the + /// file-stem suffix — never the agent name, which repeats across runs. + Subagent { + id: String, + /// Sub-agent definition id (e.g. `code_executor`). + #[serde(skip_serializing_if = "Option::is_none")] + agent_id: Option, + /// Spawn task id (`sub-…`), when the transcript recorded one. + #[serde(skip_serializing_if = "Option::is_none")] + task_id: Option, + /// The parent tool call that spawned this run, when correlated. + #[serde(skip_serializing_if = "Option::is_none")] + call_id: Option, + /// Terminal state of the run, derived from its own transcript and the + /// spawning call's result. + status: SubagentStatus, + #[serde(skip_serializing_if = "Option::is_none")] + request_id: Option, + #[serde(skip_serializing_if = "Option::is_none")] + ts: Option, + items: Vec, + }, + /// A turn boundary — emitted when the `request_id` changes between lines. + TurnBoundary { request_id: String }, + /// A partial assistant answer captured when a turn was interrupted. + InterruptedPartial { + text: String, + #[serde(skip_serializing_if = "Option::is_none")] + thinking: Option, + }, + /// A context-compaction marker: the reduced set replaced everything before + /// it. Counts describe what the record superseded/installed. + Compaction { + replaced_count: usize, + kept_count: usize, + #[serde(skip_serializing_if = "Option::is_none")] + ts: Option, + #[serde(skip_serializing_if = "Option::is_none")] + request_id: Option, + }, +} + +#[allow(clippy::trivially_copy_pass_by_ref)] +fn is_false(b: &bool) -> bool { + !*b +} + +/// A projected transcript for one thread, before pagination. Chronological +/// (file) order; the RPC layer paginates newest-first. +#[derive(Debug, Clone)] +pub struct ProjectedTranscript { + pub thread_id: String, + /// All top-level display items in chronological order. + pub items: Vec, +} diff --git a/crates/tinyagents-session/src/turn_state/mod.rs b/crates/tinyagents-session/src/turn_state/mod.rs new file mode 100644 index 00000000..4d25a30a --- /dev/null +++ b/crates/tinyagents-session/src/turn_state/mod.rs @@ -0,0 +1,23 @@ +//! Restart-survivable snapshots of in-flight agent turns. +//! +//! [`store`] is the persistence layer: one JSON file per turn under the +//! workspace, written atomically and serialized through a process-wide mutex; +//! [`store::mark_all_interrupted`] flags any non-terminal snapshot left over +//! from an unclean shutdown at cold boot. [`types`] holds the wire/storage +//! shapes (camelCase, mirroring a chat-runtime UI slice). +//! +//! Filling a snapshot from a host's progress events is host work: the host +//! builds a [`types::TurnState`], mutates it, and calls [`store::put`] at +//! iteration / tool boundaries. + +pub mod store; +pub mod types; + +pub use store::TurnStateStore; +pub use types::{ + SubagentActivity, SubagentToolCall, ToolTimelineEntry, ToolTimelineStatus, TurnLifecycle, + TurnPhase, TurnState, +}; + +#[cfg(test)] +mod shape_test; diff --git a/crates/tinyagents-session/src/turn_state/shape_test.rs b/crates/tinyagents-session/src/turn_state/shape_test.rs new file mode 100644 index 00000000..d72146d9 --- /dev/null +++ b/crates/tinyagents-session/src/turn_state/shape_test.rs @@ -0,0 +1,38 @@ +//! Persisted snapshot shape is a wire/storage contract: pin it as literal JSON. + +use serde_json::json; + +use super::types::{TurnLifecycle, TurnState}; + +#[test] +fn started_snapshot_serializes_to_the_documented_camel_case_shape() { + let mut state = TurnState::started("t1", "r1", 25, "2026-05-04T10:00:00Z"); + state.lifecycle = TurnLifecycle::Streaming; + assert_eq!( + serde_json::to_value(&state).unwrap(), + json!({ + "threadId": "t1", + "requestId": "r1", + "lifecycle": "streaming", + "iteration": 0, + "maxIterations": 25, + "streamingText": "", + "thinking": "", + "toolTimeline": [], + "startedAt": "2026-05-04T10:00:00Z", + "updatedAt": "2026-05-04T10:00:00Z" + }) + ); +} + +#[test] +fn snapshot_written_before_optional_fields_still_loads() { + let raw = json!({ + "threadId": "t1", "requestId": "r1", "lifecycle": "interrupted", + "iteration": 2, "maxIterations": 10, "streamingText": "x", "thinking": "", + "toolTimeline": [], "startedAt": "a", "updatedAt": "b" + }); + let state: TurnState = serde_json::from_value(raw).unwrap(); + assert_eq!(state.lifecycle, TurnLifecycle::Interrupted); + assert!(state.transcript.is_empty()); +} diff --git a/crates/tinyagents-session/src/turn_state/store.rs b/crates/tinyagents-session/src/turn_state/store.rs new file mode 100644 index 00000000..69bdee81 --- /dev/null +++ b/crates/tinyagents-session/src/turn_state/store.rs @@ -0,0 +1,727 @@ +//! Filesystem-backed snapshot store for [`super::types::TurnState`]. +//! +//! **Per-turn ring layout.** One JSON file per *turn* under a per-thread +//! directory: +//! `/memory/conversations/turn_states//.json`. +//! Each turn keeps its own snapshot so a multi-turn thread retains every turn's +//! tool timeline (the "Agentic task insights" trail), not just the latest. +//! Completed turns are pruned to the newest [`COMPLETED_RETENTION`] per thread so +//! history stays bounded. +//! +//! The `get(thread_id)` / `list()` / `delete(thread_id)` / `clear_all` / +//! `mark_all_interrupted` surface is unchanged so existing callers (RPC layer, +//! mirror, cold-boot) keep working: `get`/`list` resolve the *latest* turn per +//! thread. New `get_turn(thread_id, request_id)` and `list_thread(thread_id)` +//! expose the per-turn history. +//! +//! **Legacy migration.** Snapshots written by older cores live as flat files +//! `turn_states/.json`. They are migrated in place — read once, +//! rewritten under `/.json`, and the flat file +//! removed — on first access. Migration is idempotent. +//! +//! Mutations are serialised through a single process-wide mutex so the progress +//! consumer cannot interleave a flush against an RPC handler reading the same +//! file. + +use std::fs::{self, File}; +use std::io::{Read, Write}; +use std::path::{Path, PathBuf}; + +use parking_lot::Mutex; +use std::sync::LazyLock; +use tempfile::NamedTempFile; +use tracing::{debug, warn}; + +use super::types::{TurnLifecycle, TurnState}; + +const LOG_PREFIX: &str = "[threads:turn_state]"; +const TURN_STATE_DIR: &str = "turn_states"; +const SNAPSHOT_EXTENSION: &str = "json"; +/// Newest completed turns retained per thread. Older completed turns are pruned +/// on the next completed write so a long-lived thread's history stays bounded +/// (mirrors the timeline registry's soft-cap philosophy — never unbounded). +const COMPLETED_RETENTION: usize = 20; +static TURN_STATE_LOCK: LazyLock> = LazyLock::new(|| Mutex::new(())); + +fn compare_rfc3339(left: &str, right: &str) -> std::cmp::Ordering { + match ( + chrono::DateTime::parse_from_rfc3339(left), + chrono::DateTime::parse_from_rfc3339(right), + ) { + (Ok(left), Ok(right)) => left.cmp(&right), + _ => left.cmp(right), + } +} + +/// Workspace-rooted handle that reads and writes per-thread turn snapshots. +#[derive(Debug, Clone)] +pub struct TurnStateStore { + workspace_dir: PathBuf, +} + +impl TurnStateStore { + pub fn new(workspace_dir: PathBuf) -> Self { + Self { workspace_dir } + } + + /// Workspace root this store persists under. Exposed so the mirror can + /// resolve sibling session transcripts (append the interrupted partial to + /// `session_raw/{root}.jsonl`) without re-plumbing the path. + pub fn workspace_dir(&self) -> &std::path::Path { + &self.workspace_dir + } + + /// Atomically write the snapshot for `state.request_id` under + /// `state.thread_id`. On a `Completed` write, prune the thread's completed + /// turns to the newest [`COMPLETED_RETENTION`]. + pub fn put(&self, state: &TurnState) -> Result<(), String> { + let _guard = TURN_STATE_LOCK.lock(); + // Fold any pre-existing flat file for this thread into the per-turn + // layout first so the directory is the single source of truth. + self.migrate_thread_locked(&state.thread_id); + let dir = self.ensure_thread_dir(&state.thread_id)?; + let path = self.turn_path(&state.thread_id, &state.request_id); + let mut tmp = NamedTempFile::new_in(&dir) + .map_err(|e| format!("create turn-state tempfile in {}: {e}", dir.display()))?; + let bytes = + serde_json::to_vec_pretty(state).map_err(|e| format!("serialize turn state: {e}"))?; + tmp.write_all(&bytes) + .map_err(|e| format!("write turn-state tempfile: {e}"))?; + tmp.as_file() + .sync_all() + .map_err(|e| format!("fsync turn-state tempfile: {e}"))?; + persist_temp_file(tmp, &path)?; + // Sync the directory entry created by the rename — without this a crash + // or power loss between persist() and the next fs flush can drop the + // snapshot, defeating the cold-boot recovery guarantee. Best-effort on + // platforms where opening a directory for sync is not supported. + if let Err(err) = sync_dir(&dir) { + tracing::warn!("{LOG_PREFIX} failed to fsync {}: {err}", dir.display()); + } + debug!( + "{LOG_PREFIX} wrote snapshot thread={} request={} lifecycle={:?} iter={}/{} timeline={}", + state.thread_id, + state.request_id, + state.lifecycle, + state.iteration, + state.max_iterations, + state.tool_timeline.len() + ); + if state.lifecycle == TurnLifecycle::Completed { + self.prune_completed_locked(&state.thread_id); + } + Ok(()) + } + + /// Return the latest turn for `thread_id`, or `None` if none exists. + /// "Latest" is the turn with the greatest `started_at` (ties broken by + /// `updated_at`) — the in-flight or most-recent turn. + pub fn get(&self, thread_id: &str) -> Result, String> { + let _guard = TURN_STATE_LOCK.lock(); + self.migrate_thread_locked(thread_id); + Ok(latest_turn(self.read_thread_turns(thread_id)?)) + } + + /// Return a specific turn by `request_id`, or `None` if absent. + pub fn get_turn(&self, thread_id: &str, request_id: &str) -> Result, String> { + let _guard = TURN_STATE_LOCK.lock(); + self.migrate_thread_locked(thread_id); + let path = self.turn_path(thread_id, request_id); + if !path.exists() { + return Ok(None); + } + read_snapshot(&path).map(Some) + } + + /// Delete every turn for `thread_id` (and any legacy flat file). Returns + /// `true` if anything was removed. + pub fn delete(&self, thread_id: &str) -> Result { + let _guard = TURN_STATE_LOCK.lock(); + let mut removed = false; + let flat = self.legacy_flat_path(thread_id); + if flat.exists() { + fs::remove_file(&flat) + .map_err(|e| format!("remove legacy turn-state {}: {e}", flat.display()))?; + removed = true; + } + let dir = self.thread_dir(thread_id); + if dir.exists() { + fs::remove_dir_all(&dir) + .map_err(|e| format!("remove turn-state dir {}: {e}", dir.display()))?; + removed = true; + } + if removed { + debug!("{LOG_PREFIX} deleted snapshots thread={}", thread_id); + } + Ok(removed) + } + + /// Delete one turn's snapshot by `request_id`, leaving every other turn on + /// the thread untouched. Returns `true` if a file was removed. + /// + /// Backs edit/regenerate (`threads.edit_message` / `threads.regenerate`): + /// truncating the message log after a cut point orphans the turn + /// snapshots for every dropped request — `delete(thread_id)` would also + /// discard the turns kept *before* the cut, which a client's "Agentic + /// task insights" trail for an earlier answer still needs. + pub fn delete_turn(&self, thread_id: &str, request_id: &str) -> Result { + let _guard = TURN_STATE_LOCK.lock(); + self.migrate_thread_locked(thread_id); + let path = self.turn_path(thread_id, request_id); + if !path.exists() { + return Ok(false); + } + fs::remove_file(&path).map_err(|e| format!("remove turn-state {}: {e}", path.display()))?; + debug!("{LOG_PREFIX} deleted snapshot thread={thread_id} request={request_id}"); + Ok(true) + } + + /// List the latest turn for every thread. Used by the UI on cold boot to + /// surface interrupted turns from a previous process (one entry per thread, + /// preserving the pre-ring-store contract). + pub fn list(&self) -> Result, String> { + let _guard = TURN_STATE_LOCK.lock(); + self.migrate_all_legacy_locked(); + let dir = self.dir(); + if !dir.exists() { + return Ok(Vec::new()); + } + let mut snapshots = Vec::new(); + for thread_id in self.thread_ids()? { + if let Some(latest) = latest_turn(self.read_thread_turns(&thread_id)?) { + snapshots.push(latest); + } + } + Ok(snapshots) + } + + /// List every turn for one thread, newest first (by `started_at`). + pub fn list_thread(&self, thread_id: &str) -> Result, String> { + let _guard = TURN_STATE_LOCK.lock(); + self.migrate_thread_locked(thread_id); + let mut turns = self.read_thread_turns(thread_id)?; + turns.sort_by(|a, b| { + compare_rfc3339(&b.started_at, &a.started_at) + .then_with(|| compare_rfc3339(&b.updated_at, &a.updated_at)) + }); + Ok(turns) + } + + /// Remove every snapshot file, readable or not (per-turn files, thread + /// directories, and any legacy flat files). Used by `threads_purge` to + /// guarantee a destructive cleanup leaves nothing — `list()` only returns + /// parseable snapshots, so iterating list+delete would silently leave + /// half-written or schema-skewed files behind. Returns the count of JSON + /// files removed. + pub fn clear_all(&self) -> Result { + let _guard = TURN_STATE_LOCK.lock(); + let dir = self.dir(); + if !dir.exists() { + return Ok(0); + } + let mut removed = 0usize; + for entry in + fs::read_dir(&dir).map_err(|e| format!("read turn-state dir {}: {e}", dir.display()))? + { + let entry = entry.map_err(|e| format!("read turn-state entry: {e}"))?; + let path = entry.path(); + let file_type = entry + .file_type() + .map_err(|e| format!("stat turn-state entry {}: {e}", path.display()))?; + if file_type.is_dir() { + // A per-thread directory: count and remove its JSON files, then + // drop the (now empty) directory. + for sub in fs::read_dir(&path) + .map_err(|e| format!("read thread dir {}: {e}", path.display()))? + { + let sub = sub.map_err(|e| format!("read thread entry: {e}"))?; + let sub_path = sub.path(); + if sub_path.extension().and_then(|s| s.to_str()) == Some(SNAPSHOT_EXTENSION) { + fs::remove_file(&sub_path).map_err(|e| { + format!("remove turn-state file {}: {e}", sub_path.display()) + })?; + removed += 1; + } + } + fs::remove_dir_all(&path) + .map_err(|e| format!("remove thread dir {}: {e}", path.display()))?; + } else if path.extension().and_then(|s| s.to_str()) == Some(SNAPSHOT_EXTENSION) { + // A legacy flat snapshot. + fs::remove_file(&path) + .map_err(|e| format!("remove turn-state file {}: {e}", path.display()))?; + removed += 1; + } + } + if removed > 0 { + debug!( + "{LOG_PREFIX} cleared {removed} snapshots from {}", + dir.display() + ); + } + Ok(removed) + } + + /// Mark every non-terminal turn as `Interrupted`. Intended to run on startup + /// so the UI can distinguish stale turns left behind by a previous process + /// from turns currently being driven. `Completed`/`Interrupted` turns are + /// left as-is (idempotent; completed turns are intentionally kept so the + /// processing panel can replay a finished turn after a reboot). + pub fn mark_all_interrupted(&self, now_rfc3339: &str) -> Result { + let turns = { + let _guard = TURN_STATE_LOCK.lock(); + self.migrate_all_legacy_locked(); + self.all_turns_locked()? + }; + let mut count = 0usize; + for mut snapshot in turns { + if matches!( + snapshot.lifecycle, + TurnLifecycle::Interrupted | TurnLifecycle::Completed + ) { + continue; + } + snapshot.lifecycle = TurnLifecycle::Interrupted; + snapshot.updated_at = now_rfc3339.to_string(); + snapshot.active_tool = None; + snapshot.active_subagent = None; + self.put(&snapshot)?; + count += 1; + } + if count > 0 { + debug!("{LOG_PREFIX} marked {count} snapshots as interrupted on startup"); + } + Ok(count) + } + + /// Force one turn's snapshot to a terminal `lifecycle` if it is still + /// `Started`/`Streaming`. Returns `true` when it changed something; a + /// missing or already-terminal snapshot is a no-op. + /// + /// The progress bridge is normally the only writer that marks a snapshot + /// terminal, and it does so on its way out — but it only exits once its + /// progress sender drops, and for a cached per-thread session that does not + /// happen until the *next* turn replaces the sink. The last turn of a thread + /// would otherwise keep a non-terminal snapshot on disk indefinitely + /// (`prune_completed_locked` only prunes `Completed` turns, and the startup + /// sweep runs once per process), so re-entering the thread hydrates a + /// live-looking "Thinking…" indicator under a reply that already landed. + /// The turn driver calls this the moment the turn ends, which is the + /// earliest point that is known for certain. + pub fn settle_turn( + &self, + thread_id: &str, + request_id: &str, + lifecycle: TurnLifecycle, + now_rfc3339: &str, + ) -> Result { + let Some(mut snapshot) = self.get_turn(thread_id, request_id)? else { + return Ok(false); + }; + if matches!( + snapshot.lifecycle, + TurnLifecycle::Interrupted | TurnLifecycle::Completed + ) { + return Ok(false); + } + snapshot.lifecycle = lifecycle; + snapshot.phase = None; + snapshot.active_tool = None; + snapshot.active_subagent = None; + snapshot.updated_at = now_rfc3339.to_string(); + self.put(&snapshot)?; + debug!( + "{LOG_PREFIX} settled non-terminal snapshot thread={thread_id} request={request_id} lifecycle={lifecycle:?}" + ); + Ok(true) + } + + // --- internals ------------------------------------------------------- + + fn ensure_thread_dir(&self, thread_id: &str) -> Result { + let dir = self.thread_dir(thread_id); + fs::create_dir_all(&dir) + .map_err(|e| format!("create thread turn-state dir {}: {e}", dir.display()))?; + Ok(dir) + } + + fn dir(&self) -> PathBuf { + self.workspace_dir + .join("memory") + .join("conversations") + .join(TURN_STATE_DIR) + } + + fn thread_dir(&self, thread_id: &str) -> PathBuf { + self.dir().join(hex::encode(thread_id.as_bytes())) + } + + fn turn_path(&self, thread_id: &str, request_id: &str) -> PathBuf { + self.thread_dir(thread_id).join(format!( + "{}.{}", + hex::encode(request_id.as_bytes()), + SNAPSHOT_EXTENSION + )) + } + + fn legacy_flat_path(&self, thread_id: &str) -> PathBuf { + self.dir().join(format!( + "{}.{}", + hex::encode(thread_id.as_bytes()), + SNAPSHOT_EXTENSION + )) + } + + /// Read every parseable turn snapshot in one thread's directory. + /// Unreadable files are logged and skipped (mirrors `list()`'s resilience). + fn read_thread_turns(&self, thread_id: &str) -> Result, String> { + let dir = self.thread_dir(thread_id); + if !dir.exists() { + return Ok(Vec::new()); + } + let mut turns = Vec::new(); + for entry in fs::read_dir(&dir) + .map_err(|e| format!("read thread turn-state dir {}: {e}", dir.display()))? + { + let entry = entry.map_err(|e| format!("read thread turn-state entry: {e}"))?; + let path = entry.path(); + if path.extension().and_then(|s| s.to_str()) != Some(SNAPSHOT_EXTENSION) { + continue; + } + match read_snapshot(&path) { + Ok(snapshot) => turns.push(snapshot), + Err(err) => warn!( + "{LOG_PREFIX} skip unreadable snapshot {}: {err}", + path.display() + ), + } + } + Ok(turns) + } + + /// hex(thread_id) directory names under the root, decoded back to the + /// thread-id string. Skips legacy flat files (handled by migration). + fn thread_ids(&self) -> Result, String> { + let dir = self.dir(); + if !dir.exists() { + return Ok(Vec::new()); + } + let mut ids = Vec::new(); + for entry in + fs::read_dir(&dir).map_err(|e| format!("read turn-state dir {}: {e}", dir.display()))? + { + let entry = entry.map_err(|e| format!("read turn-state entry: {e}"))?; + if !entry + .file_type() + .map_err(|e| format!("stat turn-state entry: {e}"))? + .is_dir() + { + continue; + } + let name = entry.file_name(); + let Some(name) = name.to_str() else { continue }; + match hex::decode(name) + .ok() + .and_then(|b| String::from_utf8(b).ok()) + { + Some(thread_id) => ids.push(thread_id), + None => warn!("{LOG_PREFIX} skip non-hex thread dir {name}"), + } + } + Ok(ids) + } + + /// Every turn across every thread. Caller holds the lock. + fn all_turns_locked(&self) -> Result, String> { + let mut turns = Vec::new(); + for thread_id in self.thread_ids()? { + turns.append(&mut self.read_thread_turns(&thread_id)?); + } + Ok(turns) + } + + /// If a legacy flat file exists for `thread_id`, fold it into the per-turn + /// layout and remove the flat file. Best-effort; failures are logged. Caller + /// holds the lock. + fn migrate_thread_locked(&self, thread_id: &str) { + let flat = self.legacy_flat_path(thread_id); + if !flat.exists() { + return; + } + match read_snapshot(&flat) { + Ok(state) => { + if let Err(err) = self.write_turn_file(&state) { + warn!( + "{LOG_PREFIX} legacy migrate write failed thread={thread_id}: {err} (flat file kept)" + ); + return; + } + if let Err(err) = fs::remove_file(&flat) { + warn!( + "{LOG_PREFIX} legacy migrate: removed-into-dir but flat delete failed {}: {err}", + flat.display() + ); + } else { + debug!( + "{LOG_PREFIX} migrated legacy snapshot thread={thread_id} request={}", + state.request_id + ); + } + } + Err(err) => warn!( + "{LOG_PREFIX} legacy migrate: unreadable flat file {} left in place: {err}", + flat.display() + ), + } + } + + /// Migrate every legacy flat file under the root. Caller holds the lock. + fn migrate_all_legacy_locked(&self) { + let dir = self.dir(); + if !dir.exists() { + return; + } + let entries = match fs::read_dir(&dir) { + Ok(entries) => entries, + Err(err) => { + warn!("{LOG_PREFIX} migrate scan failed {}: {err}", dir.display()); + return; + } + }; + for entry in entries.flatten() { + let path = entry.path(); + if path.extension().and_then(|s| s.to_str()) != Some(SNAPSHOT_EXTENSION) { + continue; // directories and non-json files + } + match read_snapshot(&path) { + Ok(state) => { + if self.write_turn_file(&state).is_ok() { + let _ = fs::remove_file(&path); + debug!( + "{LOG_PREFIX} migrated legacy snapshot thread={} request={}", + state.thread_id, state.request_id + ); + } + } + Err(err) => warn!( + "{LOG_PREFIX} migrate: unreadable flat file {} left in place: {err}", + path.display() + ), + } + } + } + + /// Atomic per-turn write without migration/retention side effects. Used by + /// migration to relocate a snapshot. Caller holds the lock. + fn write_turn_file(&self, state: &TurnState) -> Result<(), String> { + let dir = self.ensure_thread_dir(&state.thread_id)?; + let path = self.turn_path(&state.thread_id, &state.request_id); + let mut tmp = NamedTempFile::new_in(&dir) + .map_err(|e| format!("create turn-state tempfile in {}: {e}", dir.display()))?; + let bytes = + serde_json::to_vec_pretty(state).map_err(|e| format!("serialize turn state: {e}"))?; + tmp.write_all(&bytes) + .map_err(|e| format!("write turn-state tempfile: {e}"))?; + tmp.as_file() + .sync_all() + .map_err(|e| format!("fsync turn-state tempfile: {e}"))?; + persist_temp_file(tmp, &path)?; + if let Err(err) = sync_dir(&dir) { + tracing::warn!("{LOG_PREFIX} failed to fsync {}: {err}", dir.display()); + } + Ok(()) + } + + /// Prune a thread's `Completed` turns to the newest [`COMPLETED_RETENTION`] + /// by `updated_at`. Non-completed turns (at most the one live turn) are kept. + /// Caller holds the lock. Best-effort — failures are logged, not fatal. + fn prune_completed_locked(&self, thread_id: &str) { + let turns = match self.read_thread_turns(thread_id) { + Ok(turns) => turns, + Err(err) => { + warn!("{LOG_PREFIX} prune read failed thread={thread_id}: {err}"); + return; + } + }; + let mut completed: Vec = turns + .into_iter() + .filter(|t| t.lifecycle == TurnLifecycle::Completed) + .collect(); + if completed.len() <= COMPLETED_RETENTION { + return; + } + // Newest first, then drop everything past the retention window. + completed.sort_by(|a, b| compare_rfc3339(&b.updated_at, &a.updated_at)); + for stale in completed.into_iter().skip(COMPLETED_RETENTION) { + let path = self.turn_path(&stale.thread_id, &stale.request_id); + if let Err(err) = fs::remove_file(&path) { + warn!("{LOG_PREFIX} prune remove failed {}: {err}", path.display()); + } else { + debug!( + "{LOG_PREFIX} pruned completed turn thread={thread_id} request={}", + stale.request_id + ); + } + } + } +} + +#[cfg(windows)] +fn persist_temp_file(tmp: NamedTempFile, path: &Path) -> Result<(), String> { + use std::os::windows::ffi::OsStrExt; + use windows_sys::Win32::Storage::FileSystem::{MOVEFILE_REPLACE_EXISTING, MoveFileExW}; + + let wide_path = |path: &Path| -> Result, String> { + let filename = path + .file_name() + .ok_or_else(|| format!("resolve turn-state filename {}", path.display()))?; + let parent = path + .parent() + .ok_or_else(|| format!("resolve turn-state parent {}", path.display()))? + // Canonicalizing the existing parent gives Windows a verbatim + // long-path form before appending the not-yet-existing filename. + .canonicalize() + .map_err(|e| format!("resolve turn-state parent {}: {e}", path.display()))?; + let resolved = parent.join(filename); + let raw: Vec = resolved.as_os_str().encode_wide().collect(); + let mut extended = + if raw.starts_with(&[b'\\' as u16, b'\\' as u16, b'?' as u16, b'\\' as u16]) { + raw + } else if raw.starts_with(&[b'\\' as u16, b'\\' as u16]) { + // Convert a UNC path from `\\server\share` to + // `\\?\UNC\server\share`. + let mut prefixed: Vec = r"\\?\UNC\".encode_utf16().collect(); + prefixed.extend_from_slice(&raw[2..]); + prefixed + } else { + let mut prefixed: Vec = r"\\?\".encode_utf16().collect(); + prefixed.extend_from_slice(&raw); + prefixed + }; + extended.push(0); + Ok(extended) + }; + // Compute both paths while `tmp` still owns its file, so any preparation + // error lets NamedTempFile clean the tempfile up automatically. + let source = wide_path(tmp.path())?; + let destination = wide_path(path)?; + let (file, temp_path) = tmp + .keep() + .map_err(|e| format!("persist turn-state file {}: {e}", path.display()))?; + // `keep` transfers ownership to us, so close the handle before replacing + // the destination and explicitly clean it up if the replacement fails. + drop(file); + // SAFETY: both buffers are NUL-terminated and remain alive for the call. + // The paths share a directory, so this is an atomic replacement rather + // than a cross-volume copy-and-delete move. + let moved = unsafe { + MoveFileExW( + source.as_ptr(), + destination.as_ptr(), + MOVEFILE_REPLACE_EXISTING, + ) + }; + if moved != 0 { + return Ok(()); + } + + let error = std::io::Error::last_os_error(); + Err({ + if let Err(cleanup_err) = fs::remove_file(&temp_path) { + warn!( + "{LOG_PREFIX} failed to remove turn-state tempfile {} after rename failure: {cleanup_err}", + temp_path.display() + ); + } + format!("persist turn-state file {}: {error}", path.display()) + }) +} + +#[cfg(not(windows))] +fn persist_temp_file(tmp: NamedTempFile, path: &Path) -> Result<(), String> { + tmp.persist(path) + .map(|_| ()) + .map_err(|e| format!("persist turn-state file {}: {e}", path.display())) +} + +/// Pick the latest turn (greatest `started_at`, ties broken by `updated_at`). +fn latest_turn(turns: Vec) -> Option { + turns.into_iter().max_by(|a, b| { + a.started_at + .cmp(&b.started_at) + .then_with(|| a.updated_at.cmp(&b.updated_at)) + }) +} + +/// Best-effort `fsync` of a directory entry. On Unix, opens the directory for +/// read and calls `sync_all` on the file handle. On Windows this is a no-op — +/// directory fsync is not exposed by the platform and the rename's durability is +/// provided by NTFS journaling. +#[cfg(unix)] +fn sync_dir(dir: &Path) -> std::io::Result<()> { + File::open(dir)?.sync_all() +} + +#[cfg(not(unix))] +fn sync_dir(_dir: &Path) -> std::io::Result<()> { + Ok(()) +} + +fn read_snapshot(path: &Path) -> Result { + let mut file = + File::open(path).map_err(|e| format!("open turn-state {}: {e}", path.display()))?; + let mut buf = String::new(); + file.read_to_string(&mut buf) + .map_err(|e| format!("read turn-state {}: {e}", path.display()))?; + serde_json::from_str(&buf).map_err(|e| format!("parse turn-state {}: {e}", path.display())) +} + +// Free-function wrappers mirroring `memory::conversations::store` so callers +// at the RPC layer don't have to instantiate `TurnStateStore` themselves. + +pub fn put(workspace_dir: PathBuf, state: &TurnState) -> Result<(), String> { + TurnStateStore::new(workspace_dir).put(state) +} + +pub fn get(workspace_dir: PathBuf, thread_id: &str) -> Result, String> { + TurnStateStore::new(workspace_dir).get(thread_id) +} + +pub fn get_turn( + workspace_dir: PathBuf, + thread_id: &str, + request_id: &str, +) -> Result, String> { + TurnStateStore::new(workspace_dir).get_turn(thread_id, request_id) +} + +pub fn delete(workspace_dir: PathBuf, thread_id: &str) -> Result { + TurnStateStore::new(workspace_dir).delete(thread_id) +} + +pub fn delete_turn( + workspace_dir: PathBuf, + thread_id: &str, + request_id: &str, +) -> Result { + TurnStateStore::new(workspace_dir).delete_turn(thread_id, request_id) +} + +pub fn list(workspace_dir: PathBuf) -> Result, String> { + TurnStateStore::new(workspace_dir).list() +} + +pub fn list_thread(workspace_dir: PathBuf, thread_id: &str) -> Result, String> { + TurnStateStore::new(workspace_dir).list_thread(thread_id) +} + +pub fn clear_all(workspace_dir: PathBuf) -> Result { + TurnStateStore::new(workspace_dir).clear_all() +} + +pub fn mark_all_interrupted(workspace_dir: PathBuf, now_rfc3339: &str) -> Result { + TurnStateStore::new(workspace_dir).mark_all_interrupted(now_rfc3339) +} + +#[cfg(test)] +#[path = "store_test.rs"] +mod tests; diff --git a/crates/tinyagents-session/src/turn_state/store_test.rs b/crates/tinyagents-session/src/turn_state/store_test.rs new file mode 100644 index 00000000..c966c6a8 --- /dev/null +++ b/crates/tinyagents-session/src/turn_state/store_test.rs @@ -0,0 +1,544 @@ +//! Unit tests for [`super::TurnStateStore`]. + +use super::*; +use crate::turn_state::types::{ + SubagentActivity, SubagentToolCall, SubagentTranscriptItem, ToolTimelineEntry, + ToolTimelineStatus, TurnLifecycle, TurnPhase, TurnState, +}; +use tempfile::tempdir; + +#[cfg(windows)] +use std::os::windows::ffi::OsStrExt; + +fn sample_state(thread_id: &str) -> TurnState { + TurnState::started(thread_id.to_string(), "req-1", 25, "2026-05-04T10:00:00Z") +} + +/// A turn with an explicit request id + started/updated timestamps. +fn turn(thread_id: &str, request_id: &str, started_at: &str) -> TurnState { + let mut s = TurnState::started(thread_id.to_string(), request_id, 25, started_at); + s.updated_at = started_at.to_string(); + s +} + +fn turn_states_root(dir: &tempfile::TempDir) -> std::path::PathBuf { + dir.path() + .join("memory") + .join("conversations") + .join("turn_states") +} + +#[test] +fn put_then_get_roundtrips_state() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let mut state = sample_state("thread-abc"); + state.lifecycle = TurnLifecycle::Streaming; + state.iteration = 3; + state.streaming_text = "hello".into(); + state.tool_timeline.push(ToolTimelineEntry { + id: "tc-1".into(), + name: "shell".into(), + round: 1, + status: ToolTimelineStatus::Running, + args_buffer: Some("{".into()), + display_name: None, + detail: None, + source_tool_name: None, + subagent: None, + failure: None, + output: None, + seq: Some(0), + }); + + store.put(&state).expect("put"); + let loaded = store.get("thread-abc").expect("get").expect("present"); + assert_eq!(loaded, state); +} + +#[cfg(windows)] +#[test] +fn put_roundtrips_snapshot_with_a_path_longer_than_max_path() { + let dir = tempdir().expect("tempdir"); + let mut workspace = dir.path().to_path_buf(); + let thread_id = "t".repeat(43); + let request_id = "r".repeat(36); + while workspace + .join("memory") + .join("conversations") + .join("turn_states") + .join(hex::encode(&thread_id)) + .join(format!("{}.json", hex::encode(&request_id))) + .as_os_str() + .encode_wide() + .count() + <= 260 + { + workspace.push("long_workspace_segment"); + } + let mut state = turn(&thread_id, &request_id, "2026-05-04T10:00:00Z"); + let store = TurnStateStore::new(workspace); + + store.put(&state).expect("persist a snapshot past MAX_PATH"); + assert_eq!(store.get(&thread_id).expect("get"), Some(state.clone())); + + // A regular progress flush replaces the same snapshot repeatedly. + state.iteration = 2; + store.put(&state).expect("replace snapshot past MAX_PATH"); + assert_eq!(store.get(&thread_id).expect("get replacement"), Some(state)); +} + +#[test] +fn roundtrips_subagent_interleaved_transcript_with_full_fidelity() { + // A settled turn whose subagent streamed reasoning, called a tool, then + // narrated — the interleaved transcript (not just the flat tool rows) plus + // the per-row `seq` ordering keys must survive a disk round-trip verbatim, + // so a reopened transcript rehydrates without losing the subagent's + // reasoning (the gap SubagentDrawer/chatRuntimeSlice documented). + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let mut state = sample_state("thread-sub"); + state.lifecycle = TurnLifecycle::Completed; + + let activity = SubagentActivity { + task_id: "sub-1".into(), + agent_id: "researcher".into(), + status: Some("completed".into()), + mode: Some("typed".into()), + dedicated_thread: Some(false), + child_iteration: Some(1), + child_max_iterations: Some(8), + iterations: Some(2), + elapsed_ms: Some(1234), + output_chars: Some(42), + worker_thread_id: Some("worker-thread-9".into()), + parent_call_id: Some("call-1".into()), + output: Some("done".into()), + tool_calls: vec![SubagentToolCall { + call_id: "c1".into(), + tool_name: "search".into(), + status: ToolTimelineStatus::Success, + iteration: Some(1), + elapsed_ms: Some(12), + output_chars: Some(6), + display_name: Some("Searching".into()), + detail: None, + args: Some(serde_json::json!({ "query": "rust serde" })), + failure: None, + output: Some("3 hits".into()), + }], + transcript: vec![ + SubagentTranscriptItem::Thinking { + iteration: Some(1), + text: "let me search.".into(), + }, + SubagentTranscriptItem::Tool { + iteration: Some(1), + call_id: "c1".into(), + tool_name: "search".into(), + status: ToolTimelineStatus::Success, + elapsed_ms: Some(12), + output_chars: Some(6), + display_name: Some("Searching".into()), + detail: None, + }, + SubagentTranscriptItem::Text { + iteration: Some(1), + text: "Found it.".into(), + }, + ], + }; + + state.tool_timeline.push(ToolTimelineEntry { + id: "subagent:sub-1".into(), + name: "subagent:researcher".into(), + round: 1, + status: ToolTimelineStatus::Success, + args_buffer: None, + display_name: Some("Researcher".into()), + detail: None, + source_tool_name: Some("spawn_subagent".into()), + subagent: Some(activity), + failure: None, + output: None, + seq: Some(3), + }); + + store.put(&state).expect("put"); + let loaded = store.get("thread-sub").expect("get").expect("present"); + // Structural equality proves nothing in the interleaved transcript, + // subagent activity, or the `seq` ordering keys was dropped or reordered. + assert_eq!(loaded, state); + + // Spot-check the interleaving explicitly so a future regression that keeps + // the fields but loses the ordering still fails here. + let activity = loaded.tool_timeline[0] + .subagent + .as_ref() + .expect("subagent activity restored"); + assert_eq!(activity.transcript.len(), 3); + assert!(matches!( + activity.transcript[0], + SubagentTranscriptItem::Thinking { .. } + )); + assert!(matches!( + activity.transcript[1], + SubagentTranscriptItem::Tool { .. } + )); + assert!(matches!( + activity.transcript[2], + SubagentTranscriptItem::Text { .. } + )); + // The child call's arguments are what the reloaded row's "Input" block + // renders from. Spot-checked by value (like the interleaving above) so a + // regression names the field instead of failing as one opaque struct + // inequality (#5987). + assert_eq!( + activity.tool_calls[0].args, + Some(serde_json::json!({ "query": "rust serde" })), + "child tool arguments must survive the disk round-trip" + ); + assert_eq!(loaded.tool_timeline[0].seq, Some(3)); +} + +#[test] +fn get_returns_none_when_absent() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + assert!(store.get("missing").expect("get").is_none()); +} + +#[test] +fn delete_removes_snapshot_and_reports_presence() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let state = sample_state("thread-x"); + store.put(&state).expect("put"); + assert!(store.delete("thread-x").expect("delete")); + assert!(!store.delete("thread-x").expect("delete-again")); + assert!(store.get("thread-x").expect("get").is_none()); +} + +#[test] +fn list_returns_every_snapshot() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + store.put(&sample_state("a")).expect("put a"); + store.put(&sample_state("b")).expect("put b"); + let mut ids: Vec = store + .list() + .expect("list") + .into_iter() + .map(|s| s.thread_id) + .collect(); + ids.sort(); + assert_eq!(ids, vec!["a".to_string(), "b".to_string()]); +} + +#[test] +fn list_on_missing_dir_is_empty() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + assert!(store.list().expect("list").is_empty()); +} + +#[test] +fn mark_all_interrupted_promotes_lifecycle_and_clears_active_fields() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let mut state = sample_state("t"); + state.lifecycle = TurnLifecycle::Streaming; + state.active_tool = Some("shell".into()); + state.active_subagent = Some("researcher".into()); + store.put(&state).expect("put"); + + let count = store + .mark_all_interrupted("2026-05-04T10:01:00Z") + .expect("mark"); + assert_eq!(count, 1); + + let loaded = store.get("t").expect("get").expect("present"); + assert_eq!(loaded.lifecycle, TurnLifecycle::Interrupted); + assert_eq!(loaded.updated_at, "2026-05-04T10:01:00Z"); + assert!(loaded.active_tool.is_none()); + assert!(loaded.active_subagent.is_none()); + + // Re-running is a no-op for already-interrupted snapshots. + let count = store + .mark_all_interrupted("2026-05-04T10:02:00Z") + .expect("mark again"); + assert_eq!(count, 0); +} + +#[test] +fn mark_all_interrupted_leaves_completed_snapshots_untouched() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let mut state = sample_state("t"); + // A finished turn is kept as `Completed` so its processing can be replayed; + // startup interrupted-marking must not flip it to `Interrupted`. + state.lifecycle = TurnLifecycle::Completed; + store.put(&state).expect("put"); + + let count = store + .mark_all_interrupted("2026-05-04T10:01:00Z") + .expect("mark"); + assert_eq!(count, 0); + let loaded = store.get("t").expect("get").expect("present"); + assert_eq!(loaded.lifecycle, TurnLifecycle::Completed); +} + +#[test] +fn clear_all_removes_corrupted_snapshots_too() { + use std::io::Write as _; + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + store.put(&sample_state("a")).expect("put a"); + store.put(&sample_state("b")).expect("put b"); + + // Drop a corrupted JSON file alongside — `list()` would skip it, + // but a destructive purge must still remove it. + let corrupt_path = dir + .path() + .join("memory") + .join("conversations") + .join("turn_states") + .join("deadbeef.json"); + let mut f = std::fs::File::create(&corrupt_path).expect("create corrupt"); + f.write_all(b"{ not valid json").expect("write corrupt"); + drop(f); + assert!(corrupt_path.exists()); + + let removed = store.clear_all().expect("clear_all"); + assert_eq!(removed, 3, "all three snapshots must be removed"); + assert!(!corrupt_path.exists(), "corrupted snapshot must be cleared"); + assert!(store.list().expect("list").is_empty()); +} + +#[test] +fn clear_all_on_missing_dir_returns_zero() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + assert_eq!(store.clear_all().expect("clear"), 0); +} + +#[test] +fn keeps_a_separate_snapshot_per_turn_and_get_returns_latest() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + store + .put(&turn("t", "req-1", "2026-05-04T10:00:00Z")) + .expect("put turn 1"); + store + .put(&turn("t", "req-2", "2026-05-04T10:05:00Z")) + .expect("put turn 2"); + + // Both turns are retained... + let history = store.list_thread("t").expect("list_thread"); + assert_eq!(history.len(), 2); + // ...newest first. + assert_eq!(history[0].request_id, "req-2"); + assert_eq!(history[1].request_id, "req-1"); + + // get(thread) resolves the latest turn (greatest started_at). + let latest = store.get("t").expect("get").expect("present"); + assert_eq!(latest.request_id, "req-2"); + + // get_turn fetches a specific earlier turn. + let earlier = store + .get_turn("t", "req-1") + .expect("get_turn") + .expect("present"); + assert_eq!(earlier.request_id, "req-1"); + assert!(store.get_turn("t", "nope").expect("get_turn").is_none()); + + // list() surfaces exactly one (latest) entry per thread for cold boot. + let latest_per_thread = store.list().expect("list"); + assert_eq!(latest_per_thread.len(), 1); + assert_eq!(latest_per_thread[0].request_id, "req-2"); +} + +#[test] +fn completed_turns_are_pruned_to_the_retention_window() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + // Write 25 completed turns; only the newest COMPLETED_RETENTION (20) survive. + for i in 0..25 { + let mut s = turn( + "t", + &format!("req-{i:02}"), + &format!("2026-05-04T10:{i:02}:00Z"), + ); + s.lifecycle = TurnLifecycle::Completed; + s.updated_at = format!("2026-05-04T10:{i:02}:00Z"); + store.put(&s).expect("put"); + } + let history = store.list_thread("t").expect("list_thread"); + assert_eq!(history.len(), super::COMPLETED_RETENTION); + // The oldest five (req-00..req-04) are gone; the newest survive. + assert!(history.iter().any(|t| t.request_id == "req-24")); + assert!(history.iter().all(|t| t.request_id != "req-00")); + assert!(store.get_turn("t", "req-04").expect("get_turn").is_none()); + assert!(store.get_turn("t", "req-05").expect("get_turn").is_some()); +} + +#[test] +fn a_live_turn_is_not_pruned_alongside_completed_history() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + for i in 0..COMPLETED_RETENTION_PLUS { + let mut s = turn( + "t", + &format!("done-{i:02}"), + &format!("2026-05-04T10:{i:02}:00Z"), + ); + s.lifecycle = TurnLifecycle::Completed; + s.updated_at = format!("2026-05-04T10:{i:02}:00Z"); + store.put(&s).expect("put completed"); + } + // A running turn coexists and is never pruned (only completed turns are). + let mut live = turn("t", "live", "2026-05-04T11:00:00Z"); + live.lifecycle = TurnLifecycle::Streaming; + store.put(&live).expect("put live"); + assert!(store.get_turn("t", "live").expect("get_turn").is_some()); + assert_eq!( + store.get("t").expect("get").expect("present").request_id, + "live" + ); +} + +#[test] +fn migrates_a_legacy_flat_snapshot_into_the_per_turn_layout() { + use std::io::Write as _; + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let root = turn_states_root(&dir); + std::fs::create_dir_all(&root).expect("mkdir root"); + + // Hand-write an old-style flat snapshot at `.json`. + let legacy = turn("legacy-thread", "req-legacy", "2026-05-04T09:00:00Z"); + let flat_path = root.join(format!("{}.json", hex::encode("legacy-thread".as_bytes()))); + let mut f = std::fs::File::create(&flat_path).expect("create flat"); + f.write_all(serde_json::to_vec_pretty(&legacy).unwrap().as_slice()) + .expect("write flat"); + drop(f); + + // First access migrates it into the dir and removes the flat file. + let loaded = store.get("legacy-thread").expect("get").expect("present"); + assert_eq!(loaded.request_id, "req-legacy"); + assert!( + !flat_path.exists(), + "flat file must be removed after migration" + ); + let per_turn = root + .join(hex::encode("legacy-thread".as_bytes())) + .join(format!("{}.json", hex::encode("req-legacy".as_bytes()))); + assert!( + per_turn.exists(), + "snapshot must live under the per-turn path" + ); + + // Migration is idempotent — a second access is a no-op. + assert_eq!( + store + .get("legacy-thread") + .expect("get2") + .expect("present") + .request_id, + "req-legacy" + ); +} + +const COMPLETED_RETENTION_PLUS: usize = super::COMPLETED_RETENTION + 3; + +#[test] +fn put_overwrites_previous_snapshot() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let mut state = sample_state("t"); + state.iteration = 1; + store.put(&state).expect("put 1"); + state.iteration = 7; + state.updated_at = "2026-05-04T10:05:00Z".into(); + store.put(&state).expect("put 2"); + + let loaded = store.get("t").expect("get").expect("present"); + assert_eq!(loaded.iteration, 7); + assert_eq!(loaded.updated_at, "2026-05-04T10:05:00Z"); +} + +/// The last turn of a thread is left `Streaming` on disk when its progress +/// bridge outlives the turn (the bridge only exits when its progress sender +/// drops, which for a cached session waits for the *next* turn). Re-entering +/// the thread then hydrates that snapshot into a permanent "Thinking…" +/// indicator under a reply that already landed, so the turn driver settles it. +#[test] +fn settle_turn_marks_a_live_snapshot_terminal() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let mut state = turn("thread-settle", "req-live", "2026-05-04T10:00:00Z"); + state.lifecycle = TurnLifecycle::Streaming; + state.phase = Some(TurnPhase::Thinking); + state.active_tool = Some("shell".into()); + store.put(&state).expect("put"); + + let changed = store + .settle_turn( + "thread-settle", + "req-live", + TurnLifecycle::Completed, + "2026-05-04T10:05:00Z", + ) + .expect("settle_turn"); + assert!(changed, "a Streaming snapshot must be settled"); + + // `get` is what `threads_turn_state_get` serves to the UI, and the UI reads + // `lifecycle` alone to decide whether the turn is still running. + let loaded = store.get("thread-settle").expect("get").expect("present"); + assert_eq!(loaded.lifecycle, TurnLifecycle::Completed); + assert_eq!(loaded.phase, None); + assert_eq!(loaded.active_tool, None); + assert_eq!(loaded.updated_at, "2026-05-04T10:05:00Z"); +} + +/// Settling is a floor, not an override: a bridge that already recorded the +/// true outcome must win, so a failed turn can never downgrade a `Completed` +/// snapshot to `Interrupted` and raise a spurious retry banner. +#[test] +fn settle_turn_leaves_an_already_terminal_snapshot_alone() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let mut state = turn("thread-done", "req-done", "2026-05-04T10:00:00Z"); + state.lifecycle = TurnLifecycle::Completed; + store.put(&state).expect("put"); + + let changed = store + .settle_turn( + "thread-done", + "req-done", + TurnLifecycle::Interrupted, + "2026-05-04T10:05:00Z", + ) + .expect("settle_turn"); + assert!(!changed, "a terminal snapshot must not be rewritten"); + let loaded = store.get("thread-done").expect("get").expect("present"); + assert_eq!(loaded.lifecycle, TurnLifecycle::Completed); + assert_eq!(loaded.updated_at, "2026-05-04T10:00:00Z"); +} + +/// An absent snapshot is a no-op rather than an error — a turn can end before +/// the bridge ever flushed one (an immediate failure writes nothing). +#[test] +fn settle_turn_is_a_noop_when_no_snapshot_exists() { + let dir = tempdir().expect("tempdir"); + let store = TurnStateStore::new(dir.path().to_path_buf()); + let changed = store + .settle_turn( + "thread-missing", + "req-missing", + TurnLifecycle::Completed, + "2026-05-04T10:05:00Z", + ) + .expect("settle_turn"); + assert!(!changed); +} diff --git a/crates/tinyagents-session/src/turn_state/types.rs b/crates/tinyagents-session/src/turn_state/types.rs new file mode 100644 index 00000000..55bff292 --- /dev/null +++ b/crates/tinyagents-session/src/turn_state/types.rs @@ -0,0 +1,372 @@ +//! Wire/storage types for per-thread agent-turn snapshots. +//! +//! A [`TurnState`] mirrors the live state held by the web-channel +//! progress consumer so the UI can rehydrate after a cold boot or +//! after the user navigates away mid-turn. The shape intentionally +//! parallels `app/src/store/chatRuntimeSlice.ts` so a snapshot can +//! be applied directly to that slice. + +use serde::{Deserialize, Serialize}; + +/// Lifecycle of an in-flight (or formerly in-flight) turn. +/// +/// `Started` is set when the user sends and the agent loop is about +/// to enter the iteration loop. `Streaming` is set after the first +/// progress signal arrives. `Interrupted` is stamped at startup on +/// any snapshot that survived a process restart — there is no live +/// driver to resume it, so the UI should surface a retry affordance. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum TurnLifecycle { + Started, + Streaming, + Interrupted, + /// The turn finished normally. The snapshot is **kept** (not deleted) so + /// the chat "View processing" panel can replay the full transcript + + /// tool timeline after a reload / cold boot — startup interrupted-marking + /// skips this state, and the next turn on the thread overwrites it. + Completed, +} + +/// High-level phase the agent is in within an iteration. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum TurnPhase { + Thinking, + ToolUse, + Subagent, +} + +/// Per-tool entry shown in the live timeline UI. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ToolTimelineStatus { + Running, + Success, + Error, +} + +/// Persisted, plain-language explanation of a FAILED tool row (#4459). +/// +/// Mirrors the live socket `failure` object and the frontend +/// `PersistedToolFailure` (`app/src/types/turnState.ts`) 1:1 — camelCase on the +/// wire, `class`/`category` as the taxonomy's stable variant names — so a +/// settled/reloaded turn keeps its "why + what to do next" copy across a thread +/// switch or a cold boot. Absent on successful rows and on snapshots written +/// before this field. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct PersistedToolFailure { + /// Stable failure-class variant name, e.g. `"Timeout"`, `"Denied"`. + pub class: String, + /// Stable category variant name, e.g. `"Recoverable"`, `"UserDeclined"`. + pub category: String, + /// Whether the core considers the failure automatically recoverable. + pub recoverable: bool, + /// Plain-language cause (`causePlain` on the wire). + pub cause_plain: String, + /// Plain-language next action (`nextAction` on the wire). + pub next_action: String, +} + +/// One row in the per-turn tool timeline. +/// +/// Field names use camelCase on the wire so a snapshot can be applied +/// directly to `chatRuntimeSlice.toolTimelineByThread` without a +/// translation layer in the UI. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct ToolTimelineEntry { + pub id: String, + pub name: String, + pub round: u32, + pub status: ToolTimelineStatus, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub args_buffer: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub display_name: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub detail: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub source_tool_name: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub subagent: Option, + /// Plain-language failure explanation for a FAILED row, carried in the + /// snapshot so it survives a thread switch / cold boot (#4459). `None` on + /// success and on legacy snapshots. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub failure: Option, + /// Size-capped tool result text, persisted so the "View processing" + /// panel can show what a tool returned after a thread switch / cold + /// boot — the live socket forwards the same capped payload on + /// `tool_result`. `None` while running and on legacy snapshots. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub output: Option, + /// Per-turn monotonic ordering key stamped at the moment the row is first + /// created, so a rehydrated timeline can order rows identically to the live + /// stream (conversations-timeline-refactor, Phase 4 amendment). Shares the + /// per-turn ordering space with the transcript item's `seq` field. `None` on snapshots + /// written before this field. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub seq: Option, +} + +/// Live sub-agent activity nested under a `subagent:*` timeline row. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct SubagentActivity { + pub task_id: String, + pub agent_id: String, + /// High-level status: `"running"`, `"awaiting_user"`, `"completed"`, + /// `"failed"`. `None` for legacy snapshots written before this field. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub status: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub mode: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub dedicated_thread: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub child_iteration: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub child_max_iterations: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub iterations: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub elapsed_ms: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub output_chars: Option, + /// Persistent worker sub-thread backing this delegation, when one was + /// created. Lets the UI reopen the full parent↔subagent conversation + /// from memory after a cold boot / interrupted turn. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub worker_thread_id: Option, + /// The parent turn's tool-call id (the `spawn_subagent` / dispatch call) + /// this delegation is attributed to. Mirrors + /// the host's `AgentProgress::SubagentSpawned::parent_call_id`. + /// `None` for legacy snapshots and spawn sites the harness gave no call + /// context (e.g. `orchestration::ops`). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub parent_call_id: Option, + /// Size-capped final assistant text, persisted so a rehydrated row can + /// still show what the sub-agent answered — mirrors the live + /// `subagent_completed` socket payload's `subagent.output`. `None` + /// while running, on failure, and on legacy snapshots. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub output: Option, + #[serde(default)] + pub tool_calls: Vec, + /// Ordered reasoning/narration/tool transcript for this sub-agent — what + /// the inline "Agentic task insights" thoughts render from. Persisted (not + /// live-only) so the thoughts survive a settled turn / reload. + /// `#[serde(default)]` so snapshots written before this field load empty. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub transcript: Vec, +} + +/// One child tool call performed by a running sub-agent. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct SubagentToolCall { + pub call_id: String, + pub tool_name: String, + pub status: ToolTimelineStatus, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub iteration: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub elapsed_ms: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub output_chars: Option, + /// Server-computed human label for this child call (e.g. "Reading file"), + /// or `None` to defer to the client formatter. Mirrors the parent + /// [`ToolTimelineEntry::display_name`] so the same reusable row renderer + /// reads the same field for both main-agent and sub-agent calls. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub display_name: Option, + /// Server-computed contextual detail (e.g. the path / recipient). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub detail: Option, + /// Size-capped arguments the child invoked the tool with, so a rehydrated + /// row shows *what* the sub-agent did and not just which tool it reached + /// for (#5987). Mirrors the live `subagent_tool_call` event's `args` + /// verbatim when it fits the cap; an oversized payload degrades to a + /// truncated string. Taken from the started event when it carries the + /// arguments, otherwise backfilled from + /// `SubagentToolCallCompleted.arguments` — the tinyagents + /// path emits `Value::Null` at start and only captures the input on + /// completion. `None` when the harness captured no input at all + /// (`PayloadCapture::tool_io` off) and on legacy snapshots. + /// + /// Deliberately absent from [`SubagentTranscriptItem::Tool`]: like + /// `output` and `failure`, this is a heavy payload that lives once on the + /// call row, and the frontend grafts it onto the matching transcript item + /// by `call_id` when it rehydrates. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub args: Option, + /// Plain-language failure explanation for a FAILED child call, so a + /// sub-agent's failed row carries the same "why + next" copy as a + /// main-agent row and it survives a snapshot round-trip (#4459). `None` on + /// success and on legacy snapshots. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub failure: Option, + /// Size-capped child tool result text, persisted for the same reason as + /// [`ToolTimelineEntry::output`]. `None` while running and on legacy + /// snapshots. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub output: Option, +} + +/// One ordered item in a sub-agent's processing transcript — its streamed +/// reasoning (`thinking`), visible narration (`text`), or a tool call, in the +/// exact order they occurred. Mirrors the frontend `SubagentTranscriptItem` +/// union 1:1 (order = push order; no `seq` is needed because each sub-agent's +/// transcript is built as a single ordered list). Persisting these lets the +/// inline "Agentic task insights" thoughts survive a settled turn / reload. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +// `rename_all` renames the variant tags; `rename_all_fields` renames the +// fields *inside* the struct variants (call_id → callId, …) — without the +// latter the FE would read `undefined` for camelCase fields. +#[serde( + rename_all = "camelCase", + rename_all_fields = "camelCase", + tag = "kind" +)] +pub enum SubagentTranscriptItem { + /// The sub-agent's hidden reasoning. + Thinking { + #[serde(default, skip_serializing_if = "Option::is_none")] + iteration: Option, + text: String, + }, + /// The sub-agent's visible narration. + Text { + #[serde(default, skip_serializing_if = "Option::is_none")] + iteration: Option, + text: String, + }, + /// A child tool call at the point it occurred (self-contained so a + /// rehydrated row renders without cross-referencing `tool_calls`). + Tool { + #[serde(default, skip_serializing_if = "Option::is_none")] + iteration: Option, + call_id: String, + tool_name: String, + status: ToolTimelineStatus, + #[serde(default, skip_serializing_if = "Option::is_none")] + elapsed_ms: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + output_chars: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + display_name: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + detail: Option, + }, +} + +/// One ordered item in the parent turn's processing transcript. +/// +/// Unlike [`ToolTimelineEntry`] (a flat list of tool rows), the transcript +/// preserves the **interleaving** of the agent's visible narration, its +/// hidden reasoning, and its tool calls in the exact order they streamed — +/// so the chat "View processing" panel can render prose between tool groups +/// the way Claude / Hermes does. `seq` is a monotonic per-turn ordering key +/// (round alone can't order narration vs thinking within one round). Tool +/// items hold only a `call_id` pointer into [`TurnState::tool_timeline`] so +/// the row's status/label live in exactly one place. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +// `rename_all_fields` is required so the `ToolCall.call_id` field serializes as +// `callId` (the FE reads camelCase) — `rename_all` alone only renames variants. +#[serde( + rename_all = "camelCase", + rename_all_fields = "camelCase", + tag = "kind" +)] +pub enum TranscriptItem { + /// The agent's visible assistant text between tool calls. + Narration { round: u32, seq: u32, text: String }, + /// The agent's hidden reasoning (when the model emits it). + /// + /// `started_at` / `ended_at` are epoch milliseconds of the block's first + /// and latest delta, so the UI can show "Thought for 12s" after a reload. + /// Absent on rows written before timing was recorded. + Thinking { + round: u32, + seq: u32, + text: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + started_at: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + ended_at: Option, + }, + /// A pointer to a tool row in [`TurnState::tool_timeline`]. + ToolCall { + round: u32, + seq: u32, + call_id: String, + }, +} + +/// Persisted snapshot of an in-flight (or just-finished) agent turn for one +/// thread. +/// +/// Written to disk by the web-channel progress consumer at iteration +/// boundaries, tool start/complete, and on terminal events. On normal +/// completion it is marked [`TurnLifecycle::Completed`] and **kept** (so the +/// "View processing" panel can replay the finished turn's transcript); a +/// non-terminal snapshot surviving startup is marked +/// [`TurnLifecycle::Interrupted`]. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct TurnState { + pub thread_id: String, + pub request_id: String, + pub lifecycle: TurnLifecycle, + pub iteration: u32, + pub max_iterations: u32, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub phase: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub active_tool: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub active_subagent: Option, + #[serde(default)] + pub streaming_text: String, + #[serde(default)] + pub thinking: String, + #[serde(default)] + pub tool_timeline: Vec, + /// Ordered, interleaved record of the agent's narration, reasoning, and + /// tool calls for the "View processing" panel. `#[serde(default)]` so + /// snapshots written before this field still load (as empty). + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub transcript: Vec, + pub started_at: String, + pub updated_at: String, +} + +impl TurnState { + /// Build a fresh `Started` snapshot for a new turn. + pub fn started( + thread_id: impl Into, + request_id: impl Into, + max_iterations: u32, + now_rfc3339: impl Into, + ) -> Self { + let now = now_rfc3339.into(); + Self { + thread_id: thread_id.into(), + request_id: request_id.into(), + lifecycle: TurnLifecycle::Started, + iteration: 0, + max_iterations, + phase: None, + active_tool: None, + active_subagent: None, + streaming_text: String::new(), + thinking: String::new(), + tool_timeline: Vec::new(), + transcript: Vec::new(), + started_at: now.clone(), + updated_at: now, + } + } +}