diff --git a/crates/xgeny-cli/src/composition.rs b/crates/xgeny-cli/src/composition.rs index 4da6197..29069c5 100644 --- a/crates/xgeny-cli/src/composition.rs +++ b/crates/xgeny-cli/src/composition.rs @@ -69,7 +69,7 @@ use crate::material_catalog::{ WORKSPACE_READ_MATERIAL_PROVIDER_ID, WORKSPACE_READ_RECIPE_DOMAIN, WORKSPACE_READ_RECIPE_FORMAT_VERSION, WorkspaceReadMaterialProvider, WorkspaceReadMaterializer, }; -use crate::model_profile::InferenceLimits; +use crate::model_profile::{InferenceLimits, RequestOptions}; use crate::run_layout::{RunLayout, discover_state_root, generate_run_id}; const WORKSPACE_ID: &str = "primary"; @@ -111,6 +111,7 @@ pub struct LocalRunRequest { pub tokenizer: String, pub credential: Option, pub inference_limits: InferenceLimits, + pub request_options: RequestOptions, pub allow_files: Vec, pub allow_dirs: Vec, pub allow_executables: Vec, @@ -140,6 +141,7 @@ impl LocalRunRequest { tokenizer, credential: None, inference_limits: InferenceLimits::default(), + request_options: RequestOptions::default(), allow_files, allow_dirs: Vec::new(), allow_executables: Vec::new(), @@ -160,6 +162,7 @@ pub struct LocalResumeRequest { pub base_url: Option, pub credential: Option, pub inference_limits: InferenceLimits, + pub request_options: RequestOptions, pub allow_files: Vec, pub allow_dirs: Vec, pub allow_executables: Vec, @@ -170,6 +173,15 @@ pub struct LocalResumeRequest { pub max_ticks: u32, } +/// Deferred provider settings resolved only when an incomplete Run needs model access. +/// The resumed request digest must match the immutable manifest before any inference. +pub struct ResolvedModelEndpoint { + pub base_url: String, + pub credential: Option, + pub inference_limits: InferenceLimits, + pub request_options: RequestOptions, +} + /// Process executable/environment snapshot reused by one interactive host session. /// /// Construction hashes the selected executable targets and binds them to the physical workspace. @@ -227,6 +239,7 @@ pub struct ModelCheckRequest { pub tokenizer: String, pub credential: Option, pub inference_limits: InferenceLimits, + pub request_options: RequestOptions, } /// Stable, redacted failure classes returned by `xgeny model check`. @@ -335,6 +348,7 @@ pub fn check_openai_model(request: ModelCheckRequest) -> Result<(), ModelCheckEr tokenizer, credential, inference_limits: _, + request_options: _, } = request; let config = OpenAiPlannerConfig::new(&base_url, DEFAULT_PLANNER_ID, &model, &tokenizer) .and_then(|config| config.with_timeout(MODEL_CHECK_TIMEOUT)) @@ -357,6 +371,7 @@ pub fn list_openai_models(request: ModelCheckRequest) -> Result, Mod tokenizer, credential, inference_limits: _, + request_options: _, } = request; let config = OpenAiPlannerConfig::new(&base_url, DEFAULT_PLANNER_ID, &model, &tokenizer) .and_then(|config| config.with_timeout(MODEL_CHECK_TIMEOUT)) @@ -367,7 +382,7 @@ pub fn list_openai_models(request: ModelCheckRequest) -> Result, Mod .map_err(map_model_check_failure) } -/// Send one explicit strict-JSON Chat Completions compatibility probe without Run state. +/// Send one explicit, host-validated Chat Completions compatibility probe without Run state. /// /// # Errors /// @@ -379,8 +394,15 @@ pub fn check_openai_compatibility(request: ModelCheckRequest) -> Result<(), Mode tokenizer, credential, inference_limits, + request_options, } = request; - let config = compatibility_probe_config(&base_url, &model, &tokenizer, inference_limits)?; + let config = compatibility_probe_config( + &base_url, + &model, + &tokenizer, + inference_limits, + request_options, + )?; OpenAiCompatibilityChecker::new(config, credential) .map_err(map_model_check_config)? .check() @@ -398,10 +420,13 @@ fn compatibility_probe_config( model: &str, tokenizer: &str, limits: InferenceLimits, + options: RequestOptions, ) -> Result { OpenAiPlannerConfig::new(base_url, DEFAULT_PLANNER_ID, model, tokenizer) .and_then(|config| config.with_max_output_tokens(limits.max_output_tokens())) .and_then(|config| config.with_timeout(limits.timeout())) + .and_then(|config| config.with_response_format(options.response_format)) + .and_then(|config| config.with_thinking(options.thinking)) .map_err(map_model_check_config) } @@ -878,6 +903,7 @@ where &request.model, &request.tokenizer, request.inference_limits, + request.request_options, planning_constraints_required, )?; let local_execution_profile_digest = @@ -958,7 +984,7 @@ pub fn resume_local_with_model_resolver( resolve_model: F, ) -> Result where - F: FnOnce() -> Result<(String, Option), PublicRunError>, + F: FnOnce() -> Result, { resume_local_with_model_resolver_and_progress(request, resolve_model, |_| { DriverProgressControl::Continue @@ -980,7 +1006,7 @@ pub fn resume_local_with_model_resolver_and_progress( on_progress: O, ) -> Result where - F: FnOnce() -> Result<(String, Option), PublicRunError>, + F: FnOnce() -> Result, O: FnMut(DriverProgress) -> DriverProgressControl, { resume_local_composed(request, None, resolve_model, on_progress) @@ -1002,7 +1028,7 @@ pub fn resume_local_with_process_session_and_model_resolver_progress( on_progress: O, ) -> Result where - F: FnOnce() -> Result<(String, Option), PublicRunError>, + F: FnOnce() -> Result, O: FnMut(DriverProgress) -> DriverProgressControl, { resume_local_composed(request, Some(process_session), resolve_model, on_progress) @@ -1016,7 +1042,7 @@ fn resume_local_composed( mut on_progress: O, ) -> Result where - F: FnOnce() -> Result<(String, Option), PublicRunError>, + F: FnOnce() -> Result, O: FnMut(DriverProgress) -> DriverProgressControl, { validate_max_ticks(request.max_ticks)?; @@ -1124,22 +1150,28 @@ where return Err(PublicRunError::Configuration); } let planner = if request.allow_remote_model_egress { - let (base_url, credential) = match request.base_url.as_ref() { - Some(base_url) => (base_url.clone(), request.credential.clone()), + let resolved = match request.base_url.as_ref() { + Some(base_url) => ResolvedModelEndpoint { + base_url: base_url.clone(), + credential: request.credential.clone(), + inference_limits: request.inference_limits, + request_options: request.request_options, + }, None => resolve_model()?, }; let config = planner_config( - &base_url, + &resolved.base_url, manifest.planner_id(), manifest.model(), manifest.tokenizer(), - request.inference_limits, + resolved.inference_limits, + resolved.request_options, catalog.workspace_discovery() || process.is_some(), )?; if manifest.request_profile_digest() != config.request_profile_digest() { return Err(PublicRunError::Configuration); } - Some(remote_planner(config, credential)?) + Some(remote_planner(config, resolved.credential)?) } else { None }; @@ -1688,11 +1720,14 @@ fn planner_config( model: &str, tokenizer: &str, limits: InferenceLimits, + options: RequestOptions, planning_constraints_required: bool, ) -> Result { let config = OpenAiPlannerConfig::new(base_url, planner_id, model, tokenizer) .and_then(|config| config.with_max_output_tokens(limits.max_output_tokens())) .and_then(|config| config.with_timeout(limits.timeout())) + .and_then(|config| config.with_response_format(options.response_format)) + .and_then(|config| config.with_thinking(options.thinking)) .map_err(map_provider_config)?; if planning_constraints_required { config @@ -2681,15 +2716,21 @@ mod tests { #[test] fn probe_and_planner_follow_the_profile_limits_and_stay_digest_identical() { let limits = InferenceLimits::new(Duration::from_secs(300), 2_048).unwrap(); - let probe = - compatibility_probe_config("http://127.0.0.1:1/v1", "model", "tokenizer", limits) - .unwrap(); + let probe = compatibility_probe_config( + "http://127.0.0.1:1/v1", + "model", + "tokenizer", + limits, + RequestOptions::default(), + ) + .unwrap(); let production = planner_config( "http://127.0.0.1:1/v1", DEFAULT_PLANNER_ID, "model", "tokenizer", limits, + RequestOptions::default(), false, ) .unwrap(); @@ -2705,6 +2746,7 @@ mod tests { "model", "tokenizer", other, + RequestOptions::default(), false, ) .unwrap(); @@ -2722,6 +2764,7 @@ mod tests { "model", "tokenizer", InferenceLimits::default(), + RequestOptions::default(), ) .expect("probe config should validate"); let production = planner_config( @@ -2730,6 +2773,7 @@ mod tests { "model", "tokenizer", InferenceLimits::default(), + RequestOptions::default(), false, ) .expect("production config should validate"); @@ -2741,6 +2785,45 @@ mod tests { ); } + #[test] + fn explicit_wire_options_have_identical_probe_and_planner_digests() { + use xgeny_provider_openai::{ResponseFormat, ThinkingMode}; + for response_format in [ResponseFormat::JsonSchema, ResponseFormat::JsonObject] { + for thinking in [ + ThinkingMode::Default, + ThinkingMode::Disabled, + ThinkingMode::Enabled, + ] { + let options = RequestOptions { + response_format, + thinking, + }; + let probe = compatibility_probe_config( + "https://provider.example/v1", + "model", + "tokenizer", + InferenceLimits::default(), + options, + ) + .unwrap(); + let planner = planner_config( + "https://provider.example/v1", + DEFAULT_PLANNER_ID, + "model", + "tokenizer", + InferenceLimits::default(), + options, + false, + ) + .unwrap(); + assert_eq!( + probe.request_profile_digest(), + planner.request_profile_digest() + ); + } + } + } + #[test] fn model_call_unknown_class_reaches_the_public_result_code() { // Journaled path: the driver carries the durable ModelCallUnknownReason. @@ -2890,6 +2973,7 @@ mod tests { tokenizer: "tokenizer".to_owned(), credential: None, inference_limits: InferenceLimits::default(), + request_options: RequestOptions::default(), allow_files: vec!["README.md".to_owned()], allow_dirs: Vec::new(), allow_executables: Vec::new(), @@ -2944,6 +3028,7 @@ mod tests { tokenizer: "tokenizer".to_owned(), credential: None, inference_limits: InferenceLimits::default(), + request_options: RequestOptions::default(), allow_files: Vec::new(), allow_dirs: vec![".".to_owned()], allow_executables: vec![specification], diff --git a/crates/xgeny-cli/src/main.rs b/crates/xgeny-cli/src/main.rs index 47bbaf4..254c762 100644 --- a/crates/xgeny-cli/src/main.rs +++ b/crates/xgeny-cli/src/main.rs @@ -10,14 +10,15 @@ use xgeny_cli::{ DriverProgress, DriverProgressControl, InferenceLimits, LocalCommandResult, LocalProcessSession, LocalResumeRequest, LocalRunRequest, ModelCheckError, ModelCheckRequest, ModelCredentialStore, ModelProfile, ModelProfileError, ModelProfileStore, - OsModelCredentialStore, PublicRunError, check_openai_compatibility, check_openai_model, - discard_local_model_call, discard_local_model_call_at_head, inspect_local_model_call, - list_openai_models, new_credential_reference, prepare_local_process_session, resume_local, + OsModelCredentialStore, PublicRunError, RequestOptions, ResolvedModelEndpoint, + check_openai_compatibility, check_openai_model, discard_local_model_call, + discard_local_model_call_at_head, inspect_local_model_call, list_openai_models, + new_credential_reference, prepare_local_process_session, resume_local, resume_local_with_model_resolver, resume_local_with_model_resolver_and_progress, resume_local_with_process_session_and_model_resolver_progress, run_local_with_process_session_progress, run_local_with_started, }; -use xgeny_provider_openai::BearerCredential; +use xgeny_provider_openai::{BearerCredential, ResponseFormat, ThinkingMode}; use zeroize::Zeroizing; mod repl; @@ -104,6 +105,8 @@ enum ModelCommand { after_long_help = "Resolution order: explicit options, XGENY_OPENAI_BASE_URL / XGENY_OPENAI_MODEL / XGENY_OPENAI_TOKENIZER environment, then the selected/active profile. Planner inference limits follow XGENY_OPENAI_INFERENCE_TIMEOUT / XGENY_OPENAI_MAX_OUTPUT_TOKENS, then the profile (default 300s / 1024 tokens). HTTPS authentication uses --token-stdin, XGENY_OPENAI_API_KEY, then the profile secure store; no token value is accepted as a command argument." )] struct ModelCheckArgs { + #[command(flatten)] + request_options: RequestOptionArgs, /// OpenAI-compatible API base URL ending in /v1. #[arg(long)] base_url: Option, @@ -119,7 +122,7 @@ struct ModelCheckArgs { /// Read one API token line from standard input; the value is never persisted. #[arg(long)] token_stdin: bool, - /// Also send one strict JSON Schema Chat Completions compatibility probe. + /// Also send one host-validated Chat Completions probe using the selected response format. #[arg(long)] compatibility: bool, } @@ -129,6 +132,8 @@ struct ModelCheckArgs { after_long_help = "Interactive setup hides token input and stores it only in the platform secure store. In automation, --token-stdin or XGENY_OPENAI_API_KEY is ephemeral unless --store-token is explicitly supplied." )] struct ModelSetupArgs { + #[command(flatten)] + request_options: RequestOptionArgs, /// Profile name to create or replace. #[arg(long, default_value = "default")] name: String, @@ -179,6 +184,8 @@ struct ModelOptionalNameArgs { after_long_help = "Resolution order: explicit options, XGENY_OPENAI_BASE_URL / XGENY_OPENAI_MODEL / XGENY_OPENAI_TOKENIZER environment, then the selected/active profile. Planner inference limits follow XGENY_OPENAI_INFERENCE_TIMEOUT / XGENY_OPENAI_MAX_OUTPUT_TOKENS, then the profile (default 300s / 1024 tokens). HTTPS authentication uses --token-stdin, XGENY_OPENAI_API_KEY, then the profile secure store. Credentials are ignored for loopback HTTP and cannot be passed as a command-line value." )] struct RunArgs { + #[command(flatten)] + request_options: RequestOptionArgs, /// Goal sent to the bounded planner. goal: String, /// Workspace root opened as the local filesystem capability. @@ -234,6 +241,8 @@ struct RunArgs { after_long_help = "For an incomplete Run, endpoint resolution is explicit --base-url, XGENY_OPENAI_BASE_URL, then the selected/active profile. HTTPS authentication uses --token-stdin, XGENY_OPENAI_API_KEY, then the matching profile secure store. Credentials are ignored for loopback HTTP." )] struct ResumeArgs { + #[command(flatten)] + request_options: RequestOptionArgs, /// Durable Run identifier printed by `xgeny run`. run_id: String, /// Original physical workspace root; unnecessary for completed replay. @@ -274,6 +283,16 @@ struct ResumeArgs { max_ticks: u32, } +#[derive(Debug, Args, Default)] +struct RequestOptionArgs { + /// Structured output transport; `json_object` is validated by `XGENy`, not server-enforced schema. + #[arg(long, value_parser = ["json_schema", "json_object"])] + response_format: Option, + /// Explicit provider thinking setting; default omits the provider-specific setting. + #[arg(long, value_parser = ["default", "disabled", "enabled"])] + thinking: Option, +} + fn main() -> ExitCode { let cli = Cli::parse(); match cli.command { @@ -380,7 +399,7 @@ impl repl::ReplHost for InteractiveHost { grants: repl::InvocationGrants, progress: &mut dyn FnMut(DriverProgress) -> DriverProgressControl, ) -> Result { - let model = resolve_model(None, None, None, None, false) + let model = resolve_model(None, None, None, None, false, RequestOptionArgs::default()) .map_err(|error| repl::ReplFailure::new(error.code()))?; let process_session = self.process_session()?; run_local_with_process_session_progress( @@ -393,6 +412,7 @@ impl repl::ReplHost for InteractiveHost { tokenizer: model.tokenizer, credential: model.credential, inference_limits: model.inference_limits, + request_options: model.request_options, allow_files: Vec::new(), allow_dirs: vec![".".to_owned()], allow_executables: Vec::new(), @@ -421,8 +441,8 @@ impl repl::ReplHost for InteractiveHost { workspace: Some(self.workspace.clone()), base_url: None, credential: None, - inference_limits: resolve_inference_limits(None, None, None) - .map_err(|error| repl::ReplFailure::new(error.code()))?, + inference_limits: InferenceLimits::default(), + request_options: RequestOptions::default(), allow_files: Vec::new(), allow_dirs: vec![".".to_owned()], allow_executables: if process_session.is_some() { @@ -442,12 +462,10 @@ impl repl::ReplHost for InteractiveHost { if !grants.model { return Err(PublicRunError::Configuration); } - resolve_endpoint(None, None, false) - .map(|resolved| (resolved.base_url, resolved.credential)) - .map_err(|error| { - resolution_error = Some(error); - PublicRunError::Configuration - }) + resolve_endpoint(None, None, false, RequestOptionArgs::default()).map_err(|error| { + resolution_error = Some(error); + PublicRunError::Configuration + }) }; if let Some(process_session) = process_session.as_ref() { resume_local_with_process_session_and_model_resolver_progress( @@ -504,6 +522,7 @@ fn ensure_interactive_model() -> Result<(), ModelCliError> { return Ok(()); } let (profile, stored) = try_model_setup(ModelSetupArgs { + request_options: RequestOptionArgs::default(), name: "default".to_owned(), base_url: None, model: None, @@ -557,6 +576,7 @@ fn run_command(args: RunArgs) -> ExitCode { args.tokenizer, args.profile, args.token_stdin, + args.request_options, ) { Ok(resolved) => resolved, Err(error) => return present_model_configuration_error(error), @@ -571,6 +591,7 @@ fn run_command(args: RunArgs) -> ExitCode { tokenizer: resolved.tokenizer, credential: resolved.credential, inference_limits: resolved.inference_limits, + request_options: resolved.request_options, allow_files: args.allow_files, allow_dirs: args.allow_dirs, allow_executables: args.allow_executables, @@ -586,6 +607,7 @@ fn run_command(args: RunArgs) -> ExitCode { fn resume_command(args: ResumeArgs) -> ExitCode { let ResumeArgs { + request_options, run_id, workspace, base_url, @@ -600,16 +622,13 @@ fn resume_command(args: ResumeArgs) -> ExitCode { allow_execute, max_ticks, } = args; - let inference_limits = match resolve_inference_limits(None, None, None) { - Ok(limits) => limits, - Err(error) => return present_model_configuration_error(error), - }; let request = LocalResumeRequest { run_id, workspace, base_url: None, credential: None, - inference_limits, + inference_limits: InferenceLimits::default(), + request_options: RequestOptions::default(), allow_files, allow_dirs, allow_executables, @@ -625,12 +644,10 @@ fn resume_command(args: ResumeArgs) -> ExitCode { let mut resolution_error = None; let result = resume_local_with_model_resolver(request, || { - resolve_endpoint(base_url, profile, token_stdin) - .map(|resolved| (resolved.base_url, resolved.credential)) - .map_err(|error| { - resolution_error = Some(error); - PublicRunError::Configuration - }) + resolve_endpoint(base_url, profile, token_stdin, request_options).map_err(|error| { + resolution_error = Some(error); + PublicRunError::Configuration + }) }); if let Some(error) = resolution_error { present_model_configuration_error(error) @@ -671,11 +688,7 @@ struct ResolvedModel { tokenizer: String, credential: Option, inference_limits: InferenceLimits, -} - -struct ResolvedEndpoint { - base_url: String, - credential: Option, + request_options: RequestOptions, } #[derive(Clone, Copy, PartialEq, Eq)] @@ -702,6 +715,7 @@ enum ModelCliError { InvalidCredential, CredentialRequiresHttps, InvalidInferenceLimits, + InvalidRequestOptions, } impl ModelCliError { @@ -715,6 +729,7 @@ impl ModelCliError { Self::InvalidCredential => "api_key_invalid", Self::CredentialRequiresHttps => "api_key_requires_https", Self::InvalidInferenceLimits => "inference_limits_invalid", + Self::InvalidRequestOptions => "request_options_invalid", } } @@ -734,7 +749,8 @@ impl ModelCliError { | Self::InputUnavailable | Self::InvalidCredential | Self::CredentialRequiresHttps - | Self::InvalidInferenceLimits => 64, + | Self::InvalidInferenceLimits + | Self::InvalidRequestOptions => 64, Self::Check(error) => error.exit_code(), } } @@ -759,7 +775,14 @@ fn model_setup(args: ModelSetupArgs) -> ExitCode { println!(" profile: {}", profile.name()); println!(" model: {}", profile.model()); println!(" catalog: exact model advertised"); - println!(" chat completions: strict JSON compatible"); + println!( + " chat completions: {}", + compatibility_label(profile.request_options()) + ); + println!( + " thinking: {}", + thinking_label(profile.request_options().thinking) + ); println!( " inference limits: timeout={}s max_output_tokens={}", profile.inference_limits().timeout().as_secs(), @@ -822,6 +845,7 @@ fn try_model_setup(args: ModelSetupArgs) -> Result<(ModelProfile, bool), ModelCl tokenizer: catalog_identity, credential: credential.clone(), inference_limits: InferenceLimits::default(), + request_options: RequestOptions::default(), })?; let model = match requested_model { Some(model) if models.iter().any(|candidate| candidate == &model) => model, @@ -845,12 +869,14 @@ fn try_model_setup(args: ModelSetupArgs) -> Result<(ModelProfile, bool), ModelCl args.max_output_tokens, existing.as_ref(), )?; + let request_options = resolve_request_options(args.request_options, existing.as_ref())?; check_openai_compatibility(ModelCheckRequest { base_url: base_url.clone(), model: model.clone(), tokenizer: tokenizer.clone(), credential, inference_limits, + request_options, })?; let _lock = store.try_lock()?; @@ -865,6 +891,7 @@ fn try_model_setup(args: ModelSetupArgs) -> Result<(ModelProfile, bool), ModelCl let credentials = OsModelCredentialStore; let mut profile = ModelProfile::new(&args.name, base_url, model, tokenizer)?; profile.set_inference_limits(inference_limits)?; + profile.set_request_options(request_options)?; let retain_existing = secret.source == SetupSecretSource::SecureStore; let should_store = args.store_token || secret.source == SetupSecretSource::Interactive; let mut new_reference = None; @@ -914,12 +941,14 @@ fn model_list() -> ExitCode { " " }; println!( - "{marker} {} model={} tokenizer={} timeout={}s max_output_tokens={} authentication={}", + "{marker} {} model={} tokenizer={} timeout={}s max_output_tokens={} response_format={} thinking={} authentication={}", profile.name(), profile.model(), profile.tokenizer(), profile.inference_limits().timeout().as_secs(), profile.inference_limits().max_output_tokens(), + response_format_label(profile.request_options().response_format), + thinking_label(profile.request_options().thinking), if profile.has_stored_credential() { "secure_store" } else { @@ -1019,6 +1048,7 @@ fn model_check(args: ModelCheckArgs) -> ExitCode { args.tokenizer, args.profile, args.token_stdin, + args.request_options, ) { Ok(resolved) => resolved, Err(error) => return present_model_command_error("check", error), @@ -1029,6 +1059,7 @@ fn model_check(args: ModelCheckArgs) -> ExitCode { tokenizer: resolved.tokenizer.clone(), credential: resolved.credential.clone(), inference_limits: resolved.inference_limits, + request_options: resolved.request_options, }; if let Err(error) = check_openai_model(request) { return present_model_check_error(error); @@ -1040,6 +1071,7 @@ fn model_check(args: ModelCheckArgs) -> ExitCode { tokenizer: resolved.tokenizer, credential: resolved.credential, inference_limits: resolved.inference_limits, + request_options: resolved.request_options, }) { return present_model_check_error(error); @@ -1049,7 +1081,7 @@ fn model_check(args: ModelCheckArgs) -> ExitCode { println!( " chat completions: {}", if args.compatibility { - "strict JSON compatible" + compatibility_label(resolved.request_options) } else { "not requested" } @@ -1064,6 +1096,7 @@ fn resolve_model( tokenizer: Option, profile_name: Option, token_stdin: bool, + request_options: RequestOptionArgs, ) -> Result { let profile = select_profile(profile_name)?; let base_url = base_url @@ -1088,12 +1121,14 @@ fn resolve_model( .unwrap_or_else(|| model.clone()); let credential = resolve_credential(&base_url, token_stdin, profile.as_ref())?; let inference_limits = resolve_inference_limits(None, None, profile.as_ref())?; + let request_options = resolve_request_options(request_options, profile.as_ref())?; Ok(ResolvedModel { base_url, model, tokenizer, credential, inference_limits, + request_options, }) } @@ -1135,7 +1170,8 @@ fn resolve_endpoint( base_url: Option, profile_name: Option, token_stdin: bool, -) -> Result { + request_options: RequestOptionArgs, +) -> Result { let profile = select_profile(profile_name)?; let base_url = base_url .or(read_environment("XGENY_OPENAI_BASE_URL")?) @@ -1146,12 +1182,72 @@ fn resolve_endpoint( }) .ok_or(ModelCliError::MissingConfiguration)?; let credential = resolve_credential(&base_url, token_stdin, profile.as_ref())?; - Ok(ResolvedEndpoint { + Ok(ResolvedModelEndpoint { base_url, credential, + inference_limits: resolve_inference_limits(None, None, profile.as_ref())?, + request_options: resolve_request_options(request_options, profile.as_ref())?, + }) +} + +fn resolve_request_options( + explicit: RequestOptionArgs, + profile: Option<&ModelProfile>, +) -> Result { + let base = profile + .map(ModelProfile::request_options) + .unwrap_or_default(); + let response_format = match explicit + .response_format + .or(read_environment("XGENY_OPENAI_RESPONSE_FORMAT")?) + .as_deref() + { + None => base.response_format, + Some("json_schema") => ResponseFormat::JsonSchema, + Some("json_object") => ResponseFormat::JsonObject, + Some(_) => return Err(ModelCliError::InvalidRequestOptions), + }; + let thinking = match explicit + .thinking + .or(read_environment("XGENY_OPENAI_THINKING")?) + .as_deref() + { + None => base.thinking, + Some("default") => ThinkingMode::Default, + Some("disabled") => ThinkingMode::Disabled, + Some("enabled") => ThinkingMode::Enabled, + Some(_) => return Err(ModelCliError::InvalidRequestOptions), + }; + Ok(RequestOptions { + response_format, + thinking, }) } +const fn response_format_label(format: ResponseFormat) -> &'static str { + match format { + ResponseFormat::JsonSchema => "json_schema", + ResponseFormat::JsonObject => "json_object", + } +} + +const fn thinking_label(thinking: ThinkingMode) -> &'static str { + match thinking { + ThinkingMode::Default => "default", + ThinkingMode::Disabled => "disabled", + ThinkingMode::Enabled => "enabled", + } +} + +const fn compatibility_label(options: RequestOptions) -> &'static str { + match options.response_format { + ResponseFormat::JsonSchema => "strict JSON compatible", + ResponseFormat::JsonObject => { + "JSON object compatible (host-validated; no server schema guarantee)" + } + } +} + fn select_profile(name: Option) -> Result, ModelCliError> { let requested = name.or(read_environment("XGENY_MODEL_PROFILE")?); let store = match ModelProfileStore::discover() { diff --git a/crates/xgeny-cli/src/model_profile.rs b/crates/xgeny-cli/src/model_profile.rs index db82711..c1bda89 100644 --- a/crates/xgeny-cli/src/model_profile.rs +++ b/crates/xgeny-cli/src/model_profile.rs @@ -12,7 +12,7 @@ use keyring::{Entry, Error as KeyringError}; use serde::{Deserialize, Serialize}; use sha2::{Digest, Sha256}; use thiserror::Error; -use xgeny_provider_openai::OpenAiPlannerConfig; +use xgeny_provider_openai::{OpenAiPlannerConfig, ResponseFormat, ThinkingMode}; use zeroize::Zeroizing; const PROFILE_FILE: &str = "model-profiles.json"; @@ -83,6 +83,13 @@ impl Default for InferenceLimits { } } +/// Explicit, non-secret provider wire options bound into a Run's request profile digest. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct RequestOptions { + pub response_format: ResponseFormat, + pub thinking: ThinkingMode, +} + /// One non-secret OpenAI-compatible model profile. #[derive(Clone, PartialEq, Eq)] pub struct ModelProfile { @@ -92,6 +99,7 @@ pub struct ModelProfile { tokenizer: String, credential_ref: Option, inference_limits: InferenceLimits, + request_options: RequestOptions, } impl ModelProfile { @@ -113,6 +121,7 @@ impl ModelProfile { tokenizer: tokenizer.into(), credential_ref: None, inference_limits: InferenceLimits::default(), + request_options: RequestOptions::default(), }; profile.validate()?; Ok(profile) @@ -123,6 +132,26 @@ impl ModelProfile { self.inference_limits } + #[must_use] + pub const fn request_options(&self) -> RequestOptions { + self.request_options + } + + /// Replace explicit provider wire options after validating them. + /// + /// # Errors + /// Returns `InvalidProfile` if the provider configuration rejects the options. + pub fn set_request_options( + &mut self, + options: RequestOptions, + ) -> Result<(), ModelProfileError> { + let mut candidate = self.clone(); + candidate.request_options = options; + candidate.validate()?; + self.request_options = options; + Ok(()) + } + /// Replace the planner inference limits. /// /// # Errors @@ -203,6 +232,8 @@ impl ModelProfile { ) .and_then(|config| config.with_timeout(self.inference_limits.timeout)) .and_then(|config| config.with_max_output_tokens(self.inference_limits.max_output_tokens)) + .and_then(|config| config.with_response_format(self.request_options.response_format)) + .and_then(|config| config.with_thinking(self.request_options.thinking)) .map(|_| ()) .map_err(|_| ModelProfileError::InvalidProfile) } @@ -221,6 +252,7 @@ impl std::fmt::Debug for ModelProfile { &self.credential_ref.as_ref().map(|_| ""), ) .field("inference_limits", &self.inference_limits) + .field("request_options", &self.request_options) .finish() } } @@ -716,6 +748,10 @@ struct StoredProfile { inference_timeout_seconds: u64, #[serde(default = "default_max_output_tokens")] max_output_tokens: u32, + #[serde(default)] + response_format: ResponseFormat, + #[serde(default)] + thinking: ThinkingMode, } fn default_inference_timeout_seconds() -> u64 { @@ -736,6 +772,8 @@ impl StoredProfile { credential_ref: profile.credential_ref.clone(), inference_timeout_seconds: profile.inference_limits.timeout.as_secs(), max_output_tokens: profile.inference_limits.max_output_tokens, + response_format: profile.request_options.response_format, + thinking: profile.request_options.thinking, } } @@ -752,6 +790,10 @@ impl StoredProfile { tokenizer: self.tokenizer, credential_ref: self.credential_ref, inference_limits, + request_options: RequestOptions { + response_format: self.response_format, + thinking: self.thinking, + }, }; profile .validate() @@ -1115,6 +1157,32 @@ mod tests { ); } + #[test] + fn request_options_round_trip_and_reject_unknown_modes() { + for response_format in [ResponseFormat::JsonSchema, ResponseFormat::JsonObject] { + for thinking in [ + ThinkingMode::Default, + ThinkingMode::Disabled, + ThinkingMode::Enabled, + ] { + let mut original = profile("wire"); + let options = RequestOptions { + response_format, + thinking, + }; + original.set_request_options(options).unwrap(); + let encoded = serde_json::to_value(StoredProfile::from_profile(&original)).unwrap(); + let decoded: StoredProfile = serde_json::from_value(encoded.clone()).unwrap(); + assert_eq!(decoded.into_profile().unwrap().request_options(), options); + for field in ["responseFormat", "thinking"] { + let mut malformed = encoded.clone(); + malformed[field] = serde_json::json!("auto-detect"); + assert!(serde_json::from_value::(malformed).is_err()); + } + } + } + } + #[test] fn inference_limits_round_trip_and_legacy_files_load_with_defaults() { let directory = tempdir().unwrap(); @@ -1168,6 +1236,10 @@ mod tests { legacy_loaded.active().unwrap().inference_limits(), InferenceLimits::default() ); + assert_eq!( + legacy_loaded.active().unwrap().request_options(), + RequestOptions::default() + ); // Out-of-range stored values fail closed like any other invalid profile field. let out_of_range = br#"{ diff --git a/crates/xgeny-cli/tests/model_profiles.rs b/crates/xgeny-cli/tests/model_profiles.rs index 363f619..d30a4fc 100644 --- a/crates/xgeny-cli/tests/model_profiles.rs +++ b/crates/xgeny-cli/tests/model_profiles.rs @@ -301,10 +301,182 @@ fn xgeny(config: &Path, state: &Path) -> Command { .env_remove("XGENY_OPENAI_BASE_URL") .env_remove("XGENY_OPENAI_MODEL") .env_remove("XGENY_OPENAI_TOKENIZER") + .env_remove("XGENY_OPENAI_RESPONSE_FORMAT") + .env_remove("XGENY_OPENAI_THINKING") + .env_remove("XGENY_OPENAI_INFERENCE_TIMEOUT") + .env_remove("XGENY_OPENAI_MAX_OUTPUT_TOKENS") .env_remove("XGENY_OPENAI_API_KEY"); command } +#[test] +#[allow(clippy::too_many_lines)] // One vertical setup/run/resume fixture checks both wire modes. +fn explicit_wire_profiles_round_trip_and_resume_with_committed_settings() { + // Independent transport/thinking pairs, not hostname- or model-name-specific behavior. + for (format, thinking) in [("json_object", "disabled"), ("json_schema", "enabled")] { + let fixture = tempdir().unwrap(); + let config = fixture.path().join("config"); + let state = fixture.path().join("state"); + let workspace = fixture.path().join("workspace"); + fs::create_dir(&workspace).unwrap(); + fs::write(workspace.join("README.md"), "wire profile fixture").unwrap(); + let server = ModelServer::spawn(3, true); + let setup = xgeny(&config, &state) + .args([ + "model", + "setup", + "--base-url", + &server.base_url, + "--model", + MODEL, + "--response-format", + format, + "--thinking", + thinking, + "--inference-timeout", + "600", + "--max-output-tokens", + "2048", + ]) + .output() + .unwrap(); + assert_success(&setup); + let output = String::from_utf8_lossy(&setup.stdout); + if format == "json_object" { + assert!(output.contains("host-validated; no server schema guarantee")); + assert!(!output.contains("strict JSON compatible")); + } + let profiles: Value = + serde_json::from_slice(&fs::read(config.join("model-profiles.json")).unwrap()).unwrap(); + assert_eq!(profiles["profiles"][0]["responseFormat"], format); + assert_eq!(profiles["profiles"][0]["thinking"], thinking); + let run = xgeny(&config, &state) + .current_dir(&workspace) + .args([ + "run", + "--allow-file", + "README.md", + "--allow-remote-model-egress", + "read profile fixture", + ]) + .output() + .unwrap(); + assert_eq!( + run.status.code(), + Some(10), + "{}", + String::from_utf8_lossy(&run.stderr) + ); + let stderr = String::from_utf8_lossy(&run.stderr); + let run_id = stderr + .split_whitespace() + .find_map(|part| part.strip_prefix("run_id=")) + .unwrap(); + // A changed option is rejected before model invocation or journal mutation. + for override_args in [ + [ + "--response-format", + if format == "json_object" { + "json_schema" + } else { + "json_object" + }, + ], + ["--thinking", "default"], + ] { + let changed = xgeny(&config, &state) + .current_dir(&workspace) + .args([ + "resume", + run_id, + "--allow-file", + "README.md", + "--allow-remote-model-egress", + ]) + .args(override_args) + .output() + .unwrap(); + assert_eq!(changed.status.code(), Some(64)); + assert!(String::from_utf8_lossy(&changed.stderr).contains("configuration_mismatch")); + } + let resumed = xgeny(&config, &state) + .current_dir(&workspace) + .args([ + "resume", + run_id, + "--allow-file", + "README.md", + "--allow-remote-model-egress", + ]) + .output() + .unwrap(); + assert_eq!( + resumed.status.code(), + Some(10), + "{}", + String::from_utf8_lossy(&resumed.stderr) + ); + assert!(String::from_utf8_lossy(&resumed.stderr).contains("read_approval_required")); + let requests = server.handle.join().unwrap(); + for request in &requests[1..] { + let body = request_body(request); + assert_eq!(body["response_format"]["type"], format); + assert_eq!(body["thinking"]["type"], thinking); + assert_eq!(body["max_tokens"], 2048); + assert!(body.get("seed").is_none()); + if thinking == "enabled" { + assert!(body.get("temperature").is_none()); + assert_eq!(body["reasoning_effort"], "low"); + } + } + } +} + +#[test] +fn request_option_precedence_and_invalid_values_are_explicit() { + let fixture = tempdir().unwrap(); + let config = fixture.path().join("config"); + let state = fixture.path().join("state"); + let server = ModelServer::spawn(2, false); + let checked = xgeny(&config, &state) + .env("XGENY_OPENAI_RESPONSE_FORMAT", "json_schema") + .env("XGENY_OPENAI_THINKING", "enabled") + .args([ + "model", + "check", + "--base-url", + &server.base_url, + "--model", + MODEL, + "--compatibility", + "--response-format", + "json_object", + "--thinking", + "disabled", + ]) + .output() + .unwrap(); + assert_success(&checked); + let requests = server.handle.join().unwrap(); + let probe = request_body(&requests[1]); + assert_eq!(probe["response_format"]["type"], "json_object"); + assert_eq!(probe["thinking"]["type"], "disabled"); + let invalid = xgeny(&config, &state) + .env("XGENY_OPENAI_RESPONSE_FORMAT", "auto") + .args([ + "model", + "check", + "--base-url", + "http://127.0.0.1:1/v1", + "--model", + MODEL, + ]) + .output() + .unwrap(); + assert_eq!(invalid.status.code(), Some(64)); + assert!(String::from_utf8_lossy(&invalid.stderr).contains("request_options_invalid")); +} + fn assert_success(output: &Output) { assert!( output.status.success(), diff --git a/crates/xgeny-cli/tests/public_run_resume.rs b/crates/xgeny-cli/tests/public_run_resume.rs index d2c299a..468f338 100644 --- a/crates/xgeny-cli/tests/public_run_resume.rs +++ b/crates/xgeny-cli/tests/public_run_resume.rs @@ -401,6 +401,9 @@ fn separate_processes_read_once_continue_with_exact_output_and_replay_offline() .args(["resume", &run_id, "--allow-remote-model-egress"]) .env("XGENY_OPENAI_BASE_URL", "not-a-provider-url") .env("XGENY_OPENAI_API_KEY", "invalid\ncredential") + .env("XGENY_OPENAI_RESPONSE_FORMAT", "invalid-format") + .env("XGENY_OPENAI_THINKING", "invalid-thinking") + .env("XGENY_OPENAI_INFERENCE_TIMEOUT", "invalid-timeout") .bounded_output() .expect("offline replay process should run"); assert_exit(&replay, 0); diff --git a/crates/xgeny-provider-openai/src/lib.rs b/crates/xgeny-provider-openai/src/lib.rs index 4f1707e..1410dd9 100644 --- a/crates/xgeny-provider-openai/src/lib.rs +++ b/crates/xgeny-provider-openai/src/lib.rs @@ -1,5 +1,6 @@ #![doc = "Bounded OpenAI-compatible planner adapter for `XGENy`."] +use std::borrow::Cow; use std::collections::BTreeSet; use std::fmt; use std::time::Duration; @@ -27,6 +28,7 @@ const PROMPT_TEMPLATE_REVISION: &str = "xgeny.openai-planner-prompt/v4-compact"; const CONSTRAINED_PROMPT_TEMPLATE_REVISION: &str = "xgeny.openai-planner-prompt/v4-compact-constrained"; const PROVIDER_DIALECT: &str = "openai.chat-completions/json-schema-v1"; +const JSON_OBJECT_DIALECT: &str = "openai.chat-completions/json-object-v1"; const DEFAULT_MAX_OUTPUT_TOKENS: u32 = 4_096; const DEFAULT_MAX_REQUEST_BYTES: usize = 1024 * 1024; const DEFAULT_MAX_RESPONSE_BYTES: usize = 512 * 1024; @@ -62,6 +64,31 @@ const COMPATIBILITY_SYSTEM_PROMPT: &str = "This is an XGENy connectivity probe. /// silently drops the grammar lets the model follow the instruction, and the production document /// parser rejects the unknown field. const COMPATIBILITY_USER_PROMPT: &str = "This is an XGENy connectivity probe with no planning context. Return a completion_candidate: set formatVersion to 1, kind to completion_candidate, steps to an empty array, and summary to the string ok. Also add one more top-level field named probe with the string value unconstrained."; +const JSON_OBJECT_COMPATIBILITY_USER_PROMPT: &str = "This is an XGENy connectivity probe with no planning context. Return exactly a completion_candidate: set formatVersion to 1, kind to completion_candidate, steps to an empty array, and summary to the string ok. Do not add any other fields."; + +/// Explicit provider output dialect; local proposal validation is identical in both modes. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ResponseFormat { + /// Ask the provider to enforce the production schema (legacy default). + #[default] + JsonSchema, + /// Ask for JSON syntax only; include the schema in the committed system prompt. + JsonObject, +} + +/// Opt-in vendor thinking extension, never inferred from an endpoint hostname. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum ThinkingMode { + /// Do not send thinking parameters; preserve existing request semantics. + #[default] + Default, + /// Explicit `thinking.type=disabled` for compatible providers. + Disabled, + /// Explicit thinking with low reasoning effort and the existing output/time limits. + Enabled, +} /// A bearer credential retained only as a sensitive HTTP header value. #[derive(Clone)] @@ -105,6 +132,8 @@ pub struct OpenAiPlannerConfig { max_proposal_bytes: usize, max_json_depth: usize, proposal_schema: Value, + response_format: ResponseFormat, + thinking: ThinkingMode, planning_constraints_required: bool, request_profile_digest: String, } @@ -146,6 +175,8 @@ impl OpenAiPlannerConfig { max_proposal_bytes: DEFAULT_MAX_PROPOSAL_BYTES, max_json_depth: DEFAULT_MAX_JSON_DEPTH, proposal_schema: proposal_schema(), + response_format: ResponseFormat::default(), + thinking: ThinkingMode::default(), planning_constraints_required: false, request_profile_digest: String::new(), }; @@ -153,6 +184,44 @@ impl OpenAiPlannerConfig { Ok(config) } + /// Return a copy with an explicit output dialect and a matching request commitment. + /// + /// # Errors + /// Returns an error if the updated request-profile digest cannot be constructed. + pub fn with_response_format( + mut self, + response_format: ResponseFormat, + ) -> Result { + self.response_format = response_format; + self.refresh_profile_digest()?; + Ok(self) + } + + /// Return a copy with an explicit thinking extension; no endpoint inference or fallback. + /// + /// # Errors + /// Returns an error if the updated request-profile digest cannot be constructed. + pub fn with_thinking( + mut self, + thinking: ThinkingMode, + ) -> Result { + self.thinking = thinking; + self.refresh_profile_digest()?; + Ok(self) + } + + /// The committed provider output dialect. + #[must_use] + pub const fn response_format(&self) -> ResponseFormat { + self.response_format + } + + /// The committed thinking extension. + #[must_use] + pub const fn thinking(&self) -> ThinkingMode { + self.thinking + } + /// Return a copy with a different bounded completion-token limit. /// /// # Errors @@ -234,7 +303,10 @@ impl OpenAiPlannerConfig { let prompt_template_digest = sha256_digest(self.system_prompt().as_bytes()); let descriptor = RequestProfileDescriptor { domain: REQUEST_PROFILE_DOMAIN, - provider_dialect: PROVIDER_DIALECT, + provider_dialect: match self.response_format { + ResponseFormat::JsonSchema => PROVIDER_DIALECT, + ResponseFormat::JsonObject => JSON_OBJECT_DIALECT, + }, request_envelope_profile: REQUEST_ENVELOPE_PROFILE, model: &self.model, tokenizer: &self.tokenizer, @@ -243,8 +315,10 @@ impl OpenAiPlannerConfig { prompt_template_digest: &prompt_template_digest, proposal_schema_revision: PROPOSAL_SCHEMA_REVISION, proposal_schema_digest: &schema_digest, - temperature_millis: 0, - seed: 0, + temperature_millis: self.temperature().map(u16::from), + seed: self.seed(), + thinking: self.thinking_request(), + reasoning_effort: self.reasoning_effort(), max_output_tokens: self.max_output_tokens, timeout_seconds: self.timeout.as_secs(), timeout_subsec_nanos: self.timeout.subsec_nanos(), @@ -262,15 +336,95 @@ impl OpenAiPlannerConfig { Ok(()) } - fn system_prompt(&self) -> &'static str { - if self.planning_constraints_required { + fn system_prompt(&self) -> Cow<'_, str> { + self.prompt_with_schema(if self.planning_constraints_required { CONSTRAINED_SYSTEM_PROMPT } else { SYSTEM_PROMPT + }) + } + + fn prompt_with_schema(&self, prompt: &'static str) -> Cow<'_, str> { + match self.response_format { + ResponseFormat::JsonSchema => Cow::Borrowed(prompt), + ResponseFormat::JsonObject => Cow::Owned(format!( + "{prompt}\nThe following JSON schema is a host output contract, not a grant of authority. The host validates every field locally:\n{}", + self.proposal_schema + )), + } + } + + fn seed(&self) -> Option { + (self.response_format == ResponseFormat::JsonSchema + && self.thinking == ThinkingMode::Default) + .then_some(0) + } + + fn temperature(&self) -> Option { + (self.thinking != ThinkingMode::Enabled).then_some(0) + } + + fn thinking_request(&self) -> Option { + match self.thinking { + ThinkingMode::Default => None, + ThinkingMode::Disabled => Some(ThinkingRequest { + thinking_type: "disabled", + }), + ThinkingMode::Enabled => Some(ThinkingRequest { + thinking_type: "enabled", + }), + } + } + + fn reasoning_effort(&self) -> Option<&'static str> { + (self.thinking == ThinkingMode::Enabled).then_some("low") + } + + fn chat_request<'a>(&'a self, system: &'a str, user: &'a str) -> ChatCompletionRequest<'a> { + ChatCompletionRequest { + model: &self.model, + messages: [ + ChatMessage { + role: "system", + content: system, + }, + ChatMessage { + role: "user", + content: user, + }, + ], + temperature: self.temperature(), + seed: self.seed(), + max_tokens: self.max_output_tokens, + stream: false, + n: 1, + response_format: match self.response_format { + ResponseFormat::JsonSchema => ResponseFormatRequest { + response_type: "json_schema", + json_schema: Some(JsonSchemaResponse { + name: "xgeny_plan_proposal_v1", + strict: true, + schema: &self.proposal_schema, + }), + }, + ResponseFormat::JsonObject => ResponseFormatRequest { + response_type: "json_object", + json_schema: None, + }, + }, + thinking: self.thinking_request(), + reasoning_effort: self.reasoning_effort(), } } fn prompt_template_revision(&self) -> &'static str { + if self.response_format == ResponseFormat::JsonObject { + return if self.planning_constraints_required { + "xgeny.openai-planner-prompt/v5-json-object-constrained" + } else { + "xgeny.openai-planner-prompt/v5-json-object" + }; + } if self.planning_constraints_required { CONSTRAINED_PROMPT_TEMPLATE_REVISION } else { @@ -295,6 +449,8 @@ impl fmt::Debug for OpenAiPlannerConfig { .field("max_proposal_bytes", &self.max_proposal_bytes) .field("max_json_depth", &self.max_json_depth) .field("proposal_schema", &"") + .field("response_format", &self.response_format) + .field("thinking", &self.thinking) .field( "planning_constraints_required", &self.planning_constraints_required, @@ -429,7 +585,7 @@ impl OpenAiModelChecker { } } -/// One explicit Chat Completions and strict JSON Schema compatibility probe. +/// One explicit Chat Completions probe for the selected dialect and local proposal validation. pub struct OpenAiCompatibilityChecker { config: OpenAiPlannerConfig, credential: Option, @@ -473,45 +629,27 @@ impl OpenAiCompatibilityChecker { }) } - /// Send one non-streaming Chat Completions request and validate strict JSON Schema behavior. + /// Send one non-streaming request and validate the selected output dialect. /// /// This probe deliberately has no local workspace or Run state. It sends the byte-identical - /// production proposal schema and validates the answer with the production document rules, so - /// a provider that accepts a trivial schema but cannot enforce the real one fails here instead - /// of at the first planner call. It verifies the endpoint, selected model, Chat Completions - /// envelope, strict `json_schema` enforcement, and exact response-model identity. + /// production proposal schema and validates the answer with the production document rules. + /// The default `json_schema` probe also challenges provider-side schema enforcement; + /// `json_object` verifies only JSON output and local conformance, not server enforcement. + /// Both verify the endpoint, selected model, envelope, and exact response-model identity. /// /// # Errors /// /// Returns only a fixed redacted failure class. Provider response bodies are never exposed. pub fn check(&mut self) -> Result<(), OpenAiCompatibilityCheckFailure> { - let body = serde_json::to_vec(&ChatCompletionRequest { - model: &self.config.model, - messages: [ - ChatMessage { - role: "system", - content: COMPATIBILITY_SYSTEM_PROMPT, - }, - ChatMessage { - role: "user", - content: COMPATIBILITY_USER_PROMPT, - }, - ], - temperature: 0, - seed: 0, - max_tokens: self.config.max_output_tokens, - stream: false, - n: 1, - response_format: ResponseFormat { - response_type: "json_schema", - json_schema: JsonSchemaResponse { - name: "xgeny_plan_proposal_v1", - strict: true, - schema: &self.config.proposal_schema, - }, - }, - }) - .map_err(|_| OpenAiCompatibilityCheckFailure::InvalidResponse)?; + let system = self.config.prompt_with_schema(COMPATIBILITY_SYSTEM_PROMPT); + // JSON-object APIs promise syntax, not provider-side schema enforcement. + // Asking them to emit an extra field would deliberately fail this probe. + let user = match self.config.response_format { + ResponseFormat::JsonSchema => COMPATIBILITY_USER_PROMPT, + ResponseFormat::JsonObject => JSON_OBJECT_COMPATIBILITY_USER_PROMPT, + }; + let body = serde_json::to_vec(&self.config.chat_request(&system, user)) + .map_err(|_| OpenAiCompatibilityCheckFailure::InvalidResponse)?; if body.len() > self.config.max_request_bytes { return Err(OpenAiCompatibilityCheckFailure::InvalidResponse); } @@ -635,33 +773,9 @@ impl PlannerPort for OpenAiPlanner { planning_context: request.context(), }) .map_err(|_| PlannerPortFailure::ProviderLimit)?; - let body = serde_json::to_vec(&ChatCompletionRequest { - model: &self.config.model, - messages: [ - ChatMessage { - role: "system", - content: self.config.system_prompt(), - }, - ChatMessage { - role: "user", - content: &prompt, - }, - ], - temperature: 0, - seed: 0, - max_tokens: self.config.max_output_tokens, - stream: false, - n: 1, - response_format: ResponseFormat { - response_type: "json_schema", - json_schema: JsonSchemaResponse { - name: "xgeny_plan_proposal_v1", - strict: true, - schema: &self.config.proposal_schema, - }, - }, - }) - .map_err(|_| PlannerPortFailure::ProviderLimit)?; + let system = self.config.system_prompt(); + let body = serde_json::to_vec(&self.config.chat_request(&system, &prompt)) + .map_err(|_| PlannerPortFailure::ProviderLimit)?; if body.len() > self.config.max_request_bytes { return Err(PlannerPortFailure::ProviderLimit); } @@ -715,8 +829,14 @@ struct RequestProfileDescriptor<'a> { prompt_template_digest: &'a str, proposal_schema_revision: &'static str, proposal_schema_digest: &'a str, - temperature_millis: u16, - seed: u64, + #[serde(skip_serializing_if = "Option::is_none")] + temperature_millis: Option, + #[serde(skip_serializing_if = "Option::is_none")] + seed: Option, + #[serde(skip_serializing_if = "Option::is_none")] + thinking: Option, + #[serde(skip_serializing_if = "Option::is_none")] + reasoning_effort: Option<&'static str>, max_output_tokens: u32, timeout_seconds: u64, timeout_subsec_nanos: u32, @@ -744,12 +864,18 @@ struct PlannerPrompt<'a> { struct ChatCompletionRequest<'a> { model: &'a str, messages: [ChatMessage<'a>; 2], - temperature: u8, - seed: u64, + #[serde(skip_serializing_if = "Option::is_none")] + temperature: Option, + #[serde(skip_serializing_if = "Option::is_none")] + seed: Option, max_tokens: u32, stream: bool, n: u8, - response_format: ResponseFormat<'a>, + response_format: ResponseFormatRequest<'a>, + #[serde(skip_serializing_if = "Option::is_none")] + thinking: Option, + #[serde(skip_serializing_if = "Option::is_none")] + reasoning_effort: Option<&'static str>, } #[derive(Serialize)] @@ -759,10 +885,17 @@ struct ChatMessage<'a> { } #[derive(Serialize)] -struct ResponseFormat<'a> { +struct ResponseFormatRequest<'a> { #[serde(rename = "type")] response_type: &'static str, - json_schema: JsonSchemaResponse<'a>, + #[serde(skip_serializing_if = "Option::is_none")] + json_schema: Option>, +} + +#[derive(Serialize)] +struct ThinkingRequest { + #[serde(rename = "type")] + thinking_type: &'static str, } #[derive(Serialize)] @@ -1546,6 +1679,228 @@ mod tests { .to_string() } + fn request_body(config: &OpenAiPlannerConfig) -> Value { + let system = config.system_prompt(); + serde_json::to_value(config.chat_request(&system, "held-out task context")).unwrap() + } + + #[test] + fn explicit_default_options_preserve_legacy_request_bytes_and_digest() { + let original = config("https://provider.example/v1"); + let explicit = config("https://provider.example/v1") + .with_response_format(ResponseFormat::JsonSchema) + .unwrap() + .with_thinking(ThinkingMode::Default) + .unwrap(); + let prompt = original.system_prompt(); + let expected = json!({ + "model": MODEL, + "messages": [{"role":"system", "content":SYSTEM_PROMPT}, {"role":"user", "content":"task"}], + "temperature":0,"seed":0,"max_tokens":DEFAULT_MAX_OUTPUT_TOKENS, + "stream":false,"n":1, + "response_format":{"type":"json_schema","json_schema":{"name":"xgeny_plan_proposal_v1","strict":true,"schema":proposal_schema()}} + }); + assert_eq!( + serde_json::to_value(original.chat_request(&prompt, "task")).unwrap(), + expected + ); + assert_eq!( + original.request_profile_digest(), + explicit.request_profile_digest() + ); + assert_eq!( + original.request_profile_digest(), + "sha256:be4331e9fe9c0e2645f99aa5e0e3987a946c887cb142e53586a2b1451f2bf7e9" + ); + assert_eq!( + serde_json::to_vec(&original.chat_request(&prompt, "task")).unwrap(), + serde_json::to_vec(&explicit.chat_request(&explicit.system_prompt(), "task")).unwrap() + ); + } + + #[test] + fn json_object_commits_schema_prompt_and_omits_unsupported_schema_and_seed() { + let profile = config("https://provider.example/v1") + .with_response_format(ResponseFormat::JsonObject) + .unwrap(); + let body = request_body(&profile); + assert_eq!(body["response_format"], json!({"type":"json_object"})); + assert!(body.get("seed").is_none()); + assert!(body.get("thinking").is_none()); + assert_eq!(body["temperature"], 0); + let system = body["messages"][0]["content"].as_str().unwrap(); + assert!(system.starts_with(SYSTEM_PROMPT)); + assert!(system.ends_with(&proposal_schema().to_string())); + assert_ne!( + profile.request_profile_digest(), + config("https://provider.example/v1").request_profile_digest() + ); + assert_eq!( + profile.request_profile_digest(), + config("http://127.0.0.1:9988/v1") + .with_response_format(ResponseFormat::JsonObject) + .unwrap() + .request_profile_digest() + ); + let constrained = profile.with_planning_constraints_required().unwrap(); + assert!( + constrained + .system_prompt() + .starts_with(CONSTRAINED_SYSTEM_PROMPT) + ); + assert!( + constrained + .system_prompt() + .ends_with(&proposal_schema().to_string()) + ); + assert_ne!( + constrained.request_profile_digest(), + config("https://provider.example/v1") + .with_response_format(ResponseFormat::JsonObject) + .unwrap() + .request_profile_digest() + ); + } + + #[test] + fn explicit_thinking_modes_are_committed_bounded_vendor_options() { + let base = || { + config("https://provider.example/v1") + .with_response_format(ResponseFormat::JsonObject) + .unwrap() + }; + let disabled = base().with_thinking(ThinkingMode::Disabled).unwrap(); + let enabled = base().with_thinking(ThinkingMode::Enabled).unwrap(); + let fast_body = request_body(&disabled); + assert_eq!(fast_body["thinking"], json!({"type":"disabled"})); + assert!(fast_body.get("reasoning_effort").is_none()); + let thinking_body = request_body(&enabled); + assert_eq!(thinking_body["thinking"], json!({"type":"enabled"})); + assert_eq!(thinking_body["reasoning_effort"], "low"); + assert!(thinking_body.get("seed").is_none()); + assert!(thinking_body.get("temperature").is_none()); + assert_eq!(thinking_body["max_tokens"], DEFAULT_MAX_OUTPUT_TOKENS); + assert_eq!(thinking_body["stream"], false); + let digests = [ + base().request_profile_digest().to_owned(), + disabled.request_profile_digest().to_owned(), + enabled.request_profile_digest().to_owned(), + ]; + assert_eq!( + digests + .iter() + .collect::>() + .len(), + 3 + ); + let reset = enabled + .with_response_format(ResponseFormat::JsonSchema) + .unwrap() + .with_thinking(ThinkingMode::Default) + .unwrap(); + assert_eq!( + reset.request_profile_digest(), + config("https://provider.example/v1").request_profile_digest() + ); + } + + #[test] + fn output_options_have_explicit_serialized_names_and_reject_unknown_values() { + assert_eq!( + serde_json::to_value(ResponseFormat::JsonSchema).unwrap(), + "json_schema" + ); + assert_eq!( + serde_json::from_str::("\"json_object\"").unwrap(), + ResponseFormat::JsonObject + ); + for (name, expected) in [ + ("default", ThinkingMode::Default), + ("disabled", ThinkingMode::Disabled), + ("enabled", ThinkingMode::Enabled), + ] { + assert_eq!( + serde_json::from_value::(json!(name)).unwrap(), + expected + ); + assert_eq!(serde_json::to_value(expected).unwrap(), name); + } + assert!(serde_json::from_value::(json!("auto")).is_err()); + assert!(serde_json::from_value::(json!("auto")).is_err()); + } + + struct JsonObjectProbeTransport { + response: Vec, + } + + impl Transport for JsonObjectProbeTransport { + fn send(&mut self, request: TransportRequest<'_>) -> Result, PlannerPortFailure> { + let body: Value = serde_json::from_slice(request.body).unwrap(); + assert_eq!(body["response_format"], json!({"type":"json_object"})); + assert_eq!(body["thinking"], json!({"type":"disabled"})); + assert_eq!( + body["messages"][1]["content"], + JSON_OBJECT_COMPATIBILITY_USER_PROMPT + ); + assert!( + !body["messages"][1]["content"] + .as_str() + .unwrap() + .contains("field named probe") + ); + assert!( + body["messages"][0]["content"] + .as_str() + .unwrap() + .ends_with(&proposal_schema().to_string()) + ); + assert!(!self.response.is_empty(), "a request must never be retried"); + Ok(std::mem::take(&mut self.response)) + } + } + + #[test] + fn json_object_probe_uses_local_validation_without_server_enforcement_claim() { + for (content, finish, expected) in [ + (COMPLETION_OK, "stop", Ok(())), + ( + r#"{"formatVersion":1,"kind":"completion_candidate","steps":[],"summary":"별도 과제 결과"}"#, + "stop", + Ok(()), + ), + ( + r#"{"formatVersion":1,"kind":"completion_candidate","steps":[],"summary":"ok","unknown":true}"#, + "stop", + Err(OpenAiCompatibilityCheckFailure::InvalidResponse), + ), + ( + "```json\n{}\n```", + "stop", + Err(OpenAiCompatibilityCheckFailure::InvalidResponse), + ), + ( + COMPLETION_OK, + "length", + Err(OpenAiCompatibilityCheckFailure::OutputTruncated), + ), + ] { + let profile = config("https://provider.example/v1") + .with_response_format(ResponseFormat::JsonObject) + .unwrap() + .with_thinking(ThinkingMode::Disabled) + .unwrap(); + let mut checker = OpenAiCompatibilityChecker::with_transport( + profile, + None, + JsonObjectProbeTransport { + response: response(content, finish), + }, + ) + .unwrap(); + assert_eq!(checker.check(), expected); + } + } + fn read_complete_test_request(stream: &mut TcpStream) { let mut request = Vec::new(); let mut chunk = [0_u8; 4096]; diff --git a/crates/xgeny-provider-openai/tests/http_contract.rs b/crates/xgeny-provider-openai/tests/http_contract.rs index fcd6eae..6ebb144 100644 --- a/crates/xgeny-provider-openai/tests/http_contract.rs +++ b/crates/xgeny-provider-openai/tests/http_contract.rs @@ -12,7 +12,7 @@ use xgeny_local_store::{ Commit, ExpectedHead, MemoryRunStore, RunPlanningSnapshot, RunSnapshot, RunStore, StoreError, }; use xgeny_policy::{ResourceResolutionFailure, ResourceResolver}; -use xgeny_provider_openai::{OpenAiPlanner, OpenAiPlannerConfig}; +use xgeny_provider_openai::{OpenAiPlanner, OpenAiPlannerConfig, ResponseFormat, ThinkingMode}; use xgeny_runtime::{ AgentLoop, AgentLoopTick, CapabilityRegistry, EventFactory, EventFactoryError, EventMetadata, PlanMaterializationRequest, PlanMaterializer, PlanMaterializerFailure, PlannerPortFailure, @@ -477,6 +477,144 @@ fn planner(base_url: &str) -> OpenAiPlanner { OpenAiPlanner::new(config, None).expect("planner should build") } +fn json_object_planner(base_url: &str, timeout: Duration) -> OpenAiPlanner { + let config = OpenAiPlannerConfig::new( + base_url, + "xgeny.test.json-object", + "qwen3.8-27b", + "test-tokenizer", + ) + .unwrap() + .with_response_format(ResponseFormat::JsonObject) + .unwrap() + .with_thinking(ThinkingMode::Disabled) + .unwrap() + .with_max_output_tokens(512) + .unwrap() + .with_timeout(timeout) + .unwrap(); + OpenAiPlanner::new(config, None).unwrap() +} + +#[test] +fn json_object_native_calls_still_reserve_validate_and_settle_exactly_once() { + let valid = json!({ + "formatVersion":1,"kind":"plan","steps":[{ + "key":"record_sample","objective":"Record a held-out sample path","dependsOn":[], + "capability":{"capabilityId":"xgeny.test/record-path","contractVersion":"1.0.0"}, + "arguments":{"path":"/workspace/sample.csv"} + }],"summary":"" + }); + let mut invalid = valid.clone(); + invalid["untrusted_extra"] = json!(true); + for (content, finish, expected_failure) in [ + (valid.clone(), "stop", None), + (invalid, "stop", Some(PlannerPortFailure::InvalidResponse)), + (valid, "length", Some(PlannerPortFailure::ProviderLimit)), + ] { + let mut envelope: Value = serde_json::from_slice(&provider_response(&content)).unwrap(); + envelope["choices"][0]["finish_reason"] = json!(finish); + envelope["choices"][0]["message"]["reasoning_content"] = json!(RAW_RESPONSE_SENTINEL); + let server = TestServer::spawn("200 OK", serde_json::to_vec(&envelope).unwrap()); + let mut planner = json_object_planner(&server.base_url, Duration::from_secs(3)); + let mut store = seed_store(); + let loop_runtime = configured_loop(&mut store, &mut planner); + let tick = loop_runtime + .tick( + &mut store, + &mut DeterministicEvents, + &FixedLease, + &synthetic_registry(), + &IdentityResolver::default(), + &mut planner, + &mut EphemeralMaterializer, + ) + .unwrap(); + match expected_failure { + None => assert!(matches!(tick, AgentLoopTick::PlanAccepted { .. })), + Some(expected) => assert!( + matches!(tick, AgentLoopTick::PlannerUnavailable { failure, .. } if failure == expected) + ), + } + let request = server.finish(); + let offset = find_header_end(&request).unwrap() + 4; + let body: Value = serde_json::from_slice(&request[offset..]).unwrap(); + assert_eq!(body["response_format"], json!({"type":"json_object"})); + assert_eq!(body["thinking"], json!({"type":"disabled"})); + assert!(body.get("seed").is_none()); + assert!( + body["messages"][0]["content"] + .as_str() + .unwrap() + .contains("additionalProperties") + ); + let snapshot = store.load().unwrap().unwrap(); + let lifecycle = snapshot + .state + .agent_loop + .as_ref() + .unwrap() + .model_calls + .as_ref() + .unwrap(); + assert_eq!(lifecycle.reserved_calls, 1); + assert_eq!(lifecycle.settled_calls, 1); + assert_eq!(lifecycle.unknown_calls, 0); + assert!(lifecycle.active_call.is_none()); + assert!( + !serde_json::to_string(&snapshot.records) + .unwrap() + .contains(RAW_RESPONSE_SENTINEL) + ); + } +} + +#[test] +fn json_object_timeout_retains_unknown_call_without_automatic_replay() { + let (base_url, handle) = spawn_stalling_server(Duration::from_millis(150)); + let mut planner = json_object_planner(&base_url, Duration::from_millis(40)); + let mut store = seed_store(); + let loop_runtime = configured_loop(&mut store, &mut planner); + let tick = loop_runtime + .tick( + &mut store, + &mut DeterministicEvents, + &FixedLease, + &synthetic_registry(), + &IdentityResolver::default(), + &mut planner, + &mut EphemeralMaterializer, + ) + .unwrap(); + assert!(matches!( + tick, + AgentLoopTick::PlannerUnavailable { + failure: PlannerPortFailure::Timeout, + .. + } + )); + let snapshot = store.load().unwrap().unwrap(); + assert!(matches!( + snapshot.records.last().unwrap().event.body, + RunEventBody::ModelCallBecameUnknown { + reason: ModelCallUnknownReason::Timeout, + .. + } + )); + let lifecycle = snapshot + .state + .agent_loop + .as_ref() + .unwrap() + .model_calls + .as_ref() + .unwrap(); + assert_eq!(lifecycle.reserved_calls, 1); + assert_eq!(lifecycle.unknown_calls, 1); + assert_eq!(lifecycle.settled_calls, 0); + handle.join().unwrap(); +} + fn assert_strict_request_contract(request: &[u8]) -> Value { let header_end = find_header_end(request).expect("request headers should exist"); let header = std::str::from_utf8(&request[..header_end]).unwrap(); diff --git a/docs/adr/0042-explicit-provider-output-dialects.md b/docs/adr/0042-explicit-provider-output-dialects.md new file mode 100644 index 0000000..4efa321 --- /dev/null +++ b/docs/adr/0042-explicit-provider-output-dialects.md @@ -0,0 +1,62 @@ +# ADR-0042: Explicit provider output dialects and thinking options + +- Date: 2026-09-21 +- Status: Proposed +- Extends: ADR-0017, ADR-0032 +- Protocol / journal / SQLite schema changes: none + +## Decision + +Keep model access inside the native OpenAI-compatible planner adapter. A provider +accepting Chat Completions does not necessarily support server-enforced JSON +Schema. Add explicit immutable `ResponseFormat` (`json_schema`, `json_object`) +and `ThinkingMode` (`default`, `disabled`, `enabled`) request settings. Do not +infer capabilities from endpoint names or model names, automatically retry a +different dialect, or silently switch models. + +The default remains `json_schema` with `default` thinking. Its wire body and +request-profile digest remain unchanged, including temperature 0 and seed 0, so +existing journal commitments and resume checks remain valid. + +`json_object` requests JSON syntax and includes the entire production proposal +schema in the committed system prompt. It omits the schema-only wire fields and +seed. The host still runs the identical strict duplicate-key, unknown-field, +model-identity, size, depth, proposal, capability, and policy validation. JSON +syntax support does **not** mean provider-side schema enforcement or permission +to execute a model suggestion. + +Thinking extensions are operator opt-ins for providers implementing them: + +- `default`: no extra thinking parameters. +- `disabled`: `thinking: {"type":"disabled"}`, temperature 0, no seed. +- `enabled`: `thinking: {"type":"enabled"}`, `reasoning_effort: "low"`, no seed + or temperature. Existing completion-token, timeout, response-size and call + budgets remain unchanged; this is not an unlimited reasoning mode. + +Every changed prompt, response dialect, sampling omission, and thinking option +is committed in the request-profile digest. Model profile persistence and CLI +resume must restore these options before verifying the committed digest. +Credentials and endpoint locations remain outside the digest and logs. + +## Compatibility checks and lifecycle + +The JSON Schema compatibility probe retains its adversarial extra-field request +to detect providers ignoring strict schema. JSON Object's probe instead asks for +a conforming completion and includes the schema in its prompt. Success means +one locally valid answer, **not** a claim that the server enforces the schema. + +Both modes use the same native reservation, result validation, settlement and +Unknown lifecycle. Truncation does not accept a partial proposal; a timed-out +call remains Unknown and is not automatically retried. Raw model reasoning is +not exposed or retained as a tool result. No streaming, browser, network tool, +new host authority, or parallel direct-HTTP application runtime is introduced. + +## Verification + +Offline tests cover legacy golden commitments, explicit default wire identity, +JSON Object and both thinking options, distinct/restorable commitments, unknown +enum rejection, constrained prompts, valid and invalid compatibility answers, +and truncation. Loopback HTTP tests exercise native plan acceptance, invalid +proposal rejection, output limits, exact single-call settlement, reasoning +redaction and timeout-to-Unknown behavior. These are domain-independent cases; +no contest, customer dataset or paid endpoint is required. diff --git a/docs/adr/0043-provider-wire-profile-resolution.md b/docs/adr/0043-provider-wire-profile-resolution.md new file mode 100644 index 0000000..9485a29 --- /dev/null +++ b/docs/adr/0043-provider-wire-profile-resolution.md @@ -0,0 +1,64 @@ +# ADR-0043: Provider wire options in CLI profiles and restart resolution + +- Status: Accepted +- Date: 2026-09-21 +- Scope: OpenAI-compatible provider options, CLI profiles, restart validation + +## Decision + +CLI integration of the provider modes defined in ADR-0042. + +Keep all planning inside the existing XGENy provider → validated proposal → journal → capability +flow. Do not bypass the harness or infer behavior from an endpoint hostname or a model ID. + +`model setup`, `model check`, `run`, and `resume` accept two non-secret options: + +- `--response-format json_schema|json_object` (default `json_schema`) +- `--thinking default|disabled|enabled` (default `default`, which omits the provider extension) + +Resolve each field in order: explicit option, `XGENY_OPENAI_RESPONSE_FORMAT` / +`XGENY_OPENAI_THINKING`, selected profile, default. Invalid values fail closed; there is no automatic +fallback after provider rejection. Profiles persist `responseFormat` and `thinking` with serde defaults, +so pre-existing format-v1 files still load with the original wire behavior and request digest. + +JSON Object mode is **host-validated**, not proof that the server enforces a JSON Schema. It carries +the proposal schema in the request and retains the exact same host envelope, proposal, invocation, +policy and capability validation. Setup/check output states the weaker server guarantee explicitly. +Explicit thinking options are provider extensions, not a claim that every compatible server supports +them. `enabled` requests low reasoning effort and omits temperature and seed; `disabled` omits seed. +The default does not introduce any of these extensions. No reasoning content is shown as progress. + +## Restart and credentials + +Wire options and inference limits are inputs to the committed request profile digest. The manifest +remains the authority; profile settings cannot silently change an existing Run. Incomplete resume +resolves the currently selected profile plus explicit/environment overrides and compares the digest +before calling the model. A mismatch fails without a retry or relaxed-validation mode. The deferred +resolver now includes profile inference limits instead of silently replacing them with defaults. +Completed replay needs neither profile resolution nor credential/model access. + +Credential precedence, exact-URL secure-store matching, HTTPS/loopback restrictions, budget limits, +model-call recovery, and capability permissions are unchanged. No token is stored in profile JSON. + +## Usage + +Use an ephemeral environment credential or `--token-stdin`, never an argument containing a key: + +```sh +xgeny model setup --name fast --base-url https://api.deepseek.com/v1 \ + --model deepseek-chat --response-format json_object --thinking disabled +xgeny model check --profile fast --compatibility +xgeny run --profile fast --allow-dir . --allow-remote-model-egress 'Inspect this workspace' +xgeny resume RUN_ID --profile fast --allow-dir . --allow-remote-model-egress +``` + +The endpoint/model above is an example, not a default or a tested live-service claim. A caller must +select an advertised model and validate its compatibility. Catalog checks are one GET, optional +compatibility is one POST, and no automatic paid retries are added. + +## Verification + +Offline tests cover old profiles, enum rejection, precedence, both response transports, two thinking +settings, setup/run body parity, unchanged production/probe digests, resume with non-default limits, +and changed-profile rejection. Existing completed-replay tests protect the no-model-access path. +Local HTTP fixtures are not live DeepSeek quality or availability evidence.