From 87df5668d609b81b42c5bb3a9908e33e55cd8d54 Mon Sep 17 00:00:00 2001 From: DJConnect Date: Sat, 12 Sep 2026 15:24:05 +0200 Subject: [PATCH] docs: design task and role based model policy under SA-ROLE Specify EP-owned provider/model/effort selection, qualified task profiles, immutable effective-policy binding, fallback and recovery boundaries, requested/observed evidence and Operations Console administration. Add a seven-node non-executable RMP decomposition and sixteen future qualification scenarios with offline documentary guards. NO_BUMP; no runtime model activation, credentials, production code, workflow, canary or PR175 change. --- .../POLICY_GOVERNANCE_V1_ROADMAP.md | 18 + .../ROLE_TASK_MODEL_POLICY_V1_DAG.json | 328 ++++++++++++++++++ .../ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md | 97 ++++++ .../SUBAGENT_ORCHESTRATION_V1_DAG.json | 13 +- .../SUBAGENT_ORCHESTRATION_V1_ROADMAP.md | 19 + docs/engineering/ROLE_TASK_MODEL_POLICY_V1.md | 248 +++++++++++++ tests/test_role_task_model_policy_design.py | 141 ++++++++ 7 files changed, 863 insertions(+), 1 deletion(-) create mode 100644 docs/development/ROLE_TASK_MODEL_POLICY_V1_DAG.json create mode 100644 docs/development/ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md create mode 100644 docs/engineering/ROLE_TASK_MODEL_POLICY_V1.md create mode 100644 tests/test_role_task_model_policy_design.py diff --git a/docs/development/POLICY_GOVERNANCE_V1_ROADMAP.md b/docs/development/POLICY_GOVERNANCE_V1_ROADMAP.md index 723a51b6..f483b8da 100644 --- a/docs/development/POLICY_GOVERNANCE_V1_ROADMAP.md +++ b/docs/development/POLICY_GOVERNANCE_V1_ROADMAP.md @@ -65,3 +65,21 @@ GP-E/GP-Q/GP-X are PLANNED; a Workspace decision is not a substitute for the external target's gate. No duplicate workflow, live policy or executable DAG is introduced. The current Action assurance and bounded repair requirements remain intact and distinct from post-Action human review. + +## Task/role model allocation — existing SA-ROLE consumer + +The [role/task model policy design](../engineering/ROLE_TASK_MODEL_POLICY_V1.md) +and [scoped roadmap/DAG](ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md) define the +EP_ROLE_TASK_MODEL_POLICY_V1 family under SA-ROLE. Its RMP-EFFECTIVE-POLICY gate +requires only the qualified assignment/activation/snapshot subset owned by POL-E; +it does not duplicate those services or make the entire POL-B/POL-Q/Workspace +programme a prerequisite. Existing compatible implementations can satisfy that +evidence gate after inspection; a policy name or documentation alone cannot. + +Task/role/risk preferences never override admission requirements or current host +timeout ceilings. Per-invocation model/effort selection and approved alternatives +are bound to the admitted effective profile. Console edits affect new runs unless +an explicit owning migration is qualified. Model/provider changes do not reset +three-round run/continuation consumption, grant authority or silently select a +metered API. Concrete binding activation, representative model evaluation and +later provider adapters remain separately governed implementation work. diff --git a/docs/development/ROLE_TASK_MODEL_POLICY_V1_DAG.json b/docs/development/ROLE_TASK_MODEL_POLICY_V1_DAG.json new file mode 100644 index 00000000..878be99f --- /dev/null +++ b/docs/development/ROLE_TASK_MODEL_POLICY_V1_DAG.json @@ -0,0 +1,328 @@ +{ + "schema_version": 1, + "increment": "EP_ROLE_TASK_MODEL_POLICY_V1", + "owner": "engineering-platform", + "parent_graph": "docs/development/SUBAGENT_ORCHESTRATION_V1_DAG.json", + "parent_node": "SA-ROLE", + "finding": "SA-F09", + "status": "PLANNED", + "documentary_only": true, + "execution_authority": false, + "automatic_dispatch": false, + "version_change": "NO_BUMP", + "first_canary_prerequisite": false, + "source_pin": "4a6dace73ccc15146a9afe5db356957d8af4bbbb", + "architecture": "docs/engineering/ROLE_TASK_MODEL_POLICY_V1.md", + "roadmap": "docs/development/ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md", + "policy_roadmap": "docs/development/POLICY_GOVERNANCE_V1_ROADMAP.md", + "first_delivery": "QUALIFIED_EXISTING_EP_MANAGED_CODEX_MULTIMODEL", + "later_provider_families": [ + "OTHER_QUALIFIED_CLI", + "METERED_API_EXPLICIT_OPT_IN", + "LOCAL_MODEL_SERVICE_QUALIFIED_ADAPTER" + ], + "actual_model_ids_selected": [], + "runtime_activation": false, + "invariants": { + "ep_owns_execution_policy": true, + "forge_planning_policy_unchanged": true, + "reuse_effective_profiles_and_central": true, + "resolve_deterministically": true, + "role_selection_is_model_selection": false, + "automatic_api_fallback": false, + "provider_default_requires_explicit_policy": true, + "requested_is_observed": false, + "missing_usage_is_zero": false, + "unknown_observed_model_can_satisfy_exact_requirement": false, + "per_invocation_settings": true, + "quality_security_separate_invocations": true, + "different_models_required_for_independence": false, + "allow_ambiguous_fallback": false, + "allow_review_shopping": false, + "allow_repair_budget_reset": false, + "current_runwide_repair_limit": 3, + "current_timeout_constants_editable": false, + "live_profile_edits_change_active_runs": false, + "page_refresh_generates": false, + "model_switch_grants_authority": false, + "qualified_binding_required_before_activation": true, + "metadata_redefines_terminal_v12_without_versioning": false, + "all_optional_providers_required_for_v1": false + }, + "external_requirements": [ + { + "id": "RMP-EFFECTIVE-POLICY", + "owner": "engineering-platform", + "kind": "qualified_subset_evidence", + "status": "REQUIRED_EVIDENCE_UNVERIFIED", + "owning_lane": "POL-E", + "requirement": "Reuse qualified assignment/activation/effective-profile snapshot services required by this family, not all POL-B/POL-Q/Workspace work or a second policy engine.", + "evidence": [] + } + ], + "external_dependencies": { + "RMP-RESOLVE": [ + "SA-CTX", + "SA-OBS", + "RMP-EFFECTIVE-POLICY" + ] + }, + "sibling_joins": { + "role_selection": "SA-SEL", + "findings_consumption": "SA-LOOP", + "deterministic_validation": "SA-VAL", + "deterministic_publication": "SA-PUB", + "parallel_mandatory_assurance": "SA-PAR" + }, + "task_matrix": [ + { + "task": "IMPLEMENTATION_CODE", + "role": "implementer", + "mode": "BOUNDED_WRITE", + "profile_intent": "implementation_balanced" + }, + { + "task": "IMPLEMENTATION_DOCUMENTATION", + "role": "implementer", + "mode": "BOUNDED_WRITE", + "profile_intent": "documentation_efficient" + }, + { + "task": "REPAIR_CORRECTIVE", + "role": "implementer", + "mode": "BOUNDED_WRITE", + "profile_intent": "repair_diagnostic" + }, + { + "task": "QUALITY_REVIEW", + "role": "quality", + "mode": "READ_ONLY", + "profile_intent": "quality_correctness" + }, + { + "task": "SECURITY_REVIEW", + "role": "security", + "mode": "READ_ONLY", + "profile_intent": "security_boundaries" + }, + { + "task": "SPECIALIST_REVIEW", + "role": "SELECTED_REGISTERED_ROLE", + "mode": "READ_ONLY", + "profile_intent": "specialist_" + }, + { + "task": "FAILURE_DIAGNOSIS", + "role": "EXISTING_DIAGNOSTIC_RESPONSIBILITY", + "mode": "READ_ONLY", + "profile_intent": "diagnosis" + }, + { + "task": "FINALIZATION_REASONING", + "role": "EXISTING_FINALIZATION_RESPONSIBILITY", + "mode": "PHASE_SCOPED", + "profile_intent": "finalization_reasoning" + }, + { + "task": "VALIDATION_CONTROL", + "role": "HOST_DETERMINISTIC", + "mode": "NO_LLM_TARGET", + "profile_intent": null + }, + { + "task": "PUBLICATION_CONTROL", + "role": "HOST_DETERMINISTIC", + "mode": "NO_LLM_TARGET", + "profile_intent": null + }, + { + "task": "RECONCILIATION_CONTROL", + "role": "HOST_DETERMINISTIC", + "mode": "NO_LLM_TARGET", + "profile_intent": null + } + ], + "scenarios": [ + { + "id": "RMT-01", + "required": true, + "status": "PLANNED", + "requirement": "Deterministic task/role/risk selection; equal-priority conflict and prompt-supplied model override rejected" + }, + { + "id": "RMT-02", + "required": true, + "status": "PLANNED", + "requirement": "Two different model/effort bindings through the same adapter; per-call arguments and observed metadata remain distinct" + }, + { + "id": "RMT-03", + "required": true, + "status": "PLANNED", + "requirement": "Unsupported model/tool/sandbox/effort/context/qualification denies dispatch rather than weakening requirements" + }, + { + "id": "RMT-04", + "required": true, + "status": "PLANNED", + "requirement": "Override intersection and stale activation rejected; edits do not rewrite an active run snapshot" + }, + { + "id": "RMT-05", + "required": true, + "status": "PLANNED", + "requirement": "Explicit provider-default and absent observations remain labelled; strict observed-identity requirements fail closed" + }, + { + "id": "RMT-06", + "required": true, + "status": "PLANNED", + "requirement": "Independent full Q/S rubric/context and candidate matching; cheap model cannot drop criteria or approve its own repair" + }, + { + "id": "RMT-07", + "required": true, + "status": "PLANNED", + "requirement": "Controlled concurrent adapter interleavings do not mix models/results/usage; no implicit SA-PAR enablement" + }, + { + "id": "RMT-08", + "required": true, + "status": "PLANNED", + "requirement": "Finite pre-dispatch fallback: approved alternate succeeds, cycle/unauthorized alternate and exhausted capacity denied" + }, + { + "id": "RMT-09", + "required": true, + "status": "PLANNED", + "requirement": "Timeout/ambiguous handoff never automatically switches provider; late superseded result cannot materialize twice" + }, + { + "id": "RMT-10", + "required": true, + "status": "PLANNED", + "requirement": "Repair across roles/models/SHA/PR/restart retains one runwide three-round ceiling and existing timeout limits" + }, + { + "id": "RMT-11", + "required": true, + "status": "PLANNED", + "requirement": "Qualified deterministic control paths use zero model calls; missing SA-VAL/SA-PUB integration is not mocked into success" + }, + { + "id": "RMT-12", + "required": true, + "status": "PLANNED", + "requirement": "Failure/cancel/replay usage attributed once; unknown usage/cost/actual model never becomes zero or requested value" + }, + { + "id": "RMT-13", + "required": true, + "status": "PLANNED", + "requirement": "Console authority, concurrent edits, five locales, redaction and preview/readback; refresh creates zero generations" + }, + { + "id": "RMT-14", + "required": true, + "status": "PLANNED", + "requirement": "Installed identity, persisted policy and restart evidence; unqualified/failed comparative corpus blocks binding activation" + }, + { + "id": "RMT-15", + "required": true, + "status": "PLANNED", + "requirement": "Versioned producer evidence compatibility; no silent v1.2 schema break or consumer authority expansion" + }, + { + "id": "RMT-16", + "required": true, + "status": "PLANNED", + "requirement": "Account/data/billing boundary enforced; unavailable subscription never silently selects a metered API" + } + ], + "nodes": [ + { + "id": "RMP-CONTRACT", + "owner": "engineering-platform", + "status": "PLANNED", + "depends_on": [], + "delivery": "Task/role matrix and versioned rubrics", + "completion_evidence": [ + "Registered phase/role mappings, complete Q/S rubrics and typed model-selection/observation contracts" + ] + }, + { + "id": "RMP-CATALOG", + "owner": "engineering-platform", + "status": "PLANNED", + "depends_on": [ + "RMP-CONTRACT" + ], + "delivery": "Qualified provider/model capability catalogue", + "completion_evidence": [ + "EP-local adapter/account/version provenance, supported effort/output/data capabilities and honest unknown/default states" + ] + }, + { + "id": "RMP-RESOLVE", + "owner": "engineering-platform", + "status": "PLANNED", + "depends_on": [ + "RMP-CATALOG" + ], + "delivery": "Effective policy and deterministic selection", + "completion_evidence": [ + "Reuse effective-policy authority, constrained precedence, immutable run snapshots, explicit conflicts and finite allowed alternatives" + ] + }, + { + "id": "RMP-DISPATCH", + "owner": "engineering-platform", + "status": "PLANNED", + "depends_on": [ + "RMP-RESOLVE" + ], + "delivery": "Per-invocation model and effort application", + "completion_evidence": [ + "Actual managed-Codex adapter binding and isolation, candidate/rubric matching, no ambiguity replay and shared repair budget" + ] + }, + { + "id": "RMP-EVIDENCE", + "owner": "engineering-platform", + "status": "PLANNED", + "depends_on": [ + "RMP-DISPATCH" + ], + "delivery": "Requested-versus-observed ledger/readback", + "completion_evidence": [ + "Reuse SA-OBS ledger, trustworthy observation provenance, missing usage and failure attribution, restart-safe lineage" + ] + }, + { + "id": "RMP-CONSOLE", + "owner": "engineering-platform", + "status": "PLANNED", + "depends_on": [ + "RMP-RESOLVE" + ], + "delivery": "Task/role administration and explanation", + "completion_evidence": [ + "EP-native five-language preview/activation/readback, no in-flight drift or generated calls on refresh" + ] + }, + { + "id": "RMP-Q", + "owner": "engineering-platform", + "status": "PLANNED", + "depends_on": [ + "RMP-EVIDENCE", + "RMP-CONSOLE" + ], + "delivery": "Installed and comparative role qualification", + "completion_evidence": [ + "All required RMT scenarios, real-core CI plus installed proof and separately authorized representative model comparisons" + ] + } + ] +} diff --git a/docs/development/ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md b/docs/development/ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md new file mode 100644 index 00000000..b3fc81e2 --- /dev/null +++ b/docs/development/ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md @@ -0,0 +1,97 @@ +# EP role/task model policy — SA-ROLE implementation roadmap + +**Increment:** `EP_ROLE_TASK_MODEL_POLICY_V1`. **Status:** PLANNED. +This is a decomposition of existing **SA-ROLE**, not an additional execution +programme or a new first-canary gate. The [architecture](../engineering/ROLE_TASK_MODEL_POLICY_V1.md) +and [documentary DAG](ROLE_TASK_MODEL_POLICY_V1_DAG.json) define the task/role +matrix, effective provider/model/effort policy, fallback and admin contract. +The [parent roadmap](SUBAGENT_ORCHESTRATION_V1_ROADMAP.md) retains every existing +node and dependency; SA-F09 is not closed by documenting its solution. + +## Delivery sequence + + +| Node | Delivery | Internal dependencies | +| --- | --- | --- | +| RMP-CONTRACT | Task/role matrix and versioned rubrics | none | +| RMP-CATALOG | Qualified provider/model capability catalogue | RMP-CONTRACT | +| RMP-RESOLVE | Effective policy and deterministic selection | RMP-CATALOG | +| RMP-DISPATCH | Per-invocation model and effort application | RMP-RESOLVE | +| RMP-EVIDENCE | Requested-versus-observed ledger/readback | RMP-DISPATCH | +| RMP-CONSOLE | Task/role administration and explanation | RMP-RESOLVE | +| RMP-Q | Installed and comparative role qualification | RMP-EVIDENCE, RMP-CONSOLE | + +```text +RMP-CONTRACT -> RMP-CATALOG -> RMP-RESOLVE -> RMP-DISPATCH -> RMP-EVIDENCE --+ + | | + +-------> RMP-CONSOLE ----------------+-> RMP-Q +``` + +RMP-RESOLVE additionally requires SA-CTX, SA-OBS and RMP-EFFECTIVE-POLICY. +SA-OBS retains its SA-ISO predecessor. RMP-EFFECTIVE-POLICY is a narrow evidence +gate for the needed POL-E service subset; it neither claims that subset is +already qualified nor requires the full cross-product policy programme/UI. +The parent SA-ROLE completes only when RMP-Q and its original dependencies +are satisfied. No child depends on SA-ROLE/SA-Q, avoiding a decomposition cycle. +Existing SA-VAL/SA-PUB/SA-SEL/SA-PAR gates are consumed where their paths are +claimed, not reimplemented or required wholesale before model routing. + +First qualify multiple model/effort profiles on the SAME managed Codex adapter +and existing session mode. This first slice does not purchase/require API tokens +or pretend other vendors have qualified execution adapters. Later CLI/API/local +families reuse the same contract but each needs explicit scope/auth/cost and +adapter qualification. Concrete model names and supported effort values are +selected from actual capability evidence at activation, not invented in docs. + +## Mandatory qualification families + + +| ID | Required behavior | +| --- | --- | +| RMT-01 | Deterministic task/role/risk selection; equal-priority conflict and prompt-supplied model override rejected | +| RMT-02 | Two different model/effort bindings through the same adapter; per-call arguments and observed metadata remain distinct | +| RMT-03 | Unsupported model/tool/sandbox/effort/context/qualification denies dispatch rather than weakening requirements | +| RMT-04 | Override intersection and stale activation rejected; edits do not rewrite an active run snapshot | +| RMT-05 | Explicit provider-default and absent observations remain labelled; strict observed-identity requirements fail closed | +| RMT-06 | Independent full Q/S rubric/context and candidate matching; cheap model cannot drop criteria or approve its own repair | +| RMT-07 | Controlled concurrent adapter interleavings do not mix models/results/usage; no implicit SA-PAR enablement | +| RMT-08 | Finite pre-dispatch fallback: approved alternate succeeds, cycle/unauthorized alternate and exhausted capacity denied | +| RMT-09 | Timeout/ambiguous handoff never automatically switches provider; late superseded result cannot materialize twice | +| RMT-10 | Repair across roles/models/SHA/PR/restart retains one runwide three-round ceiling and existing timeout limits | +| RMT-11 | Qualified deterministic control paths use zero model calls; missing SA-VAL/SA-PUB integration is not mocked into success | +| RMT-12 | Failure/cancel/replay usage attributed once; unknown usage/cost/actual model never becomes zero or requested value | +| RMT-13 | Console authority, concurrent edits, five locales, redaction and preview/readback; refresh creates zero generations | +| RMT-14 | Installed identity, persisted policy and restart evidence; unqualified/failed comparative corpus blocks binding activation | +| RMT-15 | Versioned producer evidence compatibility; no silent v1.2 schema break or consumer authority expansion | +| RMT-16 | Account/data/billing boundary enforced; unavailable subscription never silently selects a metered API | + +RMP-DISPATCH may deliver a bounded headless pilot before the admin UI if its own +applicable runtime/installed/model qualification is complete. Do not mark full +SA-ROLE or RMP-Q complete from that pilot; the complete task/role matrix, Console, +observation provenance and comparative evidence remain required. + +RMP-CONSOLE reuses the EP design system, existing API/services, en/nl/de/fr/es and +current UI/security/Playwright requirements. Configuration preview is read-only; +activation requires current authority and expected revision. The existing timeout +policy is displayed as a fixed ceiling, not a new editable phase-timeout form. + +RMP-Q uses the real policy/host/adapter in deterministic CI and installed +qualification outside a checkout. External fixture models prove routing behavior, +not the quality/availability of commercial models. Real allocation qualification +uses a versioned comparative corpus, predeclared quality floors and measured +usage under separate explicit authority. No budget increase, metrics fabrication, +mandatory-review weakening or retry-until-green. A required missing capability +is a gap, not an expected-failure counted as completion. + +## Pickup and closure + +Refresh EP main/open PRs and reconcile SA-F09 plus existing provider/profile/ +ledger code before implementation. Reuse prior qualified evidence where exact +source/profile/adapter bindings remain valid. Keep source, installed, active +policy and measured quality distinct. Do not mutate current runs, grants, +credentials, default account/model, PR175 or the Forge canary for this plan. + +Documentation completion is NO_BUMP and may add offline documentary guards only. +It does not implement any RMP runtime node or alter executable programme DAGs. +The canonical EP roadmap already reaches this plan through its subagent section +and SA-ROLE; the policy roadmap links the same family instead of duplicating it. diff --git a/docs/development/SUBAGENT_ORCHESTRATION_V1_DAG.json b/docs/development/SUBAGENT_ORCHESTRATION_V1_DAG.json index 33876e89..06e9f501 100644 --- a/docs/development/SUBAGENT_ORCHESTRATION_V1_DAG.json +++ b/docs/development/SUBAGENT_ORCHESTRATION_V1_DAG.json @@ -221,5 +221,16 @@ "acceptance": "Exact-source/artifact/provider/profile comparative corpus, predeclared thresholds, complete usage coverage and wall time, unique verified findings and non-regression; installed proof separate from source merge.", "qualification_evidence": [] } - ] + ], + "role_task_model_policy": { + "increment": "EP_ROLE_TASK_MODEL_POLICY_V1", + "architecture": "docs/engineering/ROLE_TASK_MODEL_POLICY_V1.md", + "roadmap": "docs/development/ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md", + "graph": "docs/development/ROLE_TASK_MODEL_POLICY_V1_DAG.json", + "parent_node": "SA-ROLE", + "completion_requires": [ + "RMP-Q" + ], + "documentary_only": true + } } diff --git a/docs/development/SUBAGENT_ORCHESTRATION_V1_ROADMAP.md b/docs/development/SUBAGENT_ORCHESTRATION_V1_ROADMAP.md index 93caf3a1..1b90b7f5 100644 --- a/docs/development/SUBAGENT_ORCHESTRATION_V1_ROADMAP.md +++ b/docs/development/SUBAGENT_ORCHESTRATION_V1_ROADMAP.md @@ -62,6 +62,25 @@ activation, verify the actual owning publication contract and exact qualified source/artifact. A local candidate, a handoff, or this documentation is not proof. Do not reimplement the existing repair as part of this design. +## SA-ROLE task/model design refinement — 2026-09-12 + +`EP_ROLE_TASK_MODEL_POLICY_V1` now refines SA-ROLE with the +[role/task matrix and policy design](../engineering/ROLE_TASK_MODEL_POLICY_V1.md), +[seven-package roadmap](ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md) and +[documentary sub-DAG](ROLE_TASK_MODEL_POLICY_V1_DAG.json). It separates specialist +selection (SA-SEL), model/effort allocation (SA-ROLE) and execution authority. +First delivery is qualified multi-model routing on EP's existing managed Codex +session; other adapters are separately qualified later, never automatic API fallback. + +All original statuses/edges and SA-F09 remain unchanged. SA-ROLE additionally +requires completion of RMP-Q via the parent JSON's decomposition reference. +RMP-RESOLVE uses SA-CTX/SA-OBS and the narrowly evidenced existing POL-E subset; +no requirement to finish all policy UI or SA-Q first. Task/role/risk assignments, +Q/S rubrics, constrained fallback, immutable snapshots, requested versus observed +metadata, measured quality/cost and Console administration are covered together. +The extension adds documentary tests, not runtime/CI behavior or policy activation. +Current immutable timeout ceilings and one runwide repair budget remain mandatory. + ## Existing contract joins Reuse the [policy/profile lane](POLICY_GOVERNANCE_V1_ROADMAP.md) for effective diff --git a/docs/engineering/ROLE_TASK_MODEL_POLICY_V1.md b/docs/engineering/ROLE_TASK_MODEL_POLICY_V1.md new file mode 100644 index 00000000..c0906c59 --- /dev/null +++ b/docs/engineering/ROLE_TASK_MODEL_POLICY_V1.md @@ -0,0 +1,248 @@ +# EP role/task model policy V1 + +**Increment:** `EP_ROLE_TASK_MODEL_POLICY_V1`. **Owner:** Engineering Platform. +**Parent:** `SA-ROLE` / finding `SA-F09` in the +[subagent roadmap](../development/SUBAGENT_ORCHESTRATION_V1_ROADMAP.md). +**Status:** design only; implementation, activation and installed qualification +are **PLANNED**. Canonical after protected merge; **NO_BUMP**. + +The owner requested an architecture/design increment for selecting different +AI providers, models and reasoning effort by engineering task and agent role, +including operational configuration. This record adds no live settings, model +purchase, provider invocation, credential, Mission, run or retry authority. +The [scoped roadmap](../development/ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md) and +[non-executable DAG](../development/ROLE_TASK_MODEL_POLICY_V1_DAG.json) refine +SA-ROLE rather than establish another orchestrator or policy platform. + +## 1. Current evidence versus target + +Source pin: `pcvantol/engineering-platform` +`4a6dace73ccc15146a9afe5db356957d8af4bbbb`, inspected 2026-09-12 UTC. +These are source observations, not installed-service or model-quality tests. + +| Observed source | Meaning / retained boundary | +| --- | --- | +| [execution_executor.py](../../src/engineering_platform/execution_executor.py), `CodexCliClient.review/invoke/validate` | Inspected command assembly uses Codex, role prompts and sandboxing but no explicit per-role model/effort arguments. Metadata/usage helpers do not prove complete attribution. | +| [capability_review.py](../../src/engineering_platform/capability_review.py) | Separate quality/security labels plus twelve optional specialists. Existing role selection is not model selection; keyword-selection repair remains SA-SEL. | +| [providers.py](../../src/engineering_platform/providers.py) | Codex resolves EP's managed launcher. A deterministic validation executor exists. Provider status grants neither execution nor network authority. | +| [codex_observability.py](../../src/engineering_platform/codex_observability.py) | Explicit runtime metadata/usage is a foundation. Requested values, provider observations and model self-description are not interchangeable. | +| [execution_timeout_policy.py](../../src/engineering_platform/execution_timeout_policy.py) | Host-owned immutable per-action deadlines, not editable dashboard preferences. Routing cannot raise or disable them. | +| [policy architecture](POLICY_GOVERNANCE_AND_ASSURANCE_PROFILES.md) | EP owns effective profiles, admission snapshots and the shared repair budget. Reuse this authority and CENTRAL. | + +SA-F09 stays historical and OPEN until implementation and qualification exist. +This inspection proves no particular model better, cheaper, available on an +account or independent of another reviewer. No commercial model ID, price or +unsupported CLI flag is selected by this design. + +## 2. Ownership and delivery boundary + +**SA-SEL** chooses relevant optional specialists and their capacity allowance. +**SA-ROLE** binds an already selected task/role to an eligible provider/model/ +effort profile. **The existing EP host** authorizes and dispatches that invocation. +These are separate decisions; none creates a Forge Action or another scheduler. + +EP owns execution/model policy, provider installations, assurance, leases, +consumption and evidence. Forge supplies supported constraints in an immutable +request; it cannot override EP ceilings or control an EP provider directly. +Forge's own planning policy remains Forge-owned and may use different models +on another machine. Workspace retains human project/governance UX; the EP +Operations Console administers the same EP services. Forge Platform retains +deployment authority. No shared policy DB, direct peer SQL or credential broker. + +The first shipping slice supports different qualified models/efforts through the +existing EP-managed Codex adapter/session. No separate API subscription is required +or silently selected. Other CLI providers, metered APIs and local-model services +are extension classes only, each requiring a real qualified adapter, approved +profile and execution/data/auth/cost contract. A text-generation API alone does +not provide tool-use or repository execution. Unsupported integrations remain +explicitly unavailable, not usable because a provider name exists in a dropdown. + +## 3. Logical records and identity + +These TARGET records refine existing EP configuration, effective profiles and +invocation/evidence storage. They are not new SQL tables, implemented classes or +active wire schemas. Reuse owning records; version a genuinely missing seam in +its later implementation rather than creating a second authoritative copy. + +| Record | Required logical fields | +| --- | --- | +| `ModelCapabilityObservation` | provider/adapter version, installation/host, opaque account reference, model ID/alias, tools/sandbox/output/context/effort capabilities, supported metadata/usage signals, source/time/expiry and UNKNOWN/unsupported states | +| `RoleModelProfile` | immutable ID/revision/digest; provider/account; model selection mode/reference; supported effort/default; task/risk applicability; rubric/output refs; complete mandatory context; data boundary; finite ceilings; ordered qualified alternates; qualification refs | +| `RoleTaskAssignment` | activation/source revision; project/repository scope; canonical phase, registered role, task/risk predicates; unique priority within scope; profile ref and permitted overrides | +| `EffectiveRoleModelPlan` | admitted assignment/profile set and alternatives, contributing policy revisions, requirement intersection/conflicts and digest, linked to the existing effective run profile | +| `InvocationSelection` | invocation/attempt/run/Action/repair ordinal; phase/role/task; candidate SHA; profile/rubric/context digests; selected profile; requested model/effort; capability evidence; reason and remaining allowance before handoff | +| `InvocationObservation` | provider-owned outcome; separately observed model/effort or NOT_REPORTED; identity-match classification; usage/timing/enforcement sources; original/retry/supersession links | + +An explicit model ID can be a vendor alias, not immutable model weights. Preserve +that distinction. `PROVIDER_DEFAULT` must be explicitly approved and qualified; +it is not absent configuration. Where exact observed identity is required but +not provable, deny admission or retain required assurance as UNRESOLVED. Never +copy requested values into observed fields. Free-form agent claims are not +runtime attestation; validate versioned provider-owned event envelopes. + +No authfile scraping or secret enumeration for discovery. Supported non-generating +model/session reads establish only what they actually observe. An unsupported +model list stays UNKNOWN, not proof against a separately qualified explicit +binding. A generative probe requires separate authority/budget, never page refresh. + +## 4. Task/role matrix + +Task keys below are documentary classes to map to actual host phases and registered +roles in RMP-CONTRACT; they do not add lifecycle phases. Profile names describe +purpose, not a ranking of named models. Before activation each must bind an actual +supported model/effort and qualification; unresolved placeholders are not runnable. + +| Task key | Role / mode | Profile intent and invariant | +| --- | --- | --- | +| IMPLEMENTATION_CODE | implementer / bounded writer | `implementation_balanced`: qualified code/tool use; approved deeper-analysis profile for evidenced complexity/risk, never a second writer | +| IMPLEMENTATION_DOCUMENTATION | implementer / bounded writer | `documentation_efficient`: smaller/faster candidate only after docs-scope qualification; file extension alone cannot downgrade code/config risk | +| REPAIR_CORRECTIVE | implementer / bounded writer | `repair_diagnostic`: compatible initial binding, preapproved escalation only; same budget and full revalidation/re-review | +| QUALITY_REVIEW | quality / read-only | `quality_correctness`: correctness, edge cases, architecture/maintainability and relevant tests against every applicable criterion | +| SECURITY_REVIEW | security / read-only | `security_boundaries`: trust, authorization, credentials/data and abuse cases; no cheaper substitute unless independently qualified for this role | +| SPECIALIST_REVIEW | selected registered specialist / read-only | `specialist_`: domain rubric and real consumer; selection/capacity stays SA-SEL; advisory is not mandatory PASS | +| FAILURE_DIAGNOSIS | existing diagnostic responsibility / read-only | `diagnosis`: bounded interpretation; no self-authorized correction or retry | +| FINALIZATION_REASONING | existing finalization responsibility / phase-scoped | `finalization_reasoning`: unresolved engineering reasoning only, preserving existing host write gates | +| VALIDATION_CONTROL | deterministic host / no LLM target | Known commands and authoritative control receipts; SA-VAL owns general integration | +| PUBLICATION_CONTROL | deterministic host / no LLM target | Known post-assurance first-PR operation; SA-PUB retains its independent qualification gate | +| RECONCILIATION_CONTROL | deterministic host / no LLM target | Known identity/receipt transitions; ambiguous diagnosis is separate, never a forced-success model call | + +The deterministic targets do not claim current LLM-mediated paths were removed. +Quality and Security retain separate versioned rubrics, contexts and invocation +identities. They may use the same qualified model: a different name alone does +not prove independence. No private-reasoning/verdict sharing, inherited approvals +or implementer self-approval. Required context cannot be omitted to fit a cheaper +model; explicit overflow blocks a complete mandatory-review claim. + +Effort is an adapter-supported enum, not a universal low/medium/high scale. Show +supported values and actual translation; a missing knob is N/A, not an inferred +equivalent. Profile names do not promise a latency or cost saving before measurement. + +## 5. Deterministic selection, snapshots and dispatch + +```text +host phase + registered role + validated task/risk evidence + -> admitted EP policy + supported producer constraints + -> candidate profiles + -> capability/authority/data/qualification/capacity eligibility + -> deterministic preference resolution or explicit conflict + -> persist InvocationSelection; reserve existing allowances + -> same EP provider boundary with per-invocation settings + -> validate output AND observation; preserve outcome/evidence +``` + +V1 uses explicit rules, not an LLM choosing its own model or an online cost learner. +Versioned host facts determine risk/complexity: validated write scope, affected +capabilities, trust changes and contract size. Task prose cannot inject model flags +or label itself low risk. Recheck a changed candidate without widening authority. + +Hard obligations combine by union, allowed capabilities/data scopes by intersection, +ceilings by minimum applicable limit and remaining allowance. Preference precedence: +EP default -> permitted project/repository override -> most-specific allowed +phase/task/role assignment. Equally scoped matching rules require a unique explicit +priority; ties or conflicting mandatory model constraints fail closed. No lower +scope may grant a new account, provider, network target, sandbox or exception. +Bound selectors; no arbitrary uploaded routing code. + +Freeze the assignment/profile/rubric set at admission. Each invocation binds the +current candidate and repair ordinal and uses that set or a prequalified alternate. +Global edits apply to new runs by default. In-flight policy migration requires a +separate owning operation and requalification, never a browser Save. Live revocation, +credential trust, availability and expiry can still deny the next dispatch. Use +existing atomic reservation/lease/ledger mechanisms; concurrent workers cannot +spend the same remaining allowance. + +Resolve arguments per invocation, not by mutating shared client state, process-wide +environment or the user's Codex config. Preserve EP-managed executable/account +identity independently of Forge. A Quality call cannot inherit another role's +model due to a race. SA-PAR still gates parallel mandatory reviews; this design +does not enable parallelism or native nested agents. + +## 6. Fallback, escalation, recovery and limits + +Fallback chooses an admitted qualified alternative. Escalation changes reasoning +capability for a real unresolved task. Recovery determines whether another attempt +is safe. None grants a retry, wider scope, extra spend or another corrective round. + +| Observed condition | Required behavior | +| --- | --- | +| Primary unavailable before handoff | Use next eligible preapproved alternate only under current authority/capacity; persist reason first, otherwise WAIT/BLOCK | +| No qualified profile or unsupported required knob/observation | Explicit denial/unresolved; skip only genuinely optional work with a disposition, never mandatory PASS | +| Proven failure before provider execution | Existing bounded recovery may retry/change binding; keep attempt evidence and usage classification even if generation is proven zero | +| Timeout/lost response/interruption/MAY_HAVE_HAPPENED | No automatic alternate-provider call. Use owning ambiguity/readback/operator recovery, retain possible usage and reject late superseded results | +| Findings, valid failure or malformed completed result | Preserve the original; corrective work uses existing repair accounting, no reviewer shopping until PASS | +| Observed model/effort contradicts strict binding | Required assurance is invalid; reconcile possible effects/usage without blind replay or editing the observation | + +Primary and alternates are a finite acyclic sequence with explicit attempt caps, +subordinate to existing stricter limits. No A -> B -> A loop, hidden quorum or +paid discovery loop. Every provider handoff counts as an invocation; every actual +corrective dispatch uses the shared allowance. Current EP permits at most three +total corrective rounds per run/continuation lineage, not per role, phase or model. +No reset via provider, SHA, PR, resume, activation or replacement run identity. + +The current timeout policy stays a read-only hard ceiling. Later qualified shorter +budgets may tighten it, never raise it for a slower model. This design makes no +existing constant editable. Invocation timeout is not the run deadline or proof +of no remote work. Keep enforced admission limits, estimates and provider-reported +subscription allowance distinct. UNKNOWN usage/cost is not zero; no hard spend/ +subscription cap without an actually enforceable boundary and evidence. + +## 7. Operations Console and service contract + +Add **Configuration -> AI task/role policies** in the existing EP Console, not +another dashboard. The same typed service can serve other authorized clients; +standalone EP does not require Forge or Workspace. Backend permission is decisive. + +The matrix shows task, role, mandatory/optional, provider/account, model mode, +effort, rubric, fallback/escalation, source/override, effective revision, freshness +and qualification. Row details explain the selected/rejected alternatives. Separate +catalogue observation, adapter/session readiness, role qualification and active +policy. Missing/unavailable capabilities stay visible with reasons. + +Edit -> validate -> preview redacted diff and effects on NEW runs -> required +approval -> activate with expected revision/digest -> authoritative readback. +Conflicting updates fail; writes are idempotent. Rollback appends activation history. +Active runs show their own snapshot. Save/Refresh never installs models, acquires +credentials, invokes AI or migrates running work. The existing subscription mode +stays unchanged without an explicit provider/billing decision; no API fallback. + +Run/step details show requested, resolved, translated and observed settings, +observation status, invocation IDs, fallback/repair cause, usage source/coverage, +wall time, deadline and remaining allowance. Filter/export redacted history by +run/phase/role/provider/model. No raw prompts, private reasoning, authfiles or +secrets. Efficiency comparisons require comparable tasks/criteria and complete +usage; a faster incomplete review is not an improvement. + +Reuse EP's design system, en/nl/de/fr/es translations, accessible modals, safe +confirmation/error patterns and browser tests. Actual new routes must join the +versioned API/OpenAPI/collection contracts; this design invents none. A needed +terminal-evidence extension is producer-versioned and consumer-qualified, not a +silent change to the current peer v1.2 contract. + +## 8. Qualification and rollout + +Implement the scenario families in the scoped DAG against real host/policy/adapter +modules and installed-wheel composition outside a checkout. CI may fake external +processes, not internal authorization/context/evidence gates. Exercise two distinct +model/effort bindings on one adapter, every role class, independent Q/S contexts, +no-LLM controls where qualified, and all negative paths. Keep existing coverage, +security, installed and five-language/Playwright requirements for production code. +Documentary tests alone are not execution qualification. + +Before activating a real binding, qualify comparative performance on a versioned +representative defect/attack corpus with predeclared floors and permitted regression +tolerances. Include severity-weighted missed defects, false positives, unresolved +criteria, repairs, review independence, latency/failure and measured usage with +provenance. No post-hoc threshold selection or savings claim from one toy result. +Live comparison is separately authorized and budgeted; none runs in this increment. + +Roll out catalogue/readback -> shadow resolution (no extra provider calls) -> +qualified same-Codex multi-model dispatch on bounded new runs -> guarded Console +activation -> integrated role-quality/evidence proof. Exact historical run snapshots +and source/installed/live availability stay separate. A config example is not a +completed migration or qualified model allocation. + +Additional provider families, learned routing, speculative model races/ensembles, +new repair loops, native nested agents and general billing products are later work. +SA-VAL/SA-PUB/SA-SEL/SA-PAR retain their own scope. SA-ROLE can be qualified without +finishing SA-Q, but keeps SA-CTX/SA-OBS and the required effective-policy evidence. +No new first Forge->EP canary prerequisite and no change to parked PR #175. diff --git a/tests/test_role_task_model_policy_design.py b/tests/test_role_task_model_policy_design.py new file mode 100644 index 00000000..8c652927 --- /dev/null +++ b/tests/test_role_task_model_policy_design.py @@ -0,0 +1,141 @@ +"""Offline documentation guards only: no runtime/model-policy qualification.""" +from graphlib import TopologicalSorter +import json +from pathlib import Path +import re +import unittest + +ROOT = Path(__file__).resolve().parents[1] +GRAPH_PATH = ROOT / "docs/development/ROLE_TASK_MODEL_POLICY_V1_DAG.json" + + +class RoleTaskModelPolicyDesignTests(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.graph = json.loads(GRAPH_PATH.read_text(encoding="utf-8")) + cls.parent = json.loads((ROOT / cls.graph["parent_graph"]).read_text(encoding="utf-8")) + cls.design = (ROOT / cls.graph["architecture"]).read_text(encoding="utf-8") + cls.roadmap = (ROOT / cls.graph["roadmap"]).read_text(encoding="utf-8") + + def test_documentary_only_without_activation_or_canary_dependency(self): + g = self.graph + self.assertEqual(g["increment"], "EP_ROLE_TASK_MODEL_POLICY_V1") + self.assertEqual(g["parent_node"], "SA-ROLE") + self.assertEqual(g["finding"], "SA-F09") + self.assertEqual(g["status"], "PLANNED") + self.assertEqual(g["version_change"], "NO_BUMP") + self.assertTrue(g["documentary_only"]) + for flag in ("execution_authority", "automatic_dispatch", "runtime_activation", "first_canary_prerequisite"): + self.assertIs(g[flag], False) + self.assertEqual(g["actual_model_ids_selected"], []) + + def test_parent_decomposition_preserves_original_dependencies(self): + ext = self.parent["role_task_model_policy"] + self.assertEqual(ext["graph"], GRAPH_PATH.relative_to(ROOT).as_posix()) + self.assertEqual(ext["completion_requires"], ["RMP-Q"]) + parent_nodes = {n["id"]: n for n in self.parent["nodes"]} + self.assertEqual(len(parent_nodes), 11) + self.assertEqual(parent_nodes["SA-ROLE"]["depends_on"], ["SA-CTX", "SA-OBS"]) + self.assertEqual(parent_nodes["SA-ROLE"]["status"], "PLANNED") + self.assertEqual(parent_nodes["SA-ROLE"]["qualification_evidence"], []) + self.assertEqual(parent_nodes["SA-OBS"]["depends_on"], ["SA-ISO"]) + self.assertIn("SA-ROLE", parent_nodes["SA-Q"]["depends_on"]) + + def test_seven_nodes_and_full_nested_graph_remain_acyclic(self): + nodes = self.graph["nodes"] + ids = {n["id"] for n in nodes} + self.assertEqual(len(nodes), 7) + self.assertEqual(len(ids), 7) + deps = {n["id"]: list(n["depends_on"]) for n in nodes} + for n in nodes: + self.assertEqual(n["owner"], "engineering-platform") + self.assertEqual(n["status"], "PLANNED") + self.assertTrue(n["completion_evidence"]) + self.assertTrue(set(n["depends_on"]) <= ids) + self.assertNotIn(n["id"], n["depends_on"]) + self.assertEqual(len(n["depends_on"]), len(set(n["depends_on"]))) + self.assertEqual(set(TopologicalSorter(deps).static_order()), ids) + combined = {n["id"]: list(n["depends_on"]) for n in self.parent["nodes"]} + combined.update(deps) + for n, external in self.graph["external_dependencies"].items(): + combined[n].extend(external) + for e in self.graph["external_requirements"]: + self.assertEqual(e["owning_lane"], "POL-E") + self.assertEqual(e["evidence"], []) + combined[e["id"]] = [] + for e in self.parent["external_requirements"]: + combined[e["id"]] = e["depends_on"] + combined["SA-ROLE"].extend(self.parent["role_task_model_policy"]["completion_requires"]) + self.assertTrue(all(set(v) <= combined.keys() for v in combined.values())) + self.assertEqual(set(TopologicalSorter(combined).static_order()), set(combined)) + + def test_markdown_and_json_delivery_edges_match(self): + rows = {} + for line in self.roadmap.splitlines(): + if line.startswith("| RMP-"): + cells = [c.strip() for c in line.strip("|").split("|")] + rows[cells[0]] = set(re.findall(r"RMP-[A-Z]+", cells[2])) + self.assertEqual(rows, {n["id"]: set(n["depends_on"]) for n in self.graph["nodes"]}) + + def test_task_role_matrix_keeps_controls_and_reviews_distinct(self): + tasks = {n["task"]: n for n in self.graph["task_matrix"]} + self.assertEqual(len(tasks), 11) + for key, row in tasks.items(): + self.assertIn(f"| {key} |", self.design) + for task in ("QUALITY_REVIEW", "SECURITY_REVIEW", "SPECIALIST_REVIEW", "FAILURE_DIAGNOSIS"): + self.assertEqual(tasks[task]["mode"], "READ_ONLY") + for task in ("VALIDATION_CONTROL", "PUBLICATION_CONTROL", "RECONCILIATION_CONTROL"): + self.assertEqual(tasks[task]["mode"], "NO_LLM_TARGET") + self.assertIsNone(tasks[task]["profile_intent"]) + self.assertNotEqual(tasks["QUALITY_REVIEW"]["profile_intent"], tasks["SECURITY_REVIEW"]["profile_intent"]) + + def test_identity_default_and_subscription_safety(self): + inv = self.graph["invariants"] + for key in ("provider_default_requires_explicit_policy", "per_invocation_settings", "qualified_binding_required_before_activation"): + self.assertIs(inv[key], True) + for key in ("requested_is_observed", "missing_usage_is_zero", "automatic_api_fallback", + "unknown_observed_model_can_satisfy_exact_requirement", "all_optional_providers_required_for_v1"): + self.assertIs(inv[key], False) + for text in ("PROVIDER_DEFAULT", "NOT_REPORTED", "No authfile scraping", "existing EP-managed Codex"): + self.assertIn(text, self.design) + + def test_no_new_retry_authority_or_budget_reset(self): + inv = self.graph["invariants"] + for key in ("allow_ambiguous_fallback", "allow_review_shopping", "allow_repair_budget_reset", + "model_switch_grants_authority", "current_timeout_constants_editable", + "live_profile_edits_change_active_runs"): + self.assertIs(inv[key], False) + self.assertEqual(inv["current_runwide_repair_limit"], 3) + for text in ("MAY_HAVE_HAPPENED", "finite acyclic", "run/continuation lineage", "same EP provider boundary"): + self.assertIn(text, self.design) + + def test_sixteen_required_scenario_families_are_visible(self): + scenarios = self.graph["scenarios"] + self.assertEqual([s["id"] for s in scenarios], [f"RMT-{i:02}" for i in range(1, 17)]) + for s in scenarios: + self.assertIs(s["required"], True) + self.assertEqual(s["status"], "PLANNED") + self.assertIn(f'| {s["id"]} |', self.roadmap) + self.assertIn("Documentary tests alone are not execution qualification", self.design) + + def test_console_and_source_provenance_are_explicit(self): + self.assertRegex(self.graph["source_pin"], r"^[0-9a-f]{40}$") + self.assertIn(self.graph["source_pin"], self.design) + for phrase in ("AI task/role policies", "en/nl/de/fr/es", "expected revision/digest", + "predeclared floors", "No separate API subscription", "direct peer SQL"): + self.assertIn(phrase, self.design) + self.assertIs(self.graph["invariants"]["page_refresh_generates"], False) + self.assertIs(self.graph["invariants"]["forge_planning_policy_unchanged"], True) + self.assertIs(self.graph["invariants"]["metadata_redefines_terminal_v12_without_versioning"], False) + + def test_navigation_reaches_one_design_from_both_owning_lanes(self): + for key in ("architecture", "roadmap", "parent_graph", "policy_roadmap"): + self.assertTrue((ROOT / self.graph[key]).is_file()) + for path in ("docs/development/SUBAGENT_ORCHESTRATION_V1_ROADMAP.md", self.graph["policy_roadmap"]): + text = (ROOT / path).read_text(encoding="utf-8") + self.assertIn("ROLE_TASK_MODEL_POLICY_V1.md", text) + self.assertIn("ROLE_TASK_MODEL_POLICY_V1_ROADMAP.md", text) + + +if __name__ == "__main__": + unittest.main()