From 3e735809884de6036918c0f20fb7716b88bb54b5 Mon Sep 17 00:00:00 2001 From: Jay Dev Date: Tue, 25 Aug 2026 06:31:06 +0530 Subject: [PATCH 1/5] fix(v10): integrate self-healing repair with compute and agent runtime --- src/myai/v10_engine.py | 98 ++++++++++++++++++++++++------------------ 1 file changed, 55 insertions(+), 43 deletions(-) diff --git a/src/myai/v10_engine.py b/src/myai/v10_engine.py index 56b815a..d0685ad 100644 --- a/src/myai/v10_engine.py +++ b/src/myai/v10_engine.py @@ -1,9 +1,9 @@ from __future__ import annotations -from .agent_runtime import AgentRuntimeResult +from .agent_runtime import AgentRuntimeResult, MultiModelAgentRuntime from .schemas import ChatMessage -from .v9_engine import V9AIEngine from .model_router import RoutingRequest +from .v9_engine import V9AIEngine class V10AIEngine(V9AIEngine): @@ -11,64 +11,76 @@ class V10AIEngine(V9AIEngine): version = "v10.0" + def repair_compute_policy(self, traceback_text: str): + """Choose bounded compute from the real failure evidence before launching repair agents.""" + diagnosis = self.diagnose_failure(traceback_text) + complexity = 0.75 if diagnosis.primary_frame else 0.6 + uncertainty = max(0.0, min(1.0, 1.0 - diagnosis.confidence)) + decision_request = RoutingRequest( + task_kind="debugging", + complexity=complexity, + uncertainty=uncertainty, + context_size=len(traceback_text), + risk="high", + latency_sensitive=False, + quality_priority=True, + ) + return self.compute_policy(decision_request) + def repair_context_v10(self, traceback_text: str) -> tuple[ChatMessage, ...]: - """Extend V9's targeted repair context with V10 health/signature/compute evidence.""" + """Build compact repair context from V9 diagnosis plus V10 health/signature state.""" diagnosis = self.diagnose_failure(traceback_text) - context = list(self.repair_context_v9(traceback_text)) - symbol = diagnosis.primary_frame.symbol if diagnosis.primary_frame else "" + base = list(super().repair_context_v9(traceback_text)) signature = self.failure_signature(traceback_text) - policy = self.compute_policy( - RoutingRequest( - task_kind="coding", - complexity=min(1.0, 0.75 + (0.2 if diagnosis.impact.nodes else 0.0)), - uncertainty=max(0.0, 1.0 - diagnosis.confidence), - risk="high", - ) + symbol = diagnosis.primary_frame.symbol if diagnosis.primary_frame else "" + inspection = self.inspection_mode(symbol) if symbol else "deep" + policy = self.repair_compute_policy(traceback_text) + history = self.repair_memory.similar( + diagnosis.error_type, + diagnosis.message, + limit=3, ) - prior = self.failure_signatures.similar(signature, limit=3) - trace_count = 0 - if diagnosis.primary_frame: - trace_count = sum( - 1 - for event in self.runtime_traces.events.values() - if event.path == diagnosis.primary_frame.path - and (event.line is None or abs(event.line - diagnosis.primary_frame.line) <= 8) - ) - context.append( + base.append( ChatMessage( role="system", content=( - "V10 repair control plane: preserve verified code, use causal evidence first, " - "and prefer the smallest justified repair. Do not promote code automatically." - ), - ) - ) - context.append( - ChatMessage( - role="user", - content=( + "V10 repair constraints:\n" f"FAILURE SIGNATURE: {signature.value}\n" - f"CODE HEALTH MODE: {self.inspection_mode(symbol) if symbol else 'deep'}\n" - f"TRACE EVENTS NEAR FAILURE: {trace_count}\n" - f"SIMILAR FAILURE SIGNATURES: {len(prior)}\n" - f"COMPUTE POLICY: depth={policy.reasoning_depth}; parallel={policy.max_parallel}; " - f"retries={policy.max_retries}; verification={policy.verification_passes}; tier={policy.preferred_tier}\n" - f"REPAIR ATTEMPTS BUDGET: {self.settings.self_healing_max_repair_attempts}" + f"CODE HEALTH INSPECTION MODE: {inspection}\n" + f"COMPUTE POLICY: tier={policy.preferred_tier}; depth={policy.reasoning_depth}; " + f"parallel={policy.max_parallel}; retries={policy.max_retries}; " + f"verification_passes={policy.verification_passes}\n" + f"SIMILAR FAILURES FOUND: {len(history)}\n" + "Preserve verified/stable code. Read only the causal impact slice. " + "A repair proposal is not a verified patch; execution and promotion remain validation-gated." ), ) ) - return tuple(context) + return tuple(base) def propose_repair(self, traceback_text: str) -> AgentRuntimeResult: - """Run V10 repair through the existing multi-model runtime using bounded evidence context.""" + """Use the existing recursive multi-model runtime with V10's bounded compute policy.""" + if not self.settings.self_healing_enabled: + return super().propose_repair(traceback_text) + diagnosis = self.diagnose_failure(traceback_text) + policy = self.repair_compute_policy(traceback_text) context = self.repair_context_v10(traceback_text) - return self.agent_runtime.run( + runtime = MultiModelAgentRuntime( + self.model_pool, + self.router, + policy.execution_budget(base=self.agent_runtime.budget), + ) + result = runtime.run( objective=( - f"Repair {diagnosis.error_type} with the minimum causally justified change. " - f"Primary symbol: {diagnosis.primary_frame.symbol if diagnosis.primary_frame else 'unknown'}. " - f"Confidence: {diagnosis.confidence:.2f}." + f"Repair {diagnosis.error_type} in " + f"{diagnosis.primary_frame.path if diagnosis.primary_frame else 'unknown'} " + f"with the smallest causally justified change. " + f"Hypothesis: {diagnosis.root_cause_hypothesis}" ), task_kind="coding", context=context, + state=self.cognitive_state, ) + self.cognitive_state.active_strategy = "v10-self-healing:" + policy.preferred_tier + return result From e3b9fd0ef422f20fd04ee9ead1feea670434dffb Mon Sep 17 00:00:00 2001 From: Jay Dev Date: Tue, 25 Aug 2026 06:31:23 +0530 Subject: [PATCH 2/5] feat(runtime): allow self-healing compute policy to pin worker tier --- src/myai/agent_runtime.py | 46 ++++++++++++++++++++++++--------------- 1 file changed, 28 insertions(+), 18 deletions(-) diff --git a/src/myai/agent_runtime.py b/src/myai/agent_runtime.py index 50c631d..49b7cba 100644 --- a/src/myai/agent_runtime.py +++ b/src/myai/agent_runtime.py @@ -20,10 +20,17 @@ class AgentRuntimeResult: class MultiModelAgentRuntime: """Connect the recursive graph to real tier-specific model endpoints.""" - def __init__(self, pool: TieredModelPool, router: AdaptiveModelRouter, budget: ExecutionBudget) -> None: + def __init__( + self, + pool: TieredModelPool, + router: AdaptiveModelRouter, + budget: ExecutionBudget, + worker_tier_override: ModelTier | None = None, + ) -> None: self.pool = pool self.router = router self.budget = budget + self.worker_tier_override = worker_tier_override self._tiers: dict[str, ModelTier] = {} def run( @@ -68,19 +75,23 @@ def _decompose(self, node: TaskNode, budget: ExecutionBudget) -> Sequence[TaskNo ) def _choose_tier(self, node: TaskNode, context: Sequence[ChatMessage]) -> ModelTier: - decision = self.router.choose( - RoutingRequest( - task_kind=node.role, - complexity=0.85 if node.depth == 0 else 0.7, - uncertainty=0.75 if node.role in {"critic", "reviewer", "countercheck"} else 0.45, - context_size=sum(len(m.content) for m in context), - risk="high" if node.role in {"reviewer", "countercheck", "synthesis"} else "medium", - latency_sensitive=False, - quality_priority=node.role in {"critic", "reviewer", "synthesis"}, + if self.worker_tier_override is not None and node.role not in {"critic", "reviewer", "countercheck", "synthesis"}: + tier = self.worker_tier_override + else: + decision = self.router.choose( + RoutingRequest( + task_kind=node.role, + complexity=0.85 if node.depth == 0 else 0.7, + uncertainty=0.75 if node.role in {"critic", "reviewer", "countercheck"} else 0.45, + context_size=sum(len(m.content) for m in context), + risk="high" if node.role in {"reviewer", "countercheck", "synthesis"} else "medium", + latency_sensitive=False, + quality_priority=node.role in {"critic", "reviewer", "synthesis"}, + ) ) - ) - self._tiers[node.id] = decision.tier - return decision.tier + tier = decision.tier + self._tiers[node.id] = tier + return tier def _worker( self, @@ -98,10 +109,9 @@ def _worker( system = ChatMessage( role="system", content=( - "You are a specialist worker in MY-AI V9.1. Return only useful work for the " - "assigned role. Do not claim tools or evidence you did not receive. Preserve " - "uncertainty and disagreements. Use the shared cognitive state as context, not as " - "proof; distinguish beliefs from observations and avoid inventing missing facts." + "You are a specialist worker in MY-AI V10. Return only useful work for the assigned role. " + "Do not claim tools or evidence you did not receive. Preserve uncertainty and disagreements. " + "Use the shared cognitive state as context, not as proof; distinguish beliefs from observations and avoid inventing missing facts." ), ) prompt = ChatMessage( @@ -138,7 +148,7 @@ def _judge( ChatMessage( role="system", content=( - "You are an independent judge for MY-AI V9.1. Evaluate whether the worker output is " + "You are an independent judge for MY-AI V10. Evaluate whether the worker output is " "consistent, useful, evidence-aware and complete. Treat the cognitive state as " "context rather than ground truth. Return JSON only: " '{"passed":true|false,"confidence":0.0-1.0,"feedback":"..."}' From fe30b4f0ade6881236d33c0c826e14c59d38509c Mon Sep 17 00:00:00 2001 From: Jay Dev Date: Tue, 25 Aug 2026 06:31:32 +0530 Subject: [PATCH 3/5] fix(v10): honor compute policy tier during repair workers --- src/myai/v10_engine.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/src/myai/v10_engine.py b/src/myai/v10_engine.py index d0685ad..9a99558 100644 --- a/src/myai/v10_engine.py +++ b/src/myai/v10_engine.py @@ -1,8 +1,8 @@ from __future__ import annotations from .agent_runtime import AgentRuntimeResult, MultiModelAgentRuntime -from .schemas import ChatMessage from .model_router import RoutingRequest +from .schemas import ChatMessage from .v9_engine import V9AIEngine @@ -70,6 +70,7 @@ def propose_repair(self, traceback_text: str) -> AgentRuntimeResult: self.model_pool, self.router, policy.execution_budget(base=self.agent_runtime.budget), + worker_tier_override=policy.preferred_tier, ) result = runtime.run( objective=( From 7a8ea79e058c86c027b47d8c08c7bc14a21330f1 Mon Sep 17 00:00:00 2001 From: Jay Dev Date: Tue, 25 Aug 2026 06:31:38 +0530 Subject: [PATCH 4/5] test(v10): cover integrated self-healing repair path --- tests/test_v10_integration.py | 37 +++++++++++++++++++++++++++++++++++ 1 file changed, 37 insertions(+) create mode 100644 tests/test_v10_integration.py diff --git a/tests/test_v10_integration.py b/tests/test_v10_integration.py new file mode 100644 index 0000000..f3d63ea --- /dev/null +++ b/tests/test_v10_integration.py @@ -0,0 +1,37 @@ +from __future__ import annotations + +from myai import AIEngine, V10AIEngine + + +def test_v10_public_engine_and_repair_policy() -> None: + engine = AIEngine() + assert isinstance(engine, V10AIEngine) + assert engine.version == "v10.0" + policy = engine.repair_compute_policy( + "Traceback (most recent call last):\n" + " File 'src/myai/engine.py', line 10, in generate\n" + " value = session['id']\n" + "TypeError: NoneType is not subscriptable" + ) + assert policy.preferred_tier in {"fast", "balanced", "frontier"} + assert 1 <= policy.reasoning_depth <= 3 + assert 1 <= policy.verification_passes <= 2 + + +def test_v10_repair_context_contains_operational_guards() -> None: + engine = AIEngine() + context = engine.repair_context_v10( + "Traceback (most recent call last):\n" + " File 'src/myai/engine.py', line 10, in generate\n" + "TypeError: NoneType is not subscriptable" + ) + rendered = "\n".join(message.content for message in context) + assert "FAILURE SIGNATURE:" in rendered + assert "CODE HEALTH INSPECTION MODE:" in rendered + assert "COMPUTE POLICY:" in rendered + assert "validation-gated" in rendered + + +def test_v10_unknown_symbols_require_deep_inspection() -> None: + engine = AIEngine() + assert engine.inspection_mode("unknown_symbol") == "deep" From ac4ba48b4e6c6b721b161b6b048122b451782c6d Mon Sep 17 00:00:00 2001 From: Jay Dev Date: Tue, 25 Aug 2026 06:31:47 +0530 Subject: [PATCH 5/5] docs(state): record V10 self-healing capabilities and evidence targets --- MODEL_STATE.json | 37 +++++++++++++++++++------------------ 1 file changed, 19 insertions(+), 18 deletions(-) diff --git a/MODEL_STATE.json b/MODEL_STATE.json index 87a72c4..7e7d493 100644 --- a/MODEL_STATE.json +++ b/MODEL_STATE.json @@ -1,55 +1,56 @@ { - "schema_version": 5, - "model_version": "v10-cognitive-compute-report", + "schema_version": 6, + "model_version": "v10.0-self-healing-runtime", "purpose": "Canonical architecture and measurement baseline for evidence-driven MY-AI upgrades.", "reporting": { "entrypoint": "myai.build_model_report / AIEngine.model_report", - "source_of_truth": ["live engine feature presence", "CapabilityLedger", "CapabilityBenchmark cases", "EvolutionMemory strategy scores", "CognitiveComputeController policies"], + "source_of_truth": ["live engine feature presence", "CapabilityLedger", "CapabilityBenchmark cases", "EvolutionMemory strategy scores", "CognitiveComputeController policies", "SelfHealingRuntime episodes", "CodeHealthStore"], "manual_capability_scores": false, "numeric_claim_rule": "Numeric capability claims require executable evidence; architecture presence alone is not a score." }, "architecture": { - "core": "recursive multi-model cognitive mesh with bounded cognitive context, persistent evidence-gated memory, capability-aware compute allocation, and self-derived reporting", + "core": "recursive multi-model cognitive mesh with bounded cognitive context, persistent evidence-gated memory, capability-aware compute allocation, causal repository intelligence, and bounded self-healing runtime", "model_routing": ["fast", "balanced", "frontier"], - "compute_control": ["capability-gap pressure", "uncertainty pressure", "risk pressure", "historical strategy reliability", "bounded execution budgets"], - "repository_representation": ["persistent symbol graph", "repository twin", "program graph", "call edges", "data-flow references", "control-flow edges"], - "runtime_evidence": ["persistent trace events", "causal links", "failure neighborhoods"], - "memory": ["semantic knowledge", "typed cognitive memory", "persistent cognitive memory", "evidence-gated lifecycle", "repair memory", "evolution memory", "capability ledger"], - "verification": ["specialist workers", "frontier judge", "targeted validation", "promotion gates", "deterministic capability checks"], - "self_improvement": ["strategy scoring", "verified strategy feedback", "baseline comparison", "capability deltas", "regression detection", "executable architecture benchmarks", "compute-policy adaptation"] + "compute_control": ["capability-gap pressure", "uncertainty pressure", "risk pressure", "historical strategy reliability", "bounded execution budgets", "repair-specific tier enforcement"], + "repository_representation": ["persistent symbol graph", "repository twin", "program graph", "call edges", "data-flow references", "control-flow edges", "causal impact slices"], + "runtime_evidence": ["persistent trace events", "causal links", "failure neighborhoods", "failure signatures", "repair episodes"], + "memory": ["semantic knowledge", "typed cognitive memory", "persistent cognitive memory", "evidence-gated lifecycle", "repair memory", "evolution memory", "capability ledger", "failure signature memory"], + "self_healing": ["failure detection", "causal diagnosis", "bounded reproduction", "policy-governed recursive repair workers", "independent frontier judge", "fault injection", "stable-code health", "validation-gated promotion"], + "verification": ["specialist workers", "frontier judge", "targeted validation", "promotion gates", "deterministic capability checks", "fault recovery verification"], + "self_improvement": ["strategy scoring", "verified strategy feedback", "baseline comparison", "capability deltas", "regression detection", "executable architecture benchmarks", "compute-policy adaptation", "repair lessons"] }, "capability_dimensions": { "reasoning": {"status": "architecture_benchmarked", "score_source": "executable architecture evidence"}, "coding": {"status": "architecture_benchmarked_via_code_localization", "score_source": "executable repository evidence"}, "repository_understanding": {"status": "architecture_benchmarked", "score_source": "program/repository twin evidence"}, - "debugging": {"status": "architecture_benchmarked", "score_source": "impact/dependency evidence"}, + "debugging": {"status": "architecture_benchmarked_with_causal_trace", "score_source": "impact/dependency + runtime trace evidence"}, "memory": {"status": "architecture_benchmarked", "score_source": "typed lifecycle + bounded retrieval evidence"}, "tool_use": {"status": "multi_model_provider_layer_unmeasured", "score": null}, "planning": {"status": "architecture_benchmarked", "score_source": "cognitive planning + bounded graph evidence"}, "verification": {"status": "architecture_benchmarked", "score_source": "deterministic cognitive verification evidence"}, - "efficiency": {"status": "architecture_benchmarked", "score_source": "bounded-context evidence"}, + "efficiency": {"status": "architecture_benchmarked", "score_source": "bounded-context + stable-code health evidence"}, "self_improvement": {"status": "architecture_benchmarked", "score_source": "evidence-gated strategy promotion"} }, "evidence_runner": { "entrypoint": "myai.run_architecture_benchmark", "mode": "deterministic", "requires_provider_model": false, - "measured_cases": ["frontier-routing", "recursive-agent-graph", "cross-episode-repair-memory", "trace-verification", "program-graph-localization", "causal-trace-diagnosis", "targeted-context", "strategy-promotion"], - "unmeasured_provider_capabilities": ["tool_use"], + "measured_cases": ["frontier-routing", "recursive-agent-graph", "cross-episode-repair-memory", "trace-verification", "program-graph-localization", "causal-trace-diagnosis", "targeted-context", "strategy-promotion", "self-healing-runtime", "fault-injection-recovery", "stable-code-inspection"], + "unmeasured_provider_capabilities": ["tool_use", "provider-backed task quality"], "principle": "Executed evidence, not architecture presence, is required for numeric capability claims." }, "compute_policy": { "entrypoint": "myai.CognitiveComputeController", "inputs": ["capability ledger gap", "task uncertainty", "task risk", "historical strategy reliability", "routing decision"], "outputs": ["reasoning depth", "parallelism", "retry budget", "verification passes", "preferred tier"], - "constraints": ["bounded depth", "bounded parallelism", "bounded retries", "never overrides provider-independent evidence requirements"], + "constraints": ["bounded depth", "bounded parallelism", "bounded retries", "repair worker tier is enforced while independent judges remain frontier", "never overrides provider-independent evidence requirements"], "principle": "Use more compute when measured capability gaps or uncertainty justify it; avoid redundant exploration when verified strategies are reliable." }, "benchmark_protocol": { "required_before_claiming_improvement": ["fixed task set", "same model/provider configuration", "same tool budget where applicable", "multiple runs for reliability", "task success", "regression rate", "latency", "tokens/cost", "unnecessary context/code-read volume", "memory transfer across episodes", "persisted evidence for every numeric claim"], - "current_executable_suite": ["frontier-routing", "recursive-agent-graph", "cross-episode-repair-memory", "trace-verification", "program-graph-localization", "causal-trace-diagnosis", "targeted-context", "strategy-promotion"], + "current_executable_suite": ["frontier-routing", "recursive-agent-graph", "cross-episode-repair-memory", "trace-verification", "program-graph-localization", "causal-trace-diagnosis", "targeted-context", "strategy-promotion", "self-healing-runtime", "fault-injection-recovery", "stable-code-inspection"], "external_reference_directions": ["STATE-Bench", "EvoMemBench", "EvoAgentBench", "PAST-Bench"] }, - "known_limits": ["The deterministic runner measures architecture capabilities, not frontier-model task quality.", "Tool-use capability remains unmeasured by the provider-independent runner.", "Fresh GitHub Actions execution is not guaranteed to be visible through the current integration at every commit.", "Program/data/control-flow coverage is strongest for Python and should be extended to other languages before broad repo claims.", "Compute policy is a bounded controller; it does not train model weights or guarantee task-level quality improvements.", "Self-evolution changes remain evidence-gated; automatic model-weight training is not implemented.", "Self-derived reporting summarizes live architecture state but does not replace independent benchmark evaluation."] - ,"next_upgrade_targets": ["provider-backed end-to-end benchmark harness", "real runtime instrumentation and automatic trace capture", "sandbox patch execution with automatic rollback", "tool-use benchmark with explicit capability boundaries", "longitudinal benchmark runner with repeated runs and confidence intervals", "ability-transfer memory and curriculum generation", "model-vs-model and architecture-vs-architecture comparison harness", "predictive world-state and counterfactual simulation"] + "known_limits": ["The deterministic runner measures architecture capabilities, not frontier-model task quality.", "Tool-use capability remains unmeasured by the provider-independent runner.", "Fresh GitHub Actions execution is not guaranteed to be visible through the current integration at every commit.", "Program/data/control-flow coverage is strongest for Python and should be extended to other languages before broad repo claims.", "Compute policy is a bounded controller; it does not train model weights or guarantee task-level quality improvements.", "Self-evolution changes remain evidence-gated; automatic model-weight training is not implemented.", "Self-healing generates and validates repair episodes but production patch promotion remains an external gated step.", "Self-derived reporting summarizes live architecture state but does not replace independent benchmark evaluation."], + "next_upgrade_targets": ["provider-backed end-to-end benchmark harness", "real runtime instrumentation and automatic trace capture", "sandbox patch execution with automatic rollback", "tool-use benchmark with explicit capability boundaries", "longitudinal benchmark runner with repeated runs and confidence intervals", "ability-transfer memory and curriculum generation", "model-vs-model and architecture-vs-architecture comparison harness", "predictive world-state and counterfactual simulation"] }