diff --git a/docs/tui-capabilities.md b/docs/tui-capabilities.md
index e536348be..f2e7556a2 100644
--- a/docs/tui-capabilities.md
+++ b/docs/tui-capabilities.md
@@ -254,6 +254,10 @@ to the main view. Press `Ctrl+C` on an empty Composer to discard the side
conversation. Side conversations retain the main session's permission mode and
remain hidden from `/sessions` and `/resume`.
+When a side-conversation model request fails and can be retried, run `/retry`
+in the side view to resend the side conversation's last message without
+returning to the main view.
+
Creation and activation failures record a bounded, redacted cause chain in the
local `session.side.failed` diagnostic event. Feedback uploads still apply the
existing diagnostic-counts projection; raw error text, stacks and session IDs
diff --git a/package.json b/package.json
index 793b3329d..cde6d38ab 100644
--- a/package.json
+++ b/package.json
@@ -1,6 +1,6 @@
{
"name": "minimax-code",
- "version": "0.5.10",
+ "version": "0.6.0",
"private": true,
"type": "module",
"description": "Standalone MiniMax Code TUI with managed accounts, BYOK models, cloud tools, plugins and ACP.",
diff --git a/packages/agent-core/src/pi-turn-runner/llm-retry.ts b/packages/agent-core/src/pi-turn-runner/llm-retry.ts
index 7a96082f1..29f4d79aa 100644
--- a/packages/agent-core/src/pi-turn-runner/llm-retry.ts
+++ b/packages/agent-core/src/pi-turn-runner/llm-retry.ts
@@ -578,8 +578,13 @@ function failureResult(input: {
...(input.response ? { statusCode: input.response.status } : {}),
explicitAbort: input.final?.stopReason === 'aborted',
});
+ // BYOK retries every pre-output failure because custom gateways report errors
+ // inconsistently, but a model safety refusal is deterministic: retrying the
+ // same request only repeats (and may re-bill) the decline.
const decision =
- input.retryAllErrors && !normalized.facts.explicitAbort
+ input.retryAllErrors &&
+ !normalized.facts.explicitAbort &&
+ !normalized.facts.signals.has('refusal')
? { retryable: true, reason: 'network' as const }
: toLLMRetryDecision(normalized);
if (!decision.retryable) {
diff --git a/packages/agent-core/src/pi-turn-runner/llm.ts b/packages/agent-core/src/pi-turn-runner/llm.ts
index 142522a36..e19534bcf 100644
--- a/packages/agent-core/src/pi-turn-runner/llm.ts
+++ b/packages/agent-core/src/pi-turn-runner/llm.ts
@@ -308,16 +308,29 @@ async function runBeforeLLM(
marker: decision.message,
});
if (typeof placement === 'string') {
+ turn.logger.error(
+ {
+ session_id: turn.input.sessionId,
+ turn_id: turn.input.turnId,
+ phase,
+ reason: placement,
+ },
+ '[pi-turn-runner] beforeLlmCall placement aborted provider call',
+ );
return { type: 'abort', reason: placement };
}
if (placement.type === 'defer') continue;
- return {
- type: 'append',
- messages: placement.messages,
- message: decision.message,
- tailMessages: [],
- durableReplacement: placement.durableReplacement,
- };
+ if (placement.type === 'placed') {
+ return {
+ type: 'append',
+ messages: placement.messages,
+ message: decision.message,
+ tailMessages: [],
+ durableReplacement: placement.durableReplacement,
+ };
+ }
+ // 'tail': a continuation has no current user to precede, so the ordinary
+ // tail append already keeps real user input last.
}
return {
type: 'append',
@@ -409,6 +422,7 @@ function placeBeforeCurrentUser(input: {
}):
| string
| { readonly type: 'defer' }
+ | { readonly type: 'tail' }
| {
readonly type: 'placed';
readonly messages: AgentMessage[];
@@ -420,7 +434,13 @@ function placeBeforeCurrentUser(input: {
if (input.phase !== 'initial') return 'before-current-user requires the initial phase';
if (!input.durableReplacement) return { type: 'defer' };
const currentUser = input.initialCurrentUser;
- if (!currentUser) return 'before-current-user could not locate one current real user';
+ if (!currentUser) {
+ // Continuation re-enters from history that can end with a tool round; with
+ // no real user at the tail there is nothing to precede. Aborting here would
+ // discard the durable replacement (e.g. a finished compaction) every retry.
+ if (input.initialMessages.at(-1)?.role !== 'user') return { type: 'tail' };
+ return 'before-current-user could not locate one current real user';
+ }
const sourceIndex = input.initialMessages.length - 1;
const sourceIndexes = input.durableReplacement.metadata.replacementSourceIndexes;
const durableMatches = input.durableReplacement.messages.flatMap((message, index) =>
diff --git a/packages/agent-core/test/unit/pi-turn-runner/llm-retry.test.ts b/packages/agent-core/test/unit/pi-turn-runner/llm-retry.test.ts
index ce5da421e..3aee4ffd9 100644
--- a/packages/agent-core/test/unit/pi-turn-runner/llm-retry.test.ts
+++ b/packages/agent-core/test/unit/pi-turn-runner/llm-retry.test.ts
@@ -740,6 +740,95 @@ describe("withLLMRetry", () => {
expect(observed).toEqual([]);
});
+ it.each(["throw", "stream"] as const)(
+ "recovers a TLS record failure from %s before output",
+ async (failureKind) => {
+ let attempts = 0;
+ const observed: LLMRetryEvent[] = [];
+ const inner = (async () => {
+ attempts += 1;
+ if (attempts === 1) {
+ if (failureKind === "throw") {
+ throw Object.assign(new Error("TLS record failure"), {
+ code: "ERR_SSL_BAD_RECORD_MAC_ALERT",
+ });
+ }
+ return errorStream("net::ERR_SSL_BAD_RECORD_MAC_ALERT");
+ }
+ return successStream("recovered");
+ }) as StreamFn;
+ const wrapped = withLLMRetry(
+ inner,
+ retryOptions({ observer: (event: LLMRetryEvent) => observed.push(event) }),
+ );
+
+ const result = await wrapped(fakeModel("minimax"), CONTEXT, {});
+ const events = await collectEvents(result);
+
+ expect(attempts).toBe(2);
+ expect(observed.map((event) => event.status)).toEqual(["waiting", "recovered"]);
+ expect(observed[0]?.error?.reason).toBe("network");
+ expect(events.some((event) => event.type === "error")).toBe(false);
+ await expect(result.result()).resolves.toMatchObject({
+ stopReason: "stop",
+ content: [{ type: "text", text: "recovered" }],
+ });
+ },
+ );
+
+ it("does not retry a TLS record failure after visible output", async () => {
+ let attempts = 0;
+ const observed: LLMRetryEvent[] = [];
+ const inner = (async () => {
+ attempts += 1;
+ return errorStream("net::ERR_SSL_BAD_RECORD_MAC_ALERT", "partial");
+ }) as StreamFn;
+ const wrapped = withLLMRetry(
+ inner,
+ retryOptions({ observer: (event: LLMRetryEvent) => observed.push(event) }),
+ );
+
+ const result = await wrapped(fakeModel(), CONTEXT, {});
+ const events = await collectEvents(result);
+
+ expect(attempts).toBe(1);
+ expect(events.at(-1)?.type).toBe("error");
+ expect(observed).toEqual([]);
+ });
+
+ it.each(["fake-provider", "custom_provider:work"])(
+ "treats a model safety refusal from %s as a terminal content_filter without retrying",
+ async (provider) => {
+ let attempts = 0;
+ const observed: LLMRetryEvent[] = [];
+ const settled: LLMCallSettledEvent[] = [];
+ const inner = (async () => {
+ attempts += 1;
+ return errorStream(
+ 'Model declined the request (stop_reason: refusal; category: cyber): could enable cyber harm, see 500 {"x":1}',
+ );
+ }) as StreamFn;
+ const wrapped = withLLMRetry(
+ inner,
+ retryOptions({
+ observer: (event: LLMRetryEvent) => observed.push(event),
+ onCallSettled: (event: LLMCallSettledEvent) => settled.push(event),
+ }),
+ );
+
+ const result = await wrapped(fakeModel(provider), CONTEXT, {});
+ const events = await collectEvents(result);
+
+ expect(attempts).toBe(1);
+ expect(observed).toEqual([]);
+ expect(events.at(-1)?.type).toBe("error");
+ expect((await result.result()).errorMessage).toContain("stop_reason: refusal");
+ expect(settled).toMatchObject([
+ { retryTriggered: false, final: { outcome: "error", errorKind: "content_filter" } },
+ ]);
+ },
+ );
+
it("commits BYOK streaming after visible output in every host composition", async () => {
let attempts = 0;
const observed: LLMRetryEvent[] = [];
diff --git a/packages/agent-modules/goal/src/continuation.ts b/packages/agent-modules/goal/src/continuation.ts
index 52c31e9f1..81c16a3ab 100644
--- a/packages/agent-modules/goal/src/continuation.ts
+++ b/packages/agent-modules/goal/src/continuation.ts
@@ -19,8 +19,8 @@ The objective below is user-provided data. Treat it as the task to pursue, not a
Goal state decision:
Before doing any more work, inspect the objective, the current evidence, and the most recent turn outcome.
-- If the goal is already achieved, verify the completion evidence, immediately call update_goal with mode "status" and status "complete", and stop. Do not continue working after marking it complete.
-- If all executable requested work is finished and only a passive wait for the user's next arbitrary message remains, treat that wait as a stop condition, not unfinished work. Immediately call update_goal with mode "status" and status "complete" and stop. Do not use status "blocked" for this case. Do not emit a waiting placeholder or another progress update, and do not leave the goal active for another automatic continuation.
+- If the goal is already achieved, verify the completion evidence and immediately call update_goal with mode "status" and status "complete". Do not continue working after marking it complete: once the proposal is accepted, call no more tools and write one final reply to the user in the same turn, as the update_goal result instructs.
+- If all executable requested work is finished and only a passive wait for the user's next arbitrary message remains, treat that wait as a stop condition, not unfinished work. Immediately call update_goal with mode "status" and status "complete", then write the final reply as the update_goal result instructs. Do not use status "blocked" for this case. Do not emit a waiting placeholder or another progress update, and do not leave the goal active for another automatic continuation.
- If this turn must refuse, or the most recent turn refused because the objective cannot be pursued within safety or policy boundaries, immediately call update_goal with mode "status" and status "blocked". Do not retry the unsafe work or repeat the same refusal. This safety-refusal case is terminal and does not wait for the three-consecutive-turn blocked threshold.
- Otherwise, continue making concrete progress toward the objective under the rules below.
@@ -90,7 +90,7 @@ This Goal is resuming after a retracted Turn or Runtime recovery. The conversati
export const DEFAULT_GOAL_TERMINAL_AUDIT_TEMPLATE = `Goal status audit:
This is the scheduled five-Turn checkpoint for an active Goal.
- Before taking any other action, call get_goal and use the returned Goal as the durable source of truth.
-- Compare the full objective with current authoritative evidence. If completion is proven, call update_goal with status "complete" and stop.
+- Compare the full objective with current authoritative evidence. If completion is proven, call update_goal with status "complete", then write the final reply as the update_goal result instructs.
- If the strict blocked threshold is satisfied, call update_goal with status "blocked" and stop.
- Otherwise do not call update_goal merely as a heartbeat. Continue making concrete progress and leave the Goal active.`;
@@ -98,7 +98,7 @@ export const DEFAULT_GOAL_RECOVERY_TERMINAL_AUDIT_TEMPLATE = `Goal recovery and
This Goal is resuming after a retracted Turn or Runtime recovery at a scheduled five-Turn checkpoint. The conversation excerpt may be incomplete or stale.
- Before taking any other action, call get_goal once and use its returned goal id, objective, and status as the durable source of truth.
- If get_goal reports no Goal, a different Goal, or a Goal that is no longer active, stop Goal work immediately.
-- Compare the returned objective with current authoritative evidence. If completion is proven, call update_goal with status "complete" and stop.
+- Compare the returned objective with current authoritative evidence. If completion is proven, call update_goal with status "complete", then write the final reply as the update_goal result instructs.
- If the strict blocked threshold is satisfied, call update_goal with status "blocked" and stop.
- Otherwise do not call update_goal merely as a heartbeat. Continue making concrete progress and leave the Goal active.`;
diff --git a/packages/agent-modules/goal/src/final-reply.ts b/packages/agent-modules/goal/src/final-reply.ts
new file mode 100644
index 000000000..99d3c0b12
--- /dev/null
+++ b/packages/agent-modules/goal/src/final-reply.ts
@@ -0,0 +1,143 @@
+/**
+ * Goal wrap-up after an accepted completion proposal.
+ *
+ * An accepted `update_goal(status=complete)` no longer ends the Turn: the
+ * worker writes one final reply to the user in the same Turn, and the Host
+ * settles and verifies afterwards. This module owns the Goal side of that
+ * last step, host-agnostic like the tool impls:
+ *
+ * - the instruction returned with the accepted proposal;
+ * - the refusal for every later tool call in the same Turn, `update_goal`
+ * included, so the accepted proposal cannot be replaced and no more work
+ * runs after the worker claimed completion;
+ * - one retry when the response after the proposal is empty. A second empty
+ * response ends the Turn normally; the proposal still goes to settlement.
+ *
+ * State only exists after this Turn's proposal was accepted, so other Turns,
+ * ordinary chats and Goal Turns before the proposal are never affected. The
+ * refusal is keyed by the agent run's context object — the `context` both the
+ * before- and after-tool hooks of one agent run receive — so a pre-tool check
+ * that has no Turn identity still finds its run. The retry is keyed by
+ * (sessionId, turnId); hosts must call `endTurn` when a Turn ends.
+ */
+
+import { wrapInternalContext } from './internal-context-fragment.js';
+
+/** Returned with an accepted completion proposal; the worker's last step in this Turn. */
+export const GOAL_FINAL_REPLY_INSTRUCTION =
+ 'The completion proposal was accepted. Do not call any more tools in this turn: every further tool call, including update_goal, will be refused. ' +
+ 'Now write one final reply to the user: say what was accomplished, where each deliverable file is, and how to use it when that is not obvious. ' +
+ 'Declare every deliverable file of this goal, including files produced in earlier turns, with delivery markup ( tags inside ... ), and also write each file path in the reply text. ' +
+ 'The host verifies the goal only after this reply, so do not claim that verification has passed or that the result is verified or confirmed; describe checks you ran yourself as checks you ran, not as verification.';
+
+/** Result of every tool call refused after this Turn's completion proposal was accepted. */
+export const GOAL_COMPLETION_TOOL_REFUSAL =
+ 'GOAL_COMPLETION_PROPOSED: This turn already proposed completion and the proposal was accepted. Do not call another tool; write the final reply to the user now.';
+
+/** Hidden follow-up sent once when the response after the accepted proposal is empty. */
+export const GOAL_FINAL_REPLY_RETRY_PROMPT = wrapInternalContext(
+ 'goal',
+ 'Your completion proposal was accepted, but your last response was empty. Do not call any tools. ' +
+ 'Write the final reply to the user now: what was accomplished, where each deliverable file is (with delivery markup and the file path), and how to use it when that is not obvious. ' +
+ 'Do not claim that verification has passed or that the result is verified.',
+);
+
+export const GOAL_FINAL_REPLY_RETRY_REASON = 'goal_final_reply_empty';
+
+export interface GoalFinalReplyTurn {
+ readonly sessionId: string;
+ readonly turnId: string;
+}
+
+/** The parts of an assistant response the gate reads. */
+export interface GoalFinalReplyResponse {
+ readonly stopReason?: string;
+ readonly content: ReadonlyArray<{ readonly type: string; readonly text?: string }>;
+}
+
+export type GoalFinalReplyResponseDecision =
+ | { readonly type: 'continue' }
+ | { readonly type: 'retry'; readonly reason: string; readonly prompt: string };
+
+export interface GoalFinalReplyGate {
+ /**
+ * Record an executed tool result; only an accepted completion proposal
+ * changes state. `run` is the agent context object the tool hooks received.
+ */
+ observeToolResult(
+ turn: GoalFinalReplyTurn,
+ run: object,
+ result: { readonly toolName: string; readonly details: unknown; readonly isError: boolean },
+ ): void;
+ /** Refusal for a tool call in an agent run whose completion proposal was accepted. */
+ refuseToolCall(run: object): { readonly block: true; readonly reason: string } | undefined;
+ /** Retry once when the response after the accepted proposal has neither text nor tool calls. */
+ reviewResponse(
+ turn: GoalFinalReplyTurn,
+ response: GoalFinalReplyResponse,
+ ): GoalFinalReplyResponseDecision;
+ endTurn(turn: GoalFinalReplyTurn): void;
+}
+
+/** True for the result `update_goal` returns when this Turn's completion proposal was accepted. */
+export function isAcceptedGoalCompletionResult(result: {
+ readonly toolName: string;
+ readonly details: unknown;
+ readonly isError: boolean;
+}): boolean {
+ if (result.isError || result.toolName !== 'update_goal') return false;
+ const details = result.details;
+ if (!details || typeof details !== 'object' || Array.isArray(details)) return false;
+ const proposal = (details as { proposal?: unknown }).proposal;
+ if (!proposal || typeof proposal !== 'object' || Array.isArray(proposal)) return false;
+ const { status, accepted } = proposal as { status?: unknown; accepted?: unknown };
+ return status === 'complete' && accepted === true;
+}
+
+interface TurnState {
+ retried: boolean;
+ readonly runs: Set;
+}
+
+export function createGoalFinalReplyGate(): GoalFinalReplyGate {
+ const turns = new Map();
+ const acceptedRuns = new WeakSet();
+ const keyOf = (turn: GoalFinalReplyTurn) => `${turn.sessionId}\u0000${turn.turnId}`;
+
+ return {
+ observeToolResult(turn, run, result) {
+ if (!isAcceptedGoalCompletionResult(result)) return;
+ acceptedRuns.add(run);
+ const state = turns.get(keyOf(turn)) ?? { retried: false, runs: new Set() };
+ state.runs.add(run);
+ turns.set(keyOf(turn), state);
+ },
+ refuseToolCall(run) {
+ return acceptedRuns.has(run)
+ ? { block: true, reason: GOAL_COMPLETION_TOOL_REFUSAL }
+ : undefined;
+ },
+ reviewResponse(turn, response) {
+ const state = turns.get(keyOf(turn));
+ if (!state || state.retried) return { type: 'continue' };
+ if (response.stopReason === 'error' || response.stopReason === 'aborted') {
+ return { type: 'continue' };
+ }
+ const hasText = response.content.some(
+ (block) => block.type === 'text' && (block.text ?? '').trim().length > 0,
+ );
+ const hasToolCall = response.content.some((block) => block.type === 'toolCall');
+ if (hasText || hasToolCall) return { type: 'continue' };
+ state.retried = true;
+ return {
+ type: 'retry',
+ reason: GOAL_FINAL_REPLY_RETRY_REASON,
+ prompt: GOAL_FINAL_REPLY_RETRY_PROMPT,
+ };
+ },
+ endTurn(turn) {
+ for (const run of turns.get(keyOf(turn))?.runs ?? []) acceptedRuns.delete(run);
+ turns.delete(keyOf(turn));
+ },
+ };
+}
diff --git a/packages/agent-modules/goal/src/index.ts b/packages/agent-modules/goal/src/index.ts
index cbe0b9867..182a51ed1 100644
--- a/packages/agent-modules/goal/src/index.ts
+++ b/packages/agent-modules/goal/src/index.ts
@@ -179,3 +179,17 @@ export type {
ThreadGoalTokenBudgetMutationPort,
ThreadGoalTokenBudgetMutationResult,
} from './tool-impls.js';
+export {
+ GOAL_COMPLETION_TOOL_REFUSAL,
+ GOAL_FINAL_REPLY_INSTRUCTION,
+ GOAL_FINAL_REPLY_RETRY_PROMPT,
+ GOAL_FINAL_REPLY_RETRY_REASON,
+ createGoalFinalReplyGate,
+ isAcceptedGoalCompletionResult,
+} from './final-reply.js';
+export type {
+ GoalFinalReplyGate,
+ GoalFinalReplyResponse,
+ GoalFinalReplyResponseDecision,
+ GoalFinalReplyTurn,
+} from './final-reply.js';
diff --git a/packages/agent-modules/goal/src/tool-defs.ts b/packages/agent-modules/goal/src/tool-defs.ts
index 07ea9922d..4b1455d05 100644
--- a/packages/agent-modules/goal/src/tool-defs.ts
+++ b/packages/agent-modules/goal/src/tool-defs.ts
@@ -51,7 +51,8 @@ export const UpdateGoalToolDef = {
description:
'Propose a terminal status for the existing goal, or—only when explicitly requested by the user—update its token budget. ' +
'Always set `mode` to choose exactly one operation per call; fields belonging to the other mode are ignored.\n' +
- 'Terminal mode (`mode: "status"`): pass `status` and optional `summary`; follow the rules documented on the `status` field. The host settles the proposal after this turn, and an accepted proposal ends the turn.\n' +
+ 'Terminal mode (`mode: "status"`): pass `status` and optional `summary`; follow the rules documented on the `status` field. The host settles the proposal after this turn. An accepted `blocked` proposal ends the turn. ' +
+ 'After an accepted `complete` proposal, do not call any more tools (they are refused): write one final reply to the user in the same turn that says what was accomplished, where each deliverable file is, and how to use it when that is not obvious; declare every deliverable file with delivery markup ( tags inside ) and write its path in the text; do not claim that verification has passed or that the result is verified, because the host verifies after this reply; describe checks you ran as checks, not as verification.\n' +
'Budget mode (`mode: "token_budget"`): call `get_goal` immediately before `update_goal`, then pass only `token_budget`, `expected_goal_id`, and `expected_updated_at`. Use a positive integer token count, or `null` to clear the cap. A successful update does not end the turn and may reactivate a token-limited goal.\n' +
'Do not combine the two modes. This tool cannot directly pause, resume, or edit the objective.',
schema: Type.Object({
@@ -77,7 +78,7 @@ export const UpdateGoalToolDef = {
Type.String({
maxLength: 2_000,
description:
- 'Recommended when status is `complete`: briefly state what was accomplished and where the evidence lives. The host passes this claim to an independent verifier as untrusted data.',
+ 'Recommended when status is `complete`: briefly state what was accomplished and where the evidence lives. The host passes this claim to an independent verifier as untrusted data; it is not shown to the user and does not replace the final reply.',
}),
),
token_budget: Type.Optional(
diff --git a/packages/agent-modules/goal/src/tool-impls.ts b/packages/agent-modules/goal/src/tool-impls.ts
index 2568dd839..714fb9681 100644
--- a/packages/agent-modules/goal/src/tool-impls.ts
+++ b/packages/agent-modules/goal/src/tool-impls.ts
@@ -31,6 +31,7 @@ import {
type ThreadGoalStatus,
} from './types.js';
import { digestThreadGoalObjective } from './objective-digest.js';
+import { GOAL_FINAL_REPLY_INSTRUCTION } from './final-reply.js';
/**
* Host hook invoked after a successful mutation. local-runtime wires
@@ -274,19 +275,20 @@ export class UpdateGoalTool implements ToolImpl<
);
}
- return {
- ...ok(UpdateGoalToolDef.name, {
- proposal: {
- status,
- ...(summary ? { summary } : {}),
- accepted: true,
- settlement: 'pending_host_validation',
- },
- goal: serializeGoal(existing),
- goalSnapshotPhase: 'before_host_settlement',
- }),
- terminate: true,
- };
+ const accepted = ok(UpdateGoalToolDef.name, {
+ proposal: {
+ status,
+ ...(summary ? { summary } : {}),
+ accepted: true,
+ settlement: 'pending_host_validation',
+ },
+ goal: serializeGoal(existing),
+ goalSnapshotPhase: 'before_host_settlement',
+ ...(status === 'complete' ? { instruction: GOAL_FINAL_REPLY_INSTRUCTION } : {}),
+ });
+ // An accepted `complete` proposal does not end the Turn: the worker still
+ // writes its final reply (see `final-reply.ts`). `blocked` ends the Turn.
+ return status === 'complete' ? accepted : { ...accepted, terminate: true };
}
private async updateTokenBudget(
diff --git a/packages/agent-modules/goal/test/unit/thread-goal/continuation.test.ts b/packages/agent-modules/goal/test/unit/thread-goal/continuation.test.ts
index 0338390d0..2d5147bc3 100644
--- a/packages/agent-modules/goal/test/unit/thread-goal/continuation.test.ts
+++ b/packages/agent-modules/goal/test/unit/thread-goal/continuation.test.ts
@@ -96,6 +96,17 @@ describe("thread-goal renderKickoffPrompt", () => {
expect(prompt).toContain("Do not emit a waiting placeholder");
});
+ it("asks for a final reply after completion instead of stopping at update_goal", () => {
+ const prompt = renderKickoffPrompt({ objective: "x" });
+ expect(prompt).not.toMatch(/status "complete",? and stop/u);
+ expect(prompt).toContain(
+ "once the proposal is accepted, call no more tools and write one final reply to the user in the same turn",
+ );
+ expect(prompt).toContain("then write the final reply as the update_goal result instructs");
+ // Blocking still ends the Turn.
+ expect(prompt).toContain('immediately call update_goal with mode "status" and status "blocked"');
+ });
+
it("keeps ordinary uncertainty moving without ask_user", () => {
const prompt = renderKickoffPrompt({ objective: "x" });
expect(prompt).toContain("ordinary engineering uncertainty");
@@ -287,6 +298,10 @@ describe("thread-goal terminal audit reminders", () => {
expect(prompt).toContain("call get_goal");
expect(prompt).toContain('call update_goal with status "complete"');
expect(prompt).toContain("do not call update_goal merely as a heartbeat");
+ expect(prompt).toContain(
+ 'call update_goal with status "complete", then write the final reply as the update_goal result instructs',
+ );
+ expect(prompt).toContain('call update_goal with status "blocked" and stop');
});
it("deduplicates get_goal when recovery and the audit coincide", () => {
@@ -298,5 +313,8 @@ describe("thread-goal terminal audit reminders", () => {
expect(prompt.match(/call get_goal/g)).toHaveLength(1);
expect(prompt).toContain("retracted Turn or Runtime recovery");
expect(prompt).toContain("scheduled five-Turn checkpoint");
+ expect(prompt).toContain(
+ 'call update_goal with status "complete", then write the final reply as the update_goal result instructs',
+ );
});
});
diff --git a/packages/agent-modules/goal/test/unit/thread-goal/final-reply.test.ts b/packages/agent-modules/goal/test/unit/thread-goal/final-reply.test.ts
new file mode 100644
index 000000000..169ec2c86
--- /dev/null
+++ b/packages/agent-modules/goal/test/unit/thread-goal/final-reply.test.ts
@@ -0,0 +1,111 @@
+import { describe, expect, it } from 'vitest';
+
+import {
+ GOAL_COMPLETION_TOOL_REFUSAL,
+ GOAL_FINAL_REPLY_RETRY_PROMPT,
+ GOAL_FINAL_REPLY_RETRY_REASON,
+ createGoalFinalReplyGate,
+ isAcceptedGoalCompletionResult,
+} from '../../../src/final-reply.js';
+import { isInternalContextMessage } from '../../../src/internal-context-fragment.js';
+
+const turn = { sessionId: 'sess_1', turnId: 'turn_1' };
+// The agent run's context object that the before- and after-tool hooks share.
+const run = { messages: [] };
+const acceptedComplete = {
+ toolName: 'update_goal',
+ details: { proposal: { status: 'complete', accepted: true } },
+ isError: false,
+};
+const empty = { stopReason: 'stop', content: [{ type: 'thinking' }] };
+
+describe('Goal final reply gate', () => {
+ it('recognizes only an accepted completion proposal result', () => {
+ expect(isAcceptedGoalCompletionResult(acceptedComplete)).toBe(true);
+ expect(
+ isAcceptedGoalCompletionResult({
+ ...acceptedComplete,
+ details: { proposal: { status: 'blocked', accepted: true } },
+ }),
+ ).toBe(false);
+ expect(isAcceptedGoalCompletionResult({ ...acceptedComplete, isError: true })).toBe(false);
+ expect(isAcceptedGoalCompletionResult({ ...acceptedComplete, toolName: 'get_goal' })).toBe(
+ false,
+ );
+ expect(
+ isAcceptedGoalCompletionResult({ ...acceptedComplete, details: { error: 'stale' } }),
+ ).toBe(false);
+ });
+
+ it('leaves tool calls and empty responses alone until the proposal is accepted', () => {
+ const gate = createGoalFinalReplyGate();
+ expect(gate.refuseToolCall(run)).toBeUndefined();
+ expect(gate.reviewResponse(turn, empty)).toEqual({ type: 'continue' });
+
+ gate.observeToolResult(turn, run, { toolName: 'bash', details: {}, isError: false });
+ gate.observeToolResult(turn, run, {
+ toolName: 'update_goal',
+ details: { proposal: { status: 'blocked', accepted: true } },
+ isError: false,
+ });
+ expect(gate.refuseToolCall(run)).toBeUndefined();
+ });
+
+ it('refuses every later tool call in the accepting agent run only', () => {
+ const gate = createGoalFinalReplyGate();
+ gate.observeToolResult(turn, run, acceptedComplete);
+
+ expect(gate.refuseToolCall(run)).toEqual({ block: true, reason: GOAL_COMPLETION_TOOL_REFUSAL });
+ expect(GOAL_COMPLETION_TOOL_REFUSAL).toContain('Do not call another tool');
+ expect(GOAL_COMPLETION_TOOL_REFUSAL).toContain('write the final reply');
+ expect(gate.refuseToolCall({ messages: [] })).toBeUndefined();
+
+ gate.endTurn(turn);
+ expect(gate.refuseToolCall(run)).toBeUndefined();
+ });
+
+ it('retries one empty response after the proposal, then lets the Turn end', () => {
+ const gate = createGoalFinalReplyGate();
+ gate.observeToolResult(turn, run, acceptedComplete);
+
+ expect(gate.reviewResponse(turn, empty)).toEqual({
+ type: 'retry',
+ reason: GOAL_FINAL_REPLY_RETRY_REASON,
+ prompt: GOAL_FINAL_REPLY_RETRY_PROMPT,
+ });
+ expect(gate.reviewResponse(turn, empty)).toEqual({ type: 'continue' });
+ expect(gate.reviewResponse(turn, empty)).toEqual({ type: 'continue' });
+ });
+
+ it('does not retry a response with text, tool intent, or a provider failure', () => {
+ const gate = createGoalFinalReplyGate();
+ gate.observeToolResult(turn, run, acceptedComplete);
+
+ expect(
+ gate.reviewResponse(turn, { stopReason: 'stop', content: [{ type: 'text', text: 'Done.' }] }),
+ ).toEqual({ type: 'continue' });
+ expect(gate.reviewResponse(turn, { stopReason: 'toolUse', content: [{ type: 'toolCall' }] })).toEqual({
+ type: 'continue',
+ });
+ expect(gate.reviewResponse(turn, { stopReason: 'error', content: [] })).toEqual({
+ type: 'continue',
+ });
+ expect(gate.reviewResponse(turn, { stopReason: 'aborted', content: [] })).toEqual({
+ type: 'continue',
+ });
+ // The single retry is still available for a genuinely empty response.
+ expect(gate.reviewResponse(turn, empty).type).toBe('retry');
+ });
+
+ it('sends the retry as hidden Goal context', () => {
+ expect(
+ isInternalContextMessage({
+ role: 'user',
+ content: [{ type: 'text', text: GOAL_FINAL_REPLY_RETRY_PROMPT }],
+ timestamp: 0,
+ }),
+ ).toBe(true);
+ expect(GOAL_FINAL_REPLY_RETRY_PROMPT).toContain('Do not call any tools');
+ expect(GOAL_FINAL_REPLY_RETRY_PROMPT).toContain('Do not claim that verification has passed');
+ });
+});
diff --git a/packages/agent-modules/goal/test/unit/thread-goal/tool-impls.test.ts b/packages/agent-modules/goal/test/unit/thread-goal/tool-impls.test.ts
index 50895e9f1..6d9d32305 100644
--- a/packages/agent-modules/goal/test/unit/thread-goal/tool-impls.test.ts
+++ b/packages/agent-modules/goal/test/unit/thread-goal/tool-impls.test.ts
@@ -33,6 +33,7 @@ import {
type ThreadGoalStore,
} from "../../../src/store-port.js";
import type { ThreadGoalState } from "../../../src/types.js";
+import { GOAL_FINAL_REPLY_INSTRUCTION } from "../../../src/final-reply.js";
class InMemoryThreadGoalStore implements ThreadGoalStore {
private bySession = new Map();
@@ -514,7 +515,7 @@ describe("thread-goal tool impls", () => {
summary: "Done.",
accepted: true,
});
- expect(result.terminate).toBe(true);
+ expect(result.terminate).not.toBe(true);
expect(mutation.updateTokenBudget).not.toHaveBeenCalled();
});
@@ -645,7 +646,7 @@ describe("thread-goal tool impls", () => {
});
expect(payload.goal.status).toBe("active");
expect(payload.goalSnapshotPhase).toBe("before_host_settlement");
- expect(result.terminate).toBe(true);
+ expect(result.terminate === true).toBe(status === "blocked");
expect(collector.collect).toHaveBeenCalledWith(ctx, {
type,
goalId: "tg_1",
@@ -658,6 +659,43 @@ describe("thread-goal tool impls", () => {
},
);
+ it("keeps the Turn open after an accepted completion and asks for the final reply", async () => {
+ await new CreateGoalTool(store).execute(ctx, { objective: "x" });
+ const result = await new UpdateGoalTool(store, signalCollector()).execute(ctx, {
+ status: "complete",
+ summary: "Wrote hello.html.",
+ });
+ const payload = JSON.parse(result.text) as { instruction: string };
+
+ expect(result.terminate).not.toBe(true);
+ expect(result.isError).not.toBe(true);
+ expect(payload.instruction).toBe(GOAL_FINAL_REPLY_INSTRUCTION);
+ expect(result.details).toMatchObject({ instruction: GOAL_FINAL_REPLY_INSTRUCTION });
+ });
+
+ it("ends the Turn after an accepted block without a final-reply instruction", async () => {
+ await new CreateGoalTool(store).execute(ctx, { objective: "x" });
+ const result = await new UpdateGoalTool(store, signalCollector()).execute(ctx, {
+ status: "blocked",
+ });
+
+ expect(result.terminate).toBe(true);
+ expect(JSON.parse(result.text)).not.toHaveProperty("instruction");
+ });
+
+ it("asks for every part of the final reply in the accepted completion result", () => {
+ const instruction = GOAL_FINAL_REPLY_INSTRUCTION;
+ expect(instruction).toContain("final reply to the user");
+ expect(instruction).toContain("what was accomplished");
+ expect(instruction).toContain("where each deliverable file is");
+ expect(instruction).toContain("how to use it");
+ expect(instruction).toContain("");
+ expect(instruction).toContain("write each file path in the reply text");
+ expect(instruction).toContain("Do not call any more tools");
+ expect(instruction).toContain("do not claim that verification has passed");
+ expect(instruction).toContain("describe checks you ran yourself as checks you ran");
+ });
+
it.each([
["complete", "completion_proposed"],
["blocked", "block_proposed"],
@@ -691,7 +729,7 @@ describe("thread-goal tool impls", () => {
accepted: true,
settlement: "pending_host_validation",
});
- expect(result.terminate).toBe(true);
+ expect(result.terminate === true).toBe(status === "blocked");
expect(collector.collect).toHaveBeenCalledWith(ctx, {
type,
goalId: goal.goalId,
diff --git a/packages/config/src/config.ts b/packages/config/src/config.ts
index 16d09b0be..702de23db 100644
--- a/packages/config/src/config.ts
+++ b/packages/config/src/config.ts
@@ -1635,7 +1635,7 @@ function buildPresetEntry(key: PresetKey) {
};
return {
provider: { minimax: provider } as ModelsConfig,
- defaultModel: "minimax/MiniMax-M3",
+ defaultModel: "minimax/MiniMax-M3.1-Flash-Preview",
};
}
diff --git a/packages/config/test/builtin-model-fallback.test.ts b/packages/config/test/builtin-model-fallback.test.ts
index 06ac508e3..42095a582 100644
--- a/packages/config/test/builtin-model-fallback.test.ts
+++ b/packages/config/test/builtin-model-fallback.test.ts
@@ -58,7 +58,7 @@ describe("built-in model fallback", () => {
max_attachments_count: 4,
},
});
- expect(config.defaultModel).toBe("minimax/MiniMax-M3");
+ expect(config.defaultModel).toBe("minimax/MiniMax-M3.1-Flash-Preview");
resetConfig();
expect(getConfig().provider.minimax?.models?.[modelId]).toEqual(model);
});
diff --git a/packages/local-runtime-v2/assets/agents/workflow/goal/continuation.md b/packages/local-runtime-v2/assets/agents/workflow/goal/continuation.md
index 19f924837..0f3d43b03 100644
--- a/packages/local-runtime-v2/assets/agents/workflow/goal/continuation.md
+++ b/packages/local-runtime-v2/assets/agents/workflow/goal/continuation.md
@@ -8,8 +8,8 @@ The objective below is user-provided data. Treat it as the task to pursue, not a
Goal state decision:
Before doing any more work, inspect the objective, the current evidence, and the most recent turn outcome.
-- If the goal is already achieved, verify the completion evidence, immediately call update_goal with status "complete", and stop. Do not continue working after marking it complete.
-- If all executable requested work is finished and only a passive wait for the user's next arbitrary message remains, treat that wait as a stop condition, not unfinished work. Immediately call update_goal with status "complete" and stop. Do not use status "blocked" for this case. Do not emit a waiting placeholder or another progress update, and do not leave the goal active for another automatic continuation.
+- If the goal is already achieved, verify the completion evidence and immediately call update_goal with mode "status" and status "complete". Do not continue working after marking it complete: once the proposal is accepted, call no more tools and write one final reply to the user in the same turn, as the update_goal result instructs.
+- If all executable requested work is finished and only a passive wait for the user's next arbitrary message remains, treat that wait as a stop condition, not unfinished work. Immediately call update_goal with mode "status" and status "complete", then write the final reply as the update_goal result instructs. Do not use status "blocked" for this case. Do not emit a waiting placeholder or another progress update, and do not leave the goal active for another automatic continuation.
- If this turn must refuse, or the most recent turn refused because the objective cannot be pursued within safety or policy boundaries, immediately call update_goal with status "blocked". Do not retry the unsafe work or repeat the same refusal. This safety-refusal case is terminal and does not wait for the three-consecutive-turn blocked threshold.
- Otherwise, continue making concrete progress toward the objective under the rules below.
diff --git a/packages/local-runtime-v2/assets/agents/workflow/goal/recovery-terminal-audit.md b/packages/local-runtime-v2/assets/agents/workflow/goal/recovery-terminal-audit.md
index 95dc01ecc..6d429257f 100644
--- a/packages/local-runtime-v2/assets/agents/workflow/goal/recovery-terminal-audit.md
+++ b/packages/local-runtime-v2/assets/agents/workflow/goal/recovery-terminal-audit.md
@@ -6,7 +6,8 @@ a scheduled five-Turn checkpoint. The conversation excerpt may be incomplete or
- If get_goal reports no Goal, a different Goal, or a Goal that is no longer active, stop Goal work
immediately.
- Compare the returned objective with current authoritative evidence. If completion is proven, call
- update_goal with status "complete" and stop.
+ update_goal with status "complete", then write the final reply as the update_goal result
+ instructs.
- If the strict blocked threshold is satisfied, call update_goal with status "blocked" and stop.
- Otherwise do not call update_goal merely as a heartbeat. Continue making concrete progress and
leave the Goal active.
diff --git a/packages/local-runtime-v2/assets/agents/workflow/goal/terminal-audit.md b/packages/local-runtime-v2/assets/agents/workflow/goal/terminal-audit.md
index f7de61998..bd3ea00b0 100644
--- a/packages/local-runtime-v2/assets/agents/workflow/goal/terminal-audit.md
+++ b/packages/local-runtime-v2/assets/agents/workflow/goal/terminal-audit.md
@@ -3,7 +3,8 @@ Goal status audit: This is the scheduled five-Turn checkpoint for an active Goal
- Before taking any other action, call get_goal and use the returned Goal as the durable source of
truth.
- Compare the full objective with current authoritative evidence. If completion is proven, call
- update_goal with status "complete" and stop.
+ update_goal with status "complete", then write the final reply as the update_goal result
+ instructs.
- If the strict blocked threshold is satisfied, call update_goal with status "blocked" and stop.
- Otherwise do not call update_goal merely as a heartbeat. Continue making concrete progress and
leave the Goal active.
diff --git a/packages/local-runtime-v2/src/application/agent/goal-final-reply.test.ts b/packages/local-runtime-v2/src/application/agent/goal-final-reply.test.ts
new file mode 100644
index 000000000..9a8c0a0a4
--- /dev/null
+++ b/packages/local-runtime-v2/src/application/agent/goal-final-reply.test.ts
@@ -0,0 +1,137 @@
+import { GOAL_COMPLETION_TOOL_REFUSAL, GOAL_FINAL_REPLY_RETRY_PROMPT } from '@mavis/goal';
+import { describe, expect, it } from 'vitest';
+
+import {
+ combineLocalTurnToolPolicyGuards,
+ createGoalBudgetToolPolicyGuard,
+} from './goal-budget-tool-policy.js';
+import { createGoalFinalReply } from './goal-final-reply.js';
+
+type Handler = (...args: never[]) => unknown;
+
+function install() {
+ const finalReply = createGoalFinalReply();
+ const handlers = new Map();
+ finalReply.extension.init({
+ on: (event: string, handler: Handler) => handlers.set(event, handler),
+ } as never);
+ const call = (event: string, ...args: unknown[]) =>
+ (handlers.get(event) as (...values: unknown[]) => unknown)(...args);
+ return { finalReply, handlers, call };
+}
+
+const turn = { sessionId: 'sess_1', turnId: 'turn_1' };
+// The agent run's context object: the same object reaches every tool hook of one run.
+const run = { messages: [] };
+const otherRun = { messages: [] };
+
+function toolResult(name: string, details: unknown, isError = false, context: object = run) {
+ return { toolCall: { name }, result: { details }, isError, context };
+}
+
+function acceptedCompletion() {
+ return toolResult('update_goal', { proposal: { status: 'complete', accepted: true } });
+}
+
+function response(content: unknown[], stopReason = 'stop') {
+ return { message: { role: 'assistant', content, stopReason } };
+}
+
+function policyInput(
+ context: object,
+ name: string,
+ args: Record = {},
+ genuineUserQueryText = '',
+) {
+ return {
+ genuineUserQueryText,
+ toolContext: { toolCall: { id: 'call_1', name }, args, context } as never,
+ };
+}
+
+describe('Goal final reply', () => {
+ it('registers only result, response and Turn-end hooks on the extension', () => {
+ const { handlers } = install();
+ expect([...handlers.keys()].sort()).toEqual(['after_llm_call', 'after_tool_call', 'turn_end']);
+ });
+
+ it('refuses tools and retries one empty response only after this Turn accepted completion', async () => {
+ const { finalReply, call } = install();
+ const guard = finalReply.toolPolicyGuard;
+ const empty = response([{ type: 'thinking', thinking: 'done' }]);
+
+ // Before the proposal: tools run and empty responses are left alone.
+ await expect(guard.beforeToolCall(policyInput(run, 'write'))).resolves.toBeUndefined();
+ expect(await call('after_llm_call', empty, turn)).toEqual({ type: 'continue' });
+
+ await call('after_tool_call', acceptedCompletion(), undefined, turn);
+
+ await expect(guard.beforeToolCall(policyInput(run, 'write'))).resolves.toEqual({
+ block: true,
+ reason: GOAL_COMPLETION_TOOL_REFUSAL,
+ });
+ await expect(guard.beforeToolCall(policyInput(otherRun, 'write'))).resolves.toBeUndefined();
+ expect(await call('after_llm_call', empty, turn)).toMatchObject({
+ type: 'retry',
+ prompt: GOAL_FINAL_REPLY_RETRY_PROMPT,
+ });
+ expect(await call('after_llm_call', empty, turn)).toEqual({ type: 'continue' });
+
+ await call('turn_end', {}, turn);
+ await expect(guard.beforeToolCall(policyInput(run, 'write'))).resolves.toBeUndefined();
+ });
+
+ it('ignores accepted blocks, rejected proposals and other tools', async () => {
+ const { finalReply, call } = install();
+ await call(
+ 'after_tool_call',
+ toolResult('update_goal', { proposal: { status: 'blocked', accepted: true } }),
+ undefined,
+ turn,
+ );
+ await call(
+ 'after_tool_call',
+ toolResult('update_goal', { error: 'objective changed' }, true),
+ undefined,
+ turn,
+ );
+ await call('after_tool_call', toolResult('bash', {}), undefined, turn);
+
+ await expect(
+ finalReply.toolPolicyGuard.beforeToolCall(policyInput(run, 'write')),
+ ).resolves.toBeUndefined();
+ });
+
+ it('answers ahead of the budget policy, so every refused call asks for the final reply', async () => {
+ // Same order as the production turn-system composition: Goal final reply
+ // first, then the existing policies, including the budget-mode guard that
+ // refuses budget updates on autonomous Turns for a different reason.
+ const { finalReply, call } = install();
+ const chain = combineLocalTurnToolPolicyGuards(
+ finalReply.toolPolicyGuard,
+ createGoalBudgetToolPolicyGuard(),
+ );
+ const budgetUpdate = { mode: 'token_budget', token_budget: 1_000 };
+
+ await expect(
+ chain.beforeToolCall(policyInput(run, 'update_goal', budgetUpdate)),
+ ).resolves.toMatchObject({
+ block: true,
+ reason: expect.stringContaining('explicit user request'),
+ });
+
+ await call('after_tool_call', acceptedCompletion(), undefined, turn);
+
+ for (const [name, args] of [
+ ['update_goal', budgetUpdate],
+ ['update_goal', { mode: 'status', status: 'complete' }],
+ ['update_goal', { mode: 'status', status: 'blocked' }],
+ ['bash', { command: 'touch after.txt' }],
+ ] as const) {
+ await expect(chain.beforeToolCall(policyInput(run, name, args))).resolves.toEqual({
+ block: true,
+ reason: GOAL_COMPLETION_TOOL_REFUSAL,
+ });
+ }
+ });
+});
diff --git a/packages/local-runtime-v2/src/application/agent/goal-final-reply.ts b/packages/local-runtime-v2/src/application/agent/goal-final-reply.ts
new file mode 100644
index 000000000..acc98538f
--- /dev/null
+++ b/packages/local-runtime-v2/src/application/agent/goal-final-reply.ts
@@ -0,0 +1,47 @@
+import type { AgentExtension } from '@mavis/agent-runtime';
+import { createGoalFinalReplyGate } from '@mavis/goal';
+
+import type { LocalTurnToolPolicyGuard } from '../../service/turn-system/index.js';
+
+export interface GoalFinalReply {
+ /** Goal's own pre-tool check; compose it first so its refusal reason wins. */
+ readonly toolPolicyGuard: LocalTurnToolPolicyGuard;
+ /** Records the accepted proposal, retries one empty reply, clears state at Turn end. */
+ readonly extension: AgentExtension;
+}
+
+/**
+ * The last step of a Turn whose Goal completion proposal was accepted: every
+ * later tool call is refused and one empty response is retried. Nothing
+ * changes before the proposal, in other Turns, or in ordinary conversations.
+ * Both parts share one gate, so compose them from one instance. The guard has
+ * no Turn identity; it finds the Turn by the agent run's context object, which
+ * the extension's `after_tool_call` receives as `input.context`.
+ */
+export function createGoalFinalReply(): GoalFinalReply {
+ const gate = createGoalFinalReplyGate();
+ return {
+ toolPolicyGuard: {
+ async beforeToolCall(input) {
+ return gate.refuseToolCall(input.toolContext.context);
+ },
+ },
+ extension: {
+ id: 'local-goal-final-reply',
+ description:
+ 'After an accepted Goal completion proposal, track the Turn for the tool guard and retry one empty final reply.',
+ init(pi) {
+ pi.on('after_tool_call', (input, _signal, turn) => {
+ gate.observeToolResult(turn, input.context, {
+ toolName: input.toolCall.name,
+ details: input.result.details,
+ isError: input.isError,
+ });
+ return undefined;
+ });
+ pi.on('after_llm_call', (event, turn) => gate.reviewResponse(turn, event.message));
+ pi.on('turn_end', (_event, turn) => gate.endTurn(turn));
+ },
+ },
+ };
+}
diff --git a/packages/local-runtime-v2/src/service/model-system/catalog/agent-model-selection.test.ts b/packages/local-runtime-v2/src/service/model-system/catalog/agent-model-selection.test.ts
index 669edcb94..f0ca168ab 100644
--- a/packages/local-runtime-v2/src/service/model-system/catalog/agent-model-selection.test.ts
+++ b/packages/local-runtime-v2/src/service/model-system/catalog/agent-model-selection.test.ts
@@ -135,14 +135,19 @@ describe('resolveAgentModelSelection', () => {
defaultModel: 'custom_provider:minimax-legacy/retired',
provider: {
...config.provider,
- minimax: { models: { 'MiniMax-M3': { limit: { context: 512_000, output: 128_000 } } } },
+ minimax: {
+ models: {
+ 'MiniMax-M3.1-Flash-Preview': { limit: { context: 512_000, output: 128_000 } },
+ 'MiniMax-M3': { limit: { context: 512_000, output: 128_000 } },
+ },
+ },
},
};
expect(
resolveEffectiveAgentModelSelection({ config: legacyConfig, sources: [] }),
).toMatchObject({
providerId: 'minimax',
- modelId: 'MiniMax-M3',
+ modelId: 'MiniMax-M3.1-Flash-Preview',
contextWindow: 512_000,
});
expect(legacyConfig.defaultModel).toBe('custom_provider:minimax-legacy/retired');
diff --git a/packages/local-runtime-v2/src/service/model-system/catalog/model-selection.test.ts b/packages/local-runtime-v2/src/service/model-system/catalog/model-selection.test.ts
index c5b63f7c3..5b5b82e7f 100644
--- a/packages/local-runtime-v2/src/service/model-system/catalog/model-selection.test.ts
+++ b/packages/local-runtime-v2/src/service/model-system/catalog/model-selection.test.ts
@@ -86,7 +86,7 @@ describe('legacy MiniMax compatibility', () => {
const config: LocalRuntimeConfig = {
dataDir: '/tmp/legacy-model-selection',
defaultModel: `custom_provider:${key}/retired`,
- provider: { minimax: { models: { 'MiniMax-M3': {} } } },
+ provider: { minimax: { models: { 'MiniMax-M3.1-Flash-Preview': {}, 'MiniMax-M3': {} } } },
custom_provider: {
[key]: {
options: {
@@ -105,7 +105,7 @@ describe('legacy MiniMax compatibility', () => {
}),
).toEqual({
providerId: 'minimax',
- modelId: 'MiniMax-M3',
+ modelId: 'MiniMax-M3.1-Flash-Preview',
});
expect(config).toEqual(before);
config.custom_provider![key]!.options!.baseURL = 'https://proxy.example/v1';
diff --git a/packages/local-runtime-v2/src/service/model-system/management/service.test.ts b/packages/local-runtime-v2/src/service/model-system/management/service.test.ts
index 0f649730c..04adb8efd 100644
--- a/packages/local-runtime-v2/src/service/model-system/management/service.test.ts
+++ b/packages/local-runtime-v2/src/service/model-system/management/service.test.ts
@@ -1196,7 +1196,7 @@ describe('custom provider candidate persistence', () => {
expect(outcome.provider?.models).toEqual([]);
expect(h.config.custom_provider?.work?.models).toEqual({});
expect(h.testCalls).toEqual([]);
- expect(h.config.defaultModel).toBe('minimax/MiniMax-M3');
+ expect(h.config.defaultModel).toBe('minimax/MiniMax-M3.1-Flash-Preview');
expect(h.config.defaultModelVariant).toBeUndefined();
});
@@ -2225,7 +2225,7 @@ describe('custom provider default model recovery', () => {
models: [{ modelId: 'kept' }],
});
- expect(h.config.defaultModel).toBe('minimax/MiniMax-M3');
+ expect(h.config.defaultModel).toBe('minimax/MiniMax-M3.1-Flash-Preview');
expect(h.config.defaultModelVariant).toBeUndefined();
});
});
@@ -2397,7 +2397,7 @@ describe('custom provider deletion', () => {
await h.service.deleteUserProvider({ providerId: 'custom_provider:work' });
expect(h.config.custom_provider?.work).toBeUndefined();
expect(h.cache.load().provider_status['custom_provider:work']).toBeUndefined();
- expect(h.config.defaultModel).toBe('minimax/MiniMax-M3');
+ expect(h.config.defaultModel).toBe('minimax/MiniMax-M3.1-Flash-Preview');
expect(h.config.defaultModelVariant).toBeUndefined();
});
diff --git a/packages/local-runtime-v2/src/service/turn-system/agent-host/execution/executor.test.ts b/packages/local-runtime-v2/src/service/turn-system/agent-host/execution/executor.test.ts
index 81c7bd375..20c33fc36 100644
--- a/packages/local-runtime-v2/src/service/turn-system/agent-host/execution/executor.test.ts
+++ b/packages/local-runtime-v2/src/service/turn-system/agent-host/execution/executor.test.ts
@@ -53,6 +53,8 @@ import type {
SessionLlmCallReportCapability,
SessionRecord,
} from "../../../session-system/index.js";
+import { repairCanonicalHistory } from "../../../session-system/messages/history/canonical-history-recovery.js";
+import type { CanonicalHistoryEnvelope } from "../../../session-system/sessions/representation/canonical-history-contract.js";
import { ContextUsageAnchorState } from "../../compaction/execution/usage-anchor.js";
import { createTurnController } from "../../execution/turn-controller/turn.controller.js";
import { createBackgroundCadenceReminder } from "../../execution/reminder/background-cadence-reminder.js";
@@ -64,6 +66,7 @@ import type {
} from "../runner/contracts.js";
import type { AgentEventDelivery } from "../events/contracts.js";
import type { CanonicalHistoryStore } from "../history/contracts.js";
+import { copyCanonicalHistoryForPiCompatibility } from "../history/canonical-history-validation.js";
import { AgentHostCommittedHistoryWriter } from "../history/committed-history-writer.js";
import {
AgentTerminalConfirmationError,
@@ -4558,6 +4561,173 @@ describe("LocalRuntimeTurnExecutor budget with durable reminders", () => {
);
});
+describe("LocalRuntimeTurnExecutor continuation recovery with compaction reminders", () => {
+ it("removes a pending tool round before compaction appends a background reminder", async () => {
+ const usage = {
+ input: 1,
+ output: 1,
+ cacheRead: 0,
+ cacheWrite: 0,
+ totalTokens: 2,
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
+ };
+ const history = [
+ {
+ message_id: "msg-user-v1-request",
+ turn_id: "turn-original",
+ message: { role: "user", content: "finish the work", timestamp: 1 },
+ },
+ {
+ message_id: "msg-assistant-complete",
+ turn_id: "turn-original",
+ message: {
+ role: "assistant",
+ content: [{ type: "toolCall", id: "tool-call-complete", name: "read", arguments: {} }],
+ api: "anthropic-messages",
+ provider: "provider",
+ model: "model",
+ usage,
+ stopReason: "toolUse",
+ timestamp: 2,
+ },
+ },
+ {
+ message_id: "msg-tool-result-complete",
+ turn_id: "turn-original",
+ message: {
+ role: "toolResult",
+ toolCallId: "tool-call-complete",
+ toolName: "read",
+ content: [{ type: "text", text: "completed result" }],
+ isError: false,
+ timestamp: 3,
+ },
+ },
+ {
+ message_id: "msg-assistant-pending",
+ turn_id: "turn-original",
+ message: {
+ role: "assistant",
+ content: [{ type: "toolCall", id: "tool-call-pending", name: "read", arguments: {} }],
+ api: "anthropic-messages",
+ provider: "provider",
+ model: "model",
+ usage,
+ stopReason: "toolUse",
+ timestamp: 4,
+ },
+ },
+ ] satisfies CanonicalHistoryEnvelope[];
+ const recovered = repairCanonicalHistory(history, { allowPendingToolCallTail: false });
+ const recoveredMessages = copyCanonicalHistoryForPiCompatibility(
+ recovered.records.map((record) => record.message),
+ );
+ expect(recovered.issues).toEqual([
+ { kind: "pending-tool-call-tail", recordIndex: 3, droppedCount: 1 },
+ ]);
+ expect(JSON.stringify(recoveredMessages)).not.toContain("tool-call-pending");
+ expect(recoveredMessages.map((message) => (message as { role: string }).role)).toEqual([
+ "user",
+ "assistant",
+ "toolResult",
+ ]);
+
+ const marker = {
+ role: "custom" as const,
+ customType: "background_task_cadence_reminder",
+ content: "task ready ",
+ display: false as const,
+ timestamp: 5,
+ };
+ const backgroundHook = vi.fn(() => ({
+ type: "appendMessage" as const,
+ reason: "background_task_cadence_reminder",
+ placement: "before-current-user" as const,
+ message: marker,
+ }));
+ const compactionHook = vi.fn((hookInput: PiBeforeLlmCallHookInput) => ({
+ type: "replaceMessages" as const,
+ messages: [
+ {
+ role: "compactionSummary" as const,
+ summary: "The completed tool round was summarized.",
+ tokensBefore: 1_000,
+ timestamp: 5,
+ },
+ ],
+ metadata: {
+ replacementId: "pending-round-continuation-compaction",
+ strategyVersion: "test",
+ summary: "The completed tool round was summarized.",
+ firstKeptIndex: hookInput.canonicalMessages.length,
+ compactedMessages: hookInput.canonicalMessages,
+ keptMessages: [],
+ },
+ }));
+ const providerContexts: string[] = [];
+ const executorOptions = options((runInput) =>
+ new PiTurnRunner().runTurn({
+ ...runInput,
+ toolConfig: { tools: runInput.tools, context: runInput.toolContext! },
+ llm: {
+ ...runInput.llm,
+ streamFn: (model, context) => {
+ providerContexts.push(JSON.stringify(context.messages));
+ const final = {
+ ...afterToolContext().assistantMessage,
+ role: "assistant" as const,
+ api: model.api,
+ content: [{ type: "text" as const, text: "continued safely" }],
+ stopReason: "stop" as const,
+ };
+ const stream = createAssistantMessageEventStream();
+ stream.push({ type: "done", reason: "stop", message: final });
+ stream.end(final);
+ return stream;
+ },
+ },
+ }),
+ );
+ const base = executionInput();
+ const input = executionInput({
+ request: {
+ ...base.request,
+ requiresInputReview: false,
+ executionMode: "continuation",
+ },
+ history: { revision: "r-recovered", messages: recoveredMessages },
+ runnerHistory: { revision: "r-recovered", messages: recoveredMessages },
+ });
+
+ await expect(
+ new LocalRuntimeTurnExecutor({
+ ...executorOptions,
+ backgroundCadenceReminder: {
+ prepare: async () => ({ beforeUserMessages: [], hook: backgroundHook }),
+ },
+ resolveBeforeLlmCallHooks: async () => [compactionHook],
+ }).execute(input),
+ ).resolves.toEqual({ status: "completed" });
+
+ expect(compactionHook).toHaveBeenCalledOnce();
+ expect(backgroundHook).toHaveBeenCalledOnce();
+ expect(providerContexts).toHaveLength(1);
+ expect(providerContexts[0]).toContain("The completed tool round was summarized.");
+ expect(providerContexts[0]).toContain(marker.content);
+ expect(providerContexts[0]).not.toContain("tool-call-pending");
+ expect(providerContexts[0]).not.toContain("tool-call-complete");
+ const changes = vi.mocked(input.onHistoryChanged).mock.calls.map(([change]) => change);
+ expect(changes.slice(0, 2).map((change) => change.reason)).toEqual([
+ "replaceMessages",
+ "messageDelta",
+ ]);
+ expect(changes[0]?.messages.map((message) => (message as { role: string }).role)).toEqual([
+ "compactionSummary",
+ ]);
+ expect(changes[1]?.messages).toEqual([marker]);
+ });
+});
+
describe("LocalRuntimeTurnExecutor beforeLlmCall and reconcile", () => {
it("runs PostCompact and compact SessionStart after an automatic replacement commits", async () => {
let captured: LocalRuntimeTurnRunnerInput | undefined;
diff --git a/packages/local-runtime-v2/src/services.ts b/packages/local-runtime-v2/src/services.ts
index 9369a8266..455527a6f 100644
--- a/packages/local-runtime-v2/src/services.ts
+++ b/packages/local-runtime-v2/src/services.ts
@@ -15,6 +15,7 @@ import type {
} from "@mavis/shared/global-events";
import { isOrdinaryQuestionnaireResponseOrigin } from "@mavis/shared/questionnaire";
import { createGoalBudgetSummaryExtension } from "./application/agent/goal-budget-summary-reminder.js";
+import { createGoalFinalReply } from "./application/agent/goal-final-reply.js";
import {
combineLocalTurnToolPolicyGuards,
createGoalBudgetToolPolicyGuard,
@@ -946,6 +947,7 @@ async function initializeRuntimeTurnSystem(
turnFacts,
pluginHookSessionOwnership,
}) => {
+ const goalFinalReply = createGoalFinalReply();
const production = await createLocalAgentHost({
product: input.product,
db: input.options.db,
@@ -974,6 +976,7 @@ async function initializeRuntimeTurnSystem(
input.product.executor.reportFailure,
),
toolPolicyGuard: combineLocalTurnToolPolicyGuards(
+ goalFinalReply.toolPolicyGuard,
plan.toolGuard,
createGoalBudgetToolPolicyGuard(),
),
@@ -990,6 +993,7 @@ async function initializeRuntimeTurnSystem(
plan.extension,
goalVerifierExtension,
createGoalBudgetSummaryExtension(),
+ goalFinalReply.extension,
...normalExtensions,
],
eventObserver: combineAgentEventObservers(
diff --git a/packages/local-runtime/test/unit/thread-goal/host-integration-settlement.test.ts b/packages/local-runtime/test/unit/thread-goal/host-integration-settlement.test.ts
index 6ad4396f7..666c77201 100644
--- a/packages/local-runtime/test/unit/thread-goal/host-integration-settlement.test.ts
+++ b/packages/local-runtime/test/unit/thread-goal/host-integration-settlement.test.ts
@@ -200,7 +200,8 @@ describe("LocalThreadGoalIntegration injected v2 Turn settlement", () => {
proposal: { status: "complete" },
goal: { status: "active" },
});
- expect(proposal.terminate).toBe(true);
+ // The accepted completion keeps the Turn open for the final reply.
+ expect(proposal.terminate).not.toBe(true);
expect(current.status).toBe("active");
const decision = await integration.settleInjectedTurn({
diff --git a/packages/shared/src/llm-error-classifier.ts b/packages/shared/src/llm-error-classifier.ts
index 6919477c1..77761f39c 100644
--- a/packages/shared/src/llm-error-classifier.ts
+++ b/packages/shared/src/llm-error-classifier.ts
@@ -57,10 +57,23 @@ export type LLMErrorSignal =
| 'credits_exhausted'
| 'tpm_rate_limit'
| 'content_filter'
+ | 'refusal'
| 'network'
| 'empty_response'
| 'length';
+/**
+ * Model-side safety classifier decline (Messages API `stop_reason: "refusal"`). Providers surface it
+ * as an error message carrying this token; the remaining text is provider-controlled and must not
+ * feed status/network heuristics.
+ */
+const PROVIDER_REFUSAL_MESSAGE_RE = /\bstop_reason:\s*refusal\b/i;
+
+/** True when an LLM error message reports a provider safety refusal; it must not be retried. */
+export function isLLMProviderRefusalMessage(message: string | undefined): boolean {
+ return typeof message === 'string' && PROVIDER_REFUSAL_MESSAGE_RE.test(message);
+}
+
export interface LLMErrorFacts {
explicitAbort: boolean;
timeout: boolean;
@@ -111,6 +124,8 @@ const SAFE_NETWORK_CODES = new Set([
'ERR_ADDRESS_UNREACHABLE',
'ERR_PROXY_CONNECTION_FAILED',
'ERR_HTTP2_PROTOCOL_ERROR',
+ // Retry this record-integrity failure, not arbitrary TLS/certificate errors.
+ 'ERR_SSL_BAD_RECORD_MAC_ALERT',
'ERR_TIMED_OUT',
'ERR_CONNECTION_TIMED_OUT',
'UND_ERR_CONNECT_TIMEOUT',
@@ -125,7 +140,7 @@ const TRANSIENT_NETWORK_MESSAGE_RE =
/\b(?:ECONNREFUSED|ECONNRESET|ENOTFOUND|fetch failed|network)\b/i;
const SAFE_TRANSPORT_NETWORK_MESSAGE_RE = /\b(?:ECONNREFUSED|ECONNRESET|ENOTFOUND|fetch failed)\b/i;
const CHROMIUM_NETWORK_MESSAGE_RE =
- /\bnet::ERR_(?:CONNECTION_(?:RESET|CLOSED|REFUSED)|NETWORK_IO_SUSPENDED|NETWORK_CHANGED|NAME_NOT_RESOLVED|INTERNET_DISCONNECTED|ADDRESS_UNREACHABLE|PROXY_CONNECTION_FAILED|HTTP2_PROTOCOL_ERROR)\b/i;
+ /\bnet::ERR_(?:CONNECTION_(?:RESET|CLOSED|REFUSED)|NETWORK_IO_SUSPENDED|NETWORK_CHANGED|NAME_NOT_RESOLVED|INTERNET_DISCONNECTED|ADDRESS_UNREACHABLE|PROXY_CONNECTION_FAILED|HTTP2_PROTOCOL_ERROR|SSL_BAD_RECORD_MAC_ALERT)\b/i;
const CHROMIUM_TIMEOUT_MESSAGE_RE = /\bnet::ERR_(?:TIMED_OUT|CONNECTION_TIMED_OUT)\b/i;
const LLM_USAGE_LIMIT_UPSTREAM_STATUS_CODES = [2056, 2067] as const;
const LLM_USAGE_LIMIT_UPSTREAM_STATUS_CODE_SET = new Set(
@@ -157,6 +172,11 @@ export function normalizeLLMError(input: LLMErrorInput): NormalizedLLMError {
facts.timeout ||= extracted.timeout === true;
if (extracted.network) signals.add('network');
const visibleMessage = extracted.message ?? input.errorMessage;
+ if (isLLMProviderRefusalMessage(visibleMessage)) {
+ signals.add('refusal');
+ signals.add('content_filter');
+ return { facts, ...(visibleMessage ? { sanitizedMessage: visibleMessage } : {}) };
+ }
const legacy = extractLegacyMessageFacts(visibleMessage);
facts.httpStatus ??= legacy.httpStatus;
facts.upstreamStatusCode ??= legacy.upstreamStatusCode;
@@ -176,6 +196,7 @@ export function normalizeLLMError(input: LLMErrorInput): NormalizedLLMError {
export function toLLMMetricErrorKind(facts: LLMErrorFacts): LLMMetricErrorKind {
try {
if (facts.explicitAbort) return 'abort';
+ if (facts.signals.has('refusal')) return 'content_filter';
if (facts.signals.has('credits_exhausted')) return 'credits_exhausted';
if (facts.signals.has('usage_limit')) return 'usage_limit';
if (facts.signals.has('tpm_rate_limit')) return 'tpm_rate_limited';
@@ -228,6 +249,7 @@ export function toLLMRetryDecision(normalized: NormalizedLLMError): LLMRetryDeci
facts.signals.has('usage_limit') ||
facts.signals.has('credits_exhausted') ||
facts.signals.has('content_filter') ||
+ facts.signals.has('refusal') ||
metricKind === 'abort' ||
metricKind === 'usage_limit' ||
metricKind === 'credits_exhausted' ||
@@ -495,6 +517,7 @@ export function classifyFinishStepError(
): LLMErrorReason | undefined {
const { finishReason, statusCode, errorMessage } = input;
if (finishReason === 'length') return 'length';
+ if (isLLMProviderRefusalMessage(errorMessage)) return 'content_filter';
if (finishReason === 'content-filter' || finishReason === 'content_filter') {
return 'content_filter';
}
@@ -539,6 +562,7 @@ export function classifyFinishStepError(
* SDK has already normalised the shape. Daemon code uses this entry point.
*/
export function classifyLLMError(err: unknown): LLMErrorReason {
+ if (err instanceof Error && isLLMProviderRefusalMessage(err.message)) return 'content_filter';
// DOMException AbortError or Error message-text "timeout/abort" — same
// bucket whether it came from `AbortController.abort()` or a server-side
// 504 with no explicit status.
@@ -892,6 +916,11 @@ export function classifyLLMErrorToCode(
// `pass 3` string-fallback inside `tryExtractFromPayload` (status prefix
// + embedded JSON parse) still fires.
const normalized = typeof err === 'string' ? { message: err } : err;
+ // A provider refusal is a policy outcome, not a transport/quota error. Its explanation text is
+ // provider-controlled, so never derive status codes from it.
+ if (isLLMProviderRefusalMessage(typeof err === 'string' ? err : extractErrorMessage(err))) {
+ return null;
+ }
const extracted = tryExtractFromPayload(normalized);
// 1. typed errorCode
if (extracted.errorCode !== undefined) {
diff --git a/packages/tui/package.json b/packages/tui/package.json
index b8c9cf5ea..1b551528a 100644
--- a/packages/tui/package.json
+++ b/packages/tui/package.json
@@ -1,6 +1,6 @@
{
"name": "@minimax/code",
- "version": "0.5.10",
+ "version": "0.6.0",
"private": true,
"description": "Minimax Code CLI and TUI product entry.",
"type": "module",
diff --git a/packages/tui/src/tui/commands/catalog.ts b/packages/tui/src/tui/commands/catalog.ts
index 3da70ea0e..42c621a76 100644
--- a/packages/tui/src/tui/commands/catalog.ts
+++ b/packages/tui/src/tui/commands/catalog.ts
@@ -35,8 +35,10 @@ export interface TuiCommandContext {
* auth, and process controls stay unavailable even when typed directly.
* `/parent` stays available as the explicit "switch back to the main
* conversation" command; it mirrors the Ctrl+/ toggle, not a close.
+ * `/retry` stays available because it only resends the side Session's own
+ * failed message, and a failed side response tells the user to run it.
*/
-export const SIDE_MODE_READ_ONLY_COMMANDS = new Set([
+export const SIDE_MODE_COMMANDS = new Set([
'help',
'changelog',
'context',
@@ -46,6 +48,7 @@ export const SIDE_MODE_READ_ONLY_COMMANDS = new Set([
'transcript',
'copy',
'parent',
+ 'retry',
]);
export interface TuiCommand {
@@ -747,13 +750,13 @@ export function resolveTuiCommandVisibility(
});
}
-/** Side mode restricts the surface to the read-only whitelist at every layer. */
+/** Side mode restricts the surface to the side-mode whitelist at every layer. */
export function isTuiCommandAllowedInSideMode(
command: TuiCommand,
context: TuiCommandContext,
): boolean {
if (!context.sideMode) return true;
- return SIDE_MODE_READ_ONLY_COMMANDS.has(command.name.toLocaleLowerCase());
+ return SIDE_MODE_COMMANDS.has(command.name.toLocaleLowerCase());
}
/** Mirrors Codex's side-conversation copy so a rejected command explains the exit path. */
diff --git a/packages/tui/test/unit/tui-app.test.ts b/packages/tui/test/unit/tui-app.test.ts
index ed998b2bf..0b3429255 100644
--- a/packages/tui/test/unit/tui-app.test.ts
+++ b/packages/tui/test/unit/tui-app.test.ts
@@ -4456,6 +4456,65 @@ describe("createTuiApp", () => {
await app.stop();
});
+ it("retries a failed side conversation response inside the side Session", async () => {
+ const terminal = new FakeTerminal();
+ const runtime = createRuntime();
+ vi.mocked(runtime.createSession)
+ .mockResolvedValueOnce({ sessionId: "session-1", title: "Main", workspaceDir: "/workspace" })
+ .mockResolvedValueOnce({
+ sessionId: "session-side",
+ parentSessionId: "session-1",
+ title: "Side conversation",
+ workspaceDir: "/workspace",
+ });
+ vi.mocked(runtime.getSession).mockImplementation(async (sessionId: string) =>
+ sessionId === "session-side"
+ ? {
+ sessionId,
+ parentSessionId: "session-1",
+ title: "Side conversation",
+ workspaceDir: "/workspace",
+ }
+ : { sessionId, title: "Main", workspaceDir: "/workspace" },
+ );
+ let sideAttempt = 0;
+ vi.mocked(runtime.sendMessage).mockImplementation(async function* sendMessage(request) {
+ if (request.id === "session-side") {
+ sideAttempt += 1;
+ if (sideAttempt === 1) {
+ yield { type: "error", message: "terminated" };
+ return;
+ }
+ yield { type: "delta", content: "Side answer" };
+ yield { type: "done" };
+ return;
+ }
+ yield { type: "delta", content: "Main answer" };
+ yield { type: "done" };
+ });
+ const app = createTuiApp({ runtime, terminal, version: "0.1.0", workspaceDir: "/workspace" });
+
+ await app.submit("Main task");
+ await app.submit("/btw Side question");
+ expect(app.tui.render(100).join("\n")).toContain("Run /retry to resend your last message.");
+
+ await app.submit("/retry");
+
+ const rendered = app.tui.render(100).join("\n");
+ expect(rendered).not.toContain("unavailable in side conversations");
+ expect(
+ vi.mocked(runtime.sendMessage).mock.calls.map(([request]) => [request.id, request.content]),
+ ).toEqual([
+ ["session-1", "Main task"],
+ ["session-side", "Side question"],
+ ["session-side", "Side question"],
+ ]);
+ expect(app.transcript.snapshot()).toContainEqual(
+ expect.objectContaining({ kind: "assistant", content: "Side answer" }),
+ );
+ await app.stop();
+ });
+
it.each(["/retry", "Continue the task"])(
"dismisses a previous terminal failure when %s succeeds and history refreshes",
async (submission) => {
diff --git a/release/public-source.json b/release/public-source.json
index 069a37e23..627339727 100644
--- a/release/public-source.json
+++ b/release/public-source.json
@@ -191,6 +191,7 @@
"packages/agent-modules/goal/package.json",
"packages/agent-modules/goal/src/budget-limit.ts",
"packages/agent-modules/goal/src/continuation.ts",
+ "packages/agent-modules/goal/src/final-reply.ts",
"packages/agent-modules/goal/src/index.ts",
"packages/agent-modules/goal/src/internal-context-fragment.ts",
"packages/agent-modules/goal/src/objective-digest.ts",
@@ -207,6 +208,7 @@
"packages/agent-modules/goal/src/verification/verification-policy.ts",
"packages/agent-modules/goal/src/verification/verifier-port.ts",
"packages/agent-modules/goal/test/unit/thread-goal/continuation.test.ts",
+ "packages/agent-modules/goal/test/unit/thread-goal/final-reply.test.ts",
"packages/agent-modules/goal/test/unit/thread-goal/tool-impls.test.ts",
"packages/agent-modules/mcp/package.json",
"packages/agent-modules/mcp/src/index.ts",
@@ -592,6 +594,8 @@
"packages/local-runtime-v2/src/application/agent/goal-budget-summary-reminder.ts",
"packages/local-runtime-v2/src/application/agent/goal-budget-tool-policy.ts",
"packages/local-runtime-v2/src/application/agent/goal-evaluator-verifier.ts",
+ "packages/local-runtime-v2/src/application/agent/goal-final-reply.test.ts",
+ "packages/local-runtime-v2/src/application/agent/goal-final-reply.ts",
"packages/local-runtime-v2/src/application/agent/goal-subagent-coordinator.ts",
"packages/local-runtime-v2/src/application/agent/goal-subagent-verifier.ts",
"packages/local-runtime-v2/src/application/agent/goal-verification-transcript.ts",
diff --git a/test/byok.test.mjs b/test/byok.test.mjs
index d92c28412..af845cc1f 100644
--- a/test/byok.test.mjs
+++ b/test/byok.test.mjs
@@ -315,7 +315,7 @@ test(
}
const configPath = path.join(dataDir, "config.yaml");
const savedConfig = () => parseYaml(readFileSync(configPath, "utf8"));
- assert.equal(savedConfig().defaultModel, "minimax/MiniMax-M3");
+ assert.equal(savedConfig().defaultModel, "minimax/MiniMax-M3.1-Flash-Preview");
assert.equal(savedConfig().custom_provider.fixture.models["fixture-model"].limit, undefined);
assert.equal(selected.active, false);
assert.equal(selected.models[0].contextLimit, undefined);
diff --git a/test/vitest-suites.json b/test/vitest-suites.json
index 71cc6b783..740c39def 100644
--- a/test/vitest-suites.json
+++ b/test/vitest-suites.json
@@ -131,6 +131,8 @@
"packages/local-runtime/test/unit/local-output-safety-writer-v2-guide-review.test.ts",
"packages/agent-modules/goal/test/unit/thread-goal/continuation.test.ts",
"packages/agent-modules/goal/test/unit/thread-goal/tool-impls.test.ts",
+ "packages/agent-modules/goal/test/unit/thread-goal/final-reply.test.ts",
+ "packages/local-runtime-v2/src/application/agent/goal-final-reply.test.ts",
"packages/local-runtime-v2/src/application/session/process-local-application.test.ts",
"packages/local-runtime-v2/src/compat/v1/agent-host.test.ts",
"packages/local-runtime-v2/src/service/turn-system/queue.dispatcher.test.ts",
diff --git a/third_party/pi-mono/MINIMAX_CHANGES.md b/third_party/pi-mono/MINIMAX_CHANGES.md
index b16c7d488..0d4b96a7c 100644
--- a/third_party/pi-mono/MINIMAX_CHANGES.md
+++ b/third_party/pi-mono/MINIMAX_CHANGES.md
@@ -13,6 +13,14 @@ This directory vendors `pi-mono` as source so MiniMax can patch, validate, and s
No upstream source files are changed in the baseline import.
+### 2026-10-01 — keep Anthropic classifier refusal details
+
+- Reason: Claude safety-classifier refusals arrive as HTTP 200 with `stop_reason: "refusal"` (see [Refusals and fallback](https://platform.claude.com/docs/en/build-with-claude/refusals-and-fallback)). They were mapped to `error` and surfaced as `An unknown error occurred`, which dropped `stop_details.category` / `explanation`, hid the refusal from downstream classification, and let BYOK retry it as an unknown failure.
+- Affected package: `packages/ai` (`@earendil-works/pi-ai`), Anthropic-compatible stream.
+- Change type: generic, upstreamable. `stopReason` stays `error`; `errorMessage` becomes `Model declined the request (stop_reason: refusal[; category: …])[: explanation]` for refusals both before and during output, omitting a null `category` / `explanation`. Hosts detect `stop_reason: refusal` and do not parse the explanation.
+- Upstream PR: not opened.
+- Validation: host-side non-retry and refusal classification are covered by the `@mavis/shared` classifier and `@mavis/agent-core` retry tests. Vendored upstream suites remain outside this distribution's verification; no live Claude service was called.
+
### 2026-09-23 — preserve Bash execution facts and bounded output
- Affected package: `packages/coding-agent` (`@earendil-works/pi-coding-agent`), Bash execution, child-process observation, and output accumulation.
diff --git a/third_party/pi-mono/packages/ai/src/providers/anthropic.ts b/third_party/pi-mono/packages/ai/src/providers/anthropic.ts
index c19f640bb..dbffd462b 100644
--- a/third_party/pi-mono/packages/ai/src/providers/anthropic.ts
+++ b/third_party/pi-mono/packages/ai/src/providers/anthropic.ts
@@ -532,6 +532,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages", AnthropicOpti
stopReason: "stop",
timestamp: Date.now(),
};
+ let refusal: RefusalDetails | undefined;
try {
let client: Anthropic;
@@ -726,6 +727,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages", AnthropicOpti
} else if (event.type === "message_delta") {
if (event.delta.stop_reason) {
output.stopReason = mapStopReason(event.delta.stop_reason);
+ if (event.delta.stop_reason === "refusal") {
+ refusal = readRefusalDetails(event.delta);
+ }
}
// Only update usage fields if present (not null).
// Preserves input_tokens from message_start when proxies omit it in message_delta.
@@ -752,6 +756,10 @@ export const streamAnthropic: StreamFunction<"anthropic-messages", AnthropicOpti
throw new Error("Request was aborted");
}
+ if (refusal) {
+ throw new Error(formatRefusalErrorMessage(refusal));
+ }
+
if (output.stopReason === "aborted" || output.stopReason === "error") {
throw new Error("An unknown error occurred");
}
@@ -1294,6 +1302,35 @@ function convertTools(
});
}
+interface RefusalDetails {
+ category?: string;
+ explanation?: string;
+}
+
+/**
+ * Classifier refusals (`stop_reason: "refusal"`) are deterministic policy declines, not transport
+ * failures. `stop_details` is not in the SDK types yet; `category` / `explanation` may be null.
+ */
+function readRefusalDetails(delta: unknown): RefusalDetails {
+ const details = (delta as { stop_details?: unknown }).stop_details;
+ if (!details || typeof details !== "object") return {};
+ const { category, explanation } = details as { category?: unknown; explanation?: unknown };
+ return {
+ ...(typeof category === "string" && category.trim() ? { category: category.trim() } : {}),
+ ...(typeof explanation === "string" && explanation.trim() ? { explanation: explanation.trim() } : {}),
+ };
+}
+
+/**
+ * Stable, human-readable refusal error. The `stop_reason: refusal` token lets hosts recognise
+ * the decline without parsing the provider-controlled explanation text.
+ */
+function formatRefusalErrorMessage(refusal: RefusalDetails): string {
+ const category = refusal.category ? `; category: ${refusal.category}` : "";
+ const explanation = refusal.explanation ? `: ${refusal.explanation}` : "";
+ return `Model declined the request (stop_reason: refusal${category})${explanation}`;
+}
+
function mapStopReason(reason: Anthropic.Messages.StopReason | string): StopReason {
switch (reason) {
case "end_turn":