Files
fusion/packages/engine/src/__tests__/executor-tool-failure-retry.test.ts
gsxdsm 3ed31e9907 FN-8815: retry execution after first tool failure
Retry executor work after the first terminal tool-call failure by default.

- Set the default tool-failure threshold to one while preserving explicit project overrides.
- Expose and document the first-error default in Settings, translations, and the settings reference.
- Cover threshold resolution, save lifecycle, and executor retry behavior.

Files changed: .changeset/fn-8815-retry-first-tool-failure.md     |  7 +++
 docs/settings-reference.md                         |  4 +-
 .../core/src/__tests__/settings-defaults.test.ts   |  6 +-
 packages/core/src/config/settings-schema.ts        | 10 ++-
 packages/core/src/tasks/in-review-stall.ts         | 15 ++++-
 packages/core/src/types/settings/settings-scope.ts | 10 ++-
 .../dashboard/app/components/SettingsModal.tsx     |  7 ++-
 .../SettingsModal.scheduling-merge.test.tsx        | 61 +++++++++++++++++-
 .../settings/sections/SchedulingSection.search.ts  |  2 +-
 .../settings/sections/SchedulingSection.tsx        |  4 +-
 .../settings-default-descriptions.test.tsx         | 12 ++++
 .../__tests__/executor-tool-failure-retry.test.ts  | 73 +++++++++++++++++++---
 packages/i18n/locales/en/app.json                  |  2 +-
 packages/i18n/src/resources.d.ts                   | 17 +++--
 14 files changed, 200 insertions(+), 30 deletions(-)

Fusion-Task-Id: FN-8815

Fusion-Task-Lineage: 909181e7-2da5-4a27-92ee-4182732d9695

Co-authored-by: Fusion (runfusion.ai) <noreply@runfusion.ai>
2026-08-06 08:35:36 -07:00

323 lines
16 KiB
TypeScript

import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { resolveConsecutiveToolFailureRetryBackoffMs, resolveMaxConsecutiveToolFailureRetries, type TaskDetail } from "@fusion/core";
import "./executor-test-helpers.js";
import { TaskExecutor } from "../executor.js";
import { createMockStore, resetExecutorMocks } from "./executor-test-helpers.js";
const now = "2026-07-16T00:00:00.000Z";
function makeTask(overrides: Partial<TaskDetail> = {}): TaskDetail {
return {
id: "FN-7996",
title: "Tool failure retry",
description: "Reproduce executor tool errors",
column: "in-progress",
dependencies: [],
steps: [{ name: "Implement", status: "in-progress" }],
currentStep: 0,
log: [],
branch: "fusion/fn-7996",
baseBranch: "main",
worktree: "/tmp/fusion-fn-7996",
status: null,
error: null,
paused: false,
userPaused: false,
toolFailureDetectorLogCursor: 0,
autoMerge: true,
mergeRetries: 0,
createdAt: now,
updatedAt: now,
...overrides,
} as TaskDetail;
}
function graphFailure(nodeId = "steps#0:step-execute") {
return {
disposition: "failed" as const,
outcome: "failure" as const,
visitedNodeIds: [nodeId],
context: { [`node:${nodeId}:value`]: "failure" },
};
}
function makeHarness(options: { retries: number; entries: Array<{ type: string }>; settings?: Record<string, unknown>; task?: Partial<TaskDetail> }) {
const store = createMockStore();
const task = makeTask(options.task);
store.getTask.mockResolvedValue(task);
store.getSettings.mockResolvedValue({
maxConcurrent: 2,
maxWorktrees: 4,
pollIntervalMs: 15_000,
autoMerge: true,
executorToolFailureRetryCount: options.retries,
executorToolFailureRetryBackoffMs: 0,
...options.settings,
});
store.getAgentLogCount = vi.fn().mockResolvedValue(options.entries.length);
store.getAgentLogs = vi.fn().mockResolvedValue(options.entries);
store.claimNextToolFailureRetry = vi.fn().mockResolvedValue({ outcome: "claimed", attempt: 1 });
store.updateTask.mockImplementation(async (_id: string, patch: Partial<TaskDetail>) => Object.assign(task, patch));
store.updateTaskAtomic = vi.fn(async (_id: string, updater: (current: TaskDetail) => Partial<TaskDetail> | null) => {
const updates = updater(task);
if (updates) Object.assign(task, updates);
return task;
});
store.markToolFailureRetryExhaustedAudit = vi.fn().mockResolvedValue(true);
store.recordRunAuditEvent = vi.fn().mockResolvedValue(undefined);
const executor = new TaskExecutor(store, "/tmp/test");
(executor as any).graphToolFailureRunCursors.set(task.id, 0);
return { executor, store, task };
}
describe("executor consecutive tool-failure retry (FN-7996)", () => {
beforeEach(() => {
resetExecutorMocks();
vi.useFakeTimers();
});
afterEach(() => vi.useRealTimers());
it("retries one post-cursor tool_error by default instead of terminal parking", async () => {
const { executor, store, task } = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }],
});
const execute = vi.spyOn(executor as any, "execute").mockResolvedValue(undefined);
await (executor as any).handleGraphFailure(task, graphFailure());
await vi.advanceTimersByTimeAsync(0);
expect(store.claimNextToolFailureRetry).toHaveBeenCalledWith(task.id, 0, 2);
expect(task).toMatchObject({ status: null, error: null });
expect(execute).toHaveBeenCalledWith(task);
expect(store.updateTask).not.toHaveBeenCalledWith(task.id, expect.objectContaining({ status: "failed" }), expect.anything());
expect(store.recordRunAuditEvent).toHaveBeenCalledWith(expect.objectContaining({
mutationType: "task:execution-tool-failure-retry",
metadata: {
taskId: task.id,
nodeId: "steps#0:step-execute",
attempt: 1,
maxAttempts: 2,
consecutiveToolFailures: 1,
mode: "same-model",
},
}));
});
it.each(["execute", "step-execute", "steps#0:step-execute"])("recognizes trailing errors at execute-family node %s", async (nodeId) => {
const { executor, store, task } = makeHarness({ retries: 2, entries: [{ type: "tool_error" }] });
await (executor as any).handleGraphFailure(task, graphFailure(nodeId));
expect(store.claimNextToolFailureRetry).toHaveBeenCalledWith(task.id, 0, 2);
});
it("honors an explicit threshold above the first-error default", async () => {
const belowThreshold = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }],
settings: { executorToolFailureThreshold: 2 },
});
await (belowThreshold.executor as any).handleGraphFailure(belowThreshold.task, graphFailure());
expect(belowThreshold.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
expect(belowThreshold.task).toMatchObject({ status: "failed" });
const qualifying = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }, { type: "tool_error" }],
settings: { executorToolFailureThreshold: 2 },
});
await (qualifying.executor as any).handleGraphFailure(qualifying.task, graphFailure());
expect(qualifying.store.claimNextToolFailureRetry).toHaveBeenCalledWith(qualifying.task.id, 0, 2);
});
it("ignores invocation/text markers but a later tool result resets the trailing error streak", async () => {
const qualifying = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }, { type: "tool" }, { type: "text" }, { type: "thinking" }],
});
await (qualifying.executor as any).handleGraphFailure(qualifying.task, graphFailure());
expect(qualifying.store.claimNextToolFailureRetry).toHaveBeenCalled();
const reset = makeHarness({ retries: 2, entries: [{ type: "tool_error" }, { type: "tool_result" }] });
await (reset.executor as any).handleGraphFailure(reset.task, graphFailure());
expect(reset.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
expect(reset.task).toMatchObject({ status: "failed" });
});
it("fails closed to the ordinary terminal path when logs cannot prove a post-cursor failure", async () => {
const noError = makeHarness({ retries: 2, entries: [] });
await (noError.executor as any).handleGraphFailure(noError.task, graphFailure());
expect(noError.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
expect(noError.task).toMatchObject({ status: "failed" });
const missingLogApis = makeHarness({ retries: 2, entries: [{ type: "tool_error" }] });
delete (missingLogApis.store as any).getAgentLogCount;
delete (missingLogApis.store as any).getAgentLogs;
await (missingLogApis.executor as any).handleGraphFailure(missingLogApis.task, graphFailure());
expect(missingLogApis.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
expect(missingLogApis.task).toMatchObject({ status: "failed" });
});
it("normalizes the configured backoff and waits before retrying", async () => {
const { executor, task } = makeHarness({
retries: 2.9,
entries: [{ type: "tool_error" }, { type: "tool_error" }, { type: "tool_error" }],
settings: { executorToolFailureRetryBackoffMs: 2500.9 },
});
const execute = vi.spyOn(executor as any, "execute").mockResolvedValue(undefined);
expect(resolveMaxConsecutiveToolFailureRetries({ executorToolFailureRetryCount: 2.9 })).toBe(2);
expect(resolveMaxConsecutiveToolFailureRetries({ executorToolFailureRetryCount: -1 })).toBe(2);
expect(resolveConsecutiveToolFailureRetryBackoffMs({ executorToolFailureRetryBackoffMs: 2500.9 })).toBe(2500);
expect(resolveConsecutiveToolFailureRetryBackoffMs({ executorToolFailureRetryBackoffMs: -1 })).toBe(2000);
await (executor as any).handleGraphFailure(task, graphFailure());
await vi.advanceTimersByTimeAsync(2499);
expect(execute).not.toHaveBeenCalled();
await vi.advanceTimersByTimeAsync(1);
expect(execute).toHaveBeenCalledWith(task);
});
it("parks unchanged after a spent retry budget and emits one exhaustion audit", async () => {
const { executor, store, task } = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }, { type: "tool_error" }, { type: "tool_error" }],
});
store.claimNextToolFailureRetry.mockResolvedValue({ outcome: "exhausted" });
await (executor as any).handleGraphFailure(task, graphFailure());
expect(store.recordRunAuditEvent).toHaveBeenCalledWith(expect.objectContaining({
mutationType: "task:execution-tool-failure-retry-exhausted",
metadata: expect.objectContaining({ taskId: task.id, attempts: 2, limit: 2, outcome: "terminal-park" }),
}));
expect(store.updateTaskAtomic).toHaveBeenCalledWith(task.id, expect.any(Function), undefined);
expect(task).toMatchObject({
status: "failed",
error: "Workflow graph terminated with failure at node 'steps#0:step-execute'",
});
});
it("escalates once to a configured model after same-model retries exhaust", async () => {
const { executor, store, task } = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }, { type: "tool_error" }, { type: "tool_error" }],
settings: { executorModelEscalationEnabled: true, executorEscalationProvider: "anthropic", executorEscalationModelId: "claude-sonnet" },
});
store.claimNextToolFailureRetry.mockResolvedValue({ outcome: "exhausted" });
const execute = vi.spyOn(executor as any, "execute").mockResolvedValue(undefined);
await (executor as any).handleGraphFailure(task, graphFailure());
await vi.advanceTimersByTimeAsync(0);
expect(task).toMatchObject({ modelProvider: "anthropic", modelId: "claude-sonnet", executorEscalationAttempted: true, status: null, error: null });
expect(execute).toHaveBeenCalledWith(task);
expect(store.recordRunAuditEvent).toHaveBeenCalledWith(expect.objectContaining({ mutationType: "task:execution-escalation-retry", metadata: expect.objectContaining({ taskId: task.id, hasModelTarget: true, hasNodeTarget: false }) }));
expect(store.updateTaskAtomic).toHaveBeenCalledWith(task.id, expect.any(Function), undefined);
});
it("requeues a node escalation for scheduler effective-node resolution", async () => {
const { executor, store, task } = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }, { type: "tool_error" }, { type: "tool_error" }],
settings: { executorModelEscalationEnabled: true, executorEscalationNodeId: "cursor-node" },
});
store.claimNextToolFailureRetry.mockResolvedValue({ outcome: "exhausted" });
const execute = vi.spyOn(executor as any, "execute").mockResolvedValue(undefined);
await (executor as any).handleGraphFailure(task, graphFailure());
expect(task).toMatchObject({ nodeId: "cursor-node", column: "todo", executorEscalationAttempted: true, status: null, error: null });
expect(execute).not.toHaveBeenCalled();
expect(store.recordRunAuditEvent).toHaveBeenCalledWith(expect.objectContaining({ mutationType: "task:execution-escalation-retry", metadata: expect.objectContaining({ hasNodeTarget: true }) }));
});
it("parks the single escalated attempt and records escalation exhaustion", async () => {
const { executor, store, task } = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }, { type: "tool_error" }, { type: "tool_error" }],
task: { executorEscalationAttempted: true },
settings: { executorModelEscalationEnabled: true, executorEscalationProvider: "anthropic", executorEscalationModelId: "claude-sonnet" },
});
store.claimNextToolFailureRetry.mockResolvedValue({ outcome: "exhausted" });
await (executor as any).handleGraphFailure(task, graphFailure());
expect(task.status).toBe("failed");
expect(store.recordRunAuditEvent).toHaveBeenCalledWith(expect.objectContaining({ mutationType: "task:execution-escalation-exhausted" }));
});
it("does not let a concurrent exhausted handler park the escalation it lost", async () => {
const { executor, store, task } = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }, { type: "tool_error" }, { type: "tool_error" }],
settings: { executorModelEscalationEnabled: true, executorEscalationProvider: "anthropic", executorEscalationModelId: "claude-sonnet" },
});
store.claimNextToolFailureRetry.mockResolvedValue({ outcome: "exhausted" });
task.executorEscalationAttempted = true;
// The atomic escalation claim invalidates its exhausted cursor before scheduling.
task.toolFailureDetectorLogCursor = null;
await (executor as any).handleGraphFailure(task, graphFailure());
expect(task).toMatchObject({ status: null, error: null, executorEscalationAttempted: true });
expect(store.updateTask).not.toHaveBeenCalledWith(task.id, expect.objectContaining({ status: "failed" }), expect.anything());
expect(store.recordRunAuditEvent).not.toHaveBeenCalledWith(expect.objectContaining({
mutationType: "task:execution-escalation-exhausted",
}));
});
it("audits a terminal escalated failure even after escalation is disabled", async () => {
const { executor, store, task } = makeHarness({
retries: 0,
entries: [],
task: { executorEscalationAttempted: true, modelProvider: "anthropic", modelId: "claude-sonnet" },
});
await (executor as any).handleGraphFailure(task, graphFailure());
expect(task.status).toBe("failed");
expect(store.recordRunAuditEvent).toHaveBeenCalledWith(expect.objectContaining({
mutationType: "task:execution-escalation-exhausted",
metadata: expect.objectContaining({ hadModelTarget: true, hadNodeTarget: false }),
}));
});
it("does not let an exhausted stale handler park a newer cursor-owned run", async () => {
const { executor, store, task } = makeHarness({
retries: 2,
entries: [{ type: "tool_error" }, { type: "tool_error" }, { type: "tool_error" }],
});
store.claimNextToolFailureRetry.mockResolvedValue({ outcome: "exhausted" });
const newRun = makeTask({ toolFailureDetectorLogCursor: 99 });
store.updateTaskAtomic.mockImplementation(async (_id: string, updater: (current: TaskDetail) => Partial<TaskDetail> | null) => {
// A new execution captured its own log cursor after the old run's claim exhausted.
expect(updater(newRun)).toBeNull();
return newRun;
});
await (executor as any).handleGraphFailure(task, graphFailure());
expect(newRun).toMatchObject({ status: null, error: null, toolFailureDetectorLogCursor: 99 });
expect(store.markToolFailureRetryExhaustedAudit).not.toHaveBeenCalled();
expect(store.recordRunAuditEvent).not.toHaveBeenCalledWith(expect.objectContaining({
mutationType: "task:execution-tool-failure-retry-exhausted",
}));
expect(store.updateTask).not.toHaveBeenCalledWith(task.id, expect.objectContaining({ status: "failed" }), expect.anything());
});
it("preserves the immediate legacy park when disabled or errors are not consecutive", async () => {
const disabled = makeHarness({ retries: 0, entries: [{ type: "tool_error" }, { type: "tool_error" }, { type: "tool_error" }] });
await (disabled.executor as any).handleGraphFailure(disabled.task, graphFailure());
expect(disabled.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
expect(disabled.store.updateTask).toHaveBeenCalledWith(disabled.task.id, expect.objectContaining({ status: "failed" }), undefined);
const interleaved = makeHarness({ retries: 2, entries: [{ type: "tool_error" }, { type: "tool_result" }] });
await (interleaved.executor as any).handleGraphFailure(interleaved.task, graphFailure());
expect(interleaved.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
expect(interleaved.store.updateTask).toHaveBeenCalledWith(interleaved.task.id, expect.objectContaining({ status: "failed" }), undefined);
});
});