FN-8815: retry execution after first tool failure
Retry executor work after the first terminal tool-call failure by default. - Set the default tool-failure threshold to one while preserving explicit project overrides. - Expose and document the first-error default in Settings, translations, and the settings reference. - Cover threshold resolution, save lifecycle, and executor retry behavior. Files changed: .changeset/fn-8815-retry-first-tool-failure.md | 7 +++ docs/settings-reference.md | 4 +- .../core/src/__tests__/settings-defaults.test.ts | 6 +- packages/core/src/config/settings-schema.ts | 10 ++- packages/core/src/tasks/in-review-stall.ts | 15 ++++- packages/core/src/types/settings/settings-scope.ts | 10 ++- .../dashboard/app/components/SettingsModal.tsx | 7 ++- .../SettingsModal.scheduling-merge.test.tsx | 61 +++++++++++++++++- .../settings/sections/SchedulingSection.search.ts | 2 +- .../settings/sections/SchedulingSection.tsx | 4 +- .../settings-default-descriptions.test.tsx | 12 ++++ .../__tests__/executor-tool-failure-retry.test.ts | 73 +++++++++++++++++++--- packages/i18n/locales/en/app.json | 2 +- packages/i18n/src/resources.d.ts | 17 +++-- 14 files changed, 200 insertions(+), 30 deletions(-) Fusion-Task-Id: FN-8815 Fusion-Task-Lineage: 909181e7-2da5-4a27-92ee-4182732d9695 Co-authored-by: Fusion (runfusion.ai) <noreply@runfusion.ai>
This commit is contained in:
7
.changeset/fn-8815-retry-first-tool-failure.md
Normal file
7
.changeset/fn-8815-retry-first-tool-failure.md
Normal file
@@ -0,0 +1,7 @@
|
||||
---
|
||||
"@runfusion/fusion": patch
|
||||
---
|
||||
|
||||
summary: Retry execution after the first terminal tool-call failure by default.
|
||||
category: fix
|
||||
dev: The project threshold remains configurable and existing explicit overrides are preserved.
|
||||
@@ -1849,9 +1849,9 @@ Standardize executor/validator pairs; auto-selectable by task size (Small → Bu
|
||||
| --- | --- | --- |
|
||||
| `executorToolFailureRetryCount` | integer, `2` | Same-model retries before terminal executor parking; `0` disables this policy entirely. |
|
||||
| `executorToolFailureRetryBackoffMs` | integer, `2000` | Unref'd delay before the rerun. |
|
||||
| `executorToolFailureThreshold` | integer, `3` | Consecutive terminal tool failures required to qualify. |
|
||||
| `executorToolFailureThreshold` | integer, `1` | Consecutive terminal tool failures required to qualify. |
|
||||
|
||||
Values are project-scoped and finite values are floored; count/backoff must be at least `0`, and threshold at least `1`, otherwise their defaults apply. The executor evaluates this bounded policy before its terminal graph-failure park: it counts `tool_error` completion entries, resets only on `tool_result`, and ignores `tool` invocation markers. The detector is scoped to the current executor-run agent-log cursor. Its project-scoped atomic claim prevents concurrent retries and classifies cursor mismatch before an exhausted cap so stale handlers do not park newer work. The exhausted audit is compare-and-set deduplicated while the terminal park remains idempotent.
|
||||
One terminal `tool_error` after the current execution-run cursor therefore qualifies for bounded same-model recovery by default. Operators may set a higher explicit threshold; existing persisted overrides are preserved. Values are project-scoped and finite values are floored; count/backoff must be at least `0`, and threshold at least `1`, otherwise their defaults apply. The executor evaluates this bounded policy before its terminal graph-failure park: it counts `tool_error` completion entries, resets only on `tool_result`, and ignores `tool` invocation markers. The detector is scoped to the current executor-run agent-log cursor. Its project-scoped atomic claim prevents concurrent retries and classifies cursor mismatch before an exhausted cap so stale handlers do not park newer work. Pause, deletion, column, and live-session cancellation fences are rechecked before delayed re-entry. The exhausted audit is compare-and-set deduplicated while the terminal park remains idempotent.
|
||||
|
||||
### Executor escalation after tool-failure retry exhaustion
|
||||
|
||||
|
||||
@@ -347,8 +347,12 @@ describe("settings defaults invariants", () => {
|
||||
expect(resolveMaxConsecutiveToolFailureRetries({ executorToolFailureRetryCount: -1 })).toBe(2);
|
||||
expect(resolveConsecutiveToolFailureRetryBackoffMs({ executorToolFailureRetryBackoffMs: 2500.9 })).toBe(2500);
|
||||
expect(resolveConsecutiveToolFailureRetryBackoffMs({ executorToolFailureRetryBackoffMs: Infinity })).toBe(2000);
|
||||
expect(resolveConsecutiveToolFailureThreshold(undefined)).toBe(1);
|
||||
expect(resolveConsecutiveToolFailureThreshold({})).toBe(1);
|
||||
expect(resolveConsecutiveToolFailureThreshold({ executorToolFailureThreshold: Number.NaN })).toBe(1);
|
||||
expect(resolveConsecutiveToolFailureThreshold({ executorToolFailureThreshold: 0.5 })).toBe(1);
|
||||
expect(resolveConsecutiveToolFailureThreshold({ executorToolFailureThreshold: 3.9 })).toBe(3);
|
||||
expect(resolveConsecutiveToolFailureThreshold({ executorToolFailureThreshold: 0.5 })).toBe(3);
|
||||
expect(resolveConsecutiveToolFailureThreshold({ executorToolFailureThreshold: 4 })).toBe(4);
|
||||
});
|
||||
|
||||
});
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { DEFAULT_MAX_AUTO_MERGE_RETRIES } from "../tasks/in-review-stall.js";
|
||||
import { CONSECUTIVE_TOOL_FAILURE_RETRY_THRESHOLD, DEFAULT_MAX_AUTO_MERGE_RETRIES } from "../tasks/in-review-stall.js";
|
||||
import type { CliAgentSettings, GlobalSettings, McpSecretRef, McpServerDefinition, ProjectSettings, Settings } from "../types.js";
|
||||
|
||||
export interface MergeRequestContractShadowSettingsSource {
|
||||
@@ -585,9 +585,15 @@ export const DEFAULT_PROJECT_SETTINGS = {
|
||||
* Project settings own the auto-merge conflict retry cap because existing engine/dashboard consumers already resolve project settings; the default imports core's stall-detection fallback to keep every surface on the historical value of 3.
|
||||
*/
|
||||
maxAutoMergeRetries: DEFAULT_MAX_AUTO_MERGE_RETRIES,
|
||||
/*
|
||||
FNXC:ExecutorToolFailureRetry 2026-08-06-14:56:
|
||||
Fresh projects retry the existing bounded same-model continuation after one
|
||||
terminal tool error. Persisted project values are not migrated, so explicit
|
||||
operator thresholds remain authoritative.
|
||||
*/
|
||||
executorToolFailureRetryCount: 2,
|
||||
executorToolFailureRetryBackoffMs: 2000,
|
||||
executorToolFailureThreshold: 3,
|
||||
executorToolFailureThreshold: CONSECUTIVE_TOOL_FAILURE_RETRY_THRESHOLD,
|
||||
executorModelEscalationEnabled: false,
|
||||
executorEscalationProvider: undefined,
|
||||
executorEscalationModelId: undefined,
|
||||
|
||||
@@ -65,7 +65,13 @@ export const DEFAULT_STALE_MERGING_MIN_AGE_MS = 5 * 60_000;
|
||||
export const DEFAULT_MAX_AUTO_MERGE_RETRIES = 3;
|
||||
export const DEFAULT_MAX_CONSECUTIVE_TOOL_FAILURE_RETRIES = 2;
|
||||
export const DEFAULT_CONSECUTIVE_TOOL_FAILURE_RETRY_BACKOFF_MS = 2_000;
|
||||
export const CONSECUTIVE_TOOL_FAILURE_RETRY_THRESHOLD = 3;
|
||||
/*
|
||||
FNXC:ExecutorToolFailureRetry 2026-08-06-14:56:
|
||||
One terminal tool_error after an execute-family run cursor must qualify for the
|
||||
existing bounded same-model recovery. Explicit thresholds, retry budgets, and
|
||||
cursor ownership remain the operator-controlled safety fences.
|
||||
*/
|
||||
export const CONSECUTIVE_TOOL_FAILURE_RETRY_THRESHOLD = 1;
|
||||
|
||||
/**
|
||||
* FNXC:AutoMergeRetries 2026-06-17-04:20:
|
||||
@@ -79,7 +85,12 @@ export function resolveMaxAutoMergeRetries(settings?: { maxAutoMergeRetries?: un
|
||||
return DEFAULT_MAX_AUTO_MERGE_RETRIES;
|
||||
}
|
||||
|
||||
/** FNXC:ExecutorToolFailureRetry 2026-07-16-12:00: normalize the project policy identically in engine and settings UI; finite in-range fractions floor, invalid values retain safe defaults. */
|
||||
/**
|
||||
* FNXC:ExecutorToolFailureRetry 2026-08-06-14:56:
|
||||
* Normalize project policy identically in engine and settings UI. Invalid or
|
||||
* undefined thresholds fall back to the first-error default; finite explicit
|
||||
* thresholds at or above one retain operator intent.
|
||||
*/
|
||||
function resolveNonNegativeInteger(value: unknown, fallback: number): number {
|
||||
const numeric = Number(value);
|
||||
return Number.isFinite(numeric) && Math.floor(numeric) >= 0 ? Math.floor(numeric) : fallback;
|
||||
|
||||
@@ -1136,11 +1136,17 @@ export interface ProjectSettings {
|
||||
* (triage specification, task execution, and merge operations). */
|
||||
maxConcurrent: number;
|
||||
/**
|
||||
* FNXC:ExecutorToolFailureRetry 2026-07-16-12:00:
|
||||
* Bounded same-model retry before the executor terminal park. Tool markers are ignored, terminal tool_error counts, tool_result resets; per-run cursor claims prevent concurrent over-retry and count 0 preserves prior behavior. Values are floored and the backoff timer is unref'd. FN-7998 consumes this stable policy shape for escalation.
|
||||
* FNXC:ExecutorToolFailureRetry 2026-08-06-14:56:
|
||||
* Bounded same-model retry before the executor terminal park. Tool markers
|
||||
* are ignored, terminal tool_error counts, and tool_result resets; one
|
||||
* terminal error qualifies by default. Per-run cursor claims prevent
|
||||
* concurrent over-retry, count 0 preserves disabled recovery, explicit
|
||||
* thresholds remain honored, and the unref'd backoff supports FN-7998
|
||||
* escalation after exhaustion.
|
||||
*/
|
||||
executorToolFailureRetryCount?: number;
|
||||
executorToolFailureRetryBackoffMs?: number;
|
||||
/** Consecutive terminal tool errors required to retry. Default: 1. */
|
||||
executorToolFailureThreshold?: number;
|
||||
/**
|
||||
* FNXC:ExecutorEscalation 2026-07-16-21:00:
|
||||
|
||||
@@ -986,7 +986,8 @@ export function SettingsModal({
|
||||
maxAutoMergeRetries: 3,
|
||||
executorToolFailureRetryCount: 2,
|
||||
executorToolFailureRetryBackoffMs: 2000,
|
||||
executorToolFailureThreshold: 3,
|
||||
// FNXC:ExecutorToolFailureRetry 2026-08-06-14:56: mirror the core first-terminal-error default so an unset project setting never displays or persists the retired threshold of three.
|
||||
executorToolFailureThreshold: 1,
|
||||
executorModelEscalationEnabled: false,
|
||||
executorEscalationProvider: "",
|
||||
executorEscalationModelId: "",
|
||||
@@ -1652,7 +1653,7 @@ export function SettingsModal({
|
||||
maxAutoMergeRetries: resolveMaxAutoMergeRetriesForSettingsForm(s),
|
||||
executorToolFailureRetryCount: resolveNonNegativeExecutorToolFailureSetting(s.executorToolFailureRetryCount, 2),
|
||||
executorToolFailureRetryBackoffMs: resolveNonNegativeExecutorToolFailureSetting(s.executorToolFailureRetryBackoffMs, 2000),
|
||||
executorToolFailureThreshold: Math.max(1, Math.floor(Number(s.executorToolFailureThreshold ?? 3) || 3)),
|
||||
executorToolFailureThreshold: Math.max(1, Math.floor(Number(s.executorToolFailureThreshold ?? 1) || 1)),
|
||||
executorModelEscalationEnabled: s.executorModelEscalationEnabled === true,
|
||||
executorEscalationProvider: s.executorEscalationProvider ?? "",
|
||||
executorEscalationModelId: s.executorEscalationModelId ?? "",
|
||||
@@ -3442,7 +3443,7 @@ export function SettingsModal({
|
||||
maxAutoMergeRetries: resolveMaxAutoMergeRetriesForSettingsForm(formSnapshot),
|
||||
executorToolFailureRetryCount: resolveNonNegativeExecutorToolFailureSetting(formSnapshot.executorToolFailureRetryCount, 2),
|
||||
executorToolFailureRetryBackoffMs: resolveNonNegativeExecutorToolFailureSetting(formSnapshot.executorToolFailureRetryBackoffMs, 2000),
|
||||
executorToolFailureThreshold: Math.max(1, Math.floor(Number(formSnapshot.executorToolFailureThreshold ?? 3) || 3)),
|
||||
executorToolFailureThreshold: Math.max(1, Math.floor(Number(formSnapshot.executorToolFailureThreshold ?? 1) || 1)),
|
||||
executorModelEscalationEnabled: formSnapshot.executorModelEscalationEnabled === true,
|
||||
executorEscalationProvider: formSnapshot.executorEscalationProvider?.trim() || undefined,
|
||||
executorEscalationModelId: formSnapshot.executorEscalationModelId?.trim() || undefined,
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, it, expect, vi, beforeEach } from "vitest";
|
||||
import { cleanup, render, screen, fireEvent, waitFor, within } from "@testing-library/react";
|
||||
import { act, cleanup, render, screen, fireEvent, waitFor, within } from "@testing-library/react";
|
||||
import { EditorView } from "@codemirror/view";
|
||||
import path from "path";
|
||||
import { SettingsModal } from "../SettingsModal";
|
||||
@@ -208,6 +208,65 @@ describe("SettingsModal", () => {
|
||||
});
|
||||
|
||||
describe("Scheduling overlap ignore paths", () => {
|
||||
/*
|
||||
FNXC:ExecutorToolFailureRetry 2026-08-06-15:10:
|
||||
Exercise the SettingsModal API lifecycle, not only the SchedulingSection fallback.
|
||||
An omitted project threshold must render as one, while an explicit operator value
|
||||
must remain unchanged when another scheduling edit triggers the project save.
|
||||
*/
|
||||
it("loads the first-error threshold default and saves an explicit threshold", async () => {
|
||||
const { executorToolFailureThreshold: _omitted, ...settingsWithoutThreshold } = defaultSettings;
|
||||
mockFetchSettings.mockResolvedValue(settingsWithoutThreshold);
|
||||
mockFetchSettingsByScope.mockResolvedValue({ global: settingsWithoutThreshold, project: {} });
|
||||
|
||||
renderModal();
|
||||
await waitForSettingsModalReady();
|
||||
await act(async () => {
|
||||
await Promise.resolve();
|
||||
});
|
||||
await settingsModalUser.click(screen.getByRole("button", { name: "Scheduling" }));
|
||||
|
||||
const threshold = screen.getByLabelText("Consecutive tool failures") as HTMLInputElement;
|
||||
expect(threshold.value).toBe("1");
|
||||
|
||||
fireEvent.change(threshold, { target: { value: "4" } });
|
||||
expect(threshold.value).toBe("4");
|
||||
await settingsModalUser.click(document.querySelector(".modal-close") as HTMLButtonElement);
|
||||
|
||||
await waitFor(() => expect(mockUpdateSettings).toHaveBeenCalled());
|
||||
expect(mockUpdateSettings.mock.calls.map((call) => call[0])).toContainEqual(
|
||||
expect.objectContaining({ executorToolFailureThreshold: 4 }),
|
||||
);
|
||||
});
|
||||
|
||||
it("does not overwrite an explicit threshold when saving another scheduling setting", async () => {
|
||||
mockFetchSettings.mockResolvedValue({
|
||||
...defaultSettings,
|
||||
executorToolFailureThreshold: 4,
|
||||
engineerBacklogAutoClaim: false,
|
||||
});
|
||||
mockFetchSettingsByScope.mockResolvedValue({
|
||||
global: defaultSettings,
|
||||
project: { executorToolFailureThreshold: 4, engineerBacklogAutoClaim: false },
|
||||
});
|
||||
|
||||
renderModal();
|
||||
await waitForSettingsModalReady();
|
||||
await act(async () => {
|
||||
await Promise.resolve();
|
||||
});
|
||||
await settingsModalUser.click(screen.getByRole("button", { name: "Scheduling" }));
|
||||
|
||||
expect((screen.getByLabelText("Consecutive tool failures") as HTMLInputElement).value).toBe("4");
|
||||
await settingsModalUser.click(screen.getByLabelText("Let engineer agents auto-claim backlog tasks"));
|
||||
await settingsModalUser.click(document.querySelector(".modal-close") as HTMLButtonElement);
|
||||
|
||||
await waitFor(() => expect(mockUpdateSettings).toHaveBeenCalled());
|
||||
const payload = mockUpdateSettings.mock.calls[0][0] as Record<string, unknown>;
|
||||
expect(payload.engineerBacklogAutoClaim).toBe(true);
|
||||
expect(payload).not.toHaveProperty("executorToolFailureThreshold");
|
||||
});
|
||||
|
||||
it("defaults hidden overlap path filtering checked when settings omit the key", async () => {
|
||||
const { ignoreHiddenOverlapPaths: _omitted, ...settingsWithoutHiddenDefault } = defaultSettings;
|
||||
mockFetchSettings.mockResolvedValue(settingsWithoutHiddenDefault);
|
||||
|
||||
@@ -54,7 +54,7 @@ export const schedulingSearchEntries: SettingsSearchEntry[] = [
|
||||
labelKey: "settings.scheduling.executorToolFailureThreshold",
|
||||
labelFallback: "Consecutive tool failures",
|
||||
helpKey: "settings.scheduling.executorToolFailureThresholdHelp",
|
||||
helpFallback: "Terminal tool errors required before retrying. Default: 3.",
|
||||
helpFallback: "Terminal tool errors required before retrying. Default: 1.",
|
||||
keywords: ["executor", "tool error", "threshold", "auto retry"],
|
||||
},
|
||||
{
|
||||
|
||||
@@ -34,10 +34,10 @@ export function SchedulingSection({ form, setForm, concurrencyLoading = false, o
|
||||
const { t } = useTranslation("app");
|
||||
return (<>
|
||||
<h4 className="settings-section-heading">{t("settings.scheduling.scheduling", "Scheduling")}</h4>
|
||||
{/* FNXC:ExecutorToolFailureRetry 2026-07-16-12:00: project controls tune the bounded same-model retry before terminal executor parking; values floor to core's resolver contract. */}
|
||||
{/* FNXC:ExecutorToolFailureRetry 2026-08-06-14:56: project controls tune bounded same-model retry before terminal executor parking; one terminal tool error qualifies by default while values still floor to core's resolver contract. */}
|
||||
<SettingsNumberRow descriptor={{ key: "executorToolFailureRetryCount", label: t("settings.scheduling.executorToolFailureRetryCount", "Executor tool-failure retries"), help: t("settings.scheduling.executorToolFailureRetryCountHelp", "Same-model retries after consecutive tool-call failures. Set 0 to disable. Default: 2."), scope: "project", min: 0, step: 1 }} value={form.executorToolFailureRetryCount ?? 2} onChange={(v) => setForm((f) => ({ ...f, executorToolFailureRetryCount: Math.max(0, Math.floor(v ?? 2)) } as SettingsFormState))} />
|
||||
<SettingsNumberRow descriptor={{ key: "executorToolFailureRetryBackoffMs", label: t("settings.scheduling.executorToolFailureRetryBackoffMs", "Tool-failure retry backoff (ms)"), help: t("settings.scheduling.executorToolFailureRetryBackoffMsHelp", "Unref'd wait before retrying. Default: 2000."), scope: "project", min: 0, step: 1 }} value={form.executorToolFailureRetryBackoffMs ?? 2000} onChange={(v) => setForm((f) => ({ ...f, executorToolFailureRetryBackoffMs: Math.max(0, Math.floor(v ?? 2000)) } as SettingsFormState))} />
|
||||
<SettingsNumberRow descriptor={{ key: "executorToolFailureThreshold", label: t("settings.scheduling.executorToolFailureThreshold", "Consecutive tool failures"), help: t("settings.scheduling.executorToolFailureThresholdHelp", "Terminal tool errors required before retrying. Default: 3."), scope: "project", min: 1, step: 1 }} value={form.executorToolFailureThreshold ?? 3} onChange={(v) => setForm((f) => ({ ...f, executorToolFailureThreshold: Math.max(1, Math.floor(v ?? 3)) } as SettingsFormState))} />
|
||||
<SettingsNumberRow descriptor={{ key: "executorToolFailureThreshold", label: t("settings.scheduling.executorToolFailureThreshold", "Consecutive tool failures"), help: t("settings.scheduling.executorToolFailureThresholdHelp", "Terminal tool errors required before retrying. Default: 1."), scope: "project", min: 1, step: 1 }} value={form.executorToolFailureThreshold ?? 1} onChange={(v) => setForm((f) => ({ ...f, executorToolFailureThreshold: Math.max(1, Math.floor(v ?? 1)) } as SettingsFormState))} />
|
||||
{/* FNXC:ExecutorEscalation 2026-07-16-21:00: Keep alternate model/node escalation opt-in and adjacent to its FN-7996 retry policy; a complete model pair or node id is required before the executor consumes its one extra attempt. */}
|
||||
<SettingsToggleRow descriptor={{ key: "executorModelEscalationEnabled", label: t("settings.scheduling.executorModelEscalationEnabled", "Escalate after tool-failure retries"), help: t("settings.scheduling.executorModelEscalationEnabledHelp", "After same-model retries are exhausted, try one configured alternate model or node. Disabled by default."), scope: "project" }} value={form.executorModelEscalationEnabled === true} onChange={(value) => setForm((f) => ({ ...f, executorModelEscalationEnabled: value === true } as SettingsFormState))} />
|
||||
<SettingsTextRow descriptor={{ key: "executorEscalationNodeId", label: t("settings.scheduling.executorEscalationNodeId", "Escalation node ID"), help: t("settings.scheduling.executorEscalationNodeIdHelp", "Optional configured node; a node target re-enters scheduler routing."), scope: "project" }} value={form.executorEscalationNodeId ?? ""} onChange={(value) => setForm((f) => ({ ...f, executorEscalationNodeId: value ?? "" } as SettingsFormState))} />
|
||||
|
||||
@@ -611,6 +611,18 @@ const NOT_SURFACED_ALLOWLIST: Record<string, string> = {
|
||||
};
|
||||
|
||||
describe("FN-7505 settings default-value description guard", () => {
|
||||
it("uses the active English catalog's first-error tool retry default", () => {
|
||||
/*
|
||||
* FNXC:ExecutorToolFailureRetry 2026-08-06-14:56:
|
||||
* Import the active runtime English catalog rather than inspecting a
|
||||
* component fallback. A stale translation otherwise overrides the correct
|
||||
* form value and tells desktop and mobile operators the retired default.
|
||||
*/
|
||||
expect(resolveDescription(realEnApp.settings as SettingsDict, "scheduling.executorToolFailureThresholdHelp"))
|
||||
.toBe("Terminal tool errors required before retrying. Default: 1.");
|
||||
expect(DEFAULT_PROJECT_SETTINGS.executorToolFailureThreshold).toBe(1);
|
||||
});
|
||||
|
||||
it("every surfaced setting's resolved English description states its default", () => {
|
||||
const missing: string[] = [];
|
||||
const noIndicator: string[] = [];
|
||||
|
||||
@@ -32,12 +32,12 @@ function makeTask(overrides: Partial<TaskDetail> = {}): TaskDetail {
|
||||
} as TaskDetail;
|
||||
}
|
||||
|
||||
function graphFailure() {
|
||||
function graphFailure(nodeId = "steps#0:step-execute") {
|
||||
return {
|
||||
disposition: "failed" as const,
|
||||
outcome: "failure" as const,
|
||||
visitedNodeIds: ["steps#0:step-execute"],
|
||||
context: { "node:steps#0:step-execute:value": "failure" },
|
||||
visitedNodeIds: [nodeId],
|
||||
context: { [`node:${nodeId}:value`]: "failure" },
|
||||
};
|
||||
}
|
||||
|
||||
@@ -52,7 +52,6 @@ function makeHarness(options: { retries: number; entries: Array<{ type: string }
|
||||
autoMerge: true,
|
||||
executorToolFailureRetryCount: options.retries,
|
||||
executorToolFailureRetryBackoffMs: 0,
|
||||
executorToolFailureThreshold: 3,
|
||||
...options.settings,
|
||||
});
|
||||
store.getAgentLogCount = vi.fn().mockResolvedValue(options.entries.length);
|
||||
@@ -79,19 +78,20 @@ describe("executor consecutive tool-failure retry (FN-7996)", () => {
|
||||
|
||||
afterEach(() => vi.useRealTimers());
|
||||
|
||||
it("retries a qualifying terminal step failure and records metadata-only audit evidence", async () => {
|
||||
it("retries one post-cursor tool_error by default instead of terminal parking", async () => {
|
||||
const { executor, store, task } = makeHarness({
|
||||
retries: 2,
|
||||
entries: [{ type: "tool_error" }, { type: "tool_error" }, { type: "tool_error" }],
|
||||
entries: [{ type: "tool_error" }],
|
||||
});
|
||||
const execute = vi.spyOn(executor as any, "execute").mockResolvedValue(undefined);
|
||||
|
||||
await (executor as any).handleGraphFailure(task, graphFailure());
|
||||
await vi.advanceTimersByTimeAsync(0);
|
||||
|
||||
expect(store.claimNextToolFailureRetry).toHaveBeenCalledWith(task.id, 0, 2);
|
||||
expect(task).toMatchObject({ status: null, error: null });
|
||||
expect(execute).toHaveBeenCalledWith(task);
|
||||
expect(store.updateTask).not.toHaveBeenCalledWith(task.id, expect.objectContaining({ status: "failed" }), expect.anything());
|
||||
expect(store.updateTask).not.toHaveBeenCalledWith(task.id, expect.objectContaining({ graphResumeRetryCount: expect.anything() }), expect.anything());
|
||||
expect(store.recordRunAuditEvent).toHaveBeenCalledWith(expect.objectContaining({
|
||||
mutationType: "task:execution-tool-failure-retry",
|
||||
metadata: {
|
||||
@@ -99,12 +99,67 @@ describe("executor consecutive tool-failure retry (FN-7996)", () => {
|
||||
nodeId: "steps#0:step-execute",
|
||||
attempt: 1,
|
||||
maxAttempts: 2,
|
||||
consecutiveToolFailures: 3,
|
||||
consecutiveToolFailures: 1,
|
||||
mode: "same-model",
|
||||
},
|
||||
}));
|
||||
});
|
||||
|
||||
it.each(["execute", "step-execute", "steps#0:step-execute"])("recognizes trailing errors at execute-family node %s", async (nodeId) => {
|
||||
const { executor, store, task } = makeHarness({ retries: 2, entries: [{ type: "tool_error" }] });
|
||||
|
||||
await (executor as any).handleGraphFailure(task, graphFailure(nodeId));
|
||||
|
||||
expect(store.claimNextToolFailureRetry).toHaveBeenCalledWith(task.id, 0, 2);
|
||||
});
|
||||
|
||||
it("honors an explicit threshold above the first-error default", async () => {
|
||||
const belowThreshold = makeHarness({
|
||||
retries: 2,
|
||||
entries: [{ type: "tool_error" }],
|
||||
settings: { executorToolFailureThreshold: 2 },
|
||||
});
|
||||
await (belowThreshold.executor as any).handleGraphFailure(belowThreshold.task, graphFailure());
|
||||
expect(belowThreshold.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
|
||||
expect(belowThreshold.task).toMatchObject({ status: "failed" });
|
||||
|
||||
const qualifying = makeHarness({
|
||||
retries: 2,
|
||||
entries: [{ type: "tool_error" }, { type: "tool_error" }],
|
||||
settings: { executorToolFailureThreshold: 2 },
|
||||
});
|
||||
await (qualifying.executor as any).handleGraphFailure(qualifying.task, graphFailure());
|
||||
expect(qualifying.store.claimNextToolFailureRetry).toHaveBeenCalledWith(qualifying.task.id, 0, 2);
|
||||
});
|
||||
|
||||
it("ignores invocation/text markers but a later tool result resets the trailing error streak", async () => {
|
||||
const qualifying = makeHarness({
|
||||
retries: 2,
|
||||
entries: [{ type: "tool_error" }, { type: "tool" }, { type: "text" }, { type: "thinking" }],
|
||||
});
|
||||
await (qualifying.executor as any).handleGraphFailure(qualifying.task, graphFailure());
|
||||
expect(qualifying.store.claimNextToolFailureRetry).toHaveBeenCalled();
|
||||
|
||||
const reset = makeHarness({ retries: 2, entries: [{ type: "tool_error" }, { type: "tool_result" }] });
|
||||
await (reset.executor as any).handleGraphFailure(reset.task, graphFailure());
|
||||
expect(reset.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
|
||||
expect(reset.task).toMatchObject({ status: "failed" });
|
||||
});
|
||||
|
||||
it("fails closed to the ordinary terminal path when logs cannot prove a post-cursor failure", async () => {
|
||||
const noError = makeHarness({ retries: 2, entries: [] });
|
||||
await (noError.executor as any).handleGraphFailure(noError.task, graphFailure());
|
||||
expect(noError.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
|
||||
expect(noError.task).toMatchObject({ status: "failed" });
|
||||
|
||||
const missingLogApis = makeHarness({ retries: 2, entries: [{ type: "tool_error" }] });
|
||||
delete (missingLogApis.store as any).getAgentLogCount;
|
||||
delete (missingLogApis.store as any).getAgentLogs;
|
||||
await (missingLogApis.executor as any).handleGraphFailure(missingLogApis.task, graphFailure());
|
||||
expect(missingLogApis.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
|
||||
expect(missingLogApis.task).toMatchObject({ status: "failed" });
|
||||
});
|
||||
|
||||
it("normalizes the configured backoff and waits before retrying", async () => {
|
||||
const { executor, task } = makeHarness({
|
||||
retries: 2.9,
|
||||
@@ -259,7 +314,7 @@ describe("executor consecutive tool-failure retry (FN-7996)", () => {
|
||||
expect(disabled.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
|
||||
expect(disabled.store.updateTask).toHaveBeenCalledWith(disabled.task.id, expect.objectContaining({ status: "failed" }), undefined);
|
||||
|
||||
const interleaved = makeHarness({ retries: 2, entries: [{ type: "tool_error" }, { type: "tool_result" }, { type: "tool_error" }, { type: "tool_error" }] });
|
||||
const interleaved = makeHarness({ retries: 2, entries: [{ type: "tool_error" }, { type: "tool_result" }] });
|
||||
await (interleaved.executor as any).handleGraphFailure(interleaved.task, graphFailure());
|
||||
expect(interleaved.store.claimNextToolFailureRetry).not.toHaveBeenCalled();
|
||||
expect(interleaved.store.updateTask).toHaveBeenCalledWith(interleaved.task.id, expect.objectContaining({ status: "failed" }), undefined);
|
||||
|
||||
@@ -6751,7 +6751,7 @@
|
||||
"executorToolFailureRetryBackoffMs": "Tool-failure retry backoff (ms)",
|
||||
"executorToolFailureRetryBackoffMsHelp": "Unref'd wait before retrying. Default: 2000.",
|
||||
"executorToolFailureThreshold": "Consecutive tool failures",
|
||||
"executorToolFailureThresholdHelp": "Terminal tool errors required before retrying. Default: 3.",
|
||||
"executorToolFailureThresholdHelp": "Terminal tool errors required before retrying. Default: 1.",
|
||||
"executorModelEscalationEnabled": "Escalate after tool-failure retries",
|
||||
"executorModelEscalationEnabledHelp": "After same-model retries are exhausted, try one configured alternate model or node. Default: disabled.",
|
||||
"executorEscalationProvider": "Escalation provider",
|
||||
|
||||
17
packages/i18n/src/resources.d.ts
vendored
17
packages/i18n/src/resources.d.ts
vendored
@@ -6565,11 +6565,9 @@ export default interface Resources {
|
||||
"defaultWorkflowModelLanes": "Default workflow model lanes",
|
||||
"delete": " Delete ",
|
||||
"edit": " Edit ",
|
||||
"executorModel": "Executor model",
|
||||
"executorEscalationModel": "Executor Escalation Model",
|
||||
"executorEscalationModelHelp": "Alternate model used once tool-failure retries are exhausted. No default — unset means no alternate model; configure escalation policy and an optional node target in Scheduling.",
|
||||
"selectExecutorEscalationModel": "Select an escalation model",
|
||||
"noExecutorEscalationModel": "No escalation model",
|
||||
"executorModel": "Executor model",
|
||||
"fallsBackTo": " Falls back to: ",
|
||||
"loadingAgents": "Loading agents…",
|
||||
"loadingAvailableModels": "Loading available models…",
|
||||
@@ -6578,6 +6576,7 @@ export default interface Resources {
|
||||
"modelPresets": "Model Presets",
|
||||
"name": "Name",
|
||||
"noCap": "No cap",
|
||||
"noExecutorEscalationModel": "No escalation model",
|
||||
"noModelsAvailableConfigureAuthenticationBeforeSelectingWorkflow": " No models available. Configure authentication before selecting workflow model lanes. ",
|
||||
"noModelsAvailableConfigureAuthenticationFirst": " No models available. Configure authentication first. ",
|
||||
"noModelsAvailableConfigureAuthenticationFirst2": "No models available. Configure authentication first.",
|
||||
@@ -6595,6 +6594,7 @@ export default interface Resources {
|
||||
"reviewerModel": "Reviewer model",
|
||||
"selectChatDefaultAgent": "Select a chat default agent",
|
||||
"selectChatDefaultModel": "Select a chat default model",
|
||||
"selectExecutorEscalationModel": "Select an escalation model",
|
||||
"taskDefinitionInInputLanguage": "Write task definitions in the operator's input language",
|
||||
"taskDefinitionInInputLanguageHelp": "When enabled, generated task-definition prose uses supported detectable input languages (Spanish, French, Korean, or Chinese as zh-CN). Headings, markers, and code stay English. Unsupported or undetectable input stays English. Default: disabled.",
|
||||
"theseProjectOverridesApplyToTheActiveDefault": " These project overrides apply to the active default workflow. ",
|
||||
@@ -6810,7 +6810,7 @@ export default interface Resources {
|
||||
"executorToolFailureRetryCount": "Executor tool-failure retries",
|
||||
"executorToolFailureRetryCountHelp": "Same-model retries after consecutive tool-call failures. Set 0 to disable. Default: 2.",
|
||||
"executorToolFailureThreshold": "Consecutive tool failures",
|
||||
"executorToolFailureThresholdHelp": "Terminal tool errors required before retrying. Default: 3.",
|
||||
"executorToolFailureThresholdHelp": "Terminal tool errors required before retrying. Default: 1.",
|
||||
"fullAgentLog": "Full agent log",
|
||||
"globalMaxConcurrent": "Global Max Concurrent",
|
||||
"heartbeatScopeDiscipline": "Heartbeat Scope Discipline",
|
||||
@@ -6882,10 +6882,16 @@ export default interface Resources {
|
||||
"installed": "Installed",
|
||||
"modelActions": "Model management",
|
||||
"modelActionsHelp": "Download or remove the Parakeet v3 speech model.",
|
||||
"modelFailed": "Fix or retry the model installation before enabling voice input.",
|
||||
"modelPreparing": "Voice input becomes available after the model installation finishes.",
|
||||
"modelRequired": "Download the Parakeet model before enabling voice input.",
|
||||
"modelStatus": "Parakeet v3 model status",
|
||||
"modelStatusHelp": "The speech model is installed and managed locally on this device.",
|
||||
"notInstalled": "Not installed",
|
||||
"remove": "Remove",
|
||||
"runtimeIncompatible": "Update or reinstall Fusion because the installed voice runtime is incompatible.",
|
||||
"runtimeModuleMissing": "Install a Fusion release that includes the optional voice runtime, then reopen Settings.",
|
||||
"runtimePlatformLoadFailed": "Reinstall Fusion for this platform so the optional voice runtime can load.",
|
||||
"runtimeUnavailable": "Voice runtime unavailable",
|
||||
"statusUnavailable": "Voice runtime status could not be determined; voice mode stays disabled.",
|
||||
"title": "Voice Input",
|
||||
@@ -8587,8 +8593,11 @@ export default interface Resources {
|
||||
"reverted": "Reverted {{taskId}} in commit {{sha}}",
|
||||
"revertedBadge": "Reverted",
|
||||
"revertedBadgeTitle": "This task's changes were reverted",
|
||||
"revertedResolutionActions": "Reverted task resolution actions",
|
||||
"revertedTasks": "Reverted Tasks",
|
||||
"reviewBudgetExhausted": "Review budget exhausted",
|
||||
"reviewerModel": "Reviewer Model",
|
||||
"revise": "Revise",
|
||||
"save": "Save",
|
||||
"saving": "Saving...",
|
||||
"searchTasksPlaceholder": "Search tasks…",
|
||||
|
||||
Reference in New Issue
Block a user