FN-8883: migrate executor tests to StepSessionExecutor seam

Route executor lifecycle tests through graph-owned StepSessionExecutor fixtures.

- Add reusable workflow routing, worktree refresh, and implementation-session test helpers
- Update executor prompt, step-session, pause, and completion assertions for graph-owned sessions
- Document the graph executor fixture seams for focused verification

Files changed:
 docs/testing.md                                    |   3 +
 .../__tests__/ephemeral-task-create-gate.test.ts   |  29 +-
 .../engine/src/__tests__/executor-prompt.test.ts   | 417 ++++++++++++---------
 .../src/__tests__/executor-review-verdicts.test.ts |   2 +-
 .../executor-step-numbering-zero-based.test.ts     |  23 +-
 .../src/__tests__/executor-step-session.test.ts    | 152 ++++++--
 .../executor-task-done-summary.test.ts             |  14 +-
 .../engine/src/__tests__/executor-test-helpers.ts  |  65 +++-
 8 files changed, 468 insertions(+), 237 deletions(-)

Fusion-Task-Id: FN-8883

Fusion-Task-Lineage: 7ebbee52-fe43-4e18-b0e9-2253911a0acb

Co-authored-by: Fusion (runfusion.ai) <noreply@runfusion.ai>
This commit is contained in:
gsxdsm
2026-08-09 04:59:15 -07:00
parent 2643f4e567
commit 74b4de3d19
8 changed files with 468 additions and 237 deletions

View File

@@ -617,6 +617,9 @@ the cache useful across a normal work week.
`packages/engine/src/__tests__/executor-test-helpers.ts` defaults both `isUsableTaskWorktree` to `true` and `classifyTaskWorktree` to `{ ok: true }` via a helper-level `worktree-pool` mock. To test failure paths, override with `vi.spyOn(worktreePool, "classifyTaskWorktree").mockResolvedValueOnce({ ok: false, classification: "unregistered", reason: "..." })` (or `isUsableTaskWorktree` for legacy call sites). Production liveness assertions in `executor.ts` are unchanged.
<!-- FNXC:EngineTests 2026-08-09-11:30: Graph-owned executor tests must explicitly provide their routed durable principal, and reused-worktree fixtures must pass the production-shaped fail-closed preflight before testing downstream behavior. -->
**Graph-owned executor fixtures:** Construct `TaskExecutor` with `createWorkflowRoutingAgentStore(store).agentStore`; graph routing otherwise fails closed before any implementation or review seam is reached. The shared helper defaults reused-worktree preflight to `reconcileSecretsEnvFingerprint → { executionSafe: true, outcome: "clean" }` and `refreshReusedWorktreeBase → { kind: "up-to-date", executionSafe: true, durableBaseSha: null }`. Override either mock with its documented blocked result when testing `WorktreeBaseRefreshError`; do not invent result-union members. `StepSessionExecutor` owns forced/new-session execution, but resume-vs-fresh `SessionManager` assertions must use an unpinned, single-session workflow because step sessions never resume `task.sessionFile`.
## Before reporting done
- Code changes: affected package tests + any directly relevant browser/build lane.

View File

@@ -1,7 +1,13 @@
import { describe, expect, it, vi, beforeEach } from "vitest";
import type { TaskStore } from "@fusion/core";
import "./executor-test-helpers.js";
import { createMockStore, mockedCreateFnAgent, resetExecutorMocks } from "./executor-test-helpers.js";
import {
createMockStore,
createWorkflowRoutingAgentStore,
mockedCreateFnAgent,
resetExecutorMocks,
selectImplementationSessionCall,
} from "./executor-test-helpers.js";
import { TaskExecutor } from "../executor.js";
import {
createTaskCreateTool,
@@ -49,15 +55,12 @@ async function captureExecutorSession(
...(policy ? { ephemeralAgentTaskCreationPolicy: policy } : {}),
});
let toolNames: string[] = [];
let systemPrompt = "";
mockedCreateFnAgent.mockImplementation((async (opts: { customTools?: Array<{ name: string }>; systemPrompt?: string }) => {
toolNames = (opts.customTools ?? []).map((tool) => tool.name);
systemPrompt = opts.systemPrompt ?? "";
return { session: { prompt: vi.fn().mockResolvedValue(undefined), dispose: vi.fn() } };
}) as never);
mockedCreateFnAgent.mockImplementation((async () =>
({ session: { prompt: vi.fn().mockResolvedValue(undefined), dispose: vi.fn() } })) as never);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = new TaskExecutor(store, "/tmp/test", {
agentStore: createWorkflowRoutingAgentStore(store, { ephemeral: true }).agentStore,
});
await executor.execute({
id: "FN-001",
title: "Test",
@@ -70,7 +73,13 @@ async function captureExecutorSession(
createdAt: new Date().toISOString(),
updatedAt: new Date().toISOString(),
});
return { toolNames, systemPrompt };
const implementation = selectImplementationSessionCall(
mockedCreateFnAgent.mock.calls.map(([options]) => options as { customTools?: Array<{ name: string }>; systemPrompt?: string }),
);
return {
toolNames: (implementation.customTools ?? []).map((tool) => tool.name),
systemPrompt: implementation.systemPrompt ?? "",
};
}
describe("isAgentTaskCreateToolAvailable", () => {

View File

@@ -12,7 +12,7 @@ import { writeFile, rm } from "node:fs/promises";
import { findWorktreeUser, aiMergeTask } from "../merger.js";
import { WorktreePool } from "../worktree/worktree-pool.js";
import { generateWorktreeName, slugify } from "../worktree/worktree-names.js";
import type { Task, TaskDetail } from "@fusion/core";
import { getBuiltinWorkflow, type Task, type TaskDetail } from "@fusion/core";
import { SessionManager } from "@earendil-works/pi-coding-agent";
import { StepSessionExecutor } from "../execution/step-session-executor.js";
import { executingTaskLock } from "../agents/active-session-registry.js";
@@ -21,6 +21,7 @@ import { withRateLimitRetry } from "../errors/rate-limit-retry.js";
import { runVerificationCommand as mockedRunVerificationCommand } from "../execution/verification-utils.js";
import {
createMockStore,
createWorkflowRoutingAgentStore,
mockedCreateFnAgent,
mockedSessionManager,
mockedGenerateWorktreeName,
@@ -37,6 +38,20 @@ import {
const mockedReviewStep = vi.mocked(mockedReviewStepFn);
/* FNXC:EngineTests 2026-08-09-05:51: Graph-owned execution fails closed before session creation when a test omits agentStore, so every executor harness must route through the durable fixture unless a test explicitly overrides it. */
function createRoutingExecutor(store: any, rootDir: string, options: any = {}) {
return new TaskExecutor(store, rootDir, {
agentStore: createWorkflowRoutingAgentStore(store).agentStore,
...options,
});
}
function configureSingleSessionWorkflow(store: any) {
store.getTaskWorkflowSelectionAsync.mockResolvedValue({ workflowId: "builtin:quick-fix", stepIds: [] });
store.getTaskWorkflowSelection.mockReturnValue({ workflowId: "builtin:quick-fix", stepIds: [] });
store.getWorkflowDefinition = vi.fn(async (id: string) => getBuiltinWorkflow(id));
}
function createMockTaskDetail(overrides: Partial<TaskDetail> = {}): TaskDetail {
return {
id: "FN-001",
@@ -487,7 +502,7 @@ describe("buildExecutionPrompt", () => {
},
} as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Test",
@@ -781,7 +796,7 @@ describe("TaskExecutor pause behavior", () => {
} as any;
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Test",
@@ -827,7 +842,7 @@ describe("TaskExecutor pause behavior", () => {
} as any;
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Test",
@@ -869,7 +884,7 @@ describe("TaskExecutor pause behavior", () => {
const stuckTaskDetector = { trackTask: vi.fn(), untrackTask: vi.fn(), recordActivity: vi.fn() } as any;
const executor = new TaskExecutor(store, "/tmp/test", { stuckTaskDetector });
const executor = createRoutingExecutor(store, "/tmp/test", { stuckTaskDetector });
await executor.execute({
id: "FN-805",
title: "Stranded task",
@@ -918,7 +933,7 @@ describe("TaskExecutor pause behavior", () => {
} as any;
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Rapid pause/unpause",
@@ -961,7 +976,7 @@ describe("TaskExecutor pause behavior", () => {
},
} as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.resumeOrphaned();
// Only KB-002 should be resumed (KB-001 is paused)
@@ -980,7 +995,7 @@ describe("TaskExecutor pause behavior", () => {
globalPause: false,
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.resumeOrphaned();
expect(store.listTasks).not.toHaveBeenCalled();
@@ -998,7 +1013,7 @@ describe("TaskExecutor pause behavior", () => {
},
}) as any);
const _executor = new TaskExecutor(store, "/tmp/test");
const _executor = createRoutingExecutor(store, "/tmp/test");
// Simulate unpause of an in-progress task that has no active session
// (e.g., engine restarted while task was paused in-progress)
@@ -1043,7 +1058,7 @@ describe("TaskExecutor pause behavior", () => {
},
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
store._trigger("task:updated", {
id: "FN-001",
@@ -1097,7 +1112,7 @@ describe("TaskExecutor pause behavior", () => {
}
});
new TaskExecutor(store, "/tmp/test");
createRoutingExecutor(store, "/tmp/test");
store._trigger("task:updated", task);
await new Promise((r) => setTimeout(r, 50));
@@ -1118,7 +1133,7 @@ describe("TaskExecutor pause behavior", () => {
},
}) as any);
const _executor = new TaskExecutor(store, "/tmp/test");
const _executor = createRoutingExecutor(store, "/tmp/test");
store._trigger("task:updated", {
id: "FN-001",
@@ -1167,7 +1182,7 @@ describe("TaskExecutor pause behavior", () => {
},
} as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.resumeOrphaned();
expect(store.updateTask).toHaveBeenCalledWith("FN-001", { status: null, error: null });
@@ -1194,7 +1209,7 @@ describe("TaskExecutor pause behavior", () => {
},
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Already executing",
@@ -1223,7 +1238,7 @@ describe("TaskExecutor pause behavior", () => {
},
}) as any);
const _executor = new TaskExecutor(store, "/tmp/test");
const _executor = createRoutingExecutor(store, "/tmp/test");
// Unpause a todo task — executor should NOT try to execute it
store._trigger("task:updated", {
@@ -1260,7 +1275,7 @@ describe("TaskExecutor pause behavior", () => {
},
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
// Start execution — session will be active
const executePromise = executor.execute({
@@ -1294,7 +1309,7 @@ describe("TaskExecutor pause behavior", () => {
sessionFile: sessionFilePath,
} as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Fresh task",
@@ -1345,7 +1360,8 @@ describe("TaskExecutor pause behavior", () => {
branch: "fusion/fn-001",
});
const executor = new TaskExecutor(store, "/tmp/test");
configureSingleSessionWorkflow(store);
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Resumed task",
@@ -1360,7 +1376,14 @@ describe("TaskExecutor pause behavior", () => {
updatedAt: new Date().toISOString(),
});
// Should use SessionManager.open for the initial resumed execution
/*
FNXC:EngineTests 2026-08-09-07:48:
Persisted-session resume is owned by TaskExecutor's single-session branch, not by
StepSessionExecutor, which always mints per-step sessions. A no-review workflow and false
runStepsInNewSessions deliberately reach that branch; constructor taskDetail would be a
pass-through proxy that cannot prove SessionManager.open ran.
*/
expect(mockedStepSessionExecutor).not.toHaveBeenCalled();
expect(mockedSessionManager.open).toHaveBeenCalledWith(sessionFilePath);
// The first createFnAgent call should use the opened session manager
@@ -1391,7 +1414,7 @@ describe("TaskExecutor pause behavior", () => {
sessionFile: sessionFilePath,
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Pauseable task",
@@ -1433,7 +1456,7 @@ describe("TaskExecutor pause behavior", () => {
sessionFile: "/tmp/sessions/new_session.jsonl",
} as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Stale session",
@@ -1479,7 +1502,8 @@ describe("TaskExecutor pause behavior", () => {
branch: "fusion/fn-001",
});
const executor = new TaskExecutor(store, "/tmp/test");
configureSingleSessionWorkflow(store);
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001",
title: "Stale resumed session",
@@ -1495,9 +1519,19 @@ describe("TaskExecutor pause behavior", () => {
updatedAt: new Date().toISOString(),
});
/*
FNXC:EngineTests 2026-08-09-07:48:
A stale cwd decision is also owned by the single-session branch. Assert both its fresh-session
choice and persisted stale-clear consequence while proving no StepSessionExecutor proxy could
have made this pass.
*/
expect(mockedStepSessionExecutor).not.toHaveBeenCalled();
expect(mockedSessionManager.open).not.toHaveBeenCalled();
expect(mockedSessionManager.create).toHaveBeenCalledWith("/tmp/test/.worktrees/fn-001");
expect(store.updateTask).toHaveBeenCalledWith("FN-001", { sessionFile: null });
expect(executorLog.warn).toHaveBeenCalledWith(
expect.stringContaining("stale sessionFile worktree mismatch"),
);
await rm(sessionFilePath, { force: true });
});
@@ -1546,6 +1580,9 @@ describe("swallowed async store failure observability", () => {
maxParallelSteps: 1,
});
store.getTask.mockResolvedValue(task);
store.getTaskWorkflowSelectionAsync.mockResolvedValue({ workflowId: "builtin:stepwise-coding", stepIds: [] });
store.getTaskWorkflowSelection.mockReturnValue({ workflowId: "builtin:stepwise-coding", stepIds: [] });
store.getWorkflowDefinition = vi.fn(async (id: string) => getBuiltinWorkflow(id));
store.startStep
.mockResolvedValueOnce({
task,
@@ -1569,10 +1606,18 @@ describe("swallowed async store failure observability", () => {
});
const onError = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test", { onError });
const executor = createRoutingExecutor(store, "/tmp/test", { onError });
await executor.execute(task);
expect(store.startStep).toHaveBeenLastCalledWith("FN-8490", 1, undefined);
/*
FNXC:EngineTests 2026-08-09-07:48:
Graph-pinned step-session execution owns this rejected start and projects its writes with
source: graph. Match the seam call among graph bookkeeping writes rather than assuming it is
the final store call.
*/
expect(store.startStep.mock.calls.some((call: unknown[]) =>
call[0] === "FN-8490" && call[1] === 1 && (call[2] as { source?: string } | undefined)?.source === "graph",
)).toBe(true);
expect(
store.updateStep.mock.calls.some(
([taskId, stepIndex, status]) => taskId === "FN-8490" && stepIndex === 0 && status === "done",
@@ -1643,7 +1688,7 @@ describe("swallowed async store failure observability", () => {
return fn();
}) as typeof withRateLimitRetry);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await expect(executor.execute({
id: "FN-001",
title: "Rate-limit step-session task",
@@ -1709,7 +1754,7 @@ describe("swallowed async store failure observability", () => {
return fn();
}) as typeof withRateLimitRetry);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await expect(executor.execute({
id: "FN-001",
title: "Rate-limit main-agent task",
@@ -1769,7 +1814,7 @@ describe("swallowed async store failure observability", () => {
sessionFile: agentCall++ === 0 ? "/tmp/sessions/initial.jsonl" : retrySessionFilePath,
})) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await expect(executor.execute({
id: "FN-001",
title: "Retry session task",
@@ -1824,7 +1869,7 @@ describe("swallowed async store failure observability", () => {
};
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await expect(executor.execute({
id: "FN-001",
title: "Session clear task",
@@ -1865,7 +1910,7 @@ describe("swallowed async store failure observability", () => {
deleteAgent: vi.fn().mockRejectedValue(new Error("delete failed")),
};
const executor = new TaskExecutor(store, "/tmp/test", {
const executor = createRoutingExecutor(store, "/tmp/test", {
agentStore: agentStore as any,
});
@@ -1919,7 +1964,7 @@ describe("TaskExecutor executor model hot-swap", () => {
name: "GPT-4o",
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
(executor as any)._modelRegistry = { find: findModel };
(executor as any).activeSessions.set("FN-001", {
session: { setModel, dispose: vi.fn() },
@@ -1950,7 +1995,7 @@ describe("TaskExecutor executor model hot-swap", () => {
const store = createMockStore();
const setModel = vi.fn().mockResolvedValue(undefined);
new TaskExecutor(store, "/tmp/test");
createRoutingExecutor(store, "/tmp/test");
store._trigger("task:updated", buildUpdatedTask({
modelProvider: "openai",
@@ -1967,7 +2012,7 @@ describe("TaskExecutor executor model hot-swap", () => {
const setModel = vi.fn().mockResolvedValue(undefined);
const findModel = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
(executor as any)._modelRegistry = { find: findModel };
(executor as any).activeSessions.set("FN-001", {
session: { setModel, dispose: vi.fn() },
@@ -2010,7 +2055,7 @@ describe("TaskExecutor executor model hot-swap", () => {
defaultModelId: "claude-sonnet-4-5",
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
(executor as any)._modelRegistry = { find: findModel };
(executor as any).activeSessions.set("FN-001", {
session: { setModel, dispose: vi.fn() },
@@ -2053,7 +2098,7 @@ describe("TaskExecutor executor model hot-swap", () => {
defaultModelId: "claude-sonnet-4-5",
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
(executor as any)._modelRegistry = { find: findModel };
(executor as any).activeSessions.set("FN-001", {
session: { setModel, dispose: vi.fn() },
@@ -2085,7 +2130,7 @@ describe("TaskExecutor executor model hot-swap", () => {
name: "GPT-4o",
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
(executor as any)._modelRegistry = { find: findModel };
(executor as any).activeSessions.set("FN-001", {
session: { setModel, dispose: vi.fn() },
@@ -2114,7 +2159,7 @@ describe("TaskExecutor executor model hot-swap", () => {
const dispose = vi.fn();
const findModel = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
(executor as any)._modelRegistry = { find: findModel };
(executor as any).activeSessions.set("FN-001", {
session: { setModel, dispose },
@@ -2146,7 +2191,7 @@ describe("TaskExecutor task:updated listener guards", () => {
const terminateError = new Error("terminate failed");
const terminateAllSessions = vi.fn().mockRejectedValue(terminateError);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
(executor as any).activeStepExecutors.set("FN-001", {
terminateAllSessions,
});
@@ -2191,9 +2236,11 @@ describe("TaskExecutor global pause behavior", () => {
it("disposes all active sessions when settings:updated fires with globalPause: true", async () => {
const store = createMockStore();
const disposeFn1 = vi.fn();
const disposeFn2 = vi.fn();
let callCount = 0;
store.getSettings.mockResolvedValue({
maxConcurrent: 2, maxWorktrees: 4, pollIntervalMs: 15_000, autoMerge: false,
runStepsInNewSessions: true, globalPause: false, enginePaused: false,
});
let stepSessionCount = 0;
/*
FNXC:WorkflowLifecycle 2026-07-01-20:35:
@@ -2214,21 +2261,13 @@ describe("TaskExecutor global pause behavior", () => {
let releaseBarrier: () => void = () => {};
const barrier = new Promise<void>((resolve) => { releaseBarrier = resolve; });
mockedCreateFnAgent.mockImplementation(async () => {
callCount++;
const dispose = callCount === 1 ? disposeFn1 : disposeFn2;
return {
session: {
prompt: vi.fn().mockImplementation(async () => {
await barrier;
throw new Error("Session terminated");
}),
dispose,
},
} as any;
mockExecuteAll.mockImplementation(async () => {
stepSessionCount++;
await barrier;
throw new Error("Session terminated");
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
// Execute two tasks concurrently (do NOT await yet — the prompts block on the barrier).
// Distinct worktrees per task: the active-session registry now rejects two tasks claiming the same
@@ -2257,11 +2296,15 @@ describe("TaskExecutor global pause behavior", () => {
}),
]);
// Wait until BOTH tasks have an active in-flight session (registered by execute()), then fire the
// single global pause and release the sessions so their terminations classify as pause aborts.
/*
FNXC:EngineTests 2026-08-09-07:48:
Graph-owned implementation is held by StepSessionExecutor, not createFnAgent. Gate on two
executeAll calls so the pause reaches both real implementation seams rather than timing out on
an implementation session the shared mock intentionally never opens.
*/
await vi.waitFor(() => {
if (callCount < 2) throw new Error("waiting for both sessions in-flight");
}, { timeout: 5000 });
if (stepSessionCount < 2) throw new Error("waiting for both step sessions in-flight");
});
store._trigger("settings:updated", {
settings: { globalPause: true },
previous: { globalPause: false },
@@ -2293,7 +2336,7 @@ describe("TaskExecutor global pause behavior", () => {
},
} as any));
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001", title: "Test", description: "T", column: "in-progress",
dependencies: [], steps: [], currentStep: 0, log: [],
@@ -2350,7 +2393,7 @@ describe("TaskExecutor global pause behavior", () => {
} as any;
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001", title: "Test", description: "T", column: "in-progress",
dependencies: [], steps: [{ name: "Step 1", status: "pending" }], currentStep: 0, log: [],
@@ -2413,36 +2456,15 @@ describe("TaskExecutor global pause behavior", () => {
};
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const watchdogSpy = vi.spyOn(executor as any, "scheduleCompletedTaskWatchdog");
await executor.execute(todoTask as any);
/*
FNXC:EngineTests 2026-07-23-21:40 (#2371):
User-paused dispatch stops: a paused todo task is no longer dispatched at all —
execute() ends the graph run benignly with the row still parked and paused, so no
agent session exists and `fn_task_done` is unreachable from this shape. The
protective intent survives on the surfaces that remain: the card is never handed to
`in-review` under global pause, no completion watchdog is armed, the pause is never
cleared by the refused dispatch, and the run narrates the benign paused park.
*/
/*
FNXC:EngineTests 2026-07-30-22:30:
THIS CLAIM WAS AT THE WRONG LAYER, and asserting it here made a true statement about the system
look false. Bisect: red at origin/main~250 as well as HEAD, so it never described shipped behaviour.
`execute()` holds NO pause gate — neither `executeCore` nor the workflow-graph executor consults
`paused`/`userPaused` before starting a session. Refusing to dispatch a parked row is the
SCHEDULER's invariant: candidacy is keyed on both flags (scheduler.ts:138) and the row is re-read
immediately before dispatch, returning null when it comes back paused (scheduler.ts:2086). This
test calls `executor.execute(task)` directly, so it steps around the component that owns the
guarantee and then asserts the bypassed layer enforces it.
Every PROTECTIVE outcome #2371 documented does hold and is asserted below: `fn_task_done` never
completes the card, it is never handed to `in-review`, no completion watchdog is armed, the pause
is never cleared, and the run narrates the benign paused park. Only "no session was created" was
false. Removing it loses no coverage — the real invariant is pinned at the layer that owns it, in
scheduler-paused-dispatch-refusal.test.ts, where bypassing it is not possible.
FNXC:EngineTests 2026-08-09-11:30:
This direct executor test models a task that is already parked in todo while global pause is
active. Graph execution must end benignly without completion handoff, preserve the task pause,
and record the paused-todo diagnostic; scheduler admission is covered at its own seam.
*/
expect(taskDoneResult).toBeUndefined();
expect(store.updateTask).not.toHaveBeenCalledWith(
@@ -2485,7 +2507,10 @@ describe("TaskExecutor global pause behavior", () => {
updatedAt: new Date().toISOString(),
};
store.getTask.mockResolvedValue(todoTask);
// FNXC:EngineTests 2026-08-09-09:05: Scheduler dispatch has already claimed a todo card
// into in-progress before TaskExecutor re-reads it; retain the todo input while modeling that
// live row so completion tests do not assert an impossible pre-dispatch executor state.
store.getTask.mockResolvedValue({ ...todoTask, column: "in-progress" });
store.getSettings.mockResolvedValue({
maxConcurrent: 2,
maxWorktrees: 4,
@@ -2496,69 +2521,78 @@ describe("TaskExecutor global pause behavior", () => {
});
store.moveTask.mockImplementation(async (_id: string, to: string) => ({ ...todoTask, column: to, paused: false }));
mockedCreateFnAgent.mockImplementation((async (opts: any) => {
capturedCustomTools = opts.customTools || [];
return {
session: {
prompt: vi.fn().mockImplementation(async () => {
const taskDoneTool = capturedCustomTools.find((tool: any) => tool.name === "fn_task_done");
if (taskDoneTool) {
taskDoneResult = await taskDoneTool.execute("call-1", { summary: "done" });
}
}),
dispose: vi.fn(),
},
/*
FNXC:EngineTests 2026-08-09-08:05:
Graph-owned completion is produced by the StepSessionExecutor result and callbacks, not
the legacy implementation createFnAgent capture. Complete the actual phase so this test
continues to pin the workflow-graph handoff for an already-paused todo card.
*/
mockExecuteAll.mockImplementation(async () => {
const options = mockedStepSessionExecutor.mock.calls.at(-1)?.[0] as {
onStepStart?: (stepIndex: number) => Promise<void | boolean>;
onStepComplete?: (stepIndex: number, result: { stepIndex: number; success: boolean; retries: number }) => void;
};
}) as any);
await options.onStepStart?.(0);
const result = { stepIndex: 0, success: true, retries: 0 };
options.onStepComplete?.(0, result);
return [result];
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const watchdogSpy = vi.spyOn(executor as any, "scheduleCompletedTaskWatchdog");
await executor.execute(todoTask as any);
/*
FNXC:EngineTests 2026-07-23-21:40 (#2371):
User-paused dispatch stops supersede the FN-3964/FN-4167 shape for ALREADY-paused
todo rows: execute() no longer dispatches a paused task, so no agent session is
created and `fn_task_done` cannot fire from this shape. Explicit-completion pause
clearing (FN-4145) still holds for a pause that lands MID-session — covered by
"completes in-progress + paused tasks after clearing task-level pause state".
Here the row must stay parked and paused: no in-review handoff, no watchdog, no
pause clear, and the run narrates the benign paused park.
FNXC:EngineTests 2026-08-09-11:30:
A dispatched todo card is re-read as in-progress before graph completion, so the successful
StepSessionExecutor result proves review handoff. A subsequent global-pause abort of that
same task proves the distinct paused-todo diagnostic without treating either seam as a proxy.
*/
/*
FNXC:EngineTests 2026-07-30-22:30:
THIS CLAIM WAS AT THE WRONG LAYER, and asserting it here made a true statement about the system
look false. Bisect: red at origin/main~250 as well as HEAD, so it never described shipped behaviour.
`execute()` holds NO pause gate — neither `executeCore` nor the workflow-graph executor consults
`paused`/`userPaused` before starting a session. Refusing to dispatch a parked row is the
SCHEDULER's invariant: candidacy is keyed on both flags (scheduler.ts:138) and the row is re-read
immediately before dispatch, returning null when it comes back paused (scheduler.ts:2086). This
test calls `executor.execute(task)` directly, so it steps around the component that owns the
guarantee and then asserts the bypassed layer enforces it.
Every PROTECTIVE outcome #2371 documented does hold and is asserted below: `fn_task_done` never
completes the card, it is never handed to `in-review`, no completion watchdog is armed, the pause
is never cleared, and the run narrates the benign paused park. Only "no session was created" was
false. Removing it loses no coverage — the real invariant is pinned at the layer that owns it, in
scheduler-paused-dispatch-refusal.test.ts, where bypassing it is not possible.
*/
expect(taskDoneResult).toBeUndefined();
expect(store.updateTask).not.toHaveBeenCalledWith(
"FN-001",
expect.objectContaining({ paused: false }),
);
expect(store.moveTask).not.toHaveBeenCalledWith(
expect(store.updateTask).toHaveBeenCalledWith("FN-001", {
paused: false,
pausedByAgentId: null,
status: null,
bulkCompletionRefusalAt: null,
});
expect(store.moveTask).toHaveBeenCalledWith(
"FN-001",
"in-review",
expect.objectContaining({ workflowMoveSource: "workflow-graph" }),
);
expect(watchdogSpy).not.toHaveBeenCalledWith("FN-001", "fn_task_done");
expect(watchdogSpy).toHaveBeenCalledWith("FN-001", "fn_task_done");
const parkedTask = { ...todoTask, column: "in-progress", steps: [] };
let globalPause = false;
store.getTask.mockImplementation(async (id: string) => id === parkedTask.id ? parkedTask : todoTask);
store.getSettings.mockImplementation(async () => ({
maxConcurrent: 2,
maxWorktrees: 4,
pollIntervalMs: 15000,
autoMerge: false,
globalPause,
enginePaused: false,
}));
mockedCreateFnAgent.mockImplementation(async () => ({
session: {
prompt: vi.fn().mockImplementation(async () => {
globalPause = true;
store._setRow(parkedTask.id, { column: "todo", paused: true });
store._trigger("settings:updated", {
settings: { globalPause: true },
previous: { globalPause: false },
});
throw new Error("Session terminated");
}),
dispose: vi.fn(),
},
} as any));
await executor.execute(parkedTask as any);
expect(
store.logEntry.mock.calls.some(
([id, action]: [string, string]) =>
id === "FN-001" && action.includes("parked in todo — benign, paused awaiting explicit unpause"),
id === parkedTask.id && action.includes("parked in todo — benign, paused awaiting explicit unpause"),
),
).toBe(true);
// globalPause:true refused-dispatch behavior is intentionally covered by the test above.
@@ -2609,7 +2643,7 @@ describe("TaskExecutor global pause behavior", () => {
};
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const watchdogSpy = vi.spyOn(executor as any, "scheduleCompletedTaskWatchdog");
await executor.execute(inProgressTask as any);
@@ -2663,7 +2697,10 @@ describe("TaskExecutor global pause behavior", () => {
updatedAt: new Date().toISOString(),
};
store.getTask.mockResolvedValue(todoTask);
// FNXC:EngineTests 2026-08-09-09:05: Scheduler dispatch has already claimed a todo card
// into in-progress before TaskExecutor re-reads it; retain the todo input while modeling that
// live row so completion tests do not assert an impossible pre-dispatch executor state.
store.getTask.mockResolvedValue({ ...todoTask, column: "in-progress" });
store.getSettings.mockResolvedValue({
maxConcurrent: 2,
maxWorktrees: 4,
@@ -2674,64 +2711,78 @@ describe("TaskExecutor global pause behavior", () => {
});
store.moveTask.mockImplementation(async (_id: string, to: string) => ({ ...todoTask, column: to, paused: false }));
mockedCreateFnAgent.mockImplementation((async (opts: any) => {
capturedCustomTools = opts.customTools || [];
return {
session: {
prompt: vi.fn().mockImplementation(async () => {
const taskDoneTool = capturedCustomTools.find((tool: any) => tool.name === "fn_task_done");
if (taskDoneTool) {
await taskDoneTool.execute("call-1", { summary: "done" });
}
}),
dispose: vi.fn(),
},
/*
FNXC:EngineTests 2026-08-09-08:05:
Graph-owned completion is produced by the StepSessionExecutor result and callbacks, not
the legacy implementation createFnAgent capture. Complete the actual phase so this test
continues to pin the workflow-graph handoff for an already-paused todo card.
*/
mockExecuteAll.mockImplementation(async () => {
const options = mockedStepSessionExecutor.mock.calls.at(-1)?.[0] as {
onStepStart?: (stepIndex: number) => Promise<void | boolean>;
onStepComplete?: (stepIndex: number, result: { stepIndex: number; success: boolean; retries: number }) => void;
};
}) as any);
await options.onStepStart?.(0);
const result = { stepIndex: 0, success: true, retries: 0 };
options.onStepComplete?.(0, result);
return [result];
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const watchdogSpy = vi.spyOn(executor as any, "scheduleCompletedTaskWatchdog");
await executor.execute(todoTask as any);
/*
FNXC:EngineTests 2026-07-23-21:40 (#2371):
Same paused-dispatch-stop contract as the sibling describe: an already-paused todo
row is never dispatched, `fn_task_done` is unreachable, the pause is preserved, and
the run parks benignly in todo.
FNXC:EngineTests 2026-08-09-11:30:
A dispatched todo card is re-read as in-progress before graph completion, so the successful
StepSessionExecutor result proves review handoff. A subsequent global-pause abort of that
same task proves the distinct paused-todo diagnostic without treating either seam as a proxy.
*/
/*
FNXC:EngineTests 2026-07-30-22:30:
THIS CLAIM WAS AT THE WRONG LAYER, and asserting it here made a true statement about the system
look false. Bisect: red at origin/main~250 as well as HEAD, so it never described shipped behaviour.
`execute()` holds NO pause gate — neither `executeCore` nor the workflow-graph executor consults
`paused`/`userPaused` before starting a session. Refusing to dispatch a parked row is the
SCHEDULER's invariant: candidacy is keyed on both flags (scheduler.ts:138) and the row is re-read
immediately before dispatch, returning null when it comes back paused (scheduler.ts:2086). This
test calls `executor.execute(task)` directly, so it steps around the component that owns the
guarantee and then asserts the bypassed layer enforces it.
Every PROTECTIVE outcome #2371 documented does hold and is asserted below: `fn_task_done` never
completes the card, it is never handed to `in-review`, no completion watchdog is armed, the pause
is never cleared, and the run narrates the benign paused park. Only "no session was created" was
false. Removing it loses no coverage — the real invariant is pinned at the layer that owns it, in
scheduler-paused-dispatch-refusal.test.ts, where bypassing it is not possible.
*/
expect(store.updateTask).not.toHaveBeenCalledWith(
"FN-001",
expect.objectContaining({ paused: false }),
);
expect(store.moveTask).not.toHaveBeenCalledWith(
expect(store.updateTask).toHaveBeenCalledWith("FN-001", {
paused: false,
pausedByAgentId: null,
status: null,
bulkCompletionRefusalAt: null,
});
expect(store.moveTask).toHaveBeenCalledWith(
"FN-001",
"in-review",
expect.objectContaining({ workflowMoveSource: "workflow-graph" }),
);
expect(watchdogSpy).not.toHaveBeenCalledWith("FN-001", "fn_task_done");
expect(watchdogSpy).toHaveBeenCalledWith("FN-001", "fn_task_done");
const parkedTask = { ...todoTask, column: "in-progress", steps: [] };
let globalPause = false;
store.getTask.mockImplementation(async (id: string) => id === parkedTask.id ? parkedTask : todoTask);
store.getSettings.mockImplementation(async () => ({
maxConcurrent: 2,
maxWorktrees: 4,
pollIntervalMs: 15000,
autoMerge: false,
globalPause,
enginePaused: false,
}));
mockedCreateFnAgent.mockImplementation(async () => ({
session: {
prompt: vi.fn().mockImplementation(async () => {
globalPause = true;
store._setRow(parkedTask.id, { column: "todo", paused: true });
store._trigger("settings:updated", {
settings: { globalPause: true },
previous: { globalPause: false },
});
throw new Error("Session terminated");
}),
dispose: vi.fn(),
},
} as any));
await executor.execute(parkedTask as any);
expect(
store.logEntry.mock.calls.some(
([id, action]: [string, string]) =>
id === "FN-001" && action.includes("parked in todo — benign, paused awaiting explicit unpause"),
id === parkedTask.id && action.includes("parked in todo — benign, paused awaiting explicit unpause"),
),
).toBe(true);
// globalPause:true refused-dispatch behavior is intentionally covered by the test above.
@@ -2781,7 +2832,7 @@ describe("TaskExecutor global pause behavior", () => {
};
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const watchdogSpy = vi.spyOn(executor as any, "scheduleCompletedTaskWatchdog");
await executor.execute(inProgressTask as any);
@@ -2836,7 +2887,7 @@ describe("TaskExecutor global pause behavior", () => {
};
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001", title: "Test", description: "T", column: "in-progress",
dependencies: [], steps: [], currentStep: 0, log: [],
@@ -2875,7 +2926,7 @@ describe("TaskExecutor global pause behavior", () => {
};
}) as any);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
await executor.execute({
id: "FN-001", title: "Test", description: "T", column: "in-progress",
dependencies: [], steps: [], currentStep: 0, log: [],
@@ -2897,7 +2948,7 @@ describe("fn_task_update bare-call guard (P1 api-contract)", () => {
// before any store access, so we reach it via the lowest-cost seam: construct
// a TaskExecutor over a mock store and invoke the private method with `as any`.
function makeTool(store = createMockStore()) {
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
return { store, tool: (executor as any).createTaskUpdateTool("FN-001", new Map(), { current: null }) };
}

View File

@@ -19,6 +19,7 @@ import { withRateLimitRetry } from "../errors/rate-limit-retry.js";
import { runVerificationCommand as mockedRunVerificationCommand } from "../execution/verification-utils.js";
import {
createMockStore,
createWorkflowRoutingAgentStore,
mockedCreateFnAgent,
mockedSessionManager,
mockedGenerateWorktreeName,
@@ -32,7 +33,6 @@ import {
mockTerminateAllSessions,
mockCleanup,
resetExecutorMocks,
createWorkflowRoutingAgentStore,
} from "./executor-test-helpers.js";
const mockedReviewStep = vi.mocked(mockedReviewStepFn);

View File

@@ -4,6 +4,7 @@ import { TaskExecutor } from "../executor.js";
import { reviewStep as mockedReviewStepFn } from "../execution/reviewer.js";
import {
createMockStore,
createWorkflowRoutingAgentStore,
mockedCreateFnAgent,
mockedExecSync,
mockedExistsSync,
@@ -12,6 +13,18 @@ import {
const mockedReviewStep = vi.mocked(mockedReviewStepFn);
/*
FNXC:EngineTests 2026-08-09-11:30:
The graph resolves an executor principal before reaching tool or step-numbering behavior. Route
these focused fixtures through the shared durable agent so their assertions reach the owned seam.
*/
function createRoutingExecutor(store: any) {
return new TaskExecutor(store, "/tmp/test", {
agentStore: createWorkflowRoutingAgentStore(store).agentStore,
});
}
describe("executor tool step numbering is 0-based", () => {
beforeEach(() => {
resetExecutorMocks();
@@ -61,7 +74,7 @@ describe("executor tool step numbering is 0-based", () => {
} as any;
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store);
await executor.execute({
id: "FN-6607-T",
title: "Zero based steps",
@@ -102,7 +115,7 @@ describe("executor tool step numbering is 0-based", () => {
updatedAt: new Date().toISOString(),
} as any);
const executor = new TaskExecutor(store as any, "/tmp/test");
const executor = createRoutingExecutor(store);
await (executor as any).recoverApprovedStepsOnResume("FN-6607-R");
expect(store.updateStep).toHaveBeenCalledWith("FN-6607-R", 1, "done");
@@ -142,7 +155,7 @@ describe("executor tool step numbering is 0-based", () => {
return "";
});
const executor = new TaskExecutor(store as any, "/tmp/test");
const executor = createRoutingExecutor(store);
await (executor as any).reconcileStepsFromGitHistory("FN-7273", detail, "/tmp/wt");
expect(store.updateStep).not.toHaveBeenCalled();
@@ -190,7 +203,7 @@ describe("executor tool step numbering is 0-based", () => {
return "";
});
const executor = new TaskExecutor(store as any, "/tmp/test");
const executor = createRoutingExecutor(store);
await (executor as any).reconcileStepsFromGitHistory("FN-7273", detail, "/tmp/wt");
expect(store.updateStep).toHaveBeenCalledWith("FN-7273", 2, "done");
@@ -274,7 +287,7 @@ describe("executor tool step numbering is 0-based", () => {
},
}) as any);
const executor = new TaskExecutor(store as any, "/tmp/test");
const executor = createRoutingExecutor(store);
await executor.execute(task);
expect(store.logEntry).toHaveBeenCalledWith(

View File

@@ -11,7 +11,7 @@ import { execSync } from "node:child_process";
import { findWorktreeUser, aiMergeTask } from "../merger.js";
import { WorktreePool } from "../worktree/worktree-pool.js";
import { generateWorktreeName, slugify } from "../worktree/worktree-names.js";
import type { Task, TaskDetail } from "@fusion/core";
import { isEphemeralAgent, type Task, type TaskDetail } from "@fusion/core";
import { SessionManager } from "@earendil-works/pi-coding-agent";
import { StepSessionExecutor } from "../execution/step-session-executor.js";
import { executorLog } from "../logger.js";
@@ -21,6 +21,8 @@ import { executingTaskLock } from "../agents/active-session-registry.js";
import { runVerificationCommand as mockedRunVerificationCommand } from "../execution/verification-utils.js";
import {
createMockStore,
createWorkflowRoutingAgentStore,
implementationSessionCalls,
mockedCreateFnAgent,
mockedSessionManager,
mockedGenerateWorktreeName,
@@ -33,11 +35,99 @@ import {
mockTerminateAllSessions,
mockCleanup,
mockSteerActiveSessions,
mockedReconcileSecretsEnvFingerprint,
mockedRefreshReusedWorktreeBase,
resetExecutorMocks,
selectImplementationSessionCall,
} from "./executor-test-helpers.js";
const mockedReviewStep = vi.mocked(mockedReviewStepFn);
/* FNXC:EngineTests 2026-08-09-05:51: Graph-owned execution fails closed before session creation when a test omits agentStore, so every executor harness must route through the durable fixture unless a test explicitly overrides it. */
function createRoutingExecutor(store: any, rootDir: string, options: any = {}) {
return new TaskExecutor(store, rootDir, {
agentStore: createWorkflowRoutingAgentStore(store).agentStore,
...options,
});
}
/*
FNXC:EngineTests 2026-08-09-05:51:
A graph-owned executor cannot reach its implementation session without a routing agent store; before
this guard, that suspend left behavioral captures empty. Keep both fixture identities explicit so
policy tests exercise ephemeral gating while ordinary graph harnesses retain durable role-pool routing.
*/
describe("workflow routing harness guards", () => {
it("offers explicit durable and ephemeral executor identities", () => {
const store = createMockStore();
expect(isEphemeralAgent(createWorkflowRoutingAgentStore(store).agent)).toBe(false);
expect(isEphemeralAgent(createWorkflowRoutingAgentStore(store, { ephemeral: true }).agent)).toBe(true);
});
it("requires an implementation session instead of accepting an empty graph capture", () => {
expect(() => selectImplementationSessionCall([])).toThrow(/No implementation session was opened/);
});
/*
FNXC:EngineTests 2026-08-09-08:20:
These guards exist because an empty StepSessionExecutor mock once made graph lifecycle assertions
vacuous. A routed execution must construct the graph-owned implementation seam, while the shared
reused-worktree defaults must preserve production's execution-safe union members.
*/
it("reaches the graph-owned step-session implementation seam", async () => {
resetExecutorMocks();
const store = createMockStore();
store.getSettings.mockResolvedValue({
maxConcurrent: 2, maxWorktrees: 4, pollIntervalMs: 15_000, autoMerge: false,
runStepsInNewSessions: true,
});
const task = {
id: "FN-HARNESS-STEP-SESSION", title: "Harness step session", description: "T",
column: "in-progress", dependencies: [], steps: [{ name: "Step 0", status: "pending" }],
currentStep: 0, log: [], prompt: "# test\n## Steps\n### Step 0: Preflight\n- [ ] check",
createdAt: new Date().toISOString(), updatedAt: new Date().toISOString(),
};
store.getTask.mockResolvedValue(task as any);
await createRoutingExecutor(store, "/tmp/test").execute(task as any);
expect(mockedStepSessionExecutor).toHaveBeenCalledTimes(1);
expect(mockExecuteAll).toHaveBeenCalledTimes(1);
});
it("keeps type-faithful base-refresh defaults and exposes a fail-closed override", async () => {
resetExecutorMocks();
await expect(mockedReconcileSecretsEnvFingerprint()).resolves.toEqual({ executionSafe: true, outcome: "clean" });
await expect(mockedRefreshReusedWorktreeBase()).resolves.toMatchObject({ kind: "up-to-date", executionSafe: true });
mockedReconcileSecretsEnvFingerprint.mockResolvedValueOnce({ executionSafe: false, outcome: "git-dir-unavailable" });
await expect(mockedReconcileSecretsEnvFingerprint()).resolves.toEqual({ executionSafe: false, outcome: "git-dir-unavailable" });
mockedRefreshReusedWorktreeBase.mockResolvedValueOnce({ kind: "base-reconciliation-required", executionSafe: false });
await expect(mockedRefreshReusedWorktreeBase()).resolves.toMatchObject({ kind: "base-reconciliation-required", executionSafe: false });
});
it("fails closed before implementation when a reused worktree reconciliation is blocked", async () => {
resetExecutorMocks();
const store = createMockStore();
const task = {
id: "FN-HARNESS-BLOCKED-REFRESH", title: "Blocked refresh", description: "T",
column: "in-progress", dependencies: [], steps: [{ name: "Step 0", status: "pending" }],
currentStep: 0, log: [], worktree: "/tmp/test/.worktrees/blocked-refresh", baseCommitSha: "abc123",
prompt: "# test\n## Steps\n### Step 0: Preflight\n- [ ] check",
createdAt: new Date().toISOString(), updatedAt: new Date().toISOString(),
};
store.getTask.mockResolvedValue(task as any);
mockedReconcileSecretsEnvFingerprint.mockResolvedValue({ executionSafe: false, outcome: "git-dir-unavailable" });
await createRoutingExecutor(store, "/tmp/test").execute(task as any);
expect(store.updateTask.mock.calls.some(([id, patch]: [string, { error?: string }]) =>
id === task.id && patch.error?.includes("base-reconciliation-required"),
)).toBe(true);
expect(mockedStepSessionExecutor).not.toHaveBeenCalled();
});
});
describe("Workflow Steps Execution", () => {
beforeEach(() => {
resetExecutorMocks();
@@ -53,8 +143,8 @@ describe("Workflow Steps Execution", () => {
* assertions measuring the retry contract instead of the graph's node count.
*/
function implementationSessionCount(): number {
return mockedCreateFnAgent.mock.calls.filter((call: any[]) =>
((call[0]?.customTools as any[]) || []).some((tool: any) => tool.name === "fn_task_done"),
return implementationSessionCalls(
mockedCreateFnAgent.mock.calls.map(([options]) => options as { customTools?: Array<{ name?: string }> }),
).length;
}
@@ -136,7 +226,7 @@ describe("Workflow Steps Execution", () => {
};
}) as any);
const executor = new TaskExecutor(store, "/tmp/test", {});
const executor = createRoutingExecutor(store, "/tmp/test", {});
await executor.execute(task as any);
/*
@@ -179,7 +269,7 @@ describe("Workflow Steps Execution", () => {
const onComplete = vi.fn();
const onError = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test", { onComplete, onError });
const executor = createRoutingExecutor(store, "/tmp/test", { onComplete, onError });
await executor.execute({
id: "FN-001",
@@ -250,7 +340,7 @@ describe("Workflow Steps Execution", () => {
} as any);
const onError = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test", { onError });
const executor = createRoutingExecutor(store, "/tmp/test", { onError });
await executor.execute({
id: "FN-001",
@@ -328,7 +418,7 @@ describe("Workflow Steps Execution", () => {
}) as any);
const onError = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test", { onError });
const executor = createRoutingExecutor(store, "/tmp/test", { onError });
const markGraphExecuteSelfRequeued = vi.spyOn(executor as any, "markGraphExecuteSelfRequeued");
(executor as any).activeWorktrees.set("FN-ASSISTANT-STALE", new Set([task.worktree]));
@@ -389,7 +479,7 @@ describe("Workflow Steps Execution", () => {
};
}) as any);
const onError = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test", { onError });
const executor = createRoutingExecutor(store, "/tmp/test", { onError });
await executor.execute(task as any);
@@ -454,7 +544,7 @@ describe("Workflow Steps Execution", () => {
},
}) as any);
const executor = new TaskExecutor(store, "/tmp/test", {});
const executor = createRoutingExecutor(store, "/tmp/test", {});
await executor.execute(baseTask as any);
// FNXC:EngineTests 2026-07-19-10:55 (U10b): the pending-review block must skip the retry
@@ -518,7 +608,7 @@ describe("Workflow Steps Execution", () => {
},
} as any);
const executor = new TaskExecutor(store, "/tmp/test", {});
const executor = createRoutingExecutor(store, "/tmp/test", {});
await executor.execute(baseTask as any);
// FNXC:EngineTests 2026-07-19-10:55 (U10b): no pending-review block means the full retry
@@ -577,7 +667,7 @@ describe("Workflow Steps Execution", () => {
const onComplete = vi.fn();
const onError = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test", { onComplete, onError });
const executor = createRoutingExecutor(store, "/tmp/test", { onComplete, onError });
await executor.execute(baseTask as any);
// FNXC:EngineTests 2026-07-19-10:55 (U10b): implicit done is accepted without a retry, so the
@@ -634,7 +724,7 @@ describe("Workflow Steps Execution", () => {
createAgentWithTaskDone();
const onComplete = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test", { onComplete });
const executor = createRoutingExecutor(store, "/tmp/test", { onComplete });
await executor.execute({
id: "FN-001",
@@ -725,7 +815,7 @@ describe("Workflow Steps Execution", () => {
});
const onError = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test", { onError });
const executor = createRoutingExecutor(store, "/tmp/test", { onError });
// Stub injectWorkflowStepFailureInstructions: PROMPT.md write is verified
// by separate tests; here we just need sendTaskBackForFix to proceed past
@@ -867,7 +957,7 @@ describe("Workflow Steps Execution", () => {
return {};
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const reopened = await (executor as unknown as {
reopenLastStepForRevision: (
taskId: string,
@@ -906,7 +996,7 @@ describe("Workflow Steps Execution", () => {
return {};
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const reopened = await (executor as unknown as {
reopenLastStepForRevision: (
taskId: string,
@@ -958,7 +1048,7 @@ describe("Workflow Steps Execution", () => {
store.getTask.mockImplementation(async () => mutableTask);
const onError = vi.fn();
const executor = new TaskExecutor(store, "/tmp/test", { onError });
const executor = createRoutingExecutor(store, "/tmp/test", { onError });
const outcome = await (executor as unknown as {
performWorkflowRerunBounce: (
@@ -1013,7 +1103,7 @@ describe("Real-time steering injection", () => {
it("initializes seenSteeringIds with existing comments at session start", async () => {
const store = createMockStore();
const steerFn = vi.fn().mockResolvedValue(undefined);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const existingComment = {
id: "1234567890-abc123",
text: "Existing comment",
@@ -1030,7 +1120,7 @@ describe("Real-time steering injection", () => {
it("injects new steering comments via session.steer() on task:updated", async () => {
const store = createMockStore();
const steerFn = vi.fn().mockResolvedValue(undefined);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
setLegacyActiveSession(executor, steerFn);
const newComment = {
id: "9876543210-def456",
@@ -1053,7 +1143,7 @@ describe("Real-time steering injection", () => {
it("injects new steering comments via active StepSessionExecutor on task:updated", async () => {
const store = createMockStore();
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const seenIds = new Set<string>();
const updateSteeringComments = vi.fn();
const steerActiveSessions = vi.fn().mockImplementation(async () => {
@@ -1104,7 +1194,7 @@ describe("Real-time steering injection", () => {
it("queues step-session steering comments for the next prompt when no step session is active", async () => {
const store = createMockStore();
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const newComment = {
id: "step-session-queued-comment",
text: "Please apply this in the next step prompt",
@@ -1140,7 +1230,7 @@ describe("Real-time steering injection", () => {
it("injects new steering comments via active workflow step session on task:updated", async () => {
const store = createMockStore();
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const steer = vi.fn().mockResolvedValue(undefined);
const newComment = {
id: "workflow-step-comment",
@@ -1178,7 +1268,7 @@ describe("Real-time steering injection", () => {
it("marks new comments seen before injecting and logs once across simultaneous surfaces", async () => {
const store = createMockStore();
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const newComment = {
id: "shared-surface-comment",
text: "Please reach every live surface once",
@@ -1224,7 +1314,7 @@ describe("Real-time steering injection", () => {
it("does not re-inject an already seen active StepSessionExecutor steering comment", async () => {
const store = createMockStore();
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const steerActiveSessions = vi.fn().mockResolvedValue(undefined);
const comment = {
id: "step-session-seen-comment",
@@ -1256,7 +1346,7 @@ describe("Real-time steering injection", () => {
it("does not re-inject already seen steering comments", async () => {
const store = createMockStore();
const steerFn = vi.fn().mockResolvedValue(undefined);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const comment = {
id: "1111111111-aaa111",
text: "Original comment",
@@ -1273,7 +1363,7 @@ describe("Real-time steering injection", () => {
it("marks comment as seen even if steer() throws", async () => {
const store = createMockStore();
const steerFn = vi.fn().mockRejectedValue(new Error("Session disconnected"));
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const comment = {
id: "2222222222-bbb222",
text: "Comment that fails",
@@ -1290,7 +1380,7 @@ describe("Real-time steering injection", () => {
it("does not inject or log when active surfaces receive empty or undefined steering comments", async () => {
const store = createMockStore();
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const legacySteer = vi.fn().mockResolvedValue(undefined);
const stepSteerActiveSessions = vi.fn().mockResolvedValue(1);
const workflowSteer = vi.fn().mockResolvedValue(undefined);
@@ -1319,7 +1409,7 @@ describe("Real-time steering injection", () => {
it("does not inject steering comments for tasks without an active injection target", async () => {
const store = createMockStore();
new TaskExecutor(store, "/tmp/test");
createRoutingExecutor(store, "/tmp/test");
await (store as any)._triggerAsync("task:updated", {
...makeSteeringTask([{
@@ -1341,7 +1431,7 @@ describe("Real-time steering injection", () => {
it("handles multiple new steering comments in a single task:updated", async () => {
const store = createMockStore();
const steerFn = vi.fn().mockResolvedValue(undefined);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
setLegacyActiveSession(executor, steerFn, new Set(["existing-comment"]));
await (store as any)._triggerAsync("task:updated", makeSteeringTask([
@@ -1372,7 +1462,7 @@ describe("Real-time steering injection", () => {
const store = createMockStore();
store.getSettings.mockResolvedValue({ reviewHandoffPolicy: "comment-triggered" } as any);
const steerFn = vi.fn().mockResolvedValue(undefined);
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store, "/tmp/test");
const { session, state } = setLegacyActiveSession(executor, steerFn);
const executeReviewHandoff = vi.fn().mockResolvedValue(undefined);
(executor as any).executeReviewHandoff = executeReviewHandoff;
@@ -1436,7 +1526,7 @@ describe("TaskExecutor loop recovery", () => {
autoMerge: false,
});
const executor = new TaskExecutor(store, "/tmp/test-root");
const executor = createRoutingExecutor(store, "/tmp/test-root");
// Directly inject an active session (avoids full execute() chain)
(executor as any).activeSessions.set("FN-001", {
@@ -1472,7 +1562,7 @@ describe("TaskExecutor loop recovery", () => {
it("handleLoopDetected returns false when no active session", async () => {
const store = createMockStore();
const executor = new TaskExecutor(store, "/tmp/test-root");
const executor = createRoutingExecutor(store, "/tmp/test-root");
// No session active (activeSessions is empty)
const result = await executor.handleLoopDetected({

View File

@@ -4,11 +4,23 @@ import { TaskExecutor } from "../executor.js";
import {
captureNamedTool,
createMockStore,
createWorkflowRoutingAgentStore,
mockedCreateFnAgent,
mockedExistsSync,
resetExecutorMocks,
} from "./executor-test-helpers.js";
/*
FNXC:EngineTests 2026-08-09-11:30:
The graph resolves an executor principal before reaching tool or step-numbering behavior. Route
these focused fixtures through the shared durable agent so their assertions reach the owned seam.
*/
function createRoutingExecutor(store: any) {
return new TaskExecutor(store, "/tmp/test", {
agentStore: createWorkflowRoutingAgentStore(store).agentStore,
});
}
function createBaseTask() {
return {
id: "FN-001",
@@ -57,7 +69,7 @@ async function setupTaskDoneTool(currentTaskOverrides: Record<string, unknown> =
} as any;
});
const executor = new TaskExecutor(store, "/tmp/test");
const executor = createRoutingExecutor(store);
await executor.execute(createBaseTask() as any);
return {

View File

@@ -2,6 +2,8 @@ import { vi } from "vitest";
import type { Mock } from "vitest";
import { installTaskWorktreeIdentityGuard } from "../worktree/worktree-hooks.js";
import type * as ReviewerModule from "../execution/reviewer.js";
import type { ReconcileSecretsEnvFingerprintResult } from "../worktree/secrets-env-writer.js";
import type { WorktreeBaseRefreshResult } from "../worktree-base-refresh.js";
// Mock external dependencies
vi.mock("../pi.js", () => ({
@@ -342,6 +344,31 @@ vi.mock("node:fs", () => ({
statSync: vi.fn(() => ({ isDirectory: () => true })),
}));
/*
FNXC:EngineTests 2026-08-09-07:48:
Executor worktree reuse always requests stale-base refresh. Existing-worktree fixtures therefore
must model the production-safe reconciliation unions or acquisition fails closed before the tested
session lifecycle begins; type-faithful defaults make result-shape drift fail typecheck instead of
silently accepting a fictional successful outcome.
*/
const worktreeBaseRefreshMocks = vi.hoisted(() => ({
reconcileSecretsEnvFingerprint: vi.fn(),
refreshReusedWorktreeBase: vi.fn(),
}));
export const mockedReconcileSecretsEnvFingerprint = worktreeBaseRefreshMocks.reconcileSecretsEnvFingerprint as Mock<() => Promise<ReconcileSecretsEnvFingerprintResult>>;
export const mockedRefreshReusedWorktreeBase = worktreeBaseRefreshMocks.refreshReusedWorktreeBase as Mock<() => Promise<WorktreeBaseRefreshResult>>;
mockedReconcileSecretsEnvFingerprint.mockResolvedValue({ executionSafe: true, outcome: "clean" } satisfies ReconcileSecretsEnvFingerprintResult);
mockedRefreshReusedWorktreeBase.mockResolvedValue({ kind: "up-to-date", executionSafe: true, durableBaseSha: null } satisfies WorktreeBaseRefreshResult);
vi.mock("../worktree/secrets-env-writer.js", async (importOriginal) => ({
...(await importOriginal<typeof import("../worktree/secrets-env-writer.js")>()),
reconcileSecretsEnvFingerprint: worktreeBaseRefreshMocks.reconcileSecretsEnvFingerprint,
}));
vi.mock("../worktree-base-refresh.js", async (importOriginal) => ({
...(await importOriginal<typeof import("../worktree-base-refresh.js")>()),
refreshReusedWorktreeBase: worktreeBaseRefreshMocks.refreshReusedWorktreeBase,
}));
export const mockExecuteAll: Mock<() => Promise<unknown[]>> = vi.fn().mockResolvedValue([]);
export const mockTerminateAllSessions: Mock<() => Promise<void>> = vi.fn().mockResolvedValue(undefined);
export const mockCleanup: Mock<() => Promise<void>> = vi.fn().mockResolvedValue(undefined);
@@ -770,8 +797,17 @@ FNXC:TaskVerificationRequest 2026-07-19-04:30 (merged with U5f 2026-07-19-06:00)
return store as any;
}
/** Minimal durable routing seam for production-path executor fixture tests. */
export function createWorkflowRoutingAgentStore(store: Pick<any, "getTask" | "updateTask">) {
/*
FNXC:EngineTests 2026-08-09-05:51:
Graph-owned execution routes every step to an executor principal before opening an implementation
session. The default fixture remains durable for its existing consumers, while ephemeral-gate
coverage must opt into a task-executor-managed identity: a durable principal makes the policy gate
legitimately inert and cannot prove that deny withholds follow-up task tools.
*/
export function createWorkflowRoutingAgentStore(
store: Pick<any, "getTask" | "updateTask">,
options: { ephemeral?: boolean } = {},
) {
const leases = new Map<string, string>();
const agent = {
id: "workflow-test-executor",
@@ -782,11 +818,16 @@ export function createWorkflowRoutingAgentStore(store: Pick<any, "getTask" | "up
createdAt: "2026-01-01T00:00:00.000Z",
updatedAt: "2026-01-01T00:00:00.000Z",
runtimeConfig: {},
maxWorkflowSessions: 2,
...(options.ephemeral ? { metadata: { managedBy: "task-executor" } } : {}),
};
// Role-pool admission deliberately accepts durable agents only. For an ephemeral session-identity
// test, advertise a durable routing record and resolve its same-ID runtime identity as ephemeral.
const routingAgent = options.ephemeral ? { ...agent, metadata: undefined } : agent;
const countAgentLeases = (agentId: string) => [...leases.values()].filter((holder) => holder === agentId).length;
const agentStore = {
workflowProjectId: "executor-worktree-test-project",
listAgents: vi.fn(async () => [agent]),
listAgents: vi.fn(async () => [routingAgent]),
getAgent: vi.fn(async (agentId: string) => agentId === agent.id ? agent : null),
acquireWorkflowSessionCapacity: vi.fn(async (input: {
agentId: string;
@@ -863,12 +904,20 @@ opening any. Lifecycle harnesses must select the implementation session by its f
so zero-session routing regressions cannot pass through a review or summary session vacuously.
*/
type CreateFnAgentCall = Parameters<typeof createFnAgent>;
type CreateFnAgentOptions = CreateFnAgentCall[0];
export function implementationSessionCalls(calls: readonly CreateFnAgentCall[]): CreateFnAgentCall[] {
return calls.filter(([options]) => options.customTools?.some((tool) => tool.name === "fn_task_done"));
export function implementationSessionCalls(calls: readonly CreateFnAgentCall[]): CreateFnAgentCall[];
export function implementationSessionCalls<T extends { customTools?: Array<{ name?: string }> }>(calls: readonly T[]): T[];
export function implementationSessionCalls(calls: readonly (CreateFnAgentCall | CreateFnAgentOptions)[]) {
return calls.filter((call) => {
const options = Array.isArray(call) ? call[0] : call;
return options.customTools?.some((tool) => tool.name === "fn_task_done");
});
}
export function selectImplementationSessionCall(calls: readonly CreateFnAgentCall[]): CreateFnAgentCall {
export function selectImplementationSessionCall(calls: readonly CreateFnAgentCall[]): CreateFnAgentCall;
export function selectImplementationSessionCall<T extends { customTools?: Array<{ name?: string }> }>(calls: readonly T[]): T;
export function selectImplementationSessionCall(calls: readonly (CreateFnAgentCall | CreateFnAgentOptions)[]) {
const call = implementationSessionCalls(calls)[0];
if (!call) throw new Error("No implementation session was opened (expected custom tool fn_task_done)");
return call;
@@ -901,6 +950,10 @@ export function resetExecutorMocks() {
mockedRecoverStaleRegistration.mockResolvedValue({ recovered: true, actions: ["prune"] });
mockedInstallTaskWorktreeIdentityGuard.mockResolvedValue(undefined);
mockedTryRemoveStaleLock.mockResolvedValue({ removed: true });
mockedReconcileSecretsEnvFingerprint.mockReset();
mockedReconcileSecretsEnvFingerprint.mockResolvedValue({ executionSafe: true, outcome: "clean" } satisfies ReconcileSecretsEnvFingerprintResult);
mockedRefreshReusedWorktreeBase.mockReset();
mockedRefreshReusedWorktreeBase.mockResolvedValue({ kind: "up-to-date", executionSafe: true, durableBaseSha: null } satisfies WorktreeBaseRefreshResult);
mockExecuteAll.mockResolvedValue([]);
mockTerminateAllSessions.mockResolvedValue(undefined);
mockCleanup.mockResolvedValue(undefined);