Files
fusion/packages/engine/src/executor/create-authoritative-workflow-seams.ts
Fusion Agent bfaa0f42da FN-158: enforce workspace multi-repo merge boundaries
Support workspace tasks across a shared root with scope-driven repositories and verifiable merge gates.

- Route task work through one workspace directory with per-repository acquisition and isolation.
- Add sandbox session policies and per-repository verification command handling.
- Enforce required pre-merge checks and honest merge blocking for workspace changes.
- Update workspace, workflow, and sandbox documentation and release metadata.

Files changed:
 .changeset/fn-158-workspace-single-root.md         |   7 +
 docs/sandbox.md                                    |   6 +-
 docs/workflow-steps.md                             |   6 +-
 docs/workspaces.md                                 |  12 +-
 .../core/src/__tests__/legacy-adoption.test.ts     |  15 +-
 .../src/__tests__/required-pre-merge-steps.test.ts |  25 ++++
 .../core/src/__tests__/store-bypass-review.test.ts |  24 ++-
 packages/core/src/__tests__/task-merge.test.ts     |  24 +++
 .../core/src/__tests__/worktree-layout.test.ts     |  29 ++++
 packages/core/src/db/legacy-adoption.ts            |  20 ++-
 packages/core/src/index.gate.ts                    |   4 +
 packages/core/src/index.ts                         |   4 +
 .../core/src/merge/required-pre-merge-steps.ts     |  26 ++++
 packages/core/src/merge/task-merge.ts              |  35 ++++-
 packages/core/src/store.ts                         |  59 ++++++--
 packages/core/src/task-store/lifecycle-ops.ts      |   1 +
 packages/core/src/task-store/merge-queue-ops.ts    |   5 +-
 packages/core/src/task-store/moves.ts              |  13 +-
 packages/core/src/task-store/task-artifacts-ops.ts |  11 +-
 packages/core/src/tasks/worktree-layout.ts         |  43 +++++-
 packages/core/src/types/workflow/workflow-steps.ts |   3 +-
 .../executor-workspace-session-cwd.test.ts         |  42 ++++--
 .../src/__tests__/node-worktree-isolation.test.ts  |  13 +-
 .../src/__tests__/pi-create-fn-agent.test.ts       |  16 ++
 .../engine/src/__tests__/project-engine.test.ts    |  20 ++-
 .../src/__tests__/reviewer-workspace.test.ts       |  30 +++-
 .../src/__tests__/run-verification-command.test.ts |  90 +++++++++++-
 .../__tests__/sandbox/sandbox-exec-policy.test.ts  |  16 +-
 .../src/__tests__/sandbox/session-policy.test.ts   |  45 ++++++
 .../__tests__/workspace-add-repo-midflight.test.ts |   9 ++
 .../engine/src/__tests__/workspace-e2e.test.ts     |  13 +-
 .../workspace-root-worktree-routing.test.ts        |  18 +--
 packages/engine/src/agent-tools.ts                 |  15 +-
 packages/engine/src/agents/agent-runtime.ts        |  19 +++
 .../engine/src/agents/agent-session-helpers.ts     |  15 ++
 packages/engine/src/execution/hold-release.ts      |  29 ++++
 .../engine/src/execution/run-verification-tool.ts  | 114 ++++++++++++++-
 .../create-authoritative-workflow-seams.ts         |  20 +--
 packages/engine/src/executor/deps-bags.ts          |   5 +-
 .../executor/ensure-graph-custom-node-worktree.ts  |  24 ++-
 .../executor/ensure-task-worktree-for-planning.ts  |  36 ++---
 .../engine/src/executor/execute-workflow-step.ts   |   4 +-
 .../src/executor/finalize-already-reviewed-task.ts |   6 +-
 .../src/executor/prepare-graph-node-execution.ts   |  14 +-
 .../engine/src/executor/run-graph-custom-node.ts   | 161 +++++++++++++--------
 packages/engine/src/executor/run-implementation.ts | 117 ++++++++++++---
 packages/engine/src/merge/merger-ai.ts             |  16 +-
 packages/engine/src/merger.ts                      |  20 ++-
 packages/engine/src/pi.ts                          | 150 ++++++++++++++++---
 packages/engine/src/project-engine.ts              |  10 +-
 packages/engine/src/runtimes/in-process-runtime.ts |   4 +-
 packages/engine/src/sandbox/bubblewrap-backend.ts  |  55 ++++++-
 packages/engine/src/sandbox/bubblewrap-policy.ts   |  10 +-
 packages/engine/src/sandbox/index.ts               |   1 +
 .../engine/src/sandbox/sandbox-exec-backend.ts     |  45 +++++-
 packages/engine/src/sandbox/sandbox-exec-policy.ts |  18 ++-
 packages/engine/src/sandbox/session-policy.ts      |  41 ++++++
 packages/engine/src/sandbox/types.ts               |  11 ++
 packages/engine/src/self-healing.ts                |   1 +
 packages/engine/src/triage.ts                      |  10 ++
 .../engine/src/worktree/worktree-acquisition.ts    |  53 ++++---
 61 files changed, 1393 insertions(+), 315 deletions(-)

Fusion-Task-Id: FN-158

Fusion-Task-Lineage: ba57f5a2-fa69-4210-8ea7-3d124be3deb2

Co-authored-by: Fusion <noreply@runfusion.ai>
2026-08-23 01:37:58 +00:00

621 lines
35 KiB
TypeScript

/**
* FNXC:CodeOrganization 2026-08-03-14:20:
* createAuthoritativeWorkflowSeams peeled from TaskExecutor (U4).
*
* FNXC:WorkflowExecutionOwnership 2026-07-27-16:25 / 2026-07-28-20:25:
* Seam return vocabulary is the ownership boundary; exit events announce without changing outcomes.
*/
import type { AgentStore, ResolvedTaskOutputLanguage, Settings, TaskStore, ThinkingLevel, WorkspaceConfig } from "@fusion/core";
import { emitWorkflowLifecycleEvent, resolveTaskOutputLanguage, THINKING_LEVELS } from "@fusion/core";
import type { ImplementationExit } from "./implementation-exit.js";
import type { WorkflowLegacySeams } from "../workflows/workflow-node-handlers.js";
import type { AgentSemaphore } from "../concurrency/concurrency.js";
import {
FOREACH_ACTIVE_CONTEXT_KEY,
SEAM_GOVERNING_NODE_CONTEXT_KEY,
SEAM_SKILL_NAME_CONTEXT_KEY,
SEAM_THINKING_LEVEL_CONTEXT_KEY,
type ForeachActiveContext,
} from "../workflows/workflow-node-handlers.js";
import { graphActiveContextKey } from "./task-predicates.js";
import { WorkflowReviewService } from "../workflows/workflow-review-service.js";
import { MERGE_BOUNDARY_UNPROVEN_VALUE } from "../workflows/workflow-merge-nodes.js";
import { mergeEffectiveSettings } from "../project/effective-settings.js";
import { resolveReviewCheckoutCwd } from "../execution/review-checkout.js";
import { logReviewCheckoutRouting } from "./review-checkout-routing.js";
import { selectUserCommentsForAgentContext } from "../agents/agent-user-comments.js";
import {
resolveValidatorThinkingLevel,
resolveValidatorFallbackThinkingLevel,
} from "../agents/agent-session-helpers.js";
import type { ReviewResult } from "../execution/reviewer.js";
import {
buildReviewUnavailableMessage,
buildPlanVerifiedMessage,
buildReviewVerdictMessage,
emitProactiveStatus,
sanitizeFailureReason,
} from "../project/proactive-status.js";
import type { EngineRunContext } from "../util/run-audit.js";
import { executorLog, reviewerLog } from "../logger.js";
import { normalizeWorkspaceTaskRouting } from "./workspace-config-resolver.js";
const WORKFLOW_THINKING_LEVEL_SET: ReadonlySet<string> = new Set(THINKING_LEVELS);
// eslint-disable-next-line @typescript-eslint/no-explicit-any -- mirror TaskExecutor method surface
type AnyFn = (...args: any[]) => any;
/*
FNXC:WorkspaceReviewEvidence 2026-08-21-19:52:
Both authoritative and graph review producers must publish the exact approval shape landing reads.
Keep this fenced writer exported so the real producer-to-consumer regression cannot recreate it in a test.
*/
export async function persistWorkspaceCodeReviewApproval(
store: TaskStore,
taskId: string,
review: Pick<ReviewResult, "verdict" | "repositoryScopeRevision" | "repositoryDiffFingerprints" | "repositoryModifiedFiles">,
): Promise<boolean> {
if (review.repositoryScopeRevision === undefined) return false;
let superseded = false;
const approvedAt = new Date().toISOString();
await store.updateTaskAtomic(taskId, (current) => {
const scope = current.repositoryScope;
if (!scope || scope.revision !== review.repositoryScopeRevision) {
superseded = true;
return null;
}
if (review.verdict !== "APPROVE" || !review.repositoryDiffFingerprints || Object.keys(review.repositoryDiffFingerprints).length === 0) return null;
return {
repositoryScope: {
...scope,
reviewEvidence: Object.fromEntries(Object.entries(review.repositoryDiffFingerprints).map(([repo, fingerprint]) => [repo, { fingerprint, approvedAt }])),
...(scope.reviewRemediation?.scopeRevision === review.repositoryScopeRevision ? { reviewRemediation: undefined } : {}),
},
...(review.repositoryModifiedFiles ? { modifiedFiles: review.repositoryModifiedFiles } : {}),
};
});
return superseded;
}
export type CreateAuthoritativeWorkflowSeamsDeps = {
store: TaskStore;
rootDir: string;
options: {
agentStore?: AgentStore | null;
pluginRunner?: unknown;
semaphore?: AgentSemaphore;
mergeRequester?: unknown;
[k: string]: unknown;
};
workspaceConfig: WorkspaceConfig | null | undefined;
ensureWorkspaceConfig?: () => Promise<WorkspaceConfig | null>;
activeWorkflowPrincipals: Map<string, { agentId: string; nodeInstanceId: string; agent?: import("@fusion/core").Agent }>;
graphSeamGoverningNodeId: Map<string, string>;
graphSeamThinkingLevel: Map<string, ThinkingLevel>;
graphStepActiveContext: Map<string, unknown>;
graphRethinkNarrations: Map<string, unknown>;
pausedAborted: Set<string>;
// eslint-disable-next-line @typescript-eslint/no-explicit-any
mergeRequester?: ((taskId: string, opts?: any) => Promise<any>) | null;
getRunContextFor: (taskId: string) => EngineRunContext | undefined;
persistTokenUsage: AnyFn;
runImplementationPhase: AnyFn;
handoffTaskToReview: AnyFn;
ensureWorkflowMergeBoundaryTask: AnyFn;
getWorkflowMergeImplementationProofFailure: AnyFn;
runProjectedGraphTaskStep: AnyFn;
updateStepGraph: AnyFn;
reviewWorkspacePerRepo: AnyFn;
registerSubagentSession: AnyFn;
unregisterSubagentSession: AnyFn;
};
export function createAuthoritativeWorkflowSeams(
deps: CreateAuthoritativeWorkflowSeamsDeps,
settings: Settings,
outputLanguage?: ResolvedTaskOutputLanguage,
): WorkflowLegacySeams {
return {
// Built-in triage/spec generation runs upstream of the interpreter today,
// so planning is a no-op for already-specified tasks. Custom planning
// behavior is expressed as a custom prompt node before the execute seam.
planning: async () => ({ outcome: "success", value: "pre-specified" }),
execute: async (seamTask, context) => {
// Column-agent seam wiring (U4, R4): record the governing node id (the
// execute-seam prompt node, stamped into context by createPromptLikeHandler)
// so execute()'s session build can resolve the column-agent binding for the
// node's DECLARED column. Cleared after the pass so a later seam without a
// binding cannot inherit a stale node id.
const governingNodeId = context?.[SEAM_GOVERNING_NODE_CONTEXT_KEY];
if (typeof governingNodeId === "string") {
deps.graphSeamGoverningNodeId.set(seamTask.id, governingNodeId);
}
const seamThinkingLevel = context?.[SEAM_THINKING_LEVEL_CONTEXT_KEY];
if (typeof seamThinkingLevel === "string" && WORKFLOW_THINKING_LEVEL_SET.has(seamThinkingLevel)) {
deps.graphSeamThinkingLevel.set(seamTask.id, seamThinkingLevel as ThinkingLevel);
}
let result: { taskDone: boolean; modifiedFiles: string[]; exit?: ImplementationExit };
try {
result = await deps.runImplementationPhase(seamTask);
} finally {
deps.graphSeamGoverningNodeId.delete(seamTask.id);
deps.graphSeamThinkingLevel.delete(seamTask.id);
}
/*
FNXC:WorkflowExecutionOwnership 2026-07-27-16:25 (U8 / R4):
THIS BOOLEAN IS THE OWNERSHIP BOUNDARY, and it is too narrow. `runImplementation` has
28 measured ways of disposing of a task (16 column moves, 3 review handoffs, 9 terminal
parks — counted by `executor-lifecycle-ownership-ledger.test.ts`) and exactly 3 ways of
telling the graph anything, all of which collapse to `taskDone: true` here.
The consequence is not a missing feature, it is a second lifecycle owner. Because the
seam has no value for "the agent stopped because a step is blocked on a pending review"
or "the session was paused after the work was already complete", the implementation
phase performs those transitions ITSELF (`executor-exit-while-review-pending`,
`paused-after-completion`) and the graph learns about them afterwards — which is why
`handleGraphFailure` carries `alreadyFinalizedToReview` / `completionFinalized`
classifiers whose whole job is to recognise a move the graph did not make.
U8's direction: widen this vocabulary so a disposition is REPORTED here and the graph
routes it, rather than performed upstream and compensated for downstream. The
compensating classifiers are the acceptance test — they become unreachable, and then
deletable, exactly when the last out-of-band transition is gone.
*/
/*
FNXC:WorkflowExecutionOwnership 2026-07-28-20:25 (U8 / R4, R5):
Announce the exit on the U3 lifecycle bus. Until this, the two out-of-band review
handoffs left NO trace anywhere that the executor — not the graph — moved the card;
they surfaced as an ordinary `implementation-incomplete` failure that
`handleGraphFailure` then quietly compensated for. An operator could not tell the two
apart, and neither could a test.
Emission is deliberately AFTER the phase and BEFORE the return, and it changes nothing:
the outcome/value below are byte-identical to what this seam returned before, for every
exit, which `executor-implementation-exit-events.test.ts` pins by driving each exit and
asserting the seam's return. Per R5 an exit id is a REACTION — dropping every subscriber
must change no execution outcome, and that is asserted too.
*/
emitWorkflowLifecycleEvent({
type: "NodeCompleted",
taskId: seamTask.id,
at: new Date().toISOString(),
runId: deps.getRunContextFor(seamTask.id)?.runId,
nodeId: typeof governingNodeId === "string" ? governingNodeId : "execute",
outcome: result.taskDone ? "success" : "failure",
...(result.exit ? { exit: result.exit } : {}),
});
if (result.taskDone) {
return { outcome: "success", value: "implemented" };
}
// Distinguish pause/abort from genuine implementation failure so the
// failure handler can leave paused tasks to the pause machinery.
let paused = deps.pausedAborted.has(seamTask.id);
if (!paused) {
try {
paused = Boolean((await deps.store.getTask(seamTask.id)).paused);
} catch {
// Best-effort pause probe; fall through to the failure value.
}
}
return {
outcome: "failure",
value: paused ? "implementation-paused" : "implementation-incomplete",
};
},
// FNXC:WorkflowExecution 2026-06-25-00:00: U4 (KTD-2) — the legacy
// `workflowStep` seam was removed. Workflow quality gates run as the graph's
// own optional-group / gate nodes (builtin:coding replaced its `workflow-step`
// seam node with optional-group nodes) which record into
// `task.workflowStepResults` (U2). `WorkflowLegacySeams.workflowStep` no
// longer exists, and `resolveSeamName` no longer recognizes the
// `workflow-step` seam (an IR node still declaring it now fails loudly via
// WorkflowIrError rather than silently no-opping).
review: async (seamTask) => {
// The legacy "review" stage is the in-review handoff: the in-review column is
// the staging state the merge queue consumes.
const live = await deps.store.getTask(seamTask.id);
await deps.persistTokenUsage(seamTask.id);
/*
FNXC:TaskOutputLanguage 2026-08-19-16:25:
Legacy graph seams receive the invocation's resolved target rather than re-detecting a
mutable live description. Direct seam callers retain the compatibility fallback.
*/
await deps.handoffTaskToReview(live, "workflow-graph-review", undefined, outputLanguage ?? resolveTaskOutputLanguage(settings, live.description));
return { outcome: "success", value: "in-review" };
},
"review-handoff": async (seamTask) => {
/*
* FNXC:WorkflowPrPolicy 2026-06-29-16:42:
* Compound Engineering can run an optional manual PR review lane after implementation. That lane must start from the review column without invoking the generic reviewer again; this seam is a pure lifecycle handoff so PR creation/feedback nodes run while the card is visibly in review.
*/
const live = await deps.store.getTask(seamTask.id);
await deps.persistTokenUsage(seamTask.id);
/* FNXC:TaskOutputLanguage 2026-08-19-16:25: Manual handoff uses the same invocation snapshot as the review seam. */
await deps.handoffTaskToReview(live, "workflow-graph-review-handoff", undefined, outputLanguage ?? resolveTaskOutputLanguage(settings, live.description));
return { outcome: "success", value: "in-review" };
},
merge: async (seamTask, _context, signal) => {
if (!deps.mergeRequester) {
return { outcome: "failure", value: "merge-unavailable" };
}
// FNXC:WorkflowCancellation 2026-07-15-10:42: fail fast before the boundary-task mutation and the merge request — an abandoned walk must not enqueue a merge. Mirrors the `requestMerge` primitive.
if (signal?.aborted) {
return { outcome: "failure", value: "merge-cancelled" };
}
const mergeBoundary = await deps.ensureWorkflowMergeBoundaryTask(seamTask, {
reason: "workflow-merge-boundary",
nodeId: "legacy-merge-seam",
workflowId: "legacy-seams",
runId: deps.getRunContextFor(seamTask.id)?.runId ?? "legacy-seam",
});
/* FNXC:WorkflowMerge 2026-08-20-00:50: FN-9157 keeps the legacy seam terminal too; a retry cannot create missing boundary proof. */
if (mergeBoundary.blocked) return { outcome: "failure", value: MERGE_BOUNDARY_UNPROVEN_VALUE };
const mergeTask = mergeBoundary.task;
const missingImplementationProof = await deps.getWorkflowMergeImplementationProofFailure(mergeTask);
if (missingImplementationProof) {
await deps.store.logEntry(
mergeTask.id,
`Workflow merge blocked before requester: ${missingImplementationProof}`,
undefined,
deps.getRunContextFor(mergeTask.id),
);
return { outcome: "failure", value: "implementation-incomplete" };
}
// Bound the wait: a wedged merge queue must not strand the graph walk
// holding the routing claim. On timeout the run fails cleanly and the
// task is parked for human review; the queue can still finish later.
// FNXC:WorkflowCancellation 2026-07-15-10:42: the timeout is the wedged-queue bound, `signal` is the cancellation path — both must stay live. See the `requestMerge` primitive for the stall this prevents.
const GRAPH_MERGE_TIMEOUT_MS = 30 * 60 * 1000;
let timeoutHandle: ReturnType<typeof setTimeout> | undefined;
const timeout = new Promise<"timeout">((resolve) => {
timeoutHandle = setTimeout(() => resolve("timeout"), GRAPH_MERGE_TIMEOUT_MS);
timeoutHandle.unref?.();
});
let onGraphAbort: (() => void) | undefined;
const cancelled = new Promise<"cancelled">((resolve) => {
if (!signal) return;
onGraphAbort = () => resolve("cancelled");
signal.addEventListener("abort", onGraphAbort, { once: true });
});
try {
const result = await Promise.race([deps.mergeRequester(mergeTask.id, signal ? { signal } : undefined), timeout, cancelled]);
if (result === "cancelled") {
executorLog.warn(`${mergeTask.id}: graph merge seam cancelled by graph abort`);
return { outcome: "failure", value: "merge-cancelled" };
}
if (result === "timeout") {
executorLog.warn(`${mergeTask.id}: graph merge seam timed out after ${GRAPH_MERGE_TIMEOUT_MS}ms`);
return { outcome: "failure", value: "merge-timeout" };
}
if (result.merged || result.noOp) {
return { outcome: "success", value: result.noOp ? "merge-noop" : "merged" };
}
return { outcome: "failure", value: result.reason ?? result.error ?? "merge-failed" };
} finally {
if (timeoutHandle) clearTimeout(timeoutHandle);
if (onGraphAbort) signal?.removeEventListener("abort", onGraphAbort);
}
},
schedule: async () => ({ outcome: "success" }),
// Step-inversion (KTD-2/KTD-4, U3): run exactly the foreach-active step.
// The foreach sub-walk has set `foreach:active` with the step index; here
// we drive runTaskStep (step-runner.ts) over the task's worktree, then
// capture the per-step baselineSha/checkpointId back INTO the active
// context object so a later RETHINK (U5) can reset the step. The full
// single-step session physics (a StepSessionExecutor scoped to one step)
// is U5/U7 territory; U3 wires the seam and the context capture, using the
// existing implementation phase as the single-pass step driver.
stepExecute: async (seamTask, context) => {
const active = context[FOREACH_ACTIVE_CONTEXT_KEY] as ForeachActiveContext | undefined;
if (!active || typeof active.stepIndex !== "number") {
return { outcome: "failure", value: "no-active-step-instance" };
}
const live = await deps.store.getTask(seamTask.id);
// Worktree isolation (KTD-11, U10): run the instance's session in ITS OWN
// worktree when the foreach allocated one; otherwise the task's main
// worktree (shared isolation — unchanged). The file-scope guard the session
// machinery installs applies to either worktree unchanged (not bypassed).
// Stamp the active instance so `runGraphTaskStep` can honor
// `deferDoneToReview` when judging a non-terminal step (FIX 3).
deps.graphStepActiveContext.set(graphActiveContextKey(seamTask.id, active.instanceId), active);
// Column-agent seam wiring (U4, R4): the governing node id — the foreach
// INSTANCE node id (`<foreachId>#<i>:<templateNodeId>`) stamped into
// context by createPromptLikeHandler — threads INTO runGraphTaskStep,
// which stamps the per-task slot only when it CREATES the memoized
// implementation pass and clears it when that pass settles (PR #1432
// review). One step-session pass serves every instance, so the
// session-identity binding is deterministically the pass-initiating
// instance's; per-invocation set/delete here would race under parallel
// foreach (overwrite mid-build, or clear while the shared pass is live).
const stepGoverningNodeId = context[SEAM_GOVERNING_NODE_CONTEXT_KEY];
const seamThinkingLevel = context[SEAM_THINKING_LEVEL_CONTEXT_KEY];
const seamSkillName = context[SEAM_SKILL_NAME_CONTEXT_KEY];
const result = await deps.runProjectedGraphTaskStep(
seamTask,
live,
active.stepIndex,
active,
typeof stepGoverningNodeId === "string" ? stepGoverningNodeId : undefined,
typeof seamThinkingLevel === "string" && WORKFLOW_THINKING_LEVEL_SET.has(seamThinkingLevel)
? (seamThinkingLevel as ThinkingLevel)
: undefined,
typeof seamSkillName === "string" && seamSkillName.trim() ? seamSkillName.trim() : undefined,
);
// Capture baseline/checkpoint back into the reserved active context so the
// foreach sub-walk threads them to later template nodes (step-review/reset).
active.baselineSha = result.baselineSha;
active.checkpointId = result.checkpointId;
/*
FNXC:WorkflowExecutionOwnership 2026-07-29-11:30 (U8 / R4):
`step-done` / `step-failed` was a two-value flattening of every possible ending, and it
is why the pending-review ending could never reach an edge on the stepwise shape. A
blocked-on-pending-review pass is a WAIT, not a step defect: the outcome stays `failure`
(the step genuinely did not complete) while the VALUE names the ending, which is what the
foreach propagates upward — `runForeach` returns a failing instance's value as its own —
so the `steps` node can carry an `outcome:review-pending` edge to the park node.
Every other ending keeps `step-failed` exactly as before.
*/
const failureValue = result.exit === "review-handoff-pending-review" ? "review-pending" : "step-failed";
return {
outcome: result.outcome,
value: result.outcome === "success" ? "step-done" : failureValue,
contextPatch: {
[FOREACH_ACTIVE_CONTEXT_KEY]: active,
},
};
},
// Step-inversion (KTD-4, U5): review the foreach-active step. Mirrors the
// legacy in-session review call (deleted in U10): run
// reviewStep under semaphore.runNested against the instance's step number/
// name and the task's PROMPT content. On an authoritative (non-advisory)
// APPROVE, mark the step done through the projection (updateStep, KTD-7) —
// the step-execute seam left it in-progress (markDoneOnSuccess:false) so the
// review is the single done authority. The handler maps the returned verdict
// to outcome edges and applies the UNAVAILABLE bounded-retry limiter.
stepReview: async (seamTask, context, config) => {
const active = context[FOREACH_ACTIVE_CONTEXT_KEY] as ForeachActiveContext | undefined;
if (!active || typeof active.stepIndex !== "number") {
// No active instance — surface UNAVAILABLE so the handler routes it
// rather than fabricating an authoritative verdict.
return { verdict: "UNAVAILABLE", review: "no active step instance" };
}
const stepIndex = active.stepIndex;
let detail = await deps.store.getTask(seamTask.id);
const workspaceConfig = deps.ensureWorkspaceConfig
? await deps.ensureWorkspaceConfig()
: deps.workspaceConfig;
if (workspaceConfig) {
/*
FNXC:WorkspaceRootRouting 2026-08-19-12:15:
A code review must not interpret historical singular root metadata as its checkout. Repair
that metadata before selecting a cwd, then let reviewWorkspacePerRepo consume only the
durable declared-repository entries; an empty map remains a real unavailable review.
*/
detail = await normalizeWorkspaceTaskRouting(deps.store, seamTask.id) as typeof detail;
}
// Workspace Code Review fans out over durable repository entries. Plan Review
// reads the workspace under its declared read-only boundary; only single-repo
// review resolves a checkout here.
const worktreePath = active.worktreePath || detail.worktree || deps.rootDir;
const reviewCwd = workspaceConfig ? deps.rootDir : resolveReviewCheckoutCwd(detail, worktreePath);
if (reviewCwd) logReviewCheckoutRouting(seamTask.id, detail, reviewCwd, worktreePath);
const stepName = detail.steps[stepIndex]?.name ?? `Step ${stepIndex}`;
const promptContent = detail.prompt ?? "";
const planScopeContext = workspaceConfig && config.type === "plan"
? `\n\nRepository scope (task-level; review this plan once): ${detail.repositoryScope?.repositories.join(", ") || "unconfirmed"}.`
: "";
const userComments = selectUserCommentsForAgentContext(detail, { limit: null });
// Merge per-task effective workflow settings (U3, KTD-3) so the validator
// model-lane reads below pick up workflow values. Behavior-inert by default.
const settings = await mergeEffectiveSettings(deps.store, detail, await deps.store.getSettings());
/*
FNXC:AgentSteering 2026-06-30-12:37:
Workflow graph step-review nodes are optional or mandatory reviewer gates. Pass canonical user comments and legacy steering into each per-cwd reviewer so workspace aggregation never drops operator requirements.
FNXC:AgentSteering 2026-06-30-13:20:
Graph reviewer gates request uncapped comment context because every user-authored requirement can affect approval, including older steering retained on long-running tasks.
*/
const sem = deps.options.semaphore;
// FNXC:Workspace 2026-06-22-00:30: KTD3 — step-inversion review seam loops per sub-repo.
// `reviewStep` stays single-cwd; THIS CALLER loops. Single-cwd by default reviews
// `worktreePath`; in workspace mode that is the browse-only non-git root, so we instead spawn
// one reviewer per acquired sub-repo (cwd = repo.worktreePath) via reviewWorkspacePerRepo and
// aggregate as a conjunction. `invokeReviewerForCwd` is the per-cwd reviewStep call both modes share.
const reviewService = new WorkflowReviewService();
const invokeReviewerForCwd = (cwd: string) =>
reviewService.reviewStep({
cwd,
taskId: seamTask.id,
stepIndex,
stepName,
type: config.type,
promptContent: `${promptContent}${planScopeContext}`,
// Code reviews diff against the per-step baseline captured at
// step-execute; plan reviews pass no baseline (advisory).
baselineSha: config.type === "code" ? active.baselineSha : undefined,
options: {
defaultProvider: settings.defaultProvider,
defaultModelId: settings.defaultModelId,
fallbackProvider: settings.fallbackProvider,
fallbackModelId: settings.fallbackModelId,
/*
* FNXC:Settings-ThinkingLevel 2026-07-13-00:27:
* Step-review model sessions honor per-node `config.thinkingLevel` before the task validator override, then shared task thinking, validator workflow lane, global lane, and default thinking settings.
*/
defaultThinkingLevel: resolveValidatorThinkingLevel(
typeof config.thinkingLevel === "string" && WORKFLOW_THINKING_LEVEL_SET.has(config.thinkingLevel)
? (config.thinkingLevel as ThinkingLevel)
: detail.validatorThinkingLevel ?? detail.thinkingLevel,
settings,
),
fallbackThinkingLevel: resolveValidatorFallbackThinkingLevel(
typeof config.thinkingLevel === "string" && WORKFLOW_THINKING_LEVEL_SET.has(config.thinkingLevel)
? (config.thinkingLevel as ThinkingLevel)
: detail.validatorThinkingLevel ?? detail.thinkingLevel,
settings,
),
taskValidatorProvider: detail.validatorModelProvider,
taskValidatorModelId: detail.validatorModelId,
taskValidatorCredentialInstanceId: detail.validatorCredentialInstanceId,
projectValidatorProvider: settings.validatorProvider,
projectValidatorModelId: settings.validatorModelId,
projectValidatorFallbackProvider: settings.validatorFallbackProvider,
projectValidatorFallbackModelId: settings.validatorFallbackModelId,
globalValidatorProvider: settings.validatorGlobalProvider,
globalValidatorModelId: settings.validatorGlobalModelId,
projectDefaultOverrideProvider: settings.defaultProviderOverride,
projectDefaultOverrideModelId: settings.defaultModelIdOverride,
store: deps.store,
taskId: seamTask.id,
task: detail,
userComments: userComments.length > 0 ? userComments : undefined,
agentPrompts: settings.agentPrompts,
agentStore: deps.options.agentStore ?? undefined,
rootDir: deps.rootDir,
settings,
/* FNXC:WorkflowAgentRouting 2026-08-07-04:45: reviewer sessions inherit the exact graph-fenced principal, including a node-local override. */
agentId: deps.activeWorkflowPrincipals.get(seamTask.id)?.agentId,
onSessionCreated: (s) => deps.registerSubagentSession(seamTask.id, s),
onSessionEnded: (s) => deps.unregisterSubagentSession(seamTask.id, s),
},
});
const runForCwd = (cwd: string): Promise<ReviewResult> => {
const invoke = () => invokeReviewerForCwd(cwd);
return sem ? sem.runNested(invoke) : invoke();
};
/*
FNXC:RepositoryScope 2026-08-20-23:40:
Plan Review is one task-document session even in a workspace. Code Review alone aggregates
modified scoped repositories. This prevents a clean acquired checkout from producing either
an extra plan session or an unavailable verdict before implementation starts.
*/
const invokeReviewer = () =>
workspaceConfig && config.type === "code" && detail.repositoryScope?.state !== "confirmed"
? Promise.resolve({
verdict: "UNAVAILABLE" as const,
retryable: false,
review: "Workspace Code Review requires a confirmed repository scope.",
summary: "Unavailable: repository scope is not confirmed",
})
: workspaceConfig && config.type === "code"
? deps.reviewWorkspacePerRepo(detail, (cwd: string) => runForCwd(cwd), {
workspaceRepos: workspaceConfig.repos,
workspaceRootDir: deps.rootDir,
settings,
/*
FNXC:Workspace 2026-08-15-04:49:
fn_task_done persists an accepted no-op sentinel as noCommitsExpected
before scheduling this review handoff. Carry that durable provenance
into the shared classifier so workspace review agrees with completion
even when the task acquired no sub-repo worktree.
*/
noOpCompletion: detail.noCommitsExpected === true,
noOpCompletionReason: "verified no-op completion persisted by fn_task_done",
})
: workspaceConfig && !reviewCwd
? Promise.resolve({ verdict: "UNAVAILABLE" as const, retryable: false, review: "Workspace Plan Review requires a confirmed scoped repository checkout.", summary: "Unavailable: no scoped repository checkout" })
: runForCwd(reviewCwd);
let review: ReviewResult;
try {
review = await invokeReviewer();
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
reviewerLog.error(`${seamTask.id}: step-review failed: ${message}`);
const narration = buildReviewUnavailableMessage(err);
void emitProactiveStatus(deps.store, seamTask.id, narration, "reviewer", sanitizeFailureReason(err));
return { verdict: "UNAVAILABLE", review: `reviewer error: ${message}` };
}
/*
FNXC:RepositoryScope 2026-08-21-02:35:
A workspace Code Review callback belongs to the scope generation used to capture its
per-repository diff evidence. Check that generation under the task lock before persisting
approval or advancing the graph: an operator scope change supersedes the whole callback.
*/
const reviewSuperseded = workspaceConfig && config.type === "code"
? await persistWorkspaceCodeReviewApproval(deps.store, seamTask.id, review)
: false;
if (reviewSuperseded) {
review = {
verdict: "UNAVAILABLE",
retryable: false,
review: "Workspace Code Review result superseded by a repository scope change.",
summary: "Unavailable: repository scope changed during review",
repositoryReviewOutcomes: review.repositoryReviewOutcomes,
repositoryScopeRevision: review.repositoryScopeRevision,
};
}
await deps.store.logEntry(
seamTask.id,
`${config.type} step-review Step ${stepIndex}: ${review.verdict}${config.advisory ? " (advisory)" : ""}`,
review.summary,
);
const narration = config.type === "plan" && review.verdict === "APPROVE"
? buildPlanVerifiedMessage()
: review.verdict === "UNAVAILABLE"
? buildReviewUnavailableMessage(review.summary)
: buildReviewVerdictMessage(review.verdict, review.summary);
if (review.verdict === "RETHINK") {
// RETHINK's rollback claim is emitted by applyGraphRethinkReset only after reset succeeds.
deps.graphRethinkNarrations.set(graphActiveContextKey(seamTask.id, active.instanceId), review.summary);
} else {
void emitProactiveStatus(deps.store, seamTask.id, narration, "reviewer", narration ? sanitizeFailureReason(review.summary) : undefined);
}
// Single-writer rule (KTD-4): advisory (split-branch) reviews never write
// the projection — they are fan-out checks that cannot clobber the
// authoritative verdict. Only an on-path APPROVE marks the step done.
if (review.verdict === "APPROVE" && !config.advisory) {
try {
const cur = await deps.store.getTask(seamTask.id);
/*
FNXC:RepositoryScope 2026-08-21-02:48:
The step-inversion path has its own graph-advance projection. Re-read
the scope generation after evidence persistence and before marking the
step done, so an intervening scope mutation routes this callback as
unavailable instead of allowing its old APPROVE edge to advance.
*/
if (workspaceConfig && config.type === "code" && review.repositoryScopeRevision !== undefined
&& cur.repositoryScope?.revision !== review.repositoryScopeRevision) {
return {
verdict: "UNAVAILABLE",
retryable: false,
review: "Workspace Code Review result superseded by a repository scope change.",
summary: "Unavailable: repository scope changed during review before graph advancement",
repositoryReviewOutcomes: review.repositoryReviewOutcomes,
repositoryScopeRevision: review.repositoryScopeRevision,
};
}
const status = cur.steps[stepIndex]?.status;
if (stepIndex >= 0 && stepIndex < cur.steps.length && status !== "done" && status !== "skipped") {
await deps.updateStepGraph(seamTask.id, stepIndex, "done");
await deps.store.logEntry(
seamTask.id,
`Step ${stepIndex} (${stepName}) marked done by step-review APPROVE (graph)`,
);
}
} catch (err) {
reviewerLog.warn(
`${seamTask.id}: failed to mark Step ${stepIndex} done after APPROVE: ${err instanceof Error ? err.message : String(err)}`,
);
}
}
return {
verdict: review.verdict,
review: review.review,
summary: review.summary,
retryable: review.retryable,
repositoryDiffFingerprints: review.repositoryDiffFingerprints,
repositoryModifiedFiles: review.repositoryModifiedFiles,
repositoryReviewOutcomes: review.repositoryReviewOutcomes,
repositoryScopeRevision: review.repositoryScopeRevision,
};
},
};
}