Support workspace tasks across a shared root with scope-driven repositories and verifiable merge gates. - Route task work through one workspace directory with per-repository acquisition and isolation. - Add sandbox session policies and per-repository verification command handling. - Enforce required pre-merge checks and honest merge blocking for workspace changes. - Update workspace, workflow, and sandbox documentation and release metadata. Files changed: .changeset/fn-158-workspace-single-root.md | 7 + docs/sandbox.md | 6 +- docs/workflow-steps.md | 6 +- docs/workspaces.md | 12 +- .../core/src/__tests__/legacy-adoption.test.ts | 15 +- .../src/__tests__/required-pre-merge-steps.test.ts | 25 ++++ .../core/src/__tests__/store-bypass-review.test.ts | 24 ++- packages/core/src/__tests__/task-merge.test.ts | 24 +++ .../core/src/__tests__/worktree-layout.test.ts | 29 ++++ packages/core/src/db/legacy-adoption.ts | 20 ++- packages/core/src/index.gate.ts | 4 + packages/core/src/index.ts | 4 + .../core/src/merge/required-pre-merge-steps.ts | 26 ++++ packages/core/src/merge/task-merge.ts | 35 ++++- packages/core/src/store.ts | 59 ++++++-- packages/core/src/task-store/lifecycle-ops.ts | 1 + packages/core/src/task-store/merge-queue-ops.ts | 5 +- packages/core/src/task-store/moves.ts | 13 +- packages/core/src/task-store/task-artifacts-ops.ts | 11 +- packages/core/src/tasks/worktree-layout.ts | 43 +++++- packages/core/src/types/workflow/workflow-steps.ts | 3 +- .../executor-workspace-session-cwd.test.ts | 42 ++++-- .../src/__tests__/node-worktree-isolation.test.ts | 13 +- .../src/__tests__/pi-create-fn-agent.test.ts | 16 ++ .../engine/src/__tests__/project-engine.test.ts | 20 ++- .../src/__tests__/reviewer-workspace.test.ts | 30 +++- .../src/__tests__/run-verification-command.test.ts | 90 +++++++++++- .../__tests__/sandbox/sandbox-exec-policy.test.ts | 16 +- .../src/__tests__/sandbox/session-policy.test.ts | 45 ++++++ .../__tests__/workspace-add-repo-midflight.test.ts | 9 ++ .../engine/src/__tests__/workspace-e2e.test.ts | 13 +- .../workspace-root-worktree-routing.test.ts | 18 +-- packages/engine/src/agent-tools.ts | 15 +- packages/engine/src/agents/agent-runtime.ts | 19 +++ .../engine/src/agents/agent-session-helpers.ts | 15 ++ packages/engine/src/execution/hold-release.ts | 29 ++++ .../engine/src/execution/run-verification-tool.ts | 114 ++++++++++++++- .../create-authoritative-workflow-seams.ts | 20 +-- packages/engine/src/executor/deps-bags.ts | 5 +- .../executor/ensure-graph-custom-node-worktree.ts | 24 ++- .../executor/ensure-task-worktree-for-planning.ts | 36 ++--- .../engine/src/executor/execute-workflow-step.ts | 4 +- .../src/executor/finalize-already-reviewed-task.ts | 6 +- .../src/executor/prepare-graph-node-execution.ts | 14 +- .../engine/src/executor/run-graph-custom-node.ts | 161 +++++++++++++-------- packages/engine/src/executor/run-implementation.ts | 117 ++++++++++++--- packages/engine/src/merge/merger-ai.ts | 16 +- packages/engine/src/merger.ts | 20 ++- packages/engine/src/pi.ts | 150 ++++++++++++++++--- packages/engine/src/project-engine.ts | 10 +- packages/engine/src/runtimes/in-process-runtime.ts | 4 +- packages/engine/src/sandbox/bubblewrap-backend.ts | 55 ++++++- packages/engine/src/sandbox/bubblewrap-policy.ts | 10 +- packages/engine/src/sandbox/index.ts | 1 + .../engine/src/sandbox/sandbox-exec-backend.ts | 45 +++++- packages/engine/src/sandbox/sandbox-exec-policy.ts | 18 ++- packages/engine/src/sandbox/session-policy.ts | 41 ++++++ packages/engine/src/sandbox/types.ts | 11 ++ packages/engine/src/self-healing.ts | 1 + packages/engine/src/triage.ts | 10 ++ .../engine/src/worktree/worktree-acquisition.ts | 53 ++++--- 61 files changed, 1393 insertions(+), 315 deletions(-) Fusion-Task-Id: FN-158 Fusion-Task-Lineage: ba57f5a2-fa69-4210-8ea7-3d124be3deb2 Co-authored-by: Fusion <noreply@runfusion.ai>
621 lines
35 KiB
TypeScript
621 lines
35 KiB
TypeScript
/**
|
|
* FNXC:CodeOrganization 2026-08-03-14:20:
|
|
* createAuthoritativeWorkflowSeams peeled from TaskExecutor (U4).
|
|
*
|
|
* FNXC:WorkflowExecutionOwnership 2026-07-27-16:25 / 2026-07-28-20:25:
|
|
* Seam return vocabulary is the ownership boundary; exit events announce without changing outcomes.
|
|
*/
|
|
import type { AgentStore, ResolvedTaskOutputLanguage, Settings, TaskStore, ThinkingLevel, WorkspaceConfig } from "@fusion/core";
|
|
import { emitWorkflowLifecycleEvent, resolveTaskOutputLanguage, THINKING_LEVELS } from "@fusion/core";
|
|
import type { ImplementationExit } from "./implementation-exit.js";
|
|
import type { WorkflowLegacySeams } from "../workflows/workflow-node-handlers.js";
|
|
import type { AgentSemaphore } from "../concurrency/concurrency.js";
|
|
import {
|
|
FOREACH_ACTIVE_CONTEXT_KEY,
|
|
SEAM_GOVERNING_NODE_CONTEXT_KEY,
|
|
SEAM_SKILL_NAME_CONTEXT_KEY,
|
|
SEAM_THINKING_LEVEL_CONTEXT_KEY,
|
|
|
|
type ForeachActiveContext,
|
|
} from "../workflows/workflow-node-handlers.js";
|
|
import { graphActiveContextKey } from "./task-predicates.js";
|
|
import { WorkflowReviewService } from "../workflows/workflow-review-service.js";
|
|
import { MERGE_BOUNDARY_UNPROVEN_VALUE } from "../workflows/workflow-merge-nodes.js";
|
|
import { mergeEffectiveSettings } from "../project/effective-settings.js";
|
|
import { resolveReviewCheckoutCwd } from "../execution/review-checkout.js";
|
|
import { logReviewCheckoutRouting } from "./review-checkout-routing.js";
|
|
import { selectUserCommentsForAgentContext } from "../agents/agent-user-comments.js";
|
|
import {
|
|
resolveValidatorThinkingLevel,
|
|
resolveValidatorFallbackThinkingLevel,
|
|
} from "../agents/agent-session-helpers.js";
|
|
import type { ReviewResult } from "../execution/reviewer.js";
|
|
import {
|
|
buildReviewUnavailableMessage,
|
|
buildPlanVerifiedMessage,
|
|
buildReviewVerdictMessage,
|
|
emitProactiveStatus,
|
|
sanitizeFailureReason,
|
|
} from "../project/proactive-status.js";
|
|
import type { EngineRunContext } from "../util/run-audit.js";
|
|
import { executorLog, reviewerLog } from "../logger.js";
|
|
import { normalizeWorkspaceTaskRouting } from "./workspace-config-resolver.js";
|
|
|
|
const WORKFLOW_THINKING_LEVEL_SET: ReadonlySet<string> = new Set(THINKING_LEVELS);
|
|
|
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any -- mirror TaskExecutor method surface
|
|
type AnyFn = (...args: any[]) => any;
|
|
|
|
/*
|
|
FNXC:WorkspaceReviewEvidence 2026-08-21-19:52:
|
|
Both authoritative and graph review producers must publish the exact approval shape landing reads.
|
|
Keep this fenced writer exported so the real producer-to-consumer regression cannot recreate it in a test.
|
|
*/
|
|
export async function persistWorkspaceCodeReviewApproval(
|
|
store: TaskStore,
|
|
taskId: string,
|
|
review: Pick<ReviewResult, "verdict" | "repositoryScopeRevision" | "repositoryDiffFingerprints" | "repositoryModifiedFiles">,
|
|
): Promise<boolean> {
|
|
if (review.repositoryScopeRevision === undefined) return false;
|
|
let superseded = false;
|
|
const approvedAt = new Date().toISOString();
|
|
await store.updateTaskAtomic(taskId, (current) => {
|
|
const scope = current.repositoryScope;
|
|
if (!scope || scope.revision !== review.repositoryScopeRevision) {
|
|
superseded = true;
|
|
return null;
|
|
}
|
|
if (review.verdict !== "APPROVE" || !review.repositoryDiffFingerprints || Object.keys(review.repositoryDiffFingerprints).length === 0) return null;
|
|
return {
|
|
repositoryScope: {
|
|
...scope,
|
|
reviewEvidence: Object.fromEntries(Object.entries(review.repositoryDiffFingerprints).map(([repo, fingerprint]) => [repo, { fingerprint, approvedAt }])),
|
|
...(scope.reviewRemediation?.scopeRevision === review.repositoryScopeRevision ? { reviewRemediation: undefined } : {}),
|
|
},
|
|
...(review.repositoryModifiedFiles ? { modifiedFiles: review.repositoryModifiedFiles } : {}),
|
|
};
|
|
});
|
|
return superseded;
|
|
}
|
|
|
|
export type CreateAuthoritativeWorkflowSeamsDeps = {
|
|
store: TaskStore;
|
|
rootDir: string;
|
|
options: {
|
|
agentStore?: AgentStore | null;
|
|
pluginRunner?: unknown;
|
|
semaphore?: AgentSemaphore;
|
|
mergeRequester?: unknown;
|
|
[k: string]: unknown;
|
|
};
|
|
workspaceConfig: WorkspaceConfig | null | undefined;
|
|
ensureWorkspaceConfig?: () => Promise<WorkspaceConfig | null>;
|
|
activeWorkflowPrincipals: Map<string, { agentId: string; nodeInstanceId: string; agent?: import("@fusion/core").Agent }>;
|
|
graphSeamGoverningNodeId: Map<string, string>;
|
|
graphSeamThinkingLevel: Map<string, ThinkingLevel>;
|
|
graphStepActiveContext: Map<string, unknown>;
|
|
graphRethinkNarrations: Map<string, unknown>;
|
|
pausedAborted: Set<string>;
|
|
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
|
mergeRequester?: ((taskId: string, opts?: any) => Promise<any>) | null;
|
|
getRunContextFor: (taskId: string) => EngineRunContext | undefined;
|
|
persistTokenUsage: AnyFn;
|
|
runImplementationPhase: AnyFn;
|
|
handoffTaskToReview: AnyFn;
|
|
ensureWorkflowMergeBoundaryTask: AnyFn;
|
|
getWorkflowMergeImplementationProofFailure: AnyFn;
|
|
runProjectedGraphTaskStep: AnyFn;
|
|
updateStepGraph: AnyFn;
|
|
reviewWorkspacePerRepo: AnyFn;
|
|
registerSubagentSession: AnyFn;
|
|
unregisterSubagentSession: AnyFn;
|
|
};
|
|
|
|
export function createAuthoritativeWorkflowSeams(
|
|
deps: CreateAuthoritativeWorkflowSeamsDeps,
|
|
settings: Settings,
|
|
outputLanguage?: ResolvedTaskOutputLanguage,
|
|
): WorkflowLegacySeams {
|
|
return {
|
|
// Built-in triage/spec generation runs upstream of the interpreter today,
|
|
// so planning is a no-op for already-specified tasks. Custom planning
|
|
// behavior is expressed as a custom prompt node before the execute seam.
|
|
planning: async () => ({ outcome: "success", value: "pre-specified" }),
|
|
execute: async (seamTask, context) => {
|
|
// Column-agent seam wiring (U4, R4): record the governing node id (the
|
|
// execute-seam prompt node, stamped into context by createPromptLikeHandler)
|
|
// so execute()'s session build can resolve the column-agent binding for the
|
|
// node's DECLARED column. Cleared after the pass so a later seam without a
|
|
// binding cannot inherit a stale node id.
|
|
const governingNodeId = context?.[SEAM_GOVERNING_NODE_CONTEXT_KEY];
|
|
if (typeof governingNodeId === "string") {
|
|
deps.graphSeamGoverningNodeId.set(seamTask.id, governingNodeId);
|
|
}
|
|
const seamThinkingLevel = context?.[SEAM_THINKING_LEVEL_CONTEXT_KEY];
|
|
if (typeof seamThinkingLevel === "string" && WORKFLOW_THINKING_LEVEL_SET.has(seamThinkingLevel)) {
|
|
deps.graphSeamThinkingLevel.set(seamTask.id, seamThinkingLevel as ThinkingLevel);
|
|
}
|
|
let result: { taskDone: boolean; modifiedFiles: string[]; exit?: ImplementationExit };
|
|
try {
|
|
result = await deps.runImplementationPhase(seamTask);
|
|
} finally {
|
|
deps.graphSeamGoverningNodeId.delete(seamTask.id);
|
|
deps.graphSeamThinkingLevel.delete(seamTask.id);
|
|
}
|
|
/*
|
|
FNXC:WorkflowExecutionOwnership 2026-07-27-16:25 (U8 / R4):
|
|
THIS BOOLEAN IS THE OWNERSHIP BOUNDARY, and it is too narrow. `runImplementation` has
|
|
28 measured ways of disposing of a task (16 column moves, 3 review handoffs, 9 terminal
|
|
parks — counted by `executor-lifecycle-ownership-ledger.test.ts`) and exactly 3 ways of
|
|
telling the graph anything, all of which collapse to `taskDone: true` here.
|
|
|
|
The consequence is not a missing feature, it is a second lifecycle owner. Because the
|
|
seam has no value for "the agent stopped because a step is blocked on a pending review"
|
|
or "the session was paused after the work was already complete", the implementation
|
|
phase performs those transitions ITSELF (`executor-exit-while-review-pending`,
|
|
`paused-after-completion`) and the graph learns about them afterwards — which is why
|
|
`handleGraphFailure` carries `alreadyFinalizedToReview` / `completionFinalized`
|
|
classifiers whose whole job is to recognise a move the graph did not make.
|
|
|
|
U8's direction: widen this vocabulary so a disposition is REPORTED here and the graph
|
|
routes it, rather than performed upstream and compensated for downstream. The
|
|
compensating classifiers are the acceptance test — they become unreachable, and then
|
|
deletable, exactly when the last out-of-band transition is gone.
|
|
*/
|
|
/*
|
|
FNXC:WorkflowExecutionOwnership 2026-07-28-20:25 (U8 / R4, R5):
|
|
Announce the exit on the U3 lifecycle bus. Until this, the two out-of-band review
|
|
handoffs left NO trace anywhere that the executor — not the graph — moved the card;
|
|
they surfaced as an ordinary `implementation-incomplete` failure that
|
|
`handleGraphFailure` then quietly compensated for. An operator could not tell the two
|
|
apart, and neither could a test.
|
|
|
|
Emission is deliberately AFTER the phase and BEFORE the return, and it changes nothing:
|
|
the outcome/value below are byte-identical to what this seam returned before, for every
|
|
exit, which `executor-implementation-exit-events.test.ts` pins by driving each exit and
|
|
asserting the seam's return. Per R5 an exit id is a REACTION — dropping every subscriber
|
|
must change no execution outcome, and that is asserted too.
|
|
*/
|
|
emitWorkflowLifecycleEvent({
|
|
type: "NodeCompleted",
|
|
taskId: seamTask.id,
|
|
at: new Date().toISOString(),
|
|
runId: deps.getRunContextFor(seamTask.id)?.runId,
|
|
nodeId: typeof governingNodeId === "string" ? governingNodeId : "execute",
|
|
outcome: result.taskDone ? "success" : "failure",
|
|
...(result.exit ? { exit: result.exit } : {}),
|
|
});
|
|
if (result.taskDone) {
|
|
return { outcome: "success", value: "implemented" };
|
|
}
|
|
// Distinguish pause/abort from genuine implementation failure so the
|
|
// failure handler can leave paused tasks to the pause machinery.
|
|
let paused = deps.pausedAborted.has(seamTask.id);
|
|
if (!paused) {
|
|
try {
|
|
paused = Boolean((await deps.store.getTask(seamTask.id)).paused);
|
|
} catch {
|
|
// Best-effort pause probe; fall through to the failure value.
|
|
}
|
|
}
|
|
return {
|
|
outcome: "failure",
|
|
value: paused ? "implementation-paused" : "implementation-incomplete",
|
|
};
|
|
},
|
|
// FNXC:WorkflowExecution 2026-06-25-00:00: U4 (KTD-2) — the legacy
|
|
// `workflowStep` seam was removed. Workflow quality gates run as the graph's
|
|
// own optional-group / gate nodes (builtin:coding replaced its `workflow-step`
|
|
// seam node with optional-group nodes) which record into
|
|
// `task.workflowStepResults` (U2). `WorkflowLegacySeams.workflowStep` no
|
|
// longer exists, and `resolveSeamName` no longer recognizes the
|
|
// `workflow-step` seam (an IR node still declaring it now fails loudly via
|
|
// WorkflowIrError rather than silently no-opping).
|
|
review: async (seamTask) => {
|
|
// The legacy "review" stage is the in-review handoff: the in-review column is
|
|
// the staging state the merge queue consumes.
|
|
const live = await deps.store.getTask(seamTask.id);
|
|
await deps.persistTokenUsage(seamTask.id);
|
|
/*
|
|
FNXC:TaskOutputLanguage 2026-08-19-16:25:
|
|
Legacy graph seams receive the invocation's resolved target rather than re-detecting a
|
|
mutable live description. Direct seam callers retain the compatibility fallback.
|
|
*/
|
|
await deps.handoffTaskToReview(live, "workflow-graph-review", undefined, outputLanguage ?? resolveTaskOutputLanguage(settings, live.description));
|
|
return { outcome: "success", value: "in-review" };
|
|
},
|
|
"review-handoff": async (seamTask) => {
|
|
/*
|
|
* FNXC:WorkflowPrPolicy 2026-06-29-16:42:
|
|
* Compound Engineering can run an optional manual PR review lane after implementation. That lane must start from the review column without invoking the generic reviewer again; this seam is a pure lifecycle handoff so PR creation/feedback nodes run while the card is visibly in review.
|
|
*/
|
|
const live = await deps.store.getTask(seamTask.id);
|
|
await deps.persistTokenUsage(seamTask.id);
|
|
/* FNXC:TaskOutputLanguage 2026-08-19-16:25: Manual handoff uses the same invocation snapshot as the review seam. */
|
|
await deps.handoffTaskToReview(live, "workflow-graph-review-handoff", undefined, outputLanguage ?? resolveTaskOutputLanguage(settings, live.description));
|
|
return { outcome: "success", value: "in-review" };
|
|
},
|
|
merge: async (seamTask, _context, signal) => {
|
|
if (!deps.mergeRequester) {
|
|
return { outcome: "failure", value: "merge-unavailable" };
|
|
}
|
|
// FNXC:WorkflowCancellation 2026-07-15-10:42: fail fast before the boundary-task mutation and the merge request — an abandoned walk must not enqueue a merge. Mirrors the `requestMerge` primitive.
|
|
if (signal?.aborted) {
|
|
return { outcome: "failure", value: "merge-cancelled" };
|
|
}
|
|
const mergeBoundary = await deps.ensureWorkflowMergeBoundaryTask(seamTask, {
|
|
reason: "workflow-merge-boundary",
|
|
nodeId: "legacy-merge-seam",
|
|
workflowId: "legacy-seams",
|
|
runId: deps.getRunContextFor(seamTask.id)?.runId ?? "legacy-seam",
|
|
});
|
|
/* FNXC:WorkflowMerge 2026-08-20-00:50: FN-9157 keeps the legacy seam terminal too; a retry cannot create missing boundary proof. */
|
|
if (mergeBoundary.blocked) return { outcome: "failure", value: MERGE_BOUNDARY_UNPROVEN_VALUE };
|
|
const mergeTask = mergeBoundary.task;
|
|
const missingImplementationProof = await deps.getWorkflowMergeImplementationProofFailure(mergeTask);
|
|
if (missingImplementationProof) {
|
|
await deps.store.logEntry(
|
|
mergeTask.id,
|
|
`Workflow merge blocked before requester: ${missingImplementationProof}`,
|
|
undefined,
|
|
deps.getRunContextFor(mergeTask.id),
|
|
);
|
|
return { outcome: "failure", value: "implementation-incomplete" };
|
|
}
|
|
// Bound the wait: a wedged merge queue must not strand the graph walk
|
|
// holding the routing claim. On timeout the run fails cleanly and the
|
|
// task is parked for human review; the queue can still finish later.
|
|
// FNXC:WorkflowCancellation 2026-07-15-10:42: the timeout is the wedged-queue bound, `signal` is the cancellation path — both must stay live. See the `requestMerge` primitive for the stall this prevents.
|
|
const GRAPH_MERGE_TIMEOUT_MS = 30 * 60 * 1000;
|
|
let timeoutHandle: ReturnType<typeof setTimeout> | undefined;
|
|
const timeout = new Promise<"timeout">((resolve) => {
|
|
timeoutHandle = setTimeout(() => resolve("timeout"), GRAPH_MERGE_TIMEOUT_MS);
|
|
timeoutHandle.unref?.();
|
|
});
|
|
let onGraphAbort: (() => void) | undefined;
|
|
const cancelled = new Promise<"cancelled">((resolve) => {
|
|
if (!signal) return;
|
|
onGraphAbort = () => resolve("cancelled");
|
|
signal.addEventListener("abort", onGraphAbort, { once: true });
|
|
});
|
|
try {
|
|
const result = await Promise.race([deps.mergeRequester(mergeTask.id, signal ? { signal } : undefined), timeout, cancelled]);
|
|
if (result === "cancelled") {
|
|
executorLog.warn(`${mergeTask.id}: graph merge seam cancelled by graph abort`);
|
|
return { outcome: "failure", value: "merge-cancelled" };
|
|
}
|
|
if (result === "timeout") {
|
|
executorLog.warn(`${mergeTask.id}: graph merge seam timed out after ${GRAPH_MERGE_TIMEOUT_MS}ms`);
|
|
return { outcome: "failure", value: "merge-timeout" };
|
|
}
|
|
if (result.merged || result.noOp) {
|
|
return { outcome: "success", value: result.noOp ? "merge-noop" : "merged" };
|
|
}
|
|
return { outcome: "failure", value: result.reason ?? result.error ?? "merge-failed" };
|
|
} finally {
|
|
if (timeoutHandle) clearTimeout(timeoutHandle);
|
|
if (onGraphAbort) signal?.removeEventListener("abort", onGraphAbort);
|
|
}
|
|
},
|
|
schedule: async () => ({ outcome: "success" }),
|
|
// Step-inversion (KTD-2/KTD-4, U3): run exactly the foreach-active step.
|
|
// The foreach sub-walk has set `foreach:active` with the step index; here
|
|
// we drive runTaskStep (step-runner.ts) over the task's worktree, then
|
|
// capture the per-step baselineSha/checkpointId back INTO the active
|
|
// context object so a later RETHINK (U5) can reset the step. The full
|
|
// single-step session physics (a StepSessionExecutor scoped to one step)
|
|
// is U5/U7 territory; U3 wires the seam and the context capture, using the
|
|
// existing implementation phase as the single-pass step driver.
|
|
stepExecute: async (seamTask, context) => {
|
|
const active = context[FOREACH_ACTIVE_CONTEXT_KEY] as ForeachActiveContext | undefined;
|
|
if (!active || typeof active.stepIndex !== "number") {
|
|
return { outcome: "failure", value: "no-active-step-instance" };
|
|
}
|
|
const live = await deps.store.getTask(seamTask.id);
|
|
// Worktree isolation (KTD-11, U10): run the instance's session in ITS OWN
|
|
// worktree when the foreach allocated one; otherwise the task's main
|
|
// worktree (shared isolation — unchanged). The file-scope guard the session
|
|
// machinery installs applies to either worktree unchanged (not bypassed).
|
|
// Stamp the active instance so `runGraphTaskStep` can honor
|
|
// `deferDoneToReview` when judging a non-terminal step (FIX 3).
|
|
deps.graphStepActiveContext.set(graphActiveContextKey(seamTask.id, active.instanceId), active);
|
|
// Column-agent seam wiring (U4, R4): the governing node id — the foreach
|
|
// INSTANCE node id (`<foreachId>#<i>:<templateNodeId>`) stamped into
|
|
// context by createPromptLikeHandler — threads INTO runGraphTaskStep,
|
|
// which stamps the per-task slot only when it CREATES the memoized
|
|
// implementation pass and clears it when that pass settles (PR #1432
|
|
// review). One step-session pass serves every instance, so the
|
|
// session-identity binding is deterministically the pass-initiating
|
|
// instance's; per-invocation set/delete here would race under parallel
|
|
// foreach (overwrite mid-build, or clear while the shared pass is live).
|
|
const stepGoverningNodeId = context[SEAM_GOVERNING_NODE_CONTEXT_KEY];
|
|
const seamThinkingLevel = context[SEAM_THINKING_LEVEL_CONTEXT_KEY];
|
|
const seamSkillName = context[SEAM_SKILL_NAME_CONTEXT_KEY];
|
|
const result = await deps.runProjectedGraphTaskStep(
|
|
seamTask,
|
|
live,
|
|
active.stepIndex,
|
|
active,
|
|
typeof stepGoverningNodeId === "string" ? stepGoverningNodeId : undefined,
|
|
typeof seamThinkingLevel === "string" && WORKFLOW_THINKING_LEVEL_SET.has(seamThinkingLevel)
|
|
? (seamThinkingLevel as ThinkingLevel)
|
|
: undefined,
|
|
typeof seamSkillName === "string" && seamSkillName.trim() ? seamSkillName.trim() : undefined,
|
|
);
|
|
// Capture baseline/checkpoint back into the reserved active context so the
|
|
// foreach sub-walk threads them to later template nodes (step-review/reset).
|
|
active.baselineSha = result.baselineSha;
|
|
active.checkpointId = result.checkpointId;
|
|
/*
|
|
FNXC:WorkflowExecutionOwnership 2026-07-29-11:30 (U8 / R4):
|
|
`step-done` / `step-failed` was a two-value flattening of every possible ending, and it
|
|
is why the pending-review ending could never reach an edge on the stepwise shape. A
|
|
blocked-on-pending-review pass is a WAIT, not a step defect: the outcome stays `failure`
|
|
(the step genuinely did not complete) while the VALUE names the ending, which is what the
|
|
foreach propagates upward — `runForeach` returns a failing instance's value as its own —
|
|
so the `steps` node can carry an `outcome:review-pending` edge to the park node.
|
|
Every other ending keeps `step-failed` exactly as before.
|
|
*/
|
|
const failureValue = result.exit === "review-handoff-pending-review" ? "review-pending" : "step-failed";
|
|
return {
|
|
outcome: result.outcome,
|
|
value: result.outcome === "success" ? "step-done" : failureValue,
|
|
contextPatch: {
|
|
[FOREACH_ACTIVE_CONTEXT_KEY]: active,
|
|
},
|
|
};
|
|
},
|
|
// Step-inversion (KTD-4, U5): review the foreach-active step. Mirrors the
|
|
// legacy in-session review call (deleted in U10): run
|
|
// reviewStep under semaphore.runNested against the instance's step number/
|
|
// name and the task's PROMPT content. On an authoritative (non-advisory)
|
|
// APPROVE, mark the step done through the projection (updateStep, KTD-7) —
|
|
// the step-execute seam left it in-progress (markDoneOnSuccess:false) so the
|
|
// review is the single done authority. The handler maps the returned verdict
|
|
// to outcome edges and applies the UNAVAILABLE bounded-retry limiter.
|
|
stepReview: async (seamTask, context, config) => {
|
|
const active = context[FOREACH_ACTIVE_CONTEXT_KEY] as ForeachActiveContext | undefined;
|
|
if (!active || typeof active.stepIndex !== "number") {
|
|
// No active instance — surface UNAVAILABLE so the handler routes it
|
|
// rather than fabricating an authoritative verdict.
|
|
return { verdict: "UNAVAILABLE", review: "no active step instance" };
|
|
}
|
|
const stepIndex = active.stepIndex;
|
|
let detail = await deps.store.getTask(seamTask.id);
|
|
const workspaceConfig = deps.ensureWorkspaceConfig
|
|
? await deps.ensureWorkspaceConfig()
|
|
: deps.workspaceConfig;
|
|
if (workspaceConfig) {
|
|
/*
|
|
FNXC:WorkspaceRootRouting 2026-08-19-12:15:
|
|
A code review must not interpret historical singular root metadata as its checkout. Repair
|
|
that metadata before selecting a cwd, then let reviewWorkspacePerRepo consume only the
|
|
durable declared-repository entries; an empty map remains a real unavailable review.
|
|
*/
|
|
detail = await normalizeWorkspaceTaskRouting(deps.store, seamTask.id) as typeof detail;
|
|
}
|
|
// Workspace Code Review fans out over durable repository entries. Plan Review
|
|
// reads the workspace under its declared read-only boundary; only single-repo
|
|
// review resolves a checkout here.
|
|
const worktreePath = active.worktreePath || detail.worktree || deps.rootDir;
|
|
const reviewCwd = workspaceConfig ? deps.rootDir : resolveReviewCheckoutCwd(detail, worktreePath);
|
|
if (reviewCwd) logReviewCheckoutRouting(seamTask.id, detail, reviewCwd, worktreePath);
|
|
const stepName = detail.steps[stepIndex]?.name ?? `Step ${stepIndex}`;
|
|
const promptContent = detail.prompt ?? "";
|
|
const planScopeContext = workspaceConfig && config.type === "plan"
|
|
? `\n\nRepository scope (task-level; review this plan once): ${detail.repositoryScope?.repositories.join(", ") || "unconfirmed"}.`
|
|
: "";
|
|
const userComments = selectUserCommentsForAgentContext(detail, { limit: null });
|
|
// Merge per-task effective workflow settings (U3, KTD-3) so the validator
|
|
// model-lane reads below pick up workflow values. Behavior-inert by default.
|
|
const settings = await mergeEffectiveSettings(deps.store, detail, await deps.store.getSettings());
|
|
|
|
/*
|
|
FNXC:AgentSteering 2026-06-30-12:37:
|
|
Workflow graph step-review nodes are optional or mandatory reviewer gates. Pass canonical user comments and legacy steering into each per-cwd reviewer so workspace aggregation never drops operator requirements.
|
|
|
|
FNXC:AgentSteering 2026-06-30-13:20:
|
|
Graph reviewer gates request uncapped comment context because every user-authored requirement can affect approval, including older steering retained on long-running tasks.
|
|
*/
|
|
const sem = deps.options.semaphore;
|
|
// FNXC:Workspace 2026-06-22-00:30: KTD3 — step-inversion review seam loops per sub-repo.
|
|
// `reviewStep` stays single-cwd; THIS CALLER loops. Single-cwd by default reviews
|
|
// `worktreePath`; in workspace mode that is the browse-only non-git root, so we instead spawn
|
|
// one reviewer per acquired sub-repo (cwd = repo.worktreePath) via reviewWorkspacePerRepo and
|
|
// aggregate as a conjunction. `invokeReviewerForCwd` is the per-cwd reviewStep call both modes share.
|
|
const reviewService = new WorkflowReviewService();
|
|
const invokeReviewerForCwd = (cwd: string) =>
|
|
reviewService.reviewStep({
|
|
cwd,
|
|
taskId: seamTask.id,
|
|
stepIndex,
|
|
stepName,
|
|
type: config.type,
|
|
promptContent: `${promptContent}${planScopeContext}`,
|
|
// Code reviews diff against the per-step baseline captured at
|
|
// step-execute; plan reviews pass no baseline (advisory).
|
|
baselineSha: config.type === "code" ? active.baselineSha : undefined,
|
|
options: {
|
|
defaultProvider: settings.defaultProvider,
|
|
defaultModelId: settings.defaultModelId,
|
|
fallbackProvider: settings.fallbackProvider,
|
|
fallbackModelId: settings.fallbackModelId,
|
|
/*
|
|
* FNXC:Settings-ThinkingLevel 2026-07-13-00:27:
|
|
* Step-review model sessions honor per-node `config.thinkingLevel` before the task validator override, then shared task thinking, validator workflow lane, global lane, and default thinking settings.
|
|
*/
|
|
defaultThinkingLevel: resolveValidatorThinkingLevel(
|
|
typeof config.thinkingLevel === "string" && WORKFLOW_THINKING_LEVEL_SET.has(config.thinkingLevel)
|
|
? (config.thinkingLevel as ThinkingLevel)
|
|
: detail.validatorThinkingLevel ?? detail.thinkingLevel,
|
|
settings,
|
|
),
|
|
fallbackThinkingLevel: resolveValidatorFallbackThinkingLevel(
|
|
typeof config.thinkingLevel === "string" && WORKFLOW_THINKING_LEVEL_SET.has(config.thinkingLevel)
|
|
? (config.thinkingLevel as ThinkingLevel)
|
|
: detail.validatorThinkingLevel ?? detail.thinkingLevel,
|
|
settings,
|
|
),
|
|
taskValidatorProvider: detail.validatorModelProvider,
|
|
taskValidatorModelId: detail.validatorModelId,
|
|
taskValidatorCredentialInstanceId: detail.validatorCredentialInstanceId,
|
|
projectValidatorProvider: settings.validatorProvider,
|
|
projectValidatorModelId: settings.validatorModelId,
|
|
projectValidatorFallbackProvider: settings.validatorFallbackProvider,
|
|
projectValidatorFallbackModelId: settings.validatorFallbackModelId,
|
|
globalValidatorProvider: settings.validatorGlobalProvider,
|
|
globalValidatorModelId: settings.validatorGlobalModelId,
|
|
projectDefaultOverrideProvider: settings.defaultProviderOverride,
|
|
projectDefaultOverrideModelId: settings.defaultModelIdOverride,
|
|
store: deps.store,
|
|
taskId: seamTask.id,
|
|
task: detail,
|
|
userComments: userComments.length > 0 ? userComments : undefined,
|
|
agentPrompts: settings.agentPrompts,
|
|
agentStore: deps.options.agentStore ?? undefined,
|
|
rootDir: deps.rootDir,
|
|
settings,
|
|
/* FNXC:WorkflowAgentRouting 2026-08-07-04:45: reviewer sessions inherit the exact graph-fenced principal, including a node-local override. */
|
|
agentId: deps.activeWorkflowPrincipals.get(seamTask.id)?.agentId,
|
|
onSessionCreated: (s) => deps.registerSubagentSession(seamTask.id, s),
|
|
onSessionEnded: (s) => deps.unregisterSubagentSession(seamTask.id, s),
|
|
},
|
|
});
|
|
const runForCwd = (cwd: string): Promise<ReviewResult> => {
|
|
const invoke = () => invokeReviewerForCwd(cwd);
|
|
return sem ? sem.runNested(invoke) : invoke();
|
|
};
|
|
/*
|
|
FNXC:RepositoryScope 2026-08-20-23:40:
|
|
Plan Review is one task-document session even in a workspace. Code Review alone aggregates
|
|
modified scoped repositories. This prevents a clean acquired checkout from producing either
|
|
an extra plan session or an unavailable verdict before implementation starts.
|
|
*/
|
|
const invokeReviewer = () =>
|
|
workspaceConfig && config.type === "code" && detail.repositoryScope?.state !== "confirmed"
|
|
? Promise.resolve({
|
|
verdict: "UNAVAILABLE" as const,
|
|
retryable: false,
|
|
review: "Workspace Code Review requires a confirmed repository scope.",
|
|
summary: "Unavailable: repository scope is not confirmed",
|
|
})
|
|
: workspaceConfig && config.type === "code"
|
|
? deps.reviewWorkspacePerRepo(detail, (cwd: string) => runForCwd(cwd), {
|
|
workspaceRepos: workspaceConfig.repos,
|
|
workspaceRootDir: deps.rootDir,
|
|
settings,
|
|
/*
|
|
FNXC:Workspace 2026-08-15-04:49:
|
|
fn_task_done persists an accepted no-op sentinel as noCommitsExpected
|
|
before scheduling this review handoff. Carry that durable provenance
|
|
into the shared classifier so workspace review agrees with completion
|
|
even when the task acquired no sub-repo worktree.
|
|
*/
|
|
noOpCompletion: detail.noCommitsExpected === true,
|
|
noOpCompletionReason: "verified no-op completion persisted by fn_task_done",
|
|
})
|
|
: workspaceConfig && !reviewCwd
|
|
? Promise.resolve({ verdict: "UNAVAILABLE" as const, retryable: false, review: "Workspace Plan Review requires a confirmed scoped repository checkout.", summary: "Unavailable: no scoped repository checkout" })
|
|
: runForCwd(reviewCwd);
|
|
|
|
let review: ReviewResult;
|
|
try {
|
|
review = await invokeReviewer();
|
|
} catch (err) {
|
|
const message = err instanceof Error ? err.message : String(err);
|
|
reviewerLog.error(`${seamTask.id}: step-review failed: ${message}`);
|
|
const narration = buildReviewUnavailableMessage(err);
|
|
void emitProactiveStatus(deps.store, seamTask.id, narration, "reviewer", sanitizeFailureReason(err));
|
|
return { verdict: "UNAVAILABLE", review: `reviewer error: ${message}` };
|
|
}
|
|
|
|
/*
|
|
FNXC:RepositoryScope 2026-08-21-02:35:
|
|
A workspace Code Review callback belongs to the scope generation used to capture its
|
|
per-repository diff evidence. Check that generation under the task lock before persisting
|
|
approval or advancing the graph: an operator scope change supersedes the whole callback.
|
|
*/
|
|
const reviewSuperseded = workspaceConfig && config.type === "code"
|
|
? await persistWorkspaceCodeReviewApproval(deps.store, seamTask.id, review)
|
|
: false;
|
|
if (reviewSuperseded) {
|
|
review = {
|
|
verdict: "UNAVAILABLE",
|
|
retryable: false,
|
|
review: "Workspace Code Review result superseded by a repository scope change.",
|
|
summary: "Unavailable: repository scope changed during review",
|
|
repositoryReviewOutcomes: review.repositoryReviewOutcomes,
|
|
repositoryScopeRevision: review.repositoryScopeRevision,
|
|
};
|
|
}
|
|
|
|
await deps.store.logEntry(
|
|
seamTask.id,
|
|
`${config.type} step-review Step ${stepIndex}: ${review.verdict}${config.advisory ? " (advisory)" : ""}`,
|
|
review.summary,
|
|
);
|
|
const narration = config.type === "plan" && review.verdict === "APPROVE"
|
|
? buildPlanVerifiedMessage()
|
|
: review.verdict === "UNAVAILABLE"
|
|
? buildReviewUnavailableMessage(review.summary)
|
|
: buildReviewVerdictMessage(review.verdict, review.summary);
|
|
if (review.verdict === "RETHINK") {
|
|
// RETHINK's rollback claim is emitted by applyGraphRethinkReset only after reset succeeds.
|
|
deps.graphRethinkNarrations.set(graphActiveContextKey(seamTask.id, active.instanceId), review.summary);
|
|
} else {
|
|
void emitProactiveStatus(deps.store, seamTask.id, narration, "reviewer", narration ? sanitizeFailureReason(review.summary) : undefined);
|
|
}
|
|
|
|
// Single-writer rule (KTD-4): advisory (split-branch) reviews never write
|
|
// the projection — they are fan-out checks that cannot clobber the
|
|
// authoritative verdict. Only an on-path APPROVE marks the step done.
|
|
if (review.verdict === "APPROVE" && !config.advisory) {
|
|
try {
|
|
const cur = await deps.store.getTask(seamTask.id);
|
|
/*
|
|
FNXC:RepositoryScope 2026-08-21-02:48:
|
|
The step-inversion path has its own graph-advance projection. Re-read
|
|
the scope generation after evidence persistence and before marking the
|
|
step done, so an intervening scope mutation routes this callback as
|
|
unavailable instead of allowing its old APPROVE edge to advance.
|
|
*/
|
|
if (workspaceConfig && config.type === "code" && review.repositoryScopeRevision !== undefined
|
|
&& cur.repositoryScope?.revision !== review.repositoryScopeRevision) {
|
|
return {
|
|
verdict: "UNAVAILABLE",
|
|
retryable: false,
|
|
review: "Workspace Code Review result superseded by a repository scope change.",
|
|
summary: "Unavailable: repository scope changed during review before graph advancement",
|
|
repositoryReviewOutcomes: review.repositoryReviewOutcomes,
|
|
repositoryScopeRevision: review.repositoryScopeRevision,
|
|
};
|
|
}
|
|
const status = cur.steps[stepIndex]?.status;
|
|
if (stepIndex >= 0 && stepIndex < cur.steps.length && status !== "done" && status !== "skipped") {
|
|
await deps.updateStepGraph(seamTask.id, stepIndex, "done");
|
|
await deps.store.logEntry(
|
|
seamTask.id,
|
|
`Step ${stepIndex} (${stepName}) marked done by step-review APPROVE (graph)`,
|
|
);
|
|
}
|
|
} catch (err) {
|
|
reviewerLog.warn(
|
|
`${seamTask.id}: failed to mark Step ${stepIndex} done after APPROVE: ${err instanceof Error ? err.message : String(err)}`,
|
|
);
|
|
}
|
|
}
|
|
|
|
return {
|
|
verdict: review.verdict,
|
|
review: review.review,
|
|
summary: review.summary,
|
|
retryable: review.retryable,
|
|
repositoryDiffFingerprints: review.repositoryDiffFingerprints,
|
|
repositoryModifiedFiles: review.repositoryModifiedFiles,
|
|
repositoryReviewOutcomes: review.repositoryReviewOutcomes,
|
|
repositoryScopeRevision: review.repositoryScopeRevision,
|
|
};
|
|
},
|
|
};
|
|
}
|