diff --git a/.changeset/workflow-principal-stale-fence-reroute.md b/.changeset/workflow-principal-stale-fence-reroute.md new file mode 100644 index 0000000000..6e11a4a528 --- /dev/null +++ b/.changeset/workflow-principal-stale-fence-reroute.md @@ -0,0 +1,7 @@ +--- +"@runfusion/fusion": patch +--- + +summary: Tasks assigned to an engineer agent now execute continuously instead of stalling until a heartbeat. +category: fix +dev: Workflow principal routing separates structural capability from availability. A named principal that can never satisfy a node (wrong role, deleted, authority edited away) is no longer authority for it — routing falls through to the column binding and role pool, and a resumed continuation discards the stale fence and re-routes. A role-capable principal that is only paused/disabled/at capacity still holds, unchanged. An explicitly assigned engineer-role agent is now valid `task-assignee` authority for an executor node; the role pool stays strict. diff --git a/packages/core/src/agents/agent-role-policy.ts b/packages/core/src/agents/agent-role-policy.ts index f92ca8d1ea..2eb30db602 100644 --- a/packages/core/src/agents/agent-role-policy.ts +++ b/packages/core/src/agents/agent-role-policy.ts @@ -141,6 +141,56 @@ export function isEngineerRoleAgent(agent: RoleTaggedAgent): boolean { return agentRoles(agent).includes("engineer"); } +/* +FNXC:WorkflowAgentRouting 2026-08-10-07:50: +STRUCTURAL capability for a workflow stage, kept separate from AVAILABILITY. The distinction decides +whether an unroutable named principal is a WAIT or a DEAD END, and conflating the two wedged the board: + +FN-8869/FN-8928/FN-8845 were each explicitly assigned to a permanent ENGINEER-role agent, which +`canAgentTakeImplementationTaskForExplicitRouting` allows by design. The workflow `step-execute` node then +took that owner as `task-assignee` named authority, found no `executor` tag, and held closed — and a NAMED +principal never falls through to the role pool. Two idle `Workflow Executor` pool agents sat unused while the +cards re-dispatched and re-held every ~15 minutes for hours. The only thing still touching them was the owner +agent's own hourly heartbeat, which logged "progressing, no blockers" and exited: heartbeat observation had +silently replaced execution. + +An agent that lacks the role can NEVER satisfy the node, so waiting on it is unbounded by construction. It was +never authority for this node in the first place — routing must skip it and continue precedence. Only an agent +that HAS the role but is momentarily unusable (paused/errored/disabled runtime/at session capacity) earns a +hold, because that hold ends on its own. +*/ +export interface WorkflowRoleCapabilityOptions { + /* + FNXC:WorkflowAgentRouting 2026-08-10-07:50: + Accept an ENGINEER-role owner as capable of an executor node. Set ONLY for named task-assignee authority, + never for the role pool. + + A durable engineer explicitly assigned to a task is already allowed to take implementation work + (`canAgentTakeImplementationTaskForExplicitRouting`), so the executor node it owns must run — and run + CONTINUOUSLY under graph dispatch. The alternative, which is what actually shipped, is that the owner's + hourly heartbeat becomes the only thing that ever touches the card: it wakes, logs "progressing, no + blockers", calls fn_heartbeat_done, and the work never advances. An agent holding a task executes it; a + heartbeat is a liveness tick, not a work loop. + + The POOL stays strict, because automatic backlog pickup by engineers is a separate opt-in + (`canAgentTakeImplementationTaskForBacklogPickup`) and unassigned work must not silently land on engineers. + */ + readonly allowEngineerAsExecutor?: boolean; +} + +export function hasWorkflowRoleCapability( + agent: RoleTaggedAgent, + role: AgentCapability, + options: WorkflowRoleCapabilityOptions = {}, +): boolean { + const roles = agentRoles(agent); + const tagged = roles.includes(role) + || (role === "executor" && options.allowEngineerAsExecutor === true && roles.includes("engineer")); + if (!tagged) return false; + // Assignment policy "none" is a hard floor on implementation work; such an agent can never run an executor node. + return role !== "executor" || canAgentReceiveImplementationTasks(agent); +} + export function canAgentTakeImplementationTaskForExplicitRouting( agent: AgentAssignmentPolicyInput, task: Pick, diff --git a/packages/core/src/index.ts b/packages/core/src/index.ts index 15bb940258..385524548a 100644 --- a/packages/core/src/index.ts +++ b/packages/core/src/index.ts @@ -797,12 +797,13 @@ export { isAgentAutoAssignable, canAgentReceiveImplementationTasks, isWorkflowPrincipalEligible, + hasWorkflowRoleCapability, isBuiltinWorkflowRoleAgent, evaluateImplementationTaskBind, assertImplementationTaskBindAllowed, AgentTaskRoutingPolicyError, } from "./agents/agent-role-policy.js"; -export type { AgentAssignmentPolicy, ImplementationTaskBindContext, ImplementationTaskBindVerdict } from "./agents/agent-role-policy.js"; +export type { AgentAssignmentPolicy, ImplementationTaskBindContext, ImplementationTaskBindVerdict, WorkflowRoleCapabilityOptions } from "./agents/agent-role-policy.js"; export { ReflectionStore } from "./agents/reflection-store.js"; export type { ReflectionStoreEvents } from "./agents/reflection-store.js"; export { MessageStore } from "./stores/message-store.js"; diff --git a/packages/engine/src/__tests__/workflow-agent-routing.test.ts b/packages/engine/src/__tests__/workflow-agent-routing.test.ts index 29a7766922..f2205dff3f 100644 --- a/packages/engine/src/__tests__/workflow-agent-routing.test.ts +++ b/packages/engine/src/__tests__/workflow-agent-routing.test.ts @@ -14,13 +14,74 @@ describe("routeWorkflowPrincipal", () => { expect(routeWorkflowPrincipal({ task: { assignedAgentId: "owner" }, ir, node: { id: "e", kind: "prompt", config: { seam: "execute" } }, agents: [owner, reviewer] })).toMatchObject({ status: "routed", route: { agent: owner, authority: "task-assignee" } }); }); - it("holds rather than falling back when a named owner is not eligible to execute", () => { + /* + FNXC:WorkflowAgentRouting 2026-08-10-07:50: + The invariant is STRUCTURAL-INCAPABILITY-FALLS-THROUGH / UNAVAILABILITY-HOLDS, and it is asserted across + every named authority because the wedge reached the board through only one of them. An owner that cannot + ever run the node was never authority for it; an owner that could but is momentarily unusable is a real + wait whose end is in sight. The previous behaviour — hold on both — stranded FN-8869/FN-8928/FN-8845 for + hours apiece behind two idle pool executors. + */ + it("falls through to the pool when a named principal can NEVER satisfy the node", () => { const pool = agent("pool", ["executor"]); - const node = { id: "e", kind: "prompt", config: { seam: "execute" } }; + const reviewerPool = agent("reviewer-pool", ["reviewer"]); + const executeNode = { id: "e", kind: "prompt", config: { seam: "execute" } } as any; + const reviewNode = { id: "r", kind: "prompt", config: { workflowRole: "reviewer" } } as any; + + // Wrong role for the node, assignment policy hard-floored, and an owner id that resolves to nothing. for (const owner of [ agent("triage-owner", ["triage"]), - { ...agent("disabled-owner", ["executor"]), runtimeConfig: { enabled: false } }, { ...agent("policy-denied-owner", ["executor"]), runtimeConfig: { assignmentPolicy: "none" } }, + ]) { + expect(routeWorkflowPrincipal({ task: { assignedAgentId: owner.id }, ir, node: executeNode, agents: [owner, pool] })) + .toMatchObject({ status: "routed", route: { agent: pool, authority: "role-pool" } }); + } + expect(routeWorkflowPrincipal({ task: { assignedAgentId: "deleted-owner" }, ir, node: executeNode, agents: [pool] })) + .toMatchObject({ status: "routed", route: { agent: pool, authority: "role-pool" } }); + + // Same rule for a review override and a column binding — a dead name is not a wait on any surface. + expect(routeWorkflowPrincipal({ + task: {}, ir, node: { ...reviewNode, reviewerAgentId: "triage-owner" }, agents: [agent("triage-owner", ["triage"]), reviewerPool], + })).toMatchObject({ status: "routed", route: { agent: reviewerPool, authority: "role-pool" } }); + const boundNode = { ...executeNode, column: "todo" }; + const boundIr = { + ...ir, + columns: [{ id: "todo", name: "Todo", traits: [], agent: { agentId: "triage-owner", mode: "override" } }], + nodes: [boundNode], + } as any; + expect(routeWorkflowPrincipal({ + task: {}, ir: boundIr, node: boundNode, agents: [agent("triage-owner", ["triage"]), pool], + })).toMatchObject({ status: "routed", route: { agent: pool, authority: "role-pool" } }); + }); + + /* + FNXC:WorkflowAgentRouting 2026-08-10-07:50: + A durable ENGINEER explicitly assigned to a task executes it, continuously, through graph dispatch. + `canAgentTakeImplementationTaskForExplicitRouting` already permits that assignment, so refusing it at the + executor node left the owner's hourly heartbeat as the only thing touching the card — it logged + "progressing, no blockers" and exited while nothing progressed. The POOL stays strict: an unassigned + executor node must not silently land on an engineer, since backlog pickup is a separate opt-in. + */ + it("routes an assigned engineer owner to its own executor node, but never through the pool", () => { + const engineer = agent("backend-engineer", ["engineer"]); + const node = { id: "e", kind: "prompt", config: { seam: "execute" } } as any; + expect(routeWorkflowPrincipal({ task: { assignedAgentId: "backend-engineer" }, ir, node, agents: [engineer] })) + .toMatchObject({ status: "routed", route: { agent: engineer, authority: "task-assignee" } }); + expect(routeWorkflowPrincipal({ task: {}, ir, node, agents: [engineer] })) + .toEqual({ status: "held", role: "executor", reason: "role-pool-exhausted" }); + // The hard floor still wins over ownership. + const denied = { ...agent("liaison", ["engineer"]), runtimeConfig: { assignmentPolicy: "none" } }; + expect(routeWorkflowPrincipal({ task: { assignedAgentId: "liaison" }, ir, node, agents: [denied] })) + .toEqual({ status: "held", role: "executor", reason: "role-pool-exhausted" }); + }); + + it("still holds a role-capable owner that is only momentarily unusable", () => { + const pool = agent("pool", ["executor"]); + const node = { id: "e", kind: "prompt", config: { seam: "execute" } } as any; + for (const owner of [ + { ...agent("disabled-owner", ["executor"]), runtimeConfig: { enabled: false } }, + { ...agent("paused-owner", ["executor"]), state: "paused" }, + { ...agent("errored-owner", ["executor"]), state: "error" }, ]) { expect(routeWorkflowPrincipal({ task: { assignedAgentId: owner.id }, ir, node, agents: [owner, pool] })) .toEqual({ status: "held", role: "executor", reason: "named-principal-unavailable" }); @@ -51,7 +112,53 @@ describe("routeWorkflowPrincipal", () => { })).toMatchObject({ status: "routed", route: { agent: reviewer, authority: "review-node-override" } }); expect(validateFencedWorkflowPrincipal({ task: {}, node: { ...node, reviewerAgentId: "other" }, principalAgentId: "reviewer", role: "reviewer", authority: "review-node-override", agents: [reviewer], - })).toEqual({ status: "held", role: "reviewer", reason: "named-principal-unavailable" }); + })).toEqual({ status: "held", role: "reviewer", reason: "named-principal-unavailable", staleFence: true }); + }); + + /* + FNXC:WorkflowAgentRouting 2026-08-10-07:50: + `staleFence` is the caller's signal to re-route instead of re-asserting a dead principal every dispatch. + It must be set on EVERY branch that proves the fence no longer describes reality — node reclassified, + ownership moved, override edited, binding redirected, agent deleted or stripped of the role — and must NOT + be set when the fenced principal is still the right one and merely unavailable. + */ + it("marks a fence stale on every branch that proves it no longer describes reality", () => { + const executeNode = { id: "e", kind: "prompt", config: { seam: "execute" } } as any; + const owner = agent("owner", ["executor"]); + const base = { task: { assignedAgentId: "owner" }, ir, node: executeNode, role: "executor", authority: "task-assignee", agents: [owner] } as any; + + // Ownership moved away from the fenced principal. + expect(validateFencedWorkflowPrincipal({ ...base, task: { assignedAgentId: "someone-else" }, principalAgentId: "owner" })) + .toMatchObject({ status: "held", staleFence: true }); + // The node is no longer the role the fence claims. + expect(validateFencedWorkflowPrincipal({ ...base, node: { id: "r", kind: "prompt", config: { workflowRole: "reviewer" } }, principalAgentId: "owner" })) + .toMatchObject({ status: "held", staleFence: true }); + // The fenced agent is gone, or lost the role. + expect(validateFencedWorkflowPrincipal({ ...base, agents: [], principalAgentId: "owner" })) + .toMatchObject({ status: "held", staleFence: true }); + expect(validateFencedWorkflowPrincipal({ ...base, agents: [agent("owner", ["reviewer"])], principalAgentId: "owner" })) + .toMatchObject({ status: "held", staleFence: true }); + // The column binding was redirected. + expect(validateFencedWorkflowPrincipal({ + ...base, + task: {}, + authority: "column-binding", + node: { ...executeNode, column: "todo" }, + ir: { ...ir, columns: [{ id: "todo", name: "Todo", traits: [], agent: { agentId: "other", mode: "override" } }] }, + principalAgentId: "owner", + })).toMatchObject({ status: "held", staleFence: true }); + + // Still the right principal, merely unusable right now — a real wait, never a re-route. + expect(validateFencedWorkflowPrincipal({ ...base, agents: [{ ...owner, state: "paused" }], principalAgentId: "owner" })) + .toEqual({ status: "held", role: "executor", reason: "named-principal-unavailable" }); + }); + + it("keeps a fenced engineer owner on its own executor node", () => { + const engineer = agent("backend-engineer", ["engineer"]); + expect(validateFencedWorkflowPrincipal({ + task: { assignedAgentId: "backend-engineer" }, ir, node: { id: "e", kind: "prompt", config: { seam: "execute" } } as any, + principalAgentId: "backend-engineer", role: "executor", authority: "task-assignee", agents: [engineer], + })).toMatchObject({ status: "routed", route: { agent: engineer, authority: "task-assignee" } }); }); it("revokes a review authority when its exact durable override changes", () => { @@ -110,7 +217,7 @@ describe("routeWorkflowPrincipal", () => { expect(validateFencedWorkflowPrincipal({ task: {}, ir: editedIr, node: nested, nodeInstanceId: "foreach#0:review", principalAgentId: "reviewer", role: "reviewer", authority: "review-node-override", agents: [reviewer], - })).toEqual({ status: "held", role: "reviewer", reason: "named-principal-unavailable" }); + })).toEqual({ status: "held", role: "reviewer", reason: "named-principal-unavailable", staleFence: true }); }); it("moves a raced role-pool route to the next deterministic candidate", () => { @@ -123,12 +230,12 @@ describe("routeWorkflowPrincipal", () => { .toMatchObject({ status: "routed", route: { agent: second, authority: "role-pool" } }); }); - it("fails a task-assignee fence after assignment changes instead of rerouting", () => { + it("fails a task-assignee fence after assignment changes, and marks it stale so the caller re-routes", () => { const owner = agent("owner", ["custom"]); expect(validateFencedWorkflowPrincipal({ task: { assignedAgentId: "other" }, node: { id: "execute", kind: "prompt", config: { seam: "execute" } } as any, principalAgentId: "owner", role: "executor", authority: "task-assignee", agents: [owner], - })).toEqual({ status: "held", role: "executor", reason: "named-principal-unavailable" }); + })).toEqual({ status: "held", role: "executor", reason: "named-principal-unavailable", staleFence: true }); }); it("fails a column fence after the durable binding is redirected", () => { @@ -145,6 +252,6 @@ describe("routeWorkflowPrincipal", () => { expect(validateFencedWorkflowPrincipal({ task: {}, ir: { ...boundIr, columns: [{ ...boundIr.columns[0], agent: { agentId: "replacement", mode: "override" } }] }, node, principalAgentId: "bound", role: "executor", authority: "column-binding", agents: [bound], - })).toEqual({ status: "held", role: "executor", reason: "named-principal-unavailable" }); + })).toEqual({ status: "held", role: "executor", reason: "named-principal-unavailable", staleFence: true }); }); }); diff --git a/packages/engine/src/__tests__/workflow-principal-stale-fence-reroute.test.ts b/packages/engine/src/__tests__/workflow-principal-stale-fence-reroute.test.ts new file mode 100644 index 0000000000..9a8193beef --- /dev/null +++ b/packages/engine/src/__tests__/workflow-principal-stale-fence-reroute.test.ts @@ -0,0 +1,121 @@ +import { describe, expect, it } from "vitest"; +import { admitWorkflowPrincipalBeforeNode } from "../executor/workflow-principal-before-node.js"; + +/* +FNXC:WorkflowAgentRouting 2026-08-10-07:50: +The dispatch surface, asserted separately from the pure router because this is where the wedge was VISIBLE. + +FN-8869 was assigned to a permanent engineer-role agent, fenced its `step-execute` continuation to that owner, +and then held `named-principal-unavailable:executor` on every dispatch for hours — each self-healing bounce +re-read the same dead fence and re-asserted it, while two idle `Workflow Executor` pool agents were never +consulted. A router-only test cannot catch that: the fence path never calls the router at all. + +Invariant: a node admission NEVER terminates in a hold that re-dispatch cannot clear. Either the fenced +principal still runs the node, or the fence is discarded and routing continues. +*/ + +const agent = (id: string, roles: string[]) => ({ + id, name: id, roles, role: roles[0], state: "idle", + createdAt: "2026-01-01T00:00:00.000Z", updatedAt: "2026-01-01T00:00:00.000Z", metadata: {}, +}) as any; + +const executeNode = { id: "step-execute", kind: "prompt", config: { seam: "execute" } } as any; + +function deps(agents: any[]) { + const written: any[] = []; + return { + written, + deps: { + store: { + getRootDir: () => "/repo", + logEntry: async () => undefined, + replaceActiveTaskWorkflowContinuation: async (input: any) => { + written.push(input); + return { id: `wi-${written.length}` }; + }, + }, + options: { agentStore: { listAgents: async () => agents, workflowProjectId: "proj" } }, + workflowAgentCapacity: { + activeSessions: () => 0, + acquire: async () => ({ status: "acquired" }), + release: () => undefined, + }, + activeWorkflowAuthorities: new Map(), + activeWorkflowPrincipals: new Map(), + workflowCapacityAttemptIds: new Set(), + directWorkflowPrincipalWorkItemIds: new Set(), + directWorkflowPrincipalHeldWorkItemIds: new Set(), + columnAgentIr: { version: "v2", name: "t", columns: [{ id: "todo", name: "Todo", traits: [] }], nodes: [executeNode] }, + resolveBindingForNode: () => undefined, + resolvedRunId: "run-1", + settings: { maxConcurrent: 12 }, + } as any, + }; +} + +const fencedContext = (principalAgentId: string) => ({ + "workflow:principal-agent-id": principalAgentId, + "workflow:principal-role": "executor", + "workflow:principal-authority": "task-assignee", +} as Record); + +describe("admitWorkflowPrincipalBeforeNode — stale fences re-route instead of wedging", () => { + it("admits an assigned engineer owner rather than holding its own executor node", async () => { + const engineer = agent("agent-backend-engineer", ["engineer"]); + const { deps: d } = deps([engineer]); + const context = fencedContext(engineer.id); + + const result = await admitWorkflowPrincipalBeforeNode( + d, executeNode, { id: "FN-8869", assignedAgentId: engineer.id } as any, context, + ); + + expect(result).toBeUndefined(); + expect(context["workflow:principal-agent-id"]).toBe(engineer.id); + expect(context["workflow:principal-authority"]).toBe("task-assignee"); + }); + + it("discards a fence whose principal lost the role and re-routes to the pool", async () => { + const stale = agent("agent-was-executor", ["reviewer"]); + const pool = agent("agent-workflow-executor", ["executor"]); + const { deps: d, written } = deps([stale, pool]); + const context = fencedContext(stale.id); + + const result = await admitWorkflowPrincipalBeforeNode( + d, executeNode, { id: "FN-8869", assignedAgentId: stale.id } as any, context, + ); + + expect(result).toBeUndefined(); + expect(context["workflow:principal-agent-id"]).toBe(pool.id); + expect(context["workflow:principal-authority"]).toBe("role-pool"); + // The durable continuation records the live principal, not the dead one. + expect(written[written.length - 1]).toMatchObject({ state: "running", principalAgentId: pool.id }); + }); + + it("discards a fence after ownership moves and runs under the new owner", async () => { + const previous = agent("agent-previous-owner", ["executor"]); + const current = agent("agent-current-owner", ["executor"]); + const { deps: d } = deps([previous, current]); + const context = fencedContext(previous.id); + + const result = await admitWorkflowPrincipalBeforeNode( + d, executeNode, { id: "FN-8869", assignedAgentId: current.id } as any, context, + ); + + expect(result).toBeUndefined(); + expect(context["workflow:principal-agent-id"]).toBe(current.id); + }); + + it("still holds — and records the hold — when the fenced principal is merely paused", async () => { + const paused = { ...agent("agent-owner", ["executor"]), state: "paused" }; + const pool = agent("agent-workflow-executor", ["executor"]); + const { deps: d, written } = deps([paused, pool]); + + const result = await admitWorkflowPrincipalBeforeNode( + d, executeNode, { id: "FN-8869", assignedAgentId: paused.id } as any, fencedContext(paused.id), + ); + + // A capable-but-unusable owner is a bounded wait, and is never silently replaced by the pool. + expect(result).toEqual({ outcome: "failure", value: "workflow-principal-named-principal-unavailable:executor" }); + expect(written[written.length - 1]).toMatchObject({ state: "held", principalAgentId: paused.id }); + }); +}); diff --git a/packages/engine/src/agents/workflow-agent-router.ts b/packages/engine/src/agents/workflow-agent-router.ts index dfbc7ffc13..ff8cc7e763 100644 --- a/packages/engine/src/agents/workflow-agent-router.ts +++ b/packages/engine/src/agents/workflow-agent-router.ts @@ -1,6 +1,6 @@ import { - canAgentReceiveImplementationTasks, classifyWorkflowAgentNode, + hasWorkflowRoleCapability, isEphemeralAgent, isWorkflowPrincipalEligible, resolveColumnAgentBinding, @@ -21,7 +21,19 @@ export interface WorkflowPrincipalRoute { export type WorkflowPrincipalRouteResult = | { status: "unclassified" } - | { status: "held"; role: WorkflowAgentRole; reason: "named-principal-unavailable" | "role-pool-exhausted" } + | { + status: "held"; + role: WorkflowAgentRole; + reason: "named-principal-unavailable" | "role-pool-exhausted"; + /** + * FNXC:WorkflowAgentRouting 2026-08-10-07:50: + * Set when a persisted fence names a principal that can NEVER satisfy this node — the agent is gone, + * lost the role, or had its authority edited away. The fence is stale, not waiting: the caller must + * re-route from scratch instead of holding. A capable-but-busy principal never sets this, so the + * "never silently replace a valid named principal" guarantee is unchanged. + */ + staleFence?: true; + } | { status: "routed"; route: WorkflowPrincipalRoute }; /** @@ -92,21 +104,26 @@ export function validateFencedWorkflowPrincipal(input: { nodeInstanceId?: string; activeSessions?: ReadonlyMap; }): WorkflowPrincipalRouteResult { + /* + * FNXC:WorkflowAgentRouting 2026-08-10-07:50: + * Every branch below proves the fence no longer describes reality — the node changed role, ownership moved, + * or the review override was edited away. None of those resolve by waiting, so they are marked `staleFence` + * and the caller re-routes. Holding on them is what turned an ordinary reassignment into a permanent wedge. + */ + const staleFence = { status: "held", role: input.role, reason: "named-principal-unavailable", staleFence: true } as const; const classifiedRole = classifyWorkflowAgentNode(input.node); - if (classifiedRole !== input.role) { - return { status: "held", role: input.role, reason: "named-principal-unavailable" }; - } + if (classifiedRole !== input.role) return staleFence; if (input.authority === "task-assignee" && ( input.role !== "executor" || input.task.assignedAgentId !== input.principalAgentId )) { - return { status: "held", role: input.role, reason: "named-principal-unavailable" }; + return staleFence; } if (input.authority === "review-node-override" && ( input.nodeInstanceId ? !isCurrentReviewerNodeOverride(input.ir, input.nodeInstanceId, input.principalAgentId) : input.node.reviewerAgentId !== input.principalAgentId )) { - return { status: "held", role: input.role, reason: "named-principal-unavailable" }; + return staleFence; } /* * FNXC:WorkflowAgentRouting 2026-08-07-04:45: @@ -115,12 +132,16 @@ export function validateFencedWorkflowPrincipal(input: { * could keep acting after an operator changed workflow routing. */ if (input.authority === "column-binding" && (!input.ir || resolveColumnAgentBinding(input.ir, input.node.id)?.agentId !== input.principalAgentId)) { - return { status: "held", role: input.role, reason: "named-principal-unavailable" }; + return staleFence; } const agent = input.agents.find((candidate) => candidate.id === input.principalAgentId); + // A vanished or role-less principal is a dead fence; a capable one that is merely busy is a real wait. + if (!agent || !hasWorkflowRoleCapability(agent, input.role, { + allowEngineerAsExecutor: input.authority === "task-assignee", + })) { + return staleFence; + } return available(agent, input.activeSessions ?? new Map()) - && agent.roles.includes(input.role) - && (input.role !== "executor" || canAgentReceiveImplementationTasks(agent)) ? { status: "routed", route: { agent, role: input.role, authority: input.authority } } : { status: "held", role: input.role, reason: "named-principal-unavailable" }; } @@ -178,9 +199,20 @@ export function routeWorkflowPrincipal(input: { // FNXC:IntakeOwnership 2026-08-09-18:15: A persisted executor owner is // revalidated at dispatch. Runtime disablement and assignment policy changes // revoke implementation authority instead of letting a stale owner run work. + /* + * FNXC:WorkflowAgentRouting 2026-08-10-07:50: + * `undefined` means "not authority for this node" and precedence CONTINUES to the column binding and then + * the role pool. A named identity that lacks the role can never run this node, so holding on it waits + * forever — that is the deadlock that stranded FN-8869/FN-8928/FN-8845 for hours behind idle pool agents, + * each because an operator legitimately assigned the card to an ENGINEER-role permanent agent (which + * `canAgentTakeImplementationTaskForExplicitRouting` permits) and the executor node then took that owner as + * `task-assignee` authority. Availability is still fail-closed below: a principal that HAS the role but is + * paused, disabled, or at session capacity holds, and is never silently replaced by a pool member. + */ + if (!agent || !hasWorkflowRoleCapability(agent, role, { allowEngineerAsExecutor: authority === "task-assignee" })) { + return undefined; + } return available(agent, activeSessions) - && agent.roles.includes(role) - && (role !== "executor" || canAgentReceiveImplementationTasks(agent)) ? { status: "routed", route: { agent, role, authority } } : { status: "held", role, reason: "named-principal-unavailable" }; }; @@ -202,7 +234,7 @@ export function routeWorkflowPrincipal(input: { if (bound) return bound; const pool = input.agents .filter((agent) => !input.excludedPoolAgentIds?.has(agent.id) - && agent.roles.includes(role) && available(agent, activeSessions)) + && hasWorkflowRoleCapability(agent, role) && available(agent, activeSessions)) .sort((left, right) => (activeSessions.get(left.id) ?? 0) - (activeSessions.get(right.id) ?? 0) || left.createdAt.localeCompare(right.createdAt) || left.id.localeCompare(right.id)); const agent = pool[0]; diff --git a/packages/engine/src/executor/workflow-principal-before-node.ts b/packages/engine/src/executor/workflow-principal-before-node.ts index 0485a7ff32..3dd80a2ebf 100644 --- a/packages/engine/src/executor/workflow-principal-before-node.ts +++ b/packages/engine/src/executor/workflow-principal-before-node.ts @@ -123,6 +123,30 @@ let routed = hasFencedPrincipal agents, activeSessions, }); +/** Cleared when a stale fence is discarded, so the pool-capacity retry below is no longer fenced off. */ +let fenceStillGoverns = Boolean(hasFencedPrincipal); +/* + * FNXC:WorkflowAgentRouting 2026-08-10-07:50: + * A fence that no longer describes reality is not a wait — the principal is gone, lost the role, or had its + * authority edited away, and no amount of re-dispatching fixes that. Re-route from scratch so the card keeps + * executing under whoever IS authority now. Without this, a resumed continuation kept re-asserting a dead + * principal every dispatch and the task only ever advanced when a human moved it. + */ +if (routed.status === "held" && routed.staleFence) { + executorLog.warn( + `[workflow-graph] ${nodeTask.id}: fenced principal '${fencedPrincipalId}' can no longer run node '${node.id}' ` + + `as '${classifiedRole}' — discarding the stale fence and re-routing.`, + ); + routed = routeWorkflowPrincipal({ + task: nodeTask, + ir: deps.columnAgentIr, + node, + agents, + activeSessions, + }); + // The fence is discarded, so a fresh role-pool route may take the cross-engine capacity retry below. + fenceStillGoverns = false; +} if (routed.status === "unclassified") return undefined; /* * FNXC:WorkflowAgentRouting 2026-08-07-23:50: @@ -264,7 +288,7 @@ let capacity = await deps.workflowAgentCapacity.acquire({ * Fenced and named principals never take this fallback. */ if (capacity.status === "held" && capacity.reason === "agent-capacity" - && routed.route.authority === "role-pool" && !hasFencedPrincipal) { + && routed.route.authority === "role-pool" && !fenceStillGoverns) { const excludedPoolAgentIds = new Set(); while (capacity.status === "held" && capacity.reason === "agent-capacity" && routed.route.authority === "role-pool") {